diff --git a/src/sentence.rs b/src/sentence.rs index 4a51a08..9223a32 100644 --- a/src/sentence.rs +++ b/src/sentence.rs @@ -176,8 +176,17 @@ mod fwd { #[inline] fn size_hint(&self) -> (usize, Option) { let slen = self.string.len(); - // A sentence could be one character - (cmp::min(slen, 2), Some(slen + 1)) + // `next` advances `pos` and never shortens `string`, so only the + // breaks at or past `pos` are still to come. + let remaining = slen - self.pos; + // A sentence could be one character, and while `pos` is 0 the + // start-of-text break is still to come as well. + let lower = if self.pos == 0 { + cmp::min(remaining, 2) + } else { + cmp::min(remaining, 1) + }; + (lower, Some(slen + 1)) } #[inline] @@ -415,3 +424,45 @@ impl<'a> Iterator for USentenceBoundIndices<'a> { self.iter.size_hint() } } + +#[test] +fn test_sentence_breaks_size_hint_tracks_what_is_left() { + // `SentenceBreaks` walks `string` with `pos` instead of reslicing it, so a + // bound measured from `string.len()` keeps describing the whole input. + for s in [ + "", + "a", + "ab", + "\r\r", + "Hi. There.", + "Mr. Fox jumped. [...] The dog was too lazy.", + ] { + let total = fwd::new_sentence_breaks(s).count(); + let mut it = fwd::new_sentence_breaks(s); + for taken in 0..=total { + let left = total - taken; + let (lower, upper) = it.size_hint(); + assert!( + lower <= left, + "with {} of {} breaks taken, size_hint lower {} exceeds the {} left on {:?}", + taken, + total, + lower, + left, + s + ); + if let Some(upper) = upper { + assert!( + left <= upper, + "with {} of {} breaks taken, the {} left exceed size_hint upper {} on {:?}", + taken, + total, + left, + upper, + s + ); + } + it.next(); + } + } +} diff --git a/tests/test.rs b/tests/test.rs index 6424bd5..24462b4 100644 --- a/tests/test.rs +++ b/tests/test.rs @@ -380,3 +380,91 @@ fn test_size_hint_is_a_valid_bound() { check!("unicode_sentences", s, || s.unicode_sentences()); } } + +/// `size_hint` has to bracket the items *still to come*, not just the ones a +/// fresh iterator started with. `test_size_hint_is_a_valid_bound` only queries +/// iterators that have not been advanced, so a bound measured from the whole +/// input instead of from the part left to scan passes it at every input. +#[test] +fn test_size_hint_is_a_valid_bound_while_iterating() { + use crate::testdata::{TEST_SAME, TEST_SENTENCE, TEST_WORD}; + + fn check(label: &str, s: &str, mut it: I, total: usize) { + for taken in 0..=total { + let left = total - taken; + let (lo, hi) = it.size_hint(); + assert!( + lo <= left, + "{}: with {} of {} items taken, size_hint lower {} exceeds the {} left on {:?}", + label, + taken, + total, + lo, + left, + s + ); + if let Some(hi) = hi { + assert!( + left <= hi, + "{}: with {} of {} items taken, the {} left exceed size_hint upper {} on {:?}", + label, + taken, + total, + left, + hi, + s + ); + } + it.next(); + } + } + + macro_rules! check { + ($label:expr, $s:expr, $mk:expr) => {{ + let mk = $mk; + let total = mk().count(); + check($label, $s, mk(), total); + }}; + } + + let corpus = [ + "", + " ", + "a", + "ab", + "\r\n", + "0", + "\u{1f600}", + "Mr. Fox jumped. [...] The dog was too lazy.", + ] + .iter() + .copied() + .chain(TEST_SAME.iter().map(|&(s, _)| s)) + .chain(TEST_WORD.iter().map(|&(s, _)| s)) + .chain(TEST_SENTENCE.iter().map(|&(s, _)| s)); + + for s in corpus { + check!("graphemes(true)", s, || s.graphemes(true)); + check!("graphemes(false)", s, || s.graphemes(false)); + check!("grapheme_indices(true)", s, || s.grapheme_indices(true)); + check!("split_word_bounds", s, || s.split_word_bounds()); + check!("split_word_bound_indices", s, || s + .split_word_bound_indices()); + check!("unicode_words", s, || s.unicode_words()); + check!("unicode_word_indices", s, || s.unicode_word_indices()); + check!("split_sentence_bounds", s, || s.split_sentence_bounds()); + check!("split_sentence_bound_indices", s, || s + .split_sentence_bound_indices()); + check!("unicode_sentences", s, || s.unicode_sentences()); + + // Tightening the bound must not cost the strength it already had: two + // or more bytes still promise a sentence before anything is taken. + if s.len() >= 2 { + assert!( + s.split_sentence_bounds().size_hint().0 >= 1, + "split_sentence_bounds: fresh size_hint lower bound regressed to 0 on {:?}", + s + ); + } + } +}