diff --git a/crates/fff-core/src/grep/grep_tests.rs b/crates/fff-core/src/grep/grep_tests.rs index a7c42da34..3ea6f0e2b 100644 --- a/crates/fff-core/src/grep/grep_tests.rs +++ b/crates/fff-core/src/grep/grep_tests.rs @@ -518,3 +518,78 @@ fn regex_fallback_keeps_file_path_scope_issue_756() { "regex fallback must not leak outside the FilePath scope" ); } + +#[test] +fn multi_grep_keeps_files_for_a_pattern_without_bigrams() { + let dir = tempfile::tempdir().unwrap(); + let base = crate::path_utils::canonicalize(dir.path()).unwrap(); + + // Only a/b/c carry the ASCII needle, so its bigrams stay below the + // ubiquity threshold that `compress` drops. + let base_contents: &[(&str, &str)] = &[ + ("a.txt", "hello unicorn world"), + ("b.txt", "another unicorn line"), + ("c.txt", "one more unicorn here"), + ("d.txt", "nothing special in here"), + ("e.txt", "日本語のテキスト"), + ("f.txt", "rainbow sky above"), + ]; + for (name, content) in base_contents { + let mut f = std::fs::File::create(base.join(name)).unwrap(); + writeln!(f, "{}", content).unwrap(); + } + + let mut picker = FilePicker::new(FilePickerOptions { + base_path: base.to_str().unwrap().into(), + watch: false, + ..Default::default() + }) + .unwrap(); + picker.collect_files().unwrap(); + assert_eq!(picker.get_files().len(), base_contents.len()); + + let consec_builder = BigramIndexBuilder::new(base_contents.len()); + let skip_builder = BigramIndexBuilder::new(base_contents.len()); + for (i, (_, content)) in base_contents.iter().enumerate() { + consec_builder.add_file_content(&skip_builder, i, content.as_bytes()); + } + // Compressed as build_bigram_index does, so this is the production prefilter. + let mut index = consec_builder.compress(None); + index.set_skip_index(skip_builder.compress(Some( + crate::index::bigram_filter::SKIP_INDEX_MIN_DENSITY_PCT, + ))); + picker.set_bigram_index(index); + + let options = crate::GrepSearchOptions { + mode: super::GrepMode::PlainText, + smart_case: true, + page_limit: 100, + ..Default::default() + }; + + let grep = |patterns: &[&str]| -> (Vec, usize) { + let result = picker.multi_grep(patterns, &[], &options); + let mut paths: Vec = result + .files + .iter() + .map(|f| f.relative_path(&picker)) + .collect(); + paths.sort(); + (paths, result.total_files_searched) + }; + + // The CJK needle has no printable-ASCII bigram, so nothing can be + // prefiltered: every file is searched and e.txt is found. + let (paths, searched) = grep(&["unicorn", "日本語"]); + assert_eq!(paths, vec!["a.txt", "b.txt", "c.txt", "e.txt"]); + assert_eq!(searched, base_contents.len()); + + // Every pattern indexable: only the files carrying some pattern's + // bigrams are searched, so the prefilter still narrows. + let (paths, searched) = grep(&["unicorn", "rainbow"]); + assert_eq!(paths, vec!["a.txt", "b.txt", "c.txt", "f.txt"]); + assert_eq!(searched, 4); + let (paths, searched) = grep(&["unicorn"]); + assert_eq!(paths, vec!["a.txt", "b.txt", "c.txt"]); + assert_eq!(searched, 3); +} diff --git a/crates/fff-core/src/index/bigram_filter.rs b/crates/fff-core/src/index/bigram_filter.rs index 2c409151b..e85078011 100644 --- a/crates/fff-core/src/index/bigram_filter.rs +++ b/crates/fff-core/src/index/bigram_filter.rs @@ -794,7 +794,7 @@ const BIGRAM_CHUNK_FILES: usize = 4 * 64; /// Sparse-column cutoff for the skip-1 sub-index. Rare skip columns add /// little filtering power but ~25-30% of index memory, so we drop /// anything appearing in < 12 % of populated files. -const SKIP_INDEX_MIN_DENSITY_PCT: u32 = 12; +pub(crate) const SKIP_INDEX_MIN_DENSITY_PCT: u32 = 12; thread_local! { /// Reusable read buffer that is allocated per thread and used for reading files diff --git a/crates/fff-core/src/index/candidates.rs b/crates/fff-core/src/index/candidates.rs index e6a2669bd..1542d51fd 100644 --- a/crates/fff-core/src/index/candidates.rs +++ b/crates/fff-core/src/index/candidates.rs @@ -38,17 +38,18 @@ pub(crate) fn literal_candidates( let mut combined: Option> = None; for pattern in patterns { - if let Some(candidates) = index.query(pattern.as_bytes()) { - combined = Some(match combined { - None => candidates, - Some(mut acc) => { - acc.iter_mut() - .zip(candidates.iter()) - .for_each(|(a, b)| *a |= *b); - acc - } - }); - } + // A pattern the index can't narrow matches any file, and the union with + // it is every file — so bail out of prefiltering instead of dropping it. + let candidates = index.query(pattern.as_bytes())?; + combined = Some(match combined { + None => candidates, + Some(mut acc) => { + acc.iter_mut() + .zip(candidates.iter()) + .for_each(|(a, b)| *a |= *b); + acc + } + }); } let mut candidates = combined?;