Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
75 changes: 75 additions & 0 deletions crates/fff-core/src/grep/grep_tests.rs
Original file line number Diff line number Diff line change
Expand Up @@ -518,3 +518,78 @@ fn regex_fallback_keeps_file_path_scope_issue_756() {
"regex fallback must not leak outside the FilePath scope"
);
}

#[test]
fn multi_grep_keeps_files_for_a_pattern_without_bigrams() {
let dir = tempfile::tempdir().unwrap();
let base = crate::path_utils::canonicalize(dir.path()).unwrap();

// Only a/b/c carry the ASCII needle, so its bigrams stay below the
// ubiquity threshold that `compress` drops.
let base_contents: &[(&str, &str)] = &[
("a.txt", "hello unicorn world"),
("b.txt", "another unicorn line"),
("c.txt", "one more unicorn here"),
("d.txt", "nothing special in here"),
("e.txt", "日本語のテキスト"),
("f.txt", "rainbow sky above"),
];
for (name, content) in base_contents {
let mut f = std::fs::File::create(base.join(name)).unwrap();
writeln!(f, "{}", content).unwrap();
}

let mut picker = FilePicker::new(FilePickerOptions {
base_path: base.to_str().unwrap().into(),
watch: false,
..Default::default()
})
.unwrap();
picker.collect_files().unwrap();
assert_eq!(picker.get_files().len(), base_contents.len());

let consec_builder = BigramIndexBuilder::new(base_contents.len());
let skip_builder = BigramIndexBuilder::new(base_contents.len());
for (i, (_, content)) in base_contents.iter().enumerate() {
consec_builder.add_file_content(&skip_builder, i, content.as_bytes());
}
// Compressed as build_bigram_index does, so this is the production prefilter.
let mut index = consec_builder.compress(None);
index.set_skip_index(skip_builder.compress(Some(
crate::index::bigram_filter::SKIP_INDEX_MIN_DENSITY_PCT,
)));
picker.set_bigram_index(index);

let options = crate::GrepSearchOptions {
mode: super::GrepMode::PlainText,
smart_case: true,
page_limit: 100,
..Default::default()
};

let grep = |patterns: &[&str]| -> (Vec<String>, usize) {
let result = picker.multi_grep(patterns, &[], &options);
let mut paths: Vec<String> = result
.files
.iter()
.map(|f| f.relative_path(&picker))
.collect();
paths.sort();
(paths, result.total_files_searched)
};

// The CJK needle has no printable-ASCII bigram, so nothing can be
// prefiltered: every file is searched and e.txt is found.
let (paths, searched) = grep(&["unicorn", "日本語"]);
assert_eq!(paths, vec!["a.txt", "b.txt", "c.txt", "e.txt"]);
assert_eq!(searched, base_contents.len());

// Every pattern indexable: only the files carrying some pattern's
// bigrams are searched, so the prefilter still narrows.
let (paths, searched) = grep(&["unicorn", "rainbow"]);
assert_eq!(paths, vec!["a.txt", "b.txt", "c.txt", "f.txt"]);
assert_eq!(searched, 4);
let (paths, searched) = grep(&["unicorn"]);
assert_eq!(paths, vec!["a.txt", "b.txt", "c.txt"]);
assert_eq!(searched, 3);
}
2 changes: 1 addition & 1 deletion crates/fff-core/src/index/bigram_filter.rs
Original file line number Diff line number Diff line change
Expand Up @@ -794,7 +794,7 @@ const BIGRAM_CHUNK_FILES: usize = 4 * 64;
/// Sparse-column cutoff for the skip-1 sub-index. Rare skip columns add
/// little filtering power but ~25-30% of index memory, so we drop
/// anything appearing in < 12 % of populated files.
const SKIP_INDEX_MIN_DENSITY_PCT: u32 = 12;
pub(crate) const SKIP_INDEX_MIN_DENSITY_PCT: u32 = 12;

thread_local! {
/// Reusable read buffer that is allocated per thread and used for reading files
Expand Down
23 changes: 12 additions & 11 deletions crates/fff-core/src/index/candidates.rs
Original file line number Diff line number Diff line change
Expand Up @@ -38,17 +38,18 @@ pub(crate) fn literal_candidates(

let mut combined: Option<Vec<u64>> = None;
for pattern in patterns {
if let Some(candidates) = index.query(pattern.as_bytes()) {
combined = Some(match combined {
None => candidates,
Some(mut acc) => {
acc.iter_mut()
.zip(candidates.iter())
.for_each(|(a, b)| *a |= *b);
acc
}
});
}
// A pattern the index can't narrow matches any file, and the union with
// it is every file — so bail out of prefiltering instead of dropping it.
let candidates = index.query(pattern.as_bytes())?;
combined = Some(match combined {
None => candidates,
Some(mut acc) => {
acc.iter_mut()
.zip(candidates.iter())
.for_each(|(a, b)| *a |= *b);
acc
}
});
}

let mut candidates = combined?;
Expand Down
Loading