From d8ab6da1ec13878d01de4041249b3635f1f6dcd5 Mon Sep 17 00:00:00 2001 From: Dmitriy Kovalenko Date: Sat, 26 Sep 2026 11:30:21 -0700 Subject: [PATCH 1/3] perf(search): rare bigram columns, openat grep reads, rayon fuzzy matching, extension fast path --- crates/fff-core/src/file_picker.rs | 1 + crates/fff-core/src/grep/fuzzy_grep.rs | 14 +- crates/fff-core/src/grep/grep.rs | 40 +-- crates/fff-core/src/grep/grep_tests.rs | 100 ++++++++ crates/fff-core/src/index/bigram_filter.rs | 68 ++++-- crates/fff-core/src/index/bigram_query.rs | 4 +- crates/fff-core/src/index/constraints.rs | 13 +- crates/fff-core/src/parallelism.rs | 175 +++++++++++--- crates/fff-core/src/score.rs | 269 +++++++++++++++++++-- crates/fff-core/src/simd_path.rs | 109 +++++++++ crates/fff-core/src/types.rs | 89 ++++++- 11 files changed, 773 insertions(+), 109 deletions(-) diff --git a/crates/fff-core/src/file_picker.rs b/crates/fff-core/src/file_picker.rs index 0d6a6d624..459d2d5d5 100644 --- a/crates/fff-core/src/file_picker.rs +++ b/crates/fff-core/src/file_picker.rs @@ -1402,6 +1402,7 @@ impl FilePicker { SEARCH_THREAD_POOL.install(|| { grep_search( self.get_files(), + self.live_file_count(), query, options, self.cache_budget(), diff --git a/crates/fff-core/src/grep/fuzzy_grep.rs b/crates/fff-core/src/grep/fuzzy_grep.rs index 57e8931c5..d3910f900 100644 --- a/crates/fff-core/src/grep/fuzzy_grep.rs +++ b/crates/fff-core/src/grep/fuzzy_grep.rs @@ -23,6 +23,10 @@ pub(super) fn fuzzy_grep_search<'a>( arena: ArenaPtr, overflow_arena: ArenaPtr, ) -> GrepResult<'a> { + #[cfg(unix)] + let base_dir = std::fs::File::open(base_path).ok(); + #[cfg(not(unix))] + let base_dir = None; // max_typos controls how many *needle* characters can be unmatched. // A transposition (e.g. "shcema" -> "schema") costs ~1 typo with // default gap penalties. We scale max_typos by needle length: @@ -169,8 +173,14 @@ pub(super) fn fuzzy_grep_search<'a>( arena }; - let file_bytes = - file.get_content_for_search(buf, mmap_slot, file_arena, base_path, budget)?; + let file_bytes = file.get_content_for_search( + buf, + mmap_slot, + file_arena, + base_path, + budget, + base_dir.as_ref(), + )?; if min_chars_required > 0 { let mut chars_found = 0usize; diff --git a/crates/fff-core/src/grep/grep.rs b/crates/fff-core/src/grep/grep.rs index 08a7e46dd..df696d571 100644 --- a/crates/fff-core/src/grep/grep.rs +++ b/crates/fff-core/src/grep/grep.rs @@ -7,7 +7,7 @@ use crate::index::{ regex_candidates, }; use crate::simd_string_utils::memmem; -use crate::types::{ContentCacheBudget, FileItem, FileSliceExt, MmapSlot}; +use crate::types::{ContentCacheBudget, FileItem, MmapSlot}; use fff_grep::{ Searcher, SearcherBuilder, Sink, SinkMatch, matcher::{Match, Matcher, NoError}, @@ -191,6 +191,7 @@ impl Sink for PlainTextSink<'_> { #[allow(clippy::too_many_arguments)] pub(crate) fn grep_search<'a>( files: &'a [FileItem], + total_files: usize, query: &FFFQuery<'_>, options: &GrepSearchOptions, budget: &ContentCacheBudget, @@ -203,6 +204,7 @@ pub(crate) fn grep_search<'a>( ) -> GrepResult<'a> { let result = grep_search_parsed( files, + total_files, query, options, budget, @@ -248,6 +250,7 @@ pub(crate) fn grep_search<'a>( let mut fallback = grep_search_parsed( files, + total_files, &literal_query, options, budget, @@ -270,6 +273,7 @@ pub(crate) fn grep_search<'a>( #[allow(clippy::too_many_arguments)] fn grep_search_parsed<'a>( files: &'a [FileItem], + total_files: usize, query: &FFFQuery<'_>, options: &GrepSearchOptions, budget: &ContentCacheBudget, @@ -280,7 +284,6 @@ fn grep_search_parsed<'a>( arena: crate::simd_path::ArenaPtr, overflow_arena: crate::simd_path::ArenaPtr, ) -> GrepResult<'a> { - let total_files = files.live_count(); let constraints_from_query = &query.constraints[..]; let grep_text = extract_grep_text(query); @@ -377,7 +380,10 @@ fn grep_search_parsed<'a>( ); if files_to_search.is_empty() { - return GrepResult::empty(total_files, filtered_file_count); + return GrepResult { + regex_fallback_error, + ..GrepResult::empty(total_files, filtered_file_count) + }; } // `PlainTextMatcher` is used by the grep-searcher engine for line detection. @@ -557,6 +563,10 @@ where }; let search_start = std::time::Instant::now(); + #[cfg(unix)] + let base_dir = std::fs::File::open(ctx.base_path).ok(); + #[cfg(not(unix))] + let base_dir = None; let page_limit = options.page_limit; // Lowest index an abort skipped; everything below it was searched, so it is // the resume point for the next page. usize::MAX means no abort happened. @@ -568,21 +578,11 @@ where let mut page_filled = false; let mut aborted = false; - // Each chunk is a rayon barrier. A flat small chunk over 500k files = ~7800 - // barriers; x2 growth makes it logarithmic. But a too-aggressive growth - // over-scans: when a page fills mid-chunk, the whole submitted chunk still - // runs. - // - // So only grow when the prefilter is weak (large candidate set); - // when bigram cut the set in half, keep fixed small chunks for cheap page-fill termination. + // Grow empty batches to avoid repeated barriers on absent queries. + // A strong prefilter returns to small batches as soon as matches appear. let base_chunk = rayon::current_num_threads() * 4; let prefilter_strong = ctx.total_files > 0 && files_to_search.len() * 2 < ctx.total_files; - let max_chunk = if prefilter_strong { - base_chunk - } else { - (base_chunk * 256).max(8 * 1024) - }; - let growth = if prefilter_strong { 1 } else { 2 }; + let max_chunk = (base_chunk * 256).max(8 * 1024); let mut chunk_size = base_chunk; let mut chunk_start = 0; @@ -590,7 +590,6 @@ where let chunk_end = (chunk_start + chunk_size).min(files_to_search.len()); let chunk = &files_to_search[chunk_start..chunk_end]; chunk_start = chunk_end; - chunk_size = (chunk_size * growth).min(max_chunk); let chunk_offset = files_consumed; // Unless enforced, the budget stays dormant until something matched. @@ -624,6 +623,7 @@ where ctx.arena_for_file(file), ctx.base_path, ctx.budget, + base_dir.as_ref(), )?; // Fast whole-file memmem check before entering the @@ -647,6 +647,12 @@ where .flatten() .collect(); + chunk_size = if prefilter_strong && !chunk_results.is_empty() { + base_chunk + } else { + (chunk_size * 2).min(max_chunk) + }; + // Every file in the chunk was visited unless an abort cut it short. let resume_at = first_skipped.load(Ordering::Relaxed); aborted = resume_at != usize::MAX; diff --git a/crates/fff-core/src/grep/grep_tests.rs b/crates/fff-core/src/grep/grep_tests.rs index 886028613..d736e1b55 100644 --- a/crates/fff-core/src/grep/grep_tests.rs +++ b/crates/fff-core/src/grep/grep_tests.rs @@ -6,6 +6,106 @@ use crate::index::BigramIndexBuilder; use std::io::Write; use std::sync::atomic::AtomicBool; +#[test] +fn invalid_regex_error_survives_empty_literal_prefilter() { + let root = tempfile::tempdir().unwrap(); + for index in 0..128 { + std::fs::write( + root.path().join(format!("file_{index}.txt")), + "ordinary text", + ) + .unwrap(); + } + std::fs::write(root.path().join("file_0.txt"), "Qq marker").unwrap(); + std::fs::write(root.path().join("file_1.txt"), "Zz marker").unwrap(); + let mut picker = FilePicker::new(FilePickerOptions { + base_path: root.path().to_string_lossy().into_owned(), + watch: false, + ..Default::default() + }) + .unwrap(); + picker.collect_files().unwrap(); + let query = parse_grep_query("QqZz("); + let options = GrepSearchOptions { + mode: GrepMode::Regex, + ..Default::default() + }; + let before = picker.grep(&query, &options); + assert!(before.matches.is_empty()); + assert!(before.regex_fallback_error.is_some()); + + let builder = BigramIndexBuilder::new(picker.get_files().len()); + let skip = BigramIndexBuilder::new(picker.get_files().len()); + for (index, file) in picker.get_files().iter().enumerate() { + let content = std::fs::read(root.path().join(file.relative_path(&picker))).unwrap(); + builder.add_file_content(&skip, index, &content); + } + picker.set_bigram_index(builder.compress(None)); + let after = picker.grep(&query, &options); + assert!(after.matches.is_empty()); + assert_eq!(after.total_files_searched, 0); + assert!(after.regex_fallback_error.is_some()); +} + +#[test] +fn sparse_matches_after_empty_batches_preserve_pagination() { + let root = tempfile::tempdir().unwrap(); + for index in 0..1536 { + let extension = if index < 512 { "rs" } else { "png" }; + let path = root.path().join(format!("file_{index}.{extension}")); + let content = if [32, 100, 180, 240].contains(&index) { + "needle\n" + } else { + "ordinary content\n" + }; + std::fs::write(&path, content).unwrap(); + std::fs::File::open(path) + .unwrap() + .set_times(std::fs::FileTimes::new().set_modified( + std::time::UNIX_EPOCH + std::time::Duration::from_secs(1_700_000_000 + index), + )) + .unwrap(); + } + let mut picker = FilePicker::new(FilePickerOptions { + base_path: root.path().to_string_lossy().into_owned(), + watch: false, + ..Default::default() + }) + .unwrap(); + picker.collect_files().unwrap(); + + for mode in [GrepMode::PlainText, GrepMode::Regex] { + let query = parse_grep_query("needle *.rs"); + let mut options = GrepSearchOptions { + mode, + page_limit: usize::MAX, + ..Default::default() + }; + let result = picker.grep(&query, &options); + let expected: Vec<_> = result + .files + .iter() + .map(|file| file.relative_path(&picker)) + .collect(); + assert_eq!(expected.len(), 4); + assert_eq!(result.filtered_file_count, 512); + + options.page_limit = 1; + let mut actual = Vec::new(); + loop { + let page = picker.grep(&query, &options); + actual.extend(page.files.iter().map(|file| file.relative_path(&picker))); + assert_eq!(page.total_files, 1536); + if page.next_file_offset == 0 { + break; + } + assert!(page.next_file_offset > options.file_offset); + options.file_offset = page.next_file_offset; + } + assert_eq!(actual, expected); + } +} + #[test] fn test_replace_newline_escapes() { // Single \n → multiline: replaced with a real newline at byte 3 diff --git a/crates/fff-core/src/index/bigram_filter.rs b/crates/fff-core/src/index/bigram_filter.rs index 41c335050..2d0ebcd5f 100644 --- a/crates/fff-core/src/index/bigram_filter.rs +++ b/crates/fff-core/src/index/bigram_filter.rs @@ -370,10 +370,8 @@ impl BigramIndexBuilder { /// Compress the dense builder into a compact `BigramFilter`. /// - /// Retains columns where the bigram appears in ≥`min_density_pct`% (or - /// the default ~3.1% heuristic when `None`) and <90% of indexed files. - /// Sparse columns carry too little data to justify their memory; - /// ubiquitous columns (≥90%) are nearly all-ones and barely filter. + /// Keeps rare columns as gap lists unless `min_density_pct` excludes them. + /// Ubiquitous columns (≥90%) are omitted because they barely filter. #[inline(always)] pub fn compress(self, min_density_pct: Option) -> BigramFilter { let cols = self.columns_used() as usize; @@ -405,15 +403,9 @@ impl BigramIndexBuilder { let bitset = &col_data[col_start..col_start + words]; let popcount: u32 = bitset.iter().map(|w| w.count_ones()).sum(); - // drop bigrams appearing in too few files - let not_to_rare = if let Some(min_pct) = min_density_pct { - // Percentage-based: require ≥ min_pct% of populated files. - populated > 0 && (popcount as usize) * 100 >= populated * min_pct as usize - } else { - // Default: popcount ≥ words × 2 (~3.1% of files). - (popcount as usize * 4) >= dense_bytes - }; - if !not_to_rare { + if min_density_pct.is_some_and(|min_pct| { + populated == 0 || (popcount as usize) * 100 < populated * min_pct as usize + }) { continue; } @@ -427,6 +419,14 @@ impl BigramIndexBuilder { } } kept.sort_unstable_by_key(|&(_, old_col, _)| old_col); + let mut rare_keys = vec![0u64; BIGRAM_KEY_SLOTS.div_ceil(64)]; + if min_density_pct.is_none() { + for &(slot, _, count) in &kept { + if (count as usize) * 4 < dense_bytes { + rare_keys[slot / 64] |= 1 << (slot % 64); + } + } + } let column = |old_col: u16| -> &[u64] { let start = old_col as usize * words; &col_data.as_deref().expect("kept columns imply a slab")[start..start + words] @@ -491,6 +491,7 @@ impl BigramIndexBuilder { sparse_data.shrink_to_fit(); BigramFilter { + rare_keys, lookup, dense_data, dense_count, @@ -574,6 +575,7 @@ unsafe impl Send for BigramIndexBuilder {} #[derive(Debug)] pub struct BigramFilter { lookup: Vec, + rare_keys: Vec, /// Flat buffer of all dense column data laid out at fixed stride `words`. /// Column `i` starts at `i * words`. dense_data: ColumnSlab, // do not try to change this to u8 it has to be wordsize @@ -697,6 +699,18 @@ impl BigramFilter { } } + // Regex and fuzzy filters keep their original columns; literals can use rare ones. + pub(crate) fn common_column_bitset(&self, key: u16) -> Option> { + if self.column(key) == NO_COLUMN { + return None; + } + let slot = key_slot(key); + if self.rare_keys[slot / 64] & (1 << (slot % 64)) != 0 { + return None; + } + self.column_bitset(key) + } + /// Column as a materialized bitset (borrowed for dense, decoded for sparse). pub(crate) fn column_bitset(&self, key: u16) -> Option> { Some(match self.column_ref(key)? { @@ -752,7 +766,7 @@ impl BigramFilter { let lookup_bytes = self.lookup.len() * std::mem::size_of::(); let dense_bytes = self.dense_data.len() * std::mem::size_of::(); let skip_bytes = self.skip_index.as_ref().map_or(0, |s| s.heap_bytes()); - lookup_bytes + dense_bytes + self.sparse_bytes() + skip_bytes + lookup_bytes + self.rare_keys.len() * 8 + dense_bytes + self.sparse_bytes() + skip_bytes } /// Check whether a bigram key is present in this index. @@ -818,6 +832,7 @@ impl BigramFilter { }; Self { lookup, + rare_keys: vec![0; BIGRAM_KEY_SLOTS.div_ceil(64)], dense_data, dense_count, sparse_offsets: vec![0], @@ -1241,6 +1256,31 @@ mod tests { assert_eq!(anded, expected); } + #[test] + fn rare_bigrams_reject_absent_queries_without_losing_matches() { + let builder = BigramIndexBuilder::new(4096); + let skip = BigramIndexBuilder::new(4096); + for file_idx in 0..4096 { + let content: &[u8] = match file_idx { + 31 => b"marker Qq blah", + 500 => b"marker qz blah", + 1000 => b"marker zy blah", + 3000 => b"marker yx blah", + _ => b"ordinary content", + }; + builder.add_file_content(&skip, file_idx, content); + } + let index = builder.compress(None); + let absent = index.query(b"Qqzyx").expect("rare bigrams retained"); + assert_eq!(BigramFilter::count_candidates(&absent), 0); + let present = index.query(b"qq").unwrap(); + assert_eq!(BigramFilter::count_candidates(&present), 1); + assert!(BigramFilter::is_candidate(&present, 31)); + assert!(index.common_column_bitset(key(b'q', b'q')).is_none()); + assert!(index.sparse_count() > 0); + assert!(!skip.compress(Some(12)).has_key(key(b'q', b' '))); + } + #[test] fn sparse_column_roundtrip_variants() { sparse_roundtrip(&[]); diff --git a/crates/fff-core/src/index/bigram_query.rs b/crates/fff-core/src/index/bigram_query.rs index d3963b2fa..cb017b1da 100644 --- a/crates/fff-core/src/index/bigram_query.rs +++ b/crates/fff-core/src/index/bigram_query.rs @@ -92,9 +92,9 @@ impl BigramQuery { match self { BigramQuery::Any => None, - BigramQuery::Consec(key) => index.column_bitset(*key), + BigramQuery::Consec(key) => index.common_column_bitset(*key), - BigramQuery::Skip1(key) => index.skip_index()?.column_bitset(*key), + BigramQuery::Skip1(key) => index.skip_index()?.common_column_bitset(*key), BigramQuery::And(children) => { let mut result: Option> = None; diff --git a/crates/fff-core/src/index/constraints.rs b/crates/fff-core/src/index/constraints.rs index cdcab3b05..d91759fcc 100644 --- a/crates/fff-core/src/index/constraints.rs +++ b/crates/fff-core/src/index/constraints.rs @@ -14,6 +14,11 @@ pub(crate) trait Constrainable { fn git_status(&self) -> Option; fn write_relative_path(&self, arena: ArenaPtr, out: &mut String); fn is_overflow(&self) -> bool; + + fn has_extension(&self, arena: ArenaPtr, extension: &str, scratch: &mut String) -> bool { + self.write_file_name(arena, scratch); + file_has_extension(scratch, extension) + } } /// Stored/canonical paths use `/`; also accept `\` so a Windows user typing @@ -289,10 +294,9 @@ impl<'q, 'c> ConstraintPlan<'q, 'c> { if self.extensions.is_empty() { return true; } - item.write_file_name(arena, &mut scratch.fname); self.extensions .iter() - .any(|ext| file_has_extension(&scratch.fname, ext)) + .any(|ext| item.has_extension(arena, ext, &mut scratch.fname)) } } @@ -330,10 +334,7 @@ fn evaluate( } // Reachable only via `Not(Extension(_))` — bare extensions are split out // up front and handled in `passes_extensions`. - Constraint::Extension(ext) => { - item.write_file_name(arena, &mut scratch.fname); - file_has_extension(&scratch.fname, ext) - } + Constraint::Extension(ext) => item.has_extension(arena, ext, &mut scratch.fname), Constraint::PathSegment(segment) => { item.write_relative_path(arena, &mut scratch.path); path_contains_segment(&scratch.path, segment) diff --git a/crates/fff-core/src/parallelism.rs b/crates/fff-core/src/parallelism.rs index 7f6c3bbd4..d673efb1f 100644 --- a/crates/fff-core/src/parallelism.rs +++ b/crates/fff-core/src/parallelism.rs @@ -1,8 +1,3 @@ -//! Dedicated rayon pools. The global pool spans every logical core, which -//! oversubscribes asymmetric chips (Apple P+E): E-cores are ~2× slower and -//! `open()` contends on a per-VFS lock past P-core count, so a larger pool is -//! slower on file-heavy work. - use std::sync::LazyLock; /// Dedicated thread pool for background work (scan, warmup, bigram build). @@ -18,8 +13,7 @@ pub static BACKGROUND_THREAD_POOL: LazyLock = LazyLock::new(| .num_threads(bg_threads) .thread_name(|i| format!("fff-bg-{i}")) .start_handler(|_| { - // QoS pin keeps workers on P-cores; the kernel otherwise drifts - // them to ~2× slower E-cores. + // Request user-initiated scheduling priority. #[cfg(target_os = "macos")] unsafe { let _ = libc::pthread_set_qos_class_self_np( @@ -32,36 +26,10 @@ pub static BACKGROUND_THREAD_POOL: LazyLock = LazyLock::new(| .expect("failed to create background rayon pool") }); -/// Physical performance-core count via sysctl, falling back to logical cores. -/// On a 12P+4E M4 Max, grep runs 16t=6.2s vs 13t=4.9s — fewer threads win. -#[cfg(target_os = "macos")] -fn performance_core_count() -> usize { - let mut count: libc::c_int = 0; - let mut size = std::mem::size_of::(); - let name = c"hw.perflevel0.physicalcpu"; - let ok = unsafe { - libc::sysctlbyname( - name.as_ptr(), - &mut count as *mut _ as *mut libc::c_void, - &mut size, - std::ptr::null_mut(), - 0, - ) - }; - if ok == 0 && count > 0 { - count as usize - } else { - std::thread::available_parallelism() - .map(|p| p.get()) - .unwrap_or(4) - } -} - -/// Pool for grep content search: P-core sized and QoS-pinned on macOS, full -/// parallelism elsewhere. Avoids E-core drag and VFS-lock contention. +/// Grep pool sized to non-efficiency cores on macOS and full parallelism elsewhere. pub static SEARCH_THREAD_POOL: LazyLock = LazyLock::new(|| { #[cfg(target_os = "macos")] - let threads = performance_core_count(); + let threads = non_efficiency_core_count(); #[cfg(not(target_os = "macos"))] let threads = std::thread::available_parallelism() .map(|p| p.get()) @@ -82,3 +50,140 @@ pub static SEARCH_THREAD_POOL: LazyLock = LazyLock::new(|| { .build() .expect("failed to create search rayon pool") }); + +#[cfg(target_os = "macos")] +fn non_efficiency_core_count() -> usize { + let detected = sysctl_count(c"hw.nperflevels").and_then(|levels| { + count_non_efficiency_cores((0..levels).map(|level| { + let name = std::ffi::CString::new(format!("hw.perflevel{level}.name")).ok()?; + let count = std::ffi::CString::new(format!("hw.perflevel{level}.physicalcpu")).ok()?; + Some((sysctl_name(&name)?, sysctl_count(&count)?)) + })) + }); + + // Unknown topology falls back to all logical cores, never to a single thread. + detected + .or_else(|| sysctl_count(c"hw.perflevel0.physicalcpu").filter(|&count| count > 0)) + .unwrap_or_else(|| { + std::thread::available_parallelism() + .map(|p| p.get()) + .unwrap_or(4) + }) +} + +#[cfg(any(target_os = "macos", test))] +fn count_non_efficiency_cores( + levels: impl IntoIterator>, +) -> Option { + let count = levels.into_iter().try_fold(0usize, |total, level| { + let (name, count) = level?; + if name.eq_ignore_ascii_case("Efficiency") { + Some(total) + } else if ["Super", "Performance", "Standard"] + .iter() + .any(|known| name.eq_ignore_ascii_case(known)) + { + total.checked_add(count) + } else { + None + } + })?; + (count > 0).then_some(count) +} + +#[cfg(target_os = "macos")] +fn sysctl_count(name: &std::ffi::CStr) -> Option { + let mut value: libc::c_int = 0; + let mut size = std::mem::size_of_val(&value); + let status = unsafe { + libc::sysctlbyname( + name.as_ptr(), + &mut value as *mut _ as *mut libc::c_void, + &mut size, + std::ptr::null_mut(), + 0, + ) + }; + if status != 0 || size != std::mem::size_of_val(&value) { + return None; + } + usize::try_from(value).ok() +} + +#[cfg(target_os = "macos")] +fn sysctl_name(name: &std::ffi::CStr) -> Option { + let mut value = [0u8; 64]; + let mut size = value.len(); + let status = unsafe { + libc::sysctlbyname( + name.as_ptr(), + value.as_mut_ptr().cast(), + &mut size, + std::ptr::null_mut(), + 0, + ) + }; + if status != 0 || size > value.len() { + return None; + } + std::ffi::CStr::from_bytes_with_nul(&value[..size]) + .ok()? + .to_str() + .ok() + .filter(|name| !name.is_empty()) + .map(str::to_owned) +} + +#[cfg(test)] +mod tests { + use super::count_non_efficiency_cores; + + #[test] + fn includes_super_and_performance_cores() { + assert_eq!(count(&[("Super", 6), ("Performance", 12)]), Some(18)); + assert_eq!(count(&[("Performance", 12), ("Super", 6)]), Some(18)); + } + + #[test] + fn excludes_efficiency_cores_in_every_position() { + assert_eq!(count(&[("Performance", 12), ("Efficiency", 4)]), Some(12)); + assert_eq!(count(&[("Efficiency", 4), ("Performance", 4)]), Some(4)); + assert_eq!( + count(&[("Super", 6), ("Efficiency", 4), ("Performance", 12)]), + Some(18) + ); + assert_eq!(count(&[("Performance", 8), ("efficiency", 4)]), Some(8)); + } + + #[test] + fn accepts_standard_cores() { + assert_eq!(count(&[("Performance", 8)]), Some(8)); + assert_eq!(count(&[("Standard", 8)]), Some(8)); + } + + #[test] + fn unknown_core_types_fall_back() { + assert_eq!(count(&[("Faster", 6), ("Performance", 12)]), None); + assert_eq!(count(&[("Super", 6), ("PowerSaving", 4)]), None); + } + + #[test] + fn incomplete_or_unusable_topology_falls_back() { + assert_eq!(count(&[]), None); + assert_eq!(count(&[("Efficiency", 4)]), None); + assert_eq!(count(&[("Performance", 0)]), None); + assert_eq!( + count_non_efficiency_cores([Some(("Super".into(), 6)), None]), + None + ); + assert_eq!(count(&[("Super", usize::MAX), ("Performance", 1)]), None); + } + + fn count(levels: &[(&str, usize)]) -> Option { + count_non_efficiency_cores( + levels + .iter() + .map(|&(name, count)| Some((name.to_owned(), count))), + ) + } +} diff --git a/crates/fff-core/src/score.rs b/crates/fff-core/src/score.rs index 5bed0ea94..ffc73c637 100644 --- a/crates/fff-core/src/score.rs +++ b/crates/fff-core/src/score.rs @@ -70,15 +70,23 @@ fn match_fuzzy_parts( return vec![]; } - // Index-based resolver: no `Vec<&FileItem>` materialization for either the - // full list or the per-part subsets, and a single monomorphized hot loop. - let first_part_matches = neo_frizbee::match_range_parallel_resolved( - valid_parts[0], + let reverse_pair = valid_parts.len() == 2 && valid_parts[1].len() > valid_parts[0].len(); + let first_part = usize::from(reverse_pair); + let mut first_options = *options; + if reverse_pair { + first_options.max_typos = options + .max_typos + .map(|t| t.min(valid_parts[1].len() as u16)); + } + + // Narrow two-part queries with the longer part before scoring the shorter part. + let first_part_matches = match_file_range( + valid_parts[first_part], working_files.len(), &|index, buf: &mut [*const u8; MAX_PATH_CHUNKS]| { resolve_file_chunks(working_files.index(index as usize), arena, buf) }, - options, + &first_options, max_threads, ); @@ -88,14 +96,20 @@ fn match_fuzzy_parts( let total_parts = valid_parts.len() as u32; let mut matches = first_part_matches; - for part in valid_parts[1..].iter() { + for (part_index, part) in valid_parts + .iter() + .enumerate() + .filter(|(i, _)| *i != first_part) + { let mut part_options = *options; - part_options.max_typos = options.max_typos.map(|t| t.min(part.len() as u16)); + if part_index > 0 { + part_options.max_typos = options.max_typos.map(|t| t.min(part.len() as u16)); + } // Match only the files that survived the previous round, addressed // through the previous matches without collecting a subset. let survivors = &matches; - let sub_matches = neo_frizbee::match_range_parallel_resolved( + let sub_matches = match_file_range( part, survivors.len(), &|index, buf: &mut [*const u8; MAX_PATH_CHUNKS]| { @@ -121,7 +135,11 @@ fn match_fuzzy_parts( neo_frizbee::Match { index: prev.index, score: avg.min(u16::MAX as u32) as u16, - end_col: prev.end_col, // keep first part's position for filename bonus + end_col: if reverse_pair { + sm.end_col + } else { + prev.end_col + }, exact: prev.exact && sm.exact, } }) @@ -756,7 +774,6 @@ fn match_and_score_in_arena_inner<'a, const WITH_CURRENT_FILE: bool>( let ScoreBuffers { dir_buf, fname_buf, - path_buf, last_dir_penalty, } = bufs; let file_idx = path_match.index as usize; @@ -867,19 +884,10 @@ fn match_and_score_in_arena_inner<'a, const WITH_CURRENT_FILE: bool>( } }; - // Path alignment bonus: when the query looks like a file path, - // reward candidates whose path closely matches the typed query. - // Uses suffix overlap — bytes matching from the end. A full prefix - // match is just the 100% coverage case, so no separate branch needed. - let path_alignment_bonus = if query_contains_path_separator { - let rel_path = file.path.read_to_buf(arena, path_buf); - let path_bytes = rel_path.as_bytes(); - let common_suffix = main_needle - .iter() - .rev() - .zip(path_bytes.iter().rev()) - .take_while(|(n, p): &(&u8, &u8)| n.eq_ignore_ascii_case(p)) - .count(); + let path_alignment_bonus = if query_contains_path_separator && main_needle.len() > 10 { + let common_suffix = file + .path + .common_suffix_len_ignore_ascii_case(arena, main_needle); let needle_len = main_needle.len(); if common_suffix > 10 && needle_len > 0 { @@ -963,7 +971,6 @@ const PARALLEL_SCORING_MIN_MATCHES: usize = 8192; struct ScoreBuffers { dir_buf: String, fname_buf: String, - path_buf: [u8; crate::simd_path::PATH_BUF_SIZE], last_dir_penalty: Option<(u32, i32)>, } @@ -972,7 +979,6 @@ impl ScoreBuffers { Self { dir_buf: String::with_capacity(64), fname_buf: String::with_capacity(32), - path_buf: [0u8; crate::simd_path::PATH_BUF_SIZE], last_dir_penalty: None, } } @@ -1145,12 +1151,225 @@ fn sort_and_paginate<'a, S>( (items, scores, total_matched) } +fn match_file_range( + needle: &str, + len: usize, + resolve: &F, + options: &neo_frizbee::Config, + max_threads: usize, +) -> Vec +where + F: Fn(u32, &mut [*const u8; MAX_PATH_CHUNKS]) -> Option<(usize, u16)> + Sync, +{ + // The rayon path below flattens per-worker results and never sorts, so + // any sorted strategy stays on frizbee's own k-merge implementation. + let unsorted = matches!(options.sort, neo_frizbee::SortStrategy::Unsorted); + if len < 32_768 || !unsorted { + return neo_frizbee::match_range_parallel_resolved( + needle, + len, + resolve, + options, + max_threads, + ); + } + assert!( + len <= u32::MAX as usize, + "too many files for fuzzy matching" + ); + let mut matcher = neo_frizbee::Matcher::new(needle, options); + let max_threads = if max_threads == 0 { + std::thread::available_parallelism() + .map(|n| n.get().saturating_sub(2)) + .unwrap_or(1) + } else { + max_threads + }; + let workers = max_threads.max(1).min(len.div_ceil(2000)); + if workers <= 1 { + return matcher.match_range_resolved(len, resolve); + } + let chunk_size = len.div_ceil(workers * 4).clamp(2048, 16_384); + let next = std::sync::atomic::AtomicUsize::new(0); + (0..workers) + .into_par_iter() + .map(|_| { + let mut matcher = matcher.clone(); + let mut matches = Vec::new(); + loop { + let start = next.fetch_add(chunk_size, std::sync::atomic::Ordering::Relaxed); + if start >= len { + break; + } + let end = (start + chunk_size).min(len); + matcher.match_range_resolved_into(start as u32..end as u32, resolve, &mut matches); + } + matches + }) + .collect::>() + .into_iter() + .flatten() + .collect() +} + #[cfg(test)] mod tests { use super::*; use crate::types::PaginationArgs; use fff_query_parser::QueryParser; + #[test] + fn longer_second_part_preserves_matches_and_first_part_positions() { + let paths: Vec<_> = [ + "src/controller.rs", + "src/contrller.rs", + "test/src.rs", + "日本語/controller.tsx", + "src/utils.rs", + "src/controller/controller.rs", + ] + .into_iter() + .map(String::from) + .collect(); + let mut files: Vec<_> = paths + .iter() + .map(|path| FileItem::new_raw(path.rfind('/').unwrap() as u16 + 1, 0, 0, None, false)) + .collect(); + let (store, strings) = + crate::simd_path::build_chunked_path_store_from_strings(&paths, &files); + for (file, path) in files.iter_mut().zip(strings) { + file.set_path(path); + } + let arena = store.as_arena_ptr(); + let working = FileItems::All(&files); + for parts in [ + ["src", "controller"], + ["rs", "contrller"], + ["rs", "日本語"], + ["no", "absent"], + ["src", "src/controller.rs"], + ] { + for max_typos in [0, 2, 4] { + let options = neo_frizbee::Config { + max_typos: Some(max_typos), + sort: neo_frizbee::SortStrategy::Unsorted, + ..Default::default() + }; + let first = match_fuzzy_parts(&parts[..1], &working, &options, 1, arena); + let mut second_options = options; + second_options.max_typos = Some(max_typos.min(parts[1].len() as u16)); + let second = match_fuzzy_parts(&parts[1..], &working, &second_options, 1, arena); + let mut expected: Vec<_> = first + .iter() + .filter_map(|a| { + let b = second.iter().find(|b| b.index == a.index)?; + Some(( + a.index, + ((a.score as u32 + b.score as u32) / 2) as u16, + a.end_col, + a.exact && b.exact, + )) + }) + .collect(); + let mut actual: Vec<_> = match_fuzzy_parts(&parts, &working, &options, 4, arena) + .iter() + .map(|m| (m.index, m.score, m.end_col, m.exact)) + .collect(); + expected.sort_unstable(); + actual.sort_unstable(); + assert_eq!(actual, expected, "{parts:?}, typos={max_typos}"); + } + } + } + + #[test] + fn fuzzy_parallel_and_sequential_scores_match() { + let paths: Vec<_> = (0..36_000) + .map(|index| { + format!( + "src/components/group_{}/controller_é_{index}.tsx", + index / 100 + ) + }) + .collect(); + let mut files: Vec<_> = paths + .iter() + .enumerate() + .map(|(index, path)| { + let mut file = FileItem::new_raw( + path.rfind('/').unwrap() as u16 + 1, + 0, + index as u64, + Some(git2::Status::WT_MODIFIED), + false, + ); + file.parent_dir_index = (index / 100) as u32; + file.access_frecency_score = (index % 97) as i16; + file.modification_frecency_score = (index % 29) as i16; + file.git_recency_score = (index % 13) as i16; + file.set_deleted(index % 97 == 0); + file + }) + .collect(); + let (store, strings) = + crate::simd_path::build_chunked_path_store_from_strings(&paths, &files); + for (file, path) in files.iter_mut().zip(strings) { + file.set_path(path); + } + let arena = store.as_arena_ptr(); + let parser = QueryParser::default(); + for text in [ + "mo", + "controller", + "src controller", + "group_0/controller_é_1.tsx", + "*.tsx", + ] { + let query = parser.parse(text); + for current_file in [None, Some(paths[1].as_str())] { + let mut context = ScoringContext { + query: &query, + max_threads: 4, + max_typos: 2, + project_path: None, + current_file, + last_same_query_match: None, + combo_boost_score_multiplier: 100, + min_combo_count: 3, + pagination: PaginationArgs::default(), + }; + let mut parallel = match_and_score_in_arena(&files, &context, arena); + context.max_threads = 1; + let mut sequential = match_and_score_in_arena(&files, &context, arena); + parallel.sort_unstable_by_key(|(file, _)| file.modified); + sequential.sort_unstable_by_key(|(file, _)| file.modified); + assert_eq!(parallel.len(), sequential.len(), "{text}"); + for ((a, a_score), (b, b_score)) in parallel.iter().zip(&sequential) { + assert!(std::ptr::eq(*a, *b)); + assert_eq!(format!("{a_score:?}"), format!("{b_score:?}"), "{text}"); + } + let (parallel_items, parallel_scores, _) = + sort_and_paginate(parallel, &context, |score| score.total); + let (sequential_items, sequential_scores, _) = + sort_and_paginate(sequential, &context, |score| score.total); + assert_eq!( + parallel_items + .iter() + .map(|f| f.modified) + .collect::>(), + sequential_items + .iter() + .map(|f| f.modified) + .collect::>() + ); + assert_eq!( + format!("{parallel_scores:?}"), + format!("{sequential_scores:?}") + ); + } + } + } + #[test] fn frecency_parallel_and_sequential_scores_match() { let files: Vec<_> = (0..33_000) diff --git a/crates/fff-core/src/simd_path.rs b/crates/fff-core/src/simd_path.rs index 6e3057000..f131f54e6 100644 --- a/crates/fff-core/src/simd_path.rs +++ b/crates/fff-core/src/simd_path.rs @@ -120,6 +120,49 @@ impl ChunkedString { }) } + #[inline] + pub(crate) fn common_suffix_len_ignore_ascii_case( + &self, + arena: ArenaPtr, + other: &[u8], + ) -> usize { + let indices = self.indices(arena); + let mut end = self.byte_len as usize; + let mut matched = 0; + while end > 0 && matched < other.len() { + let chunk = (end - 1) / SIMD_CHUNK_BYTES; + let len = end - chunk * SIMD_CHUNK_BYTES; + let bytes = + unsafe { core::slice::from_raw_parts(arena.chunk_ptr(indices[chunk]), len) }; + for (&a, &b) in bytes + .iter() + .rev() + .zip(other[..other.len() - matched].iter().rev()) + { + if !a.eq_ignore_ascii_case(&b) { + return matched; + } + matched += 1; + } + end -= len; + } + matched + } + + #[inline] + pub(crate) fn has_extension(&self, arena: ArenaPtr, extension: &str) -> bool { + let filename_len = (self.byte_len - self.filename_offset) as usize; + if filename_len <= extension.len() + 1 { + return false; + } + let dot = self.byte_len as usize - extension.len() - 1; + let chunk = self.indices(arena)[dot / SIMD_CHUNK_BYTES]; + let byte = unsafe { *arena.chunk_ptr(chunk).add(dot % SIMD_CHUNK_BYTES) }; + byte == b'.' + && self.common_suffix_len_ignore_ascii_case(arena, extension.as_bytes()) + == extension.len() + } + #[inline] fn indices<'a>(&self, arena: ArenaPtr) -> &'a [u32] { let count = self.chunk_count(); @@ -402,6 +445,36 @@ pub(crate) fn build_chunked_path_store_from_strings( mod tests { use super::*; + #[test] + fn chunked_suffix_matches_contiguous_bytes() { + let paths = [ + "", + "src/main.rs", + "src/components/Controller.tsx", + "src/日本語/éController.tsx", + "abcdefghijklmnopABCDEFGHIJKLMNOP.rs", + ]; + let (store, strings, _) = build_test_store(&paths); + for (path, chunked) in paths.iter().zip(&strings) { + for candidate in paths { + for start in 0..=candidate.len() { + let suffix = &candidate.as_bytes()[start..]; + let expected = path + .as_bytes() + .iter() + .rev() + .zip(suffix.iter().rev()) + .take_while(|(a, b)| a.eq_ignore_ascii_case(b)) + .count(); + assert_eq!( + chunked.common_suffix_len_ignore_ascii_case(store.as_arena_ptr(), suffix), + expected + ); + } + } + } + } + fn make_file_item(path: &str) -> crate::types::FileItem { let filename_start = path .rfind(std::path::is_separator) @@ -434,6 +507,42 @@ mod tests { assert_eq!(store.unique_chunks(), 0); } + #[test] + fn chunked_extensions_match_contiguous_filenames() { + let paths = [ + "", + ".rs", + "a.", + "src/long_directory/main.RS", + "src/é.日本語", + "dir.rs/file", + "abcdefghijklmn.rs", + "abcdefghijklmno.rs", + "abcdefghijklmnop.rs", + "a.tar.gz", + ]; + let (store, strings, _) = build_test_store(&paths); + for (path, chunked) in paths.iter().zip(&strings) { + let filename = path.rsplit('/').next().unwrap(); + for extension in [ + "", + "rs", + "RS", + "gz", + "tar.gz", + "日本語", + "語", + "a/more.than.a.chunk", + ] { + assert_eq!( + chunked.has_extension(store.as_arena_ptr(), extension), + crate::index::constraints::file_has_extension(filename, extension), + "{path}, {extension}" + ); + } + } + } + #[test] fn test_chunked_store_basic() { let (store, strings, _files) = diff --git a/crates/fff-core/src/types.rs b/crates/fff-core/src/types.rs index cae200293..45d0578f2 100644 --- a/crates/fff-core/src/types.rs +++ b/crates/fff-core/src/types.rs @@ -750,11 +750,7 @@ impl FileItem { self.cached_content() } - /// Get file content for searching — **always returns content** for eligible - /// files, even when the persistent cache budget is exhausted. - /// - /// The caller provides a reusable `path_buf` (pre-filled with `base_path/`) - /// and its `base_len` to avoid allocations when constructing the absolute path. + // Uncached files remain searchable after the persistent cache budget fills. #[inline] pub(crate) fn get_content_for_search<'a>( &'a self, @@ -763,6 +759,7 @@ impl FileItem { arena: ArenaPtr, base_path: &Path, budget: &ContentCacheBudget, + base_dir: Option<&std::fs::File>, ) -> Option<&'a [u8]> { #[cfg(not(target_os = "windows"))] { @@ -779,11 +776,10 @@ impl FileItem { return None; } - let abs = self.absolute_path(arena, base_path); + let mut file = self.open_for_search(arena, base_path, base_dir)?; #[cfg(not(target_os = "windows"))] if self.size >= FRESH_MMAP_THRESHOLD { - let file = std::fs::File::open(&abs).ok()?; let mmap = unsafe { memmap2::Mmap::map(&file) }.ok()?; let stored = mmap_slot.insert(mmap); return Some(&stored[..]); @@ -794,10 +790,46 @@ impl FileItem { let len = self.size as usize; buf.resize(len, 0); - let mut file = std::fs::File::open(&abs).ok()?; file.read_exact(buf).ok()?; Some(buf.as_slice()) } + + fn open_for_search( + &self, + arena: ArenaPtr, + base_path: &Path, + #[cfg_attr(not(unix), allow(unused_variables))] base_dir: Option<&std::fs::File>, + ) -> Option { + let mut path_buf = [0u8; PATH_BUF_SIZE]; + #[cfg(unix)] + if let Some(dir) = base_dir + && self.relative_path_len() < path_buf.len() + { + use std::os::fd::{AsRawFd, FromRawFd}; + let path = self.write_relative_cstr(arena, &mut path_buf); + // The directory FD stays borrowed; a successful openat returns an owned FD. + loop { + let fd = unsafe { + libc::openat( + dir.as_raw_fd(), + path.as_ptr(), + libc::O_RDONLY | libc::O_CLOEXEC, + ) + }; + if fd >= 0 { + return Some(unsafe { std::fs::File::from_raw_fd(fd) }); + } + if std::io::Error::last_os_error().kind() != std::io::ErrorKind::Interrupted { + return None; + } + } + } + if base_path.as_os_str().len() + self.relative_path_len() + 1 < path_buf.len() { + std::fs::File::open(self.write_absolute_path(arena, base_path, &mut path_buf)).ok() + } else { + std::fs::File::open(self.absolute_path(arena, base_path)).ok() + } + } } /// Per-thread scratch slot owning a transient mmap returned from @@ -809,6 +841,11 @@ pub type MmapSlot = Option; pub type MmapSlot = (); impl Constrainable for FileItem { + #[inline] + fn has_extension(&self, arena: ArenaPtr, extension: &str, _scratch: &mut String) -> bool { + self.path.has_extension(arena, extension) + } + #[inline] fn write_file_name(&self, arena: ArenaPtr, out: &mut String) { self.path.write_filename_to(arena, out); @@ -1043,6 +1080,42 @@ impl Default for ContentCacheBudget { mod content_cache_budget_tests { use super::*; + #[cfg(unix)] + #[test] + fn uncached_search_reads_relative_and_absolute_paths() { + let root = tempfile::tempdir().unwrap(); + std::fs::create_dir(root.path().join("nested")).unwrap(); + let path = "nested/é file.txt"; + let directory = std::fs::File::open(root.path()).unwrap(); + let budget = ContentCacheBudget::with_max_files(0); + for size in [17, FRESH_MMAP_THRESHOLD as usize + 17] { + let contents = vec![b'x'; size]; + std::fs::write(root.path().join(path), &contents).unwrap(); + let mut file = FileItem::new_raw(7, size as u64, 0, None, false); + let (store, strings) = crate::simd_path::build_chunked_path_store_from_strings( + &[path.to_string()], + std::slice::from_ref(&file), + ); + file.set_path(strings.into_iter().next().unwrap()); + for base_dir in [Some(&directory), None] { + let mut buf = Vec::new(); + let mut mmap = MmapSlot::default(); + let actual = file + .get_content_for_search( + &mut buf, + &mut mmap, + store.as_arena_ptr(), + root.path(), + &budget, + base_dir, + ) + .unwrap(); + assert_eq!(actual, contents); + assert_eq!(budget.cached_count.load(Ordering::Relaxed), 0); + } + } + } + #[test] fn with_max_files_applies_the_cap_verbatim() { // regression: the cap used to be routed through new_for_repo, which From 7ff01fef26cf21ca32cfbf55de8d75752f6bb1b6 Mon Sep 17 00:00:00 2001 From: Dmitriy Kovalenko Date: Sat, 26 Sep 2026 12:03:25 -0700 Subject: [PATCH 2/3] fix(test): use write-capable handle for set_times on Windows --- crates/fff-core/src/grep/grep_tests.rs | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/crates/fff-core/src/grep/grep_tests.rs b/crates/fff-core/src/grep/grep_tests.rs index d736e1b55..a90b57045 100644 --- a/crates/fff-core/src/grep/grep_tests.rs +++ b/crates/fff-core/src/grep/grep_tests.rs @@ -59,7 +59,10 @@ fn sparse_matches_after_empty_batches_preserve_pagination() { "ordinary content\n" }; std::fs::write(&path, content).unwrap(); - std::fs::File::open(path) + // `File::open` is read-only and Windows requires write access to set file times. + std::fs::OpenOptions::new() + .write(true) + .open(path) .unwrap() .set_times(std::fs::FileTimes::new().set_modified( std::time::UNIX_EPOCH + std::time::Duration::from_secs(1_700_000_000 + index), From f296c4b9e193d99e188bdf87207f18170722a5ae Mon Sep 17 00:00:00 2001 From: Dmitriy Kovalenko Date: Sat, 26 Sep 2026 20:16:44 -0700 Subject: [PATCH 3/3] perf(grep): cap chunk growth for strong prefilters to bound over-scan --- crates/fff-core/src/grep/grep.rs | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/crates/fff-core/src/grep/grep.rs b/crates/fff-core/src/grep/grep.rs index df696d571..0e08590c4 100644 --- a/crates/fff-core/src/grep/grep.rs +++ b/crates/fff-core/src/grep/grep.rs @@ -578,11 +578,15 @@ where let mut page_filled = false; let mut aborted = false; - // Grow empty batches to avoid repeated barriers on absent queries. - // A strong prefilter returns to small batches as soon as matches appear. + // Grow empty batches to avoid repeated barriers. A strong prefilter caps + // growth low (bounds over-scan past a filled page) and resets on a match. let base_chunk = rayon::current_num_threads() * 4; let prefilter_strong = ctx.total_files > 0 && files_to_search.len() * 2 < ctx.total_files; - let max_chunk = (base_chunk * 256).max(8 * 1024); + let max_chunk = if prefilter_strong { + base_chunk * 8 + } else { + (base_chunk * 256).max(8 * 1024) + }; let mut chunk_size = base_chunk; let mut chunk_start = 0;