From aace86ef16a172f3a5246442621bb9827a512912 Mon Sep 17 00:00:00 2001 From: Dmitriy Kovalenko Date: Thu, 24 Sep 2026 21:18:01 -0700 Subject: [PATCH 1/2] perf(search): skip rayon on small indexes, cache directory distance, add fuzzy search bench --- Makefile | 8 +- crates/fff-core/Cargo.toml | 5 +- crates/fff-core/benches/fuzzy_search_bench.rs | 122 ++++++ crates/fff-core/benches/grep_bench.rs | 67 +++ crates/fff-core/benches/support/mod.rs | 45 ++ crates/fff-core/src/grep/fuzzy_grep.rs | 6 +- crates/fff-core/src/grep/sink.rs | 46 -- crates/fff-core/src/lib.rs | 2 + crates/fff-core/src/match_offsets.rs | 81 ++++ crates/fff-core/src/path_utils.rs | 67 ++- crates/fff-core/src/score.rs | 396 +++++++++++++----- crates/fff-core/src/simd_path.rs | 89 +++- crates/fff-core/src/types.rs | 11 +- 13 files changed, 771 insertions(+), 174 deletions(-) create mode 100644 crates/fff-core/benches/fuzzy_search_bench.rs create mode 100644 crates/fff-core/benches/support/mod.rs create mode 100644 crates/fff-core/src/match_offsets.rs diff --git a/Makefile b/Makefile index a28b9cc7f..4a03590bf 100644 --- a/Makefile +++ b/Makefile @@ -16,7 +16,7 @@ SHELL := bash # string rather than the literal `-o` / `pipefail` tokens. .SHELLFLAGS := -o pipefail -euc -.PHONY: build build-c-lib install uninstall test test-rust test-rescan test-rescan-known-defects rescan-probe test-c-smoke test-c-api test-lua test-lua-snap test-version test-bun test-node prepare-bun prepare-bun-packaged prepare-node set-npm-version header test-stress test-stress-seeded test-stress-random test-stress-regressions test-stress-repos test-node-stress sync-js-api sync-js-api-check bump-homebrew-formula bump-install-mcp-sh test-bun-compile +.PHONY: build build-c-lib install uninstall test test-rust test-rescan test-rescan-known-defects rescan-probe test-c-smoke test-c-api test-lua test-lua-snap test-version test-bun test-node prepare-bun prepare-bun-packaged prepare-node set-npm-version header test-stress test-stress-seeded test-stress-random test-stress-regressions test-stress-repos test-node-stress sync-js-api sync-js-api-check bump-homebrew-formula bump-install-mcp-sh test-bun-compile bench-search bench-grep all: format test lint @@ -86,6 +86,12 @@ test-setup: test-rust: cargo test --workspace --no-default-features --features zlob --exclude fff-nvim --exclude fff-python +bench-search: + cargo bench -p fff-search --no-default-features --features zlob --bench fuzzy_search_bench -- $(BENCH_ARGS) + +bench-grep: + cargo bench -p fff-search --no-default-features --features zlob --bench grep_bench -- $(BENCH_ARGS) + # Watcher rescan harness: asserts that editing, build output, git activity and # preview reads all stay on the incremental path instead of re-walking the tree. test-rescan: diff --git a/crates/fff-core/Cargo.toml b/crates/fff-core/Cargo.toml index ae6271f15..96921f6e3 100644 --- a/crates/fff-core/Cargo.toml +++ b/crates/fff-core/Cargo.toml @@ -34,6 +34,10 @@ required-features = ["zlob"] name = "grep_bench" harness = false +[[bench]] +name = "fuzzy_search_bench" +harness = false + [features] # `ripgrep` is the pure-Rust walker/glob backend and is on by default so # consumers build without a Zig toolchain. CI/release opt into zlob via @@ -102,4 +106,3 @@ proptest = { version = "1", default-features = false, features = ["std", "fork"] rand = { version = "0.8", features = ["small_rng"] } tempfile = "3.8" tracing-subscriber = { version = "0.3", features = ["env-filter"] } - diff --git a/crates/fff-core/benches/fuzzy_search_bench.rs b/crates/fff-core/benches/fuzzy_search_bench.rs new file mode 100644 index 000000000..c71f9d9e9 --- /dev/null +++ b/crates/fff-core/benches/fuzzy_search_bench.rs @@ -0,0 +1,122 @@ +use criterion::{BenchmarkId, Criterion, criterion_group, criterion_main}; +use fff_search::{FilePicker, FilePickerOptions, FuzzySearchOptions, PaginationArgs, QueryParser}; +use std::{hint::black_box, time::Duration}; + +mod support; + +fn bench_fuzzy_search(c: &mut Criterion) { + if let Some(picker) = support::repo_picker() { + let current_file = + std::env::var("FFF_BENCH_CURRENT_FILE").unwrap_or_else(|_| "README.md".into()); + assert!(picker.base_path().join(¤t_file).is_file()); + bench_picker(c, &picker, "fuzzy_search_repo", ¤t_file, true); + return; + } + for count in [1_000, 10_000, 100_000] { + let (_root, picker) = make_picker(count); + bench_picker( + c, + &picker, + &format!("fuzzy_search_{count}"), + "src/components/group_0/controller_0.rs", + false, + ); + } +} + +fn bench_picker( + c: &mut Criterion, + picker: &FilePicker, + name: &str, + current_file: &str, + repo: bool, +) { + let parser = QueryParser::default(); + let mut group = c.benchmark_group(name); + group.sample_size(20); + group.warm_up_time(Duration::from_millis(if repo { 1000 } else { 300 })); + group.measurement_time(Duration::from_secs(if repo { 3 } else { 1 })); + + let path_query = + std::env::var("FFF_BENCH_PATH_QUERY").unwrap_or_else(|_| "src/components".into()); + let multipart_query = std::env::var("FFF_BENCH_MULTIPART_QUERY") + .unwrap_or_else(|_| "components controller".into()); + for (name, text) in [ + ("empty", ""), + ("short", "mo"), + ("filename", "controller"), + ("path", path_query.as_str()), + ("multipart", multipart_query.as_str()), + ("extension", "*.rs"), + ] { + let query = parser.parse(text); + for current_file in [None, Some(current_file)] { + let options = FuzzySearchOptions { + max_threads: 4, + current_file, + pagination: PaginationArgs { + offset: 0, + limit: 50, + }, + ..Default::default() + }; + let suffix = if current_file.is_some() { + "current_file" + } else { + "no_current_file" + }; + if repo { + let result = picker.fuzzy_search(&query, None, options); + eprintln!("Query {name}/{suffix}: matched={}", result.total_matched); + } + group.bench_function(BenchmarkId::new(name, suffix), |b| { + b.iter(|| black_box(picker.fuzzy_search(black_box(&query), None, options))); + }); + } + } + group.finish(); +} + +fn make_picker(count: usize) -> (tempfile::TempDir, FilePicker) { + let root = tempfile::tempdir().unwrap(); + let areas = [ + "src/components", + "src/core", + "tests/integration", + "packages/server", + ]; + let names = ["controller", "model", "service", "utils", "index"]; + let extensions = ["rs", "ts", "lua", "py"]; + for group in 0..count.div_ceil(100) { + let directory = root + .path() + .join(format!("{}/group_{group}", areas[group % areas.len()])); + std::fs::create_dir_all(&directory).unwrap(); + for index in group * 100..((group + 1) * 100).min(count) { + let name = format!( + "{}_{index}.{}", + names[index % names.len()], + extensions[index % extensions.len()] + ); + let file = std::fs::File::create(directory.join(name)).unwrap(); + let modified = + std::time::UNIX_EPOCH + Duration::from_secs(1_700_000_000 + (index % 1000) as u64); + file.set_times(std::fs::FileTimes::new().set_modified(modified)) + .unwrap(); + } + } + let mut picker = FilePicker::new(FilePickerOptions { + base_path: root.path().to_string_lossy().into_owned(), + enable_mmap_cache: false, + enable_content_indexing: false, + watch: false, + ..Default::default() + }) + .unwrap(); + picker.collect_files().unwrap(); + assert_eq!(picker.live_file_count(), count); + (root, picker) +} + +criterion_group!(benches, bench_fuzzy_search); +criterion_main!(benches); diff --git a/crates/fff-core/benches/grep_bench.rs b/crates/fff-core/benches/grep_bench.rs index e04db4d0e..3e1a01e29 100644 --- a/crates/fff-core/benches/grep_bench.rs +++ b/crates/fff-core/benches/grep_bench.rs @@ -3,6 +3,8 @@ use fff_search::file_picker::{FilePicker, FilePickerOptions}; use fff_search::{GrepMode, GrepSearchOptions, parse_grep_query}; use std::io::Write; +mod support; + /// Synthetic repo: half the files contain the needle on every line (stresses /// the per-match find/highlight path), half are pure noise (stresses the /// whole-file prefilter path). @@ -41,6 +43,10 @@ fn options(mode: GrepMode) -> GrepSearchOptions { } fn bench_grep(c: &mut Criterion) { + if let Some(picker) = support::repo_picker() { + bench_repo(c, &picker); + return; + } let dir = tempfile::tempdir().unwrap(); setup_repo(dir.path()); @@ -98,6 +104,67 @@ fn bench_grep(c: &mut Criterion) { }); }); + let fuzzy_opts = options(GrepMode::Fuzzy); + for (name, text) in [ + ("fuzzy_exact_many_matches", "controller"), + ("fuzzy_typo_many_matches", "contrller"), + ] { + let query = parse_grep_query(text); + group.bench_function(name, |b| { + b.iter(|| { + let result = picker.grep(&query, &fuzzy_opts); + assert_eq!(result.files_with_matches, 400); + std::hint::black_box(result.matches.len()) + }); + }); + } + + group.finish(); +} + +fn bench_repo(c: &mut Criterion, picker: &FilePicker) { + let mut group = c.benchmark_group("grep_repo"); + group.sample_size(10); + group.sampling_mode(criterion::SamplingMode::Flat); + group.warm_up_time(std::time::Duration::from_secs(1)); + group.measurement_time(std::time::Duration::from_secs(3)); + for (name, text, mode) in [ + ("plain_sensitive", "Controller", GrepMode::PlainText), + ("plain_insensitive", "controller", GrepMode::PlainText), + ( + "plain_no_matches", + "FFF_BENCH_absent_f891c75d", + GrepMode::PlainText, + ), + ("regex", "Contr[a-z]+ller", GrepMode::Regex), + ("fuzzy_exact", "controller", GrepMode::Fuzzy), + ("fuzzy_typo", "contrller", GrepMode::Fuzzy), + ] { + let query = parse_grep_query(text); + let options = GrepSearchOptions { + max_file_size: 10 * 1024 * 1024, + mode, + ..Default::default() + }; + let result = picker.grep(&query, &options); + eprintln!( + "Query {name}: matches={} files_with_matches={} searched={} filtered={}", + result.matches.len(), + result.files_with_matches, + result.total_files_searched, + result.filtered_file_count + ); + assert!(result.regex_fallback_error.is_none()); + if name == "plain_no_matches" { + assert!(result.matches.is_empty()); + } else { + assert!(!result.matches.is_empty(), "query {name} must match"); + } + drop(result); + group.bench_function(name, |b| { + b.iter(|| std::hint::black_box(picker.grep(&query, &options))); + }); + } group.finish(); } diff --git a/crates/fff-core/benches/support/mod.rs b/crates/fff-core/benches/support/mod.rs new file mode 100644 index 000000000..5e711b8bc --- /dev/null +++ b/crates/fff-core/benches/support/mod.rs @@ -0,0 +1,45 @@ +use fff_search::{FilePicker, FilePickerOptions}; +use std::hash::{DefaultHasher, Hash, Hasher}; + +pub(super) fn repo_picker() -> Option { + let path = std::env::var_os("FFF_BENCH_PATH")?; + let path = std::fs::canonicalize(path).expect("FFF_BENCH_PATH must exist"); + assert!(path.is_dir(), "FFF_BENCH_PATH must be a directory"); + eprintln!("Indexing {}", path.display()); + let mut picker = FilePicker::new(FilePickerOptions { + base_path: path.to_str().expect("UTF-8 repository path").into(), + enable_mmap_cache: false, + enable_content_indexing: false, + watch: false, + ..Default::default() + }) + .unwrap(); + picker.collect_files().unwrap(); + assert!( + picker.live_file_count() > 0, + "repository must contain files" + ); + + let mut files: Vec<_> = picker + .get_files() + .iter() + .map(|file| { + ( + file.relative_path(&picker), + file.size, + file.modified, + file.git_status.map(|status| status.bits()), + file.git_recency_score, + ) + }) + .collect(); + files.sort_unstable(); + let mut fingerprint = DefaultHasher::new(); + files.hash(&mut fingerprint); + eprintln!( + "Dataset: files={} fingerprint={:016x}", + files.len(), + fingerprint.finish() + ); + Some(picker) +} diff --git a/crates/fff-core/src/grep/fuzzy_grep.rs b/crates/fff-core/src/grep/fuzzy_grep.rs index c8f0e2d29..57e8931c5 100644 --- a/crates/fff-core/src/grep/fuzzy_grep.rs +++ b/crates/fff-core/src/grep/fuzzy_grep.rs @@ -1,3 +1,4 @@ +use crate::match_offsets::char_indices_to_byte_offsets; use crate::simd_path::ArenaPtr; use crate::types::{ContentCacheBudget, FileItem, MmapSlot}; use fff_grep::lines::LineStep; @@ -5,10 +6,7 @@ use rayon::prelude::*; use std::path::Path; use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; -use super::sink::{ - char_indices_to_byte_offsets, classify_definition, strip_line_terminators, - truncate_display_bytes, -}; +use super::sink::{classify_definition, strip_line_terminators, truncate_display_bytes}; use super::types::{GrepMatch, GrepResult, GrepSearchOptions}; #[allow(clippy::too_many_arguments)] diff --git a/crates/fff-core/src/grep/sink.rs b/crates/fff-core/src/grep/sink.rs index 3fbd28cc0..75e60e36d 100644 --- a/crates/fff-core/src/grep/sink.rs +++ b/crates/fff-core/src/grep/sink.rs @@ -190,52 +190,6 @@ pub(super) fn split_multiline_blob(display_bytes: &[u8]) -> (&[u8], Vec) } } -/// Convert character-position indices from neo_frizbee into byte-offset -/// pairs (start, end) suitable for `match_byte_offsets`. -/// -/// frizbee returns character positions (0-based index into the char -/// iterator). We need byte ranges because the UI renderer and Lua layer -/// use byte offsets for extmark highlights. -/// -/// Each matched character becomes its own (byte_start, byte_end) pair. -/// Adjacent characters are merged into a single contiguous range. -pub(super) fn char_indices_to_byte_offsets( - line: &str, - char_indices: &[u32], -) -> SmallVec<[(u32, u32); 4]> { - if char_indices.is_empty() { - return SmallVec::new(); - } - - // Build a map: char_index -> (byte_start, byte_end) for all chars. - // Iterating all chars is O(n) in the line length which is bounded by MAX_LINE_DISPLAY_LEN (512). - let char_byte_ranges: Vec<(usize, usize)> = line - .char_indices() - .map(|(byte_pos, ch)| (byte_pos, byte_pos + ch.len_utf8())) - .collect(); - - // Convert char indices to byte ranges, merging adjacent ranges - let mut result: SmallVec<[(u32, u32); 4]> = SmallVec::with_capacity(char_indices.len()); - - for &ci in char_indices { - let ci = ci as usize; - if ci >= char_byte_ranges.len() { - continue; // out of bounds (shouldn't happen with valid data) - } - let (start, end) = char_byte_ranges[ci]; - // Merge with previous range if adjacent - if let Some(last) = result.last_mut() - && last.1 == start as u32 - { - last.1 = end as u32; - continue; - } - result.push((start as u32, end as u32)); - } - - result -} - // copied from the rust u8 private method #[inline] const fn is_utf8_char_boundary(b: u8) -> bool { diff --git a/crates/fff-core/src/lib.rs b/crates/fff-core/src/lib.rs index d963603ca..6e590da5b 100644 --- a/crates/fff-core/src/lib.rs +++ b/crates/fff-core/src/lib.rs @@ -136,6 +136,8 @@ pub mod path_utils; pub mod types; pub use types::*; +mod match_offsets; + pub mod constants; /// Watcher rescan request accounting. diff --git a/crates/fff-core/src/match_offsets.rs b/crates/fff-core/src/match_offsets.rs new file mode 100644 index 000000000..e7105101f --- /dev/null +++ b/crates/fff-core/src/match_offsets.rs @@ -0,0 +1,81 @@ +use smallvec::SmallVec; + +pub(crate) fn char_indices_to_byte_offsets( + line: &str, + char_indices: &[u32], +) -> SmallVec<[(u32, u32); 4]> { + debug_assert!(char_indices.windows(2).all(|pair| pair[0] <= pair[1])); + let mut result: SmallVec<[(u32, u32); 4]> = SmallVec::new(); + let mut chars = line.char_indices().enumerate().peekable(); + + for &index in char_indices { + while chars.peek().is_some_and(|&(i, _)| i < index as usize) { + chars.next(); + } + let Some(&(_, (start, ch))) = chars.peek() else { + break; + }; + let end = (start + ch.len_utf8()) as u32; + if let Some(last) = result.last_mut() + && last.1 == start as u32 + { + last.1 = end; + } else { + result.push((start as u32, end)); + } + } + + result +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn offsets_match_character_ranges() { + for line in ["", "abcdefghij", "aé文🦀e\u{301}z", "é🦀a文bc"] { + let ranges: Vec<_> = line + .char_indices() + .map(|(start, ch)| (start as u32, (start + ch.len_utf8()) as u32)) + .collect(); + for mask in 0..1 << (ranges.len() + 1) { + let indices: Vec<_> = (0..=ranges.len()) + .filter(|i| mask & (1 << i) != 0) + .map(|i| i as u32) + .collect(); + let actual = char_indices_to_byte_offsets(line, &indices); + let mut expected: Vec<(u32, u32)> = Vec::new(); + for &index in &indices { + let Some(&(start, end)) = ranges.get(index as usize) else { + continue; + }; + if let Some(last) = expected.last_mut() + && last.1 == start + { + last.1 = end; + } else { + expected.push((start, end)); + } + } + assert_eq!(actual.as_slice(), expected, "{line:?}, {indices:?}"); + } + } + } + + #[test] + fn contiguous_matches_stay_inline() { + let indices: Vec<_> = (0..100).collect(); + let ranges = char_indices_to_byte_offsets(&"x".repeat(100), &indices); + assert_eq!(ranges.as_slice(), &[(0, 100)]); + assert!(!ranges.spilled()); + } + + #[test] + fn duplicate_and_out_of_bounds_indices() { + assert_eq!( + char_indices_to_byte_offsets("aé🦀", &[0, 0, 1, 2, 3, u32::MAX]).as_slice(), + &[(0, 1), (0, 7)] + ); + } +} diff --git a/crates/fff-core/src/path_utils.rs b/crates/fff-core/src/path_utils.rs index a63d655b0..e4015e218 100644 --- a/crates/fff-core/src/path_utils.rs +++ b/crates/fff-core/src/path_utils.rs @@ -1,4 +1,28 @@ -use std::path::{Path, PathBuf}; +use std::path::{Components, Path, PathBuf}; + +pub(crate) struct DirectoryDistance<'a> { + components: Components<'a>, + depth: usize, +} + +impl<'a> DirectoryDistance<'a> { + pub(crate) fn new(current_file: &'a str) -> Self { + let directory = Path::new(current_file).parent().unwrap_or(Path::new("")); + let components = directory.components(); + let depth = components.clone().count(); + Self { components, depth } + } + + pub(crate) fn penalty(&self, candidate_dir: &str) -> i32 { + let common = self + .components + .clone() + .zip(Path::new(candidate_dir).components()) + .take_while(|(a, b)| a == b) + .count(); + -((self.depth - common).min(20) as i32) + } +} #[cfg(windows)] pub fn canonicalize(path: impl AsRef) -> std::io::Result { @@ -76,13 +100,8 @@ pub fn expand_tilde(path: &str) -> PathBuf { PathBuf::from(path) } -/// Calculate distance penalty based on directory proximity. -/// Returns a negative penalty score based on how far the candidate is from the current file. -/// -/// `candidate_dir` is the directory portion of the candidate path (e.g. `"src/components/"`). -/// It may have a trailing `/` which is stripped internally. -/// -/// Zero-allocation: walks both directory part iterators in lockstep. +/// Calculate the directory proximity penalty without allocating. +/// `candidate_dir` may include a trailing separator. pub fn calculate_distance_penalty(current_file: Option<&str>, candidate_dir: &str) -> i32 { let Some(current_path) = current_file else { return 0; @@ -134,6 +153,38 @@ pub fn calculate_distance_penalty(current_file: Option<&str>, candidate_dir: &st mod tests { use super::*; + #[test] + fn prepared_distance_matches_existing_scoring() { + let paths = [ + "", + "/", + "main.rs", + "./main.rs", + "src/lib.rs", + "src/./lib.rs", + "src//nested/lib.rs", + "src/../other/lib.rs", + "/src/nested/lib.rs", + "目录/é/lib.rs", + "C:\\src\\nested\\lib.rs", + "\\\\server\\share\\lib.rs", + ]; + for current in paths { + let distance = DirectoryDistance::new(current); + for candidate in paths { + assert_eq!( + distance.penalty(candidate), + calculate_distance_penalty(Some(current), candidate), + "{current:?}, {candidate:?}" + ); + } + } + assert_eq!( + DirectoryDistance::new(&format!("{}file", "dir/".repeat(40))).penalty("elsewhere/"), + -20 + ); + } + #[test] #[cfg(not(target_family = "windows"))] fn test_calculate_distance_penalty() { diff --git a/crates/fff-core/src/score.rs b/crates/fff-core/src/score.rs index e387a85e0..137849791 100644 --- a/crates/fff-core/src/score.rs +++ b/crates/fff-core/src/score.rs @@ -1,7 +1,8 @@ use crate::{ git::is_modified_status, index::constraints::apply_constraints, - path_utils::calculate_distance_penalty, + match_offsets::char_indices_to_byte_offsets, + path_utils::DirectoryDistance, simd_path::{ArenaPtr, MAX_PATH_CHUNKS}, sort_buffer::{sort_by_key_with_buffer, sort_with_buffer}, types::{DirItem, FileItem, Score, ScoringContext}, @@ -58,7 +59,7 @@ fn match_fuzzy_parts( max_threads: usize, arena: ArenaPtr, ) -> Vec { - let valid_parts: Vec<&str> = fuzzy_parts + let valid_parts: SmallVec<[&str; 4]> = fuzzy_parts .iter() .copied() .filter(|p| p.len() >= 2) @@ -169,10 +170,10 @@ pub(crate) fn fuzzy_match_byte_offsets_for_page<'q>( base_arena: ArenaPtr, overflow_arena: ArenaPtr, ) -> Vec> { - let parts: Vec<&str> = match &query.fuzzy_query { - FuzzyQuery::Text(text) if text.len() >= 2 => vec![*text], + let parts: SmallVec<[&str; 4]> = match &query.fuzzy_query { + FuzzyQuery::Text(text) if text.len() >= 2 => smallvec::smallvec![*text], FuzzyQuery::Parts(parts) => parts.iter().copied().filter(|p| p.len() >= 2).collect(), - _ => Vec::new(), + _ => SmallVec::new(), }; let mut ranges_by_item = vec![SmallVec::new(); items.len()]; @@ -180,16 +181,23 @@ pub(crate) fn fuzzy_match_byte_offsets_for_page<'q>( return ranges_by_item; } - let paths: Vec = items + let mut paths = String::with_capacity(items.iter().map(|item| item.relative_path_len()).sum()); + for item in items { + let arena = if item.is_overflow() { + overflow_arena + } else { + base_arena + }; + item.path.append_to_string(arena, &mut paths); + } + + let mut start = 0; + let path_strs: Vec<&str> = items .iter() .map(|item| { - let arena = if item.is_overflow() { - overflow_arena - } else { - base_arena - }; - let mut path = String::with_capacity(item.relative_path_len()); - item.write_relative_path_from_arena(arena, &mut path); + let end = start + item.relative_path_len(); + let path = &paths[start..end]; + start = end; path }) .collect(); @@ -208,8 +216,6 @@ pub(crate) fn fuzzy_match_byte_offsets_for_page<'q>( ..Default::default() }; - // Match on `&str` so this shares frizbee's instantiation with fuzzy grep. - let path_strs: Vec<&str> = paths.iter().map(String::as_str).collect(); for (idx, part) in parts.iter().copied().enumerate() { let mut part_config = config; if idx > 0 { @@ -219,7 +225,7 @@ pub(crate) fn fuzzy_match_byte_offsets_for_page<'q>( let mut matcher = neo_frizbee::Matcher::new(part, &part_config); for mut matched in matcher.match_list_indices(&path_strs) { let item_idx = matched.index as usize; - let Some(path) = paths.get(item_idx) else { + let Some(path) = path_strs.get(item_idx) else { continue; }; @@ -235,55 +241,31 @@ pub(crate) fn fuzzy_match_byte_offsets_for_page<'q>( ranges_by_item } -fn char_indices_to_byte_offsets(line: &str, char_indices: &[u32]) -> SmallVec<[(u32, u32); 4]> { - let char_byte_ranges: Vec<(usize, usize)> = line - .char_indices() - .map(|(byte_pos, ch)| (byte_pos, byte_pos + ch.len_utf8())) - .collect(); - let mut result: SmallVec<[(u32, u32); 4]> = SmallVec::with_capacity(char_indices.len()); - - for &char_idx in char_indices { - let Some(&(start, end)) = char_byte_ranges.get(char_idx as usize) else { - continue; - }; - - if let Some(last) = result.last_mut() - && last.1 == start as u32 - { - last.1 = end as u32; - continue; - } - - result.push((start as u32, end as u32)); - } - - result -} - fn merge_byte_offsets(mut ranges: SmallVec<[(u32, u32); 4]>) -> SmallVec<[(u32, u32); 4]> { if ranges.len() <= 1 { return ranges; } ranges.sort_unstable_by(|a, b| a.0.cmp(&b.0).then(a.1.cmp(&b.1))); - let mut merged: SmallVec<[(u32, u32); 4]> = SmallVec::with_capacity(ranges.len()); - - for (start, end) in ranges { + let mut merged = 0; + for index in 0..ranges.len() { + let (start, end) = ranges[index]; if end <= start { continue; } - if let Some(last) = merged.last_mut() - && start <= last.1 - { + if merged > 0 && start <= ranges[merged - 1].1 { + let last = &mut ranges[merged - 1]; last.1 = last.1.max(end); continue; } - merged.push((start, end)); + ranges[merged] = (start, end); + merged += 1; } - merged + ranges.truncate(merged); + ranges } /// Resolve a DirItem's chunked path into frizbee's pointer buffer. @@ -313,7 +295,7 @@ fn match_fuzzy_parts_dirs( arena: ArenaPtr, overflow_arena: ArenaPtr, ) -> Vec { - let valid_parts: Vec<&str> = fuzzy_parts + let valid_parts: SmallVec<[&str; 4]> = fuzzy_parts .iter() .copied() .filter(|p| p.len() >= 2) @@ -417,7 +399,7 @@ pub(crate) fn fuzzy_match_and_score_dirs<'a>( } }; - let valid_parts: Vec<&str> = fuzzy_parts + let valid_parts: SmallVec<[&str; 4]> = fuzzy_parts .iter() .copied() .filter(|p| p.len() >= 2) @@ -456,6 +438,7 @@ pub(crate) fn fuzzy_match_and_score_dirs<'a>( let mut dir_buf = String::with_capacity(64); let mut dirname_buf = String::with_capacity(32); + let distance = context.current_file.map(DirectoryDistance::new); let results: Vec<(&DirItem, Score)> = path_matches .into_iter() @@ -470,9 +453,9 @@ pub(crate) fn fuzzy_match_and_score_dirs<'a>( let frecency_boost = base_score.saturating_mul(dir.max_access_frecency()) / 100; // Distance penalty from current file's directory. - let distance_penalty = if context.current_file.is_some() { + let distance_penalty = if let Some(distance) = &distance { dir.path.write_to_string(dir_arena, &mut dir_buf); - calculate_distance_penalty(context.current_file, &dir_buf) + distance.penalty(&dir_buf) } else { 0 }; @@ -584,14 +567,8 @@ fn sort_and_paginate_dirs<'a>( sort_with_buffer(&mut results, |a, b| b.1.total.cmp(&a.1.total)); - if results.len() > limit { - let page_end = std::cmp::min(offset + limit, results.len()); - let page_size = page_end - offset; - results.drain(0..offset); - results.truncate(page_size); - } - - let (items, scores): (Vec<&DirItem>, Vec) = results.into_iter().unzip(); + let (items, scores): (Vec<&DirItem>, Vec) = + results.into_iter().skip(offset).take(limit).unzip(); (items, scores, total_matched) } @@ -599,6 +576,18 @@ fn match_and_score_in_arena<'a>( files: &'a [FileItem], context: &ScoringContext, arena: ArenaPtr, +) -> Vec<(&'a FileItem, Score)> { + if context.current_file.is_some() { + match_and_score_in_arena_inner::(files, context, arena) + } else { + match_and_score_in_arena_inner::(files, context, arena) + } +} + +fn match_and_score_in_arena_inner<'a, const WITH_CURRENT_FILE: bool>( + files: &'a [FileItem], + context: &ScoringContext, + arena: ArenaPtr, ) -> Vec<(&'a FileItem, Score)> { if files.is_empty() { return vec![]; @@ -700,6 +689,15 @@ fn match_and_score_in_arena<'a>( let mut dir_buf = String::with_capacity(64); let mut fname_buf = String::with_capacity(32); let mut path_buf = [0u8; crate::simd_path::PATH_BUF_SIZE]; + let distance = context + .current_file + .filter(|_| WITH_CURRENT_FILE) + .map(DirectoryDistance::new); + let mut last_dir_penalty = None; + let combo_path = context + .last_same_query_match + .as_ref() + .map(|entry| entry.file_path.to_string_lossy()); let results: Vec<_> = path_matches .into_iter() @@ -718,12 +716,18 @@ fn match_and_score_in_arena<'a>( }; let git_recency_boost = file.git_recency_score as i32; - if context.current_file.is_some() || context.last_same_query_match.is_some() { - file.write_dir_str(arena, &mut dir_buf); - } - - let distance_penalty = if context.current_file.is_some() { - calculate_distance_penalty(context.current_file, &dir_buf) + let distance_penalty = if WITH_CURRENT_FILE && let Some(distance) = &distance { + if let Some((parent, penalty)) = last_dir_penalty + && parent == file.parent_dir_index + && parent != u32::MAX + { + penalty + } else { + file.write_dir_str(arena, &mut dir_buf); + let penalty = distance.penalty(&dir_buf); + last_dir_penalty = Some((file.parent_dir_index, penalty)); + penalty + } } else { 0 }; @@ -787,18 +791,21 @@ fn match_and_score_in_arena<'a>( 0 }; - let current_file_penalty = - calculate_current_file_penalty(file, base_score / 4, context, arena); + let current_file_penalty = if WITH_CURRENT_FILE { + calculate_current_file_penalty(file, base_score / 4, context, arena) + } else { + 0 + }; let combo_match_boost = { - let last_same_query_match = context.last_same_query_match.as_ref().filter(|m| { - let file_path_str = m.file_path.to_string_lossy(); - let total_len = file.path.byte_len as usize; - if file_path_str.len() < total_len { - return false; - } - // Reuse dir_buf (already has capacity) for the full path - file.write_relative_path_from_arena(arena, &mut dir_buf); - file_path_str.ends_with(dir_buf.as_str()) + let last_same_query_match = context.last_same_query_match.as_ref().filter(|_| { + combo_path + .as_ref() + .and_then(|path| { + path.len() + .checked_sub(file.relative_path_len()) + .and_then(|start| path.get(start..)) + }) + .is_some_and(|suffix| file.relative_path_eq(arena, suffix)) }); match last_same_query_match { @@ -952,19 +959,23 @@ fn score_filtered_by_frecency<'a>( }; match files { - FileItems::All(s) => s + // Small indexes cannot amortize Rayon scheduling for this cheap scoring pass. + FileItems::All(s) if s.len() >= 32_768 && context.max_threads > 1 => s .par_iter() - .filter_map(|f| { - let live = !f.is_deleted(); - live.then_some(score_file(f)) - }) + .with_min_len(4096) + .filter(|f| !f.is_deleted()) + .map(score_file) + .collect(), + FileItems::All(s) => s + .iter() + .filter(|f| !f.is_deleted()) + .map(score_file) .collect(), FileItems::Filtered(v) => v .iter() - .filter_map(|f| { - let live = !f.is_deleted(); - live.then_some(score_file(f)) - }) + .copied() + .filter(|f| !f.is_deleted()) + .map(score_file) .collect(), } } @@ -1041,16 +1052,8 @@ fn sort_and_paginate<'a>( .then_with(|| b.0.modified.cmp(&a.0.modified)) }); - // in the best scenario truncation happened in the select_nth step - if results.len() > limit { - let page_end = std::cmp::min(offset + limit, results.len()); - let page_size = page_end - offset; - - results.drain(0..offset); - results.truncate(page_size); - } - - let (items, scores): (Vec<&FileItem>, Vec) = results.into_iter().unzip(); + let (items, scores): (Vec<&FileItem>, Vec) = + results.into_iter().skip(offset).take(limit).unzip(); (items, scores, total_matched) } @@ -1060,6 +1063,144 @@ mod tests { use crate::types::PaginationArgs; use fff_query_parser::QueryParser; + #[test] + fn frecency_parallel_and_sequential_scores_match() { + let files: Vec<_> = (0..33_000) + .map(|index| { + let mut file = + FileItem::new_raw(0, 0, index, Some(git2::Status::WT_MODIFIED), false); + file.access_frecency_score = (index % 97) as i16; + file.modification_frecency_score = (index % 29) as i16; + file.git_recency_score = (index % 13) as i16; + file.set_deleted(index % 7 == 0); + file + }) + .collect(); + let parser = QueryParser::default(); + let query = parser.parse(""); + let mut context = ScoringContext { + query: &query, + max_threads: 4, + max_typos: 0, + project_path: None, + current_file: None, + last_same_query_match: None, + combo_boost_score_multiplier: 0, + min_combo_count: 0, + pagination: PaginationArgs { + offset: 0, + limit: 50, + }, + }; + for count in [0, 1, 1000, 33_000] { + let files = FileItems::All(&files[..count]); + context.max_threads = 4; + let parallel = score_filtered_by_frecency(&files, &context, ArenaPtr::null()); + context.max_threads = 1; + let sequential = score_filtered_by_frecency(&files, &context, ArenaPtr::null()); + assert_eq!(parallel.len(), count - count.div_ceil(7)); + assert_eq!(parallel.len(), sequential.len()); + for ((a, a_score), (b, b_score)) in parallel.iter().zip(&sequential) { + assert!(std::ptr::eq(*a, *b)); + assert!(!a.is_deleted()); + assert_eq!(a_score.total, b_score.total); + assert_eq!(a_score.frecency_boost, b_score.frecency_boost); + assert_eq!(a_score.git_status_boost, b_score.git_status_boost); + assert_eq!(a_score.git_recency_boost, b_score.git_recency_boost); + } + } + } + + #[test] + fn pagination_matches_full_ranking() { + let files: Vec<_> = (0..257) + .map(|i| FileItem::new_raw(0, 0, i as u64, None, false)) + .collect(); + let dirs: Vec<_> = (0..257) + .map(|_| DirItem::new(crate::simd_path::ChunkedString::empty(), 0)) + .collect(); + let parser = QueryParser::default(); + let query = parser.parse(""); + for count in [0, 1, 3, 100, 101, 257] { + let results: Vec<_> = files[..count] + .iter() + .enumerate() + .map(|(index, file)| { + ( + file, + Score { + total: (index * 73 % 257) as i32, + ..Default::default() + }, + ) + }) + .collect(); + let mut expected = results.clone(); + expected.sort_by(|a, b| { + b.1.total + .cmp(&a.1.total) + .then_with(|| b.0.modified.cmp(&a.0.modified)) + }); + for offset in [0, 1, count / 2, count, usize::MAX] { + for limit in [0, 1, 2, 50, count, usize::MAX] { + let context = ScoringContext { + query: &query, + max_threads: 1, + max_typos: 0, + project_path: None, + current_file: None, + last_same_query_match: None, + combo_boost_score_multiplier: 0, + min_combo_count: 0, + pagination: PaginationArgs { offset, limit }, + }; + let expected_page: Vec<_> = expected + .iter() + .skip(offset) + .take(if limit == 0 { count } else { limit }) + .map(|(_, score)| score.total) + .collect(); + let (items, scores, total) = sort_and_paginate(results.clone(), &context); + assert_eq!(total, count); + assert_eq!(items.len(), expected_page.len()); + assert_eq!( + scores.iter().map(|score| score.total).collect::>(), + expected_page + ); + let dir_results = results + .iter() + .enumerate() + .map(|(index, (_, score))| (&dirs[index], score.clone())) + .collect(); + let (items, scores, total) = sort_and_paginate_dirs(dir_results, &context); + assert_eq!(total, count); + assert_eq!(items.len(), expected_page.len()); + assert_eq!( + scores.iter().map(|score| score.total).collect::>(), + expected_page + ); + } + } + } + } + + #[test] + fn merge_offsets_reuses_storage() { + let mut ranges: SmallVec<[(u32, u32); 4]> = smallvec::smallvec![ + (20, 25), + (2, 6), + (0, 3), + (6, 10), + (11, 11), + (24, 30), + (40, 45), + ]; + let storage = ranges.as_mut_ptr(); + let merged = merge_byte_offsets(ranges); + assert_eq!(merged.as_slice(), &[(0, 10), (20, 30), (40, 45)]); + assert_eq!(merged.as_ptr(), storage); + } + fn make_test_files(specs: &[(&str, i32, u64)]) -> (Vec<(FileItem, Score)>, ArenaPtr) { let path_strings: Vec = specs.iter().map(|(p, _, _)| p.to_string()).collect(); let items: Vec = specs @@ -1263,6 +1404,67 @@ mod filename_bonus_tests { use super::*; use crate::types::PaginationArgs; use fff_query_parser::QueryParser; + + #[test] + fn page_highlights_use_each_files_arena() { + let base_path = "src/ui_controller.rs"; + let overflow_path = "other/test_controller.rs"; + let (base, base_arena) = make_files(&[base_path]); + let (overflow, overflow_arena) = make_files(&[overflow_path]); + overflow[0].set_overflow(true); + let parser = QueryParser::default(); + let query = parser.parse("controller"); + let ranges = fuzzy_match_byte_offsets_for_page( + &query, + &[&base[0], &overflow[0]], + 0, + base_arena, + overflow_arena, + ); + for (path, ranges) in [base_path, overflow_path].into_iter().zip(ranges) { + let start = path.find("controller").unwrap() as u32; + assert_eq!(ranges.as_slice(), &[(start, start + 10)]); + } + } + + #[test] + fn distance_scoring_handles_repeated_and_unknown_parents() { + let paths = [ + "src/widgets/file_a.rs", + "src/widgets/file_b.rs", + "src/file_c.rs", + "tests/file_d.rs", + ]; + let (mut files, arena) = make_files(&paths); + files[0].parent_dir_index = 0; + files[1].parent_dir_index = 0; + let parser = QueryParser::default(); + let query = parser.parse("file"); + let current_file = Some("src/widgets/current.rs"); + let context = ScoringContext { + query: &query, + current_file, + max_threads: 1, + max_typos: 0, + project_path: None, + last_same_query_match: None, + combo_boost_score_multiplier: 0, + min_combo_count: 0, + pagination: PaginationArgs { + offset: 0, + limit: 50, + }, + }; + let results = match_and_score_in_arena(&files, &context, arena); + assert_eq!(results.len(), files.len()); + for (file, score) in results { + assert_eq!( + score.distance_penalty, + crate::path_utils::calculate_distance_penalty(current_file, &file.dir_str(arena)) + ); + } + } + fn make_files(paths: &[&str]) -> (Vec, ArenaPtr) { let path_strings: Vec = paths.iter().map(|p| p.to_string()).collect(); let items: Vec = paths diff --git a/crates/fff-core/src/simd_path.rs b/crates/fff-core/src/simd_path.rs index 399fa516d..6e3057000 100644 --- a/crates/fff-core/src/simd_path.rs +++ b/crates/fff-core/src/simd_path.rs @@ -105,6 +105,21 @@ impl ChunkedString { chunks_needed(self.byte_len as usize) } + #[inline] + pub(crate) fn equals(&self, arena: ArenaPtr, other: &str) -> bool { + if other.len() != self.byte_len as usize { + return false; + } + self.indices(arena) + .iter() + .zip(other.as_bytes().chunks(SIMD_CHUNK_BYTES)) + .all(|(&index, expected)| { + let actual = + unsafe { core::slice::from_raw_parts(arena.chunk_ptr(index), expected.len()) }; + actual == expected + }) + } + #[inline] fn indices<'a>(&self, arena: ArenaPtr) -> &'a [u32] { let count = self.chunk_count(); @@ -143,11 +158,7 @@ impl ChunkedString { } } - /// Return the filename portion as a `Cow`. - /// - /// When the filename starts at a chunk boundary and fits in one chunk we - /// borrow directly from the arena (zero-copy). Otherwise we allocate. - /// Filenames are almost always <=16 bytes so the fast path dominates. + /// Borrow filenames contained in one chunk; copy filenames spanning chunks. #[inline] pub fn filename_cow<'a>(&self, arena: ArenaPtr) -> Cow<'a, str> { let fname_offset = self.filename_offset as usize; @@ -160,9 +171,9 @@ impl ChunkedString { let start_chunk = fname_offset / SIMD_CHUNK_BYTES; let offset_in_chunk = fname_offset % SIMD_CHUNK_BYTES; - if offset_in_chunk == 0 && fname_len <= SIMD_CHUNK_BYTES { + if offset_in_chunk + fname_len <= SIMD_CHUNK_BYTES { let ptr = arena.chunk_ptr(indices[start_chunk]); - let slice = unsafe { core::slice::from_raw_parts(ptr, fname_len) }; + let slice = unsafe { core::slice::from_raw_parts(ptr.add(offset_in_chunk), fname_len) }; return Cow::Borrowed(unsafe { core::str::from_utf8_unchecked(slice) }); } @@ -241,7 +252,10 @@ impl ChunkedString { #[inline] pub fn write_to_string(&self, arena: ArenaPtr, out: &mut String) { out.clear(); + self.append_to_string(arena, out); + } + pub(crate) fn append_to_string(&self, arena: ArenaPtr, out: &mut String) { let total = self.byte_len as usize; if total == 0 { return; @@ -612,6 +626,67 @@ mod tests { assert_eq!(&*fname, "Cargo.toml"); } + #[test] + fn chunked_equals_matches_string_equality() { + for len in 0..=100 { + let path = format!("{}é🦀", "x".repeat(len)); + let (store, strings, _) = build_test_store(&[&path, "src/lib.rs"]); + let arena = store.as_arena_ptr(); + for other in [&path, "", "src/lib.rs", &format!("y{}é🦀", "x".repeat(len))] { + assert_eq!(strings[0].equals(arena, other), path == other); + } + for index in 0..len { + let mut other = path.clone().into_bytes(); + other[index] = b'y'; + assert!(!strings[0].equals(arena, &String::from_utf8(other).unwrap())); + } + } + assert!(ChunkedString::empty().equals(ArenaPtr::null(), "")); + assert!(!ChunkedString::empty().equals(ArenaPtr::null(), "x")); + } + + #[test] + fn append_paths_preserves_existing_text() { + let paths = [ + "", + "src/lib.rs", + "目录/é🦀.rs", + "long/path/with/many/components/file.rs", + ]; + let (store, strings, _) = build_test_store(&paths); + let mut output = String::from("prefix:"); + for string in &strings { + string.append_to_string(store.as_arena_ptr(), &mut output); + } + assert_eq!(output, format!("prefix:{}", paths.concat())); + } + + #[test] + fn filename_cow_borrows_at_every_offset_within_a_chunk() { + for offset in 0..SIMD_CHUNK_BYTES { + for name in [ + "x", + "é", + "file.rs", + "sixteenbytes.txt", + "filename_over_sixteen.rs", + ] { + let prefix = "d".repeat(offset); + let path = format!("{prefix}{name}"); + let mut builder = ChunkedPathStoreBuilder::new(1); + let string = builder.add_file_immediate(&path, offset as u16); + let store = builder.finish(); + let filename = string.filename_cow(store.as_arena_ptr()); + + assert_eq!(filename, name); + assert_eq!( + matches!(filename, Cow::Borrowed(_)), + offset + name.len() <= SIMD_CHUNK_BYTES, + ); + } + } + } + #[test] fn test_chunked_string_long_path() { let path = "very/deeply/nested/directory/structure/with/many/levels/file.txt"; diff --git a/crates/fff-core/src/types.rs b/crates/fff-core/src/types.rs index 31fbad521..cae200293 100644 --- a/crates/fff-core/src/types.rs +++ b/crates/fff-core/src/types.rs @@ -367,10 +367,6 @@ impl FileItem { s } - pub(crate) fn write_relative_path_from_arena(&self, arena: ArenaPtr, out: &mut String) { - self.path.write_to_string(arena, out); - } - pub fn relative_path_len(&self) -> usize { self.path.byte_len as usize } @@ -380,12 +376,7 @@ impl FileItem { } pub(crate) fn relative_path_eq(&self, arena: ArenaPtr, other: &str) -> bool { - if other.len() != self.path.byte_len as usize { - return false; - } - let mut buf = [0u8; PATH_BUF_SIZE]; - let mine = self.path.read_to_buf(arena, &mut buf); - mine == other + self.path.equals(arena, other) } pub(crate) fn relative_path_starts_with(&self, arena: ArenaPtr, prefix: &str) -> bool { From 27be137ef48dab42c8fe3da0a892ff05460979fe Mon Sep 17 00:00:00 2001 From: Dmitriy Kovalenko Date: Thu, 24 Sep 2026 22:01:17 -0700 Subject: [PATCH 2/2] perf(search): rank frecency on totals, parallelize match scoring, address review --- _typos.toml | 1 + crates/fff-core/benches/fuzzy_search_bench.rs | 19 +- crates/fff-core/benches/grep_bench.rs | 5 - crates/fff-core/src/path_utils.rs | 48 +-- crates/fff-core/src/score.rs | 402 +++++++++++------- crates/fff-core/src/sort_buffer.rs | 21 - 6 files changed, 298 insertions(+), 198 deletions(-) diff --git a/_typos.toml b/_typos.toml index 6c280a33a..eeb15b90d 100644 --- a/_typos.toml +++ b/_typos.toml @@ -12,6 +12,7 @@ thm = "thm" comparsion = "comparsion" modfiers = "modfiers" shcema = "shcema" +contrller = "contrller" [default] extend-ignore-re = [ diff --git a/crates/fff-core/benches/fuzzy_search_bench.rs b/crates/fff-core/benches/fuzzy_search_bench.rs index c71f9d9e9..ca03ff565 100644 --- a/crates/fff-core/benches/fuzzy_search_bench.rs +++ b/crates/fff-core/benches/fuzzy_search_bench.rs @@ -6,9 +6,22 @@ mod support; fn bench_fuzzy_search(c: &mut Criterion) { if let Some(picker) = support::repo_picker() { - let current_file = - std::env::var("FFF_BENCH_CURRENT_FILE").unwrap_or_else(|_| "README.md".into()); - assert!(picker.base_path().join(¤t_file).is_file()); + let current_file = match std::env::var("FFF_BENCH_CURRENT_FILE") { + Ok(file) => { + assert!( + picker.base_path().join(&file).is_file(), + "FFF_BENCH_CURRENT_FILE must exist in FFF_BENCH_PATH" + ); + file + } + Err(_) => picker + .get_files() + .iter() + .find(|file| !file.is_deleted()) + .map(|file| file.relative_path(&picker)) + .expect("repository must contain files"), + }; + eprintln!("Current file: {current_file}"); bench_picker(c, &picker, "fuzzy_search_repo", ¤t_file, true); return; } diff --git a/crates/fff-core/benches/grep_bench.rs b/crates/fff-core/benches/grep_bench.rs index 3e1a01e29..7f8f27e48 100644 --- a/crates/fff-core/benches/grep_bench.rs +++ b/crates/fff-core/benches/grep_bench.rs @@ -155,11 +155,6 @@ fn bench_repo(c: &mut Criterion, picker: &FilePicker) { result.filtered_file_count ); assert!(result.regex_fallback_error.is_none()); - if name == "plain_no_matches" { - assert!(result.matches.is_empty()); - } else { - assert!(!result.matches.is_empty(), "query {name} must match"); - } drop(result); group.bench_function(name, |b| { b.iter(|| std::hint::black_box(picker.grep(&query, &options))); diff --git a/crates/fff-core/src/path_utils.rs b/crates/fff-core/src/path_utils.rs index e4015e218..6021fab6b 100644 --- a/crates/fff-core/src/path_utils.rs +++ b/crates/fff-core/src/path_utils.rs @@ -1,29 +1,5 @@ use std::path::{Components, Path, PathBuf}; -pub(crate) struct DirectoryDistance<'a> { - components: Components<'a>, - depth: usize, -} - -impl<'a> DirectoryDistance<'a> { - pub(crate) fn new(current_file: &'a str) -> Self { - let directory = Path::new(current_file).parent().unwrap_or(Path::new("")); - let components = directory.components(); - let depth = components.clone().count(); - Self { components, depth } - } - - pub(crate) fn penalty(&self, candidate_dir: &str) -> i32 { - let common = self - .components - .clone() - .zip(Path::new(candidate_dir).components()) - .take_while(|(a, b)| a == b) - .count(); - -((self.depth - common).min(20) as i32) - } -} - #[cfg(windows)] pub fn canonicalize(path: impl AsRef) -> std::io::Result { dunce::canonicalize(path) @@ -149,6 +125,30 @@ pub fn calculate_distance_penalty(current_file: Option<&str>, candidate_dir: &st (-(depth_from_common as i32)).max(-20) } +pub(crate) struct DirectoryDistance<'a> { + components: Components<'a>, + depth: usize, +} + +impl<'a> DirectoryDistance<'a> { + pub(crate) fn new(current_file: &'a str) -> Self { + let directory = Path::new(current_file).parent().unwrap_or(Path::new("")); + let components = directory.components(); + let depth = components.clone().count(); + Self { components, depth } + } + + pub(crate) fn penalty(&self, candidate_dir: &str) -> i32 { + let common = self + .components + .clone() + .zip(Path::new(candidate_dir).components()) + .take_while(|(a, b)| a == b) + .count(); + -((self.depth - common).min(20) as i32) + } +} + #[cfg(test)] mod tests { use super::*; diff --git a/crates/fff-core/src/score.rs b/crates/fff-core/src/score.rs index 137849791..5bed0ea94 100644 --- a/crates/fff-core/src/score.rs +++ b/crates/fff-core/src/score.rs @@ -4,7 +4,7 @@ use crate::{ match_offsets::char_indices_to_byte_offsets, path_utils::DirectoryDistance, simd_path::{ArenaPtr, MAX_PATH_CHUNKS}, - sort_buffer::{sort_by_key_with_buffer, sort_with_buffer}, + sort_buffer::sort_with_buffer, types::{DirItem, FileItem, Score, ScoringContext}, }; use fff_query_parser::{FFFQuery, FuzzyQuery}; @@ -140,6 +140,10 @@ pub(crate) fn fuzzy_match_and_score_files<'a>( base_arena: ArenaPtr, overflow_arena: ArenaPtr, ) -> (Vec<&'a FileItem>, Vec, usize) { + if fuzzy_parts(context.query).is_none() { + return rank_by_frecency(files, context, base_count, base_arena, overflow_arena); + } + // Process overflow files first: newly added files (created after the // initial scan) live in the overflow arena and are more likely to be // relevant to the current search. @@ -160,7 +164,70 @@ pub(crate) fn fuzzy_match_and_score_files<'a>( match_and_score_in_arena(files, context, base_arena) }; - sort_and_paginate(results, context) + sort_and_paginate(results, context, |score| score.total) +} + +// Ranks on a 16-byte (file, total) pair and builds the full Score only for the page. +fn rank_by_frecency<'a>( + files: &'a [FileItem], + context: &ScoringContext, + base_count: usize, + base_arena: ArenaPtr, + overflow_arena: ArenaPtr, +) -> (Vec<&'a FileItem>, Vec, usize) { + let base_count = base_count.min(files.len()); + let mut totals: Vec<(&FileItem, i32)> = Vec::new(); + for (slice, arena) in [ + (&files[base_count..], overflow_arena), + (&files[..base_count], base_arena), + ] { + if slice.is_empty() { + continue; + } + let Some(working_files) = filter_by_constraints(slice, context, arena) else { + continue; + }; + frecency_totals(&working_files, context, arena, &mut totals); + } + + let (items, _, total_matched) = sort_and_paginate(totals, context, |total| *total); + let scores = items + .iter() + .map(|file| { + let arena = if file.is_overflow() { + overflow_arena + } else { + base_arena + }; + frecency_score(file, context, arena) + }) + .collect(); + (items, scores, total_matched) +} + +fn fuzzy_parts<'q>(query: &'q FFFQuery<'q>) -> Option<&'q [&'q str]> { + match &query.fuzzy_query { + FuzzyQuery::Text(t) if t.len() >= 2 => Some(std::slice::from_ref(t)), + FuzzyQuery::Parts(parts) if !parts.is_empty() => Some(parts.as_slice()), + _ => None, + } +} + +// None when constraints exclude every file. +fn filter_by_constraints<'a>( + files: &'a [FileItem], + context: &ScoringContext, + arena: ArenaPtr, +) -> Option> { + let constraints = &context.query.constraints; + if constraints.is_empty() { + return Some(FileItems::All(files)); + } + match apply_constraints(files, constraints, arena, arena) { + Some(filtered) if !filtered.is_empty() => Some(FileItems::Filtered(filtered)), + Some(_) => None, + None => Some(FileItems::All(files)), + } } pub(crate) fn fuzzy_match_byte_offsets_for_page<'q>( @@ -593,28 +660,13 @@ fn match_and_score_in_arena_inner<'a, const WITH_CURRENT_FILE: bool>( return vec![]; } - let parsed = context.query; - let working_files: FileItems<'a> = if parsed.constraints.is_empty() { - FileItems::All(files) - } else { - match apply_constraints(files, &parsed.constraints, arena, arena) { - Some(filtered) if !filtered.is_empty() => FileItems::Filtered(filtered), - Some(_) => { - return vec![]; - } - None => FileItems::All(files), - } + let Some(working_files) = filter_by_constraints(files, context, arena) else { + return vec![]; }; - - let fuzzy_parts: &[&str] = match &parsed.fuzzy_query { - FuzzyQuery::Text(t) if t.len() >= 2 => std::slice::from_ref(t), - FuzzyQuery::Parts(parts) if !parts.is_empty() => parts.as_slice(), - _ => { - return score_filtered_by_frecency(&working_files, context, arena); - } + let Some(fuzzy_parts) = fuzzy_parts(context.query) else { + // Frecency-only queries are ranked by `rank_by_frecency` before reaching here. + return frecency_scores(&working_files, context, arena); }; - - debug_assert!(!fuzzy_parts.is_empty()); let has_uppercase = fuzzy_parts .iter() .any(|p| p.chars().any(|c| c.is_uppercase())); @@ -670,39 +722,43 @@ fn match_and_score_in_arena_inner<'a, const WITH_CURRENT_FILE: bool>( // Match on `&str` so frizbee reuses the instantiation its index // resolver already emits instead of a separate `Cow` copy. let filename_strs: Vec<&str> = fallback_filenames.iter().map(Cow::as_ref).collect(); - let mut matches = neo_frizbee::Matcher::new(fuzzy_parts[0], &options) - .match_list_parallel( - &filename_strs, - if path_matches.len() > 4096 { - context.max_threads.div_ceil(2048) - } else { - 1 - }, - ); - - sort_by_key_with_buffer(&mut matches, |m| fallback_indices[m.index as usize]); - matches + neo_frizbee::Matcher::new(fuzzy_parts[0], &options).match_list_parallel( + &filename_strs, + if path_matches.len() > 4096 { + context.max_threads.div_ceil(2048) + } else { + 1 + }, + ) } }; - let mut next_filename_match_cursor = 0; - let mut dir_buf = String::with_capacity(64); - let mut fname_buf = String::with_capacity(32); - let mut path_buf = [0u8; crate::simd_path::PATH_BUF_SIZE]; + // path-match index -> position in filename_fallback_matches (u32::MAX = none) + let mut fallback_by_match: Vec = Vec::new(); + if !filename_fallback_matches.is_empty() { + fallback_by_match.resize(path_matches.len(), u32::MAX); + for (position, m) in filename_fallback_matches.iter().enumerate() { + fallback_by_match[fallback_indices[m.index as usize] as usize] = position as u32; + } + } + let distance = context .current_file .filter(|_| WITH_CURRENT_FILE) .map(DirectoryDistance::new); - let mut last_dir_penalty = None; let combo_path = context .last_same_query_match .as_ref() .map(|entry| entry.file_path.to_string_lossy()); - let results: Vec<_> = path_matches - .into_iter() - .enumerate() - .map(|(match_idx, path_match)| { + let score_match = + |bufs: &mut ScoreBuffers, (match_idx, path_match): (usize, &neo_frizbee::Match)| { + let ScoreBuffers { + dir_buf, + fname_buf, + path_buf, + last_dir_penalty, + } = bufs; let file_idx = path_match.index as usize; let file = working_files.index(file_idx); @@ -717,15 +773,15 @@ fn match_and_score_in_arena_inner<'a, const WITH_CURRENT_FILE: bool>( let git_recency_boost = file.git_recency_score as i32; let distance_penalty = if WITH_CURRENT_FILE && let Some(distance) = &distance { - if let Some((parent, penalty)) = last_dir_penalty + if let Some((parent, penalty)) = *last_dir_penalty && parent == file.parent_dir_index && parent != u32::MAX { penalty } else { - file.write_dir_str(arena, &mut dir_buf); - let penalty = distance.penalty(&dir_buf); - last_dir_penalty = Some((file.parent_dir_index, penalty)); + file.write_dir_str(arena, dir_buf); + let penalty = distance.penalty(dir_buf); + *last_dir_penalty = Some((file.parent_dir_index, penalty)); penalty } } else { @@ -737,16 +793,9 @@ fn match_and_score_in_arena_inner<'a, const WITH_CURRENT_FILE: bool>( let end_col_filename_match = match_start_approx >= filename_start; let simd_filename_match = if !end_col_filename_match { - filename_fallback_matches - .get(next_filename_match_cursor) - .and_then(|m| { - if fallback_indices[m.index as usize] == match_idx as u32 { - next_filename_match_cursor += 1; - Some(m) - } else { - None - } - }) + fallback_by_match + .get(match_idx) + .and_then(|&position| filename_fallback_matches.get(position as usize)) } else { None }; @@ -756,7 +805,7 @@ fn match_and_score_in_arena_inner<'a, const WITH_CURRENT_FILE: bool>( let is_exact_filename = simd_filename_match.is_some_and(|m| m.exact) || (end_col_filename_match && main_needle_len as usize == fname_len && { - file.write_file_name_from_arena(arena, &mut fname_buf); + file.write_file_name_from_arena(arena, fname_buf); main_needle.eq_ignore_ascii_case(fname_buf.as_bytes()) }); @@ -778,10 +827,10 @@ fn match_and_score_in_arena_inner<'a, const WITH_CURRENT_FILE: bool>( max_bonus } } else if !is_filename_match && (5..=11).contains(&fname_len) { - file.write_file_name_from_arena(arena, &mut fname_buf); + file.write_file_name_from_arena(arena, fname_buf); // 5% bonus for special file but not as much as file name to avoid situations // when you have /user_service/server.rs and /user_service/server/mod.rs - if is_special_entry_point_file(&fname_buf) { + if is_special_entry_point_file(fname_buf) { has_special_filename_bonus = true; base_score * 5 / 100 } else { @@ -823,7 +872,7 @@ fn match_and_score_in_arena_inner<'a, const WITH_CURRENT_FILE: bool>( // Uses suffix overlap — bytes matching from the end. A full prefix // match is just the 100% coverage case, so no separate branch needed. let path_alignment_bonus = if query_contains_path_separator { - let rel_path = file.path.read_to_buf(arena, &mut path_buf); + let rel_path = file.path.read_to_buf(arena, path_buf); let path_bytes = rel_path.as_bytes(); let common_suffix = main_needle .iter() @@ -886,12 +935,49 @@ fn match_and_score_in_arena_inner<'a, const WITH_CURRENT_FILE: bool>( }; (file, score) - }) - .collect(); + }; + + // Scoring is a pure per-match function; only the scratch buffers are per-thread. + let results: Vec<_> = + if path_matches.len() >= PARALLEL_SCORING_MIN_MATCHES && context.max_threads > 1 { + path_matches + .par_iter() + .enumerate() + .with_min_len((path_matches.len() / context.max_threads).max(4096)) + .map_init(ScoreBuffers::new, score_match) + .collect() + } else { + let mut bufs = ScoreBuffers::new(); + path_matches + .iter() + .enumerate() + .map(|item| score_match(&mut bufs, item)) + .collect() + }; results } +const PARALLEL_SCORING_MIN_MATCHES: usize = 8192; + +struct ScoreBuffers { + dir_buf: String, + fname_buf: String, + path_buf: [u8; crate::simd_path::PATH_BUF_SIZE], + last_dir_penalty: Option<(u32, i32)>, +} + +impl ScoreBuffers { + fn new() -> Self { + Self { + dir_buf: String::with_capacity(64), + fname_buf: String::with_capacity(32), + path_buf: [0u8; crate::simd_path::PATH_BUF_SIZE], + last_dir_penalty: None, + } + } +} + fn is_special_entry_point_file(filename: &str) -> bool { matches!( filename, @@ -915,71 +1001,79 @@ fn is_special_entry_point_file(filename: &str) -> bool { ) } -fn score_filtered_by_frecency<'a>( - files: &FileItems<'a>, - context: &ScoringContext, - arena: ArenaPtr, -) -> Vec<(&'a FileItem, Score)> { - let score_file = |file: &'a FileItem| { - let total_frecency_score = file.access_frecency_score as i32 - + (file.modification_frecency_score as i32).saturating_mul(4); +fn frecency_total(file: &FileItem, context: &ScoringContext, arena: ArenaPtr) -> i32 { + frecency_score(file, context, arena).total +} - // Give modified/dirty files a boost even in frecency-only mode - let git_status_boost = if file.git_status.is_some_and(is_modified_status) { - total_frecency_score * 15 / 100 - } else { - 0 - }; - let git_recency_boost = file.git_recency_score as i32; - - let current_file_penalty = - calculate_current_file_penalty(file, total_frecency_score, context, arena); - let total = total_frecency_score - .saturating_add(git_status_boost) - .saturating_add(git_recency_boost) - .saturating_add(current_file_penalty); - - let score = Score { - total, - base_score: 0, - filename_bonus: 0, - distance_penalty: 0, - special_filename_bonus: 0, - combo_match_boost: 0, - path_alignment_bonus: 0, - current_file_penalty, - frecency_boost: total_frecency_score, - git_status_boost, - git_recency_boost, - exact_match: false, - match_type: "frecency", - }; +fn frecency_score(file: &FileItem, context: &ScoringContext, arena: ArenaPtr) -> Score { + let frecency_boost = file.access_frecency_score as i32 + + (file.modification_frecency_score as i32).saturating_mul(4); - (file, score) + // Give modified/dirty files a boost even in frecency-only mode + let git_status_boost = if file.git_status.is_some_and(is_modified_status) { + frecency_boost * 15 / 100 + } else { + 0 }; + let git_recency_boost = file.git_recency_score as i32; + let current_file_penalty = calculate_current_file_penalty(file, frecency_boost, context, arena); + let total = frecency_boost + .saturating_add(git_status_boost) + .saturating_add(git_recency_boost) + .saturating_add(current_file_penalty); + + Score { + total, + base_score: 0, + filename_bonus: 0, + distance_penalty: 0, + special_filename_bonus: 0, + combo_match_boost: 0, + path_alignment_bonus: 0, + current_file_penalty, + frecency_boost, + git_status_boost, + git_recency_boost, + exact_match: false, + match_type: "frecency", + } +} +fn frecency_totals<'a>( + files: &FileItems<'a>, + context: &ScoringContext, + arena: ArenaPtr, + out: &mut Vec<(&'a FileItem, i32)>, +) { + let total = |file: &'a FileItem| (file, frecency_total(file, context, arena)); match files { // Small indexes cannot amortize Rayon scheduling for this cheap scoring pass. - FileItems::All(s) if s.len() >= 32_768 && context.max_threads > 1 => s - .par_iter() - .with_min_len(4096) - .filter(|f| !f.is_deleted()) - .map(score_file) - .collect(), - FileItems::All(s) => s - .iter() - .filter(|f| !f.is_deleted()) - .map(score_file) - .collect(), - FileItems::Filtered(v) => v - .iter() - .copied() - .filter(|f| !f.is_deleted()) - .map(score_file) - .collect(), + FileItems::All(s) if s.len() >= 32_768 && context.max_threads > 1 => out.par_extend( + s.par_iter() + .with_min_len(4096) + .filter(|f| !f.is_deleted()) + .map(total), + ), + FileItems::All(s) => out.extend(s.iter().filter(|f| !f.is_deleted()).map(total)), + FileItems::Filtered(v) => { + out.extend(v.iter().copied().filter(|f| !f.is_deleted()).map(total)) + } } } +fn frecency_scores<'a>( + files: &FileItems<'a>, + context: &ScoringContext, + arena: ArenaPtr, +) -> Vec<(&'a FileItem, Score)> { + let mut totals = Vec::new(); + frecency_totals(files, context, arena, &mut totals); + totals + .into_iter() + .map(|(file, _)| (file, frecency_score(file, context, arena))) + .collect() +} + #[inline] fn calculate_current_file_penalty( file: &FileItem, @@ -1002,10 +1096,11 @@ fn calculate_current_file_penalty( /// Always returns results in descending order (best scores first). /// The UI layer handles rendering order based on prompt position. #[tracing::instrument(skip_all, level = tracing::Level::DEBUG)] -fn sort_and_paginate<'a>( - mut results: Vec<(&'a FileItem, Score)>, +fn sort_and_paginate<'a, S>( + mut results: Vec<(&'a FileItem, S)>, context: &ScoringContext, -) -> (Vec<&'a FileItem>, Vec, usize) { + total: impl Fn(&S) -> i32, +) -> (Vec<&'a FileItem>, Vec, usize) { let total_matched = results.len(); if total_matched == 0 { @@ -1031,28 +1126,21 @@ fn sort_and_paginate<'a>( } let items_needed = offset.saturating_add(limit).min(total_matched); + let compare = |a: &(&FileItem, S), b: &(&FileItem, S)| { + total(&b.1) + .cmp(&total(&a.1)) + .then_with(|| b.0.modified.cmp(&a.0.modified)) + }; // Use partial sort if we need less than half the results and dataset is large - let use_partial_sort = items_needed < total_matched / 2 && total_matched > 100; - // Always sort in descending order (best scores first) - if use_partial_sort { - // Partition at position (items_needed - 1) with descending comparator - // This puts the highest N needed items at the front - results.select_nth_unstable_by(items_needed - 1, |a, b| { - b.1.total - .cmp(&a.1.total) - .then_with(|| b.0.modified.cmp(&a.0.modified)) - }); + if items_needed < total_matched / 2 && total_matched > 100 { + results.select_nth_unstable_by(items_needed - 1, compare); results.truncate(items_needed); } // select nth does not sort the results, we have to sort accordingly anyway - sort_with_buffer(&mut results, |a, b| { - b.1.total - .cmp(&a.1.total) - .then_with(|| b.0.modified.cmp(&a.0.modified)) - }); + sort_with_buffer(&mut results, compare); - let (items, scores): (Vec<&FileItem>, Vec) = + let (items, scores): (Vec<&FileItem>, Vec) = results.into_iter().skip(offset).take(limit).unzip(); (items, scores, total_matched) } @@ -1093,11 +1181,11 @@ mod tests { }, }; for count in [0, 1, 1000, 33_000] { - let files = FileItems::All(&files[..count]); + let items = FileItems::All(&files[..count]); context.max_threads = 4; - let parallel = score_filtered_by_frecency(&files, &context, ArenaPtr::null()); + let parallel = frecency_scores(&items, &context, ArenaPtr::null()); context.max_threads = 1; - let sequential = score_filtered_by_frecency(&files, &context, ArenaPtr::null()); + let sequential = frecency_scores(&items, &context, ArenaPtr::null()); assert_eq!(parallel.len(), count - count.div_ceil(7)); assert_eq!(parallel.len(), sequential.len()); for ((a, a_score), (b, b_score)) in parallel.iter().zip(&sequential) { @@ -1108,6 +1196,29 @@ mod tests { assert_eq!(a_score.git_status_boost, b_score.git_status_boost); assert_eq!(a_score.git_recency_boost, b_score.git_recency_boost); } + + // Keyed ranking must produce the same page as ranking full scores. + context.max_threads = 4; + let (expected_items, expected_scores, expected_total) = + sort_and_paginate(parallel, &context, |score| score.total); + let (items, scores, total) = rank_by_frecency( + &files[..count], + &context, + count / 2, + ArenaPtr::null(), + ArenaPtr::null(), + ); + assert_eq!(total, expected_total); + assert_eq!(items.len(), expected_items.len()); + for ((a, a_score), (b, b_score)) in items + .iter() + .zip(&scores) + .zip(expected_items.iter().zip(&expected_scores)) + { + assert!(std::ptr::eq(*a, *b)); + assert_eq!(a_score.total, b_score.total); + assert_eq!(a_score.match_type, "frecency"); + } } } @@ -1160,7 +1271,8 @@ mod tests { .take(if limit == 0 { count } else { limit }) .map(|(_, score)| score.total) .collect(); - let (items, scores, total) = sort_and_paginate(results.clone(), &context); + let (items, scores, total) = + sort_and_paginate(results.clone(), &context, |s| s.total); assert_eq!(total, count); assert_eq!(items.len(), expected_page.len()); assert_eq!( @@ -1287,7 +1399,7 @@ mod tests { }; // Test with full sort - returns all results sorted descending - let (items, scores, total) = sort_and_paginate(results.clone(), &context); + let (items, scores, total) = sort_and_paginate(results.clone(), &context, |s| s.total); // Should return all 10 items sorted by score descending assert_eq!(total, 10); @@ -1337,7 +1449,7 @@ mod tests { }, }; - let (items, scores, _) = sort_and_paginate(results, &context); + let (items, scores, _) = sort_and_paginate(results, &context, |s| s.total); // Should return all 5 items sorted: 200(9000), 200(1000), 100(8000), 100(5000), 100(3000) assert_eq!(scores.len(), 5); @@ -1387,7 +1499,7 @@ mod tests { }; // Returns all results sorted descending - let (items, scores, _) = sort_and_paginate(results, &context); + let (items, scores, _) = sort_and_paginate(results, &context, |s| s.total); assert_eq!(scores.len(), 3); assert_eq!(scores[0].total, 200); diff --git a/crates/fff-core/src/sort_buffer.rs b/crates/fff-core/src/sort_buffer.rs index 940169b61..8e9b7367b 100644 --- a/crates/fff-core/src/sort_buffer.rs +++ b/crates/fff-core/src/sort_buffer.rs @@ -53,20 +53,6 @@ where } } -pub fn sort_by_key_with_buffer(slice: &mut [T], key_fn: F) -where - K: Ord, - F: FnMut(&T) -> K, -{ - match try_lock_shared_buf() { - Some(mut buf) => { - let typed = buf.as_slice_mut::(slice.len()); - glidesort::sort_with_buffer_by_key(slice, typed, key_fn); - } - None => glidesort::sort_by_key(slice, key_fn), - } -} - #[cfg(test)] mod tests { use super::*; @@ -78,13 +64,6 @@ mod tests { assert_eq!(data, vec![1, 2, 5, 8, 9]); } - #[test] - fn test_sort_by_key_with_buffer() { - let mut data = vec![(1, 50), (2, 20), (3, 80), (4, 10), (5, 90)]; - sort_by_key_with_buffer(&mut data, |a| a.1); - assert_eq!(data, vec![(4, 10), (2, 20), (1, 50), (3, 80), (5, 90)]); - } - #[test] fn test_reverse_sort() { let mut data = vec![1, 2, 3, 4, 5];