Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 7 additions & 1 deletion Makefile
Original file line number Diff line number Diff line change
Expand Up @@ -16,7 +16,7 @@ SHELL := bash
# string rather than the literal `-o` / `pipefail` tokens.
.SHELLFLAGS := -o pipefail -euc

.PHONY: build build-c-lib install uninstall test test-rust test-rescan test-rescan-known-defects rescan-probe test-c-smoke test-c-api test-lua test-lua-snap test-version test-bun test-node prepare-bun prepare-bun-packaged prepare-node set-npm-version header test-stress test-stress-seeded test-stress-random test-stress-regressions test-stress-repos test-node-stress sync-js-api sync-js-api-check bump-homebrew-formula bump-install-mcp-sh test-bun-compile
.PHONY: build build-c-lib install uninstall test test-rust test-rescan test-rescan-known-defects rescan-probe test-c-smoke test-c-api test-lua test-lua-snap test-version test-bun test-node prepare-bun prepare-bun-packaged prepare-node set-npm-version header test-stress test-stress-seeded test-stress-random test-stress-regressions test-stress-repos test-node-stress sync-js-api sync-js-api-check bump-homebrew-formula bump-install-mcp-sh test-bun-compile bench-search bench-grep

all: format test lint

Expand Down Expand Up @@ -86,6 +86,12 @@ test-setup:
test-rust:
cargo test --workspace --no-default-features --features zlob --exclude fff-nvim --exclude fff-python

bench-search:
cargo bench -p fff-search --no-default-features --features zlob --bench fuzzy_search_bench -- $(BENCH_ARGS)

bench-grep:
cargo bench -p fff-search --no-default-features --features zlob --bench grep_bench -- $(BENCH_ARGS)

# Watcher rescan harness: asserts that editing, build output, git activity and
# preview reads all stay on the incremental path instead of re-walking the tree.
test-rescan:
Expand Down
1 change: 1 addition & 0 deletions _typos.toml
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,7 @@ thm = "thm"
comparsion = "comparsion"
modfiers = "modfiers"
shcema = "shcema"
contrller = "contrller"

[default]
extend-ignore-re = [
Expand Down
5 changes: 4 additions & 1 deletion crates/fff-core/Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -34,6 +34,10 @@ required-features = ["zlob"]
name = "grep_bench"
harness = false

[[bench]]
name = "fuzzy_search_bench"
harness = false

[features]
# `ripgrep` is the pure-Rust walker/glob backend and is on by default so
# consumers build without a Zig toolchain. CI/release opt into zlob via
Expand Down Expand Up @@ -102,4 +106,3 @@ proptest = { version = "1", default-features = false, features = ["std", "fork"]
rand = { version = "0.8", features = ["small_rng"] }
tempfile = "3.8"
tracing-subscriber = { version = "0.3", features = ["env-filter"] }

135 changes: 135 additions & 0 deletions crates/fff-core/benches/fuzzy_search_bench.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,135 @@
use criterion::{BenchmarkId, Criterion, criterion_group, criterion_main};
use fff_search::{FilePicker, FilePickerOptions, FuzzySearchOptions, PaginationArgs, QueryParser};
use std::{hint::black_box, time::Duration};

mod support;

fn bench_fuzzy_search(c: &mut Criterion) {
if let Some(picker) = support::repo_picker() {
let current_file = match std::env::var("FFF_BENCH_CURRENT_FILE") {
Ok(file) => {
assert!(
picker.base_path().join(&file).is_file(),
"FFF_BENCH_CURRENT_FILE must exist in FFF_BENCH_PATH"
);
file
}
Err(_) => picker
.get_files()
.iter()
.find(|file| !file.is_deleted())
.map(|file| file.relative_path(&picker))
.expect("repository must contain files"),
};
eprintln!("Current file: {current_file}");
bench_picker(c, &picker, "fuzzy_search_repo", &current_file, true);
return;
}
for count in [1_000, 10_000, 100_000] {
let (_root, picker) = make_picker(count);
bench_picker(
c,
&picker,
&format!("fuzzy_search_{count}"),
"src/components/group_0/controller_0.rs",
false,
);
}
}

fn bench_picker(
c: &mut Criterion,
picker: &FilePicker,
name: &str,
current_file: &str,
repo: bool,
) {
let parser = QueryParser::default();
let mut group = c.benchmark_group(name);
group.sample_size(20);
group.warm_up_time(Duration::from_millis(if repo { 1000 } else { 300 }));
group.measurement_time(Duration::from_secs(if repo { 3 } else { 1 }));

let path_query =
std::env::var("FFF_BENCH_PATH_QUERY").unwrap_or_else(|_| "src/components".into());
let multipart_query = std::env::var("FFF_BENCH_MULTIPART_QUERY")
.unwrap_or_else(|_| "components controller".into());
for (name, text) in [
("empty", ""),
("short", "mo"),
("filename", "controller"),
("path", path_query.as_str()),
("multipart", multipart_query.as_str()),
("extension", "*.rs"),
] {
let query = parser.parse(text);
for current_file in [None, Some(current_file)] {
let options = FuzzySearchOptions {
max_threads: 4,
current_file,
pagination: PaginationArgs {
offset: 0,
limit: 50,
},
..Default::default()
};
let suffix = if current_file.is_some() {
"current_file"
} else {
"no_current_file"
};
if repo {
let result = picker.fuzzy_search(&query, None, options);
eprintln!("Query {name}/{suffix}: matched={}", result.total_matched);
}
group.bench_function(BenchmarkId::new(name, suffix), |b| {
b.iter(|| black_box(picker.fuzzy_search(black_box(&query), None, options)));
});
}
}
group.finish();
}

fn make_picker(count: usize) -> (tempfile::TempDir, FilePicker) {
let root = tempfile::tempdir().unwrap();
let areas = [
"src/components",
"src/core",
"tests/integration",
"packages/server",
];
let names = ["controller", "model", "service", "utils", "index"];
let extensions = ["rs", "ts", "lua", "py"];
for group in 0..count.div_ceil(100) {
let directory = root
.path()
.join(format!("{}/group_{group}", areas[group % areas.len()]));
std::fs::create_dir_all(&directory).unwrap();
for index in group * 100..((group + 1) * 100).min(count) {
let name = format!(
"{}_{index}.{}",
names[index % names.len()],
extensions[index % extensions.len()]
);
let file = std::fs::File::create(directory.join(name)).unwrap();
let modified =
std::time::UNIX_EPOCH + Duration::from_secs(1_700_000_000 + (index % 1000) as u64);
file.set_times(std::fs::FileTimes::new().set_modified(modified))
.unwrap();
}
}
let mut picker = FilePicker::new(FilePickerOptions {
base_path: root.path().to_string_lossy().into_owned(),
enable_mmap_cache: false,
enable_content_indexing: false,
watch: false,
..Default::default()
})
.unwrap();
picker.collect_files().unwrap();
assert_eq!(picker.live_file_count(), count);
(root, picker)
}

criterion_group!(benches, bench_fuzzy_search);
criterion_main!(benches);
62 changes: 62 additions & 0 deletions crates/fff-core/benches/grep_bench.rs
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,8 @@ use fff_search::file_picker::{FilePicker, FilePickerOptions};
use fff_search::{GrepMode, GrepSearchOptions, parse_grep_query};
use std::io::Write;

mod support;

/// Synthetic repo: half the files contain the needle on every line (stresses
/// the per-match find/highlight path), half are pure noise (stresses the
/// whole-file prefilter path).
Expand Down Expand Up @@ -41,6 +43,10 @@ fn options(mode: GrepMode) -> GrepSearchOptions {
}

fn bench_grep(c: &mut Criterion) {
if let Some(picker) = support::repo_picker() {
bench_repo(c, &picker);
return;
}
let dir = tempfile::tempdir().unwrap();
setup_repo(dir.path());

Expand Down Expand Up @@ -98,6 +104,62 @@ fn bench_grep(c: &mut Criterion) {
});
});

let fuzzy_opts = options(GrepMode::Fuzzy);
for (name, text) in [
("fuzzy_exact_many_matches", "controller"),
("fuzzy_typo_many_matches", "contrller"),
] {
let query = parse_grep_query(text);
group.bench_function(name, |b| {
b.iter(|| {
let result = picker.grep(&query, &fuzzy_opts);
assert_eq!(result.files_with_matches, 400);
std::hint::black_box(result.matches.len())
});
});
}

group.finish();
}

fn bench_repo(c: &mut Criterion, picker: &FilePicker) {
let mut group = c.benchmark_group("grep_repo");
group.sample_size(10);
group.sampling_mode(criterion::SamplingMode::Flat);
group.warm_up_time(std::time::Duration::from_secs(1));
group.measurement_time(std::time::Duration::from_secs(3));
for (name, text, mode) in [
("plain_sensitive", "Controller", GrepMode::PlainText),
("plain_insensitive", "controller", GrepMode::PlainText),
(
"plain_no_matches",
"FFF_BENCH_absent_f891c75d",
GrepMode::PlainText,
),
("regex", "Contr[a-z]+ller", GrepMode::Regex),
("fuzzy_exact", "controller", GrepMode::Fuzzy),
("fuzzy_typo", "contrller", GrepMode::Fuzzy),
] {
let query = parse_grep_query(text);
let options = GrepSearchOptions {
max_file_size: 10 * 1024 * 1024,
mode,
..Default::default()
};
let result = picker.grep(&query, &options);
eprintln!(
"Query {name}: matches={} files_with_matches={} searched={} filtered={}",
result.matches.len(),
result.files_with_matches,
result.total_files_searched,
result.filtered_file_count
);
assert!(result.regex_fallback_error.is_none());
drop(result);
group.bench_function(name, |b| {
b.iter(|| std::hint::black_box(picker.grep(&query, &options)));
});
}
group.finish();
}

Expand Down
45 changes: 45 additions & 0 deletions crates/fff-core/benches/support/mod.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,45 @@
use fff_search::{FilePicker, FilePickerOptions};
use std::hash::{DefaultHasher, Hash, Hasher};

pub(super) fn repo_picker() -> Option<FilePicker> {
let path = std::env::var_os("FFF_BENCH_PATH")?;
let path = std::fs::canonicalize(path).expect("FFF_BENCH_PATH must exist");
assert!(path.is_dir(), "FFF_BENCH_PATH must be a directory");
eprintln!("Indexing {}", path.display());
let mut picker = FilePicker::new(FilePickerOptions {
base_path: path.to_str().expect("UTF-8 repository path").into(),
enable_mmap_cache: false,
enable_content_indexing: false,
watch: false,
..Default::default()
})
.unwrap();
picker.collect_files().unwrap();
assert!(
picker.live_file_count() > 0,
"repository must contain files"
);

let mut files: Vec<_> = picker
.get_files()
.iter()
.map(|file| {
(
file.relative_path(&picker),
file.size,
file.modified,
file.git_status.map(|status| status.bits()),
file.git_recency_score,
)
})
.collect();
files.sort_unstable();
let mut fingerprint = DefaultHasher::new();
files.hash(&mut fingerprint);
eprintln!(
"Dataset: files={} fingerprint={:016x}",
files.len(),
fingerprint.finish()
);
Some(picker)
}
6 changes: 2 additions & 4 deletions crates/fff-core/src/grep/fuzzy_grep.rs
Original file line number Diff line number Diff line change
@@ -1,14 +1,12 @@
use crate::match_offsets::char_indices_to_byte_offsets;
use crate::simd_path::ArenaPtr;
use crate::types::{ContentCacheBudget, FileItem, MmapSlot};
use fff_grep::lines::LineStep;
use rayon::prelude::*;
use std::path::Path;
use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering};

use super::sink::{
char_indices_to_byte_offsets, classify_definition, strip_line_terminators,
truncate_display_bytes,
};
use super::sink::{classify_definition, strip_line_terminators, truncate_display_bytes};
use super::types::{GrepMatch, GrepResult, GrepSearchOptions};

#[allow(clippy::too_many_arguments)]
Expand Down
46 changes: 0 additions & 46 deletions crates/fff-core/src/grep/sink.rs
Original file line number Diff line number Diff line change
Expand Up @@ -190,52 +190,6 @@ pub(super) fn split_multiline_blob(display_bytes: &[u8]) -> (&[u8], Vec<String>)
}
}

/// Convert character-position indices from neo_frizbee into byte-offset
/// pairs (start, end) suitable for `match_byte_offsets`.
///
/// frizbee returns character positions (0-based index into the char
/// iterator). We need byte ranges because the UI renderer and Lua layer
/// use byte offsets for extmark highlights.
///
/// Each matched character becomes its own (byte_start, byte_end) pair.
/// Adjacent characters are merged into a single contiguous range.
pub(super) fn char_indices_to_byte_offsets(
line: &str,
char_indices: &[u32],
) -> SmallVec<[(u32, u32); 4]> {
if char_indices.is_empty() {
return SmallVec::new();
}

// Build a map: char_index -> (byte_start, byte_end) for all chars.
// Iterating all chars is O(n) in the line length which is bounded by MAX_LINE_DISPLAY_LEN (512).
let char_byte_ranges: Vec<(usize, usize)> = line
.char_indices()
.map(|(byte_pos, ch)| (byte_pos, byte_pos + ch.len_utf8()))
.collect();

// Convert char indices to byte ranges, merging adjacent ranges
let mut result: SmallVec<[(u32, u32); 4]> = SmallVec::with_capacity(char_indices.len());

for &ci in char_indices {
let ci = ci as usize;
if ci >= char_byte_ranges.len() {
continue; // out of bounds (shouldn't happen with valid data)
}
let (start, end) = char_byte_ranges[ci];
// Merge with previous range if adjacent
if let Some(last) = result.last_mut()
&& last.1 == start as u32
{
last.1 = end as u32;
continue;
}
result.push((start as u32, end as u32));
}

result
}

// copied from the rust u8 private method
#[inline]
const fn is_utf8_char_boundary(b: u8) -> bool {
Expand Down
2 changes: 2 additions & 0 deletions crates/fff-core/src/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -136,6 +136,8 @@ pub mod path_utils;
pub mod types;
pub use types::*;

mod match_offsets;

pub mod constants;

/// Watcher rescan request accounting.
Expand Down
Loading
Loading