From ea4066190054ff9dd585919d7e6d653433274f3d Mon Sep 17 00:00:00 2001 From: gustav-fff <286169375+gustav-fff@users.noreply.github.com> Date: Sat, 3 Oct 2026 07:42:03 -0700 Subject: [PATCH 1/2] fix(core): cap detect_binary_per_byte at MAX_FFFILE_SIZE and stop on first NUL chunk --- crates/fff-core/src/types.rs | 33 ++++++++++++++++++++++++++++++++- 1 file changed, 32 insertions(+), 1 deletion(-) diff --git a/crates/fff-core/src/types.rs b/crates/fff-core/src/types.rs index 45d0578f2..6bae0943c 100644 --- a/crates/fff-core/src/types.rs +++ b/crates/fff-core/src/types.rs @@ -561,7 +561,8 @@ impl FileItem { /// Chunked classifier of the binary content of the file chunk by chunk /// accepts path which to reuse the allocated buffer for absolute path read pub(crate) fn detect_binary_per_byte(&self, path: &Path, chunk: &mut [u8]) { - if self.size == 0 { + // files above the grep cap are never searched, don't read them to EOF + if self.size == 0 || self.size > MAX_FFFILE_SIZE { return; } @@ -584,6 +585,7 @@ impl FileItem { Ok(n) => { if detect_binary_content(&chunk[..n]) { self.set_binary(true); + break; } } } @@ -1146,3 +1148,32 @@ mod content_cache_budget_tests { assert_eq!(budget.max_file_size, default.max_file_size); } } + +#[cfg(test)] +mod detect_binary_tests { + use super::*; + + fn classify(content: &[u8]) -> bool { + let root = tempfile::tempdir().unwrap(); + let path = root.path().join("f"); + std::fs::write(&path, content).unwrap(); + let (item, _) = FileItem::new(path.clone(), root.path(), None); + item.detect_binary_per_byte(&path, &mut [0u8; BINARY_CLASSIFICATION_CHUNK_SIZE]); + item.is_binary() + } + + #[test] + fn detects_nul_in_small_file() { + let mut content = vec![b'a'; BINARY_CLASSIFICATION_CHUNK_SIZE * 3]; + content[10] = 0; + assert!(classify(&content)); + assert!(!classify(b"plain text\n")); + } + + #[test] + fn skips_files_above_max_size() { + let mut content = vec![b'a'; MAX_FFFILE_SIZE as usize + 1]; + content[0] = 0; + assert!(!classify(&content)); + } +} From a2965d4731e501ae31e10be1580bb97cd441a058 Mon Sep 17 00:00:00 2001 From: gustav-fff <286169375+gustav-fff@users.noreply.github.com> Date: Sat, 3 Oct 2026 20:33:04 -0700 Subject: [PATCH 2/2] fix(core): classify first MAX_FFFILE_SIZE bytes of large files instead of skipping --- crates/fff-core/src/types.rs | 13 +++++++++---- 1 file changed, 9 insertions(+), 4 deletions(-) diff --git a/crates/fff-core/src/types.rs b/crates/fff-core/src/types.rs index 6bae0943c..88f2adc53 100644 --- a/crates/fff-core/src/types.rs +++ b/crates/fff-core/src/types.rs @@ -561,12 +561,11 @@ impl FileItem { /// Chunked classifier of the binary content of the file chunk by chunk /// accepts path which to reuse the allocated buffer for absolute path read pub(crate) fn detect_binary_per_byte(&self, path: &Path, chunk: &mut [u8]) { - // files above the grep cap are never searched, don't read them to EOF - if self.size == 0 || self.size > MAX_FFFILE_SIZE { + if self.size == 0 { return; } - let Ok(mut file) = std::fs::OpenOptions::new() + let Ok(file) = std::fs::OpenOptions::new() .write(false) .read(true) .open(path) @@ -574,6 +573,8 @@ impl FileItem { tracing::error!(path = ?path.display(), "Failed to open indexed file"); return; }; + // only classify the first MAX_FFFILE_SIZE bytes, never read large files to EOF + let mut file = file.take(MAX_FFFILE_SIZE); loop { match file.read(chunk) { @@ -1171,9 +1172,13 @@ mod detect_binary_tests { } #[test] - fn skips_files_above_max_size() { + fn reads_only_first_max_size_bytes() { let mut content = vec![b'a'; MAX_FFFILE_SIZE as usize + 1]; content[0] = 0; + assert!(classify(&content)); + + content[0] = b'a'; + content[MAX_FFFILE_SIZE as usize] = 0; assert!(!classify(&content)); } }