From 224ce7863fa6f62ef7c1338a99af9c6e91f9b6d2 Mon Sep 17 00:00:00 2001 From: Dmitriy Kovalenko Date: Tue, 14 Jul 2026 16:59:06 -0700 Subject: [PATCH] feat: Mulitline search --- AGENTS.md | 7 + README.md | 1 + crates/fff-core/Cargo.toml | 4 + crates/fff-core/benches/grep_bench.rs | 105 ++ crates/fff-core/src/file_picker.rs | 2 +- crates/fff-core/src/grep/fuzzy_grep.rs | 6 +- crates/fff-core/src/grep/grep.rs | 1666 ++++------------- crates/fff-core/src/grep/grep_tests.rs | 135 +- crates/fff-core/src/grep/mod.rs | 19 +- crates/fff-core/src/grep/multi_pattern.rs | 190 ++ crates/fff-core/src/grep/prefilter.rs | 204 ++ crates/fff-core/src/grep/regex.rs | 130 ++ crates/fff-core/src/grep/sink.rs | 242 +++ crates/fff-core/src/grep/types.rs | 239 +++ crates/fff-core/src/grep/utils.rs | 122 -- .../fff-core/src/{ => index}/bigram_filter.rs | 0 .../fff-core/src/{ => index}/bigram_query.rs | 4 +- crates/fff-core/src/index/candidates.rs | 118 ++ .../fff-core/src/{ => index}/constraints.rs | 0 crates/fff-core/src/index/mod.rs | 11 + crates/fff-core/src/lib.rs | 7 +- crates/fff-core/src/scan.rs | 2 +- crates/fff-core/src/score.rs | 2 +- crates/fff-core/src/types.rs | 2 +- lua/fff/conf.lua | 4 + lua/fff/picker_ui/ui_creator.lua | 10 + 26 files changed, 1777 insertions(+), 1455 deletions(-) create mode 100644 crates/fff-core/benches/grep_bench.rs create mode 100644 crates/fff-core/src/grep/multi_pattern.rs create mode 100644 crates/fff-core/src/grep/prefilter.rs create mode 100644 crates/fff-core/src/grep/regex.rs create mode 100644 crates/fff-core/src/grep/sink.rs create mode 100644 crates/fff-core/src/grep/types.rs delete mode 100644 crates/fff-core/src/grep/utils.rs rename crates/fff-core/src/{ => index}/bigram_filter.rs (100%) rename crates/fff-core/src/{ => index}/bigram_query.rs (99%) create mode 100644 crates/fff-core/src/index/candidates.rs rename crates/fff-core/src/{ => index}/constraints.rs (100%) create mode 100644 crates/fff-core/src/index/mod.rs diff --git a/AGENTS.md b/AGENTS.md index 5da8e1e1a..87bfb91b2 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -31,6 +31,13 @@ When doing code make sure to REDUCE SIZE OF COMMENTS. This is very important. Ev - Do not make public structs if something can be private +## Style guide + +- NO MODULES COMMENTS +- NO TOP FILE COMMENTS +- NO COMMENT LONGER THAN 2 LINES UNLESS ASKED EXPLICITLY +- UTILITY FUNCTIONS GO INTO THE END OF FILE + ## Architecture Everything that is performance critical happens in rust world, everything that is neovim specific happens in the lua code. diff --git a/README.md b/README.md index c7bb11e01..e84a5360d 100644 --- a/README.md +++ b/README.md @@ -317,6 +317,7 @@ require('fff').setup({ preview_scroll_down = '', toggle_debug = '', cycle_grep_modes = '', + insert_newline_escape = '', -- grep mode only: jump cursor to first match of next/prev file group grep_jump_to_next_file = { '', '' }, grep_jump_to_prev_file = { '', '' }, diff --git a/crates/fff-core/Cargo.toml b/crates/fff-core/Cargo.toml index 5600fd327..fc07dd830 100644 --- a/crates/fff-core/Cargo.toml +++ b/crates/fff-core/Cargo.toml @@ -27,6 +27,10 @@ name = "glob_bench" harness = false required-features = ["zlob"] +[[bench]] +name = "grep_bench" +harness = false + [features] # `ripgrep` is the pure-Rust walker/glob backend and is on by default so # consumers build without a Zig toolchain. CI/release opt into zlob via diff --git a/crates/fff-core/benches/grep_bench.rs b/crates/fff-core/benches/grep_bench.rs new file mode 100644 index 000000000..e04db4d0e --- /dev/null +++ b/crates/fff-core/benches/grep_bench.rs @@ -0,0 +1,105 @@ +use criterion::{Criterion, criterion_group, criterion_main}; +use fff_search::file_picker::{FilePicker, FilePickerOptions}; +use fff_search::{GrepMode, GrepSearchOptions, parse_grep_query}; +use std::io::Write; + +/// Synthetic repo: half the files contain the needle on every line (stresses +/// the per-match find/highlight path), half are pure noise (stresses the +/// whole-file prefilter path). +fn setup_repo(dir: &std::path::Path) { + for i in 0..400 { + let mut f = std::fs::File::create(dir.join(format!("match_{i}.rs"))).unwrap(); + for j in 0..100 { + writeln!( + f, + "fn handle_{j}() {{ let controller = Controller::new({j}); controller.run(); }}" + ) + .unwrap(); + } + } + for i in 0..400 { + let mut f = std::fs::File::create(dir.join(format!("noise_{i}.rs"))).unwrap(); + for j in 0..100 { + writeln!( + f, + "fn compute_{j}() {{ let value = {j} * 42; process(value); }}" + ) + .unwrap(); + } + } +} + +fn options(mode: GrepMode) -> GrepSearchOptions { + GrepSearchOptions { + // Force a full scan of every file so we measure matcher/sink work, + // not pagination early-exit. + page_limit: usize::MAX, + max_matches_per_file: 0, + mode, + ..Default::default() + } +} + +fn bench_grep(c: &mut Criterion) { + let dir = tempfile::tempdir().unwrap(); + setup_repo(dir.path()); + + let mut picker = FilePicker::new(FilePickerOptions { + base_path: dir.path().to_str().unwrap().into(), + watch: false, + ..Default::default() + }) + .unwrap(); + picker.collect_files().unwrap(); + assert_eq!(picker.get_files().len(), 800); + + let mut group = c.benchmark_group("grep_e2e"); + group.sample_size(30); + + // Case-sensitive, 40k matched lines: hottest find_at/highlight path + let query = parse_grep_query("Controller"); + let opts = options(GrepMode::PlainText); + group.bench_function("plain_case_sensitive_many_matches", |b| { + b.iter(|| { + let r = picker.grep(&query, &opts); + assert_eq!(r.files_with_matches, 400); + std::hint::black_box(r.matches.len()) + }); + }); + + // Case-insensitive (SIMD folding path), 120k matched spans + let query = parse_grep_query("controller"); + group.bench_function("plain_case_insensitive_many_matches", |b| { + b.iter(|| { + let r = picker.grep(&query, &opts); + assert_eq!(r.files_with_matches, 400); + std::hint::black_box(r.matches.len()) + }); + }); + + // No matches anywhere: whole-file prefilter dominates + let query = parse_grep_query("Qqzyx"); + group.bench_function("plain_no_matches", |b| { + b.iter(|| { + let r = picker.grep(&query, &opts); + assert_eq!(r.files_with_matches, 0); + std::hint::black_box(r.total_files_searched) + }); + }); + + // Regex mode: must be unaffected by NeedleFinder changes + let query = parse_grep_query("Contr[a-z]+ller"); + let regex_opts = options(GrepMode::Regex); + group.bench_function("regex_many_matches", |b| { + b.iter(|| { + let r = picker.grep(&query, ®ex_opts); + assert_eq!(r.files_with_matches, 400); + std::hint::black_box(r.matches.len()) + }); + }); + + group.finish(); +} + +criterion_group!(benches, bench_grep); +criterion_main!(benches); diff --git a/crates/fff-core/src/file_picker.rs b/crates/fff-core/src/file_picker.rs index 0d51745de..126352f04 100644 --- a/crates/fff-core/src/file_picker.rs +++ b/crates/fff-core/src/file_picker.rs @@ -32,12 +32,12 @@ use crate::FFFStringStorage; use crate::background_watcher::BackgroundWatcher; -use crate::bigram_filter::{BigramFilter, BigramOverlay}; use crate::constants::{MAX_OVERFLOW_FILES, PATH_BUF_SIZE}; use crate::error::Error; use crate::frecency::FrecencyTracker; use crate::git::GitStatusCache; use crate::grep::{GrepResult, GrepSearchOptions, grep_search, multi_grep_search}; +use crate::index::{BigramFilter, BigramOverlay}; use crate::query_tracker::QueryTracker; use crate::scan::{ScanConfig, ScanJob, ScanSignals}; use crate::score::fuzzy_match_and_score_files; diff --git a/crates/fff-core/src/grep/fuzzy_grep.rs b/crates/fff-core/src/grep/fuzzy_grep.rs index 75e1f8f6e..6f6becbeb 100644 --- a/crates/fff-core/src/grep/fuzzy_grep.rs +++ b/crates/fff-core/src/grep/fuzzy_grep.rs @@ -5,11 +5,11 @@ use rayon::prelude::*; use std::path::Path; use std::sync::atomic::{AtomicBool, Ordering}; -use super::grep::{ - GrepMatch, GrepSearchOptions, char_indices_to_byte_offsets, classify_definition, +use super::sink::{ + char_indices_to_byte_offsets, classify_definition, strip_line_terminators, truncate_display_bytes, }; -use super::utils::{GrepResult, strip_line_terminators}; +use super::types::{GrepMatch, GrepResult, GrepSearchOptions}; #[allow(clippy::too_many_arguments)] pub(super) fn fuzzy_grep_search<'a>( diff --git a/crates/fff-core/src/grep/grep.rs b/crates/fff-core/src/grep/grep.rs index 27f72bdd6..4c825c6b5 100644 --- a/crates/fff-core/src/grep/grep.rs +++ b/crates/fff-core/src/grep/grep.rs @@ -1,292 +1,93 @@ -use crate::{ - bigram_filter::{BigramFilter, BigramOverlay, extract_bigrams}, - bigram_query::{fuzzy_to_bigram_query, regex_to_bigram_query}, - constraints::{ConstraintPlan, ConstraintsBuffers}, - simd_string_utils::memmem, - sort_buffer::sort_with_buffer, - types::{ContentCacheBudget, FileItem, FileSliceExt, MmapSlot}, +use super::prefilter::prefilter_with_filepath_retry; +use super::regex::{RegexMatcher, RegexSink, build_regex}; +use super::sink::{SinkState, debug_assert_newline_terminator}; +use super::types::{GrepMatch, GrepMode, GrepResult, GrepSearchOptions}; +use crate::index::{ + BigramFilter, BigramOverlay, bigram_boundary, fuzzy_candidates, literal_candidates, + regex_candidates, }; -use aho_corasick::AhoCorasick; +use crate::simd_string_utils::memmem; +use crate::types::{ContentCacheBudget, FileItem, FileSliceExt, MmapSlot}; use fff_grep::{ Searcher, SearcherBuilder, Sink, SinkMatch, matcher::{Match, Matcher, NoError}, }; -use fff_query_parser::{Constraint, FFFQuery, GrepConfig, QueryParser}; +use fff_query_parser::{FFFQuery, GrepConfig, QueryParser}; use rayon::prelude::*; use smallvec::SmallVec; use std::path::Path; -use std::sync::Arc; use std::sync::atomic::{AtomicBool, Ordering}; use tracing::Level; -use super::utils::{GrepResult, strip_line_terminators}; - -#[cfg(feature = "definitions")] -#[inline] -pub(super) fn classify_definition(enabled: bool, line: &str) -> bool { - enabled && super::classify::is_definition_line(line) -} - -#[cfg(not(feature = "definitions"))] -#[inline] -pub(super) fn classify_definition(_enabled: bool, _line: &str) -> bool { - false -} - -/// Check if `text` contains `\n` that is NOT preceded by another `\`. -/// -/// `\n` -> true (user wants multiline search) -/// `\\n` -> false (escaped backslash followed by literal `n`, e.g. `\\nvim-data`) -#[inline] -pub(super) fn has_unescaped_newline_escape(text: &str) -> bool { - let bytes = text.as_bytes(); - let mut i = 0; - while i < bytes.len().saturating_sub(1) { - if bytes[i] == b'\\' { - if bytes[i + 1] == b'n' { - // Count consecutive backslashes ending at position i - let mut backslash_count = 1; - while backslash_count <= i && bytes[i - backslash_count] == b'\\' { - backslash_count += 1; - } - // Odd number of backslashes before 'n' -> real \n escape - if backslash_count % 2 == 1 { - return true; - } - } - // Skip past the escaped character - i += 2; - } else { - i += 1; - } - } - false +#[allow(clippy::large_enum_variant)] +pub(super) enum NeedleFinder<'a> { + CaseSensitive(memchr::memmem::Finder<'a>), + /// Pre-lowered needle bytes for the SIMD case-insensitive search. + CaseInsensitive(&'a [u8]), } -/// Replace only unescaped `\n` sequences with real newlines. -/// -/// `\n` -> newline character -/// `\\n` -> preserved as-is (literal backslash + `n`) -pub(super) fn replace_unescaped_newline_escapes(text: &str) -> String { - let bytes = text.as_bytes(); - let mut result = Vec::with_capacity(bytes.len()); - let mut i = 0; - while i < bytes.len() { - if bytes[i] == b'\\' && i + 1 < bytes.len() { - if bytes[i + 1] == b'n' { - let mut backslash_count = 1; - while backslash_count <= i && bytes[i - backslash_count] == b'\\' { - backslash_count += 1; - } - if backslash_count % 2 == 1 { - result.push(b'\n'); - i += 2; - continue; - } - } - result.push(bytes[i]); - i += 1; +impl<'a> NeedleFinder<'a> { + fn new(needle: &'a [u8], case_insensitive: bool) -> Self { + if case_insensitive { + Self::CaseInsensitive(needle) } else { - result.push(bytes[i]); - i += 1; - } - } - String::from_utf8(result).unwrap_or_else(|_| text.to_string()) -} - -/// Controls how the grep pattern is interpreted. -#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] -pub enum GrepMode { - /// Literal plain text match: default path that doesn't require any regex machinery - #[default] - PlainText, - /// Regex mode: uses the same exact matching engine as ripgrep - Regex, - /// Smart fuzzy mode, allows user to make either a couple of single char typos or long gaps - /// e.g. shcema -> shcema, or UserController -> UserAuthController - /// - /// Significatnly slower than plain text, especially on unindexed FilePicker - Fuzzy, -} - -/// A single content match within a file -#[derive(Debug, Clone)] -pub struct GrepMatch { - /// Index into the deduplicated `files` vec of the GrepResult. - pub file_index: usize, - /// 1-based line number. - pub line_number: u64, - /// 0-based byte column of first match start within the line. - pub col: usize, - /// Absolute byte offset of the matched line from the start of the file. - /// Can be used by the preview to seek directly without scanning from the top. - pub byte_offset: u64, - /// The matched line text, truncated to `MAX_LINE_DISPLAY_LEN`. - pub line_content: String, - /// Byte offsets `(start, end)` within `line_content` for each match. - /// Stack-allocated for the common case of ≤4 spans per line. - pub match_byte_offsets: SmallVec<[(u32, u32); 4]>, - /// Fuzzy match score from neo_frizbee (only set in Fuzzy grep mode). - pub fuzzy_score: Option, - /// Whether the matched line looks like a definition (struct, fn, class, etc.). - /// Computed at match time so output formatters don't need to re-scan. - pub is_definition: bool, - /// Lines before the match (for context display). Empty when context is 0. - pub context_before: Vec, - /// Lines after the match (for context display). Empty when context is 0. - pub context_after: Vec, -} - -impl GrepMatch { - /// Strip leading whitespace from `line_content` and all context lines, - /// adjusting `col` and `match_byte_offsets` so highlights remain correct. - pub fn trim_leading_whitespace(&mut self) { - let strip_len = self.line_content.len() - self.line_content.trim_start().len(); - if strip_len > 0 { - self.line_content.drain(..strip_len); - let off = strip_len as u32; - self.col = self.col.saturating_sub(strip_len); - for range in &mut self.match_byte_offsets { - range.0 = range.0.saturating_sub(off); - range.1 = range.1.saturating_sub(off); - } - } - for line in &mut self.context_before { - let n = line.len() - line.trim_start().len(); - if n > 0 { - line.drain(..n); - } - } - for line in &mut self.context_after { - let n = line.len() - line.trim_start().len(); - if n > 0 { - line.drain(..n); - } + Self::CaseSensitive(memchr::memmem::Finder::new(needle)) } } -} - -pub use crate::constants::MAX_FFFILE_SIZE; - -/// Options for grep search. -#[derive(Debug, Clone)] -pub struct GrepSearchOptions { - pub max_file_size: u64, - pub max_matches_per_file: usize, - pub smart_case: bool, - /// File-based pagination offset: index into the sorted/filtered file list - /// to start searching from. Pass 0 for the first page, then use - /// `GrepResult::next_file_offset` for subsequent pages. - pub file_offset: usize, - /// Maximum number of matches to collect before stopping. - pub page_limit: usize, - /// How to interpret the search pattern. Defaults to `PlainText`. - pub mode: GrepMode, - /// Maximum time in milliseconds to spend searching before returning partial - /// results. Prevents UI freezes on pathological queries. 0 = no limit. - pub time_budget_ms: u64, - /// Number of context lines to include before each match. 0 = disabled. - pub before_context: usize, - /// Number of context lines to include after each match. 0 = disabled. - pub after_context: usize, - /// Whether to classify each match as a definition line. Adds ~2% overhead - /// on large repos; disable for interactive grep where it is not needed. - pub classify_definitions: bool, - /// Strip leading whitespace from matched lines and context lines, adjusting - /// highlight byte offsets accordingly. Useful for AI/MCP consumers and UIs - /// that don't need indentation. Default: false. - pub trim_whitespace: bool, - /// External abort signal. When provided, overrides the picker's internal - /// cancellation flag. Set to `true` to stop the search early and return - /// partial results. Omit (or use `..Default::default()`) to let the - /// picker manage cancellation. - pub abort_signal: Option>, -} -impl Default for GrepSearchOptions { - fn default() -> Self { - Self { - max_file_size: MAX_FFFILE_SIZE, - max_matches_per_file: 200, - smart_case: true, - file_offset: 0, - page_limit: 50, - mode: GrepMode::default(), - time_budget_ms: 0, - before_context: 0, - after_context: 0, - classify_definitions: false, - trim_whitespace: false, - abort_signal: None, + #[inline] + fn find(&self, haystack: &[u8]) -> Option { + match self { + Self::CaseSensitive(finder) => finder.find(haystack), + Self::CaseInsensitive(needle_lower) => memmem::find(haystack, needle_lower), } } -} - -#[derive(Clone, Copy)] -struct GrepContext<'a, 'b> { - total_files: usize, - filtered_file_count: usize, - budget: &'a ContentCacheBudget, - base_path: &'a Path, - arena: crate::simd_path::ArenaPtr, - overflow_arena: crate::simd_path::ArenaPtr, - prefilter: Option<&'a memchr::memmem::Finder<'b>>, - prefilter_case_insensitive: bool, - abort_signal: &'a AtomicBool, -} -impl GrepContext<'_, '_> { #[inline] - fn arena_for_file(&self, file: &FileItem) -> crate::simd_path::ArenaPtr { - if file.is_overflow() { - self.overflow_arena - } else { - self.arena + fn needle(&self) -> &[u8] { + match self { + Self::CaseSensitive(finder) => finder.needle(), + Self::CaseInsensitive(needle_lower) => needle_lower, } } -} - -struct RegexMatcher<'r> { - regex: &'r regex::bytes::Regex, - is_multiline: bool, -} - -impl Matcher for RegexMatcher<'_> { - type Error = NoError; + /// Compare `haystack` against a slice of the needle with the same case + /// semantics as `find`. #[inline] - fn find_at(&self, haystack: &[u8], at: usize) -> Result, NoError> { - Ok(self - .regex - .find_at(haystack, at) - .map(|m| Match::new(m.start(), m.end()))) + fn eq_fold(&self, haystack: &[u8], needle_seg: &[u8]) -> bool { + match self { + Self::CaseSensitive(_) => haystack == needle_seg, + Self::CaseInsensitive(_) => { + haystack.len() == needle_seg.len() && memmem::find(haystack, needle_seg) == Some(0) + } + } } + /// Collect highlight spans for every needle occurrence within a line. + /// The case branch is resolved once per line, not once per occurrence. #[inline] - fn line_terminator(&self) -> Option { - if self.is_multiline { - None - } else { - Some(fff_grep::LineTerminator::byte(b'\n')) + fn for_each_occurrence(&self, haystack: &[u8], mut on_match: impl FnMut(usize)) { + match self { + Self::CaseSensitive(finder) => { + let mut start_pos = 0usize; + while let Some(pos) = finder.find(&haystack[start_pos..]) { + on_match(start_pos + pos); + start_pos += pos + 1; + } + } + Self::CaseInsensitive(needle_lower) => { + let mut start_pos = 0usize; + while let Some(pos) = memmem::find(&haystack[start_pos..], needle_lower) { + on_match(start_pos + pos); + start_pos += pos + 1; + } + } } } } -/// A `grep_matcher::Matcher` backed by `memchr::memmem` for literal search. -/// -/// This is used in `PlainText` mode and is significantly faster than regex -/// for literal patterns: memchr uses SIMD (AVX2/NEON) two-way substring -/// search internally, avoiding the overhead of regex compilation and DFA -/// state transitions. -/// -/// Always reports `\n` as line terminator so the searcher uses the fast -/// candidate-line path (plain text can never span lines unless `\n` is -/// literally in the needle, which we handle separately). struct PlainTextMatcher<'a> { - /// Case-folded needle bytes for case-insensitive matching. - /// When case-sensitive, this is the original pattern bytes. - needle: &'a [u8], - case_insensitive: bool, + finder: &'a NeedleFinder<'a>, } impl Matcher for PlainTextMatcher<'_> { @@ -295,14 +96,12 @@ impl Matcher for PlainTextMatcher<'_> { #[inline] fn find_at(&self, haystack: &[u8], at: usize) -> Result, NoError> { let hay = &haystack[at..]; + let needle_len = self.finder.needle().len(); - let found = if self.case_insensitive { - memmem::find(hay, self.needle) - } else { - memchr::memmem::find(hay, self.needle) - }; - - Ok(found.map(|pos| Match::new(at + pos, at + pos + self.needle.len()))) + Ok(self + .finder + .find(hay) + .map(|pos| Match::new(at + pos, at + pos + needle_len))) } #[inline] @@ -311,238 +110,22 @@ impl Matcher for PlainTextMatcher<'_> { } } -/// Maximum bytes of a matched line to keep for display. Prevents minified -/// JS or huge single-line files from blowing up memory. -const MAX_LINE_DISPLAY_LEN: usize = 512; - -struct SinkState { - file_index: usize, - matches: Vec, - max_matches: usize, - before_context: usize, - after_context: usize, - classify_definitions: bool, -} - -impl SinkState { - #[inline] - fn prepare_line<'a>(line_bytes: &'a [u8], mat: &SinkMatch<'_>) -> (&'a [u8], u32, u64, u64) { - let line_number = mat.line_number().unwrap_or(0); - let byte_offset = mat.absolute_byte_offset(); - - // Trim trailing newline/CR directly on bytes to avoid UTF-8 conversion. - let trimmed_bytes = strip_line_terminators(line_bytes); - - // Truncate for display (floor to a char boundary). - let display_bytes = truncate_display_bytes(trimmed_bytes); - - let display_len = display_bytes.len() as u32; - (display_bytes, display_len, line_number, byte_offset) - } - - #[inline] - #[allow(clippy::too_many_arguments)] - fn push_match( - &mut self, - line_number: u64, - col: usize, - byte_offset: u64, - line_content: String, - match_byte_offsets: SmallVec<[(u32, u32); 4]>, - context_before: Vec, - context_after: Vec, - ) { - let is_definition = classify_definition(self.classify_definitions, &line_content); - self.matches.push(GrepMatch { - file_index: self.file_index, - line_number, - col, - byte_offset, - line_content, - match_byte_offsets, - fuzzy_score: None, - is_definition, - context_before, - context_after, - }); - } - - /// Extract context lines from the full buffer around a matched region. - fn extract_context(&self, mat: &SinkMatch<'_>) -> (Vec, Vec) { - if self.before_context == 0 && self.after_context == 0 { - return (Vec::new(), Vec::new()); - } - - let buffer = mat.buffer(); - let range = mat.bytes_range_in_buffer(); - - let mut before = Vec::new(); - if self.before_context > 0 && range.start > 0 { - // Walk backward from the start of the match line to find preceding lines - let mut pos = range.start; - let mut lines_found = 0; - while lines_found < self.before_context && pos > 0 { - // Skip the newline just before our current position - pos -= 1; - // Find the previous newline - let line_start = match memchr::memrchr(b'\n', &buffer[..pos]) { - Some(nl) => nl + 1, - None => 0, - }; - let line = &buffer[line_start..pos]; - // Trim trailing \r - let line = if line.last() == Some(&b'\r') { - &line[..line.len() - 1] - } else { - line - }; - let truncated = truncate_display_bytes(line); - before.push(String::from_utf8_lossy(truncated).into_owned()); - pos = line_start; - lines_found += 1; - } - before.reverse(); - } - - let mut after = Vec::new(); - if self.after_context > 0 && range.end < buffer.len() { - let mut pos = range.end; - let mut lines_found = 0; - while lines_found < self.after_context && pos < buffer.len() { - // Find the next newline - let line_end = match memchr::memchr(b'\n', &buffer[pos..]) { - Some(nl) => pos + nl, - None => buffer.len(), - }; - let line = &buffer[pos..line_end]; - // Trim trailing \r - let line = if line.last() == Some(&b'\r') { - &line[..line.len() - 1] - } else { - line - }; - let truncated = truncate_display_bytes(line); - after.push(String::from_utf8_lossy(truncated).into_owned()); - pos = if line_end < buffer.len() { - line_end + 1 // skip past \n - } else { - buffer.len() - }; - lines_found += 1; - } - } - - (before, after) - } -} - -/// Truncate a byte slice for display, respecting UTF-8 char boundaries. -#[inline] -pub(super) fn truncate_display_bytes(bytes: &[u8]) -> &[u8] { - if bytes.len() <= MAX_LINE_DISPLAY_LEN { - bytes - } else { - let mut end = MAX_LINE_DISPLAY_LEN; - while end > 0 && !is_utf8_char_boundary(bytes[end]) { - end -= 1; - } - &bytes[..end] - } -} - -/// Sink for `PlainText` mode. -/// -/// Highlights are extracted with `memchr::memmem::Finder` (case-sensitive) -/// or the SIMD `simd_string_utils::memmem` search (case-insensitive). No regex engine is -/// involved at any point. struct PlainTextSink<'r> { state: SinkState, - finder: &'r memchr::memmem::Finder<'r>, + finder: &'r NeedleFinder<'r>, pattern_len: u32, - case_insensitive: bool, + multiline_segment_len: Option, } impl Sink for PlainTextSink<'_> { type Error = std::io::Error; - fn matched(&mut self, _searcher: &Searcher, mat: &SinkMatch<'_>) -> Result { - if self.state.max_matches != 0 && self.state.matches.len() >= self.state.max_matches { - return Ok(false); - } - - let line_bytes = mat.bytes(); - let (display_bytes, display_len, line_number, byte_offset) = - SinkState::prepare_line(line_bytes, mat); - - let line_content = String::from_utf8_lossy(display_bytes).into_owned(); - let mut match_byte_offsets: SmallVec<[(u32, u32); 4]> = SmallVec::new(); - let mut col = 0usize; - let mut first = true; - - if self.case_insensitive { - // The finder was built over the lowered pattern, so its needle is - // exactly the `needle_lower` expected by `memmem::find`. - let needle_lower = self.finder.needle(); - let mut start_pos = 0usize; - while let Some(pos) = memmem::find(&display_bytes[start_pos..], needle_lower) { - let abs_start = (start_pos + pos) as u32; - let abs_end = (abs_start + self.pattern_len).min(display_len); - if first { - col = abs_start as usize; - first = false; - } - match_byte_offsets.push((abs_start, abs_end)); - start_pos += pos + 1; - } - } else { - let mut start_pos = 0usize; - while let Some(pos) = self.finder.find(&display_bytes[start_pos..]) { - let abs_start = (start_pos + pos) as u32; - let abs_end = (abs_start + self.pattern_len).min(display_len); - if first { - col = abs_start as usize; - first = false; - } - match_byte_offsets.push((abs_start, abs_end)); - start_pos += pos + 1; - } - } - - let (context_before, context_after) = self.state.extract_context(mat); - self.state.push_match( - line_number, - col, - byte_offset, - line_content, - match_byte_offsets, - context_before, - context_after, - ); - Ok(true) - } - - fn finish(&mut self, _: &Searcher, _: &fff_grep::SinkFinish) -> Result<(), Self::Error> { - Ok(()) - } -} - -/// Sink for `Regex` mode. -/// -/// Uses the compiled regex to extract precise variable-length highlight spans -/// from each matched line. No `memmem` finder is involved. -struct RegexSink<'r> { - state: SinkState, - re: &'r regex::bytes::Regex, -} - -impl Sink for RegexSink<'_> { - type Error = std::io::Error; - fn matched( &mut self, - _searcher: &Searcher, + searcher: &Searcher, sink_match: &SinkMatch<'_>, ) -> Result { + debug_assert_newline_terminator(searcher); if self.state.max_matches != 0 && self.state.matches.len() >= self.state.max_matches { return Ok(false); } @@ -556,650 +139,47 @@ impl Sink for RegexSink<'_> { let mut col = 0usize; let mut first = true; - for m in self.re.find_iter(display_bytes) { - let abs_start = m.start() as u32; - let abs_end = (m.end() as u32).min(display_len); - if first { - col = abs_start as usize; - first = false; + if let Some(seg_len) = self.multiline_segment_len { + // Multiline needle: the match starts on this line, so the needle's + // first segment must be a suffix of the line. Highlight that suffix. + let seg = &self.finder.needle()[..seg_len]; + if !seg.is_empty() + && display_bytes.len() >= seg.len() + && self + .finder + .eq_fold(&display_bytes[display_bytes.len() - seg.len()..], seg) + { + col = display_bytes.len() - seg.len(); + match_byte_offsets.push((col as u32, display_len)); } - match_byte_offsets.push((abs_start, abs_end)); + } else { + let pattern_len = self.pattern_len; + self.finder.for_each_occurrence(display_bytes, |pos| { + let abs_start = pos as u32; + let abs_end = (abs_start + pattern_len).min(display_len); + if first { + col = pos; + first = false; + } + match_byte_offsets.push((abs_start, abs_end)); + }); } let (context_before, context_after) = self.state.extract_context(sink_match); self.state.push_match( - line_number, - col, - byte_offset, - line_content, - match_byte_offsets, - context_before, - context_after, - ); - Ok(true) - } - - fn finish(&mut self, _: &Searcher, _: &fff_grep::SinkFinish) -> Result<(), Self::Error> { - Ok(()) - } -} - -/// A `grep_matcher::Matcher` backed by Aho-Corasick for multi-pattern search. -/// -/// Finds the first occurrence of any pattern starting at the given offset. -/// Always reports `\n` as the line terminator for the fast candidate-line path. -struct AhoCorasickMatcher<'a> { - ac: &'a AhoCorasick, -} - -impl Matcher for AhoCorasickMatcher<'_> { - type Error = NoError; - - #[inline] - fn find_at(&self, haystack: &[u8], at: usize) -> std::result::Result, NoError> { - let hay = &haystack[at..]; - let found: Option = self.ac.find(hay); - Ok(found.map(|m| Match::new(at + m.start(), at + m.end()))) - } - - #[inline] - fn line_terminator(&self) -> Option { - Some(fff_grep::LineTerminator::byte(b'\n')) - } -} - -/// Sink for Aho-Corasick multi-pattern mode. -/// -/// Collects all pattern match positions on each matched line for highlighting. -struct AhoCorasickSink<'a> { - state: SinkState, - ac: &'a AhoCorasick, -} - -impl Sink for AhoCorasickSink<'_> { - type Error = std::io::Error; - - fn matched(&mut self, _searcher: &Searcher, mat: &SinkMatch<'_>) -> Result { - if self.state.max_matches != 0 && self.state.matches.len() >= self.state.max_matches { - return Ok(false); - } - - let line_bytes = mat.bytes(); - let (display_bytes, display_len, line_number, byte_offset) = - SinkState::prepare_line(line_bytes, mat); - - let line_content = String::from_utf8_lossy(display_bytes).into_owned(); - let mut match_byte_offsets: SmallVec<[(u32, u32); 4]> = SmallVec::new(); - let mut col = 0usize; - let mut first = true; - - for m in self.ac.find_iter(display_bytes as &[u8]) { - let abs_start = m.start() as u32; - let abs_end = (m.end() as u32).min(display_len); - if first { - col = abs_start as usize; - first = false; - } - match_byte_offsets.push((abs_start, abs_end)); - } - - let (context_before, context_after) = self.state.extract_context(mat); - self.state.push_match( - line_number, - col, - byte_offset, - line_content, - match_byte_offsets, - context_before, - context_after, - ); - Ok(true) - } - - fn finish(&mut self, _: &Searcher, _: &fff_grep::SinkFinish) -> Result<(), Self::Error> { - Ok(()) - } -} - -/// Multi-pattern OR search using Aho-Corasick. -/// -/// Builds a single automaton from all patterns and searches each file in one -/// pass. This is significantly faster than regex alternation for literal text -/// searches because Aho-Corasick uses SIMD-accelerated multi-needle matching. -/// -/// Returns the same `GrepResult` type as `grep_search`. -#[allow(clippy::too_many_arguments)] -pub(crate) fn multi_grep_search<'a>( - files: &'a [FileItem], - patterns: &[&str], - constraints: &[fff_query_parser::Constraint<'_>], - options: &GrepSearchOptions, - budget: &ContentCacheBudget, - bigram_index: Option<&BigramFilter>, - bigram_overlay: Option<&BigramOverlay>, - abort_signal: &AtomicBool, - base_path: &Path, - arena: crate::simd_path::ArenaPtr, - overflow_arena: crate::simd_path::ArenaPtr, -) -> GrepResult<'a> { - let total_files = files.live_count(); - - if patterns.is_empty() || patterns.iter().all(|p| p.is_empty()) { - return GrepResult::empty(total_files, total_files); - } - - // Bigram prefiltering: OR the candidate bitsets for each pattern. - // A file is a candidate if it matches ANY of the patterns' bigrams. - let bigram_candidates = if let Some(idx) = bigram_index - && idx.is_ready() - { - let mut combined: Option> = None; - for pattern in patterns { - if let Some(candidates) = idx.query(pattern.as_bytes()) { - combined = Some(match combined { - None => candidates, - Some(mut acc) => { - // OR: file is candidate if it matches any pattern - acc.iter_mut() - .zip(candidates.iter()) - .for_each(|(a, b)| *a |= *b); - acc - } - }); - } - } - - if let Some(ref mut candidates) = combined - && let Some(overlay) = bigram_overlay - { - for pattern in patterns { - let pattern_bigrams = extract_bigrams(pattern.as_bytes()); - for file_idx in overlay.query_modified(&pattern_bigrams) { - let word = file_idx / 64; - if word < candidates.len() { - candidates[word] |= 1u64 << (file_idx % 64); - } - } - } - } - - combined - } else { - None - }; - - let base_file_count = match bigram_overlay { - Some(bigram_overlay) => bigram_overlay.base_file_count(), - None => files.len(), - }; - - let (mut files_to_search, mut filtered_file_count) = prefilter_files( - files, - constraints, - bigram_candidates.as_deref(), - base_file_count, - options, - arena, - overflow_arena, - ); - - // If constraints yielded 0 files and we had FilePath constraints, - // retry without them (the path token was likely part of the search text). - if files_to_search.is_empty() - && let Some(stripped) = strip_file_path_constraint_if_present(constraints) - { - let (retry_files, retry_count) = prefilter_files( - files, - &stripped, - bigram_candidates.as_deref(), - base_file_count, - options, - arena, - overflow_arena, - ); - files_to_search = retry_files; - filtered_file_count = retry_count; - } - - if files_to_search.is_empty() { - return GrepResult::empty(total_files, filtered_file_count); - } - - // Smart case: case-insensitive when all patterns are lowercase - let case_insensitive = if options.smart_case { - !patterns.iter().any(|p| p.chars().any(|c| c.is_uppercase())) - } else { - false - }; - - let ac = aho_corasick::AhoCorasickBuilder::new() - .ascii_case_insensitive(case_insensitive) - .build(patterns) - .expect("Aho-Corasick build should not fail for literal patterns"); - - let searcher = { - let mut b = SearcherBuilder::new(); - b.line_number(true); - b - } - .build(); - - let ac_matcher = AhoCorasickMatcher { ac: &ac }; - perform_grep( - &files_to_search, - options, - &GrepContext { - total_files, - filtered_file_count, - budget, - base_path, - arena, - overflow_arena, - prefilter: None, // no memmem prefilter for multi-pattern search - prefilter_case_insensitive: false, - abort_signal, - }, - |file_bytes: &[u8], max_matches: usize| { - let state = SinkState { - file_index: 0, - matches: Vec::with_capacity(4), - max_matches, - before_context: options.before_context, - after_context: options.after_context, - classify_definitions: options.classify_definitions, - }; - - let mut sink = AhoCorasickSink { state, ac: &ac }; - - if let Err(e) = searcher.search_slice(&ac_matcher, file_bytes, &mut sink) { - tracing::error!(error = %e, "Grep (aho-corasick multi) search failed"); - } - - sink.state.matches - }, - ) -} - -// copied from the rust u8 private method -#[inline] -const fn is_utf8_char_boundary(b: u8) -> bool { - (b as i8) >= -0x40 -} - -fn build_regex(pattern: &str, smart_case: bool) -> Result { - if pattern.is_empty() { - return Err("empty pattern".to_string()); - } - - let regex_pattern = if pattern.contains("\\n") { - pattern.replace("\\n", "\n") - } else { - pattern.to_string() - }; - - let case_insensitive = if smart_case { - !pattern.chars().any(|c| c.is_uppercase()) - } else { - false - }; - - regex::bytes::RegexBuilder::new(®ex_pattern) - .case_insensitive(case_insensitive) - .multi_line(true) - .unicode(false) - .build() - .map_err(|e| e.to_string()) -} - -/// Convert character-position indices from neo_frizbee into byte-offset -/// pairs (start, end) suitable for `match_byte_offsets`. -/// -/// frizbee returns character positions (0-based index into the char -/// iterator). We need byte ranges because the UI renderer and Lua layer -/// use byte offsets for extmark highlights. -/// -/// Each matched character becomes its own (byte_start, byte_end) pair. -/// Adjacent characters are merged into a single contiguous range. -pub(super) fn char_indices_to_byte_offsets( - line: &str, - char_indices: &[usize], -) -> SmallVec<[(u32, u32); 4]> { - if char_indices.is_empty() { - return SmallVec::new(); - } - - // Build a map: char_index -> (byte_start, byte_end) for all chars. - // Iterating all chars is O(n) in the line length which is bounded by MAX_LINE_DISPLAY_LEN (512). - let char_byte_ranges: Vec<(usize, usize)> = line - .char_indices() - .map(|(byte_pos, ch)| (byte_pos, byte_pos + ch.len_utf8())) - .collect(); - - // Convert char indices to byte ranges, merging adjacent ranges - let mut result: SmallVec<[(u32, u32); 4]> = SmallVec::with_capacity(char_indices.len()); - - for &ci in char_indices { - if ci >= char_byte_ranges.len() { - continue; // out of bounds (shouldn't happen with valid data) - } - let (start, end) = char_byte_ranges[ci]; - // Merge with previous range if adjacent - if let Some(last) = result.last_mut() - && last.1 == start as u32 - { - last.1 = end as u32; - continue; - } - result.push((start as u32, end as u32)); - } - - result -} - -#[tracing::instrument( - skip_all, - level = Level::DEBUG, - fields(prefiltered_count = files_to_search.len()) -)] -fn perform_grep<'a, F>( - files_to_search: &[&'a FileItem], - options: &GrepSearchOptions, - ctx: &GrepContext<'_, '_>, - search_file: F, -) -> GrepResult<'a> -where - F: Fn(&[u8], usize) -> Vec + Sync, -{ - let time_budget = if options.time_budget_ms > 0 { - Some(std::time::Duration::from_millis(options.time_budget_ms)) - } else { - None - }; - - let search_start = std::time::Instant::now(); - let page_limit = options.page_limit; - let budget_exceeded = AtomicBool::new(false); - - let mut result_files: Vec<&'a FileItem> = Vec::new(); - let mut all_matches: Vec = Vec::new(); - let mut files_consumed: usize = 0; - let mut page_filled = false; - - // Each chunk is a rayon barrier. A flat small chunk over 500k files = ~7800 - // barriers; x2 growth makes it logarithmic. But a too-aggressive growth - // over-scans: when a page fills mid-chunk, the whole submitted chunk still - // runs. - // - // So only grow when the prefilter is weak (large candidate set); - // when bigram cut the set in half, keep fixed small chunks for cheap page-fill termination. - let base_chunk = rayon::current_num_threads() * 4; - let prefilter_strong = ctx.total_files > 0 && files_to_search.len() * 2 < ctx.total_files; - let max_chunk = if prefilter_strong { - base_chunk - } else { - (base_chunk * 256).max(8 * 1024) - }; - let growth = if prefilter_strong { 1 } else { 2 }; - let mut chunk_size = base_chunk; - let mut chunk_start = 0; - - while chunk_start < files_to_search.len() { - let chunk_end = (chunk_start + chunk_size).min(files_to_search.len()); - let chunk = &files_to_search[chunk_start..chunk_end]; - chunk_start = chunk_end; - chunk_size = (chunk_size * growth).min(max_chunk); - let chunk_offset = files_consumed; - - let chunk_results: Vec<(usize, &'a FileItem, Vec)> = chunk - .par_iter() - .enumerate() - .map_init( - // tested it out a few times, this is just fine for rayon worker in this specific - // case it doesn't reallocate this many times and it is actually faster than using - // scoped threads with a predefined local scratch buffers because of spawn cost - || (Vec::with_capacity(64 * 1024), MmapSlot::default()), - |(buf, mmap_slot), (local_idx, file)| { - // perform all the atomic machinery on every 8th - if local_idx % 8 == 0 { - let mut need_abort = ctx.abort_signal.load(Ordering::Relaxed); - if !need_abort - && let Some(budget) = time_budget - && all_matches.len() > 1 - && search_start.elapsed() > budget - { - need_abort = true; - } - - if need_abort { - budget_exceeded.store(true, Ordering::Relaxed); - return None; - } - } - - let content = file.get_content_for_search( - buf, - mmap_slot, - ctx.arena_for_file(file), - ctx.base_path, - ctx.budget, - )?; - - // Fast whole-file memmem check before entering the - // grep-searcher machinery. Skips Vec alloc, Searcher - // setup, and line-splitting for files that can't match. - if let Some(pf) = ctx.prefilter { - let found = if ctx.prefilter_case_insensitive { - memmem::find(content, pf.needle()).is_some() - } else { - pf.find(content).is_some() - }; - if !found { - return None; - } - } - - let file_matches = search_file(content, options.max_matches_per_file); - - if file_matches.is_empty() { - return None; - } - - Some((chunk_offset + local_idx, *file, file_matches)) - }, - ) - .flatten() - .collect(); - - // Every file in the chunk was visited by rayon (matched or not). - files_consumed = chunk_offset + chunk.len(); - - // Flatten this chunk's results into the accumulator. - for (batch_idx, file, file_matches) in chunk_results { - let file_result_idx = result_files.len(); - result_files.push(file); - - for mut m in file_matches { - m.file_index = file_result_idx; - if options.trim_whitespace { - m.trim_leading_whitespace(); - } - all_matches.push(m); - } - - if all_matches.len() >= page_limit { - // Tighten files_consumed to the file that tipped us over so - // the next page resumes right after it. - files_consumed = batch_idx + 1; - page_filled = true; - break; - } - } - - if page_filled || budget_exceeded.load(Ordering::Relaxed) { - break; - } - } - - // If no file had any match, we searched the entire slice. - if result_files.is_empty() { - files_consumed = files_to_search.len(); - } - - let has_more = budget_exceeded.load(Ordering::Relaxed) - || (page_filled && files_consumed < files_to_search.len()); - - let next_file_offset = if has_more { - options.file_offset + files_consumed - } else { - 0 - }; - - GrepResult { - matches: all_matches, - files_with_matches: result_files.len(), - files: result_files, - total_files_searched: files_consumed, - total_files: ctx.total_files, - filtered_file_count: ctx.filtered_file_count, - next_file_offset, - regex_fallback_error: None, - } -} - -/// Single pass prefilter that doesn't involve file reading -/// allocates only amount of memory required for storing references of the FileItems have to be -/// opened for grepping unaviodably, in the worst case allocates N * memory if no prefilter needed -fn prefilter_files<'a>( - files: &'a [FileItem], - constraints: &[fff_query_parser::Constraint<'_>], - bigram_candidates: Option<&[u64]>, - base_count: usize, - options: &GrepSearchOptions, - arena: crate::simd_path::ArenaPtr, - overflow_arena: crate::simd_path::ArenaPtr, -) -> (Vec<&'a FileItem>, usize) { - let max_file_size = options.max_file_size; - let plan = if constraints.is_empty() { - None - } else { - Some(ConstraintPlan::build( - constraints, - files, - arena, - overflow_arena, - )) - }; - - let mut scratch = ConstraintsBuffers::new(); - - #[inline(always)] - fn basic_prefilter(file: &FileItem, max: u64) -> bool { - !file.is_deleted() && !file.is_binary() && file.size > 0 && file.size <= max - } - - // squeeze as much prefilters into a single loop as possible - let mut prefiltered: Vec<&FileItem> = match bigram_candidates { - Some(candidates) => { - let boundary = base_count.min(files.len()); - let (indexed, tail) = files.split_at(boundary); - - let cap = BigramFilter::count_candidates(candidates) + tail.len(); - let mut out: Vec<&FileItem> = Vec::with_capacity(cap); - - let full_words = boundary / 64; - let last_word_bits = boundary % 64; - - // we need this because we already had a regression of the wrong bit - // has been set for the very last word based on the overlay, it's pretty cheap - macro_rules! evaluate_bigram_match_word { - ($word:expr, $base:expr) => {{ - let mut bits: u64 = $word; - while bits != 0 { - let bit = bits.trailing_zeros() as usize; - let file_idx = $base + bit; - bits &= bits - 1; - - let f = unsafe { indexed.get_unchecked(file_idx) }; - if !basic_prefilter(f, max_file_size) { - continue; - } - if let Some(plan) = plan.as_ref() - && !plan.matches(f, file_idx, arena, overflow_arena, &mut scratch) - { - continue; - } - out.push(f); - } - }}; - } - - // Full words: every set bit guaranteed `< boundary`. - for (word_idx, &word) in candidates.iter().take(full_words).enumerate() { - if word != 0 { - evaluate_bigram_match_word!(word, word_idx * 64); - } - } - - // Last partial word: mask bits past `boundary` once at word load. - if last_word_bits != 0 { - // this will get only (mod 64) bits from the last word guaratee that it's 0 padded - let last_mask: u64 = (1u64 << last_word_bits) - 1; - let word = candidates[full_words] & last_mask; - if word != 0 { - evaluate_bigram_match_word!(word, full_words * 64); - } - } - - // Sequential processing for non-bigrammable files: they are always in the end - for (offset, f) in tail.iter().enumerate() { - if !basic_prefilter(f, max_file_size) { - continue; - } - if let Some(ref p) = plan - && !p.matches(f, boundary + offset, arena, overflow_arena, &mut scratch) - { - continue; - } - out.push(f); - } - - out - } - // this will be executed if there is no bigram, in the worst case it will allocate - // whole array of files but probability in the real repo of NO preflter working is so - // low that we just ignore that, usually there would be at least a few files excluded - None => { - let mut out: Vec<&FileItem> = Vec::new(); - for (idx, f) in files.iter().enumerate() { - if !basic_prefilter(f, max_file_size) { - continue; - } - if let Some(ref p) = plan - && !p.matches(f, idx, arena, overflow_arena, &mut scratch) - { - continue; - } - out.push(f); - } - out - } - }; - - let total_count = prefiltered.len(); - - sort_with_buffer(&mut prefiltered, |a, b| { - b.total_frecency_score() - .cmp(&a.total_frecency_score()) - .then(b.modified.cmp(&a.modified)) - }); + line_number, + col, + byte_offset, + line_content, + match_byte_offsets, + context_before, + context_after, + ); + Ok(true) + } - if options.file_offset > 0 && options.file_offset < total_count { - let paginated = prefiltered.split_off(options.file_offset); - (paginated, total_count) - } else if options.file_offset >= total_count { - (Vec::new(), total_count) - } else { - (prefiltered, total_count) + fn finish(&mut self, _: &Searcher, _: &fff_grep::SinkFinish) -> Result<(), Self::Error> { + Ok(()) } } @@ -1222,32 +202,9 @@ pub(crate) fn grep_search<'a>( overflow_arena: crate::simd_path::ArenaPtr, ) -> GrepResult<'a> { let total_files = files.live_count(); - - // Extract the grep text and file constraints from the parsed query. - // For grep, the search pattern is the original query with constraint tokens - // removed. All non-constraint text tokens are collected and joined with - // spaces to form the grep pattern: - // "name = *.rs someth" -> grep "name = someth" with constraint Extension("rs") let constraints_from_query = &query.constraints[..]; - let grep_text = if !matches!(query.fuzzy_query, fff_query_parser::FuzzyQuery::Empty) { - query.grep_text() - } else { - // if constraint-only or empty query we use raw_query for backslash-escape handling - let t = query.raw_query.trim(); - if t.starts_with('\\') && t.len() > 1 { - let suffix = &t[1..]; - let parser = QueryParser::new(GrepConfig); - if !parser.parse(suffix).constraints.is_empty() { - suffix.to_string() - } else { - t.to_string() - } - } else { - t.to_string() - } - }; - + let grep_text = extract_grep_text(query); if grep_text.is_empty() { return GrepResult::empty(total_files, total_files); } @@ -1258,46 +215,15 @@ pub(crate) fn grep_search<'a>( false }; + let base_count = bigram_boundary(bigram_overlay, files.len()); + let mut regex_fallback_error: Option = None; let regex = match options.mode { GrepMode::PlainText => None, GrepMode::Fuzzy => { - // Bigram prefilter: pick 5 evenly-spaced probe bigrams, require - // (5 - max_typos) of them to appear. Widely-spaced probes are - // far more selective than sliding windows of adjacent bigrams. - let bigram_candidates = if let Some(idx) = bigram_index - && idx.is_ready() - { - let bq = fuzzy_to_bigram_query(&grep_text, 7); - if !bq.is_any() - && let Some(mut candidates) = bq.evaluate(idx) - { - if let Some(overlay) = bigram_overlay { - for (r, t) in candidates.iter_mut().zip(overlay.tombstones().iter()) { - *r &= !t; - } - // Fuzzy: conservatively add all modified files - for file_idx in overlay.modified_indices() { - let word = file_idx / 64; - if word < candidates.len() { - candidates[word] |= 1u64 << (file_idx % 64); - } - } - } - Some(candidates) - } else { - None - } - } else { - None - }; - - let base_count = match bigram_overlay { - Some(bigram_overlay) => bigram_overlay.base_file_count(), - None => files.len(), - }; + let bigram_candidates = fuzzy_candidates(bigram_index, bigram_overlay, &grep_text); - let (mut files_to_search, mut filtered_file_count) = prefilter_files( + let (files_to_search, filtered_file_count) = prefilter_with_filepath_retry( files, constraints_from_query, bigram_candidates.as_deref(), @@ -1307,24 +233,6 @@ pub(crate) fn grep_search<'a>( overflow_arena, ); - if files_to_search.is_empty() - && let Some(stripped) = - strip_file_path_constraint_if_present(constraints_from_query) - { - let (retry_files, retry_count) = prefilter_files( - files, - &stripped, - bigram_candidates.as_deref(), - base_count, - options, - arena, - overflow_arena, - ); - - files_to_search = retry_files; - filtered_file_count = retry_count; - } - if files_to_search.is_empty() { return GrepResult::empty(total_files, filtered_file_count); } @@ -1352,12 +260,18 @@ pub(crate) fn grep_search<'a>( .ok(), }; - let is_multiline = has_unescaped_newline_escape(&grep_text); + let (multiline_segment_len, effective_pattern) = match replace_newline_escapes(&grep_text) { + Some((replaced, first_newline_pos)) => (Some(first_newline_pos), replaced), + None => (None, grep_text), + }; + + let is_multiline = multiline_segment_len.is_some(); - let effective_pattern = if is_multiline { - replace_unescaped_newline_escapes(&grep_text) + // when there is multiple line requested automatically expand the context to include all the lines + let after_context = if is_multiline && regex.is_none() && options.after_context == 0 { + effective_pattern.bytes().filter(|&b| b == b'\n').count() } else { - grep_text.to_string() + options.after_context }; let finder_pattern: Vec = if case_insensitive { @@ -1365,100 +279,34 @@ pub(crate) fn grep_search<'a>( } else { effective_pattern.as_bytes().to_vec() }; - let finder = memchr::memmem::Finder::new(&finder_pattern); + let finder = NeedleFinder::new(&finder_pattern, case_insensitive); let pattern_len = finder_pattern.len() as u32; - // Bigram prefiltering: query the inverted index + merge overlay. - // For PlainText mode: extract bigrams directly from the literal pattern. - // For Regex mode: decompose the regex HIR into an AND/OR bigram query tree - // and evaluate it against the inverted index (supports alternation, optional - // groups, character classes, and sparse-1 bigrams across single-byte wildcards). - let bigram_candidates = if let Some(idx) = bigram_index - && idx.is_ready() - { - let raw_candidates = if regex.is_none() { - // PlainText or regex-fallback-to-plain: literal bigram query - idx.query(effective_pattern.as_bytes()) - } else { - // Regex mode: decompose pattern into bigram query tree - let bq = regex_to_bigram_query(&effective_pattern); - if !bq.is_any() { bq.evaluate(idx) } else { None } - }; - - if let Some(mut candidates) = raw_candidates { - if let Some(overlay) = bigram_overlay { - // Clear tombstoned (deleted) files from candidates - for (r, t) in candidates.iter_mut().zip(overlay.tombstones().iter()) { - *r &= !t; - } - - if regex.is_none() { - let pattern_bigrams = extract_bigrams(effective_pattern.as_bytes()); - for file_idx in overlay.query_modified(&pattern_bigrams) { - let word = file_idx / 64; - if word < candidates.len() { - candidates[word] |= 1u64 << (file_idx % 64); - } - } - } else { - for file_idx in overlay.modified_indices() { - let word = file_idx / 64; - if word < candidates.len() { - candidates[word] |= 1u64 << (file_idx % 64); - } - } - } - } - Some(candidates) - } else { - None - } + // PlainText (or regex-fallback-to-plain): literal bigram query. + // Regex: decompose the pattern HIR into an AND/OR bigram query tree. + let bigram_candidates = if regex.is_none() { + literal_candidates(bigram_index, bigram_overlay, &[&effective_pattern]) } else { - None + regex_candidates(bigram_index, bigram_overlay, &effective_pattern) }; - // Bigram bitset only covers `files[..bigram_boundary]`, new files aka overflow - // (max 1024 always scanned) - let bigram_boundary = bigram_overlay - .map(|o| o.base_file_count()) - .unwrap_or(files.len()); - - let (mut files_to_search, mut filtered_file_count) = prefilter_files( + let (files_to_search, filtered_file_count) = prefilter_with_filepath_retry( files, constraints_from_query, bigram_candidates.as_deref(), - bigram_boundary, + base_count, options, arena, overflow_arena, ); - if files_to_search.is_empty() - && let Some(stripped) = strip_file_path_constraint_if_present(constraints_from_query) - { - let (retry_files, retry_count) = prefilter_files( - files, - &stripped, - bigram_candidates.as_deref(), - bigram_boundary, - options, - arena, - overflow_arena, - ); - files_to_search = retry_files; - filtered_file_count = retry_count; - } - if files_to_search.is_empty() { return GrepResult::empty(total_files, filtered_file_count); } // `PlainTextMatcher` is used by the grep-searcher engine for line detection. // `PlainTextSink` / `RegexSink` handle highlight extraction independently via ripgrep create - let plain_matcher = PlainTextMatcher { - needle: &finder_pattern, - case_insensitive, - }; + let plain_matcher = PlainTextMatcher { finder: &finder }; let searcher = { let mut b = SearcherBuilder::new(); @@ -1479,16 +327,17 @@ pub(crate) fn grep_search<'a>( arena, overflow_arena, prefilter: should_prefilter.then_some(&finder), - prefilter_case_insensitive: case_insensitive, abort_signal, }, + // The single sink-selection point: every mode's matcher/sink pairing + // is decided here based on the compiled pattern. |file_bytes: &[u8], max_matches: usize| { let state = SinkState { file_index: 0, matches: Vec::with_capacity(4), max_matches, before_context: options.before_context, - after_context: options.after_context, + after_context, classify_definitions: options.classify_definitions, }; @@ -1509,7 +358,7 @@ pub(crate) fn grep_search<'a>( state, finder: &finder, pattern_len, - case_insensitive, + multiline_segment_len, }; if let Err(e) = searcher.search_slice(&plain_matcher, file_bytes, &mut sink) { tracing::error!(error = %e, "Grep (plain text) search failed"); @@ -1523,26 +372,255 @@ pub(crate) fn grep_search<'a>( result } +/// Replace unescaped `\n` escapes with real newlines in a single pass. +/// +/// Returns `Some((replaced, first_newline_pos))` when the pattern contained at +/// least one real `\n` escape (the user wants multiline search), where +/// `first_newline_pos` is the byte offset of the first inserted newline in the +/// replaced string. Returns `None` when nothing had to be replaced: `\\n` is +/// preserved as-is (escaped backslash + literal `n`, e.g. `\\nvim-data`). +pub(super) fn replace_newline_escapes(text: &str) -> Option<(String, usize)> { + let bytes = text.as_bytes(); + let mut result = Vec::with_capacity(bytes.len()); + let mut first_newline_pos: Option = None; + let mut i = 0; + while i < bytes.len() { + if bytes[i] == b'\\' && i + 1 < bytes.len() { + if bytes[i + 1] == b'n' { + // Odd number of consecutive backslashes before 'n' -> real \n escape + let mut backslash_count = 1; + while backslash_count <= i && bytes[i - backslash_count] == b'\\' { + backslash_count += 1; + } + if backslash_count % 2 == 1 { + first_newline_pos.get_or_insert(result.len()); + result.push(b'\n'); + i += 2; + continue; + } + } + result.push(bytes[i]); + i += 1; + } else { + result.push(bytes[i]); + i += 1; + } + } + + let first_newline_pos = first_newline_pos?; + let replaced = String::from_utf8(result).unwrap_or_else(|_| text.to_string()); + Some((replaced, first_newline_pos)) +} + pub fn parse_grep_query(query: &str) -> FFFQuery<'_> { let parser = QueryParser::new(GrepConfig); parser.parse(query) } -fn strip_file_path_constraint_if_present<'a>( - constraints: &[Constraint<'a>], -) -> Option> { - if !constraints - .iter() - .any(|c| matches!(c, Constraint::FilePath(_))) - { - return None; +/// Extract the grep pattern text from the parsed query: all non-constraint +/// tokens joined with spaces, e.g. `"name = *.rs someth"` -> `"name = someth"` +/// with constraint `Extension("rs")`. +fn extract_grep_text(query: &FFFQuery<'_>) -> String { + if !matches!(query.fuzzy_query, fff_query_parser::FuzzyQuery::Empty) { + return query.grep_text(); + } + + // if constraint-only or empty query we use raw_query for backslash-escape handling + let t = query.raw_query.trim(); + if t.starts_with('\\') && t.len() > 1 { + let suffix = &t[1..]; + let parser = QueryParser::new(GrepConfig); + if !parser.parse(suffix).constraints.is_empty() { + return suffix.to_string(); + } + } + t.to_string() +} + +#[derive(Clone, Copy)] +pub(super) struct GrepContext<'a, 'b> { + pub(super) total_files: usize, + pub(super) filtered_file_count: usize, + pub(super) budget: &'a ContentCacheBudget, + pub(super) base_path: &'a Path, + pub(super) arena: crate::simd_path::ArenaPtr, + pub(super) overflow_arena: crate::simd_path::ArenaPtr, + pub(super) prefilter: Option<&'a NeedleFinder<'b>>, + pub(super) abort_signal: &'a AtomicBool, +} + +impl GrepContext<'_, '_> { + #[inline] + fn arena_for_file(&self, file: &FileItem) -> crate::simd_path::ArenaPtr { + if file.is_overflow() { + self.overflow_arena + } else { + self.arena + } + } +} + +#[tracing::instrument( + skip_all, + level = Level::DEBUG, + fields(prefiltered_count = files_to_search.len()) +)] +pub(super) fn perform_grep<'a, F>( + files_to_search: &[&'a FileItem], + options: &GrepSearchOptions, + ctx: &GrepContext<'_, '_>, + search_file: F, +) -> GrepResult<'a> +where + F: Fn(&[u8], usize) -> Vec + Sync, +{ + let time_budget = if options.time_budget_ms > 0 { + Some(std::time::Duration::from_millis(options.time_budget_ms)) + } else { + None + }; + + let search_start = std::time::Instant::now(); + let page_limit = options.page_limit; + let budget_exceeded = AtomicBool::new(false); + + let mut result_files: Vec<&'a FileItem> = Vec::new(); + let mut all_matches: Vec = Vec::new(); + let mut files_consumed: usize = 0; + let mut page_filled = false; + + // Each chunk is a rayon barrier. A flat small chunk over 500k files = ~7800 + // barriers; x2 growth makes it logarithmic. But a too-aggressive growth + // over-scans: when a page fills mid-chunk, the whole submitted chunk still + // runs. + // + // So only grow when the prefilter is weak (large candidate set); + // when bigram cut the set in half, keep fixed small chunks for cheap page-fill termination. + let base_chunk = rayon::current_num_threads() * 4; + let prefilter_strong = ctx.total_files > 0 && files_to_search.len() * 2 < ctx.total_files; + let max_chunk = if prefilter_strong { + base_chunk + } else { + (base_chunk * 256).max(8 * 1024) + }; + let growth = if prefilter_strong { 1 } else { 2 }; + let mut chunk_size = base_chunk; + let mut chunk_start = 0; + + while chunk_start < files_to_search.len() { + let chunk_end = (chunk_start + chunk_size).min(files_to_search.len()); + let chunk = &files_to_search[chunk_start..chunk_end]; + chunk_start = chunk_end; + chunk_size = (chunk_size * growth).min(max_chunk); + let chunk_offset = files_consumed; + + let chunk_results: Vec<(usize, &'a FileItem, Vec)> = chunk + .par_iter() + .enumerate() + .map_init( + // tested it out a few times, this is just fine for rayon worker in this specific + // case it doesn't reallocate this many times and it is actually faster than using + // scoped threads with a predefined local scratch buffers because of spawn cost + || (Vec::with_capacity(64 * 1024), MmapSlot::default()), + |(buf, mmap_slot), (local_idx, file)| { + // perform all the atomic machinery on every 8th + if local_idx % 8 == 0 { + let mut need_abort = ctx.abort_signal.load(Ordering::Relaxed); + if !need_abort + && let Some(budget) = time_budget + && all_matches.len() > 1 + && search_start.elapsed() > budget + { + need_abort = true; + } + + if need_abort { + budget_exceeded.store(true, Ordering::Relaxed); + return None; + } + } + + let content = file.get_content_for_search( + buf, + mmap_slot, + ctx.arena_for_file(file), + ctx.base_path, + ctx.budget, + )?; + + // Fast whole-file memmem check before entering the + // grep-searcher machinery. Skips Vec alloc, Searcher + // setup, and line-splitting for files that can't match. + if let Some(pf) = ctx.prefilter + && pf.find(content).is_none() + { + return None; + } + + let file_matches = search_file(content, options.max_matches_per_file); + + if file_matches.is_empty() { + return None; + } + + Some((chunk_offset + local_idx, *file, file_matches)) + }, + ) + .flatten() + .collect(); + + // Every file in the chunk was visited by rayon (matched or not). + files_consumed = chunk_offset + chunk.len(); + + // Flatten this chunk's results into the accumulator. + for (batch_idx, file, file_matches) in chunk_results { + let file_result_idx = result_files.len(); + result_files.push(file); + + for mut m in file_matches { + m.file_index = file_result_idx; + if options.trim_whitespace { + m.trim_leading_whitespace(); + } + all_matches.push(m); + } + + if all_matches.len() >= page_limit { + // Tighten files_consumed to the file that tipped us over so + // the next page resumes right after it. + files_consumed = batch_idx + 1; + page_filled = true; + break; + } + } + + if page_filled || budget_exceeded.load(Ordering::Relaxed) { + break; + } + } + + // If no file had any match, we searched the entire slice. + if result_files.is_empty() { + files_consumed = files_to_search.len(); } - let filtered: fff_query_parser::ConstraintVec<'a> = constraints - .iter() - .filter(|c| !matches!(c, Constraint::FilePath(_))) - .cloned() - .collect(); + let has_more = budget_exceeded.load(Ordering::Relaxed) + || (page_filled && files_consumed < files_to_search.len()); + + let next_file_offset = if has_more { + options.file_offset + files_consumed + } else { + 0 + }; - Some(filtered) + GrepResult { + matches: all_matches, + files_with_matches: result_files.len(), + files: result_files, + total_files_searched: files_consumed, + total_files: ctx.total_files, + filtered_file_count: ctx.filtered_file_count, + next_file_offset, + regex_fallback_error: None, + } } diff --git a/crates/fff-core/src/grep/grep_tests.rs b/crates/fff-core/src/grep/grep_tests.rs index d342902e3..72222a3b0 100644 --- a/crates/fff-core/src/grep/grep_tests.rs +++ b/crates/fff-core/src/grep/grep_tests.rs @@ -1,38 +1,41 @@ -use super::grep::*; +use super::grep::replace_newline_escapes; +use super::*; -use crate::bigram_filter::BigramIndexBuilder; use crate::file_picker::{FilePicker, FilePickerOptions}; +use crate::index::BigramIndexBuilder; use std::io::Write; use std::sync::atomic::AtomicBool; #[test] -fn test_unescaped_newline_detection() { - // Single \n → multiline - assert!(has_unescaped_newline_escape("foo\\nbar")); +fn test_replace_newline_escapes() { + // Single \n → multiline: replaced with a real newline at byte 3 + assert_eq!( + replace_newline_escapes("foo\\nbar"), + Some(("foo\nbar".to_string(), 3)) + ); // \\n → escaped backslash + literal n, NOT multiline // (this is what the user types when grepping Rust source with `\\nvim`) - assert!(!has_unescaped_newline_escape("foo\\\\nvim-data")); + assert_eq!(replace_newline_escapes("foo\\\\nvim-data"), None); // Real-world: source file has literal \\AppData\\Local\\nvim-data // (double backslash in the file, so user types double backslash) - assert!(!has_unescaped_newline_escape( - r#"format!("{}\\AppData\\Local\\nvim-data","# - )); + assert_eq!( + replace_newline_escapes(r#"format!("{}\\AppData\\Local\\nvim-data","#), + None + ); // No \n at all - assert!(!has_unescaped_newline_escape("hello world")); + assert_eq!(replace_newline_escapes("hello world"), None); // \\\\n → even number of backslashes before n → NOT multiline - assert!(!has_unescaped_newline_escape("foo\\\\\\\\nbar")); - // \\\n → 3 backslashes: first two pair up, third + n = \n → multiline - assert!(has_unescaped_newline_escape("foo\\\\\\nbar")); -} - -#[test] -fn test_replace_unescaped_newline() { - // \n → real newline - assert_eq!(replace_unescaped_newline_escapes("foo\\nbar"), "foo\nbar"); - // \\n → preserved as-is + assert_eq!(replace_newline_escapes("foo\\\\\\\\nbar"), None); + // \\\n → 3 backslashes: first two pair up, third + n = \n → multiline, + // newline lands after "foo" + 2 kept backslashes = byte 5 + assert_eq!( + replace_newline_escapes("foo\\\\\\nbar"), + Some(("foo\\\\\nbar".to_string(), 5)) + ); + // Position is for the FIRST newline when there are several assert_eq!( - replace_unescaped_newline_escapes("foo\\\\nvim"), - "foo\\\\nvim" + replace_newline_escapes("a\\nb\\nc"), + Some(("a\nb\nc".to_string(), 1)) ); } @@ -236,6 +239,94 @@ fn test_multi_grep_search() { ); } +/// E2E: multiline grep (`\n` in query) and escaped-backslash literals (`\\n`) +/// through the full picker.grep pipeline, in both PlainText and Regex modes. +#[test] +fn test_grep_multiline_and_escaped_newline_e2e() { + let dir = tempfile::tempdir().unwrap(); + let base = crate::path_utils::canonicalize(dir.path()).unwrap(); + + // Content spanning two lines: "hello unicorn\nrainbow world" + { + let mut f = std::fs::File::create(base.join("multi.txt")).unwrap(); + writeln!(f, "hello unicorn").unwrap(); + writeln!(f, "rainbow world").unwrap(); + } + // Content with a literal double backslash before "nvim": `C:\\Users\\nvim-data` + { + let mut f = std::fs::File::create(base.join("winpath.rs")).unwrap(); + writeln!(f, "let p = \"C:\\\\Users\\\\nvim-data\";").unwrap(); + } + { + let mut f = std::fs::File::create(base.join("noise.txt")).unwrap(); + writeln!(f, "nothing interesting here").unwrap(); + } + + let mut picker = FilePicker::new(FilePickerOptions { + base_path: base.to_str().unwrap().into(), + watch: false, + ..Default::default() + }) + .unwrap(); + picker.collect_files().unwrap(); + + let options = crate::GrepSearchOptions { + page_limit: 100, + max_matches_per_file: 0, + ..Default::default() + }; + + // 1. PlainText + `\n`: needle becomes a real newline, match spans two lines + let query = super::parse_grep_query("unicorn\\nrainbow"); + let result = picker.grep(&query, &options); + assert_eq!( + result.files.len(), + 1, + "multiline plaintext should match multi.txt" + ); + assert_eq!(result.files[0].relative_path(&picker), "multi.txt"); + let m = &result.matches[0]; + assert_eq!(m.line_content, "hello unicorn"); + // Auto after-context: the rest of the matched span is returned + assert_eq!(m.context_after, vec!["rainbow world"]); + // First needle segment highlighted as the line suffix + assert_eq!(m.match_byte_offsets.as_slice(), &[(6, 13)]); + assert_eq!(m.col, 6); + + // 2. PlainText + `\\n`: literal backslash + n, must NOT be mangled to newline + let query = super::parse_grep_query("\\\\nvim-data"); + let result = picker.grep(&query, &options); + assert_eq!( + result.files.len(), + 1, + "escaped backslash should match winpath.rs literally" + ); + assert_eq!(result.files[0].relative_path(&picker), "winpath.rs"); + assert!(result.matches[0].line_content.contains("\\\\nvim-data")); + assert!(result.matches[0].context_after.is_empty()); + + // 3. Regex + `\n`: goes through the MultiLine searcher strategy + let regex_options = super::GrepSearchOptions { + mode: super::GrepMode::Regex, + ..options.clone() + }; + let query = super::parse_grep_query("unicorn\\nrainbow"); + let result = picker.grep(&query, ®ex_options); + assert!(result.regex_fallback_error.is_none()); + assert_eq!( + result.files.len(), + 1, + "multiline regex should match multi.txt" + ); + assert_eq!(result.files[0].relative_path(&picker), "multi.txt"); + let m = &result.matches[0]; + // Blob is normalized: single-line content + remaining lines as context + assert_eq!(m.line_content, "hello unicorn"); + assert_eq!(m.context_after, vec!["rainbow world"]); + // Highlight clamped to the visible first line + assert_eq!(m.match_byte_offsets.as_slice(), &[(6, 13)]); +} + /// Regression test for issue #407: Live grep returns duplicate results /// when the bigram candidate bitset has trailing bits set beyond /// `base_file_count`. The bitset is rounded up to a multiple of 64 bits diff --git a/crates/fff-core/src/grep/mod.rs b/crates/fff-core/src/grep/mod.rs index eefe52b09..a686c0426 100644 --- a/crates/fff-core/src/grep/mod.rs +++ b/crates/fff-core/src/grep/mod.rs @@ -1,16 +1,27 @@ -mod fuzzy_grep; - -mod utils; -pub use utils::*; // contains some of the generally available functions and types +//! Live grep. `grep.rs` implements the main plain-text path, the parallel +//! scan engine, and the `grep_search` entry point that picks the matcher/sink +//! for every mode in one place; `regex`, `multi_pattern`, and `fuzzy_grep` +//! hold the mode-specific machinery on top of the shared `prefilter`/`sink`. #[allow(clippy::module_inception)] mod grep; pub use grep::*; +mod fuzzy_grep; +mod multi_pattern; +mod prefilter; +mod regex; +mod sink; +mod types; + #[cfg(feature = "definitions")] mod classify; #[cfg(feature = "definitions")] pub use classify::*; +pub(crate) use multi_pattern::multi_grep_search; +pub use regex::has_regex_metacharacters; +pub use types::*; + #[cfg(test)] mod grep_tests; diff --git a/crates/fff-core/src/grep/multi_pattern.rs b/crates/fff-core/src/grep/multi_pattern.rs new file mode 100644 index 000000000..b238b495f --- /dev/null +++ b/crates/fff-core/src/grep/multi_pattern.rs @@ -0,0 +1,190 @@ +use super::grep::{GrepContext, perform_grep}; +use super::prefilter::prefilter_with_filepath_retry; +use super::sink::{SinkState, debug_assert_newline_terminator}; +use super::types::{GrepResult, GrepSearchOptions}; +use crate::index::{BigramFilter, BigramOverlay, bigram_boundary, literal_candidates}; +use crate::types::{ContentCacheBudget, FileItem, FileSliceExt}; +use aho_corasick::AhoCorasick; +use fff_grep::{ + Searcher, SearcherBuilder, Sink, SinkMatch, + matcher::{Match, Matcher, NoError}, +}; +use smallvec::SmallVec; +use std::path::Path; +use std::sync::atomic::AtomicBool; + +/// A `grep_matcher::Matcher` backed by Aho-Corasick for multi-pattern search. +/// +/// Finds the first occurrence of any pattern starting at the given offset. +/// Always reports `\n` as the line terminator for the fast candidate-line path. +struct AhoCorasickMatcher<'a> { + ac: &'a AhoCorasick, +} + +impl Matcher for AhoCorasickMatcher<'_> { + type Error = NoError; + + #[inline] + fn find_at(&self, haystack: &[u8], at: usize) -> std::result::Result, NoError> { + let hay = &haystack[at..]; + let found: Option = self.ac.find(hay); + Ok(found.map(|m| Match::new(at + m.start(), at + m.end()))) + } + + #[inline] + fn line_terminator(&self) -> Option { + Some(fff_grep::LineTerminator::byte(b'\n')) + } +} + +/// Sink for Aho-Corasick multi-pattern mode. +/// +/// Collects all pattern match positions on each matched line for highlighting. +struct AhoCorasickSink<'a> { + state: SinkState, + ac: &'a AhoCorasick, +} + +impl Sink for AhoCorasickSink<'_> { + type Error = std::io::Error; + + fn matched(&mut self, searcher: &Searcher, mat: &SinkMatch<'_>) -> Result { + debug_assert_newline_terminator(searcher); + if self.state.max_matches != 0 && self.state.matches.len() >= self.state.max_matches { + return Ok(false); + } + + let line_bytes = mat.bytes(); + let (display_bytes, display_len, line_number, byte_offset) = + SinkState::prepare_line(line_bytes, mat); + + let line_content = String::from_utf8_lossy(display_bytes).into_owned(); + let mut match_byte_offsets: SmallVec<[(u32, u32); 4]> = SmallVec::new(); + let mut col = 0usize; + let mut first = true; + + for m in self.ac.find_iter(display_bytes as &[u8]) { + let abs_start = m.start() as u32; + let abs_end = (m.end() as u32).min(display_len); + if first { + col = abs_start as usize; + first = false; + } + match_byte_offsets.push((abs_start, abs_end)); + } + + let (context_before, context_after) = self.state.extract_context(mat); + self.state.push_match( + line_number, + col, + byte_offset, + line_content, + match_byte_offsets, + context_before, + context_after, + ); + Ok(true) + } + + fn finish(&mut self, _: &Searcher, _: &fff_grep::SinkFinish) -> Result<(), Self::Error> { + Ok(()) + } +} + +/// Multi-pattern OR search using Aho-Corasick. +/// +/// Builds a single automaton from all patterns and searches each file in one +/// pass. This is significantly faster than regex alternation for literal text +/// searches because Aho-Corasick uses SIMD-accelerated multi-needle matching. +/// +/// Returns the same `GrepResult` type as `grep_search`. +#[allow(clippy::too_many_arguments)] +pub(crate) fn multi_grep_search<'a>( + files: &'a [FileItem], + patterns: &[&str], + constraints: &[fff_query_parser::Constraint<'_>], + options: &GrepSearchOptions, + budget: &ContentCacheBudget, + bigram_index: Option<&BigramFilter>, + bigram_overlay: Option<&BigramOverlay>, + abort_signal: &AtomicBool, + base_path: &Path, + arena: crate::simd_path::ArenaPtr, + overflow_arena: crate::simd_path::ArenaPtr, +) -> GrepResult<'a> { + let total_files = files.live_count(); + + if patterns.is_empty() || patterns.iter().all(|p| p.is_empty()) { + return GrepResult::empty(total_files, total_files); + } + + let bigram_candidates = literal_candidates(bigram_index, bigram_overlay, patterns); + let base_file_count = bigram_boundary(bigram_overlay, files.len()); + + let (files_to_search, filtered_file_count) = prefilter_with_filepath_retry( + files, + constraints, + bigram_candidates.as_deref(), + base_file_count, + options, + arena, + overflow_arena, + ); + + if files_to_search.is_empty() { + return GrepResult::empty(total_files, filtered_file_count); + } + + // Smart case: case-insensitive when all patterns are lowercase + let case_insensitive = if options.smart_case { + !patterns.iter().any(|p| p.chars().any(|c| c.is_uppercase())) + } else { + false + }; + + let ac = aho_corasick::AhoCorasickBuilder::new() + .ascii_case_insensitive(case_insensitive) + .build(patterns) + .expect("Aho-Corasick build should not fail for literal patterns"); + + let searcher = { + let mut b = SearcherBuilder::new(); + b.line_number(true); + b + } + .build(); + + let ac_matcher = AhoCorasickMatcher { ac: &ac }; + perform_grep( + &files_to_search, + options, + &GrepContext { + total_files, + filtered_file_count, + budget, + base_path, + arena, + overflow_arena, + prefilter: None, // no memmem prefilter for multi-pattern search + abort_signal, + }, + |file_bytes: &[u8], max_matches: usize| { + let state = SinkState { + file_index: 0, + matches: Vec::with_capacity(4), + max_matches, + before_context: options.before_context, + after_context: options.after_context, + classify_definitions: options.classify_definitions, + }; + + let mut sink = AhoCorasickSink { state, ac: &ac }; + + if let Err(e) = searcher.search_slice(&ac_matcher, file_bytes, &mut sink) { + tracing::error!(error = %e, "Grep (aho-corasick multi) search failed"); + } + + sink.state.matches + }, + ) +} diff --git a/crates/fff-core/src/grep/prefilter.rs b/crates/fff-core/src/grep/prefilter.rs new file mode 100644 index 000000000..815b6c69c --- /dev/null +++ b/crates/fff-core/src/grep/prefilter.rs @@ -0,0 +1,204 @@ +use super::types::GrepSearchOptions; +use crate::index::BigramFilter; +use crate::index::constraints::{ConstraintPlan, ConstraintsBuffers}; +use crate::sort_buffer::sort_with_buffer; +use crate::types::FileItem; +use fff_query_parser::Constraint; + +/// Prefilter with a FilePath-constraint fallback: when constraints yield 0 +/// files and the query had FilePath constraints, retry without them (the path +/// token was likely part of the search text). +#[allow(clippy::too_many_arguments)] +pub(super) fn prefilter_with_filepath_retry<'a>( + files: &'a [FileItem], + constraints: &[Constraint<'_>], + bigram_candidates: Option<&[u64]>, + base_count: usize, + options: &GrepSearchOptions, + arena: crate::simd_path::ArenaPtr, + overflow_arena: crate::simd_path::ArenaPtr, +) -> (Vec<&'a FileItem>, usize) { + let (files_to_search, filtered_file_count) = prefilter_files( + files, + constraints, + bigram_candidates, + base_count, + options, + arena, + overflow_arena, + ); + + if !files_to_search.is_empty() { + return (files_to_search, filtered_file_count); + } + + let Some(stripped) = strip_file_path_constraint_if_present(constraints) else { + return (files_to_search, filtered_file_count); + }; + + prefilter_files( + files, + &stripped, + bigram_candidates, + base_count, + options, + arena, + overflow_arena, + ) +} + +/// Single pass prefilter that doesn't involve file reading +/// allocates only amount of memory required for storing references of the FileItems have to be +/// opened for grepping unaviodably, in the worst case allocates N * memory if no prefilter needed +fn prefilter_files<'a>( + files: &'a [FileItem], + constraints: &[Constraint<'_>], + bigram_candidates: Option<&[u64]>, + base_count: usize, + options: &GrepSearchOptions, + arena: crate::simd_path::ArenaPtr, + overflow_arena: crate::simd_path::ArenaPtr, +) -> (Vec<&'a FileItem>, usize) { + let max_file_size = options.max_file_size; + let plan = if constraints.is_empty() { + None + } else { + Some(ConstraintPlan::build( + constraints, + files, + arena, + overflow_arena, + )) + }; + + let mut scratch = ConstraintsBuffers::new(); + + #[inline(always)] + fn basic_prefilter(file: &FileItem, max: u64) -> bool { + !file.is_deleted() && !file.is_binary() && file.size > 0 && file.size <= max + } + + // squeeze as much prefilters into a single loop as possible + let mut prefiltered: Vec<&FileItem> = match bigram_candidates { + Some(candidates) => { + let boundary = base_count.min(files.len()); + let (indexed, tail) = files.split_at(boundary); + + let cap = BigramFilter::count_candidates(candidates) + tail.len(); + let mut out: Vec<&FileItem> = Vec::with_capacity(cap); + + let full_words = boundary / 64; + let last_word_bits = boundary % 64; + + // we need this because we already had a regression of the wrong bit + // has been set for the very last word based on the overlay, it's pretty cheap + macro_rules! evaluate_bigram_match_word { + ($word:expr, $base:expr) => {{ + let mut bits: u64 = $word; + while bits != 0 { + let bit = bits.trailing_zeros() as usize; + let file_idx = $base + bit; + bits &= bits - 1; + + let f = unsafe { indexed.get_unchecked(file_idx) }; + if !basic_prefilter(f, max_file_size) { + continue; + } + if let Some(plan) = plan.as_ref() + && !plan.matches(f, file_idx, arena, overflow_arena, &mut scratch) + { + continue; + } + out.push(f); + } + }}; + } + + // Full words: every set bit guaranteed `< boundary`. + for (word_idx, &word) in candidates.iter().take(full_words).enumerate() { + if word != 0 { + evaluate_bigram_match_word!(word, word_idx * 64); + } + } + + // Last partial word: mask bits past `boundary` once at word load. + if last_word_bits != 0 { + // this will get only (mod 64) bits from the last word guaratee that it's 0 padded + let last_mask: u64 = (1u64 << last_word_bits) - 1; + let word = candidates[full_words] & last_mask; + if word != 0 { + evaluate_bigram_match_word!(word, full_words * 64); + } + } + + // Sequential processing for non-bigrammable files: they are always in the end + for (offset, f) in tail.iter().enumerate() { + if !basic_prefilter(f, max_file_size) { + continue; + } + if let Some(ref p) = plan + && !p.matches(f, boundary + offset, arena, overflow_arena, &mut scratch) + { + continue; + } + out.push(f); + } + + out + } + // this will be executed if there is no bigram, in the worst case it will allocate + // whole array of files but probability in the real repo of NO preflter working is so + // low that we just ignore that, usually there would be at least a few files excluded + None => { + let mut out: Vec<&FileItem> = Vec::new(); + for (idx, f) in files.iter().enumerate() { + if !basic_prefilter(f, max_file_size) { + continue; + } + if let Some(ref p) = plan + && !p.matches(f, idx, arena, overflow_arena, &mut scratch) + { + continue; + } + out.push(f); + } + out + } + }; + + let total_count = prefiltered.len(); + + sort_with_buffer(&mut prefiltered, |a, b| { + b.total_frecency_score() + .cmp(&a.total_frecency_score()) + .then(b.modified.cmp(&a.modified)) + }); + + if options.file_offset > 0 && options.file_offset < total_count { + let paginated = prefiltered.split_off(options.file_offset); + (paginated, total_count) + } else if options.file_offset >= total_count { + (Vec::new(), total_count) + } else { + (prefiltered, total_count) + } +} + +fn strip_file_path_constraint_if_present<'a>( + constraints: &[Constraint<'a>], +) -> Option> { + if !constraints + .iter() + .any(|c| matches!(c, Constraint::FilePath(_))) + { + return None; + } + + let filtered: fff_query_parser::ConstraintVec<'a> = constraints + .iter() + .filter(|c| !matches!(c, Constraint::FilePath(_))) + .cloned() + .collect(); + + Some(filtered) +} diff --git a/crates/fff-core/src/grep/regex.rs b/crates/fff-core/src/grep/regex.rs new file mode 100644 index 000000000..ba2e136f1 --- /dev/null +++ b/crates/fff-core/src/grep/regex.rs @@ -0,0 +1,130 @@ +use super::sink::{SinkState, debug_assert_newline_terminator, split_multiline_blob}; +use fff_grep::{ + Searcher, Sink, SinkMatch, + matcher::{Match, Matcher, NoError}, +}; +use smallvec::SmallVec; + +pub fn has_regex_metacharacters(text: &str) -> bool { + regex::escape(text) != text +} + +pub(super) fn build_regex(pattern: &str, smart_case: bool) -> Result { + if pattern.is_empty() { + return Err("empty pattern".to_string()); + } + + let regex_pattern = if pattern.contains("\\n") { + pattern.replace("\\n", "\n") + } else { + pattern.to_string() + }; + + let case_insensitive = if smart_case { + !pattern.chars().any(|c| c.is_uppercase()) + } else { + false + }; + + regex::bytes::RegexBuilder::new(®ex_pattern) + .case_insensitive(case_insensitive) + .multi_line(true) + .unicode(false) + .build() + .map_err(|e| e.to_string()) +} + +pub(super) struct RegexMatcher<'r> { + pub(super) regex: &'r regex::bytes::Regex, + pub(super) is_multiline: bool, +} + +impl Matcher for RegexMatcher<'_> { + type Error = NoError; + + #[inline] + fn find_at(&self, haystack: &[u8], at: usize) -> Result, NoError> { + Ok(self + .regex + .find_at(haystack, at) + .map(|m| Match::new(m.start(), m.end()))) + } + + #[inline] + fn line_terminator(&self) -> Option { + if self.is_multiline { + None + } else { + Some(fff_grep::LineTerminator::byte(b'\n')) + } + } +} + +pub(super) struct RegexSink<'r> { + pub(super) state: SinkState, + pub(super) re: &'r regex::bytes::Regex, +} + +impl Sink for RegexSink<'_> { + type Error = std::io::Error; + + fn matched( + &mut self, + searcher: &Searcher, + sink_match: &SinkMatch<'_>, + ) -> Result { + debug_assert_newline_terminator(searcher); + if self.state.max_matches != 0 && self.state.matches.len() >= self.state.max_matches { + return Ok(false); + } + + let line_bytes = sink_match.bytes(); + let (display_bytes, _, line_number, byte_offset) = + SinkState::prepare_line(line_bytes, sink_match); + + // MultiLine strategy hands over all matched lines as one blob: keep + // `line_content` single-line, the remaining lines become after-context. + let (first_line, extra_after) = split_multiline_blob(display_bytes); + let first_len = first_line.len() as u32; + let line_content = String::from_utf8_lossy(first_line).into_owned(); + let mut match_byte_offsets: SmallVec<[(u32, u32); 4]> = SmallVec::new(); + let mut col = 0usize; + let mut first = true; + + for m in self.re.find_iter(display_bytes) { + let abs_start = m.start() as u32; + if abs_start >= first_len { + continue; // highlight only spans visible in the first line + } + let abs_end = (m.end() as u32).min(first_len); + if first { + col = abs_start as usize; + first = false; + } + match_byte_offsets.push((abs_start, abs_end)); + } + + let (context_before, context_after) = self.state.extract_context(sink_match); + let context_after = if extra_after.is_empty() { + context_after + } else { + let mut combined = extra_after; + combined.extend(context_after); + combined + }; + self.state.push_match( + line_number, + col, + byte_offset, + line_content, + match_byte_offsets, + context_before, + context_after, + ); + Ok(true) + } + + fn finish(&mut self, _: &Searcher, _: &fff_grep::SinkFinish) -> Result<(), Self::Error> { + Ok(()) + } +} diff --git a/crates/fff-core/src/grep/sink.rs b/crates/fff-core/src/grep/sink.rs new file mode 100644 index 000000000..4111dc986 --- /dev/null +++ b/crates/fff-core/src/grep/sink.rs @@ -0,0 +1,242 @@ +use super::types::GrepMatch; +use fff_grep::{Searcher, SinkMatch}; +use smallvec::SmallVec; + +/// Maximum bytes of a matched line to keep for display. Prevents minified +/// JS or huge single-line files from blowing up memory. +pub(super) const MAX_LINE_DISPLAY_LEN: usize = 512; + +#[cfg(feature = "definitions")] +#[inline] +pub(super) fn classify_definition(enabled: bool, line: &str) -> bool { + enabled && super::classify::is_definition_line(line) +} + +#[cfg(not(feature = "definitions"))] +#[inline] +pub(super) fn classify_definition(_enabled: bool, _line: &str) -> bool { + false +} + +#[inline] +pub(super) fn debug_assert_newline_terminator(searcher: &Searcher) { + debug_assert_eq!( + searcher.line_terminator(), + fff_grep::LineTerminator::byte(b'\n'), + "sink helpers assume \\n line terminators (see module invariant)" + ); +} + +#[inline] +pub(super) fn strip_line_terminators(bytes: &[u8]) -> &[u8] { + let mut len = bytes.len(); + while len > 0 && matches!(bytes[len - 1], b'\n' | b'\r') { + len -= 1; + } + &bytes[..len] +} + +pub(super) struct SinkState { + pub(super) file_index: usize, + pub(super) matches: Vec, + pub(super) max_matches: usize, + pub(super) before_context: usize, + pub(super) after_context: usize, + pub(super) classify_definitions: bool, +} + +impl SinkState { + #[inline] + pub(super) fn prepare_line<'a>( + line_bytes: &'a [u8], + mat: &SinkMatch<'_>, + ) -> (&'a [u8], u32, u64, u64) { + let line_number = mat.line_number().unwrap_or(0); + let byte_offset = mat.absolute_byte_offset(); + + // Trim trailing newline/CR directly on bytes to avoid UTF-8 conversion. + let trimmed_bytes = strip_line_terminators(line_bytes); + + // Truncate for display (floor to a char boundary). + let display_bytes = truncate_display_bytes(trimmed_bytes); + + let display_len = display_bytes.len() as u32; + (display_bytes, display_len, line_number, byte_offset) + } + + #[inline] + #[allow(clippy::too_many_arguments)] + pub(super) fn push_match( + &mut self, + line_number: u64, + col: usize, + byte_offset: u64, + line_content: String, + match_byte_offsets: SmallVec<[(u32, u32); 4]>, + context_before: Vec, + context_after: Vec, + ) { + let is_definition = classify_definition(self.classify_definitions, &line_content); + self.matches.push(GrepMatch { + file_index: self.file_index, + line_number, + col, + byte_offset, + line_content, + match_byte_offsets, + fuzzy_score: None, + is_definition, + context_before, + context_after, + }); + } + + /// Extract context lines from the full buffer around a matched region. + pub(super) fn extract_context(&self, mat: &SinkMatch<'_>) -> (Vec, Vec) { + if self.before_context == 0 && self.after_context == 0 { + return (Vec::new(), Vec::new()); + } + + let buffer = mat.buffer(); + let range = mat.bytes_range_in_buffer(); + + let mut before = Vec::new(); + if self.before_context > 0 && range.start > 0 { + // Walk backward from the start of the match line to find preceding lines + let mut pos = range.start; + let mut lines_found = 0; + while lines_found < self.before_context && pos > 0 { + // Skip the newline just before our current position + pos -= 1; + // Find the previous newline + let line_start = match memchr::memrchr(b'\n', &buffer[..pos]) { + Some(nl) => nl + 1, + None => 0, + }; + let line = &buffer[line_start..pos]; + // Trim trailing \r + let line = if line.last() == Some(&b'\r') { + &line[..line.len() - 1] + } else { + line + }; + let truncated = truncate_display_bytes(line); + before.push(String::from_utf8_lossy(truncated).into_owned()); + pos = line_start; + lines_found += 1; + } + before.reverse(); + } + + let mut after = Vec::new(); + if self.after_context > 0 && range.end < buffer.len() { + let mut pos = range.end; + let mut lines_found = 0; + while lines_found < self.after_context && pos < buffer.len() { + // Find the next newline + let line_end = match memchr::memchr(b'\n', &buffer[pos..]) { + Some(nl) => pos + nl, + None => buffer.len(), + }; + let line = &buffer[pos..line_end]; + // Trim trailing \r + let line = if line.last() == Some(&b'\r') { + &line[..line.len() - 1] + } else { + line + }; + let truncated = truncate_display_bytes(line); + after.push(String::from_utf8_lossy(truncated).into_owned()); + pos = if line_end < buffer.len() { + line_end + 1 // skip past \n + } else { + buffer.len() + }; + lines_found += 1; + } + } + + (before, after) + } +} + +/// Truncate a byte slice for display, respecting UTF-8 char boundaries. +#[inline] +pub(super) fn truncate_display_bytes(bytes: &[u8]) -> &[u8] { + if bytes.len() <= MAX_LINE_DISPLAY_LEN { + bytes + } else { + let mut end = MAX_LINE_DISPLAY_LEN; + while end > 0 && !is_utf8_char_boundary(bytes[end]) { + end -= 1; + } + &bytes[..end] + } +} + +/// Split a multiline match blob (from the MultiLine searcher strategy) into +/// the first line and the remaining lines so `line_content` stays single-line. +pub(super) fn split_multiline_blob(display_bytes: &[u8]) -> (&[u8], Vec) { + match memchr::memchr(b'\n', display_bytes) { + None => (display_bytes, Vec::new()), + Some(pos) => { + let first = strip_line_terminators(&display_bytes[..pos + 1]); + let extra = display_bytes[pos + 1..] + .split(|&b| b == b'\n') + .map(|l| String::from_utf8_lossy(strip_line_terminators(l)).into_owned()) + .collect(); + (first, extra) + } + } +} + +/// Convert character-position indices from neo_frizbee into byte-offset +/// pairs (start, end) suitable for `match_byte_offsets`. +/// +/// frizbee returns character positions (0-based index into the char +/// iterator). We need byte ranges because the UI renderer and Lua layer +/// use byte offsets for extmark highlights. +/// +/// Each matched character becomes its own (byte_start, byte_end) pair. +/// Adjacent characters are merged into a single contiguous range. +pub(super) fn char_indices_to_byte_offsets( + line: &str, + char_indices: &[usize], +) -> SmallVec<[(u32, u32); 4]> { + if char_indices.is_empty() { + return SmallVec::new(); + } + + // Build a map: char_index -> (byte_start, byte_end) for all chars. + // Iterating all chars is O(n) in the line length which is bounded by MAX_LINE_DISPLAY_LEN (512). + let char_byte_ranges: Vec<(usize, usize)> = line + .char_indices() + .map(|(byte_pos, ch)| (byte_pos, byte_pos + ch.len_utf8())) + .collect(); + + // Convert char indices to byte ranges, merging adjacent ranges + let mut result: SmallVec<[(u32, u32); 4]> = SmallVec::with_capacity(char_indices.len()); + + for &ci in char_indices { + if ci >= char_byte_ranges.len() { + continue; // out of bounds (shouldn't happen with valid data) + } + let (start, end) = char_byte_ranges[ci]; + // Merge with previous range if adjacent + if let Some(last) = result.last_mut() + && last.1 == start as u32 + { + last.1 = end as u32; + continue; + } + result.push((start as u32, end as u32)); + } + + result +} + +// copied from the rust u8 private method +#[inline] +const fn is_utf8_char_boundary(b: u8) -> bool { + (b as i8) >= -0x40 +} diff --git a/crates/fff-core/src/grep/types.rs b/crates/fff-core/src/grep/types.rs new file mode 100644 index 000000000..21bb3c437 --- /dev/null +++ b/crates/fff-core/src/grep/types.rs @@ -0,0 +1,239 @@ +use crate::types::FileItem; +use smallvec::SmallVec; +use std::sync::Arc; +use std::sync::atomic::AtomicBool; + +pub use crate::constants::MAX_FFFILE_SIZE; + +/// Controls how the grep pattern is interpreted. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub enum GrepMode { + /// Literal plain text match: default path that doesn't require any regex machinery + #[default] + PlainText, + /// Regex mode: uses the same exact matching engine as ripgrep + Regex, + /// Smart fuzzy mode, allows user to make either a couple of single char typos or long gaps + /// e.g. shcema -> shcema, or UserController -> UserAuthController + /// + /// Significatnly slower than plain text, especially on unindexed FilePicker + Fuzzy, +} + +/// A single content match within a file +#[derive(Debug, Clone)] +pub struct GrepMatch { + /// Index into the deduplicated `files` vec of the GrepResult. + pub file_index: usize, + /// 1-based line number. + pub line_number: u64, + /// 0-based byte column of first match start within the line. + pub col: usize, + /// Absolute byte offset of the matched line from the start of the file. + /// Can be used by the preview to seek directly without scanning from the top. + pub byte_offset: u64, + /// The matched line text, truncated to `MAX_LINE_DISPLAY_LEN`. + pub line_content: String, + /// Byte offsets `(start, end)` within `line_content` for each match. + /// Stack-allocated for the common case of ≤4 spans per line. + pub match_byte_offsets: SmallVec<[(u32, u32); 4]>, + /// Fuzzy match score from neo_frizbee (only set in Fuzzy grep mode). + pub fuzzy_score: Option, + /// Whether the matched line looks like a definition (struct, fn, class, etc.). + /// Computed at match time so output formatters don't need to re-scan. + pub is_definition: bool, + /// Lines before the match (for context display). Empty when context is 0. + pub context_before: Vec, + /// Lines after the match (for context display). Empty when context is 0. + pub context_after: Vec, +} + +impl GrepMatch { + /// Strip leading whitespace from `line_content` and all context lines, + /// adjusting `col` and `match_byte_offsets` so highlights remain correct. + pub fn trim_leading_whitespace(&mut self) { + let strip_len = self.line_content.len() - self.line_content.trim_start().len(); + if strip_len > 0 { + self.line_content.drain(..strip_len); + let off = strip_len as u32; + self.col = self.col.saturating_sub(strip_len); + for range in &mut self.match_byte_offsets { + range.0 = range.0.saturating_sub(off); + range.1 = range.1.saturating_sub(off); + } + } + for line in &mut self.context_before { + let n = line.len() - line.trim_start().len(); + if n > 0 { + line.drain(..n); + } + } + for line in &mut self.context_after { + let n = line.len() - line.trim_start().len(); + if n > 0 { + line.drain(..n); + } + } + } +} + +/// Options for grep search. +#[derive(Debug, Clone)] +pub struct GrepSearchOptions { + pub max_file_size: u64, + pub max_matches_per_file: usize, + pub smart_case: bool, + /// File-based pagination offset: index into the sorted/filtered file list + /// to start searching from. Pass 0 for the first page, then use + /// `GrepResult::next_file_offset` for subsequent pages. + pub file_offset: usize, + /// Maximum number of matches to collect before stopping. + pub page_limit: usize, + /// How to interpret the search pattern. Defaults to `PlainText`. + pub mode: GrepMode, + /// Maximum time in milliseconds to spend searching before returning partial + /// results. Prevents UI freezes on pathological queries. 0 = no limit. + pub time_budget_ms: u64, + /// Number of context lines to include before each match. 0 = disabled. + pub before_context: usize, + /// Number of context lines to include after each match. 0 = disabled. + pub after_context: usize, + /// Whether to classify each match as a definition line. Adds ~2% overhead + /// on large repos; disable for interactive grep where it is not needed. + pub classify_definitions: bool, + /// Strip leading whitespace from matched lines and context lines, adjusting + /// highlight byte offsets accordingly. Useful for AI/MCP consumers and UIs + /// that don't need indentation. Default: false. + pub trim_whitespace: bool, + /// External abort signal. When provided, overrides the picker's internal + /// cancellation flag. Set to `true` to stop the search early and return + /// partial results. Omit (or use `..Default::default()`) to let the + /// picker manage cancellation. + pub abort_signal: Option>, +} + +impl Default for GrepSearchOptions { + fn default() -> Self { + Self { + max_file_size: MAX_FFFILE_SIZE, + max_matches_per_file: 200, + smart_case: true, + file_offset: 0, + page_limit: 50, + mode: GrepMode::default(), + time_budget_ms: 0, + before_context: 0, + after_context: 0, + classify_definitions: false, + trim_whitespace: false, + abort_signal: None, + } + } +} + +/// Result of a grep search with a list of matches, list of matched files, and metadata. +#[derive(Debug, Clone, Default)] +pub struct GrepResult<'a> { + pub matches: Vec, + /// Deduplicated file references for the returned matches. + pub files: Vec<&'a FileItem>, + /// Number of files actually searched in this call. + pub total_files_searched: usize, + /// Total number of indexed files (before filtering). + pub total_files: usize, + /// Total number of searchable files (after filtering out binary, too-large, etc.). + pub filtered_file_count: usize, + /// Number of files that contained at least one match. + pub files_with_matches: usize, + /// The file offset to pass for the next page. `0` if there are no more files. + /// Callers should store this and pass it as `file_offset` in the next call. + pub next_file_offset: usize, + /// When regex mode fails to compile the pattern, the search falls back to + /// literal matching and this field contains the compilation error message. + /// The UI can display this to inform the user their regex was invalid. + pub regex_fallback_error: Option, +} + +impl<'a> GrepResult<'a> { + /// Empty result carrying only the file counts (empty query / prefilter miss) + pub(crate) fn empty(total_files: usize, filtered_file_count: usize) -> Self { + Self { + total_files, + filtered_file_count, + ..Default::default() + } + } + + pub(crate) fn collect( + per_file_results: Vec<(usize, &'a FileItem, Vec)>, + files_to_search_len: usize, + options: &GrepSearchOptions, + total_files: usize, + filtered_file_count: usize, + budget_exceeded: bool, + ) -> Self { + let page_limit = options.page_limit; + + // Each match stores a `file_index` pointing into `result_files` so that + // consumers (FFI JSON, Lua) can look up file metadata without duplicating + // it across every match from the same file + let mut result_files: Vec<&'a FileItem> = Vec::new(); + let mut all_matches: Vec = Vec::new(); + // files_consumed tracks how far into files_to_search we have advanced, + // counting every file whose results were emitted (with or without matches). + // We use the batch_idx of the last consumed file + 1, which is correct + // because per_file_results only contains files that had matches, and + // files between them that had no matches were still searched and can be + // safely skipped on the next page + let mut files_consumed: usize = 0; + + for (batch_idx, file, file_matches) in per_file_results { + // batch_idx is the 0-based position in files_to_search. + // Advance files_consumed to include this file and all no-match files before it. + files_consumed = batch_idx + 1; + + let file_result_idx = result_files.len(); + result_files.push(file); + + for mut m in file_matches { + m.file_index = file_result_idx; + if options.trim_whitespace { + m.trim_leading_whitespace(); + } + all_matches.push(m); + } + + // page_limit is a soft cap: we always finish the current file before + // stopping, so no matches are dropped. A page may return up to + // page_limit + max_matches_per_file - 1 matches in the worst case + if all_matches.len() >= page_limit { + break; + } + } + + // If no file had any match, we searched the entire slice. + if result_files.is_empty() { + files_consumed = files_to_search_len; + } + + let has_more = budget_exceeded + || (all_matches.len() >= page_limit && files_consumed < files_to_search_len); + + let next_file_offset = if has_more { + options.file_offset + files_consumed + } else { + 0 + }; + + Self { + matches: all_matches, + files_with_matches: result_files.len(), + files: result_files, + total_files_searched: files_consumed, + total_files, + filtered_file_count, + next_file_offset, + regex_fallback_error: None, + } + } +} diff --git a/crates/fff-core/src/grep/utils.rs b/crates/fff-core/src/grep/utils.rs deleted file mode 100644 index 37eb2b2b1..000000000 --- a/crates/fff-core/src/grep/utils.rs +++ /dev/null @@ -1,122 +0,0 @@ -use super::grep::{GrepMatch, GrepSearchOptions}; -use crate::types::FileItem; - -#[inline] -pub(crate) fn strip_line_terminators(bytes: &[u8]) -> &[u8] { - let mut len = bytes.len(); - while len > 0 && matches!(bytes[len - 1], b'\n' | b'\r') { - len -= 1; - } - &bytes[..len] -} - -/// Result of a grep search with a list of matches, list of matched files, and metadata. -#[derive(Debug, Clone, Default)] -pub struct GrepResult<'a> { - pub matches: Vec, - /// Deduplicated file references for the returned matches. - pub files: Vec<&'a FileItem>, - /// Number of files actually searched in this call. - pub total_files_searched: usize, - /// Total number of indexed files (before filtering). - pub total_files: usize, - /// Total number of searchable files (after filtering out binary, too-large, etc.). - pub filtered_file_count: usize, - /// Number of files that contained at least one match. - pub files_with_matches: usize, - /// The file offset to pass for the next page. `0` if there are no more files. - /// Callers should store this and pass it as `file_offset` in the next call. - pub next_file_offset: usize, - /// When regex mode fails to compile the pattern, the search falls back to - /// literal matching and this field contains the compilation error message. - /// The UI can display this to inform the user their regex was invalid. - pub regex_fallback_error: Option, -} - -impl<'a> GrepResult<'a> { - /// Empty result carrying only the file counts (empty query / prefilter miss) - pub(crate) fn empty(total_files: usize, filtered_file_count: usize) -> Self { - Self { - total_files, - filtered_file_count, - ..Default::default() - } - } - - pub(crate) fn collect( - per_file_results: Vec<(usize, &'a FileItem, Vec)>, - files_to_search_len: usize, - options: &GrepSearchOptions, - total_files: usize, - filtered_file_count: usize, - budget_exceeded: bool, - ) -> Self { - let page_limit = options.page_limit; - - // Each match stores a `file_index` pointing into `result_files` so that - // consumers (FFI JSON, Lua) can look up file metadata without duplicating - // it across every match from the same file - let mut result_files: Vec<&'a FileItem> = Vec::new(); - let mut all_matches: Vec = Vec::new(); - // files_consumed tracks how far into files_to_search we have advanced, - // counting every file whose results were emitted (with or without matches). - // We use the batch_idx of the last consumed file + 1, which is correct - // because per_file_results only contains files that had matches, and - // files between them that had no matches were still searched and can be - // safely skipped on the next page - let mut files_consumed: usize = 0; - - for (batch_idx, file, file_matches) in per_file_results { - // batch_idx is the 0-based position in files_to_search. - // Advance files_consumed to include this file and all no-match files before it. - files_consumed = batch_idx + 1; - - let file_result_idx = result_files.len(); - result_files.push(file); - - for mut m in file_matches { - m.file_index = file_result_idx; - if options.trim_whitespace { - m.trim_leading_whitespace(); - } - all_matches.push(m); - } - - // page_limit is a soft cap: we always finish the current file before - // stopping, so no matches are dropped. A page may return up to - // page_limit + max_matches_per_file - 1 matches in the worst case - if all_matches.len() >= page_limit { - break; - } - } - - // If no file had any match, we searched the entire slice. - if result_files.is_empty() { - files_consumed = files_to_search_len; - } - - let has_more = budget_exceeded - || (all_matches.len() >= page_limit && files_consumed < files_to_search_len); - - let next_file_offset = if has_more { - options.file_offset + files_consumed - } else { - 0 - }; - - Self { - matches: all_matches, - files_with_matches: result_files.len(), - files: result_files, - total_files_searched: files_consumed, - total_files, - filtered_file_count, - next_file_offset, - regex_fallback_error: None, - } - } -} - -pub fn has_regex_metacharacters(text: &str) -> bool { - regex::escape(text) != text -} diff --git a/crates/fff-core/src/bigram_filter.rs b/crates/fff-core/src/index/bigram_filter.rs similarity index 100% rename from crates/fff-core/src/bigram_filter.rs rename to crates/fff-core/src/index/bigram_filter.rs diff --git a/crates/fff-core/src/bigram_query.rs b/crates/fff-core/src/index/bigram_query.rs similarity index 99% rename from crates/fff-core/src/bigram_query.rs rename to crates/fff-core/src/index/bigram_query.rs index fd7c57f98..b834f9035 100644 --- a/crates/fff-core/src/bigram_query.rs +++ b/crates/fff-core/src/index/bigram_query.rs @@ -1,4 +1,4 @@ -use crate::bigram_filter::BigramFilter; +use crate::index::bigram_filter::BigramFilter; use regex_syntax::hir::{Class, Hir, HirKind}; use smallvec::SmallVec; use std::borrow::Cow; @@ -677,7 +677,7 @@ fn simplify_or(children: Vec) -> BigramQuery { #[cfg(test)] mod tests { use super::*; - use crate::bigram_filter::BigramIndexBuilder; + use crate::index::bigram_filter::BigramIndexBuilder; /// Build a tiny index from the given file contents for testing. fn build_test_index(files: &[&[u8]]) -> BigramFilter { diff --git a/crates/fff-core/src/index/candidates.rs b/crates/fff-core/src/index/candidates.rs new file mode 100644 index 000000000..e6a2669bd --- /dev/null +++ b/crates/fff-core/src/index/candidates.rs @@ -0,0 +1,118 @@ +use super::{BigramFilter, BigramOverlay, extract_bigrams}; +use super::{fuzzy_to_bigram_query, regex_to_bigram_query}; + +/// Number of evenly-spaced probe bigrams used by the fuzzy candidate query. +const FUZZY_PROBE_COUNT: usize = 7; + +#[inline] +fn set_bit(candidates: &mut [u64], file_idx: usize) { + let word = file_idx / 64; + if word < candidates.len() { + candidates[word] |= 1u64 << (file_idx % 64); + } +} + +#[inline] +fn clear_tombstones(candidates: &mut [u64], overlay: &BigramOverlay) { + for (r, t) in candidates.iter_mut().zip(overlay.tombstones().iter()) { + *r &= !t; + } +} + +/// Number of base files covered by the bigram bitset; files past this +/// boundary (overflow, max 1024) are always scanned. +#[inline] +pub(crate) fn bigram_boundary(overlay: Option<&BigramOverlay>, files_len: usize) -> usize { + overlay.map(|o| o.base_file_count()).unwrap_or(files_len) +} + +/// Candidate bitset for literal patterns, OR-ed across all of them: a file is +/// a candidate when it contains the bigrams of ANY pattern. Overlay-modified +/// files are re-checked against each pattern's bigrams. +pub(crate) fn literal_candidates( + index: Option<&BigramFilter>, + overlay: Option<&BigramOverlay>, + patterns: &[&str], +) -> Option> { + let index = ready_index(index)?; + + let mut combined: Option> = None; + for pattern in patterns { + if let Some(candidates) = index.query(pattern.as_bytes()) { + combined = Some(match combined { + None => candidates, + Some(mut acc) => { + acc.iter_mut() + .zip(candidates.iter()) + .for_each(|(a, b)| *a |= *b); + acc + } + }); + } + } + + let mut candidates = combined?; + if let Some(overlay) = overlay { + clear_tombstones(&mut candidates, overlay); + for pattern in patterns { + let pattern_bigrams = extract_bigrams(pattern.as_bytes()); + for file_idx in overlay.query_modified(&pattern_bigrams) { + set_bit(&mut candidates, file_idx); + } + } + } + Some(candidates) +} + +/// Candidate bitset for a regex pattern: the regex HIR is decomposed into an +/// AND/OR bigram query tree (supports alternation, optional groups, character +/// classes, and sparse-1 bigrams across single-byte wildcards). Since modified +/// file contents can't be re-checked against a regex cheaply, all +/// overlay-modified files are conservatively added. +pub(crate) fn regex_candidates( + index: Option<&BigramFilter>, + overlay: Option<&BigramOverlay>, + pattern: &str, +) -> Option> { + let index = ready_index(index)?; + + let bq = regex_to_bigram_query(pattern); + if bq.is_any() { + return None; + } + let candidates = bq.evaluate(index)?; + Some(add_all_modified(candidates, overlay)) +} + +/// Candidate bitset for a fuzzy pattern: evenly-spaced probe bigrams with a +/// typo allowance (widely-spaced probes are far more selective than sliding +/// windows of adjacent bigrams). All overlay-modified files are added. +pub(crate) fn fuzzy_candidates( + index: Option<&BigramFilter>, + overlay: Option<&BigramOverlay>, + pattern: &str, +) -> Option> { + let index = ready_index(index)?; + + let bq = fuzzy_to_bigram_query(pattern, FUZZY_PROBE_COUNT); + if bq.is_any() { + return None; + } + let candidates = bq.evaluate(index)?; + Some(add_all_modified(candidates, overlay)) +} + +#[inline] +fn ready_index(index: Option<&BigramFilter>) -> Option<&BigramFilter> { + index.filter(|idx| idx.is_ready()) +} + +fn add_all_modified(mut candidates: Vec, overlay: Option<&BigramOverlay>) -> Vec { + if let Some(overlay) = overlay { + clear_tombstones(&mut candidates, overlay); + for file_idx in overlay.modified_indices() { + set_bit(&mut candidates, file_idx); + } + } + candidates +} diff --git a/crates/fff-core/src/constraints.rs b/crates/fff-core/src/index/constraints.rs similarity index 100% rename from crates/fff-core/src/constraints.rs rename to crates/fff-core/src/index/constraints.rs diff --git a/crates/fff-core/src/index/mod.rs b/crates/fff-core/src/index/mod.rs new file mode 100644 index 000000000..1e19a09d1 --- /dev/null +++ b/crates/fff-core/src/index/mod.rs @@ -0,0 +1,11 @@ +#[doc(hidden)] // for bench +pub mod bigram_filter; +pub(crate) use bigram_filter::*; + +mod bigram_query; +pub use bigram_query::*; + +mod candidates; +pub(crate) use candidates::*; + +pub mod constraints; diff --git a/crates/fff-core/src/lib.rs b/crates/fff-core/src/lib.rs index 434fcd833..3f8c427b4 100644 --- a/crates/fff-core/src/lib.rs +++ b/crates/fff-core/src/lib.rs @@ -120,7 +120,7 @@ pub mod git; pub mod grep; pub use grep::*; -/// Tracing/logging initialization and panic hook setup. +/// Tracing/logging initialization pub mod log; /// Various path utils might be handy for you to work with fff paths @@ -135,13 +135,12 @@ pub mod constants; // ================================== // these are public only for benchmarks, no backward compatibility guaranteed #[doc(hidden)] -pub mod bigram_filter; +pub use index::bigram_filter; #[doc(hidden)] pub mod simd_string_utils; // ================================== mod background_watcher; -mod constraints; mod error; mod git_status_worker; mod ignore; @@ -149,7 +148,7 @@ mod scan; mod score; mod sort_buffer; -pub(crate) mod bigram_query; +pub(crate) mod index; pub(crate) mod parallelism; pub(crate) mod simd_path; pub(crate) mod stable_vec; diff --git a/crates/fff-core/src/scan.rs b/crates/fff-core/src/scan.rs index 3fe6fa37b..1ef6c00a3 100644 --- a/crates/fff-core/src/scan.rs +++ b/crates/fff-core/src/scan.rs @@ -6,9 +6,9 @@ use tracing::{error, info}; use crate::FileSync; use crate::background_watcher::BackgroundWatcher; -use crate::bigram_filter::{build_bigram_index, sniff_binary_for_non_indexable}; use crate::error::Error; use crate::file_picker::FFFMode; +use crate::index::{build_bigram_index, sniff_binary_for_non_indexable}; use crate::parallelism::BACKGROUND_THREAD_POOL; use crate::shared::{SharedFilePicker, SharedFrecency}; use crate::types::ContentCacheBudget; diff --git a/crates/fff-core/src/score.rs b/crates/fff-core/src/score.rs index f757f4816..00cbff2ab 100644 --- a/crates/fff-core/src/score.rs +++ b/crates/fff-core/src/score.rs @@ -1,6 +1,6 @@ use crate::{ - constraints::apply_constraints, git::is_modified_status, + index::constraints::apply_constraints, path_utils::calculate_distance_penalty, simd_path::{ArenaPtr, MAX_PATH_CHUNKS}, sort_buffer::{sort_by_key_with_buffer, sort_with_buffer}, diff --git a/crates/fff-core/src/types.rs b/crates/fff-core/src/types.rs index 932fb10d8..516953671 100644 --- a/crates/fff-core/src/types.rs +++ b/crates/fff-core/src/types.rs @@ -7,7 +7,7 @@ use std::sync::atomic::{AtomicI32, AtomicU8, AtomicU64, AtomicUsize, Ordering}; #[cfg(not(target_os = "windows"))] use crate::constants::{FRESH_MMAP_THRESHOLD, MMAP_THRESHOLD}; use crate::constants::{MAX_CACHED_CONTENT_BYTES, MAX_FFFILE_SIZE, PATH_BUF_SIZE}; -use crate::constraints::Constrainable; +use crate::index::constraints::Constrainable; use crate::query_tracker::QueryMatchEntry; use crate::simd_path::ArenaPtr; use fff_query_parser::{FFFQuery, FuzzyQuery, Location}; diff --git a/lua/fff/conf.lua b/lua/fff/conf.lua index ab0683f9d..6b5644471 100644 --- a/lua/fff/conf.lua +++ b/lua/fff/conf.lua @@ -34,6 +34,7 @@ local M = {} --- @field preview_scroll_down string --- @field toggle_debug string --- @field cycle_grep_modes string +--- @field insert_newline_escape string --- @field cycle_previous_query string --- @field cycle_forward_query string --- @field grep_jump_to_next_file string|string[] @@ -276,6 +277,9 @@ local function init() toggle_debug = '', -- grep mode: cycle between plain text, regex, and fuzzy search cycle_grep_modes = '', + -- grep mode only: insert a literal `\n` to search across lines + -- (requires a terminal with extended-key support to distinguish from ) + insert_newline_escape = '', -- grep mode only: jump cursor to first item of next/prev file group grep_jump_to_next_file = { '', '' }, grep_jump_to_prev_file = { '', '' }, diff --git a/lua/fff/picker_ui/ui_creator.lua b/lua/fff/picker_ui/ui_creator.lua index c6bac4e0e..502057b10 100644 --- a/lua/fff/picker_ui/ui_creator.lua +++ b/lua/fff/picker_ui/ui_creator.lua @@ -390,6 +390,16 @@ function M.setup_keymaps() set_keymap({ 'i', 'n' }, keymaps.send_to_quickfix, P.send_to_quickfix, input_opts) set_keymap({ 'i', 'n' }, keymaps.cycle_grep_modes, P.cycle_grep_modes, input_opts) + if keymaps.insert_newline_escape then + -- Inserts the literal 2-char `\n` sequence which the grep engine + -- interprets as a multiline search boundary + local newline_escape_opts = vim.tbl_extend('force', input_opts, { expr = true, replace_keycodes = false }) + set_keymap('i', keymaps.insert_newline_escape, function() + if S.mode ~= 'grep' then return '' end + return '\\n' + end, newline_escape_opts) + end + local input_mouse_opts = vim.tbl_extend('force', input_opts, { expr = true, replace_keycodes = true }) set_keymap( { 'i', 'n' },