//! The ranked search cascade. //! //! One term, four table scans, eleven ranks. Rank base = stage number: //! //! | rank | meaning | scan | //! |-----:|----------------------------------|------| //! | 1.x | exact filename, exact case | A | //! | 2.x | exact filename, any case | A | //! | 3.x | filename substring, exact case | A | //! | 4.x | filename substring, any case | A | //! | 5.x | full text occurrence, exact case | B | //! | 6.x | full text occurrence, any case | B | //! | 7.x | fuzzy filename | C | //! | 8.x | fuzzy full text | D | //! | 9.x | full path substring, exact case | A | //! | 10.x | full path substring, any case | A | //! | 11.x | fuzzy full path | C | //! //! Pass A is a single `files` scan (`LIKE`, the ASCII-nocase superset of //! its ranks) classified per-row in Rust — no index needed, the substring //! stage visits every row anyway. A path always ends in its own name, so //! `path LIKE` is a superset of `name LIKE` and that one scan covers the //! filename *and* the path tiers. Pass B is one FTS phrase MATCH verified //! against the decompressed text. Passes C/D (opt-in) iterate the whole //! table with a bitap matcher, C covering both the name and the path. //! //! Wildcard terms (`rep*rt`) rank through the same tiers, with 1/2 meaning //! the whole name matches the pattern; they skip the fuzzy passes (bitap is //! a literal matcher). A regex-only query (`regex:…` with no term) runs two //! dedicated scans that reuse tiers 4 (name), 6 (content) and 10 (path); a //! regex accompanying a term is an accept-predicate on every pass instead, //! not a rank source. //! //! The path tiers rank below everything else, so passes A and C buffer them //! instead of emitting them — stages E and F flush those buffers at the //! end, dropping files an earlier stage already emitted. Path matching //! needs a term of at least three characters, the same floor pass B has. //! //! Full-text ranks order equal-based hits by occurrence count via a //! decimal fraction: `base + (1000 - min(count, 1000)) / 1000` — more //! occurrences sorts earlier, 1000+ occurrences adds zero. Fuzzy ranks add //! `0.1 × edit_distance` instead. //! //! Every scan appends the caller's structured-filter SQL (anonymous //! placeholders over alias `f`) and checks the generation counter as it //! streams; a bumped generation aborts mid-statement. //! //! # Batches are ordered within themselves, not against each other //! //! A pass hands hits over *while* it scans (see [`FLUSH_INTERVAL`]), and a //! scan finds hits in table order — batch two can hold something better than //! anything in batch one. Each batch is sorted before it goes; the consumer //! owns the ordering *across* batches. Do not assume arrival order is rank //! order. use std::collections::HashSet; use std::hash::{BuildHasherDefault, Hasher}; use std::sync::atomic::{AtomicU64, Ordering}; use std::time::{Duration, Instant}; use rusqlite::Connection; use rusqlite::OptionalExtension; use crate::config::IgnoreSet; use crate::query::pattern::clamp_match_range; use crate::query::split::CascadeQuery; use crate::query::translator::{escape_like, quote_phrase}; use crate::snippet; use super::fuzzy::{edit_budget, pigeonhole_chunks, Bitap}; use super::{SearchHit, SearchOptions}; mod passes; /// Cancellation is checked every this many scanned rows in row-cheap /// passes; decompression-heavy passes check every row. const CANCEL_CHECK_ROWS: usize = 256; /// Snippet window budget: the GUI trims the cell text to its column width /// around the match, and the mouseover shows the rest as extended context. const SNIPPET_WINDOW_CHARS: usize = 600; /// The Content Match snippet for one document body, cut exactly as the /// full-text passes cut it. /// /// `folded` must be `text` ASCII-lowercased. That fold is byte-length /// preserving, which is the whole reason offsets found in it can slice `text`; /// the passes hold one reusable fold buffer per scan and hand it in here /// rather than paying for a second copy. /// /// Shared so that [`crate::live`], re-cutting a snippet for a file that /// changed under a result already on screen, produces the same window the /// search itself would — otherwise a row would visibly re-frame its own match /// the moment the file was touched. pub fn text_snippet( pattern: &crate::query::pattern::TermPattern, text: &str, folded: &str, ) -> Option { let opts = snippet::Options { approx_chars: SNIPPET_WINDOW_CHARS, }; match text_snippet_counted(pattern, text, folded) { // Literal terms keep the richer multi-occurrence extract; a wildcard // match marks its own first range. Some((snip, _)) => Some(snip), None => pattern.find_first_folded(folded).map(|r| { // A greedy pattern can match megabytes; clamp before the window. let r = clamp_match_range(text, r, SNIPPET_WINDOW_CHARS); snippet::window_around(text, (r.start, r.end), &opts) }), } } /// [`text_snippet`] for a literal pattern, with the case-insensitive /// occurrence count the extraction found on its way to the window. `None` /// when the pattern is not literal. /// /// The count is a by-product: [`snippet::extract_folded`] locates every /// occurrence in `folded` in order to coalesce and mark them, and that set has /// the same cardinality [`crate::query::pattern::TermPattern::count_folded`] /// would return. Taking it from here is what lets the full-text pass verify a /// row and cut its snippet in one sweep of the body rather than two — see /// `benches/search.rs`, group `cascade_row_sweeps`. It does not generalise to /// a wildcard, whose snippet comes from a single leftmost match. pub fn text_snippet_counted( pattern: &crate::query::pattern::TermPattern, text: &str, folded: &str, ) -> Option<(snippet::Snippet, usize)> { let opts = snippet::Options { approx_chars: SNIPPET_WINDOW_CHARS, }; // The pre-folded form: this runs per candidate row, and folding the term // again here would allocate once per row for a string the pattern holds. let term = pattern.literal_folded()?; Some(snippet::extract_folded(text, folded, &[term], &opts)) } /// The fuzzy full-text match in one document body: how many times the term /// occurs within the edit budget, and the Content Match window cut around /// the first occurrence at the cascade's own width. `None` when it does not /// occur at all. /// /// Shared with [`crate::live`] for the same reason as [`text_snippet`]: a /// fuzzy row whose file changes has to be re-cut the way it was cut, and /// bitap's range is what it was cut around. `bitap` is built once by the /// caller — per scan in the pass, per arm in the live watcher — since building /// it is the cost. /// /// Takes no folded copy, unlike [`text_snippet`]: the matcher folds in its /// mask table (see [`crate::search::fuzzy`]), so it reads the body as stored. /// That is the difference between this pass copying and lowercasing every /// document in the index per keystroke and not doing so. pub fn fuzzy_snippet( bitap: &crate::search::fuzzy::Bitap, text: &str, ) -> Option<(usize, snippet::Snippet)> { let opts = snippet::Options { approx_chars: SNIPPET_WINDOW_CHARS, }; // `first` is `Some` exactly when `count` is non-zero: it *is* the first // of them. let (count, first) = bitap.count_and_first(text.as_bytes()); first.map(|range| (count, snippet::window_around(text, range, &opts))) } #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub struct Outcome { pub total: usize, pub limited: bool, } /// Run the cascade, streaming rank-ordered batches into `sink`. /// `Ok(None)` means the search was cancelled (generation moved on) — the /// caller sends no completion. SQL errors are returned as strings *unless* /// the search was already cancelled (an interrupted statement is normal /// cancellation, not an error). pub fn run( conn: &Connection, query: &CascadeQuery, options: &SearchOptions, generation: u64, latest_gen: &AtomicU64, sink: &mut dyn FnMut(Vec), ) -> Result, String> { if query.is_empty() { return Ok(Some(Outcome { total: 0, limited: false, })); } let ignore = IgnoreSet::compile(&options.session_ignores) .map_err(|e| format!("session ignore filter: {}", e))?; let mut cx = Cx { conn, query, options, generation, latest_gen, ignore, emitted: IdSet::default(), deferred_path: Deferred::default(), deferred_fuzzy_path: Deferred::default(), total: 0, limited: false, sink, }; // With no term at all the regex drives its own scans; `Path` still // flushes the deferred rank-10 buffer the name pass sets aside. let passes: &[Pass] = if query.pattern.is_empty() { &[Pass::RegexName, Pass::RegexContent, Pass::Path] } else { &[ Pass::Filename, Pass::FullText, Pass::FuzzyFilename, Pass::FuzzyFullText, Pass::Path, Pass::FuzzyPath, ] }; for pass in passes { if cx.cancelled() { return Ok(None); } // Stop, but do not call it truncation. `remaining() == 0` is also // what an exactly-full result set looks like, and claiming a cut there // tells a user to raise `--limit` on a complete answer. `flush_pass` // is the authority — it sets the flag when it actually had to drop // rows — and `scan_pass` breaks its row loop the moment the limit // fills, for the same reason this break exists. if cx.remaining() == 0 { break; } let run_pass = match pass { Pass::Filename => cx.pass_filename(), Pass::FullText => cx.pass_fulltext(), Pass::FuzzyFilename => cx.pass_fuzzy_filename(), Pass::FuzzyFullText => cx.pass_fuzzy_fulltext(), Pass::RegexName => cx.pass_regex_name(), Pass::RegexContent => cx.pass_regex_content(), Pass::Path => { let d = std::mem::take(&mut cx.deferred_path); cx.flush_deferred(d) } Pass::FuzzyPath => { let d = std::mem::take(&mut cx.deferred_fuzzy_path); cx.flush_deferred(d) } }; match run_pass { Ok(true) => {} Ok(false) => return Ok(None), // cancelled mid-pass Err(e) => { // A kill from `interrupt()` arrives as an ordinary SQL error, // and SQLite's interrupt flag carries no ordering edge to the // generation counter bumped just before it: without this // fence a weakly-ordered CPU could report routine // cancellation as `Search failed: interrupted`. std::sync::atomic::fence(Ordering::Acquire); if cx.cancelled() { return Ok(None); // interrupt() killed the statement } return Err(e); } } } Ok(Some(Outcome { total: cx.total, limited: cx.limited, })) } enum Pass { Filename, FullText, FuzzyFilename, FuzzyFullText, /// Regex-only: name hits at rank 4 now, path hits deferred to rank 10. RegexName, /// Regex-only: content hits at rank 6. RegexContent, /// Flush of the rank 9–10 hits pass A set aside. Path, /// Flush of the rank 11 hits pass C set aside. FuzzyPath, } /// Occurrence-count fraction: more occurrences → smaller fraction → sorts /// earlier within a rank base; 1000+ adds zero. fn count_frac(count: usize) -> f64 { (1000usize.saturating_sub(count.min(1000))) as f64 / 1000.0 } /// `row.get` with the crate's string-error convention. fn col(row: &rusqlite::Row<'_>, idx: usize) -> Result { row.get(idx).map_err(|e| e.to_string()) } /// Rank, then name, then path — the path tiebreak makes the order total, so /// equal-rank hits cannot shuffle between otherwise-identical sorts. fn rank_order(a: &SearchHit, b: &SearchHit) -> std::cmp::Ordering { a.rank .total_cmp(&b.rank) .then_with(|| a.name.cmp(&b.name)) .then_with(|| a.path.cmp(&b.path)) } /// The `files` columns every pass selects, in the order the passes index /// them: `0` id, `1` name, `2` parent, `3` size, `4` mtime. Passes that also /// want the stored document text append `dt.text_zstd` as column `5`. A pass /// spelling its own list in a different order would compile and then quietly /// serve parents as names. /// /// There is no `path` column to select; [`Cx::scan_pass`] concatenates columns /// 2 and 1 into one reused buffer and hands every classifier the result. const HIT_COLUMNS: &str = "f.id, f.name, f.parent, f.size, f.mtime"; /// Columns 3 and 4. The clamp matters: `size` is `INTEGER` in SQLite and so /// signed; a corrupt row holding `-1` would otherwise become 18 exabytes on /// the way to `u64` and sort to the top of every size-ordered result. fn size_and_mtime(row: &rusqlite::Row<'_>) -> Result<(u64, i64), String> { let size = col::(row, 3)?.max(0) as u64; let mtime = col(row, 4)?; Ok((size, mtime)) } /// The path tiers only make sense with enough term to be specific — same /// floor the trigram full-text pass uses. Wildcards count only their /// literal content (`a*b` is two characters of specificity, not three). fn path_tiers_enabled(pattern: &crate::query::pattern::TermPattern) -> bool { pattern.literal_char_count() >= 3 } /// Hits collected by one scan but ranked below later scans, so held back /// until every better stage has emitted. #[derive(Default)] struct Deferred { hits: Vec, overflowed: bool, } /// Longest a pass may sit on hits before handing them over. A pass is a /// whole-table scan that can run for seconds; draining on a clock rather /// than only on a full buffer keeps even a sparse query painting as the /// scan reaches its matches. Short enough to land two or three batches /// inside the GUI's 250 ms result fade. const FLUSH_INTERVAL: Duration = Duration::from_millis(80); /// Hasher for the emitted-id set, which is probed **once per scanned row**. /// /// The default `HashSet` hasher is SipHash-1-3, chosen to be collision-resistant /// against hostile keys. These keys are SQLite rowids the cascade itself just /// read — nobody outside chooses them — so the resistance buys nothing, while /// the ~10 ns it costs is paid on every row of a whole-table scan. /// /// Multiplicative, by an odd constant near 2^64/φ. Odd keeps it a bijection, so /// distinct ids stay distinct; hashbrown then takes the low bits for the bucket /// and the top seven for its control byte, and the rotate is what puts real /// entropy in both halves for the near-consecutive ids a table scan produces. /// /// Worth ~5% of a fuzzy search — 17.1/17.4/17.9 ms against 18.2/18.3/18.8 over /// three runs each of `tests/search_alloc.rs`, every run of the one beating /// every run of the other. Small, but it is the whole of what a hash function /// choice can be worth, and it is bought for fifteen lines with no behaviour /// attached. The figure will grow with the row count: this is one probe per /// scanned row, and the corpus is 50,000 rows. #[derive(Default)] struct IdHasher(u64); impl Hasher for IdHasher { fn finish(&self) -> u64 { self.0 } // The only shape the set ever hashes is a single `i64`, which `Hash for // i64` delivers through `write_i64`. `write` exists because the trait // requires it and is deliberately a poor general-purpose hash: routing // anything else through here would be a bug, not a use. fn write(&mut self, bytes: &[u8]) { for &b in bytes { self.0 = (self.0 ^ u64::from(b)).wrapping_mul(0x0100_0000_01b3); } } fn write_i64(&mut self, n: i64) { self.write_u64(n as u64); } fn write_u64(&mut self, n: u64) { let mixed = n.wrapping_mul(0x9E37_79B9_7F4A_7C15); self.0 = mixed.rotate_left(31) ^ mixed; } } type IdSet = HashSet>; struct Cx<'a> { conn: &'a Connection, query: &'a CascadeQuery, options: &'a SearchOptions, generation: u64, latest_gen: &'a AtomicU64, ignore: IgnoreSet, emitted: IdSet, /// Ranks 9–10, filled by pass A. deferred_path: Deferred, /// Rank 11, filled by pass C. deferred_fuzzy_path: Deferred, total: usize, limited: bool, sink: &'a mut dyn FnMut(Vec), } /// Drives [`Cx::flush_if_due`]: when this pass last handed hits over, and /// whether it has handed over anything at all. struct FlushClock { last: Instant, sent_anything: bool, } impl FlushClock { fn new() -> FlushClock { FlushClock { last: Instant::now(), sent_anything: false, } } /// Whether a buffer of `len` hits should go now. The first batch of a /// pass goes the moment there is anything to send. fn due(&self, len: usize, batch: usize) -> bool { if len == 0 { return false; } !self.sent_anything || len >= batch || self.last.elapsed() >= FLUSH_INTERVAL } fn mark_sent(&mut self) { self.last = Instant::now(); self.sent_anything = true; } } impl<'a> Cx<'a> { fn cancelled(&self) -> bool { self.generation != self.latest_gen.load(Ordering::Relaxed) } fn remaining(&self) -> usize { self.options.limit.saturating_sub(self.total) } /// Buffer cap for scan passes: enough headroom that sorting keeps the /// best candidates, without unbounded growth on huge hit sets. fn buffer_cap(&self) -> usize { 4096.max(2 * self.remaining()) } fn params_with_filters( &self, leading: Vec, ) -> Vec { let mut p = leading; p.extend(self.query.filter_params.iter().cloned()); p } /// Skip rows already emitted at a better rank or hidden by session /// ignore chips. fn skip(&self, file_id: i64, path: &str) -> bool { self.emitted.contains(&file_id) || self.ignore.matches_path(std::path::Path::new(path)) } /// The `regex:` accept-predicate applied to every candidate row when a /// regex accompanies a term. The path contains the name, so one path /// check covers both; content is fetched (and decompressed) only for /// rows whose path missed — bounded by the pass's hit count, not its /// scan count. Pass `text` when the pass already has the content. fn regex_accepts(&self, file_id: i64, path: &str, text: Option<&str>) -> Result { let Some(re) = &self.query.regex else { return Ok(true); }; if re.is_match(path) { return Ok(true); } if let Some(text) = text { return Ok(re.is_match(text)); } let blob: Option> = self .conn .query_row( "SELECT text_zstd FROM documents_text WHERE file_id = ?1", [file_id], |r| r.get(0), ) .optional() .map_err(|e| e.to_string())?; let Some(raw) = blob.and_then(|b| zstd::decode_all(b.as_slice()).ok()) else { return Ok(false); }; Ok(re.is_match(&String::from_utf8_lossy(&raw))) } /// Hand `buf` over mid-scan if it is due, leaving it empty when it goes. /// Ordering *between* batches is the consumer's problem. fn flush_if_due(&mut self, buf: &mut Vec, clock: &mut FlushClock) { if !clock.due(buf.len(), self.options.batch.max(1)) { return; } let batch = std::mem::take(buf); // `overflowed` belongs to the pass as a whole, not to one batch; the // final flush reports it. self.flush_pass(batch, false); clock.mark_sent(); } /// Sort a finished pass buffer, truncate to what's left of the display /// limit, and stream it out in `options.batch`-sized events. fn flush_pass(&mut self, mut buf: Vec, overflowed: bool) { buf.sort_by(rank_order); let room = self.remaining(); if buf.len() > room { buf.truncate(room); self.limited = true; } if overflowed { self.limited = true; } self.total += buf.len(); for hit in &buf { self.emitted.insert(hit.file_id); } let batch = self.options.batch.max(1); let mut buf = buf.into_iter().peekable(); while buf.peek().is_some() { // A cancelled search stops emitting immediately — the newer // generation owns the UI. if self.cancelled() { return; } let chunk: Vec = buf.by_ref().take(batch).collect(); (self.sink)(chunk); } } /// Emit a buffer held back from an earlier scan. Anything a better /// stage already emitted drops out here — `emitted` was still empty (or /// smaller) when these hits were collected. fn flush_deferred(&mut self, mut deferred: Deferred) -> Result { deferred.hits.retain(|h| !self.emitted.contains(&h.file_id)); self.flush_pass(deferred.hits, deferred.overflowed); Ok(true) } /// Keep a scan buffer bounded: sort + cut back to the display-limit /// room once it doubles past it. Returns whether anything was dropped. fn enforce_cap(&self, buf: &mut Vec) -> bool { if buf.len() <= self.buffer_cap() { return false; } buf.sort_by(rank_order); buf.truncate(self.remaining()); true } }