Indexing and SQLite optimizations during cold indexing.
This commit is contained in:
parent
3d5aa2d752
commit
ad8ca3d3f2
7 changed files with 693 additions and 98 deletions
69
Cargo.toml
69
Cargo.toml
|
|
@ -4,76 +4,15 @@ members = [
|
||||||
"crates/quicksearch-core",
|
"crates/quicksearch-core",
|
||||||
"crates/quicksearch-gui",
|
"crates/quicksearch-gui",
|
||||||
]
|
]
|
||||||
# The `vendor/` crates are deliberately NOT members: they are third-party code
|
# The `vendor/` crates are deliberately NOT members: they are third-party code which we patched
|
||||||
# carried here for a patch each, not part of this workspace's lints, tests or
|
|
||||||
# release profile. `[patch.crates-io]` below is what makes the dependency graph
|
|
||||||
# resolve to them.
|
|
||||||
exclude = ["vendor/pdf-extract", "vendor/rtf-parser"]
|
exclude = ["vendor/pdf-extract", "vendor/rtf-parser"]
|
||||||
|
|
||||||
# pdf-extract 0.12.0, with the two unbounded recursions in it bounded.
|
|
||||||
#
|
|
||||||
# `get_inherited` follows `/Parent` and `process_stream`'s `Do` arm follows
|
|
||||||
# Form XObjects, neither with a depth counter or a visited set. A page whose
|
|
||||||
# `/Parent` is itself, or an XObject whose content stream draws itself, walks
|
|
||||||
# the stack until it hits the guard page — and a stack overflow is not a panic
|
|
||||||
# that `catch_unwind` can contain (`extract/pdf.rs` has one, for the parser's
|
|
||||||
# ordinary panics): Rust's handler calls `abort()`, so a ~600-byte file kills
|
|
||||||
# the process. It recurs on every run, because the row keeps
|
|
||||||
# `content_state = 0` and the feeder selects exactly those; and the live
|
|
||||||
# watcher re-extracts on-screen rows on the GUI thread, so such a file crashes
|
|
||||||
# the app when it merely appears in a result list.
|
|
||||||
#
|
|
||||||
# Vendored rather than forked-by-URL so the build stays offline, `--locked`
|
|
||||||
# keeps meaning what it means, and the cross-compile job needs no new host.
|
|
||||||
# The patch is marked LOCAL PATCH in the source and is upstreamable; the crate
|
|
||||||
# is MIT and the copy is recorded in `packaging/copyright`.
|
|
||||||
# rtf-parser 0.4.3, with its lexer taught where an RTF control word ends.
|
|
||||||
#
|
|
||||||
# The format's rule is that a control word runs `\` plus letters plus an
|
|
||||||
# optional numeric parameter, and ends at the first character that is neither —
|
|
||||||
# a space if there is one, which is swallowed as the delimiter, otherwise
|
|
||||||
# whatever that character is, which is *not* swallowed. The crate's lexer ends
|
|
||||||
# it at whitespace and nothing else. Two consequences, both of which lose
|
|
||||||
# indexed text silently rather than failing the file:
|
|
||||||
#
|
|
||||||
# * A `\uN` escape is followed by an ANSI fallback character for readers that
|
|
||||||
# predate Unicode, and the spec lets that be any character. `\u233?after`
|
|
||||||
# lexes as one unrecognised control word, so the character *and the rest of
|
|
||||||
# the word* vanish. LibreOffice writes `\uN\'3f` and dodges it; a literal
|
|
||||||
# `?` is just as legal and just as common.
|
|
||||||
# * After a `\'hh` escape the lexer re-tokenises the remainder and trims its
|
|
||||||
# leading spaces before classifying it. A remainder that is plain text
|
|
||||||
# keeps them; one that begins with another escape does not. So two adjacent
|
|
||||||
# words made entirely of escapes come back joined — `Καλημέρα κόσμε` as
|
|
||||||
# `Καλημέρακόσμε`, one FTS term where there were two. That reproduces on a
|
|
||||||
# file LibreOffice wrote, and it hits every script outside cp1252.
|
|
||||||
#
|
|
||||||
# Both were found by `tests/extraction_corpus.rs`, which is also what pins them
|
|
||||||
# fixed. A third patch replaces the two production `unwrap()`s in the parser:
|
|
||||||
# `String::from_utf16` on whatever `\uN` supplied panicked on a lone surrogate,
|
|
||||||
# and RTF is one of the two formats that also extract at *walk* time, where a
|
|
||||||
# panicking worker costs the root its whole content pass. That file now costs
|
|
||||||
# one replacement character instead of the whole document.
|
|
||||||
#
|
|
||||||
# One behaviour change is not a bug fix and is worth knowing about. Fixing the
|
|
||||||
# first bug leaves the ANSI fallback character sitting in the token stream as
|
|
||||||
# ordinary text, so the parser now counts fallbacks off against `\ucN` the way
|
|
||||||
# the specification says, rather than recognising only the `\'hh` spelling by
|
|
||||||
# guesswork. A document that writes `\u233 text` — space delimiter, no
|
|
||||||
# fallback, no `\uc0` — therefore loses the `t`, which is what Word does with
|
|
||||||
# that document too. It used to keep it.
|
|
||||||
#
|
|
||||||
# Vendored for the same reasons pdf-extract is, below: the build stays offline,
|
|
||||||
# `--locked` keeps meaning what it means, and the cross-compile job needs no
|
|
||||||
# new host. 0.4.3 is the latest release, so there is no upgrade to wait for.
|
|
||||||
# The patches are marked LOCAL PATCH in the source and are upstreamable; the
|
|
||||||
# crate is MIT and the copy is recorded in `packaging/copyright`.
|
|
||||||
[patch.crates-io]
|
[patch.crates-io]
|
||||||
pdf-extract = { path = "vendor/pdf-extract" }
|
pdf-extract = { path = "vendor/pdf-extract" } # Patched unbounded reads which could blow up on malformed files
|
||||||
rtf-parser = { path = "vendor/rtf-parser" }
|
rtf-parser = { path = "vendor/rtf-parser" } # Patched a parsing error which occurs on UTF-16 characters
|
||||||
|
|
||||||
[workspace.package]
|
[workspace.package]
|
||||||
version = "1.1.3"
|
version = "1.1.4"
|
||||||
edition = "2021"
|
edition = "2021"
|
||||||
license = "GPL-3.0-or-later"
|
license = "GPL-3.0-or-later"
|
||||||
authors = ["Jeremy <jeremy@karsttech.com>"]
|
authors = ["Jeremy <jeremy@karsttech.com>"]
|
||||||
|
|
|
||||||
|
|
@ -34,7 +34,98 @@
|
||||||
//! progress walk, no size survey — so that every syscall the trace attributes
|
//! progress walk, no size survey — so that every syscall the trace attributes
|
||||||
//! to the tree came from the indexer. The size histogram is printed by `gen`.
|
//! to the tree came from the indexer. The size histogram is printed by `gen`.
|
||||||
|
|
||||||
|
use std::alloc::{GlobalAlloc, Layout, System};
|
||||||
use std::path::{Path, PathBuf};
|
use std::path::{Path, PathBuf};
|
||||||
|
use std::sync::atomic::{AtomicU64, Ordering};
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Allocation accounting
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// `System`, counting. A global allocator is **per binary**, so this affects
|
||||||
|
/// only this probe — the shipped `quicksearch` is untouched.
|
||||||
|
///
|
||||||
|
/// Global atomics rather than the per-thread `Cell`s `tests/search_alloc.rs`
|
||||||
|
/// uses, and for the opposite reason. There the work was synchronous on one
|
||||||
|
/// thread and other *tests* ran concurrently, so per-thread counting was both
|
||||||
|
/// necessary and more precise. Here the work is spread over a walk pool, an
|
||||||
|
/// extraction pool, a feeder and a writer — per-thread counting would report a
|
||||||
|
/// fraction of it — and nothing else is running in this process, so a global
|
||||||
|
/// count is exactly the run.
|
||||||
|
///
|
||||||
|
/// The atomics cost every allocation a contended RMW, which is real overhead
|
||||||
|
/// and shows in the wall-clock line. That is acceptable because both sides of a
|
||||||
|
/// before/after comparison carry the same instrumentation; it is not acceptable
|
||||||
|
/// to quote these timings against numbers from an uninstrumented build.
|
||||||
|
struct Counting;
|
||||||
|
|
||||||
|
static ALLOCS: AtomicU64 = AtomicU64::new(0);
|
||||||
|
static ALLOC_BYTES: AtomicU64 = AtomicU64::new(0);
|
||||||
|
static LIVE: AtomicU64 = AtomicU64::new(0);
|
||||||
|
static PEAK_LIVE: AtomicU64 = AtomicU64::new(0);
|
||||||
|
|
||||||
|
#[inline]
|
||||||
|
fn note_alloc(size: usize) {
|
||||||
|
ALLOCS.fetch_add(1, Ordering::Relaxed);
|
||||||
|
ALLOC_BYTES.fetch_add(size as u64, Ordering::Relaxed);
|
||||||
|
let live = LIVE.fetch_add(size as u64, Ordering::Relaxed) + size as u64;
|
||||||
|
PEAK_LIVE.fetch_max(live, Ordering::Relaxed);
|
||||||
|
}
|
||||||
|
|
||||||
|
unsafe impl GlobalAlloc for Counting {
|
||||||
|
unsafe fn alloc(&self, l: Layout) -> *mut u8 {
|
||||||
|
let p = unsafe { System.alloc(l) };
|
||||||
|
if !p.is_null() {
|
||||||
|
note_alloc(l.size());
|
||||||
|
}
|
||||||
|
p
|
||||||
|
}
|
||||||
|
unsafe fn alloc_zeroed(&self, l: Layout) -> *mut u8 {
|
||||||
|
let p = unsafe { System.alloc_zeroed(l) };
|
||||||
|
if !p.is_null() {
|
||||||
|
note_alloc(l.size());
|
||||||
|
}
|
||||||
|
p
|
||||||
|
}
|
||||||
|
unsafe fn dealloc(&self, p: *mut u8, l: Layout) {
|
||||||
|
LIVE.fetch_sub(l.size() as u64, Ordering::Relaxed);
|
||||||
|
unsafe { System.dealloc(p, l) }
|
||||||
|
}
|
||||||
|
unsafe fn realloc(&self, p: *mut u8, l: Layout, new: usize) -> *mut u8 {
|
||||||
|
let q = unsafe { System.realloc(p, l, new) };
|
||||||
|
if !q.is_null() {
|
||||||
|
let (old, new) = (l.size() as u64, new as u64);
|
||||||
|
ALLOC_BYTES.fetch_add(new.saturating_sub(old), Ordering::Relaxed);
|
||||||
|
let live = if new >= old {
|
||||||
|
LIVE.fetch_add(new - old, Ordering::Relaxed) + (new - old)
|
||||||
|
} else {
|
||||||
|
LIVE.fetch_sub(old - new, Ordering::Relaxed) - (old - new)
|
||||||
|
};
|
||||||
|
PEAK_LIVE.fetch_max(live, Ordering::Relaxed);
|
||||||
|
}
|
||||||
|
q
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[global_allocator]
|
||||||
|
static ALLOCATOR: Counting = Counting;
|
||||||
|
|
||||||
|
/// Peak resident set size, from the kernel's own high-water mark. Unlike a
|
||||||
|
/// sampled figure this cannot miss a spike.
|
||||||
|
fn vm_hwm_bytes() -> u64 {
|
||||||
|
std::fs::read_to_string("/proc/self/status")
|
||||||
|
.ok()
|
||||||
|
.and_then(|s| {
|
||||||
|
s.lines()
|
||||||
|
.find(|l| l.starts_with("VmHWM:"))?
|
||||||
|
.split_whitespace()
|
||||||
|
.nth(1)?
|
||||||
|
.parse::<u64>()
|
||||||
|
.ok()
|
||||||
|
})
|
||||||
|
.map(|kib| kib * 1024)
|
||||||
|
.unwrap_or(0)
|
||||||
|
}
|
||||||
use std::time::{Duration, Instant};
|
use std::time::{Duration, Instant};
|
||||||
|
|
||||||
use quicksearch_core::config::Config;
|
use quicksearch_core::config::Config;
|
||||||
|
|
@ -49,6 +140,24 @@ const LARGE_TEXT: usize = 100;
|
||||||
/// the disk. A control group: their cost must not move.
|
/// the disk. A control group: their cost must not move.
|
||||||
const BINARY: usize = 100;
|
const BINARY: usize = 100;
|
||||||
|
|
||||||
|
/// Scale the generated tree by an integer factor (`QSB_SCALE`), keeping the
|
||||||
|
/// mix between the three groups fixed.
|
||||||
|
///
|
||||||
|
/// The default thousand files is enough to exercise every code path and far
|
||||||
|
/// too few to measure any of them: a run that size is dominated by fixed
|
||||||
|
/// start-up — opening the index, the config reconcile — and its per-file
|
||||||
|
/// figures carry the whole of SQLite's and FTS5's fixed structure spread over
|
||||||
|
/// a thousand rows. Anything claiming to be a per-file cost needs a tree where
|
||||||
|
/// the fixed part has been amortised away, and the difference between two
|
||||||
|
/// scales is the only way to tell the two apart.
|
||||||
|
fn scale() -> usize {
|
||||||
|
std::env::var("QSB_SCALE")
|
||||||
|
.ok()
|
||||||
|
.and_then(|v| v.parse().ok())
|
||||||
|
.filter(|n| *n >= 1)
|
||||||
|
.unwrap_or(1)
|
||||||
|
}
|
||||||
|
|
||||||
const WORDS: &[&str] = &[
|
const WORDS: &[&str] = &[
|
||||||
"alpha",
|
"alpha",
|
||||||
"beta",
|
"beta",
|
||||||
|
|
@ -144,11 +253,13 @@ fn generate(tree: &Path) {
|
||||||
|
|
||||||
let mut rng = Rng(0x5eed);
|
let mut rng = Rng(0x5eed);
|
||||||
let (mut small_bytes, mut large_bytes, mut bin_bytes) = (0usize, 0usize, 0usize);
|
let (mut small_bytes, mut large_bytes, mut bin_bytes) = (0usize, 0usize, 0usize);
|
||||||
|
let scale = scale();
|
||||||
|
let (small_text, large_text, binary) = (SMALL_TEXT * scale, LARGE_TEXT * scale, BINARY * scale);
|
||||||
|
|
||||||
// Spread across subdirectories so the walk does real directory work
|
// Spread across subdirectories so the walk does real directory work
|
||||||
// rather than one enormous readdir.
|
// rather than one enormous readdir.
|
||||||
for i in 0..SMALL_TEXT {
|
for i in 0..small_text {
|
||||||
let dir = tree.join(format!("src/mod{}", i % 40));
|
let dir = tree.join(format!("src/mod{}", i % (40 * scale)));
|
||||||
std::fs::create_dir_all(&dir).expect("mkdir");
|
std::fs::create_dir_all(&dir).expect("mkdir");
|
||||||
let ext = ["txt", "md", "rs", "json"][i % 4];
|
let ext = ["txt", "md", "rs", "json"][i % 4];
|
||||||
let size = rng.in_range(200, 8 * 1024);
|
let size = rng.in_range(200, 8 * 1024);
|
||||||
|
|
@ -157,8 +268,8 @@ fn generate(tree: &Path) {
|
||||||
std::fs::write(dir.join(format!("f{}.{}", i, ext)), body).expect("write");
|
std::fs::write(dir.join(format!("f{}.{}", i, ext)), body).expect("write");
|
||||||
}
|
}
|
||||||
|
|
||||||
for i in 0..LARGE_TEXT {
|
for i in 0..large_text {
|
||||||
let dir = tree.join(format!("docs/set{}", i % 10));
|
let dir = tree.join(format!("docs/set{}", i % (10 * scale)));
|
||||||
std::fs::create_dir_all(&dir).expect("mkdir");
|
std::fs::create_dir_all(&dir).expect("mkdir");
|
||||||
let size = rng.in_range(8 * 1024 + 1, 200 * 1024);
|
let size = rng.in_range(8 * 1024 + 1, 200 * 1024);
|
||||||
let body = prose(&mut rng, size);
|
let body = prose(&mut rng, size);
|
||||||
|
|
@ -166,8 +277,8 @@ fn generate(tree: &Path) {
|
||||||
std::fs::write(dir.join(format!("doc{}.md", i)), body).expect("write");
|
std::fs::write(dir.join(format!("doc{}.md", i)), body).expect("write");
|
||||||
}
|
}
|
||||||
|
|
||||||
for i in 0..BINARY {
|
for i in 0..binary {
|
||||||
let dir = tree.join(format!("assets/set{}", i % 10));
|
let dir = tree.join(format!("assets/set{}", i % (10 * scale)));
|
||||||
std::fs::create_dir_all(&dir).expect("mkdir");
|
std::fs::create_dir_all(&dir).expect("mkdir");
|
||||||
let n = rng.in_range(1024, 50 * 1024);
|
let n = rng.in_range(1024, 50 * 1024);
|
||||||
let blob: Vec<u8> = (0..n).map(|_| (rng.next() & 0xff) as u8).collect();
|
let blob: Vec<u8> = (0..n).map(|_| (rng.next() & 0xff) as u8).collect();
|
||||||
|
|
@ -175,21 +286,21 @@ fn generate(tree: &Path) {
|
||||||
std::fs::write(dir.join(format!("blob{}.bin", i)), blob).expect("write");
|
std::fs::write(dir.join(format!("blob{}.bin", i)), blob).expect("write");
|
||||||
}
|
}
|
||||||
|
|
||||||
let total = SMALL_TEXT + LARGE_TEXT + BINARY;
|
let total = small_text + large_text + binary;
|
||||||
eprintln!("generated {} files under {}", total, tree.display());
|
eprintln!("generated {} files under {}", total, tree.display());
|
||||||
eprintln!(
|
eprintln!(
|
||||||
" text <= 8 KiB : {:5} files, {:8.1} MiB (head covers the whole file)",
|
" text <= 8 KiB : {:5} files, {:8.1} MiB (head covers the whole file)",
|
||||||
SMALL_TEXT,
|
small_text,
|
||||||
small_bytes as f64 / (1024.0 * 1024.0)
|
small_bytes as f64 / (1024.0 * 1024.0)
|
||||||
);
|
);
|
||||||
eprintln!(
|
eprintln!(
|
||||||
" text > 8 KiB : {:5} files, {:8.1} MiB (extraction must read it)",
|
" text > 8 KiB : {:5} files, {:8.1} MiB (extraction must read it)",
|
||||||
LARGE_TEXT,
|
large_text,
|
||||||
large_bytes as f64 / (1024.0 * 1024.0)
|
large_bytes as f64 / (1024.0 * 1024.0)
|
||||||
);
|
);
|
||||||
eprintln!(
|
eprintln!(
|
||||||
" binary : {:5} files, {:8.1} MiB (no extractor; control group)",
|
" binary : {:5} files, {:8.1} MiB (no extractor; control group)",
|
||||||
BINARY,
|
binary,
|
||||||
bin_bytes as f64 / (1024.0 * 1024.0)
|
bin_bytes as f64 / (1024.0 * 1024.0)
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
@ -208,6 +319,135 @@ fn prose(rng: &mut Rng, target: usize) -> String {
|
||||||
s
|
s
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The kernel's own accounting for this process, from `/proc/self/io`.
|
||||||
|
///
|
||||||
|
/// `read_bytes`/`write_bytes` are what actually reached the block layer, so
|
||||||
|
/// they are the figures that describe the *disk* rather than the page cache —
|
||||||
|
/// a warm re-read shows as `rchar` without moving `read_bytes`. `syscr`/`syscw`
|
||||||
|
/// count the calls regardless, which is what separates "we read a lot" from
|
||||||
|
/// "we read a little, many times".
|
||||||
|
///
|
||||||
|
/// Zero everywhere on a filesystem that does not report it (virtiofs, some
|
||||||
|
/// network mounts); the caller says so rather than printing a confident 0.
|
||||||
|
#[derive(Default, Clone, Copy)]
|
||||||
|
struct Io {
|
||||||
|
rchar: u64,
|
||||||
|
wchar: u64,
|
||||||
|
syscr: u64,
|
||||||
|
syscw: u64,
|
||||||
|
read_bytes: u64,
|
||||||
|
write_bytes: u64,
|
||||||
|
cancelled: u64,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Io {
|
||||||
|
fn read() -> Io {
|
||||||
|
let mut io = Io::default();
|
||||||
|
let Ok(text) = std::fs::read_to_string("/proc/self/io") else {
|
||||||
|
return io;
|
||||||
|
};
|
||||||
|
for line in text.lines() {
|
||||||
|
let Some((key, value)) = line.split_once(':') else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
let Ok(value) = value.trim().parse::<u64>() else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
match key {
|
||||||
|
"rchar" => io.rchar = value,
|
||||||
|
"wchar" => io.wchar = value,
|
||||||
|
"syscr" => io.syscr = value,
|
||||||
|
"syscw" => io.syscw = value,
|
||||||
|
"read_bytes" => io.read_bytes = value,
|
||||||
|
"write_bytes" => io.write_bytes = value,
|
||||||
|
"cancelled_write_bytes" => io.cancelled = value,
|
||||||
|
_ => {}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
io
|
||||||
|
}
|
||||||
|
|
||||||
|
fn since(&self, start: &Io) -> Io {
|
||||||
|
Io {
|
||||||
|
rchar: self.rchar.saturating_sub(start.rchar),
|
||||||
|
wchar: self.wchar.saturating_sub(start.wchar),
|
||||||
|
syscr: self.syscr.saturating_sub(start.syscr),
|
||||||
|
syscw: self.syscw.saturating_sub(start.syscw),
|
||||||
|
read_bytes: self.read_bytes.saturating_sub(start.read_bytes),
|
||||||
|
write_bytes: self.write_bytes.saturating_sub(start.write_bytes),
|
||||||
|
cancelled: self.cancelled.saturating_sub(start.cancelled),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn mib(bytes: u64) -> String {
|
||||||
|
format!("{:.1} MiB", bytes as f64 / (1024.0 * 1024.0))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// What the write-ahead log did during a run, sampled from outside the process.
|
||||||
|
///
|
||||||
|
/// The interesting part of write amplification is not the total — that is one
|
||||||
|
/// number from `/proc/self/io` — but how it splits between **frames appended to
|
||||||
|
/// the log** and **pages copied back into the database** by a checkpoint. The
|
||||||
|
/// two want opposite fixes: more frames means the load is rewriting pages, and
|
||||||
|
/// more copy-back means it is checkpointing too often. A page rewritten five
|
||||||
|
/// times between two checkpoints costs five frames and *one* copy-back, so
|
||||||
|
/// checkpointing less often can be strictly cheaper — which is the opposite of
|
||||||
|
/// what "keep the log small" suggests.
|
||||||
|
///
|
||||||
|
/// Sampled rather than instrumented: the log is a file, its size is a `stat`,
|
||||||
|
/// and a checkpoint truncates it. Growth between samples is frames appended; a
|
||||||
|
/// drop is a checkpoint, and the size it dropped *from* bounds what that
|
||||||
|
/// checkpoint copied. Nothing in the library has to know it is being watched.
|
||||||
|
#[derive(Default, Clone, Copy)]
|
||||||
|
struct WalStats {
|
||||||
|
/// Largest the log ever got.
|
||||||
|
peak: u64,
|
||||||
|
/// Sum of every increase — bytes appended to the log over the run.
|
||||||
|
appended: u64,
|
||||||
|
/// Sum of the size before each truncation — an upper bound on the bytes
|
||||||
|
/// each checkpoint wrote back into the database.
|
||||||
|
copied_back: u64,
|
||||||
|
checkpoints: u64,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Watch `path` until `stop` is set, at `SAMPLE`.
|
||||||
|
///
|
||||||
|
/// One millisecond, because a checkpoint of a small log is quick and a sampler
|
||||||
|
/// that misses the rise and the fall reports neither. It costs one `stat` per
|
||||||
|
/// millisecond, which is nothing next to what is being measured.
|
||||||
|
fn sample_wal(path: PathBuf, stop: std::sync::Arc<std::sync::atomic::AtomicBool>) -> std::thread::JoinHandle<WalStats> {
|
||||||
|
const SAMPLE: Duration = Duration::from_millis(1);
|
||||||
|
std::thread::spawn(move || {
|
||||||
|
let mut stats = WalStats::default();
|
||||||
|
let mut last = 0u64;
|
||||||
|
while !stop.load(Ordering::Relaxed) {
|
||||||
|
let now = std::fs::metadata(&path).map(|m| m.len()).unwrap_or(0);
|
||||||
|
if now > last {
|
||||||
|
stats.appended += now - last;
|
||||||
|
} else if now < last {
|
||||||
|
// A shrink is a checkpoint landing the log. `last` is the most
|
||||||
|
// recent size seen before it, so it bounds the copy-back.
|
||||||
|
stats.checkpoints += 1;
|
||||||
|
stats.copied_back += last;
|
||||||
|
}
|
||||||
|
stats.peak = stats.peak.max(now);
|
||||||
|
last = now;
|
||||||
|
std::thread::sleep(SAMPLE);
|
||||||
|
}
|
||||||
|
stats
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Size of the index and the sidecars it leaves behind.
|
||||||
|
fn db_sizes(db: &Path) -> (u64, u64) {
|
||||||
|
let len = |p: PathBuf| std::fs::metadata(p).map(|m| m.len()).unwrap_or(0);
|
||||||
|
(
|
||||||
|
len(db.to_path_buf()),
|
||||||
|
len(PathBuf::from(format!("{}-wal", db.display()))),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
fn run(mode: &str, tree: &Path, db: &Path) {
|
fn run(mode: &str, tree: &Path, db: &Path) {
|
||||||
let config = Config::default();
|
let config = Config::default();
|
||||||
|
|
||||||
|
|
@ -221,6 +461,15 @@ fn run(mode: &str, tree: &Path, db: &Path) {
|
||||||
.expect("clear marker");
|
.expect("clear marker");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Cleared so the phase summaries below belong to this run alone.
|
||||||
|
quicksearch_core::log::clear();
|
||||||
|
let io_start = Io::read();
|
||||||
|
let (db_before, wal_before) = db_sizes(db);
|
||||||
|
|
||||||
|
let wal_path = PathBuf::from(format!("{}-wal", db.display()));
|
||||||
|
let wal_stop = std::sync::Arc::new(std::sync::atomic::AtomicBool::new(false));
|
||||||
|
let wal_sampler = sample_wal(wal_path, wal_stop.clone());
|
||||||
|
|
||||||
let service = IndexingService::new();
|
let service = IndexingService::new();
|
||||||
let start = Instant::now();
|
let start = Instant::now();
|
||||||
service
|
service
|
||||||
|
|
@ -249,7 +498,10 @@ fn run(mode: &str, tree: &Path, db: &Path) {
|
||||||
}
|
}
|
||||||
let elapsed = start.elapsed();
|
let elapsed = start.elapsed();
|
||||||
assert!(done, "indexing did not finish within the timeout");
|
assert!(done, "indexing did not finish within the timeout");
|
||||||
|
// The run's last checkpoint happens inside here, so the sampler outlives it.
|
||||||
service.stop_indexing().expect("stop");
|
service.stop_indexing().expect("stop");
|
||||||
|
wal_stop.store(true, Ordering::Relaxed);
|
||||||
|
let wal = wal_sampler.join().unwrap_or_default();
|
||||||
|
|
||||||
// Count what was actually indexed rather than assuming `gen`'s tree.
|
// Count what was actually indexed rather than assuming `gen`'s tree.
|
||||||
// The constants describe the tree this probe builds; pointing it at any
|
// The constants describe the tree this probe builds; pointing it at any
|
||||||
|
|
@ -258,11 +510,88 @@ fn run(mode: &str, tree: &Path, db: &Path) {
|
||||||
.ok()
|
.ok()
|
||||||
.and_then(|c| quicksearch_core::db::repo::row_count(&c).ok())
|
.and_then(|c| quicksearch_core::db::repo::row_count(&c).ok())
|
||||||
.unwrap_or(0);
|
.unwrap_or(0);
|
||||||
|
|
||||||
|
// Read after `stop_indexing`, so the optimize pass's checkpoint — which is
|
||||||
|
// where a run's dirty pages actually reach the file — is inside the totals.
|
||||||
|
let io = Io::read().since(&io_start);
|
||||||
|
let (db_after, wal_after) = db_sizes(db);
|
||||||
|
let per_file = |n: u64| {
|
||||||
|
if total == 0 {
|
||||||
|
"-".to_string()
|
||||||
|
} else {
|
||||||
|
format!("{:.0} B/file", n as f64 / total as f64)
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
eprintln!(
|
eprintln!(
|
||||||
"{}: {:?} ({:.0} files/sec over {} files)",
|
"\n{}: {:?} ({:.0} files/sec over {} files)",
|
||||||
mode,
|
mode,
|
||||||
elapsed,
|
elapsed,
|
||||||
total as f64 / elapsed.as_secs_f64(),
|
total as f64 / elapsed.as_secs_f64(),
|
||||||
total
|
total
|
||||||
);
|
);
|
||||||
|
|
||||||
|
// The pipeline logs one line per root per phase; they are the walk/extract
|
||||||
|
// split without a `perf` session.
|
||||||
|
for line in quicksearch_core::log::snapshot() {
|
||||||
|
let m = &line.text;
|
||||||
|
if m.contains("walk done")
|
||||||
|
|| m.contains("walk ended early")
|
||||||
|
|| m.contains("content done")
|
||||||
|
|| m.contains("stale cleanup")
|
||||||
|
|| m.contains("indexing complete")
|
||||||
|
{
|
||||||
|
eprintln!(" phase {}", m);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
let allocs = ALLOCS.load(Ordering::Relaxed);
|
||||||
|
eprintln!(
|
||||||
|
" wal peak {}, {} appended, {} copied back over {} checkpoint(s)",
|
||||||
|
mib(wal.peak),
|
||||||
|
mib(wal.appended),
|
||||||
|
mib(wal.copied_back),
|
||||||
|
wal.checkpoints,
|
||||||
|
);
|
||||||
|
eprintln!(
|
||||||
|
" memory {} allocations ({:.1} per file), {} churned, peak live {}, VmHWM {}",
|
||||||
|
allocs,
|
||||||
|
allocs as f64 / total.max(1) as f64,
|
||||||
|
mib(ALLOC_BYTES.load(Ordering::Relaxed)),
|
||||||
|
mib(PEAK_LIVE.load(Ordering::Relaxed)),
|
||||||
|
mib(vm_hwm_bytes()),
|
||||||
|
);
|
||||||
|
eprintln!(
|
||||||
|
" index {} -> {} wal {} -> {}",
|
||||||
|
mib(db_before),
|
||||||
|
mib(db_after),
|
||||||
|
mib(wal_before),
|
||||||
|
mib(wal_after),
|
||||||
|
);
|
||||||
|
eprintln!(
|
||||||
|
" syscall {} reads, {} writes ({:.1} reads/file, {:.1} writes/file)",
|
||||||
|
io.syscr,
|
||||||
|
io.syscw,
|
||||||
|
io.syscr as f64 / total.max(1) as f64,
|
||||||
|
io.syscw as f64 / total.max(1) as f64,
|
||||||
|
);
|
||||||
|
eprintln!(
|
||||||
|
" bytes rchar {} / wchar {} (through the syscall layer, cache included)",
|
||||||
|
mib(io.rchar),
|
||||||
|
mib(io.wchar),
|
||||||
|
);
|
||||||
|
if io.read_bytes == 0 && io.write_bytes == 0 {
|
||||||
|
eprintln!(
|
||||||
|
" disk not reported for this filesystem (virtiofs/tmpfs); \
|
||||||
|
use rchar/wchar and the index sizes above"
|
||||||
|
);
|
||||||
|
} else {
|
||||||
|
eprintln!(
|
||||||
|
" disk read {} / written {} (cancelled {}) -> {} written",
|
||||||
|
mib(io.read_bytes),
|
||||||
|
mib(io.write_bytes),
|
||||||
|
mib(io.cancelled),
|
||||||
|
per_file(io.write_bytes.saturating_sub(io.cancelled)),
|
||||||
|
);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
134
crates/quicksearch-core/examples/searchtime.rs
Normal file
134
crates/quicksearch-core/examples/searchtime.rs
Normal file
|
|
@ -0,0 +1,134 @@
|
||||||
|
//! Search latency against an index that already exists on disk.
|
||||||
|
//!
|
||||||
|
//! The counterweight to `indexprobe`. Every FTS write-side knob — `automerge`,
|
||||||
|
//! a final `'optimize'`, `pgsz` — buys indexing time by leaving more segments
|
||||||
|
//! behind, and a segment is a b-tree a query has to visit. Halving a cold index
|
||||||
|
//! while doubling a keystroke is a regression wearing an improvement's clothes,
|
||||||
|
//! and this is what says which one happened.
|
||||||
|
//!
|
||||||
|
//! ```text
|
||||||
|
//! cargo build -p quicksearch-core --example searchtime --release
|
||||||
|
//! ./target/release/examples/searchtime /path/to/index.db
|
||||||
|
//! ```
|
||||||
|
//!
|
||||||
|
//! Queries are run against one held connection, as the search worker holds one
|
||||||
|
//! across a typing session, and each is timed best-of-N so a single scheduling
|
||||||
|
//! hiccup does not become the headline.
|
||||||
|
|
||||||
|
use std::path::PathBuf;
|
||||||
|
use std::sync::atomic::AtomicU64;
|
||||||
|
use std::time::{Duration, Instant};
|
||||||
|
|
||||||
|
use quicksearch_core::query::split::split_for_cascade;
|
||||||
|
use quicksearch_core::search::{cascade, SearchHit, SearchOptions};
|
||||||
|
|
||||||
|
/// `(read_bytes, rchar)` from `/proc/self/io`: what reached the block layer,
|
||||||
|
/// and what passed through the read syscalls. A query whose time varies while
|
||||||
|
/// `rchar` does not is not doing more work — it is waiting on the disk.
|
||||||
|
fn proc_io() -> (u64, u64) {
|
||||||
|
let text = std::fs::read_to_string("/proc/self/io").unwrap_or_default();
|
||||||
|
let field = |key: &str| -> u64 {
|
||||||
|
text.lines()
|
||||||
|
.find_map(|l| l.strip_prefix(key)?.trim().trim_start_matches(':').trim().parse().ok())
|
||||||
|
.unwrap_or(0)
|
||||||
|
};
|
||||||
|
(field("read_bytes"), field("rchar"))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Runs per query; the best is reported.
|
||||||
|
const RUNS: u32 = 5;
|
||||||
|
|
||||||
|
/// The query set, chosen to reach the passes an FTS setting can affect.
|
||||||
|
///
|
||||||
|
/// The content queries are the point — they are the ones that go through
|
||||||
|
/// `searchabletext` and therefore through however many segments the write side
|
||||||
|
/// left behind. The filename query is the control: it never touches FTS, so it
|
||||||
|
/// must not move.
|
||||||
|
const QUERIES: &[(&str, bool, &str)] = &[
|
||||||
|
("filename (control)", false, "doc42"),
|
||||||
|
("content, common", false, "mountain"),
|
||||||
|
("content, rare", false, "quartzite"),
|
||||||
|
("content, two words", false, "ocean forest"),
|
||||||
|
("fuzzy content", true, "mountian"),
|
||||||
|
];
|
||||||
|
|
||||||
|
fn main() {
|
||||||
|
let db = PathBuf::from(
|
||||||
|
std::env::args()
|
||||||
|
.nth(1)
|
||||||
|
.expect("usage: searchtime <index.db>"),
|
||||||
|
);
|
||||||
|
let conn = quicksearch_core::db::open::open_search_reader(&db.to_string_lossy())
|
||||||
|
.expect("open the index");
|
||||||
|
// Override the profile's ceiling, to test whether a slow index is slow
|
||||||
|
// because its working set does not fit rather than because it is bigger.
|
||||||
|
if let Ok(kib) = std::env::var("QSB_CACHE_KIB") {
|
||||||
|
conn.execute_batch(&format!("PRAGMA cache_size = -{};", kib.trim()))
|
||||||
|
.expect("set cache_size");
|
||||||
|
}
|
||||||
|
|
||||||
|
let segments: i64 = conn
|
||||||
|
.query_row("SELECT COUNT(*) FROM searchabletext_idx", [], |r| r.get(0))
|
||||||
|
.unwrap_or(-1);
|
||||||
|
let rows: i64 = conn
|
||||||
|
.query_row("SELECT COUNT(*) FROM files", [], |r| r.get(0))
|
||||||
|
.unwrap_or(-1);
|
||||||
|
println!(
|
||||||
|
"{} ({} rows, {} segment-index entries)",
|
||||||
|
db.display(),
|
||||||
|
rows,
|
||||||
|
segments
|
||||||
|
);
|
||||||
|
println!("{:<22} {:>10} {:>8}", "query", "best", "hits");
|
||||||
|
|
||||||
|
let mut total = Duration::ZERO;
|
||||||
|
for (label, fuzzy, query) in QUERIES {
|
||||||
|
let split = split_for_cascade(query).expect("query parses");
|
||||||
|
// The display limit makes this benchmark unfair between indexes.
|
||||||
|
// `scan_pass` stops as soon as the limit is full, and it streams FTS
|
||||||
|
// candidates in rowid order — which is `file_id` order, which is the
|
||||||
|
// order the *walk* happened to insert rows. So an index where the large
|
||||||
|
// documents drew low ids decompresses megabytes to fill 1000 hits while
|
||||||
|
// one where the small documents did reads a few hundred kilobytes, and
|
||||||
|
// the two differ by 13x for reasons that have nothing to do with what is
|
||||||
|
// being compared. Measured: 2.4 MiB against 110.9 MiB of `rchar` for the
|
||||||
|
// same query and the same hit count.
|
||||||
|
//
|
||||||
|
// Raising the limit past the corpus makes every index examine every
|
||||||
|
// candidate, which is the only way two of them are doing equal work.
|
||||||
|
let limit = std::env::var("QSB_LIMIT")
|
||||||
|
.ok()
|
||||||
|
.and_then(|v| v.parse().ok())
|
||||||
|
.unwrap_or(1000);
|
||||||
|
let options = SearchOptions {
|
||||||
|
fuzzy: *fuzzy,
|
||||||
|
limit,
|
||||||
|
..SearchOptions::default()
|
||||||
|
};
|
||||||
|
let mut best = Duration::MAX;
|
||||||
|
let mut hits = 0usize;
|
||||||
|
let io_before = proc_io();
|
||||||
|
for _ in 0..RUNS {
|
||||||
|
let latest = AtomicU64::new(1);
|
||||||
|
let mut count = 0usize;
|
||||||
|
let mut sink = |h: Vec<SearchHit>| count += h.len();
|
||||||
|
let start = Instant::now();
|
||||||
|
cascade::run(&conn, &split, &options, 1, &latest, &mut sink).expect("cascade runs");
|
||||||
|
best = best.min(start.elapsed());
|
||||||
|
hits = count;
|
||||||
|
}
|
||||||
|
total += best;
|
||||||
|
let (rd, rc) = proc_io();
|
||||||
|
let (rd0, rc0) = io_before;
|
||||||
|
println!(
|
||||||
|
"{:<22} {:>10.1?} {:>8} disk-read {:>8.1} MiB rchar {:>8.1} MiB (over {} runs)",
|
||||||
|
label,
|
||||||
|
best,
|
||||||
|
hits,
|
||||||
|
(rd - rd0) as f64 / 1048576.0,
|
||||||
|
(rc - rc0) as f64 / 1048576.0,
|
||||||
|
RUNS,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
println!("{:<22} {:>10.1?}", "TOTAL", total);
|
||||||
|
}
|
||||||
|
|
@ -72,9 +72,9 @@ struct Shared {
|
||||||
queue: Mutex<Queue>,
|
queue: Mutex<Queue>,
|
||||||
idle: Condvar,
|
idle: Condvar,
|
||||||
/// What the range held when the pass began: rows still to extract and
|
/// What the range held when the pass began: rows still to extract and
|
||||||
/// rows already done. Set by the feeder before it pages anything, so
|
/// rows already done. Set by the feeder just *behind* its first page, so
|
||||||
/// `already_done + rows written this pass` stays exact; never set if the
|
/// the pool is never blocked on the scan that produces it; never set if the
|
||||||
/// feeder could not count.
|
/// feeder could not count. See `feeder` for what deferring it costs.
|
||||||
totals: std::sync::OnceLock<ExtractScope>,
|
totals: std::sync::OnceLock<ExtractScope>,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -173,9 +173,10 @@ impl ContentPass {
|
||||||
/// The range's pending and already-done counts as they stood when the
|
/// The range's pending and already-done counts as they stood when the
|
||||||
/// pass began.
|
/// pass began.
|
||||||
///
|
///
|
||||||
/// `None` until the feeder has counted — a scan that takes seconds on a
|
/// `None` until the feeder has counted — a scan measured at 513 ms over a
|
||||||
/// large root, which is why it happens here on the pass's own connection
|
/// million rows, which is why it happens on the pass's own connection
|
||||||
/// and not on the indexer's writer — and forever if it could not.
|
/// rather than the indexer's writer, and behind the first page rather than
|
||||||
|
/// in front of the pool — and forever if it could not.
|
||||||
pub fn totals(&self) -> Option<ExtractScope> {
|
pub fn totals(&self) -> Option<ExtractScope> {
|
||||||
self.shared.totals.get().copied()
|
self.shared.totals.get().copied()
|
||||||
}
|
}
|
||||||
|
|
@ -224,16 +225,36 @@ fn feeder(shared: &Shared, db_path: &str, mut cursor: ExtractCursor, config: &Co
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
// Before the first page, so nothing this pass writes is inside the count.
|
// The count happens *behind* the first page, not in front of it.
|
||||||
// The workers cannot run ahead of this: they block in `take` until the
|
//
|
||||||
// first page lands. A failure here costs the progress figure, not the
|
// It used to run here, before anything was fetched, and the workers block
|
||||||
// pass.
|
// in `take` until a page lands — so every thread in the pool sat idle for
|
||||||
match crate::file_handling::count_extract_scope(&conn, &cursor, config) {
|
// the whole of it, to compute a progress-bar denominator. That is not
|
||||||
Ok(totals) => {
|
// free: the scan walks the root's entire parent range fetching a row per
|
||||||
let _ = shared.totals.set(totals);
|
// entry, measured at 20 ms over 100,000 rows and **513 ms over a million**
|
||||||
|
// with the index already in cache, and the count was moved onto this
|
||||||
|
// connection in the first place because on a large root it takes seconds
|
||||||
|
// cold. The move took it off the writer and left the stall one level down.
|
||||||
|
//
|
||||||
|
// What it costs to defer: rows this pass writes during the count can be
|
||||||
|
// seen by it as `already_done` rather than `pending`. The two move in
|
||||||
|
// opposite directions and `extract_total` is their **sum**, so the
|
||||||
|
// denominator is unaffected; only the numerator can run briefly ahead of
|
||||||
|
// itself, which is a shape `RootProgress` already reports and deliberately
|
||||||
|
// does not clamp — see the note on `snapshot`.
|
||||||
|
let mut counted = false;
|
||||||
|
let mut count_now = |conn: &rusqlite::Connection, cursor: &ExtractCursor| {
|
||||||
|
if counted {
|
||||||
|
return;
|
||||||
}
|
}
|
||||||
Err(e) => crate::log_warn!("content reader: {}", e),
|
counted = true;
|
||||||
}
|
match crate::file_handling::count_extract_scope(conn, cursor, config) {
|
||||||
|
Ok(totals) => {
|
||||||
|
let _ = shared.totals.set(totals);
|
||||||
|
}
|
||||||
|
Err(e) => crate::log_warn!("content reader: {}", e),
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
let max_size = crate::file_handling::max_text_file_size(config);
|
let max_size = crate::file_handling::max_text_file_size(config);
|
||||||
while shared.take_feed_slot().is_some() {
|
while shared.take_feed_slot().is_some() {
|
||||||
|
|
@ -261,10 +282,16 @@ fn feeder(shared: &Shared, db_path: &str, mut cursor: ExtractCursor, config: &Co
|
||||||
})
|
})
|
||||||
.collect();
|
.collect();
|
||||||
shared.finish_feed(rows, last_page);
|
shared.finish_feed(rows, last_page);
|
||||||
|
// The pool is running now; the denominator can be worked out behind it.
|
||||||
|
count_now(&conn, &cursor);
|
||||||
if last_page {
|
if last_page {
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
// A range whose first `take_feed_slot` said the pass was already over
|
||||||
|
// still deserves its figure — `an_empty_range_terminates_immediately`
|
||||||
|
// pins that an empty root reports a known zero rather than an unknown.
|
||||||
|
count_now(&conn, &cursor);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn worker(
|
fn worker(
|
||||||
|
|
|
||||||
|
|
@ -27,7 +27,8 @@ pub(crate) use paths::{
|
||||||
};
|
};
|
||||||
pub use records::{
|
pub use records::{
|
||||||
classify_by_mtime, classify_for_indexing, content_extractable, decide_content,
|
classify_by_mtime, classify_for_indexing, content_extractable, decide_content,
|
||||||
extract_and_store, fts_finalize_after_text_indexing, hash_failure_counts, outcome_body,
|
extract_and_store, fts_begin_bulk_write, fts_finalize_after_text_indexing,
|
||||||
|
fts_set_automerge, hash_failure_counts, outcome_body,
|
||||||
prepare_file_record, prepare_file_record_from_path, reset_run_warnings, store_content_outcome,
|
prepare_file_record, prepare_file_record_from_path, reset_run_warnings, store_content_outcome,
|
||||||
ContentOutcome, DirRows, FileIndexAction, OwnedNewFile,
|
ContentOutcome, DirRows, FileIndexAction, OwnedNewFile,
|
||||||
};
|
};
|
||||||
|
|
|
||||||
|
|
@ -92,17 +92,120 @@ pub(super) fn get_file_hash(
|
||||||
Ok((hasher.finalize().to_vec(), head))
|
Ok((hasher.finalize().to_vec(), head))
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Nudge FTS5 to merge its index segments. Best-effort; failure is logged
|
/// FTS5's automerge threshold, applied before a run writes anything.
|
||||||
/// and swallowed.
|
///
|
||||||
pub fn fts_finalize_after_text_indexing(conn: &Connection) {
|
/// The number of index segments that must accumulate at one level before FTS5
|
||||||
|
/// merges them: 2..=16, or 0 to disable incremental merging. Every merge
|
||||||
|
/// rewrites segments to disk, so this is a **write-amplification** knob rather
|
||||||
|
/// than a CPU one, and the FTS index is where essentially all of an indexing
|
||||||
|
/// run's writing goes — `searchabletext_data` measured 228 MiB of a 265 MiB
|
||||||
|
/// index, against 2.8 MiB for all six `files` indexes put together.
|
||||||
|
///
|
||||||
|
/// 16 is the maximum FTS5 accepts and is measured, on a 10,000-file tree, three
|
||||||
|
/// builds per setting:
|
||||||
|
///
|
||||||
|
/// | automerge | cold index | written | search (unlimited) |
|
||||||
|
/// |---:|---:|---:|---:|
|
||||||
|
/// | 4 (FTS5 default) | 8.53 / 8.65 / 8.58 s | 1862 / 1890 / 1818 MiB | 763–772 ms |
|
||||||
|
/// | 16 | 7.10 / 7.10 / 6.67 s | 1235 / 1221 / 1213 MiB | 756–769 ms |
|
||||||
|
///
|
||||||
|
/// **19% faster and a third fewer bytes written, for no search cost.** The
|
||||||
|
/// search column has to be read carefully, and is the reason this comment
|
||||||
|
/// exists: measured at the *default* display limit the same six indexes span
|
||||||
|
/// 40 ms to 540 ms, and none of that spread is automerge. `scan_pass` stops as
|
||||||
|
/// soon as the limit fills and streams FTS candidates in rowid order — which is
|
||||||
|
/// `file_id` order, which is whatever order the concurrent walk inserted rows
|
||||||
|
/// in — so an index where the large documents drew low ids decompresses
|
||||||
|
/// megabytes to fill 1000 hits where another reads a few hundred kilobytes.
|
||||||
|
/// Two builds of one configuration differed by 13x that way. Comparing
|
||||||
|
/// anything about FTS against a limited search measures that lottery instead;
|
||||||
|
/// raise the limit past the corpus so every index examines every candidate.
|
||||||
|
const WRITE_AUTOMERGE: u8 = 16;
|
||||||
|
|
||||||
|
/// Set FTS5's automerge threshold. Best-effort; failure is logged.
|
||||||
|
///
|
||||||
|
/// **This sets a parameter — it does not merge anything.** With a value bound
|
||||||
|
/// to `rank`, `INSERT INTO ft(ft, rank) VALUES('automerge', N)` writes N into
|
||||||
|
/// the table's `%_config`, where it persists. The merging command is
|
||||||
|
/// `'merge'`, and the merge-everything command is `'optimize'`; see
|
||||||
|
/// [`fts_finalize_after_text_indexing`].
|
||||||
|
///
|
||||||
|
/// That distinction was worth a great deal. This used to be called once, at the
|
||||||
|
/// *end* of a run, under the name `fts_finalize_after_text_indexing` and the
|
||||||
|
/// comment "nudge FTS5 to merge its index segments" — so a fresh index did its
|
||||||
|
/// entire first bulk load at FTS5's default threshold of 4, every later run
|
||||||
|
/// silently inherited 8 from the config table, and no merge was ever performed
|
||||||
|
/// at all.
|
||||||
|
pub fn fts_set_automerge(conn: &Connection, segments: u8) {
|
||||||
if let Err(e) = conn.execute(
|
if let Err(e) = conn.execute(
|
||||||
"INSERT INTO searchabletext(searchabletext, rank) VALUES('automerge', 8)",
|
"INSERT INTO searchabletext(searchabletext, rank) VALUES('automerge', ?1)",
|
||||||
[],
|
[segments as i64],
|
||||||
) {
|
) {
|
||||||
crate::log_warn!("FTS automerge failed (non-fatal): {}", e);
|
crate::log_warn!("FTS automerge failed (non-fatal): {}", e);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// FTS5's crisis-merge threshold: the segment count at which it stops deferring
|
||||||
|
/// and *forces* a merge, whatever `automerge` would have preferred.
|
||||||
|
///
|
||||||
|
/// Raised from FTS5's default of 16 to 32 — half way to the 64 that measured
|
||||||
|
/// identically, so the safety valve this is stays nearer where SQLite put it.
|
||||||
|
/// Two builds each, 10,000-file tree, everything else equal:
|
||||||
|
///
|
||||||
|
/// | crisismerge | index | written | cold | search (unlimited) |
|
||||||
|
/// |---:|---:|---:|---:|---:|
|
||||||
|
/// | 16 (default) | 311.8 / 289.9 MiB | 1008 / 973 MiB | 5.54 / 5.51 s | 755–769 ms |
|
||||||
|
/// | 32 | **268.2 / 266.1 MiB** | 946 / 964 MiB | 5.53 / 5.52 s | 756–762 ms |
|
||||||
|
///
|
||||||
|
/// A **13% smaller index for no cost in time or search**, and — the part worth
|
||||||
|
/// noticing — a far more *stable* one: the default's size swings 290–312 MiB
|
||||||
|
/// between builds where this lands within 2 MiB of itself. Fewer forced merges
|
||||||
|
/// mid-load leave the final merge a tidier structure to consolidate.
|
||||||
|
const WRITE_CRISISMERGE: u8 = 32;
|
||||||
|
|
||||||
|
/// Apply the write-side FTS5 settings, before a run starts writing.
|
||||||
|
///
|
||||||
|
/// `pgsz` was swept here too and **rejected**: at 8192 and 16384 it wrote
|
||||||
|
/// 1061 MiB and 1008 MiB against the default's 943 MiB, for an index the same
|
||||||
|
/// size. It is a runtime option like these two — settable on an existing table,
|
||||||
|
/// not creation-time-only — so trying it cost nothing and needed no schema
|
||||||
|
/// change; it simply does not pay on this workload.
|
||||||
|
pub fn fts_begin_bulk_write(conn: &Connection) {
|
||||||
|
fts_set_automerge(conn, WRITE_AUTOMERGE);
|
||||||
|
if let Err(e) = conn.execute(
|
||||||
|
"INSERT INTO searchabletext(searchabletext, rank) VALUES('crisismerge', ?1)",
|
||||||
|
[WRITE_CRISISMERGE as i64],
|
||||||
|
) {
|
||||||
|
crate::log_warn!("FTS crisismerge failed (non-fatal): {}", e);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Merge FTS5 segments once a run has finished writing.
|
||||||
|
///
|
||||||
|
/// A real merge, which is what this function's name has always claimed and what
|
||||||
|
/// it never did. Cheap next to the load — measured at +0.2 s and +5 MiB on a
|
||||||
|
/// 20,000-file tree — and it is what reclaims the tombstones a
|
||||||
|
/// `contentless_delete` table accumulates, which is why the incremental callers
|
||||||
|
/// (`scope`, `cleanup_stale_index_entries`) want it after removing rows.
|
||||||
|
///
|
||||||
|
/// Deliberately **not** `'optimize'`. That merges everything into one segment
|
||||||
|
/// and costs, on the same tree, +1.8 s and +900 MiB written — for no measurable
|
||||||
|
/// search gain: it took the segment-index from 7,337 rows to 730 and left an
|
||||||
|
/// unlimited search within noise of where it started.
|
||||||
|
///
|
||||||
|
/// Best-effort; failure is logged and swallowed, because an unconsolidated
|
||||||
|
/// index is slower to search and still correct.
|
||||||
|
pub fn fts_finalize_after_text_indexing(conn: &Connection) {
|
||||||
|
// A negative page budget means "keep merging until there is nothing left
|
||||||
|
// worth merging", rather than doing a fixed slice of the work.
|
||||||
|
if let Err(e) = conn.execute(
|
||||||
|
"INSERT INTO searchabletext(searchabletext, rank) VALUES('merge', -16)",
|
||||||
|
[],
|
||||||
|
) {
|
||||||
|
crate::log_warn!("FTS merge failed (non-fatal): {}", e);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// An owned, fully-derived file record: everything needed to insert or
|
/// An owned, fully-derived file record: everything needed to insert or
|
||||||
/// update a `files` row, produced by [`prepare_file_record`].
|
/// update a `files` row, produced by [`prepare_file_record`].
|
||||||
#[derive(Debug, Clone)]
|
#[derive(Debug, Clone)]
|
||||||
|
|
|
||||||
|
|
@ -14,7 +14,8 @@ use crate::db;
|
||||||
use crate::db::repo;
|
use crate::db::repo;
|
||||||
use crate::extract::Registry;
|
use crate::extract::Registry;
|
||||||
use crate::file_handling::{
|
use crate::file_handling::{
|
||||||
cleanup_stale_index_entries, count_tree_entries_fast, fts_finalize_after_text_indexing,
|
cleanup_stale_index_entries, count_tree_entries_fast, fts_begin_bulk_write,
|
||||||
|
fts_finalize_after_text_indexing,
|
||||||
mark_oversize_pending_na, normalize_root_string, process_batch_inserts, process_batch_updates,
|
mark_oversize_pending_na, normalize_root_string, process_batch_inserts, process_batch_updates,
|
||||||
store_extracted, ExtractCursor, ExtractScope, FileIndexAction, OwnedNewFile,
|
store_extracted, ExtractCursor, ExtractScope, FileIndexAction, OwnedNewFile,
|
||||||
};
|
};
|
||||||
|
|
@ -280,6 +281,17 @@ impl RootPipeline {
|
||||||
},
|
},
|
||||||
// Earlier runs' rows count once the pass has counted them; until
|
// Earlier runs' rows count once the pass has counted them; until
|
||||||
// then only this run's, so the figure never goes backwards.
|
// then only this run's, so the figure never goes backwards.
|
||||||
|
//
|
||||||
|
// Not clamped against `extract_total`, deliberately. The count now
|
||||||
|
// runs *behind* the pass's first page rather than blocking its
|
||||||
|
// workers in front of it (see `content::feeder`), so a row written
|
||||||
|
// while it was in flight is seen by it as already done and counted
|
||||||
|
// again in `written` — the numerator can briefly overshoot. That is
|
||||||
|
// already a shape this reports: a pass fed rows from outside its own
|
||||||
|
// range writes them with an `extract_total` of zero, which
|
||||||
|
// `an_extracting_turn_lands_its_leftovers_one_slice_at_a_time`
|
||||||
|
// pins. Clamping here broke that test and would have hidden the
|
||||||
|
// case it exists to describe.
|
||||||
extracted: totals.map_or(self.written, |t| t.already_done + self.written),
|
extracted: totals.map_or(self.written, |t| t.already_done + self.written),
|
||||||
extract_total: totals.map(|t| t.pending + t.already_done),
|
extract_total: totals.map(|t| t.pending + t.already_done),
|
||||||
current_file: self.current_file.clone(),
|
current_file: self.current_file.clone(),
|
||||||
|
|
@ -810,6 +822,45 @@ impl IndexingService {
|
||||||
}
|
}
|
||||||
Self::update_config(&conn, config, &roots)?;
|
Self::update_config(&conn, config, &roots)?;
|
||||||
|
|
||||||
|
// Before a single row is written, so the whole load runs at the
|
||||||
|
// write-side threshold. Setting it afterwards — which is what this
|
||||||
|
// used to do — left every fresh index's first run at FTS5's default.
|
||||||
|
fts_begin_bulk_write(&conn);
|
||||||
|
|
||||||
|
// Turn off SQLite's automatic checkpointing for the duration of the
|
||||||
|
// run, because during a run it cannot do its job and charges full
|
||||||
|
// price for failing.
|
||||||
|
//
|
||||||
|
// The default fires every 1000 pages (~4 MB) and copies the log back
|
||||||
|
// into the database — but it can only *reset* the log at an instant no
|
||||||
|
// reader holds a read mark, and this run keeps a reader per root from
|
||||||
|
// start to finish (the walk's row prefetcher, then the content pass's
|
||||||
|
// feeder). So it copied pages back perpetually and never truncated
|
||||||
|
// anything: measured on a 10,000-file tree, the log grew to 144 MiB and
|
||||||
|
// stayed there while the process wrote **1,220 MiB** — the same log
|
||||||
|
// copied back some seven times over.
|
||||||
|
//
|
||||||
|
// Safe here and nowhere else, which is why it is set on this connection
|
||||||
|
// rather than in `PRAGMAS_FAST`: this is the one writer that already
|
||||||
|
// owns the machinery to land its own log. `wal_cap_for_volume` bounds
|
||||||
|
// how large it may grow — by free space, not just by
|
||||||
|
// `maximum_wal_size` — the loop below forces a checkpoint at that cap,
|
||||||
|
// and the optimize pass checkpoints again at the end. A writer without
|
||||||
|
// all three (`cli::clear_path`, say) must keep the automatic one.
|
||||||
|
//
|
||||||
|
// Measured, two runs each, same tree:
|
||||||
|
//
|
||||||
|
// | | cold | written | log peak |
|
||||||
|
// |---|---:|---:|---:|
|
||||||
|
// | autocheckpoint on (default) | 7.19 / 7.37 s | 1219 / 1227 MiB | 144 MiB |
|
||||||
|
// | off | **5.59 / 5.50 s** | **990 / 992 MiB** | 512 MiB |
|
||||||
|
//
|
||||||
|
// The log gets larger and the run gets cheaper, which is the trade the
|
||||||
|
// default is making backwards for this workload.
|
||||||
|
if let Err(e) = conn.execute_batch("PRAGMA wal_autocheckpoint = 0;") {
|
||||||
|
crate::log_warn!("could not disable autocheckpoint (non-fatal): {}", e);
|
||||||
|
}
|
||||||
|
|
||||||
// No up-front load of the whole `files` table: each walk's prefetcher
|
// No up-front load of the whole `files` table: each walk's prefetcher
|
||||||
// fetches one directory's rows at a time.
|
// fetches one directory's rows at a time.
|
||||||
let conn_mutex = Arc::new(Mutex::new(conn));
|
let conn_mutex = Arc::new(Mutex::new(conn));
|
||||||
|
|
@ -974,6 +1025,17 @@ impl IndexingService {
|
||||||
.is_some();
|
.is_some();
|
||||||
// Nothing walking (extracting passes have no such handle);
|
// Nothing walking (extracting passes have no such handle);
|
||||||
// fall back to the sleep.
|
// fall back to the sleep.
|
||||||
|
//
|
||||||
|
// Parking on an extracting root's channel here was tried and
|
||||||
|
// **measured at nothing**: over a cold run of a 10,000-file
|
||||||
|
// tree the loop found nothing 91 times and only *one* of those
|
||||||
|
// reached this sleep, because the writer is the bottleneck
|
||||||
|
// during extraction and is almost never idle. One 2 ms sleep a
|
||||||
|
// run — 15.6 ms on Windows, where the timer granularity is what
|
||||||
|
// makes this comment worth having — did not justify a second
|
||||||
|
// `wait_ready` and the `pending` slot it needs. Re-measure with
|
||||||
|
// an extraction-bound corpus (PDFs, a network share) before
|
||||||
|
// concluding otherwise.
|
||||||
if !waited {
|
if !waited {
|
||||||
thread::sleep(IDLE_BACKOFF);
|
thread::sleep(IDLE_BACKOFF);
|
||||||
}
|
}
|
||||||
|
|
|
||||||
Loading…
Add table
Reference in a new issue