perf: runtime performance pass, phase 1 (DB pragmas, batched scan writes)

Executes docs/superpowers/specs/2026-07-07-runtime-performance-pass-design.md:

- database.rs: every pooled connection now runs WAL + synchronous=NORMAL
  + busy_timeout=5000 + foreign_keys=ON via with_init (in-memory pools
  get the latter two). Fixes the SQLITE_BUSY failure mode under the
  parallel scanner and makes FK cascades work on every connection, not
  just whichever one last ran the old one-shot pragma in delete_folder.
- database.rs: prepare_cached for the per-video info_cache queries; new
  info_cache_put_many bulk upsert (single transaction) replaces the
  per-row autocommitted info_cache_put, with a round-trip unit test.
- library.rs: enrich_with_cache collects cache misses and flushes them
  once per channel; title sort uses sort_by_cached_key.
- app.rs: Title/ChannelAsc sorts use sort_by_cached_key instead of
  allocating two lowercased strings per comparison.
- Cargo.toml: panic=abort in release (crash-hook still fires; Cargo
  forces unwind for test harnesses). PACKAGING.md documents the local
  RUSTFLAGS target-cpu=native opt-in; repo stays portable.
- .gitignore: catacomb.db-wal / catacomb.db-shm sidecars.

Verified: 146 tests pass (integration suite exercises the pragmas
end-to-end), x86_64-pc-windows-gnu check green, live smoke shows
journal_mode=wal + batched cache writes landing correct rows.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Luna 2026-07-08 01:22:40 -07:00
parent 891b1e0d3c
commit d4bed019b2
No known key found for this signature in database
7 changed files with 299 additions and 34 deletions

View file

@ -938,7 +938,9 @@ impl App {
}
match self.sort_mode {
SortMode::Title => cards.sort_by(|a, b| a.title.to_lowercase().cmp(&b.title.to_lowercase())),
// sort_by_cached_key lowercases each element once instead of
// twice per comparison (O(n) allocations, not O(n log n)).
SortMode::Title => cards.sort_by_cached_key(|c| c.title.to_lowercase()),
SortMode::DurationAsc => cards.sort_by(|a, b| {
a.duration_secs.unwrap_or(0.0)
.partial_cmp(&b.duration_secs.unwrap_or(0.0))
@ -977,9 +979,8 @@ impl App {
cards.sort_by(|a, b| a.mtime_unix.unwrap_or(u64::MAX).cmp(&b.mtime_unix.unwrap_or(u64::MAX)));
}
SortMode::ChannelAsc => {
cards.sort_by(|a, b| {
a.channel_name.to_lowercase().cmp(&b.channel_name.to_lowercase())
.then_with(|| a.title.to_lowercase().cmp(&b.title.to_lowercase()))
cards.sort_by_cached_key(|c| {
(c.channel_name.to_lowercase(), c.title.to_lowercase())
});
}
}

View file

@ -138,7 +138,25 @@ impl Database {
/// hash and resume positions aren't readable by other local users. A
/// best-effort: failure is logged but doesn't abort startup.
pub fn open(path: &Path) -> Result<Self> {
let manager = SqliteConnectionManager::file(path);
// Per-connection pragmas. `synchronous`, `busy_timeout` and
// `foreign_keys` are connection-scoped in SQLite, so they must run on
// every pooled connection (`with_init`), not once at open.
// - WAL: write throughput + readers don't block the writer. Persists
// in the file header; re-asserting per connection is idempotent.
// Creates `catacomb.db-wal`/`-shm` sidecars (gitignored).
// - synchronous=NORMAL: safe with WAL, drops the per-commit fsync.
// - busy_timeout: the parallel scanner's concurrent writers wait for
// the lock instead of failing SQLITE_BUSY immediately.
// - foreign_keys: enforcement is per-connection; without this, only
// whichever connection ran init_schema had it on.
let manager = SqliteConnectionManager::file(path).with_init(|conn| {
conn.execute_batch(
"PRAGMA journal_mode = WAL;
PRAGMA synchronous = NORMAL;
PRAGMA busy_timeout = 5000;
PRAGMA foreign_keys = ON;",
)
});
let pool = Pool::builder()
.max_size(FILE_POOL_SIZE)
.build(manager)
@ -168,7 +186,14 @@ impl Database {
/// is capped at 1 connection here. Otherwise each `get()` would hand back
/// a fresh, empty database and our schema/data would vanish between calls.
pub fn open_in_memory() -> Result<Self> {
let manager = SqliteConnectionManager::memory();
// Same per-connection pragmas as `open`, minus WAL/synchronous —
// journaling pragmas are meaningless for an in-memory database.
let manager = SqliteConnectionManager::memory().with_init(|conn| {
conn.execute_batch(
"PRAGMA busy_timeout = 5000;
PRAGMA foreign_keys = ON;",
)
});
let pool = Pool::builder()
.max_size(1)
.build(manager)
@ -406,8 +431,7 @@ impl Database {
/// "Unfiled".
pub fn delete_folder(&self, id: i64) -> Result<()> {
let conn = self.conn();
// Enable FK cascade for this connection — SQLite has it off by default.
conn.execute("PRAGMA foreign_keys = ON", [])?;
// FK cascade is on for every pooled connection (see `with_init`).
conn.execute("DELETE FROM folders WHERE id = ?1", [id])?;
Ok(())
}
@ -750,7 +774,9 @@ impl Database {
mtime_unix: u64,
) -> Option<(Option<f64>, bool, Option<String>)> {
let conn = self.conn();
let mut stmt = conn.prepare(
// prepare_cached: this runs once per video per scan — reuse the
// compiled statement instead of re-parsing the SQL each call.
let mut stmt = conn.prepare_cached(
"SELECT mtime_unix, duration_secs, has_chapters, upload_date \
FROM info_cache WHERE path = ?1",
).ok()?;
@ -766,30 +792,39 @@ impl Database {
Some((dur, chap != 0, date))
}
/// Upsert a parsed info.json result into the cache. Called on miss
/// by the library scanner. Errors are swallowed — a cache miss next
/// time costs the same as no cache at all.
pub fn info_cache_put(
/// Bulk upsert of parsed info.json results in a single transaction.
/// Called with the scanner's cache misses (`enrich_with_cache`).
///
/// A cold-cache scan misses on every video; committing each row
/// individually costs one WAL append per video. Wrapping the batch in
/// one transaction commits once per channel instead. Errors are
/// swallowed — a lost cache write only means a re-parse next scan.
pub fn info_cache_put_many(
&self,
path: &str,
mtime_unix: u64,
duration_secs: Option<f64>,
has_chapters: bool,
upload_date: Option<&str>,
rows: &[(String, u64, Option<f64>, bool, Option<String>)],
) {
let conn = self.conn();
let _ = conn.execute(
"INSERT OR REPLACE INTO info_cache \
(path, mtime_unix, duration_secs, has_chapters, upload_date) \
VALUES (?1, ?2, ?3, ?4, ?5)",
rusqlite::params![
path,
mtime_unix as i64,
duration_secs,
has_chapters as i64,
upload_date,
],
);
if rows.is_empty() {
return;
}
let mut conn = self.conn();
let Ok(tx) = conn.transaction() else { return };
{
let Ok(mut stmt) = tx.prepare_cached(
"INSERT OR REPLACE INTO info_cache \
(path, mtime_unix, duration_secs, has_chapters, upload_date) \
VALUES (?1, ?2, ?3, ?4, ?5)",
) else { return };
for (path, mtime, dur, chap, date) in rows {
let _ = stmt.execute(rusqlite::params![
path,
*mtime as i64,
dur,
*chap as i64,
date.as_deref(),
]);
}
}
let _ = tx.commit();
}
/// Idempotently merge another catacomb database into this one.
@ -1230,6 +1265,39 @@ pub struct RestoreSummary {
pub notes_added: u64,
}
#[cfg(test)]
mod info_cache_tests {
use super::*;
#[test]
fn put_many_roundtrip_and_mtime_gate() {
let db = Database::open_in_memory().unwrap();
let rows = vec![
("/lib/a.info.json".to_string(), 100u64, Some(12.5), true, Some("20260101".to_string())),
("/lib/b.info.json".to_string(), 200u64, None, false, None),
("/lib/c.info.json".to_string(), 300u64, Some(0.0), false, Some("20251231".to_string())),
];
db.info_cache_put_many(&rows);
assert_eq!(
db.info_cache_get("/lib/a.info.json", 100),
Some((Some(12.5), true, Some("20260101".to_string())))
);
assert_eq!(db.info_cache_get("/lib/b.info.json", 200), Some((None, false, None)));
// Stale mtime and unknown path both miss.
assert_eq!(db.info_cache_get("/lib/a.info.json", 101), None);
assert_eq!(db.info_cache_get("/lib/nope.info.json", 100), None);
// Re-put with a new mtime replaces the row (INSERT OR REPLACE).
db.info_cache_put_many(&[("/lib/a.info.json".to_string(), 101u64, Some(99.0), false, None)]);
assert_eq!(db.info_cache_get("/lib/a.info.json", 101), Some((Some(99.0), false, None)));
assert_eq!(db.info_cache_get("/lib/a.info.json", 100), None);
// Empty batch is a no-op, not an error.
db.info_cache_put_many(&[]);
}
}
#[cfg(test)]
mod search_tests {
use super::*;

View file

@ -449,6 +449,10 @@ fn collect_raw_videos(entries: impl Iterator<Item = std::fs::DirEntry>) -> Vec<R
}
fn enrich_with_cache(raws: Vec<RawVideo>, cache: Option<&Database>) -> Vec<Video> {
// Cache misses are collected here and flushed in ONE transaction after
// the loop (`info_cache_put_many`). On a cold cache every video is a
// miss, so per-row autocommitted upserts would pay one commit per video.
let mut cache_misses: Vec<(String, u64, Option<f64>, bool, Option<String>)> = Vec::new();
let mut videos: Vec<Video> = raws.into_iter().map(|raw| {
// Parse info.json once for both duration and chapter presence.
// The cache (if provided) lets us skip the read+JSON parse when
@ -485,8 +489,8 @@ fn enrich_with_cache(raws: Vec<RawVideo>, cache: Option<&Database>) -> Vec<Video
(dur, chap, date)
})
.unwrap_or((None, false, None));
if let (Some(mt), Some(db)) = (mtime, cache) {
db.info_cache_put(&path_str, mt, parsed.0, parsed.1, parsed.2.as_deref());
if let (Some(mt), Some(_)) = (mtime, cache) {
cache_misses.push((path_str.into_owned(), mt, parsed.0, parsed.1, parsed.2.clone()));
}
parsed
})
@ -516,7 +520,11 @@ fn enrich_with_cache(raws: Vec<RawVideo>, cache: Option<&Database>) -> Vec<Video
upload_date,
}
}).collect();
videos.sort_by_key(|v| v.title.to_lowercase());
if let Some(db) = cache {
db.info_cache_put_many(&cache_misses);
}
// sort_by_cached_key: lowercase each title once, not once per comparison.
videos.sort_by_cached_key(|v| v.title.to_lowercase());
videos
}