From 0c7744977c19e8b9f404593966374fab0fd365c2 Mon Sep 17 00:00:00 2001 From: "glm-5.3-flash" Date: Sat, 10 Oct 2026 07:46:40 +0000 Subject: [PATCH] =?UTF-8?q?SQLite=20commit-error-arm=20coverage=20(task=20?= =?UTF-8?q?sqlite-commit-error-arm):=20the=20wave-3=20review=20gate's=20de?= =?UTF-8?q?ferred=20test=20landed=20=E2=80=94=20failed=5Fcommit=5Freplenis?= =?UTF-8?q?hes=5Fthe=5Fwriter=5Fslot=20drives=20a=20failing=20COMMIT=20thr?= =?UTF-8?q?ough=20the=20engine's=20commit=20path=20and=20pins=20the=20revi?= =?UTF-8?q?ew's=20code-read=20fix=20end-to-end:=20the=20error=20surfaces?= =?UTF-8?q?=20as=20the=20opaque=20Database=20carrying=20the=20SQLITE=5FFUL?= =?UTF-8?q?L-shaped=20source=20chain,=20the=20failed=20connection=20is=20d?= =?UTF-8?q?ropped=20and=20the=20writer=20slot=20replenished=20via=20the=20?= =?UTF-8?q?handle's=20reopen=20closure=20(the=20next=20begin=5Ftx=20procee?= =?UTF-8?q?ds=20within=20a=20bounded=20timeout,=20no=20parking=20=E2=80=94?= =?UTF-8?q?=20the=20store-wide-livelock=20posture),=20no=20partial-commit?= =?UTF-8?q?=20residue=20(the=20dropped=20connection's=20uncommitted=20writ?= =?UTF-8?q?es=20read=20back=20None),=20the=20post-failure=20commit=20is=20?= =?UTF-8?q?a=20real=20clean=20commit=20(the=20fault=20disarms=20on=20consu?= =?UTF-8?q?mption),=20and=20auto-commit=20notify=20works=20afterward.=20In?= =?UTF-8?q?jection=20is=20a=20cfg(test)=20commit-fault=20seam=20in=20seam.?= =?UTF-8?q?rs=20=E2=80=94=20a=20per-store=20Arc=20arm=20(born?= =?UTF-8?q?=20disarmed,=20armed=20via=20arm=5Fcommit=5Ffault,=20take()=20d?= =?UTF-8?q?isarms=20on=20first=20consumption=20so=20exactly=20one=20commit?= =?UTF-8?q?=20faults)=20whose=20fabricated=20rusqlite=20SqliteFailure=20fe?= =?UTF-8?q?eds=20the=20production=20commit=20error=20arm=20rather=20than?= =?UTF-8?q?=20replicating=20it;=20the=20PRAGMA=20max=5Fpage=5Fcount=20rout?= =?UTF-8?q?e=20was=20probed=20live=20against=20both=20WAL=20and=20DELETE?= =?UTF-8?q?=20journal=20modes=20first=20and=20rejected:=20SQLite=20checks?= =?UTF-8?q?=20the=20page-count=20limit=20at=20page-allocation=20time,=20so?= =?UTF-8?q?=20the=20squeeze=20always=20fails=20the=20growth=20statement=20?= =?UTF-8?q?(SQLITE=5FFULL/DiskFull=20on=20the=20first=20INSERT)=20and=20le?= =?UTF-8?q?aves=20no=20transaction=20active=20for=20COMMIT=20to=20fail=20?= =?UTF-8?q?=E2=80=94=20the=20arm=20is=20unreachable=20through=20PRAGMA-spa?= =?UTF-8?q?ce=20(also=20probed:=20the=20pragma=20is=20per-connection,=20so?= =?UTF-8?q?=20pre-begin=20arming=20on=20the=20writer=20conn=20would=20have?= =?UTF-8?q?=20ridden=20into=20the=20tx=20conn;=20the=20route=20failed=20on?= =?UTF-8?q?=20error=20placement,=20not=20delivery).=20Mechanism=20choice?= =?UTF-8?q?=20and=20probes=20documented=20in=20the=20seam=20doc=20comment?= =?UTF-8?q?=20and=20the=20task=20Notes.=20Cross-test=20safety=20is=20per-s?= =?UTF-8?q?tore=20scoping;=20parallel=20stores=20never=20see=20the=20arm.?= =?UTF-8?q?=20Replay-proofed=20live:=20with=20the=20error=20arm's=20writer?= =?UTF-8?q?=5Freopen=20replenish=20temporarily=20removed=20the=20test=20fa?= =?UTF-8?q?ils=20(begin=5Ftx=20parks=20past=20the=205=20s=20timeout=20?= =?UTF-8?q?=E2=80=94=20the=20stranding=20the=20review=20identified),=20rev?= =?UTF-8?q?erted=20it=20passes.=20Plumbing=20follows=20the=20pg-fix-forwar?= =?UTF-8?q?der-reconnect=20cfg(test)=20precedent:=20fields=20on=20SqliteSt?= =?UTF-8?q?ore/SqliteTxHandle=20and=20a=20begin=20param=20are=20cfg-gated,?= =?UTF-8?q?=20production=20builds=20compile=20the=20plain=20path.=20The=20?= =?UTF-8?q?waves-1-2=20review's=20optional=20watcher=20reconnect-success?= =?UTF-8?q?=20add=20rides=20here=20(taken=20=E2=80=94=20recorded=20in=20No?= =?UTF-8?q?tes):=20reconnect=5Fsuccess=5Fresumes=5Fwake=5Fdelivery=20drive?= =?UTF-8?q?s=20run=5Fpoll=5Floop=20through=20its=20existing=20open=5Fconn?= =?UTF-8?q?=5Ffn=20seam=20(same=20instrument=20as=20the=20W-1=20failure=20?= =?UTF-8?q?test),=20with=20the=20db=20file=20present=20throughout=20becaus?= =?UTF-8?q?e=20the=20vanished-file=20route=20cannot=20reach=20the=20succes?= =?UTF-8?q?s=20body=20(file=20reappearance=20trips=20the=20dead-man's=20id?= =?UTF-8?q?entity=20switch=20first):=20initial=20open=20+=20first=20two=20?= =?UTF-8?q?reconnects=20fail=20by=20injection,=20the=20third=20reconnect?= =?UTF-8?q?=20succeeds,=20and=20a=20subsequent=20commit=20wakes=20on=5Fcha?= =?UTF-8?q?nge=20=E2=80=94=20the=20success=20arm's=20data=5Fversion=20re-b?= =?UTF-8?q?aseline=20and=20restored=20delivery=20pinned.=20Watcher=20shape?= =?UTF-8?q?=20untouched.=20Verified:=20cargo=20test=20-p=20alkstore-sqlite?= =?UTF-8?q?=20green=20server-less=20(191=20lib=20+=2025=20suite),=20worksp?= =?UTF-8?q?ace=20cargo=20test=20399/0,=20clippy=20-D=20warnings,=20fmt=20c?= =?UTF-8?q?lean?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- alkstore-sqlite/src/seam.rs | 39 +++++++++++ alkstore-sqlite/src/store.rs | 24 ++++++- alkstore-sqlite/src/store/tx_tests.rs | 68 ++++++++++++++++++ alkstore-sqlite/src/substrate/watcher.rs | 88 ++++++++++++++++++++++++ alkstore-sqlite/src/tx.rs | 26 +++++-- tasks/sqlite-commit-error-arm.md | 83 +++++++++++++++++++++- 6 files changed, 320 insertions(+), 8 deletions(-) diff --git a/alkstore-sqlite/src/seam.rs b/alkstore-sqlite/src/seam.rs index 1bc6aea..3e86b49 100644 --- a/alkstore-sqlite/src/seam.rs +++ b/alkstore-sqlite/src/seam.rs @@ -160,3 +160,42 @@ pub(crate) fn is_closed_err(e: &rusqlite::Error) -> bool { pub(crate) fn closed_store_error() -> Error { Error::database(std::io::Error::other("the store is closed")) } + +/// The `cfg(test)` commit-fault seam (`sqlite-commit-error-arm`): a +/// per-store arm that, when set, substitutes a `SQLITE_FULL`-shaped +/// driver error for one `COMMIT`, feeding the real commit error arm +/// (drop + `reopen` replenish) instead of fabricating its behavior. +/// Chosen over the `PRAGMA max_page_count` squeeze after live probing +/// both WAL and DELETE journal modes: SQLite checks the page-count +/// limit at page-allocation time, so the squeeze always fails the +/// growth statement first and leaves no transaction for `COMMIT` to +/// fail on — the arm is unreachable through PRAGMA-space. The arm +/// holds a flag per [`crate::SqliteStore`] (born disarmed), so +/// parallel test stores cannot interfere; `take` disarms on read — +/// exactly one commit faults, and the post-failure commits are real. +#[cfg(test)] +pub(crate) mod commit_fault { + use std::sync::Arc; + use std::sync::atomic::{AtomicBool, Ordering}; + + pub(crate) type Flag = Arc; + + pub(crate) fn disarmed() -> Flag { + Arc::new(AtomicBool::new(false)) + } + + pub(crate) fn arm(flag: &Flag) { + flag.store(true, Ordering::Release); + } + + pub(crate) fn take(flag: &Flag) -> bool { + flag.swap(false, Ordering::AcqRel) + } + + pub(crate) fn failure() -> rusqlite::Error { + rusqlite::Error::SqliteFailure( + rusqlite::ffi::Error::new(rusqlite::ffi::SQLITE_FULL), + Some("database or disk is full".to_string()), + ) + } +} diff --git a/alkstore-sqlite/src/store.rs b/alkstore-sqlite/src/store.rs index 2775690..277d08b 100644 --- a/alkstore-sqlite/src/store.rs +++ b/alkstore-sqlite/src/store.rs @@ -50,6 +50,12 @@ pub struct SqliteStore { watcher: Arc, db_path: PathBuf, closed: Arc, + /// The store's commit-fault arm (`sqlite-commit-error-arm`) — born + /// disarmed; a test arms it through [`SqliteStore::arm_commit_fault`] + /// so the next handle commit fails through the real commit error + /// arm. Dead in production builds. + #[cfg(test)] + commit_fault: crate::seam::commit_fault::Flag, } impl std::fmt::Debug for SqliteStore { @@ -71,6 +77,16 @@ impl SqliteStore { self.readers.close(); self.writer.close(); } + + /// Arm the commit-fault seam (`sqlite-commit-error-arm`): the + /// next `commit` through a handle of this store fails with a + /// `SQLITE_FULL`-shaped driver error and replenishes the writer + /// slot through the real commit error arm. One-store scoping keeps + /// parallel test stores unaffected. + #[cfg(test)] + pub(crate) fn arm_commit_fault(&self) { + crate::seam::commit_fault::arm(&self.commit_fault) + } } impl Drop for SqliteStore { @@ -109,6 +125,8 @@ pub(crate) fn open_store(path: &str, opts: SqliteOpts) -> alkstore::Result alkstore::Result + Send + Sync, > = std::sync::Arc::new(move || crate::seam::open_writer_connection(&db_path)); - crate::tx::SqliteTxHandle::begin(writer, reopen) + #[cfg(test)] + let fut = crate::tx::SqliteTxHandle::begin(writer, reopen, self.commit_fault.clone()); + #[cfg(not(test))] + let fut = crate::tx::SqliteTxHandle::begin(writer, reopen); + fut } fn notify<'a>( diff --git a/alkstore-sqlite/src/store/tx_tests.rs b/alkstore-sqlite/src/store/tx_tests.rs index d7d703c..cf14dfb 100644 --- a/alkstore-sqlite/src/store/tx_tests.rs +++ b/alkstore-sqlite/src/store/tx_tests.rs @@ -1199,6 +1199,74 @@ async fn cancelling_a_tx_future_rolls_back() { cleanup(&dir); } +/// Acceptance: a failing `COMMIT` surfaces as the opaque `Database` +/// (`sqlite-commit-error-arm`, the wave-3 review's deferred coverage — +/// the replenish fix pinned by code-read there, now test-exercised), +/// does not strand the writer slot — the failed connection is dropped +/// (its state unknowable) and the slot replenished via the handle's +/// `reopen` closure — and the store stays fully usable: the next +/// `begin_tx` proceeds, finds no partial-commit residue (the failed +/// tx's writes rolled back with the dropped connection), commits +/// cleanly, and auto-commit ops work. +/// +/// Injection is the engine's `cfg(test)` commit-fault seam +/// (seam.rs): a `SQLITE_FULL`-shaped driver error substituted for one +/// `COMMIT`, feeding the production error arm (drop + `reopen` +/// replenish) rather than fabricating its behavior. The seam over the +/// `PRAGMA max_page_count` squeeze: live probes of both WAL and +/// DELETE journal modes showed SQLite checks the page-count limit at +/// page-allocation time — the squeeze always fails the growth +/// statement (first `INSERT`) and leaves no transaction for `COMMIT` +/// to fail on, so the PRAGMA route cannot reach this arm +/// deterministically. The seam carries no cross-test risk: the flag +/// is per-store and disarms on its first consumption. +#[tokio::test(flavor = "multi_thread")] +async fn failed_commit_replenishes_the_writer_slot() { + let dir = temp_dir("commit-err"); + let concrete = open_store(dir.join("store.db").to_str().unwrap(), Default::default()).unwrap(); + + concrete.arm_commit_fault(); + + let mut tx = concrete.begin_tx().await.unwrap(); + let id = tx + .enqueue_tx("q", EnqueueOpts::default(), serde_json::json!({"n": 1})) + .await + .unwrap(); + assert!(id > 0, "the in-tx write landed before the failed commit"); + + let err = tx.commit().await.unwrap_err(); + match err { + Error::Database(source) => { + assert!( + source.to_string().contains("full"), + "the commit error carries the SQLITE_FULL-shaped source chain: {source}" + ); + } + other => panic!("a failed commit errs opaque Database, got {other:?}"), + } + + let next = tokio::time::timeout(std::time::Duration::from_secs(5), concrete.begin_tx()) + .await + .expect("the slot must be grantable after the failed commit (replenished via reopen)") + .unwrap(); + let mut next = next; + let residue = next.get_job_tx("q", id).await.unwrap(); + assert_eq!( + residue, None, + "the failed commit's writes left no partial-commit residue" + ); + next.commit() + .await + .expect("the post-failure commit must be a real, clean commit (fault disarmed)"); + + concrete + .notify("probe", serde_json::json!({"after": "commit-error"})) + .await + .expect("auto-commit ops work after the failed commit"); + drop(concrete); + cleanup(&dir); +} + /// Acceptance: a tokio task cancelled *mid-op* (its future dropped /// while an op's blocking round trip is in flight) cannot strand the /// writer slot — the op's RAII connection lease replenishes the slot diff --git a/alkstore-sqlite/src/substrate/watcher.rs b/alkstore-sqlite/src/substrate/watcher.rs index 866451f..4b1cd3e 100644 --- a/alkstore-sqlite/src/substrate/watcher.rs +++ b/alkstore-sqlite/src/substrate/watcher.rs @@ -820,6 +820,94 @@ mod watcher_tests { assert_eq!(idx, 0); } + /// The reconnect loop's *success* body (the coverage doc's open + /// line 187–202 — `shared_update_watcher`'s W-1 test drives only + /// the failure path, and a vanished-file drive cannot test the + /// success body: after the file reappears the identity check reads + /// a fresh `(dev, ino)` and kills the watcher through the + /// dead-man's switch). Driven here without the vanished-file trick + /// and without touching the engine's watcher shape: the db file + /// exists for the whole run (identity stable), and the + /// `run_poll_loop`'s own `open_conn_fn` seam — the same mechanism + /// the W-1 test instruments — fails the initial open and the first + /// two reconnects, with the third reconnect succeeding. A commit + /// from a separate writer connection after that success must wake + /// `on_change` again, proving the success arm re-baselines + /// `data_version` and restores wake delivery (and that injected + /// failed opens do not poison the re-established connection). + #[test] + fn reconnect_success_resumes_wake_delivery() { + static OPENS: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(0); + const FAIL_FIRST: u64 = 3; + fn instrumented_open(path: &Path) -> rusqlite::Result { + let n = OPENS.fetch_add(1, Ordering::Relaxed) + 1; + if n <= FAIL_FIRST { + return Err(rusqlite::Error::InvalidPath(path.to_path_buf())); + } + open_watcher_conn(path) + } + + let tmp = temp_db("reconnect-success"); + wal_db(&tmp); + let writer = Connection::open(&tmp).unwrap(); + writer + .execute("CREATE TABLE IF NOT EXISTS _test_reconnect(x INTEGER)", []) + .unwrap(); + + let wakes = Arc::new(AtomicU64::new(0)); + let wakes_t = wakes.clone(); + let stop = Arc::new(AtomicBool::new(false)); + let stop_t = stop.clone(); + let (ready_tx, ready_rx) = std::sync::mpsc::sync_channel::<()>(1); + let db_path = tmp.clone(); + let handle = std::thread::spawn(move || { + run_poll_loop( + db_path, + move || { + wakes_t.fetch_add(1, Ordering::Relaxed); + }, + stop_t, + ready_tx, + Duration::from_millis(5), + instrumented_open, + ) + }); + ready_rx + .recv_timeout(Duration::from_secs(2)) + .expect("watcher loop baseline capture"); + + // The reconnect schedule retries every MAX_RECONNECT_TICKS + // ticks — 500 ms apart at the 5 ms cadence. Let the first two + // reconnect attempts fail (opens 2 and 3) and the third + // reconnect succeed (open 4). + std::thread::sleep(Duration::from_millis(1_800)); + let opens = OPENS.load(Ordering::Relaxed); + assert!( + opens > FAIL_FIRST, + "a reconnect open must have succeeded after the injected failures, saw {opens}" + ); + + writer + .execute("INSERT INTO _test_reconnect(x) VALUES (1)", []) + .unwrap(); + + let deadline = std::time::Instant::now() + Duration::from_secs(3); + while wakes.load(Ordering::Relaxed) == 0 && std::time::Instant::now() < deadline { + std::thread::sleep(Duration::from_millis(5)); + } + assert!( + wakes.load(Ordering::Relaxed) > 0, + "no wake delivered after the successful reconnect — the success \ + body must re-baseline data_version and restore delivery" + ); + + stop.store(true, Ordering::Release); + handle.join().unwrap(); + let _ = std::fs::remove_file(&tmp); + let _ = std::fs::remove_file(format!("{}-wal", tmp.display())); + let _ = std::fs::remove_file(format!("{}-shm", tmp.display())); + } + /// W-1: with the watcher connection down and the db file vanished, /// the reconnect path is throttled — not one open attempt per /// poll tick. Driven directly: a missing db path fails every open, diff --git a/alkstore-sqlite/src/tx.rs b/alkstore-sqlite/src/tx.rs index 7d3db2d..da128dc 100644 --- a/alkstore-sqlite/src/tx.rs +++ b/alkstore-sqlite/src/tx.rs @@ -47,6 +47,11 @@ pub struct SqliteTxHandle { conn: Option, writer: Arc, reopen: Arc Result + Send + Sync>, + /// This store's commit-fault arm (`sqlite-commit-error-arm`) — + /// read once per commit by the test-only fault branch; dead in + /// production builds. + #[cfg(test)] + commit_fault: crate::seam::commit_fault::Flag, } impl std::fmt::Debug for SqliteTxHandle { @@ -61,6 +66,7 @@ impl SqliteTxHandle { pub(crate) fn begin( writer: Arc, reopen: Arc Result + Send + Sync>, + #[cfg(test)] commit_fault: crate::seam::commit_fault::Flag, ) -> BoxedFuture<'static, Result>> { Box::pin(async move { let writer2 = writer.clone(); @@ -83,6 +89,8 @@ impl SqliteTxHandle { conn: Some(conn), writer, reopen, + #[cfg(test)] + commit_fault, }) as _) }) } @@ -392,9 +400,19 @@ impl TxHandle for SqliteTxHandle { }; let writer = self.writer.clone(); let writer_reopen = self.reopen.clone(); + #[cfg(test)] + let commit_fault = self.commit_fault.clone(); Box::pin(async move { - blocking( - move || match conn.execute_batch("COMMIT").map_err(sqlite_error) { + blocking(move || { + #[cfg(test)] + let committed = if crate::seam::commit_fault::take(&commit_fault) { + Err(crate::seam::commit_fault::failure()) + } else { + conn.execute_batch("COMMIT") + }; + #[cfg(not(test))] + let committed = conn.execute_batch("COMMIT"); + match committed.map_err(sqlite_error) { Ok(()) => { writer.release(conn); Ok(()) @@ -406,8 +424,8 @@ impl TxHandle for SqliteTxHandle { } Err(e) } - }, - ) + } + }) .await }) } diff --git a/tasks/sqlite-commit-error-arm.md b/tasks/sqlite-commit-error-arm.md index 1737a75..167d539 100644 --- a/tasks/sqlite-commit-error-arm.md +++ b/tasks/sqlite-commit-error-arm.md @@ -1,7 +1,7 @@ --- id: sqlite-commit-error-arm name: SQLite engine — commit-error-arm coverage (wave-3 review's deferred test) -status: pending +status: completed depends_on: [] scope: narrow risk: medium @@ -61,8 +61,85 @@ suggestion, not an order). ## Notes -> To be filled by implementation agent +> Decisions of record the implementation made that the description +> didn't pin: + +- **Injection mechanism: the `cfg(test)` commit-fault seam, not the + PRAGMA route — decided by live probe.** Both PRAGMA candidates were + tested against real SQLite before choosing: in *both* WAL and DELETE + journal modes, `PRAGMA max_page_count` squeezed to the current page + count fails the first growth statement (`SQLITE_FULL` / + `DiskFull` at page-allocation time), leaving "no transaction is + active" for the subsequent `COMMIT` — the commit arm is unreachable + through PRAGMA-space. (Also probed: `max_page_count` is + per-connection, so arming on the writer conn pre-`begin_tx` would + have ridden into the tx connection — the route failed on placement + of the error, not on delivery.) The seam substitutes a + `SQLITE_FULL`-shaped `rusqlite::Error` for **one** commit and feeds + it into the *production* commit error arm — the test pins the real + drop + `reopen` replenish code, not a replica. +- **Scoping design** (the seam's cross-test-safety shape): the fault + flag is a per-`SqliteStore` `Arc` (born disarmed, armed + via the `#[cfg(test)]` `arm_commit_fault`, disarmed on first + consumption — exactly one commit faults and every later commit is + real, which the test also pins). Parallel test stores can never + interfere. Plumbing follows the `pg-fix-forwarder-reconnect` + precedent exactly: `#[cfg(test)]` field on `SqliteStore` and on + `SqliteTxHandle`, `#[cfg(test)]` param on `SqliteTxHandle::begin`, + cfg'd statement branches in `store.rs::begin_tx` and + `src/tx.rs::commit` — production builds compile the plain path. +- **Replay-proofed live**: with the commit error arm's + `writer_reopen` replenish temporarily removed, the new test fails + (the next `begin_tx` parks past its 5 s timeout — the stranding the + review identified by code-read); reverted, it passes. +- **The watcher reconnect-success add rides here — taken.** The + vanished-file route the waves-1–2 review sketched could not reach + the success body (on file reappearance the identity check reads a + fresh `(dev, ino)` and the watcher dies through the dead-man's + switch before the reconnect would matter). Instead the test drives + `run_poll_loop` directly through the `open_conn_fn` seam the loop + already takes — the same instrument as the W-1 failure test — with + the db file present throughout (identity stable): initial open and + first two reconnects injected as failures, third reconnect + succeeds, and a commit from a writer connection then wakes + `on_change` (proving the success arm re-baselines `data_version` + and restores delivery). Watcher shape untouched. +- `cargo build`; `cargo test -p alkstore-sqlite` green server-less + (191 lib + 25 suite, both new tests included); workspace `cargo + test` 399 passed / 0 failed (core 25 + harness 3; postgres 121 + + 25 suite + 9 schema; sqlite 191 + 25 suite); `cargo clippy + --all-targets -- -D warnings`; `cargo fmt --check` — all green. ## Summary -> To be filled on completion \ No newline at end of file +> What landed, verified how: + +- **`alkstore-sqlite/src/seam.rs`**: the `#[cfg(test)]` + `commit_fault` module — per-store `Flag` (`Arc`), + `disarmed`/`arm`/`take` (disarm-on-consume), and the + `SQLITE_FULL`-shaped fabricated `rusqlite::Error` — with the + mechanism-choice rationale (the PRAGMA probes) in its doc comment. +- **`alkstore-sqlite/src/tx.rs`**: `#[cfg(test)]` fault field on + `SqliteTxHandle` + `begin` param; `commit`'s blocking body gains a + cfg'd branch that feeds the fault into the existing production + match — the error arm (drop + `reopen` replenish, `sqlite_error` + mapping) runs unmodified. +- **`alkstore-sqlite/src/store.rs`**: `#[cfg(test)]` + `commit_fault` field on `SqliteStore`, the `arm_commit_fault` + surface, and the cfg-branched `begin_tx` call. +- **`alkstore-sqlite/src/store/tx_tests.rs`**: + `failed_commit_replenishes_the_writer_slot` — failed `COMMIT` errs + opaque `Database` with the SQLITE_FULL source chain; the next + `begin_tx` proceeds (slot replenished, no stranding); no + partial-commit residue; the post-failure commit is real and clean + (fault disarmed); auto-commit `notify` works afterward. +- **`alkstore-sqlite/src/substrate/watcher.rs`**: + `reconnect_success_resumes_wake_delivery` — the waves-1–2 review's + optional reconnect-success add, driven through the loop's + `open_conn_fn` seam (injected failed opens, then a successful + reconnect; wake delivery resumes and is pinned). Recorded as + *taken* per the task's optional clause. +- **Gates**: `cargo build`, `cargo test` (server-less: core 25 + core + suite 25 + sqlite 191 + postgres 121 + 9 schema — 399 passed / 0 + failed, postgres harness skipped server-less), `cargo clippy + --all-targets -- -D warnings`, `cargo fmt --check` — all green. \ No newline at end of file