Implement the full ADR-0006 plan: three-layer separation of transaction
correctness, retry policy, and cross-process writer coordination.
Layer 1 — scripts/tx.ts (transaction correctness):
- runWriteTransaction executes work exactly once; no internal retry.
- BEGIN IMMEDIATE takes the write lock up front (avoids SQLITE_BUSY_SNAPSHOT).
- Guarded rollback: checks inTransaction() via adapter before attempting
ROLLBACK; never masks the primary exception.
- WriteTxDiagnostics attached to errors: phase, code, label,
rollbackSucceeded, rollbackError, transactionActive.
- Binding adapters (betterSqliteTransactionAdapter, nodeSqliteTransactionAdapter)
mapping better-sqlite3's `.inTransaction` and node:sqlite's `.isTransaction`.
- configureConnection centralizes WAL + synchronous + busy_timeout.
Layer 2 — scripts/write-coordinator.ts (retry policy):
- runRetryableWriteTransaction: bounded retry with total time budget.
- Only retries when the transaction confirmed ended (transactionActive=false)
and the error is SQLITE_BUSY during work/commit phase.
- BEGIN-phase BUSY = abort entire build (isBeginBusyFailure); the caller
returns `{ deferred: true, reason: 'writer_busy' }` instead of waiting.
- hasUnusableTransaction detects a still-active transaction after failure;
aborts the build immediately, never retries.
Layer 3 — scripts/writer-lease.ts (cross-process coordination):
- acquireWriterLease: dedicated writer.lock.sqlite with busy_timeout=0 +
BEGIN IMMEDIATE. Non-blocking attempt; bounded wait with retryDelayMs.
- writerLockPathFor derives lock path from the target DB path.
- Lease held for the entire build; released on completion or failure.
- Lock DB uses DELETE journal (not WAL); crash/close auto-releases.
- All consumers obey: skill acquires at build start (returns deferred if
unavailable); app daemon (via worker) acquires for its build cycle.
Build semantics changes:
- affectedSessionIds updated only after successful commit.
- BuildIndexResult gains skipped/skippedFiles for observability.
- Skill finalize failure now fails the build (was silently warned).
- Checkpoint changed to PASSIVE (TRUNCATE reserved for maintenance/exit).
- Skill buildIndex returns { deferred, reason } on lease contention;
indexer-service reschedules the build (deferredRetryMs) without publishing
a heartbeat (so the build-deferred state is visible to cross-process
arbitration).
- Service publishes heartbeat immediately on start() for correct arbitration.
Tests:
- tests/write-transaction.test.mjs: single-shot execution, diagnostics
propagation, auto-rolled-back transaction detected, rollback failure
captured as metadata, BEGIN IMMEDIATE semantics.
- tests/writer-lease.test.mjs: acquire/release, contention returns null,
bounded wait with release during budget.
- tests/app-writer-lease.test.mjs: better-sqlite3 adapter integration.
- tests/app-rollback-guard.test.mjs: rewritten — transient BUSY recovered
by coordinator, persistent BUSY skips file, begin-busy aborts build,
live-transaction aborts build, phantom affectedSessionIds prevented.
- tests/daemon-arbitration.test.mjs: skill defers to fresh app heartbeat,
builds when heartbeat is stale.
- tests/app-indexer-service.test.mjs: new cases for deferred-retry
scheduling and immediate heartbeat on start.
- app/tests/electron-concurrency.mjs + child: dual-child IPC structure for
real better-sqlite3 contention (holder acquires lock → build child starts
→ delayed release → result collected; persistent contention bounded).
ADR-0006 updated to reflect the implemented design.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
96 lines
2.6 KiB
TypeScript
96 lines
2.6 KiB
TypeScript
// Bounded retry policy above the transaction primitive. Callers opt in only for
|
|
// idempotent work; BEGIN contention and an uncertain/live transaction are never
|
|
// retried here.
|
|
|
|
import { runWriteTransaction, type WriteTxDb, type WriteTxOptions } from './tx.ts';
|
|
|
|
interface TransactionDiagnostics {
|
|
phase?: string;
|
|
code?: string | null;
|
|
transactionActive?: boolean | null;
|
|
attempts?: number;
|
|
}
|
|
|
|
export interface WriteRetryOptions {
|
|
maxAttempts?: number;
|
|
budgetMs?: number;
|
|
retryDelayMs?: number;
|
|
now?: () => number;
|
|
sleep?: (ms: number) => void;
|
|
}
|
|
|
|
function diagnostics(error: unknown): TransactionDiagnostics | null {
|
|
if (!error || typeof error !== 'object') return null;
|
|
return (error as { obelisk?: TransactionDiagnostics }).obelisk ?? null;
|
|
}
|
|
|
|
function isBusyCode(code: unknown): boolean {
|
|
return typeof code === 'string' && code.startsWith('SQLITE_BUSY');
|
|
}
|
|
|
|
function syncSleep(ms: number): void {
|
|
if (ms <= 0) return;
|
|
try {
|
|
Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, ms);
|
|
} catch {
|
|
// Bounded attempts still prevent an infinite retry loop.
|
|
}
|
|
}
|
|
|
|
export function isBeginBusyFailure(error: unknown): boolean {
|
|
const info = diagnostics(error);
|
|
return (
|
|
info?.phase === 'begin' &&
|
|
isBusyCode(info.code) &&
|
|
info.transactionActive === false
|
|
);
|
|
}
|
|
|
|
export function hasUnusableTransaction(error: unknown): boolean {
|
|
const info = diagnostics(error);
|
|
return Boolean(info && info.transactionActive !== false);
|
|
}
|
|
|
|
export function isRetryableWriteFailure(error: unknown): boolean {
|
|
const info = diagnostics(error);
|
|
return (
|
|
(info?.phase === 'work' || info?.phase === 'commit') &&
|
|
isBusyCode(info.code) &&
|
|
info.transactionActive === false
|
|
);
|
|
}
|
|
|
|
export function runWithWriteRetry<T>(operation: () => T, {
|
|
maxAttempts = 3,
|
|
budgetMs = 1000,
|
|
retryDelayMs = 25,
|
|
now = Date.now,
|
|
sleep = syncSleep,
|
|
}: WriteRetryOptions = {}): T {
|
|
const startedAt = now();
|
|
for (let attempt = 1; ; attempt += 1) {
|
|
try {
|
|
return operation();
|
|
} catch (error) {
|
|
const info = diagnostics(error);
|
|
if (info) info.attempts = attempt;
|
|
if (!isRetryableWriteFailure(error) || attempt >= maxAttempts) throw error;
|
|
const remaining = budgetMs - (now() - startedAt);
|
|
if (remaining <= 0) throw error;
|
|
sleep(Math.min(retryDelayMs * attempt, remaining));
|
|
}
|
|
}
|
|
}
|
|
|
|
export function runRetryableWriteTransaction<T>(
|
|
db: WriteTxDb,
|
|
work: () => T,
|
|
transactionOptions: WriteTxOptions = {},
|
|
retryOptions: WriteRetryOptions = {},
|
|
): T {
|
|
return runWithWriteRetry(
|
|
() => runWriteTransaction(db, work, transactionOptions),
|
|
retryOptions,
|
|
);
|
|
}
|