Implement the full ADR-0006 plan: three-layer separation of transaction
correctness, retry policy, and cross-process writer coordination.
Layer 1 — scripts/tx.ts (transaction correctness):
- runWriteTransaction executes work exactly once; no internal retry.
- BEGIN IMMEDIATE takes the write lock up front (avoids SQLITE_BUSY_SNAPSHOT).
- Guarded rollback: checks inTransaction() via adapter before attempting
ROLLBACK; never masks the primary exception.
- WriteTxDiagnostics attached to errors: phase, code, label,
rollbackSucceeded, rollbackError, transactionActive.
- Binding adapters (betterSqliteTransactionAdapter, nodeSqliteTransactionAdapter)
mapping better-sqlite3's `.inTransaction` and node:sqlite's `.isTransaction`.
- configureConnection centralizes WAL + synchronous + busy_timeout.
Layer 2 — scripts/write-coordinator.ts (retry policy):
- runRetryableWriteTransaction: bounded retry with total time budget.
- Only retries when the transaction confirmed ended (transactionActive=false)
and the error is SQLITE_BUSY during work/commit phase.
- BEGIN-phase BUSY = abort entire build (isBeginBusyFailure); the caller
returns `{ deferred: true, reason: 'writer_busy' }` instead of waiting.
- hasUnusableTransaction detects a still-active transaction after failure;
aborts the build immediately, never retries.
Layer 3 — scripts/writer-lease.ts (cross-process coordination):
- acquireWriterLease: dedicated writer.lock.sqlite with busy_timeout=0 +
BEGIN IMMEDIATE. Non-blocking attempt; bounded wait with retryDelayMs.
- writerLockPathFor derives lock path from the target DB path.
- Lease held for the entire build; released on completion or failure.
- Lock DB uses DELETE journal (not WAL); crash/close auto-releases.
- All consumers obey: skill acquires at build start (returns deferred if
unavailable); app daemon (via worker) acquires for its build cycle.
Build semantics changes:
- affectedSessionIds updated only after successful commit.
- BuildIndexResult gains skipped/skippedFiles for observability.
- Skill finalize failure now fails the build (was silently warned).
- Checkpoint changed to PASSIVE (TRUNCATE reserved for maintenance/exit).
- Skill buildIndex returns { deferred, reason } on lease contention;
indexer-service reschedules the build (deferredRetryMs) without publishing
a heartbeat (so the build-deferred state is visible to cross-process
arbitration).
- Service publishes heartbeat immediately on start() for correct arbitration.
Tests:
- tests/write-transaction.test.mjs: single-shot execution, diagnostics
propagation, auto-rolled-back transaction detected, rollback failure
captured as metadata, BEGIN IMMEDIATE semantics.
- tests/writer-lease.test.mjs: acquire/release, contention returns null,
bounded wait with release during budget.
- tests/app-writer-lease.test.mjs: better-sqlite3 adapter integration.
- tests/app-rollback-guard.test.mjs: rewritten — transient BUSY recovered
by coordinator, persistent BUSY skips file, begin-busy aborts build,
live-transaction aborts build, phantom affectedSessionIds prevented.
- tests/daemon-arbitration.test.mjs: skill defers to fresh app heartbeat,
builds when heartbeat is stale.
- tests/app-indexer-service.test.mjs: new cases for deferred-retry
scheduling and immediate heartbeat on start.
- app/tests/electron-concurrency.mjs + child: dual-child IPC structure for
real better-sqlite3 contention (holder acquires lock → build child starts
→ delayed release → result collected; persistent contention bounded).
ADR-0006 updated to reflect the implemented design.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
92 lines
2.6 KiB
TypeScript
92 lines
2.6 KiB
TypeScript
// Cross-process single-writer lease for a complete Obelisk index build. The
|
|
// lock lives in a dedicated SQLite database so node:sqlite and better-sqlite3
|
|
// share identical locking semantics on every supported platform.
|
|
|
|
import { mkdirSync } from 'node:fs';
|
|
import { dirname, join } from 'node:path';
|
|
|
|
export interface WriterLeaseDb {
|
|
exec(sql: string): unknown;
|
|
close(): void;
|
|
}
|
|
|
|
export interface WriterLease {
|
|
release(): void;
|
|
}
|
|
|
|
export interface AcquireWriterLeaseOptions {
|
|
lockPath: string;
|
|
openDb: (path: string) => WriterLeaseDb;
|
|
waitMs?: number;
|
|
retryDelayMs?: number;
|
|
now?: () => number;
|
|
sleep?: (ms: number) => void;
|
|
}
|
|
|
|
const BUSY_MESSAGE = /SQLITE_BUSY|database is locked|database is busy/i;
|
|
|
|
function isBusy(error: unknown): boolean {
|
|
const raw = error as { code?: unknown; errcode?: unknown; message?: unknown } | null;
|
|
const code = raw?.code ?? raw?.errcode;
|
|
return (
|
|
(typeof code === 'string' && code.startsWith('SQLITE_BUSY')) ||
|
|
(typeof raw?.message === 'string' && BUSY_MESSAGE.test(raw.message))
|
|
);
|
|
}
|
|
|
|
function syncSleep(ms: number): void {
|
|
if (ms <= 0) return;
|
|
try {
|
|
Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, ms);
|
|
} catch {
|
|
// If synchronous sleeping is unavailable, the bounded attempt count below
|
|
// still prevents an infinite acquisition loop.
|
|
}
|
|
}
|
|
|
|
export function writerLockPathFor(dbPath: string): string {
|
|
return join(dirname(dbPath), 'writer.lock.sqlite');
|
|
}
|
|
|
|
export function acquireWriterLease({
|
|
lockPath,
|
|
openDb,
|
|
waitMs = 0,
|
|
retryDelayMs = 25,
|
|
now = Date.now,
|
|
sleep = syncSleep,
|
|
}: AcquireWriterLeaseOptions): WriterLease | null {
|
|
mkdirSync(dirname(lockPath), { recursive: true });
|
|
const startedAt = now();
|
|
const maxAttempts = waitMs > 0 ? Math.ceil(waitMs / Math.max(1, retryDelayMs)) + 1 : 1;
|
|
|
|
for (let attempt = 0; attempt < maxAttempts; attempt += 1) {
|
|
const db = openDb(lockPath);
|
|
try {
|
|
db.exec('PRAGMA busy_timeout=0');
|
|
db.exec('BEGIN IMMEDIATE');
|
|
let released = false;
|
|
return {
|
|
release() {
|
|
if (released) return;
|
|
released = true;
|
|
try {
|
|
db.exec('ROLLBACK');
|
|
} catch {
|
|
// Closing the connection releases any remaining SQLite lock.
|
|
} finally {
|
|
db.close();
|
|
}
|
|
},
|
|
};
|
|
} catch (error) {
|
|
db.close();
|
|
if (!isBusy(error)) throw error;
|
|
const remaining = waitMs - (now() - startedAt);
|
|
if (remaining <= 0 || attempt + 1 >= maxAttempts) return null;
|
|
sleep(Math.min(retryDelayMs, remaining));
|
|
}
|
|
}
|
|
return null;
|
|
}
|