feat(providers): migrate codex indexing to adapter + persist, remove legacy indexers (Phase 5c)
Complete the skill-side provider migration: codex now goes through a pure adapter
and the shared persist layer, and the two original monolithic indexers are gone.
New:
- scripts/providers/codex.ts — pure codex adapter. Full-reparse (buffers the whole
file) because the event_msg↔response_item dedup needs whole-file, bidirectional
knowledge; emits SessionRecord with countMode 'total'. Handles guardian threads
(→ delete-session), agent spawns/tool calls (→ tool_call/subagent), token_count
(patched onto the message record) and task_complete (→ message-turn-duration).
Contract:
- SessionRecord.countMode ('total' | 'delta') tells persist whether to replace or
accumulate message_count — claude is line-incremental (delta), codex full-reparse
(total). SubagentRecord non-key fields are optional; persist merges them
column-wise with COALESCE. MessageTurnDurationRecord.turn_duration_ms is nullable.
Orchestration:
- buildIndex's codex branch parses via the adapter and writes via persist. An
unchanged file is skipped but still swept for stale guardian rows (routed through
persist as a delete-session), preserving prior behavior.
Cleanup:
- Remove the now-unused indexJsonl, indexCodexJsonl, deleteCodexThreadRows and
upsertCodexSubagent — their semantics now live in the adapters + persist.
indexer.mjs drops from ~840 to 428 lines. Codex pure helpers stay exported for
codex.ts and the guardian sweep (physical move deferred to the app-side reorg).
- Migrate the upsert drift test off indexJsonl to the claude.parse + persist path,
keeping the rowid-stability and count-replace regression guards.
Tests: tests/codex-parse.test.mjs (record-stream golden: dedup, tools, token patch,
turn-duration, guardian→delete) and tests/codex-index.test.mjs (full buildIndex
path: fresh build + incremental full-reparse, total-count replace, no duplicates).
Verified equivalent on the real ~/.obelisk index: codex messages 82476 and
subagents 522 identical before/after, zero guardian leakage; real incremental
confirmed (touch a codex file → reparsed idempotently, unchanged files skipped).
lint + typecheck clean, 119/119.
This commit is contained in:
@@ -1,64 +1,54 @@
|
||||
// Regression test for the indexer silent-drift fix.
|
||||
//
|
||||
// scripts/indexer.mjs and app/indexer.js had diverged in indexJsonl's message
|
||||
// write: scripts used INSERT OR REPLACE (churns rowid → FTS churn) and always
|
||||
// carried the previous message_count forward (inflating it on a full re-scan),
|
||||
// while app used ON CONFLICT DO UPDATE and reset the count when skip===0. app's
|
||||
// semantics are canonical; this pins them so the two cannot drift again and so
|
||||
// the Phase 5 provider-adapter merge inherits one known-correct behavior.
|
||||
// Regression test for the message write semantics (formerly the indexJsonl
|
||||
// INSERT-OR-REPLACE vs ON-CONFLICT drift; now enforced through the shared
|
||||
// persist layer). Re-indexing a claude session must upsert messages (stable
|
||||
// rowid, no FTS churn) and, because claude parses fresh from an empty cursor
|
||||
// (countMode 'total'), must replace message_count rather than accumulate.
|
||||
|
||||
import { test } from 'node:test';
|
||||
import assert from 'node:assert/strict';
|
||||
import { createRequire } from 'node:module';
|
||||
import { mkdtempSync, writeFileSync } from 'node:fs';
|
||||
import { mkdtempSync, writeFileSync, readFileSync } from 'node:fs';
|
||||
import { tmpdir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
|
||||
import { indexJsonl } from '../scripts/indexer.mjs';
|
||||
import { parse } from '../scripts/providers/claude.ts';
|
||||
import { persist } from '../scripts/persist.ts';
|
||||
|
||||
const require = createRequire(import.meta.url);
|
||||
const { DatabaseSync } = require('node:sqlite');
|
||||
const SCHEMA = require('node:fs').readFileSync(new URL('../scripts/schema.sql', import.meta.url), 'utf8');
|
||||
const SCHEMA = readFileSync(new URL('../scripts/schema.sql', import.meta.url), 'utf8');
|
||||
|
||||
function writeSessionJsonl() {
|
||||
function fixtureUnit() {
|
||||
const dir = mkdtempSync(join(tmpdir(), 'obelisk-drift-'));
|
||||
const jsonlPath = join(dir, 'sid-drift.jsonl');
|
||||
const jsonlPath = join(dir, 'sess.jsonl');
|
||||
const lines = [
|
||||
{ uuid: 'u-1', type: 'user', timestamp: '2026-06-10T10:00:00Z', cwd: '/tmp/proj', message: { role: 'user', content: 'first question' } },
|
||||
{ uuid: 'a-1', type: 'assistant', timestamp: '2026-06-10T10:00:05Z', message: { role: 'assistant', model: 'claude-opus', content: 'first answer' } },
|
||||
{ uuid: 'u-2', type: 'user', timestamp: '2026-06-10T10:00:10Z', cwd: '/tmp/proj', message: { role: 'user', content: 'second question' } },
|
||||
];
|
||||
writeFileSync(jsonlPath, lines.map(l => JSON.stringify(l)).join('\n') + '\n');
|
||||
return jsonlPath;
|
||||
return { key: jsonlPath, sessionId: 'sid-drift', project: 'quiet-zero' };
|
||||
}
|
||||
|
||||
test('re-indexing a session upserts messages (stable rowid) and does not inflate message_count', () => {
|
||||
test('re-indexing upserts messages (stable rowid) and replaces message_count', () => {
|
||||
const db = new DatabaseSync(':memory:');
|
||||
db.exec(SCHEMA);
|
||||
const fi = { path: writeSessionJsonl(), sessionId: 'sid-drift', project: 'quiet-zero' };
|
||||
|
||||
indexJsonl(db, fi);
|
||||
const unit = fixtureUnit();
|
||||
|
||||
persist(db, unit, parse(unit, null));
|
||||
const countAfterFirst = db.prepare('SELECT message_count FROM sessions WHERE id=?').get('sid-drift').message_count;
|
||||
const rowidAfterFirst = db.prepare('SELECT rowid FROM messages WHERE uuid=?').get('u-1').rowid;
|
||||
const totalMessages = db.prepare('SELECT COUNT(*) AS c FROM messages').get().c;
|
||||
assert.equal(countAfterFirst, 3, 'three user/assistant messages counted');
|
||||
assert.equal(totalMessages, 3);
|
||||
assert.equal(countAfterFirst, 3);
|
||||
assert.equal(db.prepare('SELECT COUNT(*) AS c FROM messages').get().c, 3);
|
||||
|
||||
// Simulate a fresh full re-scan (force / lost index_state): skip resets to 0.
|
||||
db.prepare('DELETE FROM index_state').run();
|
||||
indexJsonl(db, fi);
|
||||
// Re-index the same session from scratch (fresh parse → countMode 'total').
|
||||
persist(db, unit, parse(unit, null));
|
||||
|
||||
const countAfterSecond = db.prepare('SELECT message_count FROM sessions WHERE id=?').get('sid-drift').message_count;
|
||||
const rowidAfterSecond = db.prepare('SELECT rowid FROM messages WHERE uuid=?').get('u-1').rowid;
|
||||
const totalAfterSecond = db.prepare('SELECT COUNT(*) AS c FROM messages').get().c;
|
||||
|
||||
// message_count is reset+recounted, not accumulated (would be 6 under the old bug).
|
||||
assert.equal(countAfterSecond, 3, 'message_count must not inflate on re-scan');
|
||||
// No duplicate rows.
|
||||
assert.equal(totalAfterSecond, 3);
|
||||
// Upsert preserves rowid; INSERT OR REPLACE would have churned it.
|
||||
assert.equal(rowidAfterSecond, rowidAfterFirst, 'upsert must preserve message rowid (no REPLACE churn)');
|
||||
assert.equal(countAfterSecond, 3, 'message_count is replaced, not accumulated (would be 6 under the old bug)');
|
||||
assert.equal(db.prepare('SELECT COUNT(*) AS c FROM messages').get().c, 3, 'no duplicate rows');
|
||||
assert.equal(rowidAfterSecond, rowidAfterFirst, 'upsert preserves rowid (INSERT OR REPLACE would churn it)');
|
||||
|
||||
db.close();
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user