Files
obelisk/scripts/providers/types.ts
T
tommy0103 2b30d9596d feat(providers): add pure claude adapter with record-stream golden tests (5b-1)
claude.parse mirrors indexJsonl line-for-line but yields IndexRecords (no db).
Adds MessageTurnDurationRecord op. Purely additive — buildIndex still uses the
old path. 111/111 green.
2026-07-08 18:27:28 +08:00

217 lines
7.7 KiB
TypeScript

// Phase 5 target contract (see docs/adr/0001).
//
// The indexing layer splits along two orthogonal axes:
// - Provider axis: pure per-source adapters (claude, codex, later opencode,
// pi, …) that discover their own work and parse it into records. A source is
// NOT assumed to be a single JSONL file — an adapter may read a SQLite store,
// a directory tree, etc. So discovery, change-detection, and resume cursoring
// are all adapter-owned and format-specific.
// - Persist axis: one shared, provider- and binding-agnostic orchestration
// that consumes the records and writes them (index_state, FTS, upsert).
//
// This file defines only the shapes crossing that boundary. Record fields mirror
// the columns in scripts/schema.sql; keep them in sync. Types only — no runtime
// code — so consumers must import with `import type`.
// Opaque per-unit resume/watermark token. The orchestration stores it verbatim
// (in index_state) and hands it back on the next run; ONLY the adapter that
// produced it interprets it. A JSONL adapter might encode `"${mtime}:${lines}"`;
// a SQLite-backed adapter might encode a rowid or timestamp high-water mark.
export type Cursor = string | null;
// One unit of work an adapter has discovered. It is not necessarily a file: for
// a file-based source `key` is the path; for a DB-backed source it might be
// `"${dbPath}#${internalId}"`. `meta` carries adapter-private data (e.g. the
// resolved file path or source handle) that the orchestration passes back to
// parse() untouched.
export interface IndexUnit {
/** Stable identity used as the index_state cursor key. */
key: string;
/** Session id this unit indexes into. */
sessionId: string;
/** Project slug (dash-encoded path), when the source exposes one. */
project?: string;
/** Set for subagent transcripts, whose messages carry an agent id. */
isSubagent?: boolean;
agentId?: string;
/** Adapter-private payload, opaque to the orchestration. */
meta?: unknown;
}
/** Context the orchestration provides to discovery. */
export interface DiscoverContext {
/** Look up the cursor persisted for a unit key on a previous run. */
lastCursor(key: string): Cursor;
/** When set (daemon changed-path mode), restrict discovery to these paths. */
changedPaths?: string[];
}
/** Discriminated union of everything an adapter's parse can emit. Each record
* kind maps to one schema table (see scripts/schema.sql); `delete-session` is
* the exception — a retraction op, not a table. Sources without a table
* (history.jsonl, codex session_index.jsonl) are not records: adapters fold them
* into the SessionRecord they already emit. */
export type IndexRecord =
| SessionRecord
| MessageRecord
| ToolCallRecord
| ToolResultRecord
| SummaryRecord
| SubagentRecord
| WorkflowRecord
| WorkflowAgentRecord
| MessageTurnDurationRecord
| DeleteSessionRecord;
export interface MessageRecord {
kind: 'message';
uuid: string;
session_id: string;
type: string;
parent_uuid: string | null;
timestamp: string | null;
role: string | null;
text: string | null;
content_type: string | null;
is_meta: 0 | 1;
model: string | null;
is_sidechain: 0 | 1;
agent_id: string | null;
input_tokens: number | null;
output_tokens: number | null;
cwd: string | null;
skill: string | null;
source: string;
}
export interface ToolCallRecord {
kind: 'tool_call';
id: string;
message_uuid: string;
session_id: string;
name: string;
input_json: string;
file_path: string | null;
}
export interface ToolResultRecord {
kind: 'tool_result';
tool_use_id: string;
message_uuid: string;
session_id: string;
content: string;
file_path: string | null;
is_error: 0 | 1;
}
export interface SummaryRecord {
kind: 'summary';
id: string;
session_id: string;
timestamp: string | null;
source: string;
content: string;
}
export interface SubagentRecord {
kind: 'subagent';
agent_id: string;
session_id: string;
parent_tool_use_id: string | null;
agent_type: string | null;
description: string | null;
duration_ms: number | null;
total_tokens: number | null;
}
// A workflow run. `agent_count` is intentionally absent: it is a derived
// aggregate (COUNT of workflow_agents for this run) that persist computes, since
// the agents may be indexed on different runs than the workflow metadata.
export interface WorkflowRecord {
kind: 'workflow';
run_id: string;
session_id: string;
task_id: string | null;
script: string | null;
result_json: string | null;
timestamp: string | null;
duration_ms: number | null;
total_tokens: number | null;
status: string | null;
workflow_name: string | null;
}
// One workflow agent. A single row is contributed by TWO independent units, in
// any order: the subagent .meta.json unit fills agent_type/description; the
// workflow run json unit fills phase/label/model/state/duration_ms/tokens/
// tool_calls. So every optional field a unit does not know is omitted, and
// persist merges column-wise (ON CONFLICT(agent_id) DO UPDATE SET
// col=COALESCE(excluded.col, col)). All contributors MUST use the same unified
// agent_id key so the merge lands on the same row.
export interface WorkflowAgentRecord {
kind: 'workflow_agent';
agent_id: string;
run_id: string;
session_id: string;
agent_type?: string | null;
description?: string | null;
phase?: string | null;
label?: string | null;
model?: string | null;
state?: string | null;
duration_ms?: number | null;
tokens?: number | null;
tool_calls?: number | null;
}
// Update op (not a table): sets messages.turn_duration_ms for a message that was
// (or will be) inserted by a separate line, possibly on a different run. Persist
// applies it as a targeted UPDATE, so it never clobbers other message columns.
export interface MessageTurnDurationRecord {
kind: 'message-turn-duration';
uuid: string;
turn_duration_ms: number;
}
// Retraction op (not a table). The adapter emits this when a previously-indexed
// session must be removed — e.g. a Codex guardian/auto-review thread. Persist
// executes the cascade delete across all tables for that session.
export interface DeleteSessionRecord {
kind: 'delete-session';
sessionId: string;
}
// Session-level aggregate. Emitted once, after the unit's records are produced,
// because started_at/ended_at/message_count are computed across the stream.
// title/ended_at may be enriched by the adapter from source-specific auxiliary
// files (claude history.jsonl, codex session_index.jsonl); persist upserts with
// fill-if-null (COALESCE) so those never clobber a value already present.
// project_path is NOT set here — the orchestration's global pass derives it from
// persisted message cwds (refreshSessionProjectPaths).
export interface SessionRecord {
kind: 'session';
id: string;
title: string | null;
project: string | null;
started_at: string | null;
ended_at: string | null;
git_branch: string | null;
version: string | null;
message_count: number;
jsonl_path: string;
source: string;
}
// A transcript source. Pure: it never touches the Obelisk database. It owns its
// own discovery, change-detection, and resume cursoring, because those are
// format-specific (file mtime, DB watermark, …). `parse` is a generator that
// yields records for one unit and RETURNS the new cursor to persist.
export interface Provider {
/** Stable source tag stored on rows, e.g. 'claude' | 'codex'. */
readonly name: string;
/** Discover units needing (re)indexing, using stored cursors to detect change. */
discover(ctx: DiscoverContext): IndexUnit[];
/** Stream records for one unit resuming from `cursor`; return the new cursor. */
parse(unit: IndexUnit, cursor: Cursor): Generator<IndexRecord, Cursor>;
}