/** * The ACP snapshot suite factory (REPLAY by default, keyless). A suite is a * scenario table plus a snapshots directory: each scenario under * `//` ships an `input.json` (the client stdin script) and * a `session.jsonl` fixture; replay boots the real agent subprocess * (./harness.ts), drives it, and diffs the normalized stdout transcript * against the committed `stdout.golden.jsonl`. For model scenarios it ALSO * checks the re-persisted session log — against the `session.jsonl` fixture * itself, not a separate golden: the fixture doubles as the replay source * (recorded scenarios) and the expected produced log (both sides normalized * before comparing). * * Request-header content is pinned by exactly ONE scenario per HEADER CLASS — * scenarios that boot the same config compose the same header. Every JSONL * fixture scrubs the system prompt to `{{system}}`; each class's pinning * scenario stores the readable prompt in `system-prompt.golden.md` and keeps its full * tool schemas in `session.jsonl`, while every other fixture also scrubs tools * to `{{tools}}`. A per-run uniformity guard compares both artifacts against * every live header and forbids unrepresented changed headers (see the * pinned-header RFC, * docs/rfc/implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md). * * `pnpm run test:snapshot:record` (DSH_SNAPSHOT=record + -u) re-records the * `session.jsonl` fixtures against the real API and refreshes the stdout golden * in one pass. `pnpm run test:snapshot:refresh` (DSH_SNAPSHOT=refresh) instead * replays the committed model scripts keylessly and writes the current stdout * + persisted-log goldens back without calling a live LLM. The caller resolves * that env into {@link SnapshotSuiteOptions} (env reading stays at the suite * edge, not in this library). * * @module @deepseek-ai/dsh-acp-snapshot/suite */ import { readFile, readdir, rm, writeFile } from 'node:fs/promises' import { existsSync } from 'node:fs' import { join } from 'node:path' import { describe, expect, it } from 'vitest' import { type AgentUnderTest, type HarvestedLog, type InputScript, runScenario } from './harness.ts' import { type NormalizeContext, normalizeSessionLog, normalizeStdout, scrubRequestHeaders, scrubSystemPrompts, } from './normalize.ts' /** The readable system-prompt snapshot beside each header-pinning fixture. */ const SYSTEM_PROMPT_SNAPSHOT = 'system-prompt.golden.md' /** A snapshot scenario and how its fixtures are produced. */ export interface Scenario { name: string /** Whether the scenario drives at least one model turn (so a JSONL golden applies). */ hasModelTurn: boolean /** * Whether the run persists a comparable session log to diff against the * `session.jsonl` fixture. Defaults to {@link hasModelTurn} (a model turn * always produces a log worth comparing). Set it independently for a scenario * that produces a non-trivial log WITHOUT a model turn — e.g. a prompt blocked * by a `UserPromptSubmit` hook, which opens a `rejected` turn carrying `hook/*` * events but never calls the model. */ comparesLog?: boolean /** * Whether `test:snapshot:record` regenerates this scenario's `session.jsonl` * from the LIVE API. `recorded` scenarios are model-driven and reproducible; * `authored` scenarios (fixtures hand-written or hand-harvested — e.g. a * provider error or a cancel the live API can't be coaxed into * deterministically, a deterministic hook scenario, or a scripted repetition * a live model won't reproduce) are NEVER re-recorded. */ recorded: boolean /** * Whether replay is driven by a hand-written `replay.override.json` sidecar * (a `ReplayEntry[]` that REPLACES the script derived from `session.jsonl`) * — the throw/hang cases chunks cannot express. The fixture guard requires * the sidecar exactly when this is set: the harness forwards the file purely * on existence, so an unregistered stray sidecar would silently replace the * derived script — the guard fails loud on either mismatch. Defaults to * false (replay derives from the fixture's `assistant/chunk` events). */ overridden?: boolean /** * Whether THIS scenario pins its header class's model-facing request-header * content. Its actual composed prompt is maintained as a readable * `system-prompt.golden.md`; its JSONL keeps full tool schemas but stores the prompt * as `{{system}}`. Every other scenario of the class stores tools as * `{{tools}}` too ({@link scrubRequestHeaders}). A prompt or tool-schema * change therefore shows up in one focused artifact per class, not every * session fixture. One pin per class suffices because * header composition is class-uniform (parent, spawn child, and fork child * all compose the same prompt-modulo-cwd and the same tools) — and that * premise is ASSERTED, not assumed: every non-pinning run's live headers * must equal its class's pinned fixture's (normalized), so a * session-dependent header (say, a restricted subagent toolset) fails loud * until it gets its own pinning scenario. * Defaults to false. */ pinsHeader?: boolean /** * How many changed `request/header` snapshots this PINNING scenario's primary * fixture legitimately carries (default 0). Their full prompt text is kept in * the readable Markdown pin; any other count fails. Meaningless off the pin. */ expectedHeaderChanges?: number /** * Which header-composition class this scenario belongs to. Scenarios that * boot the same config compose the same header; each class has exactly one * {@link pinsHeader} scenario, and the uniformity guard compares every * other member against ITS class's pin. Defaults to `'default'`; a * scenario booting an alternate config ({@link configPath}) whose tool * list or prompt sections differ by construction carries its own class. */ headerClass?: string /** * Alternate LIVE config path (absolute) this scenario boots instead of * {@link AgentUnderTest.configPath} — an overlay composing a different * tree (its basename must still end in `cordis.yml` so the bin's replay * swap finds the sibling `*cordis.snapshot.yml`). A scenario whose * overlay changes the composed header also needs its own * {@link headerClass}. */ configPath?: string } /** One suite's inputs: the agent to boot, where its fixtures live, and its scenario table. */ export interface SnapshotSuiteOptions { /** The agent composition every scenario boots. */ agent: AgentUnderTest /** Absolute path of the suite's `snapshots/` directory (one subdir per scenario). */ snapshotsDir: string /** The scenario table; exactly one entry per header class must set `pinsHeader`. */ scenarios: Scenario[] /** * `replay` (keyless, the default tier), `record` (live API; re-records the * `recorded` scenarios' fixtures and refreshes the Vitest goldens under * `--update`), or `refresh` (keyless replay that rewrites stdout goldens and * comparable session fixtures from the replay run). The caller derives this * from `$DSH_SNAPSHOT` — env reading stays outside this library. */ mode: 'replay' | 'record' | 'refresh' } /** * Validate and order a scenario directory's session-fixture filenames. * * The primary fixture is always `session.jsonl`; child sessions are discovered * from contiguous `session.1.jsonl` … filenames. The directory is the source of * truth, so scenario tables do not duplicate a child count that can drift from * the files. A session-like JSONL with any other suffix fails loud. * * @param names File names in one scenario directory. * @returns The primary and child fixture names in replay/harvest order. */ export function sessionFixtureNames(names: readonly string[]): string[] { if (!names.includes('session.jsonl')) throw new Error('missing session.jsonl') const children: { name: string; index: number }[] = [] for (const name of names) { if (name === 'session.jsonl') continue if (!name.startsWith('session.') || !name.endsWith('.jsonl')) continue const match = /^session\.([1-9]\d*)\.jsonl$/.exec(name) if (match === null) throw new Error(`invalid child session fixture name: ${name}`) children.push({ name, index: Number(match[1]) }) } children.sort((a, b) => a.index - b.index) for (const [offset, child] of children.entries()) { const expected = offset + 1 if (child.index !== expected) { throw new Error(`child session fixtures must be contiguous: expected session.${expected}.jsonl, found ${child.name}`) } } return ['session.jsonl', ...children.map(child => child.name)] } /** Read one scenario directory's validated session-fixture inventory. */ async function sessionFixtures(dir: string): Promise { const entries = await readdir(dir, { withFileTypes: true }) return sessionFixtureNames(entries.filter(entry => entry.isFile()).map(entry => entry.name)) } /** * Derive the {@link NormalizeContext} for a `session.jsonl` fixture from its own * header line (`{ type: 'session', id, cwd }`). A committed fixture carries the * session id and cwd of the run that harvested it — different from the live * replay run — so normalizing it against the live run's ctx would leave those * recorded values unscrubbed. Reading them from the header scrubs the fixture's * own id/cwd to the same `{{sessionId}}`/`{{cwd}}` tokens the replay output gets. * An authored fixture whose header is already normalized (`id:'{{sessionId}}'`, * `cwd:'{{cwd}}'`) yields those tokens as the volatile values, so scrubbing them * is an idempotent no-op. A header with no `cwd` falls back to a sentinel that * cannot occur in a log (NOT `''`, which `String.split` would match on every * character boundary and corrupt the output). * * @param fixture The committed `session.jsonl` content. * @returns The fixture's own volatile values, ready for {@link normalizeSessionLog}. */ export function fixtureContext(fixture: string): NormalizeContext { const firstLine = fixture.split('\n').find(line => line.trim().length > 0) ?? '{}' const header = JSON.parse(firstLine) as { id?: unknown; cwd?: unknown } return { sessionIds: typeof header.id === 'string' ? [header.id] : [], cwd: typeof header.cwd === 'string' ? header.cwd : '\0no-cwd\0', } } /** * The `data.header` payload of every `request/header` event in a session * JSONL, in log order, with the log's volatile values scrubbed first * ({@link normalizeSessionLog}) so headers harvested from different runs — * each embedding its own temp cwd in the composed prompt — compare on equal * footing. * * @param rawLog The session `.jsonl` content to extract headers from. * @param ctx The volatile values of the run that produced it. * @returns The normalized `data.header` payloads, in log order. */ export function normalizedHeaders(rawLog: string, ctx: NormalizeContext): unknown[] { return normalizeSessionLog(rawLog, ctx) .split('\n') .filter(line => line.trim().length > 0) .map(line => JSON.parse(line) as { type?: unknown; data?: { header?: unknown } }) .filter(record => record.type === 'request/header') .map(record => record.data?.header) } /** * The normalized string-valued system prompts carried by request headers in a * session JSONL, in log order. Headers without a string prompt are omitted so * callers can assert one prompt per header explicitly. * * @param rawLog The session `.jsonl` content to inspect. * @param ctx The volatile values of the run that produced it. * @returns The normalized system prompts, in header order. */ export function normalizedSystemPrompts(rawLog: string, ctx: NormalizeContext): string[] { return normalizedHeaders(rawLog, ctx).flatMap((header) => { if (header === null || typeof header !== 'object') return [] const system = (header as { system?: unknown }).system return typeof system === 'string' ? [system] : [] }) } /** * Render a normalized prompt as a repository-friendly Markdown snapshot. * Prompt text is unchanged except that a missing terminal newline is added so * the committed file follows the repository newline contract. * * @param prompt The normalized system prompt. * @param changes Full normalized prompts from later changed-header snapshots. * @returns Markdown snapshot text ending in a newline. */ export function formatSystemPromptSnapshot( prompt: string, changes: readonly string[] = [], ): string { let snapshot = prompt.endsWith('\n') ? prompt : `${prompt}\n` for (const [index, change] of changes.entries()) { snapshot += `\n\n\n` snapshot += change.endsWith('\n') ? change : `${change}\n` } return snapshot } /** Return the initial-prompt portion of a possibly multi-header snapshot. */ function initialSystemPromptSnapshot(snapshot: string): string { const marker = snapshot.indexOf('\n