From 0893aaa7cb9ec075caf78e4198598da7088eba22 Mon Sep 17 00:00:00 2001 From: Tianyi Cui <53024+tianyicui@users.noreply.github.com> Date: Sun, 19 Jul 2026 17:45:49 +0800 Subject: [PATCH] test: rename golden snapshot files to expected --- .agents/skills/dsh-code-review/SKILL.md | 2 +- .../skills/dsh-find-simplifications/SKILL.md | 6 ++-- AGENTS.md | 4 +-- docs/i18n/terminology.md | 1 + ...js-expression-disabled-filesystem-tools.md | 8 ++--- docs/rfc/INDEX.md | 4 +-- .../2026-07-02-tool-render-intent-union.md | 2 +- .../2026-07-05-reconstructable-requests.md | 2 +- ...cutable-sdk-runtime-distribution.i18n.yaml | 4 +-- ...ile-executable-sdk-runtime-distribution.md | 2 +- ...-executable-sdk-runtime-distribution.zh.md | 2 +- .../feature/2026-06-29-todo-write-tool.md | 2 +- .../feature/2026-07-07-mcp-client-plugin.md | 2 +- ...6-07-09-bash-backed-grep-glob-discovery.md | 2 +- ...rop-unconsumed-llm-adapter-change-event.md | 2 +- ...-drop-unconsumed-llm-assembled-surfaces.md | 2 +- .../2026-07-04-prune-write-only-fs-surface.md | 2 +- ...2026-07-04-remove-agent-steering-mirror.md | 2 +- ...26-07-04-tighten-hook-protocol-contract.md | 2 +- ...-04-trim-acp-bridge-unreachable-surface.md | 2 +- .../testing/2026-06-19-acp-snapshot-tests.md | 12 +++---- ...-redundant-snapshot-log-expected-output.md | 35 +++++++++++++++++++ ...0-remove-redundant-snapshot-log-goldens.md | 35 ------------------- .../2026-06-22-subagent-snapshot-replay.md | 2 +- .../2026-07-04-hook-snapshot-matrix.md | 14 ++++---- ...6-07-04-single-source-acp-replay-config.md | 2 +- ...-request-header-content-in-one-scenario.md | 2 +- .../2026-07-08-shared-acp-snapshot-package.md | 6 ++-- ...-18-tui-terminal-state-snapshots.i18n.yaml | 4 +-- ...2026-07-18-tui-terminal-state-snapshots.md | 14 ++++---- ...6-07-18-tui-terminal-state-snapshots.zh.md | 14 ++++---- ...2026-06-20-drop-durable-step-boundaries.md | 2 +- .../2026-06-20-generic-tool-rendering.md | 2 +- docs/testing.md | 2 +- examples/acp-agent/tests/acp.snapshot.ts | 6 ++-- ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...pt.golden.md => system-prompt.expected.md} | 0 ...golden.json => tool-schemas.expected.json} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...pt.golden.md => system-prompt.expected.md} | 0 ...golden.json => tool-schemas.expected.json} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...pt.golden.md => system-prompt.expected.md} | 0 ...golden.json => tool-schemas.expected.json} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...pt.golden.md => system-prompt.expected.md} | 0 ...golden.json => tool-schemas.expected.json} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...pt.golden.md => system-prompt.expected.md} | 0 ...golden.json => tool-schemas.expected.json} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...pt.golden.md => system-prompt.expected.md} | 0 ...golden.json => tool-schemas.expected.json} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...pt.golden.md => system-prompt.expected.md} | 0 ...golden.json => tool-schemas.expected.json} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...pt.golden.md => system-prompt.expected.md} | 0 ...golden.json => tool-schemas.expected.json} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...pt.golden.md => system-prompt.expected.md} | 0 ...golden.json => tool-schemas.expected.json} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...pt.golden.md => system-prompt.expected.md} | 0 ...golden.json => tool-schemas.expected.json} | 0 .../headless-agent/tests/headless.snapshot.ts | 6 ++-- ...olden.jsonl => stream-json.expected.jsonl} | 0 examples/tui-agent/README.md | 2 +- ...minal.golden.txt => terminal.expected.txt} | 0 ...minal.golden.txt => terminal.expected.txt} | 0 ...minal.golden.txt => terminal.expected.txt} | 0 ...minal.golden.txt => terminal.expected.txt} | 0 ...minal.golden.txt => terminal.expected.txt} | 0 ...minal.golden.txt => terminal.expected.txt} | 0 ...minal.golden.txt => terminal.expected.txt} | 0 examples/tui-agent/tests/tui.snapshot.ts | 4 +-- .../core/agent-loop/tests/tool-calls.spec.ts | 2 +- packages/support/acp-snapshot/README.md | 8 ++--- packages/support/acp-snapshot/package.json | 2 +- packages/support/acp-snapshot/src/harness.ts | 6 ++-- packages/support/acp-snapshot/src/index.ts | 2 +- .../support/acp-snapshot/src/normalize.ts | 8 ++--- packages/support/acp-snapshot/src/suite.ts | 22 ++++++------ ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...pt.golden.md => system-prompt.expected.md} | 0 ...golden.json => tool-schemas.expected.json} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 ...pt.golden.md => system-prompt.expected.md} | 0 ...golden.json => tool-schemas.expected.json} | 0 ...out.golden.jsonl => stdout.expected.jsonl} | 0 .../support/acp-snapshot/tests/suite.spec.ts | 18 +++++----- packages/ui/acp/snapshot-replay.md | 6 ++-- packages/ui/tui/tests/headless-terminal.ts | 2 +- ... => advanced-cards-collapsed.expected.txt} | 0 ...t => advanced-cards-expanded.expected.txt} | 0 ...den.txt => code-mode-pending.expected.txt} | 0 ...xt => conversation-streaming.expected.txt} | 0 ....txt => cordis-tools-pending.expected.txt} | 0 ...den.txt => disposed-terminal.expected.txt} | 0 ... => dynamic-workflow-pending.expected.txt} | 0 ...olden.txt => errors-and-help.expected.txt} | 0 ...> question-dialog-validation.expected.txt} | 0 ...olden.txt => question-dialog.expected.txt} | 0 ...face-after-compaction-narrow.expected.txt} | 0 ...urface-after-compaction-wide.expected.txt} | 0 ...=> surface-before-compaction.expected.txt} | 0 ...en.txt => untrusted-controls.expected.txt} | 0 packages/ui/tui/tests/tui.snapshot.ts | 6 ++-- scripts/gen-doc-graphs.ts | 6 ++-- scripts/smoke-python-runtime.py | 2 +- scripts/verify-md-wrap.ts | 6 ++-- vitest.snapshot.config.ts | 4 +-- 157 files changed, 160 insertions(+), 159 deletions(-) create mode 100644 docs/rfc/implemented/testing/2026-06-20-remove-redundant-snapshot-log-expected-output.md delete mode 100644 docs/rfc/implemented/testing/2026-06-20-remove-redundant-snapshot-log-goldens.md rename examples/acp-agent/tests/snapshots/advanced-toolchain/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/advanced-toolchain/{system-prompt.golden.md => system-prompt.expected.md} (100%) rename examples/acp-agent/tests/snapshots/advanced-toolchain/{tool-schemas.golden.json => tool-schemas.expected.json} (100%) rename examples/acp-agent/tests/snapshots/bash-spill/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/both-mode-turn/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/both-mode-turn/{system-prompt.golden.md => system-prompt.expected.md} (100%) rename examples/acp-agent/tests/snapshots/both-mode-turn/{tool-schemas.golden.json => tool-schemas.expected.json} (100%) rename examples/acp-agent/tests/snapshots/cancel-tool-calls/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/cancel/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/code-mode-turn/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/code-mode-turn/{system-prompt.golden.md => system-prompt.expected.md} (100%) rename examples/acp-agent/tests/snapshots/code-mode-turn/{tool-schemas.golden.json => tool-schemas.expected.json} (100%) rename examples/acp-agent/tests/snapshots/code-mode-workspace-context/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/code-mode-workspace-context/{system-prompt.golden.md => system-prompt.expected.md} (100%) rename examples/acp-agent/tests/snapshots/code-mode-workspace-context/{tool-schemas.golden.json => tool-schemas.expected.json} (100%) rename examples/acp-agent/tests/snapshots/config-options/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/cordis-inspect-jsdoc/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/error-finish/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/escalation-approved/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/escalation-rejected/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/fs-edit/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/fs-policy-reject/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/fs-read-window/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/fs-read/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/fs-terminal-card/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/fs-write-overwrite/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/fs-write/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/handshake/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/hook-cc-posttool-block/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/hook-cc-posttool-context/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/hook-cc-pretool-ask/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/hook-cc-pretool-deny/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/hook-cc-promptsubmit-block/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/hook-cc-promptsubmit-context/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/hook-cc-stop-continue/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/hook-codex-posttool-block/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/hook-codex-posttool-context/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/hook-codex-pretool-block/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/hook-codex-promptsubmit-block/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/hook-codex-promptsubmit-context/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/hook-codex-stop-continue/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/model-switching/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/model-switching/{system-prompt.golden.md => system-prompt.expected.md} (100%) rename examples/acp-agent/tests/snapshots/model-switching/{tool-schemas.golden.json => tool-schemas.expected.json} (100%) rename examples/acp-agent/tests/snapshots/multi-turn/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/parallel-tool-calls/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/permission-switching/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/permission-switching/{system-prompt.golden.md => system-prompt.expected.md} (100%) rename examples/acp-agent/tests/snapshots/permission-switching/{tool-schemas.golden.json => tool-schemas.expected.json} (100%) rename examples/acp-agent/tests/snapshots/reject-extra-dirs/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/repeat-tool-guard/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/skill-load/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/skill-load/{system-prompt.golden.md => system-prompt.expected.md} (100%) rename examples/acp-agent/tests/snapshots/skill-load/{tool-schemas.golden.json => tool-schemas.expected.json} (100%) rename examples/acp-agent/tests/snapshots/subagent-fork/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/subagent-mixed/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/subagent-multi/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/subagent-spawn/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/text-turn/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/text-turn/{system-prompt.golden.md => system-prompt.expected.md} (100%) rename examples/acp-agent/tests/snapshots/text-turn/{tool-schemas.golden.json => tool-schemas.expected.json} (100%) rename examples/acp-agent/tests/snapshots/todo-plan/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/tool-call-turn/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/workflow-run/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/workspace-context/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/workspace-context/{system-prompt.golden.md => system-prompt.expected.md} (100%) rename examples/acp-agent/tests/snapshots/workspace-context/{tool-schemas.golden.json => tool-schemas.expected.json} (100%) rename examples/acp-agent/tests/snapshots/workspace-edit/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename examples/acp-agent/tests/snapshots/workspace-edit/{system-prompt.golden.md => system-prompt.expected.md} (100%) rename examples/acp-agent/tests/snapshots/workspace-edit/{tool-schemas.golden.json => tool-schemas.expected.json} (100%) rename examples/headless-agent/tests/snapshots/advanced-toolchain/{stream-json.golden.jsonl => stream-json.expected.jsonl} (100%) rename examples/tui-agent/tests/snapshots/bash-terminal-card/{terminal.golden.txt => terminal.expected.txt} (100%) rename examples/tui-agent/tests/snapshots/code-mode/{terminal.golden.txt => terminal.expected.txt} (100%) rename examples/tui-agent/tests/snapshots/cordis-dynamic-toolchain/{terminal.golden.txt => terminal.expected.txt} (100%) rename examples/tui-agent/tests/snapshots/dynamic-workflow/{terminal.golden.txt => terminal.expected.txt} (100%) rename examples/tui-agent/tests/snapshots/multi-turn-conversation/{terminal.golden.txt => terminal.expected.txt} (100%) rename examples/tui-agent/tests/snapshots/parallel-file-reads/{terminal.golden.txt => terminal.expected.txt} (100%) rename examples/tui-agent/tests/snapshots/todo-plan/{terminal.golden.txt => terminal.expected.txt} (100%) rename packages/support/acp-snapshot/tests/fixtures/record-suite/rec-child/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/{system-prompt.golden.md => system-prompt.expected.md} (100%) rename packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/{tool-schemas.golden.json => tool-schemas.expected.json} (100%) rename packages/support/acp-snapshot/tests/fixtures/record-suite/rec-skip/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename packages/support/acp-snapshot/tests/fixtures/suite/authored-error/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename packages/support/acp-snapshot/tests/fixtures/suite/blocked-log/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename packages/support/acp-snapshot/tests/fixtures/suite/no-model/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/{system-prompt.golden.md => system-prompt.expected.md} (100%) rename packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/{tool-schemas.golden.json => tool-schemas.expected.json} (100%) rename packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/{stdout.golden.jsonl => stdout.expected.jsonl} (100%) rename packages/ui/tui/tests/snapshots/{advanced-cards-collapsed.golden.txt => advanced-cards-collapsed.expected.txt} (100%) rename packages/ui/tui/tests/snapshots/{advanced-cards-expanded.golden.txt => advanced-cards-expanded.expected.txt} (100%) rename packages/ui/tui/tests/snapshots/{code-mode-pending.golden.txt => code-mode-pending.expected.txt} (100%) rename packages/ui/tui/tests/snapshots/{conversation-streaming.golden.txt => conversation-streaming.expected.txt} (100%) rename packages/ui/tui/tests/snapshots/{cordis-tools-pending.golden.txt => cordis-tools-pending.expected.txt} (100%) rename packages/ui/tui/tests/snapshots/{disposed-terminal.golden.txt => disposed-terminal.expected.txt} (100%) rename packages/ui/tui/tests/snapshots/{dynamic-workflow-pending.golden.txt => dynamic-workflow-pending.expected.txt} (100%) rename packages/ui/tui/tests/snapshots/{errors-and-help.golden.txt => errors-and-help.expected.txt} (100%) rename packages/ui/tui/tests/snapshots/{question-dialog-validation.golden.txt => question-dialog-validation.expected.txt} (100%) rename packages/ui/tui/tests/snapshots/{question-dialog.golden.txt => question-dialog.expected.txt} (100%) rename packages/ui/tui/tests/snapshots/{surface-after-compaction-narrow.golden.txt => surface-after-compaction-narrow.expected.txt} (100%) rename packages/ui/tui/tests/snapshots/{surface-after-compaction-wide.golden.txt => surface-after-compaction-wide.expected.txt} (100%) rename packages/ui/tui/tests/snapshots/{surface-before-compaction.golden.txt => surface-before-compaction.expected.txt} (100%) rename packages/ui/tui/tests/snapshots/{untrusted-controls.golden.txt => untrusted-controls.expected.txt} (100%) diff --git a/.agents/skills/dsh-code-review/SKILL.md b/.agents/skills/dsh-code-review/SKILL.md index 2b562cc321..6c0b2f8e8a 100644 --- a/.agents/skills/dsh-code-review/SKILL.md +++ b/.agents/skills/dsh-code-review/SKILL.md @@ -39,7 +39,7 @@ description: Use when reviewing a pull request in the deepseek-harness repo — - **Test strength:** assertions fail on the intended regression and verify external state, logs, events, or disposal rather than restating the implementation or trusting an agent's report. Coverage is necessary but not evidence that the scenario is correct. - **Changed checks have a negative control:** a new automated check, or a changed acceptance path in one, has a deliberately invalid case that reaches the real top-level runner and fails for the intended rule; a green happy path does not prove the check is wired. - **Implemented RFCs match shipped reality:** when a PR implements a proposed RFC, move and rewrite it as present-tense shipped state in the same diff, then verify paths, names, and mechanisms against the implementation. -- **Transcript changes:** editor-visible or model-visible changes update snapshots or explain why no snapshot applies. Review golden diffs as behavior changes, not formatting noise. +- **Transcript changes:** editor-visible or model-visible changes update snapshots or explain why no snapshot applies. Review expected-output diffs as behavior changes, not formatting noise. - **Bilingual changes:** compare meaning and terminology on both sides; a green pairing hash does not prove translation quality. ## Reporting findings diff --git a/.agents/skills/dsh-find-simplifications/SKILL.md b/.agents/skills/dsh-find-simplifications/SKILL.md index 59dde3df0d..623cdf18e2 100644 --- a/.agents/skills/dsh-find-simplifications/SKILL.md +++ b/.agents/skills/dsh-find-simplifications/SKILL.md @@ -24,7 +24,7 @@ A strong simplification removes, folds, or demotes something real and has clear - A seam has methods every implementation must support but no consumer uses. - A package boundary exists only for test/demo/support code and adds publish or dependency overhead. - A feature implements speculative product generality: multi-session/session-load, background task rosters, live registry invalidation, mid-turn steering, tool-owned UI rendering, and similar shapes with no product owner. -- An invariant, rollback path, goldens set, or special-case test exists only to protect an unused surface. +- An invariant, rollback path, set of expected outputs, or special-case test exists only to protect an unused surface. - The simplified behavior may differ slightly, but the new behavior is still reasonable and easier to explain. Thin candidates are usually not enough for an RFC: deleting one typo, running `knip` once, removing an intentionally documented backend/adapter, or flagging "this looks complex" without call-site proof. @@ -37,7 +37,7 @@ Use parallel subagents when the user asks for breadth or many candidates. Give e - ACP and UI surfaces: `session/*` methods, terminal `_meta`, transcript rendering, single vs multi-session state. - LLM/tools/system prompt: stream/generate surfaces, assemblers, registries, tool schema defaults, presentation hooks. - Bash and tool execution: foreground/background split, task ownership, output spill files, executor methods. -- Packages/examples/scripts/tests: package boundaries, static inventories, redundant snapshot goldens, support packages. +- Packages/examples/scripts/tests: package boundaries, static inventories, redundant snapshot expected outputs, support packages. If subagents are unavailable, simulate the same breadth yourself. Do not let the first good candidate stop the survey. @@ -54,7 +54,7 @@ For complex asynchronous code, draw the ownership graph and map each sentinel, r For every symbol or behavior, classify consumers before writing: - Production corpus: `packages/*/src`, `examples/*/src`, `examples/**/*.yml`, runtime scripts, and loader/config paths. -- Non-production corpus: tests, README/docs, RFCs, snapshots, generated goldens, and comments. +- Non-production corpus: tests, README/docs, RFCs, snapshots, generated expected outputs, and comments. - Ambiguous corpus: examples and scripts that may be product smoke paths. Inspect usage before classifying. Use `rg` first. Good searches include the exact symbol, event name, package name, config key, method name with both `.name(` and `name(`, and any wire strings. Then read the call sites. `knip` can help, but it is not a substitute for understanding public interfaces, dynamic event names, tests, docs, and Cordis loader paths. diff --git a/AGENTS.md b/AGENTS.md index efaa29bbac..5a0c7f0850 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -47,8 +47,8 @@ pnpm install # pnpm workspaces, node ^22.19 || >=24 pnpm run test # vitest unit tests pnpm run test:coverage # THE gating test run: per-file 100% coverage on packages/*/*/src pnpm run test:e2e # real-API tests; self-skip without DEEPSEEK_API_KEY -pnpm run test:snapshot # keyless ACP/headless/TUI replay vs goldens; filter: -t -pnpm run test:snapshot:record # re-record goldens (needs key) +pnpm run test:snapshot # keyless ACP/headless/TUI replay vs expected outputs; filter: -t +pnpm run test:snapshot:record # re-record expected outputs (needs key) pnpm run typecheck pnpm run lint pnpm run duplication # cross-file TypeScript clone detection diff --git a/docs/i18n/terminology.md b/docs/i18n/terminology.md index 120121e452..f11a1ff19e 100644 --- a/docs/i18n/terminology.md +++ b/docs/i18n/terminology.md @@ -106,6 +106,7 @@ | event stream | 事件流 | | | | | event-sourced | 事件溯源 | | | 沿用 DDD 社区通行译法 | | executor | 执行器 | | | | +| expected output | 预期输出 | | 金标 | 指 snapshot 比较产物;翻译语料的人工校准样例不在此列 | | extension | 扩展 | | | | | extension point | 扩展点 | | | 注意与 `seam` 区分 | | fail-fast | 快速失败 | | | | diff --git a/docs/postmortem/0002-js-expression-disabled-filesystem-tools.md b/docs/postmortem/0002-js-expression-disabled-filesystem-tools.md index 191c13459c..ccdb725bfb 100644 --- a/docs/postmortem/0002-js-expression-disabled-filesystem-tools.md +++ b/docs/postmortem/0002-js-expression-disabled-filesystem-tools.md @@ -4,7 +4,7 @@ Status: resolved ## Executive summary -The ACP example attempted to enable filesystem plugins conditionally with `disabled: !!js ...`, but Cordis evaluates JavaScript expressions only inside plugin `config`. The raw expression object was truthy, so the filesystem stack was always disabled. Snapshot refresh then accepted `UNKNOWN_TOOL` results as new goldens. The fix uses an explicit filesystem overlay and adds static-config and snapshot-result guards. +The ACP example attempted to enable filesystem plugins conditionally with `disabled: !!js ...`, but Cordis evaluates JavaScript expressions only inside plugin `config`. The raw expression object was truthy, so the filesystem stack was always disabled. Snapshot refresh then accepted `UNKNOWN_TOOL` results as new expected outputs. The fix uses an explicit filesystem overlay and adds static-config and snapshot-result guards. ## Summary @@ -22,7 +22,7 @@ The live confined default did not gain unintended filesystem access. A naive int - PR #261 consolidated ACP compositions and refreshed the filesystem snapshots while introducing conditional filesystem entries. - All unit, coverage, snapshot, documentation, build, and hygiene checks passed. -- Review of the refreshed filesystem goldens found generic failed cards and structured `UNKNOWN_TOOL` results. +- Review of the refreshed filesystem expected outputs found generic failed cards and structured `UNKNOWN_TOOL` results. - A real Loader boot confirmed that every `disabled` value remained an expression object and every filesystem fiber was absent. ## Root cause @@ -36,10 +36,10 @@ The snapshot framework treated any deterministic transcript as valid behavior. H - Filesystem scenarios boot `fs.cordis.yml`, an explicit fixed full-access overlay with a paired replay config and its own request-header class. - [`AGENTS.md`](../../AGENTS.md) and the [Cordis primer](../cordis-primer.md#loader-configuration) state that `!!js` is valid only under plugin `config` and conditional composition uses overlays. - `verify-cordis-config` parses repository Cordis YAML and rejects expression nodes in Loader entry metadata, including include patches and inserted entries. -- `dsh-acp-snapshot` rejects structured `UNKNOWN_TOOL` results in fresh runs and committed session fixtures before they can become accepted goldens. +- `dsh-acp-snapshot` rejects structured `UNKNOWN_TOOL` results in fresh runs and committed session fixtures before they can be committed as expected outputs. ## Lessons - A syntactically accepted configuration value is not necessarily evaluated at that location; document and verify interpolation boundaries. -- A snapshot refresh is fixture production, not correctness review. Semantic impossibilities such as a missing registered tool need assertions independent of the golden. +- A snapshot refresh is fixture production, not correctness review. Semantic impossibilities such as a missing registered tool need assertions independent of the expected output. - Permission controls must describe only the capabilities they actually govern. Composition-time filesystem access cannot follow a runtime bash-only preset safely. diff --git a/docs/rfc/INDEX.md b/docs/rfc/INDEX.md index 4476c47a32..d3afd838c8 100644 --- a/docs/rfc/INDEX.md +++ b/docs/rfc/INDEX.md @@ -207,11 +207,11 @@ Generated by `pnpm run gen-rfc-index` from the RFC tree — never edit by hand; | [Property-based testing for protocol-shaped code](implemented/testing/2026-06-11-property-based-testing.md) | 2026-06-11 | | [ACP snapshot tests — record-once / replay-deterministic](implemented/testing/2026-06-19-acp-snapshot-tests.md) | 2026-06-19 | | [Real-API e2e in CI against the external DeepSeek API](implemented/testing/2026-06-19-real-api-e2e-ci.md) | 2026-06-19 | -| [Use `session.jsonl` as the only snapshot session-log artifact](implemented/testing/2026-06-20-remove-redundant-snapshot-log-goldens.md) | 2026-06-20 | +| [Use `session.jsonl` as the only snapshot session-log artifact](implemented/testing/2026-06-20-remove-redundant-snapshot-log-expected-output.md) | 2026-06-20 | | [Persist the seed boundary so fork-child replay routes correctly](implemented/testing/2026-06-22-fork-child-replay-seed-boundary.md) | 2026-06-22 | | [Record fork and mixed spawn+fork snapshot scenarios](implemented/testing/2026-06-22-fork-snapshot-scenarios.md) | 2026-06-22 | | [Per-session snapshot replay for nested agents](implemented/testing/2026-06-22-subagent-snapshot-replay.md) | 2026-06-22 | -| [Hook snapshot matrix — end-to-end goldens for both bridges](implemented/testing/2026-07-04-hook-snapshot-matrix.md) | 2026-07-04 | +| [Hook snapshot matrix — end-to-end expected outputs for both bridges](implemented/testing/2026-07-04-hook-snapshot-matrix.md) | 2026-07-04 | | [Single-source the acp-agent replay config](implemented/testing/2026-07-04-single-source-acp-replay-config.md) | 2026-07-04 | | [Pin request-header content in one snapshot scenario](implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md) | 2026-07-06 | | [Extract the ACP snapshot suite into a support package](implemented/testing/2026-07-08-shared-acp-snapshot-package.md) | 2026-07-08 | diff --git a/docs/rfc/implemented/architecture/2026-07-02-tool-render-intent-union.md b/docs/rfc/implemented/architecture/2026-07-02-tool-render-intent-union.md index 37507c0c55..899d30416a 100644 --- a/docs/rfc/implemented/architecture/2026-07-02-tool-render-intent-union.md +++ b/docs/rfc/implemented/architecture/2026-07-02-tool-render-intent-union.md @@ -10,7 +10,7 @@ A tool declares how its calls render in a UI (an editor's tool-call card) throug - Which combinations are *valid* is unwritten: a `terminal` call that also sets `content` means "description above the card"; a generic call that sets `terminal` is meaningless but representable. The type permits nonsense. - There is no way to express the one file-tool affordance an editor most wants — a **diff card** (`{path, oldText, newText}`, which Zed renders as an inline diff / new-file preview). `ToolCallPresentation.content` is the *LLM* `ContentBlock[]` vocabulary (text/image), so a tool literally cannot ask for a diff. -The existing `FIXME(tool-presentation)` in `packages/core/tools/src/index.ts` named the fix: "redesign the type so a tool declares its render INTENT once (e.g. a tagged union over card kinds) rather than a bag of optional fields the bridge stitches together." The rejected RFC [Collapse tool-owned UI presentation](../../rejected/simplification/2026-06-20-generic-tool-rendering.md) deferred it explicitly: rich rendering "should return later as a tagged render-intent union after there are at least two real tools and two real consumers to validate the vocabulary." That bar is now met — two producer families (`dsh-tool-bash`, `dsh-tool-fs`) and two consumers (the ACP bridge live path + the snapshot-golden replay path). +The existing `FIXME(tool-presentation)` in `packages/core/tools/src/index.ts` named the fix: "redesign the type so a tool declares its render INTENT once (e.g. a tagged union over card kinds) rather than a bag of optional fields the bridge stitches together." The rejected RFC [Collapse tool-owned UI presentation](../../rejected/simplification/2026-06-20-generic-tool-rendering.md) deferred it explicitly: rich rendering "should return later as a tagged render-intent union after there are at least two real tools and two real consumers to validate the vocabulary." That bar is now met — two producer families (`dsh-tool-bash`, `dsh-tool-fs`) and two consumers (the ACP bridge live path + the snapshot replay path). ## Decision diff --git a/docs/rfc/implemented/architecture/2026-07-05-reconstructable-requests.md b/docs/rfc/implemented/architecture/2026-07-05-reconstructable-requests.md index 9b53369086..cbbfef6611 100644 --- a/docs/rfc/implemented/architecture/2026-07-05-reconstructable-requests.md +++ b/docs/rfc/implemented/architecture/2026-07-05-reconstructable-requests.md @@ -50,5 +50,5 @@ Like MiniCode, the conversation advances append-only and resets only when model- - The `step/start`-listener behavior change (above) is the one observable semantics change for plugins; `agent/pre-step` is the current-request seam. - Tool-result trimming (planned) needs no new mechanism: a logged single-entry surface replace (`start === end`) carrying a trimmed `tool/result` under the same `callId` — compaction-family, replay-correct, cache-bust batched by the same pressure logic. - Session logs grow one `request/header` snapshot per loop instance plus snapshots on real changes. This is larger than a delta codec but small beside chunk-heavy logs and retains one replay representation. `SESSION_FORMAT_VERSION` stays `0`; legacy delta events are rejected rather than migrated. -- Snapshot goldens changed once (every transcript gains its header events); the fs-writing fixtures are stored in the normalized authored form with cwd-relative tool arguments, because replay only round-trips cwd-independent argument paths. +- Snapshot expected outputs changed once (every transcript gains its header events); the fs-writing fixtures are stored in the normalized authored form with cwd-relative tool arguments, because replay only round-trips cwd-independent argument paths. - FIXME(call-config-shape): revisit `LlmCallConfig`'s exact field set — which fields are genuinely epoch-level for cache purposes (`model` certainly; the sampling scalars sit there out of caution), and where provider-specific extras (reasoning options, extra body params) belong when an adapter needs them. diff --git a/docs/rfc/implemented/architecture/2026-07-10-single-file-executable-sdk-runtime-distribution.i18n.yaml b/docs/rfc/implemented/architecture/2026-07-10-single-file-executable-sdk-runtime-distribution.i18n.yaml index 0655e89ba5..98131dce38 100644 --- a/docs/rfc/implemented/architecture/2026-07-10-single-file-executable-sdk-runtime-distribution.i18n.yaml +++ b/docs/rfc/implemented/architecture/2026-07-10-single-file-executable-sdk-runtime-distribution.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write -2026-07-10-single-file-executable-sdk-runtime-distribution.md: b177af24e988c6a314db522b8de0d1c09e30464f -2026-07-10-single-file-executable-sdk-runtime-distribution.zh.md: 0b964e8a748e4adcc32c017957e5294a3f258365 +2026-07-10-single-file-executable-sdk-runtime-distribution.md: 80d5ab64682f1b72ca1dfa1f96bc34f2a3db5f2d +2026-07-10-single-file-executable-sdk-runtime-distribution.zh.md: aea1fb66136b31e0a75ba51471fec4dadd960e8f diff --git a/docs/rfc/implemented/architecture/2026-07-10-single-file-executable-sdk-runtime-distribution.md b/docs/rfc/implemented/architecture/2026-07-10-single-file-executable-sdk-runtime-distribution.md index b177af24e9..80d5ab6468 100644 --- a/docs/rfc/implemented/architecture/2026-07-10-single-file-executable-sdk-runtime-distribution.md +++ b/docs/rfc/implemented/architecture/2026-07-10-single-file-executable-sdk-runtime-distribution.md @@ -21,7 +21,7 @@ The exe is packaged with the **`--sea` (enhanced SEA) mode** of [@yao-pkg/pkg](h `--sea` requires target ≥ node22; the exe uniformly targets node24. One pkg invocation packages exactly one target; multi-platform builds invoke it once per platform. -Terminology reminder: pkg's `/snapshot` VFS has nothing to do with this repo's testing-system "snapshot" (ACP replay goldens, `$DSH_SNAPSHOT`); this document says "VFS" for the former. +Terminology reminder: pkg's `/snapshot` VFS has nothing to do with this repo's testing-system "snapshot" (ACP replay expected outputs, `$DSH_SNAPSHOT`); this document says "VFS" for the former. ### The serving surface is a plugin: the two packages ui/jsonrpc + examples/jsonrpc-demo diff --git a/docs/rfc/implemented/architecture/2026-07-10-single-file-executable-sdk-runtime-distribution.zh.md b/docs/rfc/implemented/architecture/2026-07-10-single-file-executable-sdk-runtime-distribution.zh.md index 0b964e8a74..aea1fb6613 100644 --- a/docs/rfc/implemented/architecture/2026-07-10-single-file-executable-sdk-runtime-distribution.zh.md +++ b/docs/rfc/implemented/architecture/2026-07-10-single-file-executable-sdk-runtime-distribution.zh.md @@ -21,7 +21,7 @@ exe 使用 [@yao-pkg/pkg](https://github.com/yao-pkg/pkg)(vercel/pkg 归档后 `--sea` 要求构建目标 ≥ node22,exe 统一以 node24 为构建目标;每次 pkg 调用只打包一个构建目标,多平台各调用一次。 -术语提醒:pkg 的 `/snapshot` VFS 与本仓库测试体系的“快照”(ACP 回放 golden、`$DSH_SNAPSHOT`)无关,本文用“VFS”指前者。 +术语提醒:pkg 的 `/snapshot` VFS 与本仓库测试体系的“快照”(ACP 回放预期输出、`$DSH_SNAPSHOT`)无关,本文用“VFS”指前者。 ### 对外服务接口也是插件:ui/jsonrpc + examples/jsonrpc-demo 两包 diff --git a/docs/rfc/implemented/feature/2026-06-29-todo-write-tool.md b/docs/rfc/implemented/feature/2026-06-29-todo-write-tool.md index 06bc6daf6a..127da5d09f 100644 --- a/docs/rfc/implemented/feature/2026-06-29-todo-write-tool.md +++ b/docs/rfc/implemented/feature/2026-06-29-todo-write-tool.md @@ -49,7 +49,7 @@ Four tiers, designed up front: - **Real-Loader path** — the plugin run through `Loader.unwrapExports`, asserting the namespace export shape survives (it HAS `inject`, so a stray default would crash at load — postmortem/0001). - **Full-loop integration** — a scripted mock model calls `todo_write` through the real agent loop; the `todo/write` event lands and a second call replaces it. - **`session/load` replay** — a persisted `todo/write` re-emits the `plan` update when a fresh ACP bridge loads the session. -- **With-key e2e + snapshot** — a real prompt induces a `todo_write`; the snapshot golden gains the `plan` notification and the log event. +- **With-key e2e + snapshot** — a real prompt induces a `todo_write`; the snapshot expected output gains the `plan` notification and the log event. ## Alternatives considered diff --git a/docs/rfc/implemented/feature/2026-07-07-mcp-client-plugin.md b/docs/rfc/implemented/feature/2026-07-07-mcp-client-plugin.md index 1be5b225fe..27c108f0ce 100644 --- a/docs/rfc/implemented/feature/2026-07-07-mcp-client-plugin.md +++ b/docs/rfc/implemented/feature/2026-07-07-mcp-client-plugin.md @@ -199,7 +199,7 @@ Coverage is named per tier; each behavior lives at the cheapest tier that can ex - **Unit** (`tests/mcp-client.spec.ts`, `tests/apply.spec.ts`, mocked MCP SDK): the `publicToolName` algorithm (clean, normalize, truncate-and-hash, determinism, distinct-identity separation), raw-vs-public wire discipline, cross-server and native-tool coexistence, duplicate-`serverName` load failure and reservation release, invalid-tool-list rejection, generation swap/rollback, failed-re-sync retention, result mapping, cancellation, config schema validation. 100% per-file coverage gates the package. - **E2E** (`tests/mcp-client.e2e.ts`, keyless): the real MCP protocol against the in-repo fixture server, `@modelcontextprotocol/server-everything`, and `@modelcontextprotocol/server-filesystem` over stdio, and against an in-process `StreamableHTTPServerTransport` server over Streamable HTTP — discovery under the namespace, dotted-name normalization end to end, execution round-trips, duplicate-`serverName` rejection, disposal. -- **Snapshot**: deliberately none. MCP tools introduce no new transcript surface — they register as raw `ToolDefinition`s and render through the ACP bridge's generic-card fallback, which the bridge's unit suite already pins (`packages/ui/acp/tests/stream-update.spec.ts`). Adding an MCP server to the snapshot example's `cordis.yml` would mutate the pinned `text-turn` system-prompt fixture (forcing a with-key re-record of every recorded golden) and make every replay depend on spawning an external MCP server process — for zero new rendering behavior. If a later change gives MCP tools their own render intent, that change names its snapshot coverage then. +- **Snapshot**: deliberately none. MCP tools introduce no new transcript surface — they register as raw `ToolDefinition`s and render through the ACP bridge's generic-card fallback, which the bridge's unit suite already pins (`packages/ui/acp/tests/stream-update.spec.ts`). Adding an MCP server to the snapshot example's `cordis.yml` would mutate the pinned `text-turn` system-prompt fixture (forcing a with-key re-record of every recorded expected output) and make every replay depend on spawning an external MCP server process — for zero new rendering behavior. If a later change gives MCP tools their own render intent, that change names its snapshot coverage then. ## Consequences diff --git a/docs/rfc/implemented/feature/2026-07-09-bash-backed-grep-glob-discovery.md b/docs/rfc/implemented/feature/2026-07-09-bash-backed-grep-glob-discovery.md index 35f65c38f2..cd4f3b078b 100644 --- a/docs/rfc/implemented/feature/2026-07-09-bash-backed-grep-glob-discovery.md +++ b/docs/rfc/implemented/feature/2026-07-09-bash-backed-grep-glob-discovery.md @@ -144,7 +144,7 @@ If the complete logical result fits under the inline cap, no formatted spill art - The first-party tool-owned spill precedent is covered directly: spill backend present, spill backend absent, `saveText()` failure, and missing spill owner. - The package has real Loader-path coverage for the namespace plugin export shape (`name`, `inject`, `Config`, and `apply`, with no default export). - A real-executor integration suite (`dsh-bash-local` + a real `rg`) verifies the world: hostile patterns stay inert, per-session cwd resolution, VCS-metadata exclusion, modification-time ordering, and real ripgrep stderr classification. It self-skips where `rg` is not on PATH (a CI accommodation mirroring the keyless e2e skip); the fake-executor suite alone carries the per-file 100% coverage gate. -- Snapshot gap note for the transcript-visible spill notice: this landed with the gap note, not a snapshot. The snapshot tier replays the acp-agent tree, and adding the search plugin there changes the assembled system prompt — every golden would need re-recording with a real key, which the implementing environment did not hold. The spill notice's exact transcript text is pinned by unit tests (`formatGlobOutput`/`formatGrepOutput` and the through-the-registry spill tests); wiring the plugin into the acp-agent tree plus a `test:snapshot:record` pass is the follow-up for the next key-holding session. +- Snapshot gap note for the transcript-visible spill notice: this landed with the gap note, not a snapshot. The snapshot tier replays the acp-agent tree, and adding the search plugin there changes the assembled system prompt — every expected output would need re-recording with a real key, which the implementing environment did not hold. The spill notice's exact transcript text is pinned by unit tests (`formatGlobOutput`/`formatGrepOutput` and the through-the-registry spill tests); wiring the plugin into the acp-agent tree plus a `test:snapshot:record` pass is the follow-up for the next key-holding session. ## Consequences diff --git a/docs/rfc/implemented/simplification/2026-06-20-drop-unconsumed-llm-adapter-change-event.md b/docs/rfc/implemented/simplification/2026-06-20-drop-unconsumed-llm-adapter-change-event.md index ab921dd62c..9ce20ff7d4 100644 --- a/docs/rfc/implemented/simplification/2026-06-20-drop-unconsumed-llm-adapter-change-event.md +++ b/docs/rfc/implemented/simplification/2026-06-20-drop-unconsumed-llm-adapter-change-event.md @@ -24,7 +24,7 @@ If an LLM adapter browser or dynamic model-picker needs this signal later, reint ## Verification -`llm/adapter-change` and its emits are gone and the regenerated cordis catalog is fresh; HMR-safety holds (disposing a contributing fiber removes the adapter); `tools/change` and `system-prompt/change` remain documented and tested; and no production path changed observable behavior — the ACP snapshot goldens and the echo-agent smoke are byte-unchanged. +`llm/adapter-change` and its emits are gone and the regenerated cordis catalog is fresh; HMR-safety holds (disposing a contributing fiber removes the adapter); `tools/change` and `system-prompt/change` remain documented and tested; and no production path changed observable behavior — the ACP snapshot expected outputs and the echo-agent smoke are byte-unchanged. ## Consequences diff --git a/docs/rfc/implemented/simplification/2026-06-20-drop-unconsumed-llm-assembled-surfaces.md b/docs/rfc/implemented/simplification/2026-06-20-drop-unconsumed-llm-assembled-surfaces.md index 2b10bf36db..3a5dfcda26 100644 --- a/docs/rfc/implemented/simplification/2026-06-20-drop-unconsumed-llm-assembled-surfaces.md +++ b/docs/rfc/implemented/simplification/2026-06-20-drop-unconsumed-llm-assembled-surfaces.md @@ -26,7 +26,7 @@ This is the [drop-mutable-session-summary](../../implemented/simplification/2026 ## Verification -`streamBlocks`, `generate`, `llm/generate`, and the assembler helpers they alone required are gone with no new dead exports; both real adapters are exercised through `stream()` and the shared assembler; the loop behaves identically (ACP snapshot goldens unchanged); and the README, architecture doc, and module docs carry no mention of the removed surfaces. +`streamBlocks`, `generate`, `llm/generate`, and the assembler helpers they alone required are gone with no new dead exports; both real adapters are exercised through `stream()` and the shared assembler; the loop behaves identically (ACP snapshot expected outputs unchanged); and the README, architecture doc, and module docs carry no mention of the removed surfaces. ## Consequences diff --git a/docs/rfc/implemented/simplification/2026-07-04-prune-write-only-fs-surface.md b/docs/rfc/implemented/simplification/2026-07-04-prune-write-only-fs-surface.md index 0fc11a1c17..0de8f5732c 100644 --- a/docs/rfc/implemented/simplification/2026-07-04-prune-write-only-fs-surface.md +++ b/docs/rfc/implemented/simplification/2026-07-04-prune-write-only-fs-surface.md @@ -23,7 +23,7 @@ A future permission/containment layer might want the pre-resolution path for err ## Verification -The removed surfaces are gone — `STREAM_MIN_SIZE`/`streamMinSize` in `dsh-fs-local`, `FsTarget.inputPath`, `FsEditOutcome.replacements`/`.replaceAll`, and `FileReadOutcome.limit`/`.version` — while the request-side `replaceAll` (`FsEditRequest`) and the version fields on the other outcome types are untouched; the test fakes shrank with the types. `formatEditOutput`'s emitted text is unchanged for both `replace_all` branches, so no snapshot golden churned. +The removed surfaces are gone — `STREAM_MIN_SIZE`/`streamMinSize` in `dsh-fs-local`, `FsTarget.inputPath`, `FsEditOutcome.replacements`/`.replaceAll`, and `FileReadOutcome.limit`/`.version` — while the request-side `replaceAll` (`FsEditRequest`) and the version fields on the other outcome types are untouched; the test fakes shrank with the types. `formatEditOutput`'s emitted text is unchanged for both `replace_all` branches, so no snapshot expected output churned. ## Consequences diff --git a/docs/rfc/implemented/simplification/2026-07-04-remove-agent-steering-mirror.md b/docs/rfc/implemented/simplification/2026-07-04-remove-agent-steering-mirror.md index a0a383b255..6591aab8f3 100644 --- a/docs/rfc/implemented/simplification/2026-07-04-remove-agent-steering-mirror.md +++ b/docs/rfc/implemented/simplification/2026-07-04-remove-agent-steering-mirror.md @@ -8,7 +8,7 @@ Status: implemented `agent/steering` duplicated the immediately preceding durable `steering/message` with the same payload. `agent/queued` remains the live-only signal because it fires before persistence and covers work that may be cancelled before entering the log. -Steering carries real production traffic — the hook bridges' turn-continuation decisions inject their reasons through `inbox.steer()`, landing as durable `steering/message` events that the hook-matrix goldens pin — and every one of those consumers observes the durable event. Nothing observed the mirror. +Steering carries real production traffic — the hook bridges' turn-continuation decisions inject their reasons through `inbox.steer()`, landing as durable `steering/message` events that the hook-matrix expected outputs pin — and every one of those consumers observes the durable event. Nothing observed the mirror. ## Decision diff --git a/docs/rfc/implemented/simplification/2026-07-04-tighten-hook-protocol-contract.md b/docs/rfc/implemented/simplification/2026-07-04-tighten-hook-protocol-contract.md index 72172144d0..7cbf67043f 100644 --- a/docs/rfc/implemented/simplification/2026-07-04-tighten-hook-protocol-contract.md +++ b/docs/rfc/implemented/simplification/2026-07-04-tighten-hook-protocol-contract.md @@ -27,4 +27,4 @@ Unsupported vocabulary can return when a real consumer exists. `durationMs` rema ## Consequences -The `dialect`, `suppressOutput`, tunables, and semantics changes are invisible on the wire and in the goldens. The cost was churn in `dsh-hook-protocol` and both bridges — cheap under the pre-release stance, and cheaper than letting two copies of a durable event's semantics age apart. +The `dialect`, `suppressOutput`, tunables, and semantics changes are invisible on the wire and in the expected outputs. The cost was churn in `dsh-hook-protocol` and both bridges — cheap under the pre-release stance, and cheaper than letting two copies of a durable event's semantics age apart. diff --git a/docs/rfc/implemented/simplification/2026-07-04-trim-acp-bridge-unreachable-surface.md b/docs/rfc/implemented/simplification/2026-07-04-trim-acp-bridge-unreachable-surface.md index 191ddd833f..39edd8fee9 100644 --- a/docs/rfc/implemented/simplification/2026-07-04-trim-acp-bridge-unreachable-surface.md +++ b/docs/rfc/implemented/simplification/2026-07-04-trim-acp-bridge-unreachable-surface.md @@ -6,7 +6,7 @@ Status: implemented Two pieces of `dsh-acp` surface were unreachable from any shipped configuration: -1. **`AcpConfig.agentName` / `agentVersion`** (`packages/ui/acp/src/index.ts`). The shipped app package hands the bridge only `{ model }` (`packages/examples/acp-demo/src/index.ts`), so no leaf `cordis.yml` — the only production config surface — could set the knobs at all; they were settable solely by direct-mounting the bridge, which only a unit test did. Every snapshot golden — the hook-matrix scenarios included — pins the schema defaults (`deepseek-harness-acp` / `0.0.1`). The pair also carried a live `TODO(double-default)`: the literals existed twice (schema `.default(...)` plus `??` fallbacks), with the TODO asking to pick one home. +1. **`AcpConfig.agentName` / `agentVersion`** (`packages/ui/acp/src/index.ts`). The shipped app package hands the bridge only `{ model }` (`packages/examples/acp-demo/src/index.ts`), so no leaf `cordis.yml` — the only production config surface — could set the knobs at all; they were settable solely by direct-mounting the bridge, which only a unit test did. Every snapshot expected output — the hook-matrix scenarios included — pins the schema defaults (`deepseek-harness-acp` / `0.0.1`). The pair also carried a live `TODO(double-default)`: the literals existed twice (schema `.default(...)` plus `??` fallbacks), with the TODO asking to pick one home. 2. **The `toolKindFor` name heuristic** (same file) special-cased `bash*`/`read*`/`write`/`edit*` tool names in the generic-fallback path. Since the [render-intent union](../architecture/2026-07-02-tool-render-intent-union.md), every first-party tool those arms matched ships its own `presentCall` carrying its kind, and the presenter-less production tools (`subagent`, `subagent_fork`) fell through to `other` anyway. The arms were production-reachable only when a tool declined to present its own call — a `presentCall` that THROWS (the containment fallback), or model arguments that fail the tool's schema so `defineTool`'s `presentCall` wrapper returns `undefined` (e.g. a `bash` call missing the required `description`) — and the bridge's own module doc states the design rule the heuristic violated: "the bridge never special-cases tool names". ## Decision diff --git a/docs/rfc/implemented/testing/2026-06-19-acp-snapshot-tests.md b/docs/rfc/implemented/testing/2026-06-19-acp-snapshot-tests.md index 238f2044ad..2f7c0dbd8e 100644 --- a/docs/rfc/implemented/testing/2026-06-19-acp-snapshot-tests.md +++ b/docs/rfc/implemented/testing/2026-06-19-acp-snapshot-tests.md @@ -12,11 +12,11 @@ This RFC records the decision to add a third test tier — **snapshot tests** ## Decision -A snapshot test boots the real ACP example, drives its stdio protocol from a deterministic script, and compares normalized output with committed goldens. A session log recorded once from the real API supplies all later model streams. The fixture is the product's ordinary persisted JSONL. +A snapshot test boots the real ACP example, drives its stdio protocol from a deterministic script, and compares normalized output with committed expected outputs. A session log recorded once from the real API supplies all later model streams. The fixture is the product's ordinary persisted JSONL. ### The fixture is the persisted session JSONL -Each scenario's `session.jsonl` is harvested from a real run. `assistant/chunk` events reproduce the model streams; tool, message, and boundary events capture the harness behavior. One ordinary session artifact therefore serves as both replay source and behavioral golden. +Each scenario's `session.jsonl` is harvested from a real run. `assistant/chunk` events reproduce the model streams; tool, message, and boundary events capture the harness behavior. One ordinary session artifact therefore serves as both replay source and behavioral expected output. ### Replay derives the model script from the log @@ -48,12 +48,12 @@ Replay uses a `cordis.snapshot.yml` overlay that replaces the real adapter with A snapshot run asserts **two** normalized surfaces, because the harness's external surfaces are distinct: -1. The **stdout transcript** — the framed `session/update` JSON-RPC the editor sees. Catches regressions in the ACP bridge's event→update translation (`streamSessionEventUpdate`). Compared against a committed `stdout.golden.jsonl`. +1. The **stdout transcript** — the framed `session/update` JSON-RPC the editor sees. Catches regressions in the ACP bridge's event→update translation (`streamSessionEventUpdate`). Compared against a committed `stdout.expected.jsonl`. 2. The **re-persisted session JSONL**, normalized and compared with `session.jsonl`. The same fixture is both replay source and expected log. Prompt text is scrubbed; one scenario per header class pins readable prompt and tool content as described in the [header-pinning RFC](2026-07-06-pin-request-header-content-in-one-scenario.md). Override scenarios derive model behavior solely from their sidecar. The surfaces are complementary: stdout covers bridge projection, while JSONL covers loop, tool, and boundary structure that the projection omits. -Normalization replaces session, cwd, protocol-id, timestamp, path, and process volatility while preserving deterministic sequence numbers. Scenarios constrain real bash use to stable commands. The stdout golden remains wire-shaped JSONL and every raw line must parse as JSON. Vitest updates only the stdout golden; normalized session equality never overwrites the replay fixture. +Normalization replaces session, cwd, protocol-id, timestamp, path, and process volatility while preserving deterministic sequence numbers. Scenarios constrain real bash use to stable commands. The stdout expected output remains wire-shaped JSONL and every raw line must parse as JSON. Vitest updates only the stdout expected output; normalized session equality never overwrites the replay fixture. ### Isolation: normalization now, sandbox later @@ -65,11 +65,11 @@ Tool determinism comes from a temporary cwd, scrubbed environment, fresh non-log ### Two subcommands, replay in the default gate -`pnpm run test:snapshot` replays committed fixtures keylessly; `test:snapshot:record` uses the real API and rewrites the harvested session log and stdout golden. Missing fixtures fail loud. Every scenario carries `input.json`, `stdout.golden.jsonl`, and `session.jsonl`; no-model cases use a header-only log. `replay.override.json` is required only for scenarios marked `overridden`, because its presence replaces derived replay. Fixture guards reject missing, mismatched, and orphaned files. Both commands accept scenario filters. +`pnpm run test:snapshot` replays committed fixtures keylessly; `test:snapshot:record` uses the real API and rewrites the harvested session log and stdout expected output. Missing fixtures fail loud. Every scenario carries `input.json`, `stdout.expected.jsonl`, and `session.jsonl`; no-model cases use a header-only log. `replay.override.json` is required only for scenarios marked `overridden`, because its presence replaces derived replay. Fixture guards reject missing, mismatched, and orphaned files. Both commands accept scenario filters. ## Alternatives considered -- **A hand-authored `llm.json` of model chunks** — the earlier draft; reusing the real session log makes the fixture a genuine product of the system rather than a hand-built mock, and doubles it as a behavioral golden. +- **A hand-authored `llm.json` of model chunks** — the earlier draft; reusing the real session log makes the fixture a genuine product of the system rather than a hand-built mock, and doubles it as a behavioral expected output. - **A byte-level HTTP-record library (Polly/nock/MSW)** — rejected: adapter-specific, awkward with streaming SSE, and lower-level than the thing under test. - **Synthesizing throw/cancel entries from `turn/end {kind:'error'|'aborted'}`** — rejected: it couples `llm-replay` to loop-internal turn-closing semantics, and the `turn/end` reason is lossy (it cannot distinguish a thrown 401 from a finish-error); the explicit `replay.override.json` sidecar is the cleaner seam. diff --git a/docs/rfc/implemented/testing/2026-06-20-remove-redundant-snapshot-log-expected-output.md b/docs/rfc/implemented/testing/2026-06-20-remove-redundant-snapshot-log-expected-output.md new file mode 100644 index 0000000000..ef0d3bd4ab --- /dev/null +++ b/docs/rfc/implemented/testing/2026-06-20-remove-redundant-snapshot-log-expected-output.md @@ -0,0 +1,35 @@ +# RFC: Use `session.jsonl` as the only snapshot session-log artifact + +Status: implemented + +## Problem + +Model-driving ACP snapshot scenarios ship both `session.jsonl` and `session.expected.jsonl`. For normal recorded scenarios, `session.jsonl` is the replay fixture harvested from a real run, and the replay test normalizes the newly persisted log and compares it to `session.expected.jsonl`. In the current fixtures, the two normalized logs are identical for ordinary recorded scenarios. + +Authored override scenarios (`error-finish`, `cancel`) currently use `replay.override.json` to drive model behavior and keep `session.jsonl` as a minimal dummy fixture, while `session.expected.jsonl` holds the expected persisted log. The override file is a JSON array of `ReplayEntry` objects: `{ "kind": "chunks", "chunks": StreamChunk[] }`, `{ "kind": "throw", "chunks": StreamChunk[], "message": string, "code": string }`, or `{ "kind": "hang" }`. That split is also unnecessary: when an override sidecar exists, `llm-replay` replaces the derived script and does not need `session.jsonl` for model chunks, so `session.jsonl` can still be the expected session-log artifact for the scenario. + +## Decision + +The `session.expected.jsonl` concept is removed entirely. Every scenario has at most one committed session-log artifact, `session.jsonl`: + +- For recorded scenarios, `session.jsonl` remains the raw harvested log. Replay still derives model chunks from it, and the snapshot test compares the replay run's normalized persisted log against normalized `session.jsonl`. +- For authored override scenarios, `replay.override.json` drives model behavior and `session.jsonl` holds the expected produced session log. The replay adapter ignores the fixture for model chunks when the override exists, so the same file can be the expected log without affecting replay behavior. +- For no-model scenarios, `session.jsonl` can stay as the minimal fixture needed to boot `llm-replay`; no session-log comparison is needed unless the scenario creates a persisted session. + +Stdout expected outputs remain unchanged; they are the editor-facing projection and are not redundant with the session fixture. + +## Alternatives considered + +**Normalizing both sides against a shared (replay-run) context** — rejected: `normalizeSessionLog` scrubs cwd by exact string match, so the fixture's recorded cwd would survive unscrubbed and every compare would fail. Each side normalizes against its own header-derived context — the implementation note below carries the mechanics. + +## Verification + +`session.expected.jsonl` appears nowhere in the snapshot harness, fixtures, orphan guards, or docs; the snapshot test derives the expected session log from `session.jsonl` for every model scenario; authored sidecar scenarios commit their expected produced log as `session.jsonl` with `replay.override.json` as the model-behavior override; and the orphan-fixture guards know which files each scenario kind requires. The [ACP snapshot tests RFC](../../implemented/testing/2026-06-19-acp-snapshot-tests.md) describes the reduced fixture set. + +## Consequences + +Reviewers lose one artifact name that made the expected persisted log visually separate from the replay fixture. The stdout expected output still protects the editor transcript, and comparing replay output to `session.jsonl` preserves the loop/persistence regression check without duplicating files. + +## Implementation note + +Each side is normalized against its own header values because recording and replay have different ids, paths, and timestamps. `fixtureContext()` derives the fixture context from its header, making already-normalized fixtures idempotent. Session logs use plain equality rather than file-snapshot updates, so comparison never rewrites fixtures. diff --git a/docs/rfc/implemented/testing/2026-06-20-remove-redundant-snapshot-log-goldens.md b/docs/rfc/implemented/testing/2026-06-20-remove-redundant-snapshot-log-goldens.md deleted file mode 100644 index 5c9e1014b8..0000000000 --- a/docs/rfc/implemented/testing/2026-06-20-remove-redundant-snapshot-log-goldens.md +++ /dev/null @@ -1,35 +0,0 @@ -# RFC: Use `session.jsonl` as the only snapshot session-log artifact - -Status: implemented - -## Problem - -Model-driving ACP snapshot scenarios ship both `session.jsonl` and `session.golden.jsonl`. For normal recorded scenarios, `session.jsonl` is the replay fixture harvested from a real run, and the replay test normalizes the newly persisted log and compares it to `session.golden.jsonl`. In the current fixtures, the normalized recorded log and normalized golden are identical for the ordinary recorded scenarios. - -Authored override scenarios (`error-finish`, `cancel`) currently use `replay.override.json` to drive model behavior and keep `session.jsonl` as a minimal dummy fixture, while `session.golden.jsonl` holds the expected persisted log. The override file is a JSON array of `ReplayEntry` objects: `{ "kind": "chunks", "chunks": StreamChunk[] }`, `{ "kind": "throw", "chunks": StreamChunk[], "message": string, "code": string }`, or `{ "kind": "hang" }`. That split is also unnecessary: when an override sidecar exists, `llm-replay` replaces the derived script and does not need `session.jsonl` for model chunks, so `session.jsonl` can still be the expected session-log artifact for the scenario. - -## Decision - -The `session.golden.jsonl` concept is removed entirely. Every scenario has at most one committed session-log artifact, `session.jsonl`: - -- For recorded scenarios, `session.jsonl` remains the raw harvested log. Replay still derives model chunks from it, and the snapshot test compares the replay run's normalized persisted log against normalized `session.jsonl`. -- For authored override scenarios, `replay.override.json` drives model behavior and `session.jsonl` holds the expected produced session log. The replay adapter ignores the fixture for model chunks when the override exists, so the same file can be the expected log without affecting replay behavior. -- For no-model scenarios, `session.jsonl` can stay as the minimal fixture needed to boot `llm-replay`; no session-log comparison is needed unless the scenario creates a persisted session. - -Stdout goldens remain unchanged; they are the editor-facing projection and are not redundant with the session fixture. - -## Alternatives considered - -**Normalizing both sides against a shared (replay-run) context** — rejected: `normalizeSessionLog` scrubs cwd by exact string match, so the fixture's recorded cwd would survive unscrubbed and every compare would fail. Each side normalizes against its own header-derived context — the implementation note below carries the mechanics. - -## Verification - -`session.golden.jsonl` appears nowhere in the snapshot harness, fixtures, orphan guards, or docs; the snapshot test derives the expected session log from `session.jsonl` for every model scenario; authored sidecar scenarios commit their expected produced log as `session.jsonl` with `replay.override.json` as the model-behavior override; and the orphan-fixture guards know which files each scenario kind requires. The [ACP snapshot tests RFC](../../implemented/testing/2026-06-19-acp-snapshot-tests.md) describes the reduced fixture set. - -## Consequences - -Reviewers lose one artifact name that made the expected persisted log visually separate from the replay fixture. The stdout golden still protects the editor transcript, and comparing replay output to `session.jsonl` preserves the loop/persistence regression check without duplicating files. - -## Implementation note - -Each side is normalized against its own header values because recording and replay have different ids, paths, and timestamps. `fixtureContext()` derives the fixture context from its header, making already-normalized fixtures idempotent. Session logs use plain equality rather than file-snapshot updates, so comparison never rewrites fixtures. diff --git a/docs/rfc/implemented/testing/2026-06-22-subagent-snapshot-replay.md b/docs/rfc/implemented/testing/2026-06-22-subagent-snapshot-replay.md index 8135f6afd1..811930af00 100644 --- a/docs/rfc/implemented/testing/2026-06-22-subagent-snapshot-replay.md +++ b/docs/rfc/implemented/testing/2026-06-22-subagent-snapshot-replay.md @@ -4,7 +4,7 @@ Status: implemented ## Problem -The snapshot tier (`pnpm run test:snapshot`) boots the real `acp-agent` subprocess, replays a recorded session through [`dsh-llm-replay`](../../../../packages/support/llm-replay), and diffs the normalized stdout transcript + re-persisted session log against committed goldens. It is the only tier that exercises the full editor-facing transcript end to end. +The snapshot tier (`pnpm run test:snapshot`) boots the real `acp-agent` subprocess, replays a recorded session through [`dsh-llm-replay`](../../../../packages/support/llm-replay), and diffs the normalized stdout transcript + re-persisted session log against committed expected outputs. It is the only tier that exercises the full editor-facing transcript end to end. It was built for ONE session per process, and that assumption is wired into two places: diff --git a/docs/rfc/implemented/testing/2026-07-04-hook-snapshot-matrix.md b/docs/rfc/implemented/testing/2026-07-04-hook-snapshot-matrix.md index fd87377fb9..0ba04c4abb 100644 --- a/docs/rfc/implemented/testing/2026-07-04-hook-snapshot-matrix.md +++ b/docs/rfc/implemented/testing/2026-07-04-hook-snapshot-matrix.md @@ -1,10 +1,10 @@ -# RFC: Hook snapshot matrix — end-to-end goldens for both bridges +# RFC: Hook snapshot matrix — end-to-end expected outputs for both bridges Status: implemented ## Problem -The hook bridges — [`dsh-hooks-claude`](../../../../packages/hooks/hooks-claude) (7 Claude Code hook points) and [`dsh-hooks-codex`](../../../../packages/hooks/hooks-codex) (5 Codex points) — map external hook commands onto the harness interception seams. They carry deep unit and coverage-spec coverage (every decision arm, every payload dialect, driven against a mocked seam) plus one key-gated e2e (`hooks.e2e.ts`, a live `PreToolUse` block). But the full-transcript snapshot tier — the one net that boots the real `acp-agent` subprocess, replays a recorded session keyless, and diffs the normalized ACP stdout + re-persisted log against committed goldens — covered exactly ONE hook: a Claude `UserPromptSubmit` block (`hook-cc-promptsubmit-block`). +The hook bridges — [`dsh-hooks-claude`](../../../../packages/hooks/hooks-claude) (7 Claude Code hook points) and [`dsh-hooks-codex`](../../../../packages/hooks/hooks-codex) (5 Codex points) — map external hook commands onto the harness interception seams. They carry deep unit and coverage-spec coverage (every decision arm, every payload dialect, driven against a mocked seam) plus one key-gated e2e (`hooks.e2e.ts`, a live `PreToolUse` block). But the full-transcript snapshot tier — the one net that boots the real `acp-agent` subprocess, replays a recorded session keyless, and diffs the normalized ACP stdout + re-persisted log against committed expected outputs — covered exactly ONE hook: a Claude `UserPromptSubmit` block (`hook-cc-promptsubmit-block`). That is the tier a mocked unit test structurally cannot be: it exercises the REAL bridge translating a REAL hook process's outcome into the REAL seam decision, then the REAL loop's reaction, rendered exactly as an editor sees it. A bridge-translation or loop-structure regression that left every unit green would still escape it for every hook point but one — and for the Codex bridge, the ACP example did not even LOAD it, so no Codex hook could fire end-to-end at all. @@ -29,22 +29,22 @@ Thirteen scenarios under `examples/acp-agent/tests/snapshots/`, naming `hook- diff --git a/docs/rfc/implemented/testing/2026-07-04-single-source-acp-replay-config.md b/docs/rfc/implemented/testing/2026-07-04-single-source-acp-replay-config.md index 70730f0382..03c25572d4 100644 --- a/docs/rfc/implemented/testing/2026-07-04-single-source-acp-replay-config.md +++ b/docs/rfc/implemented/testing/2026-07-04-single-source-acp-replay-config.md @@ -10,7 +10,7 @@ Status: implemented `cordis.snapshot.yml` includes the live config, disables the named DeepSeek adapter by id and name, and inserts the replay adapter. Every other entry therefore comes from the shipping tree. Replay selects the overlay; recording still boots `cordis.yml`, and the load guard permits the intentionally disabled entry. -One vendored-plugin fact the overlay depends on, deliberately: the include applies `patches` when it loads the file — its `refresh()`/`internal/update` paths re-read without re-patching — which is exactly enough for a one-shot replay boot (the replay app loads no `hmr` and nothing rewrites the config mid-run). The snapshot suite is the proof: all scenarios pass unchanged on the overlay, byte-identical goldens included. +One vendored-plugin fact the overlay depends on, deliberately: the include applies `patches` when it loads the file — its `refresh()`/`internal/update` paths re-read without re-patching — which is exactly enough for a one-shot replay boot (the replay app loads no `hmr` and nothing rewrites the config mid-run). The snapshot suite is the proof: all scenarios pass unchanged on the overlay, byte-identical expected outputs included. ## Alternatives considered diff --git a/docs/rfc/implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md b/docs/rfc/implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md index 69af84b0f0..68f7fce66a 100644 --- a/docs/rfc/implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md +++ b/docs/rfc/implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md @@ -8,7 +8,7 @@ An ACP snapshot suite needs to prove the exact composed system prompt and tool-s ## Decision -Exactly one scenario per header-composition class is flagged `pinsHeader`. Its directory splits the pin by review format: `system-prompt.golden.md` contains the normalized full prompt sequence as ordinary Markdown, `tool-schemas.golden.json` contains the corresponding complete schema sequence as structured JSON, and `session.jsonl` retains config, reason, and any model-visible prefix while storing `header.system` and `header.tools` as `"{{system}}"` / `"{{tools}}"`. Every other JSONL uses the same prompt and tool tokens and also tokenizes session-prefix content. The pin mechanics live in [`dsh-acp-snapshot`](../../../../packages/support/acp-snapshot/README.md), whose suite factory enforces one pin per class. +Exactly one scenario per header-composition class is flagged `pinsHeader`. Its directory splits the pin by review format: `system-prompt.expected.md` contains the normalized full prompt sequence as ordinary Markdown, `tool-schemas.expected.json` contains the corresponding complete schema sequence as structured JSON, and `session.jsonl` retains config, reason, and any model-visible prefix while storing `header.system` and `header.tools` as `"{{system}}"` / `"{{tools}}"`. Every other JSONL uses the same prompt and tool tokens and also tokenizes session-prefix content. The pin mechanics live in [`dsh-acp-snapshot`](../../../../packages/support/acp-snapshot/README.md), whose suite factory enforces one pin per class. The pure `scrubSystemPrompts` and `scrubToolSchemas` normalizers independently tokenize every stored full header. `scrubRequestHeaders` also tokenizes session-prefix content for non-pinning scenarios while retaining header count, field presence, config, reason, and prefix message count. Record and refresh write-back apply the appropriate scrub before writing JSONL and regenerate both sidecars from the normalized live full-header sequence, so neither path can reintroduce prompt/schema bulk into JSONL or leave a review artifact stale. diff --git a/docs/rfc/implemented/testing/2026-07-08-shared-acp-snapshot-package.md b/docs/rfc/implemented/testing/2026-07-08-shared-acp-snapshot-package.md index 85d89d5f14..ecdd40f313 100644 --- a/docs/rfc/implemented/testing/2026-07-08-shared-acp-snapshot-package.md +++ b/docs/rfc/implemented/testing/2026-07-08-shared-acp-snapshot-package.md @@ -4,7 +4,7 @@ Status: implemented ## Problem -The ACP snapshot tier ([snapshot RFC](2026-06-19-acp-snapshot-tests.md)) was built from three modules living inside one example's test directory: `snapshot-harness.ts` (boot the real bin subprocess, drive it over ACP JSON-RPC, harvest the persisted logs), `snapshot-normalize.ts` (the pure golden normalizers), and the ~150-line scenario body plus fixture guards in `acp.snapshot.ts` (record/replay modes, the stdout-golden and log compares, the pinned-header uniformity guard, the orphan/required-file/single-pin meta-tests). +The ACP snapshot tier ([snapshot RFC](2026-06-19-acp-snapshot-tests.md)) was built from three modules living inside one example's test directory: `snapshot-harness.ts` (boot the real bin subprocess, drive it over ACP JSON-RPC, harvest the persisted logs), `snapshot-normalize.ts` (the pure expected-output normalizers), and the ~150-line scenario body plus fixture guards in `acp.snapshot.ts` (record/replay modes, the stdout expected-output and log comparisons, the pinned-header uniformity guard, the orphan/required-file/single-pin meta-tests). A second ACP example wanting snapshot coverage — the sandbox/approval composition is the immediate consumer — could only copy those modules, forking exactly the logic that must not drift: record write-back, header scrubbing, child-session harvest ordering. The spawn/client glue was also triplicated across `acp.e2e.ts`, `hooks.e2e.ts`, and the harness. Location decided test rigor: the per-file 100% coverage gate measures `packages/*/*/src` only, so none of this machinery was measured — the same gap that had moved `dsh-llm-replay` out of `examples/` into [packages/support](../../../../packages/support/README.md). And the harness's ACP client hardcoded `requestPermission → cancelled`, so an approval round-trip — the headline behavior of the sandbox composition — could not be expressed at the snapshot tier at all. @@ -18,7 +18,7 @@ The machinery lives in [`packages/support/acp-snapshot`](../../../../packages/su **`src/normalize.ts`** — the pure normalizers, hook-free by policy: when a future event carries a new volatile field (an approval duration, say), the shared normalizer learns it in the same change, keeping one home for what "normalized" means rather than per-suite scrub extensions. -**`src/suite.ts`** — the `Scenario` type and `defineAcpSnapshotSuite(options)`, registering the per-scenario compares, record/refresh fixture write-back, the header pin with its live uniformity guard, and the fixture guard block (no orphan scenario dirs, required files present, exactly one pin per class, every JSONL a `scrubSystemPrompts` fixed point, non-pinning fixtures also `scrubRequestHeaders` fixed points). A scenario directory's `session.jsonl` plus contiguous `session..jsonl` siblings are its ordered primary/child inventory, so the scenario table declares policy without duplicating a child count. The pinned-header contract ([pinned-header RFC](2026-07-06-pin-request-header-content-in-one-scenario.md)) is per-suite: each header class flags exactly one `pinsHeader` scenario, whose `system-prompt.golden.md` and JSONL tool list split the composed header into reviewable artifacts; the uniformity guard compares both against every live header in that class. A pinning scenario declares any legitimate changed-header count, and its Markdown artifact records every full changed prompt. The pure helpers (`sessionFixtureNames`, `fixtureContext`, `normalizedHeaders`, `normalizedSystemPrompts`, `formatSystemPromptSnapshot`, `headerChangeCount`) are exported from the module for direct unit coverage. +**`src/suite.ts`** — the `Scenario` type and `defineAcpSnapshotSuite(options)`, registering the per-scenario compares, record/refresh fixture write-back, the header pin with its live uniformity guard, and the fixture guard block (no orphan scenario dirs, required files present, exactly one pin per class, every JSONL a `scrubSystemPrompts` fixed point, non-pinning fixtures also `scrubRequestHeaders` fixed points). A scenario directory's `session.jsonl` plus contiguous `session..jsonl` siblings are its ordered primary/child inventory, so the scenario table declares policy without duplicating a child count. The pinned-header contract ([pinned-header RFC](2026-07-06-pin-request-header-content-in-one-scenario.md)) is per-suite: each header class flags exactly one `pinsHeader` scenario, whose `system-prompt.expected.md` and JSONL tool list split the composed header into reviewable artifacts; the uniformity guard compares both against every live header in that class. A pinning scenario declares any legitimate changed-header count, and its Markdown artifact records every full changed prompt. The pure helpers (`sessionFixtureNames`, `fixtureContext`, `normalizedHeaders`, `normalizedSystemPrompts`, `formatSystemPromptSnapshot`, `headerChangeCount`) are exported from the module for direct unit coverage. ## Alternatives considered @@ -26,7 +26,7 @@ The machinery lives in [`packages/support/acp-snapshot`](../../../../packages/su - **A shared module directory under `examples/`** — keeps the code outside the coverage gate and forces relative imports across example boundaries, against the package-name import convention; `examples/` leaves stay thin by design. - **A `/testing` subpath export of `dsh-acp-demo`** — couples test infrastructure into a product package's surface and dependency set; `packages/support/` exists precisely for real-but-lower-compatibility dev/test packages, with `dsh-llm-replay` as the precedent this package completes. - **Export raw test-body functions instead of a suite factory** — each example would re-own the `describe`/`it` skeleton (~80 lines of registration boilerplate per suite) for no flexibility gain; the factory keeps consumers to a scenario table plus one call, and the exported pure helpers preserve unit-testability inside the factory design. -- **An injectable ACP `Client` factory instead of declarative `permissionAnswers`** — maximally flexible, but it leaks SDK client construction to every consumer and reopens per-example drift in exactly the layer being unified; a declarative queue keeps `input.json` the single scripting surface and stays golden-normalizable. +- **An injectable ACP `Client` factory instead of declarative `permissionAnswers`** — maximally flexible, but it leaks SDK client construction to every consumer and reopens per-example drift in exactly the layer being unified; a declarative queue keeps `input.json` the single scripting surface and compatible with expected-output normalization. - **Generalize beyond ACP (a transport-agnostic snapshot harness)** — no second transport exists; the harness is ACP-shaped end to end (SDK client, JSON-RPC frames, `session/update` waiters), and a speculative abstraction would be a seam split ahead of any consumer. ## Testing diff --git a/docs/rfc/implemented/testing/2026-07-18-tui-terminal-state-snapshots.i18n.yaml b/docs/rfc/implemented/testing/2026-07-18-tui-terminal-state-snapshots.i18n.yaml index f111c337dc..6f3df04716 100644 --- a/docs/rfc/implemented/testing/2026-07-18-tui-terminal-state-snapshots.i18n.yaml +++ b/docs/rfc/implemented/testing/2026-07-18-tui-terminal-state-snapshots.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write -2026-07-18-tui-terminal-state-snapshots.md: a1363a521372c11dab8b239cad87df5ff5ca8f22 -2026-07-18-tui-terminal-state-snapshots.zh.md: aa365ba1f2ba17e5bc5409dbdc3736cbc29fe26a +2026-07-18-tui-terminal-state-snapshots.md: a609e7c167ddf7ebe594f6793e34a48c555c10a9 +2026-07-18-tui-terminal-state-snapshots.zh.md: 8690a6f83a827619682b82c2df56d360e104c53b diff --git a/docs/rfc/implemented/testing/2026-07-18-tui-terminal-state-snapshots.md b/docs/rfc/implemented/testing/2026-07-18-tui-terminal-state-snapshots.md index a1363a5213..a609e7c167 100644 --- a/docs/rfc/implemented/testing/2026-07-18-tui-terminal-state-snapshots.md +++ b/docs/rfc/implemented/testing/2026-07-18-tui-terminal-state-snapshots.md @@ -25,19 +25,19 @@ The runnable TUI has its own `examples/tui-agent` leaf beside the readline `repl ### Recorded-session replay -Each example-level scenario directory owns `session.jsonl`, optional child logs `session..jsonl`, and `terminal.golden.txt`. The primary log supplies user-authored `user/message` prompts and the recorded `assistant/chunk` sequence. `dsh-llm-replay` derives one model-call script per session, binds child logs to fresh child sessions, and is the only mocked boundary. The agent loop, bash and filesystem implementations, Code Mode worker, subagent provider, workflow worker, Cordis tools, presenters, and TUI are production implementations. +Each example-level scenario directory owns `session.jsonl`, optional child logs `session..jsonl`, and `terminal.expected.txt`. The primary log supplies user-authored `user/message` prompts and the recorded `assistant/chunk` sequence. `dsh-llm-replay` derives one model-call script per session, binds child logs to fresh child sessions, and is the only mocked boundary. The agent loop, bash and filesystem implementations, Code Mode worker, subagent provider, workflow worker, Cordis tools, presenters, and TUI are production implementations. -The suite rejects a journey when its tool-call sequence differs, an expected event count is missing, a tool result is an error, a turn ends in error, a workflow lifecycle is incomplete, or the live child-session count differs from the fixture set. These assertions prevent an attractive terminal golden from hiding a failed or bypassed production path. +The suite rejects a journey when its tool-call sequence differs, an expected event count is missing, a tool result is an error, a turn ends in error, a workflow lifecycle is incomplete, or the live child-session count differs from the fixture set. These assertions prevent an attractive terminal expected output from hiding a failed or bypassed production path. -The live-model fixtures use `DSH_SNAPSHOT=record`; record mode rewrites their primary and child JSONL logs and terminal goldens. The deterministic Cordis toolchain keeps an authored complete JSONL script because reliably coercing a live model through five exact tool boundaries and two children is not a stable recording contract. `DSH_SNAPSHOT=refresh` replays every committed script keylessly and rewrites only derived terminal goldens. Plain replay compares without writing, and unknown mode values fail loud. +The live-model fixtures use `DSH_SNAPSHOT=record`; record mode rewrites their primary and child JSONL logs and terminal expected outputs. The deterministic Cordis toolchain keeps an authored complete JSONL script because reliably coercing a live model through five exact tool boundaries and two children is not a stable recording contract. `DSH_SNAPSHOT=refresh` replays every committed script keylessly and rewrites only derived terminal expected outputs. Plain replay compares without writing, and unknown mode values fail loud. ### Semantic terminal projection The package-local `HeadlessTerminal` implements the same pi-tui `Terminal` interface as the process terminal and feeds every ANSI write into the pinned `@xterm/headless` parser. Snapshot code waits for synchronized frames to quiesce before reading state, so a checkpoint represents a completed screen rather than a timer-dependent write prefix. -Each golden projects dimensions, active-buffer and viewport coordinates, lifecycle and cursor state, rows, wrap markers, and non-default style ranges into text. Scroll-heavy cards capture the used buffer; overlays capture the visible viewport. Text and style remain separate so a reviewer can distinguish content changes from presentation changes without decoding ANSI bytes. +Each expected output projects dimensions, active-buffer and viewport coordinates, lifecycle and cursor state, rows, wrap markers, and non-default style ranges into text. Scroll-heavy cards capture the used buffer; overlays capture the visible viewport. Text and style remain separate so a reviewer can distinguish content changes from presentation changes without decoding ANSI bytes. -Every checkpoint enforces theme independence across the complete terminal state: no RGB colors, no palette entries beyond ANSI 0–15, and no explicit background colors. Reverse video remains valid for selection because it uses terminal defaults. Both suites own closed inventories that reject missing scenarios, missing checkpoints, and orphaned golden files. +Every checkpoint enforces theme independence across the complete terminal state: no RGB colors, no palette entries beyond ANSI 0–15, and no explicit background colors. Reverse video remains valid for selection because it uses terminal defaults. Both suites own closed inventories that reject missing scenarios, missing checkpoints, and orphaned expected output files. ### Required scenario matrix @@ -58,7 +58,7 @@ Every checkpoint enforces theme independence across the complete terminal state: - **Snapshot raw terminal writes** — rejected because differential rendering may change write boundaries without changing the screen, while cursor and clear sequences are unreadable in review. - **Snapshot component render lines before terminal output** — rejected because it does not test ANSI parsing, cursor movement, overlays, viewport behavior, or independent components in one frame. - **Build every completed flow by appending session events** — rejected because a hand-authored event sequence can drift from the agent loop, tool execution, child-session binding, or worker behavior while its presentation test stays green. Direct event construction remains limited to transient renderer states. -- **Reuse ACP stdout goldens as the TUI oracle** — rejected because a recorded model journey is transport-neutral but its presentation is not. TUI scenarios own terminal goldens while using the same JSONL replay vocabulary. +- **Reuse ACP stdout expected outputs as the TUI oracle** — rejected because a recorded model journey is transport-neutral but its presentation is not. TUI scenarios own terminal expected outputs while using the same JSONL replay vocabulary. - **Commit raster screenshots** — rejected because fonts, glyph metrics, antialiasing, and host terminal themes make them platform-sensitive and make semantic style changes difficult to review. - **Use only PTY end-to-end tests** — rejected because raw PTY output is a stream of historical drawing operations, not queryable final state. PTY tests retain the real Loader/input/teardown boundary, while the emulator owns broad state coverage. @@ -67,4 +67,4 @@ Every checkpoint enforces theme independence across the complete terminal state: - Completed advanced snapshots now fail when the real Code Mode, workflow, subagent, filesystem, bash, or Cordis path breaks, rather than accepting a fabricated result event. - TUI visual regressions produce readable cell-and-style diffs, while JSONL fixtures retain the exact model chunks that made the production path execute. - The emulator uses xterm's proposed buffer API. An xterm upgrade requires rerunning and reviewing the semantic projection; terminal-specific behavior still needs the PTY smoke. -- Goldens deliberately encode wrapping and viewport behavior at fixed sizes. Intentional layout changes use keyless refresh, while model-journey changes use record mode and review both JSONL and terminal diffs. +- Expected outputs deliberately encode wrapping and viewport behavior at fixed sizes. Intentional layout changes use keyless refresh, while model-journey changes use record mode and review both JSONL and terminal diffs. diff --git a/docs/rfc/implemented/testing/2026-07-18-tui-terminal-state-snapshots.zh.md b/docs/rfc/implemented/testing/2026-07-18-tui-terminal-state-snapshots.zh.md index aa365ba1f2..8690a6f83a 100644 --- a/docs/rfc/implemented/testing/2026-07-18-tui-terminal-state-snapshots.zh.md +++ b/docs/rfc/implemented/testing/2026-07-18-tui-terminal-state-snapshots.zh.md @@ -25,19 +25,19 @@ TUI 覆盖分为四个互补层次: ### 已录制会话回放 -每个示例级场景目录都包含 `session.jsonl`、可选的子会话日志 `session..jsonl`,以及 `terminal.golden.txt`。主日志提供用户来源的 `user/message` 提示词和已录制的 `assistant/chunk` 序列。`dsh-llm-replay` 为每个会话派生一份模型调用脚本,并将子日志绑定到新建的子会话;这是测试中唯一的 mock 边界。agent loop、bash 与文件系统实现、Code Mode worker、subagent 提供方、工作流 worker、Cordis 工具、呈现器和 TUI 都使用生产实现。 +每个示例级场景目录都包含 `session.jsonl`、可选的子会话日志 `session..jsonl`,以及 `terminal.expected.txt`。主日志提供用户来源的 `user/message` 提示词和已录制的 `assistant/chunk` 序列。`dsh-llm-replay` 为每个会话派生一份模型调用脚本,并将子日志绑定到新建的子会话;这是测试中唯一的 mock 边界。agent loop、bash 与文件系统实现、Code Mode worker、subagent 提供方、工作流 worker、Cordis 工具、呈现器和 TUI 都使用生产实现。 -如果工具调用顺序不符、预期事件数量不足、工具结果报错、轮次以错误结束、工作流生命周期不完整,或者实时子会话数量与 fixture(测试前置数据)集合不一致,测试都会失败。即使终端金标表面正确,这些断言也能阻止失败或被绕过的生产路径混入结果。 +如果工具调用顺序不符、预期事件数量不足、工具结果报错、轮次以错误结束、工作流生命周期不完整,或者实时子会话数量与 fixture(测试前置数据)集合不一致,测试都会失败。即使终端预期输出表面正确,这些断言也能阻止失败或被绕过的生产路径混入结果。 -真实模型 fixture 通过 `DSH_SNAPSHOT=record` 更新;录制模式会重写其主会话与子会话 JSONL 日志以及终端金标。确定性的 Cordis 工具链保留一份人工编写的完整 JSONL 脚本,因为要求真实模型稳定经过五个指定工具边界和两个子会话并不是可靠的录制契约。`DSH_SNAPSHOT=refresh` 会无密钥回放所有已提交脚本,并且只重写派生的终端金标。普通回放只比较而不写入,未知模式值会快速失败。 +真实模型 fixture 通过 `DSH_SNAPSHOT=record` 更新;录制模式会重写其主会话与子会话 JSONL 日志以及终端预期输出。确定性的 Cordis 工具链保留一份人工编写的完整 JSONL 脚本,因为要求真实模型稳定经过五个指定工具边界和两个子会话并不是可靠的录制契约。`DSH_SNAPSHOT=refresh` 会无密钥回放所有已提交脚本,并且只重写派生的终端预期输出。普通回放只比较而不写入,未知模式值会快速失败。 ### 语义终端投影 包内的 `HeadlessTerminal` 实现与进程终端相同的 pi-tui `Terminal` 接口,并把每次 ANSI 写入交给固定版本的 `@xterm/headless` 解析器。读取状态前,快照代码会等待同步帧稳定,因此每个检查点表示已经完成的画面,而不是依赖计时的写入前缀。 -每份金标把终端尺寸、活动缓冲区和视口坐标、生命周期与光标状态、各行、换行标记以及非默认样式区间投影为文本。滚动内容较多的卡片捕获已使用缓冲区;浮层捕获可见视口。文本和样式相互分离,评审人无需解码 ANSI 字节即可区分内容变化与呈现变化。 +每份预期输出把终端尺寸、活动缓冲区和视口坐标、生命周期与光标状态、各行、换行标记以及非默认样式区间投影为文本。滚动内容较多的卡片捕获已使用缓冲区;浮层捕获可见视口。文本和样式相互分离,评审人无需解码 ANSI 字节即可区分内容变化与呈现变化。 -每个检查点还会对完整终端状态强制执行主题无关性:禁止 RGB 颜色、禁止 ANSI 0–15 以外的调色板项,也禁止显式背景色。选择行使用终端默认色进行反显,因此仍然有效。两套测试都拥有封闭清单,会拒绝缺失的场景、缺失的检查点和遗留金标文件。 +每个检查点还会对完整终端状态强制执行主题无关性:禁止 RGB 颜色、禁止 ANSI 0–15 以外的调色板项,也禁止显式背景色。选择行使用终端默认色进行反显,因此仍然有效。两套测试都拥有封闭清单,会拒绝缺失的场景、缺失的检查点和遗留预期输出文件。 ### 必需场景矩阵 @@ -58,7 +58,7 @@ TUI 覆盖分为四个互补层次: - **快照原始终端写入**:不予采纳,因为差分渲染可能在画面不变时改变写入边界,而且光标与清屏序列难以评审。 - **快照进入终端输出之前的组件渲染行**:不予采纳,因为它无法测试 ANSI 解析、光标移动、浮层、视口行为,也无法测试独立组件在同一帧中的相互作用。 - **通过追加会话事件构造所有完整流程**:不予采纳,因为人工编写的事件序列可能与 agent loop、工具执行、子会话绑定或 worker 行为发生偏差,但呈现测试仍然保持绿色。直接构造事件只用于渲染器瞬态。 -- **复用 ACP stdout 金标作为 TUI 判定依据**:不予采纳,因为已录制模型流程与传输方式无关,其呈现方式却并非如此。TUI 场景使用同一套 JSONL 回放词汇,但拥有独立的终端金标。 +- **复用 ACP stdout 预期输出作为 TUI 判定依据**:不予采纳,因为已录制模型流程与传输方式无关,其呈现方式却并非如此。TUI 场景使用同一套 JSONL 回放词汇,但拥有独立的终端预期输出。 - **提交栅格截图**:不予采纳,因为字体、字形度量、抗锯齿和宿主终端主题会使结果依赖平台,也会增加语义样式变更的评审难度。 - **只使用 PTY 端到端测试**:不予采纳,因为原始 PTY 输出是一系列历史绘制操作,而不是可查询的最终状态。PTY 测试保留真实 Loader、输入与清理边界,模拟器负责广泛的状态覆盖。 @@ -67,4 +67,4 @@ TUI 覆盖分为四个互补层次: - 当真实 Code Mode、工作流、subagent、文件系统、bash 或 Cordis 路径损坏时,已完成高级快照会失败,不会继续接受伪造的结果事件。 - TUI 视觉回归会产生便于阅读的单元格和样式 diff,而 JSONL fixture 会保留触发生产路径的确切模型分片。 - 模拟器使用 xterm 的拟议缓冲区 API。升级 xterm 时必须重新运行并评审语义投影;终端特有行为仍需由 PTY 冒烟测试覆盖。 -- 金标有意固定指定尺寸下的换行与视口行为。预期布局变更使用无密钥刷新;模型流程变更使用录制模式,并同时评审 JSONL 与终端 diff。 +- 预期输出有意固定指定尺寸下的换行与视口行为。预期布局变更使用无密钥刷新;模型流程变更使用录制模式,并同时评审 JSONL 与终端 diff。 diff --git a/docs/rfc/rejected/simplification/2026-06-20-drop-durable-step-boundaries.md b/docs/rfc/rejected/simplification/2026-06-20-drop-durable-step-boundaries.md index 94313fd0ad..2aad598f6d 100644 --- a/docs/rfc/rejected/simplification/2026-06-20-drop-durable-step-boundaries.md +++ b/docs/rfc/rejected/simplification/2026-06-20-drop-durable-step-boundaries.md @@ -4,7 +4,7 @@ Status: rejected — `step/end` is the durable indication that a model step fini ## Problem -The session log stores `step/start` and `step/end` events even though every step-scoped event already carries `{ turn, step }`: assistant chunks, assistant messages, tool calls, tool results, usage, and errors. `deriveMessages()` ignores step boundaries, ACP ignores them for UI, and the main consumers are invariants, tests, snapshot goldens, and crash repair. +The session log stores `step/start` and `step/end` events even though every step-scoped event already carries `{ turn, step }`: assistant chunks, assistant messages, tool calls, tool results, usage, and errors. `deriveMessages()` ignores step boundaries, ACP ignores them for UI, and the main consumers are invariants, tests, snapshot expected outputs, and crash repair. The rejected argument was that boundary events make the log more ceremonial than informative. In practice, `step/end` is concrete information: a reader can tell whether a model request finished, crashed, or is being repaired without deriving that state from the next event. A bare `step/start` is likewise useful for a model request that began but produced no chunks before failing. diff --git a/docs/rfc/rejected/simplification/2026-06-20-generic-tool-rendering.md b/docs/rfc/rejected/simplification/2026-06-20-generic-tool-rendering.md index 6125feaa8b..87d739ba54 100644 --- a/docs/rfc/rejected/simplification/2026-06-20-generic-tool-rendering.md +++ b/docs/rfc/rejected/simplification/2026-06-20-generic-tool-rendering.md @@ -22,7 +22,7 @@ As a smaller alternative, replace the current optional-field bag with one explic - `ToolCallPresentation`, `ToolResultPresentation`, `ToolTerminal`, and `ToolCallKind` disappear unless a minimal generic UI type still needs one. - ACP no longer keeps presenter pending state or calls tool callbacks during live streaming/load replay. - `dsh-tool-bash` no longer parses rendered text to recover exit status for a UI pill. -- Snapshot goldens show generic tool cards and text results. +- Snapshot expected outputs show generic tool cards and text results. ## What we give up diff --git a/docs/testing.md b/docs/testing.md index cd3b1aa9e8..1a2d6cb192 100644 --- a/docs/testing.md +++ b/docs/testing.md @@ -7,7 +7,7 @@ How this repo tests, tier by tier, and the rules that keep a green suite meaning - **Unit** (`pnpm run test`): vitest over `packages|examples/*/tests/**/*.spec.ts`, colocated with what they test. Every registry gets an HMR-safety test (dispose the contributing fiber, assert cleanup). Prefer edge cases, error paths, event ordering, concurrency races, and permanent contract regressions (see `packages/core/agent-loop/tests/contract-regressions.spec.ts`). - **Coverage gate** (`pnpm run test:coverage`): the gating run, per-file 100% on `packages/*/*/src`. An uncovered line is often dead code the gate is correctly flagging for deletion, not a missing test to bolt on. Line coverage is necessary, never sufficient — it proves lines ran, not that the feature works as shipped. - **Real-API e2e** (`pnpm run test:e2e`): with-key tests against live provider APIs — the DeepSeek model plus provider-specific smokes that gate on their own keys (`EXA_API_KEY`, `PERPLEXITY_API_KEY`, …); each suite self-skips without its key so keyless CI stays green ([real-API e2e RFC](rfc/implemented/testing/2026-06-19-real-api-e2e-ci.md)). -- **Snapshot** (`pnpm run test:snapshot`): transport-specific keyless goldens cover external presentation. ACP suites boot the real example subprocess, replay a recorded session, and diff normalized JSON-RPC plus the re-persisted log ([ACP snapshot RFC](rfc/implemented/testing/2026-06-19-acp-snapshot-tests.md)); the headless suite independently pins `stream-json` through its real one-shot subprocess. TUI completed journeys replay recorded primary/child JSONL through the real agent loop and tools before projecting ANSI into semantic terminal-state goldens; package-local snapshots retain transient renderer states, and a real PTY conversation covers the process boundary ([TUI snapshot RFC](rfc/implemented/testing/2026-07-18-tui-terminal-state-snapshots.md)). Use `pnpm run test:snapshot:record` when a model transcript must change and `pnpm run test:snapshot:refresh` when committed replay input remains correct; review every JSONL and golden diff. System-prompt/tool-schema content is pinned by one ACP scenario (`text-turn`) and tokenized in every other fixture, so a prompt or schema edit churns one committed line ([pinned-header RFC](rfc/implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md)). +- **Snapshot** (`pnpm run test:snapshot`): transport-specific keyless expected outputs cover external presentation. ACP suites boot the real example subprocess, replay a recorded session, and diff normalized JSON-RPC plus the re-persisted log ([ACP snapshot RFC](rfc/implemented/testing/2026-06-19-acp-snapshot-tests.md)); the headless suite independently pins `stream-json` through its real one-shot subprocess. TUI completed journeys replay recorded primary/child JSONL through the real agent loop and tools before projecting ANSI into semantic terminal-state expected outputs; package-local snapshots retain transient renderer states, and a real PTY conversation covers the process boundary ([TUI snapshot RFC](rfc/implemented/testing/2026-07-18-tui-terminal-state-snapshots.md)). Use `pnpm run test:snapshot:record` when a model transcript must change and `pnpm run test:snapshot:refresh` when committed replay input remains correct; review every JSONL and expected-output diff. System-prompt/tool-schema content is pinned by one ACP scenario (`text-turn`) and tokenized in every other fixture, so a prompt or schema edit churns one committed line ([pinned-header RFC](rfc/implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md)). ## The with-key policy: inference is cheap here diff --git a/examples/acp-agent/tests/acp.snapshot.ts b/examples/acp-agent/tests/acp.snapshot.ts index eb9241cae9..37b420378c 100644 --- a/examples/acp-agent/tests/acp.snapshot.ts +++ b/examples/acp-agent/tests/acp.snapshot.ts @@ -5,10 +5,10 @@ import { defineAcpSnapshotSuite, type Scenario, type SnapshotSuiteOptions } from /** * The acp-agent example's snapshot suite: the scenario table for * `dsh-acp-snapshot`'s suite factory, which owns every compare/guard mechanic - * (golden + re-persisted-log diffs, record/refresh write-back, the pinned-header + * (expected-output + re-persisted-log diffs, record/refresh write-back, the pinned-header * uniformity guard, the fixture guards). Fixtures live under `snapshots//`; * `pnpm run test:snapshot:record` re-records model transcripts against the real - * API; `pnpm run test:snapshot:refresh` rewrites current replay goldens keyless. + * API; `pnpm run test:snapshot:refresh` rewrites current replay expected outputs keyless. * See the package README (packages/support/acp-snapshot) and the snapshot RFC, * docs/rfc/implemented/testing/2026-06-19-acp-snapshot-tests.md. */ @@ -137,7 +137,7 @@ const SCENARIOS: Scenario[] = [ // The mid-turn seams fire during a real model turn, so each is recorded with its hook active // (the model's reaction to a deny/block/force-continue is part of the captured transcript). // SessionStart/SubagentStart are excluded because detached injection races log - // order; SubagentStop writes no transcript, so a golden could not prove it ran. + // order; SubagentStop writes no transcript, so an expected output could not prove it ran. // Unit tests cover those points; the hook-snapshot-matrix RFC owns the rationale. { name: 'hook-cc-promptsubmit-context', hasModelTurn: true, recorded: true }, { name: 'hook-cc-pretool-deny', hasModelTurn: true, recorded: true }, diff --git a/examples/acp-agent/tests/snapshots/advanced-toolchain/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/advanced-toolchain/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/advanced-toolchain/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/advanced-toolchain/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/advanced-toolchain/system-prompt.golden.md b/examples/acp-agent/tests/snapshots/advanced-toolchain/system-prompt.expected.md similarity index 100% rename from examples/acp-agent/tests/snapshots/advanced-toolchain/system-prompt.golden.md rename to examples/acp-agent/tests/snapshots/advanced-toolchain/system-prompt.expected.md diff --git a/examples/acp-agent/tests/snapshots/advanced-toolchain/tool-schemas.golden.json b/examples/acp-agent/tests/snapshots/advanced-toolchain/tool-schemas.expected.json similarity index 100% rename from examples/acp-agent/tests/snapshots/advanced-toolchain/tool-schemas.golden.json rename to examples/acp-agent/tests/snapshots/advanced-toolchain/tool-schemas.expected.json diff --git a/examples/acp-agent/tests/snapshots/bash-spill/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/bash-spill/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/bash-spill/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/bash-spill/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/both-mode-turn/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/both-mode-turn/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/both-mode-turn/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/both-mode-turn/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/both-mode-turn/system-prompt.golden.md b/examples/acp-agent/tests/snapshots/both-mode-turn/system-prompt.expected.md similarity index 100% rename from examples/acp-agent/tests/snapshots/both-mode-turn/system-prompt.golden.md rename to examples/acp-agent/tests/snapshots/both-mode-turn/system-prompt.expected.md diff --git a/examples/acp-agent/tests/snapshots/both-mode-turn/tool-schemas.golden.json b/examples/acp-agent/tests/snapshots/both-mode-turn/tool-schemas.expected.json similarity index 100% rename from examples/acp-agent/tests/snapshots/both-mode-turn/tool-schemas.golden.json rename to examples/acp-agent/tests/snapshots/both-mode-turn/tool-schemas.expected.json diff --git a/examples/acp-agent/tests/snapshots/cancel-tool-calls/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/cancel-tool-calls/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/cancel-tool-calls/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/cancel-tool-calls/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/cancel/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/cancel/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/cancel/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/cancel/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/code-mode-turn/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/code-mode-turn/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/code-mode-turn/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/code-mode-turn/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/code-mode-turn/system-prompt.golden.md b/examples/acp-agent/tests/snapshots/code-mode-turn/system-prompt.expected.md similarity index 100% rename from examples/acp-agent/tests/snapshots/code-mode-turn/system-prompt.golden.md rename to examples/acp-agent/tests/snapshots/code-mode-turn/system-prompt.expected.md diff --git a/examples/acp-agent/tests/snapshots/code-mode-turn/tool-schemas.golden.json b/examples/acp-agent/tests/snapshots/code-mode-turn/tool-schemas.expected.json similarity index 100% rename from examples/acp-agent/tests/snapshots/code-mode-turn/tool-schemas.golden.json rename to examples/acp-agent/tests/snapshots/code-mode-turn/tool-schemas.expected.json diff --git a/examples/acp-agent/tests/snapshots/code-mode-workspace-context/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/code-mode-workspace-context/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/code-mode-workspace-context/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/code-mode-workspace-context/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/code-mode-workspace-context/system-prompt.golden.md b/examples/acp-agent/tests/snapshots/code-mode-workspace-context/system-prompt.expected.md similarity index 100% rename from examples/acp-agent/tests/snapshots/code-mode-workspace-context/system-prompt.golden.md rename to examples/acp-agent/tests/snapshots/code-mode-workspace-context/system-prompt.expected.md diff --git a/examples/acp-agent/tests/snapshots/code-mode-workspace-context/tool-schemas.golden.json b/examples/acp-agent/tests/snapshots/code-mode-workspace-context/tool-schemas.expected.json similarity index 100% rename from examples/acp-agent/tests/snapshots/code-mode-workspace-context/tool-schemas.golden.json rename to examples/acp-agent/tests/snapshots/code-mode-workspace-context/tool-schemas.expected.json diff --git a/examples/acp-agent/tests/snapshots/config-options/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/config-options/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/config-options/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/config-options/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/cordis-inspect-jsdoc/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/cordis-inspect-jsdoc/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/cordis-inspect-jsdoc/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/cordis-inspect-jsdoc/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/error-finish/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/error-finish/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/error-finish/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/error-finish/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/escalation-approved/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/escalation-approved/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/escalation-approved/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/escalation-approved/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/escalation-rejected/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/escalation-rejected/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/escalation-rejected/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/escalation-rejected/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/fs-edit/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/fs-edit/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/fs-edit/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/fs-edit/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/fs-policy-reject/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/fs-policy-reject/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/fs-policy-reject/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/fs-policy-reject/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/fs-read-window/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/fs-read-window/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/fs-read-window/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/fs-read-window/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/fs-read/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/fs-read/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/fs-read/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/fs-read/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/fs-terminal-card/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/fs-terminal-card/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/fs-terminal-card/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/fs-terminal-card/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/fs-write-overwrite/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/fs-write-overwrite/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/fs-write-overwrite/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/fs-write-overwrite/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/fs-write/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/fs-write/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/fs-write/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/fs-write/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/handshake/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/handshake/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/handshake/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/handshake/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/hook-cc-posttool-block/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/hook-cc-posttool-block/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/hook-cc-posttool-block/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/hook-cc-posttool-block/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/hook-cc-posttool-context/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/hook-cc-posttool-context/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/hook-cc-posttool-context/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/hook-cc-posttool-context/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/hook-cc-pretool-ask/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/hook-cc-pretool-ask/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/hook-cc-pretool-ask/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/hook-cc-pretool-ask/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/hook-cc-pretool-deny/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/hook-cc-pretool-deny/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/hook-cc-pretool-deny/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/hook-cc-pretool-deny/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/hook-cc-promptsubmit-block/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/hook-cc-promptsubmit-block/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/hook-cc-promptsubmit-block/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/hook-cc-promptsubmit-block/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/hook-cc-promptsubmit-context/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/hook-cc-promptsubmit-context/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/hook-cc-promptsubmit-context/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/hook-cc-promptsubmit-context/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/hook-cc-stop-continue/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/hook-cc-stop-continue/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/hook-cc-stop-continue/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/hook-cc-stop-continue/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/hook-codex-posttool-block/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/hook-codex-posttool-block/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/hook-codex-posttool-block/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/hook-codex-posttool-block/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/hook-codex-posttool-context/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/hook-codex-posttool-context/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/hook-codex-posttool-context/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/hook-codex-posttool-context/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/hook-codex-pretool-block/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/hook-codex-pretool-block/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/hook-codex-pretool-block/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/hook-codex-pretool-block/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/hook-codex-promptsubmit-block/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/hook-codex-promptsubmit-block/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/hook-codex-promptsubmit-block/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/hook-codex-promptsubmit-block/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/hook-codex-promptsubmit-context/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/hook-codex-promptsubmit-context/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/hook-codex-promptsubmit-context/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/hook-codex-promptsubmit-context/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/hook-codex-stop-continue/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/hook-codex-stop-continue/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/hook-codex-stop-continue/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/hook-codex-stop-continue/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/model-switching/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/model-switching/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/model-switching/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/model-switching/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/model-switching/system-prompt.golden.md b/examples/acp-agent/tests/snapshots/model-switching/system-prompt.expected.md similarity index 100% rename from examples/acp-agent/tests/snapshots/model-switching/system-prompt.golden.md rename to examples/acp-agent/tests/snapshots/model-switching/system-prompt.expected.md diff --git a/examples/acp-agent/tests/snapshots/model-switching/tool-schemas.golden.json b/examples/acp-agent/tests/snapshots/model-switching/tool-schemas.expected.json similarity index 100% rename from examples/acp-agent/tests/snapshots/model-switching/tool-schemas.golden.json rename to examples/acp-agent/tests/snapshots/model-switching/tool-schemas.expected.json diff --git a/examples/acp-agent/tests/snapshots/multi-turn/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/multi-turn/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/multi-turn/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/multi-turn/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/parallel-tool-calls/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/parallel-tool-calls/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/parallel-tool-calls/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/parallel-tool-calls/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/permission-switching/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/permission-switching/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/permission-switching/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/permission-switching/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/permission-switching/system-prompt.golden.md b/examples/acp-agent/tests/snapshots/permission-switching/system-prompt.expected.md similarity index 100% rename from examples/acp-agent/tests/snapshots/permission-switching/system-prompt.golden.md rename to examples/acp-agent/tests/snapshots/permission-switching/system-prompt.expected.md diff --git a/examples/acp-agent/tests/snapshots/permission-switching/tool-schemas.golden.json b/examples/acp-agent/tests/snapshots/permission-switching/tool-schemas.expected.json similarity index 100% rename from examples/acp-agent/tests/snapshots/permission-switching/tool-schemas.golden.json rename to examples/acp-agent/tests/snapshots/permission-switching/tool-schemas.expected.json diff --git a/examples/acp-agent/tests/snapshots/reject-extra-dirs/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/reject-extra-dirs/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/reject-extra-dirs/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/reject-extra-dirs/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/repeat-tool-guard/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/repeat-tool-guard/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/repeat-tool-guard/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/repeat-tool-guard/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/skill-load/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/skill-load/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/skill-load/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/skill-load/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/skill-load/system-prompt.golden.md b/examples/acp-agent/tests/snapshots/skill-load/system-prompt.expected.md similarity index 100% rename from examples/acp-agent/tests/snapshots/skill-load/system-prompt.golden.md rename to examples/acp-agent/tests/snapshots/skill-load/system-prompt.expected.md diff --git a/examples/acp-agent/tests/snapshots/skill-load/tool-schemas.golden.json b/examples/acp-agent/tests/snapshots/skill-load/tool-schemas.expected.json similarity index 100% rename from examples/acp-agent/tests/snapshots/skill-load/tool-schemas.golden.json rename to examples/acp-agent/tests/snapshots/skill-load/tool-schemas.expected.json diff --git a/examples/acp-agent/tests/snapshots/subagent-fork/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/subagent-fork/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/subagent-fork/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/subagent-fork/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/subagent-mixed/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/subagent-mixed/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/subagent-mixed/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/subagent-mixed/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/subagent-multi/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/subagent-multi/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/subagent-multi/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/subagent-multi/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/subagent-spawn/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/subagent-spawn/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/subagent-spawn/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/subagent-spawn/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/text-turn/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/text-turn/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/text-turn/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/text-turn/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/text-turn/system-prompt.golden.md b/examples/acp-agent/tests/snapshots/text-turn/system-prompt.expected.md similarity index 100% rename from examples/acp-agent/tests/snapshots/text-turn/system-prompt.golden.md rename to examples/acp-agent/tests/snapshots/text-turn/system-prompt.expected.md diff --git a/examples/acp-agent/tests/snapshots/text-turn/tool-schemas.golden.json b/examples/acp-agent/tests/snapshots/text-turn/tool-schemas.expected.json similarity index 100% rename from examples/acp-agent/tests/snapshots/text-turn/tool-schemas.golden.json rename to examples/acp-agent/tests/snapshots/text-turn/tool-schemas.expected.json diff --git a/examples/acp-agent/tests/snapshots/todo-plan/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/todo-plan/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/todo-plan/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/todo-plan/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/tool-call-turn/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/tool-call-turn/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/tool-call-turn/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/tool-call-turn/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/workflow-run/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/workflow-run/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/workflow-run/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/workflow-run/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/workspace-context/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/workspace-context/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/workspace-context/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/workspace-context/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/workspace-context/system-prompt.golden.md b/examples/acp-agent/tests/snapshots/workspace-context/system-prompt.expected.md similarity index 100% rename from examples/acp-agent/tests/snapshots/workspace-context/system-prompt.golden.md rename to examples/acp-agent/tests/snapshots/workspace-context/system-prompt.expected.md diff --git a/examples/acp-agent/tests/snapshots/workspace-context/tool-schemas.golden.json b/examples/acp-agent/tests/snapshots/workspace-context/tool-schemas.expected.json similarity index 100% rename from examples/acp-agent/tests/snapshots/workspace-context/tool-schemas.golden.json rename to examples/acp-agent/tests/snapshots/workspace-context/tool-schemas.expected.json diff --git a/examples/acp-agent/tests/snapshots/workspace-edit/stdout.golden.jsonl b/examples/acp-agent/tests/snapshots/workspace-edit/stdout.expected.jsonl similarity index 100% rename from examples/acp-agent/tests/snapshots/workspace-edit/stdout.golden.jsonl rename to examples/acp-agent/tests/snapshots/workspace-edit/stdout.expected.jsonl diff --git a/examples/acp-agent/tests/snapshots/workspace-edit/system-prompt.golden.md b/examples/acp-agent/tests/snapshots/workspace-edit/system-prompt.expected.md similarity index 100% rename from examples/acp-agent/tests/snapshots/workspace-edit/system-prompt.golden.md rename to examples/acp-agent/tests/snapshots/workspace-edit/system-prompt.expected.md diff --git a/examples/acp-agent/tests/snapshots/workspace-edit/tool-schemas.golden.json b/examples/acp-agent/tests/snapshots/workspace-edit/tool-schemas.expected.json similarity index 100% rename from examples/acp-agent/tests/snapshots/workspace-edit/tool-schemas.golden.json rename to examples/acp-agent/tests/snapshots/workspace-edit/tool-schemas.expected.json diff --git a/examples/headless-agent/tests/headless.snapshot.ts b/examples/headless-agent/tests/headless.snapshot.ts index ff27347246..dc448b9b23 100644 --- a/examples/headless-agent/tests/headless.snapshot.ts +++ b/examples/headless-agent/tests/headless.snapshot.ts @@ -13,7 +13,7 @@ import { describe, expect, it } from 'vitest' const snapshotsDir = join(dirname(fileURLToPath(import.meta.url)), 'snapshots') const scenarioDir = join(snapshotsDir, 'advanced-toolchain') const sessionFixture = join(scenarioDir, 'session.jsonl') -const streamGolden = join(scenarioDir, 'stream-json.golden.jsonl') +const streamExpected = join(scenarioDir, 'stream-json.expected.jsonl') const configPath = fileURLToPath(new URL('../advanced.cordis.snapshot.yml', import.meta.url)) const binScript = fileURLToPath(new URL('../../../packages/examples/cli-demo/src/bin.ts', import.meta.url)) const tsconfigPath = fileURLToPath(new URL('../../../tsconfig.json', import.meta.url)) @@ -134,7 +134,7 @@ describe('headless stream-json snapshots', () => { expect(result.stderr).toBe('') const normalized = normalizeHeadlessStream(result.stdout, runCwd) - if (refreshing) await writeFile(streamGolden, normalized) - expect(normalized).toBe(await readFile(streamGolden, 'utf8')) + if (refreshing) await writeFile(streamExpected, normalized) + expect(normalized).toBe(await readFile(streamExpected, 'utf8')) }, LOADER_SMOKE_TEST_TIMEOUT_MS) }) diff --git a/examples/headless-agent/tests/snapshots/advanced-toolchain/stream-json.golden.jsonl b/examples/headless-agent/tests/snapshots/advanced-toolchain/stream-json.expected.jsonl similarity index 100% rename from examples/headless-agent/tests/snapshots/advanced-toolchain/stream-json.golden.jsonl rename to examples/headless-agent/tests/snapshots/advanced-toolchain/stream-json.expected.jsonl diff --git a/examples/tui-agent/README.md b/examples/tui-agent/README.md index b4074881fb..6646649b1b 100644 --- a/examples/tui-agent/README.md +++ b/examples/tui-agent/README.md @@ -20,4 +20,4 @@ Run `pnpm run demo:code-mode tui` for the sibling Code Mode overlay. ## Snapshot tests -`tests/snapshots//session.jsonl` supplies recorded user prompts and model chunks; sibling child logs drive subagents and workflows. The keyless suite executes those scripts through the real loop and tool implementations, then compares readable terminal cell/style goldens. Use `pnpm run test:snapshot:refresh` for presentation-only changes and `pnpm run test:snapshot:record` with a DeepSeek key when a recorded model journey changes. The implemented [TUI snapshot RFC](../../docs/rfc/implemented/testing/2026-07-18-tui-terminal-state-snapshots.md) owns the scenario matrix and the split between recorded journeys, transient package snapshots, and PTY coverage. +`tests/snapshots//session.jsonl` supplies recorded user prompts and model chunks; sibling child logs drive subagents and workflows. The keyless suite executes those scripts through the real loop and tool implementations, then compares readable expected terminal cell/style output. Use `pnpm run test:snapshot:refresh` for presentation-only changes and `pnpm run test:snapshot:record` with a DeepSeek key when a recorded model journey changes. The implemented [TUI snapshot RFC](../../docs/rfc/implemented/testing/2026-07-18-tui-terminal-state-snapshots.md) owns the scenario matrix and the split between recorded journeys, transient package snapshots, and PTY coverage. diff --git a/examples/tui-agent/tests/snapshots/bash-terminal-card/terminal.golden.txt b/examples/tui-agent/tests/snapshots/bash-terminal-card/terminal.expected.txt similarity index 100% rename from examples/tui-agent/tests/snapshots/bash-terminal-card/terminal.golden.txt rename to examples/tui-agent/tests/snapshots/bash-terminal-card/terminal.expected.txt diff --git a/examples/tui-agent/tests/snapshots/code-mode/terminal.golden.txt b/examples/tui-agent/tests/snapshots/code-mode/terminal.expected.txt similarity index 100% rename from examples/tui-agent/tests/snapshots/code-mode/terminal.golden.txt rename to examples/tui-agent/tests/snapshots/code-mode/terminal.expected.txt diff --git a/examples/tui-agent/tests/snapshots/cordis-dynamic-toolchain/terminal.golden.txt b/examples/tui-agent/tests/snapshots/cordis-dynamic-toolchain/terminal.expected.txt similarity index 100% rename from examples/tui-agent/tests/snapshots/cordis-dynamic-toolchain/terminal.golden.txt rename to examples/tui-agent/tests/snapshots/cordis-dynamic-toolchain/terminal.expected.txt diff --git a/examples/tui-agent/tests/snapshots/dynamic-workflow/terminal.golden.txt b/examples/tui-agent/tests/snapshots/dynamic-workflow/terminal.expected.txt similarity index 100% rename from examples/tui-agent/tests/snapshots/dynamic-workflow/terminal.golden.txt rename to examples/tui-agent/tests/snapshots/dynamic-workflow/terminal.expected.txt diff --git a/examples/tui-agent/tests/snapshots/multi-turn-conversation/terminal.golden.txt b/examples/tui-agent/tests/snapshots/multi-turn-conversation/terminal.expected.txt similarity index 100% rename from examples/tui-agent/tests/snapshots/multi-turn-conversation/terminal.golden.txt rename to examples/tui-agent/tests/snapshots/multi-turn-conversation/terminal.expected.txt diff --git a/examples/tui-agent/tests/snapshots/parallel-file-reads/terminal.golden.txt b/examples/tui-agent/tests/snapshots/parallel-file-reads/terminal.expected.txt similarity index 100% rename from examples/tui-agent/tests/snapshots/parallel-file-reads/terminal.golden.txt rename to examples/tui-agent/tests/snapshots/parallel-file-reads/terminal.expected.txt diff --git a/examples/tui-agent/tests/snapshots/todo-plan/terminal.golden.txt b/examples/tui-agent/tests/snapshots/todo-plan/terminal.expected.txt similarity index 100% rename from examples/tui-agent/tests/snapshots/todo-plan/terminal.golden.txt rename to examples/tui-agent/tests/snapshots/todo-plan/terminal.expected.txt diff --git a/examples/tui-agent/tests/tui.snapshot.ts b/examples/tui-agent/tests/tui.snapshot.ts index c0537021e8..534207c875 100644 --- a/examples/tui-agent/tests/tui.snapshot.ts +++ b/examples/tui-agent/tests/tui.snapshot.ts @@ -296,7 +296,7 @@ describe('TUI recorded-session terminal snapshots', () => { it(scenario.name, async () => { observedScenarios.add(scenario.name) const result = await runScenario(scenario) - const terminalFile = join(scenarioDir(scenario), 'terminal.golden.txt') + const terminalFile = join(scenarioDir(scenario), 'terminal.expected.txt') if (MODE === 'record' || MODE === 'refresh') { await mkdir(scenarioDir(scenario), { recursive: true }) await writeFile(terminalFile, result.terminal) @@ -317,7 +317,7 @@ afterAll(async () => { for (const scenario of SCENARIOS) { const expected = [ 'session.jsonl', - 'terminal.golden.txt', + 'terminal.expected.txt', ...scenario.seedWorkspace === true ? ['workspace'] : [], ...Array.from({ length: scenario.childSessions ?? 0 }, (_, index) => `session.${index + 1}.jsonl`), ].sort() diff --git a/packages/core/agent-loop/tests/tool-calls.spec.ts b/packages/core/agent-loop/tests/tool-calls.spec.ts index 638f7fa8c7..a77e1d678e 100644 --- a/packages/core/agent-loop/tests/tool-calls.spec.ts +++ b/packages/core/agent-loop/tests/tool-calls.spec.ts @@ -1,6 +1,6 @@ /** * Exercises scheduler ordering and cancellation with deterministic gated tools. - * ACP goldens own transcript-facing coverage. + * ACP expected outputs own transcript-facing coverage. */ import { describe, expect, it } from 'vitest' diff --git a/packages/support/acp-snapshot/README.md b/packages/support/acp-snapshot/README.md index 27e0012d19..f9faaca8a7 100644 --- a/packages/support/acp-snapshot/README.md +++ b/packages/support/acp-snapshot/README.md @@ -5,9 +5,9 @@ The ACP snapshot suite kit: the shared machinery behind the keyless snapshot tie Four layers, importable separately: - **`launchAcpTestAgent` (launcher)** — boots an unbuilt ACP agent from a temp cwd, pins tsx to the repo tsconfig, connects the SDK client over a raw-byte stdout tee, collects session updates and stderr, surfaces asynchronous spawn failures through its startup lifecycle, fails closed on unhandled permission requests, and owns graceful or signalled shutdown. Shutdown waits for process exit, inherited stdio closure, and ACP parser exhaustion before resolving or propagating a child error, so captures are complete and callers can remove owned paths after either outcome. Snapshot and ordinary e2e suites share this process boundary; a test supplies only agent paths, cwd, environment overrides, and any permission policy. -- **`runScenario` (harness)** — boots the real agent bin as a subprocess via tsx (unbuilt, Loader path), drives it over ACP JSON-RPC stdio from a deterministic `input.json` script, tees raw stdout for the golden + purity check, and harvests every persisted session JSONL (parent + subagent children, primary-first) after a graceful stdin-EOF shutdown. Parameterized by `AgentUnderTest` (`binScript`, `configPath`, `tsconfigPath` — absolute paths; the subprocess cwd is a temp dir outside the repo). Startup failures preserve captured agent stderr in the rejected diagnostic. +- **`runScenario` (harness)** — boots the real agent bin as a subprocess via tsx (unbuilt, Loader path), drives it over ACP JSON-RPC stdio from a deterministic `input.json` script, tees raw stdout for the expected-output and purity checks, and harvests every persisted session JSONL (parent + subagent children, primary-first) after a graceful stdin-EOF shutdown. Parameterized by `AgentUnderTest` (`binScript`, `configPath`, `tsconfigPath` — absolute paths; the subprocess cwd is a temp dir outside the repo). Startup failures preserve captured agent stderr in the rejected diagnostic. - **Normalizers** — pure functions turning the two captured surfaces into stable text: `normalizeStdout` (JSON-RPC ids → first-seen sequence; UUIDs/cwd → tokens; doubles as the stdout-purity check), `normalizeSessionLog` (times zeroed, `seq` kept), `scrubSystemPrompts` (prompt text → `{{system}}`), `scrubToolSchemas` (schema bulk → `{{tools}}`), and `scrubRequestHeaders` (all header bulk → `{{system}}`/`{{tools}}`/`{{messagePrefix}}` outside each pin, structure kept — [pinned-header RFC](../../../docs/rfc/implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md)). -- **`defineAcpSnapshotSuite` (factory)** — registers the whole describe/it tree for a scenario table: per-scenario golden + re-persisted-log compares, record/refresh fixture write-back, rejection of structured `UNKNOWN_TOOL` results, the per-header-class pin (`system-prompt.golden.md` plus `tool-schemas.golden.json`) with its live uniformity guard, and the fixture guard block (no orphan scenario dirs, required files present, exactly one pin per class, every JSONL prompt/schema-scrubbed, non-pinning fixtures fully header-scrubbed). Each scenario directory's `session.jsonl` plus contiguous `session..jsonl` siblings are the ordered primary/child inventory; the scenario table does not duplicate their count. Must be called at vitest collection time. +- **`defineAcpSnapshotSuite` (factory)** — registers the whole describe/it tree for a scenario table: per-scenario expected-output and re-persisted-log comparisons, record/refresh fixture write-back, rejection of structured `UNKNOWN_TOOL` results, the per-header-class pin (`system-prompt.expected.md` plus `tool-schemas.expected.json`) with its live uniformity guard, and the fixture guard block (no orphan scenario dirs, required files present, exactly one pin per class, every JSONL prompt/schema-scrubbed, non-pinning fixtures fully header-scrubbed). Each scenario directory's `session.jsonl` plus contiguous `session..jsonl` siblings are the ordered primary/child inventory; the scenario table does not duplicate their count. Must be called at vitest collection time. A consuming `*.snapshot.ts` is the scenario table plus one factory call: @@ -36,9 +36,9 @@ defineAcpSnapshotSuite({ }) ``` -A scenario booting a differently-composed tree sets its own `configPath` (an overlay whose basename still ends in `cordis.yml`, so the bin's replay swap finds the sibling `*cordis.snapshot.yml`) and, when that composition changes the request header, its own `headerClass` with its own pinning scenario — the acp-agent example's Code Mode and filesystem scenarios are templates. Each pinning directory stores the normalized full prompt sequence in generated `system-prompt.golden.md` and the corresponding full tool-schema sequence in generated `tool-schemas.golden.json`; `session.jsonl` stores `"system":"{{system}}","tools":"{{tools}}"` while retaining config, reason, and any model-visible prefix. A pin with legitimate mid-run header changes declares `expectedHeaderChanges`, which fixes the length of both sidecar sequences. +A scenario booting a differently-composed tree sets its own `configPath` (an overlay whose basename still ends in `cordis.yml`, so the bin's replay swap finds the sibling `*cordis.snapshot.yml`) and, when that composition changes the request header, its own `headerClass` with its own pinning scenario — the acp-agent example's Code Mode and filesystem scenarios are templates. Each pinning directory stores the normalized full prompt sequence in generated `system-prompt.expected.md` and the corresponding full tool-schema sequence in generated `tool-schemas.expected.json`; `session.jsonl` stores `"system":"{{system}}","tools":"{{tools}}"` while retaining config, reason, and any model-visible prefix. A pin with legitimate mid-run header changes declares `expectedHeaderChanges`, which fixes the length of both sidecar sequences. -The example also ships a `cordis.snapshot.yml` replay overlay next to its `cordis.yml` (the bin swaps them under `DSH_SNAPSHOT=replay` — [single-source replay config RFC](../../../docs/rfc/implemented/testing/2026-07-04-single-source-acp-replay-config.md)); replay fixtures are served by [`dsh-llm-replay`](../llm-replay/README.md), which this package points at via the `DSH_SNAPSHOT_*` env vars it sets on the child. `pnpm run test:snapshot:record` calls the live LLM and rewrites the recorded scenarios' model fixtures; `pnpm run test:snapshot:refresh` stays keyless, runs the replay overlay, and rewrites stdout, comparable session-log goldens, and each pin's prompt and tool-schema sidecars from the committed model scripts. Fixture roles, record/replay/refresh semantics, and scenario-table fields are documented on `Scenario` and in the [snapshot RFC](../../../docs/rfc/implemented/testing/2026-06-19-acp-snapshot-tests.md). +The example also ships a `cordis.snapshot.yml` replay overlay next to its `cordis.yml` (the bin swaps them under `DSH_SNAPSHOT=replay` — [single-source replay config RFC](../../../docs/rfc/implemented/testing/2026-07-04-single-source-acp-replay-config.md)); replay fixtures are served by [`dsh-llm-replay`](../llm-replay/README.md), which this package points at via the `DSH_SNAPSHOT_*` env vars it sets on the child. `pnpm run test:snapshot:record` calls the live LLM and rewrites the recorded scenarios' model fixtures; `pnpm run test:snapshot:refresh` stays keyless, runs the replay overlay, and rewrites stdout, comparable session-log expected outputs, and each pin's prompt and tool-schema sidecars from the committed model scripts. Fixture roles, record/replay/refresh semantics, and scenario-table fields are documented on `Scenario` and in the [snapshot RFC](../../../docs/rfc/implemented/testing/2026-06-19-acp-snapshot-tests.md). Constraints: `suite.ts` imports vitest, so the package entry is importable only inside a vitest run (the launcher, harness, and normalizers have no such dependency but ship from the same entry). ACP-specific by design — the launcher speaks the SDK's `ClientSideConnection`. Permission round-trips are scriptable: `InputScript.permissionAnswers` is a FIFO queue of option-kind selections (`allow_once`, `reject_once`, …) the client maps to the agent-issued `optionId` at answer time; an absent or exhausted queue answers `cancelled`, and a kind the request never offered rejects the run (the agent is answered `cancelled`, so a tolerant agent cannot absorb the scenario bug). Session config options are scriptable too: the `setConfigOption` step switches a knob over `session/set_config_option`, and `setConfigOptionExpectError` asserts the bridge rejects an unknown id or out-of-vocabulary value (the error frame stays in the transcript). diff --git a/packages/support/acp-snapshot/package.json b/packages/support/acp-snapshot/package.json index 64e1b0cbd7..9cf18e7824 100644 --- a/packages/support/acp-snapshot/package.json +++ b/packages/support/acp-snapshot/package.json @@ -1,6 +1,6 @@ { "name": "@deepseek-ai/dsh-acp-snapshot", - "description": "ACP test kit: shared subprocess launcher, snapshot scenario harness, golden normalizers, and suite factory", + "description": "ACP test kit: shared subprocess launcher, snapshot scenario harness, expected-output normalizers, and suite factory", "version": "0.0.1", "private": true, "type": "module", diff --git a/packages/support/acp-snapshot/src/harness.ts b/packages/support/acp-snapshot/src/harness.ts index 542b6d6df3..81e02da6a3 100644 --- a/packages/support/acp-snapshot/src/harness.ts +++ b/packages/support/acp-snapshot/src/harness.ts @@ -6,7 +6,7 @@ * It boots the REAL agent bin subprocess via the cordis Loader (so the * export-shape bug class stays guarded — see docs/postmortem/0001), drives it * over real ACP JSON-RPC stdio with a deterministic input script, tees raw - * stdout (for the golden + a purity check) into an SDK `ClientSideConnection`, + * stdout (for the expected-output and purity checks) into an SDK `ClientSideConnection`, * and — in record mode — harvests the persisted session JSONL after a graceful * shutdown flush. The pure normalizers in ./normalize.ts turn the captured * stdout frames and the session-log events into stable, snapshot-able text. @@ -162,7 +162,7 @@ export async function runScenario(input: InputScript, opts: RunOptions): Promise const cwd = await mkdtemp(join(tmpdir(), 'acp-snap-cwd-')) const sessionsRoot = await mkdtemp(join(tmpdir(), 'acp-snap-sessions-')) // Fixed path length: spill-policy budgets the preview against the REAL path - // before stdout normalization, so tmpdir() length differences churn goldens. + // before stdout normalization, so tmpdir() length differences churn expected outputs. const spillRoot = '/tmp/dsh-acp-snapshot-spill' // Everything past the temp-dir creation is followed by failure-safe cleanup, // so a failure in workspace seeding, spawn, or any step never leaks resources. @@ -171,7 +171,7 @@ export async function runScenario(input: InputScript, opts: RunOptions): Promise let sessionLogs: HarvestedLog[] = [] const outcome = await (async (): Promise => { // Seed the workspace if the scenario ships one (a file the agent reads/edits). - // Copied into the temp cwd so the agent's bash tools see it; the goldens + // Copied into the temp cwd so the agent's bash tools see it; the expected outputs // normalize the cwd, so the seeded paths stay stable across runs. if (opts.workspaceDir !== undefined && existsSync(opts.workspaceDir)) { await cp(opts.workspaceDir, cwd, { recursive: true }) diff --git a/packages/support/acp-snapshot/src/index.ts b/packages/support/acp-snapshot/src/index.ts index b7267febe7..4d99cc96a2 100644 --- a/packages/support/acp-snapshot/src/index.ts +++ b/packages/support/acp-snapshot/src/index.ts @@ -2,7 +2,7 @@ * ACP snapshot suite kit — the shared machinery behind the keyless snapshot * tier (`pnpm run test:snapshot`). Four layers, composable per example: the * shared subprocess/client launcher ({@link launchAcpTestAgent}), the scripted - * scenario harness ({@link runScenario}), the pure golden normalizers + * scenario harness ({@link runScenario}), the pure expected-output normalizers * ({@link normalizeStdout} / {@link normalizeSessionLog} / * {@link scrubRequestHeaders} / {@link scrubSystemPrompts}), and the suite * factory ({@link defineAcpSnapshotSuite}) that registers a scenario table as a diff --git a/packages/support/acp-snapshot/src/normalize.ts b/packages/support/acp-snapshot/src/normalize.ts index 48761864f0..0c21046eb2 100644 --- a/packages/support/acp-snapshot/src/normalize.ts +++ b/packages/support/acp-snapshot/src/normalize.ts @@ -60,7 +60,7 @@ function scrubValue(value: unknown, ctx: NormalizeContext): unknown { } /** - * Normalize a raw stdout transcript (newline-delimited JSON-RPC frames) into a stable golden + * Normalize a raw stdout transcript (newline-delimited JSON-RPC frames) into a stable expected output * in the same shape as the wire: one compact JSON frame per line (NDJSON), with the JSON-RPC * `id` rewritten to a per-transcript sequence (1, 2, 3, …) and all volatile strings scrubbed. * Invalid JSON throws, doubling as a protocol-stdout purity check. @@ -72,7 +72,7 @@ function scrubValue(value: unknown, ctx: NormalizeContext): unknown { export function normalizeStdout(rawStdout: string, ctx: NormalizeContext): string { const lines = rawStdout.split('\n').filter(line => line.trim().length > 0) // Map each distinct JSON-RPC id (request/response correlate by id) to a stable - // sequence number, in first-seen order, so id churn doesn't perturb the golden. + // sequence number, in first-seen order, so id churn doesn't perturb the expected output. const idSeq = new Map() const stableId = (id: unknown): number => { const key = JSON.stringify(id) @@ -91,7 +91,7 @@ export function normalizeStdout(rawStdout: string, ctx: NormalizeContext): strin } /** - * Normalize a session JSONL log into a stable golden: the header line's + * Normalize a session JSONL log into a stable expected output: the header line's * volatile fields (`createdAt`, `id`, `cwd`) and every event's `time` are * zeroed/scrubbed, all volatile strings scrubbed, and `seq` is LEFT INTACT * (deterministic by contract). Output is JSONL in the same shape as the input — @@ -112,7 +112,7 @@ export function normalizeSessionLog(rawLog: string, ctx: NormalizeContext): stri // Event line: zero the epoch-ms timestamp; keep seq (deterministic). record.time = 0 // A hook/result carries the hook's wall-clock runtime (`data.durationMs`), - // which is run-to-run noise like `time` — zero it so the golden reflects + // which is run-to-run noise like `time` — zero it so the expected output reflects // the hook's decision/exit, not how long the shell took. if (record.type === 'hook/result' && record.data !== null && typeof record.data === 'object') { const data = record.data as Record diff --git a/packages/support/acp-snapshot/src/suite.ts b/packages/support/acp-snapshot/src/suite.ts index 938676e4b7..60f5cfeadb 100644 --- a/packages/support/acp-snapshot/src/suite.ts +++ b/packages/support/acp-snapshot/src/suite.ts @@ -30,10 +30,10 @@ import { } from './normalize.ts' /** The readable system-prompt snapshot beside each header-pinning fixture. */ -const SYSTEM_PROMPT_SNAPSHOT = 'system-prompt.golden.md' +const SYSTEM_PROMPT_SNAPSHOT = 'system-prompt.expected.md' /** The structured tool-schema snapshot beside each header-pinning fixture. */ -const TOOL_SCHEMAS_SNAPSHOT = 'tool-schemas.golden.json' +const TOOL_SCHEMAS_SNAPSHOT = 'tool-schemas.expected.json' /** Stable session-log token standing in for the sidecar's initial schemas. */ const TOOLS_TOKEN = '{{tools}}' @@ -41,7 +41,7 @@ const TOOLS_TOKEN = '{{tools}}' /** A snapshot scenario and how its fixtures are produced. */ export interface Scenario { name: string - /** Whether the scenario drives at least one model turn (so a JSONL golden applies). */ + /** Whether the scenario drives at least one model turn (so a JSONL expected output applies). */ hasModelTurn: boolean /** * Whether the run persists a comparable session log to diff against the @@ -112,8 +112,8 @@ export interface SnapshotSuiteOptions { scenarios: Scenario[] /** * `replay` (keyless, the default tier), `record` (live API; re-records the - * `recorded` scenarios' fixtures and refreshes the Vitest goldens under - * `--update`), or `refresh` (keyless replay that rewrites stdout goldens and + * `recorded` scenarios' fixtures and refreshes the Vitest expected outputs under + * `--update`), or `refresh` (keyless replay that rewrites stdout expected outputs and * comparable session fixtures from the replay run). The caller derives this * from `$DSH_SNAPSHOT` — env reading stays outside this library. */ @@ -427,7 +427,7 @@ export function stabilizeRefreshLog(fresh: string, existing: string, replacement } /** - * Register the suite: one test per scenario (the golden/log compares and + * Register the suite: one test per scenario (the expected-output and log comparisons and * the header-uniformity guard) plus the fixture guard block (no orphan * scenario dirs, required files present, exactly one pin per header class, * pinning fixtures well-formed, every JSONL prompt-scrubbed, non-pinning @@ -467,7 +467,7 @@ export function defineAcpSnapshotSuite(options: SnapshotSuiteOptions): void { for (const scenario of scenarios) { // In RECORD mode, only re-run the `recorded` (live-API) scenarios; the `authored` ones // (sidecar-driven errors/cancel) are never re-recorded. - it.skipIf(RECORDING && !scenario.recorded)(`snapshot: ${scenario.name} matches the goldens`, async ({ expect }) => { + it.skipIf(RECORDING && !scenario.recorded)(`snapshot: ${scenario.name} matches the expected outputs`, async ({ expect }) => { const dir = join(snapshotsDir, scenario.name) const input = JSON.parse(await readFile(join(dir, 'input.json'), 'utf8')) as InputScript const overrideFile = join(dir, 'replay.override.json') @@ -573,9 +573,9 @@ export function defineAcpSnapshotSuite(options: SnapshotSuiteOptions): void { const stdout = normalizeStdout(result.rawStdout, ctx) if (REFRESHING) { - await writeFile(join(dir, 'stdout.golden.jsonl'), stdout) + await writeFile(join(dir, 'stdout.expected.jsonl'), stdout) } - await expect(stdout).toMatchFileSnapshot(join(dir, 'stdout.golden.jsonl')) + await expect(stdout).toMatchFileSnapshot(join(dir, 'stdout.expected.jsonl')) // A model turn always produces a log worth comparing; a hook scenario can // produce one without a model turn (a `rejected` turn carrying `hook/*`). @@ -651,7 +651,7 @@ export function defineAcpSnapshotSuite(options: SnapshotSuiteOptions): void { describe('snapshot fixtures', () => { it('every scenario directory is registered (no orphans)', async () => { - // toMatchFileSnapshot does not prune orphaned golden/fixture files, so a + // toMatchFileSnapshot does not prune orphaned expected-output or fixture files, so a // renamed/removed scenario could leave a stale dir that nothing exercises. // Fail loud on any snapshots/ not present in the scenario table. const entries = await readdir(snapshotsDir, { withFileTypes: true }) @@ -665,7 +665,7 @@ export function defineAcpSnapshotSuite(options: SnapshotSuiteOptions): void { for (const { name, overridden, pinsHeader } of scenarios) { const dir = join(snapshotsDir, name) expect(existsSync(join(dir, 'input.json')), `${name}/input.json`).toBe(true) - expect(existsSync(join(dir, 'stdout.golden.jsonl')), `${name}/stdout.golden.jsonl`).toBe(true) + expect(existsSync(join(dir, 'stdout.expected.jsonl')), `${name}/stdout.expected.jsonl`).toBe(true) expect(existsSync(join(dir, 'session.jsonl')), `${name}/session.jsonl`).toBe(true) expect(existsSync(join(dir, 'replay.override.json')), `${name}/replay.override.json presence must match \`overridden\``) .toBe(overridden === true) diff --git a/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-child/stdout.golden.jsonl b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-child/stdout.expected.jsonl similarity index 100% rename from packages/support/acp-snapshot/tests/fixtures/record-suite/rec-child/stdout.golden.jsonl rename to packages/support/acp-snapshot/tests/fixtures/record-suite/rec-child/stdout.expected.jsonl diff --git a/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/stdout.golden.jsonl b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/stdout.expected.jsonl similarity index 100% rename from packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/stdout.golden.jsonl rename to packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/stdout.expected.jsonl diff --git a/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/system-prompt.golden.md b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/system-prompt.expected.md similarity index 100% rename from packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/system-prompt.golden.md rename to packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/system-prompt.expected.md diff --git a/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/tool-schemas.golden.json b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/tool-schemas.expected.json similarity index 100% rename from packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/tool-schemas.golden.json rename to packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/tool-schemas.expected.json diff --git a/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-skip/stdout.golden.jsonl b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-skip/stdout.expected.jsonl similarity index 100% rename from packages/support/acp-snapshot/tests/fixtures/record-suite/rec-skip/stdout.golden.jsonl rename to packages/support/acp-snapshot/tests/fixtures/record-suite/rec-skip/stdout.expected.jsonl diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/authored-error/stdout.golden.jsonl b/packages/support/acp-snapshot/tests/fixtures/suite/authored-error/stdout.expected.jsonl similarity index 100% rename from packages/support/acp-snapshot/tests/fixtures/suite/authored-error/stdout.golden.jsonl rename to packages/support/acp-snapshot/tests/fixtures/suite/authored-error/stdout.expected.jsonl diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/blocked-log/stdout.golden.jsonl b/packages/support/acp-snapshot/tests/fixtures/suite/blocked-log/stdout.expected.jsonl similarity index 100% rename from packages/support/acp-snapshot/tests/fixtures/suite/blocked-log/stdout.golden.jsonl rename to packages/support/acp-snapshot/tests/fixtures/suite/blocked-log/stdout.expected.jsonl diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/no-model/stdout.golden.jsonl b/packages/support/acp-snapshot/tests/fixtures/suite/no-model/stdout.expected.jsonl similarity index 100% rename from packages/support/acp-snapshot/tests/fixtures/suite/no-model/stdout.golden.jsonl rename to packages/support/acp-snapshot/tests/fixtures/suite/no-model/stdout.expected.jsonl diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/stdout.golden.jsonl b/packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/stdout.expected.jsonl similarity index 100% rename from packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/stdout.golden.jsonl rename to packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/stdout.expected.jsonl diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/system-prompt.golden.md b/packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/system-prompt.expected.md similarity index 100% rename from packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/system-prompt.golden.md rename to packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/system-prompt.expected.md diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/tool-schemas.golden.json b/packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/tool-schemas.expected.json similarity index 100% rename from packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/tool-schemas.golden.json rename to packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/tool-schemas.expected.json diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/stdout.golden.jsonl b/packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/stdout.expected.jsonl similarity index 100% rename from packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/stdout.golden.jsonl rename to packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/stdout.expected.jsonl diff --git a/packages/support/acp-snapshot/tests/suite.spec.ts b/packages/support/acp-snapshot/tests/suite.spec.ts index 92e676b1d6..ac6f944a4a 100644 --- a/packages/support/acp-snapshot/tests/suite.spec.ts +++ b/packages/support/acp-snapshot/tests/suite.spec.ts @@ -24,7 +24,7 @@ import { /** * Unit tests for the suite factory, by running it: two synthetic suites over the scripted fake * ACP bin (./fixtures/fake-acp-agent.ts) register real describe/it trees at collection time, - * so every factory path — golden and log compares, the per-suite header pin and its uniformity + * so every factory path — expected-output and log comparisons, the per-suite header pin and its uniformity * guard, record-mode fixture write-back, skip semantics, and the fixture guard block — * executes as an ordinary green test. * @@ -61,7 +61,7 @@ const RECORD_SCENARIOS: Scenario[] = [ // Record/refresh modes mutate their snapshots dir, so run them on throwaway // copies — except record's documented bootstrap knob, which regenerates the -// committed record fixtures/goldens in place. +// committed record fixtures and expected outputs in place. const BOOTSTRAP = process.env.ACP_SNAPSHOT_SPEC_BOOTSTRAP === '1' const recordDir = BOOTSTRAP ? RECORD_SRC : mkdtempSync(join(tmpdir(), 'acp-snap-record-suite-')) if (!BOOTSTRAP) { @@ -80,9 +80,9 @@ afterAll(async () => { }) function staleRefreshFixtures(dir: string): void { - writeFileSync(join(dir, 'plain-turn', 'stdout.golden.jsonl'), 'stale stdout\n') - writeFileSync(join(dir, 'pin-turn', 'system-prompt.golden.md'), 'STALE PROMPT\n') - writeFileSync(join(dir, 'pin-turn', 'tool-schemas.golden.json'), '{"initial":[{"name":"stale"}],"changes":[]}\n') + writeFileSync(join(dir, 'plain-turn', 'stdout.expected.jsonl'), 'stale stdout\n') + writeFileSync(join(dir, 'pin-turn', 'system-prompt.expected.md'), 'STALE PROMPT\n') + writeFileSync(join(dir, 'pin-turn', 'tool-schemas.expected.json'), '{"initial":[{"name":"stale"}],"changes":[]}\n') const plainBehaviorFile = join(dir, 'plain-turn', 'behavior.json') const plainBehavior = JSON.parse(readFileSync(plainBehaviorFile, 'utf8')) as Record @@ -117,7 +117,7 @@ describe('defineAcpSnapshotSuite: refresh mode', () => { describe('defineAcpSnapshotSuite: refresh write-back', () => { it('rewrites stdout and comparable logs from a replay-mode child run', () => { - const stdout = readFileSync(join(refreshDir, 'plain-turn', 'stdout.golden.jsonl'), 'utf8') + const stdout = readFileSync(join(refreshDir, 'plain-turn', 'stdout.expected.jsonl'), 'utf8') expect(stdout).not.toContain('stale stdout') expect(stdout).toContain('env:{\\"mode\\":\\"replay\\"') expect(stdout).not.toContain('\\"mode\\":\\"refresh\\"') @@ -130,7 +130,7 @@ describe('defineAcpSnapshotSuite: refresh write-back', () => { expect(authored).toContain('"error":"model exploded"') expect(authored).not.toContain('"error":"stale"') - expect(readFileSync(join(refreshDir, 'pin-turn', 'system-prompt.golden.md'), 'utf8')).toBe([ + expect(readFileSync(join(refreshDir, 'pin-turn', 'system-prompt.expected.md'), 'utf8')).toBe([ 'SYS PROMPT', '', '', @@ -140,7 +140,7 @@ describe('defineAcpSnapshotSuite: refresh write-back', () => { 'NEW PROMPT LINE', '', ].join('\n')) - const schemas = readFileSync(join(refreshDir, 'pin-turn', 'tool-schemas.golden.json'), 'utf8') + const schemas = readFileSync(join(refreshDir, 'pin-turn', 'tool-schemas.expected.json'), 'utf8') expect(schemas).toContain('"description": "D1"') expect(schemas).not.toContain('"name":"stale"') }) @@ -195,7 +195,7 @@ describe('defineAcpSnapshotSuite: registration contract', () => { describe('sessionFixtureNames', () => { it('orders the primary and contiguous child fixtures while ignoring other files', () => { expect(sessionFixtureNames([ - 'stdout.golden.jsonl', + 'stdout.expected.jsonl', 'session.2.jsonl', 'session.jsonl', 'session.1.jsonl', diff --git a/packages/ui/acp/snapshot-replay.md b/packages/ui/acp/snapshot-replay.md index 4242f2406d..9c716148aa 100644 --- a/packages/ui/acp/snapshot-replay.md +++ b/packages/ui/acp/snapshot-replay.md @@ -12,14 +12,14 @@ sequenceDiagram participant Workspace participant Replay as llm-replay adapter participant ACP as acp-agent subprocess - participant Golden as stdout golden + participant Expected as stdout expected output Recorder->>Fixture: session.jsonl + workspace inputs Fixture->>Workspace: seed files and hook configs Fixture->>Replay: recorded StreamChunk script Replay->>ACP: deterministic llm/stream chunks ACP->>Workspace: bash, fs, and hook side effects - ACP->>Golden: normalized sessionUpdate stream - Golden-->>ACP: diff must be empty + ACP->>Expected: normalized sessionUpdate stream + Expected-->>ACP: diff must be empty ``` The fs and hook snapshot matrix is valuable because it proves world state, hook decisions, and failed tool-card rendering, not just that replay returns text. diff --git a/packages/ui/tui/tests/headless-terminal.ts b/packages/ui/tui/tests/headless-terminal.ts index 1155016ab7..03a0eb9ebe 100644 --- a/packages/ui/tui/tests/headless-terminal.ts +++ b/packages/ui/tui/tests/headless-terminal.ts @@ -286,7 +286,7 @@ export class HeadlessTerminal implements Terminal { return violations } - /** Serialize terminal cells and metadata into a stable, reviewable golden. */ + /** Serialize terminal cells and metadata into a stable, reviewable expected output. */ async snapshot(options: TerminalSnapshotOptions = {}): Promise { await this.flush() const buffer = this.emulator.buffer.active diff --git a/packages/ui/tui/tests/snapshots/advanced-cards-collapsed.golden.txt b/packages/ui/tui/tests/snapshots/advanced-cards-collapsed.expected.txt similarity index 100% rename from packages/ui/tui/tests/snapshots/advanced-cards-collapsed.golden.txt rename to packages/ui/tui/tests/snapshots/advanced-cards-collapsed.expected.txt diff --git a/packages/ui/tui/tests/snapshots/advanced-cards-expanded.golden.txt b/packages/ui/tui/tests/snapshots/advanced-cards-expanded.expected.txt similarity index 100% rename from packages/ui/tui/tests/snapshots/advanced-cards-expanded.golden.txt rename to packages/ui/tui/tests/snapshots/advanced-cards-expanded.expected.txt diff --git a/packages/ui/tui/tests/snapshots/code-mode-pending.golden.txt b/packages/ui/tui/tests/snapshots/code-mode-pending.expected.txt similarity index 100% rename from packages/ui/tui/tests/snapshots/code-mode-pending.golden.txt rename to packages/ui/tui/tests/snapshots/code-mode-pending.expected.txt diff --git a/packages/ui/tui/tests/snapshots/conversation-streaming.golden.txt b/packages/ui/tui/tests/snapshots/conversation-streaming.expected.txt similarity index 100% rename from packages/ui/tui/tests/snapshots/conversation-streaming.golden.txt rename to packages/ui/tui/tests/snapshots/conversation-streaming.expected.txt diff --git a/packages/ui/tui/tests/snapshots/cordis-tools-pending.golden.txt b/packages/ui/tui/tests/snapshots/cordis-tools-pending.expected.txt similarity index 100% rename from packages/ui/tui/tests/snapshots/cordis-tools-pending.golden.txt rename to packages/ui/tui/tests/snapshots/cordis-tools-pending.expected.txt diff --git a/packages/ui/tui/tests/snapshots/disposed-terminal.golden.txt b/packages/ui/tui/tests/snapshots/disposed-terminal.expected.txt similarity index 100% rename from packages/ui/tui/tests/snapshots/disposed-terminal.golden.txt rename to packages/ui/tui/tests/snapshots/disposed-terminal.expected.txt diff --git a/packages/ui/tui/tests/snapshots/dynamic-workflow-pending.golden.txt b/packages/ui/tui/tests/snapshots/dynamic-workflow-pending.expected.txt similarity index 100% rename from packages/ui/tui/tests/snapshots/dynamic-workflow-pending.golden.txt rename to packages/ui/tui/tests/snapshots/dynamic-workflow-pending.expected.txt diff --git a/packages/ui/tui/tests/snapshots/errors-and-help.golden.txt b/packages/ui/tui/tests/snapshots/errors-and-help.expected.txt similarity index 100% rename from packages/ui/tui/tests/snapshots/errors-and-help.golden.txt rename to packages/ui/tui/tests/snapshots/errors-and-help.expected.txt diff --git a/packages/ui/tui/tests/snapshots/question-dialog-validation.golden.txt b/packages/ui/tui/tests/snapshots/question-dialog-validation.expected.txt similarity index 100% rename from packages/ui/tui/tests/snapshots/question-dialog-validation.golden.txt rename to packages/ui/tui/tests/snapshots/question-dialog-validation.expected.txt diff --git a/packages/ui/tui/tests/snapshots/question-dialog.golden.txt b/packages/ui/tui/tests/snapshots/question-dialog.expected.txt similarity index 100% rename from packages/ui/tui/tests/snapshots/question-dialog.golden.txt rename to packages/ui/tui/tests/snapshots/question-dialog.expected.txt diff --git a/packages/ui/tui/tests/snapshots/surface-after-compaction-narrow.golden.txt b/packages/ui/tui/tests/snapshots/surface-after-compaction-narrow.expected.txt similarity index 100% rename from packages/ui/tui/tests/snapshots/surface-after-compaction-narrow.golden.txt rename to packages/ui/tui/tests/snapshots/surface-after-compaction-narrow.expected.txt diff --git a/packages/ui/tui/tests/snapshots/surface-after-compaction-wide.golden.txt b/packages/ui/tui/tests/snapshots/surface-after-compaction-wide.expected.txt similarity index 100% rename from packages/ui/tui/tests/snapshots/surface-after-compaction-wide.golden.txt rename to packages/ui/tui/tests/snapshots/surface-after-compaction-wide.expected.txt diff --git a/packages/ui/tui/tests/snapshots/surface-before-compaction.golden.txt b/packages/ui/tui/tests/snapshots/surface-before-compaction.expected.txt similarity index 100% rename from packages/ui/tui/tests/snapshots/surface-before-compaction.golden.txt rename to packages/ui/tui/tests/snapshots/surface-before-compaction.expected.txt diff --git a/packages/ui/tui/tests/snapshots/untrusted-controls.golden.txt b/packages/ui/tui/tests/snapshots/untrusted-controls.expected.txt similarity index 100% rename from packages/ui/tui/tests/snapshots/untrusted-controls.golden.txt rename to packages/ui/tui/tests/snapshots/untrusted-controls.expected.txt diff --git a/packages/ui/tui/tests/tui.snapshot.ts b/packages/ui/tui/tests/tui.snapshot.ts index e0016184fd..609574e758 100644 --- a/packages/ui/tui/tests/tui.snapshot.ts +++ b/packages/ui/tui/tests/tui.snapshot.ts @@ -52,7 +52,7 @@ async function checkpoint( observedCheckpoints.add(name) expect(terminal.themeViolations(), `${name} must remain theme-agnostic`).toEqual([]) const snapshot = await terminal.snapshot(options) - const path = join(SNAPSHOTS_DIR, `${name}.golden.txt`) + const path = join(SNAPSHOTS_DIR, `${name}.expected.txt`) if (REFRESHING) { await mkdir(SNAPSHOTS_DIR, { recursive: true }) await writeFile(path, snapshot) @@ -493,7 +493,7 @@ describe('TUI terminal-state snapshots', () => { afterAll(async () => { expect([...observedCheckpoints].sort()).toEqual([...CHECKPOINTS].sort()) const files = (await readdir(SNAPSHOTS_DIR)) - .filter(file => file.endsWith('.golden.txt')) + .filter(file => file.endsWith('.expected.txt')) .sort() - expect(files).toEqual(CHECKPOINTS.map(name => `${name}.golden.txt`).sort()) + expect(files).toEqual(CHECKPOINTS.map(name => `${name}.expected.txt`).sort()) }) diff --git a/scripts/gen-doc-graphs.ts b/scripts/gen-doc-graphs.ts index 0681153330..4e3f049ac9 100644 --- a/scripts/gen-doc-graphs.ts +++ b/scripts/gen-doc-graphs.ts @@ -976,14 +976,14 @@ function renderSnapshotReplay(): string { ' participant Workspace', ' participant Replay as llm-replay adapter', ' participant ACP as acp-agent subprocess', - ' participant Golden as stdout golden', + ' participant Expected as stdout expected output', ' Recorder->>Fixture: session.jsonl + workspace inputs', ' Fixture->>Workspace: seed files and hook configs', ' Fixture->>Replay: recorded StreamChunk script', ` Replay->>ACP: deterministic ${mermaidCode('llm/stream')} chunks`, ' ACP->>Workspace: bash, fs, and hook side effects', - ' ACP->>Golden: normalized sessionUpdate stream', - ' Golden-->>ACP: diff must be empty', + ' ACP->>Expected: normalized sessionUpdate stream', + ' Expected-->>ACP: diff must be empty', '```', '', 'The fs and hook snapshot matrix is valuable because it proves world state, hook decisions, and failed tool-card rendering, not just that replay returns text.', diff --git a/scripts/smoke-python-runtime.py b/scripts/smoke-python-runtime.py index 643716515f..6061d945e7 100644 --- a/scripts/smoke-python-runtime.py +++ b/scripts/smoke-python-runtime.py @@ -628,7 +628,7 @@ def build_snapshot_files( child_ids: list[str], cwd: Path, ) -> dict[str, str]: - """Render the SDK result and three persisted logs into stable goldens.""" + """Render the SDK result and three persisted logs into stable expected outputs.""" replacements = [(str(cwd), "{{cwd}}"), (SNAPSHOT_SESSION_ID, "{{parent}}")] for index, child_id in enumerate(child_ids, start=1): replacements.append((child_id, f"{{{{child-{index}}}}}")) diff --git a/scripts/verify-md-wrap.ts b/scripts/verify-md-wrap.ts index 20586ec777..f9679e5537 100644 --- a/scripts/verify-md-wrap.ts +++ b/scripts/verify-md-wrap.ts @@ -13,15 +13,15 @@ import { uniqueRepoFiles } from './repo-files.ts' const root = resolve(import.meta.dirname, '..') -/** Files to check: doc-typecheck's scope, prompt goldens, and the AGENTS.md pair. */ +/** Files to check: doc-typecheck's scope, system-prompt expected outputs, and the AGENTS.md pair. */ const PATTERNS = [ 'README.md', 'README.zh.md', 'docs/**/*.md', 'packages/*/*.md', 'packages/*/*/*.md', - 'examples/**/system-prompt.golden.md', - 'packages/**/system-prompt.golden.md', + 'examples/**/system-prompt.expected.md', + 'packages/**/system-prompt.expected.md', 'AGENTS.md', 'packages/AGENTS.md', ] diff --git a/vitest.snapshot.config.ts b/vitest.snapshot.config.ts index fb2e890eb7..78800fb4d5 100644 --- a/vitest.snapshot.config.ts +++ b/vitest.snapshot.config.ts @@ -21,8 +21,8 @@ const snapshotMaxConcurrency = positiveIntFromEnv( ) // Replay is the keyless default: boot real example subprocesses from recorded model scripts and diff -// normalized protocol or transcript output plus persisted-log goldens. `record` calls the real API -// and updates fixtures and goldens; `refresh` replays committed scripts and updates current goldens. +// normalized protocol or transcript output plus persisted-log expected outputs. `record` calls the real API +// and updates fixtures and expected outputs; `refresh` replays committed scripts and updates current expected outputs. // Replay/refresh never load `.env`; only record reads a key from the environment or root `.env`. if (process.env.DSH_SNAPSHOT === 'record') { try {