Merge latest master into PR 555

# Conflicts:
#	docs/architecture.i18n.yaml
#	docs/core-data-structures/core.i18n.yaml
#	examples/acp-agent/tests/snapshots/cordis-inspect-jsdoc/session.jsonl
#	packages/client/runtime/README.i18n.yaml
#	packages/client/ui-conversation/README.i18n.yaml
#	packages/client/ui-conversation/src/client/chat/MessageItem.tsx
#	packages/compact/compact-basic/README.i18n.yaml
#	packages/ui/tui/README.i18n.yaml
This commit is contained in:
creatixchu
2026-07-31 19:39:22 +08:00
187 files changed
+5806 -657

No files matched your search

+51
View File
@@ -0,0 +1,51 @@
import type { Context } from 'cordis'
import type {
GenerateOptions,
LlmModelInfo,
LlmResolvedModelInfo,
StreamChunk,
} from '@deepseek-ai/dsh-llm'
import { LlmAdapter } from '@deepseek-ai/dsh-llm'
/** Terminal marker the preset smoke waits for before it asks the TUI to exit. */
export const COMPOSITION_REPLY_TEXT = 'Shipped composition acknowledged.'
// Provider id and model the keyless tail routes `main` to; that overlay is the
// only caller, so the pair lives here as plain constants.
const COMPOSITION_PROVIDER = 'composition-keyless'
const COMPOSITION_MODEL = 'composition-keyless-model'
/**
* Network-free adapter for the shipped-composition smoke. It answers every
* request — tool-ful agent turns and the tool-less auxiliary calls alike — with
* one fixed text and never calls a tool, because the assertion under test is the
* assembled tool catalog the loop logs, not any tool's behavior.
*/
class CompositionEchoAdapter extends LlmAdapter {
override listModels(provider: string): Promise<readonly LlmModelInfo[]> {
return Promise.resolve([{ provider, id: COMPOSITION_MODEL, name: 'Preset Keyless' }])
}
override resolveModel(provider: string, model: string): Promise<LlmResolvedModelInfo> {
return Promise.resolve({ provider, id: model, name: 'Preset Keyless', context: { contextWindow: 128_000 } })
}
override async * stream(_options: GenerateOptions): AsyncIterable<StreamChunk> {
yield { type: 'block-start', index: 0, blockType: 'text' }
for (const char of COMPOSITION_REPLY_TEXT) yield { type: 'text-delta', index: 0, text: char }
yield { type: 'block-end', index: 0, block: { type: 'text', text: COMPOSITION_REPLY_TEXT } }
yield { type: 'usage', usage: { inputTokens: 20, outputTokens: COMPOSITION_REPLY_TEXT.length } }
yield { type: 'finish', reason: { kind: 'stop' } }
}
}
export const name = 'composition-echo-llm'
export const inject = ['llm']
/**
* Register the network-free adapter the shipped-composition smoke routes through.
* @param ctx - the loader-mounted plugin context.
*/
export function apply(ctx: Context): void {
ctx.llm.registerAdapter([COMPOSITION_PROVIDER], new CompositionEchoAdapter())
}
@@ -0,0 +1,52 @@
# Keyless tail for the shipped-composition smoke, applied as `--config` so the
# launcher boots `base.cordis.yml` + `tui.cordis.yml` and then this file.
#
# Everything below is test isolation, never composition under test: the model is
# replaced so no request leaves the process, the settle marker gates the smoke's
# first prompt, and the session artifacts move into the smoke's temporary
# workspace so the log inspection can read them.
# A patch's `name` is an assertion rather than a replacement, so the base
# adapter row is disabled and the scripted one inserted. Relative specifiers
# resolve against the INCLUDED file's directory (apps/cli/config), not this
# file's, because the include moves baseUrl there.
- id: llm-deepseek
disabled: true
- insert:
- id: composition-echo-llm
name: '../tests/fixtures/composition-echo-llm.ts'
- id: composition-settled
name: '../tests/fixtures/composition-settled.ts'
- id: agent-loop
config:
agents:
- id: main
provider: composition-keyless
model: composition-keyless-model
cwd: !!js process.cwd()
- id: session-persistence-jsonl
config:
root: './.sessions'
compression: none
- id: session-query-sqlite
config:
path: './.sessions/session-query.db'
# The title call is a second, tool-less request that would race the log
# inspection for no coverage: the catalog under test rides the agent turn.
- id: session-title-llm
disabled: true
- id: tui
config:
sessionId: !!js configuredAgentIdentities?.main?.id ?? 'main'
welcome: 'composition smoke ready.'
showReasoning: true
# HMR watches the repository; a PTY subprocess test must not start a watcher.
- id: hmr
disabled: true
+24
View File
@@ -0,0 +1,24 @@
import type { Context } from 'cordis'
/**
* Marker the shipped-composition smoke gates its first prompt on. The TUI renders as soon as
* its own fiber starts, so a prompt typed at the banner can reach the loop while
* later rows — tool plugins, persistence — are still activating, and would
* assemble a partial catalog. Waiting for this line makes the turn observe the
* settled tree.
*/
export const COMPOSITION_SETTLED_MARKER = 'COMPOSITION_TREE_SETTLED'
export const name = 'composition-settled'
/**
* Announce settled Loader activation on the terminal byte stream, after every
* entry in the booted tree has started. The write is detached: awaiting the
* Loader from inside an entry would wait on this entry's own activation.
* @param ctx - the loader-mounted plugin context.
*/
export function apply(ctx: Context): void {
void ctx.loader.await().then(() => {
process.stdout.write(`\n${COMPOSITION_SETTLED_MARKER}\n`)
})
}
+124
View File
@@ -0,0 +1,124 @@
import { readdir, readFile } from 'node:fs/promises'
import { fileURLToPath } from 'node:url'
import { join } from 'node:path'
import { describe, expect, it } from 'vitest'
import { LOADER_SMOKE_TEST_TIMEOUT_MS } from '@deepseek-ai/dsh-loader-smoke'
import type { SessionEvent } from '@deepseek-ai/dsh-session'
import { COMPOSITION_REPLY_TEXT } from './fixtures/composition-echo-llm.ts'
import { COMPOSITION_SETTLED_MARKER } from './fixtures/composition-settled.ts'
import { runTuiPtySmoke } from './pty-harness.ts'
const dshBinScript = fileURLToPath(new URL('../src/bin.ts', import.meta.url))
const tsconfigPath = fileURLToPath(new URL('../../../tsconfig.json', import.meta.url))
// An overlay over the shipped tree, so the catalog under test is the one
// `base.cordis.yml` + `tui.cordis.yml` assemble; the tail only swaps the model
// and redirects session artifacts.
const keylessTail = fileURLToPath(new URL('./fixtures/composition-keyless-tail.cordis.yml', import.meta.url))
/**
* The catalog the shipped `dsh` TUI puts in front of the model, as the loop
* logged it, minus the ripgrep-dependent pair below.
* The absences are the composition's security decisions, not incidental gaps:
* the `cordis_*` toolset executes model-written JavaScript that no sandbox row
* confines, `web_fetch` chooses its own request target, and `mcp_*` servers
* spawn outside `ctx.bash`. The composition Agent Note owns the rationale and
* its sources.
*/
const EXPECTED_TUI_TOOLS = [
'ask_user_question',
'bash',
'create_goal',
'edit',
'exit_plan_mode',
'get_goal',
'ralph',
'read',
'session_event_read',
'session_event_search',
'session_event_trace',
'session_search',
'session_trace',
'skill',
'str_replace_editor',
'subagent',
'subagent_fork',
'task_kill',
'task_list',
'task_output',
'todo_write',
'update_goal',
'web_search',
'workflow',
'write',
]
/**
* `glob` and `grep` come from `dsh-tool-fs-search`, which probes `command -v rg`
* through the mounted bash executor at load and registers neither tool when
* ripgrep is absent. That is a host dependency, not a composition decision, so the
* pair is asserted separately — present together or absent together.
*/
const RIPGREP_TOOLS = ['glob', 'grep']
/** The assembled request header the smoke asserts on. */
interface LoggedHeader {
/** Assembled tool names, sorted. */
names: string[]
/** `bash`'s assembled parameter properties; the escalation pair is present only under a confining executor. */
bashArguments: Record<string, unknown>
}
/**
* Read the request header the loop assembled for its first request from the
* session log the smoke's workspace persisted — the model-visible composition
* itself, not a registry projection taken beside it.
* @param cwd - the smoke's temporary workspace.
* @returns the assembled catalog, system prompt, and `bash` argument shape.
*/
async function loggedHeader(cwd: string): Promise<LoggedHeader> {
const sessionsDir = join(cwd, '.sessions')
const entries = await readdir(sessionsDir, { recursive: true })
// A single keyless run writes one session log.
const logRelPath = entries.find(name => name.endsWith('.jsonl'))
if (logRelPath === undefined) throw new Error(`no session log written under ${sessionsDir}`)
const lines = (await readFile(join(sessionsDir, logRelPath), 'utf8')).split('\n').filter(Boolean)
for (const line of lines) {
const event = JSON.parse(line) as SessionEvent
if (event.type !== 'request/header') continue
const tools = event.data.header.tools ?? []
const bash = tools.find(schema => schema.name === 'bash')
return {
names: tools.map(schema => schema.name).sort(),
bashArguments: (bash?.parameters as { properties?: Record<string, unknown> } | undefined)?.properties ?? {},
}
}
throw new Error(`session log ${logRelPath} has no request/header event`)
}
describe('shipped dsh composition (real Loader tree in a PTY)', () => {
it('assembles exactly the shipped TUI catalog', async () => {
let observed: LoggedHeader | undefined
const output = await runTuiPtySmoke({
label: 'dsh shipped composition',
tempDirPrefix: 'dsh-shipped-tui-',
binScript: dshBinScript,
tsconfigPath,
configPath: keylessTail,
env: { DEEPSEEK_API_KEY: 'keyless-composition-no-call', DSH_TELEMETRY_DISABLED: '1' },
// Artifact CI builds and smokes concurrently on a contended runner.
...(process.env.DSH_EXAMPLE_MODE === 'lib' ? { timeoutMs: 60_000 } : {}),
actions: [
{ waitFor: COMPOSITION_SETTLED_MARKER, send: 'Describe the shipped composition.\r' },
{ waitFor: COMPOSITION_REPLY_TEXT, send: '/exit\r' },
],
inspect: async (cwd) => { observed = await loggedHeader(cwd) },
})
expect(output).toContain(COMPOSITION_REPLY_TEXT)
expect(observed?.names.filter(name => !RIPGREP_TOOLS.includes(name))).toEqual(EXPECTED_TUI_TOOLS)
expect([[], RIPGREP_TOOLS]).toContainEqual(observed?.names.filter(name => RIPGREP_TOOLS.includes(name)))
// The TUI mounts the unrestricted local executors, so `tool-bash` emits no
// escalation pair. Pinning its absence keeps a later sandbox change from
// arriving here unannounced.
expect(Object.keys(observed?.bashArguments ?? {})).not.toContain('sandbox_permissions')
}, LOADER_SMOKE_TEST_TIMEOUT_MS)
})
@@ -0,0 +1,128 @@
terminal 100x36 buffer=normal length=66 base=30 viewport=30
lifecycle started=1 stopped=0 progress=inactive
title "Reply with exactly the word: — DSH TUI snapshot"
cursor hidden column=7 viewportRow=35 bufferRow=65
buffer
0| " DEEPSEEK HARNESS"
style 1-8 fg=bright-magenta bold
style 10-16 bold
1| " Reply with exactly the word:"
style 1-28 dim
2| " main-session"
style 1-12 dim
3| <blank>
4| "Context · snapshot-seed"
style 0-22 dim
5| "Older snapshot context. Older snapshot context. Older snapshot context. Older snapshot context. "
style 0-99 dim
6| "Older snapshot context. Older snapshot context. Older snapshot context. Older snapshot context. "
style 0-99 dim
7| "Older snapshot context. Older snapshot context. Older snapshot context. Older snapshot context. "
style 0-99 dim
8| "Older snapshot context. Older snapshot context. Older snapshot context. Older snapshot context. "
style 0-99 dim
9| "Older snapshot context. Older snapshot context. Older snapshot context. Older snapshot context. "
style 0-99 dim
10| "Older snapshot context. Older snapshot context. Older snapshot context. Older snapshot context. "
style 0-99 dim
11| "Older snapshot context. Older snapshot context. Older snapshot context. Older snapshot context. "
style 0-99 dim
12| "Older snapshot context. Older snapshot context. Older snapshot context. Older snapshot context. "
style 0-99 dim
13| "Older snapshot context. Older snapshot context. Older snapshot context. Older snapshot context. "
style 0-99 dim
14| "Older snapshot context. Older snapshot context. Older snapshot context. Older snapshot context. "
style 0-99 dim
15| "Older snapshot context. Older snapshot context. Older snapshot context. Older snapshot context. "
style 0-99 dim
16| "Older snapshot context. Older snapshot context. Older snapshot context. Older snapshot context. "
style 0-99 dim
17| "Older snapshot context. Older snapshot context. Older snapshot context. Older snapshot context. "
style 0-99 dim
18| "Older snapshot context. Older snapshot context. Older snapshot context. Older snapshot context. "
style 0-99 dim
19| "Older snapshot context. Older snapshot context. Older snapshot context. Older snapshot context. "
style 0-94 dim
20| <blank>
21| "You "
style 0-2 fg=bright-magenta bold underline
22| "Reply with exactly the word: ONE. No tools. "
23| <blank>
24| "Assistant "
style 0-8 fg=bright-magenta bold underline
25| "Reasoning "
style 0-8 dim italic
26| "The user wants me to reply with exactly the word \"ONE\" and use no tools. "
style 0-71 dim italic
27| "ONE "
28| "Model wait 0.0s · Completed 2026-07-21 12:00:00 "
style 0-46 dim
29| <blank>
30| "Keyboard shortcuts "
style 0-17 fg=bright-magenta bold
31| "Enter send • Shift/Alt+Enter newline • Up/Down prompt history "
style 0-60 dim
32| "Esc cancel turn • Ctrl+O cycle cards (collapse/expand/hide) • Ctrl+R toggle reasoning • Ctrl+L "
style 0-99 dim
33| "redraw "
style 0-5 dim
34| "Ctrl+C cancel while running; clear input or exit while idle • Ctrl+D exit "
style 0-72 dim
35| " "
36| "/clear — Clear the transcript view (session history is unchanged) "
style 0-64 dim
37| "/compact — Compact older conversation history "
style 0-44 dim
38| "/exit — Exit after the active turn reaches idle "
style 0-46 dim
39| "/help — Show keyboard shortcuts and commands "
style 0-43 dim
40| "/model [[provider/]model] — Show or switch this session's model "
style 0-62 dim
41| "/palette — Show every color and attribute role this terminal renders "
style 0-67 dim
42| "/quit — Exit after the active turn reaches idle "
style 0-46 dim
43| "/reload — EXPERIMENTAL (dev): re-read loader config files and apply the diff (idle only) "
style 0-87 dim
44| "/resume — List this workspace's resumable sessions "
style 0-49 dim
45| "/status — Show session diagnostics, system prompt, and registered tools "
style 0-70 dim
46| "/skill:<name> [instructions] — load a skill into the conversation "
style 0-64 dim
47| <blank>
48| "Context · snapshot-injector"
style 0-26 dim
49| "Injected while compaction was running. "
style 0-37 dim
50| <blank>
51| "… earlier context was compacted … "
style 0-32 dim
52| <blank>
53| "You "
style 0-2 fg=bright-magenta bold underline
54| "Reply with exactly the word: TWO. No tools. "
55| <blank>
56| "Compacted 2 history items (~387 tokens). "
style 0-39 dim
57| <blank>
58| "Assistant "
style 0-8 fg=bright-magenta bold underline
59| "Reasoning "
style 0-8 dim italic
60| "The user wants me to reply with exactly the word \"TWO\" and no tools. "
style 0-67 dim italic
61| "TWO "
62| "Model wait 0.0s · Completed 2026-07-21 12:00:00 "
style 0-46 dim
63| <blank>
64| "/workspace/project deepseek-v4-flash ↑2.9k ↓41 cache 49% 3% cont"
style 0-49 fg=bright-magenta bold
style 52-68 dim
style 71-90 dim
style 93-99 dim
65| " dsh ◍ "
style 1-3 fg=bright-magenta bold
style 5-6 dim
style 7-7 inverse
+282 -9
View File
@@ -11,8 +11,12 @@ import { LocalBashExecutor } from '@deepseek-ai/dsh-bash-local'
import LocalSubprocessService from '@deepseek-ai/dsh-subprocess-local'
import WorkerCodeRuntime from '@deepseek-ai/dsh-code-runtime-worker'
import CommandService from '@deepseek-ai/dsh-commands'
import * as CommandCompact from '@deepseek-ai/dsh-command-compact'
import { BasicCompactService } from '@deepseek-ai/dsh-compact-basic'
import type { SummarizationInput } from '@deepseek-ai/dsh-compact-basic/src/summarizer.ts'
import LocalFileSystem from '@deepseek-ai/dsh-fs-local'
import * as FsPolicy from '@deepseek-ai/dsh-fs-policy'
import { createUserMessage } from '@deepseek-ai/dsh-llm'
import * as ToolFs from '@deepseek-ai/dsh-tool-fs'
import * as LlmDeepSeek from '@deepseek-ai/dsh-llm-deepseek'
import { installLlmReplay, parseSessionLog } from '@deepseek-ai/dsh-llm-replay'
@@ -45,6 +49,8 @@ type ScenarioInteraction = 'skill-invocation-policy'
interface Scenario {
name: string
/** Replay fixture owned by an earlier scenario, for a derived presentation case. */
fixture?: string
composition: Composition
expectedTools: string[]
expectedEventCounts?: Record<string, number>
@@ -68,6 +74,13 @@ interface Scenario {
spillMaxInlineBytes?: number
/** Run scenario-specific terminal input instead of replaying recorded user prompts. */
interaction?: ScenarioInteraction
/**
* Mount a deterministic compaction backend plus `/compact`, then run the
* human command with a held summary while a prompt and injected context
* arrive. Proves queued input waits for the standalone bracket's durability
* checkpoint instead of racing the replacement.
*/
manualCompact?: boolean
}
const SCENARIOS: Scenario[] = [
@@ -80,6 +93,14 @@ const SCENARIOS: Scenario[] = [
leavePlanModeAfterFirstTurn: true,
recorded: true,
},
{
name: 'queued-manual-compact',
fixture: 'multi-turn-conversation',
composition: 'native',
expectedTools: [],
recorded: false,
manualCompact: true,
},
{
name: 'todo-plan',
composition: 'native',
@@ -149,6 +170,44 @@ function snapshotModeFromEnv(value: string | undefined): SnapshotMode {
const MODE = snapshotModeFromEnv(process.env.DSH_SNAPSHOT)
const observedScenarios = new Set<string>()
const workerState = Reflect.get(globalThis, '__vitest_worker__') as
| { readonly config?: { readonly testNamePattern?: RegExp } }
| undefined
// Worker argv omits the parent CLI's `-t`; the serialized runner config is the
// authoritative distinction between a focused replay and the full suite.
const TEST_NAME_FILTERED = workerState?.config?.testNamePattern !== undefined
/**
* Deterministic keyless summary that pauses so the scenario can submit a real
* prompt and inject context while manual compaction holds turn admission.
*/
class DeferredSnapshotCompactService extends BasicCompactService {
readonly summaryStarted = Promise.withResolvers<undefined>()
readonly releaseSummary = Promise.withResolvers<undefined>()
override async summarize(
_input: SummarizationInput,
_agent: Agent,
signal?: AbortSignal,
): Promise<{ summary: [{ type: 'text'; text: string }]; provider: string; model: string }> {
this.summaryStarted.resolve(undefined)
await this.releaseSummary.promise
signal?.throwIfAborted()
return {
summary: [{ type: 'text', text: 'Keyless manual compaction checkpoint.' }],
provider: 'snapshot',
model: 'snapshot-compactor',
}
}
}
/** Seed between-turn model-visible history without inventing a loop execution. */
function seedCompactableHistory(agent: Agent): void {
agent.inject(createUserMessage({
content: [{ type: 'text', text: 'Older snapshot context. '.repeat(60) }],
source: { kind: 'plugin', plugin: 'snapshot-seed' },
}))
}
function snapshotDisplayPath(displayPath: string, cwd: string, displayCwd: string): string {
const rel = relative(cwd, displayPath)
@@ -161,10 +220,15 @@ function scenarioDir(scenario: Scenario): string {
return join(SNAPSHOTS_DIR, scenario.name)
}
/** Directory owning the replay fixture: the scenario's own, or the one it derives from. */
function fixtureDir(scenario: Scenario): string {
return join(SNAPSHOTS_DIR, scenario.fixture ?? scenario.name)
}
function childFixturePaths(scenario: Scenario): string[] {
return Array.from(
{ length: scenario.childSessions ?? 0 },
(_, index) => join(scenarioDir(scenario), `session.${index + 1}.jsonl`),
(_, index) => join(fixtureDir(scenario), `session.${index + 1}.jsonl`),
)
}
@@ -206,6 +270,24 @@ async function settleTerminal(terminal: HeadlessTerminal): Promise<void> {
if (stable < 3) throw new Error('TUI frames did not quiesce within 200ms')
}
/** Bound deterministic in-process coordination waits with actionable state. */
async function snapshotDeadline<T>(
operation: Promise<T>,
detail: () => string,
): Promise<T> {
let timer: ReturnType<typeof setTimeout> | undefined
try {
return await Promise.race([
operation,
new Promise<never>((_resolve, reject) => {
timer = setTimeout(() => { reject(new Error(detail())) }, 5_000)
}),
])
} finally {
if (timer !== undefined) clearTimeout(timer)
}
}
async function mountScenarioContext(
scenario: Scenario,
cwd: string,
@@ -232,6 +314,9 @@ async function mountScenarioContext(
skills: { local: { agentsHome: join(cwd, '.agents') } },
})
await ctx.plugin(TokenMeterService)
if (scenario.manualCompact === true) {
await ctx.plugin(DeferredSnapshotCompactService, { auto: false })
}
await ctx.plugin(LocalSubprocessService)
await ctx.plugin(LocalBashExecutor, { cwd, timeoutMs: 30_000 })
await ctx.plugin(SnapshotLocalFileSystem, { cwd: '/' })
@@ -249,6 +334,7 @@ async function mountScenarioContext(
await ctx.plugin(ToolWorkflow)
await ctx.plugin(ToolRalph)
await ctx.plugin(CommandService)
if (scenario.manualCompact === true) await ctx.plugin(CommandCompact)
if (scenario.enterPlanMode === true) {
await ctx.plugin(PlanModeService, { section: 'Snapshot plan mode instructions.' })
}
@@ -276,9 +362,9 @@ interface ScenarioResult {
}
async function runScenario(scenario: Scenario): Promise<ScenarioResult> {
const clock = vi.spyOn(Date, 'now').mockReturnValue(new Date(2026, 6, 21, 12, 0, 0).getTime())
const dir = scenarioDir(scenario)
const fixtureFile = join(dir, 'session.jsonl')
const snapshotTime = new Date(2026, 6, 21, 12, 0, 0).getTime()
const clock = vi.spyOn(Date, 'now').mockReturnValue(snapshotTime)
const fixtureFile = join(fixtureDir(scenario), 'session.jsonl')
const childFiles = childFixturePaths(scenario)
const prompts = userPrompts(await readFile(fixtureFile, 'utf8'))
if (scenario.interaction === undefined) {
@@ -292,7 +378,7 @@ async function runScenario(scenario: Scenario): Promise<ScenarioResult> {
const terminal = new HeadlessTerminal(100, 36)
try {
if (scenario.seedWorkspace === true) {
const source = join(scenarioDir(scenario), 'workspace')
const source = join(fixtureDir(scenario), 'workspace')
await cp(source, cwd, { recursive: true })
}
ctx = await mountScenarioContext(scenario, cwd, displayCwd, fixtureFile, childFiles)
@@ -308,6 +394,7 @@ async function runScenario(scenario: Scenario): Promise<ScenarioResult> {
agentOptions: { provider: 'deepseek-official', model: 'deepseek-v4-flash' },
})
const agent: Agent = handle.agent
if (scenario.manualCompact === true) seedCompactableHistory(agent)
controller = createTuiChat(ctx, {
sessionId: 'main-session',
theme: { color: true },
@@ -380,6 +467,14 @@ async function runScenario(scenario: Scenario): Promise<ScenarioResult> {
}
let remainingPrompts = prompts
let queuedPrompt: string | undefined
let manualOrder: string[] | undefined
let manualCommandId: string | undefined
if (scenario.manualCompact === true) {
expect(prompts.length, 'queued manual compaction needs a second replayed prompt').toBeGreaterThanOrEqual(2)
queuedPrompt = prompts.at(-1)
remainingPrompts = prompts.slice(0, -1)
}
if (scenario.enterPlanMode === true) {
const firstPrompt = prompts[0]!
terminal.send(`/plan ${firstPrompt}`)
@@ -396,12 +491,93 @@ async function runScenario(scenario: Scenario): Promise<ScenarioResult> {
}
for (const prompt of remainingPrompts) {
const admitted = agent.session.events.filter(event =>
event.type === 'user/message' && event.data.source.kind === 'user').length
terminal.send(prompt)
terminal.send('\r')
await terminal.flush()
await expect.poll(() => agent.session.events.filter(event =>
event.type === 'user/message' && event.data.source.kind === 'user').length).toBe(admitted + 1)
await agent.whenIdle()
await settleTerminal(terminal)
}
if (scenario.manualCompact === true && queuedPrompt !== undefined) {
terminal.send('/help')
terminal.send('\r')
await settleTerminal(terminal)
expect(await terminal.snapshot({ includeScrollback: true }))
.toContain('/compact — Compact older conversation history')
const compact = ctx.compact as DeferredSnapshotCompactService
const inbox: string[] = []
manualOrder = []
ctx.on('agent/inbox/enqueue', (subject, item) => {
if (subject === agent) inbox.push(`enqueue:${item.placement}:${item.id}`)
})
ctx.on('agent/inbox/dequeue', (subject, message) => {
if (subject === agent) inbox.push(`dequeue:${message.id}`)
})
ctx.on('session/event', (session, event) => {
if (session !== agent.session) return
if (event.type === 'command/run' && event.data.name === 'compact') {
manualCommandId = event.data.commandId
manualOrder?.push('command/run')
}
if (event.type === 'command/done' && event.data.commandId === manualCommandId) {
manualOrder?.push('command/done')
}
if (event.type.startsWith('compact/')) manualOrder?.push(event.type)
if (event.type === 'user/message'
&& event.data.source.kind === 'plugin'
&& event.data.source.plugin === 'compact') manualOrder?.push('checkpoint')
if (event.type === 'turn/start') manualOrder?.push(`turn/start:${event.data.trigger.kind}`)
})
ctx.on('session/flush', (session) => {
if (session === agent.session) manualOrder?.push('flush')
})
terminal.send('/compact')
terminal.send('\r')
await terminal.flush()
await snapshotDeadline(compact.summaryStarted.promise, () =>
`manual summary did not start; status=${agent.status}; tail=${
agent.session.events.slice(-8).map(event => event.type).join(',')
}`)
clock.mockReturnValue(snapshotTime + 1_000)
await settleTerminal(terminal)
await expect.poll(() => terminal.snapshot()).toContain('dsh ⊙')
await expect.poll(() => terminal.snapshot()).toContain('Context being compacted 1.0s')
const liveCompaction = await terminal.snapshot()
expect(liveCompaction.indexOf('Context being compacted 1.0s')).toBeLessThan(liveCompaction.indexOf('dsh ⊙'))
clock.mockReturnValue(snapshotTime)
// Real keystrokes: the prompt keeps its ordinary queue identity while
// admission is reserved, and an injection appends immediately.
terminal.send(queuedPrompt)
terminal.send('\r')
await terminal.flush()
await expect.poll(() => inbox.length).toBe(1)
agent.inject(createUserMessage({
content: [{ type: 'text', text: 'Injected while compaction was running.' }],
source: { kind: 'plugin', plugin: 'snapshot-injector' },
}))
expect(inbox[0]).toMatch(/^enqueue:queued:/u)
expect(agent.status).toBe('idle')
expect(agent.session.events.some(event => event.type === 'user/message'
&& event.data.source.kind === 'user'
&& event.data.content.some(block => block.type === 'text' && block.text === queuedPrompt))).toBe(false)
const idle = agent.whenIdle()
compact.releaseSummary.resolve(undefined)
await snapshotDeadline(idle, () =>
`manual compaction did not reach idle; status=${agent.status}; order=${manualOrder?.join(',') ?? ''}; tail=${
agent.session.events.slice(-12).map(event => event.type).join(',')
}`)
await settleTerminal(terminal)
expect(inbox).toEqual([inbox[0], `dequeue:${inbox[0]?.slice('enqueue:queued:'.length) ?? ''}`])
}
const events: SessionEvent[] = [...agent.session.events]
const firstHeader = events.find(event => event.type === 'request/header')
expect(firstHeader?.type === 'request/header' && firstHeader.data.header.system)
@@ -437,6 +613,87 @@ async function runScenario(scenario: Scenario): Promise<ScenarioResult> {
expect(events.filter(event => event.type === 'user/message' && event.data.source.kind === 'plugin').map(event => (event.data as { content: unknown }).content))
.toContainEqual([{ type: 'text', text: 'The user switched this session back to the default mode.' }])
}
if (scenario.manualCompact === true) {
const compactStart = events.find(event => event.type === 'compact/start')
const compactSummary = events.find(event => event.type === 'compact/summary')
const compactCheckpoint = events.find(event => event.type === 'user/message'
&& event.data.source.kind === 'plugin' && event.data.source.plugin === 'compact')
const injectedEvent = events.find(event => event.type === 'user/message'
&& event.data.source.kind === 'plugin' && event.data.source.plugin === 'snapshot-injector')
const compactEnd = events.find(event => event.type === 'compact/end')
expect(compactStart?.data.turn).toBeNull()
expect(compactEnd?.data.turn).toBeNull()
expect(events.filter(event => event.type === 'compact/summary')).toHaveLength(1)
if (compactStart === undefined || compactSummary === undefined
|| compactCheckpoint === undefined || injectedEvent === undefined
|| compactEnd === undefined) {
throw new Error('manual compaction snapshot is missing its durable marker, summary, checkpoint, or injection')
}
// The markers are time points, not an exclusive container: unrelated
// idle injection is allowed between them while the selected span stays stable.
expect(compactStart.seq).toBeLessThan(injectedEvent.seq)
expect(injectedEvent.seq).toBeLessThan(compactSummary.seq)
expect(compactSummary.seq).toBeLessThan(compactCheckpoint.seq)
expect(compactCheckpoint.seq).toBeLessThan(compactEnd.seq)
const manualTimeline = manualOrder ?? []
const commandRunIndex = manualTimeline.indexOf('command/run')
const compactStartIndex = manualTimeline.indexOf('compact/start')
const compactEndIndex = manualTimeline.indexOf('compact/end')
const firstFlushIndex = manualTimeline.indexOf('flush')
const queuedTurnIndex = manualTimeline.indexOf('turn/start:message')
const commandDoneIndex = manualTimeline.indexOf('command/done')
expect(manualTimeline.filter(item => item === 'command/run')).toHaveLength(1)
expect(manualTimeline.filter(item => item === 'command/done')).toHaveLength(1)
expect(compactStartIndex).toBeGreaterThan(commandRunIndex)
expect(compactEndIndex).toBeGreaterThan(compactStartIndex)
expect(firstFlushIndex).toBeGreaterThan(compactEndIndex)
expect(queuedTurnIndex).toBeGreaterThan(firstFlushIndex)
expect(commandDoneIndex).toBeGreaterThan(firstFlushIndex)
const commandRun = events.find(event => event.type === 'command/run'
&& event.data.name === 'compact')
const commandRunId = commandRun?.type === 'command/run'
? commandRun.data.commandId
: undefined
const commandDone = events.find(event => event.type === 'command/done'
&& event.data.commandId === commandRunId)
expect(commandRun?.type === 'command/run' && commandRun.data).toEqual({
commandId: commandRunId,
name: 'compact',
args: '',
source: { kind: 'user' },
})
expect(commandDone?.type === 'command/done' && commandDone.data).toEqual({
commandId: commandRunId,
kind: 'success',
text: 'Compacted 2 history items (~387 tokens).',
})
expect(commandRun !== undefined && commandRun.seq < compactStart.seq).toBe(true)
expect(commandDone !== undefined && commandDone.seq > compactEnd.seq).toBe(true)
expect(agent.session.surface.nodes).not.toContain(commandRun?.seq)
expect(agent.session.surface.nodes).not.toContain(commandDone?.seq)
// The command line itself never becomes a prompt.
expect(events.some(event => event.type === 'user/message'
&& event.data.source.kind === 'user'
&& event.data.content.some(block => block.type === 'text' && block.text.trim() === '/compact'))).toBe(false)
const derived = agent.session.deriveMessages().map(message => message.content
.map(block => block.type === 'text' ? block.text : '')
.join(''))
const checkpoint = derived.findIndex(text => text.includes('Keyless manual compaction checkpoint.'))
const injected = derived.findIndex(text => text.includes('Injected while compaction was running.'))
const queued = derived.findIndex(text => text === queuedPrompt)
expect(checkpoint).toBe(0)
expect(injected).toBeGreaterThan(checkpoint)
expect(queued).toBeGreaterThan(injected)
expect(derived).not.toContain('/compact')
expect(derived).not.toContain('Compacted 2 history items (~387 tokens).')
expect(derived.filter(text => text.includes('Injected while compaction was running.'))).toHaveLength(1)
expect(compactSummary.data.shadowedSeqs).not.toContain(injectedEvent.seq)
const queuedTurn = events.findLast(event => event.type === 'turn/start')
expect(queuedTurn !== undefined && compactEnd.seq < queuedTurn.seq).toBe(true)
}
if (scenario.spillMaxInlineBytes !== undefined) {
// The REAL pipeline ran (tools execute on replay too): the durable
// dispatch copy is bounded to a preview + locator under the run cwd,
@@ -514,7 +771,23 @@ describe('TUI recorded-session terminal snapshots', () => {
})
afterAll(async () => {
expect([...observedScenarios].sort()).toEqual(SCENARIOS.map(scenario => scenario.name).sort())
const scenarioNames = SCENARIOS.map(scenario => scenario.name).sort()
const observedNames = [...observedScenarios].sort()
if (TEST_NAME_FILTERED) {
expect(observedNames).not.toHaveLength(0)
expect(scenarioNames).toEqual(expect.arrayContaining(observedNames))
} else {
expect(observedNames).toEqual(scenarioNames)
}
for (const [index, scenario] of SCENARIOS.entries()) {
if (scenario.fixture === undefined) continue
const sourceIndex = SCENARIOS.findIndex(candidate => candidate.name === scenario.fixture)
expect(sourceIndex, `${scenario.name} fixture source ${scenario.fixture} must exist`).toBeGreaterThanOrEqual(0)
expect(sourceIndex, `${scenario.name} fixture source must precede it`).toBeLessThan(index)
const source = SCENARIOS[sourceIndex]
expect(source?.fixture, `${scenario.name} fixture source must own its replay files`).toBeUndefined()
expect(source?.recorded, `${scenario.name} fixture source must be recordable`).toBe(true)
}
const directories = (await readdir(SNAPSHOTS_DIR, { withFileTypes: true }))
.filter(entry => entry.isDirectory())
.map(entry => entry.name)
@@ -522,14 +795,14 @@ afterAll(async () => {
expect(directories).toEqual(SCENARIOS.map(scenario => scenario.name).sort())
for (const scenario of SCENARIOS) {
const expected = [
'session.jsonl',
...scenario.fixture === undefined ? ['session.jsonl'] : [],
'terminal.expected.txt',
...scenario.seedWorkspace === true ? ['workspace'] : [],
...scenario.seedWorkspace === true && scenario.fixture === undefined ? ['workspace'] : [],
...Array.from({ length: scenario.childSessions ?? 0 }, (_, index) => `session.${index + 1}.jsonl`),
].sort()
expect((await readdir(scenarioDir(scenario))).sort()).toEqual(expected)
for (const fixture of ['session.jsonl', ...childFixturePaths(scenario).map(path => basename(path))]) {
const content = await readFile(join(scenarioDir(scenario), fixture), 'utf8')
const content = await readFile(join(fixtureDir(scenario), fixture), 'utf8')
expect(scrubRequestHeaders(content), `${scenario.name}/${fixture} carries request-header bulk`).toBe(content)
}
}