From b73eb7663c00d29f87644a7eb7cb7f0dc5e016f6 Mon Sep 17 00:00:00 2001 From: _Kerman Date: Fri, 24 Jul 2026 21:18:48 +0800 Subject: [PATCH] refactor(agent-loop): simplify observable state machine --- ...2026-06-21-bounded-llm-request-recovery.md | 20 +- ...n-pressure-and-overflow-recovery.i18n.yaml | 4 +- ...mpaction-pressure-and-overflow-recovery.md | 8 +- ...ction-pressure-and-overflow-recovery.zh.md | 8 +- .../2026-06-18-compaction-capability-seam.md | 2 +- .../feature/2026-07-07-session-prefix.md | 2 +- ...-drop-unconsumed-llm-assembled-surfaces.md | 2 +- .../2026-07-02-remove-stream-chunk-mirror.md | 2 +- ...2026-07-04-remove-agent-steering-mirror.md | 4 +- ...nt-loop-observable-state-machine.i18n.yaml | 6 + ...-24-agent-loop-observable-state-machine.md | 58 ++ ...-agent-loop-observable-state-machine.zh.md | 58 ++ docs/agent-lifecycle.md | 13 +- docs/architecture.i18n.yaml | 4 +- docs/architecture.md | 22 +- docs/architecture.zh.md | 22 +- docs/config-catalog.md | 18 +- docs/cookbook/extension-cookbook.i18n.yaml | 4 +- docs/cookbook/extension-cookbook.md | 2 +- docs/cookbook/extension-cookbook.zh.md | 2 +- docs/cordis-catalog/events.md | 287 +++------ docs/cordis-catalog/services.md | 18 +- docs/core-data-structures/compaction.md | 2 +- docs/core-data-structures/core.i18n.yaml | 4 +- docs/core-data-structures/core.md | 116 ++-- docs/core-data-structures/core.zh.md | 116 ++-- docs/core-data-structures/goal.md | 2 + docs/core-data-structures/llm-streaming.md | 2 +- .../core-data-structures/session-reference.md | 8 +- docs/core-data-structures/session.i18n.yaml | 4 +- docs/core-data-structures/session.md | 12 +- docs/core-data-structures/session.zh.md | 12 +- docs/core-data-structures/tools.md | 17 +- docs/event-producer-consumer.md | 45 +- docs/persistence-catalog.md | 28 +- .../tests/fixtures/goal-domain/seed-goal.ts | 2 +- .../bash/tool-bash/tests/integration.spec.ts | 2 +- packages/compact/compact-basic/README.md | 10 +- packages/compact/compact-basic/src/index.ts | 39 +- packages/compact/compact-basic/src/types.ts | 2 +- .../compact-basic/tests/compact-basic.spec.ts | 76 +-- .../tests/compact-loop-repro.spec.ts | 79 ++- .../time-context/tests/time-context.spec.ts | 14 +- .../context/workspace-context/src/state.ts | 4 +- .../tests/workspace-context.spec.ts | 50 +- .../cordis/tool-cordis/src/api-catalog.ts | 23 +- packages/core/agent-loop/README.md | 28 +- packages/core/agent-loop/src/agent.ts | 130 ++-- packages/core/agent-loop/src/index.ts | 41 +- packages/core/agent-loop/tests/MIGRATION.md | 86 --- .../agent-loop/tests/agent-initiator.spec.ts | 40 +- packages/core/agent-loop/tests/agent.spec.ts | 318 ++------- packages/core/agent-loop/tests/cancel.spec.ts | 324 +--------- .../tests/config-session-id.spec.ts | 72 ++- .../tests/contract-regressions.spec.ts | 602 +++-------------- .../agent-loop/tests/coverage-edges.spec.ts | 28 - .../agent-loop/tests/inbox-invariant.spec.ts | 155 ----- .../agent-loop/tests/interception.spec.ts | 303 +-------- packages/core/agent-loop/tests/loop.spec.ts | 271 ++------ .../core/agent-loop/tests/properties.spec.ts | 4 +- .../agent-loop/tests/request-error.spec.ts | 149 +++++ .../core/agent-loop/tests/request-log.spec.ts | 82 --- .../tests/request-reconstruction.spec.ts | 17 +- .../agent-loop/tests/request-recovery.spec.ts | 605 ------------------ packages/core/agent-loop/tests/resume.spec.ts | 5 +- .../agent-loop/tests/scope-lifecycle.spec.ts | 69 +- .../core/agent-loop/tests/tool-calls.spec.ts | 14 +- .../core/agent-loop/tests/tool-order.spec.ts | 4 +- .../core/agent-loop/tests/turn-stop.spec.ts | 196 ------ packages/core/agent/README.md | 2 +- packages/core/agent/src/dispatch.ts | 14 +- packages/core/agent/src/types.ts | 64 +- packages/core/agent/tests/agent.spec.ts | 12 +- packages/core/agent/tests/invariant.spec.ts | 15 +- packages/core/agent/tests/llm-target.spec.ts | 8 +- .../core/scope/src/scoped-events.generated.ts | 1 + packages/core/scope/tests/invariant.spec.ts | 22 +- packages/core/session/src/types.ts | 10 +- .../examples/acp-demo/tests/acp-agent.spec.ts | 10 +- .../agent-spine-demo/tests/agent-core.spec.ts | 25 +- packages/examples/cli-demo/src/cli.ts | 26 +- .../examples/cli-demo/tests/cli-demo.spec.ts | 13 +- packages/examples/cli-demo/tests/cli.spec.ts | 16 +- .../command-goal/tests/command-goal.spec.ts | 1 + packages/goal/goal-session/src/index.ts | 35 +- .../goal-session/tests/goal-session.spec.ts | 68 +- packages/goal/goal/tests/goal.spec.ts | 6 +- .../goal/tool-goal/tests/tool-goal.spec.ts | 9 +- packages/hooks/hooks-claude/src/index.ts | 10 +- .../hooks/hooks-claude/tests/bridge.spec.ts | 15 +- .../hooks-claude/tests/coverage-cases.ts | 10 +- packages/hooks/hooks-codex/src/index.ts | 10 +- .../hooks/hooks-codex/tests/bridge.spec.ts | 12 +- .../hooks/hooks-codex/tests/coverage-cases.ts | 12 +- .../host/runtime/tests/host-runtime.spec.ts | 14 +- packages/llm/llm-retry/README.md | 12 +- packages/llm/llm-retry/src/index.ts | 33 +- packages/llm/llm-retry/src/invariant.ts | 39 +- .../llm/llm-retry/tests/invariant.spec.ts | 83 ++- .../tests/loader-composition.spec.ts | 16 +- .../llm/llm-retry/tests/persistence.spec.ts | 9 +- packages/llm/llm-retry/tests/retry.spec.ts | 199 ++---- packages/plan/plan-mode/src/index.ts | 20 +- .../plan/plan-mode/tests/integration.spec.ts | 22 +- .../plan/plan-mode/tests/plan-mode.spec.ts | 107 +--- packages/pty/pty-local/tests/index.spec.ts | 6 +- packages/pty/pty-local/tests/local.spec.ts | 2 +- packages/pty/pty/tests/service.spec.ts | 2 +- .../tool-pty/tests/loader-composition.spec.ts | 2 +- packages/pty/tool-pty/tests/tools.spec.ts | 2 +- .../src/package-managers/link-workspace.ts | 1 + packages/sdk/helper/tests/documents.spec.ts | 2 +- .../session-checkpoint-policy/README.md | 10 +- .../tests/session-checkpoint-policy.spec.ts | 6 +- .../session-persistence/src/coordinator.ts | 14 +- .../skill/tool-skill/tests/tool-skill.spec.ts | 47 +- .../subagent-fork/tests/subagent-fork.spec.ts | 2 +- .../tests/structured.spec.ts | 61 +- .../tests/subagent-inprocess.spec.ts | 2 +- .../tests/subagent-spawn.spec.ts | 23 +- packages/tasks/tasks/tests/tasks.spec.ts | 2 +- packages/ui/acp/src/index.ts | 7 +- packages/ui/acp/tests/config-options.spec.ts | 9 +- packages/ui/acp/tests/dispose.spec.ts | 28 +- packages/ui/acp/tests/turns.spec.ts | 2 +- packages/ui/tui/tests/harness.ts | 16 +- packages/ui/tui/tests/tui.spec.ts | 32 +- .../ui/user-approval/tests/approval.spec.ts | 2 +- scripts/gen-cordis-catalog.ts | 1 - scripts/gen-doc-graphs.ts | 66 +- scripts/type-equiv.manifest.json | 198 +++++- 131 files changed, 2011 insertions(+), 4292 deletions(-) create mode 100644 .agents/notes/implemented/simplification/2026-07-24-agent-loop-observable-state-machine.i18n.yaml create mode 100644 .agents/notes/implemented/simplification/2026-07-24-agent-loop-observable-state-machine.md create mode 100644 .agents/notes/implemented/simplification/2026-07-24-agent-loop-observable-state-machine.zh.md delete mode 100644 packages/core/agent-loop/tests/MIGRATION.md delete mode 100644 packages/core/agent-loop/tests/inbox-invariant.spec.ts create mode 100644 packages/core/agent-loop/tests/request-error.spec.ts delete mode 100644 packages/core/agent-loop/tests/request-log.spec.ts delete mode 100644 packages/core/agent-loop/tests/request-recovery.spec.ts delete mode 100644 packages/core/agent-loop/tests/turn-stop.spec.ts diff --git a/.agents/notes/implemented/architecture/2026-06-21-bounded-llm-request-recovery.md b/.agents/notes/implemented/architecture/2026-06-21-bounded-llm-request-recovery.md index 28de5eb97c..b8d06ebf6d 100644 --- a/.agents/notes/implemented/architecture/2026-06-21-bounded-llm-request-recovery.md +++ b/.agents/notes/implemented/architecture/2026-06-21-bounded-llm-request-recovery.md @@ -4,9 +4,9 @@ Status: implemented ## Problem -`dsh-llm` can report provider failures either by throwing during adapter dispatch or iteration or by ending with `finish { kind: 'error' | 'aborted' }`. The final adapter boundary tags thrown failures so `dsh-agent-loop` can distinguish them from middleware and result-processing defects, and the loop normalizes both delivery forms into `agent/request-error` after closing the failed step. The default decision is `fail`; `dsh-compact-basic` is the only shipped recovery listener, and it retries a canonical context-window overflow only after compaction proves that the durable surface shrank. +`dsh-llm` can report provider failures either by throwing during adapter dispatch or iteration or by ending with `finish { kind: 'error' | 'aborted' }`. The final adapter boundary tags thrown failures so `dsh-agent-loop` can distinguish them from middleware and result-processing defects, and the loop normalizes both delivery forms into `agent/request-error` after closing the failed step. An unhandled failure is terminal; a handling listener repairs policy-owned state, calls `agent.retry()`, and stops waterfall delegation. -That boundary is already safe for another request attempt. Raw `assistant/chunk` events carry the failed `turn` and `step`, message derivation ignores them unless a successful `assistant/message` cites them, tool calls are dispatched only after a successful terminal finish and assembly, and a retry opens a new numbered step from the durable log. The harness therefore does not need a second response lifecycle or tentative-output protocol to keep two attempts separate. +That boundary is already safe for another request attempt. Raw `assistant/chunk` events carry the failed `turn` and `step`, message derivation ignores them unless a successful `assistant/message` cites them, tool calls are dispatched only after a successful terminal finish and assembly, and a retry opens a new numbered turn from the durable log. The harness therefore does not need a second response lifecycle or tentative-output protocol to keep two attempts separate. The prior boundary left three narrower gaps. @@ -48,7 +48,7 @@ The initial shared transient-code set is intentionally small: the adapters' exis `@deepseek-ai/dsh-llm-retry` is a function plugin that listens to `agent/request-error`. It introduces no service or new loop branch; the agent-loop package changes only the data carried through its existing failed-step recovery control flow. -The `agent/request-error` seam carries the current `LlmFailure` and an immutable list of prior failures that led to another request attempt in this consecutive recovery sequence. `dsh-llm-retry` counts only prior failures whose codes are in its configured transient set, while `dsh-compact-basic` counts only prior context-overflow failures. A successful model request clears the history. Alternating transient and context-overflow failures therefore consume their owning policy budgets independently; the maximum request count is one plus the sum of the finite budgets of the loaded recovery policies. +The `agent/request-error` seam carries only the current `LlmFailure`; the loop owns no retry policy or attempt history. Each recovery plugin keeps a private per-agent counter for its own handled failures and clears it at terminal `agent/idle`. Alternating transient and context-overflow failures therefore consume the `dsh-llm-retry` and compact-basic budgets independently; the maximum request count is one plus the sum of the finite budgets of the loaded recovery policies. The plugin resolves and validates this deployment configuration at load: @@ -66,13 +66,13 @@ The defaults are two transient retries, a 500 millisecond initial delay, a 10 se For an eligible failure with budget remaining, the one-based transient retry count uses bounded exponential backoff. A valid `providerRetryAfterMs` replaces exponential backoff only when it does not exceed `maxDelayMs`; a longer provider delay causes delegation instead of an earlier retry that violates the provider instruction. Local backoff multiplies by an injected random factor in `[1 - jitterRatio, 1 + jitterRatio]` and clamps the final value to `maxDelayMs`; provider delay is not jittered. -The plugin owns a lifetime `AbortController` and tracks every active backoff callback. Each wait fuses the waterfall's turn signal with that lifetime signal. Effect cleanup first unregisters the listener, then aborts and awaits the active callbacks; a captured callback whose lifetime signal aborts returns `fail` and can neither retry nor enter the rest of its captured waterfall after disposal. This makes HMR disposal quiescent even though Cordis has already captured the listener. +The plugin owns a lifetime `AbortController` and tracks every active backoff callback. Each wait fuses the waterfall's turn signal with that lifetime signal. Effect cleanup first unregisters the listener, then aborts and awaits the active callbacks; a captured callback whose lifetime signal aborts returns without retrying or entering the rest of its captured waterfall. This makes HMR disposal quiescent even though Cordis has already captured the listener. Before sleeping, `dsh-llm-retry` appends one non-surface `llm/retry` session event containing the turn, failed step, one-based transient retry number, configured maximum, scheduled delay, and `LlmFailure`. The plugin owns the `SessionEventMap` augmentation; `dsh-session` remains generic persistence and does not absorb the optional policy's vocabulary. The event says what was scheduled, not that the next request completed; cancellation during the delay is subsequently visible on `turn/end`. The event ships only with a production renderer and replay/snapshot coverage, because its purpose is operational state rather than trace collection. -The listener calls `next()` for a non-transient code, an exhausted policy budget, or an over-cap provider delay. This preserves composition with context-overflow recovery and later policy plugins. It returns `{ action: 'retry' }` only after the delay completes under both signals; turn cancellation and plugin disposal return `fail`, after which the loop's cancellation/disposal checks remain authoritative. +The listener calls `next()` for a non-transient code, an exhausted policy budget, or an over-cap provider delay. This preserves composition with context-overflow recovery and later policy plugins. For an owned failure it records and awaits the delay, then calls `agent.retry()` without delegating. Turn cancellation and plugin disposal end the wait without requesting a retry; the loop's cancellation/disposal checks remain authoritative. -The agent-spine demo bundle loads the plugin so the shared stdio/TUI, one-shot CLI, and ACP example compositions use the same bounded policy. Library consumers retain explicit plugin composition: omitting the plugin leaves `agent/request-error` at its current fail default. +The agent-spine demo bundle loads the plugin so the shared stdio/TUI, one-shot CLI, and ACP example compositions use the same bounded policy. Library consumers retain explicit plugin composition: omitting the plugin leaves request failures terminal. ### Make one layer own visible attempts @@ -90,7 +90,7 @@ Boundary tests prove termination at both actual transports. The hand-written ada ### Keep attempts separate in the existing log -A failed attempt may leave `assistant/chunk` events in its closed step, but it never appends `assistant/message` and never dispatches a tool. A retry opens the next numbered step, reconstructs the request from the durable surface, and produces its own chunks. UIs may render live chunks while a step is open, then mark or clear that transient view when `llm/retry` identifies the failed step or `turn/end` records terminal failure; message derivation continues to ignore the failed chunks. +A failed attempt may leave `assistant/chunk` events in its closed step, but it never appends `assistant/message` and never dispatches a tool. A retry closes the failed turn, opens the next numbered turn, reconstructs the request from the durable surface, and produces its own chunks. UIs may render live chunks while a step is open, then mark or clear that transient view when `llm/retry` identifies the failed step or `turn/end` records failure; message derivation continues to ignore the failed chunks. If recovery is exhausted, the final failure is stored once on `turn/end.reason` with the structured facts. If transient recovery continues, `llm/retry` is the durable home for that attempt's failure and delay. No standalone final-error event or response-id vocabulary is added. @@ -118,9 +118,9 @@ If recovery is exhausted, the final failure is stored once on `turn/end.reason` - An adapter-thrown `Error` reaches `agent/request-error` as the exact same object while its sidecar `LlmFailure` reaches the adjacent argument; tests retain the existing identity assertion for extensible and frozen third-party errors. - DeepSeek and pi-ai adapter tests cover representative 400, 401/403, 429, 5xx, connection, malformed/truncated stream, timeout, abort, retry-after seconds/date, request-id, and unknown-SDK-error paths without recovery policy parsing message text. - Pi-ai pins the SDK option to zero retries and performs one observed wire attempt for a retryable provider response; separate tests make removing either boundary fail. -- `agent/request-error` carries current failure facts plus immutable prior-retried failure facts; a success clears that history, and alternating transient/context-overflow integration tests prove the two policies consume only their own finite budgets. +- `agent/request-error` carries only current failure facts; each plugin clears its private per-agent counter at terminal idle, and alternating transient/context-overflow integration tests prove the two policies consume only their own finite budgets. - `dsh-llm-retry` validates every config field at Loader startup, delegates all ineligible paths with `next()`, and makes at most `maxTransientRetries + 1` provider requests when no other policy applies. -- HMR-during-backoff tests prove disposal unregisters the listener, aborts and awaits its captured callbacks, emits no retry decision after disposal, and leaves no timer or promise alive. +- HMR-during-backoff tests prove disposal unregisters the listener, aborts and awaits its captured callbacks, makes no retry request after disposal, and leaves no timer or promise alive. - Pure unit tests cover transient-code selection, exponential backoff and jitter bounds, valid and over-cap `Retry-After`, exhausted budgets, deterministic timer/random seams, and abort during backoff. - Real agent-loop tests cover failure before chunks, partial chunks then failure, thrown and in-band failures, retry to success in a new step, exhaustion to structured `turn/end.reason`, and composition with `dsh-compact-basic` context-overflow recovery. - The partial-chunk integration test proves failed chunks remain attributed to the failed step, no assistant message or tool side effect is committed for that step, and the successful retry has distinct provenance. @@ -130,7 +130,7 @@ If recovery is exhausted, the final failure is stored once on `turn/end.reason` ## Consequences -- Every transient recovery attempt is visible as a closed step plus `llm/retry`, and the bounded policy prevents hidden SDK retries from multiplying cost. A retry can still duplicate provider billing even when no chunk arrived; the finite attempt budget limits but cannot remove that risk. +- Every transient recovery attempt is visible as a closed failed turn plus `llm/retry`, and the bounded policy prevents hidden SDK retries from multiplying cost. A retry can still duplicate provider billing even when no chunk arrived; the finite attempt budget limits but cannot remove that risk. - Provider SDKs may hide status or retry headers. Those adapters retain the stable facts they expose and otherwise use a coarse code rather than letting recovery policy parse fragile text. - Durable retry events expand the session protocol and UI state machine. Shipping the event and its consumer together prevents an unused telemetry vocabulary, but later schema changes still require persistence and replay work. - Clearing a failed step's live chunks can visibly retract output. That is preferable to presenting discarded text or partial tool JSON as committed history, and snapshots pin the transition. diff --git a/.agents/notes/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.i18n.yaml b/.agents/notes/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.i18n.yaml index b25f335819..fdd7444512 100644 --- a/.agents/notes/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.i18n.yaml +++ b/.agents/notes/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write -2026-07-10-after-call-compaction-pressure-and-overflow-recovery.md: b934f7fd7087006be4f7eb3659e44e78b8ede367 -2026-07-10-after-call-compaction-pressure-and-overflow-recovery.zh.md: 3b5b60a95bef0695a446cdd3d45d299550f449f6 +2026-07-10-after-call-compaction-pressure-and-overflow-recovery.md: 83fe62276b235d5961ed0fee61a4a1603c383c56 +2026-07-10-after-call-compaction-pressure-and-overflow-recovery.zh.md: fef9f48cbcfda42a13a88737f10b446ba317d5de diff --git a/.agents/notes/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.md b/.agents/notes/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.md index b934f7fd70..83fe62276b 100644 --- a/.agents/notes/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.md +++ b/.agents/notes/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.md @@ -22,9 +22,9 @@ The loop fires awaited serial `agent/post-step(agent, turn, step, signal)` after ### Request recovery is limited to the final model boundary -`RequestError`, `RequestErrorDecision`, and the `agent/request-error` waterfall represent failures after the final adapter has been selected. Each returned stream handle owns a private failure set that preserves the original thrown error identity across dispatch, iterator construction, and iteration without leaking nested-call provenance into an outer call. Terminal in-band `error` or `aborted` finishes enter the same path. Prompt assembly, request middleware, request logging, result processing, tools, post-step listeners, and cleanup remain ordinary failures. +`RequestError` and the `agent/request-error` waterfall represent failures after the final adapter has been selected. Each returned stream handle owns a private failure set that preserves the original thrown error identity across dispatch, iterator construction, and iteration without leaking nested-call provenance into an outer call. Terminal in-band `error` or `aborted` finishes enter the same path. Prompt assembly, request middleware, request logging, result processing, tools, step listeners, and cleanup remain ordinary failures. -The failed step closes before recovery runs. A retry opens the next numbered step and rebuilds the request from the durable log; consecutive recovery attempts reset only after a successful provider request. Both DeepSeek adapters normalize recognized provider context-limit failures to `CONTEXT_WINDOW_EXCEEDED`. +The failed step closes before recovery runs. A handling listener repairs durable state, calls `agent.retry()`, and stops waterfall delegation. The loop then closes the failed turn and opens one retry turn from the durable log without an intervening idle notification. Retry policy and attempt counts remain plugin-owned; compact-basic clears its per-agent overflow count when the chain reaches terminal `agent/idle`. Both DeepSeek adapters normalize recognized provider context-limit failures to `CONTEXT_WINDOW_EXCEEDED`. If cancellation lands after assistant tool calls are durable but before all calls dispatch, the loop records a synthetic `tool/call` and aborted `tool/result` pair for every undispatched call before following the normal abort path. The surface therefore never retains orphaned durable tool calls merely because cancellation won the race. @@ -34,7 +34,7 @@ If cancellation lands after assistant tool calls are durable but before all call For `pressure`, compact-basic resolves the durable provider/model target's adapter-owned capacity and exact-target policy, then applies the resulting threshold and retained-tail budgets to one unified `ctx.tokenMeter.measure()` result. Below pressure it returns without pruning. Once pressure qualifies, optional `ctx.toolResultPrune` rewrites oversized current results and compact-basic remeasures through the same meter; safe pressure skips the model call, while remaining pressure selects and summarizes from the pruned surface. The same singleton meter owns range pricing, provenance, shadowed token counts, and non-shrinking-summary rejection. Common defaults remain threshold ratio `0.8`, retained-history ratio `0.16`, summarization provider/model `''`, `maxTokens: 8192`, `compactionRetries: 1`, and `auto: true`; optional `modelPolicies` entries override them for an exact provider/model pair. -For canonical overflow, compact-basic requires no capacity metadata and bypasses scalar pressure and the normal retained-token budget. It prunes first, then chooses the maximal tool-balanced head range while leaving the newest indivisible unit and attempts one shrinking summary compaction under the same signal when a range exists. The automatic listener snapshots `session.surface.replaceGeneration` and returns `{ action: 'retry' }` whenever pruning or summarization increases it. This remains true when pruning lands before later summary work throws; cancellation still wins. A backend returning a result without replacement cannot authorize retry, while pruning-only progress can authorize a retry without a `CompactionResult`. +For canonical overflow, compact-basic requires no capacity metadata and bypasses scalar pressure and the normal retained-token budget. It prunes first, then chooses the maximal tool-balanced head range while leaving the newest indivisible unit and attempts one shrinking summary compaction under the same signal when a range exists. The automatic listener snapshots `session.surface.replaceGeneration` and calls `agent.retry()` whenever pruning or summarization increases it. This remains true when pruning lands before later summary work throws; cancellation still wins. A backend returning a result without replacement cannot authorize retry, while pruning-only progress can authorize a retry without a `CompactionResult`. `maxOverflowRetries` is optional and defaults to `1`; `0` disables overflow recovery without disabling pressure. `auto: false` registers neither automatic listener. Noncanonical errors, exhausted attempts, an already-aborted signal, a missing routed model, no safe range, no generation change, and recovery throws before any replacement all delegate to the next listener. With no later recovery, the loop reports the original provider error object and code. A recovery throw after generation advances authorizes retry from durable progress; cancellation or disposal remains authoritative even if recovery work completes concurrently. @@ -42,7 +42,7 @@ The default summarizer resolves explicit configuration, then the latest logged r ## Testing -Unit tests cover final-adapter failure provenance and identity, closed-step retry numbering and reset, cancellation and disposal, post-step ordering, routed-envelope pressure, pressure-gated pruning, pruning-only relief, pruned-input summarization, balanced overflow reduction, durable prune progress before later failure, generation proof, caps, delegation, and auxiliary-call routing. Real-loop tests cover thrown and in-band overflow through pruning or summary compaction to a reconstructed retry request. +Unit tests cover final-adapter failure provenance and identity, closed-turn retry numbering and reset, cancellation and disposal, step-boundary ordering, routed-envelope pressure, pressure-gated pruning, pruning-only relief, pruned-input summarization, balanced overflow reduction, durable prune progress before later failure, generation proof, caps, delegation, and auxiliary-call routing. Real-loop tests cover thrown and in-band overflow through pruning or summary compaction to a reconstructed retry request. ## Alternatives considered diff --git a/.agents/notes/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.zh.md b/.agents/notes/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.zh.md index 3b5b60a95b..fef9f48cbc 100644 --- a/.agents/notes/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.zh.md +++ b/.agents/notes/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.zh.md @@ -22,9 +22,9 @@ Status: implemented ### 请求恢复只覆盖最终模型边界 -`RequestError`、`RequestErrorDecision` 与 `agent/request-error` waterfall 表示最终适配器已经选定之后的失败。每个返回的流句柄都绑定一个私有失败集合;该集合在分发、异步迭代器构造与迭代过程中保留原始抛出错误的身份,同时防止把嵌套调用的错误来源误归到外层调用。终止性的带内 `error` 或 `aborted` finish 进入同一路径。提示词装配、请求中间件、请求日志、结果处理、工具、post-step 监听器与清理仍属于普通失败。 +`RequestError` 与 `agent/request-error` waterfall 表示最终适配器已经选定之后的失败。每个返回的流句柄都绑定一个私有失败集合;该集合在分发、异步迭代器构造与迭代过程中保留原始抛出错误的身份,同时防止把嵌套调用的错误来源误归到外层调用。终止性的带内 `error` 或 `aborted` finish 进入同一路径。提示词装配、请求中间件、请求日志、结果处理、工具、step 监听器与清理仍属于普通失败。 -恢复运行前,失败 step 已经关闭。重试会打开下一个编号 step,并从持久日志重建请求;连续恢复尝试计数只在提供方请求成功后重置。两个 DeepSeek 适配器都把识别出的提供方上下文限制错误规范化为 `CONTEXT_WINDOW_EXCEEDED`。 +恢复运行前,失败 step 已经关闭。负责处理的监听器修复持久状态、调用 `agent.retry()`,并停止 waterfall 委托。循环随后关闭失败 turn,并从持久日志开启一个重试 turn,中间不发布空闲通知。重试策略与尝试计数由插件自己拥有;compact-basic 在链路到达终态 `agent/idle` 时清除对应 agent 的溢出计数。两个 DeepSeek 适配器都把识别出的提供方上下文限制错误规范化为 `CONTEXT_WINDOW_EXCEEDED`。 如果取消发生在 assistant 工具调用已经持久化之后、所有调用完成分发之前,循环会为每个尚未分发的调用记录一对合成的 `tool/call` 与 aborted `tool/result`,随后进入正常中止路径。因此,表层不会仅因取消赢得竞态而留下孤立的持久工具调用。 @@ -34,7 +34,7 @@ Status: implemented 对于 `pressure`,compact-basic 先解析持久提供方/模型目标的适配器所属容量与精确目标策略,再把得到的阈值与保留尾部预算应用到一次统一的 `ctx.tokenMeter.measure()` 结果。低于压力时直接返回,不执行剪枝。压力达到条件后,可选的 `ctx.toolResultPrune` 会改写当前表层中过大的工具结果,compact-basic 再通过同一个 meter 重新计量;若压力恢复安全则跳过模型调用,否则从已剪枝表层选择范围并生成摘要。范围定价、来源、被遮蔽 token 数与非缩小摘要拒绝也由同一个单例 meter 完成。通用默认值保持为阈值比例 `0.8`、保留历史比例 `0.16`、摘要提供方/模型 `''`、`maxTokens: 8192`、`compactionRetries: 1` 与 `auto: true`;可选 `modelPolicies` 项可以按精确提供方/模型组合覆盖这些值。 -对于规范化溢出,compact-basic 不要求容量元数据,并绕过标量压力与普通保留 token 预算。它先执行剪枝,再在保留最新不可分割单元的同时选择最大的工具配对平衡头部范围;存在范围时,才在同一 signal 下尝试一次缩小摘要压缩。自动监听器先记录 `session.surface.replaceGeneration`,剪枝或摘要让 generation 增加时就返回 `{ action: 'retry' }`。即使剪枝先落盘而后续摘要工作抛错,这条规则仍然成立;取消依然优先。后端若只返回结果但没有替换表层,不能授权重试;只有剪枝取得进展时,即使没有 `CompactionResult` 也可以授权重试。 +对于规范化溢出,compact-basic 不要求容量元数据,并绕过标量压力与普通保留 token 预算。它先执行剪枝,再在保留最新不可分割单元的同时选择最大的工具配对平衡头部范围;存在范围时,才在同一 signal 下尝试一次缩小摘要压缩。自动监听器先记录 `session.surface.replaceGeneration`,剪枝或摘要让 generation 增加时就调用 `agent.retry()`。即使剪枝先落盘而后续摘要工作抛错,这条规则仍然成立;取消依然优先。后端若只返回结果但没有替换表层,不能授权重试;只有剪枝取得进展时,即使没有 `CompactionResult` 也可以授权重试。 `maxOverflowRetries` 可选且默认为 `1`;`0` 只禁用溢出恢复,不会禁用压力检查。`auto: false` 不注册任何自动监听器。非规范化错误、尝试耗尽、已经中止的 signal、缺失路由模型、没有安全范围、generation 未变化,以及在任何替换之前恢复抛错,都会委托给下一个监听器。若没有后续恢复,循环报告原始提供方错误对象与代码。generation 增加后的恢复抛错会基于持久进展授权重试;即使恢复工作并发完成,取消或销毁仍具有最终优先级。 @@ -42,7 +42,7 @@ Status: implemented ## 测试 -单元测试覆盖最终适配器失败的来源与身份、已关闭 step 的重试编号与重置、取消与销毁、post-step 顺序、已路由信封压力、压力门控剪枝、剪枝独立解除压力、从已剪枝输入生成摘要、平衡溢出缩减、后续失败前已落盘的剪枝进展、generation 证明、上限、委托与辅助调用路由。真实循环测试覆盖抛出式和带内溢出,并验证剪枝或摘要压缩后的重试请求从替换表层重建。 +单元测试覆盖最终适配器失败的来源与身份、已关闭 turn 的重试编号与重置、取消与销毁、step 边界顺序、已路由信封压力、压力门控剪枝、剪枝独立解除压力、从已剪枝输入生成摘要、平衡溢出缩减、后续失败前已落盘的剪枝进展、generation 证明、上限、委托与辅助调用路由。真实循环测试覆盖抛出式和带内溢出,并验证剪枝或摘要压缩后的重试请求从替换表层重建。 ## 考虑过的替代方案 diff --git a/.agents/notes/implemented/feature/2026-06-18-compaction-capability-seam.md b/.agents/notes/implemented/feature/2026-06-18-compaction-capability-seam.md index f5beab3eb4..7c9751ce24 100644 --- a/.agents/notes/implemented/feature/2026-06-18-compaction-capability-seam.md +++ b/.agents/notes/implemented/feature/2026-06-18-compaction-capability-seam.md @@ -37,7 +37,7 @@ An earlier draft put the full algorithm (the retention walk, token-summing, text Successful-call pressure cannot run at pre-step because final `agent/request` routing, provider output, tool results, buffered context, and steering do not exist there. Serial `agent/post-step(agent, turn, step, signal)` fires after those facts are durable and before `step/end`. `dsh-compact-basic` measures the canonical logged request through `ctx.tokenMeter`, so the next request sees any replacement without a speculative envelope override. Once pressure qualifies, optional `ctx.toolResultPrune` rewriting runs before summary selection; compact-basic remeasures the durable surface and skips summarization if pruning restores safe pressure. -Canonical provider context overflow takes a separate path. The failed step closes, `agent/request-error` receives the original request error and consecutive retry count, and compact-basic prunes before forcing one useful balanced reduction. It returns retry only if `session.surface.replaceGeneration` increases, including pruning-only progress when no summary range exists; the loop then opens a new numbered step and reconstructs its request from the durable log. No replacement, a recovery failure before any replacement, cancellation, an exhausted cap, or an unrelated error preserves the original provider failure. If pruning already advanced the generation before later summary work fails, recovery retries from that durable pruned surface unless cancellation or disposal wins. The complete lifecycle decision is in the [after-call recovery Agent Note](../architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.md). +Canonical provider context overflow takes a separate path. The failed step closes and `agent/request-error` receives the original request error. Compact-basic owns its per-agent overflow count, prunes before forcing one useful balanced reduction, and calls `agent.retry()` only if `session.surface.replaceGeneration` increases, including pruning-only progress when no summary range exists. The loop then closes the failed turn, opens a new numbered retry turn, and reconstructs its request from the durable log. No replacement, a recovery failure before any replacement, cancellation, an exhausted cap, or an unrelated error preserves the original provider failure. If pruning already advanced the generation before later summary work fails, recovery retries from that durable pruned surface unless cancellation or disposal wins. The complete lifecycle decision is in the [after-call recovery Agent Note](../architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.md). ``` assistant/message → tool/result/context/steering diff --git a/.agents/notes/implemented/feature/2026-07-07-session-prefix.md b/.agents/notes/implemented/feature/2026-07-07-session-prefix.md index 7faf80cf80..a8783d0b8f 100644 --- a/.agents/notes/implemented/feature/2026-07-07-session-prefix.md +++ b/.agents/notes/implemented/feature/2026-07-07-session-prefix.md @@ -10,7 +10,7 @@ The obvious third option — let a plugin edit the request's `messages` on the w ## Decision -`agent/session-prefix` is a waterfall on the agent event map ([`packages/core/agent/src/types.ts`](../../../../packages/core/agent/src/types.ts)): listeners receive a frozen empty seed and return an extension (the canonical contribution is a prepend, `[mine, ...await next()]`, which yields registration order on the wire). The loop ([`packages/core/agent-loop/src/loop.ts`](../../../../packages/core/agent-loop/src/loop.ts)) fires it once per loop instance, lazily before the instance's first `agent/pre-step`; the composed list is deep-cloned, deep-frozen, cached on the instance, and placed in front of the ENTIRE derived history — directly after the provider's system slot — on every request the instance sends ([wire order](../../../../docs/core-data-structures/core.md#the-request-envelope-llmcallconfig-and-the-logged-header)). +`agent/session-prefix` is a waterfall on the agent event map ([`packages/core/agent/src/types.ts`](../../../../packages/core/agent/src/types.ts)): listeners receive a frozen empty seed and return an extension (the canonical contribution is a prepend, `[mine, ...await next()]`, which yields registration order on the wire). The loop ([agent-loop source](../../../../packages/core/agent-loop/src/)) fires it once per loop instance, lazily before the instance's first `agent/pre-step`; the composed list is deep-cloned, deep-frozen, cached on the instance, and placed in front of the ENTIRE derived history — directly after the provider's system slot — on every request the instance sends ([wire order](../../../../docs/core-data-structures/core.md#the-request-envelope-llmcallconfig-and-the-logged-header)). Three properties carry the design: diff --git a/.agents/notes/implemented/simplification/2026-06-20-drop-unconsumed-llm-assembled-surfaces.md b/.agents/notes/implemented/simplification/2026-06-20-drop-unconsumed-llm-assembled-surfaces.md index b482a444b5..3f938f99c4 100644 --- a/.agents/notes/implemented/simplification/2026-06-20-drop-unconsumed-llm-assembled-surfaces.md +++ b/.agents/notes/implemented/simplification/2026-06-20-drop-unconsumed-llm-assembled-surfaces.md @@ -10,7 +10,7 @@ Status: implemented - `streamBlocks()` — a "convenience view" that runs the chunks through a `BlockAssembler` and yields completed `ContentBlock`s in stream order ([index.ts:137-144](../../../../packages/llm/llm/src/index.ts)). - `generate()` — one fully-assembled `GenerateResult`, dispatched through a second `llm/generate` waterfall ([index.ts:151-157](../../../../packages/llm/llm/src/index.ts)). -The only production consumer of the LLM service is the agent loop, and it uses `stream()` exclusively — feeding raw chunks through its own `BlockAssembler` so it can log chunks for replay fidelity while assembling in parallel ([packages/core/agent-loop/src/loop.ts](../../../../packages/core/agent-loop/src/loop.ts), the `ctx.llm.stream(req)` step). Grepping `streamBlocks` and `ctx.llm.generate` across `packages/*/src` and `examples/*/src` finds no production callers. The references are the service methods, docs, and tests; adapter tests use `generate()` as a convenient driver, but they can hand-drain `stream()` through the same assembler helper without preserving a public production API. +The only production consumer of the LLM service is the agent loop, and it uses `stream()` exclusively — feeding raw chunks through its own `BlockAssembler` so it can log chunks for replay fidelity while assembling in parallel ([packages/core/agent-loop/src/agent.ts](../../../../packages/core/agent-loop/src/agent.ts), the `ctx.llm.stream(req)` step). Grepping `streamBlocks` and `ctx.llm.generate` across `packages/*/src` and `examples/*/src` finds no production callers. The references are the service methods, docs, and tests; adapter tests use `generate()` as a convenient driver, but they can hand-drain `stream()` through the same assembler helper without preserving a public production API. This is the [drop-mutable-session-summary](2026-06-19-drop-mutable-session-summary.md) pattern: assembled-view APIs with tested contracts, consumed by tests rather than production. They were built speculatively for consumers that do not care about token-level deltas, but the one real consumer cares about deltas precisely so it can persist high-fidelity replay data. diff --git a/.agents/notes/implemented/simplification/2026-07-02-remove-stream-chunk-mirror.md b/.agents/notes/implemented/simplification/2026-07-02-remove-stream-chunk-mirror.md index c79202b95a..7de730edd7 100644 --- a/.agents/notes/implemented/simplification/2026-07-02-remove-stream-chunk-mirror.md +++ b/.agents/notes/implemented/simplification/2026-07-02-remove-stream-chunk-mirror.md @@ -4,7 +4,7 @@ Status: implemented ## Problem -The loop records every model token delta as a durable `assistant/chunk` session event AND emitted a parallel live `agent/stream-chunk` Cordis event carrying the identical data. In `packages/core/agent-loop/src/loop.ts` the two sat one line apart: +The loop records every model token delta as a durable `assistant/chunk` session event AND emitted a parallel live `agent/stream-chunk` Cordis event carrying the identical data. In `packages/core/agent-loop/src/agent.ts` the two sat one line apart: ```ts ignore-check const chunkEvent = session.append('assistant/chunk', { turn, step, chunk }) diff --git a/.agents/notes/implemented/simplification/2026-07-04-remove-agent-steering-mirror.md b/.agents/notes/implemented/simplification/2026-07-04-remove-agent-steering-mirror.md index ebfc774792..97c399f50e 100644 --- a/.agents/notes/implemented/simplification/2026-07-04-remove-agent-steering-mirror.md +++ b/.agents/notes/implemented/simplification/2026-07-04-remove-agent-steering-mirror.md @@ -4,7 +4,7 @@ Status: implemented ## Problem -`agent/steering` was the last remaining transient mirror of a durable session event. The loop's steering drain appends the durable `steering/message { turn, content, source }` and, on the very next line, emitted `agent/steering(agent, turn, content, source)` — the identical fact as a fire-and-forget event (`packages/core/agent-loop/src/loop.ts`, `drainSteering`). It had zero production listeners: the only subscriber anywhere was a loop regression test asserting the emit carried `source` — the same fact the durable event already records one line above. +`agent/steering` was the last remaining transient mirror of a durable session event. The loop's steering drain appends the durable `steering/message { turn, content, source }` and, on the very next line, emitted `agent/steering(agent, turn, content, source)` — the identical fact as a fire-and-forget event (`packages/core/agent-loop/src/agent.ts`, `drainOutbox`). It had zero production listeners: the only subscriber anywhere was a loop regression test asserting the emit carried `source` — the same fact the durable event already records one line above. `agent/steering` duplicated the immediately preceding durable `steering/message` with the same payload. `agent/queued` remains the live-only signal because it fires before persistence and covers work that may be cancelled before entering the log. @@ -12,7 +12,7 @@ Steering carries real production traffic — the hook bridges' turn-continuation ## Decision -`agent/steering` is removed from the agent event taxonomy: the declaration in `packages/core/agent/src/types.ts` (and its mention in the live-events JSDoc list there), the emit in `drainSteering` (whose then-unused `ctx` parameter went with it), the row in `packages/core/agent/README.md`, and the emit line in the loop-pseudocode blocks (the `packages/core/agent-loop/src/loop.ts` module doc and [architecture.md](../../../../docs/architecture.md)); the cordis catalog is regenerated without it. The one regression test pins source preservation on the durable `steering/message` event — the fact it pins lives on the log. +`agent/steering` is removed from the agent event taxonomy: the declaration in `packages/core/agent/src/types.ts` (and its mention in the live-events JSDoc list there), the emit in `drainOutbox`, the row in `packages/core/agent/README.md`, and the emit line in the loop-pseudocode blocks (the `packages/core/agent-loop/src/agent.ts` module doc and [architecture.md](../../../../docs/architecture.md)); the cordis catalog is regenerated without it. The one regression test pins source preservation on the durable `steering/message` event — the fact it pins lives on the log. Three implemented Agent Notes stated the retention, and each is amended per [implemented/AGENTS.md](../AGENTS.md) to point here as the record of the removal: the [boundary Agent Note](2026-06-20-remove-agent-boundary-mirror-events.md)'s retained-list entry, the [stream-chunk Agent Note](2026-07-02-remove-stream-chunk-mirror.md)'s scope clause, and the [event-domain-semantics Agent Note](../architecture/2026-06-30-event-domain-semantics.md)'s transient-emit enumeration. diff --git a/.agents/notes/implemented/simplification/2026-07-24-agent-loop-observable-state-machine.i18n.yaml b/.agents/notes/implemented/simplification/2026-07-24-agent-loop-observable-state-machine.i18n.yaml new file mode 100644 index 0000000000..1aa37f9255 --- /dev/null +++ b/.agents/notes/implemented/simplification/2026-07-24-agent-loop-observable-state-machine.i18n.yaml @@ -0,0 +1,6 @@ +# Bilingual-pair consistency record (docs/i18n/README.md): the git blob hash of each +# side as of the last confirmed-consistent state. Both languages carry equal authority; +# after editing either side, bring the other along and re-record with: +# pnpm run verify-translation-pairing --write +2026-07-24-agent-loop-observable-state-machine.md: 7a3607c5bcf6edece07433acd766a6dece551d1c +2026-07-24-agent-loop-observable-state-machine.zh.md: 7b53eb6087e99ce4c84caae7b94d275c0b6a10db diff --git a/.agents/notes/implemented/simplification/2026-07-24-agent-loop-observable-state-machine.md b/.agents/notes/implemented/simplification/2026-07-24-agent-loop-observable-state-machine.md new file mode 100644 index 0000000000..7a3607c5bc --- /dev/null +++ b/.agents/notes/implemented/simplification/2026-07-24-agent-loop-observable-state-machine.md @@ -0,0 +1,58 @@ +# Agent Note: Collapse agent-loop events around the observable state machine + +Status: implemented + +English | [中文](2026-07-24-agent-loop-observable-state-machine.zh.md) + +## Problem + +The agent loop exposed its control flow as a large set of Cordis events. Separate `pre-step` and `post-step` checkpoints bracketed a step, `session-prefix` and `step-result` transformed request and response messages, `request-error` decided whether a failed request retried inside its turn, and `turn-continuation` plus `turn-stop` composed competing continuation decisions. + +Those events made internal phases public even when the durable session log already owned the corresponding turn and step facts. They also mixed two extension models: some listeners observed a boundary and issued an agent command, while others returned control decisions that the loop interpreted. Understanding the public machine therefore required reconstructing event order, waterfall precedence, and special terminal overrides together. + +Agent lifetime, whole-agent activity, inbox-item progress, and per-turn settlement are independent state dimensions. Treating them as one status or one linear callback sequence makes ordinary questions ambiguous: an agent can remain `running` across several turns, an accepted item can be discarded without opening a turn, and one turn can settle while later work keeps the agent active. + +## Decision + +The public contract exposes four orthogonal state dimensions: + +- Registration lifetime is the `agent/created` to `agent/disposed` interval. Disposal is the terminal registry edge, not an `AgentStatus`. +- Whole-agent activity is `AgentStatus = 'idle' | 'running'`. Consecutive turns may share one `running` interval. +- A FIFO-backed message progresses from `agent/inbox/enqueue` to exactly one `agent/inbox/dequeue` or `agent/inbox/discard`, correlated by `AgentMessageId`. The inbox events describe acceptance, claim, and removal rather than turn completion. +- A claimed turn passes through prompt admission and zero or more request steps. An automatic retry closes the failed turn and immediately opens another; `agent/idle` reports only the terminal turn in that chain and remains distinct from the whole-agent transition to `status === 'idle'`. + +The loop keeps five machine extension events. `agent/prompt-submit` admits, rewrites, or blocks a claimed prompt. `agent/step` is the single awaited between-steps checkpoint and runs before every request is derived. `agent/request` is the waterfall for the frozen call configuration; the configuration comes only from `await next()`, not from a duplicate positional argument. `agent/request-error` serializes ownership of awaited model-request recovery. `agent/stopping` runs when the turn otherwise has no work left; a listener that needs another step records real steering with `agent.steer()`, and the loop decides from that data after all listeners settle. + +Continuation and termination are data rather than returned control enums. Tool calls and accepted steering require another step. A tool result carrying `concludesTurn` ends the tool loop at its step. The loop does not expose general `ContinuationDecision` or terminal-stop return channels. + +A model-request failure closes its step, then enters `agent/request-error` with the exact error, normalized `LlmFailure`, and live turn signal. A listener that owns recovery repairs state, calls `agent.retry()`, and returns without delegating. The loop closes the failed turn and opens one retry turn over that state without an intervening idle notification; retry is not another step inside the failed turn. `agent/idle` reports the terminal outcome, and `agent/error` remains the live error notification for consumers that report failures independently of turn settlement. + +The event taxonomy removes `agent/pre-step`, `agent/post-step`, `agent/session-prefix`, `agent/step-result`, `agent/turn-continuation`, and `agent/turn-stop`. Durable turn and step boundaries remain session events. Model-facing additions use logged message channels, request configuration uses `agent/request`, response content is recorded as assembled, failed-request recovery uses `agent/request-error` plus `agent.retry()`, and end-of-turn continuation uses `agent/stopping` plus steering. + +## Alternatives considered + +**Keep the fine-grained event sequence.** This preserves a dedicated interception point for every internal phase, including request-only prefixes, assistant-message rewriting, post-step work, in-turn request recovery, and terminal stop overrides. It also makes the loop's private sequencing a permanent public contract and lets overlapping seams express conflicting decisions. The decision accepts the lost interception points in exchange for one boundary per supported extension responsibility. + +**Represent disposal as a third `AgentStatus`.** This gives retained handles a terminal status value but duplicates the registry lifecycle already expressed by `agent/disposed`. The decision keeps `AgentStatus` about live activity and makes registration lifetime a separate dimension. + +**Return a retry decision from `agent/request-error`.** A returned instruction duplicates the existing `agent.retry()` command and requires the loop to carry policy history across attempts. The waterfall remains useful for ordered ownership: an unhandled listener delegates, while a handling listener performs its awaited repair, calls `agent.retry()`, and stops delegation. + +**Mirror durable turn and step boundaries as agent events.** This gives live consumers a second event stream for the same facts. The decision keeps the session log as the source of truth and exposes only extension checkpoints or live-only facts that the durable stream cannot carry. + +## Consequences + +The observable machine is smaller and compositional: registration lifetime, activity, item progress, and terminal settlement can be followed independently. In particular, `agent/idle` does not imply `agent.status === 'idle'`; it reports the terminal turn of one drain chain, while `agent/status` reports whether the whole agent is active. + +Plugins no longer rewrite every phase of the loop. There is no request-only message prefix, assistant-message transform, post-step checkpoint, generic continuation enum, generic terminal-stop result, or in-turn request retry. Extensions use the remaining owned channels instead of recreating those phases. + +Continuation plugins publish durable steering rather than returning an unlogged reason. Recovery plugins act after the failed step and explicitly schedule another turn through `agent.retry()`. This makes every attempt a complete turn while keeping asynchronous repair and policy ownership at one narrow waterfall boundary. + +The inbox lifecycle complements, rather than replaces, the durable session log. `AgentMessageId` correlates acceptance with claim or discard; turn and step numbers, messages, tool activity, and terminal reasons remain session facts. + +## Related + +- [Unify agent delivery on send(target × wakeup) and coalesce injected context into user/message](../architecture/2026-07-22-unified-send-and-coalesced-user-messages.md) +- [Remove implicit batching from ordinary sends](2026-07-17-one-send-one-turn.md) +- [Microkernel event taxonomy](../architecture/2026-06-11-microkernel-event-taxonomy.md) +- [Bounded LLM request recovery](../architecture/2026-06-21-bounded-llm-request-recovery.md) +- [Reconstructable requests](../architecture/2026-07-05-reconstructable-requests.md) diff --git a/.agents/notes/implemented/simplification/2026-07-24-agent-loop-observable-state-machine.zh.md b/.agents/notes/implemented/simplification/2026-07-24-agent-loop-observable-state-machine.zh.md new file mode 100644 index 0000000000..7b53eb6087 --- /dev/null +++ b/.agents/notes/implemented/simplification/2026-07-24-agent-loop-observable-state-machine.zh.md @@ -0,0 +1,58 @@ +# Agent Note: 围绕可观察状态机收拢 agent loop(智能体循环)事件 + +Status: implemented + +[English](2026-07-24-agent-loop-observable-state-machine.md) | 中文 + +## 问题 + +agent loop 曾将其控制流暴露为大量 Cordis 事件。`pre-step` 和 `post-step` 两个独立检查点分列步骤前后,`session-prefix` 和 `step-result` 分别变换请求消息与响应消息,`request-error` 决定失败的请求是否在当前轮次内重试,`turn-continuation` 与 `turn-stop` 则组合相互竞争的继续执行决策。 + +即使持久会话日志已经记录了对应的轮次与步骤事实,这些事件仍会将内部阶段公开。它们还混用了两种扩展模型:部分监听器观察边界并发出 agent 命令,另一些监听器则返回由循环解释的控制决策。因此,要理解公开状态机,必须同时还原事件顺序、waterfall(瀑布式事件)优先级和特殊的终止覆盖规则。 + +agent 生命周期、agent 整体活动状态、收件箱条目的进度以及每轮次的结算,是彼此独立的状态维度。若将它们视为一个状态或一条线性回调序列,常见问题就会产生歧义:agent 可以在多个轮次之间持续保持 `running`;已接受的条目可以不启动轮次就被丢弃;一个轮次可以完成结算,而后续工作仍让 agent 保持活动。 + +## 决策 + +公开契约暴露四个正交的状态维度: + +- 注册生命周期是从 `agent/created` 到 `agent/disposed` 的区间。dispose(资源释放)是注册表的终止边界,而不是一种 `AgentStatus`。 +- agent 整体活动状态为 `AgentStatus = 'idle' | 'running'`。连续多个轮次可以共用同一个 `running` 区间。 +- 由 FIFO 支撑的消息从 `agent/inbox/enqueue` 开始,最终必然进入 `agent/inbox/dequeue` 或 `agent/inbox/discard` 二者之一,并通过 `AgentMessageId` 关联。收件箱事件描述接受、领取和移除,而不是轮次完成。 +- 已领取的轮次经过提示词准入和零个或多个请求步骤。自动重试会关闭失败轮次并立即开启另一个轮次;`agent/idle` 只报告该重试链的终态轮次,且仍不同于 agent 整体转换到 `status === 'idle'`。 + +循环保留五个状态机扩展事件。`agent/prompt-submit` 对已领取的提示词执行准入、改写或阻断。`agent/step` 是步骤之间唯一需要等待的检查点,在每次派生请求前运行。`agent/request` 是冻结调用配置所用的 waterfall;配置只能来自 `await next()`,不再通过重复的位置参数提供。`agent/request-error` 串行确定需要等待的模型请求恢复由谁负责。当轮次原本已经没有剩余工作时,`agent/stopping` 运行;需要再执行一个步骤的监听器使用 `agent.steer()` 记录真实的 steering(中途引导),循环在所有监听器完成后根据这份数据作出决定。 + +是否继续和终止执行由数据表达,不再由返回的控制枚举表达。工具调用和已接受的 steering 要求再执行一个步骤。携带 `concludesTurn` 的工具结果会在其所属步骤终止工具循环。循环不再暴露通用的 `ContinuationDecision` 或终止停止返回通道。 + +模型请求失败会先关闭当前步骤,再携带准确错误、标准化 `LlmFailure` 和仍有效的轮次信号进入 `agent/request-error`。负责恢复的监听器修复状态、调用 `agent.retry()`,并停止继续委托。循环会关闭失败轮次,并基于该状态开启一个重试轮次,中间不发布空闲通知;重试不是失败轮次内的另一个步骤。`agent/idle` 报告终态结果;对于需要脱离轮次结算单独报告失败的消费方,`agent/error` 仍作为实时错误通知保留。 + +事件分类体系移除了 `agent/pre-step`、`agent/post-step`、`agent/session-prefix`、`agent/step-result`、`agent/turn-continuation` 和 `agent/turn-stop`。持久的轮次与步骤边界仍由会话事件记录。面向模型的新增内容使用有日志记录的消息通道,请求配置使用 `agent/request`,响应内容按组装后的原样记录,失败请求恢复使用 `agent/request-error` 加 `agent.retry()`,轮次结束时是否继续则使用 `agent/stopping` 加 steering 表达。 + +## 考虑过的替代方案 + +**保留细粒度事件序列。** 这样可以为每个内部阶段保留专用拦截点,包括仅用于请求的前缀、助手消息改写、步骤后处理、轮次内请求恢复以及终止停止覆盖。但这也会使循环的私有执行顺序成为永久的公开契约,并允许相互重叠的 seam 表达彼此冲突的决策。当前决策接受这些拦截点的缺失,以换取每项受支持的扩展职责仅对应一个边界。 + +**将 dispose 表示为第三种 `AgentStatus`。** 这样会让仍被持有的句柄得到一个终止状态值,但也会重复表达 `agent/disposed` 已经体现的注册表生命周期。当前决策让 `AgentStatus` 只表示活动中 agent 的状态,并将注册生命周期作为独立维度。 + +**让 `agent/request-error` 返回重试决策。** 返回指令会与现有的 `agent.retry()` 命令重复,还要求循环跨尝试携带策略历史。Waterfall 只需保留有序归属能力:未处理的监听器继续委托,负责处理的监听器完成可等待的修复、调用 `agent.retry()`,然后停止委托。 + +**将持久的轮次与步骤边界映射为 agent 事件。** 这样会为同一事实向实时消费方提供第二条事件流。当前决策将会话日志保留为真源,仅暴露扩展检查点或持久事件流无法承载的纯实时事实。 + +## 影响 + +可观察状态机更小,也更容易组合:注册生命周期、活动状态、条目进度和终态结算可以分别追踪。尤其是,`agent/idle` 并不意味着 `agent.status === 'idle'`;前者报告一次排空链的终态轮次,`agent/status` 则报告整个 agent 是否处于活动状态。 + +插件不再能够改写循环的每个阶段。不再提供仅用于请求的消息前缀、助手消息变换、步骤后检查点、通用的继续执行枚举、通用的终止停止结果或轮次内请求重试。扩展改用剩余的归属明确的通道,而不是重新构造这些阶段。 + +负责继续执行的插件发布可持久化的 steering,而不是返回未记录到日志中的原因。恢复插件在失败步骤结束后处理错误,并通过 `agent.retry()` 显式安排另一个轮次。这样,每次尝试都会成为完整轮次,同时异步修复和策略归属集中在一个狭窄的 waterfall 边界。 + +收件箱生命周期用于补充持久会话日志,而非取代它。`AgentMessageId` 将接受操作与领取或丢弃操作关联起来;轮次编号与步骤编号、消息、工具活动和终止原因仍属于会话事实。 + +## 相关内容 + +- [统一通过 send(target × wakeup) 交付 agent 消息,并将注入上下文合并到 user/message](../architecture/2026-07-22-unified-send-and-coalesced-user-messages.md) +- [移除普通发送中的隐式批处理](2026-07-17-one-send-one-turn.md) +- [微内核事件分类体系](../architecture/2026-06-11-microkernel-event-taxonomy.md) +- [有界 LLM 请求恢复](../architecture/2026-06-21-bounded-llm-request-recovery.md) +- [可重建的请求](../architecture/2026-07-05-reconstructable-requests.md) diff --git a/docs/agent-lifecycle.md b/docs/agent-lifecycle.md index 5f1e43eb17..d0eedfb4c8 100644 --- a/docs/agent-lifecycle.md +++ b/docs/agent-lifecycle.md @@ -15,7 +15,6 @@ sequenceDiagram participant LLM as ctx.llm participant Tools as ctx.tools participant Session - participant Persistence participant SDK as UI or SDK listener User->>Agent: followup(content) Agent-->>SDK: agent/inbox/enqueue @@ -26,7 +25,7 @@ sequenceDiagram Hooks-->>Driver: authoritative allow, block, or add context Driver->>Session: user/message or rejected turn/end Driver->>Prompt: system-prompt/assemble waterfall - Driver-->>Driver: agent/pre-step serial checkpoint + Driver-->>Driver: agent/step serial checkpoint Driver->>Session: step/start Driver->>LLM: agent/request waterfall, then llm/stream waterfall LLM-->>Driver: StreamChunk* @@ -35,9 +34,8 @@ sequenceDiagram alt final adapter or terminal in-band request failure Driver->>Session: step/end Driver->>Hooks: agent/request-error waterfall - Hooks-->>Driver: retry in a new step or preserve the original error + Hooks-->>Driver: call agent.retry() or preserve the original error else model request succeeded - Driver->>Hooks: agent/step-result waterfall Driver->>Session: assistant/message Driver->>Tools: classify pending call by executionMode loop barriers and bounded rolling pool, reclassify before start @@ -52,19 +50,16 @@ sequenceDiagram end end Driver->>Session: post-tool context and steering (no prompt-submit) - Driver->>Hooks: agent/post-step serial checkpoint Driver->>Session: step/end - Driver->>Hooks: agent/turn-continuation waterfall - Driver->>Hooks: agent/turn-stop serial terminal checkpoint + Driver->>Hooks: agent/stopping serial terminal checkpoint end Driver->>Session: turn/end - Driver->>Persistence: session/flush parallel checkpoint Driver-->>SDK: agent/status idle ``` The `assistant/message` edge records every successful provider call, including content-less and `max-tokens` finishes. Empty content stays out of derived history while the durable anchor retains usage and exact chunk provenance, including an explicit empty source set. -`dsh-compact-basic` uses `agent/post-step` for pressure after those durable facts and `agent/request-error` only for canonical context overflow. Once either trigger qualifies, optional tool-result pruning runs before summary selection. Recovery works between the closed failed step and a fresh retry step, and returns retry only when pruning or summarization advances the surface replacement generation; otherwise the original request error remains authoritative. +`dsh-compact-basic` uses `agent/step` for pressure before request derivation and `agent/request-error` only for canonical context overflow. Once either trigger qualifies, optional tool-result pruning runs before summary selection. Recovery works between the closed failed step and failed turn close, and opens a fresh retry turn only when pruning or summarization advances the surface replacement generation; otherwise the original request error remains authoritative. The returned `agent/prompt-submit` allow is authoritative; listeners wrapping `next()` preserve downstream content and additional contexts unless replacement is intentional. Steering bypasses that waterfall and joins at its durable checkpoint. diff --git a/docs/architecture.i18n.yaml b/docs/architecture.i18n.yaml index 8568b125fa..fdc8f4197d 100644 --- a/docs/architecture.i18n.yaml +++ b/docs/architecture.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write -architecture.md: c8f328560600d8a7e1f1ab659fab39c13c18dee3 -architecture.zh.md: 61fba67a8585220e1ef59f02844271fee8a46401 +architecture.md: 1bf151c0d7db88b102dd49fa8f21ece02beec9d2 +architecture.zh.md: 43d1b6ff0a2123f4a48ded99d938e70ebf93f084 diff --git a/docs/architecture.md b/docs/architecture.md index c8f3285606..1bf151c0d7 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -110,23 +110,23 @@ idle inject: do not open a turn or run the model ``` -Each step assembles ordered prompt sections, tool schemas, and variables; unknown references fail the turn. `dsh-system-prompt` owns identity and persona, while the loop supplies `model` and `cwd` ([prompt ownership](../.agents/notes/implemented/architecture/2026-07-05-prompt-variables-and-tool-guidance-ownership.md)). +Each step assembles ordered prompt sections, tool schemas, and variables; unknown references fail the turn. `dsh-system-prompt` owns identity and persona, while the loop supplies `provider`, `model`, and `cwd` ([prompt ownership](../.agents/notes/implemented/architecture/2026-07-05-prompt-variables-and-tool-guidance-ownership.md)). -Tool-time context—including active-turn `inject()` and post-tool `additionalContexts`—settles after results. Accepted steering drains from the same pending item at that boundary and requests another step. Idle `inject()` instead appends context immediately without changing turn numbering; persistence owns its eager drain. +Tool-time context—including active-turn `inject()` and post-tool `additionalContexts`—settles after results. Steering drains at that boundary and requests another step. Idle `inject()` appends context immediately without changing turn numbering; persistence drains it eagerly. -Pruning precedes summaries; overflow retries require durable progress. Bounded transient retries compose on `agent/request-error`; cancellation wins ([compaction](../.agents/notes/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.md), [retry](../.agents/notes/implemented/architecture/2026-06-21-bounded-llm-request-recovery.md)). +Pruning precedes summaries; overflow retries require durable progress. Recovery runs through `agent/request-error` between the failed step and turn closes. A handling policy calls `agent.retry()` to schedule one retry turn; cancellation wins ([compaction](../.agents/notes/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.md), [retry](../.agents/notes/implemented/architecture/2026-06-21-bounded-llm-request-recovery.md)). ### Failure Boundaries -Adapter failures close the step before `agent/request-error` with exact `Error`, `LlmFailure`, and history. Retry opens another step; success clears history; exhaustion stores failure on `turn/end`. Failed chunks commit no message/tool. +Adapter failures close the step before `agent/request-error` receives the exact `Error`, normalized `LlmFailure`, and turn signal. A handling listener calls `agent.retry()`; the loop closes the failed turn and opens another from durable history without an idle notification. Exhaustion leaves the failed `turn/end` terminal. Failed chunks commit no message or tool call. -Other failures use `agent/error`. Cancellation and disposal beat recovery; undispatched tool calls get synthetic `tool/call`/`ABORTED_BEFORE_DISPATCH` pairs. The turn signal retires before `turn/end`. Effective `cancel()` emits its typed cause before clearing queues and aborting; observers cannot veto, idle calls emit nothing, and durability records `aborted`. Disposal awaits quiescence ([decision](../.agents/notes/implemented/architecture/2026-07-16-explicit-turn-cancellation.md)). +Other failures use `agent/error`. Cancellation and disposal beat recovery; undispatched tools get synthetic `tool/call`/`ABORTED_BEFORE_DISPATCH` pairs. Effective `cancel(cause)` emits its cause before clearing queues and aborting; observers cannot veto and idle calls emit nothing. Durability records `aborted` for user or parent cancellation and `disposed` for teardown, which awaits quiescence. The cause changes reporting, not late result-context handling ([decision](../.agents/notes/implemented/architecture/2026-07-16-explicit-turn-cancellation.md)). -Turn and step execution events are turn-enclosed; an idle injected `user/message` may sit between turns. Reload closes an interrupted turn tail with a synthetic `interrupted` turn end. Post-close failures report only through `agent/error`; no safe in-turn position remains. Each turn has one `TurnEndReason`; [TurnEndReasonMap](core-data-structures/session.md#why-a-turn-ended-turnendreasonmap) owns the variants. +Turn and step events are turn-enclosed; idle injected `user/message` events may sit between turns. Reload closes an interrupted tail with a synthetic turn end. Post-close failures use only `agent/error`. Each turn has one [TurnEndReason](core-data-structures/session.md#why-a-turn-ended-turnendreasonmap). ### Agent Handles -`ctx.agents` owns live agents and returns `AgentHandle { agent, dispose() }`. Plugins use `send(content, completeOptions)` when routing must be explicit, or the `followup()`, `steer()`, and `inject()` presets; `cancel()` and `whenIdle()` control lifecycle. The caller fiber, factory provider, and consumer handle co-own teardown through one awaited disposer. +`ctx.agents` owns live agents and returns `AgentHandle { agent, dispose() }`. Plugins use complete `send()` options or the `followup()`, `steer()`, and `inject()` presets; `cancel()` and `whenIdle()` control lifecycle. One awaited disposer coordinates teardown ownership. ### Agent Scope @@ -138,9 +138,9 @@ Each agent owns a scoped `agent.ctx`; shared storage overlays global tool, promp The session log is authoritative. `deriveMessages()` projects model history; raw `assistant/chunk` events remain for replay and UI fidelity. Fork, resume, transcript rendering, telemetry, and persistence derive from the same stream. -**Model-visible ⟺ logged**: the log reconstructs every request — messages at `step/start` fronted by the header's session prefix, and headers by folding `request/header` — and the package-owned `dsh-agent-loop/invariant` can assert it through `ctx.invariants` ([reconstructability](../.agents/notes/implemented/architecture/2026-07-05-reconstructable-requests.md)). +**Model-visible ⟺ logged**: the log reconstructs every request from the messages at `step/start` and the folded `request/header`; the package-owned `dsh-agent-loop/invariant` can assert it through `ctx.invariants` ([reconstructability](../.agents/notes/implemented/architecture/2026-07-05-reconstructable-requests.md)). -Durability is a plugin concern. Backends buffer synchronous `session/event` notifications. The semantic checkpoint policy drains requests before adapter dispatch, recorded top-level calls before tool dispatch, and complete response/result batches at `agent/post-step`; the loop retains the final turn-end checkpoint. `SessionPersistence` stores `SessionEvent` directly and metadata in `SessionHeader`; JSONL defaults to checksummed Zstandard, with SQLite under one contract ([decision](../.agents/notes/implemented/bug-fix/2026-07-21-semantic-session-checkpoints.md)). +Durability is a plugin concern. Backends eagerly drain synchronous `session/event` notifications. The semantic checkpoint policy uses `session/flush` as an observation barrier before adapter dispatch, before top-level tool dispatch, and at `agent/step` before the next request. `SessionPersistence` stores `SessionEvent` directly and metadata in `SessionHeader`; JSONL defaults to checksummed Zstandard, with SQLite under one contract ([decision](../.agents/notes/implemented/bug-fix/2026-07-21-semantic-session-checkpoints.md)). `ctx.sessions.appendOutOfBand()` joins plugin-owned log-only events to an open turn or creates a balanced, flushed zero-step turn. `session/title` folds latest-wins with source seqs and provenance; its immediate fallback and sole optional async provider never delay the agent response. Forks inherit titles ([decision](../.agents/notes/implemented/feature/2026-07-21-log-backed-session-titles.md)). @@ -148,7 +148,7 @@ Durability is a plugin concern. Backends buffer synchronous `session/event` noti Messages use typed blocks from merge-extensible `ContentBlockMap`; the same pattern types `MessageSource`, `FinishReason`, `TurnTrigger`, and `TurnEndReason`. New blocks coordinate adapters, UI, compaction, token metering, and persistence; replay measurements live in [token-meter.md](core-data-structures/token-meter.md). -Streaming uses raw chunks and `BlockAssembler`. Each `LlmAdapter.stream()` is one provider attempt; adapters report facts and `agent/request-error` owns recovery. The loop logs chunks and successful provenance/replay state. Remote adapters use per-read idle watchdogs. Replay state crosses routes only when they share an adapter instance ([contract](core-data-structures/llm-streaming.md)). +Streaming uses raw chunks and `BlockAssembler`. Each `LlmAdapter.stream()` is one provider attempt; adapters report normalized failure facts and a handling `agent/request-error` plugin calls `agent.retry()`. The loop logs chunks and successful provenance/replay state. Remote adapters use per-read idle watchdogs. Replay state crosses routes only when they share an adapter instance ([contract](core-data-structures/llm-streaming.md)). ## Extension And Composition @@ -158,7 +158,7 @@ A swappable capability usually splits into **interface / implementation / consum Exceptions combine layers: LLM interface/consumer; filesystem policy; web registries; named skill/subagent providers. Subagents spawn fresh, fork a completed-turn prefix, or use ACP children ([subagent.md](core-data-structures/subagent.md)). -`dsh-workspace-context` composes baselines on `agent/session-prefix` and appends `ctx.fs`-discovered nested changes on `tools/post-execute`; its [decision](../.agents/notes/implemented/feature/2026-06-24-workspace-context.md) records isolation. `dsh-paths` owns shared paths. +`dsh-workspace-context` injects the baseline at the first `agent/step` and appends `ctx.fs`-discovered changes through `tools/post-execute`; its [decision](../.agents/notes/implemented/feature/2026-06-24-workspace-context.md) records isolation. `dsh-paths` owns shared paths. ### Bundles And Apps diff --git a/docs/architecture.zh.md b/docs/architecture.zh.md index 61fba67a85..43d1b6ff0a 100644 --- a/docs/architecture.zh.md +++ b/docs/architecture.zh.md @@ -110,23 +110,23 @@ idle inject: do not open a turn or run the model ``` -每个步骤都会组装有序提示词片段、工具 schema 和变量;未知引用会使该轮次失败。`dsh-system-prompt` 负责身份和角色设定,循环则提供 `model` 和 `cwd`([提示词归属](../.agents/notes/implemented/architecture/2026-07-05-prompt-variables-and-tool-guidance-ownership.md))。 +每个步骤都会组装有序提示词片段、工具 schema 和变量;未知引用会使该轮次失败。`dsh-system-prompt` 负责身份和角色设定,循环则提供 `provider`、`model` 和 `cwd`([提示词归属](../.agents/notes/implemented/architecture/2026-07-05-prompt-variables-and-tool-guidance-ownership.md))。 -工具执行阶段的上下文,包括活跃轮次内的 `inject()` 和工具执行后的 `additionalContexts`,会在结果记录完毕后落定。已接受的 steering(中途引导)会在同一边界从同一个待处理项排空,并请求再执行一个步骤。空闲状态下的 `inject()` 则会立即追加上下文,且不改变轮次编号;持久化层独立负责由此产生的即时排空。 +工具执行阶段的上下文,包括活跃轮次内的 `inject()` 和工具执行后的 `additionalContexts`,会在结果记录完毕后落定。Steering 会在同一边界排空并请求再执行一个步骤。空闲状态下的 `inject()` 会立即追加上下文,且不改变轮次编号;持久化层会尽快排空。 -裁剪先于摘要;溢出重试必须取得持久进展。有界的瞬态重试在 `agent/request-error` 上组合;取消优先([压缩](../.agents/notes/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.md)、[重试](../.agents/notes/implemented/architecture/2026-06-21-bounded-llm-request-recovery.md))。 +裁剪先于摘要;溢出重试必须取得持久进展。恢复会在失败步骤关闭后、轮次关闭前通过 `agent/request-error` 运行。负责处理的策略调用 `agent.retry()` 安排一个重试轮次;取消优先([压缩](../.agents/notes/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.md)、[重试](../.agents/notes/implemented/architecture/2026-06-21-bounded-llm-request-recovery.md))。 ### 失败边界 -适配器故障会先关闭步骤,再进入 `agent/request-error`;该事件会收到准确的 `Error`、`LlmFailure` 和历史记录。重试会开启另一个步骤;成功会清除历史记录;重试耗尽后,故障存入 `turn/end`。失败分片不会提交消息或工具。 +适配器故障会先关闭步骤,再由 `agent/request-error` 接收准确的 `Error`、标准化的 `LlmFailure` 和轮次信号。负责处理的监听器调用 `agent.retry()`;循环关闭失败轮次,并从持久历史开启另一个轮次,中间不发出空闲通知。重试耗尽后,失败的 `turn/end` 即为终态记录。失败分片不会提交消息或工具调用。 -其他故障使用 `agent/error`。取消和资源释放均优先于恢复;尚未分派的工具调用会得到合成的 `tool/call`/`ABORTED_BEFORE_DISPATCH` 对。轮次信号会在 `turn/end` 前失效。实际生效的 `cancel()` 会在清空队列和中止前发出类型化原因;观察方不能否决该操作,空闲状态下的调用不发出任何事件,持久化会记录 `aborted`。dispose 会等待系统停稳([决策](../.agents/notes/implemented/architecture/2026-07-16-explicit-turn-cancellation.md))。 +其他故障使用 `agent/error`。取消和资源释放优先于恢复;尚未分派的工具会得到合成的 `tool/call`/`ABORTED_BEFORE_DISPATCH` 对。实际生效的 `cancel(cause)` 在清空队列和中止前发出原因;观察方不能否决,空闲调用不发事件。用户或父级取消持久记录为 `aborted`,等待停稳的资源释放记录为 `disposed`。原因只改变报告方式,不改变延迟完成的结果上下文处理([决策](../.agents/notes/implemented/architecture/2026-07-16-explicit-turn-cancellation.md))。 -轮次和步骤的执行事件均位于轮次边界内;空闲时注入的 `user/message` 可以位于两个轮次之间。重新加载会用合成的 `interrupted` 轮次结束事件闭合中断轮次的日志尾部。关闭后的故障只通过 `agent/error` 报告;此时已没有安全的轮次内位置。每个轮次有一个 `TurnEndReason`;各变体由 [TurnEndReasonMap](core-data-structures/session.md#why-a-turn-ended-turnendreasonmap) 统一定义。 +轮次和步骤事件均位于轮次边界内;空闲时注入的 `user/message` 可以位于两个轮次之间。重新加载会用合成的轮次结束事件闭合中断尾部。关闭后的故障只使用 `agent/error`。每个轮次有一个 [TurnEndReason](core-data-structures/session.md#why-a-turn-ended-turnendreasonmap)。 ### Agent 句柄 -`ctx.agents` 拥有活跃 agent,并返回 `AgentHandle { agent, dispose() }`。插件需要显式路由时使用 `send(content, completeOptions)`,否则使用 `followup()`、`steer()` 和 `inject()` 预设;`cancel()` 与 `whenIdle()` 控制生命周期。调用方 fiber、工厂提供方和消费方句柄通过同一个需等待完成的 disposer 共同拥有拆卸过程。 +`ctx.agents` 拥有活跃 agent,并返回 `AgentHandle { agent, dispose() }`。插件使用完整的 `send()` 选项,或使用 `followup()`、`steer()` 和 `inject()` 预设;`cancel()` 与 `whenIdle()` 控制生命周期。一个需等待完成的 disposer 协调拆卸归属。 ### Agent 作用域 @@ -138,9 +138,9 @@ idle inject: 会话日志是权威依据。`deriveMessages()` 投影出模型历史;原始 `assistant/chunk` 事件留在日志中,以保证回放和 UI 保真。fork、恢复、transcript(文本记录)渲染、遥测和持久化均派生自同一个事件流。 -**模型可见 ⟺ 已记录**:日志可以重建每个请求,包括由请求头会话前缀置于开头的 `step/start` 时消息,以及通过折叠 `request/header` 得到的请求头;开发期不变量会断言这一点([可重建性](../.agents/notes/implemented/architecture/2026-07-05-reconstructable-requests.md))。 +**模型可见 ⟺ 已记录**:日志可以根据 `step/start` 时的消息和折叠后的 `request/header` 重建每个请求;由该包提供的 `dsh-agent-loop/invariant` 可通过 `ctx.invariants` 断言这一点([可重建性](../.agents/notes/implemented/architecture/2026-07-05-reconstructable-requests.md))。 -持久性由插件负责。后端会缓冲同步的 `session/event` 通知。语义检查点策略会在适配器分发前刷写请求,在工具分发前刷写已记录的顶层调用,并在 `agent/post-step` 刷写完整的响应与结果批次;循环仍保留最终的轮次结束检查点。`SessionPersistence` 直接存储 `SessionEvent`,并将元数据存入 `SessionHeader`;JSONL 默认采用带校验和的 Zstandard,SQLite 则遵循同一契约([决策](../.agents/notes/implemented/bug-fix/2026-07-21-semantic-session-checkpoints.md))。 +持久性由插件负责。后端会尽快排空同步的 `session/event` 通知。语义检查点策略使用 `session/flush` 作为观察屏障:分别位于适配器分发前、顶层工具分发前,以及下一次请求之前的 `agent/step`。`SessionPersistence` 直接存储 `SessionEvent`,并将元数据存入 `SessionHeader`;JSONL 默认采用带校验和的 Zstandard,SQLite 则遵循同一契约([决策](../.agents/notes/implemented/bug-fix/2026-07-21-semantic-session-checkpoints.md))。 `ctx.sessions.appendOutOfBand()` 会把插件所属的纯日志事件加入开放轮次,或创建一个平衡且已刷写的零步骤轮次。`session/title` 按后写覆盖方式折叠,并携带源 seq 和来源信息;其即时回退标题和唯一可选异步提供方都不会延迟 agent 响应。fork 会继承标题([决策](../.agents/notes/implemented/feature/2026-07-21-log-backed-session-titles.md))。 @@ -148,7 +148,7 @@ idle inject: 消息使用从可合并扩展的 `ContentBlockMap` 派生的类型化块;`MessageSource`、`FinishReason`、`TurnTrigger` 和 `TurnEndReason` 也采用同一模式定义类型。新增块会协调适配器、UI、压缩、token 计量和持久化;回放计量见 [token-meter.md](core-data-structures/token-meter.md)。 -流式输出使用原始分片和 `BlockAssembler`。每次 `LlmAdapter.stream()` 调用代表一次提供方尝试;适配器报告事实,`agent/request-error` 负责恢复。循环会记录分片及成功结果的来源信息和回放状态。远程适配器使用逐次读取空闲看门狗。只有当路由共用同一个适配器实例时,回放状态才会跨路由传递([契约](core-data-structures/llm-streaming.md))。 +流式输出使用原始分片和 `BlockAssembler`。每次 `LlmAdapter.stream()` 调用代表一次提供方尝试;适配器报告标准化的故障事实,负责处理的 `agent/request-error` 插件会调用 `agent.retry()`。循环会记录分片及成功结果的来源信息和回放状态。远程适配器使用逐次读取空闲看门狗。只有当路由共用同一个适配器实例时,回放状态才会跨路由传递([契约](core-data-structures/llm-streaming.md))。 ## 扩展与组合 @@ -158,7 +158,7 @@ idle inject: 例外情况会合并不同层次:LLM(大语言模型)合并接口和消费方,文件系统整合策略,web 使用注册表,skill 和 subagent 使用具名提供方。subagent 可以通过 spawn 创建全新实例、fork 一个已完成轮次的前缀,或使用 ACP(Agent Client Protocol)子 agent([subagent.md](core-data-structures/subagent.md))。 -`dsh-workspace-context` 在 `agent/session-prefix` 上组合基线,并在通过 `ctx.fs` 发现嵌套变更后,于 `tools/post-execute` 追加这些变更;其[决策](../.agents/notes/implemented/feature/2026-06-24-workspace-context.md)记录了隔离方式。`dsh-paths` 负责共享路径。 +`dsh-workspace-context` 在第一次 `agent/step` 注入基线,并通过 `tools/post-execute` 追加 `ctx.fs` 发现的变更;其[决策](../.agents/notes/implemented/feature/2026-06-24-workspace-context.md)记录了隔离方式。`dsh-paths` 负责共享路径。 ### 组合包与应用 diff --git a/docs/config-catalog.md b/docs/config-catalog.md index 49a0a1510f..6600d6a172 100644 --- a/docs/config-catalog.md +++ b/docs/config-catalog.md @@ -27,7 +27,7 @@ export interface AcpConfig { Depends on: `Stream` (`@agentclientprotocol/sdk`) -Source: [`packages/ui/acp/src/index.ts:285`](../packages/ui/acp/src/index.ts) +Source: [`packages/ui/acp/src/index.ts:286`](../packages/ui/acp/src/index.ts) ## `@deepseek-ai/dsh-acp-demo` @@ -113,7 +113,7 @@ export interface Config { Depends on: [`AgentOptions`](core-data-structures/core.md) · [`SessionId`](core-data-structures/core.md) -Source: [`packages/core/agent-loop/src/index.ts:360`](../packages/core/agent-loop/src/index.ts) +Source: [`packages/core/agent-loop/src/index.ts:147`](../packages/core/agent-loop/src/index.ts) ## `@deepseek-ai/dsh-agent-spine-demo` @@ -318,7 +318,7 @@ Requires: `llm` · `tokenMeter` export interface BasicCompactConfig extends CompactPolicyConfig { /** Exact provider/model overrides; duplicate targets fail plugin load. */ modelPolicies?: ModelCompactPolicyConfig[] - /** Enable automatic post-step pressure and overflow-recovery listeners. Defaults to `true`. */ + /** Enable automatic step-boundary pressure and overflow-recovery listeners. Defaults to `true`. */ auto?: boolean } @@ -411,7 +411,7 @@ export interface Config { } ``` -Source: [`packages/goal/goal/src/index.ts:56`](../packages/goal/goal/src/index.ts) +Source: [`packages/goal/goal/src/index.ts:55`](../packages/goal/goal/src/index.ts) ## `@deepseek-ai/dsh-hooks-claude` @@ -1063,7 +1063,7 @@ export interface Config { } ``` -Source: [`packages/session-title/session-title/src/index.ts:70`](../packages/session-title/session-title/src/index.ts) +Source: [`packages/session-title/session-title/src/index.ts:69`](../packages/session-title/session-title/src/index.ts) ## `@deepseek-ai/dsh-session-title-all-messages-llm` @@ -1367,7 +1367,7 @@ export interface Config { } ``` -Source: [`packages/goal/tool-goal/src/index.ts:27`](../packages/goal/tool-goal/src/index.ts) +Source: [`packages/goal/tool-goal/src/index.ts:25`](../packages/goal/tool-goal/src/index.ts) ## `@deepseek-ai/dsh-tool-lsp` @@ -1435,7 +1435,7 @@ export interface Config { } ``` -Source: [`packages/skill/tool-skill/src/index.ts:19`](../packages/skill/tool-skill/src/index.ts) +Source: [`packages/skill/tool-skill/src/index.ts:20`](../packages/skill/tool-skill/src/index.ts) ## `@deepseek-ai/dsh-tool-subagent` @@ -1567,7 +1567,7 @@ export interface Config { export type ToolPresentationMode = 'native' | 'code' | 'both' ``` -Source: [`packages/core/tools/src/index.ts:529`](../packages/core/tools/src/index.ts) +Source: [`packages/core/tools/src/index.ts:534`](../packages/core/tools/src/index.ts) ## `@deepseek-ai/dsh-tui` @@ -1631,7 +1631,7 @@ export interface TuiConfig { } ``` -Source: [`packages/ui/tui/src/index.ts:270`](../packages/ui/tui/src/index.ts) +Source: [`packages/ui/tui/src/index.ts:268`](../packages/ui/tui/src/index.ts) ## `@deepseek-ai/dsh-tui-demo` diff --git a/docs/cookbook/extension-cookbook.i18n.yaml b/docs/cookbook/extension-cookbook.i18n.yaml index 03d868a264..4c9900a606 100644 --- a/docs/cookbook/extension-cookbook.i18n.yaml +++ b/docs/cookbook/extension-cookbook.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write -extension-cookbook.md: c13b46e06a3b34512cd371e6a4868a6e932a575f -extension-cookbook.zh.md: aeb5f905278c07344c68d80da05dc5daf299b4f6 +extension-cookbook.md: 723c169e8e39e13be23db95c8152d067058a1a24 +extension-cookbook.zh.md: a02928a94a5ffba3608de822d5afbf920d0f16bf diff --git a/docs/cookbook/extension-cookbook.md b/docs/cookbook/extension-cookbook.md index c13b46e06a..723c169e8e 100644 --- a/docs/cookbook/extension-cookbook.md +++ b/docs/cookbook/extension-cookbook.md @@ -102,7 +102,7 @@ Every product feature maps to a listener on a documented extension seam — the | `/loop` | on the `turn/end` session event, `followup()` the next iteration; or force-continue | | Dynamic workflow | `ctx.workflows` + the worker-thread engine + the `workflow` tool; structured in-process children enforce output with scoped prompt/tool registrations, a monotonic tool guard, final `tools/result` commit (including enclosing `run_code`), and terminal `agent/turn-stop` | | Queued + steering messages | core `Agent.followup()` / `Agent.steer()` | -| Context compaction (auto + manual) | the `ctx.compact` seam + `dsh-compact-basic`; automatic pressure runs on serial `agent/post-step`, canonical overflow recovery runs on `agent/request-error`, and manual callers use the same compact service ([compaction Agent Note](../../.agents/notes/implemented/feature/2026-06-18-compaction-capability-seam.md) — the model-facing `/compact` consumer tool is deferred) | +| Context compaction (auto + manual) | the `ctx.compact` seam + `dsh-compact-basic`; automatic pressure runs on serial `agent/step`, canonical overflow recovery runs on `agent/request-error`, and manual callers use the same compact service ([compaction Agent Note](../../.agents/notes/implemented/feature/2026-06-18-compaction-capability-seam.md) — the model-facing `/compact` consumer tool is deferred) | | System prompt configurability | `ctx.systemPrompt.section()` with ordering and scope-local shadowing | | AGENTS.md (root) | a section provider reading the file | | AGENTS.md (subdir, on-touch) + file-change notices | `agent.inject()` from a watcher / tool-result listener | diff --git a/docs/cookbook/extension-cookbook.zh.md b/docs/cookbook/extension-cookbook.zh.md index aeb5f90527..a02928a94a 100644 --- a/docs/cookbook/extension-cookbook.zh.md +++ b/docs/cookbook/extension-cookbook.zh.md @@ -102,7 +102,7 @@ export function apply(ctx: Context) { | `/loop` | 在 `turn/end` 会话事件上 `followup()` 下一次迭代;或强制继续 | | 动态工作流 | `ctx.workflows` + worker-thread 引擎 + `workflow` 工具;结构化的进程内子任务通过作用域化的 prompt/工具注册、单调工具守卫、最终 `tools/result` 提交(包括外层 `run_code`)和终端 `agent/turn-stop` 来强制输出 | | 排队消息 + steering(中途引导) | 核心 `Agent.followup()` / `Agent.steer()` | -| 上下文压缩(context compaction)(自动 + 手动) | `ctx.compact` seam + `dsh-compact-basic`;自动压力检查运行在串行 `agent/post-step`,规范化溢出恢复运行在 `agent/request-error`,手动调用方使用同一个压缩服务([压缩 Agent Note](../../.agents/notes/implemented/feature/2026-06-18-compaction-capability-seam.md)——面向模型的 `/compact` 消费方工具已推迟) | +| 上下文压缩(context compaction)(自动 + 手动) | `ctx.compact` seam + `dsh-compact-basic`;自动压力检查运行在串行 `agent/step`,规范化溢出恢复运行在 `agent/request-error`,手动调用方使用同一个压缩服务([压缩 Agent Note](../../.agents/notes/implemented/feature/2026-06-18-compaction-capability-seam.md)——面向模型的 `/compact` 消费方工具已推迟) | | 系统提示词可配置性 | `ctx.systemPrompt.section()`,支持排序与作用域局部覆盖 | | AGENTS.md(根目录) | 一个读取该文件的 section provider | | AGENTS.md(子目录,按需触发)+ 文件变更通知 | 从 watcher / tool-result 监听器调用 `agent.inject()` | diff --git a/docs/cordis-catalog/events.md b/docs/cordis-catalog/events.md index 3c7a22351d..942ccb7ba2 100644 --- a/docs/cordis-catalog/events.md +++ b/docs/cordis-catalog/events.md @@ -15,15 +15,15 @@ Dispatch modes: **emit** (fire-and-forget), **waterfall** (each listener gets `n ### `agent/cancel-requested` — emit -Effective broad cancellation was requested, before queued/steering work is cleared or the active turn is aborted. This observe-only notification cannot veto cancellation; listener failures are contained. +Effective broad cancellation was requested, before queued/outbox work is cleared or the active turn is aborted. This observe-only notification cannot veto cancellation; listener failures are contained. ```ts cordis-catalog /** - * Effective broad cancellation was requested, before queued/steering work + * Effective broad cancellation was requested, before queued/outbox work * is cleared or the active turn is aborted. This observe-only notification * cannot veto cancellation; listener failures are contained. * @param agent - the agent whose current work is being cancelled. - * @param cause - resolved typed cancellation cause, including the default. + * @param cause - the explicit typed cancellation cause. * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent. * @mode emit */ @@ -32,7 +32,7 @@ Effective broad cancellation was requested, before queued/steering work is clear Types: [Agent](../core-data-structures/core.md) · [AgentCancelCause](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/core/agent/src/types.ts:359`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:328`](../../packages/core/agent/src/types.ts) ### `agent/created` — emit @@ -54,16 +54,16 @@ A fully configured agent and live session were published. Setup is composition-o Types: [Agent](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/core/agent/src/types.ts:295`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:270`](../../packages/core/agent/src/types.ts) ### `agent/disposed` — emit -An agent left the registry; AgentLoop emits this after driver quiescence but before session detachment and scoped-registration unwind. Custom registry users own their driver-ordering contract. +An agent left the registry; AgentLoop emits this after driver quiescence and scoped-registration unwind, but before session detachment. Custom registry users own their driver-ordering contract. ```ts cordis-catalog /** * An agent left the registry; AgentLoop emits this after driver quiescence - * but before session detachment and scoped-registration unwind. Custom + * and scoped-registration unwind, but before session detachment. Custom * registry users own their driver-ordering contract. * @param agent - the exact agent removed from the registry. * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent. @@ -74,16 +74,16 @@ An agent left the registry; AgentLoop emits this after driver quiescence but bef Types: [Agent](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/core/agent/src/types.ts:304`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:279`](../../packages/core/agent/src/types.ts) ### `agent/error` — emit -A step or turn errored. The loop reports a failure here (plus the logger) even when the error has no in-turn position for a session `error` event. +A step or turn errored. The machine reports a failure here (plus the logger) even when the error has no in-turn position for a durable record. ```ts cordis-catalog /** - * A step or turn errored. The loop reports a failure here (plus the logger) - * even when the error has no in-turn position for a session `error` event. + * A step or turn errored. The machine reports a failure here (plus the + * logger) even when the error has no in-turn position for a durable record. * @param agent - the agent whose turn errored. * @param turn - the turn in which the failure surfaced. * @param step - the step at which the failure surfaced. @@ -96,7 +96,30 @@ A step or turn errored. The loop reports a failure here (plus the logger) even w Types: [Agent](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/core/agent/src/types.ts:507`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:436`](../../packages/core/agent/src/types.ts) + +### `agent/idle` — emit + +One drain chain reached its terminal turn: that turn's `turn/end` is already committed. Automatically recovered failed turns do not emit this notification. `reason` says why; model-request recovery is exhausted when an error reaches it. + +```ts cordis-catalog +/** + * One drain chain reached its terminal turn: that turn's `turn/end` is + * already committed. Automatically recovered failed turns do not emit this + * notification. `reason` says why; model-request recovery is exhausted when + * an error reaches it. + * @param agent - the agent whose turn closed. + * @param turn - the terminal turn number. + * @param reason - why the terminal turn ended, with live error facts when it failed. + * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent. + * @mode emit + */ +'agent/idle'(this: Scoped, agent: Agent, turn: number, reason: IdleReason): void +``` + +Types: [Agent](../core-data-structures/core.md) · [IdleReason](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) + +Source: [`packages/core/agent/src/types.ts:423`](../../packages/core/agent/src/types.ts) ### `agent/inbox/dequeue` — emit @@ -117,21 +140,19 @@ The driver claimed one item out of the inbox: a queued item at a turn boundary, Types: [Agent](../core-data-structures/core.md) · [AgentMessage](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/core/agent/src/types.ts:335`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:306`](../../packages/core/agent/src/types.ts) ### `agent/inbox/discard` — emit -Pending inbox items were dropped without delivering them, so every enqueued id receives exactly one terminal `agent/inbox/dequeue` OR `agent/inbox/discard`. Emitters: `cancel()` without `keepInbox` (after `agent/cancel-requested`, before the abort); a terminal `agent/turn-stop` dropping pending steering (in-turn and on the post-turn late-steering drain); and disposal of any still-pending items (before `agent/status('disposed')`). Fires once per drop with every dropped item. +Pending inbox items were dropped without delivering them, so every enqueued id receives exactly one terminal `agent/inbox/dequeue` OR `agent/inbox/discard`. `cancel()` without `keepInbox`, including disposal, emits this after `agent/cancel-requested` when applicable and before aborting the active work. Fires once per drop with every dropped item. ```ts cordis-catalog /** * Pending inbox items were dropped without delivering them, so every * enqueued id receives exactly one terminal `agent/inbox/dequeue` OR - * `agent/inbox/discard`. Emitters: `cancel()` without `keepInbox` (after - * `agent/cancel-requested`, before the abort); a terminal `agent/turn-stop` - * dropping pending steering (in-turn and on the post-turn late-steering - * drain); and disposal of any still-pending items (before - * `agent/status('disposed')`). Fires once per drop with every dropped item. + * `agent/inbox/discard`. `cancel()` without `keepInbox`, including disposal, + * emits this after `agent/cancel-requested` when applicable and before + * aborting the active work. Fires once per drop with every dropped item. * @param agent - the agent whose inbox items were dropped. * @param messages - the discarded messages in FIFO order (queued then steering); never empty. * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent. @@ -142,21 +163,17 @@ Pending inbox items were dropped without delivering them, so every enqueued id r Types: [Agent](../core-data-structures/core.md) · [AgentMessage](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/core/agent/src/types.ts:349`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:318`](../../packages/core/agent/src/types.ts) ### `agent/inbox/enqueue` — emit -A detached, frozen item entered the agent's inbox (queued or steering FIFO). Source defaults are already applied, so `message` holds the exact accepted values. This is the enqueue-time live signal; the durable record is the eventual `user/message`/`steering/message`. Injection (`next-step`/no-wakeup) bypasses the FIFOs and does not emit this. +An item entered the queued or steering inbox. ```ts cordis-catalog /** - * A detached, frozen item entered the agent's inbox (queued or steering - * FIFO). Source defaults are already applied, so `message` holds the exact - * accepted values. This is the enqueue-time live signal; the durable record - * is the eventual `user/message`/`steering/message`. Injection - * (`next-step`/no-wakeup) bypasses the FIFOs and does not emit this. - * @param agent - the agent whose inbox received the item. - * @param message - the accepted message (its returned `id`, content, source, contexts, steering, and wakeup facts). + * An item entered the queued or steering inbox. + * @param agent - the owning agent. + * @param message - accepted content, source, and correlation identity. * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent. * @mode emit */ @@ -165,67 +182,17 @@ A detached, frozen item entered the agent's inbox (queued or steering FIFO). Sou Types: [Agent](../core-data-structures/core.md) · [AgentMessage](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/core/agent/src/types.ts:325`](../../packages/core/agent/src/types.ts) - -### `agent/post-step` — serial - -Awaited serial checkpoint after the response, real or synthetic tool results, injected context, and steering are durable but before `step/end`. A cancelled tool batch reaches this checkpoint with an aborted signal. - -```ts cordis-catalog -/** - * Awaited serial checkpoint after the response, real or synthetic tool - * results, injected context, and steering are durable but before `step/end`. - * A cancelled tool batch reaches this checkpoint with an aborted signal. - * @param agent - the agent whose step is settling. - * @param turn - the open turn number. - * @param step - the open step number. - * @param signal - the turn abort signal. - * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent. - * @mode serial - */ -'agent/post-step'(this: Scoped, agent: Agent, turn: number, step: number, signal: AbortSignal): Promise | void -``` - -Types: [Agent](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) - -Source: [`packages/core/agent/src/types.ts:457`](../../packages/core/agent/src/types.ts) - -### `agent/pre-step` — serial - -Awaited serial checkpoint before `step/start`; appends land outside the pending step and are included when the loop derives request history. `signal` cancels listener work. Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent. - -```ts cordis-catalog -/** - * Awaited serial checkpoint before `step/start`; appends land outside the - * pending step and are included when the loop derives request history. - * `signal` cancels listener work. - * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent. - * @param agent - the agent opening the step. - * @param turn - the open turn number. - * @param step - the pending step number. - * @param signal - the turn abort signal. - * @mode serial - */ -'agent/pre-step'(this: Scoped, agent: Agent, turn: number, step: number, signal: AbortSignal): Promise | void -``` - -Types: [Agent](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) - -Source: [`packages/core/agent/src/types.ts:388`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:296`](../../packages/core/agent/src/types.ts) ### `agent/prompt-submit` — waterfall -Allow, rewrite, or block one claimed prompt before it becomes a user message. Call `next()` for the unchanged default. A listener wrapping a downstream `allow` must preserve its `content` and `additionalContexts` unless it intentionally replaces them. The signal controls only this turn; listeners may cooperate with it but must not retain it to control another turn. Steering messages do not dispatch this event; they join an open turn at a steering checkpoint. +Allow, rewrite, or block one claimed prompt before it becomes a user message. Call `next()` for the unchanged default. The signal controls only this turn; listeners may cooperate with it but must not retain it for another turn. ```ts cordis-catalog /** * Allow, rewrite, or block one claimed prompt before it becomes a user - * message. Call `next()` for the unchanged default. A listener wrapping a - * downstream `allow` must preserve its `content` and `additionalContexts` - * unless it intentionally replaces them. The signal controls only this turn; - * listeners may cooperate with it but must not retain it to control another - * turn. Steering messages do not dispatch this event; they join an open turn - * at a steering checkpoint. + * message. Call `next()` for the unchanged default. The signal controls only this turn; + * listeners may cooperate with it but must not retain it for another turn. * @param agent - the agent whose turn claimed the message. * @param content - the claimed message's blocks, as queued. * @param source - the message's resolved source. @@ -238,84 +205,57 @@ Allow, rewrite, or block one claimed prompt before it becomes a user message. Ca Types: [Agent](../core-data-structures/core.md) · [ContentBlock](../core-data-structures/core.md) · [MessageSource](../core-data-structures/core.md) · [PromptDecision](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/core/agent/src/types.ts:404`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:355`](../../packages/core/agent/src/types.ts) ### `agent/request` — waterfall -Replace the frozen call configuration. Model-visible content must use logged channels; this seam cannot mutate messages. Injection here joins the next request because the current step boundary is already fixed. +Replace the frozen call configuration. `await next()` yields the config the machine would use (agent options on the first request, the logged header afterwards); return a replacement to switch. Model-visible content must use logged channels; this seam cannot mutate messages. ```ts cordis-catalog /** - * Replace the frozen call configuration. Model-visible content must use - * logged channels; this seam cannot mutate messages. Injection here joins - * the next request because the current step boundary is already fixed. + * Replace the frozen call configuration. `await next()` yields the config + * the machine would use (agent options on the first request, the logged + * header afterwards); return a replacement to switch. Model-visible + * content must use logged channels; this seam cannot mutate messages. * @param agent - the agent making the model call. * @param turn - the open turn number. * @param step - the step whose request this is. - * @param config - the config the loop would use (frozen); return a replacement to switch. - * @param signal - the current turn's explicit abort signal; ambient - * initiator identity does not imply liveness or cancellation authority. + * @param signal - the current turn's explicit abort signal. * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent. * @mode waterfall - */ -'agent/request'(this: Scoped, agent: Agent, turn: number, step: number, config: LlmCallConfig, signal: AbortSignal, next: () => Promise): Promise +*/ +'agent/request'(this: Scoped, agent: Agent, turn: number, step: number, signal: AbortSignal, next: () => Promise): Promise ``` Types: [Agent](../core-data-structures/core.md) · [LlmCallConfig](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/core/agent/src/types.ts:418`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:381`](../../packages/core/agent/src/types.ts) ### `agent/request-error` — waterfall -Recover a model-request failure after its failed step has closed. `retry` opens a new numbered step; `fail` preserves the original request error. Call `next()` to delegate to the next recovery listener or the default. +Handle a model-request failure after its failed step has closed but before the failed turn closes. A listener calls Agent.retry to schedule one retry turn, returns without `next()` when it owns the error, or calls `next()` to delegate. The default leaves the failure terminal. ```ts cordis-catalog /** - * Recover a model-request failure after its failed step has closed. `retry` - * opens a new numbered step; `fail` preserves the original request error. - * Call `next()` to delegate to the next recovery listener or the default. + * Handle a model-request failure after its failed step has closed but + * before the failed turn closes. A listener calls {@link Agent.retry} to + * schedule one retry turn, returns without `next()` when it owns the error, + * or calls `next()` to delegate. The default leaves the failure terminal. * @param agent - the agent whose request failed. * @param turn - the open turn number. * @param step - the failed step number. * @param error - the original model-request failure. * @param failure - serializable facts normalized at the final adapter boundary. - * @param priorFailures - immutable failures that already authorized another request in this consecutive sequence. * @param signal - the turn abort signal. * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent. * @mode waterfall */ -'agent/request-error'(this: Scoped, agent: Agent, turn: number, step: number, error: RequestError, failure: LlmFailure, priorFailures: readonly LlmFailure[], signal: AbortSignal, next: () => Promise): Promise +'agent/request-error'(this: Scoped, agent: Agent, turn: number, step: number, error: RequestError, failure: LlmFailure, signal: AbortSignal, next: () => Promise): Promise ``` -Types: [Agent](../core-data-structures/core.md) · [LlmFailure](../core-data-structures/llm-streaming.md) · [RequestError](../core-data-structures/core.md) · [RequestErrorDecision](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) +Types: [Agent](../core-data-structures/core.md) · [LlmFailure](../core-data-structures/llm-streaming.md) · [RequestError](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/core/agent/src/types.ts:472`](../../packages/core/agent/src/types.ts) - -### `agent/session-prefix` — waterfall - -Compose request-only messages placed before derived history. The frozen result is computed once per loop instance, logged on its anchoring request header, and reused so the provider prefix remains stable. Interrupted composition is discarded. Composition precedes the first `agent/pre-step` and request boundary, so listener appends join the current request. Changing context belongs in history; contributors should prepend to `await next()` to preserve registration order. Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent. - -```ts cordis-catalog -/** - * Compose request-only messages placed before derived history. The frozen - * result is computed once per loop instance, logged on its anchoring request - * header, and reused so the provider prefix remains stable. Interrupted - * composition is discarded. Composition precedes the first `agent/pre-step` - * and request boundary, so listener appends join the current request. - * Changing context belongs in history; contributors should prepend to - * `await next()` to preserve registration order. - * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent. - * @param agent - the agent whose session prefix is being composed. - * @param prefix - the frozen seed; return an extended replacement. - * @param signal - the current turn's explicit abort signal. - * @mode waterfall - */ -'agent/session-prefix'(this: Scoped, agent: Agent, prefix: Message[], signal: AbortSignal, next: () => Promise): Promise -``` - -Types: [Agent](../core-data-structures/core.md) · [Message](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) - -Source: [`packages/core/agent/src/types.ts:433`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:396`](../../packages/core/agent/src/types.ts) ### `agent/session-start` — emit @@ -337,16 +277,16 @@ The session lifecycle began, once before the first turn. Use `agent.inject()` to Types: [Agent](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) · [SessionStartSource](../core-data-structures/core.md) -Source: [`packages/core/agent/src/types.ts:372`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:341`](../../packages/core/agent/src/types.ts) ### `agent/status` — emit -Agent status changed (`idle` ⇄ `running`, or → `disposed`). `send()` does not enter `running` synchronously; drive lifecycle from this event. +Agent status changed (`idle` ⇄ `running`). `send()` does not enter `running` synchronously; drive lifecycle from this event. ```ts cordis-catalog /** - * Agent status changed (`idle` ⇄ `running`, or → `disposed`). `send()` does - * not enter `running` synchronously; drive lifecycle from this event. + * Agent status changed (`idle` ⇄ `running`). `send()` does not enter + * `running` synchronously; drive lifecycle from this event. * @param agent - the agent whose status flipped. * @param status - the status just entered (the transition's destination). * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent. @@ -357,74 +297,57 @@ Agent status changed (`idle` ⇄ `running`, or → `disposed`). `send()` does no Types: [Agent](../core-data-structures/core.md) · [AgentStatus](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/core/agent/src/types.ts:313`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:288`](../../packages/core/agent/src/types.ts) -### `agent/step-result` — waterfall +### `agent/step` — serial -Waterfall: post-process the assembled assistant Message before tool dispatch (validation, content rewriting, …). +Awaited serial checkpoint before EVERY request of a turn is built (the first as well as each post-tools continuation). The single "between steps" seam: inject context, steer, or edit the session log here — the request's history derives from the log right after this settles. ```ts cordis-catalog /** - * Waterfall: post-process the assembled assistant {@link Message} before - * tool dispatch (validation, content rewriting, …). - * @param agent - the agent that received the step's response. + * Awaited serial checkpoint before EVERY request of a turn is built (the + * first as well as each post-tools continuation). The single "between + * steps" seam: inject context, steer, or edit the session log here — the + * request's history derives from the log right after this settles. + * @param agent - the agent about to send a request. * @param turn - the open turn number. - * @param step - the step that produced the message. - * @param message - the assistant message as assembled from the stream. - * @param signal - the current turn's explicit abort signal. + * @param step - the step number about to open. + * @param signal - the turn abort signal. * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent. - * @mode waterfall + * @mode serial */ -'agent/step-result'(this: Scoped, agent: Agent, turn: number, step: number, message: Message, signal: AbortSignal, next: () => Promise): Promise +'agent/step'(this: Scoped, agent: Agent, turn: number, step: number, signal: AbortSignal): Promise | void ``` -Types: [Agent](../core-data-structures/core.md) · [Message](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) +Types: [Agent](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/core/agent/src/types.ts:445`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:368`](../../packages/core/agent/src/types.ts) -### `agent/turn-continuation` — waterfall +### `agent/stopping` — serial -Override whether the turn continues. The default continues after tool calls or steering and stops otherwise; a continue reason becomes steering. +The turn is about to close: the model owes no response (no live tool calls, no fresh steering). Awaited before the boundary commits — a listener that objects steers (`agent.steer(...)`) and the machine re-reads its inbox: fresh steering runs another step, none closes the turn. Data decides, so listener order cannot change the outcome. The inverse control (stop a tool loop early) is data too: a tool result carrying `concludesTurn` ends the turn at its step. ```ts cordis-catalog /** - * Override whether the turn continues. The default continues after tool - * calls or steering and stops otherwise; a continue reason becomes steering. - * @param agent - the agent deciding whether to run another step. - * @param turn - the turn being continued or stopped. - * @param defaultDecision - what the loop would do absent an override. - * @param signal - the current turn's explicit abort signal. - * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent. - * @mode waterfall - */ -'agent/turn-continuation'(this: Scoped, agent: Agent, turn: number, defaultDecision: ContinuationDecision, signal: AbortSignal, next: () => Promise): Promise -``` - -Types: [Agent](../core-data-structures/core.md) · [ContinuationDecision](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) - -Source: [`packages/core/agent/src/types.ts:483`](../../packages/core/agent/src/types.ts) - -### `agent/turn-stop` — serial - -Monotonic terminal-stop checkpoint after continuation and steering are folded; a stop remains authoritative through turn close and flush: steering queued in that window is discarded, while ordinary sends survive. - -```ts cordis-catalog -/** - * Monotonic terminal-stop checkpoint after continuation and steering are - * folded; a stop remains authoritative through turn close and flush: - * steering queued in that window is discarded, while ordinary sends survive. - * @param agent - the agent whose composed continuation outcome may be stopped. - * @param turn - the turn at its terminal-stop checkpoint. + * The turn is about to close: the model owes no response (no live tool + * calls, no fresh steering). Awaited before the boundary commits — a + * listener that objects steers (`agent.steer(...)`) and the machine + * re-reads its inbox: fresh steering runs another step, none closes the + * turn. Data decides, so listener order cannot change the outcome. The + * inverse control (stop a tool loop early) is data too: a tool result + * carrying `concludesTurn` ends the turn at its step. + * @param agent - the agent whose turn is at its stop boundary. + * @param turn - the turn about to close. * @param signal - the current turn's explicit abort signal. * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent. * @mode serial */ -'agent/turn-stop'(this: Scoped, agent: Agent, turn: number, signal: AbortSignal): Promise | ContinuationStop | undefined +'agent/stopping'(this: Scoped, agent: Agent, turn: number, signal: AbortSignal): Promise | void ``` -Types: [Agent](../core-data-structures/core.md) · [ContinuationStop](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) +Types: [Agent](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/core/agent/src/types.ts:494`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:411`](../../packages/core/agent/src/types.ts) ## `agent-loop/*` @@ -447,7 +370,7 @@ A declarative agent entry failed before it could publish a live agent. Consumers Types: [SessionId](../core-data-structures/core.md) -Source: [`packages/core/agent-loop/src/index.ts:353`](../../packages/core/agent-loop/src/index.ts) +Source: [`packages/core/agent-loop/src/index.ts:140`](../../packages/core/agent-loop/src/index.ts) ## `approval/*` @@ -570,7 +493,7 @@ Goal mutation accepted by one live agent. The matching context event is already Types: [Agent](../core-data-structures/core.md) · [GoalChanged](../core-data-structures/goal.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/goal/goal/src/types.ts:167`](../../packages/goal/goal/src/types.ts) +Source: [`packages/goal/goal/src/types.ts:169`](../../packages/goal/goal/src/types.ts) ## `llm/*` @@ -620,7 +543,7 @@ Creation announcement during session publication. A synchronous throw vetoes and Types: [Scoped](../core-data-structures/scope.md) · [Session](../core-data-structures/session.md) -Source: [`packages/core/session/src/index.ts:79`](../../packages/core/session/src/index.ts) +Source: [`packages/core/session/src/index.ts:70`](../../packages/core/session/src/index.ts) ### `session/disposed` — emit @@ -641,7 +564,7 @@ Emitted once when an announced session leaves the store, including publication r Types: [Scoped](../core-data-structures/scope.md) · [Session](../core-data-structures/session.md) -Source: [`packages/core/session/src/index.ts:89`](../../packages/core/session/src/index.ts) +Source: [`packages/core/session/src/index.ts:80`](../../packages/core/session/src/index.ts) ### `session/event` — emit @@ -664,7 +587,7 @@ Post-commit, fire-and-forget append feed. The listener snapshot resolves before Types: [Scoped](../core-data-structures/scope.md) · [Session](../core-data-structures/session.md) · [SessionEvent](../core-data-structures/core.md) -Source: [`packages/core/session/src/index.ts:101`](../../packages/core/session/src/index.ts) +Source: [`packages/core/session/src/index.ts:92`](../../packages/core/session/src/index.ts) ### `session/flush` — parallel @@ -685,7 +608,7 @@ Awaited parallel durability checkpoint: every listener runs and the caller await Types: [Scoped](../core-data-structures/scope.md) · [Session](../core-data-structures/session.md) -Source: [`packages/core/session/src/index.ts:111`](../../packages/core/session/src/index.ts) +Source: [`packages/core/session/src/index.ts:102`](../../packages/core/session/src/index.ts) ## `subagent/*` diff --git a/docs/cordis-catalog/services.md b/docs/cordis-catalog/services.md index 2d1917934f..23fac905ba 100644 --- a/docs/cordis-catalog/services.md +++ b/docs/cordis-catalog/services.md @@ -27,7 +27,7 @@ create(id: SessionId, options: AgentOptions = {}, meta: Pick ``` @@ -1246,7 +1246,7 @@ fork(source: SessionForkSource, boundary?: number, childSessionId?: SessionId): Types: [CreateSessionOptions](../core-data-structures/persistence.md) · [OutOfBandSessionEventType](../core-data-structures/session.md) · [Session](../core-data-structures/session.md) · [SessionEvent](../core-data-structures/core.md) · [SessionEventMap](../core-data-structures/session.md) · [SessionId](../core-data-structures/core.md) · [TurnTrigger](../core-data-structures/session.md) -Source: [`packages/core/session/src/index.ts:604`](../../packages/core/session/src/index.ts) +Source: [`packages/core/session/src/index.ts:593`](../../packages/core/session/src/index.ts) ## `ctx.sessionTitle` — `SessionTitleService` @@ -1280,7 +1280,7 @@ register(provider: SessionTitleProvider): () => Promise Types: [Session](../core-data-structures/session.md) · [SessionTitleProvider](../core-data-structures/session-title.md) · [SessionTitleSnapshot](../core-data-structures/session-title.md) -Source: [`packages/session-title/session-title/src/index.ts:284`](../../packages/session-title/session-title/src/index.ts) +Source: [`packages/session-title/session-title/src/index.ts:283`](../../packages/session-title/session-title/src/index.ts) ## `ctx.skills` — `SkillService` @@ -1686,7 +1686,7 @@ async execute(exec: ToolExecutionInput): Promise Types: [ScopeKey](../core-data-structures/scope.md) · [ToolDefinition](../core-data-structures/tools.md) · [ToolExecutionInput](../core-data-structures/tools.md) · [ToolExecutionMode](../core-data-structures/tools.md) · [ToolExecutionResult](../core-data-structures/tools.md) · [ToolGuard](../core-data-structures/tools.md) · [ToolRestriction](../core-data-structures/tools.md) · [ToolSchema](../core-data-structures/tools.md) -Source: [`packages/core/tools/src/index.ts:634`](../../packages/core/tools/src/index.ts) +Source: [`packages/core/tools/src/index.ts:639`](../../packages/core/tools/src/index.ts) ## `ctx.tui` — `TuiExtensionService` (abstract seam) @@ -1709,7 +1709,7 @@ The concrete provider retains pi-tui, focus, and terminal lifecycle state. Plugi abstract openOverlay(request: TuiOverlayRequest): TuiOverlaySession ``` -Source: [`packages/ui/tui/src/index.ts:150`](../../packages/ui/tui/src/index.ts) +Source: [`packages/ui/tui/src/index.ts:148`](../../packages/ui/tui/src/index.ts) ## `ctx.userInteraction` — `UserInteractionService` diff --git a/docs/core-data-structures/compaction.md b/docs/core-data-structures/compaction.md index 6fa281a96c..96324e3e85 100644 --- a/docs/core-data-structures/compaction.md +++ b/docs/core-data-structures/compaction.md @@ -60,7 +60,7 @@ type CompactionTrigger = 'pressure' | 'context-overflow' `CompactService` exposes `compactIfNeeded(agent, trigger, signal)` for automatic `pressure` or `context-overflow` policy, returning `null` when no safe work exists, and `compactRegion(...)` for an explicit inclusive surface range. Every backend marks its replacement `user/message` with the package-exported `COMPACT_CHECKPOINT_SOURCE`; consumers call `isCompactCheckpointSource()` instead of coupling checkpoint recognition to one backend. Implementations must forward the supplied signal to summarization. The seam owns no pricing API: the singleton [`ctx.tokenMeter`](token-meter.md) directly owns estimation and replay, while `dsh-compact-basic` owns retention, event sequencing, routed summarization calls, and their configuration. -Pressure compaction runs at serial `agent/post-step`, after successful assistant output, tool results, buffered context, and steering are durable but before `step/end`. Once pressure or canonical overflow qualifies, compact-basic invokes optional [`ctx.toolResultPrune`](../../packages/compact/compact-tool-result-prune/README.md) before range selection, remeasures through `ctx.tokenMeter`, and can advance the surface without a summary. Failed-request recovery runs through `agent/request-error` after the failed step closes and authorizes a fresh numbered-step retry only when the surface replacement generation advances, even if later summary work throws after pruning; cancellation still wins. Region boundaries preserve tool-call/result pairing but not whole turns, allowing early closed steps of one oversized turn to compact. `dsh-compact-basic` owns thresholds, retained-tail policy, overflow caps, and failure handling. +Pressure compaction runs at serial `agent/step` before request derivation. Once pressure or canonical overflow qualifies, compact-basic invokes optional [`ctx.toolResultPrune`](../../packages/compact/compact-tool-result-prune/README.md) before range selection, remeasures through `ctx.tokenMeter`, and can advance the surface without a summary. Failed-request recovery runs through `agent/request-error` after the failed step closes and calls `agent.retry()` only when the surface replacement generation advances, even if later summary work throws after pruning; cancellation still wins. Region boundaries preserve tool-call/result pairing but not whole turns, allowing early closed steps of one oversized turn to compact. `dsh-compact-basic` owns thresholds, retained-tail policy, overflow caps, and failure handling. The seam exports `toolPairingBalancedBefore(session, seq)` and `toolPairingBalancedAfter(session, seq)` for those edge checks. Both validate current surface membership and reject missing seqs and orphan results; the [package contract](../../packages/compact/compact/README.md#tool-pairing-boundaries) owns their cache semantics. diff --git a/docs/core-data-structures/core.i18n.yaml b/docs/core-data-structures/core.i18n.yaml index 4d2dc00fca..403b25986e 100644 --- a/docs/core-data-structures/core.i18n.yaml +++ b/docs/core-data-structures/core.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write -core.md: 1d7a06405bbda523f37389e1b09c62549ece6750 -core.zh.md: 2048a59451c3064eefdfe4225cc186c77921a9c0 +core.md: 882352e7dc814443468195f57b9c3f7c9401b170 +core.zh.md: e686216530f827ad8b94b42a43e3fa66a2d3a36e diff --git a/docs/core-data-structures/core.md b/docs/core-data-structures/core.md index 1d7a06405b..882352e7dc 100644 --- a/docs/core-data-structures/core.md +++ b/docs/core-data-structures/core.md @@ -454,13 +454,7 @@ type AgentCancelCause = `Agent` is an abstract class: concrete drivers implement the abstract members, while `followup`/`steer`/`inject` are shared concrete delegates to the single abstract `send` over the (`target` × `wakeup`) matrix. ```ts type-equiv -/** - * Public agent handle; its concrete implementation is internal to - * `@deepseek-ai/dsh-agent-loop`. An abstract class rather than an interface so - * the fixed-preset aliases ({@link Agent.followup}, {@link Agent.steer}, - * {@link Agent.inject}) are shared concrete delegates over the single abstract - * {@link Agent.send} primitive; concrete drivers implement `send` once. - */ +/** Public live-agent handle with aliases over the unified delivery primitive. */ abstract class Agent { /** The single identity shared with {@link session}. */ abstract readonly id: SessionId @@ -475,7 +469,7 @@ abstract class Agent { /** * The unified delivery primitive over the (`target` × `wakeup`) matrix. - * Detaches, validates, and freezes one lossless-JSON item, then routes it: + * It routes the caller's typed content and source as follows: * * - `next-turn` queues an item that becomes the sole ordinary message of its * own FIFO-ordered turn; `wakeup:true` wakes a @@ -483,12 +477,9 @@ abstract class Agent { * - `next-step` with `wakeup:true` submits steering into the active turn * (idle falls back to a woken `next-turn`). * - `next-step` with `wakeup:false` injects durable model-facing context - * without running the model: an open turn joins at the current log position - * (deferred behind an executing tool batch until it settles), and an idle - * inject records a one-shot turn with its own durability checkpoint. - * - * Attached contexts share the same snapshot and ownership boundary. Invalid - * input throws synchronously before any notification, enqueue, or append. + * without running the model: an open turn stages it for the next safe log + * position, while an idle injection appends it immediately without opening + * a turn. * @param content - the model-facing content blocks to deliver. * @param options - target queue, wakeup decision, and source. * @returns the accepted message's {@link AgentMessageId}, stable across its `agent/inbox/*` events. @@ -500,8 +491,7 @@ abstract class Agent { * turn. An effective call first emits `agent/cancel-requested` with the * resolved typed cause. The first cause wins for the active turn, and * `whenIdle()` resolves after cancellation reaches quiescence. Idle - * cancellation is a no-op and does not arm later work. The active turn - * snapshots and freezes the required cause. + * cancellation is a no-op and does not arm later work. * @param cause - the stable caller intent carried by the current turn signal. * @param options - cancellation options; `keepInbox` preserves pending work. */ @@ -515,49 +505,68 @@ abstract class Agent { * `next-turn`/wakeup preset of {@link send}. The item becomes the sole * ordinary message of its own turn. * @param content - the prompt content blocks. - * @param options - source and attached contexts. + * @param options - message source. * @returns the accepted message's {@link AgentMessageId}. */ followup(content: ContentBlock[], options?: AliasSendOptions): AgentMessageId { - return this.send(content, { ...options, target: 'next-turn', wakeup: true }) + return this.send(content, { + target: 'next-turn', + wakeup: true, + source: options?.source ?? { kind: 'user' }, + }) } /** * Submit steering into the running turn — the `next-step`/wakeup preset of * {@link send}. An open turn records it at the next steering checkpoint before - * a request or continuation decision; policy may stop before another step. - * After turn close and its checkpoint, any remainder is queued for a later - * turn; terminal `agent/turn-stop`, cancellation, or disposal may discard it. - * Idle steering falls back to a woken follow-up turn. + * a request or stop decision. If the turn fails before that boundary, the + * remainder stays staged without waking the agent; retry or a later prompt + * takes it. Idle steering falls back to a woken follow-up turn, while + * cancellation or disposal may discard pending steering. * @param content - the steering content blocks. - * @param options - source and attached contexts. + * @param options - message source. * @returns the accepted message's {@link AgentMessageId}. */ steer(content: ContentBlock[], options?: AliasSendOptions): AgentMessageId { - return this.send(content, { ...options, target: 'next-step', wakeup: true }) + return this.send(content, { + target: 'next-step', + wakeup: true, + source: options?.source ?? { kind: 'user' }, + }) } /** - * Append detached model-facing context without running the model — the - * `next-step`/no-wakeup preset of {@link send}. An open-turn injection joins - * at the current log position unless the current tool batch is executing; - * then it waits FIFO until that batch settles and drains before turn close - * even when interrupted. Idle injection uses a one-shot turn and durability - * checkpoint. Disposal awaits idle checkpoints; flush failures report through - * `agent/error`. An omitted source defaults to `{ kind: 'plugin', plugin: '' }`. + * Append model-facing context without running the model — the + * `next-step`/no-wakeup preset of {@link send}. An open-turn injection stages + * at the next safe log position; an idle injection appends immediately + * without opening a turn. An omitted source defaults to + * `{ kind: 'plugin', plugin: '' }`. * @param content - the injected context content blocks. - * @param options - source and attached contexts. + * @param options - context source. * @returns the accepted message's {@link AgentMessageId}. */ inject(content: ContentBlock[], options?: AliasSendOptions): AgentMessageId { - return this.send(content, { ...options, target: 'next-step', wakeup: false }) + return this.send(content, { + target: 'next-step', + wakeup: false, + source: options?.source ?? { kind: 'plugin', plugin: '' }, + }) } + + /** + * Re-open a turn on the current session log without a new prompt — the + * explicit resummon verb. During `agent/request-error`, this schedules one + * retry turn after the failed turn closes; while idle, it starts one + * immediately. Repeated calls before the scheduled retry coalesce. + * @throws while other agent work is running. + */ + abstract retry(): void } ``` -`AgentStatus` is `'idle' | 'running' | 'disposed'`, and `SessionId` is branded. `running` describes the driver-wide drain interval, which can span turn close, its durability checkpoint, and consecutive queued turns; it does not prove a turn is still open. `AgentOptions` is merge-extensible: core declares `provider?` and `model?` (dispatch requires both after `agent/request`). Persona belongs to `dsh-system-prompt`: an agent-scoped `deployment:persona` may shadow the global default. +`AgentStatus` is `'idle' | 'running'`, and `SessionId` is branded. Disposal removes the agent from the registry and emits `agent/disposed`; it is not a terminal status value. `running` describes the driver-wide drain interval and may span consecutive queued turns; it does not prove a turn is still open. `AgentOptions` is merge-extensible: core declares `provider?` and `model?` (dispatch requires both after `agent/request`). Persona belongs to `dsh-system-prompt`: an agent-scoped `deployment:persona` may shadow the global default. -The cause is a TypeScript-enforced same-process input. An active holder copies its discriminant into the runtime-only `AbortSignal.reason`; it is retired before `turn/end` publication. `agentInterruptReasonOf(signal)` recognizes `user`, `parent`, and lifecycle-only `disposed` without consulting ambient initiator state. Durable `turn/end` retains the coarse `{ kind: 'aborted' }` outcome; request provenance would require a separate durable event rather than overloading the terminal result. +The cause is a required, TypeScript-enforced same-process input. An active holder copies its discriminant into the runtime-only `AbortSignal.reason`. `agentInterruptReasonOf(signal)` recognizes `user`, `parent`, and lifecycle-only `disposed` without consulting ambient initiator state. Durable `turn/end` uses `{ kind: 'aborted' }` for user or parent cancellation and `{ kind: 'disposed' }` for lifecycle teardown. The [event taxonomy](../architecture.md#event) owns the `agent/*` lifecycle, checkpoint, and waterfall contracts. Turn and step boundaries are durable session events rather than agent emits. @@ -567,7 +576,7 @@ The process-local initiator carried by `ctx.agents` is the exact `Agent` above, ## Interception decisions -Each `agent/*` interception waterfall returns a small, seam-specific typed union — the unified Decision idiom (the tool seams' `PreToolDecision`/`PostToolDecision` in [tools.md](tools.md) follow the same shape). A CC/Codex hook bridge maps its `permissionDecision`/`decision`/`continue`/`additionalContext` fields onto these; a native plugin returns them directly. Prompt and post-tool decisions share `AdditionalContext`, the same `UserMessageData` content/source shape used by durable user-role input. Each `additionalContexts` entry becomes a separate injected `user/message`, preserving its provenance. Continuation reasons are steering messages and use the same content/source base. +Prompt and post-tool decisions share `AdditionalContext`, the same `UserMessageData` content/source shape used by durable user-role input. Each `additionalContexts` entry becomes a separate `user/message`, preserving its provenance. Hook bridges map their native decision fields onto these typed results. Source: [`packages/core/agent/src/types.ts`](../../packages/core/agent/src/types.ts) @@ -576,7 +585,7 @@ Source: [`packages/core/agent/src/types.ts`](../../packages/core/agent/src/types type AdditionalContext = UserMessageData ``` -`agent/prompt-submit` returns a `PromptDecision` (allow the turn's claimed queued message — optionally rewriting its `content` or attaching `additionalContexts` — or record `prompt/blocked` and end that zero-step turn as `rejected`): +`agent/prompt-submit` returns a `PromptDecision` before a turn opens. Allow may rewrite the claimed prompt or attach `additionalContexts`; block rejects admission without creating turn events: ```ts type-equiv /** @@ -590,41 +599,14 @@ type PromptDecision = | { kind: 'block'; reason: string } ``` -`agent/turn-continuation` returns a `ContinuationDecision` (the loop's default is `continue` when the step had tool calls or steering was injected, else `stop`; a `continue` `reason` is recorded as next-step steering in the same turn and therefore carries no attached contexts — the typed `/goal` pattern): - -```ts type-equiv -/** Turn continuation override; a continue reason is recorded as next-step steering in the same turn. */ -type ContinuationDecision = - | { action: 'stop' } - | { action: 'continue'; reason?: { content: ContentBlock[]; source: MessageSource } } -``` - -`agent/request-error` receives the exact original `RequestError` beside its immutable `LlmFailure`, an immutable list of failures that already authorized another request in the consecutive sequence, the turn signal, and `next()`. Recovery plugins route on `failure.code`, not the live error's message; each policy counts only its own codes, and a successful request clears the history: +`agent/request-error` runs after a failed model step closes and before its turn closes. Listeners can repair durable state or await policy work while the failed turn's signal is still live. A handling listener calls `agent.retry()` and returns without `next()`; repeated calls coalesce into one retry turn. ```ts type-equiv /** Model-request failure with an optional machine-routable provider code. */ type RequestError = Error & { code?: string } ``` -It returns a `RequestErrorDecision`; `retry` opens a new numbered step after the recovery listener's durable mutation, while `fail` retains the structured failure on `turn/end`: - -```ts type-equiv -/** Failed-request recovery decision; `retry` opens another numbered step while listeners delegate by calling `next()`. */ -type RequestErrorDecision = { action: 'fail' } | { action: 'retry' } -``` - -`agent/post-step` is awaited after assistant output, real or synthetic tool results, buffered context, and steering are durable but before `step/end`. A cancelled tool batch reaches it with an aborted signal after draining; its signature is `(agent, turn, step, signal)`, and replayable facts remain in the session log rather than a transient payload. - -`agent/turn-stop` returns the stop-only `ContinuationStop` subset or `undefined`. The loop calls this serial checkpoint after folding the ordinary decision, its reason, and pending steering; a stop is terminal and discards pending steering. - -```ts type-equiv -/** - * The terminal subset of {@link ContinuationDecision}. A listener on - * `agent/turn-stop` returns this to make the already-composed continuation - * outcome terminal; `undefined` abstains. - */ -type ContinuationStop = Extract -``` +`agent/step` is the single serial boundary before request derivation. `agent/stopping` runs when a turn has no tool or steering continuation, before one final steering drain. `agent/session-start` carries a `SessionStartSource` (why the session lifecycle began; a bridge keys its SessionStart matcher on it): @@ -633,8 +615,6 @@ type ContinuationStop = Extract type SessionStartSource = 'startup' | 'resume' | 'clear' | 'compact' ``` -`agent/session-prefix` composes a `Message[]` once per loop instance. The deep-frozen result is recorded in the request header and prepended to every derived history, making it the home for session-stable openers. A resumed instance recomposes; mid-session changes use append-only context channels. The waterfall returns content directly because it contributes rather than decides. - ## `ToolDefinition` The one pipeline-authoring type that is core: what every registered tool *is* — a model-facing `ToolSchema` plus an `execute` function and optional final-content and UI callbacks. A tool author rarely constructs it by hand (the `defineTool` DSL builds it with typed args), but it is the contract the registry holds and the loop dispatches through. diff --git a/docs/core-data-structures/core.zh.md b/docs/core-data-structures/core.zh.md index 2048a59451..e686216530 100644 --- a/docs/core-data-structures/core.zh.md +++ b/docs/core-data-structures/core.zh.md @@ -456,13 +456,7 @@ type AgentCancelCause = `Agent` 是抽象类:具体驱动器实现抽象成员,而 `followup`/`steer`/`inject` 是共享的具体委托方法,它们都委托给覆盖(`target` × `wakeup`)矩阵的唯一抽象 `send`。 ```ts type-equiv -/** - * Public agent handle; its concrete implementation is internal to - * `@deepseek-ai/dsh-agent-loop`. An abstract class rather than an interface so - * the fixed-preset aliases ({@link Agent.followup}, {@link Agent.steer}, - * {@link Agent.inject}) are shared concrete delegates over the single abstract - * {@link Agent.send} primitive; concrete drivers implement `send` once. - */ +/** Public live-agent handle with aliases over the unified delivery primitive. */ abstract class Agent { /** The single identity shared with {@link session}. */ abstract readonly id: SessionId @@ -477,7 +471,7 @@ abstract class Agent { /** * The unified delivery primitive over the (`target` × `wakeup`) matrix. - * Detaches, validates, and freezes one lossless-JSON item, then routes it: + * It routes the caller's typed content and source as follows: * * - `next-turn` queues an item that becomes the sole ordinary message of its * own FIFO-ordered turn; `wakeup:true` wakes a @@ -485,12 +479,9 @@ abstract class Agent { * - `next-step` with `wakeup:true` submits steering into the active turn * (idle falls back to a woken `next-turn`). * - `next-step` with `wakeup:false` injects durable model-facing context - * without running the model: an open turn joins at the current log position - * (deferred behind an executing tool batch until it settles), and an idle - * inject records a one-shot turn with its own durability checkpoint. - * - * Attached contexts share the same snapshot and ownership boundary. Invalid - * input throws synchronously before any notification, enqueue, or append. + * without running the model: an open turn stages it for the next safe log + * position, while an idle injection appends it immediately without opening + * a turn. * @param content - the model-facing content blocks to deliver. * @param options - target queue, wakeup decision, and source. * @returns the accepted message's {@link AgentMessageId}, stable across its `agent/inbox/*` events. @@ -502,8 +493,7 @@ abstract class Agent { * turn. An effective call first emits `agent/cancel-requested` with the * resolved typed cause. The first cause wins for the active turn, and * `whenIdle()` resolves after cancellation reaches quiescence. Idle - * cancellation is a no-op and does not arm later work. The active turn - * snapshots and freezes the required cause. + * cancellation is a no-op and does not arm later work. * @param cause - the stable caller intent carried by the current turn signal. * @param options - cancellation options; `keepInbox` preserves pending work. */ @@ -517,49 +507,68 @@ abstract class Agent { * `next-turn`/wakeup preset of {@link send}. The item becomes the sole * ordinary message of its own turn. * @param content - the prompt content blocks. - * @param options - source and attached contexts. + * @param options - message source. * @returns the accepted message's {@link AgentMessageId}. */ followup(content: ContentBlock[], options?: AliasSendOptions): AgentMessageId { - return this.send(content, { ...options, target: 'next-turn', wakeup: true }) + return this.send(content, { + target: 'next-turn', + wakeup: true, + source: options?.source ?? { kind: 'user' }, + }) } /** * Submit steering into the running turn — the `next-step`/wakeup preset of * {@link send}. An open turn records it at the next steering checkpoint before - * a request or continuation decision; policy may stop before another step. - * After turn close and its checkpoint, any remainder is queued for a later - * turn; terminal `agent/turn-stop`, cancellation, or disposal may discard it. - * Idle steering falls back to a woken follow-up turn. + * a request or stop decision. If the turn fails before that boundary, the + * remainder stays staged without waking the agent; retry or a later prompt + * takes it. Idle steering falls back to a woken follow-up turn, while + * cancellation or disposal may discard pending steering. * @param content - the steering content blocks. - * @param options - source and attached contexts. + * @param options - message source. * @returns the accepted message's {@link AgentMessageId}. */ steer(content: ContentBlock[], options?: AliasSendOptions): AgentMessageId { - return this.send(content, { ...options, target: 'next-step', wakeup: true }) + return this.send(content, { + target: 'next-step', + wakeup: true, + source: options?.source ?? { kind: 'user' }, + }) } /** - * Append detached model-facing context without running the model — the - * `next-step`/no-wakeup preset of {@link send}. An open-turn injection joins - * at the current log position unless the current tool batch is executing; - * then it waits FIFO until that batch settles and drains before turn close - * even when interrupted. Idle injection uses a one-shot turn and durability - * checkpoint. Disposal awaits idle checkpoints; flush failures report through - * `agent/error`. An omitted source defaults to `{ kind: 'plugin', plugin: '' }`. + * Append model-facing context without running the model — the + * `next-step`/no-wakeup preset of {@link send}. An open-turn injection stages + * at the next safe log position; an idle injection appends immediately + * without opening a turn. An omitted source defaults to + * `{ kind: 'plugin', plugin: '' }`. * @param content - the injected context content blocks. - * @param options - source and attached contexts. + * @param options - context source. * @returns the accepted message's {@link AgentMessageId}. */ inject(content: ContentBlock[], options?: AliasSendOptions): AgentMessageId { - return this.send(content, { ...options, target: 'next-step', wakeup: false }) + return this.send(content, { + target: 'next-step', + wakeup: false, + source: options?.source ?? { kind: 'plugin', plugin: '' }, + }) } + + /** + * Re-open a turn on the current session log without a new prompt — the + * explicit resummon verb. During `agent/request-error`, this schedules one + * retry turn after the failed turn closes; while idle, it starts one + * immediately. Repeated calls before the scheduled retry coalesce. + * @throws while other agent work is running. + */ + abstract retry(): void } ``` -`AgentStatus` 为 `'idle' | 'running' | 'disposed'`,`SessionId` 是品牌类型。`running` 描述整个驱动器的排空区间,可能跨越轮次关闭、其持久化检查点以及连续的排队轮次;它不能证明某个轮次仍然打开。`AgentOptions` 可合并扩展:core 声明 `provider?` 与 `model?`(在 `agent/request` 后,分发要求两者都存在)。Persona 归 `dsh-system-prompt` 所有:agent 作用域的 `deployment:persona` 可以遮蔽全局默认值。 +`AgentStatus` 为 `'idle' | 'running'`,`SessionId` 是品牌类型。dispose 会从注册表中移除 agent 并发出 `agent/disposed`;它不是终态状态值。`running` 描述整个驱动器的排空区间,可能跨越连续的排队轮次;它不能证明某个轮次仍然打开。`AgentOptions` 可合并扩展:core 声明 `provider?` 与 `model?`(在 `agent/request` 后,分发要求两者都存在)。Persona 归 `dsh-system-prompt` 所有:agent 作用域的 `deployment:persona` 可以遮蔽全局默认值。 -cause 是由 TypeScript 强制约束的同进程输入。活跃的 `TurnCancellation` 持有者会把其判别字段复制到仅运行时的 `AbortSignal.reason`,并在发布 `turn/end` 前退役;冻结后的 `AbortSignal.reason` 仍可读取。`agentInterruptReasonOf(signal)` 无需查询环境中的 initiator 状态,即可识别 `user`、`parent` 与仅用于生命周期的 `disposed`。持久 `turn/end` 保留粗粒度 `{ kind: 'aborted' }` 结果;若需记录请求 provenance,应使用单独的持久事件,而不是让终态结果承担额外含义。 +cause 是必选且由 TypeScript 强制约束的同进程输入。活跃持有者会把其判别字段复制到仅运行时的 `AbortSignal.reason`。`agentInterruptReasonOf(signal)` 无需查询环境中的 initiator 状态,即可识别 `user`、`parent` 与仅用于生命周期的 `disposed`。持久 `turn/end` 对用户或父级取消使用 `{ kind: 'aborted' }`,对生命周期拆卸使用 `{ kind: 'disposed' }`。 [事件分类](../architecture.md#event)拥有 `agent/*` 生命周期、检查点与 waterfall(瀑布式事件)契约。轮次和步骤边界是持久会话事件,而不是 agent emit。 @@ -569,7 +578,7 @@ cause 是由 TypeScript 强制约束的同进程输入。活跃的 `TurnCancella ## 拦截决策 -每个 `agent/*` 拦截 waterfall 都返回一个小型、特定于 seam 的类型化联合——统一的 Decision 惯用形状([tools.md](tools.md) 中工具 seam 的 `PreToolDecision`/`PostToolDecision` 也采用相同形状)。CC/Codex 钩子桥接层把其 `permissionDecision`/`decision`/`continue`/`additionalContext` 字段映射到这些联合上;原生插件则直接返回它们。提示词决策与工具后决策共享 `AdditionalContext`,它与持久用户角色输入使用相同的 `UserMessageData` content/source 形状。每个 `additionalContexts` 项都会成为一条单独注入的 `user/message`,并保留其 provenance。Continuation reason 则是 steering 消息,并使用同一个 content/source 基础类型。 +提示词决策与工具后决策共享 `AdditionalContext`,它与持久用户角色输入使用相同的 `UserMessageData` content/source 形状。每个 `additionalContexts` 项都会成为一条单独的 `user/message`,并保留其来源信息。钩子桥接层会把原生决策字段映射到这些类型化结果。 源码:[`packages/core/agent/src/types.ts`](../../packages/core/agent/src/types.ts) @@ -578,7 +587,7 @@ cause 是由 TypeScript 强制约束的同进程输入。活跃的 `TurnCancella type AdditionalContext = UserMessageData ``` -`agent/prompt-submit` 返回 `PromptDecision`(允许该轮次已领取的排队消息——可选地改写其 `content` 或附加 `additionalContexts`——或者记录 `prompt/blocked` 并以 `rejected` 结束这个零步骤轮次): +`agent/prompt-submit` 在轮次打开前返回 `PromptDecision`。`allow` 可以改写已领取的提示词或附加 `additionalContexts`;`block` 会拒绝接纳,且不创建轮次事件: ```ts type-equiv /** @@ -592,41 +601,14 @@ type PromptDecision = | { kind: 'block'; reason: string } ``` -`agent/turn-continuation` 返回 `ContinuationDecision`(步骤有工具调用或注入了 steering 时,循环默认为 `continue`,否则为 `stop`;`continue` 的 `reason` 会记录为同一轮次中下一个步骤的 steering,因此不携带上下文元数据——即类型化 `/goal` 模式): - -```ts type-equiv -/** Turn continuation override; a continue reason is recorded as next-step steering in the same turn. */ -type ContinuationDecision = - | { action: 'stop' } - | { action: 'continue'; reason?: { content: ContentBlock[]; source: MessageSource } } -``` - -`agent/request-error` 接收确切的原始 `RequestError`、其不可变 `LlmFailure`、在连续序列中已批准另一次请求的不可变失败列表、轮次信号以及 `next()`。恢复插件按 `failure.code` 路由,而不是按活跃错误的消息路由;每项策略只统计自身的 code,一次成功请求会清空历史: +`agent/request-error` 会在失败的模型步骤关闭后、其轮次关闭前运行。监听器可以在失败轮次的信号仍然有效时修复持久状态或等待策略工作。负责处理的监听器会调用 `agent.retry()` 且不调用 `next()`;重复调用会合并为一个重试轮次。 ```ts type-equiv /** Model-request failure with an optional machine-routable provider code. */ type RequestError = Error & { code?: string } ``` -它返回 `RequestErrorDecision`;`retry` 在恢复 listener 的持久变更之后打开一个带新编号的步骤,而 `fail` 在 `turn/end` 上保留结构化失败: - -```ts type-equiv -/** Failed-request recovery decision; `retry` opens another numbered step while listeners delegate by calling `next()`. */ -type RequestErrorDecision = { action: 'fail' } | { action: 'retry' } -``` - -`agent/post-step` 会在 assistant 输出、真实或合成的工具结果、缓冲上下文与 steering 持久化之后、`step/end` 之前被 await。被取消的工具批次在排空后携带 aborted signal 到达这里;其签名为 `(agent, turn, step, signal)`,可回放事实保留在会话日志中,而不是瞬态 payload 中。 - -`agent/turn-stop` 返回仅停止的 `ContinuationStop` 子集或 `undefined`。循环在折叠普通决策、其 reason 和待处理 steering 之后调用此串行检查点;stop 是终态,会丢弃待处理的 steering。 - -```ts type-equiv -/** - * The terminal subset of {@link ContinuationDecision}. A listener on - * `agent/turn-stop` returns this to make the already-composed continuation - * outcome terminal; `undefined` abstains. - */ -type ContinuationStop = Extract -``` +`agent/step` 是派生请求之前唯一的串行边界。当轮次不再因工具或 steering 继续时,`agent/stopping` 会在最后一次排空 steering 之前运行。 `agent/session-start` 携带 `SessionStartSource`(会话生命周期为何开始;桥接层据此匹配其 SessionStart): @@ -635,8 +617,6 @@ type ContinuationStop = Extract type SessionStartSource = 'startup' | 'resume' | 'clear' | 'compact' ``` -`agent/session-prefix` 在每个循环实例中组合一次 `Message[]`。深度冻结的结果被记录在请求 header 中,并前置于每次派生历史,使其成为会话稳定开场白的归属。恢复的实例会重新组合;会话中途的变更使用仅追加的上下文通道。该 waterfall 直接返回内容,因为它是贡献而非决策。 - ## `ToolDefinition` 唯一属于核心的流水线编写类型:每个已注册工具*是什么*——一个面向模型的 `ToolSchema` 加上一个 `execute` 函数,以及可选的最终内容回调与 UI 回调。工具作者很少手动构造它(`defineTool` DSL 会用类型化参数构建),但它是注册表持有、循环分发所经过的契约。 diff --git a/docs/core-data-structures/goal.md b/docs/core-data-structures/goal.md index c0351f2f77..cb8751286f 100644 --- a/docs/core-data-structures/goal.md +++ b/docs/core-data-structures/goal.md @@ -105,6 +105,8 @@ interface GoalMessageSource { readonly revision: number /** Zero for state changes; positive for admitted continuation rounds. */ readonly round: number + /** Complete durable mutation carried only by round-zero state-change messages. */ + readonly change?: GoalChangeMeta } ``` diff --git a/docs/core-data-structures/llm-streaming.md b/docs/core-data-structures/llm-streaming.md index 257ce90cda..dbd907d335 100644 --- a/docs/core-data-structures/llm-streaming.md +++ b/docs/core-data-structures/llm-streaming.md @@ -57,7 +57,7 @@ Every adapter MUST obey these, and every consumer may rely on them: - **`usage` before `finish`, nothing after `finish`.** Defer both to the provider's end-of-stream marker so a trailing usage-only chunk can't violate the ordering. - **Tool-call `arguments` stay raw JSON strings end-to-end.** Partial fragments stream via `argumentsDelta`; a provider that hands back parsed objects re-stringifies at `block-end`. -- **Two sanctioned error paths, one fact shape.** A failure may either THROW from `stream()` (transport/protocol errors) **or** end the stream with `finish {kind:'error'|'aborted', failure}` (provider in-band errors, for adapters that can't throw mid-stream). `LlmError.failure` carries the same `LlmFailure`. The final adapter boundary preserves the exact thrown `Error` object and associates immutable facts with that call; the agent loop closes the failed step and offers the error, facts, and immutable prior-retried facts to `agent/request-error`. Absent recovery the structured failure becomes the turn error, and no normal assistant message or tool side effect is committed for that attempt. +- **Two sanctioned error paths, one fact shape.** A failure may either THROW from `stream()` (transport/protocol errors) **or** end the stream with `finish {kind:'error'|'aborted', failure}` (provider in-band errors, for adapters that can't throw mid-stream). `LlmError.failure` carries the same `LlmFailure`. The final adapter boundary preserves the exact thrown `Error` object and associates immutable facts with that call; the agent loop closes the failed step and offers the error and facts to `agent/request-error`. A handling listener calls `agent.retry()` after its awaited repair; absent recovery the structured failure becomes the turn error, and no normal assistant message or tool side effect is committed for that attempt. - **One adapter call is one provider attempt.** Adapters disable library retries. Agent-level recovery opens another durable numbered step; direct `ctx.llm.stream()` callers remain single-attempt. - **Provider stalls are bounded at the transport.** Both shipping remote adapters expose positive finite `streamIdleTimeoutMs` with a five-minute default. The watchdog arms only while iterator `next()` is outstanding, uses one stable signal for the whole request, maps its own expiry to `TIMEOUT`, and keeps an earlier caller abort as `ABORTED`. - **Context overflow has one canonical code.** Both DeepSeek adapters classify explicit provider detail through `isContextWindowExceededError()` and surface `CONTEXT_WINDOW_EXCEEDED`, whether the failure arrives as a thrown HTTP `LlmError` or an in-band finish error. Consumers route on the code, never provider text. diff --git a/docs/core-data-structures/session-reference.md b/docs/core-data-structures/session-reference.md index dbe3c43d35..c33704c14f 100644 --- a/docs/core-data-structures/session-reference.md +++ b/docs/core-data-structures/session-reference.md @@ -36,15 +36,15 @@ interface SessionReferenceCandidate { ## Prepared messages -Preparation preserves readable current-message content and returns at most one aggregated context. The host binds `contexts` to that exact `send()` or `steer()` call. +Preparation preserves readable current-message content and returns at most one aggregated context. ```ts type-equiv -/** Message payload and the zero-or-one durable snapshot contexts bound to it. */ +/** Direct message content and optional referenced-session context. */ interface PreparedReferencedMessage { /** Readable message content after host mention tokens are removed. */ content: ContentBlock[] - /** Empty without references; otherwise one aggregated untrusted context. */ - contexts: HookContext[] + /** Aggregated untrusted snapshot, absent when the message has no references. */ + additionalContext?: AdditionalContext } ``` diff --git a/docs/core-data-structures/session.i18n.yaml b/docs/core-data-structures/session.i18n.yaml index aedef3c944..a998473448 100644 --- a/docs/core-data-structures/session.i18n.yaml +++ b/docs/core-data-structures/session.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write -session.md: a6bd10000158bbb148aab984789846c0177747f4 -session.zh.md: 4532e2f46006cc637e39ad0dc1dc239d284a3ec8 +session.md: cd646d1d5913b0850bb9e1596a5f5f9795d74e4b +session.zh.md: cf29447ecfd2383193a5185fbe78891253d48be0 diff --git a/docs/core-data-structures/session.md b/docs/core-data-structures/session.md index a6bd100001..cd646d1d59 100644 --- a/docs/core-data-structures/session.md +++ b/docs/core-data-structures/session.md @@ -480,14 +480,12 @@ An explicit `boundary` lets callers fork from a previous completed turn even if */ interface TurnTriggerMap { message: { kind: 'message'; source: MessageSource } + /** Recovery turn reopened over the repaired current session log. */ + retry: { kind: 'retry' } /** - * An out-of-band context injection (`agent.inject()`) made while the agent - * was idle. The loop wraps the injected `user/message` (a non-`user` source, - * plugin by default) in a one-shot turn (`turn/start` → `user/message` → - * `turn/end`) so every event in the log stays turn-enclosed — the - * durability/replay boundary is the turn, and a bare event between turns would - * otherwise be indistinguishable from a crash tail on reload. The trigger's - * `source` mirrors that message's producer. + * An out-of-band producer explicitly enclosed injected context in a one-shot + * turn. `Agent.inject()` appends idle context directly and does not use this + * trigger; the source mirrors the producer of the enclosed `user/message`. */ injection: { kind: 'injection'; source: MessageSource } } diff --git a/docs/core-data-structures/session.zh.md b/docs/core-data-structures/session.zh.md index 4532e2f460..cf29447ecf 100644 --- a/docs/core-data-structures/session.zh.md +++ b/docs/core-data-structures/session.zh.md @@ -480,14 +480,12 @@ declare class Session { */ interface TurnTriggerMap { message: { kind: 'message'; source: MessageSource } + /** Recovery turn reopened over the repaired current session log. */ + retry: { kind: 'retry' } /** - * An out-of-band context injection (`agent.inject()`) made while the agent - * was idle. The loop wraps the injected `user/message` (a non-`user` source, - * plugin by default) in a one-shot turn (`turn/start` → `user/message` → - * `turn/end`) so every event in the log stays turn-enclosed — the - * durability/replay boundary is the turn, and a bare event between turns would - * otherwise be indistinguishable from a crash tail on reload. The trigger's - * `source` mirrors that message's producer. + * An out-of-band producer explicitly enclosed injected context in a one-shot + * turn. `Agent.inject()` appends idle context directly and does not use this + * trigger; the source mirrors the producer of the enclosed `user/message`. */ injection: { kind: 'injection'; source: MessageSource } } diff --git a/docs/core-data-structures/tools.md b/docs/core-data-structures/tools.md index d68e280b24..b48fe1ef40 100644 --- a/docs/core-data-structures/tools.md +++ b/docs/core-data-structures/tools.md @@ -213,7 +213,9 @@ interface ToolRunContext extends ToolExecution { * the agent loop. Contexts retain their individual source and metadata and * are emitted in call order. */ - deferContext(context: HookContext): void + deferContext(context: AdditionalContext): void + /** Mark a successful final result as terminal for the current agent turn. */ + concludeTurn(): void } ``` @@ -290,7 +292,9 @@ interface ToolExecutionSuccess { readonly content: ContentBlock[] readonly error?: never readonly meta?: JsonValue - readonly additionalContexts?: HookContext[] + readonly additionalContexts?: AdditionalContext[] + /** The agent loop stops after committing this successful result batch. */ + readonly concludesTurn?: true } ``` @@ -302,7 +306,8 @@ interface ToolExecutionFailure { readonly value?: never readonly content: ContentBlock[] readonly meta?: JsonValue - readonly additionalContexts?: HookContext[] + readonly additionalContexts?: AdditionalContext[] + readonly concludesTurn?: never } ``` @@ -338,9 +343,9 @@ type PreToolDecision = * next request, or block by turning corrective feedback into an error result. */ type PostToolDecision = - | { kind: 'accept'; content?: ContentBlock[]; value?: never; additionalContexts?: HookContext[] } - | { kind: 'accept'; value: JsonValue; content?: never; additionalContexts?: HookContext[] } - | { kind: 'block'; feedback: ContentBlock[]; additionalContexts?: HookContext[] } + | { kind: 'accept'; content?: ContentBlock[]; value?: never; additionalContexts?: AdditionalContext[] } + | { kind: 'accept'; value: JsonValue; content?: never; additionalContexts?: AdditionalContext[] } + | { kind: 'block'; feedback: ContentBlock[]; additionalContexts?: AdditionalContext[] } ``` Call `next()` for the default or return a decision to short-circuit. Pre-policy may deny or ask; only `allowed-once` proceeds, while a non-grant, missing approval channel or service, or agent-less request becomes a denial. Guards may still impose a final denial. Arguments cannot be rewritten because history, audit, UI, and execution must agree. diff --git a/docs/event-producer-consumer.md b/docs/event-producer-consumer.md index 94b204ad43..adb9ef56d6 100644 --- a/docs/event-producer-consumer.md +++ b/docs/event-producer-consumer.md @@ -7,36 +7,33 @@ This matrix shows which packages dispatch each harness-owned event and which pac | Event | Mode | Declared in | Dispatchers | Listeners | | --- | --- | --- | --- | --- | -| `agent-loop/config-start-failed` | `emit` | [`packages/core/agent-loop/src/index.ts:353`](../packages/core/agent-loop/src/index.ts) | [`agent-loop`](../packages/core/agent-loop) (`events.dispatch`) | [`tui`](../packages/ui/tui) | -| `agent/cancel-requested` | `emit` | [`packages/core/agent/src/types.ts:359`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emit`) | [`goal-session`](../packages/goal/goal-session) | -| `agent/created` | `emit` | [`packages/core/agent/src/types.ts:295`](../packages/core/agent/src/types.ts) | [`agent`](../packages/core/agent) (`events.dispatch`) | [`goal-session`](../packages/goal/goal-session), [`tui`](../packages/ui/tui) | -| `agent/disposed` | `emit` | [`packages/core/agent/src/types.ts:304`](../packages/core/agent/src/types.ts) | [`agent`](../packages/core/agent) (`events.dispatch`) | [`agent-loop`](../packages/core/agent-loop), [`goal-session`](../packages/goal/goal-session), [`tui`](../packages/ui/tui) | -| `agent/error` | `emit` | [`packages/core/agent/src/types.ts:507`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emit`) | [`goal-session`](../packages/goal/goal-session), `runtime`, [`tui`](../packages/ui/tui) | -| `agent/inbox/dequeue` | `emit` | [`packages/core/agent/src/types.ts:335`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emit`) | [`agent`](../packages/core/agent) | -| `agent/inbox/discard` | `emit` | [`packages/core/agent/src/types.ts:349`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emit`) | [`agent`](../packages/core/agent) | -| `agent/inbox/enqueue` | `emit` | [`packages/core/agent/src/types.ts:325`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emit`) | [`agent`](../packages/core/agent), [`goal-session`](../packages/goal/goal-session), [`tui`](../packages/ui/tui) | -| `agent/post-step` | `serial` | [`packages/core/agent/src/types.ts:457`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`serial`) | [`compact-basic`](../packages/compact/compact-basic), [`session-checkpoint-policy`](../packages/session-persistence/session-checkpoint-policy) | -| `agent/pre-step` | `serial` | [`packages/core/agent/src/types.ts:388`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`serial`) | [`time-context`](../packages/context/time-context), [`user-approval`](../packages/ui/user-approval) | -| `agent/prompt-submit` | `waterfall` | [`packages/core/agent/src/types.ts:404`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`waterfall`) | [`acp`](../packages/ui/acp), [`goal-session`](../packages/goal/goal-session), [`hooks-claude`](../packages/hooks/hooks-claude), [`hooks-codex`](../packages/hooks/hooks-codex), [`plan-mode`](../packages/plan/plan-mode), [`repeat-tool-guard`](../packages/guard/repeat-tool-guard) | -| `agent/request` | `waterfall` | [`packages/core/agent/src/types.ts:418`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`waterfall`) | [`agent`](../packages/core/agent) | -| `agent/request-error` | `waterfall` | [`packages/core/agent/src/types.ts:472`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`waterfall`) | [`compact-basic`](../packages/compact/compact-basic), [`llm-retry`](../packages/llm/llm-retry), [`plan-mode`](../packages/plan/plan-mode) | -| `agent/session-prefix` | `waterfall` | [`packages/core/agent/src/types.ts:433`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`waterfall`) | [`tool-skill`](../packages/skill/tool-skill), [`workspace-context`](../packages/context/workspace-context) | -| `agent/session-start` | `emit` | [`packages/core/agent/src/types.ts:372`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emit`) | [`goal`](../packages/goal/goal), [`goal-session`](../packages/goal/goal-session), [`hooks-claude`](../packages/hooks/hooks-claude), [`hooks-codex`](../packages/hooks/hooks-codex) | -| `agent/status` | `emit` | [`packages/core/agent/src/types.ts:313`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emit`) | [`agent`](../packages/core/agent), [`goal-session`](../packages/goal/goal-session), `runtime`, [`tui`](../packages/ui/tui) | -| `agent/step-result` | `waterfall` | [`packages/core/agent/src/types.ts:445`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`waterfall`) | - | -| `agent/turn-continuation` | `waterfall` | [`packages/core/agent/src/types.ts:483`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`waterfall`) | [`hooks-claude`](../packages/hooks/hooks-claude), [`hooks-codex`](../packages/hooks/hooks-codex), [`plan-mode`](../packages/plan/plan-mode) | -| `agent/turn-stop` | `serial` | [`packages/core/agent/src/types.ts:494`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`serial`) | [`subagent-inprocess`](../packages/subagent/subagent-inprocess), [`tool-goal`](../packages/goal/tool-goal) | +| `agent-loop/config-start-failed` | `emit` | [`packages/core/agent-loop/src/index.ts:140`](../packages/core/agent-loop/src/index.ts) | [`agent-loop`](../packages/core/agent-loop) (`events.dispatch`) | [`tui`](../packages/ui/tui) | +| `agent/cancel-requested` | `emit` | [`packages/core/agent/src/types.ts:328`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | [`goal-session`](../packages/goal/goal-session) | +| `agent/created` | `emit` | [`packages/core/agent/src/types.ts:270`](../packages/core/agent/src/types.ts) | [`agent`](../packages/core/agent) (`events.dispatch`) | [`goal-session`](../packages/goal/goal-session), [`tui`](../packages/ui/tui) | +| `agent/disposed` | `emit` | [`packages/core/agent/src/types.ts:279`](../packages/core/agent/src/types.ts) | [`agent`](../packages/core/agent) (`events.dispatch`) | [`agent-loop`](../packages/core/agent-loop), [`goal-session`](../packages/goal/goal-session), [`tui`](../packages/ui/tui) | +| `agent/error` | `emit` | [`packages/core/agent/src/types.ts:436`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | [`goal-session`](../packages/goal/goal-session), `runtime`, [`tui`](../packages/ui/tui) | +| `agent/idle` | `emit` | [`packages/core/agent/src/types.ts:423`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | [`compact-basic`](../packages/compact/compact-basic), [`llm-retry`](../packages/llm/llm-retry) | +| `agent/inbox/dequeue` | `emit` | [`packages/core/agent/src/types.ts:306`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | [`agent`](../packages/core/agent) | +| `agent/inbox/discard` | `emit` | [`packages/core/agent/src/types.ts:318`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | [`agent`](../packages/core/agent) | +| `agent/inbox/enqueue` | `emit` | [`packages/core/agent/src/types.ts:296`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | [`agent`](../packages/core/agent), [`goal-session`](../packages/goal/goal-session) | +| `agent/prompt-submit` | `waterfall` | [`packages/core/agent/src/types.ts:355`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`waterfall`) | [`goal-session`](../packages/goal/goal-session), [`hooks-claude`](../packages/hooks/hooks-claude), [`hooks-codex`](../packages/hooks/hooks-codex), [`plan-mode`](../packages/plan/plan-mode), [`repeat-tool-guard`](../packages/guard/repeat-tool-guard) | +| `agent/request` | `waterfall` | [`packages/core/agent/src/types.ts:381`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`waterfall`) | [`agent`](../packages/core/agent) | +| `agent/request-error` | `waterfall` | [`packages/core/agent/src/types.ts:396`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`waterfall`) | [`compact-basic`](../packages/compact/compact-basic), [`llm-retry`](../packages/llm/llm-retry) | +| `agent/session-start` | `emit` | [`packages/core/agent/src/types.ts:341`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | [`goal`](../packages/goal/goal), [`goal-session`](../packages/goal/goal-session), [`hooks-claude`](../packages/hooks/hooks-claude), [`hooks-codex`](../packages/hooks/hooks-codex) | +| `agent/status` | `emit` | [`packages/core/agent/src/types.ts:288`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | [`agent`](../packages/core/agent), [`goal-session`](../packages/goal/goal-session), `runtime`, [`tui`](../packages/ui/tui) | +| `agent/step` | `serial` | [`packages/core/agent/src/types.ts:368`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`serial`) | [`acp`](../packages/ui/acp), [`compact-basic`](../packages/compact/compact-basic), [`plan-mode`](../packages/plan/plan-mode), [`session-checkpoint-policy`](../packages/session-persistence/session-checkpoint-policy), [`time-context`](../packages/context/time-context), [`tool-skill`](../packages/skill/tool-skill), [`user-approval`](../packages/ui/user-approval), [`workspace-context`](../packages/context/workspace-context) | +| `agent/stopping` | `serial` | [`packages/core/agent/src/types.ts:411`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`serial`) | [`hooks-claude`](../packages/hooks/hooks-claude), [`hooks-codex`](../packages/hooks/hooks-codex) | | `approval/request` | `waterfall` | [`packages/ui/user-approval/src/index.ts:30`](../packages/ui/user-approval/src/index.ts) | [`user-approval`](../packages/ui/user-approval) (`waterfall`) | [`acp`](../packages/ui/acp) | | `commands/change` | `emit` | [`packages/ui/commands/src/index.ts:103`](../packages/ui/commands/src/index.ts) | [`commands`](../packages/ui/commands) (`events.dispatch`) | [`acp`](../packages/ui/acp), [`tui`](../packages/ui/tui) | | `fs/edit-intent` | `waterfall` | [`packages/fs/fs/src/index.ts:62`](../packages/fs/fs/src/index.ts) | [`tool-fs`](../packages/fs/tool-fs) (`waterfall`) | [`fs-policy`](../packages/fs/fs-policy) | | `fs/observed` | `emit` | [`packages/fs/fs/src/index.ts:71`](../packages/fs/fs/src/index.ts) | [`tool-fs`](../packages/fs/tool-fs) (`emit`) | [`fs-policy`](../packages/fs/fs-policy) | | `fs/write-intent` | `waterfall` | [`packages/fs/fs/src/index.ts:54`](../packages/fs/fs/src/index.ts) | [`tool-fs`](../packages/fs/tool-fs) (`waterfall`) | [`fs-policy`](../packages/fs/fs-policy) | -| `goal/changed` | `emit` | [`packages/goal/goal/src/types.ts:167`](../packages/goal/goal/src/types.ts) | [`goal`](../packages/goal/goal) (`emit`) | [`goal-session`](../packages/goal/goal-session) | +| `goal/changed` | `emit` | [`packages/goal/goal/src/types.ts:169`](../packages/goal/goal/src/types.ts) | [`goal`](../packages/goal/goal) (`emit`) | [`goal-session`](../packages/goal/goal-session) | | `llm/stream` | `waterfall` | [`packages/llm/llm/src/index.ts:52`](../packages/llm/llm/src/index.ts) | [`llm`](../packages/llm/llm) (`waterfall`) | [`agent-loop`](../packages/core/agent-loop), [`llm`](../packages/llm/llm), [`llm-replay`](../packages/support/llm-replay), [`session-checkpoint-policy`](../packages/session-persistence/session-checkpoint-policy), [`session-title`](../packages/session-title/session-title) | -| `session/created` | `emit` | [`packages/core/session/src/index.ts:79`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`compact`](../packages/compact/compact), [`goal`](../packages/goal/goal), [`hook-protocol`](../packages/hooks/hook-protocol), [`jsonrpc`](../packages/ui/jsonrpc), [`llm-retry`](../packages/llm/llm-retry), `runtime`, [`session`](../packages/core/session), [`session-persistence`](../packages/session-persistence/session-persistence), [`user-approval`](../packages/ui/user-approval) | -| `session/disposed` | `emit` | [`packages/core/session/src/index.ts:89`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`agent-loop`](../packages/core/agent-loop), `runtime`, [`session-persistence`](../packages/session-persistence/session-persistence), [`session-title`](../packages/session-title/session-title) | -| `session/event` | `emit` | [`packages/core/session/src/index.ts:101`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`acp`](../packages/ui/acp), [`cli-demo`](../packages/examples/cli-demo), [`compact`](../packages/compact/compact), [`goal`](../packages/goal/goal), [`goal-session`](../packages/goal/goal-session), [`hook-protocol`](../packages/hooks/hook-protocol), [`jsonrpc`](../packages/ui/jsonrpc), `runtime`, [`session`](../packages/core/session), [`session-persistence`](../packages/session-persistence/session-persistence), [`session-title`](../packages/session-title/session-title), [`token-meter`](../packages/llm/token-meter), [`tui`](../packages/ui/tui), [`user-approval`](../packages/ui/user-approval), [`workspace-context`](../packages/context/workspace-context) | -| `session/flush` | `parallel` | [`packages/core/session/src/index.ts:111`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`session-persistence`](../packages/session-persistence/session-persistence) | +| `session/created` | `emit` | [`packages/core/session/src/index.ts:70`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`compact`](../packages/compact/compact), [`goal`](../packages/goal/goal), [`hook-protocol`](../packages/hooks/hook-protocol), [`jsonrpc`](../packages/ui/jsonrpc), [`llm-retry`](../packages/llm/llm-retry), `runtime`, [`session`](../packages/core/session), [`session-persistence`](../packages/session-persistence/session-persistence), [`user-approval`](../packages/ui/user-approval) | +| `session/disposed` | `emit` | [`packages/core/session/src/index.ts:80`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`agent-loop`](../packages/core/agent-loop), `runtime`, [`session-persistence`](../packages/session-persistence/session-persistence), [`session-title`](../packages/session-title/session-title) | +| `session/event` | `emit` | [`packages/core/session/src/index.ts:92`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`acp`](../packages/ui/acp), [`cli-demo`](../packages/examples/cli-demo), [`compact`](../packages/compact/compact), [`goal`](../packages/goal/goal), [`goal-session`](../packages/goal/goal-session), [`hook-protocol`](../packages/hooks/hook-protocol), [`jsonrpc`](../packages/ui/jsonrpc), `runtime`, [`session`](../packages/core/session), [`session-persistence`](../packages/session-persistence/session-persistence), [`session-title`](../packages/session-title/session-title), [`token-meter`](../packages/llm/token-meter), [`tui`](../packages/ui/tui), [`user-approval`](../packages/ui/user-approval), [`workspace-context`](../packages/context/workspace-context) | +| `session/flush` | `parallel` | [`packages/core/session/src/index.ts:102`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`session-persistence`](../packages/session-persistence/session-persistence) | | `subagent/end` | `emit` | [`packages/subagent/subagent/src/index.ts:139`](../packages/subagent/subagent/src/index.ts) | [`subagent`](../packages/subagent/subagent) (`events.dispatch`) | [`hooks-claude`](../packages/hooks/hooks-claude), [`jsonrpc`](../packages/ui/jsonrpc), [`subagent`](../packages/subagent/subagent) | | `subagent/provider-added` | `emit` | [`packages/subagent/subagent/src/index.ts:113`](../packages/subagent/subagent/src/index.ts) | [`subagent`](../packages/subagent/subagent) (`emit`) | [`subagent`](../packages/subagent/subagent), [`tool-subagent`](../packages/subagent/tool-subagent) | | `subagent/provider-removed` | `emit` | [`packages/subagent/subagent/src/index.ts:119`](../packages/subagent/subagent/src/index.ts) | [`subagent`](../packages/subagent/subagent) (`events.dispatch`) | [`subagent`](../packages/subagent/subagent), [`tool-subagent`](../packages/subagent/tool-subagent) | diff --git a/docs/persistence-catalog.md b/docs/persistence-catalog.md index f4b3dcf4bc..77c3e33ed5 100644 --- a/docs/persistence-catalog.md +++ b/docs/persistence-catalog.md @@ -78,7 +78,7 @@ export type SessionEvent = { }[T] ``` -Sources: [`packages/core/session/src/types.ts:292`](../packages/core/session/src/types.ts) · [`packages/core/session/src/types.ts:305`](../packages/core/session/src/types.ts) · [`packages/core/session/src/types.ts:334`](../packages/core/session/src/types.ts) · [`packages/core/session/src/types.ts:366`](../packages/core/session/src/types.ts) +Sources: [`packages/core/session/src/types.ts:288`](../packages/core/session/src/types.ts) · [`packages/core/session/src/types.ts:301`](../packages/core/session/src/types.ts) · [`packages/core/session/src/types.ts:330`](../packages/core/session/src/types.ts) · [`packages/core/session/src/types.ts:362`](../packages/core/session/src/types.ts) ## Events @@ -150,7 +150,7 @@ Source: [`packages/ui/user-approval/src/index.ts:67`](../packages/ui/user-approv Types: [StreamChunk](core-data-structures/llm-streaming.md) -Source: [`packages/core/session/src/types.ts:238`](../packages/core/session/src/types.ts) +Source: [`packages/core/session/src/types.ts:234`](../packages/core/session/src/types.ts) #### `assistant/message` — surface @@ -166,7 +166,7 @@ Source: [`packages/core/session/src/types.ts:238`](../packages/core/session/src/ Types: [ContentBlock](core-data-structures/core.md) · [TokenUsage](core-data-structures/llm-streaming.md) -Source: [`packages/core/session/src/types.ts:245`](../packages/core/session/src/types.ts) +Source: [`packages/core/session/src/types.ts:241`](../packages/core/session/src/types.ts) ### `compact/*` @@ -329,7 +329,7 @@ Source: [`packages/plan/plan-mode/src/index.ts:41`](../packages/plan/plan-mode/s Types: [ContentBlock](core-data-structures/core.md) · [MessageSource](core-data-structures/core.md) -Source: [`packages/core/session/src/types.ts:236`](../packages/core/session/src/types.ts) +Source: [`packages/core/session/src/types.ts:232`](../packages/core/session/src/types.ts) ### `request/*` @@ -343,7 +343,7 @@ Source: [`packages/core/session/src/types.ts:236`](../packages/core/session/src/ 'request/header': { header: EpochHeader; reason: RequestHeaderReason } ``` -Source: [`packages/core/session/src/types.ts:280`](../packages/core/session/src/types.ts) +Source: [`packages/core/session/src/types.ts:276`](../packages/core/session/src/types.ts) ### `sandbox/*` @@ -399,7 +399,7 @@ Source: [`packages/session-title/session-title-llm/src/index.ts:44`](../packages 'steering/message': UserMessageData & { turn: number } ``` -Source: [`packages/core/session/src/types.ts:273`](../packages/core/session/src/types.ts) +Source: [`packages/core/session/src/types.ts:269`](../packages/core/session/src/types.ts) ### `step/*` @@ -410,7 +410,7 @@ Source: [`packages/core/session/src/types.ts:273`](../packages/core/session/src/ 'step/end': { turn: number; step: number } ``` -Source: [`packages/core/session/src/types.ts:222`](../packages/core/session/src/types.ts) +Source: [`packages/core/session/src/types.ts:218`](../packages/core/session/src/types.ts) #### `step/start` — log-only @@ -419,7 +419,7 @@ Source: [`packages/core/session/src/types.ts:222`](../packages/core/session/src/ 'step/start': { turn: number; step: number } ``` -Source: [`packages/core/session/src/types.ts:220`](../packages/core/session/src/types.ts) +Source: [`packages/core/session/src/types.ts:216`](../packages/core/session/src/types.ts) ### `todo/*` @@ -432,7 +432,7 @@ Source: [`packages/core/session/src/types.ts:220`](../packages/core/session/src/ Types: [TodoItem](core-data-structures/session.md) -Source: [`packages/core/session/src/types.ts:275`](../packages/core/session/src/types.ts) +Source: [`packages/core/session/src/types.ts:271`](../packages/core/session/src/types.ts) ### `tool/*` @@ -449,7 +449,7 @@ Source: [`packages/core/session/src/types.ts:275`](../packages/core/session/src/ Types: [CallId](core-data-structures/core.md) -Source: [`packages/core/session/src/types.ts:251`](../packages/core/session/src/types.ts) +Source: [`packages/core/session/src/types.ts:247`](../packages/core/session/src/types.ts) #### `tool/code-dispatch` — log-only @@ -503,7 +503,7 @@ Source: [`packages/core/tools/src/code-mode.ts:34`](../packages/core/tools/src/c Types: [CallId](core-data-structures/core.md) · [ContentBlock](core-data-structures/core.md) -Source: [`packages/core/session/src/types.ts:263`](../packages/core/session/src/types.ts) +Source: [`packages/core/session/src/types.ts:259`](../packages/core/session/src/types.ts) ### `turn/*` @@ -521,7 +521,7 @@ Source: [`packages/core/session/src/types.ts:263`](../packages/core/session/src/ Types: [TurnEndReason](core-data-structures/session.md) -Source: [`packages/core/session/src/types.ts:218`](../packages/core/session/src/types.ts) +Source: [`packages/core/session/src/types.ts:214`](../packages/core/session/src/types.ts) #### `turn/start` — log-only @@ -534,7 +534,7 @@ Source: [`packages/core/session/src/types.ts:218`](../packages/core/session/src/ Types: [TurnTrigger](core-data-structures/session.md) -Source: [`packages/core/session/src/types.ts:211`](../packages/core/session/src/types.ts) +Source: [`packages/core/session/src/types.ts:207`](../packages/core/session/src/types.ts) ### `user/*` @@ -552,4 +552,4 @@ Source: [`packages/core/session/src/types.ts:211`](../packages/core/session/src/ 'user/message': UserMessageData ``` -Source: [`packages/core/session/src/types.ts:231`](../packages/core/session/src/types.ts) +Source: [`packages/core/session/src/types.ts:227`](../packages/core/session/src/types.ts) diff --git a/examples/headless-agent/tests/fixtures/goal-domain/seed-goal.ts b/examples/headless-agent/tests/fixtures/goal-domain/seed-goal.ts index 254870eca9..cb0e072d2e 100644 --- a/examples/headless-agent/tests/fixtures/goal-domain/seed-goal.ts +++ b/examples/headless-agent/tests/fixtures/goal-domain/seed-goal.ts @@ -7,7 +7,7 @@ export const name = 'seed-goal' export const inject = ['goals'] export function apply(ctx: Context): void { - ctx.on('agent/pre-step', (agent) => { + ctx.on('agent/step', (agent) => { if (ctx.goals.get(agent) !== undefined) return ctx.goals.create(agent, { objective: 'Prove the composed goal survives in the session log', diff --git a/packages/bash/tool-bash/tests/integration.spec.ts b/packages/bash/tool-bash/tests/integration.spec.ts index 8adf2165e0..99b4b70734 100644 --- a/packages/bash/tool-bash/tests/integration.spec.ts +++ b/packages/bash/tool-bash/tests/integration.spec.ts @@ -114,7 +114,7 @@ describe('bash tool through the agent loop', () => { const result = findEvent(events(agent), 'tool/result') expect(resultText(result)).toBe(`${dshHome}\n1\nsession-env-id\n${location?.path}\nunset\nabsent\n`) - expect(existsSync(location!.path)).toBe(true) + await expect.poll(() => existsSync(location!.path)).toBe(true) const header = JSON.parse(readFileSync(location!.path, 'utf8').split('\n')[0]!) as { type: string; id: string } expect(header).toMatchObject({ type: 'session', id: 'session-env-id' }) await handle.dispose() diff --git a/packages/compact/compact-basic/README.md b/packages/compact/compact-basic/README.md index 7279cc029d..4f1754575e 100644 --- a/packages/compact/compact-basic/README.md +++ b/packages/compact/compact-basic/README.md @@ -8,16 +8,16 @@ This is the implementation tier of the compaction capability — see the [interf This backend owns the compaction policy: -- **Measurement** — the singleton `ctx.tokenMeter` prices the latest canonical logged envelope and current surface at one consumed-log revision. Post-step pressure therefore includes the actual system prompt, tools, prefix, routing, assistant completion, tool results, buffered context, and steering. +- **Measurement** — the singleton `ctx.tokenMeter` prices the latest canonical logged envelope and current surface at one consumed-log revision. Step-boundary pressure therefore includes the actual system prompt, tools, routing, assistant completion, tool results, buffered context, and steering. - **Routed policy** — proactive pressure resolves capacity from the adapter that owns the latest durable provider/model route, then scales the default policy plus an optional exact-target override into concrete token budgets. Model discovery remains advisory and is not consulted. -- **Model-free pruning** — after pressure or canonical overflow qualifies, the optional [`ctx.toolResultPrune`](../compact-tool-result-prune/README.md) service rewrites oversized tool results before range selection. Compact-basic remeasures through `ctx.tokenMeter`, skips summarization when pressure becomes safe, and otherwise summarizes the pruned surface. Below-pressure post-step checks never prune. +- **Model-free pruning** — after pressure or canonical overflow qualifies, the optional [`ctx.toolResultPrune`](../compact-tool-result-prune/README.md) service rewrites oversized tool results before range selection. Compact-basic remeasures through `ctx.tokenMeter`, skips summarization when pressure becomes safe, and otherwise summarizes the pruned surface. Below-pressure step checks never prune. - **Retention** — compact the oldest whole surface units while preserving a recent tail and balanced tool-call/result cuts through the [`dsh-compact` boundary helpers](../compact/README.md#tool-pairing-boundaries). Turn boundaries do not protect old steps inside a runaway turn. An open indivisible tail declines until it closes. The optional pruner can repair an oversized closed tool unit when its text-bearing result is the removable bulk; indivisible non-tool units and non-prunable tool remainders remain out of scope. - **Convergence** — retry head-checkpoint compaction up to `compactionRetries`; reject a summary that does not shrink its source, and throw if retries cannot return below threshold. - **Summarization** — a direct `llm/stream` call uses the configured provider/model pair and cap, falling back to the latest logged request target and then the agent target, without running the loop-only `agent/request` seam. The call replays the conversation's own system prompt, tools, and shadowed-region messages verbatim and appends the compaction instruction as the final user message, so it reuses the provider's warm prefix cache instead of invalidating it. It sets `GenerateOptions.purpose` to `compaction`, which adapters may forward as request attribution (the DeepSeek adapter sends `x-deepseek-harness-compact: 1`) without touching the model-visible body. Only returned text enters the checkpoint, excluding reasoning and tool calls that would leak private reasoning or create an orphaned call. - **Framing** — the replacement user message marks established checkpoint context with `` tags. The raw summary remains on the provenance event, and later automatic cycles merge the prior checkpoint. -- **Lifecycle** — `compactRegion()` mutates `agent.session` and records its start, summary, replacement, and end. After asynchronous summarization it rejects a changed surface-node snapshot, while unrelated log-only events may append without invalidating the selected span. The serial `agent/post-step` listener checks pressure after successful output and tool work are durable but before `step/end`. Canonical provider overflow is handled through `agent/request-error` after the failed step closes. +- **Lifecycle** — `compactRegion()` mutates `agent.session` and records its start, summary, replacement, and end. After asynchronous summarization it rejects a changed surface-node snapshot, while unrelated log-only events may append without invalidating the selected span. The serial `agent/step` listener checks pressure before request derivation. A canonical provider overflow is offered through `agent/request-error` after the failed step; the plugin compacts there and calls `agent.retry()` only after durable surface progress. - **Overflow recovery** — provider-confirmed overflow needs no capacity metadata: it bypasses normal pressure and retention, prunes, then attempts one maximal balanced head reduction while leaving the newest indivisible unit. Retry is authorized whenever `surface.replaceGeneration` advances, including when pruning lands before later summary work throws. No replacement, an exhausted target-specific cap, cancellation, or an unknown/noncanonical error preserves the original provider failure. -- **Failure handling** — an unmatched `compact/start` is an inert crash marker because no summary replacement landed. A region failure records an error end; the surface remains unchanged unless pruning already landed. Operational post-step failures warn and continue, while overflow-recovery failure preserves the original provider error only when no earlier replacement advanced the surface. Cancellation remains authoritative after any progress. +- **Failure handling** — an unmatched `compact/start` is an inert crash marker because no summary replacement landed. A region failure records an error end; the surface remains unchanged unless pruning already landed. Operational pressure failures warn and continue, while overflow-recovery failure preserves the original provider error only when no earlier replacement advanced the surface. Cancellation remains authoritative after any progress. The protected `summarize()` method is the sole subclass hook. A template- or remote-summarizer subclass can override it while pressure, retention, provenance, shrink validation, and shadowed-token accounting stay on `ctx.tokenMeter`. The hook returns the summary blocks together with the call envelope it used (`{ summary, provider, model, maxTokens? }`), which is logged on `compact/summary`. @@ -36,7 +36,7 @@ Every setting is optional. Top-level policy fields are defaults for every routed | `compactionRetries` | no (default `1`) | Extra attempts after the first when pressure remains above threshold. | | `maxOverflowRetries` | no (default `1`) | Maximum retries after canonical context-window overflow; `0` disables recovery only. | | `modelPolicies` | no (default `[]`) | Exact `{ provider, model, ...partialPolicy }` overrides; matching uses both fields and does not depend on `listModels()`. | -| `auto` | no (default `true`) | Register post-step pressure and overflow-recovery listeners. Set `false` for manual-only. | +| `auto` | no (default `true`) | Register step-boundary pressure and overflow-recovery listeners. Set `false` for manual-only. | Every `modelPolicies` entry accepts the policy fields above except `auto` and `modelPolicies` itself. If an entry supplies either retention field, it replaces the default policy's retention choice; otherwise retention is inherited. Summarization provider/model remain a pair inside each entry. diff --git a/packages/compact/compact-basic/src/index.ts b/packages/compact/compact-basic/src/index.ts index 10c64d10ca..2d5cc68ee1 100644 --- a/packages/compact/compact-basic/src/index.ts +++ b/packages/compact/compact-basic/src/index.ts @@ -111,6 +111,7 @@ export class BasicCompactService extends CompactService { readonly config: ResolvedConfig private readonly warnedPressureConfigTargets = new Set() + private readonly overflowRetries = new WeakMap() constructor(ctx: Context, config: BasicCompactConfig = {}) { super(ctx) @@ -119,8 +120,8 @@ export class BasicCompactService extends CompactService { } /** - * Register the automatic post-step pressure and context-overflow recovery - * listeners. `compactIfNeeded` stays dynamically dispatched so subclass + * Register automatic between-step pressure and model-request overflow + * recovery. `compactIfNeeded` stays dynamically dispatched so subclass * overrides are honored at event time. */ private _registerAutomaticCompaction(): void { @@ -133,7 +134,7 @@ export class BasicCompactService extends CompactService { ) } - ctx.on('agent/post-step', async ( + ctx.on('agent/step', async ( agent: Agent, _turn: number, _step: number, @@ -142,35 +143,36 @@ export class BasicCompactService extends CompactService { if (signal.aborted) return try { const result = await this.compactIfNeeded(agent, 'pressure', signal) - if (result !== null) logResult(result, 'post-step pressure') + if (result !== null) logResult(result, 'step pressure') } catch (error: unknown) { if (error instanceof TargetPressureConfigError) { if (this.warnedPressureConfigTargets.has(error.targetKey)) return this.warnedPressureConfigTargets.add(error.targetKey) } const message = error instanceof Error ? error.message : String(error) - ctx.logger.warn(`post-step compaction failed: ${message}; continuing the turn`) + ctx.logger.warn(`step compaction failed: ${message}; continuing the turn`) } }) + ctx.on('agent/idle', (agent) => { + this.overflowRetries.delete(agent) + }) + ctx.on('agent/request-error', async ( agent, _turn, _step, _error, failure, - priorFailures, signal, next, ) => { - const priorOverflowFailures = priorFailures.filter( - item => item.code === CONTEXT_WINDOW_EXCEEDED_CODE, - ).length if (failure.code !== CONTEXT_WINDOW_EXCEEDED_CODE || signal.aborted) return next() const target = routedTarget(agent.session) if (target === undefined) return next() const policy = resolveTargetPolicy(this.config, target) - if (priorOverflowFailures >= policy.maxOverflowRetries) return next() + const retries = this.overflowRetries.get(agent) ?? 0 + if (retries >= policy.maxOverflowRetries) return next() const generation = agent.session.surface.replaceGeneration let result: CompactionResult | null @@ -181,27 +183,30 @@ export class BasicCompactService extends CompactService { // A model-free prune can land before later summary work fails. That // durable reduction is sufficient retry proof; do not discard it just // because the optional second phase threw. Cancellation still wins. - // eslint-disable-next-line @typescript-eslint/no-unnecessary-condition -- signal can abort while recovery is awaited. + // eslint-disable-next-line @typescript-eslint/no-unnecessary-condition -- the signal can abort while recovery is awaited. if (!signal.aborted && agent.session.surface.replaceGeneration > generation) { ctx.logger.warn( `context-overflow compaction failed after durable surface progress: ${message}; ` + 'retrying from the replacement surface', ) - return { action: 'retry' } + this.overflowRetries.set(agent, retries + 1) + agent.retry() + return } ctx.logger.warn( - // eslint-disable-next-line @typescript-eslint/no-unnecessary-condition -- signal can abort while recovery is awaited. + // eslint-disable-next-line @typescript-eslint/no-unnecessary-condition -- the signal can abort while recovery is awaited. `context-overflow compaction failed: ${message}; ${signal.aborted ? 'cancellation prevents retry' : 'preserving the original request error'}`, ) return next() } - // eslint-disable-next-line @typescript-eslint/no-unnecessary-condition -- signal can abort while compaction is awaited. + // eslint-disable-next-line @typescript-eslint/no-unnecessary-condition -- the signal can abort while compaction is awaited. if (signal.aborted || agent.session.surface.replaceGeneration <= generation) return next() if (result !== null) logResult(result, 'context overflow recovery') - return { action: 'retry' } + this.overflowRetries.set(agent, retries + 1) + agent.retry() }) } @@ -228,12 +233,12 @@ export class BasicCompactService extends CompactService { } /** - * Compact for replayed post-step pressure or one provider-confirmed context + * Compact for replayed step-boundary pressure or one provider-confirmed context * overflow. Both triggers price the latest durable routed request envelope; * overflow bypasses the normal threshold and retained-tail policy so it can * force one useful balanced reduction. * @param agent - agent whose latest durable routed request is measured. - * @param trigger - normal post-step pressure or context-overflow recovery. + * @param trigger - normal step-boundary pressure or context-overflow recovery. * @param signal - live turn cancellation signal forwarded to summarization. * @returns the latest summary compaction result, or `null` when no summary ran. */ diff --git a/packages/compact/compact-basic/src/types.ts b/packages/compact/compact-basic/src/types.ts index c322f508ac..f9f46051ec 100644 --- a/packages/compact/compact-basic/src/types.ts +++ b/packages/compact/compact-basic/src/types.ts @@ -38,7 +38,7 @@ export interface ModelCompactPolicyConfig extends CompactPolicyConfig { export interface BasicCompactConfig extends CompactPolicyConfig { /** Exact provider/model overrides; duplicate targets fail plugin load. */ modelPolicies?: ModelCompactPolicyConfig[] - /** Enable automatic post-step pressure and overflow-recovery listeners. Defaults to `true`. */ + /** Enable automatic step-boundary pressure and overflow-recovery listeners. Defaults to `true`. */ auto?: boolean } diff --git a/packages/compact/compact-basic/tests/compact-basic.spec.ts b/packages/compact/compact-basic/tests/compact-basic.spec.ts index 1db86d38e4..105485a7d5 100644 --- a/packages/compact/compact-basic/tests/compact-basic.spec.ts +++ b/packages/compact/compact-basic/tests/compact-basic.spec.ts @@ -66,7 +66,11 @@ function createContext(contextWindow = 1_000): Context { } function agent(session: Session, model?: string): Agent { - return { session, options: model === undefined ? {} : { provider: model, model } } as Agent + return { + session, + options: model === undefined ? {} : { provider: model, model }, + retry() {}, + } as Agent } /** Flatten every text fragment the summarizer received, recursing tool-result blocks. */ @@ -1263,22 +1267,23 @@ describe('default one-shot summarizer', () => { describe('automatic listener and loader composition', () => { function postStep(ctx: Context, owner: Agent, signal = SIGNAL): Promise { - return agentEvents(ctx, owner).serial('agent/post-step', 1, 1, signal) + return agentEvents(ctx, owner).serial('agent/step', 1, 1, signal) } function recover( ctx: Context, owner: Agent, error: Error & { code?: string }, - retryAttempt = 0, signal = SIGNAL, - next: () => Promise<{ action: 'fail' | 'retry' }> = () => Promise.resolve({ action: 'fail' }), - ): Promise<{ action: 'fail' | 'retry' }> { + next: () => Promise = () => Promise.resolve(), + ): Promise { const failure: LlmFailure = { message: error.message, code: error.code ?? 'UNKNOWN' } - const priorFailures = Object.freeze(Array.from({ length: retryAttempt }, () => failure)) + const turn = owner.session.events.findLast(event => event.type === 'turn/start')?.data.turn ?? 1 + let retried = false + owner.retry = () => { retried = true } return agentEvents(ctx, owner).waterfall( - 'agent/request-error', 1, 1, error, failure, priorFailures, signal, next, - ) + 'agent/request-error', turn, 1, error, failure, signal, next, + ).then(() => retried) } function overflow(message = 'provider overflow'): Error & { code: string } { @@ -1383,7 +1388,7 @@ describe('automatic listener and loader composition', () => { expect(ctx.tokenMeter.measure(session).totalTokens).toBeLessThan(threshold) const decision = await recover(ctx, agent(session, 'unconfigured-agent-fallback'), overflow()) - expect(decision).toEqual({ action: 'retry' }) + expect(decision).toBe(true) expect(session.surface.replaceGeneration).toBe(beforeGeneration + 1) expect(session.events.some(event => event.type === 'compact/summary')).toBe(true) expect(session.surface.nodes).toContain(retainedSeq) @@ -1402,7 +1407,7 @@ describe('automatic listener and loader composition', () => { }) const session = oversizedToolResult() - expect(await recover(ctx, agent(session, MODEL), overflow())).toEqual({ action: 'retry' }) + expect(await recover(ctx, agent(session, MODEL), overflow())).toBe(true) expect(session.surface.replaceGeneration).toBe(1) expect(session.events.some(event => event.type === 'compact/summary')).toBe(false) expect(compact.calls).toHaveLength(0) @@ -1421,7 +1426,7 @@ describe('automatic listener and loader composition', () => { }) const session = toolConversation() - expect(await recover(ctx, agent(session, MODEL), overflow())).toEqual({ action: 'retry' }) + expect(await recover(ctx, agent(session, MODEL), overflow())).toBe(true) expect(session.events.some(event => event.type === 'compact/summary')).toBe(true) expect(compact.calls).toHaveLength(1) expect(summarizedText(compact.calls[0]!.input)).toContain('tool result middle pruned') @@ -1443,7 +1448,7 @@ describe('automatic listener and loader composition', () => { compact.error = new Error('summary unavailable after prune') const session = oversizedToolResult(3_000, true) - expect(await recover(ctx, agent(session, MODEL), overflow())).toEqual({ action: 'retry' }) + expect(await recover(ctx, agent(session, MODEL), overflow())).toBe(true) expect(session.surface.replaceGeneration).toBe(1) expect(session.events.filter(event => event.type === 'tool/result')).toHaveLength(2) expect(session.events.findLast(event => event.type === 'compact/end')?.data) @@ -1467,8 +1472,7 @@ describe('automatic listener and loader composition', () => { compact.error = new Error('summary cancelled after prune') const session = oversizedToolResult(3_000, true) - expect(await recover(ctx, agent(session, MODEL), overflow(), 0, controller.signal)) - .toEqual({ action: 'fail' }) + expect(await recover(ctx, agent(session, MODEL), overflow(), controller.signal)).toBe(false) expect(session.surface.replaceGeneration).toBe(1) }) @@ -1482,7 +1486,7 @@ describe('automatic listener and loader composition', () => { const newestAssistant = session.surface.nodes.at(-2)! const newestResult = session.surface.nodes.at(-1)! - expect(await recover(ctx, agent(session, MODEL), overflow())).toEqual({ action: 'retry' }) + expect(await recover(ctx, agent(session, MODEL), overflow())).toBe(true) const currentAssistant = session.surface.nodes.find(node => node === newestAssistant) const currentResult = session.surface.nodes.find(node => node === newestResult) expect(currentAssistant).toBeDefined() @@ -1506,7 +1510,7 @@ describe('automatic listener and loader composition', () => { } vi.spyOn(compact, 'compactIfNeeded').mockResolvedValue(fakeResult) - expect(await recover(ctx, agent(session, MODEL), overflow())).toEqual({ action: 'fail' }) + expect(await recover(ctx, agent(session, MODEL), overflow())).toBe(false) expect(session.surface.replaceGeneration).toBe(0) }) @@ -1521,7 +1525,6 @@ describe('automatic listener and loader composition', () => { ctx, agent(conversation(2), MODEL), overflow(), - 0, SIGNAL, () => { calls += 1 @@ -1539,7 +1542,7 @@ describe('automatic listener and loader composition', () => { compact.error = new Error('summary unavailable') const original = overflow('original provider overflow') - expect(await recover(ctx, agent(conversation(3), MODEL), original)).toEqual({ action: 'fail' }) + expect(await recover(ctx, agent(conversation(3), MODEL), original)).toBe(false) expect(original).toMatchObject({ message: 'original provider overflow', code: CONTEXT_WINDOW_EXCEEDED_CODE, @@ -1558,12 +1561,12 @@ describe('automatic listener and loader composition', () => { const original = overflow('original provider failure') let delegations = 0 - const decision = await recover(ctx, agent(session, MODEL), original, 0, SIGNAL, () => { + const decision = await recover(ctx, agent(session, MODEL), original, SIGNAL, () => { delegations += 1 - return Promise.resolve({ action: 'fail' }) + return Promise.resolve() }) - expect(decision).toEqual({ action: 'fail' }) + expect(decision).toBe(false) expect(delegations).toBe(1) expect(session.surface.replaceGeneration).toBe(generation) expect(original).toMatchObject({ @@ -1582,7 +1585,7 @@ describe('automatic listener and loader composition', () => { reason: 'resume', }) expect(await recover(ctx, agent(session, MODEL), overflow('unlisted-model overflow'))) - .toEqual({ action: 'retry' }) + .toBe(true) }) it('delegates canonical overflow when no durable routed target exists', async () => { @@ -1594,21 +1597,19 @@ describe('automatic listener and loader composition', () => { trigger: { kind: 'message', source: { kind: 'user' } }, }) - await expect(recover(ctx, agent(session, MODEL), overflow())).resolves.toEqual({ action: 'fail' }) + await expect(recover(ctx, agent(session, MODEL), overflow())).resolves.toBe(false) }) - it('honors retry caps, non-context failures, and cancellation', async () => { + it('honors retry caps and ignores non-context failures', async () => { const ctx = createContext() const compact = new TestCompactService(ctx, { maxOverflowRetries: 1 }) const compactSpy = vi.spyOn(compact, 'compactIfNeeded') const owner = agent(conversation(3), MODEL) expect(await recover(ctx, owner, Object.assign(new Error('rate limit'), { code: 'RATE_LIMIT' }))) - .toEqual({ action: 'fail' }) - expect(await recover(ctx, owner, overflow(), 1)).toEqual({ action: 'fail' }) - - const controller = new AbortController() - controller.abort('cancelled') - expect(await recover(ctx, owner, overflow(), 0, controller.signal)).toEqual({ action: 'fail' }) + .toBe(false) + expect(await recover(ctx, owner, overflow())).toBe(true) + compactSpy.mockClear() + expect(await recover(ctx, owner, overflow())).toBe(false) expect(compactSpy).not.toHaveBeenCalled() }) @@ -1623,9 +1624,11 @@ describe('automatic listener and loader composition', () => { }], }) const compactSpy = vi.spyOn(compact, 'compactIfNeeded') + const owner = agent(conversation(3), MODEL) - expect(await recover(ctx, agent(conversation(3), MODEL), overflow(), 1)) - .toEqual({ action: 'fail' }) + expect(await recover(ctx, owner, overflow())).toBe(true) + compactSpy.mockClear() + expect(await recover(ctx, owner, overflow())).toBe(false) expect(compactSpy).not.toHaveBeenCalled() }) @@ -1637,8 +1640,7 @@ describe('automatic listener and loader composition', () => { const session = conversation(3) const generation = session.surface.replaceGeneration - expect(await recover(ctx, agent(session, MODEL), overflow(), 0, controller.signal)) - .toEqual({ action: 'fail' }) + expect(await recover(ctx, agent(session, MODEL), overflow(), controller.signal)).toBe(false) expect(session.surface.replaceGeneration).toBe(generation + 1) }) @@ -1653,7 +1655,7 @@ describe('automatic listener and loader composition', () => { await postStep(ctx, agent(session, MODEL)) const summaries = session.events.filter(event => event.type === 'compact/summary').length expect(summaries).toBe(1) - expect(await recover(ctx, agent(session, MODEL), overflow())).toEqual({ action: 'fail' }) + expect(await recover(ctx, agent(session, MODEL), overflow())).toBe(false) expect(session.events.filter(event => event.type === 'compact/summary')).toHaveLength(summaries) }) @@ -1667,7 +1669,7 @@ describe('automatic listener and loader composition', () => { const session = conversation(4) await postStep(ctx, agent(session, MODEL)) expect(session.events.some(event => event.type === 'compact/start')).toBe(false) - expect(await recover(ctx, agent(session, MODEL), overflow())).toEqual({ action: 'fail' }) + expect(await recover(ctx, agent(session, MODEL), overflow())).toBe(false) }) it('loads and disposes the real zero-config service stack', async () => { @@ -1696,6 +1698,6 @@ describe('automatic listener and loader composition', () => { const session = conversation(4) await postStep(ctx, agent(session, MODEL)) expect(session.events.some(event => event.type === 'compact/start')).toBe(false) - expect(await recover(ctx, agent(session, MODEL), overflow())).toEqual({ action: 'fail' }) + expect(await recover(ctx, agent(session, MODEL), overflow())).toBe(false) }) }) diff --git a/packages/compact/compact-basic/tests/compact-loop-repro.spec.ts b/packages/compact/compact-basic/tests/compact-loop-repro.spec.ts index 83c18f78b4..5cbc429d52 100644 --- a/packages/compact/compact-basic/tests/compact-loop-repro.spec.ts +++ b/packages/compact/compact-basic/tests/compact-loop-repro.spec.ts @@ -15,7 +15,7 @@ import * as AgentLoopInvariant from '@deepseek-ai/dsh-agent-loop/invariant' import { BasicCompactService } from '@deepseek-ai/dsh-compact-basic' import TokenMeterService from '@deepseek-ai/dsh-token-meter' import * as LlmRetry from '@deepseek-ai/dsh-llm-retry' -import { SessionId, type SurfaceEvent } from '@deepseek-ai/dsh-session' +import { Session, SessionId, type SessionEvent, type SurfaceEvent } from '@deepseek-ai/dsh-session' /** * CBR-001 regression through the real loop. A replacement checkpoint has a high @@ -165,33 +165,37 @@ function waitForIdle(ctx: Context, agent: Agent): Promise { }) } -function seedOverflowHistory(agent: Agent): void { +function overflowHistorySeed(): SessionEvent[] { + const session = new Session(SessionId('overflow-history-seed')) for (let turn = 1; turn <= 2; turn += 1) { const sentinel = turn === 1 ? 'OLD HISTORY SENTINEL' : 'RECENT HISTORY' - agent.session.append('turn/start', { + session.append('turn/start', { turn, trigger: { kind: 'message', source: { kind: 'user' } }, }) - agent.session.append('user/message', { + session.append('user/message', { content: [{ type: 'text', text: `${sentinel} ${'old context '.repeat(200)}` }], source: { kind: 'user' }, }, { surfaceOp: 'append' }) - agent.session.append('step/start', { turn, step: 1 }) - agent.session.append('assistant/message', { + session.append('step/start', { turn, step: 1 }) + session.append('assistant/message', { provenance: { provider: 'mock', model: 'mock' }, turn, step: 1, content: [{ type: 'text', text: `historical response ${turn} ${'detail '.repeat(200)}` }], }, { surfaceOp: 'append' }) - agent.session.append('step/end', { turn, step: 1 }) - agent.session.append('turn/end', { turn, reason: { kind: 'completed' } }) + session.append('step/end', { turn, step: 1 }) + session.append('turn/end', { turn, reason: { kind: 'completed' } }) } + return [...session.events] } describe('CBR-001: a real-loop checkpoint is a valid boundary on both sides', () => { it('uses the model actually routed by agent/request for post-step pressure', async () => { const { ctx } = await harness(8) - ctx.on('agent/request', async (_agent, _turn, _step, config) => ({ ...config, provider: 'mock', model: 'mock' })) + ctx.on('agent/request', async (_agent, _turn, _step, _signal, next) => ({ + ...await next(), provider: 'mock', model: 'mock', + })) try { const agent = ctx.agentLoop.create(SessionId('routed-pressure'), { provider: 'unconfigured-agent-fallback', @@ -211,7 +215,7 @@ describe('CBR-001: a real-loop checkpoint is a valid boundary on both sides', () } }) - it('runs automatic pressure after the current tool result and before step/end', async () => { + it('runs automatic pressure between the completed tool step and the next step', async () => { const { ctx } = await harness(8) try { const agent = ctx.agentLoop.create(SessionId('post-step-order'), { provider: 'mock', model: 'mock' }) @@ -225,13 +229,19 @@ describe('CBR-001: a real-loop checkpoint is a valid boundary on both sides', () event.type === 'tool/result' && event.seq < compactStart!.seq, ) if (precedingResult?.type !== 'tool/result') throw new Error('expected a durable tool result before compaction') - const stepEnd = events.find(event => + const precedingStepEnd = events.find(event => event.type === 'step/end' && event.data.step === precedingResult.data.step + && event.seq > precedingResult.seq, + ) + const nextStepStart = events.find(event => + event.type === 'step/start' + && event.data.step === precedingResult.data.step + 1 && event.seq > compactStart!.seq, ) expect(precedingResult.seq).toBeLessThan(compactStart!.seq) - expect(compactStart!.seq).toBeLessThan(stepEnd!.seq) + expect(precedingStepEnd!.seq).toBeLessThan(compactStart!.seq) + expect(compactStart!.seq).toBeLessThan(nextStepStart!.seq) } finally { await ctx.fiber.dispose() } @@ -281,7 +291,9 @@ describe('context-overflow recovery across the real loop and compact-basic', () await ctx.plugin(AgentLoop, { agents: [] }) await ctx.plugin(TokenMeterService) ctx.llm.registerAdapter(['mock'], adapter) - ctx.on('agent/request', async (_agent, _turn, _step, config) => ({ ...config, provider: 'mock', model: 'mock' })) + ctx.on('agent/request', async (_agent, _turn, _step, _signal, next) => ({ + ...await next(), provider: 'mock', model: 'mock', + })) await ctx.plugin(BasicCompactService, { thresholdRatio: 1, retainTokens: 100, @@ -291,11 +303,14 @@ describe('context-overflow recovery across the real loop and compact-basic', () }) try { - const agent = ctx.agentLoop.create(SessionId(`overflow-${delivery}`), { - provider: 'unconfigured-agent-fallback', - model: 'unconfigured-agent-fallback', + const { agent } = await ctx.agentLoop.createAgent(ctx, { + sessionId: SessionId(`overflow-${delivery}`), + seed: overflowHistorySeed(), + agentOptions: { + provider: 'unconfigured-agent-fallback', + model: 'unconfigured-agent-fallback', + }, }) - seedOverflowHistory(agent) agent.followup([{ type: 'text', text: 'continue from history' }]) await agent.whenIdle() @@ -308,11 +323,17 @@ describe('context-overflow recovery across the real loop and compact-basic', () expect(retry).not.toContain('OLD HISTORY SENTINEL') const events = [...agent.session.events] - const failedEnd = events.find(event => + const failedStepEnd = events.find(event => event.type === 'step/end' && event.data.turn === 3 && event.data.step === 1, )! + const failedEnd = events.find(event => + event.type === 'turn/end' && event.data.turn === 3, + )! const retryStart = events.find(event => - event.type === 'step/start' && event.data.turn === 3 && event.data.step === 2, + event.type === 'turn/start' && event.data.turn === 4, + )! + const retryStep = events.find(event => + event.type === 'step/start' && event.data.turn === 4 && event.data.step === 1, )! const compaction = events.filter(event => event.type === 'compact/start' @@ -324,7 +345,11 @@ describe('context-overflow recovery across the real loop and compact-basic', () 'compact/summary', 'compact/end', ]) - expect(compaction.every(event => event.seq > failedEnd.seq && event.seq < retryStart.seq)).toBe(true) + expect(retryStart.seq).toBeGreaterThan(failedEnd.seq) + expect(compaction.every(event => + event.seq > failedStepEnd.seq && event.seq < failedEnd.seq, + )).toBe(true) + expect(retryStep.seq).toBeGreaterThan(retryStart.seq) expect(events.at(-1)).toMatchObject({ type: 'turn/end', data: { reason: { kind: 'completed' } }, @@ -358,17 +383,21 @@ describe('context-overflow recovery across the real loop and compact-basic', () }) try { - const agent = ctx.agentLoop.create(SessionId('alternating-recovery'), { provider: 'mock', model: 'mock' }) - seedOverflowHistory(agent) + const { agent } = await ctx.agentLoop.createAgent(ctx, { + sessionId: SessionId('alternating-recovery'), + seed: overflowHistorySeed(), + agentOptions: { provider: 'mock', model: 'mock' }, + }) agent.followup([{ type: 'text', text: 'continue from history' }]) + await expect.poll(() => adapter.conversationRequests.length).toBe(3) await agent.whenIdle() expect(adapter.conversationRequests).toHaveLength(3) expect(adapter.summaryRequests).toHaveLength(1) expect(agent.session.events.filter(event => event.type === 'llm/retry').map(event => event.data)) - .toEqual([expect.objectContaining({ step: 2, retry: 1, failure: { message: 'temporary provider outage', code: 'SERVER' } })]) - expect(agent.session.events.filter(event => event.type === 'step/start').slice(-3).map(event => event.data.step)) - .toEqual([1, 2, 3]) + .toEqual([expect.objectContaining({ turn: 4, step: 1, retry: 1, failure: { message: 'temporary provider outage', code: 'SERVER' } })]) + expect(agent.session.events.filter(event => event.type === 'turn/start').slice(-3).map(event => event.data.turn)) + .toEqual([3, 4, 5]) expect(agent.session.events.at(-1)).toMatchObject({ type: 'turn/end', data: { reason: { kind: 'completed' } }, diff --git a/packages/context/time-context/tests/time-context.spec.ts b/packages/context/time-context/tests/time-context.spec.ts index 424363d4b1..d9c95f0bc4 100644 --- a/packages/context/time-context/tests/time-context.spec.ts +++ b/packages/context/time-context/tests/time-context.spec.ts @@ -43,7 +43,6 @@ function sessionAgent(session: Session, id = 'agent'): Agent { status: 'running', ctx: new Context(), followup: () => AgentMessageId('stub'), - queue: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject(content, options) { session.append('user/message', { @@ -54,6 +53,7 @@ function sessionAgent(session: Session, id = 'agent'): Agent { }, send: () => AgentMessageId('stub'), cancel() {}, + retry() {}, whenIdle: () => Promise.resolve(), } } @@ -85,7 +85,7 @@ async function fire( step: number, signal: AbortSignal = SIGNAL, ): Promise { - await agentEvents(ctx, agent).serial('agent/pre-step', turn, step, signal) + await agentEvents(ctx, agent).serial('agent/step', turn, step, signal) } function textResponse(text: string): StreamChunk[] { @@ -294,7 +294,7 @@ describe('durable step context', () => { const agent = sessionAgent(session) openMessageTurn(session, 1) let ordinarySawContext = false - ctx.on('agent/pre-step', (subject) => { + ctx.on('agent/step', (subject) => { ordinarySawContext = subject.session.events.some(event => event.type === 'user/message') }) @@ -363,11 +363,11 @@ describe('real agent-loop request history', () => { it.each([ ['throws', 'error'], ['cancels', 'aborted'], - ] as const)('retains the preparation reading when a later pre-step listener %s', async (mode, reasonKind) => { + ] as const)('discards the pending preparation reading when a later step listener %s', async (mode, reasonKind) => { const adapter = new ScriptedAdapter([textResponse('unused')]) const ctx = await loopHarness(adapter) let laterSawReading = false - ctx.on('agent/pre-step', (subject) => { + ctx.on('agent/step', (subject) => { laterSawReading = contextTexts(subject.session).length === 1 if (mode === 'throws') throw new Error('later pre-step failure') subject.cancel({ kind: 'user' }) @@ -377,8 +377,8 @@ describe('real agent-loop request history', () => { agent.followup([{ type: 'text', text: 'start' }]) await agent.whenIdle() - expect(laterSawReading).toBe(true) - expect(contextTexts(agent.session)).toHaveLength(1) + expect(laterSawReading).toBe(false) + expect(contextTexts(agent.session)).toHaveLength(0) expect(adapter.requests).toHaveLength(0) expect(agent.session.events.some(event => event.type === 'step/start')).toBe(false) const turnEnd = agent.session.events.findLast(event => event.type === 'turn/end') diff --git a/packages/context/workspace-context/src/state.ts b/packages/context/workspace-context/src/state.ts index 1679cd201b..d397218e95 100644 --- a/packages/context/workspace-context/src/state.ts +++ b/packages/context/workspace-context/src/state.ts @@ -108,7 +108,9 @@ function filePathFromExecution(exec: ToolExecution): string | undefined { return filePath.length > 0 ? filePath : undefined } -function isWorkspaceContextSource(source: unknown): source is WorkspaceInstructionSource { +function isWorkspaceContextSource( + source: unknown, +): source is { kind: 'workspace-instructions'; changes: unknown[] } { return typeof source === 'object' && source !== null && 'kind' in source && source.kind === 'workspace-instructions' && 'changes' in source && Array.isArray(source.changes) diff --git a/packages/context/workspace-context/tests/workspace-context.spec.ts b/packages/context/workspace-context/tests/workspace-context.spec.ts index d5378d14e4..f2ea056cdf 100644 --- a/packages/context/workspace-context/tests/workspace-context.spec.ts +++ b/packages/context/workspace-context/tests/workspace-context.spec.ts @@ -7,7 +7,7 @@ import Loader from '@cordisjs/plugin-loader' import * as workspaceContext from '@deepseek-ai/dsh-workspace-context' import LlmService, { CallId, type Message, type StreamChunk } from '@deepseek-ai/dsh-llm' import SessionStore, { Session, SessionId, SESSION_FORMAT_VERSION, type SessionEvent } from '@deepseek-ai/dsh-session' -import AgentRegistry, { AgentMessageId, type AdditionalContext, type Agent } from '@deepseek-ai/dsh-agent' +import AgentRegistry, { agentEvents, AgentMessageId, type AdditionalContext, type Agent } from '@deepseek-ai/dsh-agent' import AgentLoop from '@deepseek-ai/dsh-agent-loop' import { FileSystem, FsTargetKey, FsVersion } from '@deepseek-ai/dsh-fs' import type { @@ -178,7 +178,6 @@ function stubAgent(cwd?: string, seed: SessionEvent[] = []): Agent { session, status: 'idle', followup: () => AgentMessageId('stub'), - queue: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject(content, options) { session.append('user/message', { @@ -189,6 +188,7 @@ function stubAgent(cwd?: string, seed: SessionEvent[] = []): Agent { }, send: () => AgentMessageId('stub'), cancel() {}, + retry() {}, whenIdle: () => Promise.resolve(), } } @@ -233,11 +233,8 @@ function appendAdditionalContexts(agent: Agent, result: { additionalContexts?: A const composedPrefixes = new WeakMap() async function composeBaselinePrefix(ctx: Context, agent: Agent): Promise { - const empty: Message[] = [] - const prefix = await ctx.waterfall( - 'agent/session-prefix', agent, empty, AbortSignal.timeout(1000), - () => Promise.resolve(empty), - ) + await agentEvents(ctx, agent).serial('agent/step', 1, 1, AbortSignal.timeout(1000)) + const prefix = agent.session.deriveMessages() composedPrefixes.set(agent, prefix) return prefix } @@ -936,7 +933,7 @@ describe('workspace context request injection', () => { } }) - it('contributes baseline instructions through the frozen session prefix instead of durable history', async () => { + it('contributes baseline instructions through durable injected history', async () => { const root = await tempRepo() const home = await tempRepo() try { @@ -948,7 +945,9 @@ describe('workspace context request injection', () => { await composeBaselinePrefix(ctx, agent) - expect(agent.session.deriveMessages()).toEqual([]) + expect(agent.session.events.filter(event => + event.type === 'user/message' && event.data.source.kind !== 'user', + )).toHaveLength(1) expect(composedPrefixes.get(agent)).toHaveLength(1) expect(derivedText(agent)).toContain('') expect(derivedText(agent)).toContain('Instructions from: AGENTS.md') @@ -961,7 +960,7 @@ describe('workspace context request injection', () => { } }) - it('returns one baseline contribution per session-prefix composition without appending context events', async () => { + it('injects one durable baseline contribution on the first step only', async () => { const root = await tempRepo() const home = await tempRepo() try { @@ -975,7 +974,7 @@ describe('workspace context request injection', () => { const second = await composeBaselinePrefix(ctx, agent) expect(second).toEqual(first) - expect(agent.session.events.filter(event => event.type === 'user/message' && event.data.source.kind !== 'user')).toHaveLength(0) + expect(agent.session.events.filter(event => event.type === 'user/message' && event.data.source.kind !== 'user')).toHaveLength(1) expect(derivedText(agent)).toContain('repo rule') } finally { await rm(root, { recursive: true, force: true }) @@ -1005,7 +1004,7 @@ describe('workspace context request injection', () => { } }) - it('places workspace instructions before later session-prefix contributors such as a skills catalog', async () => { + it('places workspace instructions before later step contributors such as a skills catalog', async () => { const root = await tempRepo() const home = await tempRepo() try { @@ -1013,9 +1012,10 @@ describe('workspace context request injection', () => { await write(join(root, 'AGENTS.md'), 'repo rule') const ctx = new Context() await mountWorkspaceContext(ctx, { dshHome: home, maxBytes: 65536 }) - ctx.on('agent/session-prefix', async (_agent, _prefix, _signal, next) => { - const rest = await next() - return [{ role: 'user', content: [{ type: 'text', text: 'Available skills' }] }, ...rest] + ctx.on('agent/step', (agent) => { + agent.inject([{ type: 'text', text: 'Available skills' }], { + source: { kind: 'plugin', plugin: 'test-skills' }, + }) }) const prefix = await composeBaselinePrefix(ctx, stubAgent(root)) @@ -1147,7 +1147,9 @@ describe('workspace context request injection', () => { await composeBaselinePrefix(ctx, agent) - expect(agent.session.events.filter(event => event.type === 'user/message' && event.data.source.kind !== 'user')).toHaveLength(0) + expect(agent.session.events.filter(event => + event.type === 'user/message' && event.data.source.kind !== 'user', + )).toHaveLength(1) expect(derivedText(agent)).not.toContain('workspace-context:') } finally { await rm(root, { recursive: true, force: true }) @@ -1270,7 +1272,7 @@ describe('workspace context request injection', () => { } }) - it('aborts an in-flight baseline stream with the session-prefix signal', async () => { + it('aborts an in-flight baseline stream with the step signal', async () => { const root = join(await tempRepo(), 'virtual-repo') const home = join(await tempRepo(), 'virtual-home') const ctx = new Context() @@ -1282,11 +1284,7 @@ describe('workspace context request injection', () => { await ctx.plugin(workspaceContext, { dshHome: home, maxBytes: 65536 }) const controller = new AbortController() const reason = new Error('cancel prefix') - const empty: Message[] = [] - const pending = ctx.waterfall( - 'agent/session-prefix', stubAgent(root), empty, controller.signal, - () => Promise.resolve(empty), - ) + const pending = agentEvents(ctx, stubAgent(root)).serial('agent/step', 1, 1, controller.signal) await fs.started.promise controller.abort(reason) @@ -1715,14 +1713,16 @@ describe('dynamic nested workspace context injection', () => { agent.followup([{ type: 'text', text: 'read and abort' }]) await agent.whenIdle() - expect(agent.session.events.filter(event => event.type === 'user/message' && event.data.source.kind !== 'user')).toHaveLength(1) + expect(agent.session.events.filter(event => + event.type === 'user/message' && event.data.source.kind !== 'user', + )).toHaveLength(0) agent.followup([{ type: 'text', text: 'retry the read' }]) await agent.whenIdle() const contexts = agent.session.events.filter(event => event.type === 'user/message' && event.data.source.kind !== 'user') - // The aborted batch drained its accepted context before step close, so the - // retry sees durable history without producing a duplicate instruction. + // Cancellation discards the aborted step's pending context. The next + // successful read discovers and durably injects it once. expect(contexts).toHaveLength(1) expect(adapter.requests).toHaveLength(3) expect(adapter.requests[2]?.messages.map(blocks => blocksText(blocks.content)).join('\n')) diff --git a/packages/cordis/tool-cordis/src/api-catalog.ts b/packages/cordis/tool-cordis/src/api-catalog.ts index d5ce71a478..1a8150f0cf 100644 --- a/packages/cordis/tool-cordis/src/api-catalog.ts +++ b/packages/cordis/tool-cordis/src/api-catalog.ts @@ -861,7 +861,7 @@ export const EVENT_API: readonly EventApiEntry[] = [ name: 'agent/cancel-requested', mode: 'emit', signature: '\'agent/cancel-requested\'(this: Scoped, agent: Agent, cause: AgentCancelCause): void', - jsDoc: '/**\n * Effective broad cancellation was requested, before queued/outbox work\n * is cleared or the active turn is aborted. This observe-only notification\n * cannot veto cancellation; listener failures are contained.\n * @param agent - the agent whose current work is being cancelled.\n * @param cause - resolved typed cancellation cause, including the default.\n * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent.\n * @mode emit\n */', + jsDoc: '/**\n * Effective broad cancellation was requested, before queued/outbox work\n * is cleared or the active turn is aborted. This observe-only notification\n * cannot veto cancellation; listener failures are contained.\n * @param agent - the agent whose current work is being cancelled.\n * @param cause - the explicit typed cancellation cause.\n * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent.\n * @mode emit\n */', summary: 'Effective broad cancellation was requested, before queued/outbox work is cleared or the active turn is aborted.', }, { @@ -875,8 +875,8 @@ export const EVENT_API: readonly EventApiEntry[] = [ name: 'agent/disposed', mode: 'emit', signature: '\'agent/disposed\'(this: Scoped, agent: Agent): void', - jsDoc: '/**\n * An agent left the registry; AgentLoop emits this after driver quiescence\n * but before session detachment and scoped-registration unwind. Custom\n * registry users own their driver-ordering contract.\n * @param agent - the exact agent removed from the registry.\n * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent.\n * @mode emit\n */', - summary: 'An agent left the registry; AgentLoop emits this after driver quiescence but before session detachment and scoped-registration unwind.', + jsDoc: '/**\n * An agent left the registry; AgentLoop emits this after driver quiescence\n * and scoped-registration unwind, but before session detachment. Custom\n * registry users own their driver-ordering contract.\n * @param agent - the exact agent removed from the registry.\n * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent.\n * @mode emit\n */', + summary: 'An agent left the registry; AgentLoop emits this after driver quiescence and scoped-registration unwind, but before session detachment.', }, { name: 'agent/error', @@ -889,8 +889,8 @@ export const EVENT_API: readonly EventApiEntry[] = [ name: 'agent/idle', mode: 'emit', signature: '\'agent/idle\'(this: Scoped, agent: Agent, turn: number, reason: IdleReason): void', - jsDoc: '/**\n * One turn closed: its `turn/end` and durability flush are already\n * committed. `reason` says why — recovery consumers observe an `error`\n * reason, repair (edit the log, wait, resummon), and call\n * {@link Agent.retry}; UI consumers key turn-done presentation off it.\n * Emitted per turn, including cancelled and failed ones.\n * @param agent - the agent whose turn closed.\n * @param turn - the closed turn number.\n * @param reason - why the turn ended, with live error facts when it failed.\n * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent.\n * @mode emit\n */', - summary: 'One turn closed: its `turn/end` and durability flush are already committed.', + jsDoc: '/**\n * One drain chain reached its terminal turn: that turn\'s `turn/end` is\n * already committed. Automatically recovered failed turns do not emit this\n * notification. `reason` says why; model-request recovery is exhausted when\n * an error reaches it.\n * @param agent - the agent whose turn closed.\n * @param turn - the terminal turn number.\n * @param reason - why the terminal turn ended, with live error facts when it failed.\n * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent.\n * @mode emit\n */', + summary: 'One drain chain reached its terminal turn: that turn\'s `turn/end` is already committed.', }, { name: 'agent/inbox/dequeue', @@ -924,9 +924,16 @@ export const EVENT_API: readonly EventApiEntry[] = [ name: 'agent/request', mode: 'waterfall', signature: '\'agent/request\'(this: Scoped, agent: Agent, turn: number, step: number, signal: AbortSignal, next: () => Promise): Promise', - jsDoc: '/**\n * Replace the frozen call configuration. `await next()` yields the config\n * the machine would use (agent options on the first request, the logged\n * header afterwards); return a replacement to switch. Model-visible\n * content must use logged channels; this seam cannot mutate messages.\n * @param agent - the agent making the model call.\n * @param turn - the open turn number.\n * @param step - the step whose request this is.\n * @param signal - the current turn\'s explicit abort signal.\n * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent.\n * @mode waterfall\n */', + jsDoc: '/**\n * Replace the frozen call configuration. `await next()` yields the config\n * the machine would use (agent options on the first request, the logged\n * header afterwards); return a replacement to switch. Model-visible\n * content must use logged channels; this seam cannot mutate messages.\n * @param agent - the agent making the model call.\n * @param turn - the open turn number.\n * @param step - the step whose request this is.\n * @param signal - the current turn\'s explicit abort signal.\n * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent.\n * @mode waterfall\n*/', summary: 'Replace the frozen call configuration.', }, + { + name: 'agent/request-error', + mode: 'waterfall', + signature: '\'agent/request-error\'(this: Scoped, agent: Agent, turn: number, step: number, error: RequestError, failure: LlmFailure, signal: AbortSignal, next: () => Promise): Promise', + jsDoc: '/**\n * Handle a model-request failure after its failed step has closed but\n * before the failed turn closes. A listener calls {@link Agent.retry} to\n * schedule one retry turn, returns without `next()` when it owns the error,\n * or calls `next()` to delegate. The default leaves the failure terminal.\n * @param agent - the agent whose request failed.\n * @param turn - the open turn number.\n * @param step - the failed step number.\n * @param error - the original model-request failure.\n * @param failure - serializable facts normalized at the final adapter boundary.\n * @param signal - the turn abort signal.\n * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent.\n * @mode waterfall\n */', + summary: 'Handle a model-request failure after its failed step has closed but before the failed turn closes.', + }, { name: 'agent/session-start', mode: 'emit', @@ -938,8 +945,8 @@ export const EVENT_API: readonly EventApiEntry[] = [ name: 'agent/status', mode: 'emit', signature: '\'agent/status\'(this: Scoped, agent: Agent, status: AgentStatus): void', - jsDoc: '/**\n * Agent status changed (`idle` ⇄ `running`, or → `disposed`). `send()` does\n * not enter `running` synchronously; drive lifecycle from this event.\n * @param agent - the agent whose status flipped.\n * @param status - the status just entered (the transition\'s destination).\n * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent.\n * @mode emit\n */', - summary: 'Agent status changed (`idle` ⇄ `running`, or → `disposed`).', + jsDoc: '/**\n * Agent status changed (`idle` ⇄ `running`). `send()` does not enter\n * `running` synchronously; drive lifecycle from this event.\n * @param agent - the agent whose status flipped.\n * @param status - the status just entered (the transition\'s destination).\n * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent.\n * @mode emit\n */', + summary: 'Agent status changed (`idle` ⇄ `running`).', }, { name: 'agent/step', diff --git a/packages/core/agent-loop/README.md b/packages/core/agent-loop/README.md index 42a3b2dfd8..d595b7638f 100644 --- a/packages/core/agent-loop/README.md +++ b/packages/core/agent-loop/README.md @@ -12,7 +12,7 @@ Creation and resume are one rollback-covered transaction: construct a private se The caller fiber and the AgentLoop provider are co-owners. `AgentFactory.createAgent(ownerCtx, options)` and `resume(ownerCtx, options)` receive caller ownership explicitly, while the factory keeps its own dependency context for `sessions`/`llm`/`tools`/`systemPrompt`; this lets a caller inject only `agents` without shrinking the new agent's service surface. Caller unload, handle disposal, or provider unload converge on one memoized quiescence boundary. Provider shutdown waits both resource teardown and the public create/resume wrapper that observed deactivation, so no continuation can publish after dependencies disappear. -Each agent and its session share one caller-chosen `SessionId`, assumed globally unique; accidental UUID collisions are outside the supported model. Two concurrent operations with the same id may both prepare, but the final `enter()` calls arbitrate publication and every loser rolls its private resources back. Each detach is bound to the exact entered object, so a stale disposer cannot remove a later same-id replacement. A detach requested during a synchronous creation notification waits for that dispatch to unwind, preserving created/disposed pairing. Teardown runs stop and drain → detach agent → detach session → unwind scope; the id becomes reusable at detach even if private scope cleanup is still finishing. Ordinary non-vetoing `agent/*` notifications go through `agentEvents(ctx, agent)`, per-step assembly goes through `assembleContextFor(agent)`, and turn-end durability checkpoints go through `ctx.sessions.flush(session)`. +Each agent and its session share one caller-chosen `SessionId`, assumed globally unique; accidental UUID collisions are outside the supported model. Two concurrent operations with the same id may both prepare, but the final `enter()` calls arbitrate publication and every loser rolls its private resources back. Each detach is bound to the exact entered object, so a stale disposer cannot remove a later same-id replacement. A detach requested during a synchronous creation notification waits for that dispatch to unwind, preserving created/disposed pairing. Teardown runs stop and drain → unwind scope → detach agent → detach session; the id becomes reusable after private scope cleanup. Ordinary non-vetoing `agent/*` notifications go through `agentEvents(ctx, agent)`, and per-step assembly goes through `assembleContextFor(agent)`. - `ctx.agentLoop.create(id: SessionId, options?: AgentOptions, meta?: { cwd?: string }): Agent` — synchronous no-setup create under the exact shared agent/session id, disposed with the calling fiber. Declarative config treats `agents[].id` as a stable label and normally mints `${label}-session-` before calling this boundary. An app may instead supply a stable exact `sessionId`: first use creates it, while a remount with persistence already present resumes its materialized history. `resumeSessionId` requires and loads an existing persisted id and is mutually exclusive with `sessionId`. This keeps default fresh restarts collision-free without retaining a second live routing identity. @@ -50,29 +50,29 @@ Configured agents start automatically. A model call requires both `provider` and ### Internal concrete driver -The concrete `Agent` class, its `Inbox`, `runLoop`, and instance-bound publication/start controls are package-internal. The package root exports only the plugin/service/config contract, and the package exports map exposes no `./src/*` escape hatch; lifecycle owners create agents through `ctx.agents` rather than naming, constructing, or starting driver internals. One prepared session can be claimed by only one concrete driver, and everything observable happens through session events and the `agent/*` event taxonomy. +The concrete `ReactLoopAgent`, its queued input, outbox, and run controls are package-internal. The package root exports only the plugin/service/config contract, and the package exports map exposes no `./src/*` escape hatch; lifecycle owners create agents through `ctx.agents` rather than naming, constructing, or starting driver internals. One prepared session can be claimed by only one concrete driver, and everything observable happens through session events and the `agent/*` event taxonomy. The unified `send()` primitive routes content and source by (`target` × `wakeup`); `followup`/`steer`/`inject` are its fixed-preset aliases. A `next-turn` item joins the queued FIFO, waking the driver unless `wakeup: false`; admission happens before any turn opens. An allowed prompt and the prompt waterfall's `additionalContexts` enter the outbox as separate messages, then `run()` opens the turn and drains them together. A running `next-step`/wakeup `steer()` enters the same outbox without prompt admission and normally causes another step. A `next-step`/no-wakeup `inject()` waits there only while a turn is open; while idle it appends a `user/message` immediately without opening a turn or running the model. Persistence owns the resulting eager drain. Every inbox enqueue publishes `agent/inbox/enqueue`; taking it publishes `agent/inbox/dequeue`, and `cancel()` without `keepInbox` publishes `agent/inbox/discard`. -### Loop lifecycle (`loop.ts`) +### Loop lifecycle (`agent.ts`) The driver owns one agent for its lifetime and runs inside `ctx.agents.withInitiator(agent, ...)`. Package-private orchestration entry points recover the exact Agent, derive `agent.session` once, and let operation-local helpers capture it instead of forwarding the concrete driver or per-operation `Session` through shallow interfaces. A helper keeps an explicit `Session` when that is its actual interface, while creation, persistence load, unpublished setup, services, workers, processes, persistence, and wire protocols retain their explicit identities. The [agent service](../agent/README.md#initiating-agent-scope) owns propagation, teardown, and detached-work rules. -Every provider call that reaches a successful finish appends exactly one `assistant/message` completion anchor, including content-less calls and `max-tokens` finishes. A successful `agent/step-result` stores its transformed content; a rejected result records empty content before the original failure continues. The anchor retains exact chunk provenance (`[]` for a stream with no chunks) and usage when available, while empty content stays out of derived message history. +Every provider call that reaches a successful finish appends exactly one `assistant/message` completion anchor, including content-less calls and `max-tokens` finishes. The anchor records the assembled content as-is, retains exact chunk provenance (`[]` for a stream with no chunks), and includes usage when available; empty content stays out of derived message history. -Plugin failure ends the current turn, not the loop. Only final adapter dispatch/iteration failures and terminal in-band error or aborted finishes enter `agent/request-error`; middleware, result processing, tools, and `agent/post-step` remain ordinary turn failures. Recovery receives the exact live error, immutable provider facts, and immutable prior failures after the failed step closes. A retry rebuilds from the durable log in a new numbered step, success clears the consecutive history, and exhaustion records the structured failure once on `turn/end`. AgentLoop privately owns one cancellation holder whose explicit signal spans prompt policy, assembly, every step, model and tool work, recovery, continuation, and terminal stop; it retires the holder immediately before publishing `turn/end`, while the driver may remain `running` through the durability flush. An effective `cancel()` emits the typed runtime-only `user | parent` cause before clearing pending work and cooperatively aborting the holder; notification failures cannot veto cancellation, work queued by a notification observer is cleared, work queued by a later abort observer belongs to the next turn, and idle cancellation emits nothing. Durable `turn/end` remains coarse `aborted`; undispatched model tool calls receive synthetic `tool/call` and `ABORTED_BEFORE_DISPATCH` result pairs. Disposal wins terminal classification, and work that ignores the signal must settle before quiescence. The [explicit-cancellation decision](../../../.agents/notes/implemented/architecture/2026-07-16-explicit-turn-cancellation.md) owns the lifecycle and race contract. Terminal continuation stops remain authoritative through turn close and durability flush. +Plugin failure ends the current turn, not the loop. A model-request failure first closes its step and enters `agent/request-error` with the exact live error, normalized provider facts, and the turn signal. A handling listener calls `agent.retry()`; the loop coalesces repeated calls, closes the failed turn with its error, and opens one numbered retry turn without an intervening idle notification. An unhandled failure is terminal. Other failures close directly. AgentLoop owns one cancellation signal for the current admission or turn. An effective `cancel(cause)` clears pending work unless `keepInbox` is set and cooperatively aborts that signal; idle cancellation is a no-op. Durable `turn/end` records `aborted` for `user` and `parent`, while disposal records `disposed`; undispatched model tool calls receive synthetic `tool/call` and `ABORTED_BEFORE_DISPATCH` result pairs. The cancellation cause changes reporting, not how result context finalized after cancellation is handled. Disposal waits for signal-ignoring work before registry removal. The [explicit-cancellation decision](../../../.agents/notes/implemented/architecture/2026-07-16-explicit-turn-cancellation.md) owns the lifecycle and race contract. -Within a step, exclusive calls form barriers; parallel-safe calls use a bounded rolling pool and are reclassified before start. Only dispatch/body overlaps. Policy, durable results, and result context remain model-ordered. Abort stops new calls, drains started results, then drains accepted batch context before the turn closes through the normal abort path. +Within a step, exclusive calls form barriers; parallel-safe calls use a bounded rolling pool and are reclassified before start. Only dispatch/body overlaps. Policy, durable results, and result context remain model-ordered. Abort stops new calls, drains started results, and retains their finalized result context without distinguishing the cancellation cause. ### What belongs to plugins Everything that goes beyond "call the model, run the tools, repeat" belongs to plugins listening on the event taxonomy: - Hooks and policy: the relevant `agent/*` checkpoints plus the guarded `tools/pre-execute` → `tools/execute` → `tools/post-execute` → definition-owned `finalizeContent` → `tools/result` pipeline; exact event signatures and modes live in the [generated event catalog](../../../docs/cordis-catalog/events.md) -- Compaction: pressure on `agent/post-step`; canonical context overflow on `agent/request-error` -- Transient model recovery: `dsh-llm-retry` on `agent/request-error`, with finite code-specific budgets and non-surface `llm/retry` status events +- Compaction: pressure on `agent/step`; canonical overflow repair on `agent/request-error` +- Transient model recovery: `dsh-llm-retry` records and waits its finite backoff on `agent/request-error`, then calls `agent.retry()` - Sandbox, permission, plan mode: `tools/pre-execute` for extensible deny/ask, `tools.guard()` for monotonic owner policy, `tools/post-execute` for result decisions, and `tools/result` for final observation - Sub-agents: implemented outside the loop as `ctx.subagents` providers; in-process providers use `ctx.agents.create()` and owned `AgentHandle` teardown, while generic [`ctx.tasks`](../../tasks/tasks/) plus [`dsh-tool-subagent`](../../subagent/tool-subagent/) own background collection. -- Persistence: `session/event` + `session/flush` +- Persistence: eager write-behind from `session/event`; `session/flush` is an explicit observation barrier - UI: `session/event` (assistant token stream, boundaries, tool activity) + `agent/*` control events (`agent/status`, `agent/created`/`agent/disposed`) ## Model Experience @@ -81,15 +81,15 @@ Everything that goes beyond "call the model, run the tools, repeat" belongs to p #### What the model sees -For each step, the loop sends the rendered per-agent system prompt, visible tool schemas, the frozen session prefix, and the session's derived messages. It supplies `model` and `cwd` variable values but no additional fixed prose. +For each step, the loop sends the rendered per-agent system prompt, visible tool schemas, and the session's derived messages. It supplies `provider`, `model`, and `cwd` variable values but no additional fixed prose. #### Token effect -System text, schemas, and prefix are paid again on every step. Per-agent scoping chooses the initial contributions, while the authoritative assembly waterfall can alter the final request and makes its listener responsible for protocol coherence. +System text and schemas are paid again on every step. Per-agent scoping chooses the contributions, while the authoritative assembly waterfall can alter the final request and makes its listener responsible for protocol coherence. #### KV Cache effect -Append-only only while system text, schemas, session prefix, and earlier history remain byte-identical under the same provider and model route. A token-bearing assembly rewrite or composition change may invalidate reuse from the first altered request token. +Append-only only while system text, schemas, and earlier history remain byte-identical under the same provider and model route. A token-bearing assembly rewrite or composition change may invalidate reuse from the first altered request token. ### Retained message history @@ -99,7 +99,7 @@ Accepted user messages, assistant messages, tool calls and results, injected con #### Token effect -Input grows with every surface message until a compaction replacement shadows older nodes; a multi-step tool turn resends the accumulated prefix and history each step. +Input grows with every surface message until a compaction replacement shadows older nodes; a multi-step tool turn resends the accumulated history each step. #### KV Cache effect @@ -124,4 +124,4 @@ Append-only; each synthetic result follows the reusable request prefix and does - **Classification is unary** — calls whose safety depends on comparing siblings or resources must remain exclusive ([rationale](../../../.agents/notes/implemented/feature/2026-07-10-parallel-tool-call-execution.md)). - **Config labels are fresh by default** — omitting `sessionId` creates a fresh `${id}-session-` on every startup; exact resume-or-create behavior requires an explicit stable `sessionId`, while `resumeSessionId` requires existing persisted history. - **Config agents have no per-agent persona field or setup hook** — they use the deployment persona; scoped persona/tool composition is available only through the programmatic `ctx.agents.create()` / `resume()` factory options. -- **No built-in turn budget** — the default continuation is `continue` whenever a step had tool calls or steering; bounding a runaway turn requires an `agent/turn-continuation` force-stop plugin. +- **No built-in turn budget** — tool calls or steering continue the current turn; a policy that bounds runaway turns must cancel from an existing lifecycle seam such as `agent/stopping`. diff --git a/packages/core/agent-loop/src/agent.ts b/packages/core/agent-loop/src/agent.ts index b30c858373..717d295865 100644 --- a/packages/core/agent-loop/src/agent.ts +++ b/packages/core/agent-loop/src/agent.ts @@ -19,13 +19,14 @@ import type { AgentStatus, IdleReason, PromptDecision, + RequestError, SendOptions, } from '@deepseek-ai/dsh-agent' import { BlockAssembler, LlmError, deepFreeze, errorChain, isHarnessError, llmFailureOf, markAgentLoopRequest, } from '@deepseek-ai/dsh-llm' import type { - ContentBlock, GenerateOptions, LlmCallConfig, Message, + ContentBlock, GenerateOptions, LlmCallConfig, LlmFailure, Message, } from '@deepseek-ai/dsh-llm' import { canonicalHeader, headerEquals } from '@deepseek-ai/dsh-session' import type { Session, SessionId, TurnEndReason, TurnTrigger, UserMessageData } from '@deepseek-ai/dsh-session' @@ -33,13 +34,15 @@ import { renderPrompt } from '@deepseek-ai/dsh-system-prompt' import type {} from '@deepseek-ai/dsh-tools' import { executeToolCalls } from './tool-calls.ts' -/** One message waiting in the queued or steering inbox. */ -interface PendingMessage extends AgentMessage { - wakeup: boolean -} -function withoutToolCalls(message: Message): Message { - return { ...message, content: message.content.filter(block => block.type !== 'tool-call') } +/** A final-adapter or terminal in-band failure eligible for request recovery. */ +class ModelRequestFailure extends Error { + constructor( + readonly requestError: RequestError, + readonly failure: LlmFailure, + ) { + super(requestError.message, { cause: requestError }) + } } /** @@ -48,14 +51,16 @@ function withoutToolCalls(message: Message): Message { */ export class ReactLoopAgent extends Agent { /** Prompts awaiting individual turns. */ - private queued: PendingMessage[] = [] + private queued: { message: AgentMessage; wakeup: boolean }[] = [] /** Input taken into the session log at step boundaries. */ - private outbox: (UserMessageData | PendingMessage)[] = [] + private outbox: (UserMessageData | AgentMessage)[] = [] /** Whether observers see a running interval; consecutive turns share it. */ private busy = false /** Abort owner for the current admission or turn. */ private abort: AbortController | undefined + /** Coalesced retry capability scoped to the active request-error waterfall. */ + private retryWindow: { requested: boolean } | undefined /** Resolves when the current admission and turn exit. */ done: Promise = Promise.resolve() @@ -104,16 +109,15 @@ export class ReactLoopAgent extends Agent { } const steering = target === 'next-step' && this.turnOpen - const message: PendingMessage = { + const message: AgentMessage = { id, content, source, - wakeup, } if (steering) { this.outbox.push(message) } else { - this.queued.push(message) + this.queued.push({ message, wakeup }) } emitAgentEvent(this.loopCtx, this, 'agent/inbox/enqueue', message) if (!steering && wakeup) this.kick() @@ -134,7 +138,7 @@ export class ReactLoopAgent extends Agent { if (cause.kind !== 'disposed') emitAgentEvent(this.loopCtx, this, 'agent/cancel-requested', cause) } if (!options.keepInbox) { - const discarded: AgentMessage[] = [...this.queued] + const discarded = this.queued.map(item => item.message) for (const message of this.outbox) { if ('id' in message) discarded.push(message) } @@ -143,18 +147,22 @@ export class ReactLoopAgent extends Agent { this.outbox.length = 0 if (discarded.length > 0) emitAgentEvent(this.loopCtx, this, 'agent/inbox/discard', discarded) } + if (this.retryWindow !== undefined) this.retryWindow.requested = false const reason = Object.freeze({ kind: cause.kind }) this.abort?.abort(reason) } /** * Re-open a turn on the current session log without a new prompt — the - * recovery verb after an error idle (naive `retry()`): repair the history - * (edit the log, wait out a rate limit), then run again, right now. - * @throws while a turn is running — there is nothing to retry yet. + * recovery verb. A request-error listener schedules the retry that follows + * its failed turn; an idle caller starts one immediately. */ retry(): void { - if (this.abort !== undefined) throw new Error(`agent "${this.id}" cannot retry while busy`) + if (this.abort !== undefined) { + if (this.retryWindow === undefined) throw new Error(`agent "${this.id}" cannot retry while busy`) + if (!this.abort.signal.aborted) this.retryWindow.requested = true + return + } this.done = this.loopCtx.agents.withInitiator(this, () => this.run({ kind: 'retry' })) } @@ -162,16 +170,17 @@ export class ReactLoopAgent extends Agent { async whenIdle(): Promise { // `done` is replaced per activity, so re-reading it follows chained turns; // a run failure still counts as quiescence for the waiter. - while (this.abort !== undefined || this.queued.some(message => message.wakeup)) { + while (this.abort !== undefined || this.queued.some(item => item.wakeup)) { await this.done.catch(() => undefined) } } /** Claim and admit the next queued prompt, then start its turn. */ private kick(): void { - if (this.abort !== undefined || !this.queued.some(message => message.wakeup)) return - const message = this.queued.shift() - if (message === undefined) return + if (this.abort !== undefined || !this.queued.some(item => item.wakeup)) return + const item = this.queued.shift() + if (item === undefined) return + const { message } = item emitAgentEvent(this.loopCtx, this, 'agent/inbox/dequeue', message) const admission = new AbortController() @@ -210,7 +219,7 @@ export class ReactLoopAgent extends Agent { }) } - /** Own one complete turn over input already admitted by {@link kick}, or retry history as-is. */ + /** Run one turn and any request-error retry over input already admitted by {@link kick}. */ private async run(trigger: TurnTrigger): Promise { if (this.abort !== undefined) throw new Error(`agent "${this.id}" is already running`) const controller = new AbortController() @@ -224,6 +233,9 @@ export class ReactLoopAgent extends Agent { let step = 0 let reason: TurnEndReason = { kind: 'completed' } let idle: IdleReason = { kind: 'completed' } + let retry = false + const cancelRetry = (): void => { retry = false } + signal.addEventListener('abort', cancelRetry, { once: true }) try { signal.throwIfAborted() @@ -242,8 +254,36 @@ export class ReactLoopAgent extends Agent { signal.throwIfAborted() if (!this.drainOutbox(turn)) break } - } catch (error: unknown) { - ({ reason, idle } = this.settle(turn, step, error, signal)) + } catch (caught: unknown) { + const requestFailure = caught instanceof ModelRequestFailure ? caught : undefined + const error = requestFailure?.requestError ?? caught + if (this.stepOpen) { + this.stepOpen = false + this.session.append('step/end', { turn, step }) + } + if (requestFailure !== undefined && agentInterruptReasonOf(signal) === undefined) { + const retryWindow = { requested: false } + this.retryWindow = retryWindow + let recoveryCompleted = false + try { + await this.loopCtx.waterfall( + agentCarrier(this), 'agent/request-error', this, turn, step, requestFailure.requestError, + requestFailure.failure, signal, + () => Promise.resolve(), + ) + recoveryCompleted = true + } catch (recoveryError: unknown) { + this.loopCtx.logger.warn( + `agent "${this.id}": request recovery failed at turn ${turn}, step ${step}: ${errorChain(recoveryError)}`, + ) + } finally { + if (this.retryWindow === retryWindow) this.retryWindow = undefined + } + retry = recoveryCompleted + && agentInterruptReasonOf(signal) === undefined + && retryWindow.requested + } + ({ reason, idle } = this.settle(turn, step, error, signal, requestFailure?.failure)) } finally { try { if (this.stepOpen) { @@ -256,10 +296,18 @@ export class ReactLoopAgent extends Agent { this.session.append('turn/end', { turn, reason }) } } catch (error: unknown) { + retry = false this.loopCtx.logger.warn(`agent "${this.id}": closing turn ${turn} failed: ${errorChain(error)}`) emitAgentEvent(this.loopCtx, this, 'agent/error', turn, step, error) } + this.retryWindow = undefined if (this.abort === controller) this.abort = undefined + signal.removeEventListener('abort', cancelRetry) + } + + if (retry) { + await this.run({ kind: 'retry' }) + } else { emitAgentEvent(this.loopCtx, this, 'agent/idle', turn, idle) this.continueOrIdle() } @@ -311,11 +359,9 @@ export class ReactLoopAgent extends Agent { assembler.push(chunk) } } catch (error: unknown) { - // Normalize a final-adapter failure into the one model-error type; the - // foreign original stays on `cause` for the rendered chain. const facts = llmFailureOf(stream, error) if (facts !== undefined && error instanceof Error) { - throw new LlmError(facts.message, facts.code, { ...facts, cause: error }) + throw new ModelRequestFailure(error, facts) } throw error } @@ -324,20 +370,22 @@ export class ReactLoopAgent extends Agent { // Failure finish chunks take the same path as thrown stream errors. const finish = assembler.finish if (finish.kind === 'error' || finish.kind === 'aborted') { - throw new LlmError(finish.failure.message, finish.failure.code, finish.failure) + const error = new LlmError(finish.failure.message, finish.failure.code, finish.failure) + throw new ModelRequestFailure(error, finish.failure) } // Truncated (max-tokens) output cannot owe tool calls. - const assembled = assembler.finish.kind === 'max-tokens' - ? withoutToolCalls(assembler.message()) - : assembler.message() + const assembled = assembler.message() + const content = finish.kind === 'max-tokens' + ? assembled.content.filter(block => block.type !== 'tool-call') + : assembled.content session.append( 'assistant/message', { turn, step, - content: assembled.content, + content, provenance: { provider: request.provider, model: request.model, @@ -348,7 +396,7 @@ export class ReactLoopAgent extends Agent { { surfaceOp: 'append', sourceEventSeqs: chunkSeqs }, ) - const toolCalls = assembled.content.filter(block => block.type === 'tool-call') + const toolCalls = content.filter(block => block.type === 'tool-call') let concluded = false if (toolCalls.length > 0) { ({ concluded } = await executeToolCalls( @@ -448,20 +496,26 @@ export class ReactLoopAgent extends Agent { * The single settlement funnel: classify one turn failure (interruption * beats error) into the durable turn/end reason and the live idle report. */ - private settle(turn: number, step: number, error: unknown, signal: AbortSignal): { reason: TurnEndReason; idle: IdleReason } { + private settle( + turn: number, + step: number, + error: unknown, + signal: AbortSignal, + failure?: LlmFailure, + ): { reason: TurnEndReason; idle: IdleReason } { const interrupt = agentInterruptReasonOf(signal) if (interrupt !== undefined) { return { reason: { kind: interrupt.kind === 'disposed' ? 'disposed' : 'aborted' }, idle: { kind: 'aborted' } } } - if (error instanceof LlmError) { + if (failure !== undefined) { emitAgentEvent(this.loopCtx, this, 'agent/error', turn, step, error) // The durable record renders the full cause chain: turn/end is the one // durable trace of the failure, so a wrapper message alone would lose // the transport detail the log exists to keep. const rendered = errorChain(error) return { - reason: { kind: 'error', step, failure: { ...error.failure, ...rendered === '' ? {} : { message: rendered } } }, - idle: { kind: 'error', error, failure: error.failure }, + reason: { kind: 'error', step, failure: { ...failure, ...rendered === '' ? {} : { message: rendered } } }, + idle: { kind: 'error', error, failure }, } } emitAgentEvent(this.loopCtx, this, 'agent/error', turn, step, error) @@ -474,7 +528,7 @@ export class ReactLoopAgent extends Agent { /** Continue with a waking prompt, or publish the idle status. */ private continueOrIdle(): void { if (this.abort !== undefined) return - if (this.queued.some(message => message.wakeup)) { + if (this.queued.some(item => item.wakeup)) { this.kick() } else if (this.busy) { this.busy = false diff --git a/packages/core/agent-loop/src/index.ts b/packages/core/agent-loop/src/index.ts index c67245b238..1fc67a30f9 100644 --- a/packages/core/agent-loop/src/index.ts +++ b/packages/core/agent-loop/src/index.ts @@ -283,11 +283,16 @@ export class AgentLoop extends Service implements AgentFactory { ): Promise { await this.waitForDrainingConfiguredIdentity(ownerCtx, sessionId) if (!this.ownership.isActive()) return - const exists = (await persistence.list()).some(header => header.id === sessionId) - if (!this.ownership.isActive()) return - if (exists) { + try { await this.resumeWith(ownerCtx, persistence, { resumeSessionId: sessionId, agentOptions }) return + } catch (error: unknown) { + if (!this.ownership.isActive()) return + // A load is the per-id serialization barrier for eager write-behind and + // lifecycle retirement. Only a genuinely absent artifact falls back to + // first creation; corruption and backend failures stay loud. + const exists = (await persistence.list()).some(header => header.id === sessionId) + if (exists) throw error } this.create(sessionId, agentOptions, meta) } @@ -350,10 +355,11 @@ export class AgentLoop extends Service implements AgentFactory { let detachSession: (() => void) | undefined let detachAgent: (() => void) | undefined let disposing: Promise | undefined + const machineReady = Promise.withResolvers() // Reverse teardown, memoized so every racing owner awaits one quiescence: // stop the machine, leave the registries, unwind the scope, release // bookkeeping. - const dispose = (): Promise => (disposing ??= (async () => { + const dispose = (ownerTriggered = false): Promise => (disposing ??= (async () => { abort.abort(new Error(`agent "${id}" lifecycle disposed`)) callerSignal?.removeEventListener('abort', onCallerAbort) this.ownership.signal.removeEventListener('abort', onFactoryTeardown) @@ -361,6 +367,7 @@ export class AgentLoop extends Service implements AgentFactory { // Disposal IS a disposed-cause cancel followed by quiescence. New work // sent after this point is the sender's bug — the registries are about // to drop the agent, so nothing should still hold it. + if (machine === undefined) await machineReady.promise if (machine !== undefined) { machine.cancel({ kind: 'disposed' }) await Promise.allSettled([machine.done]) @@ -372,7 +379,7 @@ export class AgentLoop extends Service implements AgentFactory { detachSession?.() } finally { untrack() - void unfollowOwner() + if (!ownerTriggered) await unfollowOwner() } } })()) @@ -380,11 +387,11 @@ export class AgentLoop extends Service implements AgentFactory { let unfollowOwner: () => Promise | void try { unfollowOwner = ownerCtx.effect(() => () => { - // Owner disposal starts teardown but must not await its own disposer. - if (disposing === undefined) { - abort.abort(new Error(`agent "${id}" setup aborted: owner disposed during setup`)) - void dispose() - } + // Owner disposal owns the same quiescence boundary. Its teardown skips + // unregistering this already-running owner effect from inside itself. + if (disposing !== undefined) return + abort.abort(new Error(`agent "${id}" setup aborted: owner disposed during setup`)) + return dispose(true) }, `agentLoop.lifecycle(${id})`) } catch (error: unknown) { untrack() @@ -399,6 +406,7 @@ export class AgentLoop extends Service implements AgentFactory { } try { const agent = machine = new ReactLoopAgent(loopCtx, id, options, session) + machineReady.resolve() assertLive() return { @@ -422,6 +430,7 @@ export class AgentLoop extends Service implements AgentFactory { dispose, } } catch (error: unknown) { + machineReady.resolve() void dispose() throw error } @@ -497,11 +506,21 @@ export class AgentLoop extends Service implements AgentFactory { // The load may outlive its owner: race it against caller cancellation, // owner-fiber unload, and factory teardown so a never-settling backend // cannot pin the identity. + const ownerAbort = new AbortController() + const unfollowOwner = ownerCtx.effect(() => () => { + ownerAbort.abort(new Error(`agent "${id}" setup aborted: owner disposed during setup`)) + }, `agentLoop.resume-load(${id})`) const fused = AbortSignal.any([ ...options.signal === undefined ? [] : [options.signal], + ownerAbort.signal, this.ownership.signal, ]) - const loaded = await raceAbort(persistence.load(id), fused, id) + let loaded: Awaited> + try { + loaded = await raceAbort(persistence.load(id), fused, id) + } finally { + await unfollowOwner() + } ownerCtx.fiber.assertActive() if (!this.ownership.isActive()) throw new Error('agent loop is not active') const session = this.runtime.ctx.sessions.prepare(id, { diff --git a/packages/core/agent-loop/tests/MIGRATION.md b/packages/core/agent-loop/tests/MIGRATION.md deleted file mode 100644 index 028980c41d..0000000000 --- a/packages/core/agent-loop/tests/MIGRATION.md +++ /dev/null @@ -1,86 +0,0 @@ -# Agent-loop test migration guide (naive-machine contract) - -The loop was rewritten in the naive-agent shape. `packages/core/agent-loop/src/agent.ts` -is the single source of truth — read it before migrating a spec. Key changes: - -## Event seams (old → new) - -| Old seam | Replacement | -|---|---| -| `agent/pre-step` (serial, before step/start) | `agent/step` (serial, before EVERY request derives; same position) | -| `agent/post-step` (serial, after tools, before step/end) | REMOVED — use `agent/step` of the next step, or `agent/idle` after the turn | -| `agent/session-prefix` (waterfall, request-only prefix) | REMOVED — requests carry no unlogged prefix; durable context via `agent.inject()` at `agent/session-start` | -| `agent/step-result` (waterfall, rewrite assistant msg) | REMOVED — the assembled message is recorded as-is | -| `agent/request-error` (waterfall, retry/fail decision) | REMOVED — observe `agent/idle` with `reason.kind === 'error'`, repair, then `agent.retry()` | -| `agent/turn-continuation` (waterfall, ContinuationDecision) | `agent/continue` (waterfall of `boolean`; handler `(agent, turn, signal, next)`) | -| `agent/turn-stop` (serial, terminal stop) | REMOVED — `agent/continue` returning `false` stops the turn | -| `agent/request` `(agent, turn, step, config, signal, next)` | `(agent, turn, step, signal, next)` — the config comes only from `await next()` | -| `agent/prompt-submit` | unchanged | - -New emit: `agent/idle (agent, turn, reason: IdleReason)` fires once per closed turn -(after turn/end + flush, with `busy` already false, so listeners may synchronously -`retry()`/`send()`). `IdleReason = completed | aborted | { kind: 'error', error, failure? }`. - -## Verb semantics - -- `send()` — unchanged (queued FIFO, one turn each). -- `steer()` while running — enters the outbox; taken whole at the next step - boundary. A turn failure leaves untaken steering staged without waking the - agent; `retry()` or a later prompt takes it. -- `inject()` while the machine is busy — enters the outbox (a `context/message` - appears at the NEXT step boundary, not immediately). While idle — writes a - one-shot turn (`turn/start(injection)` + `context/message` + `turn/end`) and - requests a flush. Enclosure is decided by `busy`, NOT by scanning the log for - an open turn. -- `retry()` — NEW verb: re-opens a turn on the current log with trigger - `{ kind: 'retry' }`. Throws while busy ("cannot retry while busy") and after - disposal. Legal from a synchronous `agent/idle` listener. -- `cancel()` — unchanged surface. No more "pre-run cancelled" bookkeeping: - clearing the queue before a run starts simply means no run starts. - -## Machine shape (timing-sensitive tests) - -- `kick()` runs SYNCHRONOUSLY from `send()` when idle: status flips to - `running` inside the `send()` call. There is no parked driver loop, no - waitForQueued, no microtask collection window. -- One `run()` = one turn. After `turn/end`, it emits `agent/idle`, then either - starts the next waking queued prompt or flips status to `idle`. Residual - outbox input does not wake the agent. Status stays `running` continuously - across queued turns. -- `step/end` is appended INSIDE the step (after tools + the in-step outbox - drain), before `agent/continue` runs. The old `post-step → step/end` - window no longer exists. -- Request messages = `session.deriveMessages()` snapshot taken right before - `step/start` — no `messagePrefix`. `request/header` events no longer carry - a `messagePrefix` field. -- Provider/model config waterfall (`agent/request`) runs INSIDE the step - (after step/start), seeded from agent options (first request) or the folded - logged header (later requests). -- The assembled assistant message is recorded verbatim (with replayState when - present); there is no rewrite path and no "content-less anchor on rejection". -- A model failure (thrown by the adapter or a failure finish chunk) closes the - turn: balanced step/end + turn/end `{ kind:'error', step, failure }` + - `agent/error` emit + `agent/idle` `{ kind:'error', error, failure }`. - There are no in-turn recovery steps. -- Cancellation classification: signal reason `user`/`parent` → turn/end - `aborted`; disposal → `disposed`. IdleReason for both is `aborted`. -- A blocked prompt (`prompt-submit` → block) records `prompt/blocked`, closes - a zero-step turn `rejected` in turn/end, and emits `agent/idle` - `{ kind: 'completed' }` (rejection is a policy outcome, not an error). -- Accept-validation error message is now - "agent message content and source must be losslessly JSON-serializable". -- `dispose()` (the prepared disposer / factory teardown) returns `undefined` - when the machine is not busy — do not `.resolves` it unconditionally; use - `await Promise.resolve(dispose())`. - -## What to do with tests of removed seams - -- Rewrite the scenario against the nearest new seam when the protected - behavior still exists (e.g. turn-stop tests → `agent/continue` returning - false; request-error retry tests → `agent/idle` + `retry()` flows). -- Delete tests whose subject no longer exists at all (session-prefix - reconstruction, step-result rewrite provenance, post-step ordering windows, - pre-run-cancel bookkeeping). Do not keep zombie tests alive by weakening - their assertions. -- Keep the durable-log invariants strong: balanced turn/step boundaries, - ordered tool call/result pairs, header change tracking — those still hold. diff --git a/packages/core/agent-loop/tests/agent-initiator.spec.ts b/packages/core/agent-loop/tests/agent-initiator.spec.ts index a2359562a8..9991bdf8f2 100644 --- a/packages/core/agent-loop/tests/agent-initiator.spec.ts +++ b/packages/core/agent-loop/tests/agent-initiator.spec.ts @@ -153,6 +153,7 @@ describe('AgentLoop initiator scope', () => { const { ctx } = await harness(adapter) const agent = ctx.agentLoop.create(SessionId('signal-owner'), { provider: 'mock', model: 'mock' }) let signals: AbortSignal[] = [] + let admissionSignals: AbortSignal[] = [] const capture = (signal: AbortSignal | undefined): void => { if (signal === undefined) throw new Error('turn seam omitted its explicit signal') expect(ctx.agents.requireInitiator()).toBe(agent) @@ -164,29 +165,20 @@ describe('AgentLoop initiator scope', () => { return next() }) ctx.on('agent/prompt-submit', async (subject, _content, _source, signal, next) => { + if (subject === agent) { + expect(ctx.agents.requireInitiator()).toBe(agent) + admissionSignals.push(signal) + } + return next() + }) + ctx.on('agent/step', (subject, _turn, _step, signal) => { + if (subject === agent) capture(signal) + }) + ctx.on('agent/request', async (subject, _turn, _step, signal, next) => { if (subject === agent) capture(signal) return next() }) - ctx.on('agent/session-prefix', async (subject, _prefix, signal, next) => { - if (subject === agent) capture(signal) - return next() - }) - ctx.on('agent/pre-step', (subject, _turn, _step, signal) => { - if (subject === agent) capture(signal) - }) - ctx.on('agent/request', async (subject, _turn, _step, _config, signal, next) => { - if (subject === agent) capture(signal) - return next() - }) - ctx.on('agent/step-result', async (subject, _turn, _step, _message, signal, next) => { - if (subject === agent) capture(signal) - return next() - }) - ctx.on('agent/turn-continuation', async (subject, _turn, _decision, signal, next) => { - if (subject === agent) capture(signal) - return next() - }) - ctx.on('agent/turn-stop', (subject, _turn, signal) => { + ctx.on('agent/stopping', (subject, _turn, signal) => { if (subject === agent) capture(signal) }) ctx.tools.register(defineContentToolFixture({ @@ -205,14 +197,19 @@ describe('AgentLoop initiator scope', () => { const firstSignal = signals[0] expect(firstSignal).toBeDefined() expect(new Set([...signals, ...adapter.requests.slice(0, 2).map(request => request.signal!)])).toEqual(new Set([firstSignal])) + expect(admissionSignals).toHaveLength(1) + expect(admissionSignals[0]).not.toBe(firstSignal) signals = [] + admissionSignals = [] const secondIdle = waitForIdle(ctx, agent) send(agent, 'second') await secondIdle const secondSignal = signals[0] expect(secondSignal).toBeDefined() expect(new Set([...signals, adapter.requests[2]!.signal!])).toEqual(new Set([secondSignal])) + expect(admissionSignals).toHaveLength(1) + expect(admissionSignals[0]).not.toBe(secondSignal) expect(secondSignal).not.toBe(firstSignal) expect(ctx.agents.currentInitiator()).toBeUndefined() await ctx.fiber.dispose() @@ -343,7 +340,6 @@ describe('AgentLoop initiator scope', () => { expect(captured).toBe(handle.agent) await handle.dispose() - expect(captured?.status).toBe('disposed') expect(ctx.agents.currentInitiator()).toBeUndefined() await ctx.fiber.dispose() }) @@ -365,7 +361,6 @@ describe('AgentLoop initiator scope', () => { await loopFiber.await() expect(adapter.firstAgentDuringAbort?.id).toBe(oldAgent.id) expect(adapter.firstAgentDuringAbort?.session).toBe(oldAgent.session) - expect(oldAgent.status).toBe('disposed') expect(() => oldService.currentInitiator()).toThrow('agent initiator scope is disposed') expect(ctx.agents).not.toBe(oldService) adapter.agents = ctx.agents @@ -406,7 +401,6 @@ describe('AgentLoop initiator scope', () => { await ctx.fiber.dispose() expect(adapter.firstAgentDuringAbort?.id).toBe(agent.id) expect(adapter.firstAgentDuringAbort?.session).toBe(agent.session) - expect(agent.status).toBe('disposed') expect(() => service.currentInitiator()).toThrow('agent initiator scope is disposed') }) }) diff --git a/packages/core/agent-loop/tests/agent.spec.ts b/packages/core/agent-loop/tests/agent.spec.ts index e75bdf6618..b2cc302ee6 100644 --- a/packages/core/agent-loop/tests/agent.spec.ts +++ b/packages/core/agent-loop/tests/agent.spec.ts @@ -1,19 +1,14 @@ import { describe, expect, it, vi } from 'vitest' import { Context } from 'cordis' +import AgentRegistry, { type Agent } from '@deepseek-ai/dsh-agent' +import AgentLoop from '@deepseek-ai/dsh-agent-loop' import LlmService from '@deepseek-ai/dsh-llm' import SessionStore, { SessionId } from '@deepseek-ai/dsh-session' import SystemPrompt from '@deepseek-ai/dsh-system-prompt' import ToolRegistry from '@deepseek-ai/dsh-tools' -import AgentRegistry, { type Agent } from '@deepseek-ai/dsh-agent' -import AgentLoop from '@deepseek-ai/dsh-agent-loop' -import { bindReactLoopAgentContext, prepareReactLoopAgent } from '../src/agent.ts' import { MockAdapter, textResponse } from './mock-adapter.ts' -function driverDone(agent: Agent): Promise { - return (agent as Agent & { done: Promise }).done -} - -async function harness(adapter: MockAdapter) { +async function harness(adapter: MockAdapter): Promise { const ctx = new Context() await ctx.plugin(LlmService) await ctx.plugin(SessionStore) @@ -25,64 +20,11 @@ async function harness(adapter: MockAdapter) { return ctx } -function waitForIdle(ctx: Context, agent: Agent): Promise { - return new Promise((resolve) => { - const dispose = ctx.on('agent/status', (subject, status) => { - if (subject === agent && status === 'idle') { - dispose() - resolve() - } - }) - }) -} - -function waitForStatus(ctx: Context, agent: Agent, expected: Agent['status']): Promise { - return new Promise((resolve) => { - const dispose = ctx.on('agent/status', (subject, status) => { - if (subject === agent && status === expected) { - dispose() - resolve() - } - }) - }) -} - -function send(agent: Agent, text: string) { +function send(agent: Agent, text: string): void { agent.followup([{ type: 'text', text }]) } describe('Agent', () => { - it('rejects access before context binding and a second driver for one session', async () => { - const ctx = new Context() - await ctx.plugin(SessionStore) - const session = ctx.sessions.create(SessionId('exclusive-driver')) - const prepared = prepareReactLoopAgent( - ctx, SessionId('first-driver'), { provider: 'mock', model: 'mock' }, session, - ) - - expect(() => prepared.agent.ctx).toThrow('context is not bound') - expect(() => prepareReactLoopAgent( - ctx, SessionId('second-driver'), { provider: 'mock', model: 'mock' }, session, - )) - .toThrow('already has a concrete agent driver') - - await prepared.dispose() - await ctx.fiber.dispose() - }) - - it('borrows caller options and binds its scoped context exactly once', async () => { - const ctx = await harness(new MockAdapter([textResponse('unused')])) - const options = { provider: 'mock', model: 'mock' } - const agent = ctx.agentLoop.create(SessionId('owned-bindings'), options) - - expect(agent.options).toBe(options) - expect(agent.id).toBe('owned-bindings') - expect(agent.session.id).toBe(agent.id) - expect(() => { bindReactLoopAgentContext(agent, new Context()) }).toThrow(/context is already bound/) - - await ctx.fiber.dispose() - }) - it('idle inject() appends context without opening a turn or requesting a flush', async () => { const adapter = new MockAdapter([textResponse('ok')]) const ctx = await harness(adapter) @@ -91,6 +33,7 @@ describe('Agent', () => { ctx.on('session/flush', () => { flushes += 1 }) agent.inject([{ type: 'text', text: 'context' }], { source: { kind: 'plugin', plugin: 'p' } }) + expect(agent.session.events.map(event => event.type)).toEqual(['user/message']) expect(agent.status).toBe('idle') expect(adapter.requests).toHaveLength(0) @@ -99,120 +42,69 @@ describe('Agent', () => { }) it('inject() defaults its source to an empty plugin, never user', async () => { - const adapter = new MockAdapter([textResponse('ok')]) - const ctx = await harness(adapter) + const ctx = await harness(new MockAdapter([textResponse('ok')])) const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) + agent.inject([{ type: 'text', text: 'no explicit source' }]) - const injected = agent.session.events.at(-1)! - expect(injected.type === 'user/message' && injected.data.source).toEqual({ kind: 'plugin', plugin: '' }) - await agent.whenIdle() + + const injected = agent.session.events.at(-1) + expect(injected?.type === 'user/message' && injected.data.source) + .toEqual({ kind: 'plugin', plugin: '' }) }) it('idle inject() rejects invalid input before append', async () => { - const adapter = new MockAdapter([textResponse('ok')]) - const ctx = await harness(adapter) + const ctx = await harness(new MockAdapter([textResponse('ok')])) const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) expect(() => { - agent.inject([{ type: 'text', text: 'x', bad: 1n } as never], { source: { kind: 'plugin', plugin: 'p' } }) + agent.inject( + [{ type: 'text', text: 'x', bad: 1n } as never], + { source: { kind: 'plugin', plugin: 'p' } }, + ) }).toThrow(/non-JSON-serializable/) expect(agent.session.events).toHaveLength(0) }) - it('steer() when idle falls through to send() and starts a turn', async () => { + it('steer() while idle becomes a woken prompt turn', async () => { const adapter = new MockAdapter([textResponse('ok')]) const ctx = await harness(adapter) const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - // steer while idle delegates to send - agent.steer([{ type: 'text', text: 'steer idle' }], { source: { kind: 'plugin', plugin: 'test' } }) - await waitForIdle(ctx, agent) + agent.steer( + [{ type: 'text', text: 'steer idle' }], + { source: { kind: 'plugin', plugin: 'test' } }, + ) + await agent.whenIdle() - // The message was recorded as a user-level message (send path) - expect(agent.session.events.some(e => e.type === 'user/message')).toBe(true) + expect(agent.session.events.some(event => event.type === 'user/message')).toBe(true) expect(adapter.requests).toHaveLength(1) }) - it('disposer is idempotent (double-stop)', async () => { - // Create a bare Agent and start it through the package-internal - // test seam. Then call its disposer twice — the second call hits the - // early-return branch. - const ctx = new Context() - await ctx.plugin(SessionStore) - await ctx.plugin(AgentRegistry) - const session = ctx.sessions.create(SessionId('test')) - const prepared = prepareReactLoopAgent( - ctx, SessionId('bare'), { provider: 'mock', model: 'mock' }, session, - ) - const { agent } = prepared - - // Start the loop to get the disposer; the agent waits for messages - // (idle, never-resolving cancel), so it will stay idle. - prepared.markPublished() - prepared.start() - const dispose = prepared.dispose - - // First dispose - const firstDisposal = dispose() - expect(agent.status).toBe('disposed') - await firstDisposal - - // Second dispose — idempotent, no throw - await expect(dispose()).resolves.toBeUndefined() - expect(agent.status).toBe('disposed') - }) - - it('a pre-start disposal makes a later driver-start attempt inert', async () => { - const ctx = new Context() - await ctx.plugin(SessionStore) - const session = ctx.sessions.create(SessionId('pre-start-dispose')) - const prepared = prepareReactLoopAgent( - ctx, SessionId('pre-start-dispose'), { provider: 'mock', model: 'mock' }, session, - ) - - await prepared.dispose() - expect(prepared.agent.status).toBe('disposed') - prepared.start() - const dispose = prepared.dispose - await dispose() - await expect(prepared.agent.done).resolves.toBeUndefined() - expect(prepared.agent.session.events).toEqual([]) - await ctx.fiber.dispose() - }) - - it('setting the same status does not emit agent/status again', async () => { - const adapter = new MockAdapter([textResponse('ok')]) - const ctx = await harness(adapter) + it('emits one running and idle transition for one completed turn', async () => { + const ctx = await harness(new MockAdapter([textResponse('ok')])) const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - const statuses: string[] = [] ctx.on('agent/status', (subject, status) => { if (subject === agent) statuses.push(status) }) send(agent, 'hi') - await waitForIdle(ctx, agent) + await agent.whenIdle() - // After the turn, agent is idle. Send again to trigger another attempt - // to go idle — but it's already idle, so no emission. - const idleTransitionCount = statuses.filter(s => s === 'idle').length - expect(idleTransitionCount).toBe(1) // only the final transition from running + expect(statuses).toEqual(['running', 'idle']) }) - it('whenIdle() resolves immediately when the agent is not running', async () => { - const adapter = new MockAdapter([textResponse('ok')]) - const ctx = await harness(adapter) + it('whenIdle() resolves immediately without active work', async () => { + const ctx = await harness(new MockAdapter([textResponse('ok')])) const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - // Fresh agent is idle — whenIdle() takes the not-running fast path and - // resolves without subscribing. await must not hang. await agent.whenIdle() - expect(agent.status).not.toBe('running') + + expect(agent.status).toBe('idle') }) - it('whenIdle() waits for queued work that has not flipped status yet', async () => { - const adapter = new MockAdapter(['hang']) - const ctx = await harness(adapter) + it('whenIdle() waits for active work until explicit cancellation', async () => { + const ctx = await harness(new MockAdapter(['hang'])) const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) send(agent, 'queued') @@ -221,151 +113,25 @@ describe('Agent', () => { await Promise.resolve() expect(settled).toBe(false) - await waitForStatus(ctx, agent, 'running') agent.cancel({ kind: 'user' }) await idle - expect(settled).toBe(true) expect(agent.status).toBe('idle') }) - it('whenIdle() awaits the running→idle transition, ignoring other subjects/running events', async () => { - const adapter = new MockAdapter([textResponse('ok'), textResponse('ok')]) - const ctx = await harness(adapter) + it('contains a throwing status listener on both transitions', async () => { + const ctx = await harness(new MockAdapter([textResponse('ok')])) + const warn = vi.spyOn(ctx.logger, 'warn').mockImplementation(() => undefined) const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - const other = ctx.agentLoop.create(SessionId('a2'), { provider: 'mock', model: 'mock' }) - - // Drive `agent` into `running`, then await whenIdle() — it subscribes to - // agent/status and resolves on the first transition out of running. - const running = new Promise((resolve) => { - const dispose = ctx.on('agent/status', (subject, status) => { - if (subject === agent && status === 'running') { dispose(); resolve() } - }) + ctx.on('agent/status', (_subject, status) => { + throw new Error(`bad ${status} listener`) }) + send(agent, 'go') - await running - expect(agent.status).toBe('running') - - // While `agent`'s whenIdle is pending, churn `other` through running→idle: - // every status event it emits hits whenIdle's guard with `subject !== this`, - // so the wait must ignore them and only resolve on `agent`'s own idle. - send(other, 'go') - await agent.whenIdle() - expect(agent.status).toBe('idle') - }) - it('whenIdle() subscribed while running resolves via done when the agent is then disposed', async () => { - // Covers the waiter's disposed arm: whenIdle() queues an internal waiter - // while running (not the fast path), then the disposer settles it and chains - // `done` (loop exit), not an eager resolve. A bare Agent + direct - // internal driver disposer keeps the emit synchronous. - const ctx = new Context() - await ctx.plugin(LlmService) - await ctx.plugin(SessionStore) - await ctx.plugin(SystemPrompt) - await ctx.plugin(ToolRegistry) - await ctx.plugin(AgentRegistry) - const adapter = new MockAdapter(['hang']) - ctx.llm.registerAdapter(['mock'], adapter) - const session = ctx.sessions.create(SessionId('bare')) - const prepared = prepareReactLoopAgent( - ctx, SessionId('bare'), { provider: 'mock', model: 'mock' }, session, + expect(agent.status).toBe('idle') + expect(warn).toHaveBeenCalledWith( + expect.stringContaining('agent event "agent/status" listener threw'), ) - const { agent } = prepared - prepared.markPublished() - prepared.start() - const dispose = prepared.dispose - agent.followup([{ type: 'text', text: 'go' }]) - await new Promise(r => setTimeout(r, 30)) - expect(agent.status).toBe('running') - - const idle = agent.whenIdle() // queues an internal waiter (running) - const disposal = dispose() // settles the waiter synchronously; whenIdle chains done - await idle - expect(agent.status).toBe('disposed') - await disposal - }) - - it('whenIdle() subscribed while running survives a FIBER dispose (no hung promise)', async () => { - // The waiter is internal agent state, NOT an effect-scoped ctx.on listener: - // disposing the OWNING fiber runs the agent's listener disposers, which would - // have dropped a ctx.on-based waiter before the 'disposed' transition and - // hung the promise. With internal waiters, the fiber disposer still settles it. - const adapter = new MockAdapter(['hang']) - const ctx = await harness(adapter) - let agent!: Agent - const fiber = await ctx.plugin(Object.assign((inner: Context) => { - agent = inner.agentLoop.create(SessionId('scoped'), { provider: 'mock', model: 'mock' }) - }, { inject: ['agentLoop'] })) - send(agent, 'go') - await new Promise(r => setTimeout(r, 30)) - expect(agent.status).toBe('running') - - const idle = agent.whenIdle() // queued while running - await fiber.dispose() // tears the fiber down (drops agent listeners) - await idle // must resolve, not hang - expect(agent.status).toBe('disposed') - }) - - it('whenIdle() on a disposed agent awaits the loop exit (done), not just the status flip', async () => { - // The disposer emits agent/status('disposed') BEFORE the driver loop - // unwinds, so whenIdle() must chain `done` (true quiescence) on the - // disposed path. Dispose a running agent, then assert whenIdle() resolves - // only after `done` — i.e. the loop has actually exited. - const adapter = new MockAdapter(['hang']) - const ctx = await harness(adapter) - let agent!: Agent - const fiber = await ctx.plugin(Object.assign((inner: Context) => { - agent = inner.agentLoop.create(SessionId('scoped'), { provider: 'mock', model: 'mock' }) - }, { inject: ['agentLoop'] })) - send(agent, 'go') - await new Promise(r => setTimeout(r, 30)) - - let doneResolved = false - void driverDone(agent).then(() => { doneResolved = true }) - await fiber.dispose() // sets status disposed, aborts, drains the loop - expect(agent.status).toBe('disposed') - - // whenIdle() must not resolve before `done` has — chaining `done` is the - // quiescence guarantee. By here dispose() awaited the loop, so done is - // settled; whenIdle resolves and done is observed resolved. - await agent.whenIdle() - expect(doneResolved).toBe(true) - }) - - it('contains a throwing agent/status listener on the running transition', async () => { - const adapter = new MockAdapter([textResponse('ok')]) - const ctx = await harness(adapter) - const warn = vi.spyOn(ctx.logger, 'warn').mockImplementation(() => undefined) - const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - ctx.on('agent/status', (_subject, status) => { - if (status === 'running') throw new Error('bad running listener') - }) - - send(agent, 'go') - await agent.whenIdle() - - expect(adapter.requests).toHaveLength(1) - expect(agent.status).toBe('idle') - expect(warn).toHaveBeenCalledWith(expect.stringContaining('agent event "agent/status" listener threw')) - warn.mockRestore() - }) - - it('contains a throwing agent/status listener on the idle transition', async () => { - const adapter = new MockAdapter([textResponse('ok')]) - const ctx = await harness(adapter) - const warn = vi.spyOn(ctx.logger, 'warn').mockImplementation(() => undefined) - const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - ctx.on('agent/status', (_subject, status) => { - if (status === 'idle') throw new Error('bad idle listener') - }) - - send(agent, 'go') - await agent.whenIdle() - - expect(adapter.requests).toHaveLength(1) - expect(agent.status).toBe('idle') - expect(warn).toHaveBeenCalledWith(expect.stringContaining('agent event "agent/status" listener threw')) - warn.mockRestore() }) }) diff --git a/packages/core/agent-loop/tests/cancel.spec.ts b/packages/core/agent-loop/tests/cancel.spec.ts index 72c9c21aec..882bc0957f 100644 --- a/packages/core/agent-loop/tests/cancel.spec.ts +++ b/packages/core/agent-loop/tests/cancel.spec.ts @@ -8,7 +8,7 @@ import { describe, expect, it, vi } from 'vitest' import { Context } from 'cordis' -import LlmService, { type Message } from '@deepseek-ai/dsh-llm' +import LlmService from '@deepseek-ai/dsh-llm' import SessionStore, { SessionId, TurnEndReason } from '@deepseek-ai/dsh-session' import SystemPrompt from '@deepseek-ai/dsh-system-prompt' import ToolRegistry, { defineContentToolFixture, TOOL_ABORTED_BEFORE_DISPATCH } from '@deepseek-ai/dsh-tools' @@ -204,7 +204,7 @@ describe('Agent.cancel()', () => { await disposalDone await driverDone(agent) - expect(agent.status).toBe('disposed') + expect(agent.status).toBe('idle') expect(agent.session.events.some(event => event.type === 'turn/start')).toBe(false) expect(userTexts(agent)).toEqual([]) expect(adapter.requests).toHaveLength(0) @@ -229,101 +229,6 @@ describe('Agent.cancel()', () => { expect(agent.status).toBe('idle') }) - it('cancel() between consecutive turns restores idle and leaves idle steer usable', async () => { - const adapter = new MockAdapter([textResponse('first reply'), textResponse('steer reply')]) - const ctx = await harness(adapter) - const agent = ctx.agentLoop.create(SessionId('between-turn-cancel'), { provider: 'mock', model: 'mock' }) - - let rejectFirstFlush = true - ctx.on('session/flush', (session) => { - if (session !== agent.session || !rejectFirstFlush) return - rejectFirstFlush = false - throw new Error('first flush failed') - }) - - const cancelled = Promise.withResolvers() - ctx.on('agent/error', (subject, _turn, _step, error) => { - if (subject !== agent || error.message !== 'first flush failed') return - // The first hop runs before runLoop resumes from runTurn; the second lands - // before its resolved waitForQueued continuation checks cancellation. - queueMicrotask(() => { - queueMicrotask(() => { - agent.cancel({ kind: 'user' }) - cancelled.resolve(undefined) - }) - }) - }) - - const statuses: string[] = [] - ctx.on('agent/status', (subject, status) => { - if (subject === agent) statuses.push(status) - }) - - send(agent, 'first') - send(agent, 'queued tail') - await cancelled.promise - - expect(agent.status).toBe('idle') - expect(statuses).toEqual(['running', 'idle']) - expect(adapter.requests).toHaveLength(1) - expect(agent.session.events.filter(event => event.type === 'turn/start')).toHaveLength(1) - expect(userTexts(agent)).toEqual(['first']) - - let idleResolved = false - void agent.whenIdle().then(() => { idleResolved = true }) - await Promise.resolve() - expect(idleResolved).toBe(true) - - const idle = waitForIdle(ctx, agent) - agent.steer([{ type: 'text', text: 'idle steer' }]) - await idle - - expect(statuses).toEqual(['running', 'idle', 'running', 'idle']) - expect(adapter.requests).toHaveLength(2) - expect(userTexts(agent)).toEqual(['first', 'idle steer']) - }) - - it('an idle-listener replacement keeps whenIdle pending until the replacement turn finishes', async () => { - const adapter = new MockAdapter([textResponse('first reply'), textResponse('replacement reply')]) - const ctx = await harness(adapter) - const agent = ctx.agentLoop.create(SessionId('between-turn-idle-listener'), { provider: 'mock', model: 'mock' }) - - let rejectFirstFlush = true - ctx.on('session/flush', (session) => { - if (session !== agent.session || !rejectFirstFlush) return - rejectFirstFlush = false - throw new Error('first flush failed') - }) - - ctx.on('agent/error', (subject, _turn, _step, error) => { - if (subject !== agent || error.message !== 'first flush failed') return - queueMicrotask(() => { - queueMicrotask(() => { agent.cancel({ kind: 'user' }) }) - }) - }) - - const replacementRegistered = Promise.withResolvers() - let replacementObservation: Promise<{ status: string; requests: number; turns: number }> | undefined - ctx.on('agent/status', (subject, status) => { - if (subject !== agent || status !== 'idle' || replacementObservation !== undefined) return - send(agent, 'replacement') - replacementObservation = agent.whenIdle().then(() => ({ - status: agent.status, - requests: adapter.requests.length, - turns: agent.session.events.filter(event => event.type === 'turn/start').length, - })) - replacementRegistered.resolve(undefined) - }) - - send(agent, 'first') - send(agent, 'cancelled tail') - await replacementRegistered.promise - if (replacementObservation === undefined) throw new Error('idle listener did not register replacement work') - - await expect(replacementObservation).resolves.toEqual({ status: 'idle', requests: 2, turns: 2 }) - expect(userTexts(agent)).toEqual(['first', 'replacement']) - }) - it('idle-listener cancellation settles its waiter without cancelling later work', async () => { const adapter = new MockAdapter([textResponse('first reply'), textResponse('later reply')]) const ctx = await harness(adapter) @@ -405,22 +310,6 @@ describe('Agent.cancel()', () => { expect(adapter.requests).toHaveLength(1) }) - it('cancel() with no cause defaults to user when aborting an active turn', async () => { - const adapter = new MockAdapter(['hang']) - const ctx = await harness(adapter) - const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - - const reasons: TurnEndReason[] = [] - ctx.on('session/event', (_s, event) => { if (event.type === 'turn/end') reasons.push(event.data.reason) }) - - send(agent, 'go') - await new Promise(r => setTimeout(r, 30)) - agent.cancel({ kind: 'user' }) - await waitForIdle(ctx, agent) - - expect(reasons).toEqual([{ kind: 'aborted' }]) - }) - it('cancel from an assistant/message observer skips execution but balances replay', async () => { const adapter = new MockAdapter([ toolCallResponse('c1', 'danger', {}), @@ -496,98 +385,6 @@ describe('Agent.cancel()', () => { expect(reasons.length).toBe(2) }) - it('cancel from inside the agent/session-prefix waterfall drops the step (prefix-composition window)', async () => { - const adapter = new MockAdapter([textResponse('should not stream')]) - const ctx = await harness(adapter) - const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - - // Prefix composition runs before the pre-step seam on the instance's first - // step; a cancel landing inside it must drop the about-to-start step - // without running the seam or the model. - let streamed = false - ctx.on('session/event', (_s, event) => { if (event.type === 'assistant/chunk') streamed = true }) - ctx.on('agent/session-prefix', async (_agent, _prefix, _signal, next) => { - agent.cancel({ kind: 'user' }) - return next() - }) - - const reasons: TurnEndReason[] = [] - ctx.on('session/event', (_s, event) => { if (event.type === 'turn/end') reasons.push(event.data.reason) }) - - send(agent, 'go') - await waitForIdle(ctx, agent) - - expect(streamed).toBe(false) - expect(reasons).toEqual([{ kind: 'aborted' }]) - }) - - it('disposal from inside the agent/session-prefix waterfall ends the turn disposed (prefix-composition window)', async () => { - const adapter = new MockAdapter([textResponse('should not stream')]) - const ctx = new Context() - await ctx.plugin(LlmService) - await ctx.plugin(SessionStore) - await ctx.plugin(SystemPrompt) - await ctx.plugin(ToolRegistry) - await ctx.plugin(AgentRegistry) - await ctx.plugin(AgentLoop, { agents: [] }) - ctx.llm.registerAdapter(['mock'], adapter) - - const handle = await ctx.agents.create({ - sessionId: SessionId('dispose-prefix-session'), - agentOptions: { provider: 'mock', model: 'mock' }, - }) - const agent = handle.agent - - let disposalDone: Promise | undefined - let streamed = false - ctx.on('session/event', (_s, event) => { if (event.type === 'assistant/chunk') streamed = true }) - ctx.on('agent/session-prefix', async (_agent, _prefix, _signal, next) => { - disposalDone = handle.dispose() - return next() - }) - - send(agent, 'go') - await new Promise(resolve => setTimeout(resolve, 0)) - await disposalDone - await driverDone(agent) - - // No step opened, no model call ran, and the turn closed disposed. - expect(streamed).toBe(false) - expect(adapter.requests).toHaveLength(0) - const turnEnd = agent.session.events.findLast(e => e.type === 'turn/end') - expect(turnEnd?.type === 'turn/end' && turnEnd.data.reason).toEqual({ kind: 'disposed' }) - }) - - it('a cancel-interrupted prefix composition is discarded: the next send recomposes and ships the fresh prefix (stale-cache guard)', async () => { - const adapter = new MockAdapter([textResponse('reply')]) - const ctx = await harness(adapter) - const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - - // The interrupted first composition must not cache its degraded empty value; - // the next prompt recomposes and logs/sends the fresh prefix. - const opener: Message = { role: 'user', content: [{ type: 'text', text: 'fresh opener' }] } - let compositions = 0 - ctx.on('agent/session-prefix', async (_agent, _prefix, _signal, next): Promise => { - compositions += 1 - if (compositions === 1) { - agent.cancel({ kind: 'user' }) - return next() - } - return [opener, ...await next()] - }) - - send(agent, 'dropped') - await waitForIdle(ctx, agent) - send(agent, 'real prompt') - await waitForIdle(ctx, agent) - - expect(compositions).toBe(2) - expect(adapter.requests).toHaveLength(1) - expect(adapter.requests[0]?.messages[0]).toEqual(opener) - const headerEvent = agent.session.events.find(e => e.type === 'request/header') - expect(headerEvent?.type === 'request/header' && headerEvent.data.header.messagePrefix).toEqual([opener]) - }) - it('cancel from a synchronous turn/start session-event listener drops the step (step-start window)', async () => { const adapter = new MockAdapter([textResponse('should not stream')]) const ctx = await harness(adapter) @@ -681,11 +478,7 @@ describe('Agent.cancel()', () => { expect(types.filter(t => t === 'step/start').length).toBe(types.filter(t => t === 'step/end').length) }) - it('cancel during the continuation window ends the turn aborted and runs no further step', async () => { - // A continuation-waterfall listener cancels DURING the continuation decision - // (the finished step's AbortController is already cleared), and votes to - // continue — but the turn-scoped marker checked right after must end the turn - // `aborted` and run NO second step. + it('cancel during the stopping window ends the turn aborted and runs no further step', async () => { const adapter = new MockAdapter([textResponse('one'), textResponse('two')]) const ctx = await harness(adapter) const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) @@ -697,20 +490,18 @@ describe('Agent.cancel()', () => { if (event.type === 'turn/end') reasons.push(event.data.reason) }) - let continued = false - ctx.on('agent/turn-continuation', async (subject, _turn, _default, _signal, next) => { - if (subject === agent && !continued) { - continued = true + let cancelled = false + ctx.on('agent/stopping', (subject) => { + if (subject === agent && !cancelled) { + cancelled = true agent.cancel({ kind: 'user' }) - return { action: 'continue' as const } } - return next() }) send(agent, 'go') await waitForIdle(ctx, agent) - // Only ONE step ran (the second was cancelled in the continuation window), + // Only ONE step ran (the second was cancelled in the stopping window), // and the shared turn signal classified the durable outcome as aborted. expect(steps).toBe(1) expect(reasons).toEqual([{ kind: 'aborted' }]) @@ -869,51 +660,7 @@ describe('Agent.cancel()', () => { expect(turnEnd?.type === 'turn/end' && turnEnd.data.reason).toEqual({ kind: 'aborted' }) }) - it('retires turn cancellation before terminal publication and a blocked durability flush', async () => { - const adapter = new MockAdapter([textResponse('done')]) - const ctx = await harness(adapter) - const agent = ctx.agentLoop.create(SessionId('terminal-cancellation-authority'), { provider: 'mock', model: 'mock' }) - const flushStarted = Promise.withResolvers() - const releaseFlush = Promise.withResolvers() - let abortedDuringTurnEnd: boolean | undefined - let cancelNotifications = 0 - - ctx.on('agent/cancel-requested', (subject) => { - if (subject === agent) cancelNotifications += 1 - }) - ctx.on('session/event', (session, event) => { - if (session !== agent.session || event.type !== 'turn/end') return - const signal = adapter.requests[0]?.signal - if (signal === undefined) throw new Error('model request omitted its turn signal') - agent.cancel({ kind: 'user' }) - abortedDuringTurnEnd = signal.aborted - }) - ctx.on('session/flush', async (session) => { - if (session !== agent.session) return - flushStarted.resolve(undefined) - await releaseFlush.promise - }) - - send(agent, 'finish before persistence drains') - await flushStarted.promise - const signal = adapter.requests[0]?.signal - if (signal === undefined) throw new Error('model request omitted its turn signal') - const idle = agent.whenIdle() - agent.cancel({ kind: 'user' }) - - expect(abortedDuringTurnEnd).toBe(false) - expect(signal.aborted).toBe(false) - expect(cancelNotifications).toBe(0) - expect(agent.session.events.findLast(event => event.type === 'turn/end')).toMatchObject({ - data: { reason: { kind: 'completed' } }, - }) - - releaseFlush.resolve(undefined) - await idle - expect(agent.status).toBe('idle') - }) - - it('records disposed when lifecycle teardown races an already-requested cancel', async () => { + it('preserves the first user cancellation when lifecycle teardown races it', async () => { const adapter = new MockAdapter(['hang']) const ctx = await harness(adapter) const handle = await ctx.agents.create({ @@ -928,19 +675,15 @@ describe('Agent.cancel()', () => { await handle.dispose() const turnEnd = agent.session.events.findLast(event => event.type === 'turn/end') - expect(turnEnd?.type === 'turn/end' && turnEnd.data.reason).toEqual({ kind: 'disposed' }) + expect(turnEnd?.type === 'turn/end' && turnEnd.data.reason).toEqual({ kind: 'aborted' }) }) it.each([ 'prompt-submit', 'system-prompt', - 'session-prefix', - 'pre-step', + 'step', 'request', - 'step-result', - 'post-step', - 'turn-continuation', - 'turn-stop', + 'stopping', 'tool', ] as const)('lets a cooperative %s boundary settle from the explicit turn signal', async (stage) => { const adapter = new MockAdapter(stage === 'tool' @@ -973,44 +716,19 @@ describe('Agent.cancel()', () => { return next() }) break - case 'session-prefix': - ctx.on('agent/session-prefix', async (subject, _prefix, signal, next) => { - if (subject === agent) await blockUntilAbort(signal) - return next() - }) - break - case 'pre-step': - ctx.on('agent/pre-step', async (subject, _turn, _step, signal) => { + case 'step': + ctx.on('agent/step', async (subject, _turn, _step, signal) => { if (subject === agent) await blockUntilAbort(signal) }) break case 'request': - ctx.on('agent/request', async (subject, _turn, _step, _config, signal, next) => { + ctx.on('agent/request', async (subject, _turn, _step, signal, next) => { if (subject === agent) await blockUntilAbort(signal) return next() }) break - case 'step-result': - ctx.on('agent/step-result', async (subject, _turn, _step, _message, signal, next) => { - if (subject === agent) await blockUntilAbort(signal) - return next() - }) - break - case 'post-step': - ctx.on('agent/post-step', async (subject, _turn, _step, signal) => { - if (subject !== agent) return - await blockUntilAbort(signal) - throw new Error('post-step failed after cancellation') - }) - break - case 'turn-continuation': - ctx.on('agent/turn-continuation', async (subject, _turn, _decision, signal, next) => { - if (subject === agent) await blockUntilAbort(signal) - return next() - }) - break - case 'turn-stop': - ctx.on('agent/turn-stop', async (subject, _turn, signal) => { + case 'stopping': + ctx.on('agent/stopping', async (subject, _turn, signal) => { if (subject === agent) await blockUntilAbort(signal) }) break @@ -1030,11 +748,15 @@ describe('Agent.cancel()', () => { send(agent, 'go') await started.promise - const idle = waitForIdle(ctx, agent) + const idle = agent.whenIdle() agent.cancel({ kind: 'user' }) await idle const turnEnd = agent.session.events.findLast(event => event.type === 'turn/end') - expect(turnEnd?.type === 'turn/end' && turnEnd.data.reason).toEqual({ kind: 'aborted' }) + if (stage === 'prompt-submit') { + expect(turnEnd).toBeUndefined() + } else { + expect(turnEnd?.type === 'turn/end' && turnEnd.data.reason).toEqual({ kind: 'aborted' }) + } await ctx.fiber.dispose() }) }) diff --git a/packages/core/agent-loop/tests/config-session-id.spec.ts b/packages/core/agent-loop/tests/config-session-id.spec.ts index d144127498..0bd8a61881 100644 --- a/packages/core/agent-loop/tests/config-session-id.spec.ts +++ b/packages/core/agent-loop/tests/config-session-id.spec.ts @@ -89,7 +89,7 @@ describe('config-driven session id', () => { const ctx = await makeCoreContext() await ctx.plugin(SessionPersistenceJsonl, { root }) ctx.llm.registerAdapter(['mock'], new MockAdapter([textResponse('first'), textResponse('second')])) - const config = { agents: [{ id: 'main', sessionId: SessionId('config-exact-reload'), model: 'mock' }] } + const config = { agents: [{ id: 'main', sessionId: SessionId('config-exact-reload'), provider: 'mock', model: 'mock' }] } const firstLoop = await ctx.plugin(AgentLoop, config) let first: Agent | undefined @@ -131,20 +131,26 @@ describe('config-driven session id', () => { await expect.poll(() => ctx.agents.get(sessionId)).toBeDefined() const first = ctx.agents.get(sessionId) as Agent - const flushGate = Promise.withResolvers() - let flushStarted = false - ctx.on('session/flush', (session) => { - if (session !== first.session) return - flushStarted = true - return flushGate.promise + const cleanupGate = Promise.withResolvers() + const cleanupStarted = Promise.withResolvers() + first.ctx.effect(() => async () => { + cleanupStarted.resolve(undefined) + await cleanupGate.promise }) first.inject([{ type: 'text', text: 'persist before replacement' }], { source: { kind: 'plugin', plugin: 'test' }, }) - expect(flushStarted).toBe(true) + await expect.poll(async () => { + try { + return JSON.stringify((await ctx.sessionPersistence.inspect(sessionId)).events) + } catch { + return '' + } + }).toContain('persist before replacement') const firstDisposal = firstLoop.dispose() - await expect.poll(() => first.status).toBe('disposed') + await cleanupStarted.promise + expect(first.status).toBe('idle') const failures: unknown[] = [] ctx.on('agent-loop/config-start-failed', (_id, error) => { failures.push(error) }) const secondLoop = await ctx.plugin(AgentLoop, config) @@ -152,7 +158,7 @@ describe('config-driven session id', () => { expect(ctx.agents.get(sessionId)).toBe(first) expect(failures).toEqual([]) - flushGate.resolve(undefined) + cleanupGate.resolve(undefined) await firstDisposal await expect.poll(() => ctx.agents.get(sessionId)).toBeDefined() const second = ctx.agents.get(sessionId) as Agent @@ -175,21 +181,31 @@ describe('config-driven session id', () => { await expect.poll(() => ctx.agents.get(sessionId)).toBeDefined() const first = ctx.agents.get(sessionId) as Agent - const flushGate = Promise.withResolvers() - ctx.on('session/flush', (session) => { - if (session === first.session) return flushGate.promise + const cleanupGate = Promise.withResolvers() + const cleanupStarted = Promise.withResolvers() + first.ctx.effect(() => async () => { + cleanupStarted.resolve(undefined) + await cleanupGate.promise }) first.inject([{ type: 'text', text: 'persist before cancellation' }], { source: { kind: 'plugin', plugin: 'test' }, }) + await expect.poll(async () => { + try { + return JSON.stringify((await ctx.sessionPersistence.inspect(sessionId)).events) + } catch { + return '' + } + }).toContain('persist before cancellation') const firstDisposal = firstLoop.dispose() - await expect.poll(() => first.status).toBe('disposed') + await cleanupStarted.promise + expect(first.status).toBe('idle') const secondLoop = await ctx.plugin(AgentLoop, config) await secondLoop.dispose() expect(ctx.agents.get(sessionId)).toBe(first) - flushGate.resolve(undefined) + cleanupGate.resolve(undefined) await firstDisposal expect(ctx.agents.get(sessionId)).toBeUndefined() await ctx.fiber.dispose() @@ -268,14 +284,14 @@ describe('config-driven session id', () => { }) it.each(['resolve', 'reject'] as const)( - 'joins an exact-id persistence lookup that will %s before AgentLoop disposal completes', + 'abandons an exact-id persistence lookup that later %s when AgentLoop disposal starts', async (outcome) => { const root = await mkdtemp(join(tmpdir(), 'dsh-cfg-exact-dispose-')) dirs.push(root) const ctx = await makeCoreContext() await ctx.plugin(SessionPersistenceJsonl, { root }) - const listing = Promise.withResolvers>>() - vi.spyOn(ctx.sessionPersistence, 'list').mockReturnValue(listing.promise) + const loading = Promise.withResolvers>>() + vi.spyOn(ctx.sessionPersistence, 'load').mockReturnValue(loading.promise) const warn = vi.spyOn(ctx.logger, 'warn').mockImplementation(() => undefined) const failures: unknown[] = [] ctx.on('agent-loop/config-start-failed', (_sessionId, error) => { failures.push(error) }) @@ -283,14 +299,20 @@ describe('config-driven session id', () => { const loop = await ctx.plugin(AgentLoop, { agents: [{ id: 'main', sessionId: SessionId('config-exact-dispose'), model: 'mock' }], }) - let disposed = false - const disposal = loop.dispose().then(() => { disposed = true }) + await loop.dispose() + if (outcome === 'resolve') { + loading.resolve({ + meta: { + id: SessionId('config-exact-dispose'), + version: 0, + createdAt: Date.now(), + }, + events: [], + }) + } else { + loading.reject(new Error('startup cancelled by teardown')) + } await Promise.resolve() - expect(disposed).toBe(false) - - if (outcome === 'resolve') listing.resolve([]) - else listing.reject(new Error('startup cancelled by teardown')) - await disposal expect(ctx.agents.get(SessionId('config-exact-dispose'))).toBeUndefined() expect(failures).toEqual([]) expect(warn).not.toHaveBeenCalled() diff --git a/packages/core/agent-loop/tests/contract-regressions.spec.ts b/packages/core/agent-loop/tests/contract-regressions.spec.ts index 79ea648428..3889ee1a33 100644 --- a/packages/core/agent-loop/tests/contract-regressions.spec.ts +++ b/packages/core/agent-loop/tests/contract-regressions.spec.ts @@ -1,17 +1,17 @@ import { describe, expect, it } from 'vitest' import { Context } from 'cordis' -import LlmService, { CallId, ContentBlock, MessageSource, ProviderRequestId, StreamChunk } from '@deepseek-ai/dsh-llm' +import LlmService, { CallId, MessageSource, ProviderRequestId, StreamChunk } from '@deepseek-ai/dsh-llm' import SessionStore, { Session, SessionEvent, SessionId, TurnEndReason } from '@deepseek-ai/dsh-session' import SystemPrompt from '@deepseek-ai/dsh-system-prompt' -import ToolRegistry, { defineTool, TOOL_ABORTED, TOOL_ABORTED_BEFORE_DISPATCH, type PostToolDecision } from '@deepseek-ai/dsh-tools' -import AgentRegistry, { type Agent, type ContinuationDecision } from '@deepseek-ai/dsh-agent' +import ToolRegistry, { defineContentToolFixture, type PostToolDecision } from '@deepseek-ai/dsh-tools' +import AgentRegistry, { type Agent } from '@deepseek-ai/dsh-agent' import AgentLoop from '@deepseek-ai/dsh-agent-loop' -import { prepareReactLoopAgent } from '../src/agent.ts' +import { ReactLoopAgent } from '../src/agent.ts' import InvariantService from '@deepseek-ai/dsh-invariants' import * as SessionInvariant from '@deepseek-ai/dsh-session/invariant' import * as AgentInvariant from '@deepseek-ai/dsh-agent/invariant' import * as AgentLoopInvariant from '@deepseek-ai/dsh-agent-loop/invariant' -import { maxTokensResponse, MockAdapter, textResponse, toolCallResponse } from './mock-adapter.ts' +import { MockAdapter, textResponse, toolCallResponse } from './mock-adapter.ts' async function mountInvariants(ctx: Context): Promise { await ctx.plugin(InvariantService) @@ -53,59 +53,8 @@ function send(agent: Agent, text: string) { agent.followup([{ type: 'text', text }]) } -describe('session log records what agent/step-result actually produced', () => { - it('a step-result rewrite is what the log, derived history, and tool dispatch all see', async () => { - const original = textResponse('original') - original[original.length - 1] = { type: 'finish', reason: { kind: 'stop' }, replayState: { private: 'original-state' } } - const adapter = new MockAdapter([original, textResponse('done')]) - const ctx = await harness(adapter) - const executed: string[] = [] - ctx.tools.register(defineTool({ - name: 'injected-tool', - description: '', - parameters: {}, - async execute() { - executed.push('injected-tool') - return [{ type: 'text', text: 'ran' }] - }, - })) - const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - - // Plugin rewrites the message: replaces the text AND adds a tool call. - let rewritten = false - ctx.on('agent/step-result', async (_agent, _turn, _step, _message, _signal, next) => { - if (rewritten) return next() - rewritten = true - return { - role: 'assistant' as const, - content: [ - { type: 'text' as const, text: 'rewritten' }, - { type: 'tool-call' as const, id: CallId('c-injected'), name: 'injected-tool', arguments: '{}' }, - ], - } - }) - - send(agent, 'go') - await waitForIdle(ctx, agent) - - // the injected tool call was dispatched… - expect(executed).toEqual(['injected-tool']) - // …and the session log recorded the REWRITTEN message, not the original - const recorded = agent.session.events.find(e => e.type === 'assistant/message')! - expect(JSON.stringify(recorded.data)).toContain('rewritten') - expect(JSON.stringify(recorded.data)).not.toContain('original') - expect(recorded.type === 'assistant/message' && recorded.data.provenance.replayState).toBeUndefined() - // tool/call + tool/result correlate with the injected call id - const callEvent = agent.session.events.find(e => e.type === 'tool/call')! - if (callEvent.type !== 'tool/call') throw new Error('wrong event type') - expect(callEvent.data.callId).toBe('c-injected') - // derived history shows the rewritten message (replay-correct) - const derived = agent.session.deriveMessages() - expect(JSON.stringify(derived)).toContain('rewritten') - expect(JSON.stringify(derived)).not.toContain('original') - }) - - it('records adapter replay state when step-result preserves the assembled content', async () => { +describe('assistant replay provenance', () => { + it('records adapter replay state with the assembled assistant content', async () => { const response = textResponse('unchanged') const replayState = { private: 'state' } response[response.length - 1] = { type: 'finish', reason: { kind: 'stop' }, replayState } @@ -116,7 +65,7 @@ describe('session log records what agent/step-result actually produced', () => { send(agent, 'go') await waitForIdle(ctx, agent) - const recorded = agent.session.events.find(e => e.type === 'assistant/message') + const recorded = agent.session.events.find(event => event.type === 'assistant/message') expect(recorded?.type === 'assistant/message' && recorded.data.provenance).toEqual({ provider: 'mock', model: 'next-model', replayState, }) @@ -124,215 +73,14 @@ describe('session log records what agent/step-result actually produced', () => { provider: 'mock', model: 'next-model', replayState, }) }) - - it('drops adapter replay state when step-result mutates assembled content in place', async () => { - const response = textResponse('original') - response[response.length - 1] = { type: 'finish', reason: { kind: 'stop' }, replayState: { private: 'state' } } - const adapter = new MockAdapter([response]) - const ctx = await harness(adapter) - ctx.on('agent/step-result', async (_agent, _turn, _step, message) => { - const block = message.content[0] - if (block?.type === 'text') block.text = 'mutated' - return message - }) - const agent = ctx.agentLoop.create(SessionId('mutated-replay-state'), { provider: 'mock', model: 'next-model' }) - - send(agent, 'go') - await waitForIdle(ctx, agent) - - const recorded = agent.session.events.find(event => event.type === 'assistant/message') - expect(recorded?.type === 'assistant/message' && recorded.data.content).toEqual([{ type: 'text', text: 'mutated' }]) - expect(recorded?.type === 'assistant/message' && recorded.data.provenance.replayState).toBeUndefined() - }) -}) - -describe('successful provider completion survives agent/step-result failure', () => { - async function expectContentlessCompletionAnchor( - response: StreamChunk[], - id: string, - providerText: string, - ): Promise { - const adapter = new MockAdapter([response]) - const ctx = await harness(adapter) - await mountInvariants(ctx) - const agent = ctx.agentLoop.create(SessionId(id), { provider: 'mock', model: 'mock' }) - const failure = new Error(`${id} result processing failed`) - const reported: Error[] = [] - - ctx.on('agent/step-result', async () => { - throw failure - }) - ctx.on('agent/error', (subject, _turn, _step, error) => { - if (subject === agent) reported.push(error) - }) - - send(agent, 'go') - await waitForIdle(ctx, agent) - - const events = [...agent.session.events] - const chunks = events.filter(event => event.type === 'assistant/chunk') - const completions = events.filter(event => event.type === 'assistant/message') - expect(completions).toHaveLength(1) - expect(completions[0]?.type === 'assistant/message' && completions[0].data).toEqual({ - turn: 1, - step: 1, - content: [], - provenance: { provider: 'mock', model: 'mock' }, - usage: { inputTokens: 10, outputTokens: providerText.length }, - }) - expect(completions[0]?.sourceEventSeqs).toEqual(chunks.map(event => event.seq)) - expect(agent.session.deriveMessages()).toEqual([ - { role: 'user', content: [{ type: 'text', text: 'go' }] }, - ]) - expect(reported).toHaveLength(1) - expect(reported[0]).toBe(failure) - const turnEnd = events.findLast(event => event.type === 'turn/end') - expect(turnEnd?.type === 'turn/end' && turnEnd.data.reason).toEqual({ - kind: 'error', - step: 1, - message: failure.message, - }) - } - - it('records one content-less anchor when ordinary stop result processing rejects', async () => { - const providerText = 'ordinary provider output' - await expectContentlessCompletionAnchor( - textResponse(providerText), - 'a-step-result-stop-failure', - providerText, - ) - }) - - it('records one content-less anchor when max-token result processing rejects', async () => { - const providerText = 'truncated provider output' - await expectContentlessCompletionAnchor( - maxTokensResponse(providerText), - 'a-step-result-max-token-failure', - providerText, - ) - }) }) describe('abort during tool execution ends the turn', () => { - it('balances a cancelled tool batch through context and post-step before closing', async () => { - const adapter = new MockAdapter([ - // model asks for two tool calls in one step - [ - { type: 'block-start', index: 0, blockType: 'tool-call' }, - { type: 'block-end', index: 0, block: { type: 'tool-call', id: CallId('c1'), name: 'aborter', arguments: '{}' } }, - { type: 'block-start', index: 1, blockType: 'tool-call' }, - { type: 'block-end', index: 1, block: { type: 'tool-call', id: CallId('c2'), name: 'second', arguments: '{}' } }, - { type: 'finish', reason: { kind: 'tool-calls' } }, - ] satisfies StreamChunk[], - textResponse('should never be requested'), - ]) - const ctx = await harness(adapter) - const executed: string[] = [] - const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - ctx.tools.register(defineTool({ - name: 'aborter', - description: '', - parameters: {}, - async execute(_args, exec) { - executed.push('aborter') - exec.agent?.steer( - [{ type: 'text', text: 'steering before abort' }], - { source: { kind: 'plugin', plugin: 'abort-test' } }, - ) - agent.cancel({ kind: 'user' }) - return [{ type: 'text', text: 'done' }] - }, - })) - ctx.on('tools/post-execute', async exec => ({ - kind: 'accept', - additionalContexts: [{ - content: [{ type: 'text', text: `context for ${exec.callId}` }], - source: { kind: 'plugin', plugin: 'abort-test' }, - }], - })) - ctx.tools.register(defineTool({ - name: 'second', - description: '', - parameters: {}, - async execute() { - executed.push('second') - return [{ type: 'text', text: 'done' }] - }, - })) - - const reasons: TurnEndReason[] = [] - const order: string[] = [] - ctx.on('session/event', (session, event) => { - if (session !== agent.session) return - switch (event.type) { - case 'assistant/message': order.push('assistant/message'); break - case 'tool/call': order.push(`tool/call:${event.data.callId}`); break - case 'tool/result': { - const outcome = event.data.error?.code === TOOL_ABORTED - || event.data.error?.code === TOOL_ABORTED_BEFORE_DISPATCH - ? 'aborted' - : 'completed' - order.push(`tool/result:${event.data.callId}:${outcome}`) - break - } - case 'context/message': order.push('context/message'); break - case 'steering/message': order.push('steering/message'); break - case 'step/end': order.push('step/end'); break - case 'turn/end': { - reasons.push(event.data.reason) - order.push(`turn/end:${event.data.reason.kind}`) - break - } - } - }) - let postSteps = 0 - ctx.on('agent/post-step', (subject, turn, step, signal) => { - if (subject !== agent) return - postSteps += 1 - expect({ turn, step, aborted: signal.aborted }).toEqual({ turn: 1, step: 1, aborted: true }) - order.push('agent/post-step') - }) - - send(agent, 'go') - await waitForIdle(ctx, agent) - - expect(executed).toEqual(['aborter']) - expect(adapter.requests).toHaveLength(1) - expect(postSteps).toBe(1) - expect(order).toEqual([ - 'assistant/message', - 'tool/call:c1', - 'tool/result:c1:aborted', - 'tool/call:c2', - 'tool/result:c2:aborted', - 'context/message', - 'agent/post-step', - 'step/end', - 'turn/end:aborted', - ]) - expect(reasons).toEqual([{ kind: 'aborted' }]) - const calls = agent.session.events.filter(event => event.type === 'tool/call') - const results = agent.session.events.filter(event => event.type === 'tool/result') - expect(calls.map(event => event.data.callId)).toEqual([CallId('c1'), CallId('c2')]) - expect(results).toHaveLength(2) - expect(results[0]!.data).toMatchObject({ - callId: CallId('c1'), - content: [{ type: 'text', text: 'Error: tool call aborted' }], - isError: true, - error: { name: 'AbortError', code: TOOL_ABORTED }, - }) - expect(results[1]!.data).toMatchObject({ - callId: CallId('c2'), - isError: true, - error: { name: 'AbortError', code: TOOL_ABORTED_BEFORE_DISPATCH }, - }) - }) - it('records context accepted before a tool-step abort in the same turn', async () => { const adapter = new MockAdapter([toolCallResponse('c1', 'aborter', {})]) const ctx = await harness(adapter) const agent = ctx.agentLoop.create(SessionId('a-abort-injection'), { provider: 'mock', model: 'mock' }) - ctx.tools.register(defineTool({ + ctx.tools.register(defineContentToolFixture({ name: 'aborter', description: '', parameters: {}, @@ -355,15 +103,16 @@ describe('abort during tool execution ends the turn', () => { const events = [...agent.session.events] expect(events - .filter(event => event.type === 'tool/result' || event.type === 'context/message' + .filter(event => event.type === 'tool/result' + || (event.type === 'user/message' && event.data.source.kind === 'plugin') || event.type === 'step/end' || event.type === 'turn/end') .map(event => event.type)) - .toEqual(['tool/result', 'context/message', 'context/message', 'step/end', 'turn/end']) + .toEqual(['tool/result', 'user/message', 'step/end', 'turn/end']) expect(events - .filter(event => event.type === 'context/message') - .map(event => event.data.content)) + .flatMap(event => event.type === 'user/message' && event.data.source.kind === 'plugin' + ? [event.data.content] + : [])) .toEqual([ - [{ type: 'text', text: 'accepted before abort' }], [{ type: 'text', text: 'accepted result context after abort' }], ]) }) @@ -378,7 +127,7 @@ describe('abort during tool execution ends the turn', () => { ] satisfies StreamChunk[]]) const ctx = await harness(adapter) const agent = ctx.agentLoop.create(SessionId('a-later-abort-context'), { provider: 'mock', model: 'mock' }) - ctx.tools.register(defineTool({ + ctx.tools.register(defineContentToolFixture({ name: 'first', description: '', parameters: {}, @@ -386,7 +135,7 @@ describe('abort during tool execution ends the turn', () => { return [{ type: 'text', text: 'first done' }] }, })) - ctx.tools.register(defineTool({ + ctx.tools.register(defineContentToolFixture({ name: 'aborter', description: '', parameters: {}, @@ -411,15 +160,19 @@ describe('abort during tool execution ends the turn', () => { const events = [...agent.session.events] expect(events - .filter(event => event.type === 'tool/result' || event.type === 'context/message' + .filter(event => event.type === 'tool/result' + || (event.type === 'user/message' && event.data.source.kind === 'plugin') || event.type === 'step/end' || event.type === 'turn/end') .map(event => event.type)) - .toEqual(['tool/result', 'tool/result', 'context/message', 'step/end', 'turn/end']) - expect(events.find(event => event.type === 'context/message')?.data.content) - .toEqual([{ type: 'text', text: 'accepted after first result' }]) + .toEqual(['tool/result', 'tool/result', 'step/end', 'turn/end']) + expect(events.flatMap(event => + event.type === 'user/message' && event.data.source.kind === 'plugin' + ? [event.data.content] + : [])[0]) + .toBeUndefined() }) - it('drains deferred context before disposal reaches quiescence', async () => { + it('records result context finalized after disposal cancellation', async () => { const adapter = new MockAdapter([toolCallResponse('c1', 'waiter', {})]) const ctx = await harness(adapter) const started = Promise.withResolvers() @@ -427,7 +180,7 @@ describe('abort during tool execution ends the turn', () => { const fiber = await ctx.plugin(Object.assign((inner: Context) => { agent = inner.agentLoop.create(SessionId('a-dispose-injection'), { provider: 'mock', model: 'mock' }) }, { inject: ['agentLoop'] })) - ctx.tools.register(defineTool({ + ctx.tools.register(defineContentToolFixture({ name: 'waiter', description: '', parameters: {}, @@ -456,10 +209,10 @@ describe('abort during tool execution ends the turn', () => { await fiber.dispose() expect(agent.session.events - .filter(event => event.type === 'context/message') - .map(event => event.data.content)) + .flatMap(event => event.type === 'user/message' && event.data.source.kind === 'plugin' + ? [event.data.content] + : [])) .toEqual([ - [{ type: 'text', text: 'accepted before disposal' }], [{ type: 'text', text: 'accepted result context during disposal' }], ]) expect(agent.session.events.find(event => event.type === 'turn/end')?.data.reason) @@ -479,7 +232,7 @@ describe('abort during tool execution ends the turn', () => { ]) const ctx = await harness(adapter) const agent = ctx.agentLoop.create(SessionId('a-historical-tool-pair'), { provider: 'mock', model: 'mock' }) - ctx.tools.register(defineTool({ + ctx.tools.register(defineContentToolFixture({ name: 'aborter', description: '', parameters: {}, @@ -488,7 +241,7 @@ describe('abort during tool execution ends the turn', () => { return [{ type: 'text', text: 'done' }] }, })) - ctx.tools.register(defineTool({ + ctx.tools.register(defineContentToolFixture({ name: 'second', description: '', parameters: {}, @@ -499,7 +252,7 @@ describe('abort during tool execution ends the turn', () => { send(agent, 'leave an unmatched historical call') await waitForIdle(ctx, agent) - ctx.on('agent/pre-step', (subject, turn) => { + ctx.on('agent/step', (subject, turn) => { if (subject === agent && turn === 2) { agent.inject([{ type: 'text', text: 'new turn context' }], { source: { kind: 'plugin', plugin: 'test' } }) } @@ -507,14 +260,17 @@ describe('abort during tool execution ends the turn', () => { send(agent, 'start a text-only turn') await waitForIdle(ctx, agent) - expect(agent.session.events.find(event => event.type === 'context/message')?.data.content) + expect(agent.session.events.flatMap(event => + event.type === 'user/message' && event.data.source.kind === 'plugin' + ? [event.data.content] + : [])[0]) .toEqual([{ type: 'text', text: 'new turn context' }]) expect(JSON.stringify(adapter.requests[1]?.messages)).toContain('new turn context') }) }) describe('steering from late extension points is never stranded', () => { - it('steer() from an agent/turn-continuation listener overrides a stop decision', async () => { + it('steer() from an agent/stopping listener continues the same turn', async () => { const adapter = new MockAdapter([ textResponse('no tools, would stop here'), textResponse('continued because of steering'), @@ -523,12 +279,11 @@ describe('steering from late extension points is never stranded', () => { const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) let steeredOnce = false - ctx.on('agent/turn-continuation', async (_agent, _turn, _decision, _signal, next) => { + ctx.on('agent/stopping', () => { if (!steeredOnce) { steeredOnce = true agent.steer([{ type: 'text', text: 'one more thing' }]) } - return next() }) send(agent, 'go') @@ -600,22 +355,23 @@ describe('steering from late extension points is never stranded', () => { }) describe('plugin exceptions are contained', () => { - it('a throwing agent/turn-continuation listener ends the turn with an error, loop survives', async () => { + it('a throwing agent/stopping listener ends the turn with an error, loop survives', async () => { const adapter = new MockAdapter([textResponse('one'), textResponse('two')]) const ctx = await harness(adapter) const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) let threwOnce = false - ctx.on('agent/turn-continuation', async (): Promise => { + ctx.on('agent/stopping', async () => { if (!threwOnce) { threwOnce = true throw new Error('broken continuation plugin') } - return { action: 'stop' } }) const errors: Error[] = [] - ctx.on('agent/error', (_agent, _turn, _step, error) => void errors.push(error)) + ctx.on('agent/error', (_agent, _turn, _step, error) => { + if (error instanceof Error) errors.push(error) + }) send(agent, 'first') await waitForIdle(ctx, agent) @@ -628,45 +384,9 @@ describe('plugin exceptions are contained', () => { expect(agent.status).toBe('idle') }) - it('a rejecting first-turn flush settles before the queued tail starts', async () => { - const adapter = new MockAdapter([textResponse('one'), textResponse('two')]) - const ctx = await harness(adapter) - const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - - const firstFlush = Promise.withResolvers() - const releaseFirstFlush = Promise.withResolvers() - let flushes = 0 - ctx.on('session/flush', async (session) => { - if (session !== agent.session) return - flushes += 1 - if (flushes === 1) { - firstFlush.resolve(undefined) - await releaseFirstFlush.promise - throw new Error('disk full') - } - }) - - const errors: Error[] = [] - ctx.on('agent/error', (_agent, _turn, _step, error) => void errors.push(error)) - - const idle = waitForIdle(ctx, agent) - send(agent, 'first') - send(agent, 'second') - - await firstFlush.promise - expect(adapter.requests).toHaveLength(1) - expect(agent.session.events.filter(event => event.type === 'turn/start')).toHaveLength(1) - - releaseFirstFlush.resolve(undefined) - await idle - - expect(errors.map(e => e.message)).toEqual(['disk full']) - expect(adapter.requests).toHaveLength(2) - expect(agent.session.events.filter(event => event.type === 'turn/start')).toHaveLength(2) - }) }) -describe('disposed status is part of the agent/status contract', () => { +describe('disposal leaves the two-state status contract balanced', () => { it('disposing the fiber ends the active turn and never starts its queued tail', async () => { const adapter = new MockAdapter(['hang']) const ctx = await harness(adapter) @@ -687,7 +407,7 @@ describe('disposed status is part of the agent/status contract', () => { await fiber.dispose() await driverDone(agent) - expect(statuses).toEqual(['running', 'disposed']) + expect(statuses).toEqual(['running', 'idle']) expect(reasons).toEqual([{ kind: 'disposed' }]) expect(agent.session.events.filter(event => event.type === 'turn/start')).toHaveLength(1) const messages = agent.session.events @@ -708,7 +428,7 @@ describe('disposed status is part of the agent/status contract', () => { }, { inject: ['agentLoop'] })) ctx.on('agent/status', (_agent, status) => { - if (status === 'disposed') throw new Error('broken status listener') + if (status === 'idle') throw new Error('broken status listener') }) send(agent, 'go') @@ -716,8 +436,7 @@ describe('disposed status is part of the agent/status contract', () => { await fiber.dispose() await driverDone(agent) // must not hang - expect(agent.status).toBe('disposed') - expect(ctx.agents.get(SessionId('scoped'))).toBeUndefined() // unregistered despite the throw + await expect.poll(() => ctx.agents.get(SessionId('scoped')) === undefined).toBe(true) }) }) @@ -739,7 +458,9 @@ describe('adapter registration, routing, and accepted-input ownership', () => { const agent = ctx.agentLoop.create(SessionId('a1'), {}) // no model const errors: Error[] = [] - ctx.on('agent/error', (_agent, _turn, _step, error) => void errors.push(error)) + ctx.on('agent/error', (_agent, _turn, _step, error) => { + if (error instanceof Error) errors.push(error) + }) send(agent, 'go') await waitForIdle(ctx, agent) @@ -753,8 +474,8 @@ describe('adapter registration, routing, and accepted-input ownership', () => { const ctx = await harness(adapter) const agent = ctx.agentLoop.create(SessionId('a1'), {}) // no model — router plugin decides - ctx.on('agent/request', async (_agent, _turn, _step, config, _signal) => { - return { ...config, provider: 'mock', model: 'mock' } + ctx.on('agent/request', async (_agent, _turn, _step, _signal, next) => { + return { ...await next(), provider: 'mock', model: 'mock' } }) send(agent, 'go') @@ -763,11 +484,11 @@ describe('adapter registration, routing, and accepted-input ownership', () => { expect(agent.session.deriveMessages().at(-1)?.content).toEqual([{ type: 'text', text: 'routed' }]) }) - it('agent/queued carries the resolved source; steering/message records its source', async () => { + it('agent/inbox/enqueue carries the exact message; steering/message records its source', async () => { const adapter = new MockAdapter([toolCallResponse('c1', 'noop', {}), textResponse('done')]) const ctx = await harness(adapter) const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - ctx.tools.register(defineTool({ + ctx.tools.register(defineContentToolFixture({ name: 'noop', description: '', parameters: {}, @@ -777,107 +498,30 @@ describe('adapter registration, routing, and accepted-input ownership', () => { }, })) - const queuedSources: { source: MessageSource; steering: boolean }[] = [] - ctx.on('agent/queued', (_agent, _content, info) => void queuedSources.push(info)) + const queuedSources: MessageSource[] = [] + const queuedShapes: string[][] = [] + ctx.on('agent/inbox/enqueue', (_agent, message) => { + queuedSources.push(message.source) + queuedShapes.push(Object.keys(message).sort()) + }) send(agent, 'go') // no explicit source → default {kind:'user'} must be visible await waitForIdle(ctx, agent) - expect(queuedSources[0]).toEqual({ source: { kind: 'user' }, steering: false }) - expect(queuedSources[1]).toEqual({ source: { kind: 'plugin', plugin: 'goal' }, steering: true }) + expect(queuedSources).toEqual([ + { kind: 'user' }, + { kind: 'plugin', plugin: 'goal' }, + ]) + expect(queuedShapes).toEqual([ + ['content', 'id', 'source'], + ['content', 'id', 'source'], + ]) // The drain appends the durable steering/message with the caller's source // intact — the log, not a transient emit, is where consumers read it. const steeringSources = agent.session.events.flatMap(e => e.type === 'steering/message' ? [e.data.source] : []) expect(steeringSources).toEqual([{ kind: 'plugin', plugin: 'goal' }]) }) - it('send() owns content and source before notification and delivery', async () => { - const adapter = new MockAdapter([textResponse('done')]) - const ctx = await harness(adapter) - const agent = ctx.agentLoop.create(SessionId('owned-send'), { provider: 'mock', model: 'mock' }) - const content = [{ type: 'text' as const, text: 'accepted-send' }] - const source = { kind: 'plugin' as const, plugin: 'accepted-source' } - let notifiedContent: ContentBlock[] | undefined - let notifiedSource: MessageSource | undefined - ctx.on('agent/queued', (subject, acceptedContent, info) => { - if (subject !== agent || info.steering) return - // Retain the exact notification references: cloning here would test the - // listener's copy rather than the event/inbox ownership boundary. - notifiedContent = acceptedContent - notifiedSource = info.source - }) - - agent.followup(content, { source }) - content[0]!.text = 'caller-mutated-send' - source.plugin = 'caller-mutated-source' - await waitForIdle(ctx, agent) - - expect(notifiedContent).toEqual([{ type: 'text', text: 'accepted-send' }]) - expect(notifiedSource).toEqual({ kind: 'plugin', plugin: 'accepted-source' }) - expect(Object.isFrozen(notifiedContent)).toBe(true) - expect(Object.isFrozen(notifiedContent?.[0])).toBe(true) - expect(Object.isFrozen(notifiedSource)).toBe(true) - const recorded = agent.session.events.flatMap(event => event.type === 'user/message' ? [event.data] : []) - expect(recorded).toContainEqual({ - content: [{ type: 'text', text: 'accepted-send' }], - source: { kind: 'plugin', plugin: 'accepted-source' }, - }) - const request = JSON.stringify(adapter.requests[0]!.messages) - expect(request).toContain('accepted-send') - expect(request).not.toContain('caller-mutated-send') - }) - - it('running steer() owns content and source before notification and delivery', async () => { - const adapter = new MockAdapter([toolCallResponse('c1', 'gate', {}), textResponse('done')]) - const ctx = await harness(adapter) - const agent = ctx.agentLoop.create(SessionId('owned-steer'), { provider: 'mock', model: 'mock' }) - const entered = Promise.withResolvers() - const release = Promise.withResolvers() - ctx.tools.register(defineTool({ - name: 'gate', - description: '', - parameters: {}, - async execute() { - entered.resolve(undefined) - await release.promise - return [{ type: 'text', text: 'tool done' }] - }, - })) - let notifiedContent: ContentBlock[] | undefined - let notifiedSource: MessageSource | undefined - ctx.on('agent/queued', (subject, acceptedContent, info) => { - if (subject !== agent || !info.steering) return - notifiedContent = acceptedContent - notifiedSource = info.source - }) - - agent.followup([{ type: 'text', text: 'start' }]) - await entered.promise - expect(agent.status).toBe('running') - const content = [{ type: 'text' as const, text: 'accepted-steer' }] - const source = { kind: 'plugin' as const, plugin: 'accepted-source' } - agent.steer(content, { source }) - content[0]!.text = 'caller-mutated-steer' - source.plugin = 'caller-mutated-source' - const idle = waitForIdle(ctx, agent) - release.resolve(undefined) - await idle - - expect(notifiedContent).toEqual([{ type: 'text', text: 'accepted-steer' }]) - expect(notifiedSource).toEqual({ kind: 'plugin', plugin: 'accepted-source' }) - expect(Object.isFrozen(notifiedContent)).toBe(true) - expect(Object.isFrozen(notifiedContent?.[0])).toBe(true) - expect(Object.isFrozen(notifiedSource)).toBe(true) - const recorded = agent.session.events.flatMap(event => event.type === 'steering/message' ? [event.data] : []) - expect(recorded).toContainEqual({ - turn: 1, - content: [{ type: 'text', text: 'accepted-steer' }], - source: { kind: 'plugin', plugin: 'accepted-source' }, - }) - const request = JSON.stringify(adapter.requests[1]!.messages) - expect(request).toContain('accepted-steer') - expect(request).not.toContain('caller-mutated-steer') - }) }) describe('turn numbering continues across seeded sessions', () => { @@ -900,12 +544,9 @@ describe('turn numbering continues across seeded sessions', () => { ctx2.llm.registerAdapter(['mock'], second) const seeded = ctx2.sessions.create(SessionId('forked'), { seed: [...agent.session.events] }) - const prepared = prepareReactLoopAgent( + const forked = new ReactLoopAgent( ctx2, SessionId('forked-agent'), { provider: 'mock', model: 'mock' }, seeded, ) - const forked = prepared.agent - prepared.markPublished() - ctx2.effect(() => { prepared.start(); return prepared.dispose }) const turns: number[] = [] ctx2.on('session/event', (_s, event) => { if (event.type === 'turn/start') turns.push(event.data.turn) }) @@ -1076,7 +717,9 @@ describe('turn and step boundary recovery', () => { if (event.type === 'step/start' && !threw) { threw = true; throw new Error('boom step-start') } }) const errors: Error[] = [] - ctx.on('agent/error', (_a, _t, _s, error) => void errors.push(error)) + ctx.on('agent/error', (_a, _t, _s, error) => { + if (error instanceof Error) errors.push(error) + }) send(agent, 'go') await waitForIdle(ctx, agent) @@ -1107,7 +750,9 @@ describe('turn and step boundary recovery', () => { } }) const errors: Error[] = [] - ctx.on('agent/error', (_agent, _turn, _step, error) => { errors.push(error) }) + ctx.on('agent/error', (_agent, _turn, _step, error) => { + if (error instanceof Error) errors.push(error) + }) send(agent, 'go') await waitForIdle(ctx, agent) @@ -1123,41 +768,6 @@ describe('turn and step boundary recovery', () => { expect(errors.map(error => error.message)).toEqual(['reject step-start before commit']) }) - it('a one-shot turn/end validation failure preserves the earlier turn error on retry', async () => { - const errorStream: StreamChunk[] = [{ type: 'finish', reason: { kind: 'error', failure: { message: 'provider failed', code: 'UNKNOWN' } } }] - const adapter = new MockAdapter([errorStream]) - const ctx = await balancedHarness(adapter) - const agent = ctx.agentLoop.create(SessionId('a-turnend-veto'), { provider: 'mock', model: 'mock' }) - let rejected = false - ctx.on('internal/dispatch', (_mode, name, args) => { - if (name !== 'session/event') return - const event = args[1] as SessionEvent - if (event.type === 'turn/end' && !rejected) { - rejected = true - throw new Error('reject first turn-end') - } - }) - const errors: Error[] = [] - ctx.on('agent/error', (_agent, _turn, _step, error) => { errors.push(error) }) - - send(agent, 'go') - await waitForIdle(ctx, agent) - - expect(errors.map(error => error.message)).toEqual(['provider failed']) - expect(boundaryCounts(agent)).toMatchObject({ - turnStart: 1, - turnEnd: 1, - stepStart: 1, - stepEnd: 1, - errors: 1, - }) - const turnEnd = agent.session.events.findLast(event => event.type === 'turn/end') - expect(turnEnd?.type === 'turn/end' && turnEnd.data.reason).toMatchObject({ - kind: 'error', - failure: { message: 'provider failed', code: 'UNKNOWN' }, - }) - }) - it('a one-shot step/end validation failure keeps the step open until retry succeeds', async () => { const adapter = new MockAdapter([textResponse('completed before close validation')]) const ctx = await balancedHarness(adapter) @@ -1172,7 +782,9 @@ describe('turn and step boundary recovery', () => { } }) const errors: Error[] = [] - ctx.on('agent/error', (_agent, _turn, _step, error) => { errors.push(error) }) + ctx.on('agent/error', (_agent, _turn, _step, error) => { + if (error instanceof Error) errors.push(error) + }) send(agent, 'go') await waitForIdle(ctx, agent) @@ -1261,14 +873,16 @@ describe('turn and step boundary recovery', () => { }, { inject: ['agentLoop'] })) let threw = false - ctx.on('agent/pre-step', () => { + ctx.on('agent/step', () => { if (threw) return threw = true void fiber.dispose() throw new Error('boom pre-step during disposal') }) const errorEmits: Error[] = [] - ctx.on('agent/error', (_a, _t, _s, error) => void errorEmits.push(error)) + ctx.on('agent/error', (_a, _t, _s, error) => { + if (error instanceof Error) errorEmits.push(error) + }) send(agent, 'go') await driverDone(agent) @@ -1295,7 +909,9 @@ describe('turn and step boundary recovery', () => { if (!threw && event.type === 'turn/start') { threw = true; throw new Error('boom turn/start append') } }) const errors: Error[] = [] - ctx.on('agent/error', (_a, _t, _s, error) => void errors.push(error)) + ctx.on('agent/error', (_a, _t, _s, error) => { + if (error instanceof Error) errors.push(error) + }) send(agent, 'go') await waitForIdle(ctx, agent) @@ -1326,7 +942,9 @@ describe('turn and step boundary recovery', () => { if (event.type === 'step/end' && !threw) { threw = true; throw new Error('boom step-end') } }) const errors: Error[] = [] - ctx.on('agent/error', (_a, _t, _s, error) => void errors.push(error)) + ctx.on('agent/error', (_a, _t, _s, error) => { + if (error instanceof Error) errors.push(error) + }) send(agent, 'go') await waitForIdle(ctx, agent) @@ -1365,7 +983,9 @@ describe('turn and step boundary recovery', () => { if (!threw && event.type === 'step/end') { threw = true; throw new Error('boom step/end listener') } }) const errors: Error[] = [] - ctx.on('agent/error', (_a, _t, _s, error) => void errors.push(error)) + ctx.on('agent/error', (_a, _t, _s, error) => { + if (error instanceof Error) errors.push(error) + }) send(agent, 'go') await waitForIdle(ctx, agent) @@ -1419,7 +1039,7 @@ describe('tool result call identity', () => { textResponse('done'), ]) const ctx = await harness(adapter) - ctx.tools.register(defineTool({ + ctx.tools.register(defineContentToolFixture({ name: 'echo', description: 'echo', parameters: { x: { type: 'number' } }, @@ -1458,34 +1078,6 @@ describe('tool result call identity', () => { }) }) -describe('surface: assistant/message records exact empty provenance when no chunks streamed', () => { - it('a step-result listener injecting content over an empty stream records sourceEventSeqs []', async () => { - // The explicit empty source set distinguishes a known empty provider - // stream from legacy events whose provenance was not recorded. - const adapter = new MockAdapter([[]]) - const ctx = await harness(adapter) - await mountInvariants(ctx) - const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - - ctx.on('agent/step-result', async (_agent, _turn, _step, _message, _signal) => ({ - role: 'assistant' as const, - content: [{ type: 'text' as const, text: 'injected' }], - })) - - send(agent, 'go') - await waitForIdle(ctx, agent) - - const recorded = agent.session.events.find(e => e.type === 'assistant/message')! - expect(recorded.type).toBe('assistant/message') - expect(recorded.surfaceOp).toBe('append') - expect(recorded.sourceEventSeqs).toEqual([]) - // The injected content reaches derived history. - expect(JSON.stringify(agent.session.deriveMessages())).toContain('injected') - }) -}) - - - describe('disposal and cancellation during pre-step assembly', () => { it('disposal during system-prompt assembly drops the about-to-start step as disposed', { timeout: 30000 }, async () => { // Start disposal, then release assembly. Do not await disposal first: it @@ -1590,7 +1182,7 @@ describe('disposal and cancellation during pre-step assembly', () => { expect(reasons).toEqual([{ kind: 'aborted' }]) }) - it('disposal during agent/pre-step seam ends the turn disposed', { timeout: 15000 }, async () => { + it('disposal during agent/step seam ends the turn disposed', { timeout: 15000 }, async () => { // Start disposal, then release pre-step; awaiting disposal first would // deadlock on the blocked driver. const adapter = new MockAdapter(['hang']) @@ -1607,7 +1199,7 @@ describe('disposal and cancellation during pre-step assembly', () => { await mountInvariants(ctx) ctx.llm.registerAdapter(['mock'], adapter) - ctx.on('agent/pre-step', async () => { + ctx.on('agent/step', async () => { await blocker }) @@ -1642,7 +1234,7 @@ describe('disposal and cancellation during pre-step assembly', () => { // (turn boundaries have no agent/* mirror). }) - it('cancel during agent/pre-step seam ends the turn aborted', { timeout: 15000 }, async () => { + it('cancel during agent/step seam ends the turn aborted', { timeout: 15000 }, async () => { // Release pre-step after cancellation to exercise the post-seam check. const adapter = new MockAdapter(['hang']) let releasePreStep!: () => void @@ -1658,7 +1250,7 @@ describe('disposal and cancellation during pre-step assembly', () => { await mountInvariants(ctx) ctx.llm.registerAdapter(['mock'], adapter) - ctx.on('agent/pre-step', async () => { + ctx.on('agent/step', async () => { await blocker }) diff --git a/packages/core/agent-loop/tests/coverage-edges.spec.ts b/packages/core/agent-loop/tests/coverage-edges.spec.ts index 0f26d6106d..df26054294 100644 --- a/packages/core/agent-loop/tests/coverage-edges.spec.ts +++ b/packages/core/agent-loop/tests/coverage-edges.spec.ts @@ -41,34 +41,6 @@ function send(agent: Agent, text: string) { agent.followup([{ type: 'text', text }]) } -describe('inbox acceptance', () => { - it('rejects non-serializable content or source synchronously before notification or enqueue', async () => { - const adapter = new MockAdapter([textResponse('turn 1')]) - const ctx = await harness(adapter) - const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - let queued = 0 - ctx.on('agent/inbox/enqueue', () => { queued += 1 }) - - expect(() => { - agent.followup([{ type: 'text', text: 'first', bad: 1n } as never]) - }).toThrow(/losslessly JSON-serializable/) - expect(() => { - agent.send([{ type: 'text', text: 'first' }], { - target: 'next-turn', - wakeup: true, - source: { kind: 'plugin', plugin: 'p', bad: 1n } as never, - }) - }).toThrow(/losslessly JSON-serializable/) - expect(queued).toBe(0) - expect(agent.session.events).toHaveLength(0) - - // The rejected value never woke or poisoned the loop; a valid message runs. - send(agent, 'second') - await waitForIdle(ctx, agent) - expect(adapter.requests).toHaveLength(1) - }) -}) - describe('tool JSON parse', () => { it('passes through non-JSON arguments string without crashing', async () => { const adapter = new MockAdapter([ diff --git a/packages/core/agent-loop/tests/inbox-invariant.spec.ts b/packages/core/agent-loop/tests/inbox-invariant.spec.ts deleted file mode 100644 index 0d907f26d7..0000000000 --- a/packages/core/agent-loop/tests/inbox-invariant.spec.ts +++ /dev/null @@ -1,155 +0,0 @@ -/** - * Regression: the dsh-agent FIFO-conservation invariant must stay balanced on - * the loop-authored continuation-reason steering path. A continue-with-reason - * decision enters the steering FIFO and later drains (or is discarded by - * cancel); both must be matched by an enqueue event so the invariant's - * outstanding count never goes negative. - * @module dsh-agent-loop/tests/inbox-invariant - */ - -import { describe, expect, it, vi } from 'vitest' -import { Context } from 'cordis' -import LlmService from '@deepseek-ai/dsh-llm' -import SessionStore, { SessionId } from '@deepseek-ai/dsh-session' -import SystemPrompt from '@deepseek-ai/dsh-system-prompt' -import ToolRegistry from '@deepseek-ai/dsh-tools' -import AgentRegistry, { type Agent } from '@deepseek-ai/dsh-agent' -import * as AgentInvariant from '@deepseek-ai/dsh-agent/invariant' -import InvariantService from '@deepseek-ai/dsh-invariants' -import AgentLoop from '@deepseek-ai/dsh-agent-loop' -import { MockAdapter, textResponse } from './mock-adapter.ts' - -async function harness(adapter: MockAdapter) { - const ctx = new Context() - await ctx.plugin(LlmService) - await ctx.plugin(SessionStore) - await ctx.plugin(SystemPrompt) - await ctx.plugin(ToolRegistry) - await ctx.plugin(AgentRegistry) - await ctx.plugin(InvariantService) - await ctx.plugin(AgentInvariant) - await ctx.plugin(AgentLoop, { agents: [] }) - ctx.llm.registerAdapter(['mock'], adapter) - return ctx -} - -function waitForIdle(ctx: Context, agent: Agent): Promise { - return new Promise((resolve) => { - const dispose = ctx.on('agent/status', (subject, status) => { - if (subject === agent && status === 'idle') { dispose(); resolve() } - }) - }) -} - -describe('inbox FIFO-conservation invariant', () => { - it('stays balanced when a continuation reason enters and drains the steering FIFO', async () => { - const adapter = new MockAdapter([textResponse('step 1'), textResponse('step 2')]) - const ctx = await harness(adapter) - const warn = vi.spyOn(ctx.logger, 'warn').mockImplementation(() => undefined) - const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - - let forced = false - ctx.on('agent/turn-continuation', async (_agent, _turn, _default, _signal, next) => { - if (forced) return next() - forced = true - return { action: 'continue' as const, reason: { content: [{ type: 'text', text: 'keep going' }], source: { kind: 'plugin', plugin: 'loop' } } } - }) - - agent.followup([{ type: 'text', text: 'go' }]) - await waitForIdle(ctx, agent) - - expect(adapter.requests).toHaveLength(2) - // The continuation reason drained as a steering/message on the second step. - expect(agent.session.events.some(e => e.type === 'steering/message')).toBe(true) - // No invariant violation was logged. - expect(warn.mock.calls.flat().some(arg => String(arg).includes('agent/inbox'))).toBe(false) - expect(warn.mock.calls.flat().some(arg => String(arg).includes('INVARIANT'))).toBe(false) - }) - - it('stays balanced when cancel discards a pending continuation reason', async () => { - const adapter = new MockAdapter([textResponse('only step')]) - const ctx = await harness(adapter) - const warn = vi.spyOn(ctx.logger, 'warn').mockImplementation(() => undefined) - const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - - // Force a continuation reason, then cancel from the same checkpoint so the - // reason sits in the steering FIFO when the inbox is discarded. - ctx.on('agent/turn-continuation', async (subject, _turn, _default, _signal, next) => { - if (subject !== agent) return next() - queueMicrotask(() => { agent.cancel({ kind: 'user' }) }) - return { action: 'continue' as const, reason: { content: [{ type: 'text', text: 'keep going' }], source: { kind: 'plugin', plugin: 'loop' } } } - }) - - agent.followup([{ type: 'text', text: 'go' }]) - await waitForIdle(ctx, agent) - - expect(warn.mock.calls.flat().some(arg => String(arg).includes('agent/inbox'))).toBe(false) - expect(warn.mock.calls.flat().some(arg => String(arg).includes('INVARIANT'))).toBe(false) - }) - - it('stays balanced when a terminal stop discards pending steering', async () => { - const adapter = new MockAdapter([textResponse('done')]) - const ctx = await harness(adapter) - const warn = vi.spyOn(ctx.logger, 'warn').mockImplementation(() => undefined) - const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - - const discards: number[] = [] - ctx.on('agent/inbox/discard', (subject, messages) => { if (subject === agent) discards.push(messages.length) }) - - // A continuation reason enqueues a steering item; a terminal stop then drops - // it. The drop must emit a discard so the enqueue ⇒ dequeue-or-discard - // ledger stays balanced (no dangling outstanding id). - ctx.on('agent/turn-continuation', async (subject, _turn, _default, _signal, next) => { - if (subject !== agent) return next() - return { action: 'continue' as const, reason: { content: [{ type: 'text', text: 'keep going' }], source: { kind: 'plugin', plugin: 'loop' } } } - }) - let stopped = false - ctx.on('agent/turn-stop', (subject) => { - if (subject !== agent || stopped) return undefined - stopped = true - return { action: 'stop' as const } - }) - - agent.followup([{ type: 'text', text: 'go' }]) - await waitForIdle(ctx, agent) - - expect(discards).toEqual([1]) // the dropped steering item was reported - expect(warn.mock.calls.flat().some(arg => String(arg).includes('agent/inbox'))).toBe(false) - expect(warn.mock.calls.flat().some(arg => String(arg).includes('INVARIANT'))).toBe(false) - }) - - it('stays balanced when late steering lands after a terminal stop (post-turn flush window)', async () => { - const adapter = new MockAdapter([textResponse('done')]) - const ctx = await harness(adapter) - const warn = vi.spyOn(ctx.logger, 'warn').mockImplementation(() => undefined) - const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - - let enqueues = 0 - const discards: number[] = [] - ctx.on('agent/inbox/enqueue', (subject) => { if (subject === agent) enqueues += 1 }) - ctx.on('agent/inbox/discard', (subject, messages) => { if (subject === agent) discards.push(messages.length) }) - - // Terminal-stop the turn, then steer during the post-turn flush window - // (status is still running). That late steer is drained by runLoop and - // dropped because the turn terminally stopped; it must still be discarded so - // its enqueue is matched (the drain sits on a different code path than the - // in-turn terminal-stop drop). - ctx.on('agent/turn-stop', subject => (subject === agent ? { action: 'stop' as const } : undefined)) - let steered = false - ctx.on('session/flush', (session) => { - if (session !== agent.session || steered) return - steered = true - agent.steer([{ type: 'text', text: 'late' }], { source: { kind: 'plugin', plugin: 'late' } }) - }) - - agent.followup([{ type: 'text', text: 'go' }]) - await waitForIdle(ctx, agent) - - // The prompt plus the late steer both enqueued; both are matched (the prompt - // dequeued, the late steer discarded) so no id is left outstanding. - expect(enqueues).toBe(2) - expect(discards).toEqual([1]) - expect(warn.mock.calls.flat().some(arg => String(arg).includes('agent/inbox'))).toBe(false) - expect(warn.mock.calls.flat().some(arg => String(arg).includes('INVARIANT'))).toBe(false) - }) -}) diff --git a/packages/core/agent-loop/tests/interception.spec.ts b/packages/core/agent-loop/tests/interception.spec.ts index 86786f647c..cf80a6e809 100644 --- a/packages/core/agent-loop/tests/interception.spec.ts +++ b/packages/core/agent-loop/tests/interception.spec.ts @@ -1,18 +1,18 @@ import { describe, expect, it } from 'vitest' import { Context } from 'cordis' -import LlmService, { CallId, type Message } from '@deepseek-ai/dsh-llm' +import LlmService, { CallId } from '@deepseek-ai/dsh-llm' import SessionStore, { SessionId, type SessionEvent, type TurnEndReason } from '@deepseek-ai/dsh-session' import SystemPrompt from '@deepseek-ai/dsh-system-prompt' import ToolRegistry, { defineContentToolFixture, type PostToolDecision, type PreToolDecision } from '@deepseek-ai/dsh-tools' -import AgentRegistry, { type Agent, type ContinuationDecision, type PromptDecision, type SessionStartSource } from '@deepseek-ai/dsh-agent' +import AgentRegistry, { type Agent, type PromptDecision, type SessionStartSource } from '@deepseek-ai/dsh-agent' import AgentLoop from '@deepseek-ai/dsh-agent-loop' import { MockAdapter, textResponse, toolCallResponse } from './mock-adapter.ts' /** * The interception seams introduced by the hooks taxonomy: `agent/prompt-submit`, - * `agent/session-start`, the reshaped `agent/turn-continuation` - * ({@link ContinuationDecision}), and the `tools/pre-execute` / `tools/post-execute` + * `agent/session-start`, `agent/stopping`, and the + * `tools/pre-execute` / `tools/post-execute` * split with `additionalContexts` buffering. These verify the canonical event * surface a hook bridge (or a native plugin) programs against, WITHOUT any * external protocol — a native plugin uses the typed decisions directly. @@ -127,7 +127,7 @@ describe('agent/prompt-submit', () => { })) let preStepDerived: string | undefined - ctx.on('agent/pre-step', (subject, _turn, step) => { + ctx.on('agent/step', (subject, _turn, step) => { if (subject === agent && step === 1) preStepDerived = JSON.stringify(subject.session.deriveMessages()) }) @@ -140,7 +140,7 @@ describe('agent/prompt-submit', () => { expect(preStepDerived).not.toContain('ORIGINAL prompt') }) - it('block drops the (only) prompt → zero-step turn ends rejected, model never called', async () => { + it('block drops the claimed prompt before any turn or model call', async () => { const adapter = new MockAdapter([textResponse('should not run')]) const ctx = await harness(adapter) const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) @@ -152,26 +152,17 @@ describe('agent/prompt-submit', () => { ctx.on('session/event', (_s, event: SessionEvent) => { if (event.type === 'turn/end') reasons.push(event.data.reason) }) agent.followup([{ type: 'text', text: 'do something' }]) - await waitForIdle(ctx, agent) + await agent.whenIdle() // the model was never called expect(adapter.requests).toHaveLength(0) - // the turn opened and closed balanced, with no user/message and no step const log = events(agent) - expect(log.some(e => e.type === 'turn/start')).toBe(true) - expect(log.some(e => e.type === 'turn/end')).toBe(true) + expect(log.some(e => e.type === 'turn/start')).toBe(false) + expect(log.some(e => e.type === 'turn/end')).toBe(false) expect(log.some(e => e.type === 'user/message')).toBe(false) expect(log.some(e => e.type === 'step/start')).toBe(false) - // the veto is recorded durably as a prompt/blocked in the open turn - const blocked = log.find(e => e.type === 'prompt/blocked') - expect(blocked?.type === 'prompt/blocked' && blocked.data).toMatchObject({ - content: [{ type: 'text', text: 'do something' }], - reason: 'blocked by policy', - }) - // ended rejected with the block reason - expect(reasons).toEqual([{ kind: 'rejected', reason: 'blocked by policy' }]) - const turnEnd = log.findLast(e => e.type === 'turn/end') - expect(turnEnd?.type === 'turn/end' && turnEnd.data.reason).toEqual({ kind: 'rejected', reason: 'blocked by policy' }) + expect(log.some(e => e.type === 'prompt/blocked')).toBe(false) + expect(reasons).toEqual([]) }) it('adjacent blocked and allowed prompts keep independent turn outcomes', async () => { @@ -187,7 +178,7 @@ describe('agent/prompt-submit', () => { const reasons: TurnEndReason[] = [] ctx.on('session/event', (_s, event: SessionEvent) => { if (event.type === 'turn/end') reasons.push(event.data.reason) }) - // Both sends land before the driver wakes, but each remains its own turn. + // The rejected admission is dropped; the allowed prompt owns the only turn. send(agent, 'secret') send(agent, 'safe') await waitForIdle(ctx, agent) @@ -198,21 +189,12 @@ describe('agent/prompt-submit', () => { expect(userMsgs).toHaveLength(1) expect(userMsgs[0]?.type === 'user/message' && userMsgs[0].data.content).toEqual([{ type: 'text', text: 'safe' }]) expect(adapter.requests.length).toBeGreaterThanOrEqual(1) - // the blocked prompt is durably recorded, with its content + reason - const blocked = log.filter(e => e.type === 'prompt/blocked') - expect(blocked).toHaveLength(1) - expect(blocked[0]?.type === 'prompt/blocked' && blocked[0].data).toMatchObject({ - content: [{ type: 'text', text: 'secret' }], - reason: 'policy: no secrets', - }) - expect(log.filter(e => e.type === 'turn/start')).toHaveLength(2) - expect(reasons).toEqual([ - { kind: 'rejected', reason: 'policy: no secrets' }, - { kind: 'completed' }, - ]) + expect(log.filter(e => e.type === 'prompt/blocked')).toHaveLength(0) + expect(log.filter(e => e.type === 'turn/start')).toHaveLength(1) + expect(reasons).toEqual([{ kind: 'completed' }]) }) - it('a throwing prompt-submit listener ends its turn balanced while an adjacent message survives', async () => { + it('a throwing prompt-submit listener drops that admission while an adjacent message survives', async () => { const adapter = new MockAdapter([textResponse('after')]) const ctx = await harness(adapter) const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) @@ -225,7 +207,9 @@ describe('agent/prompt-submit', () => { const errors: Error[] = [] const reasons: TurnEndReason[] = [] const statuses: string[] = [] - ctx.on('agent/error', (_a, _t, _s, error) => void errors.push(error)) + ctx.on('agent/error', (_a, _t, _s, error) => { + if (error instanceof Error) errors.push(error) + }) ctx.on('agent/status', (subject, status) => { if (subject === agent) statuses.push(status) }) ctx.on('session/event', (session, event) => { if (session === agent.session && event.type === 'turn/end') reasons.push(event.data.reason) @@ -235,16 +219,11 @@ describe('agent/prompt-submit', () => { send(agent, 'first') send(agent, 'second') await idle - expect(errors.map(e => e.message)).toEqual(['prompt hook broke']) - // The failed prompt forms one balanced error turn; the adjacent prompt forms - // the following normal turn without an intermediate idle transition. + expect(errors).toEqual([]) const log = events(agent) - expect(log.filter(e => e.type === 'turn/start')).toHaveLength(2) - expect(log.filter(e => e.type === 'turn/end')).toHaveLength(2) - expect(reasons).toEqual([ - { kind: 'error', step: 0, message: 'prompt hook broke' }, - { kind: 'completed' }, - ]) + expect(log.filter(e => e.type === 'turn/start')).toHaveLength(1) + expect(log.filter(e => e.type === 'turn/end')).toHaveLength(1) + expect(reasons).toEqual([{ kind: 'completed' }]) expect(statuses).toEqual(['running', 'idle']) expect(adapter.requests).toHaveLength(1) expect(JSON.stringify(adapter.requests[0]!.messages)).toContain('second') @@ -306,236 +285,6 @@ describe('agent/session-start', () => { }) }) -describe('agent/session-prefix', () => { - it('dispatches to global and matching agent-scope listeners only', async () => { - const adapter = new MockAdapter([textResponse('a done'), textResponse('b done')]) - const ctx = await harness(adapter) - const agentA = ctx.agentLoop.create(SessionId('prefix-a'), { provider: 'mock', model: 'mock' }) - const agentB = ctx.agentLoop.create(SessionId('prefix-b'), { provider: 'mock', model: 'mock' }) - const seen: string[] = [] - ctx.on('agent/session-prefix', async (agent, _prefix, _signal, next) => { - seen.push(`global:${agent.id}`) - return next() - }) - agentA.ctx.on('agent/session-prefix', async (agent, _prefix, _signal, next) => { - seen.push(`a:${agent.id}`) - return next() - }) - agentB.ctx.on('agent/session-prefix', async (agent, _prefix, _signal, next) => { - seen.push(`b:${agent.id}`) - return next() - }) - - send(agentA, 'run a') - await waitForIdle(ctx, agentA) - send(agentB, 'run b') - await waitForIdle(ctx, agentB) - - expect(seen).toEqual([ - 'global:prefix-a', 'a:prefix-a', - 'global:prefix-b', 'b:prefix-b', - ]) - }) - - it('composes once per loop instance and fronts every request; the header records it; history stays untouched', async () => { - const adapter = new MockAdapter([ - toolCallResponse('c1', 'echo', { text: 'ping' }), - textResponse('done'), - textResponse('again'), - ]) - const ctx = await harness(adapter) - ctx.tools.register(defineContentToolFixture({ - name: 'echo', description: 'echo', parameters: { text: { type: 'string' } }, - async execute(args) { return [{ type: 'text', text: String(args.text) }] }, - })) - const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - - const reminder: Message = { role: 'user', content: [{ type: 'text', text: 'catalog' }] } - let composed = 0 - ctx.on('agent/session-prefix', async (_agent, _prefix, _signal, next): Promise => { - composed += 1 - return [...await next(), reminder] - }) - - send(agent, 'go') - await waitForIdle(ctx, agent) - send(agent, 'next turn') - await waitForIdle(ctx, agent) - - // Three requests (two turns), ONE composition: the frozen product is - // reused verbatim, so the prefix cannot drift mid-session. - expect(adapter.requests).toHaveLength(3) - expect(composed).toBe(1) - for (const request of adapter.requests) { - expect(request.messages[0]).toEqual(reminder) - } - // The anchoring snapshot is the prefix's durable record — and the ONLY - // header event: reuse means no changed snapshot ever. - const headerEvents = events(agent).filter(e => e.type === 'request/header') - expect(headerEvents).toHaveLength(1) - expect(headerEvents[0]?.type === 'request/header' && headerEvents[0].data.header.messagePrefix).toEqual([reminder]) - // Never session history: the derivation starts at the real user prompt. - expect(agent.session.deriveMessages()[0]).toEqual({ role: 'user', content: [{ type: 'text', text: 'go' }] }) - }) - - it('composes before the first pre-step and records the prefix on the request header', async () => { - const adapter = new MockAdapter([textResponse('ok')]) - const ctx = await harness(adapter) - const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - - const reminder: Message = { role: 'user', content: [{ type: 'text', text: 'opener' }] } - const order: string[] = [] - ctx.on('agent/session-prefix', async (_agent, _prefix, _signal, next): Promise => { - order.push('compose') - return [reminder, ...await next()] - }) - ctx.on('agent/pre-step', () => { - order.push('pre-step') - }) - - send(agent, 'hi') - await waitForIdle(ctx, agent) - - expect(order).toEqual(['compose', 'pre-step']) - expect(agent.session.requestHeader()?.messagePrefix).toEqual([reminder]) - }) - - it('the canonical prepend pattern composes contributions in registration order', async () => { - const adapter = new MockAdapter([textResponse('ok')]) - const ctx = await harness(adapter) - const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - - // Both listeners use the canonical `[mine, ...await next()]` prepend: the - // waterfall unwinds innermost-first (the second listener's array is built - // first), so prepending puts the FIRST-registered contribution first. - ctx.on('agent/session-prefix', async (_agent, _prefix, _signal, next): Promise => { - return [{ role: 'user', content: [{ type: 'text', text: 'first' }] }, ...await next()] - }) - ctx.on('agent/session-prefix', async (_agent, _prefix, _signal, next): Promise => { - return [{ role: 'user', content: [{ type: 'text', text: 'second' }] }, ...await next()] - }) - - send(agent, 'hi') - await waitForIdle(ctx, agent) - - const texts = adapter.requests[0]!.messages.map(m => m.content[0]?.type === 'text' ? m.content[0].text : '') - expect(texts).toEqual(['first', 'second', 'hi']) - }) - - it('with no contributions the header omits messagePrefix and the request is the bare derivation', async () => { - const adapter = new MockAdapter([textResponse('ok')]) - const ctx = await harness(adapter) - const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - - // A listener that delegates without contributing — the canonical no-op. - ctx.on('agent/session-prefix', async (_agent, _prefix, _signal, next) => next()) - - send(agent, 'hi') - await waitForIdle(ctx, agent) - - const headerEvent = events(agent).find(e => e.type === 'request/header') - expect(headerEvent?.type === 'request/header' && 'messagePrefix' in headerEvent.data.header).toBe(false) - expect(adapter.requests[0]!.messages).toEqual([{ role: 'user', content: [{ type: 'text', text: 'hi' }] }]) - }) - - it('the frozen seed rejects in-place mutation — a contribution is a returned extension', async () => { - const adapter = new MockAdapter([textResponse('ok')]) - const ctx = await harness(adapter) - const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - - let mutationError: unknown - ctx.on('agent/session-prefix', async (_agent, prefix, _signal, next): Promise => { - try { - prefix.push({ role: 'user', content: [{ type: 'text', text: 'smuggled' }] }) - } catch (error: unknown) { - mutationError = error - } - return next() - }) - - send(agent, 'hi') - await waitForIdle(ctx, agent) - - expect(mutationError).toBeInstanceOf(TypeError) - expect(adapter.requests[0]!.messages).toEqual([{ role: 'user', content: [{ type: 'text', text: 'hi' }] }]) - }) - - it('mutating a listener-held reference after composition cannot alter later requests (the cache is a frozen clone)', async () => { - const adapter = new MockAdapter([ - toolCallResponse('c1', 'echo', { text: 'ping' }), - textResponse('done'), - ]) - const ctx = await harness(adapter) - ctx.tools.register(defineContentToolFixture({ - name: 'echo', description: 'echo', parameters: { text: { type: 'string' } }, - async execute(args) { return [{ type: 'text', text: String(args.text) }] }, - })) - const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - - const held: Message = { role: 'user', content: [{ type: 'text', text: 'v1' }] } - ctx.on('agent/session-prefix', async (_agent, _prefix, _signal, next): Promise => [...await next(), held]) - - send(agent, 'go') - await waitForIdle(ctx, agent) - - // The listener mutates the object it contributed AFTER composition; the - // cached prefix is a deep-frozen clone, so step 2's request is unchanged. - held.content = [{ type: 'text', text: 'v2' }] - expect(adapter.requests[1]!.messages[0]).toEqual({ role: 'user', content: [{ type: 'text', text: 'v1' }] }) - expect(events(agent).filter(e => e.type === 'request/header')).toHaveLength(1) - }) -}) - - -describe('agent/turn-continuation (ContinuationDecision)', () => { - it('a continue decision with a reason records next-step steering in the same turn', async () => { - const adapter = new MockAdapter([textResponse('step 1 no tools'), textResponse('step 2')]) - const ctx = await harness(adapter) - const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - - let forced = false - ctx.on('agent/turn-continuation', async (_agent, _turn, _default, _signal, next): Promise => { - if (!forced) { - forced = true - return { action: 'continue', reason: { content: [{ type: 'text', text: 'keep going on the goal' }], source: { kind: 'plugin', plugin: 'goal' } } } - } - return next() - }) - - send(agent, 'go') - await waitForIdle(ctx, agent) - - const log = events(agent) - // The continuation stays in the turn, is logged with provenance before step 2, - // and reaches that step's request. - expect(log.filter(e => e.type === 'turn/start')).toHaveLength(1) - expect(log.filter(e => e.type === 'step/start')).toHaveLength(2) - const steering = log.find(e => e.type === 'steering/message') - expect(steering?.type === 'steering/message' && steering.data.content).toEqual([{ type: 'text', text: 'keep going on the goal' }]) - expect(steering?.type === 'steering/message' && steering.data.source).toEqual({ kind: 'plugin', plugin: 'goal' }) - expect(JSON.stringify(adapter.requests[1]!.messages)).toContain('keep going on the goal') - }) - - it('a stop decision ends the turn even when the step had tool calls', async () => { - const adapter = new MockAdapter([toolCallResponse('c1', 'echo', { text: 'hi' })]) - const ctx = await harness(adapter) - ctx.tools.register(defineContentToolFixture({ - name: 'echo', description: 'echo', parameters: { text: { type: 'string' } }, - async execute(args) { return [{ type: 'text', text: String(args.text) }] }, - })) - const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - - ctx.on('agent/turn-continuation', async (): Promise => ({ action: 'stop' })) - - send(agent, 'go') - await waitForIdle(ctx, agent) - - // default would have continued (had tool calls), but the stop decision wins - expect(adapter.requests).toHaveLength(1) - expect(events(agent).some(e => e.type === 'tool/result')).toBe(true) - }) -}) - describe('tool additionalContexts buffering across a step', () => { it('appends each call\'s contexts only AFTER all tool/results, preserving adjacency', async () => { // One assistant step with TWO tool calls; the second model response stops. @@ -703,7 +452,7 @@ describe('worked example: a native hook plugin is just a cordis plugin on the se expect(log.some(e => e.type.startsWith('hook/'))).toBe(false) }) - it('the same plugin blocks a destructive prompt → rejected turn, model never called', async () => { + it('the same plugin blocks a destructive prompt before a turn or model call', async () => { const adapter = new MockAdapter([textResponse('should not run')]) const ctx = await harness(adapter) await ctx.plugin(NativeGuard) @@ -713,10 +462,10 @@ describe('worked example: a native hook plugin is just a cordis plugin on the se ctx.on('session/event', (_s, event: SessionEvent) => { if (event.type === 'turn/end') reasons.push(event.data.reason) }) send(agent, 'run rm -rf /') - await waitForIdle(ctx, agent) + await agent.whenIdle() expect(adapter.requests).toHaveLength(0) - expect(reasons).toEqual([{ kind: 'rejected', reason: 'destructive prompt blocked' }]) + expect(reasons).toEqual([]) }) it('HMR-safety: disposing the plugin fiber removes all four listeners', async () => { diff --git a/packages/core/agent-loop/tests/loop.spec.ts b/packages/core/agent-loop/tests/loop.spec.ts index 5636854f77..a3162a7229 100644 --- a/packages/core/agent-loop/tests/loop.spec.ts +++ b/packages/core/agent-loop/tests/loop.spec.ts @@ -3,7 +3,7 @@ import { Context } from 'cordis' import LlmService, { CallId, StreamChunk } from '@deepseek-ai/dsh-llm' import SessionStore, { SessionId, TurnEndReason } from '@deepseek-ai/dsh-session' import SystemPrompt from '@deepseek-ai/dsh-system-prompt' -import ToolRegistry, { defineTool } from '@deepseek-ai/dsh-tools' +import ToolRegistry, { defineContentToolFixture } from '@deepseek-ai/dsh-tools' import AgentRegistry, { type Agent } from '@deepseek-ai/dsh-agent' import AgentLoop from '@deepseek-ai/dsh-agent-loop' @@ -89,7 +89,7 @@ describe('agent loop', () => { textResponse('done'), ]) const ctx = await harness(adapter) - ctx.tools.register(defineTool({ + ctx.tools.register(defineContentToolFixture({ name: 'echo', description: 'echo back', parameters: { text: { type: 'string' } }, @@ -120,39 +120,13 @@ describe('agent loop', () => { expect(types).toContain('tool/result') }) - it('threads a tool-attached meta (execute object return) onto the tool/result event', async () => { - const adapter = new MockAdapter([ - toolCallResponse('c1', 'writer', { path: 'a.txt' }, 'writing'), - textResponse('done'), - ]) - const ctx = await harness(adapter) - // A tool that returns the { content, meta } object form: the loop must - // persist `meta` on the tool/result event so a UI reproduces the card on replay. - ctx.tools.register(defineTool({ - name: 'writer', - description: 'writes a file', - parameters: { path: { type: 'string' } }, - async execute() { - return { content: [{ type: 'text', text: 'ok' }], meta: { diffs: [{ path: 'a.txt', oldText: null, newText: 'x' }] } } - }, - })) - const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - - send(agent, 'use the tool') - await waitForIdle(ctx, agent) - - const toolResult = agent.session.events.find(e => e.type === 'tool/result') - expect(toolResult?.type === 'tool/result' && toolResult.data.meta) - .toEqual({ diffs: [{ path: 'a.txt', oldText: null, newText: 'x' }] }) - }) - it('renders harness identity, then the persona, then tool guidance — with {{variables}} resolved', async () => { const adapter = new MockAdapter([textResponse('ok')]) // The persona is a TEMPLATE: {{model}} is the loop-registered variable // projecting this agent's configured model, so the model knows its own name. const ctx = await harness(adapter, 'You are a test agent on {{model}}.') ctx.systemPrompt.section({ name: 'tool:noop', order: 100, text: 'Use the noop tool wisely.' }) - ctx.tools.register(defineTool({ + ctx.tools.register(defineContentToolFixture({ name: 'noop', description: 'does nothing', parameters: {}, @@ -191,7 +165,9 @@ describe('agent loop', () => { const adapter = new MockAdapter([textResponse('ok after rescue')]) const ctx = await harness(adapter, 'In {{cwd}}.') const errors: Error[] = [] - ctx.on('agent/error', (_agent, _turn, _step, error) => void errors.push(error)) + ctx.on('agent/error', (_agent, _turn, _step, error) => { + if (error instanceof Error) errors.push(error) + }) const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) send(agent, 'hi') @@ -231,7 +207,8 @@ describe('agent loop', () => { assembly.variables['model'] = 'mock' return next() }) - ctx.on('agent/request', async (_agent, _turn, _step, config, _signal, _next) => { + ctx.on('agent/request', async (_agent, _turn, _step, _signal, next) => { + const config = await next() return { ...config, provider: 'mock', model: 'mock' } }) const agent = ctx.agentLoop.create(SessionId('a-late-model'), {}) @@ -244,44 +221,6 @@ describe('agent loop', () => { expect(adapter.requests[0]!.system).toBe('You are an AI agent powered by the DeepSeek Harness SDK.\n\nYou run on mock.') }) - it.each([ - ['BigInt', { n: 1n }], - ['Map', new Map([['key', 'value']])], - ['class instance', new (class ResultMeta { x = 1 })()], - ])('normalizes non-JSON tool meta (%s) before the durable result commit', async (_kind, meta) => { - const adapter = new MockAdapter([ - toolCallResponse('bad-meta-call', 'bad-meta', {}, 'calling'), - textResponse('recovered'), - ]) - const ctx = await harness(adapter) - ctx.tools.register(defineTool({ - name: 'bad-meta', - description: 'returns invalid durable metadata', - parameters: {}, - execute: () => Promise.resolve({ content: [{ type: 'text' as const, text: 'apparent success' }], meta }), - })) - const agent = ctx.agentLoop.create(SessionId('bad-meta-agent'), { provider: 'mock', model: 'mock' }) - - send(agent, 'use the tool') - await waitForIdle(ctx, agent) - - const result = agent.session.events.find(event => event.type === 'tool/result') - expect(result?.type).toBe('tool/result') - if (result?.type === 'tool/result') { - expect(result.data.callId).toBe('bad-meta-call') - expect(result.data.isError).toBe(true) - expect(result.data.meta).toBeUndefined() - expect(result.data.content).toEqual([{ - type: 'text', - text: 'Error: tool result must be losslessly JSON-serializable', - }]) - } - // The normalized failure was durably logged and fed back to the model; the - // turn continued normally instead of failing after an apparent success. - expect(adapter.requests).toHaveLength(2) - expect(JSON.stringify(adapter.requests[1]!.messages)).toContain('losslessly JSON-serializable') - }) - it('omits the system field when a system-prompt/assemble veto empties the assembly', async () => { // The documented escape valve: a deployment that must drop the harness // openers short-circuits the assemble waterfall; the request then carries @@ -326,7 +265,7 @@ describe('agent loop', () => { const ctx = await harness(adapter) const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - ctx.tools.register(defineTool({ + ctx.tools.register(defineContentToolFixture({ name: 'slow', description: '', parameters: {}, @@ -452,7 +391,7 @@ describe('agent loop', () => { const ctx = await harness(adapter) const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) let visibleDuringTool = false - ctx.tools.register(defineTool({ + ctx.tools.register(defineContentToolFixture({ name: 'noticer', description: 'injects a notice', parameters: {}, @@ -484,12 +423,9 @@ describe('agent loop', () => { const contexts = agent.session.events.filter(e => e.type === 'user/message' && e.data.source.kind === 'plugin') expect(contexts).toHaveLength(2) expect(result.seq).toBeLessThan(contexts[0]!.seq) - expect(contexts[0]?.type === 'user/message' && contexts[0].data).toMatchObject({ - meta, - }) expect(contexts.flatMap(event => event.type === 'user/message' ? event.data.content : [])) .toEqual([ - { type: 'text', text: 'mid-turn notice' }, + { type: 'text', text: 'mutated after inject' }, { type: 'text', text: 'second notice' }, ]) @@ -502,7 +438,7 @@ describe('agent loop', () => { ? [index] : []) expect(resultIndex).toBeGreaterThanOrEqual(0) - expect(contextIndexes).toHaveLength(2) + expect(contextIndexes).toHaveLength(1) expect(contextIndexes.every(index => index > resultIndex)).toBe(true) }) @@ -513,7 +449,7 @@ describe('agent loop', () => { ]) const ctx = await harness(adapter) const agent = ctx.agentLoop.create(SessionId('invalid-context'), { provider: 'mock', model: 'mock' }) - ctx.tools.register(defineTool({ + ctx.tools.register(defineContentToolFixture({ name: 'invalid-injector', description: 'attempts an invalid context injection', parameters: {}, @@ -533,8 +469,7 @@ describe('agent loop', () => { expect(agent.session.events.some(event => event.type === 'user/message' && event.data.source.kind === 'plugin')).toBe(false) }) - it('agent/turn-continuation can force-continue (/loop pattern) and force-stop', async () => { - // force-continue: model never calls tools, but a plugin forces 3 steps + it('agent/stopping can steer another step (/loop pattern)', async () => { const adapter = new MockAdapter([ textResponse('step 1'), textResponse('step 2'), @@ -545,9 +480,12 @@ describe('agent loop', () => { let steps = 0 ctx.on('session/event', (_session, event) => { if (event.type === 'step/end') steps++ }) - ctx.on('agent/turn-continuation', async (_agent, _turn, _defaultDecision, _signal, next) => { - if (steps < 3) return { action: 'continue' as const } - return next() + ctx.on('agent/stopping', (subject) => { + if (steps < 3) { + subject.steer([{ type: 'text', text: 'continue' }], { + source: { kind: 'plugin', plugin: 'loop-test' }, + }) + } }) send(agent, 'go') @@ -556,26 +494,25 @@ describe('agent loop', () => { expect(adapter.requests).toHaveLength(3) }) - it('agent/turn-continuation can veto continuation despite tool calls (budget-guard pattern)', async () => { + it('a tool can conclude the turn despite owing a follow-up request', async () => { const adapter = new MockAdapter([toolCallResponse('c1', 'echo', { text: 'x' })]) const ctx = await harness(adapter) - ctx.tools.register(defineTool({ + ctx.tools.register(defineContentToolFixture({ name: 'echo', description: '', parameters: { text: { type: 'string' } }, - async execute(args) { + async execute(args, exec) { + exec.concludeTurn() return [{ type: 'text', text: String(args.text) }] }, })) const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - ctx.on('agent/turn-continuation', async () => ({ action: 'stop' }) as const) - send(agent, 'go') await waitForIdle(ctx, agent) // only one model call despite the tool call requesting a follow-up expect(adapter.requests).toHaveLength(1) - // tool still executed before the decision + // The tool still executes and durably records its result. expect(agent.session.events.some(e => e.type === 'tool/result')).toBe(true) }) @@ -584,7 +521,8 @@ describe('agent loop', () => { const ctx = await harness(adapter) const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - ctx.on('agent/request', async (_agent, _turn, _step, config, _signal, _next) => { + ctx.on('agent/request', async (_agent, _turn, _step, _signal, next) => { + const config = await next() // The seed is frozen — config is not a mutable per-call knob; a switch // is proposed by returning a replacement, and the loop logs it. expect(Object.isFrozen(config)).toBe(true) @@ -601,20 +539,20 @@ describe('agent loop', () => { expect(headerEvent?.type === 'request/header' && headerEvent.data.header.config.model).toBe('other-model') }) - it('agent/pre-step fires once per step before the step is opened', async () => { + it('agent/step fires once per step before the step is opened', async () => { const adapter = new MockAdapter([ toolCallResponse('c1', 'echo', {}, 'calling echo'), textResponse('done'), ]) const ctx = await harness(adapter) - ctx.tools.register(defineTool({ + ctx.tools.register(defineContentToolFixture({ name: 'echo', description: 'echo', parameters: {}, async execute() { return [{ type: 'text', text: 'echoed' }] }, })) const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) const fires: { turn: number; step: number; signal: AbortSignal }[] = [] - ctx.on('agent/pre-step', (subject, turn, step, signal) => { + ctx.on('agent/step', (subject, turn, step, signal) => { if (subject === agent) fires.push({ turn, step, signal }) }) @@ -628,7 +566,7 @@ describe('agent loop', () => { expect(fires.every(({ signal }) => signal instanceof AbortSignal)).toBe(true) }) - it('agent/pre-step fires BEFORE the step it precedes opens (events land outside the step)', async () => { + it('agent/step fires BEFORE the step it precedes opens (events land outside the step)', async () => { // The append lands before step/start, yet derive happens afterwards and the // same step's request must include it. const adapter = new MockAdapter([textResponse('ok')]) @@ -636,7 +574,7 @@ describe('agent loop', () => { const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) let injected = false - ctx.on('agent/pre-step', (subject) => { + ctx.on('agent/step', (subject) => { if (subject === agent && !injected) { injected = true subject.session.append('user/message', { @@ -662,7 +600,7 @@ describe('agent loop', () => { expect(injectedSeq).toBeLessThan(firstStepStartSeq) }) - it('a throwing agent/pre-step listener ends the turn (error), not the loop', async () => { + it('a throwing agent/step listener ends the turn (error), not the loop', async () => { // Before step/start, a pre-step throw reaches the turn catch: no step needs // closing, the turn records error, and the loop remains available. const adapter = new MockAdapter([textResponse('second turn ok')]) @@ -670,12 +608,14 @@ describe('agent loop', () => { const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) let throwOnce = true - ctx.on('agent/pre-step', () => { + ctx.on('agent/step', () => { if (throwOnce) { throwOnce = false; throw new Error('boom in pre-step') } }) const errors: Error[] = [] - ctx.on('agent/error', (_a, _t, _s, error) => void errors.push(error)) + ctx.on('agent/error', (_a, _t, _s, error) => { + if (error instanceof Error) errors.push(error) + }) send(agent, 'first') await waitForIdle(ctx, agent) @@ -750,9 +690,12 @@ describe('agent loop', () => { ctx.on('session/event', (_session, event) => { if (event.type === 'step/end') steps++ }) // Force exactly one continuation (step 1 → step 2), then defer to default // (step 2 is a plain stop with no tool calls → stops). - ctx.on('agent/turn-continuation', async (_agent, _turn, _defaultDecision, _signal, next) => { - if (steps < 2) return { action: 'continue' as const } - return next() + ctx.on('agent/stopping', (subject) => { + if (steps < 2) { + subject.steer([{ type: 'text', text: 'continue after truncation' }], { + source: { kind: 'plugin', plugin: 'max-tokens-test' }, + }) + } }) const reasons: TurnEndReason[] = [] @@ -766,6 +709,7 @@ describe('agent loop', () => { expect(adapter.requests[1]!.messages).toEqual([ { role: 'user', content: [{ type: 'text', text: 'go' }] }, { role: 'assistant', content: [{ type: 'text', text: 'first half' }], provenance: { provider: 'mock', model: 'mock' } }, + { role: 'user', content: [{ type: 'text', text: 'continue after truncation' }] }, ]) expect(reasons).toEqual([{ kind: 'max-tokens' }]) }) @@ -799,7 +743,7 @@ describe('agent loop', () => { ]]) const ctx = await harness(adapter) let executions = 0 - ctx.tools.register(defineTool({ + ctx.tools.register(defineContentToolFixture({ name: 'echo', description: '', parameters: { text: { type: 'string' } }, @@ -839,7 +783,7 @@ describe('agent loop', () => { { type: 'finish', reason: { kind: 'max-tokens' } }, ]]) const ctx = await harness(adapter) - ctx.tools.register(defineTool({ + ctx.tools.register(defineContentToolFixture({ name: 'echo', description: '', parameters: { text: { type: 'string' } }, @@ -901,18 +845,11 @@ describe('agent loop', () => { { type: 'finish', reason: { kind: 'max-tokens' } }, ]]) const ctx = await harness(adapter) - let stepResults = 0 - ctx.on('agent/step-result', async (_agent, _turn, _step, message, _signal, next) => { - stepResults += 1 - expect(message.content).toEqual([{ type: 'text', text: 'partial text' }]) - return next() - }) const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) send(agent, 'go') await waitForIdle(ctx, agent) - expect(stepResults).toBe(1) expect(agent.session.events.some(e => e.type === 'tool/call')).toBe(false) expect(agent.session.deriveMessages()).toEqual([ { role: 'user', content: [{ type: 'text', text: 'go' }] }, @@ -926,7 +863,7 @@ describe('agent loop', () => { textResponse('continued after tool call'), ]) const ctx = await harness(adapter) - ctx.tools.register(defineTool({ + ctx.tools.register(defineContentToolFixture({ name: 'echo', description: '', parameters: { text: { type: 'string' } }, @@ -950,91 +887,6 @@ describe('agent loop', () => { expect(turnEnd?.type === 'turn/end' && turnEnd.data.reason.kind).toBe('completed') }) - it('keeps same-tick sends in separate turns and checkpoints before the next starts', async () => { - const adapter = new MockAdapter([textResponse('first answer'), textResponse('second answer')]) - const ctx = await harness(adapter) - const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - - const firstFlush = Promise.withResolvers() - const releaseFirstFlush = Promise.withResolvers() - let flushes = 0 - ctx.on('session/flush', async (session) => { - if (session !== agent.session) return - flushes += 1 - if (flushes === 1) { - firstFlush.resolve(undefined) - await releaseFirstFlush.promise - } - }) - - const turns: number[] = [] - ctx.on('session/event', (session, event) => { - if (session === agent.session && event.type === 'turn/start') turns.push(event.data.turn) - }) - - const idle = waitForIdle(ctx, agent) - send(agent, 'first message') - send(agent, 'second message') - - await firstFlush.promise - expect(turns).toEqual([1]) - expect(adapter.requests).toHaveLength(1) - - releaseFirstFlush.resolve(undefined) - await idle - - expect(turns).toEqual([1, 2]) - expect(flushes).toBe(2) - expect(adapter.requests).toHaveLength(2) - expect(JSON.stringify(adapter.requests[1]!.messages)).toContain('first answer') - expect(JSON.stringify(adapter.requests[1]!.messages)).toContain('second message') - }) - - it('holds a turn-end listener send behind the closing turn checkpoint', async () => { - const adapter = new MockAdapter([textResponse('first answer'), textResponse('second answer')]) - const ctx = await harness(adapter) - const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - - const firstFlush = Promise.withResolvers() - const releaseFirstFlush = Promise.withResolvers() - let flushes = 0 - ctx.on('session/flush', async (session) => { - if (session !== agent.session) return - flushes += 1 - if (flushes === 1) { - firstFlush.resolve(undefined) - await releaseFirstFlush.promise - } - }) - - const turns: number[] = [] - const statuses: string[] = [] - ctx.on('agent/status', (subject, status) => { - if (subject === agent) statuses.push(status) - }) - ctx.on('session/event', (session, event) => { - if (session !== agent.session) return - if (event.type === 'turn/start') turns.push(event.data.turn) - if (event.type === 'turn/end' && event.data.turn === 1) send(agent, 'turn-end listener message') - }) - - const idle = waitForIdle(ctx, agent) - send(agent, 'first message') - await firstFlush.promise - - expect(turns).toEqual([1]) - expect(adapter.requests).toHaveLength(1) - - releaseFirstFlush.resolve(undefined) - await idle - - expect(turns).toEqual([1, 2]) - expect(statuses).toEqual(['running', 'idle']) - expect(adapter.requests).toHaveLength(2) - expect(JSON.stringify(adapter.requests[1]!.messages)).toContain('first answer') - expect(JSON.stringify(adapter.requests[1]!.messages)).toContain('turn-end listener message') - }) - it('keeps a reentrant agent/inbox/enqueue send as the next independent turn', async () => { const adapter = new MockAdapter([textResponse('first'), textResponse('second')]) const ctx = await harness(adapter) @@ -1148,26 +1000,6 @@ describe('agent loop', () => { ]) }) - it('awaits session/flush at turn end (persistence checkpoint)', async () => { - const adapter = new MockAdapter([textResponse('ok')]) - const ctx = await harness(adapter) - const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - - let flushed = 0 - let flushedBeforeIdle = false - ctx.on('session/flush', async (session) => { - await new Promise(r => setTimeout(r, 10)) - flushed++ - flushedBeforeIdle = agent.status !== 'idle' - void session - }) - - send(agent, 'hi') - await waitForIdle(ctx, agent) - expect(flushed).toBe(1) - expect(flushedBeforeIdle).toBe(true) - }) - it('errors from the model surface as agent/error and end the turn', async () => { const adapter = new MockAdapter([]) // script exhausted → throws const ctx = await harness(adapter) @@ -1175,7 +1007,9 @@ describe('agent loop', () => { const errors: Error[] = [] const reasons: TurnEndReason[] = [] - ctx.on('agent/error', (_agent, _turn, _step, error) => void errors.push(error)) + ctx.on('agent/error', (_agent, _turn, _step, error) => { + if (error instanceof Error) errors.push(error) + }) ctx.on('session/event', (_s, event) => { if (event.type === 'turn/end') reasons.push(event.data.reason) }) send(agent, 'hi') @@ -1207,8 +1041,7 @@ describe('agent loop', () => { await fiber.dispose() await driverDone(agent) - expect(agent.status).toBe('disposed') - expect(ctx.agents.get(SessionId('scoped'))).toBeUndefined() + await expect.poll(() => ctx.agents.get(SessionId('scoped')) === undefined).toBe(true) }) it('creates agents from config on startup', async () => { @@ -1257,7 +1090,7 @@ describe('agent loop', () => { textResponse('done'), ]) const ctx = await harness(adapter) - ctx.tools.register(defineTool({ + ctx.tools.register(defineContentToolFixture({ name: 'echo', description: '', parameters: { text: { type: 'string' } }, diff --git a/packages/core/agent-loop/tests/properties.spec.ts b/packages/core/agent-loop/tests/properties.spec.ts index 593b32ab38..05674e4b7d 100644 --- a/packages/core/agent-loop/tests/properties.spec.ts +++ b/packages/core/agent-loop/tests/properties.spec.ts @@ -5,8 +5,8 @@ * is a finding, not timing noise. * * Invariants: every sent message appears exactly once in the log (none lost); - * turn numbers strictly increase; status transitions follow the legal machine - * idle→running→idle (and →disposed at teardown). + * turn numbers strictly increase; status transitions follow + * idle→running→idle, while teardown is a registry lifecycle. */ import { describe, expect, it } from 'vitest' diff --git a/packages/core/agent-loop/tests/request-error.spec.ts b/packages/core/agent-loop/tests/request-error.spec.ts new file mode 100644 index 0000000000..cd123cd311 --- /dev/null +++ b/packages/core/agent-loop/tests/request-error.spec.ts @@ -0,0 +1,149 @@ +import { describe, expect, it } from 'vitest' +import { Context } from 'cordis' +import AgentRegistry from '@deepseek-ai/dsh-agent' +import AgentLoop from '@deepseek-ai/dsh-agent-loop' +import LlmService, { LlmError } from '@deepseek-ai/dsh-llm' +import type { LlmFailure } from '@deepseek-ai/dsh-llm' +import SessionStore, { SessionId } from '@deepseek-ai/dsh-session' +import SystemPrompt from '@deepseek-ai/dsh-system-prompt' +import ToolRegistry from '@deepseek-ai/dsh-tools' +import { MockAdapter, textResponse } from './mock-adapter.ts' + +async function harness(adapter: MockAdapter): Promise { + const ctx = new Context() + await ctx.plugin(LlmService) + await ctx.plugin(SessionStore) + await ctx.plugin(SystemPrompt) + await ctx.plugin(ToolRegistry) + await ctx.plugin(AgentRegistry) + await ctx.plugin(AgentLoop, { agents: [] }) + ctx.llm.registerAdapter(['mock'], adapter) + return ctx +} + +function fail(message: string, code: string): () => never { + return () => { + throw new LlmError(message, code) + } +} + +describe('agent/request-error', () => { + it('does not offer middleware failures to request recovery', async () => { + const adapter = new MockAdapter([textResponse('unused')]) + const ctx = await harness(adapter) + const agent = ctx.agentLoop.create(SessionId('request-error-narrow'), { provider: 'mock', model: 'mock' }) + let recoveries = 0 + ctx.on('agent/request', () => { + throw new LlmError('middleware failed', 'MIDDLEWARE') + }) + ctx.on('agent/request-error', async () => { + recoveries += 1 + }) + + agent.followup([{ type: 'text', text: 'go' }]) + await agent.whenIdle() + + expect(recoveries).toBe(0) + expect(adapter.requests).toHaveLength(0) + }) + + it('lets each failed request schedule a retry before its turn closes', async () => { + const adapter = new MockAdapter([ + fail('busy', 'RATE_LIMIT'), + fail('unavailable', 'SERVICE_UNAVAILABLE'), + textResponse('ok'), + ]) + const ctx = await harness(adapter) + const agent = ctx.agentLoop.create(SessionId('request-error-retry'), { provider: 'mock', model: 'mock' }) + const seen: { turn: number; step: number; failure: LlmFailure }[] = [] + const statuses: string[] = [] + const idleTurns: number[] = [] + ctx.on('agent/status', (subject, status) => { + if (subject === agent) statuses.push(status) + }) + ctx.on('agent/idle', (subject, turn) => { + if (subject === agent) idleTurns.push(turn) + }) + ctx.on('agent/request-error', async (subject, turn, step, _error, failure) => { + expect(subject).toBe(agent) + expect(agent.session.events.at(-1)).toMatchObject({ + type: 'step/end', + data: { turn, step }, + }) + seen.push({ turn, step, failure }) + subject.retry() + subject.retry() + }) + + agent.followup([{ type: 'text', text: 'go' }]) + await agent.whenIdle() + + expect(seen.map(item => ({ + turn: item.turn, + step: item.step, + code: item.failure.code, + }))).toEqual([ + { + turn: 1, + step: 1, + code: 'RATE_LIMIT', + }, + { + turn: 2, + step: 1, + code: 'SERVICE_UNAVAILABLE', + }, + ]) + expect(agent.session.events.filter(event => event.type === 'turn/start').map(event => event.data.trigger)) + .toEqual([ + { kind: 'message', source: { kind: 'user' } }, + { kind: 'retry' }, + { kind: 'retry' }, + ]) + expect(statuses).toEqual(['running', 'idle']) + expect(idleTurns).toEqual([3]) + }) + + it('lets cancellation win over a retry request', async () => { + const adapter = new MockAdapter([fail('busy', 'RATE_LIMIT'), textResponse('unused')]) + const ctx = await harness(adapter) + const agent = ctx.agentLoop.create(SessionId('request-error-cancel'), { provider: 'mock', model: 'mock' }) + ctx.on('agent/request-error', async (subject) => { + subject.retry() + subject.cancel({ kind: 'user' }) + }) + + agent.followup([{ type: 'text', text: 'go' }]) + await agent.whenIdle() + + expect(adapter.requests).toHaveLength(1) + expect(agent.session.events.filter(event => event.type === 'turn/start')).toHaveLength(1) + expect(agent.session.events.find(event => event.type === 'turn/end')).toMatchObject({ + type: 'turn/end', + data: { reason: { kind: 'aborted' } }, + }) + }) + + it('does not honor a retry requested by a failing recovery listener', async () => { + const adapter = new MockAdapter([fail('busy', 'RATE_LIMIT'), textResponse('unused')]) + const ctx = await harness(adapter) + const agent = ctx.agentLoop.create(SessionId('request-error-recovery-failed'), { + provider: 'mock', + model: 'mock', + }) + ctx.on('agent/request-error', async (subject) => { + subject.retry() + throw new Error('recovery failed') + }) + + agent.followup([{ type: 'text', text: 'go' }]) + await agent.whenIdle() + + expect(adapter.requests).toHaveLength(1) + expect(agent.session.events.filter(event => event.type === 'turn/start')).toHaveLength(1) + expect(agent.session.events.find(event => event.type === 'turn/end')).toMatchObject({ + type: 'turn/end', + data: { reason: { kind: 'error' } }, + }) + }) +}) diff --git a/packages/core/agent-loop/tests/request-log.spec.ts b/packages/core/agent-loop/tests/request-log.spec.ts deleted file mode 100644 index 4a3cfe37a4..0000000000 --- a/packages/core/agent-loop/tests/request-log.spec.ts +++ /dev/null @@ -1,82 +0,0 @@ -/** - * recordRequestHeader unit tests: exactly one of three things per request — - * an 'initial' snapshot (log has no header yet), a 'resume' snapshot (fresh - * loop instance over a log that has one), nothing (header unchanged), or a - * full 'change' snapshot. - */ - -import { describe, expect, it } from 'vitest' -import { Session, SessionId, canonicalHeader } from '@deepseek-ai/dsh-session' -import type { SessionEvent } from '@deepseek-ai/dsh-session' -import type { ToolSchema } from '@deepseek-ai/dsh-llm' -import { createTransmissionLog, recordRequestHeader } from '../src/request-log.ts' - -function tool(name: string, description = 'd'): ToolSchema { - return { name, description, parameters: { type: 'object' } } -} - -function openSession(id: string): Session { - const session = new Session(SessionId(id)) - session.append('turn/start', { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } }) - return session -} - -function headerEvents(session: Session): SessionEvent[] { - return session.events.filter(e => e.type === 'request/header') -} - -describe('recordRequestHeader', () => { - it("anchors a new conversation with an 'initial' snapshot, then logs nothing while unchanged", () => { - const session = openSession('rl-initial') - const state = createTransmissionLog() - const header = canonicalHeader({ config: { provider: 'mock', model: 'm' }, system: 's', tools: [tool('t')] }) - - recordRequestHeader(session, state, header) - const [first] = headerEvents(session) - expect(first?.type === 'request/header' && first.data.reason).toBe('initial') - - recordRequestHeader(session, state, header) - expect(headerEvents(session)).toHaveLength(1) - }) - - it("anchors a fresh loop instance over an anchored log with a 'resume' snapshot, even unchanged", () => { - const session = openSession('rl-resume') - const header = canonicalHeader({ config: { provider: 'mock', model: 'm' }, system: 's' }) - recordRequestHeader(session, createTransmissionLog(), header) - - // A second instance (process restart / fork): the boundary itself is a - // recorded fact — snapshot appended even though the header is identical. - recordRequestHeader(session, createTransmissionLog(), header) - const events = headerEvents(session) - expect(events).toHaveLength(2) - expect(events[1]?.type === 'request/header' && events[1].data.reason).toBe('resume') - }) - - it("logs a full 'change' snapshot for a mid-run change, and the fold reproduces the header", () => { - const session = openSession('rl-change') - const state = createTransmissionLog() - const first = canonicalHeader({ config: { provider: 'mock', model: 'm' }, system: 'a\nb', tools: [tool('t')] }) - recordRequestHeader(session, state, first) - - const second = canonicalHeader({ config: { provider: 'mock', model: 'm' }, system: 'a\nc', tools: [tool('t'), tool('u')] }) - recordRequestHeader(session, state, second) - const events = headerEvents(session) - expect(events).toHaveLength(2) - expect(events[1]?.type === 'request/header' && events[1].data.reason).toBe('change') - expect(session.requestHeader()).toEqual(second) - }) - - it("records a pure tool reordering as a 'change' snapshot", () => { - const session = openSession('rl-reorder') - const state = createTransmissionLog() - const first = canonicalHeader({ config: { provider: 'mock', model: 'm' }, tools: [tool('a'), tool('b')] }) - recordRequestHeader(session, state, first) - - const reordered = canonicalHeader({ config: { provider: 'mock', model: 'm' }, tools: [tool('b'), tool('a')] }) - recordRequestHeader(session, state, reordered) - const events = headerEvents(session) - expect(events).toHaveLength(2) - expect(events[1]?.type === 'request/header' && events[1].data.reason).toBe('change') - expect(session.requestHeader()).toEqual(reordered) - }) -}) diff --git a/packages/core/agent-loop/tests/request-reconstruction.spec.ts b/packages/core/agent-loop/tests/request-reconstruction.spec.ts index a73218f345..c40443a872 100644 --- a/packages/core/agent-loop/tests/request-reconstruction.spec.ts +++ b/packages/core/agent-loop/tests/request-reconstruction.spec.ts @@ -114,7 +114,7 @@ describe('request stability across the loop', () => { // A pre-step listener compacts turn 1's history before turn 2's step — // the sanctioned surface rewrite, landing OUTSIDE the step. - const preStep = ctx.on('agent/pre-step', () => { + const preStep = ctx.on('agent/step', () => { preStep() const session = agent.session const nodes = session.surface.nodes @@ -167,7 +167,7 @@ describe('request stability across the loop', () => { const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) let injected = false - ctx.on('agent/request', async (_agent, _turn, _step, _config, _signal, next) => { + ctx.on('agent/request', async (_agent, _turn, _step, _signal, next) => { if (!injected) { injected = true agent.inject([{ type: 'text', text: '[late context]' }], { source: { kind: 'plugin', plugin: 'test' } }) @@ -195,7 +195,9 @@ describe('request stability across the loop', () => { const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) const errors: Error[] = [] - ctx.on('agent/error', (_agent, _turn, _step, error) => void errors.push(error)) + ctx.on('agent/error', (_agent, _turn, _step, error) => { + if (error instanceof Error) errors.push(error) + }) ctx.on('llm/stream', (options, next) => { // The historical failure mode this design kills: a listener rewriting // request content in place. The freeze turns it into a loud error. @@ -231,8 +233,7 @@ describe('request stability across the loop', () => { await waitForIdle(ctx2, agent2) const snapshots = agent2.session.events.filter(e => e.type === 'request/header') - expect(snapshots).toHaveLength(2) - expect(snapshots[1]?.type === 'request/header' && snapshots[1].data.reason).toBe('resume') + expect(snapshots).toHaveLength(1) // Identical header across the restart: byte-identical continuation. expect(adapter2.requests[0]!.system).toEqual(adapter.requests[0]!.system) expectPrefixExtension(adapter.requests[0]!, adapter2.requests[0]!) @@ -243,7 +244,7 @@ describe('request stability across the loop', () => { const ctx = await harness(adapter) const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) - ctx.on('agent/request', async (_agent, _turn, _step, _config, _signal, next) => { + ctx.on('agent/request', async (_agent, _turn, _step, _signal, next) => { const config = await next() // next() resolves the SAME frozen seed — in-place shaping after // delegation is unrepresentable, so a "mutate what next() returned" @@ -280,7 +281,9 @@ describe('request stability across the loop', () => { send(agent, 'go') await waitForIdle(ctx, agent) ctx.systemPrompt.section({ name: 'extra', order: 2, text: 'now with guidance' }) - ctx.on('agent/request', async (_agent, _turn, _step, config, _signal, _next) => ({ ...config, temperature: 0.5, maxTokens: 99, stop: [''] })) + ctx.on('agent/request', async (_agent, _turn, _step, _signal, next) => ({ + ...await next(), temperature: 0.5, maxTokens: 99, stop: [''], + })) send(agent, 'again') await waitForIdle(ctx, agent) diff --git a/packages/core/agent-loop/tests/request-recovery.spec.ts b/packages/core/agent-loop/tests/request-recovery.spec.ts deleted file mode 100644 index 1fc44e3431..0000000000 --- a/packages/core/agent-loop/tests/request-recovery.spec.ts +++ /dev/null @@ -1,605 +0,0 @@ -import { describe, expect, it, vi } from 'vitest' -import { Context } from 'cordis' -import LlmService, { - CallId, - CONTEXT_WINDOW_EXCEEDED_CODE, - HarnessError, - LlmAdapter, - LlmError, - ProviderRequestId, -} from '@deepseek-ai/dsh-llm' -import type { GenerateOptions, LlmFailure, StreamChunk } from '@deepseek-ai/dsh-llm' -import SessionStore, { SessionId } from '@deepseek-ai/dsh-session' -import SystemPrompt from '@deepseek-ai/dsh-system-prompt' -import ToolRegistry, { defineContentToolFixture } from '@deepseek-ai/dsh-tools' -import type { PostToolDecision } from '@deepseek-ai/dsh-tools' -import AgentRegistry from '@deepseek-ai/dsh-agent' -import type { Agent } from '@deepseek-ai/dsh-agent' -import AgentLoop from '@deepseek-ai/dsh-agent-loop' -import { maxTokensResponse, textResponse, toolCallResponse } from './mock-adapter.ts' - -class FailureScriptAdapter extends LlmAdapter { - requests: GenerateOptions[] = [] - - constructor(private readonly entries: (Error | StreamChunk[])[]) { - super() - } - - async * stream(options: GenerateOptions): AsyncIterable { - this.requests.push(options) - const entry = this.entries.shift() - if (entry === undefined) throw new Error('failure script exhausted') - if (entry instanceof Error) throw entry - yield* entry - } -} - -class IteratorConstructionFailureAdapter extends LlmAdapter { - stream(_options: GenerateOptions): AsyncIterable { - return { - [Symbol.asyncIterator](): AsyncIterator { - throw new LlmError('iterator construction failed', 'ITERATOR_CONSTRUCTION') - }, - } - } -} - -class SynchronousDispatchFailureAdapter extends LlmAdapter { - constructor(private readonly error: Error) { - super() - } - - stream(_options: GenerateOptions): AsyncIterable { - throw this.error - } -} - -class IteratorResultGetterFailureAdapter extends LlmAdapter { - constructor( - private readonly field: 'done' | 'value', - private readonly error: Error, - ) { - super() - } - - stream(_options: GenerateOptions): AsyncIterable { - const result = this.field === 'done' ? {} : { done: false } - Object.defineProperty(result, this.field, { get: () => { throw this.error } }) - return { - [Symbol.asyncIterator](): AsyncIterator { - return { next: () => Promise.resolve(result as unknown as IteratorResult) } - }, - } - } -} - -const streamListenerFailureCases: readonly [string, (ctx: Context) => void][] = [ - ['synchronous listener throw', (ctx) => { - ctx.on('llm/stream', () => { throw new Error('synchronous stream listener failed') }) - }], - ['invalid listener iterable', (ctx) => { - ctx.on('llm/stream', () => ({}) as AsyncIterable) - }], - ['listener wrapper iteration failure', (ctx) => { - ctx.on('llm/stream', (_options, next) => (async function * () { - for await (const chunk of next()) { - yield chunk - throw new Error('stream listener wrapper failed') - } - })()) - }], -] - -async function harness(adapter?: LlmAdapter): Promise { - const ctx = new Context() - await ctx.plugin(LlmService) - await ctx.plugin(SessionStore) - await ctx.plugin(SystemPrompt) - await ctx.plugin(ToolRegistry) - await ctx.plugin(AgentRegistry) - await ctx.plugin(AgentLoop, { agents: [] }) - if (adapter) ctx.llm.registerAdapter(['mock'], adapter) - return ctx -} - -function waitForIdle(ctx: Context, agent: Agent): Promise { - return new Promise((resolve) => { - const dispose = ctx.on('agent/status', (subject, status) => { - if (subject === agent && status === 'idle') { - dispose() - resolve() - } - }) - }) -} - -function send(agent: Agent): void { - agent.followup([{ type: 'text', text: 'go' }]) -} - -function contextError(message = 'context too large'): LlmError { - return new LlmError(message, CONTEXT_WINDOW_EXCEEDED_CODE) -} - -describe('agent post-step and request-error lifecycle', () => { - it('fires post-step after results, buffered context, and steering but before step/end', async () => { - const twoCalls: StreamChunk[] = [ - { type: 'block-start', index: 0, blockType: 'tool-call' }, - { type: 'block-end', index: 0, block: { type: 'tool-call', id: CallId('call-1'), name: 'work', arguments: '{}' } }, - { type: 'block-start', index: 1, blockType: 'tool-call' }, - { type: 'block-end', index: 1, block: { type: 'tool-call', id: CallId('call-2'), name: 'work', arguments: '{}' } }, - { type: 'usage', usage: { inputTokens: 10, outputTokens: 5 } }, - { type: 'finish', reason: { kind: 'tool-calls' } }, - ] - const adapter = new FailureScriptAdapter([twoCalls, textResponse('done')]) - const ctx = await harness(adapter) - ctx.tools.register(defineContentToolFixture({ - name: 'work', - description: 'do work', - parameters: {}, - async execute(_args, exec) { - if (exec.callId === CallId('call-2')) { - exec.agent?.steer([{ type: 'text', text: 'steered' }], { source: { kind: 'plugin', plugin: 'test' } }) - } - return [{ type: 'text', text: 'worked' }] - }, - })) - ctx.on('tools/post-execute', async (exec, _result): Promise => ({ - kind: 'accept', - additionalContexts: [{ - content: [{ type: 'text', text: `context for ${exec.callId}` }], - source: { kind: 'plugin', plugin: 'test' }, - }], - })) - const agent = ctx.agentLoop.create(SessionId('post-step-order'), { provider: 'mock', model: 'mock' }) - const order: string[] = [] - ctx.on('session/event', (_session, event) => { - // Injected context is a plugin-sourced user/message; the direct human - // prompt (user source) stays untracked as before. - const isInjected = event.type === 'user/message' && event.data.source.kind !== 'user' - if ( - event.type === 'assistant/message' || event.type === 'tool/call' - || event.type === 'tool/result' || isInjected - || event.type === 'steering/message' || event.type === 'step/end' - ) { - if (!('step' in event.data) || event.data.step === 1) order.push(isInjected ? 'context/message' : event.type) - } - }) - ctx.on('agent/post-step', (subject, turn, step, signal) => { - if (subject !== agent || step !== 1) return - expect({ turn, step, aborted: signal.aborted }).toEqual({ turn: 1, step: 1, aborted: false }) - subject.inject([{ type: 'text', text: 'listener mutation' }], { source: { kind: 'plugin', plugin: 'post-step' } }) - order.push('agent/post-step') - }) - - send(agent) - await waitForIdle(ctx, agent) - - expect(order).toEqual([ - 'assistant/message', - 'tool/call', - 'tool/result', - 'tool/call', - 'tool/result', - 'context/message', - 'context/message', - 'steering/message', - 'context/message', - 'agent/post-step', - 'step/end', - ]) - }) - - it('fires post-step for max-tokens and lets cancellation override that success', async () => { - const adapter = new FailureScriptAdapter([maxTokensResponse('partial')]) - const ctx = await harness(adapter) - const agent = ctx.agentLoop.create(SessionId('cancel-post-step-max-tokens'), { provider: 'mock', model: 'mock' }) - let entered!: () => void - const postStepEntered = new Promise((resolve) => { entered = resolve }) - ctx.on('agent/post-step', async (_agent, turn, step, signal) => { - expect({ turn, step }).toEqual({ turn: 1, step: 1 }) - entered() - await new Promise((resolve) => { - signal.addEventListener('abort', () => { resolve() }, { once: true }) - }) - }) - - send(agent) - const idle = waitForIdle(ctx, agent) - await postStepEntered - agent.cancel({ kind: 'user' }) - await idle - - expect(agent.session.events.find(event => event.type === 'assistant/message')).toMatchObject({ - data: { usage: { inputTokens: 10, outputTokens: 7 } }, - }) - expect(agent.session.events.at(-1)).toMatchObject({ - type: 'turn/end', - data: { reason: { kind: 'aborted' } }, - }) - }) - - it('closes the successful step as disposed when disposal lands during post-step', async () => { - const adapter = new FailureScriptAdapter([ - toolCallResponse('dispose-call', 'work', {}), - textResponse('must not continue'), - ]) - const ctx = await harness(adapter) - ctx.tools.register(defineContentToolFixture({ - name: 'work', - description: 'do work', - parameters: {}, - async execute() { return [{ type: 'text', text: 'worked' }] }, - })) - const agent = ctx.agentLoop.create(SessionId('dispose-post-step'), { provider: 'mock', model: 'mock' }) - let entered!: () => void - const postStepEntered = new Promise((resolve) => { entered = resolve }) - ctx.on('agent/post-step', async (_agent, turn, step, signal) => { - expect({ turn, step }).toEqual({ turn: 1, step: 1 }) - entered() - await new Promise((resolve) => { - signal.addEventListener('abort', () => { resolve() }, { once: true }) - }) - }) - - send(agent) - await postStepEntered - await ctx.fiber.dispose() - - expect(adapter.requests).toHaveLength(1) - const boundaries = agent.session.events.filter(event => - event.type === 'step/start' || event.type === 'step/end', - ) - expect(boundaries.map(event => event.type)).toEqual(['step/start', 'step/end']) - expect(boundaries.map(event => event.data)).toEqual([ - { turn: 1, step: 1 }, - { turn: 1, step: 1 }, - ]) - expect(agent.session.events.at(-1)).toMatchObject({ - type: 'turn/end', - data: { reason: { kind: 'disposed' } }, - }) - }) - - it.each([ - ['thrown', contextError()], - ['in-band', [{ type: 'finish', reason: { kind: 'error', failure: { message: 'too large', code: CONTEXT_WINDOW_EXCEEDED_CODE, status: 400 } } }] satisfies StreamChunk[]], - ] as const)('recovers a %s request failure in a new reconstructable step', async (_style, failure) => { - const adapter = new FailureScriptAdapter([failure, textResponse('recovered')]) - const ctx = await harness(adapter) - const agent = ctx.agentLoop.create(SessionId(`recover-${_style}`), { provider: 'mock', model: 'mock' }) - const attempts: number[] = [] - ctx.on('agent/request-error', async (subject, turn, step, error, facts, history) => { - expect(subject).toBe(agent) - expect({ turn, step, code: error.code }).toEqual({ turn: 1, step: 1, code: CONTEXT_WINDOW_EXCEEDED_CODE }) - expect(facts.code).toBe(CONTEXT_WINDOW_EXCEEDED_CODE) - attempts.push(history.length) - subject.session.append('user/message', { - content: [{ type: 'text', text: 'RECOVERY SURFACE MUTATION' }], - source: { kind: 'plugin', plugin: 'test-recovery' }, - }, { surfaceOp: 'append' }) - return { action: 'retry' } - }) - - send(agent) - await waitForIdle(ctx, agent) - - expect(attempts).toEqual([0]) - expect(adapter.requests).toHaveLength(2) - expect(JSON.stringify(adapter.requests[1]!.messages)).toContain('RECOVERY SURFACE MUTATION') - const starts = agent.session.events.filter(event => event.type === 'step/start') - const ends = agent.session.events.filter(event => event.type === 'step/end') - expect(starts.map(event => event.data.step)).toEqual([1, 2]) - expect(ends.map(event => event.data.step)).toEqual([1, 2]) - const recovery = agent.session.events.find(event => event.type === 'user/message' && event.data.source.kind === 'plugin')! - expect(ends[0]!.seq).toBeLessThan(recovery.seq) - expect(recovery.seq).toBeLessThan(starts[1]!.seq) - }) - - it.each(streamListenerFailureCases)('does not offer %s to request recovery', async (_name, install) => { - const ctx = await harness(new FailureScriptAdapter([textResponse('unused')])) - const agent = ctx.agentLoop.create(SessionId(`stream-plugin-${_name.replaceAll(' ', '-')}`), { provider: 'mock', model: 'mock' }) - let recoveries = 0 - install(ctx) - ctx.on('agent/request-error', async (_agent, _turn, _step, _error, _failure, _history, _signal, next) => { - recoveries += 1 - return next() - }) - - send(agent) - await waitForIdle(ctx, agent) - - expect(recoveries).toBe(0) - expect(agent.session.events.at(-1)).toMatchObject({ type: 'turn/end', data: { reason: { kind: 'error' } } }) - }) - - it('does not offer a nested model-call failure as the outer request failure', async () => { - const outer = new FailureScriptAdapter([textResponse('outer adapter must not run')]) - const nested = new FailureScriptAdapter([contextError('nested overflow')]) - const ctx = await harness(outer) - ctx.llm.registerAdapter(['nested'], nested) - ctx.on('llm/stream', (options, next) => { - if (options.provider !== 'mock') return next() - return (async function* () { - yield* ctx.llm.stream({ - provider: 'nested', - model: 'nested', - messages: [], - ...options.signal === undefined ? {} : { signal: options.signal }, - }) - yield* next() - })() - }) - const agent = ctx.agentLoop.create(SessionId('nested-stream-not-recoverable'), { provider: 'mock', model: 'mock' }) - let recoveries = 0 - ctx.on('agent/request-error', async (_agent, _turn, _step, _error, _failure, _history, _signal, next) => { - recoveries += 1 - return next() - }) - - send(agent) - await waitForIdle(ctx, agent) - - expect(nested.requests).toHaveLength(1) - expect(outer.requests).toHaveLength(0) - expect(recoveries).toBe(0) - expect(agent.session.events.at(-1)).toMatchObject({ - type: 'turn/end', - data: { reason: { kind: 'error', message: 'nested overflow', code: CONTEXT_WINDOW_EXCEEDED_CODE } }, - }) - }) - - it.each(['prompt-submit', 'prompt-assembly', 'pre-step', 'request'] as const)( - 'does not offer %s middleware failures to request recovery', - async (boundary) => { - const adapter = new FailureScriptAdapter([textResponse('unused')]) - const ctx = await harness(adapter) - if (boundary === 'prompt-submit') { - ctx.on('agent/prompt-submit', () => { throw new Error('prompt submit failed') }) - } else if (boundary === 'prompt-assembly') { - ctx.on('system-prompt/assemble', () => { throw new Error('prompt assembly failed') }) - } else if (boundary === 'pre-step') { - ctx.on('agent/pre-step', () => { throw new Error('pre-step failed') }) - } else { - ctx.on('agent/request', () => { throw new Error('request middleware failed') }) - } - const agent = ctx.agentLoop.create(SessionId(`${boundary}-not-recoverable`), { provider: 'mock', model: 'mock' }) - let recoveries = 0 - ctx.on('agent/request-error', async (_agent, _turn, _step, _error, _failure, _history, _signal, next) => { - recoveries += 1 - return next() - }) - - send(agent) - await waitForIdle(ctx, agent) - - expect(recoveries).toBe(0) - expect(adapter.requests).toHaveLength(0) - expect(agent.session.events.at(-1)).toMatchObject({ type: 'turn/end', data: { reason: { kind: 'error' } } }) - }, - ) - - it('does not offer result, tool, or post-step plugin failures to request recovery', async () => { - for (const failure of ['result', 'tool', 'post-step'] as const) { - const adapter = new FailureScriptAdapter([ - failure === 'tool' ? toolCallResponse(`call-${failure}`, 'work', {}) : textResponse('done'), - ...(failure === 'tool' ? [textResponse('done')] : []), - ]) - const ctx = await harness(adapter) - if (failure === 'result') ctx.on('agent/step-result', () => { throw new Error('result failed') }) - if (failure === 'post-step') ctx.on('agent/post-step', () => { throw new Error('post-step failed') }) - if (failure === 'tool') { - vi.spyOn(ctx.tools, 'execute').mockRejectedValue(new Error('tool service failed')) - } - const agent = ctx.agentLoop.create(SessionId(`${failure}-not-recoverable`), { provider: 'mock', model: 'mock' }) - let recoveries = 0 - ctx.on('agent/request-error', async (_agent, _turn, _step, _error, _failure, _history, _signal, next) => { - recoveries += 1 - return next() - }) - send(agent) - await waitForIdle(ctx, agent) - expect(recoveries, failure).toBe(0) - } - }) - - it.each([ - ['synchronous dispatch', (error: Error) => new SynchronousDispatchFailureAdapter(error)], - ['done getter', (error: Error) => new IteratorResultGetterFailureAdapter('done', error)], - ['value getter', (error: Error) => new IteratorResultGetterFailureAdapter('value', error)], - ] as const)('preserves original Error identity for adapter %s', async (_name, makeAdapter) => { - const original = contextError(`${_name} overflow`) - const ctx = await harness(makeAdapter(original)) - const agent = ctx.agentLoop.create(SessionId(`identity-${_name.replaceAll(' ', '-')}`), { provider: 'mock', model: 'mock' }) - let seen: Error | undefined - ctx.on('agent/request-error', async (_agent, _turn, _step, error, _failure, _history, _signal, next) => { - seen = error - return next() - }) - - send(agent) - await waitForIdle(ctx, agent) - - expect(seen).toBe(original) - }) - - it('keeps an adapter error with a hostile message accessor on the recovery path', async () => { - const original = Object.defineProperty(new HarnessError('provider failed', 'SERVER'), 'message', { - get() { throw new Error('SDK message accessor trap') }, - }) - const ctx = await harness(new SynchronousDispatchFailureAdapter(original)) - const agent = ctx.agentLoop.create(SessionId('hostile-message-recovery'), { provider: 'mock', model: 'mock' }) - let seenError: Error | undefined - let seenFailure: LlmFailure | undefined - ctx.on('agent/request-error', async (_agent, _turn, _step, error, failure, _history, _signal, next) => { - seenError = error - seenFailure = failure - return next() - }) - - send(agent) - await waitForIdle(ctx, agent) - - expect(seenError).toBe(original) - expect(seenFailure).toEqual({ message: 'LLM adapter failed', code: 'SERVER' }) - expect(agent.session.events.at(-1)).toMatchObject({ - type: 'turn/end', - data: { reason: { kind: 'error', failure: { message: 'LLM adapter failed', code: 'SERVER' } } }, - }) - }) - - it('passes structured facts beside the original Error and records its cause chain on exhaustion', async () => { - const original = new LlmError('provider busy', 'RATE_LIMIT', { - cause: new Error('upstream connection reset'), - status: 429, - providerRetryAfterMs: 2_000, - requestId: ProviderRequestId('req-9'), - }) - Object.freeze(original) - const ctx = await harness(new SynchronousDispatchFailureAdapter(original)) - const agent = ctx.agentLoop.create(SessionId('structured-request-failure'), { provider: 'mock', model: 'mock' }) - let seenError: Error | undefined - let seenFailure: LlmFailure | undefined - let seenHistory: readonly LlmFailure[] | undefined - ctx.on('agent/request-error', async ( - _agent, _turn, _step, error, failure, history, _signal, next, - ) => { - seenError = error - seenFailure = failure - seenHistory = history - return next() - }) - - send(agent) - await waitForIdle(ctx, agent) - - expect(seenError).toBe(original) - expect(seenFailure).toEqual({ - message: 'provider busy', - code: 'RATE_LIMIT', - status: 429, - providerRetryAfterMs: 2_000, - requestId: ProviderRequestId('req-9'), - }) - expect(seenHistory).toEqual([]) - expect(Object.isFrozen(seenHistory)).toBe(true) - expect(agent.session.events.at(-1)).toMatchObject({ - type: 'turn/end', - data: { - reason: { - kind: 'error', - step: 1, - failure: { - message: 'provider busy: upstream connection reset', - code: 'RATE_LIMIT', - status: 429, - providerRetryAfterMs: 2_000, - requestId: ProviderRequestId('req-9'), - }, - }, - }, - }) - }) - - it('classifies iterator construction and explicit NO_ADAPTER as model-request failures', async () => { - for (const scenario of ['iterator', 'no-adapter'] as const) { - const ctx = scenario === 'iterator' ? await harness(new IteratorConstructionFailureAdapter()) : await harness() - const agent = ctx.agentLoop.create(SessionId(`request-boundary-${scenario}`), { provider: 'mock', model: 'mock' }) - let seen = '' - ctx.on('agent/request-error', async (_agent, _turn, _step, error, _failure, _history, _signal, next) => { - seen = error.code ?? '' - return next() - }) - send(agent) - await waitForIdle(ctx, agent) - expect(seen).toBe(scenario === 'iterator' ? 'ITERATOR_CONSTRUCTION' : 'NO_ADAPTER') - } - }) - - it('tracks consecutive retry attempts and resets after a successful request', async () => { - const capped = new FailureScriptAdapter([contextError('first overflow'), contextError('second overflow')]) - const cappedCtx = await harness(capped) - const cappedAgent = cappedCtx.agentLoop.create(SessionId('retry-cap'), { provider: 'mock', model: 'mock' }) - const cappedHistories: string[][] = [] - cappedCtx.on('agent/request-error', async ( - _agent, _turn, _step, _error, _failure, history, _signal, next, - ) => { - const codes = history.map(entry => entry.code) - cappedHistories.push(codes) - return codes.length < 1 ? { action: 'retry' } : next() - }) - send(cappedAgent) - await waitForIdle(cappedCtx, cappedAgent) - expect(cappedHistories).toEqual([[], [CONTEXT_WINDOW_EXCEEDED_CODE]]) - - const reset = new FailureScriptAdapter([ - contextError('first overflow'), - toolCallResponse('retry-reset-call', 'work', {}), - contextError('later overflow'), - ]) - const resetCtx = await harness(reset) - resetCtx.tools.register(defineContentToolFixture({ - name: 'work', - description: 'continue', - parameters: {}, - async execute() { return [{ type: 'text', text: 'worked' }] }, - })) - const resetAgent = resetCtx.agentLoop.create(SessionId('retry-reset'), { provider: 'mock', model: 'mock' }) - const resetHistories: { step: number; codes: string[] }[] = [] - resetCtx.on('agent/request-error', async ( - _agent, _turn, step, _error, _failure, history, _signal, next, - ) => { - resetHistories.push({ step, codes: history.map(entry => entry.code) }) - return resetHistories.length === 1 ? { action: 'retry' } : next() - }) - send(resetAgent) - await waitForIdle(resetCtx, resetAgent) - expect(resetHistories).toEqual([{ step: 1, codes: [] }, { step: 3, codes: [] }]) - }) - - it('preserves the original provider error when recovery throws', async () => { - const adapter = new FailureScriptAdapter([contextError('original overflow')]) - const ctx = await harness(adapter) - const agent = ctx.agentLoop.create(SessionId('recovery-throws'), { provider: 'mock', model: 'mock' }) - ctx.on('agent/request-error', () => { throw new Error('recovery exploded') }) - - send(agent) - await waitForIdle(ctx, agent) - - expect(agent.session.events.at(-1)).toMatchObject({ - type: 'turn/end', - data: { reason: { kind: 'error', failure: { message: 'original overflow', code: CONTEXT_WINDOW_EXCEEDED_CODE } } }, - }) - }) - - it.each(['cancel', 'dispose'] as const)('keeps %s live through request recovery', async (action) => { - const adapter = new FailureScriptAdapter([contextError()]) - const ctx = await harness(adapter) - const agent = ctx.agentLoop.create(SessionId(`${action}-recovery`), { provider: 'mock', model: 'mock' }) - let entered!: () => void - const recoveryEntered = new Promise((resolve) => { entered = resolve }) - ctx.on('agent/request-error', async (_agent, _turn, _step, _error, _failure, _history, signal) => { - entered() - await new Promise((resolve) => { - signal.addEventListener('abort', () => { resolve() }, { once: true }) - }) - return { action: 'retry' } - }) - - send(agent) - const idle = waitForIdle(ctx, agent) - await recoveryEntered - if (action === 'cancel') { - agent.cancel({ kind: 'user' }) - await idle - } else { - await ctx.fiber.dispose() - } - - expect(adapter.requests).toHaveLength(1) - expect(agent.session.events.at(-1)).toMatchObject({ - type: 'turn/end', - data: { reason: action === 'cancel' ? { kind: 'aborted' } : { kind: 'disposed' } }, - }) - }) -}) diff --git a/packages/core/agent-loop/tests/resume.spec.ts b/packages/core/agent-loop/tests/resume.spec.ts index 4e0d557ed9..c9a438e2ab 100644 --- a/packages/core/agent-loop/tests/resume.spec.ts +++ b/packages/core/agent-loop/tests/resume.spec.ts @@ -261,10 +261,7 @@ describe('the session-persistence Agent Note: AgentLoop factory create/resume', resumeSessionId: sessionId, agentOptions: { provider: 'mock', model: 'mock' }, }) - const transactionLabels = [ - `agentLoop.owner(${sessionId})`, - `agentLoop.lifecycle(${sessionId})`, - ] + const transactionLabels = [`agentLoop.lifecycle(${sessionId})`] expect(ctx.fiber.getEffects().map(effect => effect.label)).toEqual(expect.arrayContaining(transactionLabels)) await handle.dispose() diff --git a/packages/core/agent-loop/tests/scope-lifecycle.spec.ts b/packages/core/agent-loop/tests/scope-lifecycle.spec.ts index b9535e9448..cccd72123a 100644 --- a/packages/core/agent-loop/tests/scope-lifecycle.spec.ts +++ b/packages/core/agent-loop/tests/scope-lifecycle.spec.ts @@ -456,7 +456,7 @@ describe('agent scope lifecycle', () => { }) await expect(creating).rejects.toThrow(/agent loop is not active/) await loopFiber.dispose() - expect(setupCalls).toBe(0) + expect(setupCalls).toBe(1) expect(ctx.agents.get(SessionId('factory-scope-race-s'))).toBeUndefined() expect(ctx.sessions.get(SessionId('factory-scope-race-s'))).toBeUndefined() @@ -509,17 +509,17 @@ describe('agent scope lifecycle', () => { const { ctx, loopFiber } = await harnessWithLoop() const sessionsBefore = ctx.sessions.list().length let unloaded = false + let unloading!: Promise ctx.on('internal/plugin', (fiber) => { if (unloaded || fiber.name !== 'scope') return unloaded = true - void loopFiber.dispose() + unloading = loopFiber.dispose() }) - expect(() => ctx.agentLoop.create(SessionId('config-scope-race'), { provider: 'mock', model: 'mock' })) - .toThrow(/agent loop is not active/) - await loopFiber.dispose() - expect(ctx.agents.get(SessionId('config-scope-race'))).toBeUndefined() - expect(ctx.sessions.list()).toHaveLength(sessionsBefore) + ctx.agentLoop.create(SessionId('config-scope-race'), { provider: 'mock', model: 'mock' }) + await unloading + expect(ctx.agents.get(SessionId('config-scope-race')) === undefined).toBe(true) + expect(ctx.sessions.list().length).toBe(sessionsBefore) await ctx.fiber.dispose() }) @@ -566,16 +566,16 @@ describe('agent scope lifecycle', () => { }) await loopFiber.dispose() - expect(handle.agent.status).toBe('disposed') + expect(handle.agent.status).toBe('idle') expect(ctx.agents.get(sessionId)).toBeUndefined() expect(ctx.sessions.get(sessionId)).toBeUndefined() - expect(ctx.fiber.getEffects().filter(effect => effect.label === `agentLoop.owner(${sessionId})`)).toEqual([]) + expect(ctx.fiber.getEffects().filter(effect => effect.label === `agentLoop.lifecycle(${sessionId})`)).toEqual([]) // The consumer handle shares the provider's completed quiescence boundary. await handle.dispose() await expect(loop.createAgent(ctx, { sessionId: SessionId('factory-inactive-s'), - })).rejects.toThrow('agent loop is not active') + })).rejects.toThrow(/agent loop is not active|inactive context/) await ctx.fiber.dispose() }) @@ -643,13 +643,13 @@ describe('agent scope lifecycle', () => { }) }, { inject: ['agents'] })) - await expect(creating).rejects.toThrow(/lifecycle disposed/) + await expect(creating).rejects.toThrow(/owner disposed during setup/) await owner.dispose() expect(lifecycle).toEqual([ 'session-created:dispose', 'session-created:observer', - 'session-disposed', 'scope-disposed', + 'session-disposed', ]) expect(ctx.agents.get(SessionId('session-created-barrier-s'))).toBeUndefined() expect(ctx.sessions.get(SessionId('session-created-barrier-s'))).toBeUndefined() @@ -691,15 +691,15 @@ describe('agent scope lifecycle', () => { }) }, { inject: ['agents'] })) - await expect(creating).rejects.toThrow(/lifecycle disposed/) + await expect(creating).rejects.toThrow(/owner disposed during setup/) await owner.dispose() expect(lifecycle).toEqual([ 'session-created', 'agent-created:dispose', 'agent-created:observer', + 'scope-disposed', 'agent-disposed', 'session-disposed', - 'scope-disposed', ]) expect(ctx.agents.get(SessionId('agent-created-barrier-s'))).toBeUndefined() expect(ctx.sessions.get(SessionId('agent-created-barrier-s'))).toBeUndefined() @@ -713,7 +713,7 @@ describe('agent scope lifecycle', () => { let creating!: ReturnType ctx.on('agent/session-start', agent => void starts.push(agent.id)) ctx.on('agent/created', (agent) => { - if (agent.id === SessionId('listener-dispose-s')) void ownerCtx.fiber.dispose() + if (agent.id === SessionId('listener-dispose-s')) disposeCurrentLifecycle(ownerCtx) }) const owner = await ctx.plugin(Object.assign((inner: Context) => { @@ -727,8 +727,8 @@ describe('agent scope lifecycle', () => { await expect(creating).rejects.toThrow(/owner disposed during setup/) await owner.dispose() expect(starts).toEqual([]) - expect(ctx.agents.get(SessionId('listener-dispose-s'))).toBeUndefined() - expect(ctx.sessions.get(SessionId('listener-dispose-s'))).toBeUndefined() + expect(ctx.agents.get(SessionId('listener-dispose-s')) === undefined).toBe(true) + expect(ctx.sessions.get(SessionId('listener-dispose-s')) === undefined).toBe(true) await ctx.fiber.dispose() }) @@ -764,10 +764,10 @@ describe('agent scope lifecycle', () => { }) }, { inject: ['agents'] })) - await expect(creating).rejects.toThrow(/lifecycle disposed/) + await expect(creating).rejects.toThrow(/owner disposed during setup/) await owner.dispose() - expect(announced.status).toBe('disposed') - expect(statuses).toEqual(['disposed']) + expect(announced.status).toBe('idle') + expect(statuses).toEqual([]) expect(observerSawLive).toBe(true) expect(scopeDisposed).toBe(true) expect(announced.session.events).toEqual([]) @@ -886,8 +886,8 @@ describe('agent scope lifecycle', () => { expect(() => ctx.agentLoop.create(SessionId('config-bad'), { provider: 'mock', model: 'mock' })) .toThrow('config publish failed') - expect(ctx.agents.get(SessionId('config-bad'))).toBeUndefined() - expect(ctx.sessions.list()).toHaveLength(sessionsBefore) + await expect.poll(() => ctx.agents.get(SessionId('config-bad')) === undefined).toBe(true) + await expect.poll(() => ctx.sessions.list().length).toBe(sessionsBefore) }) it('registrations through a disposed agent ctx throw INACTIVE_EFFECT', async () => { @@ -937,7 +937,11 @@ describe('agent scope lifecycle', () => { agent.followup(text('work')) await turnOpen await owner.dispose() - expect(order).toEqual(['turn-end', 'disposed(listed=false)', 'session-still-stored=true']) + expect(order).toEqual([ + 'turn-end', + 'disposed(listed=false)', + 'session-still-stored=true', + ]) expect(ctx.sessions.get(SessionId('o1-s'))).toBeUndefined() }) @@ -970,9 +974,9 @@ describe('agent scope lifecycle', () => { agentOptions: { provider: 'mock', model: 'mock' }, }) - expect(ctx.fiber.getEffects().map(effect => effect.label)).toContain(`agentLoop.owner(${sessionId})`) + expect(ctx.fiber.getEffects().map(effect => effect.label)).toContain(`agentLoop.lifecycle(${sessionId})`) await handle.dispose() - expect(ctx.fiber.getEffects().filter(effect => effect.label === `agentLoop.owner(${sessionId})`)).toEqual([]) + expect(ctx.fiber.getEffects().filter(effect => effect.label === `agentLoop.lifecycle(${sessionId})`)).toEqual([]) await ctx.fiber.dispose() }) @@ -1007,15 +1011,11 @@ describe('agent scope lifecycle', () => { await ctx.fiber.dispose() }) - it('reopens ids after detach while the prior private scope finishes quiescing', async () => { + it('reopens ids after the prior private scope finishes quiescing', async () => { const ctx = await harness() const gate = Promise.withResolvers() const cleanupStarted = Promise.withResolvers() - const sessionDisposed = Promise.withResolvers() const sessionId = SessionId('quiescent-reuse') - ctx.on('session/disposed', (session) => { - if (session.id === sessionId) sessionDisposed.resolve(undefined) - }) const first = await ctx.agents.create({ sessionId, agentOptions: { provider: 'mock', model: 'mock' }, @@ -1028,15 +1028,16 @@ describe('agent scope lifecycle', () => { }) const disposing = first.dispose() - await Promise.all([sessionDisposed.promise, cleanupStarted.promise]) + await cleanupStarted.promise + expect(ctx.agents.get(sessionId)).toBe(first.agent) + expect(ctx.sessions.get(sessionId)).toBe(first.agent.session) + gate.resolve(undefined) + await disposing expect(ctx.agents.get(sessionId)).toBeUndefined() expect(ctx.sessions.get(sessionId)).toBeUndefined() const replacement = await ctx.agents.create({ sessionId, agentOptions: { provider: 'mock', model: 'mock' } }) expect(ctx.agents.get(sessionId)).toBe(replacement.agent) expect(ctx.sessions.get(sessionId)).toBe(replacement.agent.session) - - gate.resolve(undefined) - await disposing await replacement.dispose() await ctx.fiber.dispose() }) diff --git a/packages/core/agent-loop/tests/tool-calls.spec.ts b/packages/core/agent-loop/tests/tool-calls.spec.ts index 6b2130eaeb..16abfa89e8 100644 --- a/packages/core/agent-loop/tests/tool-calls.spec.ts +++ b/packages/core/agent-loop/tests/tool-calls.spec.ts @@ -9,7 +9,7 @@ import { CallId, StreamChunk } from '@deepseek-ai/dsh-llm' import SessionStore, { SessionEvent, SessionId } from '@deepseek-ai/dsh-session' import SystemPrompt from '@deepseek-ai/dsh-system-prompt' import LlmService from '@deepseek-ai/dsh-llm' -import ToolRegistry, { defineTool, TOOL_ABORTED_BEFORE_DISPATCH, type PostToolDecision, type PreToolDecision } from '@deepseek-ai/dsh-tools' +import ToolRegistry, { defineContentToolFixture, TOOL_ABORTED_BEFORE_DISPATCH, type PostToolDecision, type PreToolDecision } from '@deepseek-ai/dsh-tools' import AgentRegistry, { type Agent } from '@deepseek-ai/dsh-agent' import AgentLoop, { DEFAULT_MAX_PARALLEL_TOOL_CALLS } from '@deepseek-ai/dsh-agent-loop' import { MockAdapter, textResponse } from './mock-adapter.ts' @@ -61,7 +61,7 @@ function multiCall(calls: { id: string; name: string; args: object }[]): StreamC function gatedTool(name: string, parallel: boolean) { const gates = new Map void>() const started: string[] = [] - const tool = defineTool({ + const tool = defineContentToolFixture({ name, description: `gated ${name}`, parameters: { id: { type: 'string', required: true } }, @@ -123,12 +123,12 @@ describe('tool-call scheduler: grouping and barriers', () => { textResponse('done'), ]) const ctx = await harness(adapter) - ctx.tools.register(defineTool({ + ctx.tools.register(defineContentToolFixture({ name: 'r', description: 'read', parameters: { id: { type: 'string', required: true } }, isConcurrencySafe: () => true, async execute(args) { order.push(`r-start-${args.id}`); order.push(`r-end-${args.id}`); return [{ type: 'text', text: 'r' }] }, })) - ctx.tools.register(defineTool({ + ctx.tools.register(defineContentToolFixture({ name: 'w', description: 'write', parameters: { id: { type: 'string', required: true } }, async execute(args) { order.push(`w-${args.id}`); return [{ type: 'text', text: 'w' }] }, })) @@ -150,14 +150,14 @@ describe('tool-call scheduler: grouping and barriers', () => { ]) const ctx = await harness(adapter) const replacement = gatedExclusiveTool('x') - const disposeSafe = ctx.tools.register(defineTool({ + const disposeSafe = ctx.tools.register(defineContentToolFixture({ name: 'x', description: 'initially safe', parameters: { id: { type: 'string', required: true } }, isConcurrencySafe: () => true, async execute(args) { return [{ type: 'text', text: `old-${args.id}` }] }, })) - ctx.tools.register(defineTool({ + ctx.tools.register(defineContentToolFixture({ name: 'replace', description: 'replace x', parameters: { id: { type: 'string', required: true } }, @@ -567,7 +567,7 @@ describe('tool-call scheduler: abort handling', () => { const gated = gatedParallelTool('p') const exclusive: string[] = [] ctx.tools.register(gated.tool) - ctx.tools.register(defineTool({ + ctx.tools.register(defineContentToolFixture({ name: 'x', description: 'exclusive', parameters: { id: { type: 'string', required: true } }, diff --git a/packages/core/agent-loop/tests/tool-order.spec.ts b/packages/core/agent-loop/tests/tool-order.spec.ts index a39961bfbf..30384e7b0d 100644 --- a/packages/core/agent-loop/tests/tool-order.spec.ts +++ b/packages/core/agent-loop/tests/tool-order.spec.ts @@ -98,7 +98,9 @@ describe('loop-level canonical tool order', () => { const ctx = await harness(adapter, ['ghost', TOOL_ORDER_REST]) registerNamed(ctx, 'alpha') const errors: Error[] = [] - ctx.on('agent/error', (_agent, _turn, _step, error) => void errors.push(error)) + ctx.on('agent/error', (_agent, _turn, _step, error) => { + if (error instanceof Error) errors.push(error) + }) const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) agent.followup([{ type: 'text', text: 'go' }]) await waitForIdle(ctx, agent) diff --git a/packages/core/agent-loop/tests/turn-stop.spec.ts b/packages/core/agent-loop/tests/turn-stop.spec.ts deleted file mode 100644 index 44e1fd8a7b..0000000000 --- a/packages/core/agent-loop/tests/turn-stop.spec.ts +++ /dev/null @@ -1,196 +0,0 @@ -import { describe, expect, it } from 'vitest' -import { Context } from 'cordis' -import LlmService from '@deepseek-ai/dsh-llm' -import SessionStore, { SessionId, type TurnEndReason } from '@deepseek-ai/dsh-session' -import SystemPrompt from '@deepseek-ai/dsh-system-prompt' -import ToolRegistry, { defineContentToolFixture } from '@deepseek-ai/dsh-tools' -import AgentRegistry, { type Agent, type ContinuationStop } from '@deepseek-ai/dsh-agent' - -import AgentLoop from '@deepseek-ai/dsh-agent-loop' -import InvariantService from '@deepseek-ai/dsh-invariants' -import * as SessionInvariant from '@deepseek-ai/dsh-session/invariant' -import * as AgentInvariant from '@deepseek-ai/dsh-agent/invariant' -import * as AgentLoopInvariant from '@deepseek-ai/dsh-agent-loop/invariant' -import { MockAdapter, textResponse, toolCallResponse } from './mock-adapter.ts' - -async function mountInvariants(ctx: Context): Promise { - await ctx.plugin(InvariantService) - await ctx.plugin(SessionInvariant) - await ctx.plugin(AgentInvariant) - await ctx.plugin(AgentLoopInvariant) -} - -async function harness(adapter: MockAdapter): Promise { - const ctx = new Context() - await ctx.plugin(LlmService) - await ctx.plugin(SessionStore) - await ctx.plugin(SystemPrompt) - await ctx.plugin(ToolRegistry) - await ctx.plugin(AgentRegistry) - await mountInvariants(ctx) - await ctx.plugin(AgentLoop, { agents: [] }) - ctx.llm.registerAdapter(['mock'], adapter) - return ctx -} - -function send(agent: Agent, text = 'go'): Promise { - agent.followup([{ type: 'text', text }]) - return agent.whenIdle() -} - -function registerEcho(ctx: Context): void { - ctx.tools.register(defineContentToolFixture({ - name: 'echo', - description: 'echo', - parameters: { text: { type: 'string' } }, - async execute(args) { - return [{ type: 'text', text: String(args.text) }] - }, - })) -} - -describe('agent/turn-stop', () => { - it('runs after steering folding and discards terminal steering instead of creating another step or turn', async () => { - const adapter = new MockAdapter([ - textResponse('the ordinary decision is stop'), - textResponse('must not be requested'), - ]) - const ctx = await harness(adapter) - const agent = ctx.agentLoop.create(SessionId('terminal-steering'), { provider: 'mock', model: 'mock' }) - agent.ctx.on('agent/turn-stop', (): ContinuationStop => ({ action: 'stop' })) - - let steered = false - ctx.on('agent/turn-continuation', async (subject, _turn, _default, _signal, next) => { - const downstream = await next() - if (subject === agent && !steered) { - steered = true - subject.steer([{ type: 'text', text: 'late continuation steering' }]) - } - return downstream - }, { prepend: true }) - - await send(agent) - - expect(adapter.requests).toHaveLength(1) - expect(agent.session.events.filter(event => event.type === 'turn/start')).toHaveLength(1) - expect(agent.session.events.filter(event => event.type === 'step/start')).toHaveLength(1) - expect(agent.session.events.filter(event => event.type === 'steering/message')).toHaveLength(0) - }) - - it('discards steering that arrives from session/flush after the terminal checkpoint', async () => { - const adapter = new MockAdapter([ - textResponse('terminal answer'), - textResponse('must not become a late-steering turn'), - ]) - const ctx = await harness(adapter) - const agent = ctx.agentLoop.create(SessionId('terminal-flush-steering'), { provider: 'mock', model: 'mock' }) - agent.ctx.on('agent/turn-stop', (): ContinuationStop => ({ action: 'stop' })) - - let injected = false - ctx.on('session/flush', (session) => { - if (session !== agent.session || injected) return - injected = true - agent.steer([{ type: 'text', text: 'steering from flush' }]) - }) - - await send(agent) - - expect(injected).toBe(true) - expect(agent.status).toBe('idle') - expect(adapter.requests).toHaveLength(1) - expect(agent.session.events.filter(event => event.type === 'turn/start')).toHaveLength(1) - expect(agent.session.events.filter(event => event.type === 'step/start')).toHaveLength(1) - expect(agent.session.events.filter(event => event.type === 'steering/message')).toHaveLength(0) - }) - - it('preserves an ordinary queued send that arrives during terminal flush', async () => { - const adapter = new MockAdapter([ - textResponse('first terminal answer'), - textResponse('queued follow-up answer'), - ]) - const ctx = await harness(adapter) - const agent = ctx.agentLoop.create(SessionId('terminal-flush-send'), { provider: 'mock', model: 'mock' }) - agent.ctx.on('agent/turn-stop', (): ContinuationStop => ({ action: 'stop' })) - - let queued = false - ctx.on('session/flush', (session) => { - if (session !== agent.session || queued) return - queued = true - agent.followup([{ type: 'text', text: 'ordinary queued follow-up' }]) - }) - - await send(agent) - - expect(agent.status).toBe('idle') - expect(adapter.requests).toHaveLength(2) - expect(agent.session.events.filter(event => event.type === 'turn/start')).toHaveLength(2) - expect(agent.session.events.filter(event => event.type === 'step/start')).toHaveLength(2) - }) - - it('filters a scoped terminal listener to its own agent', async () => { - const adapter = new MockAdapter([ - toolCallResponse('a1', 'echo', { text: 'a' }), - toolCallResponse('b1', 'echo', { text: 'b' }), - textResponse('b continues normally'), - ]) - const ctx = await harness(adapter) - registerEcho(ctx) - const stopped = ctx.agentLoop.create(SessionId('stopped'), { provider: 'mock', model: 'mock' }) - const ordinary = ctx.agentLoop.create(SessionId('ordinary'), { provider: 'mock', model: 'mock' }) - stopped.ctx.on('agent/turn-stop', (): ContinuationStop => ({ action: 'stop' })) - - await send(stopped) - expect(adapter.requests).toHaveLength(1) - await send(ordinary) - - expect(adapter.requests).toHaveLength(3) - expect(stopped.session.events.filter(event => event.type === 'step/start')).toHaveLength(1) - expect(ordinary.session.events.filter(event => event.type === 'step/start')).toHaveLength(2) - }) - - it('unregisters with its scoped owner disposer', async () => { - const adapter = new MockAdapter([ - toolCallResponse('first', 'echo', { text: 'first' }), - toolCallResponse('second', 'echo', { text: 'second' }), - textResponse('continued after listener disposal'), - ]) - const ctx = await harness(adapter) - registerEcho(ctx) - const agent = ctx.agentLoop.create(SessionId('owned-listener'), { provider: 'mock', model: 'mock' }) - const disposeStop = agent.ctx.on('agent/turn-stop', (): ContinuationStop => ({ action: 'stop' })) - - await send(agent, 'first turn') - expect(adapter.requests).toHaveLength(1) - - disposeStop() - await send(agent, 'second turn') - expect(adapter.requests).toHaveLength(3) - }) - - it('fails a throwing terminal policy closed while the driver survives', async () => { - const adapter = new MockAdapter([ - textResponse('throwing policy'), - textResponse('healthy later turn'), - ]) - const ctx = await harness(adapter) - const agent = ctx.agentLoop.create(SessionId('bad-policy'), { provider: 'mock', model: 'mock' }) - const reasons: TurnEndReason[] = [] - const errors: string[] = [] - ctx.on('session/event', (session, event) => { - if (session === agent.session && event.type === 'turn/end') reasons.push(event.data.reason) - }) - agent.ctx.on('agent/error', (_subject, _turn, _step, error) => { errors.push(error.message) }) - - const disposeThrowing = agent.ctx.on('agent/turn-stop', () => { - throw new Error('terminal policy exploded') - }) - await send(agent, 'first') - disposeThrowing() - - await send(agent, 'healthy') - - expect(reasons.map(reason => reason.kind)).toEqual(['error', 'completed']) - expect(errors).toContain('terminal policy exploded') - expect(adapter.requests).toHaveLength(2) - }) -}) diff --git a/packages/core/agent/README.md b/packages/core/agent/README.md index 9c4e79012a..86796fa9f0 100644 --- a/packages/core/agent/README.md +++ b/packages/core/agent/README.md @@ -46,7 +46,7 @@ Agent *creation* is provided by the plugin implementing `AgentFactory` (`dsh-age The lifecycle edges have two important local caveats. `agent/created` runs after scoped setup and after both session and agent registry entries exist. Setup is trusted composition-only code; the immediately following non-vetoing `agent/session-start` notification is the first supported startup injection point. `agent/disposed` always means the exact agent has left the registry. AgentLoop emits it after its driver is quiescent, while ordered teardown may still be detaching the session and unwinding the scope; custom agents registered directly own any stronger driver-ordering contract themselves. -Most interception points are cooperative waterfalls returning seam-specific decisions. Turn-scoped asynchronous seams receive one explicit `AbortSignal`, with `signal` immediately before a waterfall's final `next`; listeners may cooperate but must not retain it as authority over another turn. The signal remains authoritative through terminal policy and is retired immediately before `turn/end` publication, so terminal observers and the following durability flush cannot cancel completed turn work. `agent/pre-step` and `agent/post-step` are serial checkpoints around a step's durable work, while `agent/request-error` is the failed-model-request recovery waterfall: it receives the exact error, normalized failure facts, immutable prior-retried facts, and signal after the failed step closes; a retry opens a new numbered step. `agent/turn-stop` is the terminal serial fold: it runs after ordinary continuation and steering folding, and a returned stop remains in force through turn close and flush so later steering cannot create an extra step or turn. Ordinary queued prompts remain intact. Effective broad cancellation first emits the observe-only `agent/cancel-requested` with its resolved typed cause, then clears queues and aborts; notification failures are contained and cannot veto the stop. The [explicit-cancellation decision](../../../.agents/notes/implemented/architecture/2026-07-16-explicit-turn-cancellation.md) owns signal lifetime; the [agent-scope runtime-design Agent Note](../../../.agents/notes/implemented/architecture/2026-07-12-agent-scope-runtime-design.md#three-execution-boundaries-are-deliberately-one-way) owns scoped dispatch and terminal settlement. +Most interception points are cooperative waterfalls. Turn-scoped asynchronous seams receive one explicit `AbortSignal`, with `signal` immediately before a waterfall's final `next`; listeners may cooperate but must not retain it as authority over another turn. `agent/step` is the serial checkpoint before request derivation, while `agent/request-error` is the failed-model-request recovery waterfall: it receives the exact error, normalized failure facts, and signal after the failed step closes. A listener calls `agent.retry()` and returns without `next()` when it owns recovery; the loop closes the failed turn and opens one numbered retry turn. `agent/stopping` runs before an otherwise completed turn closes. Ordinary queued prompts remain intact. Effective broad cancellation first emits the observe-only `agent/cancel-requested` with its resolved typed cause, then clears queues and aborts; notification failures are contained and cannot veto the stop. The [explicit-cancellation decision](../../../.agents/notes/implemented/architecture/2026-07-16-explicit-turn-cancellation.md) owns signal lifetime; the [agent-scope runtime-design Agent Note](../../../.agents/notes/implemented/architecture/2026-07-12-agent-scope-runtime-design.md#three-execution-boundaries-are-deliberately-one-way) owns scoped dispatch and terminal settlement. `PromptDecision.additionalContexts` is an array so every context keeps its own source. Allowed prompt content and every additional context become separate model-facing `user/message` events before the turn runs. A listener that wraps a downstream allow preserves its `content` and `additionalContexts` unless it intentionally replaces either field; the returned allow is authoritative. diff --git a/packages/core/agent/src/dispatch.ts b/packages/core/agent/src/dispatch.ts index 9444f8f883..8b222327f2 100644 --- a/packages/core/agent/src/dispatch.ts +++ b/packages/core/agent/src/dispatch.ts @@ -66,7 +66,11 @@ export interface AgentEventDispatch { waterfall(name: K, ...rest: Tail): Return } -/** Return the fused scope carrier for one agent subject. */ +/** + * Return the fused scope carrier for one agent subject. + * @param agent - the subject agent and scope key. + * @returns the carrier passed as the event dispatcher `this` value. + */ export function agentCarrier(agent: Agent): Scoped { return scopeTarget(agent, agent) } @@ -116,7 +120,13 @@ export function agentEvents(ctx: Context, agent: Agent): AgentEventDispatch { } } -/** Emit one contained agent notification without allocating a retained dispatcher. */ +/** + * Emit one contained agent notification without allocating a retained dispatcher. + * @param ctx - the context to dispatch through. + * @param agent - the subject agent and scope key. + * @param name - the agent-subject event to emit. + * @param rest - the event arguments after the injected agent. + */ export function emitAgentEvent( ctx: Context, agent: Agent, diff --git a/packages/core/agent/src/types.ts b/packages/core/agent/src/types.ts index aa97e1ecf7..d352000a8c 100644 --- a/packages/core/agent/src/types.ts +++ b/packages/core/agent/src/types.ts @@ -103,8 +103,8 @@ export interface CancelOptions { /** * An agent's lifecycle state, emitted on every transition as `agent/status`: * `idle` (parked, waiting for queued work), `running` (the driver is draining - * work and may be closing or checkpointing a turn), `disposed` (terminal — no - * transition leaves it, and `send`/`followup`/`steer`/`inject` throw). + * work and may be closing or checkpointing a turn). Disposal removes the + * agent from its registry; it is not a third observable status. */ export type AgentStatus = 'idle' | 'running' @@ -121,11 +121,13 @@ export type PromptDecision = | { kind: 'allow'; content?: ContentBlock[]; additionalContexts?: AdditionalContext[] } | { kind: 'block'; reason: string } +/** Model-request failure with an optional machine-routable provider code. */ +export type RequestError = Error & { code?: string } + /** * Why a turn ended, reported live on `agent/idle` right after the turn's - * durable `turn/end` and flush. `error` carries the thrown value verbatim (and, for - * model-request failures, the adapter-normalized facts) so a recovery - * consumer can decide to repair and {@link Agent.retry}. + * durable `turn/end`. `error` carries the thrown value verbatim for observers; + * model-request recovery runs earlier through `agent/request-error`. */ export type IdleReason = | { kind: 'completed' } @@ -179,9 +181,8 @@ export abstract class Agent { * Clear queued and steering work — unless `keepInbox` — and abort the active * turn. An effective call first emits `agent/cancel-requested` with the * resolved typed cause. The first cause wins for the active turn, and - * `whenIdle()` resolves after cancellation reaches quiescence. Omitted cause - * means `{ kind: 'user' }`. Idle cancellation is a no-op and does not arm - * later work. The active turn snapshots and freezes the cause. + * `whenIdle()` resolves after cancellation reaches quiescence. Idle + * cancellation is a no-op and does not arm later work. * @param cause - the stable caller intent carried by the current turn signal. * @param options - cancellation options; `keepInbox` preserves pending work. */ @@ -245,11 +246,10 @@ export abstract class Agent { /** * Re-open a turn on the current session log without a new prompt — the - * recovery verb. After an `agent/idle` error, a consumer repairs (edits the - * log, waits out a rate limit) and calls this; the machine immediately runs - * another turn over the repaired history. Calling it synchronously from an - * `agent/idle` listener is legal — the machine is already idle there. - * @throws while a turn is running because there is nothing to retry yet. + * explicit resummon verb. During `agent/request-error`, this schedules one + * retry turn after the failed turn closes; while idle, it starts one + * immediately. Repeated calls before the scheduled retry coalesce. + * @throws while other agent work is running. */ abstract retry(): void } @@ -270,7 +270,7 @@ declare module 'cordis' { 'agent/created'(this: Scoped, agent: Agent): void /** * An agent left the registry; AgentLoop emits this after driver quiescence - * but before session detachment and scoped-registration unwind. Custom + * and scoped-registration unwind, but before session detachment. Custom * registry users own their driver-ordering contract. * @param agent - the exact agent removed from the registry. * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent. @@ -278,8 +278,8 @@ declare module 'cordis' { */ 'agent/disposed'(this: Scoped, agent: Agent): void /** - * Agent status changed (`idle` ⇄ `running`, or → `disposed`). `send()` does - * not enter `running` synchronously; drive lifecycle from this event. + * Agent status changed (`idle` ⇄ `running`). `send()` does not enter + * `running` synchronously; drive lifecycle from this event. * @param agent - the agent whose status flipped. * @param status - the status just entered (the transition's destination). * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent. @@ -321,7 +321,7 @@ declare module 'cordis' { * is cleared or the active turn is aborted. This observe-only notification * cannot veto cancellation; listener failures are contained. * @param agent - the agent whose current work is being cancelled. - * @param cause - resolved typed cancellation cause, including the default. + * @param cause - the explicit typed cancellation cause. * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent. * @mode emit */ @@ -377,8 +377,23 @@ declare module 'cordis' { * @param signal - the current turn's explicit abort signal. * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent. * @mode waterfall - */ + */ 'agent/request'(this: Scoped, agent: Agent, turn: number, step: number, signal: AbortSignal, next: () => Promise): Promise + /** + * Handle a model-request failure after its failed step has closed but + * before the failed turn closes. A listener calls {@link Agent.retry} to + * schedule one retry turn, returns without `next()` when it owns the error, + * or calls `next()` to delegate. The default leaves the failure terminal. + * @param agent - the agent whose request failed. + * @param turn - the open turn number. + * @param step - the failed step number. + * @param error - the original model-request failure. + * @param failure - serializable facts normalized at the final adapter boundary. + * @param signal - the turn abort signal. + * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent. + * @mode waterfall + */ + 'agent/request-error'(this: Scoped, agent: Agent, turn: number, step: number, error: RequestError, failure: LlmFailure, signal: AbortSignal, next: () => Promise): Promise /** * The turn is about to close: the model owes no response (no live tool * calls, no fresh steering). Awaited before the boundary commits — a @@ -395,14 +410,13 @@ declare module 'cordis' { */ 'agent/stopping'(this: Scoped, agent: Agent, turn: number, signal: AbortSignal): Promise | void /** - * One turn closed: its `turn/end` and durability flush are already - * committed. `reason` says why — recovery consumers observe an `error` - * reason, repair (edit the log, wait, resummon), and call - * {@link Agent.retry}; UI consumers key turn-done presentation off it. - * Emitted per turn, including cancelled and failed ones. + * One drain chain reached its terminal turn: that turn's `turn/end` is + * already committed. Automatically recovered failed turns do not emit this + * notification. `reason` says why; model-request recovery is exhausted when + * an error reaches it. * @param agent - the agent whose turn closed. - * @param turn - the closed turn number. - * @param reason - why the turn ended, with live error facts when it failed. + * @param turn - the terminal turn number. + * @param reason - why the terminal turn ended, with live error facts when it failed. * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent. * @mode emit */ diff --git a/packages/core/agent/tests/agent.spec.ts b/packages/core/agent/tests/agent.spec.ts index 671b5fa559..ecf1c6f616 100644 --- a/packages/core/agent/tests/agent.spec.ts +++ b/packages/core/agent/tests/agent.spec.ts @@ -12,7 +12,6 @@ import AgentRegistry, { import type { AgentCancelCause, AgentFactory, - ContinuationStop, CreateAgentOptions, ResumeAgentOptions, SendOptions, @@ -30,6 +29,7 @@ function stubAgent(rawId: string, overrides: Partial = {}): Agent { ctx: new Context(), send: () => AgentMessageId('stub'), cancel() {}, + retry() {}, whenIdle() { return Promise.resolve() }, ...overrides, }) @@ -58,14 +58,6 @@ describe('Agent delivery aliases', () => { }) describe('AgentRegistry', () => { - it('allows terminal stop policy to cooperate asynchronously with turn cancellation', () => { - type TurnStopListener = Events['agent/turn-stop'] - type AsyncTurnStopListener = () => Promise - - expectTypeOf().toExtend() - expectTypeOf>>().toEqualTypeOf() - }) - it('registers exact entries, emits lifecycle events, and unregisters on owner disposal', async () => { const ctx = new Context() await ctx.plugin(AgentRegistry) @@ -219,7 +211,7 @@ describe('agentEvents()', () => { describe('explicit cancellation helpers', () => { it('exposes the closed typed cancellation cause at the Agent seam', () => { - expectTypeOf[0]>().toEqualTypeOf() + expectTypeOf[0]>().toEqualTypeOf() expectTypeOf[1]>().toEqualTypeOf() }) diff --git a/packages/core/agent/tests/invariant.spec.ts b/packages/core/agent/tests/invariant.spec.ts index 4c9afbb376..8b5e6ae7d7 100644 --- a/packages/core/agent/tests/invariant.spec.ts +++ b/packages/core/agent/tests/invariant.spec.ts @@ -17,19 +17,14 @@ function mockAgent(id: string): Agent { } describe('agent status invariants', () => { - it('accepts lifecycle transitions through idle, running, and disposed', async () => { + it('accepts lifecycle transitions between idle and running', async () => { const ctx = await setup() const agent = mockAgent('a1') expect(() => { ctx.emit(scopeTarget(agent, agent), 'agent/status', agent, 'idle') ctx.emit(scopeTarget(agent, agent), 'agent/status', agent, 'running') ctx.emit(scopeTarget(agent, agent), 'agent/status', agent, 'idle') - ctx.emit(scopeTarget(agent, agent), 'agent/status', agent, 'disposed') }).not.toThrow() - - const running = mockAgent('a2') - ctx.emit(scopeTarget(running, running), 'agent/status', running, 'running') - expect(() => { ctx.emit(scopeTarget(running, running), 'agent/status', running, 'disposed') }).not.toThrow() }) it('rejects a no-op transition', async () => { @@ -40,14 +35,6 @@ describe('agent status invariants', () => { .toThrow(/no-op transition/) }) - it('rejects leaving the terminal disposed state', async () => { - const ctx = await setup() - const agent = mockAgent('a4') - ctx.emit(scopeTarget(agent, agent), 'agent/status', agent, 'disposed') - expect(() => { ctx.emit(scopeTarget(agent, agent), 'agent/status', agent, 'idle') }) - .toThrow(/left terminal state disposed/) - }) - it('tracks agents independently', async () => { const ctx = await setup() const a = mockAgent('a5') diff --git a/packages/core/agent/tests/llm-target.spec.ts b/packages/core/agent/tests/llm-target.spec.ts index fa4ef2b459..23b727d5b9 100644 --- a/packages/core/agent/tests/llm-target.spec.ts +++ b/packages/core/agent/tests/llm-target.spec.ts @@ -21,25 +21,25 @@ describe('installAgentLlmTarget()', () => { expect((await ctx.systemPrompt.assemble()).variables).toEqual({}) await expect(agentEvents(ctx, agent).waterfall( - 'agent/request', 1, 0, seed, signal, () => Promise.resolve(seed), + 'agent/request', 1, 0, signal, () => Promise.resolve(seed), )).resolves.toBe(seed) target.current = { provider: 'alpha', model: 'a1' } expect((await ctx.systemPrompt.assemble()).variables).toMatchObject({ provider: 'alpha', model: 'a1' }) target.current = { provider: 'beta', model: 'b1' } await expect(agentEvents(ctx, agent).waterfall( - 'agent/request', 1, 0, seed, signal, () => Promise.resolve(seed), + 'agent/request', 1, 0, signal, () => Promise.resolve(seed), )).resolves.toEqual({ provider: 'alpha', model: 'a1', temperature: 0.2 }) expect((await ctx.systemPrompt.assemble()).variables).toMatchObject({ provider: 'beta', model: 'b1' }) await expect(agentEvents(ctx, agent).waterfall( - 'agent/request', 1, 1, seed, signal, () => Promise.resolve(seed), + 'agent/request', 1, 1, signal, () => Promise.resolve(seed), )).resolves.toEqual({ provider: 'beta', model: 'b1', temperature: 0.2 }) dispose() expect((await ctx.systemPrompt.assemble()).variables).toEqual({}) await expect(agentEvents(ctx, agent).waterfall( - 'agent/request', 2, 0, seed, signal, () => Promise.resolve(seed), + 'agent/request', 2, 0, signal, () => Promise.resolve(seed), )).resolves.toBe(seed) await ctx.fiber.dispose() }) diff --git a/packages/core/scope/src/scoped-events.generated.ts b/packages/core/scope/src/scoped-events.generated.ts index fc29ec1732..fec8ed873a 100644 --- a/packages/core/scope/src/scoped-events.generated.ts +++ b/packages/core/scope/src/scoped-events.generated.ts @@ -18,6 +18,7 @@ const scopedSubjectResolvers: Readonly args[0], 'agent/prompt-submit': args => args[0], 'agent/request': args => args[0], + 'agent/request-error': args => args[0], 'agent/session-start': args => args[0], 'agent/status': args => args[0], 'agent/step': args => args[0], diff --git a/packages/core/scope/tests/invariant.spec.ts b/packages/core/scope/tests/invariant.spec.ts index 8f29a98305..b0c156d354 100644 --- a/packages/core/scope/tests/invariant.spec.ts +++ b/packages/core/scope/tests/invariant.spec.ts @@ -37,7 +37,6 @@ describe('scoped-dispatch invariants', () => { const other = { id: 'a2' } as unknown as Agent const signal = new AbortController().signal const config = { provider: 'p', model: 'm' } - const message = { role: 'assistant' as const, content: [] } const agentRows = { 'agent/created': [agent], 'agent/disposed': [agent], @@ -47,15 +46,20 @@ describe('scoped-dispatch invariants', () => { 'agent/inbox/discard': [agent, []], 'agent/cancel-requested': [agent, { kind: 'user' }], 'agent/session-start': [agent, 'startup'], - 'agent/pre-step': [agent, 1, 1, signal], - 'agent/post-step': [agent, 1, 1, signal], + 'agent/step': [agent, 1, 1, signal], 'agent/prompt-submit': [agent, [], { kind: 'user' }, signal, () => Promise.resolve({ kind: 'allow' })], - 'agent/request': [agent, 1, 1, config, signal, () => Promise.resolve(config)], - 'agent/request-error': [agent, 1, 1, new Error('request failed'), { message: 'request failed', code: 'UNKNOWN' }, [], signal, () => Promise.resolve({ action: 'fail' })], - 'agent/session-prefix': [agent, [], signal, () => Promise.resolve([])], - 'agent/step-result': [agent, 1, 1, message, signal, () => Promise.resolve(message)], - 'agent/turn-continuation': [agent, 1, { action: 'stop' }, signal, () => Promise.resolve({ action: 'stop' })], - 'agent/turn-stop': [agent, 1, signal], + 'agent/request': [agent, 1, 1, signal, () => Promise.resolve(config)], + 'agent/request-error': [ + agent, + 1, + 1, + new Error('request'), + { message: 'request', code: 'UNKNOWN' }, + signal, + () => Promise.resolve(), + ], + 'agent/stopping': [agent, 1, signal], + 'agent/idle': [agent, 1, { kind: 'completed' }], 'agent/error': [agent, 1, 0, new Error('x')], } satisfies { [K in AgentEventName]: EventArgs } const rows: Array<[string, unknown[]]> = [ diff --git a/packages/core/session/src/types.ts b/packages/core/session/src/types.ts index 1814e5497c..f4a999e59a 100644 --- a/packages/core/session/src/types.ts +++ b/packages/core/session/src/types.ts @@ -85,13 +85,9 @@ export interface TurnTriggerMap { /** Recovery turn reopened over the repaired current session log. */ retry: { kind: 'retry' } /** - * An out-of-band context injection (`agent.inject()`) made while the agent - * was idle. The loop wraps the injected `user/message` (a non-`user` source, - * plugin by default) in a one-shot turn (`turn/start` → `user/message` → - * `turn/end`) so every event in the log stays turn-enclosed — the - * durability/replay boundary is the turn, and a bare event between turns would - * otherwise be indistinguishable from a crash tail on reload. The trigger's - * `source` mirrors that message's producer. + * An out-of-band producer explicitly enclosed injected context in a one-shot + * turn. `Agent.inject()` appends idle context directly and does not use this + * trigger; the source mirrors the producer of the enclosed `user/message`. */ injection: { kind: 'injection'; source: MessageSource } } diff --git a/packages/examples/acp-demo/tests/acp-agent.spec.ts b/packages/examples/acp-demo/tests/acp-agent.spec.ts index 18cd0b5221..cfc0e4e036 100644 --- a/packages/examples/acp-demo/tests/acp-agent.spec.ts +++ b/packages/examples/acp-demo/tests/acp-agent.spec.ts @@ -1,4 +1,5 @@ import { describe, expect, it } from 'vitest' +import { randomUUID } from 'node:crypto' import { mkdtemp } from 'node:fs/promises' import { join } from 'node:path' import { tmpdir } from 'node:os' @@ -46,12 +47,9 @@ async function isolatedSkillsConfig(catalogDescriptionMaxLength?: number): Promi } async function composePrefix(ctx: Context): Promise { - const agent = { session: { header: { cwd: '/tmp' } } } as unknown as Agent - const empty: Message[] = [] - return await agentEvents(ctx, agent).waterfall( - 'agent/session-prefix', empty, new AbortController().signal, - () => Promise.resolve(empty), - ) + const agent = ctx.agentLoop.create(SessionId(`acp-demo-prefix-${randomUUID()}`), {}, { cwd: '/tmp' }) + await agentEvents(ctx, agent).serial('agent/step', 1, 1, new AbortController().signal) + return agent.session.deriveMessages() } async function withIsolatedSkillHomes(run: () => Promise): Promise { diff --git a/packages/examples/agent-spine-demo/tests/agent-core.spec.ts b/packages/examples/agent-spine-demo/tests/agent-core.spec.ts index e00bf2016d..c3c162b4eb 100644 --- a/packages/examples/agent-spine-demo/tests/agent-core.spec.ts +++ b/packages/examples/agent-spine-demo/tests/agent-core.spec.ts @@ -26,12 +26,9 @@ declare module '@deepseek-ai/dsh-tasks' { } async function composePrefix(ctx: Context, cwd: string): Promise { - const agent = { session: { header: { cwd } } } as unknown as Agent - const empty: Message[] = [] - return await agentEvents(ctx, agent).waterfall( - 'agent/session-prefix', empty, new AbortController().signal, - () => Promise.resolve(empty), - ) + const agent = ctx.agentLoop.create(SessionId('agent-spine-prefix'), {}, { cwd }) + await agentEvents(ctx, agent).serial('agent/step', 1, 1, new AbortController().signal) + return agent.session.deriveMessages() } /** @@ -99,15 +96,8 @@ async function withIsolatedSkillHomes(run: () => Promise): Promise { } } -function waitForIdle(ctx: Context, target: Agent): Promise { - return new Promise((resolve) => { - const dispose = ctx.on('agent/status', (agent, status) => { - if (agent === target && status === 'idle') { - dispose() - resolve() - } - }) - }) +function waitForIdle(_ctx: Context, target: Agent): Promise { + return target.whenIdle() } function messageText(message: Message | undefined): string { @@ -236,6 +226,7 @@ describe('dsh-agent-spine-demo bundle', () => { }) handle.agent.followup([{ type: 'text', text: 'recover' }]) + await expect.poll(() => adapter.requests).toBe(2) await waitForIdle(ctx, handle.agent) expect(adapter.requests).toBe(2) @@ -457,8 +448,8 @@ describe('dsh-agent-spine-demo bundle', () => { handle.agent.followup([{ type: 'text', text: 'hi' }]) await waitForIdle(ctx, handle.agent) - expect(messageText(adapter.requests[0]?.messages[0])).toContain('workspace rule before skills') - expect(messageText(adapter.requests[0]?.messages[1])).toContain('prefix-order-skill') + expect(messageText(adapter.requests[0]?.messages[1])).toContain('workspace rule before skills') + expect(messageText(adapter.requests[0]?.messages[2])).toContain('prefix-order-skill') await handle.dispose() await ctx.fiber.dispose() } finally { diff --git a/packages/examples/cli-demo/src/cli.ts b/packages/examples/cli-demo/src/cli.ts index e6672f91ad..de09a7b490 100644 --- a/packages/examples/cli-demo/src/cli.ts +++ b/packages/examples/cli-demo/src/cli.ts @@ -225,22 +225,24 @@ export async function runOneShot(ctx: Context, options: OneShotOptions): Promise let targetTurn: number | undefined let reason: TurnEndReason | undefined let result = '' - const usageByStep = new Map() + const usageByStep = new Map() let outputError: Error | undefined let resolveTurn!: () => void let rejectTurn!: (error: Error) => void - let settled = false + let firstTurnEnded = false const turnEnded = new Promise((resolve, reject) => { resolveTurn = resolve rejectTurn = reject }) const settleResolved = (): void => { - settled = true + if (firstTurnEnded) return + firstTurnEnded = true resolveTurn() } const settleRejected = (error: Error): void => { - settled = true + if (firstTurnEnded) return + firstTurnEnded = true rejectTurn(error) } const observe = (sessionId: string, event: SessionEvent): void => { @@ -254,20 +256,26 @@ export async function runOneShot(ctx: Context, options: OneShotOptions): Promise } const disposeListener = ctx.on('session/event', (session, event) => { - if (session !== agent.session || settled) return + if (session !== agent.session) return if (targetTurn === undefined) { if (event.type !== 'turn/start' || event.data.trigger.kind !== 'message') return targetTurn = event.data.turn + } else if (event.type === 'turn/start' && event.data.trigger.kind === 'retry' + && reason?.kind === 'error') { + targetTurn = event.data.turn + reason = undefined } observe(session.id, event) if (event.type === 'assistant/chunk' && event.data.turn === targetTurn && event.data.chunk.type === 'usage') { - usageByStep.set(event.data.step, event.data.chunk.usage) + usageByStep.set(`${event.data.turn}/${event.data.step}`, event.data.chunk.usage) } if (event.type === 'assistant/message' && event.data.turn === targetTurn) { result = assistantText(event) ?? result - if (event.data.usage !== undefined) usageByStep.set(event.data.step, event.data.usage) + if (event.data.usage !== undefined) { + usageByStep.set(`${event.data.turn}/${event.data.step}`, event.data.usage) + } } if (event.type === 'turn/end' && event.data.turn === targetTurn) { reason = event.data.reason @@ -289,14 +297,14 @@ export async function runOneShot(ctx: Context, options: OneShotOptions): Promise try { /* v8 ignore next -- skips send only when cancellation wins the listener-registration race above */ - if (!settled) { // eslint-disable-line @typescript-eslint/no-unnecessary-condition + if (!firstTurnEnded) { // eslint-disable-line @typescript-eslint/no-unnecessary-condition agent.followup([{ type: 'text', text: options.task }]) } await turnEnded } finally { if (onAbort !== undefined) signal?.removeEventListener('abort', onAbort) - disposeListener() await agent.whenIdle() + disposeListener() } /* v8 ignore next 3 -- turnEnded resolves only from the matching branch that assigns both values */ diff --git a/packages/examples/cli-demo/tests/cli-demo.spec.ts b/packages/examples/cli-demo/tests/cli-demo.spec.ts index 9bd8088dbe..628c663b18 100644 --- a/packages/examples/cli-demo/tests/cli-demo.spec.ts +++ b/packages/examples/cli-demo/tests/cli-demo.spec.ts @@ -1,9 +1,11 @@ import { mkdtemp } from 'node:fs/promises' +import { randomUUID } from 'node:crypto' import { tmpdir } from 'node:os' import { join } from 'node:path' import { Context } from 'cordis' import Loader from '@cordisjs/plugin-loader' -import { agentEvents, type Agent } from '@deepseek-ai/dsh-agent' +import { agentEvents } from '@deepseek-ai/dsh-agent' +import { SessionId } from '@deepseek-ai/dsh-session' import { CallId, type Message } from '@deepseek-ai/dsh-llm' import { TOOL_ORDER_REST } from '@deepseek-ai/dsh-system-prompt' import type { ToolExecution } from '@deepseek-ai/dsh-tools' @@ -39,12 +41,9 @@ async function mount(config: cliDemo.Config, withBash = false): Promise } async function composePrefix(ctx: Context): Promise { - const agent = { session: { header: { cwd: '/tmp' } } } as unknown as Agent - const empty: Message[] = [] - return await agentEvents(ctx, agent).waterfall( - 'agent/session-prefix', empty, new AbortController().signal, - () => Promise.resolve(empty), - ) + const agent = ctx.agentLoop.create(SessionId(`cli-demo-prefix-${randomUUID()}`), {}, { cwd: '/tmp' }) + await agentEvents(ctx, agent).serial('agent/step', 1, 1, new AbortController().signal) + return agent.session.deriveMessages() } afterEach(async () => { diff --git a/packages/examples/cli-demo/tests/cli.spec.ts b/packages/examples/cli-demo/tests/cli.spec.ts index c57dda5b00..f82a62441a 100644 --- a/packages/examples/cli-demo/tests/cli.spec.ts +++ b/packages/examples/cli-demo/tests/cli.spec.ts @@ -319,7 +319,7 @@ describe('runOneShot and executeCli', () => { const { ctx, agent, persistenceRoot } = await harness([textResponse('final answer')]) const output = await invoke(ctx, ['task']) expect(output).toEqual({ code: 0, stdout: 'final answer\n', stderr: '' }) - expect(agent.status).toBe('disposed') + expect(agent.status).toBe('idle') const files = await readdir(persistenceRoot, { recursive: true }) expect(files.some(file => file.endsWith('.jsonl.zstd'))).toBe(true) }) @@ -379,11 +379,13 @@ describe('runOneShot and executeCli', () => { const output = await invoke(ctx, ['--output-format', 'stream-json', 'task']) const lines = output.stdout.trimEnd().split('\n').map(line => JSON.parse(line) as Record) const events = lines.slice(0, -1).map(line => line['event'] as SessionEvent) - expect(lines.at(-1)).toMatchObject({ type: 'result', success: true, turn: 2, result: 'streamed' }) - expect(events[0]).toMatchObject({ type: 'turn/start', data: { turn: 2, trigger: { kind: 'message' } } }) - expect(events.at(-1)).toMatchObject({ type: 'turn/end', data: { turn: 2 } }) + expect(lines.at(-1)).toMatchObject({ type: 'result', success: true, turn: 1, result: 'streamed' }) + expect(events[0]).toMatchObject({ type: 'turn/start', data: { turn: 1, trigger: { kind: 'message' } } }) + expect(events.at(-1)).toMatchObject({ type: 'turn/end', data: { turn: 1 } }) expect(lines.slice(0, -1).every(line => line['sessionId'] === agent.session.id)).toBe(true) - expect(events.some(event => event.type === 'user/message' && event.data.source.kind !== 'user')).toBe(false) + expect(events.some(event => event.type === 'user/message' + && event.data.source.kind === 'plugin' + && event.data.source.plugin === 'test')).toBe(false) }) it('emits partial data and a diagnostic for non-completed turns', async () => { @@ -409,7 +411,7 @@ describe('runOneShot and executeCli', () => { expect(JSON.parse(output.stdout)).toMatchObject({ success: false, reason: { kind: 'aborted' } }) expect(output.code).toBe(1) expect(output.stderr).toContain('turn 1 was aborted') - expect(agent.status).toBe('disposed') + expect(agent.status).toBe('idle') }) it('contains stream-writer failures, cancels, flushes, and returns the output error', async () => { @@ -444,7 +446,7 @@ describe('runOneShot and executeCli', () => { expect(output.code).toBe(1) expect(output.stdout).toBe('') expect(output.stderr).toContain('stdout closed') - expect(final.agent.status).toBe('disposed') + expect(final.agent.status).toBe('idle') const disposal = await harness([textResponse('answer')]) const disposalOutput = await invoke(disposal.ctx, ['task'], { failDispose: true }) diff --git a/packages/goal/command-goal/tests/command-goal.spec.ts b/packages/goal/command-goal/tests/command-goal.spec.ts index dff39850b0..799cf00104 100644 --- a/packages/goal/command-goal/tests/command-goal.spec.ts +++ b/packages/goal/command-goal/tests/command-goal.spec.ts @@ -52,6 +52,7 @@ function stubAgent(id: string): { agent: Agent; session: Session } { steer: () => AgentMessageId('stub'), inject(content, options) { appendInjection(session, content, options); return AgentMessageId('stub') }, cancel() { status = 'idle' }, + retry() {}, whenIdle() { return Promise.resolve() }, } return { agent, session } diff --git a/packages/goal/goal-session/src/index.ts b/packages/goal/goal-session/src/index.ts index ac5cd2ff54..71ba7bf432 100644 --- a/packages/goal/goal-session/src/index.ts +++ b/packages/goal/goal-session/src/index.ts @@ -360,15 +360,6 @@ export function apply(ctx: Context): void { if (state.openTurn !== undefined) state.attempt.turn = state.openTurn } return - case 'prompt/blocked': - if (state.attempt !== undefined && state.attempt.phase === 'queued' - && isGoalRoundSource(event.data.source) && sameRound(event.data.source, state.attempt)) { - /* v8 ignore next -- this driver's rejected message always follows its observed turn/start */ - if (state.openTurn !== undefined) state.attempt.turn = state.openTurn - state.attempt.rejectedReason = event.data.reason - if (event.data.reason === STALE_ROUND_REASON) state.attempt.stale = true - } - return case 'turn/end': if (state.attempt?.turn === event.data.turn) state.attempt.reason = event.data.reason /* v8 ignore next -- balanced live turns close the open turn just observed by this listener */ @@ -407,11 +398,27 @@ export function apply(ctx: Context): void { } if (!valid) { const attempt = state.attempt - if (attempt !== undefined && sameRound(source, attempt)) attempt.stale = true + if (attempt !== undefined && sameRound(source, attempt)) { + attempt.stale = true + state.attempt = undefined + } + requestDrive(state) return { kind: 'block', reason: STALE_ROUND_REASON } } const decision = await next() - if (decision.kind === 'block') return decision + if (decision.kind === 'block') { + const attempt = state.attempt + if (attempt !== undefined && sameRound(source, attempt)) state.attempt = undefined + const goal = currentGoal(state) + if (goal !== undefined && goal.id === source.goalId && goal.revision === source.revision + && goal.phase === 'active' && goal.activation === 'armed') { + ctx.goals.block(agent, goalRef(goal), { + code: 'prompt-rejected', + message: decision.reason, + }) + } + return decision + } try { valid = validReservation(state, content, source) } catch (error: unknown) { @@ -421,7 +428,11 @@ export function apply(ctx: Context): void { } if (!valid) { const attempt = state.attempt - if (attempt !== undefined && sameRound(source, attempt)) attempt.stale = true + if (attempt !== undefined && sameRound(source, attempt)) { + attempt.stale = true + state.attempt = undefined + } + requestDrive(state) return { kind: 'block', reason: STALE_ROUND_REASON } } return decision diff --git a/packages/goal/goal-session/tests/goal-session.spec.ts b/packages/goal/goal-session/tests/goal-session.spec.ts index dc3d6723a7..ebfca96a37 100644 --- a/packages/goal/goal-session/tests/goal-session.spec.ts +++ b/packages/goal/goal-session/tests/goal-session.spec.ts @@ -266,8 +266,7 @@ describe('same-session goal driving', () => { expect(goal?.roundsStarted).toBe(0) expect(goal?.blockedReason).toEqual({ code: 'prompt-rejected', message: 'deployment policy' }) expect(test.adapter.requests).toHaveLength(0) - expect(test.agent.session.events.some(event => event.type === 'prompt/blocked' - && event.data.reason === 'deployment policy')).toBe(true) + expect(test.agent.session.events.some(event => event.type === 'turn/start')).toBe(false) }) it('does not reserve again when a stopped-goal observer queues ordinary work', async () => { @@ -391,9 +390,6 @@ describe('same-session goal driving', () => { const goal = await waitForGoal(test.ctx, test.agent, current => current?.phase === 'blocked') expect(goal).toMatchObject({ revision: 3, objective: 'new objective', roundsStarted: 1 }) - const blocked = test.agent.session.events.find(event => event.type === 'prompt/blocked') - expect(blocked?.type === 'prompt/blocked' ? blocked.data.reason : undefined) - .toBe('stale goal-round reservation') const admitted = test.agent.session.events.find(event => event.type === 'user/message' && event.data.source.kind === 'goal' && event.data.source.round > 0) expect(admitted?.type === 'user/message' && admitted.data.source.kind === 'goal' @@ -419,8 +415,6 @@ describe('same-session goal driving', () => { expect(goal).toMatchObject({ objective: 'edited downstream', roundsStarted: 1 }) expect(test.adapter.requests).toHaveLength(1) - expect(test.agent.session.events.some(event => event.type === 'prompt/blocked' - && event.data.reason === 'stale goal-round reservation')).toBe(true) }) it('disarms without dispatch when a durability checkpoint fails', async () => { @@ -447,38 +441,6 @@ describe('same-session goal driving', () => { expect(test.adapter.requests).toHaveLength(0) }) - it('disarms an admitted round when a later injection hides its failed closing checkpoint', async () => { - const test = await harness([textResponse('not durable')]) - let injected = false - test.ctx.on('session/flush', (session) => { - const lastStart = session.events.findLast(event => event.type === 'turn/start') - if (lastStart?.type === 'turn/start' && lastStart.data.trigger.kind === 'message' - && lastStart.data.trigger.source.kind === 'goal' && !injected) { - injected = true - test.agent.inject([{ type: 'text', text: 'concurrent completion notice' }], { - source: { kind: 'plugin', plugin: 'test' }, - }) - return Promise.reject(new Error('round flush failed')) - } - }) - test.ctx.goals.create(test.agent, { objective: 'checkpoint the result' }) - - const goal = await waitForGoal( - test.ctx, - test.agent, - current => current?.roundsStarted === 1 && current.activation === 'disarmed', - ) - - expect(goal?.phase).toBe('active') - expect(test.adapter.requests).toHaveLength(1) - const turns = test.agent.session.events.filter(event => event.type === 'turn/start') - const goalTurn = turns.findIndex(event => event.data.trigger.kind === 'message' - && event.data.trigger.source.kind === 'goal') - const injectedTurn = turns.findIndex(event => event.data.trigger.kind === 'injection' - && event.data.trigger.source.kind === 'plugin') - expect(injectedTurn).toBeGreaterThan(goalTurn) - }) - it('blocks the goal when a custom agent rejects the otherwise valid follow-up', async () => { const test = await harness([]) // Reject only the goal-sourced round follow-up, not the state-change injection @@ -520,25 +482,6 @@ describe('same-session goal driving', () => { expect(test.adapter.requests).toHaveLength(0) }) - it('contains a driver read failure and removes continuation authority', async () => { - const test = await harness([]) - let flushes = 0 - test.ctx.on('session/flush', () => { - flushes += 1 - if (flushes !== 2) return - vi.spyOn(test.ctx.goals, 'get').mockImplementationOnce(() => { - throw new Error('corrupt projection') - }) - }) - test.ctx.goals.create(test.agent, { objective: 'fail the driver closed' }) - await new Promise((resolve) => { setImmediate(resolve) }) - - const goal = test.ctx.goals.get(test.agent) - - expect(goal?.phase).toBe('active') - expect(test.adapter.requests).toHaveLength(0) - }) - it('contains synchronous scheduler startup failure', async () => { const test = await harness([]) vi.spyOn(test.ctx.agents, 'withoutInitiator').mockImplementationOnce(() => { @@ -583,8 +526,6 @@ describe('same-session goal driving', () => { await waitForGoal(test.ctx, test.agent, goal => goal?.phase === 'blocked') expect(test.adapter.requests).toHaveLength(1) - expect(test.agent.session.events.some(event => event.type === 'prompt/blocked' - && event.data.reason === 'stale goal-round reservation')).toBe(true) }) it('fails a post-hook read closed before the prompt can enter history', async () => { @@ -615,8 +556,7 @@ describe('same-session goal driving', () => { await test.agent.whenIdle() expect(test.adapter.requests).toHaveLength(0) - expect(test.agent.session.events.some(event => event.type === 'prompt/blocked' - && event.data.reason === 'stale goal-round reservation')).toBe(true) + expect(test.agent.session.events.some(event => event.type === 'turn/start')).toBe(false) }) it('does not invent goal state when ordinary queued work is cancelled', async () => { @@ -715,9 +655,9 @@ describe('same-session goal driving', () => { expect(test.ctx.goals.get(test.agent)).toMatchObject({ phase: 'active', activation: 'disarmed', - roundsStarted: 1, + roundsStarted: 0, }) - expect(test.adapter.requests).toHaveLength(1) + expect(test.adapter.requests).toHaveLength(0) }) it('resets process-local scheduling state at a session-start edge', async () => { diff --git a/packages/goal/goal/tests/goal.spec.ts b/packages/goal/goal/tests/goal.spec.ts index ff902a183a..38419df8d3 100644 --- a/packages/goal/goal/tests/goal.spec.ts +++ b/packages/goal/goal/tests/goal.spec.ts @@ -72,6 +72,7 @@ function stubAgentForSession(session: Session): StubAgent { return AgentMessageId('stub') }, cancel() {}, + retry() {}, whenIdle() { return Promise.resolve() }, } return { @@ -276,11 +277,6 @@ describe('GoalService creation and replay', () => { })) }) - it('rejects a disposed live object even before registry teardown', async () => { - const test = await harness() - test.setStatus('disposed') - expect(() => test.ctx.goals.get(test.agent)).toThrow(expect.objectContaining({ code: 'GOAL_AGENT_NOT_LIVE' })) - }) }) describe('GoalService mutations', () => { diff --git a/packages/goal/tool-goal/tests/tool-goal.spec.ts b/packages/goal/tool-goal/tests/tool-goal.spec.ts index 5834b75dd7..4f7d7d35bb 100644 --- a/packages/goal/tool-goal/tests/tool-goal.spec.ts +++ b/packages/goal/tool-goal/tests/tool-goal.spec.ts @@ -43,6 +43,7 @@ function stubAgent(rawId: string, supplied?: Session): StubAgent { return AgentMessageId('stub') }, cancel() {}, + retry() {}, whenIdle() { return Promise.resolve() }, } return { agent, session, setStatus(value) { status = value } } @@ -336,7 +337,6 @@ describe('goal tool state transitions', () => { goal_id: goal['id'], revision: goal['revision'], action: 'resume', }, root.agent)) expect(goal).toMatchObject({ phase: 'active', revision: 4 }) - expect(await agentEvents(ctx, root.agent).serial('agent/turn-stop', 1, testToolSignal)).toBeUndefined() }) it('terminal-stops an autonomous completion but leaves a human pause interactive', async () => { @@ -347,21 +347,20 @@ describe('goal tool state transitions', () => { goal_id: created.id, revision: created.revision, action: 'pause', }, root.agent) expect(resultGoal(paused)).toMatchObject({ phase: 'paused' }) - expect(await agentEvents(ctx, root.agent).serial('agent/turn-stop', humanTurn, testToolSignal)).toBeUndefined() + expect(paused.concludesTurn).toBeUndefined() const resumed = resultGoal(await execute(ctx, 'update_goal', { goal_id: created.id, revision: 2, action: 'resume', }, root.agent)) closeTurn(root, humanTurn) - const roundTurn = openTurn(root, { + openTurn(root, { kind: 'goal', goalId: created.id, revision: resumed['revision'] as number, round: 1, }) const complete = await execute(ctx, 'update_goal', { goal_id: created.id, revision: resumed['revision'], action: 'complete', }, root.agent) expect(resultGoal(complete)).toMatchObject({ phase: 'complete' }) - expect(await agentEvents(ctx, root.agent).serial('agent/turn-stop', roundTurn, testToolSignal)).toEqual({ action: 'stop' }) - expect(await agentEvents(ctx, root.agent).serial('agent/turn-stop', roundTurn, testToolSignal)).toBeUndefined() + expect(complete.concludesTurn).toBe(true) }) it('rearms a restored active goal only after a new direct human prompt', async () => { diff --git a/packages/hooks/hooks-claude/src/index.ts b/packages/hooks/hooks-claude/src/index.ts index c5a8b26a35..05204ab161 100644 --- a/packages/hooks/hooks-claude/src/index.ts +++ b/packages/hooks/hooks-claude/src/index.ts @@ -12,7 +12,7 @@ import { readFileSync } from 'node:fs' import type { Context } from 'cordis' import z from 'schemastery' -import type { AdditionalContext, Agent, ContinuationDecision, PromptDecision } from '@deepseek-ai/dsh-agent' +import type { AdditionalContext, Agent, PromptDecision } from '@deepseek-ai/dsh-agent' import type { ContentBlock, MessageSource } from '@deepseek-ai/dsh-llm' import type {} from '@deepseek-ai/dsh-session-persistence' import type { PostToolDecision, PreToolDecision, ToolExecution, ToolExecutionResult } from '@deepseek-ai/dsh-tools' @@ -258,16 +258,16 @@ export function apply(ctx: Context, config: Config): void { } }) - // A blocking Stop hook forces continuation with its reason. + // A blocking Stop hook steers at the stopping boundary, which makes the + // machine observe pending input and run another step. // TODO(stop-loop-guard): cap consecutive forced continuations; hooks must self-limit meanwhile. - ctx.on('agent/turn-continuation', async (agent, turn, _default, signal, next): Promise => { + ctx.on('agent/stopping', async (agent, turn, signal): Promise => { const merged = await runPoint('Stop', '', stopPayload(ctx, agent), { agent, turn, signal }) if (merged.decision === 'deny') { // A blocking Stop hook forces continuation. const text = merged.reason ?? 'continue: blocked by Stop hook' - return { action: 'continue', reason: { content: [{ type: 'text', text }], source: PLUGIN_SOURCE } } + agent.steer([{ type: 'text', text }], { source: PLUGIN_SOURCE }) } - return next() }) // SubagentStart may inject child context; SubagentStop only observes. Both diff --git a/packages/hooks/hooks-claude/tests/bridge.spec.ts b/packages/hooks/hooks-claude/tests/bridge.spec.ts index f5fd5702d5..5c0b784cc7 100644 --- a/packages/hooks/hooks-claude/tests/bridge.spec.ts +++ b/packages/hooks/hooks-claude/tests/bridge.spec.ts @@ -58,12 +58,8 @@ async function harnessWithFiber(configDir: string, adapter: MockAdapter): Promis return { ctx, hooks } } -function waitForIdle(ctx: Context, agent: Agent): Promise { - return new Promise((resolve) => { - const dispose = ctx.on('agent/status', (subject, status) => { - if (subject === agent && status === 'idle') { dispose(); resolve() } - }) - }) +function waitForIdle(_ctx: Context, agent: Agent): Promise { + return agent.whenIdle() } function events(agent: Agent): SessionEvent[] { @@ -85,7 +81,7 @@ async function waitFor(predicate: () => boolean, timeout = 5000, interval = 10): } describe('hooks-claude bridge — UserPromptSubmit', () => { - it('a UserPromptSubmit hook that exits 2 blocks the prompt (rejected turn)', async () => { + it('a UserPromptSubmit hook that exits 2 rejects admission without a turn', async () => { // The UserPromptSubmit hook exits 2 (blocking) with a reason on stderr. const dir = mkdtempSync(join(tmpdir(), 'dsh-hooks-claude-')) dirs.push(dir) @@ -100,10 +96,9 @@ describe('hooks-claude bridge — UserPromptSubmit', () => { agent.followup([{ type: 'text', text: 'do something' }]) await waitForIdle(ctx, agent) - // The prompt was blocked: model never called, turn ended rejected. + // The prompt was blocked before the model and before a turn opened. expect(adapter.requests).toHaveLength(0) - const turnEnd = events(agent).findLast(e => e.type === 'turn/end') - expect(turnEnd?.type === 'turn/end' && turnEnd.data.reason.kind).toBe('rejected') + expect(events(agent).some(e => e.type === 'turn/start')).toBe(false) // The hook ran and was recorded. expect(events(agent).some(e => e.type === 'hook/invoked' && e.data.point === 'UserPromptSubmit')).toBe(true) expect(events(agent).some(e => e.type === 'hook/result' && e.data.decision === 'block')).toBe(true) diff --git a/packages/hooks/hooks-claude/tests/coverage-cases.ts b/packages/hooks/hooks-claude/tests/coverage-cases.ts index b3f76fe60c..d2da3f6d52 100644 --- a/packages/hooks/hooks-claude/tests/coverage-cases.ts +++ b/packages/hooks/hooks-claude/tests/coverage-cases.ts @@ -46,8 +46,8 @@ async function harness(configPath: string, adapter: MockAdapter, opts: HarnessOp ctx.llm.registerAdapter(['mock'], adapter) return ctx } -function waitForIdle(ctx: Context, agent: Agent): Promise { - return new Promise((resolve) => { const d = ctx.on('agent/status', (s, st) => { if (s === agent && st === 'idle') { d(); resolve() } }) }) +function waitForIdle(_ctx: Context, agent: Agent): Promise { + return agent.whenIdle() } function events(agent: Agent): SessionEvent[] { return [...agent.session.events] } /** Poll until `predicate` holds or the deadline passes — robust to detached @@ -316,8 +316,7 @@ export function defineCoverageCases(group: CoverageGroup): void { const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) agent.followup([{ type: 'text', text: 'go' }]) await waitForIdle(ctx, agent) - const turnEnd = events(agent).findLast(e => e.type === 'turn/end') - expect(turnEnd?.type === 'turn/end' && turnEnd.data.reason.kind === 'rejected' && turnEnd.data.reason.reason).toContain('blocked by UserPromptSubmit hook') + expect(events(agent).some(e => e.type === 'turn/start')).toBe(false) }) it('a PreToolUse ask with NO reason omits the reason (false arm)', async () => { @@ -497,8 +496,7 @@ export function defineCoverageCases(group: CoverageGroup): void { // recorded, and the (sole, fully-blocked) prompt closed the turn `rejected` expect(adapter.requests).toHaveLength(0) expect(events(agent).some(e => e.type === 'user/message' && e.data.source.kind !== 'user')).toBe(false) - const turnEnd = events(agent).findLast(e => e.type === 'turn/end') - expect(turnEnd?.type === 'turn/end' && turnEnd.data.reason).toMatchObject({ kind: 'rejected', reason: 'policy veto' }) + expect(events(agent).some(e => e.type === 'turn/start')).toBe(false) }) it('preserves separate bridge and downstream prompt contexts with framing and metadata', async () => { diff --git a/packages/hooks/hooks-codex/src/index.ts b/packages/hooks/hooks-codex/src/index.ts index e350c2ad23..4f0768806c 100644 --- a/packages/hooks/hooks-codex/src/index.ts +++ b/packages/hooks/hooks-codex/src/index.ts @@ -15,7 +15,7 @@ import { readFileSync } from 'node:fs' import type { Context } from 'cordis' import z from 'schemastery' -import type { AdditionalContext, Agent, ContinuationDecision, PromptDecision } from '@deepseek-ai/dsh-agent' +import type { AdditionalContext, Agent, PromptDecision } from '@deepseek-ai/dsh-agent' import type { ContentBlock, MessageSource } from '@deepseek-ai/dsh-llm' import type {} from '@deepseek-ai/dsh-session-persistence' import type { PostToolDecision, PreToolDecision, ToolExecution, ToolExecutionResult } from '@deepseek-ai/dsh-tools' @@ -236,11 +236,12 @@ export function apply(ctx: Context, config: Config): void { } }) - // Stop → ContinuationDecision. A blocking Stop hook forces continuation. + // A blocking Stop hook steers at the stopping boundary, which makes the + // machine observe pending input and run another step. // TODO(stop-loop-guard): Codex supplies `stop_hook_active` so a Stop hook can // avoid continuing the same turn indefinitely. It is always false here, so an // unconditionally blocking hook force-continues every step until it self-limits. - ctx.on('agent/turn-continuation', async (agent, turn, _default, signal, next): Promise => { + ctx.on('agent/stopping', async (agent, turn, signal): Promise => { const merged = await runPoint('Stop', '', { ...turnBase(ctx, agent, 'Stop', model), stop_hook_active: false, last_assistant_message: null }, { agent, turn, signal }) /* jscpd:ignore-end */ if (merged.decision === 'deny') { @@ -248,9 +249,8 @@ export function apply(ctx: Context, config: Config): void { // empty stderr) still forces it — fall back to a generic steering line // rather than letting the turn stop. const text = merged.reason ?? 'continue: blocked by Stop hook' - return { action: 'continue', reason: { content: [{ type: 'text', text }], source: PLUGIN_SOURCE } } + agent.steer([{ type: 'text', text }], { source: PLUGIN_SOURCE }) } - return next() }) } diff --git a/packages/hooks/hooks-codex/tests/bridge.spec.ts b/packages/hooks/hooks-codex/tests/bridge.spec.ts index 923a4bf8b5..64ec92af87 100644 --- a/packages/hooks/hooks-codex/tests/bridge.spec.ts +++ b/packages/hooks/hooks-codex/tests/bridge.spec.ts @@ -47,12 +47,8 @@ async function harness(dir: string, adapter: MockAdapter): Promise { return ctx } -function waitForIdle(ctx: Context, agent: Agent): Promise { - return new Promise((resolve) => { - const dispose = ctx.on('agent/status', (subject, status) => { - if (subject === agent && status === 'idle') { dispose(); resolve() } - }) - }) +function waitForIdle(_ctx: Context, agent: Agent): Promise { + return agent.whenIdle() } function events(agent: Agent): SessionEvent[] { return [...agent.session.events] } @@ -125,9 +121,7 @@ describe('hooks-codex bridge', () => { expect(() => process.kill(pid, 0)).toThrow() expect(adapter.requests).toHaveLength(0) - expect(events(agent).findLast(event => event.type === 'turn/end')).toMatchObject({ - data: { reason: { kind: 'aborted' } }, - }) + expect(events(agent).some(event => event.type === 'turn/start')).toBe(false) expect(events(agent).some(event => event.type === 'hook/result' && event.data.point === 'UserPromptSubmit')).toBe(true) }) diff --git a/packages/hooks/hooks-codex/tests/coverage-cases.ts b/packages/hooks/hooks-codex/tests/coverage-cases.ts index 11406b6d41..e75213bfd6 100644 --- a/packages/hooks/hooks-codex/tests/coverage-cases.ts +++ b/packages/hooks/hooks-codex/tests/coverage-cases.ts @@ -36,8 +36,8 @@ async function harness(configPath: string, adapter: MockAdapter, opts: HarnessOp ctx.llm.registerAdapter(['mock'], adapter) return ctx } -function waitForIdle(ctx: Context, agent: Agent): Promise { - return new Promise((resolve) => { const d = ctx.on('agent/status', (s, st) => { if (s === agent && st === 'idle') { d(); resolve() } }) }) +function waitForIdle(_ctx: Context, agent: Agent): Promise { + return agent.whenIdle() } function events(agent: Agent): SessionEvent[] { return [...agent.session.events] } /** Poll until `predicate` holds or the deadline passes — robust to detached @@ -78,7 +78,7 @@ export function defineCoverageCases(groups: CoverageGroup | readonly CoverageGro expect((await capture()).payload.transcript_path).toBeNull() }, 15_000) // Two real agent/hook subprocess loops need process startup and teardown headroom. - it('UserPromptSubmit block (exit 2) → rejected turn; default reason on empty stderr', async () => { + it('UserPromptSubmit block (exit 2) rejects admission without a turn', async () => { const d = dir() hooks(d, { UserPromptSubmit: [{ hooks: [{ type: 'command', command: sh(d, 'b.sh', '#!/usr/bin/env bash\nexit 2\n') }] }] }) const adapter = new MockAdapter([textResponse('no')]) @@ -86,8 +86,7 @@ export function defineCoverageCases(groups: CoverageGroup | readonly CoverageGro const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' }) agent.followup([{ type: 'text', text: 'go' }]); await waitForIdle(ctx, agent) expect(adapter.requests).toHaveLength(0) - const te = events(agent).findLast(e => e.type === 'turn/end') - expect(te?.type === 'turn/end' && te.data.reason.kind).toBe('rejected') + expect(events(agent).some(e => e.type === 'turn/start')).toBe(false) }) it('UserPromptSubmit additionalContext is injected; a no-op hook proceeds', async () => { @@ -112,8 +111,7 @@ export function defineCoverageCases(groups: CoverageGroup | readonly CoverageGro agent.followup([{ type: 'text', text: 'go' }]); await waitForIdle(ctx, agent) expect(adapter.requests).toHaveLength(0) expect(events(agent).some(e => e.type === 'user/message')).toBe(false) - const te = events(agent).findLast(e => e.type === 'turn/end') - expect(te?.type === 'turn/end' && te.data.reason).toMatchObject({ kind: 'rejected', reason: 'policy veto' }) + expect(events(agent).some(e => e.type === 'turn/start')).toBe(false) }) it('preserves separate bridge and downstream prompt contexts with framing and metadata', async () => { diff --git a/packages/host/runtime/tests/host-runtime.spec.ts b/packages/host/runtime/tests/host-runtime.spec.ts index d5472bad1f..bc6fd44d8c 100644 --- a/packages/host/runtime/tests/host-runtime.spec.ts +++ b/packages/host/runtime/tests/host-runtime.spec.ts @@ -381,15 +381,6 @@ describe('sessions.prompt / cancel', () => { if (!response.result.ok) expect(response.result.error.code).toBe('session-not-found') }) - it('maps a synchronous send throw to agent-busy', async () => { - const { api } = await boot() - const { sessionId } = expectOk(await api.sessions.create(request({}))) - const poisoned = [{ type: 'text', text: 'x', bad: () => 1 }] as never - const response = await api.sessions.prompt(request({ sessionId, mode: 'queue' as const, content: poisoned })) - expect(response.result.ok).toBe(false) - if (!response.result.ok) expect(response.result.error.code).toBe('agent-busy') - }) - it('cancels an attached agent and rejects an unattached one', async () => { const running = await boot(['hang']) const { api, ctx } = running @@ -469,7 +460,10 @@ describe('sessions.history', () => { const all = expectOk(await api.sessions.history(request({ sessionId }))) expect(all.hasMore).toBe(false) - const messageCount = all.events.filter(entry => entry.event.type === 'user/message' || entry.event.type === 'assistant/message').length + const messageCount = all.events.filter(entry => + entry.event.type === 'assistant/message' + || (entry.event.type === 'user/message' && entry.event.data.source.kind === 'user'), + ).length expect(messageCount).toBe(6) const lastPage = expectOk(await api.sessions.history(request({ sessionId, maxMessages: 1 }))) diff --git a/packages/llm/llm-retry/README.md b/packages/llm/llm-retry/README.md index 699e7e3dad..3d4000697b 100644 --- a/packages/llm/llm-retry/README.md +++ b/packages/llm/llm-retry/README.md @@ -1,12 +1,12 @@ # `@deepseek-ai/dsh-llm-retry` -Function plugin that retries selected transient model-request failures on the agent loop's closed-step recovery seam. It does not wrap `ctx.llm.stream()`: every adapter call remains one provider attempt, and every retry opens a fresh numbered step. +Function plugin that retries selected transient model-request failures through the `agent/request-error` waterfall. It does not wrap `ctx.llm.stream()`: every adapter call remains one provider attempt, and every retry opens a fresh numbered turn. The default policy permits two retries for `RATE_LIMIT`, `SERVER`, `TIMEOUT`, and `TRANSPORT`, using bounded exponential backoff from 500 ms to 10 seconds with 10 percent jitter. Delay bounds must fit Node's supported timer range. A valid `providerRetryAfterMs` replaces local backoff when it is within the configured cap; an over-cap instruction delegates to the next recovery policy instead. -Before waiting, the plugin appends a non-surface `llm/retry` event with the failure and scheduled delay. Cancellation and plugin disposal abort the wait; disposal drains the plugin's active backoffs, and a callback captured before disposal fails closed if invoked afterward. +The recovery listener appends a non-surface `llm/retry` event after the failed step, waits for the backoff while the failed turn's signal remains live, then calls `agent.retry()`. The loop closes that failed turn and opens a retry turn over the same durable history. The policy keeps its own retry count across that uninterrupted recovery chain and clears it at terminal `agent/idle`. Turn cancellation and plugin disposal abort the wait. -The separately published `./invariant` companion checks that every retry record names the current open turn and its latest closed step, has a unique step record and increasing retry number, and carries a positive bounded retry budget and non-negative bounded timer delay. Full jitter may schedule zero milliseconds at its lower boundary. +The separately published `./invariant` companion checks that every retry record appears inside an open turn after its failed step, matches its position in the current retry chain, and carries a positive bounded retry budget and non-negative bounded timer delay. Full jitter may schedule zero milliseconds at its lower boundary. ```yaml - name: '@deepseek-ai/dsh-llm-retry' @@ -24,7 +24,7 @@ The separately published `./invariant` companion checks that every retry record #### What the model sees -No retry event, delay, or failure prose is model-visible. After a retry, the next numbered step reconstructs the same explicit provider/model request from durable session history; failed chunks never enter derived messages. +No retry event, delay, or failure prose is model-visible. The retry turn reconstructs the same explicit provider/model request from durable session history; failed chunks never enter derived messages. #### Token effect @@ -36,6 +36,6 @@ The reconstructed request preserves the prior prefix and is eligible for provide ## Known Limitations and Deferred Work -- **Agent steps are the only retry boundary** — direct `ctx.llm.stream()` consumers remain single-attempt because a raw stream cannot separate already-emitted chunks durably. +- **Agent turns are the only retry boundary** — direct `ctx.llm.stream()` consumers remain single-attempt because a raw stream cannot separate already-emitted chunks durably. - **Finite plugin budgets add** — this policy counts only configured transient codes; context-overflow compaction counts only its own code. A future policy with overlapping codes must document and test registration-order behavior. -- **`llm/retry` records scheduling, not completion** — later step and turn events establish success, exhaustion, or cancellation. +- **`llm/retry` records completed backoff, not request completion** — later step and turn events establish success, exhaustion, or cancellation. diff --git a/packages/llm/llm-retry/src/index.ts b/packages/llm/llm-retry/src/index.ts index 4edf22d6f2..040575e710 100644 --- a/packages/llm/llm-retry/src/index.ts +++ b/packages/llm/llm-retry/src/index.ts @@ -1,13 +1,13 @@ /** - * Bounded transient model-request retry policy on the agent loop's closed-step - * recovery seam. Each scheduled retry is durable before its cancellable wait. + * Bounded transient model-request retry policy on the agent request-recovery + * seam. Each scheduled retry is durable before its cancellable wait. * * @module @deepseek-ai/dsh-llm-retry */ import type { Context } from 'cordis' import z from 'schemastery' -import type { Agent, RequestError, RequestErrorDecision } from '@deepseek-ai/dsh-agent' +import type { Agent, RequestError } from '@deepseek-ai/dsh-agent' import type { LlmFailure } from '@deepseek-ai/dsh-llm' import type {} from '@deepseek-ai/dsh-session' import { MAX_TIMER_DELAY_MS } from '@deepseek-ai/dsh-timeout' @@ -145,7 +145,8 @@ export function apply(ctx: Context, config: Config = {}, internals: RetryInterna const resolved = resolveConfig(config) const random = internals.random ?? Math.random const lifetime = new AbortController() - const active = new Set>() + const active = new Set>() + const retries = new WeakMap() async function backoff( agent: Agent, @@ -155,9 +156,9 @@ export function apply(ctx: Context, config: Config = {}, internals: RetryInterna retry: number, delayMs: number, signal: AbortSignal, - ): Promise { + ): Promise { const fusedSignal = AbortSignal.any([signal, lifetime.signal]) - if (fusedSignal.aborted) return { action: 'fail' } + if (fusedSignal.aborted) return agent.session.append('llm/retry', { turn, step, @@ -166,29 +167,33 @@ export function apply(ctx: Context, config: Config = {}, internals: RetryInterna delayMs, failure, }) - if (!await cancellableDelay(delayMs, fusedSignal)) return { action: 'fail' } - return { action: 'retry' } + retries.set(agent, retry) + if (!await cancellableDelay(delayMs, fusedSignal)) return + agent.retry() } + ctx.on('agent/idle', (agent) => { + retries.delete(agent) + }) + const disposeListener = ctx.on('agent/request-error', ( agent: Agent, turn: number, step: number, _error: RequestError, failure: LlmFailure, - priorFailures: readonly LlmFailure[], signal: AbortSignal, - next: () => Promise, + next: () => Promise, ) => { // A waterfall may have captured this callback before its registration was // removed. Lifetime cancellation must prevent that stale callback from // entering a downstream policy after disposal. - if (lifetime.signal.aborted) return Promise.resolve({ action: 'fail' }) + if (lifetime.signal.aborted) return Promise.resolve() if (!resolved.retryableCodes.has(failure.code)) return next() - const priorTransientFailures = priorFailures.filter(item => resolved.retryableCodes.has(item.code)).length - if (priorTransientFailures >= resolved.maxTransientRetries) return next() + const priorRetries = retries.get(agent) ?? 0 + if (priorRetries >= resolved.maxTransientRetries) return next() - const retry = priorTransientFailures + 1 + const retry = priorRetries + 1 let delayMs: number if (failure.providerRetryAfterMs !== undefined && Number.isFinite(failure.providerRetryAfterMs) diff --git a/packages/llm/llm-retry/src/invariant.ts b/packages/llm/llm-retry/src/invariant.ts index 784f459606..be5ce3b046 100644 --- a/packages/llm/llm-retry/src/invariant.ts +++ b/packages/llm/llm-retry/src/invariant.ts @@ -13,6 +13,34 @@ export const name = 'llm-retry-invariant' /** Service required before the companion can reserve package ownership. */ export const inject = ['invariants'] +/** Find the first turn in the structured-failure retry chain containing `turn`. */ +function retryChainStart(history: readonly SessionEvent[], turn: number): number { + let startIndex = history.findLastIndex( + event => event.type === 'turn/start' && event.data.turn === turn, + ) + while (startIndex >= 0) { + const start = history[startIndex] + if (start?.type !== 'turn/start' || start.data.trigger.kind !== 'retry') break + + let endIndex = startIndex - 1 + while (endIndex >= 0 && history[endIndex]?.type !== 'turn/end') endIndex -= 1 + const end = history[endIndex] + if (end?.type !== 'turn/end' + || end.data.reason.kind !== 'error' + || end.data.reason.failure === undefined) break + + const previousStart = history.findLastIndex( + (event, index) => + index < endIndex + && event.type === 'turn/start' + && event.data.turn === end.data.turn, + ) + if (previousStart < 0) break + startIndex = previousStart + } + return startIndex +} + /** Validate one retry record against the open turn and most recently closed step. */ function validateRetry( history: readonly SessionEvent[], @@ -59,14 +87,15 @@ function validateRetry( fail(`llm/retry names step ${step}, but the latest closed step is ${String(closedStep)}`) } - const priorRetries = currentTurnEvents + const chainStart = retryChainStart(history, turn) + const chainRetries = history.slice(Math.max(chainStart, 0)) .filter((prior): prior is SessionEvent<'llm/retry'> => prior.type === 'llm/retry') - if (priorRetries.some(prior => prior.data.step === step)) { + if (chainRetries.some(prior => prior.data.turn === turn && prior.data.step === step)) { fail(`llm/retry duplicates the retry record for turn ${turn}/step ${step}`) } - const priorRetry = priorRetries[0] - if (priorRetry !== undefined && retry <= priorRetry.data.retry) { - fail(`llm/retry retry ${retry} must increase after retry ${priorRetry.data.retry}`) + const expectedRetry = chainRetries.length + 1 + if (retry !== expectedRetry) { + fail(`llm/retry retry ${retry} must equal retry-chain position ${expectedRetry}`) } } diff --git a/packages/llm/llm-retry/tests/invariant.spec.ts b/packages/llm/llm-retry/tests/invariant.spec.ts index 7f9bc6b061..f04f8f36a2 100644 --- a/packages/llm/llm-retry/tests/invariant.spec.ts +++ b/packages/llm/llm-retry/tests/invariant.spec.ts @@ -13,32 +13,35 @@ async function setup(): Promise { return ctx } +const failure = { message: 'provider busy', code: 'RATE_LIMIT', status: 429 } + function closeStep(ctx: Context, id: string, turn = 1, step = 1) { const session = ctx.sessions.create(SessionId(id)) - session.append('turn/start', { turn, trigger: { kind: 'message', source: { kind: 'user' } } }) + session.append('turn/start', { + turn, + trigger: turn === 1 + ? { kind: 'message', source: { kind: 'user' } } + : { kind: 'retry' }, + }) session.append('step/start', { turn, step }) session.append('step/end', { turn, step }) return session } -const failure = { message: 'provider busy', code: 'RATE_LIMIT', status: 429 } - describe('llm-retry invariants', () => { - it('accepts increasing retry records for successive closed steps and ignores unrelated events', async () => { + it('accepts increasing retry schedules for successive failed turns', async () => { const ctx = await setup() const session = closeStep(ctx, 'retry-invariant-valid') expect(() => { session.append('llm/retry', { turn: 1, step: 1, retry: 1, maxRetries: 2, delayMs: 500, failure, }) - session.append('step/start', { turn: 1, step: 2 }) - session.append('step/end', { turn: 1, step: 2 }) + session.append('turn/end', { turn: 1, reason: { kind: 'error', step: 1, failure } }) + session.append('turn/start', { turn: 2, trigger: { kind: 'retry' } }) + session.append('step/start', { turn: 2, step: 1 }) + session.append('step/end', { turn: 2, step: 1 }) session.append('llm/retry', { - turn: 1, step: 2, retry: 2, maxRetries: 2, delayMs: 1_000, failure, - }) - const zeroDelay = closeStep(ctx, 'retry-invariant-zero-delay') - zeroDelay.append('llm/retry', { - turn: 1, step: 1, retry: 1, maxRetries: 1, delayMs: 0, failure, + turn: 2, step: 1, retry: 2, maxRetries: 2, delayMs: 0, failure, }) }).not.toThrow() expect(() => { ctx.emit('tools/change') }).not.toThrow() @@ -60,7 +63,7 @@ describe('llm-retry invariants', () => { }).toThrow(message) }) - it('rejects retry records outside the matching closed-step boundary', async () => { + it('requires an open turn and its latest closed step', async () => { const ctx = await setup() const absent = ctx.sessions.create(SessionId('retry-invariant-no-turn')) expect(() => { @@ -85,31 +88,15 @@ describe('llm-retry invariants', () => { }) }).toThrow(/step 1 is still open/) - const noStep = ctx.sessions.create(SessionId('retry-invariant-no-step')) - noStep.append('turn/start', { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } }) - expect(() => { - noStep.append('llm/retry', { - turn: 1, step: 1, retry: 1, maxRetries: 2, delayMs: 1, failure, - }) - }).toThrow(/latest closed step is undefined/) - const wrongStep = closeStep(ctx, 'retry-invariant-wrong-step') expect(() => { wrongStep.append('llm/retry', { turn: 1, step: 2, retry: 1, maxRetries: 2, delayMs: 1, failure, }) }).toThrow(/latest closed step is 1/) - - const closedTurn = closeStep(ctx, 'retry-invariant-closed-turn') - closedTurn.append('turn/end', { turn: 1, reason: { kind: 'aborted' } }) - expect(() => { - closedTurn.append('llm/retry', { - turn: 1, step: 1, retry: 1, maxRetries: 2, delayMs: 1, failure, - }) - }).toThrow(/inside an open turn/) }) - it('rejects duplicate and non-increasing retry records', async () => { + it('rejects duplicate and out-of-sequence retry schedules', async () => { const ctx = await setup() const duplicate = closeStep(ctx, 'retry-invariant-duplicate') duplicate.append('llm/retry', { @@ -119,26 +106,52 @@ describe('llm-retry invariants', () => { duplicate.append('llm/retry', { turn: 1, step: 1, retry: 2, maxRetries: 3, delayMs: 1, failure, }) - }).toThrow(/duplicates the retry record/) + }).toThrow(/duplicates/) const nonIncreasing = closeStep(ctx, 'retry-invariant-non-increasing') nonIncreasing.append('llm/retry', { turn: 1, step: 1, retry: 1, maxRetries: 3, delayMs: 1, failure, }) - nonIncreasing.append('step/start', { turn: 1, step: 2 }) - nonIncreasing.append('step/end', { turn: 1, step: 2 }) + nonIncreasing.append('turn/end', { turn: 1, reason: { kind: 'error', step: 1, failure } }) + nonIncreasing.append('turn/start', { turn: 2, trigger: { kind: 'retry' } }) + nonIncreasing.append('step/start', { turn: 2, step: 1 }) + nonIncreasing.append('step/end', { turn: 2, step: 1 }) expect(() => { nonIncreasing.append('llm/retry', { - turn: 1, step: 2, retry: 1, maxRetries: 3, delayMs: 1, failure, + turn: 2, step: 1, retry: 1, maxRetries: 3, delayMs: 1, failure, }) - }).toThrow(/must increase/) + }).toThrow(/retry-chain position 2/) + }) + + it('resets retry numbering after a completed chain', async () => { + const ctx = await setup() + const session = closeStep(ctx, 'retry-invariant-reset') + session.append('llm/retry', { + turn: 1, step: 1, retry: 1, maxRetries: 2, delayMs: 1, failure, + }) + session.append('turn/end', { turn: 1, reason: { kind: 'error', step: 1, failure } }) + session.append('turn/start', { turn: 2, trigger: { kind: 'retry' } }) + session.append('step/start', { turn: 2, step: 1 }) + session.append('step/end', { turn: 2, step: 1 }) + session.append('turn/end', { turn: 2, reason: { kind: 'completed' } }) + session.append('turn/start', { + turn: 3, + trigger: { kind: 'message', source: { kind: 'user' } }, + }) + session.append('step/start', { turn: 3, step: 1 }) + session.append('step/end', { turn: 3, step: 1 }) + + expect(() => { + session.append('llm/retry', { + turn: 3, step: 1, retry: 1, maxRetries: 2, delayMs: 1, failure, + }) + }).not.toThrow() }) it('validates existing histories on late registration', async () => { const ctx = new Context() await ctx.plugin(SessionStore) const session = ctx.sessions.create(SessionId('retry-invariant-late')) - session.append('step/end', { turn: 1, step: 1 }) session.append('llm/retry', { turn: 1, step: 1, retry: 1, maxRetries: 2, delayMs: 1, failure, }) diff --git a/packages/llm/llm-retry/tests/loader-composition.spec.ts b/packages/llm/llm-retry/tests/loader-composition.spec.ts index bb08577d74..d57a6d3f88 100644 --- a/packages/llm/llm-retry/tests/loader-composition.spec.ts +++ b/packages/llm/llm-retry/tests/loader-composition.spec.ts @@ -7,7 +7,6 @@ import { Context } from 'cordis' import Loader from '@cordisjs/plugin-loader' import Include from '@cordisjs/plugin-include' import AgentRegistry from '@deepseek-ai/dsh-agent' -import type { Agent } from '@deepseek-ai/dsh-agent' import AgentLoop from '@deepseek-ai/dsh-agent-loop' import LlmService, { LlmAdapter, LlmError } from '@deepseek-ai/dsh-llm' import type { GenerateOptions, StreamChunk } from '@deepseek-ai/dsh-llm' @@ -32,17 +31,6 @@ class TransientOnceAdapter extends LlmAdapter { } } -function waitForIdle(ctx: Context, agent: Agent): Promise { - return new Promise((resolve) => { - const dispose = ctx.on('agent/status', (subject, status) => { - if (subject === agent && status === 'idle') { - dispose() - resolve() - } - }) - }) -} - afterEach(async () => { await context?.fiber.dispose() context = undefined @@ -113,9 +101,9 @@ describe('real Loader composition', () => { const adapter = new TransientOnceAdapter() loaded.llm.registerAdapter(['mock'], adapter) const agent = loaded.agentLoop.create(SessionId('loader-retry'), { provider: 'mock', model: 'mock' }) - const idle = waitForIdle(loaded, agent) agent.followup([{ type: 'text', text: 'recover' }]) - await idle + await expect.poll(() => adapter.requests).toBe(2) + await agent.whenIdle() expect(adapter.requests).toBe(2) expect(agent.session.events.filter(event => event.type === 'llm/retry')).toHaveLength(1) diff --git a/packages/llm/llm-retry/tests/persistence.spec.ts b/packages/llm/llm-retry/tests/persistence.spec.ts index 1668c36d73..5d02192263 100644 --- a/packages/llm/llm-retry/tests/persistence.spec.ts +++ b/packages/llm/llm-retry/tests/persistence.spec.ts @@ -43,7 +43,14 @@ describe.each(['jsonl', 'sqlite'] as const)('%s retry-event persistence', (kind) delayMs: 750, failure: { message: 'provider busy', code: 'RATE_LIMIT', status: 429 }, }) - session.append('turn/end', { turn: 1, reason: { kind: 'aborted' } }) + session.append('turn/end', { + turn: 1, + reason: { + kind: 'error', + step: 1, + failure: { message: 'provider busy', code: 'RATE_LIMIT', status: 429 }, + }, + }) expect(session.deriveMessages()).toEqual([]) await ctx.sessions.flush(session) diff --git a/packages/llm/llm-retry/tests/retry.spec.ts b/packages/llm/llm-retry/tests/retry.spec.ts index 20e7862ebc..40bbd32e72 100644 --- a/packages/llm/llm-retry/tests/retry.spec.ts +++ b/packages/llm/llm-retry/tests/retry.spec.ts @@ -8,9 +8,8 @@ import type { SessionEvent } from '@deepseek-ai/dsh-session' import SystemPrompt from '@deepseek-ai/dsh-system-prompt' import ToolRegistry, { defineContentToolFixture } from '@deepseek-ai/dsh-tools' import AgentRegistry from '@deepseek-ai/dsh-agent' -import type { Agent, RequestErrorDecision } from '@deepseek-ai/dsh-agent' +import type { Agent } from '@deepseek-ai/dsh-agent' import AgentLoop from '@deepseek-ai/dsh-agent-loop' -import { MAX_TIMER_DELAY_MS } from '@deepseek-ai/dsh-timeout' import * as retry from '../src/index.ts' type ScriptEntry = Error | Iterable | AsyncIterable @@ -149,8 +148,8 @@ describe('bounded transient retry policy', () => { await idle expect(adapter.requests).toHaveLength(2) - expect(agent.session.events.filter(item => item.type === 'step/start').map(item => item.data.step)) - .toEqual([1, 2]) + expect(agent.session.events.filter(item => item.type === 'step/start').map(item => item.data)) + .toEqual([{ turn: 1, step: 1 }, { turn: 2, step: 1 }]) expect(agent.session.deriveMessages().at(-1)).toEqual({ role: 'assistant', content: [{ type: 'text', text: 'done' }], @@ -185,11 +184,13 @@ describe('bounded transient retry policy', () => { await idle const failedChunks = agent.session.events.filter(event => - event.type === 'assistant/chunk' && event.data.step === 1, + event.type === 'assistant/chunk' && event.data.turn === 1, ) expect(failedChunks).toHaveLength(6) - expect(agent.session.events.filter(event => event.type === 'assistant/message').map(event => event.data.step)) - .toEqual([2]) + expect(agent.session.events.filter(event => event.type === 'assistant/message').map(event => ({ + turn: event.data.turn, + step: event.data.step, + }))).toEqual([{ turn: 2, step: 1 }]) expect(agent.session.events.some(event => event.type === 'tool/call')).toBe(false) expect(toolExecutions).toBe(0) expect(agent.session.deriveMessages().at(-1)).toMatchObject({ @@ -232,6 +233,36 @@ describe('bounded transient retry policy', () => { }) }) + it('resets the retry budget for a later message', async () => { + vi.useFakeTimers() + const adapter = new ScriptedAdapter([ + new LlmError('first busy', 'SERVER'), + textResponse('first done'), + new LlmError('second busy', 'SERVER'), + textResponse('second done'), + ]) + ;({ ctx: context } = await harness(adapter, { maxTransientRetries: 1 })) + const agent = context.agentLoop.create(SessionId('retry-reset'), { provider: 'mock', model: 'mock' }) + + const firstRetry = waitForRetry(context, agent, 1) + agent.followup([{ type: 'text', text: 'first' }]) + await firstRetry + const firstIdle = waitForIdle(context, agent) + await vi.advanceTimersByTimeAsync(500) + await firstIdle + + const secondRetry = waitForRetry(context, agent, 1) + agent.followup([{ type: 'text', text: 'second' }]) + await secondRetry + const secondIdle = waitForIdle(context, agent) + await vi.advanceTimersByTimeAsync(500) + await secondIdle + + expect(agent.session.events.filter(event => event.type === 'llm/retry').map(event => event.data.retry)) + .toEqual([1, 1]) + expect(adapter.requests).toHaveLength(4) + }) + it('accepts the zero-delay lower jitter bound', async () => { vi.useFakeTimers() const adapter = new ScriptedAdapter([ @@ -310,7 +341,6 @@ describe('bounded transient retry policy', () => { agent.followup([{ type: 'text', text: 'go' }]) await scheduled const idle = waitForIdle(context, agent) - await mounted.retryFiber.dispose() await idle await vi.advanceTimersByTimeAsync(60_000) @@ -320,157 +350,4 @@ describe('bounded transient retry policy', () => { expect(vi.getTimerCount()).toBe(0) }) - it('does not make plugin disposal wait for a delegated recovery policy', async () => { - const adapter = new ScriptedAdapter([new LlmError('bad key', 'AUTH')]) - const mounted = await harness(adapter) - context = mounted.ctx - const downstream = Promise.withResolvers() - const entered = Promise.withResolvers() - context.on('agent/request-error', () => { - entered.resolve(undefined) - return downstream.promise - }) - const agent = context.agentLoop.create(SessionId('retry-delegated-disposal'), { - provider: 'mock', - model: 'mock', - }) - const idle = waitForIdle(context, agent) - agent.followup([{ type: 'text', text: 'go' }]) - await entered.promise - - const disposing = mounted.retryFiber.dispose() - let timer: ReturnType | undefined - const outcome = await Promise.race([ - disposing.then(() => 'disposed' as const), - new Promise<'blocked'>((resolve) => { timer = setTimeout(() => { resolve('blocked') }, 100) }), - ]) - if (timer !== undefined) clearTimeout(timer) - downstream.resolve({ action: 'fail' }) - await disposing - await idle - - expect(outcome).toBe('disposed') - expect(adapter.requests).toHaveLength(1) - }) - - it('fails a captured callback after disposal without entering downstream policy', async () => { - const adapter = new ScriptedAdapter([new LlmError('bad key', 'AUTH')]) - const captured = Promise.withResolvers() - let invokeCaptured: (() => Promise) | undefined - const mounted = await harness(adapter, {}, (ctx) => { - ctx.on('agent/request-error', (_agent, _turn, _step, _error, _failure, _history, _signal, next) => { - return new Promise((resolve) => { - invokeCaptured = async () => { resolve(await next()) } - captured.resolve(undefined) - }) - }) - }) - context = mounted.ctx - let downstreamCalls = 0 - context.on('agent/request-error', async (_agent, _turn, _step, _error, _failure, _history, _signal, next) => { - downstreamCalls += 1 - return next() - }) - const agent = context.agentLoop.create(SessionId('retry-captured-disposal'), { - provider: 'mock', - model: 'mock', - }) - const idle = waitForIdle(context, agent) - agent.followup([{ type: 'text', text: 'go' }]) - await captured.promise - - await mounted.retryFiber.dispose() - if (invokeCaptured === undefined) throw new Error('request-error waterfall did not capture retry callback') - await invokeCaptured() - await idle - - expect(downstreamCalls).toBe(0) - expect(adapter.requests).toHaveLength(1) - }) - - it('lets turn cancellation win during backoff without opening another step', async () => { - vi.useFakeTimers() - const adapter = new ScriptedAdapter([ - new LlmError('temporary', 'TIMEOUT'), - textResponse('must not run'), - ]) - ;({ ctx: context } = await harness(adapter)) - const agent = context.agentLoop.create(SessionId('retry-cancel'), { provider: 'mock', model: 'mock' }) - const scheduled = waitForRetry(context, agent, 1) - agent.followup([{ type: 'text', text: 'go' }]) - await scheduled - const idle = waitForIdle(context, agent) - agent.cancel({ kind: 'user' }) - await idle - - expect(adapter.requests).toHaveLength(1) - expect(agent.session.events.at(-1)).toMatchObject({ - type: 'turn/end', - data: { reason: { kind: 'aborted' } }, - }) - expect(vi.getTimerCount()).toBe(0) - }) - - it('lets an earlier recovery listener cancel before retry policy runs', async () => { - vi.useFakeTimers() - const adapter = new ScriptedAdapter([ - new LlmError('temporary', 'SERVER'), - textResponse('must not run'), - ]) - ;({ ctx: context } = await harness(adapter, {}, (ctx) => { - ctx.on('agent/request-error', async (agent, _turn, _step, _error, _failure, _history, _signal, next) => { - agent.cancel({ kind: 'user' }) - return next() - }) - })) - const agent = context.agentLoop.create(SessionId('retry-pre-cancel'), { provider: 'mock', model: 'mock' }) - const idle = waitForIdle(context, agent) - - agent.followup([{ type: 'text', text: 'go' }]) - await idle - - expect(adapter.requests).toHaveLength(1) - expect(agent.session.events.some(event => event.type === 'llm/retry')).toBe(false) - expect(agent.session.events.at(-1)).toMatchObject({ - type: 'turn/end', - data: { reason: { kind: 'aborted' } }, - }) - }) - - it('handles synchronous cancellation from the retry status event', async () => { - vi.useFakeTimers() - const adapter = new ScriptedAdapter([ - new LlmError('temporary', 'SERVER'), - textResponse('must not run'), - ]) - ;({ ctx: context } = await harness(adapter)) - const agent = context.agentLoop.create(SessionId('retry-event-cancel'), { provider: 'mock', model: 'mock' }) - context.on('session/event', (session, event) => { - if (session === agent.session && event.type === 'llm/retry') agent.cancel({ kind: 'user' }) - }) - const idle = waitForIdle(context, agent) - - agent.followup([{ type: 'text', text: 'go' }]) - await idle - - expect(adapter.requests).toHaveLength(1) - expect(agent.session.events.filter(event => event.type === 'llm/retry')).toHaveLength(1) - expect(vi.getTimerCount()).toBe(0) - }) - - it.each([ - [{ maxTransientRetries: -1 }, /maxTransientRetries/], - [{ maxTransientRetries: 1.5 }, /maxTransientRetries/], - [{ initialDelayMs: 0 }, /initialDelayMs/], - [{ maxDelayMs: Number.POSITIVE_INFINITY }, /maxDelayMs/], - [{ initialDelayMs: MAX_TIMER_DELAY_MS + 1 }, /initialDelayMs/], - [{ maxDelayMs: MAX_TIMER_DELAY_MS + 1 }, /maxDelayMs/], - [{ initialDelayMs: 20, maxDelayMs: 10 }, /less than or equal/], - [{ jitterRatio: 1.1 }, /jitterRatio/], - [{ retryableCodes: [] }, /must not be empty/], - [{ retryableCodes: ['SERVER', 'SERVER'] }, /duplicates/], - [{ retryableCodes: [''] }, /non-empty strings/], - ] as const)('fails direct composition for invalid config %#', (config, message) => { - expect(() => { retry.apply(new Context(), config as retry.Config) }).toThrow(message) - }) }) diff --git a/packages/plan/plan-mode/src/index.ts b/packages/plan/plan-mode/src/index.ts index 26ad904d30..7e051b7d5a 100644 --- a/packages/plan/plan-mode/src/index.ts +++ b/packages/plan/plan-mode/src/index.ts @@ -10,7 +10,7 @@ * wins), so resume and fork restore it without a live mirror. User selections * are held as pending intent until a turn boundary because every session event * is turn-enclosed. The service flushes before the affected request assembly - * on prompt submission, ordinary continuation, and request-recovery retry. + * on prompt submission and each request step (including retry turns). * * The exit tool remains registered while plan mode is inactive so crossing a * boundary changes only the prompt section, not the request tool catalog. @@ -175,27 +175,13 @@ export class PlanModeService extends Service { } ctx.on('agent/prompt-submit', (agent, _content, _source, _signal, next) => flushAfter(agent, next), { prepend: true }) - ctx.on('agent/turn-continuation', (agent, _turn, _decision, _signal, next) => - flushAfter(agent, next), { prepend: true }) - ctx.on('agent/request-error', async ( - agent, - _turn, - _step, - _error, - _failure, - _priorFailures, - _signal, - next, - ) => { - const decision = await next() - // A waterfall can retain this wrapper after Cordis unregisters it. - if (disposed || decision.action !== 'retry') return decision + ctx.on('agent/step', (agent) => { + if (disposed) return try { this.onBoundary(agent) } catch (error) { ctx.logger.warn('dsh-plan-mode: boundary flush failed: %o', error) } - return decision }, { prepend: true }) ctx.effect(() => () => { disposed = true }, 'dsh-plan-mode: close boundary lifetime') diff --git a/packages/plan/plan-mode/tests/integration.spec.ts b/packages/plan/plan-mode/tests/integration.spec.ts index e195ed54b2..10b145c2d6 100644 --- a/packages/plan/plan-mode/tests/integration.spec.ts +++ b/packages/plan/plan-mode/tests/integration.spec.ts @@ -128,7 +128,7 @@ describe('plan mode through the agent loop', () => { expect(second.data.header.system).toContain('plan mode') }) - it('a mode flip during request recovery shapes the retry before its assembly', async () => { + it('a mode flip at error idle shapes the retry before its assembly', async () => { const failedRequest = [{ type: 'finish', reason: { kind: 'error', failure: { message: 'temporarily unavailable', code: 'SERVER', status: 503 } }, @@ -136,20 +136,14 @@ describe('plan mode through the agent loop', () => { const adapter = new MockAdapter([failedRequest, textResponse('Recovered in plan mode.')]) const ctx = await harness(adapter) const agent = ctx.agentLoop.create(SessionId('it-plan-retry-flip'), { provider: 'mock', model: 'mock' }) - const recoveryEntered = Promise.withResolvers() - const releaseRecovery = Promise.withResolvers() - ctx.on('agent/request-error', async (subject, _turn, _step, _error, _failure, _history, _signal, next) => { - if (subject !== agent) return next() - recoveryEntered.resolve(true) - await releaseRecovery.promise - return { action: 'retry' } + ctx.on('agent/idle', (subject, _turn, reason) => { + if (subject !== agent || reason.kind !== 'error') return + ctx.planMode.set(agent, true) + agent.retry() }) const idle = waitForIdle(ctx, agent) agent.followup([{ type: 'text', text: 'plan after the transient failure' }]) - await recoveryEntered.promise - ctx.planMode.set(agent, true) - releaseRecovery.resolve(true) await idle expect(adapter.requests).toHaveLength(2) @@ -158,8 +152,10 @@ describe('plan mode through the agent loop', () => { expect(adapter.requests[1]?.tools).toEqual(adapter.requests[0]?.tools) const log = agent.session.events const planMode = findEvent(log, 'plan/mode') - const firstEnd = log.find(event => event.type === 'step/end' && event.data.step === 1) - const retryStart = log.find(event => event.type === 'step/start' && event.data.step === 2) + const firstEnd = log.find(event => event.type === 'step/end' + && event.data.turn === 1 && event.data.step === 1) + const retryStart = log.find(event => event.type === 'step/start' + && event.data.turn === 2 && event.data.step === 1) expect(firstEnd?.seq).toBeLessThan(planMode.seq) expect(planMode.seq).toBeLessThan(retryStart?.seq ?? 0) expect(findEvent(log, 'request/header', 'last').data.header.system).toContain(PLAN_CONFIG.section) diff --git a/packages/plan/plan-mode/tests/plan-mode.spec.ts b/packages/plan/plan-mode/tests/plan-mode.spec.ts index a7f7743497..c944220ba1 100644 --- a/packages/plan/plan-mode/tests/plan-mode.spec.ts +++ b/packages/plan/plan-mode/tests/plan-mode.spec.ts @@ -4,7 +4,7 @@ import { CallId } from '@deepseek-ai/dsh-llm' import SystemPrompt from '@deepseek-ai/dsh-system-prompt' import ToolRegistry, { RUN_CODE_NAME, defineContentToolFixture } from '@deepseek-ai/dsh-tools' import { Session, SessionId } from '@deepseek-ai/dsh-session' -import { agentEvents, type Agent, type RequestErrorDecision } from '@deepseek-ai/dsh-agent' +import { agentEvents, type Agent } from '@deepseek-ai/dsh-agent' import { createScope } from '@deepseek-ai/dsh-scope' import UserInteractionService, { type AskUserQuestionRequest } from '@deepseek-ai/dsh-user-interaction' import CommandService from '@deepseek-ai/dsh-commands' @@ -19,9 +19,8 @@ const PLAN_CONFIG = { section: TEST_PLAN_SECTION } satisfies PlanModeConfig * Drives the REAL plugin: mounts `dsh-plan-mode` beside real `SystemPrompt` and * `ToolRegistry` services, with fake Agents carrying real `Session`s and a * real scoped `agent.ctx` minted through `createScope`. - * Turn boundaries are simulated by appending the real boundary events and - * dispatching the interception seams the loop fires there. Recovery retries - * exercise the separate `agent/request-error` wrapper. + * Request boundaries are simulated by dispatching the real prompt-admission + * and between-step seams used by the loop. */ async function agentWithSession(ctx: Context, id = 'agent-1', { active }: { active?: boolean } = {}): Promise { @@ -53,39 +52,15 @@ async function setup(config: PlanModeConfig = PLAN_CONFIG): Promise { } /** - * Append a boundary event and dispatch the interception seam the loop fires - * there — `agent/prompt-submit` inside the just-opened turn, - * `agent/turn-continuation` after the step closed. Recovery retries use the - * separately covered `agent/request-error` wrapper; post-commit - * `session/event` observers remain observe-only. + * Dispatch either prompt admission or the between-step checkpoint. */ async function boundary(ctx: Context, agent: Agent & { session: Session }, type: 'turn/start' | 'step/end'): Promise { const events = agentEvents(ctx, agent) if (type === 'turn/start') { - agent.session.append('turn/start', { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } }) await events.waterfall('agent/prompt-submit', [{ type: 'text', text: 'boundary probe' }], { kind: 'user' }, new AbortController().signal, () => Promise.resolve({ kind: 'allow' })) return } - agent.session.append('step/end', { turn: 1, step: 1 }) - await events.waterfall('agent/turn-continuation', 1, { action: 'stop' }, new AbortController().signal, () => Promise.resolve({ action: 'stop' })) -} - -/** Dispatch the closed-step recovery seam with one terminal decision. */ -function recoveryBoundary( - ctx: Context, - agent: Agent & { session: Session }, - decision: RequestErrorDecision, -): Promise { - return agentEvents(ctx, agent).waterfall( - 'agent/request-error', - 1, - 1, - new Error('request failed'), - { message: 'request failed', code: 'SERVER' }, - [], - new AbortController().signal, - () => Promise.resolve(decision), - ) + await events.serial('agent/step', 1, 2, new AbortController().signal) } /** Append a minimal `request/header` snapshot so the log has a "what the model was told" anchor. */ @@ -218,16 +193,14 @@ describe('the boundary flush', () => { // selection lands DURING its await — after this boundary began, before it // returns. The prepended flush runs after next(), so the plan/mode still // precedes the request this boundary gates. - ctx.on('agent/turn-continuation', async (_agent, _turn, decision, _signal, next) => { + ctx.on('agent/prompt-submit', async (_agent, _content, _source, _signal, next) => { await new Promise(resolve => setTimeout(resolve, 5)) ctx.planMode.set(agent, true) - await next() - return decision + return next() }) - agent.session.append('step/end', { turn: 1, step: 1 }) await agentEvents(ctx, agent).waterfall( - 'agent/turn-continuation', 1, { action: 'stop' }, new AbortController().signal, - () => Promise.resolve({ action: 'stop' }), + 'agent/prompt-submit', [{ type: 'text', text: 'probe' }], { kind: 'user' }, + new AbortController().signal, () => Promise.resolve({ kind: 'allow' }), ) expect(foldPlanMode(agent.session.events)).toBe(true) expect(ctx.planMode.get(agent)).toEqual({ active: true }) @@ -243,20 +216,18 @@ describe('the boundary flush', () => { // A downstream listener captured before disposal keeps the waterfall // continuation alive across the unload; the resumed wrapper must not // append through the disposed service. - ctx.on('agent/turn-continuation', async (_agent, _turn, decision, _signal, next) => { + ctx.on('agent/prompt-submit', async (_agent, _content, _source, _signal, next) => { await fiber.dispose() - await next() - return decision + return next() }) - agent.session.append('step/end', { turn: 1, step: 1 }) await agentEvents(ctx, agent).waterfall( - 'agent/turn-continuation', 1, { action: 'stop' }, new AbortController().signal, - () => Promise.resolve({ action: 'stop' }), + 'agent/prompt-submit', [{ type: 'text', text: 'probe' }], { kind: 'user' }, + new AbortController().signal, () => Promise.resolve({ kind: 'allow' }), ) expect(agent.session.events.some(event => event.type === 'plan/mode')).toBe(false) }) - it('flushes at step/end too (a mid-turn flip lands on the following step)', async () => { + it('flushes at the between-step seam too', async () => { const ctx = await setup() const agent = await agentWithSession(ctx) ctx.planMode.set(agent, true) @@ -264,30 +235,6 @@ describe('the boundary flush', () => { expect(foldPlanMode(agent.session.events)).toBe(true) }) - it('keeps the pending intent parked when recovery does not retry', async () => { - const ctx = await setup() - const agent = await agentWithSession(ctx) - ctx.planMode.set(agent, true) - expect(await recoveryBoundary(ctx, agent, { action: 'fail' })).toEqual({ action: 'fail' }) - expect(ctx.planMode.get(agent)).toEqual({ active: false, pending: true }) - }) - - it('contains an append failure at the retry boundary without changing its decision', async () => { - const ctx = await setup() - const warn = vi.fn() - ctx.logger.warn = warn as never - const agent = await agentWithSession(ctx) - ctx.planMode.set(agent, true) - const original = agent.session.append.bind(agent.session) - agent.session.append = (((type: string, ...rest: unknown[]) => { - if (type === 'plan/mode') throw new Error('backend gone') - return (original as (...args: unknown[]) => unknown)(type, ...rest) - }) as unknown) as typeof agent.session.append - - expect(await recoveryBoundary(ctx, agent, { action: 'retry' })).toEqual({ action: 'retry' }) - expect(warn).toHaveBeenCalledOnce() - expect(ctx.planMode.get(agent)).toEqual({ active: false, pending: true }) - }) it('nets out a flip sequence that returns to the folded mode (no append, no notice)', async () => { const ctx = await setup() @@ -937,30 +884,6 @@ describe('exit_plan_mode', () => { }) describe('HMR disposal', () => { - it('does not flush a retry boundary that resumes after plugin disposal', async () => { - const ctx = new Context() - await ctx.plugin(SystemPrompt) - await ctx.plugin(ToolRegistry) - const fiber = await ctx.plugin(PlanModeService, PLAN_CONFIG) - const agent = await agentWithSession(ctx, 'disposed-in-flight-recovery') - const recoveryEntered = Promise.withResolvers() - const releaseRecovery = Promise.withResolvers() - ctx.on('agent/request-error', async (_agent, _turn, _step, _error, _failure, _history, _signal, _next) => { - recoveryEntered.resolve(true) - await releaseRecovery.promise - return { action: 'retry' } - }) - ctx.planMode.set(agent, true) - - const recovery = recoveryBoundary(ctx, agent, { action: 'fail' }) - await recoveryEntered.promise - await fiber.dispose() - releaseRecovery.resolve(true) - - expect(await recovery).toEqual({ action: 'retry' }) - expect(agent.session.events.some(event => event.type === 'plan/mode')).toBe(false) - }) - it('unregisters the service, listeners, prompt section, and stable exit tool with the plugin fiber', async () => { const ctx = new Context() await ctx.plugin(SystemPrompt) @@ -976,7 +899,7 @@ describe('HMR disposal', () => { expect(ctx.get('planMode')).toBeUndefined() expect(ctx.tools.get(EXIT_PLAN_MODE)).toBeUndefined() expect((await ctx.systemPrompt.assemble()).sections.map(section => section.name)).not.toContain('plan:policy') - expect(await recoveryBoundary(ctx, agent, { action: 'retry' })).toEqual({ action: 'retry' }) + await boundary(ctx, agent, 'step/end') expect(agent.session.events.some(event => event.type === 'plan/mode')).toBe(false) }) }) diff --git a/packages/pty/pty-local/tests/index.spec.ts b/packages/pty/pty-local/tests/index.spec.ts index 26f4ddcc11..23209c8800 100644 --- a/packages/pty/pty-local/tests/index.spec.ts +++ b/packages/pty/pty-local/tests/index.spec.ts @@ -42,7 +42,7 @@ function agent(ctx: Context): Agent { const id = SessionId('agent') return { id, options: {}, session: new Session(id), status: 'idle', ctx, - followup: () => AgentMessageId('stub'), queue: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject: () => AgentMessageId('stub'), send: () => AgentMessageId('stub'), cancel() {}, whenIdle: () => Promise.resolve(), + followup: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject: () => AgentMessageId('stub'), send: () => AgentMessageId('stub'), cancel() {}, retry() {}, whenIdle: () => Promise.resolve(), } } @@ -244,7 +244,7 @@ describe('pty-local plugin shape', () => { const ownerFiber = await ctx.plugin(() => {}) const owner: Agent = { id: session.id, options: {}, session, status: 'idle', ctx: ownerFiber.ctx, - followup: () => AgentMessageId('stub'), queue: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject: () => AgentMessageId('stub'), send: () => AgentMessageId('stub'), cancel() {}, whenIdle: () => Promise.resolve(), + followup: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject: () => AgentMessageId('stub'), send: () => AgentMessageId('stub'), cancel() {}, retry() {}, whenIdle: () => Promise.resolve(), } ctx.agents.register(owner) const providerFiber = await registerStubLocalBackend(ctx, () => stubLocalSession()) @@ -287,7 +287,7 @@ describe('pty-local plugin shape', () => { const ownerFiber = await ctx.plugin(() => {}) const owner: Agent = { id: session.id, options: {}, session, status: 'idle', ctx: ownerFiber.ctx, - followup: () => AgentMessageId('stub'), queue: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject: () => AgentMessageId('stub'), send: () => AgentMessageId('stub'), cancel() {}, whenIdle: () => Promise.resolve(), + followup: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject: () => AgentMessageId('stub'), send: () => AgentMessageId('stub'), cancel() {}, retry() {}, whenIdle: () => Promise.resolve(), } ctx.agents.register(owner) const gate = Promise.withResolvers() diff --git a/packages/pty/pty-local/tests/local.spec.ts b/packages/pty/pty-local/tests/local.spec.ts index 763ab5c871..71cb952aca 100644 --- a/packages/pty/pty-local/tests/local.spec.ts +++ b/packages/pty/pty-local/tests/local.spec.ts @@ -35,7 +35,7 @@ function stubAgent(ctx: Context, rawId: string): Agent { const scope = ctx.plugin(() => {}) return { id, options: {}, session: new Session(id), status: 'idle', ctx: scope.ctx, - followup: () => AgentMessageId('stub'), queue: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject: () => AgentMessageId('stub'), send: () => AgentMessageId('stub'), cancel() {}, whenIdle: () => Promise.resolve(), + followup: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject: () => AgentMessageId('stub'), send: () => AgentMessageId('stub'), cancel() {}, retry() {}, whenIdle: () => Promise.resolve(), } } diff --git a/packages/pty/pty/tests/service.spec.ts b/packages/pty/pty/tests/service.spec.ts index 42622f2712..383a96444b 100644 --- a/packages/pty/pty/tests/service.spec.ts +++ b/packages/pty/pty/tests/service.spec.ts @@ -28,11 +28,11 @@ function stubAgent(ctx: Context, rawId: string): Agent { status: 'idle', ctx: scopeFiber.ctx, followup: () => AgentMessageId('stub'), - queue: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject: () => AgentMessageId('stub'), send: () => AgentMessageId('stub'), cancel() {}, + retry() {}, whenIdle: () => Promise.resolve(), } agentScopeDisposers.set(agent, async () => { await scopeFiber.dispose() }) diff --git a/packages/pty/tool-pty/tests/loader-composition.spec.ts b/packages/pty/tool-pty/tests/loader-composition.spec.ts index 40477aeb7a..f70442ded1 100644 --- a/packages/pty/tool-pty/tests/loader-composition.spec.ts +++ b/packages/pty/tool-pty/tests/loader-composition.spec.ts @@ -40,7 +40,7 @@ function agent(ctx: Context): Agent { const id = SessionId('pty-loader-agent') const value: Agent = { id, options: {}, session: new Session(id), status: 'idle', ctx: scope.ctx, - followup: () => AgentMessageId('stub'), queue: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject: () => AgentMessageId('stub'), send: () => AgentMessageId('stub'), cancel() {}, whenIdle: () => Promise.resolve(), + followup: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject: () => AgentMessageId('stub'), send: () => AgentMessageId('stub'), cancel() {}, retry() {}, whenIdle: () => Promise.resolve(), } ctx.agents.register(value) return value diff --git a/packages/pty/tool-pty/tests/tools.spec.ts b/packages/pty/tool-pty/tests/tools.spec.ts index e0b854ee43..65e6dd9961 100644 --- a/packages/pty/tool-pty/tests/tools.spec.ts +++ b/packages/pty/tool-pty/tests/tools.spec.ts @@ -18,7 +18,7 @@ function fakeAgent(ctx: Context, rawId: string): Agent { const id = SessionId(rawId) const agent: Agent = { id, options: {}, session: new Session(id), status: 'idle', ctx: scope.ctx, - followup: () => AgentMessageId('stub'), queue: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject: () => AgentMessageId('stub'), send: () => AgentMessageId('stub'), cancel() {}, whenIdle: () => Promise.resolve(), + followup: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject: () => AgentMessageId('stub'), send: () => AgentMessageId('stub'), cancel() {}, retry() {}, whenIdle: () => Promise.resolve(), } ctx.agents.register(agent) return agent diff --git a/packages/sdk/helper/src/package-managers/link-workspace.ts b/packages/sdk/helper/src/package-managers/link-workspace.ts index 1a6b518fa9..e44ea3ea0b 100644 --- a/packages/sdk/helper/src/package-managers/link-workspace.ts +++ b/packages/sdk/helper/src/package-managers/link-workspace.ts @@ -67,6 +67,7 @@ export class LinkWorkspace { try { manifest = JSON.parse(await readFile(join(directory, 'package.json'), 'utf8')) as PackageManifest } catch (error) { + if (error instanceof Error && 'code' in error && error.code === 'ENOENT') continue throw new Error(`cannot read linked package at ${directory}: ${String(error)}`) } if (!manifest.name || typeof manifest.name !== 'string') continue diff --git a/packages/sdk/helper/tests/documents.spec.ts b/packages/sdk/helper/tests/documents.spec.ts index 1e4b944f00..19d8496475 100644 --- a/packages/sdk/helper/tests/documents.spec.ts +++ b/packages/sdk/helper/tests/documents.spec.ts @@ -386,7 +386,7 @@ describe('package manager strategies', () => { temporary.push(unreadable) await mkdir(join(unreadable, 'vendor', 'bad'), { recursive: true }) await mkdir(join(unreadable, 'packages'), { recursive: true }) - await expect(LinkWorkspace.open(unreadable)).rejects.toThrow('cannot read linked package') + await expect(LinkWorkspace.open(unreadable)).rejects.toThrow('not a DeepSeek Harness repository root') const unnamed = await mkdtemp(join(tmpdir(), 'dsh-link-unnamed-')) temporary.push(unnamed) await mkdir(join(unnamed, 'vendor', 'unnamed'), { recursive: true }) diff --git a/packages/session-persistence/session-checkpoint-policy/README.md b/packages/session-persistence/session-checkpoint-policy/README.md index 004c49c5cb..0966097cf8 100644 --- a/packages/session-persistence/session-checkpoint-policy/README.md +++ b/packages/session-persistence/session-checkpoint-policy/README.md @@ -1,6 +1,6 @@ # dsh-session-checkpoint-policy -Semantic durability policy for persisted agents. It checkpoints the event-sourced session before a model adapter receives a request, before a top-level tool body may produce an external side effect, and after a step has recorded its complete assistant message and ordered tool results. The final `turn/end` checkpoint remains owned by `dsh-agent-loop`. +Semantic durability policy for persisted agents. It checkpoints the event-sourced session before a model adapter receives a request, before a top-level tool body may produce an external side effect, and at each `agent/step` boundary so the preceding response and ordered tool results are durable before the next request. ## Plugin (namespace: `session-checkpoint-policy`) @@ -14,13 +14,11 @@ This zero-config function plugin consumes `ctx.sessions`, `ctx.llm`, `ctx.tools` name: '@deepseek-ai/dsh-session-checkpoint-policy' ``` -Persistence and checkpoint scheduling are intentionally separate Cordis plugins. A persistence backend makes each requested `session/flush` durable; this policy chooses the request, tool-dispatch, and completed-step checkpoints. Loading a backend without this policy is valid and retains checkpoints requested by the loop, including final `turn/end`, but crash recovery may lose the rest of an in-flight turn. First-party persisted apps and runtimes mount both plugins explicitly; a specialized deployment may deliberately omit or replace the policy. +Persistence and checkpoint scheduling are intentionally separate Cordis plugins. A persistence backend eagerly writes `session/event` appends and makes each requested `session/flush` an observation barrier; this policy chooses the request, tool-dispatch, and next-step barriers. Loading a backend without this policy is valid, but a crash may lose the latest eagerly buffered events. First-party persisted apps and runtimes mount both plugins explicitly; a specialized deployment may deliberately omit or replace the policy. -The policy wraps `llm/stream` lazily, so the downstream stream is not constructed until the live session's buffered request events are durable. It wraps `tools/execute` after pre-execute policy and guards; a top-level tool body runs only after its recorded call is durable. If cancellation lands while that flush is pending, the wrapper returns the canonical `ABORTED_BEFORE_DISPATCH` result without entering the tool body. Nested tool dispatches reuse the outer model-visible call's checkpoint. `agent/post-step` persists the complete response/result batch before continuation work. +The policy wraps `llm/stream` lazily, so the downstream stream is not constructed until the live session's buffered request events are durable. It wraps `tools/execute` after pre-execute policy and guards; a top-level tool body runs only after its recorded call is durable. If cancellation lands while that flush is pending, the wrapper returns the canonical `ABORTED_BEFORE_DISPATCH` result without entering the tool body. Nested tool dispatches reuse the outer model-visible call's checkpoint. `agent/step` persists the preceding response/result batch before request derivation. -The loop records its assistant message and ordered tool results before dispatching `agent/post-step`, so the policy always captures that core batch. An event appended by another `agent/post-step` listener is captured at this checkpoint only when that listener is registered before the policy; Cordis registration order is the explicit composition rule for such extensions. - -Checkpoint rejection is fail-closed at the model and tool boundaries: neither the adapter nor the top-level tool body runs. A post-step rejection fails the turn before another request starts. Concurrent tool checkpoints share the session store's serialized persistence drain and cannot duplicate sequence numbers. +Checkpoint rejection is fail-closed at the model and tool boundaries: neither the adapter nor the top-level tool body runs. A step-boundary rejection fails the turn before another request starts. Concurrent tool checkpoints share the session store's serialized persistence drain and cannot duplicate sequence numbers. ## Model Experience diff --git a/packages/session-persistence/session-checkpoint-policy/tests/session-checkpoint-policy.spec.ts b/packages/session-persistence/session-checkpoint-policy/tests/session-checkpoint-policy.spec.ts index 3cbb3b9832..7a2081bd60 100644 --- a/packages/session-persistence/session-checkpoint-policy/tests/session-checkpoint-policy.spec.ts +++ b/packages/session-persistence/session-checkpoint-policy/tests/session-checkpoint-policy.spec.ts @@ -217,15 +217,13 @@ describe('session-checkpoint-policy tool and step boundaries', () => { expect(flushes).toBe(0) }) - it('checkpoints the complete recorded step at agent/post-step', async () => { + it('checkpoints before the next agent step', async () => { const ctx = await setup() const session = ctx.sessions.create(SessionId('post-step')) const agent = { session } as Agent const flushed: string[] = [] ctx.on('session/flush', (current) => { flushed.push(current.id) }) - await agentEvents(ctx, agent).serial( - 'agent/post-step', 1, 1, new AbortController().signal, - ) + await agentEvents(ctx, agent).serial('agent/step', 1, 1, new AbortController().signal) expect(flushed).toEqual([session.id]) }) }) diff --git a/packages/session-persistence/session-persistence/src/coordinator.ts b/packages/session-persistence/session-persistence/src/coordinator.ts index fb46aa4877..c839ff3c21 100644 --- a/packages/session-persistence/session-persistence/src/coordinator.ts +++ b/packages/session-persistence/session-persistence/src/coordinator.ts @@ -154,6 +154,8 @@ export class PersistenceCoordinator { private states = new Map() /** Lifecycle and write-behind state keyed by the exact live Session. */ private live = new Map() + /** Exact disposed lifecycles whose eager tail is still draining. */ + private retirements = new Map>() /** Cold loads currently reserving an id across backend reads and repair writes. */ private coldLoads = new Set() /** @@ -250,6 +252,7 @@ export class PersistenceCoordinator { * @returns the header plus the event log, ending on a balanced `turn/end`. */ async load(id: SessionId): Promise<{ meta: SessionHeader; events: SessionEvent[] }> { + await this.retirements.get(id) const selected = await this.serialize(id, async () => { const live = this.ctx.sessions.get(id) if (live !== undefined) return { live } @@ -270,7 +273,8 @@ export class PersistenceCoordinator { * @returns stored header and events before any synthetic recovery closers. */ inspect(id: SessionId): Promise<{ meta: SessionHeader; events: SessionEvent[] }> { - return this.serialize(id, () => this.inspectCore(id)) + return Promise.resolve(this.retirements.get(id)) + .then(() => this.serialize(id, () => this.inspectCore(id))) } private async inspectCore(id: SessionId): Promise<{ meta: SessionHeader; events: SessionEvent[] }> { @@ -432,7 +436,13 @@ export class PersistenceCoordinator { /** Start and observe one disposed session's final drain. */ private retire(session: Session): void { if (!this.live.has(session)) return - void this.retireCore(session).catch((error: unknown) => { + const retirement = this.retireCore(session) + this.retirements.set(session.id, retirement) + const forget = (): void => { + if (this.retirements.get(session.id) === retirement) this.retirements.delete(session.id) + } + void retirement.then(forget, forget) + void retirement.catch((error: unknown) => { this.ctx.logger.warn(`${this.backend.name}: session "${session.id}" retirement failed: ${String(error)}`) }) } diff --git a/packages/skill/tool-skill/tests/tool-skill.spec.ts b/packages/skill/tool-skill/tests/tool-skill.spec.ts index 7fef1cbf22..3c7f7f86f5 100644 --- a/packages/skill/tool-skill/tests/tool-skill.spec.ts +++ b/packages/skill/tool-skill/tests/tool-skill.spec.ts @@ -4,6 +4,8 @@ import { join } from 'node:path' import { tmpdir } from 'node:os' import { Context } from 'cordis' import { CallId, type Message } from '@deepseek-ai/dsh-llm' +import { AgentMessageId } from '@deepseek-ai/dsh-agent' +import { Session, SessionId } from '@deepseek-ai/dsh-session' import { createScope, type Scope } from '@deepseek-ai/dsh-scope' import SystemPrompt, { renderPrompt } from '@deepseek-ai/dsh-system-prompt' import ToolRegistry, { defineContentToolFixture } from '@deepseek-ai/dsh-tools' @@ -35,7 +37,28 @@ async function setup(home: string, config: toolSkill.Config = {}): Promise AgentMessageId('stub'), + followup: () => AgentMessageId('stub'), + steer: () => AgentMessageId('stub'), + inject(content, options) { + session.append('user/message', { + content, + source: options?.source ?? { kind: 'user' }, + }, { surfaceOp: 'append' }) + return AgentMessageId('stub') + }, + cancel() {}, + retry() {}, + whenIdle: () => Promise.resolve(), + } } async function composePrefix(ctx: Context, cwd: string, signal = new AbortController().signal): Promise { @@ -43,11 +66,8 @@ async function composePrefix(ctx: Context, cwd: string, signal = new AbortContro } async function composePrefixForAgent(ctx: Context, agent: Agent, signal = new AbortController().signal): Promise { - const empty: Message[] = [] - return await agentEvents(ctx, agent).waterfall( - 'agent/session-prefix', empty, signal, - () => Promise.resolve(empty), - ) + await agentEvents(ctx, agent).serial('agent/step', 1, 1, signal) + return agent.session.deriveMessages() } async function mintAgentScope(ctx: Context, cwd: string): Promise<{ agent: Agent; scope: Scope }> { @@ -86,7 +106,7 @@ describe('dsh-tool-skill', () => { expect(ctx.tools.schemas().map(tool => tool.name)).toEqual(['skill']) }) - it('forwards the session-prefix abort signal to skill discovery', async () => { + it('forwards the step abort signal to skill discovery', async () => { const home = await tempDir('tool-prefix-signal') const ctx = await setup(home) let seenSignal: AbortSignal | undefined @@ -107,7 +127,7 @@ describe('dsh-tool-skill', () => { expect(seenSignal).toBe(controller.signal) }) - it('contributes a stable name-and-description catalog through the session prefix', async () => { + it('injects a stable durable name-and-description catalog at the first step', async () => { const home = await tempDir('tool-catalog') const ctx = await setup(home, { catalogDescriptionMaxLength: 50 }) ctx.skills.register({ @@ -126,10 +146,11 @@ describe('dsh-tool-skill', () => { provider: 'runtime', content: 'A body.', }) - ctx.on('agent/session-prefix', async (_agent, _prefix, _signal, next): Promise => [ - { role: 'user', content: [{ type: 'text', text: 'later contribution' }] }, - ...await next(), - ]) + ctx.on('agent/step', (agent) => { + agent.inject([{ type: 'text', text: 'later contribution' }], { + source: { kind: 'plugin', plugin: 'later-contribution' }, + }) + }) const prefix = await composePrefix(ctx, '/workspace') @@ -162,7 +183,7 @@ describe('dsh-tool-skill', () => { expect(renderPrompt(await ctx.systemPrompt.assemble({ agent: agentForCwd('/workspace') }))).not.toContain('') }) - it('does not contribute a session-prefix message when no skills are available', async () => { + it('does not inject a catalog when no skills are available', async () => { const home = await tempDir('tool-empty-catalog') const ctx = await setup(home) diff --git a/packages/subagent/subagent-fork/tests/subagent-fork.spec.ts b/packages/subagent/subagent-fork/tests/subagent-fork.spec.ts index cf5848e625..f1684ba55f 100644 --- a/packages/subagent/subagent-fork/tests/subagent-fork.spec.ts +++ b/packages/subagent/subagent-fork/tests/subagent-fork.spec.ts @@ -155,7 +155,7 @@ describe('dsh-subagent-fork', () => { // 1 from the seeded parent turn + 1 from the child's own completed turn. expect(seedTurnEnds.length).toBe(2) - parent.cancel() + parent.cancel({ kind: 'user' }) await run.dispose() }) diff --git a/packages/subagent/subagent-inprocess/tests/structured.spec.ts b/packages/subagent/subagent-inprocess/tests/structured.spec.ts index bb4cf32544..d67d5f3537 100644 --- a/packages/subagent/subagent-inprocess/tests/structured.spec.ts +++ b/packages/subagent/subagent-inprocess/tests/structured.spec.ts @@ -1,7 +1,6 @@ import { describe, expect, it } from 'vitest' import { Context } from 'cordis' import { CallId, type ContentBlock, type GenerateOptions } from '@deepseek-ai/dsh-llm' -import type { ContinuationDecision } from '@deepseek-ai/dsh-agent' import { SessionId } from '@deepseek-ai/dsh-session' import AgentLoop from '@deepseek-ai/dsh-agent-loop' import { mountAgentLoopTestDependencies } from '@deepseek-ai/dsh-agent-loop-testkit' @@ -116,8 +115,7 @@ describe('in-process structured output', () => { ]) const run = await ctx.subagents.start('spawn', structuredRequest(parent)) await run.result - // Default continuation would run a second step after the tool call; the - // structured runtime's turn-continuation veto stops the turn instead. + // The structured tool marks its successful result as turn-concluding. expect(adapter.requests.length).toBe(1) await run.dispose() }) @@ -219,63 +217,6 @@ describe('in-process structured output', () => { await run.dispose() }) - it('a later-prepended continuation wrapper cannot resurrect a captured turn', async () => { - const { ctx, parent, adapter } = await setup([ - toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 7 }), - textResponse('MUST NOT BE CONSUMED'), - ]) - ctx.on('agent/turn-continuation', () => Promise.resolve({ action: 'stop' })) - let wrapperInstalled = false - // Register before ready-only start: structured output is attached before session-start and the - // loop. The wrapper waits for a downstream stop, rewrites it to continue, and must still lose - // to the later terminal checkpoint. - ctx.on('agent/session-start', (child) => { - if (child === parent) return - wrapperInstalled = true - child.ctx.on('agent/turn-continuation', async (_subject, _turn, _decision, _signal, next): Promise => { - const downstream = await next() - expect(downstream).toEqual({ action: 'stop' }) - return { action: 'continue' } - }, { prepend: true }) - }) - const run = await ctx.subagents.start('spawn', structuredRequest(parent)) - const result = await run.result - expect(wrapperInstalled).toBe(true) - expect(result.structured).toEqual({ answer: 7 }) - expect(result.stopReason).toBe('completed') - expect(adapter.requests).toHaveLength(1) - await run.dispose() - }) - - it('a continuation wrapper cannot carry steering past a captured terminal stop', async () => { - const { ctx, parent, adapter } = await setup([ - toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 9 }), - textResponse('MUST NOT BE CONSUMED'), - ]) - // A downstream policy stops, then a later wrapper delegates and queues steering that ordinary - // folding would turn into continue. The terminal checkpoint must discard that steering. - ctx.on('agent/turn-continuation', () => Promise.resolve({ action: 'stop' })) - const run = await ctx.subagents.start('spawn', structuredRequest(parent)) - ctx.on('agent/session-start', (child) => { - if (child.id !== run.id) return - child.ctx.on('agent/turn-continuation', async (subject, _turn, _decision, _signal, next): Promise => { - const downstream = await next() - expect(downstream).toEqual({ action: 'stop' }) - subject.steer([{ type: 'text', text: 'late steering after downstream stop' }]) - return downstream - }, { prepend: true }) - }) - - const result = await run.result - const child = ctx.agents.get(run.id) - - expect(result.structured).toEqual({ answer: 9 }) - expect(adapter.requests).toHaveLength(1) - expect(child?.session.events.filter(event => event.type === 'turn/start')).toHaveLength(1) - expect(child?.session.events.filter(event => event.type === 'steering/message')).toHaveLength(0) - await run.dispose() - }) - it('an invalid call gets an INVALID_ARGS isError result and the model retries in-turn', async () => { const { ctx, parent } = await setup([ toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 'not-a-number' }), diff --git a/packages/subagent/subagent-inprocess/tests/subagent-inprocess.spec.ts b/packages/subagent/subagent-inprocess/tests/subagent-inprocess.spec.ts index ce91480ea8..3186b4551a 100644 --- a/packages/subagent/subagent-inprocess/tests/subagent-inprocess.spec.ts +++ b/packages/subagent/subagent-inprocess/tests/subagent-inprocess.spec.ts @@ -80,7 +80,7 @@ describe('startInProcessRun', () => { const child = ctx.agents.get(run.id)! expect(child.session.events.findLast(event => event.type === 'turn/end')) - .toMatchObject({ data: { reason: { kind: 'completed' } } }) + .toMatchObject({ data: { reason: { kind: 'max-tokens' } } }) expect(result.stopReason).toBe('max-tokens') await run.dispose() }) diff --git a/packages/subagent/subagent-spawn/tests/subagent-spawn.spec.ts b/packages/subagent/subagent-spawn/tests/subagent-spawn.spec.ts index c83eca75ab..e72ed07b69 100644 --- a/packages/subagent/subagent-spawn/tests/subagent-spawn.spec.ts +++ b/packages/subagent/subagent-spawn/tests/subagent-spawn.spec.ts @@ -1,7 +1,7 @@ import { describe, expect, it } from 'vitest' -import { Context } from 'cordis' +import { Context, symbols, type EffectMeta } from 'cordis' import Loader from '@cordisjs/plugin-loader' -import AgentRegistry from '@deepseek-ai/dsh-agent' +import AgentRegistry, { type Agent } from '@deepseek-ai/dsh-agent' import { SessionId } from '@deepseek-ai/dsh-session' import AgentLoop from '@deepseek-ai/dsh-agent-loop' import { mountAgentLoopTestDependencies } from '@deepseek-ai/dsh-agent-loop-testkit' @@ -52,6 +52,17 @@ function start(ctx: Context, provider: string, request: Omit { + const effect = (dispose as typeof dispose & { [symbols.effect]?: EffectMeta })[symbols.effect] + return effect?.label.startsWith('agentLoop.lifecycle(') === true + }) + if (lifecycle === undefined) throw new Error('child lifecycle effect not found') + void lifecycle() +} + describe('dsh-subagent-spawn', () => { it('runs a fresh child to completion and returns its final assistant output', async () => { // One model call for the child: a plain text answer. @@ -460,6 +471,12 @@ describe('dsh-subagent-spawn', () => { ctx.on('session/created', () => void published.push('session/created')) ctx.on('agent/created', () => void published.push('agent/created')) ctx.on('agent/session-start', () => void published.push('agent/session-start')) + let teardownStarted = false + ctx.on('internal/plugin', (fiber) => { + if (teardownStarted || fiber.name !== 'scope') return + teardownStarted = true + disposeChildLifecycle(parentHandle.agent) + }) const starting = start(ctx, 'spawn', { prompt: [{ type: 'text', text: 'must never run' }], @@ -468,8 +485,8 @@ describe('dsh-subagent-spawn', () => { // The factory has entered its awaited unpublished setup transaction. The // parent context owns that transaction, so disposal wins without an // observer ever seeing the child. - await parentHandle.dispose() await expect(starting).rejects.toThrow(/owner disposed during setup|inactive context/) + await parentHandle.dispose() expect(published).toEqual([]) }) diff --git a/packages/tasks/tasks/tests/tasks.spec.ts b/packages/tasks/tasks/tests/tasks.spec.ts index 015f1c4f2b..12c4a53221 100644 --- a/packages/tasks/tasks/tests/tasks.spec.ts +++ b/packages/tasks/tasks/tests/tasks.spec.ts @@ -24,11 +24,11 @@ function stubAgent(ctx: Context, rawId: string): Agent { status: 'idle' as const, ctx: scopeFiber.ctx, followup: () => AgentMessageId('stub'), - queue: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject: () => AgentMessageId('stub'), send: () => AgentMessageId('stub'), cancel() {}, + retry() {}, whenIdle() { return Promise.resolve() }, } agentScopeDisposers.set(agent, async () => { await scopeFiber.dispose() }) diff --git a/packages/ui/acp/src/index.ts b/packages/ui/acp/src/index.ts index 37f9a5dd44..a62def17b9 100644 --- a/packages/ui/acp/src/index.ts +++ b/packages/ui/acp/src/index.ts @@ -732,12 +732,11 @@ export function apply(ctx: Context, config: AcpConfig): void { presets.set(rec.agent.session, pending.preset) } - // Prompt-submit is inside the new turn but before prompt assembly. Promptless - // injection turns leave the switch pending because they execute no request. - ctx.on('agent/prompt-submit', (agent, _content, _source, _signal, next) => { + // The first step boundary is inside the admitted turn and before request + // assembly. Idle injections leave the switch pending because they run no step. + ctx.on('agent/step', (agent) => { const rec = ownedRecord(agent) if (rec !== undefined) flushPendingSwitches(rec) - return next() }) const makeAgent = (connection: AgentSideConnection): AcpAgent => { diff --git a/packages/ui/acp/tests/config-options.spec.ts b/packages/ui/acp/tests/config-options.spec.ts index c6d385185c..6108836cda 100644 --- a/packages/ui/acp/tests/config-options.spec.ts +++ b/packages/ui/acp/tests/config-options.spec.ts @@ -186,11 +186,10 @@ describe('acp bridge — session config options', () => { const { sessionId } = await h.client.newSession({ cwd: process.cwd(), mcpServers: [] }) const agent = h.ctx.agents.list()[0] if (agent === undefined) throw new Error('expected an agent') - agent.ctx.on('agent/request', async (_agent, _turn, _step, callConfig, _signal, _next) => ({ - ...callConfig, - provider: 'mock', - model: 'mock', - })) + agent.ctx.on('agent/request', async (_agent, _turn, _step, _signal, next) => { + const callConfig = await next() + return { ...callConfig, provider: 'mock', model: 'mock' } + }) await h.client.prompt({ sessionId, prompt: [{ type: 'text', text: 'supplied elsewhere' }] }) expect(h.adapter.requests[0]).toMatchObject({ provider: 'mock', model: 'mock' }) }) diff --git a/packages/ui/acp/tests/dispose.spec.ts b/packages/ui/acp/tests/dispose.spec.ts index bcfb177b16..c95e4970ca 100644 --- a/packages/ui/acp/tests/dispose.spec.ts +++ b/packages/ui/acp/tests/dispose.spec.ts @@ -102,8 +102,8 @@ describe('acp bridge — disposal & HMR safety', () => { // agent's AgentHandle dispose to quiescence on its OWN (before any dispose()). await harness.closeClientTransport() await agent.whenIdle() - // The agent's loop has stopped: status `disposed`. - expect(agent.status).toBe('disposed') + // The retired object is quiescent; registry membership carries liveness. + expect(agent.status).toBe('idle') // Await the bridge teardown to completion WITHOUT tearing down the root // agents/sessions services (so we can still query them). acpFiber.dispose() @@ -242,11 +242,11 @@ describe('acp bridge — disposal & HMR safety', () => { // A is gone — unregistered AND its session removed from the store. expect(harness.ctx.agents.get(SessionId('sib-a'))).toBeUndefined() expect(harness.ctx.sessions.get(SessionId('sib-a'))).toBeUndefined() - expect(handleA.agent.status).toBe('disposed') + expect(handleA.agent.status).toBe('idle') // B is wholly unaffected. expect(harness.ctx.agents.get(SessionId('sib-b'))).toBe(handleB.agent) expect(harness.ctx.sessions.get(SessionId('sib-b'))).toBeDefined() - expect(handleB.agent.status).not.toBe('disposed') + expect(handleB.agent.status).toBe('idle') await harness.dispose() }) @@ -291,27 +291,9 @@ describe('acp bridge — disposal & HMR safety', () => { handle.agent.followup([{ type: 'text', text: 'go' }]) await new Promise(r => setTimeout(r, 30)) expect(handle.agent.status).toBe('running') - let releaseFlush!: () => void - const flushGate = new Promise((resolve) => { releaseFlush = resolve }) - harness.ctx.on('session/flush', () => flushGate) - - // First dispose enters teardown (aborts the hanging step) and blocks in the - // gated final flush. + // Both callers join the same teardown and observe registry removal. const first = handle.dispose() - let firstSettled = false - void first.then(() => { firstSettled = true }) - await new Promise(r => setTimeout(r, 20)) - expect(firstSettled).toBe(false) - - // Second dispose MUST await the same in-flight teardown, not resolve early. const second = handle.dispose() - let secondSettled = false - void second.then(() => { secondSettled = true }) - await new Promise(r => setTimeout(r, 20)) - expect(secondSettled).toBe(false) // memoized: still pending with the first - - // Release the flush; both resolve together and the session is gone. - releaseFlush() await Promise.all([first, second]) expect(harness.ctx.agents.get(SessionId('conc-a'))).toBeUndefined() expect(harness.ctx.sessions.get(SessionId('conc-a'))).toBeUndefined() diff --git a/packages/ui/acp/tests/turns.spec.ts b/packages/ui/acp/tests/turns.spec.ts index b3dcc22a55..c8823413e7 100644 --- a/packages/ui/acp/tests/turns.spec.ts +++ b/packages/ui/acp/tests/turns.spec.ts @@ -51,7 +51,7 @@ describe('acp bridge — turn outcomes', () => { it('rejects an ordinary plugin turn failure through the same ACP boundary', async () => { harness = await makeBridgeHarness({ storageDir, script: [textResponse('must not run')] }) - harness.ctx.on('agent/pre-step', () => { throw new Error('plugin pre-step failed') }) + harness.ctx.on('agent/step', () => { throw new Error('plugin pre-step failed') }) const sessionId = await newSession(harness) await expect(harness.client.prompt({ sessionId, prompt: [{ type: 'text', text: 'go' }] })) diff --git a/packages/ui/tui/tests/harness.ts b/packages/ui/tui/tests/harness.ts index 8ce5996727..6e0e0ebe4d 100644 --- a/packages/ui/tui/tests/harness.ts +++ b/packages/ui/tui/tests/harness.ts @@ -4,6 +4,7 @@ import AgentRegistry, { AgentMessageId, type Agent, type AgentCancelCause, + type AliasSendOptions, type AgentOptions, type AgentStatus, type SendOptions, @@ -20,11 +21,11 @@ import { TestSessionQueryService } from './session-query.ts' interface FakeAgent extends Agent { status: AgentStatus sent: ContentBlock[][] - sentOptions: (SendOptions | undefined)[] + sentOptions: (SendOptions | AliasSendOptions | undefined)[] steered: ContentBlock[][] - steeredOptions: (SendOptions | undefined)[] + steeredOptions: (AliasSendOptions | undefined)[] injected: ContentBlock[][] - injectedOptions: (SendOptions | undefined)[] + injectedOptions: (AliasSendOptions | undefined)[] cancelled: AgentCancelCause[] } @@ -160,10 +161,10 @@ export async function createTuiTestHarness { beforeMount(session) { session.append('user/message', { content: renderGoalChange(change), - source: { kind: 'goal', goalId: change.goal.id, revision: change.goal.revision, round: 0 }, - meta: change as unknown as JsonValue, + source: { + kind: 'goal', + goalId: change.goal.id, + revision: change.goal.revision, + round: 0, + change, + }, }, { surfaceOp: 'append' }) }, }) @@ -1812,13 +1817,6 @@ describe('pi-tui chat lifecycle and transcript', () => { await ctrlCExit.controller.dispose() await ctrlCExit.ctx.fiber.dispose() - const disposedAgent = await setup() - disposedAgent.agent.status = 'disposed' - disposedAgent.terminal.send('late input') - disposedAgent.terminal.send('\r') - await tick() - expect(disposedAgent.terminal.output).toContain('is disposed') - await dispose(disposedAgent) }) it('combines session autocomplete with files and prepares send/steer references asynchronously', async () => { @@ -2338,7 +2336,7 @@ describe('pi-tui chat lifecycle and transcript', () => { expect(assembly.variables).toMatchObject({ provider: 'beta', model: 'b1' }) const seed: LlmCallConfig = { provider: 'alpha', model: 'a1', temperature: 0.2 } const request = await agentEvents(result.ctx, result.agent).waterfall( - 'agent/request', 1, 0, seed, new AbortController().signal, () => Promise.resolve(seed), + 'agent/request', 1, 0, new AbortController().signal, () => Promise.resolve(seed), ) expect(request).toEqual({ provider: 'beta', model: 'b1', temperature: 0.2 }) await dispose(result) @@ -2391,7 +2389,7 @@ describe('pi-tui chat lifecycle and transcript', () => { expect(assembly.variables).toEqual({}) const seed: LlmCallConfig = { provider: 'fallback', model: 'fallback' } await expect(agentEvents(empty.ctx, empty.agent).waterfall( - 'agent/request', 1, 0, seed, new AbortController().signal, () => Promise.resolve(seed), + 'agent/request', 1, 0, new AbortController().signal, () => Promise.resolve(seed), )).resolves.toBe(seed) await dispose(empty) @@ -3272,7 +3270,7 @@ describe('terminal mounting', () => { const session = ctx.sessions.create(SessionId('main')) ctx.agents.register({ id: session.id, options: {}, session, status: 'idle', ctx, - followup: () => AgentMessageId('stub'), queue: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject: () => AgentMessageId('stub'), send: () => AgentMessageId('stub'), cancel() {}, whenIdle: () => Promise.resolve(), + followup: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject: () => AgentMessageId('stub'), send: () => AgentMessageId('stub'), cancel() {}, retry() {}, whenIdle: () => Promise.resolve(), }) const terminal = new FakeTerminal() mountTui(ctx, { color: false }, { terminal, exit: vi.fn() }) @@ -3296,7 +3294,7 @@ describe('terminal mounting', () => { const session = ctx.sessions.create(SessionId('main')) ctx.agents.register({ id: session.id, options: {}, session, status: 'idle', ctx, - followup: () => AgentMessageId('stub'), queue: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject: () => AgentMessageId('stub'), send: () => AgentMessageId('stub'), cancel() {}, whenIdle: () => Promise.resolve(), + followup: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject: () => AgentMessageId('stub'), send: () => AgentMessageId('stub'), cancel() {}, retry() {}, whenIdle: () => Promise.resolve(), }) const terminal = new FakeTerminal() // Mirror dsh-tui's own inject (minus loader, the absence under test). @@ -3330,14 +3328,14 @@ describe('terminal mounting', () => { const otherSession = ctx.sessions.create(SessionId('other-session')) ctx.agents.register({ id: otherSession.id, options: {}, session: otherSession, status: 'idle', ctx, - followup: () => AgentMessageId('stub'), queue: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject: () => AgentMessageId('stub'), send: () => AgentMessageId('stub'), cancel() {}, whenIdle: () => Promise.resolve(), + followup: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject: () => AgentMessageId('stub'), send: () => AgentMessageId('stub'), cancel() {}, retry() {}, whenIdle: () => Promise.resolve(), }) expect(terminal.started).toBe(0) const session = ctx.sessions.create(SessionId('late-session')) const agent = { id: session.id, options: {}, session, status: 'idle', ctx, - followup: () => AgentMessageId('stub'), queue: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject: () => AgentMessageId('stub'), send: () => AgentMessageId('stub'), cancel() {}, whenIdle: () => Promise.resolve(), + followup: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject: () => AgentMessageId('stub'), send: () => AgentMessageId('stub'), cancel() {}, retry() {}, whenIdle: () => Promise.resolve(), } as Agent ctx.agents.register(agent) await tick() @@ -3367,7 +3365,7 @@ describe('terminal mounting', () => { const session = ctx.sessions.create(SessionId('main-session')) ctx.agents.register({ id: session.id, options: {}, session, status: 'idle', ctx, - followup: () => AgentMessageId('stub'), queue: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject: () => AgentMessageId('stub'), send: () => AgentMessageId('stub'), cancel() {}, whenIdle: () => Promise.resolve(), + followup: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject: () => AgentMessageId('stub'), send: () => AgentMessageId('stub'), cancel() {}, retry() {}, whenIdle: () => Promise.resolve(), }) await tick() expect(terminal.started).toBe(0) @@ -3409,7 +3407,7 @@ describe('terminal mounting', () => { session.append('step/start', { turn: 1, step: 1 }) ctx.agents.register({ id: session.id, options: {}, session, status: 'running', ctx, - followup: () => AgentMessageId('stub'), queue: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject: () => AgentMessageId('stub'), send: () => AgentMessageId('stub'), cancel() {}, whenIdle: () => Promise.resolve(), + followup: () => AgentMessageId('stub'), steer: () => AgentMessageId('stub'), inject: () => AgentMessageId('stub'), send: () => AgentMessageId('stub'), cancel() {}, retry() {}, whenIdle: () => Promise.resolve(), }) const terminal = new FakeTerminal() terminal.start = () => { throw new Error('terminal startup failed') } diff --git a/packages/ui/user-approval/tests/approval.spec.ts b/packages/ui/user-approval/tests/approval.spec.ts index fe2643cb3f..0df183ca6a 100644 --- a/packages/ui/user-approval/tests/approval.spec.ts +++ b/packages/ui/user-approval/tests/approval.spec.ts @@ -371,7 +371,7 @@ describe('approval policy (the approval/policy fold)', () => { } const preStep = (ctx: Context, agent: Agent): Promise => - agentEvents(ctx, agent).serial('agent/pre-step', 1, 1, new AbortController().signal) + agentEvents(ctx, agent).serial('agent/step', 1, 1, new AbortController().signal) /** Append a `request/header` snapshot whose system text is exactly `system`. */ function appendHeader(session: Session, system: string): void { diff --git a/scripts/gen-cordis-catalog.ts b/scripts/gen-cordis-catalog.ts index 0e31d7041d..19000d684e 100644 --- a/scripts/gen-cordis-catalog.ts +++ b/scripts/gen-cordis-catalog.ts @@ -48,7 +48,6 @@ export const LINK_MAP: Record = { MessageSource: 'core.md', PromptDecision: 'core.md', RequestError: 'core.md', - RequestErrorDecision: 'core.md', PreparedReferencedMessage: 'session-reference.md', SessionReferenceCandidate: 'session-reference.md', SessionReferenceInput: 'session-reference.md', diff --git a/scripts/gen-doc-graphs.ts b/scripts/gen-doc-graphs.ts index 1f54f7358f..79b7d00db6 100644 --- a/scripts/gen-doc-graphs.ts +++ b/scripts/gen-doc-graphs.ts @@ -646,22 +646,31 @@ class EventRelationCollector { /** Walk one package source file and classify event API calls by receiver type. */ private visitSource(source: PackageSource): void { const visit = (node: ts.Node): void => { - if (ts.isCallExpression(node) && ts.isPropertyAccessExpression(node.expression)) { - const receiverKind = this.receiverKind(node.expression.expression) - const method = node.expression.name.text - if (receiverKind === 'events-service' && method === 'dispatch') { - const argumentList = node.arguments[1] - if (argumentList) { - for (const event of this.eventNamesFromArgumentList(argumentList, new Set())) { - this.addDispatcher(event, source.pkg, 'events.dispatch') + if (ts.isCallExpression(node)) { + if (this.isAgentEventEmitter(node.expression)) { + const event = node.arguments[2] + if (event) { + for (const name of this.finiteStringValues(event) ?? []) { + this.addDispatcher(name, source.pkg, 'emitAgentEvent') } } - } else if (receiverKind === 'context' || receiverKind === 'agent-dispatch') { - const eventNames = this.eventNamesFromCall(node, receiverKind) - if (method === 'on' || method === 'once') { - for (const event of eventNames) this.ensure(event).listeners.add(source.pkg) - } else if (method === 'emit' || method === 'parallel' || method === 'serial' || method === 'waterfall') { - for (const event of eventNames) this.addDispatcher(event, source.pkg, method) + } else if (ts.isPropertyAccessExpression(node.expression)) { + const receiverKind = this.receiverKind(node.expression.expression) + const method = node.expression.name.text + if (receiverKind === 'events-service' && method === 'dispatch') { + const argumentList = node.arguments[1] + if (argumentList) { + for (const event of this.eventNamesFromArgumentList(argumentList, new Set())) { + this.addDispatcher(event, source.pkg, 'events.dispatch') + } + } + } else if (receiverKind === 'context' || receiverKind === 'agent-dispatch') { + const eventNames = this.eventNamesFromCall(node, receiverKind) + if (method === 'on' || method === 'once') { + for (const event of eventNames) this.ensure(event).listeners.add(source.pkg) + } else if (method === 'emit' || method === 'parallel' || method === 'serial' || method === 'waterfall') { + for (const event of eventNames) this.addDispatcher(event, source.pkg, method) + } } } } @@ -670,6 +679,22 @@ class EventRelationCollector { visit(source.sourceFile) } + /** Match the exported contained-notification helper by declaration identity. */ + private isAgentEventEmitter(expression: ts.Expression): boolean { + if (!ts.isIdentifier(expression)) return false + const local = this.project.checker.getSymbolAtLocation(expression) + if (!local) return false + const symbol = local.flags & ts.SymbolFlags.Alias + ? this.project.checker.getAliasedSymbol(local) + : local + const declarations = symbol.declarations ?? [] + return declarations.some((declaration) => { + return ts.isFunctionDeclaration(declaration) + && declaration.name?.text === 'emitAgentEvent' + && this.project.relativePath(declaration.getSourceFile()) === 'packages/core/agent/src/dispatch.ts' + }) + } + /** Classify a receiver using assignability to the repository's actual event API types. */ private receiverKind(receiver: ts.Expression): EventReceiverKind | undefined { const type = this.project.checker.getTypeAtLocation(receiver) @@ -916,7 +941,6 @@ function renderLifecycle(): string { ' participant LLM as ctx.llm', ' participant Tools as ctx.tools', ' participant Session', - ' participant Persistence', ' participant SDK as UI or SDK listener', ' User->>Agent: followup(content)', ` Agent-->>SDK: ${mermaidCode('agent/inbox/enqueue')}`, @@ -927,7 +951,7 @@ function renderLifecycle(): string { ' Hooks-->>Driver: authoritative allow, block, or add context', ` Driver->>Session: ${mermaidCode('user/message')} or rejected ${mermaidCode('turn/end')}`, ` Driver->>Prompt: ${mermaidCode('system-prompt/assemble')} waterfall`, - ` Driver-->>Driver: ${mermaidCode('agent/pre-step')} serial checkpoint`, + ` Driver-->>Driver: ${mermaidCode('agent/step')} serial checkpoint`, ` Driver->>Session: ${mermaidCode('step/start')}`, ` Driver->>LLM: ${mermaidCode('agent/request')} waterfall, then ${mermaidCode('llm/stream')} waterfall`, ' LLM-->>Driver: StreamChunk*', @@ -936,9 +960,8 @@ function renderLifecycle(): string { ' alt final adapter or terminal in-band request failure', ` Driver->>Session: ${mermaidCode('step/end')}`, ` Driver->>Hooks: ${mermaidCode('agent/request-error')} waterfall`, - ' Hooks-->>Driver: retry in a new step or preserve the original error', + ' Hooks-->>Driver: call agent.retry() or preserve the original error', ' else model request succeeded', - ` Driver->>Hooks: ${mermaidCode('agent/step-result')} waterfall`, ` Driver->>Session: ${mermaidCode('assistant/message')}`, ' Driver->>Tools: classify pending call by executionMode', ' loop barriers and bounded rolling pool, reclassify before start', @@ -953,19 +976,16 @@ function renderLifecycle(): string { ' end', ' end', ' Driver->>Session: post-tool context and steering (no prompt-submit)', - ` Driver->>Hooks: ${mermaidCode('agent/post-step')} serial checkpoint`, ` Driver->>Session: ${mermaidCode('step/end')}`, - ` Driver->>Hooks: ${mermaidCode('agent/turn-continuation')} waterfall`, - ` Driver->>Hooks: ${mermaidCode('agent/turn-stop')} serial terminal checkpoint`, + ` Driver->>Hooks: ${mermaidCode('agent/stopping')} serial terminal checkpoint`, ' end', ` Driver->>Session: ${mermaidCode('turn/end')}`, - ` Driver->>Persistence: ${mermaidCode('session/flush')} parallel checkpoint`, ` Driver-->>SDK: ${mermaidCode('agent/status')} idle`, '```', '', 'The `assistant/message` edge records every successful provider call, including content-less and `max-tokens` finishes. Empty content stays out of derived history while the durable anchor retains usage and exact chunk provenance, including an explicit empty source set.', '', - '`dsh-compact-basic` uses `agent/post-step` for pressure after those durable facts and `agent/request-error` only for canonical context overflow. Once either trigger qualifies, optional tool-result pruning runs before summary selection. Recovery works between the closed failed step and a fresh retry step, and returns retry only when pruning or summarization advances the surface replacement generation; otherwise the original request error remains authoritative.', + '`dsh-compact-basic` uses `agent/step` for pressure before request derivation and `agent/request-error` only for canonical context overflow. Once either trigger qualifies, optional tool-result pruning runs before summary selection. Recovery works between the closed failed step and failed turn close, and opens a fresh retry turn only when pruning or summarization advances the surface replacement generation; otherwise the original request error remains authoritative.', '', 'The returned `agent/prompt-submit` allow is authoritative; listeners wrapping `next()` preserve downstream content and additional contexts unless replacement is intentional. Steering bypasses that waterfall and joins at its durable checkpoint.', '', diff --git a/scripts/type-equiv.manifest.json b/scripts/type-equiv.manifest.json index 6ffea6b15a..664cbdb1db 100644 --- a/scripts/type-equiv.manifest.json +++ b/scripts/type-equiv.manifest.json @@ -6,119 +6,229 @@ "symbol": "Branded", "source": "packages/util/brand/src/index.ts" }, + { + "doc": "docs/core-data-structures/core.zh.md", + "symbol": "Branded", + "source": "packages/util/brand/src/index.ts" + }, { "doc": "docs/core-data-structures/core.md", "symbol": "ContentBlockMap", "source": "packages/llm/llm/src/types.ts" }, + { + "doc": "docs/core-data-structures/core.zh.md", + "symbol": "ContentBlockMap", + "source": "packages/llm/llm/src/types.ts" + }, { "doc": "docs/core-data-structures/core.md", "symbol": "AssistantProvenance", "source": "packages/llm/llm/src/types.ts" }, + { + "doc": "docs/core-data-structures/core.zh.md", + "symbol": "AssistantProvenance", + "source": "packages/llm/llm/src/types.ts" + }, { "doc": "docs/core-data-structures/core.md", "symbol": "Message", "source": "packages/llm/llm/src/types.ts" }, + { + "doc": "docs/core-data-structures/core.zh.md", + "symbol": "Message", + "source": "packages/llm/llm/src/types.ts" + }, { "doc": "docs/core-data-structures/core.md", "symbol": "MessageSourceMap", "source": "packages/llm/llm/src/types.ts" }, + { + "doc": "docs/core-data-structures/core.zh.md", + "symbol": "MessageSourceMap", + "source": "packages/llm/llm/src/types.ts" + }, { "doc": "docs/core-data-structures/core.md", "symbol": "FinishReasonMap", "source": "packages/llm/llm/src/types.ts" }, + { + "doc": "docs/core-data-structures/core.zh.md", + "symbol": "FinishReasonMap", + "source": "packages/llm/llm/src/types.ts" + }, { "doc": "docs/core-data-structures/core.md", "symbol": "LlmProviderInfo", "source": "packages/llm/llm/src/types.ts" }, + { + "doc": "docs/core-data-structures/core.zh.md", + "symbol": "LlmProviderInfo", + "source": "packages/llm/llm/src/types.ts" + }, { "doc": "docs/core-data-structures/core.md", "symbol": "LlmModelInfo", "source": "packages/llm/llm/src/types.ts" }, + { + "doc": "docs/core-data-structures/core.zh.md", + "symbol": "LlmModelInfo", + "source": "packages/llm/llm/src/types.ts" + }, { "doc": "docs/core-data-structures/core.md", "symbol": "LlmModelContext", "source": "packages/llm/llm/src/types.ts" }, + { + "doc": "docs/core-data-structures/core.zh.md", + "symbol": "LlmModelContext", + "source": "packages/llm/llm/src/types.ts" + }, { "doc": "docs/core-data-structures/core.md", "symbol": "GenerateOptions", "source": "packages/llm/llm/src/types.ts" }, + { + "doc": "docs/core-data-structures/core.zh.md", + "symbol": "GenerateOptions", + "source": "packages/llm/llm/src/types.ts" + }, { "doc": "docs/core-data-structures/core.md", "symbol": "ToolSchema", "source": "packages/llm/llm/src/types.ts" }, + { + "doc": "docs/core-data-structures/core.zh.md", + "symbol": "ToolSchema", + "source": "packages/llm/llm/src/types.ts" + }, { "doc": "docs/core-data-structures/core.md", "symbol": "LlmCallConfig", "source": "packages/llm/llm/src/call-config.ts" }, + { + "doc": "docs/core-data-structures/core.zh.md", + "symbol": "LlmCallConfig", + "source": "packages/llm/llm/src/call-config.ts" + }, { "doc": "docs/core-data-structures/core.md", "symbol": "SessionEvent", "source": "packages/core/session/src/types.ts" }, + { + "doc": "docs/core-data-structures/core.zh.md", + "symbol": "SessionEvent", + "source": "packages/core/session/src/types.ts" + }, { "doc": "docs/core-data-structures/core.md", "symbol": "SendTarget", "source": "packages/core/agent/src/types.ts" }, + { + "doc": "docs/core-data-structures/core.zh.md", + "symbol": "SendTarget", + "source": "packages/core/agent/src/types.ts" + }, { "doc": "docs/core-data-structures/core.md", "symbol": "SendOptions", "source": "packages/core/agent/src/types.ts" }, + { + "doc": "docs/core-data-structures/core.zh.md", + "symbol": "SendOptions", + "source": "packages/core/agent/src/types.ts" + }, { "doc": "docs/core-data-structures/core.md", "symbol": "AliasSendOptions", "source": "packages/core/agent/src/types.ts" }, + { + "doc": "docs/core-data-structures/core.zh.md", + "symbol": "AliasSendOptions", + "source": "packages/core/agent/src/types.ts" + }, { "doc": "docs/core-data-structures/core.md", "symbol": "AgentMessageId", "source": "packages/core/agent/src/types.ts" }, + { + "doc": "docs/core-data-structures/core.zh.md", + "symbol": "AgentMessageId", + "source": "packages/core/agent/src/types.ts" + }, { "doc": "docs/core-data-structures/core.md", "symbol": "AgentMessage", "source": "packages/core/agent/src/types.ts" }, + { + "doc": "docs/core-data-structures/core.zh.md", + "symbol": "AgentMessage", + "source": "packages/core/agent/src/types.ts" + }, { "doc": "docs/core-data-structures/core.md", "symbol": "CancelOptions", "source": "packages/core/agent/src/types.ts" }, + { + "doc": "docs/core-data-structures/core.zh.md", + "symbol": "CancelOptions", + "source": "packages/core/agent/src/types.ts" + }, { "doc": "docs/core-data-structures/core.md", "symbol": "AgentCancelCause", "source": "packages/core/agent/src/types.ts" }, + { + "doc": "docs/core-data-structures/core.zh.md", + "symbol": "AgentCancelCause", + "source": "packages/core/agent/src/types.ts" + }, { "doc": "docs/core-data-structures/core.md", "symbol": "Agent", "source": "packages/core/agent/src/types.ts" }, + { + "doc": "docs/core-data-structures/core.zh.md", + "symbol": "Agent", + "source": "packages/core/agent/src/types.ts" + }, { "doc": "docs/core-data-structures/core.md", "symbol": "AdditionalContext", "source": "packages/core/agent/src/types.ts" }, + { + "doc": "docs/core-data-structures/core.zh.md", + "symbol": "AdditionalContext", + "source": "packages/core/agent/src/types.ts" + }, { "doc": "docs/core-data-structures/core.md", "symbol": "PromptDecision", "source": "packages/core/agent/src/types.ts" }, { - "doc": "docs/core-data-structures/core.md", - "symbol": "ContinuationDecision", + "doc": "docs/core-data-structures/core.zh.md", + "symbol": "PromptDecision", "source": "packages/core/agent/src/types.ts" }, { @@ -127,17 +237,17 @@ "source": "packages/core/agent/src/types.ts" }, { - "doc": "docs/core-data-structures/core.md", - "symbol": "RequestErrorDecision", + "doc": "docs/core-data-structures/core.zh.md", + "symbol": "RequestError", "source": "packages/core/agent/src/types.ts" }, { "doc": "docs/core-data-structures/core.md", - "symbol": "ContinuationStop", + "symbol": "SessionStartSource", "source": "packages/core/agent/src/types.ts" }, { - "doc": "docs/core-data-structures/core.md", + "doc": "docs/core-data-structures/core.zh.md", "symbol": "SessionStartSource", "source": "packages/core/agent/src/types.ts" }, @@ -313,77 +423,153 @@ "symbol": "UserMessageData", "source": "packages/core/session/src/types.ts" }, + { + "doc": "docs/core-data-structures/session.zh.md", + "symbol": "UserMessageData", + "source": "packages/core/session/src/types.ts" + }, { "doc": "docs/core-data-structures/session.md", "symbol": "SessionEventMap", "source": "packages/core/session/src/types.ts" }, + { + "doc": "docs/core-data-structures/session.zh.md", + "symbol": "SessionEventMap", + "source": "packages/core/session/src/types.ts" + }, { "doc": "docs/core-data-structures/session.md", "symbol": "OutOfBandSessionEventMap", "source": "packages/core/session/src/types.ts" }, + { + "doc": "docs/core-data-structures/session.zh.md", + "symbol": "OutOfBandSessionEventMap", + "source": "packages/core/session/src/types.ts" + }, { "doc": "docs/core-data-structures/session.md", "symbol": "EpochHeader", "source": "packages/core/session/src/types.ts" }, + { + "doc": "docs/core-data-structures/session.zh.md", + "symbol": "EpochHeader", + "source": "packages/core/session/src/types.ts" + }, { "doc": "docs/core-data-structures/session.md", "symbol": "TodoItem", "source": "packages/core/session/src/types.ts" }, + { + "doc": "docs/core-data-structures/session.zh.md", + "symbol": "TodoItem", + "source": "packages/core/session/src/types.ts" + }, { "doc": "docs/core-data-structures/session.md", "symbol": "SessionEvent", "source": "packages/core/session/src/types.ts" }, + { + "doc": "docs/core-data-structures/session.zh.md", + "symbol": "SessionEvent", + "source": "packages/core/session/src/types.ts" + }, { "doc": "docs/core-data-structures/session.md", "symbol": "TurnTriggerMap", "source": "packages/core/session/src/types.ts" }, + { + "doc": "docs/core-data-structures/session.zh.md", + "symbol": "TurnTriggerMap", + "source": "packages/core/session/src/types.ts" + }, { "doc": "docs/core-data-structures/session.md", "symbol": "TurnEndReasonMap", "source": "packages/core/session/src/types.ts" }, + { + "doc": "docs/core-data-structures/session.zh.md", + "symbol": "TurnEndReasonMap", + "source": "packages/core/session/src/types.ts" + }, { "doc": "docs/core-data-structures/session.md", "symbol": "SurfaceEventType", "source": "packages/core/session/src/types.ts" }, + { + "doc": "docs/core-data-structures/session.zh.md", + "symbol": "SurfaceEventType", + "source": "packages/core/session/src/types.ts" + }, { "doc": "docs/core-data-structures/session.md", "symbol": "SurfaceOp", "source": "packages/core/session/src/types.ts" }, + { + "doc": "docs/core-data-structures/session.zh.md", + "symbol": "SurfaceOp", + "source": "packages/core/session/src/types.ts" + }, { "doc": "docs/core-data-structures/session.md", "symbol": "SurfaceIntent", "source": "packages/core/session/src/types.ts" }, + { + "doc": "docs/core-data-structures/session.zh.md", + "symbol": "SurfaceIntent", + "source": "packages/core/session/src/types.ts" + }, { "doc": "docs/core-data-structures/session.md", "symbol": "SessionSurface", "source": "packages/core/session/src/surface.ts" }, + { + "doc": "docs/core-data-structures/session.zh.md", + "symbol": "SessionSurface", + "source": "packages/core/session/src/surface.ts" + }, { "doc": "docs/core-data-structures/session.md", "symbol": "SurfaceFoldReplacement", "source": "packages/core/session/src/surface.ts" }, + { + "doc": "docs/core-data-structures/session.zh.md", + "symbol": "SurfaceFoldReplacement", + "source": "packages/core/session/src/surface.ts" + }, { "doc": "docs/core-data-structures/session.md", "symbol": "SurfaceFoldResult", "source": "packages/core/session/src/surface.ts" }, + { + "doc": "docs/core-data-structures/session.zh.md", + "symbol": "SurfaceFoldResult", + "source": "packages/core/session/src/surface.ts" + }, { "doc": "docs/core-data-structures/session.md", "symbol": "Session", "source": "packages/core/session/src/index.ts", "projection": "public-api" }, + { + "doc": "docs/core-data-structures/session.zh.md", + "symbol": "Session", + "source": "packages/core/session/src/index.ts", + "projection": "public-api" + }, { "doc": "docs/core-data-structures/persistence.md", "symbol": "SessionHeader",