From 9ca0241d5d113d4e6d575219df4550f9c0f39ddb Mon Sep 17 00:00:00 2001 From: Hypatia May Date: Tue, 28 Jul 2026 12:10:48 +0800 Subject: [PATCH 01/38] feat(web): add durable session metrics (round 1) --- ...8-host-owned-web-session-metrics.i18n.yaml | 6 + ...26-07-28-host-owned-web-session-metrics.md | 35 +++ ...07-28-host-owned-web-session-metrics.zh.md | 35 +++ .../snapshots/fresh-round-trip/ui.expected.md | 16 +- docs/event-producer-consumer.md | 2 +- packages/client/connection/src/client/api.ts | 2 +- .../client/connection/src/client/index.ts | 2 +- packages/client/runtime/README.i18n.yaml | 4 +- packages/client/runtime/README.md | 2 +- packages/client/runtime/README.zh.md | 2 +- .../src/client/sessions/conversation.ts | 8 +- .../runtime/src/client/sessions/session.ts | 54 +++- packages/client/runtime/tests/fake-api.ts | 9 +- .../client/runtime/tests/queue-store.spec.ts | 7 + packages/client/runtime/tests/session.spec.ts | 112 ++++++- .../client/ui-conversation/README.i18n.yaml | 4 +- packages/client/ui-conversation/README.md | 2 + packages/client/ui-conversation/README.zh.md | 2 + .../src/client/chat/StatsLine.tsx | 104 ++++--- .../tests/chat-branch-tails.spec.tsx | 14 +- .../tests/chat-code-subcalls.spec.tsx | 2 +- .../tests/chat-stats-bash-sample.spec.tsx | 99 ++++++- .../tests/chat-toolview-slot.spec.tsx | 2 +- .../ui-conversation/tests/chat-view.spec.tsx | 2 +- .../tests/gate-branch-tails.spec.tsx | 16 +- .../ui-conversation/tests/input-bar.spec.tsx | 2 +- .../tests/input-matrix.spec.tsx | 2 +- .../tests/input-scenarios.spec.tsx | 2 +- .../ui-conversation/tests/queue-dock.spec.tsx | 2 +- .../ui-conversation/tests/skeleton.spec.tsx | 2 +- packages/host/apiproxy/README.i18n.yaml | 4 +- packages/host/apiproxy/README.md | 2 +- packages/host/apiproxy/README.zh.md | 2 +- packages/host/apiproxy/src/api-proxy.ts | 74 ++++- .../host/apiproxy/src/api/events.schema.ts | 5 +- packages/host/apiproxy/src/api/events.ts | 2 + packages/host/apiproxy/src/api/index.ts | 2 +- .../host/apiproxy/src/api/sessions.schema.ts | 15 +- packages/host/apiproxy/src/api/sessions.ts | 29 +- packages/host/apiproxy/src/session-metrics.ts | 194 ++++++++++++ .../apiproxy/tests/api-proxy-models.spec.ts | 34 +++ .../host/apiproxy/tests/rpc-schemas.spec.ts | 43 ++- .../apiproxy/tests/session-metrics.spec.ts | 277 ++++++++++++++++++ 43 files changed, 1139 insertions(+), 97 deletions(-) create mode 100644 .agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.i18n.yaml create mode 100644 .agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.md create mode 100644 .agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.zh.md create mode 100644 packages/host/apiproxy/src/session-metrics.ts create mode 100644 packages/host/apiproxy/tests/session-metrics.spec.ts diff --git a/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.i18n.yaml b/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.i18n.yaml new file mode 100644 index 0000000000..95add14bf5 --- /dev/null +++ b/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.i18n.yaml @@ -0,0 +1,6 @@ +# Bilingual-pair consistency record (docs/i18n/README.md): the git blob hash of each +# side as of the last confirmed-consistent state. Both languages carry equal authority; +# after editing either side, bring the other along and re-record with: +# pnpm run verify-translation-pairing --write .agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.md +2026-07-28-host-owned-web-session-metrics.md: 04381c7443491fd9de101a87713fa5c183800d0e +2026-07-28-host-owned-web-session-metrics.zh.md: 6ad06ea61508d3f0703c19bc7c9f14969f119334 diff --git a/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.md b/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.md new file mode 100644 index 0000000000..04381c7443 --- /dev/null +++ b/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.md @@ -0,0 +1,35 @@ +# Agent Note: Host-owned Web session metrics + +Status: implemented + +English | [中文](2026-07-28-host-owned-web-session-metrics.zh.md) + +## Problem + +A Web stats line derived from the currently loaded conversation nodes is window-dependent under pagination. Compaction can replace visible content without preserving historical usage, and route changes leave the browser without an authoritative context capacity. Cache-write tokens also risk being folded into a cache-hit formula whose denominator has different semantics. + +## Decision + +The Host owns one session-level metrics projection. It incrementally folds the complete durable event log, keys settled usage by `(turn, step)`, and replaces an earlier usage record for the same key instead of double-counting chunk and message forms. Uncached input, output, cache reads, and cache writes remain four disjoint cumulative buckets. Compaction can change the current prompt surface without erasing historical usage. + +Current context pressure is a separate point-in-time value from `tokenMeter.measure(session).totalTokens`. Capacity comes only from `llm.resolveModelInfo(provider, model).context.contextWindow` for the agent's selected route. A route change immediately publishes metrics with capacity absent, then publishes the resolved capacity behind a route generation fence; stale metadata cannot label the new route. + +The tail `session.history` response carries the projection, while older pages omit it. Live changes use `session/metrics` mux frames. Both forms carry a durable-log revision and a projection revision; the client accepts only nondecreasing revisions, preserves metrics across older-page prepend, and clears them at a new subscription baseline. Missing measurement or metadata stays absent. + +The Web stats line treats the projection as its sole token source. It renders uncached input, output, and cache reads separately, computes cache hit as `cacheRead / (uncachedInput + cacheRead)`, and shows current context as a percentage of the exact route capacity. Cache writes never enter that percentage. Visible nodes continue to supply only turn and step counts. + +## Alternatives considered + +**Fold the loaded node window in React.** This cannot survive pagination or compaction and duplicates durable-log semantics in a presentation package. + +**Send usage only with raw assistant events.** Reconnect and older-page stitching would still need the client to reconstruct a full-log aggregate, and duplicate usage forms would need protocol-specific repair there. + +**Reuse one total-token field for cache hit.** Cache reads, cache writes, and uncached input represent distinct provider accounting buckets; combining them would make the displayed rate misleading. + +**Keep the previous capacity until the new route resolves.** The old number would temporarily claim the wrong selected model. An explicit unknown state is honest and generation-safe. + +## Consequences + +Token totals remain stable across pagination, replay, compaction, and browser reconnect. The client stores a small detached projection instead of scanning the conversation window, and the status row remains readable for large histories through compact number formatting. + +The Host performs one incremental log fold per session and schedules live projection updates only for usage, request-header, or surface-changing events; text and reasoning deltas do not publish metrics. Exact capacity resolution is asynchronous and may briefly render as unknown. Deployments without a token meter or model context metadata retain the row and label the unavailable value instead of fabricating one. diff --git a/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.zh.md b/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.zh.md new file mode 100644 index 0000000000..6ad06ea615 --- /dev/null +++ b/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.zh.md @@ -0,0 +1,35 @@ +# Agent Note: Host 拥有的 Web 会话指标 + +Status: implemented + +[English](2026-07-28-host-owned-web-session-metrics.md) | 中文 + +## 问题 + +Web 统计行若根据当前加载的会话节点推导指标,其结果会随分页窗口变化。压缩(compaction)可以替换可见内容,却无法保留历史用量;路由变更会让浏览器缺少权威的上下文容量。缓存写入 token 还可能被计入缓存命中率公式,而该公式的分母具有不同语义。 + +## 决策 + +Host 拥有一项会话级指标投影。它以增量方式归并完整的持久事件日志,按 `(turn, step)` 标识已结算用量;同一标识再次出现时,会替换较早的用量记录,而不会重复统计分片和消息两种形态。未缓存输入、输出、缓存读取与缓存写入保持为四个彼此独立的累计计数项。压缩可以改变当前提示词表层,但不会抹除历史用量。 + +当前上下文压力是一个独立的即时值,取自 `tokenMeter.measure(session).totalTokens`。容量仅来自 `llm.resolveModelInfo(provider, model).context.contextWindow`,并对应 agent(智能体)所选的路由。路由变更时,Host 会立即发布不带容量的指标,再通过路由代际围栏发布解析出的容量;陈旧元数据无法标记新的路由。 + +`session.history` 尾页响应携带该投影,较早页面则省略它。实时变更使用 `session/metrics` mux 帧。两种形式都携带持久日志修订号和投影修订号;客户端只接受不减小的修订号,在向前加载较早页面时保留指标,并在建立新的订阅基线时将其清除。测量值或元数据缺失时,对应字段保持缺失。 + +Web 统计行把该投影视为唯一的 token 数据来源。它分别呈现未缓存输入、输出与缓存读取,通过 `cacheRead / (uncachedInput + cacheRead)` 计算缓存命中率,并把当前上下文显示为精确路由容量的百分比。缓存写入绝不计入缓存命中率。可见节点仍然只提供轮次和步骤计数。 + +## 备选方案 + +**在 React 中归并已加载的节点窗口。** 此方案无法跨越分页或压缩保留数据,还会在展示包中重复实现持久日志语义。 + +**只随原始 assistant 事件发送用量。** 重连和较早页面拼接仍会要求客户端重建完整日志聚合,而且重复的用量形态需要在客户端按协议专门修复。 + +**为缓存命中率复用单一的 token 总数字段。** 缓存读取、缓存写入与未缓存输入是提供方记账中的不同计数项;将它们合并会使显示的比率产生误导。 + +**在新路由解析完成前保留旧容量。** 旧数值会在短时间内错误标示所选模型。显式的「未知」状态能如实反映情况,并避免跨代串扰。 + +## 后果 + +token 总量在分页、回放、压缩和浏览器重连期间保持稳定。客户端存储一项小型脱耦投影,无需扫描会话窗口;状态行采用紧凑数字格式,因此在较长的历史记录中仍然清晰易读。 + +Host 为每个会话执行一次增量日志归并,仅为用量事件、请求头事件或表层变更事件调度实时投影更新;文本与推理(reasoning)增量不会发布指标。精确容量解析为异步操作,因此可能短暂显示「未知」。未部署 token 计量器或缺少模型上下文元数据时,系统仍保留该行,并标示不可用的值,而不会虚构数据。 diff --git a/apps/web/tests/snapshots/fresh-round-trip/ui.expected.md b/apps/web/tests/snapshots/fresh-round-trip/ui.expected.md index dc572023fd..29d039fd53 100644 --- a/apps/web/tests/snapshots/fresh-round-trip/ui.expected.md +++ b/apps/web/tests/snapshots/fresh-round-trip/ui.expected.md @@ -7,22 +7,30 @@ - tab "Trajectory" - tab "Waterfall" - text: "Use the bash tool to run exactly: echo WEB_E2E_OK. Then reply with the single word DONE and stop." +- button "复制": + - img +- button "在新对话中分支": + - img +- button "编辑": + - img +- button "▸ 上下文注入" - button "Think The user wants me to run a simple bash command and reply with \"DONE\".": - img - text: Think The user wants me to run a simple bash command and reply with "DONE". -- text: Echo the test string +- img +- text: Bash Echo the test string - button "Think The command executed successfully and output \"WEB_E2E_OK\". I just need to reply with \"DONE\".": - img - text: Think The command executed successfully and output "WEB_E2E_OK". I just need to reply with "DONE". - paragraph: DONE -- text: cache hit 99% · 15,818 tokens · 1 turns · 2 steps +- text: 219 uncached input · 111 output · 15.5k cache read · cache hit 99% · context 6% of 128k · 1 turns · 2 steps - textbox "Message the agent" - button "Add attachment": - img - combobox "Access mode": - option "Read-only" [selected] - option "Read-write" -- button "选择模型,当前 deepseek-v4-flash": - - text: deepseek-v4-flash +- button "选择模型,当前 DeepSeek-V4-Flash": + - text: DeepSeek-V4-Flash - img - button "Send message" [disabled] diff --git a/docs/event-producer-consumer.md b/docs/event-producer-consumer.md index 18ca96e1b3..122935aea9 100644 --- a/docs/event-producer-consumer.md +++ b/docs/event-producer-consumer.md @@ -9,7 +9,7 @@ This matrix shows which packages dispatch each harness-owned event and which pac | --- | --- | --- | --- | --- | | `agent-loop/config-start-failed` | `emit` | [`packages/core/agent-loop/src/index.ts:140`](../packages/core/agent-loop/src/index.ts) | [`agent-loop`](../packages/core/agent-loop) (`events.dispatch`) | [`tui`](../packages/ui/tui) | | `agent/cancel-requested` | `emit` | [`packages/core/agent/src/types.ts:308`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | [`goal-session`](../packages/goal/goal-session) | -| `agent/created` | `emit` | [`packages/core/agent/src/types.ts:247`](../packages/core/agent/src/types.ts) | [`agent`](../packages/core/agent) (`events.dispatch`) | [`goal-session`](../packages/goal/goal-session), [`tui`](../packages/ui/tui) | +| `agent/created` | `emit` | [`packages/core/agent/src/types.ts:247`](../packages/core/agent/src/types.ts) | [`agent`](../packages/core/agent) (`events.dispatch`) | `apiproxy`, [`goal-session`](../packages/goal/goal-session), [`tui`](../packages/ui/tui) | | `agent/disposed` | `emit` | [`packages/core/agent/src/types.ts:256`](../packages/core/agent/src/types.ts) | [`agent`](../packages/core/agent) (`events.dispatch`) | [`agent-loop`](../packages/core/agent-loop), [`goal-session`](../packages/goal/goal-session), [`tui`](../packages/ui/tui) | | `agent/error` | `emit` | [`packages/core/agent/src/types.ts:423`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | `apiproxy`, [`goal-session`](../packages/goal/goal-session), [`session-telemetry`](../packages/telemetry/session-telemetry), [`tui`](../packages/ui/tui) | | `agent/inbox/dequeue` | `emit` | [`packages/core/agent/src/types.ts:286`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | [`agent`](../packages/core/agent), `apiproxy`, [`tui`](../packages/ui/tui) | diff --git a/packages/client/connection/src/client/api.ts b/packages/client/connection/src/client/api.ts index a1380c4b58..497c79399e 100644 --- a/packages/client/connection/src/client/api.ts +++ b/packages/client/connection/src/client/api.ts @@ -11,7 +11,7 @@ export type { WorkspaceApi, WorkspaceId, WorkspaceView, CommandsApi, CommandDescriptor, CommandExecuteResult, SkillsApi, SkillEntry, ModelCatalogFailure, ModelCatalogModel, ModelProviderGroup, ModelReasoning, - ModelReasoningEffort, ModelTarget, SessionModels, + ModelReasoningEffort, ModelTarget, SessionMetrics, SessionModels, } from '@deepseek-ai/dsh-host-apiproxy/api' export type { ToolCallView, ToolResultView } from '@deepseek-ai/dsh-tools/presentation' export type { diff --git a/packages/client/connection/src/client/index.ts b/packages/client/connection/src/client/index.ts index 0e50d8617f..a8d09ac47c 100644 --- a/packages/client/connection/src/client/index.ts +++ b/packages/client/connection/src/client/index.ts @@ -16,7 +16,7 @@ export type { ToolCallView, ToolResultView, WorkspaceApi, WorkspaceId, WorkspaceView, CommandsApi, CommandDescriptor, CommandExecuteResult, SkillsApi, SkillEntry, ModelCatalogFailure, ModelCatalogModel, ModelProviderGroup, ModelReasoning, - ModelReasoningEffort, ModelTarget, SessionModels, + ModelReasoningEffort, ModelTarget, SessionMetrics, SessionModels, RpcRequest, RpcResponse, RpcResult, RpcError, RpcErrorCode, ClientRequest, ServerResponse, ServerRequest, ClientResponse, RpcMessage, RpcReceipt, IApiClient, SessionId, SessionEvent, ContentBlock, StreamChunk, diff --git a/packages/client/runtime/README.i18n.yaml b/packages/client/runtime/README.i18n.yaml index 3fa9c934f3..3a80af5265 100644 --- a/packages/client/runtime/README.i18n.yaml +++ b/packages/client/runtime/README.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write packages/client/runtime/README.md -README.md: 81261945cb2fd8b15f7c2f15cb1ae0b8e9928499 -README.zh.md: cbbf6eded4a5375223791275f26f3bc7b6553200 +README.md: fbb3a142b9efa50045f127d430bd2be2849f01f2 +README.zh.md: b5bd3c458ed462e01dfae5bcd7399dde5552dc8d diff --git a/packages/client/runtime/README.md b/packages/client/runtime/README.md index 81261945cb..fbb3a142b9 100644 --- a/packages/client/runtime/README.md +++ b/packages/client/runtime/README.md @@ -2,7 +2,7 @@ English | [中文](README.zh.md) -Client cordis boot and React-free object services: SlotsService wraps SlotCore and supplies renderer data sources; SessionsService owns Session objects, list/scope/history state; WorkspacesService depends on SessionsService and owns Workspace objects, list/actions, default-target derivation, and the New Session blank-reuse entry (`connectWorkspace`). The runtime fans the shared Host stream into both managers. Client sessions are always Host-born (Session+Agent+cwd in one `session.create`); the client holds no pre-entity session state — a session's Agent scope (the client mirror of host dsh-scope, keyed by the shared agent/session id) is born when its row enters the list mirror and dies with the prune. Contract: api-contracts v3 §4. `ConversationSnapshot` carries `todos` — the session's current todo projection: taken from the tail history page's full-log value (host-computed, independent of the page window), preserved across an older-page prepend, and overwritten by each live `todo/write` (last write wins). A tail response that omits the field means the log holds no `todo/write`, so the list resets to empty — a plan the log never kept (a write lost to a host crash) disappears on the next open or resync. +Client cordis boot and React-free object services: SlotsService wraps SlotCore and supplies renderer data sources; SessionsService owns Session objects, list/scope/history state; WorkspacesService depends on SessionsService and owns Workspace objects, list/actions, default-target derivation, and the New Session blank-reuse entry (`connectWorkspace`). The runtime fans the shared Host stream into both managers. Client sessions are always Host-born (Session+Agent+cwd in one `session.create`); the client holds no pre-entity session state — a session's Agent scope (the client mirror of host dsh-scope, keyed by the shared agent/session id) is born when its row enters the list mirror and dies with the prune. Contract: api-contracts v3 §4. `ConversationSnapshot` carries two Host-owned full-log projections. `todos` comes from the tail history page, survives older-page prepend, and follows live `todo/write` events. `metrics` comes from tail history and live `session/metrics` frames, survives older-page prepend, and accepts only nondecreasing log and projection revisions; a subscription baseline clears it before replay so a new stream generation can restart revisions safely. Missing metrics remain `null` rather than being inferred from the visible node window. ## Workspace and Session lists diff --git a/packages/client/runtime/README.zh.md b/packages/client/runtime/README.zh.md index cbbf6eded4..b5bd3c458e 100644 --- a/packages/client/runtime/README.zh.md +++ b/packages/client/runtime/README.zh.md @@ -2,7 +2,7 @@ [English](README.md) | 中文 -客户端 cordis 启动与不依赖 React 的对象服务:SlotsService 包装 SlotCore 并提供 renderer 数据源;SessionsService 拥有 Session 对象、列表/scope/history 状态;WorkspacesService 依赖 SessionsService,拥有 Workspace 对象、列表/操作、默认目标派生,以及 New Session 空会话复用入口(`connectWorkspace`)。运行时把共享 Host 流分发给两个 manager。客户端 Session 一律由 Host 出生(一次 `session.create` 同瞬产出 Session+Agent+cwd);客户端不持有任何实体化之前的会话状态——Agent scope(host dsh-scope 的客户端镜像,以 agent/session 共用 id 为键)在会话行进入列表镜像时出生,随 prune 死亡。契约:api-contracts v3 §4。`ConversationSnapshot` 携带 `todos`——会话当前的 todo 投影:取自尾页 history 携带的全量 log 值(host 计算,独立于分页窗口),跨往前翻页保留,并被每次实时 `todo/write` 覆盖(后写胜出)。尾页响应省略该字段即表示 log 中没有任何 `todo/write`,因此列表复位为空——log 从未留下的计划(写入因 host 崩溃丢失)会在下一次打开或 resync 时消失。 +客户端 cordis 启动与不依赖 React 的对象服务:SlotsService 包装 SlotCore 并提供 renderer 数据源;SessionsService 拥有 Session 对象、列表/scope/history 状态;WorkspacesService 依赖 SessionsService,拥有 Workspace 对象、列表/操作、默认目标派生,以及 New Session 空会话复用入口(`connectWorkspace`)。运行时把共享 Host 流分发给两个 manager。客户端 Session 一律由 Host 出生(一次 `session.create` 同瞬产出 Session+Agent+cwd);客户端不持有任何实体化之前的会话状态——Agent scope(host dsh-scope 的客户端镜像,以 agent/session 共用 id 为键)在会话行进入列表镜像时出生,随 prune 死亡。契约:api-contracts v3 §4。`ConversationSnapshot` 携带两项由 Host 拥有的完整日志投影。`todos` 来自 history 尾页,在向前加载较早页面时保留,并随实时 `todo/write` 事件更新。`metrics` 来自 history 尾页和实时 `session/metrics` 帧,在向前加载较早页面时保留,并且只接受日志修订号与投影修订号均不减小的数据;订阅基线会在回放前将其清除,使新的流代次可以安全地从头开始计数修订号。缺失的 metrics 保持为 `null`,而不是根据可见节点窗口推断。 ## Workspace 与 Session 列表 diff --git a/packages/client/runtime/src/client/sessions/conversation.ts b/packages/client/runtime/src/client/sessions/conversation.ts index 8cd57c4eb3..202ddf7df2 100644 --- a/packages/client/runtime/src/client/sessions/conversation.ts +++ b/packages/client/runtime/src/client/sessions/conversation.ts @@ -6,7 +6,7 @@ import type { ContentBlock } from '@deepseek-ai/dsh-llm/types' import type { TodoItem } from '@deepseek-ai/dsh-session/types' import type { - RpcError, SessionId, ToolCallView, ToolResultView, + RpcError, SessionId, SessionMetrics, ToolCallView, ToolResultView, } from '@deepseek-ai/dsh-client-connection/client' import type { PendingInteraction } from './pending.ts' @@ -246,4 +246,10 @@ export interface ConversationSnapshot { /** Current whole-list `todo/write` projection — the tail page's full-log value, then each live * write (last write wins); empty = the log holds no plan. */ todos: readonly TodoItem[] + /** + * Host-owned cumulative usage and current-context projection. Independent + * of `nodes` pagination; null until a tail response or live metrics frame + * supplies a current value. + */ + metrics: SessionMetrics | null } diff --git a/packages/client/runtime/src/client/sessions/session.ts b/packages/client/runtime/src/client/sessions/session.ts index 1dd0283429..a8eecd1bac 100644 --- a/packages/client/runtime/src/client/sessions/session.ts +++ b/packages/client/runtime/src/client/sessions/session.ts @@ -5,7 +5,7 @@ import type { ContentBlock } from '@deepseek-ai/dsh-llm/types' import type { SessionEvent, TodoItem } from '@deepseek-ai/dsh-session/types' import type { HistoryEntry, IApiClient, MuxFrame, RpcError, RpcId, RpcResult, - SessionId, ToolEventView, + SessionId, SessionMetrics, ToolEventView, } from '@deepseek-ai/dsh-client-connection/client' // Value import from the inline-safe wire layer (not the connection plugin): // plugin-to-plugin value imports are a bundle purity error. @@ -102,6 +102,8 @@ export class Session implements ObservableSnapshot { /** Current whole-list todo/write projection: each tail history response replaces it (an omitted * field is the authoritative empty list) and every live write overwrites it. */ private todos: readonly TodoItem[] = [] + /** Host-owned metrics projection; ordering resets on each subscribed baseline. */ + private metrics: SessionMetrics | null = null /** `run_code` sub-dispatches by parent callId (window-derived, like openCalls). Appends * copy-on-write the per-parent array so published snapshot references never mutate. */ private codeDispatches = new Map() @@ -296,6 +298,7 @@ export class Session implements ObservableSnapshot { this.events = [] this.views = [] this.baseSeq = 0 + this.metrics = null // Superseded, not settled: the baseline replay re-sends still-pending requested frames verbatim // (same rpcId), re-minting fresh waits; a stale reference's respond() still reaches the host. this.pending.clear() @@ -364,6 +367,14 @@ export class Session implements ObservableSnapshot { this.queueRev++ this.notifier.markDirty() } + if (this.metrics !== null) { + this.metrics = null + this.notifier.markDirty() + } + return + } + case 'session/metrics': { + this.installMetrics(frame.metrics) return } case 'approval/requested': { @@ -482,13 +493,20 @@ export class Session implements ObservableSnapshot { this.openError = result.error return } - this.installWindow(result.value.events, result.value.hasMore, result.value.todos) + this.installWindow(result.value.events, result.value.hasMore, result.value.todos, result.value.metrics) // Gap detection (§D.3-4): baseline past the window tail and liveBuffer did not cover it -> pull the tail page once more. const tailSeq = this.windowTailSeq() if (this.subscribedLastSeq !== null && tailSeq !== null && this.subscribedLastSeq > tailSeq) { result = (await this.api.sessions.history({ sessionId: this.sessionId, maxMessages: PAGE_MESSAGES })).result if (generation !== this.openGeneration) return - if (result.ok) this.installWindow(result.value.events, result.value.hasMore, result.value.todos) + if (result.ok) { + this.installWindow( + result.value.events, + result.value.hasMore, + result.value.todos, + result.value.metrics, + ) + } } this.openState = 'open' } catch (error) { @@ -506,7 +524,12 @@ export class Session implements ObservableSnapshot { * Stitching MUST NOT route through acceptLiveEvent: openState is still 'loading' here * (doOpen flips it after install), so recursing would push every buffered event straight * back into liveBuffer where nothing ever drains it — a silent drop loop (audit S1). */ - private installWindow(entries: HistoryEntry[], hasMore: boolean, todos: readonly TodoItem[] | undefined): void { + private installWindow( + entries: HistoryEntry[], + hasMore: boolean, + todos: readonly TodoItem[] | undefined, + metrics: SessionMetrics | undefined, + ): void { this.events = entries.map(e => e.event) this.views = entries.map(e => e.view) this.baseSeq = this.events[0]?.seq ?? 0 @@ -519,6 +542,7 @@ export class Session implements ObservableSnapshot { // field is the authoritative empty list, not a missing carrier. Assigning // it clears a plan the log never kept (a write lost to a host crash). this.todos = todos ?? [] + if (metrics !== undefined) this.installMetrics(metrics) this.foldAdapter.reset(this.events, this.baseSeq, this.views) this.rebuildDerivedFromWindow() const buffered = this.liveBuffer @@ -569,7 +593,12 @@ export class Session implements ObservableSnapshot { const { result } = await this.api.sessions.history({ sessionId: this.sessionId, maxMessages: PAGE_MESSAGES }) // Failure or superseded by a full resync: drop — the resync path rebuilds and clears the buffer itself. if (result.ok && generation === this.openGeneration && this.openState === 'open') { - this.installWindow(result.value.events, result.value.hasMore, result.value.todos) + this.installWindow( + result.value.events, + result.value.hasMore, + result.value.todos, + result.value.metrics, + ) } } catch (error) { console.error('[web-runtime] gap repair failed:', error) @@ -761,6 +790,20 @@ export class Session implements ObservableSnapshot { return tail === undefined ? null : tail.seq } + /** Install a metrics snapshot unless a newer durable or publication revision already landed. */ + private installMetrics(metrics: SessionMetrics): void { + const current = this.metrics + if ( + current !== null + && ( + metrics.logRevision < current.logRevision + || metrics.projectionRevision < current.projectionRevision + ) + ) return + this.metrics = metrics + this.notifier.markDirty() + } + private buildSnapshot(): ConversationSnapshot { const { nodes: folded, degraded } = this.foldAdapter.nodes() // Frozen interrupted nodes ride fractional seqs: a stable merge keeps them in flow order. @@ -811,6 +854,7 @@ export class Session implements ObservableSnapshot { blank: this.blankBit, lastAgentError: this.lastAgentError, todos: this.todos, + metrics: this.metrics, } } } diff --git a/packages/client/runtime/tests/fake-api.ts b/packages/client/runtime/tests/fake-api.ts index a5eecd0cf5..6676f00190 100644 --- a/packages/client/runtime/tests/fake-api.ts +++ b/packages/client/runtime/tests/fake-api.ts @@ -3,7 +3,7 @@ // deferred-controlled timing). Streams are hand pumps: pushMux/pushHost. import type { ClientResponse, CommandDescriptor, CommandExecuteResult, HostFrame, IApiClient, ModelTarget, MuxFrame, - RpcError, RpcReceipt, RpcRequest, RpcResponse, SessionId, SessionModels, SkillEntry, + RpcError, RpcReceipt, RpcRequest, RpcResponse, SessionId, SessionMetrics, SessionModels, SkillEntry, WorkspaceId, WorkspaceView, } from '@deepseek-ai/dsh-client-connection/client' import { RpcId } from '@deepseek-ai/dsh-client-connection/client' @@ -63,7 +63,12 @@ export class FakeApiClient implements IApiClient { onCreate: (payload: unknown) => Promise> = () => Promise.resolve(ok({ sessionId: 'fk-new' as SessionId })) readonly defaultModel: ModelTarget = { provider: 'deepseek', model: 'deepseek-v4-flash' } onHistory: (payload: { sessionId: SessionId; beforeSeq?: number; maxMessages?: number }) - => Promise> = + => Promise> = () => Promise.resolve(ok({ events: [], hasMore: false })) onModels: (payload: unknown) => Promise> = () => Promise.resolve(ok({ diff --git a/packages/client/runtime/tests/queue-store.spec.ts b/packages/client/runtime/tests/queue-store.spec.ts index 6fe9f33c80..0ecd7a5375 100644 --- a/packages/client/runtime/tests/queue-store.spec.ts +++ b/packages/client/runtime/tests/queue-store.spec.ts @@ -85,6 +85,13 @@ describe('queue retirement (host queuedMirror rules)', () => { expect(session.getSnapshot().queue).toHaveLength(1) }) + it('an unrelated durable event leaves the queue unchanged', () => { + const session = makeSession() + session.handleMuxEnvelope(rid('e1'), queuedFrame('留', 'p-1')) + session.handleMuxEnvelope(rid('e2'), { type: 'session/event', sessionId: SID, event: ev.user(0, 'unrelated') }) + expect(session.getSnapshot().queue.map(row => row.key)).toEqual(['p-1']) + }) + it('steering/message drains the source-matched steering row only', () => { const session = makeSession() session.handleMuxEnvelope(rid('e1'), queuedFrame('普通', 'p-1')) // idle → non-steering diff --git a/packages/client/runtime/tests/session.spec.ts b/packages/client/runtime/tests/session.spec.ts index 100c6f524f..6d3ce893c7 100644 --- a/packages/client/runtime/tests/session.spec.ts +++ b/packages/client/runtime/tests/session.spec.ts @@ -7,8 +7,9 @@ */ import { describe, expect, it, vi } from 'vitest' +import { Context } from 'cordis' import type { SessionEvent } from '@deepseek-ai/dsh-session/types' -import type { SessionId } from '@deepseek-ai/dsh-client-connection/client' +import type { SessionId, SessionMetrics } from '@deepseek-ai/dsh-client-connection/client' import { Session } from '../src/client/sessions/session.ts' import { FakeApiClient, deferred, err, ok } from './fake-api.ts' import { entries, ev, plainTurn } from './event-script.ts' @@ -22,9 +23,37 @@ function makeSession(api = new FakeApiClient()): { api: FakeApiClient; session: return { api, session: new Session(SID, api) } } -function histResponse(events: SessionEvent[], hasMore = false, todos?: { content: string; status: 'pending' | 'in_progress' | 'completed' }[]) { +function histResponse( + events: SessionEvent[], + hasMore = false, + todos?: { content: string; status: 'pending' | 'in_progress' | 'completed' }[], + metrics?: SessionMetrics, +) { // history now returns HistoryEntry[] ({event, view?}); these tests are view-less. - return Promise.resolve(ok({ events: entries(events) as never[], hasMore, ...todos === undefined ? {} : { todos } })) + return Promise.resolve(ok({ + events: entries(events) as never[], + hasMore, + ...todos === undefined ? {} : { todos }, + ...metrics === undefined ? {} : { metrics }, + })) +} + +function metrics( + projectionRevision: number, + logRevision: number, + over: Partial = {}, +): SessionMetrics { + return { + projectionRevision, + logRevision, + uncachedInputTokens: 10, + outputTokens: 4, + cacheReadTokens: 90, + cacheWriteTokens: 3, + contextTokens: 35, + contextWindow: 100, + ...over, + } } describe('open', () => { @@ -40,6 +69,19 @@ describe('open', () => { expect(snapshot.openState).toBe('open') expect(snapshot.hasMore).toBe(true) expect(snapshot.nodes.map(n => n.kind)).toEqual(['user', 'assistant']) + expect(snapshot.metrics).toBeNull() + }) + + it('installs full-log metrics independently of older history pages', async () => { + const { api, session } = makeSession() + const tailMetrics = metrics(4, 106) + api.onHistory = () => histResponse(plainTurn(100, 3, '问', '答'), true, undefined, tailMetrics) + await session.open() + expect(session.getSnapshot().metrics).toBe(tailMetrics) + + api.onHistory = () => histResponse(plainTurn(94, 2, '旧问', '旧答')) + await session.loadOlder() + expect(session.getSnapshot().metrics).toBe(tailMetrics) }) it('is idempotent: concurrent opens share one history call, reopening when open is a no-op', async () => { @@ -104,6 +146,43 @@ describe('live event path', () => { expect(session.getSnapshot().nodes).toEqual(before.nodes) }) + it('orders live metrics, rejects stale projections, and clears the value at a reconnect baseline', async () => { + const { session } = await opened() + const current = metrics(8, 10) + session.handleMuxEnvelope('m1' as never, { + type: 'session/metrics', + sessionId: SID, + metrics: current, + }) + expect(session.getSnapshot().metrics).toBe(current) + + session.handleMuxEnvelope('m2' as never, { + type: 'session/metrics', + sessionId: SID, + metrics: metrics(9, 9, { uncachedInputTokens: 1 }), + }) + session.handleMuxEnvelope('m3' as never, { + type: 'session/metrics', + sessionId: SID, + metrics: metrics(7, 11, { uncachedInputTokens: 2 }), + }) + expect(session.getSnapshot().metrics).toBe(current) + + session.handleMuxEnvelope('sub' as never, { + type: 'session/subscribed', + sessionId: SID, + lastSeq: 5, + }) + expect(session.getSnapshot().metrics).toBeNull() + const nextGeneration = metrics(0, 10, { contextTokens: 20 }) + session.handleMuxEnvelope('m4' as never, { + type: 'session/metrics', + sessionId: SID, + metrics: nextGeneration, + }) + expect(session.getSnapshot().metrics).toBe(nextGeneration) + }) + it('accumulates chunks into partial, then finalize swaps partial out as the node lands', async () => { const { session } = await opened() const feed = (event: SessionEvent) => { session.handleMuxEnvelope('r' as never, { type: 'session/event', sessionId: SID, event }) } @@ -332,6 +411,23 @@ describe('prompt and cancel errors', () => { }) describe('pending interactions', () => { + it('routes an approval wait response through the original requested rpcId', async () => { + const { api, session } = makeSession() + session.handleMuxEnvelope('ra-answer' as never, { + type: 'approval/requested', + sessionId: SID, + approvalId: 'ap-answer' as never, + toolName: 'bash', + }) + const wait = session.getSnapshot().pending[0]! + await wait.respond({ ok: true, value: { decision: 'allow' } }) + expect(api.callsOf('respond')).toEqual([{ + type: 'client-response', + rpcId: 'ra-answer', + result: { ok: true, value: { decision: 'allow' } }, + }]) + }) + it('adds approval/question on requested and removes them on resolved', async () => { const { session } = makeSession() session.handleMuxEnvelope('ra' as never, { type: 'approval/requested', sessionId: SID, approvalId: 'ap1' as never, toolName: 'rm' }) @@ -374,6 +470,16 @@ describe('pending interactions', () => { }) describe('remaining branches', () => { + it('rejects a second scope bind and allows rebinding after explicit release', () => { + const { session } = makeSession() + const first = new Context() + const second = new Context() + session.bindScope(first) + expect(() => { session.bindScope(second) }).toThrow(`session ${SID} already has a bound scope`) + session.unbindScope() + expect(() => { session.bindScope(second) }).not.toThrow() + }) + it('prompt transport throw folds to internal promptError', async () => { const { api, session } = makeSession() api.onPrompt = () => Promise.reject(new Error('prompt wire down')) diff --git a/packages/client/ui-conversation/README.i18n.yaml b/packages/client/ui-conversation/README.i18n.yaml index c1a6828d5b..dfc109ce90 100644 --- a/packages/client/ui-conversation/README.i18n.yaml +++ b/packages/client/ui-conversation/README.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write packages/client/ui-conversation/README.md -README.md: 56a445ccfa86e0b11cf5aefc37819a30746f0739 -README.zh.md: a7c160ecdd74074257c9d149630663dacd05c070 +README.md: 9a8595e693e2b49691d0d130d4db0754e1e4829b +README.zh.md: f5080314488e807877ebf0c93ea82cdd9725e8ed diff --git a/packages/client/ui-conversation/README.md b/packages/client/ui-conversation/README.md index 56a445ccfa..9a8595e693 100644 --- a/packages/client/ui-conversation/README.md +++ b/packages/client/ui-conversation/README.md @@ -18,6 +18,8 @@ Per-session UI state for selection and the active view lives in the declared cha The composer bar declares session-scoped single seats for `'conversation.input.plan'` and `'conversation.input.model'`, plus list slots for overlay, dock, left, and right input extensions. InputBar renders the model seat immediately before its pending indicator and send/stop button. Feature packages own each control and its state; ui-conversation supplies placement, the `locked` owner prop, and the standard slot shares. The resident no-session shell uses `DisabledInputBar` and therefore dispatches no session-scoped control seats. +The chat stats line reads durable token counters and current-context pressure only from `ConversationSnapshot.metrics`; visible nodes supply only the existing turn/step counts. It renders uncached input, output, and cache reads as separate compact values, computes cache hit as `cacheRead / (uncachedInput + cacheRead)` without cache writes, and shows context occupancy against the selected route's exact capacity. Missing host data is labeled unknown, never reconstructed from a paged window. + `src/client/` is organized for the future package split: `contract/` is the sole inter-domain shared face (`slots.ts` slot declarations + composed slot props including the tool-row contract, `views.ts` shared primitives, `tool-call-model.ts`); the `skeleton/`, `chat/`, and `toolviews/` (sample registrants) domain directories import contract files and never each other; `apply.ts` is the only assembly point allowed to import all three domains. The `/client` export surface is the contract only — `apply`/`inject`, the two service classes, and the `contract/` type families; implementation components (skeleton, chat rows) and the store factory stay internal and reach the page exclusively through apply's slot registrations (tests take them via the `./src/*` subpath). ## Model Experience diff --git a/packages/client/ui-conversation/README.zh.md b/packages/client/ui-conversation/README.zh.md index a7c160ecdd..f508031448 100644 --- a/packages/client/ui-conversation/README.zh.md +++ b/packages/client/ui-conversation/README.zh.md @@ -18,6 +18,8 @@ todo 两个面就是在该形状上的两个注册项,都是普通注册方插 输入栏为 `'conversation.input.plan'` 和 `'conversation.input.model'` 声明会话作用域的单实例 seat,并为 overlay、dock、left 和 right 输入扩展声明列表 slot。InputBar 将模型 seat 渲染在 pending 指示器与发送/停止按钮之前。各功能包拥有相应控件及其状态;ui-conversation 提供放置位置、`locked` owner prop 和标准 slot share。常驻无会话壳使用 `DisabledInputBar`,因此不会分发任何会话作用域的控件 seat。 +聊天统计行只从 `ConversationSnapshot.metrics` 读取持久的 token 计数与当前上下文压力;可见节点仅提供既有的轮次和步骤计数。它以相互独立的紧凑值显示未缓存输入、输出与缓存读取,通过 `cacheRead / (uncachedInput + cacheRead)` 计算缓存命中率而不计入缓存写入,并根据所选路由的精确容量显示上下文占用率。Host 数据缺失时标为「未知」,绝不根据分页窗口重建。 + `src/client/` 按未来的包拆分组织:`contract/` 是唯一的跨领域共享表层(`slots.ts` slot 声明 + 组合后的 slot props,包括工具行契约、`views.ts` 共享原语、`tool-call-model.ts`);`skeleton/`、`chat/` 和 `toolviews/`(示例注册方)领域目录只导入 contract 文件,彼此绝不导入;`apply.ts` 是唯一允许导入全部三个领域的组装点。`/client` 导出表层只包含契约:`apply`/`inject`、两个服务类和 `contract/` 类型家族;实现组件(骨架、聊天行)与 store factory 保持内部状态,只能通过 apply 的 slot 注册到达页面(测试通过 `./src/*` 子路径获取它们)。 ## 模型体验 diff --git a/packages/client/ui-conversation/src/client/chat/StatsLine.tsx b/packages/client/ui-conversation/src/client/chat/StatsLine.tsx index 45b783f19b..5c77b585e7 100644 --- a/packages/client/ui-conversation/src/client/chat/StatsLine.tsx +++ b/packages/client/ui-conversation/src/client/chat/StatsLine.tsx @@ -5,48 +5,63 @@ import type { ConversationSnapshot } from '@deepseek-ai/dsh-client-runtime/clien import type { SnapshotSelectorHook } from '@deepseek-ai/dsh-client-ui-slots' import css from './StatsLine.module.css' -interface UsageTotals { +type SessionMetrics = NonNullable + +interface VisibleCounts { turns: number steps: number - tokens: number - cacheHitPct: number | null -} - -/** Token accounting slice of assistant `usage` (typed upstream as unknown). */ -interface UsageLike { - inputTokens?: number - outputTokens?: number - cacheReadTokens?: number } /** - * Fold assistant nodes into display totals. + * Count visible assistant turns and steps without treating the paged window + * as an accounting source. * @param nodes - snapshot nodes. - * @returns totals; cacheHitPct null until any cache accounting arrives. + * @returns visible turn and step counts. */ -export function deriveStats(nodes: ConversationSnapshot['nodes']): UsageTotals { +export function deriveVisibleCounts(nodes: ConversationSnapshot['nodes']): VisibleCounts { const turns = new Set() let steps = 0 - let tokens = 0 - let input = 0 - let cacheRead = 0 for (const node of nodes) { if (node.kind !== 'assistant') continue turns.add(node.turn) steps += 1 - const usage = node.usage as UsageLike | undefined - if (usage === undefined) continue - input += usage.inputTokens ?? 0 - cacheRead += usage.cacheReadTokens ?? 0 - tokens += (usage.inputTokens ?? 0) + (usage.outputTokens ?? 0) + (usage.cacheReadTokens ?? 0) - } - const denom = input + cacheRead - return { - turns: turns.size, - steps, - tokens, - cacheHitPct: denom === 0 ? null : Math.round((cacheRead / denom) * 100), } + return { turns: turns.size, steps } +} + +/** + * Format large token values with the status surfaces' compact suffix style. + * @param value - token count or model capacity. + * @returns locale-formatted count. + */ +export function formatMetricTokens(value: number): string { + if (value < 1_000) return value.toLocaleString('en-US') + return value.toLocaleString('en-US', { + notation: 'compact', + maximumFractionDigits: 1, + }).replace('K', 'k').replace('M', 'm').replace('B', 'b') +} + +/** + * Existing Web cache-hit formula over disjoint uncached and cache-read input. + * @param metrics - Host-owned durable usage. + * @returns rounded integer percent, or null when no input was billed. + */ +export function cacheHitPercent(metrics: SessionMetrics): number | null { + const denominator = metrics.uncachedInputTokens + metrics.cacheReadTokens + return denominator === 0 + ? null + : Math.round(metrics.cacheReadTokens / denominator * 100) +} + +/** + * Current context occupancy using the TUI's integer rounding and upper clamp. + * @param metrics - Host-owned current pressure and exact route capacity. + * @returns occupancy percent, or null when either input is unavailable. + */ +export function contextPercent(metrics: SessionMetrics): number | null { + if (metrics.contextTokens === undefined || metrics.contextWindow === undefined) return null + return Math.min(100, Math.round(metrics.contextTokens / metrics.contextWindow * 100)) } /** Props: the conversation-snapshot selector hook (handed down by ChatView). */ @@ -54,12 +69,33 @@ export interface StatsLineProps { useSession: SnapshotSelectorHook s.nodes) - const stats = useMemo(() => deriveStats(nodes), [nodes]) - if (stats.steps === 0) return null + const metrics = useSession(s => s.metrics) + const counts = useMemo(() => deriveVisibleCounts(nodes), [nodes]) + if (counts.steps === 0 && ( + metrics === null + || ( + metrics.uncachedInputTokens === 0 + && metrics.outputTokens === 0 + && metrics.cacheReadTokens === 0 + && (metrics.contextTokens ?? 0) === 0 + ) + )) return null const parts: string[] = [] - if (stats.cacheHitPct !== null) parts.push(`cache hit ${stats.cacheHitPct}%`) - parts.push(`${stats.tokens.toLocaleString('en-US')} tokens`) - parts.push(`${stats.turns} turns`) - parts.push(`${stats.steps} steps`) + if (metrics === null) { + parts.push('usage unknown') + parts.push('context unknown') + } else { + parts.push(`${formatMetricTokens(metrics.uncachedInputTokens)} uncached input`) + parts.push(`${formatMetricTokens(metrics.outputTokens)} output`) + parts.push(`${formatMetricTokens(metrics.cacheReadTokens)} cache read`) + const cacheHit = cacheHitPercent(metrics) + if (cacheHit !== null) parts.push(`cache hit ${cacheHit}%`) + const context = contextPercent(metrics) + parts.push(context === null + ? 'context unknown' + : `context ${context}% of ${formatMetricTokens(metrics.contextWindow as number)}`) + } + parts.push(`${counts.turns} turns`) + parts.push(`${counts.steps} steps`) return
{parts.join(' · ')}
}) diff --git a/packages/client/ui-conversation/tests/chat-branch-tails.spec.tsx b/packages/client/ui-conversation/tests/chat-branch-tails.spec.tsx index bf1266981e..580070a1dc 100644 --- a/packages/client/ui-conversation/tests/chat-branch-tails.spec.tsx +++ b/packages/client/ui-conversation/tests/chat-branch-tails.spec.tsx @@ -129,15 +129,23 @@ describe('small branch tails', () => { }) it('StatsLine omits the cache-hit segment when no input accounting exists at all', () => { - // cacheHitPct is null only when input+cacheRead are both zero (pure - // output accounting) — any input makes it a real 0%. const snap = { nodes: [{ kind: 'assistant', seq: 1, turn: 1, step: 1, blocks: [], usage: { outputTokens: 10 } }], + metrics: { + logRevision: 2, + projectionRevision: 0, + uncachedInputTokens: 0, + outputTokens: 10, + cacheReadTokens: 0, + cacheWriteTokens: 5_000, + }, } const source = { getSnapshot: () => snap, subscribe: () => () => {} } const view = render( , ) - expect(view.getByText('10 tokens · 1 turns · 1 steps')).toBeTruthy() + expect(view.getByText( + '0 uncached input · 10 output · 0 cache read · context unknown · 1 turns · 1 steps', + )).toBeTruthy() }) }) diff --git a/packages/client/ui-conversation/tests/chat-code-subcalls.spec.tsx b/packages/client/ui-conversation/tests/chat-code-subcalls.spec.tsx index aa9451b413..e681020d0c 100644 --- a/packages/client/ui-conversation/tests/chat-code-subcalls.spec.tsx +++ b/packages/client/ui-conversation/tests/chat-code-subcalls.spec.tsx @@ -58,7 +58,7 @@ function snapshotWith( sessionId: SID, nodes, foldDegraded: false, partial: null, runningCalls, codeDispatches, pending: [], queue: [], todos: [], running: runningCalls.length > 0, composerPhase: 'active', removed: false, openState: 'open', openError: null, - hasMore: false, loadingOlder: false, promptError: null, blank: false, lastAgentError: null, + hasMore: false, loadingOlder: false, promptError: null, blank: false, lastAgentError: null, metrics: null, } } diff --git a/packages/client/ui-conversation/tests/chat-stats-bash-sample.spec.tsx b/packages/client/ui-conversation/tests/chat-stats-bash-sample.spec.tsx index 991aca36e0..fe9cba39b2 100644 --- a/packages/client/ui-conversation/tests/chat-stats-bash-sample.spec.tsx +++ b/packages/client/ui-conversation/tests/chat-stats-bash-sample.spec.tsx @@ -1,5 +1,5 @@ // @vitest-environment jsdom -// StatsLine (rendered inside the chat view body): totals derivation + the RFC +// StatsLine (rendered inside the chat view body): durable metrics presentation + the RFC // hard acceptance — zero renders during streaming. Bash sample row: the // canonical sub-agent differential decided INSIDE the component off the // standard useSessions kit (no registry predicates — tool ring dissolved). @@ -12,7 +12,10 @@ import type { import { createSnapshotStore } from '@deepseek-ai/dsh-client-runtime/client' import { bindSnapshotSelector } from '@deepseek-ai/dsh-client-web-react' import type { ToolRowProps } from '@deepseek-ai/dsh-client-ui-conversation/client' -import { StatsLine, deriveStats, type StatsLineProps } from '../src/client/chat/StatsLine.tsx' +import { + cacheHitPercent, contextPercent, deriveVisibleCounts, formatMetricTokens, + StatsLine, type StatsLineProps, +} from '../src/client/chat/StatsLine.tsx' import { BashRow } from '../src/client/toolviews/bash-sample.tsx' afterEach(cleanup) @@ -28,7 +31,7 @@ function snapshotBase(): ConversationSnapshot { return { sessionId: SID, nodes: [], foldDegraded: false, partial: null, runningCalls: [], codeDispatches: new Map(), pending: [], queue: [], todos: [], running: false, composerPhase: 'active', removed: false, openState: 'open', openError: null, - hasMore: false, loadingOlder: false, promptError: null, blank: false, lastAgentError: null, + hasMore: false, loadingOlder: false, promptError: null, blank: false, lastAgentError: null, metrics: null, } } @@ -50,27 +53,49 @@ function makeSource(init?: Partial) { } } -describe('deriveStats', () => { - it('folds turns/steps/tokens and cache hit percentage', () => { - const stats = deriveStats([ +describe('stats derivation', () => { + it('counts visible turns and steps without reading node usage', () => { + const stats = deriveVisibleCounts([ assistant(1, 1, { inputTokens: 100, outputTokens: 50, cacheReadTokens: 900 }), assistant(2, 1, { inputTokens: 100, outputTokens: 50 }), assistant(3, 2), ]) expect(stats.turns).toBe(2) expect(stats.steps).toBe(3) - expect(stats.tokens).toBe(1200) - expect(stats.cacheHitPct).toBe(82) }) - it('cache hit stays null with no cache accounting; non-assistant nodes ignored', () => { + it('ignores non-assistant nodes', () => { const tool: ToolResultNode = { kind: 'tool-result', seq: 5, time: 5_000, callId: 'c', call: null, callTime: null, content: [], isError: false, callView: null, resultView: null, } - const stats = deriveStats([tool, assistant(1, 1)]) + const stats = deriveVisibleCounts([tool, assistant(1, 1)]) expect(stats.steps).toBe(1) - expect(stats.cacheHitPct).toBeNull() + }) + + it('keeps the cache formula disjoint from cache writes and rounds/clamps context like the TUI', () => { + const durable = { + logRevision: 20, + projectionRevision: 2, + uncachedInputTokens: 100, + outputTokens: 50, + cacheReadTokens: 900, + cacheWriteTokens: 50_000, + contextTokens: 34_500, + contextWindow: 100_000, + } + expect(cacheHitPercent(durable)).toBe(90) + expect(contextPercent(durable)).toBe(35) + expect(contextPercent({ ...durable, contextTokens: 200_000 })).toBe(100) + const { contextWindow: _contextWindow, ...withoutContextWindow } = durable + expect(contextPercent(withoutContextWindow)).toBeNull() + expect(cacheHitPercent({ ...durable, uncachedInputTokens: 0, cacheReadTokens: 0 })).toBeNull() + }) + + it('formats large values compactly in the existing en-US style', () => { + expect(formatMetricTokens(999)).toBe('999') + expect(formatMetricTokens(15_962)).toBe('16k') + expect(formatMetricTokens(2_172_544)).toBe('2.2m') }) }) @@ -79,19 +104,65 @@ describe('StatsLine', () => { return { useSession: bindSnapshotSelector(source) } } - it('renders the joined stats row and hides with zero steps', () => { + it('renders separate durable counters, cache hit, context occupancy, and visible counts', () => { const { source } = makeSource({ nodes: [assistant(1, 1, { inputTokens: 10, outputTokens: 5, cacheReadTokens: 90 })], + metrics: { + logRevision: 30, + projectionRevision: 4, + uncachedInputTokens: 120_237, + outputTokens: 13_881, + cacheReadTokens: 2_172_544, + cacheWriteTokens: 99_999, + contextTokens: 89_600, + contextWindow: 256_000, + }, }) const view = render() - expect(view.getByText('cache hit 90% · 105 tokens · 1 turns · 1 steps')).toBeTruthy() + expect(view.getByText( + '120.2k uncached input · 13.9k output · 2.2m cache read · cache hit 95% · context 35% of 256k · 1 turns · 1 steps', + )).toBeTruthy() const empty = makeSource() const emptyView = render() expect(emptyView.container.textContent).toBe('') }) + it('renders honest unknowns when the host projection is missing', () => { + const { source } = makeSource({ nodes: [assistant(1, 1)] }) + const view = render() + expect(view.getByText('usage unknown · context unknown · 1 turns · 1 steps')).toBeTruthy() + }) + + it.each([ + { uncachedInputTokens: 1, outputTokens: 0, cacheReadTokens: 0, contextTokens: 0 }, + { uncachedInputTokens: 0, outputTokens: 1, cacheReadTokens: 0, contextTokens: 0 }, + { uncachedInputTokens: 0, outputTokens: 0, cacheReadTokens: 1, contextTokens: 0 }, + { uncachedInputTokens: 0, outputTokens: 0, cacheReadTokens: 0, contextTokens: 1 }, + ])('keeps a metrics-only row visible for each nonzero projection bucket', (nonzero) => { + const { source } = makeSource({ + metrics: { + logRevision: 1, + projectionRevision: 0, + cacheWriteTokens: 0, + ...nonzero, + }, + }) + const view = render() + expect(view.container.textContent).toContain('0 turns · 0 steps') + }) + it('renders ZERO times during streaming chunk frames (RFC hard acceptance)', () => { - const { set, source } = makeSource({ nodes: [assistant(1, 1)] }) + const { set, source } = makeSource({ + nodes: [assistant(1, 1)], + metrics: { + logRevision: 4, + projectionRevision: 0, + uncachedInputTokens: 1, + outputTokens: 1, + cacheReadTokens: 0, + cacheWriteTokens: 0, + }, + }) let renders = 0 function Counting(p: StatsLineProps) { renders += 1 diff --git a/packages/client/ui-conversation/tests/chat-toolview-slot.spec.tsx b/packages/client/ui-conversation/tests/chat-toolview-slot.spec.tsx index 87e188bbbd..05da39d195 100644 --- a/packages/client/ui-conversation/tests/chat-toolview-slot.spec.tsx +++ b/packages/client/ui-conversation/tests/chat-toolview-slot.spec.tsx @@ -41,7 +41,7 @@ function snapshotWith(nodes: ToolResultNode[]): ConversationSnapshot { return { sessionId: SID, nodes, foldDegraded: false, partial: null, runningCalls: [], codeDispatches: new Map(), pending: [], queue: [], todos: [], running: false, composerPhase: 'active', removed: false, openState: 'open', openError: null, - hasMore: false, loadingOlder: false, promptError: null, blank: false, lastAgentError: null, + hasMore: false, loadingOlder: false, promptError: null, blank: false, lastAgentError: null, metrics: null, } } diff --git a/packages/client/ui-conversation/tests/chat-view.spec.tsx b/packages/client/ui-conversation/tests/chat-view.spec.tsx index 20389c9e23..08a4218332 100644 --- a/packages/client/ui-conversation/tests/chat-view.spec.tsx +++ b/packages/client/ui-conversation/tests/chat-view.spec.tsx @@ -31,7 +31,7 @@ function snapshotBase(): ConversationSnapshot { return { sessionId: SID, nodes: [], foldDegraded: false, partial: null, runningCalls: [], codeDispatches: new Map(), pending: [], queue: [], todos: [], running: false, composerPhase: 'active', removed: false, openState: 'open', openError: null, - hasMore: false, loadingOlder: false, promptError: null, blank: false, lastAgentError: null, + hasMore: false, loadingOlder: false, promptError: null, blank: false, lastAgentError: null, metrics: null, } } diff --git a/packages/client/ui-conversation/tests/gate-branch-tails.spec.tsx b/packages/client/ui-conversation/tests/gate-branch-tails.spec.tsx index 8ee049f899..d34b784e01 100644 --- a/packages/client/ui-conversation/tests/gate-branch-tails.spec.tsx +++ b/packages/client/ui-conversation/tests/gate-branch-tails.spec.tsx @@ -20,7 +20,7 @@ function snapshotBase(): ConversationSnapshot { return { sessionId: SID, nodes: [], foldDegraded: false, partial: null, runningCalls: [], codeDispatches: new Map(), pending: [], queue: [], todos: [], running: false, composerPhase: 'active', removed: false, openState: 'open', openError: null, - hasMore: false, loadingOlder: false, promptError: null, blank: false, lastAgentError: null, + hasMore: false, loadingOlder: false, promptError: null, blank: false, lastAgentError: null, metrics: null, } } @@ -36,7 +36,7 @@ describe('render branch tails', () => { expect(view.container.querySelector('[data-state="ok"]')).not.toBeNull() }) - it('StatsLine skips usage-less nodes and defaults each absent counter to zero', () => { + it('StatsLine takes durable counters from metrics while keeping visible node counts', () => { const snap = { nodes: [ { kind: 'assistant', seq: 1, turn: 1, step: 1, blocks: [] }, @@ -44,12 +44,22 @@ describe('render branch tails', () => { // outputTokens absent: the tokens sum's ?? 0 arm for output. { kind: 'assistant', seq: 3, turn: 2, step: 1, blocks: [], usage: { inputTokens: 5 } }, ], + metrics: { + logRevision: 9, + projectionRevision: 1, + uncachedInputTokens: 9, + outputTokens: 6, + cacheReadTokens: 0, + cacheWriteTokens: 0, + }, } const source = { getSnapshot: () => snap, subscribe: () => () => {} } const view = render( } />, ) - expect(view.getByText('cache hit 0% · 15 tokens · 2 turns · 3 steps')).toBeTruthy() + expect(view.getByText( + '9 uncached input · 6 output · 0 cache read · cache hit 0% · context unknown · 2 turns · 3 steps', + )).toBeTruthy() }) it('AssistantMarkdown reasoning as the streaming tail renders the running ring', () => { diff --git a/packages/client/ui-conversation/tests/input-bar.spec.tsx b/packages/client/ui-conversation/tests/input-bar.spec.tsx index c50a35110a..0f10d61bf9 100644 --- a/packages/client/ui-conversation/tests/input-bar.spec.tsx +++ b/packages/client/ui-conversation/tests/input-bar.spec.tsx @@ -23,7 +23,7 @@ function snapshotOf(overrides: Partial = {}): Conversation sessionId: SID, nodes: [], foldDegraded: false, partial: null, runningCalls: [], codeDispatches: new Map(), pending: [], queue: [], todos: [], running: false, composerPhase: 'active', removed: false, openState: 'open', openError: null, hasMore: false, loadingOlder: false, - promptError: null, blank: false, lastAgentError: null, + promptError: null, blank: false, lastAgentError: null, metrics: null, ...overrides, } } diff --git a/packages/client/ui-conversation/tests/input-matrix.spec.tsx b/packages/client/ui-conversation/tests/input-matrix.spec.tsx index 284ef6c76a..65d4a69e32 100644 --- a/packages/client/ui-conversation/tests/input-matrix.spec.tsx +++ b/packages/client/ui-conversation/tests/input-matrix.spec.tsx @@ -26,7 +26,7 @@ function mountBar(shell: SessionInputShell, over?: { running?: boolean; disabled sessionId: SID, nodes: [], foldDegraded: false, partial: null, runningCalls: [], codeDispatches: new Map(), pending: [], queue: [], todos: [], running: over?.running ?? false, composerPhase: 'active', removed: over?.disabled ?? false, openState: 'open', openError: null, hasMore: false, - loadingOlder: false, promptError: null, blank: false, lastAgentError: null, + loadingOlder: false, promptError: null, blank: false, lastAgentError: null, metrics: null, }) const props: InputBarProps = { sessionId: SID, diff --git a/packages/client/ui-conversation/tests/input-scenarios.spec.tsx b/packages/client/ui-conversation/tests/input-scenarios.spec.tsx index 414f3c15b4..f2aad68296 100644 --- a/packages/client/ui-conversation/tests/input-scenarios.spec.tsx +++ b/packages/client/ui-conversation/tests/input-scenarios.spec.tsx @@ -112,7 +112,7 @@ async function scopedBench(register?: (slash: SlashService) => void) { sessionId, nodes: [], foldDegraded: false, partial: null, runningCalls: [], codeDispatches: new Map(), pending: [], queue: [], todos: [], running: false, composerPhase: 'active', removed: false, openState: 'open', openError: null, hasMore: false, loadingOlder: false, - promptError: null, blank: false, lastAgentError: null, + promptError: null, blank: false, lastAgentError: null, metrics: null, }) const barProps: InputBarProps = { sessionId, diff --git a/packages/client/ui-conversation/tests/queue-dock.spec.tsx b/packages/client/ui-conversation/tests/queue-dock.spec.tsx index fa0c871bdb..90a83b31e6 100644 --- a/packages/client/ui-conversation/tests/queue-dock.spec.tsx +++ b/packages/client/ui-conversation/tests/queue-dock.spec.tsx @@ -20,7 +20,7 @@ function snapshotWith(queue: QueuedMessage[]): ConversationSnapshot { return { sessionId: SID, nodes: [], foldDegraded: false, partial: null, runningCalls: [], codeDispatches: new Map(), pending: [], queue, todos: [], running: true, composerPhase: 'active', removed: false, openState: 'open', openError: null, - hasMore: false, loadingOlder: false, promptError: null, blank: false, lastAgentError: null, + hasMore: false, loadingOlder: false, promptError: null, blank: false, lastAgentError: null, metrics: null, } } diff --git a/packages/client/ui-conversation/tests/skeleton.spec.tsx b/packages/client/ui-conversation/tests/skeleton.spec.tsx index b777c85ac3..55e523abc3 100644 --- a/packages/client/ui-conversation/tests/skeleton.spec.tsx +++ b/packages/client/ui-conversation/tests/skeleton.spec.tsx @@ -50,7 +50,7 @@ function conversationSnapshot(overrides: Partial = {}): Co sessionId: SID, nodes: [], foldDegraded: false, partial: null, runningCalls: [], codeDispatches: new Map(), pending: [], queue: [], todos: [], running: false, composerPhase: 'active', removed: false, openState: 'open', openError: null, hasMore: false, loadingOlder: false, - promptError: null, blank: false, lastAgentError: null, + promptError: null, blank: false, lastAgentError: null, metrics: null, ...overrides, } } diff --git a/packages/host/apiproxy/README.i18n.yaml b/packages/host/apiproxy/README.i18n.yaml index 28ff470c0b..34593f89f9 100644 --- a/packages/host/apiproxy/README.i18n.yaml +++ b/packages/host/apiproxy/README.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write packages/host/apiproxy/README.md -README.md: d6db9a9541b0727b61dbe501f7234564ffef139e -README.zh.md: 4175c8fdb98aad2882718a2c95cd9e45825d787d +README.md: fa6fcb17ad3f2332e71e00387034fd7196759d8f +README.zh.md: cfdf0d772ac81a4e569a799569b5de6e509f3b6f diff --git a/packages/host/apiproxy/README.md b/packages/host/apiproxy/README.md index d6db9a9541..fa6fcb17ad 100644 --- a/packages/host/apiproxy/README.md +++ b/packages/host/apiproxy/README.md @@ -18,7 +18,7 @@ Workspace and Session lists are separate reconnect baselines. `workspace.create` `host.pickDirectory` opens one native directory picker and returns its selected path, or `null` when the user cancels. Its host implementation invokes platform tools without a shell: `osascript` on macOS, an STA PowerShell `FolderBrowserDialog` on Windows, and Zenity with a KDialog fallback on Linux. The picker function is injectable for tests. This user-paced method is the sole unary call exempt from the default 30-second timeout; caller and connection aborts still propagate to the native process. The browser carrier separately restricts this privileged method to loopback, same-origin requests. -`session.history` pages on message boundaries, and its tail page (no `beforeSeq`) carries two session-level extras the page window cannot supply: the in-flight partial's chunk events, and `todos` — the latest `todo/write` whole-list projection over the full log. Older pages omit `todos` because the projection is session-level, not per-page; a tail response that omits it means the whole log holds no `todo/write`, so clients read the absent field as the empty plan rather than as unchanged state. +`session.history` pages on message boundaries, and its tail page (no `beforeSeq`) carries session-level projections the page window cannot supply: the in-flight partial's chunk events; `todos`, the latest `todo/write` whole-list projection; and `metrics`, full-log usage deduplicated by `(turn, step)` plus current token-meter pressure and exact selected-route capacity when available. Older pages omit the session-level projections. Live `session/metrics` mux frames carry monotonic log/projection revisions, so clients reject stale frames and preserve the counters while prepending older pages. Cache reads and writes remain disjoint buckets; the cache-hit denominator is uncached input plus cache reads. The `command.*` and `skill.*` domains expose the host command registry and skill catalog to clients. Every method addresses one session's agent by `sessionId` (a served session always has an Agent; `command.*` resumes cold sessions through the same path as `session.*`, while `skill.list` resolves the project root from the session header without touching the Agent registry). `command.execute` runs a slash-command line host-side and returns a detached result; the carrier's request signal cancels the running handler. `host/commands-changed` is the catalog invalidation frame: clients refetch `command.list` instead of diffing. diff --git a/packages/host/apiproxy/README.zh.md b/packages/host/apiproxy/README.zh.md index 4175c8fdb9..cfdf0d772a 100644 --- a/packages/host/apiproxy/README.zh.md +++ b/packages/host/apiproxy/README.zh.md @@ -18,7 +18,7 @@ Workspace 列表与 Session 列表是相互独立的重连基线。`workspace.cr `host.pickDirectory` 会打开一个原生目录选择器并返回选中的路径;用户取消时返回 `null`。宿主实现不经 shell 调用平台工具:macOS 使用 `osascript`,Windows 使用以 STA 模式运行的 PowerShell `FolderBrowserDialog`,Linux 使用 Zenity,并以 KDialog 作为回退。选择器函数可在测试中注入。该方法需等待用户完成操作,是唯一不受默认 30 秒超时限制的一元调用;调用方发出的中止信号和连接中止仍会传播至原生进程。浏览器载体另行将这一特权方法限制为仅接受来自回环地址的同源请求。 -`session.history` 按消息边界分页,其尾页(不带 `beforeSeq`)额外携带两项页窗口本身无法提供的会话级数据:进行中局部消息的 chunk 事件,以及 `todos`——整份日志上最后一次 `todo/write` 的整表投影。较早的页面不带 `todos`,因为该投影是会话级而非分页级的;尾页响应缺少该字段意味着整份日志中没有任何 `todo/write`,因此客户端要把缺失字段读作空计划,而不是读作「状态未变」。 +`session.history` 按消息边界分页,其尾页(不带 `beforeSeq`)携带页窗口本身无法提供的会话级投影:进行中局部消息的分片事件;`todos`,即最后一次 `todo/write` 的整表投影;以及 `metrics`,即按 `(turn, step)` 去重的完整日志用量,并在可用时包含当前 token 计量压力和所选精确路由的容量。较早的页面省略会话级投影。实时 `session/metrics` mux 帧携带单调递增的日志修订号与投影修订号,因此客户端会拒绝陈旧帧,并在向前加载较早页面时保留计数器。缓存读取与缓存写入保持为彼此独立的计数项;缓存命中率的分母是未缓存输入加缓存读取。 `command.*` 与 `skill.*` 领域向客户端暴露宿主命令注册表和技能目录。每个方法都通过 `sessionId` 寻址一个会话的 Agent(被服务的会话必有 Agent;`command.*` 经由与 `session.*` 相同的路径恢复冷会话,而 `skill.list` 从会话头解析项目根目录,不触碰 Agent 注册表)。`command.execute` 在宿主侧运行一条斜杠命令行并返回脱耦结果;载体的请求信号可取消正在运行的处理器。`host/commands-changed` 是目录失效帧:客户端重新拉取 `command.list` 而不是做差分。 diff --git a/packages/host/apiproxy/src/api-proxy.ts b/packages/host/apiproxy/src/api-proxy.ts index 58922284be..13aa85c1fe 100644 --- a/packages/host/apiproxy/src/api-proxy.ts +++ b/packages/host/apiproxy/src/api-proxy.ts @@ -40,6 +40,7 @@ import type { } from '@deepseek-ai/dsh-user-interaction' import { UserInteractionError } from '@deepseek-ai/dsh-user-interaction' import { pickNativeDirectory } from './native-directory-picker.ts' +import { affectsSessionMetrics, SessionMetricsProjector } from './session-metrics.ts' /** Page size when history is called without maxMessages. */ const DEFAULT_MAX_MESSAGES = 50 @@ -417,6 +418,54 @@ export function createApiProxy(ctx: Context, defaults: ApiProxyDefaults): ApiPro for (const queue of muxQueues) queue.push(envelope) } + const pendingMetricSessions = new Set() + let metricFlushScheduled = false + let metricsDisposed = false + const metricsProjector = new SessionMetricsProjector( + ctx, + agent => targetFor(agent).current, + (agent) => { scheduleMetrics(agent.session) }, + ) + + /** Queue one full-log metrics publication after synchronous session listeners drain. */ + function scheduleMetrics(session: Session): void { + if (metricsDisposed || muxQueues.size === 0) return + pendingMetricSessions.add(session) + if (metricFlushScheduled) return + metricFlushScheduled = true + queueMicrotask(() => { + metricFlushScheduled = false + if (metricsDisposed) { + pendingMetricSessions.clear() + return + } + const sessions = [...pendingMetricSessions] + pendingMetricSessions.clear() + for (const current of sessions) { + broadcast({ + type: 'session/metrics', + sessionId: current.id, + metrics: metricsProjector.snapshot(current, ctx.agents.get(current.id)), + }) + } + }) + } + + ctx.effect(() => { + const disposers = [ + ctx.on('session/event', (session: Session, event: SessionEvent) => { + if (affectsSessionMetrics(event)) scheduleMetrics(session) + }), + ctx.on('agent/created', (agent: Agent) => { scheduleMetrics(agent.session) }), + ctx.on('session/disposed', (session: Session) => { pendingMetricSessions.delete(session) }), + ] + return () => { + metricsDisposed = true + pendingMetricSessions.clear() + for (const dispose of disposers) dispose() + } + }, 'api-proxy: session metrics') + /** * Per-session inbox mirror serving the mux-open queue snapshot (the same * refresh-recovery baseline as pending questions). Keyed by the stable @@ -723,7 +772,15 @@ export function createApiProxy(ctx: Context, defaults: ApiProxyDefaults): ApiPro // log (the page window may not contain the last todo/write; a paged // client cannot reconstruct session-level state from it). const todos = beforeSeq === undefined ? backscanTodos(found.agent.session.events) : undefined - return ok(request, { events: entries, hasMore: page.hasMore, ...todos === undefined ? {} : { todos } }) + const metrics = beforeSeq === undefined + ? metricsProjector.snapshot(found.agent.session, found.agent) + : undefined + return ok(request, { + events: entries, + hasMore: page.hasMore, + ...todos === undefined ? {} : { todos }, + ...metrics === undefined ? {} : { metrics }, + }) }, async models(request) { @@ -817,6 +874,11 @@ export function createApiProxy(ctx: Context, defaults: ApiProxyDefaults): ApiPro : { reasoningEffort: resolved.reasoningEffort }, } targetFor(found.agent).current = selected + broadcast({ + type: 'session/metrics', + sessionId: found.agent.session.id, + metrics: metricsProjector.snapshot(found.agent.session, found.agent), + }) return ok(request, { selected: { ...selected } }) } catch (error: unknown) { return err(request, { @@ -1102,6 +1164,11 @@ export function createApiProxy(ctx: Context, defaults: ApiProxyDefaults): ApiPro muxQueues.add(queue) for (const session of ctx.sessions.list()) { subscribeSession(queue, session) + queue.push(frame({ + type: 'session/metrics', + sessionId: session.id, + metrics: metricsProjector.snapshot(session, ctx.agents.get(session.id)), + })) } for (const pending of pendingQuestions.values()) { queue.push({ @@ -1154,6 +1221,11 @@ export function createApiProxy(ctx: Context, defaults: ApiProxyDefaults): ApiPro }), ctx.on('session/created', (session: Session) => { subscribeSession(queue, session) + queue.push(frame({ + type: 'session/metrics', + sessionId: session.id, + metrics: metricsProjector.snapshot(session, ctx.agents.get(session.id)), + })) }), ctx.on('session/disposed', (session: Session) => { openCalls.delete(session.id) diff --git a/packages/host/apiproxy/src/api/events.schema.ts b/packages/host/apiproxy/src/api/events.schema.ts index 973db5a91e..0f317128dd 100644 --- a/packages/host/apiproxy/src/api/events.schema.ts +++ b/packages/host/apiproxy/src/api/events.schema.ts @@ -10,7 +10,9 @@ import type { HostFrame, MuxFrame } from './events.ts' import type { Wire } from './rpc.schema.ts' import { rpcErrorSchema, rpcIdSchema } from './rpc.schema.ts' import { approvalRequestIdSchema } from './approvals.schema.ts' -import { contentBlockSchema, sessionEventSchema, sessionIdSchema, toolEventViewSchema } from './sessions.schema.ts' +import { + contentBlockSchema, sessionEventSchema, sessionIdSchema, sessionMetricsSchema, toolEventViewSchema, +} from './sessions.schema.ts' import { workspaceIdSchema, workspaceViewSchema } from './workspace.schema.ts' /** Question shape validated strictly against core dsh-user-interaction. */ @@ -27,6 +29,7 @@ export const askUserQuestionItemSchema = z.object({ export const muxFrameSchema = z.discriminatedUnion('type', [ z.object({ type: z.literal('session/event'), sessionId: sessionIdSchema, event: sessionEventSchema, view: toolEventViewSchema.optional() }), z.object({ type: z.literal('session/subscribed'), sessionId: sessionIdSchema, lastSeq: z.number().int() }), + z.object({ type: z.literal('session/metrics'), sessionId: sessionIdSchema, metrics: sessionMetricsSchema }), z.object({ type: z.literal('session/title'), sessionId: sessionIdSchema, title: z.string().min(1), eventSeq: z.number().int().nonnegative(), updatedAt: z.number() }), z.object({ type: z.literal('approval/requested'), sessionId: sessionIdSchema, approvalId: approvalRequestIdSchema, toolName: z.string(), callId: z.string().optional(), reason: z.string().optional() }), z.object({ type: z.literal('approval/resolved'), sessionId: sessionIdSchema, approvalId: approvalRequestIdSchema, outcome: z.union([z.literal('allowed-once'), z.literal('rejected'), z.literal('cancelled'), z.literal('unavailable')]) }), diff --git a/packages/host/apiproxy/src/api/events.ts b/packages/host/apiproxy/src/api/events.ts index 70139d0a00..73e2b42deb 100644 --- a/packages/host/apiproxy/src/api/events.ts +++ b/packages/host/apiproxy/src/api/events.ts @@ -13,6 +13,7 @@ import type { CallId } from '@deepseek-ai/dsh-llm/brand' import type { SessionEvent, SessionId } from '@deepseek-ai/dsh-session/types' import type { ToolCallView, ToolResultView } from '@deepseek-ai/dsh-tools/presentation' import type { RpcError, RpcId, RpcRequest } from './rpc.ts' +import type { SessionMetrics } from './sessions.ts' import type { WorkspaceView } from './workspace.ts' // Client-side consumers take the render-intent vocabulary from the contract; @@ -57,6 +58,7 @@ export interface EventsApi { export type MuxFrame = | { type: 'session/event'; sessionId: SessionId; event: SessionEvent; view?: ToolEventView } | { type: 'session/subscribed'; sessionId: SessionId; lastSeq: number } + | { type: 'session/metrics'; sessionId: SessionId; metrics: SessionMetrics } | { type: 'session/title'; sessionId: SessionId; title: string; eventSeq: number; updatedAt: number } | { type: 'approval/requested'; sessionId: SessionId; approvalId: ApprovalRequestId; toolName: string; callId?: CallId; reason?: string } | { type: 'approval/resolved'; sessionId: SessionId; approvalId: ApprovalRequestId; outcome: ApprovalOutcome } diff --git a/packages/host/apiproxy/src/api/index.ts b/packages/host/apiproxy/src/api/index.ts index ad5fbd3bf0..1b1268657c 100644 --- a/packages/host/apiproxy/src/api/index.ts +++ b/packages/host/apiproxy/src/api/index.ts @@ -27,7 +27,7 @@ export interface ApiProxy { // ---- Domain interfaces and payload entities ---- export type { HistoryEntry, ModelCatalogFailure, ModelCatalogModel, ModelProviderGroup, ModelReasoning, - ModelReasoningEffort, ModelTarget, SessionModels, SessionsApi, SessionSummary, + ModelReasoningEffort, ModelTarget, SessionMetrics, SessionModels, SessionsApi, SessionSummary, } from './sessions.ts' export type { HostApi } from './host.ts' export type { WorkspaceApi, WorkspaceId, WorkspaceView } from './workspace.ts' diff --git a/packages/host/apiproxy/src/api/sessions.schema.ts b/packages/host/apiproxy/src/api/sessions.schema.ts index 02efe769af..0866766da5 100644 --- a/packages/host/apiproxy/src/api/sessions.schema.ts +++ b/packages/host/apiproxy/src/api/sessions.schema.ts @@ -11,7 +11,7 @@ import type { RequestPayload, ResponseValue } from './rpc-map.ts' import type { Wire } from './rpc.schema.ts' import type { HistoryEntry, ModelCatalogFailure, ModelCatalogModel, ModelProviderGroup, ModelReasoning, - ModelReasoningEffort, ModelTarget, SessionSummary, + ModelReasoningEffort, ModelTarget, SessionMetrics, SessionSummary, } from './sessions.ts' import type { ToolEventView } from './events.ts' import type { WorkspaceId } from './workspace.ts' @@ -145,11 +145,24 @@ export const todoItemSchema = z.object({ status: z.union([z.literal('pending'), z.literal('in_progress'), z.literal('completed')]), }) +/** Host-owned durable usage and current-context projection. */ +export const sessionMetricsSchema = z.object({ + logRevision: z.number().int().nonnegative(), + projectionRevision: z.number().int().nonnegative(), + uncachedInputTokens: z.number().nonnegative(), + outputTokens: z.number().nonnegative(), + cacheReadTokens: z.number().nonnegative(), + cacheWriteTokens: z.number().nonnegative(), + contextTokens: z.number().nonnegative().optional(), + contextWindow: z.number().int().positive().optional(), +}) satisfies z.ZodType> + /** session.history response value. */ export const sessionHistoryValueSchema = z.object({ events: z.array(historyEntrySchema), hasMore: z.boolean(), todos: z.array(todoItemSchema).optional(), + metrics: sessionMetricsSchema.optional(), }) satisfies z.ZodType>> /** session.models request payload. */ diff --git a/packages/host/apiproxy/src/api/sessions.ts b/packages/host/apiproxy/src/api/sessions.ts index 88308c829b..9a0344867e 100644 --- a/packages/host/apiproxy/src/api/sessions.ts +++ b/packages/host/apiproxy/src/api/sessions.ts @@ -32,6 +32,31 @@ export interface HistoryEntry { view?: ToolEventView } +/** + * Host-owned token metrics for one durable session revision. Provider usage + * buckets are cumulative across the full log; current context fields describe + * the replayed request surface at this revision and are absent when the Host + * cannot measure pressure or resolve exact-route capacity. + */ +export interface SessionMetrics { + /** Number of durable events included in this projection. */ + logRevision: number + /** Monotone ordering within one Host process and mux subscription generation. */ + projectionRevision: number + /** Cumulative uncached provider input. */ + uncachedInputTokens: number + /** Cumulative provider output. */ + outputTokens: number + /** Cumulative provider cache reads. */ + cacheReadTokens: number + /** Cumulative provider cache writes; excluded from the Web cache-hit formula. */ + cacheWriteTokens: number + /** Current request pressure from `ctx.tokenMeter.measure(session).totalTokens`. */ + contextTokens?: number + /** Exact selected-route capacity from `ctx.llm.resolveModelInfo()`. */ + contextWindow?: number +} + /** Complete model target selected for one session. */ export interface ModelTarget { /** Registered provider route. */ @@ -153,9 +178,11 @@ export interface SessionsApi { * projection (latest `todo/write` over the FULL log, independent of the page window) — * so a paged client restores the plan without walking history; absent when the session * never wrote one. Older pages omit it (the projection is session-level, not per-page). + * The same tail-only rule carries `metrics`, whose cumulative usage and current context + * are Host projections over the full log rather than products of the returned page. */ history(request: RpcRequest<{ sessionId: SessionId; beforeSeq?: number; maxMessages?: number }>): - Promise> + Promise> /** Reads a fresh advisory model directory for this session. Provider lookups run independently. */ models(request: RpcRequest<{ sessionId: SessionId }>): Promise> diff --git a/packages/host/apiproxy/src/session-metrics.ts b/packages/host/apiproxy/src/session-metrics.ts new file mode 100644 index 0000000000..2415a80182 --- /dev/null +++ b/packages/host/apiproxy/src/session-metrics.ts @@ -0,0 +1,194 @@ +/** + * Full-log usage and current-context projection for Web clients. + * + * @module @deepseek-ai/dsh-host-apiproxy/session-metrics + */ + +import type { Context } from 'cordis' +import type { Agent, AgentLlmTarget } from '@deepseek-ai/dsh-agent' +import type { TokenUsage } from '@deepseek-ai/dsh-llm' +import type { Session, SessionEvent } from '@deepseek-ai/dsh-session' +import type { SessionMetrics } from './api/sessions.ts' + +interface UsageState { + logRevision: number + projectionRevision: number + uncachedInputTokens: number + outputTokens: number + cacheReadTokens: number + cacheWriteTokens: number + byStep: Map +} + +interface CapacityState { + routeKey: string + generation: number + status: 'pending' | 'ready' + contextWindow?: number +} + +interface TokenMeterLike { + measure(session: Session): { totalTokens: number } +} + +interface LlmLike { + resolveModelInfo(provider: string, model: string): Promise<{ + context?: { contextWindow: number } + }> +} + +function usageFrom(event: SessionEvent): { turn: number; step: number; usage: TokenUsage } | undefined { + if (event.type === 'assistant/chunk' && event.data.chunk.type === 'usage') { + return { turn: event.data.turn, step: event.data.step, usage: event.data.chunk.usage } + } + if (event.type === 'assistant/message' && event.data.usage !== undefined) { + return { turn: event.data.turn, step: event.data.step, usage: event.data.usage } + } + return undefined +} + +/** + * Whether an appended event can change cumulative usage or token-meter + * pressure. Text/reasoning stream deltas remain outside both projections. + * @param event - appended durable event. + * @returns true when the Host must publish a fresh metrics snapshot. + */ +export function affectsSessionMetrics(event: SessionEvent): boolean { + if (event.type === 'assistant/chunk') return event.data.chunk.type === 'usage' + if (event.type === 'request/header') return true + return 'surfaceOp' in event +} + +function recordUsage(state: UsageState, turn: number, step: number, usage: TokenUsage): void { + const key = `${turn}:${step}` + const previous = state.byStep.get(key) + if (previous !== undefined) { + state.uncachedInputTokens -= previous.inputTokens + state.outputTokens -= previous.outputTokens + state.cacheReadTokens -= previous.cacheReadTokens ?? 0 + state.cacheWriteTokens -= previous.cacheWriteTokens ?? 0 + } + state.byStep.set(key, usage) + state.uncachedInputTokens += usage.inputTokens + state.outputTokens += usage.outputTokens + state.cacheReadTokens += usage.cacheReadTokens ?? 0 + state.cacheWriteTokens += usage.cacheWriteTokens ?? 0 +} + +/** + * Projects durable cumulative usage and route-aware current context without + * awaiting model metadata on the session append path. + */ +export class SessionMetricsProjector { + private readonly usage = new WeakMap() + private readonly capacities = new WeakMap() + + /** + * @param ctx - Host context providing optional token-meter and LLM services. + * @param targetFor - selected route owner for one attached Web agent. + * @param onCapacityResolved - schedules a fresh live projection after exact-route metadata resolves. + */ + constructor( + private readonly ctx: Context, + private readonly targetFor: (agent: Agent) => Pick, + private readonly onCapacityResolved: (agent: Agent) => void, + ) {} + + /** + * Read a fresh detached projection through the session's durable tail. + * @param session - authoritative durable log owner. + * @param agent - attached route owner, when available. + * @returns cumulative usage and any currently available pressure/capacity. + */ + snapshot(session: Session, agent?: Agent): SessionMetrics { + const state = this.syncUsage(session) + const tokenMeter = this.ctx.get('tokenMeter') as TokenMeterLike | undefined + let contextTokens: number | undefined + if (tokenMeter !== undefined) { + try { + contextTokens = tokenMeter.measure(session).totalTokens + } catch { + // A malformed or temporarily unmeasurable replay has no honest pressure value. + } + } + const contextWindow = agent === undefined ? undefined : this.capacityFor(agent) + return { + logRevision: state.logRevision, + projectionRevision: state.projectionRevision++, + uncachedInputTokens: state.uncachedInputTokens, + outputTokens: state.outputTokens, + cacheReadTokens: state.cacheReadTokens, + cacheWriteTokens: state.cacheWriteTokens, + ...contextTokens === undefined ? {} : { contextTokens }, + ...contextWindow === undefined ? {} : { contextWindow }, + } + } + + private syncUsage(session: Session): UsageState { + let state = this.usage.get(session) + if (state === undefined) { + state = { + logRevision: 0, + projectionRevision: 0, + uncachedInputTokens: 0, + outputTokens: 0, + cacheReadTokens: 0, + cacheWriteTokens: 0, + byStep: new Map(), + } + this.usage.set(session, state) + } + while (state.logRevision < session.events.length) { + const event = session.events[state.logRevision] + /* v8 ignore next -- Session events are append-only and dense; logRevision is bounded by length. */ + if (event === undefined) break + const usage = usageFrom(event) + if (usage !== undefined) recordUsage(state, usage.turn, usage.step, usage.usage) + state.logRevision++ + } + return state + } + + private capacityFor(agent: Agent): number | undefined { + const target = this.targetFor(agent) + const routeKey = `${target.provider}\u0000${target.model}` + let state = this.capacities.get(agent) + if (state === undefined || state.routeKey !== routeKey) { + state = { + routeKey, + generation: (state?.generation ?? 0) + 1, + status: 'pending', + } + this.capacities.set(agent, state) + this.resolveCapacity(agent, target, state) + } + return state.status === 'ready' ? state.contextWindow : undefined + } + + private resolveCapacity( + agent: Agent, + target: Pick, + pending: CapacityState, + ): void { + const llm = this.ctx.get('llm') as LlmLike | undefined + if (llm === undefined) { + pending.status = 'ready' + return + } + void Promise.resolve() + .then(() => llm.resolveModelInfo(target.provider, target.model)) + .then( + (resolved) => { + if (this.capacities.get(agent)?.generation !== pending.generation) return + const current = this.targetFor(agent) + if (`${current.provider}\u0000${current.model}` !== pending.routeKey) return + pending.status = 'ready' + if (resolved.context !== undefined) pending.contextWindow = resolved.context.contextWindow + this.onCapacityResolved(agent) + }, + () => { + if (this.capacities.get(agent)?.generation === pending.generation) pending.status = 'ready' + }, + ) + } +} diff --git a/packages/host/apiproxy/tests/api-proxy-models.spec.ts b/packages/host/apiproxy/tests/api-proxy-models.spec.ts index af8791431a..e0e7a035fa 100644 --- a/packages/host/apiproxy/tests/api-proxy-models.spec.ts +++ b/packages/host/apiproxy/tests/api-proxy-models.spec.ts @@ -19,6 +19,7 @@ import SystemPrompt from '@deepseek-ai/dsh-system-prompt' import UserInteractionService from '@deepseek-ai/dsh-user-interaction' import type { RpcRequest } from '@deepseek-ai/dsh-host-apiproxy/api/rpc' import { RpcId } from '@deepseek-ai/dsh-host-apiproxy/api/rpc' +import type { MuxFrame } from '@deepseek-ai/dsh-host-apiproxy/api' import { createApiProxy } from '../src/api-proxy.ts' let nextRpc = 1 @@ -52,6 +53,7 @@ class CatalogAdapter extends LlmAdapter { provider, id: model, name: model, + context: { contextWindow: model === 'private-preview' ? 128_000 : 64_000 }, ...this.reasoning === undefined ? {} : { reasoning: this.reasoning }, }) } @@ -117,6 +119,16 @@ function expectValue(response: { result: { ok: true; value: T } | { ok: false return response.result.value } +async function nextMetrics( + iterator: AsyncIterator>, +): Promise['metrics']> { + for (;;) { + const next = await iterator.next() + if (next.done) throw new Error('mux ended before a metrics frame') + if (next.value.payload.type === 'session/metrics') return next.value.payload.metrics + } +} + describe('Web session model selection', () => { it('groups successful providers, isolates failures, and preserves an unlisted current model', async () => { const { ctx, sessionId } = await harness({ @@ -230,4 +242,26 @@ describe('Web session model selection', () => { .toEqual({ provider: 'deepseek', model: 'private-preview', reasoningEffort: 'max' }) await ctx.fiber.dispose() }) + + it('publishes unknown capacity immediately on selection, then the exact selected route capacity', async () => { + const { ctx, sessionId } = await harness() + const api = createApiProxy(ctx, { provider: 'deepseek', model: 'deepseek-chat', cwd: '/tmp', workspaceRoot: '/tmp' }) + const controller = new AbortController() + const iterator = api.events.mux(request({}), controller.signal)[Symbol.asyncIterator]() + + expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() + expect((await nextMetrics(iterator)).contextWindow).toBe(64_000) + + expectValue(await api.sessions.selectModel(request({ + sessionId, + provider: 'deepseek', + model: 'private-preview', + }))) + expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() + expect((await nextMetrics(iterator)).contextWindow).toBe(128_000) + + controller.abort() + await iterator.return?.() + await ctx.fiber.dispose() + }) }) diff --git a/packages/host/apiproxy/tests/rpc-schemas.spec.ts b/packages/host/apiproxy/tests/rpc-schemas.spec.ts index 9f1309b476..9a21341e26 100644 --- a/packages/host/apiproxy/tests/rpc-schemas.spec.ts +++ b/packages/host/apiproxy/tests/rpc-schemas.spec.ts @@ -10,7 +10,7 @@ import { sessionCreateValueSchema, sessionEventSchema, sessionHistoryRequestSchema, sessionHistoryValueSchema, sessionIdSchema, sessionListRequestSchema, sessionListValueSchema, sessionModelsRequestSchema, sessionModelsValueSchema, sessionPromptRequestSchema, sessionPromptValueSchema, - sessionSelectModelRequestSchema, sessionSelectModelValueSchema, sessionSummarySchema, + sessionSelectModelRequestSchema, sessionSelectModelValueSchema, sessionSummarySchema, sessionMetricsSchema, } from '../src/api/sessions.schema.ts' import { hostDescribeRequestSchema, hostDescribeValueSchema } from '../src/api/host.schema.ts' import { @@ -138,8 +138,35 @@ describe('sessions domain schemas', () => { expect(sessionHistoryValueSchema.parse({ events: [], hasMore: false, + metrics: { + logRevision: 12, + projectionRevision: 4, + uncachedInputTokens: 1_000, + outputTokens: 200, + cacheReadTokens: 4_000, + cacheWriteTokens: 500, + contextTokens: 8_000, + contextWindow: 128_000, + }, modelTarget: { provider: 'deepseek', model: 'deepseek-v4-flash' }, - }).hasMore).toBe(false) + }).metrics?.contextWindow).toBe(128_000) + expect(() => sessionMetricsSchema.parse({ + logRevision: 1, + projectionRevision: 0, + uncachedInputTokens: -1, + outputTokens: 0, + cacheReadTokens: 0, + cacheWriteTokens: 0, + })).toThrow() + expect(() => sessionMetricsSchema.parse({ + logRevision: 1, + projectionRevision: 0, + uncachedInputTokens: 0, + outputTokens: 0, + cacheReadTokens: 0, + cacheWriteTokens: 0, + contextWindow: 0, + })).toThrow() expect(sessionModelsRequestSchema.parse({ sessionId: 's1' }).sessionId).toBe('s1') expect(sessionModelsValueSchema.parse({ current: { provider: 'deepseek', model: 'deepseek-v4-flash', reasoningEffort: 'max' }, @@ -305,6 +332,18 @@ describe('events frame schemas', () => { { type: 'session/event', sessionId: 's', event: { type: 't', seq: 0, time: 1, data: null } }, { type: 'session/subscribed', sessionId: 's', lastSeq: -1 }, { type: 'session/title', sessionId: 's', title: 'Durable title', eventSeq: 2, updatedAt: 3 }, + { + type: 'session/metrics', + sessionId: 's', + metrics: { + logRevision: 3, + projectionRevision: 1, + uncachedInputTokens: 100, + outputTokens: 20, + cacheReadTokens: 300, + cacheWriteTokens: 40, + }, + }, { type: 'approval/requested', sessionId: 's', approvalId: 'a', toolName: 'bash', callId: 'c', reason: 'r' }, { type: 'approval/resolved', sessionId: 's', approvalId: 'a', outcome: 'allowed-once' }, { type: 'question/requested', sessionId: 's', questions: [{ id: 'q', question: 'Q?', options: [{ label: 'L' }], multiSelect: true }] }, diff --git a/packages/host/apiproxy/tests/session-metrics.spec.ts b/packages/host/apiproxy/tests/session-metrics.spec.ts new file mode 100644 index 0000000000..beea6378ea --- /dev/null +++ b/packages/host/apiproxy/tests/session-metrics.spec.ts @@ -0,0 +1,277 @@ +import { describe, expect, it, vi } from 'vitest' +import { Context } from 'cordis' +import type { Agent, AgentLlmTarget } from '@deepseek-ai/dsh-agent' +import { Session, SessionId } from '@deepseek-ai/dsh-session' +import { affectsSessionMetrics, SessionMetricsProjector } from '../src/session-metrics.ts' + +function assistant( + session: Session, + turn: number, + step: number, + usage: { + inputTokens: number + outputTokens: number + cacheReadTokens?: number + cacheWriteTokens?: number + }, +): void { + session.append('assistant/chunk', { + turn, + step, + chunk: { type: 'usage', usage }, + }) + session.append('assistant/message', { + turn, + step, + content: [{ type: 'text', text: `answer-${turn}-${step}` }], + provenance: { provider: 'test', model: 'alpha' }, + usage, + }, { surfaceOp: 'append' }) +} + +function agent(session: Session): Agent { + return { id: session.id, session } as Agent +} + +describe('SessionMetricsProjector', () => { + it('filters text/reasoning stream deltas while retaining usage, headers, and surface mutations', () => { + const session = new Session(SessionId('metrics-filter')) + const text = session.append('assistant/chunk', { + turn: 1, + step: 1, + chunk: { type: 'text-delta', index: 0, text: 'x' }, + }) + const usage = session.append('assistant/chunk', { + turn: 1, + step: 1, + chunk: { type: 'usage', usage: { inputTokens: 1, outputTokens: 1 } }, + }) + const header = session.append('request/header', { + header: { config: { provider: 'test', model: 'alpha' } }, + reason: 'initial', + }) + const surface = session.append('user/message', { + content: [{ type: 'text', text: 'question' }], + source: { kind: 'user' }, + }, { surfaceOp: 'append' }) + expect(affectsSessionMetrics(text)).toBe(false) + expect(affectsSessionMetrics(usage)).toBe(true) + expect(affectsSessionMetrics(header)).toBe(true) + expect(affectsSessionMetrics(surface)).toBe(true) + }) + + it('reconciles usage by turn:step, keeps cache writes disjoint, and survives a surface replacement', () => { + const ctx = new Context() + ctx.provide('tokenMeter', { + measure(session: Session) { + return { totalTokens: session.surface.nodes.length * 100 } + }, + }) + const session = new Session(SessionId('metrics-fold')) + const first = session.append('user/message', { + content: [{ type: 'text', text: 'large old surface' }], + source: { kind: 'user' }, + }, { surfaceOp: 'append' }) + assistant(session, 1, 1, { + inputTokens: 11, + outputTokens: 3, + cacheReadTokens: 89, + cacheWriteTokens: 8, + }) + + const current: AgentLlmTarget = { provider: 'test', model: 'alpha' } + const projector = new SessionMetricsProjector(ctx, () => current, () => {}) + const attached = agent(session) + const before = projector.snapshot(session, attached) + expect(before).toMatchObject({ + uncachedInputTokens: 11, + outputTokens: 3, + cacheReadTokens: 89, + cacheWriteTokens: 8, + contextTokens: 200, + }) + + const assistantSeq = session.surface.nodes.at(-1) + if (assistantSeq === undefined) throw new Error('assistant surface missing') + session.append('user/message', { + content: [{ type: 'text', text: 'compact summary' }], + source: { kind: 'plugin', plugin: 'test' }, + }, { + surfaceOp: { op: 'replace', start: first.seq, end: assistantSeq }, + sourceEventSeqs: [first.seq, assistantSeq], + }) + // A replayed usage event for the same step replaces the settled value. + session.append('assistant/chunk', { + turn: 1, + step: 1, + chunk: { + type: 'usage', + usage: { + inputTokens: 12, + outputTokens: 4, + cacheReadTokens: 88, + cacheWriteTokens: 9, + }, + }, + }) + const compacted = projector.snapshot(session, attached) + expect(compacted).toMatchObject({ + uncachedInputTokens: 12, + outputTokens: 4, + cacheReadTokens: 88, + cacheWriteTokens: 9, + contextTokens: 100, + }) + assistant(session, 1, 2, { + inputTokens: 1_000, + outputTokens: 500, + cacheReadTokens: 2_000, + cacheWriteTokens: 3_000, + }) + + const after = projector.snapshot(session, attached) + expect(after).toMatchObject({ + logRevision: session.events.length, + projectionRevision: 2, + uncachedInputTokens: 1_012, + outputTokens: 504, + cacheReadTokens: 2_088, + cacheWriteTokens: 3_009, + contextTokens: 200, + }) + expect(after.uncachedInputTokens).not.toBe( + after.uncachedInputTokens + after.cacheReadTokens + after.cacheWriteTokens, + ) + }) + + it('publishes only the selected route capacity when asynchronous resolutions race', async () => { + const ctx = new Context() + const resolutions = new Map void>() + ctx.provide('tokenMeter', { measure: () => ({ totalTokens: 35_000 }) }) + ctx.provide('llm', { + resolveModelInfo(_provider: string, model: string) { + return new Promise<{ context: { contextWindow: number } }>((resolve) => { + resolutions.set(model, (contextWindow) => { resolve({ context: { contextWindow } }) }) + }) + }, + }) + const session = new Session(SessionId('capacity-race')) + const attached = agent(session) + let current: AgentLlmTarget = { provider: 'test', model: 'alpha' } + const resolved = vi.fn() + const projector = new SessionMetricsProjector(ctx, () => current, resolved) + + expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() + await vi.waitFor(() => { expect(resolutions.has('alpha')).toBe(true) }) + current = { provider: 'test', model: 'beta' } + expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() + await vi.waitFor(() => { expect(resolutions.has('beta')).toBe(true) }) + + resolutions.get('alpha')?.(64_000) + await Promise.resolve() + expect(resolved).not.toHaveBeenCalled() + resolutions.get('beta')?.(128_000) + await vi.waitFor(() => { expect(resolved).toHaveBeenCalledOnce() }) + expect(projector.snapshot(session, attached)).toMatchObject({ + contextTokens: 35_000, + contextWindow: 128_000, + }) + }) + + it('omits current context fields when measurement or model metadata is unavailable', async () => { + const ctx = new Context() + ctx.provide('tokenMeter', { measure: () => { throw new Error('unmeasurable') } }) + ctx.provide('llm', { resolveModelInfo: () => Promise.reject(new Error('metadata unavailable')) }) + const session = new Session(SessionId('missing-metrics')) + const attached = agent(session) + const projector = new SessionMetricsProjector( + ctx, + () => ({ provider: 'test', model: 'missing' }), + () => {}, + ) + const metrics = projector.snapshot(session, attached) + expect(metrics.contextTokens).toBeUndefined() + expect(metrics.contextWindow).toBeUndefined() + await vi.waitFor(() => { + expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() + }) + }) + + it('keeps optional usage buckets at zero and tolerates absent host services or detached agents', async () => { + const ctx = new Context() + const session = new Session(SessionId('optional-metrics')) + assistant(session, 1, 0, { inputTokens: 7, outputTokens: 2 }) + const attached = agent(session) + const projector = new SessionMetricsProjector( + ctx, + () => ({ provider: 'test', model: 'no-service' }), + () => {}, + ) + + expect(projector.snapshot(session)).toMatchObject({ + uncachedInputTokens: 7, + outputTokens: 2, + cacheReadTokens: 0, + cacheWriteTokens: 0, + }) + expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() + expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() + await Promise.resolve() + }) + + it('publishes a resolved route with no advertised capacity as unknown', async () => { + const ctx = new Context() + ctx.provide('llm', { resolveModelInfo: () => Promise.resolve({}) }) + const session = new Session(SessionId('no-capacity')) + const attached = agent(session) + const resolved = vi.fn() + const projector = new SessionMetricsProjector( + ctx, + () => ({ provider: 'test', model: 'metadata-without-context' }), + resolved, + ) + + expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() + await vi.waitFor(() => { expect(resolved).toHaveBeenCalledOnce() }) + expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() + }) + + it('ignores stale resolution failures and route metadata after the target moves', async () => { + const ctx = new Context() + const resolutions = new Map() + ctx.provide('llm', { + resolveModelInfo(_provider: string, model: string) { + return new Promise<{ context: { contextWindow: number } }>((resolve, reject) => { + resolutions.set(model, { resolve, reject }) + }) + }, + }) + const session = new Session(SessionId('stale-capacity')) + const attached = agent(session) + let current: AgentLlmTarget = { provider: 'test', model: 'alpha' } + const resolved = vi.fn() + const targetFor = vi.fn(() => current) + const projector = new SessionMetricsProjector(ctx, targetFor, resolved) + + projector.snapshot(session, attached) + await vi.waitFor(() => { expect(resolutions.has('alpha')).toBe(true) }) + current = { provider: 'test', model: 'route-moved-before-snapshot' } + resolutions.get('alpha')?.resolve({ context: { contextWindow: 64_000 } }) + await vi.waitFor(() => { expect(targetFor).toHaveBeenCalledTimes(2) }) + expect(resolved).not.toHaveBeenCalled() + + projector.snapshot(session, attached) + await vi.waitFor(() => { expect(resolutions.has('route-moved-before-snapshot')).toBe(true) }) + current = { provider: 'test', model: 'beta' } + projector.snapshot(session, attached) + await vi.waitFor(() => { expect(resolutions.has('beta')).toBe(true) }) + resolutions.get('route-moved-before-snapshot')?.reject(new Error('stale failure')) + await Promise.resolve() + resolutions.get('beta')?.resolve({ context: { contextWindow: 128_000 } }) + await vi.waitFor(() => { expect(resolved).toHaveBeenCalledOnce() }) + expect(projector.snapshot(session, attached).contextWindow).toBe(128_000) + }) +}) From 1fb1f00ec24ab7580286cc482d455dba7156e01e Mon Sep 17 00:00:00 2001 From: Hypatia May Date: Tue, 28 Jul 2026 12:19:12 +0800 Subject: [PATCH 02/38] test(host): isolate compaction metric preservation (round 2) --- .../host/apiproxy/tests/session-metrics.spec.ts | 15 ++++++++++++--- 1 file changed, 12 insertions(+), 3 deletions(-) diff --git a/packages/host/apiproxy/tests/session-metrics.spec.ts b/packages/host/apiproxy/tests/session-metrics.spec.ts index beea6378ea..3ebfb1a0cf 100644 --- a/packages/host/apiproxy/tests/session-metrics.spec.ts +++ b/packages/host/apiproxy/tests/session-metrics.spec.ts @@ -100,6 +100,15 @@ describe('SessionMetricsProjector', () => { surfaceOp: { op: 'replace', start: first.seq, end: assistantSeq }, sourceEventSeqs: [first.seq, assistantSeq], }) + const compacted = projector.snapshot(session, attached) + expect(compacted).toMatchObject({ + uncachedInputTokens: 11, + outputTokens: 3, + cacheReadTokens: 89, + cacheWriteTokens: 8, + contextTokens: 100, + }) + // A replayed usage event for the same step replaces the settled value. session.append('assistant/chunk', { turn: 1, @@ -114,8 +123,8 @@ describe('SessionMetricsProjector', () => { }, }, }) - const compacted = projector.snapshot(session, attached) - expect(compacted).toMatchObject({ + const replayed = projector.snapshot(session, attached) + expect(replayed).toMatchObject({ uncachedInputTokens: 12, outputTokens: 4, cacheReadTokens: 88, @@ -132,7 +141,7 @@ describe('SessionMetricsProjector', () => { const after = projector.snapshot(session, attached) expect(after).toMatchObject({ logRevision: session.events.length, - projectionRevision: 2, + projectionRevision: 3, uncachedInputTokens: 1_012, outputTokens: 504, cacheReadTokens: 2_088, From 0dc0cc046b89295cf97c41c8a675bdb63edba22e Mon Sep 17 00:00:00 2001 From: Hypatia May Date: Tue, 28 Jul 2026 13:57:26 +0800 Subject: [PATCH 03/38] fix(host): keep metrics route lookup passive (round 3) --- packages/host/apiproxy/src/api-proxy.ts | 16 +++- packages/host/apiproxy/src/session-metrics.ts | 39 ++++++--- .../apiproxy/tests/api-proxy-models.spec.ts | 80 ++++++++++++++++--- .../apiproxy/tests/session-metrics.spec.ts | 35 +++++++- 4 files changed, 146 insertions(+), 24 deletions(-) diff --git a/packages/host/apiproxy/src/api-proxy.ts b/packages/host/apiproxy/src/api-proxy.ts index 13aa85c1fe..84177de3cc 100644 --- a/packages/host/apiproxy/src/api-proxy.ts +++ b/packages/host/apiproxy/src/api-proxy.ts @@ -405,6 +405,20 @@ export function createApiProxy(ctx: Context, defaults: ApiProxyDefaults): ApiPro return target } + /** + * Read the best capacity route without taking ownership of foreign routing. + * Web agents expose their live selection; other agents expose only a route + * that already crossed the durable request-header boundary. + */ + function metricsRouteFor(agent: Agent): Pick | undefined { + const installed = targets.get(agent) + if (installed !== undefined) return installed.current + const logged = agent.session.requestHeader()?.config + return logged === undefined + ? undefined + : { provider: logged.provider, model: logged.model } + } + /** Pre-publication setup used by both fresh and resumed Web agents. */ function installTarget(agentCtx: Context): void { const agent = agentCtx.agent @@ -423,7 +437,7 @@ export function createApiProxy(ctx: Context, defaults: ApiProxyDefaults): ApiPro let metricsDisposed = false const metricsProjector = new SessionMetricsProjector( ctx, - agent => targetFor(agent).current, + metricsRouteFor, (agent) => { scheduleMetrics(agent.session) }, ) diff --git a/packages/host/apiproxy/src/session-metrics.ts b/packages/host/apiproxy/src/session-metrics.ts index 2415a80182..161f92941e 100644 --- a/packages/host/apiproxy/src/session-metrics.ts +++ b/packages/host/apiproxy/src/session-metrics.ts @@ -21,12 +21,14 @@ interface UsageState { } interface CapacityState { - routeKey: string + routeKey: string | undefined generation: number status: 'pending' | 'ready' contextWindow?: number } +type CapacityTarget = Pick + interface TokenMeterLike { measure(session: Session): { totalTokens: number } } @@ -75,6 +77,10 @@ function recordUsage(state: UsageState, turn: number, step: number, usage: Token state.cacheWriteTokens += usage.cacheWriteTokens ?? 0 } +function routeKeyFor(target: CapacityTarget | undefined): string | undefined { + return target === undefined ? undefined : `${target.provider}\u0000${target.model}` +} + /** * Projects durable cumulative usage and route-aware current context without * awaiting model metadata on the session append path. @@ -85,12 +91,12 @@ export class SessionMetricsProjector { /** * @param ctx - Host context providing optional token-meter and LLM services. - * @param targetFor - selected route owner for one attached Web agent. + * @param targetFor - side-effect-free selected or logged route lookup for one attached agent. * @param onCapacityResolved - schedules a fresh live projection after exact-route metadata resolves. */ constructor( private readonly ctx: Context, - private readonly targetFor: (agent: Agent) => Pick, + private readonly targetFor: (agent: Agent) => CapacityTarget | undefined, private readonly onCapacityResolved: (agent: Agent) => void, ) {} @@ -151,23 +157,23 @@ export class SessionMetricsProjector { private capacityFor(agent: Agent): number | undefined { const target = this.targetFor(agent) - const routeKey = `${target.provider}\u0000${target.model}` + const routeKey = routeKeyFor(target) let state = this.capacities.get(agent) if (state === undefined || state.routeKey !== routeKey) { state = { routeKey, generation: (state?.generation ?? 0) + 1, - status: 'pending', + status: target === undefined ? 'ready' : 'pending', } this.capacities.set(agent, state) - this.resolveCapacity(agent, target, state) + if (target !== undefined) this.resolveCapacity(agent, target, state) } return state.status === 'ready' ? state.contextWindow : undefined } private resolveCapacity( agent: Agent, - target: Pick, + target: CapacityTarget, pending: CapacityState, ): void { const llm = this.ctx.get('llm') as LlmLike | undefined @@ -179,16 +185,27 @@ export class SessionMetricsProjector { .then(() => llm.resolveModelInfo(target.provider, target.model)) .then( (resolved) => { - if (this.capacities.get(agent)?.generation !== pending.generation) return - const current = this.targetFor(agent) - if (`${current.provider}\u0000${current.model}` !== pending.routeKey) return + if (this.capacityResolutionIsStale(agent, pending)) return pending.status = 'ready' if (resolved.context !== undefined) pending.contextWindow = resolved.context.contextWindow this.onCapacityResolved(agent) }, () => { - if (this.capacities.get(agent)?.generation === pending.generation) pending.status = 'ready' + if (!this.capacityResolutionIsStale(agent, pending)) pending.status = 'ready' }, ) } + + private capacityResolutionIsStale(agent: Agent, pending: CapacityState): boolean { + if (this.capacities.get(agent)?.generation !== pending.generation) return true + if (routeKeyFor(this.targetFor(agent)) === pending.routeKey) return false + // Unknown is the neutral generation; the next observed concrete route + // starts a fresh resolution even when it equals the route that disappeared. + this.capacities.set(agent, { + routeKey: undefined, + generation: pending.generation + 1, + status: 'ready', + }) + return true + } } diff --git a/packages/host/apiproxy/tests/api-proxy-models.spec.ts b/packages/host/apiproxy/tests/api-proxy-models.spec.ts index e0e7a035fa..1fc9f928e9 100644 --- a/packages/host/apiproxy/tests/api-proxy-models.spec.ts +++ b/packages/host/apiproxy/tests/api-proxy-models.spec.ts @@ -6,8 +6,8 @@ import { describe, expect, it } from 'vitest' import { Context } from 'cordis' -import AgentRegistry, { agentEvents } from '@deepseek-ai/dsh-agent' -import type { Agent } from '@deepseek-ai/dsh-agent' +import AgentRegistry, { agentEvents, installAgentLlmTarget } from '@deepseek-ai/dsh-agent' +import type { Agent, AgentLlmTargetRef } from '@deepseek-ai/dsh-agent' import LlmService, { LlmAdapter, ReasoningEffortId } from '@deepseek-ai/dsh-llm' import type { GenerateOptions, LlmCallConfig, LlmModelInfo, LlmModelReasoningInfo, LlmProviderInfo, @@ -72,15 +72,7 @@ const REASONING: LlmModelReasoningInfo = { defaultEffort: ReasoningEffortId('high'), } -async function harness(logged?: { - provider: string - model: string - reasoningEffort?: ReasoningEffortId -}): Promise<{ - ctx: Context - agent: Agent - sessionId: SessionId -}> { +async function hostContext(): Promise { const ctx = new Context() await ctx.plugin(SessionStore) await ctx.plugin(SystemPrompt, { persona: '' }) @@ -100,6 +92,19 @@ async function harness(logged?: { { provider: 'duplicate', id: 'same', name: 'Same' }, { provider: 'duplicate', id: 'same', name: 'Same Again' }, ])) + return ctx +} + +async function harness(logged?: { + provider: string + model: string + reasoningEffort?: ReasoningEffortId +}): Promise<{ + ctx: Context + agent: Agent + sessionId: SessionId +}> { + const ctx = await hostContext() const session = ctx.sessions.create() if (logged !== undefined) { session.append('request/header', { header: { config: logged }, reason: 'initial' }) @@ -246,6 +251,7 @@ describe('Web session model selection', () => { it('publishes unknown capacity immediately on selection, then the exact selected route capacity', async () => { const { ctx, sessionId } = await harness() const api = createApiProxy(ctx, { provider: 'deepseek', model: 'deepseek-chat', cwd: '/tmp', workspaceRoot: '/tmp' }) + expectValue(await api.sessions.models(request({ sessionId }))) const controller = new AbortController() const iterator = api.events.mux(request({}), controller.signal)[Symbol.asyncIterator]() @@ -264,4 +270,56 @@ describe('Web session model selection', () => { await iterator.return?.() await ctx.fiber.dispose() }) + + it('uses logged capacity without installing Web routing while scheduling foreign metrics', async () => { + const ctx = await hostContext() + const api = createApiProxy(ctx, { + provider: 'deepseek', + model: 'deepseek-chat', + cwd: '/tmp', + workspaceRoot: '/tmp', + }) + const controller = new AbortController() + const iterator = api.events.mux(request({}), controller.signal)[Symbol.asyncIterator]() + const initialMetrics = nextMetrics(iterator) + const session = ctx.sessions.create() + expect((await initialMetrics).contextWindow).toBeUndefined() + session.append('request/header', { + header: { config: { provider: 'deepseek', model: 'private-preview' } }, + reason: 'change', + }) + const foreign = { + id: session.id, + session, + status: 'running', + ctx, + } as Agent + const foreignTarget: AgentLlmTargetRef = { + current: { provider: 'foreign', model: 'foreign-model' }, + assembled: undefined, + } + const disposeForeignTarget = installAgentLlmTarget(foreign.ctx, foreignTarget) + const scheduledMetrics = nextMetrics(iterator) + ctx.agents.register(foreign) + + expect((await scheduledMetrics).contextWindow).toBeUndefined() + expect((await nextMetrics(iterator)).contextWindow).toBe(128_000) + expect((await ctx.systemPrompt.assemble()).variables) + .toMatchObject({ provider: 'foreign', model: 'foreign-model' }) + const seed: LlmCallConfig = { provider: 'seed', model: 'seed', temperature: 0.2 } + const signal = new AbortController().signal + await expect(agentEvents(ctx, foreign).waterfall( + 'agent/request', 1, 0, signal, () => Promise.resolve(seed), + )).resolves.toMatchObject({ provider: 'foreign', model: 'foreign-model' }) + + disposeForeignTarget() + expect((await ctx.systemPrompt.assemble()).variables).not.toHaveProperty('provider') + await expect(agentEvents(ctx, foreign).waterfall( + 'agent/request', 1, 1, signal, () => Promise.resolve(seed), + )).resolves.toBe(seed) + + controller.abort() + await iterator.return?.() + await ctx.fiber.dispose() + }) }) diff --git a/packages/host/apiproxy/tests/session-metrics.spec.ts b/packages/host/apiproxy/tests/session-metrics.spec.ts index 3ebfb1a0cf..c92e5a76e4 100644 --- a/packages/host/apiproxy/tests/session-metrics.spec.ts +++ b/packages/host/apiproxy/tests/session-metrics.spec.ts @@ -187,6 +187,37 @@ describe('SessionMetricsProjector', () => { }) }) + it('starts a fresh capacity generation when an unavailable route returns', async () => { + const ctx = new Context() + const resolutions: ((contextWindow: number) => void)[] = [] + ctx.provide('llm', { + resolveModelInfo() { + return new Promise<{ context: { contextWindow: number } }>((resolve) => { + resolutions.push((contextWindow) => { resolve({ context: { contextWindow } }) }) + }) + }, + }) + const session = new Session(SessionId('capacity-route-return')) + const attached = agent(session) + let current: AgentLlmTarget | undefined = { provider: 'test', model: 'alpha' } + const resolved = vi.fn() + const targetFor = vi.fn(() => current) + const projector = new SessionMetricsProjector(ctx, targetFor, resolved) + + expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() + await vi.waitFor(() => { expect(resolutions).toHaveLength(1) }) + current = undefined + resolutions[0]?.(64_000) + await vi.waitFor(() => { expect(targetFor).toHaveBeenCalledTimes(2) }) + expect(resolved).not.toHaveBeenCalled() + current = { provider: 'test', model: 'alpha' } + expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() + await vi.waitFor(() => { expect(resolutions).toHaveLength(2) }) + resolutions[1]?.(128_000) + await vi.waitFor(() => { expect(resolved).toHaveBeenCalledOnce() }) + expect(projector.snapshot(session, attached).contextWindow).toBe(128_000) + }) + it('omits current context fields when measurement or model metadata is unavailable', async () => { const ctx = new Context() ctx.provide('tokenMeter', { measure: () => { throw new Error('unmeasurable') } }) @@ -211,9 +242,10 @@ describe('SessionMetricsProjector', () => { const session = new Session(SessionId('optional-metrics')) assistant(session, 1, 0, { inputTokens: 7, outputTokens: 2 }) const attached = agent(session) + const selected: { current?: AgentLlmTarget } = {} const projector = new SessionMetricsProjector( ctx, - () => ({ provider: 'test', model: 'no-service' }), + () => selected.current, () => {}, ) @@ -224,6 +256,7 @@ describe('SessionMetricsProjector', () => { cacheWriteTokens: 0, }) expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() + selected.current = { provider: 'test', model: 'no-service' } expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() await Promise.resolve() }) From 765b360e264bbc7f781b3050caa759f46cca0f7b Mon Sep 17 00:00:00 2001 From: Hypatia May Date: Tue, 28 Jul 2026 14:29:11 +0800 Subject: [PATCH 04/38] fix(host): fence stale metric completions (round 4) --- packages/host/apiproxy/src/api-proxy.ts | 7 +- .../apiproxy/tests/api-proxy-models.spec.ts | 138 +++++++++++++++++- 2 files changed, 141 insertions(+), 4 deletions(-) diff --git a/packages/host/apiproxy/src/api-proxy.ts b/packages/host/apiproxy/src/api-proxy.ts index 84177de3cc..e55fc29f58 100644 --- a/packages/host/apiproxy/src/api-proxy.ts +++ b/packages/host/apiproxy/src/api-proxy.ts @@ -438,7 +438,12 @@ export function createApiProxy(ctx: Context, defaults: ApiProxyDefaults): ApiPro const metricsProjector = new SessionMetricsProjector( ctx, metricsRouteFor, - (agent) => { scheduleMetrics(agent.session) }, + (agent) => { + if (metricsDisposed) return + if (ctx.agents.get(agent.id) !== agent) return + if (ctx.sessions.get(agent.id) !== agent.session) return + scheduleMetrics(agent.session) + }, ) /** Queue one full-log metrics publication after synchronous session listeners drain. */ diff --git a/packages/host/apiproxy/tests/api-proxy-models.spec.ts b/packages/host/apiproxy/tests/api-proxy-models.spec.ts index 1fc9f928e9..366c3cd597 100644 --- a/packages/host/apiproxy/tests/api-proxy-models.spec.ts +++ b/packages/host/apiproxy/tests/api-proxy-models.spec.ts @@ -4,7 +4,7 @@ * models, and the prompt-assembly boundary for a running selection change. */ -import { describe, expect, it } from 'vitest' +import { describe, expect, it, vi } from 'vitest' import { Context } from 'cordis' import AgentRegistry, { agentEvents, installAgentLlmTarget } from '@deepseek-ai/dsh-agent' import type { Agent, AgentLlmTargetRef } from '@deepseek-ai/dsh-agent' @@ -13,8 +13,8 @@ import type { GenerateOptions, LlmCallConfig, LlmModelInfo, LlmModelReasoningInfo, LlmProviderInfo, LlmResolvedModelInfo, StreamChunk, } from '@deepseek-ai/dsh-llm' -import SessionStore from '@deepseek-ai/dsh-session' -import type { SessionId } from '@deepseek-ai/dsh-session' +import SessionStore, { SessionId } from '@deepseek-ai/dsh-session' +import type { Session } from '@deepseek-ai/dsh-session' import SystemPrompt from '@deepseek-ai/dsh-system-prompt' import UserInteractionService from '@deepseek-ai/dsh-user-interaction' import type { RpcRequest } from '@deepseek-ai/dsh-host-apiproxy/api/rpc' @@ -63,6 +63,33 @@ class CatalogAdapter extends LlmAdapter { } } +class DeferredCatalogAdapter extends CatalogAdapter { + readonly pending: PromiseWithResolvers[] = [] + + constructor() { + super('Deferred', [ + { provider: 'deferred', id: 'lifecycle-model', name: 'Lifecycle model' }, + ]) + } + + override resolveModel(_provider: string, _model: string): Promise { + const result = Promise.withResolvers() + this.pending.push(result) + return result.promise + } + + resolve(index: number, contextWindow: number): void { + const pending = this.pending[index] + if (pending === undefined) throw new Error(`no pending resolution at index ${String(index)}`) + pending.resolve({ + provider: 'deferred', + id: 'lifecycle-model', + name: 'Lifecycle model', + context: { contextWindow }, + }) + } +} + const REASONING: LlmModelReasoningInfo = { efforts: [ { id: ReasoningEffortId('off'), name: 'Off' }, @@ -134,6 +161,46 @@ async function nextMetrics( } } +function attachLifecycleSession( + ctx: Context, + sessionId: SessionId, + withMarker = false, +): { session: Session; detach: () => void } { + const session = ctx.sessions.prepare(sessionId) + session.append('request/header', { + header: { config: { provider: 'deferred', model: 'lifecycle-model' } }, + reason: 'initial', + }) + if (withMarker) { + session.append('user/message', { + content: [{ type: 'text', text: 'replacement marker' }], + source: { kind: 'plugin', plugin: 'test' }, + }, { surfaceOp: 'append' }) + } + const detach = ctx.sessions.enter(session) + ctx.sessions.announce(session) + return { session, detach } +} + +function attachLifecycleAgent( + ctx: Context, + session: Session, +): () => void { + const agent = { + id: session.id, + session, + status: 'running', + ctx, + } as Agent + const detach = ctx.agents.enter(agent, undefined) + ctx.agents.announce(agent) + return detach +} + +function settleCapacityCompletion(): Promise { + return new Promise((resolve) => { setImmediate(resolve) }) +} + describe('Web session model selection', () => { it('groups successful providers, isolates failures, and preserves an unlisted current model', async () => { const { ctx, sessionId } = await harness({ @@ -322,4 +389,69 @@ describe('Web session model selection', () => { await iterator.return?.() await ctx.fiber.dispose() }) + + it('drops capacity completion from a replaced agent that retains the exact session', async () => { + const ctx = await hostContext() + const deferred = new DeferredCatalogAdapter() + ctx.llm.registerAdapter(['deferred'], deferred) + const lifecycle = attachLifecycleSession(ctx, SessionId('capacity-agent-lifecycle')) + const retire = attachLifecycleAgent(ctx, lifecycle.session) + const api = createApiProxy(ctx, { provider: 'deepseek', model: 'deepseek-chat', cwd: '/tmp', workspaceRoot: '/tmp' }) + const controller = new AbortController() + const iterator = api.events.mux(request({}), controller.signal)[Symbol.asyncIterator]() + + expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() + await vi.waitFor(() => { expect(deferred.pending).toHaveLength(1) }) + retire() + const detachLive = attachLifecycleAgent(ctx, lifecycle.session) + expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() + await vi.waitFor(() => { expect(deferred.pending).toHaveLength(2) }) + + deferred.resolve(0, 64_000) + await settleCapacityCompletion() + deferred.resolve(1, 128_000) + await settleCapacityCompletion() + expect((await nextMetrics(iterator)).contextWindow).toBe(128_000) + + controller.abort() + await iterator.return?.() + detachLive() + lifecycle.detach() + await ctx.fiber.dispose() + }) + + it('drops capacity completion from a replaced session while its old agent remains live', async () => { + const ctx = await hostContext() + const deferred = new DeferredCatalogAdapter() + ctx.llm.registerAdapter(['deferred'], deferred) + const sessionId = SessionId('capacity-session-lifecycle') + const retiredSession = attachLifecycleSession(ctx, sessionId) + const retireAgent = attachLifecycleAgent(ctx, retiredSession.session) + const api = createApiProxy(ctx, { provider: 'deepseek', model: 'deepseek-chat', cwd: '/tmp', workspaceRoot: '/tmp' }) + const controller = new AbortController() + const iterator = api.events.mux(request({}), controller.signal)[Symbol.asyncIterator]() + + expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() + await vi.waitFor(() => { expect(deferred.pending).toHaveLength(1) }) + retiredSession.detach() + const liveSession = attachLifecycleSession(ctx, sessionId, true) + expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() + deferred.resolve(0, 64_000) + await settleCapacityCompletion() + retireAgent() + const detachLiveAgent = attachLifecycleAgent(ctx, liveSession.session) + const scheduled = await nextMetrics(iterator) + expect(scheduled.logRevision).toBe(2) + expect(scheduled.contextWindow).toBeUndefined() + await vi.waitFor(() => { expect(deferred.pending).toHaveLength(2) }) + deferred.resolve(1, 128_000) + await settleCapacityCompletion() + expect((await nextMetrics(iterator)).contextWindow).toBe(128_000) + + controller.abort() + await iterator.return?.() + detachLiveAgent() + liveSession.detach() + await ctx.fiber.dispose() + }) }) From 6f002007b08896a3641b3285d3bbeb42156a54b8 Mon Sep 17 00:00:00 2001 From: Hypatia May Date: Tue, 28 Jul 2026 14:51:11 +0800 Subject: [PATCH 05/38] fix(host): pair metric agents by lifecycle (round 5) --- packages/host/apiproxy/src/api-proxy.ts | 12 +++- .../apiproxy/tests/api-proxy-models.spec.ts | 55 +++++++++++++++++++ 2 files changed, 64 insertions(+), 3 deletions(-) diff --git a/packages/host/apiproxy/src/api-proxy.ts b/packages/host/apiproxy/src/api-proxy.ts index e55fc29f58..7eec1d8bb6 100644 --- a/packages/host/apiproxy/src/api-proxy.ts +++ b/packages/host/apiproxy/src/api-proxy.ts @@ -419,6 +419,12 @@ export function createApiProxy(ctx: Context, defaults: ApiProxyDefaults): ApiPro : { provider: logged.provider, model: logged.model } } + /** Pair a registry agent only with the exact Session lifecycle it owns. */ + function metricsAgentFor(session: Session): Agent | undefined { + const agent = ctx.agents.get(session.id) + return agent?.session === session ? agent : undefined + } + /** Pre-publication setup used by both fresh and resumed Web agents. */ function installTarget(agentCtx: Context): void { const agent = agentCtx.agent @@ -464,7 +470,7 @@ export function createApiProxy(ctx: Context, defaults: ApiProxyDefaults): ApiPro broadcast({ type: 'session/metrics', sessionId: current.id, - metrics: metricsProjector.snapshot(current, ctx.agents.get(current.id)), + metrics: metricsProjector.snapshot(current, metricsAgentFor(current)), }) } }) @@ -1186,7 +1192,7 @@ export function createApiProxy(ctx: Context, defaults: ApiProxyDefaults): ApiPro queue.push(frame({ type: 'session/metrics', sessionId: session.id, - metrics: metricsProjector.snapshot(session, ctx.agents.get(session.id)), + metrics: metricsProjector.snapshot(session, metricsAgentFor(session)), })) } for (const pending of pendingQuestions.values()) { @@ -1243,7 +1249,7 @@ export function createApiProxy(ctx: Context, defaults: ApiProxyDefaults): ApiPro queue.push(frame({ type: 'session/metrics', sessionId: session.id, - metrics: metricsProjector.snapshot(session, ctx.agents.get(session.id)), + metrics: metricsProjector.snapshot(session, metricsAgentFor(session)), })) }), ctx.on('session/disposed', (session: Session) => { diff --git a/packages/host/apiproxy/tests/api-proxy-models.spec.ts b/packages/host/apiproxy/tests/api-proxy-models.spec.ts index 366c3cd597..722c2e7412 100644 --- a/packages/host/apiproxy/tests/api-proxy-models.spec.ts +++ b/packages/host/apiproxy/tests/api-proxy-models.spec.ts @@ -454,4 +454,59 @@ describe('Web session model selection', () => { liveSession.detach() await ctx.fiber.dispose() }) + + it('does not project retired agent capacity into replacement session snapshots', async () => { + const ctx = await hostContext() + const deferred = new DeferredCatalogAdapter() + ctx.llm.registerAdapter(['deferred'], deferred) + const sessionId = SessionId('capacity-snapshot-lifecycle') + const retiredSession = attachLifecycleSession(ctx, sessionId) + const retireAgent = attachLifecycleAgent(ctx, retiredSession.session) + const api = createApiProxy(ctx, { provider: 'deepseek', model: 'deepseek-chat', cwd: '/tmp', workspaceRoot: '/tmp' }) + const primaryController = new AbortController() + const primary = api.events.mux(request({}), primaryController.signal)[Symbol.asyncIterator]() + + expect((await nextMetrics(primary)).contextWindow).toBeUndefined() + await vi.waitFor(() => { expect(deferred.pending).toHaveLength(1) }) + deferred.resolve(0, 64_000) + await settleCapacityCompletion() + expect((await nextMetrics(primary)).contextWindow).toBe(64_000) + + retiredSession.detach() + const replacement = attachLifecycleSession(ctx, sessionId) + const createdBaseline = await nextMetrics(primary) + replacement.session.append('user/message', { + content: [{ type: 'text', text: 'replacement marker' }], + source: { kind: 'plugin', plugin: 'test' }, + }, { surfaceOp: 'append' }) + const scheduledFlush = await nextMetrics(primary) + const reconnectController = new AbortController() + const reconnect = api.events.mux(request({}), reconnectController.signal)[Symbol.asyncIterator]() + const reconnectBaseline = await nextMetrics(reconnect) + expect(createdBaseline.logRevision).toBe(1) + for (const metrics of [scheduledFlush, reconnectBaseline]) { + expect(metrics.logRevision).toBe(2) + } + expect({ + created: createdBaseline.contextWindow, + scheduled: scheduledFlush.contextWindow, + reconnect: reconnectBaseline.contextWindow, + }).toEqual({ created: undefined, scheduled: undefined, reconnect: undefined }) + + retireAgent() + const detachReplacementAgent = attachLifecycleAgent(ctx, replacement.session) + expect((await nextMetrics(primary)).contextWindow).toBeUndefined() + await vi.waitFor(() => { expect(deferred.pending).toHaveLength(2) }) + deferred.resolve(1, 128_000) + await settleCapacityCompletion() + expect((await nextMetrics(primary)).contextWindow).toBe(128_000) + + primaryController.abort() + reconnectController.abort() + await primary.return?.() + await reconnect.return?.() + detachReplacementAgent() + replacement.detach() + await ctx.fiber.dispose() + }) }) From 03ca568246bff66a92f25f7d6313769338bbbfe5 Mon Sep 17 00:00:00 2001 From: Hypatia May Date: Tue, 28 Jul 2026 15:41:38 +0800 Subject: [PATCH 06/38] fix(host): refresh metric capacity metadata (round 6) --- packages/host/apiproxy/src/api-proxy.ts | 11 +++ packages/host/apiproxy/src/session-metrics.ts | 21 ++++-- .../apiproxy/tests/api-proxy-models.spec.ts | 69 +++++++++++++++++- .../apiproxy/tests/session-metrics.spec.ts | 71 +++++++++++++++++++ 4 files changed, 166 insertions(+), 6 deletions(-) diff --git a/packages/host/apiproxy/src/api-proxy.ts b/packages/host/apiproxy/src/api-proxy.ts index 7eec1d8bb6..b1c079a6d9 100644 --- a/packages/host/apiproxy/src/api-proxy.ts +++ b/packages/host/apiproxy/src/api-proxy.ts @@ -6,6 +6,7 @@ import { randomUUID } from 'node:crypto' import { mkdir, stat } from 'node:fs/promises' import { join } from 'node:path' +import { FiberState } from 'cordis' import type { Context } from 'cordis' import { installAgentLlmTarget } from '@deepseek-ai/dsh-agent' import type { @@ -483,6 +484,16 @@ export function createApiProxy(ctx: Context, defaults: ApiProxyDefaults): ApiPro }), ctx.on('agent/created', (agent: Agent) => { scheduleMetrics(agent.session) }), ctx.on('session/disposed', (session: Session) => { pendingMetricSessions.delete(session) }), + ctx.on('internal/status', (fiber) => { + if (metricsDisposed) return + if (fiber.state !== FiberState.ACTIVE + && fiber.state !== FiberState.FAILED + && fiber.state !== FiberState.DISPOSED) return + const sessions = ctx.get('sessions') + if (sessions === undefined) return + metricsProjector.invalidateCapacities() + for (const session of sessions.list()) scheduleMetrics(session) + }, { global: true }), ] return () => { metricsDisposed = true diff --git a/packages/host/apiproxy/src/session-metrics.ts b/packages/host/apiproxy/src/session-metrics.ts index 161f92941e..1ad1f488af 100644 --- a/packages/host/apiproxy/src/session-metrics.ts +++ b/packages/host/apiproxy/src/session-metrics.ts @@ -23,7 +23,8 @@ interface UsageState { interface CapacityState { routeKey: string | undefined generation: number - status: 'pending' | 'ready' + epoch: number + status: 'pending' | 'ready' | 'retryable' contextWindow?: number } @@ -88,6 +89,7 @@ function routeKeyFor(target: CapacityTarget | undefined): string | undefined { export class SessionMetricsProjector { private readonly usage = new WeakMap() private readonly capacities = new WeakMap() + private capacityEpoch = 0 /** * @param ctx - Host context providing optional token-meter and LLM services. @@ -100,6 +102,11 @@ export class SessionMetricsProjector { private readonly onCapacityResolved: (agent: Agent) => void, ) {} + /** Retire adapter-owned metadata and fence every resolution already in flight. */ + invalidateCapacities(): void { + this.capacityEpoch++ + } + /** * Read a fresh detached projection through the session's durable tail. * @param session - authoritative durable log owner. @@ -159,10 +166,14 @@ export class SessionMetricsProjector { const target = this.targetFor(agent) const routeKey = routeKeyFor(target) let state = this.capacities.get(agent) - if (state === undefined || state.routeKey !== routeKey) { + if (state === undefined + || state.routeKey !== routeKey + || state.epoch !== this.capacityEpoch + || state.status === 'retryable') { state = { routeKey, generation: (state?.generation ?? 0) + 1, + epoch: this.capacityEpoch, status: target === undefined ? 'ready' : 'pending', } this.capacities.set(agent, state) @@ -178,7 +189,7 @@ export class SessionMetricsProjector { ): void { const llm = this.ctx.get('llm') as LlmLike | undefined if (llm === undefined) { - pending.status = 'ready' + pending.status = 'retryable' return } void Promise.resolve() @@ -191,12 +202,13 @@ export class SessionMetricsProjector { this.onCapacityResolved(agent) }, () => { - if (!this.capacityResolutionIsStale(agent, pending)) pending.status = 'ready' + if (!this.capacityResolutionIsStale(agent, pending)) pending.status = 'retryable' }, ) } private capacityResolutionIsStale(agent: Agent, pending: CapacityState): boolean { + if (pending.epoch !== this.capacityEpoch) return true if (this.capacities.get(agent)?.generation !== pending.generation) return true if (routeKeyFor(this.targetFor(agent)) === pending.routeKey) return false // Unknown is the neutral generation; the next observed concrete route @@ -204,6 +216,7 @@ export class SessionMetricsProjector { this.capacities.set(agent, { routeKey: undefined, generation: pending.generation + 1, + epoch: this.capacityEpoch, status: 'ready', }) return true diff --git a/packages/host/apiproxy/tests/api-proxy-models.spec.ts b/packages/host/apiproxy/tests/api-proxy-models.spec.ts index 722c2e7412..5a1d9f02cf 100644 --- a/packages/host/apiproxy/tests/api-proxy-models.spec.ts +++ b/packages/host/apiproxy/tests/api-proxy-models.spec.ts @@ -6,6 +6,7 @@ import { describe, expect, it, vi } from 'vitest' import { Context } from 'cordis' +import type { Fiber } from 'cordis' import AgentRegistry, { agentEvents, installAgentLlmTarget } from '@deepseek-ai/dsh-agent' import type { Agent, AgentLlmTargetRef } from '@deepseek-ai/dsh-agent' import LlmService, { LlmAdapter, ReasoningEffortId } from '@deepseek-ai/dsh-llm' @@ -99,9 +100,10 @@ const REASONING: LlmModelReasoningInfo = { defaultEffort: ReasoningEffortId('high'), } -async function hostContext(): Promise { +async function hostContext(onSessions?: (fiber: Fiber) => void): Promise { const ctx = new Context() - await ctx.plugin(SessionStore) + const sessionsFiber = await ctx.plugin(SessionStore) + onSessions?.(sessionsFiber) await ctx.plugin(SystemPrompt, { persona: '' }) await ctx.plugin(LlmService) await ctx.plugin(UserInteractionService) @@ -201,6 +203,15 @@ function settleCapacityCompletion(): Promise { return new Promise((resolve) => { setImmediate(resolve) }) } +function installDeferredAdapter( + ctx: Context, + adapter: DeferredCatalogAdapter, +): Fiber & PromiseLike { + return ctx.plugin(Object.assign((inner: Context) => { + inner.llm.registerAdapter(['deferred'], adapter) + }, { inject: ['llm'] })) +} + describe('Web session model selection', () => { it('groups successful providers, isolates failures, and preserves an unlisted current model', async () => { const { ctx, sessionId } = await harness({ @@ -509,4 +520,58 @@ describe('Web session model selection', () => { replacement.detach() await ctx.fiber.dispose() }) + + it('refreshes same-route capacity after adapter owner replacement', async () => { + const ctx = await hostContext() + const retiredAdapter = new DeferredCatalogAdapter() + const retiredFiber = await installDeferredAdapter(ctx, retiredAdapter) + const lifecycle = attachLifecycleSession(ctx, SessionId('capacity-adapter-lifecycle')) + const detachAgent = attachLifecycleAgent(ctx, lifecycle.session) + const api = createApiProxy(ctx, { provider: 'deepseek', model: 'deepseek-chat', cwd: '/tmp', workspaceRoot: '/tmp' }) + const controller = new AbortController() + const iterator = api.events.mux(request({}), controller.signal)[Symbol.asyncIterator]() + + expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() + await vi.waitFor(() => { expect(retiredAdapter.pending).toHaveLength(1) }) + retiredAdapter.resolve(0, 64_000) + await settleCapacityCompletion() + expect((await nextMetrics(iterator)).contextWindow).toBe(64_000) + + await retiredFiber.dispose() + expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() + const replacementAdapter = new DeferredCatalogAdapter() + const replacementFiber = await installDeferredAdapter(ctx, replacementAdapter) + expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() + await vi.waitFor(() => { expect(replacementAdapter.pending).toHaveLength(1) }) + replacementAdapter.resolve(0, 128_000) + await settleCapacityCompletion() + expect((await nextMetrics(iterator)).contextWindow).toBe(128_000) + expect(lifecycle.session.requestHeader()?.config).toMatchObject({ + provider: 'deferred', + model: 'lifecycle-model', + }) + + controller.abort() + await iterator.return?.() + detachAgent() + lifecycle.detach() + await replacementFiber.dispose() + await ctx.fiber.dispose() + }) + + it('does not read the sessions service after its disposal status', async () => { + let sessionsFiber: Fiber | undefined + const ctx = await hostContext((fiber) => { sessionsFiber = fiber }) + if (sessionsFiber === undefined) throw new Error('sessions fiber missing') + createApiProxy(ctx, { provider: 'deepseek', model: 'deepseek-chat', cwd: '/tmp', workspaceRoot: '/tmp' }) + const sessions = ctx.get('sessions') + if (sessions === undefined) throw new Error('sessions service missing') + const list = vi.spyOn(sessions, 'list').mockImplementation(() => { + throw new Error('disposed sessions service read') + }) + + await expect(sessionsFiber.dispose()).resolves.toBeUndefined() + expect(list).not.toHaveBeenCalled() + await ctx.fiber.dispose() + }) }) diff --git a/packages/host/apiproxy/tests/session-metrics.spec.ts b/packages/host/apiproxy/tests/session-metrics.spec.ts index c92e5a76e4..3b3653aad8 100644 --- a/packages/host/apiproxy/tests/session-metrics.spec.ts +++ b/packages/host/apiproxy/tests/session-metrics.spec.ts @@ -33,6 +33,10 @@ function agent(session: Session): Agent { return { id: session.id, session } as Agent } +function settleAsyncWork(): Promise { + return new Promise((resolve) => { setImmediate(resolve) }) +} + describe('SessionMetricsProjector', () => { it('filters text/reasoning stream deltas while retaining usage, headers, and surface mutations', () => { const session = new Session(SessionId('metrics-filter')) @@ -187,6 +191,73 @@ describe('SessionMetricsProjector', () => { }) }) + it('retries a failed same-route capacity lookup only on the next snapshot', async () => { + const ctx = new Context() + const attempts: PromiseWithResolvers<{ context: { contextWindow: number } }>[] = [] + ctx.provide('llm', { + resolveModelInfo() { + const attempt = Promise.withResolvers<{ context: { contextWindow: number } }>() + attempts.push(attempt) + return attempt.promise + }, + }) + const session = new Session(SessionId('capacity-retry')) + const attached = agent(session) + const resolved = vi.fn() + const projector = new SessionMetricsProjector( + ctx, + () => ({ provider: 'test', model: 'alpha' }), + resolved, + ) + + expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() + await vi.waitFor(() => { expect(attempts).toHaveLength(1) }) + attempts[0]?.reject(new Error('metadata temporarily unavailable')) + await settleAsyncWork() + expect(attempts).toHaveLength(1) + expect(resolved).not.toHaveBeenCalled() + + expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() + await settleAsyncWork() + expect(attempts).toHaveLength(2) + attempts[1]?.resolve({ context: { contextWindow: 128_000 } }) + await vi.waitFor(() => { expect(resolved).toHaveBeenCalledOnce() }) + expect(projector.snapshot(session, attached).contextWindow).toBe(128_000) + }) + + it('invalidates same-route capacity and fences the prior epoch in flight', async () => { + const ctx = new Context() + const attempts: PromiseWithResolvers<{ context: { contextWindow: number } }>[] = [] + ctx.provide('llm', { + resolveModelInfo() { + const attempt = Promise.withResolvers<{ context: { contextWindow: number } }>() + attempts.push(attempt) + return attempt.promise + }, + }) + const session = new Session(SessionId('capacity-invalidation')) + const attached = agent(session) + const resolved = vi.fn() + const projector = new SessionMetricsProjector( + ctx, + () => ({ provider: 'test', model: 'alpha' }), + resolved, + ) + + expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() + await vi.waitFor(() => { expect(attempts).toHaveLength(1) }) + projector.invalidateCapacities() + attempts[0]?.resolve({ context: { contextWindow: 64_000 } }) + await settleAsyncWork() + expect(resolved).not.toHaveBeenCalled() + + expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() + await vi.waitFor(() => { expect(attempts).toHaveLength(2) }) + attempts[1]?.resolve({ context: { contextWindow: 128_000 } }) + await vi.waitFor(() => { expect(resolved).toHaveBeenCalledOnce() }) + expect(projector.snapshot(session, attached).contextWindow).toBe(128_000) + }) + it('starts a fresh capacity generation when an unavailable route returns', async () => { const ctx = new Context() const resolutions: ((contextWindow: number) => void)[] = [] From 8de47601415517df190187c420ae2b528d470e77 Mon Sep 17 00:00:00 2001 From: Hypatia May Date: Tue, 28 Jul 2026 15:56:28 +0800 Subject: [PATCH 07/38] fix(host): fence metric teardown reads (round 6 review) --- packages/host/apiproxy/src/api-proxy.ts | 8 +- .../apiproxy/tests/api-proxy-models.spec.ts | 114 +++++++++++++++++- 2 files changed, 116 insertions(+), 6 deletions(-) diff --git a/packages/host/apiproxy/src/api-proxy.ts b/packages/host/apiproxy/src/api-proxy.ts index b1c079a6d9..0b04395d90 100644 --- a/packages/host/apiproxy/src/api-proxy.ts +++ b/packages/host/apiproxy/src/api-proxy.ts @@ -447,8 +447,10 @@ export function createApiProxy(ctx: Context, defaults: ApiProxyDefaults): ApiPro metricsRouteFor, (agent) => { if (metricsDisposed) return - if (ctx.agents.get(agent.id) !== agent) return - if (ctx.sessions.get(agent.id) !== agent.session) return + const agents = ctx.get('agents') + if (agents?.get(agent.id) !== agent) return + const sessions = ctx.get('sessions') + if (sessions?.get(agent.id) !== agent.session) return scheduleMetrics(agent.session) }, ) @@ -489,9 +491,9 @@ export function createApiProxy(ctx: Context, defaults: ApiProxyDefaults): ApiPro if (fiber.state !== FiberState.ACTIVE && fiber.state !== FiberState.FAILED && fiber.state !== FiberState.DISPOSED) return + metricsProjector.invalidateCapacities() const sessions = ctx.get('sessions') if (sessions === undefined) return - metricsProjector.invalidateCapacities() for (const session of sessions.list()) scheduleMetrics(session) }, { global: true }), ] diff --git a/packages/host/apiproxy/tests/api-proxy-models.spec.ts b/packages/host/apiproxy/tests/api-proxy-models.spec.ts index 5a1d9f02cf..9955a91b79 100644 --- a/packages/host/apiproxy/tests/api-proxy-models.spec.ts +++ b/packages/host/apiproxy/tests/api-proxy-models.spec.ts @@ -5,7 +5,7 @@ */ import { describe, expect, it, vi } from 'vitest' -import { Context } from 'cordis' +import { Context, FiberState } from 'cordis' import type { Fiber } from 'cordis' import AgentRegistry, { agentEvents, installAgentLlmTarget } from '@deepseek-ai/dsh-agent' import type { Agent, AgentLlmTargetRef } from '@deepseek-ai/dsh-agent' @@ -100,14 +100,18 @@ const REASONING: LlmModelReasoningInfo = { defaultEffort: ReasoningEffortId('high'), } -async function hostContext(onSessions?: (fiber: Fiber) => void): Promise { +async function hostContext( + onSessions?: (fiber: Fiber) => void, + onAgents?: (fiber: Fiber) => void, +): Promise { const ctx = new Context() const sessionsFiber = await ctx.plugin(SessionStore) onSessions?.(sessionsFiber) await ctx.plugin(SystemPrompt, { persona: '' }) await ctx.plugin(LlmService) await ctx.plugin(UserInteractionService) - await ctx.plugin(AgentRegistry) + const agentsFiber = await ctx.plugin(AgentRegistry) + onAgents?.(agentsFiber) ctx.llm.registerAdapter(['deepseek'], new CatalogAdapter('DeepSeek', [ { provider: 'deepseek', id: 'deepseek-chat', name: 'DeepSeek Chat' }, { provider: 'deepseek', id: 'deepseek-reasoner', name: 'DeepSeek Reasoner', description: 'Reasoning model' }, @@ -574,4 +578,108 @@ describe('Web session model selection', () => { expect(list).not.toHaveBeenCalled() await ctx.fiber.dispose() }) + + it('invalidates pending capacity before SessionStore teardown can reach its callback', async () => { + let sessionsFiber: Fiber | undefined + const ctx = await hostContext((fiber) => { sessionsFiber = fiber }) + if (sessionsFiber === undefined) throw new Error('sessions fiber missing') + const deferred = new DeferredCatalogAdapter() + ctx.llm.registerAdapter(['deferred'], deferred) + const lifecycle = attachLifecycleSession(ctx, SessionId('capacity-session-store-teardown')) + const detachAgent = attachLifecycleAgent(ctx, lifecycle.session) + const api = createApiProxy(ctx, { provider: 'deepseek', model: 'deepseek-chat', cwd: '/tmp', workspaceRoot: '/tmp' }) + const controller = new AbortController() + const iterator = api.events.mux(request({}), controller.signal)[Symbol.asyncIterator]() + + expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() + await vi.waitFor(() => { expect(deferred.pending).toHaveLength(1) }) + const agents = ctx.get('agents') + if (agents === undefined) throw new Error('agent registry missing') + expect(agents.get(lifecycle.session.id)).toBeDefined() + const getAgent = vi.spyOn(agents, 'get') + + await sessionsFiber.dispose() + getAgent.mockClear() + deferred.resolve(0, 64_000) + await settleCapacityCompletion() + expect(getAgent).not.toHaveBeenCalled() + + const pendingFrame = iterator.next() + const outcome = await Promise.race([ + pendingFrame.then(() => 'frame' as const), + new Promise<'idle'>((resolve) => { setImmediate(() => { resolve('idle') }) }), + ]) + expect(outcome).toBe('idle') + + controller.abort() + await expect(pendingFrame).resolves.toMatchObject({ done: true }) + await iterator.return?.() + detachAgent() + lifecycle.detach() + await ctx.fiber.dispose() + }) + + it.each(['agents', 'sessions'] as const)( + 'drops capacity completion while %s is unavailable during unload', + async (serviceName) => { + let sessionsFiber: Fiber | undefined + let agentsFiber: Fiber | undefined + const ctx = await hostContext( + (fiber) => { sessionsFiber = fiber }, + (fiber) => { agentsFiber = fiber }, + ) + const heldFiber = serviceName === 'sessions' ? sessionsFiber : agentsFiber + if (heldFiber === undefined) throw new Error(`${serviceName} fiber missing`) + const unloadStarted = Promise.withResolvers() + const releaseUnload = Promise.withResolvers() + heldFiber.ctx.effect(() => () => { + unloadStarted.resolve(undefined) + return releaseUnload.promise + }, `test: hold ${serviceName} unload`) + const deferred = new DeferredCatalogAdapter() + ctx.llm.registerAdapter(['deferred'], deferred) + const lifecycle = attachLifecycleSession(ctx, SessionId(`capacity-${serviceName}-unloading`)) + const detachAgent = attachLifecycleAgent(ctx, lifecycle.session) + const api = createApiProxy(ctx, { provider: 'deepseek', model: 'deepseek-chat', cwd: '/tmp', workspaceRoot: '/tmp' }) + const controller = new AbortController() + const iterator = api.events.mux(request({}), controller.signal)[Symbol.asyncIterator]() + + expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() + await vi.waitFor(() => { expect(deferred.pending).toHaveLength(1) }) + const agents = ctx.get('agents') + if (agents === undefined) throw new Error('agent registry missing') + const sessions = ctx.get('sessions') + if (sessions === undefined) throw new Error('sessions service missing') + const getAgent = vi.spyOn(agents, 'get') + const getSession = vi.spyOn(sessions, 'get') + const disposing = heldFiber.dispose() + await unloadStarted.promise + await vi.waitFor(() => { expect(ctx.get(serviceName)).toBeUndefined() }) + expect(heldFiber.state).toBe(FiberState.UNLOADING) + + getAgent.mockClear() + getSession.mockClear() + deferred.resolve(0, 64_000) + await settleCapacityCompletion() + const agentReads = getAgent.mock.calls.length + const sessionReads = getSession.mock.calls.length + const pendingFrame = iterator.next() + const outcome = await Promise.race([ + pendingFrame.then(() => 'frame' as const), + new Promise<'idle'>((resolve) => { setImmediate(() => { resolve('idle') }) }), + ]) + + controller.abort() + await expect(pendingFrame).resolves.toMatchObject({ done: true }) + await iterator.return?.() + releaseUnload.resolve(undefined) + await disposing + detachAgent() + lifecycle.detach() + await ctx.fiber.dispose() + expect(agentReads).toBe(serviceName === 'sessions' ? 1 : 0) + expect(sessionReads).toBe(0) + expect(outcome).toBe('idle') + }, + ) }) From 5340aedb8bdce3c36f9101dd48948c1976a6c70f Mon Sep 17 00:00:00 2001 From: Hypatia May Date: Tue, 28 Jul 2026 16:05:43 +0800 Subject: [PATCH 08/38] fix(host): guard terminal metric snapshots (round 6 review) --- packages/host/apiproxy/src/api-proxy.ts | 2 +- .../apiproxy/tests/api-proxy-models.spec.ts | 43 +++++++++++++++++++ 2 files changed, 44 insertions(+), 1 deletion(-) diff --git a/packages/host/apiproxy/src/api-proxy.ts b/packages/host/apiproxy/src/api-proxy.ts index 0b04395d90..149143b900 100644 --- a/packages/host/apiproxy/src/api-proxy.ts +++ b/packages/host/apiproxy/src/api-proxy.ts @@ -422,7 +422,7 @@ export function createApiProxy(ctx: Context, defaults: ApiProxyDefaults): ApiPro /** Pair a registry agent only with the exact Session lifecycle it owns. */ function metricsAgentFor(session: Session): Agent | undefined { - const agent = ctx.agents.get(session.id) + const agent = ctx.get('agents')?.get(session.id) return agent?.session === session ? agent : undefined } diff --git a/packages/host/apiproxy/tests/api-proxy-models.spec.ts b/packages/host/apiproxy/tests/api-proxy-models.spec.ts index 9955a91b79..758669f5db 100644 --- a/packages/host/apiproxy/tests/api-proxy-models.spec.ts +++ b/packages/host/apiproxy/tests/api-proxy-models.spec.ts @@ -619,6 +619,49 @@ describe('Web session model selection', () => { await ctx.fiber.dispose() }) + it('publishes unknown metrics after AgentRegistry terminal disposal with mux active', async () => { + let agentsFiber: Fiber | undefined + const ctx = await hostContext(undefined, (fiber) => { agentsFiber = fiber }) + if (agentsFiber === undefined) throw new Error('agent registry fiber missing') + const deferred = new DeferredCatalogAdapter() + ctx.llm.registerAdapter(['deferred'], deferred) + const lifecycle = attachLifecycleSession(ctx, SessionId('capacity-agent-registry-disposed')) + const detachAgent = attachLifecycleAgent(ctx, lifecycle.session) + const api = createApiProxy(ctx, { provider: 'deepseek', model: 'deepseek-chat', cwd: '/tmp', workspaceRoot: '/tmp' }) + const controller = new AbortController() + const iterator = api.events.mux(request({}), controller.signal)[Symbol.asyncIterator]() + + expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() + await vi.waitFor(() => { expect(deferred.pending).toHaveLength(1) }) + deferred.resolve(0, 64_000) + await settleCapacityCompletion() + expect((await nextMetrics(iterator)).contextWindow).toBe(64_000) + + await agentsFiber.dispose() + const refresh = nextMetrics(iterator).then( + metrics => ({ kind: 'metrics' as const, metrics }), + () => ({ kind: 'error' as const }), + ) + const outcome = await Promise.race([ + refresh, + new Promise<{ kind: 'idle' }>((resolve) => { + setImmediate(() => { resolve({ kind: 'idle' }) }) + }), + ]) + + controller.abort() + await refresh + await iterator.return?.() + detachAgent() + lifecycle.detach() + await ctx.fiber.dispose() + expect(outcome.kind).toBe('metrics') + if (outcome.kind === 'metrics') { + expect(outcome.metrics.contextWindow).toBeUndefined() + expect(outcome.metrics.logRevision).toBe(1) + } + }) + it.each(['agents', 'sessions'] as const)( 'drops capacity completion while %s is unavailable during unload', async (serviceName) => { From 032cd2f72dc5592ec976cfa3cd78c2aafbcb6041 Mon Sep 17 00:00:00 2001 From: Hypatia May Date: Tue, 28 Jul 2026 16:19:40 +0800 Subject: [PATCH 09/38] docs: refresh event consumer graph after master merge --- docs/event-producer-consumer.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/event-producer-consumer.md b/docs/event-producer-consumer.md index fbb98faa71..2b719de804 100644 --- a/docs/event-producer-consumer.md +++ b/docs/event-producer-consumer.md @@ -67,7 +67,7 @@ This matrix shows which packages dispatch each harness-owned event and which pac | `connection/reset` | `runtime` (`emit`) | - | | `internal/dispatch` | - | [`compact`](../packages/compact/compact), [`fs`](../packages/fs/fs), [`goal`](../packages/goal/goal), [`goal-session`](../packages/goal/goal-session), [`hook-protocol`](../packages/hooks/hook-protocol), [`llm-retry`](../packages/llm/llm-retry), [`permission`](../packages/ui/permission), [`plan-mode`](../packages/plan/plan-mode), [`pty-local`](../packages/pty/pty-local), `runtime`, [`sandbox-policy`](../packages/sandbox/sandbox-policy), [`scope`](../packages/core/scope), [`session`](../packages/core/session), [`subagent`](../packages/subagent/subagent), [`time-context`](../packages/context/time-context), [`tool-todo`](../packages/todo/tool-todo), [`tools`](../packages/core/tools), [`user-approval`](../packages/ui/user-approval), [`workflow`](../packages/workflow/workflow) | | `internal/plugin` | - | `hmr`, `modules`, `webserver` | -| `internal/status` | - | [`agent`](../packages/core/agent) | +| `internal/status` | - | [`agent`](../packages/core/agent), `apiproxy` | | `locale/change` | `locale` (`emit`) | `locale`, `ui-models`, `ui-settings-general` | | `slots/changed` | `runtime` (`emit`) | - | | `theme/change` | `ui-theme` (`emit`) | `ui-layout`, `ui-theme` | From 344667cbf0c3dfd071a3f3a9ba0f32acc007f0ac Mon Sep 17 00:00:00 2001 From: Hypatia May Date: Tue, 28 Jul 2026 17:03:07 +0800 Subject: [PATCH 10/38] fix(host): cancel retired metric capacity lookups (round 7) --- packages/host/apiproxy/src/api-proxy.ts | 5 + packages/host/apiproxy/src/session-metrics.ts | 37 ++++- .../apiproxy/tests/api-proxy-models.spec.ts | 151 +++++++++++++++++- .../apiproxy/tests/session-metrics.spec.ts | 123 +++++++++++--- 4 files changed, 287 insertions(+), 29 deletions(-) diff --git a/packages/host/apiproxy/src/api-proxy.ts b/packages/host/apiproxy/src/api-proxy.ts index 149143b900..46ed485d45 100644 --- a/packages/host/apiproxy/src/api-proxy.ts +++ b/packages/host/apiproxy/src/api-proxy.ts @@ -488,6 +488,10 @@ export function createApiProxy(ctx: Context, defaults: ApiProxyDefaults): ApiPro ctx.on('session/disposed', (session: Session) => { pendingMetricSessions.delete(session) }), ctx.on('internal/status', (fiber) => { if (metricsDisposed) return + if (fiber.state === FiberState.UNLOADING) { + metricsProjector.invalidateCapacities() + return + } if (fiber.state !== FiberState.ACTIVE && fiber.state !== FiberState.FAILED && fiber.state !== FiberState.DISPOSED) return @@ -499,6 +503,7 @@ export function createApiProxy(ctx: Context, defaults: ApiProxyDefaults): ApiPro ] return () => { metricsDisposed = true + metricsProjector.dispose() pendingMetricSessions.clear() for (const dispose of disposers) dispose() } diff --git a/packages/host/apiproxy/src/session-metrics.ts b/packages/host/apiproxy/src/session-metrics.ts index 1ad1f488af..1eebc7b7d8 100644 --- a/packages/host/apiproxy/src/session-metrics.ts +++ b/packages/host/apiproxy/src/session-metrics.ts @@ -26,6 +26,7 @@ interface CapacityState { epoch: number status: 'pending' | 'ready' | 'retryable' contextWindow?: number + controller?: AbortController } type CapacityTarget = Pick @@ -35,7 +36,7 @@ interface TokenMeterLike { } interface LlmLike { - resolveModelInfo(provider: string, model: string): Promise<{ + resolveModelInfo(provider: string, model: string, signal?: AbortSignal): Promise<{ context?: { contextWindow: number } }> } @@ -89,7 +90,9 @@ function routeKeyFor(target: CapacityTarget | undefined): string | undefined { export class SessionMetricsProjector { private readonly usage = new WeakMap() private readonly capacities = new WeakMap() + private readonly pendingCapacities = new Set() private capacityEpoch = 0 + private disposed = false /** * @param ctx - Host context providing optional token-meter and LLM services. @@ -105,6 +108,13 @@ export class SessionMetricsProjector { /** Retire adapter-owned metadata and fence every resolution already in flight. */ invalidateCapacities(): void { this.capacityEpoch++ + for (const pending of this.pendingCapacities) this.abortCapacityResolution(pending) + } + + /** Permanently retire capacity projection and cancel every adapter-owned lookup. */ + dispose(): void { + this.disposed = true + this.invalidateCapacities() } /** @@ -163,6 +173,7 @@ export class SessionMetricsProjector { } private capacityFor(agent: Agent): number | undefined { + if (this.disposed) return undefined const target = this.targetFor(agent) const routeKey = routeKeyFor(target) let state = this.capacities.get(agent) @@ -170,6 +181,7 @@ export class SessionMetricsProjector { || state.routeKey !== routeKey || state.epoch !== this.capacityEpoch || state.status === 'retryable') { + this.abortCapacityResolution(state) state = { routeKey, generation: (state?.generation ?? 0) + 1, @@ -192,21 +204,42 @@ export class SessionMetricsProjector { pending.status = 'retryable' return } + const controller = new AbortController() + pending.controller = controller + this.pendingCapacities.add(pending) void Promise.resolve() - .then(() => llm.resolveModelInfo(target.provider, target.model)) + .then(() => { + controller.signal.throwIfAborted() + return llm.resolveModelInfo(target.provider, target.model, controller.signal) + }) .then( (resolved) => { + this.finishCapacityResolution(pending, controller) if (this.capacityResolutionIsStale(agent, pending)) return pending.status = 'ready' if (resolved.context !== undefined) pending.contextWindow = resolved.context.contextWindow this.onCapacityResolved(agent) }, () => { + this.finishCapacityResolution(pending, controller) if (!this.capacityResolutionIsStale(agent, pending)) pending.status = 'retryable' }, ) } + private abortCapacityResolution(pending: CapacityState | undefined): void { + if (pending === undefined || pending.controller === undefined) return + const controller = pending.controller + delete pending.controller + this.pendingCapacities.delete(pending) + controller.abort() + } + + private finishCapacityResolution(pending: CapacityState, controller: AbortController): void { + this.pendingCapacities.delete(pending) + if (pending.controller === controller) delete pending.controller + } + private capacityResolutionIsStale(agent: Agent, pending: CapacityState): boolean { if (pending.epoch !== this.capacityEpoch) return true if (this.capacities.get(agent)?.generation !== pending.generation) return true diff --git a/packages/host/apiproxy/tests/api-proxy-models.spec.ts b/packages/host/apiproxy/tests/api-proxy-models.spec.ts index 758669f5db..7b2595e528 100644 --- a/packages/host/apiproxy/tests/api-proxy-models.spec.ts +++ b/packages/host/apiproxy/tests/api-proxy-models.spec.ts @@ -65,7 +65,10 @@ class CatalogAdapter extends LlmAdapter { } class DeferredCatalogAdapter extends CatalogAdapter { - readonly pending: PromiseWithResolvers[] = [] + readonly pending: { + result: PromiseWithResolvers + signal: AbortSignal | undefined + }[] = [] constructor() { super('Deferred', [ @@ -73,16 +76,20 @@ class DeferredCatalogAdapter extends CatalogAdapter { ]) } - override resolveModel(_provider: string, _model: string): Promise { + override resolveModel( + _provider: string, + _model: string, + signal?: AbortSignal, + ): Promise { const result = Promise.withResolvers() - this.pending.push(result) + this.pending.push({ result, signal }) return result.promise } resolve(index: number, contextWindow: number): void { const pending = this.pending[index] if (pending === undefined) throw new Error(`no pending resolution at index ${String(index)}`) - pending.resolve({ + pending.result.resolve({ provider: 'deferred', id: 'lifecycle-model', name: 'Lifecycle model', @@ -563,6 +570,140 @@ describe('Web session model selection', () => { await ctx.fiber.dispose() }) + it('aborts pending capacity during adapter UNLOADING and refreshes after settlement', async () => { + const ctx = await hostContext() + const retiredAdapter = new DeferredCatalogAdapter() + const releaseUnload = Promise.withResolvers() + const releaseCancellationWait = Promise.withResolvers() + const abortObserved = Promise.withResolvers() + const retiredFiber = await ctx.plugin(Object.assign((inner: Context) => { + inner.llm.registerAdapter(['deferred'], retiredAdapter) + inner.effect( + () => () => releaseUnload.promise, + 'test: hold adapter unload', + ) + inner.effect(() => () => { + const signal = retiredAdapter.pending[0]?.signal + if (signal === undefined) throw new Error('pending capacity signal missing') + const cancellation = new Promise((resolve) => { + const finish = () => { + abortObserved.resolve(undefined) + resolve() + } + if (signal.aborted) finish() + else signal.addEventListener('abort', finish, { once: true }) + }) + return Promise.race([cancellation, releaseCancellationWait.promise]) + }, 'test: await capacity cancellation') + }, { inject: ['llm'] })) + const lifecycle = attachLifecycleSession(ctx, SessionId('capacity-adapter-unloading')) + const detachAgent = attachLifecycleAgent(ctx, lifecycle.session) + const api = createApiProxy(ctx, { + provider: 'deepseek', + model: 'deepseek-chat', + cwd: '/tmp', + workspaceRoot: '/tmp', + }) + const controller = new AbortController() + const iterator = api.events.mux(request({}), controller.signal)[Symbol.asyncIterator]() + const resolveModelInfo = vi.spyOn(ctx.llm, 'resolveModelInfo') + + expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() + await vi.waitFor(() => { expect(retiredAdapter.pending).toHaveLength(1) }) + expect(resolveModelInfo).toHaveBeenCalledOnce() + const listSessions = vi.spyOn(ctx.sessions, 'list') + listSessions.mockClear() + const pendingFrame = iterator.next() + const disposing = retiredFiber.dispose() + try { + await vi.waitFor(() => { + expect(retiredAdapter.pending[0]?.signal?.aborted).toBe(true) + }) + await abortObserved.promise + expect(retiredFiber.state).toBe(FiberState.UNLOADING) + retiredAdapter.resolve(0, 64_000) + await settleCapacityCompletion() + expect(resolveModelInfo).toHaveBeenCalledOnce() + expect(listSessions).not.toHaveBeenCalled() + const outcome = await Promise.race([ + pendingFrame.then(() => 'frame' as const), + new Promise<'idle'>((resolve) => { setImmediate(() => { resolve('idle') }) }), + ]) + expect(outcome).toBe('idle') + } finally { + releaseCancellationWait.resolve(undefined) + releaseUnload.resolve(undefined) + await disposing + } + + expect(retiredFiber.state).toBe(FiberState.DISPOSED) + const settledFrame = await pendingFrame + if (settledFrame.done || settledFrame.value.payload.type !== 'session/metrics') { + throw new Error('expected settled metrics refresh') + } + expect(settledFrame.value.payload.metrics.contextWindow).toBeUndefined() + await settleCapacityCompletion() + expect(resolveModelInfo).toHaveBeenCalledTimes(2) + + const replacementAdapter = new DeferredCatalogAdapter() + const replacementFiber = await installDeferredAdapter(ctx, replacementAdapter) + expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() + await vi.waitFor(() => { expect(replacementAdapter.pending).toHaveLength(1) }) + expect(resolveModelInfo).toHaveBeenCalledTimes(3) + replacementAdapter.resolve(0, 128_000) + await settleCapacityCompletion() + expect((await nextMetrics(iterator)).contextWindow).toBe(128_000) + + controller.abort() + await iterator.return?.() + detachAgent() + lifecycle.detach() + await replacementFiber.dispose() + await ctx.fiber.dispose() + }) + + it('aborts pending capacity when the API proxy fiber is disposed', async () => { + const ctx = await hostContext() + const deferred = new DeferredCatalogAdapter() + ctx.llm.registerAdapter(['deferred'], deferred) + const lifecycle = attachLifecycleSession(ctx, SessionId('capacity-api-proxy-teardown')) + const detachAgent = attachLifecycleAgent(ctx, lifecycle.session) + const proxy = Promise.withResolvers>() + const proxyFiber = await ctx.plugin(Object.assign((inner: Context) => { + proxy.resolve(createApiProxy(inner, { + provider: 'deepseek', + model: 'deepseek-chat', + cwd: '/tmp', + workspaceRoot: '/tmp', + })) + }, { inject: ['agents', 'sessions', 'userInteraction'] })) + const api = await proxy.promise + const controller = new AbortController() + const iterator = api.events.mux(request({}), controller.signal)[Symbol.asyncIterator]() + + expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() + await vi.waitFor(() => { expect(deferred.pending).toHaveLength(1) }) + expect(deferred.pending[0]?.signal?.aborted).toBe(false) + + await proxyFiber.dispose() + expect(deferred.pending[0]?.signal?.aborted).toBe(true) + deferred.resolve(0, 64_000) + await settleCapacityCompletion() + const pendingFrame = iterator.next() + const outcome = await Promise.race([ + pendingFrame.then(() => 'frame' as const), + new Promise<'idle'>((resolve) => { setImmediate(() => { resolve('idle') }) }), + ]) + expect(outcome).toBe('idle') + + controller.abort() + await expect(pendingFrame).resolves.toMatchObject({ done: true }) + await iterator.return?.() + detachAgent() + lifecycle.detach() + await ctx.fiber.dispose() + }) + it('does not read the sessions service after its disposal status', async () => { let sessionsFiber: Fiber | undefined const ctx = await hostContext((fiber) => { sessionsFiber = fiber }) @@ -720,7 +861,7 @@ describe('Web session model selection', () => { detachAgent() lifecycle.detach() await ctx.fiber.dispose() - expect(agentReads).toBe(serviceName === 'sessions' ? 1 : 0) + expect(agentReads).toBe(0) expect(sessionReads).toBe(0) expect(outcome).toBe('idle') }, diff --git a/packages/host/apiproxy/tests/session-metrics.spec.ts b/packages/host/apiproxy/tests/session-metrics.spec.ts index 3b3653aad8..deb5f740fe 100644 --- a/packages/host/apiproxy/tests/session-metrics.spec.ts +++ b/packages/host/apiproxy/tests/session-metrics.spec.ts @@ -159,12 +159,20 @@ describe('SessionMetricsProjector', () => { it('publishes only the selected route capacity when asynchronous resolutions race', async () => { const ctx = new Context() - const resolutions = new Map void>() + const resolutions = new Map() ctx.provide('tokenMeter', { measure: () => ({ totalTokens: 35_000 }) }) ctx.provide('llm', { - resolveModelInfo(_provider: string, model: string) { + resolveModelInfo(_provider: string, model: string, signal?: AbortSignal) { return new Promise<{ context: { contextWindow: number } }>((resolve) => { - resolutions.set(model, (contextWindow) => { resolve({ context: { contextWindow } }) }) + resolutions.set(model, { + signal, + resolve(contextWindow) { + resolve({ context: { contextWindow } }) + }, + }) }) }, }) @@ -176,14 +184,17 @@ describe('SessionMetricsProjector', () => { expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() await vi.waitFor(() => { expect(resolutions.has('alpha')).toBe(true) }) + expect(resolutions.get('alpha')?.signal?.aborted).toBe(false) current = { provider: 'test', model: 'beta' } expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() + expect(resolutions.get('alpha')?.signal?.aborted).toBe(true) await vi.waitFor(() => { expect(resolutions.has('beta')).toBe(true) }) + expect(resolutions.get('beta')?.signal?.aborted).toBe(false) - resolutions.get('alpha')?.(64_000) + resolutions.get('alpha')?.resolve(64_000) await Promise.resolve() expect(resolved).not.toHaveBeenCalled() - resolutions.get('beta')?.(128_000) + resolutions.get('beta')?.resolve(128_000) await vi.waitFor(() => { expect(resolved).toHaveBeenCalledOnce() }) expect(projector.snapshot(session, attached)).toMatchObject({ contextTokens: 35_000, @@ -225,17 +236,74 @@ describe('SessionMetricsProjector', () => { expect(projector.snapshot(session, attached).contextWindow).toBe(128_000) }) - it('invalidates same-route capacity and fences the prior epoch in flight', async () => { + it('aborts every active capacity on invalidation and resolves fresh generations', async () => { const ctx = new Context() - const attempts: PromiseWithResolvers<{ context: { contextWindow: number } }>[] = [] + const attempts: { + result: PromiseWithResolvers<{ context: { contextWindow: number } }> + signal: AbortSignal | undefined + }[] = [] ctx.provide('llm', { - resolveModelInfo() { - const attempt = Promise.withResolvers<{ context: { contextWindow: number } }>() - attempts.push(attempt) - return attempt.promise + resolveModelInfo(_provider: string, _model: string, signal?: AbortSignal) { + const result = Promise.withResolvers<{ context: { contextWindow: number } }>() + attempts.push({ result, signal }) + return result.promise }, }) - const session = new Session(SessionId('capacity-invalidation')) + const firstSession = new Session(SessionId('capacity-invalidation-first')) + const secondSession = new Session(SessionId('capacity-invalidation-second')) + const firstAgent = agent(firstSession) + const secondAgent = agent(secondSession) + const resolved = vi.fn() + const projector = new SessionMetricsProjector( + ctx, + () => ({ provider: 'test', model: 'alpha' }), + resolved, + ) + + expect(projector.snapshot(firstSession, firstAgent).contextWindow).toBeUndefined() + expect(projector.snapshot(secondSession, secondAgent).contextWindow).toBeUndefined() + await vi.waitFor(() => { expect(attempts).toHaveLength(2) }) + expect(attempts.map(attempt => attempt.signal?.aborted)).toEqual([false, false]) + + projector.invalidateCapacities() + expect(attempts.map(attempt => attempt.signal?.aborted)).toEqual([true, true]) + attempts[0]?.result.resolve({ context: { contextWindow: 32_000 } }) + attempts[1]?.result.resolve({ context: { contextWindow: 64_000 } }) + await settleAsyncWork() + expect(resolved).not.toHaveBeenCalled() + + expect(projector.snapshot(firstSession, firstAgent).contextWindow).toBeUndefined() + expect(projector.snapshot(secondSession, secondAgent).contextWindow).toBeUndefined() + await vi.waitFor(() => { expect(attempts).toHaveLength(4) }) + expect(attempts.slice(2).map(attempt => attempt.signal?.aborted)).toEqual([false, false]) + attempts[2]?.result.resolve({ context: { contextWindow: 128_000 } }) + attempts[3]?.result.resolve({ context: { contextWindow: 256_000 } }) + await vi.waitFor(() => { expect(resolved).toHaveBeenCalledTimes(2) }) + expect(projector.snapshot(firstSession, firstAgent).contextWindow).toBe(128_000) + expect(projector.snapshot(secondSession, secondAgent).contextWindow).toBe(256_000) + + projector.dispose() + expect(projector.snapshot(firstSession, firstAgent).contextWindow).toBeUndefined() + expect(attempts).toHaveLength(4) + }) + + it('skips adapter work invalidated before its deferred invocation', async () => { + const ctx = new Context() + const attempts: { + result: PromiseWithResolvers<{ context: { contextWindow: number } }> + signal: AbortSignal | undefined + }[] = [] + const resolveModelInfo = vi.fn(( + _provider: string, + _model: string, + signal?: AbortSignal, + ) => { + const result = Promise.withResolvers<{ context: { contextWindow: number } }>() + attempts.push({ result, signal }) + return result.promise + }) + ctx.provide('llm', { resolveModelInfo }) + const session = new Session(SessionId('capacity-pre-invocation-invalidation')) const attached = agent(session) const resolved = vi.fn() const projector = new SessionMetricsProjector( @@ -245,26 +313,34 @@ describe('SessionMetricsProjector', () => { ) expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() - await vi.waitFor(() => { expect(attempts).toHaveLength(1) }) projector.invalidateCapacities() - attempts[0]?.resolve({ context: { contextWindow: 64_000 } }) await settleAsyncWork() + expect(resolveModelInfo).not.toHaveBeenCalled() expect(resolved).not.toHaveBeenCalled() expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() - await vi.waitFor(() => { expect(attempts).toHaveLength(2) }) - attempts[1]?.resolve({ context: { contextWindow: 128_000 } }) + await vi.waitFor(() => { expect(attempts).toHaveLength(1) }) + expect(attempts[0]?.signal?.aborted).toBe(false) + attempts[0]?.result.resolve({ context: { contextWindow: 128_000 } }) await vi.waitFor(() => { expect(resolved).toHaveBeenCalledOnce() }) expect(projector.snapshot(session, attached).contextWindow).toBe(128_000) }) it('starts a fresh capacity generation when an unavailable route returns', async () => { const ctx = new Context() - const resolutions: ((contextWindow: number) => void)[] = [] + const resolutions: { + signal: AbortSignal | undefined + resolve(contextWindow: number): void + }[] = [] ctx.provide('llm', { - resolveModelInfo() { + resolveModelInfo(_provider: string, _model: string, signal?: AbortSignal) { return new Promise<{ context: { contextWindow: number } }>((resolve) => { - resolutions.push((contextWindow) => { resolve({ context: { contextWindow } }) }) + resolutions.push({ + signal, + resolve(contextWindow) { + resolve({ context: { contextWindow } }) + }, + }) }) }, }) @@ -278,13 +354,16 @@ describe('SessionMetricsProjector', () => { expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() await vi.waitFor(() => { expect(resolutions).toHaveLength(1) }) current = undefined - resolutions[0]?.(64_000) - await vi.waitFor(() => { expect(targetFor).toHaveBeenCalledTimes(2) }) + expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() + expect(resolutions[0]?.signal?.aborted).toBe(true) + resolutions[0]?.resolve(64_000) + await settleAsyncWork() expect(resolved).not.toHaveBeenCalled() current = { provider: 'test', model: 'alpha' } expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() await vi.waitFor(() => { expect(resolutions).toHaveLength(2) }) - resolutions[1]?.(128_000) + expect(resolutions[1]?.signal?.aborted).toBe(false) + resolutions[1]?.resolve(128_000) await vi.waitFor(() => { expect(resolved).toHaveBeenCalledOnce() }) expect(projector.snapshot(session, attached).contextWindow).toBe(128_000) }) From 35b9c454e59643f23ff33c57ffb595e8c75d4075 Mon Sep 17 00:00:00 2001 From: Hypatia May Date: Tue, 28 Jul 2026 18:35:39 +0800 Subject: [PATCH 11/38] refactor(web): publish transient model request capacity (round 1) --- ...8-host-owned-web-session-metrics.i18n.yaml | 4 +- ...26-07-28-host-owned-web-session-metrics.md | 18 +- ...07-28-host-owned-web-session-metrics.zh.md | 18 +- docs/architecture.i18n.yaml | 4 +- docs/architecture.md | 4 +- docs/architecture.zh.md | 4 +- docs/cordis-catalog/events.md | 57 +- docs/cordis-catalog/services.md | 6 +- .../llm-streaming.i18n.yaml | 4 +- docs/core-data-structures/llm-streaming.md | 8 +- docs/core-data-structures/llm-streaming.zh.md | 8 +- docs/event-producer-consumer.md | 35 +- packages/client/runtime/README.i18n.yaml | 4 +- packages/client/runtime/README.md | 2 +- packages/client/runtime/README.zh.md | 2 +- .../src/client/sessions/conversation.ts | 17 +- .../runtime/src/client/sessions/session.ts | 25 +- packages/client/runtime/tests/session.spec.ts | 59 +- .../client/ui-conversation/README.i18n.yaml | 4 +- packages/client/ui-conversation/README.md | 2 +- packages/client/ui-conversation/README.zh.md | 2 +- .../src/client/chat/StatsLine.tsx | 2 +- .../tests/chat-stats-bash-sample.spec.tsx | 20 + .../cordis/tool-cordis/src/api-catalog.ts | 13 +- packages/core/agent-loop/README.i18n.yaml | 4 +- packages/core/agent-loop/README.md | 2 +- packages/core/agent-loop/README.zh.md | 2 +- packages/core/agent-loop/src/agent.ts | 19 +- .../tests/request-reconstruction.spec.ts | 102 ++- packages/core/agent/README.i18n.yaml | 4 +- packages/core/agent/README.md | 2 +- packages/core/agent/README.zh.md | 2 +- packages/core/agent/src/types.ts | 24 + .../core/scope/src/scoped-events.generated.ts | 1 + packages/core/scope/tests/invariant.spec.ts | 1 + packages/host/apiproxy/README.i18n.yaml | 4 +- packages/host/apiproxy/README.md | 4 +- packages/host/apiproxy/README.zh.md | 4 +- packages/host/apiproxy/src/api-proxy.ts | 83 +-- .../host/apiproxy/src/api/events.schema.ts | 9 + packages/host/apiproxy/src/api/events.ts | 16 + .../host/apiproxy/src/api/sessions.schema.ts | 3 +- packages/host/apiproxy/src/api/sessions.ts | 14 +- packages/host/apiproxy/src/session-metrics.ts | 145 +--- .../tests/api-proxy-model-request.spec.ts | 104 +++ .../apiproxy/tests/api-proxy-models.spec.ts | 659 +----------------- .../host/apiproxy/tests/rpc-schemas.spec.ts | 31 +- .../apiproxy/tests/session-metrics.spec.ts | 381 +--------- packages/llm/llm/README.i18n.yaml | 4 +- packages/llm/llm/README.md | 4 +- packages/llm/llm/README.zh.md | 4 +- packages/llm/llm/src/index.ts | 128 +++- packages/llm/llm/tests/service.spec.ts | 34 + scripts/gen-cordis-catalog.ts | 1 + 54 files changed, 765 insertions(+), 1352 deletions(-) create mode 100644 packages/host/apiproxy/tests/api-proxy-model-request.spec.ts diff --git a/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.i18n.yaml b/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.i18n.yaml index 95add14bf5..100837813e 100644 --- a/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.i18n.yaml +++ b/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write .agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.md -2026-07-28-host-owned-web-session-metrics.md: 04381c7443491fd9de101a87713fa5c183800d0e -2026-07-28-host-owned-web-session-metrics.zh.md: 6ad06ea61508d3f0703c19bc7c9f14969f119334 +2026-07-28-host-owned-web-session-metrics.md: e37a7635cbdae65162308bf8a3a498b5071a4910 +2026-07-28-host-owned-web-session-metrics.zh.md: fd3e50c8cf8c0336a8cb2cd55f6629419bb8ad23 diff --git a/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.md b/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.md index 04381c7443..e37a7635cb 100644 --- a/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.md +++ b/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.md @@ -6,17 +6,19 @@ English | [中文](2026-07-28-host-owned-web-session-metrics.zh.md) ## Problem -A Web stats line derived from the currently loaded conversation nodes is window-dependent under pagination. Compaction can replace visible content without preserving historical usage, and route changes leave the browser without an authoritative context capacity. Cache-write tokens also risk being folded into a cache-hit formula whose denominator has different semantics. +A Web stats line derived from the currently loaded conversation nodes is window-dependent under pagination. Compaction can replace visible content without preserving historical usage, while the selected model does not prove that a request used its route or capacity. Cache-write tokens also risk being folded into a cache-hit formula whose denominator has different semantics. ## Decision The Host owns one session-level metrics projection. It incrementally folds the complete durable event log, keys settled usage by `(turn, step)`, and replaces an earlier usage record for the same key instead of double-counting chunk and message forms. Uncached input, output, cache reads, and cache writes remain four disjoint cumulative buckets. Compaction can change the current prompt surface without erasing historical usage. -Current context pressure is a separate point-in-time value from `tokenMeter.measure(session).totalTokens`. Capacity comes only from `llm.resolveModelInfo(provider, model).context.contextWindow` for the agent's selected route. A route change immediately publishes metrics with capacity absent, then publishes the resolved capacity behind a route generation fence; stale metadata cannot label the new route. +Current context pressure is the point-in-time `tokenMeter.measure(session).totalTokens`. Capacity instead belongs to the latest model request observed by the current live mux connection. `LlmService.prepareCall()` retains the context metadata obtained by the exact lookup that also validates reasoning/defaults, and the loop publishes it through one contained `agent/model-request` notification only after the final route has a successfully constructed stream handle. Failed or aborted iteration still counts as a dispatched request; preparation and synchronous construction failures do not. -The tail `session.history` response carries the projection, while older pages omit it. Live changes use `session/metrics` mux frames. Both forms carry a durable-log revision and a projection revision; the client accepts only nondecreasing revisions, preserves metrics across older-page prepend, and clears them at a new subscription baseline. Missing measurement or metadata stays absent. +The tail `session.history` response carries durable usage and pressure, while older pages omit them. Live changes use `session/metrics` mux frames. Both forms carry a durable-log revision and a projection revision; the client accepts only nondecreasing revisions and preserves metrics across older-page prepend. -The Web stats line treats the projection as its sole token source. It renders uncached input, output, and cache reads separately, computes cache hit as `cacheRead / (uncachedInput + cacheRead)`, and shows current context as a percentage of the exact route capacity. Cache writes never enter that percentage. Visible nodes continue to supply only turn and step counts. +ApiProxy forwards each notification as a distinct `session/model-request` frame only to mux connections already open when dispatch occurs. It never places the frame in `session.history` or a subscription baseline. The client retains that connection-local capacity across ordinary metrics updates, replaces or explicitly clears it on the next observed request, and clears it on `session/subscribed`; reconnect, restore, and a new subscription therefore start unknown until another request is observed. + +The Web stats line treats the durable projection plus live capacity overlay as its sole token source. It renders uncached input, output, and cache reads separately, computes cache hit as `cacheRead / (uncachedInput + cacheRead)`, and shows current context as a percentage only when the current connection observed a capacity. Cache writes never enter that percentage. Visible nodes continue to supply only turn and step counts. ## Alternatives considered @@ -26,10 +28,12 @@ The Web stats line treats the projection as its sole token source. It renders un **Reuse one total-token field for cache hit.** Cache reads, cache writes, and uncached input represent distinct provider accounting buckets; combining them would make the displayed rate misleading. -**Keep the previous capacity until the new route resolves.** The old number would temporarily claim the wrong selected model. An explicit unknown state is honest and generation-safe. +**Query the selected route before dispatch.** Selection may never produce a request, and a second metadata lookup can race the registration-bound lookup that actually validates and dispatches the call. + +**Persist or replay the latest request capacity.** That would make a former request look current on reconnect or restore even though the new connection observed no request. The denominator is deliberately live and opportunistic. ## Consequences -Token totals remain stable across pagination, replay, compaction, and browser reconnect. The client stores a small detached projection instead of scanning the conversation window, and the status row remains readable for large histories through compact number formatting. +Token totals remain stable across pagination, replay, compaction, and browser reconnect. The client stores a small detached durable projection plus one connection-local denominator instead of scanning the conversation window, and the status row remains readable for large histories through compact number formatting. -The Host performs one incremental log fold per session and schedules live projection updates only for usage, request-header, or surface-changing events; text and reasoning deltas do not publish metrics. Exact capacity resolution is asynchronous and may briefly render as unknown. Deployments without a token meter or model context metadata retain the row and label the unavailable value instead of fabricating one. +The Host performs one incremental log fold per session and schedules durable projection updates only for usage, request-header, or surface-changing events; text and reasoning deltas do not publish metrics. A new connection omits the percentage until it observes a request with context metadata. A later request without metadata clears the denominator, while deployments without a token meter still retain the durable counters and label context unavailable instead of fabricating pressure. diff --git a/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.zh.md b/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.zh.md index 6ad06ea615..fd3e50c8cf 100644 --- a/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.zh.md +++ b/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.zh.md @@ -6,17 +6,19 @@ Status: implemented ## 问题 -Web 统计行若根据当前加载的会话节点推导指标,其结果会随分页窗口变化。压缩(compaction)可以替换可见内容,却无法保留历史用量;路由变更会让浏览器缺少权威的上下文容量。缓存写入 token 还可能被计入缓存命中率公式,而该公式的分母具有不同语义。 +Web 统计行若根据当前加载的会话节点推导指标,其结果会随分页窗口变化。压缩(compaction)可以替换可见内容,却无法保留历史用量;所选模型也不能证明某次请求实际采用了该模型的路由或容量。缓存写入 token 还可能被计入缓存命中率公式,而该公式的分母具有不同语义。 ## 决策 Host 拥有一项会话级指标投影。它以增量方式归并完整的持久事件日志,按 `(turn, step)` 标识已结算用量;同一标识再次出现时,会替换较早的用量记录,而不会重复统计分片和消息两种形态。未缓存输入、输出、缓存读取与缓存写入保持为四个彼此独立的累计计数项。压缩可以改变当前提示词表层,但不会抹除历史用量。 -当前上下文压力是一个独立的即时值,取自 `tokenMeter.measure(session).totalTokens`。容量仅来自 `llm.resolveModelInfo(provider, model).context.contextWindow`,并对应 agent(智能体)所选的路由。路由变更时,Host 会立即发布不带容量的指标,再通过路由代际围栏发布解析出的容量;陈旧元数据无法标记新的路由。 +当前上下文压力是即时的 `tokenMeter.measure(session).totalTokens`。容量则属于当前实时 mux 连接观察到的最新模型请求。`LlmService.prepareCall()` 会保留同一次精确查询取得的上下文元数据,该查询也负责校验推理设置与默认值;仅在最终路由的流句柄成功构造后,循环才会通过一条失败会被收容的 `agent/model-request` 通知发布这些元数据。后续迭代失败或中止仍算作已分派请求;准备阶段失败和同步构造失败则不算。 -`session.history` 尾页响应携带该投影,较早页面则省略它。实时变更使用 `session/metrics` mux 帧。两种形式都携带持久日志修订号和投影修订号;客户端只接受不减小的修订号,在向前加载较早页面时保留指标,并在建立新的订阅基线时将其清除。测量值或元数据缺失时,对应字段保持缺失。 +`session.history` 尾页响应携带持久用量与压力,较早页面则省略这两项。实时变更使用 `session/metrics` mux 帧。两种形式都携带持久日志修订号和投影修订号;客户端只接受不减小的修订号,并在向前加载较早页面时保留指标。 -Web 统计行把该投影视为唯一的 token 数据来源。它分别呈现未缓存输入、输出与缓存读取,通过 `cacheRead / (uncachedInput + cacheRead)` 计算缓存命中率,并把当前上下文显示为精确路由容量的百分比。缓存写入绝不计入缓存命中率。可见节点仍然只提供轮次和步骤计数。 +ApiProxy 只把每条通知作为独立的 `session/model-request` 帧转发给分派发生时已经打开的 mux 连接。它绝不会把该帧放入 `session.history` 或订阅基线。客户端会在普通指标更新期间保留这项连接本地容量,在观察到下一次请求时替换或显式清除它,并在收到 `session/subscribed` 时将其清除;因此,重连、恢复和新订阅都会从未知容量开始,直到观察到另一次请求。 + +Web 统计行把持久投影与实时容量覆盖层视为唯一的 token 数据来源。它分别呈现未缓存输入、输出与缓存读取,通过 `cacheRead / (uncachedInput + cacheRead)` 计算缓存命中率,并且只有当前连接观察到容量时,才把当前上下文显示为该容量的百分比。缓存写入绝不计入缓存命中率。可见节点仍然只提供轮次和步骤计数。 ## 备选方案 @@ -26,10 +28,12 @@ Web 统计行把该投影视为唯一的 token 数据来源。它分别呈现未 **为缓存命中率复用单一的 token 总数字段。** 缓存读取、缓存写入与未缓存输入是提供方记账中的不同计数项;将它们合并会使显示的比率产生误导。 -**在新路由解析完成前保留旧容量。** 旧数值会在短时间内错误标示所选模型。显式的「未知」状态能如实反映情况,并避免跨代串扰。 +**在分派前查询所选路由。** 选择操作可能永远不会产生请求;第二次元数据查询还可能与实际校验并分派调用的、绑定注册项的查询发生竞态。 + +**持久化或回放最新请求的容量。** 即使新连接没有观察到任何请求,这也会让先前请求在重连或恢复后显得仍然有效。该分母刻意只采用实时且恰好可得的数据。 ## 后果 -token 总量在分页、回放、压缩和浏览器重连期间保持稳定。客户端存储一项小型脱耦投影,无需扫描会话窗口;状态行采用紧凑数字格式,因此在较长的历史记录中仍然清晰易读。 +token 总量在分页、回放、压缩和浏览器重连期间保持稳定。客户端存储一项小型、脱耦的持久投影与一个连接本地分母,无需扫描会话窗口;状态行采用紧凑数字格式,因此在较长的历史记录中仍然清晰易读。 -Host 为每个会话执行一次增量日志归并,仅为用量事件、请求头事件或表层变更事件调度实时投影更新;文本与推理(reasoning)增量不会发布指标。精确容量解析为异步操作,因此可能短暂显示「未知」。未部署 token 计量器或缺少模型上下文元数据时,系统仍保留该行,并标示不可用的值,而不会虚构数据。 +Host 为每个会话执行一次增量日志归并,仅为用量事件、请求头事件或表层变更事件调度持久投影更新;文本与推理(reasoning)增量不会发布指标。新连接在观察到带上下文元数据的请求之前不会显示百分比。后续不带元数据的请求会清除该分母;未部署 token 计量器时,系统仍保留持久计数器,并把上下文标示为不可用,而不会虚构压力值。 diff --git a/docs/architecture.i18n.yaml b/docs/architecture.i18n.yaml index e56a3f91bb..cfe3106723 100644 --- a/docs/architecture.i18n.yaml +++ b/docs/architecture.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write docs/architecture.md -architecture.md: 054985ac5ea32a44b9daca3c1abfd58dcdc5d897 -architecture.zh.md: 84876faf2ae27069ba8bd026bcfbc56e32f65574 +architecture.md: 52072633a0e81afce63c5e162dd1b0af7f6486ca +architecture.zh.md: f0122ece146c17366aa316cfb4ea4196a1db74bd diff --git a/docs/architecture.md b/docs/architecture.md index 054985ac5e..52072633a0 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -92,7 +92,7 @@ forever: assemble system prompt and tool schemas snapshot the derived messages (the reconstruction boundary) 'step/start' - agent/request (config only) -> prepare reasoning/default under turn signal -> log request/header -> llm/stream (frozen, registration-bound) + agent/request (config only) -> prepare reasoning/default + context under turn signal -> log request/header -> construct llm/stream (frozen, registration-bound) -> agent/model-request (live, contained) -> iterate 'assistant/chunk' 'assistant/message' schedule tool calls by ctx.tools.executionMode: @@ -151,7 +151,7 @@ Log-only events may sit between turns. Owners append through `Session`, flushing Messages use typed blocks from merge-extensible `ContentBlockMap`; the pattern also types `MessageSource`, `FinishReason`, `TurnTrigger`, and `TurnEndReason`. New blocks coordinate adapters, UI, compaction, token metering, and persistence; replay measurements live in [token-meter.md](core-data-structures/token-meter.md). -Streaming uses raw chunks and `BlockAssembler`. Each `LlmAdapter.stream()` is one provider attempt; adapters report normalized failure facts, and a handling `agent/request-error` plugin returns a retry action. The loop logs chunks, successful provenance, and replay state. Remote adapters use per-read idle watchdogs. Replay crosses routes only through a shared adapter instance ([contract](core-data-structures/llm-streaming.md)). +Streaming uses raw chunks and `BlockAssembler`. After final-stream construction, the loop emits contained, non-durable, non-replayed `agent/model-request` metadata. Adapters normalize failures; `agent/request-error` may retry. Remote adapters use per-read idle watchdogs. Replay crosses routes only through a shared adapter ([contract](core-data-structures/llm-streaming.md)). ## Extension And Composition diff --git a/docs/architecture.zh.md b/docs/architecture.zh.md index 84876faf2a..f0122ece14 100644 --- a/docs/architecture.zh.md +++ b/docs/architecture.zh.md @@ -92,7 +92,7 @@ forever: assemble system prompt and tool schemas snapshot the derived messages (the reconstruction boundary) 'step/start' - agent/request (config only) -> prepare reasoning/default under turn signal -> log request/header -> llm/stream (frozen, registration-bound) + agent/request (config only) -> prepare reasoning/default + context under turn signal -> log request/header -> construct llm/stream (frozen, registration-bound) -> agent/model-request (live, contained) -> iterate 'assistant/chunk' 'assistant/message' schedule tool calls by ctx.tools.executionMode: @@ -151,7 +151,7 @@ idle inject: 消息使用从可合并扩展的 `ContentBlockMap` 派生的类型化块;同一模式也为 `MessageSource`、`FinishReason`、`TurnTrigger` 和 `TurnEndReason` 定义类型。新增块会协调适配器、UI、压缩、token 计量和持久化;回放计量见 [token-meter.md](core-data-structures/token-meter.md)。 -流式输出使用原始分片和 `BlockAssembler`。每次 `LlmAdapter.stream()` 调用代表一次提供方尝试;适配器报告标准化的故障事实,负责处理的 `agent/request-error` 插件会返回重试动作。循环会记录分片、成功结果的来源信息和回放状态。远程适配器使用逐次读取空闲看门狗。回放仅通过共用的适配器实例跨路由传递([契约](core-data-structures/llm-streaming.md))。 +流式输出使用原始分片和 `BlockAssembler`。最终流构造完成后,循环会发出 `agent/model-request` 元数据;该通知的失败会被收容,元数据不会持久化或回放。适配器会规范化故障;`agent/request-error` 可以重试。远程适配器使用逐次读取空闲看门狗。回放仅通过共用适配器跨路由传递([契约](core-data-structures/llm-streaming.md))。 ## 扩展与组合 diff --git a/docs/cordis-catalog/events.md b/docs/cordis-catalog/events.md index ca0e9b82fc..e8e67da781 100644 --- a/docs/cordis-catalog/events.md +++ b/docs/cordis-catalog/events.md @@ -32,7 +32,7 @@ Effective broad cancellation was requested, before queued/outbox work is cleared Types: [Agent](../core-data-structures/core.md) · [AgentCancelCause](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/core/agent/src/types.ts:308`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:318`](../../packages/core/agent/src/types.ts) ### `agent/created` — emit @@ -54,7 +54,7 @@ A fully configured agent and live session were published. Setup is composition-o Types: [Agent](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/core/agent/src/types.ts:247`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:257`](../../packages/core/agent/src/types.ts) ### `agent/disposed` — emit @@ -74,7 +74,7 @@ An agent left the registry; AgentLoop emits this after driver quiescence and sco Types: [Agent](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/core/agent/src/types.ts:256`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:266`](../../packages/core/agent/src/types.ts) ### `agent/error` — emit @@ -96,7 +96,7 @@ A step or turn errored. The machine reports a failure here (plus the logger) eve Types: [Agent](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/core/agent/src/types.ts:423`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:447`](../../packages/core/agent/src/types.ts) ### `agent/inbox/dequeue` — emit @@ -117,7 +117,7 @@ The driver claimed one item out of the inbox: a queued item at a turn boundary, Types: [Agent](../core-data-structures/core.md) · [AgentMessage](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/core/agent/src/types.ts:286`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:296`](../../packages/core/agent/src/types.ts) ### `agent/inbox/discard` — emit @@ -140,7 +140,7 @@ Pending inbox items were dropped without delivering them, so every enqueued id r Types: [Agent](../core-data-structures/core.md) · [AgentMessage](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/core/agent/src/types.ts:298`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:308`](../../packages/core/agent/src/types.ts) ### `agent/inbox/enqueue` — emit @@ -162,7 +162,32 @@ An item entered the queued or steering inbox. `placement` is the acceptance-time Types: [Agent](../core-data-structures/core.md) · [AgentMessage](../core-data-structures/core.md) · [InboxPlacement](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/core/agent/src/types.ts:276`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:286`](../../packages/core/agent/src/types.ts) + +### `agent/model-request` — emit + +One model request constructed its final stream handle and is about to iterate it. This live notification is not durable or replayed; failed or aborted iteration still has a dispatch, while preparation and synchronous stream-construction failures do not. Listener failures are contained and cannot affect the request. + +```ts cordis-catalog +/** + * One model request constructed its final stream handle and is about to + * iterate it. This live notification is not durable or replayed; failed or + * aborted iteration still has a dispatch, while preparation and + * synchronous stream-construction failures do not. Listener failures are + * contained and cannot affect the request. + * @param agent - the agent dispatching the model request. + * @param turn - the open turn number. + * @param step - the request's step number. + * @param request - final route plus registration-bound context capacity. + * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent. + * @mode emit + */ +'agent/model-request'(this: Scoped, agent: Agent, turn: number, step: number, request: AgentModelRequest): void +``` + +Types: [Agent](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) + +Source: [`packages/core/agent/src/types.ts:386`](../../packages/core/agent/src/types.ts) ### `agent/prompt-submit` — waterfall @@ -186,7 +211,7 @@ Allow, rewrite, or block one claimed prompt before it becomes a user message or Types: [Agent](../core-data-structures/core.md) · [ContentBlock](../core-data-structures/core.md) · [MessageSource](../core-data-structures/core.md) · [PromptDecision](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/core/agent/src/types.ts:336`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:346`](../../packages/core/agent/src/types.ts) ### `agent/request` — waterfall @@ -210,7 +235,7 @@ Replace the frozen call configuration. `await next()` yields the config the mach Types: [Agent](../core-data-structures/core.md) · [LlmCallConfig](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/core/agent/src/types.ts:362`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:372`](../../packages/core/agent/src/types.ts) ### `agent/request-error` — waterfall @@ -240,7 +265,7 @@ Handle a model-request failure after its failed step has closed but before the f Types: [Agent](../core-data-structures/core.md) · [LlmFailure](../core-data-structures/llm-streaming.md) · [RequestError](../core-data-structures/core.md) · [RequestErrorAction](../core-data-structures/core.md) · [ResolvedRetryPolicy](../core-data-structures/llm-streaming.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/core/agent/src/types.ts:381`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:405`](../../packages/core/agent/src/types.ts) ### `agent/session-start` — emit @@ -262,7 +287,7 @@ The session lifecycle began, once before the first turn. Use `agent.inject()` to Types: [Agent](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) · [SessionStartSource](../core-data-structures/core.md) -Source: [`packages/core/agent/src/types.ts:321`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:331`](../../packages/core/agent/src/types.ts) ### `agent/settled` — emit @@ -287,7 +312,7 @@ One drain chain reached its terminal turn: that turn's `turn/end` is already com Types: [Agent](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) · [SettleReason](../core-data-structures/core.md) -Source: [`packages/core/agent/src/types.ts:410`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:434`](../../packages/core/agent/src/types.ts) ### `agent/status` — emit @@ -307,7 +332,7 @@ Agent status changed (`idle` ⇄ `running`). `send()` does not enter `running` s Types: [Agent](../core-data-structures/core.md) · [AgentStatus](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/core/agent/src/types.ts:265`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:275`](../../packages/core/agent/src/types.ts) ### `agent/step` — serial @@ -331,7 +356,7 @@ Awaited serial checkpoint before EVERY request of a turn is built (the first as Types: [Agent](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/core/agent/src/types.ts:349`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:359`](../../packages/core/agent/src/types.ts) ### `agent/turn-stopping` — serial @@ -357,7 +382,7 @@ The turn is about to close: the model owes no response (no live tool calls, no f Types: [Agent](../core-data-structures/core.md) · [Scoped](../core-data-structures/scope.md) -Source: [`packages/core/agent/src/types.ts:396`](../../packages/core/agent/src/types.ts) +Source: [`packages/core/agent/src/types.ts:420`](../../packages/core/agent/src/types.ts) ## `agent-loop/*` @@ -548,7 +573,7 @@ Waterfall around every streaming model call (retry, replay, routing). Bound to t Types: [GenerateOptions](../core-data-structures/core.md) · [LlmService](../core-data-structures/llm-streaming.md) · [StreamChunk](../core-data-structures/llm-streaming.md) -Source: [`packages/llm/llm/src/index.ts:56`](../../packages/llm/llm/src/index.ts) +Source: [`packages/llm/llm/src/index.ts:57`](../../packages/llm/llm/src/index.ts) ## `session/*` diff --git a/docs/cordis-catalog/services.md b/docs/cordis-catalog/services.md index 19d7d7765d..9d00ef37a0 100644 --- a/docs/cordis-catalog/services.md +++ b/docs/cordis-catalog/services.md @@ -780,14 +780,16 @@ async prepareCall(config: LlmCallConfig, signal?: AbortSignal): Promise +stream(options: GenerateOptions, onDispatched?: () => void): AsyncIterable ``` Types: [GenerateOptions](../core-data-structures/core.md) · [LlmAdapter](../core-data-structures/llm-streaming.md) · [LlmCallConfig](../core-data-structures/core.md) · [LlmModelInfo](../core-data-structures/core.md) · [LlmProviderInfo](../core-data-structures/core.md) · [LlmResolvedModelInfo](../core-data-structures/core.md) · [PreparedLlmCall](../core-data-structures/llm-streaming.md) · [ResolvedRetryPolicy](../core-data-structures/llm-streaming.md) · [StreamChunk](../core-data-structures/llm-streaming.md) -Source: [`packages/llm/llm/src/index.ts:189`](../../packages/llm/llm/src/index.ts) +Source: [`packages/llm/llm/src/index.ts:194`](../../packages/llm/llm/src/index.ts) ## `ctx.permission` — `PermissionService` diff --git a/docs/core-data-structures/llm-streaming.i18n.yaml b/docs/core-data-structures/llm-streaming.i18n.yaml index 9f1e4f440a..6206f864cf 100644 --- a/docs/core-data-structures/llm-streaming.i18n.yaml +++ b/docs/core-data-structures/llm-streaming.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write docs/core-data-structures/llm-streaming.md -llm-streaming.md: db46deee28cd053d034f889eb7625c9f222b418b -llm-streaming.zh.md: fbff50bf1f3b86afd313c2d5020c15dd88e0b449 +llm-streaming.md: 89628ebb96a2e8eec5209635cd92859427df4a8d +llm-streaming.zh.md: 9c8fcc1f24b970f3a7cdd7cd08d9ef3b934b4543 diff --git a/docs/core-data-structures/llm-streaming.md b/docs/core-data-structures/llm-streaming.md index db46deee28..89628ebb96 100644 --- a/docs/core-data-structures/llm-streaming.md +++ b/docs/core-data-structures/llm-streaming.md @@ -161,21 +161,25 @@ declare class BlockAssembler { ## The seam -`LlmAdapter` is the provider seam: subclass, implement `stream()`, and register one adapter instance with `ctx.llm.registerAdapter(providers, adapter)`. `GenerateOptions.provider` selects the registered adapter; `GenerateOptions.model` is passed to that adapter and need not be registered at lifecycle start. Duplicate provider routes fail atomically. Optional `providerRetryPolicy()` is captured per route with normal defaults, while `providerInfo()` and asynchronous `listModels()` feed `LlmService.listProviders()` / `listModels()` with detached selector metadata. That catalog is advisory rather than a request whitelist: the adapter remains authoritative and may accept unlisted model ids. One asynchronous `resolveModel()` query returns exact model identity plus optional correctness-sensitive context capacity and ordered model-owned reasoning ids with an optional deployment default; absent fields mean unavailable metadata or capability, not invalid catalog membership. The resolver receives optional cancellation and must settle promptly after abort. `LlmService.resolveModelInfo()` validates and detaches the aggregate. The service validates and materializes reasoning through `resolveCallConfig()` at the final adapter boundary, so direct calls cannot bypass unsupported-effort rejection; direct dispatch captures one registration before awaiting that resolution. The agent loop instead uses `prepareCall()` to keep the same registration across model resolution, durable header logging, and dispatch. Adapter lookup happens at the terminal continuation of the `llm/stream` waterfall, so a listener may short-circuit the call or route a mutable one-shot request before lookup. The `block-start` / `block-end` `index` correlation and the assembler together mean an adapter only has to emit well-formed chunks — block reassembly is not each adapter's problem. The consumer surface (`ctx.llm.stream()`) and the `llm/stream` waterfall are described in [architecture.md § Content blocks and streaming](../architecture.md#content-blocks-and-streaming-dsh-llm). +`LlmAdapter` is the provider seam: subclass, implement `stream()`, and register one adapter instance with `ctx.llm.registerAdapter(providers, adapter)`. `GenerateOptions.provider` selects the registered adapter; `GenerateOptions.model` is passed to that adapter and need not be registered at lifecycle start. Duplicate provider routes fail atomically. Optional `providerRetryPolicy()` is captured per route with normal defaults, while `providerInfo()` and asynchronous `listModels()` feed `LlmService.listProviders()` / `listModels()` with detached selector metadata. That catalog is advisory rather than a request whitelist: the adapter remains authoritative and may accept unlisted model ids. One asynchronous `resolveModel()` query returns exact model identity plus optional correctness-sensitive context capacity and ordered model-owned reasoning ids with an optional deployment default; absent fields mean unavailable metadata or capability, not invalid catalog membership. The resolver receives optional cancellation and must settle promptly after abort. `LlmService.resolveModelInfo()` validates and detaches the aggregate. The service validates and materializes reasoning through `resolveCallConfig()` at the final adapter boundary, so direct calls cannot bypass unsupported-effort rejection; direct dispatch captures one registration before awaiting that resolution. The agent loop instead uses `prepareCall()` to keep the same registration across model resolution, durable header logging, and dispatch, and to retain detached context metadata from that exact lookup. Its optional observer runs after a final stream handle is constructed and before adapter iteration. Adapter lookup happens at the terminal continuation of the `llm/stream` waterfall, so a listener may short-circuit the call or route a mutable one-shot request before lookup. The `block-start` / `block-end` `index` correlation and the assembler together mean an adapter only has to emit well-formed chunks — block reassembly is not each adapter's problem. The consumer surface (`ctx.llm.stream()`) and the `llm/stream` waterfall are described in [architecture.md § Content blocks and streaming](../architecture.md#content-blocks-and-streaming-dsh-llm). ```ts type-equiv /** One model call whose config and adapter registration were resolved together. */ interface PreparedLlmCall { /** Detached, deep-frozen config with any adapter-owned default materialized. */ readonly config: LlmCallConfig + /** Detached context metadata resolved with the registration-bound call. */ + readonly context?: LlmModelContext /** * Dispatch this call once through the registration captured during * preparation. The request's call-config fields must match {@link config}; * reuse or mismatch fails with `INVALID_PREPARED_CALL`. * @param options - fully assembled request carrying the prepared config. + * @param onDispatched - contained Agent-loop notification hook invoked after + * a stream handle is constructed and before its adapter is iterated. * @returns the chunk stream, including the `llm/stream` waterfall. */ - stream(options: GenerateOptions): AsyncIterable + stream(options: GenerateOptions, onDispatched?: () => void): AsyncIterable } ``` diff --git a/docs/core-data-structures/llm-streaming.zh.md b/docs/core-data-structures/llm-streaming.zh.md index fbff50bf1f..9c8fcc1f24 100644 --- a/docs/core-data-structures/llm-streaming.zh.md +++ b/docs/core-data-structures/llm-streaming.zh.md @@ -161,21 +161,25 @@ declare class BlockAssembler { ## seam -`LlmAdapter` 是提供方 seam:创建子类、实现 `stream()`,再用 `ctx.llm.registerAdapter(providers, adapter)` 注册一个适配器实例。`GenerateOptions.provider` 选择已注册适配器;`GenerateOptions.model` 会传给该适配器,无需在生命周期启动时注册。重复提供方路由会原子失败。可选的 `providerRetryPolicy()` 会按路由捕获并填入 normal 默认值,`providerInfo()` 与异步 `listModels()` 方法则为 `LlmService.listProviders()` / `listModels()` 提供分离的 selector 元数据。该目录仅供参考,不是请求白名单:适配器仍是权威,并可接受未列出的模型 id。单次异步 `resolveModel()` 查询返回确切模型身份,以及可选的对正确性敏感的上下文容量、由模型持有的有序推理强度 ID 和部署默认值;字段缺失表示元数据或能力不可用,而不表示目录成员关系无效。解析器会接收可选的取消信号,并且必须在信号中止后迅速完成结算。`LlmService.resolveModelInfo()` 会校验聚合结果并返回分离值。服务通过最终适配器边界的 `resolveCallConfig()` 校验推理强度并填入默认值,因此直接调用也无法绕过对不支持推理强度的拒绝;直接分派会在等待解析前捕获一项适配器注册。agent loop 则使用 `prepareCall()`,使模型解析、请求头持久记录和分派全程使用同一项注册。适配器查找发生在 `llm/stream` waterfall(瀑布式事件)的终端 continuation,因此 listener 可以在查找前短路调用,或路由一个可变的一次性请求。`block-start` / `block-end` 的 `index` 关联与 assembler 共同意味着适配器只需 emit 格式正确的分片——块重组不是每个适配器各自的问题。消费方 surface(`ctx.llm.stream()`)与 `llm/stream` waterfall 见 [architecture.md § 内容块与流式传输](../architecture.md#content-blocks-and-streaming-dsh-llm)。 +`LlmAdapter` 是提供方 seam:创建子类、实现 `stream()`,再用 `ctx.llm.registerAdapter(providers, adapter)` 注册一个适配器实例。`GenerateOptions.provider` 选择已注册适配器;`GenerateOptions.model` 会传给该适配器,无需在生命周期启动时注册。重复提供方路由会原子失败。可选的 `providerRetryPolicy()` 会按路由捕获并填入 normal 默认值,`providerInfo()` 与异步 `listModels()` 方法则为 `LlmService.listProviders()` / `listModels()` 提供分离的 selector 元数据。该目录仅供参考,不是请求白名单:适配器仍是权威,并可接受未列出的模型 id。单次异步 `resolveModel()` 查询返回确切模型身份,以及可选的对正确性敏感的上下文容量、由模型持有的有序推理强度 ID 和部署默认值;字段缺失表示元数据或能力不可用,而不表示目录成员关系无效。解析器会接收可选的取消信号,并且必须在信号中止后迅速完成结算。`LlmService.resolveModelInfo()` 会校验聚合结果并返回分离值。服务通过最终适配器边界的 `resolveCallConfig()` 校验推理强度并填入默认值,因此直接调用也无法绕过对不支持推理强度的拒绝;直接分派会在等待解析前捕获一项适配器注册。agent loop 则使用 `prepareCall()`,使模型解析、请求头持久记录和分派全程使用同一项注册,并保留来自同一次精确查询的分离上下文元数据。其可选观察器在最终流句柄构造完成后、适配器开始迭代前运行。适配器查找发生在 `llm/stream` waterfall(瀑布式事件)的终端 continuation,因此 listener 可以在查找前短路调用,或路由一个可变的一次性请求。`block-start` / `block-end` 的 `index` 关联与 assembler 共同意味着适配器只需 emit 格式正确的分片——块重组不是每个适配器各自的问题。消费方 surface(`ctx.llm.stream()`)与 `llm/stream` waterfall 见 [architecture.md § 内容块与流式传输](../architecture.md#content-blocks-and-streaming-dsh-llm)。 ```ts type-equiv /** One model call whose config and adapter registration were resolved together. */ interface PreparedLlmCall { /** Detached, deep-frozen config with any adapter-owned default materialized. */ readonly config: LlmCallConfig + /** Detached context metadata resolved with the registration-bound call. */ + readonly context?: LlmModelContext /** * Dispatch this call once through the registration captured during * preparation. The request's call-config fields must match {@link config}; * reuse or mismatch fails with `INVALID_PREPARED_CALL`. * @param options - fully assembled request carrying the prepared config. + * @param onDispatched - contained Agent-loop notification hook invoked after + * a stream handle is constructed and before its adapter is iterated. * @returns the chunk stream, including the `llm/stream` waterfall. */ - stream(options: GenerateOptions): AsyncIterable + stream(options: GenerateOptions, onDispatched?: () => void): AsyncIterable } ``` diff --git a/docs/event-producer-consumer.md b/docs/event-producer-consumer.md index 2b719de804..b71fdb13f9 100644 --- a/docs/event-producer-consumer.md +++ b/docs/event-producer-consumer.md @@ -8,21 +8,22 @@ This matrix shows which packages dispatch each harness-owned event and which pac | Event | Mode | Declared in | Dispatchers | Listeners | | --- | --- | --- | --- | --- | | `agent-loop/config-start-failed` | `emit` | [`packages/core/agent-loop/src/index.ts:140`](../packages/core/agent-loop/src/index.ts) | [`agent-loop`](../packages/core/agent-loop) (`events.dispatch`) | [`tui`](../packages/ui/tui) | -| `agent/cancel-requested` | `emit` | [`packages/core/agent/src/types.ts:308`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | [`goal-session`](../packages/goal/goal-session) | -| `agent/created` | `emit` | [`packages/core/agent/src/types.ts:247`](../packages/core/agent/src/types.ts) | [`agent`](../packages/core/agent) (`events.dispatch`) | `apiproxy`, [`goal-session`](../packages/goal/goal-session), [`tui`](../packages/ui/tui) | -| `agent/disposed` | `emit` | [`packages/core/agent/src/types.ts:256`](../packages/core/agent/src/types.ts) | [`agent`](../packages/core/agent) (`events.dispatch`) | [`agent-loop`](../packages/core/agent-loop), [`goal-session`](../packages/goal/goal-session), [`tui`](../packages/ui/tui) | -| `agent/error` | `emit` | [`packages/core/agent/src/types.ts:423`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | `apiproxy`, [`goal-session`](../packages/goal/goal-session), [`session-telemetry`](../packages/telemetry/session-telemetry), [`tui`](../packages/ui/tui) | -| `agent/inbox/dequeue` | `emit` | [`packages/core/agent/src/types.ts:286`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | [`agent`](../packages/core/agent), `apiproxy`, [`tui`](../packages/ui/tui) | -| `agent/inbox/discard` | `emit` | [`packages/core/agent/src/types.ts:298`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | [`agent`](../packages/core/agent), `apiproxy`, [`tui`](../packages/ui/tui) | -| `agent/inbox/enqueue` | `emit` | [`packages/core/agent/src/types.ts:276`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | [`agent`](../packages/core/agent), `apiproxy`, [`goal-session`](../packages/goal/goal-session), [`tui`](../packages/ui/tui) | -| `agent/prompt-submit` | `waterfall` | [`packages/core/agent/src/types.ts:336`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`waterfall`) | [`goal-session`](../packages/goal/goal-session), [`hooks-claude`](../packages/hooks/hooks-claude), [`hooks-codex`](../packages/hooks/hooks-codex), [`repeat-tool-guard`](../packages/guard/repeat-tool-guard), [`tui`](../packages/ui/tui) | -| `agent/request` | `waterfall` | [`packages/core/agent/src/types.ts:362`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`waterfall`) | [`agent`](../packages/core/agent) | -| `agent/request-error` | `waterfall` | [`packages/core/agent/src/types.ts:381`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`waterfall`) | [`compact-basic`](../packages/compact/compact-basic), [`llm-retry`](../packages/llm/llm-retry) | -| `agent/session-start` | `emit` | [`packages/core/agent/src/types.ts:321`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | [`goal`](../packages/goal/goal), [`goal-session`](../packages/goal/goal-session), [`hooks-claude`](../packages/hooks/hooks-claude), [`hooks-codex`](../packages/hooks/hooks-codex), [`workspace-context`](../packages/context/workspace-context) | -| `agent/settled` | `emit` | [`packages/core/agent/src/types.ts:410`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | [`compact-basic`](../packages/compact/compact-basic) | -| `agent/status` | `emit` | [`packages/core/agent/src/types.ts:265`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | [`agent`](../packages/core/agent), `apiproxy`, [`goal-session`](../packages/goal/goal-session), [`tui`](../packages/ui/tui) | -| `agent/step` | `serial` | [`packages/core/agent/src/types.ts:349`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`serial`) | [`compact-basic`](../packages/compact/compact-basic), [`plan-mode`](../packages/plan/plan-mode), [`session-checkpoint-policy`](../packages/session-persistence/session-checkpoint-policy), [`time-context`](../packages/context/time-context), [`tool-skill`](../packages/skill/tool-skill), [`user-approval`](../packages/ui/user-approval), [`workspace-context`](../packages/context/workspace-context) | -| `agent/turn-stopping` | `serial` | [`packages/core/agent/src/types.ts:396`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`serial`) | [`hooks-claude`](../packages/hooks/hooks-claude), [`hooks-codex`](../packages/hooks/hooks-codex) | +| `agent/cancel-requested` | `emit` | [`packages/core/agent/src/types.ts:318`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | [`goal-session`](../packages/goal/goal-session) | +| `agent/created` | `emit` | [`packages/core/agent/src/types.ts:257`](../packages/core/agent/src/types.ts) | [`agent`](../packages/core/agent) (`events.dispatch`) | `apiproxy`, [`goal-session`](../packages/goal/goal-session), [`tui`](../packages/ui/tui) | +| `agent/disposed` | `emit` | [`packages/core/agent/src/types.ts:266`](../packages/core/agent/src/types.ts) | [`agent`](../packages/core/agent) (`events.dispatch`) | [`agent-loop`](../packages/core/agent-loop), [`goal-session`](../packages/goal/goal-session), [`tui`](../packages/ui/tui) | +| `agent/error` | `emit` | [`packages/core/agent/src/types.ts:447`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | `apiproxy`, [`goal-session`](../packages/goal/goal-session), [`session-telemetry`](../packages/telemetry/session-telemetry), [`tui`](../packages/ui/tui) | +| `agent/inbox/dequeue` | `emit` | [`packages/core/agent/src/types.ts:296`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | [`agent`](../packages/core/agent), `apiproxy`, [`tui`](../packages/ui/tui) | +| `agent/inbox/discard` | `emit` | [`packages/core/agent/src/types.ts:308`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | [`agent`](../packages/core/agent), `apiproxy`, [`tui`](../packages/ui/tui) | +| `agent/inbox/enqueue` | `emit` | [`packages/core/agent/src/types.ts:286`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | [`agent`](../packages/core/agent), `apiproxy`, [`goal-session`](../packages/goal/goal-session), [`tui`](../packages/ui/tui) | +| `agent/model-request` | `emit` | [`packages/core/agent/src/types.ts:386`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | `apiproxy` | +| `agent/prompt-submit` | `waterfall` | [`packages/core/agent/src/types.ts:346`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`waterfall`) | [`goal-session`](../packages/goal/goal-session), [`hooks-claude`](../packages/hooks/hooks-claude), [`hooks-codex`](../packages/hooks/hooks-codex), [`repeat-tool-guard`](../packages/guard/repeat-tool-guard), [`tui`](../packages/ui/tui) | +| `agent/request` | `waterfall` | [`packages/core/agent/src/types.ts:372`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`waterfall`) | [`agent`](../packages/core/agent) | +| `agent/request-error` | `waterfall` | [`packages/core/agent/src/types.ts:405`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`waterfall`) | [`compact-basic`](../packages/compact/compact-basic), [`llm-retry`](../packages/llm/llm-retry) | +| `agent/session-start` | `emit` | [`packages/core/agent/src/types.ts:331`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | [`goal`](../packages/goal/goal), [`goal-session`](../packages/goal/goal-session), [`hooks-claude`](../packages/hooks/hooks-claude), [`hooks-codex`](../packages/hooks/hooks-codex), [`workspace-context`](../packages/context/workspace-context) | +| `agent/settled` | `emit` | [`packages/core/agent/src/types.ts:434`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | [`compact-basic`](../packages/compact/compact-basic) | +| `agent/status` | `emit` | [`packages/core/agent/src/types.ts:275`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | [`agent`](../packages/core/agent), `apiproxy`, [`goal-session`](../packages/goal/goal-session), [`tui`](../packages/ui/tui) | +| `agent/step` | `serial` | [`packages/core/agent/src/types.ts:359`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`serial`) | [`compact-basic`](../packages/compact/compact-basic), [`plan-mode`](../packages/plan/plan-mode), [`session-checkpoint-policy`](../packages/session-persistence/session-checkpoint-policy), [`time-context`](../packages/context/time-context), [`tool-skill`](../packages/skill/tool-skill), [`user-approval`](../packages/ui/user-approval), [`workspace-context`](../packages/context/workspace-context) | +| `agent/turn-stopping` | `serial` | [`packages/core/agent/src/types.ts:420`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`serial`) | [`hooks-claude`](../packages/hooks/hooks-claude), [`hooks-codex`](../packages/hooks/hooks-codex) | | `approval/request` | `waterfall` | [`packages/ui/user-approval/src/index.ts:30`](../packages/ui/user-approval/src/index.ts) | [`user-approval`](../packages/ui/user-approval) (`waterfall`) | [`acp`](../packages/acp/acp) | | `commands/change` | `emit` | [`packages/ui/commands/src/index.ts:103`](../packages/ui/commands/src/index.ts) | [`commands`](../packages/ui/commands) (`events.dispatch`) | `apiproxy`, [`tui`](../packages/ui/tui) | | `domain/changed` | `emit` | [`packages/storage/storage-domain/src/events.ts:46`](../packages/storage/storage-domain/src/events.ts) | [`storage-domain`](../packages/storage/storage-domain) (`emit`) | `apiproxy`, [`storage-domain`](../packages/storage/storage-domain), [`workspace`](../packages/workspace/workspace) | @@ -30,7 +31,7 @@ This matrix shows which packages dispatch each harness-owned event and which pac | `fs/observed` | `emit` | [`packages/fs/fs/src/index.ts:71`](../packages/fs/fs/src/index.ts) | [`tool-fs`](../packages/fs/tool-fs) (`emit`) | [`fs-policy`](../packages/fs/fs-policy) | | `fs/write-intent` | `waterfall` | [`packages/fs/fs/src/index.ts:54`](../packages/fs/fs/src/index.ts) | [`tool-fs`](../packages/fs/tool-fs) (`waterfall`) | [`fs-policy`](../packages/fs/fs-policy) | | `goal/changed` | `emit` | [`packages/goal/goal/src/types.ts:169`](../packages/goal/goal/src/types.ts) | [`goal`](../packages/goal/goal) (`emit`) | [`goal-session`](../packages/goal/goal-session) | -| `llm/stream` | `waterfall` | [`packages/llm/llm/src/index.ts:56`](../packages/llm/llm/src/index.ts) | [`llm`](../packages/llm/llm) (`waterfall`) | [`agent-loop`](../packages/core/agent-loop), [`llm`](../packages/llm/llm), [`llm-replay`](../packages/support/llm-replay), [`session-checkpoint-policy`](../packages/session-persistence/session-checkpoint-policy), [`session-title`](../packages/session-title/session-title) | +| `llm/stream` | `waterfall` | [`packages/llm/llm/src/index.ts:57`](../packages/llm/llm/src/index.ts) | [`llm`](../packages/llm/llm) (`waterfall`) | [`agent-loop`](../packages/core/agent-loop), [`llm`](../packages/llm/llm), [`llm-replay`](../packages/support/llm-replay), [`session-checkpoint-policy`](../packages/session-persistence/session-checkpoint-policy), [`session-title`](../packages/session-title/session-title) | | `session/created` | `emit` | [`packages/core/session/src/index.ts:70`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | `apiproxy`, [`compact`](../packages/compact/compact), [`goal`](../packages/goal/goal), [`hook-protocol`](../packages/hooks/hook-protocol), [`jsonrpc`](../packages/ui/jsonrpc), [`llm-retry`](../packages/llm/llm-retry), [`plan-mode`](../packages/plan/plan-mode), [`session`](../packages/core/session), [`session-persistence`](../packages/session-persistence/session-persistence), [`session-telemetry`](../packages/telemetry/session-telemetry), [`tools`](../packages/core/tools), [`user-approval`](../packages/ui/user-approval) | | `session/disposed` | `emit` | [`packages/core/session/src/index.ts:80`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`agent-loop`](../packages/core/agent-loop), `apiproxy`, [`session-persistence`](../packages/session-persistence/session-persistence), [`session-telemetry`](../packages/telemetry/session-telemetry), [`session-title`](../packages/session-title/session-title) | | `session/event` | `emit` | [`packages/core/session/src/index.ts:92`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`acp`](../packages/acp/acp), `apiproxy`, [`cli-demo`](../packages/examples/cli-demo), [`compact`](../packages/compact/compact), [`compact-basic`](../packages/compact/compact-basic), [`goal`](../packages/goal/goal), [`goal-session`](../packages/goal/goal-session), [`hook-protocol`](../packages/hooks/hook-protocol), [`jsonrpc`](../packages/ui/jsonrpc), [`plan-mode`](../packages/plan/plan-mode), [`session`](../packages/core/session), [`session-persistence`](../packages/session-persistence/session-persistence), [`session-telemetry`](../packages/telemetry/session-telemetry), [`session-title`](../packages/session-title/session-title), [`token-meter`](../packages/llm/token-meter), [`tools`](../packages/core/tools), [`tui`](../packages/ui/tui), [`user-approval`](../packages/ui/user-approval), [`workspace-context`](../packages/context/workspace-context) | @@ -67,7 +68,7 @@ This matrix shows which packages dispatch each harness-owned event and which pac | `connection/reset` | `runtime` (`emit`) | - | | `internal/dispatch` | - | [`compact`](../packages/compact/compact), [`fs`](../packages/fs/fs), [`goal`](../packages/goal/goal), [`goal-session`](../packages/goal/goal-session), [`hook-protocol`](../packages/hooks/hook-protocol), [`llm-retry`](../packages/llm/llm-retry), [`permission`](../packages/ui/permission), [`plan-mode`](../packages/plan/plan-mode), [`pty-local`](../packages/pty/pty-local), `runtime`, [`sandbox-policy`](../packages/sandbox/sandbox-policy), [`scope`](../packages/core/scope), [`session`](../packages/core/session), [`subagent`](../packages/subagent/subagent), [`time-context`](../packages/context/time-context), [`tool-todo`](../packages/todo/tool-todo), [`tools`](../packages/core/tools), [`user-approval`](../packages/ui/user-approval), [`workflow`](../packages/workflow/workflow) | | `internal/plugin` | - | `hmr`, `modules`, `webserver` | -| `internal/status` | - | [`agent`](../packages/core/agent), `apiproxy` | +| `internal/status` | - | [`agent`](../packages/core/agent) | | `locale/change` | `locale` (`emit`) | `locale`, `ui-models`, `ui-settings-general` | | `slots/changed` | `runtime` (`emit`) | - | | `theme/change` | `ui-theme` (`emit`) | `ui-layout`, `ui-theme` | diff --git a/packages/client/runtime/README.i18n.yaml b/packages/client/runtime/README.i18n.yaml index a14a451b52..865b83389e 100644 --- a/packages/client/runtime/README.i18n.yaml +++ b/packages/client/runtime/README.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write packages/client/runtime/README.md -README.md: 935bb4908fc54fc629574963e28dc0ddcfb89a6c -README.zh.md: 85d444ee5d973bb989232ecaf00f2312d3195c8d +README.md: bc1149644d6112ca82c9a27912d1a58351cf84a5 +README.zh.md: 993625614061e819495b25f0851156ea20c62601 diff --git a/packages/client/runtime/README.md b/packages/client/runtime/README.md index 935bb4908f..bc1149644d 100644 --- a/packages/client/runtime/README.md +++ b/packages/client/runtime/README.md @@ -2,7 +2,7 @@ English | [中文](README.zh.md) -Client cordis boot and React-free object services: SlotsService wraps SlotCore and supplies renderer data sources; SessionsService owns Session objects, list/scope/history state; WorkspacesService depends on SessionsService and owns Workspace objects, list/actions, default-target derivation, and the New Session blank-reuse entry (`connectWorkspace`). The runtime fans the shared Host stream into both managers. Client sessions are always Host-born (Session+Agent+cwd in one `session.create`); the client holds no pre-entity session state — a session's Agent scope (the client mirror of host dsh-scope, keyed by the shared agent/session id) is born when its row enters the list mirror and dies with the prune. Contract: api-contracts v3 §4. `ConversationSnapshot` carries two Host-owned full-log projections. `todos` comes from the tail history page, survives older-page prepend, and follows live `todo/write` events. `metrics` comes from tail history and live `session/metrics` frames, survives older-page prepend, and accepts only nondecreasing log and projection revisions; a subscription baseline clears it before replay so a new stream generation can restart revisions safely. Missing metrics remain `null` rather than being inferred from the visible node window. +Client cordis boot and React-free object services: SlotsService wraps SlotCore and supplies renderer data sources; SessionsService owns Session objects, list/scope/history state; WorkspacesService depends on SessionsService and owns Workspace objects, list/actions, default-target derivation, and the New Session blank-reuse entry (`connectWorkspace`). The runtime fans the shared Host stream into both managers. Client sessions are always Host-born (Session+Agent+cwd in one `session.create`); the client holds no pre-entity session state — a session's Agent scope (the client mirror of host dsh-scope, keyed by the shared agent/session id) is born when its row enters the list mirror and dies with the prune. Contract: api-contracts v3 §4. `ConversationSnapshot` carries two Host-owned full-log projections. `todos` comes from the tail history page, survives older-page prepend, and follows live `todo/write` events. Durable `metrics` comes from tail history and live `session/metrics` frames, survives older-page prepend, and accepts only nondecreasing log and projection revisions. The Session separately retains capacity from the latest `session/model-request` observed on its current mux connection and overlays it onto metrics across ordinary usage/pressure updates. A later request replaces or clears that value, while `session/subscribed` clears both metrics ordering and capacity; reconnect, restore, and a new subscription therefore show no percentage until another request is observed. Missing metrics remain `null` rather than being inferred from the visible node window. ## Workspace and Session lists diff --git a/packages/client/runtime/README.zh.md b/packages/client/runtime/README.zh.md index 85d444ee5d..9936256140 100644 --- a/packages/client/runtime/README.zh.md +++ b/packages/client/runtime/README.zh.md @@ -2,7 +2,7 @@ [English](README.md) | 中文 -客户端 cordis 启动与不依赖 React 的对象服务:SlotsService 包装 SlotCore 并提供 renderer 数据源;SessionsService 拥有 Session 对象、列表/scope/history 状态;WorkspacesService 依赖 SessionsService,拥有 Workspace 对象、列表/操作、默认目标派生,以及 New Session 空会话复用入口(`connectWorkspace`)。运行时把共享 Host 流分发给两个 manager。客户端 Session 一律由 Host 出生(一次 `session.create` 同瞬产出 Session+Agent+cwd);客户端不持有任何实体化之前的会话状态——Agent scope(host dsh-scope 的客户端镜像,以 agent/session 共用 id 为键)在会话行进入列表镜像时出生,随 prune 死亡。契约:api-contracts v3 §4。`ConversationSnapshot` 携带两项由 Host 拥有的完整日志投影。`todos` 来自 history 尾页,在向前加载较早页面时保留,并随实时 `todo/write` 事件更新。`metrics` 来自 history 尾页和实时 `session/metrics` 帧,在向前加载较早页面时保留,并且只接受日志修订号与投影修订号均不减小的数据;订阅基线会在回放前将其清除,使新的流代次可以安全地从头开始计数修订号。缺失的 metrics 保持为 `null`,而不是根据可见节点窗口推断。 +客户端 cordis 启动与不依赖 React 的对象服务:SlotsService 包装 SlotCore 并提供 renderer 数据源;SessionsService 拥有 Session 对象、列表/scope/history 状态;WorkspacesService 依赖 SessionsService,拥有 Workspace 对象、列表/操作、默认目标派生,以及 New Session 空会话复用入口(`connectWorkspace`)。运行时把共享 Host 流分发给两个 manager。客户端 Session 一律由 Host 出生(一次 `session.create` 同瞬产出 Session+Agent+cwd);客户端不持有任何实体化之前的会话状态——Agent scope(host dsh-scope 的客户端镜像,以 agent/session 共用 id 为键)在会话行进入列表镜像时出生,随 prune 死亡。契约:api-contracts v3 §4。`ConversationSnapshot` 携带两项由 Host 拥有的完整日志投影。`todos` 来自 history 尾页,在向前加载较早页面时保留,并随实时 `todo/write` 事件更新。持久 `metrics` 来自 history 尾页和实时 `session/metrics` 帧,在向前加载较早页面时保留,并且只接受日志修订号与投影修订号均不减小的数据。Session 另行保留当前 mux 连接观察到的最新 `session/model-request` 容量,并在普通用量/压力更新期间把它覆盖到 metrics 上。后续请求会替换或清除该值,`session/subscribed` 则同时清除指标顺序状态与容量;因此,重连、恢复和新订阅都不会显示百分比,直到观察到另一次请求。缺失的 metrics 保持为 `null`,而不是根据可见节点窗口推断。 ## Workspace 与 Session 列表 diff --git a/packages/client/runtime/src/client/sessions/conversation.ts b/packages/client/runtime/src/client/sessions/conversation.ts index 202ddf7df2..4ced165e7a 100644 --- a/packages/client/runtime/src/client/sessions/conversation.ts +++ b/packages/client/runtime/src/client/sessions/conversation.ts @@ -12,6 +12,15 @@ import type { PendingInteraction } from './pending.ts' export type { TodoItem } +/** + * Durable Host metrics with the latest capacity observed on this live mux + * connection overlaid for presentation. + */ +export interface ConversationMetrics extends SessionMetrics { + /** Latest dispatched-request capacity; absent until observed or after reset/clear. */ + contextWindow?: number +} + /** Assistant content blocks sorted by what the UI cares about * (text body / collapsible reasoning / tool-call card head / other fallback). */ export type AssistantBlock = @@ -247,9 +256,9 @@ export interface ConversationSnapshot { * write (last write wins); empty = the log holds no plan. */ todos: readonly TodoItem[] /** - * Host-owned cumulative usage and current-context projection. Independent - * of `nodes` pagination; null until a tail response or live metrics frame - * supplies a current value. + * Host-owned cumulative usage/current pressure with live mux-local capacity + * overlaid. Independent of `nodes` pagination; null until a tail response or + * live metrics frame supplies a current durable value. */ - metrics: SessionMetrics | null + metrics: ConversationMetrics | null } diff --git a/packages/client/runtime/src/client/sessions/session.ts b/packages/client/runtime/src/client/sessions/session.ts index a8eecd1bac..08ed26f2a6 100644 --- a/packages/client/runtime/src/client/sessions/session.ts +++ b/packages/client/runtime/src/client/sessions/session.ts @@ -12,7 +12,7 @@ import type { import { transportError } from '@deepseek-ai/dsh-host-apiproxy/api' import type { ObservableSnapshot } from '../contract/store.ts' import type { - CodeSubCall, ComposerPhase, ConversationNode, ConversationSnapshot, OpenState, + CodeSubCall, ComposerPhase, ConversationMetrics, ConversationNode, ConversationSnapshot, OpenState, PromptError, QueuedMessage, RunningToolCall, } from './conversation.ts' import type { PendingInteraction } from './pending.ts' @@ -102,8 +102,10 @@ export class Session implements ObservableSnapshot { /** Current whole-list todo/write projection: each tail history response replaces it (an omitted * field is the authoritative empty list) and every live write overwrites it. */ private todos: readonly TodoItem[] = [] - /** Host-owned metrics projection; ordering resets on each subscribed baseline. */ - private metrics: SessionMetrics | null = null + /** Host-owned metrics with current-connection request capacity overlaid. */ + private metrics: ConversationMetrics | null = null + /** Latest capacity observed on this mux connection, independent of durable metrics arrival. */ + private contextWindow: number | undefined /** `run_code` sub-dispatches by parent callId (window-derived, like openCalls). Appends * copy-on-write the per-parent array so published snapshot references never mutate. */ private codeDispatches = new Map() @@ -367,6 +369,7 @@ export class Session implements ObservableSnapshot { this.queueRev++ this.notifier.markDirty() } + this.contextWindow = undefined if (this.metrics !== null) { this.metrics = null this.notifier.markDirty() @@ -377,6 +380,18 @@ export class Session implements ObservableSnapshot { this.installMetrics(frame.metrics) return } + case 'session/model-request': { + if (this.contextWindow === frame.contextWindow) return + this.contextWindow = frame.contextWindow + if (this.metrics !== null) { + const { contextWindow: _previous, ...durable } = this.metrics + this.metrics = frame.contextWindow === undefined + ? durable + : { ...durable, contextWindow: frame.contextWindow } + this.notifier.markDirty() + } + return + } case 'approval/requested': { const { type: _type, sessionId: _sid, ...payload } = frame this.mint(new PendingWait('approval', rpcId, this.sessionId, payload, m => this.api.respond(m))) @@ -800,7 +815,9 @@ export class Session implements ObservableSnapshot { || metrics.projectionRevision < current.projectionRevision ) ) return - this.metrics = metrics + this.metrics = this.contextWindow === undefined + ? metrics + : { ...metrics, contextWindow: this.contextWindow } this.notifier.markDirty() } diff --git a/packages/client/runtime/tests/session.spec.ts b/packages/client/runtime/tests/session.spec.ts index 6d3ce893c7..4c93475b79 100644 --- a/packages/client/runtime/tests/session.spec.ts +++ b/packages/client/runtime/tests/session.spec.ts @@ -51,7 +51,6 @@ function metrics( cacheReadTokens: 90, cacheWriteTokens: 3, contextTokens: 35, - contextWindow: 100, ...over, } } @@ -146,7 +145,7 @@ describe('live event path', () => { expect(session.getSnapshot().nodes).toEqual(before.nodes) }) - it('orders live metrics, rejects stale projections, and clears the value at a reconnect baseline', async () => { + it('retains live capacity across metrics, replaces or clears it on requests, and resets at subscription', async () => { const { session } = await opened() const current = metrics(8, 10) session.handleMuxEnvelope('m1' as never, { @@ -156,6 +155,20 @@ describe('live event path', () => { }) expect(session.getSnapshot().metrics).toBe(current) + session.handleMuxEnvelope('request-1' as never, { + type: 'session/model-request', + sessionId: SID, + turn: 1, + step: 1, + provider: 'test', + model: 'alpha', + contextWindow: 128_000, + }) + expect(session.getSnapshot().metrics).toEqual({ + ...current, + contextWindow: 128_000, + }) + session.handleMuxEnvelope('m2' as never, { type: 'session/metrics', sessionId: SID, @@ -166,7 +179,31 @@ describe('live event path', () => { sessionId: SID, metrics: metrics(7, 11, { uncachedInputTokens: 2 }), }) - expect(session.getSnapshot().metrics).toBe(current) + expect(session.getSnapshot().metrics).toEqual({ + ...current, + contextWindow: 128_000, + }) + + const ordinaryUpdate = metrics(9, 11, { contextTokens: 40 }) + session.handleMuxEnvelope('m4' as never, { + type: 'session/metrics', + sessionId: SID, + metrics: ordinaryUpdate, + }) + expect(session.getSnapshot().metrics).toEqual({ + ...ordinaryUpdate, + contextWindow: 128_000, + }) + + session.handleMuxEnvelope('request-2' as never, { + type: 'session/model-request', + sessionId: SID, + turn: 2, + step: 1, + provider: 'test', + model: 'without-capacity', + }) + expect(session.getSnapshot().metrics).toEqual(ordinaryUpdate) session.handleMuxEnvelope('sub' as never, { type: 'session/subscribed', @@ -174,13 +211,25 @@ describe('live event path', () => { lastSeq: 5, }) expect(session.getSnapshot().metrics).toBeNull() + session.handleMuxEnvelope('request-3' as never, { + type: 'session/model-request', + sessionId: SID, + turn: 3, + step: 1, + provider: 'test', + model: 'beta', + contextWindow: 256_000, + }) const nextGeneration = metrics(0, 10, { contextTokens: 20 }) - session.handleMuxEnvelope('m4' as never, { + session.handleMuxEnvelope('m5' as never, { type: 'session/metrics', sessionId: SID, metrics: nextGeneration, }) - expect(session.getSnapshot().metrics).toBe(nextGeneration) + expect(session.getSnapshot().metrics).toEqual({ + ...nextGeneration, + contextWindow: 256_000, + }) }) it('accumulates chunks into partial, then finalize swaps partial out as the node lands', async () => { diff --git a/packages/client/ui-conversation/README.i18n.yaml b/packages/client/ui-conversation/README.i18n.yaml index 33ec0b703d..7296e8b314 100644 --- a/packages/client/ui-conversation/README.i18n.yaml +++ b/packages/client/ui-conversation/README.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write packages/client/ui-conversation/README.md -README.md: 0674a596353aac024824489e5ca25615c7426bcd -README.zh.md: 65d3eba678b7190d97244a54c629b78c7d24b8c0 +README.md: 1f47260f9034f22ba560552e5a99b938bf14ed6d +README.zh.md: 7d9a9d2ab95d93668de216d64eee704bcc569547 diff --git a/packages/client/ui-conversation/README.md b/packages/client/ui-conversation/README.md index 0674a59635..1f47260f90 100644 --- a/packages/client/ui-conversation/README.md +++ b/packages/client/ui-conversation/README.md @@ -18,7 +18,7 @@ Per-session UI state for selection and the active view lives in the declared cha The composer bar declares session-scoped single seats for `'conversation.input.plan'` and `'conversation.input.model'`, plus list slots for overlay, dock, left, and right input extensions. InputBar renders the model seat immediately before its pending indicator and send/stop button. Feature packages own each control and its state; ui-conversation supplies placement, the `locked` owner prop, and the standard slot shares. The resident no-session shell uses `DisabledInputBar` and therefore dispatches no session-scoped control seats. -The chat stats line reads durable token counters and current-context pressure only from `ConversationSnapshot.metrics`; visible nodes supply only the existing turn/step counts. It renders uncached input, output, and cache reads as separate compact values, computes cache hit as `cacheRead / (uncachedInput + cacheRead)` without cache writes, and shows context occupancy against the selected route's exact capacity. Missing host data is labeled unknown, never reconstructed from a paged window. +The chat stats line reads durable token counters/current pressure plus the runtime's connection-local capacity overlay only from `ConversationSnapshot.metrics`; visible nodes supply only the existing turn/step counts. It renders uncached input, output, and cache reads as separate compact values, computes cache hit as `cacheRead / (uncachedInput + cacheRead)` without cache writes, and shows context occupancy only after the current mux connection observes a model request with capacity. Before that request, after reconnect/restore/new subscription, or after a request without capacity, the percentage is omitted and context is labeled unknown rather than queried ahead or reconstructed from history. `src/client/` is organized for the future package split: `contract/` is the sole inter-domain shared face (`slots.ts` slot declarations + composed slot props including the tool-row contract, `views.ts` shared primitives, `tool-call-model.ts`); the `skeleton/`, `chat/`, and `toolviews/` (sample registrants) domain directories import contract files and never each other; `apply.ts` is the only assembly point allowed to import all three domains. The `/client` export surface is the contract only — `apply`/`inject`, the two service classes, and the `contract/` type families; implementation components (skeleton, chat rows) and the store factory stay internal and reach the page exclusively through apply's slot registrations (tests take them via the `./src/*` subpath). diff --git a/packages/client/ui-conversation/README.zh.md b/packages/client/ui-conversation/README.zh.md index 65d3eba678..7d9a9d2ab9 100644 --- a/packages/client/ui-conversation/README.zh.md +++ b/packages/client/ui-conversation/README.zh.md @@ -18,7 +18,7 @@ todo 两个面就是在该形状上的两个注册项,都是普通注册方插 输入栏为 `'conversation.input.plan'` 和 `'conversation.input.model'` 声明会话作用域的单实例 seat,并为 overlay、dock、left 和 right 输入扩展声明列表 slot。InputBar 将模型 seat 渲染在 pending 指示器与发送/停止按钮之前。各功能包拥有相应控件及其状态;ui-conversation 提供放置位置、`locked` owner prop 和标准 slot share。常驻无会话壳使用 `DisabledInputBar`,因此不会分发任何会话作用域的控件 seat。 -聊天统计行只从 `ConversationSnapshot.metrics` 读取持久的 token 计数与当前上下文压力;可见节点仅提供既有的轮次和步骤计数。它以相互独立的紧凑值显示未缓存输入、输出与缓存读取,通过 `cacheRead / (uncachedInput + cacheRead)` 计算缓存命中率而不计入缓存写入,并根据所选路由的精确容量显示上下文占用率。Host 数据缺失时标为「未知」,绝不根据分页窗口重建。 +聊天统计行只从 `ConversationSnapshot.metrics` 读取持久的 token 计数/当前压力,以及运行时提供的连接本地容量覆盖值;可见节点仅提供既有的轮次和步骤计数。它以相互独立的紧凑值显示未缓存输入、输出与缓存读取,通过 `cacheRead / (uncachedInput + cacheRead)` 计算缓存命中率而不计入缓存写入,并且只有当前 mux 连接观察到带容量的模型请求后才显示上下文占用率。在该请求之前、重连/恢复/新订阅之后,或在请求不带容量之后,系统都会省略百分比,并把上下文标为「未知」,而不会提前查询或根据历史记录重建。 `src/client/` 按未来的包拆分组织:`contract/` 是唯一的跨领域共享表层(`slots.ts` slot 声明 + 组合后的 slot props,包括工具行契约、`views.ts` 共享原语、`tool-call-model.ts`);`skeleton/`、`chat/` 和 `toolviews/`(示例注册方)领域目录只导入 contract 文件,彼此绝不导入;`apply.ts` 是唯一允许导入全部三个领域的组装点。`/client` 导出表层只包含契约:`apply`/`inject`、两个服务类和 `contract/` 类型家族;实现组件(骨架、聊天行)与 store factory 保持内部状态,只能通过 apply 的 slot 注册到达页面(测试通过 `./src/*` 子路径获取它们)。 diff --git a/packages/client/ui-conversation/src/client/chat/StatsLine.tsx b/packages/client/ui-conversation/src/client/chat/StatsLine.tsx index 5c77b585e7..3aa36e2eb8 100644 --- a/packages/client/ui-conversation/src/client/chat/StatsLine.tsx +++ b/packages/client/ui-conversation/src/client/chat/StatsLine.tsx @@ -56,7 +56,7 @@ export function cacheHitPercent(metrics: SessionMetrics): number | null { /** * Current context occupancy using the TUI's integer rounding and upper clamp. - * @param metrics - Host-owned current pressure and exact route capacity. + * @param metrics - Host-owned pressure plus current-connection request capacity. * @returns occupancy percent, or null when either input is unavailable. */ export function contextPercent(metrics: SessionMetrics): number | null { diff --git a/packages/client/ui-conversation/tests/chat-stats-bash-sample.spec.tsx b/packages/client/ui-conversation/tests/chat-stats-bash-sample.spec.tsx index 1f986efc7d..5aa737cd36 100644 --- a/packages/client/ui-conversation/tests/chat-stats-bash-sample.spec.tsx +++ b/packages/client/ui-conversation/tests/chat-stats-bash-sample.spec.tsx @@ -127,6 +127,26 @@ describe('StatsLine', () => { expect(emptyView.container.textContent).toBe('') }) + it('renders durable counters without a percentage before live capacity is observed', () => { + const { source } = makeSource({ + nodes: [assistant(1, 1)], + metrics: { + logRevision: 4, + projectionRevision: 1, + uncachedInputTokens: 120, + outputTokens: 20, + cacheReadTokens: 30, + cacheWriteTokens: 10, + contextTokens: 8_000, + }, + }) + const view = render() + expect(view.getByText( + '120 uncached input · 20 output · 30 cache read · cache hit 20% · context unknown · 1 turns · 1 steps', + )).toBeTruthy() + expect(view.container.textContent).not.toContain('% of') + }) + it('renders honest unknowns when the host projection is missing', () => { const { source } = makeSource({ nodes: [assistant(1, 1)] }) const view = render() diff --git a/packages/cordis/tool-cordis/src/api-catalog.ts b/packages/cordis/tool-cordis/src/api-catalog.ts index 47f8a1a54f..897935dad6 100644 --- a/packages/cordis/tool-cordis/src/api-catalog.ts +++ b/packages/cordis/tool-cordis/src/api-catalog.ts @@ -397,8 +397,8 @@ export const SERVICE_API: readonly ServiceApiEntry[] = [ jsDoc: '/**\n * Resolve one call under its current adapter registration. The returned\n * one-shot handle keeps that registration across header logging and dispatch,\n * so HMR cannot combine one adapter\'s capability result with another adapter.\n * @param config - provider/model route and optional request controls.\n * @param signal - optional cancellation for adapter-owned capability lookup.\n * @returns a prepared config and its registration-bound stream entry point.\n */', }, { - signature: 'stream(options: GenerateOptions): AsyncIterable', - jsDoc: '/**\n * Stream one model call as raw chunks (token-level deltas). Throws\n * `LlmError` with code `NO_ADAPTER` if no adapter is registered for\n * `options.provider`. Replay state is retained only when the same adapter\n * instance owns its historical provider and the target provider. Final\n * adapter selection remains fixed through asynchronous exact-model resolution\n * and dispatch. Selection, dispatch, and iteration failures retain their\n * original Error identity and are tagged in a call-local scope for narrow\n * agent-loop request recovery; middleware and nested-call failures remain\n * untagged for the outer call.\n * @param options - the full request; `options.provider` selects the adapter.\n * @returns the chunk stream, possibly wrapped by `llm/stream` listeners.\n */', + signature: 'stream(options: GenerateOptions, onDispatched?: () => void): AsyncIterable', + jsDoc: '/**\n * Stream one model call as raw chunks (token-level deltas). Throws\n * `LlmError` with code `NO_ADAPTER` if no adapter is registered for\n * `options.provider`. Replay state is retained only when the same adapter\n * instance owns its historical provider and the target provider. Final\n * adapter selection remains fixed through asynchronous exact-model resolution\n * and dispatch. Selection, dispatch, and iteration failures retain their\n * original Error identity and are tagged in a call-local scope for narrow\n * agent-loop request recovery; middleware and nested-call failures remain\n * untagged for the outer call.\n * @param options - the full request; `options.provider` selects the adapter.\n * @param onDispatched - contained Agent-loop notification hook invoked after\n * a stream handle is constructed and before its adapter is iterated.\n * @returns the chunk stream, possibly wrapped by `llm/stream` listeners.\n */', }, ], }, @@ -1048,6 +1048,13 @@ export const EVENT_API: readonly EventApiEntry[] = [ jsDoc: '/**\n * An item entered the queued or steering inbox. `placement` is the\n * acceptance-time routing result; listeners must not reconstruct it from\n * later agent or session state.\n * @param agent - the owning agent.\n * @param message - accepted content, source, and correlation identity.\n * @param placement - resolved queued or steering placement.\n * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent.\n * @mode emit\n */', summary: 'An item entered the queued or steering inbox.', }, + { + name: 'agent/model-request', + mode: 'emit', + signature: '\'agent/model-request\'(this: Scoped, agent: Agent, turn: number, step: number, request: AgentModelRequest): void', + jsDoc: '/**\n * One model request constructed its final stream handle and is about to\n * iterate it. This live notification is not durable or replayed; failed or\n * aborted iteration still has a dispatch, while preparation and\n * synchronous stream-construction failures do not. Listener failures are\n * contained and cannot affect the request.\n * @param agent - the agent dispatching the model request.\n * @param turn - the open turn number.\n * @param step - the request\'s step number.\n * @param request - final route plus registration-bound context capacity.\n * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent.\n * @mode emit\n */', + summary: 'One model request constructed its final stream handle and is about to iterate it.', + }, { name: 'agent/prompt-submit', mode: 'waterfall', @@ -1803,7 +1810,7 @@ export const TYPE_API: readonly TypeApiEntry[] = [ }, { name: 'PreparedLlmCall', - declaration: 'export interface PreparedLlmCall {\n readonly config: LlmCallConfig;\n stream(options: GenerateOptions): AsyncIterable;\n}', + declaration: 'export interface PreparedLlmCall {\n readonly config: LlmCallConfig;\n readonly context?: LlmModelContext;\n stream(options: GenerateOptions, onDispatched?: () => void): AsyncIterable;\n}', }, { name: 'PreparedReferencedMessage', diff --git a/packages/core/agent-loop/README.i18n.yaml b/packages/core/agent-loop/README.i18n.yaml index 8652771f93..8b72206a87 100644 --- a/packages/core/agent-loop/README.i18n.yaml +++ b/packages/core/agent-loop/README.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write packages/core/agent-loop/README.md -README.md: c12140f27aed400b0f7b4246700473e877d37632 -README.zh.md: 6394cd86f5f3241be07ef711c76079624bce1bfe +README.md: 39eaafd3abb2faef045c1a2f694d8dffe95c28b0 +README.zh.md: f8cc972e957fe95f652d54e859f8e5411288db51 diff --git a/packages/core/agent-loop/README.md b/packages/core/agent-loop/README.md index c12140f27a..39eaafd3ab 100644 --- a/packages/core/agent-loop/README.md +++ b/packages/core/agent-loop/README.md @@ -62,7 +62,7 @@ The driver owns one agent for its lifetime and runs inside `ctx.agents.withIniti Every provider call that reaches a successful finish appends exactly one `assistant/message` completion anchor, including content-less calls and `max-tokens` finishes. The anchor records the assembled content as-is, retains exact chunk provenance (`[]` for a stream with no chunks), and includes usage when available; empty content stays out of derived message history. -After `agent/request` returns a provider/model call config, the loop asks `ctx.llm.prepareCall()` to validate any adapter-owned reasoning effort and materialize its configured default under the active turn signal. The prepared call retains the exact adapter registration across this asynchronous resolution, `request/header` logging, and terminal dispatch, so HMR cannot mix one adapter's capability result with another adapter's request. The effective config is logged before dispatch, so a listener can change effort between steps without hidden request drift. A route with no registered adapter preserves the proposed config so an `llm/stream` listener can own and short-circuit it; unhandled terminal dispatch still fails with `NO_ADAPTER`. A new loop instance restores the last effort only when its initial provider/model route exactly matches the logged route; a route change discards that opaque model-owned ID and resolves the new model independently. +After `agent/request` returns a provider/model call config, the loop asks `ctx.llm.prepareCall()` to validate any adapter-owned reasoning effort, materialize its configured default, and retain available context metadata from that same exact-model lookup under the active turn signal. The prepared call retains the exact adapter registration across this asynchronous resolution, `request/header` logging, and terminal dispatch, so HMR cannot mix one adapter's capability result with another adapter's request. After the final stream handle is constructed and before adapter iteration, the loop emits one contained live `agent/model-request` notification with turn, step, final provider/model, and optional registration-bound capacity. Preparation or synchronous stream-construction failures emit nothing; later failure or abortion remains an observed dispatch. The effective config is logged before dispatch, so a listener can change effort between steps without hidden request drift. A route with no registered adapter preserves the proposed config so an `llm/stream` listener can own and short-circuit it; unhandled terminal dispatch still fails with `NO_ADAPTER`. A new loop instance restores the last effort only when its initial provider/model route exactly matches the logged route; a route change discards that opaque model-owned ID and resolves the new model independently. Plugin failure ends the current turn, not the loop. Only final adapter dispatch/iteration failures and terminal in-band error or aborted finishes enter `agent/request-error`; middleware, result processing, tools, and other extension failures close directly. Recovery receives the exact live error, immutable provider facts, immutable prior failures, the immutable retry policy of the adapter registration that served the request, and the turn signal after the failed step closes; the policy is absent if no final adapter served it. A handling listener returns `{ kind: 'retry' }`; the loop closes the failed turn with its error and opens one numbered retry turn without an intervening idle notification. Success clears the consecutive history, and an unhandled failure is terminal. AgentLoop owns one cancellation signal for the current admission or turn. An effective `cancel(cause)` clears pending work unless `keepInbox` is set and cooperatively aborts that signal; idle cancellation is a no-op. Durable `turn/end` records `aborted` for `user` and `parent`, while disposal records `disposed`; undispatched model tool calls receive synthetic `tool/call` and `ABORTED_BEFORE_DISPATCH` result pairs. The cancellation cause changes reporting, not how result context finalized after cancellation is handled. Disposal waits for signal-ignoring work before registry removal. The [explicit-cancellation decision](../../../.agents/notes/implemented/architecture/2026-07-16-explicit-turn-cancellation.md) owns the lifecycle and race contract. diff --git a/packages/core/agent-loop/README.zh.md b/packages/core/agent-loop/README.zh.md index 6394cd86f5..f8cc972e95 100644 --- a/packages/core/agent-loop/README.zh.md +++ b/packages/core/agent-loop/README.zh.md @@ -62,7 +62,7 @@ interface Config { 每次提供方调用成功结束时,都会恰好追加一个 `assistant/message` 完成锚点,包括无内容调用和以 `max-tokens` 结束的调用。该锚点原样记录组装后的内容,保留确切的 chunk 溯源(流没有 chunk 时为 `[]`),并在用量可用时包含用量;空内容不会进入派生消息历史。 -在 `agent/request` 返回提供方/模型调用配置后,循环会调用 `ctx.llm.prepareCall()`,在活跃轮次信号的控制下校验由适配器持有的推理(reasoning)强度,并填入其配置默认值。准备完成的调用会在这次异步解析、`request/header` 日志记录和最终分派期间保留同一项确切的适配器注册,因此 HMR(热模块替换)不会把某个适配器的能力解析结果与另一适配器的请求混用。生效配置会在分派前写入日志,因此监听器可以在步骤之间更改推理强度,而不会产生未记录的请求变化。没有已注册适配器的路由会保留原定配置,使 `llm/stream` 监听器可以接管并短路该请求;最终分派仍会以 `NO_ADAPTER` 拒绝未得到处理的路由。新循环实例仅在初始提供方/模型路由与日志路由完全一致时恢复上次的推理强度;路由变化会丢弃由前一模型持有的不透明 ID,并单独解析新模型。 +在 `agent/request` 返回提供方/模型调用配置后,循环会调用 `ctx.llm.prepareCall()`,在活跃轮次信号的控制下校验由适配器持有的推理(reasoning)强度、填入其配置默认值,并从同一次精确模型查询中保留可用的上下文元数据。准备完成的调用会在这次异步解析、`request/header` 日志记录和最终分派期间保留同一项确切的适配器注册,因此 HMR(热模块替换)不会把某个适配器的能力解析结果与另一适配器的请求混用。最终流句柄构造完成后、适配器开始迭代前,循环会发出一条失败会被收容的实时 `agent/model-request` 通知,其中包含轮次、步骤、最终提供方/模型,以及可选的、与注册项绑定的容量。准备阶段失败或同步流构造失败不会发出通知;之后即使失败或中止,该请求仍视为已观察到的分派。生效配置会在分派前写入日志,因此监听器可以在步骤之间更改推理强度,而不会产生未记录的请求变化。没有已注册适配器的路由会保留原定配置,使 `llm/stream` 监听器可以接管并短路该请求;最终分派仍会以 `NO_ADAPTER` 拒绝未得到处理的路由。新循环实例仅在初始提供方/模型路由与日志路由完全一致时恢复上次的推理强度;路由变化会丢弃由前一模型持有的不透明 ID,并单独解析新模型。 插件失败会结束当前轮次,而不是结束循环。只有最终适配器分发/迭代失败以及带内的终止错误或中止结束才进入 `agent/request-error`;中间件、结果处理、工具及其他扩展失败会直接关闭轮次。失败步骤关闭后,恢复逻辑会接收确切的实时错误、不可变的提供方事实、不可变的先前失败、为请求提供服务的适配器注册所对应的不可变重试策略,以及轮次信号;如果没有最终适配器为其提供服务,则该策略缺失。处理失败的监听器返回 `{ kind: 'retry' }`;循环用其错误关闭失败轮次,并在不插入空闲通知的情况下开启一个编号重试轮次。成功会清除连续失败历史;未被处理的失败是终态。AgentLoop 为当前接纳或轮次拥有一个取消信号。有效的 `cancel(cause)` 在未设置 `keepInbox` 时清除待处理工作,并以协作方式中止该信号;空闲取消是空操作。持久 `turn/end` 为 `user` 和 `parent` 记录 `aborted`,dispose(资源释放)则记录 `disposed`;未分发的模型工具调用会收到合成的 `tool/call` 与 `ABORTED_BEFORE_DISPATCH` 结果对。取消原因只改变报告方式,不改变对取消后已定案结果上下文的处理。dispose 会等待忽略信号的工作完成,然后才从注册表移除。[显式取消决策](../../../.agents/notes/implemented/architecture/2026-07-16-explicit-turn-cancellation.md)规定生命周期与竞态契约。 diff --git a/packages/core/agent-loop/src/agent.ts b/packages/core/agent-loop/src/agent.ts index 2ba2b6ab88..9c930912ba 100644 --- a/packages/core/agent-loop/src/agent.ts +++ b/packages/core/agent-loop/src/agent.ts @@ -487,7 +487,24 @@ export class ReactLoopAgent implements Agent { const assembler = new BlockAssembler() const chunkSeqs: number[] = [] - const stream = preparedCall?.stream(request) ?? this.loopCtx.llm.stream(request) + const onDispatched = (): void => { + emitAgentEvent( + this.loopCtx, + this, + 'agent/model-request', + turn, + step, + { + provider: request.provider, + model: request.model, + ...preparedCall?.context === undefined + ? {} + : { contextWindow: preparedCall.context.contextWindow }, + }, + ) + } + const stream = preparedCall?.stream(request, onDispatched) + ?? this.loopCtx.llm.stream(request, onDispatched) try { for await (const chunk of stream) { signal.throwIfAborted() diff --git a/packages/core/agent-loop/tests/request-reconstruction.spec.ts b/packages/core/agent-loop/tests/request-reconstruction.spec.ts index 0298aaa15e..9da0e30f81 100644 --- a/packages/core/agent-loop/tests/request-reconstruction.spec.ts +++ b/packages/core/agent-loop/tests/request-reconstruction.spec.ts @@ -7,8 +7,10 @@ import { describe, expect, it } from 'vitest' import { Context } from 'cordis' -import LlmService, { LlmError, ReasoningEffortId } from '@deepseek-ai/dsh-llm' -import type { GenerateOptions, LlmModelReasoningInfo, LlmResolvedModelInfo } from '@deepseek-ai/dsh-llm' +import LlmService, { LlmAdapter, LlmError, ReasoningEffortId } from '@deepseek-ai/dsh-llm' +import type { + GenerateOptions, LlmModelReasoningInfo, LlmResolvedModelInfo, StreamChunk, +} from '@deepseek-ai/dsh-llm' import SessionStore, { Session, SessionId, foldRequestHeader } from '@deepseek-ai/dsh-session' import SystemPrompt from '@deepseek-ai/dsh-system-prompt' import ToolRegistry, { defineContentToolFixture } from '@deepseek-ai/dsh-tools' @@ -179,6 +181,7 @@ describe('request stability across the loop', () => { provider, id: model, name: model, + context: { contextWindow: 64_000 }, reasoning: await reasoning.promise, } } @@ -189,6 +192,12 @@ describe('request stability across the loop', () => { }) const disposeFirst = ctx.llm.registerAdapter(['mock'], first) const agent = ctx.agentLoop.create(SessionId('effort-hmr'), { provider: 'mock', model: 'mock' }) + const dispatched: number[] = [] + ctx.on('agent/model-request', (subject, _turn, _step, request) => { + if (subject === agent && request.contextWindow !== undefined) { + dispatched.push(request.contextWindow) + } + }) send(agent, 'go') await started.promise @@ -204,6 +213,7 @@ describe('request stability across the loop', () => { ReasoningEffortId('high'), ]) expect(second.requests).toHaveLength(0) + expect(dispatched).toEqual([64_000]) const headers = agent.session.events.filter(event => event.type === 'request/header') expect(headers.at(-1)?.data.header.config.reasoningEffort).toBe(ReasoningEffortId('high')) }) @@ -308,6 +318,94 @@ describe('request stability across the loop', () => { }) }) + it('notifies one contained live model-request edge only after successful stream construction', async () => { + const ctx = new Context() + await ctx.plugin(LlmService) + await ctx.plugin(SessionStore) + await ctx.plugin(SystemPrompt, { persona: 'stable base' }) + await ctx.plugin(ToolRegistry) + await ctx.plugin(AgentRegistry) + await ctx.plugin(AgentLoop, { agents: [] }) + let resolutions = 0 + const adapter = new class extends LlmAdapter { + override resolveModel(provider: string, model: string): Promise { + resolutions += 1 + return Promise.resolve({ + provider, + id: model, + name: model, + ...model === 'capacity' + ? { context: { contextWindow: 128_000 } } + : {}, + }) + } + + override stream(options: GenerateOptions): AsyncIterable { + if (options.model === 'sync-failure') throw new LlmError('construction failed', 'CONSTRUCTION') + if (options.model === 'async-failure') { + return { + [Symbol.asyncIterator]: () => ({ + next: () => Promise.reject(new LlmError('iteration failed', 'ITERATION')), + }), + } + } + return (async function* () { + yield* textResponse(options.model) + })() + } + }() + ctx.llm.registerAdapter(['mock'], adapter) + const agent = ctx.agentLoop.create(SessionId('model-request-live'), { + provider: 'mock', + model: 'capacity', + }) + const observed: { + turn: number + step: number + provider: string + model: string + contextWindow?: number + }[] = [] + ctx.on('agent/model-request', (subject) => { + if (subject === agent) throw new Error('observer failed') + }) + ctx.on('agent/model-request', (subject, turn, step, request) => { + if (subject === agent) observed.push({ turn, step, ...request }) + }) + ctx.on('agent/request', async (_subject, turn, _step, _signal, next) => ({ + ...await next(), + model: ['capacity', 'unknown', 'async-failure', 'sync-failure'][turn - 1]!, + })) + + for (const prompt of ['one', 'two', 'three', 'four']) { + send(agent, prompt) + await waitForIdle(ctx, agent) + } + + expect(observed).toEqual([ + { + turn: 1, + step: 1, + provider: 'mock', + model: 'capacity', + contextWindow: 128_000, + }, + { + turn: 2, + step: 1, + provider: 'mock', + model: 'unknown', + }, + { + turn: 3, + step: 1, + provider: 'mock', + model: 'async-failure', + }, + ]) + expect(resolutions).toBe(4) + }) + it('a compaction replace rewrites the resend, and the log explains it', async () => { const adapter = new MockAdapter([textResponse('one'), textResponse('two')]) const ctx = await harness(adapter) diff --git a/packages/core/agent/README.i18n.yaml b/packages/core/agent/README.i18n.yaml index 4cd42522ca..00465d7021 100644 --- a/packages/core/agent/README.i18n.yaml +++ b/packages/core/agent/README.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write packages/core/agent/README.md -README.md: bb48fd8b227484a43af8f9f9e55f8adc8b990ce6 -README.zh.md: 531db9905b3c091a5c129d31e944e4095e66123b +README.md: 304347d32df4546389ee2b45230d4e80acc60942 +README.zh.md: d9adbe20015aadbad26288d43df92f80a3f550bf diff --git a/packages/core/agent/README.md b/packages/core/agent/README.md index bb48fd8b22..304347d32d 100644 --- a/packages/core/agent/README.md +++ b/packages/core/agent/README.md @@ -48,7 +48,7 @@ Agent *creation* is provided by the plugin implementing `AgentFactory` (`dsh-age The lifecycle edges have two important local caveats. `agent/created` runs after scoped setup and after both session and agent registry entries exist. Setup is trusted composition-only code; the immediately following non-vetoing `agent/session-start` notification is the first supported startup injection point. `agent/disposed` always means the exact agent has left the registry. AgentLoop emits it after its driver is quiescent, while ordered teardown may still be detaching the session and unwinding the scope; custom agents registered directly own any stronger driver-ordering contract themselves. -Most interception points are cooperative waterfalls. Turn-scoped asynchronous seams receive one explicit `AbortSignal`, with `signal` immediately before a waterfall's final `next`; listeners may cooperate but must not retain it as authority over another turn. `agent/step` is the serial checkpoint before request derivation, while `agent/request-error` is the failed-model-request recovery waterfall: it receives the exact error, normalized failure facts, and signal after the failed step closes. A listener returns `{ kind: 'retry' }` without calling `next()` when it owns recovery; the loop closes the failed turn and opens one numbered retry turn. `agent/turn-stopping` runs before an otherwise completed turn closes. Ordinary queued prompts remain intact. Effective broad cancellation first emits the observe-only `agent/cancel-requested` with its resolved typed cause, then clears queues and aborts; notification failures are contained and cannot veto the stop. The [explicit-cancellation decision](../../../.agents/notes/implemented/architecture/2026-07-16-explicit-turn-cancellation.md) owns signal lifetime; the [agent-scope runtime-design Agent Note](../../../.agents/notes/implemented/architecture/2026-07-12-agent-scope-runtime-design.md#three-execution-boundaries-are-deliberately-one-way) owns scoped dispatch and terminal settlement. +Most interception points are cooperative waterfalls. Turn-scoped asynchronous seams receive one explicit `AbortSignal`, with `signal` immediately before a waterfall's final `next`; listeners may cooperate but must not retain it as authority over another turn. `agent/step` is the serial checkpoint before request derivation, while the contained `agent/model-request` notification reports a final dispatched route and optional registration-bound context capacity without becoming durable state. `agent/request-error` is the failed-model-request recovery waterfall: it receives the exact error, normalized failure facts, and signal after the failed step closes. A listener returns `{ kind: 'retry' }` without calling `next()` when it owns recovery; the loop closes the failed turn and opens one numbered retry turn. `agent/turn-stopping` runs before an otherwise completed turn closes. Ordinary queued prompts remain intact. Effective broad cancellation first emits the observe-only `agent/cancel-requested` with its resolved typed cause, then clears queues and aborts; notification failures are contained and cannot veto the stop. The [explicit-cancellation decision](../../../.agents/notes/implemented/architecture/2026-07-16-explicit-turn-cancellation.md) owns signal lifetime; the [agent-scope runtime-design Agent Note](../../../.agents/notes/implemented/architecture/2026-07-12-agent-scope-runtime-design.md#three-execution-boundaries-are-deliberately-one-way) owns scoped dispatch and terminal settlement. `PromptDecision.additionalContexts` is an array so every context keeps its own source. Allowed prompt content and every additional context become separate model-facing `user/message` events before the turn runs. A listener that wraps a downstream allow preserves its `content` and `additionalContexts` unless it intentionally replaces either field; the returned allow is authoritative. diff --git a/packages/core/agent/README.zh.md b/packages/core/agent/README.zh.md index 531db9905b..d9adbe2001 100644 --- a/packages/core/agent/README.zh.md +++ b/packages/core/agent/README.zh.md @@ -48,7 +48,7 @@ Agent *创建* 由实现 `AgentFactory` 的插件(`dsh-agent-loop`)提供, 生命周期边有两个重要的本地注意事项。`agent/created` 在作用域 setup 之后、会话与 agent 注册表条目都存在之后运行。Setup 是受信任、仅用于组合的代码;紧随其后且不可 veto 的 `agent/session-start` 通知是第一个受支持的启动注入点。`agent/disposed` 始终表示确切 agent 已离开注册表。AgentLoop 在其驱动器静默后发出该事件,而有序 teardown 此时可能仍在分离会话并撤销作用域;直接注册的自定义 agent 自行拥有任何更强的驱动器顺序契约。 -大多数拦截点都是协作式 waterfall。轮次作用域的异步 seam 接收一个显式 `AbortSignal`,其中 `signal` 紧邻 waterfall 最终的 `next`;监听器可以配合,但不得将它保留为控制另一轮次的权限。`agent/step` 是派生请求前的串行检查点,而 `agent/request-error` 是失败模型请求的恢复 waterfall:失败步骤关闭后,它接收确切错误、规范化失败事实和信号。拥有恢复权的监听器返回 `{ kind: 'retry' }` 且不调用 `next()`;循环会关闭失败轮次,并打开一个编号重试轮次。`agent/turn-stopping` 在本可完成的轮次关闭前运行。普通排队提示词保持原样。有效的广义取消会先发出只观测的 `agent/cancel-requested` 及其解析后的类型化原因,再清空队列并中止;通知失败会被收容,不能 veto 停止。信号生命周期由[显式取消决策](../../../.agents/notes/implemented/architecture/2026-07-16-explicit-turn-cancellation.md)拥有;作用域分发与终止结算由 [agent 作用域 runtime 设计 Agent Note](../../../.agents/notes/implemented/architecture/2026-07-12-agent-scope-runtime-design.md#three-execution-boundaries-are-deliberately-one-way)拥有。 +大多数拦截点都是协作式 waterfall。轮次作用域的异步 seam 接收一个显式 `AbortSignal`,其中 `signal` 紧邻 waterfall 最终的 `next`;监听器可以配合,但不得将它保留为控制另一轮次的权限。`agent/step` 是派生请求前的串行检查点;`agent/model-request` 是失败会被收容的通知,它会报告最终已分派路由及可选的、与注册项绑定的上下文容量,但不会成为持久状态。`agent/request-error` 是失败模型请求的恢复 waterfall:失败步骤关闭后,它接收确切错误、规范化失败事实和信号。拥有恢复权的监听器返回 `{ kind: 'retry' }` 且不调用 `next()`;循环会关闭失败轮次,并打开一个编号重试轮次。`agent/turn-stopping` 在本可完成的轮次关闭前运行。普通排队提示词保持原样。有效的广义取消会先发出只观测的 `agent/cancel-requested` 及其解析后的类型化原因,再清空队列并中止;通知失败会被收容,不能 veto 停止。信号生命周期由[显式取消决策](../../../.agents/notes/implemented/architecture/2026-07-16-explicit-turn-cancellation.md)拥有;作用域分发与终止结算由 [agent 作用域 runtime 设计 Agent Note](../../../.agents/notes/implemented/architecture/2026-07-12-agent-scope-runtime-design.md#three-execution-boundaries-are-deliberately-one-way)拥有。 `PromptDecision.additionalContexts` 是数组,因此每个上下文都保留自己的来源。获准的提示词内容与每个附加上下文都会在轮次运行前成为各自独立、面向模型的 `user/message` 事件。包装下游允许决策的监听器会保留其 `content` 与 `additionalContexts`,除非有意替换任一字段;返回的允许决策是权威来源。 diff --git a/packages/core/agent/src/types.ts b/packages/core/agent/src/types.ts index b61ebadb86..5aac3152b2 100644 --- a/packages/core/agent/src/types.ts +++ b/packages/core/agent/src/types.ts @@ -116,6 +116,16 @@ export type PromptDecision = /** Model-request failure with an optional machine-routable provider code. */ export type RequestError = Error & { code?: string } +/** Live metadata for one model request that reached adapter dispatch. */ +export interface AgentModelRequest { + /** Final registered provider route. */ + readonly provider: string + /** Final adapter-owned model id. */ + readonly model: string + /** Registration-bound context capacity when the adapter exposed one. */ + readonly contextWindow?: number +} + /** Action returned by a listener that owns model-request recovery. */ export type RequestErrorAction = { kind: 'retry' } | undefined @@ -360,6 +370,20 @@ declare module 'cordis' { * @mode waterfall */ 'agent/request'(this: Scoped, agent: Agent, turn: number, step: number, signal: AbortSignal, next: () => Promise): Promise + /** + * One model request constructed its final stream handle and is about to + * iterate it. This live notification is not durable or replayed; failed or + * aborted iteration still has a dispatch, while preparation and + * synchronous stream-construction failures do not. Listener failures are + * contained and cannot affect the request. + * @param agent - the agent dispatching the model request. + * @param turn - the open turn number. + * @param step - the request's step number. + * @param request - final route plus registration-bound context capacity. + * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent. + * @mode emit + */ + 'agent/model-request'(this: Scoped, agent: Agent, turn: number, step: number, request: AgentModelRequest): void /** * Handle a model-request failure after its failed step has closed but * before the failed turn closes. A listener returns `{ kind: 'retry' }` diff --git a/packages/core/scope/src/scoped-events.generated.ts b/packages/core/scope/src/scoped-events.generated.ts index 89515bf3c2..a7bd5c93d2 100644 --- a/packages/core/scope/src/scoped-events.generated.ts +++ b/packages/core/scope/src/scoped-events.generated.ts @@ -15,6 +15,7 @@ const scopedSubjectResolvers: Readonly args[0], 'agent/inbox/discard': args => args[0], 'agent/inbox/enqueue': args => args[0], + 'agent/model-request': args => args[0], 'agent/prompt-submit': args => args[0], 'agent/request': args => args[0], 'agent/request-error': args => args[0], diff --git a/packages/core/scope/tests/invariant.spec.ts b/packages/core/scope/tests/invariant.spec.ts index bc1224d86b..d5cfea85ce 100644 --- a/packages/core/scope/tests/invariant.spec.ts +++ b/packages/core/scope/tests/invariant.spec.ts @@ -49,6 +49,7 @@ describe('scoped-dispatch invariants', () => { 'agent/step': [agent, 1, 1, signal], 'agent/prompt-submit': [agent, [], { kind: 'user' }, signal, () => Promise.resolve({ kind: 'allow' })], 'agent/request': [agent, 1, 1, signal, () => Promise.resolve(config)], + 'agent/model-request': [agent, 1, 1, { provider: 'p', model: 'm', contextWindow: 128_000 }], 'agent/request-error': [ agent, 1, diff --git a/packages/host/apiproxy/README.i18n.yaml b/packages/host/apiproxy/README.i18n.yaml index 016ad9620f..2bab687635 100644 --- a/packages/host/apiproxy/README.i18n.yaml +++ b/packages/host/apiproxy/README.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write packages/host/apiproxy/README.md -README.md: e2ba8c0e5b2620d095636503f47fac5480c6c2fa -README.zh.md: 2aae0bd683e9ea1e1fb43f73000420dc2354ee0b +README.md: d3d1711242f71832f63eb570242fdf9e149cc988 +README.zh.md: 4e52058c1795ea23a9ca8ed46b8890c85f129eb4 diff --git a/packages/host/apiproxy/README.md b/packages/host/apiproxy/README.md index e2ba8c0e5b..d3d1711242 100644 --- a/packages/host/apiproxy/README.md +++ b/packages/host/apiproxy/README.md @@ -20,7 +20,9 @@ Workspace and Session lists are separate reconnect baselines. `workspace.create` `host.openPath` opens a filesystem path with the operating system's default application (`open` on macOS, `Invoke-Item` on Windows, `xdg-open` on Linux). The opener is injectable for tests. The browser carrier applies the same loopback, same-origin restriction as `host.pickDirectory`. -`session.history` pages on message boundaries, and its tail page (no `beforeSeq`) carries session-level projections the page window cannot supply: the in-flight partial's chunk events; `todos`, the latest `todo/write` whole-list projection; and `metrics`, full-log usage deduplicated by `(turn, step)` plus current token-meter pressure and exact selected-route capacity when available. Older pages omit the session-level projections. Live `session/metrics` mux frames carry monotonic log/projection revisions, so clients reject stale frames and preserve the counters while prepending older pages. Cache reads and writes remain disjoint buckets; the cache-hit denominator is uncached input plus cache reads. +`session.history` pages on message boundaries, and its tail page (no `beforeSeq`) carries session-level projections the page window cannot supply: the in-flight partial's chunk events; `todos`, the latest `todo/write` whole-list projection; and `metrics`, full-log usage deduplicated by `(turn, step)` plus current token-meter pressure. Older pages omit the session-level projections. Live `session/metrics` mux frames carry monotonic log/projection revisions, so clients reject stale frames and preserve the counters while prepending older pages. Cache reads and writes remain disjoint buckets; the cache-hit denominator is uncached input plus cache reads. + +Context capacity uses a distinct transient `session/model-request` mux frame emitted from the contained Agent notification after an actual request reaches dispatch. It carries turn, step, final provider/model, and optional capacity only to mux connections already open at that instant. `session.history`, mux subscription baselines, reconnects, and session restore never query or replay prior capacity; a frame without capacity explicitly clears the earlier connection-local value. The `command.*` and `skill.*` domains expose the host command registry and skill catalog to clients. Every method addresses one session's agent by `sessionId` (a served session always has an Agent; `command.*` resumes cold sessions through the same path as `session.*`, while `skill.list` resolves the project root from the session header without touching the Agent registry). `command.execute` runs a slash-command line host-side and returns a detached result; the carrier's request signal cancels the running handler. `host/commands-changed` is the catalog invalidation frame: clients refetch `command.list` instead of diffing. diff --git a/packages/host/apiproxy/README.zh.md b/packages/host/apiproxy/README.zh.md index 2aae0bd683..4e52058c17 100644 --- a/packages/host/apiproxy/README.zh.md +++ b/packages/host/apiproxy/README.zh.md @@ -20,7 +20,9 @@ Workspace 列表与 Session 列表是相互独立的重连基线。`workspace.cr `host.openPath` 会用操作系统的默认应用打开一个文件系统路径(macOS 为 `open`,Windows 为 `Invoke-Item`,Linux 为 `xdg-open`)。打开器可在测试中注入。浏览器载体对其施加与 `host.pickDirectory` 相同的回环、同源限制。 -`session.history` 按消息边界分页,其尾页(不带 `beforeSeq`)携带页窗口本身无法提供的会话级投影:进行中局部消息的分片事件;`todos`,即最后一次 `todo/write` 的整表投影;以及 `metrics`,即按 `(turn, step)` 去重的完整日志用量,并在可用时包含当前 token 计量压力和所选精确路由的容量。较早的页面省略会话级投影。实时 `session/metrics` mux 帧携带单调递增的日志修订号与投影修订号,因此客户端会拒绝陈旧帧,并在向前加载较早页面时保留计数器。缓存读取与缓存写入保持为彼此独立的计数项;缓存命中率的分母是未缓存输入加缓存读取。 +`session.history` 按消息边界分页,其尾页(不带 `beforeSeq`)携带页窗口本身无法提供的会话级投影:进行中局部消息的分片事件;`todos`,即最后一次 `todo/write` 的整表投影;以及 `metrics`,即按 `(turn, step)` 去重的完整日志用量与当前 token 计量压力。较早的页面省略会话级投影。实时 `session/metrics` mux 帧携带单调递增的日志修订号与投影修订号,因此客户端会拒绝陈旧帧,并在向前加载较早页面时保留计数器。缓存读取与缓存写入保持为彼此独立的计数项;缓存命中率的分母是未缓存输入加缓存读取。 + +上下文容量使用独立的临时 `session/model-request` mux 帧;实际请求到达分派点后,该帧由失败会被收容的 Agent 通知发出。该帧携带轮次、步骤、最终提供方/模型与可选容量,且只发送给当时已经打开的 mux 连接。`session.history`、mux 订阅基线、重连和会话恢复绝不会查询或回放先前的容量;不带容量的帧会显式清除较早的连接本地值。 `command.*` 与 `skill.*` 领域向客户端暴露宿主命令注册表和技能目录。每个方法都通过 `sessionId` 寻址一个会话的 Agent(被服务的会话必有 Agent;`command.*` 经由与 `session.*` 相同的路径恢复冷会话,而 `skill.list` 从会话头解析项目根目录,不触碰 Agent 注册表)。`command.execute` 在宿主侧运行一条斜杠命令行并返回脱耦结果;载体的请求信号可取消正在运行的处理器。`host/commands-changed` 是目录失效帧:客户端重新拉取 `command.list` 而不是做差分。 diff --git a/packages/host/apiproxy/src/api-proxy.ts b/packages/host/apiproxy/src/api-proxy.ts index 440c3a1694..a9a753db66 100644 --- a/packages/host/apiproxy/src/api-proxy.ts +++ b/packages/host/apiproxy/src/api-proxy.ts @@ -6,7 +6,6 @@ import { randomUUID } from 'node:crypto' import { mkdir, stat } from 'node:fs/promises' import { join } from 'node:path' -import { FiberState } from 'cordis' import type { Context } from 'cordis' import { installAgentLlmTarget } from '@deepseek-ai/dsh-agent' import type { @@ -409,26 +408,6 @@ export function createApiProxy(ctx: Context, defaults: ApiProxyDefaults): ApiPro return target } - /** - * Read the best capacity route without taking ownership of foreign routing. - * Web agents expose their live selection; other agents expose only a route - * that already crossed the durable request-header boundary. - */ - function metricsRouteFor(agent: Agent): Pick | undefined { - const installed = targets.get(agent) - if (installed !== undefined) return installed.current - const logged = agent.session.requestHeader()?.config - return logged === undefined - ? undefined - : { provider: logged.provider, model: logged.model } - } - - /** Pair a registry agent only with the exact Session lifecycle it owns. */ - function metricsAgentFor(session: Session): Agent | undefined { - const agent = ctx.get('agents')?.get(session.id) - return agent?.session === session ? agent : undefined - } - /** Pre-publication setup used by both fresh and resumed Web agents. */ function installTarget(agentCtx: Context): void { const agent = agentCtx.agent @@ -444,39 +423,23 @@ export function createApiProxy(ctx: Context, defaults: ApiProxyDefaults): ApiPro const pendingMetricSessions = new Set() let metricFlushScheduled = false - let metricsDisposed = false - const metricsProjector = new SessionMetricsProjector( - ctx, - metricsRouteFor, - (agent) => { - if (metricsDisposed) return - const agents = ctx.get('agents') - if (agents?.get(agent.id) !== agent) return - const sessions = ctx.get('sessions') - if (sessions?.get(agent.id) !== agent.session) return - scheduleMetrics(agent.session) - }, - ) + const metricsProjector = new SessionMetricsProjector(ctx) /** Queue one full-log metrics publication after synchronous session listeners drain. */ function scheduleMetrics(session: Session): void { - if (metricsDisposed || muxQueues.size === 0) return + if (muxQueues.size === 0) return pendingMetricSessions.add(session) if (metricFlushScheduled) return metricFlushScheduled = true queueMicrotask(() => { metricFlushScheduled = false - if (metricsDisposed) { - pendingMetricSessions.clear() - return - } const sessions = [...pendingMetricSessions] pendingMetricSessions.clear() for (const current of sessions) { broadcast({ type: 'session/metrics', sessionId: current.id, - metrics: metricsProjector.snapshot(current, metricsAgentFor(current)), + metrics: metricsProjector.snapshot(current), }) } }) @@ -488,25 +451,22 @@ export function createApiProxy(ctx: Context, defaults: ApiProxyDefaults): ApiPro if (affectsSessionMetrics(event)) scheduleMetrics(session) }), ctx.on('agent/created', (agent: Agent) => { scheduleMetrics(agent.session) }), + ctx.on('agent/model-request', (agent, turn, step, request) => { + broadcast({ + type: 'session/model-request', + sessionId: agent.session.id, + turn, + step, + provider: request.provider, + model: request.model, + ...request.contextWindow === undefined + ? {} + : { contextWindow: request.contextWindow }, + }) + }), ctx.on('session/disposed', (session: Session) => { pendingMetricSessions.delete(session) }), - ctx.on('internal/status', (fiber) => { - if (metricsDisposed) return - if (fiber.state === FiberState.UNLOADING) { - metricsProjector.invalidateCapacities() - return - } - if (fiber.state !== FiberState.ACTIVE - && fiber.state !== FiberState.FAILED - && fiber.state !== FiberState.DISPOSED) return - metricsProjector.invalidateCapacities() - const sessions = ctx.get('sessions') - if (sessions === undefined) return - for (const session of sessions.list()) scheduleMetrics(session) - }, { global: true }), ] return () => { - metricsDisposed = true - metricsProjector.dispose() pendingMetricSessions.clear() for (const dispose of disposers) dispose() } @@ -819,7 +779,7 @@ export function createApiProxy(ctx: Context, defaults: ApiProxyDefaults): ApiPro // client cannot reconstruct session-level state from it). const todos = beforeSeq === undefined ? backscanTodos(found.agent.session.events) : undefined const metrics = beforeSeq === undefined - ? metricsProjector.snapshot(found.agent.session, found.agent) + ? metricsProjector.snapshot(found.agent.session) : undefined return ok(request, { events: entries, @@ -920,11 +880,6 @@ export function createApiProxy(ctx: Context, defaults: ApiProxyDefaults): ApiPro : { reasoningEffort: resolved.reasoningEffort }, } targetFor(found.agent).current = selected - broadcast({ - type: 'session/metrics', - sessionId: found.agent.session.id, - metrics: metricsProjector.snapshot(found.agent.session, found.agent), - }) return ok(request, { selected: { ...selected } }) } catch (error: unknown) { return err(request, { @@ -1235,7 +1190,7 @@ export function createApiProxy(ctx: Context, defaults: ApiProxyDefaults): ApiPro queue.push(frame({ type: 'session/metrics', sessionId: session.id, - metrics: metricsProjector.snapshot(session, metricsAgentFor(session)), + metrics: metricsProjector.snapshot(session), })) } for (const pending of pendingQuestions.values()) { @@ -1292,7 +1247,7 @@ export function createApiProxy(ctx: Context, defaults: ApiProxyDefaults): ApiPro queue.push(frame({ type: 'session/metrics', sessionId: session.id, - metrics: metricsProjector.snapshot(session, metricsAgentFor(session)), + metrics: metricsProjector.snapshot(session), })) }), ctx.on('session/disposed', (session: Session) => { diff --git a/packages/host/apiproxy/src/api/events.schema.ts b/packages/host/apiproxy/src/api/events.schema.ts index 0f317128dd..b69392165c 100644 --- a/packages/host/apiproxy/src/api/events.schema.ts +++ b/packages/host/apiproxy/src/api/events.schema.ts @@ -30,6 +30,15 @@ export const muxFrameSchema = z.discriminatedUnion('type', [ z.object({ type: z.literal('session/event'), sessionId: sessionIdSchema, event: sessionEventSchema, view: toolEventViewSchema.optional() }), z.object({ type: z.literal('session/subscribed'), sessionId: sessionIdSchema, lastSeq: z.number().int() }), z.object({ type: z.literal('session/metrics'), sessionId: sessionIdSchema, metrics: sessionMetricsSchema }), + z.object({ + type: z.literal('session/model-request'), + sessionId: sessionIdSchema, + turn: z.number().int().positive(), + step: z.number().int().positive(), + provider: z.string().min(1), + model: z.string().min(1), + contextWindow: z.number().int().positive().optional(), + }), z.object({ type: z.literal('session/title'), sessionId: sessionIdSchema, title: z.string().min(1), eventSeq: z.number().int().nonnegative(), updatedAt: z.number() }), z.object({ type: z.literal('approval/requested'), sessionId: sessionIdSchema, approvalId: approvalRequestIdSchema, toolName: z.string(), callId: z.string().optional(), reason: z.string().optional() }), z.object({ type: z.literal('approval/resolved'), sessionId: sessionIdSchema, approvalId: approvalRequestIdSchema, outcome: z.union([z.literal('allowed-once'), z.literal('rejected'), z.literal('cancelled'), z.literal('unavailable')]) }), diff --git a/packages/host/apiproxy/src/api/events.ts b/packages/host/apiproxy/src/api/events.ts index 73e2b42deb..c4babb6017 100644 --- a/packages/host/apiproxy/src/api/events.ts +++ b/packages/host/apiproxy/src/api/events.ts @@ -59,6 +59,22 @@ export type MuxFrame = | { type: 'session/event'; sessionId: SessionId; event: SessionEvent; view?: ToolEventView } | { type: 'session/subscribed'; sessionId: SessionId; lastSeq: number } | { type: 'session/metrics'; sessionId: SessionId; metrics: SessionMetrics } + /** + * One model request observed by this already-open mux connection after its + * final route and stream handle were resolved. This frame is transient: mux + * baselines, reconnects, and session history never replay it. An absent + * `contextWindow` explicitly clears a capacity observed from an earlier + * request on the same connection. + */ + | { + type: 'session/model-request' + sessionId: SessionId + turn: number + step: number + provider: string + model: string + contextWindow?: number + } | { type: 'session/title'; sessionId: SessionId; title: string; eventSeq: number; updatedAt: number } | { type: 'approval/requested'; sessionId: SessionId; approvalId: ApprovalRequestId; toolName: string; callId?: CallId; reason?: string } | { type: 'approval/resolved'; sessionId: SessionId; approvalId: ApprovalRequestId; outcome: ApprovalOutcome } diff --git a/packages/host/apiproxy/src/api/sessions.schema.ts b/packages/host/apiproxy/src/api/sessions.schema.ts index 0866766da5..544562f2e4 100644 --- a/packages/host/apiproxy/src/api/sessions.schema.ts +++ b/packages/host/apiproxy/src/api/sessions.schema.ts @@ -145,7 +145,7 @@ export const todoItemSchema = z.object({ status: z.union([z.literal('pending'), z.literal('in_progress'), z.literal('completed')]), }) -/** Host-owned durable usage and current-context projection. */ +/** Host-owned durable usage and current-pressure projection. */ export const sessionMetricsSchema = z.object({ logRevision: z.number().int().nonnegative(), projectionRevision: z.number().int().nonnegative(), @@ -154,7 +154,6 @@ export const sessionMetricsSchema = z.object({ cacheReadTokens: z.number().nonnegative(), cacheWriteTokens: z.number().nonnegative(), contextTokens: z.number().nonnegative().optional(), - contextWindow: z.number().int().positive().optional(), }) satisfies z.ZodType> /** session.history response value. */ diff --git a/packages/host/apiproxy/src/api/sessions.ts b/packages/host/apiproxy/src/api/sessions.ts index 9a0344867e..c8c10757fd 100644 --- a/packages/host/apiproxy/src/api/sessions.ts +++ b/packages/host/apiproxy/src/api/sessions.ts @@ -34,9 +34,9 @@ export interface HistoryEntry { /** * Host-owned token metrics for one durable session revision. Provider usage - * buckets are cumulative across the full log; current context fields describe - * the replayed request surface at this revision and are absent when the Host - * cannot measure pressure or resolve exact-route capacity. + * buckets are cumulative across the full log; current context pressure + * describes the replayed request surface at this revision and is absent when + * the Host cannot measure it. */ export interface SessionMetrics { /** Number of durable events included in this projection. */ @@ -53,8 +53,6 @@ export interface SessionMetrics { cacheWriteTokens: number /** Current request pressure from `ctx.tokenMeter.measure(session).totalTokens`. */ contextTokens?: number - /** Exact selected-route capacity from `ctx.llm.resolveModelInfo()`. */ - contextWindow?: number } /** Complete model target selected for one session. */ @@ -178,8 +176,10 @@ export interface SessionsApi { * projection (latest `todo/write` over the FULL log, independent of the page window) — * so a paged client restores the plan without walking history; absent when the session * never wrote one. Older pages omit it (the projection is session-level, not per-page). - * The same tail-only rule carries `metrics`, whose cumulative usage and current context - * are Host projections over the full log rather than products of the returned page. + * The same tail-only rule carries `metrics`, whose cumulative usage and + * current pressure are Host projections over the full log rather than + * products of the returned page. Live model capacity is connection-local + * telemetry and is never reconstructed here. */ history(request: RpcRequest<{ sessionId: SessionId; beforeSeq?: number; maxMessages?: number }>): Promise> diff --git a/packages/host/apiproxy/src/session-metrics.ts b/packages/host/apiproxy/src/session-metrics.ts index 1eebc7b7d8..a99535f6bc 100644 --- a/packages/host/apiproxy/src/session-metrics.ts +++ b/packages/host/apiproxy/src/session-metrics.ts @@ -5,7 +5,6 @@ */ import type { Context } from 'cordis' -import type { Agent, AgentLlmTarget } from '@deepseek-ai/dsh-agent' import type { TokenUsage } from '@deepseek-ai/dsh-llm' import type { Session, SessionEvent } from '@deepseek-ai/dsh-session' import type { SessionMetrics } from './api/sessions.ts' @@ -20,27 +19,10 @@ interface UsageState { byStep: Map } -interface CapacityState { - routeKey: string | undefined - generation: number - epoch: number - status: 'pending' | 'ready' | 'retryable' - contextWindow?: number - controller?: AbortController -} - -type CapacityTarget = Pick - interface TokenMeterLike { measure(session: Session): { totalTokens: number } } -interface LlmLike { - resolveModelInfo(provider: string, model: string, signal?: AbortSignal): Promise<{ - context?: { contextWindow: number } - }> -} - function usageFrom(event: SessionEvent): { turn: number; step: number; usage: TokenUsage } | undefined { if (event.type === 'assistant/chunk' && event.data.chunk.type === 'usage') { return { turn: event.data.turn, step: event.data.step, usage: event.data.chunk.usage } @@ -79,51 +61,19 @@ function recordUsage(state: UsageState, turn: number, step: number, usage: Token state.cacheWriteTokens += usage.cacheWriteTokens ?? 0 } -function routeKeyFor(target: CapacityTarget | undefined): string | undefined { - return target === undefined ? undefined : `${target.provider}\u0000${target.model}` -} - -/** - * Projects durable cumulative usage and route-aware current context without - * awaiting model metadata on the session append path. - */ +/** Projects durable cumulative usage and synchronous current context pressure. */ export class SessionMetricsProjector { private readonly usage = new WeakMap() - private readonly capacities = new WeakMap() - private readonly pendingCapacities = new Set() - private capacityEpoch = 0 - private disposed = false - /** - * @param ctx - Host context providing optional token-meter and LLM services. - * @param targetFor - side-effect-free selected or logged route lookup for one attached agent. - * @param onCapacityResolved - schedules a fresh live projection after exact-route metadata resolves. - */ - constructor( - private readonly ctx: Context, - private readonly targetFor: (agent: Agent) => CapacityTarget | undefined, - private readonly onCapacityResolved: (agent: Agent) => void, - ) {} - - /** Retire adapter-owned metadata and fence every resolution already in flight. */ - invalidateCapacities(): void { - this.capacityEpoch++ - for (const pending of this.pendingCapacities) this.abortCapacityResolution(pending) - } - - /** Permanently retire capacity projection and cancel every adapter-owned lookup. */ - dispose(): void { - this.disposed = true - this.invalidateCapacities() - } + /** @param ctx - Host context providing an optional token-meter service. */ + constructor(private readonly ctx: Context) {} /** * Read a fresh detached projection through the session's durable tail. * @param session - authoritative durable log owner. - * @param agent - attached route owner, when available. - * @returns cumulative usage and any currently available pressure/capacity. + * @returns cumulative usage and any currently measurable pressure. */ - snapshot(session: Session, agent?: Agent): SessionMetrics { + snapshot(session: Session): SessionMetrics { const state = this.syncUsage(session) const tokenMeter = this.ctx.get('tokenMeter') as TokenMeterLike | undefined let contextTokens: number | undefined @@ -134,7 +84,6 @@ export class SessionMetricsProjector { // A malformed or temporarily unmeasurable replay has no honest pressure value. } } - const contextWindow = agent === undefined ? undefined : this.capacityFor(agent) return { logRevision: state.logRevision, projectionRevision: state.projectionRevision++, @@ -143,7 +92,6 @@ export class SessionMetricsProjector { cacheReadTokens: state.cacheReadTokens, cacheWriteTokens: state.cacheWriteTokens, ...contextTokens === undefined ? {} : { contextTokens }, - ...contextWindow === undefined ? {} : { contextWindow }, } } @@ -171,87 +119,4 @@ export class SessionMetricsProjector { } return state } - - private capacityFor(agent: Agent): number | undefined { - if (this.disposed) return undefined - const target = this.targetFor(agent) - const routeKey = routeKeyFor(target) - let state = this.capacities.get(agent) - if (state === undefined - || state.routeKey !== routeKey - || state.epoch !== this.capacityEpoch - || state.status === 'retryable') { - this.abortCapacityResolution(state) - state = { - routeKey, - generation: (state?.generation ?? 0) + 1, - epoch: this.capacityEpoch, - status: target === undefined ? 'ready' : 'pending', - } - this.capacities.set(agent, state) - if (target !== undefined) this.resolveCapacity(agent, target, state) - } - return state.status === 'ready' ? state.contextWindow : undefined - } - - private resolveCapacity( - agent: Agent, - target: CapacityTarget, - pending: CapacityState, - ): void { - const llm = this.ctx.get('llm') as LlmLike | undefined - if (llm === undefined) { - pending.status = 'retryable' - return - } - const controller = new AbortController() - pending.controller = controller - this.pendingCapacities.add(pending) - void Promise.resolve() - .then(() => { - controller.signal.throwIfAborted() - return llm.resolveModelInfo(target.provider, target.model, controller.signal) - }) - .then( - (resolved) => { - this.finishCapacityResolution(pending, controller) - if (this.capacityResolutionIsStale(agent, pending)) return - pending.status = 'ready' - if (resolved.context !== undefined) pending.contextWindow = resolved.context.contextWindow - this.onCapacityResolved(agent) - }, - () => { - this.finishCapacityResolution(pending, controller) - if (!this.capacityResolutionIsStale(agent, pending)) pending.status = 'retryable' - }, - ) - } - - private abortCapacityResolution(pending: CapacityState | undefined): void { - if (pending === undefined || pending.controller === undefined) return - const controller = pending.controller - delete pending.controller - this.pendingCapacities.delete(pending) - controller.abort() - } - - private finishCapacityResolution(pending: CapacityState, controller: AbortController): void { - this.pendingCapacities.delete(pending) - if (pending.controller === controller) delete pending.controller - } - - private capacityResolutionIsStale(agent: Agent, pending: CapacityState): boolean { - if (pending.epoch !== this.capacityEpoch) return true - if (this.capacities.get(agent)?.generation !== pending.generation) return true - if (routeKeyFor(this.targetFor(agent)) === pending.routeKey) return false - // Unknown is the neutral generation; the next observed concrete route - // starts a fresh resolution even when it equals the route that disappeared. - this.capacities.set(agent, { - routeKey: undefined, - generation: pending.generation + 1, - epoch: this.capacityEpoch, - status: 'ready', - }) - return true - } } diff --git a/packages/host/apiproxy/tests/api-proxy-model-request.spec.ts b/packages/host/apiproxy/tests/api-proxy-model-request.spec.ts new file mode 100644 index 0000000000..2194afe131 --- /dev/null +++ b/packages/host/apiproxy/tests/api-proxy-model-request.spec.ts @@ -0,0 +1,104 @@ +import { describe, expect, it } from 'vitest' +import { Context } from 'cordis' +import AgentRegistry, { agentEvents } from '@deepseek-ai/dsh-agent' +import type { Agent } from '@deepseek-ai/dsh-agent' +import SessionStore, { SessionId } from '@deepseek-ai/dsh-session' +import UserInteractionService from '@deepseek-ai/dsh-user-interaction' +import type { MuxFrame, RpcRequest } from '@deepseek-ai/dsh-host-apiproxy/api' +import { RpcId } from '@deepseek-ai/dsh-host-apiproxy/api' +import { createApiProxy } from '../src/api-proxy.ts' + +async function nextFrame( + iterator: AsyncIterator>, + type: K, +): Promise> { + for (;;) { + const next = await iterator.next() + if (next.done) throw new Error(`mux ended before ${type}`) + if (next.value.payload.type === type) { + return next.value.payload as Extract + } + } +} + +describe('ApiProxy model-request telemetry', () => { + it('forwards only to open mux connections and never backfills history or reconnect baselines', async () => { + const ctx = new Context() + await ctx.plugin(SessionStore) + await ctx.plugin(UserInteractionService) + await ctx.plugin(AgentRegistry) + const session = ctx.sessions.create(SessionId('model-request-telemetry')) + const agent = { + id: session.id, + session, + status: 'running', + ctx, + } as Agent + ctx.agents.register(agent) + const api = createApiProxy(ctx, { + provider: 'test', + model: 'alpha', + cwd: '/tmp', + workspaceRoot: '/tmp', + }) + + const primaryAbort = new AbortController() + const primary = api.events.mux( + { rpcId: RpcId('primary'), payload: {} }, + primaryAbort.signal, + )[Symbol.asyncIterator]() + expect((await nextFrame(primary, 'session/subscribed')).sessionId).toBe(session.id) + expect((await nextFrame(primary, 'session/metrics')).metrics).not.toHaveProperty('contextWindow') + + agentEvents(ctx, agent).emit('agent/model-request', 1, 2, { + provider: 'test', + model: 'alpha', + contextWindow: 128_000, + }) + expect(await nextFrame(primary, 'session/model-request')).toEqual({ + type: 'session/model-request', + sessionId: session.id, + turn: 1, + step: 2, + provider: 'test', + model: 'alpha', + contextWindow: 128_000, + }) + + const history = await api.sessions.history({ + rpcId: RpcId('history'), + payload: { sessionId: session.id }, + }) + if (!history.result.ok) throw new Error('history failed') + expect(history.result.value.metrics).not.toHaveProperty('contextWindow') + + const reconnectAbort = new AbortController() + const reconnect = api.events.mux( + { rpcId: RpcId('reconnect'), payload: {} }, + reconnectAbort.signal, + )[Symbol.asyncIterator]() + expect((await nextFrame(reconnect, 'session/subscribed')).sessionId).toBe(session.id) + expect((await nextFrame(reconnect, 'session/metrics')).metrics).not.toHaveProperty('contextWindow') + + agentEvents(ctx, agent).emit('agent/model-request', 2, 1, { + provider: 'test', + model: 'without-capacity', + }) + for (const iterator of [primary, reconnect]) { + expect(await nextFrame(iterator, 'session/model-request')).toEqual({ + type: 'session/model-request', + sessionId: session.id, + turn: 2, + step: 1, + provider: 'test', + model: 'without-capacity', + }) + } + + primaryAbort.abort() + reconnectAbort.abort() + await primary.return?.() + await reconnect.return?.() + await ctx.fiber.dispose() + }) +}) diff --git a/packages/host/apiproxy/tests/api-proxy-models.spec.ts b/packages/host/apiproxy/tests/api-proxy-models.spec.ts index 7b2595e528..9d3e6ef34c 100644 --- a/packages/host/apiproxy/tests/api-proxy-models.spec.ts +++ b/packages/host/apiproxy/tests/api-proxy-models.spec.ts @@ -4,23 +4,20 @@ * models, and the prompt-assembly boundary for a running selection change. */ -import { describe, expect, it, vi } from 'vitest' -import { Context, FiberState } from 'cordis' -import type { Fiber } from 'cordis' -import AgentRegistry, { agentEvents, installAgentLlmTarget } from '@deepseek-ai/dsh-agent' -import type { Agent, AgentLlmTargetRef } from '@deepseek-ai/dsh-agent' +import { describe, expect, it } from 'vitest' +import { Context } from 'cordis' +import AgentRegistry, { agentEvents } from '@deepseek-ai/dsh-agent' +import type { Agent } from '@deepseek-ai/dsh-agent' import LlmService, { LlmAdapter, ReasoningEffortId } from '@deepseek-ai/dsh-llm' import type { GenerateOptions, LlmCallConfig, LlmModelInfo, LlmModelReasoningInfo, LlmProviderInfo, LlmResolvedModelInfo, StreamChunk, } from '@deepseek-ai/dsh-llm' import SessionStore, { SessionId } from '@deepseek-ai/dsh-session' -import type { Session } from '@deepseek-ai/dsh-session' import SystemPrompt from '@deepseek-ai/dsh-system-prompt' import UserInteractionService from '@deepseek-ai/dsh-user-interaction' import type { RpcRequest } from '@deepseek-ai/dsh-host-apiproxy/api/rpc' import { RpcId } from '@deepseek-ai/dsh-host-apiproxy/api/rpc' -import type { MuxFrame } from '@deepseek-ai/dsh-host-apiproxy/api' import { createApiProxy } from '../src/api-proxy.ts' let nextRpc = 1 @@ -64,40 +61,6 @@ class CatalogAdapter extends LlmAdapter { } } -class DeferredCatalogAdapter extends CatalogAdapter { - readonly pending: { - result: PromiseWithResolvers - signal: AbortSignal | undefined - }[] = [] - - constructor() { - super('Deferred', [ - { provider: 'deferred', id: 'lifecycle-model', name: 'Lifecycle model' }, - ]) - } - - override resolveModel( - _provider: string, - _model: string, - signal?: AbortSignal, - ): Promise { - const result = Promise.withResolvers() - this.pending.push({ result, signal }) - return result.promise - } - - resolve(index: number, contextWindow: number): void { - const pending = this.pending[index] - if (pending === undefined) throw new Error(`no pending resolution at index ${String(index)}`) - pending.result.resolve({ - provider: 'deferred', - id: 'lifecycle-model', - name: 'Lifecycle model', - context: { contextWindow }, - }) - } -} - const REASONING: LlmModelReasoningInfo = { efforts: [ { id: ReasoningEffortId('off'), name: 'Off' }, @@ -107,18 +70,13 @@ const REASONING: LlmModelReasoningInfo = { defaultEffort: ReasoningEffortId('high'), } -async function hostContext( - onSessions?: (fiber: Fiber) => void, - onAgents?: (fiber: Fiber) => void, -): Promise { +async function hostContext(): Promise { const ctx = new Context() - const sessionsFiber = await ctx.plugin(SessionStore) - onSessions?.(sessionsFiber) + await ctx.plugin(SessionStore) await ctx.plugin(SystemPrompt, { persona: '' }) await ctx.plugin(LlmService) await ctx.plugin(UserInteractionService) - const agentsFiber = await ctx.plugin(AgentRegistry) - onAgents?.(agentsFiber) + await ctx.plugin(AgentRegistry) ctx.llm.registerAdapter(['deepseek'], new CatalogAdapter('DeepSeek', [ { provider: 'deepseek', id: 'deepseek-chat', name: 'DeepSeek Chat' }, { provider: 'deepseek', id: 'deepseek-reasoner', name: 'DeepSeek Reasoner', description: 'Reasoning model' }, @@ -164,65 +122,6 @@ function expectValue(response: { result: { ok: true; value: T } | { ok: false return response.result.value } -async function nextMetrics( - iterator: AsyncIterator>, -): Promise['metrics']> { - for (;;) { - const next = await iterator.next() - if (next.done) throw new Error('mux ended before a metrics frame') - if (next.value.payload.type === 'session/metrics') return next.value.payload.metrics - } -} - -function attachLifecycleSession( - ctx: Context, - sessionId: SessionId, - withMarker = false, -): { session: Session; detach: () => void } { - const session = ctx.sessions.prepare(sessionId) - session.append('request/header', { - header: { config: { provider: 'deferred', model: 'lifecycle-model' } }, - reason: 'initial', - }) - if (withMarker) { - session.append('user/message', { - content: [{ type: 'text', text: 'replacement marker' }], - source: { kind: 'plugin', plugin: 'test' }, - }, { surfaceOp: 'append' }) - } - const detach = ctx.sessions.enter(session) - ctx.sessions.announce(session) - return { session, detach } -} - -function attachLifecycleAgent( - ctx: Context, - session: Session, -): () => void { - const agent = { - id: session.id, - session, - status: 'running', - ctx, - } as Agent - const detach = ctx.agents.enter(agent, undefined) - ctx.agents.announce(agent) - return detach -} - -function settleCapacityCompletion(): Promise { - return new Promise((resolve) => { setImmediate(resolve) }) -} - -function installDeferredAdapter( - ctx: Context, - adapter: DeferredCatalogAdapter, -): Fiber & PromiseLike { - return ctx.plugin(Object.assign((inner: Context) => { - inner.llm.registerAdapter(['deferred'], adapter) - }, { inject: ['llm'] })) -} - describe('Web session model selection', () => { it('groups successful providers, isolates failures, and preserves an unlisted current model', async () => { const { ctx, sessionId } = await harness({ @@ -230,7 +129,12 @@ describe('Web session model selection', () => { model: 'private-preview', reasoningEffort: ReasoningEffortId('max'), }) - const api = createApiProxy(ctx, { provider: 'deepseek', model: 'deepseek-chat', cwd: '/tmp', workspaceRoot: '/tmp' }) + const api = createApiProxy(ctx, { + provider: 'deepseek', + model: 'deepseek-chat', + cwd: '/tmp', + workspaceRoot: '/tmp', + }) const catalog = expectValue(await api.sessions.models(request({ sessionId }))) expect(catalog.current).toEqual({ @@ -271,7 +175,12 @@ describe('Web session model selection', () => { it('accepts an advisory-unlisted model, rejects an unavailable provider, and switches only after the next assembly', async () => { const { ctx, agent, sessionId } = await harness() - const api = createApiProxy(ctx, { provider: 'deepseek', model: 'deepseek-chat', cwd: '/tmp', workspaceRoot: '/tmp' }) + const api = createApiProxy(ctx, { + provider: 'deepseek', + model: 'deepseek-chat', + cwd: '/tmp', + workspaceRoot: '/tmp', + }) const seed: LlmCallConfig = { provider: 'seed', model: 'seed', temperature: 0.2 } const signal = new AbortController().signal @@ -336,534 +245,4 @@ describe('Web session model selection', () => { .toEqual({ provider: 'deepseek', model: 'private-preview', reasoningEffort: 'max' }) await ctx.fiber.dispose() }) - - it('publishes unknown capacity immediately on selection, then the exact selected route capacity', async () => { - const { ctx, sessionId } = await harness() - const api = createApiProxy(ctx, { provider: 'deepseek', model: 'deepseek-chat', cwd: '/tmp', workspaceRoot: '/tmp' }) - expectValue(await api.sessions.models(request({ sessionId }))) - const controller = new AbortController() - const iterator = api.events.mux(request({}), controller.signal)[Symbol.asyncIterator]() - - expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() - expect((await nextMetrics(iterator)).contextWindow).toBe(64_000) - - expectValue(await api.sessions.selectModel(request({ - sessionId, - provider: 'deepseek', - model: 'private-preview', - }))) - expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() - expect((await nextMetrics(iterator)).contextWindow).toBe(128_000) - - controller.abort() - await iterator.return?.() - await ctx.fiber.dispose() - }) - - it('uses logged capacity without installing Web routing while scheduling foreign metrics', async () => { - const ctx = await hostContext() - const api = createApiProxy(ctx, { - provider: 'deepseek', - model: 'deepseek-chat', - cwd: '/tmp', - workspaceRoot: '/tmp', - }) - const controller = new AbortController() - const iterator = api.events.mux(request({}), controller.signal)[Symbol.asyncIterator]() - const initialMetrics = nextMetrics(iterator) - const session = ctx.sessions.create() - expect((await initialMetrics).contextWindow).toBeUndefined() - session.append('request/header', { - header: { config: { provider: 'deepseek', model: 'private-preview' } }, - reason: 'change', - }) - const foreign = { - id: session.id, - session, - status: 'running', - ctx, - } as Agent - const foreignTarget: AgentLlmTargetRef = { - current: { provider: 'foreign', model: 'foreign-model' }, - assembled: undefined, - } - const disposeForeignTarget = installAgentLlmTarget(foreign.ctx, foreignTarget) - const scheduledMetrics = nextMetrics(iterator) - ctx.agents.register(foreign) - - expect((await scheduledMetrics).contextWindow).toBeUndefined() - expect((await nextMetrics(iterator)).contextWindow).toBe(128_000) - expect((await ctx.systemPrompt.assemble()).variables) - .toMatchObject({ provider: 'foreign', model: 'foreign-model' }) - const seed: LlmCallConfig = { provider: 'seed', model: 'seed', temperature: 0.2 } - const signal = new AbortController().signal - await expect(agentEvents(ctx, foreign).waterfall( - 'agent/request', 1, 0, signal, () => Promise.resolve(seed), - )).resolves.toMatchObject({ provider: 'foreign', model: 'foreign-model' }) - - disposeForeignTarget() - expect((await ctx.systemPrompt.assemble()).variables).not.toHaveProperty('provider') - await expect(agentEvents(ctx, foreign).waterfall( - 'agent/request', 1, 1, signal, () => Promise.resolve(seed), - )).resolves.toBe(seed) - - controller.abort() - await iterator.return?.() - await ctx.fiber.dispose() - }) - - it('drops capacity completion from a replaced agent that retains the exact session', async () => { - const ctx = await hostContext() - const deferred = new DeferredCatalogAdapter() - ctx.llm.registerAdapter(['deferred'], deferred) - const lifecycle = attachLifecycleSession(ctx, SessionId('capacity-agent-lifecycle')) - const retire = attachLifecycleAgent(ctx, lifecycle.session) - const api = createApiProxy(ctx, { provider: 'deepseek', model: 'deepseek-chat', cwd: '/tmp', workspaceRoot: '/tmp' }) - const controller = new AbortController() - const iterator = api.events.mux(request({}), controller.signal)[Symbol.asyncIterator]() - - expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() - await vi.waitFor(() => { expect(deferred.pending).toHaveLength(1) }) - retire() - const detachLive = attachLifecycleAgent(ctx, lifecycle.session) - expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() - await vi.waitFor(() => { expect(deferred.pending).toHaveLength(2) }) - - deferred.resolve(0, 64_000) - await settleCapacityCompletion() - deferred.resolve(1, 128_000) - await settleCapacityCompletion() - expect((await nextMetrics(iterator)).contextWindow).toBe(128_000) - - controller.abort() - await iterator.return?.() - detachLive() - lifecycle.detach() - await ctx.fiber.dispose() - }) - - it('drops capacity completion from a replaced session while its old agent remains live', async () => { - const ctx = await hostContext() - const deferred = new DeferredCatalogAdapter() - ctx.llm.registerAdapter(['deferred'], deferred) - const sessionId = SessionId('capacity-session-lifecycle') - const retiredSession = attachLifecycleSession(ctx, sessionId) - const retireAgent = attachLifecycleAgent(ctx, retiredSession.session) - const api = createApiProxy(ctx, { provider: 'deepseek', model: 'deepseek-chat', cwd: '/tmp', workspaceRoot: '/tmp' }) - const controller = new AbortController() - const iterator = api.events.mux(request({}), controller.signal)[Symbol.asyncIterator]() - - expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() - await vi.waitFor(() => { expect(deferred.pending).toHaveLength(1) }) - retiredSession.detach() - const liveSession = attachLifecycleSession(ctx, sessionId, true) - expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() - deferred.resolve(0, 64_000) - await settleCapacityCompletion() - retireAgent() - const detachLiveAgent = attachLifecycleAgent(ctx, liveSession.session) - const scheduled = await nextMetrics(iterator) - expect(scheduled.logRevision).toBe(2) - expect(scheduled.contextWindow).toBeUndefined() - await vi.waitFor(() => { expect(deferred.pending).toHaveLength(2) }) - deferred.resolve(1, 128_000) - await settleCapacityCompletion() - expect((await nextMetrics(iterator)).contextWindow).toBe(128_000) - - controller.abort() - await iterator.return?.() - detachLiveAgent() - liveSession.detach() - await ctx.fiber.dispose() - }) - - it('does not project retired agent capacity into replacement session snapshots', async () => { - const ctx = await hostContext() - const deferred = new DeferredCatalogAdapter() - ctx.llm.registerAdapter(['deferred'], deferred) - const sessionId = SessionId('capacity-snapshot-lifecycle') - const retiredSession = attachLifecycleSession(ctx, sessionId) - const retireAgent = attachLifecycleAgent(ctx, retiredSession.session) - const api = createApiProxy(ctx, { provider: 'deepseek', model: 'deepseek-chat', cwd: '/tmp', workspaceRoot: '/tmp' }) - const primaryController = new AbortController() - const primary = api.events.mux(request({}), primaryController.signal)[Symbol.asyncIterator]() - - expect((await nextMetrics(primary)).contextWindow).toBeUndefined() - await vi.waitFor(() => { expect(deferred.pending).toHaveLength(1) }) - deferred.resolve(0, 64_000) - await settleCapacityCompletion() - expect((await nextMetrics(primary)).contextWindow).toBe(64_000) - - retiredSession.detach() - const replacement = attachLifecycleSession(ctx, sessionId) - const createdBaseline = await nextMetrics(primary) - replacement.session.append('user/message', { - content: [{ type: 'text', text: 'replacement marker' }], - source: { kind: 'plugin', plugin: 'test' }, - }, { surfaceOp: 'append' }) - const scheduledFlush = await nextMetrics(primary) - const reconnectController = new AbortController() - const reconnect = api.events.mux(request({}), reconnectController.signal)[Symbol.asyncIterator]() - const reconnectBaseline = await nextMetrics(reconnect) - expect(createdBaseline.logRevision).toBe(1) - for (const metrics of [scheduledFlush, reconnectBaseline]) { - expect(metrics.logRevision).toBe(2) - } - expect({ - created: createdBaseline.contextWindow, - scheduled: scheduledFlush.contextWindow, - reconnect: reconnectBaseline.contextWindow, - }).toEqual({ created: undefined, scheduled: undefined, reconnect: undefined }) - - retireAgent() - const detachReplacementAgent = attachLifecycleAgent(ctx, replacement.session) - expect((await nextMetrics(primary)).contextWindow).toBeUndefined() - await vi.waitFor(() => { expect(deferred.pending).toHaveLength(2) }) - deferred.resolve(1, 128_000) - await settleCapacityCompletion() - expect((await nextMetrics(primary)).contextWindow).toBe(128_000) - - primaryController.abort() - reconnectController.abort() - await primary.return?.() - await reconnect.return?.() - detachReplacementAgent() - replacement.detach() - await ctx.fiber.dispose() - }) - - it('refreshes same-route capacity after adapter owner replacement', async () => { - const ctx = await hostContext() - const retiredAdapter = new DeferredCatalogAdapter() - const retiredFiber = await installDeferredAdapter(ctx, retiredAdapter) - const lifecycle = attachLifecycleSession(ctx, SessionId('capacity-adapter-lifecycle')) - const detachAgent = attachLifecycleAgent(ctx, lifecycle.session) - const api = createApiProxy(ctx, { provider: 'deepseek', model: 'deepseek-chat', cwd: '/tmp', workspaceRoot: '/tmp' }) - const controller = new AbortController() - const iterator = api.events.mux(request({}), controller.signal)[Symbol.asyncIterator]() - - expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() - await vi.waitFor(() => { expect(retiredAdapter.pending).toHaveLength(1) }) - retiredAdapter.resolve(0, 64_000) - await settleCapacityCompletion() - expect((await nextMetrics(iterator)).contextWindow).toBe(64_000) - - await retiredFiber.dispose() - expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() - const replacementAdapter = new DeferredCatalogAdapter() - const replacementFiber = await installDeferredAdapter(ctx, replacementAdapter) - expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() - await vi.waitFor(() => { expect(replacementAdapter.pending).toHaveLength(1) }) - replacementAdapter.resolve(0, 128_000) - await settleCapacityCompletion() - expect((await nextMetrics(iterator)).contextWindow).toBe(128_000) - expect(lifecycle.session.requestHeader()?.config).toMatchObject({ - provider: 'deferred', - model: 'lifecycle-model', - }) - - controller.abort() - await iterator.return?.() - detachAgent() - lifecycle.detach() - await replacementFiber.dispose() - await ctx.fiber.dispose() - }) - - it('aborts pending capacity during adapter UNLOADING and refreshes after settlement', async () => { - const ctx = await hostContext() - const retiredAdapter = new DeferredCatalogAdapter() - const releaseUnload = Promise.withResolvers() - const releaseCancellationWait = Promise.withResolvers() - const abortObserved = Promise.withResolvers() - const retiredFiber = await ctx.plugin(Object.assign((inner: Context) => { - inner.llm.registerAdapter(['deferred'], retiredAdapter) - inner.effect( - () => () => releaseUnload.promise, - 'test: hold adapter unload', - ) - inner.effect(() => () => { - const signal = retiredAdapter.pending[0]?.signal - if (signal === undefined) throw new Error('pending capacity signal missing') - const cancellation = new Promise((resolve) => { - const finish = () => { - abortObserved.resolve(undefined) - resolve() - } - if (signal.aborted) finish() - else signal.addEventListener('abort', finish, { once: true }) - }) - return Promise.race([cancellation, releaseCancellationWait.promise]) - }, 'test: await capacity cancellation') - }, { inject: ['llm'] })) - const lifecycle = attachLifecycleSession(ctx, SessionId('capacity-adapter-unloading')) - const detachAgent = attachLifecycleAgent(ctx, lifecycle.session) - const api = createApiProxy(ctx, { - provider: 'deepseek', - model: 'deepseek-chat', - cwd: '/tmp', - workspaceRoot: '/tmp', - }) - const controller = new AbortController() - const iterator = api.events.mux(request({}), controller.signal)[Symbol.asyncIterator]() - const resolveModelInfo = vi.spyOn(ctx.llm, 'resolveModelInfo') - - expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() - await vi.waitFor(() => { expect(retiredAdapter.pending).toHaveLength(1) }) - expect(resolveModelInfo).toHaveBeenCalledOnce() - const listSessions = vi.spyOn(ctx.sessions, 'list') - listSessions.mockClear() - const pendingFrame = iterator.next() - const disposing = retiredFiber.dispose() - try { - await vi.waitFor(() => { - expect(retiredAdapter.pending[0]?.signal?.aborted).toBe(true) - }) - await abortObserved.promise - expect(retiredFiber.state).toBe(FiberState.UNLOADING) - retiredAdapter.resolve(0, 64_000) - await settleCapacityCompletion() - expect(resolveModelInfo).toHaveBeenCalledOnce() - expect(listSessions).not.toHaveBeenCalled() - const outcome = await Promise.race([ - pendingFrame.then(() => 'frame' as const), - new Promise<'idle'>((resolve) => { setImmediate(() => { resolve('idle') }) }), - ]) - expect(outcome).toBe('idle') - } finally { - releaseCancellationWait.resolve(undefined) - releaseUnload.resolve(undefined) - await disposing - } - - expect(retiredFiber.state).toBe(FiberState.DISPOSED) - const settledFrame = await pendingFrame - if (settledFrame.done || settledFrame.value.payload.type !== 'session/metrics') { - throw new Error('expected settled metrics refresh') - } - expect(settledFrame.value.payload.metrics.contextWindow).toBeUndefined() - await settleCapacityCompletion() - expect(resolveModelInfo).toHaveBeenCalledTimes(2) - - const replacementAdapter = new DeferredCatalogAdapter() - const replacementFiber = await installDeferredAdapter(ctx, replacementAdapter) - expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() - await vi.waitFor(() => { expect(replacementAdapter.pending).toHaveLength(1) }) - expect(resolveModelInfo).toHaveBeenCalledTimes(3) - replacementAdapter.resolve(0, 128_000) - await settleCapacityCompletion() - expect((await nextMetrics(iterator)).contextWindow).toBe(128_000) - - controller.abort() - await iterator.return?.() - detachAgent() - lifecycle.detach() - await replacementFiber.dispose() - await ctx.fiber.dispose() - }) - - it('aborts pending capacity when the API proxy fiber is disposed', async () => { - const ctx = await hostContext() - const deferred = new DeferredCatalogAdapter() - ctx.llm.registerAdapter(['deferred'], deferred) - const lifecycle = attachLifecycleSession(ctx, SessionId('capacity-api-proxy-teardown')) - const detachAgent = attachLifecycleAgent(ctx, lifecycle.session) - const proxy = Promise.withResolvers>() - const proxyFiber = await ctx.plugin(Object.assign((inner: Context) => { - proxy.resolve(createApiProxy(inner, { - provider: 'deepseek', - model: 'deepseek-chat', - cwd: '/tmp', - workspaceRoot: '/tmp', - })) - }, { inject: ['agents', 'sessions', 'userInteraction'] })) - const api = await proxy.promise - const controller = new AbortController() - const iterator = api.events.mux(request({}), controller.signal)[Symbol.asyncIterator]() - - expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() - await vi.waitFor(() => { expect(deferred.pending).toHaveLength(1) }) - expect(deferred.pending[0]?.signal?.aborted).toBe(false) - - await proxyFiber.dispose() - expect(deferred.pending[0]?.signal?.aborted).toBe(true) - deferred.resolve(0, 64_000) - await settleCapacityCompletion() - const pendingFrame = iterator.next() - const outcome = await Promise.race([ - pendingFrame.then(() => 'frame' as const), - new Promise<'idle'>((resolve) => { setImmediate(() => { resolve('idle') }) }), - ]) - expect(outcome).toBe('idle') - - controller.abort() - await expect(pendingFrame).resolves.toMatchObject({ done: true }) - await iterator.return?.() - detachAgent() - lifecycle.detach() - await ctx.fiber.dispose() - }) - - it('does not read the sessions service after its disposal status', async () => { - let sessionsFiber: Fiber | undefined - const ctx = await hostContext((fiber) => { sessionsFiber = fiber }) - if (sessionsFiber === undefined) throw new Error('sessions fiber missing') - createApiProxy(ctx, { provider: 'deepseek', model: 'deepseek-chat', cwd: '/tmp', workspaceRoot: '/tmp' }) - const sessions = ctx.get('sessions') - if (sessions === undefined) throw new Error('sessions service missing') - const list = vi.spyOn(sessions, 'list').mockImplementation(() => { - throw new Error('disposed sessions service read') - }) - - await expect(sessionsFiber.dispose()).resolves.toBeUndefined() - expect(list).not.toHaveBeenCalled() - await ctx.fiber.dispose() - }) - - it('invalidates pending capacity before SessionStore teardown can reach its callback', async () => { - let sessionsFiber: Fiber | undefined - const ctx = await hostContext((fiber) => { sessionsFiber = fiber }) - if (sessionsFiber === undefined) throw new Error('sessions fiber missing') - const deferred = new DeferredCatalogAdapter() - ctx.llm.registerAdapter(['deferred'], deferred) - const lifecycle = attachLifecycleSession(ctx, SessionId('capacity-session-store-teardown')) - const detachAgent = attachLifecycleAgent(ctx, lifecycle.session) - const api = createApiProxy(ctx, { provider: 'deepseek', model: 'deepseek-chat', cwd: '/tmp', workspaceRoot: '/tmp' }) - const controller = new AbortController() - const iterator = api.events.mux(request({}), controller.signal)[Symbol.asyncIterator]() - - expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() - await vi.waitFor(() => { expect(deferred.pending).toHaveLength(1) }) - const agents = ctx.get('agents') - if (agents === undefined) throw new Error('agent registry missing') - expect(agents.get(lifecycle.session.id)).toBeDefined() - const getAgent = vi.spyOn(agents, 'get') - - await sessionsFiber.dispose() - getAgent.mockClear() - deferred.resolve(0, 64_000) - await settleCapacityCompletion() - expect(getAgent).not.toHaveBeenCalled() - - const pendingFrame = iterator.next() - const outcome = await Promise.race([ - pendingFrame.then(() => 'frame' as const), - new Promise<'idle'>((resolve) => { setImmediate(() => { resolve('idle') }) }), - ]) - expect(outcome).toBe('idle') - - controller.abort() - await expect(pendingFrame).resolves.toMatchObject({ done: true }) - await iterator.return?.() - detachAgent() - lifecycle.detach() - await ctx.fiber.dispose() - }) - - it('publishes unknown metrics after AgentRegistry terminal disposal with mux active', async () => { - let agentsFiber: Fiber | undefined - const ctx = await hostContext(undefined, (fiber) => { agentsFiber = fiber }) - if (agentsFiber === undefined) throw new Error('agent registry fiber missing') - const deferred = new DeferredCatalogAdapter() - ctx.llm.registerAdapter(['deferred'], deferred) - const lifecycle = attachLifecycleSession(ctx, SessionId('capacity-agent-registry-disposed')) - const detachAgent = attachLifecycleAgent(ctx, lifecycle.session) - const api = createApiProxy(ctx, { provider: 'deepseek', model: 'deepseek-chat', cwd: '/tmp', workspaceRoot: '/tmp' }) - const controller = new AbortController() - const iterator = api.events.mux(request({}), controller.signal)[Symbol.asyncIterator]() - - expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() - await vi.waitFor(() => { expect(deferred.pending).toHaveLength(1) }) - deferred.resolve(0, 64_000) - await settleCapacityCompletion() - expect((await nextMetrics(iterator)).contextWindow).toBe(64_000) - - await agentsFiber.dispose() - const refresh = nextMetrics(iterator).then( - metrics => ({ kind: 'metrics' as const, metrics }), - () => ({ kind: 'error' as const }), - ) - const outcome = await Promise.race([ - refresh, - new Promise<{ kind: 'idle' }>((resolve) => { - setImmediate(() => { resolve({ kind: 'idle' }) }) - }), - ]) - - controller.abort() - await refresh - await iterator.return?.() - detachAgent() - lifecycle.detach() - await ctx.fiber.dispose() - expect(outcome.kind).toBe('metrics') - if (outcome.kind === 'metrics') { - expect(outcome.metrics.contextWindow).toBeUndefined() - expect(outcome.metrics.logRevision).toBe(1) - } - }) - - it.each(['agents', 'sessions'] as const)( - 'drops capacity completion while %s is unavailable during unload', - async (serviceName) => { - let sessionsFiber: Fiber | undefined - let agentsFiber: Fiber | undefined - const ctx = await hostContext( - (fiber) => { sessionsFiber = fiber }, - (fiber) => { agentsFiber = fiber }, - ) - const heldFiber = serviceName === 'sessions' ? sessionsFiber : agentsFiber - if (heldFiber === undefined) throw new Error(`${serviceName} fiber missing`) - const unloadStarted = Promise.withResolvers() - const releaseUnload = Promise.withResolvers() - heldFiber.ctx.effect(() => () => { - unloadStarted.resolve(undefined) - return releaseUnload.promise - }, `test: hold ${serviceName} unload`) - const deferred = new DeferredCatalogAdapter() - ctx.llm.registerAdapter(['deferred'], deferred) - const lifecycle = attachLifecycleSession(ctx, SessionId(`capacity-${serviceName}-unloading`)) - const detachAgent = attachLifecycleAgent(ctx, lifecycle.session) - const api = createApiProxy(ctx, { provider: 'deepseek', model: 'deepseek-chat', cwd: '/tmp', workspaceRoot: '/tmp' }) - const controller = new AbortController() - const iterator = api.events.mux(request({}), controller.signal)[Symbol.asyncIterator]() - - expect((await nextMetrics(iterator)).contextWindow).toBeUndefined() - await vi.waitFor(() => { expect(deferred.pending).toHaveLength(1) }) - const agents = ctx.get('agents') - if (agents === undefined) throw new Error('agent registry missing') - const sessions = ctx.get('sessions') - if (sessions === undefined) throw new Error('sessions service missing') - const getAgent = vi.spyOn(agents, 'get') - const getSession = vi.spyOn(sessions, 'get') - const disposing = heldFiber.dispose() - await unloadStarted.promise - await vi.waitFor(() => { expect(ctx.get(serviceName)).toBeUndefined() }) - expect(heldFiber.state).toBe(FiberState.UNLOADING) - - getAgent.mockClear() - getSession.mockClear() - deferred.resolve(0, 64_000) - await settleCapacityCompletion() - const agentReads = getAgent.mock.calls.length - const sessionReads = getSession.mock.calls.length - const pendingFrame = iterator.next() - const outcome = await Promise.race([ - pendingFrame.then(() => 'frame' as const), - new Promise<'idle'>((resolve) => { setImmediate(() => { resolve('idle') }) }), - ]) - - controller.abort() - await expect(pendingFrame).resolves.toMatchObject({ done: true }) - await iterator.return?.() - releaseUnload.resolve(undefined) - await disposing - detachAgent() - lifecycle.detach() - await ctx.fiber.dispose() - expect(agentReads).toBe(0) - expect(sessionReads).toBe(0) - expect(outcome).toBe('idle') - }, - ) }) diff --git a/packages/host/apiproxy/tests/rpc-schemas.spec.ts b/packages/host/apiproxy/tests/rpc-schemas.spec.ts index 9a21341e26..ce2b4db320 100644 --- a/packages/host/apiproxy/tests/rpc-schemas.spec.ts +++ b/packages/host/apiproxy/tests/rpc-schemas.spec.ts @@ -146,10 +146,9 @@ describe('sessions domain schemas', () => { cacheReadTokens: 4_000, cacheWriteTokens: 500, contextTokens: 8_000, - contextWindow: 128_000, }, modelTarget: { provider: 'deepseek', model: 'deepseek-v4-flash' }, - }).metrics?.contextWindow).toBe(128_000) + }).metrics?.contextTokens).toBe(8_000) expect(() => sessionMetricsSchema.parse({ logRevision: 1, projectionRevision: 0, @@ -158,15 +157,6 @@ describe('sessions domain schemas', () => { cacheReadTokens: 0, cacheWriteTokens: 0, })).toThrow() - expect(() => sessionMetricsSchema.parse({ - logRevision: 1, - projectionRevision: 0, - uncachedInputTokens: 0, - outputTokens: 0, - cacheReadTokens: 0, - cacheWriteTokens: 0, - contextWindow: 0, - })).toThrow() expect(sessionModelsRequestSchema.parse({ sessionId: 's1' }).sessionId).toBe('s1') expect(sessionModelsValueSchema.parse({ current: { provider: 'deepseek', model: 'deepseek-v4-flash', reasoningEffort: 'max' }, @@ -344,6 +334,23 @@ describe('events frame schemas', () => { cacheWriteTokens: 40, }, }, + { + type: 'session/model-request', + sessionId: 's', + turn: 2, + step: 1, + provider: 'deepseek', + model: 'deepseek-chat', + contextWindow: 128_000, + }, + { + type: 'session/model-request', + sessionId: 's', + turn: 3, + step: 1, + provider: 'deepseek', + model: 'unknown-capacity', + }, { type: 'approval/requested', sessionId: 's', approvalId: 'a', toolName: 'bash', callId: 'c', reason: 'r' }, { type: 'approval/resolved', sessionId: 's', approvalId: 'a', outcome: 'allowed-once' }, { type: 'question/requested', sessionId: 's', questions: [{ id: 'q', question: 'Q?', options: [{ label: 'L' }], multiSelect: true }] }, @@ -358,6 +365,8 @@ describe('events frame schemas', () => { { type: 'session/title', sessionId: 's', title: '', eventSeq: 0, updatedAt: 1 }, { type: 'session/title', sessionId: 's', title: 'x', eventSeq: -1, updatedAt: 1 }, { type: 'session/title', sessionId: 's', title: 'x', eventSeq: 0.5, updatedAt: 1 }, + { type: 'session/model-request', sessionId: 's', turn: 0, step: 1, provider: 'p', model: 'm' }, + { type: 'session/model-request', sessionId: 's', turn: 1, step: 1, provider: 'p', model: 'm', contextWindow: 0 }, { type: 'session/title', sessionId: 's', title: 'x', eventSeq: 0, updatedAt: 'now' }, { type: 'session/title', sessionId: 's', title: 'x', eventSeq: 0, updatedAt: Number.NaN }, ]) expect(() => muxFrameSchema.parse(invalid)).toThrow() diff --git a/packages/host/apiproxy/tests/session-metrics.spec.ts b/packages/host/apiproxy/tests/session-metrics.spec.ts index deb5f740fe..34cb1cb430 100644 --- a/packages/host/apiproxy/tests/session-metrics.spec.ts +++ b/packages/host/apiproxy/tests/session-metrics.spec.ts @@ -1,6 +1,5 @@ -import { describe, expect, it, vi } from 'vitest' +import { describe, expect, it } from 'vitest' import { Context } from 'cordis' -import type { Agent, AgentLlmTarget } from '@deepseek-ai/dsh-agent' import { Session, SessionId } from '@deepseek-ai/dsh-session' import { affectsSessionMetrics, SessionMetricsProjector } from '../src/session-metrics.ts' @@ -29,16 +28,8 @@ function assistant( }, { surfaceOp: 'append' }) } -function agent(session: Session): Agent { - return { id: session.id, session } as Agent -} - -function settleAsyncWork(): Promise { - return new Promise((resolve) => { setImmediate(resolve) }) -} - describe('SessionMetricsProjector', () => { - it('filters text/reasoning stream deltas while retaining usage, headers, and surface mutations', () => { + it('filters text/reasoning deltas while retaining usage, headers, and surface mutations', () => { const session = new Session(SessionId('metrics-filter')) const text = session.append('assistant/chunk', { turn: 1, @@ -58,13 +49,16 @@ describe('SessionMetricsProjector', () => { content: [{ type: 'text', text: 'question' }], source: { kind: 'user' }, }, { surfaceOp: 'append' }) + const plain = session.append('step/start', { turn: 1, step: 1 }) + expect(affectsSessionMetrics(text)).toBe(false) expect(affectsSessionMetrics(usage)).toBe(true) expect(affectsSessionMetrics(header)).toBe(true) expect(affectsSessionMetrics(surface)).toBe(true) + expect(affectsSessionMetrics(plain)).toBe(false) }) - it('reconciles usage by turn:step, keeps cache writes disjoint, and survives a surface replacement', () => { + it('folds usage by turn and step while synchronous pressure follows surface replacement', () => { const ctx = new Context() ctx.provide('tokenMeter', { measure(session: Session) { @@ -83,11 +77,8 @@ describe('SessionMetricsProjector', () => { cacheWriteTokens: 8, }) - const current: AgentLlmTarget = { provider: 'test', model: 'alpha' } - const projector = new SessionMetricsProjector(ctx, () => current, () => {}) - const attached = agent(session) - const before = projector.snapshot(session, attached) - expect(before).toMatchObject({ + const projector = new SessionMetricsProjector(ctx) + expect(projector.snapshot(session)).toMatchObject({ uncachedInputTokens: 11, outputTokens: 3, cacheReadTokens: 89, @@ -104,8 +95,7 @@ describe('SessionMetricsProjector', () => { surfaceOp: { op: 'replace', start: first.seq, end: assistantSeq }, sourceEventSeqs: [first.seq, assistantSeq], }) - const compacted = projector.snapshot(session, attached) - expect(compacted).toMatchObject({ + expect(projector.snapshot(session)).toMatchObject({ uncachedInputTokens: 11, outputTokens: 3, cacheReadTokens: 89, @@ -113,357 +103,38 @@ describe('SessionMetricsProjector', () => { contextTokens: 100, }) - // A replayed usage event for the same step replaces the settled value. session.append('assistant/chunk', { turn: 1, step: 1, chunk: { type: 'usage', - usage: { - inputTokens: 12, - outputTokens: 4, - cacheReadTokens: 88, - cacheWriteTokens: 9, - }, + usage: { inputTokens: 12, outputTokens: 4, cacheReadTokens: 88, cacheWriteTokens: 9 }, }, }) - const replayed = projector.snapshot(session, attached) - expect(replayed).toMatchObject({ - uncachedInputTokens: 12, - outputTokens: 4, - cacheReadTokens: 88, - cacheWriteTokens: 9, - contextTokens: 100, - }) - assistant(session, 1, 2, { - inputTokens: 1_000, - outputTokens: 500, - cacheReadTokens: 2_000, - cacheWriteTokens: 3_000, - }) - - const after = projector.snapshot(session, attached) - expect(after).toMatchObject({ - logRevision: session.events.length, - projectionRevision: 3, - uncachedInputTokens: 1_012, - outputTokens: 504, - cacheReadTokens: 2_088, - cacheWriteTokens: 3_009, - contextTokens: 200, - }) - expect(after.uncachedInputTokens).not.toBe( - after.uncachedInputTokens + after.cacheReadTokens + after.cacheWriteTokens, - ) - }) - - it('publishes only the selected route capacity when asynchronous resolutions race', async () => { - const ctx = new Context() - const resolutions = new Map() - ctx.provide('tokenMeter', { measure: () => ({ totalTokens: 35_000 }) }) - ctx.provide('llm', { - resolveModelInfo(_provider: string, model: string, signal?: AbortSignal) { - return new Promise<{ context: { contextWindow: number } }>((resolve) => { - resolutions.set(model, { - signal, - resolve(contextWindow) { - resolve({ context: { contextWindow } }) - }, - }) - }) - }, - }) - const session = new Session(SessionId('capacity-race')) - const attached = agent(session) - let current: AgentLlmTarget = { provider: 'test', model: 'alpha' } - const resolved = vi.fn() - const projector = new SessionMetricsProjector(ctx, () => current, resolved) - - expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() - await vi.waitFor(() => { expect(resolutions.has('alpha')).toBe(true) }) - expect(resolutions.get('alpha')?.signal?.aborted).toBe(false) - current = { provider: 'test', model: 'beta' } - expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() - expect(resolutions.get('alpha')?.signal?.aborted).toBe(true) - await vi.waitFor(() => { expect(resolutions.has('beta')).toBe(true) }) - expect(resolutions.get('beta')?.signal?.aborted).toBe(false) - - resolutions.get('alpha')?.resolve(64_000) - await Promise.resolve() - expect(resolved).not.toHaveBeenCalled() - resolutions.get('beta')?.resolve(128_000) - await vi.waitFor(() => { expect(resolved).toHaveBeenCalledOnce() }) - expect(projector.snapshot(session, attached)).toMatchObject({ - contextTokens: 35_000, - contextWindow: 128_000, - }) - }) - - it('retries a failed same-route capacity lookup only on the next snapshot', async () => { - const ctx = new Context() - const attempts: PromiseWithResolvers<{ context: { contextWindow: number } }>[] = [] - ctx.provide('llm', { - resolveModelInfo() { - const attempt = Promise.withResolvers<{ context: { contextWindow: number } }>() - attempts.push(attempt) - return attempt.promise - }, - }) - const session = new Session(SessionId('capacity-retry')) - const attached = agent(session) - const resolved = vi.fn() - const projector = new SessionMetricsProjector( - ctx, - () => ({ provider: 'test', model: 'alpha' }), - resolved, - ) - - expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() - await vi.waitFor(() => { expect(attempts).toHaveLength(1) }) - attempts[0]?.reject(new Error('metadata temporarily unavailable')) - await settleAsyncWork() - expect(attempts).toHaveLength(1) - expect(resolved).not.toHaveBeenCalled() - - expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() - await settleAsyncWork() - expect(attempts).toHaveLength(2) - attempts[1]?.resolve({ context: { contextWindow: 128_000 } }) - await vi.waitFor(() => { expect(resolved).toHaveBeenCalledOnce() }) - expect(projector.snapshot(session, attached).contextWindow).toBe(128_000) - }) - - it('aborts every active capacity on invalidation and resolves fresh generations', async () => { - const ctx = new Context() - const attempts: { - result: PromiseWithResolvers<{ context: { contextWindow: number } }> - signal: AbortSignal | undefined - }[] = [] - ctx.provide('llm', { - resolveModelInfo(_provider: string, _model: string, signal?: AbortSignal) { - const result = Promise.withResolvers<{ context: { contextWindow: number } }>() - attempts.push({ result, signal }) - return result.promise - }, - }) - const firstSession = new Session(SessionId('capacity-invalidation-first')) - const secondSession = new Session(SessionId('capacity-invalidation-second')) - const firstAgent = agent(firstSession) - const secondAgent = agent(secondSession) - const resolved = vi.fn() - const projector = new SessionMetricsProjector( - ctx, - () => ({ provider: 'test', model: 'alpha' }), - resolved, - ) - - expect(projector.snapshot(firstSession, firstAgent).contextWindow).toBeUndefined() - expect(projector.snapshot(secondSession, secondAgent).contextWindow).toBeUndefined() - await vi.waitFor(() => { expect(attempts).toHaveLength(2) }) - expect(attempts.map(attempt => attempt.signal?.aborted)).toEqual([false, false]) - - projector.invalidateCapacities() - expect(attempts.map(attempt => attempt.signal?.aborted)).toEqual([true, true]) - attempts[0]?.result.resolve({ context: { contextWindow: 32_000 } }) - attempts[1]?.result.resolve({ context: { contextWindow: 64_000 } }) - await settleAsyncWork() - expect(resolved).not.toHaveBeenCalled() - - expect(projector.snapshot(firstSession, firstAgent).contextWindow).toBeUndefined() - expect(projector.snapshot(secondSession, secondAgent).contextWindow).toBeUndefined() - await vi.waitFor(() => { expect(attempts).toHaveLength(4) }) - expect(attempts.slice(2).map(attempt => attempt.signal?.aborted)).toEqual([false, false]) - attempts[2]?.result.resolve({ context: { contextWindow: 128_000 } }) - attempts[3]?.result.resolve({ context: { contextWindow: 256_000 } }) - await vi.waitFor(() => { expect(resolved).toHaveBeenCalledTimes(2) }) - expect(projector.snapshot(firstSession, firstAgent).contextWindow).toBe(128_000) - expect(projector.snapshot(secondSession, secondAgent).contextWindow).toBe(256_000) - - projector.dispose() - expect(projector.snapshot(firstSession, firstAgent).contextWindow).toBeUndefined() - expect(attempts).toHaveLength(4) - }) - - it('skips adapter work invalidated before its deferred invocation', async () => { - const ctx = new Context() - const attempts: { - result: PromiseWithResolvers<{ context: { contextWindow: number } }> - signal: AbortSignal | undefined - }[] = [] - const resolveModelInfo = vi.fn(( - _provider: string, - _model: string, - signal?: AbortSignal, - ) => { - const result = Promise.withResolvers<{ context: { contextWindow: number } }>() - attempts.push({ result, signal }) - return result.promise - }) - ctx.provide('llm', { resolveModelInfo }) - const session = new Session(SessionId('capacity-pre-invocation-invalidation')) - const attached = agent(session) - const resolved = vi.fn() - const projector = new SessionMetricsProjector( - ctx, - () => ({ provider: 'test', model: 'alpha' }), - resolved, - ) - - expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() - projector.invalidateCapacities() - await settleAsyncWork() - expect(resolveModelInfo).not.toHaveBeenCalled() - expect(resolved).not.toHaveBeenCalled() - - expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() - await vi.waitFor(() => { expect(attempts).toHaveLength(1) }) - expect(attempts[0]?.signal?.aborted).toBe(false) - attempts[0]?.result.resolve({ context: { contextWindow: 128_000 } }) - await vi.waitFor(() => { expect(resolved).toHaveBeenCalledOnce() }) - expect(projector.snapshot(session, attached).contextWindow).toBe(128_000) - }) - - it('starts a fresh capacity generation when an unavailable route returns', async () => { - const ctx = new Context() - const resolutions: { - signal: AbortSignal | undefined - resolve(contextWindow: number): void - }[] = [] - ctx.provide('llm', { - resolveModelInfo(_provider: string, _model: string, signal?: AbortSignal) { - return new Promise<{ context: { contextWindow: number } }>((resolve) => { - resolutions.push({ - signal, - resolve(contextWindow) { - resolve({ context: { contextWindow } }) - }, - }) - }) - }, - }) - const session = new Session(SessionId('capacity-route-return')) - const attached = agent(session) - let current: AgentLlmTarget | undefined = { provider: 'test', model: 'alpha' } - const resolved = vi.fn() - const targetFor = vi.fn(() => current) - const projector = new SessionMetricsProjector(ctx, targetFor, resolved) - - expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() - await vi.waitFor(() => { expect(resolutions).toHaveLength(1) }) - current = undefined - expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() - expect(resolutions[0]?.signal?.aborted).toBe(true) - resolutions[0]?.resolve(64_000) - await settleAsyncWork() - expect(resolved).not.toHaveBeenCalled() - current = { provider: 'test', model: 'alpha' } - expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() - await vi.waitFor(() => { expect(resolutions).toHaveLength(2) }) - expect(resolutions[1]?.signal?.aborted).toBe(false) - resolutions[1]?.resolve(128_000) - await vi.waitFor(() => { expect(resolved).toHaveBeenCalledOnce() }) - expect(projector.snapshot(session, attached).contextWindow).toBe(128_000) - }) - - it('omits current context fields when measurement or model metadata is unavailable', async () => { - const ctx = new Context() - ctx.provide('tokenMeter', { measure: () => { throw new Error('unmeasurable') } }) - ctx.provide('llm', { resolveModelInfo: () => Promise.reject(new Error('metadata unavailable')) }) - const session = new Session(SessionId('missing-metrics')) - const attached = agent(session) - const projector = new SessionMetricsProjector( - ctx, - () => ({ provider: 'test', model: 'missing' }), - () => {}, - ) - const metrics = projector.snapshot(session, attached) - expect(metrics.contextTokens).toBeUndefined() - expect(metrics.contextWindow).toBeUndefined() - await vi.waitFor(() => { - expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() - }) - }) - - it('keeps optional usage buckets at zero and tolerates absent host services or detached agents', async () => { - const ctx = new Context() - const session = new Session(SessionId('optional-metrics')) - assistant(session, 1, 0, { inputTokens: 7, outputTokens: 2 }) - const attached = agent(session) - const selected: { current?: AgentLlmTarget } = {} - const projector = new SessionMetricsProjector( - ctx, - () => selected.current, - () => {}, - ) + assistant(session, 1, 2, { inputTokens: 1_000, outputTokens: 500 }) expect(projector.snapshot(session)).toMatchObject({ - uncachedInputTokens: 7, - outputTokens: 2, - cacheReadTokens: 0, - cacheWriteTokens: 0, + logRevision: session.events.length, + projectionRevision: 2, + uncachedInputTokens: 1_012, + outputTokens: 504, + cacheReadTokens: 88, + cacheWriteTokens: 9, + contextTokens: 200, }) - expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() - selected.current = { provider: 'test', model: 'no-service' } - expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() - await Promise.resolve() }) - it('publishes a resolved route with no advertised capacity as unknown', async () => { - const ctx = new Context() - ctx.provide('llm', { resolveModelInfo: () => Promise.resolve({}) }) - const session = new Session(SessionId('no-capacity')) - const attached = agent(session) - const resolved = vi.fn() - const projector = new SessionMetricsProjector( - ctx, - () => ({ provider: 'test', model: 'metadata-without-context' }), - resolved, - ) + it('omits pressure when the token meter is absent or cannot measure the replay', () => { + const session = new Session(SessionId('metrics-pressure-unknown')) + const withoutMeter = new SessionMetricsProjector(new Context()).snapshot(session) + expect(withoutMeter.contextTokens).toBeUndefined() - expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() - await vi.waitFor(() => { expect(resolved).toHaveBeenCalledOnce() }) - expect(projector.snapshot(session, attached).contextWindow).toBeUndefined() - }) - - it('ignores stale resolution failures and route metadata after the target moves', async () => { const ctx = new Context() - const resolutions = new Map() - ctx.provide('llm', { - resolveModelInfo(_provider: string, model: string) { - return new Promise<{ context: { contextWindow: number } }>((resolve, reject) => { - resolutions.set(model, { resolve, reject }) - }) + ctx.provide('tokenMeter', { + measure() { + throw new Error('unmeasurable replay') }, }) - const session = new Session(SessionId('stale-capacity')) - const attached = agent(session) - let current: AgentLlmTarget = { provider: 'test', model: 'alpha' } - const resolved = vi.fn() - const targetFor = vi.fn(() => current) - const projector = new SessionMetricsProjector(ctx, targetFor, resolved) - - projector.snapshot(session, attached) - await vi.waitFor(() => { expect(resolutions.has('alpha')).toBe(true) }) - current = { provider: 'test', model: 'route-moved-before-snapshot' } - resolutions.get('alpha')?.resolve({ context: { contextWindow: 64_000 } }) - await vi.waitFor(() => { expect(targetFor).toHaveBeenCalledTimes(2) }) - expect(resolved).not.toHaveBeenCalled() - - projector.snapshot(session, attached) - await vi.waitFor(() => { expect(resolutions.has('route-moved-before-snapshot')).toBe(true) }) - current = { provider: 'test', model: 'beta' } - projector.snapshot(session, attached) - await vi.waitFor(() => { expect(resolutions.has('beta')).toBe(true) }) - resolutions.get('route-moved-before-snapshot')?.reject(new Error('stale failure')) - await Promise.resolve() - resolutions.get('beta')?.resolve({ context: { contextWindow: 128_000 } }) - await vi.waitFor(() => { expect(resolved).toHaveBeenCalledOnce() }) - expect(projector.snapshot(session, attached).contextWindow).toBe(128_000) + expect(new SessionMetricsProjector(ctx).snapshot(session).contextTokens).toBeUndefined() }) }) diff --git a/packages/llm/llm/README.i18n.yaml b/packages/llm/llm/README.i18n.yaml index d35885f2a2..6ef87e4ab8 100644 --- a/packages/llm/llm/README.i18n.yaml +++ b/packages/llm/llm/README.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write packages/llm/llm/README.md -README.md: 2328188e420df6de60f024982a31d37a858a303e -README.zh.md: 586767a9e790fa68d27673836700fd789ce6a180 +README.md: d28a5632a3fbdbf11c7dba2ee0c57a704f2ba6f4 +README.zh.md: fd93fa43d5bfabd6e8751d4efbad229096ced08d diff --git a/packages/llm/llm/README.md b/packages/llm/llm/README.md index 2328188e42..d28a5632a3 100644 --- a/packages/llm/llm/README.md +++ b/packages/llm/llm/README.md @@ -16,7 +16,7 @@ An adapter registry plus a single streaming call surface, interceptable via a wa - `ctx.llm.listModels(provider: string): Promise` Discover the models one registered provider currently advertises. - `ctx.llm.resolveModelInfo(provider: string, model: string, signal?: AbortSignal): Promise` Resolve validated exact-model identity plus available context and reasoning metadata from the owning adapter, with optional cancellation for asynchronous adapters. - `ctx.llm.resolveCallConfig(config: LlmCallConfig, signal?: AbortSignal): Promise` Validate an explicit effort and materialize an adapter-configured default without clamping. -- `ctx.llm.prepareCall(config: LlmCallConfig, signal?: AbortSignal): Promise` Resolve a config and capture its current adapter registration as one cancellable, one-shot call. +- `ctx.llm.prepareCall(config: LlmCallConfig, signal?: AbortSignal): Promise` Resolve a config plus available context metadata in one exact-model lookup and capture its current adapter registration as one cancellable, one-shot call. - `ctx.llm.stream(options: GenerateOptions): AsyncIterable` Stream one model call as raw chunks (token-level deltas). Consumers assemble the chunks into blocks/messages with `BlockAssembler`. `LlmService` preserves errors from final adapter selection, synchronous dispatch, iterator construction, and iteration, and binds their provenance to the exact stream handle returned for that model call. `isLlmAdapterFailure(stream, value)` reports only errors from that call's final adapter boundary; `llmFailureOf(stream, value)` returns the adjacent immutable `LlmFailure`; `llmRetryPolicyOf(stream)` returns the immutable policy of the exact registration selected at that boundary, even if the route is later disposed or replaced. A call that never reaches a final adapter has no serving policy. Nested model calls, `llm/stream` middleware, and downstream consumer failures remain unclassified for the outer call. Classification never replaces or mutates the adapter's original coded `Error`. @@ -25,7 +25,7 @@ Provider and model metadata is a discovery surface, not a routing whitelist. `re Exact-model metadata is a separate correctness query, not a catalog decoration or global LLM setting. `resolveModelInfo()` asks the adapter that owns the exact provider/model route once; an adapter can describe an unlisted dynamic model, and absent `context` or `reasoning` fields mean only that those capabilities are unavailable. Invalid identity, context, or reasoning metadata fails with `INVALID_MODEL_INFO`, `INVALID_MODEL_CONTEXT`, or `INVALID_MODEL_REASONING`. -Reasoning identifiers are opaque adapter-owned strings rather than a core enum. An adapter publishes its ordered selectable list, including an `off` id when that model's capability API exposes one. `resolveCallConfig()` accepts only an exact advertised identifier, materializes `defaultEffort` when present, and otherwise preserves the provider default. Asynchronous model resolvers receive the caller's signal and must settle promptly after cancellation. `prepareCall()` additionally retains the exact adapter registration through header logging and terminal dispatch, so HMR cannot combine one adapter's capability result with another adapter's request; reusing its one-shot handle or changing its call-config fields fails with `INVALID_PREPARED_CALL`. An unsupported explicit or configured effort fails with `UNSUPPORTED_REASONING_EFFORT` before provider I/O. +Reasoning identifiers are opaque adapter-owned strings rather than a core enum. An adapter publishes its ordered selectable list, including an `off` id when that model's capability API exposes one. `resolveCallConfig()` accepts only an exact advertised identifier, materializes `defaultEffort` when present, and otherwise preserves the provider default. Asynchronous model resolvers receive the caller's signal and must settle promptly after cancellation. `prepareCall()` additionally exposes the detached context metadata from that same lookup and retains the exact adapter registration through header logging and terminal dispatch, so HMR cannot combine one adapter's capability result with another adapter's request; reusing its one-shot handle or changing its call-config fields fails with `INVALID_PREPARED_CALL`. Its dispatch observer runs after a final stream handle is constructed and before adapter iteration. An unsupported explicit or configured effort fails with `UNSUPPORTED_REASONING_EFFORT` before provider I/O. ### Events diff --git a/packages/llm/llm/README.zh.md b/packages/llm/llm/README.zh.md index 586767a9e7..fd93fa43d5 100644 --- a/packages/llm/llm/README.zh.md +++ b/packages/llm/llm/README.zh.md @@ -16,7 +16,7 @@ - `ctx.llm.listModels(provider: string): Promise` 发现某个已注册提供方当前公布的模型。 - `ctx.llm.resolveModelInfo(provider: string, model: string, signal?: AbortSignal): Promise` 从拥有精确路由的适配器解析经校验的确切模型身份、可用上下文和推理(reasoning)元数据;异步适配器可选地支持取消。 - `ctx.llm.resolveCallConfig(config: LlmCallConfig, signal?: AbortSignal): Promise` 校验显式推理强度,并填入适配器配置的默认值,但不自动调整。 -- `ctx.llm.prepareCall(config: LlmCallConfig, signal?: AbortSignal): Promise` 解析配置并将其当前适配器注册捕获为一次可取消、一次性调用。 +- `ctx.llm.prepareCall(config: LlmCallConfig, signal?: AbortSignal): Promise` 在一次精确模型查询中解析配置与可用上下文元数据,并将其当前适配器注册捕获为一次可取消、一次性调用。 - `ctx.llm.stream(options: GenerateOptions): AsyncIterable` 将一次模型调用流式输出为原始 chunk(token 级 delta)。消费方使用 `BlockAssembler` 将 chunk 组装为块/消息。 `LlmService` 保留来自最终适配器选择、同步 dispatch、iterator 构造与迭代的错误,并将其溯源绑定到该次模型调用返回的精确流句柄。`isLlmAdapterFailure(stream, value)` 只报告该调用最终适配器边界的错误;`llmFailureOf(stream, value)` 返回相邻的不可变 `LlmFailure`;`llmRetryPolicyOf(stream)` 返回在该边界选中的确切注册所对应的不可变策略,即使之后释放或替换路由也不变。未到达最终适配器的调用没有服务策略。嵌套模型调用、`llm/stream` middleware 和下游消费方失败对外层调用仍未分类。分类绝不替换或更改适配器的原始编码 `Error`。 @@ -25,7 +25,7 @@ 确切模型元数据是独立的正确性查询,不是 catalog 装饰或全局 LLM 设置。`resolveModelInfo()` 会向拥有精确提供方/模型路由的适配器查询一次;适配器可以描述未列出的动态模型,缺少 `context` 或 `reasoning` 字段只表示相应能力不可用。无效的身份、上下文或推理元数据会以 `INVALID_MODEL_INFO`、`INVALID_MODEL_CONTEXT` 或 `INVALID_MODEL_REASONING` 失败。 -推理标识符是由适配器持有的不透明字符串,而非核心枚举。适配器会公布有序可选列表;模型能力 API 提供 `off` id 时,列表也会包含它。`resolveCallConfig()` 只接受与已公布标识符完全一致的值,在存在 `defaultEffort` 时填入它,否则保留提供方默认值。异步模型解析器会接收调用方的 signal,并且必须在取消后迅速完成结算。`prepareCall()` 还会让精确适配器注册跨越请求头记录和最终分派,因此 HMR(热模块替换)不会将一个适配器的能力结果与另一个适配器的请求混用;复用其一次性句柄或更改调用配置字段会以 `INVALID_PREPARED_CALL` 失败。不支持的显式或配置推理强度会在提供方 I/O 前以 `UNSUPPORTED_REASONING_EFFORT` 失败。 +推理标识符是由适配器持有的不透明字符串,而非核心枚举。适配器会公布有序可选列表;模型能力 API 提供 `off` id 时,列表也会包含它。`resolveCallConfig()` 只接受与已公布标识符完全一致的值,在存在 `defaultEffort` 时填入它,否则保留提供方默认值。异步模型解析器会接收调用方的 signal,并且必须在取消后迅速完成结算。`prepareCall()` 还会公开同一次查询得到的脱耦上下文元数据,并让精确适配器注册跨越请求头记录和最终分派,因此 HMR(热模块替换)不会将一个适配器的能力结果与另一个适配器的请求混用;复用其一次性句柄或更改调用配置字段会以 `INVALID_PREPARED_CALL` 失败。其分派观察器在最终流句柄构造完成后、适配器开始迭代前运行。不支持的显式或配置推理强度会在提供方 I/O 前以 `UNSUPPORTED_REASONING_EFFORT` 失败。 ### 事件 diff --git a/packages/llm/llm/src/index.ts b/packages/llm/llm/src/index.ts index 103bab747f..12bd2ec6ea 100644 --- a/packages/llm/llm/src/index.ts +++ b/packages/llm/llm/src/index.ts @@ -10,6 +10,7 @@ import { Context, Service } from 'cordis' import type { GenerateOptions, LlmFailure, + LlmModelContext, LlmModelInfo, LlmResolvedModelInfo, LlmProviderInfo, @@ -111,14 +112,18 @@ export class LlmError extends HarnessError { export interface PreparedLlmCall { /** Detached, deep-frozen config with any adapter-owned default materialized. */ readonly config: LlmCallConfig + /** Detached context metadata resolved with the registration-bound call. */ + readonly context?: LlmModelContext /** * Dispatch this call once through the registration captured during * preparation. The request's call-config fields must match {@link config}; * reuse or mismatch fails with `INVALID_PREPARED_CALL`. * @param options - fully assembled request carrying the prepared config. + * @param onDispatched - contained Agent-loop notification hook invoked after + * a stream handle is constructed and before its adapter is iterated. * @returns the chunk stream, including the `llm/stream` waterfall. */ - stream(options: GenerateOptions): AsyncIterable + stream(options: GenerateOptions, onDispatched?: () => void): AsyncIterable } /** @@ -392,15 +397,16 @@ export class LlmService extends Service { * @returns a detached config only when a default must be materialized. */ async resolveCallConfig(config: LlmCallConfig, signal?: AbortSignal): Promise { - return this.resolveCallConfigFor(this.registration(config.provider), config, signal) + return (await this.resolveCallFor(this.registration(config.provider), config, signal)).config } - private async resolveCallConfigFor( + private async resolveCallFor( registration: AdapterRegistration, config: LlmCallConfig, signal?: AbortSignal, - ): Promise { - const reasoning = (await this.resolveModelInfoFor(registration, config.model, signal)).reasoning + ): Promise<{ config: LlmCallConfig; context?: LlmModelContext }> { + const resolved = await this.resolveModelInfoFor(registration, config.model, signal) + const reasoning = resolved.reasoning const requested = config.reasoningEffort if (reasoning === undefined) { if (requested !== undefined) { @@ -409,17 +415,28 @@ export class LlmService extends Service { 'UNSUPPORTED_REASONING_EFFORT', ) } - return config + return { + config, + ...resolved.context === undefined ? {} : { context: resolved.context }, + } } const effective = requested ?? reasoning.defaultEffort - if (effective === undefined) return config + if (effective === undefined) { + return { + config, + ...resolved.context === undefined ? {} : { context: resolved.context }, + } + } if (!reasoning.efforts.some(effort => effort.id === effective)) { throw new LlmError( `provider "${config.provider}" model "${config.model}" does not support reasoning effort "${effective}"`, 'UNSUPPORTED_REASONING_EFFORT', ) } - return requested === effective ? config : { ...config, reasoningEffort: effective } + return { + config: requested === effective ? config : { ...config, reasoningEffort: effective }, + ...resolved.context === undefined ? {} : { context: resolved.context }, + } } /** @@ -432,18 +449,25 @@ export class LlmService extends Service { */ async prepareCall(config: LlmCallConfig, signal?: AbortSignal): Promise { const registration = this.registration(config.provider) - const resolvedConfig = deepFreeze(structuredClone( - await this.resolveCallConfigFor(registration, config, signal), - )) + const resolved = await this.resolveCallFor(registration, config, signal) + const resolvedConfig = deepFreeze(structuredClone(resolved.config)) + const context = resolved.context === undefined + ? undefined + : Object.freeze(structuredClone(resolved.context)) let dispatched = false return Object.freeze({ config: resolvedConfig, - stream: (options: GenerateOptions): AsyncIterable => { + ...context === undefined ? {} : { context }, + stream: (options: GenerateOptions, onDispatched?: () => void): AsyncIterable => { if (dispatched) { throw new LlmError('a prepared LLM call can only be dispatched once', 'INVALID_PREPARED_CALL') } dispatched = true - return this.streamWithRegistration(options, { registration, config: resolvedConfig }) + return this.streamWithRegistration( + options, + { registration, config: resolvedConfig }, + onDispatched, + ) }, }) } @@ -478,25 +502,51 @@ export class LlmService extends Service { * so it cannot suppress the primary provider error. A downstream close awaits * adapter cleanup, whose failures remain ordinary untagged work. */ - private async * adapterStream( + private adapterStream( options: GenerateOptions, failures: AdapterFailureScope, prepared?: { registration: AdapterRegistration; config: LlmCallConfig }, - ): AsyncGenerator { + onDispatched?: () => void, + ): AsyncIterable { + if (prepared === undefined) { + return this.resolveAndStream(options, failures, onDispatched) + } let iterator: AsyncIterator try { - const registration = prepared?.registration ?? this.registration(options.provider) + const registration = prepared.registration failures.retryPolicy = registration.retryPolicy - const resolvedConfig = prepared === undefined - ? await this.resolveCallConfigFor(registration, options, options.signal) - : prepared.config - if (prepared !== undefined && !callConfigEquals(options, resolvedConfig)) { + const resolvedConfig = prepared.config + if (!callConfigEquals(options, resolvedConfig)) { throw new LlmError( 'prepared LLM call config changed before adapter dispatch', 'INVALID_PREPARED_CALL', ) } - const resolvedOptions = prepared !== undefined || callConfigEquals(options, resolvedConfig) + const adapter = registration.adapter + const stream = adapter.stream(this.forAdapter(options, adapter)) + iterator = stream[Symbol.asyncIterator]() + } catch (error: unknown) { + return this.failedAdapterStream(markLlmAdapterFailure(failures, error)) + } + this.notifyDispatched(onDispatched) + return this.iterateAdapter(iterator, failures) + } + + private async * resolveAndStream( + options: GenerateOptions, + failures: AdapterFailureScope, + onDispatched?: () => void, + ): AsyncGenerator { + let iterator: AsyncIterator + try { + const registration = this.registration(options.provider) + failures.retryPolicy = registration.retryPolicy + const resolvedConfig = (await this.resolveCallFor( + registration, + options, + options.signal, + )).config + const resolvedOptions = callConfigEquals(options, resolvedConfig) ? options : Object.isFrozen(options) ? deepFreeze({ ...options, ...resolvedConfig }) @@ -507,7 +557,19 @@ export class LlmService extends Service { } catch (error: unknown) { throw markLlmAdapterFailure(failures, error) } + this.notifyDispatched(onDispatched) + yield* this.iterateAdapter(iterator, failures) + } + private async * failedAdapterStream(error: Error): AsyncGenerator { + await Promise.resolve() + throw error + } + + private async * iterateAdapter( + iterator: AsyncIterator, + failures: AdapterFailureScope, + ): AsyncGenerator { let completed = false let iterationFailed = false try { @@ -537,6 +599,15 @@ export class LlmService extends Service { } } + private notifyDispatched(onDispatched: (() => void) | undefined): void { + if (onDispatched === undefined) return + try { + onDispatched() + } catch (error: unknown) { + this.ctx.logger.warn(`llm dispatch observer threw: ${String(error)}`) + } + } + /** * Stream one model call as raw chunks (token-level deltas). Throws * `LlmError` with code `NO_ADAPTER` if no adapter is registered for @@ -548,23 +619,32 @@ export class LlmService extends Service { * agent-loop request recovery; middleware and nested-call failures remain * untagged for the outer call. * @param options - the full request; `options.provider` selects the adapter. + * @param onDispatched - contained Agent-loop notification hook invoked after + * a stream handle is constructed and before its adapter is iterated. * @returns the chunk stream, possibly wrapped by `llm/stream` listeners. */ - stream(options: GenerateOptions): AsyncIterable { - return this.streamWithRegistration(options) + stream(options: GenerateOptions, onDispatched?: () => void): AsyncIterable { + return this.streamWithRegistration(options, undefined, onDispatched) } private streamWithRegistration( options: GenerateOptions, prepared?: { registration: AdapterRegistration; config: LlmCallConfig }, + onDispatched?: () => void, ): AsyncIterable { const failures: AdapterFailureScope = { failures: new WeakMap() } + let terminalEntered = false const stream = this.ctx.waterfall( this, 'llm/stream', options, - () => this.adapterStream(options, failures, prepared), + () => { + terminalEntered = true + return this.adapterStream(options, failures, prepared, onDispatched) + }, ) + // eslint-disable-next-line @typescript-eslint/no-unnecessary-condition -- waterfall mutates this latch. + if (!terminalEntered) this.notifyDispatched(onDispatched) return bindAdapterFailureScope(stream, failures) } } diff --git a/packages/llm/llm/tests/service.spec.ts b/packages/llm/llm/tests/service.spec.ts index dc4cf1d9c0..85d07be6fe 100644 --- a/packages/llm/llm/tests/service.spec.ts +++ b/packages/llm/llm/tests/service.spec.ts @@ -1058,6 +1058,40 @@ describe('LlmService', () => { })).toThrow(expect.objectContaining({ code: 'INVALID_PREPARED_CALL' })) }) + it('reuses one exact-model lookup for prepared config and context metadata', async () => { + const ctx = new Context() + await ctx.plugin(LlmService) + let resolutions = 0 + const source = { contextWindow: 128_000 } + const adapter = new class extends ScriptedAdapter { + override resolveModel(provider: string, model: string): Promise { + resolutions += 1 + return Promise.resolve({ + provider, + id: model, + name: model, + context: source, + reasoning: { + efforts: [{ id: ReasoningEffortId('high'), name: 'High' }], + defaultEffort: ReasoningEffortId('high'), + }, + }) + } + }(SCRIPT) + ctx.llm.registerAdapter(['route'], adapter) + + const prepared = await ctx.llm.prepareCall({ provider: 'route', model: 'model' }) + source.contextWindow = 64_000 + expect(prepared.config.reasoningEffort).toBe(ReasoningEffortId('high')) + expect(prepared.context).toEqual({ contextWindow: 128_000 }) + expect(Object.isFrozen(prepared.context)).toBe(true) + for await (const _chunk of prepared.stream({ + ...prepared.config, + messages: [], + })) { /* drain */ } + expect(resolutions).toBe(1) + }) + it('passes cancellation through exact-model resolution', async () => { const ctx = new Context() await ctx.plugin(LlmService) diff --git a/scripts/gen-cordis-catalog.ts b/scripts/gen-cordis-catalog.ts index b10ad9dfac..5a1eae1faa 100644 --- a/scripts/gen-cordis-catalog.ts +++ b/scripts/gen-cordis-catalog.ts @@ -217,6 +217,7 @@ const FOUNDATION_TYPE_NAMES = new Set([ /** Project types deliberately documented outside the core-data catalog. */ const TYPE_LINK_EXEMPTIONS: Readonly> = { AgentFactory: 'agent creation seam is owned by packages/core/agent/README.md', + AgentModelRequest: 'event-local live request metadata is owned by packages/core/agent/README.md', BeginCommandRequest: 'event-local request contract is owned by packages/client/ui-slash/src/types.ts', InsertReferenceRequest: 'event-local request contract is owned by packages/client/ui-slash/src/types.ts', ConsumeTokenRequest: 'event-local request contract is owned by packages/client/ui-slash/src/types.ts', From e6ce6abd7d4b3747cf3f7610c836fc3435dc20bf Mon Sep 17 00:00:00 2001 From: Hypatia May Date: Tue, 28 Jul 2026 19:03:45 +0800 Subject: [PATCH 12/38] refactor(llm): simplify live request telemetry (round 2) --- ...8-host-owned-web-session-metrics.i18n.yaml | 4 +- ...26-07-28-host-owned-web-session-metrics.md | 2 +- ...07-28-host-owned-web-session-metrics.zh.md | 2 +- docs/architecture.i18n.yaml | 4 +- docs/architecture.md | 4 +- docs/architecture.zh.md | 4 +- docs/cordis-catalog/events.md | 12 +- docs/cordis-catalog/services.md | 6 +- .../llm-streaming.i18n.yaml | 4 +- docs/core-data-structures/llm-streaming.md | 6 +- docs/core-data-structures/llm-streaming.zh.md | 6 +- .../src/client/sessions/conversation.ts | 2 +- .../cordis/tool-cordis/src/api-catalog.ts | 10 +- packages/core/agent-loop/README.i18n.yaml | 4 +- packages/core/agent-loop/README.md | 2 +- packages/core/agent-loop/README.zh.md | 4 +- packages/core/agent-loop/src/agent.ts | 33 +++-- .../tests/request-reconstruction.spec.ts | 52 +++++++- packages/core/agent/README.i18n.yaml | 4 +- packages/core/agent/README.md | 2 +- packages/core/agent/README.zh.md | 2 +- packages/core/agent/src/types.ts | 18 +-- packages/host/apiproxy/README.i18n.yaml | 4 +- packages/host/apiproxy/README.md | 2 +- packages/host/apiproxy/README.zh.md | 2 +- packages/host/apiproxy/src/api/events.ts | 10 +- packages/llm/llm/README.i18n.yaml | 4 +- packages/llm/llm/README.md | 2 +- packages/llm/llm/README.zh.md | 2 +- packages/llm/llm/src/index.ts | 117 ++++-------------- packages/llm/llm/tests/service.spec.ts | 16 ++- 31 files changed, 159 insertions(+), 187 deletions(-) diff --git a/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.i18n.yaml b/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.i18n.yaml index 100837813e..4f7d187049 100644 --- a/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.i18n.yaml +++ b/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write .agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.md -2026-07-28-host-owned-web-session-metrics.md: e37a7635cbdae65162308bf8a3a498b5071a4910 -2026-07-28-host-owned-web-session-metrics.zh.md: fd3e50c8cf8c0336a8cb2cd55f6629419bb8ad23 +2026-07-28-host-owned-web-session-metrics.md: bc66ee72dc278196af8bab65c5e118d8f5cf039c +2026-07-28-host-owned-web-session-metrics.zh.md: a103a8e24bd806ded6131b94f4c123c8d745513d diff --git a/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.md b/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.md index e37a7635cb..bc66ee72dc 100644 --- a/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.md +++ b/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.md @@ -12,7 +12,7 @@ A Web stats line derived from the currently loaded conversation nodes is window- The Host owns one session-level metrics projection. It incrementally folds the complete durable event log, keys settled usage by `(turn, step)`, and replaces an earlier usage record for the same key instead of double-counting chunk and message forms. Uncached input, output, cache reads, and cache writes remain four disjoint cumulative buckets. Compaction can change the current prompt surface without erasing historical usage. -Current context pressure is the point-in-time `tokenMeter.measure(session).totalTokens`. Capacity instead belongs to the latest model request observed by the current live mux connection. `LlmService.prepareCall()` retains the context metadata obtained by the exact lookup that also validates reasoning/defaults, and the loop publishes it through one contained `agent/model-request` notification only after the final route has a successfully constructed stream handle. Failed or aborted iteration still counts as a dispatched request; preparation and synchronous construction failures do not. +Current context pressure is the point-in-time `tokenMeter.measure(session).totalTokens`. Capacity instead belongs to the latest model request attempt observed by the current live mux connection. `LlmService.prepareCall()` retains the context metadata obtained by the exact lookup that also validates reasoning/defaults. After the final provider/model is fixed and the outer `llm/stream` call returns a handle, the loop publishes one contained `agent/model-request` notification. This boundary observes an attempt, not proof of provider I/O: preparation or a synchronous outer waterfall failure emits nothing, while short-circuit handles and later lazy adapter construction, iteration failure, or abort still count. The tail `session.history` response carries durable usage and pressure, while older pages omit them. Live changes use `session/metrics` mux frames. Both forms carry a durable-log revision and a projection revision; the client accepts only nondecreasing revisions and preserves metrics across older-page prepend. diff --git a/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.zh.md b/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.zh.md index fd3e50c8cf..a103a8e24b 100644 --- a/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.zh.md +++ b/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.zh.md @@ -12,7 +12,7 @@ Web 统计行若根据当前加载的会话节点推导指标,其结果会随 Host 拥有一项会话级指标投影。它以增量方式归并完整的持久事件日志,按 `(turn, step)` 标识已结算用量;同一标识再次出现时,会替换较早的用量记录,而不会重复统计分片和消息两种形态。未缓存输入、输出、缓存读取与缓存写入保持为四个彼此独立的累计计数项。压缩可以改变当前提示词表层,但不会抹除历史用量。 -当前上下文压力是即时的 `tokenMeter.measure(session).totalTokens`。容量则属于当前实时 mux 连接观察到的最新模型请求。`LlmService.prepareCall()` 会保留同一次精确查询取得的上下文元数据,该查询也负责校验推理设置与默认值;仅在最终路由的流句柄成功构造后,循环才会通过一条失败会被收容的 `agent/model-request` 通知发布这些元数据。后续迭代失败或中止仍算作已分派请求;准备阶段失败和同步构造失败则不算。 +当前上下文压力是即时的 `tokenMeter.measure(session).totalTokens`。容量则属于当前实时 mux 连接观察到的最新模型请求尝试。`LlmService.prepareCall()` 会保留同一次精确查询取得的上下文元数据,该查询也负责校验推理设置与默认值。最终提供方/模型确定且外层 `llm/stream` 调用返回句柄后,循环会发布一条失败会被收容的 `agent/model-request` 通知。这个边界观察到的是一次尝试,并不能证明提供方 I/O 已开始:准备阶段或外层 waterfall(瀑布式事件)的同步失败不会发出通知,而短路句柄以及之后的惰性适配器构造、迭代失败或中止仍会计入。 `session.history` 尾页响应携带持久用量与压力,较早页面则省略这两项。实时变更使用 `session/metrics` mux 帧。两种形式都携带持久日志修订号和投影修订号;客户端只接受不减小的修订号,并在向前加载较早页面时保留指标。 diff --git a/docs/architecture.i18n.yaml b/docs/architecture.i18n.yaml index cfe3106723..166f5388d6 100644 --- a/docs/architecture.i18n.yaml +++ b/docs/architecture.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write docs/architecture.md -architecture.md: 52072633a0e81afce63c5e162dd1b0af7f6486ca -architecture.zh.md: f0122ece146c17366aa316cfb4ea4196a1db74bd +architecture.md: e9ae1e27d6f1f1170b5da4823f3e988f4e778022 +architecture.zh.md: 3a31a2e6694d39f38e9c88de44f59ce7f753871f diff --git a/docs/architecture.md b/docs/architecture.md index 52072633a0..e9ae1e27d6 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -92,7 +92,7 @@ forever: assemble system prompt and tool schemas snapshot the derived messages (the reconstruction boundary) 'step/start' - agent/request (config only) -> prepare reasoning/default + context under turn signal -> log request/header -> construct llm/stream (frozen, registration-bound) -> agent/model-request (live, contained) -> iterate + agent/request (config only) -> prepare reasoning/default + context under turn signal -> log request/header -> obtain outer llm/stream handle (frozen request; prepared calls registration-bound) -> agent/model-request (live, contained attempt) -> iterate 'assistant/chunk' 'assistant/message' schedule tool calls by ctx.tools.executionMode: @@ -151,7 +151,7 @@ Log-only events may sit between turns. Owners append through `Session`, flushing Messages use typed blocks from merge-extensible `ContentBlockMap`; the pattern also types `MessageSource`, `FinishReason`, `TurnTrigger`, and `TurnEndReason`. New blocks coordinate adapters, UI, compaction, token metering, and persistence; replay measurements live in [token-meter.md](core-data-structures/token-meter.md). -Streaming uses raw chunks and `BlockAssembler`. After final-stream construction, the loop emits contained, non-durable, non-replayed `agent/model-request` metadata. Adapters normalize failures; `agent/request-error` may retry. Remote adapters use per-read idle watchdogs. Replay crosses routes only through a shared adapter ([contract](core-data-structures/llm-streaming.md)). +Streaming uses chunks and `BlockAssembler`. When the outer `llm/stream` returns a handle, AgentLoop emits contained, non-durable, non-replayed `agent/model-request` attempt metadata—not proof of provider I/O. `agent/request-error` may retry. Replay crosses routes only through a shared adapter ([contract](core-data-structures/llm-streaming.md)). ## Extension And Composition diff --git a/docs/architecture.zh.md b/docs/architecture.zh.md index f0122ece14..3a31a2e669 100644 --- a/docs/architecture.zh.md +++ b/docs/architecture.zh.md @@ -92,7 +92,7 @@ forever: assemble system prompt and tool schemas snapshot the derived messages (the reconstruction boundary) 'step/start' - agent/request (config only) -> prepare reasoning/default + context under turn signal -> log request/header -> construct llm/stream (frozen, registration-bound) -> agent/model-request (live, contained) -> iterate + agent/request (config only) -> prepare reasoning/default + context under turn signal -> log request/header -> obtain outer llm/stream handle (frozen request; prepared calls registration-bound) -> agent/model-request (live, contained attempt) -> iterate 'assistant/chunk' 'assistant/message' schedule tool calls by ctx.tools.executionMode: @@ -151,7 +151,7 @@ idle inject: 消息使用从可合并扩展的 `ContentBlockMap` 派生的类型化块;同一模式也为 `MessageSource`、`FinishReason`、`TurnTrigger` 和 `TurnEndReason` 定义类型。新增块会协调适配器、UI、压缩、token 计量和持久化;回放计量见 [token-meter.md](core-data-structures/token-meter.md)。 -流式输出使用原始分片和 `BlockAssembler`。最终流构造完成后,循环会发出 `agent/model-request` 元数据;该通知的失败会被收容,元数据不会持久化或回放。适配器会规范化故障;`agent/request-error` 可以重试。远程适配器使用逐次读取空闲看门狗。回放仅通过共用适配器跨路由传递([契约](core-data-structures/llm-streaming.md))。 +流式输出使用分片和 `BlockAssembler`。外层 `llm/stream` 返回句柄时,AgentLoop 会发出 `agent/model-request` 尝试元数据;该通知的失败会被收容,元数据不会持久化或回放,但这并不能证明提供方 I/O 已开始。`agent/request-error` 可以重试。回放仅通过共用适配器跨路由传递([契约](core-data-structures/llm-streaming.md))。 ## 扩展与组合 diff --git a/docs/cordis-catalog/events.md b/docs/cordis-catalog/events.md index e8e67da781..fe9ca59c9a 100644 --- a/docs/cordis-catalog/events.md +++ b/docs/cordis-catalog/events.md @@ -166,15 +166,15 @@ Source: [`packages/core/agent/src/types.ts:286`](../../packages/core/agent/src/t ### `agent/model-request` — emit -One model request constructed its final stream handle and is about to iterate it. This live notification is not durable or replayed; failed or aborted iteration still has a dispatch, while preparation and synchronous stream-construction failures do not. Listener failures are contained and cannot affect the request. +One model request obtained its outer `llm/stream` handle and is about to iterate it. This observes an Agent-loop request attempt, not proof that provider I/O began. The notification is live, contained, and not replayed. Preparation or a synchronous outer waterfall failure emits nothing; failures or abortion after the handle returns still count. ```ts cordis-catalog /** - * One model request constructed its final stream handle and is about to - * iterate it. This live notification is not durable or replayed; failed or - * aborted iteration still has a dispatch, while preparation and - * synchronous stream-construction failures do not. Listener failures are - * contained and cannot affect the request. + * One model request obtained its outer `llm/stream` handle and is about to + * iterate it. This observes an Agent-loop request attempt, not proof that + * provider I/O began. The notification is live, contained, and not replayed. + * Preparation or a synchronous outer waterfall failure emits nothing; + * failures or abortion after the handle returns still count. * @param agent - the agent dispatching the model request. * @param turn - the open turn number. * @param step - the request's step number. diff --git a/docs/cordis-catalog/services.md b/docs/cordis-catalog/services.md index 9d00ef37a0..7572a3ffd7 100644 --- a/docs/cordis-catalog/services.md +++ b/docs/cordis-catalog/services.md @@ -780,16 +780,14 @@ async prepareCall(config: LlmCallConfig, signal?: AbortSignal): Promise void): AsyncIterable +stream(options: GenerateOptions): AsyncIterable ``` Types: [GenerateOptions](../core-data-structures/core.md) · [LlmAdapter](../core-data-structures/llm-streaming.md) · [LlmCallConfig](../core-data-structures/core.md) · [LlmModelInfo](../core-data-structures/core.md) · [LlmProviderInfo](../core-data-structures/core.md) · [LlmResolvedModelInfo](../core-data-structures/core.md) · [PreparedLlmCall](../core-data-structures/llm-streaming.md) · [ResolvedRetryPolicy](../core-data-structures/llm-streaming.md) · [StreamChunk](../core-data-structures/llm-streaming.md) -Source: [`packages/llm/llm/src/index.ts:194`](../../packages/llm/llm/src/index.ts) +Source: [`packages/llm/llm/src/index.ts:192`](../../packages/llm/llm/src/index.ts) ## `ctx.permission` — `PermissionService` diff --git a/docs/core-data-structures/llm-streaming.i18n.yaml b/docs/core-data-structures/llm-streaming.i18n.yaml index 6206f864cf..5b4865b2ca 100644 --- a/docs/core-data-structures/llm-streaming.i18n.yaml +++ b/docs/core-data-structures/llm-streaming.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write docs/core-data-structures/llm-streaming.md -llm-streaming.md: 89628ebb96a2e8eec5209635cd92859427df4a8d -llm-streaming.zh.md: 9c8fcc1f24b970f3a7cdd7cd08d9ef3b934b4543 +llm-streaming.md: 30230d208c582463b3680d760c27f33a71b4cf6f +llm-streaming.zh.md: 32b3485f967b849371f916f6a669b97c97701cc1 diff --git a/docs/core-data-structures/llm-streaming.md b/docs/core-data-structures/llm-streaming.md index 89628ebb96..30230d208c 100644 --- a/docs/core-data-structures/llm-streaming.md +++ b/docs/core-data-structures/llm-streaming.md @@ -161,7 +161,7 @@ declare class BlockAssembler { ## The seam -`LlmAdapter` is the provider seam: subclass, implement `stream()`, and register one adapter instance with `ctx.llm.registerAdapter(providers, adapter)`. `GenerateOptions.provider` selects the registered adapter; `GenerateOptions.model` is passed to that adapter and need not be registered at lifecycle start. Duplicate provider routes fail atomically. Optional `providerRetryPolicy()` is captured per route with normal defaults, while `providerInfo()` and asynchronous `listModels()` feed `LlmService.listProviders()` / `listModels()` with detached selector metadata. That catalog is advisory rather than a request whitelist: the adapter remains authoritative and may accept unlisted model ids. One asynchronous `resolveModel()` query returns exact model identity plus optional correctness-sensitive context capacity and ordered model-owned reasoning ids with an optional deployment default; absent fields mean unavailable metadata or capability, not invalid catalog membership. The resolver receives optional cancellation and must settle promptly after abort. `LlmService.resolveModelInfo()` validates and detaches the aggregate. The service validates and materializes reasoning through `resolveCallConfig()` at the final adapter boundary, so direct calls cannot bypass unsupported-effort rejection; direct dispatch captures one registration before awaiting that resolution. The agent loop instead uses `prepareCall()` to keep the same registration across model resolution, durable header logging, and dispatch, and to retain detached context metadata from that exact lookup. Its optional observer runs after a final stream handle is constructed and before adapter iteration. Adapter lookup happens at the terminal continuation of the `llm/stream` waterfall, so a listener may short-circuit the call or route a mutable one-shot request before lookup. The `block-start` / `block-end` `index` correlation and the assembler together mean an adapter only has to emit well-formed chunks — block reassembly is not each adapter's problem. The consumer surface (`ctx.llm.stream()`) and the `llm/stream` waterfall are described in [architecture.md § Content blocks and streaming](../architecture.md#content-blocks-and-streaming-dsh-llm). +`LlmAdapter` is the provider seam: subclass, implement `stream()`, and register one adapter instance with `ctx.llm.registerAdapter(providers, adapter)`. `GenerateOptions.provider` selects the registered adapter; `GenerateOptions.model` is passed to that adapter and need not be registered at lifecycle start. Duplicate provider routes fail atomically. Optional `providerRetryPolicy()` is captured per route with normal defaults, while `providerInfo()` and asynchronous `listModels()` feed `LlmService.listProviders()` / `listModels()` with detached selector metadata. That catalog is advisory rather than a request whitelist: the adapter remains authoritative and may accept unlisted model ids. One asynchronous `resolveModel()` query returns exact model identity plus optional correctness-sensitive context capacity and ordered model-owned reasoning ids with an optional deployment default; absent fields mean unavailable metadata or capability, not invalid catalog membership. The resolver receives optional cancellation and must settle promptly after abort. `LlmService.resolveModelInfo()` validates and detaches the aggregate. The service validates and materializes reasoning through `resolveCallConfig()` at the final adapter boundary, so direct calls cannot bypass unsupported-effort rejection; direct dispatch captures one registration before awaiting that resolution. The agent loop instead uses `prepareCall()` to keep the same registration across model resolution, durable header logging, and dispatch, and to retain detached context metadata from that exact lookup. Adapter lookup happens at the terminal continuation of the `llm/stream` waterfall, so a listener may short-circuit the call or route a mutable one-shot request before lookup. AgentLoop observes a request attempt once the outer waterfall returns a stream handle; that limited boundary does not prove a lazy terminal adapter was constructed or began provider I/O. The `block-start` / `block-end` `index` correlation and the assembler together mean an adapter only has to emit well-formed chunks — block reassembly is not each adapter's problem. The consumer surface (`ctx.llm.stream()`) and the `llm/stream` waterfall are described in [architecture.md § Content blocks and streaming](../architecture.md#content-blocks-and-streaming-dsh-llm). ```ts type-equiv /** One model call whose config and adapter registration were resolved together. */ @@ -175,11 +175,9 @@ interface PreparedLlmCall { * preparation. The request's call-config fields must match {@link config}; * reuse or mismatch fails with `INVALID_PREPARED_CALL`. * @param options - fully assembled request carrying the prepared config. - * @param onDispatched - contained Agent-loop notification hook invoked after - * a stream handle is constructed and before its adapter is iterated. * @returns the chunk stream, including the `llm/stream` waterfall. */ - stream(options: GenerateOptions, onDispatched?: () => void): AsyncIterable + stream(options: GenerateOptions): AsyncIterable } ``` diff --git a/docs/core-data-structures/llm-streaming.zh.md b/docs/core-data-structures/llm-streaming.zh.md index 9c8fcc1f24..32b3485f96 100644 --- a/docs/core-data-structures/llm-streaming.zh.md +++ b/docs/core-data-structures/llm-streaming.zh.md @@ -161,7 +161,7 @@ declare class BlockAssembler { ## seam -`LlmAdapter` 是提供方 seam:创建子类、实现 `stream()`,再用 `ctx.llm.registerAdapter(providers, adapter)` 注册一个适配器实例。`GenerateOptions.provider` 选择已注册适配器;`GenerateOptions.model` 会传给该适配器,无需在生命周期启动时注册。重复提供方路由会原子失败。可选的 `providerRetryPolicy()` 会按路由捕获并填入 normal 默认值,`providerInfo()` 与异步 `listModels()` 方法则为 `LlmService.listProviders()` / `listModels()` 提供分离的 selector 元数据。该目录仅供参考,不是请求白名单:适配器仍是权威,并可接受未列出的模型 id。单次异步 `resolveModel()` 查询返回确切模型身份,以及可选的对正确性敏感的上下文容量、由模型持有的有序推理强度 ID 和部署默认值;字段缺失表示元数据或能力不可用,而不表示目录成员关系无效。解析器会接收可选的取消信号,并且必须在信号中止后迅速完成结算。`LlmService.resolveModelInfo()` 会校验聚合结果并返回分离值。服务通过最终适配器边界的 `resolveCallConfig()` 校验推理强度并填入默认值,因此直接调用也无法绕过对不支持推理强度的拒绝;直接分派会在等待解析前捕获一项适配器注册。agent loop 则使用 `prepareCall()`,使模型解析、请求头持久记录和分派全程使用同一项注册,并保留来自同一次精确查询的分离上下文元数据。其可选观察器在最终流句柄构造完成后、适配器开始迭代前运行。适配器查找发生在 `llm/stream` waterfall(瀑布式事件)的终端 continuation,因此 listener 可以在查找前短路调用,或路由一个可变的一次性请求。`block-start` / `block-end` 的 `index` 关联与 assembler 共同意味着适配器只需 emit 格式正确的分片——块重组不是每个适配器各自的问题。消费方 surface(`ctx.llm.stream()`)与 `llm/stream` waterfall 见 [architecture.md § 内容块与流式传输](../architecture.md#content-blocks-and-streaming-dsh-llm)。 +`LlmAdapter` 是提供方 seam:创建子类、实现 `stream()`,再用 `ctx.llm.registerAdapter(providers, adapter)` 注册一个适配器实例。`GenerateOptions.provider` 选择已注册适配器;`GenerateOptions.model` 会传给该适配器,无需在生命周期启动时注册。重复提供方路由会原子失败。可选的 `providerRetryPolicy()` 会按路由捕获并填入 normal 默认值,`providerInfo()` 与异步 `listModels()` 方法则为 `LlmService.listProviders()` / `listModels()` 提供分离的 selector 元数据。该目录仅供参考,不是请求白名单:适配器仍是权威,并可接受未列出的模型 id。单次异步 `resolveModel()` 查询返回确切模型身份,以及可选的对正确性敏感的上下文容量、由模型持有的有序推理强度 ID 和部署默认值;字段缺失表示元数据或能力不可用,而不表示目录成员关系无效。解析器会接收可选的取消信号,并且必须在信号中止后迅速完成结算。`LlmService.resolveModelInfo()` 会校验聚合结果并返回分离值。服务通过最终适配器边界的 `resolveCallConfig()` 校验推理强度并填入默认值,因此直接调用也无法绕过对不支持推理强度的拒绝;直接分派会在等待解析前捕获一项适配器注册。agent loop 则使用 `prepareCall()`,使模型解析、请求头持久记录和分派全程使用同一项注册,并保留来自同一次精确查询的分离上下文元数据。适配器查找发生在 `llm/stream` waterfall(瀑布式事件)的终端 continuation,因此 listener 可以在查找前短路调用,或路由一个可变的一次性请求。AgentLoop 在外层 waterfall 返回流句柄时观察到一次请求尝试;这个有限边界不能证明惰性终端适配器已构造完成或开始提供方 I/O。`block-start` / `block-end` 的 `index` 关联与 assembler 共同意味着适配器只需 emit 格式正确的分片——块重组不是每个适配器各自的问题。消费方 surface(`ctx.llm.stream()`)与 `llm/stream` waterfall 见 [architecture.md § 内容块与流式传输](../architecture.md#content-blocks-and-streaming-dsh-llm)。 ```ts type-equiv /** One model call whose config and adapter registration were resolved together. */ @@ -175,11 +175,9 @@ interface PreparedLlmCall { * preparation. The request's call-config fields must match {@link config}; * reuse or mismatch fails with `INVALID_PREPARED_CALL`. * @param options - fully assembled request carrying the prepared config. - * @param onDispatched - contained Agent-loop notification hook invoked after - * a stream handle is constructed and before its adapter is iterated. * @returns the chunk stream, including the `llm/stream` waterfall. */ - stream(options: GenerateOptions, onDispatched?: () => void): AsyncIterable + stream(options: GenerateOptions): AsyncIterable } ``` diff --git a/packages/client/runtime/src/client/sessions/conversation.ts b/packages/client/runtime/src/client/sessions/conversation.ts index 4ced165e7a..3277f42979 100644 --- a/packages/client/runtime/src/client/sessions/conversation.ts +++ b/packages/client/runtime/src/client/sessions/conversation.ts @@ -17,7 +17,7 @@ export type { TodoItem } * connection overlaid for presentation. */ export interface ConversationMetrics extends SessionMetrics { - /** Latest dispatched-request capacity; absent until observed or after reset/clear. */ + /** Latest observed request-attempt capacity; absent until observed or after reset/clear. */ contextWindow?: number } diff --git a/packages/cordis/tool-cordis/src/api-catalog.ts b/packages/cordis/tool-cordis/src/api-catalog.ts index 897935dad6..05d631c87c 100644 --- a/packages/cordis/tool-cordis/src/api-catalog.ts +++ b/packages/cordis/tool-cordis/src/api-catalog.ts @@ -397,8 +397,8 @@ export const SERVICE_API: readonly ServiceApiEntry[] = [ jsDoc: '/**\n * Resolve one call under its current adapter registration. The returned\n * one-shot handle keeps that registration across header logging and dispatch,\n * so HMR cannot combine one adapter\'s capability result with another adapter.\n * @param config - provider/model route and optional request controls.\n * @param signal - optional cancellation for adapter-owned capability lookup.\n * @returns a prepared config and its registration-bound stream entry point.\n */', }, { - signature: 'stream(options: GenerateOptions, onDispatched?: () => void): AsyncIterable', - jsDoc: '/**\n * Stream one model call as raw chunks (token-level deltas). Throws\n * `LlmError` with code `NO_ADAPTER` if no adapter is registered for\n * `options.provider`. Replay state is retained only when the same adapter\n * instance owns its historical provider and the target provider. Final\n * adapter selection remains fixed through asynchronous exact-model resolution\n * and dispatch. Selection, dispatch, and iteration failures retain their\n * original Error identity and are tagged in a call-local scope for narrow\n * agent-loop request recovery; middleware and nested-call failures remain\n * untagged for the outer call.\n * @param options - the full request; `options.provider` selects the adapter.\n * @param onDispatched - contained Agent-loop notification hook invoked after\n * a stream handle is constructed and before its adapter is iterated.\n * @returns the chunk stream, possibly wrapped by `llm/stream` listeners.\n */', + signature: 'stream(options: GenerateOptions): AsyncIterable', + jsDoc: '/**\n * Stream one model call as raw chunks (token-level deltas). Throws\n * `LlmError` with code `NO_ADAPTER` if no adapter is registered for\n * `options.provider`. Replay state is retained only when the same adapter\n * instance owns its historical provider and the target provider. Final\n * adapter selection remains fixed through asynchronous exact-model resolution\n * and dispatch. Selection, dispatch, and iteration failures retain their\n * original Error identity and are tagged in a call-local scope for narrow\n * agent-loop request recovery; middleware and nested-call failures remain\n * untagged for the outer call.\n * @param options - the full request; `options.provider` selects the adapter.\n * @returns the chunk stream, possibly wrapped by `llm/stream` listeners.\n */', }, ], }, @@ -1052,8 +1052,8 @@ export const EVENT_API: readonly EventApiEntry[] = [ name: 'agent/model-request', mode: 'emit', signature: '\'agent/model-request\'(this: Scoped, agent: Agent, turn: number, step: number, request: AgentModelRequest): void', - jsDoc: '/**\n * One model request constructed its final stream handle and is about to\n * iterate it. This live notification is not durable or replayed; failed or\n * aborted iteration still has a dispatch, while preparation and\n * synchronous stream-construction failures do not. Listener failures are\n * contained and cannot affect the request.\n * @param agent - the agent dispatching the model request.\n * @param turn - the open turn number.\n * @param step - the request\'s step number.\n * @param request - final route plus registration-bound context capacity.\n * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent.\n * @mode emit\n */', - summary: 'One model request constructed its final stream handle and is about to iterate it.', + jsDoc: '/**\n * One model request obtained its outer `llm/stream` handle and is about to\n * iterate it. This observes an Agent-loop request attempt, not proof that\n * provider I/O began. The notification is live, contained, and not replayed.\n * Preparation or a synchronous outer waterfall failure emits nothing;\n * failures or abortion after the handle returns still count.\n * @param agent - the agent dispatching the model request.\n * @param turn - the open turn number.\n * @param step - the request\'s step number.\n * @param request - final route plus registration-bound context capacity.\n * Scope-filtered dispatch (`@deepseek-ai/dsh-scope`): agent-scoped listeners receive only that agent.\n * @mode emit\n */', + summary: 'One model request obtained its outer `llm/stream` handle and is about to iterate it.', }, { name: 'agent/prompt-submit', @@ -1810,7 +1810,7 @@ export const TYPE_API: readonly TypeApiEntry[] = [ }, { name: 'PreparedLlmCall', - declaration: 'export interface PreparedLlmCall {\n readonly config: LlmCallConfig;\n readonly context?: LlmModelContext;\n stream(options: GenerateOptions, onDispatched?: () => void): AsyncIterable;\n}', + declaration: 'export interface PreparedLlmCall {\n readonly config: LlmCallConfig;\n readonly context?: LlmModelContext;\n stream(options: GenerateOptions): AsyncIterable;\n}', }, { name: 'PreparedReferencedMessage', diff --git a/packages/core/agent-loop/README.i18n.yaml b/packages/core/agent-loop/README.i18n.yaml index 8b72206a87..c6911a4c55 100644 --- a/packages/core/agent-loop/README.i18n.yaml +++ b/packages/core/agent-loop/README.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write packages/core/agent-loop/README.md -README.md: 39eaafd3abb2faef045c1a2f694d8dffe95c28b0 -README.zh.md: f8cc972e957fe95f652d54e859f8e5411288db51 +README.md: 6e23d9f543998d5c0266c73197ee52ede51260aa +README.zh.md: cf561a03281f80b7afd5f245752e9ebf70333a8b diff --git a/packages/core/agent-loop/README.md b/packages/core/agent-loop/README.md index 39eaafd3ab..6e23d9f543 100644 --- a/packages/core/agent-loop/README.md +++ b/packages/core/agent-loop/README.md @@ -62,7 +62,7 @@ The driver owns one agent for its lifetime and runs inside `ctx.agents.withIniti Every provider call that reaches a successful finish appends exactly one `assistant/message` completion anchor, including content-less calls and `max-tokens` finishes. The anchor records the assembled content as-is, retains exact chunk provenance (`[]` for a stream with no chunks), and includes usage when available; empty content stays out of derived message history. -After `agent/request` returns a provider/model call config, the loop asks `ctx.llm.prepareCall()` to validate any adapter-owned reasoning effort, materialize its configured default, and retain available context metadata from that same exact-model lookup under the active turn signal. The prepared call retains the exact adapter registration across this asynchronous resolution, `request/header` logging, and terminal dispatch, so HMR cannot mix one adapter's capability result with another adapter's request. After the final stream handle is constructed and before adapter iteration, the loop emits one contained live `agent/model-request` notification with turn, step, final provider/model, and optional registration-bound capacity. Preparation or synchronous stream-construction failures emit nothing; later failure or abortion remains an observed dispatch. The effective config is logged before dispatch, so a listener can change effort between steps without hidden request drift. A route with no registered adapter preserves the proposed config so an `llm/stream` listener can own and short-circuit it; unhandled terminal dispatch still fails with `NO_ADAPTER`. A new loop instance restores the last effort only when its initial provider/model route exactly matches the logged route; a route change discards that opaque model-owned ID and resolves the new model independently. +After `agent/request` returns a provider/model call config, the loop asks `ctx.llm.prepareCall()` to validate any adapter-owned reasoning effort, materialize its configured default, and retain available context metadata from that same exact-model lookup under the active turn signal. The prepared call retains the exact adapter registration across this asynchronous resolution, `request/header` logging, and terminal dispatch, so HMR cannot mix one adapter's capability result with another adapter's request. Once the final provider/model is fixed and the outer `llm/stream` call returns a handle, the loop emits one contained live `agent/model-request` notification with turn, step, route, and optional registration-bound capacity. This is an observed Agent-loop attempt, not proof of provider I/O: preparation or a synchronous outer waterfall failure emits nothing, while a short-circuit handle or later lazy adapter construction, failure, or abortion still counts. The effective config is logged before dispatch, so a listener can change effort between steps without hidden request drift. A route with no registered adapter preserves the proposed config so an `llm/stream` listener can own and short-circuit it; unhandled terminal dispatch still fails with `NO_ADAPTER`. A new loop instance restores the last effort only when its initial provider/model route exactly matches the logged route; a route change discards that opaque model-owned ID and resolves the new model independently. Plugin failure ends the current turn, not the loop. Only final adapter dispatch/iteration failures and terminal in-band error or aborted finishes enter `agent/request-error`; middleware, result processing, tools, and other extension failures close directly. Recovery receives the exact live error, immutable provider facts, immutable prior failures, the immutable retry policy of the adapter registration that served the request, and the turn signal after the failed step closes; the policy is absent if no final adapter served it. A handling listener returns `{ kind: 'retry' }`; the loop closes the failed turn with its error and opens one numbered retry turn without an intervening idle notification. Success clears the consecutive history, and an unhandled failure is terminal. AgentLoop owns one cancellation signal for the current admission or turn. An effective `cancel(cause)` clears pending work unless `keepInbox` is set and cooperatively aborts that signal; idle cancellation is a no-op. Durable `turn/end` records `aborted` for `user` and `parent`, while disposal records `disposed`; undispatched model tool calls receive synthetic `tool/call` and `ABORTED_BEFORE_DISPATCH` result pairs. The cancellation cause changes reporting, not how result context finalized after cancellation is handled. Disposal waits for signal-ignoring work before registry removal. The [explicit-cancellation decision](../../../.agents/notes/implemented/architecture/2026-07-16-explicit-turn-cancellation.md) owns the lifecycle and race contract. diff --git a/packages/core/agent-loop/README.zh.md b/packages/core/agent-loop/README.zh.md index f8cc972e95..cf561a0328 100644 --- a/packages/core/agent-loop/README.zh.md +++ b/packages/core/agent-loop/README.zh.md @@ -62,7 +62,7 @@ interface Config { 每次提供方调用成功结束时,都会恰好追加一个 `assistant/message` 完成锚点,包括无内容调用和以 `max-tokens` 结束的调用。该锚点原样记录组装后的内容,保留确切的 chunk 溯源(流没有 chunk 时为 `[]`),并在用量可用时包含用量;空内容不会进入派生消息历史。 -在 `agent/request` 返回提供方/模型调用配置后,循环会调用 `ctx.llm.prepareCall()`,在活跃轮次信号的控制下校验由适配器持有的推理(reasoning)强度、填入其配置默认值,并从同一次精确模型查询中保留可用的上下文元数据。准备完成的调用会在这次异步解析、`request/header` 日志记录和最终分派期间保留同一项确切的适配器注册,因此 HMR(热模块替换)不会把某个适配器的能力解析结果与另一适配器的请求混用。最终流句柄构造完成后、适配器开始迭代前,循环会发出一条失败会被收容的实时 `agent/model-request` 通知,其中包含轮次、步骤、最终提供方/模型,以及可选的、与注册项绑定的容量。准备阶段失败或同步流构造失败不会发出通知;之后即使失败或中止,该请求仍视为已观察到的分派。生效配置会在分派前写入日志,因此监听器可以在步骤之间更改推理强度,而不会产生未记录的请求变化。没有已注册适配器的路由会保留原定配置,使 `llm/stream` 监听器可以接管并短路该请求;最终分派仍会以 `NO_ADAPTER` 拒绝未得到处理的路由。新循环实例仅在初始提供方/模型路由与日志路由完全一致时恢复上次的推理强度;路由变化会丢弃由前一模型持有的不透明 ID,并单独解析新模型。 +在 `agent/request` 返回提供方/模型调用配置后,循环会调用 `ctx.llm.prepareCall()`,在活跃轮次信号的控制下校验由适配器持有的推理(reasoning)强度、填入其配置默认值,并从同一次精确模型查询中保留可用的上下文元数据。准备完成的调用会在这次异步解析、`request/header` 日志记录和最终分派期间保留同一项确切的适配器注册,因此 HMR(热模块替换)不会把某个适配器的能力解析结果与另一适配器的请求混用。最终提供方/模型确定且外层 `llm/stream` 调用返回句柄后,循环会发出一条失败会被收容的实时 `agent/model-request` 通知,其中包含轮次、步骤、路由,以及可选的、与注册项绑定的容量。这是 agent loop 观察到的一次尝试,并不能证明提供方 I/O 已开始:准备阶段或外层 waterfall(瀑布式事件)的同步失败不会发出通知,而短路句柄或之后的惰性适配器构造、失败或中止仍会计入。生效配置会在分派前写入日志,因此监听器可以在步骤之间更改推理强度,而不会产生未记录的请求变化。没有已注册适配器的路由会保留原定配置,使 `llm/stream` 监听器可以接管并短路该请求;最终分派仍会以 `NO_ADAPTER` 拒绝未得到处理的路由。新循环实例仅在初始提供方/模型路由与日志路由完全一致时恢复上次的推理强度;路由变化会丢弃由前一模型持有的不透明 ID,并单独解析新模型。 插件失败会结束当前轮次,而不是结束循环。只有最终适配器分发/迭代失败以及带内的终止错误或中止结束才进入 `agent/request-error`;中间件、结果处理、工具及其他扩展失败会直接关闭轮次。失败步骤关闭后,恢复逻辑会接收确切的实时错误、不可变的提供方事实、不可变的先前失败、为请求提供服务的适配器注册所对应的不可变重试策略,以及轮次信号;如果没有最终适配器为其提供服务,则该策略缺失。处理失败的监听器返回 `{ kind: 'retry' }`;循环用其错误关闭失败轮次,并在不插入空闲通知的情况下开启一个编号重试轮次。成功会清除连续失败历史;未被处理的失败是终态。AgentLoop 为当前接纳或轮次拥有一个取消信号。有效的 `cancel(cause)` 在未设置 `keepInbox` 时清除待处理工作,并以协作方式中止该信号;空闲取消是空操作。持久 `turn/end` 为 `user` 和 `parent` 记录 `aborted`,dispose(资源释放)则记录 `disposed`;未分发的模型工具调用会收到合成的 `tool/call` 与 `ABORTED_BEFORE_DISPATCH` 结果对。取消原因只改变报告方式,不改变对取消后已定案结果上下文的处理。dispose 会等待忽略信号的工作完成,然后才从注册表移除。[显式取消决策](../../../.agents/notes/implemented/architecture/2026-07-16-explicit-turn-cancellation.md)规定生命周期与竞态契约。 @@ -89,7 +89,7 @@ interface Config { #### Token 影响 -每个步骤都会再次计入系统文本与 schema。逐 agent 作用域决定贡献,而权威组装 waterfall(瀑布式事件)可以改变最终请求,并使其监听器负责保持协议连贯。 +每个步骤都会再次计入系统文本与 schema。逐 agent 作用域决定贡献,而权威组装 waterfall 可以改变最终请求,并使其监听器负责保持协议连贯。 #### KV Cache 影响 diff --git a/packages/core/agent-loop/src/agent.ts b/packages/core/agent-loop/src/agent.ts index 9c930912ba..65f3fe4b2a 100644 --- a/packages/core/agent-loop/src/agent.ts +++ b/packages/core/agent-loop/src/agent.ts @@ -487,24 +487,21 @@ export class ReactLoopAgent implements Agent { const assembler = new BlockAssembler() const chunkSeqs: number[] = [] - const onDispatched = (): void => { - emitAgentEvent( - this.loopCtx, - this, - 'agent/model-request', - turn, - step, - { - provider: request.provider, - model: request.model, - ...preparedCall?.context === undefined - ? {} - : { contextWindow: preparedCall.context.contextWindow }, - }, - ) - } - const stream = preparedCall?.stream(request, onDispatched) - ?? this.loopCtx.llm.stream(request, onDispatched) + const stream = preparedCall?.stream(request) ?? this.loopCtx.llm.stream(request) + emitAgentEvent( + this.loopCtx, + this, + 'agent/model-request', + turn, + step, + { + provider: request.provider, + model: request.model, + ...preparedCall?.context === undefined + ? {} + : { contextWindow: preparedCall.context.contextWindow }, + }, + ) try { for await (const chunk of stream) { signal.throwIfAborted() diff --git a/packages/core/agent-loop/tests/request-reconstruction.spec.ts b/packages/core/agent-loop/tests/request-reconstruction.spec.ts index 9da0e30f81..51ddc4decf 100644 --- a/packages/core/agent-loop/tests/request-reconstruction.spec.ts +++ b/packages/core/agent-loop/tests/request-reconstruction.spec.ts @@ -293,6 +293,7 @@ describe('request stability across the loop', () => { await ctx.plugin(AgentRegistry) await ctx.plugin(AgentLoop, { agents: [] }) let observed: GenerateOptions | undefined + let observedRequest: { provider: string; model: string; contextWindow?: number } | undefined ctx.on('llm/stream', (options) => { observed = options return (async function* () { @@ -303,11 +304,15 @@ describe('request stability across the loop', () => { provider: 'listener', model: 'virtual', }) + ctx.on('agent/model-request', (subject, _turn, _step, request) => { + if (subject === agent) observedRequest = { ...request } + }) send(agent, 'go') await waitForIdle(ctx, agent) expect(observed).toMatchObject({ provider: 'listener', model: 'virtual' }) + expect(observedRequest).toEqual({ provider: 'listener', model: 'virtual' }) expect(agent.session.requestHeader()?.config).toEqual({ provider: 'listener', model: 'virtual', @@ -318,7 +323,7 @@ describe('request stability across the loop', () => { }) }) - it('notifies one contained live model-request edge only after successful stream construction', async () => { + it('notifies one contained request attempt after the outer stream handle returns', async () => { const ctx = new Context() await ctx.plugin(LlmService) await ctx.plugin(SessionStore) @@ -341,7 +346,9 @@ describe('request stability across the loop', () => { } override stream(options: GenerateOptions): AsyncIterable { - if (options.model === 'sync-failure') throw new LlmError('construction failed', 'CONSTRUCTION') + if (options.model === 'lazy-sync-failure') { + throw new LlmError('lazy construction failed', 'CONSTRUCTION') + } if (options.model === 'async-failure') { return { [Symbol.asyncIterator]: () => ({ @@ -359,6 +366,22 @@ describe('request stability across the loop', () => { provider: 'mock', model: 'capacity', }) + const returnedHandles = new Set() + ctx.on('llm/stream', (options, next) => { + if (options.model === 'outer-failure') { + throw new Error('outer waterfall failed before returning a handle') + } + if (options.model === 'lazy-sync-failure') { + const stream = (async function* () { + yield* next() + })() + returnedHandles.add(options.model) + return stream + } + const stream = next() + returnedHandles.add(options.model) + return stream + }) const observed: { turn: number step: number @@ -366,18 +389,27 @@ describe('request stability across the loop', () => { model: string contextWindow?: number }[] = [] + const observedBeforeHandleReturn: string[] = [] ctx.on('agent/model-request', (subject) => { if (subject === agent) throw new Error('observer failed') }) ctx.on('agent/model-request', (subject, turn, step, request) => { - if (subject === agent) observed.push({ turn, step, ...request }) + if (subject !== agent) return + if (!returnedHandles.has(request.model)) observedBeforeHandleReturn.push(request.model) + observed.push({ turn, step, ...request }) }) ctx.on('agent/request', async (_subject, turn, _step, _signal, next) => ({ ...await next(), - model: ['capacity', 'unknown', 'async-failure', 'sync-failure'][turn - 1]!, + model: [ + 'capacity', + 'unknown', + 'async-failure', + 'lazy-sync-failure', + 'outer-failure', + ][turn - 1]!, })) - for (const prompt of ['one', 'two', 'three', 'four']) { + for (const prompt of ['one', 'two', 'three', 'four', 'five']) { send(agent, prompt) await waitForIdle(ctx, agent) } @@ -402,8 +434,16 @@ describe('request stability across the loop', () => { provider: 'mock', model: 'async-failure', }, + { + turn: 4, + step: 1, + provider: 'mock', + model: 'lazy-sync-failure', + }, ]) - expect(resolutions).toBe(4) + expect(resolutions).toBe(5) + expect(observedBeforeHandleReturn).toEqual([]) + expect(returnedHandles.has('outer-failure')).toBe(false) }) it('a compaction replace rewrites the resend, and the log explains it', async () => { diff --git a/packages/core/agent/README.i18n.yaml b/packages/core/agent/README.i18n.yaml index 00465d7021..0d537a147a 100644 --- a/packages/core/agent/README.i18n.yaml +++ b/packages/core/agent/README.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write packages/core/agent/README.md -README.md: 304347d32df4546389ee2b45230d4e80acc60942 -README.zh.md: d9adbe20015aadbad26288d43df92f80a3f550bf +README.md: ad79e101974eb187df3093092a282fc6e5850857 +README.zh.md: 84849df23dfdc668f0d18c05ed6d9fd4d8c315d6 diff --git a/packages/core/agent/README.md b/packages/core/agent/README.md index 304347d32d..ad79e10197 100644 --- a/packages/core/agent/README.md +++ b/packages/core/agent/README.md @@ -48,7 +48,7 @@ Agent *creation* is provided by the plugin implementing `AgentFactory` (`dsh-age The lifecycle edges have two important local caveats. `agent/created` runs after scoped setup and after both session and agent registry entries exist. Setup is trusted composition-only code; the immediately following non-vetoing `agent/session-start` notification is the first supported startup injection point. `agent/disposed` always means the exact agent has left the registry. AgentLoop emits it after its driver is quiescent, while ordered teardown may still be detaching the session and unwinding the scope; custom agents registered directly own any stronger driver-ordering contract themselves. -Most interception points are cooperative waterfalls. Turn-scoped asynchronous seams receive one explicit `AbortSignal`, with `signal` immediately before a waterfall's final `next`; listeners may cooperate but must not retain it as authority over another turn. `agent/step` is the serial checkpoint before request derivation, while the contained `agent/model-request` notification reports a final dispatched route and optional registration-bound context capacity without becoming durable state. `agent/request-error` is the failed-model-request recovery waterfall: it receives the exact error, normalized failure facts, and signal after the failed step closes. A listener returns `{ kind: 'retry' }` without calling `next()` when it owns recovery; the loop closes the failed turn and opens one numbered retry turn. `agent/turn-stopping` runs before an otherwise completed turn closes. Ordinary queued prompts remain intact. Effective broad cancellation first emits the observe-only `agent/cancel-requested` with its resolved typed cause, then clears queues and aborts; notification failures are contained and cannot veto the stop. The [explicit-cancellation decision](../../../.agents/notes/implemented/architecture/2026-07-16-explicit-turn-cancellation.md) owns signal lifetime; the [agent-scope runtime-design Agent Note](../../../.agents/notes/implemented/architecture/2026-07-12-agent-scope-runtime-design.md#three-execution-boundaries-are-deliberately-one-way) owns scoped dispatch and terminal settlement. +Most interception points are cooperative waterfalls. Turn-scoped asynchronous seams receive one explicit `AbortSignal`, with `signal` immediately before a waterfall's final `next`; listeners may cooperate but must not retain it as authority over another turn. `agent/step` is the serial checkpoint before request derivation, while the contained `agent/model-request` notification reports the route and optional registration-bound context capacity for an attempt whose outer stream handle returned. It is neither durable state nor proof of provider I/O. `agent/request-error` is the failed-model-request recovery waterfall: it receives the exact error, normalized failure facts, and signal after the failed step closes. A listener returns `{ kind: 'retry' }` without calling `next()` when it owns recovery; the loop closes the failed turn and opens one numbered retry turn. `agent/turn-stopping` runs before an otherwise completed turn closes. Ordinary queued prompts remain intact. Effective broad cancellation first emits the observe-only `agent/cancel-requested` with its resolved typed cause, then clears queues and aborts; notification failures are contained and cannot veto the stop. The [explicit-cancellation decision](../../../.agents/notes/implemented/architecture/2026-07-16-explicit-turn-cancellation.md) owns signal lifetime; the [agent-scope runtime-design Agent Note](../../../.agents/notes/implemented/architecture/2026-07-12-agent-scope-runtime-design.md#three-execution-boundaries-are-deliberately-one-way) owns scoped dispatch and terminal settlement. `PromptDecision.additionalContexts` is an array so every context keeps its own source. Allowed prompt content and every additional context become separate model-facing `user/message` events before the turn runs. A listener that wraps a downstream allow preserves its `content` and `additionalContexts` unless it intentionally replaces either field; the returned allow is authoritative. diff --git a/packages/core/agent/README.zh.md b/packages/core/agent/README.zh.md index d9adbe2001..84849df23d 100644 --- a/packages/core/agent/README.zh.md +++ b/packages/core/agent/README.zh.md @@ -48,7 +48,7 @@ Agent *创建* 由实现 `AgentFactory` 的插件(`dsh-agent-loop`)提供, 生命周期边有两个重要的本地注意事项。`agent/created` 在作用域 setup 之后、会话与 agent 注册表条目都存在之后运行。Setup 是受信任、仅用于组合的代码;紧随其后且不可 veto 的 `agent/session-start` 通知是第一个受支持的启动注入点。`agent/disposed` 始终表示确切 agent 已离开注册表。AgentLoop 在其驱动器静默后发出该事件,而有序 teardown 此时可能仍在分离会话并撤销作用域;直接注册的自定义 agent 自行拥有任何更强的驱动器顺序契约。 -大多数拦截点都是协作式 waterfall。轮次作用域的异步 seam 接收一个显式 `AbortSignal`,其中 `signal` 紧邻 waterfall 最终的 `next`;监听器可以配合,但不得将它保留为控制另一轮次的权限。`agent/step` 是派生请求前的串行检查点;`agent/model-request` 是失败会被收容的通知,它会报告最终已分派路由及可选的、与注册项绑定的上下文容量,但不会成为持久状态。`agent/request-error` 是失败模型请求的恢复 waterfall:失败步骤关闭后,它接收确切错误、规范化失败事实和信号。拥有恢复权的监听器返回 `{ kind: 'retry' }` 且不调用 `next()`;循环会关闭失败轮次,并打开一个编号重试轮次。`agent/turn-stopping` 在本可完成的轮次关闭前运行。普通排队提示词保持原样。有效的广义取消会先发出只观测的 `agent/cancel-requested` 及其解析后的类型化原因,再清空队列并中止;通知失败会被收容,不能 veto 停止。信号生命周期由[显式取消决策](../../../.agents/notes/implemented/architecture/2026-07-16-explicit-turn-cancellation.md)拥有;作用域分发与终止结算由 [agent 作用域 runtime 设计 Agent Note](../../../.agents/notes/implemented/architecture/2026-07-12-agent-scope-runtime-design.md#three-execution-boundaries-are-deliberately-one-way)拥有。 +大多数拦截点都是协作式 waterfall。轮次作用域的异步 seam 接收一个显式 `AbortSignal`,其中 `signal` 紧邻 waterfall 最终的 `next`;监听器可以配合,但不得将它保留为控制另一轮次的权限。`agent/step` 是派生请求前的串行检查点;`agent/model-request` 是失败会被收容的通知,它会报告外层流句柄已返回的尝试所用路由,以及可选的、与注册项绑定的上下文容量。它既不是持久状态,也不能证明提供方 I/O 已开始。`agent/request-error` 是失败模型请求的恢复 waterfall:失败步骤关闭后,它接收确切错误、规范化失败事实和信号。拥有恢复权的监听器返回 `{ kind: 'retry' }` 且不调用 `next()`;循环会关闭失败轮次,并打开一个编号重试轮次。`agent/turn-stopping` 在本可完成的轮次关闭前运行。普通排队提示词保持原样。有效的广义取消会先发出只观测的 `agent/cancel-requested` 及其解析后的类型化原因,再清空队列并中止;通知失败会被收容,不能 veto 停止。信号生命周期由[显式取消决策](../../../.agents/notes/implemented/architecture/2026-07-16-explicit-turn-cancellation.md)拥有;作用域分发与终止结算由 [agent 作用域 runtime 设计 Agent Note](../../../.agents/notes/implemented/architecture/2026-07-12-agent-scope-runtime-design.md#three-execution-boundaries-are-deliberately-one-way)拥有。 `PromptDecision.additionalContexts` 是数组,因此每个上下文都保留自己的来源。获准的提示词内容与每个附加上下文都会在轮次运行前成为各自独立、面向模型的 `user/message` 事件。包装下游允许决策的监听器会保留其 `content` 与 `additionalContexts`,除非有意替换任一字段;返回的允许决策是权威来源。 diff --git a/packages/core/agent/src/types.ts b/packages/core/agent/src/types.ts index 5aac3152b2..f7b1447352 100644 --- a/packages/core/agent/src/types.ts +++ b/packages/core/agent/src/types.ts @@ -116,13 +116,13 @@ export type PromptDecision = /** Model-request failure with an optional machine-routable provider code. */ export type RequestError = Error & { code?: string } -/** Live metadata for one model request that reached adapter dispatch. */ +/** Live metadata for one model request whose outer stream handle was obtained. */ export interface AgentModelRequest { - /** Final registered provider route. */ + /** Final request provider route; a short-circuit listener may own it. */ readonly provider: string - /** Final adapter-owned model id. */ + /** Final request model id; a short-circuit listener may own it. */ readonly model: string - /** Registration-bound context capacity when the adapter exposed one. */ + /** Registration-bound context capacity when preparation exposed one. */ readonly contextWindow?: number } @@ -371,11 +371,11 @@ declare module 'cordis' { */ 'agent/request'(this: Scoped, agent: Agent, turn: number, step: number, signal: AbortSignal, next: () => Promise): Promise /** - * One model request constructed its final stream handle and is about to - * iterate it. This live notification is not durable or replayed; failed or - * aborted iteration still has a dispatch, while preparation and - * synchronous stream-construction failures do not. Listener failures are - * contained and cannot affect the request. + * One model request obtained its outer `llm/stream` handle and is about to + * iterate it. This observes an Agent-loop request attempt, not proof that + * provider I/O began. The notification is live, contained, and not replayed. + * Preparation or a synchronous outer waterfall failure emits nothing; + * failures or abortion after the handle returns still count. * @param agent - the agent dispatching the model request. * @param turn - the open turn number. * @param step - the request's step number. diff --git a/packages/host/apiproxy/README.i18n.yaml b/packages/host/apiproxy/README.i18n.yaml index 2bab687635..681058233b 100644 --- a/packages/host/apiproxy/README.i18n.yaml +++ b/packages/host/apiproxy/README.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write packages/host/apiproxy/README.md -README.md: d3d1711242f71832f63eb570242fdf9e149cc988 -README.zh.md: 4e52058c1795ea23a9ca8ed46b8890c85f129eb4 +README.md: 263cb2f661e2f5b91cc4885ac99fe3afe61bbfbc +README.zh.md: 400031b7d6d668d6ec7d2922b8a6abb855fe5b7b diff --git a/packages/host/apiproxy/README.md b/packages/host/apiproxy/README.md index d3d1711242..263cb2f661 100644 --- a/packages/host/apiproxy/README.md +++ b/packages/host/apiproxy/README.md @@ -22,7 +22,7 @@ Workspace and Session lists are separate reconnect baselines. `workspace.create` `session.history` pages on message boundaries, and its tail page (no `beforeSeq`) carries session-level projections the page window cannot supply: the in-flight partial's chunk events; `todos`, the latest `todo/write` whole-list projection; and `metrics`, full-log usage deduplicated by `(turn, step)` plus current token-meter pressure. Older pages omit the session-level projections. Live `session/metrics` mux frames carry monotonic log/projection revisions, so clients reject stale frames and preserve the counters while prepending older pages. Cache reads and writes remain disjoint buckets; the cache-hit denominator is uncached input plus cache reads. -Context capacity uses a distinct transient `session/model-request` mux frame emitted from the contained Agent notification after an actual request reaches dispatch. It carries turn, step, final provider/model, and optional capacity only to mux connections already open at that instant. `session.history`, mux subscription baselines, reconnects, and session restore never query or replay prior capacity; a frame without capacity explicitly clears the earlier connection-local value. +Context capacity uses a distinct transient `session/model-request` mux frame emitted from the contained Agent notification after an observed request attempt returns its outer stream handle. This boundary does not prove provider I/O began. The frame carries turn, step, final provider/model, and optional capacity only to mux connections already open at that instant. `session.history`, mux subscription baselines, reconnects, and session restore never query or replay prior capacity; a frame without capacity explicitly clears the earlier connection-local value. The `command.*` and `skill.*` domains expose the host command registry and skill catalog to clients. Every method addresses one session's agent by `sessionId` (a served session always has an Agent; `command.*` resumes cold sessions through the same path as `session.*`, while `skill.list` resolves the project root from the session header without touching the Agent registry). `command.execute` runs a slash-command line host-side and returns a detached result; the carrier's request signal cancels the running handler. `host/commands-changed` is the catalog invalidation frame: clients refetch `command.list` instead of diffing. diff --git a/packages/host/apiproxy/README.zh.md b/packages/host/apiproxy/README.zh.md index 4e52058c17..400031b7d6 100644 --- a/packages/host/apiproxy/README.zh.md +++ b/packages/host/apiproxy/README.zh.md @@ -22,7 +22,7 @@ Workspace 列表与 Session 列表是相互独立的重连基线。`workspace.cr `session.history` 按消息边界分页,其尾页(不带 `beforeSeq`)携带页窗口本身无法提供的会话级投影:进行中局部消息的分片事件;`todos`,即最后一次 `todo/write` 的整表投影;以及 `metrics`,即按 `(turn, step)` 去重的完整日志用量与当前 token 计量压力。较早的页面省略会话级投影。实时 `session/metrics` mux 帧携带单调递增的日志修订号与投影修订号,因此客户端会拒绝陈旧帧,并在向前加载较早页面时保留计数器。缓存读取与缓存写入保持为彼此独立的计数项;缓存命中率的分母是未缓存输入加缓存读取。 -上下文容量使用独立的临时 `session/model-request` mux 帧;实际请求到达分派点后,该帧由失败会被收容的 Agent 通知发出。该帧携带轮次、步骤、最终提供方/模型与可选容量,且只发送给当时已经打开的 mux 连接。`session.history`、mux 订阅基线、重连和会话恢复绝不会查询或回放先前的容量;不带容量的帧会显式清除较早的连接本地值。 +上下文容量使用独立的临时 `session/model-request` mux 帧;观察到的请求尝试返回外层流句柄后,该帧由失败会被收容的 Agent 通知发出。这个边界不能证明提供方 I/O 已开始。该帧携带轮次、步骤、最终提供方/模型与可选容量,且只发送给当时已经打开的 mux 连接。`session.history`、mux 订阅基线、重连和会话恢复绝不会查询或回放先前的容量;不带容量的帧会显式清除较早的连接本地值。 `command.*` 与 `skill.*` 领域向客户端暴露宿主命令注册表和技能目录。每个方法都通过 `sessionId` 寻址一个会话的 Agent(被服务的会话必有 Agent;`command.*` 经由与 `session.*` 相同的路径恢复冷会话,而 `skill.list` 从会话头解析项目根目录,不触碰 Agent 注册表)。`command.execute` 在宿主侧运行一条斜杠命令行并返回脱耦结果;载体的请求信号可取消正在运行的处理器。`host/commands-changed` 是目录失效帧:客户端重新拉取 `command.list` 而不是做差分。 diff --git a/packages/host/apiproxy/src/api/events.ts b/packages/host/apiproxy/src/api/events.ts index c4babb6017..2eec50c257 100644 --- a/packages/host/apiproxy/src/api/events.ts +++ b/packages/host/apiproxy/src/api/events.ts @@ -60,11 +60,11 @@ export type MuxFrame = | { type: 'session/subscribed'; sessionId: SessionId; lastSeq: number } | { type: 'session/metrics'; sessionId: SessionId; metrics: SessionMetrics } /** - * One model request observed by this already-open mux connection after its - * final route and stream handle were resolved. This frame is transient: mux - * baselines, reconnects, and session history never replay it. An absent - * `contextWindow` explicitly clears a capacity observed from an earlier - * request on the same connection. + * One request attempt observed by this already-open mux connection after its + * final route and outer `llm/stream` handle were obtained. This does not prove + * provider I/O began. The frame is transient: mux baselines, reconnects, and + * session history never replay it. An absent `contextWindow` explicitly clears + * a capacity observed from an earlier request on the same connection. */ | { type: 'session/model-request' diff --git a/packages/llm/llm/README.i18n.yaml b/packages/llm/llm/README.i18n.yaml index 6ef87e4ab8..fd8147dc3f 100644 --- a/packages/llm/llm/README.i18n.yaml +++ b/packages/llm/llm/README.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write packages/llm/llm/README.md -README.md: d28a5632a3fbdbf11c7dba2ee0c57a704f2ba6f4 -README.zh.md: fd93fa43d5bfabd6e8751d4efbad229096ced08d +README.md: 5d0459b722c4c5f472231ee86e775bbc4ff7b7f2 +README.zh.md: 3dd7153d63c8c4b40b737ed58d5d6d1a29f3153a diff --git a/packages/llm/llm/README.md b/packages/llm/llm/README.md index d28a5632a3..5d0459b722 100644 --- a/packages/llm/llm/README.md +++ b/packages/llm/llm/README.md @@ -25,7 +25,7 @@ Provider and model metadata is a discovery surface, not a routing whitelist. `re Exact-model metadata is a separate correctness query, not a catalog decoration or global LLM setting. `resolveModelInfo()` asks the adapter that owns the exact provider/model route once; an adapter can describe an unlisted dynamic model, and absent `context` or `reasoning` fields mean only that those capabilities are unavailable. Invalid identity, context, or reasoning metadata fails with `INVALID_MODEL_INFO`, `INVALID_MODEL_CONTEXT`, or `INVALID_MODEL_REASONING`. -Reasoning identifiers are opaque adapter-owned strings rather than a core enum. An adapter publishes its ordered selectable list, including an `off` id when that model's capability API exposes one. `resolveCallConfig()` accepts only an exact advertised identifier, materializes `defaultEffort` when present, and otherwise preserves the provider default. Asynchronous model resolvers receive the caller's signal and must settle promptly after cancellation. `prepareCall()` additionally exposes the detached context metadata from that same lookup and retains the exact adapter registration through header logging and terminal dispatch, so HMR cannot combine one adapter's capability result with another adapter's request; reusing its one-shot handle or changing its call-config fields fails with `INVALID_PREPARED_CALL`. Its dispatch observer runs after a final stream handle is constructed and before adapter iteration. An unsupported explicit or configured effort fails with `UNSUPPORTED_REASONING_EFFORT` before provider I/O. +Reasoning identifiers are opaque adapter-owned strings rather than a core enum. An adapter publishes its ordered selectable list, including an `off` id when that model's capability API exposes one. `resolveCallConfig()` accepts only an exact advertised identifier, materializes `defaultEffort` when present, and otherwise preserves the provider default. Asynchronous model resolvers receive the caller's signal and must settle promptly after cancellation. `prepareCall()` additionally exposes the detached context metadata from that same lookup and retains the exact adapter registration through header logging and terminal dispatch, so HMR cannot combine one adapter's capability result with another adapter's request; reusing its one-shot handle or changing its call-config fields fails with `INVALID_PREPARED_CALL`. An unsupported explicit or configured effort fails with `UNSUPPORTED_REASONING_EFFORT` before provider I/O. ### Events diff --git a/packages/llm/llm/README.zh.md b/packages/llm/llm/README.zh.md index fd93fa43d5..3dd7153d63 100644 --- a/packages/llm/llm/README.zh.md +++ b/packages/llm/llm/README.zh.md @@ -25,7 +25,7 @@ 确切模型元数据是独立的正确性查询,不是 catalog 装饰或全局 LLM 设置。`resolveModelInfo()` 会向拥有精确提供方/模型路由的适配器查询一次;适配器可以描述未列出的动态模型,缺少 `context` 或 `reasoning` 字段只表示相应能力不可用。无效的身份、上下文或推理元数据会以 `INVALID_MODEL_INFO`、`INVALID_MODEL_CONTEXT` 或 `INVALID_MODEL_REASONING` 失败。 -推理标识符是由适配器持有的不透明字符串,而非核心枚举。适配器会公布有序可选列表;模型能力 API 提供 `off` id 时,列表也会包含它。`resolveCallConfig()` 只接受与已公布标识符完全一致的值,在存在 `defaultEffort` 时填入它,否则保留提供方默认值。异步模型解析器会接收调用方的 signal,并且必须在取消后迅速完成结算。`prepareCall()` 还会公开同一次查询得到的脱耦上下文元数据,并让精确适配器注册跨越请求头记录和最终分派,因此 HMR(热模块替换)不会将一个适配器的能力结果与另一个适配器的请求混用;复用其一次性句柄或更改调用配置字段会以 `INVALID_PREPARED_CALL` 失败。其分派观察器在最终流句柄构造完成后、适配器开始迭代前运行。不支持的显式或配置推理强度会在提供方 I/O 前以 `UNSUPPORTED_REASONING_EFFORT` 失败。 +推理标识符是由适配器持有的不透明字符串,而非核心枚举。适配器会公布有序可选列表;模型能力 API 提供 `off` id 时,列表也会包含它。`resolveCallConfig()` 只接受与已公布标识符完全一致的值,在存在 `defaultEffort` 时填入它,否则保留提供方默认值。异步模型解析器会接收调用方的 signal,并且必须在取消后迅速完成结算。`prepareCall()` 还会公开同一次查询得到的脱耦上下文元数据,并让精确适配器注册跨越请求头记录和最终分派,因此 HMR(热模块替换)不会将一个适配器的能力结果与另一个适配器的请求混用;复用其一次性句柄或更改调用配置字段会以 `INVALID_PREPARED_CALL` 失败。不支持的显式或配置推理强度会在提供方 I/O 前以 `UNSUPPORTED_REASONING_EFFORT` 失败。 ### 事件 diff --git a/packages/llm/llm/src/index.ts b/packages/llm/llm/src/index.ts index 12bd2ec6ea..8120ac5085 100644 --- a/packages/llm/llm/src/index.ts +++ b/packages/llm/llm/src/index.ts @@ -119,11 +119,9 @@ export interface PreparedLlmCall { * preparation. The request's call-config fields must match {@link config}; * reuse or mismatch fails with `INVALID_PREPARED_CALL`. * @param options - fully assembled request carrying the prepared config. - * @param onDispatched - contained Agent-loop notification hook invoked after - * a stream handle is constructed and before its adapter is iterated. * @returns the chunk stream, including the `llm/stream` waterfall. */ - stream(options: GenerateOptions, onDispatched?: () => void): AsyncIterable + stream(options: GenerateOptions): AsyncIterable } /** @@ -408,6 +406,7 @@ export class LlmService extends Service { const resolved = await this.resolveModelInfoFor(registration, config.model, signal) const reasoning = resolved.reasoning const requested = config.reasoningEffort + let resolvedConfig = config if (reasoning === undefined) { if (requested !== undefined) { throw new LlmError( @@ -415,26 +414,20 @@ export class LlmService extends Service { 'UNSUPPORTED_REASONING_EFFORT', ) } - return { - config, - ...resolved.context === undefined ? {} : { context: resolved.context }, + } else { + const effective = requested ?? reasoning.defaultEffort + if (effective !== undefined) { + if (!reasoning.efforts.some(effort => effort.id === effective)) { + throw new LlmError( + `provider "${config.provider}" model "${config.model}" does not support reasoning effort "${effective}"`, + 'UNSUPPORTED_REASONING_EFFORT', + ) + } + if (requested !== effective) resolvedConfig = { ...config, reasoningEffort: effective } } } - const effective = requested ?? reasoning.defaultEffort - if (effective === undefined) { - return { - config, - ...resolved.context === undefined ? {} : { context: resolved.context }, - } - } - if (!reasoning.efforts.some(effort => effort.id === effective)) { - throw new LlmError( - `provider "${config.provider}" model "${config.model}" does not support reasoning effort "${effective}"`, - 'UNSUPPORTED_REASONING_EFFORT', - ) - } return { - config: requested === effective ? config : { ...config, reasoningEffort: effective }, + config: resolvedConfig, ...resolved.context === undefined ? {} : { context: resolved.context }, } } @@ -458,16 +451,12 @@ export class LlmService extends Service { return Object.freeze({ config: resolvedConfig, ...context === undefined ? {} : { context }, - stream: (options: GenerateOptions, onDispatched?: () => void): AsyncIterable => { + stream: (options: GenerateOptions): AsyncIterable => { if (dispatched) { throw new LlmError('a prepared LLM call can only be dispatched once', 'INVALID_PREPARED_CALL') } dispatched = true - return this.streamWithRegistration( - options, - { registration, config: resolvedConfig }, - onDispatched, - ) + return this.streamWithRegistration(options, { registration, config: resolvedConfig }) }, }) } @@ -502,50 +491,24 @@ export class LlmService extends Service { * so it cannot suppress the primary provider error. A downstream close awaits * adapter cleanup, whose failures remain ordinary untagged work. */ - private adapterStream( + private async * adapterStream( options: GenerateOptions, failures: AdapterFailureScope, prepared?: { registration: AdapterRegistration; config: LlmCallConfig }, - onDispatched?: () => void, - ): AsyncIterable { - if (prepared === undefined) { - return this.resolveAndStream(options, failures, onDispatched) - } + ): AsyncGenerator { let iterator: AsyncIterator try { - const registration = prepared.registration + const registration = prepared?.registration ?? this.registration(options.provider) failures.retryPolicy = registration.retryPolicy - const resolvedConfig = prepared.config - if (!callConfigEquals(options, resolvedConfig)) { + const resolvedConfig = prepared === undefined + ? (await this.resolveCallFor(registration, options, options.signal)).config + : prepared.config + if (prepared !== undefined && !callConfigEquals(options, resolvedConfig)) { throw new LlmError( 'prepared LLM call config changed before adapter dispatch', 'INVALID_PREPARED_CALL', ) } - const adapter = registration.adapter - const stream = adapter.stream(this.forAdapter(options, adapter)) - iterator = stream[Symbol.asyncIterator]() - } catch (error: unknown) { - return this.failedAdapterStream(markLlmAdapterFailure(failures, error)) - } - this.notifyDispatched(onDispatched) - return this.iterateAdapter(iterator, failures) - } - - private async * resolveAndStream( - options: GenerateOptions, - failures: AdapterFailureScope, - onDispatched?: () => void, - ): AsyncGenerator { - let iterator: AsyncIterator - try { - const registration = this.registration(options.provider) - failures.retryPolicy = registration.retryPolicy - const resolvedConfig = (await this.resolveCallFor( - registration, - options, - options.signal, - )).config const resolvedOptions = callConfigEquals(options, resolvedConfig) ? options : Object.isFrozen(options) @@ -557,19 +520,7 @@ export class LlmService extends Service { } catch (error: unknown) { throw markLlmAdapterFailure(failures, error) } - this.notifyDispatched(onDispatched) - yield* this.iterateAdapter(iterator, failures) - } - private async * failedAdapterStream(error: Error): AsyncGenerator { - await Promise.resolve() - throw error - } - - private async * iterateAdapter( - iterator: AsyncIterator, - failures: AdapterFailureScope, - ): AsyncGenerator { let completed = false let iterationFailed = false try { @@ -599,15 +550,6 @@ export class LlmService extends Service { } } - private notifyDispatched(onDispatched: (() => void) | undefined): void { - if (onDispatched === undefined) return - try { - onDispatched() - } catch (error: unknown) { - this.ctx.logger.warn(`llm dispatch observer threw: ${String(error)}`) - } - } - /** * Stream one model call as raw chunks (token-level deltas). Throws * `LlmError` with code `NO_ADAPTER` if no adapter is registered for @@ -619,32 +561,23 @@ export class LlmService extends Service { * agent-loop request recovery; middleware and nested-call failures remain * untagged for the outer call. * @param options - the full request; `options.provider` selects the adapter. - * @param onDispatched - contained Agent-loop notification hook invoked after - * a stream handle is constructed and before its adapter is iterated. * @returns the chunk stream, possibly wrapped by `llm/stream` listeners. */ - stream(options: GenerateOptions, onDispatched?: () => void): AsyncIterable { - return this.streamWithRegistration(options, undefined, onDispatched) + stream(options: GenerateOptions): AsyncIterable { + return this.streamWithRegistration(options) } private streamWithRegistration( options: GenerateOptions, prepared?: { registration: AdapterRegistration; config: LlmCallConfig }, - onDispatched?: () => void, ): AsyncIterable { const failures: AdapterFailureScope = { failures: new WeakMap() } - let terminalEntered = false const stream = this.ctx.waterfall( this, 'llm/stream', options, - () => { - terminalEntered = true - return this.adapterStream(options, failures, prepared, onDispatched) - }, + () => this.adapterStream(options, failures, prepared), ) - // eslint-disable-next-line @typescript-eslint/no-unnecessary-condition -- waterfall mutates this latch. - if (!terminalEntered) this.notifyDispatched(onDispatched) return bindAdapterFailureScope(stream, failures) } } diff --git a/packages/llm/llm/tests/service.spec.ts b/packages/llm/llm/tests/service.spec.ts index 85d07be6fe..b0372b298e 100644 --- a/packages/llm/llm/tests/service.spec.ts +++ b/packages/llm/llm/tests/service.spec.ts @@ -1070,11 +1070,14 @@ describe('LlmService', () => { provider, id: model, name: model, + description: 'Resolved model', context: source, - reasoning: { - efforts: [{ id: ReasoningEffortId('high'), name: 'High' }], - defaultEffort: ReasoningEffortId('high'), - }, + reasoning: model === 'no-default' + ? { efforts: [{ id: ReasoningEffortId('high'), name: 'High' }] } + : { + efforts: [{ id: ReasoningEffortId('high'), name: 'High' }], + defaultEffort: ReasoningEffortId('high'), + }, }) } }(SCRIPT) @@ -1090,6 +1093,11 @@ describe('LlmService', () => { messages: [], })) { /* drain */ } expect(resolutions).toBe(1) + + const noDefault = await ctx.llm.prepareCall({ provider: 'route', model: 'no-default' }) + expect(noDefault.config).toEqual({ provider: 'route', model: 'no-default' }) + expect(noDefault.context).toEqual({ contextWindow: 64_000 }) + expect(resolutions).toBe(2) }) it('passes cancellation through exact-model resolution', async () => { From fc6eb4e7a3598c1380f0dd7200f8a043fc9bb2a6 Mon Sep 17 00:00:00 2001 From: Hypatia May Date: Tue, 28 Jul 2026 19:44:29 +0800 Subject: [PATCH 13/38] fix(web): retain live capacity before session creation --- .../runtime/src/client/sessions/manager.ts | 16 +++ .../runtime/src/client/sessions/session.ts | 9 +- packages/client/runtime/tests/manager.spec.ts | 121 ++++++++++++++++++ 3 files changed, 144 insertions(+), 2 deletions(-) diff --git a/packages/client/runtime/src/client/sessions/manager.ts b/packages/client/runtime/src/client/sessions/manager.ts index 66a0d00dd9..b6daecb97c 100644 --- a/packages/client/runtime/src/client/sessions/manager.ts +++ b/packages/client/runtime/src/client/sessions/manager.ts @@ -57,6 +57,12 @@ export class SessionManager { * drop-and-backfill path; replayed and cleared on instantiation. Bounded per session (these * frames are low-frequency; overflow drops oldest) and dropped on session-removed (audit S7). */ private readonly pendingBuffers = new Map[]>() + /** + * Latest model capacity observed for an uninstantiated session on the + * current mux generation. Unlike durable history, this transient frame + * cannot be backfilled when get() lazily creates the Session. + */ + private readonly modelRequestContextWindows = new Map() /** Per-session projection value stores, retained independently of instance arrival (the * title-snapshot precedent, generalized): push frames land here whether or not the Session * is instantiated (list rows read the 'title' key), and an instantiated Session adopts the @@ -160,6 +166,7 @@ export class SessionManager { } private createSession(sessionId: SessionId): Session { + const modelRequestContextWindow = this.modelRequestContextWindows.get(sessionId) return new Session(sessionId, this.api, { // The sender's local first-send flip mirrors into the list row so the // session surfaces (lists filter on blank) before any host frame lands. @@ -167,6 +174,7 @@ export class SessionManager { this.recordMutation({ kind: 'engaged', sessionId: engaged.sessionId }) }, projections: this.projectionStore(sessionId), + ...(modelRequestContextWindow === undefined ? {} : { modelRequestContextWindow }), }) } @@ -328,7 +336,14 @@ export class SessionManager { this.notifier.markDirty() return } + if (frame.type === 'session/model-request') { + // Transient and non-replayable: retain the latest capacity until lazy + // instantiation. An absent value explicitly clears an earlier one. + if (frame.contextWindow === undefined) this.modelRequestContextWindows.delete(frame.sessionId) + else this.modelRequestContextWindows.set(frame.sessionId, frame.contextWindow) + } if (frame.type === 'session/subscribed') { + this.modelRequestContextWindows.delete(frame.sessionId) // Rows past the host's durable baseline rode state a restart lost; drop // them so last-wins cannot pin a phantom value over recomputed truth. this.projectionStores.get(frame.sessionId)?.truncate(frame.lastSeq) @@ -391,6 +406,7 @@ export class SessionManager { this.recordMutation({ kind: 'remove', sessionId: frame.sessionId }) this.sessions.get(frame.sessionId)?.handleRemoved() // instance survives (resident-instance rule), only flagged in the snapshot this.pendingBuffers.delete(frame.sessionId) // a removed session's buffered frames must not replay on a future instantiation + this.modelRequestContextWindows.delete(frame.sessionId) // connection-local request capacity dies with the Host session this.projectionStores.delete(frame.sessionId) // removed sessions drop their projection rows with the instance return } diff --git a/packages/client/runtime/src/client/sessions/session.ts b/packages/client/runtime/src/client/sessions/session.ts index f3f4891de2..8c4cee798d 100644 --- a/packages/client/runtime/src/client/sessions/session.ts +++ b/packages/client/runtime/src/client/sessions/session.ts @@ -43,6 +43,8 @@ export interface SessionOptions { * private store (bare object-layer construction). */ projections?: ProjectionValueStore + /** Model capacity already observed on this mux generation before lazy construction. */ + modelRequestContextWindow?: number } /** Queue-row preview cap: the dock renders one line, the full content never leaves the host mirror. */ @@ -172,6 +174,7 @@ export class Session implements ObservableSnapshot { private readonly options: SessionOptions = {}, ) { this.projections = options.projections ?? new ProjectionValueStore() + this.contextWindow = options.modelRequestContextWindow this.snapshotCache = this.buildSnapshot() } @@ -479,10 +482,12 @@ export class Session implements ObservableSnapshot { this.notifier.markDirty() } - /** host/session-removed relay: flag the snapshot (instance survives — resident-instance rule). */ + /** host/session-removed relay: flag the resident snapshot and clear connection-local capacity. */ handleRemoved(): void { + const changed = !this.removed || this.contextWindow !== undefined this.removed = true - this.notifier.markDirty() + this.contextWindow = undefined + if (changed) this.notifier.markDirty() } /** diff --git a/packages/client/runtime/tests/manager.spec.ts b/packages/client/runtime/tests/manager.spec.ts index 2923cd0d3c..410058228d 100644 --- a/packages/client/runtime/tests/manager.spec.ts +++ b/packages/client/runtime/tests/manager.spec.ts @@ -41,6 +41,127 @@ describe('instances', () => { expect(manager.get(S2).getSnapshot().pending).toEqual([]) }) + it('retains the latest transient model capacity until lazy instantiation', () => { + const api = new FakeApiClient() + const manager = new SessionManager(api) + manager.handleMuxEnvelope({ + rpcId: 'request-1' as never, + payload: { + type: 'session/model-request', + sessionId: S1, + turn: 1, + step: 1, + provider: 'test', + model: 'alpha', + contextWindow: 128_000, + }, + }) + manager.handleMuxEnvelope({ + rpcId: 'request-2' as never, + payload: { + type: 'session/model-request', + sessionId: S1, + turn: 1, + step: 2, + provider: 'test', + model: 'beta', + contextWindow: 256_000, + }, + }) + + expect(manager.get(S1).getSnapshot().modelRequestContextWindow).toBe(256_000) + }) + + it('retains explicit capacity clearing before lazy instantiation', () => { + const api = new FakeApiClient() + const manager = new SessionManager(api) + manager.handleMuxEnvelope({ + rpcId: 'request-with-capacity' as never, + payload: { + type: 'session/model-request', + sessionId: S1, + turn: 1, + step: 1, + provider: 'test', + model: 'alpha', + contextWindow: 128_000, + }, + }) + manager.handleMuxEnvelope({ + rpcId: 'request-without-capacity' as never, + payload: { + type: 'session/model-request', + sessionId: S1, + turn: 1, + step: 2, + provider: 'test', + model: 'unknown-capacity', + }, + }) + + expect(manager.get(S1).getSnapshot().modelRequestContextWindow).toBeUndefined() + }) + + it('clears retained capacity on subscribed and resident capacity on removal', () => { + const api = new FakeApiClient() + const manager = new SessionManager(api) + manager.handleMuxEnvelope({ + rpcId: 'request-before-subscribe' as never, + payload: { + type: 'session/model-request', + sessionId: S1, + turn: 1, + step: 1, + provider: 'test', + model: 'alpha', + contextWindow: 128_000, + }, + }) + manager.handleMuxEnvelope({ + rpcId: 'subscribed' as never, + payload: { type: 'session/subscribed', sessionId: S1, lastSeq: 0 }, + }) + const session = manager.get(S1) + expect(session.getSnapshot().modelRequestContextWindow).toBeUndefined() + + manager.handleMuxEnvelope({ + rpcId: 'request-after-subscribe' as never, + payload: { + type: 'session/model-request', + sessionId: S1, + turn: 1, + step: 2, + provider: 'test', + model: 'beta', + contextWindow: 256_000, + }, + }) + expect(session.getSnapshot().modelRequestContextWindow).toBe(256_000) + manager.handleHostEnvelope({ + rpcId: 'removed' as never, + payload: { type: 'host/session-removed', sessionId: S1 }, + }) + expect(session.getSnapshot().modelRequestContextWindow).toBeUndefined() + + manager.handleMuxEnvelope({ + rpcId: 'request-before-lazy-removal' as never, + payload: { + type: 'session/model-request', + sessionId: S2, + turn: 1, + step: 1, + provider: 'test', + model: 'gamma', + contextWindow: 64_000, + }, + }) + manager.handleHostEnvelope({ + rpcId: 'lazy-removed' as never, + payload: { type: 'host/session-removed', sessionId: S2 }, + }) + expect(manager.get(S2).getSnapshot().modelRequestContextWindow).toBeUndefined() + }) + it('caps the pending buffer at 32 keeping the newest, and drops it on session-removed', () => { const api = new FakeApiClient() const manager = new SessionManager(api) From 47205cbb82f095dbcbd42506ca99c4b86c0c5e32 Mon Sep 17 00:00:00 2001 From: Hypatia May Date: Tue, 28 Jul 2026 20:17:28 +0800 Subject: [PATCH 14/38] fix(web): preserve reconnect metrics baseline --- .../runtime/src/client/sessions/session.ts | 10 +++--- packages/client/runtime/tests/session.spec.ts | 34 +++++++++++++++++++ 2 files changed, 38 insertions(+), 6 deletions(-) diff --git a/packages/client/runtime/src/client/sessions/session.ts b/packages/client/runtime/src/client/sessions/session.ts index 1471aae91e..2eec667e46 100644 --- a/packages/client/runtime/src/client/sessions/session.ts +++ b/packages/client/runtime/src/client/sessions/session.ts @@ -309,11 +309,10 @@ export class Session implements ObservableSnapshot { * in-flight open first — its history request rode the dead connection and must not settle * the fresh generation into 'error' (audit S4). */ async resync(): Promise { - // The queue mirror is NOT cleared here: onConnected (which drives resync) - // races the mux frames — the fresh generation's baseline may have landed - // already, and the host never resends it. The mirror re-baselines on the - // session/subscribed frame instead (same stream as the queue snapshot - // that follows it, so ordering is guaranteed). + // Queue, metrics, and request capacity are NOT cleared here: onConnected + // (which drives resync) races the mux frames — fresh-generation state may + // have landed already, and the host never resends it. session/subscribed + // owns the generation reset before the queue snapshot and metrics frames. if (this.openState === 'cold') return // never opened: no window to rebuild (doOpen flips to 'loading' synchronously, so cold implies no in-flight open) this.openGeneration++ this.openPromise = null @@ -322,7 +321,6 @@ export class Session implements ObservableSnapshot { this.events = [] this.views = [] this.baseSeq = 0 - this.metrics = null // Superseded, not settled: the baseline replay re-sends still-pending requested frames verbatim // (same rpcId), re-minting fresh waits; a stale reference's respond() still reaches the host. this.pending.clear() diff --git a/packages/client/runtime/tests/session.spec.ts b/packages/client/runtime/tests/session.spec.ts index 5f01dc9cc2..cbce270597 100644 --- a/packages/client/runtime/tests/session.spec.ts +++ b/packages/client/runtime/tests/session.spec.ts @@ -813,6 +813,40 @@ describe('remaining branches', () => { }) describe('resync', () => { + it('preserves fresh-generation metrics that arrive before a failing history refresh', async () => { + const { api, session } = makeSession() + const oldMetrics = metrics(8, 10) + api.onHistory = () => histResponse(plainTurn(0, 0, 'a', 'b'), false, undefined, oldMetrics) + await session.open() + expect(session.getSnapshot().metrics).toBe(oldMetrics) + + session.handleMuxEnvelope('sub' as never, { + type: 'session/subscribed', + sessionId: SID, + lastSeq: 5, + }) + expect(session.getSnapshot().metrics).toBeNull() + + const freshMetrics = metrics(0, 10, { contextTokens: 20 }) + session.handleMuxEnvelope('fresh-metrics' as never, { + type: 'session/metrics', + sessionId: SID, + metrics: freshMetrics, + }) + api.onHistory = () => Promise.resolve(err({ + code: 'internal', + message: 'history refresh failed', + details: {}, + })) + + await session.resync() + + expect(session.getSnapshot()).toMatchObject({ + openState: 'error', + metrics: freshMetrics, + }) + }) + it('rebuilds the window and clears pending; cold instances no-op', async () => { const { api, session } = makeSession() api.onHistory = () => histResponse(plainTurn(0, 0, 'a', 'b')) From ef12895cf1c82cf2e6d233924bfc9df03020df57 Mon Sep 17 00:00:00 2001 From: Hypatia May Date: Tue, 28 Jul 2026 20:28:01 +0800 Subject: [PATCH 15/38] fix(web): reset live metrics between connections --- .../connection/src/client/connection.ts | 9 ++- .../connection/tests/connection.spec.ts | 52 +++++++++++++++++- packages/client/runtime/src/client/index.ts | 1 + .../runtime/src/client/sessions/manager.ts | 6 ++ .../runtime/src/client/sessions/service.ts | 5 ++ .../runtime/src/client/sessions/session.ts | 8 +++ .../client/runtime/tests/client-apply.spec.ts | 47 ++++++++++++++++ packages/client/runtime/tests/manager.spec.ts | 55 ++++++++++++++++++- 8 files changed, 178 insertions(+), 5 deletions(-) diff --git a/packages/client/connection/src/client/connection.ts b/packages/client/connection/src/client/connection.ts index 6eb6491e2f..a15312c9f9 100644 --- a/packages/client/connection/src/client/connection.ts +++ b/packages/client/connection/src/client/connection.ts @@ -46,6 +46,8 @@ export interface ConnectionSinks { onHostEnvelope?: (envelope: RpcRequest) => void /** After each connection generation is established (both streams open + describe succeeded), first connect included. */ onConnected?: () => void + /** After every failed generation closes and before retry starts. Not emitted when the controller is stopped. */ + onDisconnected?: () => void /** Coarse state transitions (deduplicated: fires only on change). The initial pre-connect * span reports nothing — the UI treats "no state yet" as connecting, not as an outage. */ onStateChange?: (state: ConnectionState) => void @@ -120,8 +122,8 @@ export class ConnectionController { if (gen === this.generation && !ac.signal.aborted) ac.abort() resolve() } - void this.pumpStream(this.api.events.mux({}, ac.signal, muxOpened), this.sinks.onMuxEnvelope, settle) - void this.pumpStream(this.api.events.host({}, ac.signal, hostOpened), this.sinks.onHostEnvelope, settle) + void this.pumpStream(this.api.events.mux({}, ac.signal, muxOpened), this.sinks.onMuxEnvelope, ac.signal, settle) + void this.pumpStream(this.api.events.host({}, ac.signal, hostOpened), this.sinks.onHostEnvelope, ac.signal, settle) }) try { @@ -147,6 +149,7 @@ export class ConnectionController { await failed if (!this.isRunning()) return + this.callSink(this.sinks.onDisconnected) this.emitState('reconnecting') this.attempt += 1 console.warn(`[web-runtime] connection lost, retry #${this.attempt}`) @@ -165,10 +168,12 @@ export class ConnectionController { private async pumpStream( stream: AsyncIterable>, sink: ((envelope: RpcRequest) => void) | undefined, + signal: AbortSignal, onEnd: () => void, ): Promise { try { for await (const envelope of stream) { + if (signal.aborted) break if (envelope.payload.type === 'stream/error') break if (sink !== undefined) this.callSink(() => { sink(envelope) }) } diff --git a/packages/client/connection/tests/connection.spec.ts b/packages/client/connection/tests/connection.spec.ts index 4de4a31f25..9c54fcbcff 100644 --- a/packages/client/connection/tests/connection.spec.ts +++ b/packages/client/connection/tests/connection.spec.ts @@ -7,7 +7,7 @@ */ import { describe, expect, it, vi } from 'vitest' -import type { SessionId } from '../src/client/api.ts' +import type { IApiClient, SessionId } from '../src/client/api.ts' import type { ConnectionState } from '../src/client/connection.ts' import { ConnectionController } from '../src/client/connection.ts' import { FakeApiClient, deferred, ok } from './fake-api.ts' @@ -104,6 +104,51 @@ describe('connection lifecycle', () => { } }) + it('drops a sibling stream frame buffered behind a generation failure', async () => { + const api = new FakeApiClient() + const lateMux = deferred() + const originalEvents = api.events + Object.defineProperty(api, 'events', { + value: { + host: (...args: Parameters) => originalEvents.host(...args), + mux: (_payload: unknown, _signal: AbortSignal, onOpen?: () => void) => (async function* () { + onOpen?.() + await lateMux.promise + yield { rpcId: 'late-mux' as never, payload: subscribedFrame(2) } + })(), + } satisfies IApiClient['events'], + }) + const muxSeen: number[] = [] + let connected = 0 + let disconnected = 0 + const warnSpy = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + const controller = new ConnectionController(api, { + onMuxEnvelope: (envelope) => { + if (envelope.payload.type === 'session/subscribed') muxSeen.push(envelope.payload.lastSeq) + }, + onConnected: () => { connected++ }, + onDisconnected: () => { + disconnected++ + lateMux.resolve(undefined) + controller.stop() + }, + }, FAST) + controller.start() + try { + await vi.waitFor(() => { expect(connected).toBe(1) }) + api.pushHost({ + type: 'stream/error', + error: { code: 'internal', message: 'host stream failed', details: {} }, + }) + await vi.waitFor(() => { expect(disconnected).toBe(1) }) + await new Promise(resolve => setTimeout(resolve, 0)) + expect(muxSeen).toEqual([]) + } finally { + controller.stop() + warnSpy.mockRestore() + } + }) + it('isolates sink exceptions from the pump', async () => { const api = new FakeApiClient() const seen: string[] = [] @@ -181,7 +226,7 @@ describe('connection lifecycle', () => { } }) - it('deduplicates consecutive reconnecting emissions across two straight failures', async () => { + it('reports every failed generation while deduplicating consecutive reconnecting state', async () => { const api = new FakeApiClient() const gate = deferred>>() let describeCalls = 0 @@ -190,10 +235,12 @@ describe('connection lifecycle', () => { return describeCalls <= 2 ? Promise.reject(new Error('down')) : gate.promise } const states: ConnectionState[] = [] + let disconnected = 0 let connected = 0 const warnSpy = vi.spyOn(console, 'warn').mockImplementation(() => undefined) const controller = new ConnectionController(api, { onConnected: () => { connected++ }, + onDisconnected: () => { disconnected++ }, onStateChange: state => states.push(state), }, FAST) controller.start() @@ -201,6 +248,7 @@ describe('connection lifecycle', () => { await vi.waitFor(() => { expect(describeCalls).toBe(3) }) gate.resolve(ok({ version: '0', cwd: '/f', attachedSessions: 0 })) await vi.waitFor(() => { expect(connected).toBe(1) }) + expect(disconnected).toBe(2) expect(states).toEqual(['reconnecting', 'connected']) // two failures, one reconnecting emission } finally { controller.stop() diff --git a/packages/client/runtime/src/client/index.ts b/packages/client/runtime/src/client/index.ts index 48514a9df4..f971331ba8 100644 --- a/packages/client/runtime/src/client/index.ts +++ b/packages/client/runtime/src/client/index.ts @@ -143,6 +143,7 @@ export function apply(ctx: Context): void { workspaces.handleConnected() ctx.emit('connection/reset') }, + onDisconnected: () => { sessions.handleReconnecting() }, }) ctx.effect(() => () => { loop.stop() }, 'runtime: connection stream loop') } diff --git a/packages/client/runtime/src/client/sessions/manager.ts b/packages/client/runtime/src/client/sessions/manager.ts index b6daecb97c..3443f60dd2 100644 --- a/packages/client/runtime/src/client/sessions/manager.ts +++ b/packages/client/runtime/src/client/sessions/manager.ts @@ -430,6 +430,12 @@ export class SessionManager { for (const session of this.sessions.values()) void session.resync() } + /** Before a replacement stream generation, discard values the Host does not replay. */ + handleReconnecting(): void { + this.modelRequestContextWindows.clear() + for (const session of this.sessions.values()) session.handleReconnecting() + } + private buildListSnapshot(): SessionListSnapshot { const merged: TitledSessionSummary[] = this.summaries.map((summary) => { // List rows read the generic 'title' projection key (host-computed unit diff --git a/packages/client/runtime/src/client/sessions/service.ts b/packages/client/runtime/src/client/sessions/service.ts index 19e022d450..e342450b94 100644 --- a/packages/client/runtime/src/client/sessions/service.ts +++ b/packages/client/runtime/src/client/sessions/service.ts @@ -393,6 +393,11 @@ export class SessionsService { this.manager.handleConnected() } + /** Clear connection-local Session state before the next stream generation starts. */ + handleReconnecting(): void { + this.manager.handleReconnecting() + } + /** * Create a session on the host. Resolution guarantee: by the time the * promise resolves, the created session is in the list store and diff --git a/packages/client/runtime/src/client/sessions/session.ts b/packages/client/runtime/src/client/sessions/session.ts index 2eec667e46..49035736dc 100644 --- a/packages/client/runtime/src/client/sessions/session.ts +++ b/packages/client/runtime/src/client/sessions/session.ts @@ -481,6 +481,14 @@ export class Session implements ObservableSnapshot { this.notifier.markDirty() } + /** Connection-loss boundary: clear values that are not replayed before the next stream starts. */ + handleReconnecting(): void { + if (this.metrics === null && this.contextWindow === undefined) return + this.metrics = null + this.contextWindow = undefined + this.notifier.markDirty() + } + /** host/session-removed relay: flag the resident snapshot and clear connection-local capacity. */ handleRemoved(): void { const changed = !this.removed || this.contextWindow !== undefined diff --git a/packages/client/runtime/tests/client-apply.spec.ts b/packages/client/runtime/tests/client-apply.spec.ts index d5b29f10a9..7399c46f6b 100644 --- a/packages/client/runtime/tests/client-apply.spec.ts +++ b/packages/client/runtime/tests/client-apply.spec.ts @@ -102,6 +102,53 @@ describe('runtime client apply', () => { expect(bench.api.callsOf('session.create')).toHaveLength(1) }) + it('clears connection-local Session state on disconnect but not connected', async () => { + const bench = await mount() + const sessions = bench.ctx.get('sessions') as SessionsService + bench.sinks?.onHostEnvelope?.({ + rpcId: 'session' as never, + payload: { type: 'host/session-added', blank: true, sessionId: 's-state' } as never, + }) + await Promise.resolve() + const session = sessions.binding('s-state' as never)?.session + if (session === undefined) throw new Error('session binding missing') + const currentMetrics = { + projectionRevision: 4, + logRevision: 10, + uncachedInputTokens: 10, + outputTokens: 4, + cacheReadTokens: 90, + cacheWriteTokens: 3, + contextTokens: 35, + } + bench.sinks?.onMuxEnvelope?.({ + rpcId: 'metrics' as never, + payload: { type: 'session/metrics', sessionId: 's-state', metrics: currentMetrics } as never, + }) + bench.sinks?.onMuxEnvelope?.({ + rpcId: 'capacity' as never, + payload: { + type: 'session/model-request', + sessionId: 's-state', + turn: 1, + step: 1, + provider: 'test', + model: 'alpha', + contextWindow: 128_000, + } as never, + }) + + bench.sinks?.onConnected?.() + expect(session.getSnapshot()).toMatchObject({ + metrics: currentMetrics, + modelRequestContextWindow: 128_000, + }) + + bench.sinks?.onDisconnected?.() + expect(session.getSnapshot().metrics).toBeNull() + expect(session.getSnapshot().modelRequestContextWindow).toBeUndefined() + }) + it('stops the stream loop when the plugin fiber unloads', async () => { const bench = await mount() const fiber = [...bench.ctx.registry.values()].find(f => f.name?.includes('client')) diff --git a/packages/client/runtime/tests/manager.spec.ts b/packages/client/runtime/tests/manager.spec.ts index 410058228d..874e0be477 100644 --- a/packages/client/runtime/tests/manager.spec.ts +++ b/packages/client/runtime/tests/manager.spec.ts @@ -4,7 +4,7 @@ */ import { describe, expect, it, vi } from 'vitest' -import type { SessionId } from '@deepseek-ai/dsh-client-connection/client' +import type { SessionId, SessionMetrics } from '@deepseek-ai/dsh-client-connection/client' import { SessionManager } from '../src/client/sessions/manager.ts' import { FakeApiClient, deferred, err, ok } from './fake-api.ts' import { entries, plainTurn } from './event-script.ts' @@ -162,6 +162,59 @@ describe('instances', () => { expect(manager.get(S2).getSnapshot().modelRequestContextWindow).toBeUndefined() }) + it('clears resident metrics and capacity plus lazy capacity before reconnect', () => { + const api = new FakeApiClient() + const manager = new SessionManager(api) + const session = manager.get(S1) + const currentMetrics: SessionMetrics = { + projectionRevision: 4, + logRevision: 10, + uncachedInputTokens: 10, + outputTokens: 4, + cacheReadTokens: 90, + cacheWriteTokens: 3, + contextTokens: 35, + } + manager.handleMuxEnvelope({ + rpcId: 'metrics' as never, + payload: { type: 'session/metrics', sessionId: S1, metrics: currentMetrics }, + }) + manager.handleMuxEnvelope({ + rpcId: 'resident-capacity' as never, + payload: { + type: 'session/model-request', + sessionId: S1, + turn: 1, + step: 1, + provider: 'test', + model: 'resident', + contextWindow: 128_000, + }, + }) + manager.handleMuxEnvelope({ + rpcId: 'lazy-capacity' as never, + payload: { + type: 'session/model-request', + sessionId: S2, + turn: 1, + step: 1, + provider: 'test', + model: 'lazy', + contextWindow: 256_000, + }, + }) + expect(session.getSnapshot()).toMatchObject({ + metrics: currentMetrics, + modelRequestContextWindow: 128_000, + }) + + manager.handleReconnecting() + + expect(session.getSnapshot().metrics).toBeNull() + expect(session.getSnapshot().modelRequestContextWindow).toBeUndefined() + expect(manager.get(S2).getSnapshot().modelRequestContextWindow).toBeUndefined() + }) + it('caps the pending buffer at 32 keeping the newest, and drops it on session-removed', () => { const api = new FakeApiClient() const manager = new SessionManager(api) From c2543de85b1b94e42f32d2cb3dcc6a4883545f17 Mon Sep 17 00:00:00 2001 From: Hypatia May Date: Tue, 28 Jul 2026 20:43:26 +0800 Subject: [PATCH 16/38] fix(web): fence stale history after disconnect --- .../runtime/src/client/sessions/session.ts | 18 ++- packages/client/runtime/tests/session.spec.ts | 111 ++++++++++++++++-- 2 files changed, 115 insertions(+), 14 deletions(-) diff --git a/packages/client/runtime/src/client/sessions/session.ts b/packages/client/runtime/src/client/sessions/session.ts index 49035736dc..6708202276 100644 --- a/packages/client/runtime/src/client/sessions/session.ts +++ b/packages/client/runtime/src/client/sessions/session.ts @@ -82,9 +82,8 @@ export class Session implements ObservableSnapshot { private openState: OpenState = 'cold' private openError: RpcError | null = null private openPromise: Promise | null = null - /** Bumped by resync to invalidate an in-flight doOpen: a reconnect must rebuild, never adopt - * a pre-disconnect open whose history request is already doomed (audit S4). Stale doOpen - * passes drop all writes once the generation moves on. */ + /** Bumped at disconnect and resync to invalidate in-flight history work: a reconnect must + * rebuild, never adopt a pre-disconnect response (audit S4). */ private openGeneration = 0 private loadingOlder = false private readonly foldAdapter = new FoldAdapter() @@ -270,12 +269,14 @@ export class Session implements ObservableSnapshot { /** Page up: pull one earlier page with the window's first seq as beforeSeq and prepend (§D.2). */ async loadOlder(): Promise { if (this.openState !== 'open' || !this.hasMore || this.loadingOlder) return + const generation = this.openGeneration this.loadingOlder = true this.notifier.markDirty() try { const { result } = await this.api.sessions.history({ sessionId: this.sessionId, beforeSeq: this.baseSeq, maxMessages: PAGE_MESSAGES, }) + if (generation !== this.openGeneration) return if (!result.ok) return // keep the window as-is; do not overwrite openError (open already succeeded) const older = result.value.events if (older.length === 0) { @@ -299,8 +300,10 @@ export class Session implements ObservableSnapshot { } catch (error) { console.error('[web-runtime] loadOlder failed:', error) } finally { - this.loadingOlder = false - this.notifier.markDirty() + if (generation === this.openGeneration) { + this.loadingOlder = false + this.notifier.markDirty() + } } } @@ -321,6 +324,8 @@ export class Session implements ObservableSnapshot { this.events = [] this.views = [] this.baseSeq = 0 + this.loadingOlder = false + this.stitching = false // Superseded, not settled: the baseline replay re-sends still-pending requested frames verbatim // (same rpcId), re-minting fresh waits; a stale reference's respond() still reaches the host. this.pending.clear() @@ -483,6 +488,7 @@ export class Session implements ObservableSnapshot { /** Connection-loss boundary: clear values that are not replayed before the next stream starts. */ handleReconnecting(): void { + this.openGeneration++ if (this.metrics === null && this.contextWindow === undefined) return this.metrics = null this.contextWindow = undefined @@ -649,7 +655,7 @@ export class Session implements ObservableSnapshot { } catch (error) { console.error('[web-runtime] gap repair failed:', error) } finally { - this.stitching = false + if (generation === this.openGeneration) this.stitching = false } } diff --git a/packages/client/runtime/tests/session.spec.ts b/packages/client/runtime/tests/session.spec.ts index cbce270597..d879362990 100644 --- a/packages/client/runtime/tests/session.spec.ts +++ b/packages/client/runtime/tests/session.spec.ts @@ -392,6 +392,27 @@ describe('paging', () => { await Promise.all([first, second]) expect(api.callsOf('session.history')).toHaveLength(2) // open + one page, not two }) + + it('drops an older page from the disconnected generation', async () => { + const { api, session } = makeSession() + api.onHistory = () => histResponse(plainTurn(6, 1, '新问', '新答'), true) + await session.open() + const stale = deferred>>() + api.onHistory = () => stale.promise + const loading = session.loadOlder() + + session.handleReconnecting() + stale.resolve(ok({ + events: entries(plainTurn(0, 0, '旧问', '旧答')) as never[], + hasMore: false, + })) + await loading + expect(session.getSnapshot().nodes.map(node => node.seq)).toEqual([7, 9]) + + api.onHistory = () => histResponse(plainTurn(12, 2, '重连问', '重连答')) + await session.resync() + expect(session.getSnapshot().nodes.map(node => node.seq)).toEqual([13, 15]) + }) }) describe('prompt and cancel errors', () => { @@ -733,22 +754,49 @@ describe('remaining branches', () => { expect(session.getSnapshot().openState).toBe('open') }) - it('drops a gap repair superseded by a full resync while its pull was in flight', async () => { + it('drops a stale gap repair without clearing a newer generation repair', async () => { const { api, session } = makeSession() api.onHistory = () => histResponse(plainTurn(0, 0, 'a', 'b')) await session.open() - const repairPull = deferred>>() - api.onHistory = () => repairPull.promise + const staleRepair = deferred>>() + api.onHistory = () => staleRepair.promise session.handleMuxEnvelope('r' as never, { type: 'session/event', sessionId: SID, event: ev.user(9, '洞') }) // starts repairGap + session.handleReconnecting() api.onHistory = () => histResponse(plainTurn(6, 1, 'c', 'd')) - const resynced = session.resync() // bumps the generation - repairPull.resolve(ok({ + await session.resync() + + const freshRepair = deferred>>() + let freshRepairCalls = 0 + api.onHistory = () => { + freshRepairCalls++ + return freshRepair.promise + } + session.handleMuxEnvelope('fresh-gap' as never, { + type: 'session/event', + sessionId: SID, + event: ev.user(15, '新洞'), + }) + expect(freshRepairCalls).toBe(1) + + staleRepair.resolve(ok({ events: entries(plainTurn(0, 0, '旧', '页')) as never[], hasMore: false, - modelTarget: { provider: 'deepseek', model: 'stale' }, })) // repair result: stale, dropped - await resynced - expect(session.getSnapshot().nodes.map(n => n.seq)).toEqual([7, 9]) + await Promise.resolve() + session.handleMuxEnvelope('fresh-buffer' as never, { + type: 'session/event', + sessionId: SID, + event: ev.user(16, '继续缓存'), + }) + expect(freshRepairCalls).toBe(1) // stale finally did not clear the newer stitching owner + + freshRepair.resolve(ok({ + events: entries([...plainTurn(6, 1, 'c', 'd'), ...plainTurn(12, 2, 'e', 'f')]) as never[], + hasMore: false, + })) + await vi.waitFor(() => { + expect(session.getSnapshot().nodes.map(n => n.seq)).toEqual([7, 9, 13, 15]) + }) }) it('successful cancel leaves no promptError; tool/result for an unknown callId is a no-op', async () => { @@ -813,6 +861,53 @@ describe('remaining branches', () => { }) describe('resync', () => { + it('fences pre-disconnect history behind a fresh mux metrics baseline', async () => { + const { api, session } = makeSession() + const stale = deferred>>() + api.onHistory = () => stale.promise + const opening = session.open() + const oldLiveMetrics = metrics(8, 10) + session.handleMuxEnvelope('old-metrics' as never, { + type: 'session/metrics', + sessionId: SID, + metrics: oldLiveMetrics, + }) + session.handleMuxEnvelope('old-capacity' as never, { + type: 'session/model-request', + sessionId: SID, + turn: 1, + step: 1, + provider: 'test', + model: 'old', + contextWindow: 128_000, + }) + + session.handleReconnecting() + expect(session.getSnapshot().metrics).toBeNull() + expect(session.getSnapshot().modelRequestContextWindow).toBeUndefined() + const freshMetrics = metrics(0, 1, { contextTokens: 20 }) + session.handleMuxEnvelope('fresh-metrics' as never, { + type: 'session/metrics', + sessionId: SID, + metrics: freshMetrics, + }) + + stale.resolve(ok({ + events: entries(plainTurn(0, 0, '旧问', '旧答')) as never[], + hasMore: false, + metrics: metrics(99, 99, { contextTokens: 999 }), + })) + await opening + expect(session.getSnapshot().nodes).toEqual([]) + expect(session.getSnapshot().metrics).toBe(freshMetrics) + + api.onHistory = () => histResponse(plainTurn(6, 1, '新问', '新答')) + await session.resync() + expect(session.getSnapshot().openState).toBe('open') + expect(session.getSnapshot().nodes.map(node => node.seq)).toEqual([7, 9]) + expect(session.getSnapshot().metrics).toBe(freshMetrics) + }) + it('preserves fresh-generation metrics that arrive before a failing history refresh', async () => { const { api, session } = makeSession() const oldMetrics = metrics(8, 10) From e37cb233362266acda5c6db3bbd6f2940b887abf Mon Sep 17 00:00:00 2001 From: Hypatia May Date: Wed, 29 Jul 2026 14:35:24 +0800 Subject: [PATCH 17/38] feat(token-meter): project durable token usage --- packages/llm/token-meter/package.json | 10 +- packages/llm/token-meter/src/client.ts | 7 + packages/llm/token-meter/src/index.ts | 9 + packages/llm/token-meter/src/projection.ts | 25 ++ packages/llm/token-meter/src/types.ts | 2 + .../llm/token-meter/src/usage-projection.ts | 100 ++++++++ .../tests/token-usage-projection.spec.ts | 222 ++++++++++++++++++ packages/llm/token-meter/tsconfig.json | 3 + pnpm-lock.yaml | 6 + 9 files changed, 383 insertions(+), 1 deletion(-) create mode 100644 packages/llm/token-meter/src/client.ts create mode 100644 packages/llm/token-meter/src/projection.ts create mode 100644 packages/llm/token-meter/src/usage-projection.ts create mode 100644 packages/llm/token-meter/tests/token-usage-projection.spec.ts diff --git a/packages/llm/token-meter/package.json b/packages/llm/token-meter/package.json index dadd5e8f8d..99e90e4c3b 100644 --- a/packages/llm/token-meter/package.json +++ b/packages/llm/token-meter/package.json @@ -15,12 +15,17 @@ "types": "./lib/types/invariant.d.ts", "default": "./lib/invariant.js" }, + "./client": { + "types": "./lib/types/client.d.ts", + "default": "./lib/client.js" + }, "./src/*": "./src/*", "./package.json": "./package.json" }, "files": [ "lib/index.js", "lib/invariant.js", + "lib/client.js", "lib/types/**/*.d.ts", "lib/types/**/*.d.ts.map", "src" @@ -30,15 +35,18 @@ "@deepseek-ai/dsh-invariants": "^0.0.1", "@deepseek-ai/dsh-llm": "^0.0.1", "@deepseek-ai/dsh-session": "^0.0.1", + "@deepseek-ai/dsh-session-projection": "^0.0.1", "cordis": "^4.0.0-rc.7" }, "dependencies": { - "schemastery": "^3.18.0" + "schemastery": "^3.18.0", + "zod": "^4.4.3" }, "devDependencies": { "@deepseek-ai/dsh-invariants": "workspace:^", "@deepseek-ai/dsh-llm": "workspace:^", "@deepseek-ai/dsh-session": "workspace:^", + "@deepseek-ai/dsh-session-projection": "workspace:^", "cordis": "^4.0.0-rc.7" } } diff --git a/packages/llm/token-meter/src/client.ts b/packages/llm/token-meter/src/client.ts new file mode 100644 index 0000000000..1bc02e3073 --- /dev/null +++ b/packages/llm/token-meter/src/client.ts @@ -0,0 +1,7 @@ +/** + * Client-namespace projection of token-meter's browser-safe types. + * + * @module @deepseek-ai/dsh-token-meter/client + */ + +export type * from './projection.ts' diff --git a/packages/llm/token-meter/src/index.ts b/packages/llm/token-meter/src/index.ts index 533ebd2453..9b269b5cae 100644 --- a/packages/llm/token-meter/src/index.ts +++ b/packages/llm/token-meter/src/index.ts @@ -10,12 +10,15 @@ import { BlockAssembler, deepFreeze } from '@deepseek-ai/dsh-llm' import type { ContentBlock, Message, TokenUsage } from '@deepseek-ai/dsh-llm' import type { EpochHeader, Session, SessionEvent, SurfaceEvent } from '@deepseek-ai/dsh-session' import { canonicalHeader, headerEquals, isSurfaceEvent } from '@deepseek-ai/dsh-session' +// Type-only: resolves the optional projection registry Context seam. +import type {} from '@deepseek-ai/dsh-session-projection' import type { TokenMeasurement, TokenMeasurementBaseline, TokenMeterConfig, TokenSurfaceNode, } from './types.ts' +import { tokenUsageProjectionDefinition } from './usage-projection.ts' export type * from './types.ts' @@ -90,6 +93,12 @@ export class TokenMeterService extends Service { super(ctx, 'tokenMeter') validateConfigKeys(config) + // Projection registration is an optional child: headless and TUI + // compositions without the generic registry keep the meter's old shape. + ctx.inject(['sessionProjections'], (projectionCtx) => { + projectionCtx.sessionProjections.register(tokenUsageProjectionDefinition) + }) + // Readers catch up independently, while eager observation bounds ordinary // read latency without creating state for sessions no consumer has read. ctx.on('session/event', (session) => { diff --git a/packages/llm/token-meter/src/projection.ts b/packages/llm/token-meter/src/projection.ts new file mode 100644 index 0000000000..93c52297e8 --- /dev/null +++ b/packages/llm/token-meter/src/projection.ts @@ -0,0 +1,25 @@ +/** + * Pure client-safe token-usage projection vocabulary. + * + * @module @deepseek-ai/dsh-token-meter/projection + */ + +/** + * Durable cumulative provider usage for a complete session log. + * + * The four buckets are disjoint. In particular, reasoning tokens are already + * included in `outputTokens` and are not accumulated again. + */ +export interface TokenUsageProjection { + uncachedInputTokens: number + outputTokens: number + cacheReadTokens: number + cacheWriteTokens: number +} + +declare module '@deepseek-ai/dsh-session-projection/types' { + interface SessionProjectionMap { + /** Provider-reported usage accumulated across the complete durable log. */ + tokenUsage: TokenUsageProjection + } +} diff --git a/packages/llm/token-meter/src/types.ts b/packages/llm/token-meter/src/types.ts index 255425b639..15d2d50c41 100644 --- a/packages/llm/token-meter/src/types.ts +++ b/packages/llm/token-meter/src/types.ts @@ -6,6 +6,8 @@ import type { TokenUsage } from '@deepseek-ai/dsh-llm' +export type { TokenUsageProjection } from './projection.ts' + /** Token-meter plugin configuration; the fixed estimator has no settings. */ export type TokenMeterConfig = Record diff --git a/packages/llm/token-meter/src/usage-projection.ts b/packages/llm/token-meter/src/usage-projection.ts new file mode 100644 index 0000000000..bbd231fa01 --- /dev/null +++ b/packages/llm/token-meter/src/usage-projection.ts @@ -0,0 +1,100 @@ +/** + * Pure fold for durable provider-reported token usage. + */ + +import { z } from 'zod' +import type { TokenUsage } from '@deepseek-ai/dsh-llm' +import type { ProjectionDefinition } from '@deepseek-ai/dsh-session-projection' +import type { TokenUsageProjection } from './projection.ts' + +interface UsageSample { + turn: number + step: number + buckets: TokenUsageProjection +} + +interface TokenUsageState { + totals: TokenUsageProjection + last: UsageSample | null +} + +const zeroBuckets = (): TokenUsageProjection => ({ + uncachedInputTokens: 0, + outputTokens: 0, + cacheReadTokens: 0, + cacheWriteTokens: 0, +}) + +const bucketsFrom = (usage: TokenUsage): TokenUsageProjection => ({ + uncachedInputTokens: usage.inputTokens, + outputTokens: usage.outputTokens, + cacheReadTokens: usage.cacheReadTokens ?? 0, + cacheWriteTokens: usage.cacheWriteTokens ?? 0, +}) + +const bucketsEqual = (left: TokenUsageProjection, right: TokenUsageProjection): boolean => + left.uncachedInputTokens === right.uncachedInputTokens + && left.outputTokens === right.outputTokens + && left.cacheReadTokens === right.cacheReadTokens + && left.cacheWriteTokens === right.cacheWriteTokens + +const addReplacing = ( + totals: TokenUsageProjection, + previous: TokenUsageProjection | undefined, + next: TokenUsageProjection, +): TokenUsageProjection => ({ + uncachedInputTokens: totals.uncachedInputTokens - (previous?.uncachedInputTokens ?? 0) + next.uncachedInputTokens, + outputTokens: totals.outputTokens - (previous?.outputTokens ?? 0) + next.outputTokens, + cacheReadTokens: totals.cacheReadTokens - (previous?.cacheReadTokens ?? 0) + next.cacheReadTokens, + cacheWriteTokens: totals.cacheWriteTokens - (previous?.cacheWriteTokens ?? 0) + next.cacheWriteTokens, +}) + +const projectionSchema = z.object({ + uncachedInputTokens: z.number().int().nonnegative(), + outputTokens: z.number().int().nonnegative(), + cacheReadTokens: z.number().int().nonnegative(), + cacheWriteTokens: z.number().int().nonnegative(), +}).strict() + +/** + * Token-meter's session projection unit. + * + * Usage chunks provide an early sample that survives a later request failure; + * an assistant message provides the final sample for the same turn/step. A + * repeated sample replaces that step's earlier value instead of double + * counting it. + */ +export const tokenUsageProjectionDefinition: +ProjectionDefinition<'tokenUsage', TokenUsageState> = { + key: 'tokenUsage', + schema: projectionSchema, + init: () => ({ totals: zeroBuckets(), last: null }), + apply: (state, event) => { + let turn: number + let step: number + let usage: TokenUsage + if (event.type === 'assistant/chunk' && event.data.chunk.type === 'usage') { + ;({ turn, step } = event.data) + usage = event.data.chunk.usage + } else if (event.type === 'assistant/message' && event.data.usage !== undefined) { + ;({ turn, step, usage } = event.data) + } else { + return state + } + + const buckets = bucketsFrom(usage) + const previous = state.last !== null + && state.last.turn === turn + && state.last.step === step + ? state.last.buckets + : undefined + if (previous !== undefined && bucketsEqual(previous, buckets)) return state + + return { + totals: addReplacing(state.totals, previous, buckets), + last: { turn, step, buckets }, + } + }, + view: state => state.totals, + stateVersion: 1, +} diff --git a/packages/llm/token-meter/tests/token-usage-projection.spec.ts b/packages/llm/token-meter/tests/token-usage-projection.spec.ts new file mode 100644 index 0000000000..07f745736a --- /dev/null +++ b/packages/llm/token-meter/tests/token-usage-projection.spec.ts @@ -0,0 +1,222 @@ +import { describe, expect, it } from 'vitest' +import { Context } from 'cordis' +import { createMessage, createUserMessage } from '@deepseek-ai/dsh-llm' +import type { TokenUsage } from '@deepseek-ai/dsh-llm' +import SessionStore from '@deepseek-ai/dsh-session' +import type { Session } from '@deepseek-ai/dsh-session' +import SessionProjectionRegistry from '@deepseek-ai/dsh-session-projection' +import TokenMeterService from '@deepseek-ai/dsh-token-meter' +import type { TokenUsageProjection } from '@deepseek-ai/dsh-token-meter/client' + +const ZERO: TokenUsageProjection = { + uncachedInputTokens: 0, + outputTokens: 0, + cacheReadTokens: 0, + cacheWriteTokens: 0, +} + +async function harness(): Promise<{ + ctx: Context + session: Session + meterFiber: Awaited> +}> { + const ctx = new Context() + await ctx.plugin(SessionStore) + await ctx.plugin(SessionProjectionRegistry) + const meterFiber = await ctx.plugin(TokenMeterService) + return { ctx, session: ctx.sessions.create(), meterFiber } +} + +function startStep(session: Session, turn: number, step: number): void { + session.append('step/start', { turn, step }) +} + +function usageChunk( + session: Session, + usage: TokenUsage, + turn: number, + step: number, +): number { + return session.append('assistant/chunk', { + turn, + step, + chunk: { type: 'usage', usage }, + }).seq +} + +function finalUsage( + session: Session, + usage: TokenUsage, + turn: number, + step: number, + sourceSeqs: number[], +): void { + session.append('assistant/message', { + turn, + step, + message: createMessage({ + role: 'assistant', + content: [], + source: { kind: 'model', provider: 'mock', model: 'mock' }, + }), + usage, + }, { surfaceOp: 'append', sourceEventSeqs: sourceSeqs }) + session.append('step/end', { turn, step }) +} + +const projected = (ctx: Context, session: Session): TokenUsageProjection => { + const value = ctx.sessionProjections.snapshot(session).values.tokenUsage + if (value === undefined) throw new Error('tokenUsage projection is not registered') + return value +} + +describe('tokenUsage session projection', () => { + it('serves zero buckets for an empty log', async () => { + const { ctx, session } = await harness() + expect(projected(ctx, session)).toEqual(ZERO) + }) + + it('does not count a usage chunk and identical final usage twice', async () => { + const { ctx, session } = await harness() + const changes: unknown[] = [] + ctx.sessionProjections.onChanged((_session, key, value) => { + if (key === 'tokenUsage') changes.push(value) + }) + const usage = { + inputTokens: 10, + outputTokens: 4, + cacheReadTokens: 7, + cacheWriteTokens: 2, + reasoningTokens: 3, + } + startStep(session, 1, 1) + const source = usageChunk(session, usage, 1, 1) + finalUsage(session, usage, 1, 1, [source]) + + expect(projected(ctx, session)).toEqual({ + uncachedInputTokens: 10, + outputTokens: 4, + cacheReadTokens: 7, + cacheWriteTokens: 2, + }) + expect(changes).toHaveLength(1) + }) + + it('replaces an earlier same-step chunk sample with the final usage', async () => { + const { ctx, session } = await harness() + startStep(session, 1, 1) + const source = usageChunk(session, { + inputTokens: 10, + outputTokens: 2, + cacheReadTokens: 3, + }, 1, 1) + finalUsage(session, { + inputTokens: 14, + outputTokens: 5, + cacheReadTokens: 8, + cacheWriteTokens: 1, + }, 1, 1, [source]) + + expect(projected(ctx, session)).toEqual({ + uncachedInputTokens: 14, + outputTokens: 5, + cacheReadTokens: 8, + cacheWriteTokens: 1, + }) + }) + + it('accumulates disjoint buckets across steps without adding reasoning twice', async () => { + const { ctx, session } = await harness() + startStep(session, 1, 1) + const first = usageChunk(session, { + inputTokens: 10, + outputTokens: 6, + reasoningTokens: 5, + cacheReadTokens: 2, + }, 1, 1) + finalUsage(session, { + inputTokens: 10, + outputTokens: 6, + reasoningTokens: 5, + cacheReadTokens: 2, + }, 1, 1, [first]) + startStep(session, 1, 2) + const second = usageChunk(session, { + inputTokens: 20, + outputTokens: 9, + reasoningTokens: 7, + cacheWriteTokens: 4, + }, 1, 2) + finalUsage(session, { + inputTokens: 20, + outputTokens: 9, + reasoningTokens: 7, + cacheWriteTokens: 4, + }, 1, 2, [second]) + + expect(projected(ctx, session)).toEqual({ + uncachedInputTokens: 30, + outputTokens: 15, + cacheReadTokens: 2, + cacheWriteTokens: 4, + }) + }) + + it('retains a usage chunk when the request produces no final assistant message', async () => { + const { ctx, session } = await harness() + startStep(session, 1, 1) + usageChunk(session, { inputTokens: 9, outputTokens: 1 }, 1, 1) + session.append('step/end', { turn: 1, step: 1 }) + expect(projected(ctx, session)).toEqual({ + uncachedInputTokens: 9, + outputTokens: 1, + cacheReadTokens: 0, + cacheWriteTokens: 0, + }) + }) + + it('does not erase historical billing when the visible surface is replaced', async () => { + const { ctx, session } = await harness() + startStep(session, 1, 1) + const source = usageChunk(session, { inputTokens: 12, outputTokens: 3 }, 1, 1) + finalUsage(session, { inputTokens: 12, outputTokens: 3 }, 1, 1, [source]) + const before = session.append('user/message', createUserMessage({ + content: [{ type: 'text', text: 'before compaction' }], + source: { kind: 'user' }, + }), { surfaceOp: 'append' }) + session.append('user/message', createUserMessage({ + content: [{ type: 'text', text: 'compacted' }], + source: { kind: 'plugin', plugin: 'test' }, + }), { + surfaceOp: { op: 'replace', start: before.seq, end: before.seq }, + sourceEventSeqs: [before.seq], + }) + + expect(projected(ctx, session)).toEqual({ + uncachedInputTokens: 12, + outputTokens: 3, + cacheReadTokens: 0, + cacheWriteTokens: 0, + }) + }) + + it('unregisters with the token-meter fiber and restores from a JSON checkpoint', async () => { + const { ctx, session, meterFiber } = await harness() + startStep(session, 1, 1) + usageChunk(session, { inputTokens: 8, outputTokens: 2, cacheReadTokens: 5 }, 1, 1) + const checkpoint = JSON.parse(JSON.stringify( + ctx.sessionProjections.checkpoint(session), + )) as ReturnType + + await meterFiber.dispose() + expect(ctx.sessionProjections.snapshot(session).values).not.toHaveProperty('tokenUsage') + + await ctx.plugin(TokenMeterService) + expect(ctx.sessionProjections.viewCheckpoint(checkpoint).tokenUsage).toEqual({ + uncachedInputTokens: 8, + outputTokens: 2, + cacheReadTokens: 5, + cacheWriteTokens: 0, + }) + }) +}) diff --git a/packages/llm/token-meter/tsconfig.json b/packages/llm/token-meter/tsconfig.json index 481fad6e15..92081a860b 100644 --- a/packages/llm/token-meter/tsconfig.json +++ b/packages/llm/token-meter/tsconfig.json @@ -23,6 +23,9 @@ { "path": "../../core/session" }, + { + "path": "../../session-projection/session-projection" + }, { "path": "../../support/invariants" } diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 66ea5bea26..919576a360 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -3144,6 +3144,9 @@ importers: schemastery: specifier: ^3.18.0 version: 3.18.0 + zod: + specifier: ^4.4.3 + version: 4.4.3 devDependencies: '@deepseek-ai/dsh-invariants': specifier: workspace:^ @@ -3154,6 +3157,9 @@ importers: '@deepseek-ai/dsh-session': specifier: workspace:^ version: link:../../core/session + '@deepseek-ai/dsh-session-projection': + specifier: workspace:^ + version: link:../../session-projection/session-projection cordis: specifier: ^4.0.0-rc.7 version: 4.0.0-rc.7(@cordisjs/plugin-include@1.0.4)(@cordisjs/plugin-loader@1.0.0-rc.5) From bf618dabf96d9e2f3037fdc264a308100d7fd8c0 Mon Sep 17 00:00:00 2001 From: Hypatia May Date: Wed, 29 Jul 2026 15:27:59 +0800 Subject: [PATCH 18/38] refactor(web): project usage and snapshot request context --- ...26-07-28-host-owned-web-session-metrics.md | 39 ---- ...07-28-host-owned-web-session-metrics.zh.md | 39 ---- ...token-usage-and-request-context.i18n.yaml} | 6 +- ...ojected-token-usage-and-request-context.md | 45 +++++ ...cted-token-usage-and-request-context.zh.md | 45 +++++ apps/web/tests/question-composer.e2e.ts | 2 +- .../snapshots/code-mode-round/ui.expected.md | 3 +- .../cordis-tool-round/ui.expected.md | 3 +- .../lifecycle-chrome/reloaded.expected.md | 3 +- .../live-interactions/cancel.expected.md | 3 +- .../live-interactions/error-auth.expected.md | 2 + .../live-interactions/retry.expected.md | 3 +- .../question-composer/answered.expected.md | 3 +- .../snapshots/seeded-history/ui.expected.md | 29 ++- .../snapshots/steering/mid-steer.expected.md | 5 +- .../snapshots/steering/settled.expected.md | 3 +- docs/config-catalog.md | 2 +- docs/cordis-catalog/services.md | 2 +- docs/event-producer-consumer.md | 2 +- packages/client/connection/src/client/api.ts | 2 +- .../client/connection/src/client/fixture.ts | 117 +++++++++++- .../client/connection/src/client/index.ts | 2 +- .../client/connection/tests/fixture.spec.ts | 34 +++- packages/client/runtime/README.i18n.yaml | 4 +- packages/client/runtime/README.md | 2 +- packages/client/runtime/README.zh.md | 2 +- .../src/client/sessions/conversation.ts | 16 +- .../runtime/src/client/sessions/manager.ts | 27 +-- .../runtime/src/client/sessions/session.ts | 77 +++----- .../client/runtime/tests/client-apply.spec.ts | 30 +-- packages/client/runtime/tests/fake-api.ts | 3 +- packages/client/runtime/tests/manager.spec.ts | 69 +++---- packages/client/runtime/tests/session.spec.ts | 162 ++++++---------- packages/client/test-runtime/src/fixtures.ts | 1 + .../client/ui-conversation/README.i18n.yaml | 4 +- packages/client/ui-conversation/README.md | 2 +- packages/client/ui-conversation/README.zh.md | 2 +- packages/client/ui-conversation/package.json | 2 + .../src/client/chat/ChatView.tsx | 6 +- .../src/client/chat/StatsLine.tsx | 74 ++++---- .../tests/chat-branch-tails.spec.tsx | 20 +- .../tests/chat-code-subcalls.spec.tsx | 2 +- .../tests/chat-stats-bash-sample.spec.tsx | 174 ++++++++++++------ .../ui-conversation/tests/chat-view.spec.tsx | 2 +- .../tests/gate-branch-tails.spec.tsx | 24 +-- .../ui-conversation/tests/input-bar.spec.tsx | 2 +- .../tests/input-matrix.spec.tsx | 2 +- .../tests/input-scenarios.spec.tsx | 2 +- .../ui-conversation/tests/queue-dock.spec.tsx | 2 +- .../ui-conversation/tests/skeleton.spec.tsx | 2 +- packages/client/ui-conversation/tsconfig.json | 3 + packages/core/agent-loop/README.i18n.yaml | 4 +- packages/core/agent-loop/README.md | 2 +- packages/core/agent-loop/README.zh.md | 2 +- packages/core/agent/README.i18n.yaml | 4 +- packages/core/agent/README.md | 2 +- packages/core/agent/README.zh.md | 2 +- packages/host/apiproxy/README.i18n.yaml | 4 +- packages/host/apiproxy/README.md | 4 +- packages/host/apiproxy/README.zh.md | 4 +- packages/host/apiproxy/src/api-proxy.ts | 93 +++------- .../host/apiproxy/src/api/events.schema.ts | 4 +- packages/host/apiproxy/src/api/events.ts | 28 +-- packages/host/apiproxy/src/api/index.ts | 6 +- .../host/apiproxy/src/api/sessions.schema.ts | 17 +- packages/host/apiproxy/src/api/sessions.ts | 30 +-- packages/host/apiproxy/src/session-metrics.ts | 122 ------------ .../tests/api-proxy-model-request.spec.ts | 30 ++- .../host/apiproxy/tests/rpc-schemas.spec.ts | 38 +--- .../apiproxy/tests/session-metrics.spec.ts | 143 -------------- packages/llm/llm/README.i18n.yaml | 4 +- packages/llm/llm/README.md | 2 +- packages/llm/llm/README.zh.md | 2 +- packages/llm/token-meter/README.i18n.yaml | 4 +- packages/llm/token-meter/README.md | 6 + packages/llm/token-meter/README.zh.md | 6 + packages/llm/token-meter/package.json | 4 +- pnpm-lock.yaml | 3 + 78 files changed, 748 insertions(+), 934 deletions(-) delete mode 100644 .agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.md delete mode 100644 .agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.zh.md rename .agents/notes/implemented/architecture/{2026-07-28-host-owned-web-session-metrics.i18n.yaml => 2026-07-29-projected-token-usage-and-request-context.i18n.yaml} (52%) create mode 100644 .agents/notes/implemented/architecture/2026-07-29-projected-token-usage-and-request-context.md create mode 100644 .agents/notes/implemented/architecture/2026-07-29-projected-token-usage-and-request-context.zh.md delete mode 100644 packages/host/apiproxy/src/session-metrics.ts delete mode 100644 packages/host/apiproxy/tests/session-metrics.spec.ts diff --git a/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.md b/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.md deleted file mode 100644 index f0a0bdb8ea..0000000000 --- a/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.md +++ /dev/null @@ -1,39 +0,0 @@ -# Agent Note: Host-owned Web session metrics - -Status: implemented - -English | [中文](2026-07-28-host-owned-web-session-metrics.zh.md) - -## Problem - -A Web stats line derived from the currently loaded conversation nodes is window-dependent under pagination. Compaction can replace visible content without preserving historical usage, while the selected model does not prove that a request used its route or capacity. Cache-write tokens also risk being folded into a cache-hit formula whose denominator has different semantics. - -## Decision - -The Host owns one session-level metrics projection. It incrementally folds the complete durable event log, keys settled usage by `(turn, step)`, and replaces an earlier usage record for the same key instead of double-counting chunk and message forms. Uncached input, output, cache reads, and cache writes remain four disjoint cumulative buckets. Compaction can change the current prompt surface without erasing historical usage. - -Current context pressure is the point-in-time `tokenMeter.measure(session).totalTokens`. Capacity instead belongs to the latest model request attempt observed by the current live mux connection. `LlmService.prepareCall()` retains the context metadata obtained by the exact lookup that also validates reasoning/defaults. After the final provider/model is fixed and the outer `llm/stream` call returns a handle, the loop publishes one contained `agent/model-request` notification. This boundary observes an attempt, not proof of provider I/O: preparation or a synchronous outer waterfall failure emits nothing, while short-circuit handles and later lazy adapter construction, iteration failure, or abort still count. - -The tail `session.history` response carries durable usage and pressure, while older pages omit them. Live changes use `session/metrics` mux frames. Both forms carry a durable-log revision and a projection revision; the client accepts only nondecreasing revisions and preserves metrics across older-page prepend. - -ApiProxy forwards each notification as a distinct `session/model-request` frame only to mux connections already open when dispatch occurs. It never places the frame in `session.history` or a subscription baseline. The client keeps durable `metrics` and transient `modelRequestContextWindow` as separate snapshot fields, replaces or explicitly clears the capacity on the next observed request, and clears both fields on `session/subscribed`; reconnect, restore, and a new subscription therefore start unknown until another request is observed. - -The Web stats line joins the durable projection and live capacity only at presentation. It renders uncached input, output, and cache reads separately, computes cache hit as `cacheRead / (uncachedInput + cacheRead)`, and shows current context as a percentage only when the current connection observed a capacity. Cache writes never enter that percentage. Visible nodes continue to supply only turn and step counts. - -## Alternatives considered - -**Fold the loaded node window in React.** This cannot survive pagination or compaction and duplicates durable-log semantics in a presentation package. - -**Send usage only with raw assistant events.** Reconnect and older-page stitching would still need the client to reconstruct a full-log aggregate, and duplicate usage forms would need protocol-specific repair there. - -**Reuse one total-token field for cache hit.** Cache reads, cache writes, and uncached input represent distinct provider accounting buckets; combining them would make the displayed rate misleading. - -**Query the selected route before dispatch.** Selection may never produce a request, and a second metadata lookup can race the registration-bound lookup that actually validates and dispatches the call. - -**Persist or replay the latest request capacity.** That would make a former request look current on reconnect or restore even though the new connection observed no request. The denominator is deliberately live and opportunistic. - -## Consequences - -Token totals remain stable across pagination, replay, compaction, and browser reconnect. The client stores a small detached durable projection plus one connection-local denominator instead of scanning the conversation window, and the status row remains readable for large histories through compact number formatting. - -The Host performs one incremental log fold per session and schedules durable projection updates only for usage, request-header, or surface-changing events; text and reasoning deltas do not publish metrics. A new connection omits the percentage until it observes a request with context metadata. A later request without metadata clears the denominator, while deployments without a token meter still retain the durable counters and label context unavailable instead of fabricating pressure. diff --git a/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.zh.md b/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.zh.md deleted file mode 100644 index 4535fd8a4b..0000000000 --- a/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.zh.md +++ /dev/null @@ -1,39 +0,0 @@ -# Agent Note: Host 拥有的 Web 会话指标 - -Status: implemented - -[English](2026-07-28-host-owned-web-session-metrics.md) | 中文 - -## 问题 - -Web 统计行若根据当前加载的会话节点推导指标,其结果会随分页窗口变化。压缩(compaction)可以替换可见内容,却无法保留历史用量;所选模型也不能证明某次请求实际采用了该模型的路由或容量。缓存写入 token 还可能被计入缓存命中率公式,而该公式的分母具有不同语义。 - -## 决策 - -Host 拥有一项会话级指标投影。它以增量方式归并完整的持久事件日志,按 `(turn, step)` 标识已结算用量;同一标识再次出现时,会替换较早的用量记录,而不会重复统计分片和消息两种形态。未缓存输入、输出、缓存读取与缓存写入保持为四个彼此独立的累计计数项。压缩可以改变当前提示词表层,但不会抹除历史用量。 - -当前上下文压力是即时的 `tokenMeter.measure(session).totalTokens`。容量则属于当前实时 mux 连接观察到的最新模型请求尝试。`LlmService.prepareCall()` 会保留同一次精确查询取得的上下文元数据,该查询也负责校验推理设置与默认值。最终提供方/模型确定且外层 `llm/stream` 调用返回句柄后,循环会发布一条 `agent/model-request` 通知,并收容该通知的失败。这个边界观察到的是一次尝试,并不能证明提供方 I/O 已开始:准备阶段或外层 waterfall(瀑布式事件)的同步失败不会发出通知,而短路句柄以及之后的惰性适配器构造、迭代失败或中止仍会计入。 - -`session.history` 尾页响应携带持久用量与压力,较早页面则省略这两项。实时变更使用 `session/metrics` mux 帧。两种形式都携带持久日志修订号和投影修订号;客户端只接受不减小的修订号,并在向前加载较早页面时保留指标。 - -ApiProxy 只把每条通知作为独立的 `session/model-request` 帧转发给分派发生时已经打开的 mux 连接。它绝不会把该帧放入 `session.history` 或订阅基线。客户端把持久 `metrics` 与临时 `modelRequestContextWindow` 保存在彼此独立的快照字段中,在观察到下一次请求时替换或显式清除容量,并在收到 `session/subscribed` 时清除这两个字段;因此,重连、恢复和新订阅都会从未知容量开始,直到观察到另一次请求。 - -Web 统计行只在展示时结合持久投影与实时容量。它分别呈现未缓存输入、输出与缓存读取,通过 `cacheRead / (uncachedInput + cacheRead)` 计算缓存命中率,并且只有当前连接观察到容量时,才把当前上下文显示为该容量的百分比。缓存写入绝不计入缓存命中率。可见节点仍然只提供轮次和步骤计数。 - -## 备选方案 - -**在 React 中归并已加载的节点窗口。** 此方案无法跨越分页或压缩保留数据,还会在展示包中重复实现持久日志语义。 - -**只随原始 assistant 事件发送用量。** 重连和较早页面拼接仍会要求客户端重建完整日志聚合,而且重复的用量形态需要在客户端按协议专门修复。 - -**为缓存命中率复用单一的 token 总数字段。** 缓存读取、缓存写入与未缓存输入是提供方记账中的不同计数项;将它们合并会使显示的比率产生误导。 - -**在分派前查询所选路由。** 选择操作可能永远不会产生请求;第二次元数据查询还可能与实际校验并分派调用的、绑定注册项的查询发生竞态。 - -**持久化或回放最新请求的容量。** 即使新连接没有观察到任何请求,这也会让先前请求在重连或恢复后显得仍然有效。该分母刻意只采用实时且恰好可得的数据。 - -## 后果 - -token 总量在分页、回放、压缩和浏览器重连期间保持稳定。客户端存储一项小型、脱耦的持久投影与一个连接本地分母,无需扫描会话窗口;状态行采用紧凑数字格式,因此在较长的历史记录中仍然清晰易读。 - -Host 为每个会话执行一次增量日志归并,仅为用量事件、请求头事件或表层变更事件调度持久投影更新;文本与推理(reasoning)增量不会发布指标。新连接在观察到带上下文元数据的请求之前不会显示百分比。后续不带元数据的请求会清除该分母;未部署 token 计量器时,系统仍保留持久计数器,并把上下文标示为不可用,而不会虚构压力值。 diff --git a/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.i18n.yaml b/.agents/notes/implemented/architecture/2026-07-29-projected-token-usage-and-request-context.i18n.yaml similarity index 52% rename from .agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.i18n.yaml rename to .agents/notes/implemented/architecture/2026-07-29-projected-token-usage-and-request-context.i18n.yaml index 8c72912377..6aa9498b9c 100644 --- a/.agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.i18n.yaml +++ b/.agents/notes/implemented/architecture/2026-07-29-projected-token-usage-and-request-context.i18n.yaml @@ -1,6 +1,6 @@ # Bilingual-pair consistency record (docs/i18n/README.md): the git blob hash of each # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: -# pnpm run verify-translation-pairing --write .agents/notes/implemented/architecture/2026-07-28-host-owned-web-session-metrics.md -2026-07-28-host-owned-web-session-metrics.md: f0a0bdb8ea4c1ba1f983d016c3cfde9c67a0424d -2026-07-28-host-owned-web-session-metrics.zh.md: 4535fd8a4b760aa031c11ab766b6be583b9b740d +# pnpm run verify-translation-pairing --write .agents/notes/implemented/architecture/2026-07-29-projected-token-usage-and-request-context.md +2026-07-29-projected-token-usage-and-request-context.md: 3ac8c29f7752828f2d833359293c7b1c513d75ca +2026-07-29-projected-token-usage-and-request-context.zh.md: 02b03d1516a39b7727b214b63a9039623b69c56b diff --git a/.agents/notes/implemented/architecture/2026-07-29-projected-token-usage-and-request-context.md b/.agents/notes/implemented/architecture/2026-07-29-projected-token-usage-and-request-context.md new file mode 100644 index 0000000000..3ac8c29f77 --- /dev/null +++ b/.agents/notes/implemented/architecture/2026-07-29-projected-token-usage-and-request-context.md @@ -0,0 +1,45 @@ +# Agent Note: Projected token usage and request context + +Status: implemented + +English | [中文](2026-07-29-projected-token-usage-and-request-context.zh.md) + +## Problem + +A Web stats line derived from the currently loaded conversation nodes is window-dependent under pagination. Compaction can replace visible content without preserving historical usage. Conversely, context occupancy describes one real request boundary: a selected model is only an intention, and combining token pressure from one moment with capacity resolved for another route creates a false percentage. + +These two values therefore have different lifetimes. Provider-reported billing is durable, replayable session state. Request pressure and registration-bound capacity are an opportunistic live observation that must disappear across a connection generation. + +## Decision + +`@deepseek-ai/dsh-token-meter` registers the generic `tokenUsage` session projection when `ctx.sessionProjections` is present. The projection folds the complete durable log into uncached input, output, cache-read, and cache-write buckets. An `assistant/chunk` usage sample survives a later failed request; an `assistant/message` usage value replaces the earlier value for the same `(turn, step)` instead of being counted twice. Reasoning tokens remain an output subdivision and are not added again. Compaction and surface replacement do not erase earlier billing. + +The projection uses the standard projection lifecycle and wire path. History tail baselines, `session/projection` live frames, higher-seq-wins client storage, JSON checkpoints, cache recovery, and unit unload all remain generic. There is no token-specific history field, mux frame, projector, revision counter, or client fence. + +`LlmService.prepareCall()` retains context metadata from the exact lookup that also validates reasoning and captures the adapter registration. After the outer stream call returns its handle and before iteration begins, AgentLoop emits one contained `agent/model-request` notification. Preparation or a synchronous outer waterfall failure emits nothing; short-circuit handles and later iterator construction, iteration, or abort failures still count as an observed request attempt. + +ApiProxy handles that notification synchronously. It reads `tokenMeter.measure(agent.session).totalTokens` once when the optional service is present and combines the result with the same prepared call's registration-bound `contextWindow`. It broadcasts one atomic `session/model-request` frame containing the route, turn, step, and whichever of `contextTokens` and `contextWindow` are available. Measurement failure omits only the numerator. The frame goes only to mux connections already open at that instant; history, subscription baselines, reconnect, and restore never replay it. + +The client stores the complete latest request frame as `ConversationSnapshot.modelRequest`. Every later frame replaces the entire snapshot, so omitted fields clear earlier values. `SessionManager` temporarily holds a pre-instantiation frame, while a new subscription generation, disconnect, or session removal clears both resident and pending values. Model selection alone does not change this snapshot. + +The Web `StatsLine` reads `tokenUsage` through the standard `useProjection` hook and reads request telemetry plus visible nodes through `useSession`. It renders uncached input, output, and cache reads separately, computes cache hit as `cacheRead / (uncachedInput + cacheRead)`, and shows context occupancy only when one request snapshot contains both numerator and capacity. Visible nodes continue to supply only turn and step counts. The existing inline text UI is retained; the model selector gains no circle or other accessory. + +## Alternatives considered + +**A custom session metrics history field and mux frame.** This duplicated the generic projection protocol, cache, recovery, and seq fencing while coupling durable billing to transient request pressure. + +**Fold the loaded node window in React.** This cannot survive pagination or compaction and makes a presentation package reconstruct log semantics. + +**Publish usage only with final assistant messages.** A request that reports a usage chunk and then fails would lose provider billing. + +**Query capacity from the selected model.** Selection may never produce a request, and a second metadata lookup can disagree with the registration-bound lookup used by the actual call. + +**Persist or replay the latest request snapshot.** A request from a prior connection would appear current after restore even though no new request was observed. + +**Add a context circle beside the model selector.** That placement suggests selected-model state. The existing stats line expresses the request-scoped semantics without introducing a duplicate UI or data path. + +## Consequences + +Token totals stay stable across pagination, compaction, replay, and reconnect because they are ordinary durable projection state. Context occupancy is deliberately unknown after reconnect until a new real request is observed. Deployments without token-meter or without model capacity still publish the request route and clear stale optional fields instead of fabricating a percentage. + +ApiProxy performs one synchronous optional measurement and one frame conversion per observed request. It owns no per-session metrics cache or refresh queue. The browser keeps one generic projection value plus one small connection-local request snapshot, and streaming text deltas do not force the stats line to recompute. diff --git a/.agents/notes/implemented/architecture/2026-07-29-projected-token-usage-and-request-context.zh.md b/.agents/notes/implemented/architecture/2026-07-29-projected-token-usage-and-request-context.zh.md new file mode 100644 index 0000000000..02b03d1516 --- /dev/null +++ b/.agents/notes/implemented/architecture/2026-07-29-projected-token-usage-and-request-context.zh.md @@ -0,0 +1,45 @@ +# Agent Note:token 用量投影与请求上下文 + +Status: implemented + +[English](2026-07-29-projected-token-usage-and-request-context.md) | 中文 + +## 问题 + +Web 统计行若根据当前已加载的会话节点推导,其结果会在分页时依赖当前窗口。压缩(compaction)可以替换可见内容,却不保留历史用量。另一方面,上下文占用率描述的是一个真实请求边界:所选模型只代表意图;若把某一时刻的 token 压力与另一路由解析出的容量组合,就会产生虚假百分比。 + +因此,这两个值具有不同的生命周期。提供方报告的计费用量属于持久、可回放的会话状态。请求压力和与注册项绑定的容量则是恰好可得的实时观测,跨连接代次时必须消失。 + +## 决策 + +当 `ctx.sessionProjections` 存在时,`@deepseek-ai/dsh-token-meter` 会注册通用的 `tokenUsage` 会话投影。该投影将完整持久日志归并为未缓存输入、输出、缓存读取和缓存写入四类计数项。即使后续请求失败,`assistant/chunk` 用量样本仍会保留;同一 `(turn, step)` 的 `assistant/message` 用量值会替换先前值,不会重复计数。推理(reasoning)token 仍是输出的细分项,不会再次累加。压缩和表层替换不会抹除先前的计费用量。 + +该投影使用标准的投影生命周期与协议路径。历史尾页基线、`session/projection` 实时帧、seq 高者胜的客户端存储、JSON 检查点、缓存恢复和单元卸载均保持通用机制。系统没有任何 token 专用的历史字段、mux 帧、投影器、修订计数器或客户端 seq 防护机制。 + +`LlmService.prepareCall()` 会保留精确查询得到的上下文元数据;同一次查询还会校验推理强度并捕获适配器注册项。外层流调用返回句柄后、开始迭代前,AgentLoop 会发出一条失败受收容的 `agent/model-request` 通知。准备阶段或外层 waterfall(瀑布式事件)的同步失败不会发出通知;短路句柄以及之后的迭代器构造失败、迭代失败或中止仍算作一次已观测的请求尝试。 + +ApiProxy 会同步处理该通知。当可选服务存在时,它会读取一次 `tokenMeter.measure(agent.session).totalTokens`,并将结果与同一准备完成调用中绑定注册项的 `contextWindow` 合并。它会广播一个原子 `session/model-request` 帧,其中包含路由、轮次、步骤,以及可用的 `contextTokens`/`contextWindow` 字段。测量失败时只省略分子。该帧只发送给当时已经打开的 mux 连接;历史记录、订阅基线、重连和恢复都绝不回放该帧。 + +客户端将最新的完整请求帧存储为 `ConversationSnapshot.modelRequest`。每个后续帧都会替换整个快照,因此省略字段会清除先前值。`SessionManager` 会临时保存一个实例化前帧;新订阅代次、断开连接或移除会话时,则会同时清除常驻值和待处理值。仅选择模型不会改变该快照。 + +Web `StatsLine` 通过标准 `useProjection` 钩子读取 `tokenUsage`,并通过 `useSession` 读取请求观测数据与可见节点。它分别显示未缓存输入、输出与缓存读取,通过 `cacheRead / (uncachedInput + cacheRead)` 计算缓存命中率,并且只有同一份请求快照同时包含分子与容量时才显示上下文占用率。可见节点仍只提供轮次和步骤计数。系统保留现有的行内文本 UI;模型选择器不增加圆环或其他附属控件。 + +## 备选方案 + +**自定义会话指标历史字段和 mux 帧。** 这会重复实现通用投影协议、缓存、恢复和 seq 防护机制,并将持久计费用量与临时请求压力耦合。 + +**在 React 中归并已加载的节点窗口。** 此方案无法跨分页或压缩保留数据,还会迫使展示包重建日志语义。 + +**仅随最终 assistant 消息发布用量。** 如果请求报告一个用量分片后失败,就会丢失提供方计费用量。 + +**从所选模型查询容量。** 选择操作可能永远不会产生请求;第二次元数据查询还可能与实际调用所用的、绑定注册项的查询不一致。 + +**持久化或回放最新请求快照。** 即使没有观察到任何新请求,先前连接的请求也会在恢复后显得仍是当前请求。 + +**在模型选择器旁增加上下文圆环。** 该位置会让人以为这是所选模型的状态。现有统计行可以表达按请求作用域的语义,无需引入重复的 UI 或数据路径。 + +## 后果 + +token 总量在分页、压缩、回放和重连期间保持稳定,因为它们属于普通的持久投影状态。重连后,上下文占用率会刻意保持未知,直到系统观察到新的真实请求。未部署 token-meter 或模型不提供容量时,系统仍会发布请求路由,并清除陈旧的可选字段,而不会虚构百分比。 + +ApiProxy 会为每次已观测请求执行一次可选的同步测量和一次帧转换。它不拥有任何逐会话指标缓存或刷新队列。浏览器只保留一个通用投影值和一个小型连接本地请求快照;流式文本增量不会迫使统计行重新计算。 diff --git a/apps/web/tests/question-composer.e2e.ts b/apps/web/tests/question-composer.e2e.ts index 6ecdb683b3..8689a74fc2 100644 --- a/apps/web/tests/question-composer.e2e.ts +++ b/apps/web/tests/question-composer.e2e.ts @@ -104,7 +104,7 @@ describe('web e2e: resident question composer round trip', () => { const inner = child.getBoundingClientRect() return Math.max(box.top - inner.top, inner.bottom - box.bottom) }))) - const list = rows[0]?.parentElement ?? null + const list = card.querySelector('[data-question-scroll]') return { rows: rows.length, spill: Math.max(...spill), diff --git a/apps/web/tests/snapshots/code-mode-round/ui.expected.md b/apps/web/tests/snapshots/code-mode-round/ui.expected.md index 4847924619..dccd3fd7dc 100644 --- a/apps/web/tests/snapshots/code-mode-round/ui.expected.md +++ b/apps/web/tests/snapshots/code-mode-round/ui.expected.md @@ -12,6 +12,7 @@ - img - button "编辑": - img +- button "▸ 上下文注入" - 'button "Think The user wants me to write a single `run_code` program that:"': - img - img @@ -28,7 +29,7 @@ - img - text: Think The program ran successfully. Let me now reply DONE as instructed. - paragraph: DONE -- text: cache hit 52% · 17,490 tokens · 1 turns · 2 steps +- text: 8.3k uncached input · 252 output · 9k cache read · cache hit 52% · context 7% of 128k · 1 turns · 2 steps - textbox "Message the agent" - button "Add attachment": - img diff --git a/apps/web/tests/snapshots/cordis-tool-round/ui.expected.md b/apps/web/tests/snapshots/cordis-tool-round/ui.expected.md index e5e5626be3..b9c1ea7b2d 100644 --- a/apps/web/tests/snapshots/cordis-tool-round/ui.expected.md +++ b/apps/web/tests/snapshots/cordis-tool-round/ui.expected.md @@ -12,6 +12,7 @@ - img - button "编辑": - img +- button "▸ 上下文注入" - button "Think The user wants me to:": - img - img @@ -42,7 +43,7 @@ - img - text: Think All three calls succeeded. I should now reply exactly "CORDIS_UI_DONE" and stop. - paragraph: CORDIS_UI_DONE -- text: cache hit 77% · 66,813 tokens · 1 turns · 4 steps +- text: 15.3k uncached input · 312 output · 51.2k cache read · cache hit 77% · context 13% of 128k · 1 turns · 4 steps - textbox "Message the agent" - button "Add attachment": - img diff --git a/apps/web/tests/snapshots/lifecycle-chrome/reloaded.expected.md b/apps/web/tests/snapshots/lifecycle-chrome/reloaded.expected.md index 33d1f7e6bf..1a0b740ce6 100644 --- a/apps/web/tests/snapshots/lifecycle-chrome/reloaded.expected.md +++ b/apps/web/tests/snapshots/lifecycle-chrome/reloaded.expected.md @@ -12,12 +12,13 @@ - img - button "编辑": - img +- button "▸ 上下文注入" - button "Think The user wants me to reply with a single word. Let me comply.": - img - img - text: Think The user wants me to reply with a single word. Let me comply. - paragraph: LIGHTHOUSE -- text: cache hit 99% · 7,810 tokens · 1 turns · 1 steps +- text: 109 uncached input · 21 output · 7.7k cache read · cache hit 99% · context unknown · 1 turns · 1 steps - textbox "Message the agent" - button "Add attachment": - img diff --git a/apps/web/tests/snapshots/live-interactions/cancel.expected.md b/apps/web/tests/snapshots/live-interactions/cancel.expected.md index 3d092b17ec..bf87981ea2 100644 --- a/apps/web/tests/snapshots/live-interactions/cancel.expected.md +++ b/apps/web/tests/snapshots/live-interactions/cancel.expected.md @@ -12,8 +12,9 @@ - img - button "编辑": - img +- button "▸ 上下文注入" - paragraph: partial -- text: 已停止 0 tokens · 1 turns · 1 steps +- text: 已停止 0 uncached input · 0 output · 0 cache read · context 4% of 128k · 1 turns · 1 steps - textbox "Message the agent" - button "Add attachment": - img diff --git a/apps/web/tests/snapshots/live-interactions/error-auth.expected.md b/apps/web/tests/snapshots/live-interactions/error-auth.expected.md index 5272bcf2d1..9fe1195192 100644 --- a/apps/web/tests/snapshots/live-interactions/error-auth.expected.md +++ b/apps/web/tests/snapshots/live-interactions/error-auth.expected.md @@ -12,6 +12,8 @@ - img - button "编辑": - img +- button "▸ 上下文注入" +- text: 0 uncached input · 0 output · 0 cache read · context 4% of 128k · 0 turns · 0 steps - textbox "Message the agent" - button "Add attachment": - img diff --git a/apps/web/tests/snapshots/live-interactions/retry.expected.md b/apps/web/tests/snapshots/live-interactions/retry.expected.md index 5935872557..6ad6d27487 100644 --- a/apps/web/tests/snapshots/live-interactions/retry.expected.md +++ b/apps/web/tests/snapshots/live-interactions/retry.expected.md @@ -12,12 +12,13 @@ - img - button "编辑": - img +- button "▸ 上下文注入" - button "Think The user is asking for a one-sentence description of event sourcing. This is a straightforward knowledge question that doesn't require any skill loading or tool calls.": - img - img - text: Think The user is asking for a one-sentence description of event sourcing. This is a straightforward knowledge question that doesn't require any skill loading or tool calls. - paragraph: Event sourcing is a pattern where all changes to an application's state are stored as an immutable, append-only sequence of events, rather than persisting only the current state, enabling full auditability, temporal queries, and event-driven architectures. -- text: cache hit 99% · 7,869 tokens · 1 turns · 1 steps +- text: 110 uncached input · 79 output · 7.7k cache read · cache hit 99% · context 4% of 128k · 1 turns · 1 steps - textbox "Message the agent" - button "Add attachment": - img diff --git a/apps/web/tests/snapshots/question-composer/answered.expected.md b/apps/web/tests/snapshots/question-composer/answered.expected.md index 91ff2cdf88..a243c38c48 100644 --- a/apps/web/tests/snapshots/question-composer/answered.expected.md +++ b/apps/web/tests/snapshots/question-composer/answered.expected.md @@ -12,6 +12,7 @@ - img - button "编辑": - img +- button "▸ 上下文注入" - button "Think The user wants me to use the ask_user_question tool with specific parameters. Let me do exactly that.": - img - img @@ -25,7 +26,7 @@ - img - text: Think The user answered "Blue". I should now reply with the single word DONE and stop. - paragraph: DONE -- text: cache hit 95% · 8,769 tokens · 1 turns · 2 steps +- text: 397 uncached input · 180 output · 8.2k cache read · cache hit 95% · context 4% of 128k · 1 turns · 2 steps - textbox "Message the agent" - button "Add attachment": - img diff --git a/apps/web/tests/snapshots/seeded-history/ui.expected.md b/apps/web/tests/snapshots/seeded-history/ui.expected.md index 4f9181f702..ac9c92dbfd 100644 --- a/apps/web/tests/snapshots/seeded-history/ui.expected.md +++ b/apps/web/tests/snapshots/seeded-history/ui.expected.md @@ -1,32 +1,41 @@ - banner: - navigation "Session hierarchy": - button "Use the read tool twice" [disabled] - - text: · 1 turns - tablist: - tab "Chat" [selected] - tab "Trajectory" - tab "Waterfall" - text: "Use the read tool twice in one assistant message: read a.txt and b.txt. Then reply with the single word DONE and stop." +- button "复制": + - img +- button "在新对话中分支": + - img +- button "编辑": + - img - button "Think The user wants me to read a.txt and b.txt, then reply with \"DONE\". Let me do both reads in parallel.": + - img - img - text: Think The user wants me to read a.txt and b.txt, then reply with "DONE". Let me do both reads in parallel. -- button: - - img -- text: Read a.txt -- button: - - img -- text: Read b.txt +- img +- text: Read +- button "a.txt" +- img +- text: Read +- button "b.txt" - button "Think Both files have been read. a.txt contains \"alpha\" and b.txt contains \"beta\". I'll now reply with DONE as instructed.": + - img - img - text: Think Both files have been read. a.txt contains "alpha" and b.txt contains "beta". I'll now reply with DONE as instructed. - paragraph: DONE -- text: cache hit 98% · 15,962 tokens · 1 turns · 2 steps +- text: 339 uncached input · 135 output · 15.5k cache read · cache hit 98% · context unknown · 1 turns · 2 steps - textbox "Message the agent" - button "Add attachment": - img +- text: Danger Full Access - combobox "Access mode": - - option "Read-only" [selected] - - option "Read-write" + - option "Read Only" + - option "Workspace Write" + - option "Danger Full Access" [selected] - button "选择模型,当前 deepseek-v4-flash": - text: deepseek-v4-flash - img diff --git a/apps/web/tests/snapshots/steering/mid-steer.expected.md b/apps/web/tests/snapshots/steering/mid-steer.expected.md index 8d33ea6283..bbfae558c9 100644 --- a/apps/web/tests/snapshots/steering/mid-steer.expected.md +++ b/apps/web/tests/snapshots/steering/mid-steer.expected.md @@ -12,6 +12,7 @@ - img - button "编辑": - img +- button "▸ 上下文注入" - button "Think The user wants me to use the ask_user_question tool to ask them a specific question with the given parameters. Let me do exactly that.": - img - img @@ -19,9 +20,7 @@ - button: - img - img -- text: "Tool call ask_user_question · {\"questions\": [{\"id\": \"checkpoint\", \"question\": \"Ready to continue?\", \"header\": \"Checkpoint\", \"options\": [{\"label\": \"Yes\"}, {\"label\": \"No\"}]}]} 等待回答(1 题)" -- button "▸ 问题内容" -- text: cache hit 98% · 7,946 tokens · 1 turns · 1 steps +- text: "Tool call ask_user_question · {\"questions\": [{\"id\": \"checkpoint\", \"question\": \"Ready to continue?\", \"header\": \"Checkpoint\", \"options\": [{\"label\": \"Yes\"}, {\"label\": \"No\"}]}]} 151 uncached input · 115 output · 7.7k cache read · cache hit 98% · context 4% of 128k · 1 turns · 1 steps" - region "Ready to continue?": - text: Checkpoint - heading "Ready to continue?" [level=2] diff --git a/apps/web/tests/snapshots/steering/settled.expected.md b/apps/web/tests/snapshots/steering/settled.expected.md index f08fc518e8..f6ed68511a 100644 --- a/apps/web/tests/snapshots/steering/settled.expected.md +++ b/apps/web/tests/snapshots/steering/settled.expected.md @@ -12,6 +12,7 @@ - img - button "编辑": - img +- button "▸ 上下文注入" - button "Think The user wants me to use the ask_user_question tool to ask them a specific question with the given parameters. Let me do exactly that.": - img - img @@ -25,7 +26,7 @@ - img - text: Think The user selected "Yes" and wants me to include the word "BANANA" in my final reply. Let me acknowledge their answer. - paragraph: Great, let's move forward. BANANA! -- text: cache hit 98% · 15,967 tokens · 1 turns · 2 steps +- text: 323 uncached input · 156 output · 15.5k cache read · cache hit 98% · context 6% of 128k · 1 turns · 2 steps - textbox "Message the agent" - button "Add attachment": - img diff --git a/docs/config-catalog.md b/docs/config-catalog.md index 3c0543e4ca..1026e1478b 100644 --- a/docs/config-catalog.md +++ b/docs/config-catalog.md @@ -1548,7 +1548,7 @@ Source: [`packages/context/time-context/src/index.ts:20`](../packages/context/ti export type TokenMeterConfig = Record ``` -Source: [`packages/llm/token-meter/src/types.ts:10`](../packages/llm/token-meter/src/types.ts) +Source: [`packages/llm/token-meter/src/types.ts:12`](../packages/llm/token-meter/src/types.ts) ## `@deepseek-ai/dsh-tool-bash` diff --git a/docs/cordis-catalog/services.md b/docs/cordis-catalog/services.md index 31574e0eb4..3ab05fc0a1 100644 --- a/docs/cordis-catalog/services.md +++ b/docs/cordis-catalog/services.md @@ -2031,7 +2031,7 @@ estimateMessage(message: Message): number Types: [EpochHeader](../core-data-structures/session.md) · [Message](../core-data-structures/core.md) · [Session](../core-data-structures/session.md) · [TokenMeasurement](../core-data-structures/token-meter.md) -Source: [`packages/llm/token-meter/src/index.ts:82`](../../packages/llm/token-meter/src/index.ts) +Source: [`packages/llm/token-meter/src/index.ts:85`](../../packages/llm/token-meter/src/index.ts) ## `ctx.toolResultPrune` — `ToolResultPruneService` diff --git a/docs/event-producer-consumer.md b/docs/event-producer-consumer.md index 1c36786cce..391059ec2a 100644 --- a/docs/event-producer-consumer.md +++ b/docs/event-producer-consumer.md @@ -9,7 +9,7 @@ This matrix shows which packages dispatch each harness-owned event and which pac | --- | --- | --- | --- | --- | | `agent-loop/config-start-failed` | `emit` | [`packages/core/agent-loop/src/index.ts:148`](../packages/core/agent-loop/src/index.ts) | [`agent-loop`](../packages/core/agent-loop) (`events.dispatch`) | [`tui`](../packages/ui/tui) | | `agent/cancel-requested` | `emit` | [`packages/core/agent/src/types.ts:296`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | [`goal-session`](../packages/goal/goal-session) | -| `agent/created` | `emit` | [`packages/core/agent/src/types.ts:228`](../packages/core/agent/src/types.ts) | [`agent`](../packages/core/agent) (`events.dispatch`) | `apiproxy`, [`goal-session`](../packages/goal/goal-session), [`tui`](../packages/ui/tui) | +| `agent/created` | `emit` | [`packages/core/agent/src/types.ts:228`](../packages/core/agent/src/types.ts) | [`agent`](../packages/core/agent) (`events.dispatch`) | [`goal-session`](../packages/goal/goal-session), [`tui`](../packages/ui/tui) | | `agent/disposed` | `emit` | [`packages/core/agent/src/types.ts:237`](../packages/core/agent/src/types.ts) | [`agent`](../packages/core/agent) (`events.dispatch`) | [`agent-loop`](../packages/core/agent-loop), [`goal-session`](../packages/goal/goal-session), [`tui`](../packages/ui/tui) | | `agent/error` | `emit` | [`packages/core/agent/src/types.ts:424`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | `apiproxy`, [`goal-session`](../packages/goal/goal-session), [`session-telemetry`](../packages/telemetry/session-telemetry), [`tui`](../packages/ui/tui) | | `agent/inbox/dequeue` | `emit` | [`packages/core/agent/src/types.ts:269`](../packages/core/agent/src/types.ts) | [`agent-loop`](../packages/core/agent-loop) (`emitAgentEvent`) | [`agent`](../packages/core/agent), `apiproxy`, [`tui`](../packages/ui/tui) | diff --git a/packages/client/connection/src/client/api.ts b/packages/client/connection/src/client/api.ts index 05ad4325ff..a773269257 100644 --- a/packages/client/connection/src/client/api.ts +++ b/packages/client/connection/src/client/api.ts @@ -12,7 +12,7 @@ export type { WorkspaceApi, WorkspaceId, WorkspaceView, CommandsApi, CommandDescriptor, SkillsApi, SkillEntry, ModelCatalogFailure, ModelCatalogModel, ModelProviderGroup, ModelReasoning, - ModelReasoningEffort, ModelTarget, SessionMetrics, SessionModels, SessionProjectionsBlock, + ModelReasoningEffort, ModelRequestTelemetry, ModelTarget, SessionModels, SessionProjectionsBlock, GoalsApi, GoalRef, } from '@deepseek-ai/dsh-host-apiproxy/api' export type { ToolCallView, ToolResultView } from '@deepseek-ai/dsh-tools/presentation' diff --git a/packages/client/connection/src/client/fixture.ts b/packages/client/connection/src/client/fixture.ts index cab616aabd..0843341191 100644 --- a/packages/client/connection/src/client/fixture.ts +++ b/packages/client/connection/src/client/fixture.ts @@ -15,6 +15,7 @@ import type { AssistantMessage, ContentBlock, MessageSource, + TokenUsage, ToolResultMessage, UserMessage, } from '@deepseek-ai/dsh-llm' @@ -103,6 +104,16 @@ function sid(id: string): SessionId { return id as SessionId } +/** Deterministic provider billing attached to fixture assistant messages. */ +function fixtureUsage(turn: number, step: number): TokenUsage { + return { + inputTokens: 20 + turn % 5, + outputTokens: 8 + step, + cacheReadTokens: turn === 0 ? 0 : 80, + cacheWriteTokens: turn % 10 === 0 ? 4 : 0, + } +} + /** fx-alpha history script: 60 turns (~130+ messages -> 3 pages at PAGE_MESSAGES=50), * mixing reasoning blocks / tool call+result / steering / context. */ function buildAlphaLog(): SessionEvent[] { @@ -110,7 +121,17 @@ function buildAlphaLog(): SessionEvent[] { let time = Date.now() - 3_600_000 const push = (e: Record): number => { const seq = events.length - events.push({ seq, time: (time += 800), ...e }) + const data = e['data'] as Record | undefined + const authored = e['type'] === 'assistant/message' && data !== undefined + ? { + ...e, + data: { + ...data, + usage: fixtureUsage(data['turn'] as number, data['step'] as number), + }, + } + : e + events.push({ seq, time: (time += 800), ...authored }) return seq } for (let turn = 0; turn < 60; turn++) { @@ -369,6 +390,60 @@ function permissionSelectOf( } } +interface FixtureTokenUsageProjection { + uncachedInputTokens: number + outputTokens: number + cacheReadTokens: number + cacheWriteTokens: number +} + +/** Fixture parallel of token-meter's last-sample-replacing usage projection. */ +function tokenUsageOf(log: readonly SessionEvent[]): FixtureTokenUsageProjection { + const totals: FixtureTokenUsageProjection = { + uncachedInputTokens: 0, + outputTokens: 0, + cacheReadTokens: 0, + cacheWriteTokens: 0, + } + let last: { + turn: number + step: number + buckets: FixtureTokenUsageProjection + } | null = null + for (const event of log) { + const item = event as unknown as { + type: string + data: { + turn?: number + step?: number + usage?: TokenUsage + chunk?: { type?: string; usage?: TokenUsage } + } + } + const usage = item.type === 'assistant/chunk' && item.data.chunk?.type === 'usage' + ? item.data.chunk.usage + : item.type === 'assistant/message' + ? item.data.usage + : undefined + if (usage === undefined || item.data.turn === undefined || item.data.step === undefined) continue + const buckets: FixtureTokenUsageProjection = { + uncachedInputTokens: usage.inputTokens, + outputTokens: usage.outputTokens, + cacheReadTokens: usage.cacheReadTokens ?? 0, + cacheWriteTokens: usage.cacheWriteTokens ?? 0, + } + const previous = last?.turn === item.data.turn && last.step === item.data.step + ? last.buckets + : undefined + totals.uncachedInputTokens += buckets.uncachedInputTokens - (previous?.uncachedInputTokens ?? 0) + totals.outputTokens += buckets.outputTokens - (previous?.outputTokens ?? 0) + totals.cacheReadTokens += buckets.cacheReadTokens - (previous?.cacheReadTokens ?? 0) + totals.cacheWriteTokens += buckets.cacheWriteTokens - (previous?.cacheWriteTokens ?? 0) + last = { turn: item.data.turn, step: item.data.step, buckets } + } + return totals +} + function projectionValuesOf(log: readonly SessionEvent[]): Record { const values: Record = {} const titleEvent = log.findLast(item => (item as { type: string }).type === 'session/title') @@ -383,12 +458,28 @@ function projectionValuesOf(log: readonly SessionEvent[]): Record[] { const type = (event as { type: string }).type + if ( + (type === 'assistant/chunk' + && (event as unknown as { data: { chunk?: { type?: string } } }).data.chunk?.type === 'usage') + || (type === 'assistant/message' + && (event as unknown as { data: { usage?: TokenUsage } }).data.usage !== undefined) + ) { + return [{ + type: 'session/projection', + sessionId: id, + key: 'tokenUsage', + value: tokenUsageOf(log), + seq: event.seq, + }] + } if (type === 'session/title') { const values = projectionValuesOf(log) /* v8 ignore next -- the advancing title event is in the log, so the key is present. */ @@ -853,7 +944,16 @@ export function createFixtureApi(options: FixtureOptions = {}): ApiProxy { replays.delete(id) const done = pieces.slice(0, i).join('') append(id, { type: 'assistant/chunk', data: { turn, step, chunk: { type: 'block-end', index: 0, block: { type: 'text', text: done } } } }) - append(id, { type: 'assistant/message', surfaceOp: 'append', data: { turn, step, message: assistantMessage(text(aborted ? `${done}(已中断)` : done)) } }) + append(id, { + type: 'assistant/message', + surfaceOp: 'append', + data: { + turn, + step, + message: assistantMessage(text(aborted ? `${done}(已中断)` : done)), + usage: fixtureUsage(turn, step), + }, + }) append(id, { type: 'step/end', data: { turn, step } }) append(id, { type: 'turn/end', data: { turn, reason: { kind: aborted ? 'cancelled' : 'completed' } } }) setRunning(id, false) @@ -1036,6 +1136,19 @@ export function createFixtureApi(options: FixtureOptions = {}): ApiProxy { append(id, { type: 'plan/mode', data: { active: plan.wanted } }) } append(id, { type: 'user/message', surfaceOp: 'append', data: userMessage(content) }) + const target = modelTargets.get(id) ?? { provider: 'deepseek', model: 'deepseek-v4-flash' } + const usage = tokenUsageOf(logOf(id)) + emitMux({ + type: 'session/model-request', + sessionId: id, + turn, + step: 0, + provider: target.provider, + model: target.model, + contextTokens: usage.uncachedInputTokens + usage.outputTokens + + usage.cacheReadTokens + usage.cacheWriteTokens, + contextWindow: 128_000, + }) startReply( id, turn, diff --git a/packages/client/connection/src/client/index.ts b/packages/client/connection/src/client/index.ts index 29cb02c6fa..4097a23036 100644 --- a/packages/client/connection/src/client/index.ts +++ b/packages/client/connection/src/client/index.ts @@ -17,7 +17,7 @@ export type { ToolCallView, ToolResultView, WorkspaceApi, WorkspaceId, WorkspaceView, CommandsApi, CommandDescriptor, SkillsApi, SkillEntry, ModelCatalogFailure, ModelCatalogModel, ModelProviderGroup, ModelReasoning, - ModelReasoningEffort, ModelTarget, SessionMetrics, SessionModels, SessionProjectionsBlock, + ModelReasoningEffort, ModelRequestTelemetry, ModelTarget, SessionModels, SessionProjectionsBlock, RpcRequest, RpcResponse, RpcResult, RpcError, RpcErrorCode, ClientRequest, ServerResponse, ServerRequest, ClientResponse, RpcMessage, RpcReceipt, IApiClient, SessionId, SessionEvent, ContentBlock, StreamChunk, diff --git a/packages/client/connection/tests/fixture.spec.ts b/packages/client/connection/tests/fixture.spec.ts index 09a3efecd3..d25a4974ca 100644 --- a/packages/client/connection/tests/fixture.spec.ts +++ b/packages/client/connection/tests/fixture.spec.ts @@ -84,6 +84,12 @@ describe('createFixtureApi', () => { }, plan: { active: false, pending: false }, goal: null, + tokenUsage: { + uncachedInputTokens: 0, + outputTokens: 0, + cacheReadTokens: 0, + cacheWriteTokens: 0, + }, } }, }) }) @@ -188,6 +194,20 @@ describe('createFixtureApi', () => { expect(types).toContain('assistant/chunk') expect(types).toContain('assistant/message') expect(types.at(-1)).toBe('turn/end') + expect(frames).toContainEqual({ + type: 'session/model-request', + sessionId: id, + turn: 0, + step: 0, + provider: 'deepseek', + model: 'deepseek-v4-flash', + contextTokens: 0, + contextWindow: 128_000, + }) + expect(frames.some(frame => + frame.type === 'session/projection' + && frame.key === 'tokenUsage' + && (frame.value as { outputTokens?: number }).outputTokens === 8)).toBe(true) const finalize = frames.find((f): f is Extract => f.type === 'session/event' && f.event.type === 'assistant/message') expect(JSON.stringify(finalize?.event.data)).toContain('(已中断)') // Idle cancel: no replay in flight, must not explode; running flips false. @@ -219,7 +239,7 @@ describe('createFixtureApi', () => { const envelopes: RpcRequest[] = [] for await (const envelope of api.events.mux(req({}), abort.signal)) { envelopes.push(envelope) - if (envelopes.length >= 8) abort.abort() + if (envelopes.length >= 9) abort.abort() } return envelopes } @@ -227,16 +247,18 @@ describe('createFixtureApi', () => { const second = await openOnce() expect(first[0]?.payload).toMatchObject({ type: 'session/subscribed', sessionId: 'fx-alpha' }) expect((first[0]?.payload as { lastSeq: number }).lastSeq).toBeGreaterThan(0) - // Projection baseline frames follow the subscribed frame (title + todos + permissions + plan + goal units). + // Projection baseline frames follow subscribed (domain units + token usage). expect(first[1]?.payload).toMatchObject({ type: 'session/projection', sessionId: 'fx-alpha', key: 'title', value: 'Fixture 历史会话' }) expect(first[2]?.payload).toMatchObject({ type: 'session/projection', sessionId: 'fx-alpha', key: 'todos' }) expect(first[3]?.payload).toMatchObject({ type: 'session/projection', sessionId: 'fx-alpha', key: 'permissions' }) expect(first[4]?.payload).toMatchObject({ type: 'session/projection', sessionId: 'fx-alpha', key: 'plan', value: { active: false, pending: false } }) expect(first[5]?.payload).toMatchObject({ type: 'session/projection', sessionId: 'fx-alpha', key: 'goal', value: null }) - expect(first[6]?.payload).toMatchObject({ type: 'approval/requested', toolName: 'dangerous_tool' }) - expect(second[6]?.rpcId).toBe(first[6]?.rpcId) // stable rpcId across replays (host replay semantics) - expect(first[7]?.payload).toMatchObject({ type: 'question/requested', sessionId: 'fx-alpha' }) - expect(second[7]?.rpcId).toBe(first[7]?.rpcId) + expect(first[6]?.payload).toMatchObject({ type: 'session/projection', sessionId: 'fx-alpha', key: 'tokenUsage' }) + expect(first[7]?.payload).toMatchObject({ type: 'approval/requested', toolName: 'dangerous_tool' }) + expect(second[7]?.rpcId).toBe(first[7]?.rpcId) // stable rpcId across replays (host replay semantics) + expect(first[8]?.payload).toMatchObject({ type: 'question/requested', sessionId: 'fx-alpha' }) + expect(second[8]?.rpcId).toBe(first[8]?.rpcId) + expect(first.some(envelope => envelope.payload.type === 'session/model-request')).toBe(false) }) it('steer with no replay in flight falls through to a fresh queued turn; non-text blocks stringify empty', async () => { diff --git a/packages/client/runtime/README.i18n.yaml b/packages/client/runtime/README.i18n.yaml index e74dacded2..9f33a7e7ae 100644 --- a/packages/client/runtime/README.i18n.yaml +++ b/packages/client/runtime/README.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write packages/client/runtime/README.md -README.md: 879730733f3d89c3962d8c54bcfd53795a980049 -README.zh.md: 1b1d7f03d1f5f03c054dfeaa790a9f6f91e0dca2 +README.md: ba9a7d455e8a193f23884411eb1928a10f21ddd0 +README.zh.md: 873fefca48585efed010589917f2c63653b08e5b diff --git a/packages/client/runtime/README.md b/packages/client/runtime/README.md index 879730733f..ba9a7d455e 100644 --- a/packages/client/runtime/README.md +++ b/packages/client/runtime/README.md @@ -2,7 +2,7 @@ English | [中文](README.zh.md) -Client cordis boot and React-free object services: SlotsService wraps SlotCore and supplies renderer data sources; SessionsService owns Session objects, list/scope/history state; WorkspacesService depends on SessionsService and owns Workspace objects, list/actions, default-target derivation, and the New Session blank-reuse entry (`connectWorkspace`). The runtime fans the shared Host stream into both managers. Client sessions are always Host-born (Session+Agent+cwd in one `session.create`); the client holds no pre-entity session state — a session's Agent scope (the client mirror of host dsh-scope, keyed by the shared agent/session id) is born when its row enters the list mirror and dies with the prune. Contract: api-contracts v3 §4. Each `Session` holds a generic `ProjectionValueStore` seeded from the history-tail `projections` block and updated by `session/projection` frames under higher-seq-wins; domain keys (including `todos` and `title`) are read via `projections.faceOf` / `useProjection`, not via `ConversationSnapshot`. Durable `ConversationSnapshot.metrics` instead comes from the separate history-tail value and live `session/metrics` frames because point-in-time token-meter pressure can advance at the same durable log revision; only nondecreasing log and projection revisions are accepted. `ConversationSnapshot.modelRequestContextWindow` separately retains capacity from the latest `session/model-request` observed on the current mux connection. A later request replaces or clears that value, while `session/subscribed` clears both metrics ordering and capacity; reconnect, restore, and a new subscription therefore show no percentage until another request is observed. Missing metrics remain `null` rather than being inferred from the visible node window. +Client cordis boot and React-free object services: SlotsService wraps SlotCore and supplies renderer data sources; SessionsService owns Session objects, list/scope/history state; WorkspacesService depends on SessionsService and owns Workspace objects, list/actions, default-target derivation, and the New Session blank-reuse entry (`connectWorkspace`). The runtime fans the shared Host stream into both managers. Client sessions are always Host-born (Session+Agent+cwd in one `session.create`); the client holds no pre-entity session state — a session's Agent scope (the client mirror of host dsh-scope, keyed by the shared agent/session id) is born when its row enters the list mirror and dies with the prune. Contract: api-contracts v3 §4. Each `Session` holds a generic `ProjectionValueStore` seeded from the history-tail `projections` block and updated by `session/projection` frames under higher-seq-wins; domain keys (including `todos`, `title`, and `tokenUsage`) are read via `projections.faceOf` / `useProjection`, not via `ConversationSnapshot`. `ConversationSnapshot.modelRequest` separately retains the complete latest `session/model-request` observed on the current mux connection. Each frame replaces the whole snapshot, so omitted numerator or capacity fields clear an earlier value. `SessionManager` buffers one pre-instantiation snapshot, while `session/subscribed`, disconnect, and removal clear resident and pending values; reconnect, restore, and a new subscription therefore show no context percentage until another request is observed. Model selection alone does not alter request telemetry. ## Workspace and Session lists diff --git a/packages/client/runtime/README.zh.md b/packages/client/runtime/README.zh.md index 1b1d7f03d1..873fefca48 100644 --- a/packages/client/runtime/README.zh.md +++ b/packages/client/runtime/README.zh.md @@ -2,7 +2,7 @@ [English](README.md) | 中文 -客户端 cordis 启动与不依赖 React 的对象服务:SlotsService 包装 SlotCore 并提供 renderer 数据源;SessionsService 拥有 Session 对象、列表/scope/history 状态;WorkspacesService 依赖 SessionsService,拥有 Workspace 对象、列表/操作、默认目标派生,以及 New Session 空会话复用入口(`connectWorkspace`)。运行时把共享 Host 流分发给两个 manager。客户端 Session 一律由 Host 出生(一次 `session.create` 同瞬产出 Session+Agent+cwd);客户端不持有任何实体化之前的会话状态——Agent scope(host dsh-scope 的客户端镜像,以 agent/session 共用 id 为键)在会话行进入列表镜像时出生,随 prune 死亡。契约:api-contracts v3 §4。每个 `Session` 持有一个通用的 `ProjectionValueStore`,由历史尾页的 `projections` 块播种,并经 `session/projection` 帧按 seq 高者胜更新;领域键(含 `todos` 与 `title`)经 `projections.faceOf`/`useProjection` 读取,不经 `ConversationSnapshot`。持久的 `ConversationSnapshot.metrics` 则来自独立的 history 尾页值与实时 `session/metrics` 帧,因为即时 token-meter 压力可以在相同持久日志修订号上继续变化;客户端只接受日志修订号与投影修订号均不减小的数据。`ConversationSnapshot.modelRequestContextWindow` 另行保留当前 mux 连接观察到的最新 `session/model-request` 容量。后续请求会替换或清除该值,`session/subscribed` 则同时清除指标顺序状态与容量;因此,重连、恢复和新订阅都不会显示百分比,直到观察到另一次请求。缺失的 metrics 保持为 `null`,而不是根据可见节点窗口推断。 +客户端 cordis 启动与不依赖 React 的对象服务:SlotsService 包装 SlotCore 并提供 renderer 数据源;SessionsService 拥有 Session 对象、列表/scope/history 状态;WorkspacesService 依赖 SessionsService,拥有 Workspace 对象、列表/操作、默认目标派生,以及 New Session 空会话复用入口(`connectWorkspace`)。运行时把共享 Host 流分发给两个 manager。客户端 Session 一律由 Host 出生(一次 `session.create` 同瞬产出 Session+Agent+cwd);客户端不持有任何实体化之前的会话状态——Agent scope(host dsh-scope 的客户端镜像,以 agent/session 共用 id 为键)在会话行进入列表镜像时出生,随 prune 死亡。契约:api-contracts v3 §4。每个 `Session` 持有一个通用的 `ProjectionValueStore`,由历史尾页的 `projections` 块播种,并经 `session/projection` 帧按 seq 高者胜更新;领域键(含 `todos`、`title` 与 `tokenUsage`)经 `projections.faceOf`/`useProjection` 读取,不经 `ConversationSnapshot`。`ConversationSnapshot.modelRequest` 另行保留当前 mux 连接观察到的最新完整 `session/model-request`。每个帧都会替换整个快照,因此分子或容量字段一旦缺失,就会清除先前值。`SessionManager` 会缓冲一个实例化前快照;`session/subscribed`、断开连接和移除会话则会清除常驻值与待处理值;因此,重连、恢复和新订阅都不会显示上下文百分比,直到观察到另一次请求。仅选择模型不会改变请求观测数据。 ## Workspace 与 Session 列表 diff --git a/packages/client/runtime/src/client/sessions/conversation.ts b/packages/client/runtime/src/client/sessions/conversation.ts index 78ec9e2cd8..f211b209fd 100644 --- a/packages/client/runtime/src/client/sessions/conversation.ts +++ b/packages/client/runtime/src/client/sessions/conversation.ts @@ -6,7 +6,7 @@ import type { CommandId } from '@deepseek-ai/dsh-commands/brand' import type { ContentBlock } from '@deepseek-ai/dsh-llm/types' import type { - RpcError, SessionId, SessionMetrics, ToolCallView, ToolResultView, + ModelRequestTelemetry, RpcError, SessionId, ToolCallView, ToolResultView, } from '@deepseek-ai/dsh-client-connection/client' import type { PendingInteraction } from './pending.ts' @@ -267,16 +267,6 @@ export interface ConversationSnapshot { */ blank: boolean lastAgentError: string | null - /** - * Host-owned cumulative usage/current pressure. Independent of `nodes` - * pagination; null until a tail response or live metrics frame supplies a - * current durable value. - */ - metrics: SessionMetrics | null - /** - * Capacity from the latest model-request attempt observed on this mux - * generation. Absent before the first such request, after a request whose - * registration exposes no capacity, and after `session/subscribed`. - */ - modelRequestContextWindow?: number + /** Latest atomic model-request snapshot on this mux generation. */ + modelRequest: ModelRequestTelemetry | null } diff --git a/packages/client/runtime/src/client/sessions/manager.ts b/packages/client/runtime/src/client/sessions/manager.ts index ff22ca92db..d53c9c4e94 100644 --- a/packages/client/runtime/src/client/sessions/manager.ts +++ b/packages/client/runtime/src/client/sessions/manager.ts @@ -2,7 +2,10 @@ // dispatch entry + list state, constructed and held by SessionsService (one per client runtime). // List data never enters zustand; React connects via subscribe/getListSnapshot. -import type { IApiClient, HostFrame, MuxFrame, RpcError, RpcRequest, RpcResult, SessionId, SessionSummary, WorkspaceId } from '@deepseek-ai/dsh-client-connection/client' +import type { + HostFrame, IApiClient, ModelRequestTelemetry, MuxFrame, RpcError, RpcRequest, + RpcResult, SessionId, SessionSummary, WorkspaceId, +} from '@deepseek-ai/dsh-client-connection/client' // Value import from the inline-safe wire layer (not the connection plugin): // plugin-to-plugin value imports are a bundle purity error. import { transportError } from '@deepseek-ai/dsh-host-apiproxy/api' @@ -58,11 +61,11 @@ export class SessionManager { * frames are low-frequency; overflow drops oldest) and dropped on session-removed (audit S7). */ private readonly pendingBuffers = new Map[]>() /** - * Latest model capacity observed for an uninstantiated session on the + * Latest request telemetry observed for an uninstantiated session on the * current mux generation. Unlike durable history, this transient frame * cannot be backfilled when get() lazily creates the Session. */ - private readonly modelRequestContextWindows = new Map() + private readonly modelRequests = new Map() /** Outstanding approval questions per session, keyed by approvalId (idempotent under mux-open * replays of the same requested frame). Manager-owned rather than read off Session instances * because the sidebar must light up for sessions never instantiated. Cleared per connection @@ -171,7 +174,7 @@ export class SessionManager { } private createSession(sessionId: SessionId): Session { - const modelRequestContextWindow = this.modelRequestContextWindows.get(sessionId) + const modelRequest = this.modelRequests.get(sessionId) return new Session(sessionId, this.api, { // The sender's local first-send flip mirrors into the list row so the // session surfaces (lists filter on blank) before any host frame lands. @@ -179,7 +182,7 @@ export class SessionManager { this.recordMutation({ kind: 'engaged', sessionId: engaged.sessionId }) }, projections: this.projectionStore(sessionId), - ...(modelRequestContextWindow === undefined ? {} : { modelRequestContextWindow }), + ...(modelRequest === undefined ? {} : { modelRequest }), }) } @@ -355,13 +358,13 @@ export class SessionManager { return } if (frame.type === 'session/model-request') { - // Transient and non-replayable: retain the latest capacity until lazy - // instantiation. An absent value explicitly clears an earlier one. - if (frame.contextWindow === undefined) this.modelRequestContextWindows.delete(frame.sessionId) - else this.modelRequestContextWindows.set(frame.sessionId, frame.contextWindow) + // Transient and non-replayable: retain the whole latest request until + // lazy instantiation. Missing fields replace rather than inherit. + const { type: _type, sessionId, ...modelRequest } = frame + this.modelRequests.set(sessionId, modelRequest) } if (frame.type === 'session/subscribed') { - this.modelRequestContextWindows.delete(frame.sessionId) + this.modelRequests.delete(frame.sessionId) // Rows past the host's durable baseline rode state a restart lost; drop // them so last-wins cannot pin a phantom value over recomputed truth. this.projectionStores.get(frame.sessionId)?.truncate(frame.lastSeq) @@ -440,7 +443,7 @@ export class SessionManager { this.recordMutation({ kind: 'remove', sessionId: frame.sessionId }) this.sessions.get(frame.sessionId)?.handleRemoved() // instance survives (resident-instance rule), only flagged in the snapshot this.pendingBuffers.delete(frame.sessionId) // a removed session's buffered frames must not replay on a future instantiation - this.modelRequestContextWindows.delete(frame.sessionId) // connection-local request capacity dies with the Host session + this.modelRequests.delete(frame.sessionId) // connection-local request telemetry dies with the Host session this.waitingApprovals.delete(frame.sessionId) // a removed session cannot wait on anyone this.projectionStores.delete(frame.sessionId) // removed sessions drop their projection rows with the instance return @@ -481,7 +484,7 @@ export class SessionManager { if (kept.length === 0) this.pendingBuffers.delete(sessionId) else this.pendingBuffers.set(sessionId, kept) } - this.modelRequestContextWindows.clear() + this.modelRequests.clear() for (const session of this.sessions.values()) session.handleReconnecting() } diff --git a/packages/client/runtime/src/client/sessions/session.ts b/packages/client/runtime/src/client/sessions/session.ts index 8b99940009..765c3f5fa1 100644 --- a/packages/client/runtime/src/client/sessions/session.ts +++ b/packages/client/runtime/src/client/sessions/session.ts @@ -5,7 +5,7 @@ import type { ContentBlock } from '@deepseek-ai/dsh-llm/types' import type { SessionEvent } from '@deepseek-ai/dsh-session/types' import type { HistoryEntry, IApiClient, MuxFrame, RpcError, RpcId, RpcResult, - SessionId, SessionMetrics, ToolEventView, + ModelRequestTelemetry, SessionId, ToolEventView, } from '@deepseek-ai/dsh-client-connection/client' // Value import from the inline-safe wire layer (not the connection plugin): // plugin-to-plugin value imports are a bundle purity error. @@ -43,8 +43,8 @@ export interface SessionOptions { * private store (bare object-layer construction). */ projections?: ProjectionValueStore - /** Model capacity already observed on this mux generation before lazy construction. */ - modelRequestContextWindow?: number + /** Request telemetry already observed on this mux generation before lazy construction. */ + modelRequest?: ModelRequestTelemetry } /** Queue-row preview cap: the dock renders one line, the full content never leaves the host mirror. */ @@ -110,10 +110,8 @@ export class Session implements SessionFace { private queueCache: { rev: number; value: QueuedMessage[] } | null = null private frozenRev = 0 private nodesCache: { folded: readonly ConversationNode[]; frozenRev: number; value: readonly ConversationNode[] } | null = null - /** Host-owned durable usage/current-pressure projection. */ - private metrics: SessionMetrics | null = null - /** Latest capacity observed on this mux connection, independent of durable metrics arrival. */ - private contextWindow: number | undefined + /** Latest atomic request snapshot observed on this mux connection. */ + private modelRequest: ModelRequestTelemetry | null /** `run_code` sub-dispatches by parent callId (window-derived, like openCalls). Appends * copy-on-write the per-parent array so published snapshot references never mutate. */ private codeDispatches = new Map() @@ -175,7 +173,7 @@ export class Session implements SessionFace { private readonly options: SessionOptions = {}, ) { this.projections = options.projections ?? new ProjectionValueStore() - this.contextWindow = options.modelRequestContextWindow + this.modelRequest = options.modelRequest ?? null this.snapshotCache = this.buildSnapshot() } @@ -331,10 +329,10 @@ export class Session implements SessionFace { * in-flight open first — its history request rode the dead connection and must not settle * the fresh generation into 'error' (audit S4). */ async resync(): Promise { - // Queue, metrics, and request capacity are NOT cleared here: onConnected + // Queue and request telemetry are NOT cleared here: onConnected // (which drives resync) races the mux frames — fresh-generation state may - // have landed already, and the host never resends it. session/subscribed - // owns the generation reset before the queue snapshot and metrics frames. + // have landed already, and the host never resends request telemetry. + // session/subscribed owns the reset before the queue snapshot. if (this.openState === 'cold') return // never opened: no window to rebuild (doOpen flips to 'loading' synchronously, so cold implies no in-flight open) this.openGeneration++ this.openPromise = null @@ -415,24 +413,22 @@ export class Session implements SessionFace { this.queueRev++ changed = true } - if (this.contextWindow !== undefined) { - this.contextWindow = undefined - changed = true - } - if (this.metrics !== null) { - this.metrics = null + if (this.modelRequest !== null) { + this.modelRequest = null changed = true } if (changed) this.notifier.markDirty() return } - case 'session/metrics': { - this.installMetrics(frame.metrics) - return - } case 'session/model-request': { - if (this.contextWindow === frame.contextWindow) return - this.contextWindow = frame.contextWindow + const { + type: _type, + sessionId: _sessionId, + ...modelRequest + } = frame + // Whole-frame replacement is load-bearing: an omitted numerator or + // capacity clears that field from the preceding request. + this.modelRequest = modelRequest this.notifier.markDirty() return } @@ -508,17 +504,16 @@ export class Session implements SessionFace { /** Connection-loss boundary: clear values that are not replayed before the next stream starts. */ handleReconnecting(): void { this.openGeneration++ - if (this.metrics === null && this.contextWindow === undefined) return - this.metrics = null - this.contextWindow = undefined + if (this.modelRequest === null) return + this.modelRequest = null this.notifier.markDirty() } - /** host/session-removed relay: flag the resident snapshot and clear connection-local capacity. */ + /** host/session-removed relay: flag the resident snapshot and clear request telemetry. */ handleRemoved(): void { - const changed = !this.removed || this.contextWindow !== undefined + const changed = !this.removed || this.modelRequest !== null this.removed = true - this.contextWindow = undefined + this.modelRequest = null if (changed) this.notifier.markDirty() } @@ -567,7 +562,6 @@ export class Session implements SessionFace { result.value.events, result.value.hasMore, result.value.projections, - result.value.metrics, ) // Gap detection (§D.3-4): baseline past the window tail and liveBuffer did not cover it -> pull the tail page once more. const tailSeq = this.windowTailSeq() @@ -579,7 +573,6 @@ export class Session implements SessionFace { result.value.events, result.value.hasMore, result.value.projections, - result.value.metrics, ) } } @@ -606,7 +599,6 @@ export class Session implements SessionFace { entries: HistoryEntry[], hasMore: boolean, projections: ProjectionsBaseline | undefined, - metrics: SessionMetrics | undefined, ): void { this.events = entries.map(e => e.event) this.views = entries.map(e => e.view) @@ -615,7 +607,6 @@ export class Session implements SessionFace { this.foldAdapter.reset(this.events, this.baseSeq, this.views) this.rebuildDerivedFromWindow() if (projections !== undefined) this.projections.seed(projections) - if (metrics !== undefined) this.installMetrics(metrics) const buffered = this.liveBuffer this.liveBuffer = [] for (const item of buffered) this.appendLive(item.event, item.view) @@ -668,7 +659,6 @@ export class Session implements SessionFace { result.value.events, result.value.hasMore, result.value.projections, - result.value.metrics, ) } } catch (error) { @@ -854,20 +844,6 @@ export class Session implements SessionFace { return tail === undefined ? null : tail.seq } - /** Install a metrics snapshot unless a newer durable or publication revision already landed. */ - private installMetrics(metrics: SessionMetrics): void { - const current = this.metrics - if ( - current !== null - && ( - metrics.logRevision < current.logRevision - || metrics.projectionRevision < current.projectionRevision - ) - ) return - this.metrics = metrics - this.notifier.markDirty() - } - private buildSnapshot(): ConversationSnapshot { const { nodes: folded, degraded } = this.foldAdapter.nodes() // Frozen interrupted nodes ride fractional seqs: a stable merge keeps them in flow order. @@ -920,10 +896,7 @@ export class Session implements SessionFace { promptError: this.promptError, blank: this.blankBit, lastAgentError: this.lastAgentError, - metrics: this.metrics, - ...(this.contextWindow === undefined - ? {} - : { modelRequestContextWindow: this.contextWindow }), + modelRequest: this.modelRequest, } } } diff --git a/packages/client/runtime/tests/client-apply.spec.ts b/packages/client/runtime/tests/client-apply.spec.ts index 7399c46f6b..1e2937b464 100644 --- a/packages/client/runtime/tests/client-apply.spec.ts +++ b/packages/client/runtime/tests/client-apply.spec.ts @@ -102,7 +102,7 @@ describe('runtime client apply', () => { expect(bench.api.callsOf('session.create')).toHaveLength(1) }) - it('clears connection-local Session state on disconnect but not connected', async () => { + it('clears connection-local request telemetry on reconnect but not connected', async () => { const bench = await mount() const sessions = bench.ctx.get('sessions') as SessionsService bench.sinks?.onHostEnvelope?.({ @@ -112,21 +112,8 @@ describe('runtime client apply', () => { await Promise.resolve() const session = sessions.binding('s-state' as never)?.session if (session === undefined) throw new Error('session binding missing') - const currentMetrics = { - projectionRevision: 4, - logRevision: 10, - uncachedInputTokens: 10, - outputTokens: 4, - cacheReadTokens: 90, - cacheWriteTokens: 3, - contextTokens: 35, - } bench.sinks?.onMuxEnvelope?.({ - rpcId: 'metrics' as never, - payload: { type: 'session/metrics', sessionId: 's-state', metrics: currentMetrics } as never, - }) - bench.sinks?.onMuxEnvelope?.({ - rpcId: 'capacity' as never, + rpcId: 'request' as never, payload: { type: 'session/model-request', sessionId: 's-state', @@ -134,19 +121,20 @@ describe('runtime client apply', () => { step: 1, provider: 'test', model: 'alpha', + contextTokens: 32_000, contextWindow: 128_000, } as never, }) bench.sinks?.onConnected?.() - expect(session.getSnapshot()).toMatchObject({ - metrics: currentMetrics, - modelRequestContextWindow: 128_000, + expect(session.getSnapshot().modelRequest).toMatchObject({ + model: 'alpha', + contextTokens: 32_000, + contextWindow: 128_000, }) - bench.sinks?.onDisconnected?.() - expect(session.getSnapshot().metrics).toBeNull() - expect(session.getSnapshot().modelRequestContextWindow).toBeUndefined() + bench.sinks?.onStateChange?.('reconnecting') + expect(session.getSnapshot().modelRequest).toBeNull() }) it('stops the stream loop when the plugin fiber unloads', async () => { diff --git a/packages/client/runtime/tests/fake-api.ts b/packages/client/runtime/tests/fake-api.ts index 829417e44b..0654446c5b 100644 --- a/packages/client/runtime/tests/fake-api.ts +++ b/packages/client/runtime/tests/fake-api.ts @@ -4,7 +4,7 @@ import type { CommandId } from '@deepseek-ai/dsh-commands/brand' import type { ClientResponse, CommandDescriptor, HostFrame, IApiClient, ModelTarget, MuxFrame, - RpcError, RpcReceipt, RpcRequest, RpcResponse, SessionId, SessionMetrics, SessionModels, + RpcError, RpcReceipt, RpcRequest, RpcResponse, SessionId, SessionModels, SessionProjectionsBlock, SkillEntry, WorkspaceId, WorkspaceView, } from '@deepseek-ai/dsh-client-connection/client' @@ -69,7 +69,6 @@ export class FakeApiClient implements IApiClient { events: never[] hasMore: boolean projections?: SessionProjectionsBlock - metrics?: SessionMetrics }>> = () => Promise.resolve(ok({ events: [], hasMore: false })) diff --git a/packages/client/runtime/tests/manager.spec.ts b/packages/client/runtime/tests/manager.spec.ts index 62f4cc32d1..d5a4237fc1 100644 --- a/packages/client/runtime/tests/manager.spec.ts +++ b/packages/client/runtime/tests/manager.spec.ts @@ -4,7 +4,7 @@ */ import { describe, expect, it, vi } from 'vitest' -import type { SessionId, SessionMetrics } from '@deepseek-ai/dsh-client-connection/client' +import type { SessionId } from '@deepseek-ai/dsh-client-connection/client' import { SessionManager } from '../src/client/sessions/manager.ts' import { FakeApiClient, deferred, err, ok } from './fake-api.ts' import { entries, plainTurn } from './event-script.ts' @@ -41,7 +41,7 @@ describe('instances', () => { expect(manager.get(S2).getSnapshot().pending).toEqual([]) }) - it('retains the latest transient model capacity until lazy instantiation', () => { + it('retains the latest transient request snapshot until lazy instantiation', () => { const api = new FakeApiClient() const manager = new SessionManager(api) manager.handleMuxEnvelope({ @@ -53,6 +53,7 @@ describe('instances', () => { step: 1, provider: 'test', model: 'alpha', + contextTokens: 12_000, contextWindow: 128_000, }, }) @@ -65,14 +66,22 @@ describe('instances', () => { step: 2, provider: 'test', model: 'beta', + contextTokens: 32_000, contextWindow: 256_000, }, }) - expect(manager.get(S1).getSnapshot().modelRequestContextWindow).toBe(256_000) + expect(manager.get(S1).getSnapshot().modelRequest).toEqual({ + turn: 1, + step: 2, + provider: 'test', + model: 'beta', + contextTokens: 32_000, + contextWindow: 256_000, + }) }) - it('retains explicit capacity clearing before lazy instantiation', () => { + it('retains whole-frame replacement before lazy instantiation', () => { const api = new FakeApiClient() const manager = new SessionManager(api) manager.handleMuxEnvelope({ @@ -99,10 +108,15 @@ describe('instances', () => { }, }) - expect(manager.get(S1).getSnapshot().modelRequestContextWindow).toBeUndefined() + expect(manager.get(S1).getSnapshot().modelRequest).toEqual({ + turn: 1, + step: 2, + provider: 'test', + model: 'unknown-capacity', + }) }) - it('clears retained capacity on subscribed and resident capacity on removal', () => { + it('clears retained request telemetry on subscribed and removal', () => { const api = new FakeApiClient() const manager = new SessionManager(api) manager.handleMuxEnvelope({ @@ -122,7 +136,7 @@ describe('instances', () => { payload: { type: 'session/subscribed', sessionId: S1, lastSeq: 0 }, }) const session = manager.get(S1) - expect(session.getSnapshot().modelRequestContextWindow).toBeUndefined() + expect(session.getSnapshot().modelRequest).toBeNull() manager.handleMuxEnvelope({ rpcId: 'request-after-subscribe' as never, @@ -136,12 +150,12 @@ describe('instances', () => { contextWindow: 256_000, }, }) - expect(session.getSnapshot().modelRequestContextWindow).toBe(256_000) + expect(session.getSnapshot().modelRequest?.contextWindow).toBe(256_000) manager.handleHostEnvelope({ rpcId: 'removed' as never, payload: { type: 'host/session-removed', sessionId: S1 }, }) - expect(session.getSnapshot().modelRequestContextWindow).toBeUndefined() + expect(session.getSnapshot().modelRequest).toBeNull() manager.handleMuxEnvelope({ rpcId: 'request-before-lazy-removal' as never, @@ -159,28 +173,15 @@ describe('instances', () => { rpcId: 'lazy-removed' as never, payload: { type: 'host/session-removed', sessionId: S2 }, }) - expect(manager.get(S2).getSnapshot().modelRequestContextWindow).toBeUndefined() + expect(manager.get(S2).getSnapshot().modelRequest).toBeNull() }) - it('clears resident metrics and capacity plus lazy capacity before reconnect', () => { + it('clears resident and lazy request telemetry on disconnect', () => { const api = new FakeApiClient() const manager = new SessionManager(api) const session = manager.get(S1) - const currentMetrics: SessionMetrics = { - projectionRevision: 4, - logRevision: 10, - uncachedInputTokens: 10, - outputTokens: 4, - cacheReadTokens: 90, - cacheWriteTokens: 3, - contextTokens: 35, - } manager.handleMuxEnvelope({ - rpcId: 'metrics' as never, - payload: { type: 'session/metrics', sessionId: S1, metrics: currentMetrics }, - }) - manager.handleMuxEnvelope({ - rpcId: 'resident-capacity' as never, + rpcId: 'resident-request' as never, payload: { type: 'session/model-request', sessionId: S1, @@ -188,11 +189,12 @@ describe('instances', () => { step: 1, provider: 'test', model: 'resident', + contextTokens: 35, contextWindow: 128_000, }, }) manager.handleMuxEnvelope({ - rpcId: 'lazy-capacity' as never, + rpcId: 'lazy-request' as never, payload: { type: 'session/model-request', sessionId: S2, @@ -200,19 +202,20 @@ describe('instances', () => { step: 1, provider: 'test', model: 'lazy', + contextTokens: 70, contextWindow: 256_000, }, }) - expect(session.getSnapshot()).toMatchObject({ - metrics: currentMetrics, - modelRequestContextWindow: 128_000, + expect(session.getSnapshot().modelRequest).toMatchObject({ + model: 'resident', + contextTokens: 35, + contextWindow: 128_000, }) - manager.handleReconnecting() + manager.handleDisconnected() - expect(session.getSnapshot().metrics).toBeNull() - expect(session.getSnapshot().modelRequestContextWindow).toBeUndefined() - expect(manager.get(S2).getSnapshot().modelRequestContextWindow).toBeUndefined() + expect(session.getSnapshot().modelRequest).toBeNull() + expect(manager.get(S2).getSnapshot().modelRequest).toBeNull() }) it('caps the pending buffer at 32 keeping the newest, and drops it on session-removed', () => { diff --git a/packages/client/runtime/tests/session.spec.ts b/packages/client/runtime/tests/session.spec.ts index 827e9f853e..3994a691cb 100644 --- a/packages/client/runtime/tests/session.spec.ts +++ b/packages/client/runtime/tests/session.spec.ts @@ -10,7 +10,7 @@ import { describe, expect, it, vi } from 'vitest' import { Context } from 'cordis' import type { SessionEvent } from '@deepseek-ai/dsh-session/types' import type { - SessionId, SessionMetrics, SessionProjectionsBlock, + SessionId, SessionProjectionsBlock, } from '@deepseek-ai/dsh-client-connection/client' import { Session } from '../src/client/sessions/session.ts' import { FakeApiClient, deferred, err, ok } from './fake-api.ts' @@ -29,34 +29,15 @@ function histResponse( events: SessionEvent[], hasMore = false, projections?: SessionProjectionsBlock, - metrics?: SessionMetrics, ) { // history now returns HistoryEntry[] ({event, view?}); these tests are view-less. return Promise.resolve(ok({ events: entries(events) as never[], hasMore, ...projections === undefined ? {} : { projections }, - ...metrics === undefined ? {} : { metrics }, })) } -function metrics( - projectionRevision: number, - logRevision: number, - over: Partial = {}, -): SessionMetrics { - return { - projectionRevision, - logRevision, - uncachedInputTokens: 10, - outputTokens: 4, - cacheReadTokens: 90, - cacheWriteTokens: 3, - contextTokens: 35, - ...over, - } -} - describe('open', () => { it('installs the tail page: cold → loading → open with window and nodes in place', async () => { const { api, session } = makeSession() @@ -70,19 +51,7 @@ describe('open', () => { expect(snapshot.openState).toBe('open') expect(snapshot.hasMore).toBe(true) expect(snapshot.nodes.map(n => n.kind)).toEqual(['user', 'assistant']) - expect(snapshot.metrics).toBeNull() - }) - - it('installs full-log metrics independently of older history pages', async () => { - const { api, session } = makeSession() - const tailMetrics = metrics(4, 106) - api.onHistory = () => histResponse(plainTurn(100, 3, '问', '答'), true, undefined, tailMetrics) - await session.open() - expect(session.getSnapshot().metrics).toBe(tailMetrics) - - api.onHistory = () => histResponse(plainTurn(94, 2, '旧问', '旧答')) - await session.loadOlder() - expect(session.getSnapshot().metrics).toBe(tailMetrics) + expect(snapshot.modelRequest).toBeNull() }) it('is idempotent: concurrent opens share one history call, reopening when open is a no-op', async () => { @@ -147,17 +116,8 @@ describe('live event path', () => { expect(session.getSnapshot().nodes).toEqual(before.nodes) }) - it('keeps live capacity separate from durable metrics, replaces or clears it on requests, and resets at subscription', async () => { + it('replaces the whole request snapshot, clears omitted fields, and resets at subscription', async () => { const { session } = await opened() - const current = metrics(8, 10) - session.handleMuxEnvelope('m1' as never, { - type: 'session/metrics', - sessionId: SID, - metrics: current, - }) - expect(session.getSnapshot().metrics).toBe(current) - expect(session.getSnapshot().modelRequestContextWindow).toBeUndefined() - session.handleMuxEnvelope('request-1' as never, { type: 'session/model-request', sessionId: SID, @@ -165,32 +125,17 @@ describe('live event path', () => { step: 1, provider: 'test', model: 'alpha', + contextTokens: 32_000, contextWindow: 128_000, }) - expect(session.getSnapshot().metrics).toBe(current) - expect(session.getSnapshot().modelRequestContextWindow).toBe(128_000) - - session.handleMuxEnvelope('m2' as never, { - type: 'session/metrics', - sessionId: SID, - metrics: metrics(9, 9, { uncachedInputTokens: 1 }), + expect(session.getSnapshot().modelRequest).toEqual({ + turn: 1, + step: 1, + provider: 'test', + model: 'alpha', + contextTokens: 32_000, + contextWindow: 128_000, }) - session.handleMuxEnvelope('m3' as never, { - type: 'session/metrics', - sessionId: SID, - metrics: metrics(7, 11, { uncachedInputTokens: 2 }), - }) - expect(session.getSnapshot().metrics).toBe(current) - expect(session.getSnapshot().modelRequestContextWindow).toBe(128_000) - - const ordinaryUpdate = metrics(9, 11, { contextTokens: 40 }) - session.handleMuxEnvelope('m4' as never, { - type: 'session/metrics', - sessionId: SID, - metrics: ordinaryUpdate, - }) - expect(session.getSnapshot().metrics).toBe(ordinaryUpdate) - expect(session.getSnapshot().modelRequestContextWindow).toBe(128_000) session.handleMuxEnvelope('request-2' as never, { type: 'session/model-request', @@ -200,16 +145,19 @@ describe('live event path', () => { provider: 'test', model: 'without-capacity', }) - expect(session.getSnapshot().metrics).toEqual(ordinaryUpdate) - expect(session.getSnapshot().modelRequestContextWindow).toBeUndefined() + expect(session.getSnapshot().modelRequest).toEqual({ + turn: 2, + step: 1, + provider: 'test', + model: 'without-capacity', + }) session.handleMuxEnvelope('sub' as never, { type: 'session/subscribed', sessionId: SID, lastSeq: 5, }) - expect(session.getSnapshot().metrics).toBeNull() - expect(session.getSnapshot().modelRequestContextWindow).toBeUndefined() + expect(session.getSnapshot().modelRequest).toBeNull() session.handleMuxEnvelope('request-3' as never, { type: 'session/model-request', sessionId: SID, @@ -217,19 +165,17 @@ describe('live event path', () => { step: 1, provider: 'test', model: 'beta', + contextTokens: 20, contextWindow: 256_000, }) - const nextGeneration = metrics(0, 10, { contextTokens: 20 }) - session.handleMuxEnvelope('m5' as never, { - type: 'session/metrics', - sessionId: SID, - metrics: nextGeneration, + expect(session.getSnapshot().modelRequest).toMatchObject({ + turn: 3, + contextTokens: 20, + contextWindow: 256_000, }) - expect(session.getSnapshot().metrics).toBe(nextGeneration) - expect(session.getSnapshot().modelRequestContextWindow).toBe(256_000) }) - it('publishes a subscribed reset when capacity arrived before durable metrics', async () => { + it('publishes a subscribed reset when request telemetry arrived first', async () => { const { session } = await opened() session.handleMuxEnvelope('request' as never, { type: 'session/model-request', @@ -238,17 +184,17 @@ describe('live event path', () => { step: 1, provider: 'test', model: 'alpha', + contextTokens: 8_000, contextWindow: 128_000, }) - expect(session.getSnapshot().metrics).toBeNull() - expect(session.getSnapshot().modelRequestContextWindow).toBe(128_000) + expect(session.getSnapshot().modelRequest?.contextWindow).toBe(128_000) session.handleMuxEnvelope('sub' as never, { type: 'session/subscribed', sessionId: SID, lastSeq: 5, }) - expect(session.getSnapshot().modelRequestContextWindow).toBeUndefined() + expect(session.getSnapshot().modelRequest).toBeNull() }) it('materializes a command node from live lifecycle frames and reproduces it from a history window', async () => { @@ -876,72 +822,61 @@ describe('remaining branches', () => { }) describe('resync', () => { - it('fences pre-disconnect history behind a fresh mux metrics baseline', async () => { + it('clears request telemetry on reconnect and drops a stale in-flight history response', async () => { const { api, session } = makeSession() const stale = deferred>>() api.onHistory = () => stale.promise const opening = session.open() - const oldLiveMetrics = metrics(8, 10) - session.handleMuxEnvelope('old-metrics' as never, { - type: 'session/metrics', - sessionId: SID, - metrics: oldLiveMetrics, - }) - session.handleMuxEnvelope('old-capacity' as never, { + session.handleMuxEnvelope('old-request' as never, { type: 'session/model-request', sessionId: SID, turn: 1, step: 1, provider: 'test', model: 'old', + contextTokens: 20, contextWindow: 128_000, }) session.handleReconnecting() - expect(session.getSnapshot().metrics).toBeNull() - expect(session.getSnapshot().modelRequestContextWindow).toBeUndefined() - const freshMetrics = metrics(0, 1, { contextTokens: 20 }) - session.handleMuxEnvelope('fresh-metrics' as never, { - type: 'session/metrics', - sessionId: SID, - metrics: freshMetrics, - }) + expect(session.getSnapshot().modelRequest).toBeNull() stale.resolve(ok({ events: entries(plainTurn(0, 0, '旧问', '旧答')) as never[], hasMore: false, - metrics: metrics(99, 99, { contextTokens: 999 }), })) await opening expect(session.getSnapshot().nodes).toEqual([]) - expect(session.getSnapshot().metrics).toBe(freshMetrics) + expect(session.getSnapshot().modelRequest).toBeNull() api.onHistory = () => histResponse(plainTurn(6, 1, '新问', '新答')) await session.resync() expect(session.getSnapshot().openState).toBe('open') expect(session.getSnapshot().nodes.map(node => node.seq)).toEqual([7, 9]) - expect(session.getSnapshot().metrics).toBe(freshMetrics) + expect(session.getSnapshot().modelRequest).toBeNull() }) - it('preserves fresh-generation metrics that arrive before a failing history refresh', async () => { + it('preserves a fresh-generation request snapshot when history resync fails', async () => { const { api, session } = makeSession() - const oldMetrics = metrics(8, 10) - api.onHistory = () => histResponse(plainTurn(0, 0, 'a', 'b'), false, undefined, oldMetrics) + api.onHistory = () => histResponse(plainTurn(0, 0, 'a', 'b')) await session.open() - expect(session.getSnapshot().metrics).toBe(oldMetrics) session.handleMuxEnvelope('sub' as never, { type: 'session/subscribed', sessionId: SID, lastSeq: 5, }) - expect(session.getSnapshot().metrics).toBeNull() + expect(session.getSnapshot().modelRequest).toBeNull() - const freshMetrics = metrics(0, 10, { contextTokens: 20 }) - session.handleMuxEnvelope('fresh-metrics' as never, { - type: 'session/metrics', + session.handleMuxEnvelope('fresh-request' as never, { + type: 'session/model-request', sessionId: SID, - metrics: freshMetrics, + turn: 2, + step: 1, + provider: 'test', + model: 'fresh', + contextTokens: 20, + contextWindow: 256_000, }) api.onHistory = () => Promise.resolve(err({ code: 'internal', @@ -953,7 +888,14 @@ describe('resync', () => { expect(session.getSnapshot()).toMatchObject({ openState: 'error', - metrics: freshMetrics, + modelRequest: { + turn: 2, + step: 1, + provider: 'test', + model: 'fresh', + contextTokens: 20, + contextWindow: 256_000, + }, }) }) diff --git a/packages/client/test-runtime/src/fixtures.ts b/packages/client/test-runtime/src/fixtures.ts index 4219d233e2..7c9d74aae7 100644 --- a/packages/client/test-runtime/src/fixtures.ts +++ b/packages/client/test-runtime/src/fixtures.ts @@ -62,6 +62,7 @@ export function conversationSnapshot(sessionId: SessionId): ConversationSnapshot promptError: null, blank: false, lastAgentError: null, + modelRequest: null, } } diff --git a/packages/client/ui-conversation/README.i18n.yaml b/packages/client/ui-conversation/README.i18n.yaml index 6508a18b98..388b2044b8 100644 --- a/packages/client/ui-conversation/README.i18n.yaml +++ b/packages/client/ui-conversation/README.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write packages/client/ui-conversation/README.md -README.md: b74127ffe11df40fcad0c97e2fc6896ec79ad9d1 -README.zh.md: f6dfc3ca61ea80758c9f64ca98f034b0832ee89e +README.md: c60a38ec26d4bca15ce23e51dbaedf3ef36022e4 +README.zh.md: 65c5554a78960fc4f86a6a370772c95429c648c4 diff --git a/packages/client/ui-conversation/README.md b/packages/client/ui-conversation/README.md index b74127ffe1..c60a38ec26 100644 --- a/packages/client/ui-conversation/README.md +++ b/packages/client/ui-conversation/README.md @@ -20,7 +20,7 @@ Per-session UI state for selection and the active view lives in the declared cha The composer bar declares session-scoped single seats for `'conversation.input.plan'` (right of the local access-mode control) and `'conversation.input.model'` (immediately before the pending indicator and send/stop button), plus list slots for overlay, dock, left, and right input extensions. Feature packages own each control and its state; ui-conversation supplies placement, the `locked` owner prop, and the standard slot shares. While the `plan` projection's effective target is plan mode, InputBar swaps its textarea placeholder to the plan-task wording (a host-folded value read through the standard-kit `useProjection`; owner-supplied placeholders win). The resident no-session shell uses `DisabledInputBar` and therefore dispatches no session-scoped control seats. -The chat stats line reads durable token counters/current pressure from `ConversationSnapshot.metrics` and joins them only at presentation with the separate connection-local `modelRequestContextWindow`; visible nodes supply only the existing turn/step counts. It renders uncached input, output, and cache reads as separate compact values, computes cache hit as `cacheRead / (uncachedInput + cacheRead)` without cache writes, and shows context occupancy only after the current mux connection observes a model request with capacity. Before that request, after reconnect/restore/new subscription, or after a request without capacity, the percentage is omitted and context is labeled unknown rather than queried ahead or reconstructed from history. +The chat stats line reads full-log billing from the generic `tokenUsage` projection and joins it only at presentation with the connection-local atomic `ConversationSnapshot.modelRequest`; visible nodes supply only the existing turn/step counts. It renders uncached input, output, and cache reads as separate compact values, computes cache hit as `cacheRead / (uncachedInput + cacheRead)` without cache writes, and shows context occupancy only when the same observed request snapshot contains both `contextTokens` and `contextWindow`. Before that request, after reconnect/restore/new subscription, or after a request missing either field, context is labeled unknown rather than queried from the selected model or reconstructed from history. The existing inline stats row remains the sole context UI; the model selector has no circle or accessory. `src/client/` is organized for the future package split: `contract/` is the sole inter-domain shared face (`slots.ts` slot declarations + composed slot props including the tool-row contract, `views.ts` shared primitives, `tool-call-model.ts`); the `skeleton/`, `chat/`, and `toolviews/` (sample registrants) domain directories import contract files and never each other; `apply.ts` is the only assembly point allowed to import all three domains. The `/client` export surface is the contract only — `apply`/`inject`, the two service classes, and the `contract/` type families; implementation components (skeleton, chat rows) and the store factory stay internal and reach the page exclusively through apply's slot registrations (tests take them via the `./src/*` subpath). diff --git a/packages/client/ui-conversation/README.zh.md b/packages/client/ui-conversation/README.zh.md index f6dfc3ca61..65c5554a78 100644 --- a/packages/client/ui-conversation/README.zh.md +++ b/packages/client/ui-conversation/README.zh.md @@ -20,7 +20,7 @@ todo 两个面就是在该形状上的两个注册项,都是普通注册方插 输入栏为 `'conversation.input.plan'`(位于本地 access 模式控件右侧)和 `'conversation.input.model'`(渲染在 pending 指示器与发送/停止按钮之前)声明会话作用域的单实例 seat,并为 overlay、dock、left 和 right 输入扩展声明列表 slot。各功能包拥有相应控件及其状态;ui-conversation 提供放置位置、`locked` owner prop 和标准 slot share。当 `plan` 投影的有效目标为 plan mode 时,InputBar 将文本框 placeholder 切换为 plan 任务措辞(经标准套件 `useProjection` 读取的 host 折叠值;owner 提供的 placeholder 优先)。常驻无会话壳使用 `DisabledInputBar`,因此不会分发任何会话作用域的控件 seat。 -聊天统计行从 `ConversationSnapshot.metrics` 读取持久的 token 计数/当前压力,并且只在展示时把它们与独立的连接本地 `modelRequestContextWindow` 结合;可见节点仅提供既有的轮次和步骤计数。它以相互独立的紧凑值显示未缓存输入、输出与缓存读取,通过 `cacheRead / (uncachedInput + cacheRead)` 计算缓存命中率而不计入缓存写入,并且只有当前 mux 连接观察到带容量的模型请求后才显示上下文占用率。在该请求之前、重连/恢复/新订阅之后,或在请求不带容量之后,系统都会省略百分比,并把上下文标为「未知」,而不会提前查询或根据历史记录重建。 +聊天统计行从通用 `tokenUsage` 投影读取完整日志计费用量,并且只在展示时把它与连接本地的原子快照 `ConversationSnapshot.modelRequest` 结合;可见节点仅提供既有的轮次和步骤计数。它以相互独立的紧凑值显示未缓存输入、输出与缓存读取,通过 `cacheRead / (uncachedInput + cacheRead)` 计算缓存命中率而不计入缓存写入,并且只有同一份已观测请求快照同时包含 `contextTokens` 与 `contextWindow` 时才显示上下文占用率。在该请求之前、重连/恢复/新订阅之后,或在请求缺少任一字段之后,系统都会把上下文标为「未知」,而不会从所选模型查询或根据历史记录重建。现有的行内统计行仍是唯一的上下文 UI;模型选择器不增加圆环或附属控件。 `src/client/` 按未来的包拆分组织:`contract/` 是唯一的跨领域共享表层(`slots.ts` slot 声明 + 组合后的 slot props,包括工具行契约、`views.ts` 共享原语、`tool-call-model.ts`);`skeleton/`、`chat/` 和 `toolviews/`(示例注册方)领域目录只导入 contract 文件,彼此绝不导入;`apply.ts` 是唯一允许导入全部三个领域的组装点。`/client` 导出表层只包含契约:`apply`/`inject`、两个服务类和 `contract/` 类型家族;实现组件(骨架、聊天行)与 store factory 保持内部状态,只能通过 apply 的 slot 注册到达页面(测试通过 `./src/*` 子路径获取它们)。 diff --git a/packages/client/ui-conversation/package.json b/packages/client/ui-conversation/package.json index 42812dce24..88be12b5af 100644 --- a/packages/client/ui-conversation/package.json +++ b/packages/client/ui-conversation/package.json @@ -44,6 +44,7 @@ "@deepseek-ai/dsh-client-ui-slash": "^0.0.1", "@deepseek-ai/dsh-client-ui-slots": "^0.0.1", "@deepseek-ai/dsh-invariants": "^0.0.1", + "@deepseek-ai/dsh-token-meter": "^0.0.1", "cordis": "^4.0.0-rc.7", "react": "^18.2.0" }, @@ -52,6 +53,7 @@ "@deepseek-ai/dsh-plan-mode": "workspace:^", "@deepseek-ai/dsh-session-projection": "workspace:^", "@deepseek-ai/dsh-tool-todo": "workspace:^", + "@deepseek-ai/dsh-token-meter": "workspace:^", "@deepseek-ai/dsh-client-ui-layout": "workspace:^", "@deepseek-ai/dsh-client-ui-primitives": "workspace:^", "@deepseek-ai/dsh-client-ui-slash": "workspace:^", diff --git a/packages/client/ui-conversation/src/client/chat/ChatView.tsx b/packages/client/ui-conversation/src/client/chat/ChatView.tsx index 1cbb00a4ce..36a2cde7b9 100644 --- a/packages/client/ui-conversation/src/client/chat/ChatView.tsx +++ b/packages/client/ui-conversation/src/client/chat/ChatView.tsx @@ -221,7 +221,9 @@ function StreamingTail({ useSession, onGrow }: { * The chat view slot entry: pure component over the composed props (tool rows * render through the declared keyed hole's renderSlot share). */ -export function ChatView({ useSession, useSessions, useStore, renderSlot, sessionId, openFile, loadOlder }: ChatViewSlotProps) { +export function ChatView({ + useProjection, useSession, useSessions, useStore, renderSlot, sessionId, openFile, loadOlder, +}: ChatViewSlotProps) { const nodes = useSession(s => s.nodes) // Workspace root off the session list row: path summaries display relative to it. const cwd = useSessions(s => s.byId[sessionId]?.cwd) @@ -381,7 +383,7 @@ export function ChatView({ useSession, useSessions, useStore, renderSlot, sessio {running && } - + {!atBottom && (