From 8372340f9c58cd12d73559ec8426846c39e14471 Mon Sep 17 00:00:00 2001 From: Yichen Jiang Date: Sat, 25 Jul 2026 07:47:51 +0800 Subject: [PATCH 1/8] feat(llm): add model-specific reasoning effort controls --- ...ed-reasoning-effort-capabilities.i18n.yaml | 6 + ...ter-owned-reasoning-effort-capabilities.md | 33 +++++ ...-owned-reasoning-effort-capabilities.zh.md | 33 +++++ docs/architecture.i18n.yaml | 4 +- docs/architecture.md | 2 +- docs/architecture.zh.md | 2 +- docs/config-catalog.md | 9 +- docs/cookbook/adding-an-llm-adapter.i18n.yaml | 4 +- docs/cookbook/adding-an-llm-adapter.md | 2 +- docs/cookbook/adding-an-llm-adapter.zh.md | 2 +- docs/cordis-catalog/events.md | 2 +- docs/cordis-catalog/services.md | 25 +++- docs/core-data-structures/core.md | 46 ++++++- docs/core-data-structures/llm-streaming.md | 13 +- docs/core-data-structures/session.md | 2 +- docs/event-producer-consumer.md | 2 +- .../develop/practice/llm-adapter.i18n.yaml | 4 +- docs/user/develop/practice/llm-adapter.md | 4 +- docs/user/develop/practice/llm-adapter.zh.md | 4 +- .../tests/fixtures/cli-mock-llm.ts | 26 +++- .../headless-agent/tests/headless.snapshot.ts | 41 ++++++ .../cordis/tool-cordis/src/api-catalog.ts | 24 +++- packages/core/agent-loop/README.md | 2 + packages/core/agent-loop/src/loop.ts | 39 +++++- .../core/agent-loop/tests/mock-adapter.ts | 14 +- .../tests/request-reconstruction.spec.ts | 78 ++++++++++- packages/core/session/src/index.ts | 5 + packages/core/session/src/types.ts | 2 +- .../core/session/tests/request-header.spec.ts | 5 + packages/core/session/tests/session.spec.ts | 36 ++++- packages/llm/llm-deepseek/README.md | 8 +- packages/llm/llm-deepseek/src/adapter.ts | 25 +++- packages/llm/llm-deepseek/src/index.ts | 10 +- packages/llm/llm-deepseek/src/serialize.ts | 18 ++- .../llm/llm-deepseek/tests/adapter.e2e.ts | 6 +- .../llm/llm-deepseek/tests/adapter.spec.ts | 76 ++++++++++- .../llm/llm-deepseek/tests/serialize.spec.ts | 22 ++- packages/llm/llm-pi-ai/README.md | 3 + packages/llm/llm-pi-ai/src/adapter.ts | 81 ++++++++++- packages/llm/llm-pi-ai/tests/adapter.e2e.ts | 13 +- packages/llm/llm-pi-ai/tests/adapter.spec.ts | 70 +++++++++- packages/llm/llm/README.md | 10 +- packages/llm/llm/src/brand.ts | 12 ++ packages/llm/llm/src/call-config.ts | 23 +++- packages/llm/llm/src/index.ts | 112 ++++++++++++++- packages/llm/llm/src/types.ts | 25 +++- packages/llm/llm/tests/call-config.spec.ts | 6 + packages/llm/llm/tests/service.spec.ts | 127 +++++++++++++++++- scripts/gen-cordis-catalog.ts | 1 + scripts/type-equiv.manifest.json | 15 +++ 50 files changed, 1046 insertions(+), 88 deletions(-) create mode 100644 .agents/notes/implemented/architecture/2026-07-24-adapter-owned-reasoning-effort-capabilities.i18n.yaml create mode 100644 .agents/notes/implemented/architecture/2026-07-24-adapter-owned-reasoning-effort-capabilities.md create mode 100644 .agents/notes/implemented/architecture/2026-07-24-adapter-owned-reasoning-effort-capabilities.zh.md diff --git a/.agents/notes/implemented/architecture/2026-07-24-adapter-owned-reasoning-effort-capabilities.i18n.yaml b/.agents/notes/implemented/architecture/2026-07-24-adapter-owned-reasoning-effort-capabilities.i18n.yaml new file mode 100644 index 0000000000..489f9845f6 --- /dev/null +++ b/.agents/notes/implemented/architecture/2026-07-24-adapter-owned-reasoning-effort-capabilities.i18n.yaml @@ -0,0 +1,6 @@ +# Bilingual-pair consistency record (docs/i18n/README.md): the git blob hash of each +# side as of the last confirmed-consistent state. Both languages carry equal authority; +# after editing either side, bring the other along and re-record with: +# pnpm run verify-translation-pairing --write +2026-07-24-adapter-owned-reasoning-effort-capabilities.md: 806e808ab18d617003f757200f0cdee5853476f5 +2026-07-24-adapter-owned-reasoning-effort-capabilities.zh.md: 9be8b6f4f7bcbe063ff90492ac44b9590860f67f diff --git a/.agents/notes/implemented/architecture/2026-07-24-adapter-owned-reasoning-effort-capabilities.md b/.agents/notes/implemented/architecture/2026-07-24-adapter-owned-reasoning-effort-capabilities.md new file mode 100644 index 0000000000..806e808ab1 --- /dev/null +++ b/.agents/notes/implemented/architecture/2026-07-24-adapter-owned-reasoning-effort-capabilities.md @@ -0,0 +1,33 @@ +# Agent Note: Adapter-owned reasoning effort capabilities + +Status: implemented + +English | [中文](2026-07-24-adapter-owned-reasoning-effort-capabilities.zh.md) + +## Problem + +Reasoning strength was adapter configuration only, so a conversation could not discover or change the selected model's supported levels between requests. Promoting one adapter's level union into `dsh-llm` would make every provider and model adopt names it may not support, while a provider-specific options bag would make the loop unable to validate or durably reconstruct the effective request. + +## Decision + +`dsh-llm` represents a reasoning effort as the opaque branded `ReasoningEffortId`. An adapter's `resolveModelReasoning(provider, model)` returns a non-empty ordered list of ids with display metadata and may name one configured default. The core validates metadata, requires an explicit or configured effort to appear exactly in that list, and never clamps or aliases a value. + +`LlmCallConfig` and `GenerateOptions` carry the optional effort. The agent loop resolves the post-`agent/request` config before writing `request/header`, so defaults and dynamic changes are model-visible only after becoming durable facts. A route with no registered adapter retains its proposed config so an `llm/stream` middleware can own and short-circuit it; terminal dispatch still rejects an unhandled route. A resumed loop retains the logged effort only when its initial provider/model route is unchanged; a route change discards the previous model's opaque id. The terminal `LlmService` adapter boundary repeats resolution for direct calls that do not pass through the loop. + +The native DeepSeek adapter advertises `high` and `max`, defaults to configured effort or `high`, and exposes no effort capability while thinking is disabled. The pi-ai adapter derives each exact model's list from `getSupportedThinkingLevels()`, excludes `off`, preserves an absent profile default as a provider default, and leaves provider wire-value mapping inside pi-ai. + +## Alternatives considered + +**Define the pi-ai `ThinkingLevel` union in core.** Rejected because current pi-ai canonical names are an adapter implementation detail; a future provider can expose a different identifier without requiring a core release. + +**Carry an untyped provider options object.** Rejected because the loop could neither validate a selected value nor put a stable provider-neutral fact in the request header. + +**Clamp unsupported levels.** Rejected because a silent substitution makes the user's selected control differ from the logged request intent and hides stale deployment configuration. + +**Include `off` as an effort.** Rejected because disabling reasoning is a mode capability with different request and output semantics, not a reasoning-strength level. + +## Consequences + +Clients can query one exact route and render the adapter's order and names without knowing a global enum. Adapter configuration remains the deployment-default owner, while `agent/request` can replace the effective effort on each step. Invalid metadata fails with `INVALID_MODEL_REASONING`, and unsupported explicit or configured values fail with `UNSUPPORTED_REASONING_EFFORT` before provider I/O. + +The capability query is asynchronous and exact-model resolution may fail for adapters backed by authoritative catalogs. Keyless service, adapter, loop, session, and request-header tests pin validation, defaulting, dynamic changes, logging, and resume behavior; runnable snapshots pin the resolved effort in real assembled request headers, while key-gated adapter tests exercise provider serialization. diff --git a/.agents/notes/implemented/architecture/2026-07-24-adapter-owned-reasoning-effort-capabilities.zh.md b/.agents/notes/implemented/architecture/2026-07-24-adapter-owned-reasoning-effort-capabilities.zh.md new file mode 100644 index 0000000000..9be8b6f4f7 --- /dev/null +++ b/.agents/notes/implemented/architecture/2026-07-24-adapter-owned-reasoning-effort-capabilities.zh.md @@ -0,0 +1,33 @@ +# Agent Note:适配器持有的推理强度能力 + +Status: implemented + +[English](2026-07-24-adapter-owned-reasoning-effort-capabilities.md) | 中文 + +## 问题 + +推理强度过去只能在适配器中配置,因此对话无法在多次请求之间发现或更改所选模型支持的等级。若将某个适配器的等级联合类型提升到 `dsh-llm`,所有提供方和模型都必须采用一套自身可能并不支持的名称;若改用提供方特有的 options 对象,主循环又无法校验最终生效的请求,也无法通过持久化记录准确重建该请求。 + +## 决策 + +`dsh-llm` 使用不透明的品牌类型 `ReasoningEffortId` 表示推理强度。适配器的 `resolveModelReasoning(provider, model)` 返回非空的有序 ID 列表及其展示元数据,并可指定一个由配置确定的默认值。核心会校验元数据,要求显式指定或配置指定的推理强度与列表中的某个 ID 完全一致,且绝不自动调整或为值提供别名。 + +`LlmCallConfig` 和 `GenerateOptions` 携带可选的推理强度。agent loop(智能体循环)在 `agent/request` 处理完成后、写入 `request/header` 前解析配置,因此默认值和动态变更只有成为持久化事实后才对模型可见。没有已注册适配器的路由会保留原定配置,使 `llm/stream` 中间件可以接管并短路该请求;若仍未得到处理,最终分发会拒绝该路由。恢复后的主循环仅在初始提供方/模型路由未变时保留日志中记录的推理强度;如果路由发生变化,则丢弃上一模型的不透明 ID。最终的 `LlmService` 适配器边界会再次执行解析,以覆盖未经过主循环的直接调用。 + +原生 DeepSeek 适配器声明 `high` 和 `max`,默认使用配置指定的推理强度,若未配置则使用 `high`;禁用思考时不暴露推理强度能力。pi-ai 适配器通过 `getSupportedThinkingLevels()` 按具体模型推导等级列表,排除 `off`,在 profile 未指定默认值时保留提供方默认行为,并将提供方协议值的映射留在 pi-ai 内部。 + +## 备选方案 + +**在核心中定义 pi-ai 的 `ThinkingLevel` 联合类型。** 不予采纳:pi-ai 当前的规范名称属于适配器实现细节;未来的提供方可以暴露不同的标识符,而无需为此发布新的核心版本。 + +**携带无类型约束的提供方 options 对象。** 不予采纳:主循环既无法校验选定值,也无法在请求头中写入稳定且与提供方无关的事实。 + +**自动调整不支持的等级。** 不予采纳:静默替换会导致用户选定的控制项与日志记录的请求意图不一致,还会掩盖陈旧的部署配置。 + +**将 `off` 列为推理强度。** 不予采纳:禁用推理属于具有不同请求和输出语义的模式能力,而不是推理强度等级。 + +## 影响 + +客户端可以查询一条确切路由,并按适配器给出的顺序和名称渲染等级,而无需了解全局枚举。适配器配置仍负责提供部署默认值,`agent/request` 则可以在每个步骤替换实际生效的推理强度。元数据无效时抛出 `INVALID_MODEL_REASONING`;显式指定或配置指定的值不受支持时,会在提供方 I/O 前抛出 `UNSUPPORTED_REASONING_EFFORT`。 + +能力查询采用异步方式;对于由权威目录支持的适配器,确切模型解析可能失败。无密钥的服务、适配器、主循环、会话和请求头测试为校验、默认值解析、动态变更、日志记录和恢复行为提供回归保障;可运行快照锁定实际组装请求头中的已解析推理强度,仅在有密钥时运行的适配器测试则覆盖提供方序列化。 diff --git a/docs/architecture.i18n.yaml b/docs/architecture.i18n.yaml index 20553e1b90..e7c0dd02b3 100644 --- a/docs/architecture.i18n.yaml +++ b/docs/architecture.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write -architecture.md: d1051eecf51d8d1c7f8c235b0c5dd80478b43316 -architecture.zh.md: 502c0248a9d2c62af165ce07eb19489b76c5f6ee +architecture.md: 0bc015afcf583224b7162540ae53f2bb19952cd6 +architecture.zh.md: 20ed3229e0a8d5997cc982ae5c82fb90f900c948 diff --git a/docs/architecture.md b/docs/architecture.md index d1051eecf5..0bc015afcf 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -91,7 +91,7 @@ forever: agent/pre-step snapshot the derived messages (the reconstruction boundary) 'step/start' - agent/request (config only) -> log request/header -> checkpoint -> llm/stream (frozen) + agent/request -> resolve reasoning/default -> log request/header -> checkpoint -> llm/stream (frozen) on final adapter-path or terminal in-band failure: 'step/end' agent/request-error(original error, failure facts, immutable prior failures, signal) diff --git a/docs/architecture.zh.md b/docs/architecture.zh.md index 502c0248a9..20ed3229e0 100644 --- a/docs/architecture.zh.md +++ b/docs/architecture.zh.md @@ -91,7 +91,7 @@ forever: agent/pre-step snapshot the derived messages (the reconstruction boundary) 'step/start' - agent/request (config only) -> log request/header -> checkpoint -> llm/stream (frozen) + agent/request -> resolve reasoning/default -> log request/header -> checkpoint -> llm/stream (frozen) on final adapter-path or terminal in-band failure: 'step/end' agent/request-error(original error, failure facts, immutable prior failures, signal) diff --git a/docs/config-catalog.md b/docs/config-catalog.md index 985e493184..b1f3f969c4 100644 --- a/docs/config-catalog.md +++ b/docs/config-catalog.md @@ -520,8 +520,9 @@ Requires: `llm` /** * Plugin config, validated by the same-named schemastery schema. Every field * is optional in yml: credentials/endpoint fall back to the environment (a - * missing API key fails plugin load, not the first call), and omitted - * thinking fields send nothing on the wire, so the provider default applies. + * missing API key fails plugin load, not the first call), omitted thinking + * mode uses the provider default, and omitted reasoning effort resolves to + * `high`. */ export interface Config { /** API key; falls back to $DEEPSEEK_API_KEY. Required one way or the other. */ @@ -530,7 +531,7 @@ export interface Config { baseURL?: string /** Thinking-mode default for every request (provider default: enabled). */ thinking?: 'enabled' | 'disabled' - /** Thinking effort (only meaningful with thinking enabled). */ + /** Default thinking effort when thinking is enabled (default `high`). */ reasoningEffort?: 'high' | 'max' /** Positive context capacity used when the selected model has no exact value. */ defaultContextWindow?: number @@ -553,7 +554,7 @@ export interface DeepSeekCatalogModel { } ``` -Source: [`packages/llm/llm-deepseek/src/index.ts:34`](../packages/llm/llm-deepseek/src/index.ts) +Source: [`packages/llm/llm-deepseek/src/index.ts:35`](../packages/llm/llm-deepseek/src/index.ts) ## `@deepseek-ai/dsh-llm-pi-ai` diff --git a/docs/cookbook/adding-an-llm-adapter.i18n.yaml b/docs/cookbook/adding-an-llm-adapter.i18n.yaml index 497ae08c32..15c62ae966 100644 --- a/docs/cookbook/adding-an-llm-adapter.i18n.yaml +++ b/docs/cookbook/adding-an-llm-adapter.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write -adding-an-llm-adapter.md: f20442b8c2ce823452a3ea13409f202185d12d04 -adding-an-llm-adapter.zh.md: 2864dd1e18742c7449e24f22504a5a38976ab450 +adding-an-llm-adapter.md: 3adae89f360cd81e264912d74bd86f5a06cd42f8 +adding-an-llm-adapter.zh.md: 80329da5e7c89bfb5ee2e2aa5b9931a3ef0e83f6 diff --git a/docs/cookbook/adding-an-llm-adapter.md b/docs/cookbook/adding-an-llm-adapter.md index f20442b8c2..3adae89f36 100644 --- a/docs/cookbook/adding-an-llm-adapter.md +++ b/docs/cookbook/adding-an-llm-adapter.md @@ -32,7 +32,7 @@ Registration is effect-based (HMR-safe); one adapter per provider route — dupl - A `GenerateOptions` field your provider cannot honor (e.g. a `stop` list on a provider without stop sequences): throw `LlmError(..., 'UNSUPPORTED')` rather than silently dropping it. - If the provider requires response ids, signatures, or other native metadata on follow-up calls, emit the minimal lossless-JSON projection as `finish.replayState`. Validate it when rebuilding history. `LlmService` passes it only when the historical provider route and target provider route are currently owned by the exact same adapter instance; your adapter decides whether same-model, cross-model, or cross-provider restoration is legal. Never infer native replay from provider/model names alone when state is absent. -Provider-specific request knobs (thinking modes, effort levels) belong in the ADAPTER's Config, not in `GenerateOptions` — the core vocabulary stays provider-neutral. +Provider-specific thinking-mode toggles remain in the adapter's Config. Selectable reasoning strength uses the provider-neutral capability seam: return ordered opaque ids from `resolveModelReasoning()`, declare a configured `defaultEffort` only when one exists, and map `GenerateOptions.reasoningEffort` to the provider wire value. Do not expose provider wire spellings, clamp unsupported values, or include an `off` mode as an effort. ## Structure that worked diff --git a/docs/cookbook/adding-an-llm-adapter.zh.md b/docs/cookbook/adding-an-llm-adapter.zh.md index 2864dd1e18..80329da5e7 100644 --- a/docs/cookbook/adding-an-llm-adapter.zh.md +++ b/docs/cookbook/adding-an-llm-adapter.zh.md @@ -32,7 +32,7 @@ export function apply(ctx: Context, config: Config) { - 如果 `GenerateOptions` 中某个字段你的提供方无法支持(例如提供方不支持 stop sequences 时收到 `stop` 列表):抛出 `LlmError(..., 'UNSUPPORTED')`,而非静默丢弃。 - 如果提供方在后续调用中需要响应 ID、签名或其他原生元数据,请将其最小无损 JSON 投影作为 `finish.replayState` 发出。重建历史时验证该状态。只有历史提供方路由和目标提供方路由当前由完全相同的适配器实例拥有时,`LlmService` 才会传递该状态;由适配器决定同模型、跨模型或跨提供方恢复是否合法。状态缺失时,切勿仅根据提供方/模型名称推断原生回放。 -提供方特有的请求旋钮(thinking 模式、effort 级别)放在**适配器**的 Config 中,而非 `GenerateOptions` 中——核心词汇保持提供方无关。 +提供方特有的 thinking 模式开关仍放在适配器的 Config 中。可选的推理强度使用提供方无关的能力 seam:`resolveModelReasoning()` 返回有序的不透明 ID;仅当存在配置指定的默认值时才声明 `defaultEffort`;并将 `GenerateOptions.reasoningEffort` 映射为提供方协议值。不得暴露提供方协议值的具体拼写、自动调整不支持的值,也不得把 `off` 模式列为推理强度。 ## 经验证有效的结构 diff --git a/docs/cordis-catalog/events.md b/docs/cordis-catalog/events.md index 6becfd9434..30cc12af6b 100644 --- a/docs/cordis-catalog/events.md +++ b/docs/cordis-catalog/events.md @@ -547,7 +547,7 @@ Waterfall around every streaming model call (retry, replay, routing). Bound to t Types: [GenerateOptions](../core-data-structures/core.md) · [LlmService](../core-data-structures/llm-streaming.md) · [StreamChunk](../core-data-structures/llm-streaming.md) -Source: [`packages/llm/llm/src/index.ts:52`](../../packages/llm/llm/src/index.ts) +Source: [`packages/llm/llm/src/index.ts:54`](../../packages/llm/llm/src/index.ts) ## `session/*` diff --git a/docs/cordis-catalog/services.md b/docs/cordis-catalog/services.md index c5d6d88d5c..718a45189b 100644 --- a/docs/cordis-catalog/services.md +++ b/docs/cordis-catalog/services.md @@ -671,6 +671,25 @@ async listModels(provider: string): Promise */ async resolveModelContext( provider: string, model: string, ): Promise +/** + * Resolve selectable reasoning efforts from the adapter that owns one exact + * route. Metadata is validated and detached; an absent result means an + * effort selector is unsupported for that model. + * @param provider - registered provider route to inspect. + * @param model - exact model id passed to the adapter. + * @returns detached reasoning metadata, or `undefined` when unsupported. + */ +async resolveModelReasoning( provider: string, model: string, ): Promise + +/** + * Validate a conversation call config against its exact model capability and + * materialize an adapter-configured default. Unsupported explicit efforts + * reject before provider I/O; no clamping or aliasing is performed. + * @param config - provider/model route and optional request controls. + * @returns a detached config only when a default must be materialized. + */ +async resolveCallConfig(config: LlmCallConfig): Promise + /** * Stream one model call as raw chunks (token-level deltas). Throws * `LlmError` with code `NO_ADAPTER` if no adapter is registered for @@ -686,9 +705,9 @@ async resolveModelContext( provider: string, model: string, ): Promise ``` -Types: [GenerateOptions](../core-data-structures/core.md) · [LlmAdapter](../core-data-structures/llm-streaming.md) · [LlmModelContext](../core-data-structures/core.md) · [LlmModelInfo](../core-data-structures/core.md) · [LlmProviderInfo](../core-data-structures/core.md) · [StreamChunk](../core-data-structures/llm-streaming.md) +Types: [GenerateOptions](../core-data-structures/core.md) · [LlmAdapter](../core-data-structures/llm-streaming.md) · [LlmCallConfig](../core-data-structures/core.md) · [LlmModelContext](../core-data-structures/core.md) · [LlmModelInfo](../core-data-structures/core.md) · [LlmModelReasoningInfo](../core-data-structures/core.md) · [LlmProviderInfo](../core-data-structures/core.md) · [StreamChunk](../core-data-structures/llm-streaming.md) -Source: [`packages/llm/llm/src/index.ts:159`](../../packages/llm/llm/src/index.ts) +Source: [`packages/llm/llm/src/index.ts:175`](../../packages/llm/llm/src/index.ts) ## `ctx.permission` — `PermissionService` @@ -1238,7 +1257,7 @@ fork(source: SessionForkSource, boundary?: number, childSessionId?: SessionId): Types: [CreateSessionOptions](../core-data-structures/persistence.md) · [OutOfBandSessionEventType](../core-data-structures/session.md) · [Session](../core-data-structures/session.md) · [SessionEvent](../core-data-structures/core.md) · [SessionEventMap](../core-data-structures/session.md) · [SessionId](../core-data-structures/core.md) · [TurnTrigger](../core-data-structures/session.md) -Source: [`packages/core/session/src/index.ts:605`](../../packages/core/session/src/index.ts) +Source: [`packages/core/session/src/index.ts:610`](../../packages/core/session/src/index.ts) ## `ctx.sessionTitle` — `SessionTitleService` diff --git a/docs/core-data-structures/core.md b/docs/core-data-structures/core.md index 86809ddc8b..5380573c55 100644 --- a/docs/core-data-structures/core.md +++ b/docs/core-data-structures/core.md @@ -205,12 +205,46 @@ interface LlmModelContext { } ``` +Reasoning effort is another exact-route capability. The core brands identifiers but does not enumerate their values; each adapter owns the ordered set, display names, and optional deployment default. + +```ts type-equiv +/** Adapter-owned identifier for one model's selectable reasoning effort. */ +type ReasoningEffortId = Branded<'ReasoningEffortId'> +``` + +```ts type-equiv +/** Display metadata for one adapter-owned reasoning effort. */ +interface LlmReasoningEffortInfo { + /** Opaque stable value accepted by {@link GenerateOptions.reasoningEffort}. */ + id: ReasoningEffortId + /** Human-readable effort name for selectors and diagnostics. */ + name: string + /** Optional user-facing distinction from otherwise similar efforts. */ + description?: string +} +``` + +```ts type-equiv +/** Selectable reasoning efforts for one exact provider/model route. */ +interface LlmModelReasoningInfo { + /** Supported efforts in adapter-preferred display order. */ + efforts: readonly LlmReasoningEffortInfo[] + /** + * Adapter-configured default materialized into requests when callers omit + * an effort. Absence preserves the provider's own default. + */ + defaultEffort?: ReasoningEffortId +} +``` + ```ts type-equiv /** A single model request, fully assembled. */ interface GenerateOptions { /** Registered provider route selecting the adapter instance. */ provider: string model: string + /** Adapter-owned reasoning effort selected for this exact model. */ + reasoningEffort?: ReasoningEffortId /** * Ordered conversation messages, exactly as the provider sees them (after * the `system` slot). A loop-built request assembles them as @@ -287,21 +321,23 @@ The model-facing `ToolSchema` is the wire shape; the registered `ToolDefinition` The loop builds each request from logged state. `EpochHeader` records call config, rendered prompt, authoritative returned tool order (configured by `toolOrder`, or lexicographic when unset), and session prefix through full `request/header` snapshots. Together with derived history, this makes the request reconstructable from the session log. See [session.md](session.md#the-request-header-event-requestheader) and the [reconstructability Agent Note](../../.agents/notes/implemented/architecture/2026-07-05-reconstructable-requests.md). -`agent/request` receives a frozen call-config seed and may return a replacement to switch provider, model, or sampling. `agent/session-prefix` composes request-only prefix messages once per loop instance, and the header records the exact result used. Requests reaching `llm/stream` are deep-frozen, so mutation throws, and carry a process-local loop identity so observers do not confuse separately logged frozen auxiliary calls with conversation requests. +`agent/request` receives a frozen call-config seed and may return a replacement to switch provider, model, reasoning effort, or sampling. The loop resolves the exact model capability after the waterfall, rejects unsupported explicit effort ids without clamping, materializes an adapter-configured default, and logs the effective value. `agent/session-prefix` composes request-only prefix messages once per loop instance, and the header records the exact result used. Requests reaching `llm/stream` are deep-frozen, so mutation throws, and carry a process-local loop identity so observers do not confuse separately logged frozen auxiliary calls with conversation requests. On the wire, a loop-built request reads in this order: the `system` slot (the rendered prompt assembly) → `messagePrefix` (the frozen session prefix) → the derived history — the boundary snapshot, whose tail is the newest `user/message` on a turn's first step and the previous step's tool results on later steps. The prefix never enters the derived history; its durable record is the header events, and the dev invariant recomputes exactly this equation against every loop-built request. -FIXME(call-config-shape): revisit the exact definition of this type — which fields are genuinely epoch-level for cache purposes (`model` certainly; the sampling scalars sit here out of caution), and where provider-specific extras (reasoning options, extra body params) belong when an adapter needs them. +FIXME(call-config-shape): revisit which remaining fields are genuinely epoch-level for cache purposes (`model` and the model-owned reasoning effort are explicit; the sampling scalars sit here out of caution). ```ts type-equiv /** - * Provider + model + sampling scalars of one conversation's requests. Every field maps - * 1:1 onto the same-named `GenerateOptions` field; the loop builds requests - * from the logged header rather than accepting these per call. + * Provider, model, reasoning effort, and sampling scalars of one conversation's + * requests. Every field maps 1:1 onto the same-named `GenerateOptions` field; + * the loop builds requests from the logged header rather than accepting these + * per call. */ interface LlmCallConfig { provider: string model: string + reasoningEffort?: ReasoningEffortId temperature?: number maxTokens?: number stop?: string[] diff --git a/docs/core-data-structures/llm-streaming.md b/docs/core-data-structures/llm-streaming.md index 257ce90cda..b3c285ebba 100644 --- a/docs/core-data-structures/llm-streaming.md +++ b/docs/core-data-structures/llm-streaming.md @@ -154,7 +154,7 @@ declare class BlockAssembler { ## The seam -`LlmAdapter` is the provider seam: subclass, implement `stream()`, and register one adapter instance with `ctx.llm.registerAdapter(providers, adapter)`. `GenerateOptions.provider` selects the registered adapter; `GenerateOptions.model` is passed to that adapter and need not be registered at lifecycle start. Duplicate provider routes fail atomically. Optional `providerInfo()` and asynchronous `listModels()` methods feed `LlmService.listProviders()` / `listModels()` with detached selector metadata. That catalog is advisory rather than a request whitelist: the adapter remains authoritative and may accept unlisted model ids. The separate `resolveModelContext()` query exposes correctness-sensitive capacity for an exact route without making catalog membership authoritative; absence means unknown metadata, not invalid routing. Adapter lookup happens at the terminal continuation of the `llm/stream` waterfall, so a listener may short-circuit the call or route a mutable one-shot request before lookup. The `block-start` / `block-end` `index` correlation and the assembler together mean an adapter only has to emit well-formed chunks — block reassembly is not each adapter's problem. The consumer surface (`ctx.llm.stream()`) and the `llm/stream` waterfall are described in [architecture.md § Content blocks and streaming](../architecture.md#content-blocks-and-streaming-dsh-llm). +`LlmAdapter` is the provider seam: subclass, implement `stream()`, and register one adapter instance with `ctx.llm.registerAdapter(providers, adapter)`. `GenerateOptions.provider` selects the registered adapter; `GenerateOptions.model` is passed to that adapter and need not be registered at lifecycle start. Duplicate provider routes fail atomically. Optional `providerInfo()` and asynchronous `listModels()` methods feed `LlmService.listProviders()` / `listModels()` with detached selector metadata. That catalog is advisory rather than a request whitelist: the adapter remains authoritative and may accept unlisted model ids. The separate `resolveModelContext()` query exposes correctness-sensitive capacity, while `resolveModelReasoning()` exposes ordered model-owned effort ids and an optional deployment default; absence from either query means unavailable metadata or capability, not invalid catalog membership. The service validates and materializes reasoning through `resolveCallConfig()` at the final adapter boundary, so direct calls cannot bypass unsupported-effort rejection. Adapter lookup happens at the terminal continuation of the `llm/stream` waterfall, so a listener may short-circuit the call or route a mutable one-shot request before lookup. The `block-start` / `block-end` `index` correlation and the assembler together mean an adapter only has to emit well-formed chunks — block reassembly is not each adapter's problem. The consumer surface (`ctx.llm.stream()`) and the `llm/stream` waterfall are described in [architecture.md § Content blocks and streaming](../architecture.md#content-blocks-and-streaming-dsh-llm). ```ts public-api /** @@ -189,6 +189,17 @@ declare abstract class LlmAdapter { _provider: string, _model: string, ): Promise; + /** + * Resolve selectable reasoning efforts for one exact model. Absence means + * the model has no selectable reasoning-effort capability. + * @param _provider - one provider route owned by this adapter. + * @param _model - exact model id passed to {@link GenerateOptions.model}. + * @returns adapter-owned effort metadata, or `undefined` when unsupported. + */ + resolveModelReasoning( + _provider: string, + _model: string, + ): Promise; /** * Stream one model call as raw chunks. The only required method. * @param options - the fully-assembled request; implementations must honor `options.signal`. diff --git a/docs/core-data-structures/session.md b/docs/core-data-structures/session.md index ba1dcb98ba..eb530f7709 100644 --- a/docs/core-data-structures/session.md +++ b/docs/core-data-structures/session.md @@ -167,7 +167,7 @@ The request envelope — the `EpochHeader` (call config + rendered system prompt * canonical empty optional fields are absent. */ interface EpochHeader { - /** The conversation's call configuration (provider, model, and sampling scalars). */ + /** The conversation's call configuration (provider, model, reasoning effort, and sampling scalars). */ config: LlmCallConfig /** Rendered system prompt text; absent for a system-less request. */ system?: string diff --git a/docs/event-producer-consumer.md b/docs/event-producer-consumer.md index 179ad0dcee..5659a8c7bf 100644 --- a/docs/event-producer-consumer.md +++ b/docs/event-producer-consumer.md @@ -30,7 +30,7 @@ This matrix shows which packages dispatch each harness-owned event and which pac | `fs/observed` | `emit` | [`packages/fs/fs/src/index.ts:71`](../packages/fs/fs/src/index.ts) | [`tool-fs`](../packages/fs/tool-fs) (`emit`) | [`fs-policy`](../packages/fs/fs-policy) | | `fs/write-intent` | `waterfall` | [`packages/fs/fs/src/index.ts:54`](../packages/fs/fs/src/index.ts) | [`tool-fs`](../packages/fs/tool-fs) (`waterfall`) | [`fs-policy`](../packages/fs/fs-policy) | | `goal/changed` | `emit` | [`packages/goal/goal/src/types.ts:167`](../packages/goal/goal/src/types.ts) | [`goal`](../packages/goal/goal) (`emit`) | [`goal-session`](../packages/goal/goal-session) | -| `llm/stream` | `waterfall` | [`packages/llm/llm/src/index.ts:52`](../packages/llm/llm/src/index.ts) | [`llm`](../packages/llm/llm) (`waterfall`) | [`agent-loop`](../packages/core/agent-loop), [`llm`](../packages/llm/llm), [`llm-replay`](../packages/support/llm-replay), [`session-checkpoint-policy`](../packages/session-persistence/session-checkpoint-policy), [`session-title`](../packages/session-title/session-title) | +| `llm/stream` | `waterfall` | [`packages/llm/llm/src/index.ts:54`](../packages/llm/llm/src/index.ts) | [`llm`](../packages/llm/llm) (`waterfall`) | [`agent-loop`](../packages/core/agent-loop), [`llm`](../packages/llm/llm), [`llm-replay`](../packages/support/llm-replay), [`session-checkpoint-policy`](../packages/session-persistence/session-checkpoint-policy), [`session-title`](../packages/session-title/session-title) | | `session/created` | `emit` | [`packages/core/session/src/index.ts:79`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`compact`](../packages/compact/compact), [`goal`](../packages/goal/goal), [`hook-protocol`](../packages/hooks/hook-protocol), [`jsonrpc`](../packages/ui/jsonrpc), [`llm-retry`](../packages/llm/llm-retry), `runtime`, [`session`](../packages/core/session), [`session-persistence`](../packages/session-persistence/session-persistence), [`user-approval`](../packages/ui/user-approval) | | `session/disposed` | `emit` | [`packages/core/session/src/index.ts:89`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`agent-loop`](../packages/core/agent-loop), `runtime`, [`session-persistence`](../packages/session-persistence/session-persistence), [`session-title`](../packages/session-title/session-title) | | `session/event` | `emit` | [`packages/core/session/src/index.ts:101`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`acp`](../packages/ui/acp), [`cli-demo`](../packages/examples/cli-demo), [`compact`](../packages/compact/compact), [`goal`](../packages/goal/goal), [`goal-session`](../packages/goal/goal-session), [`hook-protocol`](../packages/hooks/hook-protocol), [`jsonrpc`](../packages/ui/jsonrpc), `runtime`, [`session`](../packages/core/session), [`session-persistence`](../packages/session-persistence/session-persistence), [`session-title`](../packages/session-title/session-title), [`token-meter`](../packages/llm/token-meter), [`tui`](../packages/ui/tui), [`user-approval`](../packages/ui/user-approval), [`workspace-context`](../packages/context/workspace-context) | diff --git a/docs/user/develop/practice/llm-adapter.i18n.yaml b/docs/user/develop/practice/llm-adapter.i18n.yaml index 8735e8d5a6..c179b693f8 100644 --- a/docs/user/develop/practice/llm-adapter.i18n.yaml +++ b/docs/user/develop/practice/llm-adapter.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write -llm-adapter.md: 3e83289b8072ef231f83c0fa3cfe3260547b42fa -llm-adapter.zh.md: 92fcf9b22f4bb356ada4c46f9a03ef0cc2d159da +llm-adapter.md: f133a6ea748fd7d9bfa6a7adaa35b1cccbe40c8e +llm-adapter.zh.md: 3a4fd9a516126d1f9fa8675fdc87be10c85f3489 diff --git a/docs/user/develop/practice/llm-adapter.md b/docs/user/develop/practice/llm-adapter.md index 3e83289b80..f133a6ea74 100644 --- a/docs/user/develop/practice/llm-adapter.md +++ b/docs/user/develop/practice/llm-adapter.md @@ -110,7 +110,9 @@ async function* exampleChunks(): AsyncIterable { ## GenerateOptions -`stream()` receives the exported `GenerateOptions` type. It includes the model, conversation history, system prompt, tool schemas, generation parameters, stop sequences, and abort signal; treat the TypeScript type exported by `@deepseek-ai/dsh-llm` as authoritative. Map supported fields to the provider API. If the provider cannot honor a field, throw `LlmError` with a stable code instead of silently dropping it. +`stream()` receives the exported `GenerateOptions` type. It includes the model, adapter-owned reasoning-effort id, conversation history, system prompt, tool schemas, generation parameters, stop sequences, and abort signal; treat the TypeScript type exported by `@deepseek-ai/dsh-llm` as authoritative. Map supported fields to the provider API. If the provider cannot honor a field, throw `LlmError` with a stable code instead of silently dropping it. + +Override `resolveModelReasoning(provider, model)` when an exact model exposes selectable reasoning strengths. Return ordered opaque ids and display names plus an optional configured default; do not promote provider names into a core enum. The service validates the metadata and rejects unsupported explicit values before `stream()`. Returning `undefined` means that model has no selectable reasoning-effort capability. ## Register an adapter diff --git a/docs/user/develop/practice/llm-adapter.zh.md b/docs/user/develop/practice/llm-adapter.zh.md index 92fcf9b22f..3a4fd9a516 100644 --- a/docs/user/develop/practice/llm-adapter.zh.md +++ b/docs/user/develop/practice/llm-adapter.zh.md @@ -110,7 +110,9 @@ async function* exampleChunks(): AsyncIterable { ## GenerateOptions -`stream()` 接收仓库导出的 `GenerateOptions`。它包含模型名、对话历史、系统提示词、tool schema、生成参数、停止序列和中止信号;完整字段以 `@deepseek-ai/dsh-llm` 导出的 TypeScript 类型为准。适配器必须将支持的字段映射到具体 API;无法支持的字段应抛出带稳定 code 的 `LlmError`,不能静默丢弃。 +`stream()` 接收仓库导出的 `GenerateOptions`。它包含模型名、由适配器持有的推理强度 ID、对话历史、系统提示词、tool schema、生成参数、停止序列和中止信号;完整字段以 `@deepseek-ai/dsh-llm` 导出的 TypeScript 类型为准。适配器必须将支持的字段映射到具体 API;无法支持的字段应抛出带稳定 code 的 `LlmError`,不能静默丢弃。 + +当某个具体模型提供可选推理强度时,请覆写 `resolveModelReasoning(provider, model)`。返回有序的不透明 ID、展示名称,以及可选的配置默认值;不要将提供方使用的等级名称提升为核心枚举。服务会校验元数据,并在调用 `stream()` 前拒绝显式指定但不受支持的值。返回 `undefined` 表示该模型没有可选的推理强度能力。 ## 注册适配器 diff --git a/examples/headless-agent/tests/fixtures/cli-mock-llm.ts b/examples/headless-agent/tests/fixtures/cli-mock-llm.ts index 5238e67374..23b5e18fc9 100644 --- a/examples/headless-agent/tests/fixtures/cli-mock-llm.ts +++ b/examples/headless-agent/tests/fixtures/cli-mock-llm.ts @@ -1,8 +1,28 @@ import type { Context } from 'cordis' -import { CallId, LlmAdapter, type GenerateOptions, type StreamChunk } from '@deepseek-ai/dsh-llm' +import { + CallId, + LlmAdapter, + ReasoningEffortId, + type GenerateOptions, + type LlmModelReasoningInfo, + type StreamChunk, +} from '@deepseek-ai/dsh-llm' + +const HIGH = ReasoningEffortId('high') +const MAX = ReasoningEffortId('max') /** Keyless headless-agent adapter: one real bash call followed by a final answer. */ class CliMockAdapter extends LlmAdapter { + override async resolveModelReasoning(): Promise { + return { + efforts: [ + { id: HIGH, name: 'High' }, + { id: MAX, name: 'Max' }, + ], + defaultEffort: HIGH, + } + } + async * stream(options: GenerateOptions): AsyncIterable { const toolResult = options.messages.at(-1)?.content.find(block => block.type === 'tool-result') if (toolResult === undefined) { @@ -34,4 +54,8 @@ export const inject = ['llm'] /** Register the keyless `cli-mock` adapter. */ export function apply(ctx: Context): void { ctx.llm.registerAdapter(['cli-mock'], new CliMockAdapter()) + ctx.on('agent/request', async (_agent, _turn, step, _config, _signal, next) => { + const config = await next() + return step === 2 ? { ...config, reasoningEffort: MAX } : config + }) } diff --git a/examples/headless-agent/tests/headless.snapshot.ts b/examples/headless-agent/tests/headless.snapshot.ts index 1626489bf5..70c1fa8bfe 100644 --- a/examples/headless-agent/tests/headless.snapshot.ts +++ b/examples/headless-agent/tests/headless.snapshot.ts @@ -28,6 +28,7 @@ const ralphScenarioDir = join(snapshotsDir, 'ralph-loop') const ralphConfigPath = fileURLToPath(new URL('../ralph.cordis.snapshot.yml', import.meta.url)) const binScript = fileURLToPath(new URL('../../../packages/examples/cli-demo/src/bin.ts', import.meta.url)) const tsconfigPath = fileURLToPath(new URL('../../../tsconfig.json', import.meta.url)) +const reasoningConfigPath = fileURLToPath(new URL('./fixtures/cli.cordis.yml', import.meta.url)) const refreshing = process.env.DSH_SNAPSHOT === 'refresh' interface JsonObject { @@ -124,6 +125,46 @@ async function persistedLogs(cwd: string): Promise { } describe('headless stream-json snapshots', () => { + it('logs the model default and a dynamic next-step reasoning effort', async () => { + const result = await runLoaderSmoke({ + label: 'reasoning effort headless stream-json snapshot', + tempDirPrefix: 'headless-snapshot-reasoning-effort-', + binScript, + configPath: reasoningConfigPath, + binArgs: ['--config', reasoningConfigPath, '--output-format', 'stream-json', 'prove dynamic reasoning effort'], + tsconfigPath, + }) + + expect(result.stderr).toBe('') + const headers = parseJsonl(result.stdout) + .map(record => record.event) + .filter((event): event is JsonObject => ( + event !== null + && typeof event === 'object' + && !Array.isArray(event) + && 'type' in event + && event.type === 'request/header' + )) + .map((event) => { + const data = event.data as JsonObject + return (data.header as JsonObject).config + }) + expect(headers).toMatchInlineSnapshot(` + [ + { + "model": "cli-mock", + "provider": "cli-mock", + "reasoningEffort": "high", + }, + { + "model": "cli-mock", + "provider": "cli-mock", + "reasoningEffort": "max", + }, + ] + `) + }, LOADER_SMOKE_TEST_TIMEOUT_MS) + it('replays the advanced toolchain through the one-shot app', async () => { const prompt = await scenarioPrompt(advancedScenarioDir, 'advanced-toolchain') const fixtureFiles = [ diff --git a/packages/cordis/tool-cordis/src/api-catalog.ts b/packages/cordis/tool-cordis/src/api-catalog.ts index ffc1670f4e..604a490405 100644 --- a/packages/cordis/tool-cordis/src/api-catalog.ts +++ b/packages/cordis/tool-cordis/src/api-catalog.ts @@ -344,6 +344,14 @@ export const SERVICE_API: readonly ServiceApiEntry[] = [ signature: 'async resolveModelContext( provider: string, model: string, ): Promise', jsDoc: '/**\n * Resolve context capacity from the adapter that owns one exact route.\n * This query is independent of the advisory model catalog: an unlisted model\n * may return metadata, while `undefined` never rejects later routing.\n * @param provider - registered provider route to inspect.\n * @param model - exact model id passed to the adapter.\n * @returns detached context metadata, or `undefined` when the adapter has none.\n */', }, + { + signature: 'async resolveModelReasoning( provider: string, model: string, ): Promise', + jsDoc: '/**\n * Resolve selectable reasoning efforts from the adapter that owns one exact\n * route. Metadata is validated and detached; an absent result means an\n * effort selector is unsupported for that model.\n * @param provider - registered provider route to inspect.\n * @param model - exact model id passed to the adapter.\n * @returns detached reasoning metadata, or `undefined` when unsupported.\n */', + }, + { + signature: 'async resolveCallConfig(config: LlmCallConfig): Promise', + jsDoc: '/**\n * Validate a conversation call config against its exact model capability and\n * materialize an adapter-configured default. Unsupported explicit efforts\n * reject before provider I/O; no clamping or aliasing is performed.\n * @param config - provider/model route and optional request controls.\n * @returns a detached config only when a default must be materialized.\n */', + }, { signature: 'stream(options: GenerateOptions): AsyncIterable', jsDoc: '/**\n * Stream one model call as raw chunks (token-level deltas). Throws\n * `LlmError` with code `NO_ADAPTER` if no adapter is registered for\n * `options.provider`. Replay state is retained only when the same adapter\n * instance owns its historical provider and the target provider. Final\n * adapter selection, dispatch, and iteration failures retain their original\n * Error identity and are tagged in a call-local scope for narrow agent-loop\n * request recovery; middleware and nested-call failures remain untagged for\n * the outer call.\n * @param options - the full request; `options.provider` selects the adapter.\n * @returns the chunk stream, possibly wrapped by `llm/stream` listeners.\n */', @@ -1455,7 +1463,7 @@ export const TYPE_API: readonly TypeApiEntry[] = [ }, { name: 'GenerateOptions', - declaration: 'export interface GenerateOptions {\n provider: string;\n model: string;\n messages: Message[];\n system?: string;\n tools?: ToolSchema[];\n temperature?: number;\n maxTokens?: number;\n stop?: string[];\n signal?: AbortSignal;\n sessionId?: Branded<\'SessionId\'>;\n purpose?: \'compaction\' | \'session-title\';\n}', + declaration: 'export interface GenerateOptions {\n provider: string;\n model: string;\n reasoningEffort?: ReasoningEffortId;\n messages: Message[];\n system?: string;\n tools?: ToolSchema[];\n temperature?: number;\n maxTokens?: number;\n stop?: string[];\n signal?: AbortSignal;\n sessionId?: Branded<\'SessionId\'>;\n purpose?: \'compaction\' | \'session-title\';\n}', }, { name: 'GenericCallView', @@ -1527,7 +1535,7 @@ export const TYPE_API: readonly TypeApiEntry[] = [ }, { name: 'LlmCallConfig', - declaration: 'export interface LlmCallConfig {\n provider: string;\n model: string;\n temperature?: number;\n maxTokens?: number;\n stop?: string[];\n}', + declaration: 'export interface LlmCallConfig {\n provider: string;\n model: string;\n reasoningEffort?: ReasoningEffortId;\n temperature?: number;\n maxTokens?: number;\n stop?: string[];\n}', }, { name: 'LlmFailure', @@ -1541,10 +1549,18 @@ export const TYPE_API: readonly TypeApiEntry[] = [ name: 'LlmModelInfo', declaration: 'export interface LlmModelInfo {\n provider: string;\n id: string;\n name: string;\n description?: string;\n}', }, + { + name: 'LlmModelReasoningInfo', + declaration: 'export interface LlmModelReasoningInfo {\n efforts: readonly LlmReasoningEffortInfo[];\n defaultEffort?: ReasoningEffortId;\n}', + }, { name: 'LlmProviderInfo', declaration: 'export interface LlmProviderInfo {\n id: string;\n name: string;\n}', }, + { + name: 'LlmReasoningEffortInfo', + declaration: 'export interface LlmReasoningEffortInfo {\n id: ReasoningEffortId;\n name: string;\n description?: string;\n}', + }, { name: 'Message', declaration: 'export interface Message {\n role: \'system\' | \'user\' | \'assistant\';\n content: ContentBlock[];\n provenance?: AssistantProvenance;\n}', @@ -1689,6 +1705,10 @@ export const TYPE_API: readonly TypeApiEntry[] = [ name: 'ReasoningBlock', declaration: 'export interface ReasoningBlock {\n type: \'reasoning\';\n text: string;\n}', }, + { + name: 'ReasoningEffortId', + declaration: 'export type ReasoningEffortId = Branded<\'ReasoningEffortId\'>;', + }, { name: 'ResumeAgentOptions', declaration: 'export interface ResumeAgentOptions {\n readonly resumeSessionId: SessionId;\n readonly agentOptions?: AgentOptions;\n readonly signal?: AbortSignal;\n readonly setup?: (agentCtx: Context) => Promise | void;\n}', diff --git a/packages/core/agent-loop/README.md b/packages/core/agent-loop/README.md index 79fbbd7824..0390d54f94 100644 --- a/packages/core/agent-loop/README.md +++ b/packages/core/agent-loop/README.md @@ -60,6 +60,8 @@ The driver owns one agent for its lifetime and runs inside `ctx.agents.withIniti Every provider call that reaches a successful finish appends exactly one `assistant/message` completion anchor, including content-less calls and `max-tokens` finishes. A successful `agent/step-result` stores its transformed content; a rejected result records empty content before the original failure continues. The anchor retains exact chunk provenance (`[]` for a stream with no chunks) and usage when available, while empty content stays out of derived message history. +After `agent/request` returns a provider/model call config, the loop asks `ctx.llm.resolveCallConfig()` to validate any adapter-owned reasoning effort and materialize its configured default. The effective config is logged in the full `request/header` before dispatch, so a listener can change effort between steps without hidden request drift. A route with no registered adapter preserves the proposed config so an `llm/stream` listener can own and short-circuit it; unhandled terminal dispatch still fails with `NO_ADAPTER`. A new loop instance restores the last effort only when its initial provider/model route exactly matches the logged route; a route change discards that opaque model-owned ID and resolves the new model independently. + Plugin failure ends the current turn, not the loop. Only final adapter dispatch/iteration failures and terminal in-band error or aborted finishes enter `agent/request-error`; middleware, result processing, tools, and `agent/post-step` remain ordinary turn failures. Recovery receives the exact live error, immutable provider facts, and immutable prior failures after the failed step closes. A retry rebuilds from the durable log in a new numbered step, success clears the consecutive history, and exhaustion records the structured failure once on `turn/end`. AgentLoop privately owns one cancellation holder whose explicit signal spans prompt policy, assembly, every step, model and tool work, recovery, continuation, and terminal stop; it retires the holder immediately before publishing `turn/end`, while the driver may remain `running` through the durability flush. An effective `cancel()` emits the typed runtime-only `user | parent` cause before clearing pending work and cooperatively aborting the holder; notification failures cannot veto cancellation, work queued by a notification observer is cleared, work queued by a later abort observer belongs to the next turn, and idle cancellation emits nothing. Durable `turn/end` remains coarse `aborted`; undispatched model tool calls receive synthetic `tool/call` and `ABORTED_BEFORE_DISPATCH` result pairs. Disposal wins terminal classification, and work that ignores the signal must settle before quiescence. The [explicit-cancellation decision](../../../.agents/notes/implemented/architecture/2026-07-16-explicit-turn-cancellation.md) owns the lifecycle and race contract. Terminal continuation stops remain authoritative through turn close and durability flush. Within a step, exclusive calls form barriers; parallel-safe calls use a bounded rolling pool and are reclassified before start. Only dispatch/body overlaps. Policy, durable results, and result context remain model-ordered. Abort stops new calls, drains started results, then drains accepted batch context before the turn closes through the normal abort path. diff --git a/packages/core/agent-loop/src/loop.ts b/packages/core/agent-loop/src/loop.ts index 1cfd913b77..dae3cf0efc 100644 --- a/packages/core/agent-loop/src/loop.ts +++ b/packages/core/agent-loop/src/loop.ts @@ -621,19 +621,43 @@ async function runStep( // Seed the first request from agent options and later requests from the logged header; // detach and freeze so listeners must return an attributable replacement. - const seedConfig: LlmCallConfig = deepFreeze(structuredClone(transmission.loggedHeader - // eslint-disable-next-line @typescript-eslint/no-non-null-assertion -- loggedHeader ⟹ a snapshot is in the log - ? session.requestHeader()!.config - : { provider: options.provider ?? '', model: options.model ?? '' })) + const loggedConfig = session.requestHeader()?.config + const initialProvider = options.provider ?? '' + const initialModel = options.model ?? '' + const initialConfig: LlmCallConfig = { + provider: initialProvider, + model: initialModel, + ...loggedConfig?.provider === initialProvider + && loggedConfig.model === initialModel + && loggedConfig.reasoningEffort !== undefined + ? { reasoningEffort: loggedConfig.reasoningEffort } + : {}, + } + const seedConfig: LlmCallConfig = deepFreeze(structuredClone( + transmission.loggedHeader + // eslint-disable-next-line @typescript-eslint/no-non-null-assertion -- loggedHeader ⟹ a snapshot is in the log + ? session.requestHeader()!.config + : initialConfig, + )) // Listener replacements are recorded in the request header before dispatch. - const config = await events.waterfall( + const proposedConfig = await events.waterfall( 'agent/request', turn, step, seedConfig, signal, () => Promise.resolve(seedConfig), ) interruptionCheckpoint(signal) - if (!config.provider || !config.model) { + if (!proposedConfig.provider || !proposedConfig.model) { throw new Error(`agent "${agent.id}" has no provider/model: set AgentOptions.provider and AgentOptions.model or supply both via the agent/request waterfall`) } + let config: LlmCallConfig + try { + config = await ctx.llm.resolveCallConfig(proposedConfig) + } catch (error: unknown) { + // A waterfall listener may own and short-circuit a route with no adapter. + // Terminal dispatch still raises NO_ADAPTER when no listener handles it. + if (!(error instanceof LlmError) || error.code !== 'NO_ADAPTER') throw error + config = proposedConfig + } + interruptionCheckpoint(signal) // eslint-disable-next-line @typescript-eslint/no-non-null-assertion -- runTurn composes the prefix before every runStep call const sessionPrefix = transmission.sessionPrefix! @@ -651,6 +675,9 @@ async function runStep( const request: GenerateOptions = markAgentLoopRequest(deepFreeze({ provider: header.config.provider, model: header.config.model, + ...header.config.reasoningEffort !== undefined + ? { reasoningEffort: header.config.reasoningEffort } + : {}, messages: [...header.messagePrefix ?? [], ...boundaryMessages], ...header.system !== undefined ? { system: header.system } : {}, ...header.tools !== undefined ? { tools: header.tools } : {}, diff --git a/packages/core/agent-loop/tests/mock-adapter.ts b/packages/core/agent-loop/tests/mock-adapter.ts index 4aaff72415..8b47a9880d 100644 --- a/packages/core/agent-loop/tests/mock-adapter.ts +++ b/packages/core/agent-loop/tests/mock-adapter.ts @@ -1,4 +1,4 @@ -import type { GenerateOptions, StreamChunk } from '@deepseek-ai/dsh-llm' +import type { GenerateOptions, LlmModelReasoningInfo, StreamChunk } from '@deepseek-ai/dsh-llm' import { CallId, LlmAdapter } from '@deepseek-ai/dsh-llm' /** Helpers to write scripted responses tersely. */ @@ -64,10 +64,20 @@ export function toolCallResponse(rawCallId: string, name: string, args: object, export class MockAdapter extends LlmAdapter { requests: GenerateOptions[] = [] - constructor(private script: (StreamChunk[] | ((options: GenerateOptions) => StreamChunk[]) | 'hang')[]) { + constructor( + private script: (StreamChunk[] | ((options: GenerateOptions) => StreamChunk[]) | 'hang')[], + private readonly reasoning?: LlmModelReasoningInfo, + ) { super() } + override resolveModelReasoning( + _provider: string, + _model: string, + ): Promise { + return Promise.resolve(this.reasoning) + } + async * stream(options: GenerateOptions): AsyncIterable { this.requests.push(options) const entry = this.script.shift() diff --git a/packages/core/agent-loop/tests/request-reconstruction.spec.ts b/packages/core/agent-loop/tests/request-reconstruction.spec.ts index 63c59618e3..781ee908a7 100644 --- a/packages/core/agent-loop/tests/request-reconstruction.spec.ts +++ b/packages/core/agent-loop/tests/request-reconstruction.spec.ts @@ -7,7 +7,7 @@ import { describe, expect, it } from 'vitest' import { Context } from 'cordis' -import LlmService from '@deepseek-ai/dsh-llm' +import LlmService, { LlmError, ReasoningEffortId } from '@deepseek-ai/dsh-llm' import type { GenerateOptions } from '@deepseek-ai/dsh-llm' import SessionStore, { Session, SessionId, foldRequestHeader } from '@deepseek-ai/dsh-session' import SystemPrompt from '@deepseek-ai/dsh-system-prompt' @@ -104,6 +104,81 @@ describe('request stability across the loop', () => { expectPrefixExtension(adapter.requests[0]!, adapter.requests[1]!) }) + it('logs adapter defaults, supports per-turn effort changes, and restores the effective value', async () => { + const reasoning = { + efforts: [ + { id: ReasoningEffortId('high'), name: 'High' }, + { id: ReasoningEffortId('max'), name: 'Max' }, + ], + defaultEffort: ReasoningEffortId('high'), + } + const adapter = new MockAdapter([textResponse('one'), textResponse('two')], reasoning) + const ctx = await harness(adapter) + const agent = ctx.agentLoop.create(SessionId('effort'), { provider: 'mock', model: 'mock' }) + ctx.on('agent/request', async (_agent, turn, _step, _config, _signal, next) => { + const config = await next() + return turn === 2 ? { ...config, reasoningEffort: ReasoningEffortId('max') } : config + }) + + send(agent, 'first') + await waitForIdle(ctx, agent) + send(agent, 'second') + await waitForIdle(ctx, agent) + + expect(adapter.requests.map(request => request.reasoningEffort)).toEqual([ + ReasoningEffortId('high'), + ReasoningEffortId('max'), + ]) + const headers = agent.session.events.filter(event => event.type === 'request/header') + expect(headers.map(event => event.data.header.config.reasoningEffort)).toEqual([ + ReasoningEffortId('high'), + ReasoningEffortId('max'), + ]) + expect(headers.map(event => event.data.reason)).toEqual(['initial', 'change']) + + const resumedAdapter = new MockAdapter([textResponse('three')], reasoning) + const resumedCtx = await harness(resumedAdapter) + const resumedHandle = await resumedCtx.agents.create({ + sessionId: SessionId('effort-resumed'), + seed: structuredClone(agent.session.events), + agentOptions: { provider: 'mock', model: 'mock' }, + }) + send(resumedHandle.agent, 'third') + await waitForIdle(resumedCtx, resumedHandle.agent) + + expect(resumedAdapter.requests[0]?.reasoningEffort).toBe(ReasoningEffortId('max')) + const resumedHeaders = resumedHandle.agent.session.events.filter(event => event.type === 'request/header') + expect(resumedHeaders.at(-1)?.data.header.config.reasoningEffort).toBe(ReasoningEffortId('max')) + expect(resumedHeaders.at(-1)?.data.reason).toBe('resume') + }) + + it.each(['plain error', 'LLM error'] as const)( + 'does not swallow a %s from reasoning resolution', + async (kind) => { + const failure = kind === 'plain error' + ? new Error('reasoning metadata failed') + : new LlmError('unsupported effort', 'UNSUPPORTED_REASONING_EFFORT') + const adapter = new class extends MockAdapter { + override resolveModelReasoning(): Promise { + return Promise.reject(failure) + } + }([]) + const ctx = await harness(adapter) + const errors: Error[] = [] + ctx.on('agent/error', (_agent, _turn, _step, error) => void errors.push(error)) + const agent = ctx.agentLoop.create(SessionId(`reasoning-${kind}`), { + provider: 'mock', + model: 'mock', + }) + + send(agent, 'go') + await waitForIdle(ctx, agent) + + expect(errors).toContain(failure) + expect(adapter.requests).toHaveLength(0) + }, + ) + it('a compaction replace rewrites the resend, and the log explains it', async () => { const adapter = new MockAdapter([textResponse('one'), textResponse('two')]) const ctx = await harness(adapter) @@ -301,6 +376,7 @@ describe('request stability across the loop', () => { const firstChunk = events.find(e => e.type === 'assistant/chunk' && e.seq > stepStart.seq)! const header = foldRequestHeader(events.slice(0, firstChunk.seq))! expect(request.model).toBe(header.config.model) + expect(request.reasoningEffort).toBe(header.config.reasoningEffort) expect(request.system).toEqual(header.system) expect(structuredClone(request.tools ?? [])).toEqual(structuredClone(header.tools ?? [])) expect(request.temperature).toBe(header.config.temperature) diff --git a/packages/core/session/src/index.ts b/packages/core/session/src/index.ts index 9c5831bb18..4f287b1517 100644 --- a/packages/core/session/src/index.ts +++ b/packages/core/session/src/index.ts @@ -181,6 +181,11 @@ function assertCurrentLlmShape(event: Record, index: number): v const header = record['header'] const config = typeof header === 'object' && header !== null ? (header as Record)['config'] : undefined if (!hasProviderModel(config)) throw new Error(`seed request/header at index ${index} lacks provider/model`) + const reasoningEffort = (config as Record)['reasoningEffort'] + if (reasoningEffort !== undefined + && (typeof reasoningEffort !== 'string' || reasoningEffort.length === 0)) { + throw new Error(`seed request/header at index ${index} has an invalid reasoningEffort`) + } } if (event['type'] === 'assistant/message' && !hasProviderModel(record['provenance'])) { throw new Error(`seed assistant/message at index ${index} lacks provider/model provenance`) diff --git a/packages/core/session/src/types.ts b/packages/core/session/src/types.ts index 8b3a1e8cb6..4bc1c7eb04 100644 --- a/packages/core/session/src/types.ts +++ b/packages/core/session/src/types.ts @@ -156,7 +156,7 @@ export interface TodoItem { * canonical empty optional fields are absent. */ export interface EpochHeader { - /** The conversation's call configuration (provider, model, and sampling scalars). */ + /** The conversation's call configuration (provider, model, reasoning effort, and sampling scalars). */ config: LlmCallConfig /** Rendered system prompt text; absent for a system-less request. */ system?: string diff --git a/packages/core/session/tests/request-header.spec.ts b/packages/core/session/tests/request-header.spec.ts index b185cf6e56..e2fd4cc781 100644 --- a/packages/core/session/tests/request-header.spec.ts +++ b/packages/core/session/tests/request-header.spec.ts @@ -4,6 +4,7 @@ import { describe, expect, it } from 'vitest' import { Session, SessionId, canonicalHeader, foldRequestHeader, headerEquals } from '@deepseek-ai/dsh-session' import type { EpochHeader, SessionEvent } from '@deepseek-ai/dsh-session' import type { Message, ToolSchema } from '@deepseek-ai/dsh-llm' +import { ReasoningEffortId } from '@deepseek-ai/dsh-llm' const CONFIG = { provider: 'mock', model: 'm' } @@ -29,6 +30,10 @@ describe('headerEquals', () => { it('compares every canonical field and preserves tool order', () => { expect(headerEquals(base, structuredClone(base))).toBe(true) expect(headerEquals(base, { ...base, config: { provider: 'mock', model: 'other' } })).toBe(false) + expect(headerEquals(base, { + ...base, + config: { ...base.config, reasoningEffort: ReasoningEffortId('high') }, + })).toBe(false) expect(headerEquals(base, { ...base, system: 'other' })).toBe(false) expect(headerEquals(base, { ...base, messagePrefix: [msg('other')] })).toBe(false) expect(headerEquals(base, { ...base, tools: [] })).toBe(false) diff --git a/packages/core/session/tests/session.spec.ts b/packages/core/session/tests/session.spec.ts index d880153dd3..880a5df42f 100644 --- a/packages/core/session/tests/session.spec.ts +++ b/packages/core/session/tests/session.spec.ts @@ -1,6 +1,6 @@ import { describe, expect, expectTypeOf, it, vi } from 'vitest' import { Context } from 'cordis' -import { CallId } from '@deepseek-ai/dsh-llm' +import { CallId, ReasoningEffortId } from '@deepseek-ai/dsh-llm' import SessionStore, { displayPromptContent, findLastMessageTurnEnd, @@ -118,6 +118,11 @@ describe('Session', () => { }) it('renders context and steering messages as plain user content', () => { + expect(displayPromptContent({ + content: [{ type: 'text', text: 'plain prompt' }], + source: { kind: 'user' }, + })).toEqual([{ type: 'text', text: 'plain prompt' }]) + const session = new Session(SessionId('s2')) session.append('context/message', { content: [{ type: 'text', text: 'file changed: a.ts' }], @@ -228,6 +233,35 @@ describe('Session', () => { .toEqual([unrelatedPrimitiveData]) }) + it('round-trips a non-empty reasoning effort and rejects invalid durable values', () => { + const valid = { + type: 'request/header', + seq: 0, + time: 1, + data: { + header: { + config: { + provider: 'mock', + model: 'model', + reasoningEffort: ReasoningEffortId('adapter-owned'), + }, + }, + reason: 'initial', + }, + } as const + expect(new Session(SessionId('reasoning-effort'), [valid]).events[0]) + .toEqual(valid) + + for (const reasoningEffort of ['', 1]) { + const invalid = structuredClone(valid) as unknown as SessionEvent + if (invalid.type !== 'request/header') throw new Error('test fixture must be a request header') + const config = invalid.data.header.config as unknown as Record + config.reasoningEffort = reasoningEffort + expect(() => new Session(SessionId('invalid-reasoning-effort'), [invalid])) + .toThrow('seed request/header at index 0 has an invalid reasoningEffort') + } + }) + it('isolates the log from mutation through a derived message (append-only contract)', () => { const session = new Session(SessionId('s4')) session.append('user/message', { content: [{ type: 'text', text: 'original' }], source: { kind: 'user' } }, { surfaceOp: 'append' }) diff --git a/packages/llm/llm-deepseek/README.md b/packages/llm/llm-deepseek/README.md index 5a9a1bfdbc..13e889847f 100644 --- a/packages/llm/llm-deepseek/README.md +++ b/packages/llm/llm-deepseek/README.md @@ -15,7 +15,7 @@ The package root exposes the Cordis plugin contract and `DeepSeekAdapter`; wire apiKey: !!js process.env.DEEPSEEK_API_KEY # or rely on the env fallback baseURL: !!js process.env.DEEPSEEK_BASE_URL # default: https://api.deepseek.com thinking: enabled # optional; provider default is enabled - reasoningEffort: high # optional; high | max — omitted ⇒ not sent + reasoningEffort: high # optional; high | max — omitted ⇒ high streamIdleTimeoutMs: 300000 # optional; positive finite Node timer delay; five-minute default defaultContextWindow: 256000 # optional positive-integer fallback for models without an exact value models: # optional; defaults to V4 Flash and V4 Pro @@ -30,9 +30,9 @@ The plugin registers the single provider route `deepseek`. A request selects it `contextWindow` is optional per configured model and is not exposed through the advisory catalog. `ctx.llm.resolveModelContext('deepseek', model)` returns an exact model value first, then `defaultContextWindow` for an entry without capacity or an unlisted pass-through id. When neither value exists it returns `undefined` without invalidating routing. Pressure-sensitive plugins therefore get deployment-owned capacity without treating the model selector as authoritative. Registering another adapter for `deepseek` throws `LlmError('DUPLICATE_ADAPTER')`. -`reasoningEffort` is **omitted by default** — when unset, the `reasoning_effort` wire field is not sent and the server applies its own default for the model. The only accepted values are `high` and `max` (DeepSeek's official effort levels). It is meaningful only with thinking enabled (the provider default). +`ctx.llm.resolveModelReasoning('deepseek', model)` returns the ordered `high` and `max` efforts for every pass-through model while thinking is enabled. `reasoningEffort` selects the deployment default and falls back to `high` when omitted. `agent/request` can replace it on each conversation step; the resolved value is logged in `request/header` and serialized as the official top-level `reasoning_effort` field. An unsupported value fails with `UNSUPPORTED_REASONING_EFFORT` before network I/O. -`thinking`/`reasoningEffort` are adapter-level request defaults serialized as the official top-level `thinking: {type}` / `reasoning_effort` wire fields. They live in adapter config (not `GenerateOptions`) to keep the core vocabulary provider-neutral. A request with `GenerateOptions.purpose: 'session-title'` forces thinking disabled and omits `reasoning_effort`, reserving its bounded output for visible title text without changing conversation or compaction defaults. +`thinking: disabled` removes the reasoning capability and omits `reasoning_effort`; combining it with a configured default fails plugin loading, and a per-request effort fails as unsupported. A request with `GenerateOptions.purpose: 'session-title'` also forces thinking disabled and omits the already-resolved effort, reserving its bounded output for visible title text without changing conversation or compaction defaults. `streamIdleTimeoutMs` bounds each outstanding provider read, including the initial `fetch`, without counting time the consumer spends between chunks. One stable abort signal reaches the request and body reader for the whole call; expiry stops the transport and throws `LlmError('TIMEOUT')`, while an earlier caller abort throws `LlmError('ABORTED')`. The adapter makes exactly one provider request per `stream()` call; agent-level retry is a separate plugin policy. @@ -79,7 +79,7 @@ Reasoning, text, and raw-string tool arguments are translated into harness chunk #### Token effect -Generated tokens follow provider thinking and effort settings plus the request's `maxTokens`; only loop-retained blocks affect later input. +Generated tokens follow the request's logged reasoning effort and `maxTokens`; only loop-retained blocks affect later input. #### KV Cache effect diff --git a/packages/llm/llm-deepseek/src/adapter.ts b/packages/llm/llm-deepseek/src/adapter.ts index 64faa3b725..d8ca534c89 100644 --- a/packages/llm/llm-deepseek/src/adapter.ts +++ b/packages/llm/llm-deepseek/src/adapter.ts @@ -5,11 +5,12 @@ * @module dsh-llm-deepseek/adapter */ -import { attributionHeaders, CONTEXT_WINDOW_EXCEEDED_CODE, isContextWindowExceededError, isQuotaExceededError, LlmAdapter, LlmError, ProviderRequestId, QUOTA_EXCEEDED_CODE } from '@deepseek-ai/dsh-llm' +import { attributionHeaders, CONTEXT_WINDOW_EXCEEDED_CODE, isContextWindowExceededError, isQuotaExceededError, LlmAdapter, LlmError, ProviderRequestId, QUOTA_EXCEEDED_CODE, ReasoningEffortId } from '@deepseek-ai/dsh-llm' import type { GenerateOptions, LlmModelContext, LlmModelInfo, + LlmModelReasoningInfo, LlmProviderInfo, StreamChunk, } from '@deepseek-ai/dsh-llm' @@ -51,6 +52,12 @@ export interface DeepSeekAdapterOptions { /** Default maximum idle interval while an adapter stream read is outstanding. */ export const DEFAULT_STREAM_IDLE_TIMEOUT_MS = 300_000 const STREAM_IDLE_TIMEOUT_CODE = 'LLM_STREAM_IDLE_TIMEOUT' +const HIGH_REASONING_EFFORT = ReasoningEffortId('high') +const MAX_REASONING_EFFORT = ReasoningEffortId('max') +const REASONING_EFFORTS = [ + { id: HIGH_REASONING_EFFORT, name: 'High' }, + { id: MAX_REASONING_EFFORT, name: 'Max' }, +] as const function providerRetryAfterMs(value: string | null): number | undefined { if (value === null) return undefined @@ -98,6 +105,9 @@ export class DeepSeekAdapter extends LlmAdapter { constructor(private readonly options: DeepSeekAdapterOptions) { super() + if (options.defaults?.thinking === 'disabled' && options.defaults.reasoningEffort !== undefined) { + throw new Error('llm-deepseek: reasoningEffort cannot be configured when thinking is disabled') + } if (options.defaultContextWindow !== undefined && (!Number.isInteger(options.defaultContextWindow) || options.defaultContextWindow <= 0)) { throw new Error('llm-deepseek: defaultContextWindow must be a positive integer') @@ -134,6 +144,19 @@ export class DeepSeekAdapter extends LlmAdapter { return Promise.resolve(contextWindow === undefined ? undefined : { contextWindow }) } + override resolveModelReasoning( + _provider: string, + _model: string, + ): Promise { + if (this.options.defaults?.thinking === 'disabled') return Promise.resolve(undefined) + return Promise.resolve({ + efforts: REASONING_EFFORTS, + defaultEffort: this.options.defaults?.reasoningEffort === 'max' + ? MAX_REASONING_EFFORT + : HIGH_REASONING_EFFORT, + }) + } + async * stream(options: GenerateOptions): AsyncIterable { const consumer = new AbortController() const upstream = options.signal === undefined diff --git a/packages/llm/llm-deepseek/src/index.ts b/packages/llm/llm-deepseek/src/index.ts index 66828fc954..1dad768440 100644 --- a/packages/llm/llm-deepseek/src/index.ts +++ b/packages/llm/llm-deepseek/src/index.ts @@ -28,8 +28,9 @@ const DEFAULT_MODELS: DeepSeekCatalogModel[] = [ /** * Plugin config, validated by the same-named schemastery schema. Every field * is optional in yml: credentials/endpoint fall back to the environment (a - * missing API key fails plugin load, not the first call), and omitted - * thinking fields send nothing on the wire, so the provider default applies. + * missing API key fails plugin load, not the first call), omitted thinking + * mode uses the provider default, and omitted reasoning effort resolves to + * `high`. */ export interface Config { /** API key; falls back to $DEEPSEEK_API_KEY. Required one way or the other. */ @@ -38,7 +39,7 @@ export interface Config { baseURL?: string /** Thinking-mode default for every request (provider default: enabled). */ thinking?: 'enabled' | 'disabled' - /** Thinking effort (only meaningful with thinking enabled). */ + /** Default thinking effort when thinking is enabled (default `high`). */ reasoningEffort?: 'high' | 'max' /** Positive context capacity used when the selected model has no exact value. */ defaultContextWindow?: number @@ -94,6 +95,9 @@ function resolveModels(models: readonly DeepSeekCatalogModel[] | undefined): Dee } export function apply(ctx: Context, config: Config): void { + if (config.thinking === 'disabled' && config.reasoningEffort !== undefined) { + throw new Error('llm-deepseek: reasoningEffort cannot be configured when thinking is disabled') + } const apiKey = config.apiKey ?? process.env.DEEPSEEK_API_KEY if (apiKey === undefined || apiKey.length === 0) { throw new Error('llm-deepseek: an API key is required (Config.apiKey or $DEEPSEEK_API_KEY)') diff --git a/packages/llm/llm-deepseek/src/serialize.ts b/packages/llm/llm-deepseek/src/serialize.ts index bf9515c942..70d51c8549 100644 --- a/packages/llm/llm-deepseek/src/serialize.ts +++ b/packages/llm/llm-deepseek/src/serialize.ts @@ -6,6 +6,7 @@ * @module dsh-llm-deepseek/serialize */ +import { LlmError } from '@deepseek-ai/dsh-llm' import type { ContentBlock, GenerateOptions, Message } from '@deepseek-ai/dsh-llm' import type { WireMessage, WireRequest, WireTool } from './types.ts' @@ -15,6 +16,17 @@ export interface RequestDefaults { reasoningEffort?: 'high' | 'max' | undefined } +/** Validate the adapter-owned effort before assigning its narrower wire type. */ +function reasoningEffort(options: GenerateOptions): 'high' | 'max' | undefined { + const effort = options.reasoningEffort + if (effort === undefined) return undefined + if (effort === 'high' || effort === 'max') return effort as 'high' | 'max' + throw new LlmError( + `DeepSeek does not support reasoning effort "${effort}"`, + 'UNSUPPORTED_REASONING_EFFORT', + ) +} + /** Join the text blocks of a message (used for user/tool-result content). */ function flattenText(blocks: ContentBlock[]): string { return blocks @@ -121,7 +133,9 @@ export function serializeRequest(options: GenerateOptions, defaults: RequestDefa // A short title budget must produce visible text; conversation and // compaction calls continue to inherit the adapter's thinking defaults. const thinking = options.purpose === 'session-title' ? 'disabled' : defaults.thinking - const reasoningEffort = options.purpose === 'session-title' ? undefined : defaults.reasoningEffort + const resolvedReasoningEffort = options.purpose === 'session-title' + ? undefined + : reasoningEffort(options) return { model: options.model, @@ -129,7 +143,7 @@ export function serializeRequest(options: GenerateOptions, defaults: RequestDefa stream: true, stream_options: { include_usage: true }, ...thinking !== undefined ? { thinking: { type: thinking } } : {}, - ...reasoningEffort !== undefined ? { reasoning_effort: reasoningEffort } : {}, + ...resolvedReasoningEffort !== undefined ? { reasoning_effort: resolvedReasoningEffort } : {}, ...tools !== undefined && tools.length > 0 ? { tools } : {}, ...options.temperature !== undefined ? { temperature: options.temperature } : {}, ...options.maxTokens !== undefined ? { max_tokens: options.maxTokens } : {}, diff --git a/packages/llm/llm-deepseek/tests/adapter.e2e.ts b/packages/llm/llm-deepseek/tests/adapter.e2e.ts index e02476eaef..8897d500d5 100644 --- a/packages/llm/llm-deepseek/tests/adapter.e2e.ts +++ b/packages/llm/llm-deepseek/tests/adapter.e2e.ts @@ -1,6 +1,6 @@ import { afterEach, describe, expect, it } from 'vitest' import { Context } from 'cordis' -import LlmService, { CallId } from '@deepseek-ai/dsh-llm' +import LlmService, { CallId, ReasoningEffortId } from '@deepseek-ai/dsh-llm' import type { Message, ToolSchema } from '@deepseek-ai/dsh-llm' import * as LlmDeepSeek from '@deepseek-ai/dsh-llm-deepseek' import type { Config } from '@deepseek-ai/dsh-llm-deepseek' @@ -80,11 +80,12 @@ describe.skipIf(!process.env.DEEPSEEK_API_KEY)('llm-deepseek e2e (real API)', () it.each(['high', 'max'] as const)( 'pro + thinking enabled (effort %s): tool-call round trip with reasoning passback', async (effort) => { - const ctx = await harness(PRO, { thinking: 'enabled', reasoningEffort: effort }) + const ctx = await harness(PRO, { thinking: 'enabled' }) // Turn 1: the model must call the tool (and think before it). const first = await assemble(ctx,{ model: PRO, + reasoningEffort: ReasoningEffortId(effort), messages: ask('What is the weather in Paris right now? Use the get_weather tool.'), tools: [weatherTool], maxTokens: 2000, @@ -99,6 +100,7 @@ describe.skipIf(!process.env.DEEPSEEK_API_KEY)('llm-deepseek e2e (real API)', () // block in history (the official thinking+tools passback rule). const second = await assemble(ctx,{ model: PRO, + reasoningEffort: ReasoningEffortId(effort), messages: [ ...ask('What is the weather in Paris right now? Use the get_weather tool.'), { role: 'assistant', content: first.message.content }, diff --git a/packages/llm/llm-deepseek/tests/adapter.spec.ts b/packages/llm/llm-deepseek/tests/adapter.spec.ts index 145017ea3d..7d3d8f9f08 100644 --- a/packages/llm/llm-deepseek/tests/adapter.spec.ts +++ b/packages/llm/llm-deepseek/tests/adapter.spec.ts @@ -8,6 +8,7 @@ import LlmService, { LlmError, ProviderRequestId, QUOTA_EXCEEDED_CODE, + ReasoningEffortId, userAgent, } from '@deepseek-ai/dsh-llm' import { MAX_TIMER_DELAY_MS } from '@deepseek-ai/dsh-timeout' @@ -120,6 +121,7 @@ describe('DeepSeekAdapter against a mock server', () => { // The wire request carried the auth header contents we configured. expect(server.requests[0]).toMatchObject({ model: 'deepseek-v4-flash', + reasoning_effort: 'high', stream: true, stream_options: { include_usage: true }, }) @@ -173,9 +175,35 @@ describe('DeepSeekAdapter against a mock server', () => { expect(server.headers[0]?.['x-deepseek-harness-compact']).toBe('1') }) - it('forwards thinking config onto the wire', async () => { + it('forwards the configured reasoning default and a dynamic request override', async () => { + const server = await mockServer([ + { kind: 'sse', events: textEvents }, + { kind: 'sse', events: textEvents }, + ]) + const ctx = await harness(server.url, { thinking: 'enabled', reasoningEffort: 'max' }) + + await assemble(ctx,{ + model: 'deepseek-v4-flash', + messages: [{ role: 'user', content: [{ type: 'text', text: 'hi' }] }], + }) + await assemble(ctx,{ + model: 'deepseek-v4-flash', + reasoningEffort: ReasoningEffortId('high'), + messages: [{ role: 'user', content: [{ type: 'text', text: 'hi again' }] }], + }) + expect(server.requests[0]).toMatchObject({ + thinking: { type: 'enabled' }, + reasoning_effort: 'max', + }) + expect(server.requests[1]).toMatchObject({ + thinking: { type: 'enabled' }, + reasoning_effort: 'high', + }) + }) + + it('omits reasoning capability and effort when thinking is disabled', async () => { const server = await mockServer([{ kind: 'sse', events: textEvents }]) - const ctx = await harness(server.url, { thinking: 'disabled', reasoningEffort: 'high' }) + const ctx = await harness(server.url, { thinking: 'disabled' }) await assemble(ctx,{ model: 'deepseek-v4-flash', @@ -183,8 +211,22 @@ describe('DeepSeekAdapter against a mock server', () => { }) expect(server.requests[0]).toMatchObject({ thinking: { type: 'disabled' }, - reasoning_effort: 'high', }) + expect(server.requests[0]).not.toHaveProperty('reasoning_effort') + await expect(ctx.llm.resolveModelReasoning('deepseek', 'deepseek-v4-flash')) + .resolves.toBeUndefined() + }) + + it('rejects a per-request effort before I/O when thinking is disabled', async () => { + const server = await mockServer([]) + const ctx = await harness(server.url, { thinking: 'disabled' }) + + await expect(assemble(ctx, { + model: 'deepseek-v4-flash', + reasoningEffort: ReasoningEffortId('high'), + messages: [{ role: 'user', content: [{ type: 'text', text: 'hi' }] }], + })).rejects.toMatchObject({ code: 'UNSUPPORTED_REASONING_EFFORT' }) + expect(server.requests).toHaveLength(0) }) it.each([ @@ -533,6 +575,34 @@ describe('plugin registration and config', () => { ]) await expect(ctx.llm.resolveModelContext('deepseek', 'deepseek-v4-flash')) .resolves.toEqual({ contextWindow: 128_000 }) + await expect(ctx.llm.resolveModelReasoning('deepseek', 'deepseek-v4-flash')) + .resolves.toEqual({ + efforts: [ + { id: ReasoningEffortId('high'), name: 'High' }, + { id: ReasoningEffortId('max'), name: 'Max' }, + ], + defaultEffort: ReasoningEffortId('high'), + }) + }) + + it('rejects a configured reasoning effort when thinking is disabled', async () => { + const ctx = new Context() + await ctx.plugin(LlmService) + await expect(ctx.plugin(LlmDeepSeek, { + apiKey: 'k', + baseURL: 'http://127.0.0.1:1', + thinking: 'disabled', + reasoningEffort: 'high', + })).rejects.toThrow(/reasoningEffort cannot be configured/) + expect(ctx.llm.listProviders()).toEqual([]) + }) + + it('rejects a disabled-thinking effort at the direct constructor boundary', () => { + expect(() => new DeepSeekAdapter({ + apiKey: 'k', + baseURL: 'http://127.0.0.1:1', + defaults: { thinking: 'disabled', reasoningEffort: 'high' }, + })).toThrow(/reasoningEffort cannot be configured/) }) it('uses the default model catalog when apply is called directly', async () => { diff --git a/packages/llm/llm-deepseek/tests/serialize.spec.ts b/packages/llm/llm-deepseek/tests/serialize.spec.ts index 566c71b7f2..0f294a5e3a 100644 --- a/packages/llm/llm-deepseek/tests/serialize.spec.ts +++ b/packages/llm/llm-deepseek/tests/serialize.spec.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from 'vitest' -import { CallId } from '@deepseek-ai/dsh-llm' +import { CallId, ReasoningEffortId } from '@deepseek-ai/dsh-llm' import type { ContentBlock, GenerateOptions, Message } from '@deepseek-ai/dsh-llm' import { serializeMessages, serializeRequest } from '../src/serialize.ts' @@ -174,15 +174,22 @@ describe('serializeRequest', () => { expect(wire.tools).toBeUndefined() }) - it('applies adapter defaults for thinking and effort', () => { - const wire = serializeRequest(request({ messages: history }), { thinking: 'enabled', reasoningEffort: 'max' }) + it('maps adapter-default thinking and the request reasoning effort', () => { + const wire = serializeRequest( + request({ messages: history, reasoningEffort: ReasoningEffortId('max') }), + { thinking: 'enabled', reasoningEffort: 'high' }, + ) expect(wire.thinking).toEqual({ type: 'enabled' }) expect(wire.reasoning_effort).toBe('max') }) it('disables thinking for session-title requests without changing adapter defaults', () => { const wire = serializeRequest( - request({ messages: history, purpose: 'session-title' }), + request({ + messages: history, + purpose: 'session-title', + reasoningEffort: ReasoningEffortId('max'), + }), { thinking: 'enabled', reasoningEffort: 'max' }, ) expect(wire.thinking).toEqual({ type: 'disabled' }) @@ -194,6 +201,13 @@ describe('serializeRequest', () => { expect(wire.thinking).toBeUndefined() expect(wire.reasoning_effort).toBeUndefined() }) + + it('rejects an effort outside the DeepSeek capability', () => { + expect(() => serializeRequest(request({ + messages: history, + reasoningEffort: ReasoningEffortId('medium'), + }))).toThrow(expect.objectContaining({ code: 'UNSUPPORTED_REASONING_EFFORT' })) + }) }) describe('review fixes: assistant content shapes', () => { diff --git a/packages/llm/llm-pi-ai/README.md b/packages/llm/llm-pi-ai/README.md index 8a6736f112..d32f06a9a1 100644 --- a/packages/llm/llm-pi-ai/README.md +++ b/packages/llm/llm-pi-ai/README.md @@ -30,6 +30,8 @@ Each provider name must exist in pi-ai's installed catalog and may appear only o The adapter exposes each configured provider's installed pi-ai models through `ctx.llm.listModels(provider)`. This is provider-neutral selector metadata derived from `getModels(provider)`; request-time resolution still performs the authoritative catalog lookup, so discovery does not create a second model registry. `ctx.llm.resolveModelContext(provider, model)` performs the same exact descriptor lookup and returns its context window, keeping capacity metadata on the route-owning adapter rather than a consuming plugin. +`ctx.llm.resolveModelReasoning(provider, model)` uses pi-ai's `getSupportedThinkingLevels(model)` and returns that model's ordered levels after filtering the separate `off` control. The Harness exposes the canonical pi-ai level as an opaque ID; provider/model wire spellings remain inside pi-ai's `thinkingLevelMap`. A non-reasoning model returns `undefined`. The profile `reasoning` value is the deployment default when configured; omitting it preserves the provider default. Per-request `GenerateOptions.reasoningEffort` takes precedence, and any explicit value absent from the exact model capability fails with `UNSUPPORTED_REASONING_EFFORT` before network I/O instead of being clamped. + Supported profile fields are `provider`, `apiKey`, `baseURL`, `headers`, `reasoning`, `thinkingBudgets`, `cacheRetention`, `transport`, `timeoutMs`, `websocketConnectTimeoutMs`, and `streamIdleTimeoutMs`. The stream-idle interval is a positive finite Node timer delay, defaults to five minutes, and covers only an outstanding provider read, not consumer think time. Harness app attribution wins a conflicting configured header name. The adapter forces pi-ai's SDK `maxRetries` to zero so one `stream()` call makes one provider request. The removed profile fields `maxRetries` and `maxRetryDelayMs` fail load instead of silently multiplying or hiding the separately composed agent-level retry budget. Idle expiry aborts the SDK's stable request signal and surfaces `TIMEOUT`; an earlier caller abort remains `ABORTED`. @@ -47,6 +49,7 @@ If a listener rewrites assembled assistant content, the loop drops replay state - pi-ai tool-call arguments are parsed objects; the harness stores raw JSON strings. The adapter parses input and re-stringifies output. - pi-ai reports failures as in-stream error events; these map to `finish {kind:'error'|'aborted', failure}` chunks. Provider-specific error text distinguishes terminal `QUOTA` from transient `RATE_LIMIT`, while text and usage signals evaluated against the resolved model's context window normalize overflow to `CONTEXT_WINDOW_EXCEEDED`. - pi-ai folds reasoning tokens into output usage; there is no separate reasoning count to map. +- pi-ai may internally support an `off` thinking level, but the Harness reasoning-effort capability deliberately excludes mode changes. - `GenerateOptions.stop` is rejected with `UNSUPPORTED_OPTION` because pi-ai's common streaming surface cannot guarantee it across providers. ## App attribution diff --git a/packages/llm/llm-pi-ai/src/adapter.ts b/packages/llm/llm-pi-ai/src/adapter.ts index ed0fb9fae4..1dde9d9dc7 100644 --- a/packages/llm/llm-pi-ai/src/adapter.ts +++ b/packages/llm/llm-pi-ai/src/adapter.ts @@ -7,13 +7,27 @@ import { streamSimple } from '@earendil-works/pi-ai/compat' import { getBuiltinModels } from '@earendil-works/pi-ai/providers/all' import type { BuiltinProvider } from '@earendil-works/pi-ai/providers/all' +import { getSupportedThinkingLevels } from '@earendil-works/pi-ai' import type { Api, Model, SimpleStreamOptions, + ThinkingLevel, } from '@earendil-works/pi-ai' -import { attributionHeaders, LlmAdapter, LlmError } from '@deepseek-ai/dsh-llm' -import type { GenerateOptions, LlmModelContext, LlmModelInfo, StreamChunk } from '@deepseek-ai/dsh-llm' +import { + attributionHeaders, + LlmAdapter, + LlmError, + ReasoningEffortId, +} from '@deepseek-ai/dsh-llm' +import type { + GenerateOptions, + LlmModelContext, + LlmModelInfo, + LlmModelReasoningInfo, + ReasoningEffortId as ReasoningEffortIdType, + StreamChunk, +} from '@deepseek-ai/dsh-llm' import { idleWatchdog, timeoutOf } from '@deepseek-ai/dsh-timeout' import { resolveProfiles } from './config.ts' import type { PiAiProviderProfile, ResolvedPiAiProviderProfile } from './config.ts' @@ -39,10 +53,13 @@ function resolveModel(profile: PiAiProviderProfile, modelId: string): Model } /** Copy profile stream knobs into pi-ai's common option vocabulary. */ -function profileOptions(profile: PiAiProviderProfile): SimpleStreamOptions { +function profileOptions( + profile: PiAiProviderProfile, + reasoning: ThinkingLevel | undefined, +): SimpleStreamOptions { return { ...profile.apiKey === undefined ? {} : { apiKey: profile.apiKey }, - ...profile.reasoning === undefined ? {} : { reasoning: profile.reasoning }, + ...reasoning === undefined ? {} : { reasoning }, ...profile.thinkingBudgets === undefined ? {} : { thinkingBudgets: profile.thinkingBudgets }, ...profile.cacheRetention === undefined ? {} : { cacheRetention: profile.cacheRetention }, ...profile.transport === undefined ? {} : { transport: profile.transport }, @@ -53,6 +70,25 @@ function profileOptions(profile: PiAiProviderProfile): SimpleStreamOptions { } } +/** Selectable pi-ai levels exclude the separate on/off control. */ +function supportedReasoningLevels(model: Model): ThinkingLevel[] { + return getSupportedThinkingLevels(model).filter((level): level is ThinkingLevel => level !== 'off') +} + +/** Validate an explicit Harness/profile effort without invoking pi-ai's clamp. */ +function resolveReasoningLevel( + model: Model, + effort: ReasoningEffortIdType | ThinkingLevel | undefined, +): ThinkingLevel | undefined { + if (effort === undefined) return undefined + const supported = supportedReasoningLevels(model) + if (supported.some(level => level === effort)) return effort as ThinkingLevel + throw new LlmError( + `pi-ai provider "${model.provider}" model "${model.id}" does not support reasoning effort "${effort}"`, + 'UNSUPPORTED_REASONING_EFFORT', + ) +} + /** Merge deployment headers while removing case-insensitive attribution collisions. */ function requestHeaders(headers: Readonly> | undefined): Record { const attribution = attributionHeaders() @@ -103,6 +139,37 @@ export class PiAiAdapter extends LlmAdapter { })) } + override resolveModelReasoning( + provider: string, + model: string, + ): Promise { + const profile = this.profiles.get(provider) + if (profile === undefined) { + return Promise.reject(new LlmError( + `pi-ai adapter does not own provider "${provider}"`, + 'NO_ADAPTER', + )) + } + return Promise.resolve().then(() => { + const resolvedModel = resolveModel(profile, model) + const levels = supportedReasoningLevels(resolvedModel) + if (levels.length === 0) { + resolveReasoningLevel(resolvedModel, profile.reasoning) + return undefined + } + const defaultLevel = resolveReasoningLevel(resolvedModel, profile.reasoning) + return { + efforts: levels.map(level => ({ + id: ReasoningEffortId(level), + name: `${level.charAt(0).toUpperCase()}${level.slice(1)}`, + })), + ...defaultLevel === undefined + ? {} + : { defaultEffort: ReasoningEffortId(defaultLevel) }, + } + }) + } + async * stream(options: GenerateOptions): AsyncIterable { if (options.stop !== undefined) { throw new LlmError('llm-pi-ai does not support GenerateOptions.stop', 'UNSUPPORTED_OPTION') @@ -112,6 +179,10 @@ export class PiAiAdapter extends LlmAdapter { throw new LlmError(`pi-ai adapter does not own provider "${options.provider}"`, 'NO_ADAPTER') } const model = resolveModel(profile, options.model) + const reasoning = resolveReasoningLevel( + model, + options.reasoningEffort ?? profile.reasoning, + ) const consumer = new AbortController() const upstream = options.signal === undefined @@ -122,7 +193,7 @@ export class PiAiAdapter extends LlmAdapter { try { const events = streamSimple(model, toPiContext(options), { - ...profileOptions(profile), + ...profileOptions(profile, reasoning), ...options.temperature === undefined ? {} : { temperature: options.temperature }, ...options.maxTokens === undefined ? {} : { maxTokens: options.maxTokens }, ...options.sessionId === undefined ? {} : { sessionId: String(options.sessionId) }, diff --git a/packages/llm/llm-pi-ai/tests/adapter.e2e.ts b/packages/llm/llm-pi-ai/tests/adapter.e2e.ts index 77d2cc81dd..e91a951d85 100644 --- a/packages/llm/llm-pi-ai/tests/adapter.e2e.ts +++ b/packages/llm/llm-pi-ai/tests/adapter.e2e.ts @@ -1,6 +1,6 @@ import { afterEach, describe, expect, it } from 'vitest' import { Context } from 'cordis' -import LlmService, { CallId } from '@deepseek-ai/dsh-llm' +import LlmService, { CallId, ReasoningEffortId } from '@deepseek-ai/dsh-llm' import type { Message, ToolSchema } from '@deepseek-ai/dsh-llm' import * as LlmPiAi from '@deepseek-ai/dsh-llm-pi-ai' import type { PiAiProviderProfile } from '@deepseek-ai/dsh-llm-pi-ai' @@ -9,7 +9,7 @@ import { assemble, type AssembledResult } from './assemble.ts' /** * Real-API e2e for the pi-ai-backed adapter: V4 Flash + V4 Pro with provider - * defaults and representative high/xhigh reasoning. Mirrors the native + * defaults and representative high/max reasoning. Mirrors the native * adapter's StreamChunk contract and exercises a replayed tool follow-up. * Key-gated. */ @@ -75,9 +75,10 @@ describe.skipIf(!process.env.DEEPSEEK_API_KEY)('llm-pi-ai e2e (real API)', () => }) it.each([FLASH, PRO])('%s + reasoning high: reasoning blocks present', async (model) => { - const ctx = await harness(model, { reasoning: 'high' }) + const ctx = await harness(model) const result = await assemble(ctx,{ model, + reasoningEffort: ReasoningEffortId('high'), messages: ask('Which is larger, 9.11 or 9.8? Answer with just the number.'), maxTokens: 2000, }) @@ -86,11 +87,12 @@ describe.skipIf(!process.env.DEEPSEEK_API_KEY)('llm-pi-ai e2e (real API)', () => expect(textOf(result)).toContain('9.8') }) - it('pro + reasoning xhigh (wire max): tool-call round trip', async () => { - const ctx = await harness(PRO, { reasoning: 'xhigh' }) + it('pro + reasoning max: tool-call round trip', async () => { + const ctx = await harness(PRO) const first = await assemble(ctx,{ model: PRO, + reasoningEffort: ReasoningEffortId('max'), messages: ask('What is the weather in Paris right now? Use the get_weather tool.'), tools: [weatherTool], maxTokens: 2000, @@ -103,6 +105,7 @@ describe.skipIf(!process.env.DEEPSEEK_API_KEY)('llm-pi-ai e2e (real API)', () => const second = await assemble(ctx,{ model: PRO, + reasoningEffort: ReasoningEffortId('max'), messages: [ ...ask('What is the weather in Paris right now? Use the get_weather tool.'), first.message, diff --git a/packages/llm/llm-pi-ai/tests/adapter.spec.ts b/packages/llm/llm-pi-ai/tests/adapter.spec.ts index f37b07f624..0bf8d34da7 100644 --- a/packages/llm/llm-pi-ai/tests/adapter.spec.ts +++ b/packages/llm/llm-pi-ai/tests/adapter.spec.ts @@ -2,7 +2,7 @@ import { createServer } from 'node:http' import type { IncomingMessage, Server, ServerResponse } from 'node:http' import { afterEach, describe, expect, it, vi } from 'vitest' import { Context } from 'cordis' -import LlmService, { CONTEXT_WINDOW_EXCEEDED_CODE, LlmError, userAgent } from '@deepseek-ai/dsh-llm' +import LlmService, { CONTEXT_WINDOW_EXCEEDED_CODE, LlmError, ReasoningEffortId, userAgent } from '@deepseek-ai/dsh-llm' import * as LlmPiAi from '@deepseek-ai/dsh-llm-pi-ai' import { PiAiAdapter } from '@deepseek-ai/dsh-llm-pi-ai' import { MAX_TIMER_DELAY_MS } from '@deepseek-ai/dsh-timeout' @@ -124,7 +124,7 @@ describe('PiAiAdapter provider routing', () => { it('forwards common stream options and profile reasoning', async () => { const server = await mockServer([{ events: textEvents }]) const ctx = await harness(server.url, { - reasoning: 'xhigh', + reasoning: 'max', cacheRetention: 'none', transport: 'sse', timeoutMs: 5000, @@ -148,6 +148,25 @@ describe('PiAiAdapter provider routing', () => { }) }) + it('uses a dynamic request effort and rejects unsupported efforts before network I/O', async () => { + const server = await mockServer([{ events: textEvents }]) + const ctx = await harness(server.url, { reasoning: 'max' }) + + await assemble(ctx, { + model: 'deepseek-v4-flash', + reasoningEffort: ReasoningEffortId('high'), + messages: [], + }) + expect(server.requests[0]).toMatchObject({ reasoning_effort: 'high' }) + + await expect(assemble(ctx, { + model: 'deepseek-v4-flash', + reasoningEffort: ReasoningEffortId('xhigh'), + messages: [], + })).rejects.toMatchObject({ code: 'UNSUPPORTED_REASONING_EFFORT' }) + expect(server.requests).toHaveLength(1) + }) + it('preserves omitted profile options when constructing the adapter directly', async () => { const server = await mockServer([{ events: textEvents }]) const ctx = new Context() @@ -327,6 +346,51 @@ describe('provider profile lifecycle', () => { expect(typeof context?.contextWindow).toBe('number') }) + it('exposes model-specific reasoning levels without off or an invented provider default', async () => { + const ctx = new Context() + await ctx.plugin(LlmService) + await ctx.plugin(LlmPiAi, { + providers: [{ provider: 'deepseek' }, { provider: 'openai' }], + }) + + await expect(ctx.llm.resolveModelReasoning('deepseek', 'deepseek-v4-flash')) + .resolves.toEqual({ + efforts: [ + { id: ReasoningEffortId('high'), name: 'High' }, + { id: ReasoningEffortId('max'), name: 'Max' }, + ], + }) + const extended = await ctx.llm.resolveModelReasoning('openai', 'gpt-5.6-sol') + expect(extended?.efforts.map(effort => effort.id)).toEqual([ + ReasoningEffortId('minimal'), + ReasoningEffortId('low'), + ReasoningEffortId('medium'), + ReasoningEffortId('high'), + ReasoningEffortId('xhigh'), + ReasoningEffortId('max'), + ]) + await expect(ctx.llm.resolveModelReasoning('openai', 'gpt-4.1')) + .resolves.toBeUndefined() + }) + + it('uses a supported profile reasoning value as the model default and rejects an unsupported one', async () => { + const supported = new Context() + await supported.plugin(LlmService) + await supported.plugin(LlmPiAi, { + providers: [{ provider: 'deepseek', reasoning: 'max' }], + }) + await expect(supported.llm.resolveModelReasoning('deepseek', 'deepseek-v4-flash')) + .resolves.toMatchObject({ defaultEffort: ReasoningEffortId('max') }) + + const unsupported = new Context() + await unsupported.plugin(LlmService) + await unsupported.plugin(LlmPiAi, { + providers: [{ provider: 'deepseek', reasoning: 'medium' }], + }) + await expect(unsupported.llm.resolveModelReasoning('deepseek', 'deepseek-v4-flash')) + .rejects.toMatchObject({ code: 'UNSUPPORTED_REASONING_EFFORT' }) + }) + it('accepts absent credentials for pi-ai ambient authentication', async () => { vi.stubEnv('DEEPSEEK_API_KEY', 'ambient-key') const server = await mockServer([{ events: textEvents }]) @@ -378,6 +442,8 @@ describe('provider profile lifecycle', () => { await expect(adapter.listModels('anthropic')).rejects.toMatchObject({ code: 'NO_ADAPTER' }) await expect(adapter.resolveModelContext('anthropic', 'claude-sonnet-4')) .rejects.toMatchObject({ code: 'NO_ADAPTER' }) + await expect(adapter.resolveModelReasoning('anthropic', 'claude-sonnet-4')) + .rejects.toMatchObject({ code: 'NO_ADAPTER' }) await expect(adapter.resolveModelContext('openai', 'not-a-catalog-model')) .rejects.toMatchObject({ code: 'UNKNOWN_MODEL' }) await expect((async () => { diff --git a/packages/llm/llm/README.md b/packages/llm/llm/README.md index beac6d5e8e..d15ce7f0ac 100644 --- a/packages/llm/llm/README.md +++ b/packages/llm/llm/README.md @@ -12,6 +12,8 @@ An adapter registry plus a single streaming call surface, interceptable via a wa - `ctx.llm.listProviders(): LlmProviderInfo[]` Describe registered provider routes in registration order. - `ctx.llm.listModels(provider: string): Promise` Discover the models one registered provider currently advertises. - `ctx.llm.resolveModelContext(provider: string, model: string): Promise` Resolve authoritative context capacity for one exact route from its owning adapter. +- `ctx.llm.resolveModelReasoning(provider: string, model: string): Promise` Resolve ordered adapter-owned reasoning efforts and an optional deployment default for one exact route. +- `ctx.llm.resolveCallConfig(config: LlmCallConfig): Promise` Validate an explicit effort and materialize an adapter-configured default without clamping. - `ctx.llm.stream(options: GenerateOptions): AsyncIterable` Stream one model call as raw chunks (token-level deltas). Consumers assemble the chunks into blocks/messages with `BlockAssembler`. `LlmService` preserves errors from final adapter selection, synchronous dispatch, iterator construction, and iteration, and binds their provenance to the exact stream handle returned for that model call. `isLlmAdapterFailure(stream, value)` reports only errors from that call's final adapter boundary; `llmFailureOf(stream, value)` returns the adjacent immutable `LlmFailure`. Nested model calls, `llm/stream` middleware, and downstream consumer failures remain unclassified for the outer call. Classification never replaces or mutates the adapter's original coded `Error`. @@ -20,6 +22,8 @@ Provider and model metadata is a discovery surface, not a routing whitelist. `re Context capacity is a separate correctness query, not a catalog decoration or global LLM setting. `resolveModelContext()` asks the adapter that owns the exact provider/model route; an adapter can describe an unlisted dynamic model, and `undefined` means only that capacity is unavailable. Invalid returned capacity fails with `INVALID_MODEL_CONTEXT`. +Reasoning effort is also an exact-route capability, but its identifiers are opaque adapter-owned strings rather than a core enum. `resolveModelReasoning()` validates and detaches the ordered display metadata; `undefined` means the model has no selectable effort. `resolveCallConfig()` accepts only an exact advertised identifier, materializes `defaultEffort` when present, and otherwise preserves the provider default. Invalid capability metadata fails with `INVALID_MODEL_REASONING`; an unsupported explicit or configured effort fails with `UNSUPPORTED_REASONING_EFFORT` before provider I/O. + ### Events | Event | Mode | Purpose | @@ -28,7 +32,7 @@ Context capacity is a separate correctness query, not a catalog decoration or gl ### Extension points -- Subclass `LlmAdapter` and call `ctx.llm.registerAdapter(providers, adapter)` to add one or more provider routes. `GenerateOptions.provider` selects the adapter; `GenerateOptions.model` is adapter-owned and may be resolved dynamically. Override `providerInfo()` and asynchronous `listModels()` to expose selector metadata, and `resolveModelContext()` when exact capacity is known; the defaults use the route id as its name, advertise no models, and return no capacity. +- Subclass `LlmAdapter` and call `ctx.llm.registerAdapter(providers, adapter)` to add one or more provider routes. `GenerateOptions.provider` selects the adapter; `GenerateOptions.model` is adapter-owned and may be resolved dynamically. Override `providerInfo()` and asynchronous `listModels()` to expose selector metadata, `resolveModelContext()` when exact capacity is known, and `resolveModelReasoning()` when a model exposes selectable efforts; the defaults use the route id as its name, advertise no models, and return neither capacity nor reasoning metadata. - Wrap `llm/stream` via `ctx.on()` waterfall listeners for caching, logging, or routing. A wrapper that retries after emitting a chunk has no durable attempt boundary; shipped agent retry policy therefore uses `agent/request-error` instead. ### Content-block vocabulary (`types.ts`) @@ -39,7 +43,7 @@ Streaming is a raw chunk protocol (`block-start`, `text-delta`, `reasoning-delta ### Call configuration (`call-config.ts`) -`LlmCallConfig` is the provider + model + sampling scalars of one conversation's requests (`provider`, `model`, `temperature`, `maxTokens`, `stop` — each mapping 1:1 onto the same-named `GenerateOptions` field). It is per-conversation state recorded in the session log as part of the request header (see the dsh-session `request/header` events), never a silently-adjustable per-call knob: the `agent/request` waterfall proposes a replacement and the loop logs a real change. `callConfigEquals(a, b)` is the field-wise real-change detector; `deepFreeze(value)` is the ownership helper the loop applies to every built request before dispatch (`llm/stream` listeners and adapters read, never rewrite). `markAgentLoopRequest()` gives that exact object process-local loop provenance, and `isAgentLoopRequest()` lets observers distinguish it from independently logged auxiliary calls that may also be frozen and session-associated. `GenerateOptions.purpose` classifies logged auxiliary compaction and session-title calls so adapters can apply purpose-specific transport policy without changing ordinary conversation requests. +`LlmCallConfig` is the provider, model, optional adapter-owned reasoning effort, and sampling scalars of one conversation's requests (`provider`, `model`, `reasoningEffort`, `temperature`, `maxTokens`, `stop` — each mapping 1:1 onto the same-named `GenerateOptions` field). It is per-conversation state recorded in the session log as part of the request header (see the dsh-session `request/header` events), never a silently-adjustable per-call knob: the `agent/request` waterfall proposes a replacement, `resolveCallConfig()` validates and defaults it, and the loop logs the effective value before dispatch. `callConfigEquals(a, b)` is the field-wise real-change detector; `deepFreeze(value)` is the ownership helper the loop applies to every built request before dispatch (`llm/stream` listeners and adapters read, never rewrite). `markAgentLoopRequest()` gives that exact object process-local loop provenance, and `isAgentLoopRequest()` lets observers distinguish it from independently logged auxiliary calls that may also be frozen and session-associated. `GenerateOptions.purpose` classifies logged auxiliary compaction and session-title calls so adapters can apply purpose-specific transport policy without changing ordinary conversation requests. ### App attribution (`attribution.ts`) @@ -61,7 +65,7 @@ Two adapters implement `LlmAdapter` on different internals: [`@deepseek-ai/dsh-l ## Model Experience -None, as this adapter registry forwards an already assembled request without adding or changing any model-bound text, schema, or message. +None, as the service adds no model-bound text, schema, or message; it only materializes and logs an adapter-configured reasoning effort. #### KV Cache effect diff --git a/packages/llm/llm/src/brand.ts b/packages/llm/llm/src/brand.ts index 259dc49bce..0c0190a325 100644 --- a/packages/llm/llm/src/brand.ts +++ b/packages/llm/llm/src/brand.ts @@ -38,3 +38,15 @@ export type ProviderRequestId = Branded<'ProviderRequestId'> export function ProviderRequestId(id: string): ProviderRequestId { return id as ProviderRequestId } + +/** Adapter-owned identifier for one model's selectable reasoning effort. */ +export type ReasoningEffortId = Branded<'ReasoningEffortId'> + +/** + * Brand an adapter-owned reasoning-effort identifier. + * @param id - the opaque identifier exposed by one model capability. + * @returns the same string, branded; no validation is performed. + */ +export function ReasoningEffortId(id: string): ReasoningEffortId { + return id as ReasoningEffortId +} diff --git a/packages/llm/llm/src/call-config.ts b/packages/llm/llm/src/call-config.ts index fd6ecf9df4..c5247af143 100644 --- a/packages/llm/llm/src/call-config.ts +++ b/packages/llm/llm/src/call-config.ts @@ -1,24 +1,27 @@ /** * Conversation call configuration and freeze utilities. Provider routing, - * model, and sampling values are request-header state that can affect cache - * reuse; request waterfalls replace them and the loop logs changed snapshots - * instead of allowing silent per-call drift. + * model, reasoning effort, and sampling values are request-header state that + * can affect cache reuse; request waterfalls replace them and the loop logs + * changed snapshots instead of allowing silent per-call drift. * @module dsh-llm/call-config */ import type { GenerateOptions } from './types.ts' +import type { ReasoningEffortId } from './brand.ts' /** Process-local identities of request objects assembled by dsh-agent-loop. */ const AGENT_LOOP_REQUESTS = new WeakSet() /** - * Provider + model + sampling scalars of one conversation's requests. Every field maps - * 1:1 onto the same-named `GenerateOptions` field; the loop builds requests - * from the logged header rather than accepting these per call. + * Provider, model, reasoning effort, and sampling scalars of one conversation's + * requests. Every field maps 1:1 onto the same-named `GenerateOptions` field; + * the loop builds requests from the logged header rather than accepting these + * per call. */ export interface LlmCallConfig { provider: string model: string + reasoningEffort?: ReasoningEffortId temperature?: number maxTokens?: number stop?: string[] @@ -33,7 +36,13 @@ export interface LlmCallConfig { * @returns whether every field (including the `stop` list, element-wise) matches. */ export function callConfigEquals(a: LlmCallConfig, b: LlmCallConfig): boolean { - if (a.provider !== b.provider || a.model !== b.model || a.temperature !== b.temperature || a.maxTokens !== b.maxTokens) return false + if ( + a.provider !== b.provider + || a.model !== b.model + || a.reasoningEffort !== b.reasoningEffort + || a.temperature !== b.temperature + || a.maxTokens !== b.maxTokens + ) return false if (a.stop === undefined || b.stop === undefined) return a.stop === b.stop return a.stop.length === b.stop.length && a.stop.every((s, i) => s === b.stop?.[i]) } diff --git a/packages/llm/llm/src/index.ts b/packages/llm/llm/src/index.ts index 6765833e8c..e56c2d17a6 100644 --- a/packages/llm/llm/src/index.ts +++ b/packages/llm/llm/src/index.ts @@ -12,12 +12,14 @@ import type { LlmFailure, LlmModelContext, LlmModelInfo, + LlmModelReasoningInfo, LlmProviderInfo, Message, StreamChunk, } from './types.ts' import type { ProviderRequestId } from './brand.ts' -import { deepFreeze } from './call-config.ts' +import { callConfigEquals, deepFreeze } from './call-config.ts' +import type { LlmCallConfig } from './call-config.ts' import { HarnessError } from './error.ts' import { bindAdapterFailureScope, markLlmAdapterFailure } from './adapter-failure.ts' import type { AdapterFailureScope } from './adapter-failure.ts' @@ -144,6 +146,20 @@ export abstract class LlmAdapter { return Promise.resolve(undefined) } + /** + * Resolve selectable reasoning efforts for one exact model. Absence means + * the model has no selectable reasoning-effort capability. + * @param _provider - one provider route owned by this adapter. + * @param _model - exact model id passed to {@link GenerateOptions.model}. + * @returns adapter-owned effort metadata, or `undefined` when unsupported. + */ + resolveModelReasoning( + _provider: string, + _model: string, + ): Promise { + return Promise.resolve(undefined) + } + /** * Stream one model call as raw chunks. The only required method. * @param options - the fully-assembled request; implementations must honor `options.signal`. @@ -262,6 +278,90 @@ export class LlmService extends Service { return { contextWindow: context.contextWindow } } + /** + * Resolve selectable reasoning efforts from the adapter that owns one exact + * route. Metadata is validated and detached; an absent result means an + * effort selector is unsupported for that model. + * @param provider - registered provider route to inspect. + * @param model - exact model id passed to the adapter. + * @returns detached reasoning metadata, or `undefined` when unsupported. + */ + async resolveModelReasoning( + provider: string, + model: string, + ): Promise { + const reasoning = await this.registration(provider).adapter.resolveModelReasoning(provider, model) + if (reasoning === undefined) return undefined + if (reasoning.efforts.length === 0) { + throw new LlmError( + `adapter returned invalid reasoning metadata for provider "${provider}" model "${model}"`, + 'INVALID_MODEL_REASONING', + ) + } + const seen = new Set() + const efforts = reasoning.efforts.map((effort) => { + if ( + typeof effort.id !== 'string' + || effort.id.length === 0 + || typeof effort.name !== 'string' + || effort.name.length === 0 + || (effort.description !== undefined && typeof effort.description !== 'string') + || seen.has(effort.id) + ) { + throw new LlmError( + `adapter returned invalid or duplicate reasoning effort metadata for provider "${provider}" model "${model}"`, + 'INVALID_MODEL_REASONING', + ) + } + seen.add(effort.id) + return { + id: effort.id, + name: effort.name, + ...effort.description === undefined ? {} : { description: effort.description }, + } + }) + if (reasoning.defaultEffort !== undefined && !seen.has(reasoning.defaultEffort)) { + throw new LlmError( + `adapter returned an unknown default reasoning effort for provider "${provider}" model "${model}"`, + 'INVALID_MODEL_REASONING', + ) + } + return { + efforts, + ...reasoning.defaultEffort === undefined ? {} : { defaultEffort: reasoning.defaultEffort }, + } + } + + /** + * Validate a conversation call config against its exact model capability and + * materialize an adapter-configured default. Unsupported explicit efforts + * reject before provider I/O; no clamping or aliasing is performed. + * @param config - provider/model route and optional request controls. + * @returns a detached config only when a default must be materialized. + */ + async resolveCallConfig(config: LlmCallConfig): Promise { + const reasoning = await this.resolveModelReasoning(config.provider, config.model) + const requested = config.reasoningEffort + if (reasoning === undefined) { + if (requested !== undefined) { + throw new LlmError( + `provider "${config.provider}" model "${config.model}" does not support reasoning effort "${requested}"`, + 'UNSUPPORTED_REASONING_EFFORT', + ) + } + return config + } + const effective = requested ?? reasoning.defaultEffort + if (effective === undefined) return config + if (!reasoning.efforts.some(effort => effort.id === effective)) { + throw new LlmError( + `provider "${config.provider}" model "${config.model}" does not support reasoning effort "${effective}"`, + 'UNSUPPORTED_REASONING_EFFORT', + ) + } + return requested === effective ? config : { ...config, reasoningEffort: effective } + } + private registration(provider: string): { adapter: LlmAdapter; provider: LlmProviderInfo } { const registration = this.adapters.get(provider) if (!registration) throw new LlmError(`no adapter registered for provider "${provider}"`, 'NO_ADAPTER') @@ -298,8 +398,14 @@ export class LlmService extends Service { ): AsyncGenerator { let iterator: AsyncIterator try { - const adapter = this.registration(options.provider).adapter - const stream = adapter.stream(this.forAdapter(options, adapter)) + const resolvedConfig = await this.resolveCallConfig(options) + const resolvedOptions = callConfigEquals(options, resolvedConfig) + ? options + : Object.isFrozen(options) + ? deepFreeze({ ...options, ...resolvedConfig }) + : { ...options, ...resolvedConfig } + const adapter = this.registration(resolvedOptions.provider).adapter + const stream = adapter.stream(this.forAdapter(resolvedOptions, adapter)) iterator = stream[Symbol.asyncIterator]() } catch (error: unknown) { throw markLlmAdapterFailure(failures, error) diff --git a/packages/llm/llm/src/types.ts b/packages/llm/llm/src/types.ts index f9f97d532a..7882985288 100644 --- a/packages/llm/llm/src/types.ts +++ b/packages/llm/llm/src/types.ts @@ -5,7 +5,7 @@ */ import type { Branded } from '@deepseek-ai/dsh-brand' -import type { CallId, ProviderRequestId } from './brand.ts' +import type { CallId, ProviderRequestId, ReasoningEffortId } from './brand.ts' /** Serializable provider-boundary facts; policy decides whether they are retryable. */ export interface LlmFailure { @@ -161,6 +161,27 @@ export interface LlmModelContext { contextWindow: number } +/** Display metadata for one adapter-owned reasoning effort. */ +export interface LlmReasoningEffortInfo { + /** Opaque stable value accepted by {@link GenerateOptions.reasoningEffort}. */ + id: ReasoningEffortId + /** Human-readable effort name for selectors and diagnostics. */ + name: string + /** Optional user-facing distinction from otherwise similar efforts. */ + description?: string +} + +/** Selectable reasoning efforts for one exact provider/model route. */ +export interface LlmModelReasoningInfo { + /** Supported efforts in adapter-preferred display order. */ + efforts: readonly LlmReasoningEffortInfo[] + /** + * Adapter-configured default materialized into requests when callers omit + * an effort. Absence preserves the provider's own default. + */ + defaultEffort?: ReasoningEffortId +} + /** * Raw streaming protocol emitted by adapters. * Block indexes correlate interleaved deltas, and `block-end` carries the @@ -201,6 +222,8 @@ export interface GenerateOptions { /** Registered provider route selecting the adapter instance. */ provider: string model: string + /** Adapter-owned reasoning effort selected for this exact model. */ + reasoningEffort?: ReasoningEffortId /** * Ordered conversation messages, exactly as the provider sees them (after * the `system` slot). A loop-built request assembles them as diff --git a/packages/llm/llm/tests/call-config.spec.ts b/packages/llm/llm/tests/call-config.spec.ts index a5426e815f..2237d4639f 100644 --- a/packages/llm/llm/tests/call-config.spec.ts +++ b/packages/llm/llm/tests/call-config.spec.ts @@ -6,6 +6,7 @@ import { describe, expect, it } from 'vitest' import { callConfigEquals, deepFreeze, isAgentLoopRequest, markAgentLoopRequest } from '../src/call-config.ts' +import { ReasoningEffortId } from '../src/brand.ts' import type { GenerateOptions } from '../src/types.ts' describe('callConfigEquals', () => { @@ -14,6 +15,11 @@ describe('callConfigEquals', () => { expect(callConfigEquals(base, base)).toBe(true) expect(callConfigEquals(base, { provider: 'x', model: 'm' })).toBe(false) expect(callConfigEquals(base, { provider: 'p', model: 'x' })).toBe(false) + expect(callConfigEquals({ ...base, reasoningEffort: ReasoningEffortId('high') }, base)).toBe(false) + expect(callConfigEquals( + { ...base, reasoningEffort: ReasoningEffortId('high') }, + { ...base, reasoningEffort: ReasoningEffortId('high') }, + )).toBe(true) expect(callConfigEquals({ ...base, temperature: 0.5 }, base)).toBe(false) expect(callConfigEquals({ ...base, maxTokens: 1 }, { ...base, maxTokens: 2 })).toBe(false) expect(callConfigEquals({ ...base, stop: ['a'] }, base)).toBe(false) diff --git a/packages/llm/llm/tests/service.spec.ts b/packages/llm/llm/tests/service.spec.ts index 90be1ffcb0..59684ac8b8 100644 --- a/packages/llm/llm/tests/service.spec.ts +++ b/packages/llm/llm/tests/service.spec.ts @@ -11,9 +11,15 @@ import LlmService, { LlmError, llmFailureOf, ProviderRequestId, + ReasoningEffortId, StreamChunk, } from '@deepseek-ai/dsh-llm' -import type { LlmModelContext, LlmModelInfo, LlmProviderInfo } from '@deepseek-ai/dsh-llm' +import type { + LlmModelContext, + LlmModelInfo, + LlmModelReasoningInfo, + LlmProviderInfo, +} from '@deepseek-ai/dsh-llm' class ScriptedAdapter extends LlmAdapter { constructor(private script: StreamChunk[]) { @@ -49,6 +55,7 @@ class CatalogAdapter extends ScriptedAdapter { private readonly provider: LlmProviderInfo, private readonly models: readonly LlmModelInfo[], private readonly contexts: Readonly> = {}, + private readonly reasoning: Readonly> = {}, ) { super(SCRIPT) } @@ -67,6 +74,13 @@ class CatalogAdapter extends ScriptedAdapter { ): Promise { return Promise.resolve(this.contexts[model]) } + + override resolveModelReasoning( + _provider: string, + model: string, + ): Promise { + return Promise.resolve(this.reasoning[model]) + } } const SCRIPT: StreamChunk[] = [ @@ -674,6 +688,117 @@ describe('LlmService', () => { await expect(ctx.llm.resolveModelContext('route', 'other')).resolves.toBeUndefined() }) + it('resolves detached adapter-owned reasoning metadata and materializes its default', async () => { + const ctx = new Context() + await ctx.plugin(LlmService) + const source = { + efforts: [ + { id: ReasoningEffortId('standard'), name: 'Standard' }, + { id: ReasoningEffortId('ultra'), name: 'Ultra', description: 'Largest budget' }, + ], + defaultEffort: ReasoningEffortId('standard'), + } + ctx.llm.registerAdapter(['route'], new CatalogAdapter( + { id: 'route', name: 'Route' }, + [], + {}, + { model: source }, + )) + + const resolved = await ctx.llm.resolveModelReasoning('route', 'model') + expect(resolved).toEqual(source) + source.efforts[0]!.name = 'mutated' + expect(resolved?.efforts[0]?.name).toBe('Standard') + await expect(ctx.llm.resolveCallConfig({ provider: 'route', model: 'model' })).resolves.toEqual({ + provider: 'route', + model: 'model', + reasoningEffort: ReasoningEffortId('standard'), + }) + const explicit = { provider: 'route', model: 'model', reasoningEffort: ReasoningEffortId('ultra') } + await expect(ctx.llm.resolveCallConfig(explicit)).resolves.toBe(explicit) + }) + + it.each([ + [{ efforts: [] }, 'empty effort list'], + [{ efforts: [{ id: '', name: 'Empty' }] }, 'empty id'], + [{ efforts: [{ id: 'valid', name: '' }] }, 'empty name'], + [{ efforts: [{ id: 'valid', name: 'Valid', description: 1 }] }, 'non-string description'], + [{ efforts: [{ id: 'same', name: 'One' }, { id: 'same', name: 'Two' }] }, 'duplicate id'], + [{ efforts: [{ id: 'valid', name: 'Valid' }], defaultEffort: 'other' }, 'unknown default'], + ] as const)('rejects invalid model reasoning metadata (%s: %s)', async (metadata, _label) => { + const ctx = new Context() + await ctx.plugin(LlmService) + ctx.llm.registerAdapter(['route'], new CatalogAdapter( + { id: 'route', name: 'Route' }, + [], + {}, + { model: metadata as unknown as LlmModelReasoningInfo }, + )) + await expect(ctx.llm.resolveModelReasoning('route', 'model')) + .rejects.toMatchObject({ code: 'INVALID_MODEL_REASONING' }) + }) + + it('rejects unsupported reasoning efforts without clamping', async () => { + const ctx = new Context() + await ctx.plugin(LlmService) + ctx.llm.registerAdapter(['route'], new CatalogAdapter( + { id: 'route', name: 'Route' }, + [], + {}, + { model: { efforts: [{ id: ReasoningEffortId('ultra'), name: 'Ultra' }] } }, + )) + + await expect(ctx.llm.resolveCallConfig({ + provider: 'route', + model: 'model', + reasoningEffort: ReasoningEffortId('standard'), + })).rejects.toMatchObject({ code: 'UNSUPPORTED_REASONING_EFFORT' }) + await expect(ctx.llm.resolveCallConfig({ + provider: 'route', + model: 'plain', + reasoningEffort: ReasoningEffortId('standard'), + })).rejects.toMatchObject({ code: 'UNSUPPORTED_REASONING_EFFORT' }) + }) + + it('resolves reasoning defaults at the final adapter boundary after routing middleware', async () => { + const ctx = new Context() + await ctx.plugin(LlmService) + const adapter = new class extends RecordingAdapter { + override resolveModelReasoning( + _provider: string, + _model: string, + ): Promise { + return Promise.resolve({ + efforts: [{ id: ReasoningEffortId('standard'), name: 'Standard' }], + defaultEffort: ReasoningEffortId('standard'), + }) + } + }(SCRIPT) + ctx.llm.registerAdapter(['routed'], adapter) + const disposeRouting = ctx.on('llm/stream', (options, next) => { + options.provider = 'routed' + return next() + }) + + for await (const _chunk of ctx.llm.stream({ + provider: 'initial', + model: 'model', + messages: [], + })) { /* drain */ } + + expect(adapter.lastOptions?.reasoningEffort).toBe(ReasoningEffortId('standard')) + disposeRouting() + + const frozenRequest: GenerateOptions = Object.freeze({ + provider: 'routed', + model: 'model', + messages: [], + }) + for await (const _chunk of ctx.llm.stream(frozenRequest)) { /* drain */ } + expect(adapter.lastOptions?.reasoningEffort).toBe(ReasoningEffortId('standard')) + expect(Object.isFrozen(adapter.lastOptions)).toBe(true) + }) + it.each([0, -1, 1.5, Number.NaN])( 'rejects invalid adapter model context %s', async (contextWindow) => { diff --git a/scripts/gen-cordis-catalog.ts b/scripts/gen-cordis-catalog.ts index 8318d59996..88f7ebcf4c 100644 --- a/scripts/gen-cordis-catalog.ts +++ b/scripts/gen-cordis-catalog.ts @@ -38,6 +38,7 @@ export const LINK_MAP: Record = { HookContext: 'core.md', LlmCallConfig: 'core.md', LlmModelContext: 'core.md', + LlmModelReasoningInfo: 'core.md', LlmFailure: 'llm-streaming.md', LlmModelInfo: 'core.md', LlmProviderInfo: 'core.md', diff --git a/scripts/type-equiv.manifest.json b/scripts/type-equiv.manifest.json index 89a0778827..f94dba52c4 100644 --- a/scripts/type-equiv.manifest.json +++ b/scripts/type-equiv.manifest.json @@ -46,6 +46,21 @@ "symbol": "LlmModelContext", "source": "packages/llm/llm/src/types.ts" }, + { + "doc": "docs/core-data-structures/core.md", + "symbol": "ReasoningEffortId", + "source": "packages/llm/llm/src/brand.ts" + }, + { + "doc": "docs/core-data-structures/core.md", + "symbol": "LlmReasoningEffortInfo", + "source": "packages/llm/llm/src/types.ts" + }, + { + "doc": "docs/core-data-structures/core.md", + "symbol": "LlmModelReasoningInfo", + "source": "packages/llm/llm/src/types.ts" + }, { "doc": "docs/core-data-structures/core.md", "symbol": "GenerateOptions", From 1bfca8612835e9c246ebc1c3c2a7e180ed3ac0fd Mon Sep 17 00:00:00 2001 From: Yichen Jiang Date: Sat, 25 Jul 2026 08:32:38 +0800 Subject: [PATCH 2/8] feat(tui): select model reasoning effort --- ...cated-full-screen-tui-front-door.i18n.yaml | 4 +- ...17-dedicated-full-screen-tui-front-door.md | 6 +- ...dedicated-full-screen-tui-front-door.zh.md | 6 +- docs/config-catalog.md | 2 +- docs/cordis-catalog/services.md | 2 +- examples/tui-agent/README.md | 2 +- .../tests/fixtures/tui-scripted-llm.ts | 32 +++- .../tui-agent/tests/tui-keyless-smoke.e2e.ts | 3 +- packages/core/agent/README.md | 2 +- packages/core/agent/src/llm-target.ts | 25 ++- packages/core/agent/tests/llm-target.spec.ts | 23 ++- packages/ui/tui/README.md | 10 +- packages/ui/tui/src/index.ts | 154 ++++++++++++++--- packages/ui/tui/tests/harness.ts | 15 +- .../model-effort-switching.expected.txt | 42 +++++ .../snapshots/model-selector.expected.txt | 38 ++--- .../snapshots/model-switching.expected.txt | 12 +- .../status-diagnostics-narrow.expected.txt | 6 +- .../snapshots/status-diagnostics.expected.txt | 62 +++---- packages/ui/tui/tests/tui.snapshot.ts | 24 ++- packages/ui/tui/tests/tui.spec.ts | 159 +++++++++++++++++- 21 files changed, 497 insertions(+), 132 deletions(-) create mode 100644 packages/ui/tui/tests/snapshots/model-effort-switching.expected.txt diff --git a/.agents/notes/implemented/feature/2026-07-17-dedicated-full-screen-tui-front-door.i18n.yaml b/.agents/notes/implemented/feature/2026-07-17-dedicated-full-screen-tui-front-door.i18n.yaml index 8d6c7be831..56b2e3536d 100644 --- a/.agents/notes/implemented/feature/2026-07-17-dedicated-full-screen-tui-front-door.i18n.yaml +++ b/.agents/notes/implemented/feature/2026-07-17-dedicated-full-screen-tui-front-door.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write -2026-07-17-dedicated-full-screen-tui-front-door.md: ecfda138593fc2b98ac42929acc586b11e437ee2 -2026-07-17-dedicated-full-screen-tui-front-door.zh.md: 6b8cc63f7657672a6da542e2033d765b54bd4f07 +2026-07-17-dedicated-full-screen-tui-front-door.md: 5ab38d2f9dc97d22ab35b42b04e6da68210690eb +2026-07-17-dedicated-full-screen-tui-front-door.zh.md: 8aee779d10370efd1bef832dbcaebdc25b34a595 diff --git a/.agents/notes/implemented/feature/2026-07-17-dedicated-full-screen-tui-front-door.md b/.agents/notes/implemented/feature/2026-07-17-dedicated-full-screen-tui-front-door.md index ecfda13859..5ab38d2f9d 100644 --- a/.agents/notes/implemented/feature/2026-07-17-dedicated-full-screen-tui-front-door.md +++ b/.agents/notes/implemented/feature/2026-07-17-dedicated-full-screen-tui-front-door.md @@ -22,9 +22,9 @@ The selected front door receives the exact generated or resumed `SessionId` used The TUI rebuilds the transcript from the active `session.surface` and reprojects it whenever an event carries a `surfaceOp`, so resumed and compacted history matches the model-visible conversation. It renders Markdown text and reasoning, token totals, the latest `todo/write` plan, and tool cards produced through each tool definition's `presentCall` and `presentResult` methods. Long card bodies retain a configurable head/tail preview with the hidden-line count; one terminal control expands or collapses every card. Pending chunks and tool calls update the same components that completed events settle. -Editor input calls `agent.send()` while idle and `agent.steer()` while a turn is running. Cancellation, reasoning visibility, tool-card expansion, redraw, transcript clearing, and exit are terminal-only controls. The idle footer derives context occupancy from `tokenMeter` and shows the selected model; during a run, elapsed activity and the Escape interrupt hint replace that summary. `/status` remains available in either state and appends a detailed terminal-only snapshot: session identity and timestamps, selected model and reasoning visibility, lifecycle counts folded from the event log, the same deduplicated usage buckets and KV-cache rate as the footer, and context use from `tokenMeter` plus the selected model's advertised capacity. The plugin registers the shared `userInteraction` provider and presents queued questions in a wide bottom-left keyboard panel with batch progress, numbered options, and aligned descriptions; agent behavior and answer logging remain owned by their existing services. +Editor input calls `agent.send()` while idle and `agent.steer()` while a turn is running. Cancellation, reasoning visibility, tool-card expansion, redraw, transcript clearing, and exit are terminal-only controls. The idle footer derives context occupancy from `tokenMeter` and shows the selected model and explicit reasoning effort; during a run, elapsed activity and the Escape interrupt hint replace that summary. `/status` remains available in either state and appends a detailed terminal-only snapshot: session identity and timestamps, selected model, reasoning effort/default state and reasoning visibility, lifecycle counts folded from the event log, the same deduplicated usage buckets and KV-cache rate as the footer, and context use from `tokenMeter` plus the selected model's advertised capacity. The plugin registers the shared `userInteraction` provider and presents queued questions in a wide bottom-left keyboard panel with batch progress, numbered options, and aligned descriptions; agent behavior and answer logging remain owned by their existing services. -The `/model` command presents the advisory `ctx.llm` catalog as a keyboard selector and changes only this TUI session's target; argument forms remain available for direct selection. Agent-scoped prompt-assembly and request waterfalls snapshot one provider/model pair per step, so `{{provider}}` / `{{model}}` interpolation and request routing cannot split when a command arrives during assembly. The latest logged request header restores a used target; a selection that never reaches a request remains process-local. +The `/model` command presents the advisory `ctx.llm` catalog as a keyboard selector and changes only this TUI session's target; argument forms remain available for direct selection. Each model row owns the adapter-advertised reasoning-effort order and default: Shift+Tab cycles that row's efforts, while a model without selectable metadata keeps default behavior. Agent-scoped prompt-assembly and request waterfalls snapshot one provider/model/reasoning-effort target per step, so `{{provider}}` / `{{model}}` interpolation and request routing cannot split when a command arrives during assembly. The latest logged request header restores a used target; a selection that never reaches a request remains process-local. ### Terminal ownership @@ -49,4 +49,4 @@ The implemented [TUI terminal-state snapshot Agent Note](../testing/2026-07-18-t - The TUI carries a pi-tui dependency and a strict TTY requirement; non-TTY deployments use the Headless app or a structured protocol. - Session projection makes resume and compaction consistent with the durable conversation, but one configured session owns the transcript and editor. - Tool packages extend terminal cards through their existing presentation methods without adding tool-specific branches to the TUI. -- Model selection uses adapter-advertised metadata without turning catalog membership into request validation; unused selections are not durable state. +- Model and reasoning-effort selection use adapter-advertised metadata without turning catalog membership into request validation; unused selections are not durable state. diff --git a/.agents/notes/implemented/feature/2026-07-17-dedicated-full-screen-tui-front-door.zh.md b/.agents/notes/implemented/feature/2026-07-17-dedicated-full-screen-tui-front-door.zh.md index 6b8cc63f76..8aee779d10 100644 --- a/.agents/notes/implemented/feature/2026-07-17-dedicated-full-screen-tui-front-door.zh.md +++ b/.agents/notes/implemented/feature/2026-07-17-dedicated-full-screen-tui-front-door.zh.md @@ -22,9 +22,9 @@ DeepSeek Harness 将 [`@deepseek-ai/dsh-tui`](../../../../packages/ui/tui/README TUI 从活跃的 `session.surface` 重建 transcript(文本记录),并在事件携带 `surfaceOp` 时重新投影,因此恢复或压缩后的历史与模型可见会话保持一致。TUI 渲染 Markdown 文本与推理、token 用量、最新 `todo/write` 计划,以及各工具定义通过 `presentCall` 和 `presentResult` 方法生成的工具卡片。较长的工具卡片正文会保留可配置的头尾预览,并显示隐藏行数;一个终端控制可以展开或收起全部卡片。进行中的分片与工具调用会更新同一组组件,随后由完成事件收束状态。 -agent 空闲时,编辑器输入调用 `agent.send()`;轮次运行中则调用 `agent.steer()`。取消、推理显隐、工具卡片展开、重绘、清空 transcript 和退出都只是终端控制。空闲态页脚根据 `tokenMeter` 得出上下文占用率,并显示所选模型;agent 运行期间,该摘要会替换为带已用时长的活动指示和 Escape 中断提示。`/status` 在这两种状态下均可用,并会追加一份仅在终端显示的详细快照,其中包括会话标识与时间戳、所选模型及推理显隐状态、从事件日志归并得出的生命周期计数、与页脚一致的去重用量分项和 KV 缓存命中率,以及 `tokenMeter` 给出的上下文用量和所选模型公布的容量。插件注册共享的 `userInteraction` 提供方,在左下角宽幅键盘操作面板中呈现排队的问题,面板显示批次进度、带编号的选项和对齐的描述;agent 行为和答案日志仍由既有服务负责。 +agent 空闲时,编辑器输入调用 `agent.send()`;轮次运行中则调用 `agent.steer()`。取消、推理显隐、工具卡片展开、重绘、清空 transcript 和退出都只是终端控制。空闲态页脚根据 `tokenMeter` 得出上下文占用率,并显示所选模型和显式选定的推理强度;agent 运行期间,该摘要会替换为带已用时长的活动指示和 Escape 中断提示。`/status` 在这两种状态下均可用,并会追加一份仅在终端显示的详细快照,其中包括会话标识与时间戳、所选模型、推理强度(或默认状态)及推理显隐状态、从事件日志归并得出的生命周期计数、与页脚一致的去重用量分项和 KV 缓存命中率,以及 `tokenMeter` 给出的上下文用量和所选模型公布的容量。插件注册共享的 `userInteraction` 提供方,在左下角宽幅键盘操作面板中呈现排队的问题,面板显示批次进度、带编号的选项和对齐的描述;agent 行为和答案日志仍由既有服务负责。 -`/model` 命令将建议性的 `ctx.llm` 目录呈现为键盘选择器,并且只更改当前 TUI 会话的目标;带参数的形式仍可直接选择目标。agent 作用域内的 prompt 组装和请求两条 waterfall(瀑布式事件)会为每个 step 快照一次同一个提供方/模型字段组合,因此即使命令在组装期间到达,`{{provider}}` / `{{model}}` 插值与请求路由也不会分裂。系统通过日志中最新的请求头恢复已经使用过的目标;未被请求使用的选择只保留在当前进程中。 +`/model` 命令将建议性的 `ctx.llm` 目录呈现为键盘选择器,并且只更改当前 TUI 会话的目标;带参数的形式仍可直接选择目标。每个模型行都持有适配器公布的推理强度顺序和默认值:按 Shift+Tab 可循环切换该行的推理强度;没有可选元数据的模型保持默认行为。agent 作用域内的 prompt 组装和请求两条 waterfall(瀑布式事件)会为每个步骤快照一次同一个提供方/模型/推理强度目标,因此即使命令在组装期间到达,`{{provider}}` / `{{model}}` 插值与请求路由也不会分裂。系统通过日志中最新的请求头恢复已经使用过的目标;未被请求使用的选择只保留在当前进程中。 ### 终端所有权 @@ -49,4 +49,4 @@ agent 空闲时,编辑器输入调用 `agent.send()`;轮次运行中则调 - TUI 会引入 pi-tui 依赖并严格要求 TTY;非 TTY 部署使用 Headless app 或结构化协议。 - 会话投影使恢复和压缩与持久会话保持一致,但只有一个已配置会话拥有 transcript 和编辑器。 - 工具包通过既有呈现方法扩展终端卡片,无需在 TUI 中增加工具专用分支。 -- 模型选择使用适配器提供的目录元数据,但不会把目录成员关系变成请求校验;未使用的选择不属于持久化状态。 +- 模型和推理强度选择使用适配器公布的元数据,但不会把目录成员关系变成请求校验;未使用的选择不属于持久化状态。 diff --git a/docs/config-catalog.md b/docs/config-catalog.md index 0ad683d2f4..90da953d66 100644 --- a/docs/config-catalog.md +++ b/docs/config-catalog.md @@ -1632,7 +1632,7 @@ export interface TuiConfig { } ``` -Source: [`packages/ui/tui/src/index.ts:270`](../packages/ui/tui/src/index.ts) +Source: [`packages/ui/tui/src/index.ts:273`](../packages/ui/tui/src/index.ts) ## `@deepseek-ai/dsh-tui-demo` diff --git a/docs/cordis-catalog/services.md b/docs/cordis-catalog/services.md index 537365a477..b4ba1a6698 100644 --- a/docs/cordis-catalog/services.md +++ b/docs/cordis-catalog/services.md @@ -1728,7 +1728,7 @@ The concrete provider retains pi-tui, focus, and terminal lifecycle state. Plugi abstract openOverlay(request: TuiOverlayRequest): TuiOverlaySession ``` -Source: [`packages/ui/tui/src/index.ts:150`](../../packages/ui/tui/src/index.ts) +Source: [`packages/ui/tui/src/index.ts:153`](../../packages/ui/tui/src/index.ts) ## `ctx.userInteraction` — `UserInteractionService` diff --git a/examples/tui-agent/README.md b/examples/tui-agent/README.md index 2e87df0a27..106f1412a2 100644 --- a/examples/tui-agent/README.md +++ b/examples/tui-agent/README.md @@ -17,7 +17,7 @@ Type a coding task. The agent works through the `read`/`write`/`edit` filesystem The `todo_write` task tracker is opt-in and not in the shipped config: add `@deepseek-ai/dsh-tool-todo` to `cordis.yml` (or a personal-config overlay under `~/.dsh`) to expose it. Once loaded, the model records a whole-list plan to the session log and the TUI renders it. -The TUI renders Markdown history, reasoning, tool-owned terminal/diff/generic cards, token totals, and — when `todo_write` is loaded — the latest plan. Long tool bodies keep a head/tail preview; Ctrl+O expands or collapses every card. Enter submits or steers while the agent runs, Ctrl+R toggles reasoning, Escape cancels, and `/help` lists commands. `/plan` selects plan mode for the next step; `/plan ` also submits the message into that step, while `/plan off` selects the default mode without model input. `/status` expands the current session's identity, activity counts, exact token/cache buckets, context use, and timestamps without interrupting a running turn. `/model` opens a keyboard selector for the current provider catalog; use Up/Down and Enter, or `/model ` and `/model /` for direct selection. `ask_user_question` opens a wide bottom-left keyboard panel with batch progress and numbered options. +The TUI renders Markdown history, reasoning, tool-owned terminal/diff/generic cards, token totals, and — when `todo_write` is loaded — the latest plan. Long tool bodies keep a head/tail preview; Ctrl+O expands or collapses every card. Enter submits or steers while the agent runs, Ctrl+R toggles reasoning, Escape cancels, and `/help` lists commands. `/plan` selects plan mode for the next step; `/plan ` also submits the message into that step, while `/plan off` selects the default mode without model input. `/status` expands the current session's identity, activity counts, exact token/cache buckets, context use, and timestamps without interrupting a running turn. `/model` opens a keyboard selector for the current provider catalog; use Up/Down to focus a model, Shift+Tab to cycle its advertised reasoning efforts, and Enter to select, or use `/model ` and `/model /` for direct selection. `ask_user_question` opens a wide bottom-left keyboard panel with batch progress and numbered options. ### Resuming a prior session diff --git a/examples/tui-agent/tests/fixtures/tui-scripted-llm.ts b/examples/tui-agent/tests/fixtures/tui-scripted-llm.ts index c3bdbfd1b1..e668689ca9 100644 --- a/examples/tui-agent/tests/fixtures/tui-scripted-llm.ts +++ b/examples/tui-agent/tests/fixtures/tui-scripted-llm.ts @@ -1,6 +1,12 @@ import type { Context } from 'cordis' -import type { GenerateOptions, LlmModelContext, LlmModelInfo, StreamChunk } from '@deepseek-ai/dsh-llm' -import { CallId, LlmAdapter } from '@deepseek-ai/dsh-llm' +import type { + GenerateOptions, + LlmModelContext, + LlmModelInfo, + LlmModelReasoningInfo, + StreamChunk, +} from '@deepseek-ai/dsh-llm' +import { CallId, LlmAdapter, ReasoningEffortId } from '@deepseek-ai/dsh-llm' const CONTROL_PROBE = '\u001b]2;MODEL_CONTROLLED\u0007\u001b[999CMODEL_CURSOR\u009b31mMODEL_C1' const INITIAL_TEXT = `I need one decision before I continue. ${CONTROL_PROBE}` @@ -39,6 +45,20 @@ class ScriptedTuiAdapter extends LlmAdapter { return Promise.resolve({ contextWindow: 128_000 }) } + override resolveModelReasoning( + _provider: string, + model: string, + ): Promise { + if (model !== 'tui-scripted-model-pro') return Promise.resolve(undefined) + return Promise.resolve({ + efforts: [ + { id: ReasoningEffortId('high'), name: 'High' }, + { id: ReasoningEffortId('max'), name: 'Max' }, + ], + defaultEffort: ReasoningEffortId('high'), + }) + } + override async * stream(options: GenerateOptions): AsyncIterable { // The session-title provider's auxiliary request carries no tool schemas, // unlike every agent turn; answer it with a fixed title so the PTY test can @@ -47,8 +67,12 @@ class ScriptedTuiAdapter extends LlmAdapter { for (const chunk of textChunks(TITLE_TEXT)) yield chunk return } - if (options.model !== 'tui-scripted-model-pro' || !options.system?.includes('tui-scripted-model-pro')) { - throw new Error('the scripted TUI request did not apply the selected model to routing and prompt variables') + if ( + options.model !== 'tui-scripted-model-pro' + || !options.system?.includes('tui-scripted-model-pro') + || options.reasoningEffort !== ReasoningEffortId('max') + ) { + throw new Error('the scripted TUI request did not apply the selected model and reasoning effort') } const lastMessage = options.messages.at(-1) const lastText = (lastMessage?.content ?? []) diff --git a/examples/tui-agent/tests/tui-keyless-smoke.e2e.ts b/examples/tui-agent/tests/tui-keyless-smoke.e2e.ts index 43ac327d8b..951cce98c5 100644 --- a/examples/tui-agent/tests/tui-keyless-smoke.e2e.ts +++ b/examples/tui-agent/tests/tui-keyless-smoke.e2e.ts @@ -104,7 +104,7 @@ function smoke(overrides: Partial & { label: string }): Prom // other route (see fixtures/tui-scripted-llm.ts). const SELECT_PRO_MODEL = [ { waitFor: 'scripted TUI ready.', send: '/model\r' }, - { waitFor: 'Select model', send: '\x1b[B\r' }, + { waitFor: 'Select model', send: '\x1b[B\x1b[Z\r' }, ] as const describe('tui-agent keyless smoke (real Loader tree in a PTY)', () => { @@ -157,6 +157,7 @@ describe('tui-agent keyless smoke (real Loader tree in a PTY)', () => { ], }) expect(output).toContain('I need one decision before I continue.') + expect(output).toContain('Reasoning effort: Max.') expect(output).toContain('Entering plan mode (applies from the next step). Use /plan off to leave.') expect(output).toContain('Leaving plan mode (applies from the next step).') expect(output).toContain('Default mode confirmed.') diff --git a/packages/core/agent/README.md b/packages/core/agent/README.md index 511ce32235..75a16039cd 100644 --- a/packages/core/agent/README.md +++ b/packages/core/agent/README.md @@ -10,7 +10,7 @@ Tracks live agents and carries the initiating Agent through asynchronous driver ### Public API -The scoped-registration surface: `Agent.ctx` is the agent's scope context (`dsh-scope`, key = the agent) — register tools/sections/variables/listeners through it for that agent alone, all unwound on disposal. `agentEvents(ctx, agent)` is the fused dispatcher for ordinary agent-subject operations (carrier + injected subject in one move); its notification mode invokes every listener and contains both synchronous throws and returned-promise rejections. The registry lifecycle pair reuses one stable routing carrier. `assembleContextFor(agent)` builds the per-agent assembly context (`agent` + `scope` together). `installAgentLlmTarget(agentCtx, target)` snapshots a mutable provider/model selection during prompt assembly and applies that pair to both prompt variables and request routing for one step. `CreateAgentOptions.setup(agentCtx)` and `ResumeAgentOptions.setup(agentCtx)` compose a fresh or resumed agent's scoped world while both objects remain unpublished. Setup is trusted, composition-only same-process code: drive the agent only after creation resolves. +The scoped-registration surface: `Agent.ctx` is the agent's scope context (`dsh-scope`, key = the agent) — register tools/sections/variables/listeners through it for that agent alone, all unwound on disposal. `agentEvents(ctx, agent)` is the fused dispatcher for ordinary agent-subject operations (carrier + injected subject in one move); its notification mode invokes every listener and contains both synchronous throws and returned-promise rejections. The registry lifecycle pair reuses one stable routing carrier. `assembleContextFor(agent)` builds the per-agent assembly context (`agent` + `scope` together). `installAgentLlmTarget(agentCtx, target)` snapshots a mutable provider/model/reasoning-effort selection during prompt assembly, applies the route to prompt variables, and applies the complete target to request routing for one step; an absent selected effort clears an inherited effort so the target uses adapter/provider defaults. `CreateAgentOptions.setup(agentCtx)` and `ResumeAgentOptions.setup(agentCtx)` compose a fresh or resumed agent's scoped world while both objects remain unpublished. Setup is trusted, composition-only same-process code: drive the agent only after creation resolves. - `ctx.agents.register(agent: Agent): () => void` — record an **already-constructed** agent. Disposed with the calling fiber. - Advanced ordered lifecycle: `enter(agent, owner): () => void` enforces `agent.id === agent.session.id`, performs the authoritative ID collision check, and inserts without announcing; `owner` explicitly records the live creator-agent relation (or `undefined` for a root), independently of durable session lineage. `announce(agent)` emits `agent/created` exactly once. A detach requested synchronously by a creation listener is deferred until that dispatch unwinds, and every detach checks the captured entry object, so a stale capability cannot delete a later same-ID replacement. The async factory uses this split; ordinary plugins use `register()`. diff --git a/packages/core/agent/src/llm-target.ts b/packages/core/agent/src/llm-target.ts index 18287a3ff5..e0492a71d2 100644 --- a/packages/core/agent/src/llm-target.ts +++ b/packages/core/agent/src/llm-target.ts @@ -1,17 +1,19 @@ /** - * Agent-scoped provider/model target snapshot shared by interactive front doors. + * Agent-scoped LLM target snapshot shared by interactive front doors. * @module @deepseek-ai/dsh-agent/llm-target */ import type { Context } from 'cordis' -import type { LlmCallConfig } from '@deepseek-ai/dsh-llm' +import type { LlmCallConfig, ReasoningEffortId } from '@deepseek-ai/dsh-llm' -/** Complete provider/model route selected for one live agent. */ +/** Complete provider/model route and optional reasoning effort selected for one live agent. */ export interface AgentLlmTarget { /** Registered provider route. */ provider: string /** Provider-owned model id. */ model: string + /** Adapter-owned reasoning effort, or provider/default behavior when absent. */ + reasoningEffort?: ReasoningEffortId } /** Mutable selection plus the target captured for the current step. */ @@ -24,9 +26,11 @@ export interface AgentLlmTargetRef { /** * Couple one mutable target to agent-scoped prompt assembly and request routing. - * Prompt assembly snapshots the selected pair before delegating, then applies - * both prompt variables and request config to that snapshot so a concurrent - * switch takes effect on a later step instead of splitting the two surfaces. + * Prompt assembly snapshots the selected target before delegating, then applies + * its route to prompt variables and its route/effort to request config so a + * concurrent switch takes effect on a later step instead of splitting the two + * surfaces. An absent selected effort clears any inherited effort so a model + * switch can restore that target's provider/default behavior. * * @param agentCtx - The target agent's scoped context. * @param target - Mutable selection owned by the calling front door. @@ -52,10 +56,15 @@ export function installAgentLlmTarget(agentCtx: Context, target: AgentLlmTargetR async (_agent, _turn, _step, _config, _signal, next): Promise => { const resolved = await next() const selected = target.assembled - return selected === undefined ? resolved : { - ...resolved, + if (selected === undefined) return resolved + const { reasoningEffort: _inheritedEffort, ...withoutInheritedEffort } = resolved + return { + ...withoutInheritedEffort, provider: selected.provider, model: selected.model, + ...selected.reasoningEffort === undefined + ? {} + : { reasoningEffort: selected.reasoningEffort }, } }, ) diff --git a/packages/core/agent/tests/llm-target.spec.ts b/packages/core/agent/tests/llm-target.spec.ts index fa4ef2b459..10d688d6ed 100644 --- a/packages/core/agent/tests/llm-target.spec.ts +++ b/packages/core/agent/tests/llm-target.spec.ts @@ -7,7 +7,7 @@ import { type Agent, type AgentLlmTargetRef, } from '../src/index.ts' -import type { LlmCallConfig } from '@deepseek-ai/dsh-llm' +import { ReasoningEffortId, type LlmCallConfig } from '@deepseek-ai/dsh-llm' describe('installAgentLlmTarget()', () => { it('snapshots prompt variables and request routing together, then disposes both listeners', async () => { @@ -24,16 +24,31 @@ describe('installAgentLlmTarget()', () => { 'agent/request', 1, 0, seed, signal, () => Promise.resolve(seed), )).resolves.toBe(seed) - target.current = { provider: 'alpha', model: 'a1' } + target.current = { + provider: 'alpha', + model: 'a1', + reasoningEffort: ReasoningEffortId('high'), + } expect((await ctx.systemPrompt.assemble()).variables).toMatchObject({ provider: 'alpha', model: 'a1' }) target.current = { provider: 'beta', model: 'b1' } await expect(agentEvents(ctx, agent).waterfall( 'agent/request', 1, 0, seed, signal, () => Promise.resolve(seed), - )).resolves.toEqual({ provider: 'alpha', model: 'a1', temperature: 0.2 }) + )).resolves.toEqual({ + provider: 'alpha', + model: 'a1', + reasoningEffort: ReasoningEffortId('high'), + temperature: 0.2, + }) expect((await ctx.systemPrompt.assemble()).variables).toMatchObject({ provider: 'beta', model: 'b1' }) + const inherited: LlmCallConfig = { + provider: 'alpha', + model: 'a1', + reasoningEffort: ReasoningEffortId('max'), + temperature: 0.2, + } await expect(agentEvents(ctx, agent).waterfall( - 'agent/request', 1, 1, seed, signal, () => Promise.resolve(seed), + 'agent/request', 1, 1, inherited, signal, () => Promise.resolve(inherited), )).resolves.toEqual({ provider: 'beta', model: 'b1', temperature: 0.2 }) dispose() diff --git a/packages/ui/tui/README.md b/packages/ui/tui/README.md index 616544779f..3326c548eb 100644 --- a/packages/ui/tui/README.md +++ b/packages/ui/tui/README.md @@ -10,7 +10,7 @@ This package owns interactive terminal presentation and input only. It injects ` After terminal startup succeeds, the package provides the terminal-local `ctx.tui` extension service. A plugin that injects it can call `openOverlay()` with a component factory and constrained layout options; the host exposes the viewport, semantic theme, display-text escaping, redraw, close, and a lifetime signal, but not the pi-tui tree, terminal, focus controller, or overlay handle. Plugin overlays, the model selector, and user questions share one FIFO modal queue. Each request is an effect of the calling plugin fiber, so unload removes queued work or closes visible work before cleanup settles; terminal shutdown unloads dependents before stopping pi-tui. Overlay state is not logged or replayed. Component code is trusted and may render ANSI styling, but must pass untrusted text through `host.display()`. The [interactive-extension Agent Note](../../../.agents/notes/implemented/architecture/2026-07-22-tui-interactive-extension-service.md) owns the boundary and rejected alternatives. -The TUI rebuilds resumed history from the active session surface, renders Markdown responses and reasoning, applies each tool's `presentCall` / `presentResult` intent to terminal, diff, or generic cards, keeps the latest `todo/write` plan above the editor, and presents `ctx.userInteraction` questions in a wide bottom-left keyboard panel with progress, numbered options, and aligned descriptions. The latest logged session title becomes the header subtitle, with `welcome` before a title exists, and the terminal window title becomes ``. A durable `llm/retry` event retracts the failed step's live chunks and renders the scheduled retry count, delay, and failure in the transcript; success, exhaustion, and cancellation then settle through ordinary session events. The footer totals each logged model step's usage once, including failed attempts, while treating committed-message usage as a fallback for logs without a usage chunk. Its idle view compares token-meter pressure with `ctx.llm.resolveModelContext()` for the current route, displays `context unknown` when the adapter has no capacity metadata, and also shows tool-card mode and the current model with reasoning state; while the agent runs, an elapsed working indicator and `esc interrupt` replace that summary. Surface replacement events rebuild the transcript so compacted history does not reappear. +The TUI rebuilds resumed history from the active session surface, renders Markdown responses and reasoning, applies each tool's `presentCall` / `presentResult` intent to terminal, diff, or generic cards, keeps the latest `todo/write` plan above the editor, and presents `ctx.userInteraction` questions in a wide bottom-left keyboard panel with progress, numbered options, and aligned descriptions. The latest logged session title becomes the header subtitle, with `welcome` before a title exists, and the terminal window title becomes ``. A durable `llm/retry` event retracts the failed step's live chunks and renders the scheduled retry count, delay, and failure in the transcript; success, exhaustion, and cancellation then settle through ordinary session events. The footer totals each logged model step's usage once, including failed attempts, while treating committed-message usage as a fallback for logs without a usage chunk. Its idle view compares token-meter pressure with `ctx.llm.resolveModelContext()` for the current route, displays `context unknown` when the adapter has no capacity metadata, and also shows tool-card mode plus the current model and any explicitly selected reasoning effort; while the agent runs, an elapsed working indicator and `esc interrupt` replace that summary. Surface replacement events rebuild the transcript so compacted history does not reappear. An embedding may provide `TuiRuntime.formatCwd` when its logical workspace label differs from the session's host directory. The override changes only the footer label; tools continue to use the session `cwd`. @@ -22,13 +22,13 @@ When optional `ctx.sessionReferences` is mounted, the same `@` menu also offers While the agent is running, ordinary editor submissions call `agent.steer()`; otherwise they call `agent.followup()`. A slash at the start of the submitted line enters `ctx.commands` instead: known commands execute directly, unknown commands produce a warning, and neither path automatically reaches the model. A command producer may explicitly schedule agent work; [`dsh-plan-mode`](../../plan/plan-mode/README.md#model-and-human-surfaces) uses that contract for `/plan [message]`. The TUI registers `/help`, `/model`, `/clear`, `/reasoning`, `/tools`, `/redraw`, `/reload`, `/resume`, `/status`, and `/exit` as agent-scoped definitions; every other effective command joins autocomplete and `/help` dynamically, as do `/skill:` completions. A status line above the editor reports the turn phase the TUI derives from session events — waiting for the first token, thinking, responding, or executing tools — with the elapsed time in that phase and the running step total, refreshed each second, and ends with the `Enter sends steering, Esc cancels` hint; while steering messages wait to reach the model it inserts a `N queued ·` badge before the hint that clears as each drains. Ctrl+C or Escape cancels a running turn. Tool cards collapse long bodies into a configurable head/tail preview; Ctrl+O toggles every card between its preview and full output. Ctrl+R toggles reasoning, Ctrl+L redraws, and Ctrl+D exits while idle. -`/model` opens the advisory `ctx.llm` catalog as a keyboard selector: Up/Down moves, Enter selects, and Escape closes it. `/model ` still selects an unambiguous model id directly, while `/model /` selects an exact target. The configured target or latest logged request header initializes the selector, and an unlisted current model remains visible because catalogs are advisory. Selection is local to this TUI session. Prompt assembly snapshots the target for one step, replaces `{{provider}}` and `{{model}}`, and applies the same pair through `agent/request`; a switch during assembly therefore starts with a later step. The request header durably records targets that reach the model, while an unused selection remains process-local. +`/model` opens the advisory `ctx.llm` catalog as a keyboard selector: Up/Down moves, Shift+Tab cycles the focused model's adapter-advertised reasoning efforts in display order, Enter selects the model and effort, and Escape closes it. Models without selectable effort metadata ignore Shift+Tab; the selector does not synthesize `off`, clamp a value, or transfer an effort between models. `/model ` still selects an unambiguous model id directly, while `/model /` selects an exact target and uses its adapter default when one exists. The configured target or latest logged request header initializes the selector, and an unlisted current model remains visible because catalogs are advisory. Selection is local to this TUI session. Prompt assembly snapshots the target for one step, replaces `{{provider}}` and `{{model}}`, and applies the same provider/model/reasoning-effort target through `agent/request`; a switch during assembly therefore starts with a later step. The request header durably records targets that reach the model, while an unused selection remains process-local. `/reload` (EXPERIMENTAL, dev-only) re-reads every file-backed loader config tree and applies the diff to the running app — the HMR watcher's config path, invoked manually; it needs the cordis Loader in the context and degrades to a warning without one, runs only while the agent is idle, and refuses re-entry while a reload is in flight. Module-source hot reload remains watcher-owned. When a `skills` service is mounted, `/skill: [instructions]` loads that skill's instructions into the conversation as a user turn; autocomplete lists the model-invocable skills, and any skill (including a model-disabled one) is loadable by its exact name. The footer sums the session's reported usage as `↑`, followed by `cache %` once any input has been billed — the share of billed prompt tokens (uncached input plus cache reads and writes) served from the provider cache, rounded to a percent. It also compares token-meter pressure with `ctx.llm.resolveModelContext()` for the current route (omitting the context share when the adapter has no capacity metadata) and shows the current model and tool-card mode; the right side clips first when the footer is narrow. -`/status` adds a point-in-time diagnostics card to the transcript and remains available while the agent runs. It reports the session id, title, working directory, selected provider/model, reasoning-block visibility, agent state, event/turn/step/tool-call counts, exact input/output/cache token buckets, KV-cache hit rate, token-meter context use and capacity, creation time, and latest event time. Missing titles, models, cache input, or context capacity are labeled instead of inferred. The card is terminal-only and does not duplicate the compact footer. +`/status` adds a point-in-time diagnostics card to the transcript and remains available while the agent runs. It reports the session id, title, working directory, selected provider/model, selected reasoning effort or default behavior, reasoning-block visibility, agent state, event/turn/step/tool-call counts, exact input/output/cache token buckets, KV-cache hit rate, token-meter context use and capacity, creation time, and latest event time. Missing titles, models, cache input, or context capacity are labeled instead of inferred. The card is terminal-only and does not duplicate the compact footer. `/resume` opens a full-viewport keyboard selector over the current workspace instead of a centered dialog. Its focused search field starts immediately after the search glyph and emits pi-tui's cursor marker, so terminal IME composition remains anchored inside the field. Candidates are sorted by last logged activity and searchable by log-backed title or session id; each row reports current/live/persisted state, last turn outcome, recent provider/model, and durable goal phase when present. Up/Down and Page Up/Page Down navigate, Enter resumes, Escape clears a non-empty search before a second Escape cancels, and Ctrl+C cancels directly. The current session, a session already live in this runtime, an unreadable log, a mismatched cwd, or a session whose logged provider has no current adapter remains visible but disabled. Selection repeats those checks and requires the current agent to be idle before flushing the current session. The TUI then stops the terminal UI and calls the optional host-owned `TuiRuntime.handoffResume`; where `process.execve` is available, the shipped `dsh` host disposes the app and replaces its process. Resume restores the same `SessionId`, transcript, title, todos, and durable goal; goal activation remains disarmed and the TUI asks for human confirmation or `/goal resume`. @@ -47,7 +47,7 @@ The footer sums the session's reported usage as `↑ | `maxResumeOptions` | `8` | Visible sessions in the resume selector | | `questionDialogWidth` | `200` | Question-panel width in columns, clamped to the terminal | | `questionDialogMaxHeight` | `20` | Question-panel maximum rows | -| `modelDialogWidth` | `72` | Model-selector width in columns | +| `modelDialogWidth` | `76` | Model-selector width in columns | | `modelDialogMaxHeight` | `20` | Model-selector maximum rows | | `fileSearchMaxResults` | `20` | Maximum file and directory candidates shown for one `@` query | | `fileSearchMaxEntries` | `10000` | Maximum paths retained in the bounded workspace index used by bare fuzzy queries | @@ -114,7 +114,7 @@ The fixed instruction is part of the stable system-prompt prefix and is reusable #### What the model sees -The `/model` command text and keyboard-selector input are not logged or sent. New steps receive the selected provider/model pair in both prompt variables and request routing. +The `/model` command text and keyboard-selector input are not logged or sent. New steps receive the selected provider/model route in prompt variables and the selected provider/model/reasoning-effort target in request routing. #### Token effect diff --git a/packages/ui/tui/src/index.ts b/packages/ui/tui/src/index.ts index de0ef02f7e..21db41789c 100644 --- a/packages/ui/tui/src/index.ts +++ b/packages/ui/tui/src/index.ts @@ -31,6 +31,7 @@ import { type EditorTheme, type Focusable, type MarkdownTheme, + type SelectItem, type SelectListTheme, type SlashCommand, type Terminal, @@ -53,6 +54,8 @@ import { assertNever, errorChain } from '@deepseek-ai/dsh-llm' import type { ContentBlock, LlmModelInfo, + LlmModelReasoningInfo, + ReasoningEffortId, StreamChunk, TokenUsage, } from '@deepseek-ai/dsh-llm' @@ -233,7 +236,7 @@ const maxModelOptionsSchema = z.number().step(1).min(1).default(8) const maxResumeOptionsSchema = z.number().step(1).min(1).default(8) const questionDialogWidthSchema = z.number().step(1).min(20).default(200) const questionDialogMaxHeightSchema = z.number().step(1).min(6).default(20) -const modelDialogWidthSchema = z.number().step(1).min(20).default(72) +const modelDialogWidthSchema = z.number().step(1).min(20).default(76) const modelDialogMaxHeightSchema = z.number().step(1).min(6).default(20) const fileSearchMaxResultsSchema = z.number().step(1).min(1).default(DEFAULT_FILE_SEARCH_MAX_RESULTS) const fileSearchMaxEntriesSchema = z.number().step(1).min(1).default(DEFAULT_FILE_SEARCH_MAX_ENTRIES) @@ -356,7 +359,7 @@ export function resolveTuiConfig(config: TuiConfig | undefined): ResolvedTuiConf maxResumeOptions: config?.maxResumeOptions ?? 8, questionDialogWidth: config?.questionDialogWidth ?? 200, questionDialogMaxHeight: config?.questionDialogMaxHeight ?? 20, - modelDialogWidth: config?.modelDialogWidth ?? 72, + modelDialogWidth: config?.modelDialogWidth ?? 76, modelDialogMaxHeight: config?.modelDialogMaxHeight ?? 20, fileSearchMaxResults: config?.fileSearchMaxResults ?? DEFAULT_FILE_SEARCH_MAX_RESULTS, fileSearchMaxEntries: config?.fileSearchMaxEntries ?? DEFAULT_FILE_SEARCH_MAX_ENTRIES, @@ -580,15 +583,34 @@ function textBlocks(content: readonly ContentBlock[], type: 'text' | 'reasoning' interface ModelChoice extends AgentLlmTarget { modelName: string description?: string + reasoning?: LlmModelReasoningInfo } function targetLabel(target: AgentLlmTarget): string { return `${target.provider}/${target.model}` } +function compactTargetLabel(target: AgentLlmTarget): string { + return `${target.model}${target.reasoningEffort === undefined ? '' : ` ${target.reasoningEffort}`}` +} + +function targetReasoningLabel(choice: ModelChoice, effort: ReasoningEffortId | undefined): string | undefined { + if (effort === undefined) return choice.reasoning === undefined ? undefined : 'provider default' + return choice.reasoning?.efforts.find(candidate => candidate.id === effort)?.name ?? effort +} + function initialTarget(agent: Agent): AgentLlmTarget | undefined { const logged = agent.session.requestHeader()?.config - if (logged !== undefined) return { provider: logged.provider, model: logged.model } + if (logged !== undefined) { + if (logged.reasoningEffort === undefined) { + return { provider: logged.provider, model: logged.model } + } + return { + provider: logged.provider, + model: logged.model, + reasoningEffort: logged.reasoningEffort, + } + } if (agent.options.provider === undefined || agent.options.model === undefined) return undefined return { provider: agent.options.provider, model: agent.options.model } } @@ -607,11 +629,15 @@ async function readModelChoices( ) { models.push({ provider: provider.id, id: current.model, name: current.model }) } - return models.map((model): ModelChoice => ({ - provider: provider.id, - model: model.id, - modelName: model.name, - ...model.description === undefined ? {} : { description: model.description }, + return Promise.all(models.map(async (model): Promise => { + const reasoning = await ctx.llm.resolveModelReasoning(provider.id, model.id) + return { + provider: provider.id, + model: model.id, + modelName: model.name, + ...model.description === undefined ? {} : { description: model.description }, + ...reasoning === undefined ? {} : { reasoning }, + } })) })) return groups.flat() @@ -1226,6 +1252,10 @@ function renderDialog( class ModelDialog implements Component { private readonly list: SelectList + private readonly items: Map + private readonly choices: Map + private readonly efforts: Map + private readonly currentValue: string | undefined constructor( choices: readonly ModelChoice[], @@ -1235,15 +1265,27 @@ class ModelDialog implements Component { done: (choice: ModelChoice) => void, cancel: () => void, ) { - this.list = new SelectList(choices.map(choice => ({ - value: targetLabel(choice), - label: displayText(targetLabel(choice)), - description: [ - displayText(choice.modelName), - ...choice.description === undefined ? [] : [displayText(choice.description)], - ...current?.provider === choice.provider && current.model === choice.model ? ['current'] : [], - ].join(' — '), - })), maxVisible, dialogSelectTheme(palette)) + this.items = new Map() + this.choices = new Map() + this.efforts = new Map() + this.currentValue = current === undefined ? undefined : targetLabel(current) + for (const choice of choices) { + const value = targetLabel(choice) + const isCurrent = current?.provider === choice.provider && current.model === choice.model + this.choices.set(value, choice) + this.efforts.set( + value, + isCurrent + ? current.reasoningEffort ?? choice.reasoning?.defaultEffort + : choice.reasoning?.defaultEffort, + ) + this.items.set(value, { + value, + label: displayText(value), + description: this.describeChoice(choice, isCurrent), + }) + } + this.list = new SelectList([...this.items.values()], maxVisible, dialogSelectTheme(palette)) const currentIndex = current === undefined ? 0 : choices.findIndex(choice => choice.provider === current.provider && choice.model === current.model) @@ -1252,17 +1294,57 @@ class ModelDialog implements Component { const selected = choices.find(choice => targetLabel(choice) === item.value) /* v8 ignore next -- SelectList only returns values built from `choices`. */ if (selected === undefined) return - done(selected) + const effort = this.efforts.get(item.value) + done({ + ...selected, + ...effort === undefined ? {} : { reasoningEffort: effort }, + }) } this.list.onCancel = cancel } + private describeChoice(choice: ModelChoice, isCurrent: boolean): string { + const selectedEffort = this.efforts.get(targetLabel(choice)) + const effort = choice.reasoning?.efforts.find(candidate => candidate.id === selectedEffort) + const effortLabel = selectedEffort === undefined + ? choice.reasoning === undefined ? undefined : 'provider default' + : effort?.name ?? selectedEffort + return [ + displayText(choice.modelName), + ...choice.description === undefined ? [] : [displayText(choice.description)], + ...effortLabel === undefined ? [] : [displayText(effortLabel)], + ...isCurrent ? ['current'] : [], + ].join(' — ') + } + + private cycleReasoningEffort(): void { + const selectedItem = this.list.getSelectedItem() + /* v8 ignore next -- the dialog is opened only for a non-empty catalog. */ + if (selectedItem === null) return + const choice = this.choices.get(selectedItem.value) + if (choice?.reasoning === undefined) return + const current = this.efforts.get(selectedItem.value) + const currentIndex = choice.reasoning.efforts.findIndex(effort => effort.id === current) + const next = choice.reasoning.efforts[(currentIndex + 1) % choice.reasoning.efforts.length] + /* v8 ignore next -- validated reasoning metadata always carries at least one effort. */ + if (next === undefined) return + this.efforts.set(selectedItem.value, next.id) + const item = this.items.get(selectedItem.value) + /* v8 ignore next -- items and choices are constructed from the same values. */ + if (item === undefined) return + item.description = this.describeChoice(choice, selectedItem.value === this.currentValue) + } + invalidate(): void { this.list.invalidate() } handleInput(data: string): void { - this.list.handleInput(data) + if (matchesKey(data, Key.shift(Key.tab))) { + this.cycleReasoningEffort() + } else { + this.list.handleInput(data) + } this.invalidate() } @@ -1271,7 +1353,7 @@ class ModelDialog implements Component { return renderDialog('Select model', [ ...this.list.render(innerWidth), '', - this.palette.dim('↑/↓ navigate • Enter select • Esc cancel'), + this.palette.dim('↑/↓ navigate • Shift+Tab reasoning • Enter select • Esc cancel'), ], width, this.palette) } } @@ -1938,7 +2020,7 @@ export function createTuiChat( () => sessionTitle ?? config.welcome, palette, resolved.color && resolved.truecolor, - () => target.current?.model, + () => target.current === undefined ? undefined : compactTargetLabel(target.current), ) const footer = new FooterComponent( agent, @@ -1946,7 +2028,7 @@ export function createTuiChat( () => toolsExpanded, () => tokens, runtime.formatCwd, - () => target.current?.model, + () => target.current === undefined ? undefined : compactTargetLabel(target.current), () => contextWindow === undefined ? undefined : Math.min(100, Math.round(ctx.tokenMeter.measure(agent.session).totalTokens / contextWindow * 100)), @@ -2037,13 +2119,26 @@ export function createTuiChat( resolveContextWindow(target.current) const selectModel = (selected: ModelChoice): void => { - if (target.current?.provider === selected.provider && target.current.model === selected.model) { - appendNotice(`Model is already ${targetLabel(selected)}.`) + const sameRoute = target.current?.provider === selected.provider && target.current.model === selected.model + const reasoningEffort = selected.reasoningEffort + ?? (sameRoute ? target.current?.reasoningEffort ?? selected.reasoning?.defaultEffort : selected.reasoning?.defaultEffort) + if (sameRoute && target.current?.reasoningEffort === reasoningEffort) { + const reasoning = targetReasoningLabel(selected, reasoningEffort) + appendNotice(`Model is already ${targetLabel(selected)}${reasoning === undefined ? '' : ` with reasoning effort ${displayText(reasoning)}`}.`) return } - target.current = { provider: selected.provider, model: selected.model } + target.current = { + provider: selected.provider, + model: selected.model, + ...reasoningEffort === undefined ? {} : { reasoningEffort }, + } resolveContextWindow(target.current) - appendNotice(`Model selected: ${targetLabel(selected)}. New steps will use it.`) + const reasoning = targetReasoningLabel(selected, reasoningEffort) + appendNotice([ + `Model selected: ${targetLabel(selected)}.`, + ...reasoning === undefined ? [] : [`Reasoning effort: ${displayText(reasoning)}.`], + 'New steps will use it.', + ].join(' ')) } const showModelSelector = (choices: readonly ModelChoice[]): void => { @@ -2633,12 +2728,17 @@ export function createTuiChat( const steps = events.filter(event => event.type === 'step/start').length const toolCalls = events.filter(event => event.type === 'tool/call').length const model = target.current === undefined ? 'unset' : displayText(targetLabel(target.current)) + const effort = target.current === undefined + ? 'unset' + : target.current.reasoningEffort === undefined + ? 'default' + : displayText(target.current.reasoningEffort) const groups: readonly (readonly StatusCardRow[])[] = [ [ ['Session', displayText(agent.session.id)], ['Title', displayText(sessionTitle ?? 'untitled')], ['Directory', displayText(cwd)], - ['Model', `${model} ${palette.dim(`(reasoning ${showReasoning ? 'shown' : 'hidden'})`)}`], + ['Model', `${model} ${palette.dim(`(effort ${effort}; reasoning blocks ${showReasoning ? 'shown' : 'hidden'})`)}`], ], [ ['Agent', [ diff --git a/packages/ui/tui/tests/harness.ts b/packages/ui/tui/tests/harness.ts index 37eee99735..e42610644c 100644 --- a/packages/ui/tui/tests/harness.ts +++ b/packages/ui/tui/tests/harness.ts @@ -8,7 +8,13 @@ import AgentRegistry, { type AgentStatus, type SendOptions, } from '@deepseek-ai/dsh-agent' -import type { ContentBlock, LlmModelContext, LlmModelInfo, LlmProviderInfo } from '@deepseek-ai/dsh-llm' +import type { + ContentBlock, + LlmModelContext, + LlmModelInfo, + LlmModelReasoningInfo, + LlmProviderInfo, +} from '@deepseek-ai/dsh-llm' import CommandService from '@deepseek-ai/dsh-commands' import SessionStore, { SessionId, type Session, type SessionHeader } from '@deepseek-ai/dsh-session' import SystemPrompt from '@deepseek-ai/dsh-system-prompt' @@ -48,6 +54,10 @@ export interface TuiHarnessOptions { models: LlmModelInfo[] listModels?: (provider: string) => Promise resolveModelContext?: (provider: string, model: string) => Promise + resolveModelReasoning?: ( + provider: string, + model: string, + ) => Promise } /** Provide a fake `sessionPersistence` service so resume surfaces can list sessions. */ sessionPersistence?: { @@ -122,6 +132,9 @@ export async function createTuiTestHarness