Merge remote-tracking branch 'origin/master' into compact-basic-refactor

# Conflicts:
#	docs/cordis-catalog/events-and-services.md
#	examples/AGENTS.md
#	examples/coding-agent/cordis.yml
#	examples/coding-agent/tests/harness.ts
#	scripts/gen-cordis-catalog.ts
This commit is contained in:
Hypatia May
2026-06-30 09:13:15 +08:00
44 changed files with 1169 additions and 23 deletions
+1 -1
View File
@@ -12,7 +12,7 @@ The first REAL agent wiring: DeepSeek V4 + the bash tool suite + stdio chat
pnpm run demo:coding
```
Type a coding task. The agent's only tools are `bash` (+ `bash_output` / `bash_kill` for background tasks): file reads, writes, searches, and test runs all happen through shell commands, each in a fresh `bash -c` (the system prompt tells the model to pass `workdir` instead of `cd`). Reasoning streams dimmed; tool calls/results render inline.
Type a coding task. The agent works through `bash` (+ `bash_output` / `bash_kill` for background tasks): file reads, writes, searches, and test runs all happen through shell commands, each in a fresh `bash -c` (the system prompt tells the model to pass `workdir` instead of `cd`). It can also delegate with `subagent`/`subagent_fork` and track multi-step work with `todo_write` (a whole-list task tracker rendered as a checklist). Reasoning streams dimmed; tool calls/results render inline.
```
> fix the failing test in /path/to/project
+15 -2
View File
@@ -28,7 +28,9 @@
- deepseek-v4-flash
- deepseek-v4-pro
# Local bash executor (the model's only tool, via agent-core's tool-bash schema).
# Local bash executor for agent-core's tool-bash schema.
# FIXME(config-comments): keep this executor note from implying bash is the
# whole tool set; subagent and todo_write are loaded below.
- id: bash
name: '@deepseek-ai/dsh-bash-local'
config:
@@ -44,7 +46,7 @@
# under ./.sessions); unset starts a fresh session each run.
resumeSessionId: !!js process.env.RESUME_SESSION_ID
persistenceRoot: './.sessions'
welcome: 'coding-agent ready. Give it a coding task (its tools are bash and subagent).'
welcome: 'coding-agent ready. Give it a coding task (its tools are bash, subagent, and todo_write).'
systemPrompt: |
You are coding-agent, a CLI coding assistant.
@@ -65,6 +67,12 @@
failures before moving on. Verify your work by running the code or
tests. Keep answers brief and factual.
For multi-step work, use the todo_write tool to track a task list:
send the WHOLE list each call (it replaces the previous one), keep at
most one task in_progress (exactly one while work remains), and mark a
task completed as soon as it is done. Skip it for trivial single-step
tasks.
# Automatic context compaction: when the derived history approaches the model's
# context window, summarize an older range into a checkpoint so a long-running
# or tool-heavy session keeps fitting. A leaf entry (needs ctx.llm + the
@@ -106,3 +114,8 @@
config:
provider: fork
toolName: subagent_fork
# The model-facing todo_write tool: whole-list task tracking written to the
# session log (todo/write), rendered as a stdio checklist / ACP plan.
- id: tool-todo
name: '@deepseek-ai/dsh-tool-todo'
+13 -4
View File
@@ -8,6 +8,7 @@ import AgentRegistry from '@deepseek-ai/dsh-agent'
import AgentLoop, { ReactLoopAgent } from '@deepseek-ai/dsh-agent-loop'
import { LocalBashExecutor } from '@deepseek-ai/dsh-bash-local'
import * as ToolBash from '@deepseek-ai/dsh-tool-bash'
import * as ToolTodo from '@deepseek-ai/dsh-tool-todo'
import * as LlmDeepSeek from '@deepseek-ai/dsh-llm-deepseek'
import SessionPersistenceJsonl from '@deepseek-ai/dsh-session-persistence-jsonl'
import { BasicCompactService } from '@deepseek-ai/dsh-compact-basic'
@@ -15,14 +16,21 @@ import type { BasicCompactConfig } from '@deepseek-ai/dsh-compact-basic'
/**
* Shared harness for the coding-agent e2e suites: the full plugin stack
* with the real DeepSeek adapter and the real bash tool. Lives outside the
* *.e2e.ts pattern so importing it never re-registers another file's tests.
* with the real DeepSeek adapter and the real bash + todo_write tools. Lives
* outside the *.e2e.ts pattern so importing it never re-registers another
* file's tests.
*/
export const SYSTEM_PROMPT = 'You are a coding agent. Your only tool is bash; '
+ 'do file operations with cat/grep/heredocs, check [exit code: N] markers, '
export const SYSTEM_PROMPT = 'You are a coding agent. Use bash for file operations '
+ 'with cat/grep/heredocs; check [exit code: N] markers, '
+ 'and report results briefly.'
/** System prompt for the todo_write e2e: nudges the model to plan with the tool. */
export const TODO_SYSTEM_PROMPT = 'You are a coding agent. For multi-step work, '
+ 'use the todo_write tool to track a task list: send the WHOLE list each call, '
+ 'keep at most one task in_progress (exactly one while work remains), and mark '
+ 'a task completed as soon as it is done.'
/** Options for {@link codingHarness}. */
export interface CodingHarnessOptions {
/** Durable JSONL persistence root (the resume suite needs it; others stay file-free). */
@@ -46,6 +54,7 @@ export async function codingHarness(workdir: string, options: CodingHarnessOptio
await ctx.plugin(LlmDeepSeek, { models: ['deepseek-v4-flash'] })
await ctx.plugin(LocalBashExecutor, { cwd: workdir, timeoutMs: 30_000 })
await ctx.plugin(ToolBash)
await ctx.plugin(ToolTodo)
// Compaction is opt-in: only the compaction e2e loads it, with a lowered
// contextWindow/retainTokens so a short real session crosses the threshold.
if (options.compact !== undefined) await ctx.plugin(BasicCompactService, options.compact)
@@ -0,0 +1,49 @@
import { afterEach, describe, expect, it } from 'vitest'
import type { Context } from 'cordis'
import { AgentId } from '@deepseek-ai/dsh-agent'
import { codingHarness, TODO_SYSTEM_PROMPT, waitForIdle } from './harness.ts'
/**
* A REAL model drives the REAL todo_write tool: verify the WORLD (the session
* log gains a todo/write event whose snapshot the model actually produced), not
* the agent's self-report. Key-gated (see vitest.e2e.config.ts).
*/
let ctx: Context | undefined
afterEach(async () => {
await ctx?.fiber.dispose()
ctx = undefined
})
describe.skipIf(!process.env.DEEPSEEK_API_KEY)('todo_write: real model records a plan', () => {
it('appends a todo/write event with the model-produced task list', async () => {
ctx = await codingHarness(process.cwd())
const agent = ctx.agentLoop.create(AgentId('e2e-todo'), {
model: 'deepseek-v4-flash',
systemPrompt: TODO_SYSTEM_PROMPT,
})
agent.send([{ type: 'text', text:
'Use the todo_write tool to record a plan of exactly two steps: first '
+ '"inspect the failing test" (in_progress), then "apply the fix" (pending). '
+ 'Send both in one todo_write call, then reply with the single word DONE.' }])
await waitForIdle(ctx, agent)
const events = [...agent.session.events]
// The model actually called the tool.
const calls = events.filter(event => event.type === 'tool/call')
expect(calls.some(event => event.data.name === 'todo_write')).toBe(true)
// And the tool wrote a todo/write event to the log — verify the WORLD.
const todoEvents = events.filter(event => event.type === 'todo/write')
expect(todoEvents.length).toBeGreaterThan(0)
const todos = (todoEvents.at(-1)!).data.todos
expect(todos).toEqual([
{ content: 'inspect the failing test', status: 'in_progress' },
{ content: 'apply the fix', status: 'pending' },
])
}, 120_000)
})