Type-only change (brands are zero-cost casts; no runtime/wire impact). Closes the two gaps in the "brand ids that cross package boundaries" policy and fixes the dependency direction so a capability package never pulls in an unrelated one. - Extract the `Branded<B>` primitive into a new standalone type-only package `@deepseek-ai/dsh-brand` (packages/util/brand) with no harness-package deps. dsh-llm keeps its owned CallId but imports Branded from dsh-brand; dsh-session, dsh-agent, and dsh-bash all import Branded from there. dsh-bash depends on dsh-brand ALONE — never on dsh-llm or dsh-session (the architectural fix: a generic execution backend must not couple to the LLM or session vocabulary). - Mint BashTaskId + OwnerToken in dsh-bash and thread them through BashTask.id, the get/ownerOf/list/readOutput/kill seam, the bash-local generation site, and the dsh-tool-bash validate/access surface. OwnerToken is a DISTINCT brand from SessionId so the seam stays decoupled; dsh-tool-bash is the single boundary that casts SessionId -> OwnerToken. - Brand at the SOURCE, not via mid-pipeline casts: agent-loop's Config types agents[].id as AgentId and resumeSessionId as SessionId, so the brand enters at the config boundary and the inner create()/resume casts disappear (only the genuinely-new per-run session-id string is cast). - Stop brand erosion: propagate CallId/SessionId/AgentId to the registry/store Map keys and public params/exports (SessionStore, AgentRegistry + factory options, the ACP session-id surface + ToolPresenter CallId map, the persistence coordinator, invariants pendingCalls, the pi-ai tool-call maps). - Docs: document BashTaskId/OwnerToken in bash.md (type-equiv re-pasted), point the Branded type-equiv at dsh-brand, fix stale param types in the session/ agent/bash READMEs, regenerate the cordis catalog + module graph. Implements docs/rfc/proposed/architecture/2026-06-20-branded-ids.md
88 lines
3.1 KiB
TypeScript
88 lines
3.1 KiB
TypeScript
import { spawnSync } from 'node:child_process'
|
|
import { mkdtemp, readFile, rm, writeFile } from 'node:fs/promises'
|
|
import { tmpdir } from 'node:os'
|
|
import { join } from 'node:path'
|
|
import { afterEach, describe, expect, it } from 'vitest'
|
|
import type { Context } from 'cordis'
|
|
import { AgentId } from '@deepseek-ai/dsh-agent'
|
|
import { codingHarness, finalText, SYSTEM_PROMPT, waitForIdle } from './harness.ts'
|
|
|
|
/**
|
|
* The swebench-style smoke test: a real model fixes a real bug in a temp
|
|
* directory using only the bash tool, and the fix is verified OUTSIDE the
|
|
* agent by re-running the test script. Key-gated.
|
|
*/
|
|
|
|
const TEST_FILE = [
|
|
"const assert = require('node:assert');",
|
|
"const { add } = require('./add.js');",
|
|
'assert.strictEqual(add(2, 3), 5);',
|
|
'assert.strictEqual(add(-1, 1), 0);',
|
|
"console.log('PASS');",
|
|
'',
|
|
].join('\n')
|
|
|
|
const BUGGY_ADD = [
|
|
'// A tiny module with an obvious bug.',
|
|
'function add(a, b) {',
|
|
' return a - b;',
|
|
'}',
|
|
'module.exports = { add };',
|
|
'',
|
|
].join('\n')
|
|
|
|
let workdir: string | undefined
|
|
let ctx: Context | undefined
|
|
|
|
afterEach(async () => {
|
|
// Dispose the harness even on failure/retry: agent-loop teardown stops the
|
|
// loop and LocalBashExecutor teardown kills anything the model left running.
|
|
await ctx?.fiber.dispose()
|
|
ctx = undefined
|
|
if (workdir !== undefined) await rm(workdir, { recursive: true, force: true })
|
|
workdir = undefined
|
|
})
|
|
|
|
describe.skipIf(!process.env.DEEPSEEK_API_KEY)('coding task: fix a failing test via bash', () => {
|
|
it('repairs add.js so node add.test.js passes', async () => {
|
|
workdir = await mkdtemp(join(tmpdir(), 'dsh-coding-task-'))
|
|
await writeFile(join(workdir, 'add.js'), BUGGY_ADD)
|
|
await writeFile(join(workdir, 'add.test.js'), TEST_FILE)
|
|
|
|
// Confirm the fixture actually fails before the agent touches it.
|
|
const before = spawnSync('node', ['add.test.js'], { cwd: workdir })
|
|
expect(before.status).not.toBe(0)
|
|
|
|
ctx = await codingHarness(workdir)
|
|
const agent = ctx.agentLoop.create(AgentId('e2e-task'), {
|
|
model: 'deepseek-v4-flash',
|
|
systemPrompt: SYSTEM_PROMPT,
|
|
})
|
|
|
|
agent.send([{
|
|
type: 'text',
|
|
text: 'In the current directory, `node add.test.js` fails because add.js has a bug. '
|
|
+ 'Fix add.js so the test passes, run `node add.test.js` to verify, and report the result. '
|
|
+ 'Do not modify add.test.js.',
|
|
}])
|
|
await waitForIdle(ctx, agent)
|
|
|
|
// The agent claims success…
|
|
const summary = finalText([...agent.session.events]).toLowerCase()
|
|
expect(summary.length).toBeGreaterThan(0)
|
|
|
|
// …and the world agrees: the test passes when WE run it, and the test
|
|
// file is byte-identical (an agent that neutered the test instead of
|
|
// fixing the bug fails here, not just on a keyword probe).
|
|
const untouchedTest = await readFile(join(workdir, 'add.test.js'), 'utf8')
|
|
expect(untouchedTest).toBe(TEST_FILE)
|
|
|
|
const after = spawnSync('node', ['add.test.js'], { cwd: workdir, encoding: 'utf8' })
|
|
expect(after.stdout).toContain('PASS')
|
|
expect(after.status).toBe(0)
|
|
|
|
const fixed = await readFile(join(workdir, 'add.js'), 'utf8')
|
|
expect(fixed).not.toMatch(/a\s*-\s*b/)
|
|
}, 180_000)
|
|
})
|