fix(code-runtime): bound hostile output accounting

This commit is contained in:
Tianyi Cui
2026-07-21 21:56:38 +08:00
parent dcda5114de
commit 627eb6e00b
4 changed files with 156 additions and 43 deletions
@@ -15,7 +15,7 @@ import { CodeRuntime } from '@deepseek-ai/dsh-code-runtime'
import type { CodeBindingFunction, CodeJsonValue, CodeRunFailure, CodeRunRequest, CodeRunResult } from '@deepseek-ai/dsh-code-runtime'
import { snapshotJsonValue } from '@deepseek-ai/dsh-session'
import type { ReplyMessage, WorkerBootData, WorkerToHost } from './protocol.ts'
import { truncateJsonStringBytes } from './output-json.ts'
import { jsonStringBytesUpTo, jsonValueBytesUpTo, truncateJsonStringBytes } from './output-json.ts'
/** Plugin config: every execution cap, changeable from `cordis.yml` (no hardcoded tunables). */
export interface Config {
@@ -141,11 +141,6 @@ function parseWorkerMessage(raw: unknown): WorkerToHost | undefined {
}
/** Serialized byte size of one lossless JSON value. */
function jsonBytes(value: CodeJsonValue): number {
return Buffer.byteLength(JSON.stringify(value), 'utf8')
}
/** One run's combined outer-output ledger; binding values never enter it. */
class OutputLedger {
private bytes = 2 // JSON serialization of the empty logs array: []
@@ -155,9 +150,10 @@ class OutputLedger {
/** Admit one exact log entry, or report that the hard cap was crossed. */
admit(text: string, sink: string[]): boolean {
const cost = Buffer.byteLength(JSON.stringify(text), 'utf8') + (this.entries > 0 ? 1 : 0)
if (this.bytes + cost > this.maxBytes) return false
this.bytes += cost
const separatorBytes = this.entries > 0 ? 1 : 0
const stringBytes = jsonStringBytesUpTo(text, this.maxBytes - this.bytes - separatorBytes)
if (stringBytes === undefined) return false
this.bytes += stringBytes + separatorBytes
this.entries += 1
sink.push(text)
return true
@@ -165,46 +161,45 @@ class OutputLedger {
/** Finalize a successful absent-or-JSON completion against the combined cap. */
success(logs: string[], value?: CodeJsonValue): CodeRunResult {
if (value !== undefined && this.bytes + jsonBytes(value) > this.maxBytes) return this.limit(logs)
if (value !== undefined && jsonValueBytesUpTo(value, this.maxBytes - this.bytes) === undefined) return this.limit(logs)
return { logs, ...value !== undefined ? { value } : {} }
}
/** Finalize a failure diagnostic, with output-limit taking precedence when combined bytes exceed the cap. */
failure(logs: string[], error: CodeRunFailure): CodeRunResult {
if (this.bytes + Buffer.byteLength(JSON.stringify(error.message), 'utf8') > this.maxBytes) return this.limit(logs)
if (jsonStringBytesUpTo(error.message, this.maxBytes - this.bytes) === undefined) return this.limit(logs)
return { logs, error }
}
/** Build the explicit output-limit failure while retaining a fitting prefix of the final log. */
limit(logs: string[]): CodeRunResult {
const fullMessage = `outer output exceeded ${this.maxBytes} bytes`
const messageBytes = Buffer.byteLength(JSON.stringify(fullMessage), 'utf8')
const retained = [...logs]
let retainedBytes = jsonBytes(retained)
// The fixed diagnostic is ASCII, so every character is one byte plus the quotes.
const messageBytes = fullMessage.length + 2
const retained: string[] = []
let retainedBytes = 2
const logBudget = this.maxBytes - messageBytes
while (retained.length > 0 && retainedBytes > logBudget) {
const removed = retained.pop()
/* v8 ignore next -- the while guard proves pop cannot return undefined. */
if (removed === undefined) throw new Error('output ledger lost its final log entry')
for (const text of logs) {
const separatorBytes = retained.length > 0 ? 1 : 0
retainedBytes -= Buffer.byteLength(JSON.stringify(removed), 'utf8') + separatorBytes
const prefix = truncateJsonStringBytes(removed, logBudget - retainedBytes - separatorBytes)
if (prefix.length > 0) {
retained.push(prefix)
retainedBytes += Buffer.byteLength(JSON.stringify(prefix), 'utf8') + separatorBytes
break
const availableBytes = logBudget - retainedBytes - separatorBytes
const stringBytes = jsonStringBytesUpTo(text, availableBytes)
if (stringBytes !== undefined) {
retained.push(text)
retainedBytes += stringBytes + separatorBytes
continue
}
}
if (logBudget < 2) {
retained.length = 0
retainedBytes = 2
const prefix = truncateJsonStringBytes(text, availableBytes)
if (prefix.length > 0) {
const prefixBytes = jsonStringBytesUpTo(prefix, availableBytes)
/* v8 ignore next -- truncateJsonStringBytes guarantees its returned prefix fits the same budget. */
if (prefixBytes === undefined) throw new Error('output ledger produced an oversized log prefix')
retained.push(prefix)
retainedBytes += prefixBytes + separatorBytes
}
break
}
const availableMessageBytes = this.maxBytes - retainedBytes
// This fixed diagnostic is ASCII with no JSON escapes, so two bytes are
// the surrounding quotes and every retained character costs one byte.
const message = messageBytes <= availableMessageBytes
? fullMessage
: fullMessage.slice(0, availableMessageBytes - 2)
const message = truncateJsonStringBytes(fullMessage, availableMessageBytes)
return { logs: retained, error: { kind: 'output-limit', message } }
}
}
@@ -410,12 +405,9 @@ export class WorkerCodeRuntime extends CodeRuntime {
reply({ type: 'reply', id: message.id, ok: false, message: `unknown binding ${JSON.stringify(`${message.global}.${message.name}`)}` })
return
}
let args: CodeJsonValue | undefined
try {
args = snapshotJsonValue(message.args) as CodeJsonValue | undefined
} catch {
args = undefined
}
// Structured clone has already removed accessors and proxies, so the
// host can repeat the lossless snapshot without a reflective throw.
const args = snapshotJsonValue(message.args) as CodeJsonValue | undefined
if (args === undefined) {
reply({ type: 'reply', id: message.id, ok: false, message: 'binding arguments must be lossless JSON' })
return
@@ -1,5 +1,7 @@
/** JSON string-prefix accounting for the outer-output ledger. @module @deepseek-ai/dsh-code-runtime-worker/output-json */
import type { CodeJsonValue } from '@deepseek-ai/dsh-code-runtime'
/** Control characters with a two-byte short JSON escape instead of `\u00XX`. */
const SHORT_ESCAPE_CODES = new Set([0x08, 0x09, 0x0a, 0x0c, 0x0d])
@@ -13,6 +15,69 @@ function serializedCharacterBytes(character: string): number {
return Buffer.byteLength(character, 'utf8')
}
/**
* Measure one JSON string without materializing its complete escaped form.
* @param text - the candidate string.
* @param maxBytes - largest serialized size the caller can admit.
* @returns Exact serialized bytes, or `undefined` as soon as the cap is crossed.
*/
export function jsonStringBytesUpTo(text: string, maxBytes: number): number | undefined {
if (maxBytes < 2) return undefined
let bytes = 2
for (const character of text) {
bytes += serializedCharacterBytes(character)
if (bytes > maxBytes) return undefined
}
return bytes
}
/**
* Measure one lossless JSON value without allocating its serialized form.
* @param value - already validated lossless JSON.
* @param maxBytes - largest serialized size the caller can admit.
* @returns Exact serialized bytes, or `undefined` as soon as the cap is crossed.
*/
export function jsonValueBytesUpTo(value: CodeJsonValue, maxBytes: number): number | undefined {
if (value === null) return maxBytes >= 4 ? 4 : undefined
if (typeof value === 'string') return jsonStringBytesUpTo(value, maxBytes)
if (typeof value === 'number') {
const bytes = Buffer.byteLength(String(value), 'utf8')
return bytes <= maxBytes ? bytes : undefined
}
if (typeof value === 'boolean') {
const bytes = value ? 4 : 5
return bytes <= maxBytes ? bytes : undefined
}
let bytes = 2
if (bytes > maxBytes) return undefined
if (Array.isArray(value)) {
for (let index = 0; index < value.length; index++) {
if (index > 0 && ++bytes > maxBytes) return undefined
const item = value[index]
if (item === undefined) return undefined
const itemBytes = jsonValueBytesUpTo(item, maxBytes - bytes)
if (itemBytes === undefined) return undefined
bytes += itemBytes
}
return bytes
}
let entries = 0
for (const [key, item] of Object.entries(value)) {
if (entries > 0 && ++bytes > maxBytes) return undefined
const keyBytes = jsonStringBytesUpTo(key, maxBytes - bytes)
if (keyBytes === undefined) return undefined
bytes += keyBytes + 1
if (bytes > maxBytes) return undefined
const itemBytes = jsonValueBytesUpTo(item, maxBytes - bytes)
if (itemBytes === undefined) return undefined
bytes += itemBytes
entries += 1
}
return bytes
}
/**
* Return the longest code-point-aligned prefix whose JSON string encoding,
* including its surrounding quotes, fits `maxBytes`.
@@ -23,7 +88,6 @@ function serializedCharacterBytes(character: string): number {
*/
export function truncateJsonStringBytes(text: string, maxBytes: number): string {
if (maxBytes < 2) return ''
if (Buffer.byteLength(JSON.stringify(text), 'utf8') <= maxBytes) return text
let bytes = 2
let end = 0
for (const character of text) {
@@ -32,5 +96,5 @@ export function truncateJsonStringBytes(text: string, maxBytes: number): string
bytes += cost
end += character.length
}
return text.slice(0, end)
return end === text.length ? text : text.slice(0, end)
}
@@ -1,10 +1,12 @@
import { describe, expect, it } from 'vitest'
import { truncateJsonStringBytes } from '../src/output-json.ts'
import { describe, expect, it, vi } from 'vitest'
import { jsonStringBytesUpTo, jsonValueBytesUpTo, truncateJsonStringBytes } from '../src/output-json.ts'
describe('truncateJsonStringBytes', () => {
it('returns a fitting string whole and rejects budgets without JSON quotes', () => {
expect(truncateJsonStringBytes('fits', 6)).toBe('fits')
expect(truncateJsonStringBytes('x', 1)).toBe('')
expect(jsonStringBytesUpTo('fits', 6)).toBe(6)
expect(jsonStringBytesUpTo('fits', 5)).toBeUndefined()
})
it('accounts every JSON escape and cuts only between complete code points', () => {
@@ -15,4 +17,43 @@ describe('truncateJsonStringBytes', () => {
expect(truncateJsonStringBytes(text, budget)).toBe(prefix)
expect(Buffer.byteLength(JSON.stringify(truncateJsonStringBytes(text, budget)), 'utf8')).toBe(budget)
})
it('bounds hostile strings without materializing their complete escaped form', () => {
const stringify = vi.spyOn(JSON, 'stringify').mockImplementation(() => { throw new Error('must not stringify') })
try {
expect(jsonStringBytesUpTo('"'.repeat(10_000), 32)).toBeUndefined()
expect(truncateJsonStringBytes('"'.repeat(10_000), 32)).toBe('"'.repeat(15))
} finally {
stringify.mockRestore()
}
})
})
describe('jsonValueBytesUpTo', () => {
it('matches JSON serialization for every lossless value branch and stops at the cap', () => {
const value = {
empty: {},
nil: null,
yes: true,
no: false,
number: 1.5,
text: '"\n😀',
array: [1, 'x'],
}
const bytes = Buffer.byteLength(JSON.stringify(value), 'utf8')
expect(jsonValueBytesUpTo(value, bytes)).toBe(bytes)
expect(jsonValueBytesUpTo(value, bytes - 1)).toBeUndefined()
expect(jsonValueBytesUpTo({}, 1)).toBeUndefined()
expect(jsonValueBytesUpTo(null, 3)).toBeUndefined()
expect(jsonValueBytesUpTo(10, 1)).toBeUndefined()
expect(jsonValueBytesUpTo(false, 4)).toBeUndefined()
expect(jsonValueBytesUpTo(new Array<never>(1), 10)).toBeUndefined()
expect(jsonValueBytesUpTo([null], 5)).toBeUndefined()
expect(jsonValueBytesUpTo([0, 0], 3)).toBeUndefined()
expect(jsonValueBytesUpTo({ a: null, b: null }, 10)).toBeUndefined()
expect(jsonValueBytesUpTo({ long: null }, 2)).toBeUndefined()
expect(jsonValueBytesUpTo({ '': null }, 4)).toBeUndefined()
expect(jsonValueBytesUpTo({ a: null }, 9)).toBeUndefined()
})
})
@@ -401,6 +401,22 @@ describe('WorkerCodeRuntime — hostile programs (real workers)', () => {
expect(Buffer.byteLength(JSON.stringify(result.logs), 'utf8')).toBeLessThan(200)
})
it('bounds one oversized forged log while retaining its fitting escaped prefix', async () => {
const { runtime } = await setup({ maxOutputBytes: 96 })
const result = await runtime.run({
program: `
const { parentPort } = await import('node:worker_threads');
parentPort.postMessage({ type: 'log', text: '"'.repeat(1_000_000) });
for (;;) {}
`,
bindings: [],
})
expect(result.error).toEqual({ kind: 'output-limit', message: 'outer output exceeded 96 bytes' })
expect(result.logs).toHaveLength(1)
expect(result.logs[0]).toMatch(/^"+$/)
expect(Buffer.byteLength(JSON.stringify(result.logs), 'utf8') + Buffer.byteLength(JSON.stringify('outer output exceeded 96 bytes'), 'utf8')).toBeLessThanOrEqual(96)
})
it('drops a malformed forged done carrying both value and error', async () => {
const { runtime } = await setup()
const result = await runtime.run({