fix(ralph): harden execution boundaries

This commit is contained in:
Tianyi Cui
2026-07-20 12:56:09 +08:00
parent c6f8247ac1
commit 60eea35df5
37 changed files with 535 additions and 81 deletions
@@ -1,4 +1,4 @@
import { describe, expect, it } from 'vitest'
import { describe, expect, it, vi } from 'vitest'
import { Context } from 'cordis'
import type { Agent } from '@deepseek-ai/dsh-agent'
import AgentLoop from '@deepseek-ai/dsh-agent-loop'
@@ -10,9 +10,31 @@ import SubagentService from '@deepseek-ai/dsh-subagent'
import { STRUCTURED_OUTPUT_TOOL } from '@deepseek-ai/dsh-subagent-inprocess'
import * as spawn from '@deepseek-ai/dsh-subagent-spawn'
import WorkerWorkflowEngine from '@deepseek-ai/dsh-workflow-workerthread'
import { MockAdapter, textResponse, toolCallResponse } from '../../../core/agent-loop/tests/mock-adapter.ts'
import { MockAdapter, maxTokensResponse, textResponse, toolCallResponse } from '../../../core/agent-loop/tests/mock-adapter.ts'
import * as toolRalph from '../src/index.ts'
type MockScript = ConstructorParameters<typeof MockAdapter>[0]
/** Mount the shipped Ralph execution stack around one keyless model script. */
async function mountRalph(script: MockScript, config: toolRalph.Config) {
const ctx = new Context()
const adapter = new MockAdapter(script)
await mountAgentLoopTestDependencies(ctx)
await ctx.plugin(Invariants)
await ctx.plugin(AgentLoop, { agents: [] })
await ctx.plugin(SubagentService)
await ctx.plugin(spawn, { providerName: 'spawn' })
await ctx.plugin(WorkerWorkflowEngine, {})
await ctx.plugin(toolRalph, config)
ctx.llm.registerAdapter(['mock'], adapter)
const parentHandle = await ctx.agents.create({
sessionId: SessionId('ralph-parent'),
meta: { cwd: '/tmp/ralph-shared-workspace' },
agentOptions: { provider: 'mock', model: 'mock' },
})
return { ctx, adapter, parentHandle, parent: parentHandle.agent }
}
describe('dsh-tool-ralph over the real spawn and worker-thread stack', () => {
it('uses distinct empty-seed children, shared cwd, and only the prior bounded handoff', async () => {
const firstReport = {
@@ -54,6 +76,8 @@ describe('dsh-tool-ralph over the real spawn and worker-thread stack', () => {
await parent.whenIdle()
const children: Agent[] = []
const phases: string[] = []
ctx.on('workflow/phase', (_run, title) => { phases.push(title) })
ctx.on('workflow/agent-start', (_run, child) => {
const agent = ctx.agents.get(child.childId)
expect(agent).toBeDefined()
@@ -67,7 +91,9 @@ describe('dsh-tool-ralph over the real spawn and worker-thread stack', () => {
})
expect(result.isError).toBe(false)
expect((result.content[0] as { text: string }).text).toContain('Ralph completed after 2 rounds.')
expect((result.content[0] as { text: string }).text)
.toContain('Ralph worker reported completion after 2 rounds.')
expect(phases).toEqual(['Fresh-agent rounds'])
expect(children).toHaveLength(2)
expect(new Set(children.map(child => child.id)).size).toBe(2)
for (const child of children) {
@@ -89,4 +115,151 @@ describe('dsh-tool-ralph over the real spawn and worker-thread stack', () => {
await parentHandle.dispose()
})
it('reports the failed round and last good handoff when a child fails', async () => {
const firstReport = {
status: 'continue',
summary: 'ROUND_ONE_HANDOFF',
evidence: ['Created migration-a.ts.'],
nextSteps: ['Finish migration-b.ts.'],
blocker: '',
}
const { ctx, parent, parentHandle } = await mountRalph([
toolCallResponse('round-1', STRUCTURED_OUTPUT_TOOL, firstReport),
maxTokensResponse('unfinished child output'),
], { maxRounds: 2 })
const children: Agent[] = []
ctx.on('workflow/agent-start', (_run, child) => {
const agent = ctx.agents.get(child.childId)
if (agent !== undefined) children.push(agent)
})
const result = await ctx.tools.execute({
callId: CallId('ralph-child-failure'),
name: 'ralph',
arguments: { objective: 'Complete both migration slices.', maxRounds: 2 },
agent: parent,
})
expect(result.isError).toBe(true)
const text = (result.content[0] as { text: string }).text
expect(text).toContain('Ralph round 2 child failed before producing a structured report.')
expect(text).toContain('Last successful handoff:')
expect(text).toContain('ROUND_ONE_HANDOFF')
expect(children).toHaveLength(2)
for (const child of children) expect(ctx.agents.get(child.id)).toBeUndefined()
await parentHandle.dispose()
})
it.each([
{
name: 'blocked',
report: {
status: 'blocked',
summary: 'External authorization is required.',
evidence: ['The local implementation is ready.'],
nextSteps: ['Continue after authorization.'],
blocker: 'The required external authorization is unavailable.',
},
config: { maxRounds: 2 },
expectedError: false,
expectedText: 'Ralph worker reported a blocker after 1 round.',
},
{
name: 'budget-limited',
report: {
status: 'continue',
summary: 'One slice is complete.',
evidence: ['The first focused test passes.'],
nextSteps: ['Implement the remaining slice.'],
blocker: '',
},
config: { maxRounds: 1 },
expectedError: false,
expectedText: 'Ralph reached its 1 round limit; the worker reported work remaining.',
},
{
name: 'unnormalized report',
report: {
status: 'continue',
summary: ' padded summary ',
evidence: ['A focused test passes.'],
nextSteps: ['Continue implementation.'],
blocker: '',
},
config: { maxRounds: 1 },
expectedError: true,
expectedText: 'summary must be non-empty and normalized',
},
{
name: 'invalid continuing report',
report: {
status: 'continue',
summary: 'Work remains.',
evidence: ['A focused test passes.'],
nextSteps: [],
blocker: '',
},
config: { maxRounds: 1 },
expectedError: true,
expectedText: 'a continuing Ralph report needs nextSteps and an empty blocker',
},
{
name: 'oversized report',
report: {
status: 'continue',
summary: 'x'.repeat(300),
evidence: ['A focused test passes.'],
nextSteps: ['Continue implementation.'],
blocker: '',
},
config: { maxRounds: 1, maxHandoffChars: 100 },
expectedError: true,
expectedText: 'Ralph round report exceeds maxHandoffChars',
},
])('enforces the fixed script for $name', async ({ report, config, expectedError, expectedText }) => {
const { ctx, parent, parentHandle } = await mountRalph([
toolCallResponse('round-report', STRUCTURED_OUTPUT_TOOL, report),
], config)
const result = await ctx.tools.execute({
callId: CallId('ralph-script-enforcement'),
name: 'ralph',
arguments: { objective: 'Complete the scoped work.', maxRounds: config.maxRounds },
agent: parent,
})
expect(result.isError).toBe(expectedError)
expect((result.content[0] as { text: string }).text).toContain(expectedText)
await parentHandle.dispose()
})
it('cancels the real worker and fresh child to quiescence', async () => {
const { ctx, parent, parentHandle } = await mountRalph(['hang'], { maxRounds: 2 })
const children: Agent[] = []
const outcomes: string[] = []
ctx.on('workflow/agent-start', (_run, child) => {
const agent = ctx.agents.get(child.childId)
if (agent !== undefined) children.push(agent)
})
ctx.on('workflow/agent-end', (_run, child) => { outcomes.push(child.outcome) })
const controller = new AbortController()
const pending = ctx.tools.execute({
callId: CallId('ralph-real-cancel'),
name: 'ralph',
arguments: { objective: 'Keep working until cancelled.', maxRounds: 2 },
agent: parent,
signal: controller.signal,
})
await vi.waitFor(() => { expect(children).toHaveLength(1) })
controller.abort()
const result = await pending
expect(result.isError).toBe(true)
expect((result.content[0] as { text: string }).text).toContain('Ralph workflow was cancelled')
expect(outcomes).toEqual(['cancelled'])
expect(ctx.agents.get(children[0]!.id)).toBeUndefined()
await parentHandle.dispose()
})
})
@@ -82,6 +82,7 @@ async function setup(options?: SetupOptions) {
if (options?.config?.subagentProvider !== undefined) config.subagentProvider = options.config.subagentProvider
if (options?.config?.maxRounds !== undefined) config.maxRounds = options.config.maxRounds
if (options?.config?.maxHandoffChars !== undefined) config.maxHandoffChars = options.config.maxHandoffChars
if (options?.config?.maxResultChars !== undefined) config.maxResultChars = options.config.maxResultChars
const fiber = await ctx.plugin(toolRalph, config)
const parent = { id: SessionId('caller'), options: {} } as unknown as Agent
return { ctx, engine: ctx.workflows as StubEngine, parent, fiber }
@@ -145,6 +146,7 @@ describe('dsh-tool-ralph', () => {
meta: { name: 'ralph-loop' },
args: { objective: 'Finish the migration.', maxRounds: 4, maxHandoffChars: 9000 },
subagentProvider: 'fresh',
maxTotalAgents: 4,
parent,
})
expect(engine.requests[0]!.script).toContain("status: 'budget-limited'")
@@ -154,7 +156,8 @@ describe('dsh-tool-ralph', () => {
report: COMPLETE,
})
expect(result.isError).toBe(false)
expect((result.content[0] as { text: string }).text).toContain('Ralph completed after 1 round.')
expect((result.content[0] as { text: string }).text)
.toContain('Ralph worker reported completion after 1 round.')
expect((result.content[0] as { text: string }).text).toContain('All required gates pass.')
expect(engine.disposed).toBe(1)
})
@@ -167,7 +170,8 @@ describe('dsh-tool-ralph', () => {
roundsStarted: 2,
report: BLOCKED,
}, 2)
expect((blockedResult.content[0] as { text: string }).text).toContain('Ralph blocked after 2 rounds.')
expect((blockedResult.content[0] as { text: string }).text)
.toContain('Ralph worker reported a blocker after 2 rounds.')
const limited = execute(ctx, { objective: 'Ship it.' }, { agent: parent })
await vi.waitFor(() => { expect(engine.requests).toHaveLength(2) })
@@ -177,7 +181,54 @@ describe('dsh-tool-ralph', () => {
report: CONTINUE,
}, 2)
expect((limitedResult.content[0] as { text: string }).text)
.toContain('Ralph reached its 2 rounds limit with work remaining.')
.toContain('Ralph reached its 2 rounds limit; the worker reported work remaining.')
})
it('bounds the complete parent result and labels worker-reported completion', async () => {
const { ctx, engine, parent } = await setup({ config: { maxResultChars: 160 } })
const pending = execute(ctx, { objective: 'Ship it.' }, { agent: parent })
const result = await settleCompleted(engine, pending, {
status: 'complete',
roundsStarted: 1,
report: { ...COMPLETE, evidence: ['x'.repeat(500)] },
})
const text = (result.content[0] as { text: string }).text
expect(text).toHaveLength(160)
expect(text).toContain('Ralph worker reported completion after 1 round.')
expect(text).toMatch(/… \[truncated\]$/)
})
it('honors a result limit shorter than the truncation marker', async () => {
const { ctx, engine, parent } = await setup({ config: { maxResultChars: 5 } })
const result = await settleCompleted(engine, execute(ctx, { objective: 'Ship it.' }, { agent: parent }), {
status: 'complete',
roundsStarted: 1,
report: COMPLETE,
})
expect((result.content[0] as { text: string }).text).toBe('\n… [t')
})
it('reports an ordinary child failure with the failed round and last durable handoff', async () => {
const { ctx, engine, parent } = await setup({ config: { maxRounds: 2 } })
const first = execute(ctx, { objective: 'Ship it.', maxRounds: 2 }, { agent: parent })
const firstResult = await settleCompleted(engine, first, {
status: 'round-failed',
roundsStarted: 1,
lastReport: null,
})
expect(firstResult.isError).toBe(true)
expect((firstResult.content[0] as { text: string }).text).toContain('Ralph round 1 child failed')
expect((firstResult.content[0] as { text: string }).text).toContain('No previous handoff was available.')
const later = execute(ctx, { objective: 'Ship it.', maxRounds: 2 }, { agent: parent })
const laterResult = await settleCompleted(engine, later, {
status: 'round-failed',
roundsStarted: 2,
lastReport: CONTINUE,
})
expect(laterResult.isError).toBe(true)
expect((laterResult.content[0] as { text: string }).text).toContain('Ralph round 2 child failed')
expect((laterResult.content[0] as { text: string }).text).toContain('Implemented the first slice.')
})
it('maps workflow error and cancellation reasons to tool errors and always disposes', async () => {
@@ -251,6 +302,7 @@ describe('dsh-tool-ralph', () => {
expect(() => { toolRalph.apply(new Context(), { subagentProvider: ' ' }) }).toThrow('non-empty normalized')
expect(() => { toolRalph.apply(new Context(), { maxRounds: 0 }) }).toThrow('positive safe integer')
expect(() => { toolRalph.apply(new Context(), { maxHandoffChars: 1.5 }) }).toThrow('positive safe integer')
expect(() => { toolRalph.apply(new Context(), { maxResultChars: 0 }) }).toThrow('positive safe integer')
})
it('turns malformed fixed-workflow terminal values and reports into errors', async () => {
@@ -261,11 +313,18 @@ describe('dsh-tool-ralph', () => {
{ value: { status: 'mystery', roundsStarted: 1, report: COMPLETE }, message: 'unknown terminal status' },
{ value: { status: 'budget-limited', roundsStarted: 1, report: CONTINUE }, message: 'before the round limit', config: { maxRounds: 2 } },
{ value: { status: 'complete', roundsStarted: 1, report: null }, message: 'malformed round report' },
{ value: { status: 'complete', roundsStarted: 1, report: COMPLETE, extra: true }, message: 'malformed terminal result' },
{ value: { status: 'blocked', roundsStarted: 1, report: BLOCKED, extra: true }, message: 'malformed terminal result' },
{ value: { status: 'budget-limited', roundsStarted: 1, report: CONTINUE, extra: true }, message: 'malformed terminal result', config: { maxRounds: 1 } },
{ value: { status: 'complete', roundsStarted: 1, report: { ...COMPLETE, status: 'continue' } }, message: 'malformed round report' },
{ value: { status: 'budget-limited', roundsStarted: 1, report: { ...CONTINUE, nextSteps: [] } }, message: 'invalid continuing report', config: { maxRounds: 1 } },
{ value: { status: 'complete', roundsStarted: 1, report: { ...COMPLETE, evidence: [] } }, message: 'invalid completion report' },
{ value: { status: 'blocked', roundsStarted: 1, report: { ...BLOCKED, blocker: '' } }, message: 'invalid blocked report' },
{ value: { status: 'complete', roundsStarted: 1, report: { ...COMPLETE, summary: 'x'.repeat(500) } }, message: 'oversized handoff', config: { maxHandoffChars: 100 } },
{ value: { status: 'round-failed', roundsStarted: 1 }, message: 'malformed terminal result' },
{ value: { status: 'round-failed', roundsStarted: 1, lastReport: CONTINUE }, message: 'invalid first-round failure' },
{ value: { status: 'round-failed', roundsStarted: 2, lastReport: null }, message: 'without its last handoff', config: { maxRounds: 2 } },
{ value: { status: 'round-failed', roundsStarted: 2, lastReport: { ...CONTINUE, nextSteps: [] } }, message: 'invalid continuing report', config: { maxRounds: 2 } },
]
for (const testCase of cases) {
const { ctx, engine, parent } = await setup(
@@ -294,7 +353,9 @@ describe('dsh-tool-ralph', () => {
const { ctx, fiber } = await setup()
const section = (await ctx.systemPrompt.assemble()).sections.find(candidate => candidate.name === 'tool:ralph')
expect(section?.text).toContain('ONLY when the direct human explicitly asks')
expect(section?.text).toContain('worker reports, not independent evaluation')
const tool = ctx.tools.get('ralph')!
expect(tool.description).toContain('worker reports completion')
expect(tool.presentCall!({ objective: 'Finish it.' })).toEqual({
card: 'generic',
title: 'ralph',