2026-07-09 16:05:44 +08:00
import { chmodSync , mkdirSync , mkdtempSync } from 'node:fs'
2026-06-12 23:28:44 +08:00
import { tmpdir } from 'node:os'
import { join } from 'node:path'
import { describe , expect , it , vi } from 'vitest'
import { Context } from 'cordis'
import { CallId } from '@deepseek-ai/dsh-llm'
2026-07-09 16:41:03 +08:00
import { BashExecutor , BashTaskId , setSandboxMode } from '@deepseek-ai/dsh-bash'
2026-06-21 07:17:25 +08:00
import type { BashExecRequest , BashExecSpec , BashRunResult , BashTask , BashTaskRead , OwnerToken } from '@deepseek-ai/dsh-bash'
2026-07-09 16:41:03 +08:00
import { Session , SessionId } from '@deepseek-ai/dsh-session'
2026-06-12 23:28:44 +08:00
import SystemPrompt from '@deepseek-ai/dsh-system-prompt'
import ToolRegistry from '@deepseek-ai/dsh-tools'
2026-06-20 08:14:27 +08:00
import AgentRegistry from '@deepseek-ai/dsh-agent'
import type { Agent } from '@deepseek-ai/dsh-agent'
2026-06-12 23:28:44 +08:00
import { LocalBashExecutor } from '@deepseek-ai/dsh-bash-local'
2026-07-09 16:05:44 +08:00
import { SandboxBashExecutor } from '@deepseek-ai/dsh-bash-sandbox'
import { SandboxProvider } from '@deepseek-ai/dsh-sandbox'
import type { ConfinedArgv } from '@deepseek-ai/dsh-sandbox'
import { LocalSandboxProvider } from '@deepseek-ai/dsh-sandbox-local'
2026-07-11 21:37:38 +08:00
import ApprovalService from '@deepseek-ai/dsh-user-approval'
import type { ApprovalOutcome } from '@deepseek-ai/dsh-user-approval'
2026-06-12 23:28:44 +08:00
import * as ToolBash from '@deepseek-ai/dsh-tool-bash'
import { renderResult } from '@deepseek-ai/dsh-tool-bash'
const spillDir = mkdtempSync ( join ( tmpdir ( ) , 'dsh-tool-bash-spec-' ) )
2026-07-09 16:05:44 +08:00
// Pure-config passthrough runner (same knob the snapshot tier uses): skips the
// profile args up to `--` and execs the command unconfined — deterministic
// without a host bwrap.
const PASSTHROUGH_RUNNER = [ 'bash' , '-c' , 'while [ "$1" != "--" ]; do shift; done; shift; exec "$@"' , 'passthrough-runner' ]
2026-07-11 21:37:38 +08:00
const PASSTHROUGH_RUNNER_CONFIG = {
runnerCommand : PASSTHROUGH_RUNNER ,
// The script has no pre-exec failure path; the provider still requires an
// explicit dialect so a future script change cannot silently turn runner
// failure into an ordinary command result.
runnerFailureSignatures : [ 'passthrough-runner: profile rejected' ] ,
}
2026-07-09 16:05:44 +08:00
2026-06-12 23:28:44 +08:00
async function setup() {
const ctx = new Context ( )
await ctx . plugin ( SystemPrompt )
await ctx . plugin ( ToolRegistry )
2026-06-20 08:14:27 +08:00
await ctx . plugin ( AgentRegistry )
2026-07-04 17:37:23 +08:00
await ctx . plugin ( LocalBashExecutor , { timeoutMs : 10_000 , graceMs : 200 } )
; ( ctx . bash as LocalBashExecutor ) . internals = { spillDir }
2026-06-12 23:28:44 +08:00
await ctx . plugin ( ToolBash )
return ctx
}
2026-06-20 08:14:27 +08:00
/**
* Build a fake {@link Agent} whose session token is `sessionId`, REGISTER it in
* `ctx.agents` (the completion-notice path finds the owning agent by scanning
* the registry for a matching `session.header.id`), and return it. The returned
* agent is also passed to `execute` as `exec.agent` so it owns the spawned task.
* The registration disposer is tracked so {@link unregisterFakeAgents} can drop
* it (simulating the owning session disconnecting before a task completes).
*/
2026-07-09 13:05:44 +08:00
const fakeAgentDisposers = new Map < Context , ( ( ) = > Promise < void > | void ) [ ] > ( )
2026-06-20 08:14:27 +08:00
function registerFakeAgent ( ctx : Context , sessionId : string , inject : ( . . . args : unknown [ ] ) = > void ) : Agent {
2026-06-20 09:57:05 +08:00
// The registry KEY (agent.id) is deliberately DIFFERENT from the session
// token (session.header.id) — a config agent has `agentId !== sessionId`. The
// owner token IS the session id, so the notice path must find the agent by
// `session.header.id`, NOT the registry key. Using distinct values here makes
// the test fail if a regression matched on the wrong field (a same-value fake
// would pass either way — the "hits the line but not the scenario" trap).
2026-06-21 11:08:10 +08:00
const agent = { id : ` agent- ${ sessionId } ` , inject , session : { header : { version : 0 , id : sessionId , createdAt : 0 } } } as unknown as Agent
2026-06-20 08:14:27 +08:00
const dispose = ctx . agents . register ( agent )
const list = fakeAgentDisposers . get ( ctx ) ? ? [ ]
list . push ( dispose )
fakeAgentDisposers . set ( ctx , list )
return agent
}
/** Unregister every fake agent in this ctx (simulate the owning session disconnecting). */
function unregisterFakeAgents ( ctx : Context ) : void {
2026-07-09 13:05:44 +08:00
for ( const dispose of fakeAgentDisposers . get ( ctx ) ? ? [ ] ) void dispose ( )
2026-06-20 08:14:27 +08:00
fakeAgentDisposers . delete ( ctx )
}
2026-06-12 23:28:44 +08:00
let callCounter = 0
function call ( ctx : Context , name : string , args : unknown ) {
return ctx . tools . execute ( { callId : CallId ( ` call- ${ ++ callCounter } ` ) , name , arguments : args } )
}
function text ( result : { content : { type : string ; text? : string } [ ] } ) : string {
return result . content . filter ( block = > block . type === 'text' ) . map ( block = > block . text ) . join ( '' )
}
2026-06-23 16:27:49 +08:00
async function callUntilText (
ctx : Context ,
name : string ,
args : unknown ,
expected : string ,
timeoutMs = 5 _000 ,
) : Promise < Awaited < ReturnType < typeof call > > > {
const deadline = Date . now ( ) + timeoutMs
let last : Awaited < ReturnType < typeof call > > | undefined
while ( Date . now ( ) < deadline ) {
last = await call ( ctx , name , args )
if ( text ( last ) . includes ( expected ) ) return last
await new Promise ( resolve = > setTimeout ( resolve , 20 ) )
}
throw new Error ( ` ${ name } output did not include ${ JSON . stringify ( expected ) } ; last text was ${ JSON . stringify ( last !== undefined ? text ( last ) : '' ) } ` )
}
2026-07-13 23:42:54 +08:00
abstract class TestBashExecutor extends BashExecutor {
2026-06-19 01:54:57 +08:00
resolve ( request : BashExecRequest ) : BashExecSpec {
return {
command : request.command ,
workdir : request.workdir ? ? process . cwd ( ) ,
timeoutMs : request.timeoutMs ? ? 0 ,
2026-07-10 11:53:02 +08:00
stdoutMaxBytes : request.stdoutMaxBytes ? ? 64 _000 ,
2026-06-19 01:54:57 +08:00
. . . request . signal ? { signal : request.signal } : { } ,
2026-06-20 08:14:27 +08:00
owner : request.owner ,
2026-07-09 16:05:44 +08:00
sandboxMode : request.sandboxMode ,
2026-06-19 01:54:57 +08:00
}
}
2026-07-13 23:42:54 +08:00
}
class LossyReadBashExecutor extends TestBashExecutor {
private readonly task : BashTask = {
id : BashTaskId ( 'bash-lossy' ) ,
command : 'fake' ,
status : 'running' ,
exitCode : null ,
signal : null ,
done : Promise.resolve ( ) ,
}
2026-06-19 01:54:57 +08:00
run ( ) : Promise < BashRunResult > {
return Promise . reject ( new Error ( 'not used' ) )
}
start ( ) : BashTask {
return this . task
}
2026-06-21 07:17:25 +08:00
get ( id : BashTaskId ) : BashTask | undefined {
2026-06-19 01:54:57 +08:00
return id === this . task . id ? this . task : undefined
}
2026-06-21 07:17:25 +08:00
ownerOf ( ) : OwnerToken | undefined {
2026-06-20 08:14:27 +08:00
return undefined
}
2026-06-19 01:54:57 +08:00
list ( ) : BashTask [ ] {
return [ this . task ]
}
2026-06-21 07:17:25 +08:00
readOutput ( id : BashTaskId ) : BashTaskRead {
2026-06-19 01:54:57 +08:00
if ( id !== this . task . id ) throw new Error ( ` unknown bash task " ${ id } " ` )
return { task : this.task , delta : 'tail' , lossy : true }
}
kill ( ) : boolean {
return false
}
}
2026-06-12 23:28:44 +08:00
describe ( 'bash tool' , ( ) = > {
it ( 'returns stdout for a successful command' , async ( ) = > {
const ctx = await setup ( )
const result = await call ( ctx , 'bash' , { command : 'echo hello' , description : 'test command' } )
expect ( result . isError ) . toBe ( false )
expect ( text ( result ) ) . toBe ( 'hello\n' )
} )
it ( 'reports (no output) for silent commands' , async ( ) = > {
const ctx = await setup ( )
const result = await call ( ctx , 'bash' , { command : 'true' , description : 'test command' } )
expect ( text ( result ) ) . toBe ( '(no output)' )
} )
it ( 'marks stderr sections' , async ( ) = > {
const ctx = await setup ( )
const result = await call ( ctx , 'bash' , { command : 'echo out; echo err >&2' , description : 'test command' } )
expect ( text ( result ) ) . toBe ( 'out\n[stderr]\nerr\n' )
expect ( result . isError ) . toBe ( false )
} )
it ( 'reports non-zero exits without isError' , async ( ) = > {
const ctx = await setup ( )
const result = await call ( ctx , 'bash' , { command : 'echo failing; exit 3' , description : 'test command' } )
expect ( result . isError ) . toBe ( false )
expect ( text ( result ) ) . toBe ( 'failing\n[exit code: 3]' )
} )
it ( 'reports timeout kills with both markers (timeout first)' , async ( ) = > {
const ctx = await setup ( )
const result = await call ( ctx , 'bash' , { command : 'sleep 60' , description : 'test command' , timeoutMs : 100 } )
expect ( result . isError ) . toBe ( false )
expect ( text ( result ) ) . toBe ( '(no output)\n[timed out after 100ms]\n[killed by signal: SIGTERM]' )
} )
it ( 'reports a timeout even when the command traps the signal and exits 0' , async ( ) = > {
// The signal-independent timeout marker: a trapped SIGTERM that exits 0
// after our timer fired must NOT look like a clean success. (bash may
// print "Terminated" to stderr for the killed sleep — environment
// dependent — so assert the marker, not the exact body.)
const ctx = await setup ( )
const result = await call ( ctx , 'bash' , { command : 'trap "exit 0" TERM; sleep 60' , description : 'test command' , timeoutMs : 100 } )
expect ( result . isError ) . toBe ( false )
expect ( text ( result ) ) . toContain ( '[timed out after 100ms]' )
expect ( text ( result ) ) . not . toContain ( '[exit code:' )
} )
it ( 'reports truncation with the spill path' , async ( ) = > {
const ctx = new Context ( )
await ctx . plugin ( SystemPrompt )
await ctx . plugin ( ToolRegistry )
2026-07-04 17:37:23 +08:00
await ctx . plugin ( LocalBashExecutor , { maxOutputBytes : 100 , graceMs : 200 } )
; ( ctx . bash as LocalBashExecutor ) . internals = { spillDir }
2026-06-12 23:28:44 +08:00
await ctx . plugin ( ToolBash )
const result = await call ( ctx , 'bash' , { command : 'for i in $(seq 1 100); do printf "line-%04d\\n" $i; done' , description : 'test command' } )
expect ( text ( result ) ) . toContain ( '[output truncated; full output: ' )
expect ( text ( result ) ) . toContain ( 'line-0100' )
} )
it ( 'honors workdir' , async ( ) = > {
const ctx = await setup ( )
const result = await call ( ctx , 'bash' , { command : 'pwd' , description : 'test command' , workdir : '/tmp' } )
expect ( text ( result ) . trim ( ) ) . toMatch ( /\/tmp$/ )
} )
it ( 'surfaces spawn failures as isError' , async ( ) = > {
const ctx = await setup ( )
const result = await call ( ctx , 'bash' , { command : 'true' , description : 'test command' , workdir : '/nonexistent-dsh' } )
expect ( result . isError ) . toBe ( true )
expect ( text ( result ) ) . toMatch ( /ENOENT/ )
} )
it ( 'surfaces aborts as isError' , async ( ) = > {
const ctx = await setup ( )
const controller = new AbortController ( )
const pending = ctx . tools . execute ( {
callId : CallId ( 'call-abort' ) ,
name : 'bash' ,
arguments : { command : 'sleep 60' , description : 'test command' } ,
signal : controller.signal ,
} )
setTimeout ( ( ) = > { controller . abort ( ) } , 50 )
const result = await pending
expect ( result . isError ) . toBe ( true )
expect ( text ( result ) ) . toMatch ( /aborted/ )
} )
2026-06-13 23:00:42 +08:00
// Type and required-key violations are now rejected by the harness
2026-06-18 02:18:24 +08:00
// (defineTool validates against the SchemaSpec — the arg-validation RFC) before execute.
2026-06-12 23:28:44 +08:00
it . each ( [
2026-06-13 23:00:42 +08:00
[ { } , /missing required property "command"/ ] ,
[ { command : 42 , description : 'd' } , /"command" must be a string/ ] ,
[ { command : 'x' } , /missing required property "description"/ ] ,
[ { command : 'x' , description : 7 } , /"description" must be a string/ ] ,
[ { command : 'x' , description : 'd' , timeoutMs : 'soon' } , /"timeoutMs" must be a number/ ] ,
[ { command : 'x' , description : 'd' , workdir : 7 } , /"workdir" must be a string/ ] ,
[ { command : 'x' , description : 'd' , run_in_background : 'yes' } , /"run_in_background" must be a boolean/ ] ,
] ) ( 'rejects schema-invalid args %j' , async ( args , pattern ) = > {
const ctx = await setup ( )
const result = await call ( ctx , 'bash' , args )
expect ( result . isError ) . toBe ( true )
expect ( text ( result ) ) . toMatch ( pattern )
} )
// Value constraints the SchemaSpec can't express stay in the tool body.
it . each ( [
[ { command : ' ' , description : 'd' } , /invalid command/ ] ,
[ { command : 'x' , description : ' ' } , /invalid description/ ] ,
2026-06-12 23:28:44 +08:00
[ { command : 'x' , description : 'd' , timeoutMs : - 1 } , /invalid timeoutMs/ ] ,
2026-06-13 23:00:42 +08:00
] ) ( 'rejects value-invalid args %j' , async ( args , pattern ) = > {
2026-06-12 23:28:44 +08:00
const ctx = await setup ( )
const result = await call ( ctx , 'bash' , args )
expect ( result . isError ) . toBe ( true )
expect ( text ( result ) ) . toMatch ( pattern )
} )
2026-07-11 22:55:26 +08:00
it ( 'rejects a non-JSON numeric argument before tool-specific validation' , async ( ) = > {
const ctx = await setup ( )
const result = await call ( ctx , 'bash' , {
command : 'x' , description : 'd' , timeoutMs : Number.NaN ,
} )
expect ( result . isError ) . toBe ( true )
expect ( text ( result ) ) . toContain ( 'tool execution arguments must be losslessly JSON-serializable' )
} )
2026-06-12 23:28:44 +08:00
it ( 'registers all three schemas in the system prompt assembly' , async ( ) = > {
const ctx = await setup ( )
const names = ctx . tools . schemas ( ) . map ( schema = > schema . name )
expect ( names ) . toEqual ( [ 'bash' , 'bash_output' , 'bash_kill' ] )
const bashSchema = ctx . tools . schemas ( ) [ 0 ] !
expect ( bashSchema . parameters ) . toMatchObject ( {
type : 'object' ,
required : [ 'command' , 'description' ] ,
} )
} )
2026-07-05 01:54:46 +08:00
it ( 'contributes the exit-code habit as its prompt section (guidance the descriptions cannot carry)' , async ( ) = > {
const ctx = await setup ( )
const assembly = await ctx . systemPrompt . assemble ( )
const section = assembly . sections . find ( s = > s . name === 'tool:bash' )
expect ( section ? . order ) . toBe ( 105 )
expect ( section ? . text ) . toContain ( '[exit code: N]' )
} )
2026-06-12 23:28:44 +08:00
it ( 'unregisters everything when the plugin fiber is disposed (HMR safety)' , async ( ) = > {
const ctx = new Context ( )
await ctx . plugin ( SystemPrompt )
await ctx . plugin ( ToolRegistry )
await ctx . plugin ( LocalBashExecutor , { } )
const fiber = await ctx . plugin ( ToolBash )
expect ( ctx . tools . schemas ( ) ) . toHaveLength ( 3 )
2026-07-05 23:23:46 +08:00
expect ( ( await ctx . systemPrompt . assemble ( ) ) . sections . map ( s = > s . name ) ) . toEqual ( [ 'harness:identity' , 'deployment:persona' , 'tool:bash' ] )
2026-06-12 23:28:44 +08:00
await fiber . dispose ( )
expect ( ctx . tools . schemas ( ) ) . toHaveLength ( 0 )
2026-07-05 23:23:46 +08:00
// Only the system-prompt plugin's own built-in sections remain.
expect ( ( await ctx . systemPrompt . assemble ( ) ) . sections . map ( s = > s . name ) ) . toEqual ( [ 'harness:identity' , 'deployment:persona' ] )
2026-06-12 23:28:44 +08:00
} )
it ( 'tools depend on the executor: no registration without ctx.bash' , async ( ) = > {
const ctx = new Context ( )
await ctx . plugin ( SystemPrompt )
await ctx . plugin ( ToolRegistry )
// inject: ['tools', 'bash'] keeps the plugin pending until bash exists.
await ctx . plugin ( ToolBash )
expect ( ctx . tools . schemas ( ) ) . toHaveLength ( 0 )
await ctx . plugin ( LocalBashExecutor , { } )
await new Promise ( resolve = > setTimeout ( resolve , 0 ) )
expect ( ctx . tools . schemas ( ) ) . toHaveLength ( 3 )
} )
} )
describe ( 'background tools' , ( ) = > {
it ( 'bash with run_in_background returns a task id immediately' , async ( ) = > {
const ctx = await setup ( )
const result = await call ( ctx , 'bash' , { command : 'sleep 0.2; echo bg-done' , description : 'test command' , run_in_background : true } )
expect ( result . isError ) . toBe ( false )
expect ( text ( result ) ) . toMatch ( /^started background task bash-\d+$/ )
} )
it ( 'bash_output polls incrementally and reports status' , async ( ) = > {
const ctx = await setup ( )
2026-06-23 16:27:49 +08:00
const started = await call ( ctx , 'bash' , { command : 'echo first; sleep 1; echo second' , description : 'test command' , run_in_background : true } )
2026-06-21 07:17:25 +08:00
const id = BashTaskId ( /task (bash-\d+)/ . exec ( text ( started ) ) ! [ 1 ] ! )
2026-06-12 23:28:44 +08:00
2026-06-23 16:27:49 +08:00
const first = await callUntilText ( ctx , 'bash_output' , { task_id : id } , 'first' )
2026-06-12 23:28:44 +08:00
expect ( text ( first ) ) . toContain ( 'first' )
expect ( text ( first ) ) . toContain ( '[status: running]' )
await ctx . bash . get ( id ) ! . done
const second = await call ( ctx , 'bash_output' , { task_id : id } )
expect ( text ( second ) ) . toContain ( 'second' )
expect ( text ( second ) ) . not . toContain ( 'first' )
expect ( text ( second ) ) . toContain ( '[status: completed, exit code: 0]' )
const third = await call ( ctx , 'bash_output' , { task_id : id } )
expect ( text ( third ) ) . toContain ( '(no new output)' )
} )
it ( 'bash_output flags lossy reads with spill paths' , async ( ) = > {
const ctx = new Context ( )
await ctx . plugin ( SystemPrompt )
await ctx . plugin ( ToolRegistry )
2026-07-04 17:37:23 +08:00
await ctx . plugin ( LocalBashExecutor , { maxOutputBytes : 100 , graceMs : 200 } )
; ( ctx . bash as LocalBashExecutor ) . internals = { spillDir }
2026-06-12 23:28:44 +08:00
await ctx . plugin ( ToolBash )
const started = await call ( ctx , 'bash' , { command : 'for i in $(seq 1 200); do printf "line-%04d\\n" $i; done' , description : 'test command' , run_in_background : true } )
2026-06-21 07:17:25 +08:00
const id = BashTaskId ( /task (bash-\d+)/ . exec ( text ( started ) ) ! [ 1 ] ! )
2026-06-12 23:28:44 +08:00
await ctx . bash . get ( id ) ! . done
const read = await call ( ctx , 'bash_output' , { task_id : id } )
expect ( text ( read ) ) . toContain ( '[some output was dropped from memory; full output: ' )
} )
2026-06-19 01:54:57 +08:00
it ( 'bash_output reports unavailable when a lossy read has no safe spill path' , async ( ) = > {
const ctx = new Context ( )
await ctx . plugin ( SystemPrompt )
await ctx . plugin ( ToolRegistry )
await ctx . plugin ( LossyReadBashExecutor )
await ctx . plugin ( ToolBash )
const read = await call ( ctx , 'bash_output' , { task_id : 'bash-lossy' } )
expect ( text ( read ) ) . toBe ( 'tail\n[some output was dropped from memory; full output: (unavailable)]\n[status: running]' )
} )
2026-06-12 23:28:44 +08:00
it ( 'bash_kill stops a running task; repeat reports already-finished' , async ( ) = > {
const ctx = await setup ( )
const started = await call ( ctx , 'bash' , { command : 'sleep 60' , description : 'test command' , run_in_background : true } )
2026-06-21 07:17:25 +08:00
const id = BashTaskId ( /task (bash-\d+)/ . exec ( text ( started ) ) ! [ 1 ] ! )
2026-06-12 23:28:44 +08:00
const killed = await call ( ctx , 'bash_kill' , { task_id : id } )
expect ( text ( killed ) ) . toBe ( ` killed background task ${ id } ` )
await ctx . bash . get ( id ) ! . done
const again = await call ( ctx , 'bash_kill' , { task_id : id } )
expect ( text ( again ) ) . toBe ( ` task ${ id } had already finished ` )
const status = await call ( ctx , 'bash_output' , { task_id : id } )
expect ( text ( status ) ) . toContain ( '[status: killed by SIGTERM]' )
} )
it ( 'unknown task ids are isError for both tools' , async ( ) = > {
const ctx = await setup ( )
const read = await call ( ctx , 'bash_output' , { task_id : 'bash-999' } )
expect ( read . isError ) . toBe ( true )
expect ( text ( read ) ) . toMatch ( /unknown bash task/ )
const kill = await call ( ctx , 'bash_kill' , { task_id : 'bash-999' } )
expect ( kill . isError ) . toBe ( true )
} )
it . each ( [
2026-06-13 23:00:42 +08:00
[ 'bash_output' , { } , /missing required property "task_id"/ ] ,
[ 'bash_output' , { task_id : 9 } , /"task_id" must be a string/ ] ,
[ 'bash_kill' , { task_id : '' } , /invalid task_id/ ] ,
] ) ( '%s rejects invalid task_id %j' , async ( tool , args , pattern ) = > {
2026-06-12 23:28:44 +08:00
const ctx = await setup ( )
const result = await call ( ctx , tool , args )
expect ( result . isError ) . toBe ( true )
2026-06-13 23:00:42 +08:00
expect ( text ( result ) ) . toMatch ( pattern )
2026-06-12 23:28:44 +08:00
} )
2026-06-20 08:14:27 +08:00
it ( 'injects a completion notice into the owning agent (found via the registry by session token)' , async ( ) = > {
2026-06-12 23:28:44 +08:00
const ctx = await setup ( )
const inject = vi . fn ( )
2026-06-20 08:14:27 +08:00
// The notice path looks the agent up in ctx.agents by its session token, so
// the agent must be REGISTERED (not merely passed to execute). Mount a
// registry and register a fake whose session.header.id IS the owner token.
const agent = registerFakeAgent ( ctx , 'bg' , inject )
2026-06-12 23:28:44 +08:00
const started = await ctx . tools . execute ( {
callId : CallId ( 'call-bg' ) ,
name : 'bash' ,
arguments : { command : 'true' , description : 'test command' , run_in_background : true } ,
agent ,
} )
2026-06-21 07:17:25 +08:00
const id = BashTaskId ( /task (bash-\d+)/ . exec ( text ( started ) ) ! [ 1 ] ! )
2026-06-12 23:28:44 +08:00
await ctx . bash . get ( id ) ! . done
expect ( inject ) . toHaveBeenCalledTimes ( 1 )
const [ content , options ] = inject . mock . calls [ 0 ] as [
{ type : string ; text : string } [ ] ,
{ source : { kind : string ; plugin : string } } ,
]
expect ( content [ 0 ] ! . text ) . toContain ( ` background bash task ${ id } finished ` )
expect ( content [ 0 ] ! . text ) . toContain ( 'bash_output' )
expect ( options . source ) . toEqual ( { kind : 'plugin' , plugin : 'tool-bash' } )
} )
it ( 'swallows ONLY the disposed-agent inject error' , async ( ) = > {
const ctx = await setup ( )
2026-06-20 08:14:27 +08:00
const agent = registerFakeAgent ( ctx , 'bg' , ( ) = > { throw new Error ( 'agent "x" is disposed' ) } )
2026-06-12 23:28:44 +08:00
const started = await ctx . tools . execute ( {
callId : CallId ( 'call-bg2' ) ,
name : 'bash' ,
arguments : { command : 'true' , description : 'test command' , run_in_background : true } ,
agent ,
} )
2026-06-21 07:17:25 +08:00
const id = BashTaskId ( /task (bash-\d+)/ . exec ( text ( started ) ) ! [ 1 ] ! )
2026-06-12 23:28:44 +08:00
await expect ( ctx . bash . get ( id ) ! . done ) . resolves . toBeUndefined ( )
} )
it ( 'rethrows a non-disposed inject failure (not blindly swallowed)' , async ( ) = > {
const ctx = await setup ( )
// A real bug in inject (not the benign disposed race) must surface — the
// base-class notifier contains it (logs, does not reject task.done), but
// the listener itself must have thrown rather than silently eaten it.
const errorSpy = vi . spyOn ( console , 'error' ) . mockImplementation ( ( ) = > undefined )
try {
2026-06-20 08:14:27 +08:00
const agent = registerFakeAgent ( ctx , 'bg' , ( ) = > { throw new Error ( 'unexpected inject bug' ) } )
2026-06-12 23:28:44 +08:00
const started = await ctx . tools . execute ( {
callId : CallId ( 'call-bg3' ) ,
name : 'bash' ,
arguments : { command : 'true' , description : 'test command' , run_in_background : true } ,
agent ,
} )
2026-06-21 07:17:25 +08:00
const id = BashTaskId ( /task (bash-\d+)/ . exec ( text ( started ) ) ! [ 1 ] ! )
2026-06-12 23:28:44 +08:00
await ctx . bash . get ( id ) ! . done
// notifyTaskDone caught and logged the rethrown error.
expect ( errorSpy ) . toHaveBeenCalled ( )
const logged = errorSpy . mock . calls . flat ( ) . some ( arg = > arg instanceof Error && arg . message === 'unexpected inject bug' )
expect ( logged ) . toBe ( true )
} finally {
errorSpy . mockRestore ( )
}
} )
2026-06-20 08:14:27 +08:00
it ( 'drops the notice cleanly when the owning agent is gone from the registry by completion' , async ( ) = > {
// A bash task (owned by the host-scoped bash-local fiber) can OUTLIVE its
// per-session agent — e.g. the ACP session disconnects and its AgentHandle
// disposes while the background task is still running. The owner token is
// still on the task, but no live agent carries it anymore, so the registry
// lookup finds nothing and the notice is dropped (no throw).
const ctx = await setup ( )
const inject = vi . fn ( )
const agent = registerFakeAgent ( ctx , 'bg' , inject )
const started = await ctx . tools . execute ( {
callId : CallId ( 'call-bg4' ) ,
name : 'bash' ,
arguments : { command : 'true' , description : 'test command' , run_in_background : true } ,
agent ,
} )
2026-06-21 07:17:25 +08:00
const id = BashTaskId ( /task (bash-\d+)/ . exec ( text ( started ) ) ! [ 1 ] ! )
2026-06-20 08:14:27 +08:00
// Unregister the agent BEFORE the task completes (simulate disconnect).
unregisterFakeAgents ( ctx )
2026-06-21 06:17:11 +08:00
await expect ( ctx . bash . get ( id ) ! . done ) . resolves . toBeUndefined ( )
2026-06-20 08:14:27 +08:00
expect ( inject ) . not . toHaveBeenCalled ( )
} )
2026-06-12 23:28:44 +08:00
it ( 'does not notify when no agent owned the task' , async ( ) = > {
const ctx = await setup ( )
const started = await call ( ctx , 'bash' , { command : 'true' , description : 'test command' , run_in_background : true } )
2026-06-21 07:17:25 +08:00
const id = BashTaskId ( /task (bash-\d+)/ . exec ( text ( started ) ) ! [ 1 ] ! )
2026-06-12 23:28:44 +08:00
await expect ( ctx . bash . get ( id ) ! . done ) . resolves . toBeUndefined ( )
} )
} )
2026-06-16 19:23:21 +08:00
describe ( 'background task ownership (cross-session isolation)' , ( ) = > {
/** Run a tool on behalf of a specific agent (sets exec.agent). */
function callAs ( ctx : Context , agent : import ( '@deepseek-ai/dsh-agent' ) . Agent | undefined , name : string , args : unknown ) {
return ctx . tools . execute ( { callId : CallId ( ` own- ${ ++ callCounter } ` ) , name , arguments : args , . . . agent ? { agent } : { } } )
}
2026-06-20 08:14:27 +08:00
// Ownership is by TOKEN (session.header.id), NOT agent object identity — so
// each agent needs a DISTINCT session id, else every fake yields the same
// token and the isolation tests pass for the wrong reason (all tasks owned by
// the same token). The impl reads `session.header.id`, so the fakes MUST carry
// it.
const fakeAgent = ( sessionId : string ) = >
2026-06-21 11:08:10 +08:00
( { inject : ( ) = > undefined , session : { header : { version : 0 , id : sessionId , createdAt : 0 } } } ) as unknown as import ( '@deepseek-ai/dsh-agent' ) . Agent
2026-06-20 08:14:27 +08:00
it ( 'rejects bash_output/bash_kill for a task owned by a DIFFERENT session token' , async ( ) = > {
2026-06-16 19:23:21 +08:00
const ctx = await setup ( )
2026-06-20 08:14:27 +08:00
const a = fakeAgent ( 'sess-a' )
const b = fakeAgent ( 'sess-b' )
2026-06-16 19:23:21 +08:00
// Agent A starts a long-running background task.
const started = await callAs ( ctx , a , 'bash' , { command : 'sleep 60' , description : 'bg' , run_in_background : true } )
2026-06-21 07:17:25 +08:00
const id = BashTaskId ( /task (bash-\d+)/ . exec ( text ( started ) ) ! [ 1 ] ! )
2026-06-16 19:23:21 +08:00
2026-06-20 08:14:27 +08:00
// Agent B (a different session token) cannot read or kill A's task.
2026-06-16 19:23:21 +08:00
const readByB = await callAs ( ctx , b , 'bash_output' , { task_id : id } )
expect ( readByB . isError ) . toBe ( true )
expect ( text ( readByB ) ) . toMatch ( /belongs to another session/ )
const killByB = await callAs ( ctx , b , 'bash_kill' , { task_id : id } )
expect ( killByB . isError ) . toBe ( true )
expect ( text ( killByB ) ) . toMatch ( /belongs to another session/ )
// The task is still running (B's kill did nothing) — A can still kill it.
const killByA = await callAs ( ctx , a , 'bash_kill' , { task_id : id } )
expect ( killByA . isError ) . toBe ( false )
expect ( text ( killByA ) ) . toBe ( ` killed background task ${ id } ` )
} )
2026-06-20 13:52:02 +08:00
it ( 'a DIFFERENT Agent object with the SAME session token may access the task (ownership is by token, not object identity)' , async ( ) = > {
// Ownership fences by session.header.id, NOT Agent object identity. Two
// distinct Agent objects sharing one session token (e.g. an agent re-created
// on the same session) are the SAME owner.
2026-06-20 08:14:27 +08:00
const ctx = await setup ( )
const a1 = fakeAgent ( 'sess-shared' )
const a2 = fakeAgent ( 'sess-shared' ) // distinct object, same token
const started = await callAs ( ctx , a1 , 'bash' , { command : 'sleep 60' , description : 'bg' , run_in_background : true } )
2026-06-21 07:17:25 +08:00
const id = BashTaskId ( /task (bash-\d+)/ . exec ( text ( started ) ) ! [ 1 ] ! )
2026-06-20 08:14:27 +08:00
const readByA2 = await callAs ( ctx , a2 , 'bash_output' , { task_id : id } )
expect ( readByA2 . isError ) . toBe ( false )
await callAs ( ctx , a1 , 'bash_kill' , { task_id : id } ) // cleanup
} )
2026-06-16 19:23:21 +08:00
it ( 'the no-agent (non-loop) caller cannot access an owned task' , async ( ) = > {
const ctx = await setup ( )
2026-06-20 08:14:27 +08:00
const a = fakeAgent ( 'sess-a' )
2026-06-16 19:23:21 +08:00
const started = await callAs ( ctx , a , 'bash' , { command : 'sleep 60' , description : 'bg' , run_in_background : true } )
2026-06-21 07:17:25 +08:00
const id = BashTaskId ( /task (bash-\d+)/ . exec ( text ( started ) ) ! [ 1 ] ! )
2026-06-20 08:14:27 +08:00
// A call with no exec.agent has no token → cannot prove ownership of an owned task.
2026-06-16 19:23:21 +08:00
const read = await callAs ( ctx , undefined , 'bash_output' , { task_id : id } )
expect ( read . isError ) . toBe ( true )
expect ( text ( read ) ) . toMatch ( /belongs to another session/ )
await callAs ( ctx , a , 'bash_kill' , { task_id : id } ) // cleanup
} )
it ( 'an UNOWNED task (started with no agent) is accessible to anyone' , async ( ) = > {
const ctx = await setup ( )
2026-06-20 08:14:27 +08:00
// Started by a non-loop caller (no exec.agent) → no owner token recorded.
2026-06-16 19:23:21 +08:00
const started = await callAs ( ctx , undefined , 'bash' , { command : 'sleep 60' , description : 'bg' , run_in_background : true } )
2026-06-21 07:17:25 +08:00
const id = BashTaskId ( /task (bash-\d+)/ . exec ( text ( started ) ) ! [ 1 ] ! )
2026-06-16 19:23:21 +08:00
// Any agent (and the no-agent caller) may read/kill it.
2026-06-20 08:14:27 +08:00
const read = await callAs ( ctx , fakeAgent ( 'sess-x' ) , 'bash_output' , { task_id : id } )
2026-06-16 19:23:21 +08:00
expect ( read . isError ) . toBe ( false )
const killed = await callAs ( ctx , undefined , 'bash_kill' , { task_id : id } )
expect ( killed . isError ) . toBe ( false )
} )
2026-06-20 08:14:27 +08:00
it ( 'the owner can still access its task AFTER it completes (owner token persists on the task)' , async ( ) = > {
2026-06-16 19:23:21 +08:00
const ctx = await setup ( )
2026-06-20 08:14:27 +08:00
const a = fakeAgent ( 'sess-a' )
const b = fakeAgent ( 'sess-b' )
2026-06-16 19:23:21 +08:00
const started = await callAs ( ctx , a , 'bash' , { command : 'echo done' , description : 'bg' , run_in_background : true } )
2026-06-21 07:17:25 +08:00
const id = BashTaskId ( /task (bash-\d+)/ . exec ( text ( started ) ) ! [ 1 ] ! )
2026-06-16 19:23:21 +08:00
await ctx . bash . get ( id ) ! . done
// Completion does NOT clear ownership: B is still rejected, A still allowed.
const readByB = await callAs ( ctx , b , 'bash_output' , { task_id : id } )
expect ( readByB . isError ) . toBe ( true )
expect ( text ( readByB ) ) . toMatch ( /belongs to another session/ )
const readByA = await callAs ( ctx , a , 'bash_output' , { task_id : id } )
expect ( readByA . isError ) . toBe ( false )
} )
2026-06-20 08:14:27 +08:00
it ( 'ownership SURVIVES an independent tool-bash HMR reload (token lives on the executor)' , async ( ) = > {
// The owner token lives on the TASK inside the executor (dsh-bash fiber), NOT
// in a tool-bash plugin-local map. So reloading ONLY tool-bash (executor +
2026-06-20 13:52:02 +08:00
// task survive) preserves ownership. This is the regression guard: a
// plugin-local map would make B accessible after reload, and this test would
// catch it.
2026-06-16 19:23:21 +08:00
const ctx = new Context ( )
await ctx . plugin ( SystemPrompt )
await ctx . plugin ( ToolRegistry )
2026-07-04 17:37:23 +08:00
await ctx . plugin ( LocalBashExecutor , { timeoutMs : 10_000 , graceMs : 200 } )
; ( ctx . bash as LocalBashExecutor ) . internals = { spillDir }
2026-06-16 19:23:21 +08:00
const fiber = await ctx . plugin ( ToolBash )
2026-06-20 08:14:27 +08:00
const a = fakeAgent ( 'sess-a' )
const b = fakeAgent ( 'sess-b' )
2026-06-16 19:23:21 +08:00
const started = await callAs ( ctx , a , 'bash' , { command : 'sleep 60' , description : 'bg' , run_in_background : true } )
2026-06-21 07:17:25 +08:00
const id = BashTaskId ( /task (bash-\d+)/ . exec ( text ( started ) ) ! [ 1 ] ! )
2026-06-16 19:23:21 +08:00
// Before reload: B is rejected (A owns it).
expect ( ( await callAs ( ctx , b , 'bash_output' , { task_id : id } ) ) . isError ) . toBe ( true )
2026-06-20 08:14:27 +08:00
// Reload ONLY tool-bash; the executor and its running task (with its owner
// token) survive.
2026-06-16 19:23:21 +08:00
await fiber . dispose ( )
await ctx . plugin ( ToolBash )
expect ( ctx . bash . get ( id ) ? . status ) . toBe ( 'running' )
2026-06-20 08:14:27 +08:00
expect ( ctx . bash . ownerOf ( id ) ) . toBe ( 'sess-a' )
2026-06-16 19:23:21 +08:00
2026-06-20 08:14:27 +08:00
// After reload, ownership is INTACT → B is STILL rejected.
expect ( ( await callAs ( ctx , b , 'bash_output' , { task_id : id } ) ) . isError ) . toBe ( true )
await callAs ( ctx , a , 'bash_kill' , { task_id : id } ) // cleanup
2026-06-16 19:23:21 +08:00
} )
} )
2026-06-17 10:01:18 +08:00
describe ( 'session-cwd routing (per-session workdir)' , ( ) = > {
function callAs ( ctx : Context , agent : import ( '@deepseek-ai/dsh-agent' ) . Agent | undefined , args : unknown ) {
return ctx . tools . execute ( { callId : CallId ( ` cwd- ${ ++ callCounter } ` ) , name : 'bash' , arguments : args , . . . agent ? { agent } : { } } )
}
// An agent whose session header carries a cwd (what session/new records).
const agentInCwd = ( cwd : string ) = >
2026-06-21 11:08:10 +08:00
( { inject : ( ) = > undefined , session : { header : { version : 0 , id : 'c' , createdAt : 0 , cwd } } } ) as unknown as import ( '@deepseek-ai/dsh-agent' ) . Agent
2026-06-17 10:01:18 +08:00
it ( 'defaults bash to the agent\'s session cwd (not the server launch dir)' , async ( ) = > {
const ctx = await setup ( )
const result = await callAs ( ctx , agentInCwd ( '/tmp' ) , { command : 'pwd' , description : 'pwd' } )
expect ( text ( result ) . trim ( ) ) . toMatch ( /\/tmp$/ )
} )
it ( 'an explicit absolute workdir overrides the session cwd' , async ( ) = > {
const ctx = await setup ( )
const result = await callAs ( ctx , agentInCwd ( '/' ) , { command : 'pwd' , description : 'pwd' , workdir : '/tmp' } )
expect ( text ( result ) . trim ( ) ) . toMatch ( /\/tmp$/ )
} )
it ( 'a relative workdir is resolved against the session cwd' , async ( ) = > {
const ctx = await setup ( )
// session cwd /usr + relative 'bin' → /usr/bin
const result = await callAs ( ctx , agentInCwd ( '/usr' ) , { command : 'pwd' , description : 'pwd' , workdir : 'bin' } )
expect ( text ( result ) . trim ( ) ) . toMatch ( /\/usr\/bin$/ )
} )
it ( 'two sessions with different cwds each run bash in their own dir' , async ( ) = > {
const ctx = await setup ( )
const inUsr = await callAs ( ctx , agentInCwd ( '/usr' ) , { command : 'pwd' , description : 'pwd' } )
const inTmp = await callAs ( ctx , agentInCwd ( '/tmp' ) , { command : 'pwd' , description : 'pwd' } )
expect ( text ( inUsr ) . trim ( ) ) . toMatch ( /\/usr$/ )
expect ( text ( inTmp ) . trim ( ) ) . toMatch ( /\/tmp$/ )
} )
it ( 'falls back to the executor default when the agent has no session cwd' , async ( ) = > {
const ctx = await setup ( )
// No exec.agent at all → executor uses its config/process.cwd() default.
const result = await ctx . tools . execute ( { callId : CallId ( 'cwd-noagent' ) , name : 'bash' , arguments : { command : 'pwd' , description : 'pwd' } } )
expect ( result . isError ) . toBe ( false )
expect ( text ( result ) . trim ( ) . length ) . toBeGreaterThan ( 0 )
} )
} )
2026-06-12 23:28:44 +08:00
describe ( 'renderResult' , ( ) = > {
const base = {
exitCode : 0 as number | null ,
signal : null as NodeJS . Signals | null ,
timedOut : false ,
aborted : false ,
timeoutMs : 1000 ,
stdout : { text : '' , truncated : false } ,
stderr : { text : '' , truncated : false } ,
}
it ( 'renders stderr-only output without a stdout prefix' , ( ) = > {
expect ( renderResult ( { . . . base , stderr : { text : 'err\n' , truncated : false } } ) )
. toBe ( '[stderr]\nerr\n' )
} )
it ( 'adds a separator when stdout does not end with a newline' , ( ) = > {
expect ( renderResult ( {
. . . base ,
stdout : { text : 'out' , truncated : false } ,
stderr : { text : 'err' , truncated : false } ,
} ) ) . toBe ( 'out\n[stderr]\nerr' )
} )
it ( 'appends exit-code markers after a newline for unterminated output' , ( ) = > {
expect ( renderResult ( { . . . base , exitCode : 7 , stdout : { text : 'x' , truncated : false } } ) )
. toBe ( 'x\n[exit code: 7]' )
} )
it ( 'renders signal kills without the timeout marker when not timed out' , ( ) = > {
expect ( renderResult ( { . . . base , exitCode : null , signal : 'SIGKILL' } ) )
. toBe ( '(no output)\n[killed by signal: SIGKILL]' )
} )
it ( 'reports a timeout that exited 0 (trapped signal) without a kill marker' , ( ) = > {
expect ( renderResult ( { . . . base , exitCode : 0 , signal : null , timedOut : true } ) )
. toBe ( '(no output)\n[timed out after 1000ms]' )
} )
it ( 'orders the timeout marker before a kill marker' , ( ) = > {
expect ( renderResult ( { . . . base , exitCode : null , signal : 'SIGTERM' , timedOut : true } ) )
. toBe ( '(no output)\n[timed out after 1000ms]\n[killed by signal: SIGTERM]' )
} )
it ( 'notes truncation with a fallback when the spill path is missing' , ( ) = > {
expect ( renderResult ( { . . . base , stdout : { text : 'tail' , truncated : true } } ) )
. toBe ( 'tail\n[output truncated; full output: (unavailable)]' )
} )
} )
describe ( 'status lines' , ( ) = > {
it ( 'reports kills without a recorded signal (executor raced process exit)' , async ( ) = > {
const ctx = await setup ( )
const started = await call ( ctx , 'bash' , { command : 'sleep 60' , description : 'test command' , run_in_background : true } )
2026-06-21 07:17:25 +08:00
const id = BashTaskId ( /task (bash-\d+)/ . exec ( text ( started ) ) ! [ 1 ] ! )
2026-06-12 23:28:44 +08:00
const task = ctx . bash . get ( id ) !
await call ( ctx , 'bash_kill' , { task_id : id } )
await task . done
// Simulate the variant where the close event carried no signal.
task . signal = null
const read = await call ( ctx , 'bash_output' , { task_id : id } )
expect ( text ( read ) ) . toContain ( '[status: killed]' )
} )
it ( 'reports completed tasks with a null exit code as exit 0' , async ( ) = > {
const ctx = await setup ( )
const started = await call ( ctx , 'bash' , { command : 'true' , description : 'test command' , run_in_background : true } )
2026-06-21 07:17:25 +08:00
const id = BashTaskId ( /task (bash-\d+)/ . exec ( text ( started ) ) ! [ 1 ] ! )
2026-06-12 23:28:44 +08:00
const task = ctx . bash . get ( id ) !
await task . done
// Defensive: completed tasks always carry an exit code in practice; the
// ?? 0 fallback covers task shapes from other executor implementations.
task . exitCode = null
const read = await call ( ctx , 'bash_output' , { task_id : id } )
expect ( text ( read ) ) . toContain ( '[status: completed, exit code: 0]' )
} )
} )
2026-06-18 09:01:36 +08:00
describe ( 'tool-owned UI presentation (presentCall / presentResult)' , ( ) = > {
2026-07-03 02:04:03 +08:00
it ( 'bash presentCall: a foreground run is a terminal card (command title, description, workdir → cwd absolute or relative)' , async ( ) = > {
2026-06-18 09:01:36 +08:00
const ctx = await setup ( )
2026-07-03 02:04:03 +08:00
// No explicit workdir → a terminal card with no cwd (the UI bridge fills the
// session cwd it owns; the pure presenter can't see it).
2026-06-18 17:25:09 +08:00
expect ( ctx . tools . get ( 'bash' ) ? . presentCall ? . ( { command : 'ls -la src' , description : 'List files in src' } ) )
2026-07-03 02:04:03 +08:00
. toEqual ( { card : 'terminal' , title : 'ls -la src' , description : 'List files in src' } )
2026-06-18 18:54:32 +08:00
// An ABSOLUTE workdir is surfaced verbatim as the terminal cwd header.
2026-06-18 17:25:09 +08:00
expect ( ctx . tools . get ( 'bash' ) ? . presentCall ? . ( { command : 'pwd' , description : 'Print dir' , workdir : '/tmp/x' } ) )
2026-07-03 02:04:03 +08:00
. toEqual ( { card : 'terminal' , title : 'pwd' , description : 'Print dir' , cwd : '/tmp/x' } )
2026-06-18 18:54:32 +08:00
// A RELATIVE workdir is passed through AS-IS (the bridge resolves it against
// the session cwd, matching where execution runs) — not dropped.
2026-06-18 17:25:09 +08:00
expect ( ctx . tools . get ( 'bash' ) ? . presentCall ? . ( { command : 'pwd' , description : 'Print dir' , workdir : 'sub' } ) )
2026-07-03 02:04:03 +08:00
. toEqual ( { card : 'terminal' , title : 'pwd' , description : 'Print dir' , cwd : 'sub' } )
2026-06-18 09:01:36 +08:00
} )
2026-07-03 02:04:03 +08:00
it ( 'bash presentResult: a terminal result carries RAW output (newlines intact) + parsed exit code' , async ( ) = > {
2026-06-18 09:01:36 +08:00
const ctx = await setup ( )
const present = ctx . tools . get ( 'bash' ) ! . presentResult ! (
{ command : 'echo hi' , description : 'echo' } ,
{ content : [ { type : 'text' , text : 'hi\n[exit code: 0]\n\n' } ] , isError : false } ,
)
2026-07-03 02:04:03 +08:00
// A terminal result keeps the RAW bytes (newlines intact) a terminal renderer
// needs; the bridge derives the fenced fallback. exitCode is parsed back from
// the [exit code: N] marker.
expect ( present ) . toEqual ( { card : 'terminal' , output : 'hi\n[exit code: 0]\n\n' , exitCode : 0 } )
2026-06-18 09:01:36 +08:00
} )
2026-06-18 18:54:32 +08:00
it ( 'bash presentResult: a non-zero exit and a signal kill parse into exitCode / signal' , async ( ) = > {
const ctx = await setup ( )
const args = { command : 'x' , description : 'x' }
const nonzero = ctx . tools . get ( 'bash' ) ! . presentResult ! ( args , { content : [ { type : 'text' , text : 'oops\n[exit code: 3]' } ] , isError : false } )
2026-07-03 02:04:03 +08:00
expect ( nonzero ) . toEqual ( { card : 'terminal' , output : 'oops\n[exit code: 3]' , exitCode : 3 } )
2026-06-18 18:54:32 +08:00
const killed = ctx . tools . get ( 'bash' ) ! . presentResult ! ( args , { content : [ { type : 'text' , text : 'gone\n[killed by signal: SIGKILL]' } ] , isError : false } )
2026-07-03 02:04:03 +08:00
expect ( killed ) . toEqual ( { card : 'terminal' , output : 'gone\n[killed by signal: SIGKILL]' , signal : 'SIGKILL' } )
2026-06-18 18:54:32 +08:00
} )
it ( 'bash presentResult exit parse is the inverse of renderResult markers (round-trip)' , async ( ) = > {
const ctx = await setup ( )
const present = ctx . tools . get ( 'bash' ) !
// For each renderResult outcome, the rendered text fed back through
// presentResult recovers the matching structured exit — the parse and the
// marker emission co-evolve in one file, so this pins the pair.
const base = {
aborted : false ,
timeoutMs : 1000 ,
stdout : { text : 'out' , truncated : false } ,
stderr : { text : '' , truncated : false } ,
}
const cases = [
{ result : { . . . base , exitCode : 0 , signal : null , timedOut : false } , expect : { exitCode : 0 } } ,
{ result : { . . . base , exitCode : 7 , signal : null , timedOut : false } , expect : { exitCode : 7 } } ,
{ result : { . . . base , exitCode : null , signal : 'SIGTERM' as const , timedOut : false } , expect : { signal : 'SIGTERM' } } ,
// A trapped-timeout run that exits 0 has no signal/exit marker → reads as exit 0 (it did exit 0).
{ result : { . . . base , exitCode : 0 , signal : null , timedOut : true } , expect : { exitCode : 0 } } ,
]
for ( const c of cases ) {
const rendered = renderResult ( c . result )
const out = present . presentResult ! ( { command : 'x' , description : 'x' } , { content : [ { type : 'text' , text : rendered } ] , isError : false } )
2026-07-03 02:04:03 +08:00
// Drop card + output; the remaining fields are the parsed exit.
const { card : _c , output : _o , . . . exit } = out as { card : string ; output? : string ; exitCode? : number ; signal? : string }
2026-06-18 18:54:32 +08:00
expect ( exit ) . toEqual ( c . expect )
}
} )
2026-06-18 19:35:15 +08:00
it ( 'bash presentResult: a clean exit-0 whose output ENDS in marker-like text is NOT read as a failure' , async ( ) = > {
const ctx = await setup ( )
const args = { command : 'printf "[exit code: 5]"' , description : 'print' }
// A successful command can print text that looks like a marker. renderResult
// for a clean exit 0 appends NOTHING (and no trailing newline), so the body's
// own tail is `[exit code: 5]`. The parse requires a LEADING newline before
// the marker (renderResult always inserts one before a REAL marker), so this
// no-trailing-newline body is NOT mistaken for a failure → exitCode 0.
const out = ctx . tools . get ( 'bash' ) ! . presentResult ! ( args , { content : [ { type : 'text' , text : '[exit code: 5]' } ] , isError : false } )
2026-07-03 02:04:03 +08:00
expect ( out ) . toEqual ( { card : 'terminal' , output : '[exit code: 5]' , exitCode : 0 } )
2026-06-18 19:35:15 +08:00
// Same for a fake signal marker with no leading newline.
const sig = ctx . tools . get ( 'bash' ) ! . presentResult ! ( args , { content : [ { type : 'text' , text : '[killed by signal: SIGKILL]' } ] , isError : false } )
2026-07-03 02:04:03 +08:00
expect ( sig ) . toEqual ( { card : 'terminal' , output : '[killed by signal: SIGKILL]' , exitCode : 0 } )
2026-06-18 19:35:15 +08:00
} )
2026-07-03 02:04:03 +08:00
it ( 'bash presentCall/presentResult: a run_in_background call is a generic card and its ack carries no exit pill' , async ( ) = > {
2026-06-18 19:35:15 +08:00
const ctx = await setup ( )
2026-07-03 02:04:03 +08:00
// The background start returns a task-id ack, not a streamed run — a generic
// execute card with the command as rawInput and the description as content.
2026-06-18 19:35:15 +08:00
const call = ctx . tools . get ( 'bash' ) ! . presentCall ! ( { command : 'sleep 100' , description : 'wait' , run_in_background : true } )
2026-07-03 02:04:03 +08:00
expect ( call ) . toEqual ( { card : 'generic' , title : 'sleep 100' , kind : 'execute' , rawInput : 'sleep 100' , content : [ { type : 'text' , text : 'wait' } ] } )
// The ack result is a generic fenced-text card — no terminal output / exit pill.
2026-06-18 19:35:15 +08:00
const result = ctx . tools . get ( 'bash' ) ! . presentResult ! (
{ command : 'sleep 100' , description : 'wait' , run_in_background : true } ,
{ content : [ { type : 'text' , text : 'started background task bash-1' } ] , isError : false } ,
)
2026-07-03 02:04:03 +08:00
expect ( result ) . toEqual ( { card : 'generic' , content : [ { type : 'text' , text : '```console\nstarted background task bash-1\n```' } ] } )
2026-06-18 19:35:15 +08:00
} )
2026-07-03 02:04:03 +08:00
it ( 'bash presentResult: an isError result is a generic card (no real process exit to report)' , async ( ) = > {
2026-06-18 19:35:15 +08:00
const ctx = await setup ( )
// A spawn failure / abort has no process exit — the body is an error message,
2026-07-03 02:04:03 +08:00
// not renderResult output, so a generic fenced card, no terminal output/exit.
2026-06-18 19:35:15 +08:00
const out = ctx . tools . get ( 'bash' ) ! . presentResult ! (
{ command : 'x' , description : 'x' } ,
{ content : [ { type : 'text' , text : 'command aborted' } ] , isError : true } ,
)
2026-07-03 02:04:03 +08:00
expect ( out ) . toEqual ( { card : 'generic' , content : [ { type : 'text' , text : '```console\ncommand aborted\n```' } ] } )
2026-06-18 19:35:15 +08:00
} )
2026-06-18 09:01:36 +08:00
it ( 'bash presentResult: leaves a non-text (unexpected) result untouched → undefined (UI keeps raw content)' , async ( ) = > {
const ctx = await setup ( )
const present = ctx . tools . get ( 'bash' ) ! . presentResult ! (
{ command : 'x' , description : 'x' } ,
2026-07-04 17:21:13 +08:00
{ content : [ { type : 'reasoning' , text : 'unexpected' } ] , isError : false } ,
2026-06-18 09:01:36 +08:00
)
expect ( present ) . toBeUndefined ( )
} )
it ( 'bash presentResult: a result that is not exactly one block → undefined (no single text to fence)' , async ( ) = > {
const ctx = await setup ( )
const args = { command : 'x' , description : 'x' }
// Empty content (no block) and multi-block content both fall through.
expect ( ctx . tools . get ( 'bash' ) ! . presentResult ! ( args , { content : [ ] , isError : false } ) ) . toBeUndefined ( )
expect ( ctx . tools . get ( 'bash' ) ! . presentResult ! ( args , {
content : [ { type : 'text' , text : 'a' } , { type : 'text' , text : 'b' } ] ,
isError : false ,
} ) ) . toBeUndefined ( )
} )
it ( 'bash_output / bash_kill presentCall: a readable task-scoped title, task id as rawInput' , async ( ) = > {
const ctx = await setup ( )
expect ( ctx . tools . get ( 'bash_output' ) ! . presentCall ! ( { task_id : 'bash-3' } ) )
2026-07-03 02:04:03 +08:00
. toEqual ( { card : 'generic' , title : 'Read output from background task bash-3' , kind : 'execute' , rawInput : 'bash-3' } )
2026-06-18 09:01:36 +08:00
expect ( ctx . tools . get ( 'bash_kill' ) ! . presentCall ! ( { task_id : 'bash-3' } ) )
2026-07-03 02:04:03 +08:00
. toEqual ( { card : 'generic' , title : 'Kill background task bash-3' , kind : 'execute' , rawInput : 'bash-3' } )
2026-06-18 09:01:36 +08:00
} )
it ( 'presentCall validates softly: malformed args (missing required description) return undefined, never throw' , async ( ) = > {
const ctx = await setup ( )
// defineTool wraps presentCall to soft-validate against the schema and fall
// back to undefined (a generic UI presentation) rather than throwing on the
// display path — it may run on replay of arbitrary logged args. The
// ToolDefinition.presentCall takes `unknown`, so a malformed shape needs no cast.
expect ( ctx . tools . get ( 'bash' ) ? . presentCall ? . ( { command : 'ls' } ) ) . toBeUndefined ( )
} )
} )
2026-06-30 13:52:25 +08:00
2026-07-02 04:08:24 +08:00
describe ( 'the model-facing bash tool builds its request from named args only (no {...args} forward)' , ( ) = > {
2026-06-30 13:52:25 +08:00
/**
* Records every {@link BashExecRequest} the consumer hands to `resolve()`, so a
2026-07-02 04:08:24 +08:00
* test can assert what the model-facing tool DID and DID NOT forward. The `bash`
2026-07-10 11:53:02 +08:00
* tool does not expose trusted-plugin fields (`stdoutMaxBytes`, `stdin`, or
* `env`) as parameters, so it must build its request from named args only and
2026-07-02 04:08:24 +08:00
* never spread unknown tool-call keys into it. This guard's job is to catch a
* future refactor that blindly forwards `...args` — which would silently thread
2026-07-10 11:53:02 +08:00
* model input into the post-scrub `env` merge or per-run capture budget — NOT
* to defend a trust boundary
2026-07-02 04:08:24 +08:00
* (the credential scrub in dsh-bash-local is the security control; see the
* bash-stdin-env RFC). Foreground `run()` returns a canned result; `start()` is
* unused here.
2026-06-30 13:52:25 +08:00
*/
class RecordingBashExecutor extends BashExecutor {
readonly requests : BashExecRequest [ ] = [ ]
resolve ( request : BashExecRequest ) : BashExecSpec {
this . requests . push ( request )
return {
command : request.command ,
workdir : request.workdir ? ? process . cwd ( ) ,
timeoutMs : request.timeoutMs ? ? 0 ,
2026-07-10 11:53:02 +08:00
stdoutMaxBytes : request.stdoutMaxBytes ? ? 64 _000 ,
2026-06-30 13:52:25 +08:00
. . . request . signal ? { signal : request.signal } : { } ,
. . . request . stdin !== undefined ? { stdin : request.stdin } : { } ,
. . . request . env !== undefined ? { env : request.env } : { } ,
owner : request.owner ,
2026-07-09 16:05:44 +08:00
sandboxMode : request.sandboxMode ,
2026-06-30 13:52:25 +08:00
}
}
run ( ) : Promise < BashRunResult > {
return Promise . resolve ( {
exitCode : 0 , signal : null , timedOut : false , aborted : false , timeoutMs : 0 ,
stdout : { text : 'ok' , truncated : false } , stderr : { text : '' , truncated : false } ,
} )
}
start ( ) : BashTask { throw new Error ( 'unused' ) }
get ( ) : BashTask | undefined { return undefined }
ownerOf ( ) : OwnerToken | undefined { return undefined }
list ( ) : BashTask [ ] { return [ ] }
readOutput ( ) : BashTaskRead { throw new Error ( 'unused' ) }
kill ( ) : boolean { return false }
}
async function setupRecording() {
const ctx = new Context ( )
await ctx . plugin ( SystemPrompt )
await ctx . plugin ( ToolRegistry )
await ctx . plugin ( AgentRegistry )
await ctx . plugin ( RecordingBashExecutor )
await ctx . plugin ( ToolBash )
return { ctx , bash : ctx.bash as RecordingBashExecutor }
}
2026-07-10 11:53:02 +08:00
it ( 'does not forward trusted-only fields even when the model includes them as extra arguments' , async ( ) = > {
2026-06-30 13:52:25 +08:00
const { ctx , bash } = await setupRecording ( )
2026-07-10 11:53:02 +08:00
// Extra args: the model includes trusted-plugin keys hoping they reach the
2026-07-02 04:08:24 +08:00
// executor. The bash tool's schema ignores unknown keys, and execute() builds
// the request from only command/workdir/timeoutMs/signal — so the recorded
2026-07-10 11:53:02 +08:00
// request carries NONE. (Not a security wall — the model could set an env
2026-07-02 04:08:24 +08:00
// var or feed stdin via shell syntax anyway; this just keeps the request
// shape honest so a future `...args` spread can't silently forward input.)
2026-06-30 13:52:25 +08:00
await ctx . tools . execute ( {
2026-07-02 04:30:40 +08:00
callId : CallId ( 'no-forward-1' ) ,
2026-06-30 13:52:25 +08:00
name : 'bash' ,
arguments : {
command : 'echo hi' ,
description : 'echo' ,
env : { SNEAKY_API_KEY : 'leak' } ,
stdin : 'malicious payload' ,
2026-07-10 11:53:02 +08:00
stdoutMaxBytes : 999_999 ,
2026-06-30 13:52:25 +08:00
} ,
} )
expect ( bash . requests ) . toHaveLength ( 1 )
const request = bash . requests [ 0 ] !
expect ( request . command ) . toBe ( 'echo hi' )
expect ( 'env' in request ) . toBe ( false )
expect ( 'stdin' in request ) . toBe ( false )
2026-07-10 11:53:02 +08:00
expect ( 'stdoutMaxBytes' in request ) . toBe ( false )
2026-06-30 13:52:25 +08:00
} )
2026-07-10 11:53:02 +08:00
it ( 'a background bash call likewise carries no trusted-only fields' , async ( ) = > {
2026-06-30 13:52:25 +08:00
const { ctx , bash } = await setupRecording ( )
// start() throws in this recorder, but resolve() runs first and records the
2026-07-02 04:30:40 +08:00
// request — which is all this no-forward assertion needs.
2026-06-30 13:52:25 +08:00
await ctx . tools . execute ( {
2026-07-02 04:30:40 +08:00
callId : CallId ( 'no-forward-2' ) ,
2026-06-30 13:52:25 +08:00
name : 'bash' ,
arguments : {
command : 'sleep 1' ,
description : 'sleep' ,
run_in_background : true ,
env : { TOKEN : 'leak' } ,
stdin : 'x' ,
2026-07-10 11:53:02 +08:00
stdoutMaxBytes : 999_999 ,
2026-06-30 13:52:25 +08:00
} ,
} )
expect ( bash . requests ) . toHaveLength ( 1 )
const request = bash . requests [ 0 ] !
expect ( 'env' in request ) . toBe ( false )
expect ( 'stdin' in request ) . toBe ( false )
2026-07-10 11:53:02 +08:00
expect ( 'stdoutMaxBytes' in request ) . toBe ( false )
2026-06-30 13:52:25 +08:00
// The owner token IS set on a background call (the isolation fence) — proving
// the recorder sees the real request the consumer built, so the absent
2026-07-10 11:53:02 +08:00
// trusted-only fields above are a real negative, not a recorder that drops everything.
2026-06-30 13:52:25 +08:00
expect ( 'owner' in request ) . toBe ( true )
} )
} )
2026-07-09 16:05:44 +08:00
describe ( 'sandbox rendering' , ( ) = > {
const sandboxResult = ( denied : boolean , exitCode : number ) : BashRunResult = > ( {
exitCode ,
signal : null ,
timedOut : false ,
aborted : false ,
timeoutMs : 1000 ,
stdout : { text : '' , truncated : false } ,
stderr : { text : denied ? 'bash: /x: Read-only file system' : 'boom' , truncated : false } ,
sandbox : { mode : 'read-only' , denied } ,
} )
it ( 'renders a denial marker BEFORE the exit-code marker (the $-anchored parse survives)' , ( ) = > {
const text = renderResult ( sandboxResult ( true , 1 ) )
expect ( text ) . toMatch ( /\[sandbox: file access denied under read-only mode\]\n\[exit code: 1\]$/ )
} )
2026-07-09 16:37:10 +08:00
it ( 'appends the same-turn escalation hint to a denial exactly when the fields are advertised' , ( ) = > {
const hinted = renderResult ( sandboxResult ( true , 1 ) , [ 'workspace-write' , 'danger-full-access' ] )
expect ( hinted ) . toMatch (
/denied under read-only mode\]\n\[sandbox: escalation available — retry this exact command once with sandbox_permissions [^\n]+\]\n\[exit code: 1\]$/ , // eslint-disable-line @stylistic/max-len -- the hint sentence is pinned verbatim
)
// Default (no advertisement): no hint — a lever the schema does not offer is never suggested.
expect ( renderResult ( sandboxResult ( true , 1 ) ) ) . not . toContain ( 'escalation available' )
// A non-denied result never hints, advertised or not.
expect ( renderResult ( sandboxResult ( false , 2 ) , [ 'danger-full-access' ] ) ) . not . toContain ( 'escalation available' )
} )
2026-07-09 16:05:44 +08:00
it ( 'renders no sandbox marker for a plain failure under a sandboxed mode' , ( ) = > {
expect ( renderResult ( sandboxResult ( false , 2 ) ) ) . not . toContain ( '[sandbox:' )
} )
it ( 'bash_output reports a settled background denial with the same marker' , async ( ) = > {
const ctx = new Context ( )
await ctx . plugin ( SystemPrompt )
await ctx . plugin ( ToolRegistry )
await ctx . plugin ( AgentRegistry )
2026-07-11 21:37:38 +08:00
await ctx . plugin ( LocalSandboxProvider , PASSTHROUGH_RUNNER_CONFIG )
2026-07-09 16:05:44 +08:00
await ctx . plugin ( SandboxBashExecutor , { graceMs : 200 } )
const bash = ctx . bash as SandboxBashExecutor
bash . internals = { spillDir }
await ctx . plugin ( ToolBash )
const started = await call ( ctx , 'bash' , { command : 'echo "x: Permission denied" >&2; exit 1' , description : 'test command' , run_in_background : true } )
const id = text ( started ) . match ( /started background task (bash-\d+)/ ) ! [ 1 ]
await bash . list ( ) . find ( task = > task . id === id ) ! . done
const read = await call ( ctx , 'bash_output' , { task_id : id } )
2026-07-09 16:37:10 +08:00
expect ( text ( read ) ) . toMatch (
/\[status: completed, exit code: 1\]\n\[sandbox: file access denied under read-only mode\]\n\[sandbox: escalation available[^\n]+\]$/ ,
)
} )
it ( 'a settled background denial renders no escalation hint without a confining executor (defensive arm)' , async ( ) = > {
// Structurally near-unreachable through the real stack — every confining
// default advertises the static target set — but the read path guards
// it anyway: an executor that reports no sandboxMode (fields never
// advertised) whose task nonetheless carries denial facts must render
// the marker without suggesting a lever the schema does not offer.
2026-07-13 23:42:54 +08:00
class FactsOnlyExecutor extends TestBashExecutor {
2026-07-09 16:37:10 +08:00
private readonly task : BashTask = {
id : BashTaskId ( 'bash-facts' ) ,
command : 'fake' ,
status : 'completed' ,
exitCode : 1 ,
signal : null ,
done : Promise.resolve ( ) ,
sandbox : { mode : 'read-only' , denied : true } ,
}
run ( ) : Promise < BashRunResult > { return Promise . reject ( new Error ( 'not used' ) ) }
start ( ) : BashTask { return this . task }
get ( id : string ) : BashTask | undefined { return id === this . task . id ? this . task : undefined }
list ( ) : BashTask [ ] { return [ this . task ] }
kill ( ) : boolean { return false }
ownerOf ( ) : OwnerToken | undefined { return undefined }
readOutput ( ) : BashTaskRead {
return { task : this.task , delta : '' , lossy : false }
}
}
const ctx = new Context ( )
await ctx . plugin ( SystemPrompt )
await ctx . plugin ( ToolRegistry )
await ctx . plugin ( AgentRegistry )
await ctx . plugin ( FactsOnlyExecutor )
await ctx . plugin ( ToolBash )
const read = await call ( ctx , 'bash_output' , { task_id : 'bash-facts' } )
expect ( text ( read ) ) . toMatch ( /\[sandbox: file access denied under read-only mode\]$/ )
expect ( text ( read ) ) . not . toContain ( 'escalation available' )
2026-07-09 16:05:44 +08:00
} )
it ( 'bash_output reports a settled background RUNNER failure as a sandbox problem, outranking the denial marker' , async ( ) = > {
// A provider whose wrap carries a runner-failure signature: the settled
// task's stderr matching it means the sandbox itself broke and the
// command never ran — even though the same stderr also carries denial
// words (a runner's error text may contain them).
class FakeProvider extends SandboxProvider {
confine ( argv : readonly string [ ] ) : ConfinedArgv {
return { argv : [ . . . argv ] , enforcement : 'full' , denialSignatures : [ 'permission denied' ] , runnerFailureSignatures : [ 'fake-runner: ' ] }
}
}
const ctx = new Context ( )
await ctx . plugin ( SystemPrompt )
await ctx . plugin ( ToolRegistry )
await ctx . plugin ( AgentRegistry )
await ctx . plugin ( FakeProvider )
await ctx . plugin ( SandboxBashExecutor , { graceMs : 200 } )
const bash = ctx . bash as SandboxBashExecutor
bash . internals = { spillDir }
await ctx . plugin ( ToolBash )
const started = await call ( ctx , 'bash' , { command : 'echo "fake-runner: cannot open rule path: /x: Permission denied" >&2; exit 125' , description : 'test command' , run_in_background : true } )
const id = text ( started ) . match ( /started background task (bash-\d+)/ ) ! [ 1 ]
await bash . list ( ) . find ( task = > task . id === id ) ! . done
const read = await call ( ctx , 'bash_output' , { task_id : id } )
expect ( text ( read ) ) . toMatch ( /\[sandbox: the sandbox runner itself failed under read-only mode — the command did not run; / )
expect ( text ( read ) ) . toMatch ( /this is a sandbox problem, not a command failure\]$/ )
expect ( text ( read ) ) . not . toContain ( 'file access denied' )
} )
2026-07-11 21:37:38 +08:00
it ( 'classifies an executable configured runner that refuses its profile before the command runs' , async ( ) = > {
const signature = 'custom-runner-rejected'
const ctx = new Context ( )
await ctx . plugin ( LocalSandboxProvider , {
runnerCommand : [ 'bash' , '-c' , ` printf ' ${ signature } \\ n' >&2; exit 125 ` , 'custom-runner' ] ,
runnerFailureSignatures : [ signature ] ,
} )
await ctx . plugin ( SandboxBashExecutor , { graceMs : 200 } )
const bash = ctx . bash as SandboxBashExecutor
bash . internals = { spillDir }
await expect ( bash . run ( bash . resolve ( { command : 'echo command-must-not-run' } ) ) )
. rejects . toMatchObject ( { code : 'SANDBOX_UNAVAILABLE' } )
const task = bash . start ( bash . resolve ( { command : 'echo command-must-not-run' } ) )
await task . done
expect ( task . sandbox ) . toEqual ( { mode : 'read-only' , denied : false , enforcement : 'full' , runnerFailed : true } )
} )
2026-07-09 16:05:44 +08:00
it ( 'reports a real denial end-to-end through the shipping sandbox executor' , async ( ) = > {
const ctx = new Context ( )
await ctx . plugin ( SystemPrompt )
await ctx . plugin ( ToolRegistry )
await ctx . plugin ( AgentRegistry )
2026-07-11 21:37:38 +08:00
await ctx . plugin ( LocalSandboxProvider , PASSTHROUGH_RUNNER_CONFIG )
2026-07-09 16:05:44 +08:00
await ctx . plugin ( SandboxBashExecutor , { graceMs : 200 } )
const bash = ctx . bash as SandboxBashExecutor
bash . internals = { spillDir }
await ctx . plugin ( ToolBash )
const lockedDir = join ( mkdtempSync ( join ( tmpdir ( ) , 'dsh-tool-bash-denied-' ) ) , 'locked' )
mkdirSync ( lockedDir )
chmodSync ( lockedDir , 0 o555 )
const result = await call ( ctx , 'bash' , { command : ` echo x > ${ lockedDir } /f ` , description : 'Write into a locked directory' } )
expect ( result . isError ) . toBe ( false )
2026-07-09 16:37:10 +08:00
expect ( text ( result ) ) . toMatch (
/denied under read-only mode\]\n\[sandbox: escalation available[^\n]+\]\n\[exit code: \d+\]$/ ,
)
2026-07-09 16:05:44 +08:00
} )
} )
2026-07-09 16:37:10 +08:00
describe ( 'sandbox escalation (sandbox_permissions / justification)' , ( ) = > {
/** Compose the real sandbox stack (passthrough runner) at a given default mode. */
2026-07-09 16:41:03 +08:00
async function setupSandboxed ( mode ? : 'read-only' | 'workspace-write' | 'danger-full-access' , opts : { approval? : boolean ; policy ? : 'ask' | 'never' } = { } ) {
2026-07-09 16:37:10 +08:00
const ctx = new Context ( )
await ctx . plugin ( SystemPrompt )
await ctx . plugin ( ToolRegistry )
await ctx . plugin ( AgentRegistry )
2026-07-11 21:37:38 +08:00
await ctx . plugin ( LocalSandboxProvider , PASSTHROUGH_RUNNER_CONFIG )
2026-07-09 16:37:10 +08:00
await ctx . plugin ( SandboxBashExecutor , { graceMs : 200 , . . . mode !== undefined ? { mode } : { } } )
const bash = ctx . bash as SandboxBashExecutor
bash . internals = { spillDir }
2026-07-09 16:41:03 +08:00
if ( opts . approval === true ) await ctx . plugin ( ApprovalService , opts . policy !== undefined ? { policy : opts.policy } : { } )
2026-07-09 16:37:10 +08:00
await ctx . plugin ( ToolBash )
return { ctx , bash }
}
/** The registered bash tool's wire schema (what the model actually sees). */
function bashSchema ( ctx : Context ) {
const schema = ctx . tools . schemas ( ) . find ( s = > s . name === 'bash' )
if ( ! schema ) throw new Error ( 'bash tool not registered' )
return schema as unknown as { description : string ; parameters : { properties : Record < string , { enum ? : string [ ] } > } }
}
/**
* A fake agent whose session records appends — the approval audit surface.
* Seeded mid-turn: an escalating call always runs inside one, and request()
* enforces the enclosure.
*/
function escalationAgent ( events : Array < { type : string ; data : Record < string , unknown > } > ) : Agent {
return {
id : 'agent-esc' ,
session : {
header : { version : 0 , id : 'sess-esc' , createdAt : 0 } ,
events : [ { type : 'turn/start' } ] ,
append : ( type : string , data : Record < string , unknown > ) = > { events . push ( { type , data } ) } ,
} ,
} as unknown as Agent
}
let escCall = 0
function callAs ( ctx : Context , agent : Agent | undefined , args : unknown ) {
return ctx . tools . execute ( { callId : CallId ( ` call-esc- ${ ++ escCall } ` ) , name : 'bash' , arguments : args , . . . agent ? { agent } : { } } )
}
const ESCALATE = { command : 'true' , description : 'test escalation' , sandbox_permissions : 'workspace-write' , justification : 'the test needs it' }
it ( 'advertises no escalation surface under a non-sandboxing executor' , async ( ) = > {
const ctx = await setup ( )
expect ( ctx . bash . sandboxMode ) . toBeUndefined ( )
const schema = bashSchema ( ctx )
expect ( schema . parameters . properties [ 'sandbox_permissions' ] ) . toBeUndefined ( )
expect ( schema . parameters . properties [ 'justification' ] ) . toBeUndefined ( )
expect ( schema . description ) . not . toContain ( 'sanctioned exception' )
} )
it ( 'advertises the full closed target vocabulary under any confining default' , async ( ) = > {
// The enum is deliberately NOT default-relative: a session's effective
// mode is per-session and switchable, so every confining composition
// advertises every possible target — strict widening is checked at
// execution against the call's effective mode instead.
for ( const mode of [ undefined , 'workspace-write' , 'danger-full-access' ] as const ) {
const { ctx } = await setupSandboxed ( mode )
const schema = bashSchema ( ctx )
expect ( schema . parameters . properties [ 'sandbox_permissions' ] ? . enum ) . toEqual ( [ 'workspace-write' , 'danger-full-access' ] )
expect ( schema . parameters . properties [ 'justification' ] ) . toBeDefined ( )
expect ( schema . description ) . toContain ( 'sanctioned exception' )
}
} )
it ( 'a non-widening request fails at execution with its own text and prompts no one' , async ( ) = > {
const { ctx } = await setupSandboxed ( 'danger-full-access' , { approval : true } )
const consulted = vi . fn ( )
ctx . on ( 'approval/request' , ( _req , next ) = > { consulted ( ) ; return next ( ) } )
const result = await callAs ( ctx , escalationAgent ( [ ] ) , { command : 'true' , description : 'd' , sandbox_permissions : 'workspace-write' , justification : 'already wider' } )
expect ( result . isError ) . toBe ( true )
expect ( text ( result ) ) . toContain ( 'not strictly wider than this call\'s current "danger-full-access" mode' )
expect ( consulted ) . not . toHaveBeenCalled ( )
} )
it ( 'rejects sandbox_permissions without a justification, and vice versa, and a blank justification' , async ( ) = > {
const { ctx } = await setupSandboxed ( )
const missing = await callAs ( ctx , undefined , { command : 'true' , description : 'd' , sandbox_permissions : 'workspace-write' } )
expect ( missing . isError ) . toBe ( true )
expect ( text ( missing ) ) . toContain ( 'sandbox_permissions requires a justification' )
const orphan = await callAs ( ctx , undefined , { command : 'true' , description : 'd' , justification : 'why not' } )
expect ( orphan . isError ) . toBe ( true )
expect ( text ( orphan ) ) . toContain ( 'only valid together with sandbox_permissions' )
const blank = await callAs ( ctx , undefined , { command : 'true' , description : 'd' , sandbox_permissions : 'workspace-write' , justification : ' ' } )
expect ( blank . isError ) . toBe ( true )
expect ( text ( blank ) ) . toContain ( 'expected a non-empty sentence' )
} )
it ( 'the schema enum rejects a mode outside the target vocabulary before execute (registry-level, any caller)' , async ( ) = > {
const { ctx } = await setupSandboxed ( )
const result = await callAs ( ctx , undefined , { command : 'true' , description : 'd' , sandbox_permissions : 'read-only' , justification : 'narrow' } )
expect ( result . isError ) . toBe ( true )
expect ( text ( result ) ) . toContain ( 'must be one of' )
} )
it ( 'rejects an unadvertised sandbox_permissions injection under a non-sandboxing executor' , async ( ) = > {
const ctx = await setup ( )
const result = await callAs ( ctx , undefined , { command : 'true' , description : 'd' , sandbox_permissions : 'workspace-write' , justification : 'sneaky' } )
expect ( result . isError ) . toBe ( true )
expect ( text ( result ) ) . toContain ( 'not available in this composition' )
} )
it ( 'fails closed with its own text when no approval service is composed' , async ( ) = > {
const { ctx } = await setupSandboxed ( )
const result = await callAs ( ctx , escalationAgent ( [ ] ) , ESCALATE )
expect ( result . isError ) . toBe ( true )
expect ( text ( result ) ) . toContain ( 'no approval service is composed' )
} )
it ( 'fails closed with its own text for an agent-less escalating call' , async ( ) = > {
const { ctx } = await setupSandboxed ( 'read-only' , { approval : true } )
const result = await callAs ( ctx , undefined , ESCALATE )
expect ( result . isError ) . toBe ( true )
expect ( text ( result ) ) . toContain ( 'no agent to route it through' )
} )
it ( 'fails closed with its own text when the service has no answerer' , async ( ) = > {
const { ctx } = await setupSandboxed ( 'read-only' , { approval : true } )
const result = await callAs ( ctx , escalationAgent ( [ ] ) , ESCALATE )
expect ( result . isError ) . toBe ( true )
expect ( text ( result ) ) . toContain ( 'no approval channel is available' )
} )
it ( 'a grant runs THAT call under the wider mode — the denial marker names it — and lands the audit pair' , async ( ) = > {
const { ctx } = await setupSandboxed ( 'read-only' , { approval : true } )
ctx . on ( 'approval/request' , ( ) = > Promise . resolve < ApprovalOutcome > ( 'allowed-once' ) )
const events : Array < { type : string ; data : Record < string , unknown > } > = [ ]
// A real unix denial under the passthrough runner: the marker's mode can
// only say workspace-write if the override actually rode the spec.
const lockedDir = join ( mkdtempSync ( join ( tmpdir ( ) , 'dsh-esc-denied-' ) ) , 'locked' )
mkdirSync ( lockedDir )
chmodSync ( lockedDir , 0 o555 )
const result = await callAs ( ctx , escalationAgent ( events ) , {
command : ` echo x > ${ lockedDir } /f ` ,
description : 'write into a locked directory' ,
sandbox_permissions : 'workspace-write' ,
justification : 'must write outside the workspace' ,
} )
expect ( result . isError ) . toBe ( false )
expect ( text ( result ) ) . toMatch ( /\[sandbox: file access denied under workspace-write mode\]/ )
expect ( events . map ( e = > e . type ) ) . toEqual ( [ 'approval/asked' , 'approval/decided' ] )
expect ( events [ 0 ] ? . data [ 'toolName' ] ) . toBe ( 'bash' )
expect ( events [ 0 ] ? . data [ 'reason' ] ) . toBe ( 'escalate sandbox to workspace-write: must write outside the workspace' )
expect ( events [ 1 ] ? . data [ 'outcome' ] ) . toBe ( 'allowed-once' )
} )
it ( 'a granted background start settles with the wider mode\'s facts' , async ( ) = > {
const { ctx , bash } = await setupSandboxed ( 'read-only' , { approval : true } )
ctx . on ( 'approval/request' , ( ) = > Promise . resolve < ApprovalOutcome > ( 'allowed-once' ) )
const started = await callAs ( ctx , escalationAgent ( [ ] ) , { . . . ESCALATE , run_in_background : true } )
expect ( started . isError ) . toBe ( false )
const id = text ( started ) . match ( /started background task (bash-\d+)/ ) ? . [ 1 ]
const task = bash . list ( ) . find ( t = > t . id === id )
if ( ! task ) throw new Error ( 'escalated task not tracked' )
await task . done
expect ( task . sandbox ) . toMatchObject ( { mode : 'workspace-write' , denied : false } )
} )
it ( 'a rejection denies with the user-said-no text and runs nothing' , async ( ) = > {
const { ctx } = await setupSandboxed ( 'read-only' , { approval : true } )
ctx . on ( 'approval/request' , ( ) = > Promise . resolve < ApprovalOutcome > ( 'rejected' ) )
// A live (non-aborted) signal rides the execution: the gate threads it
// into the approval request so a turn cancellation can withdraw the ask.
const result = await ctx . tools . execute ( {
callId : CallId ( ` call-esc- ${ ++ escCall } ` ) ,
name : 'bash' ,
arguments : ESCALATE ,
agent : escalationAgent ( [ ] ) ,
signal : new AbortController ( ) . signal ,
} )
expect ( result . isError ) . toBe ( true )
expect ( text ( result ) ) . toContain ( 'the user rejected escalating this command to "workspace-write"' )
} )
it ( 'a cancellation denies with the cancelled text' , async ( ) = > {
const { ctx } = await setupSandboxed ( 'read-only' , { approval : true } )
ctx . on ( 'approval/request' , ( ) = > Promise . resolve < ApprovalOutcome > ( 'cancelled' ) )
const result = await callAs ( ctx , escalationAgent ( [ ] ) , ESCALATE )
expect ( result . isError ) . toBe ( true )
expect ( text ( result ) ) . toContain ( 'approval for escalating to "workspace-write" was cancelled' )
} )
it ( 'a rogue approval stand-in returning a non-vocabulary outcome hits the exhaustiveness backstop' , async ( ) = > {
const { ctx } = await setupSandboxed ( )
ctx . provide ( 'approval' , { request : ( ) = > Promise . resolve ( 'yolo' ) } as unknown as InstanceType < typeof ApprovalService > )
const result = await callAs ( ctx , escalationAgent ( [ ] ) , ESCALATE )
expect ( result . isError ) . toBe ( true )
expect ( text ( result ) ) . toContain ( 'unreachable' )
} )
2026-07-09 16:41:03 +08:00
it ( 'a never policy rejects an escalation deterministically without consulting any answerer' , async ( ) = > {
// The live-session e.md case: the model requests escalation against a
// 'never' session — the prepend gate answers rejected before any
// interactive answerer, the fail-closed text is the ordinary rejection
// wording, and the audit pair still lands.
const { ctx } = await setupSandboxed ( 'read-only' , { approval : true , policy : 'never' } )
const consulted = vi . fn ( )
ctx . on ( 'approval/request' , ( _req , next ) = > { consulted ( ) ; return next ( ) } )
const events : Array < { type : string ; data : Record < string , unknown > } > = [ ]
const result = await callAs ( ctx , escalationAgent ( events ) , ESCALATE )
expect ( result . isError ) . toBe ( true )
expect ( text ( result ) ) . toContain ( 'the user rejected escalating this command to "workspace-write"' )
expect ( consulted ) . not . toHaveBeenCalled ( )
expect ( events . map ( e = > e . type ) ) . toEqual ( [ 'approval/asked' , 'approval/decided' ] )
expect ( events [ 1 ] ? . data ) . toMatchObject ( { outcome : 'rejected' } )
} )
2026-07-09 16:37:10 +08:00
it ( 'a plain call under a sandboxing executor never consults approval' , async ( ) = > {
const { ctx } = await setupSandboxed ( 'read-only' , { approval : true } )
const asked = vi . fn ( )
ctx . on ( 'approval/request' , ( _req , next ) = > { asked ( ) ; return next ( ) } )
const result = await callAs ( ctx , escalationAgent ( [ ] ) , { command : 'echo plain' , description : 'plain run' } )
expect ( result . isError ) . toBe ( false )
expect ( text ( result ) ) . toContain ( 'plain' )
expect ( asked ) . not . toHaveBeenCalled ( )
} )
} )
2026-07-09 16:41:03 +08:00
describe ( 'per-session sandbox mode (the bash/sandbox-mode fold)' , ( ) = > {
/** Compose the real sandbox stack (passthrough runner) at a given default mode. */
async function setupModal ( mode : 'read-only' | 'workspace-write' | 'danger-full-access' = 'read-only' , opts : { approval? : boolean } = { } ) {
const ctx = new Context ( )
await ctx . plugin ( SystemPrompt )
await ctx . plugin ( ToolRegistry )
await ctx . plugin ( AgentRegistry )
2026-07-11 21:37:38 +08:00
await ctx . plugin ( LocalSandboxProvider , PASSTHROUGH_RUNNER_CONFIG )
2026-07-09 16:41:03 +08:00
await ctx . plugin ( SandboxBashExecutor , { graceMs : 200 , mode } )
; ( ctx . bash as SandboxBashExecutor ) . internals = { spillDir }
if ( opts . approval === true ) await ctx . plugin ( ApprovalService )
await ctx . plugin ( ToolBash )
return ctx
}
/**
* An agent stand-in over a REAL Session — the stamping folds real events;
* the opened turn satisfies approval's enclosure precondition on escalating
* calls.
*/
function sessionAgent ( id : string ) : { agent : Agent ; session : Session ; injected : string [ ] } {
const session = new Session ( SessionId ( id ) )
session . append ( 'turn/start' , { turn : 1 , trigger : { kind : 'message' , source : { kind : 'user' } } } )
const injected : string [ ] = [ ]
const agent = {
id ,
session ,
inject : ( content : { type : string ; text : string } [ ] ) = > { injected . push ( content [ 0 ] ? . text ? ? '' ) } ,
} as unknown as Agent
return { agent , session , injected }
}
let modeCall = 0
const callAs = ( ctx : Context , agent : Agent | undefined , args : unknown ) = >
ctx . tools . execute ( { callId : CallId ( ` call-mode- ${ ++ modeCall } ` ) , name : 'bash' , arguments : args , . . . agent ? { agent } : { } } )
it ( 'stamps calls with grant > session override > nothing (executor default)' , async ( ) = > {
const ctx = await setupModal ( 'read-only' , { approval : true } )
ctx . on ( 'approval/request' , ( ) = > Promise . resolve < ApprovalOutcome > ( 'allowed-once' ) )
const seen : ( string | undefined ) [ ] = [ ]
const original = ctx . bash . resolve . bind ( ctx . bash )
vi . spyOn ( ctx . bash , 'resolve' ) . mockImplementation ( ( req ) = > {
seen . push ( req . sandboxMode )
return original ( req )
} )
const { agent , session } = sessionAgent ( 'sess-stamp-1' )
const run = { command : 'true' , description : 'stamp probe' }
await callAs ( ctx , agent , run ) // no override yet
setSandboxMode ( session , 'workspace-write' )
await callAs ( ctx , agent , run ) // standing override
await callAs ( ctx , undefined , run ) // agent-less caller: no session to fold
await callAs ( ctx , agent , { . . . run , sandbox_permissions : 'danger-full-access' , justification : 'grant outranks override' } )
expect ( seen ) . toEqual ( [ undefined , 'workspace-write' , undefined , 'danger-full-access' ] )
} )
it ( 'escalates relative to the session effective mode, not the executor default (narrower override)' , async ( ) = > {
// The blocker scenario: a workspace-write default with a read-only
// override — the sensible escalation is workspace-write, which a
// default-relative ladder could not even express. The static target
// vocabulary advertises it and the execution check accepts it as
// strictly wider than the CALL's effective (overridden) mode.
const ctx = await setupModal ( 'workspace-write' , { approval : true } )
ctx . on ( 'approval/request' , ( ) = > Promise . resolve < ApprovalOutcome > ( 'allowed-once' ) )
const seen : ( string | undefined ) [ ] = [ ]
const original = ctx . bash . resolve . bind ( ctx . bash )
vi . spyOn ( ctx . bash , 'resolve' ) . mockImplementation ( ( req ) = > {
seen . push ( req . sandboxMode )
return original ( req )
} )
const { agent , session } = sessionAgent ( 'sess-esc-narrow' )
setSandboxMode ( session , 'read-only' )
const result = await callAs ( ctx , agent , { command : 'true' , description : 'd' , sandbox_permissions : 'workspace-write' , justification : 'the override is narrower than the default' } )
expect ( result . isError ) . toBe ( false )
expect ( seen ) . toEqual ( [ 'workspace-write' ] )
} )
it ( 'a danger-full-access default still offers the lever to a narrower-switched session' , async ( ) = > {
// Under the default-relative ladder these fields VANISHED (nothing is
// wider than the default), stranding a read-only-overridden session
// with no escalation path at all.
const ctx = await setupModal ( 'danger-full-access' , { approval : true } )
ctx . on ( 'approval/request' , ( ) = > Promise . resolve < ApprovalOutcome > ( 'allowed-once' ) )
const schema = ctx . tools . schemas ( ) . find ( t = > t . name === 'bash' ) as unknown as { parameters : { properties : Record < string , { enum ? : string [ ] } > } }
expect ( schema . parameters . properties [ 'sandbox_permissions' ] ? . enum ) . toEqual ( [ 'workspace-write' , 'danger-full-access' ] )
const { agent , session } = sessionAgent ( 'sess-esc-dfa' )
setSandboxMode ( session , 'read-only' )
const result = await callAs ( ctx , agent , { command : 'true' , description : 'd' , sandbox_permissions : 'workspace-write' , justification : 'confined by override under a wide default' } )
expect ( result . isError ) . toBe ( false )
} )
it ( 'rejects a non-widening request against the OVERRIDDEN effective mode without prompting' , async ( ) = > {
const ctx = await setupModal ( 'read-only' , { approval : true } )
const consulted = vi . fn ( )
ctx . on ( 'approval/request' , ( _req , next ) = > { consulted ( ) ; return next ( ) } )
const { agent , session } = sessionAgent ( 'sess-esc-nonwide' )
setSandboxMode ( session , 'danger-full-access' )
const result = await callAs ( ctx , agent , { command : 'true' , description : 'd' , sandbox_permissions : 'workspace-write' , justification : 'already wider via override' } )
expect ( result . isError ) . toBe ( true )
expect ( text ( result ) ) . toContain ( 'not strictly wider than this call\'s current "danger-full-access" mode' )
expect ( consulted ) . not . toHaveBeenCalled ( )
} )
it ( 'never stamps an override under a non-sandboxing executor (nothing honors it)' , async ( ) = > {
const ctx = await setup ( )
const seen : ( string | undefined ) [ ] = [ ]
const original = ctx . bash . resolve . bind ( ctx . bash )
vi . spyOn ( ctx . bash , 'resolve' ) . mockImplementation ( ( req ) = > {
seen . push ( req . sandboxMode )
return original ( req )
} )
const { agent , session } = sessionAgent ( 'sess-stamp-2' )
setSandboxMode ( session , 'danger-full-access' )
await callAs ( ctx , agent , { command : 'true' , description : 'plain probe' } )
expect ( seen ) . toEqual ( [ undefined ] )
} )
} )