2026-07-28 13:55:59 +08:00
import { createUserMessage } from '@deepseek-ai/dsh-llm'
2026-07-06 03:07:34 +08:00
import { afterEach , describe , expect , it } from 'vitest'
2026-08-10 22:04:06 +08:00
import { Context } from '@deepseek-ai/cordis'
2026-08-13 00:36:22 +08:00
import LlmRuntime from '@deepseek-ai/dsh-llm'
2026-07-14 01:59:21 +08:00
import SessionStore , { SessionId } from '@deepseek-ai/dsh-session'
2026-07-06 03:07:34 +08:00
import SystemPrompt from '@deepseek-ai/dsh-system-prompt'
2026-08-13 00:36:22 +08:00
import ToolRuntime , { defineContentToolFixture } from '@deepseek-ai/dsh-tools'
2026-07-14 01:59:21 +08:00
import AgentRegistry , { type Agent } from '@deepseek-ai/dsh-agent'
2026-07-06 03:07:34 +08:00
import AgentLoop from '@deepseek-ai/dsh-agent-loop'
import * as LlmDeepSeek from '@deepseek-ai/dsh-llm-deepseek'
/**
2026-07-12 03:36:43 +08:00
* With-key proof that log-derived requests translate into real provider cache hits: a
* multi-step tool turn (plus a follow-up turn) against the live DeepSeek API must report
* `cacheReadTokens > 0` on every request after the first — the adapter maps the provider's
* `prompt_cache_hit_tokens`, and the per-step usage recorded on `assistant/message` events is
2026-07-19 22:50:49 +08:00
* the production observable for cache behavior (the reconstructability Agent Note's measurement
2026-07-13 23:27:00 +08:00
* layer: prefix stability is corollary #1). Mocks establish append-extension;
* this key-gated test establishes a real provider cache hit.
2026-07-06 03:07:34 +08:00
*/
// Long enough that the shared request prefix comfortably spans the provider's
// cache-block granularity (64 tokens) from the very first request.
const SYSTEM = 'You are a terse coding assistant used in an automated cache test. '
+ 'Always follow instructions literally and exactly. When the user asks you to look '
+ 'something up, call the lookup tool with the requested key and wait for its result '
+ 'before answering. Never invent a value the tool has not returned. After the tool '
+ 'returns, answer with a single short sentence that repeats the returned value '
+ 'verbatim. Do not add explanations, do not use markdown, do not ask follow-up '
+ 'questions. If the user asks anything else, answer in one short sentence.'
let ctx : Context | undefined
afterEach ( async ( ) = > {
await ctx ? . fiber . dispose ( )
ctx = undefined
} )
async function loopHarness ( ) : Promise < Context > {
const created = new Context ( )
2026-08-13 00:36:22 +08:00
await created . plugin ( LlmRuntime )
2026-07-06 03:07:34 +08:00
await created . plugin ( SessionStore )
await created . plugin ( SystemPrompt , { persona : SYSTEM } )
2026-08-13 00:36:22 +08:00
await created . plugin ( ToolRuntime )
2026-07-06 03:07:34 +08:00
await created . plugin ( AgentRegistry )
await created . plugin ( AgentLoop , { agents : [ ] } )
2026-07-14 21:57:52 +08:00
await created . plugin ( LlmDeepSeek )
2026-07-21 03:08:35 +08:00
created . tools . register ( defineContentToolFixture ( {
2026-07-06 03:07:34 +08:00
name : 'lookup' ,
description : 'Look up the stored value for a key.' ,
parameters : { key : { type : 'string' , description : 'The key to look up.' } } ,
async execute ( args ) {
return [ { type : 'text' , text : ` value( ${ String ( args . key ) } ) = azure-falcon-42 ` } ]
} ,
} ) )
return created
}
function waitForIdle ( context : Context , agent : Agent ) : Promise < void > {
return new Promise ( ( resolve ) = > {
2026-08-06 12:13:14 +08:00
const dispose = context . on ( 'agent/status' , ( { agent : subject , status } ) = > {
2026-07-06 03:07:34 +08:00
if ( subject === agent && status === 'idle' ) {
dispose ( )
resolve ( )
}
} )
} )
}
describe . skipIf ( ! process . env . DEEPSEEK_API_KEY ) ( 'log-derived request cache hits (real API)' , ( ) = > {
it ( 'every request after the first hits the provider prefix cache' , async ( ) = > {
ctx = await loopHarness ( )
2026-07-29 16:36:07 +08:00
const agent = ctx . agentLoop . create ( SessionId ( 'cache-e2e' ) , { provider : 'deepseek-official' , model : 'deepseek-v4-flash' } )
2026-07-06 03:07:34 +08:00
// Turn 1: forces a tool call → at least two steps (two model requests).
2026-07-28 13:55:59 +08:00
agent . followup ( createUserMessage ( { content : [ { type : 'text' , text : 'Look up the key "deploy-color" with the lookup tool and tell me the value.' } ] , source : { kind : 'user' } } ) )
2026-07-06 03:07:34 +08:00
await waitForIdle ( ctx , agent )
// Turn 2: a follow-up over the same (longer) prefix.
2026-07-28 13:55:59 +08:00
agent . followup ( createUserMessage ( { content : [ { type : 'text' , text : 'Thanks. Repeat that value one more time.' } ] , source : { kind : 'user' } } ) )
2026-07-06 03:07:34 +08:00
await waitForIdle ( ctx , agent )
const usages = [ . . . agent . session . events ]
. filter ( e = > e . type === 'assistant/message' )
. map ( e = > e . data . usage )
expect ( usages . length ) . toBeGreaterThanOrEqual ( 3 ) // 2 steps in turn 1 + ≥1 in turn 2
for ( const usage of usages ) expect ( usage ) . toBeDefined ( )
// The first request has nothing to hit; every later one shares its
// predecessor as a byte-identical prefix, so the provider must report
// cached prompt tokens (prompt_cache_hit_tokens → cacheReadTokens).
for ( const usage of usages . slice ( 1 ) ) {
expect ( usage ! . cacheReadTokens ? ? 0 ) . toBeGreaterThan ( 0 )
}
// World-verification of the conversation itself: the tool value made it
// through the loop into the final answer.
const finalText = agent . session . deriveMessages ( ) . at ( - 1 ) ! . content
. filter ( block = > block . type === 'text' )
. map ( block = > block . text )
. join ( '' )
expect ( finalText ) . toContain ( 'azure-falcon-42' )
} , 180 _000 )
} )