feat(llm): add model-specific reasoning effort controls

This commit is contained in:
Yichen Jiang
2026-07-25 07:47:51 +08:00
parent 65d29da8a1
commit 8372340f9c
50 changed files with 1046 additions and 88 deletions
+24 -1
View File
@@ -5,11 +5,12 @@
* @module dsh-llm-deepseek/adapter
*/
import { attributionHeaders, CONTEXT_WINDOW_EXCEEDED_CODE, isContextWindowExceededError, isQuotaExceededError, LlmAdapter, LlmError, ProviderRequestId, QUOTA_EXCEEDED_CODE } from '@deepseek-ai/dsh-llm'
import { attributionHeaders, CONTEXT_WINDOW_EXCEEDED_CODE, isContextWindowExceededError, isQuotaExceededError, LlmAdapter, LlmError, ProviderRequestId, QUOTA_EXCEEDED_CODE, ReasoningEffortId } from '@deepseek-ai/dsh-llm'
import type {
GenerateOptions,
LlmModelContext,
LlmModelInfo,
LlmModelReasoningInfo,
LlmProviderInfo,
StreamChunk,
} from '@deepseek-ai/dsh-llm'
@@ -51,6 +52,12 @@ export interface DeepSeekAdapterOptions {
/** Default maximum idle interval while an adapter stream read is outstanding. */
export const DEFAULT_STREAM_IDLE_TIMEOUT_MS = 300_000
const STREAM_IDLE_TIMEOUT_CODE = 'LLM_STREAM_IDLE_TIMEOUT'
const HIGH_REASONING_EFFORT = ReasoningEffortId('high')
const MAX_REASONING_EFFORT = ReasoningEffortId('max')
const REASONING_EFFORTS = [
{ id: HIGH_REASONING_EFFORT, name: 'High' },
{ id: MAX_REASONING_EFFORT, name: 'Max' },
] as const
function providerRetryAfterMs(value: string | null): number | undefined {
if (value === null) return undefined
@@ -98,6 +105,9 @@ export class DeepSeekAdapter extends LlmAdapter {
constructor(private readonly options: DeepSeekAdapterOptions) {
super()
if (options.defaults?.thinking === 'disabled' && options.defaults.reasoningEffort !== undefined) {
throw new Error('llm-deepseek: reasoningEffort cannot be configured when thinking is disabled')
}
if (options.defaultContextWindow !== undefined
&& (!Number.isInteger(options.defaultContextWindow) || options.defaultContextWindow <= 0)) {
throw new Error('llm-deepseek: defaultContextWindow must be a positive integer')
@@ -134,6 +144,19 @@ export class DeepSeekAdapter extends LlmAdapter {
return Promise.resolve(contextWindow === undefined ? undefined : { contextWindow })
}
override resolveModelReasoning(
_provider: string,
_model: string,
): Promise<LlmModelReasoningInfo | undefined> {
if (this.options.defaults?.thinking === 'disabled') return Promise.resolve(undefined)
return Promise.resolve({
efforts: REASONING_EFFORTS,
defaultEffort: this.options.defaults?.reasoningEffort === 'max'
? MAX_REASONING_EFFORT
: HIGH_REASONING_EFFORT,
})
}
async * stream(options: GenerateOptions): AsyncIterable<StreamChunk> {
const consumer = new AbortController()
const upstream = options.signal === undefined
+7 -3
View File
@@ -28,8 +28,9 @@ const DEFAULT_MODELS: DeepSeekCatalogModel[] = [
/**
* Plugin config, validated by the same-named schemastery schema. Every field
* is optional in yml: credentials/endpoint fall back to the environment (a
* missing API key fails plugin load, not the first call), and omitted
* thinking fields send nothing on the wire, so the provider default applies.
* missing API key fails plugin load, not the first call), omitted thinking
* mode uses the provider default, and omitted reasoning effort resolves to
* `high`.
*/
export interface Config {
/** API key; falls back to $DEEPSEEK_API_KEY. Required one way or the other. */
@@ -38,7 +39,7 @@ export interface Config {
baseURL?: string
/** Thinking-mode default for every request (provider default: enabled). */
thinking?: 'enabled' | 'disabled'
/** Thinking effort (only meaningful with thinking enabled). */
/** Default thinking effort when thinking is enabled (default `high`). */
reasoningEffort?: 'high' | 'max'
/** Positive context capacity used when the selected model has no exact value. */
defaultContextWindow?: number
@@ -94,6 +95,9 @@ function resolveModels(models: readonly DeepSeekCatalogModel[] | undefined): Dee
}
export function apply(ctx: Context, config: Config): void {
if (config.thinking === 'disabled' && config.reasoningEffort !== undefined) {
throw new Error('llm-deepseek: reasoningEffort cannot be configured when thinking is disabled')
}
const apiKey = config.apiKey ?? process.env.DEEPSEEK_API_KEY
if (apiKey === undefined || apiKey.length === 0) {
throw new Error('llm-deepseek: an API key is required (Config.apiKey or $DEEPSEEK_API_KEY)')
+16 -2
View File
@@ -6,6 +6,7 @@
* @module dsh-llm-deepseek/serialize
*/
import { LlmError } from '@deepseek-ai/dsh-llm'
import type { ContentBlock, GenerateOptions, Message } from '@deepseek-ai/dsh-llm'
import type { WireMessage, WireRequest, WireTool } from './types.ts'
@@ -15,6 +16,17 @@ export interface RequestDefaults {
reasoningEffort?: 'high' | 'max' | undefined
}
/** Validate the adapter-owned effort before assigning its narrower wire type. */
function reasoningEffort(options: GenerateOptions): 'high' | 'max' | undefined {
const effort = options.reasoningEffort
if (effort === undefined) return undefined
if (effort === 'high' || effort === 'max') return effort as 'high' | 'max'
throw new LlmError(
`DeepSeek does not support reasoning effort "${effort}"`,
'UNSUPPORTED_REASONING_EFFORT',
)
}
/** Join the text blocks of a message (used for user/tool-result content). */
function flattenText(blocks: ContentBlock[]): string {
return blocks
@@ -121,7 +133,9 @@ export function serializeRequest(options: GenerateOptions, defaults: RequestDefa
// A short title budget must produce visible text; conversation and
// compaction calls continue to inherit the adapter's thinking defaults.
const thinking = options.purpose === 'session-title' ? 'disabled' : defaults.thinking
const reasoningEffort = options.purpose === 'session-title' ? undefined : defaults.reasoningEffort
const resolvedReasoningEffort = options.purpose === 'session-title'
? undefined
: reasoningEffort(options)
return {
model: options.model,
@@ -129,7 +143,7 @@ export function serializeRequest(options: GenerateOptions, defaults: RequestDefa
stream: true,
stream_options: { include_usage: true },
...thinking !== undefined ? { thinking: { type: thinking } } : {},
...reasoningEffort !== undefined ? { reasoning_effort: reasoningEffort } : {},
...resolvedReasoningEffort !== undefined ? { reasoning_effort: resolvedReasoningEffort } : {},
...tools !== undefined && tools.length > 0 ? { tools } : {},
...options.temperature !== undefined ? { temperature: options.temperature } : {},
...options.maxTokens !== undefined ? { max_tokens: options.maxTokens } : {},