fix(tool-web): bound HTML conversion work
This commit is contained in:
@@ -30,6 +30,52 @@ const turndown = new TurndownService({
|
||||
turndown.use(gfm)
|
||||
turndown.remove(['script', 'style', 'noscript'])
|
||||
|
||||
/** Render one GFM table cell without interpreting HTML span counts. */
|
||||
function renderTableCell(content: string, index: number): string {
|
||||
const prefix = index === 0 ? '| ' : ' '
|
||||
const escaped = content.trim().replace(/\n\r/g, '<br>').replace(/\n/g, '<br>').replace(/\|+/g, '\\|').padEnd(3, ' ')
|
||||
return `${prefix}${escaped} |`
|
||||
}
|
||||
|
||||
/** Whether a row is the table's Markdown heading row. */
|
||||
function isTableHeadingRow(row: HTMLTableRowElement): boolean {
|
||||
const cells = Array.from(row.cells)
|
||||
const section = row.parentElement as HTMLTableSectionElement
|
||||
const table = section.parentElement as HTMLTableElement
|
||||
return (section.nodeName === 'THEAD' || table.rows[0] === row)
|
||||
&& cells.every(cell => cell.nodeName === 'TH')
|
||||
}
|
||||
|
||||
/** Map an HTML table-cell alignment to the GFM separator marker. */
|
||||
function tableBorder(cell: HTMLTableCellElement): string {
|
||||
const alignment = (cell.getAttribute('align') || cell.style.textAlign || '').toLowerCase()
|
||||
if (alignment === 'left') return ':---'
|
||||
if (alignment === 'right') return '---:'
|
||||
if (alignment === 'center') return ':---:'
|
||||
return '---'
|
||||
}
|
||||
|
||||
turndown.addRule('tableCellWithoutSpanExpansion', {
|
||||
filter: ['th', 'td'],
|
||||
replacement(content, node) {
|
||||
const cell = node as HTMLTableCellElement
|
||||
const row = cell.parentNode as HTMLTableRowElement
|
||||
// GFM cannot represent spanning cells. Ignoring colspan keeps conversion
|
||||
// work and output proportional to the source instead of the numeric attribute.
|
||||
return renderTableCell(content, Array.prototype.indexOf.call(row.childNodes, cell))
|
||||
},
|
||||
})
|
||||
turndown.addRule('tableRowWithoutSpanExpansion', {
|
||||
filter: 'tr',
|
||||
replacement(content, node) {
|
||||
const row = node as HTMLTableRowElement
|
||||
const border = isTableHeadingRow(row)
|
||||
? Array.from(row.cells, (cell, index) => renderTableCell(tableBorder(cell), index)).join('')
|
||||
: ''
|
||||
return `\n${content}${border.length > 0 ? `\n${border}` : ''}`
|
||||
},
|
||||
})
|
||||
|
||||
/**
|
||||
* Validate value constraints the schema DSL can't express: a non-blank `url`.
|
||||
* Throws a plain `Error` otherwise. No timeout parameter — the tool-call budget
|
||||
@@ -55,63 +101,142 @@ export function parseFetchArgs(args: { url: string }): { url: string } {
|
||||
*/
|
||||
const MAX_CONVERSION_DEPTH = 512
|
||||
|
||||
/** Elements that never take a closing tag, so they must not count toward nesting depth. */
|
||||
/** Elements that never take a closing tag, so they do not grow the lexical stack. */
|
||||
const VOID_ELEMENTS = new Set([
|
||||
'area', 'base', 'br', 'col', 'embed', 'hr', 'img', 'input',
|
||||
'link', 'meta', 'param', 'source', 'track', 'wbr',
|
||||
])
|
||||
|
||||
/** Elements whose contents HTML parses as text until their matching end tag. */
|
||||
const RAW_TEXT_ELEMENTS = new Set(['script', 'style', 'noscript'])
|
||||
|
||||
/** Whether a character can occur after a raw-text end-tag name. */
|
||||
function isTagBoundary(char: string | undefined): boolean {
|
||||
return char === undefined || char === '>' || char === '/' || /\s/.test(char)
|
||||
}
|
||||
|
||||
/** Find the matching raw-text end tag without interpreting markup-like body text. */
|
||||
function findRawTextEnd(lowerHtml: string, name: string, from: number): number {
|
||||
const prefix = `</${name}`
|
||||
let candidate = lowerHtml.indexOf(prefix, from)
|
||||
while (candidate !== -1 && !isTagBoundary(lowerHtml[candidate + prefix.length])) {
|
||||
candidate = lowerHtml.indexOf(prefix, candidate + prefix.length)
|
||||
}
|
||||
return candidate
|
||||
}
|
||||
|
||||
/**
|
||||
* Estimate the maximum element nesting depth of an HTML string with one linear
|
||||
* tag scan. Overestimates when markup-like text sits inside `script`/`style`
|
||||
* bodies or comments (the scan does not parse those), which can only cause a
|
||||
* spurious raw-HTML fallback, never a missed bound.
|
||||
* Conservatively reject HTML whose lexical element stack crosses the conversion
|
||||
* depth ceiling. The single pass ignores closing tags inside comments, skips
|
||||
* raw-text bodies, respects quoted `>` characters, and only accepts a closing
|
||||
* tag for the current element; malformed input therefore over-counts rather
|
||||
* than hiding nesting.
|
||||
*
|
||||
* @param html - the decoded HTML body.
|
||||
* @returns the deepest open-element count the scan reaches.
|
||||
* @returns whether the body crosses {@link MAX_CONVERSION_DEPTH}.
|
||||
*/
|
||||
export function htmlNestingDepth(html: string): number {
|
||||
let depth = 0
|
||||
let max = 0
|
||||
for (const tag of html.matchAll(/<(\/?)([a-zA-Z][a-zA-Z0-9-]*)[^>]*?(\/?)>/g)) {
|
||||
const [, closing, rawName = '', selfClosing] = tag
|
||||
const name = rawName.toLowerCase()
|
||||
if (VOID_ELEMENTS.has(name) || selfClosing === '/') continue
|
||||
if (closing === '/') {
|
||||
if (depth > 0) depth -= 1
|
||||
} else {
|
||||
depth += 1
|
||||
if (depth > max) max = depth
|
||||
function exceedsConversionDepth(html: string): boolean {
|
||||
const lowerHtml = html.toLowerCase()
|
||||
const openElements: string[] = []
|
||||
let offset = 0
|
||||
let inComment = false
|
||||
|
||||
while (offset < html.length) {
|
||||
const start = html.indexOf('<', offset)
|
||||
if (inComment) {
|
||||
const end = html.indexOf('-->', offset)
|
||||
if (end !== -1 && (start === -1 || end < start)) {
|
||||
inComment = false
|
||||
offset = end + 3
|
||||
continue
|
||||
}
|
||||
}
|
||||
if (start === -1) break
|
||||
if (!inComment && html.startsWith('<!--', start)) {
|
||||
inComment = true
|
||||
offset = start + 4
|
||||
continue
|
||||
}
|
||||
|
||||
let cursor = start + 1
|
||||
const closing = html[cursor] === '/'
|
||||
if (closing) cursor += 1
|
||||
const nameStart = cursor
|
||||
while (/[a-zA-Z0-9-]/.test(html[cursor] ?? '')) cursor += 1
|
||||
if (cursor === nameStart || !/[a-zA-Z]/.test(html.charAt(nameStart))) {
|
||||
offset = start + 1
|
||||
continue
|
||||
}
|
||||
|
||||
const name = lowerHtml.slice(nameStart, cursor)
|
||||
let quote: '"' | "'" | undefined
|
||||
while (cursor < html.length) {
|
||||
const char = html[cursor]
|
||||
cursor += 1
|
||||
if (quote !== undefined) {
|
||||
if (char === quote) quote = undefined
|
||||
} else if (char === '"' || char === "'") {
|
||||
quote = char
|
||||
} else if (char === '>') {
|
||||
break
|
||||
}
|
||||
}
|
||||
if (html[cursor - 1] !== '>') break
|
||||
|
||||
if (closing) {
|
||||
if (!inComment && openElements.at(-1) === name) openElements.pop()
|
||||
} else {
|
||||
let last = cursor - 2
|
||||
while (/\s/.test(html.charAt(last))) last -= 1
|
||||
if (!VOID_ELEMENTS.has(name) && html[last] !== '/') {
|
||||
openElements.push(name)
|
||||
if (openElements.length > MAX_CONVERSION_DEPTH) return true
|
||||
if (!inComment && RAW_TEXT_ELEMENTS.has(name)) {
|
||||
const end = findRawTextEnd(lowerHtml, name, cursor)
|
||||
if (end === -1) break
|
||||
offset = end
|
||||
continue
|
||||
}
|
||||
}
|
||||
}
|
||||
offset = cursor
|
||||
}
|
||||
return max
|
||||
return false
|
||||
}
|
||||
|
||||
interface RenderedBody {
|
||||
/** Converted text, or raw HTML when conversion is unsafe or fails. */
|
||||
text: string
|
||||
/** Whether the source was cut before conversion to bound synchronous work. */
|
||||
sourceTruncated: boolean
|
||||
}
|
||||
|
||||
/**
|
||||
* Render a fetched body to model-facing markdown text.
|
||||
*
|
||||
* @param body - the decoded body; `html` is converted via turndown, `text`
|
||||
* passes through verbatim. HTML nested beyond {@link MAX_CONVERSION_DEPTH}
|
||||
* skips conversion up front (the synchronous walk over such trees is
|
||||
* superlinear and blocks the event loop past the cooperative timeout), and
|
||||
* when turndown itself throws the raw HTML passes through instead — a
|
||||
* degraded page beats an error for a body the provider already decoded.
|
||||
* @returns the text for the tool's output block.
|
||||
* passes through verbatim.
|
||||
* @param maxInputChars - maximum source characters processed synchronously.
|
||||
* @returns the rendered prefix and whether the source was cut. HTML nested
|
||||
* beyond {@link MAX_CONVERSION_DEPTH} or rejected by turndown passes through
|
||||
* raw; a degraded page beats an error for a body the provider decoded.
|
||||
*/
|
||||
export function renderBody(body: WebFetchBody): string {
|
||||
function renderBody(body: WebFetchBody, maxInputChars: number): RenderedBody {
|
||||
const content = body.content.slice(0, maxInputChars)
|
||||
const sourceTruncated = content.length !== body.content.length
|
||||
switch (body.kind) {
|
||||
case 'html':
|
||||
if (htmlNestingDepth(body.content) > MAX_CONVERSION_DEPTH) return body.content
|
||||
if (exceedsConversionDepth(content)) return { text: content, sourceTruncated }
|
||||
try {
|
||||
return turndown.turndown(body.content)
|
||||
return { text: turndown.turndown(content), sourceTruncated }
|
||||
} catch {
|
||||
// turndown's DOM walk recurses per element; malformed markup the depth
|
||||
// scan cannot see can still throw RangeError. Provider errors stay
|
||||
// turndown's DOM walk recurses per element; malformed markup the lexical
|
||||
// guard cannot model can still throw RangeError. Provider errors stay
|
||||
// structured WebErrors upstream; conversion failure downgrades to raw HTML.
|
||||
return body.content
|
||||
return { text: content, sourceTruncated }
|
||||
}
|
||||
case 'text':
|
||||
return body.content
|
||||
return { text: content, sourceTruncated }
|
||||
/* v8 ignore next 2 -- WebFetchBody is a closed union; this arm is unreachable and only makes adding a kind a compile error. */
|
||||
default:
|
||||
return assertNever(body, 'unhandled web fetch body kind')
|
||||
@@ -123,9 +248,8 @@ const TRUNCATION_FOOTER = '\n\n(Content truncated. Fetch a more specific URL or
|
||||
|
||||
/**
|
||||
* Format a fetch result as one model-facing text block, bounded as a whole.
|
||||
* Markdown escaping can expand converted HTML (worst case ~2× the provider's
|
||||
* body cap), so the bound applies here, where the complete output — header,
|
||||
* rendered body, and footer — is known.
|
||||
* The same cap limits the source prefix processed synchronously, then applies
|
||||
* again where the complete output — header, rendered body, and footer — is known.
|
||||
*
|
||||
* @param result - the seam's fetch outcome.
|
||||
* @param maxOutputChars - cap on the complete returned string; a cut body gets
|
||||
@@ -135,11 +259,13 @@ const TRUNCATION_FOOTER = '\n\n(Content truncated. Fetch a more specific URL or
|
||||
*/
|
||||
export function formatFetchOutput(result: WebFetchResult, maxOutputChars: number): string {
|
||||
const header = `Fetched ${result.url} (HTTP ${result.statusCode})\n\n`
|
||||
const body = renderBody(result.body)
|
||||
const full = `${header}${body}${result.truncated ? TRUNCATION_FOOTER : ''}`
|
||||
const rendered = renderBody(result.body, maxOutputChars)
|
||||
const prefix = `${header}${rendered.text}`
|
||||
const truncated = result.truncated || rendered.sourceTruncated || prefix.length > maxOutputChars
|
||||
const full = `${prefix}${truncated ? TRUNCATION_FOOTER : ''}`
|
||||
if (full.length <= maxOutputChars) return full
|
||||
const budget = Math.max(0, maxOutputChars - header.length - TRUNCATION_FOOTER.length)
|
||||
return `${header}${body.slice(0, budget)}${TRUNCATION_FOOTER}`
|
||||
if (maxOutputChars < TRUNCATION_FOOTER.length) return full.slice(0, maxOutputChars)
|
||||
return `${prefix.slice(0, maxOutputChars - TRUNCATION_FOOTER.length)}${TRUNCATION_FOOTER}`
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -160,8 +286,7 @@ export function presentFetchCall(args: { url: string }): GenericCallView {
|
||||
* @param timeoutMs - the cooperative tool-call budget (ms) attached as the tool's
|
||||
* `ToolDefinition.timeoutMs` for `@deepseek-ai/dsh-timeout-policy` to enforce.
|
||||
* @param maxOutputChars - cap on the complete rendered tool output (see
|
||||
* {@link formatFetchOutput}); markdown escaping can outgrow the provider's
|
||||
* body cap, so the model-context bound is enforced on the rendered result.
|
||||
* {@link formatFetchOutput}) and on source characters converted synchronously.
|
||||
*/
|
||||
export function applyWebFetchTool(ctx: Context, timeoutMs: number, maxOutputChars: number): void {
|
||||
ctx.systemPrompt.section({
|
||||
|
||||
Reference in New Issue
Block a user