Sfoglia il codice sorgente

feat(core): replace webfetch markdown renderer (#42229)

Kit Langton 4 giorni fa
parent
commit
90fd61225e

+ 0 - 8
bun.lock

@@ -393,7 +393,6 @@
         "mime-types": "3.0.2",
         "tree-sitter-bash": "0.25.0",
         "tree-sitter-powershell": "0.25.10",
-        "turndown": "7.2.0",
         "venice-ai-sdk-provider": "2.1.1",
         "web-tree-sitter": "0.25.10",
         "which": "6.0.1",
@@ -414,7 +413,6 @@
         "@tsconfig/bun": "catalog:",
         "@types/bun": "catalog:",
         "@types/node": "catalog:",
-        "@types/turndown": "5.0.5",
         "@types/which": "3.0.4",
         "drizzle-kit": "catalog:",
       },
@@ -1966,8 +1964,6 @@
 
     "@mermaid-js/parser": ["@mermaid-js/parser@1.2.0", "", { "dependencies": { "@chevrotain/types": "~11.1.2" } }, "sha512-oYPyv8A4As1yH5Bx+04iQEQxXuIQDe0GKCNSRgao6z8AM9jixXIfP0vsppRLvGf+nKIOb9/LdpWA4YuJiVvESA=="],
 
-    "@mixmark-io/domino": ["@mixmark-io/domino@2.2.0", "", {}, "sha512-Y28PR25bHXUg88kCV7nivXrP2Nj2RueZ3/l/jdx6J9f8J4nsEGcgX0Qe6lt7Pa+J79+kPiJU3LguR6O/6zrLOw=="],
-
     "@modelcontextprotocol/sdk": ["@modelcontextprotocol/sdk@1.29.0", "", { "dependencies": { "@hono/node-server": "^1.19.9", "ajv": "^8.17.1", "ajv-formats": "^3.0.1", "content-type": "^1.0.5", "cors": "^2.8.5", "cross-spawn": "^7.0.5", "eventsource": "^3.0.2", "eventsource-parser": "^3.0.0", "express": "^5.2.1", "express-rate-limit": "^8.2.1", "hono": "^4.11.4", "jose": "^6.1.3", "json-schema-typed": "^8.0.2", "pkce-challenge": "^5.0.0", "raw-body": "^3.0.0", "zod": "^3.25 || ^4.0", "zod-to-json-schema": "^3.25.1" }, "peerDependencies": { "@cfworker/json-schema": "^4.1.1" }, "optionalPeers": ["@cfworker/json-schema"] }, "sha512-zo37mZA9hJWpULgkRpowewez1y6ML5GsXJPY8FI0tBBCd77HEvza4jDqRKOXgHNn867PVGCyTdzqpz0izu5ZjQ=="],
 
     "@motionone/animation": ["@motionone/animation@10.18.0", "", { "dependencies": { "@motionone/easing": "^10.18.0", "@motionone/types": "^10.17.1", "@motionone/utils": "^10.18.0", "tslib": "^2.3.1" } }, "sha512-9z2p5GFGCm0gBsZbi8rVMOAJCtw1WqBTIPw3ozk06gDvZInBPIsQcHgYogEJ4yuHJ+akuW8g1SEIOpTOvYs8hw=="],
@@ -3258,8 +3254,6 @@
 
     "@types/tunnel": ["@types/tunnel@0.0.3", "", { "dependencies": { "@types/node": "*" } }, "sha512-sOUTGn6h1SfQ+gbgqC364jLFBw2lnFqkgF3q0WovEHRLMrVD1sd5aufqi/aJObLekJO+Aq5z646U4Oxy6shXMA=="],
 
-    "@types/turndown": ["@types/turndown@5.0.5", "", {}, "sha512-TL2IgGgc7B5j78rIccBtlYAnkuv8nUQqhQc+DSYV5j9Be9XOcm/SKOVRuA47xAVI3680Tk9B1d8flK2GWT2+4w=="],
-
     "@types/unist": ["@types/unist@3.0.3", "", {}, "sha512-ko/gIFJRv177XgZsZcBwnqJN5x/Gien8qNOn0D5bQU/zAzVf9Zt3BlcUiLqhV9y4ARk0GbT3tnUiPNgnTXzc/Q=="],
 
     "@types/whatwg-mimetype": ["@types/whatwg-mimetype@3.0.2", "", {}, "sha512-c2AKvDT8ToxLIOUlN51gTiHXflsfIFisS4pO7pDPoKouJCESkhZnEy623gwP9laCy5lnLDAw1vAzu2vM2YLOrA=="],
@@ -5774,8 +5768,6 @@
 
     "turbo": ["turbo@2.10.2", "", { "optionalDependencies": { "@turbo/darwin-64": "2.10.2", "@turbo/darwin-arm64": "2.10.2", "@turbo/linux-64": "2.10.2", "@turbo/linux-arm64": "2.10.2", "@turbo/windows-64": "2.10.2", "@turbo/windows-arm64": "2.10.2" }, "bin": { "turbo": "bin/turbo" } }, "sha512-wTExrNrRjB8qzIcg+ZLm0A3GFNLDsWNwdS/RBXB0FPrBDyzk3i96Yx+TxWZC7a0k1SIreFB8ciUbxjmEqTH8IQ=="],
 
-    "turndown": ["turndown@7.2.0", "", { "dependencies": { "@mixmark-io/domino": "^2.2.0" } }, "sha512-eCZGBN4nNNqM9Owkv9HAtWRYfLA4h909E/WGAWWBpmB275ehNhZyk87/Tpvjbp0jjNl9XwCsbe6bm6CqFsgD+A=="],
-
     "tw-animate-css": ["tw-animate-css@1.4.0", "", {}, "sha512-7bziOlRqH0hJx80h/3mbicLW7o8qLsH5+RaLR2t+OHM3D0JlWGODQKQ4cxbK7WlvmUxpcj6Kgu6EKqjrGFe3QQ=="],
 
     "tw-to-css": ["tw-to-css@0.0.12", "", { "dependencies": { "postcss": "8.4.31", "postcss-css-variables": "0.18.0", "tailwindcss": "3.3.2" } }, "sha512-rQAsQvOtV1lBkyCw+iypMygNHrShYAItES5r8fMsrhhaj5qrV2LkZyXc8ccEH+u5bFjHjQ9iuxe90I7Kykf6pw=="],

+ 0 - 2
packages/core/package.json

@@ -79,7 +79,6 @@
     "@tsconfig/bun": "catalog:",
     "@types/bun": "catalog:",
     "@types/node": "catalog:",
-    "@types/turndown": "5.0.5",
     "@types/which": "3.0.4",
     "@parcel/watcher-darwin-arm64": "2.5.1",
     "@parcel/watcher-darwin-x64": "2.5.1",
@@ -137,7 +136,6 @@
     "ignore": "7.0.5",
     "jsonc-parser": "3.3.1",
     "mime-types": "3.0.2",
-    "turndown": "7.2.0",
     "tree-sitter-bash": "0.25.0",
     "tree-sitter-powershell": "0.25.10",
     "venice-ai-sdk-provider": "2.1.1",

+ 662 - 0
packages/core/src/tool/html-markdown.ts

@@ -0,0 +1,662 @@
+import { Parser } from "htmlparser2"
+
+const omitted = new Set(["script", "style", "noscript", "iframe", "object", "embed", "meta", "link", "template"])
+const blocks = new Set([
+  "address",
+  "article",
+  "aside",
+  "details",
+  "dialog",
+  "div",
+  "dl",
+  "fieldset",
+  "figcaption",
+  "figure",
+  "footer",
+  "form",
+  "header",
+  "main",
+  "nav",
+  "p",
+  "section",
+  "summary",
+])
+
+type Frame = {
+  suppressed: boolean
+  link?: { href: string; title?: string }
+  suspendedLink?: Frame["link"]
+  marker?: { index: number; block: number; leadingSpace?: boolean; previous?: Frame["marker"] }
+  code?: { inline: boolean; text: string; language?: string }
+  linkCode?: NonNullable<Frame["code"]>
+  resumedCode?: NonNullable<Frame["code"]>
+  list?: { ordered: boolean; next: number; previous?: Frame["list"] }
+  item?: { indent: string; previous?: Frame["item"] }
+  table?: {
+    start: number
+    rows: string[][]
+    row?: string[]
+    caption?: string
+    fallback: boolean
+    previous?: Frame["table"]
+  }
+  cell?: { start: number }
+  caption?: { start: number }
+  details?: { open: boolean; summary: boolean; previous?: Frame["details"] }
+}
+
+type Chunk = string | { raw: string }
+
+export const MAX_MARKDOWN_BYTES = 5 * 1024 * 1024
+const CONTENT_BYTES = MAX_MARKDOWN_BYTES - 64 * 1024
+
+export function convertHTMLToMarkdown(html: string) {
+  const output: Chunk[] = []
+  const stack: Frame[] = []
+  const encoder = new TextEncoder()
+  let pendingSpace = false
+  let pendingIndent = ""
+  let last = ""
+  let quoteDepth = 0
+  let needsQuotePrefix = false
+  let blockCount = 0
+  let depth = 0
+  let stopped = false
+  let outputBytes = 0
+  let activeCode: NonNullable<Frame["code"]> | undefined
+  let activeLink: Frame["link"] | undefined
+  let linkOpen = false
+  let activeMarker: Frame["marker"] | undefined
+  let activeList: Frame["list"] | undefined
+  let activeItem: Frame["item"] | undefined
+  let activeTable: NonNullable<Frame["table"]> | undefined
+  let activeCell: Frame["cell"] | undefined
+  let tableDepth = 0
+  let fallbackSuppressedDepth = 0
+  let fallbackOmittedDepth = 0
+  let activeDetails: Frame["details"] | undefined
+
+  const sliceBytes = (value: string, bytes: number) => {
+    if (encoder.encode(value).byteLength <= bytes) return value
+    const characters = Array.from(value)
+    let low = 0
+    let high = characters.length
+    while (low < high) {
+      const middle = Math.ceil((low + high) / 2)
+      if (encoder.encode(characters.slice(0, middle).join("")).byteLength <= bytes) low = middle
+      else high = middle - 1
+    }
+    return characters.slice(0, low).join("")
+  }
+  const append = (value: string, content = false) => {
+    const limit = content ? CONTENT_BYTES : MAX_MARKDOWN_BYTES
+    if (!value || outputBytes >= limit) return
+    const bytes = encoder.encode(value)
+    const remaining = limit - outputBytes
+    const next = bytes.byteLength <= remaining ? value : sliceBytes(value, remaining)
+    output.push(next)
+    outputBytes += encoder.encode(next).byteLength
+    last = next.at(-1) ?? last
+  }
+  const appendRaw = (value: string) => {
+    const before = output.length
+    append(value)
+    if (output.length > before) output[output.length - 1] = { raw: output[output.length - 1] as string }
+  }
+  const take = (start: number) => {
+    const chunks = output.splice(start)
+    const value = chunks.map((chunk) => (typeof chunk === "string" ? chunk : chunk.raw)).join("")
+    outputBytes -= encoder.encode(value).byteLength
+    return value
+  }
+  const prefixQuote = () => {
+    if (!needsQuotePrefix || quoteDepth === 0 || activeCell) return
+    append(`${"> ".repeat(Math.min(8, quoteDepth))}`)
+    needsQuotePrefix = false
+  }
+  const flushSpace = () => {
+    if (!pendingSpace) return
+    const marker = activeMarker
+    if (marker && output.length === marker.index + 1 && last !== " " && last !== "\n") {
+      const value = output[marker.index]
+      if (typeof value === "string") output[marker.index] = ` ${value}`
+      outputBytes++
+      marker.leadingSpace = true
+      pendingSpace = false
+      return
+    }
+    if (last && last !== "\n" && last !== " ") append(" ")
+    pendingSpace = false
+  }
+  const inline = (value: string, open = false) => {
+    if (open) flushSpace()
+    prefixQuote()
+    if (pendingIndent) {
+      append(pendingIndent)
+      pendingIndent = ""
+    }
+    if (activeLink && !linkOpen) {
+      append("[")
+      linkOpen = true
+    }
+    append(value)
+  }
+  const block = () => {
+    if (activeLink && linkOpen) {
+      append(`](${destination(activeLink.href)}${title(activeLink.title)})`)
+      linkOpen = false
+    }
+    pendingSpace = false
+    append("\n\n")
+    blockCount++
+    needsQuotePrefix = quoteDepth > 0
+    pendingIndent = activeItem?.indent ?? ""
+  }
+  const suspendLink = (frame: Frame) => {
+    if (!activeLink) return
+    frame.suspendedLink = activeLink
+    block()
+    activeLink = undefined
+    linkOpen = false
+  }
+  const text = (value: string) => {
+    if (activeCode) {
+      activeCode.text += value
+      return
+    }
+    for (const part of value.split(/([\t\n\f\r ]+)/)) {
+      if (!part) continue
+      if (/^[\t\n\f\r ]+$/.test(part)) {
+        pendingSpace = true
+        continue
+      }
+      flushSpace()
+      prefixQuote()
+      if (pendingIndent) {
+        append(pendingIndent)
+        pendingIndent = ""
+      }
+      if (activeLink && !linkOpen) {
+        append("[")
+        linkOpen = true
+      }
+      const escaped = part
+        .replace(/([\\`*_[\]<>|])/g, "\\$1")
+        .replace(/~/g, "\\~")
+        .replace(/^([#+-])/, "\\$1")
+        .replace(/^(\d+)\./, "$1\\.")
+      append(escaped, true)
+    }
+  }
+  const destination = (value: string) => value.replace(/([\\()])/g, "\\$1").replace(/[\t\n\r ]+/g, "%20")
+  const title = (value: string | undefined) =>
+    value
+      ? ` "${value
+          .replace(/[\t\n\r ]+/g, " ")
+          .trim()
+          .replace(/([\\"])/g, "\\$1")}"`
+      : ""
+  const finishCode = (code: NonNullable<Frame["code"]>) => {
+    if (code.inline && !code.text) return
+    let backticks = 0
+    let tildes = 0
+    let currentBackticks = 0
+    let currentTildes = 0
+    for (const character of code.text) {
+      currentBackticks = character === "`" ? currentBackticks + 1 : 0
+      currentTildes = character === "~" ? currentTildes + 1 : 0
+      backticks = Math.max(backticks, currentBackticks)
+      tildes = Math.max(tildes, currentTildes)
+    }
+    if (code.inline) {
+      const fence = "`".repeat(Math.max(1, backticks + 1))
+      const padding = /^ | $/.test(code.text) && !/^ +$/.test(code.text) ? " " : ""
+      flushSpace()
+      prefixQuote()
+      const wrapper = encoder.encode(`${fence}${padding}${padding}${fence}`).byteLength
+      appendRaw(
+        `${fence}${padding}${sliceBytes(code.text, Math.max(0, CONTENT_BYTES - outputBytes - wrapper))}${padding}${fence}`,
+      )
+      return
+    }
+    if (activeCell) {
+      text(code.text)
+      return
+    }
+    const marker = backticks <= tildes ? "`" : "~"
+    const length = Math.max(3, (marker === "`" ? backticks : tildes) + 1)
+    const fence = marker.repeat(length)
+    block()
+    const prefix = `${fence}${code.language ?? ""}\n`
+    const quote = quoteDepth > 0 ? `${"> ".repeat(Math.min(8, quoteDepth))}` : ""
+    const closing = `${code.text.endsWith("\n") ? "" : "\n"}${fence}`
+    let payload = code.text
+    for (;;) {
+      const candidate = `${prefix}${payload}${payload.endsWith("\n") ? "" : "\n"}${fence}`
+      const value = quote ? candidate.replace(/^/gm, quote) : candidate
+      const valueBytes = encoder.encode(value).byteLength
+      if (outputBytes + valueBytes <= CONTENT_BYTES) {
+        appendRaw(value)
+        block()
+        return
+      }
+      const excess = valueBytes - Math.max(0, CONTENT_BYTES - outputBytes)
+      payload = sliceBytes(payload, Math.max(0, encoder.encode(payload).byteLength - Math.ceil(excess)))
+    }
+  }
+
+  const parser = new Parser({
+    onopentag(name, attributes) {
+      depth++
+      if (depth > 10_000) {
+        if (stack.at(-1)?.suppressed) {
+          const visibleParent = stack.findLastIndex((frame) => !frame.suppressed)
+          fallbackSuppressedDepth = visibleParent + 2
+        }
+        activeCode = undefined
+        stopped = true
+      }
+      if (stopped) {
+        if (omitted.has(name)) fallbackOmittedDepth++
+        else pendingSpace = true
+        return
+      }
+      const suppressed = (stack.at(-1)?.suppressed ?? false) || omitted.has(name)
+      const frame: Frame = { suppressed }
+      const hidden = "hidden" in attributes || attributes["aria-hidden"]?.toLowerCase() === "true" || name === "head"
+      const details = activeDetails
+      if (hidden || (details && !details.open && !details.summary && name !== "summary")) frame.suppressed = true
+      stack.push(frame)
+      if (frame.suppressed) return
+
+      if (activeCode && !activeCode.inline) {
+        if (name === "br") activeCode.text += "\n"
+        if (name === "code" && attributes.class)
+          activeCode.language = attributes.class.match(/(?:language-|lang-)([^\s]+)/)?.[1]
+        return
+      }
+      if (name === "details") {
+        frame.details = { open: "open" in attributes, summary: false, previous: activeDetails }
+        activeDetails = frame.details
+        block()
+        return
+      }
+      if (name === "summary") {
+        if (details) details.summary = true
+        block()
+        return
+      }
+      if (name === "pre") {
+        suspendLink(frame)
+        frame.code = { inline: false, text: "" }
+        activeCode = frame.code
+        return
+      }
+      if (name === "code") {
+        if (activeCode?.inline) return
+        frame.code = { inline: true, text: "" }
+        activeCode = frame.code
+        return
+      }
+      if (/^h[1-6]$/.test(name)) {
+        block()
+        inline(`${"#".repeat(Number(name[1]))} `)
+        return
+      }
+      if (blocks.has(name)) {
+        suspendLink(frame)
+        if (name === "p" && last === " ") return
+        block()
+        return
+      }
+      if (name === "br") {
+        pendingSpace = false
+        inline("  \n")
+        needsQuotePrefix = quoteDepth > 0
+        return
+      }
+      if (name === "hr") {
+        block()
+        inline("---")
+        block()
+        return
+      }
+      if (name === "strong" || name === "b") {
+        inline("**", true)
+        frame.marker = { index: output.length - 1, block: blockCount, previous: activeMarker }
+        activeMarker = frame.marker
+        return
+      }
+      if (name === "em" || name === "i") {
+        inline("*", true)
+        frame.marker = { index: output.length - 1, block: blockCount, previous: activeMarker }
+        activeMarker = frame.marker
+        return
+      }
+      if (name === "s" || name === "strike" || name === "del") {
+        inline("~~", true)
+        frame.marker = { index: output.length - 1, block: blockCount, previous: activeMarker }
+        activeMarker = frame.marker
+        return
+      }
+      if (name === "a") {
+        if (activeLink && activeCode?.inline) {
+          const parent = stack.findLast(
+            (candidate) => candidate.link === activeLink && candidate.linkCode === activeCode,
+          )
+          if (parent) {
+            finishCode(activeCode)
+            if (linkOpen) append(`](${destination(activeLink.href)}${title(activeLink.title)})`)
+            activeCode = parent.resumedCode
+            parent.link = undefined
+            parent.linkCode = undefined
+            activeLink = undefined
+            linkOpen = false
+          }
+        }
+        if (activeCode?.inline) {
+          frame.resumedCode = activeCode
+          if (activeCode.text) finishCode(activeCode)
+          activeCode.text = ""
+          activeCode = undefined
+          frame.link = { href: attributes.href ?? "", title: attributes.title }
+          activeLink = frame.link
+          linkOpen = true
+          inline("[", true)
+          frame.linkCode = { inline: true, text: "" }
+          activeCode = frame.linkCode
+          return
+        }
+        if (activeLink) {
+          if (linkOpen) append(`](${destination(activeLink.href)}${title(activeLink.title)})`)
+          const parent = stack.findLast((candidate) => candidate.link === activeLink)
+          if (parent) parent.link = undefined
+          activeLink = undefined
+          linkOpen = false
+        }
+        frame.link = { href: attributes.href ?? "", title: attributes.title }
+        activeLink = frame.link
+        linkOpen = true
+        return inline("[", true)
+      }
+      if (name === "img") {
+        const alt = (attributes.alt ?? "").replace(/([\\\]])/g, "\\$1")
+        const close = `](${destination(attributes.src ?? "")}${title(attributes.title)})`
+        const open = "!["
+        const available = CONTENT_BYTES - outputBytes - encoder.encode(open + close).byteLength
+        inline(`${open}${sliceBytes(alt, Math.max(0, available))}${close}`, true)
+        return
+      }
+      if (name === "blockquote") {
+        suspendLink(frame)
+        block()
+        quoteDepth++
+        needsQuotePrefix = true
+        return
+      }
+      if (name === "ul" || name === "ol") {
+        suspendLink(frame)
+        const start = Number.parseInt(attributes.start ?? "1")
+        frame.list = { ordered: name === "ol", next: Number.isNaN(start) ? 1 : start, previous: activeList }
+        activeList = frame.list
+        block()
+        return
+      }
+      if (name === "li") {
+        block()
+        const value = Number.parseInt(attributes.value ?? "")
+        if (activeList?.ordered && !Number.isNaN(value)) activeList.next = value
+        const marker = activeList?.ordered ? `${activeList.next++}.` : "-"
+        const prefix = `${(activeItem?.indent ?? "").slice(0, 24)}${marker} `
+        frame.item = { indent: " ".repeat(prefix.length), previous: activeItem }
+        activeItem = frame.item
+        pendingIndent = ""
+        inline(prefix)
+        return
+      }
+      if (name === "table") {
+        suspendLink(frame)
+        tableDepth++
+        if (tableDepth === 1) {
+          block()
+          frame.table = { start: output.length, rows: [], fallback: false, previous: activeTable }
+          activeTable = frame.table
+        } else pendingSpace = true
+        return
+      }
+      if (name === "tr") {
+        if (tableDepth !== 1) {
+          pendingSpace = true
+          return
+        }
+        if (activeTable) activeTable.row = []
+        return
+      }
+      if (name === "th" || name === "td") {
+        if (tableDepth !== 1) {
+          pendingSpace = true
+          return
+        }
+        if (attributes.colspan || attributes.rowspan) activeTable!.fallback = true
+        frame.cell = { start: output.length }
+        activeCell = frame.cell
+        return
+      }
+      if (name === "caption") {
+        frame.caption = { start: output.length }
+        return
+      }
+      if (name === "dt") {
+        block()
+        inline("**")
+        return
+      }
+      if (name === "dd") {
+        inline("\n: ")
+        return
+      }
+    },
+    ontext(value) {
+      if (stopped) {
+        if (fallbackSuppressedDepth === 0 && fallbackOmittedDepth === 0) text(value)
+        return
+      }
+      if (stack.at(-1)?.suppressed) return
+      text(value)
+    },
+    onclosetag(name) {
+      depth--
+      if (stopped) {
+        if (fallbackOmittedDepth > 0 && omitted.has(name)) fallbackOmittedDepth--
+        if (fallbackSuppressedDepth > 0 && depth < fallbackSuppressedDepth) fallbackSuppressedDepth = 0
+        return
+      }
+      const frame = stack.pop()
+      if (!frame || frame.suppressed) return
+      if (frame.linkCode) {
+        if (activeLink !== frame.link || !linkOpen) {
+          frame.resumedCode!.text += frame.linkCode.text
+          activeCode = frame.resumedCode
+          if (activeLink === frame.link) activeLink = undefined
+          linkOpen = false
+          return
+        }
+        activeCode = undefined
+        finishCode(frame.linkCode)
+        if (frame.link && linkOpen) append(`](${destination(frame.link.href)}${title(frame.link.title)})`)
+        activeLink = undefined
+        linkOpen = false
+        activeCode = frame.resumedCode
+        return
+      }
+      if (activeCode && !activeCode.inline && !frame.code) return
+      if (frame.code) {
+        activeCode = undefined
+        finishCode(frame.code)
+        if (frame.suspendedLink) activeLink = frame.suspendedLink
+        return
+      }
+      if (name === "summary") {
+        if (activeDetails) activeDetails.summary = false
+        return block()
+      }
+      if (name === "details") {
+        activeDetails = frame.details?.previous
+        return block()
+      }
+      if (name === "dt") {
+        inline("**")
+        return
+      }
+      if (name === "dd") return block()
+      if (
+        name === "strong" ||
+        name === "b" ||
+        name === "em" ||
+        name === "i" ||
+        name === "s" ||
+        name === "strike" ||
+        name === "del"
+      ) {
+        const value = name === "strong" || name === "b" ? "**" : name === "em" || name === "i" ? "*" : "~~"
+        const trailingSpace = pendingSpace
+        pendingSpace = false
+        if (frame.marker) activeMarker = frame.marker.previous
+        if (frame.marker && (frame.marker.block !== blockCount || output.length === frame.marker.index + 1)) {
+          output[frame.marker.index] = ""
+          pendingSpace = trailingSpace || frame.marker.leadingSpace === true
+          return
+        }
+        inline(value)
+        pendingSpace = trailingSpace || frame.marker?.leadingSpace === true
+        return
+      }
+      if (name === "a") {
+        if (frame.link && (activeLink === frame.link || !activeLink)) {
+          activeLink = frame.link
+          if (linkOpen) append(`](${destination(frame.link?.href ?? "")}${title(frame.link?.title)})`)
+          else if (last && last !== "\n") append(`](${destination(frame.link?.href ?? "")}${title(frame.link?.title)})`)
+          linkOpen = false
+          activeLink = undefined
+        }
+        return
+      }
+      if (/^h[1-6]$/.test(name) || blocks.has(name)) {
+        block()
+        if (frame.suspendedLink) activeLink = frame.suspendedLink
+        return
+      }
+      if (name === "blockquote") {
+        quoteDepth--
+        block()
+        if (frame.suspendedLink) activeLink = frame.suspendedLink
+        return
+      }
+      if (name === "li") {
+        activeItem = frame.item?.previous
+        return block()
+      }
+      if (name === "ul" || name === "ol") {
+        activeList = frame.list?.previous
+        block()
+        if (frame.suspendedLink) activeLink = frame.suspendedLink
+        return
+      }
+      if ((name === "th" || name === "td") && tableDepth === 1) {
+        activeCell = undefined
+        if (frame.cell) {
+          const value = take(frame.cell.start)
+            .replace(/[\t\r\n ]+/g, " ")
+            .trim()
+            .replace(/(?<!\\)\|/g, "\\|")
+          activeTable?.row?.push(value)
+        }
+        return
+      }
+      if (name === "tr") {
+        if (tableDepth !== 1) return
+        if (activeTable?.row) activeTable.rows.push(activeTable.row)
+        if (activeTable) activeTable.row = undefined
+        pendingSpace = true
+        return
+      }
+      if (name === "caption" && frame.caption && activeTable) {
+        activeTable.caption = take(frame.caption.start)
+          .replace(/[\t\r\n ]+/g, " ")
+          .trim()
+        return
+      }
+      if (name === "table") {
+        tableDepth--
+        if (tableDepth === 0) {
+          const table = frame.table
+          activeTable = table?.previous
+          if (table) {
+            const loose = take(table.start)
+              .replace(/[\t\r\n ]+/g, " ")
+              .trim()
+            const width = table.rows[0]?.length ?? 0
+            const rectangular = width > 0 && table.rows.every((row) => row.length === width)
+            if (loose) {
+              append(loose)
+              block()
+            }
+            if (table.caption) {
+              append(table.caption)
+              block()
+            }
+            if (!table.fallback && rectangular) {
+              const prefix = `${quoteDepth > 0 ? `${"> ".repeat(Math.min(8, quoteDepth))}` : ""}${pendingIndent}`
+              pendingIndent = ""
+              append(`${prefix}| ${table.rows[0].join(" | ")} |\n${prefix}|${" --- |".repeat(width)}`)
+              for (const row of table.rows.slice(1)) append(`\n${prefix}| ${row.join(" | ")} |`)
+            } else {
+              for (const [index, row] of table.rows.entries()) {
+                if (index > 0) block()
+                append(row.join(" | "))
+              }
+            }
+          }
+          block()
+          if (frame.suspendedLink) activeLink = frame.suspendedLink
+          return
+        }
+        pendingSpace = true
+      }
+    },
+  })
+  for (let index = 0; index < html.length; index += 64 * 1024) parser.write(html.slice(index, index + 64 * 1024))
+  parser.end()
+
+  const normalized: string[] = []
+  let pendingText = ""
+  const flushText = () => {
+    if (!pendingText) return
+    const lines = pendingText
+      .replace(/[ \t]+\n/g, (space) => (space.startsWith("  ") ? "  \n" : "\n"))
+      .replace(/\n{3,}/g, "\n\n")
+      .split("\n")
+    normalized.push(
+      lines
+        .map((line, index) => {
+          if (line) return line
+          const before = lines[index - 1]?.match(/^(?:> )+/)?.[0]
+          const after = lines[index + 1]?.match(/^(?:> )+/)?.[0]
+          if (!before || !after || before.length === after.length) return line
+          return "> ".repeat(Math.min(before.length, after.length) / 2).trimEnd()
+        })
+        .join("\n"),
+    )
+    pendingText = ""
+  }
+  for (const chunk of output) {
+    if (typeof chunk !== "string") {
+      flushText()
+      normalized.push(chunk.raw)
+      continue
+    }
+    pendingText += chunk
+  }
+  flushText()
+  return sliceBytes(normalized.join("").trim(), MAX_MARKDOWN_BYTES)
+}

+ 3 - 13
packages/core/src/tool/plugin/webfetch.ts

@@ -5,12 +5,12 @@ import { ToolFailure } from "@opencode-ai/ai"
 import { Duration, Effect, Schema } from "effect"
 import { HttpClient, HttpClientRequest, HttpClientResponse } from "effect/unstable/http"
 import { Parser } from "htmlparser2"
-import TurndownService from "turndown"
 import { Permission } from "../../permission.js"
+import { convertHTMLToMarkdown, MAX_MARKDOWN_BYTES } from "../html-markdown.js"
 import { collectBoundedResponseBody } from "../http-body.js"
 
 export const name = "webfetch"
-export const MAX_RESPONSE_BYTES = 5 * 1024 * 1024
+export const MAX_RESPONSE_BYTES = MAX_MARKDOWN_BYTES
 export const DEFAULT_TIMEOUT_SECONDS = 30
 export const MAX_TIMEOUT_SECONDS = 120
 
@@ -196,14 +196,4 @@ export function extractTextFromHTML(html: string) {
   return text.trim()
 }
 
-export function convertHTMLToMarkdown(html: string) {
-  const turndown = new TurndownService({
-    headingStyle: "atx",
-    hr: "---",
-    bulletListMarker: "-",
-    codeBlockStyle: "fenced",
-    emDelimiter: "*",
-  })
-  turndown.remove(["script", "style", "meta", "link"])
-  return turndown.turndown(html)
-}
+export { convertHTMLToMarkdown }

+ 313 - 6
packages/core/test/tool-webfetch.test.ts

@@ -68,9 +68,316 @@ describe("WebFetchTool helpers", () => {
   })
 
   test("ports HTML text and markdown conversions without active content", () => {
-    const html = "<h1>Hello</h1><script>bad()</script><p>world <strong>wide</strong></p><style>.bad {}</style>"
-    expect(WebFetchTool.extractTextFromHTML(html)).toBe("Helloworld wide")
-    expect(WebFetchTool.convertHTMLToMarkdown(html)).toBe("# Hello\n\nworld **wide**")
+    const html =
+      "<h1>Hello</h1><script>bad()</script><p>world <strong>wide</strong> <product-name>today</product-name></p><style>.bad {}</style>"
+    expect(WebFetchTool.extractTextFromHTML(html)).toBe("Helloworld wide today")
+    expect(WebFetchTool.convertHTMLToMarkdown(html)).toBe("# Hello\n\nworld **wide** today")
+  })
+
+  test("renders headings, inline semantics, links, images, breaks, and thematic breaks", () => {
+    const html = `<h2>Read <em>this</em></h2><p><a href="https://example.com/a (b)" title="Example">docs</a><br><img src="diagram.png" alt="a ] b"></p><hr><p><del>old</del></p>`
+    expect(WebFetchTool.convertHTMLToMarkdown(html)).toBe(
+      `## Read *this*\n\n[docs](https://example.com/a%20\\(b\\) "Example")  \n![a \\] b](diagram.png)\n\n---\n\n~~old~~`,
+    )
+  })
+
+  test("preserves inline and preformatted code verbatim with safe fences", () => {
+    const html = `<p>Use <code>say(\`hello\`)</code> now.</p><pre><code class="language-ts">const fence = \`\`\`\n&amp; stays decoded</code></pre>`
+    expect(WebFetchTool.convertHTMLToMarkdown(html)).toBe(
+      `Use \`\`say(\`hello\`)\`\` now.\n\n~~~ts\nconst fence = \`\`\`\n& stays decoded\n~~~`,
+    )
+  })
+
+  test("keeps nested ordered and unordered lists structurally readable", () => {
+    const html = `<ol start="3"><li>alpha<ul><li>nested <strong>item</strong></li></ul></li><li><p>beta first</p><p>beta second</p></li></ol>`
+    expect(WebFetchTool.convertHTMLToMarkdown(html)).toBe(
+      `3. alpha\n\n   - nested **item**\n\n4. beta first\n\n   beta second`,
+    )
+  })
+
+  test("renders blockquotes and tables as readable Markdown", () => {
+    const html = `<blockquote><p>quoted <em>text</em></p><ul><li>point</li></ul></blockquote><table><thead><tr><th>Name</th><th>Value</th></tr></thead><tbody><tr><td>one</td><td><code>1</code></td></tr></tbody></table>`
+    expect(WebFetchTool.convertHTMLToMarkdown(html)).toBe(
+      `> quoted *text*\n\n> - point\n\n| Name | Value |\n| --- | --- |\n| one | \`1\` |`,
+    )
+  })
+
+  test("decodes entities and normalizes prose whitespace without joining words", () => {
+    const html = `<p>alpha\n  <span>&amp; beta</span> <unknown>caf&eacute;</unknown>&nbsp;gamma 😀</p><p>delta</p>`
+    expect(WebFetchTool.convertHTMLToMarkdown(html)).toBe(`alpha & beta café gamma 😀\n\ndelta`)
+  })
+
+  test("omits active and fallback content while retaining surrounding prose", () => {
+    const html = `<p>before <script><b>bad</b></script><style>bad</style><noscript>bad</noscript><iframe>bad</iframe><object>bad</object><embed src="bad"><meta content="bad"><link href="bad"><template>bad</template> after</p>`
+    expect(WebFetchTool.convertHTMLToMarkdown(html)).toBe("before after")
+  })
+
+  test("is deterministic and bounded for malformed maximum-size input", () => {
+    const html = `<main><p>${"visible &amp; text ".repeat(250_000)}</main></p></unknown>`
+    const first = WebFetchTool.convertHTMLToMarkdown(html)
+    expect(WebFetchTool.convertHTMLToMarkdown(html)).toBe(first)
+    expect(first.startsWith("visible & text visible & text")).toBe(true)
+    expect(first.length).toBeLessThanOrEqual(html.length)
+  })
+
+  test("bounds deeply nested list output and fragmented code fences", () => {
+    const lists = `${"<ul><li>item".repeat(2_000)}${"</li></ul>".repeat(2_000)}`
+    const quotes = `${"<blockquote><p>item".repeat(2_000)}${"</p></blockquote>".repeat(2_000)}`
+    const code = `<pre>${"` x ".repeat(250_000)}</pre>`
+    expect(WebFetchTool.convertHTMLToMarkdown(lists).length).toBeLessThan(lists.length * 4)
+    expect(WebFetchTool.convertHTMLToMarkdown(quotes).length).toBeLessThan(quotes.length * 4)
+    expect(() => WebFetchTool.convertHTMLToMarkdown(code)).not.toThrow()
+    expect(
+      WebFetchTool.convertHTMLToMarkdown(
+        "<div>".repeat(20_000) + "safe<script><b>bad</b>&amp;</script><p>tail &amp;</p>",
+      ),
+    ).toBe("safe tail &")
+  })
+
+  test("escapes prose that would otherwise become Markdown structure", () => {
+    expect(WebFetchTool.convertHTMLToMarkdown(`<p># heading</p><p>1. item</p><p>---</p><p>a | b</p>`)).toBe(
+      `\\# heading\n\n1\\. item\n\n\\---\n\na \\| b`,
+    )
+  })
+
+  test("preserves code whitespace and quotes every line of multiline blocks", () => {
+    const html = `<blockquote><pre>line  \n\n\nnext</pre><table><tr><td>a|b</td><td>c</td></tr></table></blockquote>`
+    expect(WebFetchTool.convertHTMLToMarkdown(html)).toBe(
+      `> \`\`\`\n> line  \n> \n> \n> next\n> \`\`\`\n\n> | a\\|b | c |\n> | --- | --- |`,
+    )
+  })
+
+  test("keeps nested blockquotes inside their outer quote", () => {
+    const html = `<blockquote><p>outer</p><blockquote><p>inner</p></blockquote><p>end</p></blockquote>`
+    expect(WebFetchTool.convertHTMLToMarkdown(html)).toBe(`> outer\n>\n> > inner\n>\n> end`)
+  })
+
+  test("keeps visible whitespace around inline emphasis", () => {
+    expect(WebFetchTool.convertHTMLToMarkdown(`<p>a<strong> b</strong> c a <em>b </em>c</p>`)).toBe(`a **b** c a *b* c`)
+    expect(WebFetchTool.convertHTMLToMarkdown(`a<strong> </strong>b a<em> </em>b`)).toBe(`a b a b`)
+  })
+
+  test("captures formatting elements inside preformatted content as code only", () => {
+    expect(WebFetchTool.convertHTMLToMarkdown(`<pre><b>x</b><i>y</i><del>z</del></pre>`)).toBe(`\`\`\`\nxyz\n\`\`\``)
+  })
+
+  test("normalizes multiline table cells without changing their columns", () => {
+    const html = `<table><tr><td>x<br>y</td><td><code>a|b</code></td><td><p>first</p><p>second</p></td></tr></table>`
+    expect(WebFetchTool.convertHTMLToMarkdown(html)).toBe(`| x y | \`a\\|b\` | first second |\n| --- | --- | --- |`)
+  })
+
+  test("flattens nested tables without corrupting the outer table", () => {
+    const html = `<table><tr><th>Parent</th><th>Sibling</th></tr><tr><td>Before<table><tr><th>Key</th><th>Value</th></tr><tr><td>A</td><td>1</td></tr></table>After</td><td>Tail</td></tr></table>`
+    expect(WebFetchTool.convertHTMLToMarkdown(html)).toBe(
+      `| Parent | Sibling |\n| --- | --- |\n| Before Key Value A 1 After | Tail |`,
+    )
+  })
+
+  test("preserves loose text around malformed table rows", () => {
+    expect(WebFetchTool.convertHTMLToMarkdown(`<table>before<tr><td>cell</td></tr>after</table>`)).toBe(
+      `before after\n\n| cell |\n| --- |`,
+    )
+    expect(WebFetchTool.convertHTMLToMarkdown(`<table>alpha</table>`)).toBe(`alpha`)
+  })
+
+  test("escapes tilde fences and removes empty emphasis markers", () => {
+    expect(WebFetchTool.convertHTMLToMarkdown(`<p>~~~</p><p><strong></strong>content</p><p>~~~</p>`)).toBe(
+      `\\~\\~\\~\n\ncontent\n\n\\~\\~\\~`,
+    )
+  })
+
+  test("parses malformed tag prefixes in linear time without a regex prepass", () => {
+    const small = "<a".repeat(250_000)
+    const large = "<a".repeat(1_000_000)
+    const start = Bun.nanoseconds()
+    WebFetchTool.convertHTMLToMarkdown(small)
+    const smallDuration = Bun.nanoseconds() - start
+    const next = Bun.nanoseconds()
+    WebFetchTool.convertHTMLToMarkdown(large)
+    const largeDuration = Bun.nanoseconds() - next
+    expect(largeDuration).toBeLessThan(smallDuration * 10)
+  })
+
+  test("caps escaped prose and backtick-heavy pre output at the webfetch response ceiling", () => {
+    const prose = `<p>${"*".repeat(WebFetchTool.MAX_RESPONSE_BYTES)}</p>`
+    const code = `<pre>${"`".repeat(WebFetchTool.MAX_RESPONSE_BYTES - 11)}</pre>`
+    const proseOutput = WebFetchTool.convertHTMLToMarkdown(prose)
+    const codeOutput = WebFetchTool.convertHTMLToMarkdown(code)
+    expect(Buffer.byteLength(proseOutput)).toBeLessThanOrEqual(WebFetchTool.MAX_RESPONSE_BYTES)
+    expect(Buffer.byteLength(codeOutput)).toBeLessThanOrEqual(WebFetchTool.MAX_RESPONSE_BYTES)
+    expect(codeOutput.startsWith("~~~\n")).toBe(true)
+  })
+
+  test("does not confuse source NUL text with buffered code", () => {
+    expect(WebFetchTool.convertHTMLToMarkdown(`<p>before \u00000\u0000 after</p><pre>code</pre>`)).toBe(
+      `before \u00000\u0000 after\n\n\`\`\`\ncode\n\`\`\``,
+    )
+  })
+
+  test("preserves multiline inline code verbatim", () => {
+    expect(WebFetchTool.convertHTMLToMarkdown(`<p><code>first\n\n\nsecond  </code></p>`)).toBe(
+      "` first\n\n\nsecond   `",
+    )
+  })
+
+  test("prefixes inline code at the start of a blockquote line", () => {
+    expect(WebFetchTool.convertHTMLToMarkdown(`<blockquote><code>x</code> y</blockquote>`)).toBe(`> \`x\` y`)
+  })
+
+  test("keeps links nested in inline code associated with their text", () => {
+    const html = `<dl><dt><code>socket = new <a href="#constructor">WebSocket</a>(url)</code><dd>Creates one.</dl>`
+    expect(WebFetchTool.convertHTMLToMarkdown(html)).toBe(
+      `**\` socket = new  \`[\`WebSocket\`](#constructor)\`(url)\`**\n: Creates one.`,
+    )
+    expect(WebFetchTool.convertHTMLToMarkdown(`<code><a href="#x">x</a></code> after`)).toBe(`[\`x\`](#x) after`)
+    expect(
+      WebFetchTool.convertHTMLToMarkdown(
+        `<dl><dt><code><var>socket</var> = new <code><a href="#constructor">WebSocket</a></code>(<var>url</var>)</code><dd>Creates one.</dl>`,
+      ),
+    ).toBe(`**\` socket = new  \`[\`WebSocket\`](#constructor)\`(url)\`**\n: Creates one.`)
+    expect(WebFetchTool.convertHTMLToMarkdown(`<code>a<a href="/x">b<a href="/y">c</a>d</a>e</code>`)).toBe(
+      `\`a\`[\`b\`](\/x)[\`c\`](\/y)\`de\``,
+    )
+    expect(WebFetchTool.convertHTMLToMarkdown(`<code>a<a href="/x">b</code>c`)).toBe(`\`a\`[\`b\`](\/x)c`)
+    expect(WebFetchTool.convertHTMLToMarkdown(`<code>a<a href="/x"><div>b</div>c</a>d</code>`)).toBe(
+      `\`a\`[](\/x)\n\n\`bcd\``,
+    )
+  })
+
+  test("indents nested list continuations and preserves ordered numbering", () => {
+    const html = `<ol start="0"><li value="4"><p>first</p><p>continued</p><ul><li><p>nested</p><p>continued nested</p></li></ul></li><li>next</li></ol>`
+    expect(WebFetchTool.convertHTMLToMarkdown(html)).toBe(
+      `4. first\n\n   continued\n\n   - nested\n\n     continued nested\n\n5. next`,
+    )
+  })
+
+  test("renders block content outside link syntax", () => {
+    expect(WebFetchTool.convertHTMLToMarkdown(`<a href="/docs">before<div>block</div>after</a>`)).toBe(
+      `[before](/docs)\n\nblock\n\n[after](/docs)`,
+    )
+  })
+
+  test("recovers nested anchors without unmatched Markdown syntax", () => {
+    expect(WebFetchTool.convertHTMLToMarkdown(`<a href="/a">x<a href="/b">y</a>z</a>`)).toBe(`[x](/a)[y](/b)z`)
+  })
+
+  test("keeps emphasis whitespace through neutral wrappers", () => {
+    expect(WebFetchTool.convertHTMLToMarkdown(`<p>a<strong><span> bold</span></strong>c</p>`)).toBe(`a **bold** c`)
+  })
+
+  test("flattens preformatted content inside table cells", () => {
+    const html = `<table><tr><td><pre>a|b\nnext</pre></td><td><code>x|y</code></td></tr></table>`
+    expect(WebFetchTool.convertHTMLToMarkdown(html)).toBe(`| a\\|b next | \`x\\|y\` |\n| --- | --- |`)
+  })
+
+  test("keeps each near-boundary inline construct closed and UTF-8-safe", () => {
+    const payload = "😀".repeat(WebFetchTool.MAX_RESPONSE_BYTES / 4)
+    const cases = [
+      [`<strong>${payload}</strong>`, /^\*\*[\s\S]*\*\*$/],
+      [`<a href="/docs">${payload}</a>`, /^\[[\s\S]*\]\(\/docs\)$/],
+      [`<img src="image.png" alt="${payload}">`, /^!\[[\s\S]*\]\(image\.png\)$/],
+      [`<code>${payload}</code>`, /^`[\s\S]*`$/],
+    ] as const
+    for (const [html, pattern] of cases) {
+      const output = WebFetchTool.convertHTMLToMarkdown(html)
+      expect(Buffer.byteLength(output)).toBeLessThanOrEqual(WebFetchTool.MAX_RESPONSE_BYTES)
+      expect(output).not.toContain("�")
+      expect(output).toMatch(pattern)
+    }
+  })
+
+  test("keeps near-boundary block constructs syntactically complete", () => {
+    const payload = "x".repeat(WebFetchTool.MAX_RESPONSE_BYTES)
+    const table = WebFetchTool.convertHTMLToMarkdown(
+      `<table><tr><th>Name</th></tr><tr><td>${payload}</td></tr></table>`,
+    )
+    const list = WebFetchTool.convertHTMLToMarkdown(`<ul><li>${payload}</li></ul><ul><li>nested</li></ul>`)
+    const code = WebFetchTool.convertHTMLToMarkdown(`<pre>${payload}</pre>`)
+    for (const output of [table, list, code]) {
+      expect(Buffer.byteLength(output)).toBeLessThanOrEqual(WebFetchTool.MAX_RESPONSE_BYTES)
+      expect(output).not.toContain("�")
+    }
+    expect(table).toMatch(/^\| Name \|\n\| --- \|\n\| [\s\S]* \|$/)
+    expect(list).toMatch(/^- [\s\S]*$/)
+    expect(list.includes("nested")).toBe(false)
+    expect(code.match(/^(`{3,}|~{3,})$/gm)).toHaveLength(2)
+  })
+
+  test("keeps quoted code within budget with a safe closed fence", () => {
+    const html = `<blockquote><pre>${"`".repeat(32)}${"~".repeat(32)}${"x".repeat(WebFetchTool.MAX_RESPONSE_BYTES)}</pre></blockquote>`
+    const output = WebFetchTool.convertHTMLToMarkdown(html)
+    expect(Buffer.byteLength(output)).toBeLessThanOrEqual(WebFetchTool.MAX_RESPONSE_BYTES)
+    const lines = output.split("\n")
+    expect(lines[0]).toMatch(/^> (`{33}|~{33})$/)
+    expect(lines.at(-1)).toBe(lines[0])
+  })
+
+  test("separates reconstructed tables from adjacent inline and quoted content", () => {
+    const html = `intro<table><tr><td>x</td></tr></table>outro<blockquote>quote<table><tr><td>cell</td></tr></table></blockquote><ul><li>item<table><tr><td>cell</td></tr></table></li></ul>`
+    expect(WebFetchTool.convertHTMLToMarkdown(html)).toBe(
+      `intro\n\n| x |\n| --- |\n\noutro\n\n> quote\n\n> | cell |\n> | --- |\n\n- item\n\n| cell |\n| --- |`,
+    )
+  })
+
+  test("keeps multiline quoted code closed at the content budget", () => {
+    const html = `<blockquote><pre>${"x\n".repeat(WebFetchTool.MAX_RESPONSE_BYTES / 2)}</pre></blockquote><p>tail</p>`
+    const output = WebFetchTool.convertHTMLToMarkdown(html)
+    expect(Buffer.byteLength(output)).toBeLessThanOrEqual(WebFetchTool.MAX_RESPONSE_BYTES)
+    expect((output.match(/(`{3}|~{3})/g) ?? []).length).toBe(2)
+    expect(output.includes("\uFFFD")).toBe(false)
+    expect(output.endsWith("tail")).toBe(true)
+  })
+
+  test("keeps active content suppressed when depth fallback begins", () => {
+    const html = `<object>${"<div>".repeat(10_001)}LEAK${"</div>".repeat(10_001)}</object><p>visible</p>`
+    expect(WebFetchTool.convertHTMLToMarkdown(html)).toBe("visible")
+  })
+
+  test("keeps visible text after depth fallback begins inside preformatted content", () => {
+    const html = `<pre>${"<i>".repeat(10_001)}visible${"</i>".repeat(10_001)}</pre><p>after</p>`
+    expect(WebFetchTool.convertHTMLToMarkdown(html)).toBe("visible after")
+  })
+
+  test("resumes links around every block structure", () => {
+    const html = `<a href="/x">before<blockquote><p>quote</p></blockquote><ul><li>item</li></ul><pre>code</pre><table><tr><td>cell</td></tr></table>after</a>`
+    expect(WebFetchTool.convertHTMLToMarkdown(html)).toBe(
+      `[before](/x)\n\n> quote\n\n- item\n\n\`\`\`\ncode\n\`\`\`\n\n| cell |\n| --- |\n\n[after](/x)`,
+    )
+  })
+
+  test("indents child lists from the actual parent marker width", () => {
+    expect(WebFetchTool.convertHTMLToMarkdown(`<ol start="100"><li>outer<ul><li>inner</li></ul></li></ol>`)).toBe(
+      `100. outer\n\n     - inner`,
+    )
+  })
+
+  test("renders captions and definition lists with readable boundaries", () => {
+    const html = `<table><caption>Cache modes</caption><tr><th>Name</th><th>Meaning</th></tr><tr><td>A</td><td>Local</td></tr></table><dl><dt>Cache</dt><dd>A local store</dd><dt>Origin</dt><dd>The remote source</dd></dl>`
+    expect(WebFetchTool.convertHTMLToMarkdown(html)).toBe(
+      `Cache modes\n\n| Name | Meaning |\n| --- | --- |\n| A | Local |\n\n**Cache**\n: A local store\n\n**Origin**\n: The remote source`,
+    )
+  })
+
+  test("falls back to row-oriented text for table spans", () => {
+    const html = `<table><tr><th colspan="2">Group</th></tr><tr><td>A</td><td rowspan="2">Shared</td></tr><tr><td>B</td></tr></table>`
+    expect(WebFetchTool.convertHTMLToMarkdown(html)).toBe(`Group\n\nA | Shared\n\nB`)
+  })
+
+  test("suppresses head and hidden subtrees while retaining visible body content", () => {
+    const html = `<head><title>noise</title></head><body><p>visible</p><div hidden>hidden</div><div aria-hidden="true">aria</div><div aria-hidden="false">shown</div></body>`
+    expect(WebFetchTool.convertHTMLToMarkdown(html)).toBe(`visible\n\nshown`)
+  })
+
+  test("preserves pre breaks and normalizes multiline link titles", () => {
+    const html = `<pre>first<br>second</pre><p><a href="/x" title="line one\n  line two">link</a></p>`
+    expect(WebFetchTool.convertHTMLToMarkdown(html)).toBe(
+      `\`\`\`\nfirst\nsecond\n\`\`\`\n\n[link](/x "line one line two")`,
+    )
+  })
+
+  test("renders closed and open details according to visibility", () => {
+    const html = `<details><summary>Closed</summary><p>secret</p></details><details open><summary>Open</summary><p>visible</p></details>`
+    expect(WebFetchTool.convertHTMLToMarkdown(html)).toBe(`Closed\n\nOpen\n\nvisible`)
   })
 })
 
@@ -178,7 +485,7 @@ describe("WebFetchTool registration", () => {
     }),
   )
 
-  it.effect("returns an error result when HTML-to-Markdown conversion throws", () =>
+  it.effect("converts deeply nested HTML without overflowing", () =>
     Effect.gen(function* () {
       reset()
       respond = () =>
@@ -191,8 +498,8 @@ describe("WebFetchTool registration", () => {
       const url = "https://1.1.1.1/deep-html"
 
       expect(yield* executeTool(registry, call({ url, format: "markdown" }))).toMatchObject({
-        status: "error",
-        error: { type: "unknown" },
+        status: "completed",
+        content: [{ type: "text", text: "content" }],
       })
     }),
   )