Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
31 commits
Select commit Hold shift + click to select a range
b4e1727
fix(vscode-lm): sanitize surrogates, recover leaked tool calls, and w…
Aug 7, 2026
306976d
test(vscode-lm): cover leaked tool-call salvage and tool_result trunc…
Aug 8, 2026
ed3e8ec
fix(vscode-lm): guard leaked tool-call recovery against quoted markup
Aug 8, 2026
cbac74d
chore(knip): exclude .roo skill assets from unused-file analysis
Aug 8, 2026
220ee89
fix(vscode-lm): address review feedback on leaked tool-call recovery
Aug 9, 2026
8c80252
docs(skill): drop probe transcripts from repo
Aug 9, 2026
d587392
chore: move probe skill scripts under scripts/, drop .roo knip ignore
Aug 9, 2026
14e8556
fix(vscode-lm): harden quoted-markup detection and bound the salvage …
Aug 10, 2026
d4e639d
test(vscode-lm): assert the salvage buffer flushes mid-stream
Aug 10, 2026
aa57a19
test(vscode-lm): cover fence-close branch of isInsideCodeFence
Aug 12, 2026
fbe5079
fix(vscode-lm): clamp messages budget to a positive floor
Aug 13, 2026
87d8a68
fix(vscode-lm): require function_calls wrapper and sanitize tool-call…
Aug 15, 2026
1511498
chore: remove probe skill and harness from PR
Aug 16, 2026
8eb1d93
docs: drop dangling probe skill path from vscode-lm comment
Aug 16, 2026
9660bc1
fix(vscode-lm): harden streaming tool-call recovery and token budgeting
Sep 7, 2026
4c9c0af
Merge remote-tracking branch 'origin/main' into port/vscode-lm-reliab…
Sep 9, 2026
4f4e27a
fix: derive mutation gate base from the PR merge commit's first parent
Sep 9, 2026
3f7ccce
fix(vscode-lm): admit requests against the raw context budget, not th…
Sep 9, 2026
e340cb6
fix(vscode-lm): accept an explicit null for a nullable leaked-tool pa…
Sep 10, 2026
0fa5f01
Merge branch 'main' into port/vscode-lm-reliability
simurg79 Sep 10, 2026
7349ba3
fix(vscode-lm): support null-only parameter schemas in leaked tool-ca…
Sep 10, 2026
03ebc16
refactor(vscode-lm): narrow this branch to leaked tool-call recovery
Sep 11, 2026
a012b6a
refactor(vscode-lm): drop surrogate sanitization from the transform l…
Sep 11, 2026
34e16a4
Merge branch 'main' into port/vscode-lm-reliability
simurg79 Sep 11, 2026
c0b3351
refactor(vscode-lm): defer leaked tool-call streaming integration
Sep 11, 2026
8565210
feat(vscode-lm): recover leaked tool calls during streaming
Sep 11, 2026
5719b63
Merge branch 'main' into split/vscode-lm-streaming-b
edelauna Sep 12, 2026
fc24775
Merge origin/main into PR #1608; preserve guarded streaming recovery
Sep 30, 2026
59eb1ba
fix(vscode-lm): remove unreachable empty-buffer guard in flushSalvage
Sep 30, 2026
c23ea4e
fix: preserve large leaked calls and exercise fence guard (#1608)
Oct 1, 2026
2195006
fix(vscode-lm): drain decided salvage prefixes past the buffer cap
Oct 1, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
392 changes: 392 additions & 0 deletions src/api/providers/__tests__/vscode-lm.spec.ts
Original file line number Diff line number Diff line change
Expand Up @@ -293,6 +293,398 @@ describe("VsCodeLmHandler", () => {
})
})

describe("leaked tool-call recovery during streaming", () => {
const salvageTools = [
{
type: "function" as const,
function: {
name: "calculator",
description: "A simple calculator",
parameters: { type: "object", properties: { operation: { type: "string" } } },
},
},
]

const streamTextParts = (parts: string[]) => {
mockLanguageModelChat.sendRequest.mockResolvedValueOnce({
stream: (async function* () {
for (const part of parts) {
yield new vscode.LanguageModelTextPart(part)
}
return
})(),
text: (async function* () {
yield parts.join("")
return
})(),
})
}

const streamMixedParts = (parts: Array<string | { name: string; input: object }>) => {
mockLanguageModelChat.sendRequest.mockResolvedValueOnce({
stream: (async function* () {
for (const part of parts) {
yield typeof part === "string"
? new vscode.LanguageModelTextPart(part)
: new vscode.LanguageModelToolCallPart("native-1", part.name, part.input)
}
return
})(),
text: (async function* () {
yield ""
return
})(),
})
}

const drain = async () => {
const stream = handler.createMessage("system", [{ role: "user" as const, content: "hi" }], {
taskId: "test-task",
tools: salvageTools,
})
const chunks = []
for await (const chunk of stream) {
chunks.push(chunk)
}
return chunks
}

const collect = async (parts: string[]) => {
streamTextParts(parts)
return drain()
}

it("recovers a tool call the model streamed as raw invoke XML", async () => {
const chunks = await collect([
"Thinking. ",
'<function_calls><invoke name="calculator"><parameter name="operation">add</parameter></invoke></function_calls>',
])

expect(chunks.filter((chunk) => chunk.type === "text")).toEqual([{ type: "text", text: "Thinking. " }])
expect(chunks.filter((chunk) => chunk.type === "tool_call")).toEqual([
{
type: "tool_call",
id: expect.stringContaining("vscodelm-salvaged-"),
name: "calculator",
arguments: JSON.stringify({ operation: "add" }),
},
])
})

it("detects a marker split across stream chunks", async () => {
const chunks = await collect([
"abc <function_calls><inv",
'oke name="calculator"><parameter name="operation">sub</parameter></invoke></function_calls>',
])

expect(chunks.filter((chunk) => chunk.type === "text")).toEqual([{ type: "text", text: "abc " }])
expect(chunks.filter((chunk) => chunk.type === "tool_call")).toMatchObject([
{ name: "calculator", arguments: JSON.stringify({ operation: "sub" }) },
])
})

it("emits a carried tail as plain text when it never becomes a marker", async () => {
const chunks = await collect(["hello <par"])

expect(chunks.filter((chunk) => chunk.type === "text")).toEqual([
{ type: "text", text: "hello " },
{ type: "text", text: "<par" },
])
expect(chunks.some((chunk) => chunk.type === "tool_call")).toBe(false)
})

it("buffers across chunks that arrive after the marker", async () => {
const chunks = await collect([
'prose <function_calls><invoke name="calculator">',
'<parameter name="operation">',
"mul</parameter>",
"</invoke></function_calls>",
])

expect(chunks.filter((chunk) => chunk.type === "text")).toEqual([{ type: "text", text: "prose " }])
expect(chunks.filter((chunk) => chunk.type === "tool_call")).toMatchObject([
{ name: "calculator", arguments: JSON.stringify({ operation: "mul" }) },
])
})

it("recovers a null-only declared parameter as JSON null through createMessage", async () => {
mockLanguageModelChat.sendRequest.mockResolvedValueOnce({
stream: (async function* () {
yield new vscode.LanguageModelTextPart(
'<function_calls><invoke name="nuller"><parameter name="cursor">null</parameter></invoke></function_calls>',
)
return
})(),
text: (async function* () {
yield ""
return
})(),
})

const stream = handler.createMessage("system", [{ role: "user" as const, content: "hi" }], {
taskId: "test-task",
tools: [
{
type: "function" as const,
function: {
name: "nuller",
description: "",
parameters: { type: "object", properties: { cursor: { type: "null" } } },
},
},
],
})
const chunks = []
for await (const chunk of stream) {
chunks.push(chunk)
}

expect(chunks.filter((chunk) => chunk.type === "tool_call")).toMatchObject([
{ name: "nuller", arguments: JSON.stringify({ cursor: null }) },
])
})

it("keeps an invoke block for an unknown tool as literal text", async () => {
const block = '<invoke name="not_our_tool"><parameter name="a">1</parameter></invoke>'
const chunks = await collect([block])

expect(chunks.filter((chunk) => chunk.type === "text")).toEqual([{ type: "text", text: block }])
expect(chunks.some((chunk) => chunk.type === "tool_call")).toBe(false)
})

it("emits prose before the recovered tool call", async () => {
const chunks = await collect([
"Thinking. ",
'<function_calls><invoke name="calculator"><parameter name="operation">add</parameter></invoke></function_calls>',
])

expect(chunks.map((chunk) => chunk.type)).toEqual(["text", "tool_call", "usage"])
})

it("flushes buffered text before a native tool call so no text follows a tool_use", async () => {
streamMixedParts([
'partial <invoke name="calculator">',
{ name: "calculator", input: { operation: "div" } },
])
const chunks = await drain()

// The ordering comparison is only meaningful once both kinds of chunk exist: a
// silently broken flush emits no text at all, and -1 < firstToolCall would still hold.
expect(chunks.filter((chunk) => chunk.type === "text")).toEqual([
{ type: "text", text: "partial " },
{ type: "text", text: '<invoke name="calculator">' },
])
const types = chunks.map((chunk) => chunk.type)
const lastText = types.lastIndexOf("text")
const firstToolCall = types.indexOf("tool_call")
expect(lastText).toBeGreaterThanOrEqual(0)
expect(firstToolCall).toBeGreaterThanOrEqual(0)
expect(firstToolCall).toBeGreaterThan(lastText)
})

it("does not latch buffering on prose that merely mentions the tag", async () => {
const chunks = await collect(["never emit <invoke> markup as text. ", "Streaming continues."])

expect(chunks.filter((chunk) => chunk.type === "text")).toEqual([
{ type: "text", text: "never emit <invoke> markup as text. " },
{ type: "text", text: "Streaming continues." },
])
expect(chunks.some((chunk) => chunk.type === "tool_call")).toBe(false)
})

it("does not recover an invoke block quoted inside a fenced code block", async () => {
const block = '<invoke name="calculator"><parameter name="operation">add</parameter></invoke>'
// Open the wrapper outside the fence so only the quoting guard prevents recovery.
const chunks = await collect(["<function_calls>\n```\n" + block + "\n```\n</function_calls>"])

expect(chunks.some((chunk) => chunk.type === "tool_call")).toBe(false)
})
Comment thread
coderabbitai[bot] marked this conversation as resolved.

it.each([
{ name: "write_to_file", parameter: "content" },
{ name: "update_todo_list", parameter: "todos" },
])(
"recovers a large $name call split across chunks without leaking markup",
async ({ name, parameter }) => {
const payload = "x".repeat(5000)
streamTextParts([
`<function_calls><invoke name="${name}"><parameter name="${parameter}">`,
payload,
"</parameter></invoke></function_calls>",
])
const chunks = []
for await (const chunk of handler.createMessage("system", [{ role: "user", content: "hi" }], {
taskId: "test-task",
tools: [
{
type: "function",
function: {
name,
parameters: { type: "object", properties: { [parameter]: { type: "string" } } },
},
},
],
})) {
chunks.push(chunk)
}

expect(chunks.filter((chunk) => chunk.type !== "usage")).toEqual([
{
type: "tool_call",
id: expect.stringContaining("vscodelm-salvaged-"),
name,
arguments: JSON.stringify({ [parameter]: payload }),
},
])
},
)

it("flushes an over-long never-closing invoke as plain text before the stream ends", async () => {
// End-of-stream flushing produces identical text, so track production timing to
// prove the buffer releases text before the stream ends.
const filler = "x".repeat(256 * 1024)
const parts = ['<invoke name="calculator">', filler, filler, filler, filler]
let partsProduced = 0

mockLanguageModelChat.sendRequest.mockResolvedValueOnce({
stream: (async function* () {
for (const part of parts) {
partsProduced++
yield new vscode.LanguageModelTextPart(part)
}
return
})(),
text: (async function* () {
yield ""
return
})(),
})

const stream = handler.createMessage("system", [{ role: "user" as const, content: "hi" }], {
taskId: "test-task",
tools: salvageTools,
})

let sawTextBeforeStreamEnd = false
let streamedText = ""
for await (const chunk of stream) {
if (chunk.type === "text") {
streamedText += chunk.text
if (partsProduced < parts.length) {
sawTextBeforeStreamEnd = true
}
}
}

expect(sawTextBeforeStreamEnd).toBe(true)
expect(streamedText).toContain('<invoke name="calculator">')
expect(streamedText).toContain(filler)
})

it("recovers a completed call and releases text when the buffer passes the cap", async () => {
// The cap used to be bypassed whenever a complete block sat in the buffer, so both the
// call and 512 KB of trailing prose were withheld until the stream ended.
const filler = "y".repeat(256 * 1024)
const parts = [
'<function_calls><invoke name="calculator"><parameter name="operation">add</parameter></invoke></function_calls>',
filler,
filler,
]
let partsProduced = 0

mockLanguageModelChat.sendRequest.mockResolvedValueOnce({
stream: (async function* () {
for (const part of parts) {
partsProduced++
yield new vscode.LanguageModelTextPart(part)
}
return
})(),
text: (async function* () {
yield ""
return
})(),
})

const stream = handler.createMessage("system", [{ role: "user" as const, content: "hi" }], {
taskId: "test-task",
tools: salvageTools,
})

let sawTextBeforeStreamEnd = false
const toolCalls = []
for await (const chunk of stream) {
if (chunk.type === "text" && partsProduced < parts.length) {
sawTextBeforeStreamEnd = true
}
if (chunk.type === "tool_call") {
toolCalls.push(chunk)
}
}

expect(sawTextBeforeStreamEnd).toBe(true)
expect(toolCalls).toMatchObject([
{ name: "calculator", arguments: JSON.stringify({ operation: "add" }) },
])
})

it("keeps an invoke still in flight buffered when the cap is passed", async () => {
// Only the decided prefix may drain; emitting the open block as text would strand the
// call whose closing tag arrives in a later chunk. Every complete block in the
// over-cap buffer must drain now, so timing is asserted rather than final order.
const bulky = "z".repeat(256 * 1024)
const parts = [
`<function_calls>\n<invoke name="calculator"><parameter name="operation">${bulky}</parameter></invoke>\n` +
'<invoke name="calculator"><parameter name="operation">mid</parameter></invoke>\n',
'<invoke name="calculator"><parameter name="operation">sub</parameter>',
"</invoke></function_calls>",
]
let partsProduced = 0

mockLanguageModelChat.sendRequest.mockResolvedValueOnce({
stream: (async function* () {
for (const part of parts) {
partsProduced++
yield new vscode.LanguageModelTextPart(part)
}
return
})(),
text: (async function* () {
yield ""
return
})(),
})

const stream = handler.createMessage("system", [{ role: "user" as const, content: "hi" }], {
taskId: "test-task",
tools: salvageTools,
})

const toolCalls = []
const drainedEarly = []
for await (const chunk of stream) {
if (chunk.type === "tool_call") {
toolCalls.push(chunk)
if (partsProduced < parts.length) {
drainedEarly.push(chunk)
}
}
}

// Both complete blocks sit in the buffer when the cap is crossed, so both must drain
// before the stream ends; the still-open third only resolves at the final chunk.
expect(drainedEarly).toMatchObject([
{ arguments: JSON.stringify({ operation: bulky }) },
{ arguments: JSON.stringify({ operation: "mid" }) },
])
expect(toolCalls).toMatchObject([
{ name: "calculator", arguments: JSON.stringify({ operation: bulky }) },
{ name: "calculator", arguments: JSON.stringify({ operation: "mid" }) },
{ name: "calculator", arguments: JSON.stringify({ operation: "sub" }) },
])
})
})

it("returns the original registry name for a tool declared with an encoded name", async () => {
const systemPrompt = "You are a helpful assistant"
const originalName = `read\uD800file`
Expand Down
Loading
Loading