fix(sse): preserve coherent markdown token boundaries across stream translation chunks (#11606) (#11851)

Preserves markdown token boundaries (bold/italic markers, code fences/spans, escape-backslash parity) split across stream translation chunks by tracking boundary state and flushing held buffers correctly instead of emitting broken markdown mid-stream. Closes #11606. 38/38 + 74/74 focused tests passing. Thanks!
This commit is contained in:
Bob.Hou
2026-08-28 11:22:09 -04:00
committed by GitHub
parent 13afbfafb2
commit 1b39873ea9
4 changed files with 1113 additions and 36 deletions

View File

@@ -0,0 +1,267 @@
/**
* Markdown boundary buffering for streaming text deltas.
*
* Upstream SSE chunks can split in the middle of Markdown tokens such as
* fenced code blocks (```language) or bold markers (**). Emitting those
* partial tokens as separate text_delta events causes clients to render the
* stream with broken Markdown until the next delta arrives.
*
* This helper identifies a trailing suffix that is an *incomplete* Markdown
* boundary token and defers it to the next chunk so the token is emitted in
* one piece.
*
* Rules (held suffixes are bounded by MAX_HOLD_CHARS):
* - 1-2 trailing backticks in an "opener" context (start, whitespace, or
* punctuation before the run) are held.
* - Three trailing backticks followed by a non-empty fence info string are
* held; plain "```" is emitted so a closing fence is not accidentally
* merged with following text.
* - One to three trailing asterisks in an opener context are held. We never
* hold when preceded by an alphanumeric character (which indicates a
* closing delimiter).
*/
const MAX_HOLD_CHARS = 32;
function isOpenerContext(text: string, suffixStart: number): boolean {
if (suffixStart <= 0) return true;
const prev = text[suffixStart - 1];
// Alphanumeric preceding characters usually mean the delimiter is closing
// (e.g. "`code`" or "**bold**"), so do not hold those suffixes.
return !/[A-Za-z0-9_]/.test(prev);
}
function scanBacktickState(
text: string,
initialRun: number,
initialTrailingBackslash: boolean,
initialFenceRun: number,
initialFenceOpening: boolean,
initialFenceClosingRun: number,
initialLineIndent: number
): {
backtickRun: number;
trailingBackslash: boolean;
fenceRun: number;
fenceOpening: boolean;
fenceClosingRun: number;
lineIndent: number;
} {
let openRun = initialRun;
let fenceRun = initialFenceRun;
let fenceOpening = initialFenceOpening;
let fenceClosingRun = initialFenceClosingRun;
let lineIndent = initialLineIndent;
let backslashes = openRun === 0 && fenceRun === 0 && initialTrailingBackslash ? 1 : 0;
for (let index = 0; index < text.length;) {
const char = text[index];
if (fenceOpening) {
if (char === "\n" || char === "\r") {
fenceOpening = false;
lineIndent = 0;
} else if (char === "`" && lineIndent === -1) {
let runEnd = index + 1;
while (runEnd < text.length && text[runEnd] === "`") runEnd++;
fenceRun += runEnd - index;
index = runEnd;
continue;
} else if (char === "`") {
openRun = fenceRun;
fenceRun = 0;
fenceOpening = false;
continue;
} else lineIndent = 4;
index++;
continue;
}
if (fenceClosingRun) {
if (char === "\n" || char === "\r") {
fenceRun = 0;
fenceClosingRun = 0;
lineIndent = 0;
} else if (char === "`" && lineIndent === -1) {
let runEnd = index + 1;
while (runEnd < text.length && text[runEnd] === "`") runEnd++;
fenceClosingRun += runEnd - index;
index = runEnd;
continue;
} else if (char === " " || char === "\t") {
lineIndent = 4;
} else if (char !== " " && char !== "\t") {
fenceClosingRun = 0;
lineIndent = 4;
}
index++;
continue;
}
if (fenceRun) {
if (char === "\n" || char === "\r") {
lineIndent = 0;
index++;
continue;
}
if (char === " " && lineIndent < 4) {
lineIndent++;
index++;
continue;
}
if (char === "`" && lineIndent <= 3) {
let runEnd = index + 1;
while (runEnd < text.length && text[runEnd] === "`") runEnd++;
const runLength = runEnd - index;
if (runLength >= fenceRun) {
fenceClosingRun = runLength;
lineIndent = -1;
} else lineIndent = 4;
index = runEnd;
continue;
}
lineIndent = 4;
index++;
continue;
}
if (char !== "`") {
if (char === "\n" || char === "\r") lineIndent = 0;
else if (char === " " && lineIndent < 4) lineIndent++;
else lineIndent = 4;
if (openRun === 0) backslashes = char === "\\" ? backslashes + 1 : 0;
index++;
continue;
}
if (openRun === 0 && backslashes % 2 === 1) {
backslashes = 0;
lineIndent = 4;
index++;
continue;
}
let runEnd = index + 1;
while (runEnd < text.length && text[runEnd] === "`") runEnd++;
const runLength = runEnd - index;
if (openRun === 0 && runLength >= 3 && lineIndent <= 3) {
fenceRun = runLength;
fenceOpening = true;
lineIndent = -1;
} else if (openRun === 0) openRun = runLength;
else if (openRun === runLength) openRun = 0;
if (!fenceOpening) lineIndent = 4;
backslashes = 0;
index = runEnd;
}
return {
backtickRun: openRun,
trailingBackslash: openRun === 0 && fenceRun === 0 && backslashes % 2 === 1,
fenceRun,
fenceOpening,
fenceClosingRun,
lineIndent,
};
}
export function splitMarkdownBoundary(
text: string,
priorBacktickRun = 0,
priorTrailingBackslash = false,
priorFenceRun = 0,
priorFenceOpening = false,
priorFenceClosingRun = 0,
priorLineIndent = 0
): {
emit: string;
hold: string;
backtickRun?: number;
trailingBackslash?: boolean;
fenceRun?: number;
fenceOpening?: boolean;
fenceClosingRun?: number;
lineIndent?: number;
} {
if (!text) return { emit: "", hold: "" };
// 1) Incomplete fenced code block opener or inline code opener:
// - ` or `` (incomplete delimiter)
// - `code or ``code (incomplete inline code run)
// - ```info (fence delimiter + partial info string; any non-backtick,
// non-line-ending CommonMark info character)
// Do NOT hold plain "```" by itself to avoid gluing a closing fence to
// the next line of normal text.
const fenceMatch = text.match(/(?<!`)(`{1,2}[A-Za-z0-9_+#-]*|`{3,}[^`\r\n]+)$/);
if (fenceMatch) {
const suffix = fenceMatch[0];
const suffixStart = text.length - suffix.length;
const runLength = suffix.match(/^`+/)?.[0].length ?? 0;
const { backtickRun: openRun, fenceRun } = scanBacktickState(
text.slice(0, suffixStart),
priorBacktickRun,
priorTrailingBackslash,
priorFenceRun,
priorFenceOpening,
priorFenceClosingRun,
priorLineIndent
);
const closesKnownRun = openRun === runLength;
const mayCompleteKnownRun = openRun > runLength;
if (
suffix.length <= MAX_HOLD_CHARS &&
fenceRun === 0 &&
!closesKnownRun &&
(mayCompleteKnownRun || isOpenerContext(text, suffixStart))
) {
const emit = text.slice(0, -suffix.length);
return {
emit,
hold: suffix,
...scanBacktickState(
emit,
priorBacktickRun,
priorTrailingBackslash,
priorFenceRun,
priorFenceOpening,
priorFenceClosingRun,
priorLineIndent
),
};
}
}
// 2) Incomplete emphasis/bold opener: 1 to 3 asterisks in an opener context.
const emphMatch = text.match(/(?<!\*)\*{1,3}$/);
if (emphMatch) {
const suffix = emphMatch[0];
if (isOpenerContext(text, text.length - suffix.length)) {
const emit = text.slice(0, -suffix.length);
return {
emit,
hold: suffix,
...scanBacktickState(
emit,
priorBacktickRun,
priorTrailingBackslash,
priorFenceRun,
priorFenceOpening,
priorFenceClosingRun,
priorLineIndent
),
};
}
}
return {
emit: text,
hold: "",
...scanBacktickState(
text,
priorBacktickRun,
priorTrailingBackslash,
priorFenceRun,
priorFenceOpening,
priorFenceClosingRun,
priorLineIndent
),
};
}

View File

@@ -6,6 +6,7 @@ import {
buildGeminiThoughtSignatureKey,
storeGeminiThoughtSignature,
} from "../../services/geminiThoughtSignatureStore.ts";
import { splitMarkdownBoundary } from "../helpers/markdownBoundary.ts";
function normalizeToolName(name: string, toolNameMap?: Map<string, string> | null): string {
return restoreClaudeToolName(name, toolNameMap);
@@ -95,6 +96,33 @@ function extractXmlInvokeBlocks(
return { cleaned, toolCalls };
}
// Helper: flush any buffered Markdown boundary text before closing the open text block
function flushMarkdownBuffer(state, results) {
const buffered = state._markdownBuffer;
state._markdownCodeSpanRun = 0;
state._markdownTrailingBackslash = false;
state._markdownFenceRun = 0;
state._markdownFenceOpening = false;
state._markdownFenceClosingRun = 0;
state._markdownLineIndent = 0;
if (!buffered) return;
state._markdownBuffer = "";
if (state.openTextBlockIdx === null) {
const idx = state.contentBlockIndex++;
state.openTextBlockIdx = idx;
results.push({
type: "content_block_start",
index: idx,
content_block: { type: "text", text: "" },
});
}
results.push({
type: "content_block_delta",
index: state.openTextBlockIdx,
delta: { type: "text_delta", text: buffered },
});
}
/**
* Direct Gemini → Claude response translator.
* Converts Gemini streaming chunks directly to Claude Messages API
@@ -122,6 +150,13 @@ export function geminiToClaudeResponse(chunk, state) {
state.contentBlockIndex = 0;
// Track open text block so we can keep it open across chunks
state.openTextBlockIdx = null;
state._markdownBuffer = "";
state._markdownCodeSpanRun = 0;
state._markdownTrailingBackslash = false;
state._markdownFenceRun = 0;
state._markdownFenceOpening = false;
state._markdownFenceClosingRun = 0;
state._markdownLineIndent = 0;
results.push({
type: "message_start",
@@ -153,6 +188,7 @@ export function geminiToClaudeResponse(chunk, state) {
// Thinking content → thinking block (always open+close per chunk)
if (isThought && part.text) {
// Close any open text block first
flushMarkdownBuffer(state, results);
if (state.openTextBlockIdx !== null) {
results.push({ type: "content_block_stop", index: state.openTextBlockIdx });
state.openTextBlockIdx = null;
@@ -186,6 +222,7 @@ export function geminiToClaudeResponse(chunk, state) {
// Function call → tool_use block
if (part.functionCall) {
// Close any open text block first
flushMarkdownBuffer(state, results);
if (state.openTextBlockIdx !== null) {
results.push({ type: "content_block_stop", index: state.openTextBlockIdx });
state.openTextBlockIdx = null;
@@ -250,6 +287,7 @@ export function geminiToClaudeResponse(chunk, state) {
// Process any extracted text-format tool calls (<tool_call>, TOOL_CALL, <invoke>)
if (textToolCalls.length > 0) {
flushMarkdownBuffer(state, results);
if (state.openTextBlockIdx !== null) {
results.push({ type: "content_block_stop", index: state.openTextBlockIdx });
state.openTextBlockIdx = null;
@@ -290,21 +328,57 @@ export function geminiToClaudeResponse(chunk, state) {
}
if (cleaned) {
// Open a new text block only if none is open yet
if (state.openTextBlockIdx === null) {
const idx = state.contentBlockIndex++;
state.openTextBlockIdx = idx;
// Rehydrate buffered Markdown boundary prefix before emitting.
const bufferedPrefix = state._markdownBuffer || "";
state._markdownBuffer = "";
const combinedText = bufferedPrefix + cleaned;
const {
emit: textToEmit,
hold: textToHold,
backtickRun,
trailingBackslash,
fenceRun,
fenceOpening,
fenceClosingRun,
lineIndent,
} = splitMarkdownBoundary(
combinedText,
state._markdownCodeSpanRun || 0,
state._markdownTrailingBackslash === true,
state._markdownFenceRun || 0,
state._markdownFenceOpening === true,
state._markdownFenceClosingRun || 0,
state._markdownLineIndent || 0,
);
state._markdownBuffer = textToHold;
state._markdownCodeSpanRun = backtickRun || 0;
state._markdownTrailingBackslash = trailingBackslash === true;
state._markdownFenceRun = fenceRun || 0;
state._markdownFenceOpening = fenceOpening === true;
state._markdownFenceClosingRun = fenceClosingRun || 0;
state._markdownLineIndent = lineIndent || 0;
// Fully-held chunk (e.g. "`" + "code" -> whole text deferred to the
// boundary buffer): emitting an empty delta here opens a text block
// and fires a zero-length text_delta for nothing. The held content
// flushes on the next chunk; skip the event pair entirely.
if (textToEmit) {
// Open a new text block only if none is open yet
if (state.openTextBlockIdx === null) {
const idx = state.contentBlockIndex++;
state.openTextBlockIdx = idx;
results.push({
type: "content_block_start",
index: idx,
content_block: { type: "text", text: "" },
});
}
results.push({
type: "content_block_start",
index: idx,
content_block: { type: "text", text: "" },
type: "content_block_delta",
index: state.openTextBlockIdx,
delta: { type: "text_delta", text: textToEmit },
});
}
results.push({
type: "content_block_delta",
index: state.openTextBlockIdx,
delta: { type: "text_delta", text: cleaned },
});
}
}
}
@@ -334,6 +408,7 @@ export function geminiToClaudeResponse(chunk, state) {
// ── Finish reason → close open blocks + message_delta + message_stop ──
if (candidate.finishReason) {
// Close any still-open text block before finishing
flushMarkdownBuffer(state, results);
if (state.openTextBlockIdx !== null) {
results.push({ type: "content_block_stop", index: state.openTextBlockIdx });
state.openTextBlockIdx = null;

View File

@@ -10,6 +10,7 @@ import {
} from "../../utils/reasoningPlaceholder.ts";
import { REVERSE_MAP, restoreClaudeToolName } from "../../services/claudeCodeToolRemapper.ts";
import { sanitizeToolId } from "../helpers/schemaCoercion.ts";
import { splitMarkdownBoundary } from "../helpers/markdownBoundary.ts";
function normalizeToolName(name: string): string {
return REVERSE_MAP[name] ?? name;
@@ -151,8 +152,39 @@ function stopThinkingBlock(state, results) {
state.thinkingBlockStarted = false;
}
// Helper: flush any buffered Markdown boundary text before closing a text block
function flushMarkdownBuffer(state, results) {
const buffered = state._markdownBuffer;
state._markdownCodeSpanRun = 0;
state._markdownTrailingBackslash = false;
state._markdownFenceRun = 0;
state._markdownFenceOpening = false;
state._markdownFenceClosingRun = 0;
state._markdownLineIndent = 0;
if (!buffered) return;
state._markdownBuffer = "";
if (!state.textBlockStarted) {
state.textBlockIndex = state.nextBlockIndex++;
state.textBlockStarted = true;
state.textBlockClosed = false;
results.push({
type: "content_block_start",
index: state.textBlockIndex,
content_block: { type: "text", text: "" },
});
}
if (!state.textBlockClosed) {
results.push({
type: "content_block_delta",
index: state.textBlockIndex,
delta: { type: "text_delta", text: buffered },
});
}
}
// Helper: stop text block if started
function stopTextBlock(state, results) {
flushMarkdownBuffer(state, results);
if (!state.textBlockStarted || state.textBlockClosed) return;
state.textBlockClosed = true;
results.push({
@@ -218,6 +250,13 @@ export function openaiToClaudeResponse(chunk, state) {
state.nextBlockIndex = 0;
state._pendingXmlToolCalls = [];
state._xmlInvokeBuffer = "";
state._markdownBuffer = "";
state._markdownCodeSpanRun = 0;
state._markdownTrailingBackslash = false;
state._markdownFenceRun = 0;
state._markdownFenceOpening = false;
state._markdownFenceClosingRun = 0;
state._markdownLineIndent = 0;
results.push({
type: "message_start",
message: {
@@ -279,8 +318,16 @@ export function openaiToClaudeResponse(chunk, state) {
if (strippedContent) {
stopThinkingBlock(state, results);
// Rehydrate any Markdown boundary suffix buffered from the previous chunk
// before searching for XML tool calls, so the prefix is not lost.
const bufferedPrefix = state._markdownBuffer || "";
state._markdownBuffer = "";
// Check for XML <invoke> blocks that some models emit instead of JSON tool_calls
const { cleaned, toolCalls: xmlToolCalls } = extractXmlInvokeBlocks(strippedContent, state);
const { cleaned, toolCalls: xmlToolCalls } = extractXmlInvokeBlocks(
bufferedPrefix + strippedContent,
state
);
// Accumulate extracted tool calls for emission at finish
if (xmlToolCalls.length > 0) {
@@ -289,12 +336,35 @@ export function openaiToClaudeResponse(chunk, state) {
state._pendingXmlToolCalls.push(...xmlToolCalls);
}
// Defer any trailing incomplete Markdown boundary token to the next chunk.
const {
emit: textToEmit,
hold: textToHold,
backtickRun,
trailingBackslash,
fenceRun,
fenceOpening,
fenceClosingRun,
lineIndent,
} = splitMarkdownBoundary(
cleaned,
state._markdownCodeSpanRun || 0,
state._markdownTrailingBackslash === true,
state._markdownFenceRun || 0,
state._markdownFenceOpening === true,
state._markdownFenceClosingRun || 0,
state._markdownLineIndent || 0,
);
state._markdownBuffer = textToHold;
state._markdownCodeSpanRun = backtickRun || 0;
state._markdownTrailingBackslash = trailingBackslash === true;
state._markdownFenceRun = fenceRun || 0;
state._markdownFenceOpening = fenceOpening === true;
state._markdownFenceClosingRun = fenceClosingRun || 0;
state._markdownLineIndent = lineIndent || 0;
// Emit remaining non-XML text content
if (!cleaned) {
// All content was XML invoke blocks — skip text block entirely
// (tool calls will be emitted at finish)
} else if (xmlToolCalls.length > 0) {
// Text before/between/after XML blocks — (re)start a text block
if (textToEmit) {
if (!state.textBlockStarted) {
state.textBlockIndex = state.nextBlockIndex++;
state.textBlockStarted = true;
@@ -308,24 +378,7 @@ export function openaiToClaudeResponse(chunk, state) {
results.push({
type: "content_block_delta",
index: state.textBlockIndex,
delta: { type: "text_delta", text: cleaned },
});
} else {
// No XML — emit as regular text (original behaviour)
if (!state.textBlockStarted) {
state.textBlockIndex = state.nextBlockIndex++;
state.textBlockStarted = true;
state.textBlockClosed = false;
results.push({
type: "content_block_start",
index: state.textBlockIndex,
content_block: { type: "text", text: "" },
});
}
results.push({
type: "content_block_delta",
index: state.textBlockIndex,
delta: { type: "text_delta", text: cleaned },
delta: { type: "text_delta", text: textToEmit },
});
}
}

View File

@@ -0,0 +1,682 @@
import test from "node:test";
import assert from "node:assert/strict";
const { splitMarkdownBoundary } =
await import("../../open-sse/translator/helpers/markdownBoundary.ts");
const { openaiToClaudeResponse } =
await import("../../open-sse/translator/response/openai-to-claude.ts");
const { geminiToClaudeResponse } =
await import("../../open-sse/translator/response/gemini-to-claude.ts");
function flatten(items: (unknown[] | null)[]) {
return items.flatMap((item) => item || []);
}
function getTextDeltas(events: unknown[]) {
return events
.filter(
(e) =>
(e as Record<string, unknown>)?.type === "content_block_delta" &&
((e as Record<string, unknown>).delta as Record<string, unknown>)?.type === "text_delta"
)
.map(
(e) =>
(((e as Record<string, unknown>).delta as Record<string, unknown>).text as string) ?? ""
);
}
// -- splitMarkdownBoundary unit cases ---------------------------------------
test("splitMarkdownBoundary: no boundary emits everything", () => {
const { emit, hold } = splitMarkdownBoundary("Hello world");
assert.equal(emit, "Hello world");
assert.equal(hold, "");
});
test("splitMarkdownBoundary: defers single trailing backtick in opener context", () => {
const { emit, hold } = splitMarkdownBoundary("Use `git");
assert.equal(emit, "Use ");
assert.equal(hold, "`git");
});
test("splitMarkdownBoundary: defers two trailing backticks", () => {
const { emit, hold } = splitMarkdownBoundary("code ``");
assert.equal(emit, "code ");
assert.equal(hold, "``");
});
test("splitMarkdownBoundary: defers fence opener plus partial language", () => {
const { emit, hold } = splitMarkdownBoundary("\n```p");
assert.equal(emit, "\n");
assert.equal(hold, "```p");
});
test("splitMarkdownBoundary: defers fence info containing CommonMark punctuation", () => {
const { emit, hold } = splitMarkdownBoundary("\n```text/x-c");
assert.equal(emit, "\n");
assert.equal(hold, "```text/x-c");
});
test("splitMarkdownBoundary: emits plain triple backticks unchanged", () => {
const { emit, hold } = splitMarkdownBoundary("code\n```");
assert.equal(emit, "code\n```");
assert.equal(hold, "");
});
test("splitMarkdownBoundary: defers single trailing asterisk in opener context", () => {
const { emit, hold } = splitMarkdownBoundary("This is *");
assert.equal(emit, "This is ");
assert.equal(hold, "*");
});
test("splitMarkdownBoundary: does not defer closing delimiter after alphanumerics", () => {
const { emit, hold } = splitMarkdownBoundary("code`");
assert.equal(emit, "code`");
assert.equal(hold, "");
});
test("splitMarkdownBoundary: does not defer a matched closing backtick after punctuation", () => {
const text = "`(foo)`";
const { emit, hold } = splitMarkdownBoundary(text);
assert.equal(emit, text);
assert.equal(hold, "");
});
test("splitMarkdownBoundary: conservatively defers an unmatched backtick after punctuation", () => {
const { emit, hold } = splitMarkdownBoundary("(foo)`");
assert.equal(emit, "(foo)");
assert.equal(hold, "`");
});
test("splitMarkdownBoundary: keeps different backtick run lengths distinct", () => {
const { emit, hold } = splitMarkdownBoundary("``(foo)`");
assert.equal(emit, "``(foo)");
assert.equal(hold, "`");
});
test("splitMarkdownBoundary: preserves whitespace boundaries", () => {
const { emit, hold } = splitMarkdownBoundary("Hello, ");
assert.equal(emit, "Hello, ");
assert.equal(hold, "");
});
// -- OpenAI to Claude streaming boundary cases -------------------------------
function createOpenAIState() {
return {
toolCalls: new Map(),
_pendingXmlToolCalls: [],
_xmlInvokeBuffer: "",
_markdownBuffer: "",
_markdownCodeSpanRun: 0,
_markdownFenceRun: 0,
};
}
test("OpenAI to Claude: code fence language is not split across chunks", () => {
const state = createOpenAIState();
const chunk1 = openaiToClaudeResponse(
{
id: "chatcmpl-md1",
model: "gpt-4.1",
choices: [{ index: 0, delta: { content: "Here is code:\n\n```p" }, finish_reason: null }],
},
state
);
const chunk2 = openaiToClaudeResponse(
{
id: "chatcmpl-md1",
model: "gpt-4.1",
choices: [{ index: 0, delta: { content: "ython\nprint(1)\n```" }, finish_reason: "stop" }],
usage: { prompt_tokens: 2, completion_tokens: 10, total_tokens: 12 },
},
state
);
const result = flatten([chunk1, chunk2]);
const textDeltas = getTextDeltas(result);
assert.deepEqual(textDeltas, ["Here is code:\n\n", "```python\nprint(1)\n```"]);
});
test("OpenAI to Claude: bold marker is not split across chunks", () => {
const state = createOpenAIState();
const chunk1 = openaiToClaudeResponse(
{
id: "chatcmpl-md2",
model: "gpt-4.1",
choices: [{ index: 0, delta: { content: "This is **" }, finish_reason: null }],
},
state
);
const chunk2 = openaiToClaudeResponse(
{
id: "chatcmpl-md2",
model: "gpt-4.1",
choices: [{ index: 0, delta: { content: "bold** text" }, finish_reason: "stop" }],
usage: { prompt_tokens: 2, completion_tokens: 5, total_tokens: 7 },
},
state
);
const result = flatten([chunk1, chunk2]);
const textDeltas = getTextDeltas(result);
assert.deepEqual(textDeltas, ["This is ", "**bold** text"]);
});
test("OpenAI to Claude: flushes held boundary on finish", () => {
const state = createOpenAIState();
const chunk1 = openaiToClaudeResponse(
{
id: "chatcmpl-md3",
model: "gpt-4.1",
choices: [{ index: 0, delta: { content: "inline `code" }, finish_reason: null }],
},
state
);
const chunk2 = openaiToClaudeResponse(
{
id: "chatcmpl-md3",
model: "gpt-4.1",
choices: [{ index: 0, delta: {}, finish_reason: "stop" }],
usage: { prompt_tokens: 2, completion_tokens: 3, total_tokens: 5 },
},
state
);
const result = flatten([chunk1, chunk2]);
const textDeltas = getTextDeltas(result);
assert.deepEqual(textDeltas, ["inline ", "`code"]);
});
test("OpenAI to Claude: finish flushes a fully-held boundary before message stop", () => {
const state = createOpenAIState();
const chunk1 = openaiToClaudeResponse(
{
id: "chatcmpl-held-finish",
model: "gpt-4.1",
choices: [{ index: 0, delta: { content: "`" }, finish_reason: null }],
},
state
);
const chunk2 = openaiToClaudeResponse(
{
id: "chatcmpl-held-finish",
model: "gpt-4.1",
choices: [{ index: 0, delta: {}, finish_reason: "stop" }],
},
state
);
const result = flatten([chunk1, chunk2]);
assert.deepEqual(getTextDeltas(result), ["`"]);
assert.equal(state._markdownBuffer, "");
assert.deepEqual(
result.slice(1).map((event) => (event as Record<string, unknown>).type),
[
"content_block_start",
"content_block_delta",
"content_block_stop",
"message_delta",
"message_stop",
]
);
});
test("OpenAI to Claude: tool call flushes a fully-held boundary before tool use", () => {
const state = createOpenAIState();
const chunk1 = openaiToClaudeResponse(
{
id: "chatcmpl-held-tool",
model: "gpt-4.1",
choices: [{ index: 0, delta: { content: "`" }, finish_reason: null }],
},
state
);
const chunk2 = openaiToClaudeResponse(
{
id: "chatcmpl-held-tool",
model: "gpt-4.1",
choices: [
{
index: 0,
delta: {
tool_calls: [
{
index: 0,
id: "call_held_tool",
function: { name: "bash", arguments: '{"command":"pwd"}' },
},
],
},
finish_reason: "tool_calls",
},
],
},
state
);
const result = flatten([chunk1, chunk2]);
const contentEvents = result.filter((event) =>
String((event as Record<string, unknown>).type).startsWith("content_block_")
);
assert.deepEqual(getTextDeltas(result), ["`"]);
assert.equal(state._markdownBuffer, "");
assert.deepEqual(
contentEvents.map((event) => {
const record = event as Record<string, unknown>;
const contentBlock = record.content_block as Record<string, unknown> | undefined;
const delta = record.delta as Record<string, unknown> | undefined;
return [record.type, contentBlock?.type ?? delta?.type ?? null];
}),
[
["content_block_start", "text"],
["content_block_delta", "text_delta"],
["content_block_stop", null],
["content_block_start", "tool_use"],
["content_block_delta", "input_json_delta"],
["content_block_stop", null],
]
);
});
test("OpenAI to Claude: reasoning flushes a fully-held boundary before thinking", () => {
const state = createOpenAIState();
const chunk1 = openaiToClaudeResponse(
{
id: "chatcmpl-held-reasoning",
model: "gpt-4.1",
choices: [{ index: 0, delta: { content: "`" }, finish_reason: null }],
},
state
);
const chunk2 = openaiToClaudeResponse(
{
id: "chatcmpl-held-reasoning",
model: "gpt-4.1",
choices: [{ index: 0, delta: { reasoning_content: "Thinking" }, finish_reason: null }],
},
state
);
const result = flatten([chunk1, chunk2]);
assert.deepEqual(getTextDeltas(result), ["`"]);
assert.equal(state._markdownBuffer, "");
assert.deepEqual(
result.slice(1).map((event) => {
const record = event as Record<string, unknown>;
const contentBlock = record.content_block as Record<string, unknown> | undefined;
const delta = record.delta as Record<string, unknown> | undefined;
return [record.type, contentBlock?.type ?? delta?.type ?? null];
}),
[
["content_block_start", "text"],
["content_block_delta", "text_delta"],
["content_block_stop", null],
["content_block_start", "thinking"],
["content_block_delta", "thinking_delta"],
]
);
});
test("OpenAI to Claude: extends a held code span across multiple chunks", () => {
const state = createOpenAIState();
const chunks = ["`", "c", "ode` body"].map((content, index) =>
openaiToClaudeResponse(
{
id: "chatcmpl-multistep",
model: "gpt-4.1",
choices: [{ index: 0, delta: { content }, finish_reason: index === 2 ? "stop" : null }],
},
state
)
);
assert.deepEqual(getTextDeltas(flatten(chunks)), ["`code` body"]);
assert.equal(state._markdownBuffer, "");
});
test("OpenAI to Claude: emits punctuation-adjacent closer opened in a prior chunk", () => {
const state = createOpenAIState();
const chunks = ["Use `foo ", "(bar)`", " done"].map((content, index) =>
openaiToClaudeResponse(
{
id: "chatcmpl-cross-chunk-code",
model: "gpt-4.1",
choices: [{ index: 0, delta: { content }, finish_reason: index === 2 ? "stop" : null }],
},
state
)
);
assert.deepEqual(getTextDeltas(flatten(chunks)), ["Use `foo ", "(bar)`", " done"]);
});
test("OpenAI to Claude: longer fence closer restores inline code parsing", () => {
const state = createOpenAIState();
const contents = ["```\nfoo\n", "````\n", "`(bar)`", " done"];
const chunks = contents.map((content) =>
openaiToClaudeResponse(
{
id: "chatcmpl-long-fence-close",
model: "gpt-4.1",
choices: [{ index: 0, delta: { content }, finish_reason: null }],
},
state
)
);
assert.deepEqual(getTextDeltas(flatten(chunks)), contents);
assert.equal(state._markdownFenceRun, 0);
});
test("OpenAI to Claude: shorter fence run does not close a longer fence", () => {
const state = createOpenAIState();
const contents = ["````\nfoo\n", "```\n", "`(bar)`", " done"];
const chunks = contents.map((content) =>
openaiToClaudeResponse(
{
id: "chatcmpl-short-fence-close",
model: "gpt-4.1",
choices: [{ index: 0, delta: { content }, finish_reason: null }],
},
state
)
);
assert.deepEqual(getTextDeltas(flatten(chunks)), contents);
assert.equal(state._markdownFenceRun, 4);
});
test("OpenAI to Claude: ignores escaped backticks while tracking cross-chunk code spans", () => {
const state = createOpenAIState();
const chunks = ["\\` foo `bar", "`"].map((content) =>
openaiToClaudeResponse(
{
id: "chatcmpl-escaped-code",
model: "gpt-4.1",
choices: [{ index: 0, delta: { content }, finish_reason: null }],
},
state
)
);
assert.deepEqual(getTextDeltas(flatten(chunks)), ["\\` foo ", "`bar`"]);
});
test("OpenAI to Claude: backslash does not escape a code span closing backtick", () => {
const state = createOpenAIState();
const chunks = ["`foo\\`", "(bar)`", " done"].map((content) =>
openaiToClaudeResponse(
{
id: "chatcmpl-code-backslash",
model: "gpt-4.1",
choices: [{ index: 0, delta: { content }, finish_reason: null }],
},
state
)
);
assert.deepEqual(getTextDeltas(flatten(chunks)), ["`foo\\`", "(bar)", "` done"]);
});
test("OpenAI to Claude: escaped opener is equivalent when split after backslash", () => {
const splitState = createOpenAIState();
const splitChunks = ["\\", "` foo ", "(bar)`"].map((content) =>
openaiToClaudeResponse(
{
id: "chatcmpl-split-escape",
model: "gpt-4.1",
choices: [{ index: 0, delta: { content }, finish_reason: null }],
},
splitState
)
);
const joinedState = createOpenAIState();
const joinedChunks = ["\\` foo ", "(bar)`"].map((content) =>
openaiToClaudeResponse(
{
id: "chatcmpl-joined-escape",
model: "gpt-4.1",
choices: [{ index: 0, delta: { content }, finish_reason: null }],
},
joinedState
)
);
assert.equal(
getTextDeltas(flatten(splitChunks)).join(""),
getTextDeltas(flatten(joinedChunks)).join("")
);
assert.equal(splitState._markdownBuffer, "`");
assert.equal(splitState._markdownBuffer, joinedState._markdownBuffer);
assert.equal(splitState._markdownCodeSpanRun || 0, joinedState._markdownCodeSpanRun || 0);
});
test("OpenAI to Claude: ignores literal backtick runs inside longer code spans", () => {
const state = createOpenAIState();
const chunks = ["`` ` `` `foo", "`"].map((content) =>
openaiToClaudeResponse(
{
id: "chatcmpl-nested-code",
model: "gpt-4.1",
choices: [{ index: 0, delta: { content }, finish_reason: null }],
},
state
)
);
assert.deepEqual(getTextDeltas(flatten(chunks)), ["`` ` `` ", "`foo`"]);
});
test("OpenAI to Claude: whitespace between chunks is still preserved", () => {
const state = createOpenAIState();
const chunk1 = openaiToClaudeResponse(
{
id: "chatcmpl-space",
model: "gpt-4.1",
choices: [{ index: 0, delta: { content: "Hello, " }, finish_reason: null }],
},
state
);
const chunk2 = openaiToClaudeResponse(
{
id: "chatcmpl-space",
model: "gpt-4.1",
choices: [{ index: 0, delta: { content: "world." }, finish_reason: "stop" }],
usage: { prompt_tokens: 2, completion_tokens: 3, total_tokens: 5 },
},
state
);
const result = flatten([chunk1, chunk2]);
const textDeltas = getTextDeltas(result);
assert.deepEqual(textDeltas, ["Hello, ", "world."]);
assert.equal(textDeltas.join(""), "Hello, world.");
});
// -- Gemini to Claude streaming boundary cases -------------------------------
function createGeminiState() {
return {
_xmlInvokeBuffer: "",
_markdownBuffer: "",
_markdownCodeSpanRun: 0,
_markdownFenceRun: 0,
};
}
function geminiChunk(text: string, finish = false) {
return {
responseId: "msg-md-gemini",
modelVersion: "gemini-2.0",
candidates: [
{
content: { parts: [{ text }] },
finishReason: finish ? "STOP" : undefined,
},
],
};
}
test("Gemini to Claude: code fence language is not split across chunks", () => {
const state = createGeminiState();
const chunk1 = geminiToClaudeResponse(geminiChunk("Here is code:\n\n```p"), state);
const chunk2 = geminiToClaudeResponse(geminiChunk("ython\nprint(1)\n```", true), state);
const result = flatten([chunk1, chunk2]);
const textDeltas = getTextDeltas(result);
assert.deepEqual(textDeltas, ["Here is code:\n\n", "```python\nprint(1)\n```"]);
});
test("Gemini to Claude: bold marker is not split across chunks", () => {
const state = createGeminiState();
const chunk1 = geminiToClaudeResponse(geminiChunk("This is **"), state);
const chunk2 = geminiToClaudeResponse(geminiChunk("bold** text", true), state);
const result = flatten([chunk1, chunk2]);
const textDeltas = getTextDeltas(result);
assert.deepEqual(textDeltas, ["This is ", "**bold** text"]);
});
test("Gemini to Claude: emits punctuation-adjacent closer opened in a prior chunk", () => {
const state = createGeminiState();
const chunks = ["Use `foo ", "(bar)`", " done"].map((text, index) =>
geminiToClaudeResponse(geminiChunk(text, index === 2), state)
);
assert.deepEqual(getTextDeltas(flatten(chunks)), ["Use `foo ", "(bar)`", " done"]);
});
test("Gemini to Claude: longer fence closer restores inline code parsing", () => {
const state = createGeminiState();
const contents = ["```\nfoo\n", "````\n", "`(bar)`", " done"];
const chunks = contents.map((text) => geminiToClaudeResponse(geminiChunk(text), state));
assert.deepEqual(getTextDeltas(flatten(chunks)), contents);
assert.equal(state._markdownFenceRun, 0);
});
test("Gemini to Claude: shorter fence run does not close a longer fence", () => {
const state = createGeminiState();
const contents = ["````\nfoo\n", "```\n", "`(bar)`", " done"];
const chunks = contents.map((text) => geminiToClaudeResponse(geminiChunk(text), state));
assert.deepEqual(getTextDeltas(flatten(chunks)), contents);
assert.equal(state._markdownFenceRun, 4);
});
test("Gemini to Claude: joins a closing backtick run split across chunks", () => {
const state = createGeminiState();
const chunks = ["``a`", "`", " done"].map((text, index) =>
geminiToClaudeResponse(geminiChunk(text, index === 2), state)
);
assert.deepEqual(getTextDeltas(flatten(chunks)), ["``a", "``", " done"]);
});
test("Gemini to Claude: backslash does not escape a code span closing backtick", () => {
const state = createGeminiState();
const chunks = ["`foo\\`", "(bar)`", " done"].map((text) =>
geminiToClaudeResponse(geminiChunk(text), state)
);
assert.deepEqual(getTextDeltas(flatten(chunks)), ["`foo\\`", "(bar)", "` done"]);
});
test("Gemini to Claude: escaped opener is equivalent when split after backslash", () => {
const splitState = createGeminiState();
const splitChunks = ["\\", "` foo ", "(bar)`"].map((text) =>
geminiToClaudeResponse(geminiChunk(text), splitState)
);
const joinedState = createGeminiState();
const joinedChunks = ["\\` foo ", "(bar)`"].map((text) =>
geminiToClaudeResponse(geminiChunk(text), joinedState)
);
assert.equal(
getTextDeltas(flatten(splitChunks)).join(""),
getTextDeltas(flatten(joinedChunks)).join("")
);
assert.equal(splitState._markdownBuffer, "`");
assert.equal(splitState._markdownBuffer, joinedState._markdownBuffer);
assert.equal(splitState._markdownCodeSpanRun || 0, joinedState._markdownCodeSpanRun || 0);
});
test("Gemini to Claude: flushes held boundary before tool call transition", () => {
const state = createGeminiState();
const chunk1 = geminiToClaudeResponse(geminiChunk("Run `ls"), state);
const chunk2 = geminiToClaudeResponse(
{
responseId: "msg-md-gemini",
modelVersion: "gemini-2.0",
candidates: [
{
content: {
parts: [
{ text: "` then" },
{
functionCall: {
name: "bash",
args: { command: "ls -la" },
},
},
],
},
finishReason: "STOP",
},
],
},
state
);
const result = flatten([chunk1, chunk2]);
const textDeltas = getTextDeltas(result);
assert.deepEqual(textDeltas, ["Run ", "`ls` then"]);
const toolStart = result.find(
(e) =>
(e as Record<string, unknown>)?.type === "content_block_start" &&
((e as Record<string, unknown>).content_block as Record<string, unknown>)?.type === "tool_use"
);
assert.ok(toolStart, "expected tool_use block after flushed text");
});
test("Gemini to Claude: fully-held chunk emits no empty text_delta (#11606 R1)", () => {
const state = createGeminiState();
// Chunk 1 ends with a single backtick (opener context) -> fully held.
const chunk1 = geminiToClaudeResponse(geminiChunk("Run `", false), state);
// Chunk 2 continues with the inline code body + closing backtick.
const chunk2 = geminiToClaudeResponse(geminiChunk("ls` done", true), state);
const result = flatten([chunk1, chunk2]);
const textDeltas = getTextDeltas(result);
// No zero-length delta may appear; the boundary flushes joined on chunk 2.
assert.ok(
textDeltas.every((d) => d.length > 0),
`zero-length text_delta emitted: ${JSON.stringify(textDeltas)}`
);
assert.deepEqual(textDeltas, ["Run ", "`ls` done"]);
// The trailing backtick is held; only the real text "Run " is emitted on
// chunk 1. In particular NO zero-length text_delta may appear (the R1
// finding: a fully-held "cleaned" chunk used to open a text block and fire
// an empty delta for nothing).
const chunk1TextDeltas = (chunk1 as unknown as Record<string, unknown>[])
.filter(
(e) =>
(e as Record<string, unknown>)?.type === "content_block_delta" &&
((e as Record<string, unknown>).delta as Record<string, unknown>)?.type === "text_delta"
)
.map((e) => ((e as Record<string, unknown>).delta as Record<string, unknown>).text ?? "");
assert.deepEqual(chunk1TextDeltas, ["Run "]);
});
test("Gemini to Claude: finish flushes a fully-held boundary before message stop", () => {
const state = createGeminiState();
const chunk1 = geminiToClaudeResponse(geminiChunk("`"), state);
const chunk2 = geminiToClaudeResponse(geminiChunk("", true), state);
const result = flatten([chunk1, chunk2]);
assert.deepEqual(getTextDeltas(result), ["`"]);
assert.equal(state._markdownBuffer, "");
assert.deepEqual(
result.slice(1).map((event) => (event as Record<string, unknown>).type),
[
"content_block_start",
"content_block_delta",
"content_block_stop",
"message_delta",
"message_stop",
]
);
});