mirror of
https://github.com/diegosouzapw/OmniRoute.git
synced 2026-08-06 23:32:12 +03:00
Scope textual thinking tag extraction (#5224)
Scope textual thinking-tag extraction to tag-native model families; preserve GEMINI_CLI registration. Resubmit of #5216 without the regression. Integrated into release/v3.8.40.
This commit is contained in:
@@ -234,7 +234,11 @@ import { buildCodexQuotaPersistence } from "./chatCore/codexQuota.ts";
|
||||
import { invalidateCodexQuotaCache } from "../services/codexQuotaFetcher.ts";
|
||||
import { translateNonStreamingResponse } from "./responseTranslator.ts";
|
||||
import { extractUsageFromResponse } from "./usageExtractor.ts";
|
||||
import { sanitizeOpenAIResponse, sanitizeResponsesApiResponse } from "./responseSanitizer.ts";
|
||||
import {
|
||||
sanitizeOpenAIResponse,
|
||||
sanitizeResponsesApiResponse,
|
||||
shouldParseTextualReasoningTags,
|
||||
} from "./responseSanitizer.ts";
|
||||
import {
|
||||
withRateLimit,
|
||||
updateFromHeaders,
|
||||
@@ -3613,7 +3617,10 @@ export async function handleChatCore({
|
||||
// that non-standard field. Reasoning replay cache is captured above this
|
||||
// sanitize step, so the cache feature is unaffected.
|
||||
const stripReasoning = isStripReasoningRequested(clientRawRequest?.headers ?? null);
|
||||
translatedResponse = sanitizeOpenAIResponse(translatedResponse, { stripReasoning });
|
||||
translatedResponse = sanitizeOpenAIResponse(translatedResponse, {
|
||||
stripReasoning,
|
||||
parseTextualReasoningTags: shouldParseTextualReasoningTags(provider, model),
|
||||
});
|
||||
}
|
||||
|
||||
applyClientUsageBuffer(translatedResponse, body, clientResponseFormat);
|
||||
|
||||
@@ -10,7 +10,7 @@ import { normalizeOpenAICompatibleFinishReason } from "../utils/finishReason.ts"
|
||||
* Fixes Issues:
|
||||
* 1. Strips non-standard fields (x_groq, usage_breakdown, service_tier) that
|
||||
* break OpenAI Python SDK v1.83+ Pydantic validation (returns str instead of object)
|
||||
* 2. Extracts <think> tags from thinking models into reasoning_content
|
||||
* 2. Optionally extracts native textual reasoning tags from known tag-style models
|
||||
* 3. Normalizes response id, object, and usage fields
|
||||
* 4. Converts developer role → system for non-OpenAI providers
|
||||
*/
|
||||
@@ -32,6 +32,7 @@ const ALLOWED_RESPONSES_USAGE_FIELDS = new Set([
|
||||
]);
|
||||
|
||||
type JsonRecord = Record<string, unknown>;
|
||||
type ParseOptions = { parseTextualReasoningTags?: boolean };
|
||||
|
||||
export const OMIT_STREAMING_CHUNK_MARKER = "__omniroute_omit_streaming_chunk";
|
||||
|
||||
@@ -297,6 +298,37 @@ export function extractThinkingFromContent(text: string): {
|
||||
};
|
||||
}
|
||||
|
||||
function normalizeReasoningRouteId(value: unknown): string {
|
||||
return typeof value === "string" ? value.toLowerCase() : "";
|
||||
}
|
||||
|
||||
function isAntigravityReasoningRoute(providerId: string, modelId: string): boolean {
|
||||
return (
|
||||
providerId.includes("antigravity") ||
|
||||
providerId === "agy" ||
|
||||
modelId.includes("antigravity/") ||
|
||||
modelId.startsWith("agy/")
|
||||
);
|
||||
}
|
||||
|
||||
function isTextualReasoningTagNativeRoute(providerId: string, modelId: string): boolean {
|
||||
const routeId = `${providerId}/${modelId}`;
|
||||
return (
|
||||
/deepseek[-_/]?r1\b/.test(routeId) ||
|
||||
/r1[-_/]?distill\b/.test(routeId) ||
|
||||
/(?:^|[/:_-])qwq(?:[/._:-]|$)/.test(routeId)
|
||||
);
|
||||
}
|
||||
|
||||
export function shouldParseTextualReasoningTags(provider?: unknown, model?: unknown): boolean {
|
||||
const providerId = normalizeReasoningRouteId(provider);
|
||||
const modelId = normalizeReasoningRouteId(model);
|
||||
return (
|
||||
!isAntigravityReasoningRoute(providerId, modelId) &&
|
||||
isTextualReasoningTagNativeRoute(providerId, modelId)
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Sanitize a non-streaming OpenAI ChatCompletion response.
|
||||
* Strips non-standard fields and normalizes required fields.
|
||||
@@ -311,6 +343,12 @@ export interface SanitizeOpenAIResponseOptions {
|
||||
* Ported from upstream 9router#517 (closes upstream #509).
|
||||
*/
|
||||
stripReasoning?: boolean;
|
||||
/**
|
||||
* Keep disabled for generic OpenAI-compatible responses: prompt-format tags
|
||||
* can be user-requested visible content. Enable only for routes/models whose
|
||||
* upstream contract uses textual tags as the native reasoning channel.
|
||||
*/
|
||||
parseTextualReasoningTags?: boolean;
|
||||
}
|
||||
|
||||
export function sanitizeOpenAIResponse(
|
||||
@@ -320,6 +358,7 @@ export function sanitizeOpenAIResponse(
|
||||
const bodyRecord = toRecord(body);
|
||||
if (!bodyRecord) return body;
|
||||
const stripReasoning = options.stripReasoning === true;
|
||||
const parseTextualReasoningTags = options.parseTextualReasoningTags === true;
|
||||
|
||||
// Build sanitized response with only allowed top-level fields
|
||||
const sanitized: JsonRecord = {};
|
||||
@@ -333,7 +372,7 @@ export function sanitizeOpenAIResponse(
|
||||
// Sanitize choices
|
||||
if (Array.isArray(bodyRecord.choices)) {
|
||||
sanitized.choices = bodyRecord.choices.map((choice, idx) => {
|
||||
const sanitizedChoice = sanitizeChoice(choice, idx);
|
||||
const sanitizedChoice = sanitizeChoice(choice, idx, { parseTextualReasoningTags });
|
||||
const message = toRecord(sanitizedChoice.message);
|
||||
if (
|
||||
message &&
|
||||
@@ -426,7 +465,11 @@ export function sanitizeResponsesApiResponse(body: unknown): unknown {
|
||||
/**
|
||||
* Sanitize a single choice object.
|
||||
*/
|
||||
function sanitizeChoice(choice: unknown, defaultIndex: number): JsonRecord {
|
||||
function sanitizeChoice(
|
||||
choice: unknown,
|
||||
defaultIndex: number,
|
||||
options: ParseOptions = {}
|
||||
): JsonRecord {
|
||||
const choiceRecord = toRecord(choice);
|
||||
const sanitized: JsonRecord = {
|
||||
index: defaultIndex,
|
||||
@@ -441,15 +484,13 @@ function sanitizeChoice(choice: unknown, defaultIndex: number): JsonRecord {
|
||||
sanitized.finish_reason = normalizeOpenAICompatibleFinishReason(choiceRecord.finish_reason);
|
||||
}
|
||||
|
||||
// Sanitize message (non-streaming) or delta (streaming)
|
||||
if (choiceRecord?.message !== undefined) {
|
||||
sanitized.message = sanitizeMessage(choiceRecord.message);
|
||||
sanitized.message = sanitizeMessage(choiceRecord.message, options);
|
||||
}
|
||||
if (choiceRecord?.delta !== undefined) {
|
||||
sanitized.delta = sanitizeMessage(choiceRecord.delta);
|
||||
sanitized.delta = sanitizeMessage(choiceRecord.delta, options);
|
||||
}
|
||||
|
||||
// Keep logprobs if present
|
||||
if (choiceRecord?.logprobs !== undefined) {
|
||||
sanitized.logprobs = choiceRecord.logprobs;
|
||||
}
|
||||
@@ -457,37 +498,23 @@ function sanitizeChoice(choice: unknown, defaultIndex: number): JsonRecord {
|
||||
return sanitized;
|
||||
}
|
||||
|
||||
/**
|
||||
* Sanitize a message object, extracting <think> tags if present.
|
||||
*/
|
||||
function sanitizeMessage(msg: unknown): unknown {
|
||||
const msgRecord = toRecord(msg);
|
||||
if (!msgRecord) return msg;
|
||||
|
||||
const sanitized: JsonRecord = {};
|
||||
|
||||
// Copy only allowed fields
|
||||
if (msgRecord.role) sanitized.role = msgRecord.role;
|
||||
if (msgRecord.refusal !== undefined) sanitized.refusal = msgRecord.refusal;
|
||||
|
||||
// Handle content — extract <think> tags
|
||||
function sanitizeMessageContent(msgRecord: JsonRecord, options: ParseOptions = {}): JsonRecord {
|
||||
if (typeof msgRecord.content === "string") {
|
||||
const { content, thinking } = extractThinkingFromContent(
|
||||
stripInternalToolEnvelopeText(msgRecord.content)
|
||||
);
|
||||
sanitized.content = collapseExcessiveNewlines(content);
|
||||
|
||||
// Set reasoning_content from prompt-format tags only when the provider did
|
||||
// not also return a native OpenAI-compatible reasoning field.
|
||||
if (thinking && !getReadableReasoningValue(msgRecord)) {
|
||||
sanitized.reasoning_content = thinking;
|
||||
}
|
||||
} else if (msgRecord.content !== undefined) {
|
||||
sanitized.content = msgRecord.content;
|
||||
const strippedContent = stripInternalToolEnvelopeText(msgRecord.content);
|
||||
const nativeReasoning = getReadableReasoningValue(msgRecord);
|
||||
const { content, thinking } =
|
||||
options.parseTextualReasoningTags === true && !nativeReasoning
|
||||
? extractThinkingFromContent(strippedContent)
|
||||
: { content: strippedContent, thinking: null };
|
||||
const sanitized: JsonRecord = { content: collapseExcessiveNewlines(content) };
|
||||
if (thinking) sanitized.reasoning_content = thinking;
|
||||
return sanitized;
|
||||
}
|
||||
|
||||
copyOpenAICompatibleReasoningFields(msgRecord, sanitized);
|
||||
return msgRecord.content !== undefined ? { content: msgRecord.content } : {};
|
||||
}
|
||||
|
||||
function applyTextualToolCallSanitization(sanitized: JsonRecord, msgRecord: JsonRecord): void {
|
||||
const textualToolCall = parseTextualToolCallContent(sanitized.content);
|
||||
if (textualToolCall && !msgRecord.tool_calls) {
|
||||
sanitized.content = null;
|
||||
@@ -504,13 +531,26 @@ function sanitizeMessage(msg: unknown): unknown {
|
||||
} else if (containsTextualToolCallContent(sanitized.content) && !msgRecord.tool_calls) {
|
||||
sanitized.content = null;
|
||||
}
|
||||
}
|
||||
|
||||
function sanitizeMessage(msg: unknown, options: ParseOptions = {}): unknown {
|
||||
const msgRecord = toRecord(msg);
|
||||
if (!msgRecord) return msg;
|
||||
|
||||
const sanitized: JsonRecord = {};
|
||||
|
||||
if (msgRecord.role) sanitized.role = msgRecord.role;
|
||||
if (msgRecord.refusal !== undefined) sanitized.refusal = msgRecord.refusal;
|
||||
|
||||
Object.assign(sanitized, sanitizeMessageContent(msgRecord, options));
|
||||
|
||||
copyOpenAICompatibleReasoningFields(msgRecord, sanitized);
|
||||
applyTextualToolCallSanitization(sanitized, msgRecord);
|
||||
|
||||
// Preserve tool_calls
|
||||
if (msgRecord.tool_calls) {
|
||||
sanitized.tool_calls = msgRecord.tool_calls;
|
||||
}
|
||||
|
||||
// Preserve function_call (legacy)
|
||||
if (msgRecord.function_call) {
|
||||
sanitized.function_call = msgRecord.function_call;
|
||||
}
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
* Context Manager — Phase 4
|
||||
*
|
||||
* Pre-flight context compression to prevent "prompt too long" errors.
|
||||
* 3 layers: trim tool messages, compress thinking, aggressive purification.
|
||||
* 3 layers: trim tool messages, compress structured thinking, aggressive purification.
|
||||
*/
|
||||
|
||||
import { REGISTRY } from "../config/providerRegistry.ts";
|
||||
@@ -147,7 +147,7 @@ export function resolveComboContextLimit(options: {
|
||||
* Operates in 3 layers of increasing aggressiveness:
|
||||
*
|
||||
* Layer 1: Trim tool_result messages (truncate long outputs)
|
||||
* Layer 2: Compress thinking blocks (remove from history, keep last)
|
||||
* Layer 2: Compress structured thinking blocks (remove from history, keep last)
|
||||
* Layer 3: Aggressive purification (drop old messages until fitting)
|
||||
*
|
||||
* @param {object} body - Request body with messages[]
|
||||
@@ -194,7 +194,7 @@ export function compressContext(
|
||||
};
|
||||
}
|
||||
|
||||
// Layer 2: Compress thinking blocks (remove from non-last assistant messages)
|
||||
// Layer 2: Compress structured thinking blocks (remove from non-last assistant messages)
|
||||
messages = compressThinking(messages);
|
||||
currentTokens = estimateTokens(JSON.stringify(messages));
|
||||
stats.layers.push({ name: "compress_thinking", tokens: currentTokens });
|
||||
@@ -249,7 +249,7 @@ function trimToolMessages(messages: Record<string, unknown>[], maxChars: number)
|
||||
});
|
||||
}
|
||||
|
||||
// ─── Layer 2: Compress Thinking Blocks ──────────────────────────────────────
|
||||
// ─── Layer 2: Compress Structured Thinking Blocks ───────────────────────────
|
||||
|
||||
function compressThinking(messages: Record<string, unknown>[]) {
|
||||
// Find last assistant message index
|
||||
@@ -274,28 +274,6 @@ function compressThinking(messages: Record<string, unknown>[]) {
|
||||
return { ...msg, content: filtered };
|
||||
}
|
||||
|
||||
// Remove thinking XML tags from string content
|
||||
if (typeof msg.content === "string") {
|
||||
let cleaned = msg.content;
|
||||
for (const [start, end] of [
|
||||
["<thinking>", "</thinking>"],
|
||||
["<antThinking>", "</antThinking>"],
|
||||
]) {
|
||||
while (true) {
|
||||
const s = cleaned.indexOf(start);
|
||||
if (s === -1) break;
|
||||
const e = cleaned.indexOf(end, s + start.length);
|
||||
if (e === -1) {
|
||||
cleaned = cleaned.slice(0, s);
|
||||
break;
|
||||
}
|
||||
cleaned = cleaned.slice(0, s) + cleaned.slice(e + end.length);
|
||||
}
|
||||
}
|
||||
cleaned = cleaned.trim();
|
||||
return { ...msg, content: cleaned || "[thinking compressed]" };
|
||||
}
|
||||
|
||||
return msg;
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import { appendToolCallArgumentDelta } from "../utils/toolCallArguments.ts";
|
||||
import { shouldParseTextualReasoningTags } from "../handlers/responseSanitizer.ts";
|
||||
import * as fs from "fs";
|
||||
import * as path from "path";
|
||||
/**
|
||||
@@ -92,6 +93,7 @@ export function createResponsesApiTransformStream(logger = null, keepaliveInterv
|
||||
reasoningPartAdded: false,
|
||||
reasoningDone: false,
|
||||
inThinking: false,
|
||||
parseTextualReasoningTags: false,
|
||||
funcArgsBuf: {},
|
||||
funcNames: {},
|
||||
funcCallIds: {},
|
||||
@@ -407,6 +409,13 @@ export function createResponsesApiTransformStream(logger = null, keepaliveInterv
|
||||
const choice = parsed.choices[0];
|
||||
const idx = choice.index || 0;
|
||||
const delta = choice.delta || {};
|
||||
if (state.parseTextualReasoningTags !== true && typeof parsed.model === "string") {
|
||||
state.parseTextualReasoningTags = shouldParseTextualReasoningTags(
|
||||
undefined,
|
||||
parsed.model
|
||||
);
|
||||
}
|
||||
const parseTextualReasoningTags = state.parseTextualReasoningTags === true;
|
||||
|
||||
// Emit initial events
|
||||
if (!state.started) {
|
||||
@@ -443,41 +452,45 @@ export function createResponsesApiTransformStream(logger = null, keepaliveInterv
|
||||
emitReasoningDelta(controller, delta.reasoning_content);
|
||||
}
|
||||
|
||||
// Handle text content (may contain <think> tags)
|
||||
// Handle text content. Generic prompt-format tags are visible text;
|
||||
// only tag-native models opt into textual reasoning extraction.
|
||||
if (delta.content) {
|
||||
// Close reasoning if it was opened via native reasoning_content
|
||||
// and is still open, before emitting message content. Without this
|
||||
// the reasoning item is never closed and the message reuses the
|
||||
// reasoning output_index, producing a protocol-invalid stream.
|
||||
// Guard on !inThinking: reasoning opened via <think> tags is closed by
|
||||
// its matching </think> below — force-closing it here would snapshot a
|
||||
// partial buffer (dense output records the item at close time). (#4848 + #4906)
|
||||
if (state.reasoningId && !state.reasoningDone && !state.inThinking) {
|
||||
if (
|
||||
state.reasoningId &&
|
||||
!state.reasoningDone &&
|
||||
(!parseTextualReasoningTags || !state.inThinking)
|
||||
) {
|
||||
closeReasoning(controller);
|
||||
}
|
||||
|
||||
let content = delta.content;
|
||||
|
||||
if (content.includes("<think>")) {
|
||||
state.inThinking = true;
|
||||
content = content.replaceAll("<think>", "");
|
||||
startReasoning(controller, idx);
|
||||
}
|
||||
if (parseTextualReasoningTags) {
|
||||
if (content.includes("<think>")) {
|
||||
state.inThinking = true;
|
||||
content = content.replaceAll("<think>", "");
|
||||
startReasoning(controller, idx);
|
||||
}
|
||||
|
||||
if (content.includes("</think>")) {
|
||||
const parts = content.split("</think>");
|
||||
const thinkPart = parts[0];
|
||||
const textPart = parts.slice(1).join("</think>");
|
||||
if (content.includes("</think>")) {
|
||||
const parts = content.split("</think>");
|
||||
const thinkPart = parts[0];
|
||||
const textPart = parts.slice(1).join("</think>");
|
||||
|
||||
if (thinkPart) emitReasoningDelta(controller, thinkPart);
|
||||
closeReasoning(controller);
|
||||
state.inThinking = false;
|
||||
content = textPart;
|
||||
}
|
||||
if (thinkPart) emitReasoningDelta(controller, thinkPart);
|
||||
closeReasoning(controller);
|
||||
state.inThinking = false;
|
||||
content = textPart;
|
||||
}
|
||||
|
||||
if (state.inThinking && content) {
|
||||
emitReasoningDelta(controller, content);
|
||||
continue;
|
||||
if (state.inThinking && content) {
|
||||
emitReasoningDelta(controller, content);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
// Regular text content
|
||||
|
||||
@@ -289,8 +289,7 @@ export function translateRequest(
|
||||
// requested upstream; generic/implicit-cache OpenAI providers stay stripped.
|
||||
result = filterToOpenAIFormat(result, {
|
||||
preserveCacheControl:
|
||||
options?.preserveCacheControl === true &&
|
||||
providerHonorsOpenAIFormatCacheControl(provider),
|
||||
options?.preserveCacheControl === true && providerHonorsOpenAIFormatCacheControl(provider),
|
||||
// #4849 regression guard: keep client reasoning_content for replay providers.
|
||||
preserveReasoningContent: isReasoner,
|
||||
});
|
||||
@@ -587,6 +586,7 @@ export function initState(sourceFormat) {
|
||||
reasoningPartAdded: false,
|
||||
reasoningDone: false,
|
||||
inThinking: false,
|
||||
parseTextualReasoningTags: false,
|
||||
funcArgsBuf: {},
|
||||
funcNames: {},
|
||||
funcCallIds: {},
|
||||
|
||||
@@ -297,6 +297,9 @@ export function geminiToOpenAIResponse(chunk, state) {
|
||||
const response = chunk.response || chunk;
|
||||
if (!response) return null;
|
||||
|
||||
const modelVersion =
|
||||
typeof response.modelVersion === "string" ? response.modelVersion.toLowerCase() : "";
|
||||
const parseTextualReasoningTags = !chunk.response && !modelVersion.startsWith("antigravity/");
|
||||
const results = [];
|
||||
const candidate = response.candidates?.[0];
|
||||
|
||||
@@ -438,16 +441,18 @@ export function geminiToOpenAIResponse(chunk, state) {
|
||||
}
|
||||
|
||||
if (hasFunctionCall) {
|
||||
// Flush any still-open textual reasoning wrapper as reasoning_content BEFORE
|
||||
// the tool call. A signed native functionCall arriving while a `<thinking>`
|
||||
// (etc.) tag opened in an earlier chunk is still buffered must not silently
|
||||
// drop that buffered reasoning — flushOpenTextualReasoning emits it and clears
|
||||
// the active-tag/content buffers. (LEDGER-4 / #3821-review)
|
||||
flushOpenTextualReasoning(state, results);
|
||||
// Also drop any partial open-tag fragment buffered at a chunk boundary
|
||||
// (flushOpenTextualReasoning early-returns when only this is set), matching the
|
||||
// pre-fix branch which cleared all three buffers. (#3821-review convergence)
|
||||
state.textualReasoningTagBuffer = undefined;
|
||||
if (parseTextualReasoningTags) {
|
||||
// Flush any still-open textual reasoning wrapper as reasoning_content BEFORE
|
||||
// the tool call. A signed native functionCall arriving while a `<thinking>`
|
||||
// (etc.) tag opened in an earlier chunk is still buffered must not silently
|
||||
// drop that buffered reasoning — flushOpenTextualReasoning emits it and clears
|
||||
// the active-tag/content buffers. (LEDGER-4 / #3821-review)
|
||||
flushOpenTextualReasoning(state, results);
|
||||
// Also drop any partial open-tag fragment buffered at a chunk boundary
|
||||
// (flushOpenTextualReasoning early-returns when only this is set), matching the
|
||||
// pre-fix branch which cleared all three buffers. (#3821-review convergence)
|
||||
state.textualReasoningTagBuffer = undefined;
|
||||
}
|
||||
emitFunctionCallPart(part, state, results);
|
||||
}
|
||||
continue;
|
||||
@@ -459,7 +464,9 @@ export function geminiToOpenAIResponse(chunk, state) {
|
||||
// back to a structured OpenAI tool call so clients/tools do not see it as
|
||||
// assistant prose.
|
||||
if (part.text !== undefined && part.text !== "") {
|
||||
const afterReasoning = consumeTextualReasoningTags(part.text, state, results);
|
||||
const afterReasoning = parseTextualReasoningTags
|
||||
? consumeTextualReasoningTags(part.text, state, results)
|
||||
: part.text;
|
||||
if (!afterReasoning) continue;
|
||||
|
||||
let accumulated = (state.textualToolCallBuffer || "") + afterReasoning;
|
||||
@@ -677,7 +684,9 @@ export function geminiToOpenAIResponse(chunk, state) {
|
||||
|
||||
// Finish reason - include usage in final chunk
|
||||
if (candidate.finishReason) {
|
||||
flushOpenTextualReasoning(state, results);
|
||||
if (parseTextualReasoningTags) {
|
||||
flushOpenTextualReasoning(state, results);
|
||||
}
|
||||
|
||||
if (state.textualToolCallBuffer) {
|
||||
const remainingText = state.textualToolCallBuffer;
|
||||
|
||||
@@ -92,7 +92,6 @@ export function convertKiroToOpenAI(chunk, state) {
|
||||
const content = data.reasoningContentEvent?.content || data.content || "";
|
||||
if (!content) return null;
|
||||
|
||||
// Convert to thinking block format (Claude-style)
|
||||
const openaiChunk = {
|
||||
id: state.responseId,
|
||||
object: "chat.completion.chunk",
|
||||
@@ -103,7 +102,7 @@ export function convertKiroToOpenAI(chunk, state) {
|
||||
index: 0,
|
||||
delta: {
|
||||
...(state.chunkIndex === 0 ? { role: "assistant" } : {}),
|
||||
content: `<thinking>${content}</thinking>`,
|
||||
reasoning_content: content,
|
||||
},
|
||||
finish_reason: null,
|
||||
},
|
||||
|
||||
@@ -6,6 +6,7 @@ import { register } from "../registry.ts";
|
||||
import { FORMATS } from "../formats.ts";
|
||||
import { appendToolCallArgumentDelta } from "../../utils/toolCallArguments.ts";
|
||||
import { fallbackToolCallId } from "../helpers/toolCallHelper.ts";
|
||||
import { shouldParseTextualReasoningTags } from "../../handlers/responseSanitizer.ts";
|
||||
|
||||
function normalizeToolName(value) {
|
||||
return typeof value === "string" ? value.trim() : "";
|
||||
@@ -87,6 +88,10 @@ export function openaiToOpenAIResponsesResponse(chunk, state) {
|
||||
const choice = chunk.choices[0];
|
||||
const idx = choice.index || 0;
|
||||
const delta = choice.delta || {};
|
||||
if (state.parseTextualReasoningTags !== true && typeof chunk.model === "string") {
|
||||
state.parseTextualReasoningTags = shouldParseTextualReasoningTags(undefined, chunk.model);
|
||||
}
|
||||
const parseTextualReasoningTags = state.parseTextualReasoningTags === true;
|
||||
|
||||
// Emit initial events
|
||||
if (!state.started) {
|
||||
@@ -117,50 +122,45 @@ export function openaiToOpenAIResponsesResponse(chunk, state) {
|
||||
});
|
||||
}
|
||||
|
||||
// Handle reasoning_content
|
||||
if (delta.reasoning_content) {
|
||||
startReasoning(state, emit, idx);
|
||||
emitReasoningDelta(state, emit, delta.reasoning_content);
|
||||
}
|
||||
|
||||
// Handle text content
|
||||
if (delta.content) {
|
||||
// Close reasoning if it was opened via native reasoning_content and is
|
||||
// still open, before emitting message content. Otherwise the reasoning
|
||||
// item is never closed and the message reuses its output_index.
|
||||
// Guard on !inThinking: reasoning opened via <think> tags is closed by its
|
||||
// matching </think> below — force-closing it here would snapshot a partial
|
||||
// buffer (dense output records the item at close time). (#4848 + #4906)
|
||||
if (state.reasoningId && !state.reasoningDone && !state.inThinking) {
|
||||
if (
|
||||
state.reasoningId &&
|
||||
!state.reasoningDone &&
|
||||
(!parseTextualReasoningTags || !state.inThinking)
|
||||
) {
|
||||
closeReasoning(state, emit);
|
||||
}
|
||||
|
||||
let content = delta.content;
|
||||
|
||||
if (content.includes("<think>")) {
|
||||
state.inThinking = true;
|
||||
content = content.replaceAll("<think>", "");
|
||||
startReasoning(state, emit, idx);
|
||||
}
|
||||
if (parseTextualReasoningTags) {
|
||||
if (content.includes("<think>")) {
|
||||
state.inThinking = true;
|
||||
content = content.replaceAll("<think>", "");
|
||||
startReasoning(state, emit, idx);
|
||||
}
|
||||
|
||||
if (content.includes("</think>")) {
|
||||
const parts = content.split("</think>");
|
||||
const thinkPart = parts[0];
|
||||
const textPart = parts.slice(1).join("</think>");
|
||||
if (thinkPart) emitReasoningDelta(state, emit, thinkPart);
|
||||
closeReasoning(state, emit);
|
||||
state.inThinking = false;
|
||||
content = textPart;
|
||||
}
|
||||
if (content.includes("</think>")) {
|
||||
const parts = content.split("</think>");
|
||||
const thinkPart = parts[0];
|
||||
const textPart = parts.slice(1).join("</think>");
|
||||
if (thinkPart) emitReasoningDelta(state, emit, thinkPart);
|
||||
closeReasoning(state, emit);
|
||||
state.inThinking = false;
|
||||
content = textPart;
|
||||
}
|
||||
|
||||
if (state.inThinking && content) {
|
||||
emitReasoningDelta(state, emit, content);
|
||||
return events;
|
||||
if (state.inThinking && content) {
|
||||
emitReasoningDelta(state, emit, content);
|
||||
return events;
|
||||
}
|
||||
}
|
||||
|
||||
if (content) {
|
||||
// Use a distinct output_index for the message when reasoning was
|
||||
// emitted, so the message item does not collide with the reasoning item.
|
||||
const msgIdx = state.reasoningId ? state.reasoningIndex + 1 : idx;
|
||||
emitTextContent(state, emit, msgIdx, content);
|
||||
}
|
||||
|
||||
@@ -122,6 +122,38 @@ test("compressContext: Layer 2 — compresses thinking in old messages", () => {
|
||||
}
|
||||
});
|
||||
|
||||
test("compressContext: Layer 2 preserves prompt-format thinking tags in string content", () => {
|
||||
const body = {
|
||||
model: "test",
|
||||
messages: [
|
||||
{ role: "user", content: "q1" },
|
||||
{
|
||||
role: "assistant",
|
||||
content: "<thinking>visible prompt protocol</thinking><content>answer1</content>",
|
||||
},
|
||||
{ role: "user", content: "q2" },
|
||||
{
|
||||
role: "assistant",
|
||||
content: [
|
||||
{ type: "thinking", thinking: "lots of structured thinking here ".repeat(500) },
|
||||
{ type: "text", text: "answer2" },
|
||||
],
|
||||
},
|
||||
{ role: "user", content: "q3" },
|
||||
{ role: "assistant", content: "answer3" },
|
||||
],
|
||||
};
|
||||
const result = compressContext(body, { maxTokens: 2000, reserveTokens: 500 });
|
||||
const firstAssistant = (result.body as any).messages.find(
|
||||
(m: any) => m.role === "assistant" && typeof m.content === "string"
|
||||
);
|
||||
|
||||
assert.equal(
|
||||
firstAssistant.content,
|
||||
"<thinking>visible prompt protocol</thinking><content>answer1</content>"
|
||||
);
|
||||
});
|
||||
|
||||
test("compressContext: Layer 3 — drops old messages to fit", () => {
|
||||
const messages = [
|
||||
{ role: "system", content: "You are helpful" },
|
||||
|
||||
@@ -6,6 +6,7 @@ const {
|
||||
sanitizeOpenAIResponse,
|
||||
sanitizeResponsesApiResponse,
|
||||
sanitizeStreamingChunk,
|
||||
shouldParseTextualReasoningTags,
|
||||
} = await import("../../open-sse/handlers/responseSanitizer.ts");
|
||||
|
||||
test("extractThinkingFromContent separates think blocks from visible content", () => {
|
||||
@@ -65,7 +66,7 @@ test("sanitizeOpenAIResponse strips non-standard fields and preserves required t
|
||||
});
|
||||
});
|
||||
|
||||
test("sanitizeOpenAIResponse extracts thinking, collapses newlines, preserves reasoning_content with tool_calls, and preserves tool calls", () => {
|
||||
test("sanitizeOpenAIResponse preserves prompt-format thinking tags by default", () => {
|
||||
const sanitized = sanitizeOpenAIResponse({
|
||||
id: "chatcmpl_test",
|
||||
model: "gpt-4.1",
|
||||
@@ -75,7 +76,7 @@ test("sanitizeOpenAIResponse extracts thinking, collapses newlines, preserves re
|
||||
finish_reason: "tool_calls",
|
||||
message: {
|
||||
role: "assistant",
|
||||
content: "Hello\n\n\n<think>internal chain</think>\n\nworld",
|
||||
content: "Hello\n\n\n<think>visible protocol</think>\n\nworld",
|
||||
tool_calls: [{ id: "call_1" }],
|
||||
function_call: { name: "legacy" },
|
||||
},
|
||||
@@ -85,44 +86,76 @@ test("sanitizeOpenAIResponse extracts thinking, collapses newlines, preserves re
|
||||
|
||||
assert.equal((sanitized as any).choices[0].index, 2);
|
||||
assert.equal((sanitized as any).choices[0].finish_reason, "tool_calls");
|
||||
(assert as any).equal((sanitized as any).choices[0].message.content, "Hello\n\nworld");
|
||||
assert.equal((sanitized as any).choices[0].message.reasoning_content, "internal chain");
|
||||
(assert as any).equal(
|
||||
(sanitized as any).choices[0].message.content,
|
||||
"Hello\n\n<think>visible protocol</think>\n\nworld"
|
||||
);
|
||||
assert.equal((sanitized as any).choices[0].message.reasoning_content, undefined);
|
||||
(assert as any).deepEqual((sanitized as any).choices[0].message.tool_calls, [{ id: "call_1" }]);
|
||||
assert.deepEqual((sanitized as any).choices[0].message.function_call, { name: "legacy" });
|
||||
});
|
||||
|
||||
test("sanitizeOpenAIResponse extracts unclosed reasoning wrappers into reasoning_content", () => {
|
||||
const sanitized = sanitizeOpenAIResponse({
|
||||
model: "gpt-4.1",
|
||||
choices: [
|
||||
{
|
||||
message: {
|
||||
role: "assistant",
|
||||
content: "§54§ <thought\ninternal planning\n",
|
||||
test("sanitizeOpenAIResponse extracts textual reasoning only when explicitly enabled", () => {
|
||||
const sanitized = sanitizeOpenAIResponse(
|
||||
{
|
||||
model: "deepseek-r1",
|
||||
choices: [
|
||||
{
|
||||
message: {
|
||||
role: "assistant",
|
||||
content: "Hello\n\n\n<think>internal chain</think>\n\nworld",
|
||||
},
|
||||
},
|
||||
},
|
||||
],
|
||||
});
|
||||
],
|
||||
},
|
||||
{ parseTextualReasoningTags: true }
|
||||
);
|
||||
|
||||
assert.equal((sanitized as any).choices[0].message.content, "");
|
||||
assert.equal((sanitized as any).choices[0].message.content, "Hello\n\nworld");
|
||||
assert.equal((sanitized as any).choices[0].message.reasoning_content, "internal chain");
|
||||
});
|
||||
|
||||
test("sanitizeOpenAIResponse extracts unclosed reasoning wrappers only when enabled", () => {
|
||||
const sanitized = sanitizeOpenAIResponse(
|
||||
{
|
||||
model: "deepseek-r1",
|
||||
choices: [
|
||||
{
|
||||
message: {
|
||||
role: "assistant",
|
||||
content: "§54§ <thought\ninternal planning\n",
|
||||
},
|
||||
},
|
||||
],
|
||||
},
|
||||
{ parseTextualReasoningTags: true }
|
||||
);
|
||||
|
||||
assert.equal(((sanitized as any).choices[0].message as any).content, "");
|
||||
assert.equal((sanitized as any).choices[0].message.reasoning_content, "internal planning");
|
||||
});
|
||||
|
||||
test("sanitizeOpenAIResponse preserves native reasoning_content when no visible content remains", () => {
|
||||
const sanitized = sanitizeOpenAIResponse({
|
||||
model: "gpt-4.1",
|
||||
choices: [
|
||||
{
|
||||
message: {
|
||||
role: "assistant",
|
||||
content: "<think>discard me</think>",
|
||||
reasoning_content: "provider reasoning",
|
||||
test("sanitizeOpenAIResponse preserves native reasoning_content without stripping content tags", () => {
|
||||
const sanitized = sanitizeOpenAIResponse(
|
||||
{
|
||||
model: "gpt-4.1",
|
||||
choices: [
|
||||
{
|
||||
message: {
|
||||
role: "assistant",
|
||||
content: "<think>visible protocol</think>",
|
||||
reasoning_content: "provider reasoning",
|
||||
},
|
||||
},
|
||||
},
|
||||
],
|
||||
});
|
||||
],
|
||||
},
|
||||
{ parseTextualReasoningTags: true }
|
||||
);
|
||||
|
||||
assert.equal(((sanitized as any).choices[0].message as any).content, "");
|
||||
assert.equal(
|
||||
((sanitized as any).choices[0].message as any).content,
|
||||
"<think>visible protocol</think>"
|
||||
);
|
||||
assert.equal((sanitized as any).choices[0].message.reasoning_content, "provider reasoning");
|
||||
});
|
||||
|
||||
@@ -246,7 +279,10 @@ test("sanitizeOpenAIResponse preserves OpenRouter native reasoning and signature
|
||||
assert.deepEqual((sanitized as any).choices[0].message.reasoning_details, [
|
||||
{ type: "reasoning.encrypted", data: "sig" },
|
||||
]);
|
||||
assert.equal((sanitized as any).choices[0].message.content, "<content>Visible answer</content>");
|
||||
assert.equal(
|
||||
(sanitized as any).choices[0].message.content,
|
||||
"<thinking>tag-derived</thinking><content>Visible answer</content>"
|
||||
);
|
||||
});
|
||||
|
||||
test("sanitizeOpenAIResponse keeps reasoning_details-derived reasoning_content for reasoning-only messages", () => {
|
||||
@@ -618,6 +654,18 @@ test("sanitize functions return non-object inputs unchanged", () => {
|
||||
assert.equal(sanitizeStreamingChunk("raw text"), "raw text");
|
||||
});
|
||||
|
||||
test("shouldParseTextualReasoningTags is limited to tag-native model families", () => {
|
||||
assert.equal(shouldParseTextualReasoningTags("together", "deepseek-ai/DeepSeek-R1"), true);
|
||||
assert.equal(shouldParseTextualReasoningTags("cloudflare-ai", "@cf/qwen/qwq-32b"), true);
|
||||
assert.equal(shouldParseTextualReasoningTags("openrouter", "deepseek/deepseek-v4-pro"), false);
|
||||
assert.equal(shouldParseTextualReasoningTags("antigravity", "deepseek-r1"), false);
|
||||
assert.equal(shouldParseTextualReasoningTags(undefined, "antigravity/deepseek-r1"), false);
|
||||
assert.equal(
|
||||
shouldParseTextualReasoningTags("openai-compatible-custom", "claude-opus-4.7"),
|
||||
false
|
||||
);
|
||||
});
|
||||
|
||||
test("sanitizeOpenAIResponse converts textual pseudo tool-call content into structured tool_calls", () => {
|
||||
const sanitized = sanitizeOpenAIResponse({
|
||||
id: "chatcmpl_textual_tool_call",
|
||||
|
||||
@@ -83,9 +83,30 @@ test("createResponsesApiTransformStream converts plain chat deltas into Response
|
||||
assert.equal(doneMarker.data, "[DONE]");
|
||||
});
|
||||
|
||||
test("createResponsesApiTransformStream converts think tags into reasoning summaries", async () => {
|
||||
test("createResponsesApiTransformStream preserves prompt-format think tags by default", async () => {
|
||||
const output = await runTransformStream([
|
||||
'data: {"choices":[{"index":0,"delta":{"content":"<think>plan"}}]}\n\n',
|
||||
'data: {"id":"chatcmpl_1","model":"gpt-4.1","choices":[{"index":0,"delta":{"content":"<think>plan"}}]}\n\n',
|
||||
'data: {"choices":[{"index":0,"delta":{"content":"ning</think>answer"}}]}\n\n',
|
||||
'data: {"choices":[{"index":0,"delta":{},"finish_reason":"stop"}]}\n\n',
|
||||
]);
|
||||
|
||||
const events = parseSseOutput(output);
|
||||
const reasoningDeltas = events
|
||||
.filter((event) => event.event === "response.reasoning_summary_text.delta")
|
||||
.map((event) => JSON.parse(event.data).delta);
|
||||
const completed = JSON.parse(
|
||||
events.find((event) => event.event === "response.completed").data
|
||||
).response;
|
||||
|
||||
assert.deepEqual(reasoningDeltas, []);
|
||||
assert.deepEqual(completed.output[0].content, [
|
||||
{ type: "output_text", annotations: [], logprobs: [], text: "<think>planning</think>answer" },
|
||||
]);
|
||||
});
|
||||
|
||||
test("createResponsesApiTransformStream extracts think tags for tag-native models", async () => {
|
||||
const output = await runTransformStream([
|
||||
'data: {"id":"chatcmpl_1","model":"deepseek-ai/DeepSeek-R1","choices":[{"index":0,"delta":{"content":"<think>plan"}}]}\n\n',
|
||||
'data: {"choices":[{"index":0,"delta":{"content":"ning</think>answer"}}]}\n\n',
|
||||
'data: {"choices":[{"index":0,"delta":{},"finish_reason":"stop"}]}\n\n',
|
||||
]);
|
||||
@@ -99,11 +120,7 @@ test("createResponsesApiTransformStream converts think tags into reasoning summa
|
||||
).response;
|
||||
|
||||
assert.deepEqual(reasoningDeltas, ["plan", "ning"]);
|
||||
assert.deepEqual(completed.output[0], {
|
||||
id: completed.output[0].id,
|
||||
type: "reasoning",
|
||||
summary: [{ type: "summary_text", text: "planning" }],
|
||||
});
|
||||
assert.equal(completed.output[0].type, "reasoning");
|
||||
assert.deepEqual(completed.output[1].content, [
|
||||
{ type: "output_text", annotations: [], logprobs: [], text: "answer" },
|
||||
]);
|
||||
|
||||
@@ -510,7 +510,7 @@ test("Responses→Chat streaming: Copilot mode emits reasoning_text for summary
|
||||
assert.equal(result.choices[0].delta.reasoning, undefined);
|
||||
});
|
||||
|
||||
test("Chat→Responses streaming: multiple <think> tags in one chunk handled", () => {
|
||||
test("Chat→Responses streaming: generic prompt-format <think> tags remain text", () => {
|
||||
const state = initState(FORMATS.OPENAI_RESPONSES);
|
||||
|
||||
// Chunk with multiple think tags
|
||||
@@ -523,14 +523,44 @@ test("Chat→Responses streaming: multiple <think> tags in one chunk handled", (
|
||||
},
|
||||
],
|
||||
id: "c1",
|
||||
model: "gpt-4.1",
|
||||
};
|
||||
const events = openaiToOpenAIResponsesResponse(chunk, state);
|
||||
// Should not have literal <think> in any text delta
|
||||
const textDeltas = events
|
||||
.filter((e) => e.event === "response.output_text.delta")
|
||||
.map((e) => e.data.delta);
|
||||
const combined = textDeltas.join("");
|
||||
assert.ok(!combined.includes("<think>"), `text should not contain <think> tag, got: ${combined}`);
|
||||
assert.equal(combined, "<think>first</think>middle<think>second</think>end");
|
||||
assert.equal(
|
||||
events.some((e) => e.event === "response.reasoning_summary_text.delta"),
|
||||
false
|
||||
);
|
||||
});
|
||||
|
||||
test("Chat→Responses streaming: tag-native models still split <think> tags", () => {
|
||||
const state = initState(FORMATS.OPENAI_RESPONSES);
|
||||
|
||||
const chunk = {
|
||||
choices: [
|
||||
{
|
||||
index: 0,
|
||||
delta: { content: "<think>first</think>end" },
|
||||
finish_reason: null,
|
||||
},
|
||||
],
|
||||
id: "c1",
|
||||
model: "deepseek-r1",
|
||||
};
|
||||
const events = openaiToOpenAIResponsesResponse(chunk, state);
|
||||
const textDeltas = events
|
||||
.filter((e) => e.event === "response.output_text.delta")
|
||||
.map((e) => e.data.delta);
|
||||
const reasoningDeltas = events
|
||||
.filter((e) => e.event === "response.reasoning_summary_text.delta")
|
||||
.map((e) => e.data.delta);
|
||||
|
||||
assert.deepEqual(reasoningDeltas, ["first"]);
|
||||
assert.equal(textDeltas.join(""), "end");
|
||||
});
|
||||
|
||||
// Regression: a tool call was announced (response.output_item.added set currentToolCallId)
|
||||
|
||||
@@ -0,0 +1,75 @@
|
||||
import test from "node:test";
|
||||
import assert from "node:assert/strict";
|
||||
|
||||
const { geminiToOpenAIResponse } =
|
||||
await import("../../open-sse/translator/response/gemini-to-openai.ts");
|
||||
|
||||
function createStreamingState() {
|
||||
return {
|
||||
toolCalls: new Map(),
|
||||
};
|
||||
}
|
||||
|
||||
test("Antigravity stream preserves prompt-format thinking tags as content", () => {
|
||||
const state = createStreamingState();
|
||||
const first = geminiToOpenAIResponse(
|
||||
{
|
||||
response: {
|
||||
responseId: "resp-ag-visible-thinking",
|
||||
modelVersion: "antigravity/gemini-3-pro",
|
||||
candidates: [{ content: { parts: [{ text: "<thinking>\n[metacognition" }] } }],
|
||||
},
|
||||
},
|
||||
state
|
||||
);
|
||||
const second = geminiToOpenAIResponse(
|
||||
{
|
||||
response: {
|
||||
responseId: "resp-ag-visible-thinking",
|
||||
modelVersion: "antigravity/gemini-3-pro",
|
||||
candidates: [{ content: { parts: [{ text: "]\n\nVisible answer" }] } }],
|
||||
},
|
||||
},
|
||||
state
|
||||
);
|
||||
|
||||
const deltas = [...first, ...second].map((event: any) => event.choices?.[0]?.delta || {});
|
||||
assert.deepEqual(
|
||||
deltas.filter((delta: any) => delta.content).map((delta: any) => delta.content),
|
||||
["<thinking>\n[metacognition", "]\n\nVisible answer"]
|
||||
);
|
||||
assert.equal(
|
||||
deltas.some((delta: any) => delta.reasoning_content !== undefined),
|
||||
false
|
||||
);
|
||||
});
|
||||
|
||||
test("Antigravity stream keeps native Gemini thought parts as reasoning_content", () => {
|
||||
const result = geminiToOpenAIResponse(
|
||||
{
|
||||
response: {
|
||||
responseId: "resp-ag-native-thought",
|
||||
modelVersion: "antigravity/gemini-3-pro",
|
||||
candidates: [
|
||||
{
|
||||
content: {
|
||||
parts: [{ thought: true, text: "Native plan" }, { text: "Visible answer" }],
|
||||
},
|
||||
finishReason: "STOP",
|
||||
},
|
||||
],
|
||||
},
|
||||
},
|
||||
createStreamingState()
|
||||
);
|
||||
|
||||
assert.equal(
|
||||
result.find((event: any) => event.choices?.[0]?.delta?.reasoning_content)?.choices[0].delta
|
||||
.reasoning_content,
|
||||
"Native plan"
|
||||
);
|
||||
assert.equal(
|
||||
result.find((event: any) => event.choices?.[0]?.delta?.content)?.choices[0].delta.content,
|
||||
"Visible answer"
|
||||
);
|
||||
});
|
||||
@@ -26,13 +26,14 @@ test("Kiro -> OpenAI: subsequent assistantResponseEvent omits role", () => {
|
||||
assert.equal(result.choices[0].delta.content, "lo");
|
||||
});
|
||||
|
||||
test("Kiro -> OpenAI: reasoningContentEvent is wrapped as thinking tags", () => {
|
||||
test("Kiro -> OpenAI: reasoningContentEvent emits native reasoning_content", () => {
|
||||
const result = convertKiroToOpenAI(
|
||||
'event:reasoningContentEvent\ndata:{"content":"Need to inspect first"}\n\n',
|
||||
{}
|
||||
);
|
||||
|
||||
assert.equal(result.choices[0].delta.content, "<thinking>Need to inspect first</thinking>");
|
||||
assert.equal(result.choices[0].delta.reasoning_content, "Need to inspect first");
|
||||
assert.equal(result.choices[0].delta.content, undefined);
|
||||
});
|
||||
|
||||
test("Kiro -> OpenAI: toolUseEvent becomes OpenAI tool_calls", () => {
|
||||
|
||||
@@ -112,7 +112,7 @@ test("OpenAI -> Responses: flush on null closes text content and emits response.
|
||||
assert.ok(events.some((event) => event.event === "response.completed"));
|
||||
});
|
||||
|
||||
test("OpenAI -> Responses: <think> tags become reasoning events and normal text still streams", () => {
|
||||
test("OpenAI -> Responses: prompt-format <think> tags remain text by default", () => {
|
||||
const events = collectEvents([
|
||||
{
|
||||
id: "chatcmpl-3",
|
||||
@@ -127,6 +127,34 @@ test("OpenAI -> Responses: <think> tags become reasoning events and normal text
|
||||
},
|
||||
]);
|
||||
|
||||
assert.equal(
|
||||
events.some((event) => event.event === "response.reasoning_summary_text.delta"),
|
||||
false
|
||||
);
|
||||
assert.ok(
|
||||
events.some(
|
||||
(event) =>
|
||||
event.event === "response.output_text.delta" &&
|
||||
event.data.delta === "<think>Plan it</think>Done."
|
||||
)
|
||||
);
|
||||
});
|
||||
|
||||
test("OpenAI -> Responses: tag-native models still emit <think> text as reasoning", () => {
|
||||
const events = collectEvents([
|
||||
{
|
||||
id: "chatcmpl-3b",
|
||||
model: "Qwen/QwQ-32B",
|
||||
choices: [
|
||||
{
|
||||
index: 0,
|
||||
delta: { content: "<think>Plan it</think>Done." },
|
||||
finish_reason: "stop",
|
||||
},
|
||||
],
|
||||
},
|
||||
]);
|
||||
|
||||
assert.ok(
|
||||
events.some(
|
||||
(event) =>
|
||||
|
||||
Reference in New Issue
Block a user