mirror of
https://github.com/diegosouzapw/OmniRoute.git
synced 2026-08-14 03:02:14 +03:00
Closed #9183 in favor of this branch at the operator's request. Folds in the exact PR diff (verified byte-identical via `gh pr diff 9183`, applied cleanly with `git apply`): - Fixed output_index collisions between reasoning/message/tool-call items in Responses API streaming, which caused clients to ignore tool calls. - Fixed a race in chunk processing where tool calls and finish signals were lost when they arrived in the same chunk as reasoning/text content. - Enabled full processing of multi-choice chunks (previously truncated to a single choice). - Added reasoning_content capture/replay for DeepSeek-based models (big-pickle), keyed off the real position in the next turn's replayed messages array instead of a hardcoded index 0 — prevents history corruption that caused models to lose context and stop prematurely. - Added big-pickle to models recognized for native textual reasoning tags so <think> blocks convert to reasoning items correctly. - Improved Responses-to-Chat translation to group reasoning, content, and tool calls into a single assistant turn, as strict OpenAI-compatible upstreams require. - Added a descriptive placeholder for encrypted Responses API reasoning blocks so downgraded Chat API requests keep context. Test plan: - All 82 tests across reasoning-cache.test.ts, translator-helper-branches.test.ts, translator-request-openai-responses.test.ts, and the 3 new test files (responses-api-truncation.test.ts, responses-replay-fixes.test.ts, responses-request-translation.test.ts) pass - npm run typecheck:core — clean - npm run check:file-size — chatCore.ts (5024->5032) and translator/response/openai-responses.ts (1174->1180) rebaselined, both cohesive additions at the two reasoning-cache capture call sites and the turn-grouping fix; tests/unit/reasoning-cache.test.ts rebaselined at 1035 (crosses the 1000-line new-file test cap, entirely this fold's diff)
109 lines
3.7 KiB
TypeScript
109 lines
3.7 KiB
TypeScript
/**
|
|
* Thinking-mode upstreams (DeepSeek V4 Flash, Kimi, MiniMax, xiaomi-tokenplan
|
|
* mimo, ...) require
|
|
* `reasoning_content` to be echoed back on every assistant message in the
|
|
* conversation history. Standard OpenAI clients do not preserve that field
|
|
* across turns, so we inject a non-empty placeholder before forwarding.
|
|
*
|
|
* Without the placeholder these upstreams return:
|
|
* 400 Bad Request — reasoning_content must be passed back
|
|
*
|
|
* Ported from decolua/9router#1099 (issue #1543). Pure helper — no I/O, no
|
|
* cross-module deps — kept narrow so it can be reused by other meta-providers
|
|
* that proxy to thinking-mode models.
|
|
*/
|
|
|
|
const PLACEHOLDER = " ";
|
|
|
|
type JsonRecord = Record<string, unknown>;
|
|
|
|
/**
|
|
* Model-id predicates for thinking-mode families that need the echo.
|
|
* Matched case-insensitively against the resolved model id (post upstream
|
|
* routing, e.g. `oc/deepseek-v4-flash-free` for the OpenCode meta-provider).
|
|
*/
|
|
const THINKING_MODEL_PATTERNS: RegExp[] = [
|
|
/deepseek/i,
|
|
/\bkimi\b/i,
|
|
/\bk2\b/i, // moonshot kimi k2 family alias
|
|
/\bminimax\b/i,
|
|
/\bmimo\b/i, // xiaomi-tokenplan mimo family (e.g. xiaomi-tokenplan/mimo-v2.5-pro)
|
|
/\bbig-pickle\b/i,
|
|
];
|
|
|
|
const AUTHENTIC_REASONING_MODEL_PATTERN = /(?:^|\/)kimi-k(?:3|2\.7-code)(?:$|-)/i;
|
|
|
|
/**
|
|
* Native Moonshot K3/K2.7 replay must use the original reasoning content.
|
|
* A fabricated placeholder changes preserved-thinking history and is not a
|
|
* valid substitute when the client and reasoning cache both lack the field.
|
|
*/
|
|
export function requiresAuthenticReasoningContent(provider: unknown, model: unknown): boolean {
|
|
const normalizedProvider = String(provider ?? "")
|
|
.trim()
|
|
.toLowerCase();
|
|
const normalizedModel = String(model ?? "").trim();
|
|
return (
|
|
(normalizedProvider === "moonshot" || normalizedProvider === "kimi") &&
|
|
AUTHENTIC_REASONING_MODEL_PATTERN.test(normalizedModel)
|
|
);
|
|
}
|
|
|
|
export function isThinkingMessageModel(model: string | undefined | null): boolean {
|
|
if (!model || typeof model !== "string") return false;
|
|
return THINKING_MODEL_PATTERNS.some((re) => re.test(model));
|
|
}
|
|
|
|
export function shouldInjectReasoningContentPlaceholder(
|
|
provider: unknown,
|
|
model: string | undefined | null
|
|
): boolean {
|
|
const normalizedProvider = String(provider ?? "")
|
|
.trim()
|
|
.toLowerCase();
|
|
return (
|
|
(normalizedProvider === "moonshot" || normalizedProvider === "kimi") &&
|
|
!requiresAuthenticReasoningContent(normalizedProvider, model) &&
|
|
isThinkingMessageModel(model)
|
|
);
|
|
}
|
|
|
|
function hasNonEmptyReasoningContent(message: JsonRecord): boolean {
|
|
return (
|
|
typeof message.reasoning_content === "string" &&
|
|
(message.reasoning_content as string).trim().length > 0
|
|
);
|
|
}
|
|
|
|
function isAssistantMessage(value: unknown): value is JsonRecord {
|
|
return (
|
|
!!value &&
|
|
typeof value === "object" &&
|
|
!Array.isArray(value) &&
|
|
(value as JsonRecord).role === "assistant"
|
|
);
|
|
}
|
|
|
|
/**
|
|
* Inject a placeholder `reasoning_content` on every assistant message in
|
|
* `body.messages` that lacks one. Returns the original object if no mutation
|
|
* was needed, or a shallow-copied body with a new messages array otherwise.
|
|
*
|
|
* No-op when the body shape is unexpected (defensive).
|
|
*/
|
|
export function injectReasoningContentForThinkingModel(body: unknown): unknown {
|
|
if (!body || typeof body !== "object" || Array.isArray(body)) return body;
|
|
const record = body as JsonRecord;
|
|
if (!Array.isArray(record.messages)) return body;
|
|
|
|
let modified = false;
|
|
const messages = record.messages.map((message) => {
|
|
if (!isAssistantMessage(message)) return message;
|
|
if (hasNonEmptyReasoningContent(message)) return message;
|
|
modified = true;
|
|
return { ...message, reasoning_content: PLACEHOLDER };
|
|
});
|
|
|
|
return modified ? { ...record, messages } : body;
|
|
}
|