Files
OmniRoute/tests/unit/context-manager.test.mjs
Diego Rodrigues de Sa e Souza b100325fe0 chore(release): v3.5.2 — Qoder DashScope Native Integration & Stability (#999)
* feat(qoder): native cosy integration

* feat(qoder): implement native COSY encryption algorithm and remove CLI child instances, plus workflow bumps

* feat(resilience): context overflow fallback, OAuth token detection, empty content guard & context-optimized combo strategy

- Add isContextOverflowError + isContextOverflow detectors (400 + token-limit signals)
- Auto-fallback to next family model on context overflow in chatCore
- Add isEmptyContentResponse to catch fake-success empty responses, trigger fallback + recursive retry
- Add OAUTH_INVALID_TOKEN error type (T11) with isOAuthInvalidToken signal matching; warn instead of deactivating node
- Add getModelContextLimit helper in modelsDevSync (reads limit_context from synced capabilities)
- Upgrade getTokenLimit in contextManager to check models.dev DB before registry (fixes gemini-2.5-pro: 1000000→1048576)
- Add findLargerContextModel in modelFamilyFallback for context-aware model selection
- Add sortModelsByContextSize + context-optimized combo strategy in combo.ts
- Update context-manager unit test for corrected gemini-2.5-pro limit

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>

* fix(review): address Gemini code review — tool_calls path, infinite recursion, dedup signals, findLargerContextModel

- Fix isEmptyContentResponse: check message.tool_calls/delta.tool_calls instead
  of firstChoice.tool_calls (wrong OpenAI API path, caused tool-call responses
  to be falsely flagged as empty)
- Fix empty content fallback: replace recursive handleChatCore call (infinite
  recursion risk + wrong model due to original body.model) with non-recursive
  pattern — call executeProviderRequest, parse fallback response body, reassign
  responseBody and fall through to existing processing
- Fix context overflow: use findLargerContextModel over family candidates first,
  fall back to getNextFamilyFallback — ensures we pick a model with actually
  larger context window on overflow
- Fix signal dedup: export CONTEXT_OVERFLOW_SIGNALS + CONTEXT_OVERFLOW_REGEX
  from errorClassifier.ts; import shared regex in modelFamilyFallback.ts,
  removing duplicate signal list and per-call RegExp construction

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>

* fix(UI): add context-optimized strategy to frontend schema and options

* fix(sse): preserve Responses API events in stream translation

When translating Claude-format responses (e.g. GLM) to Responses API
format for Codex CLI, the sanitizer stripped {event, data} structured
items to {"object":"chat.completion.chunk"}, losing all content and
the critical response.completed event.

Only run sanitizeStreamingChunk on OpenAI Chat Completions chunks,
skipping items that have the Responses API {event, data} structure.

* test(sse): add regression test for Claude→Responses stream sanitization

Verifies that {event,data} structured items from the Responses API
translator bypass sanitizeStreamingChunk when translating Claude-format
providers (e.g. GLM) to Responses API format for Codex CLI.

* fix(sse): strengthen Responses API event detection with response. prefix check

Use explicit `response.` prefix check instead of generic `event && data`
presence check, as recommended in PR review.

* fix: pin Next.js to 16.0.10 to prevent Turbopack hashed module bug

Remove ^ prefix from next and eslint-config-next to prevent
automatic upgrades to 16.1.x+ which introduced content-based
hashing for external module references in Turbopack.

Also remove duplicate Material Symbols @import from globals.css
(font already loaded via <link> in layout.tsx).

Fixes #509

* align cc-compatible cache handling with client passthrough

* chore: integrate resilience and turbopack fixes (PRs #992, #990, #987)

* chore(release): bump to v3.5.2 — changelog, docs, version sync

* docs(i18n): sync documentation updates to 33 languages

* fix(qoder): replace any with unknown to comply with strict any-budget

---------

Co-authored-by: diegosouzapw <diegosouzapw@users.noreply.github.com>
Co-authored-by: oyi77 <oyi77@users.noreply.github.com>
Co-authored-by: Claude Sonnet 4.6 <noreply@anthropic.com>
Co-authored-by: Chris Staley <christopher-s@users.noreply.github.com>
Co-authored-by: Ivan <shanin-i2011@yandex.ru>
Co-authored-by: R.D. <rogerproself@gmail.com>
2026-04-05 02:54:44 -03:00

117 lines
4.2 KiB
JavaScript

import test from "node:test";
import assert from "node:assert/strict";
const { compressContext, estimateTokens, getTokenLimit } =
await import("../../open-sse/services/contextManager.ts");
// ─── estimateTokens ─────────────────────────────────────────────────────────
test("estimateTokens: estimates from string", () => {
assert.equal(estimateTokens("hello"), 2); // 5/4 = 2
assert.ok(estimateTokens("a".repeat(100)) === 25);
});
test("estimateTokens: handles null", () => {
assert.equal(estimateTokens(null), 0);
assert.equal(estimateTokens(""), 0);
});
// ─── getTokenLimit ──────────────────────────────────────────────────────────
test("getTokenLimit: detects claude", () => {
assert.equal(getTokenLimit("claude", "claude-sonnet-4"), 200000);
});
test("getTokenLimit: detects gemini", () => {
assert.equal(getTokenLimit("gemini", "gemini-2.5-pro"), 1048576);
});
test("getTokenLimit: default fallback", () => {
assert.equal(getTokenLimit("unknown"), 128000);
});
// ─── compressContext ────────────────────────────────────────────────────────
test("compressContext: returns unchanged if fits", () => {
const body = {
model: "claude-sonnet-4",
messages: [
{ role: "system", content: "You are helpful." },
{ role: "user", content: "Hello" },
],
};
const result = compressContext(body);
assert.equal(result.compressed, false);
});
test("compressContext: handles null/empty body", () => {
assert.equal(compressContext(null).compressed, false);
assert.equal(compressContext({}).compressed, false);
assert.equal(compressContext({ messages: null }).compressed, false);
});
test("compressContext: Layer 1 — trims long tool messages", () => {
const longContent = "x".repeat(10000);
const body = {
model: "test",
messages: [
{ role: "user", content: "run tool" },
{ role: "tool", content: longContent, tool_call_id: "t1" },
{ role: "user", content: "done?" },
],
};
// Use very tight limit to force compression
const result = compressContext(body, { maxTokens: 500, reserveTokens: 100 });
assert.ok(result.compressed);
const toolMsg = result.body.messages.find((m) => m.role === "tool");
assert.ok(toolMsg.content.length < longContent.length);
assert.ok(toolMsg.content.includes("[truncated]"));
});
test("compressContext: Layer 2 — compresses thinking in old messages", () => {
const body = {
model: "test",
messages: [
{ role: "user", content: "q1" },
{
role: "assistant",
content: [
{ type: "thinking", thinking: "lots of thinking here ".repeat(500) },
{ type: "text", text: "answer1" },
],
},
{ role: "user", content: "q2" },
{
role: "assistant",
content: [
{ type: "thinking", thinking: "more thinking" },
{ type: "text", text: "answer2" },
],
},
],
};
const result = compressContext(body, { maxTokens: 2000, reserveTokens: 500 });
// First assistant should have thinking removed
const firstAssistant = result.body.messages.find((m) => m.role === "assistant");
if (Array.isArray(firstAssistant.content)) {
const hasThinking = firstAssistant.content.some((b) => b.type === "thinking");
assert.equal(hasThinking, false);
}
});
test("compressContext: Layer 3 — drops old messages to fit", () => {
const messages = [
{ role: "system", content: "You are helpful" },
...Array.from({ length: 100 }, (_, i) => [
{ role: "user", content: `Message ${i}: ${"content ".repeat(50)}` },
{ role: "assistant", content: `Response ${i}: ${"answer ".repeat(50)}` },
]).flat(),
];
const body = { model: "test", messages };
const result = compressContext(body, { maxTokens: 3000, reserveTokens: 500 });
assert.ok(result.compressed);
assert.ok(result.body.messages.length < messages.length);
// System message preserved
assert.equal(result.body.messages[0].role, "system");
});