mirror of
https://github.com/diegosouzapw/OmniRoute.git
synced 2026-07-31 04:12:10 +03:00
* test: resolve typescript strictness complaints in unit tests * Update Claude Code obfuscation to version 2.1.114 (#1403) * fix(cloud-code): scope thinking stripping to executor boundaries (#1401) * fix(cloud-code): scope thinking stripping to executors * fix(cloud-code): guard antigravity normalized body * Update Claude Code obfuscation to version 2.1.114 - Update Claude Code version from 2.1.87 to 2.1.114 - Update X-Stainless-Package-Version from 0.80.0 to 0.81.0 - Add new beta flags: redact-thinking-2026-02-12, advisor-tool-2026-03-01, advanced-tool-use-2025-11-20 - Add missing headers: anthropic-version, anthropic-dangerous-direct-browser-access, x-app, X-Stainless-Timeout - Add all X-Stainless-* headers (Arch, Lang, OS, Runtime, Runtime-Version, Retry-Count) - Fix accept-encoding header: identity -> gzip, deflate, br, zstd - Add connection: keep-alive header - Update tool name mapping: add lsp, apply_patch, websearch These changes ensure that requests from OpenCode through Omniroute are indistinguishable from genuine Claude Code 2.1.114 requests, allowing proper authentication with Anthropic's API without triggering extra credits errors. * fix: resolve CodeQL password hash alert and TruffleHog CI failure --------- Co-authored-by: Randi <55005611+rdself@users.noreply.github.com> Co-authored-by: Diego Rodrigues de Sa e Souza <8016841+diegosouzapw@users.noreply.github.com> Co-authored-by: Nikolay Popov <ekklesio.dev@gmail.com> Co-authored-by: diegosouzapw <diegosouzapw@users.noreply.github.com> * fix(claude-code): scope obfuscation to cli clients and fix tests * docs(workflows): enforce PR merge instead of manual close * docs(changelog): update 3.6.9 notes with missing PR 1403 and fixes * docs(workflows): update generate-release to use full changelog for PR body * fix(tsc): silence baseUrl deprecation warnings for TS 5.5+ * fix(chatcore): apply proactive compression before provider translation (#1406) Integrated into release/v3.6.9 * docs(changelog): add PR 1406 * Makes text visible in dark-mode (#1409) Integrated into release/v3.6.9 * docs(changelog): add PR 1409 * chore: save local work * chore(release): sync version references to 3.6.9 * fix(codex): prevent proactive token refresh consumption and strip background parameter * ci: shard long-running suites and relax timeouts * ci: allow manual CI dispatch for release branches * feat(skills): provider-aware marketplace UX, scored AUTO injection, and memory pipeline hardening (#1411) * fix/400 for GeminiCLI(add "ref" in GEMINI_UNSUPPORTED_SCHEMA_KEYS) * feat(cc-compatible): align request shape with Claude CLI * fix(cc-compatible): add Claude CLI system skeleton for OpenAI input * preserve reasoning when translating chat to responses (#1414) Integrated into release/v3.6.9 * fix(skills): optimize AUTO scoring and include Responses input context (#1418) Integrated into release/v3.6.9 * chore: fix TS errors and update review-prs workflow * fix(api): stop sending unsupported Gemini and Codex parameters Prevent Gemini request translation from injecting default thoughtSignature values that the upstream API strictly validates and rejects. Only preserve real signatures resolved from prior upstream responses, and strip additionalProperties from Gemini function schemas to avoid 400 "Unknown name" errors. Also remove fallback-injected session_id and conversation_id fields before sending Codex requests, and restore compatibility with the legacy OUTBOUND_SSRF_GUARD_ENABLED flag when determining whether private provider URLs are allowed. Updates the Gemini translator and regression tests for issue #1410 and related 400 error cases. * fix(core): stabilization fixes for token refresh, usage translation, and testing - Update Codex token refresh detection logic - Mark provider connections invalid on unrecoverable refresh error - Fix Claude usage translation under-reporting cached tokens - Update test expectations - Update CHANGELOG.md for v3.6.9 * fix(auth): reload fresh token state and unify expiry persistence Refresh checks now re-read the latest stored provider connection before attempting rotation so they do not use stale refresh tokens captured by an earlier sweep. Token updates also persist both expiresAt and tokenExpiresAt across the health check, usage-limit refresh path, and SSE refresh flow. This keeps known token expiry metadata in sync and avoids interval-based refreshes for connections whose tokens are still valid well into the future. * fix: resolve SSRF environment static evaluation bug (#1427) Fix import aliases and strict TS typings for tests and ACP agents. * test: resolve remaining strict type errors in test files * test: fix provider service assertion for anthropic-compatible header * fix(codex): respect openaiStoreEnabled setting during native passthrough (#1432) * fix(codex): fix token refresh unrecoverable detection for expired tokens * fix(ci): restore release v3.6.9 build and flaky tests * fix(cc-compatible): trim default OpenAI system skeleton (#1433) Integrated into release/v3.6.9 * fix: prevent masked API keys from being written to CLI tool configs (#1435) * feat: mark Qwen provider as deprecated and add deprecation warning to CLI tool (#1437) * docs(changelog): comprehensive v3.6.9 update with all 59 commits since v3.6.8 * test(ci): align qwen guide settings assertions * fix(security): resolve CodeQL alert 163 for incomplete URL sanitization in Qwen CLI settings --------- Co-authored-by: diegosouzapw <diegosouzapw@users.noreply.github.com> Co-authored-by: Nikolay Popov <74762779+nikolay-popov-ideogram@users.noreply.github.com> Co-authored-by: Randi <55005611+rdself@users.noreply.github.com> Co-authored-by: Nikolay Popov <ekklesio.dev@gmail.com> Co-authored-by: Paijo <14921983+oyi77@users.noreply.github.com> Co-authored-by: Tim Massey <tim-massey@users.noreply.github.com> Co-authored-by: Paijo <oyi77@users.noreply.github.com> Co-authored-by: dail45 <dail45@yandex.ru> Co-authored-by: R.D. <rogerproself@gmail.com>
314 lines
8.3 KiB
TypeScript
314 lines
8.3 KiB
TypeScript
/**
|
|
* Cache Control Policy
|
|
*
|
|
* Determines when to preserve client-side prompt caching headers (cache_control)
|
|
* vs. applying OmniRoute's own caching strategy.
|
|
*
|
|
* Client-side caching (e.g., Claude Code) should be preserved when:
|
|
* 1. Client is Claude Code or similar caching-aware client
|
|
* 2. Request will hit a deterministic target (single model or deterministic combo strategy)
|
|
* 3. Provider supports prompt caching (Anthropic, Alibaba Qwen, etc.)
|
|
*/
|
|
|
|
import type { RoutingStrategyValue } from "../../src/shared/constants/routingStrategies";
|
|
|
|
/**
|
|
* Cache control preservation modes
|
|
*/
|
|
export type CacheControlMode = "auto" | "always" | "never";
|
|
|
|
/**
|
|
* Cache control settings from the database
|
|
*/
|
|
export interface CacheControlSettings {
|
|
alwaysPreserveClientCache?: CacheControlMode;
|
|
}
|
|
|
|
/**
|
|
* Cache metrics for tracking effectiveness
|
|
*/
|
|
export interface CacheControlMetrics {
|
|
// Totals
|
|
totalRequests: number;
|
|
requestsWithCacheControl: number;
|
|
|
|
// Token counts
|
|
totalInputTokens: number;
|
|
totalCachedTokens: number;
|
|
totalCacheCreationTokens: number;
|
|
|
|
// Savings
|
|
tokensSaved: number;
|
|
estimatedCostSaved: number;
|
|
|
|
// Breakdowns
|
|
byProvider: Record<
|
|
string,
|
|
{
|
|
requests: number;
|
|
inputTokens: number;
|
|
cachedTokens: number;
|
|
cacheCreationTokens: number;
|
|
}
|
|
>;
|
|
byStrategy: Record<
|
|
string,
|
|
{
|
|
requests: number;
|
|
inputTokens: number;
|
|
cachedTokens: number;
|
|
cacheCreationTokens: number;
|
|
}
|
|
>;
|
|
|
|
lastUpdated: string;
|
|
}
|
|
|
|
/**
|
|
* Routing strategies that are deterministic (same request → same provider)
|
|
*/
|
|
const DETERMINISTIC_STRATEGIES: Set<RoutingStrategyValue> = new Set(["priority", "cost-optimized"]);
|
|
|
|
/**
|
|
* Providers that support prompt caching
|
|
*/
|
|
const CACHING_PROVIDERS = new Set(["claude", "anthropic", "zai", "qwen", "deepseek"]);
|
|
|
|
/**
|
|
* Detect if the client is Claude Code or another caching-aware client
|
|
*/
|
|
export function isClaudeCodeClient(userAgent: string | null | undefined): boolean {
|
|
if (!userAgent) return false;
|
|
const ua = userAgent.toLowerCase();
|
|
|
|
// Claude Code user agents
|
|
if (ua.includes("claude-code") || ua.includes("claude_code")) return true;
|
|
if (ua.includes("claude-cli/")) return true;
|
|
if (ua.includes("sdk-cli")) return true;
|
|
if (ua.includes("anthropic") && ua.includes("cli")) return true;
|
|
|
|
return false;
|
|
}
|
|
|
|
/**
|
|
* Check if a provider supports prompt caching
|
|
* Supports caching if:
|
|
* 1. Provider is in the known caching providers list, OR
|
|
* 2. Provider uses Claude protocol (detected via targetFormat)
|
|
*/
|
|
export function providerSupportsCaching(
|
|
provider: string | null | undefined,
|
|
targetFormat?: string | null
|
|
): boolean {
|
|
if (!provider) return false;
|
|
if (CACHING_PROVIDERS.has(provider.toLowerCase())) return true;
|
|
// All Claude-protocol providers support prompt caching
|
|
if (targetFormat === "claude") return true;
|
|
return false;
|
|
}
|
|
|
|
/**
|
|
* Check if a routing strategy is deterministic
|
|
*/
|
|
export function isDeterministicStrategy(
|
|
strategy: RoutingStrategyValue | null | undefined
|
|
): boolean {
|
|
if (!strategy) return false;
|
|
return DETERMINISTIC_STRATEGIES.has(strategy);
|
|
}
|
|
|
|
/**
|
|
* Determine if client-side cache_control headers should be preserved
|
|
*
|
|
* @param userAgent - User-Agent header from the request
|
|
* @param isCombo - Whether this is a combo model
|
|
* @param comboStrategy - The combo's routing strategy (if applicable)
|
|
* @param targetProvider - The target provider for the request
|
|
* @param settings - Cache control settings from database (optional)
|
|
* @returns true if cache_control should be preserved, false if OmniRoute should manage it
|
|
*/
|
|
export function shouldPreserveCacheControl({
|
|
userAgent,
|
|
isCombo,
|
|
comboStrategy,
|
|
targetProvider,
|
|
targetFormat,
|
|
settings,
|
|
}: {
|
|
userAgent: string | null | undefined;
|
|
isCombo: boolean;
|
|
comboStrategy?: RoutingStrategyValue | null;
|
|
targetProvider: string | null | undefined;
|
|
targetFormat?: string | null;
|
|
settings?: CacheControlSettings;
|
|
}): boolean {
|
|
// User override takes precedence
|
|
if (settings?.alwaysPreserveClientCache === "always") {
|
|
return true;
|
|
}
|
|
if (settings?.alwaysPreserveClientCache === "never") {
|
|
return false;
|
|
}
|
|
|
|
// Auto mode: use automatic detection (existing logic)
|
|
// Must be a caching-aware client
|
|
if (!isClaudeCodeClient(userAgent)) {
|
|
return false;
|
|
}
|
|
|
|
// Target provider must support caching
|
|
if (!providerSupportsCaching(targetProvider, targetFormat)) {
|
|
return false;
|
|
}
|
|
|
|
// Single model: always preserve (deterministic)
|
|
if (!isCombo) {
|
|
return true;
|
|
}
|
|
|
|
// Combo: only preserve if strategy is deterministic
|
|
return isDeterministicStrategy(comboStrategy);
|
|
}
|
|
|
|
/**
|
|
* Track cache control metrics for a request
|
|
*/
|
|
export function trackCacheMetrics({
|
|
preserved,
|
|
provider,
|
|
strategy,
|
|
metrics,
|
|
inputTokens,
|
|
cachedTokens,
|
|
cacheCreationTokens,
|
|
}: {
|
|
preserved: boolean;
|
|
provider: string;
|
|
strategy: string | null | undefined;
|
|
metrics: CacheControlMetrics;
|
|
inputTokens?: number;
|
|
cachedTokens?: number;
|
|
cacheCreationTokens?: number;
|
|
}): CacheControlMetrics {
|
|
const now = new Date().toISOString();
|
|
|
|
// Initialize metrics if empty
|
|
if (!metrics) {
|
|
metrics = {
|
|
totalRequests: 0,
|
|
requestsWithCacheControl: 0,
|
|
totalInputTokens: 0,
|
|
totalCachedTokens: 0,
|
|
totalCacheCreationTokens: 0,
|
|
tokensSaved: 0,
|
|
estimatedCostSaved: 0,
|
|
byProvider: {},
|
|
byStrategy: {},
|
|
lastUpdated: now,
|
|
};
|
|
}
|
|
|
|
// Increment total requests
|
|
metrics.totalRequests++;
|
|
|
|
// Track token counts
|
|
const input = inputTokens || 0;
|
|
const cached = cachedTokens || 0;
|
|
const creation = cacheCreationTokens || 0;
|
|
|
|
metrics.totalInputTokens += input;
|
|
metrics.totalCachedTokens += cached;
|
|
metrics.totalCacheCreationTokens += creation;
|
|
|
|
// Calculate tokens saved (cached tokens are reused, not charged)
|
|
if (cached > 0) {
|
|
metrics.tokensSaved += cached;
|
|
}
|
|
|
|
// Only track requests where cache_control was preserved
|
|
if (preserved) {
|
|
metrics.requestsWithCacheControl++;
|
|
|
|
// Initialize provider tracking
|
|
if (!metrics.byProvider[provider]) {
|
|
metrics.byProvider[provider] = {
|
|
requests: 0,
|
|
inputTokens: 0,
|
|
cachedTokens: 0,
|
|
cacheCreationTokens: 0,
|
|
};
|
|
}
|
|
metrics.byProvider[provider].requests++;
|
|
metrics.byProvider[provider].inputTokens += input;
|
|
metrics.byProvider[provider].cachedTokens += cached;
|
|
metrics.byProvider[provider].cacheCreationTokens += creation;
|
|
|
|
// Initialize strategy tracking
|
|
if (strategy && !metrics.byStrategy[strategy]) {
|
|
metrics.byStrategy[strategy] = {
|
|
requests: 0,
|
|
inputTokens: 0,
|
|
cachedTokens: 0,
|
|
cacheCreationTokens: 0,
|
|
};
|
|
}
|
|
if (strategy) {
|
|
metrics.byStrategy[strategy].requests++;
|
|
metrics.byStrategy[strategy].inputTokens += input;
|
|
metrics.byStrategy[strategy].cachedTokens += cached;
|
|
metrics.byStrategy[strategy].cacheCreationTokens += creation;
|
|
}
|
|
}
|
|
|
|
metrics.lastUpdated = now;
|
|
return metrics;
|
|
}
|
|
|
|
/**
|
|
* Record cache token usage and update metrics
|
|
*/
|
|
export function updateCacheTokenMetrics({
|
|
metrics,
|
|
provider,
|
|
strategy,
|
|
inputTokens,
|
|
cachedTokens,
|
|
cacheCreationTokens,
|
|
costSaved,
|
|
}: {
|
|
metrics: CacheControlMetrics;
|
|
provider: string;
|
|
strategy: string | null | undefined;
|
|
inputTokens: number;
|
|
cachedTokens: number;
|
|
cacheCreationTokens: number;
|
|
costSaved?: number;
|
|
}): CacheControlMetrics {
|
|
metrics.totalCachedTokens += cachedTokens;
|
|
metrics.totalCacheCreationTokens += cacheCreationTokens;
|
|
metrics.totalInputTokens += inputTokens;
|
|
|
|
// Cached tokens are reused (saved), creation tokens are new cache writes
|
|
metrics.tokensSaved += cachedTokens;
|
|
if (costSaved !== undefined) {
|
|
metrics.estimatedCostSaved += costSaved;
|
|
}
|
|
|
|
// Update provider tracking
|
|
if (metrics.byProvider[provider]) {
|
|
metrics.byProvider[provider].cachedTokens += cachedTokens;
|
|
metrics.byProvider[provider].cacheCreationTokens += cacheCreationTokens;
|
|
metrics.byProvider[provider].inputTokens += inputTokens;
|
|
}
|
|
|
|
// Update strategy tracking
|
|
if (strategy && metrics.byStrategy[strategy]) {
|
|
metrics.byStrategy[strategy].cachedTokens += cachedTokens;
|
|
metrics.byStrategy[strategy].cacheCreationTokens += cacheCreationTokens;
|
|
metrics.byStrategy[strategy].inputTokens += inputTokens;
|
|
}
|
|
|
|
metrics.lastUpdated = new Date().toISOString();
|
|
return metrics;
|
|
}
|