From 7a7f3be0d262bc25a41e179ebd62b4970ff521de Mon Sep 17 00:00:00 2001 From: diegosouzapw Date: Sun, 22 Mar 2026 20:55:35 -0300 Subject: [PATCH] feat(sub2api): implement T01-T15 gap analysis tasks (3.0.0-rc.6) T01 (P1): requested_model column in call_logs - Migration 009_requested_model.sql: ALTER TABLE call_logs ADD COLUMN requested_model - callLogs.ts: INSERT + SELECT updated to include requestedModel field T02 (P1): Strip empty text blocks from nested tool_result.content - New stripEmptyTextBlocks() recursive helper in openai-to-claude.ts - Applied on tool_result content before forwarding to Anthropic - Prevents 400 'text content blocks must be non-empty' errors T03 (P1): Parse x-codex-5h-*/x-codex-7d-* headers for precise quota reset - parseCodexQuotaHeaders() in codex.ts extracts usage/limit/resetAt - getCodexResetTime() returns furthest-out reset timestamp for safe unblocking T04 (P1): X-Session-Id header for external sticky routing - extractExternalSessionId() in sessionManager.ts reads x-session-id, x-omniroute-session, session-id headers with 'ext:' prefix to avoid collisions T06 (P2): account_deactivated permanent expired status on 401 - ACCOUNT_DEACTIVATED_SIGNALS constant + isAccountDeactivated() in accountFallback.ts - Returns 1-year cooldown (effectively permanent) to prevent retrying dead accounts T07 (P2): X-Forwarded-For IP validation - New src/lib/ipUtils.ts with extractClientIp() and getClientIpFromRequest() - Skips 'unknown'/non-IP entries in X-Forwarded-For chain T10 (P2): credits_exhausted distinct account status - CREDITS_EXHAUSTED_SIGNALS + isCreditsExhausted() in accountFallback.ts - Returns 1h cooldown with creditsExhausted flag, distinct from rate_limit 429 T11 (P1): max reasoning_effort -> budget_tokens: 131072 - EFFORT_BUDGETS and THINKING_LEVEL_MAP updated with max: 131072, xhigh: 131072 - Reverse mapping now returns 'max' for full-budget responses - Unit test updated to expect 'max' (was 'high') T12 (P3): Model pricing updates - MiniMax M2.7 / MiniMax-M2.7 / minimax-m2.7-highspeed pricing added T15 (P1): Array content normalization for system/tool messages - normalizeContentToString() helper exported from openai-to-claude.ts - System messages with array content now correctly collapsed to string --- open-sse/executors/codex.ts | 66 +++++++++++++++++ open-sse/services/accountFallback.ts | 68 ++++++++++++++++++ open-sse/services/sessionManager.ts | 26 +++++++ open-sse/services/thinkingBudget.ts | 10 ++- .../translator/request/openai-to-claude.ts | 70 +++++++++++++++++-- src/lib/db/migrations/009_requested_model.sql | 9 +++ src/lib/ipUtils.ts | 56 +++++++++++++++ src/lib/usage/callLogs.ts | 6 +- src/shared/constants/pricing.ts | 24 +++++++ tests/unit/thinking-budget.test.mjs | 3 +- 10 files changed, 329 insertions(+), 9 deletions(-) create mode 100644 src/lib/db/migrations/009_requested_model.sql create mode 100644 src/lib/ipUtils.ts diff --git a/open-sse/executors/codex.ts b/open-sse/executors/codex.ts index eeae18bd39..4dbaf170bf 100644 --- a/open-sse/executors/codex.ts +++ b/open-sse/executors/codex.ts @@ -3,6 +3,72 @@ import { CODEX_DEFAULT_INSTRUCTIONS } from "../config/codexInstructions.ts"; import { PROVIDERS } from "../config/constants.ts"; import { refreshCodexToken } from "../services/tokenRefresh.ts"; +/** + * T03: Parsed quota snapshot from Codex response headers. + * Codex includes per-account usage windows that allow precise reset scheduling. + * Ref: sub2api PR #357 (feat(oauth): persist usage snapshots and window cooldown) + */ +export interface CodexQuotaSnapshot { + usage5h: number; // tokens used in 5h window + limit5h: number; // token limit for 5h window + resetAt5h: string | null; // ISO timestamp when 5h window resets + usage7d: number; // tokens used in 7d window + limit7d: number; // token limit for 7d window + resetAt7d: string | null; // ISO timestamp when 7d window resets +} + +/** + * T03: Parse Codex-specific quota headers from a provider response. + * Returns null if none of the relevant headers are present. + * + * Extracts: + * x-codex-5h-usage / x-codex-5h-limit / x-codex-5h-reset-at + * x-codex-7d-usage / x-codex-7d-limit / x-codex-7d-reset-at + */ +export function parseCodexQuotaHeaders(headers: Headers): CodexQuotaSnapshot | null { + const usage5h = headers.get("x-codex-5h-usage"); + const limit5h = headers.get("x-codex-5h-limit"); + const resetAt5h = headers.get("x-codex-5h-reset-at"); + const usage7d = headers.get("x-codex-7d-usage"); + const limit7d = headers.get("x-codex-7d-limit"); + const resetAt7d = headers.get("x-codex-7d-reset-at"); + + // Return null if none of the quota headers are present (not a quota-aware response) + if (!usage5h && !limit5h && !resetAt5h && !usage7d && !limit7d && !resetAt7d) { + return null; + } + + return { + usage5h: usage5h ? parseFloat(usage5h) : 0, + limit5h: limit5h ? parseFloat(limit5h) : Infinity, + resetAt5h: resetAt5h ?? null, + usage7d: usage7d ? parseFloat(usage7d) : 0, + limit7d: limit7d ? parseFloat(limit7d) : Infinity, + resetAt7d: resetAt7d ?? null, + }; +} + +/** + * T03: Get the soonest quota reset time from a CodexQuotaSnapshot. + * 7d window takes priority (wider window, harder limit) but we use whichever + * is further in the future to avoid releasing the block too early. + * + * @returns Unix timestamp (ms) of the soonest effective reset, or null + */ +export function getCodexResetTime(quota: CodexQuotaSnapshot): number | null { + const times: number[] = []; + if (quota.resetAt7d) { + const t = new Date(quota.resetAt7d).getTime(); + if (!isNaN(t) && t > Date.now()) times.push(t); + } + if (quota.resetAt5h) { + const t = new Date(quota.resetAt5h).getTime(); + if (!isNaN(t) && t > Date.now()) times.push(t); + } + if (times.length === 0) return null; + return Math.max(...times); // Use furthest-out reset to avoid premature unblock +} + // Ordered list of effort levels from lowest to highest const EFFORT_ORDER = ["none", "low", "medium", "high", "xhigh"] as const; type EffortLevel = (typeof EFFORT_ORDER)[number]; diff --git a/open-sse/services/accountFallback.ts b/open-sse/services/accountFallback.ts index 77839ffca4..415f3bbfa1 100644 --- a/open-sse/services/accountFallback.ts +++ b/open-sse/services/accountFallback.ts @@ -8,6 +8,46 @@ import { } from "../config/constants.ts"; import { getProviderCategory } from "../config/providerRegistry.ts"; +// T06 (sub2api PR #1037): Signals that indicate permanent account deactivation. +// When a 401 body contains these strings, the account is permanently dead +// and should NOT be retried after token refresh. +export const ACCOUNT_DEACTIVATED_SIGNALS = [ + "account_deactivated", + "account has been deactivated", + "account has been disabled", + "your account has been suspended", + "this account is deactivated", +]; + +// T10 (sub2api PR #1169): Signals that indicate billing credits are exhausted. +// Distinct from rate-limit 429 — the account won't recover until credits are added. +export const CREDITS_EXHAUSTED_SIGNALS = [ + "insufficient_quota", + "billing_hard_limit_reached", + "exceeded your current quota", + "credit_balance_too_low", + "your credit balance is too low", + "credits exhausted", + "out of credits", + "payment required", +]; + +/** + * T06: Returns true if response body indicates the account is permanently deactivated. + */ +export function isAccountDeactivated(errorText: string): boolean { + const lower = String(errorText || "").toLowerCase(); + return ACCOUNT_DEACTIVATED_SIGNALS.some((sig) => lower.includes(sig)); +} + +/** + * T10: Returns true if response body indicates credits/quota are permanently exhausted. + */ +export function isCreditsExhausted(errorText: string): boolean { + const lower = String(errorText || "").toLowerCase(); + return CREDITS_EXHAUSTED_SIGNALS.some((sig) => lower.includes(sig)); +} + // ─── Provider Profile Helper ──────────────────────────────────────────────── /** @@ -201,6 +241,14 @@ export function classifyErrorText(errorText) { ) { return RateLimitReason.QUOTA_EXHAUSTED; } + // T10: credits_exhausted signals + if (isCreditsExhausted(errorText)) { + return RateLimitReason.QUOTA_EXHAUSTED; + } + // T06: account_deactivated signals + if (isAccountDeactivated(errorText)) { + return RateLimitReason.AUTH_ERROR; + } if ( lower.includes("rate limit") || lower.includes("too many requests") || @@ -301,6 +349,26 @@ export function checkFallbackError( const errorStr = typeof errorText === "string" ? errorText : JSON.stringify(errorText); const lowerError = errorStr.toLowerCase(); + // T06 (sub2api #1037): Permanent account deactivation — do NOT retry, mark as permanent failure + if (isAccountDeactivated(errorStr)) { + return { + shouldFallback: true, + cooldownMs: 365 * 24 * 60 * 60 * 1000, // 1 year = effectively permanent + reason: RateLimitReason.AUTH_ERROR, + permanent: true, + }; + } + + // T10 (sub2api #1169): Credits/quota exhausted — long cooldown, distinct from rate limit + if (isCreditsExhausted(errorStr)) { + return { + shouldFallback: true, + cooldownMs: COOLDOWN_MS.paymentRequired ?? 3600 * 1000, // 1h cooldown + reason: RateLimitReason.QUOTA_EXHAUSTED, + creditsExhausted: true, + }; + } + if (lowerError.includes("no credentials")) { return { shouldFallback: true, diff --git a/open-sse/services/sessionManager.ts b/open-sse/services/sessionManager.ts index b571084ebd..a196265ba3 100644 --- a/open-sse/services/sessionManager.ts +++ b/open-sse/services/sessionManager.ts @@ -175,6 +175,32 @@ export function clearSessions(): void { sessions.clear(); } +/** + * T04: Extract an external session ID from request headers. + * Accepts both hyphenated and underscore forms for Nginx compatibility. + * Nginx drops headers with underscores by default — use `underscores_in_headers on` + * in nginx.conf, or use X-Session-Id (hyphenated) which passes cleanly. + * + * Ref: sub2api README + PR #634 + * + * @param headers - Request headers (Headers object or plain object with .get()) + * @returns External session ID with "ext:" prefix, or null + */ +export function extractExternalSessionId( + headers: Headers | { get?: (n: string) => string | null } | null | undefined +): string | null { + if (!headers || typeof (headers as Headers).get !== "function") return null; + const h = headers as Headers; + const raw = + h.get("x-session-id") ?? // Preferred: hyphenated (passes through Nginx) + h.get("x-omniroute-session") ?? // OmniRoute-specific form + h.get("session-id") ?? // Bare session-id + null; + if (!raw || !raw.trim()) return null; + // Prefix "ext:" to ensure no collision with internal SHA-256 hash IDs + return `ext:${raw.trim().slice(0, 64)}`; // max 64 chars to avoid abuse +} + // ─── Internal Helpers ─────────────────────────────────────────────────────── function hashShort(text: string): string { diff --git a/open-sse/services/thinkingBudget.ts b/open-sse/services/thinkingBudget.ts index b5f0994b04..fc906ba38e 100644 --- a/open-sse/services/thinkingBudget.ts +++ b/open-sse/services/thinkingBudget.ts @@ -19,6 +19,8 @@ export const EFFORT_BUDGETS = { low: 1024, medium: 10240, high: 131072, + max: 131072, // T11: Claude "max" / "xhigh" — full budget + xhigh: 131072, // T11: explicit alias used internally }; // thinkingLevel string → budget token mapping @@ -28,6 +30,8 @@ export const THINKING_LEVEL_MAP = { low: 1024, medium: 10240, high: 131072, + max: 131072, // T11: max = full Claude budget (sub2api: xhigh) + xhigh: 131072, // T11: explicit xhigh alias }; // Default config (passthrough = backward compatible) @@ -198,7 +202,7 @@ function setCustomBudget(body, budget) { }; } - // OpenAI reasoning_effort mapping + // OpenAI reasoning_effort mapping (T11: add 'max' tier for full budget) if (result.reasoning_effort !== undefined || result.reasoning !== undefined) { if (budget <= 0) { delete result.reasoning_effort; @@ -207,8 +211,10 @@ function setCustomBudget(body, budget) { result.reasoning_effort = "low"; } else if (budget <= 10240) { result.reasoning_effort = "medium"; - } else { + } else if (budget < 131072) { result.reasoning_effort = "high"; + } else { + result.reasoning_effort = "max"; // T11: full budget → "max" } } diff --git a/open-sse/translator/request/openai-to-claude.ts b/open-sse/translator/request/openai-to-claude.ts index 0a314a8ac0..4b002e36da 100644 --- a/open-sse/translator/request/openai-to-claude.ts +++ b/open-sse/translator/request/openai-to-claude.ts @@ -27,6 +27,60 @@ type ClaudeTool = { defer_loading?: boolean; }; +/** + * T02: Recursively strips empty text blocks from content arrays. + * Anthropic returns 400 "text content blocks must be non-empty" if any + * text block has text: "". Must also recurse into nested tool_result.content. + * Ref: sub2api PR #1212 + */ +export function stripEmptyTextBlocks(content: unknown[] | undefined): unknown[] { + if (!Array.isArray(content)) return content ?? []; + return content + .filter((block: unknown) => { + if ( + block && + typeof block === "object" && + (block as Record).type === "text" + ) { + const text = (block as Record).text; + if (text === "" || text == null) return false; + } + return true; + }) + .map((block: unknown) => { + if ( + block && + typeof block === "object" && + (block as Record).type === "tool_result" && + Array.isArray((block as Record).content) + ) { + // Recurse into nested tool_result.content + return { + ...(block as Record), + content: stripEmptyTextBlocks((block as Record).content as unknown[]), + }; + } + return block; + }); +} + +/** + * T15: Normalize content to string form. + * Handles both string and array-of-blocks forms (Cursor, Codex 2.x, etc.). + * Ref: sub2api PR #1197 + */ +export function normalizeContentToString(content: string | unknown[] | null | undefined): string { + if (!content) return ""; + if (typeof content === "string") return content; + if (Array.isArray(content)) { + return (content as Array>) + .filter((b) => b.type === "text") + .map((b) => String(b.text ?? "")) + .join("\n"); + } + return ""; +} + // Convert OpenAI request to Claude format export function openaiToClaudeRequest(model, body, stream) { // Check if tool prefix should be disabled (configured per-provider or global) @@ -61,11 +115,11 @@ export function openaiToClaudeRequest(model, body, stream) { const systemParts = []; if (body.messages && Array.isArray(body.messages)) { - // Extract system messages + // Extract system messages (T15: handle both string and array content) for (const msg of body.messages) { if (msg.role === "system") { systemParts.push( - typeof msg.content === "string" ? msg.content : extractTextContent(msg.content) + typeof msg.content === "string" ? msg.content : normalizeContentToString(msg.content) ); } } @@ -270,10 +324,14 @@ function getContentBlocksFromMessage(msg, toolNameMap = new Map(), disableToolPr const blocks = []; if (msg.role === "tool") { + // T02: Strip empty text blocks from nested tool_result content to avoid Anthropic 400 + const toolContent = Array.isArray(msg.content) + ? stripEmptyTextBlocks(msg.content) + : msg.content; blocks.push({ type: "tool_result", tool_use_id: msg.tool_call_id, - content: msg.content, + content: toolContent, }); } else if (msg.role === "user") { if (typeof msg.content === "string") { @@ -287,10 +345,14 @@ function getContentBlocksFromMessage(msg, toolNameMap = new Map(), disableToolPr } else if (part.type === "tool_result") { // Skip tool_result with no tool_use_id (would be useless and may cause errors) if (!part.tool_use_id) continue; + // T02: strip empty text blocks from nested content before passing to Anthropic + const resultContent = Array.isArray(part.content) + ? stripEmptyTextBlocks(part.content) + : part.content; blocks.push({ type: "tool_result", tool_use_id: part.tool_use_id, - content: part.content, + content: resultContent, ...(part.is_error && { is_error: part.is_error }), }); } else if (part.type === "image_url") { diff --git a/src/lib/db/migrations/009_requested_model.sql b/src/lib/db/migrations/009_requested_model.sql new file mode 100644 index 0000000000..95b82ff499 --- /dev/null +++ b/src/lib/db/migrations/009_requested_model.sql @@ -0,0 +1,9 @@ +-- Migration 009: Add requested_model to call_logs for billing transparency +-- Tracks the model the client *asked* for vs the model that was *actually routed*. +-- Needed when a combo falls back: requested_model ≠ model in call_logs. +-- Ref: sub2api commits 0b845c25 + 4edcfe1f (T01 sub2api gap analysis) +ALTER TABLE call_logs ADD COLUMN requested_model TEXT DEFAULT NULL; + +-- Index for filtering/aggregating by requested_model in Analytics +CREATE INDEX IF NOT EXISTS idx_call_logs_requested_model + ON call_logs(requested_model); diff --git a/src/lib/ipUtils.ts b/src/lib/ipUtils.ts new file mode 100644 index 0000000000..b1881b28d2 --- /dev/null +++ b/src/lib/ipUtils.ts @@ -0,0 +1,56 @@ +import { isIP } from "node:net"; + +/** + * T07: Extract the real client IP from X-Forwarded-For header. + * Skips invalid entries like "unknown" or empty strings. + * Falls back to remoteAddress if no valid IP found. + * Ref: sub2api PR #1135 + * + * @param xForwardedFor - Value of the X-Forwarded-For header (may be CSV) + * @param remoteAddress - Fallback from the raw socket (req.socket.remoteAddress) + * @returns The first valid IP address found, or "unknown" + */ +export function extractClientIp( + xForwardedFor: string | null | undefined, + remoteAddress: string | undefined +): string { + if (xForwardedFor) { + const entries = xForwardedFor.split(","); + for (const entry of entries) { + const trimmed = entry.trim(); + if (trimmed && isIP(trimmed) !== 0) { + return trimmed; // First valid IP wins + } + } + } + return remoteAddress?.trim() ?? "unknown"; +} + +/** + * Extract client IP from a Request or NextRequest object. + * Checks X-Forwarded-For, X-Real-IP, CF-Connecting-IP, then socket. + */ +export function getClientIpFromRequest(req: { + headers?: Headers | { get?: (n: string) => string | null }; + socket?: { remoteAddress?: string }; + ip?: string; +}): string { + // Helper to get header value from either Headers object or plain object + const getHeader = (name: string): string | null => { + if (!req.headers) return null; + if (typeof (req.headers as Headers).get === "function") { + return (req.headers as Headers).get(name); + } + return null; + }; + + // Priority: CF-Connecting-IP (Cloudflare) > X-Forwarded-For > X-Real-IP > socket + const cfIp = getHeader("cf-connecting-ip"); + if (cfIp && isIP(cfIp.trim()) !== 0) return cfIp.trim(); + + const xff = getHeader("x-forwarded-for"); + const realIp = getHeader("x-real-ip"); + const remoteAddress = req.ip ?? req.socket?.remoteAddress; + + return extractClientIp(xff ?? realIp, remoteAddress); +} diff --git a/src/lib/usage/callLogs.ts b/src/lib/usage/callLogs.ts index 9a787aa3a3..341dc7532d 100644 --- a/src/lib/usage/callLogs.ts +++ b/src/lib/usage/callLogs.ts @@ -180,6 +180,7 @@ export async function saveCallLog(entry: any) { path: entry.path || "/v1/chat/completions", status: entry.status || 0, model: entry.model || "-", + requestedModel: entry.requestedModel || null, // T01: model the client asked for provider: entry.provider || "-", account, connectionId: entry.connectionId || null, @@ -205,10 +206,10 @@ export async function saveCallLog(entry: any) { const db = getDbInstance(); db.prepare( ` - INSERT INTO call_logs (id, timestamp, method, path, status, model, provider, + INSERT INTO call_logs (id, timestamp, method, path, status, model, requested_model, provider, account, connection_id, duration, tokens_in, tokens_out, request_type, source_format, target_format, api_key_id, api_key_name, combo_name, request_body, response_body, error) - VALUES (@id, @timestamp, @method, @path, @status, @model, @provider, + VALUES (@id, @timestamp, @method, @path, @status, @model, @requestedModel, @provider, @account, @connectionId, @duration, @tokensIn, @tokensOut, @requestType, @sourceFormat, @targetFormat, @apiKeyId, @apiKeyName, @comboName, @requestBody, @responseBody, @error) ` @@ -374,6 +375,7 @@ export async function getCallLogs(filter: any = {}) { path: toStringOrNull(l.path), status: toNumber(l.status), model: toStringOrNull(l.model), + requestedModel: toStringOrNull(l.requested_model), // T01: original model from client provider: toStringOrNull(l.provider), account: toStringOrNull(l.account), duration: toNumber(l.duration), diff --git a/src/shared/constants/pricing.ts b/src/shared/constants/pricing.ts index bf0ffa46e3..399de81dce 100644 --- a/src/shared/constants/pricing.ts +++ b/src/shared/constants/pricing.ts @@ -802,6 +802,30 @@ export const DEFAULT_PRICING = { reasoning: 1.8, cache_creation: 0.3, }, + // T12: MiniMax M2.7 — new default model (sub2api PR #1120) + // Upgraded from M2.5, same API endpoint api.minimax.io + // Pricing estimated, check https://platform.minimaxi.com/document/Price + "minimax-m2.7": { + input: 0.4, + output: 1.6, + cached: 0.2, + reasoning: 2.4, + cache_creation: 0.4, + }, + "MiniMax-M2.7": { + input: 0.4, + output: 1.6, + cached: 0.2, + reasoning: 2.4, + cache_creation: 0.4, + }, + "minimax-m2.7-highspeed": { + input: 0.4, + output: 1.6, + cached: 0.2, + reasoning: 2.4, + cache_creation: 0.4, + }, }, // ─── Free-tier API Key Providers (nominal $0 pricing) ─── diff --git a/tests/unit/thinking-budget.test.mjs b/tests/unit/thinking-budget.test.mjs index 79b2ed2980..c18074b5ee 100644 --- a/tests/unit/thinking-budget.test.mjs +++ b/tests/unit/thinking-budget.test.mjs @@ -101,7 +101,8 @@ test("CUSTOM: sets OpenAI reasoning_effort from budget", () => { reasoning_effort: "low", }; const result = applyThinkingBudget(body); - assert.equal(result.reasoning_effort, "high"); + // T11 (sub2api gap): full budget (131072) now maps to "max" instead of "high" + assert.equal(result.reasoning_effort, "max"); setThinkingBudgetConfig(DEFAULT_THINKING_CONFIG); });