Compare commits

..

1 Commits

Author SHA1 Message Date
dependabot[bot]
0908ea5473 chore(deps): bump github/codeql-action from 4.37.4 to 4.37.6
Bumps [github/codeql-action](https://github.com/github/codeql-action) from 4.37.4 to 4.37.6.
- [Release notes](https://github.com/github/codeql-action/releases)
- [Changelog](https://github.com/github/codeql-action/blob/main/CHANGELOG.md)
- [Commits](https://github.com/github/codeql-action/compare/v4.37.4...v4.37.6)

---
updated-dependencies:
- dependency-name: github/codeql-action
  dependency-version: 4.37.6
  dependency-type: direct:production
  update-type: version-update:semver-patch
...

Signed-off-by: dependabot[bot] <support@github.com>
2026-08-14 18:24:34 +00:00
25 changed files with 53 additions and 1467 deletions

View File

@@ -350,11 +350,6 @@ ALLOW_API_KEY_REVEAL=false
# OMNIROUTE_CHAT_HARD_MAX_BODY_BYTES=52428800
# Maximum heavyweight requests simultaneously admitted in one process. Default 1.
# OMNIROUTE_CHAT_MAX_HEAVY_IN_FLIGHT=1
# Heap-pressure shed ratio (heapUsed/heap_size_limit) for the structural admission gate
# (#10183, #10268): a second concurrent heavyweight request past OMNIROUTE_CHAT_MAX_HEAVY_IN_FLIGHT
# is only shed with a retryable 503 when the heap is ALSO under this much pressure — on a
# healthy heap it is admitted instead. Range (0, 1]. Default 0.75.
# OMNIROUTE_CHAT_ADMISSION_HEAP_SHED_RATIO=0.75
# Message count that classifies an otherwise small body as heavyweight. Default 200.
# OMNIROUTE_CHAT_HEAVY_MESSAGE_COUNT=200
# Tool count that classifies an otherwise small body as heavyweight. Default 64.
@@ -2045,22 +2040,6 @@ APP_LOG_TO_FILE=true
# ALIBABA_CODING_PLAN_HOST=
# ALIBABA_CODING_PLAN_QUOTA_URL=
# ── Qwen Cloud / Model Studio personal Token Plan quota ──
# Cookie-authenticated console-gateway fetcher (issue #9603). Used by:
# open-sse/services/qwenTokenPlanQuotaFetcher.ts. Prefer the per-connection
# Dashboard fields (qwenCloudCookie / qwenCloudSecToken) — these env vars are
# global fallbacks. Cookie/sec_token are SENSITIVE session credentials.
# Getting the cookie: log in to home.qwencloud.com > Billing > Subscription,
# press F12 > Network, reload, filter by api.json, click any request to
# cs-data.qwencloud.com and copy the WHOLE Cookie value from Request Headers
# (it contains login_qwencloud_ticket). Paste it on ONE line — the value may
# contain '=' and ';'. It expires with the browser session; re-paste it when
# the dashboard reports an expired session.
# QWEN_CLOUD_COOKIE=
# QWEN_CLOUD_SEC_TOKEN=
# QWEN_TOKEN_PLAN_HOST=
# QWEN_TOKEN_PLAN_DASHBOARD_URL=
# ── Alibaba Model Studio free-tier quota sync ──
# Console front-end path overrides for the free-tier quota fetcher. Used by:
# open-sse/services/alibabaFreeTierQuotaFetcher.ts. When unset, the fetcher

View File

@@ -372,7 +372,7 @@ jobs:
- name: Upload Trivy SARIF to Security tab
if: needs.prepare.outputs.version != 'main'
continue-on-error: true
uses: github/codeql-action/upload-sarif@v4.37.4
uses: github/codeql-action/upload-sarif@v4.37.6
with:
sarif_file: trivy-results.sarif
category: trivy-image

View File

@@ -1 +0,0 @@
- fix(sse): gate structural chat admission shedding on real heap pressure instead of unconditional capacity (#10183, #10268)

View File

@@ -192,7 +192,6 @@ OmniRoute uses **SQLite** (via `better-sqlite3`) for all persistence. These vari
| `OMNIROUTE_CHAT_LARGE_BODY_BYTES` | `262144` (256 KB) | `src/shared/middleware/chatBodyAdmission.ts` | Actual request bodies at or above this threshold require an atomic process-local heavyweight admission lease before JSON parsing. |
| `OMNIROUTE_CHAT_HARD_MAX_BODY_BYTES` | `52428800` (50 MB) | `src/shared/middleware/chatBodyAdmission.ts` | Chat-route hard cap enforced against bytes read during bounded ingestion, including requests with missing, invalid, or dishonest `Content-Length`; excess receives `413`. |
| `OMNIROUTE_CHAT_MAX_HEAVY_IN_FLIGHT` | `1` | `src/shared/middleware/chatBodyAdmission.ts` | Maximum heavyweight chat requests admitted concurrently in one process. When capacity is unavailable, OmniRoute returns retryable `503` with `Retry-After`. |
| `OMNIROUTE_CHAT_ADMISSION_HEAP_SHED_RATIO` | `0.75` | `src/shared/middleware/chatBodyAdmission.ts` | Heap-pressure shed ratio (`heapUsed / heap_size_limit`) for the structural admission gate (#10183, #10268). A second concurrent heavyweight request past `OMNIROUTE_CHAT_MAX_HEAVY_IN_FLIGHT` is only shed with the retryable `503` when the heap is ALSO at or above this ratio; on a healthy heap it is admitted instead. |
| `OMNIROUTE_CHAT_HEAVY_MESSAGE_COUNT` | `200` | `src/shared/middleware/chatBodyAdmission.ts` | Message count that classifies a chat request as heavyweight even when its body is below the byte threshold. |
| `OMNIROUTE_CHAT_HEAVY_TOOL_COUNT` | `64` | `src/shared/middleware/chatBodyAdmission.ts` | Tool count that classifies a chat request as heavyweight even when its body is below the byte threshold. |
| `OMNIROUTE_CHAT_HEAVY_ESTIMATED_TOKENS` | `32000` | `src/shared/middleware/chatBodyAdmission.ts` | Conservative string-size token estimate that classifies a request as heavyweight; this is an admission-cost proxy, not provider billing tokenization. |
@@ -1143,10 +1142,6 @@ Provider quota endpoints, network tunnels (Tailscale, Ngrok, MITM debug proxy),
| `REDIS_URL` | `redis://localhost:6379` | `src/shared/utils/rateLimiter.ts` | Redis connection string for the rate limiter backend. |
| `ALIBABA_CODING_PLAN_HOST` | _(production host)_ | `open-sse/services/bailianQuotaFetcher.ts` | Override the host used to fetch Alibaba Bailian coding-plan quotas. |
| `ALIBABA_CODING_PLAN_QUOTA_URL` | derived from host | `open-sse/services/bailianQuotaFetcher.ts` | Full quota URL override for Alibaba Bailian. |
| `QWEN_CLOUD_COOKIE` | _(unset)_ | `open-sse/services/qwenTokenPlanQuotaFetcher.ts` | Console session cookie for the Qwen Cloud / Model Studio personal Token Plan quota gateway (the inference API key cannot read it). Copy the whole `Cookie` request header — it contains `login_qwencloud_ticket` — from any `api.json` call to `cs-data.qwencloud.com` on home.qwencloud.com Billing Subscription (F12 Network). Sensitive and session-scoped; prefer the per-connection `qwenCloudCookie` Dashboard field. |
| `QWEN_CLOUD_SEC_TOKEN` | _(unset)_ | `open-sse/services/qwenTokenPlanQuotaFetcher.ts` | Manual `sec_token` override for the Token Plan console gateway. Sensitive; when unset the fetcher resolves it from the dashboard HTML using the cookie. |
| `QWEN_TOKEN_PLAN_HOST` | `https://cs-data.qwencloud.com` | `open-sse/services/qwenTokenPlanQuotaFetcher.ts` | Gateway host override for the personal Token Plan quota fetcher (e.g. `bailian-singapore-cs.alibabacloud.com` for the Model Studio console). |
| `QWEN_TOKEN_PLAN_DASHBOARD_URL` | `https://home.qwencloud.com/` | `open-sse/services/qwenTokenPlanQuotaFetcher.ts` | Dashboard URL used to resolve `sec_token` from the logged-in HTML. |
| `ALIBABA_FREE_TIER_VISION_FE_PATH` | `/costing-balance/free-quota-image-video` | `open-sse/services/alibabaFreeTierQuotaFetcher.ts` | Console front-end path override for fetching Alibaba Model Studio free-tier vision/media quota. |
| `ALIBABA_FREE_TIER_MULTIMODAL_FE_PATH` | `/costing-balance/free-quota-multimodal` | `open-sse/services/alibabaFreeTierQuotaFetcher.ts` | Console front-end path override for fetching Alibaba Model Studio free-tier multimodal quota. |
| `ALIBABA_FREE_TIER_AUDIO_FE_PATH` | `/costing-balance/free-quota-audio` | `open-sse/services/alibabaFreeTierQuotaFetcher.ts` | Console front-end path override for fetching Alibaba Model Studio free-tier audio quota. |

View File

@@ -60,12 +60,7 @@ export const bailian_coding_planProvider: RegistryEntry = {
alias: "bcp",
format: "claude",
executor: "default",
// Token Plan endpoint (the catalog entry is "Alibaba Token Plan"). The former
// coding-intl.dashscope.aliyuncs.com host only accepts Coding Plan keys and rejects
// Token Plan keys with 401 invalid_api_key. Verified live 2026-08-14: this host
// returns 200 for every model below with the same key.
// Docs: https://www.alibabacloud.com/help/en/model-studio/more-tools
baseUrl: "https://token-plan.ap-southeast-1.maas.aliyuncs.com/apps/anthropic/v1",
baseUrl: "https://coding-intl.dashscope.aliyuncs.com/apps/anthropic/v1",
chatPath: "/messages",
authType: "apikey",
authHeader: "x-api-key",

View File

@@ -1,437 +0,0 @@
/**
* qwenTokenPlanQuotaFetcher.ts — Qwen Cloud / Alibaba Model Studio PERSONAL Token Plan
* quota fetcher (issue #9603, "quota is missing").
*
* The personal Token Plan (5-hour / 7-day sliding windows) has NO official OpenAPI —
* the console gateway is the only quota surface, and the inference API key does NOT
* authenticate it. Both portals read the same backend:
* - home.qwencloud.com portal → https://cs-data.qwencloud.com (default)
* - Model Studio console (intl) → https://bailian-singapore-cs.alibabacloud.com
*
* Transport (captured live 2026-08-13 from a logged-in session):
* POST {host}/data/api.json?product=sfm_bailian&action=IntlBroadScopeAspnGateway
* &api=zeldaHttp.apikeyMgr.%2Ftokenplan%2Fpersonal%2Fapi%2Fv2%2F<endpoint>
* form body: product, action, sec_token, region, params =
* {"Api":"zeldaHttp.apikeyMgr./tokenplan/personal/api/v2/<endpoint>","V":"1.0",
* "Data":{"commodityCode":"sfm_tokenplansolo_public_intl","cornerstoneParam":{...}}}
* Auth: browser session Cookie (providerSpecificData or QWEN_CLOUD_COOKIE env).
* sec_token: best-effort — resolved from the dashboard HTML (`SEC_TOKEN: "…"`) when
* not provided; some accounts reject requests without it
* (BailianGateway.Workspace.NotAuthorised).
*
* Windows: usage returns per<Window>Percentage (fraction used, 0..1) +
* per<Window>ResetTime (epoch ms). Fields are OMITTED while a window is
* "Temporarily Removed" (observed for 5-hour), so every window is optional.
*
* Cache: usage 60s per connection; subscription/quota-config (slow-moving tier data)
* 1h per connection. Registration: registerQwenTokenPlanQuotaFetcher() at startup.
*/
import { registerQuotaFetcher, registerQuotaWindows, type QuotaInfo } from "./quotaPreflight.ts";
import { registerMonitorFetcher } from "./quotaMonitor.ts";
import { throttleQuotaFetch } from "./quotaFetchThrottle.ts";
const DEFAULT_GATEWAY_HOST = "https://cs-data.qwencloud.com";
const DEFAULT_DASHBOARD_URL = "https://home.qwencloud.com/";
/**
* The same personal Token Plan is sold through two consoles that share one backend.
* The gateway validates the browser session against the console identity sent in the
* request, so an Alibaba cookie paired with the QwenCloud identity is rejected with
* `BailianGateway.Login.NotLogined` (verified live 2026-08-14).
*/
export interface TokenPlanConsoleSite {
consoleSite: "QWENCLOUD" | "ALIYUN";
domain: string;
gatewayHost: string;
dashboardUrl: string;
origin: string;
}
const CONSOLE_SITES: Record<"qwencloud" | "aliyun", TokenPlanConsoleSite> = {
qwencloud: {
consoleSite: "QWENCLOUD",
domain: "home.qwencloud.com",
gatewayHost: DEFAULT_GATEWAY_HOST,
dashboardUrl: DEFAULT_DASHBOARD_URL,
origin: "https://home.qwencloud.com",
},
aliyun: {
consoleSite: "ALIYUN",
domain: "modelstudio.console.alibabacloud.com",
gatewayHost: "https://bailian-singapore-cs.alibabacloud.com",
dashboardUrl: "https://modelstudio.console.alibabacloud.com/",
origin: "https://modelstudio.console.alibabacloud.com",
},
};
/** Providers served by the Alibaba (Model Studio) console rather than QwenCloud. */
const ALIYUN_CONSOLE_PROVIDERS = new Set(["bailian-coding-plan", "alibaba", "alibaba-cn"]);
/**
* Pick the console identity for a cookie: the login ticket names its console
* (`login_aliyunid_ticket` vs `login_qwencloud_ticket`). Unmarked cookies fall back to
* the provider, then to QwenCloud.
*/
export function resolveConsoleSite(
cookie: string,
provider: string | undefined
): TokenPlanConsoleSite {
if (/login_aliyunid_ticket=/.test(cookie)) return CONSOLE_SITES.aliyun;
if (/login_qwencloud_ticket=/.test(cookie)) return CONSOLE_SITES.qwencloud;
if (provider && ALIYUN_CONSOLE_PROVIDERS.has(provider)) return CONSOLE_SITES.aliyun;
return CONSOLE_SITES.qwencloud;
}
const GATEWAY_REGION = "ap-southeast-1";
const GATEWAY_PRODUCT = "sfm_bailian";
const GATEWAY_ACTION = "IntlBroadScopeAspnGateway";
const COMMODITY_CODE = "sfm_tokenplansolo_public_intl";
const TOKEN_PLAN_API_PREFIX = "zeldaHttp.apikeyMgr./tokenplan/personal/api/v2/";
const USAGE_CACHE_TTL_MS = 60_000;
const TIER_CACHE_TTL_MS = 60 * 60_000;
// Window keys surfaced to the dashboard / quota-window registry
export const QWEN_TOKEN_PLAN_WINDOW_5H = "window_5h";
export const QWEN_TOKEN_PLAN_WINDOW_WEEKLY = "window_weekly";
// usage payload field prefix → window key (fields: per<prefix>Percentage / per<prefix>ResetTime)
const WINDOW_FIELD_MAP: Record<string, string> = {
"5Hour": QWEN_TOKEN_PLAN_WINDOW_5H,
"1Week": QWEN_TOKEN_PLAN_WINDOW_WEEKLY,
};
export interface QwenTokenPlanQuota extends QuotaInfo {
windows: Record<string, { percentUsed: number; resetAt: string | null }>;
/** Which console served the quota — drives the plan label shown in the dashboard. */
consoleSite: TokenPlanConsoleSite["consoleSite"];
/** Subscription tier (e.g. "pro") or null when the subscription call failed. */
specCode: string | null;
/** Credit limits of the active tier (from quota-config), when resolvable. */
tierLimits: { fiveHour: number | null; weekly: number | null };
}
interface UsageCacheEntry {
quota: QwenTokenPlanQuota;
fetchedAt: number;
}
interface TierCacheEntry {
specCode: string | null;
tierLimits: { fiveHour: number | null; weekly: number | null };
fetchedAt: number;
}
const usageCache = new Map<string, UsageCacheEntry>();
const tierCache = new Map<string, TierCacheEntry>();
const secTokenCache = new Map<string, { token: string; fetchedAt: number }>();
const _cacheCleanup = setInterval(() => {
const now = Date.now();
for (const [key, entry] of usageCache) {
if (now - entry.fetchedAt > USAGE_CACHE_TTL_MS * 5) usageCache.delete(key);
}
for (const [key, entry] of tierCache) {
if (now - entry.fetchedAt > TIER_CACHE_TTL_MS * 2) tierCache.delete(key);
}
for (const [key, entry] of secTokenCache) {
if (now - entry.fetchedAt > TIER_CACHE_TTL_MS * 2) secTokenCache.delete(key);
}
}, 5 * 60_000);
if (typeof _cacheCleanup === "object" && "unref" in _cacheCleanup) {
(_cacheCleanup as { unref?: () => void }).unref?.();
}
// ─── Helpers ─────────────────────────────────────────────────────────────────
function toRecord(value: unknown): Record<string, unknown> {
return value && typeof value === "object" && !Array.isArray(value)
? (value as Record<string, unknown>)
: {};
}
function toNumberOrNull(value: unknown): number | null {
if (typeof value === "number" && Number.isFinite(value)) return value;
if (typeof value === "string") {
const parsed = parseFloat(value);
if (Number.isFinite(parsed)) return parsed;
}
return null;
}
function toTrimmedString(value: unknown): string {
return typeof value === "string" ? value.trim() : "";
}
function getCookie(providerSpecificData: Record<string, unknown> | undefined): string {
for (const key of ["qwenCloudCookie", "alibabaConsoleCookie", "cookie"]) {
const value = toTrimmedString(providerSpecificData?.[key]);
if (value) return value;
}
return process.env.QWEN_CLOUD_COOKIE?.trim() || "";
}
function getConfiguredSecToken(providerSpecificData: Record<string, unknown> | undefined): string {
for (const key of ["qwenCloudSecToken", "alibabaConsoleSecToken"]) {
const value = toTrimmedString(providerSpecificData?.[key]);
if (value) return value;
}
return process.env.QWEN_CLOUD_SEC_TOKEN?.trim() || "";
}
function getGatewayHost(site: TokenPlanConsoleSite): string {
const configured = process.env.QWEN_TOKEN_PLAN_HOST?.trim();
if (!configured) return site.gatewayHost;
return /^https?:\/\//i.test(configured) ? configured : `https://${configured}`;
}
function getDashboardUrl(site: TokenPlanConsoleSite): string {
return process.env.QWEN_TOKEN_PLAN_DASHBOARD_URL?.trim() || site.dashboardUrl;
}
/** Extract the console `SEC_TOKEN: "…"` embedded in the logged-in dashboard HTML. */
export function extractQwenSecToken(html: string): string | null {
const match = /SEC_?TOKEN["']?\s*[:=]\s*["']([^"']+)["']/i.exec(html);
return match ? match[1] : null;
}
async function resolveSecToken(
connectionId: string,
cookie: string,
site: TokenPlanConsoleSite
): Promise<string> {
const cached = secTokenCache.get(connectionId);
if (cached && Date.now() - cached.fetchedAt < TIER_CACHE_TTL_MS) {
return cached.token;
}
try {
const response = await fetch(getDashboardUrl(site), {
method: "GET",
headers: {
Cookie: cookie,
"User-Agent":
"Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0 Safari/537.36",
Accept: "text/html",
},
redirect: "follow",
signal: AbortSignal.timeout(8_000),
});
const html = await response.text();
const token = extractQwenSecToken(html);
if (token) {
secTokenCache.set(connectionId, { token, fetchedAt: Date.now() });
return token;
}
} catch {
// best-effort — some accounts work without sec_token
}
return "";
}
// ─── Gateway transport ───────────────────────────────────────────────────────
async function callGateway(
endpoint: string,
cookie: string,
secToken: string,
site: TokenPlanConsoleSite
): Promise<unknown | null> {
const api = `${TOKEN_PLAN_API_PREFIX}${endpoint}`;
const url = `${getGatewayHost(site)}/data/api.json?product=${GATEWAY_PRODUCT}&action=${GATEWAY_ACTION}&api=${encodeURIComponent(api)}`;
const params = JSON.stringify({
Api: api,
V: "1.0",
Data: {
commodityCode: COMMODITY_CODE,
cornerstoneParam: {
console: "ONE_CONSOLE",
consoleSite: site.consoleSite,
domain: site.domain,
productCode: "p_efm",
protocol: "V2",
xsp_lang: "en-US",
},
},
});
const body = new URLSearchParams({
product: GATEWAY_PRODUCT,
action: GATEWAY_ACTION,
sec_token: secToken,
region: GATEWAY_REGION,
params,
});
try {
// #6911: space concurrent upstream quota fetches (mirrors bailianQuotaFetcher.ts).
await throttleQuotaFetch();
const response = await fetch(url, {
method: "POST",
headers: {
Cookie: cookie,
"Content-Type": "application/x-www-form-urlencoded",
Accept: "application/json",
Origin: site.origin,
Referer: `${site.origin}/`,
},
body: body.toString(),
signal: AbortSignal.timeout(8_000),
});
const raw = await response.json();
return parseGatewayEnvelope(raw);
} catch {
// Network error, timeout, non-JSON (login redirect page) — fail open
return null;
}
}
/** Unwrap {code:"200", data:{DataV2:{data:{code:"SUCCESS", data:<payload>}}}} → payload. */
function parseGatewayEnvelope(raw: unknown): unknown | null {
const obj = toRecord(raw);
if (obj["code"] !== "200" && obj["code"] !== 200) return null;
const inner = toRecord(toRecord(toRecord(obj["data"])["DataV2"])["data"]);
if (inner["code"] !== "SUCCESS" || inner["success"] !== true) return null;
return inner["data"] ?? null;
}
// ─── Parsers ─────────────────────────────────────────────────────────────────
function parseUsageWindows(
payload: unknown
): Record<string, { percentUsed: number; resetAt: string | null }> {
const obj = toRecord(payload);
const windows: Record<string, { percentUsed: number; resetAt: string | null }> = {};
for (const [fieldPrefix, windowKey] of Object.entries(WINDOW_FIELD_MAP)) {
const percent = toNumberOrNull(obj[`per${fieldPrefix}Percentage`]);
if (percent === null) continue; // window omitted (e.g. 5-hour "Temporarily Removed")
const resetMs = toNumberOrNull(obj[`per${fieldPrefix}ResetTime`]);
windows[windowKey] = {
percentUsed: percent,
resetAt: resetMs && resetMs > 0 ? new Date(resetMs).toISOString() : null,
};
}
return windows;
}
async function resolveTierInfo(
connectionId: string,
cookie: string,
secToken: string,
site: TokenPlanConsoleSite
): Promise<TierCacheEntry> {
const cached = tierCache.get(connectionId);
if (cached && Date.now() - cached.fetchedAt < TIER_CACHE_TTL_MS) {
return cached;
}
const [quotaConfig, subscription] = await Promise.all([
callGateway("quota-config", cookie, secToken, site),
callGateway("subscription", cookie, secToken, site),
]);
const specCode = toTrimmedString(toRecord(subscription)["specCode"]) || null;
const tierRecord = specCode ? toRecord(toRecord(quotaConfig)[specCode]) : {};
const entry: TierCacheEntry = {
specCode,
tierLimits: {
fiveHour: toNumberOrNull(tierRecord["five_hour"]),
weekly: toNumberOrNull(tierRecord["weekly"]),
},
fetchedAt: Date.now(),
};
tierCache.set(connectionId, entry);
return entry;
}
// ─── Core fetcher ────────────────────────────────────────────────────────────
/**
* Fetch the personal Token Plan quota for a qwen-cloud-token-plan connection.
* Returns percentUsed = max across the windows present in the usage response,
* or null when no cookie is configured / the console session expired.
*/
export async function fetchQwenTokenPlanQuota(
connectionId: string,
connection?: Record<string, unknown>
): Promise<QuotaInfo | null> {
const cached = usageCache.get(connectionId);
if (cached && Date.now() - cached.fetchedAt < USAGE_CACHE_TTL_MS) {
return cached.quota;
}
const providerSpecificData =
connection?.providerSpecificData &&
typeof connection.providerSpecificData === "object" &&
!Array.isArray(connection.providerSpecificData)
? (connection.providerSpecificData as Record<string, unknown>)
: undefined;
const cookie = getCookie(providerSpecificData);
if (!cookie) return null;
const site = resolveConsoleSite(
cookie,
typeof connection?.provider === "string" ? connection.provider : undefined
);
const secToken =
getConfiguredSecToken(providerSpecificData) ||
(await resolveSecToken(connectionId, cookie, site));
const usagePayload = await callGateway("usage", cookie, secToken, site);
if (usagePayload === null) return null;
const windows = parseUsageWindows(usagePayload);
const windowEntries = Object.values(windows);
if (windowEntries.length === 0) return null;
const worst = windowEntries.reduce((max, w) => (w.percentUsed > max.percentUsed ? w : max));
const tier = await resolveTierInfo(connectionId, cookie, secToken, site);
const total = tier.tierLimits.weekly ?? 100;
const quota: QwenTokenPlanQuota = {
used: Math.round(worst.percentUsed * total),
total,
percentUsed: worst.percentUsed,
resetAt: worst.resetAt,
windows,
consoleSite: site.consoleSite,
specCode: tier.specCode,
tierLimits: tier.tierLimits,
limitReached: worst.percentUsed >= 1,
};
usageCache.set(connectionId, { quota, fetchedAt: Date.now() });
return quota;
}
// ─── Invalidation ────────────────────────────────────────────────────────────
export function invalidateQwenTokenPlanQuotaCache(connectionId: string): void {
usageCache.delete(connectionId);
tierCache.delete(connectionId);
secTokenCache.delete(connectionId);
}
// ─── Registration ────────────────────────────────────────────────────────────
/**
* Register the Qwen Token Plan quota fetcher with the preflight and monitor systems.
* Call once at server startup (src/sse/handlers/chat.ts), BEFORE registerGenericQuotaFetchers().
*/
export function registerQwenTokenPlanQuotaFetcher(): void {
registerQuotaFetcher("qwen-cloud-token-plan", fetchQwenTokenPlanQuota);
registerMonitorFetcher("qwen-cloud-token-plan", fetchQwenTokenPlanQuota);
registerQuotaWindows("qwen-cloud-token-plan", [
QWEN_TOKEN_PLAN_WINDOW_5H,
QWEN_TOKEN_PLAN_WINDOW_WEEKLY,
]);
}

View File

@@ -69,7 +69,6 @@ import { getXaiOauthUsage } from "./usage/xaiOauth.ts";
import { getGrokCliUsage } from "./usage/grokCli.ts";
import { getFirecrawlUsage } from "./usage/firecrawl.ts";
import { getCommandCodeUsage } from "./usage/command-code.ts";
import { getQwenTokenPlanUsage } from "./usage/qwen-token-plan.ts";
import { getConolUsage } from "./conolUsage.ts";
type JsonRecord = Record<string, unknown>;
@@ -112,7 +111,6 @@ export const USAGE_FETCHER_PROVIDERS = [
"minimax-cn",
"crof",
"bailian-coding-plan",
"qwen-cloud-token-plan",
"nanogpt",
"deepseek",
"opencode",
@@ -204,8 +202,6 @@ export async function getUsageForProvider(
return await getCrofUsage(apiKey || "");
case "bailian-coding-plan":
return await getBailianCodingPlanUsage(id || "", apiKey || "", providerSpecificData);
case "qwen-cloud-token-plan":
return await getQwenTokenPlanUsage(id || "", apiKey || "", providerSpecificData);
case "nanogpt":
return await getNanoGptUsage(apiKey || "");
case "deepseek":

View File

@@ -10,7 +10,6 @@
*/
import { fetchBailianQuota, type BailianTripleWindowQuota } from "../bailianQuotaFetcher.ts";
import { getQwenTokenPlanUsage } from "./qwen-token-plan.ts";
/**
* Bailian (Alibaba Token Plan) Usage
@@ -22,25 +21,11 @@ export async function getBailianCodingPlanUsage(
providerSpecificData?: Record<string, unknown>
) {
try {
// The catalog entry is "Alibaba Token Plan" and now points at the Token Plan
// endpoint, so prefer the Token Plan quota (console cookie) when one is
// configured. The Coding Plan path below stays as the fallback for accounts
// that really do hold a Coding Plan key (#9603).
const tokenPlanUsage = await getQwenTokenPlanUsage(
connectionId,
apiKey,
providerSpecificData,
"bailian-coding-plan"
);
if ("quotas" in tokenPlanUsage) return tokenPlanUsage;
const connection = { apiKey, providerSpecificData };
const quota = await fetchBailianQuota(connectionId, connection);
if (!quota) {
// Neither surface answered — surface the Token Plan guidance, which tells the
// operator how to supply the cookie the console gateway requires.
return tokenPlanUsage;
return { message: "Alibaba Token Plan connected. Unable to fetch quota." };
}
const bailianQuota = quota as BailianTripleWindowQuota;

View File

@@ -1,96 +0,0 @@
/**
* usage/qwen-token-plan.ts — Qwen Cloud / Alibaba Model Studio personal Token Plan
* usage leaf (issue #9603).
*
* Delegates to qwenTokenPlanQuotaFetcher (cookie-authenticated console gateway) and
* shapes the 5-hour / weekly sliding windows into the standard usage response. The
* inference API key cannot read this quota — the connection needs a console session
* cookie in providerSpecificData (qwenCloudCookie / alibabaConsoleCookie / cookie)
* or the QWEN_CLOUD_COOKIE env var.
*/
import {
fetchQwenTokenPlanQuota,
QWEN_TOKEN_PLAN_WINDOW_5H,
QWEN_TOKEN_PLAN_WINDOW_WEEKLY,
type QwenTokenPlanQuota,
} from "../qwenTokenPlanQuotaFetcher.ts";
import type { UsageQuota } from "./quota.ts";
function windowToQuota(
window: { percentUsed: number; resetAt: string | null } | undefined,
totalCredits: number | null,
displayName: string
): UsageQuota | null {
if (!window) return null;
const total = totalCredits ?? 100;
const used = Math.round(window.percentUsed * total);
const remaining = Math.max(0, total - used);
return {
used,
total,
remaining,
remainingPercentage: Math.round((1 - window.percentUsed) * 1000) / 10,
resetAt: window.resetAt,
unlimited: false,
displayName,
};
}
/**
* Qwen Cloud personal Token Plan usage (5-hour + weekly sliding windows).
*/
export async function getQwenTokenPlanUsage(
connectionId: string,
apiKey: string,
providerSpecificData?: Record<string, unknown>,
provider = "qwen-cloud-token-plan"
) {
try {
const quota = await fetchQwenTokenPlanQuota(connectionId, {
apiKey,
providerSpecificData,
provider,
});
if (!quota) {
return {
message:
"Qwen Token Plan connected. Quota needs a console session cookie — the inference " +
"API key cannot read it. Get it at home.qwencloud.com Billing Subscription " +
"(logged in): F12 Network, reload, filter by api.json, click a request to " +
"cs-data.qwencloud.com and copy the whole Cookie value from Request Headers " +
"(it contains login_qwencloud_ticket). Paste it into the connection's " +
"'Qwen / Model Studio console cookie' field, or set QWEN_CLOUD_COOKIE. " +
"The cookie expires with the browser session — re-paste it when this message returns.",
};
}
const tokenPlanQuota = quota as QwenTokenPlanQuota;
const quotas: Record<string, UsageQuota> = {};
const fiveHour = windowToQuota(
tokenPlanQuota.windows[QWEN_TOKEN_PLAN_WINDOW_5H],
tokenPlanQuota.tierLimits.fiveHour,
"5-hour window"
);
if (fiveHour) quotas.five_hour = fiveHour;
const weekly = windowToQuota(
tokenPlanQuota.windows[QWEN_TOKEN_PLAN_WINDOW_WEEKLY],
tokenPlanQuota.tierLimits.weekly,
"Weekly window"
);
if (weekly) quotas.weekly = weekly;
const specCode = tokenPlanQuota.specCode;
const brand = tokenPlanQuota.consoleSite === "ALIYUN" ? "Alibaba" : "Qwen";
const plan = specCode
? `${brand} Token Plan (${specCode.charAt(0).toUpperCase()}${specCode.slice(1)})`
: `${brand} Token Plan`;
return { plan, quotas };
} catch (error) {
return { message: `Qwen Token Plan error: ${(error as Error).message}` };
}
}

View File

@@ -340,8 +340,6 @@ export default function EditConnectionModal({
opencodeGoAuthCookie: "",
ollamaCloudUsageCookie: "",
alibabaConsoleCookie: stringField(connection.providerSpecificData?.alibabaConsoleCookie),
qwenCloudCookie: stringField(connection.providerSpecificData?.qwenCloudCookie),
qwenCloudSecToken: stringField(connection.providerSpecificData?.qwenCloudSecToken),
alibabaConsoleSecToken: stringField(
connection.providerSpecificData?.alibabaConsoleSecToken
),

View File

@@ -3,16 +3,44 @@
import { Input } from "@/shared/components";
import { providerText, type ProviderMessageTranslator } from "../../providerPageHelpers";
import {
assignQuotaScrapingProviderData,
EMPTY_QUOTA_SCRAPING_FIELDS,
QWEN_TOKEN_PLAN_PROVIDERS,
type QuotaScrapingFieldValues,
} from "./quotaScrapingFieldValues";
export type QuotaScrapingFieldValues = {
opencodeGoWorkspaceId: string;
opencodeGoAuthCookie: string;
ollamaCloudUsageCookie: string;
alibabaConsoleCookie: string;
alibabaConsoleSecToken: string;
};
// Re-exported so existing importers (modals, tests) keep their current paths.
export { assignQuotaScrapingProviderData, EMPTY_QUOTA_SCRAPING_FIELDS };
export type { QuotaScrapingFieldValues };
export const EMPTY_QUOTA_SCRAPING_FIELDS: QuotaScrapingFieldValues = {
opencodeGoWorkspaceId: "",
opencodeGoAuthCookie: "",
ollamaCloudUsageCookie: "",
alibabaConsoleCookie: "",
alibabaConsoleSecToken: "",
};
export function assignQuotaScrapingProviderData(
provider: string | undefined,
values: QuotaScrapingFieldValues,
target: Record<string, unknown>
) {
if (provider === "opencode-go") {
target.opencodeGoWorkspaceId = values.opencodeGoWorkspaceId.trim() || undefined;
if (values.opencodeGoAuthCookie.trim()) {
target.opencodeGoAuthCookie = values.opencodeGoAuthCookie.trim();
}
} else if (provider === "ollama-cloud" && values.ollamaCloudUsageCookie.trim()) {
target.ollamaCloudUsageCookie = values.ollamaCloudUsageCookie.trim();
} else if (
(provider === "alibaba" || provider === "alibaba-cn") &&
values.alibabaConsoleCookie.trim()
) {
target.alibabaConsoleCookie = values.alibabaConsoleCookie.trim();
if (values.alibabaConsoleSecToken.trim()) {
target.alibabaConsoleSecToken = values.alibabaConsoleSecToken.trim();
}
}
}
type QuotaScrapingFieldsProps = {
provider?: string;
@@ -142,50 +170,5 @@ export default function QuotaScrapingFields({
);
}
if (QWEN_TOKEN_PLAN_PROVIDERS.has(provider ?? "")) {
return (
<div className="flex flex-col gap-3 rounded-lg border border-border/50 bg-surface/20 p-4">
<Input
label={providerText(t, "qwenCloudCookieLabel", "Qwen / Model Studio console cookie")}
name="qwenCloudCookie"
type="password"
value={values.qwenCloudCookie}
onChange={(e) => onChange({ qwenCloudCookie: e.target.value })}
placeholder="cna=...; login_qwencloud_ticket=...; ..."
hint={providerText(
t,
"qwenCloudCookieHint",
(editMode ? "Leave blank to keep the stored cookie. " : "") +
"Required for Token Plan quota — the inference API key cannot read it. " +
"How to get it: open home.qwencloud.com Billing Subscription while logged in, " +
"press F12 Network, reload the page, filter by api.json, click any request to " +
"cs-data.qwencloud.com, then under Request Headers copy the WHOLE Cookie value " +
"(it contains login_qwencloud_ticket). It expires with the browser session — " +
"re-paste it when the quota reports an expired session."
)}
autoComplete="off"
spellCheck={false}
autoCapitalize="off"
/>
<Input
label={providerText(t, "qwenCloudSecTokenLabel", "Qwen console sec_token (optional)")}
name="qwenCloudSecToken"
type="password"
value={values.qwenCloudSecToken}
onChange={(e) => onChange({ qwenCloudSecToken: e.target.value })}
placeholder="GjRV..."
hint={providerText(
t,
"qwenCloudSecTokenHint",
"Optional — resolved automatically from the dashboard. Set it only if quota sync reports a permission error."
)}
autoComplete="off"
spellCheck={false}
autoCapitalize="off"
/>
</div>
);
}
return null;
}

View File

@@ -1,64 +0,0 @@
/**
* quotaScrapingFieldValues.ts — form-state shape + persistence rules for the
* quota-scraping credential fields (cookies / workspace ids) rendered by
* QuotaScrapingFields.tsx.
*
* Kept in a UI-free module on purpose: importing the .tsx pulls in
* `@/shared/components`, whose barrel reaches untranspiled ESM deps
* (@lobehub/icons) that the node:test runner cannot parse. Unit tests import
* this file instead; the component re-exports it for existing callers.
*/
/** Providers whose quota lives behind the Qwen/Model Studio console gateway (#9603). */
export const QWEN_TOKEN_PLAN_PROVIDERS = new Set(["qwen-cloud-token-plan", "bailian-coding-plan"]);
export type QuotaScrapingFieldValues = {
opencodeGoWorkspaceId: string;
opencodeGoAuthCookie: string;
ollamaCloudUsageCookie: string;
alibabaConsoleCookie: string;
alibabaConsoleSecToken: string;
qwenCloudCookie: string;
qwenCloudSecToken: string;
};
export const EMPTY_QUOTA_SCRAPING_FIELDS: QuotaScrapingFieldValues = {
opencodeGoWorkspaceId: "",
opencodeGoAuthCookie: "",
ollamaCloudUsageCookie: "",
alibabaConsoleCookie: "",
alibabaConsoleSecToken: "",
qwenCloudCookie: "",
qwenCloudSecToken: "",
};
export function assignQuotaScrapingProviderData(
provider: string | undefined,
values: QuotaScrapingFieldValues,
target: Record<string, unknown>
) {
if (provider === "opencode-go") {
target.opencodeGoWorkspaceId = values.opencodeGoWorkspaceId.trim() || undefined;
if (values.opencodeGoAuthCookie.trim()) {
target.opencodeGoAuthCookie = values.opencodeGoAuthCookie.trim();
}
} else if (provider === "ollama-cloud" && values.ollamaCloudUsageCookie.trim()) {
target.ollamaCloudUsageCookie = values.ollamaCloudUsageCookie.trim();
} else if (
(provider === "alibaba" || provider === "alibaba-cn") &&
values.alibabaConsoleCookie.trim()
) {
target.alibabaConsoleCookie = values.alibabaConsoleCookie.trim();
if (values.alibabaConsoleSecToken.trim()) {
target.alibabaConsoleSecToken = values.alibabaConsoleSecToken.trim();
}
} else if (QWEN_TOKEN_PLAN_PROVIDERS.has(provider ?? "") && values.qwenCloudCookie?.trim()) {
// Optional access: callers (AddApiKeyModal/EditConnectionModal form state, and
// existing tests) may pass a partial form object without the newer fields —
// bailian-coding-plan previously matched no branch here at all.
target.qwenCloudCookie = values.qwenCloudCookie.trim();
if (values.qwenCloudSecToken?.trim()) {
target.qwenCloudSecToken = values.qwenCloudSecToken.trim();
}
}
}

View File

@@ -97,9 +97,6 @@ const PROVIDER_LIMITS_APIKEY_PROVIDERS = new Set([
"command-code",
"conol-web",
"cnl",
// Alibaba Coding Plan (console API key) + Qwen personal Token Plan (console cookie) — #9603
"bailian-coding-plan",
"qwen-cloud-token-plan",
]);
const DEFAULT_PROVIDER_LIMITS_SYNC_INTERVAL_MINUTES = 70;
const PROVIDER_LIMITS_AUTO_SYNC_SETTING_KEY = "provider_limits_auto_sync_last_run";

View File

@@ -500,10 +500,6 @@ export const USAGE_SUPPORTED_PROVIDERS = [
"command-code",
"conol-web",
"cnl",
// Alibaba Coding Plan triple-window quota (#9603 UI gap — fetcher existed, list entry missing)
"bailian-coding-plan",
// Qwen Cloud / Model Studio personal Token Plan (cookie-authenticated console gateway)
"qwen-cloud-token-plan",
];
// ── Zod validation at module load (Phase 7.2) ──

View File

@@ -14,7 +14,6 @@
import { CORS_HEADERS } from "../utils/cors";
import { createHash } from "crypto";
import v8 from "node:v8";
const OMNIROUTE_CHAT_VIRTUAL_TTL_MS = parsePositiveInt(
@@ -84,41 +83,6 @@ export const CHAT_HEAVY_ESTIMATED_TOKENS = parsePositiveInt(
process.env.OMNIROUTE_CHAT_HEAVY_ESTIMATED_TOKENS,
32_000
);
/**
* Heap-pressure shed ratio for the structural admission gate (#10183, #10268).
*
* 3.8.48 only shed a heavy request once `heapUsed / heapLimit >= shedRatio` (0.75).
* 3.8.49 (#9654/#9940) replaced that heap-conditional shed with an unconditional
* `CHAT_MAX_HEAVY_IN_FLIGHT=1` structural lease, so a second concurrent "heavy"
* request (coding-agent fan-out is the common trigger) was hard-rejected with a
* retryable 503 even on a host with ample free RAM. This restores the heap
* condition as an ADDITIONAL gate layered on top of the bounded-concurrency /
* per-connection-lane protection from #9654 (that protection stays in force —
* this constant only decides whether a *busy* lease is still shed with a 503 or
* admitted anyway because the heap has real headroom).
*/
export const CHAT_ADMISSION_HEAP_SHED_RATIO = (() => {
const parsed = Number(process.env.OMNIROUTE_CHAT_ADMISSION_HEAP_SHED_RATIO);
return Number.isFinite(parsed) && parsed > 0 && parsed <= 1 ? parsed : 0.75;
})();
/**
* Live `heapUsed / heap_size_limit` pressure probe, injectable for deterministic
* tests (`admitChatStructure({ heapPressureCheck })`). Defaults to the real V8
* heap statistics. Any read failure is treated as "not under pressure" so a
* transient stats error never turns into a false structural shed.
*/
export function defaultHeapPressureCheck(): boolean {
try {
const heapUsed = process.memoryUsage().heapUsed;
const heapLimit = v8.getHeapStatistics().heap_size_limit;
if (!Number.isFinite(heapLimit) || heapLimit <= 0) return false;
return heapUsed / heapLimit >= CHAT_ADMISSION_HEAP_SHED_RATIO;
} catch {
return false;
}
}
/**
* Optional per-deployment history cap. `0` (the default) disables it.
*
@@ -523,12 +487,6 @@ export async function admitChatStructure(
heavyTokens?: number;
queueMs?: number;
signal?: AbortSignal;
/**
* Heap-pressure probe consulted only when heavyweight capacity is busy
* (#10183, #10268). Defaults to `defaultHeapPressureCheck` (live V8 heap
* stats). Tests inject a deterministic override.
*/
heapPressureCheck?: () => boolean;
} = {}
): Promise<ChatStructureAdmission> {
if (!body || typeof body !== "object" || Array.isArray(body)) return { admit: true, lease };
@@ -566,26 +524,6 @@ export async function admitChatStructure(
(options.sessionId
? perConnectionAdmissionController.getController(options.sessionId)
: defaultAdmissionController);
// Uncontended fast path: capacity is free, no need to consult heap pressure at all.
const immediate = controller.tryAcquireHeavy();
if (immediate) return { admit: true, lease: immediate };
// Heavyweight capacity is momentarily busy (a concurrent heavy request holds the
// lease). #10183 / #10268: only enter the bounded-wait / shed path — with its
// queued-bytes heap valve and abort handling (#9654) — when the heap is
// GENUINELY under pressure. This restores the 3.8.48 `heapUsed/heapLimit >=
// shedRatio` condition as an additional gate on top of (never a replacement
// for) the bounded-concurrency / per-connection-lane protection above. A
// healthy heap has real headroom for a second heavy request even while the
// single lease is momentarily busy, so admit it immediately instead of
// parking/shedding a request that has nothing to do with actual resource
// pressure.
const heapPressureCheck = options.heapPressureCheck ?? defaultHeapPressureCheck;
if (!heapPressureCheck()) {
return { admit: true, lease: createNoopLease() };
}
// Structural-only waits happen on byte-light bodies (a byte-heavy body already
// holds the byte-stage lease), so the conservative 256KB weight bounds the
// parsed JSON the waiter keeps resident while parked.
@@ -658,18 +596,14 @@ export function resolveSelfLoopBearer(): string {
* gap that kept the Zoo Code / api-key describe call failing even after the byte
* stage was bypassed. Release is a no-op; capacity was never reserved.
*/
function createNoopLease(): ChatAdmissionLease {
return {
get released() {
return true;
},
release() {
// No-op: this sentinel never reserved heavyweight capacity.
},
};
}
const NULL_LEASE: ChatAdmissionLease = createNoopLease();
const NULL_LEASE: ChatAdmissionLease = {
get released() {
return true;
},
release() {
// No-op: the sentinel never reserved heavyweight capacity.
},
};
/**
* True when the request is a trusted in-process self-loop sub-request that must

View File

@@ -328,8 +328,6 @@ export function validateProviderSpecificData(
"usageCookie",
"alibabaConsoleCookie",
"alibabaConsoleSecToken",
"qwenCloudCookie",
"qwenCloudSecToken",
] as const) {
const value = data[key];
if (value !== undefined && value !== null && typeof value !== "string") {

View File

@@ -148,7 +148,6 @@ import {
registerCodexQuotaFetcher,
} from "@omniroute/open-sse/services/codexQuotaFetcher.ts";
import { registerBailianCodingPlanQuotaFetcher } from "@omniroute/open-sse/services/bailianQuotaFetcher.ts";
import { registerQwenTokenPlanQuotaFetcher } from "@omniroute/open-sse/services/qwenTokenPlanQuotaFetcher.ts";
import { registerCrofUsageFetcher } from "@omniroute/open-sse/services/crofUsageFetcher.ts";
import { registerDeepseekQuotaFetcher } from "@omniroute/open-sse/services/deepseekQuotaFetcher.ts";
import { registerOpenrouterQuotaFetcher } from "@omniroute/open-sse/services/openrouterQuotaFetcher.ts";
@@ -172,11 +171,6 @@ registerCodexQuotaFetcher();
// can proactively switch accounts before quota is exhausted.
registerBailianCodingPlanQuotaFetcher();
// Register the Qwen Cloud / Model Studio personal Token Plan fetcher (#9603).
// Cookie-authenticated console gateway — 5-hour + weekly sliding windows.
// Runs before registerGenericQuotaFetchers so the bespoke fetcher wins.
registerQwenTokenPlanQuotaFetcher();
// Register CrofAI usage fetcher (subscription requests + credits balance).
// Surfaces usable_requests + credits in the monitor and only blocks (preflight
// opt-in) when the active bucket reaches zero.

View File

@@ -1,66 +0,0 @@
// #10183: regression 3.8.48 → 3.8.49 — chat admission rejected a second concurrent
// "heavy" request even on a healthy heap. `admitChatStructure`'s CHAT_MAX_HEAVY_IN_FLIGHT=1
// cap (#9654/#9940) sheds unconditionally once busy; this test proves shedding must be
// gated on real heap pressure (restoring 3.8.48's `heapUsed/heapLimit >= shedRatio`
// semantics) instead of firing regardless of free memory.
import { test } from "node:test";
import assert from "node:assert/strict";
import {
ChatAdmissionController,
admitChatStructure,
} from "../../src/shared/middleware/chatBodyAdmission.ts";
function heavyBody() {
return {
messages: Array.from({ length: 200 }, () => ({
role: "user",
content: "x".repeat(400),
})),
tools: [] as unknown[],
};
}
test("bug-10183: second concurrent heavy request admitted on a healthy heap", async () => {
const controller = new ChatAdmissionController(1); // default CHAT_MAX_HEAVY_IN_FLIGHT=1
const first = await admitChatStructure(heavyBody(), null, { controller });
assert.equal(first.admit, true);
assert.ok(first.admit && first.lease, "first heavy request should hold the lease");
try {
const second = await admitChatStructure(heavyBody(), null, {
controller,
queueMs: 50,
// No override: default heap probe reads live process stats, which are
// healthy in the test process — proves the fix without mocking away the
// real check.
});
assert.equal(second.admit, true, "healthy heap must not shed a 2nd heavy request");
if (second.admit) second.lease?.release();
} finally {
if (first.admit) first.lease?.release();
}
});
test("bug-10183: a genuinely pressured heap still sheds the 2nd heavy request", async () => {
const controller = new ChatAdmissionController(1);
const first = await admitChatStructure(heavyBody(), null, { controller });
assert.equal(first.admit, true);
assert.ok(first.admit && first.lease);
try {
const second = await admitChatStructure(heavyBody(), null, {
controller,
queueMs: 0,
heapPressureCheck: () => true, // simulate real heap pressure
});
assert.equal(second.admit, false, "real heap pressure must still shed the 2nd request");
if (!second.admit) {
assert.equal(second.response.status, 503);
const payload = await second.response.json();
assert.equal(payload.error.code, "chat_admission_busy");
assert.equal(payload.error.reason, "structure_limit");
}
} finally {
if (first.admit) first.lease?.release();
}
});

View File

@@ -43,9 +43,6 @@ test("a heavy structural request waits for capacity instead of failing immediate
heavyTools: 10,
heavyTokens: 10_000,
queueMs: 500,
// #10183/#10268: entry into the bounded-wait path requires real heap
// pressure now; force it so this test still exercises the wait.
heapPressureCheck: () => true,
}
);
@@ -88,9 +85,6 @@ test("waiting for admission times out into a retryable 503", async () => {
heavyTools: 10,
heavyTokens: 10_000,
queueMs: 50,
// #10183/#10268: entry into the bounded-wait/shed path requires real
// heap pressure now; force it to still exercise the timeout.
heapPressureCheck: () => true,
}
);
@@ -154,9 +148,6 @@ test("expired admission queue keeps the legacy immediate 503 behaviour", async (
heavyTools: 10,
heavyTokens: 10_000,
queueMs: 0,
// #10183/#10268: shedding now requires real heap pressure; force it to
// still exercise the legacy immediate-reject path.
heapPressureCheck: () => true,
}
);
@@ -183,9 +174,6 @@ test("admission waiters are served FIFO as capacity frees", async () => {
heavyTools: 10,
heavyTokens: 10_000,
queueMs: 500,
// #10183/#10268: entry into the bounded-wait path requires real heap
// pressure now; force it so both waiters still queue.
heapPressureCheck: () => true,
};
const first = admitChatStructure(body, null, options);
const second = admitChatStructure(body, null, options);
@@ -379,9 +367,6 @@ test("structural admission enforces the queued-bytes cap end-to-end", async () =
heavyTools: 10,
heavyTokens: 10_000,
queueMs: 2_000,
// #10183/#10268: entry into the bounded-wait path requires real heap
// pressure now; force it so the queued-bytes cap is still exercised.
heapPressureCheck: () => true,
};
// First structural wait parks, charging the conservative 256KB weight.
@@ -500,9 +485,6 @@ test("aborting the signal cancels a structural queue-wait", async () => {
heavyTokens: 10_000,
queueMs: 2_000,
signal: abortController.signal,
// #10183/#10268: entry into the bounded-wait path requires real heap
// pressure now; force it so the abort is still exercised mid-wait.
heapPressureCheck: () => true,
}
);

View File

@@ -80,7 +80,7 @@ test("a byte-light request above the message threshold acquires heavyweight capa
assert.equal(controller.activeHeavy, 0);
});
test("a byte-light request above the tool threshold is rejected when heavy capacity is busy AND the heap is genuinely under pressure (#10183/#10268)", async () => {
test("a byte-light request above the tool threshold is rejected when heavy capacity is busy", async () => {
const controller = new ChatAdmissionController(1);
const occupied = controller.tryAcquireHeavy();
assert.ok(occupied);
@@ -88,16 +88,7 @@ test("a byte-light request above the tool threshold is rejected when heavy capac
const result = await admitChatStructure(
{ messages: [], tools: [{ type: "function" }, { type: "function" }] },
null,
{
controller,
maxMessages: 10,
heavyMessages: 10,
heavyTools: 2,
heavyTokens: 10_000,
// #10183/#10268: shedding is now conditional on real heap pressure, not
// capacity alone — simulate the pressured case this test targets.
heapPressureCheck: () => true,
}
{ controller, maxMessages: 10, heavyMessages: 10, heavyTools: 2, heavyTokens: 10_000 }
);
assert.equal(result.admit, false);
@@ -144,7 +135,7 @@ test("no history cap is enforced by default; long conversations are admitted", a
result.lease?.release();
});
test("an uncapped oversized conversation still yields to occupied heavyweight capacity when the heap is genuinely under pressure (#10183/#10268)", async () => {
test("an uncapped oversized conversation still yields to occupied heavyweight capacity", async () => {
const controller = new ChatAdmissionController(1);
const occupied = controller.tryAcquireHeavy();
assert.ok(occupied);
@@ -152,15 +143,7 @@ test("an uncapped oversized conversation still yields to occupied heavyweight ca
const result = await admitChatStructure(
{ messages: Array.from({ length: 5_000 }, () => ({ role: "user", content: "x" })) },
null,
{
controller,
maxMessages: 0,
heavyMessages: 200,
heavyTools: 64,
heavyTokens: 32_000,
// #10183/#10268: shedding is now conditional on real heap pressure.
heapPressureCheck: () => true,
}
{ controller, maxMessages: 0, heavyMessages: 200, heavyTools: 64, heavyTokens: 32_000 }
);
assert.equal(result.admit, false);

View File

@@ -149,7 +149,7 @@ test("admitChatRequest with explicit controller overrides per-connection lookup"
if (result.admit) result.lease?.release();
});
test("admitChatStructure routes structural rejection to per-connection controller when heap pressure is genuinely high (#10183/#10268)", async () => {
test("admitChatStructure routes structural rejection to per-connection controller", async () => {
// occupy sess-a's per-connection controller via the module-level instance
const controller = perConnectionAdmissionController.getController("sess-a");
const occupied = controller.tryAcquireHeavy();
@@ -166,8 +166,6 @@ test("admitChatStructure routes structural rejection to per-connection controlle
heavyMessages: 1,
heavyTools: 10,
heavyTokens: 10_000,
// #10183/#10268: shedding is now conditional on real heap pressure.
heapPressureCheck: () => true,
}
);
// Session A is busy → 503

View File

@@ -1,70 +0,0 @@
// #10268: "[BUG] API call failed (attempt 1/3): InternalServerError [HTTP 503]" — Hermes
// Agent / Cursor coding-agent fan-out landed on the same structural admission gate as
// #10183 and burned its 3 retries on OmniRoute's own `chat_admission_busy` 503, which it
// misread as an upstream capacity error. Same root cause, same fix (heap-conditional
// shedding in `admitChatStructure`): this test is the permanent regression guard proving
// the exact reported 503 shape is still produced when heap pressure is GENUINELY high,
// so the #4380 heap-amplification shed path is preserved rather than removed outright.
import { test } from "node:test";
import assert from "node:assert/strict";
import {
admitChatStructure,
ChatAdmissionController,
type ChatAdmissionLease,
} from "../../src/shared/middleware/chatBodyAdmission.ts";
function heavyBody() {
const messages = Array.from({ length: 201 }, (_, i) => ({ role: "user", content: `prompt ${i}` }));
const tools = Array.from({ length: 32 }, (_, i) => ({
type: "function",
function: { name: `tool_${i}`, description: "a".repeat(64), parameters: { type: "object" } },
}));
return { model: "grok-4.5-fast-high", messages, tools, stream: true };
}
test("#10268: 2nd structurally-heavy agent request is rejected 503 (chat_admission_busy) under real heap pressure", async () => {
const controller = new ChatAdmissionController(1);
const first = await admitChatStructure(heavyBody(), null, { controller, queueMs: 0 });
assert.equal(first.admit, true);
const lease = (first as { admit: true; lease: ChatAdmissionLease | null }).lease;
assert.ok(lease);
try {
const second = await admitChatStructure(heavyBody(), null, {
controller,
queueMs: 0,
// Simulate genuine heap pressure (#10183/#10268 fix: shedding is now
// conditional on this, not unconditional on capacity alone).
heapPressureCheck: () => true,
});
assert.equal(second.admit, false); // reported failure path, still reachable under real pressure
const res = (second as { admit: false; response: Response }).response;
assert.equal(res.status, 503); // client is shown HTTP 503
const body = await res.json();
assert.equal(body.error?.message, "Structurally heavy chat request capacity is busy; retry shortly.");
assert.equal(body.error?.code, "chat_admission_busy");
assert.equal(body.error?.reason, "structure_limit");
} finally {
lease.release();
}
});
test("#10268: 2nd structurally-heavy agent request is admitted on a healthy heap (the fix)", async () => {
const controller = new ChatAdmissionController(1);
const first = await admitChatStructure(heavyBody(), null, { controller, queueMs: 0 });
assert.equal(first.admit, true);
const lease = (first as { admit: true; lease: ChatAdmissionLease | null }).lease;
assert.ok(lease);
try {
const second = await admitChatStructure(heavyBody(), null, {
controller,
queueMs: 0,
// No override: default heap probe reads live process stats (healthy here),
// reproducing legitimate Hermes/Cursor fan-out traffic that must no longer
// be shed on ample free RAM.
});
assert.equal(second.admit, true, "healthy heap must admit legitimate agent fan-out");
if (second.admit) second.lease?.release();
} finally {
lease.release();
}
});

View File

@@ -1,120 +0,0 @@
/**
* qwen-token-plan-console-site.test.ts — the personal Token Plan is sold through TWO
* consoles that share one backend, and the gateway validates the session against the
* console declared in the request. Sending the Alibaba console cookie with the
* QwenCloud console identity returns:
*
* {"errorCode":"BailianGateway.Login.NotLogined"}
*
* Verified live (2026-08-14) against both consoles: switching only consoleSite/domain/
* Origin/Referer (same cookie) turns that error into a real usage payload.
*/
import test from "node:test";
import assert from "node:assert/strict";
import {
resolveConsoleSite,
fetchQwenTokenPlanQuota,
invalidateQwenTokenPlanQuotaCache,
} from "../../open-sse/services/qwenTokenPlanQuotaFetcher.ts";
const originalFetch = globalThis.fetch;
test.afterEach(() => {
globalThis.fetch = originalFetch;
});
test("an Alibaba console cookie resolves to the Model Studio console", () => {
const site = resolveConsoleSite("cna=x; login_aliyunid_ticket=abc; aui=1", undefined);
assert.equal(site.consoleSite, "ALIYUN");
assert.equal(site.domain, "modelstudio.console.alibabacloud.com");
assert.ok(site.gatewayHost.includes("bailian-singapore-cs.alibabacloud.com"));
assert.ok(site.origin.includes("modelstudio.console.alibabacloud.com"));
});
test("a QwenCloud console cookie resolves to the QwenCloud console", () => {
const site = resolveConsoleSite("cna=x; login_qwencloud_ticket=abc", undefined);
assert.equal(site.consoleSite, "QWENCLOUD");
assert.equal(site.domain, "home.qwencloud.com");
assert.ok(site.gatewayHost.includes("cs-data.qwencloud.com"));
});
test("the provider decides when the cookie carries no console marker", () => {
assert.equal(resolveConsoleSite("session=opaque", "bailian-coding-plan").consoleSite, "ALIYUN");
assert.equal(
resolveConsoleSite("session=opaque", "qwen-cloud-token-plan").consoleSite,
"QWENCLOUD"
);
// Unknown provider + unmarked cookie keeps the QwenCloud default.
assert.equal(resolveConsoleSite("session=opaque", undefined).consoleSite, "QWENCLOUD");
});
test("fetch sends the Alibaba console identity for an aliyun cookie", async () => {
const connectionId = `console-site-${Date.now()}`;
const calls: { url: string; init?: RequestInit }[] = [];
globalThis.fetch = (async (input: RequestInfo | URL, init?: RequestInit) => {
const url = String(input);
calls.push({ url, init });
const body = {
code: "200",
data: {
DataV2: {
data: {
code: "SUCCESS",
success: true,
data: url.includes("%2Fusage")
? { per1WeekPercentage: 0.32, per1WeekResetTime: 1787254140000 }
: url.includes("%2Fsubscription")
? { specCode: "pro" }
: { pro: { five_hour: 12000, weekly: 40000 } },
},
},
success: true,
},
httpStatusCode: "200",
};
return new Response(JSON.stringify(body), {
status: 200,
headers: { "content-type": "application/json" },
});
}) as typeof globalThis.fetch;
const quota = await fetchQwenTokenPlanQuota(connectionId, {
provider: "bailian-coding-plan",
providerSpecificData: {
qwenCloudCookie: "cna=x; login_aliyunid_ticket=abc",
qwenCloudSecToken: "tok",
},
});
assert.ok(quota, "expected quota");
assert.equal(quota.percentUsed, 0.32);
const usageCall = calls.find((c) => c.url.includes("%2Fusage"));
assert.ok(usageCall, "usage call missing");
assert.ok(
usageCall.url.includes("bailian-singapore-cs.alibabacloud.com"),
`wrong gateway host: ${usageCall.url}`
);
const headers = usageCall.init?.headers as Record<string, string>;
assert.ok(String(headers.Referer).includes("modelstudio.console.alibabacloud.com"));
const params = JSON.parse(
new URLSearchParams(String(usageCall.init?.body)).get("params") ?? "{}"
);
assert.equal(params.Data.cornerstoneParam.consoleSite, "ALIYUN");
assert.equal(params.Data.cornerstoneParam.domain, "modelstudio.console.alibabacloud.com");
invalidateQwenTokenPlanQuotaCache(connectionId);
});
test("bailian-coding-plan points at the Token Plan endpoint, not the Coding Plan one", async () => {
const { bailian_coding_planProvider } =
await import("../../open-sse/config/providers/registry/bailian-coding-plan/index.ts");
// The catalog entry is named "Alibaba Token Plan" and links to token-plan-overview;
// coding-intl.dashscope.aliyuncs.com only accepts Coding Plan keys (401 otherwise).
assert.equal(
bailian_coding_planProvider.baseUrl,
"https://token-plan.ap-southeast-1.maas.aliyuncs.com/apps/anthropic/v1"
);
});

View File

@@ -1,101 +0,0 @@
/**
* qwen-token-plan-cookie-field.test.ts — the Qwen Token Plan quota fetcher is
* cookie-authenticated (the inference API key cannot read the console gateway),
* so the connection modal MUST expose a field to paste that cookie. Without it
* the quota is unconfigurable from the dashboard.
*
* Mirrors the existing ollama-cloud / alibaba console-cookie fields.
*/
import test from "node:test";
import assert from "node:assert/strict";
// Imports the UI-free module on purpose: pulling the .tsx would drag in
// `@/shared/components` → untranspiled ESM (@lobehub/icons) that node:test
// cannot parse ("SyntaxError: Unexpected token 'export'").
import {
EMPTY_QUOTA_SCRAPING_FIELDS,
assignQuotaScrapingProviderData,
} from "../../src/app/(dashboard)/dashboard/providers/[id]/components/modals/quotaScrapingFieldValues.ts";
const { updateProviderConnectionSchema } = await import("../../src/shared/validation/schemas.ts");
test("qwen-cloud-token-plan persists the console cookie and optional sec_token", () => {
const target: Record<string, unknown> = {};
assignQuotaScrapingProviderData(
"qwen-cloud-token-plan",
{
...EMPTY_QUOTA_SCRAPING_FIELDS,
qwenCloudCookie: " token=abc123; aux=1 ",
qwenCloudSecToken: " sec-tok ",
},
target
);
assert.equal(target.qwenCloudCookie, "token=abc123; aux=1", "cookie must be stored trimmed");
assert.equal(target.qwenCloudSecToken, "sec-tok", "sec_token must be stored trimmed");
});
test("bailian-coding-plan reuses the same console cookie field", () => {
const target: Record<string, unknown> = {};
assignQuotaScrapingProviderData(
"bailian-coding-plan",
{ ...EMPTY_QUOTA_SCRAPING_FIELDS, qwenCloudCookie: "token=xyz" },
target
);
assert.equal(target.qwenCloudCookie, "token=xyz");
});
test("a blank cookie does not overwrite the stored one", () => {
const target: Record<string, unknown> = {};
assignQuotaScrapingProviderData(
"qwen-cloud-token-plan",
{ ...EMPTY_QUOTA_SCRAPING_FIELDS, qwenCloudCookie: " " },
target
);
assert.equal(
Object.hasOwn(target, "qwenCloudCookie"),
false,
"blank input must leave the stored cookie untouched"
);
});
test("a form object without the newer cookie fields does not throw", () => {
// Regression: adding bailian-coding-plan to the qwen branch made older callers
// (which build a partial form object) reach code that assumed the fields exist.
const target: Record<string, unknown> = {};
const partial = { ...EMPTY_QUOTA_SCRAPING_FIELDS } as Record<string, string>;
delete partial.qwenCloudCookie;
delete partial.qwenCloudSecToken;
for (const provider of ["bailian-coding-plan", "qwen-cloud-token-plan"]) {
assert.doesNotThrow(() =>
assignQuotaScrapingProviderData(
provider,
partial as unknown as typeof EMPTY_QUOTA_SCRAPING_FIELDS,
target
)
);
}
assert.equal(Object.hasOwn(target, "qwenCloudCookie"), false);
});
test("providerSpecificData validation guards the qwen cookie fields", () => {
const ok = updateProviderConnectionSchema.safeParse({
providerSpecificData: { qwenCloudCookie: "token=abc", qwenCloudSecToken: "sec-tok" },
});
assert.equal(ok.success, true, JSON.stringify(ok.error?.issues));
const wrongType = updateProviderConnectionSchema.safeParse({
providerSpecificData: { qwenCloudCookie: 42 },
});
assert.equal(wrongType.success, false, "non-string cookie must be rejected");
const tooLong = updateProviderConnectionSchema.safeParse({
providerSpecificData: { qwenCloudCookie: "x".repeat(10_001) },
});
assert.equal(tooLong.success, false, "oversized cookie must be rejected");
});

View File

@@ -1,272 +0,0 @@
/**
* qwen-token-plan-quota-fetcher.test.ts — Qwen Cloud / Alibaba Model Studio personal
* Token Plan quota fetcher (issue #9603, Problema 1: quota is missing).
*
* Fixtures captured live (2026-08-13) from home.qwencloud.com/billing/subscription/
* token-plan-individual — console gateway POST cs-data.qwencloud.com/data/api.json
* (action=IntlBroadScopeAspnGateway, product=sfm_bailian), cookie-authenticated.
*/
import test from "node:test";
import assert from "node:assert/strict";
import {
QWEN_TOKEN_PLAN_WINDOW_5H,
QWEN_TOKEN_PLAN_WINDOW_WEEKLY,
extractQwenSecToken,
fetchQwenTokenPlanQuota,
invalidateQwenTokenPlanQuotaCache,
registerQwenTokenPlanQuotaFetcher,
} from "../../open-sse/services/qwenTokenPlanQuotaFetcher.ts";
const originalFetch = globalThis.fetch;
const RESET_MS = 1786714740000; // 2026-08-14 10:39 (captured per1WeekResetTime)
type FetchCall = { url: string; init: RequestInit | undefined };
function gatewayBody(payload: unknown, api: string): string {
return JSON.stringify({
code: "200",
data: {
DataV2: {
ret: ["SUCCESS::ok"],
data: { msg: "Success.", code: "SUCCESS", data: payload, success: true },
},
success: true,
httpStatus: 200,
errorCode: "",
api,
errorMsg: "",
},
httpStatusCode: "200",
successResponse: true,
});
}
const USAGE_PAYLOAD = { per1WeekResetTime: RESET_MS, per1WeekPercentage: 0.55 };
const QUOTA_CONFIG_PAYLOAD = {
standard: { five_hour: 3000.0, weekly: 10000.0 },
addon_quota: { extrabundle: 20000.0 },
lite: { five_hour: 700.0, weekly: 2500.0 },
pro: { five_hour: 12000.0, weekly: 40000.0 },
};
const SUBSCRIPTION_PAYLOAD = {
instanceCode: "sfm_tokenplansolo_public_intl-sg-test",
specCode: "pro",
remainingDays: 24,
startTime: 1786109803000,
endTime: 1788796800000,
autoRenewFlag: false,
status: "VALID",
};
function mockGateway(
calls: FetchCall[],
overrides?: { usagePayload?: unknown; dashboardHtml?: string }
): void {
globalThis.fetch = (async (input: RequestInfo | URL, init?: RequestInit) => {
const url = String(input);
calls.push({ url, init });
if (!url.includes("/data/api.json")) {
// Dashboard HTML fetch (sec_token resolution)
return new Response(overrides?.dashboardHtml ?? "<html>no token here</html>", {
status: 200,
headers: { "content-type": "text/html" },
});
}
const jsonHeaders = { "content-type": "application/json" };
if (url.includes("%2Fusage")) {
const payload =
overrides && "usagePayload" in overrides ? overrides.usagePayload : USAGE_PAYLOAD;
return new Response(gatewayBody(payload, "usage"), { status: 200, headers: jsonHeaders });
}
if (url.includes("%2Fquota-config")) {
return new Response(gatewayBody(QUOTA_CONFIG_PAYLOAD, "quota-config"), {
status: 200,
headers: jsonHeaders,
});
}
if (url.includes("%2Fsubscription")) {
return new Response(gatewayBody(SUBSCRIPTION_PAYLOAD, "subscription"), {
status: 200,
headers: jsonHeaders,
});
}
return new Response(JSON.stringify({ code: "404" }), { status: 404, headers: jsonHeaders });
}) as typeof globalThis.fetch;
}
test.beforeEach(() => {
delete process.env.QWEN_CLOUD_COOKIE;
delete process.env.QWEN_CLOUD_SEC_TOKEN;
});
test.afterEach(() => {
globalThis.fetch = originalFetch;
});
test("fetchQwenTokenPlanQuota returns null without any cookie configured", async () => {
const calls: FetchCall[] = [];
mockGateway(calls);
const quota = await fetchQwenTokenPlanQuota(`qwen-nocookie-${Date.now()}`, {});
assert.equal(quota, null);
assert.equal(calls.length, 0);
});
test("fetchQwenTokenPlanQuota parses the captured weekly-only usage response", async () => {
const connectionId = `qwen-weekly-${Date.now()}`;
const calls: FetchCall[] = [];
mockGateway(calls);
const quota = await fetchQwenTokenPlanQuota(connectionId, {
providerSpecificData: { qwenCloudCookie: "token=abc123; aux=1", qwenCloudSecToken: "sec-tok" },
});
assert.ok(quota, "expected quota, got null");
assert.equal(quota.percentUsed, 0.55);
assert.equal(quota.resetAt, new Date(RESET_MS).toISOString());
const windows = (
quota as { windows: Record<string, { percentUsed: number; resetAt: string | null }> }
).windows;
assert.ok(windows[QWEN_TOKEN_PLAN_WINDOW_WEEKLY], "weekly window missing");
assert.equal(windows[QWEN_TOKEN_PLAN_WINDOW_WEEKLY].percentUsed, 0.55);
assert.equal(windows[QWEN_TOKEN_PLAN_WINDOW_WEEKLY].resetAt, new Date(RESET_MS).toISOString());
// 5-hour window "Temporarily Removed" → API omits per5Hour* fields → no window
assert.equal(windows[QWEN_TOKEN_PLAN_WINDOW_5H], undefined);
// Tier totals resolved via subscription.specCode → quota-config.pro
assert.equal(quota.total, 40000);
assert.equal(quota.used, Math.round(0.55 * 40000));
assert.equal((quota as { specCode: string | null }).specCode, "pro");
// Request contract (captured shape)
const usageCall = calls.find((c) => c.url.includes("%2Fusage"));
assert.ok(usageCall, "usage gateway call missing");
assert.equal(usageCall.init?.method, "POST");
const headers = usageCall.init?.headers as Record<string, string>;
assert.ok(String(headers["Cookie"] ?? headers["cookie"]).includes("token=abc123"));
const body = String(usageCall.init?.body);
assert.ok(body.includes("product=sfm_bailian"), "body missing product");
assert.ok(body.includes("action=IntlBroadScopeAspnGateway"), "body missing action");
assert.ok(body.includes("region=ap-southeast-1"), "body missing region");
assert.ok(body.includes("sec_token=sec-tok"), "body missing sec_token");
const params = new URLSearchParams(body).get("params");
assert.ok(params, "body missing params");
const parsedParams = JSON.parse(params) as {
Api: string;
V: string;
Data: { commodityCode: string };
};
assert.equal(parsedParams.V, "1.0");
assert.ok(parsedParams.Api.includes("/tokenplan/personal/api/v2/usage"));
assert.equal(parsedParams.Data.commodityCode, "sfm_tokenplansolo_public_intl");
invalidateQwenTokenPlanQuotaCache(connectionId);
});
test("fetchQwenTokenPlanQuota includes the 5-hour window when the API returns it", async () => {
const connectionId = `qwen-5h-${Date.now()}`;
const calls: FetchCall[] = [];
mockGateway(calls, {
usagePayload: {
per1WeekResetTime: RESET_MS,
per1WeekPercentage: 0.55,
per5HourResetTime: RESET_MS - 3_600_000,
per5HourPercentage: 0.7,
},
});
const quota = await fetchQwenTokenPlanQuota(connectionId, {
providerSpecificData: { qwenCloudCookie: "token=abc", qwenCloudSecToken: "sec-tok" },
});
assert.ok(quota, "expected quota, got null");
const windows = (
quota as { windows: Record<string, { percentUsed: number; resetAt: string | null }> }
).windows;
assert.equal(windows[QWEN_TOKEN_PLAN_WINDOW_5H]?.percentUsed, 0.7);
// worst window wins
assert.equal(quota.percentUsed, 0.7);
assert.equal(quota.resetAt, new Date(RESET_MS - 3_600_000).toISOString());
invalidateQwenTokenPlanQuotaCache(connectionId);
});
test("fetchQwenTokenPlanQuota returns null when the console session expired", async () => {
const connectionId = `qwen-expired-${Date.now()}`;
globalThis.fetch = (async () =>
new Response(JSON.stringify({ code: "ConsoleNeedLogin" }), {
status: 200,
headers: { "content-type": "application/json" },
})) as typeof globalThis.fetch;
const quota = await fetchQwenTokenPlanQuota(connectionId, {
providerSpecificData: { qwenCloudCookie: "token=stale", qwenCloudSecToken: "sec-tok" },
});
assert.equal(quota, null);
});
test("fetchQwenTokenPlanQuota resolves sec_token from the dashboard when absent", async () => {
const connectionId = `qwen-sectoken-${Date.now()}`;
const calls: FetchCall[] = [];
mockGateway(calls, {
dashboardHtml:
'<script>window.X = { IS_CERTIFIED: "true", SEC_TOKEN: "resolved-tok" };</script>',
});
const quota = await fetchQwenTokenPlanQuota(connectionId, {
providerSpecificData: { qwenCloudCookie: "token=abc" },
});
assert.ok(quota, "expected quota, got null");
const dashboardCall = calls.find((c) => !c.url.includes("/data/api.json"));
assert.ok(dashboardCall, "dashboard fetch for sec_token missing");
const usageCall = calls.find((c) => c.url.includes("%2Fusage"));
assert.ok(String(usageCall?.init?.body).includes("sec_token=resolved-tok"));
invalidateQwenTokenPlanQuotaCache(connectionId);
});
test("fetchQwenTokenPlanQuota serves the second call from cache", async () => {
const connectionId = `qwen-cache-${Date.now()}`;
const calls: FetchCall[] = [];
mockGateway(calls);
const connection = {
providerSpecificData: { qwenCloudCookie: "token=abc", qwenCloudSecToken: "sec-tok" },
};
const first = await fetchQwenTokenPlanQuota(connectionId, connection);
assert.ok(first);
const callCountAfterFirst = calls.length;
const second = await fetchQwenTokenPlanQuota(connectionId, connection);
assert.ok(second);
assert.equal(calls.length, callCountAfterFirst);
invalidateQwenTokenPlanQuotaCache(connectionId);
});
test("extractQwenSecToken pulls SEC_TOKEN out of dashboard HTML", () => {
assert.equal(extractQwenSecToken('foo SEC_TOKEN: "abc-123", bar'), "abc-123");
assert.equal(extractQwenSecToken("<html>nothing</html>"), null);
});
test("registerQwenTokenPlanQuotaFetcher registers without throwing", () => {
registerQwenTokenPlanQuotaFetcher();
});
test("qwen-cloud-token-plan and bailian-coding-plan are wired into the usage/UI lists", async () => {
const { USAGE_FETCHER_PROVIDERS } = await import("../../open-sse/services/usage.ts");
const { USAGE_SUPPORTED_PROVIDERS } = await import("../../src/shared/constants/providers.ts");
assert.ok((USAGE_FETCHER_PROVIDERS as readonly string[]).includes("qwen-cloud-token-plan"));
assert.ok((USAGE_SUPPORTED_PROVIDERS as readonly string[]).includes("qwen-cloud-token-plan"));
// #9603 UI gap: coding-plan connections were filtered out of /dashboard/quota
assert.ok((USAGE_SUPPORTED_PROVIDERS as readonly string[]).includes("bailian-coding-plan"));
});