Files
OmniRoute/open-sse/executors/glm.ts
Nguyen Thanh Dat af49d4972e fix(stream): accept the buffer size glm.ts has been passing since #12179 (#12925)
Rebased onto the tip and completed, per the maintainer's call to finish the wiring rather than merge the capability alone.

What changed since your version:

The tip had already cleared the TS2554 by deleting the 16th argument, leaving a comment that the highWaterMark stays at the helper default. So the base-red you found is gone, but the 64 KB #12179 asked for was still not applied and your new parameter had no caller. glm.ts now passes it, which is what turns the capability into the fix.

Your test file also hung the runner: every stream createSSEStream builds arms a 10s idle watchdog via setInterval in start, and nothing cancelled them, so node:test waited on a non-empty event loop long after the assertions passed. Cancelling each readable in an after hook runs the cancel handler that clears the timer — the file now reports in about 7 seconds. Worth knowing for future stream tests.

Your five assertions are unchanged and all pass. Reading the writable's desiredSize to measure the queue budget the stream was actually built with, rather than standing in for it, is the detail that makes this testable at all — and the 0-budget case pinning `??` against `||` is the kind of thing that silently rots otherwise.

Thank you also for separating your own red checks from the base's and reporting what you found there. That is how #12919's identical failures got explained instead of chased.
2026-09-10 18:25:31 -03:00

624 lines
22 KiB
TypeScript

import { randomUUID } from "node:crypto";
import type { KeyHealth } from "../services/apiKeyRotator.ts";
import { DefaultExecutor } from "./default.ts";
import {
applyConfiguredUserAgent,
mergeAbortSignals,
mergeUpstreamExtraHeaders,
type CountTokensInput,
type ExecuteInput,
type ProviderCredentials,
} from "./base.ts";
import {
buildGlmBaseHeaders,
buildGlmChatUrl,
buildGlmCodingHeaders,
buildGlmCountTokensUrl,
GLM_COUNT_TOKENS_TIMEOUT_MS,
type GlmTransport,
getGlmTransport,
} from "../config/glmProvider.ts";
import { applyProviderRequestDefaults } from "../services/providerRequestDefaults.ts";
import { stripUnsupportedParams } from "../translator/paramSupport.ts";
import { getRotatingApiKey } from "../services/apiKeyRotator.ts";
import { CLAUDE_CLI_STAINLESS_PACKAGE_VERSION } from "../config/anthropicHeaders.ts";
import {
getRuntimeVersion,
normalizeStainlessArch,
normalizeStainlessPlatform,
} from "../config/providerHeaderProfiles.ts";
import { translateNonStreamingResponse } from "../handlers/responseTranslator.ts";
import { translateRequest } from "../translator/index.ts";
import { FORMATS } from "../translator/formats.ts";
import { createSSETransformStreamWithLogger } from "../utils/stream.ts";
import { ensureStreamReadiness } from "../utils/streamReadiness.ts";
import { STREAM_READINESS_TIMEOUT_MS } from "../config/constants.ts";
import { resolveSuppressThinkClose, THINKING_MARKER_HEADER } from "../utils/thinkCloseMarker.ts";
type JsonRecord = Record<string, unknown>;
type GlmExecuteResult = Awaited<ReturnType<DefaultExecutor["execute"]>> & {
targetFormat?: string;
};
function asRecord(value: unknown): JsonRecord | null {
return value && typeof value === "object" && !Array.isArray(value) ? (value as JsonRecord) : null;
}
function getEffectiveKey(credentials: ProviderCredentials): string {
const extraKeys = (credentials.providerSpecificData?.extraApiKeys as string[] | undefined) ?? [];
if (credentials.apiKey && credentials.connectionId && extraKeys.length > 0) {
return getRotatingApiKey(credentials.connectionId, credentials.apiKey, extraKeys);
}
return credentials.apiKey || credentials.accessToken || "";
}
export type GlmEffortLevel = "low" | "high" | "max";
type GlmEffortTier = {
baseModel: string;
effort: GlmEffortLevel;
/** Transport where the upstream honors the effort selector for this family. */
transport: GlmTransport;
};
/**
* GLM-5.2 effort tiers (glm-5.2-high/-max) route exclusively through the
* Anthropic transport, where Zhipu maps Claude Code effort selectors (high/max)
* to reasoning intensity. The base model ID sent upstream is always "glm-5.2".
*
* GLM-5.3 replaced tier endpoints with a documented `reasoning_effort` request
* parameter (low|high|max, default max) on the coding chat/completions endpoint,
* so its tiers stay on the OpenAI transport and inject `reasoning_effort` +
* `thinking.type=enabled` (5.3 no longer accepts thinking disabled).
*
* https://docs.z.ai/devpack/latest-model
* https://docs.z.ai/guides/llm/glm-5.3
*/
function parseGlmEffortTier(model: string): GlmEffortTier | null {
switch (model) {
case "glm-5.2-high":
return { baseModel: "glm-5.2", effort: "high", transport: "anthropic" };
case "glm-5.2-max":
return { baseModel: "glm-5.2", effort: "max", transport: "anthropic" };
case "glm-5.3-high":
return { baseModel: "glm-5.3", effort: "high", transport: "openai" };
case "glm-5.3-low":
return { baseModel: "glm-5.3", effort: "low", transport: "openai" };
case "glm-5.3-max":
return { baseModel: "glm-5.3", effort: "max", transport: "openai" };
case "glm-5.3-flash-high":
return { baseModel: "glm-5.3-flash", effort: "high", transport: "openai" };
case "glm-5.3-flash-low":
return { baseModel: "glm-5.3-flash", effort: "low", transport: "openai" };
case "glm-5.3-flash-max":
return { baseModel: "glm-5.3-flash", effort: "max", transport: "openai" };
default:
return null;
}
}
/**
* Detects GLM models that support deep thinking (5.2+).
* These models share a single max_tokens budget for reasoning + response
* (Z.AI does not document a separate thinking budget). When the client
* doesn't explicitly request max_tokens, we default to the model's full
* output capacity so reasoning isn't truncated by a low generic default.
*
* To add future models (e.g. glm-5.3, glm-5.4), just extend the regex.
* https://docs.z.ai/guides/overview/concept-param
*/
const GLM_THINKING_MODEL_PATTERN = /^glm-5\.(?:[2-9]|\d{2,})/i;
const GLM_53_OR_HIGHER_PATTERN = /^glm-5\.(?:[3-9]|\d{2,})/i;
function isGlmThinkingModel(model: string): boolean {
return GLM_THINKING_MODEL_PATTERN.test(model);
}
/**
* Z.AI's official max output for GLM-5.2+ is 131072 tokens (128K).
* This budget covers BOTH reasoning and the final response.
* https://z.ai/blog/glm-5.2
*/
const GLM_THINKING_DEFAULT_MAX_TOKENS = 131072;
function applyGlmRequestDefaults(body: unknown, defaults?: JsonRecord | null): unknown {
const record = asRecord(body);
if (!record || !defaults) return body;
const next = { ...(applyProviderRequestDefaults(record, defaults) as JsonRecord) };
const thinkingType = typeof defaults.thinkingType === "string" ? defaults.thinkingType : null;
if (thinkingType && next.thinking === undefined) {
next.thinking = { type: thinkingType };
} else if (thinkingType && asRecord(next.thinking)?.type === "enabled") {
next.thinking = { ...asRecord(next.thinking), type: thinkingType };
}
return next;
}
function hasTools(body: unknown): boolean {
const record = asRecord(body);
return Array.isArray(record?.tools) && record.tools.length > 0;
}
function isRetryableGlmFallbackStatus(status: number): boolean {
return status === 404 || status === 408 || status === 409 || status === 429 || status >= 500;
}
function isRetryableGlmFallbackError(error: unknown): boolean {
if (!error) return false;
const err = error instanceof Error ? error : new Error(String(error));
if (err.name === "AbortError") return false;
return true;
}
function cloneHeaders(headers: Headers): Headers {
const next = new Headers();
headers.forEach((value, key) => next.set(key, value));
return next;
}
function isJsonResponse(response: Response): boolean {
return (response.headers.get("content-type") || "").toLowerCase().includes("application/json");
}
async function translateJsonResponse(response: Response): Promise<Response> {
const parsed = await response.json().catch(() => null);
const translated = translateNonStreamingResponse(parsed, FORMATS.CLAUDE, FORMATS.OPENAI);
const headers = cloneHeaders(response.headers);
headers.set("content-type", "application/json");
headers.delete("content-length");
return new Response(JSON.stringify(translated), {
status: response.status,
statusText: response.statusText,
headers,
});
}
async function translateAnthropicJsonResponse(response: Response): Promise<Response> {
const parsed = await response.json().catch(() => null);
const translated = response.ok
? translateNonStreamingResponse(parsed, FORMATS.CLAUDE, FORMATS.OPENAI)
: translateAnthropicJsonError(parsed);
const headers = cloneHeaders(response.headers);
headers.set("content-type", "application/json");
headers.delete("content-length");
return new Response(JSON.stringify(translated), {
status: response.status,
statusText: response.statusText,
headers,
});
}
function translateAnthropicJsonError(parsed: unknown): JsonRecord {
const root = asRecord(parsed) || {};
const error = asRecord(root.error) || root;
const message =
typeof error.message === "string" && error.message.trim()
? error.message
: typeof root.message === "string" && root.message.trim()
? root.message
: "GLM Anthropic transport error";
const type =
typeof error.type === "string" && error.type.trim()
? error.type
: typeof root.type === "string" && root.type.trim()
? root.type
: "upstream_error";
return {
error: {
message,
type,
},
};
}
/** 64 KB queue budget for GLM streaming (#12179, wired through in #12925). */
const GLM_STREAM_BUFFER_BYTES = 65536;
export function translateSseResponse(
response: Response,
provider: string,
model: string,
suppressThinkClose: boolean = false
): Response {
if (!response.body) return response;
// GLM is a high-throughput provider: a 64 KB queue budget keeps provider ->
// client pacing ahead of the model's emission rate. #12179 asked for this by
// passing a 16th positional the helper did not take (a TS2554 that never
// reached the TransformStream); the helper now accepts it as its last
// parameter, so the request finally takes effect (#12925).
const transform = createSSETransformStreamWithLogger(
FORMATS.CLAUDE,
FORMATS.OPENAI,
provider,
null,
null,
model,
null,
null,
null,
null,
null,
false,
suppressThinkClose,
undefined,
undefined,
GLM_STREAM_BUFFER_BYTES
);
const headers = cloneHeaders(response.headers);
headers.set("content-type", "text/event-stream");
headers.delete("content-length");
return new Response(response.body.pipeThrough(transform), {
status: response.status,
statusText: response.statusText,
headers,
});
}
export class GlmExecutor extends DefaultExecutor {
constructor(provider = "glm") {
super(provider);
}
buildUrl(
_model: string,
_stream: boolean,
_urlIndex = 0,
credentials: ProviderCredentials | null = null
) {
const primaryTransport = getGlmTransport(credentials?.providerSpecificData);
const transport =
_urlIndex === 1 ? (primaryTransport === "openai" ? "anthropic" : "openai") : primaryTransport;
return buildGlmChatUrl(credentials?.providerSpecificData, transport, this.config.baseUrl);
}
buildCountTokensUrl(_model: string, credentials: ProviderCredentials | null = null) {
return buildGlmCountTokensUrl(credentials?.providerSpecificData, this.config.baseUrl);
}
getCountTokensTimeoutMs() {
return GLM_COUNT_TOKENS_TIMEOUT_MS;
}
buildHeaders(
credentials: ProviderCredentials,
stream = true,
_clientHeaders?: Record<string, string> | null,
_model?: string,
_health?: unknown,
_body?: unknown
): Record<string, string> {
const transport: GlmTransport = getGlmTransport(credentials.providerSpecificData);
if (transport === "openai") {
return buildGlmCodingHeaders(getEffectiveKey(credentials), stream);
}
return {
...buildGlmBaseHeaders(getEffectiveKey(credentials), stream),
"X-Stainless-Arch": normalizeStainlessArch(),
"X-Stainless-OS": normalizeStainlessPlatform(),
"X-Stainless-Runtime-Version": getRuntimeVersion(),
"X-Stainless-Package-Version": CLAUDE_CLI_STAINLESS_PACKAGE_VERSION,
"X-Claude-Code-Session-Id": randomUUID(),
"x-client-request-id": randomUUID(),
};
}
transformRequest(
model: string,
body: unknown,
stream: boolean,
credentials: ProviderCredentials
) {
const cleanedBody = super.transformRequest(model, body, stream, credentials);
return applyGlmRequestDefaults(cleanedBody, this.config.requestDefaults as JsonRecord | null);
}
transformForTransport(
model: string,
body: unknown,
stream: boolean,
credentials: ProviderCredentials,
transport: GlmTransport
) {
const effortTier = parseGlmEffortTier(model);
const effectiveModel = effortTier ? effortTier.baseModel : model;
const transformed = this.transformRequest(effectiveModel, body, stream, credentials);
const record = asRecord(transformed);
// #7364: unlike DefaultExecutor.execute() (default.ts), GlmExecutor.execute()
// never calls the base execute() loop — it drives its own fetch via
// executeTransport()/transformForTransport() — so stripUnsupportedParams()
// (normally applied at default.ts's execute() call site) never ran for GLM
// requests. Without this call, a STRIP_RULES clamp entry for provider "glm"
// (e.g. the glm-4.6v max_tokens ceiling) would be silently dead code.
if (record) stripUnsupportedParams(this.provider, effectiveModel, record);
// Ensure upstream receives the base model ID, not the effort-suffixed alias
if (record && effortTier) {
record.model = effectiveModel;
}
// GLM-5.2+ models share a single max_tokens budget for reasoning + response.
// When the client doesn't explicitly set max_tokens, default to the model's
// full output capacity (131072) so deep reasoning isn't truncated by the
// generic translator defaults (64000 for Anthropic, 16384 for OpenAI).
// This acts as the "transparent proxy override" described in Z.AI's own
// Terminal-Bench evaluation methodology.
// https://huggingface.co/blog/zai-org/glm-52-blog
if (record && isGlmThinkingModel(effectiveModel)) {
const clientBody = asRecord(body);
const clientMaxTokens = clientBody?.max_tokens ?? clientBody?.max_completion_tokens;
if (!clientMaxTokens) {
record.max_tokens = GLM_THINKING_DEFAULT_MAX_TOKENS;
}
}
if (transport === "openai") {
// GLM-5.3+ rejects thinking.type "disabled". Ensure thinking is enabled
// when targeting GLM-5.3 or higher.
if (record && GLM_53_OR_HIGHER_PATTERN.test(effectiveModel)) {
const existingThinking = asRecord(record.thinking);
if (existingThinking?.type === "disabled") {
record.thinking = { ...existingThinking, type: "enabled" };
}
}
// GLM-5.3 effort tiers: inject the documented `reasoning_effort` param and
// force thinking on — 5.3 rejects thinking.type "disabled", and an effort
// tier without thinking would silently drop the selector upstream.
if (record && effortTier && effortTier.transport === "openai") {
const existingThinking = asRecord(record.thinking);
record.thinking = { ...existingThinking, type: "enabled" };
record.reasoning_effort = effortTier.effort;
}
if (record && stream && hasTools(record) && record.tool_stream === undefined) {
return { ...record, tool_stream: true };
}
return transformed;
}
const translated = translateRequest(
FORMATS.OPENAI,
FORMATS.CLAUDE,
effectiveModel,
{ ...(record ?? {}), _disableToolPrefix: true },
stream,
credentials,
this.provider,
null,
{ preserveCacheControl: false }
);
// Inject effort and thinking for the Anthropic transport.
// Zhipu's Anthropic endpoint requires thinking.type=enabled to emit
// thinking_delta blocks in the SSE response. Without it, reasoning is
// not surfaced and clients see no thinking content.
// The effort-2025-11-24 beta header (in GLM_ANTHROPIC_BETA) carries
// the high/max intensity selector.
if (effortTier) {
const translatedRecord = asRecord(translated);
if (translatedRecord) {
translatedRecord.effort = effortTier.effort;
// Zhipu's Anthropic endpoint only supports thinking.type
// "enabled"/"disabled" — not "adaptive". Clients like Claude Code
// default to "adaptive" for reasoning models, so force "enabled"
// here while preserving any other fields (e.g. budget_tokens).
const existingThinking = asRecord(translatedRecord.thinking);
if (!existingThinking || existingThinking.type !== "enabled") {
translatedRecord.thinking = {
...existingThinking,
type: "enabled",
};
}
}
}
return translated;
}
private async executeTransport(
input: ExecuteInput,
transport: GlmTransport
): Promise<GlmExecuteResult> {
const credentials = input.credentials;
const url = buildGlmChatUrl(credentials?.providerSpecificData, transport, this.config.baseUrl);
// #10798 moved the transport out of buildHeaders' signature; the Anthropic
// transport must therefore be visible to buildHeaders through
// providerSpecificData (primaryTransport / anthropic-shaped baseUrl).
const headers =
transport === "anthropic"
? this.buildHeaders(
{
...credentials,
providerSpecificData: {
...credentials?.providerSpecificData,
primaryTransport: "anthropic",
},
},
input.stream,
input.clientHeaders,
input.model
)
: this.buildHeaders(credentials, input.stream, input.clientHeaders, input.model);
applyConfiguredUserAgent(headers, credentials.providerSpecificData);
mergeUpstreamExtraHeaders(headers, input.upstreamExtraHeaders);
const transformedBody = this.transformForTransport(
input.model,
input.body,
input.stream,
credentials,
transport
);
const fetchStartTimeoutMs = this.getTimeoutMs();
const timeoutController = fetchStartTimeoutMs > 0 ? new AbortController() : null;
let timeoutId: ReturnType<typeof setTimeout> | null = null;
if (timeoutController) {
timeoutId = setTimeout(() => {
const timeoutError = new Error(`Fetch timeout after ${fetchStartTimeoutMs}ms on ${url}`);
timeoutError.name = "TimeoutError";
timeoutController.abort(timeoutError);
}, fetchStartTimeoutMs);
}
const timeoutSignal = timeoutController?.signal ?? null;
const combinedSignal =
input.signal && timeoutSignal
? mergeAbortSignals(input.signal, timeoutSignal)
: input.signal || timeoutSignal;
let response: Response;
try {
this.assertOutboundUrlAllowed(url); // GHSA-4f49: glm has its own fetch path
response = await fetch(url, {
method: "POST",
headers,
body: JSON.stringify(transformedBody),
signal: combinedSignal || undefined,
});
} finally {
if (timeoutId) clearTimeout(timeoutId);
}
if (input.stream && response.ok) {
const readiness = await ensureStreamReadiness(response, {
timeoutMs: STREAM_READINESS_TIMEOUT_MS,
provider: this.provider,
model: input.model,
log: input.log,
});
response = readiness.response;
}
const result = { response, url, headers, transformedBody };
if (transport === "anthropic") {
return this.finalizeAnthropicTransportResult(input, result);
}
return {
...result,
url,
headers,
transformedBody,
targetFormat: FORMATS.OPENAI,
};
}
/**
* GLM's Anthropic transport does its own Claude→OpenAI translation
* (bypassing chatCore's stream), so the `</think>` close-marker
* suppression flag and the response translation both have to be resolved
* here from the original client headers (#5245 / #5312). Extracted from
* `executeTransport` to keep that method's cyclomatic complexity under the
* project cap.
*/
private async finalizeAnthropicTransportResult(
input: ExecuteInput,
result: {
response: Response;
url: string;
headers: Record<string, string>;
transformedBody: unknown;
}
): Promise<GlmExecuteResult> {
const { response: rawResponse, url, headers, transformedBody } = result;
const clientHeaders = input.clientHeaders ?? {};
const suppressThinkClose = resolveSuppressThinkClose({
userAgent: clientHeaders["user-agent"] ?? clientHeaders["User-Agent"] ?? null,
thinkingMarkerHeader:
clientHeaders[THINKING_MARKER_HEADER] ??
clientHeaders["x-omniroute-thinking-marker"] ??
null,
clientResponseFormat: input.clientResponseFormat ?? null,
});
const translatedResponse =
input.stream && rawResponse.ok
? translateSseResponse(rawResponse, this.provider, input.model, suppressThinkClose)
: isJsonResponse(rawResponse)
? await translateAnthropicJsonResponse(rawResponse)
: rawResponse;
return {
response: translatedResponse,
url,
headers,
transformedBody,
targetFormat: FORMATS.OPENAI,
};
}
async execute(input: ExecuteInput): Promise<GlmExecuteResult> {
const effortTier = parseGlmEffortTier(input.model);
// Effort tiers route directly through their family's transport (no fallback):
// GLM-5.2 → Anthropic (Zhipu only graduates effort there, via the
// effort-2025-11-24 beta header in GLM_ANTHROPIC_BETA); GLM-5.3 → OpenAI
// coding endpoint (`reasoning_effort` param). See parseGlmEffortTier.
if (effortTier) {
return this.executeTransport(input, effortTier.transport);
}
const primaryTransport = getGlmTransport(
input.credentials.providerSpecificData,
this.config.baseUrl
);
const fallbackTransport: GlmTransport = primaryTransport === "openai" ? "anthropic" : "openai";
let primaryResult: GlmExecuteResult | null = null;
try {
primaryResult = await this.executeTransport(input, primaryTransport);
if (!isRetryableGlmFallbackStatus(primaryResult.response.status)) {
return primaryResult;
}
input.log?.debug?.(
"GLM_FALLBACK",
`${primaryTransport} returned ${primaryResult.response.status}; trying ${fallbackTransport}`
);
} catch (error) {
if (!isRetryableGlmFallbackError(error)) throw error;
input.log?.debug?.(
"GLM_FALLBACK",
`${primaryTransport} error (${error instanceof Error ? error.message : String(error)}); trying ${fallbackTransport}`
);
}
try {
const fallbackResult = await this.executeTransport(input, fallbackTransport);
if (fallbackResult.response.ok || !primaryResult) {
return fallbackResult;
}
} catch (error) {
if (!primaryResult) throw error;
input.log?.debug?.(
"GLM_FALLBACK",
`${fallbackTransport} fallback failed (${error instanceof Error ? error.message : String(error)}); returning primary response`
);
}
return primaryResult;
}
async countTokens(input: CountTokensInput) {
return super.countTokens({
...input,
credentials: {
...input.credentials,
providerSpecificData: {
...(input.credentials.providerSpecificData || {}),
primaryTransport: "anthropic",
},
},
});
}
}
export default GlmExecutor;