diff --git a/docs/reference/PROVIDER_REFERENCE.md b/docs/reference/PROVIDER_REFERENCE.md index d25e0d1e47..63733efa31 100644 --- a/docs/reference/PROVIDER_REFERENCE.md +++ b/docs/reference/PROVIDER_REFERENCE.md @@ -1,14 +1,14 @@ --- title: "Provider Reference" -version: 3.8.42 -lastUpdated: 2026-06-30 +version: 3.8.43 +lastUpdated: 2026-07-02 --- # Provider Reference > **Auto-generated** from `src/shared/constants/providers.ts` — do not edit by hand. > Regenerate with: `npm run gen:provider-reference` -> **Last generated:** 2026-06-30 +> **Last generated:** 2026-07-02 Total providers: **237**. See category breakdown below. @@ -33,285 +33,285 @@ Use the dashboard at `/dashboard/providers` to enable, configure, and test each ## OAuth Providers (20) -| ID | Alias | Name | Tags | Website | Notes | -| -------------- | ------------ | -------------------- | ----- | ------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `agy` | `agy` | Antigravity CLI | OAuth | [link](https://antigravity.google) | Import your Antigravity CLI (`agy`) login (paste/upload its token file), auto-detect a local CLI login, or sign in with Google. Shares the Antigravity backend (incl. Claude models). | -| `amazon-q` | `aq` | Amazon Q | OAuth | [link](https://aws.amazon.com/q/developer/) | Uses the same AWS Builder ID or imported refresh-token flow as Kiro, but keeps Amazon Q connections separate. | -| `antigravity` | — | Antigravity | OAuth | — | — | -| `claude` | `cc` | Claude Code | OAuth | — | — | -| `cline` | `cl` | Cline | OAuth | — | — | -| `codebuddy-cn` | `cbcn` | CodeBuddy CN | OAuth | [link](https://copilot.tencent.com) | Tencent CodeBuddy CN (copilot.tencent.com). Sign in via the official CLI device-code flow, or paste a direct API key (sent as Authorization: Bearer). Catalog: GLM / Kimi / MiniMax / DeepSeek / Hunyuan. | -| `codex` | `cx` | OpenAI Codex | OAuth | — | — | -| `cursor` | `cu` | Cursor IDE | OAuth | — | — | -| `devin-cli` | `dv` | Devin CLI (Official) | OAuth | [link](https://cli.devin.ai) | Requires the Devin CLI binary. Run `devin auth login` to authenticate, or provide your WINDSURF_API_KEY. Install: https://cli.devin.ai | -| `github` | `gh` | GitHub Copilot | OAuth | — | — | -| `gitlab-duo` | `gitlab-duo` | GitLab Duo | OAuth | [link](https://docs.gitlab.com/user/duo_agent_platform/code_suggestions/) | OAuth application with ai_features + read_user scopes. Configure GITLAB_DUO_OAUTH_CLIENT_ID and optionally GITLAB_DUO_OAUTH_CLIENT_SECRET on this OmniRoute instance. | -| `grok-cli` | `gc` | Grok Build | OAuth | — | Paste your ~/.grok/auth.json (or the JWT access token) from the Grok Build CLI; refresh_token is rotated automatically. | -| `kilocode` | `kc` | Kilo Code | OAuth | — | — | -| `kimi-coding` | `kmc` | Kimi Coding | OAuth | — | — | -| `kiro` | `kr` | Kiro AI | OAuth | — | Free tier: 50 credits/month (~25K–100K tokens). ⚠️ Kiro ToS prohibits third-party proxy/harness use. | -| `qoder` | `if` | Qoder | OAuth | — | — | -| `qwen` | `qw` | Qwen Code | OAuth | — | ⚠️ **DEPRECATED.** Qwen OAuth free tier was discontinued on 2026-04-15. Use 'bailian-coding-plan', 'alibaba', 'alibaba-cn', or 'openrouter' provider with API key instead. | -| `trae` | `tr` | Trae | OAuth | [link](https://trae.ai) | Trae is an AI-native IDE by ByteDance (SOLO remote agent). Authorize via trae.ai in the popup, or sign in at solo.trae.ai and paste the Cloud-IDE-JWT (sent as 'Authorization: Cloud-IDE-JWT ', ~14-day lifetime) as the access token; web_id/biz_user_id/user_unique_id/scope/tenant/region propagate via providerSpecificData. No headless refresh for pasted tokens — re-paste on expiry. | -| `windsurf` | `ws` | Windsurf (Devin CLI) | OAuth | [link](https://windsurf.com) | In the Windsurf / VS Code IDE, open the command palette and run `Windsurf: Provide Auth Token` (or click the Jupyter "Get Windsurf Authentication Token" button), then copy the shown token and paste it here. Note: opening windsurf.com/show-auth-token directly only renders a "Redirecting" page — the IDE must initiate the flow (it adds a `?state=...` param) for the token to appear. | -| `zed` | `zd` | Zed IDE | OAuth | [link](https://zed.dev) | Zed stores LLM provider credentials (OpenAI, Anthropic, Google, Mistral, xAI) in the OS keychain. Use the Import button below to discover and import them automatically. | +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `agy` | `agy` | Antigravity CLI | OAuth | [link](https://antigravity.google) | Import your Antigravity CLI (`agy`) login (paste/upload its token file), auto-detect a local CLI login, or sign in with Google. Shares the Antigravity backend (incl. Claude models). | +| `amazon-q` | `aq` | Amazon Q | OAuth | [link](https://aws.amazon.com/q/developer/) | Uses the same AWS Builder ID or imported refresh-token flow as Kiro, but keeps Amazon Q connections separate. | +| `antigravity` | — | Antigravity | OAuth | — | — | +| `claude` | `cc` | Claude Code | OAuth | — | — | +| `cline` | `cl` | Cline | OAuth | — | — | +| `codebuddy-cn` | `cbcn` | CodeBuddy CN | OAuth | [link](https://copilot.tencent.com) | Tencent CodeBuddy CN (copilot.tencent.com). Sign in via the official CLI device-code flow, or paste a direct API key (sent as Authorization: Bearer). Catalog: GLM / Kimi / MiniMax / DeepSeek / Hunyuan. | +| `codex` | `cx` | OpenAI Codex | OAuth | — | — | +| `cursor` | `cu` | Cursor IDE | OAuth | — | — | +| `devin-cli` | `dv` | Devin CLI (Official) | OAuth | [link](https://cli.devin.ai) | Requires the Devin CLI binary. Run `devin auth login` to authenticate, or provide your WINDSURF_API_KEY. Install: https://cli.devin.ai | +| `github` | `gh` | GitHub Copilot | OAuth | — | — | +| `gitlab-duo` | `gitlab-duo` | GitLab Duo | OAuth | [link](https://docs.gitlab.com/user/duo_agent_platform/code_suggestions/) | OAuth application with ai_features + read_user scopes. Configure GITLAB_DUO_OAUTH_CLIENT_ID and optionally GITLAB_DUO_OAUTH_CLIENT_SECRET on this OmniRoute instance. | +| `grok-cli` | `gc` | Grok Build | OAuth | — | Paste your ~/.grok/auth.json (or the JWT access token) from the Grok Build CLI; refresh_token is rotated automatically. | +| `kilocode` | `kc` | Kilo Code | OAuth | — | — | +| `kimi-coding` | `kmc` | Kimi Coding | OAuth | — | — | +| `kiro` | `kr` | Kiro AI | OAuth | — | Free tier: 50 credits/month (~25K–100K tokens). ⚠️ Kiro ToS prohibits third-party proxy/harness use. | +| `qoder` | `if` | Qoder | OAuth | — | — | +| `qwen` | `qw` | Qwen Code | OAuth | — | ⚠️ **DEPRECATED.** Qwen OAuth free tier was discontinued on 2026-04-15. Use 'bailian-coding-plan', 'alibaba', 'alibaba-cn', or 'openrouter' provider with API key instead. | +| `trae` | `tr` | Trae | OAuth | [link](https://trae.ai) | Trae is an AI-native IDE by ByteDance (SOLO remote agent). Authorize via trae.ai in the popup, or sign in at solo.trae.ai and paste the Cloud-IDE-JWT (sent as 'Authorization: Cloud-IDE-JWT ', ~14-day lifetime) as the access token; web_id/biz_user_id/user_unique_id/scope/tenant/region propagate via providerSpecificData. No headless refresh for pasted tokens — re-paste on expiry. | +| `windsurf` | `ws` | Windsurf (Devin CLI) | OAuth | [link](https://windsurf.com) | In the Windsurf / VS Code IDE, open the command palette and run `Windsurf: Provide Auth Token` (or click the Jupyter "Get Windsurf Authentication Token" button), then copy the shown token and paste it here. Note: opening windsurf.com/show-auth-token directly only renders a "Redirecting" page — the IDE must initiate the flow (it adds a `?state=...` param) for the token to appear. | +| `zed` | `zd` | Zed IDE | OAuth | [link](https://zed.dev) | Zed stores LLM provider credentials (OpenAI, Anthropic, Google, Mistral, xAI) in the OS keychain. Use the Import button below to discover and import them automatically. | ## Web Cookie Providers (23) -| ID | Alias | Name | Tags | Website | Notes | -| ------------------ | ------------- | ------------------------------- | ---------- | ----------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `adapta-web` | `adp-web` | Adapta.org (Adapta One Web) | Web cookie | [link](https://agent.adapta.one) | Paste your __client cookie value from .clerk.agent.adapta.one (DevTools → Application → Cookies) | -| `blackbox-web` | `bb-web` | Blackbox Web (Subscription) | Web cookie | [link](https://app.blackbox.ai) | Paste your __Secure-authjs.session-token value or full cookie header from app.blackbox.ai | -| `chatgpt-web` | `cgpt-web` | ChatGPT Web (Plus/Pro) | Web cookie | [link](https://chatgpt.com) | Paste your __Secure-next-auth.session-token cookie value from chatgpt.com | -| `claude-web` | `cw` | Claude Web | Web cookie | [link](https://claude.ai) | Paste your session cookie from claude.ai | -| `copilot-m365-web` | `m365copilot` | Microsoft 365 Copilot (BizChat) | Web cookie | [link](https://m365.cloud.microsoft/chat) | Paste the access_token and account-specific Chathub path from the Microsoft 365 Copilot WebSocket URL. | -| `copilot-web` | `copilot` | Microsoft Copilot Web | Web cookie | [link](https://copilot.microsoft.com) | Paste your access_token from copilot.microsoft.com (or export a .har file from DevTools while logged in) | -| `deepseek-web` | `ds-web` | DeepSeek Web | Web cookie | [link](https://chat.deepseek.com) | Paste your userToken from chat.deepseek.com — DevTools → Application → Local Storage → userToken | -| `doubao-web` | `db` | Doubao Web (ByteDance) | Web cookie | [link](https://www.doubao.com) | Paste your session cookie from doubao.com (DevTools → Application → Cookies) | -| `gemini-business` | `gembiz` | Gemini Business (Enterprise) | Web cookie | [link](https://business.gemini.google) | From your enterprise account: open business.gemini.google/home/cid/{your-cid}, then copy __Secure-1PSID and __Secure-1PSIDTS cookies from DevTools → Application → Cookies. Paste as a cookie header below. | -| `gemini-web` | `gweb` | Gemini Web (Free) | Web cookie | [link](https://gemini.google.com) | Paste your __Secure-1PSID cookie value from gemini.google.com. Optionally add __Secure-1PSIDTS separated by semicolon. | -| `grok-web` | `gw` | Grok Web (Subscription) | Web cookie | [link](https://grok.com) | Paste the full grok.com cookie line from DevTools → Application → Cookies. Include both `sso` and `sso-rw` (e.g. `sso=...; sso-rw=...`) — Grok's anti-bot rejects `sso` on its own. | -| `huggingchat` | `huggingchat` | HuggingChat (Free) | Web cookie | [link](https://huggingface.co/chat) | Paste your hf-chat cookie value from huggingface.co/chat (DevTools → Application → Cookies → hf-chat). Optional — works without auth for basic use. | -| `inner-ai` | `in-ai` | Inner.ai (Subscription) | Web cookie | [link](https://app.innerai.com) | Paste your token cookie and email separated by a space: open DevTools → Application → Cookies → .innerai.com, copy the token value, then append a space and your Inner.ai login email. Example: eyJhbG... user@example.com | -| `kimi-web` | `kimi-web` | Kimi Web (Moonshot AI) | Web cookie | [link](https://kimi.moonshot.cn) | Paste your session cookie from kimi.moonshot.cn (DevTools → Application → Cookies) | -| `lmarena` | `lma` | LMArena (Free) | Web cookie | [link](https://lmarena.ai) | Paste the full Cookie header from lmarena.ai (DevTools → Network → request → Cookie). The session is now split across arena-auth-prod-v1.0, .1, … — copy the whole header. Optional — works with free tier for basic comparisons. | -| `muse-spark-web` | `ms-web` | Muse Spark Web (Meta AI) | Web cookie | [link](https://www.meta.ai) | Paste your ecto_1_sess value or full cookie header from meta.ai | -| `perplexity-web` | `pplx-web` | Perplexity Web (Pro/Max) | Web cookie | [link](https://www.perplexity.ai) | Paste your __Secure-next-auth.session-token cookie value from perplexity.ai | -| `poe-web` | `poe` | Poe Web (Subscription) | Web cookie | [link](https://poe.com) | Paste your p-b cookie value from poe.com (DevTools → Application → Cookies → p-b) | -| `qwen-web` | `qwen-web` | Qwen Web (Free) | Web cookie | [link](https://chat.qwen.ai) | Open chat.qwen.ai, log in, then open DevTools → Application → Local Storage → copy the "token" value (or use tongyi_sso_ticket cookie as Bearer token). | -| `t3-web` | `t3chat` | t3.chat (Pro/Free) | Web cookie | [link](https://t3.chat) | Open t3.chat in your browser, log in, then open DevTools → Application → Local Storage → https://t3.chat. Copy the value of 'convex-session-id'. Also open DevTools → Network, copy the Cookie header from any request. Paste both values here. See provider setup docs for a step-by-step guide. | -| `v0-vercel-web` | `v0` | v0 Vercel Web (Code Gen) | Web cookie | [link](https://v0.dev) | Paste your session cookie from v0.dev (DevTools → Application → Cookies) | -| `venice-web` | `ven` | Venice Web (Privacy) | Web cookie | [link](https://venice.ai) | Paste your session cookie from venice.ai (DevTools → Application → Cookies) | -| `zenmux-free` | `zmf` | ZenMux Free (Web) | Web cookie | [link](https://zenmux.ai) | Login at zenmux.ai, then export all cookies using EditThisCookie or Cookie-Editor and paste the full Cookie header string here. Refresh every ~30 days. | +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `adapta-web` | `adp-web` | Adapta.org (Adapta One Web) | Web cookie | [link](https://agent.adapta.one) | Paste your __client cookie value from .clerk.agent.adapta.one (DevTools → Application → Cookies) | +| `blackbox-web` | `bb-web` | Blackbox Web (Subscription) | Web cookie | [link](https://app.blackbox.ai) | Paste your __Secure-authjs.session-token value or full cookie header from app.blackbox.ai | +| `chatgpt-web` | `cgpt-web` | ChatGPT Web (Plus/Pro) | Web cookie | [link](https://chatgpt.com) | Paste your __Secure-next-auth.session-token cookie value from chatgpt.com | +| `claude-web` | `cw` | Claude Web | Web cookie | [link](https://claude.ai) | Paste your session cookie from claude.ai | +| `copilot-m365-web` | `m365copilot` | Microsoft 365 Copilot (BizChat) | Web cookie | [link](https://m365.cloud.microsoft/chat) | Sign in at m365.cloud.microsoft/chat, then open DevTools → Network → filter 'WS' → click the Chathub WebSocket connection. Copy both the access_token query parameter AND the account-specific Chathub path segment from its request URL (wss://…/Chathub/?…&access_token=…). It is NOT an Authorization: Bearer header on an XHR/Fetch request. The token is short-lived; this is an unofficial integration. | +| `copilot-web` | `copilot` | Microsoft Copilot Web | Web cookie | [link](https://copilot.microsoft.com) | Paste your access_token from copilot.microsoft.com (or export a .har file from DevTools while logged in) | +| `deepseek-web` | `ds-web` | DeepSeek Web | Web cookie | [link](https://chat.deepseek.com) | Paste your userToken from chat.deepseek.com — DevTools → Application → Local Storage → userToken | +| `doubao-web` | `db` | Doubao Web (ByteDance) | Web cookie | [link](https://www.doubao.com) | Paste your session cookie from doubao.com (DevTools → Application → Cookies) | +| `gemini-business` | `gembiz` | Gemini Business (Enterprise) | Web cookie | [link](https://business.gemini.google) | From your enterprise account: open business.gemini.google/home/cid/{your-cid}, then copy __Secure-1PSID and __Secure-1PSIDTS cookies from DevTools → Application → Cookies. Paste as a cookie header below. | +| `gemini-web` | `gweb` | Gemini Web (Free) | Web cookie | [link](https://gemini.google.com) | Paste your __Secure-1PSID cookie value from gemini.google.com. Optionally add __Secure-1PSIDTS separated by semicolon. | +| `grok-web` | `gw` | Grok Web (Subscription) | Web cookie | [link](https://grok.com) | Paste the full grok.com cookie line from DevTools → Application → Cookies. Include both `sso` and `sso-rw` (e.g. `sso=...; sso-rw=...`) — Grok's anti-bot rejects `sso` on its own. | +| `huggingchat` | `huggingchat` | HuggingChat (Free) | Web cookie | [link](https://huggingface.co/chat) | Paste the full Cookie header from huggingface.co/chat (DevTools → Network → /chat/conversation → Request Headers → Cookie). It should include hf-chat and may also include token / aws-waf-token. | +| `inner-ai` | `in-ai` | Inner.ai (Subscription) | Web cookie | [link](https://app.innerai.com) | Paste your token cookie and email separated by a space: open DevTools → Application → Cookies → .innerai.com, copy the token value, then append a space and your Inner.ai login email. Example: eyJhbG... user@example.com | +| `kimi-web` | `kimi-web` | Kimi Web (Moonshot AI) | Web cookie | [link](https://www.kimi.com) | Paste your Cookie header from www.kimi.com (must contain kimi-auth=...). Find it via DevTools → Network → request → Cookie. | +| `lmarena` | `lma` | LMArena (Free) | Web cookie | [link](https://lmarena.ai) | Paste the full Cookie header from lmarena.ai (DevTools → Network → request → Cookie). The session is now split across arena-auth-prod-v1.0, .1, … — copy the whole header. Optional — works with free tier for basic comparisons. | +| `muse-spark-web` | `ms-web` | Muse Spark Web (Meta AI) | Web cookie | [link](https://www.meta.ai) | Paste your ecto_1_sess value or full cookie header from meta.ai | +| `perplexity-web` | `pplx-web` | Perplexity Web (Pro/Max) | Web cookie | [link](https://www.perplexity.ai) | Paste your __Secure-next-auth.session-token cookie value from perplexity.ai | +| `poe-web` | `poe` | Poe Web (Subscription) | Web cookie | [link](https://poe.com) | Paste your p-b cookie value from poe.com (DevTools → Application → Cookies → p-b) | +| `qwen-web` | `qwen-web` | Qwen Web (Free) | Web cookie | [link](https://chat.qwen.ai) | Open chat.qwen.ai, log in, then open DevTools → Application → Local Storage → copy the "token" value (or use tongyi_sso_ticket cookie as Bearer token). | +| `t3-web` | `t3chat` | t3.chat (Pro/Free) | Web cookie | [link](https://t3.chat) | Open t3.chat in your browser, log in, then open DevTools → Application → Local Storage → https://t3.chat. Copy the value of 'convex-session-id'. Also open DevTools → Network, copy the Cookie header from any request. Paste both values here. See provider setup docs for a step-by-step guide. | +| `v0-vercel-web` | `v0` | v0 Vercel Web (Code Gen) | Web cookie | [link](https://v0.dev) | Paste your session cookie from v0.dev (DevTools → Application → Cookies) | +| `venice-web` | `ven` | Venice Web (Privacy) | Web cookie | [link](https://venice.ai) | Paste your session cookie from venice.ai (DevTools → Application → Cookies) | +| `zenmux-free` | `zmf` | ZenMux Free (Web) | Web cookie | [link](https://zenmux.ai) | Login at zenmux.ai, then export all cookies using EditThisCookie or Cookie-Editor and paste the full Cookie header string here. Refresh every ~30 days. | ## API Key Providers (paid / paid-with-free-credits) (158) -| ID | Alias | Name | Tags | Website | Notes | -| --------------------- | -------------- | ------------------------------- | --------------------- | -------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `360ai` | `360ai` | 360 AI | API key | [link](https://ai.360.cn) | Get API key at ai.360.cn | -| `agentrouter` | `agentrouter` | AgentRouter | API key, aggregator | [link](https://agentrouter.org) | $200 free credits on signup - multi-model routing gateway | -| `ai21` | `ai21` | AI21 Labs | API key | [link](https://www.ai21.com) | $10 trial credits on signup (valid 3 months), no credit card required | -| `aimlapi` | `aiml` | AI/ML API | API key, aggregator | [link](https://aimlapi.com) | Free tier paused (2026) — AI/ML API is now pay-as-you-go only (min $20 top-up); no recurring free credits. | -| `alibaba` | `ali` | Alibaba | API key | [link](https://dashscope-intl.aliyuncs.com) | — | -| `alibaba-cn` | `ali-cn` | Alibaba (China) | API key | [link](https://dashscope.aliyuncs.com) | — | -| `anthropic` | `anthropic` | Anthropic | API key | [link](https://platform.claude.com) | — | -| `api-airforce` | `af` | Api.airforce | API key | [link](https://api.airforce) | 55 free tier models including Grok-3, Claude 3.7, Qwen3, Kimi-K2, Gemini 2.5 Flash, DeepSeek-V3 | -| `arcee-ai` | `arcee` | Arcee AI | API key | [link](https://arcee.ai) | Get API key at arcee.ai | -| `azure-ai` | `azure-ai` | Azure AI Foundry | API key, enterprise | [link](https://learn.microsoft.com/azure/ai-foundry) | Use your Azure AI Foundry key. Base URL can be https://.services.ai.azure.com/openai/v1/ or https://.openai.azure.com/openai/v1/. | -| `azure-openai` | `azure` | Azure OpenAI | API key, enterprise | [link](https://azure.microsoft.com/products/ai-services/openai-service) | Use your Azure OpenAI API key. Base URL should be your resource endpoint, for example https://my-resource.openai.azure.com. | -| `baichuan` | `baichuan` | Baichuan | API key | [link](https://baichuan.com) | Get API key at platform.baichuan-ai.com | -| `baidu` | `baidu` | Baidu (ERNIE) | API key | [link](https://yiyan.baidu.com) | Get API key at console.bce.baidu.com | -| `bailian-coding-plan` | `bcp` | Alibaba Coding Plan | API key | [link](https://www.alibabacloud.com/help/en/model-studio/coding-plan) | — | -| `baseten` | `baseten` | Baseten | API key | [link](https://baseten.co) | $30 free trial credits for GPU inference | -| `bazaarlink` | `bzl` | BazaarLink | API key | [link](https://bazaarlink.ai) | Use your BazaarLink API key (starts with sk-bl-) in Authorization: Bearer . OpenAI SDK works with base URL https://bazaarlink.ai/api/v1. Models use provider/model-name format. | -| `bedrock` | `bedrock` | Amazon Bedrock | API key, enterprise | [link](https://aws.amazon.com/bedrock) | Use your Amazon Bedrock API key and configure the AWS region where your models are enabled (for example eu-west-2). OmniRoute calls Bedrock's native Converse API directly. | -| `black-forest-labs` | `bfl` | Black Forest Labs | API key, image | [link](https://blackforestlabs.ai) | — | -| `blackbox` | `bb` | Blackbox AI | API key | [link](https://blackbox.ai) | Free tier: unlimited basic chat plus Minimax-M2.5, no credit card required | -| `bluesminds` | `bm` | BluesMinds | API key | [link](https://www.bluesminds.com) | Free daily pi credits — supports 200+ models including GPT-4o, GPT-4.1, Claude Sonnet 4.5, Gemini 2.0 Flash, DeepSeek V4, Qwen, Kimi K2 | -| `byteplus` | `bpm` | BytePlus ModelArk | API key | [link](https://console.byteplus.com/ark) | — | -| `bytez` | `bytez` | Bytez | API key | [link](https://bytez.com) | $1 free credits, refreshes every 4 weeks | -| `cablyai` | `cablyai` | CablyAI | API key, aggregator | [link](https://cablyai.com) | Bearer API key for the CablyAI OpenAI-compatible gateway. | -| `cerebras` | `cerebras` | Cerebras | API key | [link](https://inference.cerebras.ai) | Free Trial: 1M tokens/day, 30K TPM, 5 RPM — no credit card. | -| `chutes` | `chutes` | Chutes.ai | API key, aggregator | [link](https://chutes.ai) | Bearer API key for the Chutes OpenAI-compatible gateway. | -| `clarifai` | `clarifai` | Clarifai | API key, enterprise | [link](https://docs.clarifai.com) | Use your Clarifai PAT or app-specific API key. OmniRoute targets the OpenAI-compatible endpoint at https://api.clarifai.com/v2/ext/openai/v1 and authenticates with Authorization: Key . | -| `cloudflare-ai` | `cf` | Cloudflare Workers AI | API key | [link](https://developers.cloudflare.com/workers-ai) | Requires API Token AND Account ID (found at dash.cloudflare.com) | -| `codestral` | `codestral` | Codestral | API key | [link](https://mistral.ai) | — | -| `cohere` | `cohere` | Cohere | API key | [link](https://cohere.com) | Free Trial: 1,000 API calls/month for testing, no credit card required | -| `command-code` | `cmd` | Command Code | API key | [link](https://commandcode.ai/) | Use a Command Code API key. Requests are sent to Command Code's /alpha/generate endpoint. | -| `coze` | `coze` | Coze | API key | [link](https://coze.com) | Get API key at coze.com/open/api | -| `crof` | `crof` | CrofAI | API key | [link](https://crof.ai) | — | -| `databricks` | `databricks` | Databricks | API key, enterprise | [link](https://www.databricks.com) | — | -| `datarobot` | `datarobot` | DataRobot | API key, enterprise | [link](https://docs.datarobot.com) | Use your DataRobot API token. Optional Base URL can be the account root (for LLM Gateway) or a deployment URL under /api/v2/deployments/. | -| `deepinfra` | `deepinfra` | DeepInfra | API key | [link](https://deepinfra.com) | Free signup credits for API testing and model exploration | -| `deepseek` | `ds` | DeepSeek | API key | [link](https://platform.deepseek.com) | 5M free tokens on signup - no credit card required | -| `dgrid` | `dgrid` | DGrid | API key | [link](https://dgrid.ai) | DGrid Free Models Router: 10 requests/minute and 100 requests/day. A $5 lifetime top-up unlocks up to 20 requests/minute and 1,000 requests/day. | -| `dify` | `dify` | Dify | API key | [link](https://dify.ai) | Get API key from your Dify instance. | -| `dit` | `dai` | DIT.ai | API key | [link](https://dit.ai) | Use your dit.ai API key in Authorization: Bearer . Fully OpenAI-compatible — a drop-in replacement, just change the base URL to https://api.dit.ai/v1. | -| `doubao` | `doubao` | Doubao | API key | [link](https://doubao.com) | Get API key at console.volcengine.com | -| `empower` | `empower` | Empower | API key, aggregator | [link](https://docs.empower.dev) | Bearer API key for the Empower OpenAI-compatible endpoint. | -| `factory` | `factory` | Factory | API key | [link](https://factory.ai) | Bearer API key for the Factory OpenAI-compatible gateway. | -| `fal-ai` | `fal` | Fal.ai | API key, image | [link](https://fal.ai) | — | -| `featherless-ai` | `featherless` | Featherless AI | API key | [link](https://featherless.ai) | Free tier available — no credit card required | -| `fenayai` | `fenayai` | FenayAI | API key, aggregator | [link](https://fenayai.com) | Bearer API key for the FenayAI OpenAI-compatible gateway. | -| `firecrawl` | `fc` | Firecrawl | API key | [link](https://firecrawl.dev) | — | -| `fireworks` | `fireworks` | Fireworks AI | API key | [link](https://fireworks.ai) | $1 free starter credits on signup for API testing | -| `freeaiapikey` | `faik` | FreeAIAPIKey | API key | [link](https://freeaiapikey.com) | — | -| `freemodel-dev` | `fmd` | FreeModel.dev | API key | [link](https://freemodel.dev) | $300 free credits on signup — no credit card required. Access GPT-5.4 and GPT-5.5 (OpenAI's latest flagship models) through an OpenAI-compatible API. | -| `friendliai` | `friendli` | FriendliAI | API key | [link](https://friendli.ai) | Free tier for serverless inference — no credit card required | -| `galadriel` | `galadriel` | Galadriel | API key | [link](https://galadriel.com) | ⚠️ **DEPRECATED.** api.galadriel.ai no longer resolves (sweep 2026-06-19); the inference API appears discontinued. | -| `gemini` | `gemini` | Gemini (Google AI Studio) | API key | [link](https://aistudio.google.com) | Free forever: 1,500 req/day for Gemini 2.5 Flash — no credit card, get key at aistudio.google.com | -| `getgoapi` | `ggo` | GoAPI | API key, aggregator | [link](https://api.getgoapi.com) | — | -| `gigachat` | `gigachat` | GigaChat (Sber) | API key | [link](https://developers.sber.ru) | — | -| `github-models` | `ghm` | GitHub Models | API key | [link](https://github.com/marketplace/models) | Create a GitHub PAT with 'models: read' scope at github.com/settings/tokens | -| `gitlab` | `gitlab` | GitLab Duo PAT | API key | [link](https://docs.gitlab.com/user/duo_agent_platform/code_suggestions/) | GitLab personal access token for the public Code Suggestions API. Configure a self-hosted base URL when not using gitlab.com. | -| `gitlawb` | `glb` | Gitlawb Opengateway (MiMo) | API key | [link](https://opengateway.gitlawb.com) | Free MiMo (xiaomi/mimo-v2.5) revoked 2026-05 — Opengateway is now a pay-as-you-go credit gateway; no recurring free model. | -| `gitlawb-gmi` | `glb-gmi` | Gitlawb Opengateway (GMI Cloud) | API key | [link](https://opengateway.gitlawb.com) | Free Nemotron promo ended 2026-06 — the GMI Cloud route is now pay-as-you-go credit only. | -| `glhf` | `glhf` | GLHF Chat | API key, aggregator | [link](https://glhf.chat) | ⚠️ **DEPRECATED.** glhf.chat shut down (2026); its api.laf.run gateway no longer serves the catalog (sweep 2026-06-19). | -| `glm` | `glm` | GLM Coding | API key | [link](https://z.ai/subscribe) | — | -| `glm-cn` | `glmcn` | GLM Coding (China) | API key | [link](https://open.bigmodel.cn) | — | -| `glmt` | `glmt` | GLM Thinking | API key | [link](https://open.bigmodel.cn) | — | -| `groq` | `groq` | Groq | API key | [link](https://groq.com) | Free tier: 30 RPM / 14.4K RPD — no credit card | -| `hackclub` | `hc` | Hackclub AI | API key, aggregator | [link](https://ai.hackclub.com) | Sign in with your Hack Club account at ai.hackclub.com. | -| `haiper` | `hp` | Haiper | API key, video | [link](https://haiper.ai) | Get API key at haiper.ai/haiper-api | -| `heroku` | `heroku` | Heroku AI | API key, enterprise | [link](https://www.heroku.com) | — | -| `huggingface` | `hf` | HuggingFace | API key | [link](https://huggingface.co) | Free Inference API for thousands of models (Whisper, VITS, SDXL…) | -| `hyperbolic` | `hyp` | Hyperbolic | API key | [link](https://hyperbolic.xyz) | $1-5 trial credits on signup for serverless inference | -| `ideogram` | `ideo` | Ideogram | API key | [link](https://ideogram.ai) | Get API key at ideogram.ai/docs/api | -| `iflytek` | `iflytek` | iFlytek Spark | API key | [link](https://xinghuo.xfyun.cn) | Get API key at console.xfyun.cn | -| `inclusionai` | `inclusion` | InclusionAI | API key | [link](https://inclusionai.com) | ⚠️ **DEPRECATED.** api.inclusionai.tech no longer resolves (sweep 2026-06-19); the inference API appears discontinued. | -| `inference-net` | `inet` | Inference.net | API key | [link](https://inference.net) | $25 free credits on signup plus research grants available | -| `jina-ai` | `jina` | Jina AI | API key, embed/rerank | [link](https://jina.ai) | Bearer API key for the Jina AI rerank API. | -| `jina-reader` | `jr` | Jina Reader | API key | [link](https://jina.ai/reader) | — | -| `kie` | `kie` | KIE.AI | API key | [link](https://kie.ai) | — | -| `kilo-gateway` | `kg` | Kilo Gateway | API key, aggregator | [link](https://kilo.ai) | — | -| `kimi` | `kimi` | Kimi | API key | [link](https://platform.moonshot.ai) | — | -| `kimi-coding-apikey` | `kmca` | Kimi Coding (API Key) | API key | [link](https://www.kimi.com/code) | — | -| `kluster` | `kluster` | Kluster AI | API key | [link](https://kluster.ai) | ⚠️ **DEPRECATED.** kluster.ai shut down (2026-06-09); api.kluster.ai no longer resolves (sweep 2026-06-19). Use another OpenAI-compatible provider. | -| `lambda-ai` | `lambda` | Lambda AI | API key | [link](https://lambda.ai) | — | -| `laozhang` | `lz` | LaoZhang AI | API key, aggregator | [link](https://api.laozhang.ai) | — | -| `leonardo` | `leo` | Leonardo AI | API key, video | [link](https://leonardo.ai) | Get API key at leonardo.ai/developer | -| `liquid` | `liquid` | Liquid AI | API key | [link](https://liquid.ai) | Get API key at liquid.ai | -| `llamagate` | `llamagate` | LlamaGate | API key | [link](https://llamagate.ai) | — | -| `llm7` | `llm7` | LLM7.io | API key | [link](https://llm7.io) | No signup required - 2 req/s, 20 RPM, 100 req/hr free tier | -| `longcat` | `lc` | LongCat AI | API key | [link](https://longcat.chat/platform/docs) | Free: one-time 10M-token grant after account signup + KYC verification (LongCat-2.0). One-time only — not a recurring daily/monthly allowance. | -| `maritalk` | `maritalk` | Maritalk | API key | [link](https://www.maritaca.ai) | — | -| `meta-llama` | `meta` | Meta Llama API | API key | [link](https://llama.developer.meta.com) | — | -| `minimax` | `minimax` | Minimax Coding | API key, video | [link](https://www.minimax.io) | — | -| `minimax-cn` | `minimax-cn` | Minimax (China) | API key | [link](https://www.minimaxi.com) | — | -| `mistral` | `mistral` | Mistral | API key | [link](https://mistral.ai) | Free Experiment tier: rate-limited access to all models, no credit card required | -| `modal` | `mdl` | Modal | API key, enterprise | [link](https://modal.com/docs) | Use the bearer token that protects your Modal deployment, if enabled. Base URL should point to your OpenAI-compatible Modal app, for example https://--.modal.run/v1. | -| `monsterapi` | `monster` | MonsterAPI | API key | [link](https://monsterapi.ai) | Get API key at monsterapi.ai | -| `moonshot` | `moonshot` | Moonshot AI | API key | [link](https://platform.moonshot.ai) | — | -| `morph` | `morph` | Morph | API key | [link](https://morphllm.com) | Free tier: 250K credits/month, $0 | -| `nanogpt` | `nanogpt` | NanoGPT | API key | [link](https://nano-gpt.com) | — | -| `nebius` | `nebius` | Nebius AI | API key | [link](https://nebius.com) | ~$1 trial credits on signup for API testing | -| `nlpcloud` | `nlpc` | NLP Cloud | API key | [link](https://docs.nlpcloud.com) | Use your NLP Cloud API key in Authorization: Token . OmniRoute targets the chatbot endpoint on https://api.nlpcloud.io/v1/gpu//chatbot by default. | -| `nomic` | `nomic` | Nomic | API key | [link](https://nomic.ai) | Get API key at atlas.nomic.ai | -| `nous-research` | `nous` | Nous Research | API key | [link](https://portal.nousresearch.com/help) | Use your Nous Portal API key. OmniRoute targets the official OpenAI-compatible inference endpoint at https://inference-api.nousresearch.com/v1. | -| `novita` | `novita` | Novita AI | API key, aggregator | [link](https://novita.ai) | $0.50 trial credits on signup (valid about 1 year) | -| `nscale` | `nscale` | nScale | API key | [link](https://nscale.com) | $5 free credits on signup for inference testing | -| `nvidia` | `nvidia` | NVIDIA NIM | API key | [link](https://build.nvidia.com) | Free dev access: ~40 RPM, 70+ models (Kimi K2.5, GLM 4.7, DeepSeek V3.2...) | -| `oci` | `oci` | OCI Generative AI | API key, enterprise | [link](https://www.oracle.com/artificial-intelligence/generative-ai) | Use your OCI Generative AI API key or IAM bearer token. Base URL can be https://inference.generativeai..oci.oraclecloud.com/openai/v1/. | -| `ollama-cloud` | `ollamacloud` | Ollama Cloud | API key | [link](https://ollama.com/settings/api-keys) | — | -| `openadapter` | `oad` | OpenAdapter | API key | [link](https://openadapter.dev) | Use your OpenAdapter API key in Authorization: Bearer sk-cv-. Fully OpenAI-compatible. API base URL: https://api.openadapter.in/v1. | -| `openai` | `openai` | OpenAI | API key | [link](https://platform.openai.com) | — | -| `opencode-go` | `opencode-go` | OpenCode Go | API key | [link](https://opencode.ai/go) | — | -| `opencode-zen` | `opencode-zen` | OpenCode Zen | API key | [link](https://opencode.ai/zen) | — | -| `openrouter` | `openrouter` | OpenRouter | API key, aggregator | [link](https://openrouter.ai) | Free models at $0/token with :free suffix - 20 RPM / 200 RPD | -| `orcarouter` | `orcarouter` | OrcaRouter | API key | [link](https://www.orcarouter.ai) | — | -| `ovhcloud` | `ovh` | OVHcloud AI | API key | [link](https://www.ovhcloud.com) | — | -| `perplexity` | `pplx` | Perplexity | API key | [link](https://www.perplexity.ai) | — | -| `piapi` | `pi` | PiAPI | API key, aggregator | [link](https://piapi.ai) | — | -| `pioneer` | `pn` | Pioneer AI | API key | [link](https://pioneer.ai) | $75 free usage credits — no credit card required | -| `poe` | `poe` | Poe | API key, aggregator | [link](https://creator.poe.com/api-reference) | Bearer API key for the Poe OpenAI-compatible API. | -| `pollinations` | `pol` | Pollinations AI | API key, video | [link](https://pollinations.ai) | Free keyless tier: openai, openai-fast, openai-large, qwen-coder, mistral, deepseek, grok, gemini-flash-lite-3.1, perplexity-fast, perplexity-reasoning. Premium models (claude, gemini, midijourney) require a Pollinations API key from enter.pollinations.ai. | -| `predibase` | `predibase` | Predibase | API key | [link](https://predibase.com) | ⚠️ **DEPRECATED.** serving.app.predibase.com no longer resolves (sweep 2026-06-19); the managed serving API appears discontinued. | -| `publicai` | `publicai` | PublicAI | API key | [link](https://publicai.co) | Requires an API key — one-time signup credit, then paid | -| `puter` | `pu` | Puter AI | API key | [link](https://puter.com) | Get token at puter.com/dashboard → Copy Auth Token | -| `qianfan` | `qianfan` | Baidu Qianfan | API key | [link](https://cloud.baidu.com/product/wenxinworkshop) | — | -| `recraft` | `recraft` | Recraft | API key, image | [link](https://recraft.ai) | — | -| `reka` | `reka` | Reka | API key | [link](https://docs.reka.ai/chat/overview) | Use your Reka API key. OmniRoute supports the OpenAI-compatible base URL https://api.reka.ai/v1 and sends both Authorization and X-Api-Key headers for compatibility. | -| `runwayml` | `runway` | Runway | API key, video | [link](https://docs.dev.runwayml.com) | Use your Runway API key in Authorization: Bearer . OmniRoute targets the current Runway API at https://api.dev.runwayml.com/v1 and sends the required X-Runway-Version header automatically. | -| `sambanova` | `samba` | SambaNova | API key | [link](https://sambanova.ai) | $5 free credits on signup (30-day validity), no credit card required | -| `sap` | `sap` | SAP Generative AI Hub | API key, enterprise | [link](https://help.sap.com/docs/sap-ai-core/sap-ai-core-service-guide/generative-ai-hub-in-sap-ai-core) | Use your SAP AI Core bearer token. Base URL can be your AI_API_URL root or a deploymentUrl from Generative AI Hub. | -| `scaleway` | `scw` | Scaleway AI | API key | [link](https://www.scaleway.com/en/ai/generative-apis) | 1M free tokens for new accounts — EU/GDPR compliant (Paris), Qwen3 235B & Llama 70B | -| `sensenova` | `sensenova` | SenseNova | API key | [link](https://platform.sensenova.cn) | Get API key at platform.sensenova.cn | -| `siliconflow` | `siliconflow` | SiliconFlow | API key | [link](https://cloud.siliconflow.com) | $1 free credits plus permanently free models after identity verification | -| `snowflake` | `snowflake` | Snowflake Cortex | API key, enterprise | [link](https://www.snowflake.com) | — | -| `sparkdesk` | `sparkdesk` | SparkDesk | API key | [link](https://xinghuo.xfyun.cn) | Get API key at console.xfyun.cn | -| `stability-ai` | `stability` | Stability AI | API key, image | [link](https://stability.ai) | — | -| `stepfun` | `stepfun` | StepFun | API key | [link](https://stepfun.com) | Get API key at platform.stepfun.com | -| `suno` | `suno` | Suno | API key | [link](https://suno.ai) | Paste session cookie from suno.ai (Clerk auth) | -| `synthetic` | `synthetic` | Synthetic | API key, aggregator | [link](https://synthetic.new) | — | -| `tencent` | `tencent` | Tencent Hunyuan | API key | [link](https://hunyuan.tencent.com) | Get API key at console.cloud.tencent.com | -| `thebai` | `thebai` | TheB.AI | API key, aggregator | [link](https://theb.ai) | Bearer API key for the TheB.AI OpenAI-compatible gateway. | -| `together` | `together` | Together AI | API key, video | [link](https://www.together.ai) | $25 signup credits + 3 permanently free models: Llama 3.3 70B, Vision, DeepSeek-R1 distill | -| `tokenrouter` | `trk` | TokenRouter | API key | [link](https://tokenrouter.com) | Use your TokenRouter API key in Authorization: Bearer . Fully OpenAI-compatible. API base URL: https://api.tokenrouter.com/v1. | -| `topaz` | `topaz` | Topaz | API key, image | [link](https://topazlabs.com) | — | -| `udio` | `udio` | Udio | API key | [link](https://udio.com) | Paste session cookie from udio.com (Supabase auth) | -| `uncloseai` | `unc` | UncloseAI | API key | [link](https://uncloseai.com) | No auth required. API accepts any non-empty string as key for identification. | -| `upstage` | `upstage` | Upstage | API key | [link](https://www.upstage.ai) | — | -| `v0-vercel` | `v0` | v0 (Vercel) | API key | [link](https://v0.dev) | — | -| `venice` | `venice` | Venice.ai | API key | [link](https://venice.ai) | — | -| `vercel-ai-gateway` | `vag` | Vercel AI Gateway | API key, aggregator | [link](https://vercel.com/docs/ai-gateway) | — | -| `vertex` | `vertex` | Vertex AI | API key, enterprise | [link](https://cloud.google.com/vertex-ai) | Provide Service Account JSON or OAuth access_token | -| `vertex-partner` | `vp` | Vertex AI Partners | API key, enterprise | [link](https://cloud.google.com/vertex-ai) | Provide the same Service Account JSON used for Vertex AI partner models. | -| `volcengine` | `volcengine` | Volcengine | API key | [link](https://www.volcengine.com) | — | -| `voyage-ai` | `voyage` | Voyage AI | API key, embed/rerank | [link](https://www.voyageai.com) | Bearer API key for Voyage AI embeddings and rerank APIs. | -| `wafer` | `wafer` | Wafer AI | API key | [link](https://wafer.ai) | — | -| `wandb` | `wandb` | Weights & Biases Inference | API key | [link](https://wandb.ai) | — | -| `watsonx` | `watsonx` | IBM watsonx.ai Gateway | API key, enterprise | [link](https://www.ibm.com/products/watsonx-ai) | Use your watsonx bearer token. Base URL can be https://.ml.cloud.ibm.com/ml/gateway/v1/ or a self-managed /ml/gateway/v1 endpoint. | -| `xai` | `xai` | xAI (Grok) | API key | [link](https://x.ai) | — | -| `xiaomi-mimo` | `mimo` | Xiaomi MiMo | API key | [link](https://mimo.mi.com) | — | -| `yi` | `yi` | Yi (01.AI) | API key | [link](https://01.ai) | Get API key at platform.lingyiwanwu.com | -| `zai` | `zai` | Z.AI | API key | [link](https://open.bigmodel.cn) | — | -| `zenmux` | `zm` | ZenMux | API key | [link](https://zenmux.ai) | Use your ZenMux API key in Authorization: Bearer . ZenMux is fully OpenAI-compatible. Base URL: https://zenmux.ai/api/v1. | +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `360ai` | `360ai` | 360 AI | API key | [link](https://ai.360.cn) | Get API key at ai.360.cn | +| `agentrouter` | `agentrouter` | AgentRouter | API key, aggregator | [link](https://agentrouter.org) | $200 free credits on signup - multi-model routing gateway | +| `ai21` | `ai21` | AI21 Labs | API key | [link](https://www.ai21.com) | $10 trial credits on signup (valid 3 months), no credit card required | +| `aimlapi` | `aiml` | AI/ML API | API key, aggregator | [link](https://aimlapi.com) | Free tier paused (2026) — AI/ML API is now pay-as-you-go only (min $20 top-up); no recurring free credits. | +| `alibaba` | `ali` | Alibaba | API key | [link](https://bailian.console.alibabacloud.com/) | — | +| `alibaba-cn` | `ali-cn` | Alibaba (China) | API key | [link](https://dashscope.console.aliyun.com/) | — | +| `anthropic` | `anthropic` | Anthropic | API key | [link](https://platform.claude.com) | — | +| `api-airforce` | `af` | Api.airforce | API key | [link](https://api.airforce) | 55 free tier models including Grok-3, Claude 3.7, Qwen3, Kimi-K2, Gemini 2.5 Flash, DeepSeek-V3 | +| `arcee-ai` | `arcee` | Arcee AI | API key | [link](https://arcee.ai) | Get API key at arcee.ai | +| `azure-ai` | `azure-ai` | Azure AI Foundry | API key, enterprise | [link](https://learn.microsoft.com/azure/ai-foundry) | Use your Azure AI Foundry key. Base URL can be https://.services.ai.azure.com/openai/v1/ or https://.openai.azure.com/openai/v1/. | +| `azure-openai` | `azure` | Azure OpenAI | API key, enterprise | [link](https://azure.microsoft.com/products/ai-services/openai-service) | Use your Azure OpenAI API key. Base URL should be your resource endpoint, for example https://my-resource.openai.azure.com. | +| `baichuan` | `baichuan` | Baichuan | API key | [link](https://baichuan.com) | Get API key at platform.baichuan-ai.com | +| `baidu` | `baidu` | Baidu (ERNIE) | API key | [link](https://yiyan.baidu.com) | Get API key at console.bce.baidu.com | +| `bailian-coding-plan` | `bcp` | Alibaba Coding Plan | API key | [link](https://www.alibabacloud.com/help/en/model-studio/coding-plan) | — | +| `baseten` | `baseten` | Baseten | API key | [link](https://baseten.co) | $30 free trial credits for GPU inference | +| `bazaarlink` | `bzl` | BazaarLink | API key | [link](https://bazaarlink.ai) | Use your BazaarLink API key (starts with sk-bl-) in Authorization: Bearer . OpenAI SDK works with base URL https://bazaarlink.ai/api/v1. Models use provider/model-name format. | +| `bedrock` | `bedrock` | Amazon Bedrock | API key, enterprise | [link](https://aws.amazon.com/bedrock) | Use your Amazon Bedrock API key and configure the AWS region where your models are enabled (for example eu-west-2). OmniRoute calls Bedrock's native Converse API directly. | +| `black-forest-labs` | `bfl` | Black Forest Labs | API key, image | [link](https://blackforestlabs.ai) | — | +| `blackbox` | `bb` | Blackbox AI | API key | [link](https://blackbox.ai) | Free tier: unlimited basic chat plus Minimax-M2.5, no credit card required | +| `bluesminds` | `bm` | BluesMinds | API key | [link](https://www.bluesminds.com) | Free daily pi credits — supports 200+ models including GPT-4o, GPT-4.1, Claude Sonnet 4.5, Gemini 2.0 Flash, DeepSeek V4, Qwen, Kimi K2 | +| `byteplus` | `bpm` | BytePlus ModelArk | API key | [link](https://console.byteplus.com/ark) | — | +| `bytez` | `bytez` | Bytez | API key | [link](https://bytez.com) | $1 free credits, refreshes every 4 weeks | +| `cablyai` | `cablyai` | CablyAI | API key, aggregator | [link](https://cablyai.com) | ⚠️ **DEPRECATED.** cablyai.com no longer resolves (DNS NXDOMAIN, verified 2026-06-30) — the domain is gone and every request fails with a DNS error (#5568). | +| `cerebras` | `cerebras` | Cerebras | API key | [link](https://inference.cerebras.ai) | Free Trial: 1M tokens/day, 30K TPM, 5 RPM — no credit card. | +| `chutes` | `chutes` | Chutes.ai | API key, aggregator | [link](https://chutes.ai) | Bearer API key for the Chutes OpenAI-compatible gateway. | +| `clarifai` | `clarifai` | Clarifai | API key, enterprise | [link](https://docs.clarifai.com) | Use your Clarifai PAT or app-specific API key. OmniRoute targets the OpenAI-compatible endpoint at https://api.clarifai.com/v2/ext/openai/v1 and authenticates with Authorization: Key . | +| `cloudflare-ai` | `cf` | Cloudflare Workers AI | API key | [link](https://developers.cloudflare.com/workers-ai) | Requires API Token AND Account ID (found at dash.cloudflare.com) | +| `codestral` | `codestral` | Codestral | API key | [link](https://mistral.ai) | — | +| `cohere` | `cohere` | Cohere | API key | [link](https://cohere.com) | Free Trial: 1,000 API calls/month for testing, no credit card required | +| `command-code` | `cmd` | Command Code | API key | [link](https://commandcode.ai/) | Use a Command Code API key. Requests are sent to Command Code's /alpha/generate endpoint. | +| `coze` | `coze` | Coze | API key | [link](https://coze.com) | Get API key at coze.com/open/api | +| `crof` | `crof` | CrofAI | API key | [link](https://crof.ai) | — | +| `databricks` | `databricks` | Databricks | API key, enterprise | [link](https://www.databricks.com) | — | +| `datarobot` | `datarobot` | DataRobot | API key, enterprise | [link](https://docs.datarobot.com) | Use your DataRobot API token. Optional Base URL can be the account root (for LLM Gateway) or a deployment URL under /api/v2/deployments/. | +| `deepinfra` | `deepinfra` | DeepInfra | API key | [link](https://deepinfra.com) | Free signup credits for API testing and model exploration | +| `deepseek` | `ds` | DeepSeek | API key | [link](https://platform.deepseek.com) | 5M free tokens on signup - no credit card required | +| `dgrid` | `dgrid` | DGrid | API key | [link](https://dgrid.ai) | DGrid Free Models Router: 10 requests/minute and 100 requests/day. A $5 lifetime top-up unlocks up to 20 requests/minute and 1,000 requests/day. | +| `dify` | `dify` | Dify | API key | [link](https://dify.ai) | Get API key from your Dify instance. | +| `dit` | `dai` | DIT.ai | API key | [link](https://dit.ai) | Use your dit.ai API key in Authorization: Bearer . Fully OpenAI-compatible — a drop-in replacement, just change the base URL to https://api.dit.ai/v1. | +| `doubao` | `doubao` | Doubao | API key | [link](https://doubao.com) | Get API key at console.volcengine.com | +| `empower` | `empower` | Empower | API key, aggregator | [link](https://docs.empower.dev) | Bearer API key for the Empower OpenAI-compatible endpoint. | +| `factory` | `factory` | Factory | API key | [link](https://factory.ai) | Bearer API key for the Factory OpenAI-compatible gateway. | +| `fal-ai` | `fal` | Fal.ai | API key, image | [link](https://fal.ai) | — | +| `featherless-ai` | `featherless` | Featherless AI | API key | [link](https://featherless.ai) | Free tier available — no credit card required | +| `fenayai` | `fenayai` | FenayAI | API key, aggregator | [link](https://fenayai.com) | Bearer API key for the FenayAI OpenAI-compatible gateway. | +| `firecrawl` | `fc` | Firecrawl | API key | [link](https://firecrawl.dev) | — | +| `fireworks` | `fireworks` | Fireworks AI | API key | [link](https://fireworks.ai) | $1 free starter credits on signup for API testing | +| `freeaiapikey` | `faik` | FreeAIAPIKey | API key | [link](https://freeaiapikey.com) | — | +| `freemodel-dev` | `fmd` | FreeModel.dev | API key | [link](https://freemodel.dev) | $300 free credits on signup — no credit card required. Access GPT-5.4 and GPT-5.5 (OpenAI's latest flagship models) through an OpenAI-compatible API. | +| `friendliai` | `friendli` | FriendliAI | API key | [link](https://friendli.ai) | Free tier for serverless inference — no credit card required | +| `galadriel` | `galadriel` | Galadriel | API key | [link](https://galadriel.com) | ⚠️ **DEPRECATED.** api.galadriel.ai no longer resolves (sweep 2026-06-19); the inference API appears discontinued. | +| `gemini` | `gemini` | Gemini (Google AI Studio) | API key | [link](https://aistudio.google.com) | Free forever: 1,500 req/day for Gemini 2.5 Flash — no credit card, get key at aistudio.google.com | +| `getgoapi` | `ggo` | GoAPI | API key, aggregator | [link](https://api.getgoapi.com) | — | +| `gigachat` | `gigachat` | GigaChat (Sber) | API key | [link](https://developers.sber.ru) | — | +| `github-models` | `ghm` | GitHub Models | API key | [link](https://github.com/marketplace/models) | Create a GitHub PAT with 'models: read' scope at github.com/settings/tokens | +| `gitlab` | `gitlab` | GitLab Duo PAT | API key | [link](https://docs.gitlab.com/user/duo_agent_platform/code_suggestions/) | GitLab personal access token for the public Code Suggestions API. Configure a self-hosted base URL when not using gitlab.com. | +| `gitlawb` | `glb` | Gitlawb Opengateway (MiMo) | API key | [link](https://opengateway.gitlawb.com) | Free MiMo (xiaomi/mimo-v2.5) revoked 2026-05 — Opengateway is now a pay-as-you-go credit gateway; no recurring free model. | +| `gitlawb-gmi` | `glb-gmi` | Gitlawb Opengateway (GMI Cloud) | API key | [link](https://opengateway.gitlawb.com) | Free Nemotron promo ended 2026-06 — the GMI Cloud route is now pay-as-you-go credit only. | +| `glhf` | `glhf` | GLHF Chat | API key, aggregator | [link](https://glhf.chat) | ⚠️ **DEPRECATED.** glhf.chat shut down (2026); its api.laf.run gateway no longer serves the catalog (sweep 2026-06-19). | +| `glm` | `glm` | GLM Coding | API key | [link](https://z.ai/subscribe) | — | +| `glm-cn` | `glmcn` | GLM Coding (China) | API key | [link](https://open.bigmodel.cn) | — | +| `glmt` | `glmt` | GLM Thinking | API key | [link](https://open.bigmodel.cn) | — | +| `groq` | `groq` | Groq | API key | [link](https://groq.com) | Free tier: 30 RPM / 14.4K RPD — no credit card | +| `hackclub` | `hc` | Hackclub AI | API key, aggregator | [link](https://ai.hackclub.com) | Sign in with your Hack Club account at ai.hackclub.com. | +| `haiper` | `hp` | Haiper | API key, video | [link](https://haiper.ai) | Get API key at haiper.ai/haiper-api | +| `heroku` | `heroku` | Heroku AI | API key, enterprise | [link](https://www.heroku.com) | — | +| `huggingface` | `hf` | HuggingFace | API key | [link](https://huggingface.co) | Free Inference API for thousands of models (Whisper, VITS, SDXL…) | +| `hyperbolic` | `hyp` | Hyperbolic | API key | [link](https://hyperbolic.xyz) | $1-5 trial credits on signup for serverless inference | +| `ideogram` | `ideo` | Ideogram | API key | [link](https://ideogram.ai) | Get API key at ideogram.ai/docs/api | +| `iflytek` | `iflytek` | iFlytek Spark | API key | [link](https://xinghuo.xfyun.cn) | Get API key at console.xfyun.cn | +| `inclusionai` | `inclusion` | InclusionAI | API key | [link](https://inclusionai.com) | ⚠️ **DEPRECATED.** api.inclusionai.tech no longer resolves (sweep 2026-06-19); the inference API appears discontinued. | +| `inference-net` | `inet` | Inference.net | API key | [link](https://inference.net) | $25 free credits on signup plus research grants available | +| `jina-ai` | `jina` | Jina AI | API key, embed/rerank | [link](https://jina.ai) | Bearer API key for the Jina AI rerank API. | +| `jina-reader` | `jr` | Jina Reader | API key | [link](https://jina.ai/reader) | — | +| `kie` | `kie` | KIE.AI | API key | [link](https://kie.ai) | — | +| `kilo-gateway` | `kg` | Kilo Gateway | API key, aggregator | [link](https://kilo.ai) | — | +| `kimi` | `kimi` | Kimi | API key | [link](https://platform.moonshot.ai) | — | +| `kimi-coding-apikey` | `kmca` | Kimi Coding (API Key) | API key | [link](https://www.kimi.com/code) | — | +| `kluster` | `kluster` | Kluster AI | API key | [link](https://kluster.ai) | ⚠️ **DEPRECATED.** kluster.ai shut down (2026-06-09); api.kluster.ai no longer resolves (sweep 2026-06-19). Use another OpenAI-compatible provider. | +| `lambda-ai` | `lambda` | Lambda AI | API key | [link](https://lambda.ai) | — | +| `laozhang` | `lz` | LaoZhang AI | API key, aggregator | [link](https://api.laozhang.ai) | — | +| `leonardo` | `leo` | Leonardo AI | API key, video | [link](https://leonardo.ai) | Get API key at leonardo.ai/developer | +| `liquid` | `liquid` | Liquid AI | API key | [link](https://liquid.ai) | Get API key at liquid.ai | +| `llamagate` | `llamagate` | LlamaGate | API key | [link](https://llamagate.ai) | — | +| `llm7` | `llm7` | LLM7.io | API key | [link](https://llm7.io) | No signup required - 2 req/s, 20 RPM, 100 req/hr free tier | +| `longcat` | `lc` | LongCat AI | API key | [link](https://longcat.chat/platform/docs) | Free: one-time 10M-token grant after account signup + KYC verification (LongCat-2.0). One-time only — not a recurring daily/monthly allowance. | +| `maritalk` | `maritalk` | Maritalk | API key | [link](https://www.maritaca.ai) | — | +| `meta-llama` | `meta` | Meta Llama API | API key | [link](https://llama.developer.meta.com) | — | +| `minimax` | `minimax` | Minimax Coding | API key, video | [link](https://www.minimax.io) | — | +| `minimax-cn` | `minimax-cn` | Minimax (China) | API key | [link](https://www.minimaxi.com) | — | +| `mistral` | `mistral` | Mistral | API key | [link](https://mistral.ai) | Free Experiment tier: rate-limited access to all models, no credit card required | +| `modal` | `mdl` | Modal | API key, enterprise | [link](https://modal.com/docs) | Use the bearer token that protects your Modal deployment, if enabled. Base URL should point to your OpenAI-compatible Modal app, for example https://--.modal.run/v1. | +| `monsterapi` | `monster` | MonsterAPI | API key | [link](https://monsterapi.ai) | Get API key at monsterapi.ai | +| `moonshot` | `moonshot` | Moonshot AI | API key | [link](https://platform.moonshot.ai) | — | +| `morph` | `morph` | Morph | API key | [link](https://morphllm.com) | Free tier: 250K credits/month, $0 | +| `nanogpt` | `nanogpt` | NanoGPT | API key | [link](https://nano-gpt.com) | — | +| `nebius` | `nebius` | Nebius AI | API key | [link](https://nebius.com) | ~$1 trial credits on signup for API testing | +| `nlpcloud` | `nlpc` | NLP Cloud | API key | [link](https://docs.nlpcloud.com) | Use your NLP Cloud API key in Authorization: Token . OmniRoute targets the chatbot endpoint on https://api.nlpcloud.io/v1/gpu//chatbot by default. | +| `nomic` | `nomic` | Nomic | API key | [link](https://nomic.ai) | Get API key at atlas.nomic.ai | +| `nous-research` | `nous` | Nous Research | API key | [link](https://portal.nousresearch.com/help) | Use your Nous Portal API key. OmniRoute targets the official OpenAI-compatible inference endpoint at https://inference-api.nousresearch.com/v1. | +| `novita` | `novita` | Novita AI | API key, aggregator | [link](https://novita.ai) | $0.50 trial credits on signup (valid about 1 year) | +| `nscale` | `nscale` | nScale | API key | [link](https://nscale.com) | $5 free credits on signup for inference testing | +| `nvidia` | `nvidia` | NVIDIA NIM | API key | [link](https://build.nvidia.com) | Free dev access: ~40 RPM, 70+ models (Kimi K2.5, GLM 4.7, DeepSeek V3.2...) | +| `oci` | `oci` | OCI Generative AI | API key, enterprise | [link](https://www.oracle.com/artificial-intelligence/generative-ai) | Use your OCI Generative AI API key or IAM bearer token. Base URL can be https://inference.generativeai..oci.oraclecloud.com/openai/v1/. | +| `ollama-cloud` | `ollamacloud` | Ollama Cloud | API key | [link](https://ollama.com/settings/keys) | — | +| `openadapter` | `oad` | OpenAdapter | API key | [link](https://openadapter.dev) | Use your OpenAdapter API key in Authorization: Bearer sk-cv-. Fully OpenAI-compatible. API base URL: https://api.openadapter.in/v1. | +| `openai` | `openai` | OpenAI | API key | [link](https://platform.openai.com) | — | +| `opencode-go` | `opencode-go` | OpenCode Go | API key | [link](https://opencode.ai/go) | — | +| `opencode-zen` | `opencode-zen` | OpenCode Zen | API key | [link](https://opencode.ai/zen) | — | +| `openrouter` | `openrouter` | OpenRouter | API key, aggregator | [link](https://openrouter.ai) | Free models at $0/token with :free suffix - 20 RPM / 200 RPD | +| `orcarouter` | `orcarouter` | OrcaRouter | API key | [link](https://www.orcarouter.ai) | — | +| `ovhcloud` | `ovh` | OVHcloud AI | API key | [link](https://www.ovhcloud.com) | — | +| `perplexity` | `pplx` | Perplexity | API key | [link](https://www.perplexity.ai) | — | +| `piapi` | `pi` | PiAPI | API key, aggregator | [link](https://piapi.ai) | — | +| `pioneer` | `pn` | Pioneer AI | API key | [link](https://pioneer.ai) | $75 free usage credits — no credit card required | +| `poe` | `poe` | Poe | API key, aggregator | [link](https://creator.poe.com/api-reference) | Bearer API key for the Poe OpenAI-compatible API. | +| `pollinations` | `pol` | Pollinations AI | API key, video | [link](https://pollinations.ai) | Free keyless tier: openai, openai-fast, openai-large, qwen-coder, mistral, deepseek, grok, gemini-flash-lite-3.1, perplexity-fast, perplexity-reasoning. Premium models (claude, gemini, midijourney) require a Pollinations API key from enter.pollinations.ai. | +| `predibase` | `predibase` | Predibase | API key | [link](https://predibase.com) | ⚠️ **DEPRECATED.** serving.app.predibase.com no longer resolves (sweep 2026-06-19); the managed serving API appears discontinued. | +| `publicai` | `publicai` | PublicAI | API key | [link](https://publicai.co) | Requires an API key — one-time signup credit, then paid | +| `puter` | `pu` | Puter AI | API key | [link](https://puter.com) | Get token at puter.com/dashboard → Copy Auth Token | +| `qianfan` | `qianfan` | Baidu Qianfan | API key | [link](https://cloud.baidu.com/product/wenxinworkshop) | — | +| `recraft` | `recraft` | Recraft | API key, image | [link](https://recraft.ai) | — | +| `reka` | `reka` | Reka | API key | [link](https://docs.reka.ai/chat/overview) | Use your Reka API key. OmniRoute supports the OpenAI-compatible base URL https://api.reka.ai/v1 and sends both Authorization and X-Api-Key headers for compatibility. | +| `runwayml` | `runway` | Runway | API key, video | [link](https://docs.dev.runwayml.com) | Use your Runway API key in Authorization: Bearer . OmniRoute targets the current Runway API at https://api.dev.runwayml.com/v1 and sends the required X-Runway-Version header automatically. | +| `sambanova` | `samba` | SambaNova | API key | [link](https://sambanova.ai) | $5 free credits on signup (30-day validity), no credit card required | +| `sap` | `sap` | SAP Generative AI Hub | API key, enterprise | [link](https://help.sap.com/docs/sap-ai-core/sap-ai-core-service-guide/generative-ai-hub-in-sap-ai-core) | Use your SAP AI Core bearer token. Base URL can be your AI_API_URL root or a deploymentUrl from Generative AI Hub. | +| `scaleway` | `scw` | Scaleway AI | API key | [link](https://www.scaleway.com/en/docs/ai-data/generative-apis/) | 1M free tokens for new accounts — EU/GDPR compliant (Paris), Qwen3 235B & Llama 70B | +| `sensenova` | `sensenova` | SenseNova | API key | [link](https://platform.sensenova.cn) | Get API key at platform.sensenova.cn | +| `siliconflow` | `siliconflow` | SiliconFlow | API key | [link](https://cloud.siliconflow.com) | $1 free credits plus permanently free models after identity verification | +| `snowflake` | `snowflake` | Snowflake Cortex | API key, enterprise | [link](https://www.snowflake.com) | — | +| `sparkdesk` | `sparkdesk` | SparkDesk | API key | [link](https://xinghuo.xfyun.cn) | Get API key at console.xfyun.cn | +| `stability-ai` | `stability` | Stability AI | API key, image | [link](https://stability.ai) | — | +| `stepfun` | `stepfun` | StepFun | API key | [link](https://stepfun.com) | Get API key at platform.stepfun.com | +| `suno` | `suno` | Suno | API key | [link](https://suno.ai) | Paste session cookie from suno.ai (Clerk auth) | +| `synthetic` | `synthetic` | Synthetic | API key, aggregator | [link](https://synthetic.new) | — | +| `tencent` | `tencent` | Tencent Hunyuan | API key | [link](https://hunyuan.tencent.com) | Get API key at console.cloud.tencent.com | +| `thebai` | `thebai` | TheB.AI | API key, aggregator | [link](https://theb.ai) | Bearer API key for the TheB.AI OpenAI-compatible gateway. | +| `together` | `together` | Together AI | API key, video | [link](https://www.together.ai) | — | +| `tokenrouter` | `trk` | TokenRouter | API key | [link](https://tokenrouter.com) | Use your TokenRouter API key in Authorization: Bearer . Fully OpenAI-compatible. API base URL: https://api.tokenrouter.com/v1. | +| `topaz` | `topaz` | Topaz | API key, image | [link](https://topazlabs.com) | — | +| `udio` | `udio` | Udio | API key | [link](https://udio.com) | Paste session cookie from udio.com (Supabase auth) | +| `uncloseai` | `unc` | UncloseAI | API key | [link](https://uncloseai.com) | No auth required. API accepts any non-empty string as key for identification. | +| `upstage` | `upstage` | Upstage | API key | [link](https://www.upstage.ai) | — | +| `v0-vercel` | `v0` | v0 (Vercel) | API key | [link](https://v0.dev) | — | +| `venice` | `venice` | Venice.ai | API key | [link](https://venice.ai) | — | +| `vercel-ai-gateway` | `vag` | Vercel AI Gateway | API key, aggregator | [link](https://vercel.com/docs/ai-gateway) | — | +| `vertex` | `vertex` | Vertex AI | API key, enterprise | [link](https://cloud.google.com/vertex-ai) | Provide Service Account JSON or OAuth access_token | +| `vertex-partner` | `vp` | Vertex AI Partners | API key, enterprise | [link](https://cloud.google.com/vertex-ai) | Provide the same Service Account JSON used for Vertex AI partner models. | +| `volcengine` | `volcengine` | Volcengine | API key | [link](https://www.volcengine.com) | — | +| `voyage-ai` | `voyage` | Voyage AI | API key, embed/rerank | [link](https://www.voyageai.com) | Bearer API key for Voyage AI embeddings and rerank APIs. | +| `wafer` | `wafer` | Wafer AI | API key | [link](https://wafer.ai) | — | +| `wandb` | `wandb` | Weights & Biases Inference | API key | [link](https://wandb.ai) | — | +| `watsonx` | `watsonx` | IBM watsonx.ai Gateway | API key, enterprise | [link](https://www.ibm.com/products/watsonx-ai) | Use your watsonx bearer token. Base URL can be https://.ml.cloud.ibm.com/ml/gateway/v1/ or a self-managed /ml/gateway/v1 endpoint. | +| `xai` | `xai` | xAI (Grok) | API key | [link](https://x.ai) | — | +| `xiaomi-mimo` | `mimo` | Xiaomi MiMo | API key | [link](https://mimo.mi.com) | — | +| `yi` | `yi` | Yi (01.AI) | API key | [link](https://01.ai) | Get API key at platform.lingyiwanwu.com | +| `zai` | `zai` | Z.AI | API key | [link](https://open.bigmodel.cn) | — | +| `zenmux` | `zm` | ZenMux | API key | [link](https://zenmux.ai) | Use your ZenMux API key in Authorization: Bearer . ZenMux is fully OpenAI-compatible. Base URL: https://zenmux.ai/api/v1. | ## Local Providers (12) -| ID | Alias | Name | Tags | Website | Notes | -| --------------------- | ------------ | ------------------- | ------------------ | --------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `comfyui` | `comfyui` | ComfyUI | Local | [link](https://github.com/comfyanonymous/ComfyUI) | No API key required. Configure the local ComfyUI base URL (default: http://localhost:8188). | -| `docker-model-runner` | `dmr` | Docker Model Runner | Local, self-hosted | [link](https://docs.docker.com/ai/model-runner/) | API key optional. Configure the local Docker Model Runner OpenAI-compatible base URL (default: http://localhost:12434/v1). | -| `lemonade` | `lemonade` | Lemonade Server | Local, self-hosted | [link](https://lemonade-server.ai) | API key optional. Configure the local Lemonade OpenAI-compatible base URL (default: http://localhost:13305/api/v1). | -| `llama-cpp` | `llamacpp` | llama.cpp | Local, self-hosted | [link](https://github.com/ggml-org/llama.cpp) | API key optional (use any value, e.g. sk-no-key-required). Configure the llama-server OpenAI-compatible base URL (default: http://127.0.0.1:8080/v1). Note: if Llamafile is also installed, both default to port 8080 — run only one at a time or override the port. | -| `llamafile` | `llamafile` | Llamafile | Local, self-hosted | [link](https://github.com/Mozilla-Ocho/llamafile) | API key optional. Configure the local Llamafile OpenAI-compatible base URL (default: http://127.0.0.1:8080/v1). | -| `lm-studio` | `lmstudio` | LM Studio | Local, self-hosted | [link](https://lmstudio.ai) | API key optional. Configure the local LM Studio OpenAI-compatible base URL (default: http://localhost:1234/v1). | -| `ollama-local` | `ollama` | Ollama | Local, self-hosted | [link](https://ollama.com) | No API key required. Ollama runs locally — configure its OpenAI-compatible base URL (default: http://localhost:11434/v1) and make sure Ollama is running before connecting. | -| `oobabooga` | `ooba` | oobabooga | Local, self-hosted | [link](https://github.com/oobabooga/text-generation-webui) | API key optional. Configure the local oobabooga OpenAI-compatible base URL (default: http://localhost:5000/v1). | -| `sdwebui` | `sdwebui` | SD WebUI | Local | [link](https://github.com/AUTOMATIC1111/stable-diffusion-webui) | No API key required. Configure the local WebUI base URL (default: http://localhost:7860). | -| `triton` | `triton` | NVIDIA Triton | Local, self-hosted | [link](https://developer.nvidia.com/triton-inference-server) | API key optional. Configure the Triton OpenAI-compatible base URL (default: http://localhost:8000/v1). | -| `vllm` | `vllm` | vLLM | Local, self-hosted | [link](https://github.com/vllm-project/vllm) | API key optional. Configure the local vLLM OpenAI-compatible base URL (default: http://localhost:8000/v1). | -| `xinference` | `xinference` | XInference | Local, self-hosted | [link](https://inference.readthedocs.io) | API key optional. Configure the local XInference OpenAI-compatible base URL (default: http://localhost:9997/v1). | +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `comfyui` | `comfyui` | ComfyUI | Local | [link](https://github.com/comfyanonymous/ComfyUI) | No API key required. Configure the local ComfyUI base URL (default: http://localhost:8188). | +| `docker-model-runner` | `dmr` | Docker Model Runner | Local, self-hosted | [link](https://docs.docker.com/ai/model-runner/) | API key optional. Configure the local Docker Model Runner OpenAI-compatible base URL (default: http://localhost:12434/v1). | +| `lemonade` | `lemonade` | Lemonade Server | Local, self-hosted | [link](https://lemonade-server.ai) | API key optional. Configure the local Lemonade OpenAI-compatible base URL (default: http://localhost:13305/api/v1). | +| `llama-cpp` | `llamacpp` | llama.cpp | Local, self-hosted | [link](https://github.com/ggml-org/llama.cpp) | API key optional (use any value, e.g. sk-no-key-required). Configure the llama-server OpenAI-compatible base URL (default: http://127.0.0.1:8080/v1). Note: if Llamafile is also installed, both default to port 8080 — run only one at a time or override the port. | +| `llamafile` | `llamafile` | Llamafile | Local, self-hosted | [link](https://github.com/Mozilla-Ocho/llamafile) | API key optional. Configure the local Llamafile OpenAI-compatible base URL (default: http://127.0.0.1:8080/v1). | +| `lm-studio` | `lmstudio` | LM Studio | Local, self-hosted | [link](https://lmstudio.ai) | API key optional. Configure the local LM Studio OpenAI-compatible base URL (default: http://localhost:1234/v1). | +| `ollama-local` | `ollama` | Ollama | Local, self-hosted | [link](https://ollama.com) | No API key required. Ollama runs locally — configure its OpenAI-compatible base URL (default: http://localhost:11434/v1) and make sure Ollama is running before connecting. | +| `oobabooga` | `ooba` | oobabooga | Local, self-hosted | [link](https://github.com/oobabooga/text-generation-webui) | API key optional. Configure the local oobabooga OpenAI-compatible base URL (default: http://localhost:5000/v1). | +| `sdwebui` | `sdwebui` | SD WebUI | Local | [link](https://github.com/AUTOMATIC1111/stable-diffusion-webui) | No API key required. Configure the local WebUI base URL (default: http://localhost:7860). | +| `triton` | `triton` | NVIDIA Triton | Local, self-hosted | [link](https://developer.nvidia.com/triton-inference-server) | API key optional. Configure the Triton OpenAI-compatible base URL (default: http://localhost:8000/v1). | +| `vllm` | `vllm` | vLLM | Local, self-hosted | [link](https://github.com/vllm-project/vllm) | API key optional. Configure the local vLLM OpenAI-compatible base URL (default: http://localhost:8000/v1). | +| `xinference` | `xinference` | XInference | Local, self-hosted | [link](https://inference.readthedocs.io) | API key optional. Configure the local XInference OpenAI-compatible base URL (default: http://localhost:9997/v1). | ## Search Providers (11) -| ID | Alias | Name | Tags | Website | Notes | -| ------------------- | --------------- | -------------------------- | ------ | --------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------- | -| `brave-search` | `brave-search` | Brave Search | Search | [link](https://brave.com/search/api) | Subscription token from Brave Search API dashboard | -| `exa-search` | `exa-search` | Exa Search | Search | [link](https://exa.ai) | API key from dashboard.exa.ai | -| `google-pse-search` | `google-pse` | Google Programmable Search | Search | [link](https://developers.google.com/custom-search/v1/overview) | Requires a Google API key and your Programmable Search Engine ID (cx) | -| `linkup-search` | `linkup` | Linkup Search | Search | [link](https://docs.linkup.so) | Bearer API key from the Linkup dashboard | -| `ollama-search` | `ollama-search` | Ollama Search | Search | [link](https://ollama.com/settings/api-keys) | Same API key as Ollama Cloud (from ollama.com/settings/api-keys) | -| `perplexity-search` | `pplx-search` | Perplexity Search | Search | [link](https://docs.perplexity.ai/guides/search-quickstart) | Same API key as Perplexity (pplx-...) | -| `searchapi-search` | `searchapi` | SearchAPI | Search | [link](https://www.searchapi.io/docs) | API key from SearchAPI (query param or Bearer auth) | -| `searxng-search` | `searxng` | SearXNG Search | Search | [link](https://docs.searxng.org) | API key is optional. Set your SearXNG base URL. Some instances may require a bearer token for access. | -| `serper-search` | `serper-search` | Serper Search | Search | [link](https://serper.dev) | API key from serper.dev dashboard | -| `tavily-search` | `tavily-search` | Tavily Search | Search | [link](https://tavily.com) | API key from app.tavily.com (format: tvly-...) | -| `youcom-search` | `youcom-search` | You.com Search | Search | [link](https://you.com/docs/search/overview) | X-API-Key from the You.com platform dashboard | +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `brave-search` | `brave-search` | Brave Search | Search | [link](https://brave.com/search/api) | Subscription token from Brave Search API dashboard | +| `exa-search` | `exa-search` | Exa Search | Search | [link](https://exa.ai) | API key from dashboard.exa.ai | +| `google-pse-search` | `google-pse` | Google Programmable Search | Search | [link](https://developers.google.com/custom-search/v1/overview) | Requires a Google API key and your Programmable Search Engine ID (cx) | +| `linkup-search` | `linkup` | Linkup Search | Search | [link](https://docs.linkup.so) | Bearer API key from the Linkup dashboard | +| `ollama-search` | `ollama-search` | Ollama Search | Search | [link](https://ollama.com/settings/keys) | Same API key as Ollama Cloud (from ollama.com/settings/keys) | +| `perplexity-search` | `pplx-search` | Perplexity Search | Search | [link](https://docs.perplexity.ai/guides/search-quickstart) | Same API key as Perplexity (pplx-...) | +| `searchapi-search` | `searchapi` | SearchAPI | Search | [link](https://www.searchapi.io/docs/google) | API key from SearchAPI (query param or Bearer auth) | +| `searxng-search` | `searxng` | SearXNG Search | Search | [link](https://docs.searxng.org) | API key is optional. Set your SearXNG base URL. Some instances may require a bearer token for access. | +| `serper-search` | `serper-search` | Serper Search | Search | [link](https://serper.dev) | API key from serper.dev dashboard | +| `tavily-search` | `tavily-search` | Tavily Search | Search | [link](https://tavily.com) | API key from app.tavily.com (format: tvly-...) | +| `youcom-search` | `youcom-search` | You.com Search | Search | [link](https://you.com/business/api/) | X-API-Key from the You.com platform dashboard | ## Audio-only Providers (7) -| ID | Alias | Name | Tags | Website | Notes | -| ------------ | ---------- | ---------- | ----- | ------------------------------------- | ----------------------------------------------------------------------------------------------- | -| `assemblyai` | `aai` | AssemblyAI | Audio | [link](https://assemblyai.com) | — | -| `aws-polly` | `polly` | AWS Polly | Audio | [link](https://aws.amazon.com/polly/) | Use AWS Secret Access Key as API key; set providerSpecificData.accessKeyId and optional region. | -| `cartesia` | `cartesia` | Cartesia | Audio | [link](https://cartesia.ai) | — | -| `deepgram` | `dg` | Deepgram | Audio | [link](https://deepgram.com) | — | -| `elevenlabs` | `el` | ElevenLabs | Audio | [link](https://elevenlabs.io) | — | -| `inworld` | `inworld` | Inworld | Audio | [link](https://inworld.ai) | — | -| `playht` | `playht` | PlayHT | Audio | [link](https://play.ht) | — | +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `assemblyai` | `aai` | AssemblyAI | Audio | [link](https://assemblyai.com) | — | +| `aws-polly` | `polly` | AWS Polly | Audio | [link](https://aws.amazon.com/polly/) | Use AWS Secret Access Key as API key; set providerSpecificData.accessKeyId and optional region. | +| `cartesia` | `cartesia` | Cartesia | Audio | [link](https://cartesia.ai) | — | +| `deepgram` | `dg` | Deepgram | Audio | [link](https://deepgram.com) | — | +| `elevenlabs` | `el` | ElevenLabs | Audio | [link](https://elevenlabs.io) | — | +| `inworld` | `inworld` | Inworld | Audio | [link](https://inworld.ai) | — | +| `playht` | `playht` | PlayHT | Audio | [link](https://play.ht) | — | ## Upstream Proxy Providers (2) -| ID | Alias | Name | Tags | Website | Notes | -| ------------- | ----- | ----------- | -------------- | ---------------------------------------------------- | ----- | -| `9router` | `nr` | 9router | Upstream proxy | [link](https://www.npmjs.com/package/9router) | — | -| `cliproxyapi` | `cpa` | CLIProxyAPI | Upstream proxy | [link](https://github.com/router-for-me/CLIProxyAPI) | — | +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `9router` | `nr` | 9router | Upstream proxy | [link](https://www.npmjs.com/package/9router) | — | +| `cliproxyapi` | `cpa` | CLIProxyAPI | Upstream proxy | [link](https://github.com/router-for-me/CLIProxyAPI) | — | ## Cloud Agent Providers (3) -| ID | Alias | Name | Tags | Website | Notes | -| ------------- | ------------- | ------------ | ----------- | -------------------------------- | ----------------------------------------------------------- | -| `codex-cloud` | `codex-cloud` | Codex Cloud | Cloud agent | [link](https://openai.com/codex) | OpenAI API key with Codex Cloud task access. | -| `devin` | `devin` | Devin | Cloud agent | [link](https://devin.ai) | Devin API key for cloud agent sessions. | -| `jules` | `jules` | Google Jules | Cloud agent | [link](https://jules.google) | Jules API key for creating and managing cloud coding tasks. | +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `codex-cloud` | `codex-cloud` | Codex Cloud | Cloud agent | [link](https://openai.com/codex) | OpenAI API key with Codex Cloud task access. | +| `devin` | `devin` | Devin | Cloud agent | [link](https://devin.ai) | Devin API key for cloud agent sessions. | +| `jules` | `jules` | Google Jules | Cloud agent | [link](https://jules.google) | Jules API key for creating and managing cloud coding tasks. | ## System Providers (1) -| ID | Alias | Name | Tags | Website | Notes | -| ------ | ------ | ------------------ | ------ | ------- | ----- | -| `auto` | `auto` | Auto (Zero-Config) | System | — | — | +| ID | Alias | Name | Tags | Website | Notes | +|----|-------|------|------|---------|-------| +| `auto` | `auto` | Auto (Zero-Config) | System | — | — | ## Sources of truth diff --git a/open-sse/config/providers/registry/kimi/web/index.ts b/open-sse/config/providers/registry/kimi/web/index.ts index e425c62474..c86d3ce426 100644 --- a/open-sse/config/providers/registry/kimi/web/index.ts +++ b/open-sse/config/providers/registry/kimi/web/index.ts @@ -7,11 +7,15 @@ export const kimi_webProvider: RegistryEntry = { alias: "kimi-web", format: "openai", executor: "kimi-web", - baseUrl: "https://kimi.moonshot.cn/api/chat", + // International consumer chat — the legacy `kimi.moonshot.cn` domain now + // redirects every non-CN visitor to www.kimi.com, which speaks a different + // Connect-RPC API. See `open-sse/executors/kimi-web.ts` for the wire format. + baseUrl: "https://www.kimi.com", authType: "apikey", authHeader: "cookie", models: [ { id: "kimi-default", name: "Kimi Default" }, + { id: "kimi-k2.6", name: "Kimi K2.6 (Thinking)" }, { id: "kimi-128k", name: "Kimi 128K (Long Context)" }, ], }; diff --git a/open-sse/executors/kimi-web.ts b/open-sse/executors/kimi-web.ts index 6249480286..f45e9d1072 100644 --- a/open-sse/executors/kimi-web.ts +++ b/open-sse/executors/kimi-web.ts @@ -1,55 +1,261 @@ /** - * KimiWebExecutor — Moonshot AI Chat via kimi.moonshot.cn + * KimiWebExecutor — Moonshot AI Chat via www.kimi.com (international) * - * Routes requests through Kimi's consumer chat API. - * Chinese market provider with strong long-context support. + * Routes requests through Kimi's consumer chat API on the international domain. + * Originally this executor targeted `kimi.moonshot.cn` (mainland-CN consumer + * chat). That domain now redirects every visitor outside CN to + * `https://www.kimi.com/`, which speaks a completely different API surface: * - * Endpoint: POST https://kimi.moonshot.cn/api/chat - * Auth: Session cookie from kimi.moonshot.cn + * - Endpoint: POST /apiv2/kimi.gateway.chat.v1.ChatService/Chat + * - Protocol: Connect-RPC (unary envelope framing — 5-byte header + JSON) + * - Auth: `Authorization: Bearer ` + `Cookie: kimi-auth=` + * - Body: Connect-framed `{scenario, message:{role,blocks:[{text:{content}}]}, + * options:{thinking,enable_plugin}}` + * - Response: Connect-framed stream of events carrying deltas with one of + * `mask: "block.text.content"` (answer) or + * `mask: "block.think.content"` (reasoning), emitted via + * `op: "set"` (initial) and `op: "append"` (incremental). + * + * Cookie handling: the user pastes their full Cookie header from www.kimi.com. + * We extract the `kimi-auth` JWT from it (it is the only cookie the upstream + * actually consults) and use it both as the Bearer token and as the Cookie we + * send back, so we don't leak the user's analytics cookies (Ga, CF, HM, ...). + * + * The `x-msh-*` / `x-traffic-id` / `x-msh-shield-data` headers the SPA sends + * are NOT required — verified by stripping them one at a time against a live + * session; the upstream returns the same response either way. */ import { BaseExecutor, type ExecuteInput } from "./base.ts"; -import { makeExecutorErrorResult as makeErrorResult, normalizeCookie } from "../utils/error.ts"; +import { makeExecutorErrorResult as makeErrorResult, sanitizeErrorMessage } from "../utils/error.ts"; +import { extractKimiJwt } from "@/lib/providers/webCookieAuth"; -const BASE_URL = "https://kimi.moonshot.cn"; -const CHAT_URL = `${BASE_URL}/api/chat`; +export { extractKimiJwt }; + +const BASE_URL = "https://www.kimi.com"; +const CHAT_URL = `${BASE_URL}/apiv2/kimi.gateway.chat.v1.ChatService/Chat`; const USER_AGENT = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/149.0.0.0 Safari/537.36"; +const DEFAULT_SCENARIO = "SCENARIO_K2D5"; + +/** Wrap a JSON message in the 5-byte Connect streaming envelope (flags + length). */ +export function frameConnectMessage(json: string): Uint8Array { + const payload = new TextEncoder().encode(json); + const framed = new Uint8Array(5 + payload.length); + framed[0] = 0; // flags: 0 = uncompressed + const len = payload.length; + framed[1] = (len >>> 24) & 0xff; + framed[2] = (len >>> 16) & 0xff; + framed[3] = (len >>> 8) & 0xff; + framed[4] = len & 0xff; + framed.set(payload, 5); + return framed; +} + +interface ConnectFrame { + flags: number; + message: Record | null; +} + +/** + * ponytail: cap a single Connect frame at 8 MiB. Kimi's largest legitimate + * event is well under 1 KiB (a delta or stage transition); anything bigger + * means the upstream is misbehaving or an attacker controls the response and + * is trying to OOM the proxy by sending a header claiming a huge length. + * The non-streaming accumulator would otherwise grow unbounded. If you ever + * see this tripping in production, raise the ceiling and add a regression + * test — but never remove it. + */ +const MAX_FRAME_LEN = 8 * 1024 * 1024; + +/** + * Decode one Connect frame from a stream buffer. + * Returns: + * - `consumed: 0` if there isn't enough data yet (need more bytes) + * - `consumed: -1` if the frame header claims a length above MAX_FRAME_LEN + * (caller must treat this as a stream-fatal protocol error) + * - `consumed: N` + the parsed frame otherwise + */ +export function decodeConnectFrame(buf: Uint8Array, byteOffset: number): { consumed: number; frame: ConnectFrame | null } { + if (byteOffset + 5 > buf.length) return { consumed: 0, frame: null }; + const flags = buf[byteOffset]; + const len = + (buf[byteOffset + 1] << 24) | + (buf[byteOffset + 2] << 16) | + (buf[byteOffset + 3] << 8) | + buf[byteOffset + 4]; + // Sign-extend the high bit back to negative when len was read as signed. + const msgLen = len < 0 ? len + 0x100000000 : len; + if (msgLen > MAX_FRAME_LEN) return { consumed: -1, frame: null }; + if (byteOffset + 5 + msgLen > buf.length) return { consumed: 0, frame: null }; + + const payload = buf.subarray(byteOffset + 5, byteOffset + 5 + msgLen); + let message: Record | null = null; + if (msgLen > 0) { + try { + message = JSON.parse(new TextDecoder().decode(payload)); + } catch { + message = null; + } + } + return { consumed: 5 + msgLen, frame: { flags, message } }; +} + +type DeltaKind = "text" | "think" | null; + +/** + * Extract a content delta + kind from a Connect frame message. + * + * The chat stream uses two ops against two masks: + * - `op: "set"` on `block.text` / `block.think` → first chunk + * - `op: "append"` on `block.text.content` / `block.think.content` → subsequent chunks + * + * Anything else (heartbeats, chat/message metadata, stage transitions) is + * suppressed; we only surface text to the client. + */ +export function extractDelta(msg: Record | null): { kind: DeltaKind; text: string } | null { + if (!msg) return null; + const op = String(msg.op ?? ""); + const mask = String(msg.mask ?? ""); + const block = (msg.block ?? {}) as Record; + + // `op: append` carries a delta string under `block..content`. + if (op === "append") { + if (mask === "block.text.content") { + const text = String(((block.text ?? {}) as Record).content ?? ""); + return text ? { kind: "text", text } : null; + } + if (mask === "block.think.content") { + const text = String(((block.think ?? {}) as Record).content ?? ""); + return text ? { kind: "think", text } : null; + } + return null; + } + + // `op: set` on `block.text` / `block.think` carries the initial content. + if (op === "set") { + if (mask === "block.text") { + const text = String(((block.text ?? {}) as Record).content ?? ""); + return text ? { kind: "text", text } : null; + } + if (mask === "block.think") { + const text = String(((block.think ?? {}) as Record).content ?? ""); + return text ? { kind: "think", text } : null; + } + } + return null; +} + +export function isEndOfStream(msg: Record | null): boolean { + if (!msg) return false; + // Assistant message flipped to COMPLETED. + const message = (msg.message ?? null) as Record | null; + if (message && String(message.status ?? "") === "MESSAGE_STATUS_COMPLETED" && String(message.role ?? "") === "assistant") { + return true; + } + return false; +} + +/** + * Fold a multi-turn OpenAI `messages` array into a single Kimi user turn. + * + * Limitations (kimi-web is a single-turn consumer chat, not an agentic API): + * - `tool` and `function` role messages are silently dropped — Kimi's web + * chat has no concept of tool results, so agentic flows should use the + * `kimi-coding` (api.kimi.com) provider instead. + * - Assistant `tool_calls` and image content parts are stringified into + * text, which loses structure. Acceptable for free-text continuation, + * unacceptable for tool-round-trip — same workaround: use kimi-coding. + */ +export function foldMessages(messages: Array<{ role: string; content: unknown }>): string { + let system = ""; + let user = ""; + for (const m of messages) { + const text = typeof m.content === "string" ? m.content : JSON.stringify(m.content ?? ""); + if (m.role === "system") { + system += (system ? "\n\n" : "") + text; + } else if (m.role === "user") { + // Kimi's web chat is single-turn; keep only the latest user content but + // preserve prior assistant text for continuity when present. + user = user ? `${user}\n\n${text}` : text; + } else if (m.role === "assistant") { + user = user ? `${user}\n\nAssistant: ${text}` : `Assistant: ${text}`; + } + } + return system ? `${system}\n\n${user}` : user; +} + export class KimiWebExecutor extends BaseExecutor { constructor() { - super("kimi-web", { id: "kimi-web", baseUrl: "https://kimi.moonshot.cn" }); + super("kimi-web", { id: "kimi-web", baseUrl: BASE_URL }); + } + + private buildKimiHeaders(jwt: string): Record { + const headers: Record = { + "Content-Type": "application/connect+json", + Accept: "*/*", + "User-Agent": USER_AGENT, + Origin: BASE_URL, + Referer: `${BASE_URL}/`, + "connect-protocol-version": "1", + }; + if (jwt) { + headers["Authorization"] = `Bearer ${jwt}`; + headers["Cookie"] = `kimi-auth=${jwt}`; + } + return headers; + } + + private buildRequestBody(prompt: string, wantThinking: boolean): string { + return JSON.stringify({ + scenario: DEFAULT_SCENARIO, + tools: [{ type: "TOOL_TYPE_SEARCH", search: {} }, { type: "TOOL_TYPE_CRON_JOB" }], + message: { + role: "user", + blocks: [{ message_id: "", text: { content: prompt } }], + scenario: DEFAULT_SCENARIO, + }, + options: { thinking: wantThinking, enable_plugin: true }, + }); } async execute(input: ExecuteInput) { const { body, credentials, signal, stream: wantStream } = input; const bodyObj = (body || {}) as Record; - const rawCookie = normalizeCookie(String(credentials?.apiKey ?? "").trim()); - const messages = (bodyObj.messages as Array<{ role: string; content: string }>) || []; + const rawCredential = String(credentials?.apiKey ?? "").trim(); + const jwt = extractKimiJwt(rawCredential); + if (!jwt) { + return makeErrorResult( + 400, + "Missing Kimi session — paste the full Cookie header from www.kimi.com (must contain kimi-auth=) or just the JWT itself.", + body, + CHAT_URL + ); + } + + const messages = (bodyObj.messages as Array<{ role: string; content: unknown }>) || []; const modelId = (bodyObj.model as string) || "kimi-default"; + // Decide thinking intent. A user sending `reasoning_effort: "none"` is + // explicit — honour it even when the model id suggests a thinking variant. + // Otherwise thinking models (kimi-k2.6 etc.) default to thinking on. + const modelWantsThinking = /k2\.6|k2-6|think/i.test(modelId); + const wantThinking = bodyObj.reasoning_effort === "none" ? false : modelWantsThinking; - const reqBody = { - messages: messages.map((m) => ({ role: m.role, content: m.content })), - model: modelId, - stream: wantStream, - max_tokens: (bodyObj.max_tokens as number) || 4096, - }; + const prompt = foldMessages(messages); + const reqBody = this.buildRequestBody(prompt, wantThinking); + const reqHeaders = this.buildKimiHeaders(jwt); - const reqHeaders: Record = { - "Content-Type": "application/json", - "User-Agent": USER_AGENT, - Accept: wantStream ? "text/event-stream" : "application/json", - Referer: `${BASE_URL}/`, - Origin: BASE_URL, - }; - if (rawCookie) reqHeaders.Cookie = rawCookie; + // Connect framing wraps the JSON body in a 5-byte envelope. Without it the + // upstream returns `invalid_argument` for every request. + const framedBody = frameConnectMessage(reqBody); let upstream: Response; try { upstream = await fetch(CHAT_URL, { method: "POST", headers: reqHeaders, - body: JSON.stringify(reqBody), + body: new Uint8Array(framedBody), signal, }); } catch (err) { @@ -63,93 +269,171 @@ export class KimiWebExecutor extends BaseExecutor { if (!upstream.ok) { const errText = await upstream.text().catch(() => ""); - return makeErrorResult(upstream.status, `Kimi error: ${errText}`, body, CHAT_URL); + return makeErrorResult(upstream.status, `Kimi error: ${sanitizeErrorMessage(errText)}`, body, CHAT_URL); } - if (!wantStream) { - const data = (await upstream.json()) as Record; - const content = - (data?.choices as Array<{ message?: { content?: string } }>)?.[0]?.message?.content || - (data?.content as string) || - ""; + const encoder = new TextEncoder(); + const id = `chatcmpl-kimi-${Date.now()}`; + const created = Math.floor(Date.now() / 1000); + + const emitChunk = ( + controller: ReadableStreamDefaultController, + delta: Record, + finish: string | null = null + ) => { + const chunk = { + id, + object: "chat.completion.chunk", + created, + model: modelId, + choices: [{ index: 0, delta, finish_reason: finish }], + }; + controller.enqueue(encoder.encode(`data: ${JSON.stringify(chunk)}\n\n`)); + }; + + // The upstream is a Connect-framed stream regardless of whether the + // client asked for SSE — Kimi always streams. For non-streaming clients + // we buffer the full response below. + const sourceStream = upstream.body ?? new ReadableStream({ start: (c) => c.close() }); + + if (wantStream) { + const outStream = new ReadableStream({ + async start(controller) { + const reader = sourceStream.getReader(); + let buffer = new Uint8Array(0); + let emittedRole = false; + try { + while (true) { + const { done, value } = await reader.read(); + if (done) break; + if (value) { + const merged = new Uint8Array(buffer.length + value.length); + merged.set(buffer, 0); + merged.set(value, buffer.length); + buffer = merged; + + let offset = 0; + while (offset < buffer.length) { + const { consumed, frame } = decodeConnectFrame(buffer, offset); + if (consumed === -1) { + // Frame header claims a length above MAX_FRAME_LEN — stream-fatal. + controller.error(new Error("Kimi Connect frame exceeded MAX_FRAME_LEN")); + return; + } + if (consumed === 0) break; // need more bytes + offset += consumed; + if (!frame?.message) continue; + + const delta = extractDelta(frame.message); + if (delta) { + if (!emittedRole) { + emittedRole = true; + emitChunk(controller, { role: "assistant", content: "" }); + } + if (delta.kind === "think") { + emitChunk(controller, { reasoning_content: delta.text }); + } else { + emitChunk(controller, { content: delta.text }); + } + } + if (isEndOfStream(frame.message)) { + emitChunk(controller, {}, "stop"); + controller.enqueue(encoder.encode("data: [DONE]\n\n")); + controller.close(); + return; + } + } + // Compact the buffer. + buffer = buffer.subarray(offset); + } + } + // Stream ended without an explicit COMPLETED marker — flush a stop. + if (!emittedRole) { + emitChunk(controller, { role: "assistant", content: "" }); + } + emitChunk(controller, {}, "stop"); + controller.enqueue(encoder.encode("data: [DONE]\n\n")); + controller.close(); + } catch (err) { + if (!signal?.aborted) { + try { + controller.error(err); + } catch { + /* controller already closed */ + } + } + } + }, + }); + return { - response: new Response( - JSON.stringify({ - id: `chatcmpl-kimi-${Date.now()}`, - object: "chat.completion", - created: Math.floor(Date.now() / 1000), - model: modelId, - choices: [{ index: 0, message: { role: "assistant", content }, finish_reason: "stop" }], - }), - { headers: { "Content-Type": "application/json" } } - ), + response: new Response(outStream, { + headers: { + "Content-Type": "text/event-stream", + "Cache-Control": "no-cache", + Connection: "keep-alive", + }, + }), url: CHAT_URL, headers: reqHeaders, - transformedBody: reqBody, + transformedBody: JSON.parse(reqBody), }; } - // Streaming - const encoder = new TextEncoder(); - const decoder = new TextDecoder(); - const stream = new ReadableStream({ - async start(controller) { - const reader = upstream.body?.getReader(); - if (!reader) { - controller.close(); - return; - } - let buffer = ""; - try { - while (true) { - const { done, value } = await reader.read(); - if (done) break; - buffer += decoder.decode(value, { stream: true }); - const lines = buffer.split("\n"); - buffer = lines.pop() || ""; - for (const line of lines) { - if (!line.startsWith("data:")) continue; - const data = line.slice(5).trim(); - if (data === "[DONE]") { - controller.enqueue(encoder.encode("data: [DONE]\n\n")); - continue; - } - try { - const parsed = JSON.parse(data); - const text = parsed.choices?.[0]?.delta?.content || ""; - if (text) { - const chunk = { - id: `chatcmpl-kimi-${Date.now()}`, - object: "chat.completion.chunk", - created: Math.floor(Date.now() / 1000), - model: modelId, - choices: [{ index: 0, delta: { content: text }, finish_reason: null }], - }; - controller.enqueue(encoder.encode(`data: ${JSON.stringify(chunk)}\n\n`)); - } - } catch {} - } - } - } catch (err) { - if (!signal?.aborted) controller.error(err); - } finally { - controller.enqueue(encoder.encode("data: [DONE]\n\n")); - controller.close(); - } - }, - }); + // Non-streaming: collect all deltas into a single chat.completion JSON. + let answer = ""; + let reasoning = ""; + const reader = sourceStream.getReader(); + let buffer = new Uint8Array(0); + try { + while (true) { + const { done, value } = await reader.read(); + if (done) break; + if (!value) continue; + const merged = new Uint8Array(buffer.length + value.length); + merged.set(buffer, 0); + merged.set(value, buffer.length); + buffer = merged; + let offset = 0; + while (offset < buffer.length) { + const { consumed, frame } = decodeConnectFrame(buffer, offset); + if (consumed === -1) break; // oversized frame — abort, return what we have + if (consumed === 0) break; + offset += consumed; + if (!frame?.message) continue; + const delta = extractDelta(frame.message); + if (delta) { + if (delta.kind === "think") reasoning += delta.text; + else answer += delta.text; + } + if (isEndOfStream(frame.message)) { + offset = buffer.length; // drain + break; + } + } + buffer = buffer.subarray(offset); + } + } catch { + /* best-effort — return what we have */ + } + + const message: Record = { role: "assistant", content: answer }; + if (reasoning) message.reasoning_content = reasoning; + const completion = { + id, + object: "chat.completion", + created, + model: modelId, + choices: [{ index: 0, message, finish_reason: "stop" }], + }; return { - response: new Response(stream, { - headers: { - "Content-Type": "text/event-stream", - "Cache-Control": "no-cache", - Connection: "keep-alive", - }, + response: new Response(JSON.stringify(completion), { + headers: { "Content-Type": "application/json" }, }), url: CHAT_URL, headers: reqHeaders, - transformedBody: reqBody, + transformedBody: JSON.parse(reqBody), }; } } diff --git a/open-sse/services/tokenExtractionConfig.ts b/open-sse/services/tokenExtractionConfig.ts index f82804e2e0..7d141bacc1 100644 --- a/open-sse/services/tokenExtractionConfig.ts +++ b/open-sse/services/tokenExtractionConfig.ts @@ -193,14 +193,13 @@ const RAW_CONFIGS: TokenExtractionConfig[] = [ config( "kimi-web", "Kimi (Moonshot)", - "https://kimi.moonshot.cn/", - "https://kimi.moonshot.cn", + "https://www.kimi.com/", + "https://www.kimi.com", [ - { type: "cookie", name: "kimi_token", domain: ".kimi.moonshot.cn" }, - { type: "localStorage", key: "kimi_token" }, + { type: "cookie", name: "kimi-auth", domain: ".kimi.com" }, ], - "Log in to Kimi at kimi.moonshot.cn via phone/WeChat. The session token will be extracted.", - { cookieDomain: ".kimi.moonshot.cn" } + "Log in to Kimi at www.kimi.com (international). The kimi-auth JWT cookie will be extracted.", + { cookieDomain: ".kimi.com" } ), // ── Blackbox Web ────────────────────────────────────────── diff --git a/src/i18n/messages/en.json b/src/i18n/messages/en.json index 40dd6d56b6..8c1d17ae05 100644 --- a/src/i18n/messages/en.json +++ b/src/i18n/messages/en.json @@ -4836,7 +4836,7 @@ "v0VercelWebLabel": "v0 Vercel Web", "v0VercelWebDesc": "AI code generation via v0.dev", "kimiWebLabel": "Kimi Web", - "kimiWebDesc": "Chinese market AI chat via kimi.moonshot.cn", + "kimiWebDesc": "Moonshot AI consumer chat via www.kimi.com (international, Connect-RPC API)", "doubaoWebLabel": "Doubao Web", "doubaoWebDesc": "ByteDance AI chat via doubao.com" }, diff --git a/src/i18n/messages/it.json b/src/i18n/messages/it.json index b8a544b12f..751b0b4bb8 100644 --- a/src/i18n/messages/it.json +++ b/src/i18n/messages/it.json @@ -4811,7 +4811,7 @@ "v0VercelWebLabel": "__MISSING__:v0 Vercel Web", "v0VercelWebDesc": "__MISSING__:AI code generation via v0.dev", "kimiWebLabel": "__MISSING__:Kimi Web", - "kimiWebDesc": "__MISSING__:Chinese market AI chat via kimi.moonshot.cn", + "kimiWebDesc": "__MISSING__:Moonshot AI consumer chat via www.kimi.com (international, Connect-RPC API)", "doubaoWebLabel": "__MISSING__:Doubao Web", "doubaoWebDesc": "__MISSING__:ByteDance AI chat via doubao.com" }, diff --git a/src/i18n/messages/ja.json b/src/i18n/messages/ja.json index 418afb8712..90b3976cb2 100644 --- a/src/i18n/messages/ja.json +++ b/src/i18n/messages/ja.json @@ -4811,7 +4811,7 @@ "v0VercelWebLabel": "__MISSING__:v0 Vercel Web", "v0VercelWebDesc": "__MISSING__:AI code generation via v0.dev", "kimiWebLabel": "__MISSING__:Kimi Web", - "kimiWebDesc": "__MISSING__:Chinese market AI chat via kimi.moonshot.cn", + "kimiWebDesc": "__MISSING__:Moonshot AI consumer chat via www.kimi.com (international, Connect-RPC API)", "doubaoWebLabel": "__MISSING__:Doubao Web", "doubaoWebDesc": "__MISSING__:ByteDance AI chat via doubao.com" }, diff --git a/src/i18n/messages/ko.json b/src/i18n/messages/ko.json index 3185cf6f7d..98ceae49a2 100644 --- a/src/i18n/messages/ko.json +++ b/src/i18n/messages/ko.json @@ -4811,7 +4811,7 @@ "v0VercelWebLabel": "__MISSING__:v0 Vercel Web", "v0VercelWebDesc": "__MISSING__:AI code generation via v0.dev", "kimiWebLabel": "__MISSING__:Kimi Web", - "kimiWebDesc": "__MISSING__:Chinese market AI chat via kimi.moonshot.cn", + "kimiWebDesc": "__MISSING__:Moonshot AI consumer chat via www.kimi.com (international, Connect-RPC API)", "doubaoWebLabel": "__MISSING__:Doubao Web", "doubaoWebDesc": "__MISSING__:ByteDance AI chat via doubao.com" }, diff --git a/src/i18n/messages/mr.json b/src/i18n/messages/mr.json index 619f356739..16cb405cbd 100644 --- a/src/i18n/messages/mr.json +++ b/src/i18n/messages/mr.json @@ -4811,7 +4811,7 @@ "v0VercelWebLabel": "__MISSING__:v0 Vercel Web", "v0VercelWebDesc": "__MISSING__:AI code generation via v0.dev", "kimiWebLabel": "__MISSING__:Kimi Web", - "kimiWebDesc": "__MISSING__:Chinese market AI chat via kimi.moonshot.cn", + "kimiWebDesc": "__MISSING__:Moonshot AI consumer chat via www.kimi.com (international, Connect-RPC API)", "doubaoWebLabel": "__MISSING__:Doubao Web", "doubaoWebDesc": "__MISSING__:ByteDance AI chat via doubao.com" }, diff --git a/src/i18n/messages/ms.json b/src/i18n/messages/ms.json index f77759ef3a..df63b1b73a 100644 --- a/src/i18n/messages/ms.json +++ b/src/i18n/messages/ms.json @@ -4811,7 +4811,7 @@ "v0VercelWebLabel": "__MISSING__:v0 Vercel Web", "v0VercelWebDesc": "__MISSING__:AI code generation via v0.dev", "kimiWebLabel": "__MISSING__:Kimi Web", - "kimiWebDesc": "__MISSING__:Chinese market AI chat via kimi.moonshot.cn", + "kimiWebDesc": "__MISSING__:Moonshot AI consumer chat via www.kimi.com (international, Connect-RPC API)", "doubaoWebLabel": "__MISSING__:Doubao Web", "doubaoWebDesc": "__MISSING__:ByteDance AI chat via doubao.com" }, diff --git a/src/i18n/messages/nl.json b/src/i18n/messages/nl.json index 9c2d5c7a2e..c535d6d3ff 100644 --- a/src/i18n/messages/nl.json +++ b/src/i18n/messages/nl.json @@ -4811,7 +4811,7 @@ "v0VercelWebLabel": "__MISSING__:v0 Vercel Web", "v0VercelWebDesc": "__MISSING__:AI code generation via v0.dev", "kimiWebLabel": "__MISSING__:Kimi Web", - "kimiWebDesc": "__MISSING__:Chinese market AI chat via kimi.moonshot.cn", + "kimiWebDesc": "__MISSING__:Moonshot AI consumer chat via www.kimi.com (international, Connect-RPC API)", "doubaoWebLabel": "__MISSING__:Doubao Web", "doubaoWebDesc": "__MISSING__:ByteDance AI chat via doubao.com" }, diff --git a/src/i18n/messages/no.json b/src/i18n/messages/no.json index 1d2a66826a..5b6faa9760 100644 --- a/src/i18n/messages/no.json +++ b/src/i18n/messages/no.json @@ -4811,7 +4811,7 @@ "v0VercelWebLabel": "__MISSING__:v0 Vercel Web", "v0VercelWebDesc": "__MISSING__:AI code generation via v0.dev", "kimiWebLabel": "__MISSING__:Kimi Web", - "kimiWebDesc": "__MISSING__:Chinese market AI chat via kimi.moonshot.cn", + "kimiWebDesc": "__MISSING__:Moonshot AI consumer chat via www.kimi.com (international, Connect-RPC API)", "doubaoWebLabel": "__MISSING__:Doubao Web", "doubaoWebDesc": "__MISSING__:ByteDance AI chat via doubao.com" }, diff --git a/src/i18n/messages/phi.json b/src/i18n/messages/phi.json index 20903e812e..e0b76d7867 100644 --- a/src/i18n/messages/phi.json +++ b/src/i18n/messages/phi.json @@ -4811,7 +4811,7 @@ "v0VercelWebLabel": "__MISSING__:v0 Vercel Web", "v0VercelWebDesc": "__MISSING__:AI code generation via v0.dev", "kimiWebLabel": "__MISSING__:Kimi Web", - "kimiWebDesc": "__MISSING__:Chinese market AI chat via kimi.moonshot.cn", + "kimiWebDesc": "__MISSING__:Moonshot AI consumer chat via www.kimi.com (international, Connect-RPC API)", "doubaoWebLabel": "__MISSING__:Doubao Web", "doubaoWebDesc": "__MISSING__:ByteDance AI chat via doubao.com" }, diff --git a/src/i18n/messages/pl.json b/src/i18n/messages/pl.json index 105eabf5d0..fac61cfc7c 100644 --- a/src/i18n/messages/pl.json +++ b/src/i18n/messages/pl.json @@ -4811,7 +4811,7 @@ "v0VercelWebLabel": "__MISSING__:v0 Vercel Web", "v0VercelWebDesc": "__MISSING__:AI code generation via v0.dev", "kimiWebLabel": "__MISSING__:Kimi Web", - "kimiWebDesc": "__MISSING__:Chinese market AI chat via kimi.moonshot.cn", + "kimiWebDesc": "__MISSING__:Moonshot AI consumer chat via www.kimi.com (international, Connect-RPC API)", "doubaoWebLabel": "__MISSING__:Doubao Web", "doubaoWebDesc": "__MISSING__:ByteDance AI chat via doubao.com" }, diff --git a/src/i18n/messages/pt-BR.json b/src/i18n/messages/pt-BR.json index 805d8c03a9..d0d2a29383 100644 --- a/src/i18n/messages/pt-BR.json +++ b/src/i18n/messages/pt-BR.json @@ -4811,7 +4811,7 @@ "v0VercelWebLabel": "__MISSING__:v0 Vercel Web", "v0VercelWebDesc": "__MISSING__:AI code generation via v0.dev", "kimiWebLabel": "__MISSING__:Kimi Web", - "kimiWebDesc": "__MISSING__:Chinese market AI chat via kimi.moonshot.cn", + "kimiWebDesc": "__MISSING__:Moonshot AI consumer chat via www.kimi.com (international, Connect-RPC API)", "doubaoWebLabel": "__MISSING__:Doubao Web", "doubaoWebDesc": "__MISSING__:ByteDance AI chat via doubao.com" }, diff --git a/src/i18n/messages/pt.json b/src/i18n/messages/pt.json index 8bcae514c7..3986125113 100644 --- a/src/i18n/messages/pt.json +++ b/src/i18n/messages/pt.json @@ -4811,7 +4811,7 @@ "v0VercelWebLabel": "__MISSING__:v0 Vercel Web", "v0VercelWebDesc": "__MISSING__:AI code generation via v0.dev", "kimiWebLabel": "__MISSING__:Kimi Web", - "kimiWebDesc": "__MISSING__:Chinese market AI chat via kimi.moonshot.cn", + "kimiWebDesc": "__MISSING__:Moonshot AI consumer chat via www.kimi.com (international, Connect-RPC API)", "doubaoWebLabel": "__MISSING__:Doubao Web", "doubaoWebDesc": "__MISSING__:ByteDance AI chat via doubao.com" }, diff --git a/src/i18n/messages/ro.json b/src/i18n/messages/ro.json index 2668ee7132..f31287cbf3 100644 --- a/src/i18n/messages/ro.json +++ b/src/i18n/messages/ro.json @@ -4811,7 +4811,7 @@ "v0VercelWebLabel": "__MISSING__:v0 Vercel Web", "v0VercelWebDesc": "__MISSING__:AI code generation via v0.dev", "kimiWebLabel": "__MISSING__:Kimi Web", - "kimiWebDesc": "__MISSING__:Chinese market AI chat via kimi.moonshot.cn", + "kimiWebDesc": "__MISSING__:Moonshot AI consumer chat via www.kimi.com (international, Connect-RPC API)", "doubaoWebLabel": "__MISSING__:Doubao Web", "doubaoWebDesc": "__MISSING__:ByteDance AI chat via doubao.com" }, diff --git a/src/i18n/messages/ru.json b/src/i18n/messages/ru.json index 2c5c0f7568..799c1ccc57 100644 --- a/src/i18n/messages/ru.json +++ b/src/i18n/messages/ru.json @@ -4811,7 +4811,7 @@ "v0VercelWebLabel": "v0 Vercel Web", "v0VercelWebDesc": "Генерация кода ИИ через v0.dev", "kimiWebLabel": "Kimi Web", - "kimiWebDesc": "Китайский рынок AI чата через kimi.moonshot.cn", + "kimiWebDesc": "Чат Moonshot AI через www.kimi.com (международная версия, Connect-RPC API)", "doubaoWebLabel": "Doubao Web", "doubaoWebDesc": "Чат AI ByteDance через doubao.com" }, diff --git a/src/i18n/messages/sk.json b/src/i18n/messages/sk.json index bf043d2359..c07cb2bfcd 100644 --- a/src/i18n/messages/sk.json +++ b/src/i18n/messages/sk.json @@ -4811,7 +4811,7 @@ "v0VercelWebLabel": "__MISSING__:v0 Vercel Web", "v0VercelWebDesc": "__MISSING__:AI code generation via v0.dev", "kimiWebLabel": "__MISSING__:Kimi Web", - "kimiWebDesc": "__MISSING__:Chinese market AI chat via kimi.moonshot.cn", + "kimiWebDesc": "__MISSING__:Moonshot AI consumer chat via www.kimi.com (international, Connect-RPC API)", "doubaoWebLabel": "__MISSING__:Doubao Web", "doubaoWebDesc": "__MISSING__:ByteDance AI chat via doubao.com" }, diff --git a/src/i18n/messages/sv.json b/src/i18n/messages/sv.json index 3639ea65d6..405364a64f 100644 --- a/src/i18n/messages/sv.json +++ b/src/i18n/messages/sv.json @@ -4811,7 +4811,7 @@ "v0VercelWebLabel": "__MISSING__:v0 Vercel Web", "v0VercelWebDesc": "__MISSING__:AI code generation via v0.dev", "kimiWebLabel": "__MISSING__:Kimi Web", - "kimiWebDesc": "__MISSING__:Chinese market AI chat via kimi.moonshot.cn", + "kimiWebDesc": "__MISSING__:Moonshot AI consumer chat via www.kimi.com (international, Connect-RPC API)", "doubaoWebLabel": "__MISSING__:Doubao Web", "doubaoWebDesc": "__MISSING__:ByteDance AI chat via doubao.com" }, diff --git a/src/i18n/messages/zh-CN.json b/src/i18n/messages/zh-CN.json index df4cedc1ad..a3126e466f 100644 --- a/src/i18n/messages/zh-CN.json +++ b/src/i18n/messages/zh-CN.json @@ -4667,7 +4667,7 @@ "v0VercelWebLabel": "v0 Vercel Web", "v0VercelWebDesc": "通过 v0.dev 的 AI 代码生成", "kimiWebLabel": "Kimi Web", - "kimiWebDesc": "通过 kimi.moonshot.cn 访问中国市场的 AI 聊天", + "kimiWebDesc": "通过 www.kimi.com 访问 Moonshot AI 聊天(国际版,Connect-RPC API)", "doubaoWebLabel": "豆包网", "doubaoWebDesc": "字节跳动 AI 聊天通过 doubao.com", "Account Deactivated": "账户已停用", diff --git a/src/lib/providers/validation.ts b/src/lib/providers/validation.ts index de5218a6d5..d2925db557 100644 --- a/src/lib/providers/validation.ts +++ b/src/lib/providers/validation.ts @@ -32,6 +32,7 @@ import { validateChatGptWebProvider, validatePerplexityWebProvider, validateBlackboxWebProvider, + validateKimiWebProvider, } from "./validation/webProvidersA"; import { validateMuseSparkWebProvider, @@ -349,6 +350,7 @@ export async function validateProviderApiKey({ provider, apiKey, providerSpecifi "deepseek-web": validateDeepSeekWebProvider, "grok-web": validateGrokWebProvider, "qwen-web": validateQwenWebProvider, + "kimi-web": validateKimiWebProvider, "chatgpt-web": validateChatGptWebProvider, "perplexity-web": validatePerplexityWebProvider, "blackbox-web": validateBlackboxWebProvider, diff --git a/src/lib/providers/validation/webProvidersA.ts b/src/lib/providers/validation/webProvidersA.ts index c75b8e15d6..5f66320421 100644 --- a/src/lib/providers/validation/webProvidersA.ts +++ b/src/lib/providers/validation/webProvidersA.ts @@ -8,10 +8,85 @@ import { buildGrokCookieHeader, buildQwenCookieHeader, extractCookieValue, + extractKimiJwt, extractQwenToken, normalizeSessionCookieHeader, } from "@/lib/providers/webCookieAuth"; +// kimi-web uses the international `www.kimi.com` Connect-RPC API. The legacy +// `kimi.moonshot.cn` domain now 307-redirects every non-CN visitor, and even +// if you bypass the redirect the old `/api/chat` REST endpoint is gone. The +// SPA exposes a profile probe at `GET /api/user` that returns the user object +// at the top level when the `Authorization: Bearer ` header is valid. +// +// Auth source: the `kimi-auth` cookie set after login. The user pastes the +// full Cookie header; we extract `kimi-auth` and send it as both the Bearer +// token and a `Cookie: kimi-auth=` replay (the latter is what the SPA +// does, though the upstream only consults the Authorization header in +// practice — verified by stripping one of the two at a time). +export async function validateKimiWebProvider({ apiKey }: any) { + const rawCred = String(apiKey ?? "").trim(); + if (!rawCred) { + return { + valid: false, + error: + "Missing Kimi session — paste the full Cookie header from www.kimi.com (must contain kimi-auth=)", + }; + } + + const jwt = extractKimiJwt(rawCred); + if (!jwt) { + return { + valid: false, + error: + "Could not find a kimi-auth JWT in the pasted value. Re-login at https://www.kimi.com and copy the full Cookie header.", + }; + } + + try { + const resp = await fetch("https://www.kimi.com/api/user", { + headers: { + Accept: "application/json, text/plain, */*", + Authorization: `Bearer ${jwt}`, + Cookie: `kimi-auth=${jwt}`, + Origin: "https://www.kimi.com", + Referer: "https://www.kimi.com/", + "User-Agent": + "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/149.0.0.0 Safari/537.36", + }, + }); + + if (resp.status === 401 || resp.status === 403) { + return { + valid: false, + error: + "Kimi session is invalid or expired — re-login at https://www.kimi.com and paste a fresh Cookie header", + }; + } + if (!resp.ok) { + return { valid: false, error: `Kimi returned HTTP ${resp.status}` }; + } + + // Profile response: `{ id, name, email, region, ... }` at the top level. + try { + const data = await resp.json(); + if (!data?.id) { + return { + valid: false, + error: + "Kimi session token is invalid or expired — re-login at https://www.kimi.com and paste a fresh Cookie header", + }; + } + } catch { + return { valid: false, error: "Kimi returned invalid JSON response" }; + } + + return { valid: true, error: null }; + } catch (error) { + return toValidationErrorResult(error); + } +} + export async function validateDeepSeekWebProvider({ apiKey }: any) { if (!apiKey) { return { diff --git a/src/lib/providers/webCookieAuth.ts b/src/lib/providers/webCookieAuth.ts index 5236bc0ab2..37b4655777 100644 --- a/src/lib/providers/webCookieAuth.ts +++ b/src/lib/providers/webCookieAuth.ts @@ -106,6 +106,38 @@ export function extractQwenToken(rawValue: string): string { return match ? match[1] : ""; } +/** + * Pull the `kimi-auth` JWT out of whatever the user pasted for the + * international Kimi consumer chat (www.kimi.com). + * + * Accepts (all return the same JWT string): + * - bare JWT `eyJhbGci...sig` + * - full Cookie header `_ga=...; kimi-auth=eyJ...; theme=dark` + * - `Cookie:` / `Authorization: Bearer` prefixed forms + * - stray `Bearer eyJ...` without a header label + * + * Returns "" if no JWT can be located. + */ +export function extractKimiJwt(rawValue: string): string { + const trimmed = stripCookieInputPrefix(rawValue); + if (!trimmed) return ""; + + // Bare JWT — three base64url segments separated by dots. + if (/^eyJ[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+$/.test(trimmed)) { + return trimmed; + } + + // Cookie-style pair: pull `kimi-auth=` out of the blob. + const match = trimmed.match(/(?:^|[\s;])kimi-auth=([^;\s]+)/); + if (match) return match[1]; + + // Last resort: a `Bearer ` pasted without the header label. + const bearer = trimmed.match(/bearer\s+(eyJ[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+)/i); + if (bearer) return bearer[1]; + + return ""; +} + export function normalizeSessionCookieHeaders( rawValues: Array, defaultCookieName: string diff --git a/src/shared/constants/providers/web-cookie.ts b/src/shared/constants/providers/web-cookie.ts index 337d53e23f..d6b56c43a1 100644 --- a/src/shared/constants/providers/web-cookie.ts +++ b/src/shared/constants/providers/web-cookie.ts @@ -242,8 +242,8 @@ export const WEB_COOKIE_PROVIDERS = { icon: "auto_awesome", color: "#2563EB", textIcon: "KW", - website: "https://kimi.moonshot.cn", - authHint: "Paste your session cookie from kimi.moonshot.cn (DevTools → Application → Cookies)", + website: "https://www.kimi.com", + authHint: "Paste your Cookie header from www.kimi.com (must contain kimi-auth=...). Find it via DevTools → Network → request → Cookie.", subscriptionRisk: true, riskNoticeVariant: "webCookie", }, diff --git a/src/shared/providers/webSessionCredentials.ts b/src/shared/providers/webSessionCredentials.ts index 800f874d00..68064679e0 100644 --- a/src/shared/providers/webSessionCredentials.ts +++ b/src/shared/providers/webSessionCredentials.ts @@ -164,10 +164,10 @@ export const WEB_SESSION_CREDENTIAL_REQUIREMENTS = { }, "kimi-web": { kind: "cookie", - credentialName: "session", - placeholder: "session=... or full Cookie header from kimi.moonshot.cn", + credentialName: "kimi-auth", + placeholder: "kimi-auth=eyJ... (full Cookie header from www.kimi.com)", acceptsFullCookieHeader: true, - storageKeys: ["cookie", "session"], + storageKeys: ["cookie", "kimi-auth", "session"], }, "doubao-web": { kind: "cookie", diff --git a/tests/snapshots/provider/translate-path.json b/tests/snapshots/provider/translate-path.json index 9467e78189..128567dece 100644 --- a/tests/snapshots/provider/translate-path.json +++ b/tests/snapshots/provider/translate-path.json @@ -2277,23 +2277,27 @@ "format": "openai", "headers": { "apiKey": { - "Accept": "text/event-stream", + "Accept": "*/*", "Authorization": "Bearer ", - "Content-Type": "application/json" + "Content-Type": "application/connect+json", + "connect-protocol-version": "1" }, "nonStream": { + "Accept": "*/*", "Authorization": "Bearer ", - "Content-Type": "application/json" + "Content-Type": "application/connect+json", + "connect-protocol-version": "1" }, "oauth": { - "Accept": "text/event-stream", + "Accept": "*/*", "Authorization": "Bearer ", - "Content-Type": "application/json" + "Content-Type": "application/connect+json", + "connect-protocol-version": "1" } }, "url": { - "nonStream": "https://kimi.moonshot.cn/api/chat", - "stream": "https://kimi.moonshot.cn/api/chat" + "nonStream": "https://www.kimi.com/apiv2/kimi.gateway.chat.v1.ChatService/Chat", + "stream": "https://www.kimi.com/apiv2/kimi.gateway.chat.v1.ChatService/Chat" } }, "kiro": { diff --git a/tests/unit/executor-kimi-web-decoder.test.ts b/tests/unit/executor-kimi-web-decoder.test.ts new file mode 100644 index 0000000000..e6d364bae1 --- /dev/null +++ b/tests/unit/executor-kimi-web-decoder.test.ts @@ -0,0 +1,239 @@ +// Tests for the Connect frame decoder and event-delta extractor that back +// the international Kimi web executor (www.kimi.com Connect-RPC API). +// +// These tests pin the wire-format parsing that the executor relies on — +// the riskiest piece of the migration per code review (PR #5858, I3). + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; + +const { + frameConnectMessage, + decodeConnectFrame, + extractDelta, + isEndOfStream, + foldMessages, +} = await import("../../open-sse/executors/kimi-web.ts"); + +describe("frameConnectMessage + decodeConnectMessage round-trip", () => { + it("round-trips a JSON payload through frame and decode", () => { + const json = '{"hello":"world"}'; + const framed = frameConnectMessage(json); + assert.equal(framed.length, 5 + json.length); + // First byte is flags = 0 (uncompressed). + assert.equal(framed[0], 0); + // Bytes 1-4 are big-endian length. + const len = (framed[1] << 24) | (framed[2] << 16) | (framed[3] << 8) | framed[4]; + assert.equal(len, json.length); + + const { consumed, frame } = decodeConnectFrame(framed, 0); + assert.equal(consumed, framed.length); + assert.equal(frame?.flags, 0); + assert.deepEqual(frame?.message, { hello: "world" }); + }); + + it("returns consumed=0 when the buffer has fewer than 5 bytes (need more)", () => { + const short = new Uint8Array([0x00, 0x00, 0x00]); + const { consumed, frame } = decodeConnectFrame(short, 0); + assert.equal(consumed, 0); + assert.equal(frame, null); + }); + + it("returns consumed=0 when the buffer has header but not enough payload yet", () => { + // Header claims 100 bytes of payload, but we only have 5 header + 10 payload. + const partial = new Uint8Array(15); + partial[0] = 0; + partial[1] = 0; + partial[2] = 0; + partial[3] = 0; + partial[4] = 100; + const { consumed, frame } = decodeConnectFrame(partial, 0); + assert.equal(consumed, 0); + assert.equal(frame, null); + }); + + it("consumes the first frame and leaves the rest in the buffer for the next call", () => { + const a = frameConnectMessage('{"a":1}'); + const b = frameConnectMessage('{"b":2}'); + const merged = new Uint8Array(a.length + b.length); + merged.set(a, 0); + merged.set(b, a.length); + + const first = decodeConnectFrame(merged, 0); + assert.equal(first.consumed, a.length); + assert.deepEqual(first.frame?.message, { a: 1 }); + + const second = decodeConnectFrame(merged, first.consumed); + assert.equal(second.consumed, b.length); + assert.deepEqual(second.frame?.message, { b: 2 }); + }); + + it("decodes a frame whose length has the high bit (bit 31) set without sign issues", () => { + // Construct a header claiming length 2,147,483,648 (0x80000000) — the + // signed-shift bug would read this as -2147483648. With the decoder's + // correction it should be treated as MAX_FRAME_LEN+1 and consumed=-1. + const oversized = new Uint8Array(5); + oversized[0] = 0; + oversized[1] = 0x80; + oversized[2] = 0x00; + oversized[3] = 0x00; + oversized[4] = 0x00; + const { consumed } = decodeConnectFrame(oversized, 0); + assert.equal(consumed, -1, "frames above MAX_FRAME_LEN must signal -1"); + }); + + it("returns a null message (still consumed) when payload is not valid JSON", () => { + const bad = new Uint8Array(5 + 3); + bad[0] = 0; + bad[4] = 3; + bad[5] = 0x7b; // { + bad[6] = 0x7d; // } + bad[7] = 0x2c; // , (trailing — invalid JSON) + const { consumed, frame } = decodeConnectFrame(bad, 0); + assert.equal(consumed, 8); + assert.equal(frame?.message, null); + assert.equal(frame?.flags, 0); + }); +}); + +describe("extractDelta", () => { + it("returns null on null/empty input", () => { + assert.equal(extractDelta(null), null); + }); + + it("returns null on heartbeats and unrelated events", () => { + assert.equal(extractDelta({ heartbeat: {} }), null); + assert.equal(extractDelta({ op: "set", mask: "chat.name" }), null); + assert.equal(extractDelta({ op: "set", mask: "block.stage" }), null); + }); + + it("extracts initial answer text from op=set, mask=block.text", () => { + const delta = extractDelta({ + op: "set", + mask: "block.text", + block: { text: { content: "Hello" } }, + }); + assert.deepEqual(delta, { kind: "text", text: "Hello" }); + }); + + it("extracts answer delta from op=append, mask=block.text.content", () => { + const delta = extractDelta({ + op: "append", + mask: "block.text.content", + block: { text: { content: " world" } }, + }); + assert.deepEqual(delta, { kind: "text", text: " world" }); + }); + + it("extracts initial reasoning from op=set, mask=block.think", () => { + const delta = extractDelta({ + op: "set", + mask: "block.think", + block: { think: { content: "Reasoning..." } }, + }); + assert.deepEqual(delta, { kind: "think", text: "Reasoning..." }); + }); + + it("extracts reasoning delta from op=append, mask=block.think.content", () => { + const delta = extractDelta({ + op: "append", + mask: "block.think.content", + block: { think: { content: " continued" } }, + }); + assert.deepEqual(delta, { kind: "think", text: " continued" }); + }); + + it("returns null when content is empty (no useful delta)", () => { + assert.equal( + extractDelta({ op: "set", mask: "block.text", block: { text: { content: "" } } }), + null + ); + assert.equal( + extractDelta({ op: "append", mask: "block.text.content", block: { text: {} } }), + null + ); + }); +}); + +describe("isEndOfStream", () => { + it("returns true when assistant message flips to MESSAGE_STATUS_COMPLETED", () => { + assert.equal( + isEndOfStream({ + op: "set", + mask: "message", + message: { role: "assistant", status: "MESSAGE_STATUS_COMPLETED" }, + }), + true + ); + }); + + it("returns false for non-assistant completed messages (system/user)", () => { + assert.equal( + isEndOfStream({ + op: "set", + mask: "message", + message: { role: "user", status: "MESSAGE_STATUS_COMPLETED" }, + }), + false + ); + }); + + it("returns false for assistant messages that are still generating", () => { + assert.equal( + isEndOfStream({ + op: "set", + mask: "message", + message: { role: "assistant", status: "MESSAGE_STATUS_GENERATING" }, + }), + false + ); + }); + + it("returns false for non-message events", () => { + assert.equal(isEndOfStream({ heartbeat: {} }), false); + assert.equal(isEndOfStream(null), false); + }); +}); + +describe("foldMessages", () => { + it("returns empty string for empty input", () => { + assert.equal(foldMessages([]), ""); + }); + + it("returns user content as-is when only a user message is present", () => { + assert.equal(foldMessages([{ role: "user", content: "hi" }]), "hi"); + }); + + it("prepends system content to user content", () => { + const out = foldMessages([ + { role: "system", content: "Be terse." }, + { role: "user", content: "hi" }, + ]); + assert.equal(out, "Be terse.\n\nhi"); + }); + + it("labels assistant turns and concatenates with prior user content", () => { + const out = foldMessages([ + { role: "user", content: "q1" }, + { role: "assistant", content: "a1" }, + { role: "user", content: "q2" }, + ]); + assert.equal(out, "q1\n\nAssistant: a1\n\nq2"); + }); + + it("stringifies non-string content (arrays/objects) instead of dropping it", () => { + const out = foldMessages([{ role: "user", content: [{ type: "text", text: "x" }] }]); + assert.ok(out.includes("text")); + assert.ok(out.includes("x")); + }); + + it("silently drops tool/function messages (limitation: kimi-web is single-turn)", () => { + const out = foldMessages([ + { role: "user", content: "hi" }, + { role: "tool", content: "result" }, + { role: "function", content: "fn-result" }, + ]); + // Tool/function messages contribute nothing; user content survives. + assert.equal(out, "hi"); + }); +}); diff --git a/tests/unit/executor-kimi-web.test.ts b/tests/unit/executor-kimi-web.test.ts index e2515e9391..05f6e52f34 100644 --- a/tests/unit/executor-kimi-web.test.ts +++ b/tests/unit/executor-kimi-web.test.ts @@ -1,3 +1,10 @@ +// Tests for the international Kimi web executor (www.kimi.com Connect-RPC API). +// +// Previously this provider targeted kimi.moonshot.cn; that domain now redirects +// every non-CN visitor to www.kimi.com, which uses a Connect-RPC streaming API. +// These tests pin the parser behavior of the Connect envelope framing and the +// JSON event-delta extractor. + import { describe, it } from "node:test"; import assert from "node:assert/strict"; @@ -9,20 +16,77 @@ describe("KimiWebExecutor", () => { assert.ok(executor); }); - it("execute returns error on fetch failure", async () => { + it("execute returns a 400 error when no JWT is provided", async () => { const executor = new mod.KimiWebExecutor(); + const result = await executor.execute({ + model: "kimi-default", + body: { messages: [{ role: "user", content: "hi" }] }, + stream: false, + credentials: { apiKey: "" }, + signal: null, + } as never); + assert.equal(result.response.status, 400); + const body = (await result.response.json()) as { error: { code: string } }; + assert.match(body.error.code, /HTTP_400|400/); + }); + + it("execute targets www.kimi.com (not kimi.moonshot.cn)", async () => { + const executor = new mod.KimiWebExecutor(); + let capturedUrl = ""; + const originalFetch = globalThis.fetch; try { - const result = await executor.execute({ + globalThis.fetch = (async (url: any) => { + capturedUrl = String(url); + return new Response(new ReadableStream({ start: (c) => c.close() }), { + status: 200, + headers: { "content-type": "application/connect+json" }, + }); + }) as typeof fetch; + await executor.execute({ model: "kimi-default", body: { messages: [{ role: "user", content: "hi" }] }, stream: false, - credentials: { apiKey: "" }, + credentials: { apiKey: "kimi-auth=fake.jwt.token" }, signal: null, - }); - assert.ok(result.response instanceof Response); - assert.ok(result.url.includes("kimi.moonshot.cn")); - } catch { - // Network error expected + } as never); + assert.ok(capturedUrl.startsWith("https://www.kimi.com/"), `got ${capturedUrl}`); + assert.ok(!capturedUrl.includes("moonshot.cn")); + } finally { + globalThis.fetch = originalFetch; } }); }); + +describe("extractKimiJwt", () => { + const { extractKimiJwt } = mod; + + it("returns empty string for empty input", () => { + assert.equal(extractKimiJwt(""), ""); + assert.equal(extractKimiJwt(" "), ""); + }); + + it("extracts a bare JWT", () => { + const jwt = "eyJhbGci.eyJzdWIi.c2ln"; + assert.equal(extractKimiJwt(jwt), jwt); + }); + + it("extracts kimi-auth from a full Cookie header", () => { + const jwt = "eyJhbGciOiJIUzUxMiIsInR5cCI6IkpXVCJ9.eyJzdWIiOiJ4In0.signature"; + const pasted = `_ga=GA1.1.x; theme=dark; kimi-auth=${jwt}; _gcl_au=1.1.x; lang=en-US`; + assert.equal(extractKimiJwt(pasted), jwt); + }); + + it("strips a leading Cookie: header label", () => { + const jwt = "eyJhbGci.eyJzdWIi.c2ln"; + assert.equal(extractKimiJwt(`Cookie: kimi-auth=${jwt}`), jwt); + }); + + it("strips a leading Authorization: Bearer label", () => { + const jwt = "eyJhbGci.eyJzdWIi.c2ln"; + assert.equal(extractKimiJwt(`Authorization: Bearer ${jwt}`), jwt); + }); + + it("returns empty when no JWT is present", () => { + assert.equal(extractKimiJwt("foo=bar; baz=qux"), ""); + }); +}); diff --git a/tests/unit/web-cookie-providers-new.test.ts b/tests/unit/web-cookie-providers-new.test.ts index 1827b31640..1a9d340484 100644 --- a/tests/unit/web-cookie-providers-new.test.ts +++ b/tests/unit/web-cookie-providers-new.test.ts @@ -666,27 +666,45 @@ test("v0 Vercel Web: error response returns error result", async () => { // ── Kimi Web Execution Tests ───────────────────────────────────────────────── -test("Kimi Web: streaming passes through SSE", async () => { - const sseData = ['data: {"choices":[{"delta":{"content":"你好"}}]}']; - const restore = mockFetchCapture(200, mockSSEStream(sseData)); +test("Kimi Web: targets www.kimi.com (international)", async () => { + // The new executor talks to the Connect-RPC streaming endpoint on the + // international domain. A bare empty credential is rejected before the + // fetch fires, so we feed a fake JWT and let the mock absorb the request. + const restore = mockFetchCapture(200); try { const executor = new KimiWebExecutor(); const result = await executor.execute({ ...noopExecuteInput, model: "kimi-default", + credentials: { apiKey: "kimi-auth=eyJ.eyJzdWI.signature" }, }); assert.ok(result.response instanceof Response); - assert.ok(result.url.includes("kimi.moonshot.cn")); + assert.ok(result.url.includes("www.kimi.com"), `got ${result.url}`); + assert.ok(!result.url.includes("moonshot.cn")); } finally { restore.restore(); } }); +test("Kimi Web: missing JWT returns a 400 before fetching", async () => { + const executor = new KimiWebExecutor(); + const result = await executor.execute({ + ...noopExecuteInput, + model: "kimi-default", + credentials: { apiKey: "" }, + }); + assert.equal(result.response.status, 400); +}); + test("Kimi Web: error response returns error result", async () => { const restore = mockFetchCapture(401, "Unauthorized"); try { const executor = new KimiWebExecutor(); - const result = await executor.execute(noopExecuteInput); + const result = await executor.execute({ + ...noopExecuteInput, + model: "kimi-default", + credentials: { apiKey: "kimi-auth=eyJ.eyJzdWI.signature" }, + }); assert.ok(result.response instanceof Response); assert.equal(result.response.status, 401); } finally {