mirror of
https://github.com/diegosouzapw/OmniRoute.git
synced 2026-07-31 04:12:10 +03:00
* fix(gemini): preserve structured tool calls for antigravity * fix(gemini): parse prefixed textual tool calls * fix(antigravity): preserve textual SSE tool calls * fix(stream): normalize textual passthrough tool calls * fix(stream): normalize split textual tool calls * fix(stream): suppress malformed textual tool calls * fix(stream): suppress compact malformed tool calls * fix(stream): emit structured textual tool calls * fix(stream): suppress unknown textual tool calls * fix(stream): normalize responses textual tool calls * chore: ignore .claude/settings.local.json (per-user Claude Code permissions) * fix(opencode-go): route qwen3.x via claude messages + repair fixMissingToolResponses for Claude-shape upstreams (#2791) Integrated into release/v3.8.6 * fix: resolve npm install warnings — remove dead deps, relax engine constraint (#2792) Integrated into release/v3.8.6 * fix: register missing web-cookie validators (claude-web, gemini-web, copilot-web, t3-web) (#2793) Integrated into release/v3.8.6 * fix: Error: Unable to inspect existing database #2771 (#2795) Integrated into release/v3.8.6 * fix(oauth): repair Google loopback callback flow (#2796) Integrated into release/v3.8.6 * feat(logs): add clean history button (#2799) Integrated into release/v3.8.6 * [codex] home: restore settings-driven home layout and quota auto-refresh (#2800) Integrated into release/v3.8.6 * fix(gemini): emit signaturelessToolCallMode:text for GEMINI format models (#2801) Integrated into release/v3.8.6 * feat(modelSpecs): align opencode-go family with upstream provider limits (#2802) Integrated into release/v3.8.6 * chore: apply unit test fixes, polyfills, and environment precedence fixes * docs(agents): atualiza fluxos de release e triagem Expande os workflows de release para incluir auditoria de segurança, CHANGELOG completo por commits, quality gate obrigatório, homologação em VPS local, publicação oficial, deploy em Akamai e validação de artefatos. Reorganiza a triagem de features com arquivos permanentes por bucket, suporte a itens em andamento, regra de reclaim após 15 dias e novo tratamento para ideias viáveis catalogadas. Corrige a orientação de revisão de discussões para usar a ordem cronológica real dos comentários e respostas ao identificar a última atividade. * fix(lockout): classify Gemini Antigravity resource exhaustion as quota_exhausted * fix(reasoning): gate replay by interleaved field * docs(rule-16): permit human Co-authored-by, restrict only AI/bot trailers Rule #16 previously banned all `Co-Authored-By` trailers absolutely. That blocked the upstream-port workflows (`/port-upstream-features` and `/port-upstream-issues`), which must credit human upstream PR authors and issue reporters in OmniRoute commits. Refine the rule to ban only AI/bot-attributed trailers (Claude, GPT, Copilot, Bot; anthropic.com / openai.com / bot-owned noreply.github.com emails) while allowing standard human `Co-authored-by: Name <email>` attribution. Sync the rule across the source CLAUDE.md, the E2E shakedown doc note, and 41 i18n translations. * fix(gitlawb): add specialty validators for connection test — bypass /models probe GitLawB OpenGateway API (xiaomi-mimo compatible) does not expose a /models endpoint, causing validateOpenAILikeProvider to 404 on the initial probe and report 'Provider validation endpoint not supported'. Add specialty validators for both gitlawb and gitlawb-gmi that follow the same pattern as the existing xiaomi-mimo validator: skip GET /models, validate directly via POST /chat/completions with a minimal test message. Any 401/403 response means an invalid key; all other responses mean auth is OK. Fixes test-connection returning 404 for GitLawB providers. * test(gitlawb): add 12 unit tests for gitlawb and gitlawb-gmi specialty validators Covers success, auth failure (401/403), non-auth acceptance (400/422/429), network errors, and custom baseUrl overrides for both providers. * feat(gitlawb): serve models from static registry without API-unavailable warning GitLawB's OpenGateway API does not expose a /models endpoint per provider-path. Previously the models route fell through to the generic fallback which returned static catalog models with the misleading 'API unavailable — using local catalog' warning. Now gitlawb and gitlawb-gmi are handled as static model providers (same pattern as reka and qwen OAuth) — models are served from the provider registry without any warning, since all registered models are functional via POST /chat/completions. * refactor(gitlawb): extract shared opengateway validator factory, fix docs path in test - Extract gitlawb/gitlawb-gmi validators into buildOpengatewayValidator factory - Fix dockerignore-docs-coverage test: update stale docs/AUTO-COMBO.md -> docs/routing/AUTO-COMBO.md * fix(reasoning): guard interleaved capability lookup * feat(gitlawb): dynamic model fetch with gmi-cloud fallback Hybrid approach: - gitlawb (xiaomi-mimo): dynamic /models endpoint → 356 models - gitlawb-gmi (gmi-cloud): 404 fallback → local catalog gracefully Mimics Gitlawb/openclaude's model-routing pattern * i18n(pt-BR): complete missing translations and sync with en.json * feat(build): nix multi-OS package manager install (#2806) Integrated into release/v3.8.6 * fix(i18n): translate 144 new __MISSING__ pt-BR strings (#2816) Integrated into release/v3.8.6 * chore(docs): set coverage gate to 40/40/40/40 in CLAUDE.md Aligns the documented coverage gate with the v3.8.6 release decision (lowered from 75/75/75/70). Matches the threshold already set in package.json by the large feature PRs (planos 11-22). * fix(cli): respect PORT env var in serve command (#2845) Integrated into release/v3.8.6. * fix(deepseek-web): return 400 when client sends tools[] - chat.deepseek.com has no tool support (#2854) Integrated into release/v3.8.6. * fix(qoder): reject invalid/expired PATs returning Cosy 500 error (#2860) Integrated into release/v3.8.6. * fix(cli): register openclaw in tool-detector (#2833) (#2850) Integrated into release/v3.8.6. * fix(api): include noAuth providers in /v1/models catalog (#2798) (#2814) Integrated into release/v3.8.6. * fix(combo): resolve custom provider targets via combo name (#2778) (#2812) Integrated into release/v3.8.6. * fix(translator): strip safety_identifier in openai-responses cleanup (#2770) (#2809) Integrated into release/v3.8.6. * fix(quota): honor explicit per-connection preflight opt-out (#2831) (#2844) Integrated into release/v3.8.6. * fix(usage): un-invert GitHub Copilot Free/limited quota — limited_user_quotas is remaining (#2876) (#2881) Integrated into release/v3.8.6. * fix(nous-research): correct baseUrl to include /chat/completions (#2826) (#2835) Integrated into release/v3.8.6. * fix(opencode): qwen3.x max/plus models lack vision support (#2822) (#2836) Integrated into release/v3.8.6. * fix(translator): pass-through tool_search built-in tool type (#2766) (#2811) Integrated into release/v3.8.6. * fix(github): route claude-opus-4.6 via chat completions (#2821) Integrated into release/v3.8.6. * docs(oauth): add Windsurf login fix design (Phase 1 hotfix + Phase 2 Firebase OAuth) Two-phase plan to fix the broken Windsurf OAuth flow: - Phase 1: drop the dead app.devin.ai/editor/signin PKCE path, promote import-token from windsurf.com/show-auth-token as the primary path - Phase 2: port Firebase OAuth + RegisterUser flow from fendoushaonian/WindSurf-gRPC-API for full browser-based automation Spec only - no code changes yet. * docs(plan): Phase 1 windsurf login hotfix implementation plan 10 tasks covering: - TDD assertions for flowType + 410 Gone responses - Provider switch to import_token - Route handler retiring authorize/start-callback-server/poll-callback - OAuthModal UI override - i18n sync - Verification + PR steps * fix(cli): replace cli-table3 with hand-rolled formatter (#2752) (#2813) Integrated into release/v3.8.6. * fix(skills): skip interception for unregistered client-native tools (#2815) (#2817) Integrated into release/v3.8.6. * feat(sse): add RTK filters for kubectl, docker-build, composer, gh (#2824) Integrated into release/v3.8.6. * fix(geminiHelper): support rec.image content shape + warn on dropped remote URLs (refs #2807) (#2855) Integrated into release/v3.8.6. * fix(cli): allow nullable/optional apiKey in cliMitmStartSchema (#2857) Integrated into release/v3.8.6. * fix(combo): preserve system messages during context handoff summary generation (#2865) Integrated into release/v3.8.6. * fix: wire CLIProxyAPI fallback settings into chatCore routing engine (#2866) Integrated into release/v3.8.6. * fix(usage): add opencode quota fetcher (#2852) (#2867) Integrated into release/v3.8.6. * feat(claude): default xhigh support for newer Opus models (#2874) Integrated into release/v3.8.6. * fix(cli): restore omniroute logs command stream (#2756) (#2810) Integrated into release/v3.8.6. * fix(combo): normalize upstream Headers for Node 24 undici interop (#2751) (#2823) Integrated into release/v3.8.6. * Rename proxy log Public IP to Client IP (#2880) Integrated into release/v3.8.6. * fix(claude): preserve max effort for supported models (#2875) Integrated into release/v3.8.6. * fix(oauth): switch windsurf provider to import_token flow The PKCE auth URL targeting app.devin.ai/editor/signin returns 404 post-rebrand. Until Phase 2 ports Firebase OAuth + RegisterUser, the only supported path is import-token via windsurf.com/show-auth-token. - windsurf.ts: drop buildAuthUrl, set flowType=import_token - generateAuthData returns supported:false + helpful error for windsurf/devin-cli - tests: assert flowType + disabled stub * fix(oauth): return 410 Gone for retired windsurf/devin-cli PKCE actions start-callback-server, authorize, and poll-callback (GET + POST) now return 410 Gone with a pointer to /import-token. The 410 short-circuit runs before auth so the response is honest about the action being permanently gone, not gated. Codex PKCE flow unchanged. Tests: 5 new assertions cover GET + POST 410 paths and a Codex regression check. * refactor(oauth): annotate retired PKCE fields in WINDSURF_CONFIG No behaviour change - comment-only update documenting that authorizeUrl, codeChallengeMethod, callbackPort, callbackPath, apiServerUrl, and exchangePath are no longer consumed. Active fields (inferenceUrl, showAuthTokenUrl, firebaseApiKey, ideName) called out separately. * fix(cli,docs): use requireCliToolsAuth in logs route + document OPENCODE quota env Post-merge contract fixes for v3.8.6: - src/app/api/cli-tools/logs/route.ts (#2810) now uses the shared requireCliToolsAuth guard (param renamed req->request) to satisfy the cli-tools-auth-hardening contract test. - Document OMNIROUTE_OPENCODE_QUOTA_URL (#2867) in docs/reference/ENVIRONMENT.md to satisfy the env/docs sync contract. * fix(dashboard): force import-token panel for windsurf/devin-cli Phase 1 hotfix: hide the 'Browser Login' tab and start in Paste API Key mode. Removes windsurf/devin-cli from PKCE_CALLBACK_SERVER_PROVIDERS so no callback server is started for them. Codex still uses the PKCE flow. The 'Get token' link continues to point at windsurf.com/show-auth-token via the existing supportsTokenPaste form copy. * fix(oauth): windsurf import-token mapTokens signature mismatch The route at `src/app/api/oauth/[provider]/[action]/route.ts` invokes `providerData.mapTokens({ accessToken: token })` (object), matching the cursor/kiro signature. The windsurf provider was declared with `mapTokens(token: string)` instead, so the entire object was stored as `accessToken`. When the connection record reached the SQLite layer it crashed with: SQLite3 can only bind numbers, strings, bigints, buffers, and null Fix by aligning windsurf's `mapTokens` signature with the route caller and the cursor/kiro convention. Also dedupe a copy-pasted second `if (action === "import-token")` block in the route handler — the second block was unreachable but identical to the first. Adds two regression tests asserting that `provider.mapTokens({ accessToken })` returns a string `accessToken` for both windsurf and devin-cli, so a future signature drift trips the gate instead of the SQLite bind error in production. * feat(compression): expand pt-BR pack with troglodita rules (15 → 49) (#2818) Integrated into release/v3.8.6 * fix(sse): repair RTK engine defaults so dedup and direct calls work (#2825) Integrated into release/v3.8.6 * fix(mcp): redirect console.log/warn to stderr in --mcp stdio mode (#2840) Integrated into release/v3.8.6 * fix(gemini-cli): prefer real project IDs over default-project (#2841) Integrated into release/v3.8.6 * fix(opencode-go): add provider limits quota fetcher (#2861) Integrated into release/v3.8.6 * Audit & add web cookie providers: fix 4 missing registry entries + DuckDuckGo (#2862) Integrated into release/v3.8.6 * fix(antigravity): harden signatureless tool history (#2878) Integrated into release/v3.8.6 * fix: provider model sync pruning and dynamic antigravity MITM proxy mappings (#2886) Integrated into release/v3.8.6 * feat(usage): per-API-key token limits scoped to model/provider/global (#2888) Integrated into release/v3.8.6 * fix(audio): build multipart body manually to preserve Content-Type (#2842) Integrated into release/v3.8.6 * refactor: remove agent skill documentation files and streamline maintenance workflows * test(stabilization): resolve unit test failures in blackbox-web, schema-coercion, translator-helper-branches, usage-service-hardening, and audio-transcription * fix(security): mitigate Socket.dev supply-chain findings + secrets opt-in + minimal build profile (#2863) (#2871) Two real security gaps closed and four cosmetic Socket.dev fingerprints removed. See docs/security/SOCKET_DEV_FINDINGS.md for the per-finding maintainer attestation. Real bugs fixed: - cloudSync: HMAC verification of `X-Cloud-Sig` + opt-in `OMNIROUTE_CLOUD_SYNC_SECRETS=true` before overwriting `accessToken` / `refreshToken` / `providerSpecificData` from a remote response. Closes the silent-credential-swap surface (a misconfigured or hostile CLOUD_URL could previously replace local tokens unverified). - Zed import: split into 2-step `/discover` + `/import` flow. `/import` now requires `confirmedAccounts: [{ service, account, fingerprint }]` and re-reads the keychain server-side to filter by fingerprint, so a tampered discover response cannot trick the endpoint into saving an unrelated token. Cosmetic Socket.dev mitigations: - runElevatedPowerShell writes the elevated payload to a per-call temp `.ps1` file (mode 0o600) and references it via `-File`. Removes the textbook `-EncodedCommand <base64utf16le>` pattern flagged as malware by Socket's AI classifier. - Maintainer attestation `SECURITY-AUDITOR-NOTE:` blocks added at every flagged call site pointing to `docs/security/SOCKET_DEV_FINDINGS.md`. Build-time hardening: - `OMNIROUTE_BUILD_PROFILE=minimal` (`npm run build:secure`) physically removes the four sensitive modules from the standalone bundle via webpack `NormalModuleReplacementPlugin`. Stubs throw `FeatureDisabledError` at runtime. Intended for the `omniroute-secure` artifact. Tests: - 24 new unit tests in `tests/unit/security/` covering the wrapper builder, HMAC verification (4 cases), credential fingerprint determinism (5 cases), confirmedAccounts validation + fingerprint filtering (6 cases), and the minimal-build stubs (5 cases). Docs: - New `docs/security/SOCKET_DEV_FINDINGS.md` — per-finding attestation. - New `socket.yml` — Socket.dev v2 config pointing at the attestation. - Updated `SECURITY.md` — supply-chain scanner section. - Updated `.env.example` — three new env vars documented. Backwards compatibility: - Cloud sync token overwrite is OFF by default. Users who relied on it must set `OMNIROUTE_CLOUD_SYNC_SECRETS=true`. Breaking change documented in CHANGELOG. - Zed import 2-step is the new default; legacy 1-step preserved behind `OMNIROUTE_ZED_IMPORT_LEGACY_ONE_STEP=true` and will be removed in v3.9. Closes #2863 * fix(security): redact public Firebase Web key from windsurf spec; doc SHA-256 cache-key rationale (#2894) Two security-scanning findings on release/v3.8.6: - Secret-scanning alert 7 (google_api_key): the windsurf login-fix design spec embedded the literal public Firebase Web API key on two lines. Firebase Web API keys are non-sensitive by design (they identify the project; access is gated by Firebase Security Rules + key restrictions), but the literal trips secret scanning. Redacted to a placeholder; the embedded default still goes through resolvePublicCred per rule #11. - Code-scanning alert 261 (js/insufficient-password-hash): tokenCacheKey() uses SHA-256 to derive an in-memory cache key from the session token, not for password-at-rest storage. Added a comment documenting why CWE-916 KDFs do not apply (false positive). * fix(ci): resolve release/v3.8.6 gate failures (docs-sync, any-budget, pack-artifact) (#2895) * fix(ci): resolve release/v3.8.6 gate failures (docs-sync, any-budget, pack-artifact) Three CI gates failed on release/v3.8.6 (run 26630300877): - docs-sync: CHANGELOG had a spurious "## [3.8.6-patch]" section above "## [3.8.6]", so the latest release no longer matched package.json (3.8.6) and the 41 i18n CHANGELOG mirrors were flagged as missing that section. Fold the lone #2752 entry into [3.8.6] and drop the patch heading. - any-budget:t11: open-sse/handlers/chatCore.ts regressed to 1 explicit `any` (budget 0). Type the persist callback arg as Record<string, unknown>, which matches runWithOnPersist's RefreshPersistFn contract exactly. - pack-artifact: open-sse/utils/setupPolyfill.ts ships via package.json "files" (bin/omniroute.mjs imports it at startup) but was missing from the pack policy allowlist. Allow it and add a regression test. * fix(security): redact public Firebase Web key from windsurf spec Redact the literal public Firebase Web API key (secret-scanning #7) to a placeholder, mirroring the redaction on release/v3.8.6 (PR #2894) and the windsurf fix branch. Non-sensitive public Web key; trips secret scanning. * feat(combo): Zero-Latency Combos (Hedging, Proactive Compression, Predictive TTFT) (#2868) * feat(combo): implement zero-latency combo optimizations (hedging, proactive compression, predictive TTFT) * fix(combo): fix predictive TTFT skip logic and unhandled promise rejections --------- Co-authored-by: Automation <automation@omniroute> * feat: implement automated skill workflows and update system configuration and validation schemas * test: eliminate dynamic cast warnings in cloud-sync unit test * test: isolate services-branch-hardening database directory to avoid concurrency issues * feat(providers): add 7 new web-cookie providers + research catalog + discovery tool New providers: - huggingchat: free LLM chat via huggingface.co/chat (no subscription) - phind: free dev-focused AI chat via phind.com/api/agent - poe-web: multi-model chat via poe.com GraphQL (p-b cookie) - venice-web: privacy-focused AI chat via venice.ai (session cookie) - v0-vercel-web: Vercel v0 code gen via v0.dev (session cookie) - kimi-web: Moonshot Kimi chat via kimi.moonshot.cn (session cookie) - doubao-web: ByteDance Doubao chat via doubao.com (session cookie) Additional: - Research catalog: docs/research/UNLIMITED_LLM_ACCESS.md - Discovery tool design + stub: src/lib/discovery/ + migration 073 - Unit tests: 33 tests for all 7 providers - Shared helpers consolidated in error.ts (slop cleanup) - All registered in WEB_COOKIE_PROVIDERS + providerRegistry + webSessionCredentials Closes #2885 * fix(typecheck): resolve typecheck errors in combo spec and compression modules * feat(api,oauth): add `agy` (Antigravity CLI) standalone provider with CLI token import (#2899) Add a standalone OAuth provider `agy` (Antigravity CLI) next to gemini-cli/antigravity. It reuses the antigravity inference backend (identical Google client_id + daily-cloudcode-pa.googleapis.com endpoint, executor and token-refresh) but ships its own model catalog — including the Claude models the backend exposes (claude-opus-4-6-thinking, claude-sonnet-4-6) — its own account pool, and four ways to connect: - token-file import (paste/upload the agy oauth token JSON) - auto-detect a local CLI login (~/.gemini/antigravity-cli/antigravity-oauth-token) - browser OAuth (via the shared OAuthModal Google loopback flow) - bulk / ZIP import New routes: POST /api/providers/agy-auth/{import,import-bulk,zip-extract,apply-local}. Catalog pinned from the live :fetchAvailableModels endpoint. Docs (openapi.yaml, ENVIRONMENT.md, .env.example, CHANGELOG) updated; new unit tests for registration, the token parser, and route auth-hardening. * fix(security): redact public Firebase Web key from windsurf spec (#2896) Redact the literal public Firebase Web API key (secret-scanning #7) to a placeholder. Firebase Web API keys are non-sensitive by design but the literal trips GitHub secret scanning. Mirrors the redaction landed on release/v3.8.6 (PR #2894). Embedded default still flows through resolvePublicCred (rule #11). * Pr 2871 (#2897) * fix(security): mitigate Socket.dev supply-chain findings + secrets opt-in + minimal build profile (#2863) Two real security gaps closed and four cosmetic Socket.dev fingerprints removed. See docs/security/SOCKET_DEV_FINDINGS.md for the per-finding maintainer attestation. Real bugs fixed: - cloudSync: HMAC verification of `X-Cloud-Sig` + opt-in `OMNIROUTE_CLOUD_SYNC_SECRETS=true` before overwriting `accessToken` / `refreshToken` / `providerSpecificData` from a remote response. Closes the silent-credential-swap surface (a misconfigured or hostile CLOUD_URL could previously replace local tokens unverified). - Zed import: split into 2-step `/discover` + `/import` flow. `/import` now requires `confirmedAccounts: [{ service, account, fingerprint }]` and re-reads the keychain server-side to filter by fingerprint, so a tampered discover response cannot trick the endpoint into saving an unrelated token. Cosmetic Socket.dev mitigations: - runElevatedPowerShell writes the elevated payload to a per-call temp `.ps1` file (mode 0o600) and references it via `-File`. Removes the textbook `-EncodedCommand <base64utf16le>` pattern flagged as malware by Socket's AI classifier. - Maintainer attestation `SECURITY-AUDITOR-NOTE:` blocks added at every flagged call site pointing to `docs/security/SOCKET_DEV_FINDINGS.md`. Build-time hardening: - `OMNIROUTE_BUILD_PROFILE=minimal` (`npm run build:secure`) physically removes the four sensitive modules from the standalone bundle via webpack `NormalModuleReplacementPlugin`. Stubs throw `FeatureDisabledError` at runtime. Intended for the `omniroute-secure` artifact. Tests: - 24 new unit tests in `tests/unit/security/` covering the wrapper builder, HMAC verification (4 cases), credential fingerprint determinism (5 cases), confirmedAccounts validation + fingerprint filtering (6 cases), and the minimal-build stubs (5 cases). Docs: - New `docs/security/SOCKET_DEV_FINDINGS.md` — per-finding attestation. - New `socket.yml` — Socket.dev v2 config pointing at the attestation. - Updated `SECURITY.md` — supply-chain scanner section. - Updated `.env.example` — three new env vars documented. Backwards compatibility: - Cloud sync token overwrite is OFF by default. Users who relied on it must set `OMNIROUTE_CLOUD_SYNC_SECRETS=true`. Breaking change documented in CHANGELOG. - Zed import 2-step is the new default; legacy 1-step preserved behind `OMNIROUTE_ZED_IMPORT_LEGACY_ONE_STEP=true` and will be removed in v3.9. Closes #2863 * feat: implement automated skill workflows and update system configuration and validation schemas * test: eliminate dynamic cast warnings in cloud-sync unit test * test: isolate services-branch-hardening database directory to avoid concurrency issues * chore(docs): refresh generated docs collection index Update the generated Fumadocs browser collection mapping to keep documentation imports in sync with the current docs structure. * docs: update generated browser docs collection manifest Refresh the generated Fumadocs browser collection mapping so the docs site can resolve the current documentation files correctly. --------- Co-authored-by: OpenClaw <openclaw@kuzhomesrv.local> Co-authored-by: Dmitry Kuznetsov <139351986+dmitry@users.noreply.local> Co-authored-by: KuzyaBot <kuzya@local> Co-authored-by: JeferssonLemes <jeferssondev@gmail.com> Co-authored-by: Paijo <14921983+oyi77@users.noreply.github.com> Co-authored-by: Markus Hartung <mail@hartmark.se> Co-authored-by: akarray <akarray@users.noreply.github.com> Co-authored-by: Apostol Apostolov <theapoapostolov@gmail.com> Co-authored-by: Hernan Javier Ardila Sanchez <hjasgr@gmail.com> Co-authored-by: Dmitry Kuznetsov <dmitry@kuznetsov.me> Co-authored-by: Nikolay Alafuzov <alafuzov_nn@rusklimat.ru> Co-authored-by: oyi77 <oyi77@users.noreply.github.com> Co-authored-by: Ronaldo Davi <alltomatos@users.noreply.github.com> Co-authored-by: levonk <277861+levonk@users.noreply.github.com> Co-authored-by: Lenine Júnior <lenine@engrene.com.br> Co-authored-by: Annas Alghoffar <aag.annas@gmail.com> Co-authored-by: Tushar Agarwal <76201310+Tushar49@users.noreply.github.com> Co-authored-by: GreatLiu <eurasiaxz@qq.com> Co-authored-by: yuna amelia <230527278+yunaamelia@users.noreply.github.com> Co-authored-by: Randi <55005611+rdself@users.noreply.github.com> Co-authored-by: Container <78986709+disonjer@users.noreply.github.com> Co-authored-by: nickwizard <35692452+nickwizard@users.noreply.github.com> Co-authored-by: Rajvardhan Patil <rajvardhanpatil7890@gmail.com> Co-authored-by: Raxxoor <manker_lol@hotmail.com> Co-authored-by: Muhammad Mugni Hadi <mugnimaestra3@gmail.com> Co-authored-by: mi <123757457+soyelmismo@users.noreply.github.com> Co-authored-by: Automation <automation@omniroute>
771 lines
25 KiB
TypeScript
771 lines
25 KiB
TypeScript
import {
|
|
cleanupExpiredHandoffs,
|
|
getHandoff,
|
|
hasActiveHandoff,
|
|
type HandoffPayload,
|
|
upsertHandoff,
|
|
} from "../../src/lib/db/contextHandoffs.ts";
|
|
import { estimateTokens } from "./contextManager.ts";
|
|
import { stripMarkdownCodeFence } from "../utils/aiSdkCompat.ts";
|
|
|
|
export const HANDOFF_WARNING_THRESHOLD = 0.85;
|
|
export const HANDOFF_EXHAUSTION_THRESHOLD = 0.95;
|
|
|
|
const MAX_HISTORY_TOKENS_FOR_SUMMARY = 8000;
|
|
const DEFAULT_MAX_MESSAGES_FOR_SUMMARY = 30;
|
|
const DEFAULT_SUMMARY_RESPONSE_TOKENS = 800;
|
|
const MAX_SUMMARY_LENGTH = 2000;
|
|
const MAX_TASK_PROGRESS_LENGTH = 1200;
|
|
const MAX_DECISIONS = 8;
|
|
const MAX_ENTITIES = 10;
|
|
const DEFAULT_TTL_MS = 5 * 60 * 60 * 1000;
|
|
const OMNI_MODEL_TAG_PATTERN = /(?:\\n|\n|\r)*<omniModel>[^<]+<\/omniModel>(?:\\n|\n|\r)*/g;
|
|
const inflightHandoffGenerations = new Set<string>();
|
|
|
|
const HANDOFF_PROMPT_TEMPLATE = `You are a context summarizer. Analyze the conversation below and generate a structured handoff summary.
|
|
This summary will be used to restore context when this conversation is moved to a new AI account.
|
|
|
|
CONVERSATION HISTORY:
|
|
{HISTORY}
|
|
|
|
Generate a JSON object with this exact structure:
|
|
{
|
|
"summary": "A clear, dense summary of what has been discussed and accomplished (max 200 words). Focus on what the AI needs to know to continue seamlessly.",
|
|
"keyDecisions": ["decision1", "decision2"],
|
|
"taskProgress": "Current state of the task: what's done, what's pending, next steps",
|
|
"activeEntities": ["file1.ts", "feature X", "topic Y"]
|
|
}
|
|
|
|
Important: Return ONLY the JSON object, no markdown, no explanation.`;
|
|
|
|
export type MessageLike = {
|
|
role?: string;
|
|
content?: unknown;
|
|
};
|
|
|
|
export interface ContextRelayConfig {
|
|
handoffModel?: string;
|
|
handoffThreshold?: number;
|
|
handoffProviders?: string[];
|
|
maxMessagesForSummary?: number;
|
|
}
|
|
|
|
export interface UniversalHandoffConfig {
|
|
/** Enable universal context handoff for any model/provider switch */
|
|
enabled: boolean;
|
|
/** When to generate handoff: always = every turn, on-switch = only when model changes, on-error = only after error fallback */
|
|
trigger: "always" | "on-switch" | "on-error";
|
|
/** Providers allowed to generate handoffs. Empty array = all providers */
|
|
providerAllowlist: string[];
|
|
/** Max messages to include in summary generation */
|
|
maxMessagesForSummary: number;
|
|
/** Model to use for summary generation (e.g., 'gpt-4o-mini'). Empty = use same model as request */
|
|
handoffModel: string;
|
|
/** Handoff TTL in minutes */
|
|
ttlMinutes: number;
|
|
/** Preserve existing system prompt when injecting handoff */
|
|
preserveSystemPrompt: boolean;
|
|
}
|
|
|
|
export const DEFAULT_UNIVERSAL_HANDOFF_CONFIG: UniversalHandoffConfig = {
|
|
enabled: true,
|
|
trigger: "on-switch",
|
|
providerAllowlist: [],
|
|
maxMessagesForSummary: 30,
|
|
handoffModel: "",
|
|
ttlMinutes: 300,
|
|
preserveSystemPrompt: true,
|
|
};
|
|
|
|
export const SKIP_UNIVERSAL_HANDOFF_FLAG = "_omnirouteSkipUniversalHandoff";
|
|
|
|
export function resolveUniversalHandoffConfig(
|
|
comboConfig: Record<string, unknown> | null | undefined,
|
|
globalConfig: Record<string, unknown> | null | undefined
|
|
): UniversalHandoffConfig {
|
|
const rawCombo = comboConfig ?? {};
|
|
const rawGlobal = globalConfig ?? {};
|
|
|
|
const getBool = (key: keyof UniversalHandoffConfig, fallback: boolean): boolean => {
|
|
if (typeof rawCombo[key] === "boolean") return rawCombo[key] as boolean;
|
|
if (typeof rawGlobal[key] === "boolean") return rawGlobal[key] as boolean;
|
|
return fallback;
|
|
};
|
|
|
|
const getString = (key: keyof UniversalHandoffConfig, fallback: string): string => {
|
|
if (typeof rawCombo[key] === "string") {
|
|
const v = (rawCombo[key] as string).trim();
|
|
if (v.length > 0) return v;
|
|
}
|
|
if (typeof rawGlobal[key] === "string") {
|
|
const v = (rawGlobal[key] as string).trim();
|
|
if (v.length > 0) return v;
|
|
}
|
|
return fallback;
|
|
};
|
|
|
|
const getNumber = (
|
|
key: keyof UniversalHandoffConfig,
|
|
fallback: number,
|
|
min?: number,
|
|
max?: number
|
|
): number => {
|
|
let candidate: number | undefined;
|
|
const raw = rawCombo[key] ?? rawGlobal[key];
|
|
if (typeof raw === "number" && Number.isFinite(raw)) {
|
|
candidate = raw;
|
|
} else if (typeof raw === "string") {
|
|
const parsed = Number(raw);
|
|
if (Number.isFinite(parsed)) candidate = parsed;
|
|
}
|
|
if (candidate === undefined) return fallback;
|
|
if (min !== undefined && candidate < min) return min;
|
|
if (max !== undefined && candidate > max) return max;
|
|
return candidate;
|
|
};
|
|
|
|
const getStringArray = (key: keyof UniversalHandoffConfig, fallback: string[]): string[] => {
|
|
const raw = rawCombo[key] ?? rawGlobal[key];
|
|
if (!Array.isArray(raw)) return fallback;
|
|
return raw
|
|
.map((item) => (typeof item === "string" ? item.trim().toLowerCase() : ""))
|
|
.filter(Boolean);
|
|
};
|
|
|
|
const triggerRaw = getString("trigger", DEFAULT_UNIVERSAL_HANDOFF_CONFIG.trigger);
|
|
const trigger: UniversalHandoffConfig["trigger"] =
|
|
triggerRaw === "always" || triggerRaw === "on-error" ? triggerRaw : "on-switch";
|
|
|
|
return {
|
|
enabled: getBool("enabled", DEFAULT_UNIVERSAL_HANDOFF_CONFIG.enabled),
|
|
trigger,
|
|
providerAllowlist: getStringArray(
|
|
"providerAllowlist",
|
|
DEFAULT_UNIVERSAL_HANDOFF_CONFIG.providerAllowlist
|
|
),
|
|
maxMessagesForSummary: getNumber(
|
|
"maxMessagesForSummary",
|
|
DEFAULT_UNIVERSAL_HANDOFF_CONFIG.maxMessagesForSummary,
|
|
5,
|
|
100
|
|
),
|
|
handoffModel: getString("handoffModel", DEFAULT_UNIVERSAL_HANDOFF_CONFIG.handoffModel),
|
|
ttlMinutes: getNumber("ttlMinutes", DEFAULT_UNIVERSAL_HANDOFF_CONFIG.ttlMinutes, 1, 10080),
|
|
preserveSystemPrompt: getBool(
|
|
"preserveSystemPrompt",
|
|
DEFAULT_UNIVERSAL_HANDOFF_CONFIG.preserveSystemPrompt
|
|
),
|
|
};
|
|
}
|
|
export interface ParsedHandoffContent {
|
|
summary: string;
|
|
keyDecisions: string[];
|
|
taskProgress: string;
|
|
activeEntities: string[];
|
|
}
|
|
|
|
export function resolveContextRelayConfig(
|
|
config?: Record<string, unknown> | null
|
|
): Required<ContextRelayConfig> {
|
|
const rawThreshold = Number(config?.handoffThreshold);
|
|
const rawMaxMessages = Number(config?.maxMessagesForSummary);
|
|
const hasExplicitProviders = Array.isArray(config?.handoffProviders);
|
|
const handoffProviders = hasExplicitProviders
|
|
? (config?.handoffProviders as unknown[])
|
|
.map((item) => (typeof item === "string" ? item.trim().toLowerCase() : ""))
|
|
.filter(Boolean)
|
|
: ["codex"];
|
|
|
|
return {
|
|
handoffModel:
|
|
typeof config?.handoffModel === "string" && config.handoffModel.trim().length > 0
|
|
? config.handoffModel.trim()
|
|
: "",
|
|
handoffThreshold:
|
|
Number.isFinite(rawThreshold) &&
|
|
rawThreshold > 0 &&
|
|
rawThreshold < HANDOFF_EXHAUSTION_THRESHOLD
|
|
? rawThreshold
|
|
: HANDOFF_WARNING_THRESHOLD,
|
|
handoffProviders: hasExplicitProviders ? handoffProviders : ["codex"],
|
|
maxMessagesForSummary:
|
|
Number.isFinite(rawMaxMessages) && rawMaxMessages >= 5 && rawMaxMessages <= 100
|
|
? Math.round(rawMaxMessages)
|
|
: DEFAULT_MAX_MESSAGES_FOR_SUMMARY,
|
|
};
|
|
}
|
|
|
|
function getInflightKey(sessionId: string, comboName: string): string {
|
|
return `${sessionId}::${comboName}`;
|
|
}
|
|
|
|
function toTextContent(content: unknown): string {
|
|
if (typeof content === "string") return content;
|
|
if (!Array.isArray(content)) return "";
|
|
|
|
return content
|
|
.map((part) => {
|
|
if (!part || typeof part !== "object") return "";
|
|
if (typeof (part as Record<string, unknown>).text === "string") {
|
|
return String((part as Record<string, unknown>).text);
|
|
}
|
|
if (typeof (part as Record<string, unknown>).content === "string") {
|
|
return String((part as Record<string, unknown>).content);
|
|
}
|
|
return "";
|
|
})
|
|
.filter(Boolean)
|
|
.join("\n");
|
|
}
|
|
|
|
function formatMessagesForPrompt(messages: MessageLike[]): string {
|
|
return messages
|
|
.map((message, index) => {
|
|
const role = typeof message.role === "string" ? message.role : "unknown";
|
|
const content = toTextContent(message.content).trim();
|
|
if (!content) return "";
|
|
return `[${index + 1}] ${role.toUpperCase()}:\n${content}`;
|
|
})
|
|
.filter(Boolean)
|
|
.join("\n\n");
|
|
}
|
|
|
|
export function selectMessagesForSummary(messages: MessageLike[], maxMessages: number): MessageLike[] {
|
|
const validMessages = messages.filter((m) => m && typeof m === "object");
|
|
const system = validMessages.filter(
|
|
(m) => typeof m.role === "string" && (m.role === "system" || m.role === "developer")
|
|
);
|
|
const nonSystem = validMessages.filter(
|
|
(m) => typeof m.role !== "string" || (m.role !== "system" && m.role !== "developer")
|
|
);
|
|
|
|
const recentMessages = [...system, ...nonSystem.slice(-maxMessages)];
|
|
let working = [...recentMessages];
|
|
|
|
while (working.length > system.length + 1) {
|
|
const history = formatMessagesForPrompt(working);
|
|
if (estimateTokens(history) <= MAX_HISTORY_TOKENS_FOR_SUMMARY) {
|
|
return working;
|
|
}
|
|
working = [...system, ...working.slice(system.length + 1)];
|
|
}
|
|
|
|
const fallbackHistory = formatMessagesForPrompt(working);
|
|
if (estimateTokens(fallbackHistory) > MAX_HISTORY_TOKENS_FOR_SUMMARY) {
|
|
// If there are system messages, return them so the caller can still produce context.
|
|
// If there are no system messages (system=[]), fall back to the single most-recent
|
|
// non-system message rather than returning [] which would silently drop the handoff.
|
|
if (system.length > 0) {
|
|
return system;
|
|
}
|
|
const lastNonSystem = nonSystem[nonSystem.length - 1];
|
|
return lastNonSystem ? [lastNonSystem] : [];
|
|
}
|
|
|
|
return working;
|
|
}
|
|
|
|
function normalizeStringArray(value: unknown, maxItems: number, maxLength = 240): string[] {
|
|
if (!Array.isArray(value)) return [];
|
|
return value
|
|
.map((item) => (typeof item === "string" ? item.trim() : ""))
|
|
.filter(Boolean)
|
|
.slice(0, maxItems)
|
|
.map((item) => item.slice(0, maxLength));
|
|
}
|
|
|
|
function sanitizeJsonCandidate(content: string): string {
|
|
return content.replace(OMNI_MODEL_TAG_PATTERN, "").trim();
|
|
}
|
|
|
|
function extractJsonCandidate(content: string): string {
|
|
const stripped = sanitizeJsonCandidate(String(stripMarkdownCodeFence(content) || ""));
|
|
if (!stripped) return "";
|
|
|
|
try {
|
|
JSON.parse(stripped);
|
|
return stripped;
|
|
} catch {
|
|
const firstBrace = stripped.indexOf("{");
|
|
const lastBrace = stripped.lastIndexOf("}");
|
|
if (firstBrace >= 0 && lastBrace > firstBrace) {
|
|
return stripped.slice(firstBrace, lastBrace + 1);
|
|
}
|
|
return stripped;
|
|
}
|
|
}
|
|
|
|
export function parseHandoffJSON(content: string): ParsedHandoffContent | null {
|
|
const candidate = extractJsonCandidate(content);
|
|
if (!candidate) return null;
|
|
|
|
try {
|
|
const parsed = JSON.parse(candidate) as Record<string, unknown>;
|
|
const summary =
|
|
typeof parsed.summary === "string" ? parsed.summary.trim().slice(0, MAX_SUMMARY_LENGTH) : "";
|
|
const taskProgress =
|
|
typeof parsed.taskProgress === "string"
|
|
? parsed.taskProgress.trim().slice(0, MAX_TASK_PROGRESS_LENGTH)
|
|
: "";
|
|
const keyDecisions = normalizeStringArray(parsed.keyDecisions, MAX_DECISIONS);
|
|
const activeEntities = normalizeStringArray(parsed.activeEntities, MAX_ENTITIES);
|
|
|
|
if (!summary) return null;
|
|
|
|
return {
|
|
summary,
|
|
keyDecisions,
|
|
taskProgress,
|
|
activeEntities,
|
|
};
|
|
} catch {
|
|
return null;
|
|
}
|
|
}
|
|
|
|
function escapeXml(value: string): string {
|
|
return value
|
|
.replaceAll("&", "&")
|
|
.replaceAll("<", "<")
|
|
.replaceAll(">", ">")
|
|
.replaceAll('"', """)
|
|
.replaceAll("'", "'");
|
|
}
|
|
|
|
function getResponseText(json: Record<string, unknown>): string {
|
|
const choices = Array.isArray(json.choices) ? json.choices : [];
|
|
const firstChoice = choices[0] as Record<string, unknown> | undefined;
|
|
const firstMessage = firstChoice?.message as Record<string, unknown> | undefined;
|
|
if (typeof firstMessage?.content === "string") {
|
|
return firstMessage.content;
|
|
}
|
|
|
|
if (Array.isArray(firstMessage?.content)) {
|
|
return toTextContent(firstMessage.content);
|
|
}
|
|
|
|
const output = Array.isArray(json.output) ? json.output : [];
|
|
for (const item of output) {
|
|
if (!item || typeof item !== "object") continue;
|
|
const content = Array.isArray((item as Record<string, unknown>).content)
|
|
? ((item as Record<string, unknown>).content as Array<Record<string, unknown>>)
|
|
: [];
|
|
for (const part of content) {
|
|
if (typeof part?.text === "string") return part.text;
|
|
}
|
|
}
|
|
|
|
const content = Array.isArray(json.content) ? json.content : [];
|
|
for (const part of content) {
|
|
if (!part || typeof part !== "object") continue;
|
|
if (typeof (part as Record<string, unknown>).text === "string") {
|
|
return String((part as Record<string, unknown>).text);
|
|
}
|
|
}
|
|
|
|
return "";
|
|
}
|
|
|
|
async function generateHandoffAsync(options: {
|
|
sessionId: string;
|
|
comboName: string;
|
|
connectionId: string;
|
|
percentUsed: number;
|
|
messages: MessageLike[];
|
|
model: string;
|
|
expiresAt: string | null;
|
|
config?: ContextRelayConfig | null;
|
|
handleSingleModel: (body: Record<string, unknown>, modelStr: string) => Promise<Response>;
|
|
}): Promise<void> {
|
|
cleanupExpiredHandoffs();
|
|
|
|
const relayConfig = resolveContextRelayConfig(options.config as Record<string, unknown>);
|
|
const summaryModel = relayConfig.handoffModel || options.model;
|
|
const selectedMessages = selectMessagesForSummary(
|
|
Array.isArray(options.messages) ? options.messages : [],
|
|
relayConfig.maxMessagesForSummary
|
|
);
|
|
const historyText = formatMessagesForPrompt(selectedMessages);
|
|
if (!historyText) return;
|
|
|
|
const summaryPrompt = HANDOFF_PROMPT_TEMPLATE.replace("{HISTORY}", historyText);
|
|
const summaryBody = {
|
|
model: summaryModel,
|
|
messages: [{ role: "user", content: summaryPrompt }],
|
|
stream: false,
|
|
max_tokens: DEFAULT_SUMMARY_RESPONSE_TOKENS,
|
|
temperature: 0.1,
|
|
_omnirouteSkipContextRelay: true,
|
|
_omnirouteInternalRequest: "context-handoff",
|
|
};
|
|
|
|
const response = await options.handleSingleModel(summaryBody, summaryModel);
|
|
if (!response.ok) return;
|
|
|
|
let content = "";
|
|
try {
|
|
const json = (await response.clone().json()) as Record<string, unknown>;
|
|
content = getResponseText(json);
|
|
} catch {
|
|
try {
|
|
content = await response.clone().text();
|
|
} catch {
|
|
content = "";
|
|
}
|
|
}
|
|
|
|
const parsed = parseHandoffJSON(content);
|
|
if (!parsed) return;
|
|
|
|
upsertHandoff({
|
|
sessionId: options.sessionId,
|
|
comboName: options.comboName,
|
|
fromAccount: options.connectionId,
|
|
summary: parsed.summary,
|
|
keyDecisions: parsed.keyDecisions,
|
|
taskProgress: parsed.taskProgress,
|
|
activeEntities: parsed.activeEntities,
|
|
messageCount: Array.isArray(options.messages) ? options.messages.length : 0,
|
|
model: summaryModel,
|
|
warningThresholdPct: relayConfig.handoffThreshold,
|
|
generatedAt: new Date().toISOString(),
|
|
expiresAt: options.expiresAt || new Date(Date.now() + DEFAULT_TTL_MS).toISOString(),
|
|
});
|
|
}
|
|
|
|
export function maybeGenerateHandoff(options: {
|
|
sessionId: string | null;
|
|
comboName: string;
|
|
connectionId: string | null;
|
|
percentUsed: number;
|
|
messages: MessageLike[];
|
|
model: string;
|
|
expiresAt: string | null;
|
|
config?: ContextRelayConfig | null;
|
|
handleSingleModel: (body: Record<string, unknown>, modelStr: string) => Promise<Response>;
|
|
}): void {
|
|
if (!options.sessionId || !options.connectionId) return;
|
|
|
|
const relayConfig = resolveContextRelayConfig(options.config as Record<string, unknown>);
|
|
if (relayConfig.handoffProviders.length === 0) return;
|
|
if (options.percentUsed < relayConfig.handoffThreshold) return;
|
|
if (options.percentUsed >= HANDOFF_EXHAUSTION_THRESHOLD) return;
|
|
|
|
cleanupExpiredHandoffs();
|
|
if (hasActiveHandoff(options.sessionId, options.comboName)) return;
|
|
const inflightKey = getInflightKey(options.sessionId, options.comboName);
|
|
if (inflightHandoffGenerations.has(inflightKey)) return;
|
|
inflightHandoffGenerations.add(inflightKey);
|
|
|
|
setImmediate(() => {
|
|
generateHandoffAsync({
|
|
...options,
|
|
sessionId: options.sessionId as string,
|
|
connectionId: options.connectionId as string,
|
|
config: relayConfig,
|
|
})
|
|
.catch((err) => {
|
|
if (process.env.NODE_ENV !== "test") {
|
|
console.warn("[context-relay] Handoff generation failed:", err?.message || err);
|
|
}
|
|
})
|
|
.finally(() => {
|
|
inflightHandoffGenerations.delete(inflightKey);
|
|
});
|
|
});
|
|
}
|
|
|
|
export function buildHandoffSystemMessage(payload: HandoffPayload): string {
|
|
const decisions = payload.keyDecisions.map((decision) => ` - ${escapeXml(decision)}`).join("\n");
|
|
const entities = payload.activeEntities.map((entity) => escapeXml(entity)).join(", ");
|
|
|
|
return `<context_handoff>
|
|
<transfer_reason>Account quota transfer - continuing from previous session</transfer_reason>
|
|
<session_summary>${escapeXml(payload.summary)}</session_summary>
|
|
<task_progress>${escapeXml(payload.taskProgress)}</task_progress>
|
|
<key_decisions>
|
|
${decisions}
|
|
</key_decisions>
|
|
<active_context>${entities}</active_context>
|
|
<messages_processed>${payload.messageCount}</messages_processed>
|
|
</context_handoff>
|
|
|
|
You are continuing a conversation that was transferred from another account due to quota limits.
|
|
The context above contains a concise summary of the prior work. Continue seamlessly from where the session left off.`;
|
|
}
|
|
|
|
export function injectHandoffIntoBody(
|
|
body: Record<string, unknown>,
|
|
payload: HandoffPayload
|
|
): Record<string, unknown> {
|
|
const handoffContent = buildHandoffSystemMessage(payload);
|
|
const isResponsesRequest =
|
|
Object.prototype.hasOwnProperty.call(body, "input") ||
|
|
Object.prototype.hasOwnProperty.call(body, "instructions");
|
|
|
|
if (isResponsesRequest) {
|
|
const existingInstructions =
|
|
typeof body.instructions === "string" && body.instructions.trim().length > 0
|
|
? body.instructions
|
|
: "";
|
|
const nextBody: Record<string, unknown> = {
|
|
...body,
|
|
instructions: existingInstructions
|
|
? `${handoffContent}\n\n${existingInstructions}`
|
|
: handoffContent,
|
|
};
|
|
|
|
if (Array.isArray(nextBody.messages) && nextBody.messages.length === 0) {
|
|
const { messages: _messages, ...rest } = nextBody;
|
|
return rest;
|
|
}
|
|
|
|
return nextBody;
|
|
}
|
|
|
|
const handoffMessage = {
|
|
role: "system",
|
|
content: handoffContent,
|
|
};
|
|
const messages = Array.isArray(body.messages) ? [...body.messages] : [];
|
|
|
|
return {
|
|
...body,
|
|
messages: [handoffMessage, ...messages],
|
|
};
|
|
}
|
|
|
|
export function buildUniversalHandoffSystemMessage(
|
|
prevModel: string,
|
|
currModel: string,
|
|
reason: string,
|
|
payload?: HandoffPayload | null
|
|
): string {
|
|
const escapedPrev = escapeXml(prevModel);
|
|
const escapedCurr = escapeXml(currModel);
|
|
const escapedReason = escapeXml(reason);
|
|
|
|
if (!payload || !payload.summary) {
|
|
return `<context_handoff>
|
|
<transfer_reason>${escapedReason}</transfer_reason>
|
|
<previous_model>${escapedPrev}</previous_model>
|
|
<current_model>${escapedCurr}</current_model>
|
|
<note>A continuación se resume toda la conversacion para continuar sin perder el hilo.</note>
|
|
</context_handoff>`;
|
|
}
|
|
|
|
const decisions = payload.keyDecisions.map((d) => ` - ${escapeXml(d)}`).join("\n");
|
|
const entities = payload.activeEntities.map((e) => escapeXml(e)).join(", ");
|
|
|
|
return `<context_handoff>
|
|
<transfer_reason>${escapedReason}</transfer_reason>
|
|
<previous_model>${escapedPrev}</previous_model>
|
|
<current_model>${escapedCurr}</current_model>
|
|
<session_summary>${escapeXml(payload.summary)}</session_summary>
|
|
<task_progress>${escapeXml(payload.taskProgress)}</task_progress>
|
|
<key_decisions>
|
|
${decisions}
|
|
</key_decisions>
|
|
<active_context>${entities}</active_context>
|
|
<messages_processed>${payload.messageCount}</messages_processed>
|
|
</context_handoff>
|
|
|
|
Continues conversation transfered from ${escapedPrev} to ${escapedCurr}.
|
|
The context above contains a concise summary of prior work.
|
|
Continue seamlessly from where the session left off.`;
|
|
}
|
|
|
|
/**
|
|
* Evaluate whether a universal handoff is needed for a model/provider switch.
|
|
*,
|
|
* @returns "generate" - need to create a new handoff summary
|
|
* "inject" - handoff already exists, just inject it
|
|
* "skip" - no handoff needed
|
|
*/
|
|
export function shouldGenerateUniversalHandoff(options: {
|
|
sessionId: string | null;
|
|
comboName: string;
|
|
previousModel: string | null;
|
|
currentModel: string;
|
|
universalConfig: UniversalHandoffConfig;
|
|
}): "generate" | "inject" | "skip" {
|
|
if (!options.universalConfig.enabled) return "skip";
|
|
if (!options.previousModel) return "skip";
|
|
if (options.previousModel === options.currentModel) return "skip";
|
|
|
|
// Check if handoff already exists for this session/combo
|
|
if (options.sessionId) {
|
|
const existing = getHandoff(options.sessionId, options.comboName);
|
|
if (existing && existing.summary) return "inject";
|
|
}
|
|
|
|
return "generate";
|
|
}
|
|
|
|
/**
|
|
* Generate a universal handoff summary for any model/provider switch.
|
|
*/
|
|
async function generateUniversalHandoffAsync(options: {
|
|
sessionId: string;
|
|
comboName: string;
|
|
messages: MessageLike[];
|
|
prevModel: string;
|
|
currModel: string;
|
|
handoffModel: string;
|
|
ttlMs: number;
|
|
maxMessages: number;
|
|
providerAllowlist: string[];
|
|
handleSingleModel: (body: Record<string, unknown>, modelStr: string) => Promise<Response>;
|
|
}): Promise<void> {
|
|
const selectedMessages = selectMessagesForSummary(
|
|
Array.isArray(options.messages) ? options.messages : [],
|
|
options.maxMessages
|
|
);
|
|
const historyText = formatMessagesForPrompt(selectedMessages);
|
|
if (!historyText) return;
|
|
|
|
const summaryPrompt = HANDOFF_PROMPT_TEMPLATE.replace("{HISTORY}", historyText);
|
|
const summaryModel = options.handoffModel || options.currModel;
|
|
const summaryBody: Record<string, unknown> = {
|
|
model: summaryModel,
|
|
messages: [{ role: "user", content: summaryPrompt }],
|
|
stream: false,
|
|
max_tokens: DEFAULT_SUMMARY_RESPONSE_TOKENS,
|
|
temperature: 0.1,
|
|
_omnirouteSkipContextRelay: true,
|
|
_omnirouteInternalRequest: "universal-handoff",
|
|
};
|
|
|
|
const response = await options.handleSingleModel(summaryBody, summaryModel);
|
|
if (!response.ok) return;
|
|
|
|
let content = "";
|
|
try {
|
|
const json = (await response.clone().json()) as Record<string, unknown>;
|
|
content = getResponseText(json);
|
|
} catch {
|
|
try {
|
|
content = await response.clone().text();
|
|
} catch {
|
|
content = "";
|
|
}
|
|
}
|
|
|
|
const parsed = parseHandoffJSON(content);
|
|
if (!parsed) return;
|
|
|
|
upsertHandoff({
|
|
sessionId: options.sessionId,
|
|
comboName: options.comboName,
|
|
fromAccount: `universal:${options.prevModel}`,
|
|
summary: parsed.summary,
|
|
keyDecisions: parsed.keyDecisions,
|
|
taskProgress: parsed.taskProgress,
|
|
activeEntities: parsed.activeEntities,
|
|
messageCount: Array.isArray(options.messages) ? options.messages.length : 0,
|
|
model: summaryModel,
|
|
lastModel: options.prevModel,
|
|
warningThresholdPct: 0,
|
|
generatedAt: new Date().toISOString(),
|
|
expiresAt: new Date(Date.now() + options.ttlMs).toISOString(),
|
|
});
|
|
}
|
|
|
|
export function maybeGenerateUniversalHandoff(options: {
|
|
sessionId: string | null;
|
|
comboName: string;
|
|
messages: MessageLike[];
|
|
prevModel: string | null;
|
|
currModel: string;
|
|
universalConfig: UniversalHandoffConfig;
|
|
handleSingleModel: (body: Record<string, unknown>, modelStr: string) => Promise<Response>;
|
|
}): void {
|
|
const decision = shouldGenerateUniversalHandoff({
|
|
sessionId: options.sessionId,
|
|
comboName: options.comboName,
|
|
previousModel: options.prevModel,
|
|
currentModel: options.currModel,
|
|
universalConfig: options.universalConfig,
|
|
});
|
|
|
|
if (decision !== "generate") return;
|
|
if (!options.sessionId) return;
|
|
|
|
const inflightKey = getInflightKey(options.sessionId, options.comboName);
|
|
if (inflightHandoffGenerations.has(inflightKey)) return;
|
|
inflightHandoffGenerations.add(inflightKey);
|
|
|
|
const ttlMs = (options.universalConfig.ttlMinutes || 300) * 60 * 1000;
|
|
|
|
setImmediate(() => {
|
|
generateUniversalHandoffAsync({
|
|
sessionId: options.sessionId as string,
|
|
comboName: options.comboName,
|
|
messages: options.messages,
|
|
prevModel: options.prevModel || "unknown",
|
|
currModel: options.currModel,
|
|
handoffModel: options.universalConfig.handoffModel || options.currModel,
|
|
ttlMs,
|
|
maxMessages: options.universalConfig.maxMessagesForSummary,
|
|
providerAllowlist: options.universalConfig.providerAllowlist,
|
|
handleSingleModel: options.handleSingleModel,
|
|
})
|
|
.catch((err) => {
|
|
if (process.env.NODE_ENV !== "test") {
|
|
console.warn("[universal-handoff] Generation failed:", err?.message || err);
|
|
}
|
|
})
|
|
.finally(() => {
|
|
inflightHandoffGenerations.delete(inflightKey);
|
|
});
|
|
});
|
|
}
|
|
|
|
export function injectUniversalHandoffBody(
|
|
body: Record<string, unknown>,
|
|
prevModel: string,
|
|
currModel: string,
|
|
reason: string,
|
|
existingPayload?: HandoffPayload | null
|
|
): Record<string, unknown> {
|
|
const handoffContent = buildUniversalHandoffSystemMessage(
|
|
prevModel,
|
|
currModel,
|
|
reason,
|
|
existingPayload
|
|
);
|
|
|
|
const isResponsesRequest =
|
|
Object.prototype.hasOwnProperty.call(body, "input") ||
|
|
Object.prototype.hasOwnProperty.call(body, "instructions");
|
|
|
|
if (isResponsesRequest) {
|
|
const existingInstructions =
|
|
typeof body.instructions === "string" && body.instructions.trim().length > 0
|
|
? body.instructions
|
|
: "";
|
|
const nextBody: Record<string, unknown> = {
|
|
...body,
|
|
instructions: existingInstructions
|
|
? `${handoffContent}\n\n${existingInstructions}`
|
|
: handoffContent,
|
|
};
|
|
if (Array.isArray(nextBody.messages) && nextBody.messages.length === 0) {
|
|
const { messages: _messages, ...rest } = nextBody;
|
|
return rest;
|
|
}
|
|
return nextBody;
|
|
}
|
|
|
|
const handoffMessage = {
|
|
role: "system",
|
|
content: handoffContent,
|
|
};
|
|
const messages = Array.isArray(body.messages) ? [...body.messages] : [];
|
|
|
|
return {
|
|
...body,
|
|
messages: [handoffMessage, ...messages],
|
|
};
|
|
}
|