Files
OmniRoute/tests/unit/usage-token-buffer.test.ts
NOXX - Commiter 7a68a7961c fix(notion-web): production-ready labels, multi-workspace, inference, usage (FINAL) (#7768)
* fix(notion-web): use real picker labels as primary model ids

Catalog /v1/models now surfaces web-picker names (fable-5, gpt-5.6-sol)
instead of Notion food codenames (acai-budino-high, orange-mousse).
Food codenames stay internal via notionCodename + resolveNotionCodename
for runInferenceTranscript. Legacy codename requests still work; responses
echo the client-facing id.

Also points discovery/inference at app.notion.com (same host as the AI picker).

Follow-up to #7696.

* fix(notion-web): explain plan-locked models like Fable 5

Notion returns Fable 5 (acai-budino-high) with isDisabled=true and
disabledReason=business_or_enterprise_plan_required. Keep it out of the
enabled catalog (requests would fail) but surface a discovery warning so
operators know why it is missing. Also warn when space_id is resolved via
getSpaces instead of the cookie.

* feat(notion-web): auto-detect workspace without pasting space_id

Operators only need the raw token_v2 value. When space_id is omitted:
- getSpaces loads all workspaces (browser-like headers + user id)
- each workspace is probed via getAvailableModels
- the richest AI catalog wins

Also softens auth hints so they no longer demand a cookie blob with =.

* fix(notion-web): pick Business workspace so Fable 5 is listed

Probe ALL workspaces instead of early-exiting on the first catalog with
>=8 models. Prefer spaces where Fable is enabled over personal spaces
where Notion returns isDisabled=business_or_enterprise_plan_required.
Cache the chosen spaceId for inference when cookie has no space_id.

* fix(notion-web): working inference + honest token estimates

- runInferenceTranscript: createThread+threadId, config/context/user
  transcript, space/user headers (fixes ValidationError 400)
- Parse modern NDJSON patch/record-map; strip lang tags
- Estimate usage from text (Notion has no metering); mark estimated
- Treat all-zero usage as missing; skip USAGE_TOKEN_BUFFER on estimated
- Keep estimated flag through response sanitizer (was stripped -> flat 2000)

Verified live: fable-5/gpt-5.6-sol chat 200; usage 7 / 65 not constant 2000.

* refactor(notion-web): extract helpers to keep discoverNotionWebModels/execute under the complexity cap

Co-authored-by: diegosouzapw <8016841+diegosouzapw@users.noreply.github.com>

---------

Co-authored-by: Diego Rodrigues de Sa e Souza <diegosouza.pw@gmail.com>
Co-authored-by: artickc <artickc@users.noreply.github.com>
Co-authored-by: diegosouzapw <8016841+diegosouzapw@users.noreply.github.com>
2026-07-19 16:21:18 -03:00

183 lines
6.2 KiB
TypeScript

import test from "node:test";
import assert from "node:assert/strict";
import {
addBufferToUsage,
invalidateBufferTokensCache,
setBufferTokensCache,
} from "../../open-sse/utils/usageTracking.ts";
// ─── Helpers ──────────────────────────────────────────────────────────────
function resetEnv(saved: string | undefined) {
if (saved === undefined) {
delete process.env.USAGE_TOKEN_BUFFER;
} else {
process.env.USAGE_TOKEN_BUFFER = saved;
}
}
// ─── addBufferToUsage — baseline / env-var path ───────────────────────────
test("addBufferToUsage — adds DEFAULT 2000 when no env var and cache is null", () => {
const saved = process.env.USAGE_TOKEN_BUFFER;
delete process.env.USAGE_TOKEN_BUFFER;
invalidateBufferTokensCache();
const result = addBufferToUsage({ prompt_tokens: 25, completion_tokens: 24, total_tokens: 49 });
// Documents the race: after invalidation the sync path falls back to DEFAULT=2000
assert.equal(result.prompt_tokens, 2025);
assert.equal(result.completion_tokens, 24);
assert.equal(result.total_tokens, 2049);
resetEnv(saved);
});
test("addBufferToUsage — respects USAGE_TOKEN_BUFFER=0 env override", () => {
const saved = process.env.USAGE_TOKEN_BUFFER;
process.env.USAGE_TOKEN_BUFFER = "0";
invalidateBufferTokensCache();
const result = addBufferToUsage({ prompt_tokens: 25, completion_tokens: 24, total_tokens: 49 });
assert.equal(result.prompt_tokens, 25);
assert.equal(result.total_tokens, 49);
resetEnv(saved);
});
test("addBufferToUsage — respects USAGE_TOKEN_BUFFER=500 env override", () => {
const saved = process.env.USAGE_TOKEN_BUFFER;
process.env.USAGE_TOKEN_BUFFER = "500";
invalidateBufferTokensCache();
const result = addBufferToUsage({ prompt_tokens: 86, completion_tokens: 52, total_tokens: 138 });
assert.equal(result.prompt_tokens, 586);
assert.equal(result.total_tokens, 638);
resetEnv(saved);
});
test("addBufferToUsage — also adds to Claude-format input_tokens", () => {
const saved = process.env.USAGE_TOKEN_BUFFER;
process.env.USAGE_TOKEN_BUFFER = "100";
invalidateBufferTokensCache();
const result = addBufferToUsage({ input_tokens: 40, output_tokens: 20 });
assert.equal(result.input_tokens, 140);
resetEnv(saved);
});
test("addBufferToUsage — returns usage unchanged when buffer is 0 via env", () => {
const saved = process.env.USAGE_TOKEN_BUFFER;
process.env.USAGE_TOKEN_BUFFER = "0";
invalidateBufferTokensCache();
const usage = { prompt_tokens: 86, completion_tokens: 52, total_tokens: 138 };
const result = addBufferToUsage(usage);
assert.equal(result.prompt_tokens, 86);
assert.equal(result.completion_tokens, 52);
assert.equal(result.total_tokens, 138);
resetEnv(saved);
});
test("addBufferToUsage — skips safety buffer for estimated usage (web/heuristic)", () => {
const saved = process.env.USAGE_TOKEN_BUFFER;
delete process.env.USAGE_TOKEN_BUFFER;
invalidateBufferTokensCache();
const result = addBufferToUsage({
prompt_tokens: 6,
completion_tokens: 1,
total_tokens: 7,
estimated: true,
});
// Must NOT inflate heuristics to DEFAULT 2000 — that made every Notion
// request report a flat 2000 tokens.
assert.equal(result.prompt_tokens, 6);
assert.equal(result.completion_tokens, 1);
assert.equal(result.total_tokens, 7);
assert.equal(result.estimated, true);
resetEnv(saved);
});
// ─── setBufferTokensCache — the fix for the race condition ────────────────
//
// The race: invalidateBufferTokensCache() sets _cachedBuffer=null; the next
// synchronous call to getBufferTokens() falls back to DEFAULT=2000 before
// _loadBufferFromDb() (async) completes.
//
// The fix: runtimeSettings.ts calls setBufferTokensCache(newValue) instead of
// invalidateBufferTokensCache() so the correct value is available synchronously.
test("setBufferTokensCache(0) — immediately prevents buffer addition (no race window)", () => {
const saved = process.env.USAGE_TOKEN_BUFFER;
delete process.env.USAGE_TOKEN_BUFFER;
// Simulates what runtimeSettings does after saving usageTokenBuffer=0 in DB
setBufferTokensCache(0);
const result = addBufferToUsage({ prompt_tokens: 25, completion_tokens: 24, total_tokens: 49 });
// With the fix: 0 is applied synchronously — no 2000-token race window
assert.equal(result.prompt_tokens, 25);
assert.equal(result.completion_tokens, 24);
assert.equal(result.total_tokens, 49);
resetEnv(saved);
invalidateBufferTokensCache();
});
test("setBufferTokensCache(500) — immediately sets custom buffer value", () => {
const saved = process.env.USAGE_TOKEN_BUFFER;
delete process.env.USAGE_TOKEN_BUFFER;
setBufferTokensCache(500);
const result = addBufferToUsage({ prompt_tokens: 86, completion_tokens: 52, total_tokens: 138 });
assert.equal(result.prompt_tokens, 586);
assert.equal(result.total_tokens, 638);
resetEnv(saved);
invalidateBufferTokensCache();
});
test("setBufferTokensCache(0) — works for Claude-format (input_tokens)", () => {
const saved = process.env.USAGE_TOKEN_BUFFER;
delete process.env.USAGE_TOKEN_BUFFER;
setBufferTokensCache(0);
const result = addBufferToUsage({ input_tokens: 40, output_tokens: 20 });
assert.equal(result.input_tokens, 40);
resetEnv(saved);
invalidateBufferTokensCache();
});
test("invalidateBufferTokensCache — still resets to null (returns DEFAULT on next sync call)", () => {
const saved = process.env.USAGE_TOKEN_BUFFER;
delete process.env.USAGE_TOKEN_BUFFER;
// First prime the cache with a custom value
setBufferTokensCache(0);
const afterSet = addBufferToUsage({ prompt_tokens: 10, completion_tokens: 5, total_tokens: 15 });
assert.equal(afterSet.prompt_tokens, 10); // 0 buffer
// Then invalidate — next sync call reverts to DEFAULT (2000) while async reload happens
invalidateBufferTokensCache();
const afterInvalidate = addBufferToUsage({ prompt_tokens: 10, completion_tokens: 5, total_tokens: 15 });
assert.equal(afterInvalidate.prompt_tokens, 2010); // DEFAULT=2000 applied (race window)
resetEnv(saved);
});