mirror of
https://github.com/diegosouzapw/OmniRoute.git
synced 2026-07-26 09:52:11 +03:00
* chore(release): continue v3.8.25 development cycle after main code-sync (r5) main fast-forwarded to release/v3.8.25 (#3863): unblocked Build+Docker via #3864, plus #3837 (mimocode proxy) and #3862 (trivy bump). This marker re-opens the umbrella PR for further v3.8.25 work. No version bump. * fix(db): persist the Keep-latest-backups retention setting (#3834) (#3867) * fix(oauth): clear GitLab Duo setup message instead of 500 (#3861) (#3868) * test(oauth): prove refresh_token preserved on real gemini-cli/antigravity dispatch (#3850) (#3869) * feat(compression-ui): unified compression config UI — per-engine pages + combos editor + menu + WS default-on (#3860) Integrated into release/v3.8.25 — feat(compression-ui): unified compression configuration UI (Compression Hub + per-engine Lite/Aggressive/Ultra pages + combos editor + sidebar entry + live-WS default-on). File-size re-baselined for sidebarVisibility.ts/chatCore.ts growth; orphan ws test relocated to a collected path. * docs(changelog): complete the v3.8.25 release notes + credit all contributors Audited every commit since v3.8.24 and filled the gaps the [3.8.25] section was missing: a New Features section (compression engines + Compression Studios #3848, compression UI #3860, injection-guard #3857, kiro discovery #3836, Veo #3839, mimocode proxy #3837, Arena ELO flag #3821), 9 more Fixed entries (#3811/#3807/#3759/#3849/#3838/#3835/#3814/#3820/#3819), a Security section (CCR IDOR #3859, supply-chain #3824), and an Internal/Quality section. Every contributor and issue reporter is now credited. * docs(changelog): restore + complete the v3.8.25 release notes Re-adds CHANGELOG.md (a prior server-side commit accidentally dropped it) with the complete, audited [3.8.25] section: New Features, the full Fixed list, Security & Hardening, and Internal/Quality — every contributor and issue reporter credited. * chore(release): finalize v3.8.25 — reconcile CHANGELOG + i18n mirrors, document OMNIROUTE_MAX_PENDING_MIGRATIONS, green the unit suite Release-gate reconciliation for v3.8.25: - CHANGELOG: dated 2026-06-14, linked #3826, rolled up file-size re-baselines (#3823/#3833), recorded the test-greening; re-synced all 41 i18n CHANGELOG mirrors. - Documented OMNIROUTE_MAX_PENDING_MIGRATIONS (#3416) in .env.example + ENVIRONMENT.md. - Greened the unit suite (was merged red on 4 CI shards): aligned 10 stale tests to this cycle's intended behavior (#3838/#3822/#3501/SOCKS5/Vertex-Express/Antigravity) and the same-provider 503 fall-through test; de-flaked the compression benchmark reproducibility and ServiceSupervisor crash tests. No production code changed. * ci(security): clear OpenSSF Scorecard code-scanning noise + harden workflow token permissions The Security tab held 155 open alerts, ALL from the advisory OpenSSF Scorecard tool (#3824) — supply-chain/posture scores, not code vulnerabilities — which drowned out real CodeQL findings. - scorecard.yml: stop uploading SARIF to the code-scanning tab (drop the upload-sarif step + the now-unused security-events: write). The run still produces the OpenSSF badge (publish_results) and a downloadable SARIF artifact. - TokenPermissions hardening (the high-severity, genuinely-valuable subset): set each workflow's top-level token to read-only and grant the exact writes at the job level that needs them — npm-publish (id-token/packages on publish jobs), docker-publish (packages on build), electron-release (contents on build/release, id-token/packages on publish-npm), build-fork (packages on build), claude (empty top-level; job grants its own). The 155 existing alerts were dismissed. Not adopting repo-wide SHA-pinning (143 PinnedDependencies advisories) — declined. * test(integration): align stale wiring/socks5 integration tests to this cycle's behavior These were red on the CI Integration job (pre-existing). No production code changed: - integration-wiring: the combos page no longer renders a per-page EmailPrivacyToggle (#3822 consolidated it into Settings → Appearance); the provider-detail test-result masking and upstream-proxy copy moved to decomposed components (#3501 BatchTestResultsModal / UpstreamProxyCard) — assertions now read the owning files. - api-routes-critical: SOCKS5 is now enabled by default (opt-out), so the disabled- rejection test must set ENABLE_SOCKS5_PROXY=false explicitly (an unset env now means enabled). (The ~32 live-Gemini integration tests are gated on OMNIROUTE_API_KEY and skip in CI; they only 'fail' locally when that key is present without a running server.)
175 lines
6.6 KiB
TypeScript
175 lines
6.6 KiB
TypeScript
import { describe, it, before } from "node:test";
|
|
import assert from "node:assert/strict";
|
|
|
|
import {
|
|
BENCHMARK_CORPUS,
|
|
engineToCompressFn,
|
|
benchmarkEngines,
|
|
compareReports,
|
|
runBenchmarkGate,
|
|
} from "../../../open-sse/services/compression/harness/benchmark.ts";
|
|
|
|
// ── RED/GREEN proof: all assertions here must hold once benchmark.ts exists ──
|
|
|
|
describe("benchmark — engineToCompressFn adapter", () => {
|
|
it("returns a function for a known engine id", () => {
|
|
const fn = engineToCompressFn("rtk");
|
|
assert.equal(typeof fn, "function");
|
|
});
|
|
|
|
it("compressFn returns a string for any text input", async () => {
|
|
const fn = engineToCompressFn("rtk");
|
|
const noisy =
|
|
"Lorem ipsum dolor sit amet, consectetur adipiscing elit.\n".repeat(10) +
|
|
"Unnecessary filler words that add no value whatsoever indeed actually.";
|
|
const out = await fn(noisy);
|
|
assert.equal(typeof out, "string");
|
|
});
|
|
|
|
it("compressFn output is shorter or equal for clearly compressible prose", async () => {
|
|
// A highly redundant input that caveman rules should shorten
|
|
const fn = engineToCompressFn("caveman");
|
|
const repetitive =
|
|
"This is a very redundant message. This is a very redundant message.\n".repeat(8);
|
|
const out = await fn(repetitive);
|
|
// The adapter must return a string and it must not be longer than the input
|
|
assert.ok(
|
|
out.length <= repetitive.length,
|
|
`expected shorter output, got ${out.length} vs ${repetitive.length}`
|
|
);
|
|
});
|
|
});
|
|
|
|
describe("benchmark — benchmarkEngines", () => {
|
|
let reports: Awaited<ReturnType<typeof benchmarkEngines>>;
|
|
|
|
before(async () => {
|
|
reports = await benchmarkEngines(BENCHMARK_CORPUS, ["rtk", "caveman", "headroom"]);
|
|
});
|
|
|
|
it("returns one report per requested engine", () => {
|
|
assert.deepEqual(Object.keys(reports).sort(), ["caveman", "headroom", "rtk"]);
|
|
});
|
|
|
|
it("each report has a valid meanSavingsPercent (a number)", () => {
|
|
for (const [engineId, report] of Object.entries(reports)) {
|
|
assert.equal(
|
|
typeof report.meanSavingsPercent,
|
|
"number",
|
|
`${engineId}: meanSavingsPercent must be a number`
|
|
);
|
|
}
|
|
});
|
|
|
|
it("each report has meanRetention in [0, 1]", () => {
|
|
for (const [engineId, report] of Object.entries(reports)) {
|
|
assert.ok(
|
|
report.meanRetention >= 0 && report.meanRetention <= 1,
|
|
`${engineId}: meanRetention ${report.meanRetention} must be in [0,1]`
|
|
);
|
|
}
|
|
});
|
|
|
|
it("each report has results for every corpus item", () => {
|
|
for (const [engineId, report] of Object.entries(reports)) {
|
|
assert.equal(
|
|
report.results.length,
|
|
BENCHMARK_CORPUS.length,
|
|
`${engineId}: expected ${BENCHMARK_CORPUS.length} result(s), got ${report.results.length}`
|
|
);
|
|
}
|
|
});
|
|
});
|
|
|
|
describe("benchmark — compareReports", () => {
|
|
it("returns one summary row per engine", async () => {
|
|
const reports = await benchmarkEngines(BENCHMARK_CORPUS, ["rtk", "caveman"]);
|
|
const summary = compareReports(reports);
|
|
assert.equal(summary.length, 2);
|
|
for (const row of summary) {
|
|
assert.ok("engine" in row);
|
|
assert.ok("meanSavingsPercent" in row);
|
|
assert.ok("meanRetention" in row);
|
|
assert.ok("totalCompressedTokens" in row);
|
|
}
|
|
});
|
|
|
|
it("is sorted by meanSavingsPercent descending (best saver first)", async () => {
|
|
const reports = await benchmarkEngines(BENCHMARK_CORPUS, ["rtk", "caveman"]);
|
|
const summary = compareReports(reports);
|
|
for (let i = 1; i < summary.length; i++) {
|
|
assert.ok(
|
|
summary[i - 1].meanSavingsPercent >= summary[i].meanSavingsPercent,
|
|
`row ${i - 1} savings ${summary[i - 1].meanSavingsPercent} should be >= row ${i} savings ${summary[i].meanSavingsPercent}`
|
|
);
|
|
}
|
|
});
|
|
});
|
|
|
|
describe("benchmark — runBenchmarkGate (N4)", () => {
|
|
it("passes when baselines match current costs", async () => {
|
|
const reports = await benchmarkEngines(BENCHMARK_CORPUS, ["rtk"]);
|
|
// Baseline = exact current tokensPerTask → must pass
|
|
const rtkReport = reports["rtk"];
|
|
const baselines: Record<string, { tasks: Record<string, number> }> = {};
|
|
const taskTotals: Record<string, { sum: number; count: number }> = {};
|
|
for (const r of rtkReport.results) {
|
|
const t = taskTotals[r.task] ?? { sum: 0, count: 0 };
|
|
t.sum += r.compressedTokens;
|
|
t.count += 1;
|
|
taskTotals[r.task] = t;
|
|
}
|
|
baselines["rtk"] = {
|
|
tasks: Object.fromEntries(
|
|
Object.entries(taskTotals).map(([k, v]) => [k, Math.round(v.sum / v.count)])
|
|
),
|
|
};
|
|
|
|
const gateResults = runBenchmarkGate(reports, baselines);
|
|
const rtkGate = gateResults.find((g) => g.engine === "rtk");
|
|
assert.ok(rtkGate, "rtk gate result missing");
|
|
assert.equal(rtkGate.gate.passed, true, "gate should pass when baseline matches current");
|
|
});
|
|
|
|
it("fails (regression) when baseline is tighter than actual cost", async () => {
|
|
const reports = await benchmarkEngines(BENCHMARK_CORPUS, ["rtk"]);
|
|
// Set an impossibly tight baseline (1 token per task) → regression guaranteed
|
|
const impossibleBaselines: Record<string, { tasks: Record<string, number> }> = {
|
|
rtk: { tasks: { prose: 1, "tool-output": 1, json: 1 } },
|
|
};
|
|
const gateResults = runBenchmarkGate(reports, impossibleBaselines);
|
|
const rtkGate = gateResults.find((g) => g.engine === "rtk");
|
|
assert.ok(rtkGate, "rtk gate result missing");
|
|
assert.equal(rtkGate.gate.passed, false, "gate should fail when baseline is impossibly tight");
|
|
assert.ok(rtkGate.gate.regressions.length > 0, "regressions array must be non-empty");
|
|
});
|
|
});
|
|
|
|
describe("benchmark — reproducibility", () => {
|
|
it("two runs on the same corpus yield identical meanSavingsPercent", async () => {
|
|
const engines = ["rtk", "caveman"];
|
|
// Run the two passes SEQUENTIALLY: this asserts determinism (same input → same
|
|
// output), not concurrency-safety. Running them in parallel (Promise.all) shares the
|
|
// engine singletons across both passes and races their internal state under load.
|
|
const r1 = await benchmarkEngines(BENCHMARK_CORPUS, engines);
|
|
const r2 = await benchmarkEngines(BENCHMARK_CORPUS, engines);
|
|
for (const id of engines) {
|
|
assert.equal(
|
|
r1[id].meanSavingsPercent,
|
|
r2[id].meanSavingsPercent,
|
|
`${id}: non-deterministic meanSavingsPercent`
|
|
);
|
|
assert.equal(
|
|
r1[id].meanRetention,
|
|
r2[id].meanRetention,
|
|
`${id}: non-deterministic meanRetention`
|
|
);
|
|
assert.equal(
|
|
r1[id].totalCompressedTokens,
|
|
r2[id].totalCompressedTokens,
|
|
`${id}: non-deterministic totalCompressedTokens`
|
|
);
|
|
}
|
|
});
|
|
});
|