mirror of
https://github.com/diegosouzapw/OmniRoute.git
synced 2026-08-19 21:52:21 +03:00
* fix(compression): persist Headroom minRows (set 5 and reload keeps 5) Fixes diegosouzapw/OmniRoute#8056. Headroom detail settings had a Save-looking form but EngineConfigPage only persisted aggressive/ultra via SETTINGS_SUBOBJECT, so minRows always reseeded to the schema default (8) after reload. - Add HeadroomConfig + DEFAULT_HEADROOM_CONFIG (minRows: 8) - Accept headroom in compressionSettingsUpdateSchema (minRows 2..10000) - Normalize/store headroom in get/updateCompressionSettings - Register headroom in EngineConfigPage SETTINGS_SUBOBJECT so Save works - Merge settings.headroom into stacked stepConfig for runtime apply - Thread minRows through preview API + EngineConfigPage preview payload - Tests: schema/DB round-trip, engine apply, stacked merge, UI Save→PUT 5 * chore(quality): rebaseline compression.ts + strategySelector.ts own-growth (#8056 headroom minRows) --------- Co-authored-by: Ravi Tharuma <RaviTharuma@users.noreply.github.com>
346 lines
13 KiB
TypeScript
346 lines
13 KiB
TypeScript
import { NextResponse } from "next/server";
|
||
import { z } from "zod";
|
||
import { requireManagementAuth } from "@/lib/api/requireManagementAuth";
|
||
import { compressionPreviewConfigSchema } from "@/shared/validation/compressionConfigSchemas";
|
||
import {
|
||
applyCompression,
|
||
applyCompressionAsync,
|
||
} from "@omniroute/open-sse/services/compression/strategySelector";
|
||
import type {
|
||
CompressionConfig,
|
||
CompressionMode,
|
||
} from "@omniroute/open-sse/services/compression/types";
|
||
import {
|
||
buildCompressionPreviewDiff,
|
||
type HeatmapMode,
|
||
} from "@omniroute/open-sse/services/compression/diffHelper";
|
||
import { sanitizeErrorMessage } from "@omniroute/open-sse/utils/error";
|
||
import { countTextTokens } from "@/shared/utils/tiktokenCounter";
|
||
import {
|
||
ensureEngineBreakdown,
|
||
reconcileSingleEngineTokens,
|
||
} from "@omniroute/open-sse/services/compression/engineBreakdown";
|
||
import { summarizeEncoderCandidates } from "@omniroute/open-sse/services/compression/engines/headroom/encoderComparison";
|
||
import { DEFAULT_MIN_ROWS } from "@omniroute/open-sse/services/compression/engines/headroom/smartcrusher";
|
||
|
||
export const PreviewCompressionConfigSchema = compressionPreviewConfigSchema;
|
||
|
||
export const PreviewRequestSchema = z.object({
|
||
messages: z
|
||
.array(
|
||
z.object({
|
||
role: z.string(),
|
||
content: z.union([z.string(), z.array(z.unknown())]),
|
||
})
|
||
)
|
||
.min(1),
|
||
mode: z
|
||
.enum(["off", "lite", "standard", "aggressive", "ultra", "rtk", "stacked", "caveman"])
|
||
.optional()
|
||
.default("stacked"),
|
||
engineId: z.string().optional(),
|
||
pipeline: z.array(z.string()).min(1).optional(),
|
||
config: PreviewCompressionConfigSchema.optional(),
|
||
// Playground fidelity-gate toggle. Only `enabled` is exposed on the API surface on purpose:
|
||
// the advanced thresholds (minTokenSurvivalPercent / minJsonKeyPercent / checkNumericIntegrity
|
||
// / checkDiffHunks on FidelityGateConfig) use their conservative defaults until the studio gets
|
||
// a config panel for them.
|
||
fidelityGate: z.object({ enabled: z.boolean() }).optional(),
|
||
// Playground risk-gate toggle → masks high-risk spans (secrets/keys) before compression and
|
||
// restores them verbatim after, so they pass through byte-identical. Reported via
|
||
// result.stats.riskGate (spansProtected + per-category counts).
|
||
riskGate: z.object({ enabled: z.boolean() }).optional(),
|
||
// Playground fuzzy near-duplicate toggle → injects `{ fuzzy: { enabled: true } }` into the
|
||
// session-dedup step config (see buildStep).
|
||
fuzzyDedup: z.object({ enabled: z.boolean() }).optional(),
|
||
// Playground QuantumLock toggle. The studio is a dry-run, so when enabled we force a caching
|
||
// context (provider: "anthropic") so the operator can SEE what would be stabilized; real
|
||
// cache-hit gains only show in production provider telemetry.
|
||
quantumLock: z.object({ enabled: z.boolean() }).optional(),
|
||
// Saliency heatmap mode. When set, the response includes a per-token heatmap.
|
||
// "ultra" uses scoreToken (0–1); "universal" uses kept/removed from the diff.
|
||
// Omit to skip heatmap computation (normal preview path — no extra cost).
|
||
heatmap: z.enum(["ultra", "universal"]).optional(),
|
||
});
|
||
|
||
function countTokens(text: string): number {
|
||
return countTextTokens(text);
|
||
}
|
||
|
||
function riskGateStatsOf(result: { stats?: { riskGate?: unknown } }): unknown {
|
||
return result.stats?.riskGate ?? null;
|
||
}
|
||
|
||
function quantumLockStatsOf(result: { stats?: { quantumLock?: unknown } | null }): unknown {
|
||
return result.stats?.quantumLock ?? null;
|
||
}
|
||
|
||
function quantumExtras(quantumLock?: { enabled: boolean }) {
|
||
return quantumLock?.enabled
|
||
? {
|
||
configPatch: { quantumLock: { enabled: true } },
|
||
applyOpts: { cachingContext: { provider: "anthropic" } },
|
||
}
|
||
: { configPatch: {}, applyOpts: {} };
|
||
}
|
||
|
||
function messagesToText(messages: Array<{ role: string; content: unknown }>): string {
|
||
return messages
|
||
.map((m) => {
|
||
const content = typeof m.content === "string" ? m.content : JSON.stringify(m.content);
|
||
return `${m.role}: ${content}`;
|
||
})
|
||
.join("\n");
|
||
}
|
||
|
||
function buildStep(
|
||
engine: string,
|
||
fuzzy?: { enabled: boolean },
|
||
/** Optional detail bag (e.g. headroom.minRows from saved settings). */
|
||
detail?: Record<string, unknown>
|
||
) {
|
||
const config: Record<string, unknown> = { ...(detail ?? {}) };
|
||
if (engine === "session-dedup" && fuzzy?.enabled) {
|
||
config.fuzzy = { enabled: true };
|
||
}
|
||
return Object.keys(config).length > 0 ? { engine, config } : { engine };
|
||
}
|
||
|
||
function headroomParticipates(
|
||
engineId: string | undefined,
|
||
pipeline: string[] | undefined,
|
||
mode: CompressionMode
|
||
): boolean {
|
||
// An explicit single-engine or pipeline override decides on its own terms:
|
||
// headroom only participates if it is the engine / is named in the pipeline.
|
||
// (effectiveMode is forced to "stacked" whenever engineId/pipeline is set, so we
|
||
// must not fall through to the mode check for those — e.g. engineId:"lite".)
|
||
if (engineId) return engineId === "headroom";
|
||
if (pipeline) return pipeline.includes("headroom");
|
||
return mode === "stacked";
|
||
}
|
||
|
||
async function dispatchCompression(
|
||
requestBody: Record<string, unknown>,
|
||
opts: {
|
||
engineId?: string;
|
||
pipeline?: string[];
|
||
effectiveMode: CompressionMode;
|
||
config?: unknown;
|
||
fidelityGate?: { enabled: boolean };
|
||
fuzzyDedup?: { enabled: boolean };
|
||
riskGate?: { enabled: boolean };
|
||
quantumLock?: { enabled: boolean };
|
||
}
|
||
) {
|
||
// resolveRiskGate reads `options.riskGate ?? options.config.riskGate`. applyCompressionAsync
|
||
// does not surface a top-level `riskGate` option, so thread it through the synthesized config
|
||
// (CompressionConfig.riskGate) — uniform across all three branches and type-safe.
|
||
// QuantumLock uses the same pattern: when enabled the studio forces cachingContext so the dry-run
|
||
// badge shows what WOULD be stabilized in production (real caching gains show in telemetry only).
|
||
// When the client/settings carry a headroom detail sub-object, thread it so
|
||
// buildStepOptions can merge minRows into the headroom engine stepConfig (#8056).
|
||
const headroomDetail =
|
||
opts.config && typeof opts.config === "object" && opts.config !== null
|
||
? (opts.config as CompressionConfig).headroom
|
||
: undefined;
|
||
const headroomStepDetail =
|
||
headroomDetail && typeof headroomDetail.minRows === "number"
|
||
? { minRows: headroomDetail.minRows }
|
||
: undefined;
|
||
|
||
if (opts.engineId) {
|
||
const q = quantumExtras(opts.quantumLock);
|
||
return applyCompressionAsync(requestBody, "stacked", {
|
||
config: {
|
||
stackedPipeline: [
|
||
buildStep(
|
||
opts.engineId,
|
||
opts.fuzzyDedup,
|
||
opts.engineId === "headroom" ? headroomStepDetail : undefined
|
||
),
|
||
],
|
||
...(headroomDetail ? { headroom: headroomDetail } : {}),
|
||
...(opts.fidelityGate ? { fidelityGate: opts.fidelityGate } : {}),
|
||
...(opts.riskGate ? { riskGate: opts.riskGate } : {}),
|
||
...q.configPatch,
|
||
} as CompressionConfig,
|
||
...q.applyOpts,
|
||
});
|
||
}
|
||
if (opts.pipeline) {
|
||
const q = quantumExtras(opts.quantumLock);
|
||
return applyCompressionAsync(requestBody, "stacked", {
|
||
config: {
|
||
stackedPipeline: opts.pipeline.map((engine) =>
|
||
buildStep(engine, opts.fuzzyDedup, engine === "headroom" ? headroomStepDetail : undefined)
|
||
),
|
||
...(headroomDetail ? { headroom: headroomDetail } : {}),
|
||
...(opts.fidelityGate ? { fidelityGate: opts.fidelityGate } : {}),
|
||
...(opts.riskGate ? { riskGate: opts.riskGate } : {}),
|
||
...q.configPatch,
|
||
} as CompressionConfig,
|
||
...q.applyOpts,
|
||
});
|
||
}
|
||
const q = quantumExtras(opts.quantumLock);
|
||
return applyCompression(requestBody, opts.effectiveMode, {
|
||
config: {
|
||
...(opts.config as CompressionConfig | undefined),
|
||
...(opts.fidelityGate ? { fidelityGate: opts.fidelityGate } : {}),
|
||
...(opts.riskGate ? { riskGate: opts.riskGate } : {}),
|
||
...q.configPatch,
|
||
} as CompressionConfig | undefined,
|
||
...q.applyOpts,
|
||
});
|
||
}
|
||
|
||
export async function POST(req: Request) {
|
||
const authError = await requireManagementAuth(req);
|
||
if (authError) return authError;
|
||
|
||
let body: unknown;
|
||
try {
|
||
body = await req.json();
|
||
} catch {
|
||
return NextResponse.json({ error: "Invalid JSON body" }, { status: 400 });
|
||
}
|
||
|
||
const parsed = PreviewRequestSchema.safeParse(body);
|
||
if (!parsed.success) {
|
||
return NextResponse.json(
|
||
{ error: "Invalid request", details: parsed.error.issues },
|
||
{ status: 400 }
|
||
);
|
||
}
|
||
|
||
const { messages, mode, engineId: rawEngineId, pipeline, config, fidelityGate, fuzzyDedup, riskGate, quantumLock, heatmap: heatmapMode } =
|
||
parsed.data;
|
||
// Alias: `mode: "caveman"` is a synonym for `engineId: "caveman"` (single-engine stacked run).
|
||
// The caveman engine is not a top-level CompressionMode, but it IS a registered engine.
|
||
const engineId = mode === "caveman" && !rawEngineId ? "caveman" : rawEngineId;
|
||
const effectiveMode: CompressionMode =
|
||
engineId || pipeline ? "stacked" : (mode as CompressionMode);
|
||
const originalText = messagesToText(messages);
|
||
const originalTokens = countTokens(originalText);
|
||
|
||
try {
|
||
const start = Date.now();
|
||
const requestBody = { messages };
|
||
const result = await dispatchCompression(requestBody as Record<string, unknown>, {
|
||
engineId,
|
||
pipeline,
|
||
effectiveMode,
|
||
config,
|
||
fidelityGate,
|
||
fuzzyDedup,
|
||
riskGate,
|
||
quantumLock,
|
||
});
|
||
const durationMs = Date.now() - start;
|
||
|
||
const compressedMessages = (result.body.messages ?? messages) as Array<{
|
||
role: string;
|
||
content: unknown;
|
||
}>;
|
||
const compressedText = messagesToText(compressedMessages);
|
||
const compressedTokens = countTokens(compressedText);
|
||
const tokensSaved = Math.max(0, originalTokens - compressedTokens);
|
||
const savingsPct = originalTokens > 0 ? Math.round((tokensSaved / originalTokens) * 100) : 0;
|
||
const techniquesUsed: string[] = result.stats?.techniquesUsed ?? [];
|
||
const engineBreakdown = result.stats
|
||
? reconcileSingleEngineTokens(
|
||
ensureEngineBreakdown(result.stats),
|
||
originalTokens,
|
||
compressedTokens,
|
||
savingsPct
|
||
)
|
||
: [];
|
||
const diff = buildCompressionPreviewDiff(
|
||
originalText,
|
||
compressedText,
|
||
result.stats,
|
||
{},
|
||
heatmapMode as HeatmapMode | undefined
|
||
);
|
||
|
||
const headroomMinRows =
|
||
typeof config?.headroom?.minRows === "number" && Number.isFinite(config.headroom.minRows)
|
||
? config.headroom.minRows
|
||
: DEFAULT_MIN_ROWS;
|
||
const encoderComparison = headroomParticipates(engineId, pipeline, effectiveMode)
|
||
? summarizeEncoderCandidates(messages, headroomMinRows, countTextTokens)
|
||
: null;
|
||
|
||
// #6461: when fallbackApplied=true, synthesize a deduped reason list from data the
|
||
// pipeline already produces on result.stats (engineBreakdown[].rejectReason,
|
||
// validationErrors, and inflation-guard entries in validationWarnings). Non-fallback
|
||
// runs return []/null — zero change on the happy path.
|
||
const fallbackReasons: string[] = [];
|
||
if (diff.fallbackApplied) {
|
||
const seen = new Set<string>();
|
||
const push = (s: unknown) => {
|
||
if (typeof s === "string" && s.length > 0 && !seen.has(s)) {
|
||
seen.add(s);
|
||
fallbackReasons.push(s);
|
||
}
|
||
};
|
||
for (const step of engineBreakdown) {
|
||
if ((step as { rejected?: boolean }).rejected === true) {
|
||
push((step as { rejectReason?: string }).rejectReason);
|
||
}
|
||
}
|
||
for (const err of diff.validationErrors ?? []) push(err);
|
||
for (const warn of diff.validationWarnings ?? []) {
|
||
if (typeof warn === "string" && warn.startsWith("pipeline-inflation-guard:")) push(warn);
|
||
}
|
||
}
|
||
const fallbackReason = fallbackReasons[0] ?? null;
|
||
|
||
return NextResponse.json({
|
||
encoderComparison,
|
||
original: originalText,
|
||
compressed: compressedText,
|
||
originalTokens,
|
||
compressedTokens,
|
||
tokensSaved,
|
||
savingsPct,
|
||
techniquesUsed,
|
||
engineBreakdown,
|
||
riskGate: riskGateStatsOf(result),
|
||
quantumLock: quantumLockStatsOf(result),
|
||
durationMs,
|
||
mode: effectiveMode,
|
||
intensity: null,
|
||
outputMode: null,
|
||
skippedReasons: fallbackReasons,
|
||
diff: diff.segments,
|
||
preservedBlocks: diff.preservedBlocks,
|
||
ruleRemovals: diff.ruleRemovals,
|
||
rulesApplied: diff.ruleRemovals,
|
||
validation: {
|
||
valid: diff.validationErrors.length === 0,
|
||
errors: diff.validationErrors,
|
||
warnings: diff.validationWarnings,
|
||
fallbackApplied: diff.fallbackApplied,
|
||
...(diff.fallbackReason && { fallbackReason: diff.fallbackReason }),
|
||
},
|
||
validationWarnings: diff.validationWarnings,
|
||
validationErrors: diff.validationErrors,
|
||
fallbackApplied: diff.fallbackApplied,
|
||
// Prefer the pipeline's canonical `diff.fallbackReason`; fall back to the
|
||
// first synthesized reason (#6461) when the pipeline did not set one.
|
||
fallbackReason: diff.fallbackReason ?? fallbackReason,
|
||
fallbackReasons,
|
||
...(diff.heatmap ? { heatmap: diff.heatmap } : {}),
|
||
});
|
||
} catch (err: unknown) {
|
||
const msg = err instanceof Error ? err.message : String(err);
|
||
console.error("[/api/compression/preview]", msg);
|
||
return NextResponse.json(
|
||
{ error: "Compression failed", details: sanitizeErrorMessage(msg) },
|
||
{ status: 500 }
|
||
);
|
||
}
|
||
}
|