fix(open-sse): stop concurrent requests colliding on dedup hash for non-OpenAI formats

computeRequestHash() in requestDedup.ts projected the prompt content from
body.messages only. The dedup site in chatCore.ts hashes the *translated*
(target-format) request body, and non-OpenAI target formats don't carry a
messages field: Gemini-translated bodies use `contents`, Responses-API
bodies use `input`. So for those formats messages was always undefined,
every prompt hashed to the same null-backed value for a given model, and
concurrent requests with different prompts joined the same in-flight
promise -- the second caller silently received the first caller's
response verbatim (#10249).

Fix: project body.messages ?? body.contents ?? body.input ?? null instead
of only body.messages, keeping the rest of the canonical hash projection
unchanged. Genuinely identical concurrent requests still dedupe (the
intended perf behavior); different prompts under Gemini/Responses-API
target formats no longer collide.

Regression test: tests/unit/request-dedup-10249.test.ts reproduces the
two collision scenarios from the plan-file (Gemini `contents`,
Responses-API `input`), confirms the OpenAI `messages` case was already
correct, and asserts identical-request dedup keeps working. Verified
RED (byte-identical hashes 0b24fd88.../dc16d5b7... pre-fix) -> GREEN
(distinct hashes, dedup preserved) against this exact diff.
This commit is contained in:
adevwithpurpose
2026-08-14 22:41:49 -03:00
parent abd4df63dc
commit fb9984760c
3 changed files with 100 additions and 1 deletions

View File

@@ -0,0 +1 @@
- fix(open-sse): stop concurrent requests colliding on the same dedup hash for non-OpenAI target formats (#10249)

View File

@@ -36,12 +36,19 @@ const inflight = new Map<string, Promise<unknown>>();
* Compute a deterministic hash for a request body.
* Includes: model, messages, temperature, tools, tool_choice, max_tokens, response_format
* Excludes: stream, user, metadata (don't affect LLM output)
*
* The prompt content can live under different keys depending on the target
* provider format the body has already been translated to: OpenAI-style
* bodies use `messages`, Gemini-translated bodies use `contents`, and
* Responses-API-translated bodies use `input`. Falling back to only
* `messages` made every non-OpenAI-format body hash the prompt as `null`,
* colliding different prompts onto the same dedup hash (#10249).
*/
export function computeRequestHash(requestBody: unknown): string {
const body = requestBody as Record<string, unknown>;
const canonical = {
model: body.model ?? null,
messages: body.messages ?? null,
messages: body.messages ?? body.contents ?? body.input ?? null,
temperature: typeof body.temperature === "number" ? body.temperature : 1.0,
tools: body.tools ?? null,
tool_choice: body.tool_choice ?? null,

View File

@@ -0,0 +1,91 @@
import { test } from "node:test";
import assert from "node:assert/strict";
import { computeRequestHash, deduplicate, clearInflight } from "../../open-sse/services/requestDedup.ts";
// Regression tests for #10249: the dedup hash used to read only `body.messages`,
// so translated (target-format) bodies that carry the prompt under a different
// key (`contents` for Gemini, `input` for the Responses API) always hashed the
// prompt as `null`. Concurrent requests with different prompts then collided on
// the same dedup hash, joined the same in-flight promise, and the second caller
// silently received the first caller's response.
test("Gemini-format translated bodies with different prompts must NOT collide on dedup hash", async () => {
clearInflight();
const bodyA = {
contents: [{ role: "user", parts: [{ text: "Summarize the Q3 financial report attached." }] }],
temperature: 0,
};
const bodyB = {
contents: [{ role: "user", parts: [{ text: "Extract every invoice number from the attached PDF." }] }],
temperature: 0,
};
const hashA = computeRequestHash({ ...bodyA, model: "gemini/gemini-2.5-flash", stream: false });
const hashB = computeRequestHash({ ...bodyB, model: "gemini/gemini-2.5-flash", stream: false });
assert.notEqual(hashA, hashB, "Different prompts must have different dedup hashes");
const [resA, resB] = await Promise.all([
deduplicate(hashA, async () => "RESPONSE_A"),
deduplicate(hashB, async () => "RESPONSE_B"),
]);
assert.equal(resA.result, "RESPONSE_A");
assert.equal(resB.result, "RESPONSE_B");
assert.equal(resB.wasDeduplicated, false);
});
test("Responses-API input-format translated bodies with different prompts must NOT collide", async () => {
clearInflight();
const bodyA = {
input: [{ role: "user", content: [{ type: "input_text", text: "What is the capital of France?" }] }],
temperature: 0,
};
const bodyB = {
input: [{ role: "user", content: [{ type: "input_text", text: "Explain quantum entanglement." }] }],
temperature: 0,
};
const hashA = computeRequestHash({ ...bodyA, model: "openai/gpt-4.1", stream: false });
const hashB = computeRequestHash({ ...bodyB, model: "openai/gpt-4.1", stream: false });
assert.notEqual(hashA, hashB, "Different prompts must have different dedup hashes");
const [resA, resB] = await Promise.all([
deduplicate(hashA, async () => "RESPONSE_A"),
deduplicate(hashB, async () => "RESPONSE_B"),
]);
assert.equal(resA.result, "RESPONSE_A");
assert.equal(resB.result, "RESPONSE_B");
assert.equal(resB.wasDeduplicated, false);
});
test("Sanity: OpenAI-format bodies with different prompts DO get distinct hashes (unchanged behavior)", () => {
const bodyA = { messages: [{ role: "user", content: "Hello there" }], temperature: 0 };
const bodyB = { messages: [{ role: "user", content: "Goodbye now" }], temperature: 0 };
const hashA = computeRequestHash({ ...bodyA, model: "openai/gpt-4.1", stream: false });
const hashB = computeRequestHash({ ...bodyB, model: "openai/gpt-4.1", stream: false });
assert.notEqual(hashA, hashB);
});
test("Genuinely identical requests still hash identically and get deduplicated (perf feature preserved)", async () => {
clearInflight();
const body = {
contents: [{ role: "user", parts: [{ text: "Same prompt text every time" }] }],
temperature: 0,
};
const hash1 = computeRequestHash({ ...body, model: "gemini/gemini-2.5-flash", stream: false });
const hash2 = computeRequestHash({ ...body, model: "gemini/gemini-2.5-flash", stream: false });
assert.equal(hash1, hash2, "Identical bodies must still produce the same hash");
let callCount = 0;
const slowFn = async () => {
callCount += 1;
await new Promise((resolve) => setTimeout(resolve, 20));
return "SHARED_RESPONSE";
};
const [resA, resB] = await Promise.all([
deduplicate(hash1, slowFn),
deduplicate(hash2, slowFn),
]);
assert.equal(resA.result, "SHARED_RESPONSE");
assert.equal(resB.result, "SHARED_RESPONSE");
assert.equal(callCount, 1, "Identical concurrent requests must share a single upstream call");
assert.equal(resA.wasDeduplicated === true || resB.wasDeduplicated === true, true);
});