diff --git a/changelog.d/fixes/10249-dedup-hash-collision.md b/changelog.d/fixes/10249-dedup-hash-collision.md new file mode 100644 index 0000000000..f118196dfe --- /dev/null +++ b/changelog.d/fixes/10249-dedup-hash-collision.md @@ -0,0 +1 @@ +- fix(open-sse): stop concurrent requests colliding on the same dedup hash for non-OpenAI target formats (#10249) diff --git a/open-sse/services/requestDedup.ts b/open-sse/services/requestDedup.ts index 5ccde19529..6bf780bffc 100644 --- a/open-sse/services/requestDedup.ts +++ b/open-sse/services/requestDedup.ts @@ -36,12 +36,19 @@ const inflight = new Map>(); * Compute a deterministic hash for a request body. * Includes: model, messages, temperature, tools, tool_choice, max_tokens, response_format * Excludes: stream, user, metadata (don't affect LLM output) + * + * The prompt content can live under different keys depending on the target + * provider format the body has already been translated to: OpenAI-style + * bodies use `messages`, Gemini-translated bodies use `contents`, and + * Responses-API-translated bodies use `input`. Falling back to only + * `messages` made every non-OpenAI-format body hash the prompt as `null`, + * colliding different prompts onto the same dedup hash (#10249). */ export function computeRequestHash(requestBody: unknown): string { const body = requestBody as Record; const canonical = { model: body.model ?? null, - messages: body.messages ?? null, + messages: body.messages ?? body.contents ?? body.input ?? null, temperature: typeof body.temperature === "number" ? body.temperature : 1.0, tools: body.tools ?? null, tool_choice: body.tool_choice ?? null, diff --git a/tests/unit/request-dedup-10249.test.ts b/tests/unit/request-dedup-10249.test.ts new file mode 100644 index 0000000000..7926723b51 --- /dev/null +++ b/tests/unit/request-dedup-10249.test.ts @@ -0,0 +1,91 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { computeRequestHash, deduplicate, clearInflight } from "../../open-sse/services/requestDedup.ts"; + +// Regression tests for #10249: the dedup hash used to read only `body.messages`, +// so translated (target-format) bodies that carry the prompt under a different +// key (`contents` for Gemini, `input` for the Responses API) always hashed the +// prompt as `null`. Concurrent requests with different prompts then collided on +// the same dedup hash, joined the same in-flight promise, and the second caller +// silently received the first caller's response. + +test("Gemini-format translated bodies with different prompts must NOT collide on dedup hash", async () => { + clearInflight(); + const bodyA = { + contents: [{ role: "user", parts: [{ text: "Summarize the Q3 financial report attached." }] }], + temperature: 0, + }; + const bodyB = { + contents: [{ role: "user", parts: [{ text: "Extract every invoice number from the attached PDF." }] }], + temperature: 0, + }; + const hashA = computeRequestHash({ ...bodyA, model: "gemini/gemini-2.5-flash", stream: false }); + const hashB = computeRequestHash({ ...bodyB, model: "gemini/gemini-2.5-flash", stream: false }); + assert.notEqual(hashA, hashB, "Different prompts must have different dedup hashes"); + + const [resA, resB] = await Promise.all([ + deduplicate(hashA, async () => "RESPONSE_A"), + deduplicate(hashB, async () => "RESPONSE_B"), + ]); + assert.equal(resA.result, "RESPONSE_A"); + assert.equal(resB.result, "RESPONSE_B"); + assert.equal(resB.wasDeduplicated, false); +}); + +test("Responses-API input-format translated bodies with different prompts must NOT collide", async () => { + clearInflight(); + const bodyA = { + input: [{ role: "user", content: [{ type: "input_text", text: "What is the capital of France?" }] }], + temperature: 0, + }; + const bodyB = { + input: [{ role: "user", content: [{ type: "input_text", text: "Explain quantum entanglement." }] }], + temperature: 0, + }; + const hashA = computeRequestHash({ ...bodyA, model: "openai/gpt-4.1", stream: false }); + const hashB = computeRequestHash({ ...bodyB, model: "openai/gpt-4.1", stream: false }); + assert.notEqual(hashA, hashB, "Different prompts must have different dedup hashes"); + + const [resA, resB] = await Promise.all([ + deduplicate(hashA, async () => "RESPONSE_A"), + deduplicate(hashB, async () => "RESPONSE_B"), + ]); + assert.equal(resA.result, "RESPONSE_A"); + assert.equal(resB.result, "RESPONSE_B"); + assert.equal(resB.wasDeduplicated, false); +}); + +test("Sanity: OpenAI-format bodies with different prompts DO get distinct hashes (unchanged behavior)", () => { + const bodyA = { messages: [{ role: "user", content: "Hello there" }], temperature: 0 }; + const bodyB = { messages: [{ role: "user", content: "Goodbye now" }], temperature: 0 }; + const hashA = computeRequestHash({ ...bodyA, model: "openai/gpt-4.1", stream: false }); + const hashB = computeRequestHash({ ...bodyB, model: "openai/gpt-4.1", stream: false }); + assert.notEqual(hashA, hashB); +}); + +test("Genuinely identical requests still hash identically and get deduplicated (perf feature preserved)", async () => { + clearInflight(); + const body = { + contents: [{ role: "user", parts: [{ text: "Same prompt text every time" }] }], + temperature: 0, + }; + const hash1 = computeRequestHash({ ...body, model: "gemini/gemini-2.5-flash", stream: false }); + const hash2 = computeRequestHash({ ...body, model: "gemini/gemini-2.5-flash", stream: false }); + assert.equal(hash1, hash2, "Identical bodies must still produce the same hash"); + + let callCount = 0; + const slowFn = async () => { + callCount += 1; + await new Promise((resolve) => setTimeout(resolve, 20)); + return "SHARED_RESPONSE"; + }; + + const [resA, resB] = await Promise.all([ + deduplicate(hash1, slowFn), + deduplicate(hash2, slowFn), + ]); + assert.equal(resA.result, "SHARED_RESPONSE"); + assert.equal(resB.result, "SHARED_RESPONSE"); + assert.equal(callCount, 1, "Identical concurrent requests must share a single upstream call"); + assert.equal(resA.wasDeduplicated === true || resB.wasDeduplicated === true, true); +});