feat(sse): surface Kiro adaptive-thinking reasoning as reasoning_content (#6213)

Kiro/CodeWhisperer streams Claude's reasoning as native `reasoningContentEvent`
frames when adaptive thinking is enabled, but the Kiro executor had no handler
for them, so `reasoning_effort` requests returned no reasoning. Wire it end to
end:

- translator (openai-to-kiro): enable Kiro thinking when the request carries
  `reasoning_effort`, Anthropic `output_config.effort`, or a `thinking` block
  (`{type:"enabled",budget_tokens}` mapped to a level; `{type:"adaptive"}`
  defaults to `high`, matching Anthropic's documented default). Prepends the
  Kiro `<thinking_mode>`/`<max_thinking_length>` prompt directive and sets
  top-level `additionalModelRequestFields` ({output_config.effort,
  thinking:{type:"adaptive"}, max_tokens}). Gated on `supportsReasoning`; drops
  non-default temperature/top_p (rejected by adaptive-only Claude models).
- executor transformRequest: forward `additionalModelRequestFields` to AWS
  (previously dropped by the strict top-level allowlist).
- executor stream loop: parse `reasoningContentEvent` (and reasoningText
  variants) into the OpenAI reasoning_content channel.

Verified against the live CodeWhisperer stream: reasoningContentEvent frames are
returned, and larger effort/budget measurably deepens reasoning up to the model
cap. Unit tests cover the effort sources, forwarding, temp/top_p stripping, and
native reasoning-frame parsing.
This commit is contained in:
VXNCXNX
2026-07-05 07:30:21 +02:00
committed by GitHub
parent b074c6d75e
commit dc7eeba717
4 changed files with 384 additions and 0 deletions

View File

@@ -214,6 +214,12 @@ export class KiroExecutor extends BaseExecutor {
if (b.conversationState !== undefined) kiroPayload.conversationState = b.conversationState;
if (b.profileArn !== undefined) kiroPayload.profileArn = b.profileArn;
if (b.inferenceConfig !== undefined) kiroPayload.inferenceConfig = b.inferenceConfig;
// Thinking control: `additionalModelRequestFields` ({output_config.effort,
// thinking:{type:"adaptive"}, max_tokens}) is a recognized top-level field on
// GenerateAssistantResponse — it steers adaptive reasoning. Built by the
// openai-to-kiro translator only when the request asked for thinking.
if (b.additionalModelRequestFields !== undefined)
kiroPayload.additionalModelRequestFields = b.additionalModelRequestFields;
// Fallback: if somehow conversationState isn't there, return the rest without model
// (for backward compatibility if something else bypasses the translator)
@@ -382,6 +388,56 @@ export class KiroExecutor extends BaseExecutor {
if (!state.totalContentLength) state.totalContentLength = 0;
if (!state.contextUsagePercentage) state.contextUsagePercentage = 0;
// Native reasoning frames. Verified against the live CodeWhisperer
// stream (2026-07): with adaptive thinking enabled (via
// additionalModelRequestFields), Kiro streams reasoning as a dedicated
// `reasoningContentEvent` frame carrying `{ text, signature }` — NOT
// inline `<thinking>` tags and NOT `assistantResponseEvent`. Some
// models/variants instead use a `reasoningText` object or a flat
// `{ text }` (cf. javargasm/pi-kiro `src/event-parser.ts`). OmniRoute
// had no handler for this event, so the reasoning was silently dropped;
// route it to the OpenAI `reasoning_content` channel.
{
const rp = event.payload as Record<string, unknown> | undefined;
const rt = rp?.reasoningText;
if (eventType === "reasoningContentEvent" || rt !== undefined) {
let nativeReasoning = "";
if (rt && typeof rt === "object") {
const rto = rt as { text?: unknown; Text?: unknown };
nativeReasoning =
typeof rto.text === "string"
? rto.text
: typeof rto.Text === "string"
? rto.Text
: "";
} else if (typeof rt === "string") {
nativeReasoning = rt;
} else if (typeof rp?.text === "string") {
nativeReasoning = rp.text as string;
}
if (nativeReasoning) {
state.hasReasoningContent = true;
const reasoningDelta: JsonRecord =
(state.reasoningChunkCount ?? 0) === 0 && chunkIndex === 0
? { role: "assistant", reasoning_content: nativeReasoning }
: { reasoning_content: nativeReasoning };
const chunk: JsonRecord = {
id: responseId,
object: "chat.completion.chunk",
created,
model,
choices: [{ index: 0, delta: reasoningDelta, finish_reason: null }],
};
chunkIndex++;
state.reasoningChunkCount = (state.reasoningChunkCount ?? 0) + 1;
controller.enqueue(TEXT_ENCODER.encode(`data: ${JSON.stringify(chunk)}\n\n`));
}
// Consume the reasoning frame (incl. signature-only) so it never
// falls through to the content handlers below.
continue;
}
}
// Handle assistantResponseEvent
if (eventType === "assistantResponseEvent") {
const content =

View File

@@ -5,6 +5,7 @@
import { register } from "../registry.ts";
import { FORMATS } from "../formats.ts";
import { v4 as uuidv4, v5 as uuidv5 } from "uuid";
import { capMaxOutputTokens, capThinkingBudget, supportsReasoning } from "@/lib/modelCapabilities";
import {
parseToolInput,
normalizeKiroToolSchema,
@@ -575,6 +576,80 @@ function convertMessages(messages, tools, model) {
return { history: alternatingHistory, currentMessage, toolsAttached };
}
/** Kiro's accepted reasoning-effort levels (`output_config.effort`). */
const KIRO_EFFORT_LEVELS = ["low", "medium", "high", "xhigh", "max"];
/**
* Resolve the Kiro effort level for a request, or "" when no reasoning was asked
* for. Effort sources, in priority order:
* 1. OpenAI-style `reasoning_effort`
* 2. Anthropic adaptive-thinking `output_config.effort` (the canonical field)
* 3. Anthropic `thinking` block — `{type:"enabled", budget_tokens}` mapped to a
* level via {@link effortFromBudget}; `{type:"adaptive"}` (no explicit
* effort) defaults to `high`, matching Anthropic's documented default
* (omitting `effort` ≡ `high`).
* OpenAI's `minimal` collapses to `low` (Kiro has no `minimal`).
*/
function resolveKiroEffort(body: Record<string, unknown>): string {
let effort = typeof body.reasoning_effort === "string" ? body.reasoning_effort.toLowerCase() : "";
if (!effort) {
const outputConfig = body.output_config as Record<string, unknown> | undefined;
if (
outputConfig &&
typeof outputConfig === "object" &&
typeof outputConfig.effort === "string"
) {
effort = outputConfig.effort.toLowerCase();
}
}
if (!effort) {
const thinking = body.thinking as Record<string, unknown> | undefined;
if (thinking && typeof thinking === "object") {
if (thinking.type === "enabled") {
effort = effortFromBudget(Number(thinking.budget_tokens) || 0);
} else if (thinking.type === "adaptive") {
effort = "high";
}
}
}
if (effort === "minimal") effort = "low";
return KIRO_EFFORT_LEVELS.includes(effort) ? effort : "";
}
/** Map an Anthropic `thinking.budget_tokens` to a coarse Kiro effort level. */
function effortFromBudget(budget: number): string {
if (budget >= 32000) return "high";
if (budget >= 16000) return "medium";
if (budget > 0) return "low";
return "";
}
/**
* Soft `<max_thinking_length>` budget for the Kiro prompt directive, per effort
* level. Anthropic publishes no effort→token mapping (effort is "a behavioral
* signal, not a strict token budget"), so this is a heuristic tuned against the
* live CodeWhisperer stream, where a larger budget measurably deepens reasoning
* up to the model cap. It is a hint the model may honor, not a hard cap (the hard
* enable signal is `<thinking_mode>`); the caller clamps it to the model's cap.
*/
function thinkingLengthForEffort(effort: string): number {
switch (effort) {
case "max":
return 120000;
case "xhigh":
return 64000;
case "high":
return 32000;
case "medium":
return 16000;
default:
return 8000; // low
}
}
/**
* Build Kiro payload from OpenAI format
*/
@@ -683,6 +758,11 @@ export function buildKiroPayload(model, body, stream, credentials) {
temperature?: number;
topP?: number;
};
additionalModelRequestFields?: {
thinking?: { type: string; display?: string };
output_config?: { effort: string };
max_tokens?: number;
};
} = {
conversationState: {
chatTriggerType: "MANUAL",
@@ -754,6 +834,55 @@ export function buildKiroPayload(model, body, stream, credentials) {
if (topP !== undefined) payload.inferenceConfig.topP = topP;
}
// Thinking mode for Claude models on Kiro (ported from javargasm/pi-kiro).
// Two coordinated signals steer reasoning on the CodeWhisperer surface:
// 1. a `<thinking_mode>enabled</thinking_mode><max_thinking_length>N</...>`
// directive prepended to the current user message — makes Claude emit its
// reasoning INLINE as `<thinking>…</thinking>`, which the Kiro executor
// splits back into the OpenAI `reasoning_content` channel (kiroThinking.ts);
// 2. top-level `additionalModelRequestFields` (output_config.effort +
// thinking:{type:"adaptive"} + a clamped max_tokens), forwarded to AWS by
// the Kiro executor's transformRequest allowlist — this is the graded
// effort lever. Gated on models that advertise thinking support.
const kiroEffort = supportsReasoning(normalizedModel) ? resolveKiroEffort(body) : "";
if (kiroEffort) {
// `<thinking_mode>` / `<max_thinking_length>` are Kiro/CodeWhisperer prompt
// conventions (NOT Anthropic API params); the length is a soft hint (the hard
// enable signal is `<thinking_mode>`), clamped to the model's thinking cap.
const thinkingLength = capThinkingBudget(normalizedModel, thinkingLengthForEffort(kiroEffort));
const directive =
`<thinking_mode>enabled</thinking_mode>` +
`<max_thinking_length>${thinkingLength}</max_thinking_length>`;
payload.conversationState.currentMessage.userInputMessage.content = `${directive}\n\n${payload.conversationState.currentMessage.userInputMessage.content}`;
const fields: {
output_config: { effort: string };
thinking: { type: string; display: string };
max_tokens?: number;
} = {
output_config: { effort: kiroEffort },
thinking: { type: "adaptive", display: "summarized" },
};
// Forward max_tokens only when the client set one, clamped to the model's
// output window (floor 1024) — matches pi-kiro and avoids an over-budget reject.
if (maxTokens > 0) {
const capped = capMaxOutputTokens(normalizedModel, maxTokens) ?? maxTokens;
fields.max_tokens = Math.max(Math.floor(capped), 1024);
}
payload.additionalModelRequestFields = fields;
// Adaptive-only Claude models (Opus 4.7/4.8, Sonnet 5, Fable 5) reject a
// non-default temperature / top_p with a 400 while thinking is active, so
// strip both. Drop inferenceConfig entirely if nothing else remains.
if (payload.inferenceConfig) {
delete payload.inferenceConfig.temperature;
delete payload.inferenceConfig.topP;
if (Object.keys(payload.inferenceConfig).length === 0) {
delete payload.inferenceConfig;
}
}
}
return payload;
}

View File

@@ -168,6 +168,31 @@ test("KiroExecutor.transformRequest removes the top-level model field", () => {
);
});
test("KiroExecutor.transformRequest forwards additionalModelRequestFields (thinking) to AWS", () => {
const executor = new KiroExecutor();
const body = {
model: "kiro-model",
conversationState: {
currentMessage: { userInputMessage: { modelId: "kiro-model" } },
},
additionalModelRequestFields: {
output_config: { effort: "high" },
thinking: { type: "adaptive", display: "summarized" },
max_tokens: 32000,
},
};
const result = executor.transformRequest("kiro-model", body, true, {}) as any;
// The thinking control must survive the strict allowlist — otherwise graded
// reasoning never reaches CodeWhisperer (the field the openai-to-kiro
// translator builds would be silently dropped).
assert.deepEqual(result.additionalModelRequestFields, {
output_config: { effort: "high" },
thinking: { type: "adaptive", display: "summarized" },
max_tokens: 32000,
});
});
test("KiroExecutor.transformEventStreamToSSE converts text, tool calls, usage and DONE", async () => {
const executor = new KiroExecutor();
const invalidPreludeFrame = buildEventFrame("assistantResponseEvent", { content: "skip me" });
@@ -202,6 +227,39 @@ test("KiroExecutor.transformEventStreamToSSE converts text, tool calls, usage an
assert.match(text, /\[DONE\]/);
});
test("KiroExecutor.transformEventStreamToSSE surfaces native reasoning frames as reasoning_content", async () => {
const executor = new KiroExecutor();
// Verified live wire format: Kiro streams adaptive-thinking reasoning as a
// dedicated `reasoningContentEvent` frame carrying `{ text, signature }`. Also
// cover the `reasoningText` object variant and a signature-only frame.
const response = buildEventStreamResponse([
buildEventFrame("reasoningContentEvent", { text: "Let me think... " }),
buildEventFrame("reasoningContentEvent", { text: "step two. " }),
buildEventFrame("reasoningContentEvent", { signature: "sig-only-frame" }),
buildEventFrame("assistantResponseEvent", { reasoningText: { text: "variant." } }),
buildEventFrame("assistantResponseEvent", { content: "The answer is 42." }),
buildEventFrame("metricsEvent", { inputTokens: 3, outputTokens: 5 }),
]);
const transformed = executor.transformEventStreamToSSE(response, "kiro-model");
const chunks = parseSSEJsonChunks(await transformed.text());
const reasoning = chunks
.map((c) => c.choices?.[0]?.delta?.reasoning_content)
.filter(Boolean)
.join("");
const content = chunks
.map((c) => c.choices?.[0]?.delta?.content)
.filter(Boolean)
.join("");
assert.equal(
reasoning,
"Let me think... step two. variant.",
"reasoningContentEvent frames + reasoningText variant must all surface"
);
assert.match(content, /The answer is 42\./, "normal content must still flow");
});
test("KiroExecutor.transformEventStreamToSSE parses fragmented frames and waits for post-stop usage", async () => {
const executor = new KiroExecutor();
const bytes = concatArrays(

View File

@@ -1090,3 +1090,144 @@ test("buildKiroPayload leaves already-two-dash Claude ids unchanged (#2270)", ()
"two-dash form (patch + date) must remain unchanged"
);
});
test("buildKiroPayload enables thinking mode for Claude models via reasoning_effort", () => {
const body = {
messages: [{ role: "user", content: "Solve a hard problem" }],
reasoning_effort: "high",
max_tokens: 64000,
};
const result = buildKiroPayload("claude-opus-4.8", body, false, null);
assert.ok(result.additionalModelRequestFields, "additionalModelRequestFields must be set");
assert.deepEqual(result.additionalModelRequestFields.thinking, {
type: "adaptive",
display: "summarized",
});
assert.equal(result.additionalModelRequestFields.output_config.effort, "high");
assert.equal(result.additionalModelRequestFields.max_tokens, 64000);
assert.match(
result.conversationState.currentMessage.userInputMessage.content,
/<thinking_mode>enabled<\/thinking_mode>/,
"thinking_mode directive must be injected into user content"
);
assert.match(
result.conversationState.currentMessage.userInputMessage.content,
/<max_thinking_length>\d+<\/max_thinking_length>/,
"max_thinking_length directive must be injected into user content"
);
});
test("buildKiroPayload drops temperature when thinking is enabled", () => {
const body = {
messages: [{ role: "user", content: "Solve a hard problem" }],
reasoning_effort: "high",
temperature: 0.5,
};
const result = buildKiroPayload("claude-opus-4.8", body, false, null);
assert.ok(result.additionalModelRequestFields, "thinking must be enabled");
assert.equal(
result.inferenceConfig?.temperature,
undefined,
"temperature must be dropped when adaptive thinking is active"
);
});
test("buildKiroPayload ignores thinking request for unsupported effort levels", () => {
const body = {
messages: [{ role: "user", content: "Hello" }],
reasoning_effort: "invalid",
};
const result = buildKiroPayload("claude-opus-4.8", body, false, null);
assert.equal(
result.additionalModelRequestFields,
undefined,
"invalid effort must not enable thinking"
);
});
test("buildKiroPayload maps body.thinking budget_tokens to effort level", () => {
const body = {
messages: [{ role: "user", content: "Deep reasoning" }],
thinking: { type: "enabled", budget_tokens: 50000 },
};
const result = buildKiroPayload("claude-opus-4.7", body, false, null);
assert.ok(result.additionalModelRequestFields, "thinking must be enabled from budget_tokens");
assert.equal(result.additionalModelRequestFields.output_config.effort, "high");
});
test("buildKiroPayload leaves thinking off when no reasoning is requested", () => {
const body = { messages: [{ role: "user", content: "Hi" }] };
const result = buildKiroPayload("claude-opus-4.8", body, false, null);
assert.equal(result.additionalModelRequestFields, undefined, "no thinking fields by default");
assert.doesNotMatch(
result.conversationState.currentMessage.userInputMessage.content,
/<thinking_mode>/,
"no directive injected by default"
);
});
test("buildKiroPayload maps reasoning_effort to the same Kiro effort level (no +1 shift)", () => {
const result = buildKiroPayload(
"claude-sonnet-5",
{ messages: [{ role: "user", content: "hard" }], reasoning_effort: "medium" },
false,
null
);
assert.equal(result.additionalModelRequestFields.output_config.effort, "medium");
});
test("buildKiroPayload reads effort from Anthropic output_config.effort", () => {
const result = buildKiroPayload(
"claude-opus-4.8",
{ messages: [{ role: "user", content: "hard" }], output_config: { effort: "xhigh" } },
false,
null
);
assert.ok(result.additionalModelRequestFields, "output_config.effort must enable thinking");
assert.equal(result.additionalModelRequestFields.output_config.effort, "xhigh");
});
test("buildKiroPayload defaults adaptive thinking (no effort) to high", () => {
const result = buildKiroPayload(
"claude-opus-4.8",
{ messages: [{ role: "user", content: "hard" }], thinking: { type: "adaptive" } },
false,
null
);
assert.equal(
result.additionalModelRequestFields.output_config.effort,
"high",
"adaptive with no explicit effort defaults to Anthropic's documented default (high)"
);
});
test("buildKiroPayload drops both temperature and top_p when thinking is enabled", () => {
const result = buildKiroPayload(
"claude-opus-4.8",
{
messages: [{ role: "user", content: "hard" }],
reasoning_effort: "high",
temperature: 0.5,
top_p: 0.9,
},
false,
null
);
assert.ok(result.additionalModelRequestFields, "thinking must be enabled");
assert.equal(result.inferenceConfig?.temperature, undefined, "temperature must be dropped");
assert.equal(result.inferenceConfig?.topP, undefined, "top_p must be dropped");
});