Files
9router/tests/unit/kiro-thinking-strip.test.js
T
Edison42 9c58ba645e fix(kiro): improve direct session cache reuse
Reshape Kiro direct requests so resumed client sessions reuse Kiro's
cache-affinity fields instead of starting unrelated CodeWhisperer
conversations.

- keep conversationState.conversationId stable when the client sends an
  explicit session id (x-session-id, session_id, conversation_id, Claude
  Code session metadata)
- add a stable conversationState.agentContinuationId per Kiro session
- send conversationState.agentTaskType: "vibe" and agentMode: "vibe",
  matching the normal Kiro CLI/KAS chat path
- move Kiro thinking instructions into Kiro-compatible systemPrompt /
  additionalModelRequestFields instead of generic top-level thinking
- keep volatile timestamp context out of the top-level systemPrompt; it
  remains only in user content fallback
- suppress additionalModelRequestFields for legacy 4.5-era Claude/Kiro
  models that reject it, while defaulting future Claude/Kiro model ids
  to supported
- preserve Kiro meteringEvent credit usage internally for accounting
  without leaking provider-specific fields into OpenAI-compatible usage
- prevent unrelated headerless Kiro requests from sharing one
  connection-wide continuation
- cap/evict continuation sessions so long-running processes do not grow
  the continuation map unbounded
- treat generated headerless Kiro sessions as one-shot so they do not
  evict real explicit-session continuations
- keep credit-only Kiro metering valid for internal persistence when
  token metrics are unavailable
2026-07-16 15:15:05 +07:00

182 lines
6.4 KiB
JavaScript

import { describe, it, expect } from "vitest";
import { KiroExecutor } from "../../open-sse/executors/kiro.js";
import "../translator/registerAll.js";
function createMockFrame(eventType, payloadObj) {
const payloadStr = JSON.stringify(payloadObj);
const payloadBytes = new TextEncoder().encode(payloadStr);
const headerName = ":event-type";
const headerNameBytes = new TextEncoder().encode(headerName);
const headerValueBytes = new TextEncoder().encode(eventType);
// nameLen(1) + name + type(1) + valueLen(2) + value
const headerLength = 1 + headerNameBytes.length + 1 + 2 + headerValueBytes.length;
const totalLength = 12 + headerLength + payloadBytes.length + 4;
const buffer = new Uint8Array(totalLength);
const view = new DataView(buffer.buffer);
view.setUint32(0, totalLength, false);
view.setUint32(4, headerLength, false);
let offset = 12;
buffer[offset++] = headerNameBytes.length;
buffer.set(headerNameBytes, offset);
offset += headerNameBytes.length;
buffer[offset++] = 7; // String type
view.setUint16(offset, headerValueBytes.length, false);
offset += 2;
buffer.set(headerValueBytes, offset);
offset += headerValueBytes.length;
buffer.set(payloadBytes, offset);
return buffer;
}
async function readAllSSE(stream) {
const reader = stream.getReader();
const decoder = new TextDecoder();
let result = "";
while (true) {
const { done, value } = await reader.read();
if (done) break;
result += decoder.decode(value, { stream: true });
}
return result;
}
async function readNextWithTimeout(reader) {
return Promise.race([
reader.read(),
new Promise((_, reject) => setTimeout(() => reject(new Error("timed out waiting for SSE chunk")), 100)),
]);
}
describe("KiroExecutor thinking tag stripping", () => {
it("strips <thinking> tags from assistantResponseEvent", async () => {
const executor = new KiroExecutor();
// Create frames
const f1 = createMockFrame("assistantResponseEvent", { content: "Here is my answer. <thinking>Let me think..." });
const f2 = createMockFrame("assistantResponseEvent", { content: "still thinking...</thinking> Yes, 42." });
const readableStream = new ReadableStream({
start(controller) {
controller.enqueue(f1);
controller.enqueue(f2);
controller.close();
}
});
const mockResponse = { body: readableStream };
const transformedResponse = executor.transformEventStreamToSSE(mockResponse, "claude-test");
const output = await readAllSSE(transformedResponse.body);
// Check that we got chat.completion.chunk outputs
expect(output).toContain("chat.completion.chunk");
// Ensure the thinking parts are gone
expect(output).not.toContain("<thinking>");
expect(output).not.toContain("Let me think...");
expect(output).not.toContain("still thinking...");
expect(output).not.toContain("</thinking>");
// Check that the normal content is preserved
// Parse the data chunks
const dataLines = output.split("\n").filter(line => line.startsWith("data: "));
const contents = dataLines.map(line => {
if (line.includes("[DONE]")) return "";
try {
return JSON.parse(line.slice(6)).choices[0].delta.content || "";
} catch {
return "";
}
});
const fullText = contents.join("");
expect(fullText).toBe("Here is my answer. Yes, 42.");
});
it("handles empty content after stripping when hasReasoningContent is true", async () => {
const executor = new KiroExecutor();
const f0 = createMockFrame("reasoningContentEvent", { text: "I am reasoning" });
const f1 = createMockFrame("assistantResponseEvent", { content: "<thinking>purely thinking...</thinking>" });
const readableStream = new ReadableStream({
start(controller) {
controller.enqueue(f0);
controller.enqueue(f1);
controller.close();
}
});
const mockResponse = { body: readableStream };
const transformedResponse = executor.transformEventStreamToSSE(mockResponse, "claude-test");
const output = await readAllSSE(transformedResponse.body);
const dataLines = output.split("\n").filter(line => line.startsWith("data: ") && !line.includes("[DONE]"));
const objects = dataLines.map(line => JSON.parse(line.slice(6)));
// First chunk should have reasoning_content
expect(objects[0].choices[0].delta.reasoning_content).toBe("I am reasoning");
// We shouldn't get an empty content chunk from f1 since it was entirely stripped and reasoning was present
const contentChunks = objects.filter(obj => obj.choices[0].delta.content !== undefined);
expect(contentChunks.length).toBe(0);
});
it("emits a terminal chunk at messageStop before the upstream stream closes", async () => {
const executor = new KiroExecutor();
const f1 = createMockFrame("assistantResponseEvent", { content: "OK" });
const f2 = createMockFrame("messageStopEvent", {});
const readableStream = new ReadableStream({
start(controller) {
controller.enqueue(f1);
controller.enqueue(f2);
}
});
const transformedResponse = executor.transformEventStreamToSSE({ body: readableStream }, "claude-test");
const reader = transformedResponse.body.getReader();
const decoder = new TextDecoder();
let output = "";
for (let i = 0; i < 4 && !output.includes("\"finish_reason\":\"stop\""); i++) {
const { value } = await readNextWithTimeout(reader);
output += decoder.decode(value, { stream: true });
}
await reader.cancel();
expect(output).toContain("\"finish_reason\":\"stop\"");
});
it("uses tool_calls finish reason for tool streams without messageStop", async () => {
const executor = new KiroExecutor();
const f1 = createMockFrame("toolUseEvent", { toolUseId: "tool-1", name: "read_file", input: { path: "a.txt" } });
const readableStream = new ReadableStream({
start(controller) {
controller.enqueue(f1);
controller.close();
}
});
const transformedResponse = executor.transformEventStreamToSSE({ body: readableStream }, "claude-test");
const output = await readAllSSE(transformedResponse.body);
const objects = output
.split("\n")
.filter(line => line.startsWith("data: ") && !line.includes("[DONE]"))
.map(line => JSON.parse(line.slice(6)));
const finalChunk = objects.at(-1);
expect(finalChunk.choices[0].finish_reason).toBe("tool_calls");
});
});