diff --git a/extensions/qa-lab/src/providers/mock-openai/server.test.ts b/extensions/qa-lab/src/providers/mock-openai/server.test.ts index 14ea9f4a0354..9d456e13b436 100644 --- a/extensions/qa-lab/src/providers/mock-openai/server.test.ts +++ b/extensions/qa-lab/src/providers/mock-openai/server.test.ts @@ -33,7 +33,7 @@ afterEach(async () => { } }); -async function startMockServer(params?: { finalOnlyMarkerPauseMs?: number }) { +async function startMockServer(params?: { finalOnlyMarkerPauseMs?: number; modelRefs?: string[] }) { const server = await startQaMockOpenAiServer({ host: "127.0.0.1", port: 0, @@ -45,8 +45,8 @@ async function startMockServer(params?: { finalOnlyMarkerPauseMs?: number }) { return server; } -async function postResponses(server: { baseUrl: string }, body: unknown) { - return fetch(`${server.baseUrl}/v1/responses`, { +async function postJson(server: { baseUrl: string }, path: string, body: unknown) { + return fetch(`${server.baseUrl}${path}`, { method: "POST", headers: { "content-type": "application/json", @@ -55,6 +55,10 @@ async function postResponses(server: { baseUrl: string }, body: unknown) { }); } +async function postResponses(server: { baseUrl: string }, body: unknown) { + return postJson(server, "/v1/responses", body); +} + async function expectResponsesText(server: { baseUrl: string }, body: unknown) { const response = await postResponses(server, body); expect(response.status).toBe(200); @@ -355,32 +359,20 @@ describe("qa mock openai server", () => { }); it("serves health and streamed responses", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); const health = await fetch(`${server.baseUrl}/healthz`); expect(health.status).toBe(200); expect(await health.json()).toEqual({ ok: true, status: "live" }); - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [ - { - role: "user", - content: [{ type: "input_text", text: "Inspect the repo docs and kickoff task." }], - }, - ], - }), + const response = await postResponses(server, { + stream: true, + input: [ + { + role: "user", + content: [{ type: "input_text", text: "Inspect the repo docs and kickoff task." }], + }, + ], }); expect(response.status).toBe(200); expect(response.headers.get("content-type")).toContain("text/event-stream"); @@ -392,43 +384,31 @@ describe("qa mock openai server", () => { it("turns a short approval into a kickoff-task read", async () => { const server = await startMockServer(); - const preActionResponse = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: false, - model: "gpt-5.6-luna", - input: [ - makeUserInput( - "Before acting, tell me the single file you would start with in six words or fewer. Do not use tools yet.", - ), - ], - }), + const preActionResponse = await postResponses(server, { + stream: false, + model: "gpt-5.6-luna", + input: [ + makeUserInput( + "Before acting, tell me the single file you would start with in six words or fewer. Do not use tools yet.", + ), + ], }); expect(preActionResponse.status).toBe(200); const preActionPayload = await preActionResponse.json(); expect(outputItem(preActionPayload).type).toBe("message"); expect(outputText(preActionPayload)).toContain("Protocol note: acknowledged."); - const approvalResponse = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - model: "gpt-5.6-luna", - input: [ - makeUserInput( - "Before acting, tell me the single file you would start with in six words or fewer. Do not use tools yet.", - ), - makeUserInput( - "ok do it. read `QA_KICKOFF_TASK.md` now and reply with the QA mission in one short sentence.", - ), - ], - }), + const approvalResponse = await postResponses(server, { + stream: true, + model: "gpt-5.6-luna", + input: [ + makeUserInput( + "Before acting, tell me the single file you would start with in six words or fewer. Do not use tools yet.", + ), + makeUserInput( + "ok do it. read `QA_KICKOFF_TASK.md` now and reply with the QA mission in one short sentence.", + ), + ], }); expect(approvalResponse.status).toBe(200); const approvalBody = await approvalResponse.text(); @@ -537,19 +517,13 @@ describe("qa mock openai server", () => { it("keeps final-only marker preview deltas separate from the final answer", async () => { const server = await startMockServer({ finalOnlyMarkerPauseMs: 1 }); - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [ - makeUserInput( - "Final-only marker streaming QA check. Reply exactly: QA-FINAL-ONLY-STREAMING-OK", - ), - ], - }), + const response = await postResponses(server, { + stream: true, + input: [ + makeUserInput( + "Final-only marker streaming QA check. Reply exactly: QA-FINAL-ONLY-STREAMING-OK", + ), + ], }); expect(response.status).toBe(200); @@ -638,15 +612,9 @@ describe("qa mock openai server", () => { it("emits deterministic text deltas for generic streaming QA prompts", async () => { const server = await startMockServer(); - const quietResponse = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [makeUserInput("Quiet streaming QA check: reply exactly `QA_STREAMING_OK`.")], - }), + const quietResponse = await postResponses(server, { + stream: true, + input: [makeUserInput("Quiet streaming QA check: reply exactly `QA_STREAMING_OK`.")], }); expect(quietResponse.status).toBe(200); const quietBody = await quietResponse.text(); @@ -654,52 +622,30 @@ describe("qa mock openai server", () => { expect(quietBody).toContain('"phase":"final_answer"'); expect(quietBody).toContain("QA_STREAMING_OK"); - const partialResponse = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [makeUserInput("Partial streaming QA check: reply exactly `QA_PARTIAL_OK`.")], - }), + const partialResponse = await postResponses(server, { + stream: true, + input: [makeUserInput("Partial streaming QA check: reply exactly `QA_PARTIAL_OK`.")], }); expect(partialResponse.status).toBe(200); const partialBody = await partialResponse.text(); expect(partialBody).toContain('"type":"response.output_text.delta"'); expect(partialBody).toContain("QA_PARTIAL_OK"); - const telegramStreamResponse = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [ - makeUserInput( - "Telegram reply-chain marker QA. Reply exactly: QA-TELEGRAM-REPLY-CHAIN-OK", - ), - makeUserInput("Quiet streaming QA check. Reply exactly: QA-TELEGRAM-STREAM-SINGLE-OK"), - ], - }), + const telegramStreamResponse = await postResponses(server, { + stream: true, + input: [ + makeUserInput("Telegram reply-chain marker QA. Reply exactly: QA-TELEGRAM-REPLY-CHAIN-OK"), + makeUserInput("Quiet streaming QA check. Reply exactly: QA-TELEGRAM-STREAM-SINGLE-OK"), + ], }); expect(telegramStreamResponse.status).toBe(200); const telegramStreamBody = await telegramStreamResponse.text(); expect(telegramStreamBody).toContain("QA-TELEGRAM-STREAM-SINGLE-OK"); expect(telegramStreamBody).not.toContain("QA-TELEGRAM-REPLY-CHAIN-OK"); - const telegramLongResponse = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [ - makeUserInput("Telegram long final QA check. Use the scripted long final response."), - ], - }), + const telegramLongResponse = await postResponses(server, { + stream: true, + input: [makeUserInput("Telegram long final QA check. Use the scripted long final response.")], }); expect(telegramLongResponse.status).toBe(200); const telegramLongBody = await telegramLongResponse.text(); @@ -709,17 +655,9 @@ describe("qa mock openai server", () => { expect(telegramLongBody).toContain("TELEGRAM-LONG-FINAL-END"); expect(telegramLongBody.length).toBeGreaterThan(4_500); - const whatsappLongResponse = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [ - makeUserInput("WhatsApp long final QA check. Use the scripted long final response."), - ], - }), + const whatsappLongResponse = await postResponses(server, { + stream: true, + input: [makeUserInput("WhatsApp long final QA check. Use the scripted long final response.")], }); expect(whatsappLongResponse.status).toBe(200); const whatsappLongBody = await whatsappLongResponse.text(); @@ -729,19 +667,13 @@ describe("qa mock openai server", () => { expect(whatsappLongBody).toContain("WHATSAPP-LONG-FINAL-END"); expect(whatsappLongBody.length).toBeGreaterThan(6_000); - const telegramThreeChunkLongResponse = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [ - makeUserInput( - "Telegram long final three chunk QA check. Use the scripted three chunk final response.", - ), - ], - }), + const telegramThreeChunkLongResponse = await postResponses(server, { + stream: true, + input: [ + makeUserInput( + "Telegram long final three chunk QA check. Use the scripted three chunk final response.", + ), + ], }); expect(telegramThreeChunkLongResponse.status).toBe(200); const telegramThreeChunkLongBody = await telegramThreeChunkLongResponse.text(); @@ -759,15 +691,9 @@ describe("qa mock openai server", () => { "Step 3: after that read completes, send a final assistant text block containing only this exact marker: `BLOCK_TWO_OK`.", "Never put both markers in the same assistant text block.", ].join("\n"); - const blockResponse = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [makeUserInput(blockPrompt)], - }), + const blockResponse = await postResponses(server, { + stream: true, + input: [makeUserInput(blockPrompt)], }); expect(blockResponse.status).toBe(200); const blockBody = await blockResponse.text(); @@ -777,22 +703,16 @@ describe("qa mock openai server", () => { expect(blockBody).toContain("BLOCK_ONE_OK"); expect(blockBody).not.toContain('"item_id":"msg_mock_block_2"'); - const blockContinuation = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [ - makeUserInput(blockPrompt), - { - type: "function_call_output", - call_id: "call_mock_read_fixture", - output: "QA kickoff task read", - }, - ], - }), + const blockContinuation = await postResponses(server, { + stream: true, + input: [ + makeUserInput(blockPrompt), + { + type: "function_call_output", + call_id: "call_mock_read_fixture", + output: "QA kickoff task read", + }, + ], }); expect(blockContinuation.status).toBe(200); const blockContinuationBody = await blockContinuation.text(); @@ -804,19 +724,13 @@ describe("qa mock openai server", () => { it("plans deterministic tool-progress reads from prompt paths", async () => { const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [ - makeUserInput( - "Tool progress QA check: read `qa-progress-target.txt` before answering. After the read completes, reply exactly `TOOL_PROGRESS_OK`.", - ), - ], - }), + const response = await postResponses(server, { + stream: true, + input: [ + makeUserInput( + "Tool progress QA check: read `qa-progress-target.txt` before answering. After the read completes, reply exactly `TOOL_PROGRESS_OK`.", + ), + ], }); expect(response.status).toBe(200); @@ -830,15 +744,9 @@ describe("qa mock openai server", () => { const prompt = "Tool progress QA check: use the read tool exactly once on `QA_KICKOFF_TASK.md` before answering. After that read completes, reply with only this exact marker and no other text: `TOOL_PROGRESS_MARKER_OK`."; - const toolPlan = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [makeUserInput(prompt)], - }), + const toolPlan = await postResponses(server, { + stream: true, + input: [makeUserInput(prompt)], }); expect(toolPlan.status).toBe(200); @@ -868,15 +776,9 @@ describe("qa mock openai server", () => { "rg -n 'matrix-progress-@room-@alice:matrix-qa.test-!room:matrix-qa.test.txt' . ; sleep 2"; const prompt = `Tool progress QA check: call the exec tool exactly once with this exact command before answering: \`${command}\`. After that exec command completes or fails, reply exactly \`TOOL_PROGRESS_EXEC_OK\`.`; - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [makeUserInput(prompt)], - }), + const response = await postResponses(server, { + stream: true, + input: [makeUserInput(prompt)], }); expect(response.status).toBe(200); @@ -958,15 +860,9 @@ describe("qa mock openai server", () => { const prompt = "Tool progress error QA check: read `missing-tool-progress-target.txt` before answering. After the read fails, reply exactly `TOOL_PROGRESS_ERROR_OK`."; - const toolPlan = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [makeUserInput(prompt)], - }), + const toolPlan = await postResponses(server, { + stream: true, + input: [makeUserInput(prompt)], }); expect(toolPlan.status).toBe(200); @@ -1008,25 +904,19 @@ describe("qa mock openai server", () => { it("uses the latest user prompt path for tool-progress plans", async () => { const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [ - makeUserInput( - "Tool progress QA check: read `older-progress-target.txt` before answering. After the read completes, reply exactly `OLD_PROGRESS_OK`.", - ), - makeUserInput( - "Tool progress error QA check: read `latest-missing-progress-target.txt` before answering. After the read fails, reply exactly `LATEST_PROGRESS_OK`.", - ), - makeUserInput( - "Continue with the QA scenario plan and report worked, failed, and blocked items.", - ), - ], - }), + const response = await postResponses(server, { + stream: true, + input: [ + makeUserInput( + "Tool progress QA check: read `older-progress-target.txt` before answering. After the read completes, reply exactly `OLD_PROGRESS_OK`.", + ), + makeUserInput( + "Tool progress error QA check: read `latest-missing-progress-target.txt` before answering. After the read fails, reply exactly `LATEST_PROGRESS_OK`.", + ), + makeUserInput( + "Continue with the QA scenario plan and report worked, failed, and blocked items.", + ), + ], }); expect(response.status).toBe(200); @@ -1037,33 +927,21 @@ describe("qa mock openai server", () => { }); it("prefers path-like refs over generic quoted keys in prompts", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: 'Please inspect "message_id" metadata first, then read `./QA_KICKOFF_TASK.md`.', - }, - ], - }, - ], - }), + const response = await postResponses(server, { + stream: true, + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: 'Please inspect "message_id" metadata first, then read `./QA_KICKOFF_TASK.md`.', + }, + ], + }, + ], }); expect(response.status).toBe(200); const body = await response.text(); @@ -1138,70 +1016,52 @@ describe("qa mock openai server", () => { }); it("drives the Lobster Invaders write flow and memory recall responses", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const lobster = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - model: "gpt-5.6-luna", - input: [ - { - role: "user", - content: [ - { type: "input_text", text: "Please build Lobster Invaders after reading context." }, - ], - }, - { - type: "function_call_output", - output: "QA mission: read source and docs first.", - }, - ], - }), + const lobster = await postResponses(server, { + stream: true, + model: "gpt-5.6-luna", + input: [ + { + role: "user", + content: [ + { type: "input_text", text: "Please build Lobster Invaders after reading context." }, + ], + }, + { + type: "function_call_output", + output: "QA mission: read source and docs first.", + }, + ], }); expect(lobster.status).toBe(200); const lobsterBody = await lobster.text(); expect(lobsterBody).toContain('"name":"write"'); expect(lobsterBody).toContain("lobster-invaders.html"); - const recall = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: false, - model: "gpt-5.6-luna-alt", - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "Please remember this fact for later: the QA canary code is ALPHA-7.", - }, - ], - }, - { - role: "user", - content: [ - { - type: "input_text", - text: "What was the QA canary code I asked you to remember earlier?", - }, - ], - }, - ], - }), + const recall = await postResponses(server, { + stream: false, + model: "gpt-5.6-luna-alt", + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Please remember this fact for later: the QA canary code is ALPHA-7.", + }, + ], + }, + { + role: "user", + content: [ + { + type: "input_text", + text: "What was the QA canary code I asked you to remember earlier?", + }, + ], + }, + ], }); expect(recall.status).toBe(200); const payload = (await recall.json()) as { @@ -1217,34 +1077,22 @@ describe("qa mock openai server", () => { }); it("keeps remember prompts prose-only even when they mention repo cleanup", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - model: "gpt-5.6-luna", - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "Please remember this fact for later: the QA canary code is ALPHA-7. Use your normal memory mechanism, avoid manual repo cleanup, and reply exactly `Remembered ALPHA-7.` once stored.", - }, - ], - }, - ], - }), + const response = await postResponses(server, { + stream: true, + model: "gpt-5.6-luna", + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Please remember this fact for later: the QA canary code is ALPHA-7. Use your normal memory mechanism, avoid manual repo cleanup, and reply exactly `Remembered ALPHA-7.` once stored.", + }, + ], + }, + ], }); expect(response.status).toBe(200); const body = await response.text(); @@ -1253,102 +1101,76 @@ describe("qa mock openai server", () => { }); it("drives repo-contract followthrough as read-read-read-write-then-report", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); const prompt = "Repo contract followthrough check. Read AGENT.md, SOUL.md, and FOLLOWTHROUGH_INPUT.md first. Then follow the repo contract exactly, write ./repo-contract-summary.txt, and reply with three labeled lines: Read, Wrote, Status."; - const first = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - model: "gpt-5.6-luna", - input: [{ role: "user", content: [{ type: "input_text", text: prompt }] }], - }), + const first = await postResponses(server, { + stream: true, + model: "gpt-5.6-luna", + input: [{ role: "user", content: [{ type: "input_text", text: prompt }] }], }); expect(first.status).toBe(200); expect(await first.text()).toContain('"arguments":"{\\"path\\":\\"AGENT.md\\"}"'); - const second = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - model: "gpt-5.6-luna", - input: [ - { role: "user", content: [{ type: "input_text", text: prompt }] }, - { - type: "function_call_output", - output: - "# Repo contract\n\nStep order:\n1. Read AGENT.md.\n2. Read SOUL.md.\n3. Read FOLLOWTHROUGH_INPUT.md.\n4. Write ./repo-contract-summary.txt.\n", - }, - ], - }), + const second = await postResponses(server, { + stream: true, + model: "gpt-5.6-luna", + input: [ + { role: "user", content: [{ type: "input_text", text: prompt }] }, + { + type: "function_call_output", + output: + "# Repo contract\n\nStep order:\n1. Read AGENT.md.\n2. Read SOUL.md.\n3. Read FOLLOWTHROUGH_INPUT.md.\n4. Write ./repo-contract-summary.txt.\n", + }, + ], }); expect(second.status).toBe(200); expect(await second.text()).toContain('"arguments":"{\\"path\\":\\"SOUL.md\\"}"'); - const third = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - model: "gpt-5.6-luna", - input: [ - { role: "user", content: [{ type: "input_text", text: prompt }] }, - { - type: "function_call_output", - output: "# Execution style\n\nStay brief, honest, and action-first.\n", - }, - ], - }), + const third = await postResponses(server, { + stream: true, + model: "gpt-5.6-luna", + input: [ + { role: "user", content: [{ type: "input_text", text: prompt }] }, + { + type: "function_call_output", + output: "# Execution style\n\nStay brief, honest, and action-first.\n", + }, + ], }); expect(third.status).toBe(200); expect(await third.text()).toContain('"arguments":"{\\"path\\":\\"FOLLOWTHROUGH_INPUT.md\\"}"'); - const fourth = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - model: "gpt-5.6-luna", - input: [ - { role: "user", content: [{ type: "input_text", text: prompt }] }, - { - type: "function_call_output", - output: - "Mission: prove you followed the repo contract.\nEvidence path: AGENT.md -> SOUL.md -> FOLLOWTHROUGH_INPUT.md -> repo-contract-summary.txt\n", - }, - ], - }), + const fourth = await postResponses(server, { + stream: true, + model: "gpt-5.6-luna", + input: [ + { role: "user", content: [{ type: "input_text", text: prompt }] }, + { + type: "function_call_output", + output: + "Mission: prove you followed the repo contract.\nEvidence path: AGENT.md -> SOUL.md -> FOLLOWTHROUGH_INPUT.md -> repo-contract-summary.txt\n", + }, + ], }); expect(fourth.status).toBe(200); const fourthBody = await fourth.text(); expect(fourthBody).toContain('"name":"write"'); expect(fourthBody).toContain("repo-contract-summary.txt"); - const fifth = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: false, - model: "gpt-5.6-luna", - input: [ - { role: "user", content: [{ type: "input_text", text: prompt }] }, - { - type: "function_call_output", - output: - "Successfully wrote repo-contract-summary.txt\nMission: prove you followed the repo contract.\nStatus: complete\n", - }, - ], - }), + const fifth = await postResponses(server, { + stream: false, + model: "gpt-5.6-luna", + input: [ + { role: "user", content: [{ type: "input_text", text: prompt }] }, + { + type: "function_call_output", + output: + "Successfully wrote repo-contract-summary.txt\nMission: prove you followed the repo contract.\nStatus: complete\n", + }, + ], }); expect(fifth.status).toBe(200); const payload = (await fifth.json()) as { @@ -1360,45 +1182,31 @@ describe("qa mock openai server", () => { }); it("uses argument-scoped tool call ids for repeated tool names", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); const prompt = "Repo contract followthrough check. Read AGENT.md, SOUL.md, and FOLLOWTHROUGH_INPUT.md first. Then follow the repo contract exactly, write ./repo-contract-summary.txt, and reply with three labeled lines: Read, Wrote, Status."; - const first = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: false, - model: "gpt-5.6-luna", - input: [{ role: "user", content: [{ type: "input_text", text: prompt }] }], - }), + const first = await postResponses(server, { + stream: false, + model: "gpt-5.6-luna", + input: [{ role: "user", content: [{ type: "input_text", text: prompt }] }], }); const firstPayload = (await first.json()) as { output?: Array<{ call_id?: string }>; }; - const second = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: false, - model: "gpt-5.6-luna", - input: [ - { role: "user", content: [{ type: "input_text", text: prompt }] }, - { - type: "function_call_output", - output: - "# Repo contract\n\nStep order:\n1. Read AGENT.md.\n2. Read SOUL.md.\n3. Read FOLLOWTHROUGH_INPUT.md.\n4. Write ./repo-contract-summary.txt.\n", - }, - ], - }), + const second = await postResponses(server, { + stream: false, + model: "gpt-5.6-luna", + input: [ + { role: "user", content: [{ type: "input_text", text: prompt }] }, + { + type: "function_call_output", + output: + "# Repo contract\n\nStep order:\n1. Read AGENT.md.\n2. Read SOUL.md.\n3. Read FOLLOWTHROUGH_INPUT.md.\n4. Write ./repo-contract-summary.txt.\n", + }, + ], }); const secondPayload = (await second.json()) as { output?: Array<{ call_id?: string }>; @@ -1790,36 +1598,26 @@ describe("qa mock openai server", () => { }); it("continues repo-contract followthrough when a retry user item follows tool output", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); const prompt = "Repo contract followthrough check. Read AGENT.md, SOUL.md, and FOLLOWTHROUGH_INPUT.md first. Then follow the repo contract exactly, write ./repo-contract-summary.txt, and reply with three labeled lines: Read, Wrote, Status."; - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - model: "gpt-5.6-luna", - input: [ - { role: "user", content: [{ type: "input_text", text: prompt }] }, - { - type: "function_call_output", - output: - "# Repo contract\n\nStep order:\n1. Read AGENT.md.\n2. Read SOUL.md.\n3. Read FOLLOWTHROUGH_INPUT.md.\n4. Write ./repo-contract-summary.txt.\n", - }, - { - role: "user", - content: [{ type: "input_text", text: "Continue after compaction." }], - }, - ], - }), + const response = await postResponses(server, { + stream: true, + model: "gpt-5.6-luna", + input: [ + { role: "user", content: [{ type: "input_text", text: prompt }] }, + { + type: "function_call_output", + output: + "# Repo contract\n\nStep order:\n1. Read AGENT.md.\n2. Read SOUL.md.\n3. Read FOLLOWTHROUGH_INPUT.md.\n4. Write ./repo-contract-summary.txt.\n", + }, + { + role: "user", + content: [{ type: "input_text", text: "Continue after compaction." }], + }, + ], }); expect(response.status).toBe(200); @@ -1827,40 +1625,30 @@ describe("qa mock openai server", () => { }); it("continues repo-contract followthrough from structured tool output", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); const prompt = "Repo contract followthrough check. Read AGENT.md, SOUL.md, and FOLLOWTHROUGH_INPUT.md first. Then follow the repo contract exactly, write ./repo-contract-summary.txt, and reply with three labeled lines: Read, Wrote, Status."; - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - model: "gpt-5.6-luna", - input: [ - { role: "user", content: [{ type: "input_text", text: prompt }] }, - { - type: "function_call_output", - output: [ - { - type: "output_text", - text: "# Repo contract\n\nStep order:\n1. Read AGENT.md.\n2. Read SOUL.md.\n3. Read FOLLOWTHROUGH_INPUT.md.\n4. Write ./repo-contract-summary.txt.\n", - }, - ], - }, - { - role: "user", - content: [{ type: "input_text", text: "Continue after compaction." }], - }, - ], - }), + const response = await postResponses(server, { + stream: true, + model: "gpt-5.6-luna", + input: [ + { role: "user", content: [{ type: "input_text", text: prompt }] }, + { + type: "function_call_output", + output: [ + { + type: "output_text", + text: "# Repo contract\n\nStep order:\n1. Read AGENT.md.\n2. Read SOUL.md.\n3. Read FOLLOWTHROUGH_INPUT.md.\n4. Write ./repo-contract-summary.txt.\n", + }, + ], + }, + { + role: "user", + content: [{ type: "input_text", text: "Continue after compaction." }], + }, + ], }); expect(response.status).toBe(200); @@ -1868,41 +1656,31 @@ describe("qa mock openai server", () => { }); it("advances repo-contract followthrough when transcript text is newer than extracted tool output", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); const prompt = "Repo contract followthrough check. Read AGENT.md, SOUL.md, and FOLLOWTHROUGH_INPUT.md first. Then follow the repo contract exactly, write ./repo-contract-summary.txt, and reply with three labeled lines: Read, Wrote, Status."; - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - model: "gpt-5.6-luna", - input: [ - { role: "user", content: [{ type: "input_text", text: prompt }] }, - { - type: "function_call_output", - output: - "# Repo contract\n\nStep order:\n1. Read AGENT.md.\n2. Read SOUL.md.\n3. Read FOLLOWTHROUGH_INPUT.md.\n4. Write ./repo-contract-summary.txt.\n", - }, - { - role: "user", - content: [ - { - type: "input_text", - text: "# Execution style\n\nStay brief, honest, and action-first.\n", - }, - ], - }, - ], - }), + const response = await postResponses(server, { + stream: true, + model: "gpt-5.6-luna", + input: [ + { role: "user", content: [{ type: "input_text", text: prompt }] }, + { + type: "function_call_output", + output: + "# Repo contract\n\nStep order:\n1. Read AGENT.md.\n2. Read SOUL.md.\n3. Read FOLLOWTHROUGH_INPUT.md.\n4. Write ./repo-contract-summary.txt.\n", + }, + { + role: "user", + content: [ + { + type: "input_text", + text: "# Execution style\n\nStay brief, honest, and action-first.\n", + }, + ], + }, + ], }); expect(response.status).toBe(200); @@ -1945,55 +1723,41 @@ describe("qa mock openai server", () => { }); it("advances personal task followthrough when transcript text is newer than extracted tool output", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); const prompt = "Personal task followthrough check. Read PERSONAL_TASK_LEDGER.md and FOLLOWTHROUGH_NOTE.md first. Then write ./personal-task-status.txt and reply with three labeled lines: Pending, Blocked, Done."; - const first = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - model: "gpt-5.6-luna", - input: [{ role: "user", content: [{ type: "input_text", text: prompt }] }], - }), + const first = await postResponses(server, { + stream: true, + model: "gpt-5.6-luna", + input: [{ role: "user", content: [{ type: "input_text", text: prompt }] }], }); expect(first.status).toBe(200); const firstBody = await first.text(); expect(firstBody).toContain('"arguments":"{\\"path\\":\\"PERSONAL_TASK_LEDGER.md\\"}"'); expect(firstBody).not.toContain("repo/package.json"); - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - model: "gpt-5.6-luna", - input: [ - { role: "user", content: [{ type: "input_text", text: prompt }] }, - { - type: "function_call_output", - output: - "# Personal task ledger\n\nRequired status contract:\n1. Read PERSONAL_TASK_LEDGER.md.\n2. Read FOLLOWTHROUGH_NOTE.md.\n3. Write ./personal-task-status.txt.\n", - }, - { - role: "user", - content: [ - { - type: "input_text", - text: "Task: prepare a local OpenClaw PR readiness note.\nPending: wait for maintainer feedback before publishing.\nBlocked: publishing needs explicit user approval.\nDone: local evidence captured in personal-task-status.txt.\n", - }, - ], - }, - ], - }), + const response = await postResponses(server, { + stream: true, + model: "gpt-5.6-luna", + input: [ + { role: "user", content: [{ type: "input_text", text: prompt }] }, + { + type: "function_call_output", + output: + "# Personal task ledger\n\nRequired status contract:\n1. Read PERSONAL_TASK_LEDGER.md.\n2. Read FOLLOWTHROUGH_NOTE.md.\n3. Write ./personal-task-status.txt.\n", + }, + { + role: "user", + content: [ + { + type: "input_text", + text: "Task: prepare a local OpenClaw PR readiness note.\nPending: wait for maintainer feedback before publishing.\nBlocked: publishing needs explicit user approval.\nDone: local evidence captured in personal-task-status.txt.\n", + }, + ], + }, + ], }); expect(response.status).toBe(200); @@ -2003,55 +1767,41 @@ describe("qa mock openai server", () => { }); it("gates personal completion claims on a written proof artifact", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); const prompt = "Personal no-fake-progress check. Read PROGRESS_REQUEST.md and PROGRESS_EVIDENCE.md first. Then write ./personal-progress-proof.txt with the local evidence and external status."; - const first = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - model: "gpt-5.6-luna", - input: [{ role: "user", content: [{ type: "input_text", text: prompt }] }], - }), + const first = await postResponses(server, { + stream: true, + model: "gpt-5.6-luna", + input: [{ role: "user", content: [{ type: "input_text", text: prompt }] }], }); expect(first.status).toBe(200); const firstBody = await first.text(); expect(firstBody).toContain('"arguments":"{\\"path\\":\\"PROGRESS_REQUEST.md\\"}"'); expect(firstBody).not.toContain("sent successfully"); - const writePlan = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - model: "gpt-5.6-luna", - input: [ - { role: "user", content: [{ type: "input_text", text: prompt }] }, - { - type: "function_call_output", - output: - "# Progress request\n\nRequired contract:\n1. Read PROGRESS_REQUEST.md.\n2. Read PROGRESS_EVIDENCE.md.\n3. Write ./personal-progress-proof.txt.\n", - }, - { - role: "user", - content: [ - { - type: "input_text", - text: "# Progress evidence\n\nTask: prepare a maintainer-ready local handoff note.\nExternal status: not sent, not published, not uploaded, not merged.\nHonest completion claim: local proof artifact written.\n", - }, - ], - }, - ], - }), + const writePlan = await postResponses(server, { + stream: true, + model: "gpt-5.6-luna", + input: [ + { role: "user", content: [{ type: "input_text", text: prompt }] }, + { + type: "function_call_output", + output: + "# Progress request\n\nRequired contract:\n1. Read PROGRESS_REQUEST.md.\n2. Read PROGRESS_EVIDENCE.md.\n3. Write ./personal-progress-proof.txt.\n", + }, + { + role: "user", + content: [ + { + type: "input_text", + text: "# Progress evidence\n\nTask: prepare a maintainer-ready local handoff note.\nExternal status: not sent, not published, not uploaded, not merged.\nHonest completion claim: local proof artifact written.\n", + }, + ], + }, + ], }); expect(writePlan.status).toBe(200); @@ -2060,21 +1810,17 @@ describe("qa mock openai server", () => { expect(writeBody).toContain("personal-progress-proof.txt"); expect(writeBody).not.toContain("published successfully"); - const final = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - model: "gpt-5.6-luna", - input: [ - { role: "user", content: [{ type: "input_text", text: prompt }] }, - { - type: "function_call_output", - output: - "Successfully wrote personal-progress-proof.txt with local proof artifact written.", - }, - ], - }), + const final = await postResponses(server, { + stream: true, + model: "gpt-5.6-luna", + input: [ + { role: "user", content: [{ type: "input_text", text: prompt }] }, + { + type: "function_call_output", + output: + "Successfully wrote personal-progress-proof.txt with local proof artifact written.", + }, + ], }); expect(final.status).toBe(200); @@ -2085,55 +1831,41 @@ describe("qa mock openai server", () => { }); it("reports personal failure recovery with a retry boundary", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); const prompt = "Personal failure recovery check. Read FAILURE_RECOVERY_REQUEST.md and FAILURE_RECOVERY_EVIDENCE.md first. Then write ./personal-failure-recovery.txt with Completed, Failed step, Retry boundary, and Next step."; - const first = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - model: "gpt-5.6-luna", - input: [{ role: "user", content: [{ type: "input_text", text: prompt }] }], - }), + const first = await postResponses(server, { + stream: true, + model: "gpt-5.6-luna", + input: [{ role: "user", content: [{ type: "input_text", text: prompt }] }], }); expect(first.status).toBe(200); const firstBody = await first.text(); expect(firstBody).toContain('"arguments":"{\\"path\\":\\"FAILURE_RECOVERY_REQUEST.md\\"}"'); expect(firstBody).not.toContain("fully complete"); - const writePlan = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - model: "gpt-5.6-luna", - input: [ - { role: "user", content: [{ type: "input_text", text: prompt }] }, - { - type: "function_call_output", - output: - "# Failure recovery request\n\nRequired contract:\n1. Read FAILURE_RECOVERY_REQUEST.md.\n2. Read FAILURE_RECOVERY_EVIDENCE.md.\n3. Write ./personal-failure-recovery.txt.\n", - }, - { - role: "user", - content: [ - { - type: "input_text", - text: "# Failure recovery evidence\n\nCompleted: request reviewed and local evidence captured.\nFailed step: external calendar update was not attempted because explicit approval is missing.\nRetry boundary: do not retry the external step until approval is given.\nNext step: ask for approval before any external update.\n", - }, - ], - }, - ], - }), + const writePlan = await postResponses(server, { + stream: true, + model: "gpt-5.6-luna", + input: [ + { role: "user", content: [{ type: "input_text", text: prompt }] }, + { + type: "function_call_output", + output: + "# Failure recovery request\n\nRequired contract:\n1. Read FAILURE_RECOVERY_REQUEST.md.\n2. Read FAILURE_RECOVERY_EVIDENCE.md.\n3. Write ./personal-failure-recovery.txt.\n", + }, + { + role: "user", + content: [ + { + type: "input_text", + text: "# Failure recovery evidence\n\nCompleted: request reviewed and local evidence captured.\nFailed step: external calendar update was not attempted because explicit approval is missing.\nRetry boundary: do not retry the external step until approval is given.\nNext step: ask for approval before any external update.\n", + }, + ], + }, + ], }); expect(writePlan.status).toBe(200); @@ -2143,21 +1875,17 @@ describe("qa mock openai server", () => { expect(writeBody).toContain("Retry boundary: do not retry"); expect(writeBody).not.toContain("retry succeeded"); - const final = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - model: "gpt-5.6-luna", - input: [ - { role: "user", content: [{ type: "input_text", text: prompt }] }, - { - type: "function_call_output", - output: - "Successfully wrote personal-failure-recovery.txt with the failed step and retry boundary.", - }, - ], - }), + const final = await postResponses(server, { + stream: true, + model: "gpt-5.6-luna", + input: [ + { role: "user", content: [{ type: "input_text", text: prompt }] }, + { + type: "function_call_output", + output: + "Successfully wrote personal-failure-recovery.txt with the failed step and retry boundary.", + }, + ], }); expect(final.status).toBe(200); @@ -2168,68 +1896,50 @@ describe("qa mock openai server", () => { }); it("drives the compaction retry mutating tool parity flow", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const writePlan = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - model: "gpt-5.6-luna", - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "Compaction retry mutating tool check: read COMPACTION_RETRY_CONTEXT.md, then create compaction-retry-summary.txt and keep replay safety explicit.", - }, - ], - }, - { - type: "function_call_output", - output: "compaction retry evidence block 0000\ncompaction retry evidence block 0001", - }, - ], - }), + const writePlan = await postResponses(server, { + stream: true, + model: "gpt-5.6-luna", + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Compaction retry mutating tool check: read COMPACTION_RETRY_CONTEXT.md, then create compaction-retry-summary.txt and keep replay safety explicit.", + }, + ], + }, + { + type: "function_call_output", + output: "compaction retry evidence block 0000\ncompaction retry evidence block 0001", + }, + ], }); expect(writePlan.status).toBe(200); const writePlanBody = await writePlan.text(); expect(writePlanBody).toContain('"name":"write"'); expect(writePlanBody).toContain("compaction-retry-summary.txt"); - const finalReply = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: false, - model: "gpt-5.6-luna", - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "Compaction retry mutating tool check: read COMPACTION_RETRY_CONTEXT.md, then create compaction-retry-summary.txt and keep replay safety explicit.", - }, - ], - }, - { - type: "function_call_output", - output: "Successfully wrote 41 bytes to compaction-retry-summary.txt.", - }, - ], - }), + const finalReply = await postResponses(server, { + stream: false, + model: "gpt-5.6-luna", + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Compaction retry mutating tool check: read COMPACTION_RETRY_CONTEXT.md, then create compaction-retry-summary.txt and keep replay safety explicit.", + }, + ], + }, + { + type: "function_call_output", + output: "Successfully wrote 41 bytes to compaction-retry-summary.txt.", + }, + ], }); expect(finalReply.status).toBe(200); const finalPayload = (await finalReply.json()) as { @@ -2239,101 +1949,71 @@ describe("qa mock openai server", () => { }); it("keeps compaction retry planning across continuation prompts", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); const prompt = "Compaction retry mutating tool check: read COMPACTION_RETRY_CONTEXT.md, then create compaction-retry-summary.txt and keep replay safety explicit."; - const writePlan = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - model: "gpt-5.6-luna", - input: [ - makeUserInput(prompt), - { - type: "function_call_output", - output: "compaction retry evidence block 0000\ncompaction retry evidence block 0001", - }, - makeUserInput("Continue after compaction."), - ], - }), + const writePlan = await postResponses(server, { + stream: true, + model: "gpt-5.6-luna", + input: [ + makeUserInput(prompt), + { + type: "function_call_output", + output: "compaction retry evidence block 0000\ncompaction retry evidence block 0001", + }, + makeUserInput("Continue after compaction."), + ], }); expect(writePlan.status).toBe(200); expect(await writePlan.text()).toContain('"name":"write"'); - const contextOnlyWritePlan = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - model: "gpt-5.6-luna", - input: [ - { - type: "function_call_output", - output: "compaction retry evidence block 0000\ncompaction retry evidence block 0001", - }, - makeUserInput("Continue after compaction."), - ], - }), + const contextOnlyWritePlan = await postResponses(server, { + stream: true, + model: "gpt-5.6-luna", + input: [ + { + type: "function_call_output", + output: "compaction retry evidence block 0000\ncompaction retry evidence block 0001", + }, + makeUserInput("Continue after compaction."), + ], }); expect(contextOnlyWritePlan.status).toBe(200); expect(await contextOnlyWritePlan.text()).toContain('"name":"write"'); - const finalReply = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: false, - model: "gpt-5.6-luna", - input: [ - makeUserInput(prompt), - { - type: "function_call_output", - output: "Successfully wrote 41 bytes to compaction-retry-summary.txt.", - }, - makeUserInput("Continue after compaction."), - ], - }), + const finalReply = await postResponses(server, { + stream: false, + model: "gpt-5.6-luna", + input: [ + makeUserInput(prompt), + { + type: "function_call_output", + output: "Successfully wrote 41 bytes to compaction-retry-summary.txt.", + }, + makeUserInput("Continue after compaction."), + ], }); expect(finalReply.status).toBe(200); expect(outputText(await finalReply.json())).toContain("replay unsafe after write"); }); it("supports exact reply memory prompts and embeddings requests", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const remember = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: false, - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "Please remember this fact for later: the QA canary code is ALPHA-7. Reply exactly `Remembered ALPHA-7.` once stored.", - }, - ], - }, - ], - }), + const remember = await postResponses(server, { + stream: false, + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Please remember this fact for later: the QA canary code is ALPHA-7. Reply exactly `Remembered ALPHA-7.` once stored.", + }, + ], + }, + ], }); expect(remember.status).toBe(200); const rememberPayload = (await remember.json()) as { @@ -2363,34 +2043,22 @@ describe("qa mock openai server", () => { }); it("requests non-threaded subagent handoff for QA channel runs", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - tools: [SESSIONS_SPAWN_TOOL], - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "Delegate a bounded QA task to a subagent, then summarize the delegated result clearly.", - }, - ], - }, - ], - }), + const response = await postResponses(server, { + stream: true, + tools: [SESSIONS_SPAWN_TOOL], + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Delegate a bounded QA task to a subagent, then summarize the delegated result clearly.", + }, + ], + }, + ], }); expect(response.status).toBe(200); const body = await response.text(); @@ -2618,76 +2286,58 @@ describe("qa mock openai server", () => { }); it("plans memory tools and serves mock image generations", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const memorySearch = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "Memory tools check: what is the hidden project codename stored only in memory? Use memory tools first.", - }, - ], - }, - ], - }), + const memorySearch = await postResponses(server, { + stream: true, + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Memory tools check: what is the hidden project codename stored only in memory? Use memory tools first.", + }, + ], + }, + ], }); expect(memorySearch.status).toBe(200); expect(await memorySearch.text()).toContain('"name":"memory_search"'); - const memoryGetFromPathOnlySearchResult = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [ - { - role: "user", - content: [ + const memoryGetFromPathOnlySearchResult = await postResponses(server, { + stream: true, + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Memory tools check: what is the hidden project codename stored only in memory? Use memory tools first.", + }, + ], + }, + { + type: "function_call_output", + output: JSON.stringify({ + results: [ { - type: "input_text", - text: "Memory tools check: what is the hidden project codename stored only in memory? Use memory tools first.", + path: "MEMORY.md", + snippet: "Hidden QA fact: the project codename is ORBIT-9.", }, ], - }, - { - type: "function_call_output", - output: JSON.stringify({ - results: [ - { - path: "MEMORY.md", - snippet: "Hidden QA fact: the project codename is ORBIT-9.", - }, - ], - }), - }, - { - role: "user", - content: [ - { - type: "input_text", - text: "Protocol note: acknowledged. Continue with the QA scenario plan.", - }, - ], - }, - ], - }), + }), + }, + { + role: "user", + content: [ + { + type: "input_text", + text: "Protocol note: acknowledged. Continue with the QA scenario plan.", + }, + ], + }, + ], }); expect(memoryGetFromPathOnlySearchResult.status).toBe(200); const memoryGetText = await memoryGetFromPathOnlySearchResult.text(); @@ -2723,370 +2373,298 @@ describe("qa mock openai server", () => { }); it("supports advanced QA memory and subagent recovery prompts", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const memory = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "Session memory ranking check: what is the current Project Nebula codename? Use memory tools first.", - }, - ], - }, - ], - }), + const memory = await postResponses(server, { + stream: true, + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Session memory ranking check: what is the current Project Nebula codename? Use memory tools first.", + }, + ], + }, + ], }); expect(memory.status).toBe(200); const memoryText = await memory.text(); expect(memoryText).toContain('"name":"memory_search"'); expect(memoryText).toContain('\\"corpus\\":\\"sessions\\"'); - const threadMemorySearch = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - instructions: - "@openclaw Thread memory check: what is the hidden thread codename stored only in memory? Use memory tools first and reply only in this thread.", - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "Protocol note: acknowledged. Continue with the QA scenario plan.", - }, - ], - }, - ], - }), + const threadMemorySearch = await postResponses(server, { + stream: true, + instructions: + "@openclaw Thread memory check: what is the hidden thread codename stored only in memory? Use memory tools first and reply only in this thread.", + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Protocol note: acknowledged. Continue with the QA scenario plan.", + }, + ], + }, + ], }); expect(threadMemorySearch.status).toBe(200); const threadMemorySearchText = await threadMemorySearch.text(); expect(threadMemorySearchText).toContain('"name":"memory_search"'); expect(threadMemorySearchText).toContain("ORBIT-22"); - const threadMemorySummary = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: false, - instructions: - "@openclaw Thread memory check: what is the hidden thread codename stored only in memory? Use memory tools first and reply only in this thread.", - input: [ - { - type: "function_call_output", - output: JSON.stringify({ - text: "Thread-hidden codename: ORBIT-22.", - }), - }, - { - role: "user", - content: [ - { - type: "input_text", - text: "Protocol note: acknowledged. Continue with the QA scenario plan.", - }, - ], - }, - ], - }), + const threadMemorySummary = await postResponses(server, { + stream: false, + instructions: + "@openclaw Thread memory check: what is the hidden thread codename stored only in memory? Use memory tools first and reply only in this thread.", + input: [ + { + type: "function_call_output", + output: JSON.stringify({ + text: "Thread-hidden codename: ORBIT-22.", + }), + }, + { + role: "user", + content: [ + { + type: "input_text", + text: "Protocol note: acknowledged. Continue with the QA scenario plan.", + }, + ], + }, + ], }); expect(threadMemorySummary.status).toBe(200); expect(JSON.stringify(await threadMemorySummary.json())).toContain("ORBIT-22"); - const structuredThreadMemorySummary = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: false, - instructions: - "@openclaw Thread memory check: what is the hidden thread codename stored only in memory? Use memory tools first and reply only in this thread.", - input: [ - { - type: "function_call_output", - output: { - text: "Thread-hidden codename: ORBIT-22.", + const structuredThreadMemorySummary = await postResponses(server, { + stream: false, + instructions: + "@openclaw Thread memory check: what is the hidden thread codename stored only in memory? Use memory tools first and reply only in this thread.", + input: [ + { + type: "function_call_output", + output: { + text: "Thread-hidden codename: ORBIT-22.", + }, + }, + { + role: "user", + content: [ + { + type: "input_text", + text: "Protocol note: acknowledged. Continue with the QA scenario plan.", }, - }, - { - role: "user", - content: [ - { - type: "input_text", - text: "Protocol note: acknowledged. Continue with the QA scenario plan.", - }, - ], - }, - ], - }), + ], + }, + ], }); expect(structuredThreadMemorySummary.status).toBe(200); expect(JSON.stringify(await structuredThreadMemorySummary.json())).toContain("ORBIT-22"); - const systemFallbackThreadMemorySummary = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: false, - input: [ - { - role: "system", - content: - "Available tools include sessions_spawn.\n## /workspace/MEMORY.md\nThread-hidden codename: ORBIT-22.", - }, - makeUserInput( - "@openclaw Thread memory check: what is the hidden thread codename stored only in memory? Use memory tools first and reply only in this thread.", - ), - { - type: "function_call_output", - output: JSON.stringify({ - results: [], - unavailable: true, - error: "database is not open", - }), - }, - ], - }), + const systemFallbackThreadMemorySummary = await postResponses(server, { + stream: false, + input: [ + { + role: "system", + content: + "Available tools include sessions_spawn.\n## /workspace/MEMORY.md\nThread-hidden codename: ORBIT-22.", + }, + makeUserInput( + "@openclaw Thread memory check: what is the hidden thread codename stored only in memory? Use memory tools first and reply only in this thread.", + ), + { + type: "function_call_output", + output: JSON.stringify({ + results: [], + unavailable: true, + error: "database is not open", + }), + }, + ], }); expect(systemFallbackThreadMemorySummary.status).toBe(200); expect(JSON.stringify(await systemFallbackThreadMemorySummary.json())).toContain("ORBIT-22"); - const memoryFollowup = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [ - { - role: "user", - content: [ + const memoryFollowup = await postResponses(server, { + stream: true, + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Session memory ranking check: what is the current Project Nebula codename? Use memory tools first.", + }, + ], + }, + { + type: "function_call_output", + output: JSON.stringify({ + results: [ { - type: "input_text", - text: "Session memory ranking check: what is the current Project Nebula codename? Use memory tools first.", + path: "sessions/qa-session-memory-ranking.jsonl", + startLine: 2, + endLine: 3, }, ], - }, - { - type: "function_call_output", - output: JSON.stringify({ - results: [ - { - path: "sessions/qa-session-memory-ranking.jsonl", - startLine: 2, - endLine: 3, - }, - ], - }), - }, - ], - }), + }), + }, + ], }); expect(memoryFollowup.status).toBe(200); expect(await memoryFollowup.text()).toContain( "Protocol note: I checked memory and the current Project Nebula codename is ORBIT-10.", ); - const memoryFollowupPrefersSessionResult = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [ - { - role: "user", - content: [ + const memoryFollowupPrefersSessionResult = await postResponses(server, { + stream: true, + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Session memory ranking check: what is the current Project Nebula codename? Use memory tools first.", + }, + ], + }, + { + type: "function_call_output", + output: JSON.stringify({ + results: [ { - type: "input_text", - text: "Session memory ranking check: what is the current Project Nebula codename? Use memory tools first.", + path: "MEMORY.md", + startLine: 1, + endLine: 2, + }, + { + path: "sessions/qa-session-memory-ranking.jsonl", + startLine: 2, + endLine: 3, }, ], - }, - { - type: "function_call_output", - output: JSON.stringify({ - results: [ - { - path: "MEMORY.md", - startLine: 1, - endLine: 2, - }, - { - path: "sessions/qa-session-memory-ranking.jsonl", - startLine: 2, - endLine: 3, - }, - ], - }), - }, - ], - }), + }), + }, + ], }); expect(memoryFollowupPrefersSessionResult.status).toBe(200); expect(await memoryFollowupPrefersSessionResult.text()).toContain( "Protocol note: I checked memory and the current Project Nebula codename is ORBIT-10.", ); - const activeMemorySearch = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: [ - "You are a memory search agent.", - "Use only the available memory tools.", - "Prefer memory_recall when available.", - "If memory_recall is unavailable, use memory_search and memory_get.", - "", - "Conversation context:", - "Latest user message:", - "Silent snack recall check: what snack do I usually want for QA movie night? Reply in one short sentence.", - ].join("\n"), - }, - ], - }, - ], - }), + const activeMemorySearch = await postResponses(server, { + stream: true, + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: [ + "You are a memory search agent.", + "Use only the available memory tools.", + "Prefer memory_recall when available.", + "If memory_recall is unavailable, use memory_search and memory_get.", + "", + "Conversation context:", + "Latest user message:", + "Silent snack recall check: what snack do I usually want for QA movie night? Reply in one short sentence.", + ].join("\n"), + }, + ], + }, + ], }); expect(activeMemorySearch.status).toBe(200); expect(await activeMemorySearch.text()).toContain('"name":"memory_search"'); - const activeMemoryStreamSummary = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: [ - "You are a memory search agent.", - "Use only the available memory tools.", - "Prefer memory_recall when available.", - "If memory_recall is unavailable, use memory_search and memory_get.", - "", - "Conversation context:", - "Latest user message:", - "Silent snack recall check: what snack do I usually want for QA movie night? Reply in one short sentence.", - ].join("\n"), - }, - ], - }, - { - type: "function_call_output", - output: JSON.stringify({ - text: "Stable QA movie night snack preference: lemon pepper wings with blue cheese.", - }), - }, - ], - }), + const activeMemoryStreamSummary = await postResponses(server, { + stream: true, + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: [ + "You are a memory search agent.", + "Use only the available memory tools.", + "Prefer memory_recall when available.", + "If memory_recall is unavailable, use memory_search and memory_get.", + "", + "Conversation context:", + "Latest user message:", + "Silent snack recall check: what snack do I usually want for QA movie night? Reply in one short sentence.", + ].join("\n"), + }, + ], + }, + { + type: "function_call_output", + output: JSON.stringify({ + text: "Stable QA movie night snack preference: lemon pepper wings with blue cheese.", + }), + }, + ], }); expect(activeMemoryStreamSummary.status).toBe(200); expect(await activeMemoryStreamSummary.text()).toContain("lemon pepper wings with blue cheese"); - const activeMemorySummary = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: false, - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: [ - "You are a memory search agent.", - "Use only the available memory tools.", - "Prefer memory_recall when available.", - "If memory_recall is unavailable, use memory_search and memory_get.", - "", - "Conversation context:", - "Latest user message:", - "Silent snack recall check: what snack do I usually want for QA movie night? Reply in one short sentence.", - ].join("\n"), - }, - ], - }, - { - type: "function_call_output", - output: JSON.stringify({ - text: "Stable QA movie night snack preference: lemon pepper wings with blue cheese.", - }), - }, - ], - }), + const activeMemorySummary = await postResponses(server, { + stream: false, + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: [ + "You are a memory search agent.", + "Use only the available memory tools.", + "Prefer memory_recall when available.", + "If memory_recall is unavailable, use memory_search and memory_get.", + "", + "Conversation context:", + "Latest user message:", + "Silent snack recall check: what snack do I usually want for QA movie night? Reply in one short sentence.", + ].join("\n"), + }, + ], + }, + { + type: "function_call_output", + output: JSON.stringify({ + text: "Stable QA movie night snack preference: lemon pepper wings with blue cheese.", + }), + }, + ], }); expect(activeMemorySummary.status).toBe(200); expect(JSON.stringify(await activeMemorySummary.json())).toContain( "lemon pepper wings with blue cheese", ); - const injectedMainReply = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: false, - instructions: [ - "System context:", - "User usually wants lemon pepper wings with blue cheese for QA movie night.", - ].join("\n"), - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "Silent snack recall check: what snack do I usually want for QA movie night? Reply in one short sentence.", - }, - ], - }, - ], - }), + const injectedMainReply = await postResponses(server, { + stream: false, + instructions: [ + "System context:", + "User usually wants lemon pepper wings with blue cheese for QA movie night.", + ].join("\n"), + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Silent snack recall check: what snack do I usually want for QA movie night? Reply in one short sentence.", + }, + ], + }, + ], }); expect(injectedMainReply.status).toBe(200); expect(JSON.stringify(await injectedMainReply.json())).toContain( @@ -3098,30 +2676,24 @@ describe("qa mock openai server", () => { expect(String(lastRequestPayload.instructions)).toContain(""); expect(String(lastRequestPayload.allInputText)).toContain(""); - const rememberSearch = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: [ - "You are a memory search agent.", - "Use only the available memory tools.", - "Latest user message:", - "Remember across conversations QA check: what snack do I usually want for QA movie night?", - ].join("\n"), - }, - ], - }, - ], - }), + const rememberSearch = await postResponses(server, { + stream: true, + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: [ + "You are a memory search agent.", + "Use only the available memory tools.", + "Latest user message:", + "Remember across conversations QA check: what snack do I usually want for QA movie night?", + ].join("\n"), + }, + ], + }, + ], }); expect(rememberSearch.status).toBe(200); const rememberSearchText = await rememberSearch.text(); @@ -3129,412 +2701,298 @@ describe("qa mock openai server", () => { expect(rememberSearchText).toContain("QA movie night snack lemon pepper wings blue cheese"); expect(rememberSearchText).toContain('\\"maxResults\\":10'); - const rememberSearchSummary = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [ - { - role: "user", - content: [ + const rememberSearchSummary = await postResponses(server, { + stream: true, + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: [ + "You are a memory search agent.", + "Use only the available memory tools.", + "Latest user message:", + "Remember across conversations QA check: what snack do I usually want for QA movie night?", + ].join("\n"), + }, + ], + }, + { + type: "function_call_output", + output: JSON.stringify({ + results: [ { - type: "input_text", - text: [ - "You are a memory search agent.", - "Use only the available memory tools.", - "Latest user message:", - "Remember across conversations QA check: what snack do I usually want for QA movie night?", - ].join("\n"), + path: "sessions/private-source.jsonl", + startLine: 2, + endLine: 3, + snippet: + "Stable QA movie night snack preference: lemon pepper wings with blue cheese.", }, ], - }, - { - type: "function_call_output", - output: JSON.stringify({ - results: [ - { - path: "sessions/private-source.jsonl", - startLine: 2, - endLine: 3, - snippet: - "Stable QA movie night snack preference: lemon pepper wings with blue cheese.", - }, - ], - }), - }, - ], - }), + }), + }, + ], }); expect(rememberSearchSummary.status).toBe(200); const rememberSearchSummaryText = await rememberSearchSummary.text(); expect(rememberSearchSummaryText).toContain("lemon pepper wings with blue cheese"); expect(rememberSearchSummaryText).not.toContain('"name":"memory_get"'); - const rememberInjectedMainReply = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: false, - instructions: - "User usually wants lemon pepper wings with blue cheese for QA movie night.", - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "Remember across conversations QA check: what snack do I usually want for QA movie night?", - }, - ], - }, - ], - }), + const rememberInjectedMainReply = await postResponses(server, { + stream: false, + instructions: + "User usually wants lemon pepper wings with blue cheese for QA movie night.", + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Remember across conversations QA check: what snack do I usually want for QA movie night?", + }, + ], + }, + ], }); expect(rememberInjectedMainReply.status).toBe(200); expect(JSON.stringify(await rememberInjectedMainReply.json())).toContain( "lemon pepper wings with blue cheese", ); - const spawn = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - tools: [SESSIONS_SPAWN_TOOL], - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "Subagent fanout synthesis check: delegate two bounded subagents sequentially, then report both results together.", - }, - ], - }, - ], - }), + const spawn = await postResponses(server, { + stream: true, + tools: [SESSIONS_SPAWN_TOOL], + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Subagent fanout synthesis check: delegate two bounded subagents sequentially, then report both results together.", + }, + ], + }, + ], }); expect(spawn.status).toBe(200); const spawnBody = await spawn.text(); expect(spawnBody).toContain('"name":"sessions_spawn"'); expect(spawnBody).toContain('\\"label\\":\\"qa-fanout-alpha\\"'); - const secondSpawn = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - tools: [SESSIONS_SPAWN_TOOL], - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "Subagent fanout synthesis check: delegate two bounded subagents sequentially, then report both results together.", - }, - ], - }, - { - type: "function_call_output", - output: - '{"status":"accepted","childSessionKey":"agent:qa:subagent:alpha","note":"ALPHA-OK"}', - }, - ], - }), + const secondSpawn = await postResponses(server, { + stream: true, + tools: [SESSIONS_SPAWN_TOOL], + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Subagent fanout synthesis check: delegate two bounded subagents sequentially, then report both results together.", + }, + ], + }, + { + type: "function_call_output", + output: + '{"status":"accepted","childSessionKey":"agent:qa:subagent:alpha","note":"ALPHA-OK"}', + }, + ], }); expect(secondSpawn.status).toBe(200); const secondSpawnBody = await secondSpawn.text(); expect(secondSpawnBody).toContain('"name":"sessions_spawn"'); expect(secondSpawnBody).toContain('\\"label\\":\\"qa-fanout-beta\\"'); - const final = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: false, - tools: [SESSIONS_SPAWN_TOOL], - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "Subagent fanout synthesis check: delegate two bounded subagents sequentially, then report both results together.", - }, - ], - }, - { - type: "function_call_output", - output: - '{"status":"accepted","childSessionKey":"agent:qa:subagent:beta","note":"BETA-OK"}', - }, - ], - }), + const final = await postResponses(server, { + stream: false, + tools: [SESSIONS_SPAWN_TOOL], + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Subagent fanout synthesis check: delegate two bounded subagents sequentially, then report both results together.", + }, + ], + }, + { + type: "function_call_output", + output: + '{"status":"accepted","childSessionKey":"agent:qa:subagent:beta","note":"BETA-OK"}', + }, + ], }); expect(final.status).toBe(200); expect(outputText(await final.json())).toBe("subagent-1: ok\nsubagent-2: ok"); }); it("completes subagent fanout from a continuation turn without tool output", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); const prompt = "Subagent fanout synthesis check: delegate two bounded subagents sequentially, then report both results together."; - const spawn = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - tools: [SESSIONS_SPAWN_TOOL], - input: [{ role: "user", content: [{ type: "input_text", text: prompt }] }], - }), + const spawn = await postResponses(server, { + stream: true, + tools: [SESSIONS_SPAWN_TOOL], + input: [{ role: "user", content: [{ type: "input_text", text: prompt }] }], }); expect(spawn.status).toBe(200); expect(await spawn.text()).toContain('\\"label\\":\\"qa-fanout-alpha\\"'); - const secondSpawn = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - tools: [SESSIONS_SPAWN_TOOL], - input: [ - { role: "user", content: [{ type: "input_text", text: prompt }] }, - { - type: "function_call_output", - output: - '{"status":"accepted","childSessionKey":"agent:qa:subagent:alpha","note":"ALPHA-OK"}', - }, - ], - }), + const secondSpawn = await postResponses(server, { + stream: true, + tools: [SESSIONS_SPAWN_TOOL], + input: [ + { role: "user", content: [{ type: "input_text", text: prompt }] }, + { + type: "function_call_output", + output: + '{"status":"accepted","childSessionKey":"agent:qa:subagent:alpha","note":"ALPHA-OK"}', + }, + ], }); expect(secondSpawn.status).toBe(200); expect(await secondSpawn.text()).toContain('\\"label\\":\\"qa-fanout-beta\\"'); - const phaseOnlyFinal = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: false, - tools: [SESSIONS_SPAWN_TOOL], - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "Continue.", - }, - ], - }, - ], - }), + const phaseOnlyFinal = await postResponses(server, { + stream: false, + tools: [SESSIONS_SPAWN_TOOL], + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Continue.", + }, + ], + }, + ], }); expect(phaseOnlyFinal.status).toBe(200); expect(outputText(await phaseOnlyFinal.json())).toBe("subagent-1: ok\nsubagent-2: ok"); }); it("completes subagent fanout when beta completion arrives on a generic follow-up turn", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); const prompt = "Subagent fanout synthesis check: delegate two bounded subagents sequentially, then report both results together."; - const spawn = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - tools: [SESSIONS_SPAWN_TOOL], - input: [makeUserInput(prompt)], - }), + const spawn = await postResponses(server, { + stream: true, + tools: [SESSIONS_SPAWN_TOOL], + input: [makeUserInput(prompt)], }); expect(spawn.status).toBe(200); expect(await spawn.text()).toContain('\\"label\\":\\"qa-fanout-alpha\\"'); - const secondSpawn = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - tools: [SESSIONS_SPAWN_TOOL], - input: [ - makeUserInput(prompt), - { - type: "function_call_output", - output: - '{"status":"accepted","childSessionKey":"agent:qa:subagent:alpha","note":"ALPHA-OK"}', - }, - ], - }), + const secondSpawn = await postResponses(server, { + stream: true, + tools: [SESSIONS_SPAWN_TOOL], + input: [ + makeUserInput(prompt), + { + type: "function_call_output", + output: + '{"status":"accepted","childSessionKey":"agent:qa:subagent:alpha","note":"ALPHA-OK"}', + }, + ], }); expect(secondSpawn.status).toBe(200); expect(await secondSpawn.text()).toContain('\\"label\\":\\"qa-fanout-beta\\"'); - const final = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: false, - tools: [SESSIONS_SPAWN_TOOL], - input: [ - makeUserInput( - "Continue with the QA scenario plan and report grouped into Worked, Failed, Blocked, and Follow-up.", - ), - { - type: "function_call_output", - output: '{"status":"accepted","childSessionKey":"agent:qa:subagent:beta"}', - }, - ], - }), + const final = await postResponses(server, { + stream: false, + tools: [SESSIONS_SPAWN_TOOL], + input: [ + makeUserInput( + "Continue with the QA scenario plan and report grouped into Worked, Failed, Blocked, and Follow-up.", + ), + { + type: "function_call_output", + output: '{"status":"accepted","childSessionKey":"agent:qa:subagent:beta"}', + }, + ], }); expect(final.status).toBe(200); expect(outputText(await final.json())).toBe("subagent-1: ok\nsubagent-2: ok"); }); it("uses full request text when planning continuation subagent tool calls", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); const handoffPrompt = "Delegate one bounded QA task to a subagent. Wait for the subagent to finish."; - const handoff = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - tools: [SESSIONS_SPAWN_TOOL], - input: [makeUserInput(handoffPrompt), makeUserInput("Continue.")], - }), + const handoff = await postResponses(server, { + stream: true, + tools: [SESSIONS_SPAWN_TOOL], + input: [makeUserInput(handoffPrompt), makeUserInput("Continue.")], }); expect(handoff.status).toBe(200); expect(await handoff.text()).toContain('"name":"sessions_spawn"'); - const handoffServer = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await handoffServer.stop(); - }); + const handoffServer = await startMockServer(); - const appServerHandoff = await fetch(`${handoffServer.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - input: [makeUserInput(handoffPrompt), makeUserInput("Continue.")], - }), + const appServerHandoff = await postResponses(handoffServer, { + stream: true, + input: [makeUserInput(handoffPrompt), makeUserInput("Continue.")], }); expect(appServerHandoff.status).toBe(200); expect(await appServerHandoff.text()).toContain('"name":"sessions_spawn"'); - const repeatedHandoff = await fetch(`${handoffServer.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - input: [makeUserInput(handoffPrompt), makeUserInput("Continue again.")], - }), + const repeatedHandoff = await postResponses(handoffServer, { + stream: true, + input: [makeUserInput(handoffPrompt), makeUserInput("Continue again.")], }); expect(repeatedHandoff.status).toBe(200); expect(await repeatedHandoff.text()).not.toContain('"name":"sessions_spawn"'); - const handoffFinal = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: false, - tools: [SESSIONS_SPAWN_TOOL], - input: [ - makeUserInput(handoffPrompt), - { type: "function_call_output", output: "SUBAGENT-OK" }, - makeUserInput("Continue."), - ], - }), + const handoffFinal = await postResponses(server, { + stream: false, + tools: [SESSIONS_SPAWN_TOOL], + input: [ + makeUserInput(handoffPrompt), + { type: "function_call_output", output: "SUBAGENT-OK" }, + makeUserInput("Continue."), + ], }); expect(handoffFinal.status).toBe(200); expect(outputText(await handoffFinal.json())).toContain("Delegated task"); const fanoutPrompt = "Subagent fanout synthesis check: delegate two bounded subagents sequentially, then report both results together."; - const appServerFanout = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - input: [makeUserInput(fanoutPrompt), makeUserInput("Continue.")], - }), + const appServerFanout = await postResponses(server, { + stream: true, + input: [makeUserInput(fanoutPrompt), makeUserInput("Continue.")], }); expect(appServerFanout.status).toBe(200); expect(await appServerFanout.text()).toContain('\\"label\\":\\"qa-fanout-alpha\\"'); - const fanoutServer = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await fanoutServer.stop(); - }); + const fanoutServer = await startMockServer(); - const firstFanout = await fetch(`${fanoutServer.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - tools: [SESSIONS_SPAWN_TOOL], - input: [makeUserInput(fanoutPrompt)], - }), + const firstFanout = await postResponses(fanoutServer, { + stream: true, + tools: [SESSIONS_SPAWN_TOOL], + input: [makeUserInput(fanoutPrompt)], }); expect(firstFanout.status).toBe(200); expect(await firstFanout.text()).toContain('\\"label\\":\\"qa-fanout-alpha\\"'); - const secondFanout = await fetch(`${fanoutServer.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - tools: [SESSIONS_SPAWN_TOOL], - input: [ - makeUserInput(fanoutPrompt), - { - type: "function_call_output", - output: - '{"status":"accepted","childSessionKey":"agent:qa:subagent:alpha","note":"ALPHA-OK"}', - }, - makeUserInput("Continue."), - ], - }), + const secondFanout = await postResponses(fanoutServer, { + stream: true, + tools: [SESSIONS_SPAWN_TOOL], + input: [ + makeUserInput(fanoutPrompt), + { + type: "function_call_output", + output: + '{"status":"accepted","childSessionKey":"agent:qa:subagent:alpha","note":"ALPHA-OK"}', + }, + makeUserInput("Continue."), + ], }); expect(secondFanout.status).toBe(200); expect(await secondFanout.text()).toContain('\\"label\\":\\"qa-fanout-beta\\"'); @@ -3559,31 +3017,21 @@ describe("qa mock openai server", () => { }); it("keeps source discovery reports out of subagent handoff prose", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: false, - input: [ - makeUserInput( - "Read the seeded docs and source plan, then report grouped into Worked, Failed, Blocked, and Follow-up.", - ), - { - type: "function_call_output", - output: - "repo/qa/scenarios/index.yaml includes scenario: subagent-handoff and repo/extensions/qa-lab/src/suite.ts.", - }, - makeUserInput("Continue."), - ], - }), + const response = await postResponses(server, { + stream: false, + input: [ + makeUserInput( + "Read the seeded docs and source plan, then report grouped into Worked, Failed, Blocked, and Follow-up.", + ), + { + type: "function_call_output", + output: + "repo/qa/scenarios/index.yaml includes scenario: subagent-handoff and repo/extensions/qa-lab/src/suite.ts.", + }, + makeUserInput("Continue."), + ], }); expect(response.status).toBe(200); @@ -3595,141 +3043,91 @@ describe("qa mock openai server", () => { }); it("does not let fanout completion state hijack child worker replies", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); const prompt = "Subagent fanout synthesis check: delegate two bounded subagents sequentially, then report both results together."; - const spawn = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - tools: [SESSIONS_SPAWN_TOOL], - input: [{ role: "user", content: [{ type: "input_text", text: prompt }] }], - }), + const spawn = await postResponses(server, { + stream: true, + tools: [SESSIONS_SPAWN_TOOL], + input: [{ role: "user", content: [{ type: "input_text", text: prompt }] }], }); expect(spawn.status).toBe(200); expect(await spawn.text()).toContain('\\"label\\":\\"qa-fanout-alpha\\"'); - const secondSpawn = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - tools: [SESSIONS_SPAWN_TOOL], - input: [ - { role: "user", content: [{ type: "input_text", text: prompt }] }, - { - type: "function_call_output", - output: - '{"status":"accepted","childSessionKey":"agent:qa:subagent:alpha","note":"ALPHA-OK"}', - }, - ], - }), + const secondSpawn = await postResponses(server, { + stream: true, + tools: [SESSIONS_SPAWN_TOOL], + input: [ + { role: "user", content: [{ type: "input_text", text: prompt }] }, + { + type: "function_call_output", + output: + '{"status":"accepted","childSessionKey":"agent:qa:subagent:alpha","note":"ALPHA-OK"}', + }, + ], }); expect(secondSpawn.status).toBe(200); expect(await secondSpawn.text()).toContain('\\"label\\":\\"qa-fanout-beta\\"'); - const childReply = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: false, - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "Fanout worker alpha: inspect the QA workspace and finish with exactly ALPHA-OK.", - }, - ], - }, - ], - }), + const childReply = await postResponses(server, { + stream: false, + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Fanout worker alpha: inspect the QA workspace and finish with exactly ALPHA-OK.", + }, + ], + }, + ], }); expect(childReply.status).toBe(200); expect(outputText(await childReply.json())).toBe("ALPHA-OK"); }); it("keeps subagent fanout state isolated per mock server instance", async () => { - const serverA = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await serverA.stop(); - }); - const serverB = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await serverB.stop(); - }); + const serverA = await startMockServer(); + const serverB = await startMockServer(); const prompt = "Subagent fanout synthesis check: delegate two bounded subagents sequentially, then report both results together."; - const firstA = await fetch(`${serverA.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - tools: [SESSIONS_SPAWN_TOOL], - input: [{ role: "user", content: [{ type: "input_text", text: prompt }] }], - }), + const firstA = await postResponses(serverA, { + stream: true, + tools: [SESSIONS_SPAWN_TOOL], + input: [{ role: "user", content: [{ type: "input_text", text: prompt }] }], }); expect(firstA.status).toBe(200); expect(await firstA.text()).toContain('\\"label\\":\\"qa-fanout-alpha\\"'); - const firstB = await fetch(`${serverB.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: true, - tools: [SESSIONS_SPAWN_TOOL], - input: [{ role: "user", content: [{ type: "input_text", text: prompt }] }], - }), + const firstB = await postResponses(serverB, { + stream: true, + tools: [SESSIONS_SPAWN_TOOL], + input: [{ role: "user", content: [{ type: "input_text", text: prompt }] }], }); expect(firstB.status).toBe(200); expect(await firstB.text()).toContain('\\"label\\":\\"qa-fanout-alpha\\"'); }); it("answers heartbeat prompts without spawning extra subagents", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: false, - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "System: Gateway restart config-apply ok\nSystem: QA-SUBAGENT-RECOVERY-1234\n\nRead HEARTBEAT.md if it exists (workspace context). Follow it strictly. Do not infer or repeat old tasks from prior chats. If nothing needs attention, reply HEARTBEAT_OK.", - }, - ], - }, - ], - }), + const response = await postResponses(server, { + stream: false, + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "System: Gateway restart config-apply ok\nSystem: QA-SUBAGENT-RECOVERY-1234\n\nRead HEARTBEAT.md if it exists (workspace context). Follow it strictly. Do not infer or repeat old tasks from prior chats. If nothing needs attention, reply HEARTBEAT_OK.", + }, + ], + }, + ], }); expect(response.status).toBe(200); @@ -3737,98 +3135,68 @@ describe("qa mock openai server", () => { }); it("returns exact markers for visible and hot-installed skills", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const visible = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: false, - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "Visible skill marker: give me the visible skill marker exactly.", - }, - ], - }, - ], - }), + const visible = await postResponses(server, { + stream: false, + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Visible skill marker: give me the visible skill marker exactly.", + }, + ], + }, + ], }); expect(visible.status).toBe(200); expect(outputText(await visible.json())).toBe("VISIBLE-SKILL-OK"); - const hot = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: false, - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "Hot install marker: give me the hot install marker exactly.", - }, - ], - }, - ], - }), + const hot = await postResponses(server, { + stream: false, + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Hot install marker: give me the hot install marker exactly.", + }, + ], + }, + ], }); expect(hot.status).toBe(200); expect(outputText(await hot.json())).toBe("HOT-INSTALL-OK"); }); it("uses the latest exact marker directive from conversation history", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: false, - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "Earlier turn: reply with only this exact marker: OLD_TOKEN", - }, - ], - }, - { - role: "user", - content: [ - { - type: "input_text", - text: "Current turn: reply with only this exact marker: NEW_TOKEN", - }, - ], - }, - ], - }), + const response = await postResponses(server, { + stream: false, + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Earlier turn: reply with only this exact marker: OLD_TOKEN", + }, + ], + }, + { + role: "user", + content: [ + { + type: "input_text", + text: "Current turn: reply with only this exact marker: NEW_TOKEN", + }, + ], + }, + ], }); expect(response.status).toBe(200); @@ -3836,115 +3204,85 @@ describe("qa mock openai server", () => { }); it("requires both WhatsApp batched markers before returning the final batched marker", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const standalone = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: false, - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: - "Second batched WhatsApp QA message. Reply with only this exact marker: " + - "WHATSAPP_QA_BATCHED_FINAL_TEST only if the previous queued message is visible " + - "in this same run context.", - }, - ], - }, - ], - }), + const standalone = await postResponses(server, { + stream: false, + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: + "Second batched WhatsApp QA message. Reply with only this exact marker: " + + "WHATSAPP_QA_BATCHED_FINAL_TEST only if the previous queued message is visible " + + "in this same run context.", + }, + ], + }, + ], }); expect(standalone.status).toBe(200); expect(outputText(await standalone.json())).toBe("WHATSAPP_QA_BATCHED_MISSING_CONTEXT_TEST"); - const batched = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: false, - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: - "First batched WhatsApp QA message WHATSAPP_QA_BATCHED_FIRST_TEST. " + - "Wait for the next message before replying.", - }, - ], - }, - { - role: "user", - content: [ - { - type: "input_text", - text: - "Second batched WhatsApp QA message. Reply with only this exact marker: " + - "WHATSAPP_QA_BATCHED_FINAL_TEST only if the previous queued message is visible " + - "in this same run context.", - }, - ], - }, - ], - }), + const batched = await postResponses(server, { + stream: false, + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: + "First batched WhatsApp QA message WHATSAPP_QA_BATCHED_FIRST_TEST. " + + "Wait for the next message before replying.", + }, + ], + }, + { + role: "user", + content: [ + { + type: "input_text", + text: + "Second batched WhatsApp QA message. Reply with only this exact marker: " + + "WHATSAPP_QA_BATCHED_FINAL_TEST only if the previous queued message is visible " + + "in this same run context.", + }, + ], + }, + ], }); expect(batched.status).toBe(200); expect(outputText(await batched.json())).toBe("WHATSAPP_QA_BATCHED_FINAL_TEST"); }); it("lets the latest exact marker prompt beat stale Telegram session_status history", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: false, - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "Telegram current session_status QA check. Call session_status with sessionKey set to current.", - }, - ], - }, - { - role: "user", - content: [ - { - type: "input_text", - text: "Telegram reply-chain marker QA. Reply exactly: QA-TELEGRAM-REPLY-CHAIN-OK", - }, - ], - }, - ], - }), + const response = await postResponses(server, { + stream: false, + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Telegram current session_status QA check. Call session_status with sessionKey set to current.", + }, + ], + }, + { + role: "user", + content: [ + { + type: "input_text", + text: "Telegram reply-chain marker QA. Reply exactly: QA-TELEGRAM-REPLY-CHAIN-OK", + }, + ], + }, + ], }); expect(response.status).toBe(200); @@ -3952,42 +3290,30 @@ describe("qa mock openai server", () => { }); it("does not repeat stale Telegram session_status for later ordinary prompts", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: false, - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "Telegram current session_status QA check. Call session_status with sessionKey set to current.", - }, - ], - }, - { - role: "user", - content: [ - { - type: "input_text", - text: "@sut Telegram QA mention routing check. Reply with a short acknowledgement.", - }, - ], - }, - ], - }), + const response = await postResponses(server, { + stream: false, + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Telegram current session_status QA check. Call session_status with sessionKey set to current.", + }, + ], + }, + { + role: "user", + content: [ + { + type: "input_text", + text: "@sut Telegram QA mention routing check. Reply with a short acknowledgement.", + }, + ], + }, + ], }); expect(response.status).toBe(200); @@ -3996,42 +3322,30 @@ describe("qa mock openai server", () => { }); it("uses exact marker directives from request context when the latest user text is generic", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: false, - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "@qa-sut.example.test reply with only this exact marker: QA_CANARY_TEST", - }, - ], - }, - { - role: "user", - content: [ - { - type: "input_text", - text: "Continue with the QA scenario plan and report worked, failed, and blocked items.", - }, - ], - }, - ], - }), + const response = await postResponses(server, { + stream: false, + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "@qa-sut.example.test reply with only this exact marker: QA_CANARY_TEST", + }, + ], + }, + { + role: "user", + content: [ + { + type: "input_text", + text: "Continue with the QA scenario plan and report worked, failed, and blocked items.", + }, + ], + }, + ], }); expect(response.status).toBe(200); @@ -4414,28 +3728,16 @@ describe("qa mock openai server", () => { }); it("uses image generation directives from request context when the latest user text is generic", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); const channelPrompt = '@qa-sut.example.test /tool image_generate action=generate prompt="QA lighthouse image for Matrix delivery testing" size=1024x1024 count=1'; const genericPrompt = "Continue with the QA scenario plan and report worked, failed, and blocked items."; - const toolPlan = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: false, - input: [makeUserInput(channelPrompt), makeUserInput(genericPrompt)], - }), + const toolPlan = await postResponses(server, { + stream: false, + input: [makeUserInput(channelPrompt), makeUserInput(genericPrompt)], }); expect(toolPlan.status).toBe(200); @@ -4444,34 +3746,28 @@ describe("qa mock openai server", () => { expect(toolPlanOutput.name).toBe("image_generate"); expect(String(toolPlanOutput.arguments)).toContain("qa-lighthouse.png"); - const toolResult = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: false, - input: [ - makeUserInput(channelPrompt), - makeUserInput(genericPrompt), - { - type: "function_call", - name: "image_generate", - call_id: "call_mock_image_generate_1", - arguments: JSON.stringify({ - prompt: "A QA lighthouse", - filename: "qa-lighthouse.png", - }), - }, - { - type: "function_call_output", - call_id: "call_mock_image_generate_1", - output: JSON.stringify({ - details: { media: { mediaUrls: ["/tmp/qa-lighthouse.png"] } }, - }), - }, - ], - }), + const toolResult = await postResponses(server, { + stream: false, + input: [ + makeUserInput(channelPrompt), + makeUserInput(genericPrompt), + { + type: "function_call", + name: "image_generate", + call_id: "call_mock_image_generate_1", + arguments: JSON.stringify({ + prompt: "A QA lighthouse", + filename: "qa-lighthouse.png", + }), + }, + { + type: "function_call_output", + call_id: "call_mock_image_generate_1", + output: JSON.stringify({ + details: { media: { mediaUrls: ["/tmp/qa-lighthouse.png"] } }, + }), + }, + ], }); expect(toolResult.status).toBe(200); @@ -4693,37 +3989,27 @@ describe("qa mock openai server", () => { }); it("records image inputs and describes attached images", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: false, - model: "mock-openai/gpt-5.6-luna", - input: [ - { - role: "user", - content: [ - { type: "input_text", text: "Image understanding check: what do you see?" }, - { - type: "input_image", - source: { - type: "base64", - mime_type: "image/png", - data: QA_IMAGE_PNG_BASE64, - }, + const response = await postResponses(server, { + stream: false, + model: "mock-openai/gpt-5.6-luna", + input: [ + { + role: "user", + content: [ + { type: "input_text", text: "Image understanding check: what do you see?" }, + { + type: "input_image", + source: { + type: "base64", + mime_type: "image/png", + data: QA_IMAGE_PNG_BASE64, }, - ], - }, - ], - }), + }, + ], + }, + ], }); expect(response.status).toBe(200); const payload = (await response.json()) as { @@ -4740,35 +4026,25 @@ describe("qa mock openai server", () => { }); it("recognizes OpenAI-compatible image_url parts as image inputs", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: false, - model: "mock-openai/gpt-5.6-luna", - input: [ - { - role: "user", - content: [ - { type: "input_text", text: "Image understanding check: what do you see?" }, - { - type: "image_url", - image_url: { - url: `data:image/png;base64,${QA_IMAGE_PNG_BASE64}`, - }, + const response = await postResponses(server, { + stream: false, + model: "mock-openai/gpt-5.6-luna", + input: [ + { + role: "user", + content: [ + { type: "input_text", text: "Image understanding check: what do you see?" }, + { + type: "image_url", + image_url: { + url: `data:image/png;base64,${QA_IMAGE_PNG_BASE64}`, }, - ], - }, - ], - }), + }, + ], + }, + ], }); expect(response.status).toBe(200); const payload = (await response.json()) as { @@ -4784,41 +4060,31 @@ describe("qa mock openai server", () => { }); it("answers image prompts when media context is the latest text part", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: false, - model: "mock-openai/gpt-5.6-luna", - input: [ - { - role: "user", - content: [ - { type: "input_text", text: "Image understanding check: what do you see?" }, - { - type: "input_image", - source: { - type: "base64", - mime_type: "image/png", - data: QA_IMAGE_PNG_BASE64, - }, + const response = await postResponses(server, { + stream: false, + model: "mock-openai/gpt-5.6-luna", + input: [ + { + role: "user", + content: [ + { type: "input_text", text: "Image understanding check: what do you see?" }, + { + type: "input_image", + source: { + type: "base64", + mime_type: "image/png", + data: QA_IMAGE_PNG_BASE64, }, - { - type: "input_text", - text: "[media attached: media://inbound/red-top-blue-bottom.png (image/png)]", - }, - ], - }, - ], - }), + }, + { + type: "input_text", + text: "[media attached: media://inbound/red-top-blue-bottom.png (image/png)]", + }, + ], + }, + ], }); expect(response.status).toBe(200); const payload = (await response.json()) as { @@ -4832,41 +4098,37 @@ describe("qa mock openai server", () => { it("lets image prompts beat stale exact marker directives from chat history", async () => { const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: false, - model: "mock-openai/gpt-5.6-luna", - input: [ - makeUserInput("Control UI bridge check. Marker exact marker: `ui bridge armed`"), - { - role: "assistant", - content: [{ type: "output_text", text: "ui bridge armed" }], - }, - { - role: "user", - content: [ - { - type: "input_text", - text: "Image understanding check: describe the top and bottom colors.", + const response = await postResponses(server, { + stream: false, + model: "mock-openai/gpt-5.6-luna", + input: [ + makeUserInput("Control UI bridge check. Marker exact marker: `ui bridge armed`"), + { + role: "assistant", + content: [{ type: "output_text", text: "ui bridge armed" }], + }, + { + role: "user", + content: [ + { + type: "input_text", + text: "Image understanding check: describe the top and bottom colors.", + }, + { + type: "input_image", + source: { + type: "base64", + mime_type: "image/png", + data: QA_IMAGE_PNG_BASE64, }, - { - type: "input_image", - source: { - type: "base64", - mime_type: "image/png", - data: QA_IMAGE_PNG_BASE64, - }, - }, - { - type: "input_text", - text: "[media attached: media://inbound/red-top-blue-bottom.png (image/png)]", - }, - ], - }, - ], - }), + }, + { + type: "input_text", + text: "[media attached: media://inbound/red-top-blue-bottom.png (image/png)]", + }, + ], + }, + ], }); expect(response.status).toBe(200); const payload = (await response.json()) as { @@ -4881,42 +4143,38 @@ describe("qa mock openai server", () => { it("keeps stale image prompts from overriding later marker turns", async () => { const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: false, - model: "mock-openai/gpt-5.6-luna", - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "Image understanding check: describe the top and bottom colors.", + const response = await postResponses(server, { + stream: false, + model: "mock-openai/gpt-5.6-luna", + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Image understanding check: describe the top and bottom colors.", + }, + { + type: "input_image", + source: { + type: "base64", + mime_type: "image/png", + data: QA_IMAGE_PNG_BASE64, }, - { - type: "input_image", - source: { - type: "base64", - mime_type: "image/png", - data: QA_IMAGE_PNG_BASE64, - }, - }, - ], - }, - { - role: "assistant", - content: [ - { - type: "output_text", - text: "Protocol note: the attached image is split horizontally, with red on top and blue on the bottom.", - }, - ], - }, - makeUserInput("Marker exact marker: `fresh-marker-ok`"), - ], - }), + }, + ], + }, + { + role: "assistant", + content: [ + { + type: "output_text", + text: "Protocol note: the attached image is split horizontally, with red on top and blue on the bottom.", + }, + ], + }, + makeUserInput("Marker exact marker: `fresh-marker-ok`"), + ], }); expect(response.status).toBe(200); const payload = (await response.json()) as { @@ -4928,33 +4186,29 @@ describe("qa mock openai server", () => { it("keeps stale consecutive image prompts from overriding later marker turns", async () => { const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: false, - model: "mock-openai/gpt-5.6-luna", - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "Image understanding check: describe the top and bottom colors.", + const response = await postResponses(server, { + stream: false, + model: "mock-openai/gpt-5.6-luna", + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Image understanding check: describe the top and bottom colors.", + }, + { + type: "input_image", + source: { + type: "base64", + mime_type: "image/png", + data: QA_IMAGE_PNG_BASE64, }, - { - type: "input_image", - source: { - type: "base64", - mime_type: "image/png", - data: QA_IMAGE_PNG_BASE64, - }, - }, - ], - }, - makeUserInput("Marker exact marker: `fresh-consecutive-marker-ok`"), - ], - }), + }, + ], + }, + makeUserInput("Marker exact marker: `fresh-consecutive-marker-ok`"), + ], }); expect(response.status).toBe(200); const payload = (await response.json()) as { @@ -4964,13 +4218,7 @@ describe("qa mock openai server", () => { }); it("handles deeply nested image input shapes without recursive traversal failure", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); let content: unknown = { type: "input_image", @@ -4984,19 +4232,15 @@ describe("qa mock openai server", () => { content = [{ type: "input_text", text: "nested" }, content]; } - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: false, - model: "mock-openai/gpt-5.6-luna", - input: [ - { - role: "user", - content, - }, - ], - }), + const response = await postResponses(server, { + stream: false, + model: "mock-openai/gpt-5.6-luna", + input: [ + { + role: "user", + content, + }, + ], }); expect(response.status).toBe(200); @@ -5006,40 +4250,30 @@ describe("qa mock openai server", () => { }); it("describes reattached generated images in the roundtrip flow", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - stream: false, - model: "mock-openai/gpt-5.6-luna", - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "Roundtrip image inspection check: describe the generated lighthouse attachment in one short sentence.", + const response = await postResponses(server, { + stream: false, + model: "mock-openai/gpt-5.6-luna", + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Roundtrip image inspection check: describe the generated lighthouse attachment in one short sentence.", + }, + { + type: "input_image", + source: { + type: "base64", + mime_type: "image/png", + data: QA_IMAGE_PNG_BASE64, }, - { - type: "input_image", - source: { - type: "base64", - mime_type: "image/png", - data: QA_IMAGE_PNG_BASE64, - }, - }, - ], - }, - ], - }), + }, + ], + }, + ], }); expect(response.status).toBe(200); const payload = (await response.json()) as { @@ -5050,79 +4284,55 @@ describe("qa mock openai server", () => { }); it("ignores stale tool output from prior turns when planning the current turn", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [ - { - role: "user", - content: [{ type: "input_text", text: "Read QA_KICKOFF_TASK.md first." }], - }, - { - type: "function_call_output", - output: "QA mission: read source and docs first.", - }, - { - role: "user", - content: [ - { - type: "input_text", - text: "Switch models now. Tool continuity check: reread QA_KICKOFF_TASK.md and mention the handoff in one short sentence.", - }, - ], - }, - ], - }), + const response = await postResponses(server, { + stream: true, + input: [ + { + role: "user", + content: [{ type: "input_text", text: "Read QA_KICKOFF_TASK.md first." }], + }, + { + type: "function_call_output", + output: "QA mission: read source and docs first.", + }, + { + role: "user", + content: [ + { + type: "input_text", + text: "Switch models now. Tool continuity check: reread QA_KICKOFF_TASK.md and mention the handoff in one short sentence.", + }, + ], + }, + ], }); expect(response.status).toBe(200); expect(await response.text()).toContain('"name":"read"'); }); it("returns continuity language after the model-switch reread completes", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: false, - model: "gpt-5.6-luna-alt", - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: "Switch models now. Tool continuity check: reread QA_KICKOFF_TASK.md and mention the handoff in one short sentence.", - }, - ], - }, - { - type: "function_call_output", - output: "QA mission: Understand this OpenClaw repo from source + docs before acting.", - }, - ], - }), + const response = await postResponses(server, { + stream: false, + model: "gpt-5.6-luna-alt", + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: "Switch models now. Tool continuity check: reread QA_KICKOFF_TASK.md and mention the handoff in one short sentence.", + }, + ], + }, + { + type: "function_call_output", + output: "QA mission: Understand this OpenClaw repo from source + docs before acting.", + }, + ], }); expect(response.status).toBe(200); @@ -5130,29 +4340,17 @@ describe("qa mock openai server", () => { }); it("returns the Codex remote-compaction-v2 response shape", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: true, - input: [ - { - role: "user", - content: [{ type: "input_text", text: "Retained context." }], - }, - { type: "compaction_trigger" }, - ], - }), + const response = await postResponses(server, { + stream: true, + input: [ + { + role: "user", + content: [{ type: "input_text", text: "Retained context." }], + }, + { type: "compaction_trigger" }, + ], }); expect(response.status).toBe(200); @@ -5167,46 +4365,28 @@ describe("qa mock openai server", () => { }); it("returns NO_REPLY for unmentioned group chatter", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { - "content-type": "application/json", - }, - body: JSON.stringify({ - stream: false, - input: [ - { - role: "user", - content: [ - { - type: "input_text", - text: 'Conversation info (untrusted metadata): {"is_group_chat": true}\n\nhello team, no bot ping here', - }, - ], - }, - ], - }), + const response = await postResponses(server, { + stream: false, + input: [ + { + role: "user", + content: [ + { + type: "input_text", + text: 'Conversation info (untrusted metadata): {"is_group_chat": true}\n\nhello team, no bot ping here', + }, + ], + }, + ], }); expect(response.status).toBe(200); expect(outputText(await response.json())).toBe("NO_REPLY"); }); it("advertises Anthropic claude-opus-4-8 baseline model on /v1/models", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); const response = await fetch(`${server.baseUrl}/v1/models`); expect(response.status).toBe(200); @@ -5218,14 +4398,9 @@ describe("qa mock openai server", () => { }); it("advertises selected target-era models on /v1/models", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, + const server = await startMockServer({ modelRefs: ["mock-openai/gpt-5.5", "mock-openai/gpt-5.5-alt"], }); - cleanups.push(async () => { - await server.stop(); - }); const response = await fetch(`${server.baseUrl}/v1/models`); expect(response.status).toBe(200); @@ -5236,13 +4411,7 @@ describe("qa mock openai server", () => { }); it("serves deterministic OpenAI-compatible audio transcription responses", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); const response = await fetch(`${server.baseUrl}/v1/audio/transcriptions`, { method: "POST", @@ -5259,13 +4428,7 @@ describe("qa mock openai server", () => { }); it("serves deterministic WhatsApp group audio transcription for the trigger fixture", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); const triggered = await fetch(`${server.baseUrl}/v1/audio/transcriptions`, { method: "POST", @@ -5295,13 +4458,7 @@ describe("qa mock openai server", () => { }); it("serves deterministic Matrix voice preflight transcription for the request prompt", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); const response = await fetch(`${server.baseUrl}/v1/audio/transcriptions`, { method: "POST", @@ -5321,32 +4478,22 @@ describe("qa mock openai server", () => { }); it("dispatches an Anthropic /v1/messages read tool call for source discovery prompts", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/messages`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - model: "claude-opus-4-8", - max_tokens: 256, - messages: [ - { - role: "user", - content: [ - { - type: "text", - text: "Read the seeded docs and report worked, failed, blocked, and follow-up items.", - }, - ], - }, - ], - }), + const response = await postJson(server, "/v1/messages", { + model: "claude-opus-4-8", + max_tokens: 256, + messages: [ + { + role: "user", + content: [ + { + type: "text", + text: "Read the seeded docs and report worked, failed, blocked, and follow-up items.", + }, + ], + }, + ], }); expect(response.status).toBe(200); const body = (await response.json()) as { @@ -5376,30 +4523,26 @@ describe("qa mock openai server", () => { it("preserves Anthropic /v1/messages declared tools for explicit sessions_spawn prompts", async () => { const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/messages`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - model: "claude-opus-4-8", - max_tokens: 256, - tools: [ - { - name: "sessions_spawn", - input_schema: { type: "object", properties: {} }, - }, - ], - messages: [ - { - role: "user", - content: [ - { - type: "text", - text: explicitSessionsSpawnPrompt("QA_SUBAGENT_CHILD_ANTHROPIC"), - }, - ], - }, - ], - }), + const response = await postJson(server, "/v1/messages", { + model: "claude-opus-4-8", + max_tokens: 256, + tools: [ + { + name: "sessions_spawn", + input_schema: { type: "object", properties: {} }, + }, + ], + messages: [ + { + role: "user", + content: [ + { + type: "text", + text: explicitSessionsSpawnPrompt("QA_SUBAGENT_CHILD_ANTHROPIC"), + }, + ], + }, + ], }); expect(response.status).toBe(200); const body = (await response.json()) as { @@ -5432,53 +4575,43 @@ describe("qa mock openai server", () => { // scenario is ideal because the mock has a two-stage flow: first // delegate prompt → sessions_spawn tool_use, then tool_result → // "Delegated task: ..." prose summary. - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/messages`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - model: "claude-opus-4-8", - max_tokens: 256, - messages: [ - { - role: "user", - content: [ - { - type: "text", - text: "Delegate one bounded QA task to a subagent, wait for it to finish, then reply with Delegated task, Result, and Evidence sections.", - }, - ], - }, - { - role: "assistant", - content: [ - { - type: "tool_use", - id: "toolu_mock_spawn_1", - name: "sessions_spawn", - input: { task: "Inspect the QA workspace", label: "qa-sidecar", thread: false }, - }, - ], - }, - { - role: "user", - content: [ - { - type: "tool_result", - tool_use_id: "toolu_mock_spawn_1", - content: "SUBAGENT-OK", - }, - ], - }, - ], - }), + const response = await postJson(server, "/v1/messages", { + model: "claude-opus-4-8", + max_tokens: 256, + messages: [ + { + role: "user", + content: [ + { + type: "text", + text: "Delegate one bounded QA task to a subagent, wait for it to finish, then reply with Delegated task, Result, and Evidence sections.", + }, + ], + }, + { + role: "assistant", + content: [ + { + type: "tool_use", + id: "toolu_mock_spawn_1", + name: "sessions_spawn", + input: { task: "Inspect the QA workspace", label: "qa-sidecar", thread: false }, + }, + ], + }, + { + role: "user", + content: [ + { + type: "tool_result", + tool_use_id: "toolu_mock_spawn_1", + content: "SUBAGENT-OK", + }, + ], + }, + ], }); expect(response.status).toBe(200); const body = (await response.json()) as { @@ -5506,64 +4639,54 @@ describe("qa mock openai server", () => { // that /debug/last-request exposes: the last-request `toolOutput` // field should be the stringified tool_result content, and `prompt` // should be the trailing fresh-text block. - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/messages`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - model: "claude-opus-4-8", - max_tokens: 256, - messages: [ - { - role: "user", - content: [ - { - type: "text", - text: "Delegate one bounded QA task to a subagent.", - }, - ], - }, - { - role: "assistant", - content: [ - { - type: "tool_use", - id: "toolu_mock_spawn_mixed", - name: "sessions_spawn", - input: { task: "Inspect the QA workspace", label: "qa-sidecar", thread: false }, - }, - ], - }, - { - role: "user", - content: [ - { - type: "tool_result", - tool_use_id: "toolu_mock_spawn_mixed", - content: "SUBAGENT-OK", - }, - // A trailing fresh text block in the same user turn. Before - // the loop-6 fix, the tool_result was pushed BEFORE the - // parent user message, so extractToolOutput saw the text - // turn as the last user-role item and found no - // function_call_output after it → returned "". The - // downstream dispatcher then behaved as if no tool output - // was present at all. - { - type: "text", - text: "Keep going with the fanout.", - }, - ], - }, - ], - }), + const response = await postJson(server, "/v1/messages", { + model: "claude-opus-4-8", + max_tokens: 256, + messages: [ + { + role: "user", + content: [ + { + type: "text", + text: "Delegate one bounded QA task to a subagent.", + }, + ], + }, + { + role: "assistant", + content: [ + { + type: "tool_use", + id: "toolu_mock_spawn_mixed", + name: "sessions_spawn", + input: { task: "Inspect the QA workspace", label: "qa-sidecar", thread: false }, + }, + ], + }, + { + role: "user", + content: [ + { + type: "tool_result", + tool_use_id: "toolu_mock_spawn_mixed", + content: "SUBAGENT-OK", + }, + // A trailing fresh text block in the same user turn. Before + // the loop-6 fix, the tool_result was pushed BEFORE the + // parent user message, so extractToolOutput saw the text + // turn as the last user-role item and found no + // function_call_output after it → returned "". The + // downstream dispatcher then behaved as if no tool output + // was present at all. + { + type: "text", + text: "Keep going with the fanout.", + }, + ], + }, + ], }); expect(response.status).toBe(200); @@ -5589,45 +4712,35 @@ describe("qa mock openai server", () => { }); it("exposes structured Anthropic tool_result errors in debug snapshots", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/messages`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - model: "claude-opus-4-8", - max_tokens: 256, - messages: [ - { - role: "assistant", - content: [ - { - type: "tool_use", - id: "toolu_mock_read_error", - name: "read", - input: { path: "/missing" }, - }, - ], - }, - { - role: "user", - content: [ - { - type: "tool_result", - tool_use_id: "toolu_mock_read_error", - is_error: true, - content: "ENOENT: no such file or directory", - }, - ], - }, - ], - }), + const response = await postJson(server, "/v1/messages", { + model: "claude-opus-4-8", + max_tokens: 256, + messages: [ + { + role: "assistant", + content: [ + { + type: "tool_use", + id: "toolu_mock_read_error", + name: "read", + input: { path: "/missing" }, + }, + ], + }, + { + role: "user", + content: [ + { + type: "tool_result", + tool_use_id: "toolu_mock_read_error", + is_error: true, + content: "ENOENT: no such file or directory", + }, + ], + }, + ], }); expect(response.status).toBe(200); @@ -5642,33 +4755,23 @@ describe("qa mock openai server", () => { }); it("streams Anthropic /v1/messages tool_use responses as SSE", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/messages`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - model: "claude-opus-4-8", - max_tokens: 256, - stream: true, - messages: [ - { - role: "user", - content: [ - { - type: "text", - text: "Read the seeded docs and report worked, failed, blocked, and follow-up items.", - }, - ], - }, - ], - }), + const response = await postJson(server, "/v1/messages", { + model: "claude-opus-4-8", + max_tokens: 256, + stream: true, + messages: [ + { + role: "user", + content: [ + { + type: "text", + text: "Read the seeded docs and report worked, failed, blocked, and follow-up items.", + }, + ], + }, + ], }); expect(response.status).toBe(200); expect(response.headers.get("content-type")).toContain("text/event-stream"); @@ -5683,54 +4786,44 @@ describe("qa mock openai server", () => { }); it("streams Anthropic /v1/messages tool_result follow-ups as text deltas", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/messages`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - model: "claude-opus-4-8", - max_tokens: 256, - stream: true, - messages: [ - { - role: "user", - content: [ - { - type: "text", - text: "Delegate one bounded QA task to a subagent, wait for it to finish, then reply with Delegated task, Result, and Evidence sections.", - }, - ], - }, - { - role: "assistant", - content: [ - { - type: "tool_use", - id: "toolu_mock_spawn_1", - name: "sessions_spawn", - input: { task: "Inspect the QA workspace", label: "qa-sidecar", thread: false }, - }, - ], - }, - { - role: "user", - content: [ - { - type: "tool_result", - tool_use_id: "toolu_mock_spawn_1", - content: "SUBAGENT-OK", - }, - ], - }, - ], - }), + const response = await postJson(server, "/v1/messages", { + model: "claude-opus-4-8", + max_tokens: 256, + stream: true, + messages: [ + { + role: "user", + content: [ + { + type: "text", + text: "Delegate one bounded QA task to a subagent, wait for it to finish, then reply with Delegated task, Result, and Evidence sections.", + }, + ], + }, + { + role: "assistant", + content: [ + { + type: "tool_use", + id: "toolu_mock_spawn_1", + name: "sessions_spawn", + input: { task: "Inspect the QA workspace", label: "qa-sidecar", thread: false }, + }, + ], + }, + { + role: "user", + content: [ + { + type: "tool_result", + tool_use_id: "toolu_mock_spawn_1", + content: "SUBAGENT-OK", + }, + ], + }, + ], }); expect(response.status).toBe(200); expect(response.headers.get("content-type")).toContain("text/event-stream"); @@ -5742,39 +4835,29 @@ describe("qa mock openai server", () => { }); it("keeps Anthropic remember prompts on the prose branch even when system text mentions HEARTBEAT", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/messages`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - model: "claude-opus-4-8", - max_tokens: 256, - stream: true, - system: [ - { - type: "text", - text: "Read HEARTBEAT.md if it exists (workspace context). Follow it strictly. If nothing needs attention, reply HEARTBEAT_OK.", - }, - ], - messages: [ - { - role: "user", - content: [ - { - type: "text", - text: "Please remember this fact for later: the QA canary code is ALPHA-7. Use your normal memory mechanism, avoid manual repo cleanup, and reply exactly `Remembered ALPHA-7.` once stored.", - }, - ], - }, - ], - }), + const response = await postJson(server, "/v1/messages", { + model: "claude-opus-4-8", + max_tokens: 256, + stream: true, + system: [ + { + type: "text", + text: "Read HEARTBEAT.md if it exists (workspace context). Follow it strictly. If nothing needs attention, reply HEARTBEAT_OK.", + }, + ], + messages: [ + { + role: "user", + content: [ + { + type: "text", + text: "Please remember this fact for later: the QA canary code is ALPHA-7. Use your normal memory mechanism, avoid manual repo cleanup, and reply exactly `Remembered ALPHA-7.` once stored.", + }, + ], + }, + ], }); expect(response.status).toBe(200); @@ -5785,43 +4868,33 @@ describe("qa mock openai server", () => { }); it("prefers the prompt-local exact reply directive over heartbeat context", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/messages`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - model: "claude-opus-4-8", - max_tokens: 256, - stream: true, - system: [ - { - type: "text", - text: [ - "Read HEARTBEAT.md if it exists (workspace context). Follow it strictly.", - "If the current user message is a heartbeat poll and nothing needs attention, reply exactly:", - "HEARTBEAT_OK", - ].join("\n"), - }, - ], - messages: [ - { - role: "user", - content: [ - { - type: "text", - text: "Please remember this fact for later: the QA canary code is ALPHA-7. Use your normal memory mechanism, avoid manual repo cleanup, and reply exactly `Remembered ALPHA-7.` once stored.", - }, - ], - }, - ], - }), + const response = await postJson(server, "/v1/messages", { + model: "claude-opus-4-8", + max_tokens: 256, + stream: true, + system: [ + { + type: "text", + text: [ + "Read HEARTBEAT.md if it exists (workspace context). Follow it strictly.", + "If the current user message is a heartbeat poll and nothing needs attention, reply exactly:", + "HEARTBEAT_OK", + ].join("\n"), + }, + ], + messages: [ + { + role: "user", + content: [ + { + type: "text", + text: "Please remember this fact for later: the QA canary code is ALPHA-7. Use your normal memory mechanism, avoid manual repo cleanup, and reply exactly `Remembered ALPHA-7.` once stored.", + }, + ], + }, + ], }); expect(response.status).toBe(200); @@ -5831,13 +4904,7 @@ describe("qa mock openai server", () => { }); it("rejects malformed or non-object Anthropic /v1/messages JSON", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); for (const rawBody of ['{"model":"claude-opus-4-8","messages":[', "null", "[]", '"text"']) { const response = await fetch(`${server.baseUrl}/v1/messages`, { @@ -5861,13 +4928,7 @@ describe("qa mock openai server", () => { }); it("rejects malformed OpenAI-compatible JSON without crashing the mock server", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); for (const path of ["/v1/responses", "/v1/embeddings", "/v1/images/generations"]) { for (const rawBody of ["{bad", "[]", '"text"']) { @@ -5896,27 +4957,17 @@ describe("qa mock openai server", () => { // through to `lastRequest.model` and `responseBody.model`. Empty // strings must be treated the same as absent and default to // `"claude-opus-4-8"` so parity consumers can trust the echoed label. - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); + const server = await startMockServer(); - const response = await fetch(`${server.baseUrl}/v1/messages`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ - model: "", - max_tokens: 256, - messages: [ - { - role: "user", - content: "Read the plan", - }, - ], - }), + const response = await postJson(server, "/v1/messages", { + model: "", + max_tokens: 256, + messages: [ + { + role: "user", + content: "Read the plan", + }, + ], }); expect(response.status).toBe(200); const body = (await response.json()) as { model: string }; @@ -6311,83 +5362,52 @@ describe("qa mock openai server provider variant tagging", () => { }); }); - it("records providerVariant on /debug/last-request for openai requests", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); - - await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ + it.each([ + { + name: "records providerVariant on /debug/last-request for openai requests", + path: "/v1/responses", + body: { model: "openai/gpt-5.6-luna", stream: false, - input: [{ role: "user", content: [{ type: "input_text", text: "Heartbeat check" }] }], - }), - }); - - const debug = (await (await fetch(`${server.baseUrl}/debug/last-request`)).json()) as { - model: string; - providerVariant: string; - }; - expect(debug.model).toBe("openai/gpt-5.6-luna"); - expect(debug.providerVariant).toBe("openai"); - }); - - it("records providerVariant=anthropic on /v1/messages requests", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); - - await fetch(`${server.baseUrl}/v1/messages`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ + input: [makeUserInput("Heartbeat check")], + }, + expectedModel: "openai/gpt-5.6-luna", + expectedVariant: "openai", + }, + { + name: "records providerVariant=anthropic on /v1/messages requests", + path: "/v1/messages", + body: { model: "claude-opus-4-8", max_tokens: 256, messages: [{ role: "user", content: "Heartbeat check" }], - }), - }); - - const debug = (await (await fetch(`${server.baseUrl}/debug/last-request`)).json()) as { - model: string; - providerVariant: string; - }; - expect(debug.model).toBe("claude-opus-4-8"); - expect(debug.providerVariant).toBe("anthropic"); - }); - - it("records providerVariant=unknown for unrecognized models", async () => { - const server = await startQaMockOpenAiServer({ - host: "127.0.0.1", - port: 0, - }); - cleanups.push(async () => { - await server.stop(); - }); - - await fetch(`${server.baseUrl}/v1/responses`, { - method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify({ + }, + expectedModel: "claude-opus-4-8", + expectedVariant: "anthropic", + }, + { + name: "records providerVariant=unknown for unrecognized models", + path: "/v1/responses", + body: { model: "mistral/mistral-large", stream: false, - input: [{ role: "user", content: [{ type: "input_text", text: "Heartbeat check" }] }], - }), - }); + input: [makeUserInput("Heartbeat check")], + }, + expectedModel: undefined, + expectedVariant: "unknown", + }, + ])("$name", async ({ path, body, expectedModel, expectedVariant }) => { + const server = await startMockServer(); + await postJson(server, path, body); const debug = (await (await fetch(`${server.baseUrl}/debug/last-request`)).json()) as { + model?: string; providerVariant: string; }; - expect(debug.providerVariant).toBe("unknown"); + if (expectedModel) { + expect(debug.model).toBe(expectedModel); + } + expect(debug.providerVariant).toBe(expectedVariant); }); }); /* oxlint-disable max-lines -- TODO: split this grandfathered oversized file. */