fix(codex): preserve inbound audio for automatic voice replies (#135884)

* fix(codex): preserve inbound audio in dynamic tool context

Carry inbound-audio state from the shared attempt context so reconstructed
Codex message tools apply automatic TTS to voice input. Keep accepted audio
steering bound to its originating reply operation and remove the embedded
runner's duplicate forwarding path.

Add ordered real-tool regression coverage and a Gateway scenario that sends
audio followed by text through the native Codex app server.

Refs #135571. Thanks to @hyper-sdn for reporting the missing voice reply.

* test: run native voice QA scenario serially

* test(codex): cover inbound TTS after accepted audio steering
This commit is contained in:
Peter Steinberger 2026-09-01 22:33:04 -07:00 • committed by GitHub
parent f1f299a40b
commit 51177301f2
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
8 changed files with 434 additions and 64 deletions

View file

@ -14,9 +14,20 @@ import { readMemoryArtifactProvenance } from "openclaw/plugin-sdk/memory-core-ho
import {
createAgentHarnessHostCapabilitiesForTest,
createMockPluginRegistry,
createOutboundTestPlugin,
createTestRegistry,
getActivePluginRegistry,
initializeGlobalHookRunner,
resetGlobalHookRunner,
resetPluginRuntimeStateForTest,
setActivePluginRegistry,
} from "openclaw/plugin-sdk/plugin-test-runtime";
import {
clearRuntimeConfigSnapshot,
getRuntimeConfigSnapshot,
getRuntimeConfigSourceSnapshot,
setRuntimeConfigSnapshot,
} from "openclaw/plugin-sdk/runtime-config-snapshot";
import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
import { dynamicToolBuildState } from "./dynamic-tool-build-state.js";
import {
@ -2773,6 +2784,121 @@ describe("Codex app-server dynamic tool build", () => {
).toBe(false);
});
it.each(["text", "initial audio", "accepted audio steering"])(
"applies inbound TTS to a final dynamic message for %s",
async (input) => {
const workspaceDir = path.join(tempDir, "workspace");
vi.stubEnv("OPENCLAW_STATE_DIR", path.join(tempDir, "state"));
const synthesize = vi.fn(async (_request: { text: string }) => ({
audioBuffer: Buffer.from("synthetic speech"),
fileExtension: ".ogg",
outputFormat: "ogg",
voiceCompatible: true,
}));
const sendMedia = vi.fn(async (_context: { mediaUrl?: string }) => ({
channel: "whatsapp",
messageId: "voice-reply",
}));
const sendText = vi.fn(async () => ({ channel: "whatsapp", messageId: "text-reply" }));
const channel = createOutboundTestPlugin({
id: "whatsapp",
capabilities: {
chatTypes: ["direct"],
media: true,
tts: { voice: { synthesisTarget: "voice-note" } },
},
outbound: {
deliveryMode: "direct",
resolveTarget: ({ to }) => ({ ok: true, to: to ?? "+12025550123" }),
sendText,
sendMedia,
},
});
channel.config.listAccountIds = () => ["default"];
const registry = createTestRegistry([
{ pluginId: "whatsapp", source: "test", plugin: channel },
]);
registry.speechProviders.push({
pluginId: "test-speech",
source: "test",
provider: { id: "test-speech", label: "Test speech", isConfigured: () => true, synthesize },
});
const previousRegistry = getActivePluginRegistry();
const previousRuntimeConfig = getRuntimeConfigSnapshot();
const previousSourceConfig = getRuntimeConfigSourceSnapshot();
setActivePluginRegistry(registry);
try {
const params = createParams(path.join(tempDir, "session.jsonl"), workspaceDir);
params.disableTools = false;
params.runtimePlan = createCodexRuntimePlanFixture();
params.config = {
tts: { auto: "inbound", provider: "test-speech" },
channels: { whatsapp: { allowFrom: ["*"] } },
};
// Match Gateway config ownership so earlier tool construction cannot supply TTS policy.
setRuntimeConfigSnapshot(params.config, params.config);
params.messageChannel = "whatsapp";
params.currentInboundAudio = input === "initial audio";
const replyOperation = { acceptedSteeredInboundAudio: false };
params.replyOperation = replyOperation as EmbeddedRunAttemptParams["replyOperation"];
params.sourceReplyDeliveryMode = "message_tool_only";
setOpenClawCodingToolsFactoryForTests((options) =>
createOpenClawCodingTools(options).filter((tool) => tool.name === "message"),
);
const tools = await buildDynamicToolsForTest(params, workspaceDir, {
sandbox: null as never,
});
// Accepted steering must reach tools that were already constructed.
replyOperation.acceptedSteeredInboundAudio = input === "accepted audio steering";
const bridge = createCodexDynamicToolBridge({
tools,
signal: new AbortController().signal,
});
const result = await bridge.handleToolCall({
threadId: "thread-1",
turnId: "turn-1",
callId: "voice-final",
namespace: null,
tool: "message",
arguments: {
action: "send",
channel: "whatsapp",
target: "+12025550123",
message: "Here is the requested spoken reply.",
final: true,
},
});
const expectsVoice = input !== "text";
expect(result.success, JSON.stringify(result.contentItems)).toBe(true);
expect(synthesize).toHaveBeenCalledTimes(expectsVoice ? 1 : 0);
expect(sendMedia).toHaveBeenCalledTimes(expectsVoice ? 1 : 0);
if (expectsVoice) {
expect(sendMedia).toHaveBeenCalledWith(expect.objectContaining({ audioAsVoice: true }));
} else {
expect(sendText).toHaveBeenCalledOnce();
}
} finally {
if (previousRuntimeConfig) {
setRuntimeConfigSnapshot(previousRuntimeConfig, previousSourceConfig ?? undefined);
} else {
clearRuntimeConfigSnapshot();
}
for (const [sent] of sendMedia.mock.calls) {
if (sent.mediaUrl) {
await fs.rm(sent.mediaUrl, { force: true });
}
}
if (previousRegistry) {
setActivePluginRegistry(previousRegistry);
} else {
resetPluginRuntimeStateForTest();
}
}
},
);
it("preserves the core final delivery control only on message-tool-only schemas", async () => {
const workspaceDir = path.join(tempDir, "workspace");
const params = createParams(path.join(tempDir, "session.jsonl"), workspaceDir);

View file

@ -0,0 +1,35 @@
title: Codex inbound-audio message TTS delivery
scenario:
id: codex-inbound-message-auto-tts
surface: media
category: media.text-to-speech-delivery
coverage:
# Native-binary proof stays explicitly selected instead of joining default coverage profiles.
secondary:
- media.tts
- media.outbound-voice-audio-delivery
- channels.inbound-media-normalization
objective: Verify native Codex dynamic message tools retain the current inbound-audio fact for automatic TTS.
successCriteria:
- A synthetic audio attachment enters through QA Channel with transcription disabled.
- The native Codex app server calls message(action=send, final=true) against a local mock model.
- Inbound-mode TTS synthesizes exactly once and QA Channel receives the exact synthetic WAV bytes.
- A subsequent text-only turn in the same session delivers one text reply without another synthesis.
docsRefs:
- docs/tools/tts.md
- docs/channels/qa-channel.md
codeRefs:
- test/e2e/qa-lab/runtime/media-talk-gateway.ts
- src/agents/embedded-agent-runner/run/attempt-tool-run-context.ts
- extensions/codex/src/app-server/dynamic-tool-build.ts
- src/infra/outbound/message-action-tts.ts
execution:
kind: script
path: test/e2e/qa-lab/runtime/media-talk-gateway.ts
summary: Starts an isolated Gateway, QA Channel, native Codex app server, and synthetic speech provider. The fixed openai/gpt-5.6-luna fixture model uses only the managed local mock endpoint and catalog; no live credentials or transcription requests are used.
args:
- --scenario
- codex-inbound-message-auto-tts
- --artifact-base
- ${outputDir}

View file

@ -346,14 +346,6 @@ export function prepareEmbeddedAttemptToolBase(params: {
currentMessagingTarget: attempt.currentMessagingTarget,
currentThreadTs: attempt.currentThreadTs,
currentMessageId: attempt.currentMessageId,
currentInboundAudio: attempt.currentInboundAudio,
...(attempt.replyOperation
? {
hasCurrentInboundAudio: () =>
attempt.currentInboundAudio === true ||
attempt.replyOperation?.acceptedSteeredInboundAudio === true,
}
: {}),
includeCoreTools: toolConstructionPlan.includeCoreTools,
includeToolSearchControls: toolSearchControlsEnabledForRun,
toolSearchCatalogExecutor: params.toolSearchCatalogExecutor,

View file

@ -0,0 +1,57 @@
import { describe, expect, it } from "vitest";
import { buildEmbeddedAttemptToolRunContext } from "./attempt-tool-run-context.js";
describe("buildEmbeddedAttemptToolRunContext", () => {
it("carries runtime toolsAllow into coding tool construction", () => {
const context = buildEmbeddedAttemptToolRunContext({
trigger: "manual",
jobId: "job-1",
memoryFlushWritePath: "memory/log.md",
toolsAllow: ["memory_search", "memory_get"],
});
expect(context.trigger).toBe("manual");
expect(context.jobId).toBe("job-1");
expect(context.memoryFlushWritePath).toBe("memory/log.md");
expect(context.runtimeToolAllowlist).toEqual(["memory_search", "memory_get"]);
});
it.each([undefined, false, true])(
"preserves originating inbound audio %s",
(currentInboundAudio) => {
const context = buildEmbeddedAttemptToolRunContext({ currentInboundAudio });
const withOperation = buildEmbeddedAttemptToolRunContext({
currentInboundAudio,
replyOperation: { acceptedSteeredInboundAudio: false },
});
expect(context.hasCurrentInboundAudio?.()).toBe(currentInboundAudio === true);
expect(withOperation.hasCurrentInboundAudio?.()).toBe(currentInboundAudio === true);
},
);
it("reads accepted steering at execution without crossing operation owners", () => {
let acceptedSteeredInboundAudio = false;
const attempt = {
currentInboundAudio: false,
replyOperation: {
get acceptedSteeredInboundAudio() {
return acceptedSteeredInboundAudio;
},
},
};
const context = buildEmbeddedAttemptToolRunContext(attempt);
expect(context.hasCurrentInboundAudio?.()).toBe(false);
attempt.replyOperation = { acceptedSteeredInboundAudio: true };
expect(context.hasCurrentInboundAudio?.()).toBe(false);
acceptedSteeredInboundAudio = true;
expect(context.hasCurrentInboundAudio?.()).toBe(true);
expect(
buildEmbeddedAttemptToolRunContext({
currentInboundAudio: false,
replyOperation: { acceptedSteeredInboundAudio: false },
}).hasCurrentInboundAudio?.(),
).toBe(false);
});
});

View file

@ -8,7 +8,7 @@ import { mergeForcedEmbeddedAttemptToolsAllow } from "./attempt-tool-constructio
import type { EmbeddedRunTrigger } from "./params.js";
/**
* Builds the stable tool-run context forwarded into an embedded-attempt execution.
* Builds the shared tool-run context for embedded and plugin harness attempts.
*/
export function buildEmbeddedAttemptToolRunContext(params: {
thinkLevel?: ThinkLevel;
@ -21,7 +21,10 @@ export function buildEmbeddedAttemptToolRunContext(params: {
swarmOutputSchema?: Record<string, unknown>;
conversationToolPolicy?: GroupToolPolicyConfig;
trace?: DiagnosticTraceContext;
currentInboundAudio?: boolean;
replyOperation?: { readonly acceptedSteeredInboundAudio: boolean };
}) {
const { currentInboundAudio, replyOperation } = params;
// Collector output is mandatory result transport, even on a narrowed tool surface.
const runtimeToolAllowlist = mergeForcedEmbeddedAttemptToolsAllow(params.toolsAllow, {
forceMessageTool: params.forceMessageTool,
@ -35,6 +38,10 @@ export function buildEmbeddedAttemptToolRunContext(params: {
memoryFlushWritePath: params.memoryFlushWritePath,
swarmCollector: params.swarmCollector,
swarmOutputSchema: params.swarmOutputSchema,
currentInboundAudio,
// Read accepted steering from the captured owner when the tool executes.
hasCurrentInboundAudio: () =>
currentInboundAudio === true || replyOperation?.acceptedSteeredInboundAudio === true,
...(runtimeToolAllowlist ? { runtimeToolAllowlist } : {}),
...(params.conversationToolPolicy
? { conversationToolPolicy: params.conversationToolPolicy }

View file

@ -33,7 +33,6 @@ import {
import { composeSystemPromptWithHookContext } from "./attempt-thread-helpers.js";
import { wrapStreamFnSanitizeMalformedToolCalls } from "./attempt-tool-call-replay-sanitization.js";
import { wrapStreamFnTrimToolCallNames } from "./attempt-tool-call-stream-normalization.js";
import { buildEmbeddedAttemptToolRunContext } from "./attempt-tool-run-context.js";
import { wrapStreamFnRepairMalformedToolCallArguments } from "./attempt.tool-call-argument-repair.js";
const llmRuntime = {
@ -121,21 +120,6 @@ function firstBaseContext(baseFn: ReturnType<typeof vi.fn>): { messages: unknown
return call[1] as { messages: unknown[] };
}
describe("buildEmbeddedAttemptToolRunContext", () => {
it("carries runtime toolsAllow into coding tool construction", () => {
const context = buildEmbeddedAttemptToolRunContext({
trigger: "manual",
jobId: "job-1",
memoryFlushWritePath: "memory/log.md",
toolsAllow: ["memory_search", "memory_get"],
});
expect(context.trigger).toBe("manual");
expect(context.jobId).toBe("job-1");
expect(context.memoryFlushWritePath).toBe("memory/log.md");
expect(context.runtimeToolAllowlist).toEqual(["memory_search", "memory_get"]);
});
});
describe("resolvePromptBuildHookResult", () => {
it("preserves prompt-build context fields", async () => {
const hookRunner = {

View file

@ -29,20 +29,18 @@ import {
resetGlobalHookRunner,
} from "../../plugins/hook-runner-global.js";
import { createMockPluginRegistry } from "../../plugins/hooks.test-fixtures.js";
import { resetPluginRuntimeStateForTest, setActivePluginRegistry } from "../../plugins/runtime.js";
import { createTestRegistry } from "../../test-utils/channel-plugins.js";
import { withTempDir } from "../../test-utils/temp-dir.js";
import {
consumePreExecutionBlockedToolCall,
wrapToolWithBeforeToolCallHook,
} from "../agent-tools.before-tool-call.js";
import { createOpenClawTools } from "../openclaw-tools.js";
import { withGatewayToolCallerIdentity } from "./gateway-caller-context.js";
type CreateMessageTool = typeof import("./message-tool-execution.js").createMessageTool;
type CreateOpenClawTools = typeof import("../openclaw-tools.js").createOpenClawTools;
type ResetPluginRuntimeStateForTest =
typeof import("../../plugins/runtime.js").resetPluginRuntimeStateForTest;
type SetActivePluginRegistry = typeof import("../../plugins/runtime.js").setActivePluginRegistry;
type CreateTestRegistry = typeof import("../../test-utils/channel-plugins.js").createTestRegistry;
type RunMessageAction =
typeof import("../../infra/outbound/message-action-runner.js").runMessageAction;
import { createMessageTool } from "./message-tool-execution.js";
type CreateMessageTool = typeof createMessageTool;
const ROOM_EVENT_DELIVERY_HINT = MESSAGE_TOOL_DELIVERY_HINTS[3];
const CRITICAL_THRESHOLD = 20;
@ -52,13 +50,6 @@ const EMPTY_PREPARED_MESSAGE_TOOL_CATALOG = {
getChannel: () => undefined,
} as const;
let createMessageTool: CreateMessageTool;
let createOpenClawTools: CreateOpenClawTools;
let resetPluginRuntimeStateForTest: ResetPluginRuntimeStateForTest;
let setActivePluginRegistry: SetActivePluginRegistry;
let createTestRegistry: CreateTestRegistry;
let actualRunMessageAction: RunMessageAction;
type DescribeMessageTool = NonNullable<
NonNullable<ChannelPlugin["actions"]>["describeMessageTool"]
>;
@ -395,16 +386,9 @@ function expectStringSchema(
}
}
beforeAll(async () => {
({ resetPluginRuntimeStateForTest, setActivePluginRegistry } =
await import("../../plugins/runtime.js"));
({ createTestRegistry } = await import("../../test-utils/channel-plugins.js"));
({ createMessageTool } = await import("./message-tool-execution.js"));
({ createOpenClawTools } = await import("../openclaw-tools.js"));
({ runMessageAction: actualRunMessageAction } = await vi.importActual(
"../../infra/outbound/message-action-runner.js",
));
});
const { runMessageAction: actualRunMessageAction } = await vi.importActual<
typeof import("../../infra/outbound/message-action-runner.js")
>("../../infra/outbound/message-action-runner.js");
const mintedTurnCapabilities: string[] = [];

View file

@ -1,5 +1,6 @@
import assert from "node:assert/strict";
import { randomUUID } from "node:crypto";
// QA Lab producer exercises WebChat media delivery and Talk run control through a real Gateway.
// QA Lab producer exercises speech delivery and Talk run control through a real Gateway.
import fs from "node:fs/promises";
import os from "node:os";
import path from "node:path";
@ -7,15 +8,18 @@ import { pathToFileURL } from "node:url";
import type { OpenClawConfig } from "openclaw/plugin-sdk/config-contracts";
import { formatErrorMessage } from "openclaw/plugin-sdk/error-runtime";
import {
QA_EVIDENCE_FILENAME,
type QaEvidenceSummaryJson,
} from "../../../../extensions/qa-lab/src/evidence-summary.js";
import {
createQaBusState,
createQaChannelTransport,
createQaGatewayChild,
QA_EVIDENCE_FILENAME,
startQaBusServer,
startQaMockOpenAiServer,
type MockOpenAiRequestSnapshot,
type QaEvidenceSummaryJson,
type QaGatewayChild,
} from "../../../../extensions/qa-lab/src/gateway-child.js";
import { startQaMockOpenAiServer } from "../../../../extensions/qa-lab/src/providers/mock-openai/server.js";
} from "../../../../extensions/qa-lab/api.js";
import { GatewayClient, type GatewayClientOptions } from "../../../../src/gateway/client.js";
import type { SessionsListResult } from "../../../../src/gateway/session-utils.types.js";
import type { DiagnosticStabilitySnapshot } from "../../../../src/logging/diagnostic-stability.js";
import {
GATEWAY_CLIENT_MODES,
@ -23,7 +27,7 @@ import {
type GatewayClientMode,
type GatewayClientName,
} from "../../../../src/utils/message-channel.js";
import { stopQaGatewayFixture } from "../../../helpers/qa-gateway-cleanup.js";
import { runQaGatewayFixture, stopQaGatewayFixture } from "../../../helpers/qa-gateway-cleanup.js";
import { createQaScriptEvidenceWriter, type QaScriptEvidenceStatus } from "./script-evidence.js";
const FIXTURE_PLUGIN_ID = "qa-media-talk-runtime";
@ -32,8 +36,9 @@ const FIXTURE_REALTIME_PROVIDER_ID = "qa-realtime";
const FIXTURE_WAV_BASE64 =
"UklGRsQAAABXQVZFZm10IBAAAAABAAEAQB8AAIA+AAACABAAZGF0YaAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA";
const SOURCE_PATH = "test/e2e/qa-lab/runtime/media-talk-gateway.ts";
const CODEX_TTS_MODEL_REF = "openai/gpt-5.6-luna";
type ScenarioId = "webchat-auto-tts" | "active-talk-agent-run-status";
type ScenarioId = keyof typeof SCENARIOS;
type ProducerOptions = {
artifactBase: string;
@ -50,6 +55,7 @@ type ProofResult = {
const SCENARIOS = {
"webchat-auto-tts": {
title: "WebChat auto TTS delivery",
run: runWebchatAutoTtsProof,
sourcePath: "qa/scenarios/media/webchat-auto-tts.yaml",
docsRefs: ["docs/tools/tts.md", "docs/tools/media-overview.md"],
codeRefs: [
@ -59,8 +65,21 @@ const SCENARIOS = {
"src/gateway/server-methods/artifacts.ts",
],
},
"codex-inbound-message-auto-tts": {
title: "Codex inbound-audio message TTS delivery",
run: runCodexInboundMessageAutoTtsProof,
sourcePath: "qa/scenarios/media/codex-inbound-message-auto-tts.yaml",
docsRefs: ["docs/tools/tts.md", "docs/channels/qa-channel.md"],
codeRefs: [
SOURCE_PATH,
"src/agents/embedded-agent-runner/run/attempt-tool-run-context.ts",
"extensions/codex/src/app-server/dynamic-tool-build.ts",
"src/infra/outbound/message-action-tts.ts",
],
},
"active-talk-agent-run-status": {
title: "Active Talk agent-run control boundaries",
run: runActiveTalkAgentRunProof,
sourcePath: "qa/scenarios/runtime/active-talk-agent-run-status.yaml",
docsRefs: ["docs/nodes/talk.md", "docs/web/control-ui.md"],
codeRefs: [
@ -411,6 +430,172 @@ async function runWebchatAutoTtsProof(options: ProducerOptions): Promise<string>
}
}
async function runCodexInboundMessageAutoTtsProof(options: ProducerOptions): Promise<string> {
const fixtureRoot = await fs.mkdtemp(path.join(os.tmpdir(), "openclaw-codex-inbound-tts-"));
const state = createQaBusState();
const transport = createQaChannelTransport(state);
const gatewayOwner = createQaGatewayChild();
let bus: Awaited<ReturnType<typeof startQaBusServer>> | undefined;
let mock: Awaited<ReturnType<typeof startQaMockOpenAiServer>> | undefined;
let details = "";
await runQaGatewayFixture(
async () => {
const fixture = await createFixturePlugin(fixtureRoot);
bus = await startQaBusServer({ state });
mock = await startQaMockOpenAiServer({ modelRefs: [CODEX_TTS_MODEL_REF] });
const mockBaseUrl = mock.baseUrl;
const providerBaseUrl = `${mockBaseUrl}/v1`;
const gateway = await gatewayOwner.start({
repoRoot: options.repoRoot,
forcedRuntime: "codex",
providerMode: "mock-openai",
providerBaseUrl,
primaryModel: CODEX_TTS_MODEL_REF,
alternateModel: CODEX_TTS_MODEL_REF,
transport,
transportBaseUrl: bus.baseUrl,
controlUiEnabled: false,
runtimeEnvPatch: {
OPENCLAW_QA_SPEECH_CALLS_PATH: fixture.speechCallsPath,
OPENCLAW_TTS_PREFS: path.join(fixtureRoot, "tts-prefs.json"),
},
mutateConfig: (config) => {
const withPlugin = withFixturePlugin(config, fixture.pluginDir);
return {
...withPlugin,
messages: { ...withPlugin.messages, visibleReplies: "message_tool" },
tools: {
...withPlugin.tools,
alsoAllow: ["message"],
// The ingress media fact must survive without an STT request.
media: { audio: { enabled: false } },
},
tts: {
auto: "inbound",
mode: "final",
provider: FIXTURE_SPEECH_PROVIDER_ID,
// A broken fixture must never fall back to an external speech endpoint.
providers: { openai: { baseUrl: providerBaseUrl } },
},
};
},
});
await transport.waitReady({ gateway });
const readRequests = async () => {
const response = await fetch(`${mockBaseUrl}/debug/requests`);
assert.equal(response.status, 200, "mock request evidence must remain available");
return (await response.json()) as MockOpenAiRequestSnapshot[];
};
const conversation = { id: "codex-inbound-tts", kind: "direct" as const };
const expectedText = "QA-MESSAGE-DELIVERY-OK";
// The audit fixture authors both message text and additive presentation text.
const expectedBody = `${expectedText}\n\n${expectedText}`;
let previousRunId: string | undefined;
for (const inboundAudio of [true, false]) {
const beforeRequests = await readRequests();
const requestCursor = beforeRequests.at(-1)?.cursor ?? 0;
const sinceIndex = state
.getSnapshot()
.messages.filter((m) => m.direction === "outbound").length;
const beforeSpeech = (await readJsonLines(fixture.speechCallsPath)).length;
await transport.sendInbound({
conversation,
senderId: conversation.id,
text: `message delivery decision send qa check: ${inboundAudio ? "audio" : "text"} turn`,
...(inboundAudio
? {
attachments: [
{
id: "synthetic-voice-note",
kind: "audio" as const,
mimeType: "audio/wav",
fileName: "voice-note.wav",
contentBase64: FIXTURE_WAV_BASE64,
},
],
}
: {}),
});
const outbound = await transport.waitForOutbound({
conversation,
sinceIndex,
textIncludes: expectedText,
timeoutMs: 60_000,
});
// Wait for this admitted turn to finish before testing the next ingress.
// Sending on message delivery alone could accidentally exercise steering.
const session = await transport.waitForCondition(async () => {
const result = (await gateway.call("sessions.list", { limit: 20 })) as SessionsListResult;
const row = result.sessions.find((entry) => entry.lastChannel === transport.id);
return row &&
!row.hasActiveRun &&
row.status === "done" &&
row.lastRunId &&
row.lastRunId !== previousRunId
? row
: undefined;
}, 60_000);
assert.equal(session.agentRuntime?.id, "codex", "the real Codex runtime must own the turn");
assert.equal(session.status, "done");
assert.ok(session.lastRunId, "the completed turn must have an owner-recorded run ID");
assert.notEqual(session.lastRunId, previousRunId, "each ingress must complete a fresh run");
assert.ok(!session.lastRunError);
assert.notEqual(session.abortedLastRun, true);
previousRunId = session.lastRunId;
const requests = (await readRequests()).filter((request) => request.cursor > requestCursor);
const sends = requests.filter((request) => request.plannedToolName === "message");
assert.equal(sends.length, 1, "one dynamic message tool must deliver the reply");
const send = sends[0];
assert.ok(send?.plannedToolCallId, "the mock must record the dynamic call identity");
assert.equal(send.plannedToolArgs?.action, "send");
assert.equal(send.plannedToolArgs?.final, true);
assert.equal(send.plannedToolArgs?.voiceText, undefined, "speech must be automatic");
assert.equal(send.plannedToolArgs?.asVoice, undefined, "speech must not be forced");
// Final source delivery closes the native turn after its tool response;
// it does not require another provider request carrying that result.
const history = await gateway.call("chat.history", { sessionKey: session.key, limit: 20 });
assert.ok(
collectRecords(history).some(
(record) =>
record.role === "toolResult" &&
record.toolCallId === send.plannedToolCallId &&
record.toolName === "message" &&
record.isError === false,
),
"the Gateway must record the successful dynamic message tool result",
);
assert.equal(
state.getSnapshot().messages.filter((message) => message.direction === "outbound")
.length - sinceIndex,
1,
"the admitted turn must deliver exactly one visible reply",
);
assert.equal(outbound.text, expectedBody);
const attachments = outbound.attachments ?? [];
assert.equal(
attachments.length,
inboundAudio ? 1 : 0,
`${inboundAudio ? "audio" : "text"} ingress must control automatic speech delivery`,
);
const speechCalls = await readJsonLines(fixture.speechCallsPath);
assert.equal(speechCalls.length - beforeSpeech, inboundAudio ? 1 : 0);
if (inboundAudio) {
assert.equal(attachments[0]?.kind, "audio");
assert.equal(attachments[0]?.mimeType, "audio/wav");
assert.equal(attachments[0]?.contentBase64, FIXTURE_WAV_BASE64);
assert.equal(speechCalls.at(-1)?.text, expectedText);
}
}
details = `real Codex app-server and Gateway pid=${gateway.pid ?? "unknown"}; two final message tool replies; audio ingress synthesized and delivered exact WAV bytes; subsequent text ingress stayed text-only; syntheses=1`;
},
() => stopQaGatewayFixture(gatewayOwner),
() => mock?.stop(),
() => bus?.stop(),
() => fs.rm(fixtureRoot, { force: true, recursive: true }),
);
return details;
}
function assertControlResult(
value: unknown,
expected: { mode: string; active?: boolean; queued?: boolean; aborted?: boolean },
@ -606,10 +791,7 @@ async function runActiveTalkAgentRunProof(options: ProducerOptions): Promise<str
async function produceProof(options: ProducerOptions): Promise<ProofResult> {
const startedAt = Date.now();
try {
const details =
options.scenarioId === "webchat-auto-tts"
? await runWebchatAutoTtsProof(options)
: await runActiveTalkAgentRunProof(options);
const details = await SCENARIOS[options.scenarioId].run(options);
return { details, durationMs: Math.max(1, Date.now() - startedAt), status: "pass" };
} catch (error) {
return {
@ -627,7 +809,10 @@ async function runMediaTalkGatewayProducer(
const writer = createQaScriptEvidenceWriter({
artifactBase: options.artifactBase,
logFileName: `${options.scenarioId}.log`,
primaryModel: "mock-openai/gpt-5.6-luna",
primaryModel:
options.scenarioId === "codex-inbound-message-auto-tts"
? CODEX_TTS_MODEL_REF
: "mock-openai/gpt-5.6-luna",
providerMode: "mock-openai",
repoRoot: options.repoRoot,
target: {