diff --git a/packages/adapter-utils/src/acpx-engine/execute.test.ts b/packages/adapter-utils/src/acpx-engine/execute.test.ts index 324e20cf43..b3bcd72acd 100644 --- a/packages/adapter-utils/src/acpx-engine/execute.test.ts +++ b/packages/adapter-utils/src/acpx-engine/execute.test.ts @@ -36,6 +36,8 @@ import { type AcpxEngineExecutorOptions, } from "./execute.js"; import { runChildProcess } from "../server-utils.js"; +import { createDuplexBridgeBroker } from "../duplex-bridge-broker.js"; +import type { CommandManagedDuplexChannel } from "../command-managed-runtime.js"; import { getActiveStepContext, runWithRuntimeParent, @@ -5731,3 +5733,188 @@ describe("ACPX engine run lifecycle corrections (F3: one teardown error policy)" } }); }); + +describe("ACPX engine sandbox duplex run-disposition seam (fail-closed)", () => { + beforeEach(() => { + vi.clearAllMocks(); + }); + + // A minimal in-memory duplex channel. The test drives a channel exit to latch a + // real loss in a real broker, so the seam reads a real run disposition. + function createFakeDuplexChannel() { + let exitListener: ((exit: { exitCode: number | null }) => void) | null = null; + const channel: CommandManagedDuplexChannel = { + write(): void {}, + onData(): void {}, + onExit(listener: (exit: { exitCode: number | null }) => void): void { + exitListener = listener; + }, + stop(): void {}, + close(): Promise { + return Promise.resolve(); + }, + }; + return { + channel, + emitExit: (exit: { exitCode: number | null }) => exitListener?.(exit), + }; + } + + // Wrap a started real broker in a paperclip bridge handle. The handle exposes + // the same run-disposition surface the sandbox bridge exposes, so the seam runs + // against the real latch and the real orderly-completion mark. + function bridgeOverBroker(fake: ReturnType) { + const broker = createDuplexBridgeBroker({ + channel: fake.channel, + forwardRequest: async () => ({ status: 200 }), + }); + broker.start(); + const markOrderlyCompletion = vi.fn(() => broker.markOrderlyCompletion()); + const stop = vi.fn(async () => {}); + const handle = { + env: { + PAPERCLIP_API_URL: "http://127.0.0.1:1", + PAPERCLIP_API_KEY: "bridge-token", + PAPERCLIP_API_BRIDGE_MODE: "duplex_v1", + }, + readRunDisposition: () => broker.runDisposition, + markOrderlyCompletion, + stop, + }; + return { broker, handle, markOrderlyCompletion }; + } + + // A runtime whose one turn completes cleanly. The `beforeResult` hook runs at + // the exact point the ACP terminal resolves, so the test orders a duplex loss + // before the completion when it needs to. + function runtimeWithControlledResult(beforeResult?: () => void) { + return { + ensureSession: async () => ({ + backendSessionId: "backend-session", + agentSessionId: "agent-session", + runtimeSessionName: "runtime-session", + }), + startTurn: () => ({ + events: (async function* () { + yield { type: "done", stopReason: "end_turn" }; + })(), + result: (async () => { + beforeResult?.(); + return { status: "completed" as const, stopReason: "end_turn" }; + })(), + cancel: async () => {}, + }), + setConfigOption: async () => {}, + close: async () => {}, + }; + } + + async function setupRemoteSandbox() { + const root = await makeTempRoot(); + const stateDir = path.join(root, "state"); + const localCwd = path.join(root, "worktree"); + const remoteCwd = path.join(root, "remote-workspace"); + await fs.mkdir(localCwd, { recursive: true }); + await fs.mkdir(remoteCwd, { recursive: true }); + await fs.writeFile(path.join(localCwd, "hello.txt"), "hi", "utf8"); + const runner = createLocalSandboxRunner(); + const executionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "fake-plugin", + remoteCwd, + runner, + }; + return { root, stateDir, localCwd, remoteCwd, executionTarget }; + } + + async function runRemote( + handle: unknown, + runtime: unknown, + sandbox: Awaited>, + ) { + vi.mocked(startAdapterExecutionTargetPaperclipBridge).mockImplementationOnce( + async () => handle as never, + ); + vi.mocked(startAdapterExecutionTargetProcessSessionBridge).mockImplementationOnce( + async () => ({ agentCommand: null, stop: async () => {} }) as never, + ); + const execute = createAcpxEngineExecutor({ + createRuntime: () => runtime as never, + }); + return await execute({ + runId: "run-duplex-seam", + agent: { id: "agent-1", companyId: "company-1" }, + runtime: {}, + config: { + agent: "custom", + agentCommand: "node ./fake-acp.js", + stateDir: sandbox.stateDir, + cwd: sandbox.localCwd, + }, + context: {}, + authToken: "real-run-jwt", + executionTarget: sandbox.executionTarget, + onLog: async () => {}, + onMeta: async () => {}, + onEvent: async () => {}, + } as never); + } + + it("fails a completed run when the duplex channel was lost before the completion", async () => { + const sandbox = await setupRemoteSandbox(); + const fake = createFakeDuplexChannel(); + const { handle, markOrderlyCompletion } = bridgeOverBroker(fake); + // Latch the loss before the ACP terminal resolves. + const runtime = runtimeWithControlledResult(() => fake.emitExit({ exitCode: 1 })); + + const result = await runRemote(handle, runtime, sandbox); + + // The lost channel overrides the nominally completed terminal to a failure. + expect(result.exitCode).not.toBe(0); + expect(result.errorCode).toBe("duplex_channel_lost"); + // The message carries only the typed loss reason, not raw provider text. + expect(result.errorMessage).toContain("provider_exit"); + expect(result.resultJson).toMatchObject({ status: "failed" }); + // The seam did not mark an orderly completion for a lost channel. + expect(markOrderlyCompletion).not.toHaveBeenCalled(); + }); + + it("keeps a completed run a success when the channel stays live, and a later teardown loss is benign", async () => { + const sandbox = await setupRemoteSandbox(); + const fake = createFakeDuplexChannel(); + const { broker, handle, markOrderlyCompletion } = bridgeOverBroker(fake); + // No loss before the completion. + const runtime = runtimeWithControlledResult(); + + const result = await runRemote(handle, runtime, sandbox); + + expect(result.exitCode).toBe(0); + expect(result.errorCode ?? null).toBeNull(); + // The seam marked the orderly completion for the success-eligible terminal. + expect(markOrderlyCompletion).toHaveBeenCalledTimes(1); + // A teardown loss ordered after the orderly completion is a normal teardown, + // so the run disposition stays a success. + fake.emitExit({ exitCode: 0 }); + expect(broker.runDisposition.failed).toBe(false); + }); + + it("does not let a later completion or activity clear the loss latch", async () => { + const sandbox = await setupRemoteSandbox(); + const fake = createFakeDuplexChannel(); + const { broker, handle } = bridgeOverBroker(fake); + // Latch the loss before the ACP terminal resolves. + const runtime = runtimeWithControlledResult(() => fake.emitExit({ exitCode: 1 })); + + const result = await runRemote(handle, runtime, sandbox); + + expect(result.exitCode).not.toBe(0); + expect(result.errorCode).toBe("duplex_channel_lost"); + // A later orderly-completion mark and further channel activity cannot clear + // the latched loss. + broker.markOrderlyCompletion(); + fake.emitExit({ exitCode: 0 }); + expect(broker.runDisposition.failed).toBe(true); + expect(broker.runDisposition.lossReason).toBe("provider_exit"); + }); +}); diff --git a/packages/adapter-utils/src/acpx-engine/execute.ts b/packages/adapter-utils/src/acpx-engine/execute.ts index 06ab95bb42..08ec83295b 100644 --- a/packages/adapter-utils/src/acpx-engine/execute.ts +++ b/packages/adapter-utils/src/acpx-engine/execute.ts @@ -15,6 +15,8 @@ import type { import { adapterExecutionTargetSessionIdentity, describeAdapterExecutionTarget, + adapterExecutionTargetDuplexTelemetryRecorder, + adapterExecutionTargetEnablesSandboxDuplexBridge, formatAdapterExecutionTimeoutErrorMessage, formatAdapterExecutionTimeoutStartLogLine, prepareAdapterExecutionTargetRuntime, @@ -31,6 +33,7 @@ import { type PreparedAdapterExecutionTargetRuntime, type SandboxAdditionalSource, } from "@paperclipai/adapter-utils/execution-target"; +import type { DuplexLossReason } from "../duplex-telemetry.js"; import { DEFAULT_PAPERCLIP_AGENT_PROMPT_TEMPLATE, applyPaperclipWorkspaceEnv, @@ -2047,6 +2050,8 @@ async function buildRuntime(input: { adapterKey: input.engine.adapterType, timeoutSec, hostApiToken: env.PAPERCLIP_API_KEY, + enableSandboxDuplexBridge: adapterExecutionTargetEnablesSandboxDuplexBridge(remoteTarget), + duplexTelemetryRecorder: adapterExecutionTargetDuplexTelemetryRecorder(remoteTarget), onLog: input.ctx.onLog, getRuntimeParentContext: input.getRuntimeParentContext, runtimeSpan: input.runtimeSpan, @@ -4003,6 +4008,36 @@ export function createAcpxEngineExecutor(deps: AcpxEngineExecutorOptions = {}) { if (input.kind === "terminal") { const terminal = input.terminal; const timedOut = input.timedOut; + // Read the sandbox duplex control-channel disposition at the ACP + // terminal-finalization boundary, before the bridge teardown. A control + // channel that died mid-turn latches a failure with a typed loss reason; + // a healthy channel or a normal-teardown loss reports a success. Only a + // nominally completed, non-timed-out terminal is success-eligible, so the + // seam reads the disposition only there. For that success-eligible + // terminal the seam marks the host-observed orderly completion, so a later + // teardown loss cannot flip the run to a failure. The file bridge path + // never sets these methods, so the optional calls no-op there. + let duplexLossReason: DuplexLossReason | null = null; + if (terminal.status === "completed" && !timedOut) { + const disposition = prepared.paperclipBridge?.readRunDisposition?.() ?? null; + if (disposition?.failed) { + duplexLossReason = disposition.lossReason ?? "other"; + } else { + prepared.paperclipBridge?.markOrderlyCompletion?.(); + } + } + // A terminal that reports "completed" but whose duplex control channel + // died before the completion is not a success. The seam fails it closed. + const channelLost = duplexLossReason !== null; + // Build the boundary failure message only from the closed loss-reason + // enum, so no raw provider text rides the message. + const channelLostMessage = duplexLossReason + ? `The sandbox duplex control channel was lost (${duplexLossReason}) before the run completed.` + : null; + // A completed, non-timed-out turn whose channel stayed live is the one + // success path. Every other outcome — a failed, cancelled, or timed-out + // terminal, or a completed terminal with a lost channel — is a failure. + const turnSucceeded = terminal.status === "completed" && !timedOut && !channelLost; // Read usage before the settlement can discard runtime state. const postTurnStatus = await readRuntimeStatus(runtime, sessionHandle); const turnUsage = summarizeAcpxTurnUsage({ @@ -4026,10 +4061,12 @@ export function createAcpxEngineExecutor(deps: AcpxEngineExecutorOptions = {}) { handle: sessionHandle, reason: timedOut ? "paperclip timeout cleanup" - : failedTurn - ? `paperclip turn ${terminal.status}` - : "paperclip completed turn cleanup", - discardPersistentState: terminal.status === "cancelled" || timedOut, + : channelLost + ? "paperclip duplex channel lost cleanup" + : failedTurn + ? `paperclip turn ${terminal.status}` + : "paperclip completed turn cleanup", + discardPersistentState: terminal.status === "cancelled" || timedOut || channelLost, dropWarmEntry: false, recordCloseError: false, cancelTurnReason: null, @@ -4037,23 +4074,32 @@ export function createAcpxEngineExecutor(deps: AcpxEngineExecutorOptions = {}) { const errorMessage = timedOut ? formatAdapterExecutionTimeoutErrorMessage(prepared.timeoutResolution) - : resultErrorMessage(terminal); + : channelLost + ? channelLostMessage + : resultErrorMessage(terminal); const terminalStopReason = terminal.status === "failed" ? terminal.error.message : terminal.stopReason; await emitAcpxLog(ctx, { - type: terminal.status === "completed" ? "acpx.result" : "acpx.error", - summary: terminal.status, + type: turnSucceeded ? "acpx.result" : "acpx.error", + summary: channelLost ? "duplex_channel_lost" : terminal.status, stopReason: terminalStopReason, message: errorMessage, }); // The one clean-completion path clears the run failure flag; every other - // path keeps it set, so the run root span closes with error status. - runFailed = terminal.status === "completed" && !timedOut ? false : true; + // path keeps it set, so the run root span closes with error status. A + // completed terminal with a lost duplex channel keeps the flag set. + runFailed = turnSucceeded ? false : true; capturedResult = { - exitCode: terminal.status === "completed" ? 0 : 1, + exitCode: turnSucceeded ? 0 : 1, signal: timedOut ? "SIGTERM" : null, timedOut, errorMessage, - errorCode: terminal.status === "failed" ? "acpx_turn_failed" : timedOut ? "acpx_timeout" : null, + errorCode: terminal.status === "failed" + ? "acpx_turn_failed" + : timedOut + ? "acpx_timeout" + : channelLost + ? "duplex_channel_lost" + : null, sessionId: sessionHandle.backendSessionId ?? sessionHandle.runtimeSessionName, sessionParams: buildSessionParams({ prepared, handle: sessionHandle }), sessionDisplayId: sessionHandle.agentSessionId ?? sessionHandle.backendSessionId ?? sessionHandle.runtimeSessionName, @@ -4063,7 +4109,7 @@ export function createAcpxEngineExecutor(deps: AcpxEngineExecutorOptions = {}) { ...(turnUsage.usage ? { usage: turnUsage.usage, usageBasis: "per_run" as const } : {}), costUsd: turnUsage.costUsd, resultJson: { - status: terminal.status, + status: channelLost ? "failed" : terminal.status, stopReason: terminalStopReason, permissionMode: prepared.permissionMode, mode: prepared.mode, @@ -4081,12 +4127,12 @@ export function createAcpxEngineExecutor(deps: AcpxEngineExecutorOptions = {}) { }), clearSession, }; - // The turn phase finished. A completed, non-timed-out turn is `ok`; every - // other terminal outcome is `failed`. + // The turn phase finished. A completed, non-timed-out turn with a live + // duplex channel is `ok`; every other terminal outcome is `failed`. await emitPhase( "turn", turnPhaseStart, - terminal.status === "completed" && !timedOut ? "ok" : "failed", + turnSucceeded ? "ok" : "failed", ); // Return the typed turn completion so the coordinator settles for the right // cause. The completion carries no live resources; the settlement claims the @@ -4117,6 +4163,20 @@ export function createAcpxEngineExecutor(deps: AcpxEngineExecutorOptions = {}) { resources: emptyConsumed, }; } + // A completed terminal whose duplex control channel died mid-turn returns + // a failed completion, so the coordinator settles for a failure and the + // reuse decision forbids a save. The message carries only the typed loss + // reason, so no raw provider text rides the cause. + if (channelLost) { + return { + kind: "failed", + cause: { + kind: "turn_failed", + error: new Error(channelLostMessage ?? "The sandbox duplex control channel was lost."), + }, + resources: emptyConsumed, + }; + } return { kind: "finalized" }; } const err = input.error; diff --git a/packages/adapter-utils/src/duplex-bridge-broker.test.ts b/packages/adapter-utils/src/duplex-bridge-broker.test.ts new file mode 100644 index 0000000000..d254742a26 --- /dev/null +++ b/packages/adapter-utils/src/duplex-bridge-broker.test.ts @@ -0,0 +1,631 @@ +import { afterEach, describe, expect, it } from "vitest"; + +import { + assertDuplexBrokerLimits, + createDuplexBridgeBroker, + type DuplexBridgeBroker, + type DuplexBrokerForwardResult, +} from "./duplex-bridge-broker.js"; +import { + DEFAULT_MAX_DUPLEX_REQUEST_ID_BYTES, + DUPLEX_FRAME_VERSION, + DuplexFrameDecoder, + encodeDuplexFrame, + type DuplexRequestFrame, + type DuplexResponseFrame, +} from "./duplex-frame-codec.js"; +import { + createDuplexTelemetry, + DUPLEX_SPAN_REQUEST, + type DuplexTelemetryCounterRecord, + type DuplexTelemetryEventRecord, + type DuplexTelemetrySpanRecord, +} from "./duplex-telemetry.js"; +import type { CommandManagedDuplexChannel } from "./command-managed-runtime.js"; + +/** + * Unit harness for the host broker limit gate. + * + * The harness drives frames straight into the broker through an in-memory + * channel. It does not spawn the gateway, so a test injects untrusted request + * frames directly, the same as a malicious provider that controls the transport. + * The channel records each frame the broker writes back, so a test asserts the + * exact response the broker returns for a refused request. + */ + +/** One pending forward the test controls. */ +interface PendingForward { + id: string; + resolve: (result: DuplexBrokerForwardResult) => void; + reject: (error: Error) => void; + settled: boolean; +} + +/** The in-memory channel plus the levers a test uses to drive the broker. */ +interface FakeChannelHarness { + channel: CommandManagedDuplexChannel; + /** Push one request frame into the broker read path. */ + feed: (frame: DuplexRequestFrame) => void; + /** The response frames the broker wrote back, in order. */ + responses: DuplexResponseFrame[]; + /** Every forward call the broker made, in order. */ + forwards: PendingForward[]; + /** Resolve the newest unsettled forward for one id with a 200 result. */ + resolveForward: (id: string, body?: string) => void; + /** Resolve every unsettled forward with a 200 result. */ + resolveAll: () => void; +} + +/** Build one valid request frame with distinctive, secret-looking fields. */ +function requestFrame(id: string, method = "POST"): DuplexRequestFrame { + return { + version: DUPLEX_FRAME_VERSION, + type: "request", + id, + method, + path: `/api/issues/${id}`, + query: "?secret-query=leak", + headers: { authorization: "Bearer super-secret-provider-token" }, + body: JSON.stringify({ secret: "secret-request-body" }), + }; +} + +/** + * Build the in-memory channel harness. The channel keeps the broker read + * listener, so `feed` pushes an encoded frame into the broker. The channel + * decodes each written frame, so `responses` holds the response frames only. + */ +function createFakeChannelHarness(): FakeChannelHarness { + const responses: DuplexResponseFrame[] = []; + const forwards: PendingForward[] = []; + const writtenDecoder = new DuplexFrameDecoder(); + let dataListener: ((chunk: string) => void) | null = null; + + const channel: CommandManagedDuplexChannel = { + write: (data) => { + for (const result of writtenDecoder.push(data)) { + if (result.ok && result.frame.type === "response") { + responses.push(result.frame); + } + } + }, + onData: (listener) => { + dataListener = listener; + }, + onExit: () => undefined, + stop: () => undefined, + close: () => Promise.resolve(), + }; + + return { + channel, + feed: (frame) => { + if (!dataListener) throw new Error("The broker did not bind the data listener."); + dataListener(encodeDuplexFrame(frame)); + }, + responses, + forwards, + resolveForward: (id, body = JSON.stringify({ ok: true })) => { + const forward = [...forwards].reverse().find((entry) => entry.id === id && !entry.settled); + if (!forward) throw new Error(`No unsettled forward for id ${id}.`); + forward.settled = true; + forward.resolve({ status: 200, headers: { "content-type": "application/json" }, body }); + }, + resolveAll: () => { + for (const forward of forwards) { + if (forward.settled) continue; + forward.settled = true; + forward.resolve({ + status: 200, + headers: { "content-type": "application/json" }, + body: JSON.stringify({ ok: true }), + }); + } + }, + }; +} + +/** A forward handler that hands each call to the harness and never auto-resolves. */ +function controllableForward(harness: FakeChannelHarness) { + return (request: DuplexRequestFrame): Promise => + new Promise((resolve, reject) => { + harness.forwards.push({ id: request.id, resolve, reject, settled: false }); + }); +} + +/** The telemetry sink capture. It proves the broker records nothing for a refusal. */ +interface TelemetryCapture { + spans: DuplexTelemetrySpanRecord[]; + counters: DuplexTelemetryCounterRecord[]; + events: DuplexTelemetryEventRecord[]; +} + +/** Build the real telemetry facade over a capturing recorder. */ +function createTelemetryCapture(): { telemetry: ReturnType; capture: TelemetryCapture } { + const capture: TelemetryCapture = { spans: [], counters: [], events: [] }; + const telemetry = createDuplexTelemetry({ + providerKey: "daytona", + recorder: { + recordSpan: (record) => capture.spans.push(record), + incrementCounter: (record) => capture.counters.push(record), + emitEvent: (record) => capture.events.push(record), + }, + }); + return { telemetry, capture }; +} + +describe("duplex bridge broker request limits", () => { + const brokers: DuplexBridgeBroker[] = []; + + afterEach(async () => { + while (brokers.length > 0) { + const broker = brokers.pop(); + if (broker) await broker.close(); + } + }); + + it("rejects a non-positive or non-integer limit at construction", () => { + expect(() => assertDuplexBrokerLimits({ maxInFlightRequests: 0, maxLifetimeRequests: 10 })).toThrow(); + expect(() => assertDuplexBrokerLimits({ maxInFlightRequests: 5, maxLifetimeRequests: 0 })).toThrow(); + expect(() => assertDuplexBrokerLimits({ maxInFlightRequests: 1.5, maxLifetimeRequests: 10 })).toThrow(); + expect(() => + assertDuplexBrokerLimits({ maxInFlightRequests: Number.POSITIVE_INFINITY, maxLifetimeRequests: 10 }), + ).toThrow(); + expect(() => assertDuplexBrokerLimits({ maxInFlightRequests: 8, maxLifetimeRequests: 16 })).not.toThrow(); + }); + + it("bounds the in-flight forwards and refuses the excess with a retryable terminal response", () => { + const harness = createFakeChannelHarness(); + const { telemetry, capture } = createTelemetryCapture(); + const maxInFlightRequests = 4; + const broker = createDuplexBridgeBroker({ + channel: harness.channel, + forwardRequest: controllableForward(harness), + telemetry, + maxInFlightRequests, + }); + brokers.push(broker); + broker.start(); + + // Inject more than the limit of unique valid frames. Every forward hangs, so + // the broker holds the maximum number of in-flight forwards. + const total = 10; + const ids = Array.from({ length: total }, (_unused, index) => `flight-${index}`); + for (const id of ids) harness.feed(requestFrame(id)); + + // The broker forwarded only up to the limit. It refused the rest. + expect(harness.forwards).toHaveLength(maxInFlightRequests); + expect(harness.responses).toHaveLength(total - maxInFlightRequests); + for (const response of harness.responses) { + expect(response.status).toBe(503); + expect(response.outcome).toBe("unavailable"); + expect(response.headers["x-paperclip-bridge-outcome"]).toBe("unavailable"); + expect(JSON.parse(response.body)).toEqual({ + error: "Duplex broker capacity limit reached.", + outcome: "unavailable", + retryable: true, + }); + } + // The refusal leaks no route, query, body, or token. + const refusalText = harness.responses.map((response) => encodeDuplexFrame(response)).join(""); + expect(refusalText).not.toContain("secret"); + expect(refusalText).not.toContain("super-secret-provider-token"); + expect(refusalText).not.toContain("/api/issues/"); + + // The broker recorded no request span and no counter for a refusal. Only the + // delivered requests below produce a span. + expect(capture.spans).toHaveLength(0); + expect(capture.counters).toHaveLength(0); + + // Resolve the in-flight forwards. The broker delivers one response for each + // and drains the pending bookkeeping. + harness.resolveAll(); + }); + + it("frees capacity after a delivered request and forwards a resent refused id (no-replay preserved)", async () => { + const harness = createFakeChannelHarness(); + const maxInFlightRequests = 2; + const broker = createDuplexBridgeBroker({ + channel: harness.channel, + forwardRequest: controllableForward(harness), + maxInFlightRequests, + }); + brokers.push(broker); + broker.start(); + + // Saturate the in-flight limit with two hung forwards. + harness.feed(requestFrame("keep-0")); + harness.feed(requestFrame("keep-1")); + expect(harness.forwards).toHaveLength(2); + + // A third unique id is refused, retryable. The broker did not forward it and + // did not retain its id. + harness.feed(requestFrame("resend-me")); + expect(harness.forwards).toHaveLength(2); + expect(harness.responses).toHaveLength(1); + expect(JSON.parse(harness.responses[0].body).retryable).toBe(true); + + // Free one in-flight slot. The delivered request drains the pending record. + harness.resolveForward("keep-0"); + await Promise.resolve(); + + // The gateway resends the refused id. Capacity is free now, so the broker + // forwards it exactly one time. This proves the refusal did not poison the id. + harness.feed(requestFrame("resend-me")); + expect(harness.forwards.filter((forward) => forward.id === "resend-me")).toHaveLength(1); + + // A resend of an already-forwarded id never reaches the forward twice. + const forwardedKeep1 = harness.forwards.filter((forward) => forward.id === "keep-1").length; + harness.feed(requestFrame("keep-1")); + expect(harness.forwards.filter((forward) => forward.id === "keep-1")).toHaveLength(forwardedKeep1); + + harness.resolveAll(); + }); + + it("bounds the retained request ids over the channel lifetime and refuses each new id past the limit", async () => { + const harness = createFakeChannelHarness(); + const { telemetry, capture } = createTelemetryCapture(); + const maxLifetimeRequests = 3; + const broker = createDuplexBridgeBroker({ + channel: harness.channel, + // Resolve every forward at once, so each dispatched id leaves the in-flight + // count but stays in the retained-id set. + forwardRequest: async (request: DuplexRequestFrame): Promise => { + harness.forwards.push({ id: request.id, resolve: () => undefined, reject: () => undefined, settled: true }); + return { status: 200, headers: { "content-type": "application/json" }, body: JSON.stringify({ ok: true }) }; + }, + telemetry, + maxLifetimeRequests, + }); + brokers.push(broker); + broker.start(); + + // Inject more than the lifetime limit of unique valid frames. + const total = 9; + for (let index = 0; index < total; index += 1) { + harness.feed(requestFrame(`life-${index}`)); + await Promise.resolve(); + } + + // The broker forwarded only up to the lifetime limit. The retained-id set + // stays bounded, because a refused id never joins it. + expect(harness.forwards).toHaveLength(maxLifetimeRequests); + + // The broker delivered a real response for each forwarded request, then a + // bounded terminal refusal for each id past the limit. + const refusals = harness.responses.filter((response) => response.status === 503); + expect(refusals).toHaveLength(total - maxLifetimeRequests); + for (const refusal of refusals) { + expect(refusal.outcome).toBe("unavailable"); + // A lifetime refusal is terminal, so a resend does not help. + expect(JSON.parse(refusal.body).retryable).toBe(false); + } + + // A resend of a refused id stays refused, so the set never grows. + harness.feed(requestFrame("life-8")); + await Promise.resolve(); + expect(harness.forwards).toHaveLength(maxLifetimeRequests); + + // The telemetry surface stayed on the fixed request span for the delivered + // requests only. No refusal produced a span, a counter, or a loss record. + expect(capture.spans).toHaveLength(maxLifetimeRequests); + for (const span of capture.spans) { + expect(span.name).toBe(DUPLEX_SPAN_REQUEST); + expect(span.dimensions.outcome).toBe("ok"); + expect(Object.keys(span.dimensions).sort()).toEqual(["outcome", "provider", "transport"]); + } + expect(capture.counters).toHaveLength(0); + expect(broker.lossRecord).toBeNull(); + expect(broker.state).toBe("open"); + }); + + it("forwards nothing and fails the channel closed under a flood of over-limit ids", () => { + const harness = createFakeChannelHarness(); + const broker = createDuplexBridgeBroker({ + channel: harness.channel, + forwardRequest: controllableForward(harness), + }); + brokers.push(broker); + broker.start(); + + // Send a sustained flood of distinct ids, each one byte over the id bound. The + // codec rejects each frame on the read path, so no over-limit id ever reaches + // the retained-id set or a forward. + const overLimitId = (index: number): string => + `${"a".repeat(DEFAULT_MAX_DUPLEX_REQUEST_ID_BYTES)}-${index}`; + for (let index = 0; index < 20; index += 1) { + harness.feed(requestFrame(overLimitId(index))); + } + + // The broker forwarded nothing. The retained-id set never grew, because the + // set only grows on a dispatched forward, and the broker dispatched none. + expect(harness.forwards).toHaveLength(0); + // The first over-limit frame is a protocol failure, so the broker fails the + // whole channel closed. It stays lost for the rest of the flood. + expect(broker.state).toBe("lost"); + expect(broker.lossRecord?.reason).toBe("protocol_failure"); + }); + + it("forwards a maximal-size id exactly once and does not forward a resend (no-replay preserved)", () => { + const harness = createFakeChannelHarness(); + const broker = createDuplexBridgeBroker({ + channel: harness.channel, + forwardRequest: controllableForward(harness), + }); + brokers.push(broker); + broker.start(); + + // An id at the maximum byte size is allowed. The broker forwards it one time. + const maxId = "a".repeat(DEFAULT_MAX_DUPLEX_REQUEST_ID_BYTES); + harness.feed(requestFrame(maxId)); + expect(harness.forwards.filter((forward) => forward.id === maxId)).toHaveLength(1); + + // Deliver the response, then resend the same id. The broker retained the id, + // so the resend never reaches the forward a second time. + harness.resolveForward(maxId); + harness.feed(requestFrame(maxId)); + expect(harness.forwards.filter((forward) => forward.id === maxId)).toHaveLength(1); + expect(broker.state).toBe("open"); + }); + + it("maps a forward that resolves with the indeterminate marker to a non-retryable response and retains the id", async () => { + const harness = createFakeChannelHarness(); + const broker = createDuplexBridgeBroker({ + channel: harness.channel, + forwardRequest: controllableForward(harness), + }); + brokers.push(broker); + broker.start(); + + // The forward handler resolves with a 504 that carries the indeterminate + // marker. This is the shape the host forward returns for a response-read + // failure after the host commit. The broker must keep the marker and the + // non-retryable body, so the gateway maps it to a non-retryable status and a + // caller never repeats a possibly-committed mutation. + harness.feed(requestFrame("commit-1")); + const forward = harness.forwards.find((entry) => entry.id === "commit-1" && !entry.settled); + if (!forward) throw new Error("The broker did not forward the request."); + forward.settled = true; + forward.resolve({ + status: 504, + headers: { + "content-type": "application/json", + "x-paperclip-bridge-outcome": "indeterminate", + }, + body: JSON.stringify({ + error: "Bridge response body exceeded the configured size limit of 32 bytes.", + outcome: "indeterminate", + retryable: false, + }), + }); + // The broker answers inside the forward `then` microtask, so let it settle. + await Promise.resolve(); + + expect(harness.responses).toHaveLength(1); + const response = harness.responses[0]!; + expect(response.status).toBe(504); + expect(response.outcome).toBe("indeterminate"); + expect(response.headers["x-paperclip-bridge-outcome"]).toBe("indeterminate"); + expect(JSON.parse(response.body)).toEqual({ + error: "Bridge response body exceeded the configured size limit of 32 bytes.", + outcome: "indeterminate", + retryable: false, + }); + + // The broker delivered a response, so it retained the id. A resend never + // reaches the forward a second time, so a caller that ignores the + // non-retryable status still cannot repeat the mutation through the broker. + harness.feed(requestFrame("commit-1")); + expect(harness.forwards.filter((entry) => entry.id === "commit-1")).toHaveLength(1); + expect(broker.state).toBe("open"); + }); + + it("maps a forward rejection for a mutating method to a non-retryable indeterminate response and retains the id", async () => { + const harness = createFakeChannelHarness(); + const broker = createDuplexBridgeBroker({ + channel: harness.channel, + forwardRequest: controllableForward(harness), + }); + brokers.push(broker); + broker.start(); + + // The forward rejects before the host delivers a response. A fetch can + // reject after the request reaches the host and the host commits, but + // before the response headers arrive. The broker did not abort the + // forward, so a POST must return a non-retryable indeterminate response. + // A retryable status would let a caller repeat a committed mutation. + harness.feed(requestFrame("mutate-1", "POST")); + const forward = harness.forwards.find((entry) => entry.id === "mutate-1" && !entry.settled); + if (!forward) throw new Error("The broker did not forward the request."); + forward.settled = true; + forward.reject(new Error("fetch failed before the response headers arrived")); + // The broker answers inside the forward rejection microtask, so let it settle. + await new Promise((resolve) => setTimeout(resolve, 0)); + + expect(harness.responses).toHaveLength(1); + const response = harness.responses[0]!; + expect(response.status).toBe(504); + expect(response.outcome).toBe("indeterminate"); + expect(response.headers["x-paperclip-bridge-outcome"]).toBe("indeterminate"); + expect(JSON.parse(response.body)).toEqual({ + error: "fetch failed before the response headers arrived", + outcome: "indeterminate", + retryable: false, + }); + + // The broker delivered a response, so it retained the id. A resend never + // reaches the forward a second time, so a caller that ignores the + // non-retryable status still cannot repeat the mutation through the broker. + harness.feed(requestFrame("mutate-1", "POST")); + expect(harness.forwards.filter((entry) => entry.id === "mutate-1")).toHaveLength(1); + expect(broker.state).toBe("open"); + }); + + it("maps a forward rejection for a safe method to a retryable response", async () => { + const harness = createFakeChannelHarness(); + const broker = createDuplexBridgeBroker({ + channel: harness.channel, + forwardRequest: controllableForward(harness), + }); + brokers.push(broker); + broker.start(); + + // A safe method never changes host state, so a retry cannot double-apply a + // mutation. A forward rejection for a GET stays retryable: the broker + // returns a 502 with the completed outcome, so the gateway passes it through. + harness.feed(requestFrame("read-1", "GET")); + const forward = harness.forwards.find((entry) => entry.id === "read-1" && !entry.settled); + if (!forward) throw new Error("The broker did not forward the request."); + forward.settled = true; + forward.reject(new Error("fetch failed before the response headers arrived")); + await new Promise((resolve) => setTimeout(resolve, 0)); + + expect(harness.responses).toHaveLength(1); + const response = harness.responses[0]!; + expect(response.status).toBe(502); + expect(response.outcome).toBe("completed"); + expect(response.headers["x-paperclip-bridge-outcome"]).toBeUndefined(); + expect(JSON.parse(response.body)).toEqual({ + error: "fetch failed before the response headers arrived", + }); + }); + + it("maps a forward-budget timeout for a safe method to a retryable response", async () => { + const harness = createFakeChannelHarness(); + const broker = createDuplexBridgeBroker({ + channel: harness.channel, + forwardRequest: controllableForward(harness), + // A tiny forward budget aborts the controller before the manual rejection. + budgets: { forwardTimeoutMs: 5 }, + }); + brokers.push(broker); + broker.start(); + + // A safe method never changes host state. A forward timeout for a GET must + // stay retryable: the broker returns a 504 with the completed outcome and no + // indeterminate marker, so the gateway does not map it to a terminal 409. + harness.feed(requestFrame("read-timeout-1", "GET")); + const forward = harness.forwards.find((entry) => entry.id === "read-timeout-1" && !entry.settled); + if (!forward) throw new Error("The broker did not forward the request."); + + // Wait for the forward-budget timer to abort the controller, then reject the + // forward. The rejection handler now sees the aborted signal. + await new Promise((resolve) => setTimeout(resolve, 20)); + forward.settled = true; + forward.reject(new Error("Duplex broker forward budget exceeded.")); + await new Promise((resolve) => setTimeout(resolve, 0)); + + expect(harness.responses).toHaveLength(1); + const response = harness.responses[0]!; + expect(response.status).toBe(504); + expect(response.outcome).toBe("completed"); + expect(response.headers["x-paperclip-bridge-outcome"]).toBeUndefined(); + expect(JSON.parse(response.body)).toEqual({ + error: "Duplex broker forward budget exceeded.", + retryable: true, + }); + }); + + it("maps a forward-budget timeout for a mutating method to a non-retryable indeterminate response", async () => { + const harness = createFakeChannelHarness(); + const broker = createDuplexBridgeBroker({ + channel: harness.channel, + forwardRequest: controllableForward(harness), + budgets: { forwardTimeoutMs: 5 }, + }); + brokers.push(broker); + broker.start(); + + // A POST may commit before the forward budget aborts the call. The broker + // must keep the terminal contract: a 504 with the indeterminate outcome, so + // the gateway maps it to a non-retryable 409 and no caller double-applies it. + harness.feed(requestFrame("mutate-timeout-1", "POST")); + const forward = harness.forwards.find((entry) => entry.id === "mutate-timeout-1" && !entry.settled); + if (!forward) throw new Error("The broker did not forward the request."); + + await new Promise((resolve) => setTimeout(resolve, 20)); + forward.settled = true; + forward.reject(new Error("Duplex broker forward budget exceeded.")); + await new Promise((resolve) => setTimeout(resolve, 0)); + + expect(harness.responses).toHaveLength(1); + const response = harness.responses[0]!; + expect(response.status).toBe(504); + expect(response.outcome).toBe("indeterminate"); + expect(response.headers["x-paperclip-bridge-outcome"]).toBe("indeterminate"); + expect(JSON.parse(response.body)).toEqual({ + error: "Duplex broker forward budget exceeded.", + outcome: "indeterminate", + retryable: false, + }); + }); + + it("maps a response-budget backstop for a safe method to a retryable response", async () => { + const harness = createFakeChannelHarness(); + const broker = createDuplexBridgeBroker({ + channel: harness.channel, + forwardRequest: controllableForward(harness), + // The forward budget aborts the call, but the forward promise stays + // pending. A GET receives the response headers, but the body reader stalls + // through the response budget, so the forward promise never settles. The + // response-budget backstop must answer the request. + budgets: { forwardTimeoutMs: 5, responseBudgetMs: 15, gatewayWaitMs: 40 }, + }); + brokers.push(broker); + broker.start(); + + // A safe method never changes host state, so a stalled body reader cannot + // leave a mutation half-applied. The backstop must keep the request + // retryable: a 504 with the completed outcome and no indeterminate marker, + // so the gateway does not map it to a terminal 409. + harness.feed(requestFrame("read-stall-1", "GET")); + const forward = harness.forwards.find((entry) => entry.id === "read-stall-1" && !entry.settled); + if (!forward) throw new Error("The broker did not forward the request."); + + // Never settle the forward. Wait past the response budget so the backstop + // fires on the stalled body reader. + await new Promise((resolve) => setTimeout(resolve, 30)); + + expect(harness.responses).toHaveLength(1); + const response = harness.responses[0]!; + expect(response.status).toBe(504); + expect(response.outcome).toBe("completed"); + expect(response.headers["x-paperclip-bridge-outcome"]).toBeUndefined(); + expect(JSON.parse(response.body)).toEqual({ + error: "Duplex broker response budget exceeded.", + retryable: true, + }); + }); + + it("maps a response-budget backstop for a mutating method to a non-retryable indeterminate response", async () => { + const harness = createFakeChannelHarness(); + const broker = createDuplexBridgeBroker({ + channel: harness.channel, + forwardRequest: controllableForward(harness), + budgets: { forwardTimeoutMs: 5, responseBudgetMs: 15, gatewayWaitMs: 40 }, + }); + brokers.push(broker); + broker.start(); + + // A POST may commit before the response budget passes, and the broker cannot + // prove the host applied no mutation. The backstop must keep the terminal + // contract: a 504 with the indeterminate outcome, so the gateway maps it to a + // non-retryable 409 and no caller double-applies it. + harness.feed(requestFrame("mutate-stall-1", "POST")); + const forward = harness.forwards.find((entry) => entry.id === "mutate-stall-1" && !entry.settled); + if (!forward) throw new Error("The broker did not forward the request."); + + await new Promise((resolve) => setTimeout(resolve, 30)); + + expect(harness.responses).toHaveLength(1); + const response = harness.responses[0]!; + expect(response.status).toBe(504); + expect(response.outcome).toBe("indeterminate"); + expect(response.headers["x-paperclip-bridge-outcome"]).toBe("indeterminate"); + expect(JSON.parse(response.body)).toEqual({ + error: "Duplex broker response budget exceeded.", + outcome: "indeterminate", + retryable: false, + }); + }); +}); diff --git a/packages/adapter-utils/src/duplex-bridge-broker.ts b/packages/adapter-utils/src/duplex-bridge-broker.ts new file mode 100644 index 0000000000..dbd1734d6b --- /dev/null +++ b/packages/adapter-utils/src/duplex-bridge-broker.ts @@ -0,0 +1,880 @@ +/** + * Host broker for the sandbox duplex channel. + * + * The broker owns the host end of one persistent duplex channel to the sandbox + * gateway. It reads request frames from the channel, forwards each one on the + * existing Paperclip API path, and writes one response frame back. It sends a + * heartbeat frame on an interval to prove liveness. + * + * The broker does not hold the route allowlist, the token replacement, or the + * run attribution. The caller passes a forward handler, and the broker calls it + * for each request. The handler applies the real token and the signed run + * identifier, so those rules stay in one place on the existing forward path. + * + * The broker runs a set of nested timeout budgets: + * - forward budget: the deadline for one forward call. + * - response budget: the deadline for the broker to send one response frame. + * - gateway wait budget: the deadline the in-sandbox gateway waits for the + * response frame. + * Each inner budget is smaller than its outer budget, so the broker aborts and + * answers before the gateway gives up. The broker asserts this order at + * construction and fails a configuration that breaks it. + * + * The provider controls the duplex transport directly, so the broker treats each + * request frame as untrusted. The broker bounds the work one channel can force: + * - in-flight limit: the maximum number of pending forwards at one time. It + * bounds the controllers, the timers, and the concurrent authenticated + * forwards. + * - lifetime limit: the maximum number of distinct requests over the channel + * lifetime. It bounds the retained request-id memory, because the broker keeps + * one id per distinct dispatched request for the no-replay guarantee. + * The broker checks each limit before it adds the id to the seen set, allocates + * the pending record, or calls the forward handler. On a limit it answers the + * refused request with one bounded terminal response and forwards nothing. The + * refusal preserves the no-replay and no-double-dispatch rules. + * + * Loss is terminal. The broker detects loss through channel exit, a stream + * write failure, a protocol failure, a heartbeat write failure, and a close + * timeout. On loss the broker stops the heartbeat, aborts every in-flight + * forward, records the loss, and dispatches nothing more. The broker never + * reconnects and never replays a request. The broker records the loss for + * metrics only and sends nothing about the loss to the sandbox. + */ + +import type { CommandManagedDuplexChannel } from "./command-managed-runtime.js"; +import { + DEFAULT_MAX_DUPLEX_REQUEST_ID_BYTES, + DUPLEX_FRAME_VERSION, + DuplexFrameDecoder, + encodeDuplexFrame, + type DuplexFrame, + type DuplexRequestFrame, + type DuplexResponseFrame, + type DuplexResponseOutcome, +} from "./duplex-frame-codec.js"; +import type { + DuplexLossReason, + DuplexOutcomeValue, + DuplexTelemetry, +} from "./duplex-telemetry.js"; + +/** The lifecycle states of the broker. The broker moves through them in order. */ +export type DuplexBrokerState = "opening" | "open" | "lost" | "closing" | "closed"; + +/** The reason the broker classified a loss. The broker records it for metrics only. */ +export type DuplexBrokerLossReason = + | "channel_exit" + | "stream_failure" + | "protocol_failure" + | "heartbeat_write_failure" + | "close_timeout"; + +/** + * Map one broker loss reason to the closed, typed telemetry loss reason. The + * broker records the internal reason for its own metrics; the telemetry boundary + * carries only the closed enum value. The map is total over the internal reasons, + * so no raw text ever reaches the typed reason. + * - `channel_exit` -> `provider_exit`: the provider channel process exited. + * - `stream_failure` -> `write_error`: a write to the channel failed. + * - `protocol_failure` -> `rpc_failure`: a malformed or mismatched frame. + * - `heartbeat_write_failure` -> `heartbeat_timeout`: the liveness write failed. + * - `close_timeout` -> `other`: an orderly close did not complete in the budget. + */ +const BROKER_LOSS_REASON_TO_TYPED: Readonly> = { + channel_exit: "provider_exit", + stream_failure: "write_error", + protocol_failure: "rpc_failure", + heartbeat_write_failure: "heartbeat_timeout", + close_timeout: "other", +}; + +/** + * Map one broker loss reason to the closed, typed telemetry loss reason. The host + * uses it to name the typed reason on a log line without the raw provider text. + */ +export function typedDuplexLossReason(reason: DuplexBrokerLossReason): DuplexLossReason { + return BROKER_LOSS_REASON_TO_TYPED[reason] ?? "other"; +} + +/** + * The terminal run disposition the broker computes from its ordered lifecycle. A + * `failed` disposition means a terminal loss ordered before an orderly completion, + * so the run must not report success. The typed loss reason names the cause; it is + * `null` for a success. + */ +export interface DuplexBrokerRunDisposition { + /** True when a terminal loss ordered before an orderly completion. */ + failed: boolean; + /** The typed, closed loss reason on a failure; `null` on a success. */ + lossReason: DuplexLossReason | null; +} + +/** The nested timeout budgets. Each inner budget is smaller than its outer budget. */ +export interface DuplexBrokerBudgets { + /** The deadline for one forward call, in milliseconds. */ + forwardTimeoutMs: number; + /** The deadline for the broker to send one response frame, in milliseconds. */ + responseBudgetMs: number; + /** The deadline the in-sandbox gateway waits for the response frame, in milliseconds. */ + gatewayWaitMs: number; +} + +/** The default nested budgets: forward 30 s, response 32 s, gateway wait 35 s. */ +export const DEFAULT_DUPLEX_BROKER_BUDGETS: DuplexBrokerBudgets = { + forwardTimeoutMs: 30_000, + responseBudgetMs: 32_000, + gatewayWaitMs: 35_000, +}; + +/** The default interval between two heartbeat frames, in milliseconds. */ +export const DEFAULT_DUPLEX_BROKER_HEARTBEAT_INTERVAL_MS = 5_000; + +/** The default deadline for an orderly channel close, in milliseconds. */ +export const DEFAULT_DUPLEX_BROKER_CLOSE_TIMEOUT_MS = 2_000; + +/** + * The default maximum number of in-flight requests. The broker holds this many + * pending forwards at one time. It refuses a further request until an in-flight + * request completes. The provider controls the transport, so this finite limit + * bounds the controllers, the timers, and the concurrent authenticated forwards + * a provider can force on the host. + */ +export const DEFAULT_DUPLEX_BROKER_MAX_IN_FLIGHT_REQUESTS = 64; + +/** + * The default maximum number of distinct requests over the channel lifetime. The + * broker forwards this many distinct request ids, then refuses each new distinct + * id and forwards nothing more. This limit bounds the retained request-id memory, + * because the broker keeps one id per distinct dispatched request for the no-replay + * guarantee. + * + * The retained id memory has a hard ceiling. The codec bounds each id at + * `DEFAULT_MAX_DUPLEX_REQUEST_ID_BYTES` (256 bytes), so the worst-case retained id + * bytes are this count multiplied by that bound: 50,000 * 256 = 12,800,000 bytes + * (about 12.8 MB), plus the fixed per-entry overhead of the Set. This count is + * sized against that id bound to keep the ceiling small. + */ +export const DEFAULT_DUPLEX_BROKER_MAX_LIFETIME_REQUESTS = 50_000; + +/** The result of one forward call. The broker turns it into one response frame. */ +export interface DuplexBrokerForwardResult { + status: number; + headers?: Record; + body?: string; +} + +/** + * The forward handler the broker calls for each request. The handler applies the + * real token and the run attribution, then forwards the request on the existing + * API path. The broker aborts `options.signal` when the forward budget ends or a + * loss happens, so a handler that threads the signal into its work stops early. + */ +export type DuplexBrokerForwardHandler = ( + request: DuplexRequestFrame, + options: { signal: AbortSignal }, +) => Promise; + +/** One request record. The broker captures the dispatch-start point for metrics only. */ +export interface DuplexBrokerRequestRecord { + id: string; + method: string; + path: string; + /** The point the broker started to dispatch the request, in milliseconds. */ + dispatchStartMs: number; +} + +/** One loss record. The broker reports it for metrics only. */ +export interface DuplexBrokerLossRecord { + reason: DuplexBrokerLossReason; + message: string; + /** The point the broker recorded the loss, in milliseconds. */ + atMs: number; +} + +/** The options for {@link createDuplexBridgeBroker}. */ +export interface DuplexBrokerOptions { + /** The duplex channel to the sandbox gateway. */ + channel: CommandManagedDuplexChannel; + /** The forward handler the broker calls for each request. */ + forwardRequest: DuplexBrokerForwardHandler; + /** The nested timeout budgets. The default is {@link DEFAULT_DUPLEX_BROKER_BUDGETS}. */ + budgets?: Partial; + /** The interval between two heartbeat frames, in milliseconds. */ + heartbeatIntervalMs?: number; + /** The deadline for an orderly channel close, in milliseconds. */ + closeTimeoutMs?: number; + /** The maximum size of one inbound frame, in bytes. Forwarded to the decoder. */ + maxFrameBytes?: number; + /** + * The maximum number of in-flight requests the broker holds at one time. The + * broker refuses a further request past this limit with a bounded terminal + * response and forwards nothing for it. The default is + * {@link DEFAULT_DUPLEX_BROKER_MAX_IN_FLIGHT_REQUESTS}. + */ + maxInFlightRequests?: number; + /** + * The maximum number of distinct requests the broker dispatches over the + * channel lifetime. The broker refuses each new distinct request past this + * limit with a bounded terminal response and forwards nothing more. This limit + * bounds the retained request-id memory. The default is + * {@link DEFAULT_DUPLEX_BROKER_MAX_LIFETIME_REQUESTS}. + */ + maxLifetimeRequests?: number; + /** The clock the broker reads for the metric timestamps. The default is `Date.now`. */ + now?: () => number; + /** The metrics sink for the per-request dispatch record. */ + onRequestRecord?: (record: DuplexBrokerRequestRecord) => void; + /** The metrics sink for the terminal loss record. */ + onLoss?: (record: DuplexBrokerLossRecord) => void; + /** The sink for a state change. The broker reports every transition. */ + onStateChange?: (state: DuplexBrokerState) => void; + /** The sink for a diagnostic message. The broker never writes diagnostics to the channel. */ + logger?: (message: string) => void; + /** + * The fixed observability facade. The broker records one request span per + * delivered request and one loss record per terminal loss. The facade maps each + * record to the fixed names and dimensions, so no route, query, body, token, or + * raw error rides a span or a counter. The default records nothing. + */ + telemetry?: DuplexTelemetry; +} + +/** The broker handle the factory returns. */ +export interface DuplexBridgeBroker { + /** The current state of the broker. */ + readonly state: DuplexBrokerState; + /** The loss record, or `null` when the broker never lost the channel. */ + readonly lossRecord: DuplexBrokerLossRecord | null; + /** + * The terminal run disposition from the ordered lifecycle. It reports a failure + * when a terminal loss ordered before an orderly completion, and names the typed + * loss reason. It reports a success for a healthy channel or a normal-teardown + * loss. The host reads it at the run-disposition seam. + */ + readonly runDisposition: DuplexBrokerRunDisposition; + /** + * Mark the host-observed orderly completion of the agent turn on the ordered + * lifecycle. A loss ordered before this mark latches a failure; a loss ordered + * after it stays a success. The broker also marks it on a gateway close frame + * and on a host-initiated orderly close. Safe to call more than one time. + */ + markOrderlyCompletion(): void; + /** Start the broker. It wires the channel listeners and moves to `open`. */ + start(): void; + /** Close the channel cleanly. It moves through `closing` to `closed`. */ + close(): Promise; + /** Stop the sandbox child process. Safe to call more than one time. */ + stop(): void; +} + +/** + * Assert the nested budget order. Each inner budget must be smaller than its + * outer budget, so the broker answers before the gateway gives up. The function + * throws when the order breaks. + */ +export function assertNestedDuplexBrokerBudgets(budgets: DuplexBrokerBudgets): void { + if (!(budgets.forwardTimeoutMs < budgets.responseBudgetMs)) { + throw new Error( + `Duplex broker forward budget ${budgets.forwardTimeoutMs}ms must be smaller than the response budget ${budgets.responseBudgetMs}ms.`, + ); + } + if (!(budgets.responseBudgetMs < budgets.gatewayWaitMs)) { + throw new Error( + `Duplex broker response budget ${budgets.responseBudgetMs}ms must be smaller than the gateway wait budget ${budgets.gatewayWaitMs}ms.`, + ); + } +} + +/** The resolved request limits the broker enforces. */ +export interface DuplexBrokerLimits { + /** The maximum number of in-flight requests the broker holds at one time. */ + maxInFlightRequests: number; + /** The maximum number of distinct requests the broker dispatches over the channel lifetime. */ + maxLifetimeRequests: number; +} + +/** + * Assert the request limits. Each limit must be a finite positive integer, so the + * broker fails closed on a broken configuration instead of running with an + * unbounded or a zero limit. The function throws when a limit breaks the rule. + */ +export function assertDuplexBrokerLimits(limits: DuplexBrokerLimits): void { + const entries: ReadonlyArray = [ + ["maxInFlightRequests", limits.maxInFlightRequests], + ["maxLifetimeRequests", limits.maxLifetimeRequests], + ]; + for (const [name, value] of entries) { + if (!Number.isInteger(value) || value < 1) { + throw new Error( + `Duplex broker ${name} must be a finite positive integer; got ${String(value)}.`, + ); + } + } +} + +function errorMessage(error: unknown): string { + return error instanceof Error ? error.message : String(error); +} + +/** + * The safe HTTP methods. RFC 7231 section 4.2.1 defines this set. A safe method + * does not change host state, so the host applies no mutation for it. A caller + * can retry a safe method after a forward failure without a double-apply risk. + */ +const SAFE_BRIDGE_METHODS = new Set(["GET", "HEAD", "OPTIONS", "TRACE"]); + +/** Report whether the method is safe, so a forward failure stays retryable. */ +export function isSafeBridgeMethod(method: string): boolean { + return SAFE_BRIDGE_METHODS.has(method.trim().toUpperCase()); +} + +/** The internal bookkeeping for one in-flight request. */ +interface PendingRequest { + controller: AbortController; + responded: boolean; + forwardTimer: ReturnType; + responseTimer: ReturnType; + /** The point the broker started to dispatch the request. It sets the span latency. */ + dispatchStartMs: number; +} + +/** + * Create the host duplex bridge broker. The factory asserts the budget order and + * returns a handle. Call `start` to wire the channel and open the broker. + */ +export function createDuplexBridgeBroker(options: DuplexBrokerOptions): DuplexBridgeBroker { + const budgets: DuplexBrokerBudgets = { + ...DEFAULT_DUPLEX_BROKER_BUDGETS, + ...options.budgets, + }; + assertNestedDuplexBrokerBudgets(budgets); + + const limits: DuplexBrokerLimits = { + maxInFlightRequests: options.maxInFlightRequests ?? DEFAULT_DUPLEX_BROKER_MAX_IN_FLIGHT_REQUESTS, + maxLifetimeRequests: options.maxLifetimeRequests ?? DEFAULT_DUPLEX_BROKER_MAX_LIFETIME_REQUESTS, + }; + assertDuplexBrokerLimits(limits); + + const channel = options.channel; + const forwardRequest = options.forwardRequest; + const heartbeatIntervalMs = options.heartbeatIntervalMs ?? DEFAULT_DUPLEX_BROKER_HEARTBEAT_INTERVAL_MS; + const closeTimeoutMs = options.closeTimeoutMs ?? DEFAULT_DUPLEX_BROKER_CLOSE_TIMEOUT_MS; + const now = options.now ?? (() => Date.now()); + const decoder = new DuplexFrameDecoder( + options.maxFrameBytes ? { maxFrameBytes: options.maxFrameBytes } : undefined, + ); + + let state: DuplexBrokerState = "opening"; + let stopped = false; + let started = false; + let closePromise: Promise | null = null; + let lossRecord: DuplexBrokerLossRecord | null = null; + let heartbeatTimer: ReturnType | null = null; + + // The host-owned lifecycle sequence. The broker assigns each terminal lifecycle + // event a strictly increasing sequence number at ingress, before any + // asynchronous logging. The order of these numbers, not a wall-clock or a + // provider timestamp, decides the run disposition. + let lifecycleSeq = 0; + // The sequence number of the terminal loss, or `null` when no loss ordered. The + // broker sets it one time. A later event never clears it, so the loss latches. + let lossSeq: number | null = null; + // The typed, closed loss reason for the latched loss, or `null` on a success. + let typedLossReason: DuplexLossReason | null = null; + // The sequence number of the host-observed orderly completion, or `null` when + // none ordered. The broker sets it one time, only while the channel is healthy. + let orderlyCompletionSeq: number | null = null; + const nextLifecycleSeq = (): number => { + lifecycleSeq += 1; + return lifecycleSeq; + }; + + // Mark the host-observed orderly completion of the agent turn. The broker sets + // the sequence one time, and only while no loss has ordered. A loss that already + // latched keeps the failure, so a late completion never clears the latch. + const markOrderlyCompletion = (): void => { + if (orderlyCompletionSeq !== null || lossSeq !== null) return; + orderlyCompletionSeq = nextLifecycleSeq(); + }; + // The ids the broker already dispatched. The broker forwards one id one time, + // so a repeated frame never reaches the API twice. + const seenRequestIds = new Set(); + const pending = new Map(); + + const setState = (next: DuplexBrokerState): void => { + if (state === next) return; + state = next; + options.onStateChange?.(next); + }; + + const clearHeartbeat = (): void => { + if (heartbeatTimer !== null) { + clearInterval(heartbeatTimer); + heartbeatTimer = null; + } + }; + + const clearPending = (): void => { + for (const entry of pending.values()) { + clearTimeout(entry.forwardTimer); + clearTimeout(entry.responseTimer); + entry.controller.abort(new Error("Duplex broker stopped.")); + } + pending.clear(); + }; + + const recordLoss = (reason: DuplexBrokerLossReason, message: string): void => { + // Loss is terminal. Record it one time and stop every activity. + if (stopped) return; + const afterOrderlyCompletion = orderlyCompletionSeq !== null; + // A clean channel end that orders after a host-observed orderly completion is + // a normal teardown, not a loss. Stop cleanly, emit no loss event, and leave + // the run a success. This keeps the closed telemetry contract: an orderly + // close is not a loss and emits no loss event. + if (afterOrderlyCompletion && reason === "channel_exit") { + stopped = true; + clearHeartbeat(); + clearPending(); + if (state !== "closing") setState("closed"); + return; + } + stopped = true; + // Classify the loss relative to the first dispatch. A loss after the broker + // dispatched a request is `post_dispatch`; a loss before any dispatch is + // `pre_dispatch`. The class rides the fixed loss counter, never the raw + // message. + const lossClass = seenRequestIds.size > 0 ? "post_dispatch" : "pre_dispatch"; + // Assign the loss its lifecycle sequence at ingress, before any logging. The + // loss latches the run as a failure only when no orderly completion ordered + // before it. A loss ordered after an orderly completion (for example a failed + // close) is a real channel loss for the telemetry and the leak metric, but it + // does not fail the run, because the run already completed. Once set, `lossSeq` + // never clears, so a later completion, exit, or activity callback cannot flip + // the latch. + const seq = nextLifecycleSeq(); + if (!afterOrderlyCompletion) { + lossSeq = seq; + typedLossReason = BROKER_LOSS_REASON_TO_TYPED[reason] ?? "other"; + } + clearHeartbeat(); + clearPending(); + lossRecord = { reason, message, atMs: now() }; + setState("lost"); + // Log the internal reason only. The broker never writes the raw provider + // message to a log line, so no raw provider text rides a sink here. + options.logger?.(`Duplex broker lost the channel (${reason}).`); + options.onLoss?.(lossRecord); + options.telemetry?.recordLoss(lossClass, BROKER_LOSS_REASON_TO_TYPED[reason] ?? "other"); + }; + + const writeFrame = (frame: DuplexFrame): boolean => { + try { + channel.write(encodeDuplexFrame(frame)); + return true; + } catch (error) { + recordLoss("stream_failure", errorMessage(error)); + return false; + } + }; + + const respond = ( + id: string, + result: DuplexBrokerForwardResult, + outcome: DuplexResponseOutcome, + telemetryOutcome: DuplexOutcomeValue, + ): void => { + const entry = pending.get(id); + if (!entry || entry.responded) return; + entry.responded = true; + clearTimeout(entry.forwardTimer); + clearTimeout(entry.responseTimer); + pending.delete(id); + // Do not write on a lost or closed channel. The gateway answers its own + // outstanding request on loss, so a late write would go to a dead channel. + if (state !== "open") return; + // Record the request span for the delivered request. The span carries the + // latency and the outcome only; no route, query, body, or token rides it. + options.telemetry?.recordRequest({ + latencyMs: now() - entry.dispatchStartMs, + outcome: telemetryOutcome, + }); + const frame: DuplexResponseFrame = { + version: DUPLEX_FRAME_VERSION, + type: "response", + id, + status: result.status, + headers: result.headers ?? {}, + body: result.body ?? "", + outcome, + }; + writeFrame(frame); + }; + + const respondSaturated = (id: string, retryable: boolean): void => { + // Answer a refused request with a bounded terminal response. The broker made + // no controller, no timer, and no forward for this id, so the host API stays + // untouched. The response carries no route, no query, no body, and no token; + // it holds only the fixed error shape. The `unavailable` outcome tells the + // gateway this is not a delivered host response, so it never counts as one. + if (state !== "open") return; + const frame: DuplexResponseFrame = { + version: DUPLEX_FRAME_VERSION, + type: "response", + id, + status: 503, + headers: { + "content-type": "application/json", + "x-paperclip-bridge-outcome": "unavailable", + }, + body: JSON.stringify({ + error: "Duplex broker capacity limit reached.", + outcome: "unavailable", + retryable, + }), + outcome: "unavailable", + }; + writeFrame(frame); + }; + + const dispatch = (frame: DuplexRequestFrame): void => { + // Dispatch only while open. After loss or close the broker forwards nothing. + if (state !== "open") return; + // Bound the id byte size before any retention or work. The codec already + // rejects an over-limit id on the read path, so this guard is defense in depth + // for a frame that reaches dispatch by another path. The broker never adds the + // id to the seen set, never allocates a controller or a timer, and never + // forwards. It answers with the bounded terminal refusal, which carries no + // route, query, body, or token, and records no telemetry. The refusal is not + // retryable, because a resend of the same over-limit id never gets past this + // bound. + if (Buffer.byteLength(frame.id, "utf8") > DEFAULT_MAX_DUPLEX_REQUEST_ID_BYTES) { + respondSaturated(frame.id, false); + return; + } + // Forward one id one time. A repeated id never reaches the API twice. + if (seenRequestIds.has(frame.id)) return; + // Bound the retained request-id memory. The broker keeps one id per distinct + // dispatched request for the no-replay guarantee, so the set can only grow. + // Once the broker reaches the lifetime limit, it refuses each new distinct id + // and forwards nothing more. The refusal is not retryable, because a resend + // never gets past the limit. This check runs before the id joins the set, so + // the set never grows past the limit. + if (seenRequestIds.size >= limits.maxLifetimeRequests) { + respondSaturated(frame.id, false); + return; + } + // Bound the in-flight request count. Once the broker holds the maximum number + // of pending forwards, it refuses a further request and forwards nothing for + // it. The broker does not add the id to the seen set, so the gateway can + // resend the request after an in-flight request completes. The refusal is + // retryable for that reason. This check bounds the controllers, the timers, + // and the concurrent forwards a provider can force. + if (pending.size >= limits.maxInFlightRequests) { + respondSaturated(frame.id, true); + return; + } + seenRequestIds.add(frame.id); + + const record: DuplexBrokerRequestRecord = { + id: frame.id, + method: frame.method, + path: frame.path, + dispatchStartMs: now(), + }; + options.onRequestRecord?.(record); + + const controller = new AbortController(); + const forwardTimer = setTimeout(() => { + controller.abort(new Error("Duplex broker forward budget exceeded.")); + }, budgets.forwardTimeoutMs); + const responseTimer = setTimeout(() => { + // Response-budget backstop. The forward rejection normally answers first, + // well before this deadline. This backstop answers a request whose forward + // rejection handling itself stalls, so the request never strands. One stall + // path is a response whose headers arrive but whose body reader stays + // pending through the budget, so the forward promise never settles. + if (isSafeBridgeMethod(frame.method)) { + // A safe method never changes host state, so a stalled body reader + // cannot leave a mutation half-applied. Keep the request retryable: + // return a 504 with the completed outcome and no indeterminate marker, + // so the gateway passes it through as a retryable status and never maps + // it to a terminal 409. + respond( + frame.id, + { + status: 504, + headers: { "content-type": "application/json" }, + body: JSON.stringify({ + error: "Duplex broker response budget exceeded.", + retryable: true, + }), + }, + "completed", + "error", + ); + return; + } + // The method may mutate host state, and the broker cannot prove the host + // applied no mutation once the response budget passes. Return a + // non-retryable 504 and mark the outcome indeterminate, so the gateway + // maps it to a terminal 409 and no caller double-applies the mutation. + respond( + frame.id, + { + status: 504, + headers: { + "content-type": "application/json", + "x-paperclip-bridge-outcome": "indeterminate", + }, + body: JSON.stringify({ + error: "Duplex broker response budget exceeded.", + outcome: "indeterminate", + retryable: false, + }), + }, + "indeterminate", + "error", + ); + }, budgets.responseBudgetMs); + forwardTimer.unref?.(); + responseTimer.unref?.(); + pending.set(frame.id, { + controller, + responded: false, + forwardTimer, + responseTimer, + dispatchStartMs: record.dispatchStartMs, + }); + + forwardRequest(frame, { signal: controller.signal }).then( + (result) => { + // Keep the outcome classification consistent with the file path. A + // possibly-committed mutation carries the indeterminate marker header, so + // map it to the indeterminate outcome. Any other result is completed. + const outcome: DuplexResponseOutcome = + result.headers?.["x-paperclip-bridge-outcome"] === "indeterminate" + ? "indeterminate" + : "completed"; + // The host delivered a real response, so the request span outcome is `ok`. + // A host application status (200, a 4xx, a 5xx) is still a delivered + // response; only a broker-synthesized failure below is `error`. + respond(frame.id, result, outcome, "ok"); + }, + (error) => { + if (controller.signal.aborted) { + // The forward budget aborted the call. A safe method never changes + // host state, so a forward timeout stays retryable for it. Return a + // 504 with the completed outcome and no indeterminate marker, so the + // gateway passes it through as a retryable status. + if (isSafeBridgeMethod(frame.method)) { + respond( + frame.id, + { + status: 504, + headers: { "content-type": "application/json" }, + body: JSON.stringify({ + error: errorMessage(error), + retryable: true, + }), + }, + "completed", + "error", + ); + return; + } + // The method may mutate host state, and the forward budget aborted the + // call after the request may have committed. Return a non-retryable 504 + // and mark the outcome indeterminate, so a caller does not retry a + // committed mutation. + respond( + frame.id, + { + status: 504, + headers: { + "content-type": "application/json", + "x-paperclip-bridge-outcome": "indeterminate", + }, + body: JSON.stringify({ + error: errorMessage(error), + outcome: "indeterminate", + retryable: false, + }), + }, + "indeterminate", + "error", + ); + return; + } + // The forward rejected before the host delivered a response. This + // rejection does not prove that the host applied no mutation. A fetch + // can reject after the request bytes reach the host and the host + // commits, but before the response headers arrive. A safe method never + // changes host state, so a retry stays safe for it. For any other + // method the host may have committed, so the outcome is indeterminate. + if (isSafeBridgeMethod(frame.method)) { + // The method is safe, so a retry cannot double-apply a mutation. + // Return a 502 with the completed outcome, so the gateway passes it + // through as a retryable status. + respond( + frame.id, + { + status: 502, + headers: { "content-type": "application/json" }, + body: JSON.stringify({ error: errorMessage(error) }), + }, + "completed", + "error", + ); + return; + } + // The method may mutate host state, so a retry with a new request id + // could apply the mutation twice. Return a non-retryable 504 and mark + // the outcome indeterminate, so the gateway maps it to a terminal 409. + respond( + frame.id, + { + status: 504, + headers: { + "content-type": "application/json", + "x-paperclip-bridge-outcome": "indeterminate", + }, + body: JSON.stringify({ + error: errorMessage(error), + outcome: "indeterminate", + retryable: false, + }), + }, + "indeterminate", + "error", + ); + }, + ); + }; + + const handleFrame = (frame: DuplexFrame): void => { + switch (frame.type) { + case "request": + dispatch(frame); + return; + case "close": + // The gateway asked for an orderly close. The gateway sends this frame on + // the agent's orderly completion, so mark the completion on the ordered + // lifecycle before the close, then close the channel. + markOrderlyCompletion(); + void close(); + return; + case "ready": + case "heartbeat": + // Liveness frames. The broker reads them and dispatches nothing. + return; + case "response": + case "error": + // The host never expects these on the read path. Ignore them. + return; + default: + return; + } + }; + + const onData = (chunk: string): void => { + if (stopped) return; + const results = decoder.push(chunk); + for (const result of results) { + if (stopped) return; + if (!result.ok) { + recordLoss("protocol_failure", result.error.message); + return; + } + handleFrame(result.frame); + } + }; + + const onExit = (): void => { + recordLoss("channel_exit", "The sandbox channel process exited."); + }; + + const sendHeartbeat = (): void => { + if (state !== "open") return; + try { + channel.write(encodeDuplexFrame({ version: DUPLEX_FRAME_VERSION, type: "heartbeat" })); + } catch (error) { + recordLoss("heartbeat_write_failure", errorMessage(error)); + } + }; + + const close = (): Promise => { + if (closePromise) return closePromise; + if (stopped) return Promise.resolve(); + // A host-initiated orderly close is a host-observed orderly completion. The + // host tears the channel down on its own terms, so a channel end during the + // close is a normal teardown, not a mid-run loss. `markOrderlyCompletion` + // no-ops when a loss already latched, so a lost channel stays a failure. + markOrderlyCompletion(); + closePromise = (async () => { + setState("closing"); + clearHeartbeat(); + clearPending(); + // Send an orderly close frame. Ignore a write failure here; the broker is + // already closing, so a dead channel needs no loss record. + try { + channel.write(encodeDuplexFrame({ version: DUPLEX_FRAME_VERSION, type: "close" })); + } catch (error) { + options.logger?.(`Duplex broker could not send the close frame: ${errorMessage(error)}`); + } + let closeTimer: ReturnType | undefined; + const timeout = new Promise((_resolve, reject) => { + closeTimer = setTimeout(() => { + reject(new Error("Duplex broker channel close timed out.")); + }, closeTimeoutMs); + closeTimer.unref?.(); + }); + try { + await Promise.race([channel.close(), timeout]); + if (closeTimer !== undefined) clearTimeout(closeTimer); + stopped = true; + setState("closed"); + } catch (error) { + if (closeTimer !== undefined) clearTimeout(closeTimer); + recordLoss("close_timeout", errorMessage(error)); + } + })(); + return closePromise; + }; + + const start = (): void => { + if (started) return; + started = true; + channel.onData(onData); + channel.onExit(onExit); + heartbeatTimer = setInterval(sendHeartbeat, heartbeatIntervalMs); + heartbeatTimer.unref?.(); + setState("open"); + }; + + const stop = (): void => { + try { + channel.stop(); + } catch (error) { + options.logger?.(`Duplex broker could not stop the channel: ${errorMessage(error)}`); + } + }; + + return { + get state() { + return state; + }, + get lossRecord() { + return lossRecord; + }, + get runDisposition(): DuplexBrokerRunDisposition { + // A latched loss ordered before any orderly completion is a failure. Every + // other state — a healthy channel, or a loss ordered after an orderly + // completion — is a success. + return { failed: lossSeq !== null, lossReason: typedLossReason }; + }, + markOrderlyCompletion, + start, + close, + stop, + }; +} diff --git a/packages/adapter-utils/src/duplex-bridge-local-e2e.test.ts b/packages/adapter-utils/src/duplex-bridge-local-e2e.test.ts new file mode 100644 index 0000000000..a4e2942c97 --- /dev/null +++ b/packages/adapter-utils/src/duplex-bridge-local-e2e.test.ts @@ -0,0 +1,646 @@ +import { spawn, type ChildProcessWithoutNullStreams } from "node:child_process"; +import { createServer, type IncomingMessage, type ServerResponse } from "node:http"; +import net from "node:net"; +import { mkdtemp, rm, writeFile } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { StringDecoder } from "node:string_decoder"; +import { afterEach, describe, expect, it } from "vitest"; + +import { + authorizeSandboxCallbackBridgeRequestWithRoutes, + getSandboxCallbackBridgeServerSource, + SANDBOX_CALLBACK_BRIDGE_DUPLEX_MODE, +} from "./sandbox-callback-bridge.js"; +import { + DuplexFrameDecoder, + type DuplexFrame, + type DuplexReadyFrame, + type DuplexRequestFrame, +} from "./duplex-frame-codec.js"; +import { + createDuplexBridgeBroker, + type DuplexBridgeBroker, + type DuplexBrokerForwardResult, +} from "./duplex-bridge-broker.js"; +import type { CommandManagedDuplexChannel } from "./command-managed-runtime.js"; + +/** + * Local real-process end-to-end harness for the composed duplex path. + * + * The harness spawns the real generated gateway with plain `node` from a + * temporary file, attaches the real host broker to the child stdin and stdout, + * and forwards each request to a local fake API server on a real HTTP call. It + * uses no provider credentials. + * + * The child stdio pipes are kernel pipes. They give partial reads, split a + * multi-byte UTF-8 sequence across two chunks, apply backpressure, and give a + * true end of file when the child dies. The harness proves the composed path + * handles each of these real conditions. + */ + +/** One recorded call the fake API server received. */ +interface FakeApiRequest { + method: string; + url: string; + auth: string | null; + runId: string | null; + body: string; +} + +/** The responder the fake API server calls for each request. */ +type FakeApiResponder = (req: IncomingMessage, res: ServerResponse, body: string) => void; + +/** The handle for one fake API server. */ +interface FakeApiServer { + origin: string; + requests: FakeApiRequest[]; + setResponder: (responder: FakeApiResponder) => void; + /** The count of sockets the server holds open right now. */ + openSocketCount: () => number; + /** True while the server still accepts connections. */ + listening: () => boolean; + close: () => Promise; +} + +/** + * Start a local fake API server. The forward handler targets it, so one real + * HTTP call proves the composed path end to end. The server tracks each open + * socket, so a test asserts teardown leaves no open socket handle. + */ +async function startFakeApiServer(): Promise { + const requests: FakeApiRequest[] = []; + const sockets = new Set(); + let responder: FakeApiResponder = (req, res) => { + const url = new URL(req.url ?? "/", "http://127.0.0.1"); + res.writeHead(200, { "content-type": "application/json" }); + res.end(JSON.stringify({ ok: true, path: url.pathname })); + }; + + const server = createServer((req, res) => { + const chunks: Buffer[] = []; + req.on("data", (chunk: Buffer) => chunks.push(chunk)); + req.on("end", () => { + const body = Buffer.concat(chunks).toString("utf8"); + requests.push({ + method: req.method ?? "GET", + url: req.url ?? "/", + auth: req.headers.authorization ?? null, + runId: + typeof req.headers["x-paperclip-run-id"] === "string" + ? req.headers["x-paperclip-run-id"] + : null, + body, + }); + responder(req, res, body); + }); + }); + server.on("connection", (socket) => { + sockets.add(socket); + socket.on("close", () => sockets.delete(socket)); + }); + + await new Promise((resolve, reject) => { + server.once("error", reject); + server.listen(0, "127.0.0.1", () => resolve()); + }); + const address = server.address(); + if (!address || typeof address === "string") { + throw new Error("The fake API server did not expose a TCP port."); + } + + return { + origin: `http://127.0.0.1:${address.port}`, + requests, + setResponder: (next) => { + responder = next; + }, + openSocketCount: () => sockets.size, + listening: () => server.listening, + close: () => + new Promise((resolve) => { + // Destroy each open socket first. A keep-alive client socket keeps the + // server open, so `server.close` alone could stall. The destroy makes the + // close callback fire and drives the open socket count to zero. + for (const socket of sockets) socket.destroy(); + server.close(() => resolve()); + }), + }; +} + +/** The channel view over the spawned child, plus the frames the child sent host-ward. */ +interface ChildDuplexChannel { + channel: CommandManagedDuplexChannel; + observedFrames: DuplexFrame[]; + stderr: () => string; +} + +/** + * Wrap the spawned child as a {@link CommandManagedDuplexChannel}. The broker + * writes to the child stdin, reads the child stdout, and learns of the child + * exit through this channel. + * + * The host end reads a byte stream from the stdout pipe. A pipe read can split + * one multi-byte UTF-8 character across two chunks. The `StringDecoder` holds + * the bytes of an incomplete character until the next chunk, so the broker only + * ever reads whole characters. The channel keeps a second decoder for + * observation only; it lets the harness assert the READY frame and the request + * frames the child produced, and it never feeds the broker. + */ +function attachChildDuplexChannel(child: ChildProcessWithoutNullStreams): ChildDuplexChannel { + const observed = new DuplexFrameDecoder(); + const observedFrames: DuplexFrame[] = []; + const stdoutDecoder = new StringDecoder("utf8"); + let dataListener: ((chunk: string) => void) | null = null; + let exitListener: ((exit: { exitCode: number | null }) => void) | null = null; + let pendingText = ""; + let pendingExit: { exitCode: number | null } | null = null; + let stderrText = ""; + + child.stdout.on("data", (buffer: Buffer) => { + for (const result of observed.push(buffer)) { + if (result.ok) observedFrames.push(result.frame); + } + const text = stdoutDecoder.write(buffer); + if (text.length === 0) return; + if (dataListener) dataListener(text); + else pendingText += text; + }); + child.stderr.on("data", (buffer: Buffer) => { + stderrText += buffer.toString("utf8"); + }); + child.on("exit", (code) => { + const exit = { exitCode: code }; + if (exitListener) exitListener(exit); + else pendingExit = exit; + }); + // Swallow a stdin EPIPE. The broker can write one more frame while the child + // exits; the write fails and the broker records the loss on its own path. + child.stdin.on("error", () => undefined); + + const channel: CommandManagedDuplexChannel = { + write: (data) => { + child.stdin.write(data); + }, + onData: (listener) => { + dataListener = listener; + if (pendingText.length > 0) { + const replay = pendingText; + pendingText = ""; + listener(replay); + } + }, + onExit: (listener) => { + exitListener = listener; + if (pendingExit) { + const exit = pendingExit; + pendingExit = null; + listener(exit); + } + }, + stop: () => { + child.kill("SIGKILL"); + }, + close: () => + new Promise((resolve) => { + // Close the write side. The child stdin reaches end of file, so the + // gateway sees a real EOF. + child.stdin.end(() => resolve()); + }), + }; + + return { channel, observedFrames, stderr: () => stderrText }; +} + +/** The forward-handler mode. `proxy` calls the fake API; `hang` blocks until abort. */ +type ForwardMode = "proxy" | "hang"; + +/** The options for one harness. */ +interface HarnessOptions { + hostApiToken?: string; + runId?: string; + maxBodyBytes?: number; + lossExitGraceMs?: number; +} + +/** The full harness handle for one composed duplex path. */ +interface DuplexE2EHarness { + baseUrl: string; + bridgeToken: string; + hostApiToken: string; + runId: string; + nonce: string; + broker: DuplexBridgeBroker; + api: FakeApiServer; + child: ChildProcessWithoutNullStreams; + observedFrames: DuplexFrame[]; + forwardedRequests: DuplexRequestFrame[]; + setForwardMode: (mode: ForwardMode) => void; + stderr: () => string; + killChild: () => Promise; + waitFor: (predicate: () => boolean, message: string, timeoutMs?: number) => Promise; + teardown: () => Promise; +} + +/** Reserve one free loopback port. The host assigns it to the gateway. */ +async function reserveLoopbackPort(): Promise { + return new Promise((resolve, reject) => { + const probe = net.createServer(); + probe.once("error", reject); + probe.listen(0, "127.0.0.1", () => { + const address = probe.address(); + if (!address || typeof address === "string") { + probe.close(() => reject(new Error("Could not reserve a loopback port."))); + return; + } + const reserved = address.port; + probe.close(() => resolve(reserved)); + }); + }); +} + +/** + * Build the composed duplex path: a spawned gateway child, the real broker on + * the child stdio, and a local fake API server as the forward target. + */ +async function createHarness(options: HarnessOptions = {}): Promise { + const hostApiToken = options.hostApiToken ?? "real-run-jwt"; + const runId = options.runId ?? "run-e2e"; + const maxBodyBytes = options.maxBodyBytes ?? 1_000_000; + const bridgeToken = "duplex-e2e-bridge-token"; + const nonce = "e2e112233445566778899aabbccddeeff"; + let forwardMode: ForwardMode = "proxy"; + const forwardedRequests: DuplexRequestFrame[] = []; + + const api = await startFakeApiServer(); + const assignedPort = await reserveLoopbackPort(); + + const tmpDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-e2e-")); + const entrypoint = path.join(tmpDir, "gateway.mjs"); + await writeFile(entrypoint, getSandboxCallbackBridgeServerSource(), "utf8"); + + const child = spawn(process.execPath, [entrypoint], { + stdio: ["pipe", "pipe", "pipe"], + env: { + ...process.env, + PAPERCLIP_API_BRIDGE_MODE: SANDBOX_CALLBACK_BRIDGE_DUPLEX_MODE, + PAPERCLIP_BRIDGE_HOST: "127.0.0.1", + PAPERCLIP_BRIDGE_PORT: String(assignedPort), + PAPERCLIP_BRIDGE_NONCE: nonce, + PAPERCLIP_BRIDGE_TOKEN: bridgeToken, + PAPERCLIP_BRIDGE_MAX_BODY_BYTES: String(maxBodyBytes), + ...(options.lossExitGraceMs != null + ? { PAPERCLIP_BRIDGE_LOSS_EXIT_GRACE_MS: String(options.lossExitGraceMs) } + : {}), + }, + }) as ChildProcessWithoutNullStreams; + + const { channel, observedFrames, stderr } = attachChildDuplexChannel(child); + + const forwardRequest = async ( + request: DuplexRequestFrame, + opts: { signal: AbortSignal }, + ): Promise => { + forwardedRequests.push(request); + // Keep the real route allowlist on the forward seam. The broker forwards + // only an allowed route; it denies any other route with a 403. + const denial = authorizeSandboxCallbackBridgeRequestWithRoutes(request); + if (denial) { + return { + status: 403, + headers: { "content-type": "application/json" }, + body: JSON.stringify({ error: denial }), + }; + } + if (forwardMode === "hang") { + // Block until the broker aborts the forward. This lets a test hold an + // outstanding request open while it forces a loss. + await new Promise((_resolve, reject) => { + const onAbort = () => reject(new Error("The forward call was aborted.")); + if (opts.signal.aborted) { + onAbort(); + return; + } + opts.signal.addEventListener("abort", onAbort, { once: true }); + }); + } + const method = request.method.trim().toUpperCase() || "GET"; + const headers = new Headers(); + for (const [key, value] of Object.entries(request.headers)) { + if (value.trim().length === 0) continue; + headers.set(key, value); + } + // Apply the real host token and the run id, the same as the file bridge path. + headers.set("authorization", `Bearer ${hostApiToken}`); + headers.set("x-paperclip-run-id", runId); + const target = new URL(`${request.path}${request.query ?? ""}`, api.origin); + const response = await fetch(target, { + method, + headers, + ...(method === "GET" || method === "HEAD" ? {} : { body: request.body }), + signal: opts.signal, + }); + const body = await response.text(); + const outHeaders: Record = {}; + response.headers.forEach((value, key) => { + if (key.toLowerCase() === "content-length") return; + outHeaders[key] = value; + }); + return { status: response.status, headers: outHeaders, body }; + }; + + const broker = createDuplexBridgeBroker({ + channel, + forwardRequest, + logger: () => undefined, + }); + broker.start(); + + const waitFor = async ( + predicate: () => boolean, + message: string, + timeoutMs = 5000, + ): Promise => { + const deadline = Date.now() + timeoutMs; + for (;;) { + if (predicate()) return; + if (Date.now() > deadline) { + throw new Error(`Timed out waiting for ${message}. stderr: ${stderr()}`); + } + await new Promise((resolve) => { + const timer = setTimeout(resolve, 20); + timer.unref?.(); + }); + } + }; + + const waitForExit = (): Promise => + new Promise((resolve) => { + if (child.exitCode !== null || child.signalCode !== null) { + resolve(); + return; + } + child.once("exit", () => resolve()); + }); + + const killChild = async (): Promise => { + child.kill("SIGKILL"); + await waitForExit(); + }; + + let torndown = false; + const teardown = async (): Promise => { + if (torndown) return; + torndown = true; + child.kill("SIGKILL"); + await waitForExit(); + child.stdin.destroy(); + child.stdout.destroy(); + child.stderr.destroy(); + await api.close(); + await rm(tmpDir, { recursive: true, force: true }); + }; + + return { + baseUrl: `http://127.0.0.1:${assignedPort}`, + bridgeToken, + hostApiToken, + runId, + nonce, + broker, + api, + child, + observedFrames, + forwardedRequests, + setForwardMode: (mode) => { + forwardMode = mode; + }, + stderr, + killChild, + waitFor, + teardown, + }; +} + +/** + * Build a large body of multi-byte UTF-8 characters. "€" uses three UTF-8 bytes + * and "😀" uses four. A pipe read boundary at a 65536-byte multiple lands inside + * a "€" sequence, because 65536 is not a multiple of three. This guarantees a + * multi-byte character split across two pipe chunks. + */ +function buildMultiByteBody(targetBytes: number): string { + const marker = "😀-start-😀"; + const filler = "€".repeat(Math.ceil(targetBytes / 3)); + return `${marker}${filler}${marker}`; +} + +describe("duplex bridge local end-to-end harness", () => { + const harnesses: DuplexE2EHarness[] = []; + + afterEach(async () => { + while (harnesses.length > 0) { + const harness = harnesses.pop(); + if (harness) await harness.teardown(); + } + }); + + const readyPredicate = (harness: DuplexE2EHarness) => () => + harness.observedFrames.some((frame) => frame.type === "ready"); + + it("delivers a valid READY frame from the spawned gateway child to the attached broker", async () => { + const harness = await createHarness(); + harnesses.push(harness); + + await harness.waitFor(readyPredicate(harness), "a READY frame from the gateway child"); + + const ready = harness.observedFrames.find( + (frame) => frame.type === "ready", + ) as DuplexReadyFrame; + expect(ready.version).toBe(1); + expect(ready.nonce).toBe(harness.nonce); + // READY is liveness only. It carries no address data. + expect((ready as unknown as Record).address).toBeUndefined(); + expect((ready as unknown as Record).port).toBeUndefined(); + // The broker read the READY frame and stayed open with no loss. + expect(harness.broker.state).toBe("open"); + expect(harness.broker.lossRecord).toBeNull(); + }, 20000); + + it("carries one real HTTP request through the child and the broker to the fake API unchanged", async () => { + const harness = await createHarness(); + harnesses.push(harness); + await harness.waitFor(readyPredicate(harness), "the gateway to become ready"); + + harness.api.setResponder((req, res) => { + const url = new URL(req.url ?? "/", "http://127.0.0.1"); + res.writeHead(200, { "content-type": "application/json" }); + res.end(JSON.stringify({ ok: true, echoedPath: url.pathname })); + }); + + const response = await fetch(`${harness.baseUrl}/api/agents/me`, { + headers: { authorization: `Bearer ${harness.bridgeToken}` }, + }); + expect(response.status).toBe(200); + await expect(response.json()).resolves.toEqual({ ok: true, echoedPath: "/api/agents/me" }); + + // The fake API saw one request with the real host token and the run id. The + // broker replaced the bridge token, so the fake API never saw it. + expect(harness.api.requests).toHaveLength(1); + expect(harness.api.requests[0]).toMatchObject({ + method: "GET", + url: "/api/agents/me", + auth: `Bearer ${harness.hostApiToken}`, + runId: harness.runId, + }); + }, 20000); + + it("reassembles a large response that spans many pipe chunks", async () => { + const harness = await createHarness(); + harnesses.push(harness); + await harness.waitFor(readyPredicate(harness), "the gateway to become ready"); + + // The body is larger than the pipe buffer, so the response frame crosses the + // child stdin pipe in many chunks. An ASCII body isolates the chunk-span + // behavior from the multi-byte behavior the next test covers. + const largeBody = "x".repeat(256 * 1024); + expect(Buffer.byteLength(largeBody, "utf8")).toBeGreaterThan(200 * 1024); + harness.api.setResponder((_req, res) => { + res.writeHead(200, { "content-type": "text/plain; charset=utf-8" }); + res.end(largeBody); + }); + + const response = await fetch(`${harness.baseUrl}/api/agents/me`, { + headers: { authorization: `Bearer ${harness.bridgeToken}` }, + }); + expect(response.status).toBe(200); + const received = await response.text(); + // The whole body returns complete after it crossed the pipe in many chunks. + expect(received.length).toBe(largeBody.length); + expect(received).toBe(largeBody); + }, 20000); + + it("decodes a multi-byte UTF-8 sequence split across chunk borders", async () => { + const harness = await createHarness(); + harnesses.push(harness); + await harness.waitFor(readyPredicate(harness), "the gateway to become ready"); + + const multiByteBody = buildMultiByteBody(256 * 1024); + expect(Buffer.byteLength(multiByteBody, "utf8")).toBeGreaterThan(200 * 1024); + harness.api.setResponder((_req, res) => { + res.writeHead(200, { "content-type": "text/plain; charset=utf-8" }); + res.end(multiByteBody); + }); + + const response = await fetch(`${harness.baseUrl}/api/agents/me`, { + headers: { authorization: `Bearer ${harness.bridgeToken}` }, + }); + expect(response.status).toBe(200); + const received = await response.text(); + // A multi-byte character split across a pipe chunk border decodes to the + // same character, so the whole body returns unchanged. + expect(received.length).toBe(multiByteBody.length); + expect(received).toBe(multiByteBody); + }, 20000); + + it("matches each concurrent request through one child to its own response", async () => { + const harness = await createHarness(); + harnesses.push(harness); + await harness.waitFor(readyPredicate(harness), "the gateway to become ready"); + + harness.api.setResponder((req, res) => { + const url = new URL(req.url ?? "/", "http://127.0.0.1"); + res.writeHead(200, { "content-type": "application/json" }); + res.end(JSON.stringify({ path: url.pathname })); + }); + + const ids = Array.from({ length: 8 }, (_unused, index) => `issue-${index}`); + const responses = await Promise.all( + ids.map((id) => + fetch(`${harness.baseUrl}/api/issues/${id}`, { + headers: { authorization: `Bearer ${harness.bridgeToken}` }, + }).then(async (response) => ({ status: response.status, body: await response.json() })), + ), + ); + + responses.forEach((result, index) => { + expect(result.status).toBe(200); + expect(result.body).toEqual({ path: `/api/issues/${ids[index]}` }); + }); + expect(harness.api.requests).toHaveLength(8); + }, 20000); + + it("answers 409 to outstanding and 503 to new requests when the broker closes its write side", async () => { + const harness = await createHarness({ lossExitGraceMs: 3000 }); + harnesses.push(harness); + await harness.waitFor(readyPredicate(harness), "the gateway to become ready"); + + // Hold the forward open, so the broker sends no response frame. The gateway + // keeps the request outstanding until the loss. + harness.setForwardMode("hang"); + const outstanding = fetch(`${harness.baseUrl}/api/agents/me`, { + headers: { authorization: `Bearer ${harness.bridgeToken}` }, + }); + void outstanding.catch(() => undefined); + await harness.waitFor( + () => harness.forwardedRequests.length >= 1, + "the broker to receive the forwarded request", + ); + + // The broker closes its write side, so the child stdin reaches a real EOF. + await harness.broker.close(); + + const lossResponse = await outstanding; + expect(lossResponse.status).toBe(409); + expect(lossResponse.headers.get("x-paperclip-bridge-outcome")).toBe("indeterminate"); + await expect(lossResponse.json()).resolves.toEqual({ error: "outcome_indeterminate" }); + + const afterLoss = await fetch(`${harness.baseUrl}/api/agents/me`, { + headers: { authorization: `Bearer ${harness.bridgeToken}` }, + }); + expect(afterLoss.status).toBe(503); + await expect(afterLoss.json()).resolves.toEqual({ error: "bridge_unavailable" }); + }, 20000); + + it("moves the broker to lost on a child kill, dispatches nothing more, and leaks no handle after teardown", async () => { + const harness = await createHarness(); + harnesses.push(harness); + await harness.waitFor(readyPredicate(harness), "the gateway to become ready"); + + // Hold one forward open, so a request is in flight when the child dies. + harness.setForwardMode("hang"); + const outstanding = fetch(`${harness.baseUrl}/api/agents/me`, { + headers: { authorization: `Bearer ${harness.bridgeToken}` }, + }); + void outstanding.catch(() => undefined); + await harness.waitFor( + () => harness.forwardedRequests.length >= 1, + "the broker to receive the forwarded request", + ); + const forwardedBeforeKill = harness.forwardedRequests.length; + + // A real process kill closes the stdout pipe and ends the child. + await harness.killChild(); + await harness.waitFor( + () => harness.broker.state === "lost", + "the broker to record the channel loss", + ); + expect(harness.broker.lossRecord?.reason).toBe("channel_exit"); + + // The broker dispatches nothing more after the loss. + await new Promise((resolve) => { + const timer = setTimeout(resolve, 100); + timer.unref?.(); + }); + expect(harness.forwardedRequests.length).toBe(forwardedBeforeKill); + // The outstanding fetch fails because the child died before it answered. + await expect(outstanding).rejects.toThrow(); + + await harness.teardown(); + // Teardown left no open pipe or socket handle. + expect(harness.child.stdin.destroyed).toBe(true); + expect(harness.child.stdout.destroyed).toBe(true); + expect(harness.child.stderr.destroyed).toBe(true); + expect(harness.api.listening()).toBe(false); + expect(harness.api.openSocketCount()).toBe(0); + }, 20000); +}); diff --git a/packages/adapter-utils/src/duplex-frame-codec.test.ts b/packages/adapter-utils/src/duplex-frame-codec.test.ts index 539d4898ec..e9c00b0b75 100644 --- a/packages/adapter-utils/src/duplex-frame-codec.test.ts +++ b/packages/adapter-utils/src/duplex-frame-codec.test.ts @@ -3,6 +3,7 @@ import { fileURLToPath } from "node:url"; import { describe, expect, it } from "vitest"; import { DEFAULT_MAX_DUPLEX_FRAME_BYTES, + DEFAULT_MAX_DUPLEX_REQUEST_ID_BYTES, DUPLEX_FRAME_VERSION, DuplexFrameDecoder, decodeDuplexLine, @@ -17,7 +18,14 @@ type ExpectedResult = { frame: DuplexFrame } | { error: string }; interface Vector { name: string; - category: "valid" | "invalid" | "partial" | "oversized" | "versionMismatch"; + category: + | "valid" + | "invalid" + | "partial" + | "oversized" + | "versionMismatch" + | "idBoundValid" + | "idBoundInvalid"; bytes: string; splitByteOffsets?: number[]; maxFrameBytes?: number; @@ -128,6 +136,49 @@ describe("round-trip", () => { } }); +describe("ready frame schema", () => { + // READY is a liveness signal, not an address source. The strict schema holds + // exactly the frame version and the nonce. + it("accepts a READY frame that carries exactly the version and the nonce", () => { + const result = decodeDuplexLine( + JSON.stringify({ version: DUPLEX_FRAME_VERSION, type: "ready", nonce: "a1b2c3d4e5f6a7b8" }), + ); + expect(result.ok).toBe(true); + if (result.ok) { + expect(result.frame).toEqual({ + version: DUPLEX_FRAME_VERSION, + type: "ready", + nonce: "a1b2c3d4e5f6a7b8", + }); + } + }); + + it.each([ + { name: "an absent nonce", frame: { version: DUPLEX_FRAME_VERSION, type: "ready" } }, + { + name: "a wrong-typed nonce", + frame: { version: DUPLEX_FRAME_VERSION, type: "ready", nonce: 42 }, + }, + { + name: "an extra address field", + frame: { + version: DUPLEX_FRAME_VERSION, + type: "ready", + nonce: "a1b2c3d4e5f6a7b8", + address: "http://127.0.0.1:47215", + }, + }, + { + name: "an extra port field", + frame: { version: DUPLEX_FRAME_VERSION, type: "ready", nonce: "a1b2c3d4e5f6a7b8", port: 47215 }, + }, + ])("rejects a READY frame with $name", ({ frame }) => { + const result = decodeDuplexLine(JSON.stringify(frame)); + expect(result.ok).toBe(false); + if (!result.ok) expect(result.error.code).toBe("malformed_frame"); + }); +}); + describe("streaming decoder behavior", () => { it("emits nothing until a full line arrives, then the complete frame", () => { const decoder = new DuplexFrameDecoder(); @@ -199,3 +250,63 @@ describe("streaming decoder behavior", () => { expect(results[0].ok).toBe(false); }); }); + +describe("request id byte bound", () => { + // The id byte bound caps the memory the host broker retains per distinct + // request. The bound is 256 bytes; a UTF-8 id byte can differ from a character. + function requestWithId(id: string): string { + return JSON.stringify({ + version: DUPLEX_FRAME_VERSION, + type: "request", + id, + method: "GET", + path: "/", + query: "", + headers: {}, + body: "", + }); + } + + it("accepts an id at the maximum byte size", () => { + const id = "a".repeat(DEFAULT_MAX_DUPLEX_REQUEST_ID_BYTES); + const result = decodeDuplexLine(requestWithId(id)); + expect(result.ok).toBe(true); + if (result.ok && result.frame.type === "request") expect(result.frame.id).toBe(id); + }); + + it("rejects an over-limit id with an id_too_large protocol error and never throws", () => { + const id = "a".repeat(DEFAULT_MAX_DUPLEX_REQUEST_ID_BYTES + 1); + expect(() => decodeDuplexLine(requestWithId(id))).not.toThrow(); + const result = decodeDuplexLine(requestWithId(id)); + expect(result.ok).toBe(false); + if (!result.ok) expect(result.error.code).toBe("id_too_large"); + }); + + it("measures the id in bytes, not characters, so a multi-byte id over the byte bound is rejected", () => { + // Each "😀" is four UTF-8 bytes. 65 code points make 260 bytes, one code point + // makes 4 bytes, so 65 stays over the 256-byte bound while the character count + // stays under it. + const id = "😀".repeat(65); + expect(id.length).toBeLessThan(DEFAULT_MAX_DUPLEX_REQUEST_ID_BYTES); + const result = decodeDuplexLine(requestWithId(id)); + expect(result.ok).toBe(false); + if (!result.ok) expect(result.error.code).toBe("id_too_large"); + }); + + it("applies the same bound to the response id", () => { + const id = "b".repeat(DEFAULT_MAX_DUPLEX_REQUEST_ID_BYTES + 1); + const result = decodeDuplexLine( + JSON.stringify({ + version: DUPLEX_FRAME_VERSION, + type: "response", + id, + status: 200, + headers: {}, + body: "", + outcome: "completed", + }), + ); + expect(result.ok).toBe(false); + if (!result.ok) expect(result.error.code).toBe("id_too_large"); + }); +}); diff --git a/packages/adapter-utils/src/duplex-frame-codec.ts b/packages/adapter-utils/src/duplex-frame-codec.ts index 2cc4d57c98..2efdb384a2 100644 --- a/packages/adapter-utils/src/duplex-frame-codec.ts +++ b/packages/adapter-utils/src/duplex-frame-codec.ts @@ -27,6 +27,20 @@ export const DUPLEX_FRAME_VERSION = 1; */ export const DEFAULT_MAX_DUPLEX_FRAME_BYTES = 1_000_000; +/** + * The maximum size of the frame `id` field, in bytes. The decoder rejects a + * request or a response frame that carries a longer id with an `id_too_large` + * protocol error. The read path returns the error; it never throws. + * + * The generated gateway builds each request id with `randomUUID()`, so a real id + * is 36 ASCII bytes. This bound of 256 bytes gives large headroom for that id and + * for any short future id scheme. The bound also caps the bytes the host broker + * retains per distinct request. The broker keeps one id per distinct dispatched + * request for the no-replay guarantee, so the id bound sets the per-id ceiling of + * that retained memory. + */ +export const DEFAULT_MAX_DUPLEX_REQUEST_ID_BYTES = 256; + const NEWLINE_BYTE = 0x0a; const EMPTY = Buffer.alloc(0); @@ -67,13 +81,18 @@ export interface DuplexResponseFrame { } /** - * The READY control frame. The gateway sends it one time after it validates its - * local listener address. The `address` field carries that validated address. + * The READY control frame. The gateway sends it one time after it binds the + * host-assigned listener port. READY is a liveness signal, not an address + * source. The frame carries exactly the frame version and the `nonce` string. + * The gateway echoes the nonce the host passed through the launch environment, + * so the host correlates the READY frame with this channel open. The frame + * carries no address data; the host builds the endpoint from its own stored + * port, never from the channel. */ export interface DuplexReadyFrame { version: number; type: "ready"; - address: string; + nonce: string; } /** The heartbeat control frame. Each side sends it on an interval to prove liveness. */ @@ -114,7 +133,8 @@ export type DuplexProtocolErrorCode = | "malformed_frame" | "unknown_type" | "version_mismatch" - | "frame_too_large"; + | "frame_too_large" + | "id_too_large"; /** A decode-time protocol error. The read path returns it; it never throws. */ export interface DuplexProtocolError { @@ -153,6 +173,11 @@ function isStringRecord(value: unknown): value is Record { return true; } +/** Return true when the id byte size is within {@link DEFAULT_MAX_DUPLEX_REQUEST_ID_BYTES}. */ +function idWithinLimit(id: string): boolean { + return Buffer.byteLength(id, "utf8") <= DEFAULT_MAX_DUPLEX_REQUEST_ID_BYTES; +} + /** * Encode one frame to a single line of JSON with a trailing newline. `JSON.stringify` * escapes any newline inside a string value, so the returned line holds no @@ -220,6 +245,12 @@ function validateRequest(frame: Record): DuplexDecodeResult { ) { return fail("malformed_frame", "request frame has a missing or wrong-typed field"); } + // Bound the id byte size at the protocol level. The host broker retains one id + // per distinct dispatched request, so an unbounded id is a memory-exhaustion + // path. Reject an over-limit id as a protocol error; the read path never throws. + if (!idWithinLimit(frame.id)) { + return fail("id_too_large", "request frame id exceeds the maximum size"); + } return ok(frame as unknown as DuplexRequestFrame); } @@ -234,12 +265,26 @@ function validateResponse(frame: Record): DuplexDecodeResult { ) { return fail("malformed_frame", "response frame has a missing or wrong-typed field"); } + // Apply the same id byte bound to the response id. The host echoes the request + // id on the response, so the same rule keeps both frame ids under one ceiling. + if (!idWithinLimit(frame.id)) { + return fail("id_too_large", "response frame id exceeds the maximum size"); + } return ok(frame as unknown as DuplexResponseFrame); } function validateReady(frame: Record): DuplexDecodeResult { - if (typeof frame.address !== "string") { - return fail("malformed_frame", "ready frame has a missing or wrong-typed address"); + // READY carries a liveness nonce, not an address. The schema is strict: a valid + // READY frame holds exactly `version`, `type`, and `nonce`. The decoder rejects + // an absent nonce, a wrong-typed nonce, or any extra field, so a READY frame + // that smuggles an `address`, a `port`, a `host`, or a URL never decodes. + if (typeof frame.nonce !== "string") { + return fail("malformed_frame", "ready frame has a missing or wrong-typed nonce"); + } + for (const key of Object.keys(frame)) { + if (key !== "version" && key !== "type" && key !== "nonce") { + return fail("malformed_frame", "ready frame has an unexpected field"); + } } return ok(frame as unknown as DuplexReadyFrame); } diff --git a/packages/adapter-utils/src/duplex-frame-vectors.json b/packages/adapter-utils/src/duplex-frame-vectors.json index e1e07fa0c1..74ded507fa 100644 --- a/packages/adapter-utils/src/duplex-frame-vectors.json +++ b/packages/adapter-utils/src/duplex-frame-vectors.json @@ -92,18 +92,48 @@ { "name": "valid-ready", "category": "valid", - "bytes": "{\"version\":1,\"type\":\"ready\",\"address\":\"127.0.0.1:47215\"}\n", + "bytes": "{\"version\":1,\"type\":\"ready\",\"nonce\":\"9f8e7d6c5b4a39281706f5e4d3c2b1a0\"}\n", "roundTrip": true, "expected": [ { "frame": { "version": 1, "type": "ready", - "address": "127.0.0.1:47215" + "nonce": "9f8e7d6c5b4a39281706f5e4d3c2b1a0" } } ] }, + { + "name": "invalid-ready-missing-nonce", + "category": "invalid", + "bytes": "{\"version\":1,\"type\":\"ready\"}\n", + "expected": [ + { + "error": "malformed_frame" + } + ] + }, + { + "name": "invalid-ready-with-address", + "category": "invalid", + "bytes": "{\"version\":1,\"type\":\"ready\",\"nonce\":\"9f8e7d6c5b4a39281706f5e4d3c2b1a0\",\"address\":\"http://127.0.0.1:47215\"}\n", + "expected": [ + { + "error": "malformed_frame" + } + ] + }, + { + "name": "invalid-ready-with-port", + "category": "invalid", + "bytes": "{\"version\":1,\"type\":\"ready\",\"nonce\":\"9f8e7d6c5b4a39281706f5e4d3c2b1a0\",\"port\":47215}\n", + "expected": [ + { + "error": "malformed_frame" + } + ] + }, { "name": "valid-heartbeat", "category": "valid", @@ -353,6 +383,36 @@ "error": "version_mismatch" } ] + }, + { + "name": "request-id-at-max-bytes", + "category": "idBoundValid", + "bytes": "{\"version\":1,\"type\":\"request\",\"id\":\"aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa\",\"method\":\"GET\",\"path\":\"/\",\"query\":\"\",\"headers\":{},\"body\":\"\"}\n", + "roundTrip": true, + "expected": [ + { + "frame": { + "version": 1, + "type": "request", + "id": "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + "method": "GET", + "path": "/", + "query": "", + "headers": {}, + "body": "" + } + } + ] + }, + { + "name": "request-id-over-max-bytes", + "category": "idBoundInvalid", + "bytes": "{\"version\":1,\"type\":\"request\",\"id\":\"aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa\",\"method\":\"GET\",\"path\":\"/\",\"query\":\"\",\"headers\":{},\"body\":\"\"}\n", + "expected": [ + { + "error": "id_too_large" + } + ] } ] } diff --git a/packages/adapter-utils/src/duplex-telemetry.ts b/packages/adapter-utils/src/duplex-telemetry.ts new file mode 100644 index 0000000000..4bc08abebf --- /dev/null +++ b/packages/adapter-utils/src/duplex-telemetry.ts @@ -0,0 +1,344 @@ +/** + * The fixed observability surface for the sandbox duplex transport. + * + * This module owns the one closed contract for every duplex telemetry sink: the + * span names, the counter names, the one event name, the dimension keys, and the + * enum values. Each name and value is a literal constant here, so the surface + * never drifts and a test asserts the exact set. + * + * The module also owns the provider allowlist. The `provider` dimension carries + * one approved public value, `daytona`. Any other plugin key maps to the constant + * `other`. The map runs inside this module before every sink, so a raw plugin key + * never reaches a span attribute, a counter label, or an event field. + * + * The module stays free of `@opentelemetry/api` and of the database. The host + * injects a {@link DuplexTelemetryRecorder}; the default is a no-op recorder, so + * the whole surface stays inert until the host binds a real recorder. Every + * recorder call sits inside an error swallow, so a telemetry failure never breaks + * the request path. + */ + +/** The span for one duplex channel-open attempt. */ +export const DUPLEX_SPAN_CHANNEL_OPEN = "sandbox.duplex.channel_open"; +/** The span for one duplex request. It carries the request latency. */ +export const DUPLEX_SPAN_REQUEST = "sandbox.duplex.request"; + +/** The one duplex transport event. The host emits it at each transport boundary. */ +export const DUPLEX_TRANSPORT_EVENT = "sandbox.duplex.transport"; + +/** The guarded counter for one successful channel open. */ +export const DUPLEX_COUNTER_CHANNEL_OPEN_TOTAL = "sandbox_duplex_channel_open_total"; +/** The guarded counter for one fallback to the file bridge. */ +export const DUPLEX_COUNTER_FALLBACK_TOTAL = "sandbox_duplex_fallback_total"; +/** The guarded counter for one terminal channel loss. */ +export const DUPLEX_COUNTER_LOSS_TOTAL = "sandbox_duplex_loss_total"; +/** The guarded counter for one leaked provider session on teardown. */ +export const DUPLEX_COUNTER_SESSION_LEAK_TOTAL = "sandbox_duplex_session_leak_total"; + +/** + * The closed dimension-key set. Every span attribute, counter label, and event + * field uses only these keys. A test asserts the exact set, so a new key never + * reaches a sink by accident. + */ +export const DUPLEX_DIMENSION_KEYS = [ + "provider", + "transport", + "outcome", + "fallback_reason", + "loss_class", + "loss_reason", +] as const; + +/** One dimension key from the closed set. */ +export type DuplexDimensionKey = (typeof DUPLEX_DIMENSION_KEYS)[number]; + +/** The transport a record is about. */ +export type DuplexTransportValue = "duplex" | "file"; + +/** The outcome of a record. */ +export type DuplexOutcomeValue = "ok" | "error"; + +/** + * The reason the host selected the file bridge instead of the duplex transport. + * The open-failure stage is split, so a reader groups a failed open by the exact + * stage: the process-scoped route ceiling was full (`route_busy`), the entrypoint + * sync failed (`entrypoint_sync_failed`), the broker construction failed + * (`broker_construction_failed`), or the channel open failed (`channel_open_failed`). + */ +export type DuplexFallbackReason = + | "gate_off" + | "capability_absent" + | "route_busy" + | "entrypoint_sync_failed" + | "broker_construction_failed" + | "channel_open_failed" + | "ready_invalid" + | "ready_nonce_mismatch" + | "ready_timeout" + | "contaminated"; + +/** The class of a terminal loss, relative to the first request dispatch. */ +export type DuplexLossClass = "pre_dispatch" | "post_dispatch"; + +/** + * The closed, typed reason for a terminal channel loss. The host maps every loss + * cause to one of these values before any sink reads it. The set covers the + * loss-detection modes the transport names: a gateway stdin end of file, a + * provider process exit, a heartbeat timeout, and an RPC failure. It adds + * `write_error` for a rejected host-to-sandbox write and `other` for an unknown + * cause. The host maps an unknown cause or any caught provider text to `other`, + * so no raw provider text reaches a sink. + */ +export const DUPLEX_LOSS_REASONS = [ + "stdin_eof", + "provider_exit", + "heartbeat_timeout", + "rpc_failure", + "write_error", + "other", +] as const; + +/** One typed loss reason from the closed set. */ +export type DuplexLossReason = (typeof DUPLEX_LOSS_REASONS)[number]; + +/** The host-owned closed loss-reason set. It backs {@link normalizeDuplexLossReason}. */ +const LOSS_REASONS: ReadonlySet = new Set(DUPLEX_LOSS_REASONS); + +/** + * Map a raw loss-cause value to the closed {@link DuplexLossReason} set. Return + * the value when the closed set holds it. Return `other` for any other value or a + * missing value, so a raw provider string never reaches a sink. + */ +export function normalizeDuplexLossReason(value: string | null | undefined): DuplexLossReason { + return typeof value === "string" && LOSS_REASONS.has(value) + ? (value as DuplexLossReason) + : "other"; +} + +/** The one approved public provider value. */ +export const DUPLEX_APPROVED_PROVIDER = "daytona"; +/** The constant for any provider key outside the allowlist. */ +export const DUPLEX_PROVIDER_OTHER = "other"; + +/** The `provider` dimension value after the allowlist map. */ +export type DuplexProviderValue = typeof DUPLEX_APPROVED_PROVIDER | typeof DUPLEX_PROVIDER_OTHER; + +/** The host-owned closed provider allowlist. It holds one approved public value. */ +const APPROVED_PROVIDERS: ReadonlySet = new Set([DUPLEX_APPROVED_PROVIDER]); + +/** + * Map a raw provider key to the closed `provider` dimension value. Return the key + * when the allowlist holds it. Return `other` for any other value, so a raw + * plugin key never reaches a sink. A missing key also maps to `other`. + */ +export function normalizeDuplexProvider(key: string | null | undefined): DuplexProviderValue { + return typeof key === "string" && APPROVED_PROVIDERS.has(key) + ? (key as DuplexProviderValue) + : DUPLEX_PROVIDER_OTHER; +} + +/** + * The dimension bag a sink reads. Only the closed keys appear. The optional keys + * are present only when the record defines them, so a span or a counter never + * carries an empty dimension. + */ +export interface DuplexTelemetryDimensions { + provider: DuplexProviderValue; + transport: DuplexTransportValue; + outcome?: DuplexOutcomeValue; + fallback_reason?: DuplexFallbackReason; + loss_class?: DuplexLossClass; + loss_reason?: DuplexLossReason; +} + +/** One span record the host records. The request span carries a latency. */ +export interface DuplexTelemetrySpanRecord { + name: string; + dimensions: DuplexTelemetryDimensions; + /** The request latency in milliseconds. Only the request span sets it. */ + latencyMs?: number; +} + +/** One counter increment the host records. */ +export interface DuplexTelemetryCounterRecord { + metric: string; + dimensions: DuplexTelemetryDimensions; +} + +/** One event the host emits. */ +export interface DuplexTelemetryEventRecord { + name: string; + dimensions: DuplexTelemetryDimensions; +} + +/** + * The injected low-level recorder. The host binds it to the existing telemetry + * pipeline: the span to the OTel tracer, the counter to the guarded counter + * store, and the event to the run-events bridge. The default is a no-op recorder. + * The recorder receives only already-mapped dimensions, so the raw provider key + * never reaches it. + */ +export interface DuplexTelemetryRecorder { + recordSpan(record: DuplexTelemetrySpanRecord): void; + incrementCounter(record: DuplexTelemetryCounterRecord): void; + emitEvent(record: DuplexTelemetryEventRecord): void; +} + +/** A no-op recorder. Every method does nothing, so the surface stays inert. */ +export const NOOP_DUPLEX_TELEMETRY_RECORDER: DuplexTelemetryRecorder = { + recordSpan() {}, + incrementCounter() {}, + emitEvent() {}, +}; + +/** One channel-open attempt. The caller reports exactly one terminal. */ +export interface DuplexChannelOpenAttempt { + /** + * The channel opened and readiness passed. Record the channel-open span with + * the `ok` outcome, increment the channel-open counter, and emit the transport + * event for the duplex transport. + */ + ready(): void; + /** + * The channel did not open or readiness failed. Record the channel-open span + * with the `error` outcome, increment the fallback counter, and emit the + * transport event for the file bridge. + */ + fallback(reason: DuplexFallbackReason): void; +} + +/** + * The bound telemetry facade the call sites use. It carries the normalized + * provider, so a call site never passes a raw key. It maps each semantic event to + * the fixed names and dimensions, then calls the recorder inside an error swallow. + */ +export interface DuplexTelemetry { + /** Begin a channel-open attempt. The caller reports `ready` or `fallback`. */ + startChannelOpen(): DuplexChannelOpenAttempt; + /** + * Record a fallback to the file bridge with no channel-open attempt. Use it for + * `gate_off` and `capability_absent`, where the host opens no channel. + */ + recordFallback(reason: DuplexFallbackReason): void; + /** Record one duplex request span with its latency and outcome. */ + recordRequest(record: { latencyMs: number; outcome: DuplexOutcomeValue }): void; + /** + * Record one terminal channel loss. The caller passes the loss class and the + * typed, closed loss reason. The loss counter and the transport loss event carry + * both dimensions. No raw provider text rides either sink. + */ + recordLoss(lossClass: DuplexLossClass, lossReason: DuplexLossReason): void; + /** Record one leaked provider session on teardown. */ + recordSessionLeak(): void; +} + +/** The options for {@link createDuplexTelemetry}. */ +export interface DuplexTelemetryOptions { + /** The injected recorder. The default is the no-op recorder. */ + recorder?: DuplexTelemetryRecorder | null; + /** The raw provider key. The facade maps it through the allowlist one time. */ + providerKey?: string | null; +} + +/** + * Build the bound telemetry facade. The facade normalizes the provider key one + * time, then reuses the mapped value for every record. Every recorder call sits + * inside a `try/catch`, so a throwing recorder never breaks the request path. A + * missing recorder yields a facade whose methods do nothing. + */ +export function createDuplexTelemetry(options: DuplexTelemetryOptions = {}): DuplexTelemetry { + const recorder = options.recorder ?? NOOP_DUPLEX_TELEMETRY_RECORDER; + const provider = normalizeDuplexProvider(options.providerKey); + + const safeSpan = (record: DuplexTelemetrySpanRecord): void => { + try { + recorder.recordSpan(record); + } catch { + // Observability must not break the request path. + } + }; + const safeCounter = (record: DuplexTelemetryCounterRecord): void => { + try { + recorder.incrementCounter(record); + } catch { + // Observability must not break the request path. + } + }; + const safeEvent = (record: DuplexTelemetryEventRecord): void => { + try { + recorder.emitEvent(record); + } catch { + // Observability must not break the request path. + } + }; + + const recordFallback = (reason: DuplexFallbackReason): void => { + const dimensions: DuplexTelemetryDimensions = { + provider, + transport: "file", + outcome: "error", + fallback_reason: reason, + }; + safeCounter({ metric: DUPLEX_COUNTER_FALLBACK_TOTAL, dimensions }); + safeEvent({ name: DUPLEX_TRANSPORT_EVENT, dimensions }); + }; + + return { + startChannelOpen(): DuplexChannelOpenAttempt { + let settled = false; + return { + ready(): void { + if (settled) return; + settled = true; + const dimensions: DuplexTelemetryDimensions = { + provider, + transport: "duplex", + outcome: "ok", + }; + safeSpan({ name: DUPLEX_SPAN_CHANNEL_OPEN, dimensions }); + safeCounter({ metric: DUPLEX_COUNTER_CHANNEL_OPEN_TOTAL, dimensions }); + safeEvent({ name: DUPLEX_TRANSPORT_EVENT, dimensions }); + }, + fallback(reason: DuplexFallbackReason): void { + if (settled) return; + settled = true; + // The channel-open span records the failed attempt on the duplex + // transport; the counter and the event record the file-bridge fallback. + // The span carries `fallback_reason` on this fallback path only, so a + // reader can group the failed opens by reason. `fallback_reason` is a + // closed dimension key, so no new key reaches a sink. + safeSpan({ + name: DUPLEX_SPAN_CHANNEL_OPEN, + dimensions: { provider, transport: "duplex", outcome: "error", fallback_reason: reason }, + }); + recordFallback(reason); + }, + }; + }, + recordFallback, + recordRequest(record: { latencyMs: number; outcome: DuplexOutcomeValue }): void { + safeSpan({ + name: DUPLEX_SPAN_REQUEST, + dimensions: { provider, transport: "duplex", outcome: record.outcome }, + latencyMs: record.latencyMs, + }); + }, + recordLoss(lossClass: DuplexLossClass, lossReason: DuplexLossReason): void { + const dimensions: DuplexTelemetryDimensions = { + provider, + transport: "duplex", + outcome: "error", + loss_class: lossClass, + loss_reason: lossReason, + }; + safeCounter({ metric: DUPLEX_COUNTER_LOSS_TOTAL, dimensions }); + safeEvent({ name: DUPLEX_TRANSPORT_EVENT, dimensions }); + }, + recordSessionLeak(): void { + safeCounter({ + metric: DUPLEX_COUNTER_SESSION_LEAK_TOTAL, + dimensions: { provider, transport: "duplex", outcome: "error" }, + }); + }, + }; +} diff --git a/packages/adapter-utils/src/execution-target-sandbox.test.ts b/packages/adapter-utils/src/execution-target-sandbox.test.ts index 75756182a5..3f23cc3871 100644 --- a/packages/adapter-utils/src/execution-target-sandbox.test.ts +++ b/packages/adapter-utils/src/execution-target-sandbox.test.ts @@ -14,13 +14,18 @@ import { } from "./sandbox-callback-bridge.js"; import { + __duplexReadinessTesting, + buildDuplexGatewayLaunchArgv, DEFAULT_REMOTE_SANDBOX_ADAPTER_TIMEOUT_SEC, + adapterExecutionTargetDuplexTelemetryRecorder, + adapterExecutionTargetEnablesSandboxDuplexBridge, adapterExecutionTargetSessionIdentity, adapterExecutionTargetToRemoteSpec, adapterExecutionTargetUsesPaperclipBridge, ensureAdapterExecutionTargetCommandResolvable, formatAdapterExecutionTimeoutErrorMessage, formatAdapterExecutionTimeoutStartLogLine, + parseAdapterExecutionTarget, postedIssueCommentLogMarker, resolveAdapterExecutionTargetTimeout, resolveAdapterExecutionTargetTimeoutSec, @@ -29,6 +34,7 @@ import { startAdapterExecutionTargetProcessSessionBridge, startAdapterExecutionTargetPaperclipBridge, type AdapterSandboxExecutionTarget, + type EffectiveSandboxCapabilities, } from "./execution-target.js"; import { createRuntimeSpanRunner, @@ -40,6 +46,38 @@ import { import { createSandboxRunLogTailFactory } from "./sandbox-run-log-stream.js"; import { runChildProcess } from "./server-utils.js"; import { shellQuote } from "./ssh.js"; +import type { CommandManagedDuplexChannel } from "./command-managed-runtime.js"; +import { + DEFAULT_MAX_DUPLEX_FRAME_BYTES, + DUPLEX_FRAME_VERSION, + decodeDuplexLine, + encodeDuplexFrame, + type DuplexRequestFrame, + type DuplexResponseFrame, +} from "./duplex-frame-codec.js"; +import { + assertNestedDuplexBrokerBudgets, + createDuplexBridgeBroker, + type DuplexBrokerForwardResult, + type DuplexBrokerLossRecord, + type DuplexBrokerRequestRecord, + type DuplexBrokerState, +} from "./duplex-bridge-broker.js"; +import { + createDuplexTelemetry, + DUPLEX_COUNTER_CHANNEL_OPEN_TOTAL, + DUPLEX_COUNTER_FALLBACK_TOTAL, + DUPLEX_COUNTER_LOSS_TOTAL, + DUPLEX_DIMENSION_KEYS, + DUPLEX_SPAN_CHANNEL_OPEN, + DUPLEX_SPAN_REQUEST, + DUPLEX_TRANSPORT_EVENT, + type DuplexTelemetryCounterRecord, + type DuplexTelemetryDimensions, + type DuplexTelemetryEventRecord, + type DuplexTelemetryRecorder, + type DuplexTelemetrySpanRecord, +} from "./duplex-telemetry.js"; const execFileAsync = promisify(execFile); @@ -2462,7 +2500,14 @@ describe("sandbox adapter execution targets", () => { } }); - it("fails oversized host responses with a 502 before returning them to the sandbox client", async () => { + it("fails an oversized host response with a non-retryable 409 so a committed mutation never repeats", async () => { + // The host receives the request and commits the mutation, then sends a + // response body over the size limit. The forward reads the body after the + // fetch resolves, so the read failure happens after the host commit. The + // forward must return a non-retryable 504 with the indeterminate outcome, not + // a retryable 502. The in-sandbox server maps the indeterminate 504 to a + // non-retryable 409. A retryable status would repeat the mutation with a new + // request id outside the broker deduplication set. const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-execution-target-bridge-limit-")); cleanupDirs.push(rootDir); const remoteCwd = path.join(rootDir, "workspace"); @@ -2470,7 +2515,99 @@ describe("sandbox adapter execution targets", () => { await mkdir(runtimeRootDir, { recursive: true }); const requests: Array<{ method: string; url: string; auth: string | null; runId: string | null }> = []; - const largeBody = "x".repeat(64); + // The host body sits over the size limit, so the forward read fails. The + // limit stays above the small indeterminate marker the forward returns, so the + // marker still reaches the server for the 504-to-409 map. + const largeBody = "x".repeat(1024); + const apiServer = createServer((req, res) => { + requests.push({ + method: req.method ?? "GET", + url: req.url ?? "/", + auth: req.headers.authorization ?? null, + runId: typeof req.headers["x-paperclip-run-id"] === "string" ? req.headers["x-paperclip-run-id"] : null, + }); + res.writeHead(201, { + "content-type": "application/json", + "content-length": String(Buffer.byteLength(largeBody, "utf8")), + }); + res.end(largeBody); + }); + await new Promise((resolve, reject) => { + apiServer.once("error", reject); + apiServer.listen(0, "127.0.0.1", () => resolve()); + }); + const address = apiServer.address(); + if (!address || typeof address === "string") { + throw new Error("Expected the bridge test API server to listen on a TCP port."); + } + + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "e2b", + environmentId: "env-1", + leaseId: "lease-1", + remoteCwd, + runner: createLocalSandboxRunner(), + timeoutMs: 30_000, + }; + + const bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-bridge-limit", + target, + runtimeRootDir, + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: `http://127.0.0.1:${address.port}`, + maxBodyBytes: 512, + }); + try { + const response = await fetch(`${bridge!.env.PAPERCLIP_API_URL}/api/issues/issue-1/comments`, { + method: "POST", + headers: { + authorization: `Bearer ${bridge!.env.PAPERCLIP_API_KEY}`, + "content-type": "application/json", + }, + body: JSON.stringify({ body: "Status update." }), + }); + + // The indeterminate 504 maps to a non-retryable 409, so the caller does not + // retry the committed mutation. + expect(response.status).toBe(409); + expect(response.headers.get("x-paperclip-bridge-outcome")).toBe("indeterminate"); + await expect(response.json()).resolves.toEqual({ + error: "Bridge response body exceeded the configured size limit of 512 bytes.", + outcome: "indeterminate", + retryable: false, + }); + // The host ran the mutation exactly once. It never receives a retry. + expect(requests).toEqual([{ + method: "POST", + url: "/api/issues/issue-1/comments", + auth: "Bearer real-run-jwt", + runId: "run-bridge-limit", + }]); + } finally { + await bridge?.stop(); + await new Promise((resolve) => apiServer.close(() => resolve())); + } + }); + + it("keeps an oversized host response for a safe method retryable so the read failure does not turn terminal", async () => { + // A GET never changes host state, so a retry cannot double-apply a mutation. + // The host sends a response body over the size limit, so the forward read + // fails after the fetch resolves. For a safe method the forward must return a + // retryable 502 with no indeterminate marker, not the non-retryable 504 the + // forward returns for a mutating method. The in-sandbox server passes the 502 + // through, so the caller can retry the safe read. + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-execution-target-bridge-safe-limit-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + const runtimeRootDir = path.join(remoteCwd, ".paperclip-runtime", "codex"); + await mkdir(runtimeRootDir, { recursive: true }); + + const requests: Array<{ method: string; url: string; auth: string | null; runId: string | null }> = []; + const largeBody = "x".repeat(1024); const apiServer = createServer((req, res) => { requests.push({ method: req.method ?? "GET", @@ -2505,31 +2642,34 @@ describe("sandbox adapter execution targets", () => { }; const bridge = await startAdapterExecutionTargetPaperclipBridge({ - runId: "run-bridge-limit", + runId: "run-bridge-safe-limit", target, runtimeRootDir, adapterKey: "codex", hostApiToken: "real-run-jwt", hostApiUrl: `http://127.0.0.1:${address.port}`, - maxBodyBytes: 32, + maxBodyBytes: 512, }); try { - const response = await fetch(`${bridge!.env.PAPERCLIP_API_URL}/api/agents/me`, { + const response = await fetch(`${bridge!.env.PAPERCLIP_API_URL}/api/issues/issue-1`, { + method: "GET", headers: { authorization: `Bearer ${bridge!.env.PAPERCLIP_API_KEY}`, - accept: "application/json", }, }); + // The forward returns a retryable 502 with no indeterminate marker, so the + // server passes it through instead of mapping it to a terminal 409. expect(response.status).toBe(502); + expect(response.headers.get("x-paperclip-bridge-outcome")).toBeNull(); await expect(response.json()).resolves.toEqual({ - error: "Bridge response body exceeded the configured size limit of 32 bytes.", + error: "Bridge response body exceeded the configured size limit of 512 bytes.", }); expect(requests).toEqual([{ method: "GET", - url: "/api/agents/me", + url: "/api/issues/issue-1", auth: "Bearer real-run-jwt", - runId: "run-bridge-limit", + runId: "run-bridge-safe-limit", }]); } finally { await bridge?.stop(); @@ -2753,6 +2893,1927 @@ describe("sandbox adapter execution targets", () => { await new Promise((resolve) => apiServer.close(() => resolve())); } }); + + // The full effective-capability snapshot with one flag set. The two strict + // gates read `duplexCommandStream`; the other flags stay false. + function duplexCapabilities(duplexCommandStream: boolean): EffectiveSandboxCapabilities { + return { + reusableLeases: false, + nativeSyncIn: false, + nativeSyncOut: false, + persistentProcessSessions: false, + independentControlCommands: false, + incrementalSessionOutput: false, + concurrentSyncOperations: false, + duplexCommandStream, + }; + } + + // The control surface a test uses to drive one fake duplex channel and read + // what the broker wrote back through it. + interface DuplexSelectionControl { + openCount: number; + written: DuplexResponseFrame[]; + writtenTypes: string[]; + stopCount: number; + closeCount: number; + emitData: (chunk: string) => void; + } + + // The hook a test supplies to script the first frames the fake gateway sends + // after the host binds the readiness gate. The default hook echoes a valid + // READY frame with the launch nonce, so the happy path needs no hook. + interface DuplexOpenContext { + nonce: string; + port: string; + emitRaw: (text: string) => void; + emitFrame: (frame: Record) => void; + emitExit: (exit: { exitCode: number | null }) => void; + } + + // Build a runner that runs real shell commands for the asset upload and the + // file bridge, and a fake `openDuplexChannel`. The fake parses the nonce and + // the port out of the launch command, so the test proves the host passes both + // only through the launch environment. + function makeDuplexSelectionRunner(onOpen?: (ctx: DuplexOpenContext) => void): { + runner: ReturnType & { + openDuplexChannel: (openInput: { command: readonly string[] }) => Promise; + }; + control: DuplexSelectionControl; + } { + const base = createLocalSandboxRunner(); + const control: DuplexSelectionControl = { + openCount: 0, + written: [], + writtenTypes: [], + stopCount: 0, + closeCount: 0, + emitData: () => {}, + }; + const openDuplexChannel = async (openInput: { + command: readonly string[]; + }): Promise => { + control.openCount += 1; + const joined = openInput.command.join(" "); + const nonce = /PAPERCLIP_BRIDGE_NONCE='([^']*)'/.exec(joined)?.[1] ?? ""; + const port = /PAPERCLIP_BRIDGE_PORT='([^']*)'/.exec(joined)?.[1] ?? ""; + let dataListener: ((chunk: string) => void) | null = null; + let exitListener: ((exit: { exitCode: number | null }) => void) | null = null; + const channel: CommandManagedDuplexChannel = { + write(data: string): void { + const decoded = decodeDuplexLine(data.replace(/\n$/, "")); + if (decoded.ok) { + control.writtenTypes.push(decoded.frame.type); + if (decoded.frame.type === "response") control.written.push(decoded.frame); + } + }, + onData(listener: (chunk: string) => void): void { + dataListener = listener; + control.emitData = (chunk) => dataListener?.(chunk); + // Drive the readiness emission on the next tick, after the gate also + // registers its exit listener. + setImmediate(() => { + const emitRaw = (text: string) => dataListener?.(text); + const emitFrame = (frame: Record) => + dataListener?.(`${JSON.stringify(frame)}\n`); + const emitExit = (exit: { exitCode: number | null }) => exitListener?.(exit); + if (onOpen) { + onOpen({ nonce, port, emitRaw, emitFrame, emitExit }); + } else { + emitFrame({ version: 1, type: "ready", nonce }); + } + }); + }, + onExit(listener: (exit: { exitCode: number | null }) => void): void { + exitListener = listener; + }, + stop(): void { + control.stopCount += 1; + }, + close(): Promise { + control.closeCount += 1; + return Promise.resolve(); + }, + }; + return channel; + }; + return { runner: { ...base, openDuplexChannel }, control }; + } + + // Start a host API server that records each forwarded request, so a test can + // assert the real token and the run id reach the host, or that a rejected + // request never forwards. + async function startRecordingApiServer(): Promise<{ + origin: string; + requests: Array<{ + method: string; + url: string; + auth: string | null; + runId: string | null; + headers: Record; + }>; + close: () => Promise; + }> { + const requests: Array<{ + method: string; + url: string; + auth: string | null; + runId: string | null; + headers: Record; + }> = []; + const server = createServer((req, res) => { + const headers: Record = {}; + for (const [key, value] of Object.entries(req.headers)) { + if (typeof value === "string") headers[key] = value; + } + requests.push({ + method: req.method ?? "GET", + url: req.url ?? "/", + auth: req.headers.authorization ?? null, + runId: typeof req.headers["x-paperclip-run-id"] === "string" ? req.headers["x-paperclip-run-id"] : null, + headers, + }); + res.writeHead(200, { "content-type": "application/json" }); + res.end(JSON.stringify({ ok: true })); + }); + await new Promise((resolve, reject) => { + server.once("error", reject); + server.listen(0, "127.0.0.1", () => resolve()); + }); + const address = server.address(); + if (!address || typeof address === "string") { + throw new Error("Expected the recording API server to listen on a TCP port."); + } + return { + origin: `http://127.0.0.1:${address.port}`, + requests, + close: () => new Promise((resolve) => server.close(() => resolve())), + }; + } + + it("selects the duplex transport when both strict gates pass and readiness completes", async () => { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-select-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + const api = await startRecordingApiServer(); + const { runner, control } = makeDuplexSelectionRunner(); + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "e2b", + environmentId: "env-1", + leaseId: "lease-1", + remoteCwd, + timeoutMs: 30_000, + runner, + effectiveCapabilities: duplexCapabilities(true), + }; + + const bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-duplex", + target, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: true, + }); + try { + expect(bridge).not.toBeNull(); + expect(control.openCount).toBe(1); + expect(bridge?.env.PAPERCLIP_API_BRIDGE_MODE).toBe("duplex_v1"); + // The host builds the origin from the port it assigned, never from a frame. + expect(bridge?.env.PAPERCLIP_API_URL).toMatch(/^http:\/\/127\.0\.0\.1:\d+$/); + expect(bridge?.env.PAPERCLIP_API_KEY).not.toBe("real-run-jwt"); + + // The gateway forwards one agent request as a request frame. The broker + // forwards it on the host path with the real token and the run id, then + // writes one response frame back. + control.emitData( + encodeDuplexFrame({ + version: DUPLEX_FRAME_VERSION, + type: "request", + id: "req-1", + method: "GET", + path: "/api/agents/me", + query: "", + headers: { authorization: "Bearer bridge-token" }, + body: "", + }), + ); + await waitForCondition( + () => control.written.length >= 1, + "the broker to write a duplex response frame", + 4000, + ); + expect(control.written[0]).toMatchObject({ + type: "response", + id: "req-1", + status: 200, + outcome: "completed", + }); + expect(api.requests).toHaveLength(1); + expect(api.requests[0]).toMatchObject({ + method: "GET", + url: "/api/agents/me", + auth: "Bearer real-run-jwt", + runId: "run-duplex", + }); + } finally { + await bridge?.stop(); + await api.close(); + } + // Teardown closed the channel before lease release, then stopped the child. + expect(control.closeCount).toBeGreaterThanOrEqual(1); + expect(control.stopCount).toBeGreaterThanOrEqual(1); + }, 20000); + + it("streams run logs on the duplex path under the same gate and log line as the file path", async () => { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-runlog-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + const api = await startRecordingApiServer(); + const { runner } = makeDuplexSelectionRunner(); + const logs: Array<{ stream: "stdout" | "stderr"; chunk: string }> = []; + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "daytona", + remoteCwd, + timeoutMs: 30_000, + runner, + streamRunLogs: true, + effectiveCapabilities: duplexCapabilities(true), + }; + + const bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-duplex-log", + target, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: true, + onLog: async (stream, chunk) => { + logs.push({ stream, chunk }); + }, + }); + try { + // The duplex transport served, and it still streams run logs with the same + // gate and the same log line as the file path. + expect(bridge?.env.PAPERCLIP_API_BRIDGE_MODE).toBe("duplex_v1"); + expect(bridge?.runLogTail).toBeTruthy(); + expect(combinedStream(logs, "stdout")).toContain("Sandbox run log streaming enabled"); + const wrapped = bridge!.runLogTail!.create().wrapCommand("agent-cli", ["--message", "hello world"]); + expect(wrapped.args.join("\n")).toContain("tee -a"); + expect(wrapped.args.join("\n")).toContain("agent-cli"); + } finally { + await bridge?.stop(); + await api.close(); + } + }, 20000); + + it("returns no run-log tail on the duplex path when streaming is opted out", async () => { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-runlog-off-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + const api = await startRecordingApiServer(); + const { runner } = makeDuplexSelectionRunner(); + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "daytona", + remoteCwd, + timeoutMs: 30_000, + runner, + streamRunLogs: false, + effectiveCapabilities: duplexCapabilities(true), + }; + + const bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-duplex-log-off", + target, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: true, + }); + try { + expect(bridge?.env.PAPERCLIP_API_BRIDGE_MODE).toBe("duplex_v1"); + expect(bridge?.runLogTail ?? null).toBeNull(); + } finally { + await bridge?.stop(); + await api.close(); + } + }, 20000); + + it("routes duplex channel-open and fallback records to a recorder attached on the server seam", async () => { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-recorder-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + const api = await startRecordingApiServer(); + + const counters: DuplexTelemetryCounterRecord[] = []; + const recorder: DuplexTelemetryRecorder = { + recordSpan() {}, + incrementCounter(record) { + counters.push(record); + }, + emitEvent() {}, + }; + + // A channel open reaches the recorder on the duplex success path. The host + // attaches the recorder to the sandbox target on the same seam as the + // runner; the caller reads it with the accessor and passes it to the bridge. + const openTarget: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "daytona", + remoteCwd, + timeoutMs: 30_000, + runner: makeDuplexSelectionRunner().runner, + effectiveCapabilities: duplexCapabilities(true), + duplexTelemetryRecorder: recorder, + }; + const openBridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-duplex-open", + target: openTarget, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: true, + duplexTelemetryRecorder: adapterExecutionTargetDuplexTelemetryRecorder(openTarget), + }); + try { + expect(openBridge?.env.PAPERCLIP_API_BRIDGE_MODE).toBe("duplex_v1"); + const open = counters.find((record) => record.metric === DUPLEX_COUNTER_CHANNEL_OPEN_TOTAL); + expect(open?.dimensions.transport).toBe("duplex"); + expect(open?.dimensions.provider).toBe("daytona"); + } finally { + await openBridge?.stop(); + } + + // A fallback reaches the same recorder. The kill switch off records a + // gate_off fallback with the file transport. + counters.length = 0; + const fallbackTarget: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "daytona", + remoteCwd, + timeoutMs: 30_000, + runner: makeDuplexSelectionRunner().runner, + effectiveCapabilities: duplexCapabilities(true), + duplexTelemetryRecorder: recorder, + }; + const fallbackBridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-duplex-fallback", + target: fallbackTarget, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: false, + duplexTelemetryRecorder: adapterExecutionTargetDuplexTelemetryRecorder(fallbackTarget), + }); + try { + expect(fallbackBridge?.env.PAPERCLIP_API_BRIDGE_MODE).toBe("queue_v1"); + const fallback = counters.find((record) => record.metric === DUPLEX_COUNTER_FALLBACK_TOTAL); + expect(fallback?.dimensions.fallback_reason).toBe("gate_off"); + expect(fallback?.dimensions.transport).toBe("file"); + } finally { + await fallbackBridge?.stop(); + await api.close(); + } + }, 20000); + + it.each([ + { name: "the kill switch is off with the capability granted", enable: false, capability: true }, + { name: "the capability is absent with the kill switch on", enable: true, capability: false }, + ])("selects the file bridge when $name", async ({ enable, capability }) => { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-gate-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + const api = await startRecordingApiServer(); + const { runner, control } = makeDuplexSelectionRunner(); + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "e2b", + remoteCwd, + timeoutMs: 30_000, + runner, + effectiveCapabilities: duplexCapabilities(capability), + }; + + const bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-gate", + target, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: enable, + }); + try { + expect(bridge).not.toBeNull(); + // Neither gate combination opened a duplex channel; the file bridge serves. + expect(control.openCount).toBe(0); + expect(bridge?.env.PAPERCLIP_API_BRIDGE_MODE).toBe("queue_v1"); + } finally { + await bridge?.stop(); + await api.close(); + } + }, 20000); + + it.each([ + { + name: "a mismatched nonce", + onOpen: (ctx: DuplexOpenContext) => + ctx.emitFrame({ version: 1, type: "ready", nonce: "00000000000000000000000000000000" }), + }, + { + name: "an incomplete READY frame", + onOpen: (ctx: DuplexOpenContext) => ctx.emitRaw('{"version":1,"type":"ready"}\n'), + }, + { + name: "protocol contamination before READY", + onOpen: (ctx: DuplexOpenContext) => ctx.emitFrame({ version: 1, type: "heartbeat" }), + }, + { + name: "a gateway bind failure with no READY frame", + onOpen: (ctx: DuplexOpenContext) => ctx.emitExit({ exitCode: 1 }), + }, + { + name: "a readiness timeout with no frame at all", + onOpen: () => {}, + }, + ])("fails closed to the file bridge on $name and leaves no live session", async ({ onOpen }) => { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-fail-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + const api = await startRecordingApiServer(); + const { runner, control } = makeDuplexSelectionRunner(onOpen); + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "e2b", + remoteCwd, + timeoutMs: 30_000, + runner, + effectiveCapabilities: duplexCapabilities(true), + }; + + const bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-fail", + target, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: true, + duplexReadinessTimeoutMs: 400, + }); + try { + expect(bridge).not.toBeNull(); + expect(control.openCount).toBe(1); + // Fail closed: the file bridge serves after the bounded cleanup. + expect(bridge?.env.PAPERCLIP_API_BRIDGE_MODE).toBe("queue_v1"); + // The bounded cleanup left no live provider session. + expect(control.closeCount + control.stopCount).toBeGreaterThanOrEqual(1); + } finally { + await bridge?.stop(); + await api.close(); + } + }, 20000); + + it.each([ + { + name: "an attacker-owned numeric local port", + buildReady: (nonce: string, attackerPort: number) => + `{"version":1,"type":"ready","nonce":"${nonce}","port":${attackerPort}}\n`, + }, + { + name: "a channel-supplied host URL", + buildReady: (nonce: string, attackerPort: number) => + `{"version":1,"type":"ready","nonce":"${nonce}","address":"http://127.0.0.1:${attackerPort}"}\n`, + }, + ])( + "rejects a READY frame that carries $name and never sends the bridge token there", + async ({ buildReady }) => { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-addr-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + + // An endpoint an attacker controls. No request that carries the bridge + // token may reach it, because the host never derives the endpoint from a + // channel frame. + const attackerHits: string[] = []; + const attacker = createServer((req, res) => { + attackerHits.push(req.headers.authorization ?? ""); + res.writeHead(200, { "content-type": "application/json" }); + res.end("{}"); + }); + await new Promise((resolve, reject) => { + attacker.once("error", reject); + attacker.listen(0, "127.0.0.1", () => resolve()); + }); + const attackerAddress = attacker.address(); + if (!attackerAddress || typeof attackerAddress === "string") { + throw new Error("Expected the attacker server to listen on a TCP port."); + } + const attackerPort = attackerAddress.port; + + const api = await startRecordingApiServer(); + const { runner, control } = makeDuplexSelectionRunner((ctx) => + ctx.emitRaw(buildReady(ctx.nonce, attackerPort)), + ); + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "e2b", + remoteCwd, + timeoutMs: 30_000, + runner, + effectiveCapabilities: duplexCapabilities(true), + }; + + const bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-addr", + target, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: true, + duplexReadinessTimeoutMs: 400, + }); + try { + expect(bridge).not.toBeNull(); + // The address-bearing READY frame failed the strict schema, so the host + // fell closed to the file bridge and built no channel-supplied endpoint. + expect(bridge?.env.PAPERCLIP_API_BRIDGE_MODE).toBe("queue_v1"); + expect(bridge?.env.PAPERCLIP_API_URL).not.toContain(String(attackerPort)); + expect(control.closeCount + control.stopCount).toBeGreaterThanOrEqual(1); + // Give any stray forward a moment, then assert the attacker got nothing. + await new Promise((resolve) => setTimeout(resolve, 100)); + expect(attackerHits).toEqual([]); + } finally { + await bridge?.stop(); + await api.close(); + await new Promise((resolve) => attacker.close(() => resolve())); + } + }, + 20000, + ); + + it("answers an unlisted route 403 over the duplex path without forwarding it", async () => { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-403-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + const api = await startRecordingApiServer(); + const { runner, control } = makeDuplexSelectionRunner(); + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "e2b", + remoteCwd, + timeoutMs: 30_000, + runner, + effectiveCapabilities: duplexCapabilities(true), + }; + + const bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-403", + target, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: true, + }); + try { + expect(bridge?.env.PAPERCLIP_API_BRIDGE_MODE).toBe("duplex_v1"); + control.emitData( + encodeDuplexFrame({ + version: DUPLEX_FRAME_VERSION, + type: "request", + id: "req-forbidden", + method: "POST", + path: "/api/secret-admin-route", + query: "", + headers: { authorization: "Bearer bridge-token", "content-type": "application/json" }, + body: JSON.stringify({ escalate: true }), + }), + ); + await waitForCondition( + () => control.written.length >= 1, + "the broker to write a 403 response frame", + 4000, + ); + expect(control.written[0]).toMatchObject({ type: "response", id: "req-forbidden", status: 403 }); + // The route allowlist rejected the request, so it never forwarded. + expect(api.requests).toHaveLength(0); + } finally { + await bridge?.stop(); + await api.close(); + } + }, 20000); + + // One recording telemetry recorder. It captures every span, counter, and event + // the fixed duplex surface produces, so a test asserts the exact names, + // dimensions, and values. An optional `failEvery` flag makes every method throw, + // so a test proves a telemetry failure never breaks the request path. + function createRecordingDuplexRecorder(options: { failEvery?: boolean } = {}): { + recorder: DuplexTelemetryRecorder; + spans: DuplexTelemetrySpanRecord[]; + counters: DuplexTelemetryCounterRecord[]; + events: DuplexTelemetryEventRecord[]; + } { + const spans: DuplexTelemetrySpanRecord[] = []; + const counters: DuplexTelemetryCounterRecord[] = []; + const events: DuplexTelemetryEventRecord[] = []; + const recorder: DuplexTelemetryRecorder = { + recordSpan(record) { + if (options.failEvery) throw new Error("telemetry sink down"); + spans.push(record); + }, + incrementCounter(record) { + if (options.failEvery) throw new Error("telemetry sink down"); + counters.push(record); + }, + emitEvent(record) { + if (options.failEvery) throw new Error("telemetry sink down"); + events.push(record); + }, + }; + return { recorder, spans, counters, events }; + } + + // Every dimension key a record carries must be one of the fixed keys. The set is + // closed, so a new key never reaches a sink by accident. + function assertOnlyFixedDimensionKeys(dimensions: DuplexTelemetryDimensions | undefined): void { + expect(dimensions).toBeDefined(); + for (const key of Object.keys(dimensions ?? {})) { + expect(DUPLEX_DIMENSION_KEYS).toContain(key as (typeof DUPLEX_DIMENSION_KEYS)[number]); + } + } + + it("pins the exact fixed duplex dimension-key set", () => { + // The dimension-key set is closed. This test locks the exact contract, so a + // new key never reaches a sink without an explicit change here. + expect([...DUPLEX_DIMENSION_KEYS]).toEqual([ + "provider", + "transport", + "outcome", + "fallback_reason", + "loss_class", + "loss_reason", + ]); + }); + + it("records a duplex request span with latency and the fixed dimension keys", async () => { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-obs-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + const api = await startRecordingApiServer(); + const { runner, control } = makeDuplexSelectionRunner(); + const { recorder, spans, counters, events } = createRecordingDuplexRecorder(); + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "daytona", + remoteCwd, + timeoutMs: 30_000, + runner, + effectiveCapabilities: duplexCapabilities(true), + }; + + const bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-obs", + target, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: true, + duplexTelemetryRecorder: recorder, + }); + try { + expect(bridge?.env.PAPERCLIP_API_BRIDGE_MODE).toBe("duplex_v1"); + control.emitData( + encodeDuplexFrame({ + version: DUPLEX_FRAME_VERSION, + type: "request", + id: "req-obs", + method: "GET", + path: "/api/agents/me", + query: "", + headers: { authorization: "Bearer bridge-token" }, + body: "", + }), + ); + await waitForCondition( + () => control.written.length >= 1, + "the broker to write a duplex response frame", + 4000, + ); + + // The channel-open surface: the span, the counter, and the transport event. + const openSpan = spans.find((span) => span.name === DUPLEX_SPAN_CHANNEL_OPEN); + expect(openSpan).toBeDefined(); + expect(openSpan?.dimensions).toMatchObject({ provider: "daytona", transport: "duplex", outcome: "ok" }); + expect(counters.some((c) => c.metric === DUPLEX_COUNTER_CHANNEL_OPEN_TOTAL)).toBe(true); + expect( + events.some( + (e) => + e.name === DUPLEX_TRANSPORT_EVENT && + e.dimensions.transport === "duplex" && + e.dimensions.outcome === "ok", + ), + ).toBe(true); + + // The request span carries a numeric latency and only the fixed keys. + const requestSpan = spans.find((span) => span.name === DUPLEX_SPAN_REQUEST); + expect(requestSpan).toBeDefined(); + expect(typeof requestSpan?.latencyMs).toBe("number"); + expect(requestSpan?.latencyMs).toBeGreaterThanOrEqual(0); + expect(requestSpan?.dimensions).toMatchObject({ provider: "daytona", transport: "duplex", outcome: "ok" }); + assertOnlyFixedDimensionKeys(requestSpan?.dimensions); + } finally { + await bridge?.stop(); + await api.close(); + } + }, 20000); + + it("increments the fallback counter with an approved reason when the capability is absent", async () => { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-fb-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + const api = await startRecordingApiServer(); + const { runner } = makeDuplexSelectionRunner(); + const { recorder, counters, events } = createRecordingDuplexRecorder(); + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "daytona", + remoteCwd, + timeoutMs: 30_000, + runner, + effectiveCapabilities: duplexCapabilities(false), + }; + + const bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-fb", + target, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: true, + duplexTelemetryRecorder: recorder, + }); + try { + expect(bridge?.env.PAPERCLIP_API_BRIDGE_MODE).toBe("queue_v1"); + const fallback = counters.find((c) => c.metric === DUPLEX_COUNTER_FALLBACK_TOTAL); + expect(fallback).toBeDefined(); + const approvedReasons = [ + "gate_off", + "capability_absent", + "route_busy", + "entrypoint_sync_failed", + "broker_construction_failed", + "channel_open_failed", + "ready_invalid", + "ready_nonce_mismatch", + "ready_timeout", + "contaminated", + ]; + expect(approvedReasons).toContain(fallback?.dimensions.fallback_reason); + expect(fallback?.dimensions).toMatchObject({ transport: "file", outcome: "error" }); + assertOnlyFixedDimensionKeys(fallback?.dimensions); + // The transport event mirrors the fallback. + expect( + events.some( + (e) => e.name === DUPLEX_TRANSPORT_EVENT && e.dimensions.transport === "file", + ), + ).toBe(true); + } finally { + await bridge?.stop(); + await api.close(); + } + }, 20000); + + it.each([ + { + name: "a full process-scoped route ceiling", + error: new Error("worker route rejected: DUPLEX_CHANNEL_ROUTE_BUSY"), + expectedReason: "route_busy", + }, + { + name: "a generic channel-open failure", + error: new Error("provider channel open failed"), + expectedReason: "channel_open_failed", + }, + ])( + "names the open-failure stage $expectedReason and falls back to the file bridge on $name", + async ({ error, expectedReason }) => { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-stage-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + const api = await startRecordingApiServer(); + // A runner whose duplex open rejects. The host binds the caught error and + // names the exact open-failure stage. + const base = createLocalSandboxRunner(); + const openDuplexChannel = async (): Promise => { + throw error; + }; + const runner = { ...base, openDuplexChannel }; + const { recorder, spans, counters, events } = createRecordingDuplexRecorder(); + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "daytona", + remoteCwd, + timeoutMs: 30_000, + runner, + effectiveCapabilities: duplexCapabilities(true), + }; + + const bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-stage", + target, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: true, + duplexTelemetryRecorder: recorder, + }); + try { + // The channel never opened, so the host serves the file bridge. + expect(bridge?.env.PAPERCLIP_API_BRIDGE_MODE).toBe("queue_v1"); + // The channel-open span and the fallback counter name the exact stage. + const openSpan = spans.find( + (s) => s.name === DUPLEX_SPAN_CHANNEL_OPEN && s.dimensions.outcome === "error", + ); + expect(openSpan?.dimensions.fallback_reason).toBe(expectedReason); + const fallback = counters.find((c) => c.metric === DUPLEX_COUNTER_FALLBACK_TOTAL); + expect(fallback?.dimensions.fallback_reason).toBe(expectedReason); + assertOnlyFixedDimensionKeys(fallback?.dimensions); + expect( + events.some( + (e) => e.name === DUPLEX_TRANSPORT_EVENT && e.dimensions.transport === "file", + ), + ).toBe(true); + } finally { + await bridge?.stop(); + await api.close(); + } + }, + 20000, + ); + + it.each([ + { name: "before any dispatch", dispatchFirst: false, expectedClass: "pre_dispatch" }, + { name: "after a dispatch", dispatchFirst: true, expectedClass: "post_dispatch" }, + ])("increments the loss counter with the loss class $name", async ({ dispatchFirst, expectedClass }) => { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-loss-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + const api = await startRecordingApiServer(); + const { runner, control } = makeDuplexSelectionRunner(); + const { recorder, counters } = createRecordingDuplexRecorder(); + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "daytona", + remoteCwd, + timeoutMs: 30_000, + runner, + effectiveCapabilities: duplexCapabilities(true), + }; + + const bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-loss", + target, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: true, + duplexTelemetryRecorder: recorder, + }); + try { + expect(bridge?.env.PAPERCLIP_API_BRIDGE_MODE).toBe("duplex_v1"); + if (dispatchFirst) { + control.emitData( + encodeDuplexFrame({ + version: DUPLEX_FRAME_VERSION, + type: "request", + id: "req-loss", + method: "GET", + path: "/api/agents/me", + query: "", + headers: { authorization: "Bearer bridge-token" }, + body: "", + }), + ); + await waitForCondition( + () => control.written.length >= 1, + "the broker to write a duplex response frame", + 4000, + ); + } + // A malformed frame is a protocol failure. The broker records a terminal loss. + control.emitData("@@@ not a duplex frame @@@\n"); + await waitForCondition( + () => counters.some((c) => c.metric === DUPLEX_COUNTER_LOSS_TOTAL), + "the broker to record a loss counter", + 4000, + ); + const loss = counters.find((c) => c.metric === DUPLEX_COUNTER_LOSS_TOTAL); + expect(loss?.dimensions.loss_class).toBe(expectedClass); + expect(loss?.dimensions).toMatchObject({ transport: "duplex", outcome: "error" }); + assertOnlyFixedDimensionKeys(loss?.dimensions); + } finally { + await bridge?.stop(); + await api.close(); + } + }, 20000); + + it("keeps serving the request path when the telemetry recorder throws", async () => { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-guard-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + const api = await startRecordingApiServer(); + const { runner, control } = makeDuplexSelectionRunner(); + const { recorder } = createRecordingDuplexRecorder({ failEvery: true }); + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "daytona", + remoteCwd, + timeoutMs: 30_000, + runner, + effectiveCapabilities: duplexCapabilities(true), + }; + + const bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-guard", + target, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: true, + duplexTelemetryRecorder: recorder, + }); + try { + // The throwing recorder never blocked the duplex selection. + expect(bridge?.env.PAPERCLIP_API_BRIDGE_MODE).toBe("duplex_v1"); + control.emitData( + encodeDuplexFrame({ + version: DUPLEX_FRAME_VERSION, + type: "request", + id: "req-guard", + method: "GET", + path: "/api/agents/me", + query: "", + headers: { authorization: "Bearer bridge-token" }, + body: "", + }), + ); + await waitForCondition( + () => control.written.length >= 1, + "the broker to write a duplex response frame", + 4000, + ); + // The request path still delivered a real host response. + expect(control.written[0]).toMatchObject({ type: "response", id: "req-guard", status: 200 }); + expect(api.requests).toHaveLength(1); + } finally { + await bridge?.stop(); + await api.close(); + } + }, 20000); + + it("keeps sentinel route, query, body, tokens, and provider errors off the duplex telemetry and logs", async () => { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-redact-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + const api = await startRecordingApiServer(); + + const ROUTE_SENTINEL = "sentinelroute8f21"; + const QUERY_SENTINEL = "sentinelquery3d90"; + const BODY_SENTINEL = "sentinelbodya17c"; + const BRIDGE_TOKEN_SENTINEL = "sentinelbridgetok55e2"; + const AGENT_TOKEN_SENTINEL = "sentinelagenttoke91b4"; + const PROVIDER_ERROR_SENTINEL = "sentinelprovidererr7a3d"; + const sentinels = [ + ROUTE_SENTINEL, + QUERY_SENTINEL, + BODY_SENTINEL, + BRIDGE_TOKEN_SENTINEL, + AGENT_TOKEN_SENTINEL, + PROVIDER_ERROR_SENTINEL, + ]; + + // A runner whose channel throws a provider error on the response write, so the + // broker records a stream-failure loss carrying the sentinel message. + const base = createLocalSandboxRunner(); + const control = { emitData: (_chunk: string) => {} }; + const openDuplexChannel = async (openInput: { + command: readonly string[]; + }): Promise => { + const joined = openInput.command.join(" "); + const nonce = /PAPERCLIP_BRIDGE_NONCE='([^']*)'/.exec(joined)?.[1] ?? ""; + let dataListener: ((chunk: string) => void) | null = null; + const channel: CommandManagedDuplexChannel = { + write(_data: string): void { + // Every response write fails with a provider error carrying the sentinel. + throw new Error(`provider write failed: ${PROVIDER_ERROR_SENTINEL}`); + }, + onData(listener: (chunk: string) => void): void { + dataListener = listener; + control.emitData = (chunk) => dataListener?.(chunk); + setImmediate(() => dataListener?.(`${JSON.stringify({ version: 1, type: "ready", nonce })}\n`)); + }, + onExit(_listener: (exit: { exitCode: number | null }) => void): void {}, + stop(): void {}, + close(): Promise { + return Promise.resolve(); + }, + }; + return channel; + }; + const runner = { ...base, openDuplexChannel }; + + const logLines: string[] = []; + const { recorder, spans, counters, events } = createRecordingDuplexRecorder(); + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "daytona", + remoteCwd, + timeoutMs: 30_000, + runner, + effectiveCapabilities: duplexCapabilities(true), + }; + + const previousDebug = process.env.PAPERCLIP_BRIDGE_DEBUG; + process.env.PAPERCLIP_BRIDGE_DEBUG = "1"; + let bridge: Awaited> = null; + try { + bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-redact", + target, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: AGENT_TOKEN_SENTINEL, + hostApiUrl: api.origin, + enableSandboxDuplexBridge: true, + duplexTelemetryRecorder: recorder, + onLog: async (_stream, chunk) => { + logLines.push(chunk); + }, + }); + expect(bridge?.env.PAPERCLIP_API_BRIDGE_MODE).toBe("duplex_v1"); + + // A request that carries the sentinel route, query, body, and bridge token. + // The response write then fails with the sentinel provider error. + control.emitData( + encodeDuplexFrame({ + version: DUPLEX_FRAME_VERSION, + type: "request", + id: "req-redact", + method: "GET", + path: `/api/${ROUTE_SENTINEL}`, + query: `secret=${QUERY_SENTINEL}`, + headers: { authorization: `Bearer ${BRIDGE_TOKEN_SENTINEL}` }, + body: BODY_SENTINEL, + }), + ); + // Give the forward and the failing response write time to run and record a loss. + await waitForCondition( + () => counters.some((c) => c.metric === DUPLEX_COUNTER_LOSS_TOTAL), + "the broker to record a loss counter after the failed write", + 4000, + ); + + // Serialize every telemetry record and every log line, then assert that no + // sentinel reaches any of them on the duplex path. + const telemetryDump = JSON.stringify({ spans, counters, events }); + const logDump = logLines.join(""); + for (const sentinel of sentinels) { + expect(telemetryDump).not.toContain(sentinel); + expect(logDump).not.toContain(sentinel); + } + } finally { + if (previousDebug === undefined) delete process.env.PAPERCLIP_BRIDGE_DEBUG; + else process.env.PAPERCLIP_BRIDGE_DEBUG = previousDebug; + await bridge?.stop(); + await api.close(); + } + }, 20000); + + it("maps a sentinel provider key to the constant other across every sink", async () => { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-prov-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + const api = await startRecordingApiServer(); + const { runner, control } = makeDuplexSelectionRunner(); + const { recorder, spans, counters, events } = createRecordingDuplexRecorder(); + const PROVIDER_SENTINEL = "sentinel-plugin-provider-key-9c2a"; + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: PROVIDER_SENTINEL, + remoteCwd, + timeoutMs: 30_000, + runner, + effectiveCapabilities: duplexCapabilities(true), + }; + + const bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-prov", + target, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: true, + duplexTelemetryRecorder: recorder, + }); + try { + expect(bridge?.env.PAPERCLIP_API_BRIDGE_MODE).toBe("duplex_v1"); + control.emitData( + encodeDuplexFrame({ + version: DUPLEX_FRAME_VERSION, + type: "request", + id: "req-prov", + method: "GET", + path: "/api/agents/me", + query: "", + headers: { authorization: "Bearer bridge-token" }, + body: "", + }), + ); + await waitForCondition( + () => spans.some((s) => s.name === DUPLEX_SPAN_REQUEST), + "the broker to record a request span", + 4000, + ); + + // Every recorded provider dimension is the constant `other`, never the key. + const allDimensions = [ + ...spans.map((s) => s.dimensions), + ...counters.map((c) => c.dimensions), + ...events.map((e) => e.dimensions), + ]; + expect(allDimensions.length).toBeGreaterThan(0); + for (const dimensions of allDimensions) { + expect(dimensions.provider).toBe("other"); + } + // The raw key reaches no sink. + const telemetryDump = JSON.stringify({ spans, counters, events }); + expect(telemetryDump).not.toContain(PROVIDER_SENTINEL); + } finally { + await bridge?.stop(); + await api.close(); + } + }, 20000); + + it("caps the pre-READY readiness buffer and falls back with a contaminated reason", async () => { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-cap-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + const api = await startRecordingApiServer(); + // The fake gateway sends a large pre-READY blob with no newline, then one + // more blob after the gate settles. The gate must cap the buffer, finish with + // protocol contamination, and drop the later blob. The blob is larger than + // the codec frame-size bound, so it passes the readiness buffer cap. + const oversizedBlob = "x".repeat(DEFAULT_MAX_DUPLEX_FRAME_BYTES * 2); + const { runner, control } = makeDuplexSelectionRunner((ctx) => { + ctx.emitRaw(oversizedBlob); + ctx.emitRaw(oversizedBlob); + }); + const { recorder, counters } = createRecordingDuplexRecorder(); + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "daytona", + remoteCwd, + timeoutMs: 30_000, + runner, + effectiveCapabilities: duplexCapabilities(true), + }; + + const bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-cap", + target, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: true, + // A long readiness timeout, so the buffer cap, not the timeout, drives the + // failure. + duplexReadinessTimeoutMs: 5_000, + duplexTelemetryRecorder: recorder, + }); + try { + expect(bridge).not.toBeNull(); + expect(control.openCount).toBe(1); + // The cap drove the failure, so the file bridge serves after the bounded cleanup. + expect(bridge?.env.PAPERCLIP_API_BRIDGE_MODE).toBe("queue_v1"); + const fallback = counters.find((c) => c.metric === DUPLEX_COUNTER_FALLBACK_TOTAL); + expect(fallback?.dimensions.fallback_reason).toBe("contaminated"); + // The bounded cleanup left no live provider session. + expect(control.closeCount + control.stopCount).toBeGreaterThanOrEqual(1); + } finally { + await bridge?.stop(); + await api.close(); + } + }, 20000); + + it("caps the pre-READY buffer under many small newline-less chunks", async () => { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-cap-small-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + const api = await startRecordingApiServer(); + // An adversarial provider controls the chunk size. It sends many small + // newline-less chunks that together pass the cap. The gate must scan each + // chunk in O(1) of the buffer length, so the pre-READY window stays bounded. + // The gate caps the buffer, finishes with protocol contamination, and falls + // back to the file bridge. + const readinessBufferCapBytes = DEFAULT_MAX_DUPLEX_FRAME_BYTES + 4_096; + const smallChunk = "x".repeat(64); + const chunkCount = Math.ceil(readinessBufferCapBytes / smallChunk.length) + 1; + const { runner, control } = makeDuplexSelectionRunner((ctx) => { + for (let i = 0; i < chunkCount; i += 1) { + ctx.emitRaw(smallChunk); + } + }); + const { recorder, counters } = createRecordingDuplexRecorder(); + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "daytona", + remoteCwd, + timeoutMs: 30_000, + runner, + effectiveCapabilities: duplexCapabilities(true), + }; + + const bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-cap-small", + target, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: true, + // A long readiness timeout, so the buffer cap, not the timeout, drives the + // failure. + duplexReadinessTimeoutMs: 5_000, + duplexTelemetryRecorder: recorder, + }); + try { + expect(bridge).not.toBeNull(); + expect(control.openCount).toBe(1); + // The cap drove the failure, so the file bridge serves after the bounded cleanup. + expect(bridge?.env.PAPERCLIP_API_BRIDGE_MODE).toBe("queue_v1"); + const fallback = counters.find((c) => c.metric === DUPLEX_COUNTER_FALLBACK_TOTAL); + expect(fallback?.dimensions.fallback_reason).toBe("contaminated"); + expect(control.closeCount + control.stopCount).toBeGreaterThanOrEqual(1); + } finally { + await bridge?.stop(); + await api.close(); + } + }, 20000); + + it("bounds the pre-READY newline-scan work by the bytes received", async () => { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-scan-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + const api = await startRecordingApiServer(); + // An adversarial provider sends many small newline-less chunks before the + // cap fires. Each chunk must scan only the new bytes, not the whole buffer, + // so the total newline-scan work stays linear in the bytes received. A + // per-chunk full rescan makes the work quadratic. + const readinessBufferCapBytes = DEFAULT_MAX_DUPLEX_FRAME_BYTES + 4_096; + const smallChunk = "x".repeat(64); + const chunkCount = Math.ceil(readinessBufferCapBytes / smallChunk.length) + 1; + const totalBytes = chunkCount * smallChunk.length; + const { runner, control } = makeDuplexSelectionRunner((ctx) => { + for (let i = 0; i < chunkCount; i += 1) { + ctx.emitRaw(smallChunk); + } + }); + const { recorder } = createRecordingDuplexRecorder(); + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "daytona", + remoteCwd, + timeoutMs: 30_000, + runner, + effectiveCapabilities: duplexCapabilities(true), + }; + + __duplexReadinessTesting.resetNewlineScanUnits(); + const bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-scan-bound", + target, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: true, + duplexReadinessTimeoutMs: 5_000, + duplexTelemetryRecorder: recorder, + }); + try { + expect(bridge).not.toBeNull(); + expect(control.openCount).toBe(1); + const scanUnits = __duplexReadinessTesting.readNewlineScanUnits(); + // Linear scan work reads each byte one time, so the count stays near + // totalBytes. A per-chunk full rescan is quadratic (about + // totalBytes^2 / (2 * chunkSize)), far above this bound. + expect(scanUnits).toBeLessThanOrEqual(4 * totalBytes); + } finally { + await bridge?.stop(); + await api.close(); + } + }, 20000); + + it("bounds the pre-READY skip scan work by the bytes received", async () => { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-blank-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + const api = await startRecordingApiServer(); + // An adversarial provider sends one pre-READY chunk of many blank lines and a + // noise line, then a valid READY frame. The gate must skip each blank line and + // the noise line in O(1), so the total newline-scan work stays linear in the + // bytes received. A per-line full rescan or a per-line buffer copy makes the + // work quadratic. The READY frame then settles the gate ready. + const blankLineCount = 15_000; + const noisePrefix = "\n".repeat(blankLineCount) + "a non-frame echo line\n"; + const { runner, control } = makeDuplexSelectionRunner((ctx) => { + ctx.emitRaw(noisePrefix); + ctx.emitFrame({ version: 1, type: "ready", nonce: ctx.nonce }); + }); + const readyLine = '{"version":1,"type":"ready","nonce":""}\n'; + const totalBytes = noisePrefix.length + readyLine.length; + const { recorder, counters } = createRecordingDuplexRecorder(); + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "daytona", + remoteCwd, + timeoutMs: 30_000, + runner, + effectiveCapabilities: duplexCapabilities(true), + }; + + __duplexReadinessTesting.resetNewlineScanUnits(); + const bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-blank-scan", + target, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: true, + duplexReadinessTimeoutMs: 5_000, + duplexTelemetryRecorder: recorder, + }); + try { + expect(bridge).not.toBeNull(); + expect(control.openCount).toBe(1); + const scanUnits = __duplexReadinessTesting.readNewlineScanUnits(); + // Incremental skip handling reads each byte one time, so the count stays + // near totalBytes. A per-line full rescan is quadratic (about + // blankLineCount^2 / 2), far above this bound. + expect(scanUnits).toBeLessThanOrEqual(4 * totalBytes); + // The gate skipped the noise and accepted the READY frame, so the duplex + // transport serves and no fallback fired. + expect(bridge?.env.PAPERCLIP_API_BRIDGE_MODE).toBe("duplex_v1"); + const fallback = counters.find((c) => c.metric === DUPLEX_COUNTER_FALLBACK_TOTAL); + expect(fallback).toBeUndefined(); + } finally { + await bridge?.stop(); + await api.close(); + } + }, 20000); + + it("skips pre-READY noise lines, then accepts a valid READY frame and serves the duplex transport", async () => { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-noise-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + const api = await startRecordingApiServer(); + // A PTY channel echoes the launch wrapper line before it sets raw mode, so the + // first line the host reads is a non-frame echo, not the READY frame. The gate + // must skip the echo line and a partial-JSON line, then accept the READY frame. + const { runner, control } = makeDuplexSelectionRunner((ctx) => { + ctx.emitRaw("sh -c exec env PAPERCLIP_BRIDGE_NONCE=... node gateway.mjs\n"); + ctx.emitRaw('{"version":1,"type":"ready"}\n'); + ctx.emitFrame({ version: 1, type: "ready", nonce: ctx.nonce }); + }); + const { recorder, counters } = createRecordingDuplexRecorder(); + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "daytona", + remoteCwd, + timeoutMs: 30_000, + runner, + effectiveCapabilities: duplexCapabilities(true), + }; + + const bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-noise-ready", + target, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: true, + duplexReadinessTimeoutMs: 5_000, + duplexTelemetryRecorder: recorder, + }); + try { + expect(bridge).not.toBeNull(); + expect(control.openCount).toBe(1); + // The gate skipped the echo and the partial frame, then accepted the READY + // frame, so the duplex transport serves and no fallback fired. + expect(bridge?.env.PAPERCLIP_API_BRIDGE_MODE).toBe("duplex_v1"); + const fallback = counters.find((c) => c.metric === DUPLEX_COUNTER_FALLBACK_TOTAL); + expect(fallback).toBeUndefined(); + } finally { + await bridge?.stop(); + await api.close(); + } + }, 20000); + + it("settles a wrong-nonce READY frame as a nonce mismatch, even after a noise line", async () => { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-noise-nonce-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + const api = await startRecordingApiServer(); + // The gate skips the echo line, then reads a READY frame that decodes cleanly + // but carries a wrong nonce. A wrong-nonce READY authenticates as a failure, + // not as noise, so the gate settles the handshake failed and falls back with + // the `ready_nonce_mismatch` reason. + const { runner, control } = makeDuplexSelectionRunner((ctx) => { + ctx.emitRaw("a non-frame echo line\n"); + ctx.emitFrame({ version: 1, type: "ready", nonce: "00000000000000000000000000000000" }); + }); + const { recorder, counters } = createRecordingDuplexRecorder(); + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "daytona", + remoteCwd, + timeoutMs: 30_000, + runner, + effectiveCapabilities: duplexCapabilities(true), + }; + + const bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-noise-nonce", + target, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: true, + duplexReadinessTimeoutMs: 5_000, + duplexTelemetryRecorder: recorder, + }); + try { + expect(bridge).not.toBeNull(); + expect(control.openCount).toBe(1); + // The wrong nonce failed the handshake, so the file bridge serves. + expect(bridge?.env.PAPERCLIP_API_BRIDGE_MODE).toBe("queue_v1"); + const fallback = counters.find((c) => c.metric === DUPLEX_COUNTER_FALLBACK_TOTAL); + expect(fallback?.dimensions.fallback_reason).toBe("ready_nonce_mismatch"); + // The bounded cleanup left no live provider session. + expect(control.closeCount + control.stopCount).toBeGreaterThanOrEqual(1); + } finally { + await bridge?.stop(); + await api.close(); + } + }, 20000); + + it("enforces the buffer cap on an over-cap blank prefix before it accepts a valid READY frame", async () => { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-capbypass-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + const api = await startRecordingApiServer(); + // The cap must bound every pre-READY path, including a skipped blank line. An + // adversarial provider sends one chunk: an over-cap blank prefix followed by a + // valid nonce-bound READY frame. The gate must reject on the cap before READY + // acceptance, so it falls back to the file bridge with the contaminated reason + // and the bounded cleanup. Without the per-skip cap check the blank prefix + // reaches the valid READY line in the same chunk, and the duplex transport + // opens, which is the cap bypass. A single chunk keeps the trailing READY + // newline in the buffer, so the no-newline cap check never fires here; only the + // per-skip cap check stops the bypass. + const readinessBufferCapBytes = DEFAULT_MAX_DUPLEX_FRAME_BYTES + 4_096; + const { runner, control } = makeDuplexSelectionRunner((ctx) => { + const readyLine = `${JSON.stringify({ version: 1, type: "ready", nonce: ctx.nonce })}\n`; + ctx.emitRaw("\n".repeat(readinessBufferCapBytes + 1) + readyLine); + }); + const { recorder, counters } = createRecordingDuplexRecorder(); + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "daytona", + remoteCwd, + timeoutMs: 30_000, + runner, + effectiveCapabilities: duplexCapabilities(true), + }; + + const bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-cap-bypass", + target, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: true, + // A long readiness timeout, so the cap, not the timeout, drives the failure. + duplexReadinessTimeoutMs: 5_000, + duplexTelemetryRecorder: recorder, + }); + try { + expect(bridge).not.toBeNull(); + expect(control.openCount).toBe(1); + // The cap drove the failure before READY acceptance, so the file bridge serves. + expect(bridge?.env.PAPERCLIP_API_BRIDGE_MODE).toBe("queue_v1"); + const fallback = counters.find((c) => c.metric === DUPLEX_COUNTER_FALLBACK_TOTAL); + expect(fallback?.dimensions.fallback_reason).toBe("contaminated"); + // The bounded cleanup left no live provider session. + expect(control.closeCount + control.stopCount).toBeGreaterThanOrEqual(1); + } finally { + await bridge?.stop(); + await api.close(); + } + }, 20000); + + it("records the channel-open span with the fallback_reason dimension on the fallback path", async () => { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-openspan-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + const api = await startRecordingApiServer(); + // A wrong-nonce READY frame fails the handshake, so the gate falls back. The + // channel-open span records the failed attempt on the duplex transport and now + // carries the closed `fallback_reason` dimension, so a reader can group the + // failed opens by reason. + const { runner } = makeDuplexSelectionRunner((ctx) => + ctx.emitFrame({ version: 1, type: "ready", nonce: "00000000000000000000000000000000" }), + ); + const { recorder, spans } = createRecordingDuplexRecorder(); + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "daytona", + remoteCwd, + timeoutMs: 30_000, + runner, + effectiveCapabilities: duplexCapabilities(true), + }; + + const bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-open-span", + target, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: true, + duplexReadinessTimeoutMs: 5_000, + duplexTelemetryRecorder: recorder, + }); + try { + expect(bridge?.env.PAPERCLIP_API_BRIDGE_MODE).toBe("queue_v1"); + const openSpan = spans.find((span) => span.name === DUPLEX_SPAN_CHANNEL_OPEN); + expect(openSpan).toBeDefined(); + expect(openSpan?.dimensions).toMatchObject({ + provider: "daytona", + transport: "duplex", + outcome: "error", + fallback_reason: "ready_nonce_mismatch", + }); + // The span carries only closed dimension keys. + assertOnlyFixedDimensionKeys(openSpan?.dimensions); + } finally { + await bridge?.stop(); + await api.close(); + } + }, 20000); + + it("drops a frame header outside the allowlist on the host duplex forward path", async () => { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-hdr-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + const api = await startRecordingApiServer(); + const { runner, control } = makeDuplexSelectionRunner(); + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "e2b", + remoteCwd, + timeoutMs: 30_000, + runner, + effectiveCapabilities: duplexCapabilities(true), + }; + + const bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-hdr", + target, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: true, + }); + try { + expect(bridge?.env.PAPERCLIP_API_BRIDGE_MODE).toBe("duplex_v1"); + control.emitData( + encodeDuplexFrame({ + version: DUPLEX_FRAME_VERSION, + type: "request", + id: "req-hdr", + method: "GET", + path: "/api/agents/me", + query: "", + headers: { + // An allowlisted header the host must keep. + accept: "application/json", + // A header outside the allowlist the host must drop. + "x-injected-header": "attacker", + // A sandbox-supplied auth header the host must replace with the real token. + authorization: "Bearer bridge-token", + }, + body: "", + }), + ); + await waitForCondition( + () => api.requests.length >= 1, + "the broker to forward the duplex request", + 4000, + ); + const forwarded = api.requests[0]; + // The allowlisted header reaches the host. + expect(forwarded.headers.accept).toBe("application/json"); + // The header outside the allowlist never reaches the authenticated fetch. + expect(forwarded.headers["x-injected-header"]).toBeUndefined(); + // The host applied the real token and the run id in place of the frame values. + expect(forwarded.auth).toBe("Bearer real-run-jwt"); + expect(forwarded.runId).toBe("run-hdr"); + } finally { + await bridge?.stop(); + await api.close(); + } + }, 20000); + + it("selects the duplex transport for a large forward budget and starts the broker", async () => { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-budget-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + const api = await startRecordingApiServer(); + const { runner, control } = makeDuplexSelectionRunner(); + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "e2b", + remoteCwd, + timeoutMs: 30_000, + runner, + effectiveCapabilities: duplexCapabilities(true), + }; + + // A forward budget past the default response budget (32 s). Before the budget + // derivation, this made the nested budget assertion throw and leak the channel. + const bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-budget", + target, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: true, + forwardTimeoutMs: 60_000, + }); + try { + // The broker started with derived nested budgets, so the duplex transport serves. + expect(bridge?.env.PAPERCLIP_API_BRIDGE_MODE).toBe("duplex_v1"); + control.emitData( + encodeDuplexFrame({ + version: DUPLEX_FRAME_VERSION, + type: "request", + id: "req-budget", + method: "GET", + path: "/api/agents/me", + query: "", + headers: { authorization: "Bearer bridge-token" }, + body: "", + }), + ); + await waitForCondition( + () => control.written.length >= 1, + "the broker to write a duplex response frame", + 4000, + ); + expect(control.written[0]).toMatchObject({ type: "response", id: "req-budget", status: 200 }); + } finally { + await bridge?.stop(); + await api.close(); + } + }, 20000); + + it("falls back to the file bridge when the broker construction throws", async () => { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-ctor-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + const api = await startRecordingApiServer(); + const { runner, control } = makeDuplexSelectionRunner(); + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "e2b", + remoteCwd, + timeoutMs: 30_000, + runner, + effectiveCapabilities: duplexCapabilities(true), + }; + + // A non-finite forward budget makes the nested budget assertion throw at + // broker construction. Readiness passes first, so the guarded region must + // close the channel and select the file bridge instead of leaking it. + const bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-ctor", + target, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: true, + forwardTimeoutMs: Number.NaN, + }); + try { + expect(bridge).not.toBeNull(); + expect(control.openCount).toBe(1); + // Readiness passed, then the broker construction threw; the file bridge serves. + expect(bridge?.env.PAPERCLIP_API_BRIDGE_MODE).toBe("queue_v1"); + // The bounded cleanup left no live provider session. + expect(control.closeCount + control.stopCount).toBeGreaterThanOrEqual(1); + } finally { + await bridge?.stop(); + await api.close(); + } + }, 20000); + + // --------------------------------------------------------------------------- + // Real-PTY replay. + // + // The earlier PTY-echo defect shipped because a fake PTY does not echo the way + // a real terminal does. These cases replay byte sequences captured from an + // actual `pty.fork()` + bash session driven through the same launch wrapper the + // Daytona plugin builds, so the gate is exercised against terminal output + // rather than against synthetic frames. + // --------------------------------------------------------------------------- + + async function runReadinessReplay(emit: (ctx: DuplexOpenContext) => void) { + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-pty-replay-")); + cleanupDirs.push(rootDir); + const remoteCwd = path.join(rootDir, "workspace"); + await mkdir(remoteCwd, { recursive: true }); + const api = await startRecordingApiServer(); + const { runner, control } = makeDuplexSelectionRunner(emit); + const target: AdapterSandboxExecutionTarget = { + kind: "remote", + transport: "sandbox", + providerKey: "daytona", + environmentId: "env-1", + leaseId: "lease-1", + remoteCwd, + timeoutMs: 30_000, + runner, + effectiveCapabilities: duplexCapabilities(true), + }; + const bridge = await startAdapterExecutionTargetPaperclipBridge({ + runId: "run-pty-replay", + target, + runtimeRootDir: path.join(remoteCwd, ".paperclip-runtime", "codex"), + adapterKey: "codex", + hostApiToken: "real-run-jwt", + hostApiUrl: api.origin, + enableSandboxDuplexBridge: true, + duplexReadinessTimeoutMs: 2_000, + }); + const mode = bridge?.env.PAPERCLIP_API_BRIDGE_MODE; + await bridge?.stop(); + await api.close(); + return { mode, control }; + } + + // Case 1: the shape observed on a real Daytona PTY. bash echoes its prompt and + // the wrapper line, terminated by CRLF, then the gateway's READY frame follows + // on its own clean line. + it("PTY replay: accepts READY after an echoed prompt and wrapper line", async () => { + const { mode } = await runReadinessReplay((ctx) => { + ctx.emitRaw( + "daytona@212487a7f3c9:~$ exec 2>'/tmp/paperclip-duplex-x.log'; stty raw -echo; " + + "exec 'bash' '-c' 'exec env PAPERCLIP_BRIDGE_NONCE=" + ctx.nonce + " node gateway.mjs'\r\n", + ); + ctx.emitRaw('{"version":1,"type":"ready","nonce":"' + ctx.nonce + '"}\n'); + }); + expect(mode).toBe("duplex_v1"); + }, 20000); + + // Case 2: captured from a local pty.fork() + bash on a host whose bash enables + // bracketed paste. The disable sequence and a bare CR land immediately before + // the READY frame, on the same line with no newline between them, so the whole + // line does not decode. The Daytona image in use today does not do this; another + // image or another provider can, and the gate must not depend on it. + it("PTY replay: accepts READY prefixed by a bracketed-paste disable sequence", async () => { + const { mode } = await runReadinessReplay((ctx) => { + ctx.emitRaw( + "\x1b[?2004h\x1b]0;user@host: /srv\x07user@host:/srv$ exec 2>'/tmp/d.log'; " + + "stty raw -echo; exec 'bash' '-c' 'exec env node gateway.mjs'\r\n", + ); + // No newline between the escape sequence and the frame: same line. + ctx.emitRaw('\x1b[?2004l\r{"version":1,"type":"ready","nonce":"' + ctx.nonce + '"}\n'); + }); + expect(mode).toBe("duplex_v1"); + }, 20000); + + // Case 3: the READY frame split across two chunk deliveries. + it("PTY replay: accepts READY split across chunk boundaries", async () => { + const { mode } = await runReadinessReplay((ctx) => { + ctx.emitRaw("prompt$ wrapper-line\r\n"); + const frame = '{"version":1,"type":"ready","nonce":"' + ctx.nonce + '"}\n'; + ctx.emitRaw(frame.slice(0, 12)); + ctx.emitRaw(frame.slice(12)); + }); + expect(mode).toBe("duplex_v1"); + }, 20000); + + // Case 4: a multibyte character split across two chunks in the pre-READY noise. + it("PTY replay: accepts READY when a multibyte char splits across chunks", async () => { + const { mode } = await runReadinessReplay((ctx) => { + const noise = Buffer.from("prompt ✓ done\r\n", "utf8"); + ctx.emitRaw(noise.slice(0, 8).toString("utf8")); + ctx.emitRaw(noise.slice(8).toString("utf8")); + ctx.emitRaw('{"version":1,"type":"ready","nonce":"' + ctx.nonce + '"}\n'); + }); + expect(mode).toBe("duplex_v1"); + }, 20000); + + // Case 5: a flood of short noise lines must fail closed at the cap rather than + // scanning forever. This is the hole the cursor-based scan closes. + it("PTY replay: fails closed on a pre-READY noise flood", async () => { + const { mode } = await runReadinessReplay((ctx) => { + const line = "x".repeat(64) + "\n"; + for (let i = 0; i < 80_000; i += 1) ctx.emitRaw(line); + ctx.emitRaw('{"version":1,"type":"ready","nonce":"' + ctx.nonce + '"}\n'); + }); + expect(mode).toBe("queue_v1"); + }, 30000); + }); // One decoded stdout frame from the generated duplex gateway. The gateway writes @@ -2766,7 +4827,7 @@ interface DecodedGatewayFrame { query?: string; headers?: Record; body?: string; - address?: string; + nonce?: string; __unparsed?: string; } @@ -2825,6 +4886,44 @@ describe("sandbox duplex gateway", () => { } }); + it("launches the gateway with `exec env`, so a POSIX shell runs the environment-assignment prefix", async () => { + // A POSIX shell accepts an environment-assignment prefix only on a plain + // command, never on `exec`. The form `exec NAME=value command` exits with + // status 127, so the gateway never starts. The launch argv must use + // `exec env NAME=value command`. This test runs the generated script in a real + // `sh` against a stub entrypoint that echoes one launch env var, so it fails on + // the old `exec NAME=value` form and passes on the `exec env` form. + const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-launch-")); + duplexCleanupDirs.push(rootDir); + const stub = path.join(rootDir, "stub-entrypoint.sh"); + // The stub stands in for the node gateway. It prints a READY frame that echoes + // the launch nonce, so the test proves the environment assignment reached the + // process the launch replaced the shell with. + await writeFile( + stub, + '#!/bin/sh\nprintf \'{"version":1,"type":"ready","nonce":"%s"}\\n\' "$PAPERCLIP_BRIDGE_NONCE"\n', + "utf8", + ); + const argv = buildDuplexGatewayLaunchArgv({ + shellCommand: "sh", + remoteEntrypoint: stub, + // Run the stub with `sh`, so the test needs no node runtime. The launch form + // is `exec env NAME=value 'sh' ''`, which exercises the exec-env fix. + nodeCommand: "sh", + env: { PAPERCLIP_BRIDGE_NONCE: "abc123def456", PAPERCLIP_BRIDGE_PORT: "40404" }, + }); + const [shell, ...shellArgs] = argv; + const { stdout } = await execFileAsync(shell, shellArgs, { encoding: "utf8" }); + const decoded = decodeDuplexLine(stdout.trim()); + expect(decoded.ok).toBe(true); + if (decoded.ok) { + expect(decoded.frame.type).toBe("ready"); + if (decoded.frame.type === "ready") { + expect(decoded.frame.nonce).toBe("abc123def456"); + } + } + }); + interface DuplexGatewayHandle { baseUrl: string; frames: DecodedGatewayFrame[]; @@ -2840,22 +4939,50 @@ describe("sandbox duplex gateway", () => { stop: () => Promise; } + // Reserve a free loopback port. The host assigns a positive port to the duplex + // gateway; the gateway binds exactly that port. The test opens an ephemeral + // listener, reads its port, then closes it, so the number is very likely free + // when the gateway binds it a moment later. + async function reserveLoopbackPort(): Promise { + return new Promise((resolve, reject) => { + const probe = net.createServer(); + probe.once("error", reject); + probe.listen(0, "127.0.0.1", () => { + const address = probe.address(); + if (!address || typeof address === "string") { + probe.close(() => reject(new Error("Could not reserve a loopback port."))); + return; + } + const reserved = address.port; + probe.close(() => resolve(reserved)); + }); + }); + } + // Start the generated gateway `.mjs` in duplex mode as a real child process. // The test writes response frames to the child stdin and reads request frames // from the child stdout, so it stands in for the host side of the channel. - async function startDuplexGateway(env: Record): Promise { + // The host assigns the port and the nonce; the test builds the base URL from + // the assigned port, never from the READY frame. + async function startDuplexGateway( + env: Record, + options: { port?: number; nonce?: string } = {}, + ): Promise { const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-duplex-gateway-")); duplexCleanupDirs.push(rootDir); const entrypoint = path.join(rootDir, "gateway.mjs"); await writeFile(entrypoint, getSandboxCallbackBridgeServerSource(), "utf8"); + const assignedPort = options.port ?? (await reserveLoopbackPort()); + const nonce = options.nonce ?? "d0c1b2a3e4f5061728394a5b6c7d8e9f"; const child = spawn(process.execPath, [entrypoint], { stdio: ["pipe", "pipe", "pipe"], env: { ...process.env, PAPERCLIP_API_BRIDGE_MODE: "duplex_v1", PAPERCLIP_BRIDGE_HOST: "127.0.0.1", - PAPERCLIP_BRIDGE_PORT: "0", + PAPERCLIP_BRIDGE_PORT: String(assignedPort), + PAPERCLIP_BRIDGE_NONCE: nonce, ...env, }, }); @@ -2948,7 +5075,11 @@ describe("sandbox duplex gateway", () => { }; const ready = await waitForFrame((frame) => frame.type === "ready"); - handle.baseUrl = String(ready.address); + // READY carries the echoed nonce and no address data. The host builds the + // origin from the port it assigned, never from the frame. + expect(ready.nonce).toBe(nonce); + expect((ready as Record).address).toBeUndefined(); + handle.baseUrl = `http://127.0.0.1:${assignedPort}`; return handle; } @@ -3275,3 +5406,462 @@ describe("sandbox duplex gateway", () => { await gateway.stop(); }, 20000); }); + +/** + * A scripted fake duplex channel. The test drives the read path with + * `emitData`/`emitExit`, and reads the frames the broker wrote through `written`. + * The `writeError` and `closeBehavior` hooks let a test force a stream failure + * and a close timeout. + */ +function createFakeDuplexChannel(): { + channel: CommandManagedDuplexChannel; + emitData: (chunk: string) => void; + emitExit: (exit: { exitCode: number | null }) => void; + written: DuplexResponseFrame[]; + writtenTypes: string[]; + stopped: () => number; + setWriteError: (error: Error | null) => void; + setCloseBehavior: (behavior: "resolve" | "hang" | "reject") => void; +} { + let dataListener: ((chunk: string) => void) | null = null; + let exitListener: ((exit: { exitCode: number | null }) => void) | null = null; + let writeError: Error | null = null; + let closeBehavior: "resolve" | "hang" | "reject" = "resolve"; + let stopCount = 0; + const written: DuplexResponseFrame[] = []; + const writtenTypes: string[] = []; + + const channel: CommandManagedDuplexChannel = { + write(data: string): void { + if (writeError) throw writeError; + const decoded = decodeDuplexLine(data.replace(/\n$/, "")); + if (decoded.ok) { + writtenTypes.push(decoded.frame.type); + if (decoded.frame.type === "response") written.push(decoded.frame); + } + }, + onData(listener: (chunk: string) => void): void { + dataListener = listener; + }, + onExit(listener: (exit: { exitCode: number | null }) => void): void { + exitListener = listener; + }, + stop(): void { + stopCount += 1; + }, + close(): Promise { + if (closeBehavior === "resolve") return Promise.resolve(); + if (closeBehavior === "reject") return Promise.reject(new Error("close rejected")); + return new Promise(() => {}); + }, + }; + + return { + channel, + emitData: (chunk: string) => dataListener?.(chunk), + emitExit: (exit: { exitCode: number | null }) => exitListener?.(exit), + written, + writtenTypes, + stopped: () => stopCount, + setWriteError: (error: Error | null) => { + writeError = error; + }, + setCloseBehavior: (behavior: "resolve" | "hang" | "reject") => { + closeBehavior = behavior; + }, + }; +} + +/** Build one request frame line the fake channel can emit. */ +function requestFrameLine(overrides: Partial & { id: string }): string { + const frame: DuplexRequestFrame = { + version: DUPLEX_FRAME_VERSION, + type: "request", + id: overrides.id, + method: overrides.method ?? "POST", + path: overrides.path ?? "/api/issues/PAP-1/comments", + query: overrides.query ?? "", + headers: overrides.headers ?? { authorization: "Bearer bridge-token" }, + body: overrides.body ?? JSON.stringify({ body: "hello" }), + }; + return encodeDuplexFrame(frame); +} + +/** Wait for the pending microtasks and macrotasks to settle. */ +function flushMacrotasks(): Promise { + return new Promise((resolve) => setImmediate(resolve)); +} + +describe("createDuplexBridgeBroker", () => { + it("forwards a decoded request to the handler and writes the handler response back", async () => { + const fake = createFakeDuplexChannel(); + const received: DuplexRequestFrame[] = []; + // The forward handler owns the token replacement and the run attribution. The + // broker passes the decoded request straight through, so the sandbox request + // still carries only the bridge token here, and the handler applies the real + // token and the signed run identifier on the existing forward path. + const broker = createDuplexBridgeBroker({ + channel: fake.channel, + forwardRequest: async (request): Promise => { + received.push(request); + expect(request.headers.authorization).toBe("Bearer bridge-token"); + expect(request.headers["x-paperclip-run-id"]).toBeUndefined(); + return { + status: 201, + headers: { "content-type": "application/json" }, + body: JSON.stringify({ ok: true }), + }; + }, + }); + + broker.start(); + fake.emitData(requestFrameLine({ id: "req-1" })); + await flushMacrotasks(); + + expect(received).toHaveLength(1); + expect(received[0].id).toBe("req-1"); + expect(fake.written).toHaveLength(1); + expect(fake.written[0]).toMatchObject({ + type: "response", + id: "req-1", + status: 201, + outcome: "completed", + }); + expect(JSON.parse(fake.written[0].body)).toEqual({ ok: true }); + + await broker.close(); + }); + + it("moves through opening, open, closing, closed in order", async () => { + const fake = createFakeDuplexChannel(); + const states: DuplexBrokerState[] = []; + const broker = createDuplexBridgeBroker({ + channel: fake.channel, + forwardRequest: async () => ({ status: 200 }), + onStateChange: (state) => states.push(state), + }); + + expect(broker.state).toBe("opening"); + broker.start(); + expect(broker.state).toBe("open"); + await broker.close(); + expect(broker.state).toBe("closed"); + expect(states).toEqual(["open", "closing", "closed"]); + expect(fake.writtenTypes).toContain("close"); + }); + + it.each([ + { + name: "channel exit", + reason: "channel_exit" as const, + trigger: (fake: ReturnType) => + fake.emitExit({ exitCode: 1 }), + }, + { + name: "protocol failure", + reason: "protocol_failure" as const, + trigger: (fake: ReturnType) => + fake.emitData("this is not json\n"), + }, + { + name: "stream failure", + reason: "stream_failure" as const, + trigger: (fake: ReturnType) => { + fake.setWriteError(new Error("broken pipe")); + fake.emitData(requestFrameLine({ id: "req-stream" })); + }, + }, + ])("enters lost on $name", async ({ reason, trigger }) => { + const fake = createFakeDuplexChannel(); + const losses: DuplexBrokerLossRecord[] = []; + const broker = createDuplexBridgeBroker({ + channel: fake.channel, + forwardRequest: async () => ({ status: 200 }), + onLoss: (record) => losses.push(record), + }); + broker.start(); + + trigger(fake); + await flushMacrotasks(); + + expect(broker.state).toBe("lost"); + expect(broker.lossRecord?.reason).toBe(reason); + expect(losses).toHaveLength(1); + }); + + it("enters lost on a close timeout", async () => { + const fake = createFakeDuplexChannel(); + fake.setCloseBehavior("hang"); + const broker = createDuplexBridgeBroker({ + channel: fake.channel, + forwardRequest: async () => ({ status: 200 }), + closeTimeoutMs: 20, + }); + broker.start(); + + await broker.close(); + + expect(broker.state).toBe("lost"); + expect(broker.lossRecord?.reason).toBe("close_timeout"); + }); + + it("stops the heartbeat, marks the run bridge ended, and dispatches nothing after loss", async () => { + const fake = createFakeDuplexChannel(); + const forwarded: string[] = []; + const broker = createDuplexBridgeBroker({ + channel: fake.channel, + forwardRequest: async (request) => { + forwarded.push(request.id); + return { status: 200 }; + }, + }); + broker.start(); + + fake.emitExit({ exitCode: 1 }); + await flushMacrotasks(); + expect(broker.state).toBe("lost"); + expect(broker.lossRecord).not.toBeNull(); + + // A request that arrives after loss reaches nothing. The broker never + // reconnects and never replays a request. + fake.emitData(requestFrameLine({ id: "after-loss" })); + await flushMacrotasks(); + expect(forwarded).toEqual([]); + }); + + it("forwards one request id one time, so a repeated frame never reaches the API twice", async () => { + const fake = createFakeDuplexChannel(); + const forwarded: string[] = []; + const broker = createDuplexBridgeBroker({ + channel: fake.channel, + forwardRequest: async (request) => { + forwarded.push(request.id); + return { status: 200, body: "ok" }; + }, + }); + broker.start(); + + fake.emitData(requestFrameLine({ id: "dup" })); + await flushMacrotasks(); + fake.emitData(requestFrameLine({ id: "dup" })); + await flushMacrotasks(); + + expect(forwarded).toEqual(["dup"]); + expect(fake.written).toHaveLength(1); + + await broker.close(); + }); + + it("captures the dispatch-start point per request for metrics only", async () => { + const fake = createFakeDuplexChannel(); + const records: DuplexBrokerRequestRecord[] = []; + let clock = 1000; + const broker = createDuplexBridgeBroker({ + channel: fake.channel, + forwardRequest: async () => ({ status: 200 }), + now: () => clock, + onRequestRecord: (record) => records.push(record), + }); + broker.start(); + + clock = 2500; + fake.emitData(requestFrameLine({ id: "metric-1", method: "GET", path: "/api/agents/me" })); + await flushMacrotasks(); + + expect(records).toEqual([ + { id: "metric-1", method: "GET", path: "/api/agents/me", dispatchStartMs: 2500 }, + ]); + // The record never reaches the channel. The broker writes only a response. + expect(fake.writtenTypes).not.toContain("request"); + + await broker.close(); + }); + + it("latches a failure when a loss orders before an orderly completion and names the typed reason", async () => { + const fake = createFakeDuplexChannel(); + const broker = createDuplexBridgeBroker({ + channel: fake.channel, + forwardRequest: async () => ({ status: 200 }), + }); + broker.start(); + + // The channel dies mid-turn, before any orderly completion. + fake.emitExit({ exitCode: 1 }); + await flushMacrotasks(); + expect(broker.runDisposition).toEqual({ failed: true, lossReason: "provider_exit" }); + + // A later orderly completion cannot clear the latch. A delayed activity + // callback cannot clear the latch either, because the broker dispatches + // nothing after loss. + broker.markOrderlyCompletion(); + fake.emitData(requestFrameLine({ id: "after-loss" })); + await flushMacrotasks(); + expect(broker.runDisposition).toEqual({ failed: true, lossReason: "provider_exit" }); + }); + + it("keeps a success when a loss orders after a host-observed orderly completion, and emits no loss event", async () => { + const events: DuplexTelemetryEventRecord[] = []; + const counters: DuplexTelemetryCounterRecord[] = []; + const recorder: DuplexTelemetryRecorder = { + recordSpan() {}, + incrementCounter(record) { + counters.push(record); + }, + emitEvent(record) { + events.push(record); + }, + }; + const fake = createFakeDuplexChannel(); + const broker = createDuplexBridgeBroker({ + channel: fake.channel, + forwardRequest: async () => ({ status: 200 }), + telemetry: createDuplexTelemetry({ recorder, providerKey: "daytona" }), + }); + broker.start(); + + // The agent completes its turn, then the channel ends during the teardown. + broker.markOrderlyCompletion(); + fake.emitExit({ exitCode: 0 }); + await flushMacrotasks(); + + expect(broker.runDisposition).toEqual({ failed: false, lossReason: null }); + // A normal teardown is not a loss: no loss event and no loss counter. + expect(events.some((e) => e.dimensions.loss_reason !== undefined)).toBe(false); + expect(counters.some((c) => c.metric === DUPLEX_COUNTER_LOSS_TOTAL)).toBe(false); + }); + + it("carries the typed loss_reason on the transport loss event and keeps a sentinel message off every sink", async () => { + const spans: DuplexTelemetrySpanRecord[] = []; + const counters: DuplexTelemetryCounterRecord[] = []; + const events: DuplexTelemetryEventRecord[] = []; + const recorder: DuplexTelemetryRecorder = { + recordSpan(record) { + spans.push(record); + }, + incrementCounter(record) { + counters.push(record); + }, + emitEvent(record) { + events.push(record); + }, + }; + const logLines: string[] = []; + const fake = createFakeDuplexChannel(); + const sentinel = "SENTINEL-PROVIDER-TEXT-1a2b3c"; + const broker = createDuplexBridgeBroker({ + channel: fake.channel, + forwardRequest: async () => ({ status: 200 }), + telemetry: createDuplexTelemetry({ recorder, providerKey: "daytona" }), + logger: (message) => logLines.push(message), + }); + broker.start(); + + // A stream write failure carries a raw provider message. It maps to the typed + // `write_error`; the raw message must reach no sink. + fake.setWriteError(new Error(sentinel)); + fake.emitData(requestFrameLine({ id: "req-sentinel" })); + await flushMacrotasks(); + + const lossEvent = events.find((e) => e.dimensions.loss_reason !== undefined); + expect(lossEvent?.dimensions.loss_reason).toBe("write_error"); + expect(lossEvent?.dimensions).toMatchObject({ transport: "duplex", outcome: "error" }); + + // The sentinel provider text reaches no telemetry sink and no log line. + const serializedSinks = JSON.stringify({ spans, counters, events }); + expect(serializedSinks).not.toContain(sentinel); + expect(logLines.join("\n")).not.toContain(sentinel); + }); + + it("rejects a configuration where an inner budget is not smaller than its outer budget", () => { + expect(() => + assertNestedDuplexBrokerBudgets({ + forwardTimeoutMs: 32_000, + responseBudgetMs: 32_000, + gatewayWaitMs: 35_000, + }), + ).toThrow(/forward budget/); + expect(() => + assertNestedDuplexBrokerBudgets({ + forwardTimeoutMs: 30_000, + responseBudgetMs: 35_000, + gatewayWaitMs: 35_000, + }), + ).toThrow(/response budget/); + expect(() => + createDuplexBridgeBroker({ + channel: createFakeDuplexChannel().channel, + forwardRequest: async () => ({ status: 200 }), + budgets: { forwardTimeoutMs: 40_000 }, + }), + ).toThrow(/forward budget/); + // The default budget set holds the nested order. + expect(() => + assertNestedDuplexBrokerBudgets({ + forwardTimeoutMs: 30_000, + responseBudgetMs: 32_000, + gatewayWaitMs: 35_000, + }), + ).not.toThrow(); + }); +}); + +describe("sandbox target spec parse: enableSandboxDuplexBridge", () => { + // The minimal serialized sandbox target the host stamps and the adapter parses. + // A test overrides one field per case to prove the fail-closed parse. + function serializedSandboxTarget(overrides: Record): Record { + return { + kind: "remote", + transport: "sandbox", + providerKey: "daytona", + environmentId: "env-1", + leaseId: "lease-1", + remoteCwd: "/work", + ...overrides, + }; + } + + it("reads the kill switch as a grant when the stamped field is true", () => { + const parsed = parseAdapterExecutionTarget(serializedSandboxTarget({ enableSandboxDuplexBridge: true })); + expect(parsed?.kind).toBe("remote"); + if (parsed?.kind !== "remote" || parsed.transport !== "sandbox") { + throw new Error("expected a sandbox execution target"); + } + expect(parsed.enableSandboxDuplexBridge).toBe(true); + expect(adapterExecutionTargetEnablesSandboxDuplexBridge(parsed)).toBe(true); + }); + + it("parses an absent field as no grant", () => { + const parsed = parseAdapterExecutionTarget(serializedSandboxTarget({})); + if (parsed?.kind !== "remote" || parsed.transport !== "sandbox") { + throw new Error("expected a sandbox execution target"); + } + expect(parsed.enableSandboxDuplexBridge).toBe(false); + expect(adapterExecutionTargetEnablesSandboxDuplexBridge(parsed)).toBe(false); + }); + + it("parses a false field as no grant", () => { + const parsed = parseAdapterExecutionTarget(serializedSandboxTarget({ enableSandboxDuplexBridge: false })); + if (parsed?.kind !== "remote" || parsed.transport !== "sandbox") { + throw new Error("expected a sandbox execution target"); + } + expect(parsed.enableSandboxDuplexBridge).toBe(false); + expect(adapterExecutionTargetEnablesSandboxDuplexBridge(parsed)).toBe(false); + }); + + it("fails closed on a non-boolean field", () => { + // A string "true" is not the literal boolean true, so the parse never reads + // it as a grant. This keeps a malformed round-trip on the file bridge. + const parsed = parseAdapterExecutionTarget(serializedSandboxTarget({ enableSandboxDuplexBridge: "true" })); + if (parsed?.kind !== "remote" || parsed.transport !== "sandbox") { + throw new Error("expected a sandbox execution target"); + } + expect(parsed.enableSandboxDuplexBridge).toBe(false); + expect(adapterExecutionTargetEnablesSandboxDuplexBridge(parsed)).toBe(false); + }); + + it("returns false from the reader for a non-sandbox target", () => { + const localTarget = parseAdapterExecutionTarget({ kind: "local", environmentId: "env-1", leaseId: "lease-1" }); + expect(adapterExecutionTargetEnablesSandboxDuplexBridge(localTarget)).toBe(false); + expect(adapterExecutionTargetEnablesSandboxDuplexBridge(null)).toBe(false); + }); +}); diff --git a/packages/adapter-utils/src/execution-target.ts b/packages/adapter-utils/src/execution-target.ts index aa3c1e4064..300053ccea 100644 --- a/packages/adapter-utils/src/execution-target.ts +++ b/packages/adapter-utils/src/execution-target.ts @@ -2,10 +2,11 @@ import fs from "node:fs/promises"; import net from "node:net"; import os from "node:os"; import path from "node:path"; -import { randomUUID } from "node:crypto"; +import { randomBytes, randomUUID } from "node:crypto"; import type { SshRemoteExecutionSpec } from "./ssh.js"; import { prepareCommandManagedRuntime, + type CommandManagedDuplexChannel, type CommandManagedRuntimeAsset, type CommandManagedRuntimeRunner, } from "./command-managed-runtime.js"; @@ -23,19 +24,44 @@ export type { SandboxAdditionalSource, } from "./sandbox-managed-runtime.js"; import { + authorizeSandboxCallbackBridgeRequestWithRoutes, createCommandManagedSandboxCallbackBridgeQueueClient, createSandboxCallbackBridgeAsset, createSandboxCallbackBridgeToken, DEFAULT_SANDBOX_CALLBACK_BRIDGE_MAX_BODY_BYTES, + SANDBOX_CALLBACK_BRIDGE_DUPLEX_MODE, + SANDBOX_CALLBACK_BRIDGE_ENTRYPOINT, sandboxCallbackBridgeDirectories, + sanitizeSandboxCallbackBridgeHeaders, startSandboxCallbackBridgeServer, startSandboxCallbackBridgeWorker, syncRemoteTextFileWithHashSkip, + syncSandboxCallbackBridgeEntrypoint, } from "./sandbox-callback-bridge.js"; import { createSandboxRunLogTailFactory, type SandboxRunLogTailFactory, } from "./sandbox-run-log-stream.js"; +import { + createDuplexBridgeBroker, + DEFAULT_DUPLEX_BROKER_BUDGETS, + isSafeBridgeMethod, + typedDuplexLossReason, + type DuplexBridgeBroker, + type DuplexBrokerBudgets, + type DuplexBrokerForwardResult, + type DuplexBrokerRunDisposition, +} from "./duplex-bridge-broker.js"; +import { + decodeDuplexLine, + DEFAULT_MAX_DUPLEX_FRAME_BYTES, + type DuplexRequestFrame, +} from "./duplex-frame-codec.js"; +import { + createDuplexTelemetry, + type DuplexFallbackReason, + type DuplexTelemetryRecorder, +} from "./duplex-telemetry.js"; import { createSshCommandManagedRuntimeRunner, parseSshRemoteExecutionSpec, runSshCommand, shellQuote } from "./ssh.js"; import { ensureCommandResolvable, @@ -129,6 +155,14 @@ export interface AdapterSandboxExecutionTarget extends AdapterExecutionTargetWor * narrowing, then attaches it here. Absent when no snapshot was resolved. */ readonly effectiveCapabilities?: EffectiveSandboxCapabilities | null; + /** + * Per-run duplex bridge kill switch. The host stamps it on the same seam as + * `effectiveCapabilities`. `true` selects the duplex transport only when the + * capability `duplexCommandStream` is also `true`; any other value keeps the + * file bridge. The value stays on the host and never enters the sandbox + * environment. Absent means no grant. + */ + readonly enableSandboxDuplexBridge?: boolean; shellCommand?: "bash" | "sh" | null; environmentId?: string | null; leaseId?: string | null; @@ -142,6 +176,13 @@ export interface AdapterSandboxExecutionTarget extends AdapterExecutionTargetWor * set to `false` to explicitly opt out back to batch-at-end delivery. */ streamRunLogs?: boolean | null; + /** + * The injected duplex telemetry recorder for this run. The host attaches it on + * the same seam as `runner`, so this live object stays on the host and never + * enters the sandbox environment. The bridge binds it to the fixed duplex + * observability surface. Absent means the safe no-op default. + */ + duplexTelemetryRecorder?: DuplexTelemetryRecorder | null; } export type AdapterExecutionTarget = @@ -213,6 +254,24 @@ export interface AdapterExecutionTargetPaperclipBridgeHandle { * `runAdapterExecutionTargetProcess` via `options.runLogTail`. */ runLogTail?: SandboxRunLogTailFactory | null; + /** + * Read the terminal run disposition of the duplex control channel. It reports a + * failure when the channel was lost before an orderly completion, and names the + * typed loss reason. It reports a success for a healthy channel or a + * normal-teardown loss. The file bridge path never sets it, so the method is + * absent there. The caller reads it at the run-disposition seam to fail a run + * whose control channel died mid-turn. + */ + readRunDisposition?(): DuplexBrokerRunDisposition; + /** + * Mark the host-observed orderly completion of the agent turn on the broker's + * ordered lifecycle. The caller marks it at the ACP terminal-finalization + * boundary for a still-success-eligible completion, so a later teardown loss + * cannot flip the run to a failure. A loss that already latched keeps the + * failure, because the broker no-ops the mark after a latched loss. The file + * bridge path never sets it, so the method is absent there. + */ + markOrderlyCompletion?(): void; stop(): Promise; } @@ -318,6 +377,35 @@ export function adapterExecutionTargetUsesManagedHome( return target?.kind === "remote" && target.transport === "sandbox"; } +/** + * Read the per-run duplex bridge kill switch off a target. Only a sandbox + * target with `enableSandboxDuplexBridge` set to `true` returns `true`. Every + * other target and every other value returns `false`, so the caller fails + * closed to the file bridge. + */ +export function adapterExecutionTargetEnablesSandboxDuplexBridge( + target: AdapterExecutionTarget | null | undefined, +): boolean { + return ( + target?.kind === "remote" && + target.transport === "sandbox" && + target.enableSandboxDuplexBridge === true + ); +} + +/** + * Read the injected duplex telemetry recorder off a target. Only a sandbox + * target with a recorder attached returns it. Every other target returns null, + * so the bridge falls back to the safe no-op recorder. + */ +export function adapterExecutionTargetDuplexTelemetryRecorder( + target: AdapterExecutionTarget | null | undefined, +): DuplexTelemetryRecorder | null { + return target?.kind === "remote" && target.transport === "sandbox" + ? target.duplexTelemetryRecorder ?? null + : null; +} + export function adapterExecutionTargetRemoteCwd( target: AdapterExecutionTarget | null | undefined, localCwd: string, @@ -1143,6 +1231,9 @@ export function parseAdapterExecutionTarget(value: unknown): AdapterExecutionTar remoteCwd, timeoutMs: typeof parsed.timeoutMs === "number" ? parsed.timeoutMs : null, streamRunLogs: typeof parsed.streamRunLogs === "boolean" ? parsed.streamRunLogs : null, + // Fail closed: only the literal `true` reads as a grant. An absent field + // or any other value parses as no grant, so a round-trip never invents one. + enableSandboxDuplexBridge: parsed.enableSandboxDuplexBridge === true, ...(effectiveCapabilities ? { effectiveCapabilities } : {}), }; } @@ -2184,6 +2275,440 @@ child.on("close", (code, signal) => void writeEvent({ type: "exit", code, signal ${PROCESS_SESSION_STDIN_POLL_TAIL}`; } +/** The default deadline for the duplex readiness handshake, in milliseconds. */ +const DEFAULT_DUPLEX_READINESS_TIMEOUT_MS = 10_000; + +/** The default bounded budget to close a partial duplex channel, in milliseconds. */ +const DEFAULT_DUPLEX_CLEANUP_BUDGET_MS = 2_000; + +/** + * Reserve a loopback port the host assigns to the duplex gateway. The host binds + * an ephemeral listener on `127.0.0.1`, reads the port the operating system + * chose, then closes the listener. The number is very likely free when the + * gateway binds it a moment later. The gateway binds exactly this port or exits + * nonzero, so a taken port fails closed to the file bridge and never steers the + * endpoint. + */ +async function reserveHostAssignedLoopbackPort(): Promise { + return new Promise((resolve, reject) => { + const probe = net.createServer(); + probe.once("error", reject); + probe.listen(0, "127.0.0.1", () => { + const address = probe.address(); + if (!address || typeof address === "string") { + probe.close(() => reject(new Error("Could not reserve a loopback port for the duplex gateway."))); + return; + } + const reserved = address.port; + probe.close(() => resolve(reserved)); + }); + }); +} + +/** + * Build the argument vector that launches the duplex gateway in the sandbox. The + * host passes the assigned port and the per-open nonce only through the launch + * environment, so the argument vector sets them as environment assignments in + * front of the node command. No addressing data comes from the channel. + * + * The script uses `exec env NAME=value ... command`. A POSIX shell accepts an + * environment-assignment prefix only on a plain command, never on `exec`. The + * form `exec NAME=value command` exits with status 127. The `env` utility carries + * the assignments, and `exec` still replaces the shell with the gateway process, + * so the gateway keeps the process slot and the assigned environment. + */ +export function buildDuplexGatewayLaunchArgv(input: { + shellCommand: "bash" | "sh"; + remoteEntrypoint: string; + nodeCommand?: string | null; + env: Record; +}): string[] { + const assignments = Object.entries(input.env) + .map(([key, value]) => `${key}=${shellQuote(value)}`) + .join(" "); + const nodeCommand = input.nodeCommand?.trim() || "node"; + const script = `exec env ${assignments} ${shellQuote(nodeCommand)} ${shellQuote(input.remoteEntrypoint)}`; + return [input.shellCommand, ...shellCommandArgs(script)]; +} + +/** The reason the duplex readiness handshake did not pass. */ +type DuplexReadinessFailure = + | "protocol_contamination" + | "nonce_mismatch" + | "channel_exit" + | "timeout"; + +/** The outcome of the duplex readiness handshake. */ +type DuplexReadinessResult = + | { ok: true } + | { ok: false; reason: DuplexReadinessFailure }; + +/** + * Map a readiness failure to the fixed fallback reason. Each readiness failure + * maps to exactly one reason from the closed telemetry set, so the fallback + * counter and the transport event carry only an approved value. + */ +function duplexReadinessFallbackReason(reason: DuplexReadinessFailure): DuplexFallbackReason { + switch (reason) { + case "protocol_contamination": + return "contaminated"; + case "nonce_mismatch": + return "ready_nonce_mismatch"; + case "timeout": + return "ready_timeout"; + case "channel_exit": + return "ready_invalid"; + } +} + +/** + * The fixed marker the worker manager puts in a route-busy rejection. The manager + * defines the constant; this module matches the fixed string, because the two + * layers ship in separate packages and share no import. The marker names the + * process-scoped route ceiling, so it carries no route, query, body, or token. + */ +const DUPLEX_ROUTE_BUSY_ERROR_MARKER = "DUPLEX_CHANNEL_ROUTE_BUSY"; + +/** + * Report whether the caught open error is a route-busy rejection. The host maps it + * to the `route_busy` fallback stage, so a full process-scoped route ceiling never + * folds into a generic open failure. + */ +function isDuplexRouteBusyError(error: unknown): boolean { + return error instanceof Error && error.message.includes(DUPLEX_ROUTE_BUSY_ERROR_MARKER); +} + +/** + * The cap on the pre-READY readiness buffer, in bytes. The gate reads untrusted + * bytes before the READY line arrives, so it bounds the buffer. The cap is the + * codec frame-size bound plus one line of margin, so a legitimate maximum-size + * READY frame still fits. Past the cap with no newline, the stream cannot be a + * valid READY frame, so the gate finishes with protocol contamination. + */ +const DUPLEX_READINESS_BUFFER_CAP_BYTES = DEFAULT_MAX_DUPLEX_FRAME_BYTES + 4_096; + +/** + * Derive the nested broker budgets from one forward budget. The response budget + * and the gateway wait budget each add a fixed margin, so the nested budget + * order holds for any finite forward budget. The default forward budget (30 s) + * yields the historical 32 s response budget and 35 s gateway wait budget, so a + * caller that sets no forward budget sees no change. + */ +function deriveNestedDuplexBrokerBudgets(forwardTimeoutMs: number): DuplexBrokerBudgets { + const responseMargin = + DEFAULT_DUPLEX_BROKER_BUDGETS.responseBudgetMs - DEFAULT_DUPLEX_BROKER_BUDGETS.forwardTimeoutMs; + const gatewayMargin = + DEFAULT_DUPLEX_BROKER_BUDGETS.gatewayWaitMs - DEFAULT_DUPLEX_BROKER_BUDGETS.responseBudgetMs; + const responseBudgetMs = forwardTimeoutMs + responseMargin; + const gatewayWaitMs = responseBudgetMs + gatewayMargin; + return { forwardTimeoutMs, responseBudgetMs, gatewayWaitMs }; +} + +/** + * Create the run-log directory on the sandbox before the tail starts. The file + * bridge worker creates this directory on the file path. The duplex path starts + * no worker, so the host creates the directory here. The tail then reads a real + * directory from its first tick. + * + * This step is best effort. The broker already serves the duplex transport when + * the host reaches it, and the tail wrap command runs its own `mkdir -p` as a + * backstop. So a create failure must not tear down a working duplex transport; + * the host swallows it and still builds the tail. The log line names no raw + * error, so no raw error rides a log line here. + */ +async function ensureSandboxRunLogDirectory(input: { + runner: CommandManagedRuntimeRunner; + remoteCwd: string; + logsDir: string; + shellCommand: "bash" | "sh"; + timeoutMs: number | null | undefined; +}): Promise { + try { + await input.runner.execute({ + command: input.shellCommand, + args: shellCommandArgs(`mkdir -p ${shellQuote(input.logsDir)}`), + cwd: input.remoteCwd, + timeoutMs: input.timeoutMs ?? undefined, + }); + } catch { + // Best effort: the tail wrap command creates the directory before it writes. + } +} + +/** + * The duplex readiness gate. The gate owns the single data listener and the + * single exit listener of the channel while the host waits for a valid READY + * frame. It resolves the handshake, then hands the channel to the broker. + * + * The gate authenticates readiness with the nonce and the strict READY schema, + * not with the line position. A PTY channel echoes the launch wrapper line before + * it sets raw mode, so the first line is often not the READY frame. The gate skips + * each pre-READY line that does not decode as a READY frame, then accepts the first + * line that decodes as a READY frame with the matching nonce. A line that decodes + * as a READY frame with a wrong nonce fails the handshake with a nonce mismatch. An + * early exit or a timeout also fails the handshake. The gate never dispatches a + * request; the broker does that after readiness passes. + * + * The skipped bytes stay in the capped buffer, so the O(1) cap and the readiness + * timeout still bound the wait. The gate enforces the cap on every pre-READY path + * and before it decodes a candidate line, so an over-cap prefix never reaches READY + * acceptance. The cap check has priority over READY acceptance on every path. + */ +// The count of the pre-READY newline-scan work, in UTF-16 code units. Each +// search adds the number of code units it can read. A test reads this count to +// prove the scan work stays linear in the bytes received. Production code never +// reads this count. +let duplexReadinessNewlineScanUnits = 0; + +/** + * Find the first newline in `buffer` at or after index `from` and return its + * index, or -1. `String#indexOf` reads the code units from `from` up to the + * newline it finds, or to the end of the buffer when it finds none. This helper + * counts that real scan distance for a test. The count stays linear in the bytes + * received when the caller advances `from` past each newline it consumes. + */ +function findNewlineFrom(buffer: string, from: number): number { + const newlineIndex = buffer.indexOf("\n", from); + const scanned = newlineIndex === -1 ? buffer.length - from : newlineIndex - from + 1; + duplexReadinessNewlineScanUnits += scanned; + return newlineIndex; +} + +/** + * Test-only surface for the pre-READY readiness gate. A test reads the scan + * count to prove the newline search work stays linear in the bytes received. + * Production code does not use this object. + */ +export const __duplexReadinessTesting = { + readNewlineScanUnits: (): number => duplexReadinessNewlineScanUnits, + resetNewlineScanUnits: (): void => { + duplexReadinessNewlineScanUnits = 0; + }, +}; + +interface DuplexReadinessGate { + /** Resolves with the handshake outcome. It never rejects. */ + readonly ready: Promise; + /** + * The channel view the broker consumes after readiness passes. It replays the + * bytes that followed the READY frame, then forwards each later chunk and the + * exit. The gate keeps one real data listener, so the broker never double-binds + * the channel. + */ + readonly brokerChannel: CommandManagedDuplexChannel; +} + +function createDuplexReadinessGate( + channel: CommandManagedDuplexChannel, + options: { nonce: string; timeoutMs: number }, +): DuplexReadinessGate { + let settled = false; + let readyOk = false; + // The raw bytes the host reads before the READY frame completes. The buffer is + // append-only, so the O(1) cap check on `buffer.length` stays valid. + let buffer = ""; + // The start index of the current line in `buffer`. A leading blank line + // advances this cursor past its newline without a buffer copy. + let lineStart = 0; + // The next index to search for a newline. The gate scans from here, so each + // code unit is read at most one time for the newline search. + let scanFrom = 0; + // The bytes that followed the READY frame, held until the broker binds. + let pending = ""; + // The exit that arrived after READY but before the broker bound, if any. + let pendingExit: { exitCode: number | null } | null = null; + let dataSink: ((chunk: string) => void) | null = null; + let exitSink: ((exit: { exitCode: number | null }) => void) | null = null; + let resolveReady!: (result: DuplexReadinessResult) => void; + const ready = new Promise((resolve) => { + resolveReady = resolve; + }); + + const timer = setTimeout(() => finish({ ok: false, reason: "timeout" }), options.timeoutMs); + timer.unref?.(); + + function finish(result: DuplexReadinessResult): void { + if (settled) return; + settled = true; + clearTimeout(timer); + if (result.ok) readyOk = true; + resolveReady(result); + } + + channel.onData((chunk) => { + if (dataSink) { + dataSink(chunk); + return; + } + if (readyOk) { + // READY already passed; hold the bytes until the broker binds. + pending += chunk; + return; + } + if (settled) { + // The handshake already failed. The channel is untrusted and can keep + // sending bytes until the host closes it. Drop them, so a failed handshake + // never grows the buffer after the gate settles. + return; + } + // Append the new bytes and continue the newline search from `scanFrom`, the + // first index not yet examined. Each code unit is read at most one time for + // the search, so the total scan work stays linear in the bytes received. + buffer += chunk; + for (;;) { + const newlineIndex = findNewlineFrom(buffer, scanFrom); + if (newlineIndex === -1) { + // No complete line yet. The whole buffer up to the end is now scanned. + scanFrom = buffer.length; + // The gate reads untrusted bytes, so bound the pre-READY buffer. Past + // the cap with no complete READY line, the stream cannot be a valid + // READY frame, so finish with protocol contamination. The retained + // skipped lines count against the cap; that is acceptable and fail-closed. + // + // Gate on buffer.length, the UTF-16 code-unit count, which is O(1). + // Buffer.byteLength is O(n), so a byte check on every newline-less chunk + // makes the pre-READY window quadratic in the bytes received. The UTF-8 + // byte length is greater than or equal to the UTF-16 code-unit count for + // every string, so this check never fires before the byte cap is truly + // exceeded. It can fire late by at most a factor of 3, so peak buffer + // memory stays bounded near 3 MB. + if (buffer.length > DUPLEX_READINESS_BUFFER_CAP_BYTES) { + finish({ ok: false, reason: "protocol_contamination" }); + } + return; + } + if (newlineIndex === lineStart) { + // Skip a blank line, the same as the frame decoder. Advance the line-start + // cursor past the newline without a buffer copy, then search the next line + // from the position after the newline. Enforce the cap after the advance, + // so a blank-line flood past the cap fails closed before READY acceptance. + lineStart = newlineIndex + 1; + scanFrom = newlineIndex + 1; + if (lineStart > DUPLEX_READINESS_BUFFER_CAP_BYTES) { + finish({ ok: false, reason: "protocol_contamination" }); + return; + } + continue; + } + // A complete non-blank candidate line spans `[lineStart, newlineIndex)`. + // Enforce the cap on the line's end offset before the decode, so an over-cap + // prefix never reaches READY acceptance on a completed line. + if (newlineIndex + 1 > DUPLEX_READINESS_BUFFER_CAP_BYTES) { + finish({ ok: false, reason: "protocol_contamination" }); + return; + } + // Slice only the single candidate line for the decode. + const line = buffer.slice(lineStart, newlineIndex); + let decoded = decodeDuplexLine(line); + if (!decoded.ok) { + // The whole line did not decode. A terminal can put bytes in front of the + // gateway's first frame on the same line, with no newline between them: a + // shell with bracketed paste enabled writes its disable sequence and a + // bare carriage return (`ESC [ ? 2 0 0 4 l CR`) immediately before the + // child's first output. So retry the decode from the first `{` in the + // line, which is where a frame can start. + // + // This does not weaken the handshake. The retry still runs the same + // strict decode over the remainder of the line, and readiness still + // authenticates on the nonce below, so a prefix cannot forge a frame or + // smuggle a second one — it can only be discarded. + const braceIndex = line.indexOf("{"); + if (braceIndex > 0) { + decoded = decodeDuplexLine(line.slice(braceIndex)); + } + } + if (decoded.ok && decoded.frame.type === "ready") { + // A line that decodes as a READY frame authenticates by the nonce. A wrong + // nonce fails the handshake; the matching nonce passes it. + if (decoded.frame.nonce !== options.nonce) { + finish({ ok: false, reason: "nonce_mismatch" }); + return; + } + // Hold the bytes that follow the READY line for the broker to replay. + pending = buffer.slice(newlineIndex + 1); + finish({ ok: true }); + return; + } + // The line does not decode as a READY frame. A PTY echo line or any other + // pre-READY noise reaches here. Skip it and keep scanning; the nonce and the + // strict schema, not the line position, authenticate readiness. Enforce the + // cap after the advance, so a noise-line flood past the cap fails closed. + lineStart = newlineIndex + 1; + scanFrom = newlineIndex + 1; + if (lineStart > DUPLEX_READINESS_BUFFER_CAP_BYTES) { + finish({ ok: false, reason: "protocol_contamination" }); + return; + } + } + }); + + channel.onExit((exit) => { + if (exitSink) { + exitSink(exit); + return; + } + if (readyOk) { + // The channel exited after READY but before the broker bound. Hold the + // exit so the broker still learns of the loss. + pendingExit = exit; + return; + } + finish({ ok: false, reason: "channel_exit" }); + }); + + const brokerChannel: CommandManagedDuplexChannel = { + write: (data: string) => channel.write(data), + onData: (listener: (chunk: string) => void) => { + dataSink = listener; + if (pending.length > 0) { + const replay = pending; + pending = ""; + listener(replay); + } + }, + onExit: (listener: (exit: { exitCode: number | null }) => void) => { + exitSink = listener; + if (pendingExit) { + const exit = pendingExit; + pendingExit = null; + listener(exit); + } + }, + stop: () => channel.stop(), + close: () => channel.close(), + }; + + return { ready, brokerChannel }; +} + +/** + * Close a partial duplex channel inside a bounded budget, then stop the child. + * The host runs this on any readiness failure, so a failed duplex attempt leaves + * no live provider session before the fallback to the file bridge. + */ +async function closeDuplexChannelWithinBudget( + channel: CommandManagedDuplexChannel, + budgetMs: number, +): Promise { + try { + let timer: ReturnType | undefined; + const budget = new Promise((resolve) => { + timer = setTimeout(resolve, budgetMs); + timer.unref?.(); + }); + await Promise.race([channel.close().catch(() => undefined), budget]); + if (timer !== undefined) clearTimeout(timer); + } catch { + // Best effort: the stop below still removes the child. + } finally { + try { + channel.stop(); + } catch { + // The channel is already gone; nothing more to do. + } + } +} + export async function startAdapterExecutionTargetPaperclipBridge(input: { runId: string; target: AdapterExecutionTarget | null | undefined; @@ -2194,6 +2719,21 @@ export async function startAdapterExecutionTargetPaperclipBridge(input: { hostApiUrl?: string | null; onLog?: (stream: "stdout" | "stderr", chunk: string) => Promise; maxBodyBytes?: number | null; + // The deadline for one forward call, in milliseconds. This is the inner budget + // of the duplex broker's nested budget set. The default is the forward budget + // in `DEFAULT_DUPLEX_BROKER_BUDGETS` (30 s), so the current behavior does not + // change when the caller sets no option. + forwardTimeoutMs?: number | null; + // The first strict gate for the duplex transport. The host selects duplex only + // when this is exactly `true` and the resolved capability + // `duplexCommandStream` is exactly `true`. Any other value of either gate + // selects the file bridge. The caller reads this from the experimental instance + // setting `enableSandboxDuplexBridge`. The default is the file bridge. + enableSandboxDuplexBridge?: boolean | null; + // The deadline for the duplex readiness handshake, in milliseconds. On a + // timeout the host closes the partial channel and selects the file bridge. The + // default is `DEFAULT_DUPLEX_READINESS_TIMEOUT_MS`. + duplexReadinessTimeoutMs?: number | null; // Return the current-run parent-context token. The factory threads it into the // callback bridge worker, which reads it per request so each request // `sandbox.exec` span parents to the live run span. When it is absent, the @@ -2204,6 +2744,12 @@ export async function startAdapterExecutionTargetPaperclipBridge(input: { // request's execs group under one wrapper span. When it is absent, the request // work runs under the run parent with no wrapper span. runtimeSpan?: RuntimeSpanRunner; + // The injected recorder for the fixed duplex observability surface. The factory + // binds it to a provider-scoped telemetry facade, which records the channel-open + // span, the request span, the guarded counters, and the transport event. The + // default is a no-op recorder, so the surface stays inert until the host injects + // a real recorder. + duplexTelemetryRecorder?: DuplexTelemetryRecorder | null; }): Promise { if (!adapterExecutionTargetUsesPaperclipBridge(input.target)) { return null; @@ -2218,6 +2764,10 @@ export async function startAdapterExecutionTargetPaperclipBridge(input: { if (hostApiToken.length === 0) { throw new Error("Sandbox bridge mode requires a host-side Paperclip API token."); } + // The forward budget for one relayed request. It stays at the broker's default + // forward budget (30 s) when the caller sets no option, so current behavior + // does not change. + const forwardTimeoutMs = input.forwardTimeoutMs ?? DEFAULT_DUPLEX_BROKER_BUDGETS.forwardTimeoutMs; const runtimeRootDir = input.runtimeRootDir?.trim().length @@ -2257,6 +2807,366 @@ export async function startAdapterExecutionTargetPaperclipBridge(input: { ); const bridgeAsset = await createSandboxCallbackBridgeAsset(); + + // The provider-scoped telemetry facade for the fixed duplex observability + // surface. It maps the raw provider key through the allowlist one time, so no + // raw plugin key reaches a span, a counter, or the event. The default recorder + // is a no-op, so the facade is inert until the host injects a real recorder. + const duplexProviderKey = + "providerKey" in target ? target.providerKey ?? undefined : undefined; + const duplexTelemetry = createDuplexTelemetry({ + recorder: input.duplexTelemetryRecorder ?? undefined, + providerKey: duplexProviderKey, + }); + + // PAPERCLIP_BRIDGE_DEBUG opts into verbose stdout logs of every bridge proxy + // request/response. The query string is logged verbatim, so callers who pass + // auth tokens or other sensitive values as query parameters should be aware + // those values appear in the host process's stdout when this flag is enabled. + // Only intended for active debugging in trusted environments. + const bridgeDebugEnabled = isBridgeDebugEnabled(process.env); + + // One forward of a relayed sandbox request onto the existing Paperclip API + // path. The forward applies the real host token and the signed run id, so the + // token replacement and the run attribution stay in one place for both the + // file bridge and the duplex broker. The sandbox request carries only the + // bridge token; the real agent token never leaves the host. + const forwardBridgeRequest = async ( + request: { method: string; path: string; query: string; headers: Record; body: string }, + signal?: AbortSignal, + options?: { suppressDebugLog?: boolean }, + ): Promise<{ status: number; headers: Record; body: string }> => { + const method = request.method.trim().toUpperCase() || "GET"; + // The per-request debug log prints the method, the path, and the query. The + // duplex path suppresses it, so no route or query rides a log line there. The + // file path keeps the existing behavior. + const emitDebugLog = bridgeDebugEnabled && options?.suppressDebugLog !== true; + if (emitDebugLog) { + await onLog( + "stdout", + `[paperclip] Bridge proxy ${method} ${request.path}${request.query ? `?${request.query}` : ""}\n`, + ); + } + const headers = new Headers(); + for (const [key, value] of Object.entries(request.headers)) { + if (value.trim().length === 0) continue; + headers.set(key, value); + } + headers.set("authorization", `Bearer ${hostApiToken}`); + headers.set("x-paperclip-run-id", input.runId); + // Abort the forward when the caller aborts the request (its per-iteration + // timeout or watchdog fired, or the broker's forward budget ended), or after + // the forward budget here, whichever comes first. + const timeoutSignal = AbortSignal.timeout(forwardTimeoutMs); + const forwardSignal = signal ? AbortSignal.any([signal, timeoutSignal]) : timeoutSignal; + const response = await fetch(buildBridgeForwardUrl(hostApiUrl, request), { + method, + headers, + ...(method === "GET" || method === "HEAD" ? {} : { body: request.body }), + signal: forwardSignal, + }); + if (emitDebugLog) { + await onLog( + "stdout", + `[paperclip] Bridge proxy response ${response.status} for ${method} ${request.path}${request.query ? `?${request.query}` : ""}\n`, + ); + } + // The host delivered response headers, so the response-body read starts after + // the host processed the request. A later response-body read failure (a body + // over the size limit, or a stream read error) needs a classification by + // method safety. A safe method (GET, HEAD, OPTIONS, TRACE) never changes host + // state, so a retry cannot double-apply a mutation and the failure stays + // retryable. A mutating or otherwise unsafe method may have committed on the + // host, so a retryable status is unsafe: it makes the caller repeat the + // request with a new request id outside the broker deduplication set, and the + // host applies the mutation twice. For an unsafe method the code returns a + // non-retryable 504 and marks the outcome indeterminate, exactly like an + // aborted in-flight forward. The in-sandbox server maps the indeterminate 504 + // to a non-retryable 409 for both the file bridge and the duplex broker. + let responseBody: string; + try { + responseBody = await readBridgeForwardResponseBody(response, maxBodyBytes); + } catch (error) { + if (isSafeBridgeMethod(method)) { + // The method is safe, so a retry cannot double-apply a mutation. Return a + // retryable 502 with no indeterminate marker, so the gateway passes it + // through as a retryable status. + return { + status: 502, + headers: { "content-type": "application/json" }, + body: JSON.stringify({ + error: error instanceof Error ? error.message : String(error), + }), + }; + } + return { + status: 504, + headers: { + "content-type": "application/json", + "x-paperclip-bridge-outcome": "indeterminate", + }, + body: JSON.stringify({ + error: error instanceof Error ? error.message : String(error), + outcome: "indeterminate", + retryable: false, + }), + }; + } + const commentMarker = postedIssueCommentLogMarker(method, request.path, response.status, responseBody); + if (commentMarker) await onLog("stdout", commentMarker); + return { + status: response.status, + headers: buildBridgeResponseHeaders(response), + body: responseBody, + }; + }; + + // Two strict gates guard the duplex transport. Select duplex only when the + // experimental setting is exactly `true`, the resolved capability + // `duplexCommandStream` is exactly `true`, and the runner exposes the duplex + // channel. Any other value of either gate selects the file bridge below. + const duplexRequested = input.enableSandboxDuplexBridge === true; + const capabilityGranted = + "effectiveCapabilities" in target && + target.effectiveCapabilities?.duplexCommandStream === true; + const openDuplexChannel = runner.openDuplexChannel?.bind(runner); + // Record the pre-attempt fallback for a file-bridge selection that opens no + // channel. `gate_off` marks the kill switch off; `capability_absent` marks the + // capability or the runner method absent. A later channel-open failure records + // its own fallback through the channel-open attempt below. + if (!duplexRequested) { + duplexTelemetry.recordFallback("gate_off"); + } else if (!capabilityGranted || typeof openDuplexChannel !== "function") { + duplexTelemetry.recordFallback("capability_absent"); + } + if (duplexRequested && capabilityGranted && typeof openDuplexChannel === "function") { + // Begin the channel-open attempt. The block reports exactly one terminal: + // `ready` on success, or `fallback(reason)` on an open or a readiness failure. + const duplexChannelOpen = duplexTelemetry.startChannelOpen(); + const readinessTimeoutMs = + typeof input.duplexReadinessTimeoutMs === "number" && + Number.isFinite(input.duplexReadinessTimeoutMs) && + input.duplexReadinessTimeoutMs > 0 + ? Math.trunc(input.duplexReadinessTimeoutMs) + : DEFAULT_DUPLEX_READINESS_TIMEOUT_MS; + + // The host assigns the loopback port before it opens the channel. It passes + // the port and one random per-open nonce to the gateway only through the + // launch environment. It builds the sandbox-facing origin from its own + // stored port. No field of any channel frame contributes to the endpoint. + // + // Provider-boundary rationale (a security-review condition): the Daytona SDK + // surface the plugin uses (`sandbox.process` sessions, PTY, and exec; + // `sandbox.fs`) exposes no listener identity that the provider control plane + // binds to a launched process. No channel-supplied address can be attested, + // so this design uses zero channel-supplied addressing data. The only actor + // that can pre-bind the host-assigned loopback port before the agent starts + // is the provider itself, which already delivers the agent launch + // environment. The gateway bind-or-exit contract fails closed. This is an + // accepted provider-untrusted residual, not endpoint authentication: the + // nonce is a liveness signal, because a compromised channel can read the + // launch environment; the endpoint stays safe because it never derives from + // the channel. + const assignedPort = await reserveHostAssignedLoopbackPort(); + const nonce = randomBytes(16).toString("hex"); + const sandboxOrigin = `http://127.0.0.1:${assignedPort}`; + + let channel: CommandManagedDuplexChannel | null = null; + // The open runs two stages under one try: the entrypoint sync, then the + // channel open. The stage names the exact open-failure outcome in the catch. + let openStage: "entrypoint_sync" | "channel_open" = "entrypoint_sync"; + try { + const assetSync = await syncSandboxCallbackBridgeEntrypoint({ + runner, + remoteCwd: target.remoteCwd, + assetRemoteDir, + bridgeAsset, + timeoutMs: bridgeTimeoutMs, + shellCommand, + }); + const gatewayEnv: Record = { + PAPERCLIP_API_BRIDGE_MODE: SANDBOX_CALLBACK_BRIDGE_DUPLEX_MODE, + PAPERCLIP_BRIDGE_TOKEN: bridgeToken, + PAPERCLIP_BRIDGE_HOST: "127.0.0.1", + PAPERCLIP_BRIDGE_PORT: String(assignedPort), + PAPERCLIP_BRIDGE_NONCE: nonce, + PAPERCLIP_BRIDGE_MAX_BODY_BYTES: String(maxBodyBytes), + }; + const command = buildDuplexGatewayLaunchArgv({ + shellCommand, + remoteEntrypoint: assetSync.remoteEntrypoint, + env: gatewayEnv, + }); + openStage = "channel_open"; + channel = await openDuplexChannel({ command }); + } catch (error) { + // The channel never opened, so no request could carry the bridge token. + // The open call is the last statement of the try, so `channel` is still + // null here; there is nothing to close. Fall through to the file bridge + // below. Bind and keep the caught error, so the fallback names the exact + // open-failure stage: a full process-scoped route ceiling is `route_busy`; + // otherwise the stage is the entrypoint sync or the channel open. The log + // line names only the fixed stage enum, so no raw provider error rides a log + // line on the duplex path. + const reason: DuplexFallbackReason = isDuplexRouteBusyError(error) + ? "route_busy" + : openStage === "entrypoint_sync" + ? "entrypoint_sync_failed" + : "channel_open_failed"; + duplexChannelOpen.fallback(reason); + await onLog( + "stderr", + `[paperclip] Could not open the sandbox duplex channel (${reason}). Using the file bridge.\n`, + ); + channel = null; + } + + if (channel) { + const gate = createDuplexReadinessGate(channel, { nonce, timeoutMs: readinessTimeoutMs }); + const readiness = await gate.ready; + if (!readiness.ok) { + // Fail closed. Close the partial channel inside a bounded budget, then + // select the file bridge. The broker never started, so no request that + // carries the bridge token reached the channel or any endpoint. The reason + // is a fixed enum, so it rides the log line and the fallback telemetry + // with no raw value. + await closeDuplexChannelWithinBudget(channel, DEFAULT_DUPLEX_CLEANUP_BUDGET_MS); + duplexChannelOpen.fallback(duplexReadinessFallbackReason(readiness.reason)); + await onLog( + "stderr", + `[paperclip] Sandbox duplex readiness failed (${readiness.reason}). Using the file bridge.\n`, + ); + } else { + // Readiness passed. Construct the broker inside the guarded region, so a + // construction throw closes the channel within the cleanup budget and + // selects the file bridge. The broker enforces the same route allowlist + // as the file bridge, then forwards each request with the real token and + // the run id. The agent environment below receives the bridge URL and + // token only now, after readiness passed. + let broker: DuplexBridgeBroker | null = null; + try { + broker = createDuplexBridgeBroker({ + channel: gate.brokerChannel, + // Derive the nested budgets from the forward budget, so any forward + // budget keeps the nested order the broker asserts at construction. + budgets: deriveNestedDuplexBrokerBudgets(forwardTimeoutMs), + forwardRequest: async ( + request: DuplexRequestFrame, + options: { signal: AbortSignal }, + ): Promise => { + const denialReason = authorizeSandboxCallbackBridgeRequestWithRoutes(request); + if (denialReason) { + return { + status: 403, + headers: { "content-type": "application/json" }, + body: JSON.stringify({ error: denialReason }), + }; + } + // Apply the header allowlist on the host, next to the route check. + // The host is the trust boundary: the sandbox controls the frame + // headers, so the host drops every header outside the allowlist + // before the authenticated forward. The forward then applies the + // real token and the run id. + const sanitizedRequest = { + ...request, + headers: sanitizeSandboxCallbackBridgeHeaders(request.headers), + }; + // Suppress the per-request debug log on the duplex path, so no route + // or query rides a log line here. + return forwardBridgeRequest(sanitizedRequest, options.signal, { + suppressDebugLog: true, + }); + }, + // The duplex path emits only the fixed transport telemetry. It passes no + // free-form logger, so no raw provider error rides a log line here. + telemetry: duplexTelemetry, + // Surface a terminal channel loss on the run log. The broker latches + // the failure on its ordered lifecycle; the host names only the typed, + // closed loss reason here, never the raw provider message. The caller + // reads the latched disposition through `readRunDisposition` at the + // run-disposition seam. + onLoss: (record) => { + void onLog( + "stderr", + `[paperclip] Sandbox duplex channel lost (${typedDuplexLossReason(record.reason)}). The run fails.\n`, + ); + }, + }); + } catch { + // The broker construction failed, so no broker owns the channel. Close + // the channel within the cleanup budget, then select the file bridge. + // The log line names no raw error, so no raw error rides a log line on + // the duplex path. + await closeDuplexChannelWithinBudget(channel, DEFAULT_DUPLEX_CLEANUP_BUDGET_MS); + duplexChannelOpen.fallback("broker_construction_failed"); + await onLog( + "stderr", + "[paperclip] Could not start the sandbox duplex broker. Using the file bridge.\n", + ); + } + if (broker) { + const activeBroker = broker; + activeBroker.start(); + duplexChannelOpen.ready(); + await onLog( + "stdout", + "[paperclip] Sandbox duplex transport ready; serving the host-assigned origin.\n", + ); + // Stream run logs on the duplex path with the same gate and the same + // log line as the file path. The duplex path starts no file-bridge + // worker, so create the log directory before the tail starts. + let duplexRunLogTail: SandboxRunLogTailFactory | null = null; + if (target.transport === "sandbox" && target.streamRunLogs !== false) { + const duplexLogsDir = sandboxCallbackBridgeDirectories(queueDir).logsDir; + await ensureSandboxRunLogDirectory({ + runner, + remoteCwd: target.remoteCwd, + logsDir: duplexLogsDir, + shellCommand, + timeoutMs: bridgeTimeoutMs, + }); + duplexRunLogTail = createSandboxRunLogTailFactory({ + runner, + remoteCwd: target.remoteCwd, + logsDir: duplexLogsDir, + shellCommand, + }); + await onLog("stdout", "[paperclip] Sandbox run log streaming enabled for this run.\n"); + } + return { + env: { + PAPERCLIP_API_URL: sandboxOrigin, + PAPERCLIP_API_KEY: bridgeToken, + PAPERCLIP_API_BRIDGE_MODE: SANDBOX_CALLBACK_BRIDGE_DUPLEX_MODE, + }, + runLogTail: duplexRunLogTail, + // Read the broker's ordered lifecycle latch at the run-disposition + // seam. A loss ordered before an orderly completion reports a failure + // with the typed loss reason; every other state reports a success. + readRunDisposition: (): DuplexBrokerRunDisposition => activeBroker.runDisposition, + // Surface the broker's orderly-completion mark to the run-disposition + // seam. The seam marks the completion for a success-eligible terminal, + // so a teardown loss after the completion stays a normal teardown. + markOrderlyCompletion: (): void => activeBroker.markOrderlyCompletion(), + stop: async () => { + // Close the channel before lease release. The broker sends an orderly + // close and releases the route, then stops the child, so no live + // provider session remains when the caller releases the lease. + await activeBroker.close(); + activeBroker.stop(); + // A channel that did not reach the `closed` state may leave a live + // provider session, so record one session leak. + if (activeBroker.state !== "closed") { + duplexTelemetry.recordSessionLeak(); + } + await bridgeAsset.cleanup(); + }, + }; + } + } + } + } + let server: Awaited> | null = null; let worker: Awaited> | null = null; try { @@ -2266,13 +3176,6 @@ export async function startAdapterExecutionTargetPaperclipBridge(input: { timeoutMs: bridgeTimeoutMs, shellCommand, }); - // PAPERCLIP_BRIDGE_DEBUG opts into verbose stdout logs of every bridge - // proxy request/response. The query string is logged verbatim, so callers - // who pass auth tokens or other sensitive values as query parameters - // should be aware those values appear in the host process's stdout when - // this flag is enabled. Only intended for active debugging in trusted - // environments. - const bridgeDebugEnabled = isBridgeDebugEnabled(process.env); // `startSandboxCallbackBridgeWorker` keeps its awaited queue-directory // setup on the active `bridge.paperclip` step, and runs each request under // the run parent context (see `runWithRuntimeParent` inside that function). @@ -2284,50 +3187,7 @@ export async function startAdapterExecutionTargetPaperclipBridge(input: { maxBodyBytes, getRuntimeParentContext: input.getRuntimeParentContext, runtimeSpan: input.runtimeSpan, - handleRequest: async (request, options) => { - const method = request.method.trim().toUpperCase() || "GET"; - if (bridgeDebugEnabled) { - await onLog( - "stdout", - `[paperclip] Bridge proxy ${method} ${request.path}${request.query ? `?${request.query}` : ""}\n`, - ); - } - const headers = new Headers(); - for (const [key, value] of Object.entries(request.headers)) { - if (value.trim().length === 0) continue; - headers.set(key, value); - } - headers.set("authorization", `Bearer ${hostApiToken}`); - headers.set("x-paperclip-run-id", input.runId); - // Abort the forward when the worker aborts the request (its per-iteration - // timeout or watchdog fired), or after the 30s ceiling, whichever comes - // first. The worker abort lets the bridge fail a hung forward fast - // instead of stranding the request until the 30s ceiling. - const timeoutSignal = AbortSignal.timeout(30_000); - const forwardSignal = options?.signal - ? AbortSignal.any([options.signal, timeoutSignal]) - : timeoutSignal; - const response = await fetch(buildBridgeForwardUrl(hostApiUrl, request), { - method, - headers, - ...(method === "GET" || method === "HEAD" ? {} : { body: request.body }), - signal: forwardSignal, - }); - if (bridgeDebugEnabled) { - await onLog( - "stdout", - `[paperclip] Bridge proxy response ${response.status} for ${method} ${request.path}${request.query ? `?${request.query}` : ""}\n`, - ); - } - const responseBody = await readBridgeForwardResponseBody(response, maxBodyBytes); - const commentMarker = postedIssueCommentLogMarker(method, request.path, response.status, responseBody); - if (commentMarker) await onLog("stdout", commentMarker); - return { - status: response.status, - headers: buildBridgeResponseHeaders(response), - body: responseBody, - }; - }, + handleRequest: async (request, options) => forwardBridgeRequest(request, options?.signal), }); server = await startSandboxCallbackBridgeServer({ runner, diff --git a/packages/adapter-utils/src/sandbox-callback-bridge.ts b/packages/adapter-utils/src/sandbox-callback-bridge.ts index 2c28dfe635..fc5601708a 100644 --- a/packages/adapter-utils/src/sandbox-callback-bridge.ts +++ b/packages/adapter-utils/src/sandbox-callback-bridge.ts @@ -51,7 +51,7 @@ const MAX_BACKSTOP_WRITE_ATTEMPTS = 3; // finish well under the in-sandbox 30s response deadline. const BACKSTOP_WRITE_RETRY_MS = 50; const REMOTE_WRITE_BASE64_CHUNK_SIZE = 32 * 1024; -const SANDBOX_CALLBACK_BRIDGE_ENTRYPOINT = "paperclip-bridge-server.mjs"; +export const SANDBOX_CALLBACK_BRIDGE_ENTRYPOINT = "paperclip-bridge-server.mjs"; const SANDBOX_EXEC_CHANNEL_ENV = "PAPERCLIP_SANDBOX_EXEC_CHANNEL"; const SANDBOX_EXEC_CHANNEL_BRIDGE = "bridge"; @@ -60,7 +60,7 @@ const SANDBOX_EXEC_CHANNEL_BRIDGE = "bridge"; // stdout and resolves one response frame from stdin. The generated `.mjs` // selects the mode from `PAPERCLIP_API_BRIDGE_MODE`. const SANDBOX_CALLBACK_BRIDGE_FILE_MODE = "queue_v1"; -const SANDBOX_CALLBACK_BRIDGE_DUPLEX_MODE = "duplex_v1"; +export const SANDBOX_CALLBACK_BRIDGE_DUPLEX_MODE = "duplex_v1"; // The duplex gateway HTTP wait budget default. The gateway waits this long for a // response frame before it answers the local caller with a 502 timeout. The @@ -1745,6 +1745,7 @@ export async function startSandboxCallbackBridgeServer(input: { */ const DUPLEX_GATEWAY_CODEC_SOURCE = `const DUPLEX_FRAME_VERSION = 1; const DEFAULT_MAX_DUPLEX_FRAME_BYTES = 1000000; +const DEFAULT_MAX_DUPLEX_REQUEST_ID_BYTES = 256; const DUPLEX_NEWLINE_BYTE = 0x0a; const DUPLEX_EMPTY = Buffer.alloc(0); const DUPLEX_RESPONSE_OUTCOMES = new Set(["completed", "indeterminate", "unavailable"]); @@ -1820,6 +1821,9 @@ function duplexValidateRequest(frame) { ) { return duplexFail("malformed_frame", "request frame has a missing or wrong-typed field"); } + if (Buffer.byteLength(frame.id, "utf8") > DEFAULT_MAX_DUPLEX_REQUEST_ID_BYTES) { + return duplexFail("id_too_large", "request frame id exceeds the maximum size"); + } return duplexOk(frame); } @@ -1834,12 +1838,20 @@ function duplexValidateResponse(frame) { ) { return duplexFail("malformed_frame", "response frame has a missing or wrong-typed field"); } + if (Buffer.byteLength(frame.id, "utf8") > DEFAULT_MAX_DUPLEX_REQUEST_ID_BYTES) { + return duplexFail("id_too_large", "response frame id exceeds the maximum size"); + } return duplexOk(frame); } function duplexValidateReady(frame) { - if (typeof frame.address !== "string") { - return duplexFail("malformed_frame", "ready frame has a missing or wrong-typed address"); + if (typeof frame.nonce !== "string") { + return duplexFail("malformed_frame", "ready frame has a missing or wrong-typed nonce"); + } + for (const key of Object.keys(frame)) { + if (key !== "version" && key !== "type" && key !== "nonce") { + return duplexFail("malformed_frame", "ready frame has an unexpected field"); + } } return duplexOk(frame); } @@ -1922,6 +1934,12 @@ const queueDir = process.env.PAPERCLIP_BRIDGE_QUEUE_DIR; const bridgeToken = process.env.PAPERCLIP_BRIDGE_TOKEN; const host = process.env.PAPERCLIP_BRIDGE_HOST || "127.0.0.1"; const port = Number(process.env.PAPERCLIP_BRIDGE_PORT || "0"); +// The host assigns the loopback port and passes it through the launch +// environment. The duplex gateway binds exactly this port; it never selects a +// different one. The host also passes one random per-open nonce here. The +// gateway echoes it in the READY frame so the host correlates READY with this +// channel open. The nonce is a liveness signal, not authentication. +const bridgeNonce = process.env.PAPERCLIP_BRIDGE_NONCE || ""; const pollIntervalMs = Number(process.env.PAPERCLIP_BRIDGE_POLL_INTERVAL_MS || "100"); const responseTimeoutMs = Number( process.env.PAPERCLIP_BRIDGE_RESPONSE_TIMEOUT_MS || @@ -2296,6 +2314,18 @@ function runDuplexGateway() { process.exit(0); }); + // Bind-or-exit. The host assigns a positive loopback port. The gateway binds + // exactly that port. On a non-positive assigned port or a bind failure it + // writes a diagnostic and exits with a nonzero code. It never selects a + // different port, so no untrusted workload can steer the endpoint. + if (!Number.isInteger(port) || port <= 0) { + diag("duplex gateway requires a positive assigned PAPERCLIP_BRIDGE_PORT; got " + String(port)); + process.exit(1); + } + server.on("error", (error) => { + diag("duplex gateway could not bind port " + String(port) + ": " + (error && error.message ? error.message : String(error))); + process.exit(1); + }); server.listen(port, host, () => { const address = server.address(); if (!address || typeof address === "string") { @@ -2303,12 +2333,14 @@ function runDuplexGateway() { process.exit(1); return; } - // The one READY frame carries the validated listener address. Stdout carries + // The gateway sends READY only after the listener binds. READY is a liveness + // signal: it carries the frame version and the echoed nonce, and no address + // data. The host builds the endpoint from its own stored port. Stdout carries // only frames. writeFrame({ version: DUPLEX_FRAME_VERSION, type: "ready", - address: "http://" + host + ":" + address.port, + nonce: bridgeNonce, }); }); } diff --git a/packages/adapters/claude-local/src/server/execute.ts b/packages/adapters/claude-local/src/server/execute.ts index 6dcdd81045..6e927cff21 100644 --- a/packages/adapters/claude-local/src/server/execute.ts +++ b/packages/adapters/claude-local/src/server/execute.ts @@ -15,6 +15,8 @@ import { ensureAdapterExecutionTargetCommandResolvable, ensureAdapterExecutionTargetRuntimeCommandInstalled, prepareAdapterExecutionTargetRuntime, + adapterExecutionTargetDuplexTelemetryRecorder, + adapterExecutionTargetEnablesSandboxDuplexBridge, readAdapterExecutionTarget, resolveAdapterExecutionTargetTimeoutSec, resolveAdapterExecutionTargetCommandForLogs, @@ -681,6 +683,8 @@ export async function execute(ctx: AdapterExecutionContext): Promise string }> { let output = ""; - const session = await openDaytonaDuplexChannelSession(sandbox.process, "cat"); + const session = await openDaytonaDuplexChannelSession(sandbox.process, ["cat"]); session.onData((chunk) => { output += chunk; }); @@ -191,4 +191,101 @@ describeLive("Daytona duplex channel (live)", () => { }, LIVE_TIMEOUT_MS, ); + + it( + "serves repeated channel round trips with no idle polling and no per-request session growth", + async () => { + const live = sandbox!; + const listSessions = live.process.listSessions?.bind(live.process); + const { session, readOutput } = await openCatChannel(live); + try { + // Wait for the raw-mode switch, then take the open baseline. The channel + // uses a pseudo-terminal, not an ordinary session, so the session list + // stays flat through the whole exchange. + await new Promise((resolve) => setTimeout(resolve, 4_000)); + const openSessions = listSessions ? (await listSessions()).length : 0; + + // Send a batch of request-shaped lines over the one open channel and + // measure the round-trip latency of each. The transport writes no queue + // file and runs no exec per line, so the round trip is the stream latency. + const latenciesMs: number[] = []; + const rounds = 10; + for (let index = 0; index < rounds; index += 1) { + const line = `RTT-${index}-${randomUUID()}`; + const baseline = readOutput().length; + const start = Date.now(); + session.write(`${line}\n`); + await waitFor( + readOutput, + (text) => text.slice(baseline).includes(line), + 30_000, + "channel round trip", + ); + latenciesMs.push(Date.now() - start); + } + + // Hold the channel idle. A polling transport would run a provider exec on + // each tick and raise the session count. The duplex channel polls nothing. + await new Promise((resolve) => setTimeout(resolve, 5_000)); + + if (listSessions) { + const afterSessions = (await listSessions()).length; + // Zero idle polls and zero per-request session growth: the session count + // never rose above the open baseline. + expect(afterSessions).toBe(openSessions); + } + + const min = Math.min(...latenciesMs); + const max = Math.max(...latenciesMs); + const avg = Math.round(latenciesMs.reduce((sum, value) => sum + value, 0) / latenciesMs.length); + // eslint-disable-next-line no-console + console.log( + `[duplex-live] channel round-trip latency over ${rounds} requests: min=${min}ms avg=${avg}ms max=${max}ms; idle-poll session growth=0; per-request session growth=0`, + ); + expect(latenciesMs.length).toBe(rounds); + } finally { + await session.close(); + } + }, + LIVE_TIMEOUT_MS, + ); + + it( + "leaves no leaked session and keeps the sandbox usable after a forced mid-flight disconnect", + async () => { + const live = sandbox!; + const listSessions = live.process.listSessions?.bind(live.process); + const baseline = listSessions ? (await listSessions()).length : 0; + + const { session, readOutput } = await openCatChannel(live); + // Wait for the raw-mode switch, then start a write and force a disconnect + // before its echo returns. The abrupt close models a lost provider channel. + await new Promise((resolve) => setTimeout(resolve, 4_000)); + const inFlight = `INFLIGHT-${randomUUID()}`; + session.write(`${inFlight}\n`); + // Close at once, without waiting for the echo. The pending round trip never + // settles through the stream; the channel tears down instead. + await session.close(); + + if (listSessions) { + // The forced disconnect left no leaked provider session. The count is back + // at the baseline that preceded the channel. + const after = (await listSessions()).length; + expect(after).toBe(baseline); + } + // Do not read the in-flight echo; the disconnect settled the run. Reference + // the output length only to keep the reader wired. + expect(readOutput().length).toBeGreaterThanOrEqual(0); + + // The sandbox stays usable: a fresh ordinary command still succeeds. A + // leaked channel would block or corrupt the sandbox. + const token = `post-disconnect-${randomUUID()}`; + const probe = await live.process.executeCommand(`printf %s ${token}`); + expect(probe.exitCode ?? 0).toBe(0); + expect(probe.result ?? "").toContain(token); + // eslint-disable-next-line no-console + console.log("[duplex-live] forced mid-flight disconnect: leaked sessions=0; sandbox usable after=yes"); + }, + LIVE_TIMEOUT_MS, + ); }); diff --git a/packages/plugins/sandbox-providers/daytona/src/duplex-command-stream.test.ts b/packages/plugins/sandbox-providers/daytona/src/duplex-command-stream.test.ts index 476591afd6..1610a3d83e 100644 --- a/packages/plugins/sandbox-providers/daytona/src/duplex-command-stream.test.ts +++ b/packages/plugins/sandbox-providers/daytona/src/duplex-command-stream.test.ts @@ -396,6 +396,85 @@ describe("openDaytonaDuplexChannelSession", () => { expect(process.createOptions?.cwd).toBe("/paperclip-workspace"); }); + + it("ends the channel with the typed write_error and no raw text when a write rejects", async () => { + // The fake handle accepts the launch wrapper write, then rejects every later + // write with a raw provider message. The seam must map the cause to the typed + // reason and never surface the raw text. + const rawProviderText = "RAW-PROVIDER-BROKEN-PIPE-9z8y7x"; + let sends = 0; + let killed = 0; + const handle: DaytonaPtyHandle = { + async waitForConnection(): Promise {}, + async sendInput(): Promise { + sends += 1; + if (sends === 1) return; // the launch wrapper write succeeds. + throw new Error(rawProviderText); + }, + wait(): Promise<{ exitCode?: number; error?: string }> { + return new Promise(() => {}); + }, + async kill(): Promise { + killed += 1; + }, + async disconnect(): Promise {}, + }; + const process: DaytonaPtyProcess = { + async createPty(): Promise { + return handle; + }, + }; + const writeErrors: string[] = []; + const session = await openDaytonaDuplexChannelSession(process, GATEWAY, { + onWriteError: (reason) => writeErrors.push(reason), + }); + + session.write("frame\n"); + // Let the rejected write settle. + await new Promise((resolve) => setImmediate(resolve)); + + // The seam reported exactly the typed reason, ended the channel, and leaked no + // raw provider text. + expect(writeErrors).toEqual(["write_error"]); + expect(killed).toBe(1); + expect(JSON.stringify(writeErrors)).not.toContain(rawProviderText); + }); + + it("reports a write error one time even when more writes reject", async () => { + let sends = 0; + let killed = 0; + const handle: DaytonaPtyHandle = { + async waitForConnection(): Promise {}, + async sendInput(): Promise { + sends += 1; + if (sends === 1) return; + throw new Error("broken"); + }, + wait(): Promise<{ exitCode?: number; error?: string }> { + return new Promise(() => {}); + }, + async kill(): Promise { + killed += 1; + }, + async disconnect(): Promise {}, + }; + const process: DaytonaPtyProcess = { + async createPty(): Promise { + return handle; + }, + }; + const writeErrors: string[] = []; + const session = await openDaytonaDuplexChannelSession(process, GATEWAY, { + onWriteError: (reason) => writeErrors.push(reason), + }); + + session.write("a"); + session.write("b"); + await new Promise((resolve) => setImmediate(resolve)); + + expect(writeErrors).toEqual(["write_error"]); + expect(killed).toBe(1); + }); }); describe("createDaytonaDuplexChannelSessionOpener", () => { diff --git a/packages/plugins/sandbox-providers/daytona/src/duplex-command-stream.ts b/packages/plugins/sandbox-providers/daytona/src/duplex-command-stream.ts index 9651f65714..2aba7b76c8 100644 --- a/packages/plugins/sandbox-providers/daytona/src/duplex-command-stream.ts +++ b/packages/plugins/sandbox-providers/daytona/src/duplex-command-stream.ts @@ -74,6 +74,16 @@ export type DuplexChannelSessionOpener = ( command: readonly string[], ) => Promise; +/** + * The typed, closed reason for a duplex channel write failure. The seam reports + * only this constant, never the raw provider error text. It maps to the host + * telemetry `write_error` loss reason. + */ +export const DAYTONA_DUPLEX_WRITE_ERROR_REASON = "write_error" as const; + +/** The typed reason a rejected host-to-sandbox write reports. */ +export type DaytonaDuplexWriteErrorReason = typeof DAYTONA_DUPLEX_WRITE_ERROR_REASON; + /** The options for the Daytona duplex channel session. */ export interface DaytonaDuplexChannelOptions { /** The working directory for the duplex channel PTY. Defaults to the sandbox default. */ @@ -84,6 +94,13 @@ export interface DaytonaDuplexChannelOptions { * stdout frame stream. Defaults to a per-channel path under `/tmp`. */ diagnosticsPath?: string; + /** + * The write-error seam. The session calls it one time when a host-to-sandbox + * `sendInput` rejects. The session then ends the channel at once. The seam + * carries only the typed {@link DaytonaDuplexWriteErrorReason}; the raw provider + * error never reaches it. + */ + onWriteError?: (reason: DaytonaDuplexWriteErrorReason) => void; } // The terminal size for the duplex channel PTY. The channel carries bytes, not a @@ -183,6 +200,17 @@ export async function openDaytonaDuplexChannelSession( // frame newlines. The diagnostics redirect keeps the stdout frame stream clean. await handle.sendInput(buildDuplexChannelLaunchWrapper(command, diagnosticsPath)); + // Report a host-to-sandbox write failure one time and end the channel at once. + // The seam carries only the typed reason; the raw provider error never leaves + // this scope. The channel end propagates the loss up through the exit. + let writeErrorReported = false; + const endOnWriteError = (): void => { + if (writeErrorReported) return; + writeErrorReported = true; + options?.onWriteError?.(DAYTONA_DUPLEX_WRITE_ERROR_REASON); + void handle.kill().catch(() => undefined); + }; + return { onData(next: (chunk: string) => void): void { listener = next; @@ -194,8 +222,12 @@ export async function openDaytonaDuplexChannelSession( }, write(data: string): void { // Fire the input write. A write error must not throw into the transport, so - // the transport's stream stays the single result path. - void handle.sendInput(data).catch(() => undefined); + // the transport's stream stays the single result path. On a rejected write, + // end the channel at once and report the typed reason. The raw provider + // error never reaches a sink. + void handle.sendInput(data).catch(() => { + endOnWriteError(); + }); }, async wait(): Promise<{ exitCode: number | null }> { const result = await handle.wait(); diff --git a/packages/plugins/sandbox-providers/daytona/src/plugin.test.ts b/packages/plugins/sandbox-providers/daytona/src/plugin.test.ts index 1fb64271b8..d1492ce014 100644 --- a/packages/plugins/sandbox-providers/daytona/src/plugin.test.ts +++ b/packages/plugins/sandbox-providers/daytona/src/plugin.test.ts @@ -201,11 +201,11 @@ describe("Daytona sandbox provider plugin", () => { mockCreate.mockResolvedValue(sandbox); // Capture the data and the exit the worker forwards through `ctx.duplexChannel`. - const dataChunks: Array<{ workerSessionId: string; chunk: string }> = []; + const dataChunks: Array<{ hostRouteId: string; workerSessionId: string; chunk: string }> = []; const restore = __setDaytonaPluginContextForTest({ duplexChannel: { - data: (workerSessionId: string, chunk: string) => - dataChunks.push({ workerSessionId, chunk }), + data: (hostRouteId: string, workerSessionId: string, chunk: string) => + dataChunks.push({ hostRouteId, workerSessionId, chunk }), exit: () => {}, }, } as unknown as PluginContext); @@ -231,6 +231,8 @@ describe("Daytona sandbox provider plugin", () => { command: ["node", "/paperclip/gateway.mjs"], }); expect(open?.workerSessionId).toMatch(/^duplex-/); + // The open reply echoes the host route id, so the host binds the exact pair. + expect(open?.hostRouteId).toBe("route-1"); const workerSessionId = open?.workerSessionId ?? ""; // The launch wrapper sets raw mode with echo off and redirects diagnostics. @@ -239,18 +241,39 @@ describe("Daytona sandbox provider plugin", () => { expect(inputs[0]).toContain("exec 'node' '/paperclip/gateway.mjs'"); expect(inputs[0]).toMatch(/2>'\/tmp\/paperclip-duplex-.+\.log'/); - // A host write reaches the process on the same channel. + // A host write on the exact pair reaches the process on the same channel. await plugin.definition.onDuplexChannelWrite?.({ + hostRouteId: "route-1", workerSessionId, data: '{"version":1,"type":"heartbeat"}\n', }); expect(inputs[1]).toBe('{"version":1,"type":"heartbeat"}\n'); - // Process output reaches the host as a data notification bound to the - // worker session id. - ptyOnData?.(new TextEncoder().encode('{"version":1,"type":"ready","address":"127.0.0.1:1"}\n')); + // A write whose pair does not match the bound entry applies no bytes. The + // worker acts only on the exact live pair. + const inputsBeforeForeign = inputs.length; + await plugin.definition.onDuplexChannelWrite?.({ + hostRouteId: "route-foreign", + workerSessionId, + data: "foreign\n", + }); + expect(inputs.length).toBe(inputsBeforeForeign); + // A stop whose pair does not match the bound entry stops nothing. + const killedBeforeForeign = killed; + await plugin.definition.onDuplexChannelStop?.({ + hostRouteId: "route-foreign", + workerSessionId, + }); + expect(killed).toBe(killedBeforeForeign); + + // Process output reaches the host as a data notification bound to the exact + // pair, so it echoes the host route id and the worker session id. + (ptyOnData as ((data: Uint8Array) => void) | null)?.( + new TextEncoder().encode('{"version":1,"type":"ready","address":"127.0.0.1:1"}\n'), + ); expect(dataChunks).toEqual([ { + hostRouteId: "route-1", workerSessionId, chunk: '{"version":1,"type":"ready","address":"127.0.0.1:1"}\n', }, @@ -272,6 +295,7 @@ describe("Daytona sandbox provider plugin", () => { // process. const inputsBefore = inputs.length; await plugin.definition.onDuplexChannelWrite?.({ + hostRouteId: "route-1", workerSessionId, data: "late\n", }); @@ -2758,7 +2782,7 @@ describe("Daytona sandbox provider plugin", () => { expect(sandbox.delete).not.toHaveBeenCalled(); resolveFirstExecute(); - await cancelPromise.then(() => { + await cancelPromise!.then(() => { cancelResolved = true; }); await expect(queuedExecutePromise).rejects.toThrow(/no longer active/); @@ -4496,13 +4520,13 @@ describe("daytona native file-sync hooks", () => { // Both commands ran, VERBATIM (first arg is the exact authored string — the // provider never rewrote/concatenated a shell fragment onto it: C1/C3). const findCall = (cmd: string) => - sandbox.process.executeCommand.mock.calls.find(([c]: [string]) => c === cmd); + sandbox.process.executeCommand.mock.calls.find(([c]) => c === cmd); expect(findCall("codex-auth-merge --first")).toBeDefined(); expect(findCall("chmod 600 config.txt")).toBeDefined(); // Ordered: the first command's exec precedes the second's (C4 array order). const orderOf = (cmd: string) => { - const idx = sandbox.process.executeCommand.mock.calls.findIndex(([c]: [string]) => c === cmd); + const idx = sandbox.process.executeCommand.mock.calls.findIndex(([c]) => c === cmd); return sandbox.process.executeCommand.mock.invocationCallOrder[idx]; }; expect(orderOf("codex-auth-merge --first")).toBeLessThan(orderOf("chmod 600 config.txt")); @@ -4551,7 +4575,7 @@ describe("daytona native file-sync hooks", () => { // Fail-fast: the command after the failing one never executed. expect( - sandbox.process.executeCommand.mock.calls.some(([c]: [string]) => c === "should-not-run"), + sandbox.process.executeCommand.mock.calls.some(([c]) => c === "should-not-run"), ).toBe(false); }); @@ -4598,7 +4622,7 @@ describe("daytona native file-sync hooks", () => { expect(sandbox.fs.uploadFiles).toHaveBeenCalledTimes(1); const [uploads] = sandbox.fs.uploadFiles.mock.calls[0] as [Array<{ source: string; destination: string }>]; expect(uploads).toHaveLength(2); - const mvCalls = sandbox.process.executeCommand.mock.calls.filter(([cmd]: [string]) => + const mvCalls = sandbox.process.executeCommand.mock.calls.filter(([cmd]) => String(cmd).includes("mv -f"), ); expect(mvCalls).toHaveLength(1); @@ -4606,7 +4630,7 @@ describe("daytona native file-sync hooks", () => { // Both extract commands ran, in array order, AFTER the upload (git first). const orderOf = (cmd: string) => { - const idx = sandbox.process.executeCommand.mock.calls.findIndex(([c]: [string]) => c === cmd); + const idx = sandbox.process.executeCommand.mock.calls.findIndex(([c]) => c === cmd); return sandbox.process.executeCommand.mock.invocationCallOrder[idx]; }; expect(orderOf(gitExtract)).toBeLessThan(orderOf(overlayExtract)); @@ -4702,7 +4726,7 @@ describe("daytona native file-sync hooks", () => { // Fail-fast: the overlay extract and the remove-deleted command never ran. const ran = (cmd: string) => - sandbox.process.executeCommand.mock.calls.some(([c]: [string]) => c === cmd); + sandbox.process.executeCommand.mock.calls.some(([c]) => c === cmd); expect(ran("git-history-extract")).toBe(true); expect(ran("workspace-overlay-extract")).toBe(false); expect(ran("remove-deleted-paths")).toBe(false); @@ -4733,7 +4757,7 @@ describe("daytona native file-sync hooks", () => { }), ).rejects.toThrow(/not a confined absolute path|escapes the workspace remote dir/); // The command never ran — lexical confinement rejected it before exec. - expect(sandbox.process.executeCommand.mock.calls.some(([c]: [string]) => c === "run-me")).toBe( + expect(sandbox.process.executeCommand.mock.calls.some(([c]) => c === "run-me")).toBe( false, ); } @@ -4772,7 +4796,7 @@ describe("daytona native file-sync hooks", () => { ], }), ).rejects.toThrow(/symlink-escape guard|command failed/i); - expect(sandbox.process.executeCommand.mock.calls.some(([c]: [string]) => c === "run-me")).toBe( + expect(sandbox.process.executeCommand.mock.calls.some(([c]) => c === "run-me")).toBe( false, ); }); diff --git a/packages/plugins/sandbox-providers/daytona/src/plugin.ts b/packages/plugins/sandbox-providers/daytona/src/plugin.ts index 684ec2b816..304947e562 100644 --- a/packages/plugins/sandbox-providers/daytona/src/plugin.ts +++ b/packages/plugins/sandbox-providers/daytona/src/plugin.ts @@ -2823,43 +2823,49 @@ const plugin = definePlugin({ daytonaDuplexChannelByRoute.set(params.hostRouteId, entry); daytonaDuplexChannelBySession.set(workerSessionId, entry); // Register the data listener before the first write, so no early data chunk is - // lost. The client stamps the worker session id, so the host binds the data to - // the open route. + // lost. The client echoes the host route id and the worker session id, so the + // host routes the data to the exact live pair. session.onData((chunk) => { - pluginContext?.duplexChannel.data(workerSessionId, chunk); + pluginContext?.duplexChannel.data(entry.hostRouteId, workerSessionId, chunk); }); // Forward the child exit one time. The host resolves the open route on it. void session.wait().then( - (result) => pluginContext?.duplexChannel.exit(workerSessionId, result.exitCode), - () => pluginContext?.duplexChannel.exit(workerSessionId, null), + (result) => pluginContext?.duplexChannel.exit(entry.hostRouteId, workerSessionId, result.exitCode), + () => pluginContext?.duplexChannel.exit(entry.hostRouteId, workerSessionId, null), ); - return { workerSessionId }; + // Echo the host route id on the reply, so the host binds the exact pair. + return { hostRouteId: params.hostRouteId, workerSessionId }; }, - // Write host input to an open duplex channel, keyed by the worker session id. - // Drop the input for an unknown session. + // Write host input to an open duplex channel. Act only on the exact live pair. + // A write whose pair does not match the bound entry applies no bytes. async onDuplexChannelWrite(params) { const entry = daytonaDuplexChannelBySession.get(params.workerSessionId); - if (!entry) return; + if (!entry || entry.hostRouteId !== params.hostRouteId) return; entry.session.write(params.data); }, - // Stop an open duplex channel child, keyed by the worker session id. + // Stop an open duplex channel child. Act only on the exact live pair. A stop + // whose pair does not match the bound entry stops nothing. async onDuplexChannelStop(params) { const entry = daytonaDuplexChannelBySession.get(params.workerSessionId); - if (!entry) return; + if (!entry || entry.hostRouteId !== params.hostRouteId) return; entry.session.kill(); }, - // Close an open duplex channel by the host route id and acknowledge the close - // with the same identifier. The close is idempotent: it returns the - // acknowledgement even when the entry is already gone, so the host confirms the - // channel is closed. The worker never keys the close on the worker session id. + // Close an open duplex channel by the host route id and acknowledge the close. + // The host route id is the authoritative close key, so a pre-bind close with a + // lost open reply still closes the channel. On a bound close the acknowledgement + // echoes the worker session id too, so the host verifies the exact pair. The + // close is idempotent: it returns the acknowledgement even when the entry is + // already gone, so the host confirms the channel is closed. async onDuplexChannelClose(params) { const entry = daytonaDuplexChannelByRoute.get(params.hostRouteId); if (entry) { + const boundWorkerSessionId = entry.workerSessionId; forgetDaytonaDuplexChannel(entry); await entry.session.close().catch(() => undefined); + return { hostRouteId: params.hostRouteId, workerSessionId: boundWorkerSessionId }; } return { hostRouteId: params.hostRouteId }; }, diff --git a/packages/plugins/sandbox-providers/daytona/tsconfig.json b/packages/plugins/sandbox-providers/daytona/tsconfig.json index d1a61f000a..d7b036c118 100644 --- a/packages/plugins/sandbox-providers/daytona/tsconfig.json +++ b/packages/plugins/sandbox-providers/daytona/tsconfig.json @@ -9,6 +9,5 @@ "@paperclipai/plugin-sdk": ["../../sdk/dist/index.d.ts"] } }, - "include": ["src"], - "exclude": ["src/**/*.test.ts"] + "include": ["src"] } diff --git a/packages/plugins/sdk/src/protocol.duplex-channel.test.ts b/packages/plugins/sdk/src/protocol.duplex-channel.test.ts index b9fb2529d4..d8c40f7e4c 100644 --- a/packages/plugins/sdk/src/protocol.duplex-channel.test.ts +++ b/packages/plugins/sdk/src/protocol.duplex-channel.test.ts @@ -35,11 +35,21 @@ describe("duplex channel request schemas", () => { providerLeaseId: "lease-1", command: ["paperclip-bridge"], }; - const reply: PluginDuplexChannelOpenResult = { workerSessionId: "ws-1" }; + const reply: PluginDuplexChannelOpenResult = { + hostRouteId: "route-1", + workerSessionId: "ws-1", + }; expect(open.hostRouteId).toBe("route-1"); + expect(reply.hostRouteId).toBe("route-1"); expect(reply.workerSessionId).toBe("ws-1"); }); + it("rejects an open reply that omits the echoed host route identifier", () => { + // @ts-expect-error — the open reply echoes the host route identifier. + const reply: PluginDuplexChannelOpenResult = { workerSessionId: "ws-1" }; + expect(reply).toBeDefined(); + }); + it("rejects an open request that omits the host route identifier", () => { // @ts-expect-error — hostRouteId is required. const open: PluginDuplexChannelOpenParams = { @@ -65,38 +75,68 @@ describe("duplex channel request schemas", () => { expect(open).toBeDefined(); }); - it("accepts a valid write request", () => { + it("accepts a valid write request that carries the exact pair", () => { const write: PluginDuplexChannelWriteParams = { + hostRouteId: "route-1", workerSessionId: "ws-1", data: "payload", }; expect(write.data).toBe("payload"); }); - it("rejects a write request that omits the data", () => { - // @ts-expect-error — data is required. - const write: PluginDuplexChannelWriteParams = { workerSessionId: "ws-1" }; + it("rejects a write request that omits the host route identifier", () => { + // @ts-expect-error — hostRouteId is required on a post-bind write. + const write: PluginDuplexChannelWriteParams = { + workerSessionId: "ws-1", + data: "payload", + }; expect(write).toBeDefined(); }); - it("accepts a valid stop request", () => { - const stop: PluginDuplexChannelStopParams = { workerSessionId: "ws-1" }; + it("rejects a write request that omits the data", () => { + // @ts-expect-error — data is required. + const write: PluginDuplexChannelWriteParams = { + hostRouteId: "route-1", + workerSessionId: "ws-1", + }; + expect(write).toBeDefined(); + }); + + it("accepts a valid stop request that carries the exact pair", () => { + const stop: PluginDuplexChannelStopParams = { + hostRouteId: "route-1", + workerSessionId: "ws-1", + }; expect(stop.workerSessionId).toBe("ws-1"); }); + it("rejects a stop request that omits the host route identifier", () => { + // @ts-expect-error — hostRouteId is required on a post-bind stop. + const stop: PluginDuplexChannelStopParams = { workerSessionId: "ws-1" }; + expect(stop).toBeDefined(); + }); + it("rejects a stop request that omits the worker session identifier", () => { // @ts-expect-error — workerSessionId is required. - const stop: PluginDuplexChannelStopParams = {}; + const stop: PluginDuplexChannelStopParams = { hostRouteId: "route-1" }; expect(stop).toBeDefined(); }); - it("accepts a close request keyed only by the host route identifier", () => { + it("accepts a pre-bind close request and a route-only acknowledgement", () => { const close: PluginDuplexChannelCloseParams = { hostRouteId: "route-1" }; const reply: PluginDuplexChannelCloseResult = { hostRouteId: "route-1" }; expect(close.hostRouteId).toBe("route-1"); expect(reply.hostRouteId).toBe("route-1"); }); + it("accepts a bound close acknowledgement that echoes both identifiers", () => { + const reply: PluginDuplexChannelCloseResult = { + hostRouteId: "route-1", + workerSessionId: "ws-1", + }; + expect(reply.workerSessionId).toBe("ws-1"); + }); + it("accepts a close request that also carries the worker session identifier", () => { const close: PluginDuplexChannelCloseParams = { hostRouteId: "route-1", @@ -118,26 +158,41 @@ describe("duplex channel notification schemas", () => { expect(DUPLEX_CHANNEL_EXIT_NOTIFICATION).toBe("duplexChannel.exit"); }); - it("accepts a valid data notification", () => { + it("accepts a valid data notification that carries the exact pair", () => { const data: PluginDuplexChannelDataParams = { + hostRouteId: "route-1", workerSessionId: "ws-1", chunk: "output bytes", }; expect(data.chunk).toBe("output bytes"); }); + it("rejects a data notification that omits the host route identifier", () => { + // @ts-expect-error — hostRouteId is required on a notification. + const data: PluginDuplexChannelDataParams = { + workerSessionId: "ws-1", + chunk: "output bytes", + }; + expect(data).toBeDefined(); + }); + it("rejects a data notification that omits the chunk", () => { // @ts-expect-error — chunk is required. - const data: PluginDuplexChannelDataParams = { workerSessionId: "ws-1" }; + const data: PluginDuplexChannelDataParams = { + hostRouteId: "route-1", + workerSessionId: "ws-1", + }; expect(data).toBeDefined(); }); it("accepts an exit notification with a numeric code and with null", () => { const exit: PluginDuplexChannelExitParams = { + hostRouteId: "route-1", workerSessionId: "ws-1", exitCode: 0, }; const exitNull: PluginDuplexChannelExitParams = { + hostRouteId: "route-1", workerSessionId: "ws-1", exitCode: null, }; @@ -145,8 +200,18 @@ describe("duplex channel notification schemas", () => { expect(exitNull.exitCode).toBeNull(); }); + it("rejects an exit notification that omits the host route identifier", () => { + // @ts-expect-error — hostRouteId is required on a notification. + const exit: PluginDuplexChannelExitParams = { + workerSessionId: "ws-1", + exitCode: 0, + }; + expect(exit).toBeDefined(); + }); + it("rejects an exit notification with a non-numeric, non-null exit code", () => { const exit: PluginDuplexChannelExitParams = { + hostRouteId: "route-1", workerSessionId: "ws-1", // @ts-expect-error — exitCode must be a number or null. exitCode: "0", diff --git a/packages/plugins/sdk/src/protocol.ts b/packages/plugins/sdk/src/protocol.ts index f81c2bd0dc..4c98b41088 100644 --- a/packages/plugins/sdk/src/protocol.ts +++ b/packages/plugins/sdk/src/protocol.ts @@ -1119,22 +1119,28 @@ export interface PluginDuplexChannelOpenParams { command: readonly string[]; } -/** The open reply. It returns the worker session identifier for data binding only. */ +/** The open reply. It echoes the host route identifier and returns the worker session identifier. */ export interface PluginDuplexChannelOpenResult { + /** The host route identifier the open request carried. The worker echoes it, so the host binds the exact pair. */ + hostRouteId: string; /** The worker session identifier. It binds the data and the exit notification only. */ workerSessionId: string; } -/** The write request. It carries the worker session identifier and the raw input bytes. */ +/** The write request. It carries the exact route pair and the raw input bytes. */ export interface PluginDuplexChannelWriteParams { + /** The host route identifier the open request carried. The worker acts only on the exact live pair. */ + hostRouteId: string; /** The worker session identifier that the open reply returned. */ workerSessionId: string; /** The raw input bytes to write to the channel. */ data: string; } -/** The stop request. It carries the worker session identifier. */ +/** The stop request. It carries the exact route pair. */ export interface PluginDuplexChannelStopParams { + /** The host route identifier the open request carried. The worker acts only on the exact live pair. */ + hostRouteId: string; /** The worker session identifier that the open reply returned. */ workerSessionId: string; } @@ -1155,14 +1161,22 @@ export interface PluginDuplexChannelCloseParams { workerSessionId?: string; } -/** The close reply. It acknowledges the close and carries the same host route identifier. */ +/** The close reply. It acknowledges the close and echoes the route identifiers. */ export interface PluginDuplexChannelCloseResult { /** The close acknowledgement. It carries the same host route identifier the close sent. */ hostRouteId: string; + /** + * The bound worker session identifier. The worker echoes it on a bound close, + * so the host verifies the exact pair. It is absent on a pre-bind route-only + * close, where no session bound yet. + */ + workerSessionId?: string; } /** The worker→host duplex channel data notification parameters. */ export interface PluginDuplexChannelDataParams { + /** The host route identifier the open request carried. The worker echoes it, so the host routes the exact pair. */ + hostRouteId: string; /** The worker session identifier that the open reply returned. */ workerSessionId: string; /** The raw channel output bytes. */ @@ -1171,6 +1185,8 @@ export interface PluginDuplexChannelDataParams { /** The worker→host duplex channel exit notification parameters. */ export interface PluginDuplexChannelExitParams { + /** The host route identifier the open request carried. The worker echoes it, so the host routes the exact pair. */ + hostRouteId: string; /** The worker session identifier that the open reply returned. */ workerSessionId: string; /** The child exit code, or null when the child ended with no code. */ diff --git a/packages/plugins/sdk/src/testing.ts b/packages/plugins/sdk/src/testing.ts index 43ce72f658..bd5ad5c21c 100644 --- a/packages/plugins/sdk/src/testing.ts +++ b/packages/plugins/sdk/src/testing.ts @@ -2487,10 +2487,10 @@ export function createTestHarness(options: TestHarnessOptions): TestHarness { }, }, duplexChannel: { - data(_workerSessionId: string, _chunk: string) { + data(_hostRouteId: string, _workerSessionId: string, _chunk: string) { // No-op in test harness — the host duplex route is not wired here. }, - exit(_workerSessionId: string, _exitCode: number | null) { + exit(_hostRouteId: string, _workerSessionId: string, _exitCode: number | null) { // No-op in test harness — the host duplex route is not wired here. }, }, diff --git a/packages/plugins/sdk/src/types.ts b/packages/plugins/sdk/src/types.ts index 6eda319ba5..8ea10b964e 100644 --- a/packages/plugins/sdk/src/types.ts +++ b/packages/plugins/sdk/src/types.ts @@ -2054,17 +2054,19 @@ export interface PluginDuplexChannelClient { /** * Deliver one raw data chunk of a persistent duplex channel. * + * @param hostRouteId - The host route identifier the open request carried. The worker echoes it, so the host routes the exact pair. * @param workerSessionId - The worker session identifier the open reply returned. * @param chunk - The raw channel output text. */ - data(workerSessionId: string, chunk: string): void; + data(hostRouteId: string, workerSessionId: string, chunk: string): void; /** * Deliver the child exit of a persistent duplex channel. * + * @param hostRouteId - The host route identifier the open request carried. The worker echoes it, so the host routes the exact pair. * @param workerSessionId - The worker session identifier the open reply returned. * @param exitCode - The child exit code, or null when the child ended with no code. */ - exit(workerSessionId: string, exitCode: number | null): void; + exit(hostRouteId: string, workerSessionId: string, exitCode: number | null): void; } // --------------------------------------------------------------------------- diff --git a/packages/plugins/sdk/src/worker-rpc-host.ts b/packages/plugins/sdk/src/worker-rpc-host.ts index 4c00884e11..901410a841 100644 --- a/packages/plugins/sdk/src/worker-rpc-host.ts +++ b/packages/plugins/sdk/src/worker-rpc-host.ts @@ -1403,23 +1403,26 @@ export function startWorkerRpcHost(options: WorkerRpcHostOptions): WorkerRpcHost }, duplexChannel: { - data(workerSessionId: string, chunk: string): void { + data(hostRouteId: string, workerSessionId: string, chunk: string): void { // Forward one raw data chunk of a persistent duplex channel. The - // notification carries the worker session identifier, so the host binds - // the chunk to the open route by that identifier while the route is - // open. The host drops an unknown or a mismatched identifier and never - // logs the raw bytes. This notification carries no invocation id, + // notification echoes the host route identifier and the worker session + // identifier, so the host routes the chunk to the exact live pair while + // the route is open. The host drops an unknown or a mismatched pair and + // never logs the raw bytes. This notification carries no invocation id, // because it fires after the open reply returns. + if (typeof hostRouteId !== "string" || hostRouteId.length === 0) return; if (typeof workerSessionId !== "string" || workerSessionId.length === 0) return; if (typeof chunk !== "string" || chunk.length === 0) return; - notifyHost(DUPLEX_CHANNEL_DATA_NOTIFICATION, { workerSessionId, chunk }); + notifyHost(DUPLEX_CHANNEL_DATA_NOTIFICATION, { hostRouteId, workerSessionId, chunk }); }, - exit(workerSessionId: string, exitCode: number | null): void { + exit(hostRouteId: string, workerSessionId: string, exitCode: number | null): void { // Forward the child exit of a persistent duplex channel. The host - // resolves the open route's wait promise by the worker session - // identifier while the route is open. + // resolves the open route's wait promise by the exact live pair while + // the route is open. + if (typeof hostRouteId !== "string" || hostRouteId.length === 0) return; if (typeof workerSessionId !== "string" || workerSessionId.length === 0) return; notifyHost(DUPLEX_CHANNEL_EXIT_NOTIFICATION, { + hostRouteId, workerSessionId, exitCode: typeof exitCode === "number" ? exitCode : null, }); diff --git a/packages/shared/src/telemetry/README.md b/packages/shared/src/telemetry/README.md index 6c6f8d03c7..159501a8dc 100644 --- a/packages/shared/src/telemetry/README.md +++ b/packages/shared/src/telemetry/README.md @@ -304,6 +304,73 @@ record, so a worker can never forge a parent. The host validates the trace context the worker sends no span, so the whole provider-span path is a no-op. +## Sandbox Duplex Transport Telemetry + +Paperclip opens a fixed observability surface for the sandbox duplex transport. +This surface is separate from the first-party events and from the startup spans +above. The generated telemetry contract does not cover it, so this section is its +canonical contract. The code owner is +`packages/adapter-utils/src/duplex-telemetry.ts`. That module holds each name and +each enum value as a literal constant, so the surface never drifts. + +The surface is opt-in. The host injects a recorder that binds the span to the +OTel tracer, the counter to the guarded counter store in +`server/src/services/tool-runtime-metrics.ts`, and the event to the run-events +bridge. The default recorder is a no-op, so the whole surface stays inert until +the host binds a real recorder. Every recorder call sits inside an error swallow, +so a telemetry failure never breaks the request path. + +The surface carries no user content. No route, no query, no request body, no +token, and no raw identifier rides a span, a counter, or an event. Each record +carries only the closed dimension keys below and, for the request span, a +latency. The `provider` dimension carries only the allowlisted public value +`daytona`. Any other plugin key maps to `other` before the record reaches a sink, +so a raw plugin key never reaches a span attribute, a counter label, or an event +field. + +### Spans + +| Span | Scope | Latency | +| --- | --- | --- | +| `sandbox.duplex.channel_open` | One duplex channel-open attempt. The `outcome` dimension is `ok` when the channel opened and readiness passed, or `error` when the open or readiness failed. | none | +| `sandbox.duplex.request` | One duplex request the broker forwarded to the host. | The request latency in milliseconds. | + +### Event + +| Event | Scope | +| --- | --- | +| `sandbox.duplex.transport` | The host emits it at each transport boundary: a ready duplex channel, a fallback to the file bridge, and a terminal channel loss. Its dimensions record the boundary. | + +### Counters + +| Counter | Scope | +| --- | --- | +| `sandbox_duplex_channel_open_total` | One successful duplex channel open. | +| `sandbox_duplex_fallback_total` | One fallback to the file bridge. The `fallback_reason` dimension records the cause. | +| `sandbox_duplex_loss_total` | One terminal duplex channel loss. The `loss_class` dimension records the phase. | +| `sandbox_duplex_session_leak_total` | One leaked provider session at teardown. | + +### Dimension keys + +Counters carry no dimension labels. The guarded counter store keys each counter +on `(companyId, metric)` with no label column, so the `fallback_reason` and +`loss_class` values fold into the counter metric name instead. The full closed +dimension set below rides only the spans and the `sandbox.duplex.transport` +event, which use only these closed keys. A test asserts the exact set, so a new +key never reaches a sink by accident. + +| Key | Type | Optional | Value set | +| --- | --- | --- | --- | +| `provider` | string | no | `daytona`, or `other` for any other plugin key. | +| `transport` | string | no | `duplex` or `file`. A fallback record uses `file`; every other record uses `duplex`. | +| `outcome` | string | yes | `ok` or `error`. | +| `fallback_reason` | string | yes | `gate_off`, `capability_absent`, `open_failed`, `ready_invalid`, `ready_nonce_mismatch`, `ready_timeout`, or `contaminated`. It rides only a fallback record. | +| `loss_class` | string | yes | `pre_dispatch` or `post_dispatch`, relative to the first request dispatch. It rides only a loss record. | + +To add a name or an enum value, extend the literal constant in +`duplex-telemetry.ts` first, then update the test that asserts the closed set. +Keep every dimension low-cardinality and free of user content. + ## Sandbox Startup Run-Log Event Paperclip writes one `run.startup.step` event to the run log for each bring-up diff --git a/server/src/__tests__/duplex-telemetry-recorder.test.ts b/server/src/__tests__/duplex-telemetry-recorder.test.ts new file mode 100644 index 0000000000..4edddcedce --- /dev/null +++ b/server/src/__tests__/duplex-telemetry-recorder.test.ts @@ -0,0 +1,138 @@ +import { describe, expect, it } from "vitest"; + +import { + DUPLEX_COUNTER_CHANNEL_OPEN_TOTAL, + DUPLEX_COUNTER_FALLBACK_TOTAL, + DUPLEX_COUNTER_LOSS_TOTAL, + DUPLEX_SPAN_CHANNEL_OPEN, + DUPLEX_SPAN_REQUEST, + DUPLEX_TRANSPORT_EVENT, +} from "@paperclipai/adapter-utils/duplex-telemetry"; +import { + createHostDuplexTelemetryRecorder, + foldDuplexCounterMetric, + type DuplexTelemetrySpan, + type DuplexTelemetryTracer, +} from "../services/duplex-telemetry-recorder.js"; + +// A recording tracer that captures each span's name, attributes, and end time. +function createRecordingTracer(): { + tracer: DuplexTelemetryTracer; + spans: Array<{ name: string; startTime?: number; attributes: Record; endTime?: number }>; +} { + const spans: Array<{ name: string; startTime?: number; attributes: Record; endTime?: number }> = []; + const tracer: DuplexTelemetryTracer = { + startSpan(name, options) { + const record = { name, startTime: options?.startTime, attributes: {} as Record, endTime: undefined as number | undefined }; + spans.push(record); + const span: DuplexTelemetrySpan = { + setAttribute(key, value) { + record.attributes[key] = value; + }, + end(endTime) { + record.endTime = endTime; + }, + }; + return span; + }, + }; + return { tracer, spans }; +} + +describe("createHostDuplexTelemetryRecorder", () => { + it("records the channel-open span with only the closed dimension keys", () => { + const { tracer, spans } = createRecordingTracer(); + const recorder = createHostDuplexTelemetryRecorder({ + tracer, + incrementCounter: () => {}, + emitTransportEvent: () => {}, + now: () => 1_000, + }); + + recorder.recordSpan({ + name: DUPLEX_SPAN_CHANNEL_OPEN, + dimensions: { provider: "daytona", transport: "duplex", outcome: "ok" }, + }); + + expect(spans).toHaveLength(1); + expect(spans[0].name).toBe(DUPLEX_SPAN_CHANNEL_OPEN); + expect(spans[0].attributes).toEqual({ provider: "daytona", transport: "duplex", outcome: "ok" }); + // The channel-open span carries no latency, so it opens and ends at the same instant. + expect(spans[0].startTime).toBe(1_000); + expect(spans[0].endTime).toBe(1_000); + }); + + it("makes the request span duration equal the measured latency", () => { + const { tracer, spans } = createRecordingTracer(); + const recorder = createHostDuplexTelemetryRecorder({ + tracer, + incrementCounter: () => {}, + emitTransportEvent: () => {}, + now: () => 5_000, + }); + + recorder.recordSpan({ + name: DUPLEX_SPAN_REQUEST, + dimensions: { provider: "other", transport: "duplex", outcome: "error" }, + latencyMs: 250, + }); + + expect(spans[0].startTime).toBe(4_750); + expect(spans[0].endTime).toBe(5_000); + }); + + it("folds the discriminating dimension into the counter metric name", () => { + expect( + foldDuplexCounterMetric({ + metric: DUPLEX_COUNTER_CHANNEL_OPEN_TOTAL, + dimensions: { provider: "daytona", transport: "duplex", outcome: "ok" }, + }), + ).toBe(DUPLEX_COUNTER_CHANNEL_OPEN_TOTAL); + expect( + foldDuplexCounterMetric({ + metric: DUPLEX_COUNTER_FALLBACK_TOTAL, + dimensions: { provider: "daytona", transport: "file", outcome: "error", fallback_reason: "gate_off" }, + }), + ).toBe(`${DUPLEX_COUNTER_FALLBACK_TOTAL}.gate_off`); + expect( + foldDuplexCounterMetric({ + metric: DUPLEX_COUNTER_LOSS_TOTAL, + dimensions: { provider: "daytona", transport: "duplex", outcome: "error", loss_class: "post_dispatch" }, + }), + ).toBe(`${DUPLEX_COUNTER_LOSS_TOTAL}.post_dispatch`); + }); + + it("forwards the counter through the guarded sink with the folded metric", () => { + const metrics: string[] = []; + const recorder = createHostDuplexTelemetryRecorder({ + tracer: createRecordingTracer().tracer, + incrementCounter: (metric) => metrics.push(metric), + emitTransportEvent: () => {}, + }); + + recorder.incrementCounter({ + metric: DUPLEX_COUNTER_FALLBACK_TOTAL, + dimensions: { provider: "daytona", transport: "file", outcome: "error", fallback_reason: "ready_timeout" }, + }); + + expect(metrics).toEqual([`${DUPLEX_COUNTER_FALLBACK_TOTAL}.ready_timeout`]); + }); + + it("forwards the transport event with its name and dimensions", () => { + const events: Array<{ name: string; dimensions: Record }> = []; + const recorder = createHostDuplexTelemetryRecorder({ + tracer: createRecordingTracer().tracer, + incrementCounter: () => {}, + emitTransportEvent: (event) => events.push(event), + }); + + recorder.emitEvent({ + name: DUPLEX_TRANSPORT_EVENT, + dimensions: { provider: "daytona", transport: "duplex", outcome: "ok" }, + }); + + expect(events).toEqual([ + { name: DUPLEX_TRANSPORT_EVENT, dimensions: { provider: "daytona", transport: "duplex", outcome: "ok" } }, + ]); + }); +}); diff --git a/server/src/__tests__/environment-execution-target-duplex-kill-switch.test.ts b/server/src/__tests__/environment-execution-target-duplex-kill-switch.test.ts new file mode 100644 index 0000000000..655017221e --- /dev/null +++ b/server/src/__tests__/environment-execution-target-duplex-kill-switch.test.ts @@ -0,0 +1,89 @@ +import { beforeEach, describe, expect, it, vi } from "vitest"; + +const { mockResolveEnvironmentDriverConfigForRuntime } = vi.hoisted(() => ({ + mockResolveEnvironmentDriverConfigForRuntime: vi.fn(), +})); + +vi.mock("../services/environment-config.js", () => ({ + resolveEnvironmentDriverConfigForRuntime: mockResolveEnvironmentDriverConfigForRuntime, +})); + +import { resolveEnvironmentExecutionTarget } from "../services/environment-execution-target.js"; +import type { EnvironmentRuntimeService } from "../services/environment-runtime.js"; + +// Build the host sandbox target for one duplex kill-switch state. The fake +// runtime returns the given bridge input, or throws when `rejectRead` is set. +// When `omitMethod` is set the runtime has no `readSandboxDuplexBridgeInput`, so +// the test proves the stamp fails closed on a runtime that predates the read. +async function buildSandboxTarget(input: { + bridgeInput?: { enableDuplexBridge: boolean }; + rejectRead?: boolean; + omitMethod?: boolean; +}) { + mockResolveEnvironmentDriverConfigForRuntime.mockResolvedValue({ + driver: "sandbox", + config: { provider: "daytona", timeoutMs: 30_000 }, + }); + + const readSandboxDuplexBridgeInput = vi.fn(async () => { + if (input.rejectRead) { + throw new Error("settings read failed"); + } + return input.bridgeInput ?? { enableDuplexBridge: false }; + }); + const runtime: Record = { + supportsSync: () => false, + effectiveSandboxCapabilities: vi.fn(async () => null), + }; + if (!input.omitMethod) { + runtime.readSandboxDuplexBridgeInput = readSandboxDuplexBridgeInput; + } + const environmentRuntime = runtime as unknown as EnvironmentRuntimeService; + + const target = await resolveEnvironmentExecutionTarget({ + db: {} as never, + companyId: "company-1", + adapterType: "codex_local", + environment: { id: "env-1", driver: "sandbox", config: { provider: "daytona" } }, + leaseId: "lease-1", + leaseMetadata: { remoteCwd: "/work" }, + lease: { id: "lease-1", leasePolicy: "reuse_by_environment" } as never, + environmentRuntime, + }); + + if (target?.kind !== "remote" || target.transport !== "sandbox") { + throw new Error("expected a sandbox target"); + } + return { target, readSandboxDuplexBridgeInput }; +} + +describe("resolveEnvironmentExecutionTarget duplex kill switch", () => { + beforeEach(() => { + mockResolveEnvironmentDriverConfigForRuntime.mockReset(); + }); + + it("stamps the kill switch true when the instance setting is on", async () => { + const { target, readSandboxDuplexBridgeInput } = await buildSandboxTarget({ + bridgeInput: { enableDuplexBridge: true }, + }); + expect(readSandboxDuplexBridgeInput).toHaveBeenCalledTimes(1); + expect(target.enableSandboxDuplexBridge).toBe(true); + }); + + it("stamps no grant when the setting is off", async () => { + const { target } = await buildSandboxTarget({ bridgeInput: { enableDuplexBridge: false } }); + expect(target.enableSandboxDuplexBridge).toBe(false); + }); + + it("stamps no grant when the settings read fails, and the error does not become a grant", async () => { + const { target, readSandboxDuplexBridgeInput } = await buildSandboxTarget({ rejectRead: true }); + expect(readSandboxDuplexBridgeInput).toHaveBeenCalledTimes(1); + // The resolve did not throw, and the read error stamped no grant. + expect(target.enableSandboxDuplexBridge).toBe(false); + }); + + it("stamps no grant when the runtime has no duplex read method", async () => { + const { target } = await buildSandboxTarget({ omitMethod: true }); + expect(target.enableSandboxDuplexBridge).toBe(false); + }); +}); diff --git a/server/src/__tests__/fixtures/plugin-worker-duplex-channel.cjs b/server/src/__tests__/fixtures/plugin-worker-duplex-channel.cjs index bd7b4ac8ad..914df984dd 100644 --- a/server/src/__tests__/fixtures/plugin-worker-duplex-channel.cjs +++ b/server/src/__tests__/fixtures/plugin-worker-duplex-channel.cjs @@ -8,10 +8,11 @@ // - `mode`: "normal" | "malformed-open" | "no-open-reply" | "duplicate-open-reply" | // "no-write-reply" // - `workerSessionId`: the worker session id the open reply returns (default "ws-1") -// - `data`: an array of `{ chunk, sid? }`. The fixture emits each as a data -// notification after the open reply. `sid` defaults to the real worker session -// id; a test sets a wrong `sid` to prove the host drops a mismatched -// notification and counts a protocol error. +// - `data`: an array of `{ chunk, sid?, rid? }`. The fixture emits each as a +// data notification after the open reply. `sid` defaults to the real worker +// session id and `rid` defaults to the real host route id; a test sets a wrong +// `sid` or `rid` to prove the host drops a mismatched-pair notification and +// counts a protocol error. // - `exitCode`: when set, the fixture emits an exit notification after the data. // - `dataAfterExit`: an array of `{ chunk, sid? }`, same shape as `data`. The // fixture emits these as data notifications after the exit notification, so a @@ -34,22 +35,28 @@ function send(message) { } // Serialize the scripted data and exit frames as newline-delimited lines. The -// batch mode writes these together with the open reply in one stdout write. -function scriptedFrameLines(directive, workerSessionId) { +// batch mode writes these together with the open reply in one stdout write. Each +// frame carries the exact pair, so the host binds and routes by the pair. A test +// sets a wrong `sid` or `rid` to force a mismatch. +function scriptedFrameLines(directive, hostRouteId, workerSessionId) { const data = Array.isArray(directive.data) ? directive.data : []; let lines = ""; for (const entry of data) { lines += `${JSON.stringify({ jsonrpc: "2.0", method: "duplexChannel.data", - params: { workerSessionId: entry.sid ?? workerSessionId, chunk: entry.chunk }, + params: { + hostRouteId: entry.rid ?? hostRouteId, + workerSessionId: entry.sid ?? workerSessionId, + chunk: entry.chunk, + }, })}\n`; } if (typeof directive.exitCode === "number") { lines += `${JSON.stringify({ jsonrpc: "2.0", method: "duplexChannel.exit", - params: { workerSessionId, exitCode: directive.exitCode }, + params: { hostRouteId, workerSessionId, exitCode: directive.exitCode }, })}\n`; } const dataAfterExit = Array.isArray(directive.dataAfterExit) ? directive.dataAfterExit : []; @@ -57,7 +64,11 @@ function scriptedFrameLines(directive, workerSessionId) { lines += `${JSON.stringify({ jsonrpc: "2.0", method: "duplexChannel.data", - params: { workerSessionId: entry.sid ?? workerSessionId, chunk: entry.chunk }, + params: { + hostRouteId: entry.rid ?? hostRouteId, + workerSessionId: entry.sid ?? workerSessionId, + chunk: entry.chunk, + }, })}\n`; } return lines; @@ -106,11 +117,15 @@ rl.on("line", (line) => { const mode = directive.mode ?? "normal"; const workerSessionId = directive.workerSessionId ?? "ws-1"; const closeMode = directive.closeMode ?? "ack"; - routes.set(params.hostRouteId, { + const hostRouteId = params.hostRouteId; + routes.set(hostRouteId, { + hostRouteId, workerSessionId, closeMode, echoInput: directive.echoInput === true, noWriteReply: mode === "no-write-reply", + emitAfterCloseChunk: + typeof directive.emitAfterCloseChunk === "string" ? directive.emitAfterCloseChunk : null, }); if (mode === "no-open-reply") { @@ -123,17 +138,18 @@ rl.on("line", (line) => { return; } + // Echo the host route id on the reply, so the host binds the exact pair. const openReplyLine = `${JSON.stringify({ jsonrpc: "2.0", id: message.id, - result: { workerSessionId }, + result: { hostRouteId, workerSessionId }, })}\n`; if (directive.batchWithOpenReply === true) { // Write the open reply and the scripted frames in one stdout write. The // host reads them in one batch, so a data or exit frame arrives before the // route binds. The host must hold and replay the frame after the bind. - process.stdout.write(openReplyLine + scriptedFrameLines(directive, workerSessionId)); + process.stdout.write(openReplyLine + scriptedFrameLines(directive, hostRouteId, workerSessionId)); return; } @@ -145,29 +161,38 @@ rl.on("line", (line) => { } // Emit the scripted data and the exit after the open reply, so the host - // binds the route first. + // binds the route first. Each frame echoes the exact pair; a test overrides + // `sid` or `rid` to force a mismatch. setImmediate(() => { - process.stdout.write(scriptedFrameLines(directive, workerSessionId)); + process.stdout.write(scriptedFrameLines(directive, hostRouteId, workerSessionId)); }); return; } if (method === "duplexChannelWrite") { - const entry = [...routes.values()].find( - (route) => route.workerSessionId === params.workerSessionId, - ); - if (entry && entry.noWriteReply) { + // Act only on the exact live pair. A write whose pair does not match a bound + // route applies no bytes. + const entry = routes.get(params.hostRouteId); + if (!entry || entry.workerSessionId !== params.workerSessionId) { + send({ jsonrpc: "2.0", id: message.id, result: null }); + return; + } + if (entry.noWriteReply) { // Never reply, so the host write call stays pending. The test proves the // host ends the route on the pending-request bound. return; } - if (entry && entry.echoInput) { - // Echo the input back as one data notification for the bound session, so a + if (entry.echoInput) { + // Echo the input back as one data notification for the bound pair, so a // test proves the input reaches the worker and the output routes back. send({ jsonrpc: "2.0", method: "duplexChannel.data", - params: { workerSessionId: entry.workerSessionId, chunk: `echo:${params.data}` }, + params: { + hostRouteId: entry.hostRouteId, + workerSessionId: entry.workerSessionId, + chunk: `echo:${params.data}`, + }, }); } send({ jsonrpc: "2.0", id: message.id, result: null }); @@ -191,7 +216,28 @@ rl.on("line", (line) => { send({ jsonrpc: "2.0", id: message.id, result: { hostRouteId: "mismatched-route" } }); return; } - send({ jsonrpc: "2.0", id: message.id, result: { hostRouteId: params.hostRouteId } }); + // A bound close echoes both identifiers; a pre-bind close with no entry + // echoes the host route id only. + const ack = entry + ? { hostRouteId: params.hostRouteId, workerSessionId: entry.workerSessionId } + : { hostRouteId: params.hostRouteId }; + send({ jsonrpc: "2.0", id: message.id, result: ack }); + if (entry && entry.emitAfterCloseChunk) { + // Emit one late data frame for the just-closed pair, so a test proves the + // host drops a late frame for a tombstoned pair and does not retire the + // worker. + setImmediate(() => { + send({ + jsonrpc: "2.0", + method: "duplexChannel.data", + params: { + hostRouteId: entry.hostRouteId, + workerSessionId: entry.workerSessionId, + chunk: entry.emitAfterCloseChunk, + }, + }); + }); + } return; } diff --git a/server/src/__tests__/plugin-worker-manager-duplex.test.ts b/server/src/__tests__/plugin-worker-manager-duplex.test.ts index 94ecc850cd..85bc5f3dff 100644 --- a/server/src/__tests__/plugin-worker-manager-duplex.test.ts +++ b/server/src/__tests__/plugin-worker-manager-duplex.test.ts @@ -2,7 +2,10 @@ import path from "node:path"; import { fileURLToPath } from "node:url"; import { describe, expect, it, vi } from "vitest"; import type { PaperclipPluginManifestV1 } from "@paperclipai/shared"; -import { createPluginWorkerHandle } from "../services/plugin-worker-manager.js"; +import { + createDuplexRouteSlotController, + createPluginWorkerHandle, +} from "../services/plugin-worker-manager.js"; const FIXTURES_DIR = path.join(path.dirname(fileURLToPath(import.meta.url)), "fixtures"); const DUPLEX_CHANNEL_WORKER_ENTRYPOINT = path.join( @@ -37,10 +40,10 @@ function makeDuplexHandle(extra?: Record) { // The test directive rides in `providerLeaseId`, an opaque field the manager // forwards to the worker unchanged. The duplex channel is generic, so the // command is a plain fixed string with no allowlist. -function duplexOpenInput(directive: unknown) { +function duplexOpenInput(directive: unknown, companyId = "company-1") { return { driverKey: "daytona", - companyId: "company-1", + companyId, environmentId: "env-1", providerLeaseId: JSON.stringify(directive), command: "bridge-callback", @@ -48,28 +51,45 @@ function duplexOpenInput(directive: unknown) { } describe("plugin worker manager duplex channel route", () => { - it("delivers data only for the exact bound worker session id and drops a mismatch", async () => { + it("fails closed and retires the worker on a forged worker session id", async () => { const handle = makeDuplexHandle(); try { await handle.start(); const session = await handle.openDuplexChannel( duplexOpenInput({ workerSessionId: "ws-A", - data: [ - { chunk: "good-1" }, - { chunk: "forged", sid: "ws-EVIL" }, - { chunk: "good-2" }, - ], - exitCode: 0, + // The frame carries the bound host route id but a forged worker session + // id, so its pair matches no live route. + data: [{ chunk: "forged", sid: "ws-EVIL" }], }), ); const chunks: string[] = []; session.onData((chunk) => chunks.push(chunk)); - await expect(session.wait()).resolves.toEqual({ exitCode: 0 }); - // The forged notification carries a wrong worker session id, so the host - // drops it. Only the two bound chunks reach the listener, in order. - expect(chunks).toEqual(["good-1", "good-2"]); - await session.close(); + // The forged pair is an ownership violation. The host reaches no listener and + // retires the worker, so the wait settles with the fixed non-secret null exit. + await expect(session.wait()).resolves.toEqual({ exitCode: null }); + expect(chunks).toEqual([]); + } finally { + await handle.stop().catch(() => undefined); + } + }); + + it("fails closed and retires the worker on a forged host route id", async () => { + const handle = makeDuplexHandle(); + try { + await handle.start(); + const session = await handle.openDuplexChannel( + duplexOpenInput({ + workerSessionId: "ws-A", + // The frame carries the bound worker session id but a forged host route + // id, so its pair matches no live route. + data: [{ chunk: "forged", rid: "duplex-route-forged" }], + }), + ); + const chunks: string[] = []; + session.onData((chunk) => chunks.push(chunk)); + await expect(session.wait()).resolves.toEqual({ exitCode: null }); + expect(chunks).toEqual([]); } finally { await handle.stop().catch(() => undefined); } @@ -213,20 +233,79 @@ describe("plugin worker manager duplex channel route", () => { } }); - it("permits one active duplex channel per worker", async () => { + it("routes two concurrent cross-company duplex routes by the exact pair with no cross-talk", async () => { const handle = makeDuplexHandle(); try { await handle.start(); - const first = await handle.openDuplexChannel(duplexOpenInput({ mode: "normal" })); - // A second open while the first route is not closed rejects with one fixed - // non-secret error before it reaches the worker. + const routeA = await handle.openDuplexChannel( + duplexOpenInput({ workerSessionId: "ws-A", data: [{ chunk: "a-1" }], exitCode: 0 }, "company-A"), + ); + const routeB = await handle.openDuplexChannel( + duplexOpenInput({ workerSessionId: "ws-B", data: [{ chunk: "b-1" }], exitCode: 0 }, "company-B"), + ); + const aChunks: string[] = []; + const bChunks: string[] = []; + routeA.onData((chunk) => aChunks.push(chunk)); + routeB.onData((chunk) => bChunks.push(chunk)); + await expect(routeA.wait()).resolves.toEqual({ exitCode: 0 }); + await expect(routeB.wait()).resolves.toEqual({ exitCode: 0 }); + // The host routes each frame by the exact pair, so each route receives only + // its own data. Neither route sees the other's chunk. + expect(aChunks).toEqual(["a-1"]); + expect(bChunks).toEqual(["b-1"]); + await routeA.close(); + await routeB.close(); + } finally { + await handle.stop().catch(() => undefined); + } + }); + + it("reports an explicit route-busy result when the aggregate ceiling is full", async () => { + // A process-scoped ceiling of one slot. The manager injects one shared + // controller into every worker; the test injects a small one directly. + const handle = makeDuplexHandle({ duplexRouteSlots: createDuplexRouteSlotController(1) }); + try { + await handle.start(); + const first = await handle.openDuplexChannel(duplexOpenInput({ workerSessionId: "ws-A" })); + // The ceiling is full, so the second open rejects with the fixed route-busy + // error before it reaches the worker. An active channel never downgrades. await expect( - handle.openDuplexChannel(duplexOpenInput({ mode: "normal" })), + handle.openDuplexChannel(duplexOpenInput({ workerSessionId: "ws-B" })), ).rejects.toThrow("DUPLEX_CHANNEL_ROUTE_BUSY"); + // Closing the first route releases its slot, so a later open is admitted. await first.close(); - // After the first route closes and the worker acknowledges the close, a new - // open is admitted. - const second = await handle.openDuplexChannel(duplexOpenInput({ mode: "normal" })); + const third = await handle.openDuplexChannel(duplexOpenInput({ workerSessionId: "ws-C" })); + await third.close(); + } finally { + await handle.stop().catch(() => undefined); + } + }); + + it("drops a late frame for a tombstoned pair and keeps the worker for a new open", async () => { + const handle = makeDuplexHandle(); + try { + await handle.start(); + const first = await handle.openDuplexChannel( + duplexOpenInput({ workerSessionId: "ws-A", emitAfterCloseChunk: "after-close" }), + ); + const chunks: string[] = []; + first.onData((chunk) => chunks.push(chunk)); + // Close the route. The host installs the tombstone atomically before the slot + // frees. The worker then emits one late frame for the closed pair. + await first.close(); + // Give the late frame time to arrive. The host drops it, so it reaches no + // listener and does not retire the worker. + await new Promise((resolve) => setTimeout(resolve, 60)); + expect(chunks).toEqual([]); + // The worker is still alive: a new open on the same worker succeeds and + // delivers its own data. + const second = await handle.openDuplexChannel( + duplexOpenInput({ workerSessionId: "ws-B", data: [{ chunk: "b-1" }], exitCode: 0 }), + ); + const secondChunks: string[] = []; + second.onData((chunk) => secondChunks.push(chunk)); + await expect(second.wait()).resolves.toEqual({ exitCode: 0 }); + expect(secondChunks).toEqual(["b-1"]); await second.close(); } finally { await handle.stop().catch(() => undefined); diff --git a/server/src/__tests__/workspace-runtime-start-terminality.test.ts b/server/src/__tests__/workspace-runtime-start-terminality.test.ts index 93dcc2b386..01bbcbe88f 100644 --- a/server/src/__tests__/workspace-runtime-start-terminality.test.ts +++ b/server/src/__tests__/workspace-runtime-start-terminality.test.ts @@ -129,6 +129,23 @@ describe("managed runtime start terminality", () => { ).resolves.toBe(49881); }); + it("skips a candidate port inside the runtime exposure app-port range", async () => { + // The kernel can hand an ephemeral port inside the exposure app-port range + // (42000-42999). The reconciler classifies a persisted row by its port: a + // port in that range marks the row as an exposure reservation, not a managed + // auto port. So the allocator must never return an in-range port; it drops + // the in-range candidate and takes the next out-of-range one. + const candidates = [42500, 49883]; + let index = 0; + const probe = async () => candidates[Math.min(index++, candidates.length - 1)]!; + const portOwnerLookup = async () => null; + + await expect(allocateRuntimeServicePort({ probe, portOwnerLookup })).resolves.toBe(49883); + // The allocator skipped 42500 before it reserved a candidate, so the in-range + // port stays free. A later start can still claim it through the exposure path. + expect(claimRuntimeServiceBindPort(42500, null)).toBe(true); + }); + it("refuses a configured port a sibling start is already claiming", () => { // The reported collision was on a *configured* app/HMR pair, not an auto-allocated one: // both lanes read the pair as free because neither had bound or persisted a row yet. diff --git a/server/src/services/duplex-telemetry-recorder.ts b/server/src/services/duplex-telemetry-recorder.ts new file mode 100644 index 0000000000..8a4192dae6 --- /dev/null +++ b/server/src/services/duplex-telemetry-recorder.ts @@ -0,0 +1,113 @@ +import { + DUPLEX_DIMENSION_KEYS, + DUPLEX_SPAN_REQUEST, + type DuplexTelemetryCounterRecord, + type DuplexTelemetryDimensions, + type DuplexTelemetryEventRecord, + type DuplexTelemetryRecorder, + type DuplexTelemetrySpanRecord, +} from "@paperclipai/adapter-utils/duplex-telemetry"; + +/** + * The host binding for the fixed duplex telemetry surface. This module maps each + * recorder call to a real host sink: the span to the OTel tracer, the counter to + * the guarded counter store in `tool-runtime-metrics.ts`, and the transport event + * to the run-event path. The mapping is the single boundary, so the closed + * dimension keys and the counter folding stay in one place. + * + * The recorder never throws. The facade in `duplex-telemetry.ts` wraps each + * synchronous call in a swallow, but an async sink can still reject after the + * call returns. Each async sink here runs fire-and-forget with its own catch, so + * a sink failure never breaks the request path. + */ + +/** The minimal span the recorder opens. A real OTel span satisfies it; the + * no-op tracer's span satisfies it too. */ +export interface DuplexTelemetrySpan { + setAttribute(key: string, value: string | number | boolean): void; + end(endTime?: number): void; +} + +/** The minimal tracer surface the recorder calls. `getStartupTracer` returns a + * real or a no-op implementation that satisfies it. The optional second argument + * carries an explicit start time, so the request span duration equals the + * measured latency. */ +export interface DuplexTelemetryTracer { + startSpan(name: string, options?: { startTime?: number }): DuplexTelemetrySpan; +} + +export interface HostDuplexTelemetryRecorderInput { + /** The OTel tracer for the two duplex spans. */ + tracer: DuplexTelemetryTracer; + /** + * Increment one guarded host counter by its metric name. The caller binds it + * to `incrementToolRuntimeMetricCounter` with the company id, inside a swallow. + */ + incrementCounter(metric: string): void; + /** + * Emit one transport event to the run-event path. The caller binds it to the + * run-events bridge, inside a swallow. + */ + emitTransportEvent(event: { name: string; dimensions: DuplexTelemetryDimensions }): void; + /** The wall clock. The default is `Date.now`. Tests inject a fixed clock. */ + now?: () => number; +} + +/** + * Fold the discriminating dimension into the counter metric name. The guarded + * counter store keys on `(companyId, metric)` with no label column, so the + * fallback reason or the loss class rides the metric name to keep the analytic + * breakdown. Both dimension sets are closed and low-cardinality, so the metric + * cardinality stays bounded. A record with neither dimension uses the base name. + */ +export function foldDuplexCounterMetric(record: DuplexTelemetryCounterRecord): string { + if (record.dimensions.fallback_reason) { + return `${record.metric}.${record.dimensions.fallback_reason}`; + } + if (record.dimensions.loss_class) { + return `${record.metric}.${record.dimensions.loss_class}`; + } + return record.metric; +} + +/** + * Build the host duplex telemetry recorder. The recorder receives only + * already-mapped dimensions, so a raw provider key never reaches a sink. Every + * span attribute uses only the closed dimension keys. + */ +export function createHostDuplexTelemetryRecorder( + input: HostDuplexTelemetryRecorderInput, +): DuplexTelemetryRecorder { + const now = input.now ?? Date.now; + + const setDimensionAttributes = (span: DuplexTelemetrySpan, dimensions: DuplexTelemetryDimensions): void => { + for (const key of DUPLEX_DIMENSION_KEYS) { + const value = dimensions[key]; + if (typeof value === "string") { + span.setAttribute(key, value); + } + } + }; + + return { + recordSpan(record: DuplexTelemetrySpanRecord): void { + // The request span carries a latency, so start it in the past and end it + // now, so the span duration equals the measured latency. The channel-open + // span carries no latency, so it opens and ends at the same instant. + const end = now(); + const latencyMs = + record.name === DUPLEX_SPAN_REQUEST && typeof record.latencyMs === "number" && Number.isFinite(record.latencyMs) + ? Math.max(0, record.latencyMs) + : 0; + const span = input.tracer.startSpan(record.name, { startTime: end - latencyMs }); + setDimensionAttributes(span, record.dimensions); + span.end(end); + }, + incrementCounter(record: DuplexTelemetryCounterRecord): void { + input.incrementCounter(foldDuplexCounterMetric(record)); + }, + emitEvent(record: DuplexTelemetryEventRecord): void { + input.emitTransportEvent({ name: record.name, dimensions: record.dimensions }); + }, + }; +} diff --git a/server/src/services/environment-execution-target.ts b/server/src/services/environment-execution-target.ts index fccf326a4c..90fc7ac3c9 100644 --- a/server/src/services/environment-execution-target.ts +++ b/server/src/services/environment-execution-target.ts @@ -5,6 +5,7 @@ import { adapterExecutionTargetToRemoteSpec, type AdapterExecutionTarget, } from "@paperclipai/adapter-utils/execution-target"; +import type { DuplexTelemetryRecorder } from "@paperclipai/adapter-utils/duplex-telemetry"; import { clampSpanLabel, getActiveStepContext, @@ -200,6 +201,11 @@ export async function resolveEnvironmentExecutionTarget(input: { // gated server tracer, which is a no-op when tracing is off. Tests inject a // recording tracer. tracer?: ExecTracer; + // The host duplex telemetry recorder. The seam stamps it onto the sandbox + // target next to the runner, so the live object stays on the host and never + // enters the sandbox environment. Absent keeps the safe no-op default in the + // bridge, so the surface stays inert until the host injects a real recorder. + duplexTelemetryRecorder?: DuplexTelemetryRecorder | null; }): Promise { if (input.environment.driver === "local") { return { @@ -291,12 +297,32 @@ export async function resolveEnvironmentExecutionTarget(input: { !capabilityResolutionFailed && (!effectiveCapabilities || effectiveCapabilities.persistentProcessSessions); + // Resolve the per-run duplex bridge kill switch. It rides the host-side + // sandbox target on the same seam as `effectiveCapabilities`, so the value + // stays on the host and never enters the sandbox environment. Fail closed: + // an absent runtime, an absent method, or a read error keeps the file + // bridge. The stamp never turns a read error into a grant. + let enableSandboxDuplexBridge = false; + if (input.environmentRuntime?.readSandboxDuplexBridgeInput) { + try { + const duplexBridgeInput = await input.environmentRuntime.readSandboxDuplexBridgeInput(); + enableSandboxDuplexBridge = duplexBridgeInput.enableDuplexBridge === true; + } catch { + enableSandboxDuplexBridge = false; + } + } + return { kind: "remote", transport: "sandbox", providerKey: parsed.config.provider, shellCommand, remoteCwd, + enableSandboxDuplexBridge, + // Attach the host duplex telemetry recorder next to the runner. The bridge + // binds it to the fixed observability surface. Absent keeps the no-op + // default, so the surface stays inert on a run with no injected recorder. + duplexTelemetryRecorder: input.duplexTelemetryRecorder ?? null, ...(effectiveCapabilities ? { effectiveCapabilities: Object.freeze({ ...effectiveCapabilities }) } : {}), environmentId: input.environment.id ?? null, leaseId: input.leaseId ?? null, diff --git a/server/src/services/environment-run-orchestrator.ts b/server/src/services/environment-run-orchestrator.ts index c907d4c177..3a76dcfec1 100644 --- a/server/src/services/environment-run-orchestrator.ts +++ b/server/src/services/environment-run-orchestrator.ts @@ -42,6 +42,7 @@ import { type AdapterRemoteExecutionSpec, type AdapterWorkspaceRealization, } from "@paperclipai/adapter-utils/execution-target"; +import type { DuplexTelemetryRecorder } from "@paperclipai/adapter-utils/duplex-telemetry"; import { buildWorkspaceRealizationRequest } from "./workspace-realization.js"; import { executionWorkspaceService } from "./execution-workspaces.js"; import { logActivity } from "./activity-log.js"; @@ -347,6 +348,12 @@ export function environmentRunOrchestrator( executionWorkspace: RealizedExecutionWorkspace; effectiveExecutionWorkspaceMode: string | null; persistedExecutionWorkspace: ExecutionWorkspace | null; + /** + * The host duplex telemetry recorder for this run. The orchestrator threads + * it to `resolveEnvironmentExecutionTarget`, which stamps it on the sandbox + * target. Absent keeps the safe no-op default in the bridge. + */ + duplexTelemetryRecorder?: DuplexTelemetryRecorder | null; }): Promise { const { environment, @@ -515,6 +522,7 @@ export function environmentRunOrchestrator( leaseMetadata: (lease.metadata as Record | null) ?? null, lease, environmentRuntime, + duplexTelemetryRecorder: input.duplexTelemetryRecorder ?? null, }); const realizationMode = workspaceRealization.mode === "in_place" ? "in_place" : "copy"; const authoritativeRoot = diff --git a/server/src/services/heartbeat.ts b/server/src/services/heartbeat.ts index dd8ba788f3..f5d9e1841e 100644 --- a/server/src/services/heartbeat.ts +++ b/server/src/services/heartbeat.ts @@ -70,7 +70,9 @@ import { workspaceOperations, } from "@paperclipai/db"; import { conflict, HttpError, notFound } from "../errors.js"; -import { getStartupTraceContext } from "../instrumentation.js"; +import { getStartupTraceContext, getStartupTracer } from "../instrumentation.js"; +import { createHostDuplexTelemetryRecorder } from "./duplex-telemetry-recorder.js"; +import { incrementToolRuntimeMetricCounter } from "./tool-runtime-metrics.js"; import { logger } from "../middleware/logger.js"; import { createGitRemoteAuthProvider, @@ -15345,6 +15347,33 @@ export function heartbeatService(db: Db, options: HeartbeatServiceOptions = {}) lease: acquiredEnvironment.lease, leaseContext: acquiredEnvironment.leaseContext, }; + // The host duplex telemetry recorder for this run. It binds the fixed duplex + // observability surface to real sinks: the spans to the OTel tracer, the + // guarded counters to the tool-runtime metric store, and the transport event + // to the run-event path. Each sink runs guarded and fire-and-forget, so a + // telemetry failure never breaks the run. The orchestrator stamps it on the + // sandbox target; a non-duplex run keeps the safe no-op default in the bridge. + const duplexTelemetryRecorder = createHostDuplexTelemetryRecorder({ + tracer: getStartupTracer(), + incrementCounter: (metric) => { + void incrementToolRuntimeMetricCounter(db, { + companyId: run.companyId, + metric, + }).catch(() => {}); + }, + emitTransportEvent: (event) => { + void (async () => { + const eventRun = run; + const seq = await nextRunEventSeq(eventRun.id); + await appendRunEvent(eventRun, seq, { + eventType: event.name, + stream: "system", + level: event.dimensions.outcome === "error" ? "warn" : "info", + payload: { ...event.dimensions }, + }); + })().catch(() => {}); + }, + }); const realizationResult = await envOrchestrator.realizeForRun({ environment: selectedEnvironment, lease: activeEnvironmentLease.lease, @@ -15355,6 +15384,7 @@ export function heartbeatService(db: Db, options: HeartbeatServiceOptions = {}) executionWorkspace, effectiveExecutionWorkspaceMode, persistedExecutionWorkspace, + duplexTelemetryRecorder, }); activeEnvironmentLease = { ...activeEnvironmentLease, diff --git a/server/src/services/plugin-worker-manager.ts b/server/src/services/plugin-worker-manager.ts index 96f8c67a39..76dea17f0c 100644 --- a/server/src/services/plugin-worker-manager.ts +++ b/server/src/services/plugin-worker-manager.ts @@ -176,22 +176,22 @@ const MAX_DUPLEX_CHANNEL_PRE_BIND_CHARS = 8 * 1024 * 1024; */ const MAX_DUPLEX_CHANNEL_PRE_BIND_FRAMES = 10_000; /** - * The margin the pre-open hold ceiling keeps above the pre-bind buffered-frame - * bound (`maxDuplexChannelPreBindFrames`, see below). A worker can batch frames - * with the open reply, so the host reads them before it knows the bound worker - * session id and before it can apply the per-frame bounds. The host holds these - * frames and replays them after the bind. The hold ceiling bounds that hold, so - * a worker that floods frames before it replies to the open cannot make the - * host hold an unbounded number of frames. + * The margin the pre-bind hold ceiling keeps above the pre-bind buffered-frame + * bound (`maxDuplexChannelPreBindFrames`). A worker can batch data and exit + * frames with the open reply, so the host reads them before the route binds and + * before it can apply the per-frame bounds. The host holds these frames and + * replays them after the bind. The hold ceiling bounds that hold, so a worker + * that floods frames before the open reply cannot make the host hold an + * unbounded number of frames. * - * The hold ceiling is derived from the buffered bound, not a fixed constant: a - * fixed ceiling equal to or below a caller-configured (or even the default) - * buffered bound would let the hold drop the frame that should instead trip the - * buffered bound during replay, so the route would never end. One frame of - * margin is enough — it lets the frame that exceeds the buffered bound reach - * the hold, so the replay's buffered-bound check, not the hold, ends the route. + * The hold ceiling derives from the buffered bound, not a fixed constant. A + * fixed ceiling at or below a caller-configured buffered bound would drop the + * frame that must instead trip the buffered bound during replay, so the route + * would never end. One frame of margin is enough. It lets the frame that + * exceeds the buffered bound reach the hold, so the replay's buffered-bound + * check, not the hold, ends the route. */ -const DUPLEX_CHANNEL_PRE_OPEN_HOLD_MARGIN_FRAMES = 1; +const DUPLEX_CHANNEL_PRE_BIND_HOLD_MARGIN_FRAMES = 1; /** * The default maximum number of in-flight host→worker requests for one duplex * channel route. A worker that never replies cannot make the host hold an @@ -229,6 +229,34 @@ const DUPLEX_CHANNEL_ROUTE_BUSY = "DUPLEX_CHANNEL_ROUTE_BUSY"; /** The fixed non-secret error a failed duplex channel open returns. */ const DUPLEX_CHANNEL_OPEN_FAILED = "DUPLEX_CHANNEL_OPEN_FAILED"; +// The process-scoped monotonic route-generation source. The host mints one +// strictly increasing, non-reusable `hostRouteId` on each duplex channel open, +// across every worker in the process. A retired generation never returns, so a +// late frame for a closed pair never collides with a new open. The host owns the +// value; no worker field sets it. +let duplexHostRouteIdSequence = 0; +/** Mint the next monotonic, non-reusable host route identifier for a duplex channel open. */ +function nextDuplexHostRouteId(): string { + duplexHostRouteIdSequence += 1; + return `duplex-route-${duplexHostRouteIdSequence}`; +} + +// The separator between the two identifiers of a duplex route pair key. It is a +// null byte, which appears in neither a host route id nor a worker session id, so +// two distinct pairs never collapse to one key. +const DUPLEX_PAIR_KEY_SEPARATOR = "\u0000"; +/** Build the exact-pair routing key from the host route id and the worker session id. */ +function duplexPairKey(hostRouteId: string, workerSessionId: string): string { + return `${hostRouteId}${DUPLEX_PAIR_KEY_SEPARATOR}${workerSessionId}`; +} + +// The maximum number of tombstoned duplex pairs one worker retains. The worker +// keeps every tombstone until it retires, so a closed pair never returns. Before +// the set would exceed this bound, the host retires the worker, which drops every +// route and every tombstone at once. A tombstone overflow fails closed; the host +// never evicts a tombstone and lets a pair return. +const MAX_DUPLEX_ROUTE_TOMBSTONES = 4096; + /** Minimum time between two dropped-`execute.log` debug records. The router * rate-limits the record so a flood of dropped chunks writes at most one line * per window with a running count. */ @@ -324,6 +352,18 @@ export function resolveRpcCallTimeoutMs( /** * Options for starting a worker process. */ +/** + * The process-scoped aggregate route-slot controller. The manager injects one + * shared instance into every worker handle, so the ceiling counts every + * concurrent duplex route across the process, not one agent's setting. + */ +export interface DuplexRouteSlotController { + /** Reserve one route slot. Return true when a slot was free and is now held. */ + tryAcquire(): boolean; + /** Release one held route slot. */ + release(): void; +} + export interface WorkerStartOptions { /** Absolute path to the plugin worker entrypoint (CJS bundle). */ entrypointPath: string; @@ -350,6 +390,13 @@ export interface WorkerStartOptions { execArgv?: string[]; /** Environment variables passed to the child process. */ env?: Record; + /** + * The process-scoped aggregate route-slot controller for the duplex channel + * ceiling. The manager injects one shared instance into every worker handle. + * When it is absent, the worker admits every duplex open unbounded (a unit test + * constructs a handle this way). + */ + duplexRouteSlots?: DuplexRouteSlotController | null; /** * Companies this worker may act on from proactive (no-invocation) worker→host * calls — the plugin's configured companies. Seeded onto the handle at @@ -828,11 +875,11 @@ export function createPluginWorkerHandle( const maxDuplexChannelPreBindFrames = options.duplexChannelLimits?.maxPreBindBufferedFrames ?? MAX_DUPLEX_CHANNEL_PRE_BIND_FRAMES; - // Always strictly above maxDuplexChannelPreBindFrames, including when a - // caller configures that bound at or above the module default. See - // DUPLEX_CHANNEL_PRE_OPEN_HOLD_MARGIN_FRAMES for why the margin must hold. - const maxDuplexChannelPreOpenHoldFrames = - maxDuplexChannelPreBindFrames + DUPLEX_CHANNEL_PRE_OPEN_HOLD_MARGIN_FRAMES; + // The pre-bind hold ceiling stays one frame above the buffered bound, so the + // replay, not the hold, ends the route on the buffered bound. See + // DUPLEX_CHANNEL_PRE_BIND_HOLD_MARGIN_FRAMES for why the margin must hold. + const maxDuplexChannelPreBindHoldFrames = + maxDuplexChannelPreBindFrames + DUPLEX_CHANNEL_PRE_BIND_HOLD_MARGIN_FRAMES; const maxDuplexChannelPendingRequests = options.duplexChannelLimits?.maxPendingRequests ?? MAX_DUPLEX_CHANNEL_PENDING_REQUESTS; @@ -1505,50 +1552,124 @@ export function createPluginWorkerHandle( listener: ((chunk: string) => void) | null; buffered: string[]; bufferedChars: number; - // Raw data notifications that arrive before the route binds. The host reads - // the worker stdout line by line. The open reply and a data notification can - // arrive in one read batch, so the host dispatches the notification before - // the deferred open-reply continuation flips the state to `open`. The host - // holds these frames here and replays them in order right after it binds the - // route, so a batched frame is never lost. Bounded by the pre-open hold - // ceiling — see `bufferPreOpenDuplexChannelNotification`. - preOpen: JsonRpcNotification[]; - // The most recent exit notification that arrived before the route binds, or - // null. A single slot, not an array: only the last exit a worker sends before - // the bind is ever meaningful, so the host overwrites this on every pre-open - // exit instead of holding each one. That keeps a worker that batches many - // exit notifications before it replies to the open from growing this past one - // entry — the pre-open hold ceiling bounds only `preOpen`, so an exit that - // shared that array with data frames could otherwise consume a data frame's - // hold slot and let a data frame that should trip the buffered-frame bound - // get dropped by the hold instead. Replayed after `preOpen` drains, so data - // still delivers before the exit resolves the wait, matching a real worker's - // order. - preOpenExit: JsonRpcNotification | null; pendingRequests: number; protocolErrors: number; totalDataBytes: number; lifetimeTimer: ReturnType | null; terminalized: boolean; settleWait: (value: { exitCode: number | null }) => void; + // The data frames that arrived before the bind, held in order. The bind + // replays them through the exact-pair routing, so an early frame is never + // lost and a frame whose pair does not match the bound pair still fails + // closed. The hold ceiling is `maxDuplexChannelPreBindHoldFrames`, one frame + // above the buffered bound, so the replay's buffered-bound check ends the + // route, not the hold. + preBind: JsonRpcNotification[]; + // The single exit frame that arrived before the bind. An exit never consumes + // a data hold slot, so a worker that batches an exit among enough data frames + // to fill the hold cannot crowd out a data frame. The bind replays the held + // data frames first, then this exit last. + preBindExit: JsonRpcNotification | null; + } + // The live duplex routes on this worker, keyed by the exact + // `{ hostRouteId, workerSessionId }` pair. The host binds one pair once, at + // open, and routes each data, exit, and close frame only to the route that owns + // its exact pair. The host never routes by the worker session id alone. + const liveDuplexRoutes = new Map(); + // The reserved-or-opening routes, keyed by the host route id. A route lives here + // from the open call until the worker session id binds. The host then moves it + // to `liveDuplexRoutes` under the exact pair key. + const openingDuplexRoutes = new Map(); + // The tombstoned pairs on this worker, keyed by the exact pair key. The host + // installs a tombstone atomically when it removes a live binding, and retains it + // until the worker retires. A late frame for a tombstoned pair reaches no + // listener. A closed pair never returns. + const duplexPairTombstones = new Set(); + + // The process-scoped aggregate route-slot controller. The manager injects it, so + // the ceiling counts every duplex route across every worker in the process, not + // one agent's setting. When it is absent, the worker admits every open, so the + // handle runs unbounded in isolation (a unit test constructs it this way). + const duplexRouteSlots = options.duplexRouteSlots ?? null; + // The routes that currently hold one aggregate slot. The host releases a slot one + // time per route, so a double terminalize never releases two slots. + const duplexRouteSlotHolders = new Set(); + + // Try to reserve one aggregate route slot for a route. Return true when the + // route holds a slot after the call. When no controller is present, the route + // always holds a slot. + function acquireDuplexRouteSlot(route: DuplexChannelRoute): boolean { + if (!duplexRouteSlots) return true; + if (!duplexRouteSlots.tryAcquire()) return false; + duplexRouteSlotHolders.add(route); + return true; + } + + // Release the aggregate route slot a route holds, one time. A route that never + // held a slot releases nothing. + function releaseDuplexRouteSlot(route: DuplexChannelRoute): void { + if (!duplexRouteSlotHolders.delete(route)) return; + duplexRouteSlots?.release(); + } + + // Retire the worker at once. The host kills the process, so every live route, + // every reserved route, and every tombstone drops together. The host uses this + // as the fail-closed response to an ownership violation and to a tombstone-set + // overflow. The host never logs the raw frame content on a violation. + function retireDuplexWorkerOnViolation(reason: string): void { + log.error({ pluginId }, `duplex channel ownership violation (${reason}); retiring worker`); + void killProcess(); + } + + // Install a tombstone for a bound pair atomically with the removal of its live + // binding. This runs in one synchronous step, before any slot release or reuse, + // so a late frame for the pair reaches no listener. Before the set would exceed + // its hard bound, retire the worker, which drops every route and every tombstone + // at once. A tombstone overflow fails closed; the host never evicts a tombstone + // and lets a pair return. + function tombstoneBoundDuplexPair(route: DuplexChannelRoute): void { + if (route.workerSessionId === null) return; + const pairKey = duplexPairKey(route.hostRouteId, route.workerSessionId); + liveDuplexRoutes.delete(pairKey); + if (!duplexPairTombstones.has(pairKey)) { + if (duplexPairTombstones.size >= MAX_DUPLEX_ROUTE_TOMBSTONES) { + // The set is full. Do not evict a tombstone. Retire the worker, which + // drops every route and every tombstone together, so no closed pair ever + // returns. + retireDuplexWorkerOnViolation("tombstone set overflow"); + return; + } + duplexPairTombstones.add(pairKey); + } } - // At most one active duplex channel per worker. A non-null route blocks a - // second open until the manager confirms the first route's close. - let duplexChannelRoute: DuplexChannelRoute | null = null; // Close the worker channel by the host route identifier and verify the bound // acknowledgement. Return true only when the worker returns an acknowledgement - // that carries the exact host route identifier. An absent, malformed, - // mismatched, or timed-out acknowledgement returns false, so the caller fails - // closed. - async function closeDuplexChannelTerminal(hostRouteId: string): Promise { + // for the exact pair. Before a session binds, the acknowledgement carries the + // host route identifier only, so a lost open reply still permits a route-only + // close. After a session binds, the acknowledgement must echo both the host + // route identifier and the bound worker session identifier. An absent, + // malformed, mismatched, or timed-out acknowledgement returns false, so the + // caller fails closed. + async function closeDuplexChannelTerminal( + hostRouteId: string, + boundWorkerSessionId: string | null, + ): Promise { try { const ack = await callInternal( "duplexChannelClose", { hostRouteId }, duplexChannelCloseTimeoutMs, ); - return isRecord(ack) && readNonEmptyString(ack.hostRouteId) === hostRouteId; + if (!isRecord(ack) || readNonEmptyString(ack.hostRouteId) !== hostRouteId) { + return false; + } + if (boundWorkerSessionId !== null) { + // A bound close: the acknowledgement must echo the exact worker session + // identifier. A missing or a mismatched identifier is invalid. + return readNonEmptyString(ack.workerSessionId) === boundWorkerSessionId; + } + return true; } catch { return false; } @@ -1571,20 +1692,27 @@ export function createPluginWorkerHandle( route.terminalized = true; route.state = "closed"; route.listener = null; - // Keep the buffered chunks that the host accepted before the route ended, so - // a listener that attaches after the end still drains them. A frame can end - // the route during the pre-open replay, before a listener attaches, and the - // buffered chunks the host accepted before that frame are valid data the - // listener must still receive. The buffered bytes stay bounded by the - // pre-bind buffered bound, and `onData` clears them once it drains them. - route.preOpen = []; - route.preOpenExit = null; + // Keep the buffered chunks the host accepted before the route ended, so a + // listener that attaches after the end still drains them. A frame can end the + // route during the pre-bind replay, before a listener attaches, and the + // chunks the host accepted before that frame are valid data the listener must + // still receive. The buffered bytes stay bounded by the pre-bind buffered + // bound, and `onData` clears them once it drains them. Drop the held pre-bind + // frames, which never bound to a listener. + route.preBind = []; + route.preBindExit = null; clearDuplexChannelLifetimeTimer(route); + // Remove the live binding and install the tombstone in one synchronous step, + // before the worker close and before any reuse. A reserved route that never + // bound leaves the opening map only; it has no pair to tombstone, and the + // monotonic host route id never returns. + openingDuplexRoutes.delete(route.hostRouteId); + tombstoneBoundDuplexPair(route); + releaseDuplexRouteSlot(route); // A terminalized route reports a null exit code, which the caller treats as a // failure. settleRouteWait(route, { exitCode: null }); - const confirmed = await closeDuplexChannelTerminal(route.hostRouteId); - if (duplexChannelRoute === route) duplexChannelRoute = null; + const confirmed = await closeDuplexChannelTerminal(route.hostRouteId, route.workerSessionId); if (!confirmed) { // The worker did not acknowledge the close, so the host cannot prove the // channel is gone. Fail closed: retire the worker before any reuse. @@ -1606,65 +1734,6 @@ export function createPluginWorkerHandle( } } - // Hold one data or exit notification that arrives before the route binds. The - // host replays the held notifications after it binds the route: `preOpen` - // drains first, in order, then the held exit (if any) resolves the wait last — - // see `drainPreOpenDuplexChannelNotifications`. - // - // A data notification goes on `preOpen`, bounded by a pre-open ceiling derived - // from the pre-bind buffered-frame bound, so a worker that floods data frames - // before it replies to the open cannot make the host hold an unbounded number - // of them. Count one protocol error for each data frame past the ceiling. This - // ceiling is separate from the pre-bind buffered-frame bound, but it tracks it: - // the replay after the bind applies the buffered bound to each held frame, so - // the buffered bound ends the route when a caller lowers (or raises) it. The - // hold ceiling stays above the buffered bound by construction - // (maxDuplexChannelPreOpenHoldFrames = maxDuplexChannelPreBindFrames + margin), - // or it would drop a frame before the buffered bound can end the route. - // - // An exit notification never touches `preOpen`. It overwrites the single - // `preOpenExit` slot instead, so a worker that batches an exit among enough - // data frames to fill the hold cannot consume a data frame's hold slot: the - // ceiling above bounds `preOpen` alone, so it stays exactly the margin above - // the buffered bound regardless of how many exit notifications arrive pre-open. - function bufferPreOpenDuplexChannelNotification( - route: DuplexChannelRoute, - notification: JsonRpcNotification, - ): void { - if (notification.method === DUPLEX_CHANNEL_EXIT_NOTIFICATION) { - route.preOpenExit = notification; - return; - } - if (route.preOpen.length >= maxDuplexChannelPreOpenHoldFrames) { - recordDuplexChannelProtocolError(route); - return; - } - route.preOpen.push(notification); - } - - // Replay the held pre-open notifications right after the route binds. The - // route is `open` now, so each notification passes through the normal - // per-frame bounds and the session-identifier match. Drain the held data - // frames first, in order, then replay the held exit (if any) last, so data - // still delivers before the exit resolves the wait, matching a real worker's - // order. A frame that ends the route terminalizes it, and every later - // notification in the replay is a no-op, because the routing functions drop a - // notification when the route is not `open`. - function drainPreOpenDuplexChannelNotifications(route: DuplexChannelRoute): void { - if (route.preOpen.length > 0) { - const pending = route.preOpen; - route.preOpen = []; - for (const notification of pending) { - routeDuplexChannelData(notification); - } - } - if (route.preOpenExit) { - const exitNotification = route.preOpenExit; - route.preOpenExit = null; - routeDuplexChannelExit(exitNotification); - } - } - // Deliver one duplex channel chunk to the bound listener in isolation. A // listener that throws must not escape the worker stdout notification handler // or the buffered replay, so a throw here breaks neither the notification @@ -1691,25 +1760,54 @@ export function createPluginWorkerHandle( // larger than the per-chunk limit or when the cumulative bytes pass the total // cap. Buffer a valid frame under the pre-bind bounds when no listener has // attached yet. Never log the raw bytes. + // Resolve one inbound duplex frame to its owning route by the exact pair. Return + // the live route on an exact live-pair match. Return `"tombstoned"` for a late + // frame whose exact pair the host already closed; that frame reaches no listener + // and changes no state. Return `"violation"` for any unknown, foreign, + // duplicate, ambiguous, or malformed pair; the caller fails closed and retires + // the worker. The lookup never routes by the worker session id alone. + function resolveDuplexRouteByPair( + hostRouteId: string | null, + workerSessionId: string | null, + ): DuplexChannelRoute | "tombstoned" | "opening" | "violation" { + if (!hostRouteId || !workerSessionId) return "violation"; + const pairKey = duplexPairKey(hostRouteId, workerSessionId); + const live = liveDuplexRoutes.get(pairKey); + if (live && live.state === "open") return live; + if (duplexPairTombstones.has(pairKey)) return "tombstoned"; + // A frame whose host route id names a route that is still opening arrived + // before the bind. It is the host's own reserved route, not a foreign frame. + // The caller defers it, so the bind replays it through the exact-pair routing. + if (openingDuplexRoutes.has(hostRouteId)) return "opening"; + return "violation"; + } + function routeDuplexChannelData(notification: JsonRpcNotification): void { - const route = duplexChannelRoute; - if (!route || route.terminalized) return; - if (route.state === "reserved" || route.state === "opening") { - // The route did not bind yet. Hold the frame and replay it after the bind. - bufferPreOpenDuplexChannelNotification(route, notification); + const params = isRecord(notification.params) ? notification.params : {}; + const hostRouteId = readNonEmptyString(params.hostRouteId); + const workerSessionId = readNonEmptyString(params.workerSessionId); + const resolved = resolveDuplexRouteByPair(hostRouteId, workerSessionId); + if (resolved === "tombstoned") { + // A late frame for a closed pair. It reaches no listener and changes no + // state. Never log the raw frame content. return; } - if (route.state !== "open") return; - const params = isRecord(notification.params) ? notification.params : {}; - const workerSessionId = readNonEmptyString(params.workerSessionId); + if (resolved === "violation") { + // An unknown, foreign, duplicate, ambiguous, or malformed pair. Fail closed: + // retire the worker. Never log the raw frame content. + retireDuplexWorkerOnViolation("data frame pair"); + return; + } + if (resolved === "opening") { + // The frame arrived before the bind. Hold it; the bind replays it through the + // exact-pair routing. + bufferPreBindDuplexFrame(hostRouteId, notification); + return; + } + const route = resolved; const chunk = params.chunk; - if ( - !workerSessionId || - workerSessionId !== route.workerSessionId || - typeof chunk !== "string" || - chunk.length === 0 - ) { - // A late, unknown, malformed, or mismatched frame. Drop it and count one + if (typeof chunk !== "string" || chunk.length === 0) { + // The exact pair matches, but the chunk is malformed. Count one per-route // protocol error. recordDuplexChannelProtocolError(route); return; @@ -1750,37 +1848,101 @@ export function createPluginWorkerHandle( // the route is `open` and the notification carries the exact bound worker // session identifier. function routeDuplexChannelExit(notification: JsonRpcNotification): void { - const route = duplexChannelRoute; - if (!route || route.terminalized) return; - if (route.state === "reserved" || route.state === "opening") { - // The route did not bind yet. Hold the frame and replay it after the bind. - bufferPreOpenDuplexChannelNotification(route, notification); + const params = isRecord(notification.params) ? notification.params : {}; + const hostRouteId = readNonEmptyString(params.hostRouteId); + const workerSessionId = readNonEmptyString(params.workerSessionId); + const resolved = resolveDuplexRouteByPair(hostRouteId, workerSessionId); + if (resolved === "tombstoned") { + // A late exit for a closed pair. It reaches no wait and changes no state. + return; + } + if (resolved === "violation") { + // An unknown, foreign, duplicate, ambiguous, or malformed pair. Fail closed: + // retire the worker. + retireDuplexWorkerOnViolation("exit frame pair"); + return; + } + if (resolved === "opening") { + // The exit arrived before the bind. Hold it; the bind replays it through the + // exact-pair routing. + bufferPreBindDuplexFrame(hostRouteId, notification); return; } - if (route.state !== "open") return; - const params = isRecord(notification.params) ? notification.params : {}; - const workerSessionId = readNonEmptyString(params.workerSessionId); - if (!workerSessionId || workerSessionId !== route.workerSessionId) return; const exitCode = typeof params.exitCode === "number" ? params.exitCode : null; - settleRouteWait(route, { exitCode }); + settleRouteWait(resolved, { exitCode }); } - // Close the one route on a worker exit. The worker is gone, so the manager - // resolves the wait with the fixed non-secret exit and clears the route one - // time. The pending channel calls reject through `rejectAllPending`. - function closeDuplexChannelRouteOnWorkerExit(): void { - const route = duplexChannelRoute; + // Hold one frame that arrived before its route bound, in order. The bind replays + // the held frames through the exact-pair routing. An exit frame never consumes a + // data hold slot; the host holds it in the single `preBindExit` slot instead, so + // a worker that batches an exit among enough data frames to fill the hold cannot + // crowd out a data frame. The data hold ceiling stays one frame above the + // buffered bound, so the replay's buffered-bound check, not the hold, ends the + // route. The host counts one protocol error for each data frame past the ceiling, + // so an early flood bounds the hold instead of growing without limit. + function bufferPreBindDuplexFrame( + hostRouteId: string | null, + notification: JsonRpcNotification, + ): void { + if (!hostRouteId) return; + const route = openingDuplexRoutes.get(hostRouteId); if (!route) return; - duplexChannelRoute = null; - route.terminalized = true; - route.state = "closed"; - route.listener = null; - route.buffered = []; - route.bufferedChars = 0; - route.preOpen = []; - route.preOpenExit = null; - clearDuplexChannelLifetimeTimer(route); - settleRouteWait(route, { exitCode: null }); + if (notification.method === DUPLEX_CHANNEL_EXIT_NOTIFICATION) { + route.preBindExit = notification; + return; + } + if (route.preBind.length >= maxDuplexChannelPreBindHoldFrames) { + recordDuplexChannelProtocolError(route); + return; + } + route.preBind.push(notification); + } + + // Replay the frames a route held before it bound. The route is live now, so the + // exact-pair routing delivers each held frame or fails closed on a mismatch. + // Replay the held data frames first, in order, then the held exit last, so the + // data delivers before the exit resolves the wait, which matches a real worker's + // order. A frame that ends the route terminalizes it, and every later frame in + // the replay is a no-op, because the routing functions drop a frame for a route + // that is not `open`. + function replayPreBindDuplexFrames(route: DuplexChannelRoute): void { + const held = route.preBind; + route.preBind = []; + for (const notification of held) { + if (route.terminalized) break; + routeDuplexChannelData(notification); + } + const heldExit = route.preBindExit; + route.preBindExit = null; + if (heldExit && !route.terminalized) { + routeDuplexChannelExit(heldExit); + } + } + + // Close every route on a worker exit. The worker is gone, so the manager + // resolves each wait with the fixed non-secret exit, releases each aggregate + // slot, and clears the live, opening, and tombstone state. A worker exit is a + // worker retirement, so the host drops every tombstone; the monotonic host route + // id never returns, so no closed pair can revive on a restart. The pending + // channel calls reject through `rejectAllPending`. + function closeDuplexChannelRouteOnWorkerExit(): void { + const routes = [...openingDuplexRoutes.values(), ...liveDuplexRoutes.values()]; + openingDuplexRoutes.clear(); + liveDuplexRoutes.clear(); + duplexPairTombstones.clear(); + for (const route of routes) { + if (route.terminalized) continue; + route.terminalized = true; + route.state = "closed"; + route.listener = null; + route.buffered = []; + route.bufferedChars = 0; + route.preBind = []; + route.preBindExit = null; + clearDuplexChannelLifetimeTimer(route); + releaseDuplexRouteSlot(route); + settleRouteWait(route, { exitCode: null }); + } } // Open one live generic duplex channel route. Reserve the route before the open @@ -1790,12 +1952,7 @@ export function createPluginWorkerHandle( async function openDuplexChannel( input: DuplexChannelOpenInput, ): Promise { - if (duplexChannelRoute) { - // A route for this worker is not yet closed and confirmed. Reject the - // second open with one fixed non-secret error before it reaches the worker. - throw new Error(DUPLEX_CHANNEL_ROUTE_BUSY); - } - const hostRouteId = randomUUID(); + const hostRouteId = nextDuplexHostRouteId(); let settleWait: (value: { exitCode: number | null }) => void = () => {}; const waitPromise = new Promise<{ exitCode: number | null }>((resolve) => { settleWait = resolve; @@ -1807,16 +1964,22 @@ export function createPluginWorkerHandle( listener: null, buffered: [], bufferedChars: 0, - preOpen: [], - preOpenExit: null, pendingRequests: 0, protocolErrors: 0, totalDataBytes: 0, lifetimeTimer: null, terminalized: false, settleWait, + preBind: [], + preBindExit: null, }; - duplexChannelRoute = route; + // Reserve one aggregate route slot before any work. When the process-scoped + // ceiling is full, reject with the fixed route-busy error and open nothing, so + // an active channel never downgrades and the ceiling never overcommits. + if (!acquireDuplexRouteSlot(route)) { + throw new Error(DUPLEX_CHANNEL_ROUTE_BUSY); + } + openingDuplexRoutes.set(hostRouteId, route); route.state = "opening"; let openResult: HostToWorkerMethods["duplexChannelOpen"][1]; @@ -1841,32 +2004,45 @@ export function createPluginWorkerHandle( } const workerSessionId = readBindableWorkerSessionId(route, openResult); - if (!workerSessionId) { - // A malformed reply, or a route that already left `opening`. A late or a - // duplicate reply never binds, revives, or reopens a route. + // Verify the worker echoed the exact host route id the open request carried. + // A reply with a missing or a mismatched host route id never binds, so the + // host binds only a reply that proves the worker holds the exact pair. + const echoedHostRouteId = isRecord(openResult) + ? readNonEmptyString(openResult.hostRouteId) + : null; + if (!workerSessionId || echoedHostRouteId !== hostRouteId) { + // A malformed reply, a mismatched host route id, or a route that already + // left `opening`. A late or a duplicate reply never binds, revives, or + // reopens a route. await terminalizeDuplexChannelRoute(route); throw new Error(DUPLEX_CHANNEL_OPEN_FAILED); } - // Bind the worker session identifier one time and move the route to `open`. + const pairKey = duplexPairKey(hostRouteId, workerSessionId); + if (liveDuplexRoutes.has(pairKey) || duplexPairTombstones.has(pairKey)) { + // The worker returned a pair that is already live or already tombstoned. Fail + // closed: terminalize this route and retire the worker before any reuse. + await terminalizeDuplexChannelRoute(route); + retireDuplexWorkerOnViolation("duplicate bound pair"); + throw new Error(DUPLEX_CHANNEL_OPEN_FAILED); + } + // Bind the worker session identifier one time and move the route from the + // opening map to the live map under the exact pair key. route.workerSessionId = workerSessionId; route.state = "open"; - - // Replay any data or exit frame that arrived in the open-reply read batch, - // before the route bound. The route is `open` now, so each replayed frame - // passes through the normal per-frame bounds and the session match. - drainPreOpenDuplexChannelNotifications(route); + openingDuplexRoutes.delete(hostRouteId); + liveDuplexRoutes.set(pairKey, route); // Start the route lifetime timer now the route is open. The route ends when // the timer expires. Every terminal path and the worker-exit path clears the // timer. Unreference the timer so it never blocks the host process shutdown. - // A replayed frame can end the route during the drain above, so start the - // timer only while the route is still open. - if (route.state === "open") { - route.lifetimeTimer = setTimeout(() => { - void terminalizeDuplexChannelRoute(route); - }, maxDuplexChannelDurationMs); - route.lifetimeTimer.unref?.(); - } + route.lifetimeTimer = setTimeout(() => { + void terminalizeDuplexChannelRoute(route); + }, maxDuplexChannelDurationMs); + route.lifetimeTimer.unref?.(); + + // Replay any frame that arrived before the bind. The route is live now, so the + // exact-pair routing delivers each held frame or fails closed on a mismatch. + replayPreBindDuplexFrames(route); // Send one host→worker request under the pending-request bound. End the route // when too many requests are in-flight, so a worker that never replies cannot @@ -1909,7 +2085,11 @@ export function createPluginWorkerHandle( void terminalizeDuplexChannelRoute(route); return; } - sendBoundedRequest("duplexChannelWrite", { workerSessionId: sid, data }); + sendBoundedRequest("duplexChannelWrite", { + hostRouteId: route.hostRouteId, + workerSessionId: sid, + data, + }); }, wait(): Promise<{ exitCode: number | null }> { return waitPromise; @@ -1917,7 +2097,7 @@ export function createPluginWorkerHandle( kill(): void { const sid = route.workerSessionId; if (!sid) return; - sendBoundedRequest("duplexChannelStop", { workerSessionId: sid }); + sendBoundedRequest("duplexChannelStop", { hostRouteId: route.hostRouteId, workerSessionId: sid }); }, async close(): Promise { await terminalizeDuplexChannelRoute(route); @@ -2829,6 +3009,46 @@ export interface PluginWorkerManagerOptions { signal?: string | null; willRestart?: boolean; }) => void; + /** + * The process-scoped aggregate ceiling for concurrent duplex channel routes, + * across every worker in the process. The manager builds one shared slot + * controller from it and injects it into every worker handle, so one tenant can + * never exhaust the manager-wide resource. The manager validates it and falls + * back to {@link DEFAULT_MAX_CONCURRENT_DUPLEX_ROUTES} for an absent or an + * invalid value. It is not the per-agent `heartbeat.maxConcurrentRuns`, which + * stays upstream admission only. + */ + maxConcurrentDuplexRoutes?: number | null; +} + +/** + * The default process-scoped aggregate ceiling for concurrent duplex channel + * routes. It caps the manager-wide resource, not one agent's run budget. The host + * reports an explicit route-busy outcome when the ceiling is full. + */ +export const DEFAULT_MAX_CONCURRENT_DUPLEX_ROUTES = 128; + +/** + * Build one process-scoped aggregate route-slot controller. The controller holds a + * strictly positive integer ceiling and a live count. `tryAcquire` reserves one + * slot only when a slot is free, so the count never passes the ceiling. + */ +export function createDuplexRouteSlotController(maxRoutes?: number | null): DuplexRouteSlotController { + const ceiling = + typeof maxRoutes === "number" && Number.isInteger(maxRoutes) && maxRoutes > 0 + ? maxRoutes + : DEFAULT_MAX_CONCURRENT_DUPLEX_ROUTES; + let active = 0; + return { + tryAcquire(): boolean { + if (active >= ceiling) return false; + active += 1; + return true; + }, + release(): void { + if (active > 0) active -= 1; + }, + }; } /** @@ -2864,6 +3084,12 @@ export function createPluginWorkerManager( const workers = new Map(); /** Per-plugin startup locks to prevent concurrent spawn races. */ const startupLocks = new Map>(); + // The one shared, process-scoped aggregate route-slot controller. The manager + // injects it into every worker handle, so the duplex route ceiling counts every + // concurrent route across the process, not one agent's setting. + const duplexRouteSlots = createDuplexRouteSlotController( + managerOptions?.maxConcurrentDuplexRoutes, + ); return { async startWorker( @@ -2884,7 +3110,12 @@ export function createPluginWorkerManager( ); } - const handle = createPluginWorkerHandle(pluginId, options); + const handle = createPluginWorkerHandle(pluginId, { + // Inject the shared process-scoped route-slot controller, unless the caller + // already supplied one (a test may inject its own). + duplexRouteSlots, + ...options, + }); workers.set(pluginId, handle); // Subscribe to crash/ready events for live event forwarding diff --git a/server/src/services/workspace-runtime.ts b/server/src/services/workspace-runtime.ts index cd572f164a..0d1a504765 100644 --- a/server/src/services/workspace-runtime.ts +++ b/server/src/services/workspace-runtime.ts @@ -4829,6 +4829,13 @@ export async function allocateRuntimeServicePort(overrides?: { for (let attempt = 0; attempt < PORT_ALLOCATION_ATTEMPTS; attempt += 1) { const candidate = await probe(); lastCandidate = candidate; + // Never hand back a port inside the runtime exposure app-port range. The + // reconciler classifies a persisted row by its port. A port in that range + // marks the row as an exposure reservation, not a managed auto port. An auto + // port from that range makes the reconciler read a stopped managed row as an + // exposure reservation and report drift instead of the adoption. The kernel + // can hand out an ephemeral port in that band, so skip the candidate here. + if (isRuntimeExposureAppPort(candidate)) continue; if (!reservePortIfFree(candidate)) continue; const ownerPid = await portOwnerLookup(candidate); if (!ownerPid) return candidate; diff --git a/tests/e2e/smoke-lab.spec.ts b/tests/e2e/smoke-lab.spec.ts index 1b2fed21b3..3b6c0904f0 100644 --- a/tests/e2e/smoke-lab.spec.ts +++ b/tests/e2e/smoke-lab.spec.ts @@ -301,7 +301,13 @@ async function gatewayFetch(request: APIRequestContext, path: string, token: str } test.describe.serial("Smoke Lab scenario catalog mirror", () => { - test.setTimeout(240_000); + // The lifecycle test walks all CI-safe scenarios, and each scenario records + // eight steps with a real navigation and a full-page screenshot. On a loaded + // CI runner the whole walk needs a little more than four minutes, so the + // earlier 240s budget could expire on the final step. Six minutes gives + // headroom for runner variance without slowing a healthy run, because this is + // a ceiling, not the normal run time. + test.setTimeout(360_000); test("records the P1-P7 CI-safe Smoke Lab lifecycle into the results API @smoke-lab", async ({ page, request }) => { const seed = await newCompany(request, "catalog"); diff --git a/ui/src/pages/CompanyEnvironments.test.tsx b/ui/src/pages/CompanyEnvironments.test.tsx index 1875aaa448..ee708c9783 100644 --- a/ui/src/pages/CompanyEnvironments.test.tsx +++ b/ui/src/pages/CompanyEnvironments.test.tsx @@ -290,6 +290,21 @@ function getEnvironmentFormPage(): HTMLElement | null { return document.body.querySelector("[data-testid='environment-form-page']"); } +// Opens the edit page for the environment at `index`. The environment list +// depends on an async query. Under a loaded test worker that query can resolve +// after the first flush, so a direct click can miss the Edit control and never +// open the form page. This helper first waits for the control, then clicks it, +// then waits for the form page. +async function openEnvironmentEditPage(container: HTMLElement, index = 0) { + await waitForAssertion(() => { + expect(editButtons(container)[index]).toBeTruthy(); + }); + await act(async () => click(editButtons(container)[index])); + await waitForAssertion(() => { + expect(getEnvironmentFormPage()).not.toBeNull(); + }); +} + function renderCompanyEnvironments(queryClient: QueryClient, initialPath = ENVIRONMENTS_PATH) { return ( @@ -844,7 +859,7 @@ describe("CompanyEnvironments — test provider button", () => { await flushReact(); // Daytona supports setup + capture -> "Configure image" on its edit page. - await act(async () => click(editButtons(container)[0])); + await openEnvironmentEditPage(container); await waitForAssertion(() => { expect(getEnvironmentFormPage()?.textContent).toContain("Configure image"); }); @@ -853,7 +868,7 @@ describe("CompanyEnvironments — test provider button", () => { await waitForAssertion(() => expect(getEnvironmentFormPage()).toBeNull()); // E2B does not advertise interactive setup. - await act(async () => click(editButtons(container)[1])); + await openEnvironmentEditPage(container, 1); await waitForAssertion(() => { expect(getEnvironmentFormPage()?.textContent).toContain("Unsupported provider"); }); @@ -862,7 +877,7 @@ describe("CompanyEnvironments — test provider button", () => { await waitForAssertion(() => expect(getEnvironmentFormPage()).toBeNull()); // Provider advertises setup but cannot capture an image. - await act(async () => click(editButtons(container)[2])); + await openEnvironmentEditPage(container, 2); await waitForAssertion(() => { expect(getEnvironmentFormPage()?.textContent).toContain("Setup capture unavailable"); }); @@ -916,7 +931,7 @@ describe("CompanyEnvironments — test provider button", () => { }); await flushReact(); - await act(async () => click(editButtons(container)[0])); + await openEnvironmentEditPage(container); await waitForAssertion(() => { expect(getEnvironmentFormPage()?.textContent).toContain(command); }); @@ -956,7 +971,7 @@ describe("CompanyEnvironments — test provider button", () => { }); await flushReact(); - await act(async () => click(editButtons(container)[0])); + await openEnvironmentEditPage(container); await waitForAssertion(() => { expect(getEnvironmentFormPage()?.textContent).toContain(command); expect(getEnvironmentFormPage()?.textContent).toContain("Browser terminal"); @@ -1037,7 +1052,7 @@ describe("CompanyEnvironments — test provider button", () => { }); await flushReact(); - await act(async () => click(editButtons(container)[0])); + await openEnvironmentEditPage(container); let terminalScreen: HTMLElement | null = null; await waitForAssertion(() => { terminalScreen = getEnvironmentFormPage()?.querySelector( @@ -1096,7 +1111,7 @@ describe("CompanyEnvironments — test provider button", () => { }); await flushReact(); - await act(async () => click(editButtons(container)[0])); + await openEnvironmentEditPage(container); await waitForAssertion(() => { expect(getEnvironmentFormPage()?.textContent).toContain("Setup expired"); }); @@ -1122,7 +1137,7 @@ describe("CompanyEnvironments — test provider button", () => { }); await flushReact(); - await act(async () => click(editButtons(container)[0])); + await openEnvironmentEditPage(container); await waitForAssertion(() => { expect(getEnvironmentFormPage()?.textContent).toContain("Setup connection details could not be refreshed."); }); @@ -1159,7 +1174,7 @@ describe("CompanyEnvironments — test provider button", () => { }); await flushReact(); - await act(async () => click(editButtons(container)[0])); + await openEnvironmentEditPage(container); await waitForAssertion(() => { expect(getEnvironmentFormPage()?.textContent).toContain("Browser terminal is not available for this provider connection."); }); @@ -1205,7 +1220,7 @@ describe("CompanyEnvironments — test provider button", () => { }); await flushReact(); - await act(async () => click(editButtons(container)[0])); + await openEnvironmentEditPage(container); await waitForAssertion(() => { const dialog = getEnvironmentFormPage(); expect(dialog?.textContent).toContain("Active template"); @@ -1267,7 +1282,7 @@ describe("CompanyEnvironments — test provider button", () => { }); await flushReact(); - await act(async () => click(editButtons(container)[0])); + await openEnvironmentEditPage(container); await waitForAssertion(() => { const dialog = getEnvironmentFormPage()!; expect(dialog.textContent).toContain("Capturing template"); @@ -1308,7 +1323,7 @@ describe("CompanyEnvironments — test provider button", () => { }); await flushReact(); - await act(async () => click(editButtons(container)[0])); + await openEnvironmentEditPage(container); await waitForAssertion(() => { const dialog = getEnvironmentFormPage()!; expect(dialog.textContent).toContain("Active template"); @@ -1339,7 +1354,7 @@ describe("CompanyEnvironments — test provider button", () => { }); await flushReact(); - await act(async () => click(editButtons(container)[0])); + await openEnvironmentEditPage(container); await waitForAssertion(() => { const dialog = getEnvironmentFormPage()!; expect(dialog.textContent).toContain("Not in use — Base image changed: snapshot `a` -> `b`"); @@ -1367,7 +1382,7 @@ describe("CompanyEnvironments — test provider button", () => { }); await flushReact(); - await act(async () => click(editButtons(container)[0])); + await openEnvironmentEditPage(container); await waitForAssertion(() => { const dialog = getEnvironmentFormPage()!; expect(dialog.textContent).toContain("Not in use — the environment configuration changed"); @@ -1394,7 +1409,7 @@ describe("CompanyEnvironments — test provider button", () => { }); await flushReact(); - await act(async () => click(editButtons(container)[0])); + await openEnvironmentEditPage(container); await waitForAssertion(() => { const dialog = getEnvironmentFormPage()!; expect(dialog.textContent).toContain("Active template"); @@ -1437,7 +1452,7 @@ describe("CompanyEnvironments — test provider button", () => { }); await flushReact(); - await act(async () => click(editButtons(container)[0])); + await openEnvironmentEditPage(container); await waitForAssertion(() => { const dialog = getEnvironmentFormPage(); expect(dialog?.textContent).toContain("Active template"); @@ -1496,7 +1511,7 @@ describe("CompanyEnvironments — test provider button", () => { }); await flushReact(); - await act(async () => click(editButtons(container)[0])); + await openEnvironmentEditPage(container); await waitForAssertion(() => { const dialog = getEnvironmentFormPage()!; expect(dialog.textContent).toContain("relink this image or capture a new one"); @@ -1532,7 +1547,7 @@ describe("CompanyEnvironments — test provider button", () => { root!.render(renderCompanyEnvironments(queryClient)); }); await flushReact(); - await act(async () => click(editButtons(container)[0])); + await openEnvironmentEditPage(container); await waitForAssertion(() => { expect(findButton(getEnvironmentFormPage()!, "Relink")).toBeTruthy(); }); @@ -1569,7 +1584,7 @@ describe("CompanyEnvironments — test provider button", () => { root!.render(renderCompanyEnvironments(queryClient)); }); await flushReact(); - await act(async () => click(editButtons(container)[0])); + await openEnvironmentEditPage(container); await waitForAssertion(() => { expect(findButton(getEnvironmentFormPage()!, "Relink")).toBeTruthy(); });