Merge remote-tracking branch 'origin/master' into codex/runner-e2e-s3-images

* origin/master:
  fix(runner): recover native sessions across restarts (#12845)
This commit is contained in:
Dotta 2026-09-04 15:11:01 -05:00
commit aae81f4254
38 changed files with 47365 additions and 261 deletions

View File

@ -251,6 +251,15 @@ that invariant. Removing or disabling a future native rollout flag must not
delete these records; persisted experimental runs remain available for recovery
and inspection.
`native_run_finalizations` also stores restart ownership and recovery state.
The controller owner is a server boot id, PID, operating-system process-start
timestamp, and monotonically increasing controller generation. Recovery writes
its correlated request id, current state, and a bounded JSON history. A
successor can take the lease immediately only when coordinated handoff or PID
and process-start evidence proves the prior controller is gone, or when the
lease expires. Recovery generation changes do not increment the independent
provider-attempt counter.
## Question-response delivery receipts
`issue_question_response_deliveries` is the retry-safe, content-free outbox for

View File

@ -763,6 +763,47 @@ agent workspace. The host `HOME` itself, a directory that contains it, a
filesystem root, a `CODEX_HOME` overlap, or a canonical path outside the
assigned workspace is rejected before provider startup.
### Native runner restart recovery
Paperclip Runner keeps its heartbeat run, native session, logical runner, and
provider session identities across server restarts. A coordinated hot restart
registers a correlated recovery request before it signals the dev supervisor.
An uncoordinated server restart uses the same durable recovery classifier
without trusting a handoff marker.
Startup binds the HTTP and PRP listener before it classifies native runs. Public
health reports a startup state until every candidate is reattached, dispatched
for same-run resume, finalized from durable evidence, or held for explicit
ownership evidence. Scheduling and generic orphan recovery start only after
that classification finishes.
- A verified live runner re-registers its existing PRP authority and reconnects
with the same operating-system PID. Paperclip does not spawn a competing
runner.
- A verified dead runner starts a replacement from the same durable root and
resumes the same provider checkpoint. Only the operating-system PID changes.
- A runner that died before its first authenticated connection can restart on
the same run only when its durable root proves that no provider authority or
checkpoint exists. Paperclip quarantines the incomplete root first.
- A live but mismatched or unverifiable process fails closed. Paperclip does not
signal it or spawn a replacement.
- A persisted proposed or terminal result is reconciled before any runner or
provider work starts, so restart recovery cannot submit a duplicate turn.
Run the credential-free real-process restart suite with:
```sh
pnpm --filter @paperclipai/paperclip-runner build:runner-binaries
pnpm exec vitest run server/src/services/native-runtime/native-runner-restart-recovery.integration.test.ts
pnpm --filter @paperclipai/paperclip-runner exec vitest run src/live/runnerd-codex-transport.test.ts -t 'adopts a live runner'
```
The suite uses isolated PostgreSQL state, isolated `PAPERCLIP_HOME` roots, real
`runnerd` processes, and a deterministic fake Codex app server. It covers hot
and hard restarts with live and dead runners, the result-finalization race,
incomplete bootstrap, repeated crashes with steering, and fail-closed process
identity mismatches.
## App-Shipped Skills Catalog
The Paperclip app ships a curated catalog of company skills out of the box. The

View File

@ -23,6 +23,25 @@ credential material are never written to the run log.
These records remain run-log events. They do not create an OpenTelemetry or
Paperclip Telemetry export, and legacy adapters do not use this writer.
## Native Restart Recovery Run-Log Event
Paperclip writes a `native.recovery.transition` event for every native restart
classification and for graceful restart suspension. This immutable run-log
record lets operators reconstruct recovery decisions without exporting data to
Paperclip Telemetry or OpenTelemetry.
The payload contains the restart kind, recovery request id when one exists,
runner disposition, and the controller generation and provider attempt for a
claimed recovery. Live-runner adoption also records the runner PID, process
group, and process-start fingerprint. A non-claim disposition records a bounded
reason instead. Graceful suspension records the signal and confirms that it did
not create a retry run.
The event never includes bootstrap tickets, reconnect leases, authentication
proofs, encryption keys, environment variables, provider credentials, command
arguments, or an unsanitized stderr stream. Detailed failed-attempt diagnostics
remain in the bounded `native_run_finalizations.recovery_history` ledger.
## Sandbox Startup Run-Log Event
Paperclip writes one `run.startup.step` event to the run log for each bring-up

View File

@ -0,0 +1,7 @@
ALTER TABLE "native_run_finalizations" ADD COLUMN IF NOT EXISTS "controller_boot_id" text;--> statement-breakpoint
ALTER TABLE "native_run_finalizations" ADD COLUMN IF NOT EXISTS "controller_pid" integer;--> statement-breakpoint
ALTER TABLE "native_run_finalizations" ADD COLUMN IF NOT EXISTS "controller_process_started_at" timestamp with time zone;--> statement-breakpoint
ALTER TABLE "native_run_finalizations" ADD COLUMN IF NOT EXISTS "controller_generation" integer DEFAULT 0 NOT NULL;--> statement-breakpoint
ALTER TABLE "native_run_finalizations" ADD COLUMN IF NOT EXISTS "recovery_state" text;--> statement-breakpoint
ALTER TABLE "native_run_finalizations" ADD COLUMN IF NOT EXISTS "recovery_request_id" text;--> statement-breakpoint
ALTER TABLE "native_run_finalizations" ADD COLUMN IF NOT EXISTS "recovery_history" jsonb DEFAULT '[]'::jsonb NOT NULL;

File diff suppressed because it is too large Load Diff

View File

@ -1653,6 +1653,13 @@
"when": 1788296836732,
"tag": "0237_clammy_colonel_america",
"breakpoints": true
},
{
"idx": 238,
"version": "7",
"when": 1788542862021,
"tag": "0238_graceful_infant_terrible",
"breakpoints": true
}
]
}

View File

@ -0,0 +1,155 @@
import { readFile } from "node:fs/promises";
import { afterEach, describe, expect, it } from "vitest";
import postgres from "postgres";
import {
EMBEDDED_POSTGRES_TEST_TIMEOUT_MS,
getEmbeddedPostgresTestSupport,
startEmbeddedPostgresTestDatabase,
} from "./test-embedded-postgres.js";
const cleanups: Array<() => Promise<void>> = [];
const support = await getEmbeddedPostgresTestSupport();
const describeEmbeddedPostgres = support.supported ? describe : describe.skip;
afterEach(async () => {
while (cleanups.length > 0) await cleanups.pop()?.();
});
describeEmbeddedPostgres("native runner recovery migration", () => {
it(
"repairs a partial application and can be replayed without changing the schema",
async () => {
const database = await startEmbeddedPostgresTestDatabase(
"paperclip-native-recovery-migration-",
);
cleanups.push(database.cleanup);
const sql = postgres(database.connectionString, {
max: 1,
onnotice: () => {},
});
cleanups.push(async () => sql.end());
const migration = await readFile(
new URL(
"./migrations/0238_graceful_infant_terrible.sql",
import.meta.url,
),
"utf8",
);
const statements = migration
.split("--> statement-breakpoint")
.map((statement) => statement.trim())
.filter((statement) => statement.length > 0);
// Model an interrupted migration: earlier statements committed, but one of
// the later columns did not. Reapplying must skip completed work and repair
// only the missing portion.
await sql`ALTER TABLE "native_run_finalizations" DROP COLUMN "recovery_request_id"`;
for (const statement of statements) await sql.unsafe(statement);
const columnsAfterRepair = await sql<
{
columnName: string;
dataType: string;
isNullable: "YES" | "NO";
columnDefault: string | null;
}[]
>`
SELECT
column_name AS "columnName",
udt_name AS "dataType",
is_nullable AS "isNullable",
column_default AS "columnDefault"
FROM information_schema.columns
WHERE table_schema = 'public'
AND table_name = 'native_run_finalizations'
AND column_name IN (
'controller_boot_id',
'controller_pid',
'controller_process_started_at',
'controller_generation',
'recovery_state',
'recovery_request_id',
'recovery_history'
)
ORDER BY column_name
`;
for (const statement of statements) await sql.unsafe(statement);
const columnsAfterReplay = await sql<
{
columnName: string;
dataType: string;
isNullable: "YES" | "NO";
columnDefault: string | null;
}[]
>`
SELECT
column_name AS "columnName",
udt_name AS "dataType",
is_nullable AS "isNullable",
column_default AS "columnDefault"
FROM information_schema.columns
WHERE table_schema = 'public'
AND table_name = 'native_run_finalizations'
AND column_name IN (
'controller_boot_id',
'controller_pid',
'controller_process_started_at',
'controller_generation',
'recovery_state',
'recovery_request_id',
'recovery_history'
)
ORDER BY column_name
`;
expect(columnsAfterReplay).toEqual(columnsAfterRepair);
expect(columnsAfterReplay).toEqual([
{
columnName: "controller_boot_id",
dataType: "text",
isNullable: "YES",
columnDefault: null,
},
{
columnName: "controller_generation",
dataType: "int4",
isNullable: "NO",
columnDefault: "0",
},
{
columnName: "controller_pid",
dataType: "int4",
isNullable: "YES",
columnDefault: null,
},
{
columnName: "controller_process_started_at",
dataType: "timestamptz",
isNullable: "YES",
columnDefault: null,
},
{
columnName: "recovery_history",
dataType: "jsonb",
isNullable: "NO",
columnDefault: "'[]'::jsonb",
},
{
columnName: "recovery_request_id",
dataType: "text",
isNullable: "YES",
columnDefault: null,
},
{
columnName: "recovery_state",
dataType: "text",
isNullable: "YES",
columnDefault: null,
},
]);
},
EMBEDDED_POSTGRES_TEST_TIMEOUT_MS,
);
});

View File

@ -26,6 +26,18 @@ export const nativeRunFinalizations = pgTable(
attempt: integer("attempt").notNull().default(0),
leaseOwner: text("lease_owner"),
leaseExpiresAt: timestamp("lease_expires_at", { withTimezone: true }),
controllerBootId: text("controller_boot_id"),
controllerPid: integer("controller_pid"),
controllerProcessStartedAt: timestamp("controller_process_started_at", {
withTimezone: true,
}),
controllerGeneration: integer("controller_generation").notNull().default(0),
recoveryState: text("recovery_state"),
recoveryRequestId: text("recovery_request_id"),
recoveryHistory: jsonb("recovery_history")
.$type<Array<Record<string, unknown>>>()
.notNull()
.default(sql`'[]'::jsonb`),
resultId: uuid("result_id"),
assessmentId: uuid("assessment_id"),
decisionId: uuid("decision_id"),

View File

@ -88,6 +88,11 @@ The state contains:
- lifecycle, reconnect, backpressure, harness generation, and recovery facts;
- bounded, redacted diagnostics.
Raw runner stdout and stderr are never redirected to durable files. The runner
itself redacts and bounds a terminal diagnostic before publishing it through an
atomic private-file replacement, so neither an output burst nor controller
restart can create a transient unbounded or unredacted diagnostic file.
Authentication capabilities and arbitrary command bodies are not stored. A
recent command is represented by a SHA-256 comparison digest from the vetted
RustCrypto implementation, its stable ID and controller sequence, the redacted

View File

@ -1,16 +1,108 @@
use std::path::PathBuf;
use std::fs::{self, OpenOptions};
use std::io::{self, Write};
use std::path::{Path, PathBuf};
use std::process::ExitCode;
use std::time::Duration;
#[cfg(unix)]
use std::os::unix::fs::{OpenOptionsExt, PermissionsExt};
use paperclip_runner_core::durable::{
capture_bootstrap_ticket, run_durable_runner, AcpxLaunchProfile, DurableRunnerConfig,
OpenCodeLaunchProfile, QualifiedLaunchArtifact,
capture_bootstrap_ticket, redact_diagnostic_text, run_durable_runner, AcpxLaunchProfile,
DurableRunnerConfig, OpenCodeLaunchProfile, QualifiedLaunchArtifact,
};
use paperclip_runner_core::local_runner::{run_local_runner, LocalRunnerError, RunnerConfig};
use paperclip_runner_core::native_provider_backend::NativeProviderCommandExecutor;
use serde_json::json;
const RUNNERD_BUILD_METADATA_SCHEMA: &str = "paperclip-runner/runnerd-build-metadata/v1";
const RUNNER_DIAGNOSTIC_MAX_BYTES: usize = 64 * 1024;
const RUNNER_DIAGNOSTIC_TEMP_ATTEMPTS: usize = 32;
fn diagnostic_directory(args: &[String]) -> Option<PathBuf> {
let index = args
.iter()
.position(|argument| argument == "--diagnostics-directory")?;
args.get(index + 1).map(PathBuf::from)
}
fn bounded_redacted_diagnostic(message: &str) -> String {
let mut diagnostic = redact_diagnostic_text(message);
if diagnostic.len() <= RUNNER_DIAGNOSTIC_MAX_BYTES {
return diagnostic;
}
let suffix = "…[truncated]";
let byte_limit = RUNNER_DIAGNOSTIC_MAX_BYTES.saturating_sub(suffix.len());
let boundary = diagnostic
.char_indices()
.map(|(index, _)| index)
.take_while(|index| *index <= byte_limit)
.last()
.unwrap_or(0);
diagnostic.truncate(boundary);
diagnostic.push_str(suffix);
diagnostic
}
fn verify_private_diagnostics_directory(directory: &Path) -> io::Result<()> {
let metadata = fs::symlink_metadata(directory)?;
if metadata.file_type().is_symlink() || !metadata.is_dir() {
return Err(io::Error::new(
io::ErrorKind::InvalidInput,
"runner diagnostics path is not a real directory",
));
}
#[cfg(unix)]
if metadata.permissions().mode() & 0o077 != 0 {
return Err(io::Error::new(
io::ErrorKind::PermissionDenied,
"runner diagnostics directory is accessible by group or other users",
));
}
Ok(())
}
fn persist_runner_diagnostic(directory: &Path, message: &str) -> io::Result<()> {
verify_private_diagnostics_directory(directory)?;
let destination = directory.join("runnerd.stderr.log");
let contents = bounded_redacted_diagnostic(message);
let process_id = std::process::id();
for attempt in 0..RUNNER_DIAGNOSTIC_TEMP_ATTEMPTS {
let temporary = directory.join(format!(".runnerd.stderr.log.{process_id}.{attempt}.tmp"));
let mut options = OpenOptions::new();
options.write(true).create_new(true);
#[cfg(unix)]
options.mode(0o600);
let mut file = match options.open(&temporary) {
Ok(file) => file,
Err(error) if error.kind() == io::ErrorKind::AlreadyExists => continue,
Err(error) => return Err(error),
};
let write_result = (|| {
file.write_all(contents.as_bytes())?;
file.sync_all()?;
drop(file);
fs::rename(&temporary, &destination)
})();
if write_result.is_err() {
let _ = fs::remove_file(&temporary);
}
return write_result;
}
Err(io::Error::new(
io::ErrorKind::AlreadyExists,
"could not allocate a runner diagnostic temporary file",
))
}
fn install_diagnostic_panic_hook(directory: Option<PathBuf>) {
let Some(directory) = directory else {
return;
};
std::panic::set_hook(Box::new(move |panic| {
let _ = persist_runner_diagnostic(&directory, &format!("paperclip-runnerd panic: {panic}"));
}));
}
fn build_metadata() -> serde_json::Value {
json!({
@ -237,9 +329,8 @@ fn run_durable(args: &[String]) -> Result<(), LocalRunnerError> {
.map_err(|error| LocalRunnerError::invalid(error.to_string()))
}
fn run() -> Result<(), LocalRunnerError> {
let args = std::env::args().skip(1).collect::<Vec<_>>();
if args.as_slice() == ["--build-metadata"] {
fn run(args: &[String]) -> Result<(), LocalRunnerError> {
if args == ["--build-metadata"] {
println!("{}", build_metadata());
return Ok(());
}
@ -248,22 +339,22 @@ fn run() -> Result<(), LocalRunnerError> {
|| args.iter().any(|argument| argument == "--listen-port")
|| args.iter().any(|argument| argument == "--listen-path")
{
return run_durable(&args);
return run_durable(args);
}
run_local_runner(RunnerConfig {
run_id: value(&args, "--run-id")?,
normalized_session_id: value(&args, "--session-id")?,
runner_instance_id: value(&args, "--runner-id")?,
fake_harness_path: PathBuf::from(value(&args, "--fake-harness")?),
script_path: PathBuf::from(value(&args, "--script")?),
delay_override_ms: optional_u64(&args, "--delay-ms")?,
log_max_lines: usize_value(&args, "--log-max-lines", 32)?,
log_max_bytes: usize_value(&args, "--log-max-bytes", 16_384)?,
command_history_limit: usize_value(&args, "--command-history-limit", 4096)?,
controller_max_line_bytes: usize_value(&args, "--controller-max-line-bytes", 64 * 1024)?,
harness_max_line_bytes: usize_value(&args, "--harness-max-line-bytes", 64 * 1024)?,
run_id: value(args, "--run-id")?,
normalized_session_id: value(args, "--session-id")?,
runner_instance_id: value(args, "--runner-id")?,
fake_harness_path: PathBuf::from(value(args, "--fake-harness")?),
script_path: PathBuf::from(value(args, "--script")?),
delay_override_ms: optional_u64(args, "--delay-ms")?,
log_max_lines: usize_value(args, "--log-max-lines", 32)?,
log_max_bytes: usize_value(args, "--log-max-bytes", 16_384)?,
command_history_limit: usize_value(args, "--command-history-limit", 4096)?,
controller_max_line_bytes: usize_value(args, "--controller-max-line-bytes", 64 * 1024)?,
harness_max_line_bytes: usize_value(args, "--harness-max-line-bytes", 64 * 1024)?,
shutdown_grace: Duration::from_millis(
optional_u64(&args, "--shutdown-grace-ms")?.unwrap_or(100),
optional_u64(args, "--shutdown-grace-ms")?.unwrap_or(100),
),
})
}
@ -282,13 +373,71 @@ mod tests {
json!(["dial_ws_loopback", "dial_wss", "listen_ws"])
);
}
#[test]
fn persistent_diagnostic_is_private_bounded_and_redacted() {
let unique = format!(
"paperclip-runnerd-diagnostic-test-{}-{}",
std::process::id(),
std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.expect("test clock is after the Unix epoch")
.as_nanos()
);
let directory = std::env::temp_dir().join(unique);
fs::create_dir(&directory).expect("create diagnostic test directory");
#[cfg(unix)]
fs::set_permissions(&directory, fs::Permissions::from_mode(0o700))
.expect("make diagnostic test directory private");
let path = directory.join("runnerd.stderr.log");
fs::write(&path, "").expect("create runner diagnostic placeholder");
#[cfg(unix)]
fs::set_permissions(&path, fs::Permissions::from_mode(0o600))
.expect("make runner diagnostic placeholder private");
let diagnostic = format!(
"OPENAI_API_KEY=raw-secret Authorization: Bearer another-secret {}",
"x".repeat(RUNNER_DIAGNOSTIC_MAX_BYTES * 4)
);
persist_runner_diagnostic(&directory, &diagnostic)
.expect("persist bounded runner diagnostic");
let persisted = fs::read_to_string(&path).expect("read runner diagnostic");
assert!(persisted.len() <= RUNNER_DIAGNOSTIC_MAX_BYTES);
assert!(!persisted.contains("raw-secret"));
assert!(!persisted.contains("another-secret"));
assert!(persisted.contains("[REDACTED]"));
#[cfg(unix)]
assert_eq!(
fs::metadata(&path)
.expect("read runner diagnostic metadata")
.permissions()
.mode()
& 0o777,
0o600
);
fs::remove_dir_all(directory).expect("remove diagnostic test directory");
}
}
fn main() -> ExitCode {
match run() {
let args = std::env::args().skip(1).collect::<Vec<_>>();
let diagnostics_directory = diagnostic_directory(&args);
install_diagnostic_panic_hook(diagnostics_directory.clone());
match run(&args) {
Ok(()) => ExitCode::SUCCESS,
Err(error) => {
eprintln!("paperclip-runnerd: {error}");
let message = format!("paperclip-runnerd: {error}");
if let Some(directory) = diagnostics_directory {
if let Err(persist_error) = persist_runner_diagnostic(&directory, &message) {
eprintln!(
"paperclip-runnerd: failed to persist bounded diagnostic: {persist_error}"
);
}
} else {
eprintln!("{message}");
}
ExitCode::FAILURE
}
}

View File

@ -28,6 +28,11 @@ pub const BOOTSTRAP_TICKET_ENV: &str = "PAPERCLIP_RUNNER_BOOTSTRAP_TICKET";
const MAX_OUTBOX_BYTES: usize = 512 * 1024 * 1024;
const MAX_FRAME_BYTES: usize = 16 * 1024 * 1024;
/// Redacts and bounds a diagnostic before a process boundary may persist it.
pub fn redact_diagnostic_text(input: &str) -> String {
state::redact_text(input)
}
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct DurableRunnerError(String);

View File

@ -4,7 +4,13 @@ import {
createHash,
createHmac,
} from "node:crypto";
import { mkdtempSync, rmSync } from "node:fs";
import {
chmodSync,
mkdtempSync,
readFileSync,
rmSync,
writeFileSync,
} from "node:fs";
import { connect, type Socket } from "node:net";
import { tmpdir } from "node:os";
import { resolve } from "node:path";
@ -31,6 +37,49 @@ const identity: DurableRecoveryIdentity = {
const expectedRunnerVersion = "0.3.0";
const expectedRunnerDigest = `sha256:${"a".repeat(64)}`;
it.skipIf(process.platform === "win32")(
"never persists raw child stdout or stderr as durable diagnostics",
async () => {
const root = mkdtempSync(resolve(tmpdir(), "runner-diagnostics-test-"));
const executable = resolve(root, "noisy-runner");
const diagnosticsDirectory = resolve(root, "diagnostics");
writeFileSync(
executable,
[
"#!/usr/bin/env node",
'process.stdout.write("token=raw-stdout-secret " + "o".repeat(256 * 1024));',
'process.stderr.write("Authorization: Bearer raw-stderr-secret " + "e".repeat(256 * 1024));',
].join("\n"),
{ mode: 0o700 },
);
chmodSync(executable, 0o700);
try {
const handle = spawnRunner({
connection: { mode: "connect", connectUrl: "ws://127.0.0.1:43127" },
stateDirectory: resolve(root, "state"),
identity,
ticket: "bootstrap-ticket",
maxOutboxBytes: 256 * 1024,
p0ReserveBytes: 64 * 1024,
runnerVersion: expectedRunnerVersion,
runnerDigest: expectedRunnerDigest,
runnerBinaryPath: executable,
diagnosticsDirectory,
});
const result = await handle.completion;
expect(result.stdout).toBe("");
expect(result.stderr).toBe("");
for (const name of ["runnerd.stdout.log", "runnerd.stderr.log"]) {
const filePath = resolve(diagnosticsDirectory, name);
expect(readFileSync(filePath, "utf8")).toBe("");
}
} finally {
rmSync(root, { recursive: true, force: true });
}
},
);
it("pins the ACPX launch profile in runner startup arguments and restarts", () => {
const launches: RunnerProcessLaunchSpec[] = [];
const handle = spawnRunner({

View File

@ -226,6 +226,8 @@ export interface RunnerProcessHandle {
kill(signal?: NodeJS.Signals | number): boolean;
};
completion: Promise<RunnerProcessResult>;
processGroupId?: number | null;
startedAt?: string;
/** Relaunches the same immutable process specification with a fresh ticket. */
restart?(ticket: string): RunnerProcessHandle;
}
@ -1993,6 +1995,7 @@ export function spawnRunner(options: {
};
environment?: NodeJS.ProcessEnv;
processLauncher?: (spec: RunnerProcessLaunchSpec) => RunnerProcessHandle;
diagnosticsDirectory?: string;
}): RunnerProcessHandle {
const connection =
options.connection ??
@ -2096,6 +2099,9 @@ export function spawnRunner(options: {
);
}
}
if (options.diagnosticsDirectory !== undefined) {
args.push("--diagnostics-directory", options.diagnosticsDirectory);
}
const command = options.runnerBinaryPath ?? runnerBinary;
const environment = runnerEnvironment(options.ticket, options.environment);
@ -2109,28 +2115,75 @@ export function spawnRunner(options: {
);
}
const detached = process.platform !== "win32";
const diagnosticsDirectory = options.diagnosticsDirectory;
let stdoutPath: string | null = null;
let stderrPath: string | null = null;
if (diagnosticsDirectory) {
try {
const metadata = lstatSync(diagnosticsDirectory);
if (metadata.isSymbolicLink() || !metadata.isDirectory()) {
throw new Error(
`Private state directory is not a real directory: ${diagnosticsDirectory}`,
);
}
} catch (error) {
if (!isNodeError(error, "ENOENT")) throw error;
mkdirSync(diagnosticsDirectory, { recursive: true, mode: 0o700 });
}
if (process.platform !== "win32") chmodSync(diagnosticsDirectory, 0o700);
verifyPrivateDirectory(diagnosticsDirectory);
stdoutPath = resolve(diagnosticsDirectory, "runnerd.stdout.log");
stderrPath = resolve(diagnosticsDirectory, "runnerd.stderr.log");
// runnerd owns every durable diagnostic write so it can redact and bound
// the complete value before a byte reaches disk. Raw process output is
// intentionally discarded below; these files are only the runner-owned
// restart-survivable diagnostic channel.
atomicPrivateWrite(stdoutPath, "");
atomicPrivateWrite(stderrPath, "");
}
const child = spawn(command, args, {
cwd: packageRoot,
env: environment,
stdio: "pipe",
detached,
stdio: diagnosticsDirectory ? "ignore" : "pipe",
});
child.unref();
let stdout = "";
let stderr = "";
child.stdout.setEncoding("utf8").on("data", (chunk: string) => {
child.stdout?.setEncoding("utf8").on("data", (chunk: string) => {
stdout = `${stdout}${chunk}`.slice(-16_384);
});
child.stderr.setEncoding("utf8").on("data", (chunk: string) => {
child.stderr?.setEncoding("utf8").on("data", (chunk: string) => {
stderr = `${stderr}${chunk}`.slice(-16_384);
});
const completion = new Promise<RunnerProcessResult>(
const boundedDiagnostic = (filePath: string | null): string => {
if (!filePath) return "";
try {
return (readPrivateFile(filePath) ?? "").slice(-16_384);
} catch {
return "";
}
};
const processCompletion = new Promise<RunnerProcessResult>(
(resolveCompletion, rejectCompletion) => {
child.once("error", rejectCompletion);
child.once("exit", (code, signal) =>
resolveCompletion({ code, signal, stdout, stderr }),
resolveCompletion({
code,
signal,
stdout: stdout || boundedDiagnostic(stdoutPath),
stderr: stderr || boundedDiagnostic(stderrPath),
}),
);
},
);
return withRestart({ child, completion });
return withRestart({
child,
completion: processCompletion,
processGroupId: detached ? (child.pid ?? null) : null,
startedAt: new Date().toISOString(),
});
}
export async function waitForProcess(
@ -2143,7 +2196,19 @@ export async function waitForProcess(
handle.completion,
new Promise<never>((_resolve, reject) => {
timer = setTimeout(() => {
handle.child.kill("SIGKILL");
if (
process.platform !== "win32" &&
handle.processGroupId &&
handle.processGroupId > 0
) {
try {
process.kill(-handle.processGroupId, "SIGKILL");
} catch {
handle.child.kill("SIGKILL");
}
} else {
handle.child.kill("SIGKILL");
}
reject(new Error("Durable recovery runner timed out."));
}, timeoutMs);
}),

View File

@ -9,11 +9,13 @@ import {
writeFile,
} from "node:fs/promises";
import { createHash } from "node:crypto";
import { createServer } from "node:http";
import { tmpdir } from "node:os";
import { join, resolve } from "node:path";
import { fileURLToPath } from "node:url";
import { expect, it } from "vitest";
import { expect, it, vi } from "vitest";
import type { DurablePrpControlPlane } from "../control-plane/durable-prp-control-plane.js";
import {
NATIVE_RUNTIME_ASSET_SCHEMA,
@ -1985,6 +1987,122 @@ it("cold-restores a suspended provider session under its durable run binding", a
}
}, 30_000);
it("adopts a live runner on the same durable authority without spawning a duplicate", async () => {
const stateDirectory = await mkdtemp(join(tmpdir(), "runnerd-live-adopt-"));
const server = createServer();
let authority: DurablePrpControlPlane | null = null;
server.on("upgrade", (request, socket, head) => {
if (!authority) {
socket.destroy();
return;
}
authority.handleUpgrade(request, socket, "/runner", head);
});
await new Promise<void>((resolveListen) =>
server.listen(0, "127.0.0.1", resolveListen),
);
const address = server.address();
if (!address || typeof address === "string")
throw new Error("Expected adoption test listener");
const registration = async (next: DurablePrpControlPlane) => {
authority = next;
return {
connectUrl: `ws://127.0.0.1:${address.port}/runner`,
release: async () => {
if (authority === next) authority = null;
},
};
};
const identity = {
runnerInstanceId: "runner-live-adopt",
environmentLeaseId: "lease-live-adopt",
runId: "run-live-adopt",
normalizedSessionId: "session-live-adopt",
turnId: "turn-live-adopt",
itemId: "item-live-adopt",
};
const sharedOptions = {
runnerBinary: defaultCapabilityRunnerdBinary(),
codexCommand: fakeCodex,
codexArgs: fakeCodexArgs(stateDirectory),
stateDirectory,
prpIdentity: identity,
lifecyclePolicy: { mode: "warm" as const, idleTimeoutMs: 60_000 },
controlPlaneRegistration: registration,
};
const first = createCapabilityRunnerdCodexTransport(sharedOptions);
let runnerPid: number | null = null;
let adopted: ReturnType<typeof createCapabilityRunnerdCodexTransport> | null =
null;
try {
try {
await first.transport.request("thread/start", {
cwd: tmpdir(),
dynamicTools: codexSemanticToolSpecs(),
});
} catch (error) {
const stderr = await readFile(
join(stateDirectory, "diagnostics", "runnerd.stderr.log"),
"utf8",
).catch(() => "");
throw new Error(
`${String(error)}\n${JSON.stringify(first.evidence())}${stderr ? `\n${stderr}` : ""}`,
);
}
runnerPid = first.evidence().runnerPid;
expect(runnerPid).toEqual(expect.any(Number));
await first.detachControllerForRestart();
expect(() => process.kill(runnerPid!, 0)).not.toThrow();
const duplicateLauncher = vi.fn(() => {
throw new Error("duplicate runner spawn attempted");
});
adopted = createCapabilityRunnerdCodexTransport({
...sharedOptions,
resumeDynamicTools: [],
runnerProcessLauncher: duplicateLauncher,
adoptExistingRunner: {
pid: runnerPid!,
processGroupId: runnerPid,
startedAt: new Date().toISOString(),
isAlive: () => {
try {
process.kill(runnerPid!, 0);
return true;
} catch {
return false;
}
},
},
});
await expect(adopted.transport.request("thread/read", {})).resolves.toEqual(
expect.objectContaining({
thread: expect.objectContaining({ id: "codex-thread-1" }),
}),
);
expect(adopted.evidence().runnerPid).toBe(runnerPid);
expect(duplicateLauncher).not.toHaveBeenCalled();
expect(adopted.evidence().diagnostics).toContain(
`adopted runner ${runnerPid} authenticated to its durable PRP authority`,
);
} finally {
await adopted?.transport.close().catch(() => undefined);
if (runnerPid) {
try {
process.kill(-runnerPid, "SIGKILL");
} catch {
// The adopted runner normally exits after its durable suspend command.
}
}
server.closeAllConnections();
if (server.listening) {
await new Promise<void>((resolveClose) => server.close(() => resolveClose()));
}
await rm(stateDirectory, { recursive: true, force: true });
}
}, 30_000);
it("surfaces a runner exit while provider-ingress readiness is still pending", async () => {
const neverReady = new Promise<void>(() => undefined);
const bundle = createCapabilityRunnerdCodexTransport({

View File

@ -1,3 +1,4 @@
import { execFileSync } from "node:child_process";
import { createHash, randomUUID } from "node:crypto";
import {
appendFileSync,
@ -29,7 +30,10 @@ import {
codexSemanticToolSpecs,
createIsolatedCodexAppServerArgs,
} from "../drivers/codex/codex-app-server-driver.js";
import type { DurableRecoveryIdentity } from "../contracts/durable-recovery.js";
import type {
DurableRecoveryCommittedEvent,
DurableRecoveryIdentity,
} from "../contracts/durable-recovery.js";
import type { HarnessRuntimeRequestResolution } from "../contracts/harness-driver.js";
import {
DurablePrpControlPlane,
@ -70,6 +74,50 @@ const MAX_NOTIFICATION_BYTES = 4 * 1024 * 1024;
const RUNNER_CLIENT_VERSION = "0.3.0";
const RUNNER_BOOTSTRAP_TICKET_TTL_MS = 60_000;
function readLocalProcessStartedAt(pid: number): string | null {
if (!Number.isInteger(pid) || pid <= 0) return null;
try {
if (process.platform === "linux") {
return new Date(statSync(`/proc/${pid}`).ctimeMs).toISOString();
}
if (
["darwin", "freebsd", "openbsd", "aix", "sunos"].includes(
process.platform,
)
) {
const raw = execFileSync("ps", ["-o", "lstart=", "-p", String(pid)], {
encoding: "utf8",
timeout: 1_500,
windowsHide: true,
}).trim();
const parsed = new Date(raw);
return Number.isNaN(parsed.getTime()) ? null : parsed.toISOString();
}
if (process.platform === "win32") {
const script = [
`$target = Get-Process -Id ${pid} -ErrorAction Stop`,
"$target.StartTime.ToUniversalTime().ToString('o')",
].join("; ");
for (const command of ["powershell.exe", "pwsh.exe"]) {
try {
const raw = execFileSync(
command,
["-NoLogo", "-NoProfile", "-NonInteractive", "-Command", script],
{ encoding: "utf8", timeout: 1_500, windowsHide: true },
).trim();
const parsed = new Date(raw);
if (!Number.isNaN(parsed.getTime())) return parsed.toISOString();
} catch {
// Try the other supported PowerShell host.
}
}
}
} catch {
// Process exit and restricted process metadata both produce no fingerprint.
}
return null;
}
const CODEX_COLLABORATION_RUNTIME_INSTRUCTIONS = `## Codex-style collaboration
- Before the first tool call in a turn, send a brief commentary update describing the immediate work you are starting.
@ -307,20 +355,10 @@ function recoveredRunAttachment(state: {
.reverse()
.find((candidate) => candidate.type === "run.attach");
if (!command) return null;
let providerIdentityEventIndex = -1;
if (command.status === "completed") {
for (let index = state.committedEvents.length - 1; index >= 0; index -= 1) {
const eventType = state.committedEvents[index]?.eventType;
if (
eventType === "harness.ready" ||
eventType === "session.started" ||
eventType === "session.resumed"
) {
providerIdentityEventIndex = index;
break;
}
}
}
const providerIdentityEventIndex =
command.status === "completed"
? latestProviderIdentityEventIndex(state.committedEvents)
: -1;
return {
commandId: command.commandId,
status: command.status,
@ -328,6 +366,22 @@ function recoveredRunAttachment(state: {
};
}
function latestProviderIdentityEventIndex(
events: readonly { eventType: string }[],
): number {
for (let index = events.length - 1; index >= 0; index -= 1) {
const eventType = events[index]?.eventType;
if (
eventType === "harness.ready" ||
eventType === "session.started" ||
eventType === "session.resumed"
) {
return index;
}
}
return -1;
}
function providerDrainStateFromSnapshot(state: Record<string, unknown>): {
pendingEventCount: number;
activeProviderTurnId: string | null;
@ -532,9 +586,13 @@ export interface CapabilityRunnerdProcessEvidence {
runnerPid: number | null;
runnerProcessGroupId: number | null;
providerPid: number | null;
providerProcessStartedAt: string | null;
codexPid: number | null;
codexProcessStartedAt: string | null;
sidecarPid: number | null;
sidecarProcessStartedAt: string | null;
agentPid: number | null;
agentProcessStartedAt: string | null;
providerDriver: string | null;
providerVersion: string | null;
acpxAgent: QualifiedAcpxAgent | null;
@ -669,6 +727,18 @@ export interface CapabilityRunnerdCodexTransportOptions {
readRunnerState?: () => Promise<Record<string, unknown>>;
/** Active-connection recovery budget. Omitted for the existing local mode. */
runnerReconnectGraceMs?: number;
/**
* A verified local runner that outlived its controller. Adoption registers
* the durable authority and waits for this exact process to reconnect; it
* never calls the process launcher while the process remains alive.
*/
adoptExistingRunner?: {
pid: number;
processGroupId: number | null;
startedAt: string;
isAlive: () => Promise<boolean> | boolean;
signal?: (signal: NodeJS.Signals) => Promise<boolean> | boolean;
};
}
export type RunnerdCodexTransportOptions =
@ -677,6 +747,8 @@ export type RunnerdCodexTransportOptions =
export interface CapabilityRunnerdCodexTransport {
transport: CodexAppServerTransport;
evidence(): Readonly<CapabilityRunnerdProcessEvidence>;
/** Relinquish controller authority without stopping the durable runner. */
detachControllerForRestart(): Promise<void>;
}
export type RunnerdCodexTransport = CapabilityRunnerdCodexTransport;
@ -1581,6 +1653,7 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
});
#core: DurablePrpControlPlane | null = null;
#handle: RunnerProcessHandle | null = null;
#adoptedRunnerMonitor: NodeJS.Timeout | null = null;
#pump: NodeJS.Timeout | null = null;
#eventIndex = 0;
#threadId = "";
@ -1633,9 +1706,13 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
runnerPid: null,
runnerProcessGroupId: null,
providerPid: null,
providerProcessStartedAt: null,
codexPid: null,
codexProcessStartedAt: null,
sidecarPid: null,
sidecarProcessStartedAt: null,
agentPid: null,
agentProcessStartedAt: null,
providerDriver: null,
providerVersion: null,
acpxAgent: null,
@ -1746,6 +1823,7 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
snapshot.activeProviderTurnId.length > 0
? snapshot.activeProviderTurnId
: null;
if (activeProviderTurnId !== null) this.#turnId = activeProviderTurnId;
const recoveredTurns: Array<Record<string, unknown>> =
activeProviderTurnId === null
? []
@ -1985,7 +2063,7 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
);
return false;
}
if (this.#handle?.child.exitCode !== null) return false;
if (await this.#runnerHasExited()) return false;
await new Promise((resolveWait) => setTimeout(resolveWait, 5));
}
this.#diagnostic("provider turn stop timed out before runner suspension");
@ -2050,13 +2128,37 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
return this.#closePromise;
}
async detachControllerForRestart(): Promise<void> {
if (this.#closed) return;
this.#closed = true;
if (this.#pump !== null) clearInterval(this.#pump);
this.#pump = null;
if (this.#adoptedRunnerMonitor !== null)
clearInterval(this.#adoptedRunnerMonitor);
this.#adoptedRunnerMonitor = null;
this.#core?.disconnectActiveRunner();
if (this.#controlPlaneRelease !== null) await this.#controlPlaneRelease();
await this.#core?.stop();
this.#controlPlaneRelease = null;
this.#handle = null;
this.#queue.close();
this.#diagnostic(
"controller authority detached for restart; durable runner left alive",
);
}
async #closeOnce(): Promise<void> {
this.#closed = true;
if (this.#core !== null && this.#handle !== null) {
const adoptedRunner = this.options.adoptExistingRunner;
if (
this.#core !== null &&
(this.#handle !== null || adoptedRunner !== undefined) &&
(this.#failure === null || this.#startupComplete)
) {
// A terminal provider frame can become visible one control loop before
// its durable provider suffix is ACKed. Drain it before suspension so a
// fresh run authority never inherits the prior run's pending events.
if (this.#handle.child.exitCode === null) {
if (!(await this.#runnerHasExited())) {
const stoppedActiveTurn =
await this.#stopActiveProviderTurnBeforeSuspend();
await this.#drainSettledProviderEventsBeforeSuspend(
@ -2064,7 +2166,7 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
);
}
const runnerAlreadyStopping =
this.#handle.child.exitCode !== null ||
(await this.#runnerHasExited()) ||
this.#core.store.state.commands.some(
(command) =>
(command.type === "runner.suspend" ||
@ -2079,15 +2181,26 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
this.#core.queueCommand("runner.suspend", {}, undefined, true);
}
try {
const result = await waitForProcess(
this.#handle,
this.options.closeGraceMs ?? 10_000,
);
this.#evidence.runnerExited = true;
this.#evidence.runnerExitCode = result.code;
this.#evidence.runnerSignal = result.signal as NodeJS.Signals | null;
if (result.stderr.trim())
this.#diagnostic(result.stderr.trim().slice(-4_096));
if (this.#handle) {
const result = await waitForProcess(
this.#handle,
this.options.closeGraceMs ?? 10_000,
);
this.#evidence.runnerExited = true;
this.#evidence.runnerExitCode = result.code;
this.#evidence.runnerSignal = result.signal as NodeJS.Signals | null;
if (result.stderr.trim())
this.#diagnostic(result.stderr.trim().slice(-4_096));
} else if (adoptedRunner) {
const deadline = Date.now() + (this.options.closeGraceMs ?? 10_000);
while ((await adoptedRunner.isAlive()) && Date.now() < deadline) {
await new Promise((resolveWait) => setTimeout(resolveWait, 25));
}
if (await adoptedRunner.isAlive()) {
await adoptedRunner.signal?.("SIGKILL");
}
this.#evidence.runnerExited = !(await adoptedRunner.isAlive());
}
} catch (error) {
this.#diagnostic(`runner shutdown failed: ${String(error)}`);
}
@ -2119,6 +2232,9 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
}
if (this.#pump !== null) clearInterval(this.#pump);
this.#pump = null;
if (this.#adoptedRunnerMonitor !== null)
clearInterval(this.#adoptedRunnerMonitor);
this.#adoptedRunnerMonitor = null;
this.#queue.close();
// Ensure a runner that missed or could not finish the graceful lifecycle
// command cannot keep the control-plane server alive during teardown.
@ -2524,6 +2640,7 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
}),
this.options.environment,
),
diagnosticsDirectory: resolve(this.#root, "diagnostics"),
processLauncher: this.options.runnerProcessLauncher,
});
this.#handle = handle;
@ -2538,7 +2655,7 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
}
await this.#awaitRegistrationReady(registration?.ready);
this.#evidence.runnerPid = handle.child.pid ?? null;
this.#evidence.runnerProcessGroupId = null;
this.#evidence.runnerProcessGroupId = handle.processGroupId ?? null;
this.#publish();
this.#pump = setInterval(() => this.#pumpEventsSafely(), 5);
await this.#waitCommand("run.prepare");
@ -2798,6 +2915,17 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
runAttachment !== null && runAttachment.providerIdentityEventIndex >= 0
? runAttachment.providerIdentityEventIndex
: committedEvents.length;
const adoptedProviderIdentityIndex =
latestProviderIdentityEventIndex(committedEvents);
if (
this.options.adoptExistingRunner &&
exactAuthority &&
adoptedProviderIdentityIndex >= 0
) {
this.#applyProviderIdentityEvent(
committedEvents[adoptedProviderIdentityIndex]!,
);
}
const registration = this.options.controlPlaneRegistration
? await this.options.controlPlaneRegistration(core)
: null;
@ -2805,44 +2933,50 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
registration?.startupFailureCode ?? "runner_local_connect_failed";
if (registration === null) await core.start();
else this.#controlPlaneRelease = registration.release;
const handle = spawnRunner({
connection: registration?.connection ?? {
mode: "connect",
connectUrl: registration?.connectUrl ?? core.connectUrl,
},
stateDirectory:
this.options.runnerStateDirectory ?? resolve(this.#root, "runner"),
identity,
ticket: core.issueBootstrapTicket(RUNNER_BOOTSTRAP_TICKET_TTL_MS),
maxOutboxBytes: 256 * 1024,
p0ReserveBytes: 64 * 1024,
maxRuntimeMs: 60 * 60 * 1_000,
reconnectGraceMs: this.options.runnerReconnectGraceMs,
lifecyclePolicy: this.options.lifecyclePolicy,
runnerBinaryPath,
runnerVersion: runnerArtifact.version,
runnerDigest: runnerArtifact.digest,
acpxLaunchProfile: runnerAcpxLaunchProfile,
opencodeLaunchProfile: runnerOpenCodeLaunchProfile,
environment: withRunnerdProviderTrace(
createCapabilityRunnerdProviderEnvironment({
provider,
options: {
...this.options,
stateDirectory: this.#root,
const adoptedRunner = this.options.adoptExistingRunner;
const handle = adoptedRunner
? null
: spawnRunner({
connection: registration?.connection ?? {
mode: "connect",
connectUrl: registration?.connectUrl ?? core.connectUrl,
},
stateDirectory:
this.options.runnerStateDirectory ?? resolve(this.#root, "runner"),
identity,
codexHome,
runtimeContextPath,
hasRuntimeContext: runtimeContext !== null,
acpxSidecarPath,
}),
this.options.environment,
),
processLauncher: this.options.runnerProcessLauncher,
});
this.#handle = handle;
this.#watchRunner(handle);
ticket: core.issueBootstrapTicket(RUNNER_BOOTSTRAP_TICKET_TTL_MS),
maxOutboxBytes: 256 * 1024,
p0ReserveBytes: 64 * 1024,
maxRuntimeMs: 60 * 60 * 1_000,
reconnectGraceMs: this.options.runnerReconnectGraceMs,
lifecyclePolicy: this.options.lifecyclePolicy,
runnerBinaryPath,
runnerVersion: runnerArtifact.version,
runnerDigest: runnerArtifact.digest,
acpxLaunchProfile: runnerAcpxLaunchProfile,
opencodeLaunchProfile: runnerOpenCodeLaunchProfile,
environment: withRunnerdProviderTrace(
createCapabilityRunnerdProviderEnvironment({
provider,
options: {
...this.options,
stateDirectory: this.#root,
},
identity,
codexHome,
runtimeContextPath,
hasRuntimeContext: runtimeContext !== null,
acpxSidecarPath,
}),
this.options.environment,
),
diagnosticsDirectory: resolve(this.#root, "diagnostics"),
processLauncher: this.options.runnerProcessLauncher,
});
if (handle) {
this.#handle = handle;
this.#watchRunner(handle);
}
await registration?.activate?.();
if (registration?.failure) {
void registration.failure.catch((error: unknown) => {
@ -2852,10 +2986,17 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
});
}
await this.#awaitRegistrationReady(registration?.ready);
this.#evidence.runnerPid = handle.child.pid ?? null;
this.#evidence.runnerProcessGroupId = null;
if (adoptedRunner) {
this.#evidence.runnerPid = adoptedRunner.pid;
this.#evidence.runnerProcessGroupId = adoptedRunner.processGroupId;
this.#watchAdoptedRunner(adoptedRunner);
} else {
this.#evidence.runnerPid = handle?.child.pid ?? null;
this.#evidence.runnerProcessGroupId = handle?.processGroupId ?? null;
}
this.#publish();
this.#pump = setInterval(() => this.#pumpEventsSafely(), 5);
if (adoptedRunner) await this.#awaitAdoptedRunnerConnection(adoptedRunner);
if (runAttachment) {
await this.#waitCommand("run.attach", runAttachment.commandId);
}
@ -2890,7 +3031,7 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
this.#throwIfFailed();
this.#pumpEvents();
if (this.#turnId !== pendingTurnId) break;
if (this.#handle?.child.exitCode !== null)
if (await this.#runnerHasExited())
throw new Error("runnerd exited before provider turn startup");
await new Promise((resolveWait) => setTimeout(resolveWait, 10));
}
@ -2978,7 +3119,7 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
this.#evidence.providerPid !== null)
)
return;
if (this.#handle?.child.exitCode !== null)
if (await this.#runnerHasExited())
throw new Error("runnerd exited before provider startup");
await new Promise((resolveWait) => setTimeout(resolveWait, 10));
}
@ -3000,7 +3141,7 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
`PRP command ${type} ${command.status}: ${JSON.stringify(command.result)}`,
);
}
if (this.#handle?.child.exitCode !== null)
if (await this.#runnerHasExited())
throw new Error(`runnerd exited while waiting for ${type}`);
await new Promise((resolveWait) => setTimeout(resolveWait, 10));
}
@ -3034,58 +3175,7 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
event.eventType === "session.started" ||
event.eventType === "session.resumed"
) {
const started = record(record(event.envelope.payload).payload);
const runtimeIdentity = record(started.runtimeIdentity);
const descriptor = record(started.providerDescriptor);
const {
processId: pid,
threadId,
sessionId,
} = resolveRunnerdSessionIdentity(started);
const providerIdentity = record(started.providerIdentity);
if (pid !== null) {
this.#evidence.providerPid = pid;
if (descriptor.driver === "acpx_runtime")
this.#evidence.sidecarPid = pid;
else this.#evidence.codexPid = pid;
}
if (typeof descriptor.driver === "string")
this.#evidence.providerDriver = descriptor.driver;
if (typeof descriptor.providerVersion === "string")
this.#evidence.providerVersion = descriptor.providerVersion;
if (
descriptor.agent === "pi" ||
descriptor.agent === "claude" ||
descriptor.agent === "codex"
)
this.#evidence.acpxAgent = descriptor.agent;
if (typeof descriptor.agentServerVersion === "string")
this.#evidence.agentServerVersion = descriptor.agentServerVersion;
if (typeof descriptor.agentRuntimeVersion === "string")
this.#evidence.agentRuntimeVersion = descriptor.agentRuntimeVersion;
if (typeof descriptor.acpProtocolVersion === "number")
this.#evidence.acpProtocolVersion = descriptor.acpProtocolVersion;
if (typeof descriptor.agentProcessId === "number")
this.#evidence.agentPid = descriptor.agentProcessId;
if (
runtimeIdentity.executionKind === "local_process" ||
runtimeIdentity.executionKind === "remote_service"
) {
this.#evidence.providerExecutionKind = runtimeIdentity.executionKind;
}
if (runtimeIdentity.service === "anthropic_managed_agents") {
this.#evidence.providerService = "anthropic_managed_agents";
} else if (
runtimeIdentity.service === "aws_bedrock_agentcore_harness"
) {
this.#evidence.providerService = "aws_bedrock_agentcore_harness";
}
if (threadId !== null) this.#threadId = threadId;
if (sessionId !== null) this.#sessionId = sessionId;
if (typeof providerIdentity.kind === "string") {
this.#providerIdentity = structuredClone(providerIdentity);
}
this.#publish();
this.#applyProviderIdentityEvent(event);
continue;
}
if (event.eventType === "harness.diagnostic") {
@ -3096,6 +3186,9 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
typeof diagnostic.pid === "number"
) {
this.#evidence.agentPid = diagnostic.pid;
this.#evidence.agentProcessStartedAt = readLocalProcessStartedAt(
diagnostic.pid,
);
this.#publish();
}
}
@ -3301,6 +3394,70 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
}
}
#applyProviderIdentityEvent(event: DurableRecoveryCommittedEvent): void {
const started = record(record(event.envelope.payload).payload);
const runtimeIdentity = record(started.runtimeIdentity);
const descriptor = record(started.providerDescriptor);
const {
processId: pid,
threadId,
sessionId,
} = resolveRunnerdSessionIdentity(started);
const providerIdentity = record(started.providerIdentity);
if (pid !== null) {
this.#evidence.providerPid = pid;
this.#evidence.providerProcessStartedAt = readLocalProcessStartedAt(pid);
if (descriptor.driver === "acpx_runtime") {
this.#evidence.sidecarPid = pid;
this.#evidence.sidecarProcessStartedAt =
this.#evidence.providerProcessStartedAt;
} else {
this.#evidence.codexPid = pid;
this.#evidence.codexProcessStartedAt =
this.#evidence.providerProcessStartedAt;
}
}
if (typeof descriptor.driver === "string")
this.#evidence.providerDriver = descriptor.driver;
if (typeof descriptor.providerVersion === "string")
this.#evidence.providerVersion = descriptor.providerVersion;
if (
descriptor.agent === "pi" ||
descriptor.agent === "claude" ||
descriptor.agent === "codex"
)
this.#evidence.acpxAgent = descriptor.agent;
if (typeof descriptor.agentServerVersion === "string")
this.#evidence.agentServerVersion = descriptor.agentServerVersion;
if (typeof descriptor.agentRuntimeVersion === "string")
this.#evidence.agentRuntimeVersion = descriptor.agentRuntimeVersion;
if (typeof descriptor.acpProtocolVersion === "number")
this.#evidence.acpProtocolVersion = descriptor.acpProtocolVersion;
if (typeof descriptor.agentProcessId === "number") {
this.#evidence.agentPid = descriptor.agentProcessId;
this.#evidence.agentProcessStartedAt = readLocalProcessStartedAt(
descriptor.agentProcessId,
);
}
if (
runtimeIdentity.executionKind === "local_process" ||
runtimeIdentity.executionKind === "remote_service"
) {
this.#evidence.providerExecutionKind = runtimeIdentity.executionKind;
}
if (runtimeIdentity.service === "anthropic_managed_agents") {
this.#evidence.providerService = "anthropic_managed_agents";
} else if (runtimeIdentity.service === "aws_bedrock_agentcore_harness") {
this.#evidence.providerService = "aws_bedrock_agentcore_harness";
}
if (threadId !== null) this.#threadId = threadId;
if (sessionId !== null) this.#sessionId = sessionId;
if (typeof providerIdentity.kind === "string") {
this.#providerIdentity = structuredClone(providerIdentity);
}
this.#publish();
}
#flushPendingTraceRehydrations(): void {
const tracePath = this.options.environment?.PAPERCLIP_PROVIDER_TRACE_PATH;
if (!tracePath) return;
@ -3343,6 +3500,80 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
}
}
async #runnerHasExited(): Promise<boolean> {
if (this.#handle) return this.#handle.child.exitCode !== null;
const adoptedRunner = this.options.adoptExistingRunner;
if (!adoptedRunner) return true;
try {
return !(await adoptedRunner.isAlive());
} catch {
return true;
}
}
#watchAdoptedRunner(
adoptedRunner: NonNullable<
CapabilityRunnerdCodexTransportOptions["adoptExistingRunner"]
>,
): void {
let checking = false;
this.#adoptedRunnerMonitor = setInterval(() => {
if (checking || this.#closed) return;
checking = true;
void Promise.resolve(adoptedRunner.isAlive())
.then((alive) => {
if (alive || this.#closed) return;
this.#evidence.runnerExited = true;
this.#publish();
this.#failTransport(
new Error(
"native_adopted_runner_exited: the verified runner exited while its durable authority was active",
),
);
})
.catch(() => {
if (!this.#closed) {
this.#failTransport(
new Error(
"native_adopted_runner_identity_unverifiable: runner liveness could not be revalidated",
),
);
}
})
.finally(() => {
checking = false;
});
}, 250);
this.#adoptedRunnerMonitor.unref?.();
}
async #awaitAdoptedRunnerConnection(
adoptedRunner: NonNullable<
CapabilityRunnerdCodexTransportOptions["adoptExistingRunner"]
>,
): Promise<void> {
const core = this.#core;
if (!core) throw new Error("native_runner_authority_unavailable");
this.#diagnostic(
`waiting for adopted runner ${adoptedRunner.pid} to authenticate to its durable PRP authority`,
);
while (core.activeRunnerConnectionCount() !== 1) {
this.#throwIfFailed();
if (!(await adoptedRunner.isAlive())) {
throw new Error(
"native_adopted_runner_exited: runner exited before PRP authentication",
);
}
await Promise.race([
new Promise<void>((resolveWait) => setTimeout(resolveWait, 25)),
this.#failureSignal,
]);
}
this.#diagnostic(
`adopted runner ${adoptedRunner.pid} authenticated to its durable PRP authority`,
);
}
#watchRunner(handle: RunnerProcessHandle): void {
void handle.completion.then(
(result) => {
@ -3465,6 +3696,8 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
this.#evidence.runnerExitCode = null;
this.#evidence.runnerSignal = null;
this.#evidence.runnerPid = recoveredHandle.child.pid ?? null;
this.#evidence.runnerProcessGroupId =
recoveredHandle.processGroupId ?? null;
this.#publish();
let processSettled = false;
@ -3566,7 +3799,11 @@ export function createCapabilityRunnerdCodexTransport(
options: CapabilityRunnerdCodexTransportOptions = {},
): CapabilityRunnerdCodexTransport {
const transport = new DurablePrpCodexTransport(options);
return { transport, evidence: () => transport.evidence() };
return {
transport,
evidence: () => transport.evidence(),
detachControllerForRestart: () => transport.detachControllerForRestart(),
};
}
export const createRunnerdCodexTransport =

View File

@ -14,6 +14,10 @@ import { applyDevRunnerOptions } from "./dev-runner-options.ts";
import { collectWatchedSnapshot as collectDevServerWatchedSnapshot, diffSnapshots } from "./dev-runner-snapshot.mjs";
import { createDevServiceIdentity, repoRoot } from "./dev-service-profile.ts";
import { bootstrapDevRunnerWorktreeEnv, isWorktreeSeedPending } from "../server/src/dev-runner-worktree.ts";
import {
readDevServerRestartRequest,
removeDevServerRestartRequest,
} from "../server/src/dev-server-status.ts";
import {
findAdoptableLocalService,
removeLocalServiceRegistryRecord,
@ -331,10 +335,9 @@ function clearDevServerStatus() {
rmSync(devServerRestartRequestFilePath, { force: true });
}
function consumeDevServerRestartRequest() {
if (mode !== "dev" || !existsSync(devServerRestartRequestFilePath)) return false;
rmSync(devServerRestartRequestFilePath, { force: true });
return true;
function getDevServerRestartRequest() {
if (mode !== "dev" || !existsSync(devServerRestartRequestFilePath)) return null;
return readDevServerRestartRequest(env);
}
async function updateDevServiceRecord(extra?: Record<string, unknown>) {
@ -747,8 +750,8 @@ async function startServerChild() {
async function maybeAutoRestartChild() {
if (mode !== "dev" || restartInFlight || !child) return;
const manualRestartRequested = consumeDevServerRestartRequest();
if (!manualRestartRequested && dirtyPaths.size === 0 && pendingMigrations.length === 0) return;
const manualRestartRequest = getDevServerRestartRequest();
if (!manualRestartRequest && dirtyPaths.size === 0 && pendingMigrations.length === 0) return;
restartInFlight = true;
let health: { devServer?: { enabled?: boolean; autoRestartEnabled?: boolean; activeRunCount?: number } } | null = null;
@ -764,11 +767,30 @@ async function maybeAutoRestartChild() {
restartInFlight = false;
return;
}
if (!manualRestartRequested && devServer.autoRestartEnabled !== true) {
const observedServerIdentity =
typeof (health as { serverInfo?: { processStartedAt?: unknown } })
.serverInfo?.processStartedAt === "string"
? (health as { serverInfo: { processStartedAt: string } }).serverInfo
.processStartedAt
: null;
if (
manualRestartRequest?.previousServerIdentity &&
observedServerIdentity !== manualRestartRequest.previousServerIdentity
) {
removeDevServerRestartRequest(
manualRestartRequest.requestId
? { requestId: manualRestartRequest.requestId }
: undefined,
env,
);
restartInFlight = false;
return;
}
if (!manualRestartRequested && (devServer.activeRunCount ?? 0) > 0) {
if (!manualRestartRequest && devServer.autoRestartEnabled !== true) {
restartInFlight = false;
return;
}
if (!manualRestartRequest && (devServer.activeRunCount ?? 0) > 0) {
restartInFlight = false;
return;
}
@ -780,7 +802,26 @@ async function maybeAutoRestartChild() {
exitOnDecline: false,
});
await stopChildForRestart();
const restartRequestConsumed = manualRestartRequest
? removeDevServerRestartRequest(
manualRestartRequest.requestId
? { requestId: manualRestartRequest.requestId }
: undefined,
env,
)
: true;
await startServerChild();
if (manualRestartRequest && !restartRequestConsumed) {
// A live writer may briefly hold the request lock. Starting the child is
// still correct because the requested restart already happened; retry
// correlated cleanup afterward without terminating the supervisor.
removeDevServerRestartRequest(
manualRestartRequest.requestId
? { requestId: manualRestartRequest.requestId }
: undefined,
env,
);
}
} catch (error) {
const err = toError(error, "Auto-restart failed");
process.stderr.write(`${err.stack ?? err.message}\n`);

View File

@ -1,10 +1,19 @@
import { existsSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
import {
existsSync,
mkdirSync,
mkdtempSync,
readFileSync,
rmSync,
writeFileSync,
} from "node:fs";
import os from "node:os";
import path from "node:path";
import { afterEach, describe, expect, it } from "vitest";
import {
getDevServerRestartRequestFilePath,
readDevServerRestartRequest,
readPersistedDevServerStatus,
removeDevServerRestartRequest,
toDevServerHealthStatus,
writeDevServerRestartRequest,
} from "../dev-server-status.js";
@ -36,7 +45,11 @@ describe("dev server status helpers", () => {
lastRestartAt: "2026-03-20T11:30:00.000Z",
});
expect(readPersistedDevServerStatus({ PAPERCLIP_DEV_SERVER_STATUS_FILE: filePath })).toEqual({
expect(
readPersistedDevServerStatus({
PAPERCLIP_DEV_SERVER_STATUS_FILE: filePath,
}),
).toEqual({
dirty: true,
lastChangedAt: "2026-03-20T12:00:00.000Z",
changedPathCount: 4,
@ -76,7 +89,11 @@ describe("dev server status helpers", () => {
pendingMigrations: [],
});
expect(readPersistedDevServerStatus({ PAPERCLIP_DEV_SERVER_STATUS_FILE: filePath })).toBeNull();
expect(
readPersistedDevServerStatus({
PAPERCLIP_DEV_SERVER_STATUS_FILE: filePath,
}),
).toBeNull();
});
it("writes restart requests next to the persisted status file", () => {
@ -87,17 +104,106 @@ describe("dev server status helpers", () => {
});
const env = { PAPERCLIP_DEV_SERVER_STATUS_FILE: filePath };
expect(writeDevServerRestartRequest({
requestedAt: "2026-03-20T12:05:00.000Z",
reason: "manual_restart_now",
}, env)).toBe(true);
expect(
writeDevServerRestartRequest(
{
requestedAt: "2026-03-20T12:05:00.000Z",
reason: "manual_restart_now",
},
env,
),
).toBe(true);
const requestPath = getDevServerRestartRequestFilePath(env);
expect(requestPath).toBe(path.join(path.dirname(filePath), "dev-server-restart-request.json"));
expect(requestPath).toBe(
path.join(path.dirname(filePath), "dev-server-restart-request.json"),
);
expect(requestPath && existsSync(requestPath)).toBe(true);
expect(JSON.parse(readFileSync(requestPath!, "utf8"))).toEqual({
requestedAt: "2026-03-20T12:05:00.000Z",
reason: "manual_restart_now",
});
});
it("correlates restart request cleanup so stale consumers cannot remove a replacement", () => {
const filePath = createTempStatusFile({ dirty: true });
const env = { PAPERCLIP_DEV_SERVER_STATUS_FILE: filePath };
expect(
writeDevServerRestartRequest(
{
requestedAt: "2026-09-04T12:00:00.000Z",
reason: "manual_restart_now",
requestId: "restart-new",
mode: "hot",
previousServerIdentity: "server-start-new",
},
env,
),
).toBe(true);
removeDevServerRestartRequest({ requestId: "restart-stale" }, env);
expect(readDevServerRestartRequest(env)).toEqual({
requestedAt: "2026-09-04T12:00:00.000Z",
reason: "manual_restart_now",
requestId: "restart-new",
mode: "hot",
previousServerIdentity: "server-start-new",
});
removeDevServerRestartRequest({ requestId: "restart-new" }, env);
expect(readDevServerRestartRequest(env)).toBeNull();
});
it("immediately recovers an owner-less lock from an interrupted publisher", () => {
const filePath = createTempStatusFile({ dirty: true });
const env = { PAPERCLIP_DEV_SERVER_STATUS_FILE: filePath };
const requestPath = getDevServerRestartRequestFilePath(env)!;
const lockPath = `${requestPath}.lock`;
mkdirSync(lockPath);
writeDevServerRestartRequest(
{
requestedAt: "2026-09-04T12:00:01.000Z",
reason: "manual_restart_now",
requestId: "restart-after-crash",
mode: "hot",
},
env,
);
expect(readDevServerRestartRequest(env)).toMatchObject({
requestId: "restart-after-crash",
requestedAt: "2026-09-04T12:00:01.000Z",
});
expect(existsSync(lockPath)).toBe(false);
});
it("preserves the request instead of throwing when a live writer holds the lock", () => {
const filePath = createTempStatusFile({ dirty: true });
const env = { PAPERCLIP_DEV_SERVER_STATUS_FILE: filePath };
writeDevServerRestartRequest(
{
requestedAt: "2026-09-04T12:00:02.000Z",
reason: "manual_restart_now",
requestId: "restart-contended",
mode: "hot",
},
env,
);
const requestPath = getDevServerRestartRequestFilePath(env)!;
const lockPath = `${requestPath}.lock`;
mkdirSync(lockPath);
writeFileSync(
path.join(lockPath, "owner.json"),
JSON.stringify({ pid: process.pid, acquiredAt: new Date().toISOString() }),
"utf8",
);
expect(
removeDevServerRestartRequest({ requestId: "restart-contended" }, env),
).toBe(false);
expect(readDevServerRestartRequest(env)).toMatchObject({
requestId: "restart-contended",
});
});
});

View File

@ -6,6 +6,8 @@ import request from "supertest";
import { afterEach, describe, expect, it, vi } from "vitest";
import type { Db } from "@paperclipai/db";
import { healthRoutes } from "../routes/health.js";
import * as devServerStatus from "../dev-server-status.js";
import { resolveHotRestartIntentPath } from "../services/hot-restart.js";
const tempDirs: string[] = [];
@ -18,6 +20,7 @@ function createDevServerStatusFile(payload: unknown) {
}
afterEach(() => {
vi.restoreAllMocks();
for (const dir of tempDirs.splice(0)) {
rmSync(dir, { recursive: true, force: true });
}
@ -138,6 +141,7 @@ describe("GET /health dev-server supervisor access", () => {
describe("POST /health/dev-server/restart", () => {
it("records a manual restart request for the dev runner", async () => {
const previousFile = process.env.PAPERCLIP_DEV_SERVER_STATUS_FILE;
const previousHome = process.env.PAPERCLIP_HOME;
process.env.PAPERCLIP_DEV_SERVER_STATUS_FILE = createDevServerStatusFile({
dirty: true,
lastChangedAt: "2026-03-20T12:00:00.000Z",
@ -146,15 +150,29 @@ describe("POST /health/dev-server/restart", () => {
pendingMigrations: [],
lastRestartAt: "2026-03-20T11:30:00.000Z",
});
process.env.PAPERCLIP_HOME = path.dirname(
process.env.PAPERCLIP_DEV_SERVER_STATUS_FILE,
);
try {
const app = express();
app.use("/health", healthRoutes(undefined));
const db = {
select: vi.fn(() => ({
from: vi.fn(() => ({
where: vi.fn().mockResolvedValue([]),
})),
})),
} as unknown as Db;
app.use("/health", healthRoutes(db));
const res = await request(app).post("/health/dev-server/restart");
expect(res.status).toBe(202);
expect(res.body).toEqual({ status: "restart_requested" });
expect(res.body).toMatchObject({
status: "restart_requested",
mode: "hot",
requestId: expect.any(String),
});
const requestPath = path.join(
path.dirname(process.env.PAPERCLIP_DEV_SERVER_STATUS_FILE),
@ -163,6 +181,8 @@ describe("POST /health/dev-server/restart", () => {
expect(existsSync(requestPath)).toBe(true);
expect(JSON.parse(readFileSync(requestPath, "utf8"))).toMatchObject({
reason: "manual_restart_now",
mode: "hot",
requestId: res.body.requestId,
});
} finally {
if (previousFile === undefined) {
@ -170,6 +190,47 @@ describe("POST /health/dev-server/restart", () => {
} else {
process.env.PAPERCLIP_DEV_SERVER_STATUS_FILE = previousFile;
}
if (previousHome === undefined) delete process.env.PAPERCLIP_HOME;
else process.env.PAPERCLIP_HOME = previousHome;
}
});
it("rolls back the hot intent when the supervisor request cannot be written", async () => {
const previousFile = process.env.PAPERCLIP_DEV_SERVER_STATUS_FILE;
const previousHome = process.env.PAPERCLIP_HOME;
process.env.PAPERCLIP_DEV_SERVER_STATUS_FILE = createDevServerStatusFile({
dirty: true,
changedPathCount: 1,
changedPathsSample: ["server/src/routes/health.ts"],
pendingMigrations: [],
});
const home = mkdtempSync(path.join(os.tmpdir(), "paperclip-health-restart-home-"));
tempDirs.push(home);
process.env.PAPERCLIP_HOME = home;
vi.spyOn(devServerStatus, "writeDevServerRestartRequest").mockReturnValue(false);
try {
const app = express();
const db = {
select: vi.fn(() => ({
from: vi.fn(() => ({ where: vi.fn().mockResolvedValue([]) })),
})),
} as unknown as Db;
app.use("/health", healthRoutes(db));
const res = await request(app).post("/health/dev-server/restart");
expect(res.status).toBe(404);
expect(res.body).toEqual({ error: "dev_server_supervisor_unavailable" });
expect(existsSync(resolveHotRestartIntentPath(home))).toBe(false);
} finally {
if (previousFile === undefined) {
delete process.env.PAPERCLIP_DEV_SERVER_STATUS_FILE;
} else {
process.env.PAPERCLIP_DEV_SERVER_STATUS_FILE = previousFile;
}
if (previousHome === undefined) delete process.env.PAPERCLIP_HOME;
else process.env.PAPERCLIP_HOME = previousHome;
}
});

View File

@ -46,6 +46,7 @@ import {
issueTreeHolds,
issueWorkProducts,
issues,
nativeRunFinalizations,
plugins,
projects,
projectWorkspaces,
@ -118,6 +119,7 @@ import {
} from "../services/heartbeat.ts";
import {
readHotRestartIntent,
readProcessStartedAt,
resolveLegacyHotRestartIntentPath,
resolveHotRestartReportPath,
writeHotRestartIntent,
@ -433,6 +435,7 @@ describeEmbeddedPostgres("heartbeat orphaned process recovery", () => {
await db.delete(issueRecoveryActions);
await db.delete(issueTreeHoldMembers);
await db.delete(issueTreeHolds);
await db.delete(nativeRunFinalizations);
for (let attempt = 0; attempt < 5; attempt += 1) {
await db.delete(issueComments);
await db.delete(issueDocuments);
@ -1442,6 +1445,53 @@ describeEmbeddedPostgres("heartbeat orphaned process recovery", () => {
expect(wakeup?.status).toBe("claimed");
});
it("does not reap a retryable native run while its same-run recovery path owns it", async () => {
const { companyId, agentId, runId, issueId, wakeupRequestId } =
await seedRunFixture({
adapterType: "paperclip_runner",
runtimeMode: "native",
});
const nextAttemptAt = new Date(Date.now() + 60_000);
await db
.update(heartbeatRuns)
.set({ nativeIssueId: issueId, nativePhase: "retryable_failure" })
.where(eq(heartbeatRuns.id, runId));
await db.insert(nativeRunFinalizations).values({
runId,
companyId,
issueId,
phase: "retryable_failure",
attempt: 1,
nextAttemptAt,
recoveryState: "resuming_session",
});
const result = await heartbeatService(db).reapOrphanedRuns();
expect(result).toEqual({ reaped: 0, runIds: [] });
expect(await heartbeatService(db).getRun(runId)).toMatchObject({
status: "running",
nativePhase: "retryable_failure",
});
await expect(
db
.select({ id: heartbeatRuns.id })
.from(heartbeatRuns)
.where(
and(
eq(heartbeatRuns.agentId, agentId),
eq(heartbeatRuns.retryOfRunId, runId),
),
),
).resolves.toHaveLength(0);
await expect(
db
.select({ status: agentWakeupRequests.status })
.from(agentWakeupRequests)
.where(eq(agentWakeupRequests.id, wakeupRequestId)),
).resolves.toEqual([{ status: "claimed" }]);
});
it("does not grant a dead native run legacy retry authority after adapter reassignment", async () => {
const { agentId, runId } = await seedRunFixture({
adapterType: "paperclip_runner",
@ -2268,12 +2318,14 @@ describeEmbeddedPostgres("heartbeat orphaned process recovery", () => {
return row?.status === "running" && row.processPid ? row : null;
}),
);
const observedProcessStartedAt = await readProcessStartedAt(spawnedPid!);
expect(observedProcessStartedAt).not.toBeNull();
expect(running).toMatchObject({
id: runId,
status: "running",
processPid: spawnedPid,
processGroupId: null,
processStartedAt: new Date("2026-07-30T07:00:00.000Z"),
processStartedAt: new Date(observedProcessStartedAt!),
});
await withTempPaperclipHome(async (home) => {
@ -2524,6 +2576,64 @@ describeEmbeddedPostgres("heartbeat orphaned process recovery", () => {
expect(issue?.executionRunId).toBe(retryRun?.id);
});
it("suspends native Paperclip Runner ownership on graceful restart without cancelling or creating a retry run", async () => {
const { agentId, runId, issueId, wakeupRequestId } =
await seedRunFixture({
adapterType: "paperclip_runner",
agentStatus: "running",
runtimeMode: "native",
});
await db
.update(heartbeatRuns)
.set({ nativeIssueId: issueId })
.where(eq(heartbeatRuns.id, runId));
await db.insert(nativeRunFinalizations).values({
runId,
companyId: (await heartbeatService(db).getRun(runId))!.companyId,
issueId,
phase: "observed",
});
const result = await heartbeatService(db).drainRunningRunsForShutdown(
"SIGTERM",
new Date("2026-09-04T12:00:00.000Z"),
);
expect(result).toMatchObject({
interrupted: 0,
interruptedRunIds: [],
retryRunIds: [],
restartSuspendedRunIds: [runId],
});
await expect(
db.select().from(heartbeatRuns).where(eq(heartbeatRuns.agentId, agentId)),
).resolves.toEqual([
expect.objectContaining({
id: runId,
status: "running",
retryOfRunId: null,
}),
]);
await expect(
db
.select({ recoveryState: nativeRunFinalizations.recoveryState })
.from(nativeRunFinalizations)
.where(eq(nativeRunFinalizations.runId, runId)),
).resolves.toEqual([{ recoveryState: "awaiting_runner_reattach" }]);
await expect(
db
.select({ status: agentWakeupRequests.status })
.from(agentWakeupRequests)
.where(eq(agentWakeupRequests.id, wakeupRequestId)),
).resolves.toEqual([{ status: "claimed" }]);
await expect(
db
.select({ executionRunId: issues.executionRunId })
.from(issues)
.where(eq(issues.id, issueId)),
).resolves.toEqual([{ executionRunId: runId }]);
});
it("does not overwrite a run that is no longer running during graceful shutdown drain", async () => {
const { runId, wakeupRequestId } = await seedRunFixture({
agentStatus: "running",

View File

@ -48,6 +48,13 @@ const {
}));
const heartbeatServiceMock = {
resolveSchedulingSuppression: resolveHeartbeatSchedulingSuppressionMock,
recoverNativeRunsAfterRestart: vi.fn(async () => ({
restartKind: "hard",
dispositions: [],
claims: [],
awaitingEvidenceRunIds: [],
blockedRunIds: [],
})),
reconcileHotRestartAdoption: vi.fn(async () => ({ mode: "none" })),
reapOrphanedRuns: vi.fn(async () => ({ reaped: 0, runIds: [] })),
promoteDueScheduledRetries: vi.fn(async () => ({ promoted: 0, runIds: [] })),
@ -118,7 +125,7 @@ const {
callback?.();
return fakeServer;
}),
close: vi.fn(),
close: vi.fn((callback?: (error?: Error) => void) => callback?.()),
};
const loadConfigMock = vi.fn();
@ -574,6 +581,21 @@ describe("startServer feedback export wiring", () => {
expect(heartbeatServiceMock.reapOrphanedRuns).toHaveBeenCalledTimes(2);
});
it("closes the bound listener when native startup recovery fails", async () => {
loadConfigMock.mockReturnValue(buildTestConfig({
heartbeatSchedulerEnabled: true,
heartbeatSchedulerIntervalMs: 30000,
}));
heartbeatServiceMock.recoverNativeRunsAfterRestart.mockRejectedValueOnce(
new Error("native recovery unavailable"),
);
await expect(startServer()).rejects.toThrow("native recovery unavailable");
expect(fakeServer.listen).toHaveBeenCalledTimes(1);
expect(fakeServer.close).toHaveBeenCalledTimes(1);
});
it("refuses authenticated public startup without an external database URL", async () => {
loadConfigMock.mockReturnValue(buildTestConfig({
deploymentExposure: "public",

View File

@ -1,7 +1,21 @@
import { existsSync, mkdirSync, readFileSync, statSync, writeFileSync } from "node:fs";
import { randomUUID } from "node:crypto";
import {
existsSync,
mkdirSync,
readFileSync,
renameSync,
rmSync,
statSync,
unlinkSync,
writeFileSync,
} from "node:fs";
import path from "node:path";
const MAX_PERSISTED_DEV_SERVER_STATUS_BYTES = 64 * 1024;
const DEV_RESTART_REQUEST_LOCK_STALE_MS = 30_000;
const DEV_RESTART_REQUEST_LOCK_RETRY_COUNT = 50;
const DEV_RESTART_REQUEST_LOCK_RETRY_MS = 2;
const lockWaitBuffer = new Int32Array(new SharedArrayBuffer(4));
export type PersistedDevServerStatus = {
dirty: boolean;
@ -15,7 +29,11 @@ export type PersistedDevServerStatus = {
export type DevServerHealthStatus = {
enabled: true;
restartRequired: boolean;
reason: "backend_changes" | "pending_migrations" | "backend_changes_and_pending_migrations" | null;
reason:
| "backend_changes"
| "pending_migrations"
| "backend_changes_and_pending_migrations"
| null;
lastChangedAt: string | null;
changedPathCount: number;
changedPathsSample: string[];
@ -29,14 +47,110 @@ export type DevServerHealthStatus = {
export type DevServerRestartRequest = {
requestedAt: string;
reason: "manual_restart_now";
requestId?: string;
mode?: "hot";
previousServerIdentity?: string;
};
function processIsAlive(pid: number): boolean {
try {
process.kill(pid, 0);
return true;
} catch (error) {
return (error as NodeJS.ErrnoException | undefined)?.code === "EPERM";
}
}
function tryRecoverDevRestartRequestLock(lockPath: string): boolean {
let stale = false;
try {
const lockAgeMs = Date.now() - statSync(lockPath).mtimeMs;
const owner = JSON.parse(
readFileSync(path.join(lockPath, "owner.json"), "utf8"),
) as Record<string, unknown>;
stale =
lockAgeMs >= DEV_RESTART_REQUEST_LOCK_STALE_MS ||
(typeof owner.pid === "number" && !processIsAlive(owner.pid));
} catch {
// Canonical locks are published only after owner.json is durable in a
// private candidate directory. A visible lock without valid ownership is
// therefore abandoned and can be reclaimed immediately.
stale = true;
}
if (!stale) return false;
const stalePath = `${lockPath}.${process.pid}.${randomUUID()}.stale`;
try {
renameSync(lockPath, stalePath);
} catch (error) {
if ((error as NodeJS.ErrnoException | undefined)?.code === "ENOENT") {
return true;
}
return false;
}
rmSync(stalePath, { recursive: true, force: true });
return true;
}
function withDevRestartRequestLock<T>(filePath: string, action: () => T): T {
const lockPath = `${filePath}.lock`;
for (
let attempt = 0;
attempt <= DEV_RESTART_REQUEST_LOCK_RETRY_COUNT;
attempt += 1
) {
const candidateLockPath = `${lockPath}.${process.pid}.${randomUUID()}.candidate`;
mkdirSync(candidateLockPath);
try {
writeFileSync(
path.join(candidateLockPath, "owner.json"),
`${JSON.stringify({ pid: process.pid, acquiredAt: new Date().toISOString() })}\n`,
"utf8",
);
try {
// Publishing a populated directory makes lock ownership visible in one
// rename; there is no canonical owner-less crash window.
renameSync(candidateLockPath, lockPath);
} catch (error) {
const code = (error as NodeJS.ErrnoException | undefined)?.code;
if (code !== "EEXIST" && code !== "ENOTEMPTY" && code !== "EPERM") {
throw error;
}
if (tryRecoverDevRestartRequestLock(lockPath)) continue;
if (attempt === DEV_RESTART_REQUEST_LOCK_RETRY_COUNT) {
throw new Error("dev_server_restart_request_lock_busy");
}
Atomics.wait(
lockWaitBuffer,
0,
0,
DEV_RESTART_REQUEST_LOCK_RETRY_MS,
);
continue;
}
} finally {
if (existsSync(candidateLockPath)) {
rmSync(candidateLockPath, { recursive: true, force: true });
}
}
try {
return action();
} finally {
rmSync(lockPath, { recursive: true, force: true });
}
}
throw new Error("dev_server_restart_request_lock_busy");
}
export function getDevServerRestartRequestFilePath(
env: NodeJS.ProcessEnv = process.env,
): string | null {
const statusFilePath = env.PAPERCLIP_DEV_SERVER_STATUS_FILE?.trim();
if (!statusFilePath) return null;
return path.join(path.dirname(statusFilePath), "dev-server-restart-request.json");
return path.join(
path.dirname(statusFilePath),
"dev-server-restart-request.json",
);
}
export function writeDevServerRestartRequest(
@ -47,10 +161,88 @@ export function writeDevServerRestartRequest(
if (!filePath) return false;
mkdirSync(path.dirname(filePath), { recursive: true });
writeFileSync(filePath, `${JSON.stringify(request, null, 2)}\n`, "utf8");
withDevRestartRequestLock(filePath, () => {
const tempPath = `${filePath}.${process.pid}.${randomUUID()}.tmp`;
try {
writeFileSync(tempPath, `${JSON.stringify(request, null, 2)}\n`, "utf8");
renameSync(tempPath, filePath);
} finally {
try {
unlinkSync(tempPath);
} catch (error) {
if ((error as NodeJS.ErrnoException | undefined)?.code !== "ENOENT") {
throw error;
}
}
}
});
return true;
}
function readDevServerRestartRequestAtPath(
filePath: string,
): DevServerRestartRequest | null {
try {
if (statSync(filePath).size > MAX_PERSISTED_DEV_SERVER_STATUS_BYTES)
return null;
const value = JSON.parse(readFileSync(filePath, "utf8")) as Record<
string,
unknown
>;
if (
typeof value.requestedAt !== "string" ||
value.reason !== "manual_restart_now"
) {
return null;
}
return {
requestedAt: value.requestedAt,
reason: "manual_restart_now",
...(typeof value.requestId === "string"
? { requestId: value.requestId }
: {}),
...(value.mode === "hot" ? { mode: "hot" as const } : {}),
...(typeof value.previousServerIdentity === "string"
? { previousServerIdentity: value.previousServerIdentity }
: {}),
};
} catch {
return null;
}
}
export function readDevServerRestartRequest(
env: NodeJS.ProcessEnv = process.env,
): DevServerRestartRequest | null {
const filePath = getDevServerRestartRequestFilePath(env);
if (!filePath || !existsSync(filePath)) return null;
return readDevServerRestartRequestAtPath(filePath);
}
export function removeDevServerRestartRequest(
expected?: Pick<DevServerRestartRequest, "requestId">,
env: NodeJS.ProcessEnv = process.env,
): boolean {
const filePath = getDevServerRestartRequestFilePath(env);
if (!filePath) return false;
try {
withDevRestartRequestLock(filePath, () => {
const current = readDevServerRestartRequestAtPath(filePath);
if (expected?.requestId && current?.requestId !== expected.requestId) return;
rmSync(filePath, { force: true });
});
return true;
} catch (error) {
if (
error instanceof Error &&
error.message === "dev_server_restart_request_lock_busy"
) {
return false;
}
throw error;
}
}
function normalizeStringArray(value: unknown): string[] {
if (!Array.isArray(value)) return [];
return value
@ -75,12 +267,18 @@ export function readPersistedDevServerStatus(
if (statSync(filePath).size > MAX_PERSISTED_DEV_SERVER_STATUS_BYTES) {
return null;
}
const raw = JSON.parse(readFileSync(filePath, "utf8")) as Record<string, unknown>;
const changedPathsSample = normalizeStringArray(raw.changedPathsSample).slice(0, 5);
const raw = JSON.parse(readFileSync(filePath, "utf8")) as Record<
string,
unknown
>;
const changedPathsSample = normalizeStringArray(
raw.changedPathsSample,
).slice(0, 5);
const pendingMigrations = normalizeStringArray(raw.pendingMigrations);
const changedPathCountRaw = raw.changedPathCount;
const changedPathCount =
typeof changedPathCountRaw === "number" && Number.isFinite(changedPathCountRaw)
typeof changedPathCountRaw === "number" &&
Number.isFinite(changedPathCountRaw)
? Math.max(0, Math.trunc(changedPathCountRaw))
: changedPathsSample.length;
const dirtyRaw = raw.dirty;
@ -128,7 +326,8 @@ export function toDevServerHealthStatus(
pendingMigrations: persisted.pendingMigrations,
autoRestartEnabled: opts.autoRestartEnabled,
activeRunCount: opts.activeRunCount,
waitingForIdle: restartRequired && opts.autoRestartEnabled && opts.activeRunCount > 0,
waitingForIdle:
restartRequired && opts.autoRestartEnabled && opts.activeRunCount > 0,
lastRestartAt: persisted.lastRestartAt,
};
}

View File

@ -35,6 +35,7 @@ import detectPort from "detect-port";
import { createApp } from "./app.js";
import { loadConfig } from "./config.js";
import { logger } from "./middleware/logger.js";
import { setStartupRecoveryPhase } from "./startup-recovery-state.js";
import {
StartupRefusalError,
migrationRefusalError,
@ -103,6 +104,7 @@ import { ensureDecisionSigningSecret } from "./services/decision-signing.js";
import { createDecisionRetentionNotifyOriginAgent, createDecisionWakeOriginAgent } from "./services/decision-wakeup.js";
import {
coordinateHeartbeatSchedulerShutdown,
drainRunExecutionFinalizersForShutdown,
finalizeServerShutdown,
loadWithoutCoordinatedShutdownSignalHooks,
} from "./shutdown.js";
@ -155,6 +157,7 @@ export interface StartedServer {
}
export async function startServer(): Promise<StartedServer> {
setStartupRecoveryPhase("starting");
warnIfUnsupportedNodeVersion(process.versions.node, (message) => logger.warn(message));
// Tracing must be active (or have failed and logged) before the first DB
@ -887,7 +890,9 @@ export async function startServer(): Promise<StartedServer> {
process.env.PAPERCLIP_RUNTIME_API_URL = runtimeApiUrl;
process.env.PAPERCLIP_RUNTIME_API_CANDIDATES_JSON = JSON.stringify(runtimeApiCandidates);
process.env.PAPERCLIP_API_URL = configuredApiUrl;
let startupListenerBound = false;
try {
setupRunnerPrpWebSocketServer(server, { apiUrl: configuredApiUrl });
setupEnvironmentCustomImageTerminalWebSocketServer(server, db as any, {
pluginWorkerManager,
@ -911,6 +916,28 @@ export async function startServer(): Promise<StartedServer> {
},
});
setStartupRecoveryPhase("recovering");
// Bind the shared HTTP/PRP listener before native startup recovery. A
// runnerd process that survived a controller crash is already reconnecting
// to this address; delaying listen until after orphan reconciliation makes
// authenticated adoption impossible and turns a healthy process into a
// duplicate-provider risk.
await new Promise<void>((resolveListen, rejectListen) => {
const onError = (err: Error) => {
server.off("error", onError);
rejectListen(err);
};
server.once("error", onError);
server.listen(listenPort, config.host, () => {
server.off("error", onError);
logger.info(
`Server listener bound on ${config.host}:${listenPort}; startup recovery in progress`,
);
resolveListen();
});
});
startupListenerBound = true;
try {
const result = await workspaceOperationService(db as any)
.reconcileStaleRuntimeControlOperations();
@ -1042,6 +1069,7 @@ export async function startServer(): Promise<StartedServer> {
signal: "SIGINT" | "SIGTERM",
runIds?: readonly string[] | null,
) => Promise<unknown>) | null = null;
let drainHeartbeatExecutionFinalizers: (() => Promise<void>) | null = null;
let prepareHotRestartShutdown: ((signal: "SIGINT" | "SIGTERM") => Promise<{
skipDrain: boolean;
drainRunIds?: string[];
@ -1146,6 +1174,8 @@ export async function startServer(): Promise<StartedServer> {
drainHeartbeatRunsForShutdown = (signal, runIds) => (
heartbeat.drainRunningRunsForShutdown(signal, new Date(), runIds)
);
drainHeartbeatExecutionFinalizers = () =>
heartbeat.drainActiveRunExecutions();
prepareHotRestartShutdown = heartbeat.prepareHotRestartShutdown;
const environmentCustomImages = environmentCustomImageService(db as any, { pluginWorkerManager });
const routines = routineService(db as any, { pluginWorkerManager });
@ -1290,6 +1320,32 @@ export async function startServer(): Promise<StartedServer> {
);
} else {
const startupHeartbeatRecovery = (async () => {
try {
const nativeRecovery =
await heartbeat.recoverNativeRunsAfterRestart();
if (nativeRecovery.dispositions.length > 0) {
logger.info(
{
restartKind: nativeRecovery.restartKind,
claims: nativeRecovery.claims.map((claim) => ({
runId: claim.runId,
disposition: claim.kind,
controllerGeneration: claim.controllerGeneration,
})),
awaitingEvidenceRunIds:
nativeRecovery.awaitingEvidenceRunIds,
blockedRunIds: nativeRecovery.blockedRunIds,
},
"startup native runner restart recovery classified",
);
}
} catch (err) {
logger.error(
{ err },
"startup native runner restart recovery failed closed",
);
throw err;
}
try {
const hotRestart = await heartbeat.reconcileHotRestartAdoption();
if (hotRestart.mode === "reported") {
@ -1374,6 +1430,7 @@ export async function startServer(): Promise<StartedServer> {
}
})().catch((err) => {
logger.error({ err }, "startup heartbeat recovery failed");
throw err;
});
trackHeartbeatSchedulerWork(startupHeartbeatRecovery);
await startupHeartbeatRecovery;
@ -1667,35 +1724,27 @@ export async function startServer(): Promise<StartedServer> {
throw err;
}
await new Promise<void>((resolveListen, rejectListen) => {
const onError = (err: Error) => {
server.off("error", onError);
rejectListen(err);
};
server.once("error", onError);
server.listen(listenPort, config.host, () => {
server.off("error", onError);
logger.info(`Server listening on ${config.host}:${listenPort}`);
void systemdNotify(["--ready", `--status=Listening on ${config.host}:${listenPort}`]).then((notified) => {
if (notified) logger.info("Notified systemd that Paperclip is ready");
setStartupRecoveryPhase("ready");
logger.info(`Server startup recovery complete on ${config.host}:${listenPort}`);
void systemdNotify(["--ready", `--status=Listening on ${config.host}:${listenPort}`]).then((notified) => {
if (notified) logger.info("Notified systemd that Paperclip is ready");
});
if (process.env.PAPERCLIP_OPEN_ON_LISTEN === "true") {
const openHost = config.host === "0.0.0.0" || config.host === "::" ? "127.0.0.1" : config.host;
const url = `http://${openHost}:${listenPort}`;
void import("open")
.then((mod) => mod.default(url))
.then(() => {
logger.info(`Opened browser at ${url}`);
})
.catch((err) => {
logger.warn({ err, url }, "Failed to open browser on startup");
});
if (process.env.PAPERCLIP_OPEN_ON_LISTEN === "true") {
const openHost = config.host === "0.0.0.0" || config.host === "::" ? "127.0.0.1" : config.host;
const url = `http://${openHost}:${listenPort}`;
void import("open")
.then((mod) => mod.default(url))
.then(() => {
logger.info(`Opened browser at ${url}`);
})
.catch((err) => {
logger.warn({ err, url }, "Failed to open browser on startup");
});
}
printStartupBanner({
bind: config.bind,
host: config.host,
deploymentMode: config.deploymentMode,
}
printStartupBanner({
bind: config.bind,
host: config.host,
deploymentMode: config.deploymentMode,
deploymentExposure: config.deploymentExposure,
authReady,
requestedPort: requestedListenPort,
@ -1709,27 +1758,23 @@ export async function startServer(): Promise<StartedServer> {
databaseBackupIntervalMinutes: config.databaseBackupIntervalMinutes,
databaseBackupRetentionDays: config.databaseBackupRetentionDays,
databaseBackupDir: config.databaseBackupDir,
});
const boardClaimUrl = getBoardClaimWarningUrl(config.host, listenPort);
if (boardClaimUrl) {
const red = "\x1b[41m\x1b[30m";
const yellow = "\x1b[33m";
const reset = "\x1b[0m";
console.log(
[
`${red} BOARD CLAIM REQUIRED ${reset}`,
`${yellow}This instance was previously local_trusted and still has local-board as the only admin.${reset}`,
`${yellow}Sign in with a real user and open this one-time URL to claim ownership:${reset}`,
`${yellow}${boardClaimUrl}${reset}`,
`${yellow}If you are connecting over Tailscale, replace the host in this URL with your Tailscale IP/MagicDNS name.${reset}`,
].join("\n"),
);
}
resolveListen();
});
});
const boardClaimUrl = getBoardClaimWarningUrl(config.host, listenPort);
if (boardClaimUrl) {
const red = "\x1b[41m\x1b[30m";
const yellow = "\x1b[33m";
const reset = "\x1b[0m";
console.log(
[
`${red} BOARD CLAIM REQUIRED ${reset}`,
`${yellow}This instance was previously local_trusted and still has local-board as the only admin.${reset}`,
`${yellow}Sign in with a real user and open this one-time URL to claim ownership:${reset}`,
`${yellow}${boardClaimUrl}${reset}`,
`${yellow}If you are connecting over Tailscale, replace the host in this URL with your Tailscale IP/MagicDNS name.${reset}`,
].join("\n"),
);
}
{
const shutdown = async (signal: "SIGINT" | "SIGTERM") => {
@ -1774,6 +1819,14 @@ export async function startServer(): Promise<StartedServer> {
}
}
if (!skipHeartbeatDrain) {
await drainRunExecutionFinalizersForShutdown({
signal,
drain: drainHeartbeatExecutionFinalizers,
log: logger,
});
}
// Whatever the drain did not finalize (timed-out runs, the hot-restart
// skip path) still has a local-only tail when the in-flight run-log
// mirror is enabled; upload those tails now so an orderly restart
@ -1821,6 +1874,33 @@ export async function startServer(): Promise<StartedServer> {
apiUrl: configuredApiUrl,
databaseUrl: activeDatabaseConnectionString,
};
} catch (error) {
if (startupListenerBound) {
await new Promise<void>((resolveClose) => {
try {
server.close((closeError?: Error) => {
if (
closeError &&
(closeError as NodeJS.ErrnoException).code !== "ERR_SERVER_NOT_RUNNING"
) {
logger.error(
{ err: closeError },
"failed to close HTTP listener after startup failure",
);
}
resolveClose();
});
} catch (closeError) {
logger.error(
{ err: closeError },
"failed to close HTTP listener after startup failure",
);
resolveClose();
}
});
}
throw error;
}
}
function isMainModule(metaUrl: string): boolean {

View File

@ -1,10 +1,15 @@
import { timingSafeEqual } from "node:crypto";
import { randomUUID, timingSafeEqual } from "node:crypto";
import { Router } from "express";
import type { Db } from "@paperclipai/db";
import { and, count, eq, gt, inArray, isNull, sql } from "drizzle-orm";
import { heartbeatRuns, instanceUserRoles, invites } from "@paperclipai/db";
import type { DeploymentExposure, DeploymentMode } from "@paperclipai/shared";
import { readPersistedDevServerStatus, toDevServerHealthStatus, writeDevServerRestartRequest } from "../dev-server-status.js";
import {
readPersistedDevServerStatus,
removeDevServerRestartRequest,
toDevServerHealthStatus,
writeDevServerRestartRequest,
} from "../dev-server-status.js";
import { logger } from "../middleware/logger.js";
import { getServerInfoSnapshot, type ServerInfoSnapshot } from "../server-info.js";
import {
@ -29,6 +34,12 @@ import {
WORKSPACE_READINESS_USER_ID_HEADER,
} from "../auth/workspace-login-handoff.js";
import { serverVersion } from "../version.js";
import { getStartupRecoveryState } from "../startup-recovery-state.js";
import { nativeRestartRecoverySummary } from "../services/native-runtime/native-restart-recovery.js";
import {
removeHotRestartIntent,
writeHotRestartIntent,
} from "../services/hot-restart.js";
function shouldExposeFullHealthDetails(
actorType: "none" | "board" | "agent" | null | undefined,
@ -146,16 +157,75 @@ export function healthRoutes(
return;
}
const written = writeDevServerRestartRequest({
requestedAt: new Date().toISOString(),
reason: "manual_restart_now",
});
if (!written) {
res.status(404).json({ error: "dev_server_supervisor_unavailable" });
if (!db) {
res.status(503).json({ error: "database_unavailable" });
return;
}
res.status(202).json({ status: "restart_requested" });
const requestId = randomUUID();
const requestedAt = new Date();
const serverInfo = opts.serverInfo ?? getServerInfoSnapshot();
const preflightActiveRunIds = await db
.select({ id: heartbeatRuns.id })
.from(heartbeatRuns)
.where(eq(heartbeatRuns.status, "running"))
.then((rows) => rows.map((row) => row.id));
let intent: Awaited<ReturnType<typeof writeHotRestartIntent>> | null = null;
try {
intent = await writeHotRestartIntent({
previousServerPid: process.pid,
previousServerIdentity: serverInfo.processStartedAt,
previousServerVersion: serverVersion,
preflightActiveRunIds,
recoveryRequestId: requestId,
requestedAt,
});
const written = writeDevServerRestartRequest({
requestedAt: requestedAt.toISOString(),
reason: "manual_restart_now",
requestId,
mode: "hot",
previousServerIdentity: serverInfo.processStartedAt,
});
if (!written) {
throw new Error("dev_server_supervisor_unavailable");
}
} catch (error) {
try {
removeDevServerRestartRequest({ requestId });
} catch (rollbackError) {
logger.error(
{ err: rollbackError, requestId },
"failed to roll back dev-server restart request",
);
}
if (intent) {
await removeHotRestartIntent(undefined, intent).catch(
(rollbackError) => {
logger.error(
{ err: rollbackError, requestId },
"failed to roll back hot-restart intent",
);
},
);
}
if (
error instanceof Error &&
error.message === "dev_server_supervisor_unavailable"
) {
res.status(404).json({ error: "dev_server_supervisor_unavailable" });
return;
}
logger.error({ err: error, requestId }, "failed to coordinate hot restart request");
res.status(500).json({ error: "hot_restart_intent_failed" });
return;
}
res.status(202).json({
status: "restart_requested",
requestId,
mode: "hot",
});
});
router.get("/", async (req, res) => {
@ -165,6 +235,9 @@ export function healthRoutes(
opts.deploymentMode,
);
const runtimeEnv = opts.runtimeEnv ?? process.env;
const startupRecovery = getStartupRecoveryState();
const healthStatus =
startupRecovery.phase === "ready" ? "ok" : "starting";
const cloud = getCloudHealthStatus(runtimeEnv);
// Operator-hidden settings ride every response (like `cloud`): the list
// holds UI surface names only, and the settings nav needs it before any
@ -200,7 +273,7 @@ export function healthRoutes(
res.json(
exposeFullDetails
? {
status: "ok",
status: healthStatus,
version: serverVersion,
serverVersion: serverVersion,
commit,
@ -209,7 +282,7 @@ export function healthRoutes(
...(hiddenSettings.length ? { hiddenSettings } : {}),
}
: {
status: "ok",
status: healthStatus,
deploymentMode: opts.deploymentMode,
commit,
...(cloud ? { cloud } : {}),
@ -304,12 +377,18 @@ export function healthRoutes(
? inspectDatabaseBackupHealth(opts.databaseBackupHealth)
: undefined;
const warnings = databaseBackup?.warnings.length ? databaseBackup.warnings : undefined;
const nativeRecovery = exposeFullDetails
? await nativeRestartRecoverySummary(db).catch((error) => {
logger.warn({ err: error }, "native recovery health summary failed");
return {};
})
: undefined;
if (!exposeFullDetails) {
const redactedDatabaseBackup = databaseBackup ? redactedDatabaseBackupHealth(databaseBackup) : undefined;
const redactedWarnings = redactedDatabaseBackup?.warnings.length ? redactedDatabaseBackup.warnings : undefined;
res.json({
status: "ok",
status: healthStatus,
deploymentMode: opts.deploymentMode,
deploymentExposure: opts.deploymentExposure,
commit,
@ -329,7 +408,7 @@ export function healthRoutes(
}
res.json({
status: "ok",
status: healthStatus,
version: serverVersion,
serverVersion,
commit,
@ -342,6 +421,8 @@ export function healthRoutes(
companyDeletionEnabled: opts.companyDeletionEnabled,
},
serverInfo,
startupRecovery,
nativeRecovery,
...(databaseBackup ? { databaseBackup } : {}),
...(warnings ? { warnings } : {}),
...(devServer ? { devServer } : {}),

View File

@ -127,15 +127,18 @@ import {
buildNativeExecutionInput,
buildNativeRuntimeContext,
cancelNativeSession,
claimNativeRestartRecoveries,
dispatchNativeSessionResumptions,
ensureNativeCompletionContract,
executePaperclipNativeSession,
finalizeNativeRun,
isNativeSessionId,
isUnusedLegacyNativeRetryReplacement,
isRunnerIngressAuthorized,
materializeLegacyQuestionResponseWakeProjection,
materializeNativeInteractionResponses,
NativeCancellationPendingRecoveryError,
type NativeRestartRecoveryClaim,
rebindNativeSessionCheckpoint,
reconcileNativeFinalizations,
resolveHeartbeatNativeRuntimeMode,
@ -421,6 +424,7 @@ import {
import {
findMissingHotRestartSnapshotRunIds,
readHotRestartIntent,
readProcessStartedAt,
removeHotRestartIntent,
shouldHonorHotRestartIntentForProcess,
writeHotRestartReport,
@ -7669,7 +7673,10 @@ export async function persistHeartbeatRunProcessMetadata(
runId: string,
meta: { pid: number; processGroupId: number | null; startedAt: string },
) {
const startedAt = new Date(meta.startedAt);
const observedStartedAt = await readProcessStartedAt(meta.pid).catch(
() => null,
);
const startedAt = new Date(observedStartedAt ?? meta.startedAt);
return db
.update(heartbeatRuns)
.set({
@ -8173,6 +8180,7 @@ export function heartbeatService(
db: Db,
options: HeartbeatServiceOptions = {},
) {
let shutdownInProgress = false;
const instanceSettings = instanceSettingsService(db);
const getCurrentUserRedactionOptions = async () => ({
enabled: (await instanceSettings.getGeneral()).censorUsernameInLogs,
@ -12383,6 +12391,10 @@ export function heartbeatService(
processPid: input.run.processPid ?? null,
processGroupId: input.run.processGroupId ?? null,
issueId: readNonEmptyString(context.issueId),
runtimeMode: input.run.runtimeMode,
nativeSessionId: input.run.nativeSessionId,
runnerInstanceId: input.run.runnerInstanceId,
processStartedAt: input.run.processStartedAt?.toISOString() ?? null,
};
}
@ -12420,6 +12432,7 @@ export function heartbeatService(
signal: "SIGINT" | "SIGTERM",
now = new Date(),
) {
shutdownInProgress = true;
let intent: Awaited<ReturnType<typeof readHotRestartIntent>>;
try {
intent = await readHotRestartIntent();
@ -12706,6 +12719,14 @@ export function heartbeatService(
continue;
}
if (
run.runtimeMode === "native" &&
adapterType === "paperclip_runner"
) {
classify(candidate, "skipped", "native_restart_recovery_owned", patch);
continue;
}
if (!isTrackedLocalChildProcessAdapter(adapterType)) {
classify(
candidate,
@ -12844,6 +12865,145 @@ export function heartbeatService(
};
}
async function recoverNativeRunsAfterRestart(now = new Date()) {
// A result committed before the old controller stopped outranks process
// recovery. Finish its durable workspace/status suffix before deciding
// whether any provider authority needs to be reopened.
await reconcileNativeFinalizations(db);
const intent = await readHotRestartIntent().catch((error) => {
logger.warn(
{ err: error },
"failed to read hot-restart intent before native startup recovery",
);
return null;
});
const restartKind = intent ? ("hot" as const) : ("hard" as const);
const previousStartedAt = intent?.previousServerStartedAt
? new Date(intent.previousServerStartedAt)
: null;
const scheduledNativeRetries = await db
.select({
runId: nativeRunFinalizations.runId,
nextAttemptAt: nativeRunFinalizations.nextAttemptAt,
})
.from(nativeRunFinalizations)
.innerJoin(
heartbeatRuns,
eq(heartbeatRuns.id, nativeRunFinalizations.runId),
)
.where(
and(
eq(heartbeatRuns.runtimeMode, "native"),
inArray(heartbeatRuns.status, ["running", "failed"]),
isNull(nativeRunFinalizations.resultId),
eq(nativeRunFinalizations.phase, "retryable_failure"),
gt(nativeRunFinalizations.nextAttemptAt, now),
),
);
for (const scheduled of scheduledNativeRetries) {
if (scheduled.nextAttemptAt) {
scheduleNativeSessionResumeDispatch(
scheduled.runId,
scheduled.nextAttemptAt,
);
}
}
const dispositions = await claimNativeRestartRecoveries({
db,
restartKind,
recoveryRequestId: intent?.recoveryRequestId ?? null,
coordinatedPreviousController: intent
? {
pid: intent.previousServerPid,
processStartedAt:
previousStartedAt && !Number.isNaN(previousStartedAt.getTime())
? previousStartedAt
: null,
}
: null,
now,
});
const claims = dispositions.filter(
(disposition): disposition is NativeRestartRecoveryClaim =>
disposition.kind === "reattach_existing_runner" ||
disposition.kind === "resume_dead_runner" ||
disposition.kind === "bootstrap_incomplete",
);
for (const disposition of dispositions) {
const run = await getRun(disposition.runId);
if (run) {
const isClaim =
disposition.kind === "reattach_existing_runner" ||
disposition.kind === "resume_dead_runner" ||
disposition.kind === "bootstrap_incomplete";
await appendRunEvent(run, {
eventType: "native.recovery.transition",
stream: "system",
level: disposition.kind === "blocked" ? "warn" : "info",
message:
disposition.kind === "reattach_existing_runner"
? "Recovering the existing native runner process after server restart"
: disposition.kind === "resume_dead_runner"
? "Resuming the durable native provider session after runner process loss"
: disposition.kind === "bootstrap_incomplete"
? "Restarting an incomplete native runner bootstrap on the same heartbeat run"
: disposition.kind === "awaiting_evidence"
? "Native restart recovery is waiting for safe ownership evidence"
: disposition.kind === "already_finalized"
? "Native restart recovery found an already-finalized result"
: "Native restart recovery blocked ambiguous or conflicting ownership",
payload: {
restartKind: isClaim ? disposition.restartKind : restartKind,
recoveryRequestId: isClaim
? disposition.recoveryRequestId
: (intent?.recoveryRequestId ?? null),
runnerDisposition: disposition.kind,
...(isClaim
? {
controllerGeneration: disposition.controllerGeneration,
providerAttempt: disposition.providerAttempt,
}
: { reason: disposition.reason }),
...(disposition.kind === "reattach_existing_runner"
? {
processPid: disposition.process.pid,
processGroupId: disposition.process.processGroupId,
processStartedAt: disposition.process.startedAt,
}
: {}),
},
});
}
}
for (const claim of claims) {
const execution = executeRun(claim.runId, {
nativeLeaseOwner: claim.leaseOwner,
nativeRestartRecovery: claim,
}).catch((error) => {
logger.error(
{ err: error, runId: claim.runId, disposition: claim.kind },
"native restart recovery execution failed",
);
});
activeRunExecutionPromises.add(execution);
void execution.finally(() => activeRunExecutionPromises.delete(execution));
}
return {
restartKind,
claims,
dispositions,
scheduledRetryRunIds: scheduledNativeRetries.map((entry) => entry.runId),
awaitingEvidenceRunIds: dispositions
.filter((entry) => entry.kind === "awaiting_evidence")
.map((entry) => entry.runId),
blockedRunIds: dispositions
.filter((entry) => entry.kind === "blocked")
.map((entry) => entry.runId),
};
}
async function drainRunningRunsForShutdown(
signal: "SIGINT" | "SIGTERM",
now = new Date(),
@ -12851,7 +13011,12 @@ export function heartbeatService(
) {
const selectedRunIds = runIds ? [...new Set(runIds)] : null;
if (selectedRunIds?.length === 0) {
return { interrupted: 0, interruptedRunIds: [], retryRunIds: [] };
return {
interrupted: 0,
interruptedRunIds: [],
retryRunIds: [],
restartSuspendedRunIds: [],
};
}
const activeRuns = await db
.select({
@ -12871,8 +13036,66 @@ export function heartbeatService(
const interruptedRunIds: string[] = [];
const retryRunIds: string[] = [];
const restartSuspendedRunIds: string[] = [];
for (const { run, agent } of activeRuns) {
if (
run.runtimeMode === "native" &&
agent.adapterType === "paperclip_runner"
) {
const recoveryHistoryEntry = JSON.stringify({
at: now.toISOString(),
restartKind: "graceful",
disposition: "restart_suspended",
reason: signal,
processPid: run.processPid,
processStartedAt: run.processStartedAt?.toISOString() ?? null,
});
await db
.update(nativeRunFinalizations)
.set({
recoveryState: "awaiting_runner_reattach",
recoveryHistory: sql`(
select coalesce(jsonb_agg(item order by ordinal), '[]'::jsonb)
from jsonb_array_elements(
coalesce(${nativeRunFinalizations.recoveryHistory}, '[]'::jsonb)
|| jsonb_build_array(${recoveryHistoryEntry}::jsonb)
) with ordinality as history(item, ordinal)
where ordinal > greatest(
jsonb_array_length(
coalesce(${nativeRunFinalizations.recoveryHistory}, '[]'::jsonb)
|| jsonb_build_array(${recoveryHistoryEntry}::jsonb)
) - 20,
0
)
)`,
updatedAt: now,
})
.where(
and(
eq(nativeRunFinalizations.runId, run.id),
isNull(nativeRunFinalizations.resultId),
),
);
await appendRunEvent(run, {
eventType: "native.recovery.transition",
stream: "system",
level: "info",
message:
"Server shutdown suspended native controller ownership without cancelling provider work",
payload: {
restartKind: "graceful",
signal,
runnerDisposition: "awaiting_runner_reattach",
processPid: run.processPid,
processGroupId: run.processGroupId,
processStartedAt: run.processStartedAt?.toISOString() ?? null,
retryRunCreated: false,
},
});
restartSuspendedRunIds.push(run.id);
continue;
}
const message = `Interrupted by graceful server shutdown (${signal}); retry queued for restart recovery`;
const running = runningProcesses.get(run.id);
try {
@ -12979,6 +13202,7 @@ export function heartbeatService(
interrupted: interruptedRunIds.length,
interruptedRunIds,
retryRunIds,
restartSuspendedRunIds,
};
}
@ -16740,6 +16964,7 @@ export function heartbeatService(
adapterType: agents.adapterType,
adapterConfig: agents.adapterConfig,
nativeCoordinatorPhase: nativeRunFinalizations.phase,
nativeRecoveryState: nativeRunFinalizations.recoveryState,
})
.from(heartbeatRuns)
.innerJoin(agents, eq(heartbeatRuns.agentId, agents.id))
@ -16785,6 +17010,7 @@ export function heartbeatService(
adapterType,
adapterConfig,
nativeCoordinatorPhase,
nativeRecoveryState,
} of activeRuns) {
const nativeRun = run.runtimeMode === "native";
const nativeProcessPidAlive =
@ -16795,6 +17021,20 @@ export function heartbeatService(
isProcessGroupAlive(run.processGroupId);
const locallyTracked =
runningProcesses.has(run.id) || activeRunExecutions.has(run.id);
if (
nativeRun &&
(
[
"awaiting_evidence",
"awaiting_runner_reattach",
"resuming_session",
"bootstrap_incomplete",
].includes(nativeRecoveryState ?? "") ||
nativeCoordinatorPhase === "retryable_failure"
)
) {
continue;
}
const observedOwnerUnverified =
nativeRun &&
(nativeCoordinatorPhase === "observed" ||
@ -17426,7 +17666,10 @@ export function heartbeatService(
async function executeRun(
runId: string,
runOptions: { nativeLeaseOwner?: string } = {},
runOptions: {
nativeLeaseOwner?: string;
nativeRestartRecovery?: NativeRestartRecoveryClaim;
} = {},
) {
if ((await getSchedulingSuppression()).suppressed) {
try {
@ -17453,7 +17696,11 @@ export function heartbeatService(
run = claimed;
}
if (runOptions.nativeLeaseOwner && run.runtimeMode === "native") {
if (
runOptions.nativeLeaseOwner &&
run.runtimeMode === "native" &&
runOptions.nativeRestartRecovery?.kind !== "reattach_existing_runner"
) {
// A numeric PID or process-group ID is a liveness signal, never an
// ownership capability: the OS may have recycled it after the service
// restart. A still-active in-memory child handle is also insufficient to
@ -19929,14 +20176,79 @@ export function heartbeatService(
const taskNativeSessionId = readNonEmptyString(
taskSessionDecodedParams?.sessionId,
);
const resumableTaskSessionId =
// Compatibility for native retry rows created before same-run restart
// recovery existed. Only an entirely unused replacement row may
// inherit its source checkpoint; any process/provider evidence on the
// replacement makes the ownership ambiguous and therefore ineligible.
const legacyRetrySource =
run.retryOfRunId
? await db
.select({
id: heartbeatRuns.id,
companyId: heartbeatRuns.companyId,
agentId: heartbeatRuns.agentId,
runnerInstanceId: heartbeatRuns.runnerInstanceId,
nativeSessionId: heartbeatRuns.nativeSessionId,
runnerProfileJson: heartbeatRuns.runnerProfileJson,
runtimeMode: heartbeatRuns.runtimeMode,
status: heartbeatRuns.status,
})
.from(heartbeatRuns)
.where(
and(
eq(heartbeatRuns.id, run.retryOfRunId),
eq(heartbeatRuns.companyId, agent.companyId),
eq(heartbeatRuns.agentId, agent.id),
),
)
.limit(1)
.then((rows) => rows[0] ?? null)
: null;
const legacyRetryHasProviderEvidence = legacyRetrySource
? await db
.select({ id: heartbeatRunEvents.id })
.from(heartbeatRunEvents)
.where(
and(
eq(heartbeatRunEvents.runId, run.id),
inArray(heartbeatRunEvents.eventType, [
"harness.ready",
"session.started",
"session.resumed",
"session.updated",
"turn.started",
"provider.event",
"provider.rpc_result",
]),
),
)
.limit(1)
.then((rows) => rows.length > 0)
: false;
const compatibleLegacyRetrySource =
isUnusedLegacyNativeRetryReplacement({
replacement: run,
source: legacyRetrySource,
hasProviderEvents: legacyRetryHasProviderEvidence,
})
? legacyRetrySource
: null;
const legacyRetrySessionId = compatibleLegacyRetrySource
?.nativeSessionId;
const taskResumeRunId =
taskSessionForRun?.lastRunId &&
taskSessionForRun.lastRunId !== run.id &&
isNativeSessionId(taskNativeSessionId)
? taskNativeSessionId
? taskSessionForRun.lastRunId
: null;
const resumableTaskSessionId =
taskResumeRunId
? taskNativeSessionId
: legacyRetrySessionId ?? null;
const priorNativeRunId =
taskResumeRunId ?? compatibleLegacyRetrySource?.id ?? null;
const previousNativeRun =
resumableTaskSessionId && taskSessionForRun?.lastRunId
resumableTaskSessionId && priorNativeRunId
? await db
.select({
id: heartbeatRuns.id,
@ -19949,7 +20261,7 @@ export function heartbeatService(
.from(heartbeatRuns)
.where(
and(
eq(heartbeatRuns.id, taskSessionForRun.lastRunId),
eq(heartbeatRuns.id, priorNativeRunId),
eq(heartbeatRuns.companyId, agent.companyId),
eq(heartbeatRuns.agentId, agent.id),
eq(heartbeatRuns.nativeSessionId, resumableTaskSessionId),
@ -20697,6 +21009,7 @@ export function heartbeatService(
execution: nativeExecution,
runnerInstanceId: nativeRunnerInstanceId,
leaseOwner: runOptions.nativeLeaseOwner,
restartRecovery: runOptions.nativeRestartRecovery,
backend:
options.nativeSessionBackendFactory?.(nativeExecution),
useRunnerd: agent.adapterType === "paperclip_runner",
@ -22119,7 +22432,7 @@ export function heartbeatService(
}
}
activeRunExecutions.delete(run.id);
if (!nativeSessionResumeScheduled) {
if (!nativeSessionResumeScheduled && !shutdownInProgress) {
await startNextQueuedRunForAgent(run.agentId);
}
}
@ -25592,6 +25905,7 @@ export function heartbeatService(
prepareHotRestartShutdown,
reconcileHotRestartAdoption,
recoverNativeRunsAfterRestart,
reapOrphanedRuns,
sweepPendingCleanupLeases,
// Override-aware scheduling-suppression check (honors the worktree

View File

@ -5,6 +5,7 @@ import { afterEach, describe, expect, it, vi } from "vitest";
import {
findMissingHotRestartSnapshotRunIds,
isObservedHotRestartTargetAlive,
parseHotRestartIntent,
readHotRestartIntent,
readProcessStartedAt,
removeHotRestartIntent,
@ -30,6 +31,64 @@ async function withTempHome<T>(fn: (homeDir: string) => Promise<T>) {
}
describe("hot-restart path compatibility", () => {
it("reads version-1 native recovery fields without making them mandatory", () => {
expect(
parseHotRestartIntent({
version: 1,
requestedAt: "2026-09-04T10:00:00.000Z",
recoveryRequestId: "restart-request-1",
previousServerPid: 123,
drainRequired: false,
requestedByRunId: null,
preflightActiveRunIds: ["run-1"],
shutdownSnapshot: {
capturedAt: "2026-09-04T10:00:01.000Z",
signal: "SIGTERM",
activeRuns: [
{
runId: "run-1",
companyId: "company-1",
agentId: "agent-1",
adapterType: "paperclip_runner",
status: "running",
processPid: 456,
processGroupId: 456,
issueId: "issue-1",
runtimeMode: "native",
nativeSessionId: "session-1",
runnerInstanceId: "runner-1",
processStartedAt: "2026-09-04T09:59:00.000Z",
},
],
},
}),
).toMatchObject({
version: 1,
recoveryRequestId: "restart-request-1",
shutdownSnapshot: {
activeRuns: [
{
runtimeMode: "native",
nativeSessionId: "session-1",
runnerInstanceId: "runner-1",
processStartedAt: "2026-09-04T09:59:00.000Z",
},
],
},
});
expect(
parseHotRestartIntent({
version: 1,
requestedAt: "2026-09-04T10:00:00.000Z",
previousServerPid: 123,
drainRequired: false,
requestedByRunId: null,
preflightActiveRunIds: [],
}),
).not.toBeNull();
});
it("reads Linux process start time from proc metadata", async () => {
await expect(
readProcessStartedAt(123, {

View File

@ -25,11 +25,16 @@ export type HotRestartIntentRun = {
processPid: number | null;
processGroupId: number | null;
issueId: string | null;
runtimeMode?: string | null;
nativeSessionId?: string | null;
runnerInstanceId?: string | null;
processStartedAt?: string | null;
};
export type HotRestartIntent = {
version: 1;
requestedAt: string;
recoveryRequestId?: string | null;
previousServerPid: number;
previousServerIdentity?: string | null;
previousServerStartedAt?: string | null;
@ -382,7 +387,8 @@ function isSameHotRestartRequest(left: HotRestartIntent, right: HotRestartIntent
return left.requestedAt === right.requestedAt
&& left.previousServerPid === right.previousServerPid
&& left.drainRequired === right.drainRequired
&& left.requestedByRunId === right.requestedByRunId;
&& left.requestedByRunId === right.requestedByRunId
&& (left.recoveryRequestId ?? null) === (right.recoveryRequestId ?? null);
}
function parseRun(value: unknown): HotRestartIntentRun | null {
@ -402,6 +408,10 @@ function parseRun(value: unknown): HotRestartIntentRun | null {
processPid: asNumber(value.processPid),
processGroupId: asNumber(value.processGroupId),
issueId: asString(value.issueId),
runtimeMode: asString(value.runtimeMode),
nativeSessionId: asString(value.nativeSessionId),
runnerInstanceId: asString(value.runnerInstanceId),
processStartedAt: asDateString(value.processStartedAt),
};
}
@ -414,6 +424,7 @@ export function parseHotRestartIntent(value: unknown): HotRestartIntent | null {
const intent: HotRestartIntent = {
version: 1,
requestedAt,
recoveryRequestId: asString(value.recoveryRequestId),
previousServerPid,
previousServerIdentity: asString(value.previousServerIdentity),
previousServerStartedAt: asDateString(value.previousServerStartedAt),
@ -492,6 +503,7 @@ export async function writeHotRestartIntent(input: {
requestedByRunId?: string | null;
preflightActiveRunIds?: string[];
requestedAt?: Date;
recoveryRequestId?: string | null;
homeDir?: string;
}) {
const previousServerStartedAt = input.previousServerStartedAt === undefined
@ -507,6 +519,7 @@ export async function writeHotRestartIntent(input: {
const intent: HotRestartIntent = {
version: 1,
requestedAt: (input.requestedAt ?? new Date()).toISOString(),
recoveryRequestId: input.recoveryRequestId ?? null,
previousServerPid: input.previousServerPid,
previousServerIdentity,
previousServerStartedAt,

View File

@ -8,4 +8,5 @@ export * from "./native-interaction-bridge.js";
export * from "./paperclip-control-plane-port.js";
export * from "./native-run-finalizer.js";
export * from "./native-finalization-reconciler.js";
export * from "./native-restart-recovery.js";
export * from "./status-arbiter.js";

View File

@ -0,0 +1,259 @@
import { describe, expect, it, vi } from "vitest";
import {
classifyNativeRunnerRecoveryEvidence,
evaluateNativeControllerTakeover,
evaluateNativeProviderProcesses,
nextNativeProviderAttempt,
} from "./native-restart-recovery.js";
describe("native restart recovery classification", () => {
it("keeps controller-only recovery out of the provider retry budget", () => {
expect(nextNativeProviderAttempt(2, "reattach_existing_runner")).toBe(2);
expect(nextNativeProviderAttempt(2, "bootstrap_incomplete")).toBe(2);
expect(nextNativeProviderAttempt(2, "resume_dead_runner")).toBe(3);
});
it.each([
{
name: "reattaches an exact live runner",
evidence: {
runnerPidAlive: true,
runnerGroupAlive: true,
processStartMatches: true,
hasCheckpoint: true,
hasProviderEvidence: true,
},
expected: "reattach_existing_runner",
},
{
name: "resumes a dead checkpointed runner",
evidence: {
runnerPidAlive: false,
runnerGroupAlive: false,
processStartMatches: false,
hasCheckpoint: true,
hasProviderEvidence: true,
},
expected: "resume_dead_runner",
},
{
name: "retries an incomplete bootstrap on the same run",
evidence: {
runnerPidAlive: false,
runnerGroupAlive: false,
processStartMatches: false,
hasCheckpoint: false,
hasProviderEvidence: false,
},
expected: "bootstrap_incomplete",
},
{
name: "blocks a recycled live PID",
evidence: {
runnerPidAlive: true,
runnerGroupAlive: true,
processStartMatches: false,
hasCheckpoint: true,
hasProviderEvidence: true,
},
expected: null,
},
{
name: "blocks ambiguous checkpoint evidence",
evidence: {
runnerPidAlive: false,
runnerGroupAlive: false,
processStartMatches: false,
hasCheckpoint: true,
hasProviderEvidence: false,
},
expected: null,
},
{
name: "blocks a checkpoint bound to a different native identity",
evidence: {
runnerPidAlive: false,
runnerGroupAlive: false,
processStartMatches: false,
hasCheckpoint: true,
checkpointIdentityMatches: false,
hasProviderEvidence: true,
},
expected: null,
},
{
name: "blocks while a known provider process outlives its runner",
evidence: {
runnerPidAlive: false,
runnerGroupAlive: false,
processStartMatches: false,
knownProviderProcessAlive: true,
hasCheckpoint: true,
hasProviderEvidence: true,
},
expected: null,
},
{
name: "blocks while a live provider PID has no durable fingerprint",
evidence: {
runnerPidAlive: false,
runnerGroupAlive: false,
processStartMatches: false,
knownProviderProcessIdentityAmbiguous: true,
hasCheckpoint: true,
hasProviderEvidence: true,
},
expected: null,
},
])("$name", ({ evidence, expected }) => {
expect(classifyNativeRunnerRecoveryEvidence(evidence).claimKind).toBe(
expected,
);
});
});
describe("native controller takeover fencing", () => {
const now = new Date("2026-09-04T12:00:00.000Z");
const recordedStart = new Date("2026-09-04T11:00:00.000Z");
function owner(
overrides: Partial<{
leaseOwner: string | null;
leaseExpiresAt: Date | null;
controllerPid: number | null;
controllerProcessStartedAt: Date | null;
}> = {},
) {
return {
leaseOwner: "old-controller",
leaseExpiresAt: new Date("2026-09-04T12:20:00.000Z"),
controllerPid: 123,
controllerProcessStartedAt: recordedStart,
...overrides,
};
}
it("takes over immediately when the exact prior controller process died", async () => {
await expect(
evaluateNativeControllerTakeover({
owner: owner(),
now,
isProcessAlive: () => false,
}),
).resolves.toEqual({
allowed: true,
reason: "controller_process_dead",
});
});
it("takes over a recycled controller PID", async () => {
await expect(
evaluateNativeControllerTakeover({
owner: owner(),
now,
isProcessAlive: () => true,
readProcessStartedAt: async () => new Date("2026-09-04T11:30:00.000Z"),
}),
).resolves.toEqual({
allowed: true,
reason: "controller_pid_recycled",
});
});
it("does not steal a live controller lease", async () => {
const readStartedAt = vi.fn(async () => recordedStart);
await expect(
evaluateNativeControllerTakeover({
owner: owner(),
now,
isProcessAlive: () => true,
readProcessStartedAt: readStartedAt,
}),
).resolves.toEqual({
allowed: false,
reason: "controller_still_alive",
});
expect(readStartedAt).toHaveBeenCalledWith(123);
});
it("does not let lease expiry bypass a live controller fence", async () => {
const isProcessAlive = vi.fn(() => true);
await expect(
evaluateNativeControllerTakeover({
owner: owner({
leaseExpiresAt: new Date("2026-09-04T11:59:59.000Z"),
}),
now,
isProcessAlive,
}),
).resolves.toEqual({ allowed: false, reason: "controller_still_alive" });
expect(isProcessAlive).toHaveBeenCalledWith(123);
});
it("takes over an expired lease after proving the controller died", async () => {
await expect(
evaluateNativeControllerTakeover({
owner: owner({
leaseExpiresAt: new Date("2026-09-04T11:59:59.000Z"),
}),
now,
isProcessAlive: () => false,
}),
).resolves.toEqual({
allowed: true,
reason: "expired_lease_controller_process_dead",
});
});
it("keeps a coordinated previous controller fenced until that exact process exits", async () => {
await expect(
evaluateNativeControllerTakeover({
owner: owner(),
now,
coordinatedPreviousController: {
pid: 123,
processStartedAt: recordedStart,
},
isProcessAlive: () => true,
readProcessStartedAt: async () => recordedStart,
}),
).resolves.toEqual({
allowed: false,
reason: "coordinated_controller_still_alive",
});
});
});
describe("native provider process fencing", () => {
const recordedStart = new Date("2026-09-04T11:00:00.000Z");
it("does not mistake a recycled provider PID for the old live provider", async () => {
await expect(
evaluateNativeProviderProcesses({
identities: [{ pid: 456, processStartedAt: recordedStart }],
isProcessAlive: () => true,
readProcessStartedAt: async () => new Date("2026-09-04T11:30:00.000Z"),
}),
).resolves.toEqual({
knownPids: [456],
livePids: [],
ambiguousLivePids: [],
recycledPids: [456],
});
});
it("fails closed when a live provider has no comparable fingerprint", async () => {
await expect(
evaluateNativeProviderProcesses({
identities: [{ pid: 456, processStartedAt: null }],
isProcessAlive: () => true,
readProcessStartedAt: async () => recordedStart,
}),
).resolves.toMatchObject({
livePids: [],
ambiguousLivePids: [456],
recycledPids: [],
});
});
});

View File

@ -0,0 +1,854 @@
import { randomUUID } from "node:crypto";
import { and, desc, eq, inArray, isNull, lte, or, sql } from "drizzle-orm";
import type { Db } from "@paperclipai/db";
import {
agents,
heartbeatRunEvents,
heartbeatRuns,
issues,
nativeRunFinalizations,
} from "@paperclipai/db";
import { readProcessStartedAt } from "../hot-restart.js";
import { getServerInfoSnapshot } from "../../server-info.js";
import { redactSensitiveText } from "../../redaction.js";
export type NativeControllerIdentity = {
bootId: string;
pid: number;
processStartedAt: Date;
};
export type NativeRestartKind = "hot" | "hard" | "graceful";
export type NativeRestartRecoveryClaim =
| {
kind: "reattach_existing_runner";
runId: string;
leaseOwner: string;
controllerGeneration: number;
providerAttempt: number;
restartKind: NativeRestartKind;
recoveryRequestId: string | null;
process: {
pid: number;
processGroupId: number | null;
startedAt: string;
};
}
| {
kind: "resume_dead_runner";
runId: string;
leaseOwner: string;
controllerGeneration: number;
providerAttempt: number;
restartKind: NativeRestartKind;
recoveryRequestId: string | null;
}
| {
kind: "bootstrap_incomplete";
runId: string;
leaseOwner: string;
controllerGeneration: number;
providerAttempt: number;
restartKind: NativeRestartKind;
recoveryRequestId: string | null;
};
export type NativeRestartRecoveryDisposition =
| NativeRestartRecoveryClaim
| {
kind: "awaiting_evidence" | "blocked" | "already_finalized";
runId: string;
reason: string;
};
export function nextNativeProviderAttempt(
currentAttempt: number,
recoveryKind?: NativeRestartRecoveryClaim["kind"],
): number {
return recoveryKind === "reattach_existing_runner" ||
recoveryKind === "bootstrap_incomplete"
? currentAttempt
: currentAttempt + 1;
}
const controllerBootId = randomUUID();
function processIsAlive(pid: number | null | undefined): boolean {
if (!pid || !Number.isInteger(pid) || pid <= 0) return false;
try {
process.kill(pid, 0);
return true;
} catch (error) {
return (error as NodeJS.ErrnoException | undefined)?.code === "EPERM";
}
}
function processGroupIsAlive(
processGroupId: number | null | undefined,
): boolean {
if (
process.platform === "win32" ||
!processGroupId ||
!Number.isInteger(processGroupId) ||
processGroupId <= 0
) {
return false;
}
try {
process.kill(-processGroupId, 0);
return true;
} catch (error) {
return (error as NodeJS.ErrnoException | undefined)?.code === "EPERM";
}
}
async function observedProcessStart(pid: number): Promise<Date | null> {
try {
const startedAt = await readProcessStartedAt(pid);
if (!startedAt) return null;
const parsed = new Date(startedAt);
return Number.isNaN(parsed.getTime()) ? null : parsed;
} catch {
return null;
}
}
export async function currentNativeControllerIdentity(): Promise<NativeControllerIdentity> {
const observed = await observedProcessStart(process.pid);
const fallback = new Date(getServerInfoSnapshot().processStartedAt);
return {
bootId: controllerBootId,
pid: process.pid,
processStartedAt:
observed ?? (Number.isNaN(fallback.getTime()) ? new Date() : fallback),
};
}
function sameProcessStart(left: Date | null, right: Date | null): boolean {
return left !== null && right !== null && left.getTime() === right.getTime();
}
export async function evaluateNativeControllerTakeover(input: {
owner: Pick<
typeof nativeRunFinalizations.$inferSelect,
| "leaseOwner"
| "leaseExpiresAt"
| "controllerPid"
| "controllerProcessStartedAt"
>;
now: Date;
coordinatedPreviousController?: {
pid: number;
processStartedAt: Date | null;
} | null;
isProcessAlive?: (pid: number) => boolean;
readProcessStartedAt?: (pid: number) => Promise<Date | null>;
}): Promise<{ allowed: boolean; reason: string }> {
const { owner, now } = input;
const isAlive = input.isProcessAlive ?? processIsAlive;
const readStartedAt = input.readProcessStartedAt ?? observedProcessStart;
if (!owner.leaseOwner || !owner.leaseExpiresAt) {
return { allowed: true, reason: "unowned" };
}
const priorPid = owner.controllerPid;
if (!priorPid) {
return {
allowed: false,
reason:
owner.leaseExpiresAt <= now
? "expired_lease_without_controller_identity"
: "live_legacy_lease_without_controller_identity",
};
}
if (!isAlive(priorPid)) {
return {
allowed: true,
reason:
owner.leaseExpiresAt <= now
? "expired_lease_controller_process_dead"
: "controller_process_dead",
};
}
const observedStartedAt = await readStartedAt(priorPid);
if (
owner.controllerProcessStartedAt &&
observedStartedAt &&
!sameProcessStart(owner.controllerProcessStartedAt, observedStartedAt)
) {
return { allowed: true, reason: "controller_pid_recycled" };
}
const coordinated = input.coordinatedPreviousController;
if (
coordinated &&
coordinated.pid === priorPid &&
coordinated.processStartedAt &&
owner.controllerProcessStartedAt &&
sameProcessStart(
coordinated.processStartedAt,
owner.controllerProcessStartedAt,
)
) {
return { allowed: false, reason: "coordinated_controller_still_alive" };
}
return { allowed: false, reason: "controller_still_alive" };
}
export type NativeProviderProcessIdentity = {
pid: number;
processStartedAt: Date | null;
};
export async function evaluateNativeProviderProcesses(input: {
identities: NativeProviderProcessIdentity[];
isProcessAlive?: (pid: number) => boolean;
readProcessStartedAt?: (pid: number) => Promise<Date | null>;
}): Promise<{
knownPids: number[];
livePids: number[];
ambiguousLivePids: number[];
recycledPids: number[];
}> {
const isAlive = input.isProcessAlive ?? processIsAlive;
const readStartedAt = input.readProcessStartedAt ?? observedProcessStart;
const expectedStarts = new Map<number, Date | null>();
for (const identity of input.identities) {
const current = expectedStarts.get(identity.pid);
if (
current === undefined ||
(current === null && identity.processStartedAt)
) {
expectedStarts.set(identity.pid, identity.processStartedAt);
}
}
const livePids: number[] = [];
const ambiguousLivePids: number[] = [];
const recycledPids: number[] = [];
for (const [pid, expectedStartedAt] of expectedStarts) {
if (!isAlive(pid)) continue;
const observedStartedAt = await readStartedAt(pid);
if (!expectedStartedAt || !observedStartedAt) {
ambiguousLivePids.push(pid);
} else if (sameProcessStart(expectedStartedAt, observedStartedAt)) {
livePids.push(pid);
} else {
recycledPids.push(pid);
}
}
return {
knownPids: [...expectedStarts.keys()],
livePids,
ambiguousLivePids,
recycledPids,
};
}
export function classifyNativeRunnerRecoveryEvidence(input: {
runnerPidAlive: boolean;
runnerGroupAlive: boolean;
processStartMatches: boolean;
knownProviderProcessAlive?: boolean;
knownProviderProcessIdentityAmbiguous?: boolean;
hasCheckpoint: boolean;
checkpointIdentityMatches?: boolean;
hasProviderEvidence: boolean;
}): {
claimKind: NativeRestartRecoveryClaim["kind"] | null;
reason: string;
} {
if (input.runnerPidAlive && input.processStartMatches) {
return {
claimKind: "reattach_existing_runner",
reason: "runner_process_identity_verified",
};
}
if (input.runnerPidAlive || input.runnerGroupAlive) {
return {
claimKind: null,
reason: input.runnerPidAlive
? "live_runner_identity_mismatch"
: "live_runner_process_group_without_exact_pid",
};
}
if (input.knownProviderProcessAlive) {
return {
claimKind: null,
reason: "live_provider_process_without_runner_authority",
};
}
if (input.knownProviderProcessIdentityAmbiguous) {
return {
claimKind: null,
reason: "live_provider_process_identity_unverifiable",
};
}
const checkpointIdentityMatches =
input.checkpointIdentityMatches ?? input.hasCheckpoint;
if (
input.hasCheckpoint &&
checkpointIdentityMatches &&
input.hasProviderEvidence
) {
return {
claimKind: "resume_dead_runner",
reason: "dead_runner_with_exact_provider_checkpoint",
};
}
if (!input.hasCheckpoint && !input.hasProviderEvidence) {
return {
claimKind: "bootstrap_incomplete",
reason: "dead_runner_before_provider_identity",
};
}
return { claimKind: null, reason: "ambiguous_provider_recovery_evidence" };
}
function historyEntry(input: {
now: Date;
restartKind: NativeRestartKind;
controller: NativeControllerIdentity;
generation: number;
disposition: string;
reason: string;
requestId: string | null;
providerAttempt: number;
processPid: number | null;
processStartedAt: Date | null;
hasCheckpoint: boolean;
checkpointIdentityMatches?: boolean;
hasProviderEvidence: boolean;
knownProviderPids?: number[];
liveProviderPids?: number[];
ambiguousProviderPids?: number[];
recycledProviderPids?: number[];
stderrTail?: string | null;
}) {
return {
at: input.now.toISOString(),
restartKind: input.restartKind,
controllerBootId: input.controller.bootId,
controllerPid: input.controller.pid,
controllerProcessStartedAt: input.controller.processStartedAt.toISOString(),
controllerGeneration: input.generation,
disposition: input.disposition,
reason: input.reason,
stateRootAction:
input.disposition === "reattach_existing_runner"
? "reopen_exact_root"
: input.disposition === "resume_dead_runner"
? "reuse_exact_root"
: input.disposition === "bootstrap_incomplete"
? "quarantine_incomplete_root_then_bootstrap"
: input.disposition === "awaiting_evidence"
? "preserve_awaiting_evidence"
: "preserve_fail_closed",
recoveryRequestId: input.requestId,
providerAttempt: input.providerAttempt,
processPid: input.processPid,
processStartedAt: input.processStartedAt?.toISOString() ?? null,
hasCheckpoint: input.hasCheckpoint,
checkpointIdentityMatches: input.checkpointIdentityMatches ?? false,
hasProviderEvidence: input.hasProviderEvidence,
knownProviderPids: input.knownProviderPids ?? [],
liveProviderPids: input.liveProviderPids ?? [],
ambiguousProviderPids: input.ambiguousProviderPids ?? [],
recycledProviderPids: input.recycledProviderPids ?? [],
stderrTail:
input.stderrTail && input.stderrTail.trim()
? redactSensitiveText(input.stderrTail).slice(-4_096)
: null,
} satisfies Record<string, unknown>;
}
function appendBoundedRecoveryHistory(entry: Record<string, unknown>) {
const encoded = JSON.stringify(entry);
return sql`(
select coalesce(jsonb_agg(item order by ordinal), '[]'::jsonb)
from jsonb_array_elements(
coalesce(${nativeRunFinalizations.recoveryHistory}, '[]'::jsonb)
|| jsonb_build_array(${encoded}::jsonb)
) with ordinality as history(item, ordinal)
where ordinal > greatest(
jsonb_array_length(
coalesce(${nativeRunFinalizations.recoveryHistory}, '[]'::jsonb)
|| jsonb_build_array(${encoded}::jsonb)
) - 20,
0
)
)`;
}
/**
* Claims abandoned local runner executions without changing the provider retry
* attempt. The transaction locks the run, coordinator, and issue execution
* owner so two successor servers cannot both reconstruct the same authority.
*/
export async function claimNativeRestartRecoveries(input: {
db: Db;
controller?: NativeControllerIdentity;
restartKind: NativeRestartKind;
recoveryRequestId?: string | null;
runIds?: string[];
now?: Date;
limit?: number;
coordinatedPreviousController?: {
pid: number;
processStartedAt: Date | null;
} | null;
}): Promise<NativeRestartRecoveryDisposition[]> {
const now = input.now ?? new Date();
const controller =
input.controller ?? (await currentNativeControllerIdentity());
const candidateQuery = input.db
.select({ runId: heartbeatRuns.id })
.from(heartbeatRuns)
.innerJoin(agents, eq(agents.id, heartbeatRuns.agentId))
.innerJoin(
nativeRunFinalizations,
eq(nativeRunFinalizations.runId, heartbeatRuns.id),
)
.where(
and(
eq(heartbeatRuns.runtimeMode, "native"),
eq(agents.adapterType, "paperclip_runner"),
inArray(heartbeatRuns.status, ["running", "failed"]),
isNull(nativeRunFinalizations.resultId),
or(
eq(nativeRunFinalizations.phase, "observed"),
and(
eq(nativeRunFinalizations.phase, "retryable_failure"),
or(
isNull(nativeRunFinalizations.nextAttemptAt),
lte(nativeRunFinalizations.nextAttemptAt, now),
),
),
),
...(input.runIds?.length
? [inArray(heartbeatRuns.id, input.runIds)]
: []),
),
)
.orderBy(heartbeatRuns.id);
const candidates = await (input.limit === undefined
? candidateQuery
: candidateQuery.limit(input.limit));
const dispositions: NativeRestartRecoveryDisposition[] = [];
for (const candidate of candidates) {
const disposition = await input.db.transaction(async (tx) => {
const row = await tx
.select({
run: heartbeatRuns,
coordinator: nativeRunFinalizations,
issueExecutionRunId: issues.executionRunId,
})
.from(heartbeatRuns)
.innerJoin(
nativeRunFinalizations,
eq(nativeRunFinalizations.runId, heartbeatRuns.id),
)
.innerJoin(
issues,
and(
eq(issues.id, nativeRunFinalizations.issueId),
eq(issues.companyId, nativeRunFinalizations.companyId),
),
)
.where(eq(heartbeatRuns.id, candidate.runId))
.for("update")
.limit(1)
.then((rows) => rows[0] ?? null);
if (!row) {
return {
kind: "blocked",
runId: candidate.runId,
reason: "recovery_rows_missing",
} as const;
}
if (row.coordinator.resultId) {
return {
kind: "already_finalized",
runId: row.run.id,
reason: "result_already_persisted",
} as const;
}
if (row.issueExecutionRunId !== row.run.id) {
return {
kind: "blocked",
runId: row.run.id,
reason: "issue_execution_lock_changed",
} as const;
}
const takeover = await evaluateNativeControllerTakeover({
owner: row.coordinator,
now,
coordinatedPreviousController:
input.coordinatedPreviousController ?? null,
});
if (!takeover.allowed) {
const event = historyEntry({
now,
restartKind: input.restartKind,
controller,
generation: row.coordinator.controllerGeneration,
disposition: "awaiting_evidence",
reason: takeover.reason,
requestId: input.recoveryRequestId ?? null,
providerAttempt: row.coordinator.attempt,
processPid: row.run.processPid,
processStartedAt: row.run.processStartedAt,
hasCheckpoint: false,
hasProviderEvidence: false,
stderrTail: row.run.stderrExcerpt,
});
await tx
.update(nativeRunFinalizations)
.set({
recoveryState: "awaiting_evidence",
recoveryRequestId: input.recoveryRequestId ?? null,
recoveryHistory: appendBoundedRecoveryHistory(event),
updatedAt: now,
})
.where(
and(
eq(nativeRunFinalizations.runId, row.run.id),
eq(nativeRunFinalizations.attempt, row.coordinator.attempt),
eq(nativeRunFinalizations.phase, row.coordinator.phase),
),
);
return {
kind: "awaiting_evidence",
runId: row.run.id,
reason: takeover.reason,
} as const;
}
const runnerPidAlive = processIsAlive(row.run.processPid);
const runnerGroupAlive = processGroupIsAlive(row.run.processGroupId);
const observedRunnerStart =
row.run.processPid && runnerPidAlive
? await observedProcessStart(row.run.processPid)
: null;
const exactRunnerIdentity =
runnerPidAlive &&
row.run.processPid !== null &&
sameProcessStart(row.run.processStartedAt, observedRunnerStart);
const profile = row.run.runnerProfileJson ?? {};
const checkpoint = profile.sessionCheckpoint;
const checkpointRecord =
checkpoint &&
typeof checkpoint === "object" &&
!Array.isArray(checkpoint)
? (checkpoint as Record<string, unknown>)
: {};
const hasCheckpoint = checkpoint !== undefined && checkpoint !== null;
const checkpointIdentity =
checkpointRecord.identity &&
typeof checkpointRecord.identity === "object" &&
!Array.isArray(checkpointRecord.identity)
? (checkpointRecord.identity as Record<string, unknown>)
: {};
const checkpointIdentityMatches =
hasCheckpoint &&
typeof row.run.nativeSessionId === "string" &&
row.run.nativeSessionId.length > 0 &&
checkpointIdentity.runId === row.run.id &&
checkpointIdentity.companyId === row.run.companyId &&
checkpointIdentity.issueId === row.coordinator.issueId &&
checkpointIdentity.agentId === row.run.agentId &&
checkpointIdentity.sessionId === row.run.nativeSessionId;
const hasCheckpointProviderIdentity =
(typeof checkpointRecord.providerSessionId === "string" &&
checkpointRecord.providerSessionId.length > 0) ||
(checkpointRecord.providerIdentity !== null &&
typeof checkpointRecord.providerIdentity === "object" &&
!Array.isArray(checkpointRecord.providerIdentity) &&
Object.keys(checkpointRecord.providerIdentity).length > 0);
const providerEvents = await tx
.select({
id: heartbeatRunEvents.id,
payload: heartbeatRunEvents.payload,
})
.from(heartbeatRunEvents)
.where(
and(
eq(heartbeatRunEvents.runId, row.run.id),
inArray(heartbeatRunEvents.eventType, [
"harness.ready",
"session.started",
"session.resumed",
"session.updated",
"turn.started",
"provider.event",
"provider.rpc_result",
]),
),
)
.orderBy(desc(heartbeatRunEvents.id))
.limit(100);
const checkpointProcess =
checkpointRecord.process &&
typeof checkpointRecord.process === "object" &&
!Array.isArray(checkpointRecord.process)
? (checkpointRecord.process as Record<string, unknown>)
: {};
const providerProcessIdentities: NativeProviderProcessIdentity[] = [];
const checkpointProcessFields = [
["providerPid", "providerProcessStartedAt"],
["codexPid", "codexProcessStartedAt"],
["sidecarPid", "sidecarProcessStartedAt"],
["agentPid", "agentProcessStartedAt"],
] as const;
for (const [field, startedAtField] of checkpointProcessFields) {
const value = checkpointProcess[field];
if (typeof value === "number" && Number.isInteger(value) && value > 0) {
const rawStartedAt = checkpointProcess[startedAtField];
const parsedStartedAt =
typeof rawStartedAt === "string" ? new Date(rawStartedAt) : null;
providerProcessIdentities.push({
pid: value,
processStartedAt:
parsedStartedAt && !Number.isNaN(parsedStartedAt.getTime())
? parsedStartedAt
: null,
});
}
}
for (const providerEvent of providerEvents) {
const prpEvent = providerEvent.payload?.prpEvent;
const event =
prpEvent && typeof prpEvent === "object" && !Array.isArray(prpEvent)
? (prpEvent as Record<string, unknown>)
: {};
const eventPayload =
event.payload &&
typeof event.payload === "object" &&
!Array.isArray(event.payload)
? (event.payload as Record<string, unknown>)
: {};
const processId = eventPayload.processId;
if (
typeof processId === "number" &&
Number.isInteger(processId) &&
processId > 0
) {
const rawStartedAt =
eventPayload.processStartedAt ??
eventPayload.providerProcessStartedAt;
const parsedStartedAt =
typeof rawStartedAt === "string" ? new Date(rawStartedAt) : null;
providerProcessIdentities.push({
pid: processId,
processStartedAt:
parsedStartedAt && !Number.isNaN(parsedStartedAt.getTime())
? parsedStartedAt
: null,
});
}
}
const providerProcesses = await evaluateNativeProviderProcesses({
identities: providerProcessIdentities.filter(
(identity) => identity.pid !== row.run.processPid,
),
});
const hasProviderEvidence =
hasCheckpointProviderIdentity || providerEvents.length > 0;
const classification = classifyNativeRunnerRecoveryEvidence({
runnerPidAlive,
runnerGroupAlive,
processStartMatches: exactRunnerIdentity,
knownProviderProcessAlive: providerProcesses.livePids.length > 0,
knownProviderProcessIdentityAmbiguous:
providerProcesses.ambiguousLivePids.length > 0,
hasCheckpoint,
checkpointIdentityMatches,
hasProviderEvidence,
});
const claimKind = classification.claimKind;
const reason = classification.reason;
if (!claimKind) {
const generation = row.coordinator.controllerGeneration;
const event = historyEntry({
now,
restartKind: input.restartKind,
controller,
generation,
disposition: "blocked",
reason,
requestId: input.recoveryRequestId ?? null,
providerAttempt: row.coordinator.attempt,
processPid: row.run.processPid,
processStartedAt: row.run.processStartedAt,
hasCheckpoint,
checkpointIdentityMatches,
hasProviderEvidence,
knownProviderPids: providerProcesses.knownPids,
liveProviderPids: providerProcesses.livePids,
ambiguousProviderPids: providerProcesses.ambiguousLivePids,
recycledProviderPids: providerProcesses.recycledPids,
stderrTail: row.run.stderrExcerpt,
});
await tx
.update(nativeRunFinalizations)
.set({
recoveryState: "blocked",
recoveryRequestId: input.recoveryRequestId ?? null,
recoveryHistory: appendBoundedRecoveryHistory(event),
updatedAt: now,
})
.where(
and(
eq(nativeRunFinalizations.runId, row.run.id),
eq(nativeRunFinalizations.attempt, row.coordinator.attempt),
eq(nativeRunFinalizations.phase, row.coordinator.phase),
),
);
return { kind: "blocked", runId: row.run.id, reason } as const;
}
const generation = row.coordinator.controllerGeneration + 1;
const leaseOwner = `${controller.bootId}:${generation}:${randomUUID()}`;
const event = historyEntry({
now,
restartKind: input.restartKind,
controller,
generation,
disposition: claimKind,
reason,
requestId: input.recoveryRequestId ?? null,
providerAttempt: row.coordinator.attempt,
processPid: row.run.processPid,
processStartedAt: row.run.processStartedAt,
hasCheckpoint,
checkpointIdentityMatches,
hasProviderEvidence,
knownProviderPids: providerProcesses.knownPids,
liveProviderPids: providerProcesses.livePids,
ambiguousProviderPids: providerProcesses.ambiguousLivePids,
recycledProviderPids: providerProcesses.recycledPids,
stderrTail: row.run.stderrExcerpt,
});
const claimed = await tx
.update(nativeRunFinalizations)
.set({
phase: "observed",
leaseOwner,
leaseExpiresAt: new Date(now.getTime() + 20 * 60_000),
controllerBootId: controller.bootId,
controllerPid: controller.pid,
controllerProcessStartedAt: controller.processStartedAt,
controllerGeneration: generation,
recoveryState:
claimKind === "reattach_existing_runner"
? "awaiting_runner_reattach"
: claimKind === "resume_dead_runner"
? "resuming_session"
: "bootstrap_incomplete",
recoveryRequestId: input.recoveryRequestId ?? null,
recoveryHistory: appendBoundedRecoveryHistory(event),
failureCode: null,
nextAttemptAt: null,
updatedAt: now,
})
.where(
and(
eq(nativeRunFinalizations.runId, row.run.id),
eq(nativeRunFinalizations.attempt, row.coordinator.attempt),
eq(nativeRunFinalizations.phase, row.coordinator.phase),
row.coordinator.leaseOwner === null
? isNull(nativeRunFinalizations.leaseOwner)
: eq(
nativeRunFinalizations.leaseOwner,
row.coordinator.leaseOwner,
),
),
)
.returning({ runId: nativeRunFinalizations.runId })
.then((rows) => rows[0] ?? null);
if (!claimed) {
return {
kind: "awaiting_evidence",
runId: row.run.id,
reason: "concurrent_recovery_claim",
} as const;
}
await tx
.update(heartbeatRuns)
.set({
status: "running",
finishedAt: null,
error: null,
errorCode: null,
nativePhase: "observed",
nativePhaseUpdatedAt: now,
...(claimKind === "reattach_existing_runner"
? {}
: {
processPid: null,
processGroupId: null,
processStartedAt: null,
}),
updatedAt: now,
})
.where(eq(heartbeatRuns.id, row.run.id));
const common = {
runId: row.run.id,
leaseOwner,
controllerGeneration: generation,
providerAttempt: row.coordinator.attempt,
restartKind: input.restartKind,
recoveryRequestId: input.recoveryRequestId ?? null,
};
if (claimKind === "reattach_existing_runner") {
return {
kind: claimKind,
...common,
process: {
pid: row.run.processPid!,
processGroupId: row.run.processGroupId,
startedAt: row.run.processStartedAt!.toISOString(),
},
} satisfies NativeRestartRecoveryClaim;
}
return {
kind: claimKind,
...common,
} satisfies NativeRestartRecoveryClaim;
});
dispositions.push(disposition);
}
return dispositions;
}
export async function nativeRestartRecoverySummary(db: Db) {
const rows = await db
.select({
state: nativeRunFinalizations.recoveryState,
count: sql<number>`count(*)::int`,
})
.from(nativeRunFinalizations)
.where(
inArray(nativeRunFinalizations.recoveryState, [
"awaiting_evidence",
"awaiting_runner_reattach",
"resuming_session",
"bootstrap_incomplete",
"blocked",
]),
)
.groupBy(nativeRunFinalizations.recoveryState);
return Object.fromEntries(
rows.map((row) => [row.state ?? "unknown", row.count]),
);
}

View File

@ -0,0 +1,998 @@
import { randomUUID } from "node:crypto";
import { spawn } from "node:child_process";
import { existsSync } from "node:fs";
import { mkdtemp, readFile, rename, rm } from "node:fs/promises";
import { createServer, type Server } from "node:http";
import { tmpdir } from "node:os";
import { resolve } from "node:path";
import { and, eq } from "drizzle-orm";
import {
agents,
companies,
createDb,
heartbeatRuns,
issues,
nativeRunFinalizations,
nativeRunResults,
} from "@paperclipai/db";
import { afterAll, beforeAll, describe, expect, it, vi } from "vitest";
import {
createRunnerdCodexTransport,
defaultCapabilityRunnerdBinary,
} from "../../vendor/paperclip-runner/index.js";
import {
getEmbeddedPostgresTestSupport,
startEmbeddedPostgresTestDatabase,
} from "../../__tests__/helpers/embedded-postgres.js";
import {
registerRunnerPrpAuthority,
runnerPrpWebSocketInternals,
setupRunnerPrpWebSocketServer,
} from "../../realtime/runner-prp-ws.js";
import { readProcessStartedAt } from "../hot-restart.js";
import { prepareNativeHeartbeatRun } from "./prepare-native-run.js";
import {
claimNativeRestartRecoveries,
type NativeControllerIdentity,
} from "./native-restart-recovery.js";
const embeddedPostgresSupport = await getEmbeddedPostgresTestSupport();
const describeEmbeddedPostgres = embeddedPostgresSupport.supported
? describe
: describe.skip;
const fakeCodexAppServer = resolve(
import.meta.dirname,
"../../../../packages/paperclip-runner/runner/target/debug/fake-codex-app-server",
);
const binariesAvailable =
existsSync(defaultCapabilityRunnerdBinary()) && existsSync(fakeCodexAppServer);
const realProcessIt = binariesAvailable ? it : it.skip;
async function closeServer(server: Server | null): Promise<void> {
if (!server) return;
server.closeAllConnections();
if (!server.listening) return;
await new Promise<void>((resolveClose) => server.close(() => resolveClose()));
}
function processAlive(pid: number | null): boolean {
if (!pid) return false;
try {
process.kill(pid, 0);
return true;
} catch {
return false;
}
}
async function waitForCondition(
description: string,
condition: () => boolean | Promise<boolean>,
timeoutMs = 10_000,
): Promise<void> {
const deadline = Date.now() + timeoutMs;
while (Date.now() < deadline) {
if (await condition()) return;
await new Promise<void>((resolveWait) => setTimeout(resolveWait, 25));
}
throw new Error(`Timed out waiting for ${description}`);
}
async function stopOwnedProcessGroup(
pid: number | null,
expectedStateRoot: string,
): Promise<void> {
if (!pid || !processAlive(pid)) return;
const command = await import("node:child_process").then(
({ execFileSync }) =>
execFileSync("ps", ["-p", String(pid), "-o", "command="], {
encoding: "utf8",
}).trim(),
);
if (!command.includes(expectedStateRoot)) {
throw new Error(`Refusing to signal process ${pid}; ownership changed`);
}
if (process.platform !== "win32") process.kill(-pid, "SIGKILL");
else process.kill(pid, "SIGKILL");
await waitForCondition(`owned process ${pid} to exit`, () => !processAlive(pid));
}
describeEmbeddedPostgres("native runner restart recovery with real processes", () => {
let temporary: Awaited<ReturnType<typeof startEmbeddedPostgresTestDatabase>>;
let runtimeRoot: string;
let paperclipHome: string;
let server: Server | null = null;
let apiUrl: string;
let originalPaperclipHome: string | undefined;
const companyId = randomUUID();
const agentId = randomUUID();
let successor: NativeControllerIdentity;
beforeAll(async () => {
temporary = await startEmbeddedPostgresTestDatabase(
"native-runner-restart-recovery-",
);
runtimeRoot = await mkdtemp(resolve(tmpdir(), "native-restart-runtime-"));
paperclipHome = await mkdtemp(resolve(tmpdir(), "native-restart-home-"));
originalPaperclipHome = process.env.PAPERCLIP_HOME;
process.env.PAPERCLIP_HOME = paperclipHome;
const controllerStartedAt = await readProcessStartedAt(process.pid);
successor = {
bootId: randomUUID(),
pid: process.pid,
processStartedAt: controllerStartedAt
? new Date(controllerStartedAt)
: new Date(),
};
server = createServer();
await new Promise<void>((resolveListen) =>
server!.listen(0, "127.0.0.1", resolveListen),
);
const address = server.address();
if (!address || typeof address === "string") {
throw new Error("Expected restart recovery TCP listener");
}
apiUrl = `http://127.0.0.1:${address.port}`;
setupRunnerPrpWebSocketServer(server, { apiUrl });
const db = createDb(temporary.connectionString);
await db.insert(companies).values({
id: companyId,
name: "Native restart recovery",
issuePrefix: "NRR",
requireBoardApprovalForNewAgents: false,
});
await db.insert(agents).values({
id: agentId,
companyId,
name: "Native restart runner",
role: "engineer",
status: "active",
adapterType: "paperclip_runner",
adapterConfig: { provider: "codex" },
runtimeConfig: {},
permissions: {},
});
});
afterAll(async () => {
runnerPrpWebSocketInternals.resetForTests();
await closeServer(server);
await temporary.cleanup();
await Promise.all([
rm(runtimeRoot, { recursive: true, force: true }),
rm(paperclipHome, { recursive: true, force: true }),
]);
if (originalPaperclipHome === undefined) delete process.env.PAPERCLIP_HOME;
else process.env.PAPERCLIP_HOME = originalPaperclipHome;
});
async function seedRun(
label: string,
existingDb?: ReturnType<typeof createDb>,
) {
const db = existingDb ?? createDb(temporary.connectionString);
const issueId = randomUUID();
const runId = randomUUID();
await db.insert(issues).values({
id: issueId,
companyId,
identifier: `NRR-${label}`,
title: `Recover ${label}`,
status: "in_progress",
priority: "medium",
workMode: "standard",
assigneeAgentId: agentId,
});
const [run] = await db
.insert(heartbeatRuns)
.values({
id: runId,
companyId,
agentId,
status: "running",
runtimeMode: "native",
nativeIssueId: issueId,
invocationSource: "assignment",
triggerDetail: "system",
contextSnapshot: { issueId },
})
.returning();
if (!run) throw new Error("Failed to seed native restart run");
await db
.update(issues)
.set({ executionRunId: runId })
.where(eq(issues.id, issueId));
const native = await prepareNativeHeartbeatRun({
db,
run,
issue: {
id: issueId,
title: `Recover ${label}`,
description: null,
reviewPolicy: null,
},
environmentLeaseId: `lease-${label}`,
});
await db.insert(nativeRunFinalizations).values({
runId,
companyId,
issueId,
phase: "observed",
});
return { db, issueId, runId, native };
}
function transportOptions(
fixture: Awaited<ReturnType<typeof seedRun>>,
stateDirectory: string,
) {
return {
runnerBinary: defaultCapabilityRunnerdBinary(),
codexCommand: fakeCodexAppServer,
codexArgs: [
"--state-file",
resolve(stateDirectory, "fake-codex-state.json"),
"--call-log",
resolve(stateDirectory, "fake-codex-calls.log"),
],
stateDirectory,
lifecyclePolicy: { mode: "warm" as const, idleTimeoutMs: 60_000 },
prpIdentity: {
runnerInstanceId: fixture.native.runnerInstanceId,
environmentLeaseId: fixture.native.environmentLeaseId,
runId: fixture.runId,
normalizedSessionId: fixture.native.normalizedSessionId,
turnId: fixture.native.turnId,
itemId: fixture.native.itemId,
},
controlPlaneRegistration: (authority: Parameters<typeof registerRunnerPrpAuthority>[0]["authority"]) =>
registerRunnerPrpAuthority({
companyId,
runId: fixture.runId,
authority,
}),
};
}
async function persistRestartEvidence(input: {
fixture: Awaited<ReturnType<typeof seedRun>>;
runnerPid: number;
processGroupId: number | null;
providerPid: number | null;
providerSessionId: string;
}) {
const startedAt = await readProcessStartedAt(input.runnerPid);
if (!startedAt) throw new Error("Runner process start fingerprint unavailable");
const providerStartedAt = input.providerPid
? await readProcessStartedAt(input.providerPid)
: null;
const now = new Date();
await input.fixture.db
.update(heartbeatRuns)
.set({
processPid: input.runnerPid,
processGroupId: input.processGroupId,
processStartedAt: new Date(startedAt),
runnerProfileJson: {
sessionCheckpoint: {
sessionId: input.fixture.native.normalizedSessionId,
identity: {
companyId,
issueId: input.fixture.issueId,
runId: input.fixture.runId,
agentId,
sessionId: input.fixture.native.normalizedSessionId,
},
providerSessionId: input.providerSessionId,
providerIdentity: {
kind: "codex",
providerSessionId: input.providerSessionId,
},
process: {
runnerPid: input.runnerPid,
runnerProcessGroupId: input.processGroupId,
providerPid: input.providerPid,
providerProcessStartedAt: providerStartedAt,
codexPid: input.providerPid,
codexProcessStartedAt: providerStartedAt,
},
},
},
updatedAt: now,
})
.where(eq(heartbeatRuns.id, input.fixture.runId));
await input.fixture.db
.update(nativeRunFinalizations)
.set({
leaseOwner: "dead-controller",
leaseExpiresAt: new Date(now.getTime() + 60_000),
controllerBootId: "dead-controller-boot",
controllerPid: 2_000_000_000,
controllerProcessStartedAt: new Date("2026-09-04T11:00:00.000Z"),
controllerGeneration: 1,
updatedAt: now,
})
.where(eq(nativeRunFinalizations.runId, input.fixture.runId));
return new Date(startedAt);
}
async function persistUnconnectedRunner(input: {
fixture: Awaited<ReturnType<typeof seedRun>>;
runnerPid: number;
processGroupId: number | null;
}) {
const startedAt = await readProcessStartedAt(input.runnerPid);
if (!startedAt) throw new Error("Runner process start fingerprint unavailable");
const now = new Date();
await input.fixture.db
.update(heartbeatRuns)
.set({
processPid: input.runnerPid,
processGroupId: input.processGroupId,
processStartedAt: new Date(startedAt),
updatedAt: now,
})
.where(eq(heartbeatRuns.id, input.fixture.runId));
await input.fixture.db
.update(nativeRunFinalizations)
.set({
leaseOwner: "dead-controller",
leaseExpiresAt: new Date(now.getTime() + 60_000),
controllerBootId: "dead-controller-boot",
controllerPid: 2_000_000_003,
controllerProcessStartedAt: new Date("2026-09-04T11:00:00.000Z"),
controllerGeneration: 1,
updatedAt: now,
})
.where(eq(nativeRunFinalizations.runId, input.fixture.runId));
}
realProcessIt("adopts one active runner across hot and hard controller restarts without duplicating steering", async () => {
const fixture = await seedRun("LIVE");
const stateDirectory = resolve(runtimeRoot, fixture.runId);
const baseOptions = transportOptions(fixture, stateDirectory);
const options = {
...baseOptions,
codexArgs: [...baseOptions.codexArgs, "--linger-after-turn-start"],
};
const first = createRunnerdCodexTransport(options);
let runnerPid: number | null = null;
let adopted: ReturnType<typeof createRunnerdCodexTransport> | null = null;
let adoptedAfterHardRestart:
| ReturnType<typeof createRunnerdCodexTransport>
| null = null;
try {
const started = await first.transport.request("thread/start", {
cwd: tmpdir(),
dynamicTools: [],
});
const thread = started.thread as Record<string, unknown>;
const turnStarted = await first.transport.request("turn/start", {
input: [{ type: "text", text: "Hold this turn across restarts." }],
});
const providerTurnId = String(
(turnStarted.turn as Record<string, unknown>).id,
);
await first.transport.request("turn/steer", {
input: [{ type: "text", text: "before hot restart" }],
expectedTurnId: providerTurnId,
correlationId: "steer-before-hot",
});
runnerPid = first.evidence().runnerPid;
if (!runnerPid) throw new Error("Real runner PID was not observed");
await persistRestartEvidence({
fixture,
runnerPid,
processGroupId: first.evidence().runnerProcessGroupId,
providerPid: first.evidence().providerPid,
providerSessionId: String(thread.id),
});
await first.detachControllerForRestart();
const [claim] = await claimNativeRestartRecoveries({
db: fixture.db,
controller: successor,
restartKind: "hot",
recoveryRequestId: "hot-restart-request",
now: new Date(),
runIds: [fixture.runId],
});
expect(claim).toMatchObject({
kind: "reattach_existing_runner",
runId: fixture.runId,
controllerGeneration: 2,
providerAttempt: 0,
process: { pid: runnerPid },
});
if (!claim || claim.kind !== "reattach_existing_runner") {
throw new Error("Expected live-runner recovery claim");
}
const duplicateLauncher = vi.fn(() => {
throw new Error("duplicate runner spawn attempted");
});
adopted = createRunnerdCodexTransport({
...options,
resumeDynamicTools: [],
runnerProcessLauncher: duplicateLauncher,
adoptExistingRunner: {
...claim.process,
isAlive: () => processAlive(runnerPid),
},
});
const restored = await adopted.transport.request("thread/read", {});
expect(restored.thread).toMatchObject({
id: thread.id,
turns: [{ id: providerTurnId, status: "inProgress" }],
});
expect(adopted.evidence().runnerPid).toBe(runnerPid);
expect(duplicateLauncher).not.toHaveBeenCalled();
await adopted.transport.request("turn/steer", {
input: [{ type: "text", text: "after hot restart" }],
expectedTurnId: providerTurnId,
correlationId: "steer-after-hot",
});
await adopted.detachControllerForRestart();
const hardRestartController: NativeControllerIdentity = {
...successor,
bootId: randomUUID(),
};
await fixture.db
.update(nativeRunFinalizations)
.set({
controllerPid: 2_000_000_001,
controllerProcessStartedAt: new Date(
"2026-09-04T11:30:00.000Z",
),
})
.where(eq(nativeRunFinalizations.runId, fixture.runId));
const [hardClaim] = await claimNativeRestartRecoveries({
db: fixture.db,
controller: hardRestartController,
restartKind: "hard",
now: new Date(),
runIds: [fixture.runId],
});
expect(hardClaim).toMatchObject({
kind: "reattach_existing_runner",
runId: fixture.runId,
controllerGeneration: 3,
providerAttempt: 0,
process: { pid: runnerPid },
});
if (!hardClaim || hardClaim.kind !== "reattach_existing_runner") {
throw new Error("Expected second live-runner recovery claim");
}
const secondDuplicateLauncher = vi.fn(() => {
throw new Error("duplicate runner spawn attempted after hard restart");
});
adoptedAfterHardRestart = createRunnerdCodexTransport({
...options,
resumeDynamicTools: [],
runnerProcessLauncher: secondDuplicateLauncher,
adoptExistingRunner: {
...hardClaim.process,
isAlive: () => processAlive(runnerPid),
},
});
await expect(
adoptedAfterHardRestart.transport.request("thread/read", {}),
).resolves.toMatchObject({
thread: {
id: thread.id,
turns: [{ id: providerTurnId, status: "inProgress" }],
},
});
await adoptedAfterHardRestart.transport.request("turn/steer", {
input: [{ type: "text", text: "after hard restart" }],
expectedTurnId: providerTurnId,
correlationId: "steer-after-hard",
});
expect(adoptedAfterHardRestart.evidence().runnerPid).toBe(runnerPid);
expect(secondDuplicateLauncher).not.toHaveBeenCalled();
const providerCalls = await readFile(
resolve(stateDirectory, "fake-codex-calls.log"),
"utf8",
);
expect(providerCalls.match(/^turn\/start$/gm)).toHaveLength(1);
expect(providerCalls.match(/^turn\/steer$/gm)).toHaveLength(3);
expect(
await fixture.db
.select({ id: heartbeatRuns.id })
.from(heartbeatRuns)
.where(
and(
eq(heartbeatRuns.retryOfRunId, fixture.runId),
eq(heartbeatRuns.companyId, companyId),
),
),
).toHaveLength(0);
} finally {
await adoptedAfterHardRestart?.transport.close().catch(() => undefined);
await adopted?.transport.close().catch(() => undefined);
await stopOwnedProcessGroup(runnerPid, stateDirectory).catch(() => undefined);
await rm(stateDirectory, { recursive: true, force: true });
}
}, 45_000);
realProcessIt("hard-restarts a dead runner on the same run and provider session", async () => {
const fixture = await seedRun("DEAD");
const stateDirectory = resolve(runtimeRoot, fixture.runId);
const baseOptions = transportOptions(fixture, stateDirectory);
const options = {
...baseOptions,
codexArgs: [...baseOptions.codexArgs, "--linger-after-turn-start"],
};
const first = createRunnerdCodexTransport(options);
let firstRunnerPid: number | null = null;
let recoveredRunnerPid: number | null = null;
let restored: ReturnType<typeof createRunnerdCodexTransport> | null = null;
try {
const started = await first.transport.request("thread/start", {
cwd: tmpdir(),
dynamicTools: [],
});
const thread = started.thread as Record<string, unknown>;
const turnStarted = await first.transport.request("turn/start", {
input: [{ type: "text", text: "Resume this active turn after process loss." }],
});
const providerTurnId = String(
(turnStarted.turn as Record<string, unknown>).id,
);
firstRunnerPid = first.evidence().runnerPid;
if (!firstRunnerPid) throw new Error("Real runner PID was not observed");
const providerPid = first.evidence().providerPid;
await persistRestartEvidence({
fixture,
runnerPid: firstRunnerPid,
processGroupId: first.evidence().runnerProcessGroupId,
providerPid,
providerSessionId: String(thread.id),
});
await first.detachControllerForRestart();
await stopOwnedProcessGroup(firstRunnerPid, stateDirectory);
await waitForCondition(
"provider process to observe runner pipe closure",
() => !processAlive(providerPid),
);
const [claim] = await claimNativeRestartRecoveries({
db: fixture.db,
controller: successor,
restartKind: "hard",
now: new Date(),
runIds: [fixture.runId],
});
expect(claim).toMatchObject({
kind: "resume_dead_runner",
runId: fixture.runId,
controllerGeneration: 2,
providerAttempt: 0,
});
if (!claim || claim.kind !== "resume_dead_runner") {
throw new Error("Expected dead-runner recovery claim");
}
restored = createRunnerdCodexTransport({
...options,
resumeDynamicTools: [],
});
const recovered = await restored.transport.request("thread/read", {});
recoveredRunnerPid = restored.evidence().runnerPid;
expect(recovered.thread).toMatchObject({
id: thread.id,
turns: [{ id: providerTurnId, status: "inProgress" }],
});
expect(recoveredRunnerPid).toEqual(expect.any(Number));
expect(recoveredRunnerPid).not.toBe(firstRunnerPid);
const providerCalls = await readFile(
resolve(stateDirectory, "fake-codex-calls.log"),
"utf8",
);
expect(providerCalls.match(/^turn\/start$/gm)).toHaveLength(1);
expect(
await fixture.db
.select({ id: heartbeatRuns.id })
.from(heartbeatRuns)
.where(eq(heartbeatRuns.retryOfRunId, fixture.runId)),
).toHaveLength(0);
} finally {
await restored?.transport.close().catch(() => undefined);
await stopOwnedProcessGroup(recoveredRunnerPid, stateDirectory).catch(
() => undefined,
);
await stopOwnedProcessGroup(firstRunnerPid, stateDirectory).catch(
() => undefined,
);
await rm(stateDirectory, { recursive: true, force: true });
}
}, 45_000);
realProcessIt("quarantines a pre-auth runner root and bootstraps the same run", async () => {
const fixture = await seedRun("BOOTSTRAP");
const stateDirectory = resolve(runtimeRoot, fixture.runId);
const baseOptions = transportOptions(fixture, stateDirectory);
const transport = createRunnerdCodexTransport({
...baseOptions,
runnerReconnectGraceMs: 60_000,
controlPlaneRegistration: async () => ({
connectUrl: "ws://127.0.0.1:9/api/runner/prp/unreachable",
release: () => undefined,
}),
});
let runnerPid: number | null = null;
let replacementRunnerPid: number | null = null;
let replacement: ReturnType<typeof createRunnerdCodexTransport> | null =
null;
const quarantineDirectory = `${stateDirectory}.bootstrap-incomplete`;
const starting = transport.transport
.request("thread/start", {
cwd: tmpdir(),
dynamicTools: [],
})
.then(
() => null,
(error: unknown) => error,
);
try {
await waitForCondition("pre-auth runner spawn", () => {
runnerPid = transport.evidence().runnerPid;
return processAlive(runnerPid);
});
if (!runnerPid) throw new Error("Pre-auth runner PID was not observed");
await persistUnconnectedRunner({
fixture,
runnerPid,
processGroupId: transport.evidence().runnerProcessGroupId,
});
await stopOwnedProcessGroup(runnerPid, stateDirectory);
await expect(starting).resolves.toBeInstanceOf(Error);
const durable = JSON.parse(
await readFile(
resolve(stateDirectory, "control-plane", "control-plane-state.json"),
"utf8",
),
) as Record<string, unknown>;
expect(durable).toMatchObject({
connectionCount: 0,
committedEvents: [],
});
const [claim] = await claimNativeRestartRecoveries({
db: fixture.db,
controller: successor,
restartKind: "hard",
now: new Date(),
runIds: [fixture.runId],
});
expect(claim).toMatchObject({
kind: "bootstrap_incomplete",
runId: fixture.runId,
controllerGeneration: 2,
providerAttempt: 0,
});
if (!claim || claim.kind !== "bootstrap_incomplete") {
throw new Error("Expected incomplete-bootstrap recovery claim");
}
await transport.transport.close();
await rename(stateDirectory, quarantineDirectory);
replacement = createRunnerdCodexTransport(baseOptions);
const restarted = await replacement.transport.request("thread/start", {
cwd: tmpdir(),
dynamicTools: [],
});
replacementRunnerPid = replacement.evidence().runnerPid;
expect(restarted.thread).toMatchObject({ id: expect.any(String) });
expect(replacementRunnerPid).toEqual(expect.any(Number));
expect(replacementRunnerPid).not.toBe(runnerPid);
expect(processAlive(replacementRunnerPid)).toBe(true);
expect(existsSync(quarantineDirectory)).toBe(true);
expect(existsSync(stateDirectory)).toBe(true);
const replacementDurable = JSON.parse(
await readFile(
resolve(stateDirectory, "control-plane", "control-plane-state.json"),
"utf8",
),
) as Record<string, unknown>;
expect(replacementDurable).toMatchObject({
identity: {
runId: fixture.runId,
normalizedSessionId: fixture.native.normalizedSessionId,
runnerInstanceId: fixture.native.runnerInstanceId,
environmentLeaseId: fixture.native.environmentLeaseId,
},
});
await expect(
fixture.db
.select({ id: heartbeatRuns.id })
.from(heartbeatRuns)
.where(eq(heartbeatRuns.retryOfRunId, fixture.runId)),
).resolves.toHaveLength(0);
} finally {
await replacement?.transport.close().catch(() => undefined);
await transport.transport.close().catch(() => undefined);
await starting;
await stopOwnedProcessGroup(replacementRunnerPid, stateDirectory).catch(
() => undefined,
);
await stopOwnedProcessGroup(runnerPid, stateDirectory).catch(
() => undefined,
);
await rm(stateDirectory, { recursive: true, force: true });
await rm(quarantineDirectory, { recursive: true, force: true });
}
}, 45_000);
it("prioritizes an already-proposed durable result over runner recovery", async () => {
const fixture = await seedRun("RESULT");
const [persistedRun] = await fixture.db
.select({ completionContractId: heartbeatRuns.completionContractId })
.from(heartbeatRuns)
.where(eq(heartbeatRuns.id, fixture.runId));
if (!persistedRun?.completionContractId) {
throw new Error("Prepared native run has no completion contract");
}
const resultId = randomUUID();
await fixture.db.insert(nativeRunResults).values({
id: resultId,
companyId,
issueId: fixture.issueId,
runId: fixture.runId,
turnId: fixture.native.turnId,
completionContractId: persistedRun.completionContractId,
callerResultId: "result-before-restart",
callerDedupeKey: "result-before-restart",
serverFingerprint: "result-before-restart",
schemaStatus: "accepted",
resultJson: {
schema: "paperclip.run_result.v1",
reportedWorkDisposition: "done",
summary: "Durable result proposed before restart.",
},
canonicalSha256: "a".repeat(64),
});
await fixture.db
.update(nativeRunFinalizations)
.set({
phase: "result_persisted",
resultId,
leaseOwner: null,
leaseExpiresAt: null,
})
.where(eq(nativeRunFinalizations.runId, fixture.runId));
await expect(
claimNativeRestartRecoveries({
db: fixture.db,
controller: successor,
restartKind: "hot",
recoveryRequestId: "result-race-restart",
now: new Date(),
runIds: [fixture.runId],
}),
).resolves.toEqual([]);
await expect(
fixture.db
.select({ id: nativeRunResults.id })
.from(nativeRunResults)
.where(eq(nativeRunResults.runId, fixture.runId)),
).resolves.toEqual([{ id: resultId }]);
await expect(
fixture.db
.select({ id: heartbeatRuns.id })
.from(heartbeatRuns)
.where(eq(heartbeatRuns.retryOfRunId, fixture.runId)),
).resolves.toHaveLength(0);
});
it("preserves a future same-run provider recovery wake instead of consuming it during startup", async () => {
const fixture = await seedRun("SCHEDULED");
const now = new Date();
const nextAttemptAt = new Date(now.getTime() + 60_000);
await fixture.db
.update(nativeRunFinalizations)
.set({
phase: "retryable_failure",
attempt: 1,
nextAttemptAt,
recoveryState: "resuming_session",
leaseOwner: null,
leaseExpiresAt: null,
})
.where(eq(nativeRunFinalizations.runId, fixture.runId));
await expect(
claimNativeRestartRecoveries({
db: fixture.db,
controller: successor,
restartKind: "hard",
now,
runIds: [fixture.runId],
}),
).resolves.toEqual([]);
await expect(
fixture.db
.select({
phase: nativeRunFinalizations.phase,
attempt: nativeRunFinalizations.attempt,
nextAttemptAt: nativeRunFinalizations.nextAttemptAt,
recoveryState: nativeRunFinalizations.recoveryState,
})
.from(nativeRunFinalizations)
.where(eq(nativeRunFinalizations.runId, fixture.runId)),
).resolves.toEqual([
{
phase: "retryable_failure",
attempt: 1,
nextAttemptAt,
recoveryState: "resuming_session",
},
]);
});
it("fails closed for a live recycled PID without signalling or spawning", async () => {
const fixture = await seedRun("MISMATCH");
const stateDirectory = resolve(runtimeRoot, fixture.runId);
const child = spawn(
process.execPath,
["-e", "setInterval(() => {}, 1000)", stateDirectory],
{
detached: process.platform !== "win32",
stdio: "ignore",
},
);
child.unref();
const pid = child.pid ?? null;
if (!pid) throw new Error("Failed to launch mismatch sentinel");
try {
await waitForCondition("mismatch sentinel startup", () => processAlive(pid));
const observedStart = await readProcessStartedAt(pid);
if (!observedStart) throw new Error("Sentinel fingerprint unavailable");
await fixture.db
.update(heartbeatRuns)
.set({
processPid: pid,
processGroupId: process.platform === "win32" ? null : pid,
processStartedAt: new Date(
new Date(observedStart).getTime() - 60_000,
),
})
.where(eq(heartbeatRuns.id, fixture.runId));
const [disposition] = await claimNativeRestartRecoveries({
db: fixture.db,
controller: successor,
restartKind: "hard",
now: new Date(),
runIds: [fixture.runId],
});
expect(disposition).toEqual({
kind: "blocked",
runId: fixture.runId,
reason: "live_runner_identity_mismatch",
});
expect(processAlive(pid)).toBe(true);
expect(existsSync(stateDirectory)).toBe(false);
await expect(
fixture.db
.select({ recoveryState: nativeRunFinalizations.recoveryState })
.from(nativeRunFinalizations)
.where(eq(nativeRunFinalizations.runId, fixture.runId)),
).resolves.toEqual([{ recoveryState: "blocked" }]);
await expect(
fixture.db
.select({ id: heartbeatRuns.id })
.from(heartbeatRuns)
.where(eq(heartbeatRuns.retryOfRunId, fixture.runId)),
).resolves.toHaveLength(0);
} finally {
await stopOwnedProcessGroup(pid, stateDirectory).catch(() => undefined);
}
});
it("allows only one of two concurrent successor controllers to claim a dead session", async () => {
const fixture = await seedRun("CONCURRENT");
const now = new Date();
await fixture.db
.update(heartbeatRuns)
.set({
runnerProfileJson: {
sessionCheckpoint: {
sessionId: fixture.native.normalizedSessionId,
identity: {
companyId,
issueId: fixture.issueId,
runId: fixture.runId,
agentId,
sessionId: fixture.native.normalizedSessionId,
},
providerSessionId: "provider-concurrent",
providerIdentity: {
kind: "codex",
providerSessionId: "provider-concurrent",
},
},
},
})
.where(eq(heartbeatRuns.id, fixture.runId));
await fixture.db
.update(nativeRunFinalizations)
.set({
leaseOwner: "dead-controller",
leaseExpiresAt: new Date(now.getTime() + 60_000),
controllerBootId: "dead-controller-boot",
controllerPid: 2_000_000_002,
controllerProcessStartedAt: new Date("2026-09-04T10:00:00.000Z"),
controllerGeneration: 4,
})
.where(eq(nativeRunFinalizations.runId, fixture.runId));
const controllers = [
successor,
{ ...successor, bootId: randomUUID() },
];
const dispositions = (
await Promise.all(
controllers.map((controller) =>
claimNativeRestartRecoveries({
db: fixture.db,
controller,
restartKind: "hard",
now,
runIds: [fixture.runId],
}),
),
)
).flat();
expect(
dispositions.filter((entry) => entry.kind === "resume_dead_runner"),
).toHaveLength(1);
expect(
dispositions.filter((entry) => entry.kind === "awaiting_evidence"),
).toHaveLength(1);
await expect(
fixture.db
.select({
controllerGeneration: nativeRunFinalizations.controllerGeneration,
providerAttempt: nativeRunFinalizations.attempt,
})
.from(nativeRunFinalizations)
.where(eq(nativeRunFinalizations.runId, fixture.runId)),
).resolves.toEqual([
{ controllerGeneration: 5, providerAttempt: 0 },
]);
});
it("classifies every requested recovery candidate without an implicit 100-run cap", async () => {
const db = createDb(temporary.connectionString);
const fixtures: Array<Awaited<ReturnType<typeof seedRun>>> = [];
for (let index = 0; index < 101; index += 1) {
fixtures.push(
await seedRun(`BULK-${String(index).padStart(3, "0")}`, db),
);
}
const dispositions = await claimNativeRestartRecoveries({
db,
controller: { ...successor, bootId: randomUUID() },
restartKind: "hard",
now: new Date(),
runIds: fixtures.map((fixture) => fixture.runId),
});
expect(dispositions).toHaveLength(101);
expect(
dispositions.every(
(disposition) => disposition.kind === "bootstrap_incomplete",
),
).toBe(true);
}, 30_000);
});

View File

@ -166,6 +166,7 @@ import {
runtimeQuestionFallbackFromEvent,
resolveNativeRuntimeRequest,
resolveNativeHarnessPersistenceProfile,
runnerdStateProvesIncompleteBootstrap,
semanticProviderPlanMarkdown,
sha256DirectoryTree,
stageRemoteRunnerDirectory,
@ -175,6 +176,46 @@ import {
shouldRestoreNativeHarnessBackupIntoSandbox,
} from "./native-session-executor.js";
describe("native incomplete-bootstrap evidence", () => {
it("requires zero connections, zero events, and only untouched bootstrap commands", async () => {
const root = await mkdtemp(join(tmpdir(), "native-bootstrap-evidence-"));
const controlPlaneRoot = join(root, "control-plane");
await mkdir(controlPlaneRoot, { recursive: true });
const statePath = join(controlPlaneRoot, "control-plane-state.json");
const base = {
schema: "paperclip.runner.durable.control-plane-state.v1",
connectionCount: 0,
committedEvents: [],
commands: [
{ type: "run.prepare", status: "pending" },
{ type: "session.open", status: "pending" },
],
};
try {
await writeFile(statePath, JSON.stringify(base));
expect(runnerdStateProvesIncompleteBootstrap(root)).toBe(true);
for (const ambiguous of [
{ ...base, connectionCount: 1 },
{ ...base, committedEvents: [{ eventType: "harness.ready" }] },
{
...base,
commands: [{ type: "session.open", status: "completed" }],
},
{
...base,
commands: [{ type: "turn.start", status: "pending" }],
},
]) {
await writeFile(statePath, JSON.stringify(ambiguous));
expect(runnerdStateProvesIncompleteBootstrap(root)).toBe(false);
}
} finally {
await rm(root, { recursive: true, force: true });
}
});
});
describe("native provider usage normalization", () => {
it("reads remote runner run-delta tokens and provider cost", () => {
const usage = {

View File

@ -83,6 +83,7 @@ import {
type NativeStatusDecision,
} from "./status-arbiter.js";
import { HttpError } from "../../errors.js";
import { redactSensitiveText } from "../../redaction.js";
import { resolvePaperclipRunnerBinary } from "./native-codex-runner.js";
import {
createNativeRunTrace,
@ -91,6 +92,13 @@ import {
type NativeRunTrace,
} from "./native-run-trace.js";
import { createNativeHarnessBackupStamp } from "./native-harness-backup-stamp.js";
import { readProcessStartedAt } from "../hot-restart.js";
import {
currentNativeControllerIdentity,
nextNativeProviderAttempt,
type NativeControllerIdentity,
type NativeRestartRecoveryClaim,
} from "./native-restart-recovery.js";
type ActiveNativeSession = {
session: NativeSession;
@ -154,6 +162,39 @@ const nativeRuntimeRequestResolutions = new Map<
NativeRuntimeRequestResolution
>();
async function verifiedRecoveryProcessIsAlive(input: {
pid: number;
startedAt: string;
}): Promise<boolean> {
try {
process.kill(input.pid, 0);
} catch (error) {
if ((error as NodeJS.ErrnoException | undefined)?.code !== "EPERM")
return false;
}
try {
const observed = await readProcessStartedAt(input.pid);
return (
observed !== null &&
new Date(observed).getTime() === new Date(input.startedAt).getTime()
);
} catch {
return false;
}
}
async function signalVerifiedRecoveryProcess(
input: { pid: number; startedAt: string },
signal: NodeJS.Signals,
): Promise<boolean> {
if (!(await verifiedRecoveryProcessIsAlive(input))) return false;
try {
return process.kill(input.pid, signal);
} catch {
return false;
}
}
function pruneNativeRuntimeRequestResolutionCache(): void {
const completed = [...nativeRuntimeRequestResolutions.entries()]
.filter(([, resolution]) => resolution.completedAt !== null)
@ -1378,6 +1419,7 @@ async function verifyPriorRunnerdStateForSessionScope(input: {
async function migrateRunnerdStateRootForExecution(input: {
db: Db;
execution: NativeExecutionInput;
restartRecovery?: NativeRestartRecoveryClaim;
}): Promise<void> {
const scoped = scopedRunnerdStateRoot(input.execution);
if (existsSync(scoped)) {
@ -1386,16 +1428,32 @@ async function migrateRunnerdStateRootForExecution(input: {
}
const identity = readRunnerdDurableIdentity(scoped);
if (!identity) {
quarantineRunnerdStateRoot(scoped, "identity_indeterminate");
if (input.restartRecovery?.kind !== "reattach_existing_runner") {
quarantineRunnerdStateRoot(scoped, "identity_indeterminate");
}
throw new Error("runner_state_identity_mismatch");
}
if (!durableIdentityMatchesSession(identity, input.execution)) {
quarantineRunnerdStateRoot(scoped, "identity_mismatch");
if (input.restartRecovery?.kind !== "reattach_existing_runner") {
quarantineRunnerdStateRoot(scoped, "identity_mismatch");
}
throw new Error("runner_state_identity_mismatch");
}
if (input.restartRecovery?.kind === "bootstrap_incomplete") {
if (runnerdStateProvesIncompleteBootstrap(scoped)) {
quarantineRunnerdStateRoot(scoped, "identity_indeterminate");
return;
}
// Database evidence alone cannot distinguish a never-connected runner
// from a partially-persisted provider bootstrap. Only the durable PRP
// root can authorize a fresh bootstrap; anything else stays fail-closed.
throw new Error("runner_state_identity_mismatch");
}
if (durableIdentityMatchesExecution(identity, input.execution)) {
if (runnerdAuthorityLifecycle(scoped, identity) === "indeterminate") {
quarantineRunnerdStateRoot(scoped, "identity_indeterminate");
if (input.restartRecovery?.kind !== "reattach_existing_runner") {
quarantineRunnerdStateRoot(scoped, "identity_indeterminate");
}
throw new Error("runner_state_identity_mismatch");
}
} else {
@ -1464,6 +1522,40 @@ async function migrateRunnerdStateRootForExecution(input: {
}
}
export function runnerdStateProvesIncompleteBootstrap(root: string): boolean {
try {
const statePath = resolve(root, "control-plane", "control-plane-state.json");
const state = record(
JSON.parse(
readBoundedNativeFile(
statePath,
NATIVE_DURABLE_IDENTITY_MAX_BYTES,
"runner_durable_identity_too_large",
).toString("utf8"),
),
);
const commands = Array.isArray(state.commands)
? state.commands.map(record)
: [];
const committedEvents = Array.isArray(state.committedEvents)
? state.committedEvents
: [];
const onlyUnconsumedBootstrapCommands = commands.every(
(command) =>
command.status === "pending" &&
(command.type === "run.prepare" || command.type === "session.open"),
);
return (
state.schema === RUNNERD_CONTROL_PLANE_STATE_SCHEMA &&
state.connectionCount === 0 &&
committedEvents.length === 0 &&
onlyUnconsumedBootstrapCommands
);
} catch {
return false;
}
}
function isSafeNativeStateDirectory(path: string): boolean {
if (!existsSync(path)) return false;
const stats = lstatSync(path);
@ -3079,8 +3171,10 @@ export async function renewNativeSessionExecutionLease(input: {
issueId: string;
leaseOwner: string;
attempt: number;
controller?: NativeControllerIdentity;
leaseTtlMs?: number;
}): Promise<void> {
const controller = input.controller ?? (await currentNativeControllerIdentity());
const leaseTtlMs = input.leaseTtlMs ?? NATIVE_SESSION_EXECUTION_LEASE_TTL_MS;
if (
!Number.isInteger(leaseTtlMs) ||
@ -3102,6 +3196,12 @@ export async function renewNativeSessionExecutionLease(input: {
eq(nativeRunFinalizations.issueId, input.issueId),
eq(nativeRunFinalizations.leaseOwner, input.leaseOwner),
eq(nativeRunFinalizations.attempt, input.attempt),
eq(nativeRunFinalizations.controllerBootId, controller.bootId),
eq(nativeRunFinalizations.controllerPid, controller.pid),
eq(
nativeRunFinalizations.controllerProcessStartedAt,
controller.processStartedAt,
),
gt(nativeRunFinalizations.leaseExpiresAt, sql`now()`),
),
)
@ -3116,6 +3216,7 @@ function startNativeSessionExecutionLeaseRenewal(input: {
issueId: string;
leaseOwner: string;
attempt: number;
controller: NativeControllerIdentity;
}): { stop: () => Promise<void> } {
let leaseLost: Error | null = null;
let renewal = Promise.resolve();
@ -3154,6 +3255,7 @@ export async function executePaperclipNativeSession(input: {
execution: NativeExecutionInput;
runnerInstanceId: string;
leaseOwner?: string;
restartRecovery?: NativeRestartRecoveryClaim;
onSpawn?: (meta: {
pid: number;
processGroupId: number | null;
@ -3286,6 +3388,7 @@ async function executePaperclipNativeSessionWithinScope(
await migrateRunnerdStateRootForExecution({
db: input.db,
execution: input.execution,
restartRecovery: input.restartRecovery,
});
}
const durableRunnerBinding = input.useRunnerd
@ -3310,6 +3413,7 @@ async function executePaperclipNativeSessionWithinScope(
}
const leaseOwner =
input.leaseOwner ?? `${effectiveRunnerInstanceId}:${randomUUID()}`;
const controller = await currentNativeControllerIdentity();
const leaseNow = new Date();
const leaseExpiresAt = new Date(leaseNow.getTime() + 20 * 60_000);
let attempt: number;
@ -3400,13 +3504,39 @@ async function executePaperclipNativeSessionWithinScope(
coordinator.leaseExpiresAt > leaseNow
)
throw new Error("native_finalization_lease_busy");
const recovering = input.restartRecovery;
if (
recovering &&
(recovering.runId !== coordinator.runId ||
recovering.leaseOwner !== leaseOwner ||
coordinator.leaseOwner !== leaseOwner ||
coordinator.controllerBootId !== controller.bootId ||
coordinator.controllerPid !== controller.pid ||
coordinator.controllerGeneration !==
recovering.controllerGeneration)
) {
throw new Error("native_restart_recovery_claim_changed");
}
const nextAttempt = nextNativeProviderAttempt(
coordinator.attempt,
recovering?.kind,
);
const nextControllerGeneration = recovering
? recovering.controllerGeneration
: coordinator.controllerBootId === controller.bootId
? Math.max(1, coordinator.controllerGeneration)
: coordinator.controllerGeneration + 1;
const claimed = await tx
.update(nativeRunFinalizations)
.set({
phase: "observed",
attempt: coordinator.attempt + 1,
attempt: nextAttempt,
leaseOwner,
leaseExpiresAt,
controllerBootId: controller.bootId,
controllerPid: controller.pid,
controllerProcessStartedAt: controller.processStartedAt,
controllerGeneration: nextControllerGeneration,
failureCode: null,
failureDetail: null,
nextAttemptAt: null,
@ -3438,7 +3568,7 @@ async function executePaperclipNativeSessionWithinScope(
updatedAt: leaseNow,
})
.where(eq(heartbeatRuns.id, coordinator.runId));
return coordinator.attempt + 1;
return nextAttempt;
}),
{ parentName: "task.prepare" },
);
@ -3859,6 +3989,7 @@ async function executePaperclipNativeSessionWithinScope(
issueId: input.execution.binding.issueId,
leaseOwner,
attempt,
controller,
});
try {
const runnerdBackend =
@ -4085,6 +4216,7 @@ async function executePaperclipNativeSessionWithinScope(
error instanceof Error
? error.message.slice(0, 2_000)
: String(error).slice(0, 2_000);
const sanitizedStderrTail = redactSensitiveText(message).slice(-4_096);
await input.db.transaction(async (tx) => {
const updated = await tx
.update(nativeRunFinalizations)
@ -4092,6 +4224,8 @@ async function executePaperclipNativeSessionWithinScope(
phase,
leaseOwner: null,
leaseExpiresAt: null,
recoveryState:
phase === "retryable_failure" ? "resuming_session" : "blocked",
failureCode,
failureDetail: {
message,
@ -4114,6 +4248,44 @@ async function executePaperclipNativeSessionWithinScope(
: "Resume this same run from its exact persisted native provider checkpoint after the retry delay.",
},
nextAttemptAt,
recoveryHistory: sql`(
select coalesce(jsonb_agg(item order by ordinal), '[]'::jsonb)
from jsonb_array_elements(
coalesce(${nativeRunFinalizations.recoveryHistory}, '[]'::jsonb)
|| jsonb_build_array(${JSON.stringify({
at: now.toISOString(),
disposition: phase,
reason: sourceFailureCode,
controllerBootId: controller.bootId,
controllerGeneration:
input.restartRecovery?.controllerGeneration ?? null,
providerAttempt: attempt,
stderrTail: sanitizedStderrTail,
providerSessionEstablished:
recoveryEvidence.providerSessionEstablished,
checkpointExists: recoveryEvidence.checkpointExists,
})}::jsonb)
) with ordinality as history(item, ordinal)
where ordinal > greatest(
jsonb_array_length(
coalesce(${nativeRunFinalizations.recoveryHistory}, '[]'::jsonb)
|| jsonb_build_array(${JSON.stringify({
at: now.toISOString(),
disposition: phase,
reason: sourceFailureCode,
controllerBootId: controller.bootId,
controllerGeneration:
input.restartRecovery?.controllerGeneration ?? null,
providerAttempt: attempt,
stderrTail: sanitizedStderrTail,
providerSessionEstablished:
recoveryEvidence.providerSessionEstablished,
checkpointExists: recoveryEvidence.checkpointExists,
})}::jsonb)
) - 20,
0
)
)`,
updatedAt: now,
})
.where(
@ -4126,6 +4298,12 @@ async function executePaperclipNativeSessionWithinScope(
eq(nativeRunFinalizations.issueId, input.execution.binding.issueId),
eq(nativeRunFinalizations.leaseOwner, leaseOwner),
eq(nativeRunFinalizations.attempt, attempt),
eq(nativeRunFinalizations.controllerBootId, controller.bootId),
eq(nativeRunFinalizations.controllerPid, controller.pid),
eq(
nativeRunFinalizations.controllerProcessStartedAt,
controller.processStartedAt,
),
gt(nativeRunFinalizations.leaseExpiresAt, sql`now()`),
),
)
@ -4227,6 +4405,8 @@ async function executePaperclipNativeSessionWithinScope(
.set({
leaseOwner: null,
leaseExpiresAt: null,
recoveryState: null,
recoveryRequestId: null,
updatedAt: releaseNow,
})
.where(
@ -4236,6 +4416,12 @@ async function executePaperclipNativeSessionWithinScope(
eq(nativeRunFinalizations.issueId, input.execution.binding.issueId),
eq(nativeRunFinalizations.leaseOwner, leaseOwner),
eq(nativeRunFinalizations.attempt, attempt),
eq(nativeRunFinalizations.controllerBootId, controller.bootId),
eq(nativeRunFinalizations.controllerPid, controller.pid),
eq(
nativeRunFinalizations.controllerProcessStartedAt,
controller.processStartedAt,
),
gt(nativeRunFinalizations.leaseExpiresAt, sql`now()`),
),
)
@ -5184,6 +5370,7 @@ export async function createRunnerdBackend(input: {
db: Db;
execution: NativeExecutionInput;
runnerInstanceId: string;
restartRecovery?: NativeRestartRecoveryClaim;
durableEnvironmentLeaseId?: string;
onSpawn?: (meta: {
pid: number;
@ -5225,6 +5412,7 @@ export async function createRunnerdBackend(input: {
await migrateRunnerdStateRootForExecution({
db: input.db,
execution: input.execution,
restartRecovery: input.restartRecovery,
});
return await createRunnerdBackendWithinSessionClaim(input, sessionScopeId);
} finally {
@ -6557,6 +6745,11 @@ async function createRunnerdBackendWithinSessionClaim(
}
remotePrepared = false;
};
const adoptedProcess =
target.kind === "local" &&
input.restartRecovery?.kind === "reattach_existing_runner"
? input.restartRecovery.process
: null;
const backend = createNativeSessionBackend(runnerExecution, {
runnerInstanceId: input.runnerInstanceId,
environment: effectiveRunnerEnvironment,
@ -6693,6 +6886,14 @@ async function createRunnerdBackendWithinSessionClaim(
: undefined,
runnerProcessLauncher: remoteProcessLauncher,
runnerReconnectGraceMs: remoteTarget ? 120_000 : undefined,
adoptExistingRunner: adoptedProcess
? {
...adoptedProcess,
isAlive: () => verifiedRecoveryProcessIsAlive(adoptedProcess),
signal: (signal) =>
signalVerifiedRecoveryProcess(adoptedProcess, signal),
}
: undefined,
environment: effectiveRunnerEnvironment,
lifecyclePolicy: input.execution.session.lifecyclePolicy,
runtimeContext:

View File

@ -1,7 +1,10 @@
import { describe, expect, it } from "vitest";
import { canonicalNativeRuntimeContextDigest } from "../../vendor/paperclip-runner/index.js";
import { buildNativeExecutionInput } from "./native-execution-input.js";
import { rebindNativeSessionCheckpoint } from "./native-session-resume.js";
import {
isUnusedLegacyNativeRetryReplacement,
rebindNativeSessionCheckpoint,
} from "./native-session-resume.js";
import { nativeRuntimeContextFixture } from "./runtime-context.test-fixture.js";
const companyId = "10000000-0000-4000-8000-000000000001";
@ -87,6 +90,51 @@ function previousRun(overrides: Record<string, unknown> = {}) {
}
describe("rebindNativeSessionCheckpoint", () => {
it("permits legacy retry rebinding only before the replacement acquired authority", () => {
const source = {
runtimeMode: "native",
status: "interrupted",
nativeSessionId: normalizedSessionId,
};
const replacement = {
processPid: null,
processGroupId: null,
processStartedAt: null,
runnerProfileJson: {},
};
expect(
isUnusedLegacyNativeRetryReplacement({
source,
replacement,
hasProviderEvents: false,
}),
).toBe(true);
expect(
isUnusedLegacyNativeRetryReplacement({
source,
replacement: { ...replacement, processPid: 123 },
hasProviderEvents: false,
}),
).toBe(false);
expect(
isUnusedLegacyNativeRetryReplacement({
source,
replacement,
hasProviderEvents: true,
}),
).toBe(false);
expect(
isUnusedLegacyNativeRetryReplacement({
source,
replacement: {
...replacement,
runnerProfileJson: { sessionCheckpoint: { providerSessionId: "claimed" } },
},
hasProviderEvents: false,
}),
).toBe(false);
});
it("retains provider identity but clears prior turn and event state", () => {
const rebound = rebindNativeSessionCheckpoint({
previousRun: previousRun(),

View File

@ -15,6 +15,48 @@ export function isNativeSessionId(value: unknown): value is string {
&& /^[0-9a-f]{8}-[0-9a-f]{4}-[1-5][0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$/i.test(value);
}
const LEGACY_RETRY_SOURCE_TERMINAL_STATUSES = new Set([
"succeeded",
"interrupted",
"failed",
"cancelled",
"timed_out",
]);
/**
* Legacy compatibility is deliberately narrower than ordinary task-session
* continuation: only a replacement row that never acquired any native process
* or provider authority may be rebound to a terminal native source. The exact
* checkpoint/session/workspace/provider binding is validated separately by
* rebindNativeSessionCheckpoint.
*/
export function isUnusedLegacyNativeRetryReplacement(input: {
replacement: {
processPid: number | null;
processGroupId: number | null;
processStartedAt: Date | null;
runnerProfileJson: unknown;
};
source: {
runtimeMode: string | null;
status: string;
nativeSessionId: string | null;
} | null;
hasProviderEvents: boolean;
}): boolean {
const replacementProfile = record(input.replacement.runnerProfileJson);
return Boolean(
input.source?.runtimeMode === "native" &&
LEGACY_RETRY_SOURCE_TERMINAL_STATUSES.has(input.source.status) &&
isNativeSessionId(input.source.nativeSessionId) &&
input.replacement.processPid === null &&
input.replacement.processGroupId === null &&
input.replacement.processStartedAt === null &&
replacementProfile.sessionCheckpoint == null &&
!input.hasProviderEvents,
);
}
function sameProvider(
previous: NativeExecutionInput["provider"],
current: NativeExecutionInput["provider"],

View File

@ -791,7 +791,7 @@ export function recoveryService(db: Db, deps: { enqueueWakeup: RecoveryWakeup })
}
async function hasActiveExecutionPath(companyId: string, issueId: string, agentId?: string | null) {
const [run, deferredWake] = await Promise.all([
const [run, deferredWake, nativeRecovery] = await Promise.all([
db
.select({ id: heartbeatRuns.id })
.from(heartbeatRuns)
@ -818,9 +818,35 @@ export function recoveryService(db: Db, deps: { enqueueWakeup: RecoveryWakeup })
)
.limit(1)
.then((rows) => rows[0] ?? null),
db
.select({ id: nativeRunFinalizations.runId })
.from(nativeRunFinalizations)
.innerJoin(
heartbeatRuns,
eq(heartbeatRuns.id, nativeRunFinalizations.runId),
)
.where(
and(
eq(nativeRunFinalizations.companyId, companyId),
eq(nativeRunFinalizations.issueId, issueId),
isNull(nativeRunFinalizations.resultId),
agentId ? eq(heartbeatRuns.agentId, agentId) : sql`true`,
or(
inArray(nativeRunFinalizations.recoveryState, [
"awaiting_evidence",
"awaiting_runner_reattach",
"resuming_session",
"bootstrap_incomplete",
]),
eq(nativeRunFinalizations.phase, "retryable_failure"),
),
),
)
.limit(1)
.then((rows) => rows[0] ?? null),
]);
return Boolean(run || deferredWake);
return Boolean(run || deferredWake || nativeRecovery);
}
async function hasPendingWakeInteraction(companyId: string, issueId: string) {

View File

@ -2,6 +2,7 @@ import { EventEmitter } from "node:events";
import { describe, expect, it, vi } from "vitest";
import {
coordinateHeartbeatSchedulerShutdown,
drainRunExecutionFinalizersForShutdown,
finalizeServerShutdown,
loadWithoutCoordinatedShutdownSignalHooks,
} from "./shutdown.js";
@ -145,6 +146,43 @@ describe("finalizeServerShutdown", () => {
});
});
describe("drainRunExecutionFinalizersForShutdown", () => {
it("awaits bounded execution finalizers", async () => {
const release = deferred();
const drain = vi.fn(() => release.promise);
const pending = drainRunExecutionFinalizersForShutdown({
signal: "SIGTERM",
drain,
timeoutMs: 1_000,
log: stubLogger(),
});
await vi.waitFor(() => expect(drain).toHaveBeenCalledOnce());
release.resolve();
await expect(pending).resolves.toBe("drained");
});
it("returns after the bounded timeout when an adopted run remains active", async () => {
vi.useFakeTimers();
try {
const log = stubLogger();
const pending = drainRunExecutionFinalizersForShutdown({
signal: "SIGINT",
drain: () => new Promise<void>(() => undefined),
timeoutMs: 250,
log,
});
await vi.advanceTimersByTimeAsync(250);
await expect(pending).resolves.toBe("timed_out");
expect(log.info).toHaveBeenCalledWith(
expect.objectContaining({ timeoutMs: 250 }),
expect.stringContaining("timed out"),
);
} finally {
vi.useRealTimers();
}
});
});
describe("loadWithoutCoordinatedShutdownSignalHooks", () => {
it("removes the eager signal handlers from the real embedded-postgres import", async () => {
const before = {

View File

@ -7,6 +7,35 @@ type ShutdownLogger = {
error(obj: object, msg: string): void;
};
export async function drainRunExecutionFinalizersForShutdown(input: {
signal: "SIGINT" | "SIGTERM";
drain: (() => Promise<void>) | null;
timeoutMs?: number;
log: ShutdownLogger;
}): Promise<"drained" | "timed_out" | "unavailable"> {
if (!input.drain) return "unavailable";
const timeoutMs = input.timeoutMs ?? 5_000;
let timer: NodeJS.Timeout | null = null;
try {
const result = await Promise.race([
input.drain().then(() => "drained" as const),
new Promise<"timed_out">((resolve) => {
timer = setTimeout(() => resolve("timed_out"), timeoutMs);
timer.unref?.();
}),
]);
if (result === "timed_out") {
input.log.info(
{ signal: input.signal, timeoutMs },
"bounded heartbeat execution finalizer drain timed out",
);
}
return result;
} finally {
if (timer) clearTimeout(timer);
}
}
/**
* Runs the final, ordered teardown of the server. It awaits the application
* service cleanup first, so a live setup-token login session stops and releases

View File

@ -0,0 +1,21 @@
export type StartupRecoveryPhase = "starting" | "recovering" | "ready";
let phase: StartupRecoveryPhase = "ready";
let updatedAt = new Date().toISOString();
export function setStartupRecoveryPhase(next: StartupRecoveryPhase): void {
phase = next;
updatedAt = new Date().toISOString();
}
export function getStartupRecoveryState(): {
phase: StartupRecoveryPhase;
updatedAt: string;
} {
return { phase, updatedAt };
}
export function resetStartupRecoveryStateForTests(): void {
phase = "ready";
updatedAt = new Date().toISOString();
}