Merge remote-tracking branch 'origin/master' into codex/runner-e2e-s3-images
* origin/master: fix(runner): recover native sessions across restarts (#12845)
This commit is contained in:
commit
aae81f4254
|
|
@ -251,6 +251,15 @@ that invariant. Removing or disabling a future native rollout flag must not
|
|||
delete these records; persisted experimental runs remain available for recovery
|
||||
and inspection.
|
||||
|
||||
`native_run_finalizations` also stores restart ownership and recovery state.
|
||||
The controller owner is a server boot id, PID, operating-system process-start
|
||||
timestamp, and monotonically increasing controller generation. Recovery writes
|
||||
its correlated request id, current state, and a bounded JSON history. A
|
||||
successor can take the lease immediately only when coordinated handoff or PID
|
||||
and process-start evidence proves the prior controller is gone, or when the
|
||||
lease expires. Recovery generation changes do not increment the independent
|
||||
provider-attempt counter.
|
||||
|
||||
## Question-response delivery receipts
|
||||
|
||||
`issue_question_response_deliveries` is the retry-safe, content-free outbox for
|
||||
|
|
|
|||
|
|
@ -763,6 +763,47 @@ agent workspace. The host `HOME` itself, a directory that contains it, a
|
|||
filesystem root, a `CODEX_HOME` overlap, or a canonical path outside the
|
||||
assigned workspace is rejected before provider startup.
|
||||
|
||||
### Native runner restart recovery
|
||||
|
||||
Paperclip Runner keeps its heartbeat run, native session, logical runner, and
|
||||
provider session identities across server restarts. A coordinated hot restart
|
||||
registers a correlated recovery request before it signals the dev supervisor.
|
||||
An uncoordinated server restart uses the same durable recovery classifier
|
||||
without trusting a handoff marker.
|
||||
|
||||
Startup binds the HTTP and PRP listener before it classifies native runs. Public
|
||||
health reports a startup state until every candidate is reattached, dispatched
|
||||
for same-run resume, finalized from durable evidence, or held for explicit
|
||||
ownership evidence. Scheduling and generic orphan recovery start only after
|
||||
that classification finishes.
|
||||
|
||||
- A verified live runner re-registers its existing PRP authority and reconnects
|
||||
with the same operating-system PID. Paperclip does not spawn a competing
|
||||
runner.
|
||||
- A verified dead runner starts a replacement from the same durable root and
|
||||
resumes the same provider checkpoint. Only the operating-system PID changes.
|
||||
- A runner that died before its first authenticated connection can restart on
|
||||
the same run only when its durable root proves that no provider authority or
|
||||
checkpoint exists. Paperclip quarantines the incomplete root first.
|
||||
- A live but mismatched or unverifiable process fails closed. Paperclip does not
|
||||
signal it or spawn a replacement.
|
||||
- A persisted proposed or terminal result is reconciled before any runner or
|
||||
provider work starts, so restart recovery cannot submit a duplicate turn.
|
||||
|
||||
Run the credential-free real-process restart suite with:
|
||||
|
||||
```sh
|
||||
pnpm --filter @paperclipai/paperclip-runner build:runner-binaries
|
||||
pnpm exec vitest run server/src/services/native-runtime/native-runner-restart-recovery.integration.test.ts
|
||||
pnpm --filter @paperclipai/paperclip-runner exec vitest run src/live/runnerd-codex-transport.test.ts -t 'adopts a live runner'
|
||||
```
|
||||
|
||||
The suite uses isolated PostgreSQL state, isolated `PAPERCLIP_HOME` roots, real
|
||||
`runnerd` processes, and a deterministic fake Codex app server. It covers hot
|
||||
and hard restarts with live and dead runners, the result-finalization race,
|
||||
incomplete bootstrap, repeated crashes with steering, and fail-closed process
|
||||
identity mismatches.
|
||||
|
||||
## App-Shipped Skills Catalog
|
||||
|
||||
The Paperclip app ships a curated catalog of company skills out of the box. The
|
||||
|
|
|
|||
|
|
@ -23,6 +23,25 @@ credential material are never written to the run log.
|
|||
These records remain run-log events. They do not create an OpenTelemetry or
|
||||
Paperclip Telemetry export, and legacy adapters do not use this writer.
|
||||
|
||||
## Native Restart Recovery Run-Log Event
|
||||
|
||||
Paperclip writes a `native.recovery.transition` event for every native restart
|
||||
classification and for graceful restart suspension. This immutable run-log
|
||||
record lets operators reconstruct recovery decisions without exporting data to
|
||||
Paperclip Telemetry or OpenTelemetry.
|
||||
|
||||
The payload contains the restart kind, recovery request id when one exists,
|
||||
runner disposition, and the controller generation and provider attempt for a
|
||||
claimed recovery. Live-runner adoption also records the runner PID, process
|
||||
group, and process-start fingerprint. A non-claim disposition records a bounded
|
||||
reason instead. Graceful suspension records the signal and confirms that it did
|
||||
not create a retry run.
|
||||
|
||||
The event never includes bootstrap tickets, reconnect leases, authentication
|
||||
proofs, encryption keys, environment variables, provider credentials, command
|
||||
arguments, or an unsanitized stderr stream. Detailed failed-attempt diagnostics
|
||||
remain in the bounded `native_run_finalizations.recovery_history` ledger.
|
||||
|
||||
## Sandbox Startup Run-Log Event
|
||||
|
||||
Paperclip writes one `run.startup.step` event to the run log for each bring-up
|
||||
|
|
|
|||
|
|
@ -0,0 +1,7 @@
|
|||
ALTER TABLE "native_run_finalizations" ADD COLUMN IF NOT EXISTS "controller_boot_id" text;--> statement-breakpoint
|
||||
ALTER TABLE "native_run_finalizations" ADD COLUMN IF NOT EXISTS "controller_pid" integer;--> statement-breakpoint
|
||||
ALTER TABLE "native_run_finalizations" ADD COLUMN IF NOT EXISTS "controller_process_started_at" timestamp with time zone;--> statement-breakpoint
|
||||
ALTER TABLE "native_run_finalizations" ADD COLUMN IF NOT EXISTS "controller_generation" integer DEFAULT 0 NOT NULL;--> statement-breakpoint
|
||||
ALTER TABLE "native_run_finalizations" ADD COLUMN IF NOT EXISTS "recovery_state" text;--> statement-breakpoint
|
||||
ALTER TABLE "native_run_finalizations" ADD COLUMN IF NOT EXISTS "recovery_request_id" text;--> statement-breakpoint
|
||||
ALTER TABLE "native_run_finalizations" ADD COLUMN IF NOT EXISTS "recovery_history" jsonb DEFAULT '[]'::jsonb NOT NULL;
|
||||
File diff suppressed because it is too large
Load Diff
|
|
@ -1653,6 +1653,13 @@
|
|||
"when": 1788296836732,
|
||||
"tag": "0237_clammy_colonel_america",
|
||||
"breakpoints": true
|
||||
},
|
||||
{
|
||||
"idx": 238,
|
||||
"version": "7",
|
||||
"when": 1788542862021,
|
||||
"tag": "0238_graceful_infant_terrible",
|
||||
"breakpoints": true
|
||||
}
|
||||
]
|
||||
}
|
||||
|
|
@ -0,0 +1,155 @@
|
|||
import { readFile } from "node:fs/promises";
|
||||
import { afterEach, describe, expect, it } from "vitest";
|
||||
import postgres from "postgres";
|
||||
import {
|
||||
EMBEDDED_POSTGRES_TEST_TIMEOUT_MS,
|
||||
getEmbeddedPostgresTestSupport,
|
||||
startEmbeddedPostgresTestDatabase,
|
||||
} from "./test-embedded-postgres.js";
|
||||
|
||||
const cleanups: Array<() => Promise<void>> = [];
|
||||
const support = await getEmbeddedPostgresTestSupport();
|
||||
const describeEmbeddedPostgres = support.supported ? describe : describe.skip;
|
||||
|
||||
afterEach(async () => {
|
||||
while (cleanups.length > 0) await cleanups.pop()?.();
|
||||
});
|
||||
|
||||
describeEmbeddedPostgres("native runner recovery migration", () => {
|
||||
it(
|
||||
"repairs a partial application and can be replayed without changing the schema",
|
||||
async () => {
|
||||
const database = await startEmbeddedPostgresTestDatabase(
|
||||
"paperclip-native-recovery-migration-",
|
||||
);
|
||||
cleanups.push(database.cleanup);
|
||||
const sql = postgres(database.connectionString, {
|
||||
max: 1,
|
||||
onnotice: () => {},
|
||||
});
|
||||
cleanups.push(async () => sql.end());
|
||||
|
||||
const migration = await readFile(
|
||||
new URL(
|
||||
"./migrations/0238_graceful_infant_terrible.sql",
|
||||
import.meta.url,
|
||||
),
|
||||
"utf8",
|
||||
);
|
||||
const statements = migration
|
||||
.split("--> statement-breakpoint")
|
||||
.map((statement) => statement.trim())
|
||||
.filter((statement) => statement.length > 0);
|
||||
|
||||
// Model an interrupted migration: earlier statements committed, but one of
|
||||
// the later columns did not. Reapplying must skip completed work and repair
|
||||
// only the missing portion.
|
||||
await sql`ALTER TABLE "native_run_finalizations" DROP COLUMN "recovery_request_id"`;
|
||||
for (const statement of statements) await sql.unsafe(statement);
|
||||
|
||||
const columnsAfterRepair = await sql<
|
||||
{
|
||||
columnName: string;
|
||||
dataType: string;
|
||||
isNullable: "YES" | "NO";
|
||||
columnDefault: string | null;
|
||||
}[]
|
||||
>`
|
||||
SELECT
|
||||
column_name AS "columnName",
|
||||
udt_name AS "dataType",
|
||||
is_nullable AS "isNullable",
|
||||
column_default AS "columnDefault"
|
||||
FROM information_schema.columns
|
||||
WHERE table_schema = 'public'
|
||||
AND table_name = 'native_run_finalizations'
|
||||
AND column_name IN (
|
||||
'controller_boot_id',
|
||||
'controller_pid',
|
||||
'controller_process_started_at',
|
||||
'controller_generation',
|
||||
'recovery_state',
|
||||
'recovery_request_id',
|
||||
'recovery_history'
|
||||
)
|
||||
ORDER BY column_name
|
||||
`;
|
||||
|
||||
for (const statement of statements) await sql.unsafe(statement);
|
||||
const columnsAfterReplay = await sql<
|
||||
{
|
||||
columnName: string;
|
||||
dataType: string;
|
||||
isNullable: "YES" | "NO";
|
||||
columnDefault: string | null;
|
||||
}[]
|
||||
>`
|
||||
SELECT
|
||||
column_name AS "columnName",
|
||||
udt_name AS "dataType",
|
||||
is_nullable AS "isNullable",
|
||||
column_default AS "columnDefault"
|
||||
FROM information_schema.columns
|
||||
WHERE table_schema = 'public'
|
||||
AND table_name = 'native_run_finalizations'
|
||||
AND column_name IN (
|
||||
'controller_boot_id',
|
||||
'controller_pid',
|
||||
'controller_process_started_at',
|
||||
'controller_generation',
|
||||
'recovery_state',
|
||||
'recovery_request_id',
|
||||
'recovery_history'
|
||||
)
|
||||
ORDER BY column_name
|
||||
`;
|
||||
|
||||
expect(columnsAfterReplay).toEqual(columnsAfterRepair);
|
||||
expect(columnsAfterReplay).toEqual([
|
||||
{
|
||||
columnName: "controller_boot_id",
|
||||
dataType: "text",
|
||||
isNullable: "YES",
|
||||
columnDefault: null,
|
||||
},
|
||||
{
|
||||
columnName: "controller_generation",
|
||||
dataType: "int4",
|
||||
isNullable: "NO",
|
||||
columnDefault: "0",
|
||||
},
|
||||
{
|
||||
columnName: "controller_pid",
|
||||
dataType: "int4",
|
||||
isNullable: "YES",
|
||||
columnDefault: null,
|
||||
},
|
||||
{
|
||||
columnName: "controller_process_started_at",
|
||||
dataType: "timestamptz",
|
||||
isNullable: "YES",
|
||||
columnDefault: null,
|
||||
},
|
||||
{
|
||||
columnName: "recovery_history",
|
||||
dataType: "jsonb",
|
||||
isNullable: "NO",
|
||||
columnDefault: "'[]'::jsonb",
|
||||
},
|
||||
{
|
||||
columnName: "recovery_request_id",
|
||||
dataType: "text",
|
||||
isNullable: "YES",
|
||||
columnDefault: null,
|
||||
},
|
||||
{
|
||||
columnName: "recovery_state",
|
||||
dataType: "text",
|
||||
isNullable: "YES",
|
||||
columnDefault: null,
|
||||
},
|
||||
]);
|
||||
},
|
||||
EMBEDDED_POSTGRES_TEST_TIMEOUT_MS,
|
||||
);
|
||||
});
|
||||
|
|
@ -26,6 +26,18 @@ export const nativeRunFinalizations = pgTable(
|
|||
attempt: integer("attempt").notNull().default(0),
|
||||
leaseOwner: text("lease_owner"),
|
||||
leaseExpiresAt: timestamp("lease_expires_at", { withTimezone: true }),
|
||||
controllerBootId: text("controller_boot_id"),
|
||||
controllerPid: integer("controller_pid"),
|
||||
controllerProcessStartedAt: timestamp("controller_process_started_at", {
|
||||
withTimezone: true,
|
||||
}),
|
||||
controllerGeneration: integer("controller_generation").notNull().default(0),
|
||||
recoveryState: text("recovery_state"),
|
||||
recoveryRequestId: text("recovery_request_id"),
|
||||
recoveryHistory: jsonb("recovery_history")
|
||||
.$type<Array<Record<string, unknown>>>()
|
||||
.notNull()
|
||||
.default(sql`'[]'::jsonb`),
|
||||
resultId: uuid("result_id"),
|
||||
assessmentId: uuid("assessment_id"),
|
||||
decisionId: uuid("decision_id"),
|
||||
|
|
|
|||
|
|
@ -88,6 +88,11 @@ The state contains:
|
|||
- lifecycle, reconnect, backpressure, harness generation, and recovery facts;
|
||||
- bounded, redacted diagnostics.
|
||||
|
||||
Raw runner stdout and stderr are never redirected to durable files. The runner
|
||||
itself redacts and bounds a terminal diagnostic before publishing it through an
|
||||
atomic private-file replacement, so neither an output burst nor controller
|
||||
restart can create a transient unbounded or unredacted diagnostic file.
|
||||
|
||||
Authentication capabilities and arbitrary command bodies are not stored. A
|
||||
recent command is represented by a SHA-256 comparison digest from the vetted
|
||||
RustCrypto implementation, its stable ID and controller sequence, the redacted
|
||||
|
|
|
|||
|
|
@ -1,16 +1,108 @@
|
|||
use std::path::PathBuf;
|
||||
use std::fs::{self, OpenOptions};
|
||||
use std::io::{self, Write};
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::process::ExitCode;
|
||||
use std::time::Duration;
|
||||
|
||||
#[cfg(unix)]
|
||||
use std::os::unix::fs::{OpenOptionsExt, PermissionsExt};
|
||||
|
||||
use paperclip_runner_core::durable::{
|
||||
capture_bootstrap_ticket, run_durable_runner, AcpxLaunchProfile, DurableRunnerConfig,
|
||||
OpenCodeLaunchProfile, QualifiedLaunchArtifact,
|
||||
capture_bootstrap_ticket, redact_diagnostic_text, run_durable_runner, AcpxLaunchProfile,
|
||||
DurableRunnerConfig, OpenCodeLaunchProfile, QualifiedLaunchArtifact,
|
||||
};
|
||||
use paperclip_runner_core::local_runner::{run_local_runner, LocalRunnerError, RunnerConfig};
|
||||
use paperclip_runner_core::native_provider_backend::NativeProviderCommandExecutor;
|
||||
use serde_json::json;
|
||||
|
||||
const RUNNERD_BUILD_METADATA_SCHEMA: &str = "paperclip-runner/runnerd-build-metadata/v1";
|
||||
const RUNNER_DIAGNOSTIC_MAX_BYTES: usize = 64 * 1024;
|
||||
const RUNNER_DIAGNOSTIC_TEMP_ATTEMPTS: usize = 32;
|
||||
|
||||
fn diagnostic_directory(args: &[String]) -> Option<PathBuf> {
|
||||
let index = args
|
||||
.iter()
|
||||
.position(|argument| argument == "--diagnostics-directory")?;
|
||||
args.get(index + 1).map(PathBuf::from)
|
||||
}
|
||||
|
||||
fn bounded_redacted_diagnostic(message: &str) -> String {
|
||||
let mut diagnostic = redact_diagnostic_text(message);
|
||||
if diagnostic.len() <= RUNNER_DIAGNOSTIC_MAX_BYTES {
|
||||
return diagnostic;
|
||||
}
|
||||
let suffix = "…[truncated]";
|
||||
let byte_limit = RUNNER_DIAGNOSTIC_MAX_BYTES.saturating_sub(suffix.len());
|
||||
let boundary = diagnostic
|
||||
.char_indices()
|
||||
.map(|(index, _)| index)
|
||||
.take_while(|index| *index <= byte_limit)
|
||||
.last()
|
||||
.unwrap_or(0);
|
||||
diagnostic.truncate(boundary);
|
||||
diagnostic.push_str(suffix);
|
||||
diagnostic
|
||||
}
|
||||
|
||||
fn verify_private_diagnostics_directory(directory: &Path) -> io::Result<()> {
|
||||
let metadata = fs::symlink_metadata(directory)?;
|
||||
if metadata.file_type().is_symlink() || !metadata.is_dir() {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::InvalidInput,
|
||||
"runner diagnostics path is not a real directory",
|
||||
));
|
||||
}
|
||||
#[cfg(unix)]
|
||||
if metadata.permissions().mode() & 0o077 != 0 {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::PermissionDenied,
|
||||
"runner diagnostics directory is accessible by group or other users",
|
||||
));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn persist_runner_diagnostic(directory: &Path, message: &str) -> io::Result<()> {
|
||||
verify_private_diagnostics_directory(directory)?;
|
||||
let destination = directory.join("runnerd.stderr.log");
|
||||
let contents = bounded_redacted_diagnostic(message);
|
||||
let process_id = std::process::id();
|
||||
for attempt in 0..RUNNER_DIAGNOSTIC_TEMP_ATTEMPTS {
|
||||
let temporary = directory.join(format!(".runnerd.stderr.log.{process_id}.{attempt}.tmp"));
|
||||
let mut options = OpenOptions::new();
|
||||
options.write(true).create_new(true);
|
||||
#[cfg(unix)]
|
||||
options.mode(0o600);
|
||||
let mut file = match options.open(&temporary) {
|
||||
Ok(file) => file,
|
||||
Err(error) if error.kind() == io::ErrorKind::AlreadyExists => continue,
|
||||
Err(error) => return Err(error),
|
||||
};
|
||||
let write_result = (|| {
|
||||
file.write_all(contents.as_bytes())?;
|
||||
file.sync_all()?;
|
||||
drop(file);
|
||||
fs::rename(&temporary, &destination)
|
||||
})();
|
||||
if write_result.is_err() {
|
||||
let _ = fs::remove_file(&temporary);
|
||||
}
|
||||
return write_result;
|
||||
}
|
||||
Err(io::Error::new(
|
||||
io::ErrorKind::AlreadyExists,
|
||||
"could not allocate a runner diagnostic temporary file",
|
||||
))
|
||||
}
|
||||
|
||||
fn install_diagnostic_panic_hook(directory: Option<PathBuf>) {
|
||||
let Some(directory) = directory else {
|
||||
return;
|
||||
};
|
||||
std::panic::set_hook(Box::new(move |panic| {
|
||||
let _ = persist_runner_diagnostic(&directory, &format!("paperclip-runnerd panic: {panic}"));
|
||||
}));
|
||||
}
|
||||
|
||||
fn build_metadata() -> serde_json::Value {
|
||||
json!({
|
||||
|
|
@ -237,9 +329,8 @@ fn run_durable(args: &[String]) -> Result<(), LocalRunnerError> {
|
|||
.map_err(|error| LocalRunnerError::invalid(error.to_string()))
|
||||
}
|
||||
|
||||
fn run() -> Result<(), LocalRunnerError> {
|
||||
let args = std::env::args().skip(1).collect::<Vec<_>>();
|
||||
if args.as_slice() == ["--build-metadata"] {
|
||||
fn run(args: &[String]) -> Result<(), LocalRunnerError> {
|
||||
if args == ["--build-metadata"] {
|
||||
println!("{}", build_metadata());
|
||||
return Ok(());
|
||||
}
|
||||
|
|
@ -248,22 +339,22 @@ fn run() -> Result<(), LocalRunnerError> {
|
|||
|| args.iter().any(|argument| argument == "--listen-port")
|
||||
|| args.iter().any(|argument| argument == "--listen-path")
|
||||
{
|
||||
return run_durable(&args);
|
||||
return run_durable(args);
|
||||
}
|
||||
run_local_runner(RunnerConfig {
|
||||
run_id: value(&args, "--run-id")?,
|
||||
normalized_session_id: value(&args, "--session-id")?,
|
||||
runner_instance_id: value(&args, "--runner-id")?,
|
||||
fake_harness_path: PathBuf::from(value(&args, "--fake-harness")?),
|
||||
script_path: PathBuf::from(value(&args, "--script")?),
|
||||
delay_override_ms: optional_u64(&args, "--delay-ms")?,
|
||||
log_max_lines: usize_value(&args, "--log-max-lines", 32)?,
|
||||
log_max_bytes: usize_value(&args, "--log-max-bytes", 16_384)?,
|
||||
command_history_limit: usize_value(&args, "--command-history-limit", 4096)?,
|
||||
controller_max_line_bytes: usize_value(&args, "--controller-max-line-bytes", 64 * 1024)?,
|
||||
harness_max_line_bytes: usize_value(&args, "--harness-max-line-bytes", 64 * 1024)?,
|
||||
run_id: value(args, "--run-id")?,
|
||||
normalized_session_id: value(args, "--session-id")?,
|
||||
runner_instance_id: value(args, "--runner-id")?,
|
||||
fake_harness_path: PathBuf::from(value(args, "--fake-harness")?),
|
||||
script_path: PathBuf::from(value(args, "--script")?),
|
||||
delay_override_ms: optional_u64(args, "--delay-ms")?,
|
||||
log_max_lines: usize_value(args, "--log-max-lines", 32)?,
|
||||
log_max_bytes: usize_value(args, "--log-max-bytes", 16_384)?,
|
||||
command_history_limit: usize_value(args, "--command-history-limit", 4096)?,
|
||||
controller_max_line_bytes: usize_value(args, "--controller-max-line-bytes", 64 * 1024)?,
|
||||
harness_max_line_bytes: usize_value(args, "--harness-max-line-bytes", 64 * 1024)?,
|
||||
shutdown_grace: Duration::from_millis(
|
||||
optional_u64(&args, "--shutdown-grace-ms")?.unwrap_or(100),
|
||||
optional_u64(args, "--shutdown-grace-ms")?.unwrap_or(100),
|
||||
),
|
||||
})
|
||||
}
|
||||
|
|
@ -282,13 +373,71 @@ mod tests {
|
|||
json!(["dial_ws_loopback", "dial_wss", "listen_ws"])
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn persistent_diagnostic_is_private_bounded_and_redacted() {
|
||||
let unique = format!(
|
||||
"paperclip-runnerd-diagnostic-test-{}-{}",
|
||||
std::process::id(),
|
||||
std::time::SystemTime::now()
|
||||
.duration_since(std::time::UNIX_EPOCH)
|
||||
.expect("test clock is after the Unix epoch")
|
||||
.as_nanos()
|
||||
);
|
||||
let directory = std::env::temp_dir().join(unique);
|
||||
fs::create_dir(&directory).expect("create diagnostic test directory");
|
||||
#[cfg(unix)]
|
||||
fs::set_permissions(&directory, fs::Permissions::from_mode(0o700))
|
||||
.expect("make diagnostic test directory private");
|
||||
let path = directory.join("runnerd.stderr.log");
|
||||
fs::write(&path, "").expect("create runner diagnostic placeholder");
|
||||
#[cfg(unix)]
|
||||
fs::set_permissions(&path, fs::Permissions::from_mode(0o600))
|
||||
.expect("make runner diagnostic placeholder private");
|
||||
|
||||
let diagnostic = format!(
|
||||
"OPENAI_API_KEY=raw-secret Authorization: Bearer another-secret {}",
|
||||
"x".repeat(RUNNER_DIAGNOSTIC_MAX_BYTES * 4)
|
||||
);
|
||||
persist_runner_diagnostic(&directory, &diagnostic)
|
||||
.expect("persist bounded runner diagnostic");
|
||||
|
||||
let persisted = fs::read_to_string(&path).expect("read runner diagnostic");
|
||||
assert!(persisted.len() <= RUNNER_DIAGNOSTIC_MAX_BYTES);
|
||||
assert!(!persisted.contains("raw-secret"));
|
||||
assert!(!persisted.contains("another-secret"));
|
||||
assert!(persisted.contains("[REDACTED]"));
|
||||
#[cfg(unix)]
|
||||
assert_eq!(
|
||||
fs::metadata(&path)
|
||||
.expect("read runner diagnostic metadata")
|
||||
.permissions()
|
||||
.mode()
|
||||
& 0o777,
|
||||
0o600
|
||||
);
|
||||
|
||||
fs::remove_dir_all(directory).expect("remove diagnostic test directory");
|
||||
}
|
||||
}
|
||||
|
||||
fn main() -> ExitCode {
|
||||
match run() {
|
||||
let args = std::env::args().skip(1).collect::<Vec<_>>();
|
||||
let diagnostics_directory = diagnostic_directory(&args);
|
||||
install_diagnostic_panic_hook(diagnostics_directory.clone());
|
||||
match run(&args) {
|
||||
Ok(()) => ExitCode::SUCCESS,
|
||||
Err(error) => {
|
||||
eprintln!("paperclip-runnerd: {error}");
|
||||
let message = format!("paperclip-runnerd: {error}");
|
||||
if let Some(directory) = diagnostics_directory {
|
||||
if let Err(persist_error) = persist_runner_diagnostic(&directory, &message) {
|
||||
eprintln!(
|
||||
"paperclip-runnerd: failed to persist bounded diagnostic: {persist_error}"
|
||||
);
|
||||
}
|
||||
} else {
|
||||
eprintln!("{message}");
|
||||
}
|
||||
ExitCode::FAILURE
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -28,6 +28,11 @@ pub const BOOTSTRAP_TICKET_ENV: &str = "PAPERCLIP_RUNNER_BOOTSTRAP_TICKET";
|
|||
const MAX_OUTBOX_BYTES: usize = 512 * 1024 * 1024;
|
||||
const MAX_FRAME_BYTES: usize = 16 * 1024 * 1024;
|
||||
|
||||
/// Redacts and bounds a diagnostic before a process boundary may persist it.
|
||||
pub fn redact_diagnostic_text(input: &str) -> String {
|
||||
state::redact_text(input)
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, PartialEq, Eq)]
|
||||
pub struct DurableRunnerError(String);
|
||||
|
||||
|
|
|
|||
|
|
@ -4,7 +4,13 @@ import {
|
|||
createHash,
|
||||
createHmac,
|
||||
} from "node:crypto";
|
||||
import { mkdtempSync, rmSync } from "node:fs";
|
||||
import {
|
||||
chmodSync,
|
||||
mkdtempSync,
|
||||
readFileSync,
|
||||
rmSync,
|
||||
writeFileSync,
|
||||
} from "node:fs";
|
||||
import { connect, type Socket } from "node:net";
|
||||
import { tmpdir } from "node:os";
|
||||
import { resolve } from "node:path";
|
||||
|
|
@ -31,6 +37,49 @@ const identity: DurableRecoveryIdentity = {
|
|||
const expectedRunnerVersion = "0.3.0";
|
||||
const expectedRunnerDigest = `sha256:${"a".repeat(64)}`;
|
||||
|
||||
it.skipIf(process.platform === "win32")(
|
||||
"never persists raw child stdout or stderr as durable diagnostics",
|
||||
async () => {
|
||||
const root = mkdtempSync(resolve(tmpdir(), "runner-diagnostics-test-"));
|
||||
const executable = resolve(root, "noisy-runner");
|
||||
const diagnosticsDirectory = resolve(root, "diagnostics");
|
||||
writeFileSync(
|
||||
executable,
|
||||
[
|
||||
"#!/usr/bin/env node",
|
||||
'process.stdout.write("token=raw-stdout-secret " + "o".repeat(256 * 1024));',
|
||||
'process.stderr.write("Authorization: Bearer raw-stderr-secret " + "e".repeat(256 * 1024));',
|
||||
].join("\n"),
|
||||
{ mode: 0o700 },
|
||||
);
|
||||
chmodSync(executable, 0o700);
|
||||
try {
|
||||
const handle = spawnRunner({
|
||||
connection: { mode: "connect", connectUrl: "ws://127.0.0.1:43127" },
|
||||
stateDirectory: resolve(root, "state"),
|
||||
identity,
|
||||
ticket: "bootstrap-ticket",
|
||||
maxOutboxBytes: 256 * 1024,
|
||||
p0ReserveBytes: 64 * 1024,
|
||||
runnerVersion: expectedRunnerVersion,
|
||||
runnerDigest: expectedRunnerDigest,
|
||||
runnerBinaryPath: executable,
|
||||
diagnosticsDirectory,
|
||||
});
|
||||
const result = await handle.completion;
|
||||
expect(result.stdout).toBe("");
|
||||
expect(result.stderr).toBe("");
|
||||
|
||||
for (const name of ["runnerd.stdout.log", "runnerd.stderr.log"]) {
|
||||
const filePath = resolve(diagnosticsDirectory, name);
|
||||
expect(readFileSync(filePath, "utf8")).toBe("");
|
||||
}
|
||||
} finally {
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
}
|
||||
},
|
||||
);
|
||||
|
||||
it("pins the ACPX launch profile in runner startup arguments and restarts", () => {
|
||||
const launches: RunnerProcessLaunchSpec[] = [];
|
||||
const handle = spawnRunner({
|
||||
|
|
|
|||
|
|
@ -226,6 +226,8 @@ export interface RunnerProcessHandle {
|
|||
kill(signal?: NodeJS.Signals | number): boolean;
|
||||
};
|
||||
completion: Promise<RunnerProcessResult>;
|
||||
processGroupId?: number | null;
|
||||
startedAt?: string;
|
||||
/** Relaunches the same immutable process specification with a fresh ticket. */
|
||||
restart?(ticket: string): RunnerProcessHandle;
|
||||
}
|
||||
|
|
@ -1993,6 +1995,7 @@ export function spawnRunner(options: {
|
|||
};
|
||||
environment?: NodeJS.ProcessEnv;
|
||||
processLauncher?: (spec: RunnerProcessLaunchSpec) => RunnerProcessHandle;
|
||||
diagnosticsDirectory?: string;
|
||||
}): RunnerProcessHandle {
|
||||
const connection =
|
||||
options.connection ??
|
||||
|
|
@ -2096,6 +2099,9 @@ export function spawnRunner(options: {
|
|||
);
|
||||
}
|
||||
}
|
||||
if (options.diagnosticsDirectory !== undefined) {
|
||||
args.push("--diagnostics-directory", options.diagnosticsDirectory);
|
||||
}
|
||||
|
||||
const command = options.runnerBinaryPath ?? runnerBinary;
|
||||
const environment = runnerEnvironment(options.ticket, options.environment);
|
||||
|
|
@ -2109,28 +2115,75 @@ export function spawnRunner(options: {
|
|||
);
|
||||
}
|
||||
|
||||
const detached = process.platform !== "win32";
|
||||
const diagnosticsDirectory = options.diagnosticsDirectory;
|
||||
let stdoutPath: string | null = null;
|
||||
let stderrPath: string | null = null;
|
||||
if (diagnosticsDirectory) {
|
||||
try {
|
||||
const metadata = lstatSync(diagnosticsDirectory);
|
||||
if (metadata.isSymbolicLink() || !metadata.isDirectory()) {
|
||||
throw new Error(
|
||||
`Private state directory is not a real directory: ${diagnosticsDirectory}`,
|
||||
);
|
||||
}
|
||||
} catch (error) {
|
||||
if (!isNodeError(error, "ENOENT")) throw error;
|
||||
mkdirSync(diagnosticsDirectory, { recursive: true, mode: 0o700 });
|
||||
}
|
||||
if (process.platform !== "win32") chmodSync(diagnosticsDirectory, 0o700);
|
||||
verifyPrivateDirectory(diagnosticsDirectory);
|
||||
stdoutPath = resolve(diagnosticsDirectory, "runnerd.stdout.log");
|
||||
stderrPath = resolve(diagnosticsDirectory, "runnerd.stderr.log");
|
||||
// runnerd owns every durable diagnostic write so it can redact and bound
|
||||
// the complete value before a byte reaches disk. Raw process output is
|
||||
// intentionally discarded below; these files are only the runner-owned
|
||||
// restart-survivable diagnostic channel.
|
||||
atomicPrivateWrite(stdoutPath, "");
|
||||
atomicPrivateWrite(stderrPath, "");
|
||||
}
|
||||
const child = spawn(command, args, {
|
||||
cwd: packageRoot,
|
||||
env: environment,
|
||||
stdio: "pipe",
|
||||
detached,
|
||||
stdio: diagnosticsDirectory ? "ignore" : "pipe",
|
||||
});
|
||||
child.unref();
|
||||
let stdout = "";
|
||||
let stderr = "";
|
||||
child.stdout.setEncoding("utf8").on("data", (chunk: string) => {
|
||||
child.stdout?.setEncoding("utf8").on("data", (chunk: string) => {
|
||||
stdout = `${stdout}${chunk}`.slice(-16_384);
|
||||
});
|
||||
child.stderr.setEncoding("utf8").on("data", (chunk: string) => {
|
||||
child.stderr?.setEncoding("utf8").on("data", (chunk: string) => {
|
||||
stderr = `${stderr}${chunk}`.slice(-16_384);
|
||||
});
|
||||
const completion = new Promise<RunnerProcessResult>(
|
||||
const boundedDiagnostic = (filePath: string | null): string => {
|
||||
if (!filePath) return "";
|
||||
try {
|
||||
return (readPrivateFile(filePath) ?? "").slice(-16_384);
|
||||
} catch {
|
||||
return "";
|
||||
}
|
||||
};
|
||||
const processCompletion = new Promise<RunnerProcessResult>(
|
||||
(resolveCompletion, rejectCompletion) => {
|
||||
child.once("error", rejectCompletion);
|
||||
child.once("exit", (code, signal) =>
|
||||
resolveCompletion({ code, signal, stdout, stderr }),
|
||||
resolveCompletion({
|
||||
code,
|
||||
signal,
|
||||
stdout: stdout || boundedDiagnostic(stdoutPath),
|
||||
stderr: stderr || boundedDiagnostic(stderrPath),
|
||||
}),
|
||||
);
|
||||
},
|
||||
);
|
||||
return withRestart({ child, completion });
|
||||
return withRestart({
|
||||
child,
|
||||
completion: processCompletion,
|
||||
processGroupId: detached ? (child.pid ?? null) : null,
|
||||
startedAt: new Date().toISOString(),
|
||||
});
|
||||
}
|
||||
|
||||
export async function waitForProcess(
|
||||
|
|
@ -2143,7 +2196,19 @@ export async function waitForProcess(
|
|||
handle.completion,
|
||||
new Promise<never>((_resolve, reject) => {
|
||||
timer = setTimeout(() => {
|
||||
handle.child.kill("SIGKILL");
|
||||
if (
|
||||
process.platform !== "win32" &&
|
||||
handle.processGroupId &&
|
||||
handle.processGroupId > 0
|
||||
) {
|
||||
try {
|
||||
process.kill(-handle.processGroupId, "SIGKILL");
|
||||
} catch {
|
||||
handle.child.kill("SIGKILL");
|
||||
}
|
||||
} else {
|
||||
handle.child.kill("SIGKILL");
|
||||
}
|
||||
reject(new Error("Durable recovery runner timed out."));
|
||||
}, timeoutMs);
|
||||
}),
|
||||
|
|
|
|||
|
|
@ -9,11 +9,13 @@ import {
|
|||
writeFile,
|
||||
} from "node:fs/promises";
|
||||
import { createHash } from "node:crypto";
|
||||
import { createServer } from "node:http";
|
||||
import { tmpdir } from "node:os";
|
||||
import { join, resolve } from "node:path";
|
||||
import { fileURLToPath } from "node:url";
|
||||
|
||||
import { expect, it } from "vitest";
|
||||
import { expect, it, vi } from "vitest";
|
||||
import type { DurablePrpControlPlane } from "../control-plane/durable-prp-control-plane.js";
|
||||
|
||||
import {
|
||||
NATIVE_RUNTIME_ASSET_SCHEMA,
|
||||
|
|
@ -1985,6 +1987,122 @@ it("cold-restores a suspended provider session under its durable run binding", a
|
|||
}
|
||||
}, 30_000);
|
||||
|
||||
it("adopts a live runner on the same durable authority without spawning a duplicate", async () => {
|
||||
const stateDirectory = await mkdtemp(join(tmpdir(), "runnerd-live-adopt-"));
|
||||
const server = createServer();
|
||||
let authority: DurablePrpControlPlane | null = null;
|
||||
server.on("upgrade", (request, socket, head) => {
|
||||
if (!authority) {
|
||||
socket.destroy();
|
||||
return;
|
||||
}
|
||||
authority.handleUpgrade(request, socket, "/runner", head);
|
||||
});
|
||||
await new Promise<void>((resolveListen) =>
|
||||
server.listen(0, "127.0.0.1", resolveListen),
|
||||
);
|
||||
const address = server.address();
|
||||
if (!address || typeof address === "string")
|
||||
throw new Error("Expected adoption test listener");
|
||||
const registration = async (next: DurablePrpControlPlane) => {
|
||||
authority = next;
|
||||
return {
|
||||
connectUrl: `ws://127.0.0.1:${address.port}/runner`,
|
||||
release: async () => {
|
||||
if (authority === next) authority = null;
|
||||
},
|
||||
};
|
||||
};
|
||||
const identity = {
|
||||
runnerInstanceId: "runner-live-adopt",
|
||||
environmentLeaseId: "lease-live-adopt",
|
||||
runId: "run-live-adopt",
|
||||
normalizedSessionId: "session-live-adopt",
|
||||
turnId: "turn-live-adopt",
|
||||
itemId: "item-live-adopt",
|
||||
};
|
||||
const sharedOptions = {
|
||||
runnerBinary: defaultCapabilityRunnerdBinary(),
|
||||
codexCommand: fakeCodex,
|
||||
codexArgs: fakeCodexArgs(stateDirectory),
|
||||
stateDirectory,
|
||||
prpIdentity: identity,
|
||||
lifecyclePolicy: { mode: "warm" as const, idleTimeoutMs: 60_000 },
|
||||
controlPlaneRegistration: registration,
|
||||
};
|
||||
const first = createCapabilityRunnerdCodexTransport(sharedOptions);
|
||||
let runnerPid: number | null = null;
|
||||
let adopted: ReturnType<typeof createCapabilityRunnerdCodexTransport> | null =
|
||||
null;
|
||||
try {
|
||||
try {
|
||||
await first.transport.request("thread/start", {
|
||||
cwd: tmpdir(),
|
||||
dynamicTools: codexSemanticToolSpecs(),
|
||||
});
|
||||
} catch (error) {
|
||||
const stderr = await readFile(
|
||||
join(stateDirectory, "diagnostics", "runnerd.stderr.log"),
|
||||
"utf8",
|
||||
).catch(() => "");
|
||||
throw new Error(
|
||||
`${String(error)}\n${JSON.stringify(first.evidence())}${stderr ? `\n${stderr}` : ""}`,
|
||||
);
|
||||
}
|
||||
runnerPid = first.evidence().runnerPid;
|
||||
expect(runnerPid).toEqual(expect.any(Number));
|
||||
|
||||
await first.detachControllerForRestart();
|
||||
expect(() => process.kill(runnerPid!, 0)).not.toThrow();
|
||||
|
||||
const duplicateLauncher = vi.fn(() => {
|
||||
throw new Error("duplicate runner spawn attempted");
|
||||
});
|
||||
adopted = createCapabilityRunnerdCodexTransport({
|
||||
...sharedOptions,
|
||||
resumeDynamicTools: [],
|
||||
runnerProcessLauncher: duplicateLauncher,
|
||||
adoptExistingRunner: {
|
||||
pid: runnerPid!,
|
||||
processGroupId: runnerPid,
|
||||
startedAt: new Date().toISOString(),
|
||||
isAlive: () => {
|
||||
try {
|
||||
process.kill(runnerPid!, 0);
|
||||
return true;
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
},
|
||||
},
|
||||
});
|
||||
await expect(adopted.transport.request("thread/read", {})).resolves.toEqual(
|
||||
expect.objectContaining({
|
||||
thread: expect.objectContaining({ id: "codex-thread-1" }),
|
||||
}),
|
||||
);
|
||||
expect(adopted.evidence().runnerPid).toBe(runnerPid);
|
||||
expect(duplicateLauncher).not.toHaveBeenCalled();
|
||||
expect(adopted.evidence().diagnostics).toContain(
|
||||
`adopted runner ${runnerPid} authenticated to its durable PRP authority`,
|
||||
);
|
||||
} finally {
|
||||
await adopted?.transport.close().catch(() => undefined);
|
||||
if (runnerPid) {
|
||||
try {
|
||||
process.kill(-runnerPid, "SIGKILL");
|
||||
} catch {
|
||||
// The adopted runner normally exits after its durable suspend command.
|
||||
}
|
||||
}
|
||||
server.closeAllConnections();
|
||||
if (server.listening) {
|
||||
await new Promise<void>((resolveClose) => server.close(() => resolveClose()));
|
||||
}
|
||||
await rm(stateDirectory, { recursive: true, force: true });
|
||||
}
|
||||
}, 30_000);
|
||||
|
||||
it("surfaces a runner exit while provider-ingress readiness is still pending", async () => {
|
||||
const neverReady = new Promise<void>(() => undefined);
|
||||
const bundle = createCapabilityRunnerdCodexTransport({
|
||||
|
|
|
|||
|
|
@ -1,3 +1,4 @@
|
|||
import { execFileSync } from "node:child_process";
|
||||
import { createHash, randomUUID } from "node:crypto";
|
||||
import {
|
||||
appendFileSync,
|
||||
|
|
@ -29,7 +30,10 @@ import {
|
|||
codexSemanticToolSpecs,
|
||||
createIsolatedCodexAppServerArgs,
|
||||
} from "../drivers/codex/codex-app-server-driver.js";
|
||||
import type { DurableRecoveryIdentity } from "../contracts/durable-recovery.js";
|
||||
import type {
|
||||
DurableRecoveryCommittedEvent,
|
||||
DurableRecoveryIdentity,
|
||||
} from "../contracts/durable-recovery.js";
|
||||
import type { HarnessRuntimeRequestResolution } from "../contracts/harness-driver.js";
|
||||
import {
|
||||
DurablePrpControlPlane,
|
||||
|
|
@ -70,6 +74,50 @@ const MAX_NOTIFICATION_BYTES = 4 * 1024 * 1024;
|
|||
const RUNNER_CLIENT_VERSION = "0.3.0";
|
||||
const RUNNER_BOOTSTRAP_TICKET_TTL_MS = 60_000;
|
||||
|
||||
function readLocalProcessStartedAt(pid: number): string | null {
|
||||
if (!Number.isInteger(pid) || pid <= 0) return null;
|
||||
try {
|
||||
if (process.platform === "linux") {
|
||||
return new Date(statSync(`/proc/${pid}`).ctimeMs).toISOString();
|
||||
}
|
||||
if (
|
||||
["darwin", "freebsd", "openbsd", "aix", "sunos"].includes(
|
||||
process.platform,
|
||||
)
|
||||
) {
|
||||
const raw = execFileSync("ps", ["-o", "lstart=", "-p", String(pid)], {
|
||||
encoding: "utf8",
|
||||
timeout: 1_500,
|
||||
windowsHide: true,
|
||||
}).trim();
|
||||
const parsed = new Date(raw);
|
||||
return Number.isNaN(parsed.getTime()) ? null : parsed.toISOString();
|
||||
}
|
||||
if (process.platform === "win32") {
|
||||
const script = [
|
||||
`$target = Get-Process -Id ${pid} -ErrorAction Stop`,
|
||||
"$target.StartTime.ToUniversalTime().ToString('o')",
|
||||
].join("; ");
|
||||
for (const command of ["powershell.exe", "pwsh.exe"]) {
|
||||
try {
|
||||
const raw = execFileSync(
|
||||
command,
|
||||
["-NoLogo", "-NoProfile", "-NonInteractive", "-Command", script],
|
||||
{ encoding: "utf8", timeout: 1_500, windowsHide: true },
|
||||
).trim();
|
||||
const parsed = new Date(raw);
|
||||
if (!Number.isNaN(parsed.getTime())) return parsed.toISOString();
|
||||
} catch {
|
||||
// Try the other supported PowerShell host.
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
// Process exit and restricted process metadata both produce no fingerprint.
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
const CODEX_COLLABORATION_RUNTIME_INSTRUCTIONS = `## Codex-style collaboration
|
||||
|
||||
- Before the first tool call in a turn, send a brief commentary update describing the immediate work you are starting.
|
||||
|
|
@ -307,20 +355,10 @@ function recoveredRunAttachment(state: {
|
|||
.reverse()
|
||||
.find((candidate) => candidate.type === "run.attach");
|
||||
if (!command) return null;
|
||||
let providerIdentityEventIndex = -1;
|
||||
if (command.status === "completed") {
|
||||
for (let index = state.committedEvents.length - 1; index >= 0; index -= 1) {
|
||||
const eventType = state.committedEvents[index]?.eventType;
|
||||
if (
|
||||
eventType === "harness.ready" ||
|
||||
eventType === "session.started" ||
|
||||
eventType === "session.resumed"
|
||||
) {
|
||||
providerIdentityEventIndex = index;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
const providerIdentityEventIndex =
|
||||
command.status === "completed"
|
||||
? latestProviderIdentityEventIndex(state.committedEvents)
|
||||
: -1;
|
||||
return {
|
||||
commandId: command.commandId,
|
||||
status: command.status,
|
||||
|
|
@ -328,6 +366,22 @@ function recoveredRunAttachment(state: {
|
|||
};
|
||||
}
|
||||
|
||||
function latestProviderIdentityEventIndex(
|
||||
events: readonly { eventType: string }[],
|
||||
): number {
|
||||
for (let index = events.length - 1; index >= 0; index -= 1) {
|
||||
const eventType = events[index]?.eventType;
|
||||
if (
|
||||
eventType === "harness.ready" ||
|
||||
eventType === "session.started" ||
|
||||
eventType === "session.resumed"
|
||||
) {
|
||||
return index;
|
||||
}
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
function providerDrainStateFromSnapshot(state: Record<string, unknown>): {
|
||||
pendingEventCount: number;
|
||||
activeProviderTurnId: string | null;
|
||||
|
|
@ -532,9 +586,13 @@ export interface CapabilityRunnerdProcessEvidence {
|
|||
runnerPid: number | null;
|
||||
runnerProcessGroupId: number | null;
|
||||
providerPid: number | null;
|
||||
providerProcessStartedAt: string | null;
|
||||
codexPid: number | null;
|
||||
codexProcessStartedAt: string | null;
|
||||
sidecarPid: number | null;
|
||||
sidecarProcessStartedAt: string | null;
|
||||
agentPid: number | null;
|
||||
agentProcessStartedAt: string | null;
|
||||
providerDriver: string | null;
|
||||
providerVersion: string | null;
|
||||
acpxAgent: QualifiedAcpxAgent | null;
|
||||
|
|
@ -669,6 +727,18 @@ export interface CapabilityRunnerdCodexTransportOptions {
|
|||
readRunnerState?: () => Promise<Record<string, unknown>>;
|
||||
/** Active-connection recovery budget. Omitted for the existing local mode. */
|
||||
runnerReconnectGraceMs?: number;
|
||||
/**
|
||||
* A verified local runner that outlived its controller. Adoption registers
|
||||
* the durable authority and waits for this exact process to reconnect; it
|
||||
* never calls the process launcher while the process remains alive.
|
||||
*/
|
||||
adoptExistingRunner?: {
|
||||
pid: number;
|
||||
processGroupId: number | null;
|
||||
startedAt: string;
|
||||
isAlive: () => Promise<boolean> | boolean;
|
||||
signal?: (signal: NodeJS.Signals) => Promise<boolean> | boolean;
|
||||
};
|
||||
}
|
||||
|
||||
export type RunnerdCodexTransportOptions =
|
||||
|
|
@ -677,6 +747,8 @@ export type RunnerdCodexTransportOptions =
|
|||
export interface CapabilityRunnerdCodexTransport {
|
||||
transport: CodexAppServerTransport;
|
||||
evidence(): Readonly<CapabilityRunnerdProcessEvidence>;
|
||||
/** Relinquish controller authority without stopping the durable runner. */
|
||||
detachControllerForRestart(): Promise<void>;
|
||||
}
|
||||
|
||||
export type RunnerdCodexTransport = CapabilityRunnerdCodexTransport;
|
||||
|
|
@ -1581,6 +1653,7 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
|
|||
});
|
||||
#core: DurablePrpControlPlane | null = null;
|
||||
#handle: RunnerProcessHandle | null = null;
|
||||
#adoptedRunnerMonitor: NodeJS.Timeout | null = null;
|
||||
#pump: NodeJS.Timeout | null = null;
|
||||
#eventIndex = 0;
|
||||
#threadId = "";
|
||||
|
|
@ -1633,9 +1706,13 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
|
|||
runnerPid: null,
|
||||
runnerProcessGroupId: null,
|
||||
providerPid: null,
|
||||
providerProcessStartedAt: null,
|
||||
codexPid: null,
|
||||
codexProcessStartedAt: null,
|
||||
sidecarPid: null,
|
||||
sidecarProcessStartedAt: null,
|
||||
agentPid: null,
|
||||
agentProcessStartedAt: null,
|
||||
providerDriver: null,
|
||||
providerVersion: null,
|
||||
acpxAgent: null,
|
||||
|
|
@ -1746,6 +1823,7 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
|
|||
snapshot.activeProviderTurnId.length > 0
|
||||
? snapshot.activeProviderTurnId
|
||||
: null;
|
||||
if (activeProviderTurnId !== null) this.#turnId = activeProviderTurnId;
|
||||
const recoveredTurns: Array<Record<string, unknown>> =
|
||||
activeProviderTurnId === null
|
||||
? []
|
||||
|
|
@ -1985,7 +2063,7 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
|
|||
);
|
||||
return false;
|
||||
}
|
||||
if (this.#handle?.child.exitCode !== null) return false;
|
||||
if (await this.#runnerHasExited()) return false;
|
||||
await new Promise((resolveWait) => setTimeout(resolveWait, 5));
|
||||
}
|
||||
this.#diagnostic("provider turn stop timed out before runner suspension");
|
||||
|
|
@ -2050,13 +2128,37 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
|
|||
return this.#closePromise;
|
||||
}
|
||||
|
||||
async detachControllerForRestart(): Promise<void> {
|
||||
if (this.#closed) return;
|
||||
this.#closed = true;
|
||||
if (this.#pump !== null) clearInterval(this.#pump);
|
||||
this.#pump = null;
|
||||
if (this.#adoptedRunnerMonitor !== null)
|
||||
clearInterval(this.#adoptedRunnerMonitor);
|
||||
this.#adoptedRunnerMonitor = null;
|
||||
this.#core?.disconnectActiveRunner();
|
||||
if (this.#controlPlaneRelease !== null) await this.#controlPlaneRelease();
|
||||
await this.#core?.stop();
|
||||
this.#controlPlaneRelease = null;
|
||||
this.#handle = null;
|
||||
this.#queue.close();
|
||||
this.#diagnostic(
|
||||
"controller authority detached for restart; durable runner left alive",
|
||||
);
|
||||
}
|
||||
|
||||
async #closeOnce(): Promise<void> {
|
||||
this.#closed = true;
|
||||
if (this.#core !== null && this.#handle !== null) {
|
||||
const adoptedRunner = this.options.adoptExistingRunner;
|
||||
if (
|
||||
this.#core !== null &&
|
||||
(this.#handle !== null || adoptedRunner !== undefined) &&
|
||||
(this.#failure === null || this.#startupComplete)
|
||||
) {
|
||||
// A terminal provider frame can become visible one control loop before
|
||||
// its durable provider suffix is ACKed. Drain it before suspension so a
|
||||
// fresh run authority never inherits the prior run's pending events.
|
||||
if (this.#handle.child.exitCode === null) {
|
||||
if (!(await this.#runnerHasExited())) {
|
||||
const stoppedActiveTurn =
|
||||
await this.#stopActiveProviderTurnBeforeSuspend();
|
||||
await this.#drainSettledProviderEventsBeforeSuspend(
|
||||
|
|
@ -2064,7 +2166,7 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
|
|||
);
|
||||
}
|
||||
const runnerAlreadyStopping =
|
||||
this.#handle.child.exitCode !== null ||
|
||||
(await this.#runnerHasExited()) ||
|
||||
this.#core.store.state.commands.some(
|
||||
(command) =>
|
||||
(command.type === "runner.suspend" ||
|
||||
|
|
@ -2079,15 +2181,26 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
|
|||
this.#core.queueCommand("runner.suspend", {}, undefined, true);
|
||||
}
|
||||
try {
|
||||
const result = await waitForProcess(
|
||||
this.#handle,
|
||||
this.options.closeGraceMs ?? 10_000,
|
||||
);
|
||||
this.#evidence.runnerExited = true;
|
||||
this.#evidence.runnerExitCode = result.code;
|
||||
this.#evidence.runnerSignal = result.signal as NodeJS.Signals | null;
|
||||
if (result.stderr.trim())
|
||||
this.#diagnostic(result.stderr.trim().slice(-4_096));
|
||||
if (this.#handle) {
|
||||
const result = await waitForProcess(
|
||||
this.#handle,
|
||||
this.options.closeGraceMs ?? 10_000,
|
||||
);
|
||||
this.#evidence.runnerExited = true;
|
||||
this.#evidence.runnerExitCode = result.code;
|
||||
this.#evidence.runnerSignal = result.signal as NodeJS.Signals | null;
|
||||
if (result.stderr.trim())
|
||||
this.#diagnostic(result.stderr.trim().slice(-4_096));
|
||||
} else if (adoptedRunner) {
|
||||
const deadline = Date.now() + (this.options.closeGraceMs ?? 10_000);
|
||||
while ((await adoptedRunner.isAlive()) && Date.now() < deadline) {
|
||||
await new Promise((resolveWait) => setTimeout(resolveWait, 25));
|
||||
}
|
||||
if (await adoptedRunner.isAlive()) {
|
||||
await adoptedRunner.signal?.("SIGKILL");
|
||||
}
|
||||
this.#evidence.runnerExited = !(await adoptedRunner.isAlive());
|
||||
}
|
||||
} catch (error) {
|
||||
this.#diagnostic(`runner shutdown failed: ${String(error)}`);
|
||||
}
|
||||
|
|
@ -2119,6 +2232,9 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
|
|||
}
|
||||
if (this.#pump !== null) clearInterval(this.#pump);
|
||||
this.#pump = null;
|
||||
if (this.#adoptedRunnerMonitor !== null)
|
||||
clearInterval(this.#adoptedRunnerMonitor);
|
||||
this.#adoptedRunnerMonitor = null;
|
||||
this.#queue.close();
|
||||
// Ensure a runner that missed or could not finish the graceful lifecycle
|
||||
// command cannot keep the control-plane server alive during teardown.
|
||||
|
|
@ -2524,6 +2640,7 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
|
|||
}),
|
||||
this.options.environment,
|
||||
),
|
||||
diagnosticsDirectory: resolve(this.#root, "diagnostics"),
|
||||
processLauncher: this.options.runnerProcessLauncher,
|
||||
});
|
||||
this.#handle = handle;
|
||||
|
|
@ -2538,7 +2655,7 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
|
|||
}
|
||||
await this.#awaitRegistrationReady(registration?.ready);
|
||||
this.#evidence.runnerPid = handle.child.pid ?? null;
|
||||
this.#evidence.runnerProcessGroupId = null;
|
||||
this.#evidence.runnerProcessGroupId = handle.processGroupId ?? null;
|
||||
this.#publish();
|
||||
this.#pump = setInterval(() => this.#pumpEventsSafely(), 5);
|
||||
await this.#waitCommand("run.prepare");
|
||||
|
|
@ -2798,6 +2915,17 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
|
|||
runAttachment !== null && runAttachment.providerIdentityEventIndex >= 0
|
||||
? runAttachment.providerIdentityEventIndex
|
||||
: committedEvents.length;
|
||||
const adoptedProviderIdentityIndex =
|
||||
latestProviderIdentityEventIndex(committedEvents);
|
||||
if (
|
||||
this.options.adoptExistingRunner &&
|
||||
exactAuthority &&
|
||||
adoptedProviderIdentityIndex >= 0
|
||||
) {
|
||||
this.#applyProviderIdentityEvent(
|
||||
committedEvents[adoptedProviderIdentityIndex]!,
|
||||
);
|
||||
}
|
||||
const registration = this.options.controlPlaneRegistration
|
||||
? await this.options.controlPlaneRegistration(core)
|
||||
: null;
|
||||
|
|
@ -2805,44 +2933,50 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
|
|||
registration?.startupFailureCode ?? "runner_local_connect_failed";
|
||||
if (registration === null) await core.start();
|
||||
else this.#controlPlaneRelease = registration.release;
|
||||
const handle = spawnRunner({
|
||||
connection: registration?.connection ?? {
|
||||
mode: "connect",
|
||||
connectUrl: registration?.connectUrl ?? core.connectUrl,
|
||||
},
|
||||
stateDirectory:
|
||||
this.options.runnerStateDirectory ?? resolve(this.#root, "runner"),
|
||||
identity,
|
||||
ticket: core.issueBootstrapTicket(RUNNER_BOOTSTRAP_TICKET_TTL_MS),
|
||||
maxOutboxBytes: 256 * 1024,
|
||||
p0ReserveBytes: 64 * 1024,
|
||||
maxRuntimeMs: 60 * 60 * 1_000,
|
||||
reconnectGraceMs: this.options.runnerReconnectGraceMs,
|
||||
lifecyclePolicy: this.options.lifecyclePolicy,
|
||||
runnerBinaryPath,
|
||||
runnerVersion: runnerArtifact.version,
|
||||
runnerDigest: runnerArtifact.digest,
|
||||
acpxLaunchProfile: runnerAcpxLaunchProfile,
|
||||
opencodeLaunchProfile: runnerOpenCodeLaunchProfile,
|
||||
environment: withRunnerdProviderTrace(
|
||||
createCapabilityRunnerdProviderEnvironment({
|
||||
provider,
|
||||
options: {
|
||||
...this.options,
|
||||
stateDirectory: this.#root,
|
||||
const adoptedRunner = this.options.adoptExistingRunner;
|
||||
const handle = adoptedRunner
|
||||
? null
|
||||
: spawnRunner({
|
||||
connection: registration?.connection ?? {
|
||||
mode: "connect",
|
||||
connectUrl: registration?.connectUrl ?? core.connectUrl,
|
||||
},
|
||||
stateDirectory:
|
||||
this.options.runnerStateDirectory ?? resolve(this.#root, "runner"),
|
||||
identity,
|
||||
codexHome,
|
||||
runtimeContextPath,
|
||||
hasRuntimeContext: runtimeContext !== null,
|
||||
acpxSidecarPath,
|
||||
}),
|
||||
this.options.environment,
|
||||
),
|
||||
processLauncher: this.options.runnerProcessLauncher,
|
||||
});
|
||||
this.#handle = handle;
|
||||
this.#watchRunner(handle);
|
||||
ticket: core.issueBootstrapTicket(RUNNER_BOOTSTRAP_TICKET_TTL_MS),
|
||||
maxOutboxBytes: 256 * 1024,
|
||||
p0ReserveBytes: 64 * 1024,
|
||||
maxRuntimeMs: 60 * 60 * 1_000,
|
||||
reconnectGraceMs: this.options.runnerReconnectGraceMs,
|
||||
lifecyclePolicy: this.options.lifecyclePolicy,
|
||||
runnerBinaryPath,
|
||||
runnerVersion: runnerArtifact.version,
|
||||
runnerDigest: runnerArtifact.digest,
|
||||
acpxLaunchProfile: runnerAcpxLaunchProfile,
|
||||
opencodeLaunchProfile: runnerOpenCodeLaunchProfile,
|
||||
environment: withRunnerdProviderTrace(
|
||||
createCapabilityRunnerdProviderEnvironment({
|
||||
provider,
|
||||
options: {
|
||||
...this.options,
|
||||
stateDirectory: this.#root,
|
||||
},
|
||||
identity,
|
||||
codexHome,
|
||||
runtimeContextPath,
|
||||
hasRuntimeContext: runtimeContext !== null,
|
||||
acpxSidecarPath,
|
||||
}),
|
||||
this.options.environment,
|
||||
),
|
||||
diagnosticsDirectory: resolve(this.#root, "diagnostics"),
|
||||
processLauncher: this.options.runnerProcessLauncher,
|
||||
});
|
||||
if (handle) {
|
||||
this.#handle = handle;
|
||||
this.#watchRunner(handle);
|
||||
}
|
||||
await registration?.activate?.();
|
||||
if (registration?.failure) {
|
||||
void registration.failure.catch((error: unknown) => {
|
||||
|
|
@ -2852,10 +2986,17 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
|
|||
});
|
||||
}
|
||||
await this.#awaitRegistrationReady(registration?.ready);
|
||||
this.#evidence.runnerPid = handle.child.pid ?? null;
|
||||
this.#evidence.runnerProcessGroupId = null;
|
||||
if (adoptedRunner) {
|
||||
this.#evidence.runnerPid = adoptedRunner.pid;
|
||||
this.#evidence.runnerProcessGroupId = adoptedRunner.processGroupId;
|
||||
this.#watchAdoptedRunner(adoptedRunner);
|
||||
} else {
|
||||
this.#evidence.runnerPid = handle?.child.pid ?? null;
|
||||
this.#evidence.runnerProcessGroupId = handle?.processGroupId ?? null;
|
||||
}
|
||||
this.#publish();
|
||||
this.#pump = setInterval(() => this.#pumpEventsSafely(), 5);
|
||||
if (adoptedRunner) await this.#awaitAdoptedRunnerConnection(adoptedRunner);
|
||||
if (runAttachment) {
|
||||
await this.#waitCommand("run.attach", runAttachment.commandId);
|
||||
}
|
||||
|
|
@ -2890,7 +3031,7 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
|
|||
this.#throwIfFailed();
|
||||
this.#pumpEvents();
|
||||
if (this.#turnId !== pendingTurnId) break;
|
||||
if (this.#handle?.child.exitCode !== null)
|
||||
if (await this.#runnerHasExited())
|
||||
throw new Error("runnerd exited before provider turn startup");
|
||||
await new Promise((resolveWait) => setTimeout(resolveWait, 10));
|
||||
}
|
||||
|
|
@ -2978,7 +3119,7 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
|
|||
this.#evidence.providerPid !== null)
|
||||
)
|
||||
return;
|
||||
if (this.#handle?.child.exitCode !== null)
|
||||
if (await this.#runnerHasExited())
|
||||
throw new Error("runnerd exited before provider startup");
|
||||
await new Promise((resolveWait) => setTimeout(resolveWait, 10));
|
||||
}
|
||||
|
|
@ -3000,7 +3141,7 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
|
|||
`PRP command ${type} ${command.status}: ${JSON.stringify(command.result)}`,
|
||||
);
|
||||
}
|
||||
if (this.#handle?.child.exitCode !== null)
|
||||
if (await this.#runnerHasExited())
|
||||
throw new Error(`runnerd exited while waiting for ${type}`);
|
||||
await new Promise((resolveWait) => setTimeout(resolveWait, 10));
|
||||
}
|
||||
|
|
@ -3034,58 +3175,7 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
|
|||
event.eventType === "session.started" ||
|
||||
event.eventType === "session.resumed"
|
||||
) {
|
||||
const started = record(record(event.envelope.payload).payload);
|
||||
const runtimeIdentity = record(started.runtimeIdentity);
|
||||
const descriptor = record(started.providerDescriptor);
|
||||
const {
|
||||
processId: pid,
|
||||
threadId,
|
||||
sessionId,
|
||||
} = resolveRunnerdSessionIdentity(started);
|
||||
const providerIdentity = record(started.providerIdentity);
|
||||
if (pid !== null) {
|
||||
this.#evidence.providerPid = pid;
|
||||
if (descriptor.driver === "acpx_runtime")
|
||||
this.#evidence.sidecarPid = pid;
|
||||
else this.#evidence.codexPid = pid;
|
||||
}
|
||||
if (typeof descriptor.driver === "string")
|
||||
this.#evidence.providerDriver = descriptor.driver;
|
||||
if (typeof descriptor.providerVersion === "string")
|
||||
this.#evidence.providerVersion = descriptor.providerVersion;
|
||||
if (
|
||||
descriptor.agent === "pi" ||
|
||||
descriptor.agent === "claude" ||
|
||||
descriptor.agent === "codex"
|
||||
)
|
||||
this.#evidence.acpxAgent = descriptor.agent;
|
||||
if (typeof descriptor.agentServerVersion === "string")
|
||||
this.#evidence.agentServerVersion = descriptor.agentServerVersion;
|
||||
if (typeof descriptor.agentRuntimeVersion === "string")
|
||||
this.#evidence.agentRuntimeVersion = descriptor.agentRuntimeVersion;
|
||||
if (typeof descriptor.acpProtocolVersion === "number")
|
||||
this.#evidence.acpProtocolVersion = descriptor.acpProtocolVersion;
|
||||
if (typeof descriptor.agentProcessId === "number")
|
||||
this.#evidence.agentPid = descriptor.agentProcessId;
|
||||
if (
|
||||
runtimeIdentity.executionKind === "local_process" ||
|
||||
runtimeIdentity.executionKind === "remote_service"
|
||||
) {
|
||||
this.#evidence.providerExecutionKind = runtimeIdentity.executionKind;
|
||||
}
|
||||
if (runtimeIdentity.service === "anthropic_managed_agents") {
|
||||
this.#evidence.providerService = "anthropic_managed_agents";
|
||||
} else if (
|
||||
runtimeIdentity.service === "aws_bedrock_agentcore_harness"
|
||||
) {
|
||||
this.#evidence.providerService = "aws_bedrock_agentcore_harness";
|
||||
}
|
||||
if (threadId !== null) this.#threadId = threadId;
|
||||
if (sessionId !== null) this.#sessionId = sessionId;
|
||||
if (typeof providerIdentity.kind === "string") {
|
||||
this.#providerIdentity = structuredClone(providerIdentity);
|
||||
}
|
||||
this.#publish();
|
||||
this.#applyProviderIdentityEvent(event);
|
||||
continue;
|
||||
}
|
||||
if (event.eventType === "harness.diagnostic") {
|
||||
|
|
@ -3096,6 +3186,9 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
|
|||
typeof diagnostic.pid === "number"
|
||||
) {
|
||||
this.#evidence.agentPid = diagnostic.pid;
|
||||
this.#evidence.agentProcessStartedAt = readLocalProcessStartedAt(
|
||||
diagnostic.pid,
|
||||
);
|
||||
this.#publish();
|
||||
}
|
||||
}
|
||||
|
|
@ -3301,6 +3394,70 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
|
|||
}
|
||||
}
|
||||
|
||||
#applyProviderIdentityEvent(event: DurableRecoveryCommittedEvent): void {
|
||||
const started = record(record(event.envelope.payload).payload);
|
||||
const runtimeIdentity = record(started.runtimeIdentity);
|
||||
const descriptor = record(started.providerDescriptor);
|
||||
const {
|
||||
processId: pid,
|
||||
threadId,
|
||||
sessionId,
|
||||
} = resolveRunnerdSessionIdentity(started);
|
||||
const providerIdentity = record(started.providerIdentity);
|
||||
if (pid !== null) {
|
||||
this.#evidence.providerPid = pid;
|
||||
this.#evidence.providerProcessStartedAt = readLocalProcessStartedAt(pid);
|
||||
if (descriptor.driver === "acpx_runtime") {
|
||||
this.#evidence.sidecarPid = pid;
|
||||
this.#evidence.sidecarProcessStartedAt =
|
||||
this.#evidence.providerProcessStartedAt;
|
||||
} else {
|
||||
this.#evidence.codexPid = pid;
|
||||
this.#evidence.codexProcessStartedAt =
|
||||
this.#evidence.providerProcessStartedAt;
|
||||
}
|
||||
}
|
||||
if (typeof descriptor.driver === "string")
|
||||
this.#evidence.providerDriver = descriptor.driver;
|
||||
if (typeof descriptor.providerVersion === "string")
|
||||
this.#evidence.providerVersion = descriptor.providerVersion;
|
||||
if (
|
||||
descriptor.agent === "pi" ||
|
||||
descriptor.agent === "claude" ||
|
||||
descriptor.agent === "codex"
|
||||
)
|
||||
this.#evidence.acpxAgent = descriptor.agent;
|
||||
if (typeof descriptor.agentServerVersion === "string")
|
||||
this.#evidence.agentServerVersion = descriptor.agentServerVersion;
|
||||
if (typeof descriptor.agentRuntimeVersion === "string")
|
||||
this.#evidence.agentRuntimeVersion = descriptor.agentRuntimeVersion;
|
||||
if (typeof descriptor.acpProtocolVersion === "number")
|
||||
this.#evidence.acpProtocolVersion = descriptor.acpProtocolVersion;
|
||||
if (typeof descriptor.agentProcessId === "number") {
|
||||
this.#evidence.agentPid = descriptor.agentProcessId;
|
||||
this.#evidence.agentProcessStartedAt = readLocalProcessStartedAt(
|
||||
descriptor.agentProcessId,
|
||||
);
|
||||
}
|
||||
if (
|
||||
runtimeIdentity.executionKind === "local_process" ||
|
||||
runtimeIdentity.executionKind === "remote_service"
|
||||
) {
|
||||
this.#evidence.providerExecutionKind = runtimeIdentity.executionKind;
|
||||
}
|
||||
if (runtimeIdentity.service === "anthropic_managed_agents") {
|
||||
this.#evidence.providerService = "anthropic_managed_agents";
|
||||
} else if (runtimeIdentity.service === "aws_bedrock_agentcore_harness") {
|
||||
this.#evidence.providerService = "aws_bedrock_agentcore_harness";
|
||||
}
|
||||
if (threadId !== null) this.#threadId = threadId;
|
||||
if (sessionId !== null) this.#sessionId = sessionId;
|
||||
if (typeof providerIdentity.kind === "string") {
|
||||
this.#providerIdentity = structuredClone(providerIdentity);
|
||||
}
|
||||
this.#publish();
|
||||
}
|
||||
|
||||
#flushPendingTraceRehydrations(): void {
|
||||
const tracePath = this.options.environment?.PAPERCLIP_PROVIDER_TRACE_PATH;
|
||||
if (!tracePath) return;
|
||||
|
|
@ -3343,6 +3500,80 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
|
|||
}
|
||||
}
|
||||
|
||||
async #runnerHasExited(): Promise<boolean> {
|
||||
if (this.#handle) return this.#handle.child.exitCode !== null;
|
||||
const adoptedRunner = this.options.adoptExistingRunner;
|
||||
if (!adoptedRunner) return true;
|
||||
try {
|
||||
return !(await adoptedRunner.isAlive());
|
||||
} catch {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
#watchAdoptedRunner(
|
||||
adoptedRunner: NonNullable<
|
||||
CapabilityRunnerdCodexTransportOptions["adoptExistingRunner"]
|
||||
>,
|
||||
): void {
|
||||
let checking = false;
|
||||
this.#adoptedRunnerMonitor = setInterval(() => {
|
||||
if (checking || this.#closed) return;
|
||||
checking = true;
|
||||
void Promise.resolve(adoptedRunner.isAlive())
|
||||
.then((alive) => {
|
||||
if (alive || this.#closed) return;
|
||||
this.#evidence.runnerExited = true;
|
||||
this.#publish();
|
||||
this.#failTransport(
|
||||
new Error(
|
||||
"native_adopted_runner_exited: the verified runner exited while its durable authority was active",
|
||||
),
|
||||
);
|
||||
})
|
||||
.catch(() => {
|
||||
if (!this.#closed) {
|
||||
this.#failTransport(
|
||||
new Error(
|
||||
"native_adopted_runner_identity_unverifiable: runner liveness could not be revalidated",
|
||||
),
|
||||
);
|
||||
}
|
||||
})
|
||||
.finally(() => {
|
||||
checking = false;
|
||||
});
|
||||
}, 250);
|
||||
this.#adoptedRunnerMonitor.unref?.();
|
||||
}
|
||||
|
||||
async #awaitAdoptedRunnerConnection(
|
||||
adoptedRunner: NonNullable<
|
||||
CapabilityRunnerdCodexTransportOptions["adoptExistingRunner"]
|
||||
>,
|
||||
): Promise<void> {
|
||||
const core = this.#core;
|
||||
if (!core) throw new Error("native_runner_authority_unavailable");
|
||||
this.#diagnostic(
|
||||
`waiting for adopted runner ${adoptedRunner.pid} to authenticate to its durable PRP authority`,
|
||||
);
|
||||
while (core.activeRunnerConnectionCount() !== 1) {
|
||||
this.#throwIfFailed();
|
||||
if (!(await adoptedRunner.isAlive())) {
|
||||
throw new Error(
|
||||
"native_adopted_runner_exited: runner exited before PRP authentication",
|
||||
);
|
||||
}
|
||||
await Promise.race([
|
||||
new Promise<void>((resolveWait) => setTimeout(resolveWait, 25)),
|
||||
this.#failureSignal,
|
||||
]);
|
||||
}
|
||||
this.#diagnostic(
|
||||
`adopted runner ${adoptedRunner.pid} authenticated to its durable PRP authority`,
|
||||
);
|
||||
}
|
||||
|
||||
#watchRunner(handle: RunnerProcessHandle): void {
|
||||
void handle.completion.then(
|
||||
(result) => {
|
||||
|
|
@ -3465,6 +3696,8 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
|
|||
this.#evidence.runnerExitCode = null;
|
||||
this.#evidence.runnerSignal = null;
|
||||
this.#evidence.runnerPid = recoveredHandle.child.pid ?? null;
|
||||
this.#evidence.runnerProcessGroupId =
|
||||
recoveredHandle.processGroupId ?? null;
|
||||
this.#publish();
|
||||
|
||||
let processSettled = false;
|
||||
|
|
@ -3566,7 +3799,11 @@ export function createCapabilityRunnerdCodexTransport(
|
|||
options: CapabilityRunnerdCodexTransportOptions = {},
|
||||
): CapabilityRunnerdCodexTransport {
|
||||
const transport = new DurablePrpCodexTransport(options);
|
||||
return { transport, evidence: () => transport.evidence() };
|
||||
return {
|
||||
transport,
|
||||
evidence: () => transport.evidence(),
|
||||
detachControllerForRestart: () => transport.detachControllerForRestart(),
|
||||
};
|
||||
}
|
||||
|
||||
export const createRunnerdCodexTransport =
|
||||
|
|
|
|||
|
|
@ -14,6 +14,10 @@ import { applyDevRunnerOptions } from "./dev-runner-options.ts";
|
|||
import { collectWatchedSnapshot as collectDevServerWatchedSnapshot, diffSnapshots } from "./dev-runner-snapshot.mjs";
|
||||
import { createDevServiceIdentity, repoRoot } from "./dev-service-profile.ts";
|
||||
import { bootstrapDevRunnerWorktreeEnv, isWorktreeSeedPending } from "../server/src/dev-runner-worktree.ts";
|
||||
import {
|
||||
readDevServerRestartRequest,
|
||||
removeDevServerRestartRequest,
|
||||
} from "../server/src/dev-server-status.ts";
|
||||
import {
|
||||
findAdoptableLocalService,
|
||||
removeLocalServiceRegistryRecord,
|
||||
|
|
@ -331,10 +335,9 @@ function clearDevServerStatus() {
|
|||
rmSync(devServerRestartRequestFilePath, { force: true });
|
||||
}
|
||||
|
||||
function consumeDevServerRestartRequest() {
|
||||
if (mode !== "dev" || !existsSync(devServerRestartRequestFilePath)) return false;
|
||||
rmSync(devServerRestartRequestFilePath, { force: true });
|
||||
return true;
|
||||
function getDevServerRestartRequest() {
|
||||
if (mode !== "dev" || !existsSync(devServerRestartRequestFilePath)) return null;
|
||||
return readDevServerRestartRequest(env);
|
||||
}
|
||||
|
||||
async function updateDevServiceRecord(extra?: Record<string, unknown>) {
|
||||
|
|
@ -747,8 +750,8 @@ async function startServerChild() {
|
|||
|
||||
async function maybeAutoRestartChild() {
|
||||
if (mode !== "dev" || restartInFlight || !child) return;
|
||||
const manualRestartRequested = consumeDevServerRestartRequest();
|
||||
if (!manualRestartRequested && dirtyPaths.size === 0 && pendingMigrations.length === 0) return;
|
||||
const manualRestartRequest = getDevServerRestartRequest();
|
||||
if (!manualRestartRequest && dirtyPaths.size === 0 && pendingMigrations.length === 0) return;
|
||||
|
||||
restartInFlight = true;
|
||||
let health: { devServer?: { enabled?: boolean; autoRestartEnabled?: boolean; activeRunCount?: number } } | null = null;
|
||||
|
|
@ -764,11 +767,30 @@ async function maybeAutoRestartChild() {
|
|||
restartInFlight = false;
|
||||
return;
|
||||
}
|
||||
if (!manualRestartRequested && devServer.autoRestartEnabled !== true) {
|
||||
const observedServerIdentity =
|
||||
typeof (health as { serverInfo?: { processStartedAt?: unknown } })
|
||||
.serverInfo?.processStartedAt === "string"
|
||||
? (health as { serverInfo: { processStartedAt: string } }).serverInfo
|
||||
.processStartedAt
|
||||
: null;
|
||||
if (
|
||||
manualRestartRequest?.previousServerIdentity &&
|
||||
observedServerIdentity !== manualRestartRequest.previousServerIdentity
|
||||
) {
|
||||
removeDevServerRestartRequest(
|
||||
manualRestartRequest.requestId
|
||||
? { requestId: manualRestartRequest.requestId }
|
||||
: undefined,
|
||||
env,
|
||||
);
|
||||
restartInFlight = false;
|
||||
return;
|
||||
}
|
||||
if (!manualRestartRequested && (devServer.activeRunCount ?? 0) > 0) {
|
||||
if (!manualRestartRequest && devServer.autoRestartEnabled !== true) {
|
||||
restartInFlight = false;
|
||||
return;
|
||||
}
|
||||
if (!manualRestartRequest && (devServer.activeRunCount ?? 0) > 0) {
|
||||
restartInFlight = false;
|
||||
return;
|
||||
}
|
||||
|
|
@ -780,7 +802,26 @@ async function maybeAutoRestartChild() {
|
|||
exitOnDecline: false,
|
||||
});
|
||||
await stopChildForRestart();
|
||||
const restartRequestConsumed = manualRestartRequest
|
||||
? removeDevServerRestartRequest(
|
||||
manualRestartRequest.requestId
|
||||
? { requestId: manualRestartRequest.requestId }
|
||||
: undefined,
|
||||
env,
|
||||
)
|
||||
: true;
|
||||
await startServerChild();
|
||||
if (manualRestartRequest && !restartRequestConsumed) {
|
||||
// A live writer may briefly hold the request lock. Starting the child is
|
||||
// still correct because the requested restart already happened; retry
|
||||
// correlated cleanup afterward without terminating the supervisor.
|
||||
removeDevServerRestartRequest(
|
||||
manualRestartRequest.requestId
|
||||
? { requestId: manualRestartRequest.requestId }
|
||||
: undefined,
|
||||
env,
|
||||
);
|
||||
}
|
||||
} catch (error) {
|
||||
const err = toError(error, "Auto-restart failed");
|
||||
process.stderr.write(`${err.stack ?? err.message}\n`);
|
||||
|
|
|
|||
|
|
@ -1,10 +1,19 @@
|
|||
import { existsSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
|
||||
import {
|
||||
existsSync,
|
||||
mkdirSync,
|
||||
mkdtempSync,
|
||||
readFileSync,
|
||||
rmSync,
|
||||
writeFileSync,
|
||||
} from "node:fs";
|
||||
import os from "node:os";
|
||||
import path from "node:path";
|
||||
import { afterEach, describe, expect, it } from "vitest";
|
||||
import {
|
||||
getDevServerRestartRequestFilePath,
|
||||
readDevServerRestartRequest,
|
||||
readPersistedDevServerStatus,
|
||||
removeDevServerRestartRequest,
|
||||
toDevServerHealthStatus,
|
||||
writeDevServerRestartRequest,
|
||||
} from "../dev-server-status.js";
|
||||
|
|
@ -36,7 +45,11 @@ describe("dev server status helpers", () => {
|
|||
lastRestartAt: "2026-03-20T11:30:00.000Z",
|
||||
});
|
||||
|
||||
expect(readPersistedDevServerStatus({ PAPERCLIP_DEV_SERVER_STATUS_FILE: filePath })).toEqual({
|
||||
expect(
|
||||
readPersistedDevServerStatus({
|
||||
PAPERCLIP_DEV_SERVER_STATUS_FILE: filePath,
|
||||
}),
|
||||
).toEqual({
|
||||
dirty: true,
|
||||
lastChangedAt: "2026-03-20T12:00:00.000Z",
|
||||
changedPathCount: 4,
|
||||
|
|
@ -76,7 +89,11 @@ describe("dev server status helpers", () => {
|
|||
pendingMigrations: [],
|
||||
});
|
||||
|
||||
expect(readPersistedDevServerStatus({ PAPERCLIP_DEV_SERVER_STATUS_FILE: filePath })).toBeNull();
|
||||
expect(
|
||||
readPersistedDevServerStatus({
|
||||
PAPERCLIP_DEV_SERVER_STATUS_FILE: filePath,
|
||||
}),
|
||||
).toBeNull();
|
||||
});
|
||||
|
||||
it("writes restart requests next to the persisted status file", () => {
|
||||
|
|
@ -87,17 +104,106 @@ describe("dev server status helpers", () => {
|
|||
});
|
||||
|
||||
const env = { PAPERCLIP_DEV_SERVER_STATUS_FILE: filePath };
|
||||
expect(writeDevServerRestartRequest({
|
||||
requestedAt: "2026-03-20T12:05:00.000Z",
|
||||
reason: "manual_restart_now",
|
||||
}, env)).toBe(true);
|
||||
expect(
|
||||
writeDevServerRestartRequest(
|
||||
{
|
||||
requestedAt: "2026-03-20T12:05:00.000Z",
|
||||
reason: "manual_restart_now",
|
||||
},
|
||||
env,
|
||||
),
|
||||
).toBe(true);
|
||||
|
||||
const requestPath = getDevServerRestartRequestFilePath(env);
|
||||
expect(requestPath).toBe(path.join(path.dirname(filePath), "dev-server-restart-request.json"));
|
||||
expect(requestPath).toBe(
|
||||
path.join(path.dirname(filePath), "dev-server-restart-request.json"),
|
||||
);
|
||||
expect(requestPath && existsSync(requestPath)).toBe(true);
|
||||
expect(JSON.parse(readFileSync(requestPath!, "utf8"))).toEqual({
|
||||
requestedAt: "2026-03-20T12:05:00.000Z",
|
||||
reason: "manual_restart_now",
|
||||
});
|
||||
});
|
||||
|
||||
it("correlates restart request cleanup so stale consumers cannot remove a replacement", () => {
|
||||
const filePath = createTempStatusFile({ dirty: true });
|
||||
const env = { PAPERCLIP_DEV_SERVER_STATUS_FILE: filePath };
|
||||
expect(
|
||||
writeDevServerRestartRequest(
|
||||
{
|
||||
requestedAt: "2026-09-04T12:00:00.000Z",
|
||||
reason: "manual_restart_now",
|
||||
requestId: "restart-new",
|
||||
mode: "hot",
|
||||
previousServerIdentity: "server-start-new",
|
||||
},
|
||||
env,
|
||||
),
|
||||
).toBe(true);
|
||||
|
||||
removeDevServerRestartRequest({ requestId: "restart-stale" }, env);
|
||||
expect(readDevServerRestartRequest(env)).toEqual({
|
||||
requestedAt: "2026-09-04T12:00:00.000Z",
|
||||
reason: "manual_restart_now",
|
||||
requestId: "restart-new",
|
||||
mode: "hot",
|
||||
previousServerIdentity: "server-start-new",
|
||||
});
|
||||
|
||||
removeDevServerRestartRequest({ requestId: "restart-new" }, env);
|
||||
expect(readDevServerRestartRequest(env)).toBeNull();
|
||||
});
|
||||
|
||||
it("immediately recovers an owner-less lock from an interrupted publisher", () => {
|
||||
const filePath = createTempStatusFile({ dirty: true });
|
||||
const env = { PAPERCLIP_DEV_SERVER_STATUS_FILE: filePath };
|
||||
const requestPath = getDevServerRestartRequestFilePath(env)!;
|
||||
const lockPath = `${requestPath}.lock`;
|
||||
mkdirSync(lockPath);
|
||||
|
||||
writeDevServerRestartRequest(
|
||||
{
|
||||
requestedAt: "2026-09-04T12:00:01.000Z",
|
||||
reason: "manual_restart_now",
|
||||
requestId: "restart-after-crash",
|
||||
mode: "hot",
|
||||
},
|
||||
env,
|
||||
);
|
||||
|
||||
expect(readDevServerRestartRequest(env)).toMatchObject({
|
||||
requestId: "restart-after-crash",
|
||||
requestedAt: "2026-09-04T12:00:01.000Z",
|
||||
});
|
||||
expect(existsSync(lockPath)).toBe(false);
|
||||
});
|
||||
|
||||
it("preserves the request instead of throwing when a live writer holds the lock", () => {
|
||||
const filePath = createTempStatusFile({ dirty: true });
|
||||
const env = { PAPERCLIP_DEV_SERVER_STATUS_FILE: filePath };
|
||||
writeDevServerRestartRequest(
|
||||
{
|
||||
requestedAt: "2026-09-04T12:00:02.000Z",
|
||||
reason: "manual_restart_now",
|
||||
requestId: "restart-contended",
|
||||
mode: "hot",
|
||||
},
|
||||
env,
|
||||
);
|
||||
const requestPath = getDevServerRestartRequestFilePath(env)!;
|
||||
const lockPath = `${requestPath}.lock`;
|
||||
mkdirSync(lockPath);
|
||||
writeFileSync(
|
||||
path.join(lockPath, "owner.json"),
|
||||
JSON.stringify({ pid: process.pid, acquiredAt: new Date().toISOString() }),
|
||||
"utf8",
|
||||
);
|
||||
|
||||
expect(
|
||||
removeDevServerRestartRequest({ requestId: "restart-contended" }, env),
|
||||
).toBe(false);
|
||||
expect(readDevServerRestartRequest(env)).toMatchObject({
|
||||
requestId: "restart-contended",
|
||||
});
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -6,6 +6,8 @@ import request from "supertest";
|
|||
import { afterEach, describe, expect, it, vi } from "vitest";
|
||||
import type { Db } from "@paperclipai/db";
|
||||
import { healthRoutes } from "../routes/health.js";
|
||||
import * as devServerStatus from "../dev-server-status.js";
|
||||
import { resolveHotRestartIntentPath } from "../services/hot-restart.js";
|
||||
|
||||
const tempDirs: string[] = [];
|
||||
|
||||
|
|
@ -18,6 +20,7 @@ function createDevServerStatusFile(payload: unknown) {
|
|||
}
|
||||
|
||||
afterEach(() => {
|
||||
vi.restoreAllMocks();
|
||||
for (const dir of tempDirs.splice(0)) {
|
||||
rmSync(dir, { recursive: true, force: true });
|
||||
}
|
||||
|
|
@ -138,6 +141,7 @@ describe("GET /health dev-server supervisor access", () => {
|
|||
describe("POST /health/dev-server/restart", () => {
|
||||
it("records a manual restart request for the dev runner", async () => {
|
||||
const previousFile = process.env.PAPERCLIP_DEV_SERVER_STATUS_FILE;
|
||||
const previousHome = process.env.PAPERCLIP_HOME;
|
||||
process.env.PAPERCLIP_DEV_SERVER_STATUS_FILE = createDevServerStatusFile({
|
||||
dirty: true,
|
||||
lastChangedAt: "2026-03-20T12:00:00.000Z",
|
||||
|
|
@ -146,15 +150,29 @@ describe("POST /health/dev-server/restart", () => {
|
|||
pendingMigrations: [],
|
||||
lastRestartAt: "2026-03-20T11:30:00.000Z",
|
||||
});
|
||||
process.env.PAPERCLIP_HOME = path.dirname(
|
||||
process.env.PAPERCLIP_DEV_SERVER_STATUS_FILE,
|
||||
);
|
||||
|
||||
try {
|
||||
const app = express();
|
||||
app.use("/health", healthRoutes(undefined));
|
||||
const db = {
|
||||
select: vi.fn(() => ({
|
||||
from: vi.fn(() => ({
|
||||
where: vi.fn().mockResolvedValue([]),
|
||||
})),
|
||||
})),
|
||||
} as unknown as Db;
|
||||
app.use("/health", healthRoutes(db));
|
||||
|
||||
const res = await request(app).post("/health/dev-server/restart");
|
||||
|
||||
expect(res.status).toBe(202);
|
||||
expect(res.body).toEqual({ status: "restart_requested" });
|
||||
expect(res.body).toMatchObject({
|
||||
status: "restart_requested",
|
||||
mode: "hot",
|
||||
requestId: expect.any(String),
|
||||
});
|
||||
|
||||
const requestPath = path.join(
|
||||
path.dirname(process.env.PAPERCLIP_DEV_SERVER_STATUS_FILE),
|
||||
|
|
@ -163,6 +181,8 @@ describe("POST /health/dev-server/restart", () => {
|
|||
expect(existsSync(requestPath)).toBe(true);
|
||||
expect(JSON.parse(readFileSync(requestPath, "utf8"))).toMatchObject({
|
||||
reason: "manual_restart_now",
|
||||
mode: "hot",
|
||||
requestId: res.body.requestId,
|
||||
});
|
||||
} finally {
|
||||
if (previousFile === undefined) {
|
||||
|
|
@ -170,6 +190,47 @@ describe("POST /health/dev-server/restart", () => {
|
|||
} else {
|
||||
process.env.PAPERCLIP_DEV_SERVER_STATUS_FILE = previousFile;
|
||||
}
|
||||
if (previousHome === undefined) delete process.env.PAPERCLIP_HOME;
|
||||
else process.env.PAPERCLIP_HOME = previousHome;
|
||||
}
|
||||
});
|
||||
|
||||
it("rolls back the hot intent when the supervisor request cannot be written", async () => {
|
||||
const previousFile = process.env.PAPERCLIP_DEV_SERVER_STATUS_FILE;
|
||||
const previousHome = process.env.PAPERCLIP_HOME;
|
||||
process.env.PAPERCLIP_DEV_SERVER_STATUS_FILE = createDevServerStatusFile({
|
||||
dirty: true,
|
||||
changedPathCount: 1,
|
||||
changedPathsSample: ["server/src/routes/health.ts"],
|
||||
pendingMigrations: [],
|
||||
});
|
||||
const home = mkdtempSync(path.join(os.tmpdir(), "paperclip-health-restart-home-"));
|
||||
tempDirs.push(home);
|
||||
process.env.PAPERCLIP_HOME = home;
|
||||
vi.spyOn(devServerStatus, "writeDevServerRestartRequest").mockReturnValue(false);
|
||||
|
||||
try {
|
||||
const app = express();
|
||||
const db = {
|
||||
select: vi.fn(() => ({
|
||||
from: vi.fn(() => ({ where: vi.fn().mockResolvedValue([]) })),
|
||||
})),
|
||||
} as unknown as Db;
|
||||
app.use("/health", healthRoutes(db));
|
||||
|
||||
const res = await request(app).post("/health/dev-server/restart");
|
||||
|
||||
expect(res.status).toBe(404);
|
||||
expect(res.body).toEqual({ error: "dev_server_supervisor_unavailable" });
|
||||
expect(existsSync(resolveHotRestartIntentPath(home))).toBe(false);
|
||||
} finally {
|
||||
if (previousFile === undefined) {
|
||||
delete process.env.PAPERCLIP_DEV_SERVER_STATUS_FILE;
|
||||
} else {
|
||||
process.env.PAPERCLIP_DEV_SERVER_STATUS_FILE = previousFile;
|
||||
}
|
||||
if (previousHome === undefined) delete process.env.PAPERCLIP_HOME;
|
||||
else process.env.PAPERCLIP_HOME = previousHome;
|
||||
}
|
||||
});
|
||||
|
||||
|
|
|
|||
|
|
@ -46,6 +46,7 @@ import {
|
|||
issueTreeHolds,
|
||||
issueWorkProducts,
|
||||
issues,
|
||||
nativeRunFinalizations,
|
||||
plugins,
|
||||
projects,
|
||||
projectWorkspaces,
|
||||
|
|
@ -118,6 +119,7 @@ import {
|
|||
} from "../services/heartbeat.ts";
|
||||
import {
|
||||
readHotRestartIntent,
|
||||
readProcessStartedAt,
|
||||
resolveLegacyHotRestartIntentPath,
|
||||
resolveHotRestartReportPath,
|
||||
writeHotRestartIntent,
|
||||
|
|
@ -433,6 +435,7 @@ describeEmbeddedPostgres("heartbeat orphaned process recovery", () => {
|
|||
await db.delete(issueRecoveryActions);
|
||||
await db.delete(issueTreeHoldMembers);
|
||||
await db.delete(issueTreeHolds);
|
||||
await db.delete(nativeRunFinalizations);
|
||||
for (let attempt = 0; attempt < 5; attempt += 1) {
|
||||
await db.delete(issueComments);
|
||||
await db.delete(issueDocuments);
|
||||
|
|
@ -1442,6 +1445,53 @@ describeEmbeddedPostgres("heartbeat orphaned process recovery", () => {
|
|||
expect(wakeup?.status).toBe("claimed");
|
||||
});
|
||||
|
||||
it("does not reap a retryable native run while its same-run recovery path owns it", async () => {
|
||||
const { companyId, agentId, runId, issueId, wakeupRequestId } =
|
||||
await seedRunFixture({
|
||||
adapterType: "paperclip_runner",
|
||||
runtimeMode: "native",
|
||||
});
|
||||
const nextAttemptAt = new Date(Date.now() + 60_000);
|
||||
await db
|
||||
.update(heartbeatRuns)
|
||||
.set({ nativeIssueId: issueId, nativePhase: "retryable_failure" })
|
||||
.where(eq(heartbeatRuns.id, runId));
|
||||
await db.insert(nativeRunFinalizations).values({
|
||||
runId,
|
||||
companyId,
|
||||
issueId,
|
||||
phase: "retryable_failure",
|
||||
attempt: 1,
|
||||
nextAttemptAt,
|
||||
recoveryState: "resuming_session",
|
||||
});
|
||||
|
||||
const result = await heartbeatService(db).reapOrphanedRuns();
|
||||
|
||||
expect(result).toEqual({ reaped: 0, runIds: [] });
|
||||
expect(await heartbeatService(db).getRun(runId)).toMatchObject({
|
||||
status: "running",
|
||||
nativePhase: "retryable_failure",
|
||||
});
|
||||
await expect(
|
||||
db
|
||||
.select({ id: heartbeatRuns.id })
|
||||
.from(heartbeatRuns)
|
||||
.where(
|
||||
and(
|
||||
eq(heartbeatRuns.agentId, agentId),
|
||||
eq(heartbeatRuns.retryOfRunId, runId),
|
||||
),
|
||||
),
|
||||
).resolves.toHaveLength(0);
|
||||
await expect(
|
||||
db
|
||||
.select({ status: agentWakeupRequests.status })
|
||||
.from(agentWakeupRequests)
|
||||
.where(eq(agentWakeupRequests.id, wakeupRequestId)),
|
||||
).resolves.toEqual([{ status: "claimed" }]);
|
||||
});
|
||||
|
||||
it("does not grant a dead native run legacy retry authority after adapter reassignment", async () => {
|
||||
const { agentId, runId } = await seedRunFixture({
|
||||
adapterType: "paperclip_runner",
|
||||
|
|
@ -2268,12 +2318,14 @@ describeEmbeddedPostgres("heartbeat orphaned process recovery", () => {
|
|||
return row?.status === "running" && row.processPid ? row : null;
|
||||
}),
|
||||
);
|
||||
const observedProcessStartedAt = await readProcessStartedAt(spawnedPid!);
|
||||
expect(observedProcessStartedAt).not.toBeNull();
|
||||
expect(running).toMatchObject({
|
||||
id: runId,
|
||||
status: "running",
|
||||
processPid: spawnedPid,
|
||||
processGroupId: null,
|
||||
processStartedAt: new Date("2026-07-30T07:00:00.000Z"),
|
||||
processStartedAt: new Date(observedProcessStartedAt!),
|
||||
});
|
||||
|
||||
await withTempPaperclipHome(async (home) => {
|
||||
|
|
@ -2524,6 +2576,64 @@ describeEmbeddedPostgres("heartbeat orphaned process recovery", () => {
|
|||
expect(issue?.executionRunId).toBe(retryRun?.id);
|
||||
});
|
||||
|
||||
it("suspends native Paperclip Runner ownership on graceful restart without cancelling or creating a retry run", async () => {
|
||||
const { agentId, runId, issueId, wakeupRequestId } =
|
||||
await seedRunFixture({
|
||||
adapterType: "paperclip_runner",
|
||||
agentStatus: "running",
|
||||
runtimeMode: "native",
|
||||
});
|
||||
await db
|
||||
.update(heartbeatRuns)
|
||||
.set({ nativeIssueId: issueId })
|
||||
.where(eq(heartbeatRuns.id, runId));
|
||||
await db.insert(nativeRunFinalizations).values({
|
||||
runId,
|
||||
companyId: (await heartbeatService(db).getRun(runId))!.companyId,
|
||||
issueId,
|
||||
phase: "observed",
|
||||
});
|
||||
|
||||
const result = await heartbeatService(db).drainRunningRunsForShutdown(
|
||||
"SIGTERM",
|
||||
new Date("2026-09-04T12:00:00.000Z"),
|
||||
);
|
||||
|
||||
expect(result).toMatchObject({
|
||||
interrupted: 0,
|
||||
interruptedRunIds: [],
|
||||
retryRunIds: [],
|
||||
restartSuspendedRunIds: [runId],
|
||||
});
|
||||
await expect(
|
||||
db.select().from(heartbeatRuns).where(eq(heartbeatRuns.agentId, agentId)),
|
||||
).resolves.toEqual([
|
||||
expect.objectContaining({
|
||||
id: runId,
|
||||
status: "running",
|
||||
retryOfRunId: null,
|
||||
}),
|
||||
]);
|
||||
await expect(
|
||||
db
|
||||
.select({ recoveryState: nativeRunFinalizations.recoveryState })
|
||||
.from(nativeRunFinalizations)
|
||||
.where(eq(nativeRunFinalizations.runId, runId)),
|
||||
).resolves.toEqual([{ recoveryState: "awaiting_runner_reattach" }]);
|
||||
await expect(
|
||||
db
|
||||
.select({ status: agentWakeupRequests.status })
|
||||
.from(agentWakeupRequests)
|
||||
.where(eq(agentWakeupRequests.id, wakeupRequestId)),
|
||||
).resolves.toEqual([{ status: "claimed" }]);
|
||||
await expect(
|
||||
db
|
||||
.select({ executionRunId: issues.executionRunId })
|
||||
.from(issues)
|
||||
.where(eq(issues.id, issueId)),
|
||||
).resolves.toEqual([{ executionRunId: runId }]);
|
||||
});
|
||||
|
||||
it("does not overwrite a run that is no longer running during graceful shutdown drain", async () => {
|
||||
const { runId, wakeupRequestId } = await seedRunFixture({
|
||||
agentStatus: "running",
|
||||
|
|
|
|||
|
|
@ -48,6 +48,13 @@ const {
|
|||
}));
|
||||
const heartbeatServiceMock = {
|
||||
resolveSchedulingSuppression: resolveHeartbeatSchedulingSuppressionMock,
|
||||
recoverNativeRunsAfterRestart: vi.fn(async () => ({
|
||||
restartKind: "hard",
|
||||
dispositions: [],
|
||||
claims: [],
|
||||
awaitingEvidenceRunIds: [],
|
||||
blockedRunIds: [],
|
||||
})),
|
||||
reconcileHotRestartAdoption: vi.fn(async () => ({ mode: "none" })),
|
||||
reapOrphanedRuns: vi.fn(async () => ({ reaped: 0, runIds: [] })),
|
||||
promoteDueScheduledRetries: vi.fn(async () => ({ promoted: 0, runIds: [] })),
|
||||
|
|
@ -118,7 +125,7 @@ const {
|
|||
callback?.();
|
||||
return fakeServer;
|
||||
}),
|
||||
close: vi.fn(),
|
||||
close: vi.fn((callback?: (error?: Error) => void) => callback?.()),
|
||||
};
|
||||
const loadConfigMock = vi.fn();
|
||||
|
||||
|
|
@ -574,6 +581,21 @@ describe("startServer feedback export wiring", () => {
|
|||
expect(heartbeatServiceMock.reapOrphanedRuns).toHaveBeenCalledTimes(2);
|
||||
});
|
||||
|
||||
it("closes the bound listener when native startup recovery fails", async () => {
|
||||
loadConfigMock.mockReturnValue(buildTestConfig({
|
||||
heartbeatSchedulerEnabled: true,
|
||||
heartbeatSchedulerIntervalMs: 30000,
|
||||
}));
|
||||
heartbeatServiceMock.recoverNativeRunsAfterRestart.mockRejectedValueOnce(
|
||||
new Error("native recovery unavailable"),
|
||||
);
|
||||
|
||||
await expect(startServer()).rejects.toThrow("native recovery unavailable");
|
||||
|
||||
expect(fakeServer.listen).toHaveBeenCalledTimes(1);
|
||||
expect(fakeServer.close).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it("refuses authenticated public startup without an external database URL", async () => {
|
||||
loadConfigMock.mockReturnValue(buildTestConfig({
|
||||
deploymentExposure: "public",
|
||||
|
|
|
|||
|
|
@ -1,7 +1,21 @@
|
|||
import { existsSync, mkdirSync, readFileSync, statSync, writeFileSync } from "node:fs";
|
||||
import { randomUUID } from "node:crypto";
|
||||
import {
|
||||
existsSync,
|
||||
mkdirSync,
|
||||
readFileSync,
|
||||
renameSync,
|
||||
rmSync,
|
||||
statSync,
|
||||
unlinkSync,
|
||||
writeFileSync,
|
||||
} from "node:fs";
|
||||
import path from "node:path";
|
||||
|
||||
const MAX_PERSISTED_DEV_SERVER_STATUS_BYTES = 64 * 1024;
|
||||
const DEV_RESTART_REQUEST_LOCK_STALE_MS = 30_000;
|
||||
const DEV_RESTART_REQUEST_LOCK_RETRY_COUNT = 50;
|
||||
const DEV_RESTART_REQUEST_LOCK_RETRY_MS = 2;
|
||||
const lockWaitBuffer = new Int32Array(new SharedArrayBuffer(4));
|
||||
|
||||
export type PersistedDevServerStatus = {
|
||||
dirty: boolean;
|
||||
|
|
@ -15,7 +29,11 @@ export type PersistedDevServerStatus = {
|
|||
export type DevServerHealthStatus = {
|
||||
enabled: true;
|
||||
restartRequired: boolean;
|
||||
reason: "backend_changes" | "pending_migrations" | "backend_changes_and_pending_migrations" | null;
|
||||
reason:
|
||||
| "backend_changes"
|
||||
| "pending_migrations"
|
||||
| "backend_changes_and_pending_migrations"
|
||||
| null;
|
||||
lastChangedAt: string | null;
|
||||
changedPathCount: number;
|
||||
changedPathsSample: string[];
|
||||
|
|
@ -29,14 +47,110 @@ export type DevServerHealthStatus = {
|
|||
export type DevServerRestartRequest = {
|
||||
requestedAt: string;
|
||||
reason: "manual_restart_now";
|
||||
requestId?: string;
|
||||
mode?: "hot";
|
||||
previousServerIdentity?: string;
|
||||
};
|
||||
|
||||
function processIsAlive(pid: number): boolean {
|
||||
try {
|
||||
process.kill(pid, 0);
|
||||
return true;
|
||||
} catch (error) {
|
||||
return (error as NodeJS.ErrnoException | undefined)?.code === "EPERM";
|
||||
}
|
||||
}
|
||||
|
||||
function tryRecoverDevRestartRequestLock(lockPath: string): boolean {
|
||||
let stale = false;
|
||||
try {
|
||||
const lockAgeMs = Date.now() - statSync(lockPath).mtimeMs;
|
||||
const owner = JSON.parse(
|
||||
readFileSync(path.join(lockPath, "owner.json"), "utf8"),
|
||||
) as Record<string, unknown>;
|
||||
stale =
|
||||
lockAgeMs >= DEV_RESTART_REQUEST_LOCK_STALE_MS ||
|
||||
(typeof owner.pid === "number" && !processIsAlive(owner.pid));
|
||||
} catch {
|
||||
// Canonical locks are published only after owner.json is durable in a
|
||||
// private candidate directory. A visible lock without valid ownership is
|
||||
// therefore abandoned and can be reclaimed immediately.
|
||||
stale = true;
|
||||
}
|
||||
if (!stale) return false;
|
||||
|
||||
const stalePath = `${lockPath}.${process.pid}.${randomUUID()}.stale`;
|
||||
try {
|
||||
renameSync(lockPath, stalePath);
|
||||
} catch (error) {
|
||||
if ((error as NodeJS.ErrnoException | undefined)?.code === "ENOENT") {
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
rmSync(stalePath, { recursive: true, force: true });
|
||||
return true;
|
||||
}
|
||||
|
||||
function withDevRestartRequestLock<T>(filePath: string, action: () => T): T {
|
||||
const lockPath = `${filePath}.lock`;
|
||||
for (
|
||||
let attempt = 0;
|
||||
attempt <= DEV_RESTART_REQUEST_LOCK_RETRY_COUNT;
|
||||
attempt += 1
|
||||
) {
|
||||
const candidateLockPath = `${lockPath}.${process.pid}.${randomUUID()}.candidate`;
|
||||
mkdirSync(candidateLockPath);
|
||||
try {
|
||||
writeFileSync(
|
||||
path.join(candidateLockPath, "owner.json"),
|
||||
`${JSON.stringify({ pid: process.pid, acquiredAt: new Date().toISOString() })}\n`,
|
||||
"utf8",
|
||||
);
|
||||
try {
|
||||
// Publishing a populated directory makes lock ownership visible in one
|
||||
// rename; there is no canonical owner-less crash window.
|
||||
renameSync(candidateLockPath, lockPath);
|
||||
} catch (error) {
|
||||
const code = (error as NodeJS.ErrnoException | undefined)?.code;
|
||||
if (code !== "EEXIST" && code !== "ENOTEMPTY" && code !== "EPERM") {
|
||||
throw error;
|
||||
}
|
||||
if (tryRecoverDevRestartRequestLock(lockPath)) continue;
|
||||
if (attempt === DEV_RESTART_REQUEST_LOCK_RETRY_COUNT) {
|
||||
throw new Error("dev_server_restart_request_lock_busy");
|
||||
}
|
||||
Atomics.wait(
|
||||
lockWaitBuffer,
|
||||
0,
|
||||
0,
|
||||
DEV_RESTART_REQUEST_LOCK_RETRY_MS,
|
||||
);
|
||||
continue;
|
||||
}
|
||||
} finally {
|
||||
if (existsSync(candidateLockPath)) {
|
||||
rmSync(candidateLockPath, { recursive: true, force: true });
|
||||
}
|
||||
}
|
||||
try {
|
||||
return action();
|
||||
} finally {
|
||||
rmSync(lockPath, { recursive: true, force: true });
|
||||
}
|
||||
}
|
||||
throw new Error("dev_server_restart_request_lock_busy");
|
||||
}
|
||||
|
||||
export function getDevServerRestartRequestFilePath(
|
||||
env: NodeJS.ProcessEnv = process.env,
|
||||
): string | null {
|
||||
const statusFilePath = env.PAPERCLIP_DEV_SERVER_STATUS_FILE?.trim();
|
||||
if (!statusFilePath) return null;
|
||||
return path.join(path.dirname(statusFilePath), "dev-server-restart-request.json");
|
||||
return path.join(
|
||||
path.dirname(statusFilePath),
|
||||
"dev-server-restart-request.json",
|
||||
);
|
||||
}
|
||||
|
||||
export function writeDevServerRestartRequest(
|
||||
|
|
@ -47,10 +161,88 @@ export function writeDevServerRestartRequest(
|
|||
if (!filePath) return false;
|
||||
|
||||
mkdirSync(path.dirname(filePath), { recursive: true });
|
||||
writeFileSync(filePath, `${JSON.stringify(request, null, 2)}\n`, "utf8");
|
||||
withDevRestartRequestLock(filePath, () => {
|
||||
const tempPath = `${filePath}.${process.pid}.${randomUUID()}.tmp`;
|
||||
try {
|
||||
writeFileSync(tempPath, `${JSON.stringify(request, null, 2)}\n`, "utf8");
|
||||
renameSync(tempPath, filePath);
|
||||
} finally {
|
||||
try {
|
||||
unlinkSync(tempPath);
|
||||
} catch (error) {
|
||||
if ((error as NodeJS.ErrnoException | undefined)?.code !== "ENOENT") {
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
return true;
|
||||
}
|
||||
|
||||
function readDevServerRestartRequestAtPath(
|
||||
filePath: string,
|
||||
): DevServerRestartRequest | null {
|
||||
try {
|
||||
if (statSync(filePath).size > MAX_PERSISTED_DEV_SERVER_STATUS_BYTES)
|
||||
return null;
|
||||
const value = JSON.parse(readFileSync(filePath, "utf8")) as Record<
|
||||
string,
|
||||
unknown
|
||||
>;
|
||||
if (
|
||||
typeof value.requestedAt !== "string" ||
|
||||
value.reason !== "manual_restart_now"
|
||||
) {
|
||||
return null;
|
||||
}
|
||||
return {
|
||||
requestedAt: value.requestedAt,
|
||||
reason: "manual_restart_now",
|
||||
...(typeof value.requestId === "string"
|
||||
? { requestId: value.requestId }
|
||||
: {}),
|
||||
...(value.mode === "hot" ? { mode: "hot" as const } : {}),
|
||||
...(typeof value.previousServerIdentity === "string"
|
||||
? { previousServerIdentity: value.previousServerIdentity }
|
||||
: {}),
|
||||
};
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
export function readDevServerRestartRequest(
|
||||
env: NodeJS.ProcessEnv = process.env,
|
||||
): DevServerRestartRequest | null {
|
||||
const filePath = getDevServerRestartRequestFilePath(env);
|
||||
if (!filePath || !existsSync(filePath)) return null;
|
||||
return readDevServerRestartRequestAtPath(filePath);
|
||||
}
|
||||
|
||||
export function removeDevServerRestartRequest(
|
||||
expected?: Pick<DevServerRestartRequest, "requestId">,
|
||||
env: NodeJS.ProcessEnv = process.env,
|
||||
): boolean {
|
||||
const filePath = getDevServerRestartRequestFilePath(env);
|
||||
if (!filePath) return false;
|
||||
try {
|
||||
withDevRestartRequestLock(filePath, () => {
|
||||
const current = readDevServerRestartRequestAtPath(filePath);
|
||||
if (expected?.requestId && current?.requestId !== expected.requestId) return;
|
||||
rmSync(filePath, { force: true });
|
||||
});
|
||||
return true;
|
||||
} catch (error) {
|
||||
if (
|
||||
error instanceof Error &&
|
||||
error.message === "dev_server_restart_request_lock_busy"
|
||||
) {
|
||||
return false;
|
||||
}
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
|
||||
function normalizeStringArray(value: unknown): string[] {
|
||||
if (!Array.isArray(value)) return [];
|
||||
return value
|
||||
|
|
@ -75,12 +267,18 @@ export function readPersistedDevServerStatus(
|
|||
if (statSync(filePath).size > MAX_PERSISTED_DEV_SERVER_STATUS_BYTES) {
|
||||
return null;
|
||||
}
|
||||
const raw = JSON.parse(readFileSync(filePath, "utf8")) as Record<string, unknown>;
|
||||
const changedPathsSample = normalizeStringArray(raw.changedPathsSample).slice(0, 5);
|
||||
const raw = JSON.parse(readFileSync(filePath, "utf8")) as Record<
|
||||
string,
|
||||
unknown
|
||||
>;
|
||||
const changedPathsSample = normalizeStringArray(
|
||||
raw.changedPathsSample,
|
||||
).slice(0, 5);
|
||||
const pendingMigrations = normalizeStringArray(raw.pendingMigrations);
|
||||
const changedPathCountRaw = raw.changedPathCount;
|
||||
const changedPathCount =
|
||||
typeof changedPathCountRaw === "number" && Number.isFinite(changedPathCountRaw)
|
||||
typeof changedPathCountRaw === "number" &&
|
||||
Number.isFinite(changedPathCountRaw)
|
||||
? Math.max(0, Math.trunc(changedPathCountRaw))
|
||||
: changedPathsSample.length;
|
||||
const dirtyRaw = raw.dirty;
|
||||
|
|
@ -128,7 +326,8 @@ export function toDevServerHealthStatus(
|
|||
pendingMigrations: persisted.pendingMigrations,
|
||||
autoRestartEnabled: opts.autoRestartEnabled,
|
||||
activeRunCount: opts.activeRunCount,
|
||||
waitingForIdle: restartRequired && opts.autoRestartEnabled && opts.activeRunCount > 0,
|
||||
waitingForIdle:
|
||||
restartRequired && opts.autoRestartEnabled && opts.activeRunCount > 0,
|
||||
lastRestartAt: persisted.lastRestartAt,
|
||||
};
|
||||
}
|
||||
|
|
|
|||
|
|
@ -35,6 +35,7 @@ import detectPort from "detect-port";
|
|||
import { createApp } from "./app.js";
|
||||
import { loadConfig } from "./config.js";
|
||||
import { logger } from "./middleware/logger.js";
|
||||
import { setStartupRecoveryPhase } from "./startup-recovery-state.js";
|
||||
import {
|
||||
StartupRefusalError,
|
||||
migrationRefusalError,
|
||||
|
|
@ -103,6 +104,7 @@ import { ensureDecisionSigningSecret } from "./services/decision-signing.js";
|
|||
import { createDecisionRetentionNotifyOriginAgent, createDecisionWakeOriginAgent } from "./services/decision-wakeup.js";
|
||||
import {
|
||||
coordinateHeartbeatSchedulerShutdown,
|
||||
drainRunExecutionFinalizersForShutdown,
|
||||
finalizeServerShutdown,
|
||||
loadWithoutCoordinatedShutdownSignalHooks,
|
||||
} from "./shutdown.js";
|
||||
|
|
@ -155,6 +157,7 @@ export interface StartedServer {
|
|||
}
|
||||
|
||||
export async function startServer(): Promise<StartedServer> {
|
||||
setStartupRecoveryPhase("starting");
|
||||
warnIfUnsupportedNodeVersion(process.versions.node, (message) => logger.warn(message));
|
||||
|
||||
// Tracing must be active (or have failed and logged) before the first DB
|
||||
|
|
@ -887,7 +890,9 @@ export async function startServer(): Promise<StartedServer> {
|
|||
process.env.PAPERCLIP_RUNTIME_API_URL = runtimeApiUrl;
|
||||
process.env.PAPERCLIP_RUNTIME_API_CANDIDATES_JSON = JSON.stringify(runtimeApiCandidates);
|
||||
process.env.PAPERCLIP_API_URL = configuredApiUrl;
|
||||
|
||||
|
||||
let startupListenerBound = false;
|
||||
try {
|
||||
setupRunnerPrpWebSocketServer(server, { apiUrl: configuredApiUrl });
|
||||
setupEnvironmentCustomImageTerminalWebSocketServer(server, db as any, {
|
||||
pluginWorkerManager,
|
||||
|
|
@ -911,6 +916,28 @@ export async function startServer(): Promise<StartedServer> {
|
|||
},
|
||||
});
|
||||
|
||||
setStartupRecoveryPhase("recovering");
|
||||
// Bind the shared HTTP/PRP listener before native startup recovery. A
|
||||
// runnerd process that survived a controller crash is already reconnecting
|
||||
// to this address; delaying listen until after orphan reconciliation makes
|
||||
// authenticated adoption impossible and turns a healthy process into a
|
||||
// duplicate-provider risk.
|
||||
await new Promise<void>((resolveListen, rejectListen) => {
|
||||
const onError = (err: Error) => {
|
||||
server.off("error", onError);
|
||||
rejectListen(err);
|
||||
};
|
||||
server.once("error", onError);
|
||||
server.listen(listenPort, config.host, () => {
|
||||
server.off("error", onError);
|
||||
logger.info(
|
||||
`Server listener bound on ${config.host}:${listenPort}; startup recovery in progress`,
|
||||
);
|
||||
resolveListen();
|
||||
});
|
||||
});
|
||||
startupListenerBound = true;
|
||||
|
||||
try {
|
||||
const result = await workspaceOperationService(db as any)
|
||||
.reconcileStaleRuntimeControlOperations();
|
||||
|
|
@ -1042,6 +1069,7 @@ export async function startServer(): Promise<StartedServer> {
|
|||
signal: "SIGINT" | "SIGTERM",
|
||||
runIds?: readonly string[] | null,
|
||||
) => Promise<unknown>) | null = null;
|
||||
let drainHeartbeatExecutionFinalizers: (() => Promise<void>) | null = null;
|
||||
let prepareHotRestartShutdown: ((signal: "SIGINT" | "SIGTERM") => Promise<{
|
||||
skipDrain: boolean;
|
||||
drainRunIds?: string[];
|
||||
|
|
@ -1146,6 +1174,8 @@ export async function startServer(): Promise<StartedServer> {
|
|||
drainHeartbeatRunsForShutdown = (signal, runIds) => (
|
||||
heartbeat.drainRunningRunsForShutdown(signal, new Date(), runIds)
|
||||
);
|
||||
drainHeartbeatExecutionFinalizers = () =>
|
||||
heartbeat.drainActiveRunExecutions();
|
||||
prepareHotRestartShutdown = heartbeat.prepareHotRestartShutdown;
|
||||
const environmentCustomImages = environmentCustomImageService(db as any, { pluginWorkerManager });
|
||||
const routines = routineService(db as any, { pluginWorkerManager });
|
||||
|
|
@ -1290,6 +1320,32 @@ export async function startServer(): Promise<StartedServer> {
|
|||
);
|
||||
} else {
|
||||
const startupHeartbeatRecovery = (async () => {
|
||||
try {
|
||||
const nativeRecovery =
|
||||
await heartbeat.recoverNativeRunsAfterRestart();
|
||||
if (nativeRecovery.dispositions.length > 0) {
|
||||
logger.info(
|
||||
{
|
||||
restartKind: nativeRecovery.restartKind,
|
||||
claims: nativeRecovery.claims.map((claim) => ({
|
||||
runId: claim.runId,
|
||||
disposition: claim.kind,
|
||||
controllerGeneration: claim.controllerGeneration,
|
||||
})),
|
||||
awaitingEvidenceRunIds:
|
||||
nativeRecovery.awaitingEvidenceRunIds,
|
||||
blockedRunIds: nativeRecovery.blockedRunIds,
|
||||
},
|
||||
"startup native runner restart recovery classified",
|
||||
);
|
||||
}
|
||||
} catch (err) {
|
||||
logger.error(
|
||||
{ err },
|
||||
"startup native runner restart recovery failed closed",
|
||||
);
|
||||
throw err;
|
||||
}
|
||||
try {
|
||||
const hotRestart = await heartbeat.reconcileHotRestartAdoption();
|
||||
if (hotRestart.mode === "reported") {
|
||||
|
|
@ -1374,6 +1430,7 @@ export async function startServer(): Promise<StartedServer> {
|
|||
}
|
||||
})().catch((err) => {
|
||||
logger.error({ err }, "startup heartbeat recovery failed");
|
||||
throw err;
|
||||
});
|
||||
trackHeartbeatSchedulerWork(startupHeartbeatRecovery);
|
||||
await startupHeartbeatRecovery;
|
||||
|
|
@ -1667,35 +1724,27 @@ export async function startServer(): Promise<StartedServer> {
|
|||
throw err;
|
||||
}
|
||||
|
||||
await new Promise<void>((resolveListen, rejectListen) => {
|
||||
const onError = (err: Error) => {
|
||||
server.off("error", onError);
|
||||
rejectListen(err);
|
||||
};
|
||||
|
||||
server.once("error", onError);
|
||||
server.listen(listenPort, config.host, () => {
|
||||
server.off("error", onError);
|
||||
logger.info(`Server listening on ${config.host}:${listenPort}`);
|
||||
void systemdNotify(["--ready", `--status=Listening on ${config.host}:${listenPort}`]).then((notified) => {
|
||||
if (notified) logger.info("Notified systemd that Paperclip is ready");
|
||||
setStartupRecoveryPhase("ready");
|
||||
logger.info(`Server startup recovery complete on ${config.host}:${listenPort}`);
|
||||
void systemdNotify(["--ready", `--status=Listening on ${config.host}:${listenPort}`]).then((notified) => {
|
||||
if (notified) logger.info("Notified systemd that Paperclip is ready");
|
||||
});
|
||||
if (process.env.PAPERCLIP_OPEN_ON_LISTEN === "true") {
|
||||
const openHost = config.host === "0.0.0.0" || config.host === "::" ? "127.0.0.1" : config.host;
|
||||
const url = `http://${openHost}:${listenPort}`;
|
||||
void import("open")
|
||||
.then((mod) => mod.default(url))
|
||||
.then(() => {
|
||||
logger.info(`Opened browser at ${url}`);
|
||||
})
|
||||
.catch((err) => {
|
||||
logger.warn({ err, url }, "Failed to open browser on startup");
|
||||
});
|
||||
if (process.env.PAPERCLIP_OPEN_ON_LISTEN === "true") {
|
||||
const openHost = config.host === "0.0.0.0" || config.host === "::" ? "127.0.0.1" : config.host;
|
||||
const url = `http://${openHost}:${listenPort}`;
|
||||
void import("open")
|
||||
.then((mod) => mod.default(url))
|
||||
.then(() => {
|
||||
logger.info(`Opened browser at ${url}`);
|
||||
})
|
||||
.catch((err) => {
|
||||
logger.warn({ err, url }, "Failed to open browser on startup");
|
||||
});
|
||||
}
|
||||
printStartupBanner({
|
||||
bind: config.bind,
|
||||
host: config.host,
|
||||
deploymentMode: config.deploymentMode,
|
||||
}
|
||||
printStartupBanner({
|
||||
bind: config.bind,
|
||||
host: config.host,
|
||||
deploymentMode: config.deploymentMode,
|
||||
deploymentExposure: config.deploymentExposure,
|
||||
authReady,
|
||||
requestedPort: requestedListenPort,
|
||||
|
|
@ -1709,27 +1758,23 @@ export async function startServer(): Promise<StartedServer> {
|
|||
databaseBackupIntervalMinutes: config.databaseBackupIntervalMinutes,
|
||||
databaseBackupRetentionDays: config.databaseBackupRetentionDays,
|
||||
databaseBackupDir: config.databaseBackupDir,
|
||||
});
|
||||
|
||||
const boardClaimUrl = getBoardClaimWarningUrl(config.host, listenPort);
|
||||
if (boardClaimUrl) {
|
||||
const red = "\x1b[41m\x1b[30m";
|
||||
const yellow = "\x1b[33m";
|
||||
const reset = "\x1b[0m";
|
||||
console.log(
|
||||
[
|
||||
`${red} BOARD CLAIM REQUIRED ${reset}`,
|
||||
`${yellow}This instance was previously local_trusted and still has local-board as the only admin.${reset}`,
|
||||
`${yellow}Sign in with a real user and open this one-time URL to claim ownership:${reset}`,
|
||||
`${yellow}${boardClaimUrl}${reset}`,
|
||||
`${yellow}If you are connecting over Tailscale, replace the host in this URL with your Tailscale IP/MagicDNS name.${reset}`,
|
||||
].join("\n"),
|
||||
);
|
||||
}
|
||||
|
||||
resolveListen();
|
||||
});
|
||||
});
|
||||
|
||||
const boardClaimUrl = getBoardClaimWarningUrl(config.host, listenPort);
|
||||
if (boardClaimUrl) {
|
||||
const red = "\x1b[41m\x1b[30m";
|
||||
const yellow = "\x1b[33m";
|
||||
const reset = "\x1b[0m";
|
||||
console.log(
|
||||
[
|
||||
`${red} BOARD CLAIM REQUIRED ${reset}`,
|
||||
`${yellow}This instance was previously local_trusted and still has local-board as the only admin.${reset}`,
|
||||
`${yellow}Sign in with a real user and open this one-time URL to claim ownership:${reset}`,
|
||||
`${yellow}${boardClaimUrl}${reset}`,
|
||||
`${yellow}If you are connecting over Tailscale, replace the host in this URL with your Tailscale IP/MagicDNS name.${reset}`,
|
||||
].join("\n"),
|
||||
);
|
||||
}
|
||||
|
||||
{
|
||||
const shutdown = async (signal: "SIGINT" | "SIGTERM") => {
|
||||
|
|
@ -1774,6 +1819,14 @@ export async function startServer(): Promise<StartedServer> {
|
|||
}
|
||||
}
|
||||
|
||||
if (!skipHeartbeatDrain) {
|
||||
await drainRunExecutionFinalizersForShutdown({
|
||||
signal,
|
||||
drain: drainHeartbeatExecutionFinalizers,
|
||||
log: logger,
|
||||
});
|
||||
}
|
||||
|
||||
// Whatever the drain did not finalize (timed-out runs, the hot-restart
|
||||
// skip path) still has a local-only tail when the in-flight run-log
|
||||
// mirror is enabled; upload those tails now so an orderly restart
|
||||
|
|
@ -1821,6 +1874,33 @@ export async function startServer(): Promise<StartedServer> {
|
|||
apiUrl: configuredApiUrl,
|
||||
databaseUrl: activeDatabaseConnectionString,
|
||||
};
|
||||
} catch (error) {
|
||||
if (startupListenerBound) {
|
||||
await new Promise<void>((resolveClose) => {
|
||||
try {
|
||||
server.close((closeError?: Error) => {
|
||||
if (
|
||||
closeError &&
|
||||
(closeError as NodeJS.ErrnoException).code !== "ERR_SERVER_NOT_RUNNING"
|
||||
) {
|
||||
logger.error(
|
||||
{ err: closeError },
|
||||
"failed to close HTTP listener after startup failure",
|
||||
);
|
||||
}
|
||||
resolveClose();
|
||||
});
|
||||
} catch (closeError) {
|
||||
logger.error(
|
||||
{ err: closeError },
|
||||
"failed to close HTTP listener after startup failure",
|
||||
);
|
||||
resolveClose();
|
||||
}
|
||||
});
|
||||
}
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
|
||||
function isMainModule(metaUrl: string): boolean {
|
||||
|
|
|
|||
|
|
@ -1,10 +1,15 @@
|
|||
import { timingSafeEqual } from "node:crypto";
|
||||
import { randomUUID, timingSafeEqual } from "node:crypto";
|
||||
import { Router } from "express";
|
||||
import type { Db } from "@paperclipai/db";
|
||||
import { and, count, eq, gt, inArray, isNull, sql } from "drizzle-orm";
|
||||
import { heartbeatRuns, instanceUserRoles, invites } from "@paperclipai/db";
|
||||
import type { DeploymentExposure, DeploymentMode } from "@paperclipai/shared";
|
||||
import { readPersistedDevServerStatus, toDevServerHealthStatus, writeDevServerRestartRequest } from "../dev-server-status.js";
|
||||
import {
|
||||
readPersistedDevServerStatus,
|
||||
removeDevServerRestartRequest,
|
||||
toDevServerHealthStatus,
|
||||
writeDevServerRestartRequest,
|
||||
} from "../dev-server-status.js";
|
||||
import { logger } from "../middleware/logger.js";
|
||||
import { getServerInfoSnapshot, type ServerInfoSnapshot } from "../server-info.js";
|
||||
import {
|
||||
|
|
@ -29,6 +34,12 @@ import {
|
|||
WORKSPACE_READINESS_USER_ID_HEADER,
|
||||
} from "../auth/workspace-login-handoff.js";
|
||||
import { serverVersion } from "../version.js";
|
||||
import { getStartupRecoveryState } from "../startup-recovery-state.js";
|
||||
import { nativeRestartRecoverySummary } from "../services/native-runtime/native-restart-recovery.js";
|
||||
import {
|
||||
removeHotRestartIntent,
|
||||
writeHotRestartIntent,
|
||||
} from "../services/hot-restart.js";
|
||||
|
||||
function shouldExposeFullHealthDetails(
|
||||
actorType: "none" | "board" | "agent" | null | undefined,
|
||||
|
|
@ -146,16 +157,75 @@ export function healthRoutes(
|
|||
return;
|
||||
}
|
||||
|
||||
const written = writeDevServerRestartRequest({
|
||||
requestedAt: new Date().toISOString(),
|
||||
reason: "manual_restart_now",
|
||||
});
|
||||
if (!written) {
|
||||
res.status(404).json({ error: "dev_server_supervisor_unavailable" });
|
||||
if (!db) {
|
||||
res.status(503).json({ error: "database_unavailable" });
|
||||
return;
|
||||
}
|
||||
|
||||
res.status(202).json({ status: "restart_requested" });
|
||||
const requestId = randomUUID();
|
||||
const requestedAt = new Date();
|
||||
const serverInfo = opts.serverInfo ?? getServerInfoSnapshot();
|
||||
const preflightActiveRunIds = await db
|
||||
.select({ id: heartbeatRuns.id })
|
||||
.from(heartbeatRuns)
|
||||
.where(eq(heartbeatRuns.status, "running"))
|
||||
.then((rows) => rows.map((row) => row.id));
|
||||
let intent: Awaited<ReturnType<typeof writeHotRestartIntent>> | null = null;
|
||||
try {
|
||||
intent = await writeHotRestartIntent({
|
||||
previousServerPid: process.pid,
|
||||
previousServerIdentity: serverInfo.processStartedAt,
|
||||
previousServerVersion: serverVersion,
|
||||
preflightActiveRunIds,
|
||||
recoveryRequestId: requestId,
|
||||
requestedAt,
|
||||
});
|
||||
const written = writeDevServerRestartRequest({
|
||||
requestedAt: requestedAt.toISOString(),
|
||||
reason: "manual_restart_now",
|
||||
requestId,
|
||||
mode: "hot",
|
||||
previousServerIdentity: serverInfo.processStartedAt,
|
||||
});
|
||||
if (!written) {
|
||||
throw new Error("dev_server_supervisor_unavailable");
|
||||
}
|
||||
} catch (error) {
|
||||
try {
|
||||
removeDevServerRestartRequest({ requestId });
|
||||
} catch (rollbackError) {
|
||||
logger.error(
|
||||
{ err: rollbackError, requestId },
|
||||
"failed to roll back dev-server restart request",
|
||||
);
|
||||
}
|
||||
if (intent) {
|
||||
await removeHotRestartIntent(undefined, intent).catch(
|
||||
(rollbackError) => {
|
||||
logger.error(
|
||||
{ err: rollbackError, requestId },
|
||||
"failed to roll back hot-restart intent",
|
||||
);
|
||||
},
|
||||
);
|
||||
}
|
||||
if (
|
||||
error instanceof Error &&
|
||||
error.message === "dev_server_supervisor_unavailable"
|
||||
) {
|
||||
res.status(404).json({ error: "dev_server_supervisor_unavailable" });
|
||||
return;
|
||||
}
|
||||
logger.error({ err: error, requestId }, "failed to coordinate hot restart request");
|
||||
res.status(500).json({ error: "hot_restart_intent_failed" });
|
||||
return;
|
||||
}
|
||||
|
||||
res.status(202).json({
|
||||
status: "restart_requested",
|
||||
requestId,
|
||||
mode: "hot",
|
||||
});
|
||||
});
|
||||
|
||||
router.get("/", async (req, res) => {
|
||||
|
|
@ -165,6 +235,9 @@ export function healthRoutes(
|
|||
opts.deploymentMode,
|
||||
);
|
||||
const runtimeEnv = opts.runtimeEnv ?? process.env;
|
||||
const startupRecovery = getStartupRecoveryState();
|
||||
const healthStatus =
|
||||
startupRecovery.phase === "ready" ? "ok" : "starting";
|
||||
const cloud = getCloudHealthStatus(runtimeEnv);
|
||||
// Operator-hidden settings ride every response (like `cloud`): the list
|
||||
// holds UI surface names only, and the settings nav needs it before any
|
||||
|
|
@ -200,7 +273,7 @@ export function healthRoutes(
|
|||
res.json(
|
||||
exposeFullDetails
|
||||
? {
|
||||
status: "ok",
|
||||
status: healthStatus,
|
||||
version: serverVersion,
|
||||
serverVersion: serverVersion,
|
||||
commit,
|
||||
|
|
@ -209,7 +282,7 @@ export function healthRoutes(
|
|||
...(hiddenSettings.length ? { hiddenSettings } : {}),
|
||||
}
|
||||
: {
|
||||
status: "ok",
|
||||
status: healthStatus,
|
||||
deploymentMode: opts.deploymentMode,
|
||||
commit,
|
||||
...(cloud ? { cloud } : {}),
|
||||
|
|
@ -304,12 +377,18 @@ export function healthRoutes(
|
|||
? inspectDatabaseBackupHealth(opts.databaseBackupHealth)
|
||||
: undefined;
|
||||
const warnings = databaseBackup?.warnings.length ? databaseBackup.warnings : undefined;
|
||||
const nativeRecovery = exposeFullDetails
|
||||
? await nativeRestartRecoverySummary(db).catch((error) => {
|
||||
logger.warn({ err: error }, "native recovery health summary failed");
|
||||
return {};
|
||||
})
|
||||
: undefined;
|
||||
|
||||
if (!exposeFullDetails) {
|
||||
const redactedDatabaseBackup = databaseBackup ? redactedDatabaseBackupHealth(databaseBackup) : undefined;
|
||||
const redactedWarnings = redactedDatabaseBackup?.warnings.length ? redactedDatabaseBackup.warnings : undefined;
|
||||
res.json({
|
||||
status: "ok",
|
||||
status: healthStatus,
|
||||
deploymentMode: opts.deploymentMode,
|
||||
deploymentExposure: opts.deploymentExposure,
|
||||
commit,
|
||||
|
|
@ -329,7 +408,7 @@ export function healthRoutes(
|
|||
}
|
||||
|
||||
res.json({
|
||||
status: "ok",
|
||||
status: healthStatus,
|
||||
version: serverVersion,
|
||||
serverVersion,
|
||||
commit,
|
||||
|
|
@ -342,6 +421,8 @@ export function healthRoutes(
|
|||
companyDeletionEnabled: opts.companyDeletionEnabled,
|
||||
},
|
||||
serverInfo,
|
||||
startupRecovery,
|
||||
nativeRecovery,
|
||||
...(databaseBackup ? { databaseBackup } : {}),
|
||||
...(warnings ? { warnings } : {}),
|
||||
...(devServer ? { devServer } : {}),
|
||||
|
|
|
|||
|
|
@ -127,15 +127,18 @@ import {
|
|||
buildNativeExecutionInput,
|
||||
buildNativeRuntimeContext,
|
||||
cancelNativeSession,
|
||||
claimNativeRestartRecoveries,
|
||||
dispatchNativeSessionResumptions,
|
||||
ensureNativeCompletionContract,
|
||||
executePaperclipNativeSession,
|
||||
finalizeNativeRun,
|
||||
isNativeSessionId,
|
||||
isUnusedLegacyNativeRetryReplacement,
|
||||
isRunnerIngressAuthorized,
|
||||
materializeLegacyQuestionResponseWakeProjection,
|
||||
materializeNativeInteractionResponses,
|
||||
NativeCancellationPendingRecoveryError,
|
||||
type NativeRestartRecoveryClaim,
|
||||
rebindNativeSessionCheckpoint,
|
||||
reconcileNativeFinalizations,
|
||||
resolveHeartbeatNativeRuntimeMode,
|
||||
|
|
@ -421,6 +424,7 @@ import {
|
|||
import {
|
||||
findMissingHotRestartSnapshotRunIds,
|
||||
readHotRestartIntent,
|
||||
readProcessStartedAt,
|
||||
removeHotRestartIntent,
|
||||
shouldHonorHotRestartIntentForProcess,
|
||||
writeHotRestartReport,
|
||||
|
|
@ -7669,7 +7673,10 @@ export async function persistHeartbeatRunProcessMetadata(
|
|||
runId: string,
|
||||
meta: { pid: number; processGroupId: number | null; startedAt: string },
|
||||
) {
|
||||
const startedAt = new Date(meta.startedAt);
|
||||
const observedStartedAt = await readProcessStartedAt(meta.pid).catch(
|
||||
() => null,
|
||||
);
|
||||
const startedAt = new Date(observedStartedAt ?? meta.startedAt);
|
||||
return db
|
||||
.update(heartbeatRuns)
|
||||
.set({
|
||||
|
|
@ -8173,6 +8180,7 @@ export function heartbeatService(
|
|||
db: Db,
|
||||
options: HeartbeatServiceOptions = {},
|
||||
) {
|
||||
let shutdownInProgress = false;
|
||||
const instanceSettings = instanceSettingsService(db);
|
||||
const getCurrentUserRedactionOptions = async () => ({
|
||||
enabled: (await instanceSettings.getGeneral()).censorUsernameInLogs,
|
||||
|
|
@ -12383,6 +12391,10 @@ export function heartbeatService(
|
|||
processPid: input.run.processPid ?? null,
|
||||
processGroupId: input.run.processGroupId ?? null,
|
||||
issueId: readNonEmptyString(context.issueId),
|
||||
runtimeMode: input.run.runtimeMode,
|
||||
nativeSessionId: input.run.nativeSessionId,
|
||||
runnerInstanceId: input.run.runnerInstanceId,
|
||||
processStartedAt: input.run.processStartedAt?.toISOString() ?? null,
|
||||
};
|
||||
}
|
||||
|
||||
|
|
@ -12420,6 +12432,7 @@ export function heartbeatService(
|
|||
signal: "SIGINT" | "SIGTERM",
|
||||
now = new Date(),
|
||||
) {
|
||||
shutdownInProgress = true;
|
||||
let intent: Awaited<ReturnType<typeof readHotRestartIntent>>;
|
||||
try {
|
||||
intent = await readHotRestartIntent();
|
||||
|
|
@ -12706,6 +12719,14 @@ export function heartbeatService(
|
|||
continue;
|
||||
}
|
||||
|
||||
if (
|
||||
run.runtimeMode === "native" &&
|
||||
adapterType === "paperclip_runner"
|
||||
) {
|
||||
classify(candidate, "skipped", "native_restart_recovery_owned", patch);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (!isTrackedLocalChildProcessAdapter(adapterType)) {
|
||||
classify(
|
||||
candidate,
|
||||
|
|
@ -12844,6 +12865,145 @@ export function heartbeatService(
|
|||
};
|
||||
}
|
||||
|
||||
async function recoverNativeRunsAfterRestart(now = new Date()) {
|
||||
// A result committed before the old controller stopped outranks process
|
||||
// recovery. Finish its durable workspace/status suffix before deciding
|
||||
// whether any provider authority needs to be reopened.
|
||||
await reconcileNativeFinalizations(db);
|
||||
const intent = await readHotRestartIntent().catch((error) => {
|
||||
logger.warn(
|
||||
{ err: error },
|
||||
"failed to read hot-restart intent before native startup recovery",
|
||||
);
|
||||
return null;
|
||||
});
|
||||
const restartKind = intent ? ("hot" as const) : ("hard" as const);
|
||||
const previousStartedAt = intent?.previousServerStartedAt
|
||||
? new Date(intent.previousServerStartedAt)
|
||||
: null;
|
||||
const scheduledNativeRetries = await db
|
||||
.select({
|
||||
runId: nativeRunFinalizations.runId,
|
||||
nextAttemptAt: nativeRunFinalizations.nextAttemptAt,
|
||||
})
|
||||
.from(nativeRunFinalizations)
|
||||
.innerJoin(
|
||||
heartbeatRuns,
|
||||
eq(heartbeatRuns.id, nativeRunFinalizations.runId),
|
||||
)
|
||||
.where(
|
||||
and(
|
||||
eq(heartbeatRuns.runtimeMode, "native"),
|
||||
inArray(heartbeatRuns.status, ["running", "failed"]),
|
||||
isNull(nativeRunFinalizations.resultId),
|
||||
eq(nativeRunFinalizations.phase, "retryable_failure"),
|
||||
gt(nativeRunFinalizations.nextAttemptAt, now),
|
||||
),
|
||||
);
|
||||
for (const scheduled of scheduledNativeRetries) {
|
||||
if (scheduled.nextAttemptAt) {
|
||||
scheduleNativeSessionResumeDispatch(
|
||||
scheduled.runId,
|
||||
scheduled.nextAttemptAt,
|
||||
);
|
||||
}
|
||||
}
|
||||
const dispositions = await claimNativeRestartRecoveries({
|
||||
db,
|
||||
restartKind,
|
||||
recoveryRequestId: intent?.recoveryRequestId ?? null,
|
||||
coordinatedPreviousController: intent
|
||||
? {
|
||||
pid: intent.previousServerPid,
|
||||
processStartedAt:
|
||||
previousStartedAt && !Number.isNaN(previousStartedAt.getTime())
|
||||
? previousStartedAt
|
||||
: null,
|
||||
}
|
||||
: null,
|
||||
now,
|
||||
});
|
||||
|
||||
const claims = dispositions.filter(
|
||||
(disposition): disposition is NativeRestartRecoveryClaim =>
|
||||
disposition.kind === "reattach_existing_runner" ||
|
||||
disposition.kind === "resume_dead_runner" ||
|
||||
disposition.kind === "bootstrap_incomplete",
|
||||
);
|
||||
for (const disposition of dispositions) {
|
||||
const run = await getRun(disposition.runId);
|
||||
if (run) {
|
||||
const isClaim =
|
||||
disposition.kind === "reattach_existing_runner" ||
|
||||
disposition.kind === "resume_dead_runner" ||
|
||||
disposition.kind === "bootstrap_incomplete";
|
||||
await appendRunEvent(run, {
|
||||
eventType: "native.recovery.transition",
|
||||
stream: "system",
|
||||
level: disposition.kind === "blocked" ? "warn" : "info",
|
||||
message:
|
||||
disposition.kind === "reattach_existing_runner"
|
||||
? "Recovering the existing native runner process after server restart"
|
||||
: disposition.kind === "resume_dead_runner"
|
||||
? "Resuming the durable native provider session after runner process loss"
|
||||
: disposition.kind === "bootstrap_incomplete"
|
||||
? "Restarting an incomplete native runner bootstrap on the same heartbeat run"
|
||||
: disposition.kind === "awaiting_evidence"
|
||||
? "Native restart recovery is waiting for safe ownership evidence"
|
||||
: disposition.kind === "already_finalized"
|
||||
? "Native restart recovery found an already-finalized result"
|
||||
: "Native restart recovery blocked ambiguous or conflicting ownership",
|
||||
payload: {
|
||||
restartKind: isClaim ? disposition.restartKind : restartKind,
|
||||
recoveryRequestId: isClaim
|
||||
? disposition.recoveryRequestId
|
||||
: (intent?.recoveryRequestId ?? null),
|
||||
runnerDisposition: disposition.kind,
|
||||
...(isClaim
|
||||
? {
|
||||
controllerGeneration: disposition.controllerGeneration,
|
||||
providerAttempt: disposition.providerAttempt,
|
||||
}
|
||||
: { reason: disposition.reason }),
|
||||
...(disposition.kind === "reattach_existing_runner"
|
||||
? {
|
||||
processPid: disposition.process.pid,
|
||||
processGroupId: disposition.process.processGroupId,
|
||||
processStartedAt: disposition.process.startedAt,
|
||||
}
|
||||
: {}),
|
||||
},
|
||||
});
|
||||
}
|
||||
}
|
||||
for (const claim of claims) {
|
||||
const execution = executeRun(claim.runId, {
|
||||
nativeLeaseOwner: claim.leaseOwner,
|
||||
nativeRestartRecovery: claim,
|
||||
}).catch((error) => {
|
||||
logger.error(
|
||||
{ err: error, runId: claim.runId, disposition: claim.kind },
|
||||
"native restart recovery execution failed",
|
||||
);
|
||||
});
|
||||
activeRunExecutionPromises.add(execution);
|
||||
void execution.finally(() => activeRunExecutionPromises.delete(execution));
|
||||
}
|
||||
|
||||
return {
|
||||
restartKind,
|
||||
claims,
|
||||
dispositions,
|
||||
scheduledRetryRunIds: scheduledNativeRetries.map((entry) => entry.runId),
|
||||
awaitingEvidenceRunIds: dispositions
|
||||
.filter((entry) => entry.kind === "awaiting_evidence")
|
||||
.map((entry) => entry.runId),
|
||||
blockedRunIds: dispositions
|
||||
.filter((entry) => entry.kind === "blocked")
|
||||
.map((entry) => entry.runId),
|
||||
};
|
||||
}
|
||||
|
||||
async function drainRunningRunsForShutdown(
|
||||
signal: "SIGINT" | "SIGTERM",
|
||||
now = new Date(),
|
||||
|
|
@ -12851,7 +13011,12 @@ export function heartbeatService(
|
|||
) {
|
||||
const selectedRunIds = runIds ? [...new Set(runIds)] : null;
|
||||
if (selectedRunIds?.length === 0) {
|
||||
return { interrupted: 0, interruptedRunIds: [], retryRunIds: [] };
|
||||
return {
|
||||
interrupted: 0,
|
||||
interruptedRunIds: [],
|
||||
retryRunIds: [],
|
||||
restartSuspendedRunIds: [],
|
||||
};
|
||||
}
|
||||
const activeRuns = await db
|
||||
.select({
|
||||
|
|
@ -12871,8 +13036,66 @@ export function heartbeatService(
|
|||
|
||||
const interruptedRunIds: string[] = [];
|
||||
const retryRunIds: string[] = [];
|
||||
const restartSuspendedRunIds: string[] = [];
|
||||
|
||||
for (const { run, agent } of activeRuns) {
|
||||
if (
|
||||
run.runtimeMode === "native" &&
|
||||
agent.adapterType === "paperclip_runner"
|
||||
) {
|
||||
const recoveryHistoryEntry = JSON.stringify({
|
||||
at: now.toISOString(),
|
||||
restartKind: "graceful",
|
||||
disposition: "restart_suspended",
|
||||
reason: signal,
|
||||
processPid: run.processPid,
|
||||
processStartedAt: run.processStartedAt?.toISOString() ?? null,
|
||||
});
|
||||
await db
|
||||
.update(nativeRunFinalizations)
|
||||
.set({
|
||||
recoveryState: "awaiting_runner_reattach",
|
||||
recoveryHistory: sql`(
|
||||
select coalesce(jsonb_agg(item order by ordinal), '[]'::jsonb)
|
||||
from jsonb_array_elements(
|
||||
coalesce(${nativeRunFinalizations.recoveryHistory}, '[]'::jsonb)
|
||||
|| jsonb_build_array(${recoveryHistoryEntry}::jsonb)
|
||||
) with ordinality as history(item, ordinal)
|
||||
where ordinal > greatest(
|
||||
jsonb_array_length(
|
||||
coalesce(${nativeRunFinalizations.recoveryHistory}, '[]'::jsonb)
|
||||
|| jsonb_build_array(${recoveryHistoryEntry}::jsonb)
|
||||
) - 20,
|
||||
0
|
||||
)
|
||||
)`,
|
||||
updatedAt: now,
|
||||
})
|
||||
.where(
|
||||
and(
|
||||
eq(nativeRunFinalizations.runId, run.id),
|
||||
isNull(nativeRunFinalizations.resultId),
|
||||
),
|
||||
);
|
||||
await appendRunEvent(run, {
|
||||
eventType: "native.recovery.transition",
|
||||
stream: "system",
|
||||
level: "info",
|
||||
message:
|
||||
"Server shutdown suspended native controller ownership without cancelling provider work",
|
||||
payload: {
|
||||
restartKind: "graceful",
|
||||
signal,
|
||||
runnerDisposition: "awaiting_runner_reattach",
|
||||
processPid: run.processPid,
|
||||
processGroupId: run.processGroupId,
|
||||
processStartedAt: run.processStartedAt?.toISOString() ?? null,
|
||||
retryRunCreated: false,
|
||||
},
|
||||
});
|
||||
restartSuspendedRunIds.push(run.id);
|
||||
continue;
|
||||
}
|
||||
const message = `Interrupted by graceful server shutdown (${signal}); retry queued for restart recovery`;
|
||||
const running = runningProcesses.get(run.id);
|
||||
try {
|
||||
|
|
@ -12979,6 +13202,7 @@ export function heartbeatService(
|
|||
interrupted: interruptedRunIds.length,
|
||||
interruptedRunIds,
|
||||
retryRunIds,
|
||||
restartSuspendedRunIds,
|
||||
};
|
||||
}
|
||||
|
||||
|
|
@ -16740,6 +16964,7 @@ export function heartbeatService(
|
|||
adapterType: agents.adapterType,
|
||||
adapterConfig: agents.adapterConfig,
|
||||
nativeCoordinatorPhase: nativeRunFinalizations.phase,
|
||||
nativeRecoveryState: nativeRunFinalizations.recoveryState,
|
||||
})
|
||||
.from(heartbeatRuns)
|
||||
.innerJoin(agents, eq(heartbeatRuns.agentId, agents.id))
|
||||
|
|
@ -16785,6 +17010,7 @@ export function heartbeatService(
|
|||
adapterType,
|
||||
adapterConfig,
|
||||
nativeCoordinatorPhase,
|
||||
nativeRecoveryState,
|
||||
} of activeRuns) {
|
||||
const nativeRun = run.runtimeMode === "native";
|
||||
const nativeProcessPidAlive =
|
||||
|
|
@ -16795,6 +17021,20 @@ export function heartbeatService(
|
|||
isProcessGroupAlive(run.processGroupId);
|
||||
const locallyTracked =
|
||||
runningProcesses.has(run.id) || activeRunExecutions.has(run.id);
|
||||
if (
|
||||
nativeRun &&
|
||||
(
|
||||
[
|
||||
"awaiting_evidence",
|
||||
"awaiting_runner_reattach",
|
||||
"resuming_session",
|
||||
"bootstrap_incomplete",
|
||||
].includes(nativeRecoveryState ?? "") ||
|
||||
nativeCoordinatorPhase === "retryable_failure"
|
||||
)
|
||||
) {
|
||||
continue;
|
||||
}
|
||||
const observedOwnerUnverified =
|
||||
nativeRun &&
|
||||
(nativeCoordinatorPhase === "observed" ||
|
||||
|
|
@ -17426,7 +17666,10 @@ export function heartbeatService(
|
|||
|
||||
async function executeRun(
|
||||
runId: string,
|
||||
runOptions: { nativeLeaseOwner?: string } = {},
|
||||
runOptions: {
|
||||
nativeLeaseOwner?: string;
|
||||
nativeRestartRecovery?: NativeRestartRecoveryClaim;
|
||||
} = {},
|
||||
) {
|
||||
if ((await getSchedulingSuppression()).suppressed) {
|
||||
try {
|
||||
|
|
@ -17453,7 +17696,11 @@ export function heartbeatService(
|
|||
run = claimed;
|
||||
}
|
||||
|
||||
if (runOptions.nativeLeaseOwner && run.runtimeMode === "native") {
|
||||
if (
|
||||
runOptions.nativeLeaseOwner &&
|
||||
run.runtimeMode === "native" &&
|
||||
runOptions.nativeRestartRecovery?.kind !== "reattach_existing_runner"
|
||||
) {
|
||||
// A numeric PID or process-group ID is a liveness signal, never an
|
||||
// ownership capability: the OS may have recycled it after the service
|
||||
// restart. A still-active in-memory child handle is also insufficient to
|
||||
|
|
@ -19929,14 +20176,79 @@ export function heartbeatService(
|
|||
const taskNativeSessionId = readNonEmptyString(
|
||||
taskSessionDecodedParams?.sessionId,
|
||||
);
|
||||
const resumableTaskSessionId =
|
||||
// Compatibility for native retry rows created before same-run restart
|
||||
// recovery existed. Only an entirely unused replacement row may
|
||||
// inherit its source checkpoint; any process/provider evidence on the
|
||||
// replacement makes the ownership ambiguous and therefore ineligible.
|
||||
const legacyRetrySource =
|
||||
run.retryOfRunId
|
||||
? await db
|
||||
.select({
|
||||
id: heartbeatRuns.id,
|
||||
companyId: heartbeatRuns.companyId,
|
||||
agentId: heartbeatRuns.agentId,
|
||||
runnerInstanceId: heartbeatRuns.runnerInstanceId,
|
||||
nativeSessionId: heartbeatRuns.nativeSessionId,
|
||||
runnerProfileJson: heartbeatRuns.runnerProfileJson,
|
||||
runtimeMode: heartbeatRuns.runtimeMode,
|
||||
status: heartbeatRuns.status,
|
||||
})
|
||||
.from(heartbeatRuns)
|
||||
.where(
|
||||
and(
|
||||
eq(heartbeatRuns.id, run.retryOfRunId),
|
||||
eq(heartbeatRuns.companyId, agent.companyId),
|
||||
eq(heartbeatRuns.agentId, agent.id),
|
||||
),
|
||||
)
|
||||
.limit(1)
|
||||
.then((rows) => rows[0] ?? null)
|
||||
: null;
|
||||
const legacyRetryHasProviderEvidence = legacyRetrySource
|
||||
? await db
|
||||
.select({ id: heartbeatRunEvents.id })
|
||||
.from(heartbeatRunEvents)
|
||||
.where(
|
||||
and(
|
||||
eq(heartbeatRunEvents.runId, run.id),
|
||||
inArray(heartbeatRunEvents.eventType, [
|
||||
"harness.ready",
|
||||
"session.started",
|
||||
"session.resumed",
|
||||
"session.updated",
|
||||
"turn.started",
|
||||
"provider.event",
|
||||
"provider.rpc_result",
|
||||
]),
|
||||
),
|
||||
)
|
||||
.limit(1)
|
||||
.then((rows) => rows.length > 0)
|
||||
: false;
|
||||
const compatibleLegacyRetrySource =
|
||||
isUnusedLegacyNativeRetryReplacement({
|
||||
replacement: run,
|
||||
source: legacyRetrySource,
|
||||
hasProviderEvents: legacyRetryHasProviderEvidence,
|
||||
})
|
||||
? legacyRetrySource
|
||||
: null;
|
||||
const legacyRetrySessionId = compatibleLegacyRetrySource
|
||||
?.nativeSessionId;
|
||||
const taskResumeRunId =
|
||||
taskSessionForRun?.lastRunId &&
|
||||
taskSessionForRun.lastRunId !== run.id &&
|
||||
isNativeSessionId(taskNativeSessionId)
|
||||
? taskNativeSessionId
|
||||
? taskSessionForRun.lastRunId
|
||||
: null;
|
||||
const resumableTaskSessionId =
|
||||
taskResumeRunId
|
||||
? taskNativeSessionId
|
||||
: legacyRetrySessionId ?? null;
|
||||
const priorNativeRunId =
|
||||
taskResumeRunId ?? compatibleLegacyRetrySource?.id ?? null;
|
||||
const previousNativeRun =
|
||||
resumableTaskSessionId && taskSessionForRun?.lastRunId
|
||||
resumableTaskSessionId && priorNativeRunId
|
||||
? await db
|
||||
.select({
|
||||
id: heartbeatRuns.id,
|
||||
|
|
@ -19949,7 +20261,7 @@ export function heartbeatService(
|
|||
.from(heartbeatRuns)
|
||||
.where(
|
||||
and(
|
||||
eq(heartbeatRuns.id, taskSessionForRun.lastRunId),
|
||||
eq(heartbeatRuns.id, priorNativeRunId),
|
||||
eq(heartbeatRuns.companyId, agent.companyId),
|
||||
eq(heartbeatRuns.agentId, agent.id),
|
||||
eq(heartbeatRuns.nativeSessionId, resumableTaskSessionId),
|
||||
|
|
@ -20697,6 +21009,7 @@ export function heartbeatService(
|
|||
execution: nativeExecution,
|
||||
runnerInstanceId: nativeRunnerInstanceId,
|
||||
leaseOwner: runOptions.nativeLeaseOwner,
|
||||
restartRecovery: runOptions.nativeRestartRecovery,
|
||||
backend:
|
||||
options.nativeSessionBackendFactory?.(nativeExecution),
|
||||
useRunnerd: agent.adapterType === "paperclip_runner",
|
||||
|
|
@ -22119,7 +22432,7 @@ export function heartbeatService(
|
|||
}
|
||||
}
|
||||
activeRunExecutions.delete(run.id);
|
||||
if (!nativeSessionResumeScheduled) {
|
||||
if (!nativeSessionResumeScheduled && !shutdownInProgress) {
|
||||
await startNextQueuedRunForAgent(run.agentId);
|
||||
}
|
||||
}
|
||||
|
|
@ -25592,6 +25905,7 @@ export function heartbeatService(
|
|||
|
||||
prepareHotRestartShutdown,
|
||||
reconcileHotRestartAdoption,
|
||||
recoverNativeRunsAfterRestart,
|
||||
reapOrphanedRuns,
|
||||
sweepPendingCleanupLeases,
|
||||
// Override-aware scheduling-suppression check (honors the worktree
|
||||
|
|
|
|||
|
|
@ -5,6 +5,7 @@ import { afterEach, describe, expect, it, vi } from "vitest";
|
|||
import {
|
||||
findMissingHotRestartSnapshotRunIds,
|
||||
isObservedHotRestartTargetAlive,
|
||||
parseHotRestartIntent,
|
||||
readHotRestartIntent,
|
||||
readProcessStartedAt,
|
||||
removeHotRestartIntent,
|
||||
|
|
@ -30,6 +31,64 @@ async function withTempHome<T>(fn: (homeDir: string) => Promise<T>) {
|
|||
}
|
||||
|
||||
describe("hot-restart path compatibility", () => {
|
||||
it("reads version-1 native recovery fields without making them mandatory", () => {
|
||||
expect(
|
||||
parseHotRestartIntent({
|
||||
version: 1,
|
||||
requestedAt: "2026-09-04T10:00:00.000Z",
|
||||
recoveryRequestId: "restart-request-1",
|
||||
previousServerPid: 123,
|
||||
drainRequired: false,
|
||||
requestedByRunId: null,
|
||||
preflightActiveRunIds: ["run-1"],
|
||||
shutdownSnapshot: {
|
||||
capturedAt: "2026-09-04T10:00:01.000Z",
|
||||
signal: "SIGTERM",
|
||||
activeRuns: [
|
||||
{
|
||||
runId: "run-1",
|
||||
companyId: "company-1",
|
||||
agentId: "agent-1",
|
||||
adapterType: "paperclip_runner",
|
||||
status: "running",
|
||||
processPid: 456,
|
||||
processGroupId: 456,
|
||||
issueId: "issue-1",
|
||||
runtimeMode: "native",
|
||||
nativeSessionId: "session-1",
|
||||
runnerInstanceId: "runner-1",
|
||||
processStartedAt: "2026-09-04T09:59:00.000Z",
|
||||
},
|
||||
],
|
||||
},
|
||||
}),
|
||||
).toMatchObject({
|
||||
version: 1,
|
||||
recoveryRequestId: "restart-request-1",
|
||||
shutdownSnapshot: {
|
||||
activeRuns: [
|
||||
{
|
||||
runtimeMode: "native",
|
||||
nativeSessionId: "session-1",
|
||||
runnerInstanceId: "runner-1",
|
||||
processStartedAt: "2026-09-04T09:59:00.000Z",
|
||||
},
|
||||
],
|
||||
},
|
||||
});
|
||||
|
||||
expect(
|
||||
parseHotRestartIntent({
|
||||
version: 1,
|
||||
requestedAt: "2026-09-04T10:00:00.000Z",
|
||||
previousServerPid: 123,
|
||||
drainRequired: false,
|
||||
requestedByRunId: null,
|
||||
preflightActiveRunIds: [],
|
||||
}),
|
||||
).not.toBeNull();
|
||||
});
|
||||
|
||||
it("reads Linux process start time from proc metadata", async () => {
|
||||
await expect(
|
||||
readProcessStartedAt(123, {
|
||||
|
|
|
|||
|
|
@ -25,11 +25,16 @@ export type HotRestartIntentRun = {
|
|||
processPid: number | null;
|
||||
processGroupId: number | null;
|
||||
issueId: string | null;
|
||||
runtimeMode?: string | null;
|
||||
nativeSessionId?: string | null;
|
||||
runnerInstanceId?: string | null;
|
||||
processStartedAt?: string | null;
|
||||
};
|
||||
|
||||
export type HotRestartIntent = {
|
||||
version: 1;
|
||||
requestedAt: string;
|
||||
recoveryRequestId?: string | null;
|
||||
previousServerPid: number;
|
||||
previousServerIdentity?: string | null;
|
||||
previousServerStartedAt?: string | null;
|
||||
|
|
@ -382,7 +387,8 @@ function isSameHotRestartRequest(left: HotRestartIntent, right: HotRestartIntent
|
|||
return left.requestedAt === right.requestedAt
|
||||
&& left.previousServerPid === right.previousServerPid
|
||||
&& left.drainRequired === right.drainRequired
|
||||
&& left.requestedByRunId === right.requestedByRunId;
|
||||
&& left.requestedByRunId === right.requestedByRunId
|
||||
&& (left.recoveryRequestId ?? null) === (right.recoveryRequestId ?? null);
|
||||
}
|
||||
|
||||
function parseRun(value: unknown): HotRestartIntentRun | null {
|
||||
|
|
@ -402,6 +408,10 @@ function parseRun(value: unknown): HotRestartIntentRun | null {
|
|||
processPid: asNumber(value.processPid),
|
||||
processGroupId: asNumber(value.processGroupId),
|
||||
issueId: asString(value.issueId),
|
||||
runtimeMode: asString(value.runtimeMode),
|
||||
nativeSessionId: asString(value.nativeSessionId),
|
||||
runnerInstanceId: asString(value.runnerInstanceId),
|
||||
processStartedAt: asDateString(value.processStartedAt),
|
||||
};
|
||||
}
|
||||
|
||||
|
|
@ -414,6 +424,7 @@ export function parseHotRestartIntent(value: unknown): HotRestartIntent | null {
|
|||
const intent: HotRestartIntent = {
|
||||
version: 1,
|
||||
requestedAt,
|
||||
recoveryRequestId: asString(value.recoveryRequestId),
|
||||
previousServerPid,
|
||||
previousServerIdentity: asString(value.previousServerIdentity),
|
||||
previousServerStartedAt: asDateString(value.previousServerStartedAt),
|
||||
|
|
@ -492,6 +503,7 @@ export async function writeHotRestartIntent(input: {
|
|||
requestedByRunId?: string | null;
|
||||
preflightActiveRunIds?: string[];
|
||||
requestedAt?: Date;
|
||||
recoveryRequestId?: string | null;
|
||||
homeDir?: string;
|
||||
}) {
|
||||
const previousServerStartedAt = input.previousServerStartedAt === undefined
|
||||
|
|
@ -507,6 +519,7 @@ export async function writeHotRestartIntent(input: {
|
|||
const intent: HotRestartIntent = {
|
||||
version: 1,
|
||||
requestedAt: (input.requestedAt ?? new Date()).toISOString(),
|
||||
recoveryRequestId: input.recoveryRequestId ?? null,
|
||||
previousServerPid: input.previousServerPid,
|
||||
previousServerIdentity,
|
||||
previousServerStartedAt,
|
||||
|
|
|
|||
|
|
@ -8,4 +8,5 @@ export * from "./native-interaction-bridge.js";
|
|||
export * from "./paperclip-control-plane-port.js";
|
||||
export * from "./native-run-finalizer.js";
|
||||
export * from "./native-finalization-reconciler.js";
|
||||
export * from "./native-restart-recovery.js";
|
||||
export * from "./status-arbiter.js";
|
||||
|
|
|
|||
|
|
@ -0,0 +1,259 @@
|
|||
import { describe, expect, it, vi } from "vitest";
|
||||
|
||||
import {
|
||||
classifyNativeRunnerRecoveryEvidence,
|
||||
evaluateNativeControllerTakeover,
|
||||
evaluateNativeProviderProcesses,
|
||||
nextNativeProviderAttempt,
|
||||
} from "./native-restart-recovery.js";
|
||||
|
||||
describe("native restart recovery classification", () => {
|
||||
it("keeps controller-only recovery out of the provider retry budget", () => {
|
||||
expect(nextNativeProviderAttempt(2, "reattach_existing_runner")).toBe(2);
|
||||
expect(nextNativeProviderAttempt(2, "bootstrap_incomplete")).toBe(2);
|
||||
expect(nextNativeProviderAttempt(2, "resume_dead_runner")).toBe(3);
|
||||
});
|
||||
|
||||
it.each([
|
||||
{
|
||||
name: "reattaches an exact live runner",
|
||||
evidence: {
|
||||
runnerPidAlive: true,
|
||||
runnerGroupAlive: true,
|
||||
processStartMatches: true,
|
||||
hasCheckpoint: true,
|
||||
hasProviderEvidence: true,
|
||||
},
|
||||
expected: "reattach_existing_runner",
|
||||
},
|
||||
{
|
||||
name: "resumes a dead checkpointed runner",
|
||||
evidence: {
|
||||
runnerPidAlive: false,
|
||||
runnerGroupAlive: false,
|
||||
processStartMatches: false,
|
||||
hasCheckpoint: true,
|
||||
hasProviderEvidence: true,
|
||||
},
|
||||
expected: "resume_dead_runner",
|
||||
},
|
||||
{
|
||||
name: "retries an incomplete bootstrap on the same run",
|
||||
evidence: {
|
||||
runnerPidAlive: false,
|
||||
runnerGroupAlive: false,
|
||||
processStartMatches: false,
|
||||
hasCheckpoint: false,
|
||||
hasProviderEvidence: false,
|
||||
},
|
||||
expected: "bootstrap_incomplete",
|
||||
},
|
||||
{
|
||||
name: "blocks a recycled live PID",
|
||||
evidence: {
|
||||
runnerPidAlive: true,
|
||||
runnerGroupAlive: true,
|
||||
processStartMatches: false,
|
||||
hasCheckpoint: true,
|
||||
hasProviderEvidence: true,
|
||||
},
|
||||
expected: null,
|
||||
},
|
||||
{
|
||||
name: "blocks ambiguous checkpoint evidence",
|
||||
evidence: {
|
||||
runnerPidAlive: false,
|
||||
runnerGroupAlive: false,
|
||||
processStartMatches: false,
|
||||
hasCheckpoint: true,
|
||||
hasProviderEvidence: false,
|
||||
},
|
||||
expected: null,
|
||||
},
|
||||
{
|
||||
name: "blocks a checkpoint bound to a different native identity",
|
||||
evidence: {
|
||||
runnerPidAlive: false,
|
||||
runnerGroupAlive: false,
|
||||
processStartMatches: false,
|
||||
hasCheckpoint: true,
|
||||
checkpointIdentityMatches: false,
|
||||
hasProviderEvidence: true,
|
||||
},
|
||||
expected: null,
|
||||
},
|
||||
{
|
||||
name: "blocks while a known provider process outlives its runner",
|
||||
evidence: {
|
||||
runnerPidAlive: false,
|
||||
runnerGroupAlive: false,
|
||||
processStartMatches: false,
|
||||
knownProviderProcessAlive: true,
|
||||
hasCheckpoint: true,
|
||||
hasProviderEvidence: true,
|
||||
},
|
||||
expected: null,
|
||||
},
|
||||
{
|
||||
name: "blocks while a live provider PID has no durable fingerprint",
|
||||
evidence: {
|
||||
runnerPidAlive: false,
|
||||
runnerGroupAlive: false,
|
||||
processStartMatches: false,
|
||||
knownProviderProcessIdentityAmbiguous: true,
|
||||
hasCheckpoint: true,
|
||||
hasProviderEvidence: true,
|
||||
},
|
||||
expected: null,
|
||||
},
|
||||
])("$name", ({ evidence, expected }) => {
|
||||
expect(classifyNativeRunnerRecoveryEvidence(evidence).claimKind).toBe(
|
||||
expected,
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
describe("native controller takeover fencing", () => {
|
||||
const now = new Date("2026-09-04T12:00:00.000Z");
|
||||
const recordedStart = new Date("2026-09-04T11:00:00.000Z");
|
||||
|
||||
function owner(
|
||||
overrides: Partial<{
|
||||
leaseOwner: string | null;
|
||||
leaseExpiresAt: Date | null;
|
||||
controllerPid: number | null;
|
||||
controllerProcessStartedAt: Date | null;
|
||||
}> = {},
|
||||
) {
|
||||
return {
|
||||
leaseOwner: "old-controller",
|
||||
leaseExpiresAt: new Date("2026-09-04T12:20:00.000Z"),
|
||||
controllerPid: 123,
|
||||
controllerProcessStartedAt: recordedStart,
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
it("takes over immediately when the exact prior controller process died", async () => {
|
||||
await expect(
|
||||
evaluateNativeControllerTakeover({
|
||||
owner: owner(),
|
||||
now,
|
||||
isProcessAlive: () => false,
|
||||
}),
|
||||
).resolves.toEqual({
|
||||
allowed: true,
|
||||
reason: "controller_process_dead",
|
||||
});
|
||||
});
|
||||
|
||||
it("takes over a recycled controller PID", async () => {
|
||||
await expect(
|
||||
evaluateNativeControllerTakeover({
|
||||
owner: owner(),
|
||||
now,
|
||||
isProcessAlive: () => true,
|
||||
readProcessStartedAt: async () => new Date("2026-09-04T11:30:00.000Z"),
|
||||
}),
|
||||
).resolves.toEqual({
|
||||
allowed: true,
|
||||
reason: "controller_pid_recycled",
|
||||
});
|
||||
});
|
||||
|
||||
it("does not steal a live controller lease", async () => {
|
||||
const readStartedAt = vi.fn(async () => recordedStart);
|
||||
await expect(
|
||||
evaluateNativeControllerTakeover({
|
||||
owner: owner(),
|
||||
now,
|
||||
isProcessAlive: () => true,
|
||||
readProcessStartedAt: readStartedAt,
|
||||
}),
|
||||
).resolves.toEqual({
|
||||
allowed: false,
|
||||
reason: "controller_still_alive",
|
||||
});
|
||||
expect(readStartedAt).toHaveBeenCalledWith(123);
|
||||
});
|
||||
|
||||
it("does not let lease expiry bypass a live controller fence", async () => {
|
||||
const isProcessAlive = vi.fn(() => true);
|
||||
await expect(
|
||||
evaluateNativeControllerTakeover({
|
||||
owner: owner({
|
||||
leaseExpiresAt: new Date("2026-09-04T11:59:59.000Z"),
|
||||
}),
|
||||
now,
|
||||
isProcessAlive,
|
||||
}),
|
||||
).resolves.toEqual({ allowed: false, reason: "controller_still_alive" });
|
||||
expect(isProcessAlive).toHaveBeenCalledWith(123);
|
||||
});
|
||||
|
||||
it("takes over an expired lease after proving the controller died", async () => {
|
||||
await expect(
|
||||
evaluateNativeControllerTakeover({
|
||||
owner: owner({
|
||||
leaseExpiresAt: new Date("2026-09-04T11:59:59.000Z"),
|
||||
}),
|
||||
now,
|
||||
isProcessAlive: () => false,
|
||||
}),
|
||||
).resolves.toEqual({
|
||||
allowed: true,
|
||||
reason: "expired_lease_controller_process_dead",
|
||||
});
|
||||
});
|
||||
|
||||
it("keeps a coordinated previous controller fenced until that exact process exits", async () => {
|
||||
await expect(
|
||||
evaluateNativeControllerTakeover({
|
||||
owner: owner(),
|
||||
now,
|
||||
coordinatedPreviousController: {
|
||||
pid: 123,
|
||||
processStartedAt: recordedStart,
|
||||
},
|
||||
isProcessAlive: () => true,
|
||||
readProcessStartedAt: async () => recordedStart,
|
||||
}),
|
||||
).resolves.toEqual({
|
||||
allowed: false,
|
||||
reason: "coordinated_controller_still_alive",
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe("native provider process fencing", () => {
|
||||
const recordedStart = new Date("2026-09-04T11:00:00.000Z");
|
||||
|
||||
it("does not mistake a recycled provider PID for the old live provider", async () => {
|
||||
await expect(
|
||||
evaluateNativeProviderProcesses({
|
||||
identities: [{ pid: 456, processStartedAt: recordedStart }],
|
||||
isProcessAlive: () => true,
|
||||
readProcessStartedAt: async () => new Date("2026-09-04T11:30:00.000Z"),
|
||||
}),
|
||||
).resolves.toEqual({
|
||||
knownPids: [456],
|
||||
livePids: [],
|
||||
ambiguousLivePids: [],
|
||||
recycledPids: [456],
|
||||
});
|
||||
});
|
||||
|
||||
it("fails closed when a live provider has no comparable fingerprint", async () => {
|
||||
await expect(
|
||||
evaluateNativeProviderProcesses({
|
||||
identities: [{ pid: 456, processStartedAt: null }],
|
||||
isProcessAlive: () => true,
|
||||
readProcessStartedAt: async () => recordedStart,
|
||||
}),
|
||||
).resolves.toMatchObject({
|
||||
livePids: [],
|
||||
ambiguousLivePids: [456],
|
||||
recycledPids: [],
|
||||
});
|
||||
});
|
||||
});
|
||||
|
|
@ -0,0 +1,854 @@
|
|||
import { randomUUID } from "node:crypto";
|
||||
import { and, desc, eq, inArray, isNull, lte, or, sql } from "drizzle-orm";
|
||||
import type { Db } from "@paperclipai/db";
|
||||
import {
|
||||
agents,
|
||||
heartbeatRunEvents,
|
||||
heartbeatRuns,
|
||||
issues,
|
||||
nativeRunFinalizations,
|
||||
} from "@paperclipai/db";
|
||||
import { readProcessStartedAt } from "../hot-restart.js";
|
||||
import { getServerInfoSnapshot } from "../../server-info.js";
|
||||
import { redactSensitiveText } from "../../redaction.js";
|
||||
|
||||
export type NativeControllerIdentity = {
|
||||
bootId: string;
|
||||
pid: number;
|
||||
processStartedAt: Date;
|
||||
};
|
||||
|
||||
export type NativeRestartKind = "hot" | "hard" | "graceful";
|
||||
|
||||
export type NativeRestartRecoveryClaim =
|
||||
| {
|
||||
kind: "reattach_existing_runner";
|
||||
runId: string;
|
||||
leaseOwner: string;
|
||||
controllerGeneration: number;
|
||||
providerAttempt: number;
|
||||
restartKind: NativeRestartKind;
|
||||
recoveryRequestId: string | null;
|
||||
process: {
|
||||
pid: number;
|
||||
processGroupId: number | null;
|
||||
startedAt: string;
|
||||
};
|
||||
}
|
||||
| {
|
||||
kind: "resume_dead_runner";
|
||||
runId: string;
|
||||
leaseOwner: string;
|
||||
controllerGeneration: number;
|
||||
providerAttempt: number;
|
||||
restartKind: NativeRestartKind;
|
||||
recoveryRequestId: string | null;
|
||||
}
|
||||
| {
|
||||
kind: "bootstrap_incomplete";
|
||||
runId: string;
|
||||
leaseOwner: string;
|
||||
controllerGeneration: number;
|
||||
providerAttempt: number;
|
||||
restartKind: NativeRestartKind;
|
||||
recoveryRequestId: string | null;
|
||||
};
|
||||
|
||||
export type NativeRestartRecoveryDisposition =
|
||||
| NativeRestartRecoveryClaim
|
||||
| {
|
||||
kind: "awaiting_evidence" | "blocked" | "already_finalized";
|
||||
runId: string;
|
||||
reason: string;
|
||||
};
|
||||
|
||||
export function nextNativeProviderAttempt(
|
||||
currentAttempt: number,
|
||||
recoveryKind?: NativeRestartRecoveryClaim["kind"],
|
||||
): number {
|
||||
return recoveryKind === "reattach_existing_runner" ||
|
||||
recoveryKind === "bootstrap_incomplete"
|
||||
? currentAttempt
|
||||
: currentAttempt + 1;
|
||||
}
|
||||
|
||||
const controllerBootId = randomUUID();
|
||||
|
||||
function processIsAlive(pid: number | null | undefined): boolean {
|
||||
if (!pid || !Number.isInteger(pid) || pid <= 0) return false;
|
||||
try {
|
||||
process.kill(pid, 0);
|
||||
return true;
|
||||
} catch (error) {
|
||||
return (error as NodeJS.ErrnoException | undefined)?.code === "EPERM";
|
||||
}
|
||||
}
|
||||
|
||||
function processGroupIsAlive(
|
||||
processGroupId: number | null | undefined,
|
||||
): boolean {
|
||||
if (
|
||||
process.platform === "win32" ||
|
||||
!processGroupId ||
|
||||
!Number.isInteger(processGroupId) ||
|
||||
processGroupId <= 0
|
||||
) {
|
||||
return false;
|
||||
}
|
||||
try {
|
||||
process.kill(-processGroupId, 0);
|
||||
return true;
|
||||
} catch (error) {
|
||||
return (error as NodeJS.ErrnoException | undefined)?.code === "EPERM";
|
||||
}
|
||||
}
|
||||
|
||||
async function observedProcessStart(pid: number): Promise<Date | null> {
|
||||
try {
|
||||
const startedAt = await readProcessStartedAt(pid);
|
||||
if (!startedAt) return null;
|
||||
const parsed = new Date(startedAt);
|
||||
return Number.isNaN(parsed.getTime()) ? null : parsed;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
export async function currentNativeControllerIdentity(): Promise<NativeControllerIdentity> {
|
||||
const observed = await observedProcessStart(process.pid);
|
||||
const fallback = new Date(getServerInfoSnapshot().processStartedAt);
|
||||
return {
|
||||
bootId: controllerBootId,
|
||||
pid: process.pid,
|
||||
processStartedAt:
|
||||
observed ?? (Number.isNaN(fallback.getTime()) ? new Date() : fallback),
|
||||
};
|
||||
}
|
||||
|
||||
function sameProcessStart(left: Date | null, right: Date | null): boolean {
|
||||
return left !== null && right !== null && left.getTime() === right.getTime();
|
||||
}
|
||||
|
||||
export async function evaluateNativeControllerTakeover(input: {
|
||||
owner: Pick<
|
||||
typeof nativeRunFinalizations.$inferSelect,
|
||||
| "leaseOwner"
|
||||
| "leaseExpiresAt"
|
||||
| "controllerPid"
|
||||
| "controllerProcessStartedAt"
|
||||
>;
|
||||
now: Date;
|
||||
coordinatedPreviousController?: {
|
||||
pid: number;
|
||||
processStartedAt: Date | null;
|
||||
} | null;
|
||||
isProcessAlive?: (pid: number) => boolean;
|
||||
readProcessStartedAt?: (pid: number) => Promise<Date | null>;
|
||||
}): Promise<{ allowed: boolean; reason: string }> {
|
||||
const { owner, now } = input;
|
||||
const isAlive = input.isProcessAlive ?? processIsAlive;
|
||||
const readStartedAt = input.readProcessStartedAt ?? observedProcessStart;
|
||||
if (!owner.leaseOwner || !owner.leaseExpiresAt) {
|
||||
return { allowed: true, reason: "unowned" };
|
||||
}
|
||||
|
||||
const priorPid = owner.controllerPid;
|
||||
if (!priorPid) {
|
||||
return {
|
||||
allowed: false,
|
||||
reason:
|
||||
owner.leaseExpiresAt <= now
|
||||
? "expired_lease_without_controller_identity"
|
||||
: "live_legacy_lease_without_controller_identity",
|
||||
};
|
||||
}
|
||||
if (!isAlive(priorPid)) {
|
||||
return {
|
||||
allowed: true,
|
||||
reason:
|
||||
owner.leaseExpiresAt <= now
|
||||
? "expired_lease_controller_process_dead"
|
||||
: "controller_process_dead",
|
||||
};
|
||||
}
|
||||
|
||||
const observedStartedAt = await readStartedAt(priorPid);
|
||||
if (
|
||||
owner.controllerProcessStartedAt &&
|
||||
observedStartedAt &&
|
||||
!sameProcessStart(owner.controllerProcessStartedAt, observedStartedAt)
|
||||
) {
|
||||
return { allowed: true, reason: "controller_pid_recycled" };
|
||||
}
|
||||
|
||||
const coordinated = input.coordinatedPreviousController;
|
||||
if (
|
||||
coordinated &&
|
||||
coordinated.pid === priorPid &&
|
||||
coordinated.processStartedAt &&
|
||||
owner.controllerProcessStartedAt &&
|
||||
sameProcessStart(
|
||||
coordinated.processStartedAt,
|
||||
owner.controllerProcessStartedAt,
|
||||
)
|
||||
) {
|
||||
return { allowed: false, reason: "coordinated_controller_still_alive" };
|
||||
}
|
||||
|
||||
return { allowed: false, reason: "controller_still_alive" };
|
||||
}
|
||||
|
||||
export type NativeProviderProcessIdentity = {
|
||||
pid: number;
|
||||
processStartedAt: Date | null;
|
||||
};
|
||||
|
||||
export async function evaluateNativeProviderProcesses(input: {
|
||||
identities: NativeProviderProcessIdentity[];
|
||||
isProcessAlive?: (pid: number) => boolean;
|
||||
readProcessStartedAt?: (pid: number) => Promise<Date | null>;
|
||||
}): Promise<{
|
||||
knownPids: number[];
|
||||
livePids: number[];
|
||||
ambiguousLivePids: number[];
|
||||
recycledPids: number[];
|
||||
}> {
|
||||
const isAlive = input.isProcessAlive ?? processIsAlive;
|
||||
const readStartedAt = input.readProcessStartedAt ?? observedProcessStart;
|
||||
const expectedStarts = new Map<number, Date | null>();
|
||||
for (const identity of input.identities) {
|
||||
const current = expectedStarts.get(identity.pid);
|
||||
if (
|
||||
current === undefined ||
|
||||
(current === null && identity.processStartedAt)
|
||||
) {
|
||||
expectedStarts.set(identity.pid, identity.processStartedAt);
|
||||
}
|
||||
}
|
||||
|
||||
const livePids: number[] = [];
|
||||
const ambiguousLivePids: number[] = [];
|
||||
const recycledPids: number[] = [];
|
||||
for (const [pid, expectedStartedAt] of expectedStarts) {
|
||||
if (!isAlive(pid)) continue;
|
||||
const observedStartedAt = await readStartedAt(pid);
|
||||
if (!expectedStartedAt || !observedStartedAt) {
|
||||
ambiguousLivePids.push(pid);
|
||||
} else if (sameProcessStart(expectedStartedAt, observedStartedAt)) {
|
||||
livePids.push(pid);
|
||||
} else {
|
||||
recycledPids.push(pid);
|
||||
}
|
||||
}
|
||||
return {
|
||||
knownPids: [...expectedStarts.keys()],
|
||||
livePids,
|
||||
ambiguousLivePids,
|
||||
recycledPids,
|
||||
};
|
||||
}
|
||||
|
||||
export function classifyNativeRunnerRecoveryEvidence(input: {
|
||||
runnerPidAlive: boolean;
|
||||
runnerGroupAlive: boolean;
|
||||
processStartMatches: boolean;
|
||||
knownProviderProcessAlive?: boolean;
|
||||
knownProviderProcessIdentityAmbiguous?: boolean;
|
||||
hasCheckpoint: boolean;
|
||||
checkpointIdentityMatches?: boolean;
|
||||
hasProviderEvidence: boolean;
|
||||
}): {
|
||||
claimKind: NativeRestartRecoveryClaim["kind"] | null;
|
||||
reason: string;
|
||||
} {
|
||||
if (input.runnerPidAlive && input.processStartMatches) {
|
||||
return {
|
||||
claimKind: "reattach_existing_runner",
|
||||
reason: "runner_process_identity_verified",
|
||||
};
|
||||
}
|
||||
if (input.runnerPidAlive || input.runnerGroupAlive) {
|
||||
return {
|
||||
claimKind: null,
|
||||
reason: input.runnerPidAlive
|
||||
? "live_runner_identity_mismatch"
|
||||
: "live_runner_process_group_without_exact_pid",
|
||||
};
|
||||
}
|
||||
if (input.knownProviderProcessAlive) {
|
||||
return {
|
||||
claimKind: null,
|
||||
reason: "live_provider_process_without_runner_authority",
|
||||
};
|
||||
}
|
||||
if (input.knownProviderProcessIdentityAmbiguous) {
|
||||
return {
|
||||
claimKind: null,
|
||||
reason: "live_provider_process_identity_unverifiable",
|
||||
};
|
||||
}
|
||||
const checkpointIdentityMatches =
|
||||
input.checkpointIdentityMatches ?? input.hasCheckpoint;
|
||||
if (
|
||||
input.hasCheckpoint &&
|
||||
checkpointIdentityMatches &&
|
||||
input.hasProviderEvidence
|
||||
) {
|
||||
return {
|
||||
claimKind: "resume_dead_runner",
|
||||
reason: "dead_runner_with_exact_provider_checkpoint",
|
||||
};
|
||||
}
|
||||
if (!input.hasCheckpoint && !input.hasProviderEvidence) {
|
||||
return {
|
||||
claimKind: "bootstrap_incomplete",
|
||||
reason: "dead_runner_before_provider_identity",
|
||||
};
|
||||
}
|
||||
return { claimKind: null, reason: "ambiguous_provider_recovery_evidence" };
|
||||
}
|
||||
|
||||
function historyEntry(input: {
|
||||
now: Date;
|
||||
restartKind: NativeRestartKind;
|
||||
controller: NativeControllerIdentity;
|
||||
generation: number;
|
||||
disposition: string;
|
||||
reason: string;
|
||||
requestId: string | null;
|
||||
providerAttempt: number;
|
||||
processPid: number | null;
|
||||
processStartedAt: Date | null;
|
||||
hasCheckpoint: boolean;
|
||||
checkpointIdentityMatches?: boolean;
|
||||
hasProviderEvidence: boolean;
|
||||
knownProviderPids?: number[];
|
||||
liveProviderPids?: number[];
|
||||
ambiguousProviderPids?: number[];
|
||||
recycledProviderPids?: number[];
|
||||
stderrTail?: string | null;
|
||||
}) {
|
||||
return {
|
||||
at: input.now.toISOString(),
|
||||
restartKind: input.restartKind,
|
||||
controllerBootId: input.controller.bootId,
|
||||
controllerPid: input.controller.pid,
|
||||
controllerProcessStartedAt: input.controller.processStartedAt.toISOString(),
|
||||
controllerGeneration: input.generation,
|
||||
disposition: input.disposition,
|
||||
reason: input.reason,
|
||||
stateRootAction:
|
||||
input.disposition === "reattach_existing_runner"
|
||||
? "reopen_exact_root"
|
||||
: input.disposition === "resume_dead_runner"
|
||||
? "reuse_exact_root"
|
||||
: input.disposition === "bootstrap_incomplete"
|
||||
? "quarantine_incomplete_root_then_bootstrap"
|
||||
: input.disposition === "awaiting_evidence"
|
||||
? "preserve_awaiting_evidence"
|
||||
: "preserve_fail_closed",
|
||||
recoveryRequestId: input.requestId,
|
||||
providerAttempt: input.providerAttempt,
|
||||
processPid: input.processPid,
|
||||
processStartedAt: input.processStartedAt?.toISOString() ?? null,
|
||||
hasCheckpoint: input.hasCheckpoint,
|
||||
checkpointIdentityMatches: input.checkpointIdentityMatches ?? false,
|
||||
hasProviderEvidence: input.hasProviderEvidence,
|
||||
knownProviderPids: input.knownProviderPids ?? [],
|
||||
liveProviderPids: input.liveProviderPids ?? [],
|
||||
ambiguousProviderPids: input.ambiguousProviderPids ?? [],
|
||||
recycledProviderPids: input.recycledProviderPids ?? [],
|
||||
stderrTail:
|
||||
input.stderrTail && input.stderrTail.trim()
|
||||
? redactSensitiveText(input.stderrTail).slice(-4_096)
|
||||
: null,
|
||||
} satisfies Record<string, unknown>;
|
||||
}
|
||||
|
||||
function appendBoundedRecoveryHistory(entry: Record<string, unknown>) {
|
||||
const encoded = JSON.stringify(entry);
|
||||
return sql`(
|
||||
select coalesce(jsonb_agg(item order by ordinal), '[]'::jsonb)
|
||||
from jsonb_array_elements(
|
||||
coalesce(${nativeRunFinalizations.recoveryHistory}, '[]'::jsonb)
|
||||
|| jsonb_build_array(${encoded}::jsonb)
|
||||
) with ordinality as history(item, ordinal)
|
||||
where ordinal > greatest(
|
||||
jsonb_array_length(
|
||||
coalesce(${nativeRunFinalizations.recoveryHistory}, '[]'::jsonb)
|
||||
|| jsonb_build_array(${encoded}::jsonb)
|
||||
) - 20,
|
||||
0
|
||||
)
|
||||
)`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Claims abandoned local runner executions without changing the provider retry
|
||||
* attempt. The transaction locks the run, coordinator, and issue execution
|
||||
* owner so two successor servers cannot both reconstruct the same authority.
|
||||
*/
|
||||
export async function claimNativeRestartRecoveries(input: {
|
||||
db: Db;
|
||||
controller?: NativeControllerIdentity;
|
||||
restartKind: NativeRestartKind;
|
||||
recoveryRequestId?: string | null;
|
||||
runIds?: string[];
|
||||
now?: Date;
|
||||
limit?: number;
|
||||
coordinatedPreviousController?: {
|
||||
pid: number;
|
||||
processStartedAt: Date | null;
|
||||
} | null;
|
||||
}): Promise<NativeRestartRecoveryDisposition[]> {
|
||||
const now = input.now ?? new Date();
|
||||
const controller =
|
||||
input.controller ?? (await currentNativeControllerIdentity());
|
||||
const candidateQuery = input.db
|
||||
.select({ runId: heartbeatRuns.id })
|
||||
.from(heartbeatRuns)
|
||||
.innerJoin(agents, eq(agents.id, heartbeatRuns.agentId))
|
||||
.innerJoin(
|
||||
nativeRunFinalizations,
|
||||
eq(nativeRunFinalizations.runId, heartbeatRuns.id),
|
||||
)
|
||||
.where(
|
||||
and(
|
||||
eq(heartbeatRuns.runtimeMode, "native"),
|
||||
eq(agents.adapterType, "paperclip_runner"),
|
||||
inArray(heartbeatRuns.status, ["running", "failed"]),
|
||||
isNull(nativeRunFinalizations.resultId),
|
||||
or(
|
||||
eq(nativeRunFinalizations.phase, "observed"),
|
||||
and(
|
||||
eq(nativeRunFinalizations.phase, "retryable_failure"),
|
||||
or(
|
||||
isNull(nativeRunFinalizations.nextAttemptAt),
|
||||
lte(nativeRunFinalizations.nextAttemptAt, now),
|
||||
),
|
||||
),
|
||||
),
|
||||
...(input.runIds?.length
|
||||
? [inArray(heartbeatRuns.id, input.runIds)]
|
||||
: []),
|
||||
),
|
||||
)
|
||||
.orderBy(heartbeatRuns.id);
|
||||
const candidates = await (input.limit === undefined
|
||||
? candidateQuery
|
||||
: candidateQuery.limit(input.limit));
|
||||
|
||||
const dispositions: NativeRestartRecoveryDisposition[] = [];
|
||||
for (const candidate of candidates) {
|
||||
const disposition = await input.db.transaction(async (tx) => {
|
||||
const row = await tx
|
||||
.select({
|
||||
run: heartbeatRuns,
|
||||
coordinator: nativeRunFinalizations,
|
||||
issueExecutionRunId: issues.executionRunId,
|
||||
})
|
||||
.from(heartbeatRuns)
|
||||
.innerJoin(
|
||||
nativeRunFinalizations,
|
||||
eq(nativeRunFinalizations.runId, heartbeatRuns.id),
|
||||
)
|
||||
.innerJoin(
|
||||
issues,
|
||||
and(
|
||||
eq(issues.id, nativeRunFinalizations.issueId),
|
||||
eq(issues.companyId, nativeRunFinalizations.companyId),
|
||||
),
|
||||
)
|
||||
.where(eq(heartbeatRuns.id, candidate.runId))
|
||||
.for("update")
|
||||
.limit(1)
|
||||
.then((rows) => rows[0] ?? null);
|
||||
if (!row) {
|
||||
return {
|
||||
kind: "blocked",
|
||||
runId: candidate.runId,
|
||||
reason: "recovery_rows_missing",
|
||||
} as const;
|
||||
}
|
||||
if (row.coordinator.resultId) {
|
||||
return {
|
||||
kind: "already_finalized",
|
||||
runId: row.run.id,
|
||||
reason: "result_already_persisted",
|
||||
} as const;
|
||||
}
|
||||
if (row.issueExecutionRunId !== row.run.id) {
|
||||
return {
|
||||
kind: "blocked",
|
||||
runId: row.run.id,
|
||||
reason: "issue_execution_lock_changed",
|
||||
} as const;
|
||||
}
|
||||
|
||||
const takeover = await evaluateNativeControllerTakeover({
|
||||
owner: row.coordinator,
|
||||
now,
|
||||
coordinatedPreviousController:
|
||||
input.coordinatedPreviousController ?? null,
|
||||
});
|
||||
if (!takeover.allowed) {
|
||||
const event = historyEntry({
|
||||
now,
|
||||
restartKind: input.restartKind,
|
||||
controller,
|
||||
generation: row.coordinator.controllerGeneration,
|
||||
disposition: "awaiting_evidence",
|
||||
reason: takeover.reason,
|
||||
requestId: input.recoveryRequestId ?? null,
|
||||
providerAttempt: row.coordinator.attempt,
|
||||
processPid: row.run.processPid,
|
||||
processStartedAt: row.run.processStartedAt,
|
||||
hasCheckpoint: false,
|
||||
hasProviderEvidence: false,
|
||||
stderrTail: row.run.stderrExcerpt,
|
||||
});
|
||||
await tx
|
||||
.update(nativeRunFinalizations)
|
||||
.set({
|
||||
recoveryState: "awaiting_evidence",
|
||||
recoveryRequestId: input.recoveryRequestId ?? null,
|
||||
recoveryHistory: appendBoundedRecoveryHistory(event),
|
||||
updatedAt: now,
|
||||
})
|
||||
.where(
|
||||
and(
|
||||
eq(nativeRunFinalizations.runId, row.run.id),
|
||||
eq(nativeRunFinalizations.attempt, row.coordinator.attempt),
|
||||
eq(nativeRunFinalizations.phase, row.coordinator.phase),
|
||||
),
|
||||
);
|
||||
return {
|
||||
kind: "awaiting_evidence",
|
||||
runId: row.run.id,
|
||||
reason: takeover.reason,
|
||||
} as const;
|
||||
}
|
||||
|
||||
const runnerPidAlive = processIsAlive(row.run.processPid);
|
||||
const runnerGroupAlive = processGroupIsAlive(row.run.processGroupId);
|
||||
const observedRunnerStart =
|
||||
row.run.processPid && runnerPidAlive
|
||||
? await observedProcessStart(row.run.processPid)
|
||||
: null;
|
||||
const exactRunnerIdentity =
|
||||
runnerPidAlive &&
|
||||
row.run.processPid !== null &&
|
||||
sameProcessStart(row.run.processStartedAt, observedRunnerStart);
|
||||
|
||||
const profile = row.run.runnerProfileJson ?? {};
|
||||
const checkpoint = profile.sessionCheckpoint;
|
||||
const checkpointRecord =
|
||||
checkpoint &&
|
||||
typeof checkpoint === "object" &&
|
||||
!Array.isArray(checkpoint)
|
||||
? (checkpoint as Record<string, unknown>)
|
||||
: {};
|
||||
const hasCheckpoint = checkpoint !== undefined && checkpoint !== null;
|
||||
const checkpointIdentity =
|
||||
checkpointRecord.identity &&
|
||||
typeof checkpointRecord.identity === "object" &&
|
||||
!Array.isArray(checkpointRecord.identity)
|
||||
? (checkpointRecord.identity as Record<string, unknown>)
|
||||
: {};
|
||||
const checkpointIdentityMatches =
|
||||
hasCheckpoint &&
|
||||
typeof row.run.nativeSessionId === "string" &&
|
||||
row.run.nativeSessionId.length > 0 &&
|
||||
checkpointIdentity.runId === row.run.id &&
|
||||
checkpointIdentity.companyId === row.run.companyId &&
|
||||
checkpointIdentity.issueId === row.coordinator.issueId &&
|
||||
checkpointIdentity.agentId === row.run.agentId &&
|
||||
checkpointIdentity.sessionId === row.run.nativeSessionId;
|
||||
const hasCheckpointProviderIdentity =
|
||||
(typeof checkpointRecord.providerSessionId === "string" &&
|
||||
checkpointRecord.providerSessionId.length > 0) ||
|
||||
(checkpointRecord.providerIdentity !== null &&
|
||||
typeof checkpointRecord.providerIdentity === "object" &&
|
||||
!Array.isArray(checkpointRecord.providerIdentity) &&
|
||||
Object.keys(checkpointRecord.providerIdentity).length > 0);
|
||||
const providerEvents = await tx
|
||||
.select({
|
||||
id: heartbeatRunEvents.id,
|
||||
payload: heartbeatRunEvents.payload,
|
||||
})
|
||||
.from(heartbeatRunEvents)
|
||||
.where(
|
||||
and(
|
||||
eq(heartbeatRunEvents.runId, row.run.id),
|
||||
inArray(heartbeatRunEvents.eventType, [
|
||||
"harness.ready",
|
||||
"session.started",
|
||||
"session.resumed",
|
||||
"session.updated",
|
||||
"turn.started",
|
||||
"provider.event",
|
||||
"provider.rpc_result",
|
||||
]),
|
||||
),
|
||||
)
|
||||
.orderBy(desc(heartbeatRunEvents.id))
|
||||
.limit(100);
|
||||
const checkpointProcess =
|
||||
checkpointRecord.process &&
|
||||
typeof checkpointRecord.process === "object" &&
|
||||
!Array.isArray(checkpointRecord.process)
|
||||
? (checkpointRecord.process as Record<string, unknown>)
|
||||
: {};
|
||||
const providerProcessIdentities: NativeProviderProcessIdentity[] = [];
|
||||
const checkpointProcessFields = [
|
||||
["providerPid", "providerProcessStartedAt"],
|
||||
["codexPid", "codexProcessStartedAt"],
|
||||
["sidecarPid", "sidecarProcessStartedAt"],
|
||||
["agentPid", "agentProcessStartedAt"],
|
||||
] as const;
|
||||
for (const [field, startedAtField] of checkpointProcessFields) {
|
||||
const value = checkpointProcess[field];
|
||||
if (typeof value === "number" && Number.isInteger(value) && value > 0) {
|
||||
const rawStartedAt = checkpointProcess[startedAtField];
|
||||
const parsedStartedAt =
|
||||
typeof rawStartedAt === "string" ? new Date(rawStartedAt) : null;
|
||||
providerProcessIdentities.push({
|
||||
pid: value,
|
||||
processStartedAt:
|
||||
parsedStartedAt && !Number.isNaN(parsedStartedAt.getTime())
|
||||
? parsedStartedAt
|
||||
: null,
|
||||
});
|
||||
}
|
||||
}
|
||||
for (const providerEvent of providerEvents) {
|
||||
const prpEvent = providerEvent.payload?.prpEvent;
|
||||
const event =
|
||||
prpEvent && typeof prpEvent === "object" && !Array.isArray(prpEvent)
|
||||
? (prpEvent as Record<string, unknown>)
|
||||
: {};
|
||||
const eventPayload =
|
||||
event.payload &&
|
||||
typeof event.payload === "object" &&
|
||||
!Array.isArray(event.payload)
|
||||
? (event.payload as Record<string, unknown>)
|
||||
: {};
|
||||
const processId = eventPayload.processId;
|
||||
if (
|
||||
typeof processId === "number" &&
|
||||
Number.isInteger(processId) &&
|
||||
processId > 0
|
||||
) {
|
||||
const rawStartedAt =
|
||||
eventPayload.processStartedAt ??
|
||||
eventPayload.providerProcessStartedAt;
|
||||
const parsedStartedAt =
|
||||
typeof rawStartedAt === "string" ? new Date(rawStartedAt) : null;
|
||||
providerProcessIdentities.push({
|
||||
pid: processId,
|
||||
processStartedAt:
|
||||
parsedStartedAt && !Number.isNaN(parsedStartedAt.getTime())
|
||||
? parsedStartedAt
|
||||
: null,
|
||||
});
|
||||
}
|
||||
}
|
||||
const providerProcesses = await evaluateNativeProviderProcesses({
|
||||
identities: providerProcessIdentities.filter(
|
||||
(identity) => identity.pid !== row.run.processPid,
|
||||
),
|
||||
});
|
||||
const hasProviderEvidence =
|
||||
hasCheckpointProviderIdentity || providerEvents.length > 0;
|
||||
|
||||
const classification = classifyNativeRunnerRecoveryEvidence({
|
||||
runnerPidAlive,
|
||||
runnerGroupAlive,
|
||||
processStartMatches: exactRunnerIdentity,
|
||||
knownProviderProcessAlive: providerProcesses.livePids.length > 0,
|
||||
knownProviderProcessIdentityAmbiguous:
|
||||
providerProcesses.ambiguousLivePids.length > 0,
|
||||
hasCheckpoint,
|
||||
checkpointIdentityMatches,
|
||||
hasProviderEvidence,
|
||||
});
|
||||
const claimKind = classification.claimKind;
|
||||
const reason = classification.reason;
|
||||
|
||||
if (!claimKind) {
|
||||
const generation = row.coordinator.controllerGeneration;
|
||||
const event = historyEntry({
|
||||
now,
|
||||
restartKind: input.restartKind,
|
||||
controller,
|
||||
generation,
|
||||
disposition: "blocked",
|
||||
reason,
|
||||
requestId: input.recoveryRequestId ?? null,
|
||||
providerAttempt: row.coordinator.attempt,
|
||||
processPid: row.run.processPid,
|
||||
processStartedAt: row.run.processStartedAt,
|
||||
hasCheckpoint,
|
||||
checkpointIdentityMatches,
|
||||
hasProviderEvidence,
|
||||
knownProviderPids: providerProcesses.knownPids,
|
||||
liveProviderPids: providerProcesses.livePids,
|
||||
ambiguousProviderPids: providerProcesses.ambiguousLivePids,
|
||||
recycledProviderPids: providerProcesses.recycledPids,
|
||||
stderrTail: row.run.stderrExcerpt,
|
||||
});
|
||||
await tx
|
||||
.update(nativeRunFinalizations)
|
||||
.set({
|
||||
recoveryState: "blocked",
|
||||
recoveryRequestId: input.recoveryRequestId ?? null,
|
||||
recoveryHistory: appendBoundedRecoveryHistory(event),
|
||||
updatedAt: now,
|
||||
})
|
||||
.where(
|
||||
and(
|
||||
eq(nativeRunFinalizations.runId, row.run.id),
|
||||
eq(nativeRunFinalizations.attempt, row.coordinator.attempt),
|
||||
eq(nativeRunFinalizations.phase, row.coordinator.phase),
|
||||
),
|
||||
);
|
||||
return { kind: "blocked", runId: row.run.id, reason } as const;
|
||||
}
|
||||
|
||||
const generation = row.coordinator.controllerGeneration + 1;
|
||||
const leaseOwner = `${controller.bootId}:${generation}:${randomUUID()}`;
|
||||
const event = historyEntry({
|
||||
now,
|
||||
restartKind: input.restartKind,
|
||||
controller,
|
||||
generation,
|
||||
disposition: claimKind,
|
||||
reason,
|
||||
requestId: input.recoveryRequestId ?? null,
|
||||
providerAttempt: row.coordinator.attempt,
|
||||
processPid: row.run.processPid,
|
||||
processStartedAt: row.run.processStartedAt,
|
||||
hasCheckpoint,
|
||||
checkpointIdentityMatches,
|
||||
hasProviderEvidence,
|
||||
knownProviderPids: providerProcesses.knownPids,
|
||||
liveProviderPids: providerProcesses.livePids,
|
||||
ambiguousProviderPids: providerProcesses.ambiguousLivePids,
|
||||
recycledProviderPids: providerProcesses.recycledPids,
|
||||
stderrTail: row.run.stderrExcerpt,
|
||||
});
|
||||
const claimed = await tx
|
||||
.update(nativeRunFinalizations)
|
||||
.set({
|
||||
phase: "observed",
|
||||
leaseOwner,
|
||||
leaseExpiresAt: new Date(now.getTime() + 20 * 60_000),
|
||||
controllerBootId: controller.bootId,
|
||||
controllerPid: controller.pid,
|
||||
controllerProcessStartedAt: controller.processStartedAt,
|
||||
controllerGeneration: generation,
|
||||
recoveryState:
|
||||
claimKind === "reattach_existing_runner"
|
||||
? "awaiting_runner_reattach"
|
||||
: claimKind === "resume_dead_runner"
|
||||
? "resuming_session"
|
||||
: "bootstrap_incomplete",
|
||||
recoveryRequestId: input.recoveryRequestId ?? null,
|
||||
recoveryHistory: appendBoundedRecoveryHistory(event),
|
||||
failureCode: null,
|
||||
nextAttemptAt: null,
|
||||
updatedAt: now,
|
||||
})
|
||||
.where(
|
||||
and(
|
||||
eq(nativeRunFinalizations.runId, row.run.id),
|
||||
eq(nativeRunFinalizations.attempt, row.coordinator.attempt),
|
||||
eq(nativeRunFinalizations.phase, row.coordinator.phase),
|
||||
row.coordinator.leaseOwner === null
|
||||
? isNull(nativeRunFinalizations.leaseOwner)
|
||||
: eq(
|
||||
nativeRunFinalizations.leaseOwner,
|
||||
row.coordinator.leaseOwner,
|
||||
),
|
||||
),
|
||||
)
|
||||
.returning({ runId: nativeRunFinalizations.runId })
|
||||
.then((rows) => rows[0] ?? null);
|
||||
if (!claimed) {
|
||||
return {
|
||||
kind: "awaiting_evidence",
|
||||
runId: row.run.id,
|
||||
reason: "concurrent_recovery_claim",
|
||||
} as const;
|
||||
}
|
||||
|
||||
await tx
|
||||
.update(heartbeatRuns)
|
||||
.set({
|
||||
status: "running",
|
||||
finishedAt: null,
|
||||
error: null,
|
||||
errorCode: null,
|
||||
nativePhase: "observed",
|
||||
nativePhaseUpdatedAt: now,
|
||||
...(claimKind === "reattach_existing_runner"
|
||||
? {}
|
||||
: {
|
||||
processPid: null,
|
||||
processGroupId: null,
|
||||
processStartedAt: null,
|
||||
}),
|
||||
updatedAt: now,
|
||||
})
|
||||
.where(eq(heartbeatRuns.id, row.run.id));
|
||||
|
||||
const common = {
|
||||
runId: row.run.id,
|
||||
leaseOwner,
|
||||
controllerGeneration: generation,
|
||||
providerAttempt: row.coordinator.attempt,
|
||||
restartKind: input.restartKind,
|
||||
recoveryRequestId: input.recoveryRequestId ?? null,
|
||||
};
|
||||
if (claimKind === "reattach_existing_runner") {
|
||||
return {
|
||||
kind: claimKind,
|
||||
...common,
|
||||
process: {
|
||||
pid: row.run.processPid!,
|
||||
processGroupId: row.run.processGroupId,
|
||||
startedAt: row.run.processStartedAt!.toISOString(),
|
||||
},
|
||||
} satisfies NativeRestartRecoveryClaim;
|
||||
}
|
||||
return {
|
||||
kind: claimKind,
|
||||
...common,
|
||||
} satisfies NativeRestartRecoveryClaim;
|
||||
});
|
||||
dispositions.push(disposition);
|
||||
}
|
||||
return dispositions;
|
||||
}
|
||||
|
||||
export async function nativeRestartRecoverySummary(db: Db) {
|
||||
const rows = await db
|
||||
.select({
|
||||
state: nativeRunFinalizations.recoveryState,
|
||||
count: sql<number>`count(*)::int`,
|
||||
})
|
||||
.from(nativeRunFinalizations)
|
||||
.where(
|
||||
inArray(nativeRunFinalizations.recoveryState, [
|
||||
"awaiting_evidence",
|
||||
"awaiting_runner_reattach",
|
||||
"resuming_session",
|
||||
"bootstrap_incomplete",
|
||||
"blocked",
|
||||
]),
|
||||
)
|
||||
.groupBy(nativeRunFinalizations.recoveryState);
|
||||
return Object.fromEntries(
|
||||
rows.map((row) => [row.state ?? "unknown", row.count]),
|
||||
);
|
||||
}
|
||||
|
|
@ -0,0 +1,998 @@
|
|||
import { randomUUID } from "node:crypto";
|
||||
import { spawn } from "node:child_process";
|
||||
import { existsSync } from "node:fs";
|
||||
import { mkdtemp, readFile, rename, rm } from "node:fs/promises";
|
||||
import { createServer, type Server } from "node:http";
|
||||
import { tmpdir } from "node:os";
|
||||
import { resolve } from "node:path";
|
||||
|
||||
import { and, eq } from "drizzle-orm";
|
||||
import {
|
||||
agents,
|
||||
companies,
|
||||
createDb,
|
||||
heartbeatRuns,
|
||||
issues,
|
||||
nativeRunFinalizations,
|
||||
nativeRunResults,
|
||||
} from "@paperclipai/db";
|
||||
import { afterAll, beforeAll, describe, expect, it, vi } from "vitest";
|
||||
|
||||
import {
|
||||
createRunnerdCodexTransport,
|
||||
defaultCapabilityRunnerdBinary,
|
||||
} from "../../vendor/paperclip-runner/index.js";
|
||||
import {
|
||||
getEmbeddedPostgresTestSupport,
|
||||
startEmbeddedPostgresTestDatabase,
|
||||
} from "../../__tests__/helpers/embedded-postgres.js";
|
||||
import {
|
||||
registerRunnerPrpAuthority,
|
||||
runnerPrpWebSocketInternals,
|
||||
setupRunnerPrpWebSocketServer,
|
||||
} from "../../realtime/runner-prp-ws.js";
|
||||
import { readProcessStartedAt } from "../hot-restart.js";
|
||||
import { prepareNativeHeartbeatRun } from "./prepare-native-run.js";
|
||||
import {
|
||||
claimNativeRestartRecoveries,
|
||||
type NativeControllerIdentity,
|
||||
} from "./native-restart-recovery.js";
|
||||
|
||||
const embeddedPostgresSupport = await getEmbeddedPostgresTestSupport();
|
||||
const describeEmbeddedPostgres = embeddedPostgresSupport.supported
|
||||
? describe
|
||||
: describe.skip;
|
||||
|
||||
const fakeCodexAppServer = resolve(
|
||||
import.meta.dirname,
|
||||
"../../../../packages/paperclip-runner/runner/target/debug/fake-codex-app-server",
|
||||
);
|
||||
const binariesAvailable =
|
||||
existsSync(defaultCapabilityRunnerdBinary()) && existsSync(fakeCodexAppServer);
|
||||
const realProcessIt = binariesAvailable ? it : it.skip;
|
||||
|
||||
async function closeServer(server: Server | null): Promise<void> {
|
||||
if (!server) return;
|
||||
server.closeAllConnections();
|
||||
if (!server.listening) return;
|
||||
await new Promise<void>((resolveClose) => server.close(() => resolveClose()));
|
||||
}
|
||||
|
||||
function processAlive(pid: number | null): boolean {
|
||||
if (!pid) return false;
|
||||
try {
|
||||
process.kill(pid, 0);
|
||||
return true;
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
async function waitForCondition(
|
||||
description: string,
|
||||
condition: () => boolean | Promise<boolean>,
|
||||
timeoutMs = 10_000,
|
||||
): Promise<void> {
|
||||
const deadline = Date.now() + timeoutMs;
|
||||
while (Date.now() < deadline) {
|
||||
if (await condition()) return;
|
||||
await new Promise<void>((resolveWait) => setTimeout(resolveWait, 25));
|
||||
}
|
||||
throw new Error(`Timed out waiting for ${description}`);
|
||||
}
|
||||
|
||||
async function stopOwnedProcessGroup(
|
||||
pid: number | null,
|
||||
expectedStateRoot: string,
|
||||
): Promise<void> {
|
||||
if (!pid || !processAlive(pid)) return;
|
||||
const command = await import("node:child_process").then(
|
||||
({ execFileSync }) =>
|
||||
execFileSync("ps", ["-p", String(pid), "-o", "command="], {
|
||||
encoding: "utf8",
|
||||
}).trim(),
|
||||
);
|
||||
if (!command.includes(expectedStateRoot)) {
|
||||
throw new Error(`Refusing to signal process ${pid}; ownership changed`);
|
||||
}
|
||||
if (process.platform !== "win32") process.kill(-pid, "SIGKILL");
|
||||
else process.kill(pid, "SIGKILL");
|
||||
await waitForCondition(`owned process ${pid} to exit`, () => !processAlive(pid));
|
||||
}
|
||||
|
||||
describeEmbeddedPostgres("native runner restart recovery with real processes", () => {
|
||||
let temporary: Awaited<ReturnType<typeof startEmbeddedPostgresTestDatabase>>;
|
||||
let runtimeRoot: string;
|
||||
let paperclipHome: string;
|
||||
let server: Server | null = null;
|
||||
let apiUrl: string;
|
||||
let originalPaperclipHome: string | undefined;
|
||||
|
||||
const companyId = randomUUID();
|
||||
const agentId = randomUUID();
|
||||
let successor: NativeControllerIdentity;
|
||||
|
||||
beforeAll(async () => {
|
||||
temporary = await startEmbeddedPostgresTestDatabase(
|
||||
"native-runner-restart-recovery-",
|
||||
);
|
||||
runtimeRoot = await mkdtemp(resolve(tmpdir(), "native-restart-runtime-"));
|
||||
paperclipHome = await mkdtemp(resolve(tmpdir(), "native-restart-home-"));
|
||||
originalPaperclipHome = process.env.PAPERCLIP_HOME;
|
||||
process.env.PAPERCLIP_HOME = paperclipHome;
|
||||
const controllerStartedAt = await readProcessStartedAt(process.pid);
|
||||
successor = {
|
||||
bootId: randomUUID(),
|
||||
pid: process.pid,
|
||||
processStartedAt: controllerStartedAt
|
||||
? new Date(controllerStartedAt)
|
||||
: new Date(),
|
||||
};
|
||||
server = createServer();
|
||||
await new Promise<void>((resolveListen) =>
|
||||
server!.listen(0, "127.0.0.1", resolveListen),
|
||||
);
|
||||
const address = server.address();
|
||||
if (!address || typeof address === "string") {
|
||||
throw new Error("Expected restart recovery TCP listener");
|
||||
}
|
||||
apiUrl = `http://127.0.0.1:${address.port}`;
|
||||
setupRunnerPrpWebSocketServer(server, { apiUrl });
|
||||
|
||||
const db = createDb(temporary.connectionString);
|
||||
await db.insert(companies).values({
|
||||
id: companyId,
|
||||
name: "Native restart recovery",
|
||||
issuePrefix: "NRR",
|
||||
requireBoardApprovalForNewAgents: false,
|
||||
});
|
||||
await db.insert(agents).values({
|
||||
id: agentId,
|
||||
companyId,
|
||||
name: "Native restart runner",
|
||||
role: "engineer",
|
||||
status: "active",
|
||||
adapterType: "paperclip_runner",
|
||||
adapterConfig: { provider: "codex" },
|
||||
runtimeConfig: {},
|
||||
permissions: {},
|
||||
});
|
||||
});
|
||||
|
||||
afterAll(async () => {
|
||||
runnerPrpWebSocketInternals.resetForTests();
|
||||
await closeServer(server);
|
||||
await temporary.cleanup();
|
||||
await Promise.all([
|
||||
rm(runtimeRoot, { recursive: true, force: true }),
|
||||
rm(paperclipHome, { recursive: true, force: true }),
|
||||
]);
|
||||
if (originalPaperclipHome === undefined) delete process.env.PAPERCLIP_HOME;
|
||||
else process.env.PAPERCLIP_HOME = originalPaperclipHome;
|
||||
});
|
||||
|
||||
async function seedRun(
|
||||
label: string,
|
||||
existingDb?: ReturnType<typeof createDb>,
|
||||
) {
|
||||
const db = existingDb ?? createDb(temporary.connectionString);
|
||||
const issueId = randomUUID();
|
||||
const runId = randomUUID();
|
||||
await db.insert(issues).values({
|
||||
id: issueId,
|
||||
companyId,
|
||||
identifier: `NRR-${label}`,
|
||||
title: `Recover ${label}`,
|
||||
status: "in_progress",
|
||||
priority: "medium",
|
||||
workMode: "standard",
|
||||
assigneeAgentId: agentId,
|
||||
});
|
||||
const [run] = await db
|
||||
.insert(heartbeatRuns)
|
||||
.values({
|
||||
id: runId,
|
||||
companyId,
|
||||
agentId,
|
||||
status: "running",
|
||||
runtimeMode: "native",
|
||||
nativeIssueId: issueId,
|
||||
invocationSource: "assignment",
|
||||
triggerDetail: "system",
|
||||
contextSnapshot: { issueId },
|
||||
})
|
||||
.returning();
|
||||
if (!run) throw new Error("Failed to seed native restart run");
|
||||
await db
|
||||
.update(issues)
|
||||
.set({ executionRunId: runId })
|
||||
.where(eq(issues.id, issueId));
|
||||
const native = await prepareNativeHeartbeatRun({
|
||||
db,
|
||||
run,
|
||||
issue: {
|
||||
id: issueId,
|
||||
title: `Recover ${label}`,
|
||||
description: null,
|
||||
reviewPolicy: null,
|
||||
},
|
||||
environmentLeaseId: `lease-${label}`,
|
||||
});
|
||||
await db.insert(nativeRunFinalizations).values({
|
||||
runId,
|
||||
companyId,
|
||||
issueId,
|
||||
phase: "observed",
|
||||
});
|
||||
return { db, issueId, runId, native };
|
||||
}
|
||||
|
||||
function transportOptions(
|
||||
fixture: Awaited<ReturnType<typeof seedRun>>,
|
||||
stateDirectory: string,
|
||||
) {
|
||||
return {
|
||||
runnerBinary: defaultCapabilityRunnerdBinary(),
|
||||
codexCommand: fakeCodexAppServer,
|
||||
codexArgs: [
|
||||
"--state-file",
|
||||
resolve(stateDirectory, "fake-codex-state.json"),
|
||||
"--call-log",
|
||||
resolve(stateDirectory, "fake-codex-calls.log"),
|
||||
],
|
||||
stateDirectory,
|
||||
lifecyclePolicy: { mode: "warm" as const, idleTimeoutMs: 60_000 },
|
||||
prpIdentity: {
|
||||
runnerInstanceId: fixture.native.runnerInstanceId,
|
||||
environmentLeaseId: fixture.native.environmentLeaseId,
|
||||
runId: fixture.runId,
|
||||
normalizedSessionId: fixture.native.normalizedSessionId,
|
||||
turnId: fixture.native.turnId,
|
||||
itemId: fixture.native.itemId,
|
||||
},
|
||||
controlPlaneRegistration: (authority: Parameters<typeof registerRunnerPrpAuthority>[0]["authority"]) =>
|
||||
registerRunnerPrpAuthority({
|
||||
companyId,
|
||||
runId: fixture.runId,
|
||||
authority,
|
||||
}),
|
||||
};
|
||||
}
|
||||
|
||||
async function persistRestartEvidence(input: {
|
||||
fixture: Awaited<ReturnType<typeof seedRun>>;
|
||||
runnerPid: number;
|
||||
processGroupId: number | null;
|
||||
providerPid: number | null;
|
||||
providerSessionId: string;
|
||||
}) {
|
||||
const startedAt = await readProcessStartedAt(input.runnerPid);
|
||||
if (!startedAt) throw new Error("Runner process start fingerprint unavailable");
|
||||
const providerStartedAt = input.providerPid
|
||||
? await readProcessStartedAt(input.providerPid)
|
||||
: null;
|
||||
const now = new Date();
|
||||
await input.fixture.db
|
||||
.update(heartbeatRuns)
|
||||
.set({
|
||||
processPid: input.runnerPid,
|
||||
processGroupId: input.processGroupId,
|
||||
processStartedAt: new Date(startedAt),
|
||||
runnerProfileJson: {
|
||||
sessionCheckpoint: {
|
||||
sessionId: input.fixture.native.normalizedSessionId,
|
||||
identity: {
|
||||
companyId,
|
||||
issueId: input.fixture.issueId,
|
||||
runId: input.fixture.runId,
|
||||
agentId,
|
||||
sessionId: input.fixture.native.normalizedSessionId,
|
||||
},
|
||||
providerSessionId: input.providerSessionId,
|
||||
providerIdentity: {
|
||||
kind: "codex",
|
||||
providerSessionId: input.providerSessionId,
|
||||
},
|
||||
process: {
|
||||
runnerPid: input.runnerPid,
|
||||
runnerProcessGroupId: input.processGroupId,
|
||||
providerPid: input.providerPid,
|
||||
providerProcessStartedAt: providerStartedAt,
|
||||
codexPid: input.providerPid,
|
||||
codexProcessStartedAt: providerStartedAt,
|
||||
},
|
||||
},
|
||||
},
|
||||
updatedAt: now,
|
||||
})
|
||||
.where(eq(heartbeatRuns.id, input.fixture.runId));
|
||||
await input.fixture.db
|
||||
.update(nativeRunFinalizations)
|
||||
.set({
|
||||
leaseOwner: "dead-controller",
|
||||
leaseExpiresAt: new Date(now.getTime() + 60_000),
|
||||
controllerBootId: "dead-controller-boot",
|
||||
controllerPid: 2_000_000_000,
|
||||
controllerProcessStartedAt: new Date("2026-09-04T11:00:00.000Z"),
|
||||
controllerGeneration: 1,
|
||||
updatedAt: now,
|
||||
})
|
||||
.where(eq(nativeRunFinalizations.runId, input.fixture.runId));
|
||||
return new Date(startedAt);
|
||||
}
|
||||
|
||||
async function persistUnconnectedRunner(input: {
|
||||
fixture: Awaited<ReturnType<typeof seedRun>>;
|
||||
runnerPid: number;
|
||||
processGroupId: number | null;
|
||||
}) {
|
||||
const startedAt = await readProcessStartedAt(input.runnerPid);
|
||||
if (!startedAt) throw new Error("Runner process start fingerprint unavailable");
|
||||
const now = new Date();
|
||||
await input.fixture.db
|
||||
.update(heartbeatRuns)
|
||||
.set({
|
||||
processPid: input.runnerPid,
|
||||
processGroupId: input.processGroupId,
|
||||
processStartedAt: new Date(startedAt),
|
||||
updatedAt: now,
|
||||
})
|
||||
.where(eq(heartbeatRuns.id, input.fixture.runId));
|
||||
await input.fixture.db
|
||||
.update(nativeRunFinalizations)
|
||||
.set({
|
||||
leaseOwner: "dead-controller",
|
||||
leaseExpiresAt: new Date(now.getTime() + 60_000),
|
||||
controllerBootId: "dead-controller-boot",
|
||||
controllerPid: 2_000_000_003,
|
||||
controllerProcessStartedAt: new Date("2026-09-04T11:00:00.000Z"),
|
||||
controllerGeneration: 1,
|
||||
updatedAt: now,
|
||||
})
|
||||
.where(eq(nativeRunFinalizations.runId, input.fixture.runId));
|
||||
}
|
||||
|
||||
realProcessIt("adopts one active runner across hot and hard controller restarts without duplicating steering", async () => {
|
||||
const fixture = await seedRun("LIVE");
|
||||
const stateDirectory = resolve(runtimeRoot, fixture.runId);
|
||||
const baseOptions = transportOptions(fixture, stateDirectory);
|
||||
const options = {
|
||||
...baseOptions,
|
||||
codexArgs: [...baseOptions.codexArgs, "--linger-after-turn-start"],
|
||||
};
|
||||
const first = createRunnerdCodexTransport(options);
|
||||
let runnerPid: number | null = null;
|
||||
let adopted: ReturnType<typeof createRunnerdCodexTransport> | null = null;
|
||||
let adoptedAfterHardRestart:
|
||||
| ReturnType<typeof createRunnerdCodexTransport>
|
||||
| null = null;
|
||||
try {
|
||||
const started = await first.transport.request("thread/start", {
|
||||
cwd: tmpdir(),
|
||||
dynamicTools: [],
|
||||
});
|
||||
const thread = started.thread as Record<string, unknown>;
|
||||
const turnStarted = await first.transport.request("turn/start", {
|
||||
input: [{ type: "text", text: "Hold this turn across restarts." }],
|
||||
});
|
||||
const providerTurnId = String(
|
||||
(turnStarted.turn as Record<string, unknown>).id,
|
||||
);
|
||||
await first.transport.request("turn/steer", {
|
||||
input: [{ type: "text", text: "before hot restart" }],
|
||||
expectedTurnId: providerTurnId,
|
||||
correlationId: "steer-before-hot",
|
||||
});
|
||||
runnerPid = first.evidence().runnerPid;
|
||||
if (!runnerPid) throw new Error("Real runner PID was not observed");
|
||||
await persistRestartEvidence({
|
||||
fixture,
|
||||
runnerPid,
|
||||
processGroupId: first.evidence().runnerProcessGroupId,
|
||||
providerPid: first.evidence().providerPid,
|
||||
providerSessionId: String(thread.id),
|
||||
});
|
||||
|
||||
await first.detachControllerForRestart();
|
||||
const [claim] = await claimNativeRestartRecoveries({
|
||||
db: fixture.db,
|
||||
controller: successor,
|
||||
restartKind: "hot",
|
||||
recoveryRequestId: "hot-restart-request",
|
||||
now: new Date(),
|
||||
runIds: [fixture.runId],
|
||||
});
|
||||
expect(claim).toMatchObject({
|
||||
kind: "reattach_existing_runner",
|
||||
runId: fixture.runId,
|
||||
controllerGeneration: 2,
|
||||
providerAttempt: 0,
|
||||
process: { pid: runnerPid },
|
||||
});
|
||||
if (!claim || claim.kind !== "reattach_existing_runner") {
|
||||
throw new Error("Expected live-runner recovery claim");
|
||||
}
|
||||
|
||||
const duplicateLauncher = vi.fn(() => {
|
||||
throw new Error("duplicate runner spawn attempted");
|
||||
});
|
||||
adopted = createRunnerdCodexTransport({
|
||||
...options,
|
||||
resumeDynamicTools: [],
|
||||
runnerProcessLauncher: duplicateLauncher,
|
||||
adoptExistingRunner: {
|
||||
...claim.process,
|
||||
isAlive: () => processAlive(runnerPid),
|
||||
},
|
||||
});
|
||||
const restored = await adopted.transport.request("thread/read", {});
|
||||
expect(restored.thread).toMatchObject({
|
||||
id: thread.id,
|
||||
turns: [{ id: providerTurnId, status: "inProgress" }],
|
||||
});
|
||||
expect(adopted.evidence().runnerPid).toBe(runnerPid);
|
||||
expect(duplicateLauncher).not.toHaveBeenCalled();
|
||||
await adopted.transport.request("turn/steer", {
|
||||
input: [{ type: "text", text: "after hot restart" }],
|
||||
expectedTurnId: providerTurnId,
|
||||
correlationId: "steer-after-hot",
|
||||
});
|
||||
|
||||
await adopted.detachControllerForRestart();
|
||||
const hardRestartController: NativeControllerIdentity = {
|
||||
...successor,
|
||||
bootId: randomUUID(),
|
||||
};
|
||||
await fixture.db
|
||||
.update(nativeRunFinalizations)
|
||||
.set({
|
||||
controllerPid: 2_000_000_001,
|
||||
controllerProcessStartedAt: new Date(
|
||||
"2026-09-04T11:30:00.000Z",
|
||||
),
|
||||
})
|
||||
.where(eq(nativeRunFinalizations.runId, fixture.runId));
|
||||
const [hardClaim] = await claimNativeRestartRecoveries({
|
||||
db: fixture.db,
|
||||
controller: hardRestartController,
|
||||
restartKind: "hard",
|
||||
now: new Date(),
|
||||
runIds: [fixture.runId],
|
||||
});
|
||||
expect(hardClaim).toMatchObject({
|
||||
kind: "reattach_existing_runner",
|
||||
runId: fixture.runId,
|
||||
controllerGeneration: 3,
|
||||
providerAttempt: 0,
|
||||
process: { pid: runnerPid },
|
||||
});
|
||||
if (!hardClaim || hardClaim.kind !== "reattach_existing_runner") {
|
||||
throw new Error("Expected second live-runner recovery claim");
|
||||
}
|
||||
const secondDuplicateLauncher = vi.fn(() => {
|
||||
throw new Error("duplicate runner spawn attempted after hard restart");
|
||||
});
|
||||
adoptedAfterHardRestart = createRunnerdCodexTransport({
|
||||
...options,
|
||||
resumeDynamicTools: [],
|
||||
runnerProcessLauncher: secondDuplicateLauncher,
|
||||
adoptExistingRunner: {
|
||||
...hardClaim.process,
|
||||
isAlive: () => processAlive(runnerPid),
|
||||
},
|
||||
});
|
||||
await expect(
|
||||
adoptedAfterHardRestart.transport.request("thread/read", {}),
|
||||
).resolves.toMatchObject({
|
||||
thread: {
|
||||
id: thread.id,
|
||||
turns: [{ id: providerTurnId, status: "inProgress" }],
|
||||
},
|
||||
});
|
||||
await adoptedAfterHardRestart.transport.request("turn/steer", {
|
||||
input: [{ type: "text", text: "after hard restart" }],
|
||||
expectedTurnId: providerTurnId,
|
||||
correlationId: "steer-after-hard",
|
||||
});
|
||||
expect(adoptedAfterHardRestart.evidence().runnerPid).toBe(runnerPid);
|
||||
expect(secondDuplicateLauncher).not.toHaveBeenCalled();
|
||||
const providerCalls = await readFile(
|
||||
resolve(stateDirectory, "fake-codex-calls.log"),
|
||||
"utf8",
|
||||
);
|
||||
expect(providerCalls.match(/^turn\/start$/gm)).toHaveLength(1);
|
||||
expect(providerCalls.match(/^turn\/steer$/gm)).toHaveLength(3);
|
||||
expect(
|
||||
await fixture.db
|
||||
.select({ id: heartbeatRuns.id })
|
||||
.from(heartbeatRuns)
|
||||
.where(
|
||||
and(
|
||||
eq(heartbeatRuns.retryOfRunId, fixture.runId),
|
||||
eq(heartbeatRuns.companyId, companyId),
|
||||
),
|
||||
),
|
||||
).toHaveLength(0);
|
||||
} finally {
|
||||
await adoptedAfterHardRestart?.transport.close().catch(() => undefined);
|
||||
await adopted?.transport.close().catch(() => undefined);
|
||||
await stopOwnedProcessGroup(runnerPid, stateDirectory).catch(() => undefined);
|
||||
await rm(stateDirectory, { recursive: true, force: true });
|
||||
}
|
||||
}, 45_000);
|
||||
|
||||
realProcessIt("hard-restarts a dead runner on the same run and provider session", async () => {
|
||||
const fixture = await seedRun("DEAD");
|
||||
const stateDirectory = resolve(runtimeRoot, fixture.runId);
|
||||
const baseOptions = transportOptions(fixture, stateDirectory);
|
||||
const options = {
|
||||
...baseOptions,
|
||||
codexArgs: [...baseOptions.codexArgs, "--linger-after-turn-start"],
|
||||
};
|
||||
const first = createRunnerdCodexTransport(options);
|
||||
let firstRunnerPid: number | null = null;
|
||||
let recoveredRunnerPid: number | null = null;
|
||||
let restored: ReturnType<typeof createRunnerdCodexTransport> | null = null;
|
||||
try {
|
||||
const started = await first.transport.request("thread/start", {
|
||||
cwd: tmpdir(),
|
||||
dynamicTools: [],
|
||||
});
|
||||
const thread = started.thread as Record<string, unknown>;
|
||||
const turnStarted = await first.transport.request("turn/start", {
|
||||
input: [{ type: "text", text: "Resume this active turn after process loss." }],
|
||||
});
|
||||
const providerTurnId = String(
|
||||
(turnStarted.turn as Record<string, unknown>).id,
|
||||
);
|
||||
firstRunnerPid = first.evidence().runnerPid;
|
||||
if (!firstRunnerPid) throw new Error("Real runner PID was not observed");
|
||||
const providerPid = first.evidence().providerPid;
|
||||
await persistRestartEvidence({
|
||||
fixture,
|
||||
runnerPid: firstRunnerPid,
|
||||
processGroupId: first.evidence().runnerProcessGroupId,
|
||||
providerPid,
|
||||
providerSessionId: String(thread.id),
|
||||
});
|
||||
|
||||
await first.detachControllerForRestart();
|
||||
await stopOwnedProcessGroup(firstRunnerPid, stateDirectory);
|
||||
await waitForCondition(
|
||||
"provider process to observe runner pipe closure",
|
||||
() => !processAlive(providerPid),
|
||||
);
|
||||
|
||||
const [claim] = await claimNativeRestartRecoveries({
|
||||
db: fixture.db,
|
||||
controller: successor,
|
||||
restartKind: "hard",
|
||||
now: new Date(),
|
||||
runIds: [fixture.runId],
|
||||
});
|
||||
expect(claim).toMatchObject({
|
||||
kind: "resume_dead_runner",
|
||||
runId: fixture.runId,
|
||||
controllerGeneration: 2,
|
||||
providerAttempt: 0,
|
||||
});
|
||||
if (!claim || claim.kind !== "resume_dead_runner") {
|
||||
throw new Error("Expected dead-runner recovery claim");
|
||||
}
|
||||
|
||||
restored = createRunnerdCodexTransport({
|
||||
...options,
|
||||
resumeDynamicTools: [],
|
||||
});
|
||||
const recovered = await restored.transport.request("thread/read", {});
|
||||
recoveredRunnerPid = restored.evidence().runnerPid;
|
||||
expect(recovered.thread).toMatchObject({
|
||||
id: thread.id,
|
||||
turns: [{ id: providerTurnId, status: "inProgress" }],
|
||||
});
|
||||
expect(recoveredRunnerPid).toEqual(expect.any(Number));
|
||||
expect(recoveredRunnerPid).not.toBe(firstRunnerPid);
|
||||
const providerCalls = await readFile(
|
||||
resolve(stateDirectory, "fake-codex-calls.log"),
|
||||
"utf8",
|
||||
);
|
||||
expect(providerCalls.match(/^turn\/start$/gm)).toHaveLength(1);
|
||||
expect(
|
||||
await fixture.db
|
||||
.select({ id: heartbeatRuns.id })
|
||||
.from(heartbeatRuns)
|
||||
.where(eq(heartbeatRuns.retryOfRunId, fixture.runId)),
|
||||
).toHaveLength(0);
|
||||
} finally {
|
||||
await restored?.transport.close().catch(() => undefined);
|
||||
await stopOwnedProcessGroup(recoveredRunnerPid, stateDirectory).catch(
|
||||
() => undefined,
|
||||
);
|
||||
await stopOwnedProcessGroup(firstRunnerPid, stateDirectory).catch(
|
||||
() => undefined,
|
||||
);
|
||||
await rm(stateDirectory, { recursive: true, force: true });
|
||||
}
|
||||
}, 45_000);
|
||||
|
||||
realProcessIt("quarantines a pre-auth runner root and bootstraps the same run", async () => {
|
||||
const fixture = await seedRun("BOOTSTRAP");
|
||||
const stateDirectory = resolve(runtimeRoot, fixture.runId);
|
||||
const baseOptions = transportOptions(fixture, stateDirectory);
|
||||
const transport = createRunnerdCodexTransport({
|
||||
...baseOptions,
|
||||
runnerReconnectGraceMs: 60_000,
|
||||
controlPlaneRegistration: async () => ({
|
||||
connectUrl: "ws://127.0.0.1:9/api/runner/prp/unreachable",
|
||||
release: () => undefined,
|
||||
}),
|
||||
});
|
||||
let runnerPid: number | null = null;
|
||||
let replacementRunnerPid: number | null = null;
|
||||
let replacement: ReturnType<typeof createRunnerdCodexTransport> | null =
|
||||
null;
|
||||
const quarantineDirectory = `${stateDirectory}.bootstrap-incomplete`;
|
||||
const starting = transport.transport
|
||||
.request("thread/start", {
|
||||
cwd: tmpdir(),
|
||||
dynamicTools: [],
|
||||
})
|
||||
.then(
|
||||
() => null,
|
||||
(error: unknown) => error,
|
||||
);
|
||||
try {
|
||||
await waitForCondition("pre-auth runner spawn", () => {
|
||||
runnerPid = transport.evidence().runnerPid;
|
||||
return processAlive(runnerPid);
|
||||
});
|
||||
if (!runnerPid) throw new Error("Pre-auth runner PID was not observed");
|
||||
await persistUnconnectedRunner({
|
||||
fixture,
|
||||
runnerPid,
|
||||
processGroupId: transport.evidence().runnerProcessGroupId,
|
||||
});
|
||||
await stopOwnedProcessGroup(runnerPid, stateDirectory);
|
||||
await expect(starting).resolves.toBeInstanceOf(Error);
|
||||
|
||||
const durable = JSON.parse(
|
||||
await readFile(
|
||||
resolve(stateDirectory, "control-plane", "control-plane-state.json"),
|
||||
"utf8",
|
||||
),
|
||||
) as Record<string, unknown>;
|
||||
expect(durable).toMatchObject({
|
||||
connectionCount: 0,
|
||||
committedEvents: [],
|
||||
});
|
||||
const [claim] = await claimNativeRestartRecoveries({
|
||||
db: fixture.db,
|
||||
controller: successor,
|
||||
restartKind: "hard",
|
||||
now: new Date(),
|
||||
runIds: [fixture.runId],
|
||||
});
|
||||
expect(claim).toMatchObject({
|
||||
kind: "bootstrap_incomplete",
|
||||
runId: fixture.runId,
|
||||
controllerGeneration: 2,
|
||||
providerAttempt: 0,
|
||||
});
|
||||
if (!claim || claim.kind !== "bootstrap_incomplete") {
|
||||
throw new Error("Expected incomplete-bootstrap recovery claim");
|
||||
}
|
||||
|
||||
await transport.transport.close();
|
||||
await rename(stateDirectory, quarantineDirectory);
|
||||
replacement = createRunnerdCodexTransport(baseOptions);
|
||||
const restarted = await replacement.transport.request("thread/start", {
|
||||
cwd: tmpdir(),
|
||||
dynamicTools: [],
|
||||
});
|
||||
replacementRunnerPid = replacement.evidence().runnerPid;
|
||||
expect(restarted.thread).toMatchObject({ id: expect.any(String) });
|
||||
expect(replacementRunnerPid).toEqual(expect.any(Number));
|
||||
expect(replacementRunnerPid).not.toBe(runnerPid);
|
||||
expect(processAlive(replacementRunnerPid)).toBe(true);
|
||||
expect(existsSync(quarantineDirectory)).toBe(true);
|
||||
expect(existsSync(stateDirectory)).toBe(true);
|
||||
const replacementDurable = JSON.parse(
|
||||
await readFile(
|
||||
resolve(stateDirectory, "control-plane", "control-plane-state.json"),
|
||||
"utf8",
|
||||
),
|
||||
) as Record<string, unknown>;
|
||||
expect(replacementDurable).toMatchObject({
|
||||
identity: {
|
||||
runId: fixture.runId,
|
||||
normalizedSessionId: fixture.native.normalizedSessionId,
|
||||
runnerInstanceId: fixture.native.runnerInstanceId,
|
||||
environmentLeaseId: fixture.native.environmentLeaseId,
|
||||
},
|
||||
});
|
||||
await expect(
|
||||
fixture.db
|
||||
.select({ id: heartbeatRuns.id })
|
||||
.from(heartbeatRuns)
|
||||
.where(eq(heartbeatRuns.retryOfRunId, fixture.runId)),
|
||||
).resolves.toHaveLength(0);
|
||||
} finally {
|
||||
await replacement?.transport.close().catch(() => undefined);
|
||||
await transport.transport.close().catch(() => undefined);
|
||||
await starting;
|
||||
await stopOwnedProcessGroup(replacementRunnerPid, stateDirectory).catch(
|
||||
() => undefined,
|
||||
);
|
||||
await stopOwnedProcessGroup(runnerPid, stateDirectory).catch(
|
||||
() => undefined,
|
||||
);
|
||||
await rm(stateDirectory, { recursive: true, force: true });
|
||||
await rm(quarantineDirectory, { recursive: true, force: true });
|
||||
}
|
||||
}, 45_000);
|
||||
|
||||
it("prioritizes an already-proposed durable result over runner recovery", async () => {
|
||||
const fixture = await seedRun("RESULT");
|
||||
const [persistedRun] = await fixture.db
|
||||
.select({ completionContractId: heartbeatRuns.completionContractId })
|
||||
.from(heartbeatRuns)
|
||||
.where(eq(heartbeatRuns.id, fixture.runId));
|
||||
if (!persistedRun?.completionContractId) {
|
||||
throw new Error("Prepared native run has no completion contract");
|
||||
}
|
||||
const resultId = randomUUID();
|
||||
await fixture.db.insert(nativeRunResults).values({
|
||||
id: resultId,
|
||||
companyId,
|
||||
issueId: fixture.issueId,
|
||||
runId: fixture.runId,
|
||||
turnId: fixture.native.turnId,
|
||||
completionContractId: persistedRun.completionContractId,
|
||||
callerResultId: "result-before-restart",
|
||||
callerDedupeKey: "result-before-restart",
|
||||
serverFingerprint: "result-before-restart",
|
||||
schemaStatus: "accepted",
|
||||
resultJson: {
|
||||
schema: "paperclip.run_result.v1",
|
||||
reportedWorkDisposition: "done",
|
||||
summary: "Durable result proposed before restart.",
|
||||
},
|
||||
canonicalSha256: "a".repeat(64),
|
||||
});
|
||||
await fixture.db
|
||||
.update(nativeRunFinalizations)
|
||||
.set({
|
||||
phase: "result_persisted",
|
||||
resultId,
|
||||
leaseOwner: null,
|
||||
leaseExpiresAt: null,
|
||||
})
|
||||
.where(eq(nativeRunFinalizations.runId, fixture.runId));
|
||||
|
||||
await expect(
|
||||
claimNativeRestartRecoveries({
|
||||
db: fixture.db,
|
||||
controller: successor,
|
||||
restartKind: "hot",
|
||||
recoveryRequestId: "result-race-restart",
|
||||
now: new Date(),
|
||||
runIds: [fixture.runId],
|
||||
}),
|
||||
).resolves.toEqual([]);
|
||||
await expect(
|
||||
fixture.db
|
||||
.select({ id: nativeRunResults.id })
|
||||
.from(nativeRunResults)
|
||||
.where(eq(nativeRunResults.runId, fixture.runId)),
|
||||
).resolves.toEqual([{ id: resultId }]);
|
||||
await expect(
|
||||
fixture.db
|
||||
.select({ id: heartbeatRuns.id })
|
||||
.from(heartbeatRuns)
|
||||
.where(eq(heartbeatRuns.retryOfRunId, fixture.runId)),
|
||||
).resolves.toHaveLength(0);
|
||||
});
|
||||
|
||||
it("preserves a future same-run provider recovery wake instead of consuming it during startup", async () => {
|
||||
const fixture = await seedRun("SCHEDULED");
|
||||
const now = new Date();
|
||||
const nextAttemptAt = new Date(now.getTime() + 60_000);
|
||||
await fixture.db
|
||||
.update(nativeRunFinalizations)
|
||||
.set({
|
||||
phase: "retryable_failure",
|
||||
attempt: 1,
|
||||
nextAttemptAt,
|
||||
recoveryState: "resuming_session",
|
||||
leaseOwner: null,
|
||||
leaseExpiresAt: null,
|
||||
})
|
||||
.where(eq(nativeRunFinalizations.runId, fixture.runId));
|
||||
|
||||
await expect(
|
||||
claimNativeRestartRecoveries({
|
||||
db: fixture.db,
|
||||
controller: successor,
|
||||
restartKind: "hard",
|
||||
now,
|
||||
runIds: [fixture.runId],
|
||||
}),
|
||||
).resolves.toEqual([]);
|
||||
await expect(
|
||||
fixture.db
|
||||
.select({
|
||||
phase: nativeRunFinalizations.phase,
|
||||
attempt: nativeRunFinalizations.attempt,
|
||||
nextAttemptAt: nativeRunFinalizations.nextAttemptAt,
|
||||
recoveryState: nativeRunFinalizations.recoveryState,
|
||||
})
|
||||
.from(nativeRunFinalizations)
|
||||
.where(eq(nativeRunFinalizations.runId, fixture.runId)),
|
||||
).resolves.toEqual([
|
||||
{
|
||||
phase: "retryable_failure",
|
||||
attempt: 1,
|
||||
nextAttemptAt,
|
||||
recoveryState: "resuming_session",
|
||||
},
|
||||
]);
|
||||
});
|
||||
|
||||
it("fails closed for a live recycled PID without signalling or spawning", async () => {
|
||||
const fixture = await seedRun("MISMATCH");
|
||||
const stateDirectory = resolve(runtimeRoot, fixture.runId);
|
||||
const child = spawn(
|
||||
process.execPath,
|
||||
["-e", "setInterval(() => {}, 1000)", stateDirectory],
|
||||
{
|
||||
detached: process.platform !== "win32",
|
||||
stdio: "ignore",
|
||||
},
|
||||
);
|
||||
child.unref();
|
||||
const pid = child.pid ?? null;
|
||||
if (!pid) throw new Error("Failed to launch mismatch sentinel");
|
||||
try {
|
||||
await waitForCondition("mismatch sentinel startup", () => processAlive(pid));
|
||||
const observedStart = await readProcessStartedAt(pid);
|
||||
if (!observedStart) throw new Error("Sentinel fingerprint unavailable");
|
||||
await fixture.db
|
||||
.update(heartbeatRuns)
|
||||
.set({
|
||||
processPid: pid,
|
||||
processGroupId: process.platform === "win32" ? null : pid,
|
||||
processStartedAt: new Date(
|
||||
new Date(observedStart).getTime() - 60_000,
|
||||
),
|
||||
})
|
||||
.where(eq(heartbeatRuns.id, fixture.runId));
|
||||
|
||||
const [disposition] = await claimNativeRestartRecoveries({
|
||||
db: fixture.db,
|
||||
controller: successor,
|
||||
restartKind: "hard",
|
||||
now: new Date(),
|
||||
runIds: [fixture.runId],
|
||||
});
|
||||
expect(disposition).toEqual({
|
||||
kind: "blocked",
|
||||
runId: fixture.runId,
|
||||
reason: "live_runner_identity_mismatch",
|
||||
});
|
||||
expect(processAlive(pid)).toBe(true);
|
||||
expect(existsSync(stateDirectory)).toBe(false);
|
||||
await expect(
|
||||
fixture.db
|
||||
.select({ recoveryState: nativeRunFinalizations.recoveryState })
|
||||
.from(nativeRunFinalizations)
|
||||
.where(eq(nativeRunFinalizations.runId, fixture.runId)),
|
||||
).resolves.toEqual([{ recoveryState: "blocked" }]);
|
||||
await expect(
|
||||
fixture.db
|
||||
.select({ id: heartbeatRuns.id })
|
||||
.from(heartbeatRuns)
|
||||
.where(eq(heartbeatRuns.retryOfRunId, fixture.runId)),
|
||||
).resolves.toHaveLength(0);
|
||||
} finally {
|
||||
await stopOwnedProcessGroup(pid, stateDirectory).catch(() => undefined);
|
||||
}
|
||||
});
|
||||
|
||||
it("allows only one of two concurrent successor controllers to claim a dead session", async () => {
|
||||
const fixture = await seedRun("CONCURRENT");
|
||||
const now = new Date();
|
||||
await fixture.db
|
||||
.update(heartbeatRuns)
|
||||
.set({
|
||||
runnerProfileJson: {
|
||||
sessionCheckpoint: {
|
||||
sessionId: fixture.native.normalizedSessionId,
|
||||
identity: {
|
||||
companyId,
|
||||
issueId: fixture.issueId,
|
||||
runId: fixture.runId,
|
||||
agentId,
|
||||
sessionId: fixture.native.normalizedSessionId,
|
||||
},
|
||||
providerSessionId: "provider-concurrent",
|
||||
providerIdentity: {
|
||||
kind: "codex",
|
||||
providerSessionId: "provider-concurrent",
|
||||
},
|
||||
},
|
||||
},
|
||||
})
|
||||
.where(eq(heartbeatRuns.id, fixture.runId));
|
||||
await fixture.db
|
||||
.update(nativeRunFinalizations)
|
||||
.set({
|
||||
leaseOwner: "dead-controller",
|
||||
leaseExpiresAt: new Date(now.getTime() + 60_000),
|
||||
controllerBootId: "dead-controller-boot",
|
||||
controllerPid: 2_000_000_002,
|
||||
controllerProcessStartedAt: new Date("2026-09-04T10:00:00.000Z"),
|
||||
controllerGeneration: 4,
|
||||
})
|
||||
.where(eq(nativeRunFinalizations.runId, fixture.runId));
|
||||
|
||||
const controllers = [
|
||||
successor,
|
||||
{ ...successor, bootId: randomUUID() },
|
||||
];
|
||||
const dispositions = (
|
||||
await Promise.all(
|
||||
controllers.map((controller) =>
|
||||
claimNativeRestartRecoveries({
|
||||
db: fixture.db,
|
||||
controller,
|
||||
restartKind: "hard",
|
||||
now,
|
||||
runIds: [fixture.runId],
|
||||
}),
|
||||
),
|
||||
)
|
||||
).flat();
|
||||
expect(
|
||||
dispositions.filter((entry) => entry.kind === "resume_dead_runner"),
|
||||
).toHaveLength(1);
|
||||
expect(
|
||||
dispositions.filter((entry) => entry.kind === "awaiting_evidence"),
|
||||
).toHaveLength(1);
|
||||
await expect(
|
||||
fixture.db
|
||||
.select({
|
||||
controllerGeneration: nativeRunFinalizations.controllerGeneration,
|
||||
providerAttempt: nativeRunFinalizations.attempt,
|
||||
})
|
||||
.from(nativeRunFinalizations)
|
||||
.where(eq(nativeRunFinalizations.runId, fixture.runId)),
|
||||
).resolves.toEqual([
|
||||
{ controllerGeneration: 5, providerAttempt: 0 },
|
||||
]);
|
||||
});
|
||||
|
||||
it("classifies every requested recovery candidate without an implicit 100-run cap", async () => {
|
||||
const db = createDb(temporary.connectionString);
|
||||
const fixtures: Array<Awaited<ReturnType<typeof seedRun>>> = [];
|
||||
for (let index = 0; index < 101; index += 1) {
|
||||
fixtures.push(
|
||||
await seedRun(`BULK-${String(index).padStart(3, "0")}`, db),
|
||||
);
|
||||
}
|
||||
|
||||
const dispositions = await claimNativeRestartRecoveries({
|
||||
db,
|
||||
controller: { ...successor, bootId: randomUUID() },
|
||||
restartKind: "hard",
|
||||
now: new Date(),
|
||||
runIds: fixtures.map((fixture) => fixture.runId),
|
||||
});
|
||||
|
||||
expect(dispositions).toHaveLength(101);
|
||||
expect(
|
||||
dispositions.every(
|
||||
(disposition) => disposition.kind === "bootstrap_incomplete",
|
||||
),
|
||||
).toBe(true);
|
||||
}, 30_000);
|
||||
});
|
||||
|
|
@ -166,6 +166,7 @@ import {
|
|||
runtimeQuestionFallbackFromEvent,
|
||||
resolveNativeRuntimeRequest,
|
||||
resolveNativeHarnessPersistenceProfile,
|
||||
runnerdStateProvesIncompleteBootstrap,
|
||||
semanticProviderPlanMarkdown,
|
||||
sha256DirectoryTree,
|
||||
stageRemoteRunnerDirectory,
|
||||
|
|
@ -175,6 +176,46 @@ import {
|
|||
shouldRestoreNativeHarnessBackupIntoSandbox,
|
||||
} from "./native-session-executor.js";
|
||||
|
||||
describe("native incomplete-bootstrap evidence", () => {
|
||||
it("requires zero connections, zero events, and only untouched bootstrap commands", async () => {
|
||||
const root = await mkdtemp(join(tmpdir(), "native-bootstrap-evidence-"));
|
||||
const controlPlaneRoot = join(root, "control-plane");
|
||||
await mkdir(controlPlaneRoot, { recursive: true });
|
||||
const statePath = join(controlPlaneRoot, "control-plane-state.json");
|
||||
const base = {
|
||||
schema: "paperclip.runner.durable.control-plane-state.v1",
|
||||
connectionCount: 0,
|
||||
committedEvents: [],
|
||||
commands: [
|
||||
{ type: "run.prepare", status: "pending" },
|
||||
{ type: "session.open", status: "pending" },
|
||||
],
|
||||
};
|
||||
try {
|
||||
await writeFile(statePath, JSON.stringify(base));
|
||||
expect(runnerdStateProvesIncompleteBootstrap(root)).toBe(true);
|
||||
|
||||
for (const ambiguous of [
|
||||
{ ...base, connectionCount: 1 },
|
||||
{ ...base, committedEvents: [{ eventType: "harness.ready" }] },
|
||||
{
|
||||
...base,
|
||||
commands: [{ type: "session.open", status: "completed" }],
|
||||
},
|
||||
{
|
||||
...base,
|
||||
commands: [{ type: "turn.start", status: "pending" }],
|
||||
},
|
||||
]) {
|
||||
await writeFile(statePath, JSON.stringify(ambiguous));
|
||||
expect(runnerdStateProvesIncompleteBootstrap(root)).toBe(false);
|
||||
}
|
||||
} finally {
|
||||
await rm(root, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe("native provider usage normalization", () => {
|
||||
it("reads remote runner run-delta tokens and provider cost", () => {
|
||||
const usage = {
|
||||
|
|
|
|||
|
|
@ -83,6 +83,7 @@ import {
|
|||
type NativeStatusDecision,
|
||||
} from "./status-arbiter.js";
|
||||
import { HttpError } from "../../errors.js";
|
||||
import { redactSensitiveText } from "../../redaction.js";
|
||||
import { resolvePaperclipRunnerBinary } from "./native-codex-runner.js";
|
||||
import {
|
||||
createNativeRunTrace,
|
||||
|
|
@ -91,6 +92,13 @@ import {
|
|||
type NativeRunTrace,
|
||||
} from "./native-run-trace.js";
|
||||
import { createNativeHarnessBackupStamp } from "./native-harness-backup-stamp.js";
|
||||
import { readProcessStartedAt } from "../hot-restart.js";
|
||||
import {
|
||||
currentNativeControllerIdentity,
|
||||
nextNativeProviderAttempt,
|
||||
type NativeControllerIdentity,
|
||||
type NativeRestartRecoveryClaim,
|
||||
} from "./native-restart-recovery.js";
|
||||
|
||||
type ActiveNativeSession = {
|
||||
session: NativeSession;
|
||||
|
|
@ -154,6 +162,39 @@ const nativeRuntimeRequestResolutions = new Map<
|
|||
NativeRuntimeRequestResolution
|
||||
>();
|
||||
|
||||
async function verifiedRecoveryProcessIsAlive(input: {
|
||||
pid: number;
|
||||
startedAt: string;
|
||||
}): Promise<boolean> {
|
||||
try {
|
||||
process.kill(input.pid, 0);
|
||||
} catch (error) {
|
||||
if ((error as NodeJS.ErrnoException | undefined)?.code !== "EPERM")
|
||||
return false;
|
||||
}
|
||||
try {
|
||||
const observed = await readProcessStartedAt(input.pid);
|
||||
return (
|
||||
observed !== null &&
|
||||
new Date(observed).getTime() === new Date(input.startedAt).getTime()
|
||||
);
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
async function signalVerifiedRecoveryProcess(
|
||||
input: { pid: number; startedAt: string },
|
||||
signal: NodeJS.Signals,
|
||||
): Promise<boolean> {
|
||||
if (!(await verifiedRecoveryProcessIsAlive(input))) return false;
|
||||
try {
|
||||
return process.kill(input.pid, signal);
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
function pruneNativeRuntimeRequestResolutionCache(): void {
|
||||
const completed = [...nativeRuntimeRequestResolutions.entries()]
|
||||
.filter(([, resolution]) => resolution.completedAt !== null)
|
||||
|
|
@ -1378,6 +1419,7 @@ async function verifyPriorRunnerdStateForSessionScope(input: {
|
|||
async function migrateRunnerdStateRootForExecution(input: {
|
||||
db: Db;
|
||||
execution: NativeExecutionInput;
|
||||
restartRecovery?: NativeRestartRecoveryClaim;
|
||||
}): Promise<void> {
|
||||
const scoped = scopedRunnerdStateRoot(input.execution);
|
||||
if (existsSync(scoped)) {
|
||||
|
|
@ -1386,16 +1428,32 @@ async function migrateRunnerdStateRootForExecution(input: {
|
|||
}
|
||||
const identity = readRunnerdDurableIdentity(scoped);
|
||||
if (!identity) {
|
||||
quarantineRunnerdStateRoot(scoped, "identity_indeterminate");
|
||||
if (input.restartRecovery?.kind !== "reattach_existing_runner") {
|
||||
quarantineRunnerdStateRoot(scoped, "identity_indeterminate");
|
||||
}
|
||||
throw new Error("runner_state_identity_mismatch");
|
||||
}
|
||||
if (!durableIdentityMatchesSession(identity, input.execution)) {
|
||||
quarantineRunnerdStateRoot(scoped, "identity_mismatch");
|
||||
if (input.restartRecovery?.kind !== "reattach_existing_runner") {
|
||||
quarantineRunnerdStateRoot(scoped, "identity_mismatch");
|
||||
}
|
||||
throw new Error("runner_state_identity_mismatch");
|
||||
}
|
||||
if (input.restartRecovery?.kind === "bootstrap_incomplete") {
|
||||
if (runnerdStateProvesIncompleteBootstrap(scoped)) {
|
||||
quarantineRunnerdStateRoot(scoped, "identity_indeterminate");
|
||||
return;
|
||||
}
|
||||
// Database evidence alone cannot distinguish a never-connected runner
|
||||
// from a partially-persisted provider bootstrap. Only the durable PRP
|
||||
// root can authorize a fresh bootstrap; anything else stays fail-closed.
|
||||
throw new Error("runner_state_identity_mismatch");
|
||||
}
|
||||
if (durableIdentityMatchesExecution(identity, input.execution)) {
|
||||
if (runnerdAuthorityLifecycle(scoped, identity) === "indeterminate") {
|
||||
quarantineRunnerdStateRoot(scoped, "identity_indeterminate");
|
||||
if (input.restartRecovery?.kind !== "reattach_existing_runner") {
|
||||
quarantineRunnerdStateRoot(scoped, "identity_indeterminate");
|
||||
}
|
||||
throw new Error("runner_state_identity_mismatch");
|
||||
}
|
||||
} else {
|
||||
|
|
@ -1464,6 +1522,40 @@ async function migrateRunnerdStateRootForExecution(input: {
|
|||
}
|
||||
}
|
||||
|
||||
export function runnerdStateProvesIncompleteBootstrap(root: string): boolean {
|
||||
try {
|
||||
const statePath = resolve(root, "control-plane", "control-plane-state.json");
|
||||
const state = record(
|
||||
JSON.parse(
|
||||
readBoundedNativeFile(
|
||||
statePath,
|
||||
NATIVE_DURABLE_IDENTITY_MAX_BYTES,
|
||||
"runner_durable_identity_too_large",
|
||||
).toString("utf8"),
|
||||
),
|
||||
);
|
||||
const commands = Array.isArray(state.commands)
|
||||
? state.commands.map(record)
|
||||
: [];
|
||||
const committedEvents = Array.isArray(state.committedEvents)
|
||||
? state.committedEvents
|
||||
: [];
|
||||
const onlyUnconsumedBootstrapCommands = commands.every(
|
||||
(command) =>
|
||||
command.status === "pending" &&
|
||||
(command.type === "run.prepare" || command.type === "session.open"),
|
||||
);
|
||||
return (
|
||||
state.schema === RUNNERD_CONTROL_PLANE_STATE_SCHEMA &&
|
||||
state.connectionCount === 0 &&
|
||||
committedEvents.length === 0 &&
|
||||
onlyUnconsumedBootstrapCommands
|
||||
);
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
function isSafeNativeStateDirectory(path: string): boolean {
|
||||
if (!existsSync(path)) return false;
|
||||
const stats = lstatSync(path);
|
||||
|
|
@ -3079,8 +3171,10 @@ export async function renewNativeSessionExecutionLease(input: {
|
|||
issueId: string;
|
||||
leaseOwner: string;
|
||||
attempt: number;
|
||||
controller?: NativeControllerIdentity;
|
||||
leaseTtlMs?: number;
|
||||
}): Promise<void> {
|
||||
const controller = input.controller ?? (await currentNativeControllerIdentity());
|
||||
const leaseTtlMs = input.leaseTtlMs ?? NATIVE_SESSION_EXECUTION_LEASE_TTL_MS;
|
||||
if (
|
||||
!Number.isInteger(leaseTtlMs) ||
|
||||
|
|
@ -3102,6 +3196,12 @@ export async function renewNativeSessionExecutionLease(input: {
|
|||
eq(nativeRunFinalizations.issueId, input.issueId),
|
||||
eq(nativeRunFinalizations.leaseOwner, input.leaseOwner),
|
||||
eq(nativeRunFinalizations.attempt, input.attempt),
|
||||
eq(nativeRunFinalizations.controllerBootId, controller.bootId),
|
||||
eq(nativeRunFinalizations.controllerPid, controller.pid),
|
||||
eq(
|
||||
nativeRunFinalizations.controllerProcessStartedAt,
|
||||
controller.processStartedAt,
|
||||
),
|
||||
gt(nativeRunFinalizations.leaseExpiresAt, sql`now()`),
|
||||
),
|
||||
)
|
||||
|
|
@ -3116,6 +3216,7 @@ function startNativeSessionExecutionLeaseRenewal(input: {
|
|||
issueId: string;
|
||||
leaseOwner: string;
|
||||
attempt: number;
|
||||
controller: NativeControllerIdentity;
|
||||
}): { stop: () => Promise<void> } {
|
||||
let leaseLost: Error | null = null;
|
||||
let renewal = Promise.resolve();
|
||||
|
|
@ -3154,6 +3255,7 @@ export async function executePaperclipNativeSession(input: {
|
|||
execution: NativeExecutionInput;
|
||||
runnerInstanceId: string;
|
||||
leaseOwner?: string;
|
||||
restartRecovery?: NativeRestartRecoveryClaim;
|
||||
onSpawn?: (meta: {
|
||||
pid: number;
|
||||
processGroupId: number | null;
|
||||
|
|
@ -3286,6 +3388,7 @@ async function executePaperclipNativeSessionWithinScope(
|
|||
await migrateRunnerdStateRootForExecution({
|
||||
db: input.db,
|
||||
execution: input.execution,
|
||||
restartRecovery: input.restartRecovery,
|
||||
});
|
||||
}
|
||||
const durableRunnerBinding = input.useRunnerd
|
||||
|
|
@ -3310,6 +3413,7 @@ async function executePaperclipNativeSessionWithinScope(
|
|||
}
|
||||
const leaseOwner =
|
||||
input.leaseOwner ?? `${effectiveRunnerInstanceId}:${randomUUID()}`;
|
||||
const controller = await currentNativeControllerIdentity();
|
||||
const leaseNow = new Date();
|
||||
const leaseExpiresAt = new Date(leaseNow.getTime() + 20 * 60_000);
|
||||
let attempt: number;
|
||||
|
|
@ -3400,13 +3504,39 @@ async function executePaperclipNativeSessionWithinScope(
|
|||
coordinator.leaseExpiresAt > leaseNow
|
||||
)
|
||||
throw new Error("native_finalization_lease_busy");
|
||||
const recovering = input.restartRecovery;
|
||||
if (
|
||||
recovering &&
|
||||
(recovering.runId !== coordinator.runId ||
|
||||
recovering.leaseOwner !== leaseOwner ||
|
||||
coordinator.leaseOwner !== leaseOwner ||
|
||||
coordinator.controllerBootId !== controller.bootId ||
|
||||
coordinator.controllerPid !== controller.pid ||
|
||||
coordinator.controllerGeneration !==
|
||||
recovering.controllerGeneration)
|
||||
) {
|
||||
throw new Error("native_restart_recovery_claim_changed");
|
||||
}
|
||||
const nextAttempt = nextNativeProviderAttempt(
|
||||
coordinator.attempt,
|
||||
recovering?.kind,
|
||||
);
|
||||
const nextControllerGeneration = recovering
|
||||
? recovering.controllerGeneration
|
||||
: coordinator.controllerBootId === controller.bootId
|
||||
? Math.max(1, coordinator.controllerGeneration)
|
||||
: coordinator.controllerGeneration + 1;
|
||||
const claimed = await tx
|
||||
.update(nativeRunFinalizations)
|
||||
.set({
|
||||
phase: "observed",
|
||||
attempt: coordinator.attempt + 1,
|
||||
attempt: nextAttempt,
|
||||
leaseOwner,
|
||||
leaseExpiresAt,
|
||||
controllerBootId: controller.bootId,
|
||||
controllerPid: controller.pid,
|
||||
controllerProcessStartedAt: controller.processStartedAt,
|
||||
controllerGeneration: nextControllerGeneration,
|
||||
failureCode: null,
|
||||
failureDetail: null,
|
||||
nextAttemptAt: null,
|
||||
|
|
@ -3438,7 +3568,7 @@ async function executePaperclipNativeSessionWithinScope(
|
|||
updatedAt: leaseNow,
|
||||
})
|
||||
.where(eq(heartbeatRuns.id, coordinator.runId));
|
||||
return coordinator.attempt + 1;
|
||||
return nextAttempt;
|
||||
}),
|
||||
{ parentName: "task.prepare" },
|
||||
);
|
||||
|
|
@ -3859,6 +3989,7 @@ async function executePaperclipNativeSessionWithinScope(
|
|||
issueId: input.execution.binding.issueId,
|
||||
leaseOwner,
|
||||
attempt,
|
||||
controller,
|
||||
});
|
||||
try {
|
||||
const runnerdBackend =
|
||||
|
|
@ -4085,6 +4216,7 @@ async function executePaperclipNativeSessionWithinScope(
|
|||
error instanceof Error
|
||||
? error.message.slice(0, 2_000)
|
||||
: String(error).slice(0, 2_000);
|
||||
const sanitizedStderrTail = redactSensitiveText(message).slice(-4_096);
|
||||
await input.db.transaction(async (tx) => {
|
||||
const updated = await tx
|
||||
.update(nativeRunFinalizations)
|
||||
|
|
@ -4092,6 +4224,8 @@ async function executePaperclipNativeSessionWithinScope(
|
|||
phase,
|
||||
leaseOwner: null,
|
||||
leaseExpiresAt: null,
|
||||
recoveryState:
|
||||
phase === "retryable_failure" ? "resuming_session" : "blocked",
|
||||
failureCode,
|
||||
failureDetail: {
|
||||
message,
|
||||
|
|
@ -4114,6 +4248,44 @@ async function executePaperclipNativeSessionWithinScope(
|
|||
: "Resume this same run from its exact persisted native provider checkpoint after the retry delay.",
|
||||
},
|
||||
nextAttemptAt,
|
||||
recoveryHistory: sql`(
|
||||
select coalesce(jsonb_agg(item order by ordinal), '[]'::jsonb)
|
||||
from jsonb_array_elements(
|
||||
coalesce(${nativeRunFinalizations.recoveryHistory}, '[]'::jsonb)
|
||||
|| jsonb_build_array(${JSON.stringify({
|
||||
at: now.toISOString(),
|
||||
disposition: phase,
|
||||
reason: sourceFailureCode,
|
||||
controllerBootId: controller.bootId,
|
||||
controllerGeneration:
|
||||
input.restartRecovery?.controllerGeneration ?? null,
|
||||
providerAttempt: attempt,
|
||||
stderrTail: sanitizedStderrTail,
|
||||
providerSessionEstablished:
|
||||
recoveryEvidence.providerSessionEstablished,
|
||||
checkpointExists: recoveryEvidence.checkpointExists,
|
||||
})}::jsonb)
|
||||
) with ordinality as history(item, ordinal)
|
||||
where ordinal > greatest(
|
||||
jsonb_array_length(
|
||||
coalesce(${nativeRunFinalizations.recoveryHistory}, '[]'::jsonb)
|
||||
|| jsonb_build_array(${JSON.stringify({
|
||||
at: now.toISOString(),
|
||||
disposition: phase,
|
||||
reason: sourceFailureCode,
|
||||
controllerBootId: controller.bootId,
|
||||
controllerGeneration:
|
||||
input.restartRecovery?.controllerGeneration ?? null,
|
||||
providerAttempt: attempt,
|
||||
stderrTail: sanitizedStderrTail,
|
||||
providerSessionEstablished:
|
||||
recoveryEvidence.providerSessionEstablished,
|
||||
checkpointExists: recoveryEvidence.checkpointExists,
|
||||
})}::jsonb)
|
||||
) - 20,
|
||||
0
|
||||
)
|
||||
)`,
|
||||
updatedAt: now,
|
||||
})
|
||||
.where(
|
||||
|
|
@ -4126,6 +4298,12 @@ async function executePaperclipNativeSessionWithinScope(
|
|||
eq(nativeRunFinalizations.issueId, input.execution.binding.issueId),
|
||||
eq(nativeRunFinalizations.leaseOwner, leaseOwner),
|
||||
eq(nativeRunFinalizations.attempt, attempt),
|
||||
eq(nativeRunFinalizations.controllerBootId, controller.bootId),
|
||||
eq(nativeRunFinalizations.controllerPid, controller.pid),
|
||||
eq(
|
||||
nativeRunFinalizations.controllerProcessStartedAt,
|
||||
controller.processStartedAt,
|
||||
),
|
||||
gt(nativeRunFinalizations.leaseExpiresAt, sql`now()`),
|
||||
),
|
||||
)
|
||||
|
|
@ -4227,6 +4405,8 @@ async function executePaperclipNativeSessionWithinScope(
|
|||
.set({
|
||||
leaseOwner: null,
|
||||
leaseExpiresAt: null,
|
||||
recoveryState: null,
|
||||
recoveryRequestId: null,
|
||||
updatedAt: releaseNow,
|
||||
})
|
||||
.where(
|
||||
|
|
@ -4236,6 +4416,12 @@ async function executePaperclipNativeSessionWithinScope(
|
|||
eq(nativeRunFinalizations.issueId, input.execution.binding.issueId),
|
||||
eq(nativeRunFinalizations.leaseOwner, leaseOwner),
|
||||
eq(nativeRunFinalizations.attempt, attempt),
|
||||
eq(nativeRunFinalizations.controllerBootId, controller.bootId),
|
||||
eq(nativeRunFinalizations.controllerPid, controller.pid),
|
||||
eq(
|
||||
nativeRunFinalizations.controllerProcessStartedAt,
|
||||
controller.processStartedAt,
|
||||
),
|
||||
gt(nativeRunFinalizations.leaseExpiresAt, sql`now()`),
|
||||
),
|
||||
)
|
||||
|
|
@ -5184,6 +5370,7 @@ export async function createRunnerdBackend(input: {
|
|||
db: Db;
|
||||
execution: NativeExecutionInput;
|
||||
runnerInstanceId: string;
|
||||
restartRecovery?: NativeRestartRecoveryClaim;
|
||||
durableEnvironmentLeaseId?: string;
|
||||
onSpawn?: (meta: {
|
||||
pid: number;
|
||||
|
|
@ -5225,6 +5412,7 @@ export async function createRunnerdBackend(input: {
|
|||
await migrateRunnerdStateRootForExecution({
|
||||
db: input.db,
|
||||
execution: input.execution,
|
||||
restartRecovery: input.restartRecovery,
|
||||
});
|
||||
return await createRunnerdBackendWithinSessionClaim(input, sessionScopeId);
|
||||
} finally {
|
||||
|
|
@ -6557,6 +6745,11 @@ async function createRunnerdBackendWithinSessionClaim(
|
|||
}
|
||||
remotePrepared = false;
|
||||
};
|
||||
const adoptedProcess =
|
||||
target.kind === "local" &&
|
||||
input.restartRecovery?.kind === "reattach_existing_runner"
|
||||
? input.restartRecovery.process
|
||||
: null;
|
||||
const backend = createNativeSessionBackend(runnerExecution, {
|
||||
runnerInstanceId: input.runnerInstanceId,
|
||||
environment: effectiveRunnerEnvironment,
|
||||
|
|
@ -6693,6 +6886,14 @@ async function createRunnerdBackendWithinSessionClaim(
|
|||
: undefined,
|
||||
runnerProcessLauncher: remoteProcessLauncher,
|
||||
runnerReconnectGraceMs: remoteTarget ? 120_000 : undefined,
|
||||
adoptExistingRunner: adoptedProcess
|
||||
? {
|
||||
...adoptedProcess,
|
||||
isAlive: () => verifiedRecoveryProcessIsAlive(adoptedProcess),
|
||||
signal: (signal) =>
|
||||
signalVerifiedRecoveryProcess(adoptedProcess, signal),
|
||||
}
|
||||
: undefined,
|
||||
environment: effectiveRunnerEnvironment,
|
||||
lifecyclePolicy: input.execution.session.lifecyclePolicy,
|
||||
runtimeContext:
|
||||
|
|
|
|||
|
|
@ -1,7 +1,10 @@
|
|||
import { describe, expect, it } from "vitest";
|
||||
import { canonicalNativeRuntimeContextDigest } from "../../vendor/paperclip-runner/index.js";
|
||||
import { buildNativeExecutionInput } from "./native-execution-input.js";
|
||||
import { rebindNativeSessionCheckpoint } from "./native-session-resume.js";
|
||||
import {
|
||||
isUnusedLegacyNativeRetryReplacement,
|
||||
rebindNativeSessionCheckpoint,
|
||||
} from "./native-session-resume.js";
|
||||
import { nativeRuntimeContextFixture } from "./runtime-context.test-fixture.js";
|
||||
|
||||
const companyId = "10000000-0000-4000-8000-000000000001";
|
||||
|
|
@ -87,6 +90,51 @@ function previousRun(overrides: Record<string, unknown> = {}) {
|
|||
}
|
||||
|
||||
describe("rebindNativeSessionCheckpoint", () => {
|
||||
it("permits legacy retry rebinding only before the replacement acquired authority", () => {
|
||||
const source = {
|
||||
runtimeMode: "native",
|
||||
status: "interrupted",
|
||||
nativeSessionId: normalizedSessionId,
|
||||
};
|
||||
const replacement = {
|
||||
processPid: null,
|
||||
processGroupId: null,
|
||||
processStartedAt: null,
|
||||
runnerProfileJson: {},
|
||||
};
|
||||
expect(
|
||||
isUnusedLegacyNativeRetryReplacement({
|
||||
source,
|
||||
replacement,
|
||||
hasProviderEvents: false,
|
||||
}),
|
||||
).toBe(true);
|
||||
expect(
|
||||
isUnusedLegacyNativeRetryReplacement({
|
||||
source,
|
||||
replacement: { ...replacement, processPid: 123 },
|
||||
hasProviderEvents: false,
|
||||
}),
|
||||
).toBe(false);
|
||||
expect(
|
||||
isUnusedLegacyNativeRetryReplacement({
|
||||
source,
|
||||
replacement,
|
||||
hasProviderEvents: true,
|
||||
}),
|
||||
).toBe(false);
|
||||
expect(
|
||||
isUnusedLegacyNativeRetryReplacement({
|
||||
source,
|
||||
replacement: {
|
||||
...replacement,
|
||||
runnerProfileJson: { sessionCheckpoint: { providerSessionId: "claimed" } },
|
||||
},
|
||||
hasProviderEvents: false,
|
||||
}),
|
||||
).toBe(false);
|
||||
});
|
||||
|
||||
it("retains provider identity but clears prior turn and event state", () => {
|
||||
const rebound = rebindNativeSessionCheckpoint({
|
||||
previousRun: previousRun(),
|
||||
|
|
|
|||
|
|
@ -15,6 +15,48 @@ export function isNativeSessionId(value: unknown): value is string {
|
|||
&& /^[0-9a-f]{8}-[0-9a-f]{4}-[1-5][0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$/i.test(value);
|
||||
}
|
||||
|
||||
const LEGACY_RETRY_SOURCE_TERMINAL_STATUSES = new Set([
|
||||
"succeeded",
|
||||
"interrupted",
|
||||
"failed",
|
||||
"cancelled",
|
||||
"timed_out",
|
||||
]);
|
||||
|
||||
/**
|
||||
* Legacy compatibility is deliberately narrower than ordinary task-session
|
||||
* continuation: only a replacement row that never acquired any native process
|
||||
* or provider authority may be rebound to a terminal native source. The exact
|
||||
* checkpoint/session/workspace/provider binding is validated separately by
|
||||
* rebindNativeSessionCheckpoint.
|
||||
*/
|
||||
export function isUnusedLegacyNativeRetryReplacement(input: {
|
||||
replacement: {
|
||||
processPid: number | null;
|
||||
processGroupId: number | null;
|
||||
processStartedAt: Date | null;
|
||||
runnerProfileJson: unknown;
|
||||
};
|
||||
source: {
|
||||
runtimeMode: string | null;
|
||||
status: string;
|
||||
nativeSessionId: string | null;
|
||||
} | null;
|
||||
hasProviderEvents: boolean;
|
||||
}): boolean {
|
||||
const replacementProfile = record(input.replacement.runnerProfileJson);
|
||||
return Boolean(
|
||||
input.source?.runtimeMode === "native" &&
|
||||
LEGACY_RETRY_SOURCE_TERMINAL_STATUSES.has(input.source.status) &&
|
||||
isNativeSessionId(input.source.nativeSessionId) &&
|
||||
input.replacement.processPid === null &&
|
||||
input.replacement.processGroupId === null &&
|
||||
input.replacement.processStartedAt === null &&
|
||||
replacementProfile.sessionCheckpoint == null &&
|
||||
!input.hasProviderEvents,
|
||||
);
|
||||
}
|
||||
|
||||
function sameProvider(
|
||||
previous: NativeExecutionInput["provider"],
|
||||
current: NativeExecutionInput["provider"],
|
||||
|
|
|
|||
|
|
@ -791,7 +791,7 @@ export function recoveryService(db: Db, deps: { enqueueWakeup: RecoveryWakeup })
|
|||
}
|
||||
|
||||
async function hasActiveExecutionPath(companyId: string, issueId: string, agentId?: string | null) {
|
||||
const [run, deferredWake] = await Promise.all([
|
||||
const [run, deferredWake, nativeRecovery] = await Promise.all([
|
||||
db
|
||||
.select({ id: heartbeatRuns.id })
|
||||
.from(heartbeatRuns)
|
||||
|
|
@ -818,9 +818,35 @@ export function recoveryService(db: Db, deps: { enqueueWakeup: RecoveryWakeup })
|
|||
)
|
||||
.limit(1)
|
||||
.then((rows) => rows[0] ?? null),
|
||||
db
|
||||
.select({ id: nativeRunFinalizations.runId })
|
||||
.from(nativeRunFinalizations)
|
||||
.innerJoin(
|
||||
heartbeatRuns,
|
||||
eq(heartbeatRuns.id, nativeRunFinalizations.runId),
|
||||
)
|
||||
.where(
|
||||
and(
|
||||
eq(nativeRunFinalizations.companyId, companyId),
|
||||
eq(nativeRunFinalizations.issueId, issueId),
|
||||
isNull(nativeRunFinalizations.resultId),
|
||||
agentId ? eq(heartbeatRuns.agentId, agentId) : sql`true`,
|
||||
or(
|
||||
inArray(nativeRunFinalizations.recoveryState, [
|
||||
"awaiting_evidence",
|
||||
"awaiting_runner_reattach",
|
||||
"resuming_session",
|
||||
"bootstrap_incomplete",
|
||||
]),
|
||||
eq(nativeRunFinalizations.phase, "retryable_failure"),
|
||||
),
|
||||
),
|
||||
)
|
||||
.limit(1)
|
||||
.then((rows) => rows[0] ?? null),
|
||||
]);
|
||||
|
||||
return Boolean(run || deferredWake);
|
||||
return Boolean(run || deferredWake || nativeRecovery);
|
||||
}
|
||||
|
||||
async function hasPendingWakeInteraction(companyId: string, issueId: string) {
|
||||
|
|
|
|||
|
|
@ -2,6 +2,7 @@ import { EventEmitter } from "node:events";
|
|||
import { describe, expect, it, vi } from "vitest";
|
||||
import {
|
||||
coordinateHeartbeatSchedulerShutdown,
|
||||
drainRunExecutionFinalizersForShutdown,
|
||||
finalizeServerShutdown,
|
||||
loadWithoutCoordinatedShutdownSignalHooks,
|
||||
} from "./shutdown.js";
|
||||
|
|
@ -145,6 +146,43 @@ describe("finalizeServerShutdown", () => {
|
|||
});
|
||||
});
|
||||
|
||||
describe("drainRunExecutionFinalizersForShutdown", () => {
|
||||
it("awaits bounded execution finalizers", async () => {
|
||||
const release = deferred();
|
||||
const drain = vi.fn(() => release.promise);
|
||||
const pending = drainRunExecutionFinalizersForShutdown({
|
||||
signal: "SIGTERM",
|
||||
drain,
|
||||
timeoutMs: 1_000,
|
||||
log: stubLogger(),
|
||||
});
|
||||
await vi.waitFor(() => expect(drain).toHaveBeenCalledOnce());
|
||||
release.resolve();
|
||||
await expect(pending).resolves.toBe("drained");
|
||||
});
|
||||
|
||||
it("returns after the bounded timeout when an adopted run remains active", async () => {
|
||||
vi.useFakeTimers();
|
||||
try {
|
||||
const log = stubLogger();
|
||||
const pending = drainRunExecutionFinalizersForShutdown({
|
||||
signal: "SIGINT",
|
||||
drain: () => new Promise<void>(() => undefined),
|
||||
timeoutMs: 250,
|
||||
log,
|
||||
});
|
||||
await vi.advanceTimersByTimeAsync(250);
|
||||
await expect(pending).resolves.toBe("timed_out");
|
||||
expect(log.info).toHaveBeenCalledWith(
|
||||
expect.objectContaining({ timeoutMs: 250 }),
|
||||
expect.stringContaining("timed out"),
|
||||
);
|
||||
} finally {
|
||||
vi.useRealTimers();
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe("loadWithoutCoordinatedShutdownSignalHooks", () => {
|
||||
it("removes the eager signal handlers from the real embedded-postgres import", async () => {
|
||||
const before = {
|
||||
|
|
|
|||
|
|
@ -7,6 +7,35 @@ type ShutdownLogger = {
|
|||
error(obj: object, msg: string): void;
|
||||
};
|
||||
|
||||
export async function drainRunExecutionFinalizersForShutdown(input: {
|
||||
signal: "SIGINT" | "SIGTERM";
|
||||
drain: (() => Promise<void>) | null;
|
||||
timeoutMs?: number;
|
||||
log: ShutdownLogger;
|
||||
}): Promise<"drained" | "timed_out" | "unavailable"> {
|
||||
if (!input.drain) return "unavailable";
|
||||
const timeoutMs = input.timeoutMs ?? 5_000;
|
||||
let timer: NodeJS.Timeout | null = null;
|
||||
try {
|
||||
const result = await Promise.race([
|
||||
input.drain().then(() => "drained" as const),
|
||||
new Promise<"timed_out">((resolve) => {
|
||||
timer = setTimeout(() => resolve("timed_out"), timeoutMs);
|
||||
timer.unref?.();
|
||||
}),
|
||||
]);
|
||||
if (result === "timed_out") {
|
||||
input.log.info(
|
||||
{ signal: input.signal, timeoutMs },
|
||||
"bounded heartbeat execution finalizer drain timed out",
|
||||
);
|
||||
}
|
||||
return result;
|
||||
} finally {
|
||||
if (timer) clearTimeout(timer);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Runs the final, ordered teardown of the server. It awaits the application
|
||||
* service cleanup first, so a live setup-token login session stops and releases
|
||||
|
|
|
|||
|
|
@ -0,0 +1,21 @@
|
|||
export type StartupRecoveryPhase = "starting" | "recovering" | "ready";
|
||||
|
||||
let phase: StartupRecoveryPhase = "ready";
|
||||
let updatedAt = new Date().toISOString();
|
||||
|
||||
export function setStartupRecoveryPhase(next: StartupRecoveryPhase): void {
|
||||
phase = next;
|
||||
updatedAt = new Date().toISOString();
|
||||
}
|
||||
|
||||
export function getStartupRecoveryState(): {
|
||||
phase: StartupRecoveryPhase;
|
||||
updatedAt: string;
|
||||
} {
|
||||
return { phase, updatedAt };
|
||||
}
|
||||
|
||||
export function resetStartupRecoveryStateForTests(): void {
|
||||
phase = "ready";
|
||||
updatedAt = new Date().toISOString();
|
||||
}
|
||||
Loading…
Reference in New Issue