fix(plugin-worker-manager): recover when a worker's stdin command channel dies
A plugin worker communicates with the host over newline-delimited JSON-RPC
on the child's stdin/stdout pipes. If the host->worker stdin pipe is
destroyed by an EPIPE (or otherwise closes) while the child process keeps
running, the worker becomes uncommandable: every `sendMessage` throws
`Worker process for plugin "<id>" is not writable`, so all host->worker
calls (e.g. http.fetch proxying) fail forever. Crucially `child.on("exit")`
never fires, so the existing crash-recovery/backoff path is never reached
and the worker silently zombies — observed in a Telegram-bridge plugin whose
getUpdates long-poll began failing after ~5 days of uptime and never
recovered until a manual disable/enable.
Fix (no polling watchdog — supervise the command channel the same way the
process is already supervised):
1. `sendMessage` passes a write callback so an async stdin write error is
surfaced/logged instead of being swallowed.
2. `attachStdioHandlers` attaches `error`/`close` handlers to `child.stdin`.
If the channel dies while the process is still alive (exitCode/signalCode
null) and the stop wasn't intentional, it SIGKILLs the child so the
normal `handleProcessExit() -> scheduleRestart()` recovery runs. This is
the missing third failure mode, mirroring the existing exit/error handlers.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
parent
aa6b6bcb41
commit
2b1a499678
|
|
@ -529,7 +529,14 @@ export function createPluginWorkerHandle(
|
|||
throw new Error(`Worker process for plugin "${pluginId}" is not writable`);
|
||||
}
|
||||
const serialized = serializeMessage(message as any);
|
||||
childProcess.stdin.write(serialized);
|
||||
// Pass a write callback so an async write error (e.g. EPIPE on the command
|
||||
// pipe) is surfaced rather than swallowed. The stdin "error" handler wired
|
||||
// in attachStdioHandlers() drives the actual recovery (forced restart).
|
||||
childProcess.stdin.write(serialized, (err) => {
|
||||
if (err) {
|
||||
log.warn({ err: err.message }, "failed to write message to worker stdin");
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
function errorCodeForWorkerHostError(err: unknown): number {
|
||||
|
|
@ -928,6 +935,36 @@ export function createPluginWorkerHandle(
|
|||
);
|
||||
}
|
||||
});
|
||||
|
||||
// Supervise the command channel (host -> worker stdin), not just the
|
||||
// process. If stdin errors (EPIPE) or closes while the process is still
|
||||
// alive, the worker is uncommandable: every host->worker RPC write will
|
||||
// throw "not writable" forever, yet child.on("exit") never fires, so the
|
||||
// crash-recovery path is never reached and the worker silently zombies
|
||||
// (observed after multi-day uptime). Force a real exit so the standard
|
||||
// handleProcessExit() -> scheduleRestart() recovery runs. This is the
|
||||
// missing third failure mode, mirroring the exit/error handlers above —
|
||||
// event-driven, not a polling watchdog.
|
||||
if (child.stdin) {
|
||||
const onCommandChannelLost = (err?: Error): void => {
|
||||
// Ignore during graceful stop, or if this child was already replaced.
|
||||
if (intentionalStop || childProcess !== child) return;
|
||||
// Only act while the process is still alive (a real exit is handled by
|
||||
// handleProcessExit). exitCode/signalCode are null until the child dies.
|
||||
if (child.exitCode !== null || child.signalCode !== null) return;
|
||||
log.error(
|
||||
{ err: err?.message },
|
||||
"worker stdin (command channel) lost while process alive — forcing restart",
|
||||
);
|
||||
try {
|
||||
child.kill("SIGKILL");
|
||||
} catch {
|
||||
// Best effort — handleProcessExit still runs on the eventual exit.
|
||||
}
|
||||
};
|
||||
child.stdin.on("error", onCommandChannelLost);
|
||||
child.stdin.on("close", onCommandChannelLost);
|
||||
}
|
||||
}
|
||||
|
||||
function handleProcessExit(
|
||||
|
|
|
|||
Loading…
Reference in New Issue