function getTimeMs(): number { return dateNowOverride ? dateNowOverride() : Date.now();
}
function sleepSync(ms: number): void { const timeoutMs = Math.max(0, Math.floor(ms)); if (timeoutMs <= 0) { return;
} if (sleepSyncOverride) {
sleepSyncOverride(timeoutMs); return;
} try { const lock = new Int32Array(new SharedArrayBuffer(4));
Atomics.wait(lock, 0, 0, timeoutMs);
} catch { const start = Date.now(); while (Date.now() - start < timeoutMs) { // Best-effort fallback when Atomics.wait is unavailable.
}
}
}
function getParentPid(): number { return parentPidOverride ? parentPidOverride() : process.ppid;
}
/** *ReadasingleancestorPIDfrom`/proc/<pid>/status`onLinux. *Returnsnullonanyfailure(non-Linuxplatform,restricted/proc,race *wherethetargetpidexitedbetweenthewalkstepandtheread);callers *treatanullreturnas"stopwalking"andproceedwiththeancestorset *collectedsofar.
*/ function readParentPidFromProc(pid: number): number | null { try { const status = readFileSync(`/proc/${pid}/status`, "utf8"); const match = status.match(/^PPid:\s*(\d+)/m); if (!match) { returnnull;
} const parsed = Number.parseInt(match[1] ?? "", 10); return Number.isFinite(parsed) && parsed > 0 ? parsed : null;
} catch { // Null truncates the walk at this hop. In hardened Linux (hidepid=2, // gVisor, AppArmor-locked namespaces) /proc is unreadable beyond the // caller, so the walk can stop at `process.ppid`. #68451's direct // gateway→sidecar topology is covered (ppid is captured without a // /proc read); 3-level chains (gateway→plugin-host→sidecar) are not // — pinned by the "grandparent stays killable when /proc truncates // the walk" regression test. returnnull;
}
}
/** *CollectthesetofPIDswhoseterminationwouldcascade-killthecaller: *thecurrentprocess,itsdirectparent,and—wheretheplatformpermits *—thefullancestorchainuptothetopofthepidnamespace. * *Rationale:`cleanStaleGatewayProcessesSync`alreadyrefusestokill *`process.pid`(see`parsePidsFromLsofOutput`),acknowledgingtheinvariant *"acleanupstepmustneverdestroyitsowncaller."Thatinvariantwas *appliedonlytothecalleritself,nottoitsancestors—whichishow *issue#68451arises:apluginsidecarcallsthecleanup,`lsof`reports *theparentgatewaylisteningon18789,theparent'sPIDpassesthe *`pid!==process.pid`filter,itisSIGTERM'd,thesidecaristhenreaped *bythesupervisor,thesupervisorrestartsthegateway,whichre-spawns *thesidecar,whichrunsthecleanupagain—infiniterestartloop. * *Completingtheinvarianthereremovestheloopatitssource:killingany *ancestorisexactlyasfataltothecalleraskillingitself,soancestors *mustreceivethesameexclusiontreatment.Thecheckadmitsanypositive *ancestorPID(including1),becauseinsideacontainer—afirst-class *deploymenttargetforthisproject—thegatewayisfrequentlythe *entrypointandthereforerunsasPID1ofitsownnamespace;excluding1 *unconditionallywouldrecreatethe#68451looponeverycontainerised *installwherethegatewayspawnsadirect-childsidecar. * *Thewalkisbest-effort.`process.ppid`isprovidedbyNodeviaadirect *syscallandisalwaysavailable;transitiveancestorsareonlyreadon *Linuxvia`/proc`.macOS/Windowsstopatppid,whichissufficientfor *thedirect-childsidecartopologythisbugdescribes;extendingthose *platformscanbedonewithouttouchingthecallsites. * *Thefunctiontakesnoparametersandexposesnohooks.Testsexercise *therealwalkbystubbing`process.ppid`(and,onLinux,bymocking *`node:fs`toinject`/proc/<pid>/status`payloads)—thereisno *reachableoverrideforruntimecallerstomutate.
*/ function getSelfAndAncestorPidsSync(): Set<number> { const pids = new Set<number>([process.pid]); const immediateParent = getParentPid(); if (!Number.isFinite(immediateParent) || immediateParent <= 0) { return pids;
}
pids.add(immediateParent); if (process.platform !== "linux") { return pids;
} // Transitive ancestor walk. Each hop's validity (positive pid, not already // seen) is enforced by the per-iteration `parent` check below; the entry // invariant `current > 0` is established above and preserved by `current = // parent` after the same check, so no separate top-of-loop guard is needed.
let current = immediateParent; for (let depth = 0; depth < MAX_ANCESTOR_WALK_DEPTH; depth++) { const parent = readParentPidFromProc(current); if (parent == null || parent <= 0 || pids.has(parent)) { break;
}
pids.add(parent);
current = parent;
} return pids;
}
/** *ParseopenclawgatewayPIDsfromlsof-Fpcstdout,excludingthecurrent *processanditsancestors(see`getSelfAndAncestorPidsSync`forthefull *rationale).OnLinuxtheancestorlookupreadsupto *`MAX_ANCESTOR_WALK_DEPTH`entriesfrom`/proc/<pid>/status`;eachreadis *avirtual-filesystemaccess(nodiskI/O,noexternalprocess),wrapped *intry/catchanddegradessilently.OnmacOS/Windowsthelookupis *in-memoryvia`process.ppid`only.
*/ function parsePidsFromLsofOutput(stdout: string): number[] { const pids: number[] = [];
let currentPid: number | undefined;
let currentCmd: string | undefined; for (const line of stdout.split(/\r?\n/).filter(Boolean)) { if (line.startsWith("p")) { if (
currentPid != null &&
currentCmd &&
normalizeLowercaseStringOrEmpty(currentCmd).includes("openclaw")
) {
pids.push(currentPid);
} const parsed = Number.parseInt(line.slice(1), 10);
currentPid = Number.isFinite(parsed) && parsed > 0 ? parsed : undefined;
currentCmd = undefined;
} elseif (line.startsWith("c")) {
currentCmd = line.slice(1);
}
} if (
currentPid != null &&
currentCmd &&
normalizeLowercaseStringOrEmpty(currentCmd).includes("openclaw")
) {
pids.push(currentPid);
} // Deduplicate: dual-stack listeners (IPv4 + IPv6) cause lsof to emit the // same PID twice. Return each PID at most once to avoid double-killing. // Exclude self and ancestors — terminating any ancestor cascade-kills the // caller via the supervisor, recreating the #68451 restart loop. const excluded = getSelfAndAncestorPidsSync(); return [...new Set(pids)].filter((pid) => !excluded.has(pid));
}
function pollPortOnce(port: number): PollResult { if (process.platform === "win32") { return pollPortOnceWindows(port);
} try { const lsof = resolveLsofCommandSync(); const res = spawnSync(lsof, ["-nP", `-iTCP:${port}`, "-sTCP:LISTEN", "-Fpc"], {
encoding: "utf8",
timeout: POLL_SPAWN_TIMEOUT_MS,
}); if (res.error) { // Spawn-level failure. ENOENT / EACCES means lsof is permanently // unavailable on this system; other errors (e.g. timeout) are transient. const code = (res.error as NodeJS.ErrnoException).code; const permanent = code === "ENOENT" || code === "EACCES" || code === "EPERM"; return { free: null, permanent };
} if (res.status === 1) { // lsof canonical "no matching processes" exit — port is genuinely free. // Guard: on Linux containers with restricted /proc (AppArmor, seccomp, // user namespaces), lsof can exit 1 AND still emit some output for the // processes it could read. Parse stdout when non-empty to avoid false-free. if (res.stdout) { const pids = parsePidsFromLsofOutput(res.stdout); return pids.length === 0 ? { free: true } : { free: false };
} return { free: true };
} if (res.status !== 0) { // status > 1: runtime/permission/flag error. Cannot confirm port state — // treat as a transient failure and keep polling rather than falsely // reporting the port as free (which would recreate the EADDRINUSE race). return { free: null, permanent: false };
} // status === 0: lsof found listeners. Parse pids from the stdout we // already hold — no second lsof spawn, no new failure surface. const pids = parsePidsFromLsofOutput(res.stdout); return pids.length === 0 ? { free: true } : { free: false };
} catch { return { free: null, permanent: false };
}
}
/** *Pollthegivenportuntilitisconfirmedfree,lsofisconfirmedunavailable, *orthewall-clockbudgetexpires. * *EachpollinvocationusesPOLL_SPAWN_TIMEOUT_MS(400ms),whichis *significantlyshorterthanPORT_FREE_TIMEOUT_MS(2000ms).Thisensures *thatasinglesloworhunglsofcallcannotconsumetheentirepolling *budgetandcausethefunctiontoexitprematurelywithaninconclusive *result.Uptofiveindependentlsofattemptsfitwithinthebudget. * *Exitconditions: *-`pollPortOnce`returns`{free:true}`→portconfirmedfree *-`pollPortOnce`returns`{free:null,permanent:true}`→lsofunavailable,bail *-`pollPortOnce`returns`{free:false}`→portbusy,sleep+retry *-`pollPortOnce`returns`{free:null,permanent:false}`→transienterror,sleep+retry *-Wall-clockdeadlineexceeded→logwarning,proceedanyway
*/ function waitForPortFreeSync(port: number): void { const deadline = getTimeMs() + PORT_FREE_TIMEOUT_MS; while (getTimeMs() < deadline) { const result = pollPortOnce(port); if (result.free === true) { return;
} if (result.free === null && result.permanent) { // lsof is permanently unavailable (ENOENT / EACCES) — bail immediately, // no point spinning the remaining budget. return;
} // result.free === false: port still bound. // result.free === null && !permanent: transient lsof error — keep polling.
sleepSync(PORT_FREE_POLL_INTERVAL_MS);
}
restartLog.warn(`port ${port} still in use after ${PORT_FREE_TIMEOUT_MS}ms; proceeding anyway`);
}
/** *Inspectthegatewayportandkillanystalegatewayprocessesholdingit. *Blocksuntiltheportisconfirmedfree(orthepollbudgetexpires)so *thesupervisor(systemd/launchctl)doesnotraceazombieprocessfor *theportandenteranEADDRINUSErestartloop. * *Calledbeforeservicerestartcommandstopreventportconflicts.
*/
export function cleanStaleGatewayProcessesSync(portOverride?: number): number[] { try { const port = typeof portOverride === "number" && Number.isFinite(portOverride) && portOverride > 0
? Math.floor(portOverride)
: resolveGatewayPort(undefined, process.env); const stalePids =
process.platform === "win32"
? (() => { const result = findVerifiedWindowsGatewayPidsOnPortResultSync(port); if (result.ok) { return result.pids;
}
waitForPortFreeSync(port); return [];
})()
: findGatewayPidsOnPortSync(port); if (stalePids.length === 0) { return [];
}
restartLog.warn(
`killing ${stalePids.length} stale gateway process(es) before restart: ${stalePids.join(", ")}`,
); const killed = terminateStaleProcessesSync(stalePids); // Wait for the port to be released before returning — called unconditionally // even when `killed` is empty (all pids were already dead before SIGTERM). // A process can exit before our signal arrives yet still leave its socket // in TIME_WAIT / FIN_WAIT; polling is the only reliable way to confirm the // kernel has fully released the port before systemd fires the new process.
waitForPortFreeSync(port); return killed;
} catch { return [];
}
}
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.