From 9b5afd4f501505ef9d5ecaa535d53fd43f5d7972 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 00:50:22 +0900 Subject: [PATCH] fix(desktop): the app keeps its runtime alive across restarts and unexpected exits Root cause: the desktop app only started a runtime at launch and from the failure page's retry. After Ready it merely recorded its child's exit. A restart the runtime performed itself (Connect as Child, memory or package restart, the recycle after a disconnect) handed the port to a detached grandchild the app could not see, stop or quit, and one that failed to start left no proxy. A crash or a terminal `ocx stop` left the port refusing connections until the app was quit and reopened. A failed in-app update left the app in a terminal Drained phase with no runtime. Fix: - The sidecar is spawned with OCX_DESKTOP_SUPERVISED=1. handleStart consumes it with the other start markers and records the parent pid. While that app is still the parent and alive, acceptSystemRestart (completed and deadline paths) marks recycling and exits 75, and recycleStandalone exits 75 after its cleanup, instead of spawning; both detached replacement environments drop the marker. A link-mode client runtime reclaims its port for 20 s under the marker (25 s with the pinned prefer-retry, inside the app's 30 s startup deadline), and its runtime-port.json carries an attestation secret like a standalone start. - desktop/src-tauri/src/supervisor.rs: the sidecar watch reports each child's exit once it is recorded; a pure decide() respawns only for the tracked child with the coordinator idle, the runtime wanted and no ending claimed. Exit 75 goes after 0.5 s when no recovery ran in the last healthy two minutes, otherwise it takes the 3/6/12/24/30 s backoff like other exits. Recovery re-runs the startup sequence in Mode::Recover: resolve stays the only authority, only a proven absence spawns, a live runtime is attached as a guest, and no window or takeover prompt is shown. A 5 s watchdog while Ready recovers on a different pid, or on 3 refused connections in a row (12 for a guest runtime). Decisions go to a 256 KiB runtime-supervisor.log. - A Child's client runtime (resolve role=client) is attached to as a guest at launch and in recovery, never offered a takeover, and owned when it is the app's own child. Refusing it failed every recovery on a Child whose runtime restarted outside the app, forever, one bundled `ocx resolve` per 30 s. - A recovery that finds the port held by a listener the app cannot use (bound off loopback) is not rescheduled: the supervisor parks and the watchdog only asks whether the endpoint changed. - The retry guard waits on the child the app tracks (spawned under 90 s ago, no exit reported) instead of the ownership flag attach() had just reset, so a recovery no longer spawns a second sidecar beside a still-starting one. - A recovery that lands while the update page is up leaves it on screen. - exit.rs: a sticky `wanted` intent, cleared by a tray Stop that takes the phase and by any drain claim (Quit, update), restored by the failure page's retry; abort_restart() returns a coordinated restart's settled drain to Idle, and a failed install re-runs startup in Recover mode. - POST /api/stop from a dashboard session answers 409 desktop_supervised while the desktop app supervises the proxy, before anything is touched: the app would start it again after a full native-Codex teardown. `ocx stop` (tray Stop, Quit, update drain, terminal) is unaffected. Performance: nothing on the standalone or Home request path changes. The marker is read once at start; the parent check runs only at restart time, at a link-mode start and on a dashboard stop. On the desktop, the watchdog is one loopback GET /healthz every 5 s while Ready or parked, and supervision work otherwise happens only on exits and run ends. A held port no longer costs a resolve every 30 s. Security: no new network surface. Nothing is killed or signalled; a released child handle is only dropped. The marker is honored only against the live parent recorded at start, and a runtime whose app died falls back to the detached replacement. The attestation secret goes into the same private runtime-port.json a standalone start writes. The client runtime is attached on loopback only and never taken over. The supervisor log holds pids, exit codes, verdicts and startup failure reasons, never credentials, and is not written through a symlink. Co-Authored-By: Claude Opus 5.5 (1M context) --- desktop/src-tauri/src/exit.rs | 186 +++- desktop/src-tauri/src/lib.rs | 18 + desktop/src-tauri/src/resolve.rs | 33 +- desktop/src-tauri/src/sidecar.rs | 85 +- desktop/src-tauri/src/startup.rs | 218 ++++- desktop/src-tauri/src/supervisor.rs | 896 ++++++++++++++++++ desktop/src-tauri/src/updater.rs | 60 +- desktop/src-tauri/src/window.rs | 5 + .../src/content/docs/guides/desktop-app.md | 27 +- .../src/content/docs/guides/remote-link.md | 2 +- .../src/content/docs/guides/web-dashboard.md | 4 +- .../content/docs/reference/cli/lifecycle.md | 7 + scripts/test-layout/layout.json | 1 + src/cli/restart-handoff.ts | 13 +- src/client/runtime.ts | 51 +- src/lib/system-restart-contract.ts | 68 ++ src/server/management-api.ts | 6 +- src/server/management/system-restart.ts | 29 +- src/server/stop-teardown.ts | 27 + structure/desktop-shell.md | 67 +- structure/gui-and-management-api.md | 5 + structure/ops/service-and-sidecars.md | 11 + tests/clients/desktop-cli-contracts.test.ts | 23 +- tests/clients/desktop-startup-surface.test.ts | 8 + .../desktop-supervised-restart.test.ts | 280 ++++++ tests/fixtures/test-layout-expected.json | 1 + 26 files changed, 2074 insertions(+), 57 deletions(-) create mode 100644 desktop/src-tauri/src/supervisor.rs create mode 100644 tests/clients/desktop-supervised-restart.test.ts diff --git a/desktop/src-tauri/src/exit.rs b/desktop/src-tauri/src/exit.rs index aba0aab4b39..eac40099499 100644 --- a/desktop/src-tauri/src/exit.rs +++ b/desktop/src-tauri/src/exit.rs @@ -134,12 +134,33 @@ pub fn decide(phase: ExitPhase, reason: Option, hides_to_tray: bool) } } +/// What the runtime supervisor needs to know before it brings a runtime back. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct Supervision { + pub phase: ExitPhase, + /// Whether the app still wants a runtime. True from launch; the tray's Stop, a quit's drain and + /// an update's drain clear it before the runtime's exit can arrive, so the supervisor never + /// undoes any of them. + pub wanted: bool, + /// Something has claimed the app's ending: a quit or a restart. + pub reason_set: bool, +} + +impl Supervision { + /// Nothing is in flight, nobody asked for the runtime to stop, and the app is not ending. + pub fn allowed(self) -> bool { + self.phase == ExitPhase::Idle && self.wanted && !self.reason_set + } +} + struct Inner { phase: ExitPhase, reason: Option, hides_to_tray: bool, /// An exit that arrived while a runtime was being started or stopped, and still has to happen. deferred: bool, + /// See [`Supervision::wanted`]. Sticky: finishing a stop does not restore it. + wanted: bool, } /// The exit sequence's state, managed by the app. @@ -157,6 +178,7 @@ impl ExitCoordinator { // tray that turns out not to exist is the exact failure D6 is about. hides_to_tray: TrayAvailability::assumed().hides_to_tray(), deferred: false, + wanted: true, }), } } @@ -193,6 +215,9 @@ impl ExitCoordinator { /// [`ExitCoordinator::finish_stop`] is holding the phase. pub fn claim_drain(&self, fallback: ExitReason) -> Option { let mut inner = self.inner(); + // A quit or an update is on its way, whoever ends up running the drain: the runtime it + // stops is not one to bring back. + inner.wanted = false; match inner.phase { ExitPhase::Idle => { let reason = *inner.reason.get_or_insert(fallback); @@ -226,6 +251,51 @@ impl ExitCoordinator { }; } + /// Give a coordinated restart that is not going to happen back to a running app. + /// + /// An update drains before it installs. When the install then fails, or the drain itself did, + /// the drain's phase used to be the end of the road: `Drained` is terminal, so no runtime could + /// be started again and a bare window close quit the app. This returns the app to `Idle` with + /// no claimed reason and wants a runtime again, which is what a successful update would have + /// ended in too. It touches nothing unless an update's restart holds the phase: a quit is never + /// aborted, and a drain still running belongs to whoever runs it. Returns the phase it left. + pub fn abort_restart(&self) -> Option { + let mut inner = self.inner(); + let left = inner.phase; + let restart = inner.reason == Some(ExitReason::CoordinatedRestart); + let settled = matches!( + left, + ExitPhase::Drained | ExitPhase::DrainFailed | ExitPhase::OwnershipUnknown + ); + if !restart || !settled { + return None; + } + inner.phase = ExitPhase::Idle; + inner.reason = None; + inner.deferred = false; + inner.wanted = true; + Some(left) + } + + /// What the runtime supervisor reads before it acts. + pub fn supervision(&self) -> Supervision { + let inner = self.inner(); + Supervision { + phase: inner.phase, + wanted: inner.wanted, + reason_set: inner.reason.is_some(), + } + } + + pub fn supervision_allowed(&self) -> bool { + self.supervision().allowed() + } + + /// A person asked for a runtime again (the startup page's retry). + pub fn resume(&self) { + self.inner().wanted = true; + } + /// Reserve the right to start a runtime. False once something else owns the phase. /// /// The reservation exists instead of holding the lock across the spawn. Holding it would make @@ -243,8 +313,18 @@ impl ExitCoordinator { } /// Reserve the runtime for a stop that does not end the app. + /// + /// A stop that takes the phase also stops the app wanting a runtime, in the same step, so the + /// runtime's exit can never reach the supervisor first. One that cannot take it stops nothing + /// and changes nothing: the runtime it would have stopped stays supervised. pub fn begin_stop(&self) -> bool { - self.begin(ExitPhase::Stopping) + let mut inner = self.inner(); + if inner.phase != ExitPhase::Idle { + return false; + } + inner.phase = ExitPhase::Stopping; + inner.wanted = false; + true } /// Release the stop, handing back a reason that arrived meanwhile. @@ -545,10 +625,112 @@ fn hide_windows(app: &AppHandle) { mod tests { use super::{ decide, DrainVerdict, ExitCoordinator, ExitDecision, ExitPhase, ExitReason, - RestartReadiness, + RestartReadiness, Supervision, }; use crate::tray_availability::TrayAvailability; + #[test] + fn stop_quit_and_update_stop_wanting_a_runtime_and_only_a_resume_restores_it() { + let coordinator = ExitCoordinator::new(); + assert!(coordinator.supervision_allowed()); + assert!(coordinator.begin_stop()); + assert!(!coordinator.supervision().wanted); + assert_eq!(coordinator.finish_stop(), None); + // Back to Idle and still not wanted: the stop was the person's intent, not a phase. + assert_eq!(coordinator.phase(), ExitPhase::Idle); + assert!(!coordinator.supervision_allowed()); + coordinator.resume(); + assert!(coordinator.supervision_allowed()); + + for reason in [ExitReason::UserQuit, ExitReason::CoordinatedRestart] { + let coordinator = ExitCoordinator::new(); + assert_eq!(coordinator.claim_drain(reason), Some(reason)); + assert!(!coordinator.supervision().wanted); + } + // A stop that could not take the phase stopped nothing, so it changes nothing either. + let coordinator = ExitCoordinator::new(); + assert!(coordinator.begin_spawn()); + assert!(!coordinator.begin_stop()); + assert!(coordinator.supervision().wanted); + } + + #[test] + fn supervision_is_allowed_only_idle_wanted_and_with_no_ending_claimed() { + for phase in [ + ExitPhase::Spawning, + ExitPhase::Stopping, + ExitPhase::Draining, + ExitPhase::Drained, + ExitPhase::DrainFailed, + ExitPhase::OwnershipUnknown, + ] { + let supervision = Supervision { + phase, + wanted: true, + reason_set: false, + }; + assert!(!supervision.allowed(), "{phase:?}"); + } + let idle = Supervision { + phase: ExitPhase::Idle, + wanted: true, + reason_set: false, + }; + assert!(idle.allowed()); + assert!(!Supervision { + wanted: false, + ..idle + } + .allowed()); + assert!(!Supervision { + reason_set: true, + ..idle + } + .allowed()); + let coordinator = ExitCoordinator::new(); + coordinator.claim(ExitReason::UserQuit); + assert!(!coordinator.supervision_allowed()); + } + + #[test] + fn a_failed_update_returns_the_app_to_a_running_runtime() { + for verdict in [ + DrainVerdict::Drained, + DrainVerdict::Failed, + DrainVerdict::OwnershipUnknown, + ] { + let coordinator = ExitCoordinator::new(); + coordinator.set_tray(TrayAvailability::Available); + assert_eq!( + coordinator.claim_drain(ExitReason::CoordinatedRestart), + Some(ExitReason::CoordinatedRestart) + ); + coordinator.finish_drain(verdict); + let left = coordinator.phase(); + assert_eq!(coordinator.abort_restart(), Some(left)); + assert_eq!(coordinator.phase(), ExitPhase::Idle); + // A bare close hides again instead of quitting out of a terminal phase. + assert_eq!(coordinator.decision(), ExitDecision::Hide); + assert!(coordinator.supervision_allowed()); + assert!(coordinator.begin_spawn()); + } + } + + #[test] + fn a_quit_or_a_drain_in_flight_is_never_aborted() { + let coordinator = ExitCoordinator::new(); + coordinator.claim_drain(ExitReason::UserQuit); + coordinator.finish_drain(DrainVerdict::Drained); + assert_eq!(coordinator.abort_restart(), None); + assert_eq!(coordinator.phase(), ExitPhase::Drained); + assert_eq!(coordinator.decision(), ExitDecision::Proceed); + // A restart still draining belongs to whoever runs it. + let coordinator = ExitCoordinator::new(); + coordinator.claim_drain(ExitReason::CoordinatedRestart); + assert_eq!(coordinator.abort_restart(), None); + assert_eq!(coordinator.phase(), ExitPhase::Draining); + } + #[test] fn a_bare_gesture_hides_when_there_is_a_tray_to_come_back_from() { assert_eq!(decide(ExitPhase::Idle, None, true), ExitDecision::Hide); diff --git a/desktop/src-tauri/src/lib.rs b/desktop/src-tauri/src/lib.rs index ce404b34b75..ea7ab910ada 100644 --- a/desktop/src-tauri/src/lib.rs +++ b/desktop/src-tauri/src/lib.rs @@ -36,6 +36,7 @@ mod resolve; mod runtime_stop; mod sidecar; mod startup; +mod supervisor; mod tray; mod tray_availability; mod updater; @@ -57,6 +58,9 @@ pub struct AppState { child: Mutex>, /// The pid of the child this app started, if it started one. child_pid: Mutex>, + /// When that child was spawned, so a later run can tell a child still starting from one that + /// will never answer (`startup::waits_on_child`). + child_spawned: Mutex>, /// Whether the process answering the endpoint has been confirmed to be that child. /// /// Durable consent and current process ownership are different facts. Consent is a recorded @@ -75,6 +79,7 @@ impl AppState { proxy: Mutex::new(None), child: Mutex::new(None), child_pid: Mutex::new(None), + child_spawned: Mutex::new(None), confirmed: AtomicBool::new(false), watch: sidecar::SidecarWatch::default(), } @@ -102,6 +107,12 @@ impl AppState { *Self::slot(&self.child_pid) } + /// How long ago the child this app tracks was spawned; nothing when it tracks none. + pub fn child_age(&self) -> Option { + self.child_pid()?; + Self::slot(&self.child_spawned).map(|spawned| spawned.elapsed()) + } + /// Confirm that the instance answering is the child this app started. /// /// This is the only thing that grants ownership. A spawn records a pid; it does not record that @@ -115,6 +126,7 @@ impl AppState { pub fn adopt(&self, child: CommandChild) { *Self::slot(&self.child_pid) = Some(child.pid()); + *Self::slot(&self.child_spawned) = Some(std::time::Instant::now()); *Self::slot(&self.child) = Some(child); // Spawned, not yet confirmed: the health probe is what establishes that this pid is the // one answering. @@ -128,6 +140,7 @@ impl AppState { pub fn release(&self) { self.confirmed.store(false, Ordering::Release); let _ = Self::slot(&self.child_pid).take(); + let _ = Self::slot(&self.child_spawned).take(); let _ = Self::slot(&self.child).take(); } } @@ -176,8 +189,11 @@ fn startup_phases() -> Vec { } /// Run the startup sequence again. A run already in flight is left alone. +/// +/// A person asking for a runtime again also resumes supervision, even after the tray's Stop. #[tauri::command] fn retry_startup(app: tauri::AppHandle) { + supervisor::resume(&app); startup::begin(&app); } @@ -280,6 +296,8 @@ pub fn run() { app.manage(tray::TrayState::default()); app.manage(exit::ExitCoordinator::new()); app.manage(startup::Startup::new()); + // Registers the exit hook and an idle watchdog; it starts no runtime of its own. + supervisor::watch_runtime(app.handle()); // D7: the window is created and shown before anything is registered, resolved, probed // or started, so every state below has somewhere to be reported. A login launch stays diff --git a/desktop/src-tauri/src/resolve.rs b/desktop/src-tauri/src/resolve.rs index e644b6edcc4..6d50c34150b 100644 --- a/desktop/src-tauri/src/resolve.rs +++ b/desktop/src-tauri/src/resolve.rs @@ -158,6 +158,10 @@ pub enum LiveVerdict { NotLive, /// It is a proxy, on an address this shell can reach. Attach as a guest. Attach, + /// A Child's client runtime, on loopback. It serves Codex through its Home and the Child's own + /// dashboard, not the management plane, and it is never taken over: the shell attaches to it as + /// a guest and asks nothing. When it is the child this app started, that attach is ownership. + Client, /// Something is listening and this shell cannot use it. Never a reason to start a second one. Unusable(String), } @@ -178,10 +182,12 @@ pub fn loopback_reachable(hostname: Option<&str>) -> bool { /// Read a live verdict. /// /// Liveness answers "is something there", and core's predicate accepts a connected client's -/// listener on purpose so duplicate-start avoidance can see it. This shell needs the management -/// plane, so it has to discriminate on the role the CLI carried: a client listener serves machine -/// routes, not `/api/*`, and attaching to it would report Ready against an endpoint the dashboard -/// and the tray cannot use. +/// listener on purpose so duplicate-start avoidance can see it. The shell discriminates on the role +/// the CLI carried: a client listener serves machine routes and the Child's dashboard, not `/api/*`, +/// and the takeover a proxy can be offered does not apply to it. It is also what a Child runs, +/// including this app's own sidecar after Connect as Child, so it is attached to +/// ([`LiveVerdict::Client`]) rather than refused. Refusing it failed every recovery on a Child whose +/// runtime restarted outside the app, and each failure scheduled the next. pub fn live_verdict(resolution: &Resolution) -> LiveVerdict { let Some(resolved) = resolution.resolved() else { return LiveVerdict::NotLive; @@ -189,11 +195,7 @@ pub fn live_verdict(resolution: &Resolution) -> LiveVerdict { if resolved.liveness.status != Status::Live { return LiveVerdict::NotLive; } - if resolved.liveness.role.as_deref() == Some("client") { - return LiveVerdict::Unusable( - "a connected client is listening on this port, not a proxy this app can manage".into(), - ); - } + let client = resolved.liveness.role.as_deref() == Some("client"); if !loopback_reachable(resolved.liveness.hostname.as_deref()) { return LiveVerdict::Unusable(format!( "the runtime is bound to {} and this app only speaks to loopback", @@ -204,6 +206,9 @@ pub fn live_verdict(resolution: &Resolution) -> LiveVerdict { .unwrap_or("an unknown address") )); } + if client { + return LiveVerdict::Client; + } LiveVerdict::Attach } @@ -302,17 +307,23 @@ mod tests { } #[test] - fn a_connected_client_is_live_but_not_a_runtime_to_attach_to() { + fn a_connected_client_is_attached_to_and_never_started_beside() { let client = LIVE.replace( r#""version":"2.61.0""#, r#""version":"2.61.0","role":"client""#, ); let resolution = read(Some(0), client.as_bytes(), b""); + // A Child's runtime: attached to as a guest, never taken over and never refused. + assert_eq!(live_verdict(&resolution), LiveVerdict::Client); + // Live is still live: it is never a reason to start a second one. + assert!(!may_start(&resolution)); + // Off loopback it is as unusable as any other listener there. + let elsewhere = client.replace(r#""pid":42"#, r#""pid":42,"hostname":"::1""#); + let resolution = read(Some(0), elsewhere.as_bytes(), b""); assert!(matches!( live_verdict(&resolution), LiveVerdict::Unusable(_) )); - // Live and unusable is still live: it is never a reason to start a second one. assert!(!may_start(&resolution)); } diff --git a/desktop/src-tauri/src/sidecar.rs b/desktop/src-tauri/src/sidecar.rs index 2a1da6e90c7..71f16db8f0a 100644 --- a/desktop/src-tauri/src/sidecar.rs +++ b/desktop/src-tauri/src/sidecar.rs @@ -8,6 +8,10 @@ //! //! Stopping it is not here. D4 gives that to the bundled `ocx stop`, which owns the receipt-backed //! teardown this process cannot perform on itself; see `runtime_stop.rs`. +//! +//! The exit is also where supervision starts: once it is recorded, the hook `supervisor.rs` +//! registered hears which child ended and how, so a runtime that went away is noticed at once +//! instead of only when the next startup run reads the record. use crate::endpoint::ProxyEndpoint; use std::{ @@ -23,6 +27,15 @@ use tauri_plugin_shell::{ /// refusal, bounded so a chatty runtime cannot grow the buffer for the life of the process. const MAX_LINES: usize = 40; +/// The marker that tells the runtime this app waits on it (`DESKTOP_SUPERVISED_ENV` in +/// `src/lib/system-restart-contract.ts`). Under it a restart exits with +/// [`crate::supervisor::REQUESTED_RESTART_EXIT_CODE`] instead of spawning a detached replacement +/// this app could neither see nor stop, and the supervisor starts the replacement. +pub const SUPERVISED_ENV: &str = "OCX_DESKTOP_SUPERVISED"; + +/// Told which child ended, and how, once its exit is recorded. +pub type ExitHook = Arc; + /// How the sidecar process ended. #[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] pub struct SidecarExit { @@ -75,6 +88,11 @@ impl WatchInner { #[derive(Clone, Default)] pub struct SidecarWatch { inner: Arc>, + hook: Arc>>, + /// The child this view follows. Only a view made by [`SidecarWatch::for_child`] has one, and + /// only such a view reports an exit to the hook: the record is shared across attempts, the pid + /// is what tells the supervisor which attempt ended. + pid: Option, } impl SidecarWatch { @@ -84,6 +102,40 @@ impl SidecarWatch { } } + /// Register what hears about a child's exit. One hook for the life of the app. + pub fn on_exit(&self, hook: impl Fn(u32, SidecarExit) + Send + Sync + 'static) { + if let Ok(mut slot) = self.hook.lock() { + *slot = Some(Arc::new(hook)); + } + } + + /// The same record, following one spawned child. + pub fn for_child(&self, pid: u32) -> Self { + Self { + inner: self.inner.clone(), + hook: self.hook.clone(), + pid: Some(pid), + } + } + + /// Record one event, and report an exit to the hook after it is recorded, so whatever the hook + /// starts reads the exit it was told about. + fn deliver(&self, event: SidecarEvent) { + let exit = match &event { + SidecarEvent::Exited(exit) => Some(*exit), + SidecarEvent::Line(_) => None, + }; + self.record(event); + let (Some(exit), Some(pid)) = (exit, self.pid) else { + return; + }; + // Cloned out so the hook never runs under this lock. + let hook = self.hook.lock().ok().and_then(|slot| slot.clone()); + if let Some(hook) = hook { + hook(pid, exit); + } + } + pub fn exit(&self) -> Option { self.inner.lock().ok().and_then(|inner| inner.exit) } @@ -109,7 +161,7 @@ impl SidecarWatch { tauri::async_runtime::spawn(async move { while let Some(event) = events.recv().await { if let Some(event) = translate(event) { - watch.record(event); + watch.deliver(event); } } }); @@ -151,8 +203,10 @@ pub fn start( .sidecar("ocx") .map_err(|error| error.to_string())? .args(["start", "--port", &endpoint.port.to_string()]) - .env("OPENCODEX_GUI_DIST", gui_dist); + .env("OPENCODEX_GUI_DIST", gui_dist) + .env(SUPERVISED_ENV, "1"); let (events, child) = command.spawn().map_err(|error| error.to_string())?; + let watch = watch.for_child(child.pid()); watch.follow(events); Ok(child) } @@ -160,6 +214,33 @@ pub fn start( #[cfg(test)] mod tests { use super::{SidecarEvent, SidecarExit, SidecarWatch, MAX_LINES}; + use std::sync::{Arc, Mutex}; + + #[test] + fn an_exit_reaches_the_hook_with_the_childs_pid_after_it_is_recorded() { + let watch = SidecarWatch::default(); + let seen = Arc::new(Mutex::new(Vec::new())); + let record = watch.clone(); + let sink = seen.clone(); + watch.on_exit(move |pid, exit| { + // Whatever the hook starts reads the exit it was told about. + let recorded = record.exit().and_then(|exit| exit.code); + sink.lock().unwrap().push((pid, exit.code, recorded)); + }); + // The shared record follows no child, so it names nobody to the hook. + watch.deliver(SidecarEvent::Exited(SidecarExit { + code: Some(1), + signal: None, + })); + let child = watch.for_child(4242); + child.deliver(SidecarEvent::Line("listening on 10100".into())); + child.deliver(SidecarEvent::Exited(SidecarExit { + code: Some(75), + signal: None, + })); + assert_eq!(*seen.lock().unwrap(), vec![(4242, Some(75), Some(75))]); + assert_eq!(watch.lines(), vec!["listening on 10100".to_owned()]); + } #[test] fn the_exit_code_survives_the_event_stream() { diff --git a/desktop/src-tauri/src/startup.rs b/desktop/src-tauri/src/startup.rs index afdb4453125..723e0364537 100644 --- a/desktop/src-tauri/src/startup.rs +++ b/desktop/src-tauri/src/startup.rs @@ -60,6 +60,22 @@ const POLL: Duration = Duration::from_millis(250); /// diagnostic is the one on screen. const SETTLE_GRACE: Duration = Duration::from_secs(2); +/// How long a child this app started may take to answer before a later run stops waiting on it. +/// +/// The slowest start that still ends well is a hard-pinned port reclaim (60 seconds) plus the +/// pinned prefer-retry (5 seconds). Past this, a child that neither answers nor has reported an +/// exit is wedged, or its exit is held up by a grandchild that kept its output pipes open (the +/// shell plugin reports an exit only once both pipes close), and waiting on it again would keep +/// the port empty for good. +pub const CHILD_START_GRACE: Duration = Duration::from_secs(90); + +/// Whether a run waits on the child this app already started instead of starting another; `age` +/// is how long ago the tracked child was spawned, and nothing when the app tracks none. The caller +/// also requires that the child has not reported an exit. +pub fn waits_on_child(age: Option) -> bool { + age.is_some_and(|age| age < CHILD_START_GRACE) +} + /// Where the launch came from. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub enum LaunchOrigin { @@ -73,6 +89,17 @@ pub enum LaunchOrigin { /// presence is the launch origin. pub const AUTOSTART_FLAG: &str = "--autostart"; +/// Why a run was started. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Mode { + /// The app's own launch, or a person's retry. It may ask to take a listening runtime over. + Launch, + /// The supervisor bringing a runtime back (`supervisor.rs`). Nobody is looking at this run, so + /// it never shows the window or a prompt: a runtime that answers is attached as a guest. Every + /// other gate is the launch's own — only a proven absence starts a runtime. + Recover, +} + impl LaunchOrigin { pub fn from_args(mut args: impl Iterator) -> Self { if args.any(|argument| argument == AUTOSTART_FLAG) { @@ -365,6 +392,11 @@ pub struct Startup { /// before `live`, and never held across an await. reporting: Mutex<()>, running: AtomicBool, + /// Whether the run in flight is a [`Mode::Recover`] run. + recovering: AtomicBool, + /// Set when the run failed because a listener this app cannot use holds the port. The + /// supervisor reads it when the run ends: another attempt would only find the same listener. + held: Mutex>, /// Whether this window has already left the bundled bootstrap surface. /// /// Explicit open actions can arrive repeatedly from the tray, the single-instance hook, and @@ -397,6 +429,8 @@ impl Startup { }), reporting: Mutex::new(()), running: AtomicBool::new(false), + recovering: AtomicBool::new(false), + held: Mutex::new(None), dashboard_loaded: AtomicBool::new(false), dashboard_requested: AtomicBool::new(false), generation: AtomicU64::new(0), @@ -427,6 +461,37 @@ impl Startup { self.live().latest.clone() } + /// Whether a run is in flight. + pub fn is_running(&self) -> bool { + self.running.load(Ordering::Acquire) + } + + /// Whether the last run reported Ready. + pub fn is_ready(&self) -> bool { + self.live().latest.phase == Phase::Ready.id() + } + + fn mode(&self) -> Mode { + if self.recovering.load(Ordering::Acquire) { + Mode::Recover + } else { + Mode::Launch + } + } + + fn held_slot(&self) -> MutexGuard<'_, Option> { + self.held.lock().unwrap_or_else(PoisonError::into_inner) + } + + /// Record that this run found the port held by a listener it cannot use. + fn note_held(&self, pid: Option) { + *self.held_slot() = Some(crate::supervisor::Held { pid }); + } + + fn take_held(&self) -> Option { + self.held_slot().take() + } + /// The user's answer to a pending takeover prompt. Nothing pending is a no-op: a retry /// or a late click must never be read as a decision for a prompt that is not up. pub fn decide_takeover(&self, approved: bool) { @@ -474,6 +539,7 @@ impl Startup { live.consent = ConsentState::Idle; live.reported.clear(); live.latest = Progress::new(Phase::NotStarted, 0); + *self.held_slot() = None; self.dashboard_loaded.store(false, Ordering::SeqCst); self.dashboard_requested.store(false, Ordering::SeqCst); } @@ -600,16 +666,26 @@ impl Default for Startup { /// Run the sequence, unless it is already running. This is also the retry. pub fn begin(app: &AppHandle) { + begin_with(app, Mode::Launch); +} + +/// Run the sequence for `mode`, unless one is already running. Returns whether this call started +/// a run. The run reports how it ended to the supervisor, which is what follows up a failed +/// recovery. +pub fn begin_with(app: &AppHandle, mode: Mode) -> bool { let Some(startup) = app.try_state::() else { - return; + return false; }; if startup .running .compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire) .is_err() { - return; + return false; } + startup + .recovering + .store(mode == Mode::Recover, Ordering::Release); startup.restart(); let generation = startup.generation.fetch_add(1, Ordering::AcqRel) + 1; let started = Instant::now(); @@ -668,10 +744,26 @@ pub fn begin(app: &AppHandle) { generation, "the startup sequence ended without reporting a result".to_owned(), ); + // Read before the flag drops, so a retry that starts at once cannot answer for this run. + let mut outcome = None; + let mut held = None; if let Some(startup) = app.try_state::() { + outcome = Some(startup.latest()); + held = startup.take_held(); startup.running.store(false, Ordering::Release); } + let ready = outcome + .as_ref() + .is_some_and(|progress| progress.phase == Phase::Ready.id()); + let detail = outcome.and_then(|progress| progress.detail); + let ended = match (ready, held) { + (true, _) => crate::supervisor::RunOutcome::Ready, + (false, Some(held)) => crate::supervisor::RunOutcome::Held(held), + (false, None) => crate::supervisor::RunOutcome::Failed, + }; + crate::supervisor::run_finished(&app, mode, ended, detail.as_deref()); }); + true } /// Report a terminal state for a run that did not report one itself. @@ -783,7 +875,10 @@ async fn run(app: &AppHandle, started: Instant) { Some(install_id) => ownership::consent(&answer.ownership, install_id), None => ownership::Consent::Refuse, }; - match attach_plan(consent, &answer.takeover) { + let mode = app + .try_state::() + .map_or(Mode::Launch, |startup| startup.mode()); + match attach_plan(consent, &answer.takeover, mode) { AttachPlan::Guest(detail) => { attach_as_guest( app, @@ -869,9 +964,31 @@ async fn run(app: &AppHandle, started: Instant) { } } } + // A Child's client runtime. It is never taken over, so there is nothing to ask: attach to + // it, and `bind` makes that ownership when it is the child this app started. + resolve::LiveVerdict::Client => { + attach_as_guest( + app, + started, + &target, + ®istration, + &watch, + &proxy, + endpoint, + deadline, + "a Child's client runtime is listening; it serves Codex through its Home and the Child's dashboard, so this app attached to it and asked nothing" + .to_owned(), + ) + .await; + return; + } // Something holds the port and this app cannot manage it. That is not an absence, so it - // does not authorise starting a second runtime beside it either. + // does not authorise starting a second runtime beside it either. Another attempt would + // find the same listener, which the supervisor is told so it waits for a change instead. resolve::LiveVerdict::Unusable(reason) => { + if let Some(startup) = app.try_state::() { + startup.note_held(answer.liveness.pid); + } fail( app, started, @@ -900,12 +1017,13 @@ async fn run(app: &AppHandle, started: Instant) { return; } - // A retry must not leave a second proxy behind. A child that has not reported an exit is still - // out there, whatever the last run concluded, so the retry waits on that one rather than - // starting another and racing it for the port. + // A retry or a recovery must not leave a second proxy behind. A child that has not reported an + // exit is still out there, whatever the last run concluded, so the run waits on that one rather + // than starting another and racing it for the port. This reads the child the app tracks, not + // the ownership confirmation: `attach` above has just reset that, which left the wait dead. let owns_live_child = app .try_state::() - .is_some_and(|state| state.owns_runtime()) + .is_some_and(|state| waits_on_child(state.child_age())) && watch.exit().is_none(); if owns_live_child { report( @@ -995,7 +1113,11 @@ enum AttachPlan { Ask, } -fn attach_plan(consent: ownership::Consent, takeover: &resolve::Takeover) -> AttachPlan { +fn attach_plan( + consent: ownership::Consent, + takeover: &resolve::Takeover, + mode: Mode, +) -> AttachPlan { match consent { ownership::Consent::Held => AttachPlan::Guest( "a runtime was already listening and this installation already owns it".to_owned(), @@ -1007,6 +1129,11 @@ fn attach_plan(consent: ownership::Consent, takeover: &resolve::Takeover) -> Att resolve::Takeover::Blocked { reason, detail } => AttachPlan::Guest(format!( "a runtime was already listening, but taking it over is not available ({reason}: {detail}), so this app is a guest on it" )), + // A recovery runs with nobody watching; a prompt would surface a window the person + // never asked for, over a runtime that is serving. It stays a guest instead. + resolve::Takeover::Supported { .. } if mode == Mode::Recover => AttachPlan::Guest( + "a runtime was already listening when this app came back for its own, so the recovery attached as a guest and asked nothing".to_owned(), + ), resolve::Takeover::Supported { .. } => AttachPlan::Ask, }, } @@ -1431,6 +1558,12 @@ fn finish(app: &AppHandle, started: Instant, endpoint: ProxyEndpoint) { let requested = startup .as_ref() .is_some_and(|startup| startup.dashboard_requested()); + let mode = startup + .as_ref() + .map_or(Mode::Launch, |startup| startup.mode()); + if keeps_update_page(mode, crate::window::shows_update_page(&window)) { + return; + } if loads_dashboard_on_ready(LaunchOrigin::detect(), visible, requested) { match startup { Some(startup) => { @@ -1492,6 +1625,15 @@ fn loads_dashboard_on_ready(origin: LaunchOrigin, window_visible: bool, requeste origin == LaunchOrigin::User || window_visible || requested } +/// Whether a Ready run leaves the update page on screen instead of loading the dashboard. +/// +/// A recovery nobody asked to watch lands wherever the window is. When that is the update page — +/// a failed install brings the runtime back from there — the page is showing why the install +/// failed, and its own button returns to the dashboard. +fn keeps_update_page(mode: Mode, on_update_page: bool) -> bool { + mode == Mode::Recover && on_update_page +} + /// Perform this run's single dashboard navigation through `navigate`. /// /// `navigate` reports whether the WebView accepted the script. Acceptance is not proof that the @@ -1620,10 +1762,11 @@ fn elapsed(started: Instant) -> u64 { #[cfg(test)] mod tests { use super::{ - approval_still_current, attach_plan, claim_after_silence, loads_dashboard_on_ready, - navigate_once, return_ready_dashboard, shows_window, stop_after_approval, unavailable, - AttachPlan, ConsentState, Expiry, LaunchOrigin, Phase, Progress, Startup, AUTOSTART_FLAG, - DEADLINE, PHASES, POLL, + approval_still_current, attach_plan, claim_after_silence, keeps_update_page, + loads_dashboard_on_ready, navigate_once, return_ready_dashboard, shows_window, + stop_after_approval, unavailable, waits_on_child, AttachPlan, ConsentState, Expiry, + LaunchOrigin, Mode, Phase, Progress, Startup, AUTOSTART_FLAG, CHILD_START_GRACE, DEADLINE, + PHASES, POLL, }; use crate::claim::ClaimResult; use crate::ownership::{Claim, Consent, Owner, Recorded}; @@ -1810,18 +1953,18 @@ mod tests { fn an_ask_only_arises_when_the_takeover_can_be_taken() { // Held and Refuse never ask, whatever the CLI reported about compatibility. assert!(matches!( - attach_plan(Consent::Held, &supported()), + attach_plan(Consent::Held, &supported(), Mode::Launch), AttachPlan::Guest(_) )); assert!(matches!( - attach_plan(Consent::Refuse, &supported()), + attach_plan(Consent::Refuse, &supported(), Mode::Launch), AttachPlan::Guest(_) )); assert!(matches!( - attach_plan(Consent::AskFirstTime, &supported()), + attach_plan(Consent::AskFirstTime, &supported(), Mode::Launch), AttachPlan::Ask )); - match attach_plan(Consent::AskAgain, &blocked()) { + match attach_plan(Consent::AskAgain, &blocked(), Mode::Launch) { AttachPlan::Guest(detail) => { assert!(detail.contains("managing-cli-unsupported: path uses 2.59.0")) } @@ -1829,6 +1972,26 @@ mod tests { } } + #[test] + fn a_recovery_never_asks_and_attaches_as_a_guest() { + // Nobody is looking at a recovery: a prompt would show a window nobody asked for. + for consent in [Consent::AskFirstTime, Consent::AskAgain] { + match attach_plan(consent, &supported(), Mode::Recover) { + AttachPlan::Guest(detail) => assert!(detail.contains("guest")), + AttachPlan::Ask => panic!("a recovery must not prompt"), + } + // A launch still asks. + assert!(matches!( + attach_plan(consent, &supported(), Mode::Launch), + AttachPlan::Ask + )); + } + let startup = Startup::new(); + assert_eq!(startup.mode(), Mode::Launch); + startup.recovering.store(true, Ordering::SeqCst); + assert_eq!(startup.mode(), Mode::Recover); + } + #[test] fn not_having_started_is_not_a_step_of_the_run() { // A checklist row for it would be a step that never completes, and resolving it out of a @@ -1927,6 +2090,27 @@ mod tests { )); } + #[test] + fn a_recovery_leaves_the_update_page_where_it_is() { + // A failed install brings the runtime back while the page shows why the install failed. + assert!(keeps_update_page(Mode::Recover, true)); + // Anywhere else a recovery reloads the dashboard, and a launch always moves on. + assert!(!keeps_update_page(Mode::Recover, false)); + assert!(!keeps_update_page(Mode::Launch, true)); + assert!(!keeps_update_page(Mode::Launch, false)); + } + + #[test] + fn a_run_waits_on_a_child_still_starting_and_not_on_one_that_never_will() { + // The app tracks no child: nothing to wait on. + assert!(!waits_on_child(None)); + // A child spawned moments ago, or one still inside the slowest good start, is waited on. + assert!(waits_on_child(Some(Duration::ZERO))); + assert!(waits_on_child(Some(Duration::from_secs(65)))); + // Past the grace it is wedged, or its exit event is held up: a start goes ahead. + assert!(!waits_on_child(Some(CHILD_START_GRACE))); + } + #[test] fn explicit_dashboard_navigation_is_consumed_once_per_run() { let startup = Startup::new(); diff --git a/desktop/src-tauri/src/supervisor.rs b/desktop/src-tauri/src/supervisor.rs new file mode 100644 index 00000000000..430686ee332 --- /dev/null +++ b/desktop/src-tauri/src/supervisor.rs @@ -0,0 +1,896 @@ +//! Keeping the runtime alive: what the app does when the runtime it runs goes away unasked. +//! +//! The startup sequence used to be the only thing that ever started a runtime, and it ran at +//! launch and on the failure page's retry. A runtime that exited afterwards — a crash, a terminal +//! `ocx stop`, or a restart the runtime performed itself by handing the port to a detached +//! grandchild this app could not see — was recorded in memory and nothing more. The port kept +//! refusing connections until somebody quit and reopened the app. +//! +//! Two signals start a recovery here, and both only reach the same startup sequence in +//! [`Mode::Recover`], so every gate it has still holds: only a proven absence starts a runtime, a +//! runtime that answers is attached as a guest, and nothing is ever killed or signalled. +//! +//! - The exit of the child this app started ([`on_exit`]). Exit code +//! [`REQUESTED_RESTART_EXIT_CODE`] is the runtime asking for exactly this: under the marker +//! `sidecar.rs` sets, a restart (a join into a Child, a memory restart, a disconnect) exits with +//! it instead of spawning past the app, and the replacement starts almost at once. Any other exit +//! waits a capped backoff first, so a replacement or a service wrapper that already owns the +//! port binds before the app looks. +//! - The watchdog ([`watchdog_tick`]), every few seconds while a run is Ready: a different process +//! answering the endpoint, or several refused connections in a row. It covers a runtime this app +//! is only a guest on and an exit event that never arrived. Timeouts and unauthorized or +//! unreadable answers never count; they say something is there. +//! +//! A recovery that finds the port held by a listener this app cannot use (one bound off loopback) +//! is not retried: another attempt would find the same listener, and each one costs a resolve. +//! The supervisor parks instead, and the same watchdog tick asks the endpoint only whether +//! something changed there ([`Parked`]). +//! +//! The tray's Stop, Quit and an update's drain clear the app's wish for a runtime before the +//! runtime's exit can arrive (`exit.rs`), so none of them is ever undone. Every decision is +//! appended to `runtime-supervisor.log` in the app's log directory, bounded, because the exit +//! code otherwise dies with the app. + +use crate::{ + exit::{ExitCoordinator, ExitPhase}, + sidecar::SidecarExit, + startup::{self, Mode, Startup}, + AppState, +}; +use std::{ + fs::{self, OpenOptions}, + io::Write, + path::{Path, PathBuf}, + sync::{Mutex, MutexGuard, PoisonError}, + time::{SystemTime, UNIX_EPOCH}, +}; +use tauri::{AppHandle, Manager}; +use tokio::time::{sleep, Duration, Instant}; + +/// The exit code a runtime ends with to hand its restart to this app +/// (`DESKTOP_RESTART_EXIT_CODE` in `src/lib/system-restart-contract.ts`; EX_TEMPFAIL). +pub const REQUESTED_RESTART_EXIT_CODE: i32 = 75; +/// A requested restart has already released the port, so its replacement starts almost at once. +pub const REQUESTED_RESTART_DELAY: Duration = Duration::from_millis(500); +/// The wait before bringing back a runtime that ended unasked, by how many recoveries ran recently. +/// The first step leaves a replacement or a service wrapper that owns the port time to bind first. +pub const BACKOFF: [Duration; 5] = [ + Duration::from_secs(3), + Duration::from_secs(6), + Duration::from_secs(12), + Duration::from_secs(24), + Duration::from_secs(30), +]; +/// How long a runtime has to stay healthy before the backoff starts over. +pub const HEALTHY_RESET: Duration = Duration::from_secs(120); +/// How often the watchdog asks the endpoint who it is, and only while a run is Ready. +pub const WATCHDOG_INTERVAL: Duration = Duration::from_secs(5); +/// Refused connections in a row before the watchdog treats the runtime this app started as gone. +pub const UNREACHABLE_LIMIT: u32 = 3; +/// The same for a runtime this app is only a guest on. Its own manager (a service, a terminal, an +/// update in progress) gets about a minute to bring it back before the app starts one of its own. +pub const GUEST_UNREACHABLE_LIMIT: u32 = 12; +// A runtime somebody else manages gets its manager's grace; one this app started does not. +const _: () = assert!(UNREACHABLE_LIMIT < GUEST_UNREACHABLE_LIMIT); +/// The supervisor log's cap: at this size it is emptied before the next line. +pub const LOG_CAP_BYTES: u64 = 256 * 1024; +pub const LOG_FILE: &str = "runtime-supervisor.log"; + +/// What an exit is judged on. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct Input { + pub phase: ExitPhase, + pub wanted: bool, + pub reason_set: bool, + pub startup_running: bool, + /// The exit belongs to a child this app no longer tracks. + pub stale_pid: bool, + pub requested_restart: bool, + /// Recoveries scheduled since the runtime was last healthy for [`HEALTHY_RESET`]. + pub attempts: u32, +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Verdict { + /// Leave it: the reason says why. + Ignore(&'static str), + /// A startup run is in flight. It reports how it ended, and a failure is followed up then. + Defer, + /// Bring a runtime back after this long. + RespawnAfter(Duration), +} + +/// Decide what an exit means. Only an exit of the tracked child, with nothing in flight, nobody +/// having asked for the runtime to stop and the app not ending, brings a runtime back. +pub fn decide(input: Input) -> Verdict { + if input.stale_pid { + return Verdict::Ignore("the exit belongs to a runtime this app no longer tracks"); + } + if input.phase != ExitPhase::Idle { + return Verdict::Ignore("the app is starting, stopping, draining or updating its runtime"); + } + if input.reason_set { + return Verdict::Ignore("the app is quitting or restarting"); + } + if !input.wanted { + return Verdict::Ignore("the runtime was stopped from the tray, by Quit or for an update"); + } + if input.startup_running { + return Verdict::Defer; + } + Verdict::RespawnAfter(respawn_delay(input.requested_restart, input.attempts)) +} + +/// A requested restart right after a healthy stretch goes almost at once; everything else, and a +/// requested restart that keeps recurring, waits the capped backoff. +pub fn respawn_delay(requested_restart: bool, attempts: u32) -> Duration { + if requested_restart && attempts == 0 { + return REQUESTED_RESTART_DELAY; + } + let step = usize::try_from(attempts).unwrap_or(usize::MAX); + BACKOFF[step.min(BACKOFF.len() - 1)] +} + +/// The backoff starts over once the runtime has been healthy for [`HEALTHY_RESET`]. +pub fn attempts_after(attempts: u32, healthy_for: Duration) -> u32 { + if healthy_for >= HEALTHY_RESET { + 0 + } else { + attempts + } +} + +/// What one watchdog question established. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Probe { + /// The endpoint identified itself with this pid. + Identified(u32), + /// Nothing is listening: the connection was refused. + Unreachable, + /// A timeout, an unauthorized or unreadable answer, or something that is not this proxy. + /// Something may be there, so it proves nothing. + Inconclusive, +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Watch { + /// The bound runtime answered. + Healthy, + /// Not proven gone; the refused connections in a row so far. + Counting(u32), + /// Bring the runtime back now. + Recover(&'static str), +} + +/// Judge one watchdog answer against the runtime the app is bound to; `owned` says whether this +/// app started it. +pub fn classify(probe: Probe, bound_pid: u32, streak: u32, owned: bool) -> Watch { + let limit = if owned { + UNREACHABLE_LIMIT + } else { + GUEST_UNREACHABLE_LIMIT + }; + match probe { + Probe::Identified(pid) if pid == bound_pid => Watch::Healthy, + Probe::Identified(_) => Watch::Recover("a different process answers the endpoint"), + Probe::Unreachable => { + let streak = streak.saturating_add(1); + if streak >= limit { + Watch::Recover("the endpoint refused every connection the watchdog made") + } else { + Watch::Counting(streak) + } + } + Probe::Inconclusive => Watch::Counting(0), + } +} + +/// A listener this app cannot use held the port when a run looked; its pid, when the resolve named +/// one. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct Held { + pub pid: Option, +} + +/// How a startup run ended, as far as supervision is concerned. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum RunOutcome { + Ready, + /// Anything another attempt may change: a resolve that failed, a start that did not answer. + Failed, + /// The port is held by a listener this app cannot use. Another attempt finds the same one. + Held(Held), +} + +/// What follows a run that did not reach Ready. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum FollowUp { + /// A launch somebody is looking at: their retry decides, as it always has. + Wait, + /// Another attempt after the backoff, if supervision still allows one. + Retry, + /// Watch the endpoint for a change instead of retrying ([`Parked`]). + Park, +} + +/// A failed recovery, or a failed run that swallowed an exit of this app's child, is followed up; +/// any other failed launch waits for the person. A port held by a listener this app cannot use is +/// never retried, because the retry would find the same listener again every time. +pub fn follow_up(mode: Mode, owed: bool, held: bool) -> FollowUp { + if mode != Mode::Recover && !owed { + FollowUp::Wait + } else if held { + FollowUp::Park + } else { + FollowUp::Retry + } +} + +/// A parked supervisor's view of the endpoint after a run found the port held by a listener this +/// app cannot use. Only a change there is worth another recovery: a different process answering, +/// or the listener that answered going silent. A listener that refused loopback from the start is +/// bound where loopback cannot see it, so its refusals are the steady state, not a change. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] +pub struct Parked { + holder: Option, + /// The holder has answered on loopback at least once. + answered: bool, + streak: u32, +} + +impl Parked { + pub fn new(holder: Option) -> Self { + Self { + holder, + ..Self::default() + } + } + + /// Judge one watchdog answer. `Some(reason)` is a change worth another recovery. + pub fn observe(&mut self, probe: Probe) -> Option<&'static str> { + match probe { + Probe::Identified(pid) => { + self.streak = 0; + if self.holder.is_some_and(|holder| holder != pid) { + return Some("a different process answers the endpoint"); + } + self.holder = Some(pid); + self.answered = true; + None + } + Probe::Unreachable if self.answered => { + self.streak = self.streak.saturating_add(1); + // Not ours: its own manager gets the guest's grace to bring it back first. + (self.streak >= GUEST_UNREACHABLE_LIMIT) + .then_some("the listener that held the port stopped answering") + } + Probe::Unreachable => None, + Probe::Inconclusive => { + self.streak = 0; + None + } + } + } +} + +#[derive(Default)] +struct State { + attempts: u32, + /// A recovery is scheduled and has not started yet. One at a time. + pending: bool, + /// An exit arrived while a run was in flight; a failure of that run is followed up. + owed: bool, + streak: u32, + healthy_since: Option, + /// The last run found the port held by a listener this app cannot use. + parked: Option, +} + +/// The supervisor's managed state. +pub struct Supervisor { + state: Mutex, + log: Option, +} + +impl Supervisor { + fn new(log: Option) -> Self { + Self { + state: Mutex::new(State::default()), + log, + } + } + + fn state(&self) -> MutexGuard<'_, State> { + self.state.lock().unwrap_or_else(PoisonError::into_inner) + } + + fn note(&self, message: &str) { + let Some(path) = &self.log else { + return; + }; + if let Err(error) = append_bounded(path, &log_line(SystemTime::now(), message)) { + crate::logging::log_once( + "the runtime supervisor log could not be written", + &error.to_string(), + ); + } + } +} + +/// Register the exit hook and start the watchdog. Called once from setup; it starts no runtime. +pub fn watch_runtime(app: &AppHandle) { + let log = app.path().app_log_dir().ok().map(|dir| dir.join(LOG_FILE)); + app.manage(Supervisor::new(log)); + if let Some(state) = app.try_state::() { + let handle = app.clone(); + state + .watch + .on_exit(move |pid, exit| on_exit(&handle, pid, exit)); + } + let handle = app.clone(); + tauri::async_runtime::spawn(async move { + loop { + sleep(WATCHDOG_INTERVAL).await; + watchdog_tick(&handle).await; + } + }); +} + +/// A person asked for a runtime again: supervision resumes and the backoff starts over. +pub fn resume(app: &AppHandle) { + if let Some(coordinator) = app.try_state::() { + coordinator.resume(); + } + if let Some(supervisor) = app.try_state::() { + let mut state = supervisor.state(); + state.attempts = 0; + state.parked = None; + } +} + +fn input(app: &AppHandle, stale_pid: bool, requested_restart: bool, attempts: u32) -> Input { + let supervision = app + .try_state::() + .map(|coordinator| coordinator.supervision()); + Input { + phase: supervision.map_or(ExitPhase::Draining, |s| s.phase), + wanted: supervision.is_some_and(|s| s.wanted), + reason_set: supervision.map_or(true, |s| s.reason_set), + startup_running: app + .try_state::() + .is_some_and(|startup| startup.is_running()), + stale_pid, + requested_restart, + attempts, + } +} + +/// The child this app started has exited. +fn on_exit(app: &AppHandle, pid: u32, exit: SidecarExit) { + let Some(supervisor) = app.try_state::() else { + return; + }; + let stale = app + .try_state::() + .map_or(true, |state| state.child_pid() != Some(pid)); + let requested = exit.code == Some(REQUESTED_RESTART_EXIT_CODE); + let attempts = supervisor.state().attempts; + let verdict = decide(input(app, stale, requested, attempts)); + supervisor.note(&format!( + "runtime pid {pid} ended ({}); {}", + exit.describe(), + describe(verdict, attempts) + )); + match verdict { + Verdict::Ignore(_) => {} + Verdict::Defer => supervisor.state().owed = true, + Verdict::RespawnAfter(delay) => { + // The child is gone: let go of its handle without signalling anything, so nothing + // later reads it as a runtime this app still owns. + if let Some(state) = app.try_state::() { + state.release(); + } + crate::tray::set_owned(app, false); + schedule(app, delay); + } + } +} + +fn describe(verdict: Verdict, attempts: u32) -> String { + match verdict { + Verdict::Ignore(reason) => format!("left alone: {reason}"), + Verdict::Defer => "a startup run is in flight; its outcome decides".to_owned(), + Verdict::RespawnAfter(delay) => format!( + "bringing a runtime back in {} ms (recovery {})", + delay.as_millis(), + attempts.saturating_add(1) + ), + } +} + +/// Start one recovery after `delay`, unless one is already scheduled. +fn schedule(app: &AppHandle, delay: Duration) { + let Some(supervisor) = app.try_state::() else { + return; + }; + { + let mut state = supervisor.state(); + if state.pending { + return; + } + state.pending = true; + state.attempts = state.attempts.saturating_add(1); + state.streak = 0; + state.healthy_since = None; + state.parked = None; + } + let app = app.clone(); + tauri::async_runtime::spawn(async move { + sleep(delay).await; + let Some(supervisor) = app.try_state::() else { + return; + }; + let attempts = { + let mut state = supervisor.state(); + state.pending = false; + state.attempts + }; + // The world may have moved during the wait: a Stop, a Quit or an update since then wins. + match decide(input(&app, false, false, attempts)) { + Verdict::RespawnAfter(_) => { + if !startup::begin_with(&app, Mode::Recover) { + supervisor.state().owed = true; + } + } + Verdict::Defer => supervisor.state().owed = true, + Verdict::Ignore(reason) => supervisor.note(&format!("recovery skipped: {reason}")), + } + }); +} + +/// A startup run ended. A Ready run starts the healthy clock. A failed run is followed up as +/// [`follow_up`] says: another attempt after the backoff, a parked watch when a listener this app +/// cannot use holds the port, or nothing until the person retries. `detail` is the run's own last +/// word, for the log. +pub fn run_finished(app: &AppHandle, mode: Mode, outcome: RunOutcome, detail: Option<&str>) { + let Some(supervisor) = app.try_state::() else { + return; + }; + let ready = outcome == RunOutcome::Ready; + let (owed, attempts) = { + let mut state = supervisor.state(); + let owed = std::mem::take(&mut state.owed); + state.parked = None; + if ready { + state.streak = 0; + state.healthy_since = Some(Instant::now()); + } + (owed, state.attempts) + }; + if ready { + if mode == Mode::Recover { + supervisor.note("recovery run ready"); + } + return; + } + let held = match outcome { + RunOutcome::Held(held) => Some(held), + RunOutcome::Ready | RunOutcome::Failed => None, + }; + let reason = detail.unwrap_or("no reason reported"); + match follow_up(mode, owed, held.is_some()) { + FollowUp::Wait => {} + FollowUp::Park => { + let holder = held.and_then(|held| held.pid); + supervisor.state().parked = Some(Parked::new(holder)); + supervisor.note(&format!( + "startup run failed ({reason}); the port is held by a listener this app cannot use, so no attempt is scheduled until the endpoint changes" + )); + } + FollowUp::Retry => { + let verdict = decide(input(app, false, false, attempts)); + supervisor.note(&format!( + "startup run failed ({reason}); {}", + describe(verdict, attempts) + )); + if let Verdict::RespawnAfter(delay) = verdict { + schedule(app, delay); + } + } + } +} + +/// The watchdog's one question: who answers the endpoint, if anything. +async fn probe(proxy: &crate::proxy::ProxyClient) -> Probe { + match proxy.identify().await { + Ok(identity) => Probe::Identified(identity.pid), + Err(error) if error.is_unreachable() => Probe::Unreachable, + Err(_) => Probe::Inconclusive, + } +} + +/// One parked watchdog step: a recovery only once the endpoint has changed. +async fn parked_tick(app: &AppHandle, supervisor: &Supervisor) { + let Some(proxy) = app.try_state::().and_then(|state| state.proxy()) else { + return; + }; + let answer = probe(&proxy).await; + let changed = { + let mut state = supervisor.state(); + let Some(parked) = state.parked.as_mut() else { + return; + }; + parked.observe(answer) + }; + let Some(reason) = changed else { + return; + }; + // Re-read after the await: a Stop, a Quit or a run that began meanwhile wins. + let allowed = app + .try_state::() + .is_some_and(|coordinator| coordinator.supervision_allowed()); + let idle = app + .try_state::() + .is_some_and(|startup| !startup.is_running()); + if !allowed || !idle { + return; + } + supervisor.note(&format!( + "watchdog: {reason} on the port that was held; bringing a runtime back" + )); + schedule(app, Duration::ZERO); +} + +/// One watchdog step. It asks only while supervision is allowed and a run is Ready or the +/// supervisor is parked, and it asks the unauthenticated health endpoint, so nothing secret is sent. +async fn watchdog_tick(app: &AppHandle) { + let Some(supervisor) = app.try_state::() else { + return; + }; + let allowed = app + .try_state::() + .is_some_and(|coordinator| coordinator.supervision_allowed()); + let (idle, ready) = app + .try_state::() + .map_or((false, false), |startup| { + (!startup.is_running(), startup.is_ready()) + }); + let parked = { + let mut state = supervisor.state(); + let parked = state.parked.is_some(); + if !allowed || !idle || state.pending || !(ready || parked) { + state.streak = 0; + return; + } + parked && !ready + }; + if parked { + parked_tick(app, &supervisor).await; + return; + } + let Some((proxy, owned)) = app + .try_state::() + .and_then(|state| state.proxy().map(|proxy| (proxy, state.owns_runtime()))) + else { + return; + }; + let Some(binding) = proxy.binding() else { + return; + }; + let answer = probe(&proxy).await; + let verdict = { + let mut state = supervisor.state(); + let verdict = classify(answer, binding.identity.pid, state.streak, owned); + match verdict { + Watch::Healthy => { + state.streak = 0; + let since = *state.healthy_since.get_or_insert_with(Instant::now); + state.attempts = attempts_after(state.attempts, since.elapsed()); + } + Watch::Counting(streak) => state.streak = streak, + Watch::Recover(_) => state.streak = 0, + } + verdict + }; + let Watch::Recover(reason) = verdict else { + return; + }; + // Re-read after the await: a Stop, a Quit or a run that began meanwhile wins. + if !app + .try_state::() + .is_some_and(|coordinator| coordinator.supervision_allowed()) + { + return; + } + supervisor.note(&format!( + "watchdog: {reason} (bound pid {}); bringing a runtime back", + binding.identity.pid + )); + crate::tray::set_owned(app, false); + schedule(app, Duration::ZERO); +} + +/// `2026-09-26T14:33:10Z message\n`, in UTC without a date library. +pub fn log_line(at: SystemTime, message: &str) -> String { + let seconds = at + .duration_since(UNIX_EPOCH) + .map_or(0, |elapsed| elapsed.as_secs()); + let days = i64::try_from(seconds / 86_400).unwrap_or(0); + let rest = seconds % 86_400; + let (year, month, day) = civil_from_days(days); + format!( + "{year:04}-{month:02}-{day:02}T{:02}:{:02}:{:02}Z {message}\n", + rest / 3_600, + rest % 3_600 / 60, + rest % 60 + ) +} + +/// Days since 1970-01-01 to a proleptic Gregorian date (Howard Hinnant's `civil_from_days`). +fn civil_from_days(days: i64) -> (i64, u32, u32) { + let z = days + 719_468; + let era = z.div_euclid(146_097); + let doe = z.rem_euclid(146_097); + let yoe = (doe - doe / 1_460 + doe / 36_524 - doe / 146_096) / 365; + let doy = doe - (365 * yoe + yoe / 4 - yoe / 100); + let mp = (5 * doy + 2) / 153; + let day = u32::try_from(doy - (153 * mp + 2) / 5 + 1).unwrap_or(1); + let month = u32::try_from(if mp < 10 { mp + 3 } else { mp - 9 }).unwrap_or(1); + let year = yoe + era * 400 + i64::from(month <= 2); + (year, month, day) +} + +/// Append one line, emptying the file first once it has reached [`LOG_CAP_BYTES`]. A symlink in +/// its place is refused rather than followed. +pub fn append_bounded(path: &Path, line: &str) -> std::io::Result<()> { + if let Some(parent) = path.parent() { + fs::create_dir_all(parent)?; + } + let existing = match fs::symlink_metadata(path) { + Ok(metadata) if metadata.file_type().is_symlink() => { + return Err(std::io::Error::other("the log path is a symlink")); + } + Ok(metadata) => metadata.len(), + Err(error) if error.kind() == std::io::ErrorKind::NotFound => 0, + Err(error) => return Err(error), + }; + let mut options = OpenOptions::new(); + if existing >= LOG_CAP_BYTES { + options.write(true).truncate(true); + } else { + options.append(true).create(true); + } + let mut file = options.open(path)?; + if existing >= LOG_CAP_BYTES { + file.write_all( + log_line( + SystemTime::now(), + &format!("log emptied at the {} KiB cap", LOG_CAP_BYTES / 1024), + ) + .as_bytes(), + )?; + } + file.write_all(line.as_bytes()) +} + +#[cfg(test)] +mod tests { + use super::{ + append_bounded, attempts_after, classify, decide, follow_up, log_line, respawn_delay, + FollowUp, Input, Parked, Probe, Verdict, Watch, BACKOFF, GUEST_UNREACHABLE_LIMIT, + HEALTHY_RESET, LOG_CAP_BYTES, REQUESTED_RESTART_DELAY, REQUESTED_RESTART_EXIT_CODE, + UNREACHABLE_LIMIT, + }; + use crate::exit::ExitPhase; + use crate::startup::Mode; + use std::time::{Duration, UNIX_EPOCH}; + + const PHASES: [ExitPhase; 7] = [ + ExitPhase::Idle, + ExitPhase::Spawning, + ExitPhase::Stopping, + ExitPhase::Draining, + ExitPhase::Drained, + ExitPhase::DrainFailed, + ExitPhase::OwnershipUnknown, + ]; + + fn open() -> Input { + Input { + phase: ExitPhase::Idle, + wanted: true, + reason_set: false, + startup_running: false, + stale_pid: false, + requested_restart: false, + attempts: 0, + } + } + + #[test] + fn only_an_idle_wanted_unclaimed_settled_exit_of_the_tracked_child_brings_a_runtime_back() { + for phase in PHASES { + for wanted in [false, true] { + for reason_set in [false, true] { + for startup_running in [false, true] { + for stale_pid in [false, true] { + let input = Input { + phase, + wanted, + reason_set, + startup_running, + stale_pid, + ..open() + }; + let gates_open = + phase == ExitPhase::Idle && wanted && !reason_set && !stale_pid; + let expected = match (gates_open, startup_running) { + (true, false) => "respawn", + (true, true) => "defer", + (false, _) => "ignore", + }; + let verdict = match decide(input) { + Verdict::RespawnAfter(_) => "respawn", + Verdict::Defer => "defer", + Verdict::Ignore(_) => "ignore", + }; + assert_eq!(verdict, expected, "{input:?}"); + } + } + } + } + } + } + + #[test] + fn a_requested_restart_goes_at_once_and_an_unasked_exit_backs_off() { + assert_eq!(REQUESTED_RESTART_EXIT_CODE, 75); + assert_eq!(respawn_delay(true, 0), REQUESTED_RESTART_DELAY); + assert!(REQUESTED_RESTART_DELAY < Duration::from_secs(1)); + // The first unasked step leaves a replacement or a service wrapper time to bind first. + assert_eq!(respawn_delay(false, 0), Duration::from_secs(3)); + // A requested restart that keeps recurring is not exempt from the backoff. + assert_eq!(respawn_delay(true, 1), BACKOFF[1]); + let mut previous = Duration::ZERO; + for attempts in 0..20 { + let delay = respawn_delay(false, attempts); + assert!(delay >= previous, "the backoff never shrinks"); + assert!(delay <= Duration::from_secs(30), "the backoff is capped"); + previous = delay; + } + assert_eq!(respawn_delay(false, u32::MAX), Duration::from_secs(30)); + assert_eq!( + decide(Input { + requested_restart: true, + ..open() + }), + Verdict::RespawnAfter(REQUESTED_RESTART_DELAY) + ); + } + + #[test] + fn the_backoff_starts_over_only_after_a_healthy_stretch() { + assert_eq!(attempts_after(4, HEALTHY_RESET - Duration::from_secs(1)), 4); + assert_eq!(attempts_after(4, HEALTHY_RESET), 0); + } + + #[test] + fn only_refused_connections_count_and_a_different_pid_recovers_at_once() { + for (owned, limit) in [(true, UNREACHABLE_LIMIT), (false, GUEST_UNREACHABLE_LIMIT)] { + assert_eq!( + classify(Probe::Identified(42), 42, 2, owned), + Watch::Healthy + ); + assert!(matches!( + classify(Probe::Identified(43), 42, 0, owned), + Watch::Recover(_) + )); + let mut streak = 0; + for _ in 1..limit { + match classify(Probe::Unreachable, 42, streak, owned) { + Watch::Counting(next) => streak = next, + other => panic!("recovered too early: {other:?}"), + } + } + // A timeout or an unreadable answer says something may be there: it breaks the run. + assert_eq!( + classify(Probe::Inconclusive, 42, streak, owned), + Watch::Counting(0) + ); + assert!(matches!( + classify(Probe::Unreachable, 42, streak, owned), + Watch::Recover(_) + )); + } + } + + #[test] + fn a_failed_recovery_on_a_held_port_parks_instead_of_scheduling_another() { + // A recovery that found a listener this app cannot use would find it again every time, + // and each attempt costs a resolve: it is never rescheduled. + assert_eq!(follow_up(Mode::Recover, false, true), FollowUp::Park); + assert_eq!(follow_up(Mode::Launch, true, true), FollowUp::Park); + // Anything else another attempt may change keeps its backoff. + assert_eq!(follow_up(Mode::Recover, false, false), FollowUp::Retry); + assert_eq!(follow_up(Mode::Recover, true, false), FollowUp::Retry); + assert_eq!(follow_up(Mode::Launch, true, false), FollowUp::Retry); + // A launch somebody is looking at still waits for their retry. + assert_eq!(follow_up(Mode::Launch, false, false), FollowUp::Wait); + assert_eq!(follow_up(Mode::Launch, false, true), FollowUp::Wait); + } + + #[test] + fn a_parked_supervisor_recovers_only_on_a_change_at_the_endpoint() { + // Bound off loopback: refused from the start, and forever. That is the steady state. + let mut hidden = Parked::new(Some(42)); + for _ in 0..(GUEST_UNREACHABLE_LIMIT * 4) { + assert_eq!(hidden.observe(Probe::Unreachable), None); + } + assert_eq!(hidden.observe(Probe::Inconclusive), None); + // Something this app can use now answers on loopback. + assert!(hidden.observe(Probe::Identified(43)).is_some()); + + // A holder loopback can see: the same pid is no change, a different one is. + let mut seen = Parked::new(Some(42)); + assert_eq!(seen.observe(Probe::Identified(42)), None); + assert!(seen.observe(Probe::Identified(7)).is_some()); + + // It going silent is a change too, after the grace its own manager gets. + let mut silent = Parked::new(None); + assert_eq!(silent.observe(Probe::Identified(42)), None); + for _ in 1..GUEST_UNREACHABLE_LIMIT { + assert_eq!(silent.observe(Probe::Unreachable), None); + } + // A timeout breaks the run of refusals: something may be there. + assert_eq!(silent.observe(Probe::Inconclusive), None); + for _ in 1..GUEST_UNREACHABLE_LIMIT { + assert_eq!(silent.observe(Probe::Unreachable), None); + } + assert!(silent.observe(Probe::Unreachable).is_some()); + } + + #[test] + fn a_log_line_is_utc_iso_and_carries_the_message() { + assert_eq!( + log_line(UNIX_EPOCH, "start"), + "1970-01-01T00:00:00Z start\n" + ); + assert_eq!( + log_line(UNIX_EPOCH + Duration::from_secs(951_782_400), "leap"), + "2000-02-29T00:00:00Z leap\n" + ); + assert_eq!( + log_line(UNIX_EPOCH + Duration::from_secs(1_700_000_000), "x"), + "2023-11-14T22:13:20Z x\n" + ); + } + + #[test] + fn the_log_is_bounded_and_a_symlink_is_not_followed() { + let dir = std::env::temp_dir().join(format!( + "ocx-supervisor-log-{}-{}", + std::process::id(), + uuid::Uuid::new_v4() + )); + let path = dir.join("nested").join("runtime-supervisor.log"); + append_bounded(&path, "first\n").unwrap(); + append_bounded(&path, "second\n").unwrap(); + assert_eq!(std::fs::read_to_string(&path).unwrap(), "first\nsecond\n"); + let cap = usize::try_from(LOG_CAP_BYTES).unwrap(); + std::fs::write(&path, vec![b'x'; cap]).unwrap(); + append_bounded(&path, "after\n").unwrap(); + let text = std::fs::read_to_string(&path).unwrap(); + assert!(text.len() < 1024, "the full log was emptied first"); + assert!(text.contains("log emptied at the 256 KiB cap")); + assert!(text.ends_with("after\n")); + #[cfg(unix)] + { + let target = dir.join("elsewhere"); + let link = dir.join("link.log"); + std::os::unix::fs::symlink(&target, &link).unwrap(); + assert!(append_bounded(&link, "nope\n").is_err()); + assert!(!target.exists()); + } + let _ = std::fs::remove_dir_all(&dir); + } +} diff --git a/desktop/src-tauri/src/updater.rs b/desktop/src-tauri/src/updater.rs index badd66e9431..9fd3595e5be 100644 --- a/desktop/src-tauri/src/updater.rs +++ b/desktop/src-tauri/src/updater.rs @@ -1,4 +1,7 @@ -use crate::{exit::RestartReadiness, logging, tray}; +use crate::{ + exit::{ExitCoordinator, ExitPhase, RestartReadiness}, + logging, tray, +}; use serde::Serialize; use serde_json::to_value; use std::sync::atomic::{AtomicU64, Ordering}; @@ -398,17 +401,41 @@ pub async fn install(app: &AppHandle, update: Update) -> Result<(), String> { // coordinated restart and not a quit — but the coordination has to finish first. let readiness = crate::exit::prepare_restart(app).await; if readiness != RestartReadiness::Ready { + // Not an ending after all: the app goes back to running, so a close hides again, Quit + // works and Install can be retried. A drain somebody else owns is left alone. + recover_after_failed_install(app); return Err(format!( "the update was downloaded but not installed: {}", readiness.describe() )); } - update.install(package).map_err(|error| error.to_string())?; + if let Err(error) = update.install(package) { + // The installer returned a failure. The runtime was stopped for an install that did not + // happen, so the app brings one back instead of sitting drained. + recover_after_failed_install(app); + return Err(error.to_string()); + } // Only reached where the installer returns. On Windows it does not. crate::exit::complete_restart(app) } +/// Hand a failed install back to a running app. True when the drain had already stopped the +/// runtime, so the startup sequence has to bring one back; a drain that failed left it running. +fn after_install_failure(coordinator: &ExitCoordinator) -> bool { + coordinator.abort_restart() == Some(ExitPhase::Drained) +} + +fn recover_after_failed_install(app: &AppHandle) { + let restart = app + .try_state::() + .is_some_and(|coordinator| after_install_failure(&coordinator)); + if restart { + // Recover, not Launch: nobody is waiting on a prompt, and only a proven absence starts one. + crate::startup::begin_with(app, crate::startup::Mode::Recover); + } +} + pub fn update_label(version: &str) -> String { format!("Install update v{version}") } @@ -476,13 +503,38 @@ pub async fn check_and_show(app: &AppHandle) -> Result<(), String> { #[cfg(test)] mod tests { use super::{ - linux_updater_target, update_label, CheckGeneration, DesktopUpdateState, InstallClaim, - UiProjection, + after_install_failure, linux_updater_target, update_label, CheckGeneration, + DesktopUpdateState, InstallClaim, UiProjection, }; + use crate::exit::{DrainVerdict, ExitCoordinator, ExitDecision, ExitReason}; use std::sync::atomic::{AtomicBool, Ordering}; use std::sync::{mpsc, Arc}; use tauri_utils::config::BundleType; + #[test] + fn a_failed_install_restarts_the_runtime_only_when_the_drain_had_stopped_it() { + let coordinator = ExitCoordinator::new(); + coordinator.claim_drain(ExitReason::CoordinatedRestart); + coordinator.finish_drain(DrainVerdict::Drained); + assert!(after_install_failure(&coordinator)); + assert!(coordinator.supervision_allowed()); + assert!(coordinator.begin_spawn()); + + // A drain that failed left the runtime serving: back to running, nothing to start. + let coordinator = ExitCoordinator::new(); + coordinator.claim_drain(ExitReason::CoordinatedRestart); + coordinator.finish_drain(DrainVerdict::Failed); + assert!(!after_install_failure(&coordinator)); + assert!(coordinator.supervision_allowed()); + + // A quit that took the drain is never turned back into a running app. + let coordinator = ExitCoordinator::new(); + coordinator.claim_drain(ExitReason::UserQuit); + coordinator.finish_drain(DrainVerdict::Drained); + assert!(!after_install_failure(&coordinator)); + assert_eq!(coordinator.decision(), ExitDecision::Proceed); + } + #[test] fn desktop_snapshot_serializes_the_bounded_wire_fields() { let state = DesktopUpdateState::new("2.61.0".into()); diff --git a/desktop/src-tauri/src/window.rs b/desktop/src-tauri/src/window.rs index 67a0461e021..1eb2b474653 100644 --- a/desktop/src-tauri/src/window.rs +++ b/desktop/src-tauri/src/window.rs @@ -91,6 +91,11 @@ fn is_update_page_url(url: &Url) -> bool { is_app_origin(url) && url.path() == "/update.html" } +/// Whether the window currently shows the bundled update page. An unreadable URL reads as not. +pub fn shows_update_page(window: &WebviewWindow) -> bool { + window.url().is_ok_and(|url| is_update_page_url(&url)) +} + pub fn show(window: &WebviewWindow) { let _ = window.show(); let _ = window.set_focus(); diff --git a/docs-site/src/content/docs/guides/desktop-app.md b/docs-site/src/content/docs/guides/desktop-app.md index 58c57b929fd..6f57e04ca8a 100644 --- a/docs-site/src/content/docs/guides/desktop-app.md +++ b/docs-site/src/content/docs/guides/desktop-app.md @@ -61,6 +61,30 @@ embedded dashboard and your normal browser. The tray also provides update checks On macOS, closing the dashboard keeps the app running in the menu bar. Open OpenCodex again from Dock or Finder to restore the dashboard without restarting the proxy. +## Keeping the proxy running + +The app keeps the proxy it started running. When that proxy restarts itself — after +**Connect as Child**, a memory restart from the dashboard, or disconnecting a Child — the app +starts the new proxy on the same port, usually within about a second (a few seconds when the proxy +already restarted in the last two minutes), and reloads the open dashboard, so the +tray's Stop still reaches it and Quit still ends it. If the proxy exits without being asked (a +crash, or `ocx stop` from a terminal), the app starts it again after a short delay that grows from +3 to 30 seconds while the proxy keeps failing. If the proxy stops answering on its port, the app +notices within about 15 seconds and recovers the same way; for a proxy it did not start (a +background service, or one you started yourself), it first waits about a minute for that proxy to +come back. While recovering, it never stops or replaces a proxy that something else already runs on +that port; it attaches to that one instead, without asking to take it over. That includes a Child's +proxy that a background service restarted after **Connect as Child**: the app attaches to it and +shows the Child's dashboard. If the port is held by something the app cannot use, such as a proxy +bound to an address other than `127.0.0.1`, the app stops retrying and waits for that to change. A +recovery that finishes while the update page is open leaves that page on screen. + +The tray's **Stop proxy** and **Quit** keep the proxy stopped. The dashboard's own **Stop** button +does not stop a proxy the app runs, because the app would start it again: it says so and changes +nothing. What the app decided and why is +recorded in `runtime-supervisor.log` in the app's log directory (`~/Library/Logs/com.opencodex.desktop` +on macOS). + ## Usage in the tray On macOS and Windows, click the tray icon to open a compact usage window. The tray's @@ -103,7 +127,8 @@ When the Tauri updater finds a newer app version, a blue dot appears on the macO In the desktop app, choose the dashboard's update button to open the app's update page. There you can check again, install a pending signed update, or return to the dashboard. The same install action is available from the tray menu. If installation fails, the -pending update remains available for retry. This page also works on Linux when the +pending update remains available for retry. The app also brings back the proxy it stopped for +the install, so a failed update does not leave Codex without one. This page also works on Linux when the desktop has no tray icon. A normal browser dashboard manages the package installation on that proxy instead. diff --git a/docs-site/src/content/docs/guides/remote-link.md b/docs-site/src/content/docs/guides/remote-link.md index ff8b5af94d2..408629fa6d1 100644 --- a/docs-site/src/content/docs/guides/remote-link.md +++ b/docs-site/src/content/docs/guides/remote-link.md @@ -13,7 +13,7 @@ A machine link connects an OpenCodex **Home** computer to a **Child** computer o - Both computers run macOS or Linux. - The Home dashboard has a full paired session. -Password SSH and Windows are outside the current flow. For a Child-initiated link, open the standalone Child dashboard, choose **Child** → **Find Home**, select the SSH host for Home, check and confirm the host-key fingerprint, then choose **Connect as Child**. The Child must be able to log in to Home with an SSH key (password login is not supported), and `ocx` must be running on Home. The client tunnel port is `1024` or higher. After joining, the Child restarts and connects through Home. The restart can pause new requests for up to a minute while running turns finish, and the Child waits for its own port instead of giving up while the old process releases it. If the Child still cannot start, Codex stays pointed at the Child's port rather than silently switching back to local providers: run `ocx start` on the Child, and check `~/.opencodex/restart-handoff.log` for what the restarted process printed. This option is available only on a standalone runtime. +Password SSH and Windows are outside the current flow. For a Child-initiated link, open the standalone Child dashboard, choose **Child** → **Find Home**, select the SSH host for Home, check and confirm the host-key fingerprint, then choose **Connect as Child**. The Child must be able to log in to Home with an SSH key (password login is not supported), and `ocx` must be running on Home. The client tunnel port is `1024` or higher. After joining, the Child restarts and connects through Home. The restart can pause new requests for up to a minute while running turns finish, and the Child waits for its own port instead of giving up while the old process releases it. If the Child still cannot start, Codex stays pointed at the Child's port rather than silently switching back to local providers: run `ocx start` on the Child, and check `~/.opencodex/restart-handoff.log` for what the restarted process printed. When the Child runs the OpenCodex desktop app, the app starts the restarted proxy itself, keeps retrying it if it fails, and reloads its dashboard; there is nothing to run by hand. This option is available only on a standalone runtime. ## Add a Child from `#remote` diff --git a/docs-site/src/content/docs/guides/web-dashboard.md b/docs-site/src/content/docs/guides/web-dashboard.md index b07256b1730..2e279c4a1ef 100644 --- a/docs-site/src/content/docs/guides/web-dashboard.md +++ b/docs-site/src/content/docs/guides/web-dashboard.md @@ -115,7 +115,7 @@ is visible) and never forces an upstream refresh. | **Logs** | Auto-refresh recent requests with tokens, requested effort and (when available) effective outbound effort, resolved model, provider, status, request id, duration, and error details. The detail view includes the exact reasoning wire field when the adapter emits one. Filter by opaque conversation/session id (when the client sends one) to total tokens and estimated list-price cost for the currently loaded Logs ring. | | **Usage / Debug** | Inspect token-usage coverage and trends. The Usage page's Models table also breaks each model down into input tokens, output tokens, cache hits, cache writes, and cache hit rate; a dash means cache telemetry for that metric is unavailable. Or enable opt-in provider transport and usage-extraction diagnostics. | | **Storage** | Read-only CODEX_HOME disk breakdown (sessions, archives, DBs, attachments). Optional archived cleanup: preview the oldest N%, then quarantine to `CODEX_HOME/.trash` (default) or permanently delete behind an explicit checkbox. **Auto-cleanup policy** is opt-in and **default OFF** (`storageCleanupPolicy.enabled`); configure threshold/target/schedule/mode on the Storage page, or trigger **Run now**. Quarantined entries can be restored from the Storage page (JSONL + threads). Active sessions stay read-only. Cleanup and restore are refused while Codex holds the newest/active `state_*.sqlite` locked. | -| **Stop** | Gracefully stop the proxy and installed background service, restore native Codex, and exit (`POST /api/stop`). On Windows with the Task Scheduler backend the dashboard refuses and asks you to run `ocx stop` instead: that wrapper can respawn the proxy after the task ends, and only a stop running outside this process can verify the restart window before restoring your client config. Nothing is changed when it refuses. | +| **Stop** | Gracefully stop the proxy and installed background service, restore native Codex, and exit (`POST /api/stop`). On Windows with the Task Scheduler backend the dashboard refuses and asks you to run `ocx stop` instead: that wrapper can respawn the proxy after the task ends, and only a stop running outside this process can verify the restart window before restoring your client config. Nothing is changed when it refuses. It also refuses for a proxy the OpenCodex desktop app runs, which the app would start again: use the app's tray **Stop proxy** or **Quit** there. | If some usage records cannot be included, the Usage page, Dashboard, provider workspace, provider catalog, and API key views show a warning even when no readable records remain. Counts, dates, and @@ -365,7 +365,7 @@ The GUI is a thin client over the proxy's JSON management API. Useful endpoints | `POST /api/codex-auth/login` · `GET /api/codex-auth/login-status` | Add a pool account through browser login. | | `GET /api/logs?tail=50&limit=20&offset=0&provider=...&status=5xx` | Read recent request metadata with optional tail, provider, and exact/class status filters. With `limit`/`offset`, paging walks backward from the newest row (`offset=0` returns the latest page). Response shape: `{ timeZone, generatedAt, total, logs }` where `total` is the filtered row count before pagination. | | `GET` / `PUT /api/subagent-models` | Read or set the five featured `spawn_agent` override models. | -| `POST /api/stop` | Stop the proxy/service, restore native Codex, and exit. Refused with `respawnable_service` on the Windows Task Scheduler backend, with `self_unload_service` when this proxy is itself the installed launchd/systemd job, and with `service_state_unknown` when the Task Scheduler state cannot be read; nothing is changed in any of those cases. | +| `POST /api/stop` | Stop the proxy/service, restore native Codex, and exit. Refused with `respawnable_service` on the Windows Task Scheduler backend, with `self_unload_service` when this proxy is itself the installed launchd/systemd job, and with `service_state_unknown` when the Task Scheduler state cannot be read; nothing is changed in any of those cases. A dashboard session is also refused, with `desktop_supervised`, while the OpenCodex desktop app runs the proxy. | :::tip Adding **Ollama Cloud** or another catalog provider from the dashboard copies its text-versus-vision diff --git a/docs-site/src/content/docs/reference/cli/lifecycle.md b/docs-site/src/content/docs/reference/cli/lifecycle.md index 6ae5e1f7119..08e0979cf4d 100644 --- a/docs-site/src/content/docs/reference/cli/lifecycle.md +++ b/docs-site/src/content/docs/reference/cli/lifecycle.md @@ -49,6 +49,11 @@ ocx start --socks5-off Stop the running proxy (by PID), remove the PID file, and restore native Codex. If a managed background service is installed, `ocx stop` also stops it first so it cannot respawn the proxy. +While the OpenCodex desktop app is running, `ocx stop` does not keep the proxy down: the app treats +it as an unexpected exit and starts a proxy again, within seconds for one it started and after about +a minute for one it had only attached to. Use the app's tray **Stop proxy** (for a proxy the app +started) or **Quit** to keep it stopped. For the same reason the dashboard's **Stop** button refuses +with `desktop_supervised`, and changes nothing, for a proxy the app started. The web dashboard's **Stop** button runs the same action (`POST /api/stop`) on every backend except Windows Task Scheduler. There the wrapper can respawn the proxy after the task ends, and only a stop running outside the proxy can verify that restart window before restoring @@ -86,6 +91,8 @@ When the proxy starts its own replacement (no background service supervises it) finished normally, a replacement that exits before it answers is started again up to twice. What the replacement prints goes to `~/.opencodex/restart-handoff.log`, which stays near 256 KiB: a restart empties it once it reaches that size, and the running replacement checks it once a minute. +A proxy the OpenCodex desktop app started does not start its own replacement: it exits, and the app +starts the new proxy on the same port. If a live listener cannot be attested to a runtime PID (including a pre-update proxy), restart fails closed without an `ensure` or stop/start fallback. After confirming ownership, use `ocx stop` then `ocx start` for a standalone proxy. For a service-managed proxy, use `ocx stop` followed by diff --git a/scripts/test-layout/layout.json b/scripts/test-layout/layout.json index 1def8dad126..df37c43f7d7 100644 --- a/scripts/test-layout/layout.json +++ b/scripts/test-layout/layout.json @@ -168,6 +168,7 @@ } }, "explicit": { + "desktop-supervised-restart.test.ts": "clients", "cli-restart-handoff.test.ts": "cli", "restart-replacement.test.ts": "server", "deepseek-artifact-tool-schema.test.ts": "providers", diff --git a/src/cli/restart-handoff.ts b/src/cli/restart-handoff.ts index bccd3a4c9dd..7bb611e10bc 100644 --- a/src/cli/restart-handoff.ts +++ b/src/cli/restart-handoff.ts @@ -16,25 +16,28 @@ * * {@link takeRestartHandoffMarkers} also consumes the handoff-log flag: a replacement whose output * its parent sent to `restart-handoff.log` bounds that file itself from then on - * (`armRestartHandoffLogCap` in `src/server/restart-replacement.ts`). + * (`armRestartHandoffLogCap` in `src/server/restart-replacement.ts`). It consumes the desktop app's + * supervision marker too, so a later restart exits to the app instead of spawning past it. */ import { isProcessAlive } from "../lib/process-control"; -import { takeRestartParentMarker } from "../lib/system-restart-contract"; +import { takeDesktopSupervisedMarker, takeRestartParentMarker } from "../lib/system-restart-contract"; import { armRestartHandoffLogCap, type RestartHandoffLogCapIo } from "../server/restart-replacement"; import { decideStartWithLiveOwner } from "./dispatch"; export { takeRestartParentMarker }; /** - * Consume everything a restarting parent handed this start, before its first probe: arm the - * handoff log's cap when the flag is set, and return the restart-parent pid, honored only for this - * process's real parent. Both markers are removed from `env`, so no later child inherits them. + * Consume everything a parent handed this start, before its first probe: arm the handoff log's cap + * when the flag is set, record the desktop app's supervision, and return the restart-parent pid, + * honored only for this process's real parent. Every marker is removed from `env`, so no later child + * inherits one. */ export function takeRestartHandoffMarkers( env: Record, logCap: RestartHandoffLogCapIo = {}, ): number | null { armRestartHandoffLogCap(env, logCap); + takeDesktopSupervisedMarker(env); return takeRestartParentMarker(env); } diff --git a/src/client/runtime.ts b/src/client/runtime.ts index a2fbaeb16b0..54eec8acd04 100644 --- a/src/client/runtime.ts +++ b/src/client/runtime.ts @@ -2,9 +2,16 @@ import { existsSync } from "node:fs"; import type { Server } from "bun"; import { siblingRuntimeField, withSiblingMarker } from "../codex/sibling-start"; import { loadConfig } from "../config"; -import { removePid, removeRuntimePort, writePid, writeRuntimePort } from "../config/process-state"; +import { removePid, removeRuntimePort, writePid, writeRuntimePort, type RuntimePortState } from "../config/process-state"; import { installCrashGuards } from "../lib/crash-guard"; +import { createLocalAttestationSecret } from "../lib/local-management-attestation"; import { loadServiceTokenFromFile, serviceApiTokenFingerprint } from "../lib/service-secrets"; +import { + DESKTOP_RESTART_EXIT_CODE, + DESKTOP_SUPERVISED_ENV, + DESKTOP_SUPERVISED_PORT_WAIT_MS, + isDesktopSupervised, +} from "../lib/system-restart-contract"; import { findAvailablePort, isAddrInUse, PortUnavailableError, waitForPortAvailable } from "../server/ports"; import type { ReplacementStartRequest } from "../server/restart-replacement"; import type { OcxClientConnectionConfig } from "../types"; @@ -23,6 +30,18 @@ let recycleScheduled = false; * gives `reclaimListenPort` (`src/cli/index.ts`). */ export const LINK_PORT_WAIT_MS = 60_000; + +/** How long a pinned port is re-probed after the reclaim wait before the start gives up on it. */ +export const PINNED_PREFER_RETRY_MS = 5_000; + +/** + * Link mode's port-reclaim budget. Under the desktop app the whole wait (this plus + * {@link PINNED_PREFER_RETRY_MS}) ends inside the app's 30-second startup deadline, so the app sees + * this start either serve or exit instead of giving up on it first. + */ +export function linkPortWaitMs(desktopSupervised: boolean = isDesktopSupervised()): number { + return desktopSupervised ? DESKTOP_SUPERVISED_PORT_WAIT_MS : LINK_PORT_WAIT_MS; +} /** Bind attempts when the port is taken between the free-port probe and `Bun.serve`. */ const LINK_BIND_ATTEMPTS = 3; @@ -36,6 +55,7 @@ export interface StandaloneRecycleIo { spawnReplacement?: (request: ReplacementStartRequest) => Promise; exitProcess?: (code: number) => void; configuredPort?: () => number | undefined; + isDesktopSupervised?: () => boolean; } function cleanup(): void { @@ -43,11 +63,26 @@ function cleanup(): void { removeRuntimePort(process.pid); } +/** + * What this runtime publishes in `runtime-port.json`. The attestation secret is what the desktop app + * reads to authenticate the runtime it started (`desktop/src-tauri/src/auth.rs`); a record without one + * reads as unusable there, the same as a standalone start's record would. + */ +export function clientRuntimeRecord( + pid: number, + port: number, + attestationSecret: string = createLocalAttestationSecret(), +): RuntimePortState { + return { pid, port, hostname: "127.0.0.1", attestationSecret, ...siblingRuntimeField() }; +} + export function standaloneRecycleEnv( env: NodeJS.ProcessEnv, disconnectedTokenFingerprint: string, ): NodeJS.ProcessEnv { const childEnv = { ...env }; + // A detached replacement is not the desktop app's child. + delete childEnv[DESKTOP_SUPERVISED_ENV]; const admissionToken = childEnv.OPENCODEX_API_AUTH_TOKEN?.trim(); if (admissionToken && serviceApiTokenFingerprint(admissionToken) !== disconnectedTokenFingerprint) { // A surviving env token shadows OCX_API_TOKEN_FILE entirely, so nothing below can @@ -109,8 +144,14 @@ export async function recycleStandalone( } cleanup(); // Recycling back to standalone after `ocx disconnect` must actually bring a standalone - // proxy back, under either launch shape. + // proxy back, under every launch shape. // + // Under the desktop app that spawned us: exit 75 and let the app start the standalone + // runtime, so it keeps owning it (tray Stop, Quit) instead of losing it to a detached child. + if ((io.isDesktopSupervised ?? isDesktopSupervised)()) { + exit(DESKTOP_RESTART_EXIT_CODE); + return; + } // Unsupervised: spawn the replacement ourselves and exit 0. // // Supervised (`OCX_SERVICE=1`): do NOT spawn — the supervisor owns the process, and a @@ -167,7 +208,7 @@ export async function bindClientListener( io: ClientRuntimeIo = {}, ): Promise<{ server: Server; port: number }> { const { linkMode } = request; - const portWaitMs = io.portWaitMs ?? LINK_PORT_WAIT_MS; + const portWaitMs = io.portWaitMs ?? (linkMode ? linkPortWaitMs() : LINK_PORT_WAIT_MS); const deadline = Date.now() + portWaitMs; const busy = (cause: unknown) => new Error( `link mode needs port ${request.configuredPort}; free it or change port`, @@ -186,7 +227,7 @@ export async function bindClientListener( }); } port = await findAvailablePort(request.preferred, "127.0.0.1", { - preferRetryMs: request.explicitPort ? 5_000 : 750, + preferRetryMs: request.explicitPort ? PINNED_PREFER_RETRY_MS : 750, preferRetryIntervalMs: 50, allowEphemeralFallback: linkMode ? false : !request.explicitPort, }); @@ -241,7 +282,7 @@ export async function startClientRuntime( supervisor?.start(); installCrashGuards(); writePid(process.pid); - writeRuntimePort({ pid: process.pid, port: boundPort, hostname: "127.0.0.1", ...siblingRuntimeField() }); + writeRuntimePort(clientRuntimeRecord(process.pid, boundPort)); let shuttingDown = false; const shutdown = () => { diff --git a/src/lib/system-restart-contract.ts b/src/lib/system-restart-contract.ts index 1c51c0f0e91..963b721abba 100644 --- a/src/lib/system-restart-contract.ts +++ b/src/lib/system-restart-contract.ts @@ -50,6 +50,74 @@ export function takeRestartParentMarker( return pid; } +/** + * The env var the desktop app sets on the `ocx start` it spawns and waits on + * (`desktop/src-tauri/src/sidecar.rs`). Under it a restart spawns no detached replacement: it marks + * recycling and exits {@link DESKTOP_RESTART_EXIT_CODE}, and the app starts the replacement itself. + * A detached grandchild is a process the app can neither see nor stop, and one that failed to start + * left no proxy until somebody relaunched the app. `handleStart` consumes the marker before its + * first probe, so no child of the runtime inherits it. + */ +export const DESKTOP_SUPERVISED_ENV = "OCX_DESKTOP_SUPERVISED"; +/** EX_TEMPFAIL, "run me again": the desktop's supervisor restarts promptly on exactly this code. */ +export const DESKTOP_RESTART_EXIT_CODE = 75; +/** + * The link-mode port-reclaim budget under the desktop app. With the pinned port's prefer-retry after + * it (`PINNED_PREFER_RETRY_MS` in `src/client/runtime.ts`) a start that cannot bind gives up within + * 25 seconds, inside the app's 30-second startup deadline, so the app sees it exit instead of giving + * up on it while it still waits. + */ +export const DESKTOP_SUPERVISED_PORT_WAIT_MS = 20_000; + +/** The desktop app's pid, recorded when the marker was taken; null when nothing supervises this process. */ +let desktopParentPid: number | null = null; + +/** + * Consume an inherited desktop-supervision marker, removing it from `env` either way. It is recorded + * only against a real parent process, so {@link isDesktopSupervised} can later check that the same + * app is still there. + */ +export function takeDesktopSupervisedMarker(env: RestartEnv, actualParentPid: number = process.ppid): boolean { + const raw = env[DESKTOP_SUPERVISED_ENV]?.trim(); + delete env[DESKTOP_SUPERVISED_ENV]; + desktopParentPid = raw === "1" && Number.isSafeInteger(actualParentPid) && actualParentPid > 1 + ? actualParentPid + : null; + return desktopParentPid !== null; +} + +export interface DesktopSupervisionIo { + parentPid?: () => number; + isAlive?: (pid: number) => boolean; +} + +function processExists(pid: number): boolean { + try { + process.kill(pid, 0); + return true; + } catch (error) { + return (error as NodeJS.ErrnoException | null)?.code === "EPERM"; + } +} + +/** + * Whether the desktop app that started this process is still there to start its replacement: the + * marker was taken, the parent recorded then is still this process's parent, and it is alive. An app + * that crashed leaves the runtime re-parented (POSIX) or its parent dead (Windows), and a restart then + * falls back to the detached replacement instead of exiting into nothing. Read only when a restart + * or a link-mode start needs it, never on the request path. + */ +export function isDesktopSupervised(io: DesktopSupervisionIo = {}): boolean { + if (desktopParentPid === null) return false; + if ((io.parentPid ?? (() => process.ppid))() !== desktopParentPid) return false; + return (io.isAlive ?? processExists)(desktopParentPid); +} + +/** Test seam: forget a marker a test took. */ +export function resetDesktopSupervisionForTests(): void { + desktopParentPid = null; +} + const BASE64URL_256 = /^[A-Za-z0-9_-]{43}$/; export type ExpectedSystemRestartPid = diff --git a/src/server/management-api.ts b/src/server/management-api.ts index a717ab84cba..78de49c9cef 100644 --- a/src/server/management-api.ts +++ b/src/server/management-api.ts @@ -370,7 +370,11 @@ export async function handleManagementAPI( // outcome. This process cannot verify its own post-exit respawn window; only the // receipt-backed parent `ocx stop` can, which is what the deferral exists for. const { deferralMatchesReceipt } = await import("../config/pending-teardown"); - const { deferralHonored, performStopTeardown } = await import("./stop-teardown"); + const { deferralHonored, desktopSupervisedStopRefusal, performStopTeardown } = await import("./stop-teardown"); + // The desktop app would start this proxy again within seconds; refuse before anything is + // touched and point at its tray, whose Stop it honours (#3008 refuses an undone stop the same way). + const desktopRefusal = desktopSupervisedStopRefusal(principal); + if (desktopRefusal) return jsonResponse(desktopRefusal, 409, req, config); const holdsReceipt = deferralHonored(url, deferralMatchesReceipt); // A sibling never runs under a service manager, and the installed service is the live // owner's: asking the manager to stop from here would refuse, or boot the owner's job out. diff --git a/src/server/management/system-restart.ts b/src/server/management/system-restart.ts index 80e322b8b14..55c39fbe789 100644 --- a/src/server/management/system-restart.ts +++ b/src/server/management/system-restart.ts @@ -31,6 +31,10 @@ * routes to the client runtime the next start serves on this port, so the failed * handoff still marks recycling and exit cleanup keeps that routing instead of * silently falling back to native Codex. + * - Desktop-supervised child (the desktop app spawned it with `OCX_DESKTOP_SUPERVISED=1` + * and is still its parent): no spawn. Mark recycle and exit 75; the app sees the exit + * and starts the replacement itself, so it keeps owning, stopping and quitting it. + * Checked before the service rule: that app, not a service manager, is the parent. */ import { acquireTemporaryDrain, @@ -48,7 +52,12 @@ import { readClientConnectionState } from "../../client/state"; import { withSiblingMarker } from "../../codex/sibling-start"; import { readRuntimePort } from "../../config/process-state"; import { spendLedgerRestartEnvironment } from "../../lib/spend-ledger-owner"; -import { MEMORY_DRAIN_RESTART_MS } from "../../lib/system-restart-contract"; +import { + DESKTOP_RESTART_EXIT_CODE, + DESKTOP_SUPERVISED_ENV, + MEMORY_DRAIN_RESTART_MS, + isDesktopSupervised, +} from "../../lib/system-restart-contract"; import { spawnReplacementStart } from "../restart-replacement"; export { MEMORY_DRAIN_RESTART_MS, REPLACEMENT_READY_TIMEOUT_MS } from "../../lib/system-restart-contract"; @@ -61,6 +70,8 @@ export interface SystemRestartIo { /** True when a background service can actually respawn this process after exit(1). */ isServiceViable?: () => boolean; isSupervisedServiceChild?: () => boolean; + /** True when the desktop app that spawned this process starts its replacement after exit 75. */ + isDesktopSupervised?: () => boolean; /** Ordinary start; deadline handoff may defer health until parent exit releases OS locks. */ spawnStart?: (port?: number, waitForHealthBeforeParentExit?: boolean) => void | Promise; /** Idempotent listener close; must settle before an ordinary start is spawned. */ @@ -206,6 +217,18 @@ function keepRoutingForCommittedClient(io: SystemRestartIo): void { if (connected) (io.markRecycling ?? markRecyclingForExit)(); } +/** + * Hand the restart to the desktop app that spawned this process: keep Codex routing for the + * replacement it starts on this port, and exit with the code it restarts on. False when nothing + * supervises this process that way, and then nothing happened. + */ +function handOffToDesktop(io: SystemRestartIo, exitProcess: (code: number) => void): boolean { + if (!(io.isDesktopSupervised ?? isDesktopSupervised)()) return false; + (io.markRecycling ?? markRecyclingForExit)(); + exitProcess(DESKTOP_RESTART_EXIT_CODE); + return true; +} + /** * The replacement's environment before `spawnReplacementStart` adds the restart-parent marker and * the runtime provenance: never under a service marker, a sibling's replacement stays a sibling, @@ -218,6 +241,8 @@ export function replacementStartEnvironment( // A sibling's replacement stays a sibling even if the owner is down while it probes. const sourceEnv: NodeJS.ProcessEnv = withSiblingMarker(process.env); delete sourceEnv.OCX_SERVICE; + // A detached replacement is not the desktop app's child; it must never exit to an app that is not waiting on it. + delete sourceEnv[DESKTOP_SUPERVISED_ENV]; return spendLedgerRestartEnvironment( sourceEnv, waitForHealthBeforeParentExit ? undefined : parentPid, @@ -268,6 +293,7 @@ async function completeDeadlineRestartHandoff( canHandoff: () => boolean = () => true, ): Promise { if (!canHandoff()) return; + if (handOffToDesktop(io, exitProcess)) return; const supervised = (io.isSupervisedServiceChild ?? (() => isSupervisedServiceChild(io)))(); if (supervised) { // Failure-only supervisors ignore exit(0); intentional non-zero triggers respawn. @@ -381,6 +407,7 @@ export function acceptSystemRestart(io: SystemRestartIo = restartIo, admission: // rejects, an accepted restart must still reach replacement or terminal exit. console.warn("Drain-and-restart cleanup failed; continuing terminal restart handoff"); } + if (handOffToDesktop(io, exitProcess)) return; const supervised = (io.isSupervisedServiceChild ?? (() => isSupervisedServiceChild(io)))(); if (supervised) { // Failure-only supervisors ignore exit(0); intentional non-zero triggers respawn. diff --git a/src/server/stop-teardown.ts b/src/server/stop-teardown.ts index 85447a43378..965e1659082 100644 --- a/src/server/stop-teardown.ts +++ b/src/server/stop-teardown.ts @@ -1,6 +1,8 @@ import type { CodexNativeRestoreResult } from "../codex/inject"; import { siblingOfLivePort, siblingSkipMessage } from "../codex/sibling-start"; import { deferralMatchesReceipt } from "../config/pending-teardown"; +import { isDesktopSupervised } from "../lib/system-restart-contract"; +import type { ManagementPrincipal } from "./management-auth"; /** * Shared-teardown decision and execution for `POST /api/stop` (#3008). @@ -45,6 +47,31 @@ export function deferralHonored(url: URL, ownsReceipt: (nonce: string | null) => return ownsReceipt(url.searchParams.get("teardownNonce")); } +export type StopRefusalBody = { success: false; code: string; message: string }; + +/** + * The dashboard's Stop, refused while the desktop app supervises this process. + * + * The app starts its runtime again after any exit it did not ask for + * (`desktop/src-tauri/src/supervisor.rs`), and a dashboard Stop is not the app asking: the proxy + * would restore native Codex and exit, and the app would start it again seconds later and reload + * the dashboard. Like the service-manager refusals it is refused before anything changes, and it + * names what does stop the proxy: the app's tray Stop proxy, or Quit. Only a dashboard session is + * refused. `ocx stop`, which the tray's Stop, Quit and an update's drain all run, authenticates + * with the admin token and is unaffected. Read only on this route, never on the request path. + */ +export function desktopSupervisedStopRefusal( + principal: ManagementPrincipal | null | undefined, + supervised: () => boolean = isDesktopSupervised, +): StopRefusalBody | null { + if (principal !== "gui-session" || !supervised()) return null; + return { + success: false, + code: "desktop_supervised", + message: "The OpenCodex desktop app runs this proxy and starts it again after it exits, so the dashboard does not stop it. Use Stop proxy in the app's tray menu, or quit the app. Nothing was changed.", + }; +} + /** Run (or skip) the shared teardown and describe the outcome truthfully. */ export async function performStopTeardown(url: URL, io: StopTeardownIo = {}): Promise { // Before the deferral check: a sibling owns no shared teardown to perform OR to hand over. diff --git a/structure/desktop-shell.md b/structure/desktop-shell.md index 207f0944db8..1193ba600bb 100644 --- a/structure/desktop-shell.md +++ b/structure/desktop-shell.md @@ -166,7 +166,10 @@ drains it and confirms the child is gone, and only then installs. The order is n pinned updater's Windows installer hands off to the installer process and ends this one, so a restart asked for after `install` is never reached, and the package would be replaced under a runtime still serving out of those files. A drain that did not complete refuses the install and -leaves the update pending. +leaves the update pending. Neither that refusal nor an install that fails after the drain strands +the app: `ExitCoordinator::abort_restart` takes a coordinated restart's settled drain phase back to +idle with no claimed reason, so a close hides again and Quit works, and when the drain had stopped +the runtime the startup sequence brings one back in recovery mode. A quit's drain is never aborted. The Tauri updater also publishes a bounded desktop snapshot over its identity-bound ProxyClient. A random process-session id travels in the embedded dashboard URL, and the dashboard requests GET /api/update/badge?surface=desktop&session=. A normal browser keeps the package badge. The shell posts each updater-state change and a 60-second heartbeat; if the proxy loses the snapshot or the shell stops, the desktop badge becomes unknown after 180 seconds. This display path never installs an update or replaces the signed Tauri result. The tray shows the same pending state: macOS draws a blue child NSView dot over the template status-item image; Windows/Linux swap a generated dotted PNG when a tray host exists. The Windows base glyph is unchanged. @@ -206,6 +209,68 @@ main thread while holding the menu mutex, so the handles are copied out from und any setter is called. Holding it across a setter is a cycle, and the symptom would be an app that stops answering Quit. +## Keeping the runtime alive + +`desktop/src-tauri/src/supervisor.rs` brings back a runtime that went away without the app asking. +The startup sequence used to run only at launch and from the failure page's retry, so a runtime that +exited later — a crash, a terminal `ocx stop`, or a restart the runtime carried out by handing the +port to a detached grandchild the app could not see — left the port refusing connections until the +app was quit and reopened. + +The sidecar is spawned with `OCX_DESKTOP_SUPERVISED=1`. Under it the runtime's own restarts — a join +into a Child, a memory or package restart, the recycle after a disconnect — exit 75 instead of +spawning a replacement; the drain-and-restart marks recycling first, so exit cleanup keeps Codex +routing ([restart handoff](ops/service-and-sidecars.md#restart-handoff)). +The runtime honors the marker only while the app that set it is still its parent. A link-mode client +runtime gives up on a busy port within 25 seconds under it, inside the 30-second startup deadline, and +publishes an attestation secret in `runtime-port.json` like a standalone start, so the app can +authenticate the runtime it started. + +`sidecar.rs` reports each child's exit to the supervisor once the exit is recorded. The pure `decide` +brings a runtime back only when the exit belongs to the tracked child, the exit coordinator is idle, +the app still wants a runtime and no ending is claimed; while a startup run is in flight, that run's +outcome decides instead. Exit 75 goes after half a second when no recovery has run since the last +120 healthy seconds; otherwise it takes the next backoff step like any other exit. Any other exit +waits 3, 6, 12, 24 and then 30 seconds as recoveries repeat, and the count starts over after 120 +healthy seconds; the first step leaves a replacement or a service wrapper that owns the port time to +bind first. A recovery drops the dead child's handle without signalling anything and runs the startup +sequence in `Mode::Recover`: resolve is still the only authority, only a proven absence starts a +runtime, and a runtime that answers is attached as a guest. A recovery never shows the window or the +takeover prompt, and one that finishes while the window shows the update page leaves that page up. A +failed recovery, or a failed run that swallowed an exit of this app's child, schedules the next +attempt; any other failed launch still waits for the person's retry. The exception is a run that +found the port held by a listener this app cannot use (one bound off loopback): another attempt would +find the same listener, so the supervisor parks, and the watchdog below only asks whether the +endpoint changed — a different process answering, or a holder that had answered going silent for +about a minute. + +A Child's client runtime (`role: client` in the resolve answer) is attached to the same way, at +launch and in a recovery, and never offered a takeover. It serves Codex and the Child's dashboard, +not the management plane, so the tray's usage reads have nothing to show on it. When it is the child +this app started, `bind` confirms ownership, so the tray's Stop and Quit reach it. A run also never +spawns beside the child it already tracks: while that child has reported no exit and was spawned +under 90 seconds ago (a 60-second port reclaim plus its retry fits), the run waits on it. Past that +it is wedged, or its exit event is held up by a grandchild that kept its output pipes (the shell +plugin reports an exit only once both close), and a start goes ahead. + +A watchdog asks `/healthz` every five seconds while a run is Ready and supervision is allowed. A +different pid answering starts a recovery at once. Refused connections start one after three in a row +for a runtime this app started, and after twelve (about a minute) for one it is only a guest on, so a +service or an update restarting its own runtime gets there first. Timeouts and unauthorized or +unreadable answers never count. It covers guest runtimes and an exit event that never arrived. + +The exit coordinator's `wanted` intent keeps this from fighting the person. It is true from launch; +the tray's Stop (when it takes the phase), a quit's drain and an update's drain clear it before the +runtime's exit can arrive, finishing a stop does not restore it, and the failure page's retry sets it +again. A terminal +`ocx stop` of the runtime this app started clears nothing, so the app starts it again after the +backoff; the tray's Stop and Quit keep it stopped. The dashboard's own Stop, in the app's window or +a browser, is refused with `desktop_supervised` while the app supervises the runtime +(`src/server/stop-teardown.ts`): it would be undone within seconds, after a full native-Codex +teardown. Only a dashboard session is refused; `ocx stop` authenticates with the admin token. Every +decision is appended to +`runtime-supervisor.log` in the app's log directory, emptied at 256 KiB, never through a symlink. + ## Runtime ownership, from the app's side `desktop/src-tauri/src/identity.rs` holds this installation's own install id: an opaque value minted diff --git a/structure/gui-and-management-api.md b/structure/gui-and-management-api.md index 6b4ffaf097d..d7d8baaf953 100644 --- a/structure/gui-and-management-api.md +++ b/structure/gui-and-management-api.md @@ -300,12 +300,17 @@ a client-role body: liveness answers "is one of our processes listening here", w orphan cleanup, and duplicate-start avoidance need, and narrowing it would make them blind to a real opencodex process and let them shadow-start over it. Refusing the client role belongs to the caller that needs a management plane, which is the [CLI management client](config.md#management-backed-cli-commands-need-a-management-plane). +The desktop shell needs only the dashboard, so it attaches to a client-role listener as a guest +([desktop shell](desktop-shell.md#keeping-the-runtime-alive)). ## Sidebar stop button The dashboard sidebar includes a stop button that calls `POST /api/stop`. The button shows a confirmation prompt, then fires the request and accepts the connection drop (the proxy exits). The endpoint restores native Codex config, stops any installed service to prevent respawn, and exits. +While the desktop app supervises the proxy it refuses a dashboard session with 409 +`desktop_supervised` before touching anything, because the app would start the proxy again +([desktop shell](desktop-shell.md#keeping-the-runtime-alive)); the button reports the refusal. ## Bun runtime provenance diff --git a/structure/ops/service-and-sidecars.md b/structure/ops/service-and-sidecars.md index 7eea27a9d1a..d6801ad12b5 100644 --- a/structure/ops/service-and-sidecars.md +++ b/structure/ops/service-and-sidecars.md @@ -306,6 +306,17 @@ for its supervisor. Coverage: `tests/server/restart-replacement.test.ts`, `tests/cli/cli-restart-handoff.test.ts`, `tests/server/system-restart.test.ts` and `tests/clients/client-runtime.test.ts`. +A process the desktop app spawned spawns no replacement at all. The app sets +`OCX_DESKTOP_SUPERVISED=1` on its sidecar; `handleStart` consumes it with the other start markers and +records the parent pid (`src/lib/system-restart-contract.ts`). While that parent is still this +process's parent and alive, the drain-and-restart (completed and deadline paths alike) marks +recycling and exits 75, the standalone recycle exits 75 once its cleanup ran, and the app starts the +replacement itself ([desktop shell](../desktop-shell.md#keeping-the-runtime-alive)). This check runs before the service +rule, because that app, not a service manager, is the parent. An app that crashed leaves the runtime +re-parented or its parent dead, and the restart falls back to the detached replacement. Every +detached replacement's environment drops the marker. Coverage: +`tests/clients/desktop-supervised-restart.test.ts`. + ## Package cache refresh src/update/refresh-scheduler.ts owns the package cache timer and per-channel singleflight for the running proxy. Eligible npm, pnpm and Bun installs refresh missing or 20-hour-stale `version.json` after bind, check staleness hourly and retry failures with bounded backoff. Each server start owns one scheduler reference; the last matching stop disarms the timer. A stopped automatic lookup cannot write a late result, but an explicit check joining that lookup marks explicit interest and writes its successful result even if the last listener stops before it resolves. Source/mise installs and `OCX_DISABLE_UPDATE_CHECK=1` do not start automatic lookup; explicit requests remain available. diff --git a/tests/clients/desktop-cli-contracts.test.ts b/tests/clients/desktop-cli-contracts.test.ts index f4ad4ae299b..77113d51f55 100644 --- a/tests/clients/desktop-cli-contracts.test.ts +++ b/tests/clients/desktop-cli-contracts.test.ts @@ -56,15 +56,17 @@ describe("desktop CLI contracts", () => { test("a live listener this app cannot manage is neither attached to nor started beside", () => { // Core's liveness predicate accepts a connected client's listener on purpose, so - // duplicate-start avoidance can see it; a caller that needs the management plane has to - // discriminate on the role rather than narrow that predicate. + // duplicate-start avoidance can see it; the shell discriminates on the role rather than + // narrow that predicate. expect(resolveTs).toContain("role?: string"); const verdict = resolveRs.slice(resolveRs.indexOf("pub fn live_verdict(")); const body = verdict.slice(0, verdict.indexOf("\n}")); expect(body).toContain('role.as_deref() == Some("client")'); expect(body).toContain("LiveVerdict::Unusable"); - // And an address this shell cannot reach on loopback is the same kind of answer. - expect(body).toContain("loopback_reachable(resolved.liveness.hostname.as_deref())"); + // An address this shell cannot reach on loopback is unusable, whatever the role. + const loopback = body.indexOf("loopback_reachable(resolved.liveness.hostname.as_deref())"); + expect(loopback).toBeGreaterThan(-1); + expect(body.indexOf("return LiveVerdict::Client;")).toBeGreaterThan(loopback); const startup = code(STARTUP); const unusable = startup.indexOf("resolve::LiveVerdict::Unusable(reason) =>"); const spawn = startup.indexOf("spawn_runtime(app, endpoint, &watch)"); @@ -72,6 +74,19 @@ describe("desktop CLI contracts", () => { expect(startup.slice(unusable, spawn)).toContain("return;"); }); + test("a Child's client runtime is attached to without a takeover prompt, never started beside", () => { + // Refusing it failed every recovery on a Child whose runtime restarted outside the app. + const startup = code(STARTUP); + const client = startup.indexOf("resolve::LiveVerdict::Client =>"); + const unusable = startup.indexOf("resolve::LiveVerdict::Unusable(reason) =>"); + expect(client).toBeGreaterThan(-1); + const arm = startup.slice(client, unusable); + expect(arm).toContain("attach_as_guest("); + expect(arm).toContain("return;"); + expect(arm).not.toContain("attach_plan("); + expect(arm).not.toContain("await_consent"); + }); + test("everything that can go wrong on this side folds into unknown", () => { const reader = resolveRs.slice(resolveRs.indexOf("pub fn read("), resolveRs.indexOf("pub async fn run(")); // A non-zero exit is the CLI's own refusal, including the exit 1 it uses for unknown liveness. diff --git a/tests/clients/desktop-startup-surface.test.ts b/tests/clients/desktop-startup-surface.test.ts index 693cc528e3d..76cc454bb2a 100644 --- a/tests/clients/desktop-startup-surface.test.ts +++ b/tests/clients/desktop-startup-surface.test.ts @@ -131,6 +131,14 @@ describe("desktop startup surface", () => { const spawn = startup.indexOf("sidecar::start(app, endpoint, watch)"); expect(guard).toBeGreaterThan(-1); expect(spawn).toBeGreaterThan(guard); + // The guard reads the child the app tracks. The ownership confirmation cannot stand in for + // it: the run's `attach` resets that just before, which left the wait unreachable. + const decl = startup.slice(startup.indexOf("let owns_live_child = app"), guard); + expect(decl).toContain("waits_on_child(state.child_age())"); + expect(decl).not.toContain("owns_runtime"); + const attach = startup.indexOf("state.attach(proxy.clone());"); + expect(attach).toBeGreaterThan(-1); + expect(attach).toBeLessThan(guard); }); test("the diagnostic names the state, the endpoint, the home and how the child ended", () => { diff --git a/tests/clients/desktop-supervised-restart.test.ts b/tests/clients/desktop-supervised-restart.test.ts new file mode 100644 index 00000000000..d3ced166fd4 --- /dev/null +++ b/tests/clients/desktop-supervised-restart.test.ts @@ -0,0 +1,280 @@ +/** + * A runtime the desktop app spawned hands its restarts to the app instead of spawning past it. + * + * Before this, a join into a Child, a memory restart and the recycle after a disconnect each spawned + * a detached `ocx start` grandchild and exited. The app only recorded its own child's exit, so the + * replacement was a process it could not see, stop or quit, and one that failed to start left no + * proxy until somebody relaunched the app. Under the marker the desktop sets on its sidecar + * (`desktop/src-tauri/src/sidecar.rs`), the runtime marks recycling and exits 75, and the app's + * supervisor (`desktop/src-tauri/src/supervisor.rs`) starts the replacement itself. + */ +import { afterEach, describe, expect, test } from "bun:test"; +import { mkdtempSync, readFileSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { takeRestartHandoffMarkers } from "../../src/cli/restart-handoff"; +import { + LINK_PORT_WAIT_MS, + PINNED_PREFER_RETRY_MS, + linkPortWaitMs, + recycleStandalone, + standaloneRecycleEnv, +} from "../../src/client/runtime"; +import { getDefaultConfig, saveConfig } from "../../src/config"; +import { serviceApiTokenFingerprint } from "../../src/lib/service-secrets"; +import { + DESKTOP_RESTART_EXIT_CODE, + DESKTOP_SUPERVISED_ENV, + DESKTOP_SUPERVISED_PORT_WAIT_MS, + isDesktopSupervised, + resetDesktopSupervisionForTests, + takeDesktopSupervisedMarker, +} from "../../src/lib/system-restart-contract"; +import { resetLifecycleDrainStateForTests } from "../../src/server/lifecycle"; +import { + MEMORY_DRAIN_RESTART_MS, + acceptSystemRestart, + replacementStartEnvironment, + setSystemRestartIoForTests, + type SystemRestartIo, +} from "../../src/server/management/system-restart"; +import { desktopSupervisedStopRefusal } from "../../src/server/stop-teardown"; +import type { OcxClientConnectionConfig } from "../../src/types"; +import { repoPath } from "../helpers/repo-root"; + +afterEach(() => { + setSystemRestartIoForTests(); + resetLifecycleDrainStateForTests(); + resetDesktopSupervisionForTests(); +}); + +/** One accepted restart, driven to its terminal step; `deadline` never lets the drain settle. */ +async function restartUnder(io: Partial, deadline: boolean): Promise { + const calls: string[] = []; + let scheduled: (() => void | Promise) | null = null; + let fireDeadline: (() => void) | null = null; + let now = 1_000; + acceptSystemRestart({ + isDraining: () => false, + getActiveTurnCount: () => 0, + isSupervisedServiceChild: () => false, + listenPort: () => 10123, + schedule: (fn) => { scheduled = fn; }, + scheduleDeadline: (fn) => { fireDeadline = fn; return () => {}; }, + now: () => now, + setDraining: () => { calls.push("latched"); }, + drainAndShutdown: deadline + ? () => { calls.push("drain"); return new Promise(() => {}); } + : async () => { calls.push("drain"); }, + stopListener: () => { calls.push("stop"); }, + spawnStart: (port) => { calls.push(`start:${port}`); }, + markRecycling: () => { calls.push("recycle"); }, + exitProcess: (code) => { calls.push(`exit:${code}`); }, + ...io, + }); + const running = scheduled!(); + if (deadline) { + await Promise.resolve(); + await Promise.resolve(); + now += MEMORY_DRAIN_RESTART_MS; + fireDeadline?.(); + } + await running; + return calls; +} + +describe("a desktop-supervised drain-and-restart exits to the app", () => { + test("the completed drain marks recycling and exits 75 without spawning a replacement", async () => { + const calls = await restartUnder({ isDesktopSupervised: () => true }, false); + expect(calls).toEqual(["latched", "drain", "recycle", `exit:${DESKTOP_RESTART_EXIT_CODE}`]); + expect(DESKTOP_RESTART_EXIT_CODE).toBe(75); + }); + + test("the deadline path does the same instead of a parent-exit handoff", async () => { + const calls = await restartUnder({ isDesktopSupervised: () => true }, true); + expect(calls).toEqual(["latched", "drain", "recycle", "exit:75"]); + }); + + test("the app that spawned the process wins over a service marker", async () => { + const calls = await restartUnder({ isDesktopSupervised: () => true, isSupervisedServiceChild: () => true }, false); + expect(calls).toEqual(["latched", "drain", "recycle", "exit:75"]); + }); + + test("without the app the detached replacement is unchanged", async () => { + const calls = await restartUnder({ isDesktopSupervised: () => false }, false); + expect(calls).toEqual(["latched", "drain", "stop", "start:10123", "recycle", "exit:0"]); + }); + + test("the marker handleStart took is what the default check reads", async () => { + // The test runner's parent is alive, so a marker taken against it reads as supervised. + const env: Record = { [DESKTOP_SUPERVISED_ENV]: "1" }; + expect(takeDesktopSupervisedMarker(env)).toBe(true); + expect(env[DESKTOP_SUPERVISED_ENV]).toBeUndefined(); + const calls = await restartUnder({}, false); + expect(calls).toEqual(["latched", "drain", "recycle", "exit:75"]); + }); +}); + +describe("the supervision marker", () => { + test("is consumed with the other start markers and honored only against a live parent", () => { + const env: Record = { [DESKTOP_SUPERVISED_ENV]: "1", PATH: "/usr/bin" }; + takeRestartHandoffMarkers(env, { path: join(tmpdir(), "unused-handoff.log") }); + expect(env).toEqual({ PATH: "/usr/bin" }); + expect(isDesktopSupervised()).toBe(true); + + expect(takeDesktopSupervisedMarker({ [DESKTOP_SUPERVISED_ENV]: "1" }, 4242)).toBe(true); + expect(isDesktopSupervised({ parentPid: () => 4242, isAlive: () => true })).toBe(true); + // A crashed app re-parents the runtime (POSIX) or leaves its parent dead (Windows): the + // restart falls back to the detached replacement rather than exiting into nothing. + expect(isDesktopSupervised({ parentPid: () => 1, isAlive: () => true })).toBe(false); + expect(isDesktopSupervised({ parentPid: () => 4242, isAlive: () => false })).toBe(false); + + for (const raw of [undefined, "", "0", "true", " 2"]) { + expect(takeDesktopSupervisedMarker({ [DESKTOP_SUPERVISED_ENV]: raw }, 4242)).toBe(false); + expect(isDesktopSupervised({ parentPid: () => 4242, isAlive: () => true })).toBe(false); + } + // No parent to be supervised by: init, or nothing at all. + expect(takeDesktopSupervisedMarker({ [DESKTOP_SUPERVISED_ENV]: "1" }, 1)).toBe(false); + }); + + test("never reaches a detached replacement's environment", () => { + const prev = process.env[DESKTOP_SUPERVISED_ENV]; + process.env[DESKTOP_SUPERVISED_ENV] = "1"; + try { + expect(replacementStartEnvironment(true)[DESKTOP_SUPERVISED_ENV]).toBeUndefined(); + expect(replacementStartEnvironment(false)[DESKTOP_SUPERVISED_ENV]).toBeUndefined(); + const recycled = standaloneRecycleEnv( + { [DESKTOP_SUPERVISED_ENV]: "1", OPENCODEX_API_AUTH_TOKEN: "operator-token", PATH: "/usr/bin" }, + serviceApiTokenFingerprint("disconnected-hub-token"), + ); + expect(recycled).toEqual({ OPENCODEX_API_AUTH_TOKEN: "operator-token", PATH: "/usr/bin" }); + } finally { + if (prev === undefined) delete process.env[DESKTOP_SUPERVISED_ENV]; + else process.env[DESKTOP_SUPERVISED_ENV] = prev; + } + }); +}); + +describe("the client runtime under the desktop app", () => { + test("the recycle back to standalone exits 75 and spawns nothing", async () => { + const calls: string[] = []; + await recycleStandalone(serviceApiTokenFingerprint("disconnected-hub-token"), { + configuredPort: () => 10456, + isDesktopSupervised: () => true, + spawnReplacement: async () => { calls.push("spawn"); }, + exitProcess: code => { calls.push(`exit:${code}`); }, + }); + expect(calls).toEqual(["exit:75"]); + }); + + test("link mode waits for its port inside the app's 30-second startup deadline", () => { + const startup = readFileSync(repoPath("desktop/src-tauri/src/startup.rs"), "utf8"); + const deadline = Number(/pub const DEADLINE: Duration = Duration::from_secs\((\d+)\);/.exec(startup)?.[1]); + expect(deadline).toBe(30); + expect(linkPortWaitMs(true)).toBe(DESKTOP_SUPERVISED_PORT_WAIT_MS); + // The reclaim wait and the prefer-retry after it, with room left for the spawn and the health wait. + expect(DESKTOP_SUPERVISED_PORT_WAIT_MS + PINNED_PREFER_RETRY_MS).toBeLessThanOrEqual(25_000); + expect(linkPortWaitMs(false)).toBe(LINK_PORT_WAIT_MS); + }); + + test("publishes an attestation secret, so the app can authenticate the runtime it started", async () => { + const home = mkdtempSync(join(tmpdir(), "ocx-client-attest-")); + const holder = Bun.serve({ hostname: "127.0.0.1", port: 0, fetch: () => new Response("") }); + const port = holder.port!; + holder.stop(true); + const previousHome = process.env.OPENCODEX_HOME; + process.env.OPENCODEX_HOME = home; + try { + const token = `ocx_data_${"c".repeat(40)}`; + const client: OcxClientConnectionConfig = { + serverUrl: "http://127.0.0.1:34567", + managementUrl: "http://127.0.0.1:34567", + managementTransport: "direct", + transport: "link", + link: { tunnelPort: 34567, linkId: "lnk_0123456789abcdef" }, + selectedClients: ["codex"], + tokenEnv: "OPENCODEX_API_AUTH_TOKEN", + apiKeyId: "key-1", + tokenFingerprint: serviceApiTokenFingerprint(token), + protocolVersion: 1, + connectedAt: "2026-09-25T00:00:00.000Z", + }; + const config = getDefaultConfig(); + config.port = port; + config.runtimeRole = "client"; + config.client = client; + saveConfig(config); + // A real start, in its own process: it installs signal handlers and crash guards that must + // not leak into this test runner. + const script = [ + `const { startClientRuntime } = await import(${JSON.stringify(repoPath("src/client/runtime.ts"))});`, + `const { readRuntimePort } = await import(${JSON.stringify(repoPath("src/config/process-state.ts"))});`, + "await startClientRuntime({ block: false }, { portWaitMs: 5000 });", + "console.log(JSON.stringify(readRuntimePort(process.pid)));", + "process.exit(0);", + ].join("\n"); + const child = Bun.spawn([process.execPath, "-e", script], { + env: { ...process.env, OPENCODEX_HOME: home }, + stdout: "pipe", + stderr: "pipe", + }); + const [stdout, stderr, exitCode] = await Promise.all([ + new Response(child.stdout).text(), + new Response(child.stderr).text(), + child.exited, + ]); + expect({ exitCode, stderr: exitCode === 0 ? "" : stderr }).toEqual({ exitCode: 0, stderr: "" }); + const lines = stdout.trim().split("\n"); + const record = JSON.parse(lines[lines.length - 1]!) as { port: number; attestationSecret?: string }; + expect(record.port).toBe(port); + expect(record.attestationSecret).toMatch(/^[A-Za-z0-9_-]{43}$/); + } finally { + if (previousHome === undefined) delete process.env.OPENCODEX_HOME; + else process.env.OPENCODEX_HOME = previousHome; + rmSync(home, { recursive: true, force: true }); + } + }); +}); + +describe("the dashboard's Stop under the desktop app", () => { + test("a dashboard session is refused while the app supervises this process", () => { + const refusal = desktopSupervisedStopRefusal("gui-session", () => true); + expect(refusal).toMatchObject({ success: false, code: "desktop_supervised" }); + expect(refusal!.message).toContain("Stop proxy"); + expect(refusal!.message).toContain("Nothing was changed."); + // `ocx stop` — what the tray's Stop, Quit and an update's drain run — keeps working. + expect(desktopSupervisedStopRefusal("admin-token", () => true)).toBeNull(); + expect(desktopSupervisedStopRefusal(undefined, () => true)).toBeNull(); + // Outside the desktop app the dashboard's Stop is unchanged. + expect(desktopSupervisedStopRefusal("gui-session", () => false)).toBeNull(); + // The default reads the marker handleStart took. + expect(desktopSupervisedStopRefusal("gui-session")).toBeNull(); + takeDesktopSupervisedMarker({ [DESKTOP_SUPERVISED_ENV]: "1" }); + expect(desktopSupervisedStopRefusal("gui-session")?.code).toBe("desktop_supervised"); + }); + + test("the route refuses before it touches the service manager or the teardown", () => { + const source = readFileSync(repoPath("src/server/management-api.ts"), "utf8"); + const from = source.indexOf('url.pathname === "/api/stop"'); + const handler = source.slice(from, source.indexOf("/api/native-main-profiles", from)); + const refusal = handler.indexOf("desktopSupervisedStopRefusal(principal)"); + expect(refusal).toBeGreaterThan(-1); + expect(handler.slice(refusal, refusal + 200)).toContain("return jsonResponse(desktopRefusal, 409, req, config)"); + for (const later of ["installedServiceRespawnRisk()", "stopServiceIfInstalledDetailed()", "noteExplicitShutdownRequested()", "performStopTeardown(url"]) { + expect(handler.indexOf(later)).toBeGreaterThan(refusal); + } + }); +}); + +describe("the desktop side speaks the same contract", () => { + test("the sidecar sets the marker and the supervisor restarts on the same exit code", () => { + const sidecar = readFileSync(repoPath("desktop/src-tauri/src/sidecar.rs"), "utf8"); + expect(/pub const SUPERVISED_ENV: &str = "([A-Z_]+)";/.exec(sidecar)?.[1]).toBe(DESKTOP_SUPERVISED_ENV); + // Set on the spawned sidecar itself, next to its other environment. + const start = sidecar.slice(sidecar.indexOf("pub fn start("), sidecar.indexOf("command.spawn()")); + expect(/\.env\(SUPERVISED_ENV, "1"\)/.test(start)).toBe(true); + const supervisor = readFileSync(repoPath("desktop/src-tauri/src/supervisor.rs"), "utf8"); + expect(Number(/pub const REQUESTED_RESTART_EXIT_CODE: i32 = (\d+);/.exec(supervisor)?.[1])) + .toBe(DESKTOP_RESTART_EXIT_CODE); + }); +}); diff --git a/tests/fixtures/test-layout-expected.json b/tests/fixtures/test-layout-expected.json index 93e984671dc..727f6475d98 100644 --- a/tests/fixtures/test-layout-expected.json +++ b/tests/fixtures/test-layout-expected.json @@ -1,4 +1,5 @@ { + "desktop-supervised-restart.test.ts": "clients", "cli-restart-handoff.test.ts": "cli", "restart-replacement.test.ts": "server", "deepseek-artifact-tool-schema.test.ts": "providers",