From dee6862f0cfcceae2efa0b8fc8cde75107e37b6d Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Sat, 29 Aug 2026 18:46:09 -0700 Subject: [PATCH] docs(linux): document orcad update restart safety --- .../headless-serve-shutdown-workflow.test.mjs | 32 ++++++++++ .../orcad-operations-restart-safety.test.mjs | 35 +++++++++++ docs/reference/headless-linux-server.md | 23 ++++++- docs/reference/orcad-operations.md | 60 ++++++++++++------- docs/reference/ssh-execution-boundary.md | 2 +- 5 files changed, 129 insertions(+), 23 deletions(-) create mode 100644 config/scripts/orcad-operations-restart-safety.test.mjs diff --git a/config/scripts/headless-serve-shutdown-workflow.test.mjs b/config/scripts/headless-serve-shutdown-workflow.test.mjs index edfcb028b71..c464f5248f2 100644 --- a/config/scripts/headless-serve-shutdown-workflow.test.mjs +++ b/config/scripts/headless-serve-shutdown-workflow.test.mjs @@ -5,6 +5,7 @@ import { describe, expect, it } from 'vitest' const workflow = parse(readFileSync('.github/workflows/pr.yml', 'utf8')) const headlessLinuxGuide = readFileSync('docs/reference/headless-linux-server.md', 'utf8') +<<<<<<< HEAD const signalCase = readFileSync('config/docker/headless-serve-shutdown/run-signal-case.sh', 'utf8') const shutdownDockerRunner = readFileSync( 'config/scripts/run-headless-serve-shutdown-docker.mjs', @@ -15,6 +16,11 @@ const desktopStartupOracle = readFileSync( 'config/docker/headless-serve-shutdown/run-appimage-desktop-startup-case.sh', 'utf8' ) +const headlessLinuxProse = headlessLinuxGuide.replace(/\s+/g, ' ') +||||||| parent of e24fc472f7f (docs(linux): document orcad update restart safety) +======= +const headlessLinuxProse = headlessLinuxGuide.replace(/\s+/g, ' ') +>>>>>>> e24fc472f7f (docs(linux): document orcad update restart safety) function readSystemdUnitBlocks(doc, unitName) { const escapedUnitName = unitName.replace(/[.*+?^${}()|[\]\\]/g, '\\$&') @@ -169,4 +175,30 @@ describe('headless serve shutdown PR gate', () => { expect(managedXvfbUnits).toHaveLength(1) expect(managedXvfbUnits[0]).not.toMatch(/^KillMode=/m) }) + + it('distinguishes persisted state from live work during a service restart', () => { + expect(headlessLinuxProse).toContain( + 'Every `systemctl stop` or `restart` therefore ends live terminals and agent processes' + ) + expect(headlessLinuxProse).toContain( + 'These guarantees do not preserve live processes. The service restart kills every terminal and agent in its cgroup' + ) + expect(headlessLinuxProse).toContain( + 'Proceed only when it is untruncated, has an explicit `hostScope` covering every expected execution host, has no `omittedHostIds`, and lists no terminals' + ) + expect(headlessLinuxGuide).not.toContain('Two facts make this safe and predictable') + }) + + it('uses the registered CLI name from ordinary Linux shells', () => { + const commandRule = + 'The registered Linux CLI command is `orca-ide`, not `orca`, to avoid shadowing the GNOME Orca screen reader.' + const substitutionRule = + "Bare `orca` is available only through Orca's terminal-scoped shim; from an ordinary shell, substitute `orca-ide` for `orca` in commands below." + + expect(headlessLinuxProse).toContain(commandRule) + expect(headlessLinuxProse).toContain(substitutionRule) + expect(headlessLinuxProse.indexOf(substitutionRule)).toBeLessThan( + headlessLinuxProse.indexOf('`orca terminal list --json`') + ) + }) }) diff --git a/config/scripts/orcad-operations-restart-safety.test.mjs b/config/scripts/orcad-operations-restart-safety.test.mjs new file mode 100644 index 00000000000..cbe5fcb908e --- /dev/null +++ b/config/scripts/orcad-operations-restart-safety.test.mjs @@ -0,0 +1,35 @@ +import { readFileSync } from 'node:fs' + +import { describe, expect, it } from 'vitest' + +const operationsGuide = readFileSync('docs/reference/orcad-operations.md', 'utf8') +const operationsProse = operationsGuide.replace(/\s+/g, ' ') + +describe('orcad operations restart safety', () => { + it('distinguishes PID-scoped preservation from systemd cgroup teardown', () => { + expect(operationsProse).toContain( + 'This makes a PID-scoped update, rollback or restart non-destructive to live work' + ) + expect(operationsProse).toContain( + 'The successor adopts the current endpoint and routes supported previous protocol versions through legacy adapters' + ) + expect(operationsProse).toContain('`KillMode=mixed` does **not** preserve them') + expect(operationsProse).toContain( + '`KillMode=process` leaves service-owned processes unmanaged and is not a supported preservation mechanism' + ) + }) + + it('fails closed before cgroup-wide maintenance', () => { + expect(operationsProse).toContain( + 'A safe empty census is untruncated, has an explicit `hostScope`, covers every expected execution host, has no `omittedHostIds`, and lists no terminals' + ) + expect(operationsProse).toContain( + 'Missing scope, truncation, an omitted host, a failed request or lost contact makes the result `unverifiable`' + ) + expect(operationsProse).toContain('Orca does not yet provide an atomic census-and-stop fence') + }) + + it('does not refer to the unavailable shipping design', () => { + expect(operationsGuide).not.toContain('docs/design/shipping-orcad.html') + }) +}) diff --git a/docs/reference/headless-linux-server.md b/docs/reference/headless-linux-server.md index 9f625a4db4b..0de53054011 100644 --- a/docs/reference/headless-linux-server.md +++ b/docs/reference/headless-linux-server.md @@ -233,6 +233,10 @@ clients should use. `KillMode=mixed` sends the graceful stop signal only to Orca's main process, then retains systemd's cgroup-wide `SIGKILL` fallback if shutdown times out. This lets Orca keep its owned Xvfb alive until Electron disconnects cleanly. +It does **not** preserve the detached terminal daemon: the daemon and its PTYs +remain in `orca-serve.service`'s cgroup and are killed when the stop completes. +Every `systemctl stop` or `restart` therefore ends live terminals and agent +processes, even though their persisted layout and terminal history remain. Exit status `3` means another process already owns this userData profile, so `RestartPreventExitStatus=3` stops the unit instead of retrying a launch that @@ -328,6 +332,11 @@ sudo systemctl enable --now orca-xvfb.service orca-serve.service ## CLI Install Note +The registered Linux CLI command is `orca-ide`, not `orca`, to avoid shadowing +the GNOME Orca screen reader. Bare `orca` is available only through Orca's +terminal-scoped shim; from an ordinary shell, substitute `orca-ide` for `orca` +in commands below. + On a headless host, you do not need to open the desktop UI just to run the server. Invoke the AppImage directly: @@ -380,7 +389,7 @@ at all — the built-in updater only runs in the desktop GUI, and no paired mobi or web client can trigger it remotely. Upgrading is always a deliberate step: replace the AppImage and restart the service. -Two facts make this safe and predictable: +Two facts make the persisted-state transition predictable: - **State lives in the service user's home, not next to the binary.** Persisted data is under `/home/orca/.config/` (Orca uses both an `orca` and an `Orca` @@ -392,6 +401,18 @@ Two facts make this safe and predictable: state into the current schema and writes it back in the current shape, so a forward upgrade needs no manual data step. +These guarantees do not preserve live processes. The service restart kills +every terminal and agent in its cgroup; an agent conversation may be resumable, +but its current process and any in-flight command are gone. + +Immediately before stopping the service, obtain a fresh +`orca terminal list --json` result for this environment. Proceed only when it is +untruncated, has an explicit `hostScope` covering every expected execution host, +has no `omittedHostIds`, and lists no terminals. A missing scope, omitted host, +failed request or lost connection is `unverifiable`, so defer the restart. Do +not allow new work between that census and the stop; Orca does not yet provide +an atomic census-and-stop fence. + Rolling back is the case that needs care — see [Roll back](#roll-back). ### Record the version you deploy diff --git a/docs/reference/orcad-operations.md b/docs/reference/orcad-operations.md index bbde9829514..b647e22342f 100644 --- a/docs/reference/orcad-operations.md +++ b/docs/reference/orcad-operations.md @@ -4,8 +4,6 @@ whatever supervises it: what it binds, what it owns on disk, who restarts what, and what its readiness payload actually proves. -Design background: `docs/design/shipping-orcad.html` §00c and §04. - ## Two long-lived processes, not one A deployment is **orcad** plus **the terminal daemon**. @@ -14,18 +12,22 @@ A deployment is **orcad** plus **the terminal daemon**. | ---------- | -------------------------------- | ------------------------------------- | | Started by | the supervisor | orcad, detached | | Owns | RPC, git, worktrees, persistence | every local PTY | -| Lifetime | one supervised run | **outlives orcad** | +| Lifetime | one supervised run | detached from orcad, not its service | | Endpoint | `ws://:` | `/daemon/daemon-v.sock` | -The daemon outliving orcad is the property the whole peer model is recommended for -(`docs/reference/ssh-execution-boundary.md`): daemon-backed PTYs stay `live` across a runtime -restart, so a restart, an update or a rollback does not destroy running work. Everything -below exists to keep that true. +orcad detaches the daemon and calls `disconnectDaemon()`, never `shutdownDaemon()`. The +built-in remote deployment path stops only the recorded orcad PID, so the daemon and its PTYs +survive. The successor adopts the current endpoint and routes supported previous protocol +versions through legacy adapters. This makes a PID-scoped update, rollback or restart +non-destructive to live work. -**Consequence for supervision:** orcad's shutdown path calls `disconnectDaemon()`, never -`shutdownDaemon()`. A supervisor that reaps orcad's whole process group — systemd's -`KillMode=control-group` — kills the daemon too and turns every restart back into data loss. -Use `KillMode=mixed` (the default) or `process`, and never `--send-sigkill` on the group. +Process detachment is not service isolation. A daemon forked by orcad, and every PTY it owns, +remain in the same systemd service cgroup. `KillMode=mixed` does **not** preserve them: it +sends the graceful stop signal only to the main process, then sends `SIGKILL` to every process +remaining in the cgroup when the stop timeout expires. `KillMode=control-group` is destructive +too. `KillMode=process` leaves service-owned processes unmanaged and is not a supported +preservation mechanism. Service-restart survival requires separately supervised cgroups; the +current deployment does not provide them. ## Bind policy @@ -76,6 +78,19 @@ a live daemon makes worthwhile. ## Supervision +### Process-scoped and cgroup-wide stops + +The built-in remote updater performs a PID-scoped stop and keeps the daemon's install version +pinned while it owns sessions. A combined-unit systemd stop or restart is different: it reaps +the daemon and every live terminal after the graceful window. + +Before a cgroup-wide stop, obtain a fresh `orca terminal list --json` result for the target +environment. A safe empty census is untruncated, has an explicit `hostScope`, covers every +expected execution host, has no `omittedHostIds`, and lists no terminals. Missing scope, +truncation, an omitted host, a failed request or lost contact makes the result `unverifiable`: +defer the stop. Do not admit new work after the census. Orca does not yet provide an atomic +census-and-stop fence. + ### Who supervises orcad An external supervisor (systemd, launchd, a process manager). orcad conforms to it: @@ -127,11 +142,11 @@ An external supervisor (systemd, launchd, a process manager). orcad conforms to ### Decommissioning -The daemon outliving orcad is deliberate, so stopping orcad does **not** leave the host with -zero Orca processes. A daemon that has been adopted stays resident after its runtime -disconnects — that is what makes the next start a reattach rather than a cold restore. To -retire a host completely, stop orcad and then stop the daemon named by -`health.terminalDaemon.pid`, or delete the data root and let the endpoint go stale. +After a PID-scoped stop, an adopted daemon stays resident so the next orcad can reattach. +A combined-unit systemd stop kills it instead. To retire a process-scoped deployment, apply +the census rule above, stop orcad, then stop the daemon named by `health.terminalDaemon.pid`. +Only report it `exited` after verification on the execution host; loss of contact is +`unverifiable`. ## Health @@ -145,7 +160,8 @@ nodeVersion / nodeAbi process.versions.node / .modules — the ABI native add platform / arch / pid terminalDaemon: state live | degraded | absent - ownsFreshSessions whether NEW terminals are daemon-owned, i.e. survive an orcad restart + ownsFreshSessions whether NEW terminals are daemon-owned; this supports PID-scoped + restart recovery, not supervisor or service-cgroup isolation pid the live daemon's pid, from its own PID record buildVersion the build the LIVE daemon was forked from (may legitimately predate this orcad after an update — reporting orcad's version for both would @@ -179,11 +195,13 @@ Named here so nothing reads as implemented that is not: - **A continuous health endpoint.** `health` is published once, in the readiness payload. A supervisor's periodic liveness/readiness probe needs an HTTP or RPC surface over the same `collectOrcadHealth()`; that surface does not exist yet. -- **libc slot.** §04 asks for it in the health payload. It belongs to the native strategy - (plan item 5), which owns libc detection; there is no honest value to publish until then. -- **`degradations[]`.** Plan item 2's contract, not this one. +- **Systemd-isolated daemon supervision.** orcad and its daemon currently share one service + cgroup, so a combined-unit stop cannot preserve live terminals. +- **libc slot.** There is no honest health value to publish until native libc detection owns + it. +- **`degradations[]`.** The readiness contract does not publish this collection yet. - **Credential administration** (list / revoke / rotate devices, expiring pending offers, - structured security logging) — §04, not delivered here. + structured security logging). - **Pinned-port fail-closed.** A pinned `--port` still falls back to an OS-assigned port on conflict. - **Reconciling `webClientUrl` with reachability** under the loopback default. diff --git a/docs/reference/ssh-execution-boundary.md b/docs/reference/ssh-execution-boundary.md index d24cdc38e8e..e31572f2c99 100644 --- a/docs/reference/ssh-execution-boundary.md +++ b/docs/reference/ssh-execution-boundary.md @@ -74,4 +74,4 @@ A listing is only evidence about the hosts it actually covered. When a result do An SSH host and a paired runtime (`orca environment`) imply opposite boundaries: the first is a dumb execution host driven by your client, the second is a peer that owns its own control plane. Registering the same machine both ways splits its worktrees across two identities, makes `terminal list` return different sets depending on `--environment`, and reliably confuses both humans and agents. Pick one per machine. -For work that must continue while you are offline, use the peer/headless-runtime model on the remote host instead of the direct-SSH model. Its control plane is host-local, and its daemon-backed PTYs stay `live` across a normal runtime restart so the runtime can reattach; an explicit daemon shutdown can still make them `exited`. Do not register the same machine through both models. A detached agent process outside Orca can also survive a control-plane outage, but it has no stdin, so its instructions cannot be amended mid-run. +For work that must continue while you are offline, use the peer/headless-runtime model on the remote host instead of the direct-SSH model. Its control plane is host-local, and its daemon-backed PTYs can stay `live` across a PID-scoped runtime restart so the runtime can reattach. A service manager that reaps the runtime's cgroup, or an explicit daemon shutdown, makes them `exited`; see [Running orcad](./orcad-operations.md#process-scoped-and-cgroup-wide-stops). Do not register the same machine through both models. A detached agent process outside Orca can also survive a control-plane outage, but it has no stdin, so its instructions cannot be amended mid-run.