From 853629c3ec0df5d093aaa44772e87df9cf1a6fcc Mon Sep 17 00:00:00 2001 From: Nadia Rodionova <273119990+sosidudku1@users.noreply.github.com> Date: Fri, 2 Oct 2026 17:55:15 +0300 Subject: [PATCH] =?UTF-8?q?Bring=20the=20agent-core=20fixes=20of=20the=20d?= =?UTF-8?q?esktop=20batch=2036=E2=80=9358=20to=20main?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The src/ half of backlog items 36–58 (rel/xp-build integration branch int3/desktop-36-50), moved here first per the main-first rule: - fallback chain: a refusal because the account cannot pay (402; 403/429 with funds, balance or billing words, insufficient_quota) ends the turn in the provider's own words instead of parking on the last link; rate limits stay waits; one wording rule for key refusals; a stopped call never falls over to the next link (item 40, ATO-137) - sessions: a turn is marked running in its row, its end is written however it ends — shutdown included — and marks a dead owner left behind are ended at boot (start time, host and database checked); a store that cannot add the column still boots (ATO-137) - serve: the structured log is written, an orphaned serve mutes a dead stderr instead of exiting, the server's close waits for its connections; transport errnos (ECONNREFUSED…) reach traces and logs (items 41, ATO-123) - local models: the automatic context on unified memory leaves the system max(4 GiB, 25 % of RAM) and caps the KV cache at 1/16 of RAM; --swa-full weighed against the headroom only; one --list-devices per start; the speed probe carried over for the same launch for a day; the llama.cpp release check trusted for 6 hours (items 39, 42) - config: config.json written owner-only (0600); `config set -` reads the whole file from stdin so no key travels on a command line (ATO-132) - skills: browse and search read ClawHub and the GitHub taps side by side (ATO-119 Д45) - logging: a log line carries the start of the model's text, not all of it; the stderr sink factory can no longer pass for a sink The SSE frame changes of the same items stay on the desktop branch, where the HTTP extensions they build on live. --- AGENTS.md | 34 +- eval-memory/harness/consolidator-tick.ts | 4 +- src/agent/agent-loop-cancel-race.test.ts | 204 ++++++ src/agent/agent-loop-lesson-lifecycle.test.ts | 51 ++ src/agent/agent-loop.test.ts | 231 +++++- src/agent/agent-loop.ts | 100 ++- src/agent/step-executor.test.ts | 55 +- src/agent/step-executor.ts | 31 +- src/cli/config-command.test.ts | 67 ++ src/cli/config-command.ts | 36 +- src/cli/config-help.ts | 2 + src/cli/models-handlers.ts | 17 + src/cli/run-agent.ts | 4 +- src/cli/serve-command.test.ts | 85 ++- src/cli/serve-command.ts | 17 +- src/cli/skill-browse.test.ts | 100 +++ src/cli/skill.ts | 16 +- src/cli/trace-formatter.test.ts | 59 ++ src/cli/trace-formatter.ts | 20 +- src/config/config-file.ts | 18 +- src/config/config-paths.ts | 15 +- src/config/config-schema.ts | 5 +- src/config/owner-only-file.test.ts | 125 ++++ src/config/owner-only-file.ts | 56 ++ src/error-reporting/error-reporter.test.ts | 76 +- src/error-reporting/error-reporter.ts | 95 ++- src/error-reporting/index.ts | 4 + src/http/http-server.ts | 63 +- src/http/route-sessions.ts | 4 + src/http/session-status.test.ts | 236 ++++++ src/llm/fallback/failed-links.ts | 25 + src/llm/fallback/link-failure-kind.test.ts | 90 ++- src/llm/fallback/link-failure-kind.ts | 38 +- src/llm/fallback/log-fallback-advance.ts | 14 + src/llm/fallback/partition-state.ts | 9 + .../fallback/provider-fallback-chain.test.ts | 24 + src/llm/fallback/provider-fallback-chain.ts | 11 + .../fallback/run-with-fallback-stop.test.ts | 91 +++ src/llm/fallback/run-with-fallback.test.ts | 175 +++++ src/llm/fallback/run-with-fallback.ts | 48 +- src/llm/provider/openai/openai-http.test.ts | 140 ++++ src/llm/provider/openai/openai-http.ts | 73 +- .../openai/parse-provider-error-body.test.ts | 188 +++++ .../openai/parse-provider-error-body.ts | 122 +++- src/llm/provider/provider-service-name.ts | 28 + .../reliability/provider-wait-cause.test.ts | 29 + src/llm/reliability/provider-wait-cause.ts | 30 + src/local-llm/backend-paths.ts | 10 + src/local-llm/context-size.test.ts | 223 ++++++ src/local-llm/context-size.ts | 167 ++++- src/local-llm/daemon-lifecycle.test.ts | 447 +++++++++++- src/local-llm/daemon-lifecycle.ts | 311 +++++++- src/local-llm/ensure-latest-backend.test.ts | 169 ++++- src/local-llm/ensure-latest-backend.ts | 181 +++++ src/local-llm/gpu-devices.test.ts | 129 +++- src/local-llm/gpu-devices.ts | 66 +- src/local-llm/index.ts | 12 + src/runtime/bootstrap-turn-status.test.ts | 423 +++++++++++ src/runtime/bootstrap.ts | 178 ++++- src/runtime/llm-fallback-seam.ts | 2 + src/runtime/turns-in-flight.test.ts | 78 ++ src/runtime/turns-in-flight.ts | 99 +++ src/session/index.ts | 14 +- src/session/session-retention.ts | 5 + src/session/session-store.test.ts | 6 +- src/session/session-store.ts | 562 ++++++++++++-- src/session/session-store.turns.test.ts | 683 ++++++++++++++++++ src/session/turn-owner.test.ts | 277 +++++++ src/session/turn-owner.ts | 321 ++++++++ src/skills/hub/skill-hub-catalog.test.ts | 87 +++ src/skills/hub/skill-hub-catalog.ts | 122 ++-- src/tools/fusion/worker-result.test.ts | 5 + src/tools/fusion/worker-result.ts | 2 +- src/tracing/index.ts | 3 +- src/tracing/structured-logger.ts | 25 +- src/tracing/trace/trace-bus.ts | 2 +- src/tracing/trace/trace-event.ts | 27 + src/tracing/trace/trace-recorder.test.ts | 183 +++++ src/tracing/trace/trace-recorder.ts | 63 +- src/tracing/tracing.test.ts | 48 +- src/tui/format-provider-outage.ts | 2 + .../local-models/local-models-orchestrator.ts | 10 +- 82 files changed, 7626 insertions(+), 281 deletions(-) create mode 100644 src/agent/agent-loop-cancel-race.test.ts create mode 100644 src/cli/skill-browse.test.ts create mode 100644 src/config/owner-only-file.test.ts create mode 100644 src/config/owner-only-file.ts create mode 100644 src/http/session-status.test.ts create mode 100644 src/llm/fallback/failed-links.ts create mode 100644 src/llm/fallback/run-with-fallback-stop.test.ts create mode 100644 src/llm/provider/provider-service-name.ts create mode 100644 src/runtime/bootstrap-turn-status.test.ts create mode 100644 src/runtime/turns-in-flight.test.ts create mode 100644 src/runtime/turns-in-flight.ts create mode 100644 src/session/session-store.turns.test.ts create mode 100644 src/session/turn-owner.test.ts create mode 100644 src/session/turn-owner.ts diff --git a/AGENTS.md b/AGENTS.md index e25005275..22778be6b 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -233,7 +233,9 @@ two app restarts and a fresh session — with nothing on screen to say the link fallback chain the error the loop sees is the last link's, so a dead key on the primary used to park the turn on a stopped local link's `fetch failed`; the primary's refused key now outranks it unless a fallback has been serving (§"Provider fallback chain", invariant 5). The `provider_waiting` event names the link it - waits on (`providerId`), and the trace row records it. + waits on (`providerId`), and the trace row records it. The event, its trace row (beside `cause`) and the + `provider unreachable; parking the turn` log line also carry `causeCode`, the errno the transport left on the + failure's `cause` chain (`ECONNREFUSED` for a server that is not running). 3. **Backoff is 2s doubling to 30s, clipped so the last wait ends exactly at `maxWaitMs`.** The budget the operator configured is the budget they get. 4. **Esc during a wait ends the turn at once** — the sleep is abort-aware, and the turn settles @@ -468,7 +470,7 @@ The TUI is clickable. Ink has no mouse layer, so this is built in `src/tui/mouse | `src/http/` | OpenAI-compatible HTTP API + atomic admin routes for `atomic-agent serve` | | `src/llm/` | HTTP client for external llama-server + GBNF grammar | | `src/prompt/` | Prompt builder, stable prefix, token budget. See [PROMPT.md](PROMPT.md) for full anatomy of the stable prefix and variable tail. | -| `src/session/` | Session state + sqlite persistence + the startup retention pass (`session-retention.ts`, `session-pins.ts`). See §"Session retention". | +| `src/session/` | Session state + sqlite persistence + the startup retention pass (`session-retention.ts`, `session-pins.ts`) + turn marks (`turn-owner.ts`). See §"Session retention" and §"Turn marks: the row says when a turn is running". | | `src/agent/` | Agent loop + step executor + parallel batch executor (`batch-executor.ts`) + resource-class taxonomy (`tool-resource-class.ts`) + no-progress loop detector | | `src/tools/` | Tool registry + individual tools. OS tools: `shell.run` (direct-exec by default; routes to a `sh -c` subshell when `needsShellInterpretation` sees shell metacharacters `\| & ; > < $ \`` or a pre-joined command line in `cmd` with empty `args` — the common ENOENT trap where the model puts a whole command line in `cmd`; the guard still inspects a tokenised view of the full line so hardline/dangerous rules match), `fs.read` (w/ `offset`/`limit`/`lineNumbers`), `fs.write`, `fs.list`, `fs.glob`, `fs.locate_project` (fuzzy project-name → directory over bounded sources, see §"Project path resolution"), `fs.grep` (bundled ripgrep), `fs.edit` (atomic string replace), `fs.read_document` (PDF/DOCX/XLSX/RTF/ODT/PPTX/legacy .doc → plain text via pure-JS), `fs.archive.list` / `fs.archive.read_entry` / `fs.archive.extract` (zip/tar/tar.gz/gz via pure-JS; zip-slip + bomb guards), `fs.hash` (md5/sha1/sha256/sha512 streaming), `fs.diff` (unified diff, jsdiff), `fs.patch` (dry-run default, all-or-nothing apply), `fs.watch` (chokidar one-shot, timeout-capped), `git.status` / `git.log` / `git.diff` / `git.show` / `git.blame` / `git.branch` (read-only shell-out with structured parse), `git.init` / `git.add` / `git.checkout` / `git.commit` / `git.push` and the network verbs `git.remote` / `git.fetch` / `git.pull` / `git.clone` (writes; local ones ask like a file write in the repository, `commit` forces `-c commit.gpgsign=false` because the agent has no terminal for a pinentry; the network ones need Remote sync on and ask under `git_remote`, see §"GitHub integration"), `proc.list` / `proc.kill` (ps/tasklist + approval), `http.request` (curl + host allowlist + `config.http.approvalMode`), `web.search` (configured provider; keyless Exa with a DuckDuckGo fallback by default, SearXNG/Brave selectable via `web.search.*`; Exa/Brave use an env API key when present, see §"Web search reliability"), `web.fetch` (read a known URL as markdown/text), `clipboard.*`, `window.*`, `notify`, `email.inbox` / `email.send` (the agent's own Atomic Mail inbox; sending is approval-gated, see §"Atomic Mail"). | | `src/compressor/` | Result compressor, log summariser | @@ -510,7 +512,7 @@ Invariants: `ATOMIC_MAIL_API_KEY` is read only in `atomic-mail-store.ts` / the n ## Secrets and process environment -Skills that need API keys (Notion, GitHub, etc.) read them from `process.env`. The agent populates `process.env` once at bootstrap from the optional file `/.env` via `loadDotenvFromStateDir` in [src/config/load-dotenv.ts](src/config/load-dotenv.ts), invoked from [src/config/load-config.ts](src/config/load-config.ts) immediately after `stateDir` is resolved and before `ensureUserConfigFileSync`. Shell-exported variables always win — the loader only sets a key when it is currently unset or empty. Missing file is a silent no-op. The parser is deliberately tiny (`KEY=VALUE` per line, optional surrounding quotes, `#` comments, blank lines; no interpolation, no `export ` prefix, no multiline values) so we do not depend on the `dotenv` package. +Skills that need API keys (Notion, GitHub, etc.) read them from `process.env`. The agent populates `process.env` once at bootstrap from the optional file `/.env` via `loadDotenvFromStateDir` in [src/config/load-dotenv.ts](src/config/load-dotenv.ts), invoked from [src/config/load-config.ts](src/config/load-config.ts) immediately after `stateDir` is resolved and before `ensureUserConfigFileSync`. Shell-exported variables always win — the loader only sets a key when it is currently unset or empty. Missing file is a silent no-op. The parser is deliberately tiny (`KEY=VALUE` per line, optional surrounding quotes, `#` comments, blank lines; no interpolation, no `export ` prefix, no multiline values) so we do not depend on the `dotenv` package. `config.json` is owner-only as well: both of its writers go through `writeOwnerOnlyFileAtomicSync` ([src/config/owner-only-file.ts](src/config/owner-only-file.ts) — tmp file created 0600, renamed over the target), so a file an older build left 0644 is tightened on its next save; it can carry MCP `env` blocks, hand-set provider headers and keys written inline. Pinned by [src/config/owner-only-file.test.ts](src/config/owner-only-file.test.ts). The startup read is defensive about transient locks (#59). A failing read of an existing `.env` is retried up to 3 attempts with 50/150 ms backoff when the errno code is `EPERM`, `EACCES`, `EBUSY`, or `EAGAIN` (the family Windows antivirus and sync clients surface; on POSIX an `EACCES` is almost always permanent and simply costs one ~200 ms loop before the warning). Any other code fails fast, and a missing file (`ENOENT`) stays the silent no-op above. The load outcome travels on the runtime config as `config.dotenv` (`DotenvLoadResult`: the `.env` path, `exists`, `loaded`/`skipped` variable names, and `error` carrying the errno code plus attempt count; values never cross this surface, and parse diagnostics name line numbers, not line content). A failure that survives the retries is printed to stderr by the loader and repeated by the TUI as a warn-variant system chat message with platform-specific guidance, because the stderr line scrolls away before the alt screen takes over. On the write side, after `setDotenvKey` tightens the `.env` ACL via [src/config/windows-acl.ts](src/config/windows-acl.ts) (`icacls /inheritance:r /grant:r`), it probe-reads the file as the current process and rolls the ACL back with `icacls /reset` when the probe fails, so a wrong-principal grant cannot leave behind a file the agent itself can no longer read. Pinned by [src/config/load-dotenv-retry.test.ts](src/config/load-dotenv-retry.test.ts) (retry-then-success, persistent-failure warning, fail-fast codes, silent ENOENT), [src/config/load-config.test.ts](src/config/load-config.test.ts) ("carries the .env load outcome as config.dotenv" / "reports an unreadable .env in config.dotenv.error without throwing"), and [src/config/windows-acl.test.ts](src/config/windows-acl.test.ts) (tighten/probe/rollback). @@ -700,7 +702,7 @@ Pinned by [src/llm/resolve-n-predict.test.ts](src/llm/resolve-n-predict.test.ts) `atomic-agent` supports two modes for the llama-server backend (`localModels.mode`): - `external` (default) — user runs `llama-server` out-of-band; runtime reads the URL from `localModels.url` in `config.json` (no env override). -- `managed` — `atomic-agent` downloads the llama.cpp binary from `AtomicBot-ai/atomic-llama-cpp-turboquant-nightly` GitHub Releases into `/llamacpp/backend/` and GGUF models into `/llamacpp/models//`. The server is **not** spawned by the runtime; operators control lifecycle via `atomic-agent llama start|stop|status|update`. Managed start auto-pulls a newer zip when `localModels.managed.autoUpdate` is true (default since config v41). A failed check or download never blocks start — the existing binary is used. The two entry points differ deliberately: **TUI auto-start** brings the daemon up first and runs the update afterwards, off the start path, so the user never faces a typeable prompt with no model behind it; that pass also refuses to stop the live daemon (`keepDaemonRunning`), so the swap lands on the next start. **CLI `models start`** is an explicit one-shot command, so it still updates before starting. Both bound the download with a timeout. +- `managed` — `atomic-agent` downloads the llama.cpp binary from `AtomicBot-ai/atomic-llama-cpp-turboquant-nightly` GitHub Releases into `/llamacpp/backend/` and GGUF models into `/llamacpp/models//`. The server is **not** spawned by the runtime; operators control lifecycle via `atomic-agent llama start|stop|status|update`. Managed start auto-pulls a newer zip when `localModels.managed.autoUpdate` is true (default since config v41). A failed check or download never blocks start — the existing binary is used. The two entry points differ deliberately: **TUI auto-start** brings the daemon up first and runs the update afterwards, off the start path, so the user never faces a typeable prompt with no model behind it; that pass also refuses to stop the live daemon (`keepDaemonRunning`), so the swap lands on the next start. **CLI `models start`** is an explicit one-shot command, so it still updates before starting. Both bound the download with a timeout. Both also pass `recheckAfterMs` (`AUTO_UPDATE_RECHECK_MS`, 6 h): the last check's answer is kept in `/llama-backend-check.json` (the build it was made for, whether GitHub answered, the newest tag), and a start whose build, asset and wanted variant still match a check that young answers `recent` without asking GitHub; a failed check holds off the next one for `AUTO_UPDATE_RETRY_MS` (15 min), and the TUI's Models panel, which asks GitHub itself, drops the record (`forgetBackendCheck`) when it finds an update, so the next start installs it. The desktop runs a fresh `models start` on every switch to the local model, where the process-wide release cache never survives, so before this every switch put a GitHub round trip (up to the 5 s deadline) in front of the model's load. `models update` and any caller without the option always ask. **Model and backend downloads are parallel, resumable and killable.** `downloadFile` ([src/local-llm/download-file.ts](src/local-llm/download-file.ts), attempt in `download-attempt.ts`, resume planning in `download-resume.ts`, segment workers in `download-segments.ts`) sends one plain GET; when the answer carries `Accept-Ranges: bytes` and a length, the rest of the file is split into up to `localModels.download.connections` (config v52, default 16, env override `ATOMIC_AGENT_DOWNLOAD_CONNECTIONS`) closed `Range` requests written at their final offsets into one `.part`. Hugging Face's CDN throttles per connection (~0.1 MB/s alone vs ~1.8 MB/s on 16 from the same machine), so this is where the speed-up comes from; `1` restores the single stream. `.part.json` (`download-partial.ts`) lists the **byte intervals** on disk under `done` — the file is sparse and out of order, so its size means nothing — and is written atomically every 500 ms plus on every exit path; a resumed run asks only for the holes, bound to the recorded ETag via `If-Range`. The sidecar deliberately stores the URL as `source`, not `url`: a pre-v52 build looks for `url` and would otherwise resume by file size onto a hole and publish a corrupt GGUF. A segment whose body dies re-requests only its remainder while the others keep going; a `206` naming a different file (validator or range mismatch) discards the partial and restarts; a `200` to a segment's `Range` keeps the bytes and continues on one stream. **A throttled connection cannot hold the download.** The stall watchdog only catches silence and is re-armed by every chunk, so a connection the CDN slows to a few hundred B/s never trips it — and once the other slices land it is the whole download (a 2.7 GB GGUF sat at 96% with a 116 MB remainder on one ~400 B/s connection while a fresh one ran at 1.9 MB/s). Two things cover it: every 30 s (`slowCheckMs`, `0` disables) a connection among several is judged against the best per-connection rate this attempt has seen ([download-slow-segment.ts](src/local-llm/download-slow-segment.ts)) — below `1/8` of it and under 256 KiB/s it is dropped with a `SlowSegmentError` (a `StalledError`, so a transport failure) and its remainder re-requested on a new connection at once — without `onRetry`, which the download chip and worker log render as a network outage. Windows are measured on the clock, a window with no bytes is left to the stall watchdog, only downloads with more than one connection are judged, and a segment stops being judged after 3 reconnects in a row that did not help (a reconnect helped when the new connection had a healthy window first, so a CDN that throttles every connection after a fast start is worked around indefinitely while a link slow everywhere costs at most 3 requests per segment); and a connection that finishes its slice with the queue empty takes the back half of the running slice with the most bytes left ([download-rebalance.ts](src/local-llm/download-rebalance.ts), both halves ≥ the 8 MiB floor) — the donor's `end` moves to the cut, which its stream re-reads on every chunk, and a chunk that crosses the cut is clamped out of the donor's count. Small files (< 2 × 8 MiB remaining) and servers without range support take the old single-stream path unchanged. `localModels.download.hfEndpoint` (config v53; `HF_ENDPOINT` env wins, the variable `huggingface_hub` uses) points every Hugging Face request — the repo tree listing in `huggingface-api.ts` and the file downloads — at a mirror; catalogue entries, custom-model definitions and sidecars keep canonical `https://huggingface.co/...` URLs and `rewriteHuggingFaceUrl` (`huggingface-endpoint.ts`) swaps the origin at request time, so a partial survives a mirror switch. Pinned by [src/local-llm/download-file.test.ts](src/local-llm/download-file.test.ts), [src/local-llm/download-file-parallel.test.ts](src/local-llm/download-file-parallel.test.ts), [src/local-llm/download-partial.test.ts](src/local-llm/download-partial.test.ts), [src/local-llm/download-segments.test.ts](src/local-llm/download-segments.test.ts), [src/local-llm/download-slow-segment.test.ts](src/local-llm/download-slow-segment.test.ts) and [src/local-llm/download-rebalance.test.ts](src/local-llm/download-rebalance.test.ts). @@ -1662,6 +1664,26 @@ Rules 2 and 3 have no knob and run whenever the pass does — they describe rows A periodic pass — this is startup-only, so there is still exactly one timer in the runtime (see §"Background autonomy") — and any CLI or TUI surface of its own. A new thing that points at a session by id is one reader in `session-pins.ts`, not a new option on the prune. +## Turn marks: the row says when a turn is running + +A session row used to hold only a turn's end, written by `executeTurn` once the loop returned. A turn that never got there — the app quit mid-turn and the store closed before the aborted turn could save, a process killed outright (the desktop's stop on Windows is `taskkill /F`), a crash — left the row as it was before the turn: a chat whose first message had been cancelled read back as `pending` with nothing in it. So `executeTurn` now marks the row before the loop starts (`SessionStore.beginTurn`): `status = 'running'` plus a `turn_owner` mark, JSON in its own nullable column — `{pid, host, db, hostUptime, processStart, at}`. Only the two columns are written (the payload is not rewritten at every turn start), so every reader takes the status from the column. Every way a turn ends takes the mark off: + +- **Its own end.** `finishTurn` writes the state the loop handed back. +- **A throw.** `releaseTurn` writes `failed` with the error, or `cancelled` (restoring the pre-turn `lastError`) when the turn's signal had aborted — an abort can surface as any error. The loop itself classifies a step error as `cancelled` once its signal has aborted, unless `classifyFailure` answered `tool` (an error it does not recognise, which is how a programming error arrives — that stays a reported failure). +- **Shutdown.** `releaseOwnTurns` writes `cancelled` + `lastError: "turn interrupted: …"` first thing, as a stand-in that keeps the mark, so a stop that turns into a kill partway through teardown still leaves the row right and a turn that still reaches its own end (or throws) replaces it. Then shutdown waits up to `SHUTDOWN_TURN_GRACE_MS` (1.5 s) for the turns their hosts stopped to write their own end (`TurnsInFlight.settleCancelled`; a turn nobody stopped — a scheduled task — is not waited for), releases what is left, and only then closes the store. `serve`'s `handle.close()` resolves only once every connection has closed, so its turns have been aborted by then; the scheduler's ticker stops at the top of shutdown. +- **A process that is gone.** The boot sweep (`recoverInterruptedTurns`, right after `new SessionStore()` and before the retention pass) ends live-status rows whose owner is gone: no mark; a mark written into another database file (`db` — a copy, such as the desktop's "bring your terminal setup over" import); this process's own pid; a dead pid; a host whose uptime went backwards since the mark (a reboot — uptime does not move with the wall clock); a pid whose process started at another moment (`processStart`: `/proc//stat` on Linux, `ps -o lstart=` in UTC on macOS). A mark from another pid namespace (`host`: the platform, plus `/proc/self/ns/pid` on Linux — a container sharing the state dir) is never judged, and a live pid without such evidence is left alone: cancelling a turn another window is still running is the worse mistake. + +Invariants: + +- **The status of a marked row is its turn's.** `save()` never writes a live status, and never changes the status of a row a turn has marked — a copy read before or during the turn (the TUI's model stamp after a session switch) cannot say the session is idle while it runs, or put `running` back after it ended. +- **A failed write does not forget the turn.** `finishTurn` / `releaseTurn` drop their record of a mark only after the write went through, and `releaseOwnTurns` tries every row before rethrowing. +- **An open that cannot add the column still boots.** The first open after the upgrade needs the write lock; when another process holds it past the busy timeout, or the file is one this process may only read, the store runs without marks (`turnMarksUnavailable`, logged at boot) and the next open tries again. +- **A stopped call never falls over.** `runWithFallback` rethrows once the turn's signal has aborted, without arming the link's breaker or setting the sticky override, and a cancelled turn's `loop_failed` frame says `cancelled`, not the chain's earlier failure. + +Pinned by [session-store.turns.test.ts](src/session/session-store.turns.test.ts), [turn-owner.test.ts](src/session/turn-owner.test.ts), [turns-in-flight.test.ts](src/runtime/turns-in-flight.test.ts), [bootstrap-turn-status.test.ts](src/runtime/bootstrap-turn-status.test.ts), [session-status.test.ts](src/http/session-status.test.ts), [stream-cancel-frame.test.ts](src/http/stream-cancel-frame.test.ts), [agent-loop-cancel-race.test.ts](src/agent/agent-loop-cancel-race.test.ts) and [run-with-fallback-stop.test.ts](src/llm/fallback/run-with-fallback-stop.test.ts). + +Not covered: a process killed outright loses what its turn did — the next boot ends the row, but the user's message was never written; and on Windows, where no process start time is read (it would cost a PowerShell start in the turn path), a dead owner's pid reused by a live process keeps the row `running` until a boot after that process has exited, or the next turn on that session. + ## Durable tasks A minimal durable queue of deferred `runTurn` submissions lives in [src/tasks/](src/tasks/). It is the **persistence layer** for any future scheduler / cron / agent-driven self-scheduling — but it ships **without** a background ticker on purpose: drains are always triggered explicitly (CLI `atomic-agent task run`, HTTP `POST /api/tasks/drain`) or implicitly right after `create()` when `tasks.runOnCreate=true` (the default). @@ -2668,7 +2690,7 @@ The fourth control, `workers` ([src/tui/composer-switch/composer-switch-worker-r **Declared inputs are edited in place, never replaced by a worker (F51).** The request-name refusal (§"Refused before dispatch") gives a plain session `overwrite: true` as the way past it; a worker is the wrong party to decide that, since it sees the request only as context. So the contract gains `inputs?: string[]` ([contract-inputs.ts](src/tools/fusion/contract-inputs.ts): paths relative to the working directory or absolute, ≤ 32, trimmed and deduplicated, a bad entry named by index like every other shape problem, an empty list dropped), rendered first in the shared block as `INPUTS (the operator's own files — read and edit in place, never replace; os.fs.write on one is refused):`. The worker runner resolves them as the worker's tools would (globs guard nothing) and declares them for the worker's session in a `DeclaredInputsRegistry` ([fs-declared-inputs.ts](src/tools/os/fs-declared-inputs.ts), the same shape as `FanoutScopeRegistry`, cleared in the same `finally`); `os.fs.write` on a declared input is refused for that session with no `overwrite` exemption — `refused: sales.csv is an input this fan-out declared (2,401 lines → 10); edit it in place (os.fs.edit / os.fs.patch) — a worker cannot replace a declared input; if the task needs it replaced, say so in your reply so the orchestrator can redeclare it` — without a store or a request, since the orchestrator declared it. Pinned by the inputs cases in [delegate-args.test.ts](src/tools/fusion/delegate-args.test.ts), [contract.test.ts](src/tools/fusion/contract.test.ts), [worker-prompt.test.ts](src/tools/fusion/worker-prompt.test.ts), [worker-runner.test.ts](src/tools/fusion/worker-runner.test.ts), [fusion-delegate.test.ts](src/tools/fusion/fusion-delegate.test.ts) and [fs-input-guard.test.ts](src/tools/os/fs-input-guard.test.ts). -**Measured, not guessed, where a measurement exists.** The managed daemon's start (`startDaemon`) runs one 64-token completion once `/health` is OK (`probeThroughput`, skippable with `throughputProbe: false`) and records `timings.predicted_per_second` next to the pid file (`llama-server.throughput.json`, pid-stamped so a previous daemon's figure never describes the live one). `ModelProfileManager.getTokensPerSecond()` reads it lazily and again on every `/props` refresh, and the agent loop passes it into `buildPrompt` (`fusionTokensPerSecond`), where the `### fusion` machine line says "~N tok/s single stream". It is per daemon instance, so it moves only on a restart — which drops the local cache anyway. The auto context size is likewise costed from the model's own attention layout ([src/local-llm/context-size.ts](src/local-llm/context-size.ts): `estimateKvBytesPerToken` — Σ over layers of 2 × KV heads × head dims × bits per value, sliding-window layers weighted by `min(window, ctx) / ctx`, recurrent layers free; within 2× of the 13.7 KB/token measured for Gemma 4 31B at 131,072 with turbo3), with the file-size scale kept only as the fallback when no header could be read; `MAX_AUTO_CONTEXT` is 262,144, still clamped by the model's trained context. Pinned by [src/local-llm/context-size.test.ts](src/local-llm/context-size.test.ts), [src/local-llm/daemon-lifecycle.test.ts](src/local-llm/daemon-lifecycle.test.ts), [src/llm/model-profile-manager.test.ts](src/llm/model-profile-manager.test.ts) and [src/prompt/fusion-machine-facts.test.ts](src/prompt/fusion-machine-facts.test.ts). The pool is read **after** `warmWorkerBackend` (the fallback seam's `prepareLink`): a fusion boot is cloud-active, so the local `/props` probe is deferred and the pool is still sized 1 until that warm runs — reading first would cap every fan-out at one worker. When the pool is what held the width down, the result adds a line naming how many workers were wanted, how many actually ran at once, and `localModels.managed.parallel` — it goes into the *tool result*, because the orchestrator is the party that can adapt to it. +**Measured, not guessed, where a measurement exists.** The managed daemon's start (`startDaemon`) runs one 64-token completion once `/health` is OK (`probeThroughput`, skippable with `throughputProbe: false`) and records `timings.predicted_per_second` next to the pid file (`llama-server.throughput.json`, pid-stamped so a previous daemon's figure never describes the live one). `ModelProfileManager.getTokensPerSecond()` reads it lazily and again on every `/props` refresh, and the agent loop passes it into `buildPrompt` (`fusionTokensPerSecond`), where the `### fusion` machine line says "~N tok/s single stream". It is stamped per daemon instance, but measured once per launch: the record carries `measuredOn` (`throughputBasis`: model, llama.cpp tag and install time, device, the launch's context, and whether weights and cache fit the device whole — `-fit` leaves layers on the CPU when they do not) and `alone` (the probe ran on the only slot, or `GET /slots` showed every slot idle as it began and as it ended — on eight slots a probe that decoded beside queued turns measured 1.1 tok/s against 13-22 alone), and a later start on the same basis carries an `alone` figure younger than `THROUGHPUT_REUSE_MAX_AGE_MS` (24 h) over to its own pid instead of probing (`readReusableThroughput`); the probe kept a 4B model on a 16 GB Mac from answering for another 3-5 s after every load. The auto context size is likewise costed from the model's own attention layout ([src/local-llm/context-size.ts](src/local-llm/context-size.ts): `estimateKvBytesPerToken` — Σ over layers of 2 × KV heads × head dims × bits per value, sliding-window layers weighted by `min(window, ctx) / ctx`, recurrent layers free; within 2× of the 13.7 KB/token measured for Gemma 4 31B at 131,072 with turbo3), with the file-size scale kept only as the fallback when no header could be read; `MAX_AUTO_CONTEXT` is 262,144, still clamped by the model's trained context. On a device that shares the system's RAM (`sharesSystemMemory`: Apple silicon's `MTL*`, an integrated GPU, and parts named like cards — Intel's Meteor/Lunar Lake Arc iGPUs, AMD Strix Halo, NVIDIA GB10 and Jetson Orin/Thor) the free figure is Metal's working-set ceiling, not free memory, so the KV budget also leaves the system a headroom of max(4 GiB, a quarter of RAM) (`UNIFIED_MEMORY_HEADROOM_*`, `resolveUnifiedMemoryKvRoomMiB`, applied in `resolveKvBudgetMiB`), and an auto-sized context's cache is held to 1/16 of RAM on top (`UNIFIED_MEMORY_KV_SHARE`, `resolveContextKvBudgetMiB`) — Qwen 3.5 4B (7 KiB/token at turbo3) gets 149,504 tokens and 1 GiB on a 16 GB Mac instead of 262,144 and 1.75 GiB, Gemma 4 31B 109,568 on 32 GB (three `parallel: "auto"` slots at a 16,384-token reply) and 123,904 on 36 GB, and the daemon log says so (`held to N GB of unified memory …`). The share sizes the context only: `--swa-full` is weighed against the headroom budget, so a sliding-window model on a 96-128 GB Mac keeps prefix reuse. A fixed server share of half of RAM was tried first and floored 27-31B models on 32-36 GB Macs at 32,768. A pinned `contextSize` is never capped. `models start` asks `--list-devices` once for its device pick and the context fit (`deviceTableOnce`, handed to `startDaemon` as `listDevices`; an empty answer, which a run that ran out its 5 s leaves, is asked once more); the TUI start keeps its own enumeration inside `startDaemon`, after its port clearance may have stopped a server holding VRAM. Pinned by [src/local-llm/context-size.test.ts](src/local-llm/context-size.test.ts), [src/local-llm/daemon-lifecycle.test.ts](src/local-llm/daemon-lifecycle.test.ts), [src/llm/model-profile-manager.test.ts](src/llm/model-profile-manager.test.ts) and [src/prompt/fusion-machine-facts.test.ts](src/prompt/fusion-machine-facts.test.ts). The pool is read **after** `warmWorkerBackend` (the fallback seam's `prepareLink`): a fusion boot is cloud-active, so the local `/props` probe is deferred and the pool is still sized 1 until that warm runs — reading first would cap every fan-out at one worker. When the pool is what held the width down, the result adds a line naming how many workers were wanted, how many actually ran at once, and `localModels.managed.parallel` — it goes into the *tool result*, because the orchestrator is the party that can adapt to it. The call is `approval_gated` in [tool-resource-class.ts](src/agent/tool-resource-class.ts) — not because it prompts, but because that is the class meaning "must be solo": one call runs several turns internally, for minutes. It refuses from inside a worker session (one level of fan-out), refuses when the resolver no longer says fusion, and otherwise its status summarises its tasks (`details.outcome`: `all_ok` / `partial` / `all_failed`): `ok` while any task delivered anything — an orchestrator handed a bare error learns nothing about which parts survived, and partial results are the whole value of a fan-out — and `error` only when every task failed or was cancelled; the head line of the result counts every status (`7 tasks: 5 ok, 1 no_changes, 1 failed`). Progress reaches the chat as the `fusion_worker` `AgentLoopEvent` (`started` / `tool` / `finished` / `failed` / `cancelled`, plus `usage`, which carries `contextTokens` for the live worker row and never becomes a feed line), emitted through `emitAgentLoopEventFor(parentSessionId, …)`: `started` and `tool` fire from inside the worker turn's own event hook, which runs under the **worker's** async context, and an event routed by the ambient frame would be tagged with a throwaway session that has no recorder, no hook and no UI. @@ -2704,7 +2726,7 @@ Emitted `TraceEvent` types (see [src/tracing/trace/trace-event.ts](src/tracing/t - `prompt_captured` — `{ stablePrefixHash, tail, tokens: { total, stablePrefix, tail }, slotId, cacheReused }`. The stable prefix is stored only as its salted hash (via `hashPrefix` from [src/llm/slot-manager.ts](src/llm/slot-manager.ts)) so trace files stay compact across steps; the variable tail is stored verbatim. - `llm_completion` — full completion `content` + `reasoningContent` + `timing`, with `attempt: 1 | 2` (attempt 2 == parse retry). `reasoningContent` is sourced from two mutually-exclusive channels: **Channel A** is the dedicated `reasoning_content` SSE field (QwQ / DeepSeek-R1 with `--reasoning-format deepseek`); **Channel B** is the inline `...` / `<|channel>thought...` block that the grammar-aware stream parser splits client-side. `consumeStream` accumulates both separately and prefers Channel A when both fire; the legacy `/completion` endpoint always falls back to Channel B because it never emits `reasoning_content`. Pinned by [src/agent/step-executor.test.ts](src/agent/step-executor.test.ts) "executeStep streaming reasoning accumulator". - `tool_invocation` — executed tool call with args, status, summary, and optional details. -- `parse_retry`, `loop_detected`, `error`, `trace_truncated` — diagnostics. +- `parse_retry`, `loop_detected`, `error`, `trace_truncated` — diagnostics. A `transport` `error` row carries `causeCode`, the errno read off the failure's `cause` chain (`ECONNREFUSED`, `ETIMEDOUT`, `UND_ERR_SOCKET`, …), and so does the `agent loop failed` log line. Invariants: diff --git a/eval-memory/harness/consolidator-tick.ts b/eval-memory/harness/consolidator-tick.ts index 097277770..8dbc4e64c 100644 --- a/eval-memory/harness/consolidator-tick.ts +++ b/eval-memory/harness/consolidator-tick.ts @@ -35,7 +35,7 @@ import { import { DistillRunner } from "../../src/memory/consolidator/distill-runner.js"; import { LlamaServerClient } from "../../src/llm/llama-server-client.js"; import { createLinkGeneratorRunner } from "../../src/memory/links/link-generator-runner.js"; -import { StructuredLogger, stderrSink } from "../../src/tracing/structured-logger.js"; +import { StructuredLogger, createStderrSink } from "../../src/tracing/structured-logger.js"; export interface ConsolidatorTickInput { stateDir: string; @@ -262,7 +262,7 @@ export async function runLinkSweep(deps: LinkSweepDeps): Promise // we cannot tell whether the model is rejecting the prompt, the // parser is dropping malformed output, or we just need to bump // the chunk timeout. - logger: new StructuredLogger({ level: "debug", sinks: [stderrSink()] }), + logger: new StructuredLogger({ level: "debug", sinks: [createStderrSink()] }), }); const sessionId = "eval-link-sweep"; let chunks = 0; diff --git a/src/agent/agent-loop-cancel-race.test.ts b/src/agent/agent-loop-cancel-race.test.ts new file mode 100644 index 000000000..87767d017 --- /dev/null +++ b/src/agent/agent-loop-cancel-race.test.ts @@ -0,0 +1,204 @@ +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { mkdtempSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; + +import { AgentLoop, type AgentLoopEvent } from "./agent-loop.js"; +import { buildDefaultToolRegistry } from "../tools/index.js"; +import { SlotManager } from "../llm/slot-manager.js"; +import { TransportError } from "../llm/reliability/llm-failures.js"; +import type { CompletionResult } from "../llm/llama-server-client.js"; +import { createEmptySessionState } from "../session/session-state.js"; +import type { + CapabilitiesSummary, + ToolDescriptor, +} from "../prompt/stable-prefix.js"; + +/** + * A stop does not always arrive as an abort. When it lands as a stream + * ends, the request can come back as whatever the torn-down socket said + * — `terminated`, `fetch failed` — and the loop read that as a provider + * outage: it parked the turn on a wait the stop then cut short, closing + * the turn twice, or (with the wait off) failed the turn with the + * transport's words. A turn the user stopped is `cancelled`, closed once. + */ + +const TOOLS: ToolDescriptor[] = [ + { + name: "finish", + summary: "Finish the session with a summary.", + argsSchema: '{"summary": string}', + }, +]; + +const CAPS: CapabilitiesSummary = { + platform: "darwin", + arch: "arm64", + browserChannel: "chrome", + workingDir: "/work", + hasClipboard: true, + hasWmctrl: false, + hasNotifications: true, +}; + +describe("a stop that races the end of a request", () => { + let workingDir: string; + + beforeEach(() => { + workingDir = mkdtempSync(join(tmpdir(), "atomic-agent-cancel-race-")); + }); + + afterEach(() => { + rmSync(workingDir, { recursive: true, force: true }); + }); + + function loopWith( + llmComplete: () => Promise, + events: AgentLoopEvent[], + ): AgentLoop { + return new AgentLoop({ + registry: buildDefaultToolRegistry(), + slotManager: new SlotManager(2), + grammar: 'root ::= "ok"', + llmComplete, + toolDescriptors: TOOLS, + capabilities: CAPS, + skillCatalog: [], + onEvent: (event) => events.push(event), + }); + } + + function closes(events: AgentLoopEvent[]): string[] { + return events.flatMap((event) => + event.type === "loop_completed" ? [event.reason] : [], + ); + } + + it("ends the turn cancelled, not failed, when the stopped request comes back as a transport error", async () => { + const controller = new AbortController(); + const events: AgentLoopEvent[] = []; + const loop = loopWith(async () => { + // The stop lands, and the request it tore down reports the socket. + controller.abort(); + throw new TransportError("terminated", null, ""); + }, events); + const result = await loop.runTurn( + createEmptySessionState({ id: "s-race-nowait", workingDir }), + { + userMessage: "stop me", + maxSteps: 5, + taskMaxSteps: 5, + providerWaitEnabled: false, + signal: controller.signal, + }, + ); + expect(result.reason).toBe("cancelled"); + expect(result.session.status).toBe("cancelled"); + expect(result.session.lastError).toBeNull(); + expect(result.session.turnCount).toBe(1); + expect(closes(events)).toEqual(["cancelled"]); + }); + + it("does not park a stopped turn on a provider wait", async () => { + const controller = new AbortController(); + const events: AgentLoopEvent[] = []; + const loop = loopWith(async () => { + controller.abort(); + throw new TransportError("fetch failed", null, ""); + }, events); + const result = await loop.runTurn( + createEmptySessionState({ id: "s-race-wait", workingDir }), + { + userMessage: "stop me", + maxSteps: 5, + taskMaxSteps: 5, + signal: controller.signal, + }, + ); + expect(result.reason).toBe("cancelled"); + expect(result.session.status).toBe("cancelled"); + expect(events.some((event) => event.type === "provider_waiting")).toBe( + false, + ); + expect(result.session.turnCount).toBe(1); + expect(closes(events)).toEqual(["cancelled"]); + }); + + it("closes a turn stopped while parked on an outage exactly once", async () => { + const controller = new AbortController(); + const events: AgentLoopEvent[] = []; + const loop = loopWith(async () => { + // The provider is down; the operator stops the turn while it waits. + setTimeout(() => controller.abort(), 5); + throw new TransportError("fetch failed", null, ""); + }, events); + const result = await loop.runTurn( + createEmptySessionState({ id: "s-park-stop", workingDir }), + { + userMessage: "stop me while you wait", + maxSteps: 5, + taskMaxSteps: 5, + signal: controller.signal, + }, + ); + expect(result.reason).toBe("cancelled"); + expect(result.session.status).toBe("cancelled"); + expect(events.some((event) => event.type === "provider_waiting")).toBe( + true, + ); + // One close, one turn — the wait's exit used to close it a second time. + expect(closes(events)).toEqual(["cancelled"]); + expect( + events.filter((event) => event.type === "turn_finished"), + ).toHaveLength(1); + expect(result.session.turnCount).toBe(1); + }); + + it("still fails, and reports, an error no request makes even when the turn was stopped", async () => { + // `classifyFailure` answers `tool` for an error it does not know — + // the shape a programming error arrives in. A stop happening at the + // same moment must not file it away as a cancel. + const controller = new AbortController(); + const events: AgentLoopEvent[] = []; + const loop = loopWith(async () => { + controller.abort(); + throw new Error("bug: step context is undefined"); + }, events); + const result = await loop.runTurn( + createEmptySessionState({ id: "s-bug", workingDir }), + { + userMessage: "work", + maxSteps: 5, + taskMaxSteps: 5, + signal: controller.signal, + }, + ); + expect(result.reason).toBe("failed"); + expect(result.session.status).toBe("failed"); + expect(result.session.lastError).toBe("bug: step context is undefined"); + const failure = events.find((event) => event.type === "loop_failed"); + expect(failure?.type === "loop_failed" ? failure.category : null).toBe( + "tool", + ); + }); + + it("still fails a request that breaks while nobody has stopped the turn", async () => { + const events: AgentLoopEvent[] = []; + const loop = loopWith(async () => { + throw new TransportError("terminated", null, ""); + }, events); + const result = await loop.runTurn( + createEmptySessionState({ id: "s-no-stop", workingDir }), + { + userMessage: "work", + maxSteps: 5, + taskMaxSteps: 5, + providerWaitEnabled: false, + signal: new AbortController().signal, + }, + ); + expect(result.reason).toBe("failed"); + expect(result.session.status).toBe("failed"); + expect(result.session.lastError).toBe("terminated"); + }); +}); diff --git a/src/agent/agent-loop-lesson-lifecycle.test.ts b/src/agent/agent-loop-lesson-lifecycle.test.ts index deac099b1..d555b7710 100644 --- a/src/agent/agent-loop-lesson-lifecycle.test.ts +++ b/src/agent/agent-loop-lesson-lifecycle.test.ts @@ -21,6 +21,12 @@ import type { ToolDescriptor, } from "../prompt/stable-prefix.js"; import type { LessonIndexEntry } from "../memory/lessons/lesson-store.js"; +import { + humanizeOpenAiHttpError, + OpenAiHttpError, +} from "../llm/provider/openai/openai-http.js"; +import { parseProviderErrorBody } from "../llm/provider/openai/parse-provider-error-body.js"; +import { TransportError } from "../llm/reliability/llm-failures.js"; /** * Phase 6 — lesson lifecycle integration with `AgentLoop.runTurn`. @@ -28,6 +34,7 @@ import type { LessonIndexEntry } from "../memory/lessons/lesson-store.js"; * Pins: * - `reply` / `finish` → hook fires with outcome="success". * - thrown error → hook fires with outcome="failure". + * - a refusal because the account cannot pay → hook is NOT called. * - `cancelled` (signal aborted) → hook is NOT called. * - `max_steps` → hook is NOT called. * - Once-per-turn dedup: surfacing the same lesson across many @@ -208,6 +215,50 @@ describe("AgentLoop lesson lifecycle hook (phase 6)", () => { expect(calls).toEqual([{ outcome: "failure", surfaced: [3, 4] }]); }); + it("does NOT fire the hook when the provider refused because the account cannot pay (item 40)", async () => { + // The turn fails at once on its first request, but an empty account + // says nothing about the lessons recalled for it: neutral, as the + // paused path for exhausted credit always was. + const calls: HookCall[] = []; + const body = + '{"title":"Forbidden","status":403,"message":"You\'ve run out of funds. Please top up your balance"}'; + const http = new OpenAiHttpError( + `openai provider 403: ${body}`, + 403, + "https://api.aimlapi.com/v1/chat/completions", + false, + null, + "aimlapi", + undefined, + { body: parseProviderErrorBody(body) }, + ); + const loop = new AgentLoop({ + registry: buildDefaultToolRegistry(), + slotManager: new SlotManager(2), + grammar: 'root ::= "ok"', + llmComplete: async () => { + throw new TransportError(humanizeOpenAiHttpError(http), 403, http.url, { + cause: http, + }); + }, + toolDescriptors: TOOLS, + capabilities: CAPS, + skillCatalog: SKILLS, + memoryContextProvider: makeProvider([[5, 6]]), + lessonLifecycle: captureHook(calls), + }); + const result = await loop.runTurn( + createEmptySessionState({ id: "s-no-funds", workingDir }), + { + userMessage: "x", + maxSteps: 2, + signal: new AbortController().signal, + }, + ); + expect(result.reason).toBe("failed"); + expect(calls).toEqual([]); + }); + it("does NOT fire the hook when the turn is cancelled", async () => { const calls: HookCall[] = []; const ac = new AbortController(); diff --git a/src/agent/agent-loop.test.ts b/src/agent/agent-loop.test.ts index 259691387..0c5405b7b 100644 --- a/src/agent/agent-loop.test.ts +++ b/src/agent/agent-loop.test.ts @@ -9,11 +9,15 @@ import { osFsReadTool } from "../tools/os/fs-read.js"; import { SlotManager } from "../llm/slot-manager.js"; import { TransportError } from "../llm/reliability/llm-failures.js"; import { LlamaServerError } from "../llm/llama-server-client.js"; -import { OpenAiHttpError } from "../llm/provider/openai/openai-http.js"; +import { + humanizeOpenAiHttpError, + OpenAiHttpError, +} from "../llm/provider/openai/openai-http.js"; import { parseProviderErrorBody } from "../llm/provider/openai/parse-provider-error-body.js"; import { PARSE_RECOVERY_BUDGET } from "./parse-failure-recovery.js"; import { EMPTY_COMPLETION_RECOVERY_BUDGET } from "./empty-completion-recovery.js"; import { createEmptySessionState } from "../session/session-state.js"; +import type { SessionState } from "../session/session-state.js"; import type { CompletionResult, LlamaServerClient, @@ -493,6 +497,9 @@ describe("AgentLoop end-to-end with mock LLM", () => { expect(events[0]!.nextRetryMs).toBe(2_000); expect(events[1]!.nextRetryMs).toBe(4_000); expect(events[0]!.reason).toBe("fetch failed"); + // A bare `TransportError` carries no errno anywhere on its chain, so + // the event names none rather than guessing one. + expect(events[0]).not.toHaveProperty("causeCode"); // The parked attempts are not steps and replay nothing: one tool // step plus the reply, not four steps and two noops. expect(noopRuns).toBe(1); @@ -1365,6 +1372,10 @@ describe("AgentLoop end-to-end with mock LLM", () => { // statusless shape and is exactly what the park exists for. const registry = buildDefaultToolRegistry(); const waits: unknown[] = []; + const warnings: Array<{ + message: string; + context?: Record; + }> = []; let calls = 0; const loop = new AgentLoop({ registry, @@ -1391,6 +1402,14 @@ describe("AgentLoop end-to-end with mock LLM", () => { onEvent: (event) => { if (event.type === "provider_waiting") waits.push(event); }, + logger: { + debug: () => {}, + info: () => {}, + warn: (message: string, context?: Record) => { + warnings.push({ message, context }); + }, + error: () => {}, + } as never, }); const result = await loop.runTurn( createEmptySessionState({ id: "s-econnrefused", workingDir }), @@ -1404,6 +1423,73 @@ describe("AgentLoop end-to-end with mock LLM", () => { expect(result.reason).toBe("reply"); expect(waits).toHaveLength(1); expect(calls).toBe(2); + // The reason is a bare `fetch failed` — the same line a network + // outage gives. The errno is what says the server was not running, + // and it reaches both the event (and so the trace) and the log. + expect(waits[0]).toMatchObject({ + reason: "fetch failed", + cause: { kind: "refused" }, + causeCode: "ECONNREFUSED", + }); + expect( + warnings.find( + (w) => w.message === "provider unreachable; parking the turn", + )?.context, + ).toMatchObject({ error: "fetch failed", causeCode: "ECONNREFUSED" }); + }); + + it("names the errno in the failure log when the turn does not wait", async () => { + // The same refused connection with waiting switched off: the turn + // fails at once, and `agent loop failed` is the only log line about + // it — it must not say only `fetch failed` either. + const registry = buildDefaultToolRegistry(); + const failures: Array<{ + message: string; + context?: Record; + }> = []; + const loop = new AgentLoop({ + registry, + slotManager: new SlotManager(2), + grammar: 'root ::= "ok"', + llmComplete: async () => { + throw new LlamaServerError( + "fetch failed", + null, + "http://127.0.0.1:8080/completion", + false, + "ECONNREFUSED", + ); + }, + toolDescriptors: TOOLS, + capabilities: CAPS, + skillCatalog: SKILLS, + logger: { + debug: () => {}, + info: () => {}, + warn: () => {}, + error: (message: string, context?: Record) => { + failures.push({ message, context }); + }, + } as never, + }); + const result = await loop.runTurn( + createEmptySessionState({ id: "s-econnrefused-no-wait", workingDir }), + { + userMessage: "server stopped", + maxSteps: 5, + taskMaxSteps: 5, + providerWaitEnabled: false, + signal: new AbortController().signal, + }, + ); + expect(result.reason).toBe("failed"); + expect( + failures.find((f) => f.message === "agent loop failed")?.context, + ).toMatchObject({ + error: "fetch failed", + category: "transport", + causeCode: "ECONNREFUSED", + }); }); it("still waits out a socket-level ETIMEDOUT, which is the kernel's deadline (issue #490)", async () => { @@ -1460,8 +1546,24 @@ describe("AgentLoop end-to-end with mock LLM", () => { it("stops the turn resumable when the provider's body says the credit is exhausted (F29)", async () => { // The Codex attempt: a 429 carrying `credit_balance_exhausted` was - // parked and retried as rate limiting, 42 times per worker. + // parked and retried as rate limiting, 42 times per worker. The task + // has done a step by then, so there is work to resume (item 40 ends + // a turn refused on its first request instead, see below). const registry = buildDefaultToolRegistry(); + registry.register({ + name: "noop", + description: "no-op", + readonly: true, + async run() { + return { + tool: "noop", + status: "ok", + summary: "noop", + details: {}, + truncated: false, + }; + }, + }); const events: string[] = []; let calls = 0; const body = JSON.stringify({ @@ -1479,6 +1581,9 @@ describe("AgentLoop end-to-end with mock LLM", () => { grammar: 'root ::= "ok"', llmComplete: async () => { calls += 1; + if (calls === 1) { + return makeCompletion(JSON.stringify({ tool: "noop", args: {} })); + } throw new TransportError( '"openrouter" is rate-limiting this key (429).', 429, @@ -1521,15 +1626,16 @@ describe("AgentLoop end-to-end with mock LLM", () => { signal: new AbortController().signal, }, ); - // One request, no park, no failure: paused where it stood. - expect(calls).toBe(1); + // One refused request after the step, no park, no failure: paused + // where it stood. + expect(calls).toBe(2); expect(Date.now() - started).toBeLessThan(1_000); expect(events).toEqual(["credit_exhausted", "loop_completed"]); expect(result.reason).toBe("max_steps"); expect(result.stopCause).toBe("credit_exhausted"); expect(result.session.status).toBe("stalled"); expect(result.session.lastError).toBe( - 'task_stopped:credit_exhausted: "openrouter" is out of credit after 0 steps', + 'task_stopped:credit_exhausted: "openrouter" is out of credit after 1 steps', ); const last = result.session.turns.at(-1); expect(last?.kind).toBe("assistant_reply"); @@ -1539,6 +1645,121 @@ describe("AgentLoop end-to-end with mock LLM", () => { expect((last as { text: string }).text).toContain("say `continue`"); }); + /* Item 40: a provider that refuses the very first request because the + account cannot pay leaves nothing to resume. The turn fails at once + with the provider's own sentence, as a refused key does, instead of + pausing on a "(paused …) after 0 steps" reply or parking on a 429. */ + describe("a billing refusal on the task's first request", () => { + const refusedBy = (status: number, body: string, label: string) => { + const http = new OpenAiHttpError( + `openai provider ${status}: ${body}`, + status, + "https://api.aimlapi.com/v1/chat/completions", + false, + null, + label, + undefined, + { body: parseProviderErrorBody(body) }, + ); + return new TransportError(humanizeOpenAiHttpError(http), status, http.url, { + cause: http, + }); + }; + + async function runRefused( + err: TransportError, + session: SessionState = createEmptySessionState({ id: "s-billing", workingDir }), + ) { + const events: AgentLoopEvent[] = []; + let calls = 0; + const loop = new AgentLoop({ + registry: buildDefaultToolRegistry(), + slotManager: new SlotManager(2), + grammar: 'root ::= "ok"', + llmComplete: async () => { + calls += 1; + throw err; + }, + toolDescriptors: TOOLS, + capabilities: CAPS, + skillCatalog: SKILLS, + onEvent: (event) => events.push(event), + }); + const started = Date.now(); + const result = await loop.runTurn( + session, + { + userMessage: "hello", + maxSteps: 5, + taskMaxSteps: 5, + providerWaitEnabled: true, + signal: new AbortController().signal, + }, + ); + return { result, events, calls, elapsedMs: Date.now() - started }; + } + + const AIML_403 = + '{"title":"Forbidden","status":403,"message":"You\'ve run out of funds. Please top up your balance or update your payment method to continue: https://aimlapi.com/app/billing"}'; + const SENTENCE = + "AI/ML API refused the request: you've run out of funds. Top up your balance with AI/ML API or pick another provider in the Providers panel."; + + it("fails the turn on AI/ML API's 403 with its sentence, without a wait or a pause", async () => { + const { result, events, calls, elapsedMs } = await runRefused( + refusedBy(403, AIML_403, "aimlapi"), + ); + expect(calls).toBe(1); + expect(elapsedMs).toBeLessThan(1_000); + expect(result.reason).toBe("failed"); + expect(result.stopCause).toBeUndefined(); + const types = events.map((e) => e.type); + expect(types).not.toContain("provider_waiting"); + expect(types).not.toContain("credit_exhausted"); + const failed = events.find((e) => e.type === "loop_failed"); + expect(failed?.type === "loop_failed" && failed.error.message).toBe(SENTENCE); + // No synthetic "(paused …)" reply: the transcript keeps the failed + // turn's own record, and that names the provider and its reason. + const replies = result.session.turns + .filter((t) => t.kind === "assistant_reply") + .map((t) => (t as { text: string }).text); + expect(replies.some((text) => text.includes("(paused"))).toBe(false); + expect(replies.at(-1)).toContain("AI/ML API refused the request: you've run out of funds"); + expect(result.session.status).toBe("failed"); + }); + + it("pauses again when the turn resumes a task an earlier turn stopped: the task has work to keep", async () => { + // `continue` after "(paused: … out of credit …)", the account still empty. + const stopped: SessionState = { + ...createEmptySessionState({ id: "s-billing-resumed", workingDir }), + status: "stalled", + }; + const { result, events, calls } = await runRefused( + refusedBy(403, AIML_403, "aimlapi"), + stopped, + ); + expect(calls).toBe(1); + expect(result.reason).toBe("max_steps"); + expect(result.stopCause).toBe("credit_exhausted"); + const types = events.map((e) => e.type); + expect(types).toContain("credit_exhausted"); + expect(types).not.toContain("loop_failed"); + expect(types).not.toContain("provider_waiting"); + }); + + it("does not park on a 429 that says the account is empty, though a 429 is otherwise a wait", async () => { + const { result, events, calls } = await runRefused( + refusedBy( + 429, + '{"error":{"message":"You exceeded your current quota, please check your plan and billing details.","type":"insufficient_quota","code":"insufficient_quota"}}', + "openai", + ), + ); + expect(calls).toBe(1); + expect(result.reason).toBe("failed"); + expect(events.map((e) => e.type)).not.toContain("provider_waiting"); + }); + }); + it("waits as long as the provider asked, on a 402 the outage wait would otherwise refuse (F29)", async () => { // OpenRouter's `in_flight_budget_exhausted` with a retry hint ended // a cloud-only run at 2m19s as final. The hint is honoured instead. diff --git a/src/agent/agent-loop.ts b/src/agent/agent-loop.ts index 2f7607e4f..27dcaad7e 100644 --- a/src/agent/agent-loop.ts +++ b/src/agent/agent-loop.ts @@ -41,10 +41,13 @@ import { isRequestSizeRejection, } from "../llm/index.js"; import { readFailingLink } from "../llm/fallback/failed-attempts.js"; +import { describeFailedLinks } from "../llm/fallback/failed-links.js"; +import { readErrnoCode } from "../llm/errno-code.js"; import { readProviderErrorVerdict } from "../llm/reliability/provider-error-verdict.js"; import { classifyProviderWaitCause, type ProviderWaitCause, + type ProviderWaitFailure, } from "../llm/reliability/provider-wait-cause.js"; import { composeSizeRejectionNotice, @@ -840,6 +843,16 @@ export type AgentLoopEvent = * for logs and traces. */ cause?: ProviderWaitCause; + /** + * The errno-like code the transport left on the failure's `cause` + * chain (`ECONNREFUSED`, `ETIMEDOUT`, `ENOTFOUND`, `ECONNRESET`, + * `UND_ERR_SOCKET`, …), as `readErrnoCode` reads it. `reason` is + * often a bare `fetch failed`, which looks the same for a local + * server that is not running (refused) and a network that is down + * (unreachable, timed out); this is the fact that tells them apart + * in a trace or a log. Absent when the transport left no code. + */ + causeCode?: string; /** * The provider link the turn is waiting on: the one whose failure * parked it. With a fallback chain that is the last link tried, @@ -848,6 +861,14 @@ export type AgentLoopEvent = * did not come through the chain or a pinned link. */ providerId?: string; + /** + * The links that failed before the one waited on, in the order + * they were tried, each with its own cause. The provider the user + * picked is usually the first, and why it failed (an account out + * of funds, a refused key) is the part a UI says before the link + * it waits on (item 40). Absent when nothing failed before it. + */ + fallbackFailures?: readonly ProviderWaitFailure[]; } | { /** The provider answered again; the parked turn is running on. */ @@ -1175,6 +1196,13 @@ export class AgentLoop { options: RunTurnOptions, ): Promise { let state = session; + /** + * This turn picks up a task an earlier turn stopped at a ceiling or + * for credit (`continue` after "(paused: …)"): its first request is + * not the task's first, and a billing refusal pauses it again rather + * than failing it (item 40). + */ + const resumesStoppedTask = session.status === "stalled"; // NOTE: previously called `reflectionRunner.abortPending({ sessionId })` // here on every turn to "free the reflection slot quickly". That @@ -2155,11 +2183,27 @@ export class AgentLoop { // reserved final step must keep its `cancelled` outcome // (issue #107 — cancellation semantics remain unchanged), not // be relabelled `max_steps`. + // + // Once the turn's own signal has aborted, a request that fails the + // way requests fail is the stop's doing, not a verdict on the + // provider. An abort that lands as the stream ends does not always + // surface as an abort: the socket the stop tore down can come back + // as `terminated` or `fetch failed`, a cut-off body as a parse or + // empty-completion failure — read as a transport outage (a wait, + // then a second close of the turn) or as `failed` with the + // transport's words, for a turn the user had simply stopped. The + // `tool` catch-all is left out on purpose: it is what + // `classifyFailure` answers for an error it does not recognise, + // which is how a programming error arrives, and that stays a + // failure — reported as one — whatever the signal says. + const stoppedRequest = options.signal.aborted && category !== "tool"; const cancelled = !ceilingFired && - (err instanceof CancelledError || + (stoppedRequest || + err instanceof CancelledError || (err instanceof LlmFailure && err.category === "cancelled") || category === "cancelled"); + if (cancelled) category = "cancelled"; if (ceilingFired) { if (!finalizationStep) { // Abandon the request and take the reserved summary step @@ -2476,8 +2520,20 @@ export class AgentLoop { // is told which provider refused. (A fallback link, when the // chain has one, has already been tried by the time the error // reaches here.) + // + // Paused, that is, once the task has done something to keep: a + // step of this turn, or an earlier turn's that this one resumes. + // Refused on the task's very first request, there is nothing to + // resume: the turn fails at once with the provider's own sentence + // ("… refused the request: you've run out of funds. Top up …"), + // the way a refused key does, instead of a "(paused …) after 0 + // steps" reply standing in for an answer (item 40). const verdict = cancelled ? null : readProviderErrorVerdict(err); - if (verdict?.kind === "credit_exhausted") { + const creditRefused = verdict?.kind === "credit_exhausted"; + if ( + verdict?.kind === "credit_exhausted" && + (stepsTaken > 0 || resumesStoppedTask) + ) { stopCause = "credit_exhausted"; creditStop = { provider: verdict.provider, detail: verdict.detail }; reason = "max_steps"; @@ -2519,6 +2575,7 @@ export class AgentLoop { if ( category === "transport" && !cancelled && + !creditRefused && providerWaitCfg.enabled && (isWaitableOutage(err) || retryHint !== null) && outageWaitedMs < providerWaitCfg.maxWaitMs @@ -2537,6 +2594,13 @@ export class AgentLoop { outageAttempts += 1; awaitingRecovery = true; const waitedOn = readFailingLink(err); + const failedBefore = describeFailedLinks(err); + // The errno behind the outage: `fetch failed` is the same + // sentence for a local server that is not running and for a + // network that is down. Read only here, for a `transport` + // failure (the condition above), so a user's abort, which + // classifies `cancelled`, never lends its `ABORT_ERR` to it. + const causeCode = readErrnoCode(err); this.deps.onEvent?.({ type: "provider_waiting", attempt: outageAttempts, @@ -2545,7 +2609,11 @@ export class AgentLoop { nextRetryMs, reason: runError.message, cause: classifyProviderWaitCause(err), + ...(causeCode !== undefined ? { causeCode } : {}), ...(waitedOn !== undefined ? { providerId: waitedOn } : {}), + ...(failedBefore.length > 0 + ? { fallbackFailures: failedBefore } + : {}), }); this.deps.logger?.warn("provider unreachable; parking the turn", { sessionId: state.id, @@ -2554,6 +2622,7 @@ export class AgentLoop { waitedMs: outageWaitedMs, nextRetryMs, error: runError.message, + ...(causeCode !== undefined ? { causeCode } : {}), ...(waitedOn !== undefined ? { providerId: waitedOn } : {}), }); await abortableSleep(nextRetryMs, options.signal); @@ -2563,13 +2632,11 @@ export class AgentLoop { // attempt carried. pendingNotice = noticeForThisStep; if (options.signal.aborted) { + // Stopped while parked. The close below the loop does the + // rest — status, `loop_completed`, the turn count — exactly + // once; doing it here as well closed the turn twice: two + // `loop_completed` events and a turn counted double. reason = "cancelled"; - state = { ...state, status: "cancelled" }; - this.deps.onEvent?.({ - type: "loop_completed", - reason: "cancelled", - }); - state = incrementTurnCount(state); break; } // Retry the very same step index: `i += 1` runs on `continue`, @@ -2597,11 +2664,22 @@ export class AgentLoop { runError = truncationRetry.original; category = classifyFailure(runError); } + // Same errno as the wait above, for the turn that fails instead + // of parking (waiting disabled, budget spent, a refusal that + // will not fix itself). Read off `runError`, which the swap just + // above may have replaced, and only for `transport`: any other + // category's code — an abort's `ABORT_ERR` — is not a network + // cause. + const failureCauseCode = + category === "transport" ? readErrnoCode(runError) : undefined; this.deps.logger?.error("agent loop failed", { sessionId: state.id, stepIndex: i, error: runError.message, category, + ...(failureCauseCode !== undefined + ? { causeCode: failureCauseCode } + : {}), }); this.deps.onEvent?.({ type: "loop_failed", @@ -2672,8 +2750,10 @@ export class AgentLoop { // Phase 6 — bump failure_count for every surfaced lesson. // `cancelled` is intentionally NOT routed here; that branch // returned earlier without calling the hook (cancellation - // carries neither success nor failure signal). - if (!options.ephemeral) { + // carries neither success nor failure signal). Nor is an account + // that cannot pay (item 40): it says nothing about the lessons + // recalled at turn start, as the paused path for it says nothing. + if (!options.ephemeral && !creditRefused) { invokeLessonLifecycle( this.deps, state.id, diff --git a/src/agent/step-executor.test.ts b/src/agent/step-executor.test.ts index 739bd7c67..32e66c6ae 100644 --- a/src/agent/step-executor.test.ts +++ b/src/agent/step-executor.test.ts @@ -37,6 +37,10 @@ import { replyTool } from "../tools/conversation/reply.js"; import { openAiToolCallAdapter } from "../llm/provider/openai/openai-tool-call-adapter.js"; import { reviewStallToolSet } from "./review-stall.js"; import { resetConfigCache } from "../config/index.js"; +import { + StructuredLogger, + type LogRecord, +} from "../tracing/structured-logger.js"; import { buildOpenAiChatBody } from "../llm/provider/openai/openai-build-body.js"; import type { CapabilitiesSummary, @@ -2638,7 +2642,7 @@ describe("executeStep skill.view short-circuit", () => { describe("executeStep unparseable-completion fallback", () => { const grammarsDir = join(process.cwd(), "grammars"); - async function runQwenStep(content: string) { + async function runQwenStep(content: string, logger?: StructuredLogger) { const registry = new ToolRegistry(); registry.register(replyTool); const grammar = await buildGrammar(QWEN_THINK_PROFILE, grammarsDir); @@ -2675,6 +2679,7 @@ describe("executeStep unparseable-completion fallback", () => { }), grammar, profile: QWEN_THINK_PROFILE, + ...(logger ? { logger } : {}), }, ); } @@ -2695,6 +2700,29 @@ describe("executeStep unparseable-completion fallback", () => { runQwenStep("[SFC] 分析中 rambling that never closes"), ).rejects.toThrow(/tool-call/); }); + + // agent.log is what the desktop's "Save report for support" takes the + // tail of, and a completion can quote whatever the model read. + it("logs the unparseable completion's length and start, never the whole of it", async () => { + const records: LogRecord[] = []; + const logger = new StructuredLogger({ + level: "debug", + sinks: [(record) => records.push(record)], + }); + const prose = `Here is the file: ${"x".repeat(2_000)} OPENAI_API_KEY=sk-tail-of-the-file`; + const content = `thinking about it${prose}`; + const outcome = await runQwenStep(content, logger); + expect(outcome.toolCalls[0]!.tool).toBe("reply"); + const failed = records.find( + (r) => r.message === "tool-call parse failed after retry", + ); + expect(failed?.context).toMatchObject({ rawLength: content.length }); + expect(failed?.context).not.toHaveProperty("raw"); + const preview = String(failed?.context?.["rawPreview"]); + expect(preview.length).toBeLessThanOrEqual(301); + expect(content.startsWith(preview.slice(0, -1))).toBe(true); + expect(JSON.stringify(records)).not.toContain("sk-tail-of-the-file"); + }); }); describe("parallelToolCalls derivation (issue #104)", () => { @@ -5291,6 +5319,7 @@ describe("links need a source (#581)", () => { linkEvidence?: false; claimEvidence?: boolean; worldText?: string; + logger?: StructuredLogger; } = {}, ) { const registry = new ToolRegistry(); @@ -5374,11 +5403,35 @@ describe("links need a source (#581)", () => { }), grammar, profile: PLAIN_INSTRUCT_PROFILE, + ...(options.logger ? { logger: options.logger } : {}), }, ); return { outcome, marks: () => marks, claimMarks: () => claimMarks }; } + it("logs how many links were held and a few of them, each cut short", async () => { + const records: LogRecord[] = []; + const logger = new StructuredLogger({ + level: "debug", + sinks: [(record) => records.push(record)], + }); + const links = Array.from( + { length: 7 }, + (_, i) => `${MANGLED}-${i}-${"a".repeat(300)}`, + ); + await run(`Reviews: ${links.join(" ")}`, { logger }); + const held = records.find( + (r) => r.message === "reply links a URL no tool result holds; held once", + ); + expect(held?.context?.["linkCount"]).toBe(7); + const logged = held?.context?.["links"] as string[]; + expect(logged).toHaveLength(5); + for (const [i, url] of logged.entries()) { + expect(url.length).toBeLessThanOrEqual(201); + expect(links[i]!.startsWith(url.slice(0, -1))).toBe(true); + } + }); + it("holds a reply whose link no result holds, once, with a notice naming it", async () => { const { outcome, marks } = await run(`The review: ${MANGLED}`); expect(outcome.terminal).toBeNull(); diff --git a/src/agent/step-executor.ts b/src/agent/step-executor.ts index 6f2d703dd..ea3bd8c78 100644 --- a/src/agent/step-executor.ts +++ b/src/agent/step-executor.ts @@ -1475,7 +1475,7 @@ async function executeStepInner( sessionId: ctx.session.id, stepIndex: ctx.stepIndex, rawLength: completion.content.length, - raw: completion.content, + rawPreview: logTextPreview(completion.content), }); // Last resort: the model talked instead of emitting a call (small // models routinely answer "hi" in plain prose even under the @@ -1605,7 +1605,10 @@ async function executeStepInner( deps.logger?.warn("reply links a URL no tool result holds; held once", { sessionId: ctx.session.id, stepIndex: ctx.stepIndex, - links: unsourced.map((link) => link.url), + linkCount: unsourced.length, + links: unsourced + .slice(0, LOG_LINKS_MAX) + .map((link) => logTextPreview(link.url, LOG_LINK_CHARS)), }); } if (notices.length > 0) { @@ -3111,6 +3114,23 @@ export function callArgsSchemaValid( } } +/** + * How much of the model's own text a log line carries. The log is a file + * the operator may attach to a support report (the desktop app's "Save + * report for support" takes its tail), and a completion can quote + * anything the model read — a file's contents, a key that was in one — + * so a line gives the text's length and its start, never the whole. + */ +const LOG_TEXT_PREVIEW_CHARS = 300; +/** A held reply's links in its log line: this many, each cut to this length. */ +const LOG_LINKS_MAX = 5; +const LOG_LINK_CHARS = 200; + +/** The first `max` characters of `text`, marked with `…` when cut. */ +function logTextPreview(text: string, max = LOG_TEXT_PREVIEW_CHARS): string { + return text.length > max ? `${text.slice(0, max)}…` : text; +} + /** * Trim a raw completion body to the short preview attached to every * `GrammarError` so postmortems can tell grammar misconfiguration apart @@ -3304,7 +3324,12 @@ function renderOpenReasoningBlock( */ function toLlmFailure(err: unknown, ctx: StepContext): LlmFailure { if (err instanceof LlmFailure) return err; - if (ctx.signal.aborted) { + // A stopped step's request that fails the way requests fail (an abort, + // a torn socket, a cut-off body) is the stop's doing. An error the + // classifier does not recognise — `tool`, the shape a programming error + // arrives in — stays what it is whatever the signal says, so it is + // reported as a failure, not filed away as a cancel (ATO-137). + if (ctx.signal.aborted && classifyFailure(err) !== "tool") { return new CancelledError( err instanceof Error ? err.message : "operation cancelled", { cause: err }, diff --git a/src/cli/config-command.test.ts b/src/cli/config-command.test.ts index 973108d3c..867ddc918 100644 --- a/src/cli/config-command.test.ts +++ b/src/cli/config-command.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it, beforeEach, afterEach, vi } from "vitest"; import { mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; import { join } from "node:path"; +import { Readable } from "node:stream"; import { configCommand } from "./config-command.js"; import { @@ -141,6 +142,72 @@ describe("configCommand", () => { expect(onDisk.localModels.url).toBe("http://x:1"); }); + describe("set - (the whole-file payload on stdin)", () => { + const stdinOf = (...chunks: string[]) => Readable.from(chunks); + + it("replaces the whole file with the JSON read from stdin", async () => { + const code = await configCommand( + ["set", "-"], + stdinOf( + `{"version":${USER_CONFIG_VERSION},`, + '"localModels":{"url":"http://x:2"}}', + ), + ); + expect(code).toBe(0); + const onDisk = JSON.parse( + readFileSync(join(stateDir, "config.json"), "utf8"), + ); + expect(onDisk.localModels.url).toBe("http://x:2"); + }); + + it("keeps a provider's inline key exactly as given", async () => { + // The desktop saves keys inline and writes every change this way, so + // the key never rides on a command line. + const key = "sk-dummy-stdin-0123456789"; + const code = await configCommand( + ["set", "-"], + stdinOf( + JSON.stringify({ + version: USER_CONFIG_VERSION, + llm: { + activeTextProvider: "local-llama", + providers: [ + { id: "local-llama", kind: "llama-server", url: "http://127.0.0.1:8080" }, + { id: "aimlapi", kind: "aimlapi", apiKey: key, defaultChatModel: "m" }, + ], + }, + }), + ), + ); + expect(code).toBe(0); + const onDisk = JSON.parse( + readFileSync(join(stateDir, "config.json"), "utf8"), + ); + const entry = onDisk.llm.providers.find( + (p: { id: string }) => p.id === "aimlapi", + ); + expect(entry.apiKey).toBe(key); + }); + + it("rejects malformed JSON from stdin without touching the file", async () => { + await configCommand(["get"]); + const before = readFileSync(join(stateDir, "config.json"), "utf8"); + const code = await configCommand(["set", "-"], stdinOf("{not json")); + expect(code).toBe(1); + expect(stderr).toContain("invalid JSON"); + expect(readFileSync(join(stateDir, "config.json"), "utf8")).toBe(before); + }); + + it("rejects an empty stdin as no JSON at all", async () => { + await configCommand(["get"]); + const before = readFileSync(join(stateDir, "config.json"), "utf8"); + const code = await configCommand(["set", "-"], stdinOf()); + expect(code).toBe(1); + expect(stderr).toContain("invalid JSON"); + expect(readFileSync(join(stateDir, "config.json"), "utf8")).toBe(before); + }); + }); + it("set without a payload prints usage and returns non-zero", async () => { const code = await configCommand(["set"]); expect(code).toBe(1); diff --git a/src/cli/config-command.ts b/src/cli/config-command.ts index 9bec2637f..351596333 100644 --- a/src/cli/config-command.ts +++ b/src/cli/config-command.ts @@ -24,7 +24,14 @@ import { writeRawUserConfigFileSync, } from "../config/config-paths.js"; -export async function configCommand(args: string[]): Promise { +/** + * `stdin` is where `config set -` reads its payload; the CLI passes + * `process.stdin`, a test hands in a stream of its own. + */ +export async function configCommand( + args: string[], + stdin: NodeJS.ReadableStream = process.stdin, +): Promise { const sub = args[0]; if (!sub || sub === "-h" || sub === "--help") { process.stdout.write(HELP); @@ -35,7 +42,7 @@ export async function configCommand(args: string[]): Promise { case "get": return handleGet(args.slice(1)); case "set": - return await handleSet(args.slice(1)); + return await handleSet(args.slice(1), stdin); case "unset": return handleUnset(args.slice(1)); case "list": @@ -82,14 +89,26 @@ function handleGet(args: string[]): number { return 0; } -async function handleSet(args: string[]): Promise { +async function handleSet( + args: string[], + stdin: NodeJS.ReadableStream, +): Promise { if (args.length === 0) { process.stderr.write( "usage: atomic-agent config set \n" + - " or: atomic-agent config set ''\n", + " or: atomic-agent config set ''\n" + + " or: atomic-agent config set - < config.json\n", ); return 1; } + // The whole-file payload from stdin. A file that carries API keys, + // MCP env blocks or auth headers must not travel as an argument: any + // process on the machine can read another's command line (`ps`), for + // as long as the write runs. The desktop app writes every config + // change this way. + if (args.length === 1 && args[0] === "-") { + return setWholeFile(await readAllText(stdin)); + } // Form discrimination. A leading `{` means the whole-file JSON payload, // including the case where the shell split one JSON argument across // several argv entries (`set { "version":40, ... }`), which is why the @@ -110,6 +129,15 @@ async function handleSet(args: string[]): Promise { return setWholeFile(args.join(" ")); } +/** Everything `stream` delivers until it ends, as UTF-8. */ +async function readAllText(stream: NodeJS.ReadableStream): Promise { + const chunks: Buffer[] = []; + for await (const chunk of stream) { + chunks.push(typeof chunk === "string" ? Buffer.from(chunk) : chunk); + } + return Buffer.concat(chunks).toString("utf8"); +} + async function setWholeFile(raw: string): Promise { let parsed: unknown; try { diff --git a/src/cli/config-help.ts b/src/cli/config-help.ts index afacb6638..328f21683 100644 --- a/src/cli/config-help.ts +++ b/src/cli/config-help.ts @@ -38,6 +38,8 @@ export const HELP = " get Print one value by dotted key", " set Set one value, leaving the rest of the file alone", " set '' Replace the whole config file with a JSON payload", + " set - Same, with the JSON read from stdin (keeps secrets", + " such as inline API keys off the command line)", " unset Restore one key to its default", " list Print every key as `key = value`", " path Print the path to the config file", diff --git a/src/cli/models-handlers.ts b/src/cli/models-handlers.ts index ed93c7009..5ae490ba5 100644 --- a/src/cli/models-handlers.ts +++ b/src/cli/models-handlers.ts @@ -35,6 +35,8 @@ import { isModelDownloaded, listLocalModels, listVulkanDevices, + AUTO_UPDATE_RECHECK_MS, + deviceTableOnce, maybeAutoUpdateBackend, readBackendVersion, readDownloadJob, @@ -457,6 +459,11 @@ export async function runLocalModelsStart(): Promise { // for. It still needs a deadline — a stalled-open connection would // otherwise pin the command forever with a progress bar at 12%. signal: AbortSignal.timeout(BACKEND_DOWNLOAD_TIMEOUT_MS), + // Each `models start` is a fresh process (the desktop runs one on + // every switch to the local model), so the release cache never + // survives from one to the next: a check from the last few hours, + // recorded in the data dir, stands instead of a GitHub round trip. + recheckAfterMs: AUTO_UPDATE_RECHECK_MS, onProgress: (p: number, t: number, tot: number) => { const line = renderPullProgress("backend zip", p, t, tot); if (process.stderr.isTTY) process.stderr.write(`\r${line.padEnd(79)}`); @@ -536,11 +543,15 @@ export async function runLocalModelsStart(): Promise { const multiGpu = tensorSplit.length > 0; const { binaryName } = resolvePlatformAsset(); const binPath = resolveServerBinPath(dataDir, binaryName); + // One `--list-devices` for the launch, run only when something asks: + // an `auto` pick here, the context fit in startDaemon (`deviceTableOnce`). + const listDevices = deviceTableOnce(binPath); const device = await resolveManagedDevice( binPath, cfg.localModels.managed.device, { multiGpu, + listDevices, }, ); process.stdout.write( @@ -573,6 +584,12 @@ export async function runLocalModelsStart(): Promise { ...(tpl ? { chatTemplateFile: tpl } : {}), ...(mmprojFile ? { mmprojFile } : {}), ...(dev ? { device: dev } : {}), + // The launch's device table: startDaemon reads the free memory the + // context is fitted into from it, so the binary is not asked again + // when an `auto` pick above already asked it. A `cpu` launch — + // configured, or the CPU rescue below on its swapped-in binary — + // offloads nothing and reads no device memory. + ...(dev !== "cpu" ? { listDevices } : {}), // A configured tensor split only applies while a GPU build // serves — the forced-CPU rescue retry (`startWithDevice("cpu")`) // must not hand multi-GPU split args to the CPU backend. diff --git a/src/cli/run-agent.ts b/src/cli/run-agent.ts index 153850d6e..8bdc27b59 100644 --- a/src/cli/run-agent.ts +++ b/src/cli/run-agent.ts @@ -22,7 +22,7 @@ import { type ApprovalRequest, } from "../approval/approval-gate.js"; import { formatSkillCatalogOmittedNote } from "../skills/index.js"; -import { stderrSink } from "../tracing/structured-logger.js"; +import { createStderrSink } from "../tracing/structured-logger.js"; import { isFailedSessionStatus, type SessionState, @@ -467,7 +467,7 @@ export async function runAgentCommand(args: string[]): Promise { : ""; process.stderr.write(`[${status.channel}] ${status.state}${suffix}\n`); }, - logSinks: [stderrSink()], + logSinks: [createStderrSink()], }, }); diff --git a/src/cli/serve-command.test.ts b/src/cli/serve-command.test.ts index 72644de46..5968e5b95 100644 --- a/src/cli/serve-command.test.ts +++ b/src/cli/serve-command.test.ts @@ -1,6 +1,46 @@ -import { describe, expect, it } from "vitest"; +import { describe, expect, it, vi } from "vitest"; import { resolveBootApprovalLevel } from "../approval/approval-level.js"; +import type { CreateAgentRuntimeOptions } from "../runtime/bootstrap.js"; +import { StructuredLogger } from "../tracing/structured-logger.js"; + +import { serveCommand } from "./serve-command.js"; + +/** What `serveCommand` handed `createAgentRuntime`, the last time it ran. */ +const boot = vi.hoisted(() => ({ + options: null as CreateAgentRuntimeOptions | null, +})); + +// `serveCommand` runs for real up to the runtime it would boot. The +// stand-in keeps the options serve built and refuses, so serve takes its +// failure path and returns without listening on anything. +vi.mock("../runtime/bootstrap.js", async (importOriginal) => { + const actual = + await importOriginal(); + return { + ...actual, + createAgentRuntime: async (options: CreateAgentRuntimeOptions) => { + boot.options = options; + throw new Error("runtime stand-in: not booting"); + }, + }; +}); + +/** Run `serve` up to its runtime and return what it passed there. */ +async function serveBootOptions(): Promise { + boot.options = null; + const stderr = vi + .spyOn(process.stderr, "write") + .mockImplementation(() => true); + try { + // 1: the stand-in refused, and serve said so on (silenced) stderr. + expect(await serveCommand([])).toBe(1); + } finally { + stderr.mockRestore(); + } + if (boot.options === null) throw new Error("serve never booted a runtime"); + return boot.options; +} // `serve` shares the boot contract with `run` and `tui` by calling the // same resolver; this test pins the contract at serve's import site. @@ -18,3 +58,46 @@ describe("serve boot approval level (resolveBootApprovalLevel)", () => { expect(resolveBootApprovalLevel(true, 5)).toBe(5); }); }); + +// The desktop app runs `atag serve` and relays its stderr into agent.log. +// The sinks serve passes are the runtime's only way into that file: serve +// used to pass the sink factory itself, and nothing was ever written. So +// this reads the sinks `serveCommand` really passes, not a helper's. +describe("what serve hands the runtime", () => { + it("log sinks that write the structured log to stderr, at the configured level", async () => { + const { handlers } = await serveBootOptions(); + const sinks = handlers?.logSinks ?? []; + expect(sinks).toHaveLength(1); + const stderr = vi + .spyOn(process.stderr, "write") + .mockImplementation(() => true); + try { + // The runtime builds its logger this way (`createAgentRuntime`): + // the configured level is the gate, the sinks only write. + const logger = new StructuredLogger({ level: "info", sinks }); + logger.debug("below the configured level"); + logger.warn("provider unreachable; parking the turn", { + error: "fetch failed", + causeCode: "ECONNREFUSED", + }); + const written = stderr.mock.calls.map((c) => String(c[0])).join(""); + // The shape the desktop reads a line's level from. + expect(written).toMatch( + /^\[\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}\.\d{3}Z\] WARN provider unreachable; parking the turn \{/, + ); + expect(written).toContain('"causeCode":"ECONNREFUSED"'); + expect(written).not.toContain("below the configured level"); + } finally { + stderr.mockRestore(); + } + }); + + // Under the desktop, a host that died leaves stderr a dead pipe, and + // the next log line exited the process before its teardown ran (the + // port, the session store's turn ends, the serve record). Muted, the + // orphan watch ends it the way SIGTERM would. + it("a broken stderr is muted, not a reason to exit", async () => { + const options = await serveBootOptions(); + expect(options.brokenPipe).toBe("mute"); + }); +}); diff --git a/src/cli/serve-command.ts b/src/cli/serve-command.ts index 58c2de350..a87c36d31 100644 --- a/src/cli/serve-command.ts +++ b/src/cli/serve-command.ts @@ -1,7 +1,7 @@ import { resolveBootApprovalLevel } from "../approval/approval-level.js"; import { getConfig } from "../config/index.js"; import { createAgentRuntime } from "../runtime/bootstrap.js"; -import { stderrSink } from "../tracing/structured-logger.js"; +import { createStderrSink } from "../tracing/structured-logger.js"; import type { AgentRuntime } from "../runtime/bootstrap.js"; import { HELP, parseArgs } from "./serve-args.js"; @@ -67,8 +67,21 @@ export async function serveCommand(args: string[]): Promise { getConfig().agent.approvalLevel, ), traceDefault: true, + // A host that died (the desktop app, Force Quit) leaves stderr a + // dead pipe. The next log line used to exit the process there and + // then, skipping the `finally` below — the port, the session + // store's turn ends, the serve record. Muted, serve goes on until + // the orphan watch ends it through that teardown. + brokenPipe: "mute", handlers: { - logSinks: [stderrSink], + // The structured log goes to stderr, which a host running serve + // relays into its own log (the desktop app's agent.log and its + // Diagnostics pane); `config.log.level` still decides what is + // written. Serve once passed `stderrSink` here, the factory and + // not the sink it builds, and nothing was ever written: the + // factory's parameter makes that a compile error now, and + // `serve-command.test.ts` reads what is passed here. + logSinks: [createStderrSink()], onApprovalRequest: (request) => approvalBus.publish(request), }, }); diff --git a/src/cli/skill-browse.test.ts b/src/cli/skill-browse.test.ts new file mode 100644 index 000000000..bff588eff --- /dev/null +++ b/src/cli/skill-browse.test.ts @@ -0,0 +1,100 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; +import { mkdtempSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; + +import { skillCommand } from "./skill.js"; +import { resetConfigCache } from "../config/index.js"; +import { browseHub, searchHub } from "../skills/hub/index.js"; + +// `skill browse` / `skill search` (skill.ts handleBrowse / handleSearch): +// only the GitHub-tap side is stubbed; ClawHub runs for real against a +// stubbed `fetch`. +vi.mock("../skills/hub/index.js", async (importOriginal) => ({ + ...(await importOriginal()), + browseHub: vi.fn(async () => ({ entries: [], errors: [] })), + searchHub: vi.fn(async () => ({ entries: [], errors: [] })), +})); + +const TAP_ROW = { + identifier: "o/r/x", name: "x", description: "tap row", version: "1.0.0", + repo: "o/r", dir: "x", source: "github" as const, +}; +const CLAW = { + slug: "claw-one", ownerHandle: "me", displayName: "Claw One", + summary: "claw row", stats: { downloads: 5 }, tags: { latest: "1.0.0" }, +}; +// Browse reads `items`, search reads `results`. +const CLAW_BODY = JSON.stringify({ items: [CLAW], results: [CLAW] }); + +describe("skill browse / search", () => { + let stateDir: string; + let stdout = ""; + + beforeEach(() => { + stateDir = mkdtempSync(join(tmpdir(), "atomic-cli-skill-browse-")); + process.env.ATOMIC_AGENT_STATE_DIR = stateDir; + resetConfigCache(); + stdout = ""; + vi.spyOn(process.stdout, "write").mockImplementation((chunk: unknown) => { + stdout += typeof chunk === "string" ? chunk : String(chunk); + return true; + }); + vi.spyOn(process.stderr, "write").mockImplementation(() => true); + }); + + afterEach(() => { + vi.unstubAllGlobals(); + rmSync(stateDir, { recursive: true, force: true }); + delete process.env.ATOMIC_AGENT_STATE_DIR; + resetConfigCache(); + vi.restoreAllMocks(); + }); + + /** + * Run `args` with ClawHub's answer held until the taps have been asked + * (300 ms at most): a command that waits for ClawHub before asking the + * taps shows it in the order of events. + */ + async function run(args: string[]): Promise<{ code: number; order: string[] }> { + const order: string[] = []; + let tapsAsked!: () => void; + const asked = new Promise((r) => { tapsAsked = r; }); + const taps = async () => { + order.push("taps asked"); + tapsAsked(); + return { entries: [TAP_ROW], errors: [] }; + }; + // Set on both, for this run's own `order`: the command calls one of them. + vi.mocked(browseHub).mockImplementation(taps); + vi.mocked(searchHub).mockImplementation(taps); + vi.stubGlobal("fetch", vi.fn(async () => { + order.push("clawhub asked"); + await Promise.race([asked, new Promise((r) => setTimeout(r, 300))]); + order.push("clawhub answered"); + return new Response(CLAW_BODY, { status: 200, headers: { "content-type": "application/json" } }); + })); + const code = await skillCommand(args); + return { code, order }; + } + + it("browse asks ClawHub and the taps side by side, ClawHub's rows first", async () => { + const { code, order } = await run(["browse"]); + expect(code).toBe(0); + expect(order).toEqual(["clawhub asked", "taps asked", "clawhub answered"]); + expect(stdout.split("\n").filter(Boolean)).toEqual([ + "[claw]\t@me/claw-one\t↓5\tclaw row", + "[gh]\to/r/x\t-\ttap row", + ]); + }); + + it("search asks ClawHub and the taps side by side, ClawHub's rows first", async () => { + const { code, order } = await run(["search", "claw"]); + expect(code).toBe(0); + expect(order).toEqual(["clawhub asked", "taps asked", "clawhub answered"]); + expect(stdout.split("\n").filter(Boolean)).toEqual([ + "[claw]\t@me/claw-one\t↓5\tclaw row", + "[gh]\to/r/x\t-\ttap row", + ]); + }); +}); diff --git a/src/cli/skill.ts b/src/cli/skill.ts index 387404ba0..008bb63c5 100644 --- a/src/cli/skill.ts +++ b/src/cli/skill.ts @@ -348,9 +348,13 @@ async function handleBrowse(args: string[]): Promise { } // ClawHub is the primary catalog; `--source owner/repo` narrows to a // single GitHub tap and skips ClawHub (the operator asked for a repo). - const clawEntries = source ? [] : await browseClawHubSafe(null); + // The two are fetched side by side; ClawHub's rows still print first. + const taps = configuredTaps(source); const client = new GithubSkillClient(); - const { entries, errors } = await browseHub(client, configuredTaps(source)); + const [clawEntries, { entries, errors }] = await Promise.all([ + source ? Promise.resolve([]) : browseClawHubSafe(null), + browseHub(client, taps), + ]); return printHubEntries([...clawEntries, ...entries], errors); } @@ -363,9 +367,13 @@ async function handleSearch(args: string[]): Promise { process.stderr.write("usage: atomic-agent skill search \n"); return 2; } - const clawEntries = await browseClawHubSafe(query); + // ClawHub's search and the taps side by side, as in browse. + const taps = configuredTaps(); const client = new GithubSkillClient(); - const { entries, errors } = await searchHub(client, configuredTaps(), query); + const [clawEntries, { entries, errors }] = await Promise.all([ + browseClawHubSafe(query), + searchHub(client, taps, query), + ]); return printHubEntries([...clawEntries, ...entries], errors); } diff --git a/src/cli/trace-formatter.test.ts b/src/cli/trace-formatter.test.ts index c9f3b67c4..474a388ee 100644 --- a/src/cli/trace-formatter.test.ts +++ b/src/cli/trace-formatter.test.ts @@ -62,6 +62,65 @@ describe("formatTraceChronology error", () => { / message=fetch failed \(after "openrouter" failed: openai provider 404: No endpoints found\)$/, ); }); + + it("names the errno right after the message it explains", () => { + expect(render([{ ...row, causeCode: "ECONNREFUSED" }])).toMatch( + / message=fetch failed code=ECONNREFUSED$/, + ); + }); + + it("keeps the errno with the last link's message, ahead of the links before it", () => { + expect( + render([ + { + ...row, + causeCode: "ECONNREFUSED", + fallbackFailures: [ + { + providerId: "openrouter", + reason: "openai provider 404: No endpoints found", + }, + ], + }, + ]), + ).toMatch( + / message=fetch failed code=ECONNREFUSED \(after "openrouter" failed: openai provider 404: No endpoints found\)$/, + ); + }); +}); + +describe("formatTraceChronology provider_waiting", () => { + const row = { + type: "provider_waiting" as const, + seq: 4, + sessionId: "s-1", + ts: Date.parse("2026-09-01T10:00:00.000Z"), + turnIndex: 0, + stepIndex: 2, + attempt: 1, + waitedMs: 0, + maxWaitMs: 300_000, + nextRetryMs: 2_000, + reason: "fetch failed", + }; + + it("prints the wait as it always has when the row carries no errno", () => { + expect(render([row])).toMatch( + / attempt=1 waited=0s\/300s next=2s reason=fetch failed$/, + ); + }); + + it("names the errno last, so a refused server reads apart from a dead network", () => { + expect( + render([ + { + ...row, + cause: { kind: "refused" as const }, + causeCode: "ECONNREFUSED", + }, + ]), + ).toMatch(/ reason=fetch failed code=ECONNREFUSED$/); + }); }); describe("formatTraceChronology profile rows (issue #407)", () => { diff --git a/src/cli/trace-formatter.ts b/src/cli/trace-formatter.ts index 4493311af..8b64e8b53 100644 --- a/src/cli/trace-formatter.ts +++ b/src/cli/trace-formatter.ts @@ -99,11 +99,16 @@ function formatTraceEvent(event: TraceEvent, raw: boolean): string { : "" }`; case "provider_waiting": + // The errno last: `reason` is often a bare `fetch failed`, and + // the code is what says whether the server was refused, timed out + // or never resolved. Rows without one print as they always have. return `${head} attempt=${event.attempt} waited=${Math.round( event.waitedMs / 1000, )}s/${Math.round(event.maxWaitMs / 1000)}s next=${Math.round( event.nextRetryMs / 1000, - )}s reason=${truncate(event.reason, 120, raw)}`; + )}s reason=${truncate(event.reason, 120, raw)}${causeCodeSuffix( + event.causeCode, + )}`; case "provider_recovered": return `${head} waited=${Math.round(event.waitedMs / 1000)}s`; case "completion_truncated": @@ -129,9 +134,11 @@ function formatTraceEvent(event: TraceEvent, raw: boolean): string { const after = (event.fallbackFailures ?? []).map( (f) => `"${f.providerId}" failed: ${f.reason}`, ); - return `${head} message=${event.message}${ - after.length > 0 ? ` (after ${after.join("; ")})` : "" - }`; + // The code sits right after the message it explains, ahead of the + // links that failed before: it belongs to the last link's error. + return `${head} message=${event.message}${causeCodeSuffix( + event.causeCode, + )}${after.length > 0 ? ` (after ${after.join("; ")})` : ""}`; } case "trace_truncated": // The counts are the point of the row: they tell the reader how @@ -150,6 +157,11 @@ function formatTraceEvent(event: TraceEvent, raw: boolean): string { } } +/** ` code=ECONNREFUSED`, or nothing when the row carries no cause code. */ +function causeCodeSuffix(causeCode: string | undefined): string { + return causeCode !== undefined ? ` code=${causeCode}` : ""; +} + function truncate(input: string, max: number, raw: boolean): string { if (raw) return input; if (input.length <= max) return input; diff --git a/src/config/config-file.ts b/src/config/config-file.ts index fd07fdd58..64fdd593c 100644 --- a/src/config/config-file.ts +++ b/src/config/config-file.ts @@ -1,10 +1,4 @@ -import { - existsSync, - mkdirSync, - readFileSync, - renameSync, - writeFileSync, -} from "node:fs"; +import { existsSync, mkdirSync, readFileSync } from "node:fs"; import { dirname, join } from "node:path"; import { @@ -15,6 +9,7 @@ import { USER_CONFIG_VERSION, type UserConfigFile, } from "./config-schema.js"; +import { writeOwnerOnlyFileAtomicSync } from "./owner-only-file.js"; /** Resolve the absolute path to the user config file inside a state dir. */ export function getUserConfigPath(stateDir: string): string { @@ -58,8 +53,9 @@ export function readUserConfigFileSync(path: string): UserConfigFile | null { } /** - * Atomically write the user config file: tmp file + rename. Creates - * the parent directory as needed. + * Atomically write the user config file: tmp file + rename, readable + * by its owner only (0600, `writeOwnerOnlyFileAtomicSync`). Creates the + * parent directory as needed. * * The written `version` is never lower than the one already on disk. A * dozen call sites build their payload by spreading a config object and @@ -92,9 +88,7 @@ export function writeUserConfigFileSync( null, 2, ) + "\n"; - const tmp = `${path}.tmp-${process.pid}`; - writeFileSync(tmp, payload, "utf8"); - renameSync(tmp, path); + writeOwnerOnlyFileAtomicSync(path, payload); } /** diff --git a/src/config/config-paths.ts b/src/config/config-paths.ts index 67c3c5695..9f35c362b 100644 --- a/src/config/config-paths.ts +++ b/src/config/config-paths.ts @@ -1,14 +1,9 @@ -import { - existsSync, - mkdirSync, - readFileSync, - renameSync, - writeFileSync, -} from "node:fs"; +import { existsSync, mkdirSync, readFileSync } from "node:fs"; import { dirname } from "node:path"; import { USER_CONFIG_DEFAULTS } from "./config-schema.js"; import { ConfigValidationError } from "./config-validation-error.js"; +import { writeOwnerOnlyFileAtomicSync } from "./owner-only-file.js"; /** * Dotted-key addressing for `atomic-agent config get|set|unset|list`. @@ -258,7 +253,7 @@ export function writeConfigPath( /** * Atomically write a *sparse* config tree — only the keys the user has - * actually set — using the same tmp + rename discipline as + * actually set — using the same owner-only tmp + rename discipline as * `writeUserConfigFileSync`. * * Separate from `writeUserConfigFileSync` on purpose: that one takes a @@ -276,9 +271,7 @@ export function writeRawUserConfigFileSync( ): void { mkdirSync(dirname(path), { recursive: true }); const payload = JSON.stringify(tree, null, 2) + "\n"; - const tmp = `${path}.tmp-${process.pid}`; - writeFileSync(tmp, payload, "utf8"); - renameSync(tmp, path); + writeOwnerOnlyFileAtomicSync(path, payload); } /** diff --git a/src/config/config-schema.ts b/src/config/config-schema.ts index 6c5292521..f535bf159 100644 --- a/src/config/config-schema.ts +++ b/src/config/config-schema.ts @@ -1506,7 +1506,10 @@ export interface UserManagedLocalLlmConfig { * llama-server context window (`--ctx-size`) for the managed chat * daemon. * - `0` (default) — auto: fit the context to the target device's - * free VRAM at start (see `estimateContextSize`). + * free VRAM at start (see `estimateContextSize`); on a device that + * shares the system's RAM (Apple silicon, an integrated GPU) also + * leaving the system its headroom and the cache within its share of + * physical memory (`resolveContextKvBudgetMiB`). * - a positive value — pin `--ctx-size` exactly, clamped only to the * model's trained context ceiling. */ diff --git a/src/config/owner-only-file.test.ts b/src/config/owner-only-file.test.ts new file mode 100644 index 000000000..d9936643f --- /dev/null +++ b/src/config/owner-only-file.test.ts @@ -0,0 +1,125 @@ +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { + chmodSync, + existsSync, + mkdirSync, + mkdtempSync, + readdirSync, + readFileSync, + rmSync, + statSync, + writeFileSync, +} from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; + +import { getUserConfigPath, writeUserConfigFileSync } from "./config-file.js"; +import { writeRawUserConfigFileSync } from "./config-paths.js"; +import { USER_CONFIG_DEFAULTS } from "./config-schema.js"; +import { + OWNER_ONLY_FILE_MODE, + writeOwnerOnlyFileAtomicSync, +} from "./owner-only-file.js"; + +const posix = process.platform !== "win32"; +const modeOf = (path: string): number => statSync(path).mode & 0o777; + +describe("writeOwnerOnlyFileAtomicSync", () => { + let dir: string; + + beforeEach(() => { + dir = mkdtempSync(join(tmpdir(), "owner-only-")); + }); + + afterEach(() => { + rmSync(dir, { recursive: true, force: true }); + }); + + it("creates the file with the payload, owner read/write only", () => { + const path = join(dir, "config.json"); + writeOwnerOnlyFileAtomicSync(path, '{"a":1}\n'); + expect(readFileSync(path, "utf8")).toBe('{"a":1}\n'); + if (posix) expect(modeOf(path)).toBe(OWNER_ONLY_FILE_MODE); + }); + + it("tightens a file that other accounts could read", () => { + const path = join(dir, "config.json"); + writeFileSync(path, "old\n", { mode: 0o644 }); + if (posix) { + chmodSync(path, 0o644); + expect(modeOf(path)).toBe(0o644); + } + writeOwnerOnlyFileAtomicSync(path, "new\n"); + expect(readFileSync(path, "utf8")).toBe("new\n"); + if (posix) expect(modeOf(path)).toBe(0o600); + }); + + it("leaves no tmp file behind", () => { + const path = join(dir, "config.json"); + writeOwnerOnlyFileAtomicSync(path, "one\n"); + writeOwnerOnlyFileAtomicSync(path, "two\n"); + expect(readdirSync(dir)).toEqual(["config.json"]); + }); + + it("replaces a tmp file a crash left, without inheriting its mode", () => { + const path = join(dir, "config.json"); + const leftover = `${path}.tmp-${process.pid}`; + writeFileSync(leftover, "half a write", { mode: 0o644 }); + if (posix) chmodSync(leftover, 0o644); + writeOwnerOnlyFileAtomicSync(path, "whole\n"); + expect(readFileSync(path, "utf8")).toBe("whole\n"); + expect(existsSync(leftover)).toBe(false); + if (posix) expect(modeOf(path)).toBe(0o600); + }); + + it("throws and cleans up its tmp file when the rename cannot happen", () => { + // A non-empty directory where the file should be: renameSync fails on + // every platform, and nothing is left beside it. + const path = join(dir, "taken"); + mkdirSync(join(path, "inside"), { recursive: true }); + expect(() => writeOwnerOnlyFileAtomicSync(path, "x")).toThrow(); + expect(existsSync(`${path}.tmp-${process.pid}`)).toBe(false); + expect(existsSync(join(path, "inside"))).toBe(true); + }); +}); + +describe("config.json is written owner-only", () => { + let dir: string; + + beforeEach(() => { + dir = mkdtempSync(join(tmpdir(), "owner-only-config-")); + }); + + afterEach(() => { + rmSync(dir, { recursive: true, force: true }); + }); + + it("writeUserConfigFileSync writes 0600 and tightens a 0644 file", () => { + const path = getUserConfigPath(dir); + writeFileSync(path, "{}\n", { mode: 0o644 }); + if (posix) chmodSync(path, 0o644); + writeUserConfigFileSync(path, USER_CONFIG_DEFAULTS); + expect(JSON.parse(readFileSync(path, "utf8")).version).toBe( + USER_CONFIG_DEFAULTS.version, + ); + if (posix) expect(modeOf(path)).toBe(0o600); + }); + + it("writeRawUserConfigFileSync writes 0600 and tightens a 0644 file", () => { + const path = getUserConfigPath(dir); + writeFileSync(path, "{}\n", { mode: 0o644 }); + if (posix) chmodSync(path, 0o644); + writeRawUserConfigFileSync(path, { version: 1, log: { level: "info" } }); + expect(JSON.parse(readFileSync(path, "utf8"))).toEqual({ + version: 1, + log: { level: "info" }, + }); + if (posix) expect(modeOf(path)).toBe(0o600); + }); + + it("a config written into a fresh directory is 0600 too", () => { + const path = join(dir, "nested", "config.json"); + writeUserConfigFileSync(path, USER_CONFIG_DEFAULTS); + if (posix) expect(modeOf(path)).toBe(0o600); + }); +}); diff --git a/src/config/owner-only-file.ts b/src/config/owner-only-file.ts new file mode 100644 index 000000000..23ded57c1 --- /dev/null +++ b/src/config/owner-only-file.ts @@ -0,0 +1,56 @@ +import { chmodSync, renameSync, unlinkSync, writeFileSync } from "node:fs"; + +/** Owner read/write only: the mode `config.json` and `.env` are kept at. */ +export const OWNER_ONLY_FILE_MODE = 0o600; + +/** + * Atomically replace `path` with `payload`, readable and writable by its + * owner only: a tmp file created 0600 beside it, then renamed over it. + * + * `config.json` was written with the default mode (0644 under the usual + * umask), so any other account on the machine could read it, and with it + * whatever the operator keeps there: MCP server env blocks and headers, + * hand-set provider headers, provider keys written inline. `.env` has + * always been 0600 (`setDotenvKey`); this gives the config the same mode + * on every write, so a file that was 0644 is tightened the next time + * anything saves it. + * + * The tmp file is created exclusively, after removing one a crash may + * have left: a leftover keeps the mode it was created with, and writing + * into it would put the new content on disk at that mode until the + * rename. Windows has no POSIX modes; there the chmod is a no-op and the + * file keeps its folder's ACL. + */ +export function writeOwnerOnlyFileAtomicSync( + path: string, + payload: string, +): void { + const tmp = `${path}.tmp-${process.pid}`; + try { + unlinkSync(tmp); + } catch { + // none left behind — the usual case + } + writeFileSync(tmp, payload, { + encoding: "utf8", + mode: OWNER_ONLY_FILE_MODE, + flag: "wx", + }); + try { + renameSync(tmp, path); + } catch (err) { + try { + unlinkSync(tmp); + } catch { + // best effort cleanup + } + throw err; + } + // Creation masks the mode with the umask, which only ever removes bits; + // chmod again so the result does not depend on the platform keeping it. + try { + chmodSync(path, OWNER_ONLY_FILE_MODE); + } catch { + // Windows: no POSIX modes. The rename already won. + } +} diff --git a/src/error-reporting/error-reporter.test.ts b/src/error-reporting/error-reporter.test.ts index 21a17db1f..42852045f 100644 --- a/src/error-reporting/error-reporter.test.ts +++ b/src/error-reporting/error-reporter.test.ts @@ -1,6 +1,9 @@ -import { describe, expect, it, vi } from "vitest"; +import { EventEmitter } from "node:events"; -import { captureError } from "./error-reporter.js"; +import { afterEach, describe, expect, it, vi } from "vitest"; + +import { captureError, guardStdioStream } from "./error-reporter.js"; +import type { GuardedStream } from "./error-reporter.js"; import type { SentryClient } from "./sentry-client.js"; import type { ScrubbedErrorEvent } from "./error-scrubber.js"; @@ -70,3 +73,72 @@ describe("captureError", () => { expect(captured).toHaveLength(1); }); }); + +/** A stdio stream as far as the guard can tell: events, and a write that records. */ +function fakeStdio() { + const written: string[] = []; + const stream = Object.assign(new EventEmitter(), { + write: vi.fn((chunk: unknown, ...rest: unknown[]) => { + written.push(String(chunk)); + const callback = rest.find((r) => typeof r === "function") as + | (() => void) + | undefined; + callback?.(); + return true; + }), + }); + return { stream: stream as unknown as GuardedStream, emitter: stream, written }; +} + +const brokenPipe = () => + Object.assign(new Error("write EPIPE"), { code: "EPIPE", syscall: "write" }); + +describe("guardStdioStream", () => { + afterEach(() => { + vi.restoreAllMocks(); + }); + + it("exits 0 on a broken pipe by default, when the process is its own", () => { + const exit = vi + .spyOn(process, "exit") + .mockImplementation(() => undefined as never); + const { stream, emitter } = fakeStdio(); + guardStdioStream(stream, true, "exit"); + emitter.emit("error", brokenPipe()); + expect(exit).toHaveBeenCalledWith(0); + }); + + it("only swallows a broken pipe in a process a host owns", () => { + const exit = vi + .spyOn(process, "exit") + .mockImplementation(() => undefined as never); + const { stream, emitter } = fakeStdio(); + guardStdioStream(stream, false, "exit"); + emitter.emit("error", brokenPipe()); + expect(exit).not.toHaveBeenCalled(); + }); + + // serve: a host that died must not end the server before its teardown. + it("under mute, stops writing to the broken stream and does not exit", async () => { + const exit = vi + .spyOn(process, "exit") + .mockImplementation(() => undefined as never); + const { stream, emitter, written } = fakeStdio(); + guardStdioStream(stream, true, "mute"); + stream.write("before\n"); + emitter.emit("error", brokenPipe()); + expect(exit).not.toHaveBeenCalled(); + + // Later writes are dropped, and a caller waiting on one still hears back. + const done = vi.fn(); + expect(stream.write("after\n", done)).toBe(true); + stream.write("after again\n", "utf8", done); + await new Promise((resolve) => process.nextTick(resolve)); + expect(written).toEqual(["before\n"]); + expect(done).toHaveBeenCalledTimes(2); + + // A second broken-pipe report changes nothing. + emitter.emit("error", brokenPipe()); + expect(exit).not.toHaveBeenCalled(); + }); +}); diff --git a/src/error-reporting/error-reporter.ts b/src/error-reporting/error-reporter.ts index 2fec3b564..59e9cd1d3 100644 --- a/src/error-reporting/error-reporter.ts +++ b/src/error-reporting/error-reporter.ts @@ -5,6 +5,28 @@ import type { SentryClient } from "./sentry-client.js"; let globalHandlersInstalled = false; let stdioGuardsInstalled = false; +/** + * What a process does when the reader of its stdout or stderr goes away. + * + * - `exit` (the default): shut down cleanly, exit 0 — the pipe closed, + * as `head` hanging up. Right for a command whose output is the point + * (`atag run | head`) and for a TUI whose terminal is gone. + * - `mute`: stop writing to that stream and carry on. `serve` is a + * server, and its stdio is only its log. A host that died — the desktop + * app, Force Quit — is noticed by serve's orphan watch, which ends it + * through the same teardown SIGTERM takes: the port let go, every + * turn's end written to the session store, the serve record cleared. + * Exiting at the first log line after the host died skipped all of it, + * and once serve's structured log reached stderr there was a line + * within seconds. + */ +export type BrokenPipePolicy = "exit" | "mute"; + +export interface GlobalErrorHandlerOptions { + /** See {@link BrokenPipePolicy}. Default `exit`. */ + brokenPipe?: BrokenPipePolicy; +} + /** * Capture an error through the (possibly `null`) client. No-ops when * reporting is disabled. `cancelled` failures are intentionally dropped — @@ -53,22 +75,26 @@ export function captureError( */ export function installGlobalErrorHandlers( getClient: () => SentryClient | null, + options: GlobalErrorHandlerOptions = {}, ): void { if (globalHandlersInstalled) return; globalHandlersInstalled = true; const soleUncaughtHandler = process.listenerCount("uncaughtException") === 0; + const brokenPipe = options.brokenPipe ?? "exit"; - installStdioErrorGuards(soleUncaughtHandler); + installStdioErrorGuards(soleUncaughtHandler, brokenPipe); process.on("uncaughtException", (err) => { const client = getClient(); captureError(client, err, { source: "uncaughtException" }); // Belt-and-braces for the synchronous path the stream guards below // cannot intercept: a dead pipe is not a crash, so it neither prints - // a stack (there is nowhere to print it) nor exits non-zero. + // a stack (there is nowhere to print it) nor exits non-zero. Under + // `mute` it does not end the process either: that is the orphan + // watch's to do, through the teardown. if (isBrokenPipeError(err)) { - if (soleUncaughtHandler) process.exit(0); + if (soleUncaughtHandler && brokenPipe === "exit") process.exit(0); return; } if (soleUncaughtHandler) { @@ -106,26 +132,65 @@ export function installGlobalErrorHandlers( * `ownsProcess` mirrors the `uncaughtException` policy: when this * runtime is the top-level process we shut down cleanly (exit 0 — the * pipe closed, same as `head` hanging up), and when a host embeds us we - * only swallow the error and let the host decide. A non-broken-pipe - * stream error is re-thrown so genuine bugs stay visible. + * only swallow the error and let the host decide. Under the `mute` + * policy the broken stream is silenced instead and the process goes on + * (see {@link BrokenPipePolicy}). A non-broken-pipe stream error is + * re-thrown so genuine bugs stay visible. */ -export function installStdioErrorGuards(ownsProcess: boolean): void { +export function installStdioErrorGuards( + ownsProcess: boolean, + brokenPipe: BrokenPipePolicy = "exit", +): void { if (stdioGuardsInstalled) return; stdioGuardsInstalled = true; for (const stream of [process.stdout, process.stderr]) { - stream.on("error", (err: unknown) => { - if (!isBrokenPipeError(err)) { - queueMicrotask(() => { - throw err; - }); - return; - } - if (ownsProcess) process.exit(0); - }); + guardStdioStream(stream, ownsProcess, brokenPipe); } } +/** The part of a stdio stream the guard touches. */ +export type GuardedStream = Pick; + +/** One stream's guard (see `installStdioErrorGuards`); exported for its test. */ +export function guardStdioStream( + stream: GuardedStream, + ownsProcess: boolean, + brokenPipe: BrokenPipePolicy, +): void { + stream.on("error", (err: unknown) => { + if (!isBrokenPipeError(err)) { + queueMicrotask(() => { + throw err; + }); + return; + } + if (brokenPipe === "mute") { + muteStream(stream); + return; + } + if (ownsProcess) process.exit(0); + }); +} + +/** + * Stop writing to a stream whose reader is gone. Node's stdio streams + * are never really destroyed (their `_destroy` is a no-op that revives + * them), so without this every later write would go to the dead pipe + * and fail again. Each write is dropped here instead; its callback, if + * any, still runs, so nothing waiting on one is left hanging. + */ +function muteStream(stream: GuardedStream): void { + const dropped = (...args: unknown[]): boolean => { + const callback = args.find( + (arg): arg is () => void => typeof arg === "function", + ); + if (callback) process.nextTick(callback); + return true; + }; + stream.write = dropped as GuardedStream["write"]; +} + /** Test-only reset of the idempotency guards. */ export function resetGlobalErrorHandlersForTests(): void { globalHandlersInstalled = false; diff --git a/src/error-reporting/index.ts b/src/error-reporting/index.ts index 6146da598..5a42b5b7b 100644 --- a/src/error-reporting/index.ts +++ b/src/error-reporting/index.ts @@ -33,4 +33,8 @@ export { installStdioErrorGuards, resetGlobalErrorHandlersForTests, } from "./error-reporter.js"; +export type { + BrokenPipePolicy, + GlobalErrorHandlerOptions, +} from "./error-reporter.js"; export { isBrokenPipeError } from "./broken-pipe.js"; diff --git a/src/http/http-server.ts b/src/http/http-server.ts index 85e97ef06..b2b8c3d1d 100644 --- a/src/http/http-server.ts +++ b/src/http/http-server.ts @@ -4,6 +4,7 @@ import { type Server, type ServerResponse, } from "node:http"; +import type { Socket } from "node:net"; import type { AgentRuntime } from "../runtime/bootstrap.js"; import { ApprovalBus } from "./approval-bus.js"; @@ -153,6 +154,14 @@ export function createHttpServer( } }); + // Every open connection, so `close()` can wait for each to close + // (`closeServer`). + const sockets = new Set(); + server.on("connection", (socket: Socket) => { + sockets.add(socket); + socket.once("close", () => sockets.delete(socket)); + }); + return new Promise((resolvePromise, rejectPromise) => { const onError = (err: Error): void => { server.off("listening", onListening); @@ -172,7 +181,7 @@ export function createHttpServer( approvalBus, completionRegistry, undeliveredSteers, - close: () => closeServer(server), + close: () => closeServer(server, sockets), }); }; server.once("error", onError); @@ -255,10 +264,58 @@ function handleRouteError(res: ServerResponse, err: unknown): void { ); } -function closeServer(server: Server): Promise { +/** + * How long `close()` waits for the connections it destroyed to report + * closed. A destroyed socket always does, within a turn of the event + * loop; this is only a backstop so teardown can never hang on one. + */ +const CLOSE_CONNECTIONS_BACKSTOP_MS = 1_000; + +/** + * Stop the server, and resolve once every connection it had has closed. + * + * `server.close()` calls back on the server's own 'close', which Node + * emits on the next tick once `closeAllConnections()` has destroyed the + * sockets — before any of those sockets has emitted its own 'close', + * which comes later, from the handle's close callback. Each response's + * 'close' rides its socket's, and that is what tells a running turn its + * client is gone (`onClientGone` aborts it). `serve` shuts the runtime + * down the moment this resolves, so resolving early meant the session + * store closed while those turns had not even been told to stop, and a + * turn cancelled by quitting the app lost its end. Waiting for every + * socket's 'close' means every request has seen its connection go — and + * aborted its turn — by the time the caller moves on. + */ +function closeServer(server: Server, sockets: Set): Promise { if (!server.listening) return Promise.resolve(); + const open = [...sockets]; return new Promise((resolvePromise) => { - server.close(() => resolvePromise()); + let serverClosed = false; + let pending = open.length; + let backstop: ReturnType | undefined; + const settle = (): void => { + if (!serverClosed || pending > 0) return; + clearTimeout(backstop); + resolvePromise(); + }; + // Registered after each response's own 'close' listener, so by the + // time this one runs the request has already been told. + for (const socket of open) { + socket.once("close", () => { + pending -= 1; + settle(); + }); + } + server.close(() => { + serverClosed = true; + settle(); + }); server.closeAllConnections?.(); + if (pending > 0) { + backstop = setTimeout(() => { + pending = 0; + settle(); + }, CLOSE_CONNECTIONS_BACKSTOP_MS); + } }); } diff --git a/src/http/route-sessions.ts b/src/http/route-sessions.ts index 709ae1a6e..6ae2c6b02 100644 --- a/src/http/route-sessions.ts +++ b/src/http/route-sessions.ts @@ -36,6 +36,10 @@ export function createListSessionsHandler(): HttpHandler { sessions: sessions.map((s) => ({ id: s.id, workingDir: s.workingDir, + // `running` while a turn is in flight on the session, in this + // process or another on the same state dir; after it, the status + // that turn ended on (`cancelled` also for a turn the agent was + // shut down or killed in the middle of — see `SessionStore`). status: s.status, turnCount: s.turnCount, stepCount: s.stepCount, diff --git a/src/http/session-status.test.ts b/src/http/session-status.test.ts new file mode 100644 index 000000000..acb394ed6 --- /dev/null +++ b/src/http/session-status.test.ts @@ -0,0 +1,236 @@ +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { join } from "node:path"; + +import type { CompletionResult } from "../llm/llama-server-client.js"; +import { SessionStore } from "../session/index.js"; + +import { COMPLETION_ID_HEADER } from "./openai-chat-completions.js"; +import { startTestHarness, type Harness } from "./test-harness.js"; + +/** + * The status a chat's session row holds over `serve`, for each way the + * desktop ends a turn: its Stop (`POST .../cancel`), a client that drops + * the stream, and the app quitting mid-turn — the last being the one + * that used to lose the turn's end: the store closed before the aborted + * turn could write it, and a first message in a new chat read back as + * `pending` with nothing in it. + */ + +function reply(text: string): CompletionResult { + return { + content: JSON.stringify({ tool: "reply", args: { text } }), + reasoningContent: "", + stop: true, + truncated: false, + timing: { promptMs: 0, predictedMs: 0, promptTokens: 1, predictedTokens: 1 }, + cacheHitTokens: 0, + slotId: 0, + modelId: null, + }; +} + +/** + * The turn's own model call holds until its signal aborts and then + * rejects `unwindMs` later, as an aborted request takes a moment to come + * back. Side calls (naming, reflection) on other ids answer at once. + */ +function hangingModel(unwindMs = 0) { + const state: { + target: string; + entered: number; + aborted: number; + /** The signal the turn's model call was made with. */ + signal: AbortSignal | undefined; + } = { target: "", entered: 0, aborted: 0, signal: undefined }; + const llamaComplete = async (params: { + sessionId: string; + signal?: AbortSignal; + }): Promise => { + if (params.sessionId !== state.target) return reply("ok"); + state.entered += 1; + const signal = params.signal; + state.signal = signal; + await new Promise((resolve) => { + if (signal?.aborted) { + resolve(); + return; + } + signal?.addEventListener("abort", () => resolve(), { once: true }); + }); + state.aborted += 1; + if (unwindMs > 0) { + await new Promise((resolve) => setTimeout(resolve, unwindMs)); + } + throw signal?.reason ?? new DOMException("aborted", "AbortError"); + }; + return { state, llamaComplete }; +} + +async function waitFor(check: () => boolean, ms = 5_000): Promise { + const deadline = Date.now() + ms; + while (!check()) { + if (Date.now() > deadline) return false; + await new Promise((resolve) => setTimeout(resolve, 10)); + } + return true; +} + +function chatInit(sessionId: string, signal: AbortSignal): RequestInit { + return { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ + stream: true, + session_id: sessionId, + messages: [{ role: "user", content: "write a long story" }], + }), + signal, + }; +} + +async function storedStatus(baseUrl: string, id: string): Promise { + const res = await fetch(`${baseUrl}/api/sessions/${id}`); + const body = (await res.json()) as { status?: string }; + return body.status ?? `<${res.status}>`; +} + +/** Poll until the row stops saying `running`, then hand back what it says. */ +async function settledStatus(baseUrl: string, id: string): Promise { + const deadline = Date.now() + 5_000; + let status = await storedStatus(baseUrl, id); + while (status === "running" && Date.now() < deadline) { + await new Promise((resolve) => setTimeout(resolve, 20)); + status = await storedStatus(baseUrl, id); + } + return status; +} + +describe("a chat's session status over serve", () => { + let harness: Harness; + let model: ReturnType; + + beforeEach(async () => { + model = hangingModel(20); + harness = await startTestHarness({ llamaComplete: model.llamaComplete }); + }); + + afterEach(async () => { + await harness.cleanup(); + }); + + it("is running while the turn runs, and cancelled once the desktop's Stop cancels it", async () => { + model.state.target = "status-stop"; + const res = await fetch( + `${harness.baseUrl}/v1/chat/completions`, + chatInit("status-stop", new AbortController().signal), + ); + expect(res.status).toBe(200); + const completionId = res.headers.get(COMPLETION_ID_HEADER)!; + const body = res.text().catch(() => ""); + expect(await waitFor(() => model.state.entered === 1)).toBe(true); + expect(await storedStatus(harness.baseUrl, "status-stop")).toBe("running"); + + const cancel = await fetch( + `${harness.baseUrl}/v1/chat/completions/${completionId}/cancel`, + { method: "POST" }, + ); + expect(cancel.status).toBe(200); + await cancel.body?.cancel(); + // The stream closes once the turn has ended and written its end. + await body; + expect(await storedStatus(harness.baseUrl, "status-stop")).toBe( + "cancelled", + ); + }); + + it("is cancelled once a client that dropped the stream has ended the turn", async () => { + model.state.target = "status-drop"; + const client = new AbortController(); + const res = await fetch( + `${harness.baseUrl}/v1/chat/completions`, + chatInit("status-drop", client.signal), + ); + expect(res.status).toBe(200); + expect(await waitFor(() => model.state.entered === 1)).toBe(true); + client.abort(); + expect(await waitFor(() => model.state.aborted === 1, 2_000)).toBe(true); + expect(await settledStatus(harness.baseUrl, "status-drop")).toBe( + "cancelled", + ); + }); +}); + +describe("a chat's session when serve shuts down mid-turn", () => { + it("is stored cancelled, with the message that started the turn", async () => { + // The app quitting: `serve` drops every connection (which stops the + // turn) and shuts the runtime down. The aborted model call comes + // back 50 ms later — after the store used to be closed. + const model = hangingModel(50); + model.state.target = "status-quit"; + const harness = await startTestHarness({ + llamaComplete: model.llamaComplete, + }); + try { + const res = await fetch( + `${harness.baseUrl}/v1/chat/completions`, + chatInit("status-quit", new AbortController().signal), + ); + expect(res.status).toBe(200); + const drained = res.text().catch(() => ""); + expect(await waitFor(() => model.state.entered === 1)).toBe(true); + + await harness.handle.close(); + await harness.runtime.shutdown(); + await drained; + + const store = new SessionStore({ + dbFile: join(harness.stateDir, "sessions.sqlite"), + }); + try { + const stored = store.load("status-quit"); + expect(stored?.status).toBe("cancelled"); + expect( + stored?.turns.flatMap((turn) => + turn.kind === "user" ? [turn.text] : [], + ), + ).toEqual(["write a long story"]); + // The turn's own end, not shutdown's stand-in for it. + expect(stored?.lastError).toBeNull(); + } finally { + store.close(); + } + } finally { + await harness.cleanup(); + } + }); +}); + +describe("closing serve with a streamed turn in flight", () => { + it("has told the turn its client is gone by the time close() resolves", async () => { + // `serve` shuts the runtime down the moment the server's close() + // resolves. Node's own close callback comes before the sockets have + // closed, so the turn had not yet been aborted when the runtime began + // closing its stores under it. + const model = hangingModel(); + model.state.target = "status-close"; + const harness = await startTestHarness({ + llamaComplete: model.llamaComplete, + }); + try { + const res = await fetch( + `${harness.baseUrl}/v1/chat/completions`, + chatInit("status-close", new AbortController().signal), + ); + expect(res.status).toBe(200); + const drained = res.text().catch(() => ""); + expect(await waitFor(() => model.state.entered === 1)).toBe(true); + expect(model.state.signal?.aborted).toBe(false); + + await harness.handle.close(); + expect(model.state.signal?.aborted).toBe(true); + await drained; + } finally { + await harness.cleanup(); + } + }); +}); diff --git a/src/llm/fallback/failed-links.ts b/src/llm/fallback/failed-links.ts new file mode 100644 index 000000000..55cb0ad6b --- /dev/null +++ b/src/llm/fallback/failed-links.ts @@ -0,0 +1,25 @@ +import { + classifyProviderWaitCause, + type ProviderWaitFailure, +} from "../reliability/provider-wait-cause.js"; +import { describeReason } from "./describe-reason.js"; +import { readFailedAttempts } from "./failed-attempts.js"; + +/** + * The links that failed before `err` (`readFailedAttempts`), each with + * its own cause, in the order they were tried. + * + * What a host says first when a turn waits on, or fails at, a later + * link: the provider the user picked is usually the first entry, and + * why it failed (an account out of funds, a refused key) is not the last + * link's `fetch failed`. Item 40: AI/ML API answered "You've run out of + * funds", the chain fell over to a stopped local server, and the window + * named only that server. + */ +export function describeFailedLinks(err: unknown): ProviderWaitFailure[] { + return readFailedAttempts(err).map((a) => ({ + providerId: a.providerId, + reason: describeReason(a.error), + cause: classifyProviderWaitCause(a.error), + })); +} diff --git a/src/llm/fallback/link-failure-kind.test.ts b/src/llm/fallback/link-failure-kind.test.ts index dcbdba09a..10ad708ca 100644 --- a/src/llm/fallback/link-failure-kind.test.ts +++ b/src/llm/fallback/link-failure-kind.test.ts @@ -4,7 +4,12 @@ import { LlamaServerError } from "../llama-server-client.js"; import { OpenAiHttpError } from "../provider/openai/openai-http.js"; import { SubscriptionCliAuthError } from "../provider/subscription-cli/subscription-cli-errors.js"; import { ModelError } from "../reliability/llm-failures.js"; -import { isCredentialRejection, isOutageFailure } from "./link-failure-kind.js"; +import { parseProviderErrorBody } from "../provider/openai/parse-provider-error-body.js"; +import { + isBillingRefusal, + isCredentialRejection, + isOutageFailure, +} from "./link-failure-kind.js"; function http( status: number | null, @@ -14,7 +19,83 @@ function http( return new OpenAiHttpError(message, status, "http://x/y", timedOut, null, "p"); } +/** A provider error as `httpErrorFromResponse` builds it: the body parsed beside the message. */ +function withBody(status: number, body: string, label = "aimlapi"): OpenAiHttpError { + return new OpenAiHttpError( + `openai provider ${status}: ${body}`, + status, + "https://api.aimlapi.com/v1/chat/completions", + false, + null, + label, + undefined, + { body: parseProviderErrorBody(body) }, + ); +} + +/** Item 40's field body: AI/ML API with a good key and an empty account. */ +const OUT_OF_FUNDS = JSON.stringify({ + title: "Forbidden", + status: 403, + message: + "You've run out of funds. Please top up your balance or update your payment method to continue: https://aimlapi.com/app/billing", +}); + +describe("isBillingRefusal", () => { + it("is AI/ML API's 403 for an empty account, a 402, and OpenAI's insufficient_quota", () => { + expect(isBillingRefusal(withBody(403, OUT_OF_FUNDS))).toBe(true); + expect(isBillingRefusal(withBody(402, '{"error":{"message":"Insufficient Balance"}}'))).toBe(true); + expect(isBillingRefusal(http(402))).toBe(true); + expect( + isBillingRefusal( + withBody(429, '{"error":{"message":"You exceeded your current quota","code":"insufficient_quota"}}', "openai"), + ), + ).toBe(true); + }); + + it("is not a 429 in a rate limit's words, nor a 403 about authentication that mentions billing", () => { + for (const message of [ + "Too many requests. Please top up your account to increase your rate limits.", + "Out of credits for this minute", + ]) { + const limited = withBody(429, JSON.stringify({ error: { message } })); + expect(isBillingRefusal(limited), message).toBe(false); + expect(isOutageFailure(limited), message).toBe(true); + } + for (const message of [ + "Authentication failed. Please check your billing details.", + "Invalid token. Check billing.", + ]) { + const refused = withBody(403, JSON.stringify({ error: { message } })); + expect(isBillingRefusal(refused), message).toBe(false); + expect(isCredentialRejection(refused), message).toBe(true); + } + }); + + it("is not a refused key, a rate limit, a cooldown or an outage", () => { + expect(isBillingRefusal(withBody(403, '{"error":{"message":"Invalid API key"}}'))).toBe(false); + expect(isBillingRefusal(http(401))).toBe(false); + expect(isBillingRefusal(http(429))).toBe(false); + expect( + isBillingRefusal(http(402, 'openai provider 402: {"error":{"code":"in_flight_budget_exhausted"}}')), + ).toBe(false); + expect(isBillingRefusal(http(null))).toBe(false); + expect(isBillingRefusal(http(402, "boom", true))).toBe(false); + expect(isBillingRefusal(new TypeError("fetch failed"))).toBe(false); + expect(isBillingRefusal(new LlamaServerError("nope", 402, "http://l"))).toBe(false); + }); +}); + describe("isCredentialRejection", () => { + it("is not a 403 about the account's funds, whatever else it mentions", () => { + expect(isCredentialRejection(withBody(403, OUT_OF_FUNDS))).toBe(false); + expect( + isCredentialRejection( + withBody(403, '{"error":{"message":"Your API key has insufficient balance"}}'), + ), + ).toBe(false); + }); + it("is a cloud 401: the key is wrong, missing, or could not be sent", () => { expect(isCredentialRejection(http(401))).toBe(true); expect( @@ -104,6 +185,13 @@ describe("isOutageFailure", () => { for (const status of [400, 401, 403, 404]) { expect(isOutageFailure(http(status))).toBe(false); } + // An empty account, even on a 429 (item 40). + expect(isOutageFailure(withBody(403, OUT_OF_FUNDS))).toBe(false); + expect( + isOutageFailure( + withBody(429, '{"error":{"message":"You exceeded your current quota","code":"insufficient_quota"}}', "openai"), + ), + ).toBe(false); expect( isOutageFailure( http(402, "openai provider 402: This request requires more credits"), diff --git a/src/llm/fallback/link-failure-kind.ts b/src/llm/fallback/link-failure-kind.ts index 517b9eb73..fcec952ac 100644 --- a/src/llm/fallback/link-failure-kind.ts +++ b/src/llm/fallback/link-failure-kind.ts @@ -1,20 +1,20 @@ import { LlamaServerError } from "../llama-server-client.js"; -import { OpenAiHttpError } from "../provider/openai/openai-http.js"; +import { + isCreditExhausted, + OpenAiHttpError, +} from "../provider/openai/openai-http.js"; +import { CREDENTIAL_WORDING } from "../provider/openai/parse-provider-error-body.js"; import { TransportError } from "../reliability/llm-failures.js"; import { isNetworkError } from "../reliability/network-error.js"; import { readProviderErrorVerdict } from "../reliability/provider-error-verdict.js"; /** - * Two questions the chain asks about a link that failed, beyond + * Three questions the chain asks about a link that failed, beyond * `shouldAdvance`'s "is another link worth a try". Every cloud failure * advances, so the advance decision cannot tell a service that is down - * from one that refused what it was sent; these two can. + * from one that refused what it was sent; these can. */ -/** A provider's words for a credential problem, in a 403's message. */ -const KEY_WORDING = - /\b(?:api[ _-]?key|credentials?|unauthori[sz]ed|unauthenticated|authenticat\w*|access[ _-]?token|invalid[ _-]?token)\b/i; - /** * Did the link refuse its credentials? * @@ -33,7 +33,27 @@ export function isCredentialRejection(err: unknown): boolean { if (!(err instanceof OpenAiHttpError) || err.timedOut) return false; if (err.status === 401) return true; if (err.status !== 403) return false; - return err.keyProblem !== undefined || KEY_WORDING.test(err.message); + if (err.keyProblem !== undefined) return true; + // "Please top up your balance or update your payment method": the + // account, not the key (item 40), whatever else the body mentions. + if (isCreditExhausted(err)) return false; + // The same words the billing reading defers to (one rule for both). + return CREDENTIAL_WORDING.test(err.message); +} + +/** + * Did the link refuse because the account cannot pay? + * + * A 402, or a 403 / 429 whose body says the account is out of funds or + * credit (`isCreditExhausted`: AI/ML API's 403 "You've run out of funds", + * OpenAI's 429 `insufficient_quota`). The link answered and said no; the + * key is fine and waiting changes nothing until someone tops up. Like a + * refused key it outranks the outage a later link reports when the chain + * runs out (`runWithFallback`), so the turn ends on its sentence instead + * of parking on a stopped local server (item 40). + */ +export function isBillingRefusal(err: unknown): boolean { + return err instanceof OpenAiHttpError && !err.timedOut && isCreditExhausted(err); } /** @@ -48,6 +68,8 @@ export function isCredentialRejection(err: unknown): boolean { * completion) is the link saying no, and waiting does not change it. */ export function isOutageFailure(err: unknown): boolean { + // An empty account is a "no", even on a 429. + if (isBillingRefusal(err)) return false; const status = statusOf(err); if (status === null) return true; if (status !== undefined) { diff --git a/src/llm/fallback/log-fallback-advance.ts b/src/llm/fallback/log-fallback-advance.ts index b3dba019b..351243422 100644 --- a/src/llm/fallback/log-fallback-advance.ts +++ b/src/llm/fallback/log-fallback-advance.ts @@ -1,4 +1,5 @@ import type { StructuredLogger } from "../../tracing/structured-logger.js"; +import { readErrnoCode } from "../errno-code.js"; import { describeReason } from "./describe-reason.js"; /** What `ProviderFallbackChain` logs through; a subset so tests can pass a spy. */ @@ -18,6 +19,17 @@ export type FallbackLogger = Pick; * text the switch notice already shows in chat. The HTTP clients build * their messages from the status, the URL and the response body; request * headers never enter them, so no API key reaches this line. + * + * `causeCode` is the errno the transport left on the error's `cause` + * chain (`ECONNREFUSED`, `ENOTFOUND`, `UND_ERR_SOCKET`, …), present only + * when there is one: a `status: null` with `reason: "fetch failed"` is + * the same line for a link that is not running and one the network + * cannot reach. Named as the trace's `error` row and the parking line + * name it; a bare `code` reads as a provider's own error code, which is + * what `code` means elsewhere in these lines. It needs no category gate, + * unlike the trace's `error` row: only a failure `shouldAdvance` accepted + * (`transport` or `model`) reaches this line, and a cancellation never + * advances. */ export function logFallbackAdvance( logger: FallbackLogger | undefined, @@ -29,10 +41,12 @@ export function logFallbackAdvance( error !== null && typeof error === "object" ? (error as { status?: unknown }).status : undefined; + const causeCode = readErrnoCode(error); logger.warn("provider failed; falling over to the next link", { from: advance.from, to: advance.to, ...(typeof status === "number" || status === null ? { status } : {}), + ...(causeCode !== undefined ? { causeCode } : {}), reason: describeReason(error), ...(advance.sessionId ? { sessionId: advance.sessionId } : {}), }); diff --git a/src/llm/fallback/partition-state.ts b/src/llm/fallback/partition-state.ts index 5432bf463..31b4f47a2 100644 --- a/src/llm/fallback/partition-state.ts +++ b/src/llm/fallback/partition-state.ts @@ -61,6 +61,13 @@ export interface PartitionState { * still the route this partition is running on. */ fallbackServed: boolean; + /** + * The fallback link that answered last since the chain left the + * primary: the route this partition is running on. Null on the primary + * and until a fallback answers. Unlike `overrideServed` it survives a + * failed probe moving the pointer onto the same link again. + */ + servingId: string | null; } export function freshPartition(): PartitionState { @@ -71,6 +78,7 @@ export function freshPartition(): PartitionState { overrideCause: null, overrideServed: false, fallbackServed: false, + servingId: null, }; } @@ -129,4 +137,5 @@ export function clearOverride(p: PartitionState): void { p.overrideCause = null; p.overrideServed = false; p.fallbackServed = false; + p.servingId = null; } diff --git a/src/llm/fallback/provider-fallback-chain.test.ts b/src/llm/fallback/provider-fallback-chain.test.ts index 3652f12b8..b2fc26786 100644 --- a/src/llm/fallback/provider-fallback-chain.test.ts +++ b/src/llm/fallback/provider-fallback-chain.test.ts @@ -437,6 +437,30 @@ describe("ProviderFallbackChain", () => { expect(chain.hasFallbackServed()).toBe(true); }); + it("names the fallback that answered last as the serving one, through a failed probe, until the primary is back", () => { + const chain = new ProviderFallbackChain({ + resolve: () => chainOf(["primary", "b", "c"]), + now: makeClock().now, + }); + chain.advanceFrom("primary", http(503)); + expect(chain.isServingFallback("b")).toBe(false); + chain.recordSuccess("b", false); + expect(chain.isServingFallback("b")).toBe(true); + expect(chain.isServingFallback("c")).toBe(false); + expect(chain.isServingFallback("primary")).toBe(false); + // A failed probe moves the pointer onto `b` again; `b` is still the route. + chain.advanceFrom("primary", http(503)); + expect(chain.isServingFallback("b")).toBe(true); + // `c` answers after `b` failed: `c` is the route now. + chain.advanceFrom("b", new TypeError("fetch failed")); + chain.recordSuccess("c", false); + expect(chain.isServingFallback("c")).toBe(true); + expect(chain.isServingFallback("b")).toBe(false); + // The primary answers a probe: no fallback is serving. + chain.recordSuccess("primary", true); + expect(chain.isServingFallback("c")).toBe(false); + }); + it("does not apply to a primary that is down: its cooldown holds as before", () => { const chain = new ProviderFallbackChain({ resolve: () => chainOf(["primary", "backup"]), diff --git a/src/llm/fallback/provider-fallback-chain.ts b/src/llm/fallback/provider-fallback-chain.ts index 4e054320a..e999f671d 100644 --- a/src/llm/fallback/provider-fallback-chain.ts +++ b/src/llm/fallback/provider-fallback-chain.ts @@ -190,6 +190,7 @@ export class ProviderFallbackChain { if (id === p.overrideId) { p.overrideServed = true; p.fallbackServed = true; + p.servingId = id; } const { chain } = this.resolve(); @@ -231,6 +232,15 @@ export class ProviderFallbackChain { return this.partitions.get(partitionKey)?.fallbackServed ?? false; } + /** + * Whether `id` is the fallback `partitionKey` has been running on: the + * link that answered last since the chain left the primary. + */ + isServingFallback(id: string, partitionKey = DEFAULT_PARTITION): boolean { + const servingId = this.partitions.get(partitionKey)?.servingId ?? null; + return servingId !== null && servingId === id; + } + /** The override the next call starts on (no stand-in), for the route note. */ standingOverrideFor(partitionKey: string): string | null { const p = this.partitions.get(partitionKey); @@ -252,6 +262,7 @@ export class ProviderFallbackChain { p.announcedOverride = false; p.overrideServed = false; p.fallbackServed = false; + p.servingId = null; } else if (p.overrideId) { // Chain continued past a dead deeper link — keep the override // pointed at the newest working candidate. diff --git a/src/llm/fallback/run-with-fallback-stop.test.ts b/src/llm/fallback/run-with-fallback-stop.test.ts new file mode 100644 index 000000000..8771bdb48 --- /dev/null +++ b/src/llm/fallback/run-with-fallback-stop.test.ts @@ -0,0 +1,91 @@ +import { describe, expect, it } from "vitest"; + +import { DEFAULT_FALLBACK_TIMING } from "./fallback-config.js"; +import { readFailedAttempts, readFailingLink } from "./failed-attempts.js"; +import { + ProviderFallbackChain, + type ProviderSwitchNotice, +} from "./provider-fallback-chain.js"; +import { runWithFallback } from "./run-with-fallback.js"; +import { TransportError } from "../reliability/llm-failures.js"; + +/** + * A stop is not an outage. When the user stops a turn as its stream ends, + * the request can come back as `terminated` or `fetch failed` — the + * shape of a provider that went away. Falling over on it armed the + * primary's breaker and set the sticky override, so the next turn ran on + * the fallback without anyone having asked for it. + */ + +function makeChain(notices: ProviderSwitchNotice[]): ProviderFallbackChain { + return new ProviderFallbackChain({ + resolve: () => ({ + chain: ["primary", "backup"], + timing: DEFAULT_FALLBACK_TIMING, + }), + noticeSink: (notice) => notices.push(notice), + }); +} + +describe("runWithFallback when its caller has stopped the call", () => { + it("rethrows the error without falling over or arming anything", async () => { + const notices: ProviderSwitchNotice[] = []; + const chain = makeChain(notices); + const controller = new AbortController(); + const seen: string[] = []; + + const thrown = await runWithFallback( + chain, + async (id) => { + seen.push(id); + if (id === "primary") { + controller.abort(); + throw new TransportError("terminated", null, ""); + } + return `answer from ${id}`; + }, + "s-stop", + controller.signal, + ).catch((err: unknown) => err); + + expect(thrown).toBeInstanceOf(TransportError); + expect(seen).toEqual(["primary"]); + expect(notices).toEqual([]); + expect(chain.activeOverrideFor("s-stop")).toBeNull(); + // Nothing recorded beside it: a cancelled turn is not the primary's + // failure. The link is still named, as for every thrown error. + expect(readFailedAttempts(thrown)).toEqual([]); + expect(readFailingLink(thrown)).toBe("primary"); + + // The next turn starts on the primary, as if nothing had happened. + const next: string[] = []; + const out = await runWithFallback( + chain, + async (id) => { + next.push(id); + return id; + }, + "s-stop", + new AbortController().signal, + ); + expect(out).toBe("primary"); + expect(next).toEqual(["primary"]); + }); + + it("still falls over on the same failure when nobody stopped the call", async () => { + const notices: ProviderSwitchNotice[] = []; + const chain = makeChain(notices); + const out = await runWithFallback( + chain, + async (id) => { + if (id === "primary") throw new TransportError("terminated", null, ""); + return `answer from ${id}`; + }, + "s-outage", + new AbortController().signal, + ); + expect(out).toBe("answer from backup"); + expect(notices).toHaveLength(1); + expect(chain.activeOverrideFor("s-outage")).toBe("backup"); + }); +}); diff --git a/src/llm/fallback/run-with-fallback.test.ts b/src/llm/fallback/run-with-fallback.test.ts index e660b7c99..1118ccc09 100644 --- a/src/llm/fallback/run-with-fallback.test.ts +++ b/src/llm/fallback/run-with-fallback.test.ts @@ -9,6 +9,7 @@ import { ProviderFallbackChain } from "./provider-fallback-chain.js"; import type { ProviderSwitchNotice } from "./provider-fallback-chain.js"; import { DEFAULT_FALLBACK_TIMING } from "./fallback-config.js"; import { OpenAiHttpError } from "../provider/openai/openai-http.js"; +import { parseProviderErrorBody } from "../provider/openai/parse-provider-error-body.js"; import { GrammarError } from "../reliability/llm-failures.js"; import { classifyFailure, @@ -382,6 +383,132 @@ describe("runWithFallback", () => { }); }); + /* Item 40: AI/ML API answered 403 "You've run out of funds", DashScope + had no key, and the stopped local server's `fetch failed` parked the + turn; the window named the local server. An empty account is a + refusal like a refused key, wherever on the route it comes from. */ + describe("an account that cannot pay", () => { + const outOfFunds = (label = "aimlapi"): OpenAiHttpError => { + const body = + '{"title":"Forbidden","status":403,"message":"You\'ve run out of funds. Please top up your balance or update your payment method to continue"}'; + return new OpenAiHttpError( + `openai provider 403: ${body}`, + 403, + "https://api.aimlapi.com/v1/chat/completions", + false, + null, + label, + undefined, + { body: parseProviderErrorBody(body) }, + ); + }; + function chainAt(ids: string[], now: () => number): ProviderFallbackChain { + return new ProviderFallbackChain({ + resolve: () => ({ chain: ids, timing: DEFAULT_FALLBACK_TIMING }), + now, + }); + } + + it("on the primary, throws its refusal rather than the last link's outage", async () => { + const chain = makeChain(["aimlapi", "local"]); + const refusal = outOfFunds(); + await expect( + runWithFallback(chain, async (id) => { + throw id === "aimlapi" ? refusal : new TypeError("fetch failed"); + }), + ).rejects.toBe(refusal); + expect(readFailingLink(refusal)).toBe("aimlapi"); + expect(readFailedAttempts(refusal)).toEqual([]); + }); + + it("on the primary, still lets a fallback that answers serve the call", async () => { + const chain = makeChain(["aimlapi", "backup"]); + await expect( + runWithFallback(chain, async (id) => { + if (id === "aimlapi") throw outOfFunds(); + return id; + }), + ).resolves.toBe("backup"); + }); + + it("on the primary, asks it again next call: the stand-in proved nothing", async () => { + const chain = chainAt(["aimlapi", "local"], () => 1_000); + await expect( + runWithFallback(chain, async (id) => { + throw id === "aimlapi" ? outOfFunds() : new TypeError("fetch failed"); + }), + ).rejects.toBeInstanceOf(OpenAiHttpError); + const seen: string[] = []; + await expect( + runWithFallback(chain, async (id) => { + seen.push(id); + return id; + }), + ).resolves.toBe("aimlapi"); + expect(seen).toEqual(["aimlapi"]); + }); + + it("on the fallback that has been serving, throws its refusal with the links before it", async () => { + let now = 1_000; + const chain = chainAt(["primary", "cloud2", "local"], () => now); + // The primary goes down and cloud2 serves: the route in use. + await runWithFallback(chain, async (id) => { + if (id === "primary") throw http(503); + return id; + }); + now += DEFAULT_FALLBACK_TIMING.probeThrottleMs; + // A probe finds the primary still down, cloud2's account is now + // empty, and the local server is stopped. + const refusal = outOfFunds("cloud2"); + await expect( + runWithFallback(chain, async (id) => { + if (id === "primary") throw http(503); + if (id === "cloud2") throw refusal; + throw new TypeError("fetch failed"); + }), + ).rejects.toBe(refusal); + expect(readFailingLink(refusal)).toBe("cloud2"); + expect(readFailedAttempts(refusal).map((a) => a.providerId)).toEqual([ + "primary", + ]); + }); + + it("on a fallback that never served, leaves the last link's error, as before", async () => { + const chain = chainAt(["primary", "cloud2", "local"], () => 1_000); + const down = new TypeError("fetch failed"); + await expect( + runWithFallback(chain, async (id) => { + if (id === "primary") throw http(503); + if (id === "cloud2") throw outOfFunds("cloud2"); + throw down; + }), + ).rejects.toBe(down); + expect(readFailedAttempts(down).map((a) => a.providerId)).toEqual([ + "primary", + "cloud2", + ]); + }); + + it("on the primary, leaves the outage path to a fallback that has been serving", async () => { + let now = 1_000; + const chain = chainAt(["primary", "backup"], () => now); + await runWithFallback(chain, async (id) => { + if (id === "primary") throw http(503); + return id; + }); + now += DEFAULT_FALLBACK_TIMING.probeThrottleMs; + const outage = new TypeError("fetch failed"); + await expect( + runWithFallback(chain, async (id) => { + throw id === "primary" ? outOfFunds("primary") : outage; + }), + ).rejects.toBe(outage); + expect(readFailedAttempts(outage).map((a) => a.providerId)).toEqual([ + "primary", + ]); + }); + }); + it("marks the thrown error with the link that threw it", async () => { const chain = makeChain(["cloud", "local"]); const last = new TypeError("fetch failed"); @@ -446,6 +573,54 @@ describe("runWithFallback", () => { ); }); + it("names the errno the failed link's transport left behind", async () => { + const warn = vi.fn(); + const chain = new ProviderFallbackChain({ + resolve: () => ({ + chain: ["cloud", "local"], + timing: DEFAULT_FALLBACK_TIMING, + }), + logger: { warn }, + }); + // No response, so no status: without the errno this line reads + // the same for a host the network cannot resolve and for one that + // refused the connection. + const unreachable = new OpenAiHttpError( + "fetch failed", + null, + "http://x", + false, + null, + "p", + "ENOTFOUND", + ); + + await expect( + runWithFallback( + chain, + async (id) => { + if (id === "cloud") throw unreachable; + return id; + }, + "s-2", + ), + ).resolves.toBe("local"); + + expect(warn.mock.calls).toEqual([ + [ + "provider failed; falling over to the next link", + { + from: "cloud", + to: "local", + status: null, + causeCode: "ENOTFOUND", + reason: "fetch failed", + sessionId: "s-2", + }, + ], + ]); + }); + it("does not warn about a failure that does not advance", async () => { const warn = vi.fn(); const chain = new ProviderFallbackChain({ diff --git a/src/llm/fallback/run-with-fallback.ts b/src/llm/fallback/run-with-fallback.ts index ec625a4f7..589887095 100644 --- a/src/llm/fallback/run-with-fallback.ts +++ b/src/llm/fallback/run-with-fallback.ts @@ -3,7 +3,7 @@ import { attachFailingLink, type FailedAttempt, } from "./failed-attempts.js"; -import { isCredentialRejection } from "./link-failure-kind.js"; +import { isBillingRefusal, isCredentialRejection } from "./link-failure-kind.js"; import type { ProviderFallbackChain } from "./provider-fallback-chain.js"; import { shouldAdvance } from "./should-advance.js"; @@ -38,10 +38,29 @@ import { shouldAdvance } from "./should-advance.js"; * advance log. A fallback that has been serving is the route the user * is actually on, so its outage still gets the outage wait, as before. * + * **The same for an account that cannot pay** (`isBillingRefusal`: a + * 402, AI/ML API's 403 "You've run out of funds", OpenAI's 429 + * `insufficient_quota`). The primary's billing refusal outranks the + * links after it on the same terms as its refused key (item 40: the + * turn parked on a stopped local server and the window named that + * server). And a fallback that has been serving this partition and now + * says the account is empty is the route the user is on saying no: when + * the chain runs out after it, its refusal is thrown, with the links + * that failed before it recorded beside it, rather than a later link's + * outage. Every billing refusal ends the turn without the outage wait. + * * Every thrown error also carries the id of the link that threw it * (`attachFailingLink`), for the hosts that say which link a parked turn * is waiting on. * + * **A call its caller stopped never falls over.** Once `signal` has + * aborted, whatever the attempt threw is the stop's doing — a stop that + * lands as a stream ends can come back as `terminated` or `fetch failed` + * — and says nothing about the link. Advancing on it armed that link's + * breaker and flipped the sticky override, so the next turn ran on the + * fallback for nothing; recording it beside the error made a cancelled + * turn report the primary's transport failure. It is rethrown as it is. + * * Shared by both the non-stream (`llmComplete`) and stream-opening * (`llmCompleteStream`) seams. For streaming, `attempt` must resolve only * once the stream has successfully OPENED — a stream already emitting @@ -52,6 +71,7 @@ export async function runWithFallback( chain: ProviderFallbackChain, attempt: (providerId: string) => Promise, partitionKey?: string, + signal?: AbortSignal, ): Promise { const pick = chain.pickProvider(partitionKey); let currentId = pick.providerId; @@ -68,8 +88,14 @@ export async function runWithFallback( // the primary's real refusal was last mentioned anywhere. const cause = pick.isProbe ? null : chain.overrideCause(partitionKey); const failed: FailedAttempt[] = cause ? [cause] : []; - /** The primary's own refusal of its key, when this call asked it. */ + /** The primary's own refusal of its key or its account, when this call asked it. */ let primaryRefusal: { providerId: string; error: unknown } | null = null; + /** The serving fallback's billing refusal, and the links that failed before it. */ + let routeRefusal: { + providerId: string; + error: unknown; + before: FailedAttempt[]; + } | null = null; for (;;) { try { @@ -77,6 +103,12 @@ export async function runWithFallback( chain.recordSuccess(currentId, wasProbe, partitionKey); return result; } catch (err) { + if (signal?.aborted) { + attachFailingLink(err, currentId); + throw err; + } + // Read before `advanceFrom` moves the override past this link. + const serving = chain.isServingFallback(currentId, partitionKey); const nextId = chain.advanceFrom(currentId, err, partitionKey); if (nextId === null) { // `null` is also the answer for an error that must not fall over @@ -90,12 +122,22 @@ export async function runWithFallback( attachFailingLink(primaryRefusal.error, primaryRefusal.providerId); throw primaryRefusal.error; } + if (routeRefusal !== null && shouldAdvance(err).advance) { + attachFailedAttempts(routeRefusal.error, routeRefusal.before); + attachFailingLink(routeRefusal.error, routeRefusal.providerId); + throw routeRefusal.error; + } attachFailedAttempts(err, failed); attachFailingLink(err, currentId); throw err; } - if (isCredentialRejection(err) && chain.isPrimary(currentId)) { + if ( + chain.isPrimary(currentId) && + (isCredentialRejection(err) || isBillingRefusal(err)) + ) { primaryRefusal = { providerId: currentId, error: err }; + } else if (routeRefusal === null && serving && isBillingRefusal(err)) { + routeRefusal = { providerId: currentId, error: err, before: [...failed] }; } failed.push({ providerId: currentId, error: err }); currentId = nextId; diff --git a/src/llm/provider/openai/openai-http.test.ts b/src/llm/provider/openai/openai-http.test.ts index 4d5868333..f5ea11381 100644 --- a/src/llm/provider/openai/openai-http.test.ts +++ b/src/llm/provider/openai/openai-http.test.ts @@ -9,6 +9,7 @@ import { type OpenAiHttpDeps, } from "./openai-http.js"; import { classifyFailure } from "../../reliability/classify-failure.js"; +import { parseProviderErrorBody } from "./parse-provider-error-body.js"; function depsWith( fetchImpl: typeof fetch, @@ -190,6 +191,25 @@ describe("openAiPostJson", () => { ); }); + it("does not retry a 429 whose words say the account is empty (item 40)", async () => { + const body = JSON.stringify({ + error: { + message: + "Your account is suspended due to insufficient balance, please recharge your account", + type: "exceeded_current_quota_error", + }, + }); + const fetchImpl = vi.fn().mockResolvedValue(errorResponse(429, body)); + const err = await openAiPostJson( + depsWith(fetchImpl as unknown as typeof fetch), + "/x", + {}, + {}, + ).catch((e: unknown) => e); + expect(err).toBeInstanceOf(OpenAiHttpError); + expect(fetchImpl).toHaveBeenCalledTimes(1); + }); + describe("structured RetryInfo metadata", () => { // Gemini's OpenAI-compatible endpoint sends its cooldown only in // the error JSON — google.rpc.RetryInfo with a protobuf Duration @@ -816,6 +836,126 @@ describe("humanizeOpenAiHttpError", () => { expect(said).not.toContain('"error"'); }); + /* Item 40: a refusal because the account cannot pay names the provider, + quotes its own first sentence (links cut to their domain) and says + what helps. AI/ML API's 403 read "rejected the API key (403)", and + OpenAI's 429 insufficient_quota read as rate limiting, "Tried 3 + times", for a request that was never retried. */ + describe("a refusal because the account cannot pay", () => { + const aiml = (body: string): OpenAiHttpError => + new OpenAiHttpError( + `openai provider 403: ${body}`, + 403, + "https://api.aimlapi.com/v1/chat/completions", + false, + null, + "aimlapi", + undefined, + { body: parseProviderErrorBody(body) }, + ); + + it("says AI/ML API's 403 in its own words, with the remedy", () => { + const said = humanizeOpenAiHttpError( + aiml( + JSON.stringify({ + title: "Forbidden", + status: 403, + message: + "You've run out of funds. Please top up your balance or update your payment method to continue: https://aimlapi.com/app/billing", + }), + ), + ); + expect(said).toBe( + "AI/ML API refused the request: you've run out of funds. Top up your balance with AI/ML API or pick another provider in the Providers panel.", + ); + expect(said).not.toContain("API key"); + }); + + it("names a preset or a numbered entry by its service, and quotes an id it does not know", () => { + const body = JSON.stringify({ error: { message: "Insufficient Balance" } }); + const as = (label: string): string => + humanizeOpenAiHttpError( + new OpenAiHttpError(`openai provider 403: ${body}`, 403, "https://api.example.com/v1/chat/completions", false, null, label, undefined, { + body: parseProviderErrorBody(body), + }), + ); + expect(as("deepseek")).toMatch(/^DeepSeek refused the request: insufficient Balance\. Top up your balance with DeepSeek /); + expect(as("aimlapi-2")).toMatch(/^AI\/ML API refused the request/); + expect(as("dashscope")).toMatch(/^Qwen refused the request/); + expect(as("my-proxy")).toMatch(/^"my-proxy" refused the request/); + }); + + /* OpenRouter relays the upstream vendor's refusal (a key of the user's + own at that vendor) as "Provider returned error", with the vendor's + body in metadata.raw: the F29 case. */ + it("quotes the upstream vendor OpenRouter relays, and sends the top-up there", () => { + const relayed = (metadata: Record): string => { + const body = JSON.stringify({ error: { message: "Provider returned error", code: 429, metadata } }); + return humanizeOpenAiHttpError( + new OpenAiHttpError(`openai provider 429: ${body}`, 429, "https://openrouter.ai/api/v1/chat/completions", false, null, "openrouter", undefined, { + body: parseProviderErrorBody(body), + }), + ); + }; + const raw = + '{"type":"error","error":{"type":"credit_balance_exhausted","message":"Your credit balance is too low to access the Anthropic API. Please go to Plans & Billing to upgrade or purchase credits."}}'; + expect(relayed({ provider_name: "Anthropic", raw })).toBe( + "OpenRouter refused the request: Anthropic says your credit balance is too low to access the Anthropic API. Top up your balance with Anthropic or pick another provider in the Providers panel.", + ); + expect(relayed({ raw })).toBe( + "OpenRouter refused the request: your credit balance is too low to access the Anthropic API. Top up your balance with the provider behind OpenRouter or pick another provider in the Providers panel.", + ); + expect(relayed({ provider_name: "Anthropic", raw })).not.toContain("provider returned error"); + }); + + it("cuts a link in the quoted sentence to its domain", () => { + const said = humanizeOpenAiHttpError( + aiml( + JSON.stringify({ + error: { message: "No funds left, add some at https://www.example.com/billing/top-up?ref=x" }, + }), + ), + ); + expect(said).toContain("refused the request: no funds left, add some at example.com."); + expect(said).not.toContain("https://"); + }); + + it("words OpenAI's 429 insufficient_quota as the account, not a rate limit", () => { + const body = JSON.stringify({ + error: { + message: + "You exceeded your current quota, please check your plan and billing details. For more information on this error, read the docs: https://platform.openai.com/docs/guides/error-codes/api-errors.", + type: "insufficient_quota", + code: "insufficient_quota", + }, + }); + const said = humanizeOpenAiHttpError( + new OpenAiHttpError( + `openai provider 429: ${body}`, + 429, + "https://api.openai.com/v1/chat/completions", + false, + null, + "openai", + undefined, + { body: parseProviderErrorBody(body) }, + ), + ); + expect(said).toBe( + '"openai" refused the request: you exceeded your current quota, please check your plan and billing details. Top up your balance with "openai" or pick another provider in the Providers panel.', + ); + expect(said).not.toContain("rate-limiting"); + expect(said).not.toContain("Tried"); + }); + + it("leaves a 403 about the key and a plain 429 as they were", () => { + expect( + humanizeOpenAiHttpError(aiml(JSON.stringify({ error: { message: "Invalid API key" } }))), + ).toContain("rejected the API key (403)"); + expect(humanizeOpenAiHttpError(mk(429))).toContain("rate-limiting this key (429)"); + }); + }); + it("says only what it knows when the body carried nothing", () => { /* No sentence from the provider, so none is invented — but a 402 still explains the mechanism, because that part is true whatever the body diff --git a/src/llm/provider/openai/openai-http.ts b/src/llm/provider/openai/openai-http.ts index 2e759ebf2..cc84a90a8 100644 --- a/src/llm/provider/openai/openai-http.ts +++ b/src/llm/provider/openai/openai-http.ts @@ -1,6 +1,7 @@ import { isAsciiOnly } from "./ascii-header-guard.js"; import { buildOpenAiAuthHeaders } from "./openai-auth-headers.js"; import { readErrnoCode } from "../../errno-code.js"; +import { providerServiceName } from "../provider-service-name.js"; import { creditLimitRetryContext, creditLimitRetryMessage, @@ -196,6 +197,9 @@ export function humanizeOpenAiHttpError(err: OpenAiHttpError): string { if (err.keyProblem === "missing") { return `${who} needs an API key and none is set. Add the key in the Providers panel.`; } + if ((err.status === 403 || err.status === 429) && isCreditExhausted(err)) { + return billingRefusalSentence(err); + } if (err.status === 401 || err.status === 403) { return `${who} rejected the API key (${err.status}). Check the key in the Providers panel.`; } @@ -241,6 +245,67 @@ export function humanizeOpenAiHttpError(err: OpenAiHttpError): string { : `${who} rejected the request (${err.status}).`; } +/** + * A 403 or 429 that refused because the account cannot pay + * (`isCreditExhausted`): the provider by its name, its own first + * sentence, and what helps. Item 40: AI/ML API's 403 "You've run out of + * funds" read "rejected the API key (403)", and OpenAI's 429 + * `insufficient_quota` read as rate limiting ("Tried 3 times") for a + * request that was never retried. A 402 keeps its own wording below, + * which also explains the reservation. + * + * OpenRouter relays an upstream vendor's refusal (a key of your own at + * Anthropic, say) as "Provider returned error", with the vendor's body in + * `metadata.raw`: the vendor's words are the ones quoted, and the + * account to top up is the vendor's. + */ +function billingRefusalSentence(err: OpenAiHttpError): string { + const who = err.providerLabel + ? providerServiceName(err.providerLabel) + : `"${hostOf(err.url)}"`; + const relayed = err.body?.upstreamMessage; + const vendor = relayed !== undefined ? err.body?.upstream : undefined; + // The parsed body first: it was read whole, the message keeps only its head. + const own = shortProviderSentence( + relayed ?? err.body?.message ?? providerReason(err), + ); + const said = own + ? `${lowerFirst(own)}${own.endsWith("…") ? "" : "."}` + : `the account has no funds or credit left (${err.status}).`; + const payer = + vendor ?? (relayed !== undefined ? `the provider behind ${who}` : who); + return ( + `${who} refused the request: ${vendor !== undefined ? `${vendor} says ` : ""}${said} ` + + `Top up your balance with ${payer} or pick another provider in the Providers panel.` + ); +} + +/** Longest provider sentence a billing refusal quotes. */ +const BILLING_SENTENCE_MAX_LEN = 140; + +/** + * The provider's first sentence, with every link cut to its domain: + * "You've run out of funds. Please top up …: https://aimlapi.com/app/…" + * is "You've run out of funds". The rest is the remedy the sentence + * around it already gives. + */ +function shortProviderSentence(text: string): string { + const flat = text + .replace(/\bhttps?:\/\/(?:www\.)?([^\s/?#"'<>)]+)[^\s"'<>)]*/gi, "$1") + .replace(/\s+/g, " ") + .trim(); + const first = /^(.+?[.!?])(?=\s|$)/.exec(flat)?.[1] ?? flat; + const bare = first.replace(/[\s.!?:;,]+$/, ""); + return bare.length > BILLING_SENTENCE_MAX_LEN + ? `${bare.slice(0, BILLING_SENTENCE_MAX_LEN - 1).trimEnd()}…` + : bare; +} + +/** "You've run out" → "you've run out"; an acronym ("API …") stays as it is. */ +function lowerFirst(text: string): string { + return /^[A-Z][a-z']/.test(text) ? text[0]!.toLowerCase() + text.slice(1) : text; +} + /** * The provider's own explanation, dug out of the body `httpErrorFromResponse` * folded into the message as `openai provider : `. @@ -780,7 +845,13 @@ function isRetryableOpenAiError(err: unknown): boolean { return err.status >= 500 || err.status === 429 || err.status === 408; } -function isCreditExhausted(err: OpenAiHttpError): boolean { +/** + * Did the provider refuse because the account cannot pay: a billing + * refusal (`readProviderErrorReason`'s `credit_exhausted`), as opposed to + * a refused key or a rate limit? Nothing about it changes until someone + * tops up, so it is never retried, waited out or read as a key problem. + */ +export function isCreditExhausted(err: OpenAiHttpError): boolean { return ( readProviderErrorReason({ status: err.status, diff --git a/src/llm/provider/openai/parse-provider-error-body.test.ts b/src/llm/provider/openai/parse-provider-error-body.test.ts index ddd498ec2..9ccb67a71 100644 --- a/src/llm/provider/openai/parse-provider-error-body.test.ts +++ b/src/llm/provider/openai/parse-provider-error-body.test.ts @@ -1,5 +1,6 @@ import { describe, expect, it } from "vitest"; import { + CREDENTIAL_WORDING, IN_FLIGHT_BUDGET_DEFAULT_WAIT_MS, parseProviderErrorBody, readProviderErrorReason, @@ -24,6 +25,23 @@ describe("parseProviderErrorBody", () => { expect(body.retryHintMs).toBeUndefined(); }); + it("reads the upstream vendor's own message out of OpenRouter's raw body", () => { + const relayed = (raw: unknown) => + parseProviderErrorBody( + JSON.stringify({ + error: { message: "Provider returned error", code: 429, metadata: { provider_name: "Anthropic", raw } }, + }), + ); + expect( + relayed('{"type":"error","error":{"type":"credit_balance_exhausted","message":"Your credit balance is too low"}}') + .upstreamMessage, + ).toBe("Your credit balance is too low"); + expect(relayed({ message: "Out of credit" }).upstreamMessage).toBe("Out of credit"); + expect(relayed("upstream said no").upstreamMessage).toBe("upstream said no"); + expect(relayed("502").upstreamMessage).toBeUndefined(); + expect(parseProviderErrorBody(JSON.stringify({ error: { message: "x" } })).upstreamMessage).toBeUndefined(); + }); + it("reads the OpenRouter shape, including the upstream raw body", () => { const body = parseProviderErrorBody( JSON.stringify({ @@ -60,6 +78,26 @@ describe("parseProviderErrorBody", () => { expect(body.text).toBe("Bad gateway"); expect(body.message).toBeUndefined(); }); + + it("reads the message a body with no error object carries at the top (AI/ML API)", () => { + const body = parseProviderErrorBody( + JSON.stringify({ + title: "Forbidden", + status: 403, + message: + "You've run out of funds. Please top up your balance or update your payment method to continue: https://aimlapi.com/app/billing", + }), + ); + expect(body.message).toBe( + "You've run out of funds. Please top up your balance or update your payment method to continue: https://aimlapi.com/app/billing", + ); + // An error object wins, as before. + expect( + parseProviderErrorBody( + JSON.stringify({ message: "outer", error: { message: "inner" } }), + ).message, + ).toBe("inner"); + }); }); describe("readProviderErrorReason", () => { @@ -165,6 +203,138 @@ describe("readProviderErrorReason", () => { expect(read(401, "", 5_000)).toBeNull(); }); + /* Item 40: a billing refusal that names no code. AI/ML API answered + 403 "You've run out of funds", which read as nothing at all, so the + turn fell over to a stopped local server and parked on it. */ + describe("an account that cannot pay, in words", () => { + const AIML_403 = JSON.stringify({ + title: "Forbidden", + status: 403, + message: + "You've run out of funds. Please top up your balance or update your payment method to continue: https://aimlapi.com/app/billing", + }); + + it("reads AI/ML API's 403 as exhausted credit", () => { + expect(read(403, AIML_403)).toEqual({ kind: "credit_exhausted", code: "403" }); + }); + + it("reads a 403 about billing or the payment method the same way", () => { + expect( + read(403, JSON.stringify({ error: { message: "Billing is not enabled for this account." } })), + ).toEqual({ kind: "credit_exhausted", code: "403" }); + }); + + it("leaves a 403 about the key or the input alone", () => { + expect( + read(403, JSON.stringify({ error: { message: "Invalid API key provided" } })), + ).toBeNull(); + // Billing words beside the key's are about the key. + expect( + read( + 403, + JSON.stringify({ error: { message: "Invalid API key. Check your billing settings." } }), + ), + ).toBeNull(); + expect( + read( + 403, + JSON.stringify({ error: { message: "Authentication failed. Please check your billing details." } }), + ), + ).toBeNull(); + expect(read(403, JSON.stringify({ error: { message: "Invalid token. Check billing." } }))).toBeNull(); + expect( + read( + 403, + JSON.stringify({ + error: { message: "Your chosen model requires moderation and your input was flagged" }, + }), + ), + ).toBeNull(); + }); + + it("reads any 402 that asked for no cooldown as the account, whatever its words", () => { + expect(read(402, JSON.stringify({ error: { message: "Insufficient Balance" } }))).toEqual({ + kind: "credit_exhausted", + code: "402", + }); + expect(read(402, "")).toEqual({ kind: "credit_exhausted", code: "402" }); + expect(read(402, "{}")).toEqual({ kind: "credit_exhausted", code: "402" }); + }); + + it("keeps a 402 that asked for a cooldown, and the in-flight budget, a wait", () => { + expect( + read(402, JSON.stringify({ error: { message: "Server busy, retry in 5 s" } })), + ).toEqual({ kind: "retry_after", delayMs: 5_000, code: null }); + expect( + read(402, JSON.stringify({ error: { code: "in_flight_budget_exhausted", message: "for your balance" } })), + ).toMatchObject({ kind: "retry_after", code: "in_flight_budget_exhausted" }); + }); + + it("reads a 429 that says the account is empty and asks for no cooldown as exhausted credit", () => { + expect( + read( + 429, + JSON.stringify({ + error: { + message: + "Your account is suspended due to insufficient balance, please recharge your account or check your plan and billing details", + type: "exceeded_current_quota_error", + }, + }), + ), + ).toEqual({ kind: "credit_exhausted", code: "429" }); + }); + + it("keeps an ordinary 429 rate limit transient, quota and billing words included", () => { + // Gemini's free tier: a per-minute limit in billing words, with a cooldown. + const gemini = JSON.stringify({ + error: { + code: 429, + message: + "You exceeded your current quota, please check your plan and billing details. Please retry in 33.5s.", + status: "RESOURCE_EXHAUSTED", + }, + }); + expect(read(429, gemini)).toEqual({ + kind: "retry_after", + delayMs: 33_500, + code: "429", + }); + // The same words without a cooldown are still not an empty account. + expect( + read( + 429, + JSON.stringify({ + error: { message: "You exceeded your current quota, please check your plan and billing details." }, + }), + ), + ).toBeNull(); + expect(read(429, JSON.stringify({ error: { message: "Rate limit exceeded" } }))).toBeNull(); + // A cooldown the provider asked for wins over its words. + expect( + read(429, JSON.stringify({ error: { message: "Out of credits for this minute" } }), 7_000), + ).toEqual({ kind: "retry_after", delayMs: 7_000, code: null }); + }); + + it("keeps a 429 that talks money in a rate limit's words a wait, cooldown or not", () => { + for (const message of [ + "Too many requests. Please top up your account to increase your rate limits.", + "Out of credits for this minute", + "Insufficient balance for 60 requests per minute; slow down", + "You have run out of credits for the current RPM window", + ]) { + expect(read(429, JSON.stringify({ error: { message } })), message).toBeNull(); + } + // An explicit billing code outweighs the rate limit's words. + expect( + read( + 429, + JSON.stringify({ error: { message: "Rate limit reached for requests", code: "insufficient_quota" } }), + ), + ).toEqual({ kind: "credit_exhausted", code: "insufficient_quota" }); + }); + }); + it("falls back to the error's own message when no body was kept", () => { expect( readProviderErrorReason({ @@ -177,3 +347,21 @@ describe("readProviderErrorReason", () => { ).toEqual({ kind: "credit_exhausted", code: "credit_balance_exhausted" }); }); }); + +describe("CREDENTIAL_WORDING", () => { + it("is a provider's words for the key or its authentication", () => { + for (const text of [ + "Invalid API key provided", + "Authentication failed", + "unauthenticated", + "Unauthorized", + "Invalid token", + "access token expired", + "bad credentials", + ]) { + expect(CREDENTIAL_WORDING.test(text), text).toBe(true); + } + expect(CREDENTIAL_WORDING.test("You've run out of funds")).toBe(false); + expect(CREDENTIAL_WORDING.test("Your input was flagged")).toBe(false); + }); +}); diff --git a/src/llm/provider/openai/parse-provider-error-body.ts b/src/llm/provider/openai/parse-provider-error-body.ts index 4452d460c..ba4247aca 100644 --- a/src/llm/provider/openai/parse-provider-error-body.ts +++ b/src/llm/provider/openai/parse-provider-error-body.ts @@ -18,7 +18,7 @@ * a cycle through the reliability layer. */ export interface ProviderErrorBody { - /** `error.message`, when the body parsed. */ + /** `error.message`, when the body parsed; a top-level `message` when it has no `error` object. */ readonly message?: string; /** `error.code` as a string (a numeric code is kept as its digits). */ readonly code?: string; @@ -26,6 +26,12 @@ export interface ProviderErrorBody { readonly type?: string; /** Upstream vendor name from OpenRouter's `metadata.provider_name`. */ readonly upstream?: string; + /** + * The upstream vendor's own message out of OpenRouter's `metadata.raw`, + * when it has one: "Your credit balance is too low to access the + * Anthropic API" behind OpenRouter's "Provider returned error". + */ + readonly upstreamMessage?: string; /** * A cooldown the body's *text* asked for ("retry in 120 s", "try * again in 2 minutes"). Headers and structured `RetryInfo` are read @@ -50,24 +56,73 @@ const CREDIT_CODES = /** OpenRouter's "you have too many requests in flight for your balance". */ const IN_FLIGHT_BUDGET = /\bin_flight_budget_exhausted\b/i; +/** + * An empty account in a provider's words, for the bodies that carry no + * code for it: AI/ML API's 403 "You've run out of funds", DeepSeek's 402 + * "Insufficient Balance", Moonshot's 429 "suspended due to insufficient + * balance, please recharge your account", "Your credit balance is too + * low", "Payment Required". + * + * Deliberately not "quota" or "billing" on their own: a per-minute rate + * limit says "You exceeded your current quota, please check your plan and + * billing details" too (Gemini's free tier, a 429 with a cooldown), and + * must keep its wait. + */ +const NO_FUNDS_WORDING = + /\b(?:out of (?:funds|credits?|balance|money)|insufficient[ _-]?(?:funds|balance|credits?|account[ _-]balance)|not enough (?:funds|credits?|balance|money)|(?:credit|account|wallet) balance (?:is )?(?:too low|exhausted|insufficient|empty|depleted)|(?:no|zero) (?:credits?|funds|balance) (?:left|remaining)|payment[ _-]required|top[ -]?up (?:your |the )?(?:balance|account|credits?|wallet)|recharge (?:your |the )?(?:account|balance|wallet))\b/i; + +/** + * Billing words a 402 or a 403 is read for (never a 429, see above): + * "billing is not enabled", "update your payment method". + */ +const BILLING_WORDING = + /\b(?:billing|payment[ _-]?method|payment details|add (?:a )?payment)\b/i; + +/** + * A provider's words for a credential problem. One rule for every reader: + * the fallback chain's refused-key check (`isCredentialRejection`) and the + * billing wording below, which a 403 that also talks about its key or + * authentication ("Authentication failed. Please check your billing + * details.") is not read for. + */ +export const CREDENTIAL_WORDING = + /\b(?:api[ _-]?key|credentials?|unauthori[sz]ed|unauthenticated|authenticat\w*|access[ _-]?token|invalid[ _-]?token)\b/i; + +/** + * A rate limit in so many words. On a 429 it outweighs any wording about + * money ("Too many requests. Please top up your account to increase your + * rate limits.", "Out of credits for this minute"): such a 429 is a + * billing refusal only by an explicit code (`CREDIT_CODES`). + */ +const RATE_LIMIT_WORDING = + /\b(?:rate[ _-]?limit\w*|too many requests|requests? per|tokens? per|per[ -](?:second|minute|hour|day)|(?:this|a|each|every|the next) (?:second|minute|hour)|[RT]PM|throttl\w*|slow down)\b/i; + /** "retry in 120 s", "retry after 2 minutes", "try again in 30 seconds". */ const RETRY_HINT = /\b(?:retry|try again|please wait)(?:\s+\w+){0,2}?\s+(?:in|after)\s+(\d+(?:\.\d+)?)\s*(ms|milliseconds?|s|secs?|seconds?|m|mins?|minutes?)\b/i; export function parseProviderErrorBody(text: string): ProviderErrorBody { const bounded = text.slice(0, BODY_TEXT_MAX); - const error = readErrorObject(bounded); + const root = tryParseJson(bounded); + const error = readObject(root?.error); const raw = error !== null ? readRaw(error.metadata) : null; - const message = readString(error?.message); + // A body with no `error` object says it at the top: AI/ML API answers + // `{"title": "Forbidden", "status": 403, "message": "You've run out of + // funds. …"}`, and that sentence is the one worth quoting. + const message = + readString(error?.message) ?? + (error === null ? readString(root?.message) : undefined); const code = readCode(error?.code); const type = readString(error?.type); const upstream = readString(readObject(error?.metadata)?.provider_name); + const upstreamMessage = raw !== null ? readUpstreamMessage(raw) : undefined; const hint = retryHintMs(`${message ?? ""}\n${raw ?? ""}\n${bounded}`); return { ...(message !== undefined ? { message } : {}), ...(code !== undefined ? { code } : {}), ...(type !== undefined ? { type } : {}), ...(upstream !== undefined ? { upstream } : {}), + ...(upstreamMessage !== undefined ? { upstreamMessage } : {}), ...(hint !== null ? { retryHintMs: hint } : {}), text: raw !== null && !bounded.includes(raw) ? `${bounded}\n${raw}` : bounded, }; @@ -91,6 +146,15 @@ export const IN_FLIGHT_BUDGET_DEFAULT_WAIT_MS = 30_000; * Read the reason out of a parsed body plus the transport facts around * it. `retryAfterMs` is the header / `RetryInfo` value the HTTP client * already extracted, when any. + * + * `credit_exhausted` is every billing refusal: a code that says so, a + * 402 that asked for no cooldown, and a 403 or 429 whose words say the + * account cannot pay (`NO_FUNDS_WORDING`; a 403 also `BILLING_WORDING` + * when it says nothing about its key). A 429 counts only when it asked + * for no cooldown and says nothing of a rate limit: a rate limit stays a + * wait. Item 40: AI/ML API's 403 "You've run out + * of funds" read as nothing at all, so its fallback chain parked the turn + * on a stopped local server and the window named that server. */ export function readProviderErrorReason(input: { status: number | null; @@ -109,9 +173,13 @@ export function readProviderErrorReason(input: { if (status === 402 && /\bcredits?\b/i.test(text)) { return { kind: "credit_exhausted", code: "402" }; } - if (!isCooldownStatus(status)) return null; const hinted = input.retryAfterMs ?? input.body?.retryHintMs ?? null; - if (IN_FLIGHT_BUDGET.test(text)) { + const inFlight = IN_FLIGHT_BUDGET.test(text); + if (!inFlight && saysAccountCannotPay(status, text, hinted)) { + return { kind: "credit_exhausted", code: String(status) }; + } + if (!isCooldownStatus(status)) return null; + if (inFlight) { return { kind: "retry_after", delayMs: hinted ?? IN_FLIGHT_BUDGET_DEFAULT_WAIT_MS, @@ -121,9 +189,35 @@ export function readProviderErrorReason(input: { if (hinted !== null) { return { kind: "retry_after", delayMs: hinted, code: code || null }; } + // Payment Required that asked for no cooldown: the account's answer, + // whatever its words. + if (status === 402) return { kind: "credit_exhausted", code: "402" }; return null; } +/** A 402, 403 or 429 whose words say the account cannot pay. */ +function saysAccountCannotPay( + status: number | null, + text: string, + hinted: number | null, +): boolean { + if (status === 402 || status === 403) { + if (NO_FUNDS_WORDING.test(text)) return true; + return ( + BILLING_WORDING.test(text) && + (status === 402 || !CREDENTIAL_WORDING.test(text)) + ); + } + if (status === 429) { + return ( + hinted === null && + NO_FUNDS_WORDING.test(text) && + !RATE_LIMIT_WORDING.test(text) + ); + } + return false; +} + /** Statuses whose body may legitimately ask for a cooldown. */ function isCooldownStatus(status: number | null): boolean { return status === 402 || status === 408 || status === 429 || (status !== null && status >= 500); @@ -143,10 +237,20 @@ function retryHintMs(text: string): number | null { return Math.round(ms); } -function readErrorObject(text: string): Record | null { - const parsed = tryParseJson(text); - const error = readObject(parsed?.error); - return error; +/** + * The upstream body's own message: `error.message` or a top-level + * `message` when it is JSON, the text itself when it is a short plain + * sentence. + */ +function readUpstreamMessage(raw: string): string | undefined { + const root = tryParseJson(raw); + if (root !== null) { + return readString(readObject(root.error)?.message) ?? readString(root.message); + } + const plain = raw.trim(); + return plain.length > 0 && plain.length <= 300 && !plain.includes("<") + ? plain + : undefined; } /** OpenRouter's `metadata.raw`: the upstream body, as text or as an object. */ diff --git a/src/llm/provider/provider-service-name.ts b/src/llm/provider/provider-service-name.ts new file mode 100644 index 000000000..629e096bf --- /dev/null +++ b/src/llm/provider/provider-service-name.ts @@ -0,0 +1,28 @@ +import { presetForEntryId } from "./presets/provider-presets.js"; + +/** + * The built-in kinds by the names their services go by, as the TUI's + * providers screens write them (`KIND_SERVICE_LABELS` in + * src/tui/providers/providers-wizard-target.ts). + */ +const SERVICE_NAMES: Readonly> = { + openrouter: "OpenRouter", + aimlapi: "AI/ML API", + gemini: "Gemini", +}; + +/** + * A provider entry id as a sentence names it: "AI/ML API" for `aimlapi`, + * "DeepSeek" for `deepseek` or `deepseek-2`, "Qwen" for `dashscope` (a + * preset's label up to its parenthesis, as the desktop shows it). An id + * that is none of these is quoted, the way every other failure sentence + * quotes it: a custom entry's id is the only name it has. + */ +export function providerServiceName(id: string): string { + const base = /^(.+)-\d+$/.exec(id)?.[1] ?? id; + const named = + SERVICE_NAMES[id] ?? + SERVICE_NAMES[base] ?? + presetForEntryId(id)?.label.split(" (")[0]; + return named !== undefined && named.length > 0 ? named : `"${id}"`; +} diff --git a/src/llm/reliability/provider-wait-cause.test.ts b/src/llm/reliability/provider-wait-cause.test.ts index 9194c8321..867df89a2 100644 --- a/src/llm/reliability/provider-wait-cause.test.ts +++ b/src/llm/reliability/provider-wait-cause.test.ts @@ -7,6 +7,7 @@ import { } from "../provider/openai/openai-http.js"; import { OpenAiProvider } from "../provider/openai/openai-provider.js"; import { OpenAiSseError } from "../provider/openai/openai-stream-consumer.js"; +import { parseProviderErrorBody } from "../provider/openai/parse-provider-error-body.js"; import { TransportError } from "./llm-failures.js"; import { classifyProviderWaitCause } from "./provider-wait-cause.js"; @@ -58,6 +59,34 @@ describe("classifyProviderWaitCause", () => { ).toEqual({ kind: "stream_error", status: null }); }); + it("a provider that refused because the account cannot pay is billing, through the wrapper", () => { + // Item 40: AI/ML API's 403 for an empty account, and OpenAI's 429 + // insufficient_quota. The TransportError carries the status alone. + const aiml = '{"title":"Forbidden","status":403,"message":"You\'ve run out of funds. Please top up your balance"}'; + const out = new OpenAiHttpError(`openai provider 403: ${aiml}`, 403, URL, false, null, "aimlapi", undefined, { + body: parseProviderErrorBody(aiml), + }); + expect(classifyProviderWaitCause(asStepFailure(out))).toEqual({ kind: "billing", status: 403 }); + expect(classifyProviderWaitCause(out)).toEqual({ kind: "billing", status: 403 }); + const quota = '{"error":{"message":"You exceeded your current quota","code":"insufficient_quota"}}'; + expect( + classifyProviderWaitCause( + asStepFailure( + new OpenAiHttpError(`openai provider 429: ${quota}`, 429, URL, false, null, "openai", undefined, { + body: parseProviderErrorBody(quota), + }), + ), + ), + ).toEqual({ kind: "billing", status: 429 }); + // A rate limit, and a 403 about the key, stay what they were. + expect( + classifyProviderWaitCause(asStepFailure(new OpenAiHttpError("openai provider 429: slow down", 429, URL))), + ).toEqual({ kind: "http", status: 429 }); + expect( + classifyProviderWaitCause(asStepFailure(new OpenAiHttpError("openai provider 403: Invalid API key", 403, URL))), + ).toEqual({ kind: "http", status: 403 }); + }); + it("an HTTP status only when a response carried one", () => { const err = new OpenAiHttpError("openai provider 503: busy", 503, URL); expect(classifyProviderWaitCause(asStepFailure(err))).toEqual({ diff --git a/src/llm/reliability/provider-wait-cause.ts b/src/llm/reliability/provider-wait-cause.ts index 4bc81f426..6c01e3dae 100644 --- a/src/llm/reliability/provider-wait-cause.ts +++ b/src/llm/reliability/provider-wait-cause.ts @@ -1,6 +1,7 @@ import { readErrnoCode } from "../errno-code.js"; import { LlamaServerError } from "../llama-server-client.js"; import { + isCreditExhausted, isErrorFinishMessage, OpenAiHttpError, } from "../provider/openai/openai-http.js"; @@ -19,8 +20,15 @@ import { looksLikeMidStreamDrop } from "./network-error.js"; * (502). Tried 3 times" for a 200 that was never retried. Each kind * here carries only facts the error itself holds — a status only when * a response had one. + * + * `billing` is a provider that answered and refused because the account + * cannot pay (`isCreditExhausted`: a 402, AI/ML API's 403 "You've run out + * of funds", OpenAI's 429 `insufficient_quota`). A turn never waits on + * one; it is the cause of a link that failed before the one waited on, + * and of an error that ended the turn. */ export type ProviderWaitCause = + | { readonly kind: "billing"; readonly status: number } | { readonly kind: "refused" } | { readonly kind: "dropped" } | { readonly kind: "unreachable" } @@ -31,6 +39,14 @@ export type ProviderWaitCause = | { readonly kind: "error_finish" } | { readonly kind: "unknown" }; +/** A chain link that failed before the one a turn waits on, and why. */ +export interface ProviderWaitFailure { + readonly providerId: string; + /** The failure's own line (`describeReason`), for logs and traces. */ + readonly reason: string; + readonly cause: ProviderWaitCause; +} + const MAX_CAUSE_DEPTH = 6; const UNREACHABLE_ERRNOS = new Set([ @@ -72,6 +88,20 @@ export function classifyProviderWaitCause(err: unknown): ProviderWaitCause { } } + // An account that cannot pay, read off the provider's own error under + // whatever wrapper reached here: the step executor's TransportError + // carries the status alone. + for (const link of chain) { + if ( + link instanceof OpenAiHttpError && + !link.timedOut && + link.status !== null && + isCreditExhausted(link) + ) { + return { kind: "billing", status: link.status }; + } + } + for (const link of chain) { if ( link instanceof OpenAiHttpError || diff --git a/src/local-llm/backend-paths.ts b/src/local-llm/backend-paths.ts index a4a1d0513..d7990e50a 100644 --- a/src/local-llm/backend-paths.ts +++ b/src/local-llm/backend-paths.ts @@ -44,6 +44,16 @@ export function resolveVersionFilePath(dataDir: string): string { return join(resolveBackendDir(dataDir), "backend-version.json"); } +/** + * What the last release check before a managed start found + * (`ensure-latest-backend.ts`), so a start in another process trusts a + * recent check instead of asking GitHub again. Next to the pid file, not + * in `backend/`, which an update replaces wholesale. + */ +export function resolveBackendCheckFilePath(dataDir: string): string { + return join(dataDir, "llama-backend-check.json"); +} + export function resolvePidFilePath(dataDir: string): string { return join(dataDir, "llama-server.pid"); } diff --git a/src/local-llm/context-size.test.ts b/src/local-llm/context-size.test.ts index 94d3905be..e6acdf02e 100644 --- a/src/local-llm/context-size.test.ts +++ b/src/local-llm/context-size.test.ts @@ -17,8 +17,16 @@ import { kvBitsPerValue, resolveDeviceFreeVramMiB, resolveKvBudgetMiB, + resolveContextKvBudgetMiB, + resolveUnifiedMemoryHeadroomMiB, + resolveUnifiedMemoryKvRoomMiB, + UNIFIED_MEMORY_HEADROOM_MIN_MIB, + UNIFIED_MEMORY_HEADROOM_SHARE, + UNIFIED_MEMORY_KV_SHARE, type KvCacheLayout, } from "./context-size.js"; +import { resolveSwaFullDecision } from "./swa-full.js"; +import { resolveWorkerSlots } from "./worker-slots.js"; import type { GpuDevice } from "./gpu-devices.js"; import { minUsableContextWindow } from "../prompt/token-budget.js"; import { USER_CONFIG_DEFAULTS } from "../config/config-schema.js"; @@ -509,6 +517,221 @@ describe("estimateContextSize", () => { }); }); +/** + * Backlog 42: Qwen 3.5 4B started with `--ctx-size 262144` "fitted from + * the model's KV layout" on a 16 GB Mac that was 13.5 GB into swap. + * Metal reports its working-set ceiling as free whatever else runs, so + * the fit alone had ~6.5 GB to spend; on unified memory the auto size is + * also held to two shares of physical RAM. + */ +describe("the unified-memory cap (backlog 42)", () => { + /** The catalogue's Qwen 3.5 4B, text-only (no projector on disk). */ + const QWEN35_4B = { modelSizeGb: 2.7, mmprojSizeGb: 0, maxContextLength: 262_144 }; + /** What Metal reports as free on a 16 GB Apple silicon Mac (its working-set ceiling). */ + const MAC_16GB_METAL_FREE_MIB = 10_922; + + it("costs Qwen 3.5 4B 7 KiB per token: 1.75 GiB at 262,144, 224 MiB at 32,768", () => { + // 8 attention layers × 4 KV heads × (256 + 256) dims × 3.5 bits. + expect(estimateKvBytesPerToken(QWEN35_4B_LAYOUT, MANAGED_KV_CACHE_TYPE, 262_144)).toBe(7_168); + expect(estimateKvBytesTotal(QWEN35_4B_LAYOUT, MANAGED_KV_CACHE_TYPE, 262_144)).toBe(1.75 * 1024 ** 3); + expect(estimateKvBytesTotal(QWEN35_4B_LAYOUT, MANAGED_KV_CACHE_TYPE, 32_768)).toBe(224 * 1024 ** 2); + }); + + it("gave Qwen 3.5 4B its whole trained context on a 16 GB Mac from Metal's figure alone", () => { + expect( + estimateContextSize({ + ...QWEN35_4B, + freeVramMiB: MAC_16GB_METAL_FREE_MIB, + configuredContextSize: 0, + kvLayout: QWEN35_4B_LAYOUT, + }), + ).toBe(262_144); + }); + + it("holds it to 1 GiB of cache on that Mac: 149,504 tokens", () => { + const ctx = estimateContextSize({ + ...QWEN35_4B, + freeVramMiB: MAC_16GB_METAL_FREE_MIB, + configuredContextSize: 0, + kvLayout: QWEN35_4B_LAYOUT, + systemMemoryMiB: 16_384, + }); + expect(ctx).toBe(149_504); + expect(estimateKvBytesTotal(QWEN35_4B_LAYOUT, MANAGED_KV_CACHE_TYPE, ctx)).toBeLessThanOrEqual( + 16_384 * UNIFIED_MEMORY_KV_SHARE * 1024 * 1024, + ); + }); + + it("scales with the machine: 74,752 on 8 GB, the trained 262,144 on 32 GB (1.75 GiB fits its 2 GiB)", () => { + const at = (systemMemoryMiB: number, freeVramMiB: number) => + estimateContextSize({ + ...QWEN35_4B, + freeVramMiB, + configuredContextSize: 0, + kvLayout: QWEN35_4B_LAYOUT, + systemMemoryMiB, + }); + expect(at(8_192, 5_461)).toBe(74_752); + expect(at(32_768, 21_845)).toBe(262_144); + }); + + it("keeps Gemma 4 31B above 200k on a 64 GB Mac (4 GiB of cache)", () => { + const ctx = estimateContextSize({ + ...GEMMA_31B, + freeVramMiB: 49_152, + configuredContextSize: 0, + kvLayout: GEMMA4_31B_LAYOUT, + systemMemoryMiB: 65_536, + }); + expect(ctx).toBe(229_376); + expect(estimateKvBytesTotal(GEMMA4_31B_LAYOUT, MANAGED_KV_CACHE_TYPE, ctx)).toBeLessThanOrEqual( + 4 * 1024 ** 3, + ); + }); + + it("keeps Gemma 4 31B well above the floor on a 32 GB and a 36 GB Mac, and Fusion's auto slots above one", () => { + // Metal's figure: about two thirds of 32 GB, three quarters of 36 GB. + const at = (systemMemoryMiB: number, freeVramMiB: number, mmprojSizeGb: number) => + estimateContextSize({ + ...GEMMA_31B, + mmprojSizeGb, + freeVramMiB, + configuredContextSize: 0, + kvLayout: GEMMA4_31B_LAYOUT, + systemMemoryMiB, + }); + const on32 = at(32_768, 21_845, 0); + const on32Vision = at(32_768, 21_845, GEMMA_31B.mmprojSizeGb); + const on36 = at(36_864, 27_648, 0); + // 2 GiB / 2.25 GiB of cache: the 1/16 share holds them, not the headroom. + expect(on32).toBe(109_568); + expect(on36).toBe(123_904); + // With its projector the free figure is the tighter one, as before the cap. + expect(on32Vision).toBe(88_064); + expect( + estimateContextSize({ + ...GEMMA_31B, + freeVramMiB: 21_845, + configuredContextSize: 0, + kvLayout: GEMMA4_31B_LAYOUT, + }), + ).toBe(88_064); + // Fusion's `parallel: "auto"` at the default 16,384-token reply: three + // workers' footprints fit, where the 32,768 floor held it to one. + const slots = (contextSize: number) => + resolveWorkerSlots({ contextSize, cpuOnly: false, completionMaxTokens: 16_384 }); + expect(slots(on32)).toBe(3); + expect(slots(on36)).toBe(3); + expect(slots(on32Vision)).toBe(2); + expect(slots(MIN_AUTO_CONTEXT)).toBe(1); + }); + + it("leaves the system a quarter of RAM, never less than 4 GiB", () => { + expect(UNIFIED_MEMORY_HEADROOM_SHARE).toBe(0.25); + expect(UNIFIED_MEMORY_HEADROOM_MIN_MIB).toBe(4_096); + expect(resolveUnifiedMemoryHeadroomMiB(8_192)).toBe(4_096); + expect(resolveUnifiedMemoryHeadroomMiB(16_384)).toBe(4_096); + expect(resolveUnifiedMemoryHeadroomMiB(32_768)).toBe(8_192); + expect(resolveUnifiedMemoryHeadroomMiB(131_072)).toBe(32_768); + // 16 GB less 4 GiB, the 2.7 GB of weights and the compute buffers. + expect( + resolveUnifiedMemoryKvRoomMiB({ systemMemoryMiB: 16_384, modelSizeGb: 2.7, mmprojSizeGb: 0 }), + ).toBeCloseTo(8_945.08, 2); + }); + + it("lands on the floor when the weights alone reach into the headroom", () => { + // A 3.5 GB model on an 8 GB Mac: Metal's 5.3 GB figure alone would + // give it 134,144 tokens; RAM less the 4 GiB headroom has no room left. + const input = { modelSizeGb: 3.5, mmprojSizeGb: 0, maxContextLength: 262_144 }; + expect( + resolveUnifiedMemoryKvRoomMiB({ systemMemoryMiB: 8_192, modelSizeGb: 3.5, mmprojSizeGb: 0 }), + ).toBeLessThan(0); + expect( + estimateContextSize({ + ...input, + freeVramMiB: 5_461, + configuredContextSize: 0, + kvLayout: QWEN35_4B_LAYOUT, + }), + ).toBe(134_144); + expect( + estimateContextSize({ + ...input, + freeVramMiB: 5_461, + configuredContextSize: 0, + kvLayout: QWEN35_4B_LAYOUT, + systemMemoryMiB: 8_192, + }), + ).toBe(MIN_AUTO_CONTEXT); + }); + + it("fits the context into the 1/16 share, and weighs --swa-full against the headroom alone", () => { + const input = { systemMemoryMiB: 16_384, modelSizeGb: 2.7, mmprojSizeGb: 0 }; + const plain = resolveKvBudgetMiB({ freeVramMiB: MAC_16GB_METAL_FREE_MIB, modelSizeGb: 2.7, mmprojSizeGb: 0 }); + // On 16 GB the headroom leaves more than Metal's figure does. + expect(resolveKvBudgetMiB({ ...input, freeVramMiB: MAC_16GB_METAL_FREE_MIB })).toBe(plain); + expect(resolveContextKvBudgetMiB({ ...input, freeVramMiB: MAC_16GB_METAL_FREE_MIB })).toBe(1_024); + // A tight free figure still wins over both. + expect(resolveContextKvBudgetMiB({ ...input, freeVramMiB: 4_000 })).toBe( + resolveKvBudgetMiB({ freeVramMiB: 4_000, modelSizeGb: 2.7, mmprojSizeGb: 0 }), + ); + + // Gemma 4 31B on a 128 GB Mac (Metal: three quarters of it free): the + // context is 262,144 either way, and at that context its full cache is + // 29.3 GB — inside the headroom budget's 60 %, far past the share's. + const mac128 = { systemMemoryMiB: 131_072, freeVramMiB: 98_304, modelSizeGb: GEMMA_31B.modelSizeGb, mmprojSizeGb: 0 }; + const ctx = estimateContextSize({ + ...mac128, + maxContextLength: 262_144, + configuredContextSize: 0, + kvLayout: GEMMA4_31B_LAYOUT, + }); + expect(ctx).toBe(262_144); + const pattern = Array.from({ length: 60 }, (_, i) => (i + 1) % 6 !== 0); + const decide = (budgetMiB: number) => + resolveSwaFullDecision({ + preference: "auto", + layout: { + blockCount: 60, + headCountKv: pattern.map((slides) => (slides ? 16 : 4)), + keyLength: 512, + valueLength: 512, + slidingWindow: 1024, + slidingWindowPattern: pattern, + keyLengthSwa: 256, + valueLengthSwa: 256, + }, + contextSize: ctx, + kvBudgetBytes: budgetMiB * 1024 * 1024, + }).enabled; + expect(decide(resolveKvBudgetMiB(mac128))).toBe(true); + expect(decide(resolveContextKvBudgetMiB(mac128))).toBe(false); + }); + + it("leaves a GPU with memory of its own, and a pinned context, as they were", () => { + // No system memory given: a discrete card's free VRAM is its own. + expect( + estimateContextSize({ + ...QWEN35_4B, + freeVramMiB: MAC_16GB_METAL_FREE_MIB, + configuredContextSize: 0, + kvLayout: QWEN35_4B_LAYOUT, + systemMemoryMiB: null, + }), + ).toBe(262_144); + // The operator's number stays as set, unified memory or not. + expect( + estimateContextSize({ + ...QWEN35_4B, + freeVramMiB: MAC_16GB_METAL_FREE_MIB, + configuredContextSize: 262_144, + kvLayout: QWEN35_4B_LAYOUT, + systemMemoryMiB: 16_384, + }), + ).toBe(262_144); + }); +}); + describe("resolveDeviceFreeVramMiB", () => { const devices: GpuDevice[] = [ { diff --git a/src/local-llm/context-size.ts b/src/local-llm/context-size.ts index c6c4226c1..0dce9420e 100644 --- a/src/local-llm/context-size.ts +++ b/src/local-llm/context-size.ts @@ -22,6 +22,81 @@ const VRAM_SAFETY_FRACTION = 0.92; */ const COMPUTE_OVERHEAD_MIB = 768; +// --------------------------------------------------------------------- +// Unified memory: the GPU has no memory of its own. +// +// On Apple silicon (and an integrated GPU) the "free VRAM" a device +// reports is not free memory: Metal answers with its working-set ceiling, +// about two thirds to three quarters of RAM, whatever else is running. A +// fit against that figure alone hands the model server memory the OS, +// the browser and the editor are using. So on such a machine the KV +// budget also leaves the system a headroom of physical RAM, and the +// auto-sized context's own cache is held to a share of it. +// --------------------------------------------------------------------- + +/** + * Share of physical RAM the OS and the person's other apps keep on a + * unified-memory machine: a quarter, never less than + * `UNIFIED_MEMORY_HEADROOM_MIN_MIB`. Weights, projector, compute buffers + * and the KV cache stay within the rest. A headroom for the system rather + * than a share for the server: half of RAM for the server left a 27-31B + * model (17-18 GB of weights) almost nothing on a 32-36 GB Mac — Gemma 4 + * 31B fell to 32,768 tokens on 32 GB and 57,344 on 36 GB, and Fusion's + * `parallel: "auto"` to one slot. + */ +export const UNIFIED_MEMORY_HEADROOM_SHARE = 0.25; + +/** The least the system keeps, on a small machine: 4 GiB. */ +export const UNIFIED_MEMORY_HEADROOM_MIN_MIB = 4 * 1024; + +/** + * Most of physical RAM an auto-sized context's KV cache may take on a + * unified-memory machine: 1 GiB on 16 GB, 2 GiB on 32 GB, 4 GiB on 64 GB. + * + * Without it a small model's cache is cheap enough that "what fits" is + * the model's whole trained context. Qwen 3.5 4B attends on 8 of its 32 + * layers (4 KV heads × 256 dims), 7 KiB per token at turbo3, so 262,144 + * tokens is 1.75 GiB of cache — two thirds of its 2.7 GB of weights — and + * that is what a 16 GB Mac 13.5 GB into swap launched it with, for + * prompts that measured 6-8k tokens. This share gives it 149,504 tokens + * there (1 GiB): at the 90-400 tokens a second that Mac read prompts at, + * a prompt that long already takes six minutes or more. + * + * It sizes the context only. Whether `--swa-full` fits is weighed + * against the headroom budget alone (`resolveKvBudgetMiB`): on a 96-128 + * GB Mac a sliding-window model's full cache is well inside it, and the + * prefix reuse it buys is worth the memory there. + */ +export const UNIFIED_MEMORY_KV_SHARE = 1 / 16; + +/** The RAM (MiB) the OS and other apps keep: `UNIFIED_MEMORY_HEADROOM_*`. */ +export function resolveUnifiedMemoryHeadroomMiB(systemMemoryMiB: number): number { + return Math.max( + UNIFIED_MEMORY_HEADROOM_MIN_MIB, + systemMemoryMiB * UNIFIED_MEMORY_HEADROOM_SHARE, + ); +} + +/** + * The memory (MiB) a unified-memory machine has for the KV cache once the + * headroom, the weights, the projector and the compute buffers are taken + * out. Negative when the weights alone reach into the headroom — the fit + * then lands on `MIN_AUTO_CONTEXT`. + */ +export function resolveUnifiedMemoryKvRoomMiB(input: { + systemMemoryMiB: number; + modelSizeGb: number; + mmprojSizeGb: number; +}): number { + const weightsMiB = (input.modelSizeGb + input.mmprojSizeGb) * MIB_PER_GB; + return ( + input.systemMemoryMiB - + resolveUnifiedMemoryHeadroomMiB(input.systemMemoryMiB) - + weightsMiB - + COMPUTE_OVERHEAD_MIB + ); +} + // --------------------------------------------------------------------- // KV cost from the model's own layout. // @@ -332,6 +407,14 @@ export interface EstimateContextSizeInput { cacheType?: KvCacheType | string; /** Whether the launch carries `--swa-full` (sliding layers at full size). */ swaFull?: boolean; + /** + * Physical RAM (binary MiB) when the target device shares it with the + * system — Apple silicon, an integrated GPU — so the fit also leaves + * the OS and other apps their headroom and holds the cache to its share + * (`resolveContextKvBudgetMiB`). `null` / absent for a GPU with memory + * of its own. + */ + systemMemoryMiB?: number | null; } function roundDownTo(value: number, step: number): number { @@ -340,17 +423,53 @@ function roundDownTo(value: number, step: number): number { /** * The KV budget (MiB) a launch has after weights, projector and compute - * buffers — the memory the context is fitted into. Negative when the - * weights alone do not fit. + * buffers. Negative when the weights alone do not fit. With + * `systemMemoryMiB` (a unified-memory device) it is also held to what + * the machine has once the system's headroom is kept + * (`resolveUnifiedMemoryKvRoomMiB`). This is the budget `--swa-full` is + * weighed against; an auto-sized context is fitted into the tighter + * `resolveContextKvBudgetMiB`. */ export function resolveKvBudgetMiB(input: { freeVramMiB: number; modelSizeGb: number; mmprojSizeGb: number; + systemMemoryMiB?: number | null; }): number { const usableMiB = input.freeVramMiB * VRAM_SAFETY_FRACTION; const weightsMiB = (input.modelSizeGb + input.mmprojSizeGb) * MIB_PER_GB; - return usableMiB - weightsMiB - COMPUTE_OVERHEAD_MIB; + const fitMiB = usableMiB - weightsMiB - COMPUTE_OVERHEAD_MIB; + const systemMemoryMiB = input.systemMemoryMiB; + if (typeof systemMemoryMiB !== "number" || !(systemMemoryMiB > 0)) { + return fitMiB; + } + return Math.min( + fitMiB, + resolveUnifiedMemoryKvRoomMiB({ + systemMemoryMiB, + modelSizeGb: input.modelSizeGb, + mmprojSizeGb: input.mmprojSizeGb, + }), + ); +} + +/** + * The KV budget (MiB) an auto-sized context is fitted into: + * `resolveKvBudgetMiB`, and on a unified-memory device at most + * `UNIFIED_MEMORY_KV_SHARE` of physical RAM. + */ +export function resolveContextKvBudgetMiB(input: { + freeVramMiB: number; + modelSizeGb: number; + mmprojSizeGb: number; + systemMemoryMiB?: number | null; +}): number { + const budgetMiB = resolveKvBudgetMiB(input); + const systemMemoryMiB = input.systemMemoryMiB; + if (typeof systemMemoryMiB !== "number" || !(systemMemoryMiB > 0)) { + return budgetMiB; + } + return Math.min(budgetMiB, systemMemoryMiB * UNIFIED_MEMORY_KV_SHARE); } /** @@ -392,10 +511,12 @@ export function fitContextToKvBudget( * caller. When `configuredContextSize > 0` the operator's value wins * (clamped to the model ceiling). Otherwise the size is fitted to free * VRAM: weights + projector + compute overhead are subtracted, and the - * remainder is what the KV cache may take — costed from the model's - * layout when the header was readable, from the file-size scale when - * not — then clamped into `[MIN_AUTO_CONTEXT, MAX_AUTO_CONTEXT]` and the - * model's trained ceiling. + * remainder is what the KV cache may take — on a unified-memory machine + * (`systemMemoryMiB`) also within the system's headroom and the cache's + * share of RAM (`resolveContextKvBudgetMiB`), costed from the + * model's layout when the header was readable, from the file-size scale + * when not — then clamped into `[MIN_AUTO_CONTEXT, MAX_AUTO_CONTEXT]` and + * the model's trained ceiling. */ export function estimateContextSize(input: EstimateContextSizeInput): number { const { @@ -416,10 +537,11 @@ export function estimateContextSize(input: EstimateContextSizeInput): number { return Math.min(NO_VRAM_DEFAULT_CONTEXT, ceiling); } - const kvBudgetMiB = resolveKvBudgetMiB({ + const kvBudgetMiB = resolveContextKvBudgetMiB({ freeVramMiB, modelSizeGb, mmprojSizeGb, + systemMemoryMiB: input.systemMemoryMiB ?? null, }); const kvBudgetBytes = kvBudgetMiB > 0 ? kvBudgetMiB * 1024 * 1024 : 0; @@ -446,6 +568,35 @@ export function estimateContextSize(input: EstimateContextSizeInput): number { return Math.min(roundDownTo(clamped, 1024), ceiling); } +/** + * Bytes of KV cache a launch's context costs, the way the auto-size + * costs it: from the model's layout when its header was readable (with + * `--swa-full`, when the launch carries it), from the file-size fallback + * when not. Set against `resolveKvBudgetMiB` without a system memory + * figure, it says whether the weights and the cache fit the device whole + * or llama.cpp's `-fit` will leave layers on the CPU. + */ +export function estimateLaunchKvBytes(input: { + kvLayout: KvCacheLayout | null; + modelSizeGb: number; + contextSize: number; + cacheType?: KvCacheType | string; + swaFull?: boolean; +}): number { + if (input.kvLayout && input.kvLayout.layers.length > 0) { + return estimateKvBytesTotal( + input.kvLayout, + input.cacheType ?? MANAGED_KV_CACHE_TYPE, + input.contextSize, + { swaFull: input.swaFull === true }, + ); + } + return ( + input.contextSize * + Math.max(KV_MIN_BYTES_PER_TOKEN, input.modelSizeGb * KV_BYTES_PER_TOKEN_PER_GB) + ); +} + /** * Look up the free VRAM (binary MiB) for the resolved offload device, or * `null` when there is no usable figure. Falls back to `totalMemMiB` diff --git a/src/local-llm/daemon-lifecycle.test.ts b/src/local-llm/daemon-lifecycle.test.ts index 0c46df234..acc583d49 100644 --- a/src/local-llm/daemon-lifecycle.test.ts +++ b/src/local-llm/daemon-lifecycle.test.ts @@ -19,6 +19,12 @@ vi.mock("node:child_process", () => ({ execSync: execSyncMock, execFile: execFileMock, })); +/** Physical RAM as the launch reads it (`os.totalmem`); 16 GB unless a test says otherwise. */ +const totalmemMock = vi.hoisted(() => vi.fn(() => 16 * 1024 ** 3)); +vi.mock("node:os", async (importOriginal) => { + const actual = await importOriginal(); + return { ...actual, totalmem: totalmemMock }; +}); import { resolveEmbeddingPidFilePath, @@ -36,11 +42,15 @@ import { getDaemonStatus, probeThroughput, readLaunchRecord, + readReusableThroughput, readRunningPid, readThroughputRecord, + slotsAllIdle, startDaemon, startEmbeddingDaemon, THROUGHPUT_PROBE_TOKENS, + THROUGHPUT_REUSE_MAX_AGE_MS, + throughputBasis, writeLaunchRecord, writeThroughputRecord, stopDaemon, @@ -48,9 +58,14 @@ import { type DaemonStartOptions, type EmbeddingDaemonStartOptions, } from "./daemon-lifecycle.js"; +import { writeBackendVersion } from "./backend-version.js"; import { getEmbeddingModelDef, getLocalModelDef } from "./models-catalog.js"; import { resetConfigCache } from "../config/config-cache.js"; -import { encodeSyntheticGguf, gemma4Pairs } from "./gguf-metadata.fixtures.js"; +import { + encodeSyntheticGguf, + gemma4Pairs, + qwen35Pairs, +} from "./gguf-metadata.fixtures.js"; import { EventEmitter } from "node:events"; /** A spawned llama-server that stays up until `exit` is emitted. */ @@ -783,6 +798,12 @@ describe("startDaemon throughput probe (F16)", () => { if (String(url).endsWith("/v1/models")) { return new Response(JSON.stringify({ data: [{ id: "qwen-3.5-4b" }] }), { status: 200 }); } + if (String(url).endsWith("/slots")) { + return new Response( + JSON.stringify([{ id: 0, is_processing: false }, { id: 1, is_processing: false }]), + { status: 200 }, + ); + } posts.push(String(url)); void init; return new Response( @@ -1031,6 +1052,11 @@ describe("managed launch requires an api key (#582)", () => { if (String(url).endsWith("/v1/models")) { return new Response(JSON.stringify({ data: [{ id: alias }] }), { status: 200 }); } + // A build without the endpoint: the probe's figure is not carried + // over, so each launch below probes and shows its key. + if (String(url).endsWith("/slots")) { + return new Response("Not Found", { status: 404 }); + } const headers = (init?.headers ?? {}) as Record; probeAuth.push(headers.authorization ?? null); return new Response( @@ -1128,3 +1154,422 @@ describe("managed launch requires an api key (#582)", () => { } }); }); + +/** + * Backlog 39 and 42, on the managed start: one `--list-devices` per + * launch, the auto context held to a unified-memory Mac's RAM, and the + * decode speed carried over instead of measured on every start. + */ +describe("startDaemon on a 16 GB Mac (backlog 39, 42)", () => { + afterEach(() => { + vi.unstubAllGlobals(); + spawnMock.mockReset(); + execFileMock.mockReset(); + totalmemMock.mockReset(); + totalmemMock.mockReturnValue(16 * 1024 ** 3); + }); + + const BUILD = { + tag: "turboquant-6df272c", + downloadedAt: "2026-10-01T21:39:24.093Z", + asset: "llama-turboquant-macos-arm64.zip", + }; + + /** Backend, its version record and Qwen 3.5 4B's real header on disk. */ + function stageQwen35(dataDir: string): void { + const binPath = resolveServerBinPath(dataDir, "llama-server"); + mkdirSync(dirname(binPath), { recursive: true }); + writeFileSync(binPath, "#!/bin/sh\n", "utf-8"); + writeBackendVersion(dataDir, BUILD); + const model = getLocalModelDef("qwen-3.5-4b"); + const modelPath = resolveModelFilePath(dataDir, model.id, model.filename); + mkdirSync(dirname(modelPath), { recursive: true }); + writeFileSync(modelPath, encodeSyntheticGguf(qwen35Pairs())); + } + + /** `llama-server --list-devices` answers with `rows`. */ + function listDevicesAnswers(...rows: string[]): void { + execFileMock.mockImplementation( + ( + _file: string, + _args: string[], + _opts: unknown, + done: (err: Error | null, out: { stdout: string; stderr: string }) => void, + ) => { + done(null, { stdout: ["Available devices:", ...rows.map((r) => ` ${r}`)].join("\n"), stderr: "" }); + }, + ); + } + + /** + * A healthy server; returns the probe POSTs. `busySlot`: a slot decoding + * at every look at /slots; `busyAtFirstLook`: only at the first (a turn + * that was running as the probe began, and ended during it). + */ + function healthyServer( + opts: { speed?: number; busySlot?: boolean; busyAtFirstLook?: boolean; alias?: string } = {}, + ): string[] { + const posts: string[] = []; + let looks = 0; + vi.stubGlobal( + "fetch", + vi.fn(afterSpawn(async (url: string) => { + const at = String(url); + const json = (body: unknown) => new Response(JSON.stringify(body), { status: 200 }); + if (at.endsWith("/health")) return json({ status: "ok" }); + if (at.endsWith("/v1/models")) return json({ data: [{ id: opts.alias ?? "qwen-3.5-4b" }] }); + if (at.endsWith("/slots")) { + looks += 1; + const busy = opts.busySlot === true || (opts.busyAtFirstLook === true && looks === 1); + return json([ + { id: 0, is_processing: false }, + { id: 1, is_processing: busy }, + ]); + } + posts.push(at); + return json({ timings: { predicted_n: 64, predicted_per_second: opts.speed ?? 19.45 } }); + })), + ); + return posts; + } + + /** What a `device: "cpu"` launch of Qwen 3.5 4B is: the no-VRAM 32,768, no free figure to fit against. */ + const CPU_LAUNCH = { contextSize: 32_768, fitsDevice: null }; + + /** The server stopped (its pid file gone) and the port free again before the next start. */ + function stopped(dataDir: string): void { + rmSync(resolvePidFilePath(dataDir), { force: true }); + spawnMock.mockClear(); + } + + function spawnedArg(flag: string): string | undefined { + const args = spawnMock.mock.calls[0]![1] as string[]; + const at = args.indexOf(flag); + return at < 0 ? undefined : args[at + 1]; + } + + it("asks --list-devices once, and holds Qwen 3.5 4B to the Mac's memory: 149,504 tokens, not 262,144", async () => { + const dataDir = mkdtempSync(`${tmpdir()}/atomic-daemon-mac16-`); + try { + stageQwen35(dataDir); + listDevicesAnswers("MTL0: Apple M4 (10922 MiB, 10922 MiB free)"); + spawnMock.mockReturnValue(fakeChild(6001)); + healthyServer(); + const result = await startDaemon({ + dataDir, + modelId: "qwen-3.5-4b", + port: 19089, + parallel: 1, + throughputProbe: false, + }); + // The device pick and the context fit read one table. + expect(execFileMock).toHaveBeenCalledTimes(1); + expect(execFileMock.mock.calls[0]![1]).toEqual(["--list-devices"]); + expect(spawnedArg("--device")).toBe("MTL0"); + expect(result.contextSize).toBe(149_504); + expect(spawnedArg("--ctx-size")).toBe("149504"); + const log = readFileSync(resolveLogFilePath(dataDir), "utf-8"); + expect(log).toContain( + "[atomic-agent] launch: --ctx-size 149504 fitted from the model's KV layout, held to 16 GB of unified memory", + ); + expect(log).toContain("(262144 would fit the GPU's free figure)"); + } finally { + rmSync(dataDir, { recursive: true, force: true }); + } + }); + + it("gives an 8 GB Mac 74,752 tokens and leaves a card with memory of its own at the fit", async () => { + const dataDir = mkdtempSync(`${tmpdir()}/atomic-daemon-mac8-`); + try { + stageQwen35(dataDir); + totalmemMock.mockReturnValue(8 * 1024 ** 3); + listDevicesAnswers("MTL0: Apple M2 (5461 MiB, 5461 MiB free)"); + spawnMock.mockReturnValue(fakeChild(6002)); + healthyServer(); + const mac = await startDaemon({ dataDir, modelId: "qwen-3.5-4b", port: 19088, parallel: 1, throughputProbe: false }); + expect(mac.contextSize).toBe(74_752); + + stopped(dataDir); + totalmemMock.mockReturnValue(16 * 1024 ** 3); + listDevicesAnswers("CUDA0: NVIDIA GeForce RTX 4090 (24564 MiB, 10922 MiB free)"); + spawnMock.mockReturnValue(fakeChild(6003)); + const card = await startDaemon({ dataDir, modelId: "qwen-3.5-4b", port: 19088, parallel: 1, throughputProbe: false }); + expect(card.contextSize).toBe(262_144); + expect(readFileSync(resolveLogFilePath(dataDir), "utf-8")).not.toMatch( + /--ctx-size 262144 fitted from the model's KV layout, held to/, + ); + } finally { + rmSync(dataDir, { recursive: true, force: true }); + } + }); + + it("keeps a contextSize the operator set, on the same Mac", async () => { + const dataDir = mkdtempSync(`${tmpdir()}/atomic-daemon-mac-pinned-`); + try { + stageQwen35(dataDir); + listDevicesAnswers("MTL0: Apple M4 (10922 MiB, 10922 MiB free)"); + spawnMock.mockReturnValue(fakeChild(6004)); + healthyServer(); + const result = await startDaemon({ + dataDir, + modelId: "qwen-3.5-4b", + port: 19087, + parallel: 1, + contextSize: 262_144, + throughputProbe: false, + }); + expect(result.contextSize).toBe(262_144); + expect(spawnedArg("--ctx-size")).toBe("262144"); + } finally { + rmSync(dataDir, { recursive: true, force: true }); + } + }); + + it("measures the speed once, then carries it over: the next start of the same model does not probe", async () => { + const dataDir = mkdtempSync(`${tmpdir()}/atomic-daemon-speed-`); + try { + stageQwen35(dataDir); + const posts = healthyServer({ speed: 19.45 }); + spawnMock.mockReturnValue(fakeChild(6101)); + const first = await startDaemon({ dataDir, modelId: "qwen-3.5-4b", port: 19086, device: "cpu", parallel: 1 }); + expect(first.tokensPerSecond).toBe(19.45); + expect(posts).toEqual(["http://127.0.0.1:19086/completion"]); + const measured = readThroughputRecord(dataDir, 6101); + expect(measured).toMatchObject({ + tokensPerSecond: 19.45, + alone: true, + measuredOn: throughputBasis(dataDir, "qwen-3.5-4b", "cpu", CPU_LAUNCH), + }); + + stopped(dataDir); + spawnMock.mockReturnValue(fakeChild(6102)); + const second = await startDaemon({ dataDir, modelId: "qwen-3.5-4b", port: 19086, device: "cpu", parallel: 1 }); + expect(second.tokensPerSecond).toBe(19.45); + expect(posts).toHaveLength(1); + // Stamped for the daemon now running, with the time it was measured. + expect(readThroughputRecord(dataDir, 6102)).toMatchObject({ + tokensPerSecond: 19.45, + measuredAt: measured!.measuredAt, + }); + expect(readThroughputRecord(dataDir, 6101)).toBeNull(); + } finally { + rmSync(dataDir, { recursive: true, force: true }); + } + }); + + it("measures again after a llama.cpp update, on another device or context, and once the figure is a day old", async () => { + const dataDir = mkdtempSync(`${tmpdir()}/atomic-daemon-speed-again-`); + try { + stageQwen35(dataDir); + const posts = healthyServer(); + let pid = 6200; + const start = async (device: string, contextSize?: number) => { + stopped(dataDir); + spawnMock.mockReturnValue(fakeChild(++pid)); + listDevicesAnswers("MTL0: Apple M4 (10922 MiB, 10922 MiB free)"); + return startDaemon({ + dataDir, + modelId: "qwen-3.5-4b", + port: 19085, + device, + parallel: 1, + ...(contextSize ? { contextSize } : {}), + }); + }; + await start("cpu"); + expect(posts).toHaveLength(1); + writeBackendVersion(dataDir, { ...BUILD, tag: "turboquant-7a1c0de", downloadedAt: "2026-10-03T08:00:00.000Z" }); + await start("cpu"); + expect(posts).toHaveLength(2); + await start("MTL0"); + expect(posts).toHaveLength(3); + await start("MTL0"); + expect(posts).toHaveLength(3); + // The same device with another context is another launch. + await start("MTL0", 65_536); + expect(posts).toHaveLength(4); + await start("MTL0", 65_536); + expect(posts).toHaveLength(4); + // The same record, a day and a minute old. + expect(THROUGHPUT_REUSE_MAX_AGE_MS).toBe(24 * 60 * 60 * 1000); + const record = readThroughputRecord(dataDir, pid)!; + writeThroughputRecord(dataDir, { + ...record, + measuredAt: Date.now() - THROUGHPUT_REUSE_MAX_AGE_MS - 60_000, + }); + await start("MTL0", 65_536); + expect(posts).toHaveLength(5); + } finally { + rmSync(dataDir, { recursive: true, force: true }); + } + }); + + it("measures a launch that does not fit the card whole apart from one that does", async () => { + // Free VRAM on a 12 GB card: 3,600 MiB leaves nothing for the cache + // after Qwen 3.5 4B's weights (the floor's 224 MiB spills layers to + // the CPU); 3,881 MiB leaves 227 MiB, enough for the same 32,768. + const dataDir = mkdtempSync(`${tmpdir()}/atomic-daemon-speed-fit-`); + try { + stageQwen35(dataDir); + const posts = healthyServer(); + let pid = 6250; + const start = async (freeMiB: number) => { + stopped(dataDir); + spawnMock.mockReturnValue(fakeChild(++pid)); + listDevicesAnswers(`CUDA0: NVIDIA GeForce RTX 3060 (12288 MiB, ${freeMiB} MiB free)`); + return startDaemon({ dataDir, modelId: "qwen-3.5-4b", port: 19083, parallel: 1 }); + }; + const spills = await start(3_600); + expect(spills.contextSize).toBe(32_768); + expect(readThroughputRecord(dataDir, pid)?.measuredOn).toBe( + throughputBasis(dataDir, "qwen-3.5-4b", "CUDA0", { contextSize: 32_768, fitsDevice: false }), + ); + const fits = await start(3_881); + expect(fits.contextSize).toBe(32_768); + expect(posts).toHaveLength(2); + expect(readThroughputRecord(dataDir, pid)?.measuredOn).toBe( + throughputBasis(dataDir, "qwen-3.5-4b", "CUDA0", { contextSize: 32_768, fitsDevice: true }), + ); + await start(3_881); + expect(posts).toHaveLength(2); + } finally { + rmSync(dataDir, { recursive: true, force: true }); + } + }); + + it("does not carry over a speed measured while another slot was decoding", async () => { + const dataDir = mkdtempSync(`${tmpdir()}/atomic-daemon-speed-busy-`); + try { + stageQwen35(dataDir); + let posts = healthyServer({ speed: 1.14, busySlot: true }); + spawnMock.mockReturnValue(fakeChild(6301)); + await startDaemon({ dataDir, modelId: "qwen-3.5-4b", port: 19084, device: "cpu", parallel: 8 }); + expect(posts).toHaveLength(1); + expect(readThroughputRecord(dataDir, 6301)).toMatchObject({ tokensPerSecond: 1.14, alone: false }); + + // Measured again — this time with every slot idle, which stands. + stopped(dataDir); + posts = healthyServer({ speed: 13.66 }); + spawnMock.mockReturnValue(fakeChild(6302)); + const again = await startDaemon({ dataDir, modelId: "qwen-3.5-4b", port: 19084, device: "cpu", parallel: 8 }); + expect(posts).toHaveLength(1); + expect(again.tokensPerSecond).toBe(13.66); + expect( + readReusableThroughput(dataDir, throughputBasis(dataDir, "qwen-3.5-4b", "cpu", CPU_LAUNCH)), + ).toMatchObject({ tokensPerSecond: 13.66 }); + } finally { + rmSync(dataDir, { recursive: true, force: true }); + } + }); + + it("does not carry over a speed measured while a turn was running as the probe began", async () => { + const dataDir = mkdtempSync(`${tmpdir()}/atomic-daemon-speed-busy-before-`); + try { + stageQwen35(dataDir); + // Busy at the look before the probe, idle by the look after it. + const posts = healthyServer({ speed: 4.4, busyAtFirstLook: true }); + spawnMock.mockReturnValue(fakeChild(6311)); + await startDaemon({ dataDir, modelId: "qwen-3.5-4b", port: 19082, device: "cpu", parallel: 8 }); + expect(posts).toHaveLength(1); + expect(readThroughputRecord(dataDir, 6311)).toMatchObject({ tokensPerSecond: 4.4, alone: false }); + expect( + readReusableThroughput(dataDir, throughputBasis(dataDir, "qwen-3.5-4b", "cpu", CPU_LAUNCH)), + ).toBeNull(); + } finally { + rmSync(dataDir, { recursive: true, force: true }); + } + }); + + /** Gemma 4 31B's header and the catalogue's gemma-4-31b on disk (no projector). */ + function stageGemma31(dataDir: string): void { + const binPath = resolveServerBinPath(dataDir, "llama-server"); + mkdirSync(dirname(binPath), { recursive: true }); + writeFileSync(binPath, "#!/bin/sh\n", "utf-8"); + writeBackendVersion(dataDir, BUILD); + const model = getLocalModelDef("gemma-4-31b"); + const modelPath = resolveModelFilePath(dataDir, model.id, model.filename); + mkdirSync(dirname(modelPath), { recursive: true }); + writeFileSync(modelPath, encodeSyntheticGguf(gemma4Pairs())); + } + + it("gives Gemma 4 31B on a 32 GB Mac 109,568 tokens and three auto slots, where half of RAM floored it at one", async () => { + const dataDir = mkdtempSync(`${tmpdir()}/atomic-daemon-gemma32-`); + try { + stageGemma31(dataDir); + totalmemMock.mockReturnValue(32 * 1024 ** 3); + listDevicesAnswers("MTL0: Apple M2 Max (21845 MiB, 21845 MiB free)"); + spawnMock.mockReturnValue(fakeChild(6401)); + healthyServer({ alias: "gemma-4-31b" }); + const result = await startDaemon({ + dataDir, + modelId: "gemma-4-31b", + port: 19081, + parallel: "auto", + completionMaxTokens: 16_384, + throughputProbe: false, + }); + expect(result.contextSize).toBe(109_568); + expect(spawnedArg("--parallel")).toBe("3"); + // Its full sliding-window cache (12.9 GB at this context) is far past the budget here. + expect(result.swaFull.enabled).toBe(false); + } finally { + rmSync(dataDir, { recursive: true, force: true }); + } + }); + + it("keeps --swa-full for Gemma 4 31B on a 128 GB Mac: weighed against the headroom, not the context's share", async () => { + const dataDir = mkdtempSync(`${tmpdir()}/atomic-daemon-gemma128-`); + try { + stageGemma31(dataDir); + totalmemMock.mockReturnValue(128 * 1024 ** 3); + listDevicesAnswers("MTL0: Apple M4 Max (98304 MiB, 98304 MiB free)"); + spawnMock.mockReturnValue(fakeChild(6402)); + healthyServer({ alias: "gemma-4-31b" }); + const result = await startDaemon({ + dataDir, + modelId: "gemma-4-31b", + port: 19080, + parallel: 1, + throughputProbe: false, + }); + expect(result.contextSize).toBe(262_144); + expect(result.swaFull.enabled).toBe(true); + expect(spawnMock.mock.calls[0]![1] as string[]).toContain("--swa-full"); + expect(result.prefixReuse).toBe("partial"); + expect(readFileSync(resolveLogFilePath(dataDir), "utf-8")).toContain( + "[atomic-agent] launch: swa-full: on (auto)", + ); + } finally { + rmSync(dataDir, { recursive: true, force: true }); + } + }); +}); + +describe("slotsAllIdle", () => { + const answering = (status: number, body: unknown) => + (async () => new Response(JSON.stringify(body), { status })) as unknown as typeof fetch; + + it("is true only when /slots answers and no slot is processing", async () => { + const idle = [{ id: 0, is_processing: false }, { id: 1, is_processing: false }]; + expect(await slotsAllIdle({ port: 1, fetchImpl: answering(200, idle) })).toBe(true); + const busy = [{ id: 0, is_processing: false }, { id: 1, is_processing: true }]; + expect(await slotsAllIdle({ port: 1, fetchImpl: answering(200, busy) })).toBe(false); + expect(await slotsAllIdle({ port: 1, fetchImpl: answering(200, []) })).toBe(false); + expect(await slotsAllIdle({ port: 1, fetchImpl: answering(200, { slots: idle }) })).toBe(false); + expect(await slotsAllIdle({ port: 1, fetchImpl: answering(404, "Not Found") })).toBe(false); + const hung = (async () => { + throw new DOMException("The operation was aborted due to timeout", "TimeoutError"); + }) as unknown as typeof fetch; + expect(await slotsAllIdle({ port: 1, fetchImpl: hung })).toBe(false); + }); + + it("sends the server's key", async () => { + let auth: string | null = null; + const fetchImpl = (async (_url: string, init?: RequestInit) => { + auth = ((init?.headers ?? {}) as Record).authorization ?? null; + return new Response("[]", { status: 200 }); + }) as unknown as typeof fetch; + await slotsAllIdle({ port: 1, apiKey: "k", fetchImpl }); + expect(auth).toBe("Bearer k"); + }); +}); diff --git a/src/local-llm/daemon-lifecycle.ts b/src/local-llm/daemon-lifecycle.ts index 14d431205..53de9d22c 100644 --- a/src/local-llm/daemon-lifecycle.ts +++ b/src/local-llm/daemon-lifecycle.ts @@ -11,7 +11,9 @@ import { writeFileSync, writeSync, } from "node:fs"; +import { totalmem } from "node:os"; +import { readBackendVersion } from "./backend-version.js"; import { resolveEmbeddingLogFilePath, resolveEmbeddingPidFilePath, @@ -25,9 +27,12 @@ import { import { buildKvLayout, estimateContextSize, + estimateLaunchKvBytes, MANAGED_KV_CACHE_TYPE, resolveDeviceFreeVramMiB, resolveKvBudgetMiB, + resolveUnifiedMemoryHeadroomMiB, + UNIFIED_MEMORY_KV_SHARE, type KvLayoutSource, } from "./context-size.js"; import { @@ -36,7 +41,12 @@ import { readGgufMetadataSync, type GgufMetadata, } from "./gguf-metadata.js"; -import { listVulkanDevices, resolveManagedDevice } from "./gpu-devices.js"; +import { + deviceTableOnce, + resolveManagedDevice, + sharesSystemMemory, + type ListDevices, +} from "./gpu-devices.js"; import { resolveSwaFullDecision, type SwaFullDecision, @@ -79,6 +89,13 @@ export interface DaemonStartOptions { * value passed here is treated as an explicit override. */ device?: string; + /** + * The launch's `--list-devices` table (`deviceTableOnce`) when the + * caller already asked it to pick `device`: the context fit reads the + * free memory from the same answer instead of starting the backend a + * second time. Absent: `startDaemon` enumerates, at most once. + */ + listDevices?: ListDevices; /** * Operator override for the llama-server context window * (`localModels.managed.contextSize`). `0` / `undefined` means @@ -140,7 +157,11 @@ export interface DaemonStartOptions { * `true`): one 64-token completion whose `timings.predicted_per_second` * is written next to the pid file and shown to the fusion orchestrator * as "~N tok/s single stream". Costs a few seconds of readiness — a - * 31B model at 3 tok/s spends ~20 s on it. `false` skips it. + * 31B model at 3 tok/s spends ~20 s on it, a 4B one 3-5 s on a 16 GB + * Mac — so a speed an earlier start measured on the same launch (model, + * build, device, context, whole fit) in the last day is carried over + * instead (`readReusableThroughput`). `false` skips both and leaves the + * speed unknown. */ throughputProbe?: boolean; /** @@ -167,8 +188,60 @@ export interface ThroughputRecord extends ThroughputSample { /** Pid of the daemon instance the sample belongs to. */ pid: number; modelId: string; - /** Epoch ms of the measurement. */ + /** Epoch ms of the measurement (kept when a later start carries it over). */ measuredAt: number; + /** + * What the speed was measured on (`throughputBasis`: model, llama.cpp + * build and install, device, context, whether the model fit the device + * whole). Absent on records written before speeds were carried over; + * those are never reused. + */ + measuredOn?: string; + /** + * Whether the probe had the server to itself: it ran on the only slot, + * or no slot was busy as it began and as it ended (`slotsAllIdle`). Only + * such a figure is carried over. On eight slots, requests that came in + * during the probe decoded beside it and it measured 1.1 tok/s, against + * 13-22 alone, for the same model on the same Mac. + */ + alone?: boolean; +} + +/** + * How long a measured speed is carried over to later starts of the same + * launch before it is measured again. Speed is a property of the model, + * the machine and how the model sits on it, not of one server process; + * re-measuring it on every start kept the model from answering for + * another 3-5 s after it had loaded, on every switch back to the local + * model. A day, so a figure taken on a bad afternoon (thermal limits, + * another app on the GPU) does not outlive it by much. + */ +export const THROUGHPUT_REUSE_MAX_AGE_MS = 24 * 60 * 60 * 1000; + +/** + * What a speed measurement describes: the model, the llama.cpp build on + * disk (its tag and when it was installed, so an update or a reinstall + * measures again), the device, the launch's context, and whether the + * weights and that context's cache fit the device whole (`fitsDevice`) — + * when they do not, llama.cpp's `-fit` leaves layers on the CPU and the + * model decodes at another speed; `null` when there was no free figure + * to tell. + */ +export function throughputBasis( + dataDir: string, + modelId: string, + device: string | undefined, + launch: { contextSize: number; fitsDevice: boolean | null }, +): string { + const backend = readBackendVersion(dataDir); + return JSON.stringify([ + modelId, + backend?.tag ?? null, + backend?.downloadedAt ?? null, + device ?? null, + launch.contextSize, + launch.fitsDevice, + ]); } /** Tokens the probe asks for: enough to average out the first-token cost. */ @@ -349,12 +422,98 @@ export function readThroughputRecord( predictedTokens: toFiniteNumber(parsed.predictedTokens) ?? 0, promptTokensPerSecond: toFiniteNumber(parsed.promptTokensPerSecond), measuredAt: toFiniteNumber(parsed.measuredAt) ?? 0, + ...(typeof parsed.measuredOn === "string" ? { measuredOn: parsed.measuredOn } : {}), + ...(typeof parsed.alone === "boolean" ? { alone: parsed.alone } : {}), }; } catch { return null; } } +/** + * The speed an earlier start measured on this same `basis` + * (`throughputBasis`), whichever daemon it was stamped for — `null` when + * there is none, it describes another model, build or device, the probe + * did not have the server to itself (`alone`), or it is older than + * `THROUGHPUT_REUSE_MAX_AGE_MS`. Read before a launch clears the record. + */ +export function readReusableThroughput( + dataDir: string, + basis: string, + now: number = Date.now(), +): ThroughputRecord | null { + let parsed: Partial; + try { + parsed = JSON.parse( + readFileSync(resolveThroughputFilePath(dataDir), "utf-8"), + ) as Partial; + } catch { + return null; + } + const tokensPerSecond = toFiniteNumber(parsed.tokensPerSecond); + const measuredAt = toFiniteNumber(parsed.measuredAt); + if ( + parsed.measuredOn !== basis || + parsed.alone !== true || + typeof parsed.modelId !== "string" || + typeof parsed.pid !== "number" || + tokensPerSecond === null || + tokensPerSecond <= 0 || + measuredAt === null || + now < measuredAt || + now - measuredAt > THROUGHPUT_REUSE_MAX_AGE_MS + ) { + return null; + } + return { + pid: parsed.pid, + modelId: parsed.modelId, + tokensPerSecond, + predictedTokens: toFiniteNumber(parsed.predictedTokens) ?? 0, + promptTokensPerSecond: toFiniteNumber(parsed.promptTokensPerSecond), + measuredAt, + measuredOn: basis, + alone: true, + }; +} + +/** + * Whether no slot of the server on `port` is busy: `GET /slots` answered + * in time and every slot says `is_processing: false`. A busy slot, a + * `/slots` that does not answer in time (it hangs while a slot evaluates + * a long prompt), a refusal, or a build without the endpoint all answer + * `false`. Never throws. + */ +export async function slotsAllIdle(opts: { + port: number; + apiKey?: string | null; + fetchImpl?: typeof fetch; + timeoutMs?: number; +}): Promise { + try { + const headers: Record = { accept: "application/json" }; + if (opts.apiKey) headers.authorization = `Bearer ${opts.apiKey}`; + const res = await (opts.fetchImpl ?? fetch)( + `http://127.0.0.1:${opts.port}/slots`, + { headers, signal: AbortSignal.timeout(opts.timeoutMs ?? 1_500) }, + ); + if (!res.ok) return false; + const slots = (await res.json()) as unknown; + return ( + Array.isArray(slots) && + slots.length > 0 && + slots.every( + (slot) => + typeof slot === "object" && + slot !== null && + (slot as { is_processing?: unknown }).is_processing === false, + ) + ); + } catch { + return false; + } +} + /** * Build the full llama-server CLI argv for a managed-mode launch. * Pure function — no IO, no spawn, no path validation. Extracted from @@ -464,13 +623,24 @@ export function buildLlamaServerArgs( /** * Resolve the effective `--ctx-size` for a chat daemon launch. Impure * glue around the pure `estimateContextSize`: when auto-sizing on a GPU - * device it enumerates `--list-devices` to read the target device's free - * VRAM. Best-effort — any enumeration failure degrades to the no-VRAM - * default. Skips the probe entirely when the operator pinned a value or - * offload is CPU-only. + * device it reads the target device's free VRAM from the launch's + * `--list-devices` table, and on a device that shares the system's RAM + * (Apple silicon, an integrated GPU) the machine's physical memory too. + * Best-effort — any enumeration failure degrades to the no-VRAM default. + * Skips the probe entirely when the operator pinned a value or offload + * is CPU-only. + * + * - `kvBudgetBytes`: what the cache may take, within the system's + * headroom on unified memory — the budget `--swa-full` is weighed + * against (not the context's 1/16 share). + * - `deviceBudgetBytes`: what is left of the device's own free figure + * after weights, projector and compute buffers (negative when the + * weights alone overflow it), against which a launch's cache says + * whether the model fits the device whole. + * - `note`: why a unified-memory machine got less than its free figure + * would fit, for the daemon log. */ async function resolveEffectiveContextSize( - binPath: string, device: string | undefined, model: { fileSizeGb: number; @@ -482,15 +652,26 @@ async function resolveEffectiveContextSize( hasMmproj: boolean; /** The model's attention layout from its header, when readable. */ kvLayout?: KvLayoutSource | null; + listDevices: ListDevices; }, -): Promise<{ contextSize: number; kvBudgetBytes: number | null }> { +): Promise<{ + contextSize: number; + kvBudgetBytes: number | null; + deviceBudgetBytes: number | null; + note: string | null; +}> { let freeVramMiB: number | null = null; + let systemMemoryMiB: number | null = null; if (opts.configured <= 0 && device && device !== "cpu") { - const devices = await listVulkanDevices(binPath); + const devices = await opts.listDevices(); freeVramMiB = resolveDeviceFreeVramMiB(devices, device); + const target = devices.find((d) => d.id === device); + if (freeVramMiB !== null && target && sharesSystemMemory(target)) { + systemMemoryMiB = totalmem() / (1024 * 1024); + } } const mmprojSizeGb = opts.hasMmproj ? (model.mmprojFileSizeGb ?? 0) : 0; - const contextSize = estimateContextSize({ + const input = { freeVramMiB, modelSizeGb: model.fileSizeGb, mmprojSizeGb, @@ -498,21 +679,37 @@ async function resolveEffectiveContextSize( configuredContextSize: opts.configured, kvLayout: opts.kvLayout ? buildKvLayout(opts.kvLayout) : null, cacheType: MANAGED_KV_CACHE_TYPE, - }); - const kvBudgetBytes = + }; + const contextSize = estimateContextSize({ ...input, systemMemoryMiB }); + let note: string | null = null; + if (systemMemoryMiB !== null) { + const fits = estimateContextSize(input); + if (fits > contextSize) { + note = + `held to ${Math.round(systemMemoryMiB / 1024)} GB of unified memory: ` + + `at most 1/${Math.round(1 / UNIFIED_MEMORY_KV_SHARE)} of it for the KV cache, ` + + `${Math.round(resolveUnifiedMemoryHeadroomMiB(systemMemoryMiB) / 1024)} GB left to the system ` + + `(${fits} would fit the GPU's free figure)`; + } + } + const budgetBytes = (withSystemMemory: boolean): number | null => freeVramMiB !== null && freeVramMiB > 0 - ? Math.max( - 0, - resolveKvBudgetMiB({ - freeVramMiB, - modelSizeGb: model.fileSizeGb, - mmprojSizeGb, - }), - ) * + ? resolveKvBudgetMiB({ + freeVramMiB, + modelSizeGb: model.fileSizeGb, + mmprojSizeGb, + systemMemoryMiB: withSystemMemory ? systemMemoryMiB : null, + }) * 1024 * 1024 : null; - return { contextSize, kvBudgetBytes }; + const kvBudgetBytes = budgetBytes(true); + return { + contextSize, + kvBudgetBytes: kvBudgetBytes === null ? null : Math.max(0, kvBudgetBytes), + deviceBudgetBytes: budgetBytes(false), + note, + }; } /** @@ -709,24 +906,29 @@ export async function startDaemon( // With no pinned device the context auto-sizer has no single VRAM // figure to probe and degrades to its conservative no-VRAM default; // operators splitting across GPUs can pin `contextSize` explicitly. + // One `--list-devices` for the whole launch: the device pick and the + // context fit below read the same table (`deviceTableOnce`). + const listDevices = opts.listDevices ?? deviceTableOnce(binPath); const device = await resolveManagedDevice(binPath, opts.device, { multiGpu: (opts.tensorSplit?.length ?? 0) > 0, + listDevices, }); // The header says what the KV cache really costs and whether the // model's cache can be reused partially — both decide flags below. const header = readModelHeader(modelPath); const kvLayout = header ? kvLayoutSourceFromMetadata(header) : null; const prefixReuse = header ? classifyPrefixReuse(header) : null; - const { contextSize, kvBudgetBytes } = await resolveEffectiveContextSize( - binPath, - device, - model, - { - configured: opts.contextSize ?? 0, - hasMmproj: Boolean(opts.mmprojFile), - kvLayout, - }, - ); + const { + contextSize, + kvBudgetBytes, + deviceBudgetBytes, + note: contextNote, + } = await resolveEffectiveContextSize(device, model, { + configured: opts.contextSize ?? 0, + hasMmproj: Boolean(opts.mmprojFile), + kvLayout, + listDevices, + }); const swaFull = opts.swaFullFlag !== undefined ? { @@ -751,6 +953,29 @@ export async function startDaemon( : swaFull.enabled && !prefixReuse.hybrid ? "partial" : prefixReuse.prefixReuse; + // Whether the weights and this context's cache fit the device whole, or + // llama.cpp's `-fit` will leave layers on the CPU — another speed. + // Unknown (`null`) with no free figure to weigh them against. + const fitsDevice = + deviceBudgetBytes === null + ? null + : estimateLaunchKvBytes({ + kvLayout: kvLayout ? buildKvLayout(kvLayout) : null, + modelSizeGb: model.fileSizeGb, + contextSize, + swaFull: swaFull.enabled, + }) <= deviceBudgetBytes; + // The speed an earlier start measured on this same launch, read before + // this one clears the record: carried over, it spares the probe + // (`readReusableThroughput`). + const speedBasis = throughputBasis(opts.dataDir, model.id, device, { + contextSize, + fitsDevice, + }); + const knownSpeed = + opts.throughputProbe === false + ? null + : readReusableThroughput(opts.dataDir, speedBasis); const completionMaxTokens = opts.completionMaxTokens ?? readConfiguredCompletionMaxTokens(); const apiKey = @@ -787,7 +1012,8 @@ export async function startDaemon( ? ` (${header.architecture}, ${header.blockCount ?? "?"} layers, trained context ${header.contextLength ?? "?"})` : " (header unreadable)"), `[atomic-agent] launch: --ctx-size ${contextSize || "(llama.cpp default)"}` + - (kvLayout ? " fitted from the model's KV layout" : " fitted from the file-size fallback"), + (kvLayout ? " fitted from the model's KV layout" : " fitted from the file-size fallback") + + (contextNote ? `, ${contextNote}` : ""), `[atomic-agent] launch: ${swaFull.reason}`, `[atomic-agent] launch: prefix reuse ${effectivePrefixReuse ?? "unknown"}${reuseWhy}`, "", @@ -822,25 +1048,40 @@ export async function startDaemon( // A stale record must never outlive the daemon it described: drop // it before the probe so a skipped or failed probe leaves nothing // behind that `readThroughputRecord` could mistake (it also checks - // the pid, but the file is cheap to clear). + // the pid, but the file is cheap to clear). A speed carried over was + // read above, and is stamped for this daemon instead of measured. try { unlinkSync(resolveThroughputFilePath(opts.dataDir)); } catch { /* none recorded */ } let tokensPerSecond: number | null = null; - if (opts.throughputProbe !== false) { + if (knownSpeed) { + tokensPerSecond = knownSpeed.tokensPerSecond; + writeThroughputRecord(opts.dataDir, { ...knownSpeed, pid: child.pid }); + } else if (opts.throughputProbe !== false) { + // On the only slot nothing can decode beside the probe. With more, a + // turn already running when it starts, or still running when it + // ends, shared the GPU with it: such a figure is not carried over. + const oneSlot = args[args.indexOf("--parallel") + 1] === "1"; + const idleBefore = + oneSlot || (await slotsAllIdle({ port: opts.port, apiKey })); const sample = await probeThroughput({ port: opts.port, apiKey, }); if (sample) { tokensPerSecond = sample.tokensPerSecond; + const alone = + oneSlot || + (idleBefore && (await slotsAllIdle({ port: opts.port, apiKey }))); writeThroughputRecord(opts.dataDir, { ...sample, pid: child.pid, modelId: model.id, measuredAt: Date.now(), + measuredOn: speedBasis, + alone, }); } } diff --git a/src/local-llm/ensure-latest-backend.test.ts b/src/local-llm/ensure-latest-backend.test.ts index 81400373c..2ecc50fc2 100644 --- a/src/local-llm/ensure-latest-backend.test.ts +++ b/src/local-llm/ensure-latest-backend.test.ts @@ -1,4 +1,8 @@ -import { afterEach, describe, expect, it, vi } from "vitest"; +import { existsSync, mkdtempSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; + +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; vi.mock("./backend-installer.js", async () => { const actual = await vi.importActual( @@ -42,8 +46,16 @@ import { readRunningPid, stopChatAndEmbeddingDaemons, } from "./daemon-lifecycle.js"; -import { maybeAutoUpdateBackend } from "./ensure-latest-backend.js"; +import { resolveBackendCheckFilePath } from "./backend-paths.js"; +import { writeBackendVersion } from "./backend-version.js"; +import { + AUTO_UPDATE_RECHECK_MS, + AUTO_UPDATE_RETRY_MS, + checkForBackendUpdateForPanel, + maybeAutoUpdateBackend, +} from "./ensure-latest-backend.js"; import { hasOtherLiveSessions } from "./session-registry.js"; +import { resolveDownloadAsset } from "./windows-backend-variant.js"; describe("maybeAutoUpdateBackend", () => { afterEach(() => { @@ -211,3 +223,156 @@ describe("maybeAutoUpdateBackend", () => { expect(downloadBackend).not.toHaveBeenCalled(); }); }); + +/** + * Backlog 39: every switch to the local model in the desktop runs a fresh + * `models start`, so the process-wide release cache never helped and each + * start asked GitHub (up to 5 s) before the model began to load. The + * start paths pass `recheckAfterMs`: a recent answer for the build on + * disk stands, whichever process got it. + */ +describe("maybeAutoUpdateBackend with recheckAfterMs (backlog 39)", () => { + const INSTALLED = "turboquant-6df272c"; + let dataDir: string; + let clock: number; + const now = () => clock; + const start = () => + maybeAutoUpdateBackend(dataDir, { + enabled: true, + recheckAfterMs: AUTO_UPDATE_RECHECK_MS, + now, + }); + const install = (tag: string, asset = resolveDownloadAsset().assetName) => + writeBackendVersion(dataDir, { + tag, + downloadedAt: new Date(clock).toISOString(), + asset, + }); + + beforeEach(() => { + dataDir = mkdtempSync(join(tmpdir(), "atomic-auto-update-")); + clock = 1_790_909_708_032; + install(INSTALLED); + vi.mocked(checkForBackendUpdate).mockReset(); + vi.mocked(downloadBackend).mockReset(); + vi.mocked(readRunningPid).mockReset(); + vi.mocked(readRunningPid).mockReturnValue(null); + vi.mocked(hasOtherLiveSessions).mockReturnValue(false); + vi.mocked(isBackendDownloaded).mockReturnValue(true); + }); + + afterEach(() => { + rmSync(dataDir, { recursive: true, force: true }); + }); + + function nothingNewer(): void { + vi.mocked(checkForBackendUpdate).mockResolvedValue({ + updateAvailable: false, + latestTag: INSTALLED, + currentTag: INSTALLED, + }); + } + + it("asks GitHub once, then trusts that answer for six hours", async () => { + nothingNewer(); + expect(await start()).toEqual({ action: "current", tag: INSTALLED }); + const checkedAt = clock; + clock += 60_000; + expect(await start()).toEqual({ action: "recent", tag: INSTALLED, checkedAt }); + clock = checkedAt + AUTO_UPDATE_RECHECK_MS - 1; + expect((await start()).action).toBe("recent"); + expect(checkForBackendUpdate).toHaveBeenCalledTimes(1); + clock = checkedAt + AUTO_UPDATE_RECHECK_MS; + expect(await start()).toEqual({ action: "current", tag: INSTALLED }); + expect(checkForBackendUpdate).toHaveBeenCalledTimes(2); + }); + + it("asks again once another build is on disk", async () => { + nothingNewer(); + await start(); + install("turboquant-7a1c0de"); + await start(); + expect(checkForBackendUpdate).toHaveBeenCalledTimes(2); + }); + + it("always asks while the machine wants another variant than the one installed", async () => { + // The installed asset is not what this machine resolves (on Windows: + // an NVIDIA driver installed since the Vulkan build) — an update in + // itself, however recent the last check. + install(INSTALLED, "llama-turboquant-some-other-variant.zip"); + nothingNewer(); + await start(); + clock += 60_000; + await start(); + expect(checkForBackendUpdate).toHaveBeenCalledTimes(2); + }); + + it("holds off a failed check for fifteen minutes, not six hours", async () => { + vi.mocked(checkForBackendUpdate).mockRejectedValue(new Error("fetch failed")); + expect(await start()).toEqual({ action: "check_failed", error: "fetch failed" }); + const failedAt = clock; + clock += AUTO_UPDATE_RETRY_MS - 1; + expect(await start()).toEqual({ action: "recent", tag: null, checkedAt: failedAt }); + clock = failedAt + AUTO_UPDATE_RETRY_MS; + expect((await start()).action).toBe("check_failed"); + expect(checkForBackendUpdate).toHaveBeenCalledTimes(2); + }); + + it("records the update it installed, so the next start does not ask", async () => { + vi.mocked(checkForBackendUpdate).mockResolvedValue({ + updateAvailable: true, + latestTag: "turboquant-7a1c0de", + currentTag: INSTALLED, + }); + vi.mocked(downloadBackend).mockImplementation(async () => { + install("turboquant-7a1c0de"); + return { ok: true, tag: "turboquant-7a1c0de" }; + }); + expect((await start()).action).toBe("updated"); + clock += 60_000; + expect((await start()).action).toBe("recent"); + expect(checkForBackendUpdate).toHaveBeenCalledTimes(1); + }); + + it("remembers nothing and always asks without recheckAfterMs (models update, an explicit check)", async () => { + nothingNewer(); + await maybeAutoUpdateBackend(dataDir, { enabled: true, now }); + await maybeAutoUpdateBackend(dataDir, { enabled: true, now }); + expect(checkForBackendUpdate).toHaveBeenCalledTimes(2); + expect(existsSync(resolveBackendCheckFilePath(dataDir))).toBe(false); + }); + + it("does not trust a check stamped in the future (a clock set back)", async () => { + nothingNewer(); + await start(); + clock -= 60_000; + await start(); + expect(checkForBackendUpdate).toHaveBeenCalledTimes(2); + }); + + it("drops the record when the Models panel finds an update, so the next start asks and installs it", async () => { + nothingNewer(); + await start(); + expect(existsSync(resolveBackendCheckFilePath(dataDir))).toBe(true); + vi.mocked(checkForBackendUpdate).mockResolvedValueOnce({ + updateAvailable: true, + latestTag: "turboquant-7a1c0de", + currentTag: INSTALLED, + }); + expect((await checkForBackendUpdateForPanel(dataDir)).updateAvailable).toBe(true); + expect(existsSync(resolveBackendCheckFilePath(dataDir))).toBe(false); + clock += 60_000; + await start(); + // The first start, the panel, and this start: it did not answer `recent`. + expect(checkForBackendUpdate).toHaveBeenCalledTimes(3); + }); + + it("keeps the record when the Models panel finds nothing newer", async () => { + nothingNewer(); + await start(); + expect((await checkForBackendUpdateForPanel(dataDir)).updateAvailable).toBe(false); + clock += 60_000; + expect((await start()).action).toBe("recent"); + expect(checkForBackendUpdate).toHaveBeenCalledTimes(2); + }); +}); diff --git a/src/local-llm/ensure-latest-backend.ts b/src/local-llm/ensure-latest-backend.ts index cbdd7aea7..db4fde54c 100644 --- a/src/local-llm/ensure-latest-backend.ts +++ b/src/local-llm/ensure-latest-backend.ts @@ -1,18 +1,29 @@ +import { readFileSync, unlinkSync, writeFileSync } from "node:fs"; + import { checkForBackendUpdate, downloadBackend, isBackendDownloaded, } from "./backend-installer.js"; +import { resolveBackendCheckFilePath } from "./backend-paths.js"; +import { readBackendVersion } from "./backend-version.js"; import type { DownloadProgressFn } from "./download-file.js"; import { readRunningPid, stopChatAndEmbeddingDaemons, } from "./daemon-lifecycle.js"; import { hasOtherLiveSessions } from "./session-registry.js"; +import { resolveDownloadAsset } from "./windows-backend-variant.js"; export type AutoUpdateBackendResult = | { action: "skipped" } | { action: "current"; tag: string | null } + /** + * A check recent enough (`recheckAfterMs`) already answered for the + * build on disk, so nothing was asked. `tag` is the newest release it + * found — `null` when that check failed or found none for this platform. + */ + | { action: "recent"; tag: string | null; checkedAt: number } | { action: "updated"; from: string | null; to: string } | { action: "deferred"; reason: "other_session" | "daemon_live" } | { action: "check_failed"; error: string } @@ -26,6 +37,147 @@ export type AutoUpdateBackendResult = */ | { action: "update_failed"; error: string; backendUsable: boolean }; +/** + * How long a start trusts a release check that found nothing newer for + * the build on disk (`recheckAfterMs`). The releases are nightly, and the + * check is a GitHub round trip — up to its 5 s deadline — in front of the + * model's start: the desktop starts the model through a fresh `models + * start` process on every switch back to the local model, where the + * process-wide release cache never survives, so every switch asked again. + */ +export const AUTO_UPDATE_RECHECK_MS = 6 * 60 * 60 * 1000; + +/** + * How long a check that failed (offline, rate-limited, a black-holed + * connection that ran out the deadline) holds off the next one. + */ +export const AUTO_UPDATE_RETRY_MS = 15 * 60 * 1000; + +/** The last check before a start, at `resolveBackendCheckFilePath`. */ +interface BackendCheckRecord { + /** Epoch ms of the check. */ + checkedAt: number; + /** The build on disk it was made for (`backend-version.json`). */ + tag: string; + asset: string | null; + /** Whether GitHub answered. */ + ok: boolean; + /** The newest release it found for this platform. */ + latestTag: string | null; +} + +function readBackendCheck(dataDir: string): BackendCheckRecord | null { + try { + const parsed = JSON.parse( + readFileSync(resolveBackendCheckFilePath(dataDir), "utf-8"), + ) as Partial; + if ( + typeof parsed.checkedAt !== "number" || + !Number.isFinite(parsed.checkedAt) || + typeof parsed.tag !== "string" || + typeof parsed.ok !== "boolean" + ) { + return null; + } + return { + checkedAt: parsed.checkedAt, + tag: parsed.tag, + asset: typeof parsed.asset === "string" ? parsed.asset : null, + ok: parsed.ok, + latestTag: typeof parsed.latestTag === "string" ? parsed.latestTag : null, + }; + } catch { + return null; + } +} + +/** Best-effort: a record that cannot be written only means the next start asks again. */ +function noteBackendCheck( + dataDir: string, + checkedAt: number, + outcome: { ok: boolean; latestTag: string | null }, +): void { + const installed = readBackendVersion(dataDir); + if (!installed) return; + const record: BackendCheckRecord = { + checkedAt, + tag: installed.tag, + asset: installed.asset ?? null, + ...outcome, + }; + try { + writeFileSync(resolveBackendCheckFilePath(dataDir), JSON.stringify(record), "utf-8"); + } catch { + /* asked again next time */ + } +} + +/** + * Drop the last check's record, so the next start asks GitHub again. For + * a check made elsewhere that found an update: a view offering an update + * while a start trusted an older "nothing newer" would have the two + * disagree for up to `AUTO_UPDATE_RECHECK_MS`. + */ +export function forgetBackendCheck(dataDir: string): void { + try { + unlinkSync(resolveBackendCheckFilePath(dataDir)); + } catch { + /* none recorded */ + } +} + +/** + * `checkForBackendUpdate` for a view that shows whether an update is + * available — the TUI's Models panel, which asks on every refresh. When it + * finds one, the record a start trusts is dropped (`forgetBackendCheck`), + * so the next start asks too and installs it. + */ +export async function checkForBackendUpdateForPanel( + dataDir: string, +): Promise>> { + const result = await checkForBackendUpdate(dataDir); + if (result.updateAvailable) forgetBackendCheck(dataDir); + return result; +} + +/** + * The last check, when it still stands for the build on disk: made for + * this tag and asset, younger than its window (`AUTO_UPDATE_RETRY_MS` at + * most when it failed), and the machine still wants the asset that is + * installed — a variant change (an NVIDIA driver installed since) is an + * update in itself, so that always asks. + */ +function standingCheck( + dataDir: string, + now: number, + recheckAfterMs: number, +): BackendCheckRecord | null { + const last = readBackendCheck(dataDir); + const installed = readBackendVersion(dataDir); + if ( + !last || + !installed || + last.tag !== installed.tag || + last.asset !== (installed.asset ?? null) + ) { + return null; + } + if (installed.asset !== undefined) { + let wanted: string; + try { + wanted = resolveDownloadAsset().assetName; + } catch { + return null; + } + if (wanted !== installed.asset) return null; + } + const window = last.ok + ? recheckAfterMs + : Math.min(recheckAfterMs, AUTO_UPDATE_RETRY_MS); + const age = now - last.checkedAt; + return age >= 0 && age < window ? last : null; +} + /** * When `enabled`, pull a newer llama.cpp backend from GitHub Releases * before the managed daemon starts. Missing-backend first install is @@ -57,20 +209,46 @@ export async function maybeAutoUpdateBackend( * background update would kill the model mid-turn. */ keepDaemonRunning?: boolean; + /** + * Trust a check this recent for the build on disk, from any process + * (`AUTO_UPDATE_RECHECK_MS` for the start paths): answer `recent` + * without asking GitHub, and remember what this check found. Absent + * or `0`: always ask, remember nothing — `models update` and anything + * else that means "check now". + */ + recheckAfterMs?: number; + /** Clock for the check record; tests pass one. */ + now?: () => number; }, ): Promise { if (!opts.enabled) return { action: "skipped" }; + const now = opts.now ?? Date.now; + const remember = (opts.recheckAfterMs ?? 0) > 0; + if (remember) { + const standing = standingCheck(dataDir, now(), opts.recheckAfterMs ?? 0); + if (standing) { + return { + action: "recent", + tag: standing.latestTag, + checkedAt: standing.checkedAt, + }; + } + } let check: Awaited>; try { check = await checkForBackendUpdate(dataDir); } catch (err) { + if (remember) noteBackendCheck(dataDir, now(), { ok: false, latestTag: null }); return { action: "check_failed", error: err instanceof Error ? err.message : String(err), }; } if (!check.updateAvailable) { + if (remember) { + noteBackendCheck(dataDir, now(), { ok: true, latestTag: check.latestTag }); + } return { action: "current", tag: check.latestTag }; } @@ -95,6 +273,9 @@ export async function maybeAutoUpdateBackend( onProgress: opts.onProgress, signal: opts.signal, }); + if (remember) { + noteBackendCheck(dataDir, now(), { ok: true, latestTag: downloaded.tag }); + } return { action: "updated", from: check.currentTag, diff --git a/src/local-llm/gpu-devices.test.ts b/src/local-llm/gpu-devices.test.ts index 014f21e98..fa7c21fbc 100644 --- a/src/local-llm/gpu-devices.test.ts +++ b/src/local-llm/gpu-devices.test.ts @@ -1,13 +1,15 @@ -import { mkdtempSync, rmSync, writeFileSync } from "node:fs"; +import { mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; import { join } from "node:path"; import { describe, expect, it } from "vitest"; import { + deviceTableOnce, parseListDevices, pickBestDevice, resolveManagedDevice, + sharesSystemMemory, type GpuDevice, } from "./gpu-devices.js"; @@ -313,3 +315,128 @@ describe("resolveManagedDevice", () => { ); }); }); + +describe("sharesSystemMemory", () => { + const device = (id: string, description: string): GpuDevice => ({ + id, + description, + totalMemMiB: 10_922, + freeMemMiB: 10_922, + }); + + it("is true for Apple silicon's Metal device and for an integrated GPU", () => { + expect(sharesSystemMemory(device("MTL0", "Apple M4"))).toBe(true); + expect(sharesSystemMemory(device("Metal0", "Apple M1 Max"))).toBe(true); + expect(sharesSystemMemory(device("Vulkan1", "Intel(R) Graphics (RPL-S)"))).toBe(true); + expect(sharesSystemMemory(device("Vulkan0", "AMD Radeon(TM) Graphics"))).toBe(true); + }); + + it("is true for unified-memory parts named like cards", () => { + // Intel Meteor Lake and Lunar Lake: their iGPUs are called Arc. + expect(sharesSystemMemory(device("Vulkan0", "Intel(R) Arc(TM) Graphics"))).toBe(true); + expect(sharesSystemMemory(device("Vulkan0", "Intel(R) Arc(TM) 140V GPU (16GB)"))).toBe(true); + expect(sharesSystemMemory(device("Vulkan0", "Intel(R) Arc(TM) 130V GPU"))).toBe(true); + // AMD Strix Halo. + expect(sharesSystemMemory(device("Vulkan0", "AMD Radeon(TM) 8060S Graphics"))).toBe(true); + expect(sharesSystemMemory(device("Vulkan0", "AMD Radeon(TM) 8050S Graphics"))).toBe(true); + // NVIDIA GB10 (DGX Spark) and the Jetson modules. + expect(sharesSystemMemory(device("CUDA0", "NVIDIA GB10"))).toBe(true); + expect(sharesSystemMemory(device("CUDA0", "Orin"))).toBe(true); + expect(sharesSystemMemory(device("CUDA0", "NVIDIA Jetson AGX Orin"))).toBe(true); + expect(sharesSystemMemory(device("CUDA0", "NVIDIA Thor"))).toBe(true); + }); + + it("is false for a card with memory of its own", () => { + expect(sharesSystemMemory(device("CUDA0", "NVIDIA GeForce RTX 4090"))).toBe(false); + expect(sharesSystemMemory(device("Vulkan0", "AMD Radeon RX 7900 XTX"))).toBe(false); + expect(sharesSystemMemory(device("Vulkan0", "AMD Radeon PRO W7900"))).toBe(false); + expect(sharesSystemMemory(device("Vulkan0", "Intel(R) Arc(TM) A770 Graphics"))).toBe(false); + expect(sharesSystemMemory(device("Vulkan0", "Intel(R) Arc(TM) B580 Graphics"))).toBe(false); + }); +}); + +describe("deviceTableOnce (backlog 39)", () => { + it.skipIf(process.platform === "win32")( + "runs --list-devices once for a launch: the device pick and the context fit share the answer", + async () => { + const dir = mkdtempSync(join(tmpdir(), "gpu-devices-once-")); + const bin = join(dir, "llama-server"); + const runs = join(dir, "runs"); + writeFileSync(runs, ""); + writeFileSync( + bin, + [ + "#!/bin/sh", + `echo run >> '${runs}'`, + 'echo "Available devices:"', + 'echo " MTL0: Apple M4 (10922 MiB, 10922 MiB free)"', + ].join("\n"), + { mode: 0o755 }, + ); + try { + const table = deviceTableOnce(bin); + expect(await resolveManagedDevice(bin, "auto", { listDevices: table })).toBe("MTL0"); + const again = await table(); + expect(again[0]).toMatchObject({ id: "MTL0", freeMemMiB: 10_922 }); + expect(readFileSync(runs, "utf-8").split("\n").filter(Boolean)).toHaveLength(1); + // A table nobody asks never runs the binary. + deviceTableOnce(bin); + expect(readFileSync(runs, "utf-8").split("\n").filter(Boolean)).toHaveLength(1); + } finally { + rmSync(dir, { recursive: true, force: true }); + } + }, + ); + + it.skipIf(process.platform === "win32")( + "asks once more when the first answer is empty (a cold start that ran out its deadline), and keeps that answer", + async () => { + const dir = mkdtempSync(join(tmpdir(), "gpu-devices-cold-")); + const bin = join(dir, "llama-server"); + const runs = join(dir, "runs"); + const warm = join(dir, "warm"); + writeFileSync(runs, ""); + // The first run says nothing, as one killed at its 5 s deadline does. + writeFileSync( + bin, + [ + "#!/bin/sh", + `echo run >> '${runs}'`, + `[ -f '${warm}' ] || { : > '${warm}'; exit 0; }`, + 'echo "Available devices:"', + 'echo " MTL0: Apple M4 (10922 MiB, 10922 MiB free)"', + ].join("\n"), + { mode: 0o755 }, + ); + try { + const table = deviceTableOnce(bin); + expect(await resolveManagedDevice(bin, "auto", { listDevices: table })).toBe("MTL0"); + expect((await table())[0]?.id).toBe("MTL0"); + expect(readFileSync(runs, "utf-8").split("\n").filter(Boolean)).toHaveLength(2); + } finally { + rmSync(dir, { recursive: true, force: true }); + } + }, + ); + + it.skipIf(process.platform === "win32")( + "asks at most twice when there is no GPU to report", + async () => { + const dir = mkdtempSync(join(tmpdir(), "gpu-devices-none-")); + const bin = join(dir, "llama-server"); + const runs = join(dir, "runs"); + writeFileSync(runs, ""); + writeFileSync(bin, ["#!/bin/sh", `echo run >> '${runs}'`, 'echo "Available devices:"'].join("\n"), { + mode: 0o755, + }); + try { + const table = deviceTableOnce(bin); + expect(await resolveManagedDevice(bin, "auto", { listDevices: table })).toBeUndefined(); + expect(await table()).toEqual([]); + expect(readFileSync(runs, "utf-8").split("\n").filter(Boolean)).toHaveLength(2); + } finally { + rmSync(dir, { recursive: true, force: true }); + } + }, + ); +}); diff --git a/src/local-llm/gpu-devices.ts b/src/local-llm/gpu-devices.ts index 440b64f45..47864d058 100644 --- a/src/local-llm/gpu-devices.ts +++ b/src/local-llm/gpu-devices.ts @@ -142,6 +142,33 @@ export function deviceClassRank(description: string): 0 | 1 | 2 { return 1; } +/** + * GPUs that answer to the discrete patterns (or to none) yet have no + * memory of their own: Apple silicon; Intel's Meteor Lake and Lunar Lake + * iGPUs, which are named Arc ("Intel(R) Arc(TM) Graphics", "Arc 140V" / + * "Arc 130V", unlike the A- and B-series cards); AMD Strix Halo + * ("Radeon(TM) 8060S / 8050S Graphics"); and NVIDIA's GB10 (DGX Spark) and + * Jetson Orin / Thor modules. + */ +const UNIFIED_MEMORY_DEVICE_RE = + /\bapple\b|\barc(\(tm\))?\s*(graphics|1[34]0v)\b|radeon(\(tm\))?\s*\d{4}s\b|\bgb10\b|\bjetson\b|\borin\b|\bthor\b/i; + +/** + * Whether the device's memory is the system's own RAM: Apple silicon's + * Metal device (`MTL0: Apple M4 …`), an integrated GPU, or a unified- + * memory part named like a card (`UNIFIED_MEMORY_DEVICE_RE`). The free + * figure such a device reports is a ceiling on what the GPU may map, not + * memory nobody else is using, so the context auto-size also leaves the + * system its headroom there (`context-size.ts`). Pure — no IO. + */ +export function sharesSystemMemory(device: GpuDevice): boolean { + return ( + /^(MTL|Metal)\d+$/i.test(device.id) || + UNIFIED_MEMORY_DEVICE_RE.test(device.description) || + deviceClassRank(device.description) === 0 + ); +} + /** * Pick the best single device id for offloading, or `null` when there * is no usable GPU. Heuristic: drop software rasterizers, prefer a @@ -188,6 +215,37 @@ export async function listVulkanDevices(binPath: string): Promise { } } +/** A launch's device table, read on demand (`deviceTableOnce`). */ +export type ListDevices = () => Promise; + +/** + * ` --list-devices` for one launch: run the first time something + * asks, and that answer handed to everyone after. A managed start reads + * the table twice — to pick the device, then for the free memory the + * context is fitted into — and each run starts the backend (on Apple + * silicon, Metal's device and its shader library), so the second spawn + * was pure delay before the model began to load. + * + * An empty answer is not kept: it is what a run that ran out its 5 s + * deadline leaves, and the first start after a llama.cpp install is slow + * to start the backend at all (16 s before the server printed its first + * line, on a 16 GB Mac). It is asked once more at once, and that answer + * stands either way — a machine with no GPU says nothing twice. + */ +export function deviceTableOnce(binPath: string): ListDevices { + let table: Promise | null = null; + let asked = 0; + const ask = (): Promise => { + asked += 1; + table = listVulkanDevices(binPath); + return table; + }; + return async () => { + const devices = await (table ?? ask()); + return devices.length === 0 && asked < 2 ? ask() : devices; + }; +} + /** * Resolve the configured device preference into a concrete value for the * daemon argv builder: @@ -207,16 +265,18 @@ export async function listVulkanDevices(binPath: string): Promise { * split. * * Best-effort and never throws — enumeration failures fall through to - * `undefined`. + * `undefined`. `opts.listDevices` is the launch's own table + * (`deviceTableOnce`), so the context fit after it does not enumerate + * again; without it the binary is asked here. */ export async function resolveManagedDevice( binPath: string, configured: string | undefined, - opts?: { multiGpu?: boolean }, + opts?: { multiGpu?: boolean; listDevices?: ListDevices }, ): Promise { if (configured === "cpu") return "cpu"; if (configured && configured !== "auto") return configured; if (opts?.multiGpu) return undefined; - const devices = await listVulkanDevices(binPath); + const devices = await (opts?.listDevices ?? (() => listVulkanDevices(binPath)))(); return pickBestDevice(devices) ?? undefined; } diff --git a/src/local-llm/index.ts b/src/local-llm/index.ts index 6790cc6e4..32c2fc8b2 100644 --- a/src/local-llm/index.ts +++ b/src/local-llm/index.ts @@ -44,6 +44,7 @@ export { resolveModelFilePath, resolveMmprojFilePath, resolveVersionFilePath, + resolveBackendCheckFilePath, resolvePidFilePath, resolveLogFilePath, resolveThroughputFilePath, @@ -148,6 +149,10 @@ export { type LatestReleaseInfo, } from "./backend-installer.js"; export { + AUTO_UPDATE_RECHECK_MS, + AUTO_UPDATE_RETRY_MS, + checkForBackendUpdateForPanel, + forgetBackendCheck, maybeAutoUpdateBackend, type AutoUpdateBackendResult, } from "./ensure-latest-backend.js"; @@ -167,8 +172,11 @@ export { parseListDevices, pickBestDevice, listVulkanDevices, + deviceTableOnce, resolveManagedDevice, + sharesSystemMemory, type GpuDevice, + type ListDevices, } from "./gpu-devices.js"; export { resolveGpuBudgetGb, @@ -194,6 +202,10 @@ export { stopChatAndEmbeddingDaemons, probeThroughput, readThroughputRecord, + readReusableThroughput, + slotsAllIdle, + throughputBasis, + THROUGHPUT_REUSE_MAX_AGE_MS, writeThroughputRecord, readLaunchRecord, writeLaunchRecord, diff --git a/src/runtime/bootstrap-turn-status.test.ts b/src/runtime/bootstrap-turn-status.test.ts new file mode 100644 index 000000000..52d1d6ab9 --- /dev/null +++ b/src/runtime/bootstrap-turn-status.test.ts @@ -0,0 +1,423 @@ +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { spawnSync } from "node:child_process"; +import { mkdirSync, mkdtempSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; + +import { createAgentRuntime, type AgentRuntime } from "./bootstrap.js"; +import { resetConfigCache } from "../config/index.js"; +import { + INTERRUPTED_TURN_ENDING, + SessionStore, + createEmptySessionState, + hostIdentity, + hostUptime, + type ConversationTurn, +} from "../session/index.js"; +import type { AgentLoopEvent } from "../agent/agent-loop.js"; +import type { CompletionResult } from "../llm/llama-server-client.js"; +import type { BrowserBackend } from "../tools/browser/browser-backend.js"; +import type { LogRecord } from "../tracing/structured-logger.js"; + +/** + * Every way a turn ends, as the session row records it. + * + * The row used to hold a turn only once the turn wrote its end. A turn + * that never got that far — the app quit while it ran, the process was + * killed — left the row as it was before the turn; a cancelled first + * message in a new chat read back as `pending` with nothing in it. Now + * the row says `running` while the turn runs, and its end, however it + * comes, replaces that. + */ + +function inertBackend(): BrowserBackend { + return { + ensureReady: async () => undefined, + shutdown: async () => undefined, + } as unknown as BrowserBackend; +} + +function reply(text: string): CompletionResult { + return { + content: JSON.stringify({ tool: "reply", args: { text } }), + reasoningContent: "", + stop: true, + truncated: false, + timing: { promptMs: 1, predictedMs: 1, promptTokens: 10, predictedTokens: 5 }, + cacheHitTokens: 0, + slotId: 0, + modelId: "mock", + }; +} + +/** + * A model that holds its answer for the session under test until + * `release()`, or until the request's signal aborts — and then rejects + * `unwindMs` later, the way an aborted request takes a moment to come + * back. Side calls on other ids (session naming, reflection) answer at + * once. + */ +function heldModel(unwindMs = 0) { + const state = { target: "", entered: 0 }; + let release: () => void = () => undefined; + const llamaComplete = async (params: { + sessionId: string; + signal?: AbortSignal; + }): Promise => { + if (params.sessionId !== state.target) return reply("ok"); + state.entered += 1; + const signal = params.signal; + const outcome = await new Promise<"released" | "aborted">((resolve) => { + release = () => resolve("released"); + if (signal?.aborted) { + resolve("aborted"); + return; + } + signal?.addEventListener("abort", () => resolve("aborted"), { + once: true, + }); + }); + if (outcome === "aborted") { + if (unwindMs > 0) { + await new Promise((resolve) => setTimeout(resolve, unwindMs)); + } + throw signal?.reason ?? new DOMException("aborted", "AbortError"); + } + return reply("done"); + }; + return { state, llamaComplete, release: () => release() }; +} + +async function waitFor(check: () => boolean, ms = 5_000): Promise { + const deadline = Date.now() + ms; + while (!check()) { + if (Date.now() > deadline) return false; + await new Promise((resolve) => setTimeout(resolve, 10)); + } + return true; +} + +function userTexts(turns: readonly ConversationTurn[]): string[] { + return turns.flatMap((turn) => (turn.kind === "user" ? [turn.text] : [])); +} + +describe("a turn's status in the session store", () => { + let stateDir: string; + let workingDir: string; + let dbFile: string; + + beforeEach(() => { + stateDir = mkdtempSync(join(tmpdir(), "atomic-runtime-turn-status-")); + workingDir = mkdtempSync(join(tmpdir(), "atomic-cwd-turn-status-")); + dbFile = join(stateDir, "sessions.sqlite"); + mkdirSync(join(workingDir, ".atomic-agent", "skills"), { recursive: true }); + process.env.ATOMIC_AGENT_STATE_DIR = stateDir; + process.env.ATOMIC_AGENT_GRAMMARS_DIR = join(process.cwd(), "grammars"); + resetConfigCache(); + }); + + afterEach(() => { + rmSync(stateDir, { recursive: true, force: true }); + rmSync(workingDir, { recursive: true, force: true }); + delete process.env.ATOMIC_AGENT_STATE_DIR; + delete process.env.ATOMIC_AGENT_GRAMMARS_DIR; + resetConfigCache(); + }); + + async function boot( + llamaComplete: (params: { + sessionId: string; + signal?: AbortSignal; + }) => Promise, + handlers: { + onAgentEvent?: (event: AgentLoopEvent) => void; + logs?: LogRecord[]; + } = {}, + ): Promise { + const logs = handlers.logs; + return createAgentRuntime({ + workingDir, + approvalLevel: 5, + handlers: { + ...(handlers.onAgentEvent ? { onAgentEvent: handlers.onAgentEvent } : {}), + ...(logs ? { logSinks: [(record: LogRecord) => logs.push(record)] } : {}), + }, + overrides: { + browserBackend: inertBackend(), + skipLlamaHealthCheck: true, + llamaComplete, + }, + }); + } + + function turnOwner(runtime: AgentRuntime, id: string): string | null | undefined { + const row = runtime.sessionStore + .getDatabaseHandleForRetention() + .prepare(`SELECT turn_owner AS turnOwner FROM sessions WHERE id = ?`) + .get(id) as { turnOwner: string | null } | undefined; + return row?.turnOwner; + } + + /** The row as the next process to open the file would find it. */ + function readBack(read: (store: SessionStore) => T): T { + const store = new SessionStore({ dbFile }); + try { + return read(store); + } finally { + store.close(); + } + } + + it("says running for as long as the turn runs, and the turn's end replaces it", async () => { + const model = heldModel(); + const runtime = await boot(model.llamaComplete); + try { + const session = runtime.createSession(); + model.state.target = session.id; + const turn = runtime.runTurn(session, "hello", { + origin: "tui", + maxSteps: 4, + }); + expect(await waitFor(() => model.state.entered === 1)).toBe(true); + expect(runtime.sessionStore.load(session.id)?.status).toBe("running"); + expect(turnOwner(runtime, session.id)).toContain(`"pid":${process.pid}`); + + model.release(); + const result = await turn; + expect(result.reason).toBe("reply"); + expect(runtime.sessionStore.load(session.id)?.status).toBe("pending"); + expect(turnOwner(runtime, session.id)).toBeNull(); + } finally { + await runtime.shutdown(); + } + }); + + it("stores a stopped turn as cancelled, with the message it was sent", async () => { + const model = heldModel(20); + const runtime = await boot(model.llamaComplete); + try { + const session = runtime.createSession(); + model.state.target = session.id; + const controller = new AbortController(); + const turn = runtime.runTurn(session, "stop me", { + origin: "tui", + maxSteps: 4, + signal: controller.signal, + }); + expect(await waitFor(() => model.state.entered === 1)).toBe(true); + controller.abort(); + const result = await turn; + expect(result.reason).toBe("cancelled"); + const stored = runtime.sessionStore.load(session.id); + expect(stored?.status).toBe("cancelled"); + expect(userTexts(stored?.turns ?? [])).toEqual(["stop me"]); + expect(stored?.turnCount).toBe(1); + expect(turnOwner(runtime, session.id)).toBeNull(); + } finally { + await runtime.shutdown(); + } + }); + + it("does not leave a turn that threw marked running", async () => { + let armed = false; + const model = heldModel(); + const runtime = await boot(model.llamaComplete, { + onAgentEvent: (event) => { + if (armed && event.type === "turn_started") { + throw new Error("host hook blew up"); + } + }, + }); + try { + const session = runtime.createSession(); + model.state.target = session.id; + armed = true; + await expect( + runtime.runTurn(session, "boom", { origin: "tui", maxSteps: 4 }), + ).rejects.toThrow(/host hook blew up/); + armed = false; + const stored = runtime.sessionStore.load(session.id); + expect(stored?.status).toBe("failed"); + expect(stored?.lastError).toBe("host hook blew up"); + expect(turnOwner(runtime, session.id)).toBeNull(); + } finally { + await runtime.shutdown(); + } + }); + + it("stores a turn that threw after it was stopped as cancelled", async () => { + const controller = new AbortController(); + let armed = false; + const model = heldModel(); + const runtime = await boot(model.llamaComplete, { + onAgentEvent: (event) => { + if (armed && event.type === "turn_started") { + controller.abort(); + throw new Error("hook failed while stopping"); + } + }, + }); + try { + const session = runtime.createSession(); + model.state.target = session.id; + armed = true; + await expect( + runtime.runTurn(session, "boom", { + origin: "tui", + maxSteps: 4, + signal: controller.signal, + }), + ).rejects.toThrow(/hook failed while stopping/); + armed = false; + const stored = runtime.sessionStore.load(session.id); + expect(stored?.status).toBe("cancelled"); + expect(stored?.lastError).toBeNull(); + expect(turnOwner(runtime, session.id)).toBeNull(); + } finally { + await runtime.shutdown(); + } + }); + + it("lets a turn its host stopped write its own end before shutdown closes the store", async () => { + // What every host does on quit: stop its turn, then shut the runtime + // down. The aborted request takes a moment to come back; the store + // used to be closed by then, and the row kept its pre-turn state. + const model = heldModel(50); + const runtime = await boot(model.llamaComplete); + const session = runtime.createSession(); + model.state.target = session.id; + const controller = new AbortController(); + const turn = runtime.runTurn(session, "quit mid-turn", { + origin: "tui", + maxSteps: 4, + signal: controller.signal, + }); + expect(await waitFor(() => model.state.entered === 1)).toBe(true); + controller.abort(); + await runtime.shutdown(); + const result = await turn; + expect(result.reason).toBe("cancelled"); + + const stored = readBack((store) => store.load(session.id)); + expect(stored?.status).toBe("cancelled"); + expect(userTexts(stored?.turns ?? [])).toEqual(["quit mid-turn"]); + // The turn's own end, not shutdown's stand-in for it. + expect(stored?.lastError).toBeNull(); + }); + + it("lets a stopped turn that throws during shutdown replace the interrupted stand-in", async () => { + // Shutdown writes "interrupted" first thing, as a stand-in. The turn + // then ends by throwing (here: a host hook failing as the loop closes + // it), so it has no state to save — but it was stopped, and that is + // what its row says, not the stand-in. + let armed = false; + const model = heldModel(50); + const runtime = await boot(model.llamaComplete, { + onAgentEvent: (event) => { + if (armed && event.type === "loop_completed") { + throw new Error("hook failed as the turn closed"); + } + }, + }); + const session = runtime.createSession(); + model.state.target = session.id; + const controller = new AbortController(); + armed = true; + const turn = runtime + .runTurn(session, "quit mid-turn", { + origin: "tui", + maxSteps: 4, + signal: controller.signal, + }) + .catch((err: unknown) => err); + expect(await waitFor(() => model.state.entered === 1)).toBe(true); + controller.abort(); + await runtime.shutdown(); + expect(await turn).toBeInstanceOf(Error); + + const stored = readBack((store) => store.load(session.id)); + expect(stored?.status).toBe("cancelled"); + expect(stored?.lastError).toBeNull(); + }); + + it("records a turn nobody stopped as interrupted at shutdown", async () => { + // Nothing stops a scheduled task's turn when the app quits, and + // shutdown does not wait for it: its model never answers here. + const model = heldModel(); + const runtime = await boot(model.llamaComplete); + const session = runtime.createSession(); + model.state.target = session.id; + const controller = new AbortController(); + const turn = runtime + .runTurn(session, "background work", { + origin: "scheduler", + maxSteps: 4, + signal: controller.signal, + }) + .catch(() => undefined); + expect(await waitFor(() => model.state.entered === 1)).toBe(true); + expect(runtime.sessionStore.load(session.id)?.status).toBe("running"); + await runtime.shutdown(); + + const stored = readBack((store) => store.load(session.id)); + expect(stored?.status).toBe(INTERRUPTED_TURN_ENDING.status); + expect(stored?.lastError).toBe(INTERRUPTED_TURN_ENDING.lastError); + // Its end never came, so the transcript is the one before it. + expect(stored?.turns).toEqual([]); + + // Let the abandoned turn unwind, so the test leaves nothing running. + controller.abort(); + await turn; + }); + + it("ends, at boot, a turn left running by a process that is gone, and leaves a live one", async () => { + // A process that has exited; its pid is free. + const deadPid = spawnSync(process.execPath, ["-e", ""]).pid; + const seedRunning = (id: string, pid: number): void => { + const store = new SessionStore({ + dbFile, + turnOwnerProbe: { + pid, + host: hostIdentity(), + hostUptime, + isAlive: () => true, + processStartOf: () => null, + }, + }); + try { + store.save(createEmptySessionState({ id, workingDir })); + expect(store.beginTurn(id)).toBe(true); + } finally { + // Closing is not ending the turn: that process died mid-turn. + store.close(); + } + }; + seedRunning("s-dead-owner", deadPid); + // The process that started this test's runner: alive throughout. + seedRunning("s-live-owner", process.ppid); + + const logs: LogRecord[] = []; + const model = heldModel(); + const runtime = await boot(model.llamaComplete, { logs }); + try { + const dead = runtime.sessionStore.load("s-dead-owner"); + expect(dead?.status).toBe("cancelled"); + expect(dead?.lastError).toBe(INTERRUPTED_TURN_ENDING.lastError); + expect(turnOwner(runtime, "s-dead-owner")).toBeNull(); + expect(runtime.sessionStore.load("s-live-owner")?.status).toBe("running"); + expect( + logs.some( + (record) => + record.message === + "sessions left mid-turn by a stopped agent marked cancelled", + ), + ).toBe(true); + } finally { + await runtime.shutdown(); + } + // Shutting down does not touch a turn another process owns. + expect(readBack((store) => store.load("s-live-owner")?.status)).toBe( + "running", + ); + }); +}); diff --git a/src/runtime/bootstrap.ts b/src/runtime/bootstrap.ts index babf87d96..41eed1714 100644 --- a/src/runtime/bootstrap.ts +++ b/src/runtime/bootstrap.ts @@ -13,6 +13,7 @@ import { import type { LlmStreamParams } from "../agent/step-executor.js"; import { TurnController } from "./turn-controller.js"; import { SteeringInbox } from "./steering-inbox.js"; +import { SHUTDOWN_TURN_GRACE_MS, TurnsInFlight } from "./turns-in-flight.js"; import type { TurnEventHook, TurnOrigin } from "./turn-controller.js"; import type { ChannelStatus } from "./channel-status.js"; @@ -196,6 +197,7 @@ import type { AgentLoopEvent, RunTurnResult } from "../agent/agent-loop.js"; import { SessionStore, + INTERRUPTED_TURN_ENDING, createEmptySessionState, createFusionWorkerSession, readFusionWorkerMeta, @@ -263,6 +265,7 @@ import { installGlobalErrorHandlers, captureError, } from "../error-reporting/index.js"; +import type { BrokenPipePolicy } from "../error-reporting/index.js"; import { getAppVersion } from "../version.js"; export interface RuntimeEventHandlers { @@ -337,6 +340,14 @@ export interface CreateAgentRuntimeOptions { * day. Only the TUI passes `true`. */ interactiveLaunch?: boolean; + /** + * What this process does when the reader of its stdout or stderr goes + * away (see `BrokenPipePolicy`). Default `exit`; `serve` passes `mute`, + * so a server whose host died keeps going until its orphan watch ends + * it through the teardown, instead of exiting at its next log line. + * Process-wide and first-come: the handlers are installed once. + */ + brokenPipe?: BrokenPipePolicy; /** * Analytics `surface` of the entry point (the TUI passes `tui`). * Headless entry points leave it unset so `ATOMIC_AGENT_SURFACE` from @@ -863,7 +874,9 @@ export async function createAgentRuntime( } // Read the current reporter lazily so a hot-toggle is reflected without // re-installing the process-global handlers. - installGlobalErrorHandlers(() => errorReporter); + installGlobalErrorHandlers(() => errorReporter, { + brokenPipe: options.brokenPipe, + }); /** * Hot-toggle anonymous analytics (PostHog) and error reporting @@ -1520,6 +1533,34 @@ export async function createAgentRuntime( // column-only `listRecentWorkingDirs` projection, so the store must // exist by the time `registerOsTools` wires the closure below. const sessionStore = new SessionStore(); + if (sessionStore.turnMarksUnavailable !== null) { + // The first open after an upgrade adds the `turn_owner` column, and + // could not this time. The runtime runs without turn marks — as it + // did before they existed — and the next start tries again. + logger.warn("session turn marks unavailable for this run", { + error: sessionStore.turnMarksUnavailable, + }); + } + // A row still marked `running` by a process that is gone is a turn that + // will never write its end — the app was killed or crashed mid-turn — + // and every list would show it running for ever. End those before + // anything reads the table (the retention pass below included: it + // never prunes a live row, so a ghost would also be kept for ever). + // Rows a live process still owns — a second window, a `serve` beside a + // TUI — are left alone. Never blocks boot. + try { + const recovered = sessionStore.recoverInterruptedTurns(); + if (recovered.length > 0) { + logger.info("sessions left mid-turn by a stopped agent marked cancelled", { + count: recovered.length, + sessionIds: recovered.join(","), + }); + } + } catch (err) { + logger.warn("could not end sessions left mid-turn; continuing", { + error: err instanceof Error ? err.message : String(err), + }); + } // `sessions.sqlite` and the traces beside it are the only state this // runtime never shrinks (§"Session retention"). One bounded pass here, // opt-in, and wrapped so that a prune can never be the reason a @@ -2767,10 +2808,51 @@ export async function createAgentRuntime( * `shutdownCalled`'s job, checked on both sides of the completion. */ const pendingSessionNamings = new Set(); + /** + * The turns `executeTurn` is running, so `shutdown` can let the ones + * their hosts stopped write their own end before the session store + * closes (`TurnsInFlight`). + */ + const turnsInFlight = new TurnsInFlight(); + /** + * Record every turn this runtime still has marked `running` as + * interrupted. Shutdown only: by then a turn that has not written its + * end cannot be counted on to, and a row left `running` would show a + * turn nothing is running until some later boot cleans it up. + * `keepMarks` writes it as a stand-in the turn's own end can still + * replace (`SessionStore.releaseOwnTurns`). + */ + const releaseTurnsInterrupted = ( + options: { keepMarks?: boolean } = {}, + ): void => { + try { + sessionStore.releaseOwnTurns(INTERRUPTED_TURN_ENDING, options); + } catch (err) { + logger.warn("could not record the turns this shutdown interrupted", { + error: err instanceof Error ? err.message : String(err), + }); + } + }; let shutdownCalled = false; const shutdown = async (): Promise => { if (shutdownCalled) return; shutdownCalled = true; + // Say now, while the store is certainly open, that the turns still + // running were interrupted: a stop that turns into a kill partway + // through this teardown — the desktop gives it 4 s — still leaves + // every row right. It goes in as a stand-in: a turn that writes its + // own end before the store closes (below), or that throws and ends + // through `releaseTurn`, replaces it with what really happened. + releaseTurnsInterrupted({ keepMarks: true }); + // No scheduled turn may start on a runtime that is closing: the + // ticker stops here. The tick already running — and the task turn in + // it, which nothing stops — is waited for further down, where it + // always was. + const schedulerStopped = scheduler?.stop().catch((err: unknown) => { + logger.warn("scheduler stop failed", { + error: err instanceof Error ? err.message : String(err), + }); + }); // Nothing will drain the inbox after this point; drop pending // steers so a message cannot resurface in a later process. steeringInbox.clearAll(); @@ -2836,6 +2918,23 @@ export async function createAgentRuntime( error: err instanceof Error ? err.message : String(err), }); } + // The turns their hosts stopped — `serve`'s dropped connections, the + // TUI's and the sidecar's aborts, the channels above — are unwinding + // now. Closing the store under them is how a turn cancelled by a quit + // used to lose its end: it saved a moment after `close`, and its row + // kept what it held before the turn. Give them a moment first; a + // turn nobody stopped (a scheduled task) is not waited for. + const stillEnding = await turnsInFlight.settleCancelled( + SHUTDOWN_TURN_GRACE_MS, + ); + if (stillEnding > 0) { + logger.warn("stopped turns still running at shutdown; recorded as interrupted", { + count: stillEnding, + }); + } + // A turn that did not end in time, or that started after the release + // at the top, is recorded the same way before the store goes away. + releaseTurnsInterrupted(); try { sessionStore.close(); } catch { @@ -2869,15 +2968,8 @@ export async function createAgentRuntime( } catch { // already closed } - if (scheduler) { - try { - await scheduler.stop(); - } catch (err) { - logger.warn("scheduler stop failed", { - error: err instanceof Error ? err.message : String(err), - }); - } - } + // Stopped at the top; this waits out the tick that was running then. + await schedulerStopped; if (consolidatorJob) { try { await consolidatorJob.stop(); @@ -3173,6 +3265,57 @@ export async function createAgentRuntime( } }; + /** + * Mark the session's row `running` for the turn about to run, so the + * store says what is happening while it happens and a turn cut off by + * a kill or a crash is recognised at the next boot + * (`SessionStore.beginTurn`). Bookkeeping only: a mark that cannot be + * written must not stop the turn. + */ + const markTurnRunning = (sessionId: string): void => { + try { + sessionStore.beginTurn(sessionId); + } catch (err) { + logger.warn("could not mark the session running", { + sessionId, + error: err instanceof Error ? err.message : String(err), + }); + } + }; + + /** + * End a turn that has no state to save — it threw — on the status it + * actually ended with: `cancelled` when it had been told to stop (an + * abort can surface as any error), `failed` with the error otherwise. + * The transcript stays what it was before the turn (`session`); there + * is nothing truer to put there. A cancel puts back the `lastError` + * the session had before the turn, which a shutdown's stand-in may + * have overwritten meanwhile. + */ + const releaseThrownTurn = ( + session: SessionState, + err: unknown, + signal: AbortSignal | undefined, + ): void => { + try { + sessionStore.releaseTurn( + session.id, + signal?.aborted === true + ? { status: "cancelled", lastError: session.lastError } + : { + status: "failed", + lastError: err instanceof Error ? err.message : String(err), + }, + ); + } catch (releaseErr) { + logger.warn("could not record how a turn ended", { + sessionId: session.id, + error: + releaseErr instanceof Error ? releaseErr.message : String(releaseErr), + }); + } + }; + const executeTurn = async ( session: SessionState, userMessage: string, @@ -3264,6 +3407,10 @@ export async function createAgentRuntime( }); } return turnContext.run({ sessionId: session.id }, async () => { + // Registered before the mark and ended after the turn's end is + // written, so `shutdown` waiting on it waits for the row to be right. + const inFlight = turnsInFlight.begin(runOptions.signal); + markTurnRunning(session.id); try { // Recorded for `fusion.delegate`, which quotes it to the workers. const turnRequest = pickOriginalRequest({ @@ -3305,7 +3452,8 @@ export async function createAgentRuntime( [SESSION_ROUTE_METADATA_KEY]: turnRoute, }, }; - sessionStore.save(finished); + // The turn's end replaces its `running` mark (`beginTurn`). + sessionStore.finishTurn(finished); // Name the thread once, from its first prompt, after the first // turn that actually answered. Fire-and-forget on purpose: the // turn is already saved and already returned, and an unnamed @@ -3317,7 +3465,15 @@ export async function createAgentRuntime( // `finish` ended the whole session: its kept jobs go with it. if (finished.status === "completed") shellJobs.endSession(session.id); return { ...result, session: finished }; + } catch (err) { + // The loop hands back a state for every ending it can classify — + // reply, finish, max steps, failed, cancelled — so this is a turn + // that threw, or whose save did. Its row must not go on saying + // `running`. + releaseThrownTurn(session, err, runOptions.signal); + throw err; } finally { + inFlight.end(); // The turn is over, however it ended: the shell jobs it started // and did not `keep` are stopped here — the one choke point // every turn passes through (§"A turn is a task, not a step diff --git a/src/runtime/llm-fallback-seam.ts b/src/runtime/llm-fallback-seam.ts index be0b0aa1f..34becc3e2 100644 --- a/src/runtime/llm-fallback-seam.ts +++ b/src/runtime/llm-fallback-seam.ts @@ -124,6 +124,7 @@ export function createFallbackCompleter( deps.fallbackChain, (providerId) => attempt(providerId, params), params.sessionId, + params.signal, ); } @@ -162,6 +163,7 @@ export function createFallbackStreamer( ...(await openStreamOnLink(deps, params, id)), }), params.sessionId, + params.signal, ); const { primed, transport, providerId } = opened; let result: CompletionResult; diff --git a/src/runtime/turns-in-flight.test.ts b/src/runtime/turns-in-flight.test.ts new file mode 100644 index 000000000..561938dcc --- /dev/null +++ b/src/runtime/turns-in-flight.test.ts @@ -0,0 +1,78 @@ +import { describe, expect, it } from "vitest"; + +import { TurnsInFlight } from "./turns-in-flight.js"; + +/** + * What `shutdown` waits on before it closes the session store: the turns + * their hosts stopped, so each can write its own end; never a turn + * nobody stopped, and never longer than the grace. + */ + +function stopped(): AbortSignal { + const controller = new AbortController(); + controller.abort(); + return controller.signal; +} + +describe("TurnsInFlight.settleCancelled", () => { + it("resolves at once when no turn has been stopped", async () => { + const turns = new TurnsInFlight(); + // A scheduled task's turn (no signal) and a turn still being served. + turns.begin(undefined); + turns.begin(new AbortController().signal); + const started = Date.now(); + expect(await turns.settleCancelled(5_000)).toBe(0); + expect(Date.now() - started).toBeLessThan(1_000); + expect(turns.size).toBe(2); + }); + + it("waits for a stopped turn to end", async () => { + const turns = new TurnsInFlight(); + const turn = turns.begin(stopped()); + setTimeout(() => turn.end(), 30); + expect(await turns.settleCancelled(5_000)).toBe(0); + expect(turns.size).toBe(0); + }); + + it("waits for a turn its host stops from an I/O callback just after shutdown began", async () => { + const turns = new TurnsInFlight(); + const controller = new AbortController(); + const turn = turns.begin(controller.signal); + // A closing socket's callback is what stops a served turn; it runs a + // trip round the event loop after the shutdown that caused it. + setImmediate(() => controller.abort()); + setTimeout(() => turn.end(), 30); + expect(await turns.settleCancelled(5_000)).toBe(0); + expect(turns.size).toBe(0); + }); + + it("gives up on a stopped turn that does not end within the grace", async () => { + const turns = new TurnsInFlight(); + turns.begin(stopped()); + const ending = turns.begin(stopped()); + setTimeout(() => ending.end(), 10); + const started = Date.now(); + expect(await turns.settleCancelled(80)).toBe(1); + expect(Date.now() - started).toBeGreaterThanOrEqual(70); + }); + + it("waits only for the stopped turns, not for one still running beside them", async () => { + const turns = new TurnsInFlight(); + turns.begin(undefined); + const stoppedTurn = turns.begin(stopped()); + setTimeout(() => stoppedTurn.end(), 10); + const started = Date.now(); + expect(await turns.settleCancelled(5_000)).toBe(0); + expect(Date.now() - started).toBeLessThan(1_000); + expect(turns.size).toBe(1); + }); + + it("counts a turn once however often it is ended", () => { + const turns = new TurnsInFlight(); + const turn = turns.begin(undefined); + turns.begin(undefined); + turn.end(); + turn.end(); + expect(turns.size).toBe(1); + }); +}); diff --git a/src/runtime/turns-in-flight.ts b/src/runtime/turns-in-flight.ts new file mode 100644 index 000000000..c08c6b6eb --- /dev/null +++ b/src/runtime/turns-in-flight.ts @@ -0,0 +1,99 @@ +/** + * How long `shutdown` waits for turns that were told to stop to write + * their own end before it closes the session store under them. + * + * A stopped turn settles in milliseconds — the aborted request rejects, + * the loop records `cancelled`, `executeTurn` saves — so this only runs + * out on a turn stuck in work that ignores its signal. It has to fit + * well inside the desktop app's stop, which sends SIGTERM and kills the + * process 4 s later, with the rest of teardown still to run after it. + */ +export const SHUTDOWN_TURN_GRACE_MS = 1_500; + +interface InFlightTurn { + readonly signal: AbortSignal | undefined; + readonly settled: Promise; +} + +/** + * The turns `executeTurn` is running right now, for `shutdown` to wait + * on before it closes the session store. + * + * Every host stops its own turns before it shuts the runtime down — + * `serve` drops the connections (and its server's `close()` resolves + * only once every request has seen that), the TUI and the sidecar abort + * their controllers, the channels abort theirs as they stop — but + * shutdown then closed the store at once, while those turns were still + * unwinding. + * A turn that lost that race could not write its end: its row kept + * whatever it held before the turn, so a chat cancelled by quitting the + * app looked like one where nothing had happened. Seen in a user's + * database: two turns cancelled in the same second by a quit, one row + * `cancelled`, the other still `pending` with no turn in it. + */ +export class TurnsInFlight { + private readonly turns = new Set(); + + /** + * Register a turn that is starting. Call `end()` once it has written + * its end, or failed to; calling it twice is harmless. + */ + begin(signal: AbortSignal | undefined): { end(): void } { + let resolveSettled!: () => void; + const settled = new Promise((resolve) => { + resolveSettled = resolve; + }); + const turn: InFlightTurn = { signal, settled }; + this.turns.add(turn); + let ended = false; + return { + end: () => { + if (ended) return; + ended = true; + this.turns.delete(turn); + resolveSettled(); + }, + }; + } + + /** Turns registered and not yet ended. */ + get size(): number { + return this.turns.size; + } + + /** + * Wait, at most `graceMs`, for every turn whose signal has aborted to + * end. A turn nobody stopped is not waited for: it is running work — + * a scheduled task, most likely — that will not end on its own any time + * soon, and its row is released as interrupted instead. Resolves to the + * number of stopped turns still running when the wait gave up. + * + * The timer is deliberately not `unref`'d: shutdown awaits this, and a + * process whose only remaining handle is this timer must still come + * back to finish closing its stores. + */ + async settleCancelled(graceMs: number): Promise { + if (this.turns.size === 0) return 0; + // One trip round the event loop first: a host that stops its turns + // from an I/O callback (a closing socket) has not necessarily run it + // yet, and a turn sampled a moment too early is not waited for. + await new Promise((resolve) => setTimeout(resolve, 0)); + const cancelling = [...this.turns].filter( + (turn) => turn.signal?.aborted === true, + ); + if (cancelling.length === 0) return 0; + let timer: ReturnType | undefined; + const graceOver = new Promise((resolve) => { + timer = setTimeout(resolve, graceMs); + }); + try { + await Promise.race([ + Promise.all(cancelling.map((turn) => turn.settled)), + graceOver, + ]); + } finally { + clearTimeout(timer); + } + return cancelling.filter((turn) => this.turns.has(turn)).length; + } +} diff --git a/src/session/index.ts b/src/session/index.ts index ae4f82593..90e501a11 100644 --- a/src/session/index.ts +++ b/src/session/index.ts @@ -1,8 +1,20 @@ -export { SessionStore } from "./session-store.js"; +export { SessionStore, INTERRUPTED_TURN_ENDING } from "./session-store.js"; export type { SessionStoreOptions, RecentWorkingDirRow, + TurnEnding, } from "./session-store.js"; +export { + currentTurnOwnerProbe, + hostIdentity, + hostUptime, + isTurnOwnerGone, + parseTurnOwner, + processStartOf, + serializeTurnOwner, + turnOwnerFor, +} from "./turn-owner.js"; +export type { TurnOwner, TurnOwnerProbe } from "./turn-owner.js"; export { LIVE_SESSION_STATUSES, pruneSessions } from "./session-retention.js"; export type { PruneSessionsOptions, diff --git a/src/session/session-retention.ts b/src/session/session-retention.ts index 55491e8a2..4958d381a 100644 --- a/src/session/session-retention.ts +++ b/src/session/session-retention.ts @@ -49,6 +49,11 @@ const VACUUM_FREELIST_RATIO = 0.1; * the `reason === "reply"` branch) and only an explicit `finish` writes * `completed`, so status says nothing about whether a session is done — * which is why age, not status, is the rule. + * + * `running` is written by `SessionStore.beginTurn` for as long as a turn + * runs. A row left that way by a process that is gone is ended by + * `SessionStore.recoverInterruptedTurns`, which boot runs before this + * pass, so the exemption never keeps such a row for ever. */ export const LIVE_SESSION_STATUSES: readonly SessionStatus[] = [ "running", diff --git a/src/session/session-store.test.ts b/src/session/session-store.test.ts index d92ce63b0..200fce7b5 100644 --- a/src/session/session-store.test.ts +++ b/src/session/session-store.test.ts @@ -88,14 +88,16 @@ describe("SessionStore", () => { workingDir: "/work", }); store.save(state); + // Not `running`: a live status is the turn's own to write + // (`beginTurn`), and `save` keeps it out of the row. store.save({ ...state, - status: "running", + status: "completed", stepCount: 3, updatedAt: Date.now() + 100, }); const loaded = store.load("s2")!; - expect(loaded.status).toBe("running"); + expect(loaded.status).toBe("completed"); expect(loaded.stepCount).toBe(3); }); diff --git a/src/session/session-store.ts b/src/session/session-store.ts index ab71a41c4..04a919e0f 100644 --- a/src/session/session-store.ts +++ b/src/session/session-store.ts @@ -1,10 +1,15 @@ import type Database from "better-sqlite3"; import { Database as DatabaseCtor } from "../native/load-better-sqlite3.js"; -import { mkdirSync } from "node:fs"; -import { dirname } from "node:path"; +import { mkdirSync, realpathSync } from "node:fs"; +import { dirname, resolve } from "node:path"; import { getConfig } from "../config/index.js"; -import { stripEphemeral, type SessionState } from "./session-state.js"; +import { + stripEphemeral, + type SessionState, + type SessionStatus, +} from "./session-state.js"; import { normalizeSessionState } from "./normalize-session-state.js"; +import { LIVE_SESSION_STATUSES } from "./session-retention.js"; import type { SessionSummary } from "./session-summary.js"; import { summaryPageParams, @@ -18,7 +23,20 @@ import { SESSION_TITLE_METADATA_KEY, readSessionTitle, } from "./session-title.js"; +import { + currentTurnOwnerProbe, + isTurnOwnerGone, + serializeTurnOwner, + turnOwnerFor, + type TurnOwnerProbe, +} from "./turn-owner.js"; +// `turn_owner` names the process running a turn on the row (see +// `beginTurn`) and is null otherwise — a shutdown's stand-in keeps it +// until the turn's own end or the store's last release (see +// `releaseOwnTurns`). Databases made before it existed get it from +// `ensureTurnOwnerColumn`; a binary that predates it never names the +// column, so it reads and writes such a file exactly as before. const SCHEMA = ` CREATE TABLE IF NOT EXISTS sessions ( id TEXT PRIMARY KEY, @@ -26,7 +44,8 @@ CREATE TABLE IF NOT EXISTS sessions ( status TEXT NOT NULL, payload TEXT NOT NULL, created_at INTEGER NOT NULL, - updated_at INTEGER NOT NULL + updated_at INTEGER NOT NULL, + turn_owner TEXT ); CREATE INDEX IF NOT EXISTS idx_sessions_status ON sessions(status); CREATE INDEX IF NOT EXISTS idx_sessions_working_dir ON sessions(working_dir); @@ -35,8 +54,120 @@ CREATE INDEX IF NOT EXISTS idx_sessions_updated_id ON sessions(updated_at DESC, const COUNT_UNREADABLE_SQL = `SELECT COUNT(*) AS n FROM sessions WHERE NOT json_valid(payload)`; +const LIVE_STATUS_SQL_LIST = LIVE_SESSION_STATUSES.map( + (status) => `'${status}'`, +).join(", "); + +const LIVE_STATUSES: ReadonlySet = new Set(LIVE_SESSION_STATUSES); + +/** + * A status-only write. The column is where a row's status lives — every + * reader takes it from there (`readPayload`) — so this moves the column + * and touches the payload only to set `lastError`, keeping the payload's + * own `status` in step while it is there. A payload that is not JSON is + * left as it is: no reader can parse it anyway. + */ +const END_TURN_SQL = `status = @status, + payload = CASE WHEN @set_last_error = 0 OR NOT json_valid(payload) + THEN payload + ELSE json_set(payload, '$.status', @status, '$.lastError', @last_error) + END`; + +/** + * How a turn that could not write its own end is recorded instead: the + * status it ends on, and what becomes of `lastError` — a sentence to + * set, `null` to clear it, absent to keep what the row has. + */ +export interface TurnEnding { + readonly status: SessionStatus; + readonly lastError?: string | null; +} + +function endingParams(ending: TurnEnding): { + status: SessionStatus; + set_last_error: number; + last_error: string | null; +} { + return { + status: ending.status, + set_last_error: ending.lastError === undefined ? 0 : 1, + last_error: ending.lastError ?? null, + }; +} + +/** What a write needs to know about the row it is about to replace. */ +interface StoredRow { + /** The generated title, `null` when there is none. */ + title: string | null; + status: string; + turnOwner: string | null; +} + +/** + * The status a plain `save` may write over `stored`. + * + * A row's live status belongs to the turn that set it. While a row + * carries a turn's mark the stored status stands, whatever the copy + * being saved says — a model stamp from a copy read before the turn + * would otherwise say the session is idle while it runs. And `save` + * never writes a live status of its own: a copy read while a turn was + * running would put `running` back after that turn had ended, with no + * mark left that anything would ever take off. It keeps what the row + * says instead, or `pending` where that is itself a live status nothing + * owns. + */ +function statusForSave( + incoming: SessionStatus, + stored: StoredRow | undefined, +): SessionStatus { + if (stored !== undefined && stored.turnOwner !== null) { + return stored.status as SessionStatus; + } + if (!LIVE_STATUSES.has(incoming)) return incoming; + if (stored === undefined || LIVE_STATUSES.has(stored.status)) { + return "pending"; + } + return stored.status as SessionStatus; +} + +/** + * A turn whose process stopped before the turn could write its end: the + * app quit or was killed mid-turn, or the process died. `cancelled`, the + * same status a stopped turn gets, because nothing failed — the turn was + * cut off — and the sentence says by what. + */ +export const INTERRUPTED_TURN_ENDING: TurnEnding = { + status: "cancelled", + lastError: "turn interrupted: the agent stopped before it finished", +}; + export interface SessionStoreOptions { dbFile?: string; + /** + * This process, as the turn marks it writes and the boot sweep see it. + * Tests pin it; production reads it off the process and the host. + */ + turnOwnerProbe?: TurnOwnerProbe; + /** + * How long a statement waits on another connection's lock before it + * fails (better-sqlite3's `timeout`, 5 s by default). Tests shorten it. + */ + busyTimeoutMs?: number; +} + +/** The statements that read and write turn marks (`beginTurn` and on). */ +interface TurnMarkStatements { + begin: Database.Statement; + release: Database.Statement; + standIn: Database.Statement; + listLive: Database.Statement; + recover: Database.Statement; +} + +/** A row as the readers select it: the status column and the payload. */ +interface StoredPayloadRow { + status?: unknown; + payload: string; } /** Narrow projection returned by `listRecentWorkingDirs`. */ @@ -55,7 +186,8 @@ export class SessionStore { private readonly insertStmt: Database.Statement; private readonly updateStmt: Database.Statement; private readonly selectStmt: Database.Statement; - private readonly selectTitleStmt: Database.Statement; + private readonly selectStoredStmt: Database.Statement; + private readonly selectStoredBareStmt: Database.Statement; private readonly listByWorkingDirStmt: Database.Statement; private readonly listRecentStmt: Database.Statement; private readonly listRecentDirsStmt: Database.Statement; @@ -63,6 +195,17 @@ export class SessionStore { private readonly summaryNextPageStmt: Database.Statement; private readonly countUnreadableStmt: Database.Statement; private readonly deleteStmt: Database.Statement; + private readonly finishTurnStmt: Database.Statement; + /** + * `null` while this database has no `turn_owner` column — the open that + * should have added it could not (`turnMarksUnavailable`). Every turn + * mark method is then a no-op, and the store works as it did before + * marks existed. + */ + private readonly marks: TurnMarkStatements | null; + private readonly marksUnavailable: string | null; + /** The database file's real path, as marks record it; `undefined` in memory. */ + private readonly dbIdentity: string | undefined; /** * How many rows `load` / `listRecent` / `listByWorkingDir` have skipped * because their payload would not parse. Counts every skip, so the @@ -70,15 +213,29 @@ export class SessionStore { * number of such rows in the table. */ private unreadableSkips = 0; + private readonly turnOwnerProbe: TurnOwnerProbe; + /** + * The rows this store has marked `running` and not yet ended, with the + * exact mark it wrote on each — so a release only ever clears its own + * mark, never one another process put there since. + */ + private readonly ownTurns = new Map(); constructor(options: SessionStoreOptions = {}) { const config = getConfig(); const file = options.dbFile ?? config.paths.sessionsDbFile; mkdirSync(dirname(file), { recursive: true }); - this.db = new DatabaseCtor(file); + this.db = + options.busyTimeoutMs === undefined + ? new DatabaseCtor(file) + : new DatabaseCtor(file, { timeout: options.busyTimeoutMs }); this.db.pragma("journal_mode = WAL"); this.db.pragma("foreign_keys = ON"); this.db.exec(SCHEMA); + this.marksUnavailable = ensureTurnOwnerColumn(this.db); + const withMarks = this.marksUnavailable === null; + this.turnOwnerProbe = options.turnOwnerProbe ?? currentTurnOwnerProbe(); + this.dbIdentity = databaseIdentity(file); this.insertStmt = this.db.prepare( `INSERT INTO sessions (id, working_dir, status, payload, created_at, updated_at) VALUES (@id, @working_dir, @status, @payload, @created_at, @updated_at)`, @@ -91,21 +248,28 @@ export class SessionStore { updated_at = @updated_at WHERE id = @id`, ); + // Every reader takes `status` from the column, beside the payload: + // see `readPayload`. this.selectStmt = this.db.prepare( - `SELECT payload FROM sessions WHERE id = ?`, + `SELECT status, payload FROM sessions WHERE id = ?`, ); // Projected in SQL rather than parsed in JS: `save` runs this on // every write, and a transcript payload is the one thing in this // row worth not re-parsing. - this.selectTitleStmt = this.db.prepare( - `SELECT json_extract(payload, '$.metadata.${SESSION_TITLE_METADATA_KEY}') AS title + const turnOwnerColumn = withMarks ? "turn_owner" : "NULL"; + this.selectStoredStmt = this.db.prepare( + `SELECT status, ${turnOwnerColumn} AS turnOwner, + json_extract(payload, '$.metadata.${SESSION_TITLE_METADATA_KEY}') AS title FROM sessions WHERE id = ?`, ); + this.selectStoredBareStmt = this.db.prepare( + `SELECT status, ${turnOwnerColumn} AS turnOwner FROM sessions WHERE id = ?`, + ); this.listByWorkingDirStmt = this.db.prepare( - `SELECT payload FROM sessions WHERE working_dir = ? ORDER BY updated_at DESC LIMIT ?`, + `SELECT status, payload FROM sessions WHERE working_dir = ? ORDER BY updated_at DESC LIMIT ?`, ); this.listRecentStmt = this.db.prepare( - `SELECT payload FROM sessions ORDER BY updated_at DESC LIMIT ?`, + `SELECT status, payload FROM sessions ORDER BY updated_at DESC LIMIT ?`, ); this.listRecentDirsStmt = this.db.prepare( `SELECT working_dir AS workingDir, updated_at AS updatedAt @@ -119,6 +283,71 @@ export class SessionStore { this.summaryNextPageStmt = this.db.prepare(SUMMARY_NEXT_PAGE_SQL); this.countUnreadableStmt = this.db.prepare(COUNT_UNREADABLE_SQL); this.deleteStmt = this.db.prepare(`DELETE FROM sessions WHERE id = ?`); + this.finishTurnStmt = withMarks + ? this.db.prepare( + `UPDATE sessions + SET working_dir = @working_dir, + status = @status, + payload = @payload, + updated_at = @updated_at, + turn_owner = NULL + WHERE id = @id`, + ) + : this.updateStmt; + this.marks = withMarks ? this.prepareMarkStatements() : null; + } + + /** + * Why this store runs without turn marks, or `null` when it has them: + * the open that should have added the `turn_owner` column to an older + * database could not (another process held the write lock past the + * busy timeout, a file this process may only read). Sessions are + * stored as before, no turn is marked, and the next open tries the + * column again. Bootstrap logs it. + */ + get turnMarksUnavailable(): string | null { + return this.marksUnavailable; + } + + private prepareMarkStatements(): TurnMarkStatements { + return { + // Two columns and nothing else, at every turn start: the payload is + // not rewritten (readers take the status from the column), and + // `updated_at` is left alone — the row's content has not changed, + // and it is what every list orders by and the desktop reads + // "unread" from. + begin: this.db.prepare( + `UPDATE sessions SET status = 'running', turn_owner = @owner WHERE id = @id`, + ), + release: this.db.prepare( + `UPDATE sessions + SET ${END_TURN_SQL}, + turn_owner = NULL + WHERE id = @id AND turn_owner IS @owner`, + ), + // The same, keeping the mark: a stand-in the turn's own end can + // still replace (`releaseOwnTurns` with `keepMarks`). + standIn: this.db.prepare( + `UPDATE sessions + SET ${END_TURN_SQL} + WHERE id = @id AND turn_owner IS @owner`, + ), + // By status, not by mark: `idx_sessions_status` keeps this to the + // few rows that claim a live turn, and reading `turn_owner` — stored + // after the payload — on every row would read every transcript. + listLive: this.db.prepare( + `SELECT id, status, turn_owner AS turnOwner + FROM sessions WHERE status IN (${LIVE_STATUS_SQL_LIST})`, + ), + // Guarded on what the sweep read, so a turn another process started + // on the row in between keeps its mark. + recover: this.db.prepare( + `UPDATE sessions + SET ${END_TURN_SQL}, + turn_owner = NULL + WHERE id = @id AND status = @read_status AND turn_owner IS @owner`, + ), + }; } /** @@ -138,62 +367,247 @@ export class SessionStore { * no title does not remove one. Nothing renames a session today — * `shouldNameSession` refuses to name a session twice — so "keep what * is there" is also the product behaviour. + * + * And the status a turn set is the turn's own (`statusForSave`): a + * save never writes a live status, and never changes the status of a + * row a turn has marked. Only the turn's own end — `finishTurn` or + * `releaseTurn` — moves it, and takes the mark off. */ save(state: SessionState): void { - const stored = this.storedTitle(state.id); + this.write(state, false); + } + + /** + * Mark a session `running` for the turn about to run on it, and + * remember which process is running it, in the row itself. + * + * Before this the row only ever held a turn's end, written by + * `executeTurn` once the turn returned. A turn that never got there — + * the app closed mid-turn, the process killed or crashed — left the + * row as it was before the turn, so a turn that had been sent and then + * cut off looked like a session where nothing had happened. Now the + * row says a turn is running from the moment one starts, every way the + * turn ends replaces that (`finishTurn`, `releaseTurn`), and a mark + * whose process is gone is cleared at the next boot + * (`recoverInterruptedTurns`). + * + * Only an existing row is marked: a session nobody has saved yet (the + * TUI's deferred first turn) gets its row from `finishTurn`, as before. + * Returns whether a row was marked. + */ + beginTurn(id: string, now: number = Date.now()): boolean { + if (this.marks === null) return false; + const owner = serializeTurnOwner( + turnOwnerFor(this.turnOwnerProbe, now, this.dbIdentity), + ); + const result = this.marks.begin.run({ id, owner }) as { + changes: number; + }; + if (result.changes === 0) return false; + this.ownTurns.set(id, owner); + return true; + } + + /** + * Persist the state a turn ended with, and take the turn's mark off + * the row. Same title rule as `save`; inserts the row when there is + * none (a deferred session's first turn, or one deleted mid-turn). + * + * The turn stays this store's until the write has gone through: one + * that fails (a lock held too long, a full disk) leaves it to + * `releaseTurn` and, at shutdown, `releaseOwnTurns` — rather than a + * row that says `running` for as long as this process lives, under a + * pid no sweep will ever call gone. + */ + finishTurn(state: SessionState): void { + this.write(state, true); + this.ownTurns.delete(state.id); + } + + /** + * End a turn this store marked without the state it ended with — it + * threw, or the runtime is closing under it — by writing `ending` as + * its status. Touches the row only while it still carries this store's + * own mark: a turn that wrote its end already, or one another process + * has since started on the row, is left as it is. A failed write keeps + * the turn this store's, as in `finishTurn`. Returns whether the row + * was changed. + */ + releaseTurn(id: string, ending: TurnEnding): boolean { + const owner = this.ownTurns.get(id); + if (owner === undefined || this.marks === null) return false; + const result = this.marks.release.run({ + id, + owner, + ...endingParams(ending), + }) as { changes: number }; + this.ownTurns.delete(id); + return result.changes > 0; + } + + /** + * Write `ending` on every row this store still has marked: shutdown, + * where a turn that has not written its end by now never will. + * + * `keepMarks` is for the top of a shutdown. The ending goes in as a + * stand-in — right away, so a stop that turns into a kill partway + * through teardown still leaves every row right — but the marks stay, + * and so does this store's record of them: a turn that still gets to + * its own end, through `finishTurn` or through `releaseTurn` when it + * throws, replaces the stand-in with what really happened. Without it + * the marks come off and the store forgets them, the last word before + * it closes. + * + * A row whose write fails does not stop the others; the first error is + * thrown once every row has been tried. Returns how many rows changed. + */ + releaseOwnTurns( + ending: TurnEnding, + options: { keepMarks?: boolean } = {}, + ): number { + if (this.marks === null) return 0; + const keepMarks = options.keepMarks === true; + const statement = keepMarks ? this.marks.standIn : this.marks.release; + let changed = 0; + let failure: { error: unknown } | undefined; + for (const [id, owner] of [...this.ownTurns]) { + try { + const result = statement.run({ + id, + owner, + ...endingParams(ending), + }) as { changes: number }; + if (result.changes > 0) changed += 1; + if (!keepMarks) this.ownTurns.delete(id); + } catch (err) { + failure ??= { error: err }; + } + } + if (failure !== undefined) throw failure.error; + return changed; + } + + /** + * The boot sweep: every row still claiming a live turn + * (`LIVE_SESSION_STATUSES`) whose owning process is gone gets + * `ending` — by default `INTERRUPTED_TURN_ENDING` — so lists stop + * showing a turn nothing is running. A row a live process still owns + * is left alone: the store is shared by every process on the state dir + * (a second TUI window, `serve` beside a TUI), and their turns are + * theirs. Rows this store marked itself are never touched. + * + * `isOwnerGone` defaults to `isTurnOwnerGone` against this process; + * the default is only right at boot, before this process has started a + * turn (see there). Run as one `BEGIN IMMEDIATE` transaction: it reads + * and then writes, and a deferred one would fail outright when another + * process committed in between instead of waiting for the lock. Returns + * the ids it ended. + */ + recoverInterruptedTurns( + options: { + isOwnerGone?: (owner: string | null) => boolean; + ending?: TurnEnding; + } = {}, + ): string[] { + const marks = this.marks; + if (marks === null) return []; + const probe = this.turnOwnerProbe; + const db = this.dbIdentity; + const isOwnerGone: (owner: string | null) => boolean = + options.isOwnerGone ?? ((owner) => isTurnOwnerGone(owner, probe, db)); + const ending = endingParams(options.ending ?? INTERRUPTED_TURN_ENDING); + const sweep = this.db.transaction((): string[] => { + const rows = marks.listLive.all() as Array<{ + id: string; + status: string; + turnOwner: string | null; + }>; + const recovered: string[] = []; + for (const row of rows) { + if (this.ownTurns.has(row.id)) continue; + if (!isOwnerGone(row.turnOwner)) continue; + const result = marks.recover.run({ + id: row.id, + read_status: row.status, + owner: row.turnOwner, + ...ending, + }) as { changes: number }; + if (result.changes > 0) recovered.push(row.id); + } + return recovered; + }); + return sweep.immediate(); + } + + /** + * One row write: `save` (`endsTurn` false) or a turn's own end + * (`finishTurn`). Both keep a stored title the state does not carry; + * only `save` is held to `statusForSave`. + */ + private write(state: SessionState, endsTurn: boolean): void { + const stored = this.storedRow(state.id); + const status = endsTurn + ? state.status + : statusForSave(state.status, stored); + let next: SessionState = + status === state.status ? state : { ...state, status }; if (stored === undefined) { - this.insertStmt.run(this.serialize(state)); + this.insertStmt.run(this.serialize(next)); return; } - const keep = stored !== null && readSessionTitle(state.metadata) === null; - this.updateStmt.run( - this.serialize( - keep - ? { - ...state, - metadata: { - ...state.metadata, - [SESSION_TITLE_METADATA_KEY]: stored, - }, - } - : state, - ), - ); + if (stored.title !== null && readSessionTitle(next.metadata) === null) { + next = { + ...next, + metadata: { + ...next.metadata, + [SESSION_TITLE_METADATA_KEY]: stored.title, + }, + }; + } + const update = endsTurn ? this.finishTurnStmt : this.updateStmt; + update.run(this.serialize(next)); } /** - * The stored title: `undefined` when there is no such row, `null` - * when the row has no title. + * What a write needs from the row it replaces, or `undefined` when + * there is no such row. * * `json_extract` raises on a payload that is not valid JSON, and this * table tolerates those (see `countUnreadable`) — a corrupt row must - * not make saving impossible, so it falls back to the existence check - * `save` did before. + * not make saving impossible, so it falls back to the columns alone + * and reads as having no title. */ - private storedTitle(id: string): string | null | undefined { + private storedRow(id: string): StoredRow | undefined { + let row: + | { status: string; turnOwner: string | null; title?: unknown } + | undefined; try { - const row = this.selectTitleStmt.get(id) as - | { title: string | null } - | undefined; - if (row === undefined) return undefined; - return typeof row.title === "string" && row.title.trim().length > 0 - ? row.title - : null; + row = this.selectStoredStmt.get(id) as typeof row; } catch { - return this.selectStmt.get(id) === undefined ? undefined : null; + row = this.selectStoredBareStmt.get(id) as typeof row; } + if (row === undefined) return undefined; + return { + title: + typeof row.title === "string" && row.title.trim().length > 0 + ? row.title + : null, + status: row.status, + turnOwner: row.turnOwner, + }; } load(id: string): SessionState | null { - const row = this.selectStmt.get(id) as { payload: string } | undefined; + const row = this.selectStmt.get(id) as StoredPayloadRow | undefined; if (!row) return null; return this.readPayload(row); } listByWorkingDir(workingDir: string, limit = 25): SessionState[] { - const rows = this.listByWorkingDirStmt.all(workingDir, limit) as Array<{ - payload: string; - }>; + const rows = this.listByWorkingDirStmt.all( + workingDir, + limit, + ) as StoredPayloadRow[]; return this.readPayloads(rows); } @@ -203,7 +617,7 @@ export class SessionStore { * ongoing threads from any project root. */ listRecent(limit = 25): SessionState[] { - const rows = this.listRecentStmt.all(limit) as Array<{ payload: string }>; + const rows = this.listRecentStmt.all(limit) as StoredPayloadRow[]; return this.readPayloads(rows); } @@ -281,17 +695,26 @@ export class SessionStore { * Parse one stored payload, or `null` when it will not parse. A row * that is not JSON — a truncated write, a hand edit — used to throw * out of every list and empty the rail; now it is skipped and counted. + * + * The status comes from the column, not the payload: `beginTurn` moves + * only the column (rewriting a whole transcript to flip one word at + * every turn start is not worth it), so the payload's copy can be a + * turn behind. */ - private readPayload(row: { payload: string }): SessionState | null { + private readPayload(row: StoredPayloadRow): SessionState | null { + let state: SessionState; try { - return normalizeSessionState(JSON.parse(row.payload)); + state = normalizeSessionState(JSON.parse(row.payload)); } catch { this.unreadableSkips += 1; return null; } + return typeof row.status === "string" + ? { ...state, status: row.status as SessionStatus } + : state; } - private readPayloads(rows: Array<{ payload: string }>): SessionState[] { + private readPayloads(rows: StoredPayloadRow[]): SessionState[] { const states: SessionState[] = []; for (const row of rows) { const state = this.readPayload(row); @@ -312,3 +735,46 @@ export class SessionStore { }; } } + +/** + * Give a `sessions` table made before turn marks existed its + * `turn_owner` column. Additive and nullable, so every existing row + * reads as "no turn running" and an older binary sharing the file is + * unaffected. Two processes can open an old file at once; the one that + * loses the race to add the column finds it there. + * + * Adding it takes the write lock, which the first open after an upgrade + * may not get — another process mid-`VACUUM` past the busy timeout — or + * the write can fail: a database file this process may only read opens, + * reads and passes the schema check above, and refuses only this. That + * must not keep the runtime from starting: the store runs without turn + * marks (`turnMarksUnavailable`) and the next open tries again. Returns + * why the column is missing, or `null`. + */ +function ensureTurnOwnerColumn(db: Database.Database): string | null { + try { + const columns = db.prepare(`PRAGMA table_info(sessions)`).all() as Array<{ + name: string; + }>; + if (columns.some((column) => column.name === "turn_owner")) return null; + db.exec(`ALTER TABLE sessions ADD COLUMN turn_owner TEXT`); + return null; + } catch (err) { + const message = err instanceof Error ? err.message : String(err); + return /duplicate column/i.test(message) ? null : message; + } +} + +/** + * The real path of the database file, which every mark written into it + * carries: a mark found in another file came with a copy. `undefined` + * for an in-memory database, which nothing else can open. + */ +function databaseIdentity(file: string): string | undefined { + if (file === ":memory:" || file.length === 0) return undefined; + try { + return realpathSync.native(file); + } catch { + return resolve(file); + } +} diff --git a/src/session/session-store.turns.test.ts b/src/session/session-store.turns.test.ts new file mode 100644 index 000000000..335d70af5 --- /dev/null +++ b/src/session/session-store.turns.test.ts @@ -0,0 +1,683 @@ +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import type Database from "better-sqlite3"; +import { + chmodSync, + copyFileSync, + mkdtempSync, + realpathSync, + rmSync, +} from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; + +import { Database as DatabaseCtor } from "../native/load-better-sqlite3.js"; +import { userTurn } from "./conversation-turn.js"; +import { INTERRUPTED_TURN_ENDING, SessionStore } from "./session-store.js"; +import { + createEmptySessionState, + incrementTurnCount, + recordTurn, + type SessionState, +} from "./session-state.js"; +import type { TurnOwnerProbe } from "./turn-owner.js"; + +/** + * A turn's life in the store: `beginTurn` marks the row `running` and + * says which process runs it, and every way the turn ends takes that + * off — its own end (`finishTurn`), an end without a state + * (`releaseTurn`, `releaseOwnTurns`), or, when its process is gone, the + * next boot (`recoverInterruptedTurns`). Before the mark existed a turn + * cut off before its end left the row as it was before the turn. + */ + +const UPTIME = 50_000; + +function probe( + pid: number, + isAlive: (pid: number) => boolean = () => true, +): TurnOwnerProbe { + return { + pid, + host: "darwin", + hostUptime: () => UPTIME, + isAlive, + processStartOf: () => null, + }; +} + +interface RawRow { + status: string; + turnOwner: string | null; + updatedAt: number; + payload: string; +} + +describe("SessionStore turn marks", () => { + let tmp: string; + let file: string; + let store: SessionStore; + + beforeEach(() => { + tmp = mkdtempSync(join(tmpdir(), "atomic-agent-turns-")); + file = join(tmp, "sessions.sqlite"); + store = new SessionStore({ dbFile: file, turnOwnerProbe: probe(100) }); + }); + + afterEach(() => { + store.close(); + rmSync(tmp, { recursive: true, force: true }); + }); + + function raw(id: string): RawRow | undefined { + return store + .getDatabaseHandleForRetention() + .prepare( + `SELECT status, turn_owner AS turnOwner, updated_at AS updatedAt, payload + FROM sessions WHERE id = ?`, + ) + .get(id) as RawRow | undefined; + } + + /** Make every UPDATE of the table fail, as a full disk or a held lock would. */ + function failWrites(on: boolean): void { + const db = store.getDatabaseHandleForRetention(); + if (on) { + db.exec( + `CREATE TRIGGER fail_writes BEFORE UPDATE ON sessions + BEGIN SELECT RAISE(ABORT, 'disk full'); END`, + ); + } else { + db.exec(`DROP TRIGGER fail_writes`); + } + } + + function seed(id: string, extra: Partial = {}): SessionState { + const state: SessionState = { + ...createEmptySessionState({ id, workingDir: "/w" }), + updatedAt: 5_000, + ...extra, + }; + store.save(state); + return state; + } + + /** What the loop hands back for a turn: the message recorded, the turn closed. */ + function ended( + state: SessionState, + status: SessionState["status"], + ): SessionState { + return { + ...incrementTurnCount(recordTurn(state, userTurn("do the thing"))), + status, + }; + } + + it("beginTurn marks the row running, names this process, and rewrites nothing else", () => { + seed("s1"); + const before = raw("s1"); + expect(store.beginTurn("s1", 7_000)).toBe(true); + const row = raw("s1"); + expect(row?.status).toBe("running"); + // Readers take the status from the column. + expect(store.load("s1")?.status).toBe("running"); + expect(store.listRecent(10).find((s) => s.id === "s1")?.status).toBe( + "running", + ); + expect(JSON.parse(row?.turnOwner ?? "null")).toEqual({ + pid: 100, + host: "darwin", + db: realpathSync.native(file), + hostUptime: UPTIME, + at: 7_000, + }); + // Neither the transcript nor `updated_at` (list order, the desktop's + // unread dot) moves when a turn starts. + expect(row?.payload).toBe(before?.payload); + expect(row?.updatedAt).toBe(5_000); + }); + + it("beginTurn writes nothing for a session that has no row yet", () => { + expect(store.beginTurn("never-saved")).toBe(false); + expect(raw("never-saved")).toBeUndefined(); + }); + + it("finishTurn writes the turn's end and takes the mark off", () => { + const state = seed("s2"); + store.beginTurn("s2"); + store.finishTurn(ended(state, "pending")); + const row = raw("s2"); + expect(row?.status).toBe("pending"); + expect(row?.turnOwner).toBeNull(); + const loaded = store.load("s2"); + expect(loaded?.turnCount).toBe(1); + expect(loaded?.turns.map((t) => t.kind)).toEqual(["user"]); + }); + + it("finishTurn writes the row of a session first saved by its turn", () => { + // The TUI's deferred session: no row until its first turn ends. + const state = createEmptySessionState({ id: "deferred", workingDir: "/w" }); + expect(store.beginTurn("deferred")).toBe(false); + store.finishTurn(ended(state, "cancelled")); + expect(raw("deferred")?.status).toBe("cancelled"); + expect(raw("deferred")?.turnOwner).toBeNull(); + }); + + it("finishTurn keeps a title the turn's state does not carry", () => { + const state = seed("s-title"); + store.beginTurn("s-title"); + // Named by the previous turn's naming call while this one ran. + store.save({ + ...state, + metadata: { ...state.metadata, title: "Named meanwhile" }, + }); + store.finishTurn(ended(state, "pending")); + expect(store.load("s-title")?.metadata.title).toBe("Named meanwhile"); + }); + + it("keeps the turn its own when the write of its end fails", () => { + const state = seed("s-busy"); + store.beginTurn("s-busy"); + failWrites(true); + expect(() => store.finishTurn(ended(state, "pending"))).toThrow( + /disk full/, + ); + failWrites(false); + // Not forgotten: the end can still be written, and the mark comes off. + expect(store.releaseTurn("s-busy", { status: "failed", lastError: "x" })).toBe( + true, + ); + expect(raw("s-busy")?.status).toBe("failed"); + expect(raw("s-busy")?.turnOwner).toBeNull(); + }); + + it("keeps the turn its own when a release fails, for the next one to finish", () => { + seed("s-release"); + store.beginTurn("s-release"); + failWrites(true); + expect(() => + store.releaseTurn("s-release", { status: "cancelled" }), + ).toThrow(/disk full/); + failWrites(false); + expect(store.releaseOwnTurns(INTERRUPTED_TURN_ENDING)).toBe(1); + expect(raw("s-release")?.status).toBe("cancelled"); + expect(raw("s-release")?.turnOwner).toBeNull(); + }); + + describe("save while a turn runs", () => { + it("keeps the running status a copy from before the turn would overwrite", () => { + const state = seed("s3"); + store.beginTurn("s3"); + // A model stamp from a copy read before the turn started. + store.save({ ...state, metadata: { ...state.metadata, stamped: true } }); + expect(raw("s3")?.status).toBe("running"); + expect(raw("s3")?.turnOwner).not.toBeNull(); + expect(store.load("s3")?.metadata.stamped).toBe(true); + expect(store.releaseTurn("s3", { status: "cancelled" })).toBe(true); + expect(raw("s3")?.status).toBe("cancelled"); + expect(raw("s3")?.turnOwner).toBeNull(); + }); + + it("never writes back a running status read from a copy after the turn ended", () => { + const state = seed("s4"); + store.beginTurn("s4"); + const readMidTurn = store.load("s4")!; + expect(readMidTurn.status).toBe("running"); + store.finishTurn(ended(state, "failed")); + // The copy read mid-turn, saved after the turn ended. + store.save({ ...readMidTurn, stepCount: 9 }); + expect(raw("s4")?.status).toBe("failed"); + expect(store.load("s4")?.status).toBe("failed"); + expect(store.load("s4")?.stepCount).toBe(9); + }); + + it("writes a new row with a live status as pending", () => { + store.save({ + ...createEmptySessionState({ id: "fresh", workingDir: "/w" }), + status: "running", + }); + expect(raw("fresh")?.status).toBe("pending"); + }); + + it("lets every other status through as before", () => { + const state = seed("s5"); + store.save({ ...state, status: "completed" }); + expect(store.load("s5")?.status).toBe("completed"); + }); + }); + + it("releaseTurn writes the status and keeps lastError unless it is given one", () => { + seed("keep", { lastError: "from an earlier turn" }); + store.beginTurn("keep"); + expect(store.releaseTurn("keep", { status: "cancelled" })).toBe(true); + expect(store.load("keep")?.status).toBe("cancelled"); + expect(store.load("keep")?.lastError).toBe("from an earlier turn"); + + seed("replace"); + store.beginTurn("replace"); + expect( + store.releaseTurn("replace", { status: "failed", lastError: "it threw" }), + ).toBe(true); + expect(raw("replace")?.status).toBe("failed"); + expect(store.load("replace")?.status).toBe("failed"); + expect(store.load("replace")?.lastError).toBe("it threw"); + // The transcript is the one from before the turn: there is no other. + expect(store.load("replace")?.turns).toEqual([]); + + seed("clear", { lastError: "stale" }); + store.beginTurn("clear"); + store.releaseTurn("clear", { status: "cancelled", lastError: null }); + expect(store.load("clear")?.lastError).toBeNull(); + }); + + it("releaseTurn does nothing once the turn wrote its own end", () => { + const state = seed("done"); + store.beginTurn("done"); + store.finishTurn(ended(state, "pending")); + expect(store.releaseTurn("done", INTERRUPTED_TURN_ENDING)).toBe(false); + expect(store.load("done")?.status).toBe("pending"); + expect(store.load("done")?.lastError).toBeNull(); + }); + + it("releaseTurn leaves a turn another process has since started on the row", () => { + seed("shared"); + store.beginTurn("shared"); + const other = new SessionStore({ + dbFile: file, + turnOwnerProbe: probe(200), + }); + try { + other.beginTurn("shared"); + expect(store.releaseTurn("shared", INTERRUPTED_TURN_ENDING)).toBe(false); + expect(raw("shared")?.status).toBe("running"); + expect(JSON.parse(raw("shared")?.turnOwner ?? "null").pid).toBe(200); + } finally { + other.close(); + } + }); + + it("releaseOwnTurns ends every turn this store still has marked", () => { + const finished = seed("finished"); + seed("a"); + seed("b"); + store.beginTurn("finished"); + store.beginTurn("a"); + store.beginTurn("b"); + store.finishTurn(ended(finished, "pending")); + expect(store.releaseOwnTurns(INTERRUPTED_TURN_ENDING)).toBe(2); + for (const id of ["a", "b"]) { + expect(store.load(id)?.status).toBe("cancelled"); + expect(store.load(id)?.lastError).toBe(INTERRUPTED_TURN_ENDING.lastError); + expect(raw(id)?.turnOwner).toBeNull(); + } + expect(store.load("finished")?.status).toBe("pending"); + // Nothing left to release. + expect(store.releaseOwnTurns(INTERRUPTED_TURN_ENDING)).toBe(0); + }); + + describe("a shutdown's stand-in (keepMarks)", () => { + it("records the ending at once and leaves the mark", () => { + seed("s-stand"); + store.beginTurn("s-stand"); + expect( + store.releaseOwnTurns(INTERRUPTED_TURN_ENDING, { keepMarks: true }), + ).toBe(1); + expect(store.load("s-stand")?.status).toBe("cancelled"); + expect(store.load("s-stand")?.lastError).toBe( + INTERRUPTED_TURN_ENDING.lastError, + ); + expect(raw("s-stand")?.turnOwner).not.toBeNull(); + // The final release takes the mark off and forgets the turn. + expect(store.releaseOwnTurns(INTERRUPTED_TURN_ENDING)).toBe(1); + expect(raw("s-stand")?.turnOwner).toBeNull(); + expect(store.releaseOwnTurns(INTERRUPTED_TURN_ENDING)).toBe(0); + }); + + it("gives way to the turn's own end", () => { + const state = seed("s-own"); + store.beginTurn("s-own"); + store.releaseOwnTurns(INTERRUPTED_TURN_ENDING, { keepMarks: true }); + store.finishTurn(ended(state, "cancelled")); + const loaded = store.load("s-own"); + expect(loaded?.status).toBe("cancelled"); + expect(loaded?.lastError).toBeNull(); + expect(loaded?.turns.map((t) => t.kind)).toEqual(["user"]); + expect(raw("s-own")?.turnOwner).toBeNull(); + expect(store.releaseOwnTurns(INTERRUPTED_TURN_ENDING)).toBe(0); + expect(store.load("s-own")?.lastError).toBeNull(); + }); + + it("gives way to the end of a turn that threw", () => { + seed("s-threw"); + store.beginTurn("s-threw"); + store.releaseOwnTurns(INTERRUPTED_TURN_ENDING, { keepMarks: true }); + expect( + store.releaseTurn("s-threw", { status: "failed", lastError: "boom" }), + ).toBe(true); + expect(store.load("s-threw")?.status).toBe("failed"); + expect(store.load("s-threw")?.lastError).toBe("boom"); + expect(raw("s-threw")?.turnOwner).toBeNull(); + }); + + it("is left alone by a save, like the running status before it", () => { + const state = seed("s-saved"); + store.beginTurn("s-saved"); + store.releaseOwnTurns(INTERRUPTED_TURN_ENDING, { keepMarks: true }); + store.save({ ...state, status: "pending" }); + expect(raw("s-saved")?.status).toBe("cancelled"); + }); + }); + + describe("recoverInterruptedTurns", () => { + /** A turn another process started on `id` and never ended. */ + function markedBy(pid: number, id: string): void { + seed(id); + const other = new SessionStore({ + dbFile: file, + turnOwnerProbe: probe(pid), + }); + try { + expect(other.beginTurn(id)).toBe(true); + } finally { + // Closing is not ending: that process died mid-turn. + other.close(); + } + } + + /** A row that says `running` with no mark behind it. */ + function runningUnclaimed(id: string): void { + seed(id); + store + .getDatabaseHandleForRetention() + .prepare(`UPDATE sessions SET status = 'running' WHERE id = ?`) + .run(id); + } + + it("ends a turn whose process is gone, and leaves one a live process owns", () => { + markedBy(300, "dead-owner"); + markedBy(400, "live-owner"); + const sweeper = new SessionStore({ + dbFile: file, + turnOwnerProbe: probe(500, (pid) => pid === 400), + }); + try { + expect(sweeper.recoverInterruptedTurns()).toEqual(["dead-owner"]); + } finally { + sweeper.close(); + } + const dead = store.load("dead-owner"); + expect(dead?.status).toBe("cancelled"); + expect(dead?.lastError).toBe(INTERRUPTED_TURN_ENDING.lastError); + expect(raw("dead-owner")?.turnOwner).toBeNull(); + // Its last activity was before the turn; recording how it ended + // does not move it up any list. + expect(raw("dead-owner")?.updatedAt).toBe(5_000); + expect(store.load("live-owner")?.status).toBe("running"); + expect(raw("live-owner")?.turnOwner).not.toBeNull(); + }); + + it("ends a running row nothing claims", () => { + runningUnclaimed("unclaimed"); + expect(raw("unclaimed")?.turnOwner).toBeNull(); + expect(store.recoverInterruptedTurns()).toEqual(["unclaimed"]); + expect(store.load("unclaimed")?.status).toBe("cancelled"); + }); + + it("by default judges marks against this store's process", () => { + markedBy(100, "same-pid"); + // Same pid as the sweeping store: an earlier process with the + // same number, never a live turn of this one. + const sweeper = new SessionStore({ + dbFile: file, + turnOwnerProbe: probe(100), + }); + try { + expect(sweeper.recoverInterruptedTurns()).toEqual(["same-pid"]); + } finally { + sweeper.close(); + } + }); + + it("never touches a turn this store is running itself", () => { + seed("mine"); + store.beginTurn("mine"); + expect(store.recoverInterruptedTurns({ isOwnerGone: () => true })).toEqual( + [], + ); + expect(store.load("mine")?.status).toBe("running"); + }); + + it("leaves rows that claim no live turn alone", () => { + seed("idle"); + seed("stopped", { status: "cancelled" }); + seed("broke", { status: "failed", lastError: "x" }); + expect(store.recoverInterruptedTurns({ isOwnerGone: () => true })).toEqual( + [], + ); + expect(store.load("idle")?.status).toBe("pending"); + expect(store.load("broke")?.lastError).toBe("x"); + }); + + it("moves the status column of a row whose payload is not JSON", () => { + const db = new DatabaseCtor(file); + try { + db.prepare( + `INSERT INTO sessions (id, working_dir, status, payload, created_at, updated_at) + VALUES ('corrupt', '/w', 'running', '{not json', 1, 1)`, + ).run(); + } finally { + db.close(); + } + expect(store.recoverInterruptedTurns()).toEqual(["corrupt"]); + expect(raw("corrupt")?.status).toBe("cancelled"); + expect(raw("corrupt")?.payload).toBe("{not json"); + }); + + it("takes a custom ending", () => { + markedBy(300, "custom"); + const sweeper = new SessionStore({ + dbFile: file, + turnOwnerProbe: probe(500, () => false), + }); + try { + sweeper.recoverInterruptedTurns({ + ending: { status: "failed", lastError: "custom" }, + }); + } finally { + sweeper.close(); + } + expect(store.load("custom")?.status).toBe("failed"); + expect(store.load("custom")?.lastError).toBe("custom"); + }); + + it("ends the turns a copied database carries, even while their owner lives", () => { + // The desktop's "bring your terminal setup over" import copies the + // terminal agent's sessions.sqlite, perhaps mid-turn there. That + // turn runs on the original; in the copy it would say `running` for + // as long as the terminal agent lived. + markedBy(400, "mid-turn"); + store.close(); + const copy = join(tmp, "copy.sqlite"); + copyFileSync(file, copy); + store = new SessionStore({ dbFile: file, turnOwnerProbe: probe(100) }); + const copied = new SessionStore({ + dbFile: copy, + turnOwnerProbe: probe(500, (pid) => pid === 400), + }); + try { + expect(copied.recoverInterruptedTurns()).toEqual(["mid-turn"]); + expect(copied.load("mid-turn")?.status).toBe("cancelled"); + } finally { + copied.close(); + } + // The original, whose owner is alive, is left alone. + const sweeper = new SessionStore({ + dbFile: file, + turnOwnerProbe: probe(500, (pid) => pid === 400), + }); + try { + expect(sweeper.recoverInterruptedTurns()).toEqual([]); + expect(sweeper.load("mid-turn")?.status).toBe("running"); + } finally { + sweeper.close(); + } + }); + }); +}); + +describe("SessionStore on a database from before turn marks", () => { + it("adds the column, keeps every row readable, and marks turns on it", () => { + const tmp = mkdtempSync(join(tmpdir(), "atomic-agent-turns-old-")); + const file = join(tmp, "sessions.sqlite"); + try { + // The schema as every earlier release created it. + const db = new DatabaseCtor(file); + try { + db.exec(` + CREATE TABLE sessions ( + id TEXT PRIMARY KEY, + working_dir TEXT NOT NULL, + status TEXT NOT NULL, + payload TEXT NOT NULL, + created_at INTEGER NOT NULL, + updated_at INTEGER NOT NULL + ); + `); + const old = createEmptySessionState({ id: "old", workingDir: "/w" }); + db.prepare( + `INSERT INTO sessions (id, working_dir, status, payload, created_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?)`, + ).run("old", "/w", "pending", JSON.stringify(old), 1, 1); + } finally { + db.close(); + } + + const store = new SessionStore({ dbFile: file, turnOwnerProbe: probe(100) }); + try { + expect(store.load("old")?.status).toBe("pending"); + expect(store.beginTurn("old")).toBe(true); + expect(store.load("old")?.status).toBe("running"); + } finally { + store.close(); + } + + // Opening it again finds the column already there. + const again = new SessionStore({ dbFile: file, turnOwnerProbe: probe(100) }); + try { + expect(again.recoverInterruptedTurns()).toEqual(["old"]); + expect(again.load("old")?.status).toBe("cancelled"); + } finally { + again.close(); + } + } finally { + rmSync(tmp, { recursive: true, force: true }); + } + }); + + it("runs without turn marks when the column cannot be added, and adds it on a later open", () => { + // The first open after an upgrade needs the write lock; another + // process holding it past the busy timeout (a long VACUUM) must not + // keep the agent from starting. + const tmp = mkdtempSync(join(tmpdir(), "atomic-agent-turns-locked-")); + const file = join(tmp, "sessions.sqlite"); + const holder = new DatabaseCtor(file); + try { + const old = createPreMarksDatabase(holder); + holder.exec("BEGIN IMMEDIATE"); + + const store = new SessionStore({ + dbFile: file, + turnOwnerProbe: probe(100), + busyTimeoutMs: 50, + }); + try { + expect(store.turnMarksUnavailable).toMatch(/locked|busy/i); + expect(store.load("old")?.status).toBe("pending"); + expect(store.beginTurn("old")).toBe(false); + expect(store.recoverInterruptedTurns()).toEqual([]); + expect(store.releaseOwnTurns(INTERRUPTED_TURN_ENDING)).toBe(0); + holder.exec("ROLLBACK"); + // Everything else works as it did before marks existed. + store.finishTurn({ ...old, status: "cancelled", stepCount: 2 }); + expect(store.load("old")?.status).toBe("cancelled"); + store.save({ ...old, status: "running" }); + expect(store.load("old")?.status).toBe("cancelled"); + } finally { + store.close(); + } + + // With the lock free, the next open adds the column. + const later = new SessionStore({ dbFile: file, turnOwnerProbe: probe(100) }); + try { + expect(later.turnMarksUnavailable).toBeNull(); + expect(later.beginTurn("old")).toBe(true); + expect(later.load("old")?.status).toBe("running"); + } finally { + later.close(); + } + } finally { + holder.close(); + rmSync(tmp, { recursive: true, force: true }); + } + }); + + // Root writes a read-only file anyway, and Windows locks files its own + // way: the refusal this needs is POSIX file modes for a normal user. + it.skipIf(process.platform === "win32" || process.getuid?.() === 0)( + "runs without turn marks on a database file it may only read", + () => { + // Such a file opens, reads, and passes the schema check; only + // adding the column is refused. + const tmp = mkdtempSync(join(tmpdir(), "atomic-agent-turns-readonly-")); + const file = join(tmp, "sessions.sqlite"); + try { + const seed = new DatabaseCtor(file); + try { + createPreMarksDatabase(seed); + } finally { + seed.close(); + } + chmodSync(file, 0o444); + + const store = new SessionStore({ dbFile: file, turnOwnerProbe: probe(100) }); + try { + expect(store.turnMarksUnavailable).toMatch(/readonly|read-only/i); + expect(store.load("old")?.status).toBe("pending"); + expect(store.beginTurn("old")).toBe(false); + expect(store.recoverInterruptedTurns()).toEqual([]); + expect(store.releaseOwnTurns(INTERRUPTED_TURN_ENDING)).toBe(0); + } finally { + store.close(); + } + } finally { + rmSync(tmp, { recursive: true, force: true }); + } + }, + ); +}); + +/** + * The `sessions` table as the release before turn marks created it, + * indexes included, in WAL mode, holding one session: `old`. + */ +function createPreMarksDatabase(db: Database.Database): SessionState { + db.pragma("journal_mode = WAL"); + db.exec(` + CREATE TABLE sessions ( + id TEXT PRIMARY KEY, + working_dir TEXT NOT NULL, + status TEXT NOT NULL, + payload TEXT NOT NULL, + created_at INTEGER NOT NULL, + updated_at INTEGER NOT NULL + ); + CREATE INDEX idx_sessions_status ON sessions(status); + CREATE INDEX idx_sessions_working_dir ON sessions(working_dir); + CREATE INDEX idx_sessions_updated_id ON sessions(updated_at DESC, id DESC); + `); + const old = createEmptySessionState({ id: "old", workingDir: "/w" }); + db.prepare( + `INSERT INTO sessions (id, working_dir, status, payload, created_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?)`, + ).run("old", "/w", "pending", JSON.stringify(old), 1, 1); + return old; +} diff --git a/src/session/turn-owner.test.ts b/src/session/turn-owner.test.ts new file mode 100644 index 000000000..3d0c9f44f --- /dev/null +++ b/src/session/turn-owner.test.ts @@ -0,0 +1,277 @@ +import { describe, expect, it } from "vitest"; +import { uptime } from "node:os"; + +import { + currentTurnOwnerProbe, + hostIdentity, + hostUptime, + isTurnOwnerGone, + parseTurnOwner, + processStartOf, + serializeTurnOwner, + turnOwnerFor, + type TurnOwner, + type TurnOwnerProbe, +} from "./turn-owner.js"; + +/** + * The boot sweep's one judgement: is the process a `running` row names + * gone, so the turn will never write its end? Once the pid is alive only + * certain evidence may say so — cancelling a turn another window is + * still running is the worse mistake. + */ + +const DB = "/state/sessions.sqlite"; + +function probe(overrides: Partial = {}): TurnOwnerProbe { + return { + pid: 100, + host: "darwin", + hostUptime: () => 50_000, + isAlive: () => true, + processStartOf: () => null, + ...overrides, + }; +} + +function mark(owner: Partial & { pid: number }): string { + return serializeTurnOwner({ host: "darwin", at: 1_790_000_123_456, ...owner }); +} + +describe("serializeTurnOwner / parseTurnOwner", () => { + it("reads back what was written", () => { + const owner: TurnOwner = { + pid: 4242, + host: "linux:pid:[4026531836]", + db: DB, + hostUptime: 40_000, + processStart: "ticks:123456", + at: 1_790_000_123_456, + }; + expect(parseTurnOwner(serializeTurnOwner(owner))).toEqual(owner); + }); + + it("leaves out what the platform could not say", () => { + expect( + parseTurnOwner(serializeTurnOwner({ pid: 9, at: 1_790_000_123_456 })), + ).toEqual({ pid: 9, at: 1_790_000_123_456 }); + }); + + it("is null for no mark, a mark that is not JSON, or one without a usable pid", () => { + expect(parseTurnOwner(null)).toBeNull(); + expect(parseTurnOwner("{not json")).toBeNull(); + expect(parseTurnOwner("42")).toBeNull(); + expect(parseTurnOwner("null")).toBeNull(); + expect(parseTurnOwner(JSON.stringify({ hostUptime: 5 }))).toBeNull(); + expect(parseTurnOwner(JSON.stringify({ pid: "7" }))).toBeNull(); + expect(parseTurnOwner(JSON.stringify({ pid: 0 }))).toBeNull(); + expect(parseTurnOwner(JSON.stringify({ pid: 1.5 }))).toBeNull(); + }); + + it("drops fields of the wrong shape and keeps the pid", () => { + expect( + parseTurnOwner( + JSON.stringify({ + pid: 9, + host: 3, + db: "", + hostUptime: "soon", + processStart: 12, + at: "x", + }), + ), + ).toEqual({ pid: 9, at: 0 }); + }); +}); + +describe("turnOwnerFor", () => { + it("records the probe's process and host, the database, the uptime now and the start", () => { + let up = 1_000; + const owner = turnOwnerFor( + probe({ + pid: 77, + hostUptime: () => up, + processStartOf: (pid) => (pid === 77 ? "lstart:Fri Oct 2 12:10:29 2026" : null), + }), + 123, + DB, + ); + expect(owner).toEqual({ + pid: 77, + host: "darwin", + db: DB, + hostUptime: 1_000, + processStart: "lstart:Fri Oct 2 12:10:29 2026", + at: 123, + }); + // Read at the moment of the mark, not when the probe was built. + up = 2_000; + expect(turnOwnerFor(probe({ hostUptime: () => up }), 1).hostUptime).toBe( + 2_000, + ); + }); +}); + +describe("isTurnOwnerGone", () => { + it("keeps a turn whose process is alive", () => { + expect( + isTurnOwnerGone(mark({ pid: 200, db: DB, hostUptime: 40_000 }), probe(), DB), + ).toBe(false); + }); + + it("ends a turn whose process has exited", () => { + const isAlive = (pid: number) => pid !== 200; + expect(isTurnOwnerGone(mark({ pid: 200 }), probe({ isAlive }))).toBe(true); + expect(isTurnOwnerGone(mark({ pid: 300 }), probe({ isAlive }))).toBe(false); + }); + + it("ends a turn marked with this process's own pid, without asking whether it is alive", () => { + // At boot this process has run no turn yet: the mark is an earlier + // process that had the same number. + let asked = false; + const gone = isTurnOwnerGone( + mark({ pid: 100 }), + probe({ + isAlive: () => { + asked = true; + return true; + }, + }), + ); + expect(gone).toBe(true); + expect(asked).toBe(false); + }); + + it("never judges a mark from another pid namespace, dead pid or not", () => { + // A container sharing the state dir numbers its processes on its own. + const foreign = mark({ pid: 7, host: "linux:pid:[4026532415]" }); + const here = probe({ + host: "linux:pid:[4026531836]", + isAlive: () => false, + }); + expect(isTurnOwnerGone(foreign, here)).toBe(false); + // Nor one from another platform sharing the directory. + expect(isTurnOwnerGone(mark({ pid: 7, host: "win32" }), here)).toBe(false); + }); + + it("ends a turn whose mark was written into another database file — a copy", () => { + // The desktop's import copies the terminal agent's sessions.sqlite + // while a turn may be running there; that turn runs on the original. + expect( + isTurnOwnerGone( + mark({ pid: 200, db: "/home/u/.atomic-agent/sessions.sqlite" }), + probe(), + "/home/u/.atomic-agent-desktop/sessions.sqlite", + ), + ).toBe(true); + // The same file is the same turn. + expect(isTurnOwnerGone(mark({ pid: 200, db: DB }), probe(), DB)).toBe(false); + }); + + it("ends a turn from before a reboot even when its pid is alive now", () => { + // Up for a day when the mark was written, up for an hour now. + expect( + isTurnOwnerGone( + mark({ pid: 200, hostUptime: 86_400 }), + probe({ hostUptime: () => 3_600 }), + ), + ).toBe(true); + }); + + it("does not read a reboot into an uptime a moment off, or a clock step", () => { + // Uptime is whole seconds on some platforms; and it does not move + // with the wall clock, so nothing here can turn a clock step into a + // reboot. + expect( + isTurnOwnerGone( + mark({ pid: 200, hostUptime: 3_601 }), + probe({ hostUptime: () => 3_600 }), + ), + ).toBe(false); + expect( + isTurnOwnerGone( + mark({ pid: 200, hostUptime: 3_600 }), + probe({ hostUptime: () => 90_000 }), + ), + ).toBe(false); + }); + + it("keeps a live pid when either side does not know the uptime", () => { + expect( + isTurnOwnerGone(mark({ pid: 200 }), probe({ hostUptime: () => 1 })), + ).toBe(false); + expect( + isTurnOwnerGone( + mark({ pid: 200, hostUptime: 86_400 }), + probe({ hostUptime: () => undefined }), + ), + ).toBe(false); + }); + + it("ends a turn whose pid now belongs to a process that started at another moment", () => { + const processStartOf = (pid: number) => + pid === 200 ? "lstart:Fri Oct 2 12:10:29 2026" : null; + expect( + isTurnOwnerGone( + mark({ pid: 200, processStart: "lstart:Thu Oct 1 09:00:00 2026" }), + probe({ processStartOf }), + ), + ).toBe(true); + expect( + isTurnOwnerGone( + mark({ pid: 200, processStart: "lstart:Fri Oct 2 12:10:29 2026" }), + probe({ processStartOf }), + ), + ).toBe(false); + }); + + it("keeps a live pid whose start cannot be read", () => { + expect( + isTurnOwnerGone( + mark({ pid: 200, processStart: "lstart:Thu Oct 1 09:00:00 2026" }), + probe(), + ), + ).toBe(false); + }); + + it("ends a live status with no mark, or a mark that does not parse", () => { + expect(isTurnOwnerGone(null, probe())).toBe(true); + expect(isTurnOwnerGone("garbage", probe())).toBe(true); + }); +}); + +describe("the live probe", () => { + it("reports this process, its host, a positive uptime and itself as alive", () => { + const live = currentTurnOwnerProbe(); + expect(live.pid).toBe(process.pid); + expect(live.host).toBe(hostIdentity()); + expect(live.isAlive(process.pid)).toBe(true); + const up = live.hostUptime(); + expect(up).toBeGreaterThan(0); + expect(Math.abs((up ?? 0) - uptime())).toBeLessThanOrEqual(2); + expect(hostUptime()).toBeGreaterThan(0); + }); + + it("names the host by platform, and the pid namespace on Linux", () => { + const host = hostIdentity(); + if (process.platform === "linux") { + expect(host).toMatch(/^linux(:pid:\[\d+\])?$/); + } else { + expect(host).toBe(process.platform); + } + }); + + it("reads a process's start where the platform offers it, the same on every read", () => { + const own = processStartOf(process.pid); + if (process.platform === "linux") { + expect(own).toMatch(/^ticks:\d+$/); + } else if (process.platform === "darwin") { + expect(own).toMatch(/^lstart:\S/); + } else { + expect(own).toBeNull(); + } + expect(processStartOf(process.pid)).toBe(own); + // A pid with no process behind it is unknown, never a value. + expect(processStartOf(2_147_483_646)).toBeNull(); + }); +}); diff --git a/src/session/turn-owner.ts b/src/session/turn-owner.ts new file mode 100644 index 000000000..516cd73bb --- /dev/null +++ b/src/session/turn-owner.ts @@ -0,0 +1,321 @@ +import { execFileSync } from "node:child_process"; +import { readFileSync, readlinkSync } from "node:fs"; +import { uptime } from "node:os"; + +/** + * Who is running the turn a session row is marked `running` for. + * + * A turn used to exist only in memory until it ended: `executeTurn` + * wrote the row once, from the finished state. Whatever stopped a turn + * before that write — the app quitting mid-turn, a process killed + * outright (Windows has no SIGTERM, the desktop's stop is a forced tree + * kill there), a crash — left the row exactly as it was before the turn, + * so a turn the user had sent, and that had been cancelled, read as a + * session sitting there with nothing happening. `SessionStore.beginTurn` + * now marks the row `running` and stores one of these beside it, and + * the next runtime to boot can tell a mark whose process is gone from + * one a live process still owns (`isTurnOwnerGone`). + * + * Stored as JSON in the `turn_owner` column, never in the payload: every + * `save` rewrites the payload from whatever copy its caller holds, and a + * mark that rode along in it would be dropped by the first save of a + * copy taken before the turn began. + */ +export interface TurnOwner { + /** The process running the turn. */ + readonly pid: number; + /** + * What the pid is a pid of: the platform, and on Linux the pid + * namespace (`/proc/self/ns/pid`). A container sharing the state dir + * with its host — or with another container — numbers its processes + * on its own; a mark from another namespace says nothing about the pid + * of the same number here, so it is never judged from here. The + * hostname is deliberately not part of it: on a Mac it follows the + * network, and a mark that looked foreign after a network change would + * never be cleared. + */ + readonly host?: string; + /** + * The database file the mark was written into (its real path). A mark + * found in another file came with a copy — the desktop's "bring your + * terminal setup over" import, a restored backup — and whatever turn it + * names runs on the original, never on the copy. + */ + readonly db?: string; + /** + * The host's uptime, in seconds, when the mark was written. A host + * whose uptime is now lower has rebooted since, so the pid names + * nobody, however alive the process holding that number now is. The + * uptime counts from boot on its own clock, so stepping the wall clock + * does not move it — a boot time worked out from the wall clock did, + * and could make a live owner look like one from an earlier boot. + * Absent when the platform would not say. + */ + readonly hostUptime?: number; + /** + * When the owning process started, as the kernel recorded it + * (`processStartOf`). With the pid it names exactly one process for as + * long as the host is up, so a pid since reused by another process is + * caught. Absent where it cannot be read cheaply (Windows). + */ + readonly processStart?: string; + /** When the turn started, ms since the epoch. For diagnostics. */ + readonly at: number; +} + +/** + * This process and host: what a mark records about the turn's owner, and + * what `isTurnOwnerGone` compares a mark against. Functions, not values, + * where the answer moves, so each mark and each sweep reads the host as + * it is at that moment. + */ +export interface TurnOwnerProbe { + readonly pid: number; + /** `TurnOwner.host` for this process; `undefined` when unknown. */ + readonly host: string | undefined; + /** The host's uptime now, in seconds; `undefined` when unknown. */ + readonly hostUptime: () => number | undefined; + readonly isAlive: (pid: number) => boolean; + /** `TurnOwner.processStart` of process `pid`, or `null` when it cannot be read. */ + readonly processStartOf: (pid: number) => string | null; +} + +/** + * Slack for the uptime comparison: uptime is whole seconds on some + * platforms, and two readings a moment apart can differ by one. + */ +const UPTIME_SLACK_S = 5; + +/** The host's uptime in seconds, or `undefined` when the platform will not say. */ +export function hostUptime(): number | undefined { + try { + const up = uptime(); + return Number.isFinite(up) && up > 0 ? up : undefined; + } catch { + return undefined; + } +} + +/** `TurnOwner.host` for this process (see there). */ +export function hostIdentity(): string { + if (process.platform !== "linux") return process.platform; + try { + return `linux:${readlinkSync("/proc/self/ns/pid")}`; + } catch { + return "linux"; + } +} + +/** + * The start of process `pid` as the kernel recorded it, as an opaque + * string only ever compared with another reading on the same host: + * + * - Linux: `starttime` from `/proc//stat`, in clock ticks since + * boot — a file read; + * - macOS: `ps -o lstart=`, the start the kernel stored when the process + * was created, printed in UTC with the C locale so neither a time-zone + * change nor the wall clock moving since alters it — one short `ps`; + * - anywhere else (Windows): `null`. Asking costs a PowerShell start of + * several hundred milliseconds, too much for the turn path. + * + * `null` too when the process cannot be read, which callers must treat + * as "unknown", never as "gone". + */ +export function processStartOf(pid: number): string | null { + if (process.platform === "linux") return linuxStartTicks(pid); + if (process.platform === "darwin") return darwinStart(pid); + return null; +} + +function linuxStartTicks(pid: number): string | null { + let stat: string; + try { + stat = readFileSync(`/proc/${pid}/stat`, "utf8"); + } catch { + return null; + } + // `pid (comm) state ppid …`: the command name may hold spaces and + // parentheses, so fields are counted from its closing parenthesis. + // `starttime` is field 22; the first field after `)` is field 3. + const close = stat.lastIndexOf(")"); + if (close < 0) return null; + const start = stat.slice(close + 1).trim().split(/\s+/)[19]; + return start !== undefined && /^\d+$/.test(start) ? `ticks:${start}` : null; +} + +function darwinStart(pid: number): string | null { + try { + const out = execFileSync("/bin/ps", ["-o", "lstart=", "-p", String(pid)], { + encoding: "utf8", + // Nothing of this process's environment (keys included) goes to + // `ps`; it needs only the locale and the zone to print in. + env: { LC_ALL: "C", TZ: "UTC" }, + stdio: ["ignore", "pipe", "ignore"], + timeout: 2_000, + }).trim(); + return out.length > 0 ? `lstart:${out.replace(/\s+/g, " ")}` : null; + } catch { + return null; + } +} + +let ownStart: string | null | undefined; + +/** This process's own start, read once. */ +function ownProcessStart(): string | null { + if (ownStart === undefined) ownStart = processStartOf(process.pid); + return ownStart; +} + +/** This process and host, right now. */ +export function currentTurnOwnerProbe(): TurnOwnerProbe { + return { + pid: process.pid, + host: hostIdentity(), + hostUptime, + isAlive: isProcessAlive, + processStartOf: (pid) => + pid === process.pid ? ownProcessStart() : processStartOf(pid), + }; +} + +/** + * The mark `probe`'s process writes into the database at `db` for a turn + * starting at `at`. + */ +export function turnOwnerFor( + probe: TurnOwnerProbe, + at: number, + db?: string, +): TurnOwner { + const up = probe.hostUptime(); + const start = probe.processStartOf(probe.pid); + return { + pid: probe.pid, + ...(probe.host !== undefined ? { host: probe.host } : {}), + ...(db !== undefined ? { db } : {}), + ...(up !== undefined ? { hostUptime: up } : {}), + ...(start !== null ? { processStart: start } : {}), + at, + }; +} + +/** The stored form of a mark. */ +export function serializeTurnOwner(owner: TurnOwner): string { + return JSON.stringify({ + pid: owner.pid, + ...(owner.host !== undefined ? { host: owner.host } : {}), + ...(owner.db !== undefined ? { db: owner.db } : {}), + ...(owner.hostUptime !== undefined ? { hostUptime: owner.hostUptime } : {}), + ...(owner.processStart !== undefined + ? { processStart: owner.processStart } + : {}), + at: owner.at, + }); +} + +/** A stored mark, or `null` when there is none or it does not parse. */ +export function parseTurnOwner(raw: string | null): TurnOwner | null { + if (raw === null) return null; + let value: unknown; + try { + value = JSON.parse(raw); + } catch { + return null; + } + if (typeof value !== "object" || value === null) return null; + const fields = value as Record; + const { pid, host, db, processStart, at } = fields; + const up = fields.hostUptime; + if (typeof pid !== "number" || !Number.isInteger(pid) || pid <= 0) { + return null; + } + const text = (v: unknown): v is string => + typeof v === "string" && v.length > 0; + return { + pid, + ...(text(host) ? { host } : {}), + ...(text(db) ? { db } : {}), + ...(typeof up === "number" && Number.isFinite(up) && up > 0 + ? { hostUptime: up } + : {}), + ...(text(processStart) ? { processStart } : {}), + at: typeof at === "number" && Number.isFinite(at) ? at : 0, + }; +} + +/** + * Whether the process a `running` row names can no longer be running + * that turn — so the row is a turn that will never write its end. + * `db` is the real path of the database the row was read from. + * + * Meant for the boot sweep, before this process has started a turn of + * its own: a mark carrying this process's pid then belongs to an earlier + * process that had the same number (or to an earlier runtime in this one + * that never got to release it), never to a live turn. + * + * A mark from another pid namespace is never judged at all (`host`). + * Once the pid is alive, only certain evidence counts: the host's uptime + * has gone backwards since the mark (it rebooted), or the process now + * holding the pid started at another moment than the one that wrote the + * mark. Anything less leaves the row alone — cancelling a turn another + * window is still running is the worse mistake, and a mark left behind + * is cleared by a later boot or by the next turn on that session. Gone: + * + * - no mark, or one that does not parse — a live status no turn claims + * (`beginTurn` never writes `running` without a mark, and `save` never + * writes a live status at all); + * - a mark written into another database file — this row is a copy; + * - this process's pid (see above); + * - a pid with no process behind it; + * - a host that has rebooted since the mark; + * - a pid now held by a process that started at another moment. + */ +export function isTurnOwnerGone( + raw: string | null, + probe: TurnOwnerProbe, + db?: string, +): boolean { + const owner = parseTurnOwner(raw); + if (owner === null) return true; + if ( + owner.host !== undefined && + probe.host !== undefined && + owner.host !== probe.host + ) { + return false; + } + if (owner.db !== undefined && db !== undefined && owner.db !== db) { + return true; + } + if (owner.pid === probe.pid) return true; + if (!probe.isAlive(owner.pid)) return true; + const up = probe.hostUptime(); + if ( + owner.hostUptime !== undefined && + up !== undefined && + up + UPTIME_SLACK_S < owner.hostUptime + ) { + return true; + } + if (owner.processStart !== undefined) { + const now = probe.processStartOf(owner.pid); + if (now !== null && now !== owner.processStart) return true; + } + return false; +} + +/** + * `EPERM` (POSIX) / `EACCES` (Windows) both mean the process exists but + * belongs to someone else — still alive. Only `ESRCH` means dead. Same + * probe as `cli/serve-orphan-guard.ts`. + */ +function isProcessAlive(pid: number): boolean { + try { + process.kill(pid, 0); + return true; + } catch (err) { + const code = (err as NodeJS.ErrnoException).code; + return code === "EPERM" || code === "EACCES"; + } +} diff --git a/src/skills/hub/skill-hub-catalog.test.ts b/src/skills/hub/skill-hub-catalog.test.ts index 4d21a023f..8b6ed8d42 100644 --- a/src/skills/hub/skill-hub-catalog.test.ts +++ b/src/skills/hub/skill-hub-catalog.test.ts @@ -177,3 +177,90 @@ describe("browseTap lazy branch resolution", () => { expect(resolveDefaultCalls).toBe(1); }); }); + +describe("browseHub / browseTap fan-out", () => { + /** Every microtask queued so far, and those they queue, have run. */ + const drain = () => new Promise((r) => setImmediate(r)); + + it("reads the taps side by side, not one after the other", async () => { + const started: string[] = []; + let aSawB = false; + const client: SkillHubClient = { + resolveDefaultBranch: async () => "main", + listSkillManifests: async (owner) => { + started.push(owner); + // One microtask on: a tap read alongside has begun by now, one read after this one has not. + await Promise.resolve(); + if (owner === "a") aSawB = started.includes("b"); + return [{ dir: "s", manifestPath: "s/SKILL.md" }]; + }, + fetchTextFile: async (owner) => manifest(`${owner}-s`, "x"), + downloadSkillDir: async () => [], + }; + const { entries } = await browseHub(client, [ + { repo: "a/r", path: "" }, + { repo: "b/r", path: "" }, + ]); + expect(entries.map((e) => e.name)).toEqual(["a-s", "b-s"]); + expect(aSawB).toBe(true); + }); + + it("keeps errors in the configured order whichever tap fails first (an order guard: reading the taps one after the other kept it too)", async () => { + const client: SkillHubClient = { + resolveDefaultBranch: async () => "main", + listSkillManifests: async (owner) => { + if (owner === "slow") await drain(); + throw new GithubSkillError(`${owner} down`, "rate_limited", 403); + }, + fetchTextFile: async () => "", + downloadSkillDir: async () => [], + }; + const { errors } = await browseHub(client, [ + { repo: "slow/r", path: "" }, + { repo: "fast/r", path: "" }, + ]); + expect(errors.map((e) => e.repo)).toEqual(["slow/r", "fast/r"]); + }); + + it("keeps six SKILL.md reads in flight and refills a slot as soon as it frees", async () => { + const dirs = Array.from({ length: 13 }, (_, i) => `s${String(i).padStart(2, "0")}`); + // Each read waits until the test answers it. + const pending = new Map void>(); + let most = 0; + const client: SkillHubClient = { + resolveDefaultBranch: async () => "main", + listSkillManifests: async () => + dirs.map((dir) => ({ dir, manifestPath: `${dir}/SKILL.md` })), + fetchTextFile: (_o, _r, _ref, path) => { + const dir = path.split("/")[0]!; + return new Promise((resolve) => { + pending.set(dir, () => { + pending.delete(dir); + resolve(manifest(dir, "x")); + }); + most = Math.max(most, pending.size); + }); + }, + downloadSkillDir: async () => [], + }; + const answer = async (dir: string) => { + pending.get(dir)!(); + await drain(); + }; + const browsing = browseTap(client, { repo: "o/r", path: "" }); + await drain(); + expect([...pending.keys()]).toEqual(dirs.slice(0, 6)); + // s01 answers while s00 is still out: its slot goes to s06 at once. + // Fixed batches of six would hold s06…s12 until s00 answered. + await answer("s01"); + expect([...pending.keys()]).toEqual(["s00", "s02", "s03", "s04", "s05", "s06"]); + for (let d = [...pending.keys()].find((k) => k !== "s00"); d; d = [...pending.keys()].find((k) => k !== "s00")) { + await answer(d); + } + expect([...pending.keys()]).toEqual(["s00"]); + await answer("s00"); + const entries = await browsing; + expect(entries.map((e) => e.name)).toEqual(dirs); + expect(most).toBe(6); + }); +}); diff --git a/src/skills/hub/skill-hub-catalog.ts b/src/skills/hub/skill-hub-catalog.ts index 80ebded76..3464c82e2 100644 --- a/src/skills/hub/skill-hub-catalog.ts +++ b/src/skills/hub/skill-hub-catalog.ts @@ -42,6 +42,30 @@ export interface HubSkillEntry { /** Bounded fan-out so a large tap does not open hundreds of sockets. */ const MANIFEST_FETCH_CONCURRENCY = 6; +/** + * `task` over every item with at most `limit` running at once, the results + * in the items' order. A slot is refilled the moment it frees: unlike + * fixed batches, one slow file holds up one slot, not the five beside it. + */ +async function mapBounded( + items: readonly T[], + limit: number, + task: (item: T) => Promise, +): Promise { + const out = new Array(items.length); + let next = 0; + const worker = async (): Promise => { + while (next < items.length) { + const i = next++; + out[i] = await task(items[i]!); + } + }; + await Promise.all( + Array.from({ length: Math.min(limit, items.length) }, worker), + ); + return out; +} + /** * Candidate default branches tried before spending an API call on * `GET /repos/{owner}/{repo}`. The overwhelming majority of public repos @@ -92,38 +116,33 @@ export async function browseTap( tap.path, ); - const entries: HubSkillEntry[] = []; - for (let i = 0; i < manifests.length; i += MANIFEST_FETCH_CONCURRENCY) { - const batch = manifests.slice(i, i + MANIFEST_FETCH_CONCURRENCY); - const resolved = await Promise.all( - batch.map(async (m) => { - try { - const content = await client.fetchTextFile( - owner, - repo, - ref, - m.manifestPath, - ); - const { manifest } = parseSkillFile(content); - const entry: HubSkillEntry = { - identifier: formatSkillIdentifier(owner, repo, m.dir), - name: manifest.name, - description: manifest.description, - version: manifest.version, - repo: `${owner}/${repo}`, - dir: m.dir, - source: "github", - }; - return entry; - } catch { - return null; - } - }), - ); - for (const e of resolved) { - if (e) entries.push(e); - } - } + const resolved = await mapBounded( + manifests, + MANIFEST_FETCH_CONCURRENCY, + async (m): Promise => { + try { + const content = await client.fetchTextFile( + owner, + repo, + ref, + m.manifestPath, + ); + const { manifest } = parseSkillFile(content); + return { + identifier: formatSkillIdentifier(owner, repo, m.dir), + name: manifest.name, + description: manifest.description, + version: manifest.version, + repo: `${owner}/${repo}`, + dir: m.dir, + source: "github", + }; + } catch { + return null; + } + }, + ); + const entries = resolved.filter((e): e is HubSkillEntry => e !== null); entries.sort((a, b) => a.name.localeCompare(b.name)); return entries; } @@ -131,7 +150,9 @@ export async function browseTap( /** * Browse every tap and return the union, deduplicated by identifier. * Per-tap failures are collected into `errors` so one unreachable repo - * does not blank the whole list. + * does not blank the whole list. The taps are read side by side (each + * with its own bounded fan-out) and merged in the configured order, so + * the first tap still wins a duplicate and `errors` keeps that order. */ export async function browseHub( client: SkillHubClient, @@ -140,24 +161,31 @@ export async function browseHub( entries: HubSkillEntry[]; errors: Array<{ repo: string; error: string }>; }> { + const results = await Promise.all( + taps.map((tap) => + browseTap(client, tap).then( + (tapEntries) => ({ tapEntries, error: null }), + (err: unknown) => ({ + tapEntries: [] as HubSkillEntry[], + error: err instanceof Error ? err.message : String(err), + }), + ), + ), + ); const seen = new Set(); const entries: HubSkillEntry[] = []; const errors: Array<{ repo: string; error: string }> = []; - for (const tap of taps) { - try { - const tapEntries = await browseTap(client, tap); - for (const e of tapEntries) { - if (seen.has(e.identifier)) continue; - seen.add(e.identifier); - entries.push(e); - } - } catch (err) { - errors.push({ - repo: tap.repo, - error: err instanceof Error ? err.message : String(err), - }); + results.forEach(({ tapEntries, error }, i) => { + if (error !== null) { + errors.push({ repo: taps[i]!.repo, error }); + return; } - } + for (const e of tapEntries) { + if (seen.has(e.identifier)) continue; + seen.add(e.identifier); + entries.push(e); + } + }); entries.sort((a, b) => a.name.localeCompare(b.name)); return { entries, errors }; } diff --git a/src/tools/fusion/worker-result.test.ts b/src/tools/fusion/worker-result.test.ts index ddbd218ab..9d75a7914 100644 --- a/src/tools/fusion/worker-result.test.ts +++ b/src/tools/fusion/worker-result.test.ts @@ -758,6 +758,11 @@ describe("workerFailureHint", () => { ["openrouter HTTP 402: Payment Required", WORKER_HINT_QUOTA], ["HTTP 429: Too Many Requests", WORKER_HINT_QUOTA], ["insufficient credits on this API key", WORKER_HINT_QUOTA], + // AI/ML API's 403, as the agent words a billing refusal (item 40). + [ + "AI/ML API refused the request: you've run out of funds. Top up your balance with AI/ML API or pick another provider in the Providers panel.", + WORKER_HINT_QUOTA, + ], ])("recognises %s", (message, hint) => { expect(workerFailureHint(message)).toBe(hint); }); diff --git a/src/tools/fusion/worker-result.ts b/src/tools/fusion/worker-result.ts index c02180a57..e47008b6a 100644 --- a/src/tools/fusion/worker-result.ts +++ b/src/tools/fusion/worker-result.ts @@ -472,7 +472,7 @@ const SERVER_SATURATED = const SERVER_UNREACHABLE = /stopped answering GET \/slots|it is unreachable|accepted the connection and answered nothing/i; const CREDIT_OR_QUOTA = - /\b402\b|\b429\b|payment required|insufficient (?:credits?|funds|balance|quota)|out of credits?|quota (?:exceeded|exhausted)|exceeded (?:your|the) (?:current )?quota|rate[- ]limit|too many requests/i; + /\b402\b|\b429\b|payment required|insufficient (?:credits?|funds|balance|quota)|out of (?:credits?|funds)|quota (?:exceeded|exhausted)|exceeded (?:your|the) (?:current )?quota|rate[- ]limit|too many requests/i; /** * A short remediation for a worker failure the orchestrator (or the diff --git a/src/tracing/index.ts b/src/tracing/index.ts index 9637426fb..f5d47c1db 100644 --- a/src/tracing/index.ts +++ b/src/tracing/index.ts @@ -1,5 +1,6 @@ -export { StructuredLogger, stderrSink } from "./structured-logger.js"; +export { StructuredLogger, createStderrSink } from "./structured-logger.js"; export type { + LineWriter, LogContext, LogRecord, LogSink, diff --git a/src/tracing/structured-logger.ts b/src/tracing/structured-logger.ts index 0da33ec24..f581dc00c 100644 --- a/src/tracing/structured-logger.ts +++ b/src/tracing/structured-logger.ts @@ -61,10 +61,31 @@ export class StructuredLogger { } } -export function stderrSink(): LogSink { +/** Where `createStderrSink` writes: stderr, unless a test hands in another. */ +export type LineWriter = Pick; + +/** + * A sink that writes each record as one line: + * + * [2026-10-02T07:15:29.123Z] WARN message {"context":"as JSON"} + * + * The desktop app reads the level back out of this shape (its agent.log + * and Diagnostics label `atag serve`'s lines by it), so the shape is a + * contract, pinned by `tracing.test.ts`. + * + * A factory, and named like one, because of how it was once misused: + * `serve` handed the runtime the factory itself instead of the sink it + * returns, and every record "written" built a sink and threw it away. + * The parameter is what makes that a compile error: a `LogRecord` is + * not a stream, so the factory does not type-check where a `LogSink` + * is wanted. + */ +export function createStderrSink( + stream: LineWriter = process.stderr, +): LogSink { return (record) => { const context = record.context ? ` ${JSON.stringify(record.context)}` : ""; - process.stderr.write( + stream.write( `[${new Date(record.timestamp).toISOString()}] ${record.level.toUpperCase()} ${record.message}${context}\n`, ); }; diff --git a/src/tracing/trace/trace-bus.ts b/src/tracing/trace/trace-bus.ts index b94415318..701b02180 100644 --- a/src/tracing/trace/trace-bus.ts +++ b/src/tracing/trace/trace-bus.ts @@ -12,7 +12,7 @@ export interface TraceBus { /** * Fan-out helper: relays each event to every sink and swallows - * individual sink errors. Mirrors the contract of `stderrSink` in + * individual sink errors. Mirrors the contract of `StructuredLogger` in * `structured-logger.ts` — observability must never disrupt execution. */ export function createTraceBus(sinks: readonly TraceSink[]): TraceBus { diff --git a/src/tracing/trace/trace-event.ts b/src/tracing/trace/trace-event.ts index 610477bce..41cb376c9 100644 --- a/src/tracing/trace/trace-event.ts +++ b/src/tracing/trace/trace-event.ts @@ -1,5 +1,6 @@ import type { AgentLoopReason } from "../../agent/agent-loop.js"; import type { LlmFailureCategory } from "../../llm/reliability/index.js"; +import type { ProviderWaitCause } from "../../llm/reliability/provider-wait-cause.js"; import type { MemorySubcallKind } from "../../memory/health/index.js"; /** @@ -236,6 +237,21 @@ export interface TraceProviderWaiting extends TraceEventBase { maxWaitMs: number; nextRetryMs: number; reason: string; + /** + * What the failure was, as data (`classifyProviderWaitCause`): `kind`, + * plus `status` only when a response had one — the shape the + * `provider_waiting` SSE frame carries. Absent on rows recorded before + * the trace kept it. + */ + cause?: { kind: ProviderWaitCause["kind"]; status?: number }; + /** + * The errno-like code the transport left on the failure's `cause` + * chain (`ECONNREFUSED`, `ETIMEDOUT`, `ENOTFOUND`, `UND_ERR_SOCKET`, …). + * `reason` is often a bare `fetch failed`; this is what tells a local + * server that is not running from a network that is down. Absent when + * the transport left none. + */ + causeCode?: string; /** The provider link the turn waits on, when the loop knows it. */ providerId?: string; } @@ -592,6 +608,17 @@ export interface TraceError extends TraceEventBase { * new traces always carry it. */ category?: LlmFailureCategory; + /** + * For a `transport` failure: the errno-like code (`ECONNREFUSED`, + * `ETIMEDOUT`, `ENOTFOUND`, `ECONNRESET`, `UND_ERR_SOCKET`, …) the + * transport left on the `cause` chain of the error `message` came + * from. A `message` of `fetch failed` says only that no answer came; + * the code says why — a refused connection is a server that is not + * running. Absent for every other category (an abort can carry + * `ABORT_ERR`, which is not a network cause) and when the transport + * left no code. + */ + causeCode?: string; /** * Fallback-chain links that failed before the one `message` came from — * present only when the chain fell over, or the turn was already on a diff --git a/src/tracing/trace/trace-recorder.test.ts b/src/tracing/trace/trace-recorder.test.ts index baea030b4..a1d4e0913 100644 --- a/src/tracing/trace/trace-recorder.test.ts +++ b/src/tracing/trace/trace-recorder.test.ts @@ -3,6 +3,7 @@ import { describe, expect, it } from "vitest"; import type { AgentLoopEvent } from "../../agent/agent-loop.js"; import { attachFailedAttempts } from "../../llm/fallback/failed-attempts.js"; import { attachGenerationId } from "../../llm/provider/openai/generation-id.js"; +import { TransportError } from "../../llm/reliability/llm-failures.js"; import { createTraceRecorder } from "./trace-recorder.js"; import type { TraceEvent } from "./trace-event.js"; @@ -608,6 +609,188 @@ describe("createTraceRecorder", () => { }); }); + it("records the errno a transport failure left on its cause chain, on the one row", () => { + const { events, emit } = collector(); + const rec = createTraceRecorder({ sessionId: "s-errno", emit, now }); + rec.onAgentEvent({ type: "turn_started", turnIndex: 0 }); + rec.onAgentEvent({ type: "step_started", stepIndex: 0 }); + // The field case: a local server that is not running. The step + // executor wraps undici's `fetch failed` in a `TransportError`, and + // the errno survives only two links down the `cause` chain. + const failure = new TransportError( + "fetch failed", + null, + "http://127.0.0.1:8080/completion", + { + cause: new TypeError("fetch failed", { + cause: Object.assign( + new Error("connect ECONNREFUSED 127.0.0.1:8080"), + { code: "ECONNREFUSED" }, + ), + }), + }, + ); + rec.onAgentEvent({ + type: "llm_event", + event: { type: "step_error", error: failure, category: "transport" }, + }); + // The loop rethrows the very same object: still one row, and the + // row that stays is the one carrying the code. + rec.onAgentEvent({ + type: "loop_failed", + error: failure, + category: "transport", + }); + const errors = events.filter((e) => e.type === "error"); + expect(errors).toHaveLength(1); + expect(errors[0]).toMatchObject({ + message: "fetch failed", + category: "transport", + causeCode: "ECONNREFUSED", + stepIndex: 0, + }); + }); + + it("records the errno on a loop_failed that no step_error preceded", () => { + const { events, emit } = collector(); + const rec = createTraceRecorder({ sessionId: "s-errno-loop", emit, now }); + rec.onAgentEvent({ type: "turn_started", turnIndex: 0 }); + const error = new TypeError("fetch failed", { + cause: Object.assign(new Error("read ECONNRESET"), { + code: "ECONNRESET", + }), + }); + rec.onAgentEvent({ type: "loop_failed", error, category: "transport" }); + expect(events.find((e) => e.type === "error")).toMatchObject({ + message: "fetch failed", + causeCode: "ECONNRESET", + }); + }); + + it("leaves causeCode off a row with no errno, and off any category but transport", () => { + const { events, emit } = collector(); + const rec = createTraceRecorder({ sessionId: "s-no-errno", emit, now }); + rec.onAgentEvent({ type: "turn_started", turnIndex: 0 }); + rec.onAgentEvent({ type: "step_started", stepIndex: 0 }); + rec.onAgentEvent({ + type: "llm_event", + event: { + type: "step_error", + error: new TransportError("fetch failed", null, ""), + category: "transport", + }, + }); + rec.onAgentEvent({ type: "step_started", stepIndex: 1 }); + // An abort can carry Node's `ABORT_ERR`. It says the request was + // cut off here, nothing about the provider, so it is not recorded. + rec.onAgentEvent({ + type: "llm_event", + event: { + type: "step_error", + error: new Error("This operation was aborted", { + cause: Object.assign(new Error("The operation was aborted"), { + code: "ABORT_ERR", + }), + }), + category: "cancelled", + }, + }); + const errors = events.filter((e) => e.type === "error"); + expect(errors).toHaveLength(2); + expect(errors[0]).toMatchObject({ category: "transport" }); + expect(errors[0]).not.toHaveProperty("causeCode"); + expect(errors[1]).toMatchObject({ category: "cancelled" }); + expect(errors[1]).not.toHaveProperty("causeCode"); + }); + + it("records what a parked turn waits on: the cause, its errno and the link", () => { + const { events, emit } = collector(); + const rec = createTraceRecorder({ sessionId: "s-wait", emit, now }); + rec.onAgentEvent({ type: "turn_started", turnIndex: 0 }); + rec.onAgentEvent({ type: "step_started", stepIndex: 3 }); + rec.onAgentEvent({ + type: "provider_waiting", + attempt: 1, + waitedMs: 0, + maxWaitMs: 300_000, + nextRetryMs: 2_000, + reason: "fetch failed", + cause: { kind: "refused" }, + causeCode: "ECONNREFUSED", + providerId: "local-llama", + }); + expect(events.find((e) => e.type === "provider_waiting")).toEqual({ + type: "provider_waiting", + seq: 2, + sessionId: "s-wait", + ts: 1000, + turnIndex: 0, + stepIndex: 3, + attempt: 1, + waitedMs: 0, + maxWaitMs: 300_000, + nextRetryMs: 2_000, + reason: "fetch failed", + cause: { kind: "refused" }, + causeCode: "ECONNREFUSED", + providerId: "local-llama", + }); + }); + + it("records a wait without an errno as such, and a stream status of null as absent", () => { + const { events, emit } = collector(); + const rec = createTraceRecorder({ sessionId: "s-wait-2", emit, now }); + rec.onAgentEvent({ type: "turn_started", turnIndex: 0 }); + rec.onAgentEvent({ + type: "provider_waiting", + attempt: 1, + waitedMs: 0, + maxWaitMs: 300_000, + nextRetryMs: 2_000, + reason: "the provider ended the completion with an error", + cause: { kind: "stream_error", status: null }, + }); + rec.onAgentEvent({ + type: "provider_waiting", + attempt: 2, + waitedMs: 2_000, + maxWaitMs: 300_000, + nextRetryMs: 4_000, + reason: "openai provider 503: overloaded", + cause: { kind: "http", status: 503 }, + }); + // One shape for a reader, as on the SSE frame: `status` only when a + // response had one. + expect(events.filter((e) => e.type === "provider_waiting")).toEqual([ + { + type: "provider_waiting", + seq: 1, + sessionId: "s-wait-2", + ts: 1000, + turnIndex: 0, + attempt: 1, + waitedMs: 0, + maxWaitMs: 300_000, + nextRetryMs: 2_000, + reason: "the provider ended the completion with an error", + cause: { kind: "stream_error" }, + }, + { + type: "provider_waiting", + seq: 2, + sessionId: "s-wait-2", + ts: 1000, + turnIndex: 0, + attempt: 2, + waitedMs: 2_000, + maxWaitMs: 300_000, + nextRetryMs: 4_000, + reason: "openai provider 503: overloaded", + cause: { kind: "http", status: 503 }, + }, + ]); + }); + it("records a truncation with no cap on the wire without inventing one", () => { const { events, emit } = collector(); const rec = createTraceRecorder({ sessionId: "s-trunc-nocap", emit, now }); diff --git a/src/tracing/trace/trace-recorder.ts b/src/tracing/trace/trace-recorder.ts index 4c2e1f865..9078d1ddd 100644 --- a/src/tracing/trace/trace-recorder.ts +++ b/src/tracing/trace/trace-recorder.ts @@ -2,13 +2,21 @@ import type { AgentLoopEvent } from "../../agent/agent-loop.js"; import type { StepEvent } from "../../agent/step-executor.js"; import type { ToolCallPayload } from "../../llm/grammar/tool-call-grammar.js"; import type { LlmFailureCategory } from "../../llm/reliability/index.js"; +import type { ProviderWaitCause } from "../../llm/reliability/provider-wait-cause.js"; // The module, not the fallback barrel: it has no imports of its own, so // tracing does not pull the provider clients in behind it. import { summarizeFailedAttempts } from "../../llm/fallback/failed-attempts.js"; +// A leaf for the same reason: the errno walk, without the classifier +// (`classifyProviderWaitCause`) and the clients it would bring along. +import { readErrnoCode } from "../../llm/errno-code.js"; import { readGenerationId } from "../../llm/provider/openai/generation-id.js"; -import type { TraceError, TraceEvent } from "./trace-event.js"; +import type { + TraceError, + TraceEvent, + TraceProviderWaiting, +} from "./trace-event.js"; import type { TraceSink } from "./trace-bus.js"; /** The `generationId` field of an `error` row, or nothing. */ @@ -25,6 +33,35 @@ function fallbackFailuresOf( return failures.length > 0 ? { fallbackFailures: failures } : {}; } +/** + * The `causeCode` field of an `error` row, or nothing: the errno a + * `transport` failure left on its `cause` chain. Every other category + * is left without one — a cancelled request can carry `ABORT_ERR`, + * which says nothing about the provider. + */ +function causeCodeOf( + error: unknown, + category: LlmFailureCategory, +): Pick { + if (category !== "transport") return {}; + const code = readErrnoCode(error); + return code === undefined ? {} : { causeCode: code }; +} + +/** + * A wait cause as the `provider_waiting` row records it: `kind`, plus + * `status` only when it is a number. The event's `stream_error` may hold + * `status: null` (the provider reported none); the row leaves it out, as + * the SSE frame does, so a reader meets one shape. + */ +function waitCauseOf( + cause: ProviderWaitCause, +): NonNullable { + return "status" in cause && typeof cause.status === "number" + ? { kind: cause.kind, status: cause.status } + : { kind: cause.kind }; +} + export interface TraceRecorderOptions { sessionId: string; /** @@ -178,11 +215,15 @@ export function createTraceRecorder( // loop can fail outside `executeStep`, and the truncation-retry path // swaps in `truncationRetry.original` before emitting. The memo is // cleared when a new turn or step begins, so only the failure that - // just happened can suppress anything. + // just happened can suppress anything. `causeCode` is part of what is + // compared because it is part of the row: the same object reads the + // same code twice, so the rethrow is still dropped, while a different + // failure that happens to share the message is not. let lastStepError: { message: string; stack: string | undefined; category: LlmFailureCategory; + causeCode: string | undefined; stepIndex: number | null; } | null = null; @@ -301,11 +342,13 @@ export function createTraceRecorder( reason: inner.reason, }); return; - case "step_error": + case "step_error": { + const codeField = causeCodeOf(inner.error, inner.category); lastStepError = { message: inner.error.message, stack: inner.error.stack, category: inner.category, + causeCode: codeField.causeCode, stepIndex: currentStepIndex, }; push({ @@ -318,10 +361,12 @@ export function createTraceRecorder( message: inner.error.message, ...(inner.error.stack ? { stack: inner.error.stack } : {}), category: inner.category, + ...codeField, ...generationIdOf(inner.error), ...fallbackFailuresOf(inner.error), }); return; + } default: return; } @@ -546,6 +591,15 @@ export function createTraceRecorder( maxWaitMs: event.maxWaitMs, nextRetryMs: event.nextRetryMs, reason: event.reason, + // `reason` alone was a bare `fetch failed` in the field: a + // stopped local server read as a network outage. The kind + // and the errno travel with it. + ...(event.cause !== undefined + ? { cause: waitCauseOf(event.cause) } + : {}), + ...(event.causeCode !== undefined + ? { causeCode: event.causeCode } + : {}), ...(event.providerId !== undefined ? { providerId: event.providerId } : {}), @@ -627,11 +681,13 @@ export function createTraceRecorder( }); return; case "loop_failed": { + const codeField = causeCodeOf(event.error, event.category); const duplicate = lastStepError !== null && lastStepError.message === event.error.message && lastStepError.stack === event.error.stack && lastStepError.category === event.category && + lastStepError.causeCode === codeField.causeCode && lastStepError.stepIndex === currentStepIndex; lastStepError = null; if (duplicate) return; @@ -647,6 +703,7 @@ export function createTraceRecorder( message: event.error.message, ...(event.error.stack ? { stack: event.error.stack } : {}), category: event.category, + ...codeField, ...generationIdOf(event.error), ...fallbackFailuresOf(event.error), }); diff --git a/src/tracing/tracing.test.ts b/src/tracing/tracing.test.ts index 9d57a224a..0db2219a8 100644 --- a/src/tracing/tracing.test.ts +++ b/src/tracing/tracing.test.ts @@ -1,5 +1,5 @@ -import { describe, it, expect } from "vitest"; -import { StructuredLogger } from "./structured-logger.js"; +import { describe, it, expect, vi } from "vitest"; +import { StructuredLogger, createStderrSink } from "./structured-logger.js"; import { MetricsCollector } from "./metrics-collector.js"; import { AgentMetrics, METRIC_NAMES } from "./agent-metrics.js"; import { createLogNdjsonSink, createMetricNdjsonSink } from "./ndjson-sinks.js"; @@ -34,6 +34,50 @@ describe("StructuredLogger", () => { }); }); +describe("createStderrSink", () => { + // The desktop app labels each line of `atag serve`'s stderr by the + // level in it (desktop/main/agent-output.ts), so this shape is what it + // reads: `[ISO time] LEVEL message`, then the context as one JSON + // object, one record per line. + it("writes one line per record: [time] LEVEL message {context}", () => { + const lines: string[] = []; + const sink = createStderrSink({ + write: (chunk: string | Uint8Array) => { + lines.push(String(chunk)); + return true; + }, + }); + const logger = new StructuredLogger({ level: "debug", sinks: [sink] }); + logger.debug("d"); + logger.info("i"); + logger.warn("w", { causeCode: "ECONNREFUSED" }); + logger.error("e"); + expect(lines.map((l) => l.replace(/^\[[^\]]+\] /, ""))).toEqual([ + "DEBUG d\n", + "INFO i\n", + 'WARN w {"causeCode":"ECONNREFUSED"}\n', + "ERROR e\n", + ]); + for (const line of lines) { + expect(line).toMatch(/^\[\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}\.\d{3}Z\] /); + } + }); + + it("writes to stderr by default", () => { + const stderr = vi + .spyOn(process.stderr, "write") + .mockImplementation(() => true); + try { + createStderrSink()({ level: "warn", message: "to stderr", timestamp: 0 }); + expect(stderr).toHaveBeenCalledWith( + "[1970-01-01T00:00:00.000Z] WARN to stderr\n", + ); + } finally { + stderr.mockRestore(); + } + }); +}); + describe("MetricsCollector + AgentMetrics", () => { it("records canonical step/llm/tool metric names", () => { const samples: Array<{ diff --git a/src/tui/format-provider-outage.ts b/src/tui/format-provider-outage.ts index 100e6f944..78b8b4eaf 100644 --- a/src/tui/format-provider-outage.ts +++ b/src/tui/format-provider-outage.ts @@ -34,6 +34,8 @@ const HUMANISED_REASONS: ReadonlyArray = [ */ function describeCause(cause: ProviderWaitCause): string | null { switch (cause.kind) { + case "billing": + return "the account is out of funds"; case "refused": return "connection refused"; case "dropped": diff --git a/src/tui/local-models/local-models-orchestrator.ts b/src/tui/local-models/local-models-orchestrator.ts index 1acec3559..6c74b0d79 100644 --- a/src/tui/local-models/local-models-orchestrator.ts +++ b/src/tui/local-models/local-models-orchestrator.ts @@ -8,7 +8,6 @@ import { import { hasOtherLiveSessions } from "../../local-llm/session-registry.js"; import { apiKeyForUrl } from "../../local-llm/managed-api-key.js"; import { - checkForBackendUpdate, DEFAULT_EMBEDDING_MODEL_ID, DEFAULT_LLAMACPP_MODEL_ID, downloadBackend, @@ -47,6 +46,8 @@ import { buildCustomModelDef, listLocalModels, listVulkanDevices, + AUTO_UPDATE_RECHECK_MS, + checkForBackendUpdateForPanel, maybeAutoUpdateBackend, probeNvidiaVramMiB, readBackendVersion, @@ -412,7 +413,9 @@ export class LocalModelsOrchestrator { let updateAvailable: boolean | null = null; let latestTag: string | null = null; try { - const u = await checkForBackendUpdate(dataDir); + // An update shown here is one the next start asks for too: it + // drops the record a start would otherwise trust for hours. + const u = await checkForBackendUpdateForPanel(dataDir); updateAvailable = u.updateAvailable; latestTag = u.latestTag; } catch { @@ -2587,6 +2590,9 @@ export class LocalModelsOrchestrator { const result = await maybeAutoUpdateBackend(dataDir, { enabled: getConfig().localModels.managed.autoUpdate, keepDaemonRunning: opts?.keepDaemonRunning, + // A check from the last few hours stands, whichever process made + // it (the desktop's `models start`, another TUI). + recheckAfterMs: AUTO_UPDATE_RECHECK_MS, // The zip is small (27-39 MB) but the link may not be. Without a // deadline a stalled-open connection pins the download for the // life of the process; the next start retries from scratch.