From 15b32efa765a17b2875ad221c3b934de5293204d Mon Sep 17 00:00:00 2001 From: Ovtcharov Date: Thu, 17 Sep 2026 10:49:08 -0700 Subject: [PATCH 1/5] fix(agent,tools): stop the loop reporting work it never did, and let it use the tools it has An agent run could end with "Task completed" having written nothing, and the caller had no way to tell that from a real answer. Six defects behind that, each found by running real tasks to completion and reading what the agent actually did: - the guard that re-prompts for a missing output file called a symbol nobody imported, and a blanket except swallowed the NameError -- it had never run - a repeated tool call ended the turn claiming success; it now asks for a different approach once, then says the work is unfinished - search_file could not find a file named exactly (ci.log, pyproject.toml): an extension allowlist hid it, and the agent read the empty result as an empty workspace - content search skipped .toml, .cfg, .ts and every extensionless file, so 'bump the version wherever it is declared' never saw pyproject.toml - an unpaired left a reasoning model's deliberation in the answer - the shell refused linters it already ran via 'python -m', and its hint claimed only read-only commands were allowed long after that stopped being true -- so the agent stopped trying to verify its own work Also adds the task-execution harness these were found with: real agent loop, isolated sandbox per attempt, correctness decided by a verifier's exit code and never blended with judged quality. Measured across three-run batches on the 27-task suite, shell refusals fell from 55-76% of calls to 6-10%, and accomplished rose from 19/27 to 26/27. Issues: #3941 #3942 #3943 #3945 #3946 #3957 #3967 #3971 --- .claude/skills/building-eval-dataset/SKILL.md | 198 +++ .claude/skills/driving-the-tui/SKILL.md | 62 + .github/workflows/build-installers.yml | 2 +- .github/workflows/claude-auth-canary.yml | 2 +- .github/workflows/claude-nightly-audit.yml | 4 +- .github/workflows/claude-run.yml | 4 +- .github/workflows/claude-security-audit.yml | 2 +- .../claude-weekly-doc-walkthrough.yml | 6 +- .github/workflows/claude.yml | 4 +- .github/workflows/lemonade-version-bump.yml | 2 +- .github/workflows/skill_audit.yml | 2 +- .gitignore | 28 + docs/releases/v0.15.3.mdx | 2 +- hub/agents/email/python/CHANGELOG.md | 9 - .../gaia_agent_email/answer_grounding.py | 19 +- .../email/python/packaging/gen_scorecard.py | 4 +- .../email/python/packaging/smoke_test.py | 3 +- .../python/tests/test_answer_grounding.py | 54 - package-lock.json | 20 +- package.json | 4 +- src/gaia/agents/base/agent.py | 302 +++- src/gaia/agents/base/verification.py | 99 ++ src/gaia/agents/tools/file_tools.py | 110 +- src/gaia/agents/tools/shell_tools.py | 206 ++- src/gaia/apps/example/webui/package.json | 2 +- src/gaia/apps/webui/package-lock.json | 212 +-- src/gaia/cli.py | 25 +- src/gaia/connectors/flow.py | 75 +- src/gaia/daemon/sidecars/spec.py | 71 +- src/gaia/electron/package.json | 2 +- src/gaia/eval/runner.py | 6 +- src/gaia/factory/dataset/__init__.py | 11 + src/gaia/factory/dataset/annotate.py | 397 +++++ src/gaia/factory/dataset/audit.py | 266 ++++ src/gaia/factory/dataset/axes.py | 164 ++ src/gaia/factory/dataset/build.py | 1143 ++++++++++++++ src/gaia/factory/dataset/commits.py | 206 +++ src/gaia/factory/dataset/extract.py | 498 ++++++ src/gaia/factory/dataset/harness.py | 358 +++++ src/gaia/factory/dataset/partition.py | 171 +++ src/gaia/factory/dataset/scrub.py | 320 ++++ src/gaia/factory/dataset/select.py | 145 ++ src/gaia/factory/dataset/split_report.py | 321 ++++ src/gaia/factory/dataset/usecase_split.py | 389 +++++ src/gaia/factory/dataset/verifiers.py | 301 ++++ src/gaia/factory/dataset/verify.py | 220 +++ src/gaia/factory/harvest/inventory.py | 293 ++++ src/gaia/factory/harvest/reader.py | 9 + src/gaia/factory/harvest/report.py | 12 + src/gaia/factory/harvest/systems.py | 1360 +++++++++++++++++ src/gaia/factory/harvest/task_tables.py | 150 ++ src/gaia/factory/harvest/tasks.py | 371 +++++ src/gaia/factory/harvest/use_cases.py | 793 ++++++++++ src/gaia/factory/harvest/usecase_report.py | 1244 +++++++++++++++ src/gaia/factory/tasks/__init__.py | 4 + src/gaia/factory/tasks/_child.py | 131 ++ src/gaia/factory/tasks/catalogue.py | 216 +++ src/gaia/factory/tasks/claude_code.py | 204 +++ src/gaia/factory/tasks/corpus_reference.py | 274 ++++ src/gaia/factory/tasks/enterprise.py | 483 ++++++ src/gaia/factory/tasks/extract_slides.py | 144 ++ src/gaia/factory/tasks/gateway_proxy.py | 187 +++ src/gaia/factory/tasks/master_deck.py | 122 ++ src/gaia/factory/tasks/quality.py | 280 ++++ src/gaia/factory/tasks/report.py | 857 +++++++++++ src/gaia/factory/tasks/routing.py | 163 ++ src/gaia/factory/tasks/runner.py | 629 ++++++++ src/gaia/factory/tasks/slides.py | 1093 +++++++++++++ src/gaia/factory/tasks/suite.py | 460 ++++++ src/gaia/factory/tasks/suite_extra.py | 866 +++++++++++ src/gaia/factory/tasks/suite_judged.py | 206 +++ src/gaia/installer/lemonade_installer.py | 6 - src/gaia/llm/lemonade_client.py | 91 +- src/gaia/llm/lemonade_embedded.py | 7 - src/gaia/mcp/client/transports/http.py | 7 - .../integration/test_chat_rag_pdf_chat_e2e.py | 6 +- tests/integration/test_chat_rag_pdf_e2e.py | 6 +- .../test_lemonade_embeddable_assets.py | 4 +- .../test_lemonade_embedded_lifecycle.py | 4 +- .../agents/test_agent_source_invariants.py | 39 +- tests/unit/agents/test_default_max_steps.py | 66 +- tests/unit/agents/test_idle_turn_guard.py | 48 + tests/unit/agents/test_loop_break_truthful.py | 46 +- .../unit/agents/test_output_guards_in_loop.py | 202 +++ .../agents/test_python_console_scripts.py | 144 ++ .../agents/test_reasoning_block_stripping.py | 95 ++ .../test_search_file_content_coverage.py | 100 ++ .../agents/test_search_file_extensions.py | 58 + .../agents/test_search_tool_disambiguation.py | 65 + .../unit/agents/test_unwritten_claims_3938.py | 94 ++ tests/unit/agents/test_verification_scope.py | 15 +- tests/unit/connectors/test_device_flow.py | 90 -- .../test_oauth_error_classification.py | 55 - tests/unit/factory/oracles_extra.py | 322 ++++ tests/unit/factory/test_dataset_audit.py | 260 ++++ tests/unit/factory/test_dataset_axes.py | 282 ++++ tests/unit/factory/test_dataset_extract.py | 557 +++++++ tests/unit/factory/test_dataset_gitignore.py | 126 ++ tests/unit/factory/test_dataset_scrub.py | 261 ++++ .../factory/test_dataset_usecase_split.py | 186 +++ tests/unit/factory/test_dataset_verifiers.py | 277 ++++ tests/unit/factory/test_harvest_systems.py | 610 ++++++++ tests/unit/factory/test_harvest_tasks.py | 158 ++ tests/unit/factory/test_runner_timeout.py | 63 + tests/unit/factory/test_task_suite.py | 320 ++++ tests/unit/test_daemon_dev_anchor_spec.py | 52 +- tests/unit/test_publish_workflow_contract.py | 5 +- .../unit/test_remote_disconnected_handling.py | 175 --- tests/unit/test_shell_guardrails.py | 95 ++ util/check_component_core_api.py | 49 +- util/verify_publish_pipeline.py | 7 - 111 files changed, 21134 insertions(+), 921 deletions(-) create mode 100644 .claude/skills/building-eval-dataset/SKILL.md create mode 100644 src/gaia/factory/dataset/__init__.py create mode 100644 src/gaia/factory/dataset/annotate.py create mode 100644 src/gaia/factory/dataset/audit.py create mode 100644 src/gaia/factory/dataset/axes.py create mode 100644 src/gaia/factory/dataset/build.py create mode 100644 src/gaia/factory/dataset/commits.py create mode 100644 src/gaia/factory/dataset/extract.py create mode 100644 src/gaia/factory/dataset/harness.py create mode 100644 src/gaia/factory/dataset/partition.py create mode 100644 src/gaia/factory/dataset/scrub.py create mode 100644 src/gaia/factory/dataset/select.py create mode 100644 src/gaia/factory/dataset/split_report.py create mode 100644 src/gaia/factory/dataset/usecase_split.py create mode 100644 src/gaia/factory/dataset/verifiers.py create mode 100644 src/gaia/factory/dataset/verify.py create mode 100644 src/gaia/factory/harvest/inventory.py create mode 100644 src/gaia/factory/harvest/systems.py create mode 100644 src/gaia/factory/harvest/task_tables.py create mode 100644 src/gaia/factory/harvest/tasks.py create mode 100644 src/gaia/factory/harvest/use_cases.py create mode 100644 src/gaia/factory/harvest/usecase_report.py create mode 100644 src/gaia/factory/tasks/__init__.py create mode 100644 src/gaia/factory/tasks/_child.py create mode 100644 src/gaia/factory/tasks/catalogue.py create mode 100644 src/gaia/factory/tasks/claude_code.py create mode 100644 src/gaia/factory/tasks/corpus_reference.py create mode 100644 src/gaia/factory/tasks/enterprise.py create mode 100644 src/gaia/factory/tasks/extract_slides.py create mode 100644 src/gaia/factory/tasks/gateway_proxy.py create mode 100644 src/gaia/factory/tasks/master_deck.py create mode 100644 src/gaia/factory/tasks/quality.py create mode 100644 src/gaia/factory/tasks/report.py create mode 100644 src/gaia/factory/tasks/routing.py create mode 100644 src/gaia/factory/tasks/runner.py create mode 100644 src/gaia/factory/tasks/slides.py create mode 100644 src/gaia/factory/tasks/suite.py create mode 100644 src/gaia/factory/tasks/suite_extra.py create mode 100644 src/gaia/factory/tasks/suite_judged.py create mode 100644 tests/unit/agents/test_idle_turn_guard.py create mode 100644 tests/unit/agents/test_output_guards_in_loop.py create mode 100644 tests/unit/agents/test_python_console_scripts.py create mode 100644 tests/unit/agents/test_reasoning_block_stripping.py create mode 100644 tests/unit/agents/test_search_file_content_coverage.py create mode 100644 tests/unit/agents/test_search_file_extensions.py create mode 100644 tests/unit/agents/test_search_tool_disambiguation.py create mode 100644 tests/unit/agents/test_unwritten_claims_3938.py create mode 100644 tests/unit/factory/oracles_extra.py create mode 100644 tests/unit/factory/test_dataset_audit.py create mode 100644 tests/unit/factory/test_dataset_axes.py create mode 100644 tests/unit/factory/test_dataset_extract.py create mode 100644 tests/unit/factory/test_dataset_gitignore.py create mode 100644 tests/unit/factory/test_dataset_scrub.py create mode 100644 tests/unit/factory/test_dataset_usecase_split.py create mode 100644 tests/unit/factory/test_dataset_verifiers.py create mode 100644 tests/unit/factory/test_harvest_systems.py create mode 100644 tests/unit/factory/test_harvest_tasks.py create mode 100644 tests/unit/factory/test_runner_timeout.py create mode 100644 tests/unit/factory/test_task_suite.py delete mode 100644 tests/unit/test_remote_disconnected_handling.py diff --git a/.claude/skills/building-eval-dataset/SKILL.md b/.claude/skills/building-eval-dataset/SKILL.md new file mode 100644 index 0000000000..a6d57f18cf --- /dev/null +++ b/.claude/skills/building-eval-dataset/SKILL.md @@ -0,0 +1,198 @@ +--- +name: building-eval-dataset +description: Build or regenerate the step-level agentic eval dataset from Claude Code session transcripts, and run a candidate agent harness against it. Use when asked to generate the eval dataset, label sessions by use-case, evaluate a harness or the GAIA agent step by step, or reproduce the dataset on another machine. +--- + +# Building and running the step-level eval dataset + +Turns a local Claude Code transcript corpus into **decision-point records** — one per +moment the agent had to choose its next action — and grades a candidate harness against +them. + +The parsing is deterministic Python (`gaia.factory.dataset`). **An LLM is used only where +judgement is genuinely required, and it runs as a Claude Code subagent — never as a direct +API call.** There is no `ANTHROPIC_API_KEY` path and none is needed. + +## Privacy — before anything else + +Transcripts contain absolute paths, usernames, branch names, repository content, and +whatever was pasted into a prompt. `amd/gaia` is public. + +- Everything derived goes to `~/.gaia/cache/factory/`. **Never** into a repo. +- Never paste a record, excerpt, or sample into an issue, PR, commit, or doc. +- Report aggregates only. +- The dataset is **private-tier even after scrubbing** — a regex cannot remove arbitrary + proper nouns. + +Three defences exist and all three are automatic: the output path is outside every repo, +`build.py` refuses to write into a git working tree, and `.gitignore` plus +`tests/unit/factory/test_dataset_gitignore.py` cover the rest. + +## Never run the agent under test from this repo + +This skill's own description tells a reader it is being evaluated, and subagents see skill +descriptions. Start the agent under test from a neutral directory with no `.claude/` above +it — not this repo, not the experiments directory. A score from a run that could see this +file is directional at best. + +--- + +## Run it + +```bash +cd +export PYTHONPATH=$(pwd)/src # the editable install may resolve to another worktree +``` + +### 1. Extract — deterministic, no LLM + +```bash +python -m gaia.factory.harvest.scan +``` + +Writes `traces.jsonl`, `intents.jsonl`, `stats.json` to `~/.gaia/cache/factory/`. + +**Optional but recommended:** freeze a copy (`cp -r ~/.gaia/cache/factory/*.jsonl +~/.gaia/cache/factory/snapshot-YYYY-MM-DD/`). The corpus grows while you work on it, and +Claude Code prunes old transcripts — 31 of one frozen 300-session list had already +vanished. Building from a frozen copy is the only way to reproduce a number later. + +### 2. Label — this is your job, not a script's + +`build.py` requires `labels.txt` and **will not run without it**. There is no labelling +script on purpose: assigning a use-case is a judgement call. + +1. Read `intents.jsonl`. Split into batches of ~70 sessions. +2. For each batch, dispatch a subagent with **the same taxonomy every time** (the 19 + use-cases in `§2` of the reference analysis: `pr_lifecycle`, `code_review`, + `doc_audit`, `doc_authoring`, `feature_impl`, `bug_fix`, `refactor`, `ci_debug`, + `security_fix`, `test_coverage`, `eval_quality`, `release_packaging`, `repo_ops`, + `research`, `live_validation`, `meta_agent_config`, `data_transform`, + `qa_conversational`, `other`). Ask for strict JSON: one primary use-case plus up to two + secondary tags per session. +3. Write `~/.gaia/cache/factory/labels.txt` as + `<8-char-session-prefix> `. + +**Classify from the first user message, never the auto-generated title.** The title is a +summary of what happened, so using it leaks the outcome into the label. + +Batches can run in parallel — give every one identical instructions or the taxonomy +drifts between batches and the use-case counts become meaningless. + +### 3. Build + +```bash +python -m gaia.factory.dataset.build --username +# --snapshot build from a frozen copy instead of the live scan +# --extra-name "…" repeatable; redact a colleague's name (a regex cannot infer these) +``` + +Writes `oracle/`, `pool/`, `blobs/`, and the manifests. The build **ends by running the +integrity audit and the leak sweep and aborts if either fails** — there is no override. + +### 4. Verify anytime + +```bash +python -m gaia.factory.dataset.audit # 16 integrity invariants +python -m gaia.factory.dataset.verify --username # privacy leak sweep +python -m gaia.factory.dataset.verify --spot-check 5 # records to hand-check +python -m pytest tests/unit/factory/ -q +``` + +--- + +## Running a harness against it + +Full contract, field reference and worked examples: +`~/.gaia/cache/factory/dataset/STATUS_AND_USAGE.md`. Read it before writing any grading +code. Four things matter most. + +**Use the shipped graders.** `gaia.factory.dataset.verifiers.grade`. Do not write your +own — the shipped ones are calibrated against the reference and the record's +`reference_checks` only make sense with them. + +**Smoke-test first.** Replay `record["action"]["calls"]` as if it were the candidate's +proposal. You must get **100% credited on `match_reference` and 100% on all four checks +over informative records**. Anything less means the integration is wrong, not the harness. +Do this before evaluating anything real. + +**`pool/` for development, `oracle/` only for a final number.** The split exists so a +harness tuned on one is measured on the other. Reading `oracle/` while iterating destroys +the only contamination control the dataset has. + +**Three rules that decide whether the numbers mean anything:** + +- **Two polarities, never merged.** 147 records are `avoid_reference`: the reference + action *failed*, so reproducing it scores zero and the harness wins by doing something + else that passes the checks. Timeouts are 24.8% of corpus failures — crediting a + reproduction would credit the worst habit in the data. +- **Only score a check where the reference passed it** (`checks_informative`). Ignoring + this costs a harness ~30% on `old_string_present` for the dataset's own staleness. +- **Never emit one aggregate score.** Report per axis and per use-case, with check + coverage beside every rate. + +### Evaluating GAIA specifically + +GAIA's tool names differ from Claude Code's — map them or `tool_selection` scores zero on +everything. `Read`/`Write`/`Edit` → `file_io`; `Grep`/`Glob` → `file_search` or +`code_index`; `Bash` → `shell`; `WebSearch`/`WebFetch` → `browser`. `Agent`/`Task` has no +GAIA equivalent, so the `delegation` axis (137 records) is unscoreable — report that as a +capability gap rather than a zero. + +--- + +## Honesty requirements + +Not optional; the evaluation is worthless without them. + +- **There is no ground truth for task success.** Nothing in a transcript says whether the + human's goal was met. `episode_outcome` is inference with stated evidence and + confidence — 1,688 of 3,296 records are `unknown` because 67% of sessions have exactly + one human turn and there is no reaction to read. Use it to rank; never quote it as fact. +- **The reference is what one strong harness did, not what was optimal.** Report + *agreement*, never *accuracy*. +- **Reasoning quality cannot be graded.** Extended thinking is encrypted corpus-wide — + zero of 31,935 decision points have recoverable thinking text, though 55% reasoned + invisibly. `reasoning.inferred` reconstructs the decision *context*, not the model's + thoughts. Do not build a reasoning judge on it. +- **This is a regression detector, not a release gate.** An auto-mined oracle is weaker + than a human-curated held-out set. Exact action overlap across the split is 0.87%, but + 65.9% of sessions share a repository branch across it. +- **One user, ~6 repositories, 37 days.** A portrait of one developer, not of developers. +- **Four use-case/partition groups sit below the sampling floor.** Do not report + per-use-case scores for them; `DATASHEET.md` names them. + +--- + +## The traps that will silently corrupt a rebuild + +Each cost real debugging time. `.claude/skills/analyzing-claude-sessions/SKILL.md` covers +five more for the extraction layer; these are specific to this dataset. + +1. **Subagents are not sessions.** A delegated run lives in + `/subagents/*.jsonl` and a `*/*.jsonl` glob misses it — they are a third + of all records here. +2. **One API response is several JSONL records**, one per content block. A message's text + and its `tool_use` land in *different* records, so a per-record scan finds zero + reasoning and concludes there is none. Group by `message.id`. +3. **Assign `step_index` when a message first dispatches a tool**, not when the message is + created — otherwise two decision points share an index and a record's own action shows + up inside its own `recent_steps`. +4. **`tools_available` must not be the transcript's tool union.** That is a union over the + whole run including the answer; it once made 4.6% of records advertise exactly the + tools used. Build it from environment facts instead. +5. **Hash `arg_hash` over the *scrubbed* arguments**, or the shipped record carries a hash + that does not describe its own contents. +6. **`Glob` takes globs, `Grep` takes regexes.** Compiling `**/*.go` with `re` rejects + every ordinary glob and makes the checker look 84% wrong when it is 100% right. +7. **Inline script bodies are not shell.** Splitting a `python -c "…"` or heredoc on + `;`/newline yields "binaries" like `open(p`. +8. **Keep held file content fresh.** After the agent edits a file, the last-read copy is + stale — 94% of `old_string` check failures were that, not bad arguments. + +## Verifying a rebuild + +Run the reference through its own graders. If it does not score 100% on informative +records, a *verifier* is wrong, not the data. That check found four separate bugs that +unit tests and the leak sweep both missed, because nothing failed — the graders simply +disagreed with reality. diff --git a/.claude/skills/driving-the-tui/SKILL.md b/.claude/skills/driving-the-tui/SKILL.md index 9445298dde..3d9356dd03 100644 --- a/.claude/skills/driving-the-tui/SKILL.md +++ b/.claude/skills/driving-the-tui/SKILL.md @@ -88,10 +88,72 @@ argument parsing (`error 0x80070002`), so keep it one word; and put the env vars new tab handed to an already-running Windows Terminal inherits *that* process's environment, not your shell's. +### Two things that silently break the `.bat` — both cost a bring-up cycle + +**Call the binary by absolute path.** `NoDefaultCurrentDirectoryInExePath=1` is set +in this environment and is inherited by every process you spawn, so `cmd.exe` will +*not* search the working directory for an executable. `cd /d ` then +`gaia-tui.exe` fails with `'gaia-tui.exe' is not recognized as an internal or +external command` even though the file is right there and `ls` shows it. Write +`"\gaia-tui.exe" --control-port 8815` instead. The `cd` is still worth +keeping so relative paths inside the TUI resolve. + +**Write the `.bat` from Python, not from bash.** A bash heredoc gives it LF line +endings, and `printf` is worse — it eats `\Users` as a `\U` unicode escape and +silently writes a corrupted path, which surfaces as `The filename, directory name, +or volume label syntax is incorrect.` Use: + +```python +pathlib.Path(bat).write_bytes(("\r\n".join(lines) + "\r\n").encode("ascii")) +``` + +with the path strings as raw literals. Verify by reading the file back before you +launch — a wrong path in a `.bat` only shows up as a terminal window that flashes +and dies, with no error you can see from the tool call. + +**Confirm the launch, don't assume it.** `Start-Process` returns success whether or +not the program inside the tab ever ran. Poll for `~/.gaia/tui/control.json` *and* +a `200` from `/control/v1/status`; the control file alone can be stale from an +earlier attempt. Delete it before relaunching. + To check colour rather than guess: `GET /control/v1/screen?format=ansi` and count `\x1b`. Zero on a frame that should be styled means the profile degraded — relaunch under `wt.exe`. +## Capture a trace — `--trace`, and the `=` is not optional + +Reading the screen cannot tell a correct answer from a broken one. #3576 is the proof: the +agent answered "Zero" to a question whose answer was 203, and answered "that directory +doesn't exist" about a directory that does — both rendered exactly like a good answer. +`--trace` records every agent event, **tool calls with their arguments**, results and +errors, as JSONL. + +```bash +gaia-tui.exe --control-port 8815 --trace # ~/.gaia/traces/-.jsonl +gaia-tui.exe --control-port 8815 --trace=C:\path\to\run.jsonl # explicit +``` + +**`--trace ` with a space does not work and never will.** The flag carries a +`NoOptDefVal` so that bare `--trace` parses, and pflag will not then attach a spaced value — +your path becomes a positional argument and is read as a command name. The TUI catches this +and prints the fix, so trust the error rather than re-deriving it: + +> `Error: --trace takes its path attached, not spaced: write --trace=` + +Writing the launch `.bat`, build that argument by **concatenation, not an f-string** — +`f'... --trace={PATH}'` makes Python parse `\Users` as a `\U` escape and the run dies with +`The filename, directory name, or volume label syntax is incorrect.` Use +`'... --trace=' + TRACE` with `TRACE` as a raw literal. + +What the trace does **not** hold: prompt size and token accounting, which live only in the +agent's own recorder behind `GAIA_TURN_LOG`. And note the turn recorder is off by default +and its `ok` flag is derived from the tool's own status — a tool that returns +`status: "success"` with zero results records as green. Only the trace's arguments and +result payload show the difference. + +**The file holds whatever the agent read** — file contents, shell output, email. Review +before sharing. + ## Endpoints `/control/v1/` — `status` · `screen` · `keys` · **`mouse`** · `text` · **`wait`** · diff --git a/.github/workflows/build-installers.yml b/.github/workflows/build-installers.yml index 2b4eacf3e1..434630fc50 100644 --- a/.github/workflows/build-installers.yml +++ b/.github/workflows/build-installers.yml @@ -533,7 +533,7 @@ jobs: # evaluated AFTER `if:`, which is why we can't inline the secret # forwarding here. if: matrix.platform == 'windows' && env.SIGNPATH_API_TOKEN != '' && env.SIGNPATH_ORG_ID != '' - uses: signpath/github-action-submit-signing-request@v3 + uses: signpath/github-action-submit-signing-request@v2 with: api-token: ${{ env.SIGNPATH_API_TOKEN }} organization-id: ${{ env.SIGNPATH_ORG_ID }} diff --git a/.github/workflows/claude-auth-canary.yml b/.github/workflows/claude-auth-canary.yml index 8a52e28a1d..34709eb52e 100644 --- a/.github/workflows/claude-auth-canary.yml +++ b/.github/workflows/claude-auth-canary.yml @@ -42,7 +42,7 @@ jobs: # It now defaults to the same model the real jobs use (claude-opus-5) so the # scheduled run exercises the actual path, not a cheaper proxy that could stay # green while every real job fails to resolve its model. - uses: anthropics/claude-code-action@51db78a4b844e144f8d02425cb280435c04a3474 # v1.0.224 + uses: anthropics/claude-code-action@d75b94d5ad426cb8546e6628b6f5f19b84e5cce1 # v1.0.216 with: anthropic_api_key: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN == '' && secrets.ANTHROPIC_API_KEY || '' }} claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }} diff --git a/.github/workflows/claude-nightly-audit.yml b/.github/workflows/claude-nightly-audit.yml index a971e643dd..8c1b4a1e7f 100644 --- a/.github/workflows/claude-nightly-audit.yml +++ b/.github/workflows/claude-nightly-audit.yml @@ -262,7 +262,7 @@ jobs: - name: Run Claude (${{ matrix.dimension }} lens) id: claude continue-on-error: true - uses: anthropics/claude-code-action@51db78a4b844e144f8d02425cb280435c04a3474 # v1.0.224 + uses: anthropics/claude-code-action@d75b94d5ad426cb8546e6628b6f5f19b84e5cce1 # v1.0.216 with: # Prefer subscription OAuth, fall back to API key — see claude.yml "Authentication". anthropic_api_key: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN == '' && secrets.ANTHROPIC_API_KEY || '' }} @@ -560,7 +560,7 @@ jobs: - name: Synthesize and file one issue per defect id: claude continue-on-error: true - uses: anthropics/claude-code-action@51db78a4b844e144f8d02425cb280435c04a3474 # v1.0.224 + uses: anthropics/claude-code-action@d75b94d5ad426cb8546e6628b6f5f19b84e5cce1 # v1.0.216 with: anthropic_api_key: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN == '' && secrets.ANTHROPIC_API_KEY || '' }} claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }} diff --git a/.github/workflows/claude-run.yml b/.github/workflows/claude-run.yml index b6f3df44d4..67c6fcbe82 100644 --- a/.github/workflows/claude-run.yml +++ b/.github/workflows/claude-run.yml @@ -330,7 +330,7 @@ jobs: # not red-X a contributor's PR CI. Visibility comes from the verification step. - name: Run Claude (attempt 1) id: a1 - uses: anthropics/claude-code-action@51db78a4b844e144f8d02425cb280435c04a3474 # v1.0.224 + uses: anthropics/claude-code-action@d75b94d5ad426cb8546e6628b6f5f19b84e5cce1 # v1.0.216 continue-on-error: true with: # Prefer subscription OAuth, fall back to API key — see claude.yml "Authentication". @@ -353,7 +353,7 @@ jobs: - name: Run Claude (attempt 2 — retry install-phase crash) id: a2 if: "!cancelled() && steps.a1.outcome == 'failure' && steps.a1.outputs.execution_file == ''" - uses: anthropics/claude-code-action@51db78a4b844e144f8d02425cb280435c04a3474 # v1.0.224 + uses: anthropics/claude-code-action@d75b94d5ad426cb8546e6628b6f5f19b84e5cce1 # v1.0.216 continue-on-error: true with: anthropic_api_key: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN == '' && secrets.ANTHROPIC_API_KEY || '' }} diff --git a/.github/workflows/claude-security-audit.yml b/.github/workflows/claude-security-audit.yml index a48d97fa63..8e5f328dcc 100644 --- a/.github/workflows/claude-security-audit.yml +++ b/.github/workflows/claude-security-audit.yml @@ -193,7 +193,7 @@ jobs: - name: Run Claude (${{ matrix.lens }} lens) id: claude continue-on-error: true - uses: anthropics/claude-code-action@51db78a4b844e144f8d02425cb280435c04a3474 # v1.0.224 + uses: anthropics/claude-code-action@d75b94d5ad426cb8546e6628b6f5f19b84e5cce1 # v1.0.216 with: anthropic_api_key: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN == '' && secrets.ANTHROPIC_API_KEY || '' }} claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }} diff --git a/.github/workflows/claude-weekly-doc-walkthrough.yml b/.github/workflows/claude-weekly-doc-walkthrough.yml index 15079aa19d..b12f889fe3 100644 --- a/.github/workflows/claude-weekly-doc-walkthrough.yml +++ b/.github/workflows/claude-weekly-doc-walkthrough.yml @@ -396,7 +396,7 @@ jobs: - name: "Executor: walk ${{ matrix.doc.path }}" id: executor continue-on-error: true - uses: anthropics/claude-code-action@51db78a4b844e144f8d02425cb280435c04a3474 # v1.0.224 + uses: anthropics/claude-code-action@d75b94d5ad426cb8546e6628b6f5f19b84e5cce1 # v1.0.216 with: anthropic_api_key: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN == '' && secrets.ANTHROPIC_API_KEY || '' }} claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }} @@ -535,7 +535,7 @@ jobs: id: judge if: always() continue-on-error: true - uses: anthropics/claude-code-action@51db78a4b844e144f8d02425cb280435c04a3474 # v1.0.224 + uses: anthropics/claude-code-action@d75b94d5ad426cb8546e6628b6f5f19b84e5cce1 # v1.0.216 with: anthropic_api_key: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN == '' && secrets.ANTHROPIC_API_KEY || '' }} claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }} @@ -831,7 +831,7 @@ jobs: - name: Synthesize and file one issue per defect id: claude continue-on-error: true - uses: anthropics/claude-code-action@51db78a4b844e144f8d02425cb280435c04a3474 # v1.0.224 + uses: anthropics/claude-code-action@d75b94d5ad426cb8546e6628b6f5f19b84e5cce1 # v1.0.216 with: anthropic_api_key: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN == '' && secrets.ANTHROPIC_API_KEY || '' }} claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }} diff --git a/.github/workflows/claude.yml b/.github/workflows/claude.yml index fefcd532e9..b70b6d172c 100644 --- a/.github/workflows/claude.yml +++ b/.github/workflows/claude.yml @@ -1065,7 +1065,7 @@ jobs: - name: Attempt auto-fix id: claude # referenced by the Notify step's execution_file check below - uses: anthropics/claude-code-action@51db78a4b844e144f8d02425cb280435c04a3474 # v1.0.224 + uses: anthropics/claude-code-action@d75b94d5ad426cb8546e6628b6f5f19b84e5cce1 # v1.0.216 with: # Prefer subscription OAuth, fall back to API key — see "Authentication" in the file header. anthropic_api_key: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN == '' && secrets.ANTHROPIC_API_KEY || '' }} @@ -1540,7 +1540,7 @@ jobs: # continue-on-error: a rate-limit / expired-token / outage failure here is # caught by the "Verify and validate release notes" step (which fails the job) # and the "Notify on failure" step opens a tracking issue. - uses: anthropics/claude-code-action@51db78a4b844e144f8d02425cb280435c04a3474 # v1.0.224 + uses: anthropics/claude-code-action@d75b94d5ad426cb8546e6628b6f5f19b84e5cce1 # v1.0.216 continue-on-error: true with: # Prefer subscription OAuth, fall back to API key — see "Authentication" in the file header. diff --git a/.github/workflows/lemonade-version-bump.yml b/.github/workflows/lemonade-version-bump.yml index 645f33abc5..d7402e82c6 100644 --- a/.github/workflows/lemonade-version-bump.yml +++ b/.github/workflows/lemonade-version-bump.yml @@ -137,7 +137,7 @@ jobs: - name: Open Lemonade bump PR if: steps.detect.outputs.should_bump == 'true' - uses: anthropics/claude-code-action@51db78a4b844e144f8d02425cb280435c04a3474 # v1.0.224 + uses: anthropics/claude-code-action@d75b94d5ad426cb8546e6628b6f5f19b84e5cce1 # v1.0.216 with: # Prefer subscription OAuth, fall back to API key — same wiring as claude.yml. anthropic_api_key: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN == '' && secrets.ANTHROPIC_API_KEY || '' }} diff --git a/.github/workflows/skill_audit.yml b/.github/workflows/skill_audit.yml index 619ffac3a4..fb7ccc664a 100644 --- a/.github/workflows/skill_audit.yml +++ b/.github/workflows/skill_audit.yml @@ -477,7 +477,7 @@ jobs: - name: Run Claude (skill-intent lens) id: claude continue-on-error: true # advisory: a crash here must not gate the merge - uses: anthropics/claude-code-action@51db78a4b844e144f8d02425cb280435c04a3474 # v1.0.224 + uses: anthropics/claude-code-action@d75b94d5ad426cb8546e6628b6f5f19b84e5cce1 # v1.0.216 with: anthropic_api_key: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN == '' && secrets.ANTHROPIC_API_KEY || '' }} claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }} diff --git a/.gitignore b/.gitignore index 186fd24da8..d025368fdd 100644 --- a/.gitignore +++ b/.gitignore @@ -303,3 +303,31 @@ tests/unit/eval/test_mcp_tool_reliability_scenarios.py .venv-freeze/ hub/agents/*/python/packaging/dist/ hub/agents/*/python/packaging/build/ +# Step-level eval dataset output (gaia.factory.dataset). Built from private +# transcripts: absolute paths, branch names, repository content, and anything ever +# pasted into a prompt. It belongs in ~/.gaia/cache/factory/dataset/ and must never +# reach this public repository. +# +# The builder already refuses to write into a git working tree and every record is +# scrubbed, but a regex cannot remove arbitrary proper nouns — the data is +# private-tier regardless. This is the last line of defence. +# +# The patterns are anchored (leading /) or match distinctive data filenames, so +# they do NOT hide src/gaia/factory/dataset/, which is builder *code* and stays +# tracked. tests/unit/factory/test_dataset_gitignore.py asserts exactly that. +/dataset/ +/oracle/ +/pool/ +/blobs/ +/factory-cache/ +/DATASHEET.md +/MANIFEST.json +/partition_audit.json +/contamination.json +/commit_recovery.json +/scrub_report.json +/tool_catalog.json +**/records.jsonl +**/traces.jsonl +**/intents.jsonl +**/blobs/*.txt diff --git a/docs/releases/v0.15.3.mdx b/docs/releases/v0.15.3.mdx index c401269821..670e2ed986 100644 --- a/docs/releases/v0.15.3.mdx +++ b/docs/releases/v0.15.3.mdx @@ -58,7 +58,7 @@ class ImageStoryAgent(Agent, SDToolsMixin, VLMToolsMixin): self.init_vlm() # 2 VLM tools ``` -This release shipped an SD Agent Playbook and an SD User Guide. Both pages were removed in v0.24.0 along with the standalone `sd` agent — image generation is now part of the flagship agent's tool surface. +See [SD Agent Playbook](https://amd-gaia.ai/docs/playbooks/sd-agent) for complete tutorial, and [SD User Guide](https://amd-gaia.ai/docs/guides/sd) for CLI reference. ### SDToolsMixin: Stable Diffusion SDK diff --git a/hub/agents/email/python/CHANGELOG.md b/hub/agents/email/python/CHANGELOG.md index eba56bdca7..9dbe8fe52b 100644 --- a/hub/agents/email/python/CHANGELOG.md +++ b/hub/agents/email/python/CHANGELOG.md @@ -131,15 +131,6 @@ contract version is tracked separately as anywhere in the query with no notion of quoting, so a colon word inside a quoted value was mistaken for an operator. It now only matches outside a quoted span; a real unsupported operator that follows one still raises. -- **A triage summary no longer reads as complete when one mailbox failed - during the scan (#3768).** When a provider outage skipped a connected - mailbox, the pre-scan envelope recorded it (`degraded`, `mailbox_errors`) - but the grounded fallback sentence quoted its counts unqualified — so an - urgent message in the skipped mailbox vanished behind a confident - all-covered answer. That sentence now carries the same "Outlook couldn't be - scanned (token expired); results below are from the rest of your mailboxes - only" caveat the suspicious-mail summary already used. A scan where every - mailbox answered reads exactly as before. ### Changed diff --git a/hub/agents/email/python/gaia_agent_email/answer_grounding.py b/hub/agents/email/python/gaia_agent_email/answer_grounding.py index c3dd7c000b..c4eaf689c1 100644 --- a/hub/agents/email/python/gaia_agent_email/answer_grounding.py +++ b/hub/agents/email/python/gaia_agent_email/answer_grounding.py @@ -601,9 +601,9 @@ def _mailbox_failure_caveat(mailbox_errors: Any) -> str: # ``degraded`` promised at least one entry here — a malformed # one is a broken envelope, not a normal case worth hiding. logger.warning( - "email agent: degraded scan envelope has a malformed " - "mailbox_errors entry (%r) — omitting it from the coverage " - "caveat", + "email agent: degraded check_suspicious_mail envelope has a " + "malformed mailbox_errors entry (%r) — omitting it from the " + "coverage caveat", entry, ) continue @@ -673,13 +673,7 @@ def _honest_suspicious_summary(envelope: Dict[str, Any]) -> str: def _honest_prescan_summary(envelope: Dict[str, Any]) -> str: """A minimal, always-grounded pre-scan sentence built straight from the envelope's own counts — the fallback used when the model's own framing - sentence contradicts that same envelope. - - A degraded scan gets the same ``_mailbox_failure_caveat`` as - ``_honest_suspicious_summary`` (#3768): these counts cover only the - mailboxes that answered, so stating them unqualified reads as - whole-account coverage when a mailbox was skipped. - """ + sentence contradicts that same envelope.""" urgent = len(envelope.get("urgent") or []) actionable = len(envelope.get("actionable") or []) needs_review = len(envelope.get("needs_review") or []) @@ -695,10 +689,7 @@ def _honest_prescan_summary(envelope: Dict[str, Any]) -> str: total_unread = envelope.get("total_unread") if isinstance(total_unread, int): coverage += f" · {total_unread} unread in your inbox" - lead = f"Here's your inbox pre-scan — {summary}. {coverage}." - if envelope.get("degraded"): - lead += " " + _mailbox_failure_caveat(envelope.get("mailbox_errors")) - return lead + return f"Here's your inbox pre-scan — {summary}. {coverage}." # --------------------------------------------------------------------------- diff --git a/hub/agents/email/python/packaging/gen_scorecard.py b/hub/agents/email/python/packaging/gen_scorecard.py index 85717e56a2..ff1645bf74 100644 --- a/hub/agents/email/python/packaging/gen_scorecard.py +++ b/hub/agents/email/python/packaging/gen_scorecard.py @@ -740,9 +740,7 @@ def _query_lemonade_version(base_url: str) -> str: try: with urllib.request.urlopen(url, timeout=5) as resp: data = json.loads(resp.read().decode("utf-8")) - except (urllib.error.URLError, ConnectionError, TimeoutError) as exc: - # ConnectionError covers http.client.RemoteDisconnected, which escapes - # URLError - see urllib.request.AbstractHTTPHandler.do_open. + except urllib.error.URLError as exc: raise RuntimeError( f"Cannot determine lemonade_version: health endpoint unreachable at " f"{url}: {exc}. " diff --git a/hub/agents/email/python/packaging/smoke_test.py b/hub/agents/email/python/packaging/smoke_test.py index a951a32481..62b56c174f 100644 --- a/hub/agents/email/python/packaging/smoke_test.py +++ b/hub/agents/email/python/packaging/smoke_test.py @@ -233,9 +233,8 @@ def check_triage() -> bool: return True log(f"FAIL: triage returned unexpected HTTP {e.code}: {detail[:500]}") return False - except (urllib.error.URLError, ConnectionError, TimeoutError) as e: + except (urllib.error.URLError, TimeoutError) as e: # Accepted + routed, then timed out waiting on an absent model. - # ConnectionError covers RemoteDisconnected, which escapes URLError. log(f"triage request timed out waiting on Lemonade (none reachable): {e}") log("triage check PASS (route accepted + routed the request)") return True diff --git a/hub/agents/email/python/tests/test_answer_grounding.py b/hub/agents/email/python/tests/test_answer_grounding.py index a7e4bbc0f0..db9f6bd047 100644 --- a/hub/agents/email/python/tests/test_answer_grounding.py +++ b/hub/agents/email/python/tests/test_answer_grounding.py @@ -49,7 +49,6 @@ from gaia_agent_email.agent import EmailTriageAgent, _SYSTEM_PROMPT # noqa: E402 from gaia_agent_email.answer_grounding import ( # noqa: E402 UNGROUNDED_SUCCESS_FALLBACK, - _honest_prescan_summary, decode_stray_unicode_escapes, find_attention_card_contradiction, find_fabricated_attendee_claim, @@ -594,59 +593,6 @@ def test_the_models_own_list_is_discarded_in_favor_of_the_rendered_one(self): assert out.startswith("Here's your inbox — 1 item needs attention.") -class TestHonestPrescanSummary: - """#3768 — a pre-scan that skipped a failed mailbox must say so in the - sentence the user reads, not only in the envelope's ``degraded`` flag. - """ - - def test_degraded_scan_names_the_failed_mailbox(self): - envelope = _prescan_envelope( - urgent=[{"message_id": "m1"}], - degraded=True, - mailbox_errors=[{"mailbox": "microsoft", "error": "token expired"}], - ) - summary = _honest_prescan_summary(envelope) - assert "Outlook" in summary, ( - f"must name the mailbox that failed (provider_label('microsoft') " - f"== 'Outlook'), got: {summary!r}" - ) - assert "couldn't be scanned" in summary - # A user must not be able to read this as whole-account coverage. - assert "only" in summary - - def test_non_degraded_scan_is_byte_identical_to_the_plain_summary(self): - envelope = _prescan_envelope(urgent=[{"message_id": "m1"}], scanned=25) - assert _honest_prescan_summary(envelope) == ( - "Here's your inbox pre-scan — 1 urgent. " - "25 messages scanned · 100 unread in your inbox." - ) - - def test_degraded_with_unusable_mailbox_errors_still_qualifies_the_counts(self): - for errors in ([], [{"error": "no mailbox name"}], None): - summary = _honest_prescan_summary( - _prescan_envelope(degraded=True, mailbox_errors=errors) - ) - assert "Part of your mail could not be scanned this time." in summary, ( - f"a degraded envelope with mailbox_errors={errors!r} must still " - f"carry the generic caveat, got: {summary!r}" - ) - - def test_grounded_replacement_answer_carries_the_caveat(self): - # End-to-end through the path that actually reaches the user: the - # model's contradicted claim is replaced by this summary. - envelope = _prescan_envelope( - urgent=[{"message_id": "m1"}], - degraded=True, - mailbox_errors=[{"mailbox": "microsoft", "error": "token expired"}], - ) - result = { - "result": "No urgent items today.", - "conversation": [_tool_entry("pre_scan_inbox", envelope)], - } - out = ground_final_answer(result) - assert "Outlook couldn't be scanned (token expired)" in out["result"] - - class TestNormalizeTriageList: def test_ordinary_prose_is_untouched(self): prose = "We looked at 5. Then we stopped." diff --git a/package-lock.json b/package-lock.json index dd0969a827..7978c6fd80 100644 --- a/package-lock.json +++ b/package-lock.json @@ -14,11 +14,11 @@ "dependencies": { "@prisma/client": "^7.10.0", "prisma": "^7.10.0", - "zod": "^4.6.5" + "zod": "^4.5.4" }, "devDependencies": { "cross-env": "^10.1.0", - "electron": "^44.3.0" + "electron": "^44.1.1" } }, "node_modules/@amd-gaia/electron": { @@ -3040,9 +3040,9 @@ } }, "node_modules/electron": { - "version": "44.3.0", - "resolved": "https://registry.npmjs.org/electron/-/electron-44.3.0.tgz", - "integrity": "sha512-St9EV7F2VtYaYWD2qaAjBwUgKxx39eJOUsUJ5+/1113sqbVfNqv4Dbm/W1rN7qmYSPa+mWwR6yr+b7MfgjgVfQ==", + "version": "44.1.1", + "resolved": "https://registry.npmjs.org/electron/-/electron-44.1.1.tgz", + "integrity": "sha512-N2WCq2sbOkqQgvXJYx2lS6UiO8bF+Yr67trDnS6JKa2WxTCRQsAjGl57SUtdW9h6r5PlduBFjIhxhgd3dzv1hg==", "dev": true, "license": "MIT", "dependencies": { @@ -7524,9 +7524,9 @@ } }, "node_modules/zod": { - "version": "4.6.5", - "resolved": "https://registry.npmjs.org/zod/-/zod-4.6.5.tgz", - "integrity": "sha512-v5l/aFXZQeai4awLbOpSoHecE9UiMrnfx75tEXLjNonXVARxQ5mOeipTjROUchszUNCqnE+hqAMujRsRHsut2Q==", + "version": "4.5.4", + "resolved": "https://registry.npmjs.org/zod/-/zod-4.5.4.tgz", + "integrity": "sha512-sC95tT5iHHH9gtpj6A81kh+NEaRAUFN+qlUPDUbRfOMvNf5QCBqsb3WgvnpVtK5Y+4UfA6KqufotuTvMGiTlsA==", "license": "MIT", "funding": { "url": "https://github.com/sponsors/colinhacks" @@ -7544,7 +7544,7 @@ "@electron-forge/cli": "^7.11.2", "@electron-forge/maker-squirrel": "^7.11.2", "@electron-forge/maker-zip": "^7.11.2", - "electron": "^44.3.0" + "electron": "^44.1.1" } }, "src/gaia/apps/example/webui/node_modules/@electron-forge/cli": { @@ -7902,7 +7902,7 @@ "node-fetch": "^3.3.2" }, "devDependencies": { - "electron": "^44.3.0" + "electron": "^44.1.1" }, "peerDependencies": { "electron": ">=28.0.0" diff --git a/package.json b/package.json index 0f73a45062..895c94066a 100644 --- a/package.json +++ b/package.json @@ -18,12 +18,12 @@ ], "devDependencies": { "cross-env": "^10.1.0", - "electron": "^44.3.0" + "electron": "^44.1.1" }, "dependencies": { "@prisma/client": "^7.10.0", "prisma": "^7.10.0", - "zod": "^4.6.5" + "zod": "^4.5.4" }, "overrides": { "tar": ">=7.5.8", diff --git a/src/gaia/agents/base/agent.py b/src/gaia/agents/base/agent.py index 270cb884d9..5483b68bd0 100644 --- a/src/gaia/agents/base/agent.py +++ b/src/gaia/agents/base/agent.py @@ -44,6 +44,8 @@ NOT_EXECUTED, build_verification_scope, check_was_executed, + missing_requested_outputs, + unwritten_claims, verification_check_label, ) @@ -51,8 +53,10 @@ from gaia.chat.sdk import AgentConfig, AgentSDK from gaia.llm.lemonade_client import ( DEFAULT_MODEL_NAME, + REMOTE_CTX_ASSUMPTION, budget_for_ctx, is_context_overflow_error, + is_local_host, profile_ctx_size, truncation_budget, ) @@ -104,12 +108,69 @@ def _skill_menu_description(description: str) -> str: CHUNK_TRUNCATION_THRESHOLD = 5000 CHUNK_TRUNCATION_SIZE = 2500 -# Global default for how many reasoning/tool steps an agent may take before it -# stops and reports progress. This is the single knob for the whole fleet: -# change DEFAULT_MAX_STEPS here, or set GAIA_AGENT_MAX_STEPS= at runtime to -# override every agent at once. Agents that genuinely need more (e.g. CodeAgent -# for multi-file generation) override it explicitly in their own config. -DEFAULT_MAX_STEPS = 50 +# How many reasoning/tool steps an agent may take. **0 means no limit, and that +# is the default.** +# +# A fixed cap was measured against real work and found to stop the majority of +# the tasks worth doing: the median bug fix runs 52 steps, the median refactor +# 102, and a p90 feature implementation 140. A 50-step ceiling ended 39% of all +# build-track work mid-task, having already spent the tokens — the worst of both +# outcomes. +# +# Set GAIA_AGENT_MAX_STEPS= to impose a ceiling for a specific run. Nothing +# else bounds a runaway loop today, so a cap is still the right tool when the +# agent is unattended; it is simply the wrong default for interactive work, +# where the operator can see what is happening and interrupt. +DEFAULT_MAX_STEPS = 0 + +#: An answer that announces an action instead of taking one. Only consulted when +#: the turn ran no tool at all, so a report of work already done ("I read the +#: file and it says X") cannot trip it — the future tense and the empty tool log +#: are both required. +_INTENT_WITHOUT_ACTION = re.compile( + r"(?:i'll|i will|i am going to|i need to|i should|let me|let's|next,? i)\s+" + r"(?:now\s+)?(?:go ahead and\s+)?" + r"(?:read|open|check|look|search|find|run|execute|create|write|edit|update|" + r"modify|fix|add|inspect|examine|review|list|analyz|analys)" + r"|next steps?:" + r"|here(?:'s| is) (?:my|the) plan", + re.IGNORECASE, +) + + +#: Sentinel for "no ceiling", used wherever a step budget is compared. +NO_STEP_LIMIT = 0 + +_THINK_BLOCK = re.compile(r".*?", re.DOTALL) +_ORPHAN_THINK_CLOSE = re.compile(r"^.*?", re.DOTALL) + + +def strip_reasoning_blocks(text: str) -> str: + """Remove paired ``…`` blocks from a raw model response. + + Safe to run before JSON parsing: a paired block sits outside the tool-call + payload, so removing it cannot corrupt one. + """ + return _THINK_BLOCK.sub("", text).strip() if text else text + + +def strip_orphan_reasoning(answer: str) -> str: + """Drop deliberation preceding an unpaired ```` in *answer*. + + When a reasoning model's opening tag is consumed by a separate channel, + only ```` survives inline: the text arrives as + ````, the paired pattern matches nothing, + and the deliberation ships as the answer with a stray tag inside it. + + **Answer text only — never a raw response.** Applied before parsing, this + rule is destructive: a tool call whose arguments merely contain the + characters ```` loses everything before them, and + ``{"answer": "…"}`` is reduced to ``"}``. That regression cost a + benchmark run eight tasks. + """ + if not answer or "" not in answer: + return answer + return _ORPHAN_THINK_CLOSE.sub("", answer, count=1).strip() def effective_skill_body(agent, skill) -> str: @@ -210,9 +271,10 @@ def default_max_steps() -> int: """Resolve the global default agent step limit. Reads ``GAIA_AGENT_MAX_STEPS`` at call time (not import) so the env var can - be set after this module is imported and still take effect. Returns - ``DEFAULT_MAX_STEPS`` when the var is unset; raises on a present-but-invalid - value so a typo surfaces immediately instead of silently capping agents. + be set after this module is imported and still take effect. + + ``0`` means **no limit** and is the default. A negative value raises, so a + typo surfaces immediately instead of silently capping agents. """ raw = os.environ.get("GAIA_AGENT_MAX_STEPS") if raw is None or raw == "": @@ -221,13 +283,13 @@ def default_max_steps() -> int: value = int(raw) except ValueError as e: raise ValueError( - f"GAIA_AGENT_MAX_STEPS must be a positive integer, got {raw!r}. " - f"Unset it to use the default ({DEFAULT_MAX_STEPS})." + f"GAIA_AGENT_MAX_STEPS must be 0 (no limit) or a positive integer, " + f"got {raw!r}. Unset it to use the default (no limit)." ) from e - if value <= 0: + if value < 0: raise ValueError( - f"GAIA_AGENT_MAX_STEPS must be a positive integer, got {value}. " - f"Unset it to use the default ({DEFAULT_MAX_STEPS})." + f"GAIA_AGENT_MAX_STEPS must be 0 (no limit) or a positive integer, " + f"got {value}. Unset it to use the default (no limit)." ) return value @@ -4137,6 +4199,17 @@ def _truncation_budget(self) -> tuple: ) return budget_for_ctx(CLAUDE_CTX_SIZE) + + # Same reasoning as the Claude branch, generalised: a model served from + # a cloud or gateway backend has its own window and its own memory, so + # local hardware's device profile must not cap its tool results. This + # reads the client's host rather than the model id, because the same + # model id can be served either way. + client = getattr(self, "llm_client", None) or getattr(self, "client", None) + host = getattr(client, "host", None) + if host is not None and not is_local_host(host): + return budget_for_ctx(REMOTE_CTX_ASSUMPTION) + return truncation_budget(self.device) #: Scalar annotations worth coercing, by name as well as by type: a module @@ -4959,11 +5032,56 @@ def _with_verification_scope(self, answer: Optional[str]) -> Optional[str]: Empty stays empty — a blank answer is a signal downstream (cancelled turns skip persistence), and a scope line would make it non-blank. + + A claim to have written a file that does not exist is corrected here + rather than left standing (#3938). That failure is the worst one the + agent has, because it is indistinguishable from success in the + transcript: the step count looks healthy, no tool reported an error, + and the summary is confident and specific. """ if not answer or not answer.strip(): return answer + answer = self._flag_unwritten_claims(answer) return f"{answer.rstrip()}\n\n{self.verification_scope_statement()}" + def _flag_unwritten_claims(self, answer: str) -> str: + """Correct the answer when a file that should exist does not. + + Two separate failures, and the second is invisible to the first. The + agent can *claim* a file it never wrote — caught by reading the answer. + It can also silently skip a write the request asked for: told to put a + number in ``answer.txt`` it replied "400" and created nothing, which is + a correct answer to a question nobody asked. There is no false claim to + contradict there, so the request has to be checked too. + """ + # OSError only. A blanket ``except Exception`` here once swallowed a + # NameError and turned the whole check into a no-op that still passed + # its unit tests — a programming error must crash, not disable a guard. + try: + missing = unwritten_claims(answer, os.getcwd()) + missing += [ + name + for name in missing_requested_outputs( + getattr(self, "_current_query", "") or "", os.getcwd() + ) + if name not in missing + ] + except OSError as exc: + logger.debug("could not check written-file claims: %s", exc) + return answer + if not missing: + return answer + named = ", ".join(f"`{name}`" for name in missing) + logger.warning( + "Answer claims to have written %s, but the file is absent or empty", + named, + ) + return ( + f"{answer.rstrip()}\n\n**Correction:** this answer says it wrote " + f"{named}, but that file does not exist or is empty. The work is " + "not finished — the file still needs to be written." + ) + def process_query( self, user_input: str, @@ -5119,7 +5237,11 @@ def _process_query_impl( logger.debug(f"Input prompt: {prompt[:200]}...") # Process the query in steps, allowing for multiple tool usages - while steps_taken < steps_limit and final_answer is None: + # steps_limit == NO_STEP_LIMIT means run until the agent finishes or the + # operator interrupts. See DEFAULT_MAX_STEPS for why that is the default. + while ( + steps_limit == NO_STEP_LIMIT or steps_taken < steps_limit + ) and final_answer is None: # Cooperative cancellation: if a consumer (e.g. the Agent UI's # stream-timeout/disconnect cleanup) signalled cancel, stop here so # the producer thread is torn down rather than left running. Checked @@ -5832,9 +5954,7 @@ def _process_query_impl( # finds clean input, and before the response is stored in # conversation_history so the thinking text never bleeds into the # next turn and confuses the model about the current user message. - response = re.sub( - r".*?", "", response, flags=re.DOTALL - ).strip() + response = strip_reasoning_blocks(response) # Print the LLM response to the console logger.debug(f"LLM response: {response[:200]}...") @@ -6036,9 +6156,7 @@ def _process_query_impl( self.console.stop_progress() # Strip blocks before parsing (same reason as main path) - plan_response = re.sub( - r".*?", "", plan_response, flags=re.DOTALL - ).strip() + plan_response = strip_reasoning_blocks(plan_response) # Parse the plan response try: @@ -6475,6 +6593,39 @@ def _process_query_impl( # Stop progress indicator self.console.stop_progress() + # Repeating a call is a stall, not a conclusion. Say so and + # let the model change approach once before the turn ends — + # tasks were being abandoned here with the work half done + # and the answer reading "Task completed". + if not getattr(self, "_nudged_repeat_loop", False) and ( + steps_limit == NO_STEP_LIMIT or steps_taken < steps_limit - 1 + ): + self._nudged_repeat_loop = True + logger.debug( + "[WORKFLOW] %s called identically %d times — " + "asking for a different approach", + tool_name, + consecutive_count, + ) + tool_call_history.clear() + messages.append( + { + "role": "user", + "content": ( + f"You have called `{tool_name}` " + f"{consecutive_count} times with identical " + "arguments and learned nothing new. Do not " + "repeat it. Either use a different tool or " + "different arguments to make progress, or — " + "if you already have what you need — finish " + "the task now, including writing any file " + "the request asked for." + ), + } + ) + self.console.print_repeated_tool_warning() + continue + # Force a final answer if the same tool is called repeatedly. # Branches on whether the recent calls were errors so we # never claim success on a loop of failures. @@ -6642,7 +6793,9 @@ def _process_query_impl( # Check for final answer (after collecting stats) if "answer" in parsed: - answer_candidate = parsed["answer"] + # Answer text, after the JSON is safely parsed — see + # strip_orphan_reasoning for why it cannot run before this. + answer_candidate = strip_orphan_reasoning(parsed["answer"]) # Guard against incomplete workflows: detect when the LLM outputs # planning text ("Let me now search...") as a final answer after # calling index_document but before issuing a query tool call. @@ -6703,7 +6856,7 @@ def _process_query_impl( if ( last_index_pos >= 0 and not query_after_index - and steps_taken < steps_limit - 1 + and (steps_limit == NO_STEP_LIMIT or steps_taken < steps_limit - 1) ): logger.debug( "[WORKFLOW] Post-index answer without query — forcing query tool call: %s", @@ -6793,7 +6946,9 @@ def _process_query_impl( is_planning_text = len(answer_candidate) < 500 and any( phrase in answer_candidate.lower() for phrase in _PLANNING_PHRASES ) - if is_planning_text and steps_taken < steps_limit - 1: + if is_planning_text and ( + steps_limit == NO_STEP_LIMIT or steps_taken < steps_limit - 1 + ): # Inject a correction message and continue the loop to force the answer logger.debug( "[WORKFLOW] Blocking planning-only response as final answer: %s", @@ -6826,9 +6981,8 @@ def _process_query_impl( _TOOL_ARTIFACT_PATTERN = re.compile( r"^\s*\[tool:[a-zA-Z_]+\]\s*$", re.MULTILINE ) - if ( - _TOOL_ARTIFACT_PATTERN.match(answer_candidate.strip()) - and steps_taken < steps_limit - 1 + if _TOOL_ARTIFACT_PATTERN.match(answer_candidate.strip()) and ( + steps_limit == NO_STEP_LIMIT or steps_taken < steps_limit - 1 ): logger.debug( "[WORKFLOW] Blocking tool-syntax artifact as final answer: %s", @@ -6861,7 +7015,9 @@ def _process_query_impl( re.search(p, answer_candidate, re.DOTALL) for p in _RAW_JSON_PATTERNS ) - if is_raw_json and steps_taken < steps_limit - 1: + if is_raw_json and ( + steps_limit == NO_STEP_LIMIT or steps_taken < steps_limit - 1 + ): logger.debug( "[WORKFLOW] Blocking raw-JSON hallucination as final answer: %s", answer_candidate[:120], @@ -6931,7 +7087,7 @@ def _process_query_impl( _should_block_sd = ( is_capability_claim and not outcome_acknowledged - and steps_taken < steps_limit - 1 + and (steps_limit == NO_STEP_LIMIT or steps_taken < steps_limit - 1) ) if _should_block_sd: logger.debug( @@ -7016,6 +7172,78 @@ def _process_query_impl( "start GAIA with the `--sd` flag to enable it." ) + # Stopped-before-starting guard: the turn is ending with NO tool + # having run, and the answer says what the agent is about to do + # rather than what it did. On a task-execution benchmark this + # shape showed up as "1 step, 0 calls" — a whole task failed + # because the model narrated a plan and the loop accepted it. + # + # Deliberately narrow. A conversational turn needs no tools, so + # the intent language is what separates "I'll now read the file" + # from "the answer is 42". One nudge only; if the model repeats + # itself the answer stands rather than looping. + if ( + not tool_call_log + and not getattr(self, "_nudged_idle_turn", False) + and _INTENT_WITHOUT_ACTION.search(answer_candidate or "") + and (steps_limit == NO_STEP_LIMIT or steps_taken < steps_limit - 1) + ): + self._nudged_idle_turn = True + logger.debug( + "[WORKFLOW] Answer states intent with no tool call — " + "re-prompting once: %s", + (answer_candidate or "")[:80], + ) + messages.append( + { + "role": "user", + "content": ( + "You described what you were going to do but did " + "not do it — no tool has run this turn. Carry out " + "the action now with the appropriate tool. If the " + "request genuinely needs no tool, answer it " + "directly instead of describing a plan." + ), + } + ) + continue + + # Missing-output guard: the request named a file to write and + # that file is absent. Saying so in the answer is honest but + # leaves the work undone, so ask for it once instead — the + # agent usually has the content already and only skipped the + # write. A task that asked for a diagnosis file failed twice + # this way with the correct diagnosis sitting in the chat. + if not getattr(self, "_nudged_missing_output", False) and ( + steps_limit == NO_STEP_LIMIT or steps_taken < steps_limit - 1 + ): + # OSError only — see _flag_unwritten_claims for why. + try: + _absent = missing_requested_outputs( + getattr(self, "_current_query", "") or "", os.getcwd() + ) + except OSError: + _absent = [] + if _absent: + self._nudged_missing_output = True + _names = ", ".join(_absent) + logger.debug( + "[WORKFLOW] Requested output missing — re-prompting " + "once: %s", + _names, + ) + messages.append( + { + "role": "user", + "content": ( + f"{_names} was asked for but does not exist " + "or is empty. Write it now with the content " + "from your answer, then confirm." + ), + } + ) + continue + # Scope line goes on AFTER the subclass hook: a subclass that # rewrites the answer must not be able to drop it (#3376). final_answer = self._with_verification_scope( @@ -7209,7 +7437,15 @@ def _build_loop_break_summary( consecutive_count: int, step_results: list, ) -> str: - """Final-answer text when the loop breaks on repeats; honest on errors.""" + """Final-answer text when the loop breaks on repeats; never claims success. + + This path is reached only after the model was already told it was + repeating itself and did it again, so the turn is ending on a stall. + It used to end on "Task completed with {tool}. No further action + needed." — a claim of success for work that was never finished, which + is worse than no answer at all because nothing downstream can tell the + difference. + """ last = step_results[-1] if step_results else None if Agent._is_error_result(last): err = (last or {}).get("error") or "the tool returned an error" @@ -7219,7 +7455,11 @@ def _build_loop_break_summary( "I couldn't recover from this — please rephrase the request " "or check that the underlying service is running." ) - return f"Task completed with {tool_name}. No further action needed." + return ( + f"I stopped: I called `{tool_name}` {consecutive_count} times with " + "the same arguments and stopped making progress, so the task is " + "not finished. Tell me what to try instead, or narrow the request." + ) def _dedup_mutation_call( self, diff --git a/src/gaia/agents/base/verification.py b/src/gaia/agents/base/verification.py index 397dd86646..fafeb8d23d 100644 --- a/src/gaia/agents/base/verification.py +++ b/src/gaia/agents/base/verification.py @@ -204,3 +204,102 @@ def strip_verification_scope(text: str) -> str: if not isinstance(text, str) or VERIFICATION_SCOPE_PREFIX not in text: return text return _SCOPE_LINE_RE.sub("", text) + + +#: Phrasings that claim a file was produced, as opposed to merely mentioning it. +#: Kept narrow on purpose: "see config.py" or "config.py defines X" must not +#: trip this, or every answer that names a file gets a warning nobody reads. +_WROTE_PATTERNS = ( + r"\b(?:wrote|written|saved|created|generated|produced)\b[^.\n]{0,60}?" + r"[`'\"]([\w./\-]+\.[A-Za-z0-9]{1,8})[`'\"]", + r"[`'\"]([\w./\-]+\.[A-Za-z0-9]{1,8})[`'\"][^.\n]{0,40}?" + r"\b(?:was|is|has been)\s+(?:written|saved|created|generated)\b", +) + + +def claimed_written_files(answer: str) -> List[str]: + """Files the answer says it produced, in the order claimed. + + Reads the agent's own words rather than its tool log, because that is where + the failure lives: an agent can write a script, narrate running it, and stop + without ever executing it. The tool log looks clean; the claim is false. + """ + import re + + if not answer: + return [] + seen, found = set(), [] + for pattern in _WROTE_PATTERNS: + for match in re.finditer(pattern, answer, re.IGNORECASE): + name = match.group(1) + if name not in seen: + seen.add(name) + found.append(name) + return found + + +def unwritten_claims(answer: str, workspace: Optional[str] = None) -> List[str]: + """Of the files the answer claims to have written, those that do not exist. + + An empty file counts as missing: "generated orders.json" followed by a + zero-byte file is the same broken promise as no file at all. + """ + import os + + missing = [] + for name in claimed_written_files(answer): + path = os.path.join(workspace, name) if workspace else name + try: + if not os.path.isfile(path) or os.path.getsize(path) == 0: + missing.append(name) + except OSError: + # Unreadable is not the same as absent, and guessing either way + # would either cry wolf or hide a real miss. Say nothing. + continue + return missing + + +#: A request that names where its output must go. The agent can compute the +#: right answer and simply not write it — observed on a task that asked for a +#: number "in answer.txt", where the agent replied "400" and created nothing. +#: The phantom-write check cannot catch that: there is no false claim to +#: contradict, only a silent omission. +_REQUESTED_OUTPUT = ( + r"\b(?:write|save|put|output|store|record)\b[^.\n]{0,80}?" + r"\b(?:to|in|into|as)\s+[`'\"]?([\w./\-]+\.[A-Za-z0-9]{1,8})[`'\"]?", + r"\b(?:create|produce|generate)\b[^.\n]{0,40}?" + r"[`'\"]([\w./\-]+\.[A-Za-z0-9]{1,8})[`'\"]", +) + + +def requested_output_files(request: str) -> List[str]: + """Files the *user's request* says the answer must be written to.""" + import re + + if not request: + return [] + seen, found = set(), [] + for pattern in _REQUESTED_OUTPUT: + for match in re.finditer(pattern, request, re.IGNORECASE): + name = match.group(1) + if name not in seen: + seen.add(name) + found.append(name) + return found + + +def missing_requested_outputs( + request: str, workspace: Optional[str] = None +) -> List[str]: + """Of the files the request asked for, those that were never produced.""" + import os + + missing = [] + for name in requested_output_files(request): + path = os.path.join(workspace, name) if workspace else name + try: + if not os.path.isfile(path) or os.path.getsize(path) == 0: + missing.append(name) + except OSError: + continue + return missing diff --git a/src/gaia/agents/tools/file_tools.py b/src/gaia/agents/tools/file_tools.py index 57b257bd93..3d066c6ade 100644 --- a/src/gaia/agents/tools/file_tools.py +++ b/src/gaia/agents/tools/file_tools.py @@ -32,6 +32,29 @@ logger = logging.getLogger(__name__) +#: Enough of a file to tell text from binary. Executables and archives carry a +#: NUL well inside this; source files do not. +_BINARY_SNIFF_BYTES = 4096 + + +def _looks_binary(path: Path) -> bool: + """True when *path* is binary, by the rule ``grep -r`` uses: a NUL byte. + + Content search used to filter on an extension allowlist instead, which + silently skipped ``.toml``, ``.cfg``, ``.ts``, ``.go``, ``.rs`` and every + extensionless file — so grepping a Python project for a version string + never looked in ``pyproject.toml``, and reported success. Any such list is + wrong for the next language someone searches; sniffing is not. + + Unreadable reads as binary: skipping a file we cannot open is honest, + where treating it as text would raise inside the search loop. + """ + try: + with open(path, "rb") as handle: + return b"\0" in handle.read(_BINARY_SNIFF_BYTES) + except OSError: + return True + class FileSearchToolsMixin: """ @@ -115,7 +138,13 @@ def search_file( file_types: str = None, ) -> Dict[str, Any]: """ - Find files by name or pattern. + Find files by NAME. Does not look inside them. + + If the question is "where is X used / defined / declared", or + anything that must find every occurrence of a string, use + search_file_content instead — it greps contents. This tool only + matches filenames, and returns an empty result for a string that + appears inside files but in no filename. Args: file_pattern: name, substring, glob ("*.go") or regex to match. @@ -156,8 +185,29 @@ def search_file( ".rs", ".rb", ".sh", + ".log", + ".yaml", + ".yml", + ".toml", + ".ini", + ".cfg", + ".xml", + ".html", + ".css", + ".sql", + ".tsx", + ".jsx", } + # A pattern that names its own extension outranks the default + # list. Asking for "ci.log" and being told the directory is + # empty is worse than a slow search: the agent believes it and + # stops. Observed on a CI-triage task where the log was sitting + # in the working directory the whole time. + _named = os.path.splitext(file_pattern)[1].lower() + if _named and not file_types and _named.isascii(): + doc_extensions = doc_extensions | {_named} + import re as _re matching_files = [] @@ -799,9 +849,24 @@ def search_file_content( context_lines: int = 0, ) -> Dict[str, Any]: """ - Search for text patterns within files (grep-like functionality). + Search file contents for a pattern. This is grep. - Searches actual file contents on disk, not RAG indexed documents. + Use this to answer "where is X used / declared / defined" — + anything that must find EVERY occurrence. search_file finds files + by NAME and cannot answer that. Bumping a version "wherever it is + declared", renaming a symbol, or auditing a setting all start + here, not with a name search. + + Searches files on disk, not RAG-indexed documents. Binary files + are skipped; every text file is searched whatever its extension. + + Args: + pattern: regex, or plain text if it is not a valid regex. + directory: folder to search, recursively. Defaults to ".". + file_pattern: optional glob ("*.py") to narrow the search. + Omit it to search every text file. + case_sensitive: defaults to False. + context_lines: lines of context around each match. """ try: # Enforce the --allowed-paths sandbox before the existence probe @@ -831,31 +896,6 @@ def search_file_content( ), } - # Text file extensions to search - text_extensions = { - ".txt", - ".md", - ".py", - ".js", - ".java", - ".c", - ".cpp", - ".h", - ".json", - ".xml", - ".yaml", - ".yml", - ".csv", - ".log", - ".ini", - ".conf", - ".sh", - ".bat", - ".html", - ".css", - ".sql", - } - matches = [] files_searched = 0 ctx = max(0, int(context_lines)) @@ -935,10 +975,11 @@ def search_file(file_path: Path): if file_pattern: if not fnmatch.fnmatch(file_path.name, file_pattern): continue - else: - # Only search text files - if file_path.suffix.lower() not in text_extensions: - continue + elif _looks_binary(file_path): + # Skip binaries the way grep -r does. An extension + # allowlist here silently hid .toml, .cfg, .ts and + # every extensionless file (Dockerfile, Makefile). + continue files_searched += 1 if not search_file(file_path): @@ -956,9 +997,8 @@ def search_file(file_path: Path): if file_pattern: if not fnmatch.fnmatch(_fp2.name, file_pattern): continue - else: - if _fp2.suffix.lower() not in text_extensions: - continue + elif _looks_binary(_fp2): + continue if not search_file(_fp2): break diff --git a/src/gaia/agents/tools/shell_tools.py b/src/gaia/agents/tools/shell_tools.py index f3b1f45113..06e7e886fc 100644 --- a/src/gaia/agents/tools/shell_tools.py +++ b/src/gaia/agents/tools/shell_tools.py @@ -87,8 +87,88 @@ "jobs", # Git commands (mostly safe, read-only operations) "git", # Individual git subcommands checked separately + # Language runtimes, so the agent can run a project's own tests. + # See DEVELOPER_RUNTIMES for why this is not a widening of the sandbox. + "python", + "python3", + "py", + # The verification tooling those runtimes already expose via ``-m``. + # See PYTHON_CONSOLE_SCRIPTS. ``pytest`` is deliberately NOT here — it has + # a skill-grant policy that outranks this list (see that docstring). + "tox", + "nox", + "coverage", + "ruff", + "black", + "isort", + "flake8", + "pylint", + "mypy", } +#: The runtimes above, named separately so the reasoning is auditable instead of +#: buried in one large set. +#: +#: These do **not** widen the sandbox. ``execute_python_file`` already runs +#: agent-supplied Python in the same workspace under the same ``allowed_paths``, +#: so the interpreter was always reachable — just not by the obvious command. +#: All the omission bought was friction: asked to fix a bug and prove it, the +#: agent could not run ``python -m pytest`` and had to write a scratch runner +#: and execute that instead, spending extra steps on every verification. +#: +#: Measured on a task-execution benchmark, ``run_shell_command`` failed 55-76% of +#: its calls across every model tried, against 0-2% for a comparison harness with +#: no allowlist. Those refusals were the dominant cost of the tool. +DEVELOPER_RUNTIMES = frozenset({"python", "python3", "py"}) + +#: Console scripts that are **the same program** as ``python -m ``. +#: +#: Allowing ``python`` and refusing ``pytest`` blocks nothing — ``python -m +#: pytest`` runs the identical code — so the refusal only costs the agent a +#: step and teaches it that verification is unavailable. Measured on a +#: task-execution benchmark, the agent was told "Command 'pytest' is not +#: available to this agent" and thereafter answered without running the +#: tests at all; nearly every episode ended "unverified — none of them a +#: test, lint, or build". +#: +#: Scoped deliberately to tooling that *reads and reports*. ``pip`` is +#: reachable the same way and is still excluded: it mutates the environment +#: and reaches the network, which is the read/write line this allowlist +#: already draws for ``git``. Runners outside the Python ecosystem (``npm``, +#: ``go``, ``cargo``, ``make``) are a genuine widening rather than a name for +#: something already permitted, so they stay out until something measures a +#: need for them. +#: +#: ``pytest`` is **excluded on purpose**, though it is the one the agent +#: reaches for most. It carries a ``shell:execute:pytest`` grant policy in +#: :mod:`gaia.skills.binaries` that restricts flags and outranks this list, so +#: naming it here would change nothing. That policy is bypassable via +#: ``python -m pytest`` — tracked separately; resolving it is the policy +#: owner's call, not something to route around here. +PYTHON_CONSOLE_SCRIPTS = frozenset( + { + "tox", + "nox", + "coverage", + "ruff", + "black", + "isort", + "flake8", + "pylint", + "mypy", + } +) + +#: Runaway-loop backstop, not a pace-setter. +#: +#: At the previous 3-per-10-seconds an ordinary edit-then-test cycle tripped +#: the limit, and each trip cost a step and a model round trip that the user +#: waited through — a measured 5 refusals in a single twelve-task sweep. The +#: ceiling now sits where no deliberate sequence reaches it but a loop still +#: does within a few seconds. +MAX_COMMANDS_PER_10_SECONDS = 30 +MAX_COMMANDS_PER_MINUTE = 120 + # Actions/predicates that turn otherwise read-only commands into a write, # delete, or arbitrary-command-execution primitive. The whitelist only checks # the command NAME, so these must be inspected explicitly or an allowed command @@ -124,6 +204,29 @@ "help", } +#: Git subcommands that change the local repository but cannot reach a remote. +#: +#: Read-only git made a whole class of request impossible rather than merely +#: awkward: asked to stop tracking a committed secret, the agent produced a +#: correct ``.gitignore`` and then could not run ``git rm --cached``, so the +#: file stayed tracked and the task failed. Every arm failed it for that reason. +#: +#: These touch the index and working tree only. ``commit`` is deliberately NOT +#: here — creating commits unattended is gated elsewhere in this repo and that +#: policy stands; untracking a file needs ``git rm --cached``, not a commit. +#: ``push`` and ``remote`` stay refused because they leave the machine, and +#: history rewrites (``rebase``, ``filter-branch``) stay out because recovery +#: from a wrong one is not obvious. +LOCAL_WRITE_GIT_COMMANDS = { + "add", + "rm", + "mv", + "restore", + "switch", + "checkout", + "stash", +} + # Safe PowerShell cmdlet prefixes (read-only operations) SAFE_PS_CMDLET_PREFIXES = ( "get-", @@ -253,6 +356,39 @@ def _is_granted_binary(token: str, granted: frozenset) -> bool: return normalize_binary(token) in granted +_CD_PREFIX = re.compile( + r"^\s*cd\s+(?P\"[^\"]+\"|'[^']+'|[^\s&;|]+)\s*&&\s*(?P.+)$", + re.IGNORECASE | re.DOTALL, +) + + +def split_cd_prefix(command: str): + """Peel a leading ``cd &&`` off a command. + + Returns ``(directory, remainder)``, or ``(None, command)`` when the command + does not start that way. + + This shape is the agent's universal habit and was refused as command + chaining, which is the wrong reading of it: ``cd build && pytest`` is a + working directory and one command, not two commands joined to smuggle a + second past the allowlist. The tool already takes ``working_directory``, so + the intent maps exactly onto a parameter it has -- it was rejecting a + request it could have honoured, and it was the single most common refusal. + + Only a *leading* ``cd`` is peeled, and only one. Anything after the first + ``&&`` is still validated as a normal command, so this widens nothing: a + segment that would have been refused on its own is still refused. + """ + match = _CD_PREFIX.match(command or "") + if not match: + return None, command + directory = match.group("dir").strip().strip("\"'") + rest = match.group("rest").strip() + if not directory or not rest: + return None, command + return directory, rest + + def _operator_check_text(command: str) -> str: """The part of *command* the operator blocklist applies to. @@ -313,8 +449,8 @@ def __init__(self, *args, **kwargs): # Rate limiting configuration self.shell_command_times = deque(maxlen=100) # Track last 100 command times - self.max_commands_per_minute = 10 - self.max_commands_per_10_seconds = 3 + self.max_commands_per_minute = MAX_COMMANDS_PER_MINUTE + self.max_commands_per_10_seconds = MAX_COMMANDS_PER_10_SECONDS def _validate_shell_command(self, command: str) -> tuple: """Every refusal ``command`` earns on its text alone, plus its segments. @@ -497,8 +633,8 @@ def _check_rate_limit(self) -> tuple: # Initialize if not already done (defensive programming) if not hasattr(self, "shell_command_times"): self.shell_command_times = deque(maxlen=100) - self.max_commands_per_minute = 10 - self.max_commands_per_10_seconds = 3 + self.max_commands_per_minute = MAX_COMMANDS_PER_MINUTE + self.max_commands_per_10_seconds = MAX_COMMANDS_PER_10_SECONDS current_time = time.time() @@ -611,16 +747,22 @@ def _validate_command( } return None - # Special handling for git - only allow read-only operations + # Git: read-only operations, plus local writes that cannot reach a + # remote. The line is local-versus-published, not read-versus-write. if cmd_base == "git": if len(cmd_parts) > 1: git_subcmd = cmd_parts[1].lower() - if git_subcmd not in SAFE_GIT_COMMANDS: + allowed = SAFE_GIT_COMMANDS | LOCAL_WRITE_GIT_COMMANDS + if git_subcmd not in allowed: return { "status": "error", - "error": f"Git command '{git_subcmd}' is not allowed. Only read-only git operations are permitted.", + "error": ( + f"Git command '{git_subcmd}' is not allowed. Local " + "operations are permitted; anything that publishes " + "to a remote or rewrites history is not." + ), "has_errors": True, - "allowed_git_commands": list(SAFE_GIT_COMMANDS), + "allowed_git_commands": sorted(allowed), } # Special handling for wmic - only allow read-only queries elif cmd_base == "wmic": @@ -800,8 +942,24 @@ def _validate_command( "status": "error", "error": f"Command '{cmd_base}' is not in the allowed list for security reasons", "has_errors": True, - "hint": "Only read-only, informational commands are allowed", - "examples": "ls, cat, grep, find, git status, systeminfo, powershell -Command 'Get-WmiObject ...'", + # Naming the alternative is the whole point. The old hint read + # "Only read-only, informational commands are allowed" — untrue + # since python and the git write subcommands were added, and it + # taught the agent that checking its own work was impossible, so + # it stopped trying and answered unverified. + "hint": ( + "Allowed: file inspection (ls, cat, head, grep, find, wc, " + "diff), read-only git plus add/rm/mv/restore/switch/stash, " + "python / python3 / py, and the checkers ruff, black, " + "isort, flake8, pylint, mypy, coverage, tox. Run a " + "project's tests with 'python -m pytest'. Use write_file " + "and edit_file instead of rm/cp/mv/mkdir/touch, and " + "working_directory instead of 'cd'." + ), + "examples": ( + "python -m pytest -q, ruff check ., git diff, " + "grep -rn TODO ., ls -la" + ), } return None # Command is allowed @@ -819,6 +977,25 @@ def run_shell_command( """ Execute a shell command and return the output. + Only allowlisted programs run. What is available: + + * inspect — ls, cat, head, tail, grep, find, findstr, wc, sort, + uniq, diff, stat, file, pwd, du, df + * git — status, log, diff, show, add, rm, mv, restore, switch, + stash. Not commit, push or rebase. + * python — ``python``, ``python3``, ``py``. **Run a project's + tests with ``python -m pytest``**, and a module the same way + (``python -m json.tool``, ``python -c "..."``). + * check — ruff, black, isort, flake8, pylint, mypy, coverage, + tox, nox. + * system info — uname, systeminfo, hostname, ps, whoami. + + Not available: package managers (pip, npm), other ecosystems' + runners (node, go, cargo, make), file mutation (rm, mv, cp, + mkdir, touch — use write_file and edit_file), and the shell + operators ``&&``, ``||``, ``;``, ``>`` and backticks. Pipes work. + Use ``working_directory`` instead of ``cd``. + Args: command: Shell command to execute working_directory: Directory to run command in @@ -827,6 +1004,15 @@ def run_shell_command( Returns: Dictionary with status, output, and error information """ + # `cd && ` is the agent's universal habit and was refused + # as command chaining. That reads it wrong: it is a working + # directory and one command, and this tool already takes a working + # directory. Map it onto the parameter rather than reject it. An + # explicit working_directory wins, because the caller was specific. + _cd_dir, _rest = split_cd_prefix(command) + if _cd_dir and working_directory is None: + command, working_directory = _rest, _cd_dir + try: # Check rate limits first to prevent DOS allowed, reason, wait_time = self._check_rate_limit() diff --git a/src/gaia/apps/example/webui/package.json b/src/gaia/apps/example/webui/package.json index 8a3e574202..da0b4eede1 100644 --- a/src/gaia/apps/example/webui/package.json +++ b/src/gaia/apps/example/webui/package.json @@ -47,7 +47,7 @@ "@electron-forge/cli": "^7.11.2", "@electron-forge/maker-squirrel": "^7.11.2", "@electron-forge/maker-zip": "^7.11.2", - "electron": "^44.3.0" + "electron": "^44.1.1" }, "overrides": { "tar": ">=7.5.8" diff --git a/src/gaia/apps/webui/package-lock.json b/src/gaia/apps/webui/package-lock.json index 50920f5d88..bf052b3c7b 100644 --- a/src/gaia/apps/webui/package-lock.json +++ b/src/gaia/apps/webui/package-lock.json @@ -899,13 +899,13 @@ "license": "MIT" }, "node_modules/@oxc-project/types": { - "version": "0.149.0", - "resolved": "https://registry.npmjs.org/@oxc-project/types/-/types-0.149.0.tgz", - "integrity": "sha512-Efcc+iF0j3Bf67YjEqIqWXbX5XddXoK/Mw4K1/JuXwRCZ8N16VR7iT23nlCc9XrveFVh/E5Rqs2StT0V8v9LdA==", + "version": "0.146.0", + "resolved": "https://registry.npmjs.org/@oxc-project/types/-/types-0.146.0.tgz", + "integrity": "sha512-XC0QsnnhVe7sLIWmYmdPw7x5P0h4W8vUU3Nv1ySgWXtvCz8NizoAEpGXA0sOYoJQV2Rl13LgURAHQ5cI5ILCSA==", "dev": true, "license": "MIT", "funding": { - "url": "https://github.com/sponsors/oxc-project" + "url": "https://github.com/sponsors/Boshen" } }, "node_modules/@peculiar/asn1-schema": { @@ -961,9 +961,9 @@ } }, "node_modules/@rolldown/binding-android-arm-eabi": { - "version": "1.2.8", - "resolved": "https://registry.npmjs.org/@rolldown/binding-android-arm-eabi/-/binding-android-arm-eabi-1.2.8.tgz", - "integrity": "sha512-tN5aztYkKCte4i5SIrrz5yK/HMjEuCqCSCJa418jOV8tZ1cBY3YF2otxB1ktPxzsLA1BeTqwapK0bfjxNvHJVw==", + "version": "1.2.5", + "resolved": "https://registry.npmjs.org/@rolldown/binding-android-arm-eabi/-/binding-android-arm-eabi-1.2.5.tgz", + "integrity": "sha512-DLe/i+l8ynIBY7XEQ191TeZvCoowIGa18R+dIV30GW7DiOtp74i/xX8hs8GUjW5ARV7VZuie3d6AumSmCwbeRA==", "cpu": [ "arm" ], @@ -978,9 +978,9 @@ } }, "node_modules/@rolldown/binding-android-arm64": { - "version": "1.2.8", - "resolved": "https://registry.npmjs.org/@rolldown/binding-android-arm64/-/binding-android-arm64-1.2.8.tgz", - "integrity": "sha512-dIYTWl9XprMUiQFoc55KUyk/oS8SKYH3zFl0LTR7RT0Xj4hgSVyuJcroH8JUu8RcpF8fTB6E0aOwCkZoYPcDSQ==", + "version": "1.2.5", + "resolved": "https://registry.npmjs.org/@rolldown/binding-android-arm64/-/binding-android-arm64-1.2.5.tgz", + "integrity": "sha512-zXcwKlQApYAOELHd8PwKDFkagYF9Wy4e0RJ+0qnzl9Pjnpj75TEG8ufv40p2J7kCEfwZAsNiuzRIyNNMWT38ig==", "cpu": [ "arm64" ], @@ -995,9 +995,9 @@ } }, "node_modules/@rolldown/binding-darwin-arm64": { - "version": "1.2.8", - "resolved": "https://registry.npmjs.org/@rolldown/binding-darwin-arm64/-/binding-darwin-arm64-1.2.8.tgz", - "integrity": "sha512-PCSDQGXD2IyTEFrcgPyBM8jJuGmrbCMuoIOXdbEGVemruKACXoLQJrb+A45Z0L5t1RQkdfJprAYPkikbh7dzdA==", + "version": "1.2.5", + "resolved": "https://registry.npmjs.org/@rolldown/binding-darwin-arm64/-/binding-darwin-arm64-1.2.5.tgz", + "integrity": "sha512-dK4QakI42nzWgJT5sm4y4y/O//D4OxM75/cH28RLV+nzIN9AY+YsbuUVrUTjlLjXR6vpyxFbSsbmNuJ6BP9sww==", "cpu": [ "arm64" ], @@ -1012,9 +1012,9 @@ } }, "node_modules/@rolldown/binding-darwin-x64": { - "version": "1.2.8", - "resolved": "https://registry.npmjs.org/@rolldown/binding-darwin-x64/-/binding-darwin-x64-1.2.8.tgz", - "integrity": "sha512-Uk7lRsGhPFHVX/sAUC6D5H9Ol30dFHd6iquokll2th3LpdJ3F5CzQB+7DHn0Ri2mG+U7k2zXiPHDrwZenXhwSA==", + "version": "1.2.5", + "resolved": "https://registry.npmjs.org/@rolldown/binding-darwin-x64/-/binding-darwin-x64-1.2.5.tgz", + "integrity": "sha512-fqSALaUu1Wjd1nK2uW2kJDWdLCc8lx1IcY+MTY26Aurfdx19anlzhqXOgCFbBFQnlFDTn4TC1/7Nz4Bl2mLP3A==", "cpu": [ "x64" ], @@ -1029,9 +1029,9 @@ } }, "node_modules/@rolldown/binding-freebsd-x64": { - "version": "1.2.8", - "resolved": "https://registry.npmjs.org/@rolldown/binding-freebsd-x64/-/binding-freebsd-x64-1.2.8.tgz", - "integrity": "sha512-DjszaTEVogPqA5bYzsEeqDCQxbcp2fexQwKcRspYji2yzR68fCf+e4fx6kBSRDwX5/brZaHw/hWS9+A/+/w9sQ==", + "version": "1.2.5", + "resolved": "https://registry.npmjs.org/@rolldown/binding-freebsd-x64/-/binding-freebsd-x64-1.2.5.tgz", + "integrity": "sha512-/vCnNxlkxs9tKxNDcyWUePpJ/PgTzxIaVhoM5SmG8UV+GR/IcPam4VYxi7GIMo7PSDuNqlJqvprqii9NqqVCMw==", "cpu": [ "x64" ], @@ -1046,9 +1046,9 @@ } }, "node_modules/@rolldown/binding-linux-arm-gnueabihf": { - "version": "1.2.8", - "resolved": "https://registry.npmjs.org/@rolldown/binding-linux-arm-gnueabihf/-/binding-linux-arm-gnueabihf-1.2.8.tgz", - "integrity": "sha512-zmwa7FTmdzB6aaEEuuls18H6Ap5JmJPSoPTuXixeJZV6tG40SyLkApQtz1g8ptZtiEKqj9OM0oNLPh1AgvE31Q==", + "version": "1.2.5", + "resolved": "https://registry.npmjs.org/@rolldown/binding-linux-arm-gnueabihf/-/binding-linux-arm-gnueabihf-1.2.5.tgz", + "integrity": "sha512-abk0NLA519LxRCszmbE0jYKuQ9YPocOXTiOXOo6Yr+YAT95VH+PtqYAjOJvGKt3viEd/x4qzabAlwd5bHOOARg==", "cpu": [ "arm" ], @@ -1063,9 +1063,9 @@ } }, "node_modules/@rolldown/binding-linux-arm64-gnu": { - "version": "1.2.8", - "resolved": "https://registry.npmjs.org/@rolldown/binding-linux-arm64-gnu/-/binding-linux-arm64-gnu-1.2.8.tgz", - "integrity": "sha512-KdYQDPHwJVnbFwdTGMgxsI9SqblBlz6STGM+w1We/d5B8OWWidYH0MwkU/uA1wM5fIpO2MkOVxXrNzzuZhw9ew==", + "version": "1.2.5", + "resolved": "https://registry.npmjs.org/@rolldown/binding-linux-arm64-gnu/-/binding-linux-arm64-gnu-1.2.5.tgz", + "integrity": "sha512-Y7eALiJ8lr0M2HH103Js+g7V34wf6snlpZLAsHI90uLhr3PVlNsbFVAXJC9d/V6BnPyKtpSwI+NcB/RLxsQxuA==", "cpu": [ "arm64" ], @@ -1083,9 +1083,9 @@ } }, "node_modules/@rolldown/binding-linux-arm64-musl": { - "version": "1.2.8", - "resolved": "https://registry.npmjs.org/@rolldown/binding-linux-arm64-musl/-/binding-linux-arm64-musl-1.2.8.tgz", - "integrity": "sha512-jFJTifHnNPY+yzOoNZQfSIysrVyXzEQPhPnOUjmD1bcQGHH6s7c8cViKWar8YplQImE5N9JRqMCLrM2CdxOrZA==", + "version": "1.2.5", + "resolved": "https://registry.npmjs.org/@rolldown/binding-linux-arm64-musl/-/binding-linux-arm64-musl-1.2.5.tgz", + "integrity": "sha512-xMvZgnbZg4YVnR/AX2b3oOPDTFYJvUVaJg5FedA/LuvexAtXibZQej4cnTkw3rjsJ/ggUROB64TdtETiim+FYA==", "cpu": [ "arm64" ], @@ -1103,9 +1103,9 @@ } }, "node_modules/@rolldown/binding-linux-ppc64-gnu": { - "version": "1.2.8", - "resolved": "https://registry.npmjs.org/@rolldown/binding-linux-ppc64-gnu/-/binding-linux-ppc64-gnu-1.2.8.tgz", - "integrity": "sha512-FhiOziBDWPBjbcmRzfLyIJnaP7AVMFXT7YCXPjXxj7wKU3vx24RjrCNN/zjvVa+N2vVoHJwCoUBvsrN/DG3zIA==", + "version": "1.2.5", + "resolved": "https://registry.npmjs.org/@rolldown/binding-linux-ppc64-gnu/-/binding-linux-ppc64-gnu-1.2.5.tgz", + "integrity": "sha512-GRjeqTUDHTo5GwntsLaAMcBahG3nlpjftXWZLN73HiYQlhwEowvarFgQnRnQZtIp4keXX7quXFbG38uPZBa2EA==", "cpu": [ "ppc64" ], @@ -1123,9 +1123,9 @@ } }, "node_modules/@rolldown/binding-linux-s390x-gnu": { - "version": "1.2.8", - "resolved": "https://registry.npmjs.org/@rolldown/binding-linux-s390x-gnu/-/binding-linux-s390x-gnu-1.2.8.tgz", - "integrity": "sha512-WnHfADMzOV2Y55wlx1hzzQnar/wDt/VdvWSD99r18Mz9ylNieIGOkRx3UV21h7m/eJvjySYJkO26VvGNFkwsIQ==", + "version": "1.2.5", + "resolved": "https://registry.npmjs.org/@rolldown/binding-linux-s390x-gnu/-/binding-linux-s390x-gnu-1.2.5.tgz", + "integrity": "sha512-vLNTR45F2Uwc8AufkNXPmB4VliaXs+FvcheEogIzOXzO4l+LzieXF5A/TWxLy5HtqpsRCHUfd0lPVrrdgXdLHQ==", "cpu": [ "s390x" ], @@ -1143,9 +1143,9 @@ } }, "node_modules/@rolldown/binding-linux-x64-gnu": { - "version": "1.2.8", - "resolved": "https://registry.npmjs.org/@rolldown/binding-linux-x64-gnu/-/binding-linux-x64-gnu-1.2.8.tgz", - "integrity": "sha512-H9tRr5ibfXFVLxbPOseVewewFpl28zcEdjRDt2FTUZU7odxP0gEv1ki4/kGmcGOh78oRwZuuQllGLZ9zTJp84g==", + "version": "1.2.5", + "resolved": "https://registry.npmjs.org/@rolldown/binding-linux-x64-gnu/-/binding-linux-x64-gnu-1.2.5.tgz", + "integrity": "sha512-Mgj59/HTuYeK9Gz2MA+mBWKnHsAgkBSec15ZMb1st3oIfFbX7gCjOae7GydHhzcyQi9Z/7M1QuN9bR3oFqF0jQ==", "cpu": [ "x64" ], @@ -1163,9 +1163,9 @@ } }, "node_modules/@rolldown/binding-linux-x64-musl": { - "version": "1.2.8", - "resolved": "https://registry.npmjs.org/@rolldown/binding-linux-x64-musl/-/binding-linux-x64-musl-1.2.8.tgz", - "integrity": "sha512-UefiqfM3D6IVNlZ8tSGs9+Ejjud2T+oxO0IHADU45Y+lyEjD2dVFyZHbkfX0LUb5Zugo/oIv1eCO/KVYhgYJYA==", + "version": "1.2.5", + "resolved": "https://registry.npmjs.org/@rolldown/binding-linux-x64-musl/-/binding-linux-x64-musl-1.2.5.tgz", + "integrity": "sha512-mY8AP0/ichsbhAxGnLa3d3+MwV0EfgrPND2bplI3Ym8T6R2pJ0N87bvrKVwNXmdy3jnr6eQBecdqx/HMknBmpA==", "cpu": [ "x64" ], @@ -1183,9 +1183,9 @@ } }, "node_modules/@rolldown/binding-openharmony-arm64": { - "version": "1.2.8", - "resolved": "https://registry.npmjs.org/@rolldown/binding-openharmony-arm64/-/binding-openharmony-arm64-1.2.8.tgz", - "integrity": "sha512-637Ke4kWSy6rp9cxQ9gMOXlxPgIw/c1beASV4M//3+9I4uwBVOOl74G+e3zyU3u19U7RkRl/HuewixZ/Z6+Rjg==", + "version": "1.2.5", + "resolved": "https://registry.npmjs.org/@rolldown/binding-openharmony-arm64/-/binding-openharmony-arm64-1.2.5.tgz", + "integrity": "sha512-8SLssA2oweAxyRgDp789ACfRb/3P+zNRJpzZxSizxF9m8NUDQ4+3xjo8ttjhVGGw6Qxb70oZiEtIjaKikCO7Yw==", "cpu": [ "arm64" ], @@ -1200,9 +1200,9 @@ } }, "node_modules/@rolldown/binding-win32-arm64-msvc": { - "version": "1.2.8", - "resolved": "https://registry.npmjs.org/@rolldown/binding-win32-arm64-msvc/-/binding-win32-arm64-msvc-1.2.8.tgz", - "integrity": "sha512-xWBkPOF1Q9k/Gv1nQXnVdLxKu74jXppuOM4Z3mnypVUJJJwLsMl7hNJGRAUJoG8A5MgOI1ACKM+wBFxSJzKy4A==", + "version": "1.2.5", + "resolved": "https://registry.npmjs.org/@rolldown/binding-win32-arm64-msvc/-/binding-win32-arm64-msvc-1.2.5.tgz", + "integrity": "sha512-vGbruD5zquhoc8D9SViXgN2FBJtNdTyQ4DtG+SWiEGlJiAzoKcZ2xp+xuXCffhubVdt0NJlTZqkeRuERy7g8Cw==", "cpu": [ "arm64" ], @@ -1217,9 +1217,9 @@ } }, "node_modules/@rolldown/binding-win32-x64-msvc": { - "version": "1.2.8", - "resolved": "https://registry.npmjs.org/@rolldown/binding-win32-x64-msvc/-/binding-win32-x64-msvc-1.2.8.tgz", - "integrity": "sha512-uz2ZvfgXbxqNwijjjbxrnvALwpyODDcgc1T1N8N3rf/DXKQmaFwmB4LX4yyjggpwN2obdQLb2rgirX5ffCWYng==", + "version": "1.2.5", + "resolved": "https://registry.npmjs.org/@rolldown/binding-win32-x64-msvc/-/binding-win32-x64-msvc-1.2.5.tgz", + "integrity": "sha512-e/SXpgISz+IoqVcSSI0rx/d/he8zqLex+/rCWpnHpmVfmPIUjag9H6P7zotf0gJHwPUhQxZ/mF8tr6acebT9yw==", "cpu": [ "x64" ], @@ -1496,9 +1496,9 @@ } }, "node_modules/@types/react": { - "version": "19.3.0", - "resolved": "https://registry.npmjs.org/@types/react/-/react-19.3.0.tgz", - "integrity": "sha512-N0rFCuH9YoxG9/m61l9MfpJKfmLOVU0em7ipIz6TRgSSkvReLB9vL85GB+yr8Bs5leqpvg96JSwF4ZS1s4viQg==", + "version": "19.2.18", + "resolved": "https://registry.npmjs.org/@types/react/-/react-19.2.18.tgz", + "integrity": "sha512-AnzbBERsrLKtk2XSfTbYRLjQPdy116Sty4q+T+Bp3IC4l6jNBvreVPAHmpq9qhXQM7CXZPjLVmGMw9sy+hxQ3w==", "dev": true, "license": "MIT", "dependencies": { @@ -1506,13 +1506,13 @@ } }, "node_modules/@types/react-dom": { - "version": "19.3.0", - "resolved": "https://registry.npmjs.org/@types/react-dom/-/react-dom-19.3.0.tgz", - "integrity": "sha512-ZI7bU42mZXXKHn/qNLEw2IrbiINU7X5+vfgdixBHkCNpYWXjKgfQ/P+uyGb5CjOLB9UcnTeg3rylQtV2hym44Q==", + "version": "19.2.7", + "resolved": "https://registry.npmjs.org/@types/react-dom/-/react-dom-19.2.7.tgz", + "integrity": "sha512-I8bPpDLcHBv1qiIiXDCy71Rt8eQDKJP0sMSWJphDdAcdqiJ1sGpZamavoEIRZmYzjia9LuEb2HlYdDpmoENpvQ==", "dev": true, "license": "MIT", "peerDependencies": { - "@types/react": "^19.3.0" + "@types/react": "^19.2.0" } }, "node_modules/@types/responselike": { @@ -3101,9 +3101,9 @@ } }, "node_modules/electron": { - "version": "44.3.0", - "resolved": "https://registry.npmjs.org/electron/-/electron-44.3.0.tgz", - "integrity": "sha512-St9EV7F2VtYaYWD2qaAjBwUgKxx39eJOUsUJ5+/1113sqbVfNqv4Dbm/W1rN7qmYSPa+mWwR6yr+b7MfgjgVfQ==", + "version": "44.1.1", + "resolved": "https://registry.npmjs.org/electron/-/electron-44.1.1.tgz", + "integrity": "sha512-N2WCq2sbOkqQgvXJYx2lS6UiO8bF+Yr67trDnS6JKa2WxTCRQsAjGl57SUtdW9h6r5PlduBFjIhxhgd3dzv1hg==", "dev": true, "license": "MIT", "dependencies": { @@ -4716,9 +4716,9 @@ } }, "node_modules/lucide-react": { - "version": "1.46.0", - "resolved": "https://registry.npmjs.org/lucide-react/-/lucide-react-1.46.0.tgz", - "integrity": "sha512-Bv+FZXgZPrxc/NCl1e7JJVQFLdiCxYgxNVhqoV7X0p6I8ADJo8DxBnK1auH0fZz4AmqOJ3jgneL4f1i8LJQRAA==", + "version": "1.40.0", + "resolved": "https://registry.npmjs.org/lucide-react/-/lucide-react-1.40.0.tgz", + "integrity": "sha512-MaG+8WnOXkDWz9XeElj7TnQ890tTZUB0a36i03aRRCKGWE6e7jJpmdCvtxxuxcjjdqyN6m4sL6qlHVuSUDtYgg==", "dev": true, "license": "ISC", "peerDependencies": { @@ -5874,9 +5874,9 @@ } }, "node_modules/nanoid": { - "version": "3.3.19", - "resolved": "https://registry.npmjs.org/nanoid/-/nanoid-3.3.19.tgz", - "integrity": "sha512-Y2tUNy4ouw6tq5oDSKeQYGOyhkUBhNOcGV/02KC+6kd9eDGqdZd++mjMiIDilrBYvjEnCYvVtsuHCuP+okSfug==", + "version": "3.3.18", + "resolved": "https://registry.npmjs.org/nanoid/-/nanoid-3.3.18.tgz", + "integrity": "sha512-DTg4MJbGMWkfi6VZFdNt2/caMbQy4Ou+Op/hJQvGEWcnVfoA1QA+xzRKAzw9jD6+GVOOeYr/mIcuDSdug6F6+w==", "dev": true, "funding": [ { @@ -6288,9 +6288,9 @@ } }, "node_modules/postcss": { - "version": "8.5.28", - "resolved": "https://registry.npmjs.org/postcss/-/postcss-8.5.28.tgz", - "integrity": "sha512-RRuzqDtt5Y9h3quz5hWhK+TPnsmVs6WwSU6LkJMeY4HstUEDuYTG8UJSdawMRzmzAtV+KEoG8N3Qg2qLy5vM/A==", + "version": "8.5.26", + "resolved": "https://registry.npmjs.org/postcss/-/postcss-8.5.26.tgz", + "integrity": "sha512-u82N74LFzG8ca+dD8puPnplTXoGH4fTPpVGuIbt36G3qvNlkvfD0lEAZSxaly3KX8TS/L1A1gsCEmvKmBcVbkQ==", "dev": true, "funding": [ { @@ -6308,7 +6308,7 @@ ], "license": "MIT", "dependencies": { - "nanoid": "^3.3.18", + "nanoid": "^3.3.17", "picocolors": "^1.1.1", "source-map-js": "^1.2.1" }, @@ -6584,9 +6584,9 @@ } }, "node_modules/react": { - "version": "19.3.0", - "resolved": "https://registry.npmjs.org/react/-/react-19.3.0.tgz", - "integrity": "sha512-E8LUcbtBWt20bbl2YoHfx4ZDBdxVTfOKtCZn9cDSJ4l6/nuoApcpIBcj47t2wZoVX8g2ZHuMHbiShgCR1T5Sog==", + "version": "19.2.8", + "resolved": "https://registry.npmjs.org/react/-/react-19.2.8.tgz", + "integrity": "sha512-PWaYA1L/q9u2u7xYQi+Y3L3Yfnie7XyLeaJICV1MGD6LprsBxcAqGjYyr0eY3p+QdsA+x/Irkt4Qif8D63+Sbw==", "dev": true, "license": "MIT", "engines": { @@ -6594,16 +6594,16 @@ } }, "node_modules/react-dom": { - "version": "19.3.0", - "resolved": "https://registry.npmjs.org/react-dom/-/react-dom-19.3.0.tgz", - "integrity": "sha512-JDk8dgif51OjFoDE70+OT9ICyYr+69HlmihNwp1+Nsfbna3t5sIiCa9ZJktDmQ4/1b/rn26hIAR2uYXDMr5r0Q==", + "version": "19.2.8", + "resolved": "https://registry.npmjs.org/react-dom/-/react-dom-19.2.8.tgz", + "integrity": "sha512-rVprimfGBG3DR+Tq0IQG2DT5PxKth1WIGDmj5yPmlzr4YBe7uyE+Du4oVqTDXZSHGGGXRtTJEGSSePyQCMBglQ==", "dev": true, "license": "MIT", "dependencies": { - "scheduler": "^0.28.0" + "scheduler": "^0.27.0" }, "peerDependencies": { - "react": "^19.3.0" + "react": "^19.2.8" } }, "node_modules/react-is": { @@ -6887,13 +6887,13 @@ } }, "node_modules/rolldown": { - "version": "1.2.8", - "resolved": "https://registry.npmjs.org/rolldown/-/rolldown-1.2.8.tgz", - "integrity": "sha512-Z67nTmhZe7anqnM/EjI392w5i/ANUinjip7QYsOyN37oayduxt3ksdX0hf5OOamkAd53BiIHfbfSzfUmzKFQqQ==", + "version": "1.2.5", + "resolved": "https://registry.npmjs.org/rolldown/-/rolldown-1.2.5.tgz", + "integrity": "sha512-VD2IE5PUG4Oj8zz2VGykiYd5wbnjdIiSsNQb8Qu5B+noEp+A78mu2iVvpp27g8es14Tk9rofNs5Tku9iQCS4fA==", "dev": true, "license": "MIT", "dependencies": { - "@oxc-project/types": "=0.149.0", + "@oxc-project/types": "=0.146.0", "@rolldown/pluginutils": "^1.0.0" }, "bin": { @@ -6903,21 +6903,21 @@ "node": "^20.19.0 || >=22.12.0" }, "optionalDependencies": { - "@rolldown/binding-android-arm-eabi": "1.2.8", - "@rolldown/binding-android-arm64": "1.2.8", - "@rolldown/binding-darwin-arm64": "1.2.8", - "@rolldown/binding-darwin-x64": "1.2.8", - "@rolldown/binding-freebsd-x64": "1.2.8", - "@rolldown/binding-linux-arm-gnueabihf": "1.2.8", - "@rolldown/binding-linux-arm64-gnu": "1.2.8", - "@rolldown/binding-linux-arm64-musl": "1.2.8", - "@rolldown/binding-linux-ppc64-gnu": "1.2.8", - "@rolldown/binding-linux-s390x-gnu": "1.2.8", - "@rolldown/binding-linux-x64-gnu": "1.2.8", - "@rolldown/binding-linux-x64-musl": "1.2.8", - "@rolldown/binding-openharmony-arm64": "1.2.8", - "@rolldown/binding-win32-arm64-msvc": "1.2.8", - "@rolldown/binding-win32-x64-msvc": "1.2.8" + "@rolldown/binding-android-arm-eabi": "1.2.5", + "@rolldown/binding-android-arm64": "1.2.5", + "@rolldown/binding-darwin-arm64": "1.2.5", + "@rolldown/binding-darwin-x64": "1.2.5", + "@rolldown/binding-freebsd-x64": "1.2.5", + "@rolldown/binding-linux-arm-gnueabihf": "1.2.5", + "@rolldown/binding-linux-arm64-gnu": "1.2.5", + "@rolldown/binding-linux-arm64-musl": "1.2.5", + "@rolldown/binding-linux-ppc64-gnu": "1.2.5", + "@rolldown/binding-linux-s390x-gnu": "1.2.5", + "@rolldown/binding-linux-x64-gnu": "1.2.5", + "@rolldown/binding-linux-x64-musl": "1.2.5", + "@rolldown/binding-openharmony-arm64": "1.2.5", + "@rolldown/binding-win32-arm64-msvc": "1.2.5", + "@rolldown/binding-win32-x64-msvc": "1.2.5" } }, "node_modules/safe-buffer": { @@ -6960,9 +6960,9 @@ } }, "node_modules/scheduler": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/scheduler/-/scheduler-0.28.0.tgz", - "integrity": "sha512-juorfCmIkIw8tT+p5BXSm6PJjQF/ycEYmKyzURCIt/RaZIhL+PulbQ9Yu2z1HdOJDdqDTlxA1+xKBmHXJsczAw==", + "version": "0.27.0", + "resolved": "https://registry.npmjs.org/scheduler/-/scheduler-0.27.0.tgz", + "integrity": "sha512-eNv+WrVbKu1f3vbYJT/xtiF5syA5HPIMtf9IgY/nKg0sWqzAUEvqY/xm7OcZc/qafLx/iO9FgOmeSAp4v5ti/Q==", "dev": true, "license": "MIT" }, @@ -7764,16 +7764,16 @@ } }, "node_modules/vite": { - "version": "8.3.0", - "resolved": "https://registry.npmjs.org/vite/-/vite-8.3.0.tgz", - "integrity": "sha512-lhZBVvEHefgE+HQZC9O7EBJgCU/nVzFNl7vkS4RE0APtWLP02/8QVIkQtzBxPquh7lq5/78NHipTj7ODQ6XuyQ==", + "version": "8.2.2", + "resolved": "https://registry.npmjs.org/vite/-/vite-8.2.2.tgz", + "integrity": "sha512-cFKLV/PRgAUlIRm5WjMjJ86jrftzpqcgH+Us+DS8mI3CDNiH30Whrz8uHL3+MOLPAgqbMBAqWdAHAphOAM+z/Q==", "dev": true, "license": "MIT", "dependencies": { "lightningcss": "^1.33.0", - "picomatch": "^4.0.7", - "postcss": "^8.5.28", - "rolldown": "~1.2.6", + "picomatch": "^4.0.5", + "postcss": "^8.5.26", + "rolldown": "~1.2.4", "tinyglobby": "^0.2.17" }, "bin": { @@ -7790,7 +7790,7 @@ }, "peerDependencies": { "@types/node": "^20.19.0 || >=22.12.0", - "@vitejs/devtools": "^0.7.1", + "@vitejs/devtools": "^0.4.0 || ^0.5.0", "esbuild": "^0.27.0 || ^0.28.0", "jiti": ">=1.21.0", "less": "^4.0.0", diff --git a/src/gaia/cli.py b/src/gaia/cli.py index bffb9b34ab..4b323741ed 100644 --- a/src/gaia/cli.py +++ b/src/gaia/cli.py @@ -172,7 +172,12 @@ def initialize_lemonade_for_agent( if _ctx_override: try: _ctx_int = int(_ctx_override) - if _ctx_int > 0: + if _ctx_int == 0: + # 0 means "let the model and backend decide" — skip the + # startup ctx requirement entirely rather than asserting one. + log.info("GAIA_CTX_SIZE=0: not pinning a context window") + required_ctx = 0 + elif _ctx_int > 0: log.info( "GAIA_CTX_SIZE=%d overriding agent '%s' default of %d", _ctx_int, @@ -1241,8 +1246,8 @@ def build_parser(): "--max-steps", type=int, default=None, - help="Maximum conversation steps. Defaults to the global agent step " - "limit (50, or $GAIA_AGENT_MAX_STEPS if set).", + help="Maximum conversation steps. 0 means no limit, which is the " + "default; set $GAIA_AGENT_MAX_STEPS to impose one fleet-wide.", ) parent_parser.add_argument( "--list-tools", @@ -3390,9 +3395,7 @@ def main(): if resp.status == 200 and body == "ok": print(f"Telegram adapter: healthy ({url})") return - except (urllib.error.URLError, ConnectionError, TimeoutError): - # ConnectionError catches http.client.RemoteDisconnected, which - # is not a URLError - see AbstractHTTPHandler.do_open. + except urllib.error.URLError: pass pid_path = os.path.expanduser("~/.gaia/telegram.pid") @@ -7717,7 +7720,7 @@ def handle_mcp_status(args): print("⚠️ Server is running but may not be healthy") else: raise - except (urllib.error.URLError, ConnectionError, TimeoutError): + except urllib.error.URLError: print("⚠️ Server is running but status endpoint not accessible") print(" Server may be starting up or using an older version") except Exception as e: @@ -7753,7 +7756,7 @@ def handle_mcp_test(args): print("✅ MCP server is healthy") else: print("⚠️ Server may not be fully operational") - except (urllib.error.URLError, ConnectionError, TimeoutError): + except urllib.error.URLError: print(f"❌ Cannot connect to MCP server at {args.host}:{args.port}") print(" Make sure the server is running with: gaia mcp start") return @@ -7816,8 +7819,6 @@ def handle_mcp_test(args): print(f"❌ HTTP Error: {e.code} {e.reason}") except urllib.error.URLError as e: print(f"❌ Connection error: {e.reason}") - except (ConnectionError, TimeoutError) as e: - print(f"❌ Connection dropped by the MCP server: {e}") except json.JSONDecodeError as e: print(f"❌ Invalid JSON response: {e}") except Exception as e: @@ -7851,7 +7852,7 @@ def handle_mcp_agent(args): print("✅ MCP server is healthy") else: print("⚠️ Server may not be fully operational") - except (urllib.error.URLError, ConnectionError, TimeoutError): + except urllib.error.URLError: print(f"❌ Cannot connect to MCP server at {args.host}:{args.port}") print(" Make sure the server is running with: gaia mcp start") return @@ -7948,8 +7949,6 @@ def handle_mcp_agent(args): print(f"❌ HTTP Error: {e.code} {e.reason}") except urllib.error.URLError as e: print(f"❌ Connection error: {e.reason}") - except (ConnectionError, TimeoutError) as e: - print(f"❌ Connection dropped by the MCP server: {e}") except json.JSONDecodeError as e: print(f"❌ Invalid JSON response: {e}") except Exception as e: diff --git a/src/gaia/connectors/flow.py b/src/gaia/connectors/flow.py index 05749a1222..e742efd266 100644 --- a/src/gaia/connectors/flow.py +++ b/src/gaia/connectors/flow.py @@ -569,35 +569,6 @@ def _resolve_granted_scopes( return [s for s in returned if s in requested_set] -#: Bound on each provider-supplied field, matching OAuthProviderError's own. -_MAX_PROVIDER_FIELD_LEN = 300 - - -def _structured_oauth_error(resp: Any) -> "tuple[str, str]": - """The provider's RFC 6749 ``(error, error_description)``, bounded. - - Never falls back to the raw body (#3875): every request these responses - answer carries a credential — an authorization code, a device code, a - refresh token — and providers echo request context back into error - bodies, so the body must not reach a log line or a user-visible error. - Non-string fields are dropped rather than coerced, so a provider that - nests an object under ``error`` yields no detail instead of a stringified - fragment of its body. - """ - try: - payload = resp.json() - except Exception: # noqa: BLE001 — body may be empty/non-JSON - return "", "" - if not isinstance(payload, dict): - return "", "" - error = payload.get("error") - description = payload.get("error_description") - return ( - error[:_MAX_PROVIDER_FIELD_LEN] if isinstance(error, str) else "", - description[:_MAX_PROVIDER_FIELD_LEN] if isinstance(description, str) else "", - ) - - async def _exchange_code_for_tokens(flow: _PendingFlow, code: str) -> Dict[str, Any]: """Run the token-exchange step and persist the connection.""" provider = get_provider(flow.provider_id) @@ -609,14 +580,19 @@ async def _exchange_code_for_tokens(flow: _PendingFlow, code: str) -> Dict[str, response = await client.post(provider.token_url, data=body) if response.status_code != 200: - # Structured, bounded fields only (#2590) — the request this answers - # carried the authorization code and PKCE verifier, so the raw body - # never reaches the message (#3875). - error, description = _structured_oauth_error(response) + # Structured, bounded fields (#2590) — the previous behaviour + # interpolated the ENTIRE unbounded response.text into the message, + # so a caller that must not echo arbitrary exception text (it might + # ultimately carry provider-chosen content) had no way to report the + # failure at all short of a bare type name. + try: + err_payload = response.json() + except Exception: # noqa: BLE001 — body may be empty/non-JSON + err_payload = {} raise OAuthProviderError( flow.provider_id, - error=error, - error_description=description, + error=err_payload.get("error", ""), + error_description=err_payload.get("error_description", response.text[:300]), status_code=response.status_code, ) payload = response.json() @@ -733,8 +709,7 @@ async def start_device_flow(provider_id: str, scopes: Iterable[str]) -> Dict[str # rejects it — under the split, that means it was registered for # "microsoft" (consumers) but connected via "microsoft_work" # (organizations, or a pinned Directory tenant id). Name the - # connector to use instead, never an env var. Membership test only — - # the body is matched against, never surfaced. + # connector to use instead, never an env var. if "AADSTS9002346" in resp.text: other = "microsoft" if provider_id != "microsoft" else "microsoft_work" raise ConnectorsError( @@ -752,13 +727,9 @@ async def start_device_flow(provider_id: str, scopes: Iterable[str]) -> Dict[str # all (D6); the only tenant knob left is microsoft_work's optional # Directory (tenant) ID setup field. client_id_env = f"GAIA_{provider_id.upper()}_CLIENT_ID" - # Structured, bounded fields only — never the raw body (#3875). - error, description = _structured_oauth_error(resp) - detail = description or error - reason = f" ({detail})" if detail else "" raise ConnectorsError( f"Device-code request for {provider_id} failed with status " - f"{resp.status_code}{reason}. Check the client id " + f"{resp.status_code}: {resp.text[:300]}. Check the client id " f"({client_id_env}), or the Directory (tenant) ID setup field if " f"you set one. See docs/connectors/microsoft.mdx." ) @@ -818,7 +789,11 @@ async def poll_device_flow( if resp.status_code == 200: payload = resp.json() break - err, err_description = _structured_oauth_error(resp) + try: + err_payload = resp.json() + except Exception: # noqa: BLE001 — body may be empty/non-JSON + err_payload = {} + err = err_payload.get("error", "") if err == "authorization_pending": pass elif err == "slow_down": @@ -833,15 +808,17 @@ async def poll_device_flow( f"Device-code sign-in for {provider_id} was declined." ) else: - # Structured, bounded fields only (#2590) — see - # OAuthProviderError. This is where an admin-consent-required - # rejection (AADSTS65001) surfaces during polling; the raw - # body is never used as a fallback, because this request just - # posted the device code (#3875). + # Structured, bounded fields (#2590) — see OAuthProviderError. + # This is where an admin-consent-required rejection + # (AADSTS65001) actually surfaces during polling; a bare + # ConnectorsError with the response text glued in gave + # classify_oauth_exception nothing to inspect. raise OAuthProviderError( provider_id, error=err, - error_description=err_description, + error_description=err_payload.get( + "error_description", resp.text[:300] + ), status_code=resp.status_code, ) if _time.monotonic() >= deadline: diff --git a/src/gaia/daemon/sidecars/spec.py b/src/gaia/daemon/sidecars/spec.py index bef841c31c..2c4f188657 100644 --- a/src/gaia/daemon/sidecars/spec.py +++ b/src/gaia/daemon/sidecars/spec.py @@ -205,62 +205,6 @@ def _matches_dev_src_dir_shape(path: Path, agent_id: str) -> bool: return tuple(p.lower() for p in tail) == tuple(t.lower() for t in expected_tail) -def _dev_src_dir_agent_id(path: Path) -> Optional[str]: - """The agent id *path* is the dev-src dir of, or ``None`` if it isn't one. - - Matches ``hub/agents//python`` with the agent slot wildcarded, so a - caller who pointed at ANOTHER agent's source tree can be told that, rather - than being told they passed a checkout root. - """ - parts = path.parts - if len(parts) < 4: - return None - head, agents, _, tail = (p.lower() for p in parts[-4:]) - if (head, agents, tail) != ("hub", "agents", "python"): - return None - return parts[-2] - - -def _dev_src_dir_shape_error(resolved: Path, agent_id: str) -> str: - """Explain why *resolved* isn't ``agent_id``'s dev-src dir (issue #3852). - - A concrete "pass this instead" path is only named when it exists on disk — - blindly joining the expected tail onto whatever was typed turns a typo into - a longer typo and sends the caller to a path that was never there. - """ - expected = "/".join(_dev_src_dir_tail(agent_id)) - other_id = _dev_src_dir_agent_id(resolved) - if other_id is not None: - sibling = resolved.parent.parent / agent_id / "python" - instead = ( - f"pass '{sibling}' instead" - if sibling.is_dir() - else f"pass that checkout's {expected} directory instead" - ) - return ( - f"--dev-src-dir points at the '{other_id}' agent's source " - f"directory, but the agent requested is '{agent_id}'. Got " - f"'{resolved}'; {instead}, or start the other agent with " - f"`gaia daemon start-agent {other_id}`." - ) - - corrected = agent_dev_src_dir(resolved, agent_id) - if corrected.is_dir(): - return ( - f"--dev-src-dir must point at the {expected} directory inside a " - f"checkout, not the checkout root. Got '{resolved}'; pass " - f"'{corrected}' instead." - ) - - return ( - f"--dev-src-dir must be an absolute path ending in {expected} — the " - f"agent's source directory inside a checkout. Got '{resolved}', which " - f"is neither that shape nor a checkout containing it. Check it for a " - f"typo, or pass the {expected} directory of the checkout you want to " - f"run from." - ) - - def repo_root_from_agent_dev_src_dir(dev_src_dir: Path, agent_id: str) -> Path: """Invert :func:`agent_dev_src_dir`: recover the repo root a per-agent dev-mode source dir was joined from. @@ -358,10 +302,10 @@ def resolve_caller_dev_src_dir( *explicit* is validated client-side against the ``hub/agents// python`` shape (issue #2742) before it is ever sent to the daemon — a repo - root passed by mistake fails here, naming the corrected path when that - path exists (see :func:`_dev_src_dir_shape_error`), instead of reaching - the daemon and failing with :func:`repo_root_from_agent_dev_src_dir`'s - internal "restart remedy" wording. + root passed by mistake fails here, naming the exact corrected path, + instead of reaching the daemon and failing with + :func:`repo_root_from_agent_dev_src_dir`'s internal "restart remedy" + wording. Raises: DevSrcDirResolutionError: *explicit* is not an absolute path (a @@ -382,7 +326,12 @@ def resolve_caller_dev_src_dir( ) resolved = candidate.expanduser().resolve() if not _matches_dev_src_dir_shape(resolved, agent_id): - raise DevSrcDirResolutionError(_dev_src_dir_shape_error(resolved, agent_id)) + corrected = agent_dev_src_dir(resolved, agent_id) + raise DevSrcDirResolutionError( + f"--dev-src-dir must point at the hub/agents/{agent_id}/python " + f"directory inside a checkout, not the checkout root. Got " + f"'{resolved}'; pass '{corrected}' instead." + ) return resolved resolved_cwd = cwd or Path.cwd() diff --git a/src/gaia/electron/package.json b/src/gaia/electron/package.json index f13c17c4b2..e6eb3929e0 100644 --- a/src/gaia/electron/package.json +++ b/src/gaia/electron/package.json @@ -16,6 +16,6 @@ "electron": ">=28.0.0" }, "devDependencies": { - "electron": "^44.3.0" + "electron": "^44.1.1" } } diff --git a/src/gaia/eval/runner.py b/src/gaia/eval/runner.py index 9565342c23..5cf07e88f8 100644 --- a/src/gaia/eval/runner.py +++ b/src/gaia/eval/runner.py @@ -729,9 +729,7 @@ def preflight_check(backend_url, scenarios=None): with urllib.request.urlopen(f"{backend_url}/api/health", timeout=5) as r: if r.status != 200: errors.append(f"Agent UI returned HTTP {r.status}") - except (urllib.error.URLError, ConnectionError, TimeoutError) as e: - # ConnectionError catches http.client.RemoteDisconnected, which is not a - # URLError - see urllib.request.AbstractHTTPHandler.do_open. + except urllib.error.URLError as e: errors.append(f"Agent UI not reachable at {backend_url}: {e}") # Check corpus manifest @@ -818,7 +816,7 @@ def _probe_memory_admin(backend_url: str) -> Optional[str]: f"Memory admin probe failed with HTTP {e.code} from {backend_url}: " f"{e.reason}" ) - except (urllib.error.URLError, ConnectionError, TimeoutError) as e: + except urllib.error.URLError as e: return ( f"Memory admin probe could not reach {backend_url}: {e}. " "Is the Agent UI backend running?" diff --git a/src/gaia/factory/dataset/__init__.py b/src/gaia/factory/dataset/__init__.py new file mode 100644 index 0000000000..2e4a932997 --- /dev/null +++ b/src/gaia/factory/dataset/__init__.py @@ -0,0 +1,11 @@ +# Copyright(C) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. +# SPDX-License-Identifier: MIT +"""Step-level agentic evaluation dataset built from Claude Code session transcripts. + +One record is one decision point — an assistant inference call that dispatched at +least one tool — carrying the state the agent had, the action it took, what came +back, and which capabilities the step probes. + +Everything this package writes is derived from private transcripts and belongs in +``~/.gaia/cache/factory/dataset/``. Never in a repository working tree. +""" diff --git a/src/gaia/factory/dataset/annotate.py b/src/gaia/factory/dataset/annotate.py new file mode 100644 index 0000000000..6c09fb7909 --- /dev/null +++ b/src/gaia/factory/dataset/annotate.py @@ -0,0 +1,397 @@ +# Copyright(C) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. +# SPDX-License-Identifier: MIT +"""Infer what a transcript does not state: why a human spoke, and how it went. + +A transcript records what happened, never whether it was wanted. Everything in +this module is therefore **inference with stated evidence**, never ground truth, +and every annotation ships its own ``basis``, ``confidence`` and ``evidence`` +list so a consumer can discount it. + +Two things are genuinely recoverable and worth mining: + +* **Why a human spoke again.** A follow-up turn that says "no, revert that" is a + correction; one that says "perfect, ship it" is approval. Classifying those + turns is the closest the corpus comes to a verdict on the preceding work. +* **How an episode ended.** Verification commands and their exit status, whether + the run was interrupted, whether the last actions succeeded, and what the human + said next. + +**The hard limit, measured:** 67% of sessions contain exactly one human turn, so +there is no follow-up to read. For those the outcome annotation is driven only by +indirect evidence and is labelled ``unknown`` far more often than not. That is +reported rather than papered over — an outcome label that guesses on two-thirds +of the corpus would be worse than none. +""" + +import re +from dataclasses import asdict, dataclass, field +from typing import Any, Dict, List, Optional, Sequence + +# --------------------------------------------------------------------------- +# Human-turn classification +# --------------------------------------------------------------------------- + +#: Ordered specific-before-generic, first match wins. A turn that says +#: "no, use pytest instead" is a correction, not a new instruction, even though +#: it also names a tool. +_TURN_PATTERNS: List[tuple] = [ + ( + "interruption", + re.compile(r"\[request interrupted", re.I), + "harness recorded an interrupt", + ), + ( + "correction", + re.compile( + r"\b(?:no[,.\s]|nope\b|that'?s wrong|incorrect|not what i|" + r"you (?:missed|forgot|broke|misunderstood)|revert|undo|" + r"stop\b|don'?t\b|shouldn'?t|instead of|actually,|" + r"still (?:failing|broken|not working|wrong)|didn'?t work|" + r"try again|that'?s not)", + re.I, + ), + "corrective language", + ), + ( + "approval", + re.compile( + r"^\W{0,3}(?:thanks|thank you|perfect|great|nice|lgtm|looks good|" + r"ship it|yes\b|yep\b|correct\b|exactly|good\b|ok(?:ay)?\b|" + r"sounds good|go ahead|proceed|do it|approved)\W{0,3}$", + re.I, + ), + "short affirmative with no new instruction", + ), + ( + "approval_with_followup", + re.compile( + r"^\W{0,3}(?:thanks|perfect|great|lgtm|looks good|yes|ok(?:ay)?|" + r"good|nice)\b[,.! ]", + re.I, + ), + "affirmative opening followed by more instruction", + ), + ( + "clarification_answer", + re.compile( + r"^\W{0,3}(?:option\s*\d|[a-d]\)|the (?:first|second|latter|former)\b)", + re.I, + ), + "answers a posed question", + ), +] + +#: A turn shorter than this with no verb is more likely an answer than a task. +_SHORT_TURN_CHARS = 40 + +TURN_CLASSES = ( + "initial_instruction", + "correction", + "approval", + "approval_with_followup", + "clarification_answer", + "interruption", + "follow_up_task", +) + + +@dataclass +class TurnAnnotation: + """Why a human spoke, with the evidence for saying so.""" + + index: int + turn_class: str + confidence: str # high | medium | low + evidence: List[str] = field(default_factory=list) + chars: int = 0 + basis: str = "inferred_from_lexical_signals" + + def to_dict(self) -> Dict[str, Any]: + return asdict(self) + + +def classify_turn(text: str, index: int) -> TurnAnnotation: + """Classify one human turn. + + The first turn is the instruction by definition. Later turns are read for + corrective or affirmative language; anything else is a follow-up task. + """ + stripped = (text or "").strip() + if index == 0: + return TurnAnnotation( + index=index, + turn_class="initial_instruction", + confidence="high", + evidence=["first human turn of the transcript"], + chars=len(stripped), + basis="structural", + ) + for name, pattern, reason in _TURN_PATTERNS: + if pattern.search(stripped): + # A long turn that merely opens with "thanks" still carries work. + confidence = "high" if len(stripped) < 400 else "medium" + return TurnAnnotation( + index=index, + turn_class=name, + confidence=confidence, + evidence=[reason], + chars=len(stripped), + ) + if len(stripped) < _SHORT_TURN_CHARS: + return TurnAnnotation( + index=index, + turn_class="clarification_answer", + confidence="low", + evidence=[f"very short turn ({len(stripped)} chars), no corrective signal"], + chars=len(stripped), + ) + return TurnAnnotation( + index=index, + turn_class="follow_up_task", + confidence="medium", + evidence=["no corrective or affirmative signal; reads as new work"], + chars=len(stripped), + ) + + +def annotate_turns(prompts: Sequence[str]) -> List[TurnAnnotation]: + return [classify_turn(text, i) for i, text in enumerate(prompts)] + + +# --------------------------------------------------------------------------- +# Episode outcome inference +# --------------------------------------------------------------------------- + +_VERIFY_PASS = re.compile( + r"\b(\d+)\s+passed\b|\ball checks pass|\bOK\b|\bSUCCESS\b|build succeeded", re.I +) +_VERIFY_FAIL = re.compile( + r"\b(\d+)\s+failed\b|\bFAILED\b|\berror:|\btraceback\b|build failed", re.I +) +_VERIFY_CMD = re.compile( + r"\bpytest\b|\bnpm (?:run )?test\b|\bgo test\b|\bruff\b|\bflake8\b|" + r"\bblack\b|\bisort\b|\bmake\b|\btsc\b|gh pr checks", + re.I, +) +_LANDING_CMD = re.compile(r"git (?:commit|push)\b|gh pr (?:create|merge)\b", re.I) + +OUTCOME_LABELS = ("likely_succeeded", "likely_failed", "mixed", "unknown") + + +@dataclass +class OutcomeAnnotation: + """An inferred verdict on an episode, and why. + + ``basis`` is deliberately verbose. Nothing in a transcript states whether the + human's goal was met, and this label must never be read as if it did. + """ + + label: str + confidence: str + evidence: List[str] = field(default_factory=list) + next_turn_class: Optional[str] = None + verification_ran: bool = False + verification_passed: Optional[bool] = None + landed_change: bool = False + interrupted: bool = False + final_action_ok: Optional[bool] = None + basis: str = "inferred_from_transcript_signals_NOT_ground_truth" + + def to_dict(self) -> Dict[str, Any]: + return asdict(self) + + +def infer_episode_outcome( + decisions: Sequence[Any], + next_turn: Optional[TurnAnnotation], + interrupted: bool = False, +) -> OutcomeAnnotation: + """Infer how one episode went from the evidence the transcript does hold. + + Weighting, strongest first: + + 1. **What the human said next.** A correction is the clearest negative signal + in the corpus; an approval the clearest positive. + 2. **Verification.** A test or lint run and its result — the only + machine-checkable definition of done available. + 3. **Landing.** A commit, push or merged PR implies the work was accepted. + 4. **The final action's own outcome**, which is weak: a run can end on a + successful ``ls`` and still have failed the task. + """ + evidence: List[str] = [] + verification_ran = False + verification_passed: Optional[bool] = None + landed = False + final_ok: Optional[bool] = None + + for point in decisions: + for call, obs in zip(point.calls, point.observations): + command = ( + str(call.arguments.get("command", "")) if call.tool == "Bash" else "" + ) + if command and _VERIFY_CMD.search(command): + verification_ran = True + text = obs.text or "" + if obs.ok is False or _VERIFY_FAIL.search(text): + verification_passed = False + elif verification_passed is not False and _VERIFY_PASS.search(text): + verification_passed = True + if command and _LANDING_CMD.search(command) and obs.ok: + landed = True + if decisions: + final_ok = ( + decisions[-1].observations[-1].ok if decisions[-1].observations else None + ) + + next_class = next_turn.turn_class if next_turn else None + + # 1. The human's own reaction. + if next_class == "correction": + evidence.append("the next human turn corrects the agent") + label, confidence = "likely_failed", "medium" + elif next_class in ("approval", "approval_with_followup"): + evidence.append("the next human turn approves") + label, confidence = "likely_succeeded", "medium" + elif interrupted: + evidence.append("the run was interrupted by the user") + label, confidence = "likely_failed", "low" + else: + label, confidence = "unknown", "low" + if next_class is None: + evidence.append( + "no following human turn — 67% of sessions never get one, so " + "there is no reaction to read" + ) + else: + evidence.append(f"next human turn reads as {next_class}, which is neutral") + + # 2. Verification, which can confirm or contradict. + if verification_ran: + evidence.append( + f"verification ran and {'passed' if verification_passed else 'failed'}" + if verification_passed is not None + else "verification ran, result unreadable" + ) + if verification_passed is True and label == "unknown": + label, confidence = "likely_succeeded", "low" + elif verification_passed is False: + if label == "likely_succeeded": + label, confidence = "mixed", "low" + evidence.append("human approved but verification failed — conflicting") + else: + label, confidence = "likely_failed", "medium" + + # 3. Landing the change. + if landed: + evidence.append("the episode committed, pushed or merged") + if label == "unknown": + label, confidence = "likely_succeeded", "low" + elif label == "likely_succeeded": + confidence = "high" if confidence == "medium" else confidence + + if final_ok is False: + evidence.append("the episode's final tool call errored") + + return OutcomeAnnotation( + label=label, + confidence=confidence, + evidence=evidence, + next_turn_class=next_class, + verification_ran=verification_ran, + verification_passed=verification_passed, + landed_change=landed, + interrupted=interrupted, + final_action_ok=final_ok, + ) + + +# --------------------------------------------------------------------------- +# Inferred reasoning +# --------------------------------------------------------------------------- + +_INTENT_BY_FAMILY = { + "read": "inspect a file's contents", + "search": "locate code or files matching a pattern", + "edit": "modify a file in place", + "write": "create or overwrite a file", + "shell": "run a command", + "web": "fetch information from outside the repository", + "delegate": "hand a scoped sub-task to a subagent", + "plan": "record or revise a plan", + "meta": "invoke a skill or command", + "mcp": "call an external typed service", +} + + +def infer_reasoning( + point: Any, + previous: Optional[Any], + goal: str, + depth: int, + stated_preamble: str = "", +) -> Dict[str, Any]: + """Reconstruct a rationale for an action the model never explained. + + Extended thinking is encrypted corpus-wide, so the model's actual reasoning + is gone. This synthesises a rationale from what *is* observable: the goal, the + step immediately before, what the action does, and where in the episode it + sits. + + It is explicitly **not** the model's reasoning and is marked as such. It is + useful as a grounded description of the decision context — enough to give a + candidate harness the same framing — and useless as evidence about how the + reference model actually thought. + """ + families = [c.family for c in point.calls] + tools = [c.tool for c in point.calls] + primary = _INTENT_BY_FAMILY.get(families[0], "act") if families else "act" + + situation: List[str] = [] + if depth == 0: + situation.append("opening move for this instruction") + elif depth >= 8: + situation.append(f"step {depth} of a sustained chain") + else: + situation.append(f"step {depth} of this instruction") + + if previous is not None and previous.reference_quality == "errored": + classes = [o.error_class for o in previous.observations if o.error_class] + situation.append( + f"the previous step failed ({', '.join(classes) or 'unclassified'}), " + "so this is a recovery attempt" + ) + elif previous is not None: + prior_tools = ", ".join(sorted({c.tool for c in previous.calls})) + situation.append(f"follows a successful {prior_tools}") + + if len(point.calls) > 1: + situation.append(f"dispatches {len(point.calls)} tools together") + + # The caller supplies the preamble already scrubbed. Reading + # point.reasoning_text here instead published two raw usernames: the record's + # own `reasoning.visible_text` was scrubbed, but this copy was not. + stated = (stated_preamble or "").strip()[:300] + if stated: + confidence = "medium" + source = "grounded_in_visible_preamble" + else: + confidence = "low" + source = "grounded_in_state_and_action_only" + + summary = ( + f"To advance \"{goal[:110].strip()}\", {primary} via {'/'.join(sorted(set(tools)))}; " + + "; ".join(situation) + + "." + ) + return { + "inferred_summary": summary, + "stated_preamble": stated, + "situation": situation, + "confidence": confidence, + "source": source, + "basis": ( + "reconstructed_from_observable_state — the model's own reasoning is " + "encrypted corpus-wide and is NOT recoverable; this is a description " + "of the decision context, not the model's thought process" + ), + } diff --git a/src/gaia/factory/dataset/audit.py b/src/gaia/factory/dataset/audit.py new file mode 100644 index 0000000000..bafd7d49a8 --- /dev/null +++ b/src/gaia/factory/dataset/audit.py @@ -0,0 +1,266 @@ +# Copyright(C) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. +# SPDX-License-Identifier: MIT +"""Assert the invariants a built dataset must satisfy, and fail loudly if not. + +Every check here exists because the invariant was violated at least once during +development, silently, in a way that produced a plausible-looking dataset: + +* two decision points shared a ``step_index``, so a record's own action appeared + inside its own ``recent_steps``; +* ``tools_available`` was a union over the transcript's whole life, so 4.6% of + records advertised exactly the tools the record was asking a harness to pick; +* ``arg_hash`` was computed before scrubbing, so 58% of records carried a hash + that did not describe their own contents; +* a subagent's ``parent_session_id`` held its own file stem rather than its + parent's id, breaking the replay pointer for a third of the corpus; +* ``episode_id`` was keyed on the session, so one id spanned a parent and its 19 + subagents — grouping by episode merged 20 independent runs into one. + +None of those crashes anything. That is exactly why they are asserted rather than +left to a reviewer's eye. +""" + +import argparse +import collections +import hashlib +import json +import sys +from pathlib import Path +from typing import Any, Callable, Dict, Iterator, List, Tuple + +from gaia.factory.harvest.reader import _hash_args + + +def load_records(dataset: Path) -> List[Dict[str, Any]]: + records: List[Dict[str, Any]] = [] + for side in ("oracle", "pool"): + path = dataset / side / "records.jsonl" + if not path.is_file(): + raise FileNotFoundError( + f"No records at {path}. Build the dataset first with " + "`python -m gaia.factory.dataset.build`." + ) + with path.open(encoding="utf-8") as fh: + for line in fh: + line = line.strip() + if line: + records.append(json.loads(line)) + return records + + +def _violations( + records: List[Dict[str, Any]], dataset: Path +) -> Iterator[Tuple[str, str]]: + """Yield ``(invariant, detail)`` for every breach found.""" + seen_ids = collections.Counter(r["record_id"] for r in records) + for rid, n in seen_ids.items(): + if n > 1: + yield "unique_record_id", f"{rid} appears {n} times" + + # A missing required field is reported, never raised: an audit that crashes + # on a malformed record tells you less than one that names what is missing. + for rec in records: + absent = [f for f in REQUIRED_FIELDS if f not in rec] + if absent: + yield "required_fields_present", f"{rec.get('record_id', '?')} missing {absent}" + + episodes: Dict[str, set] = collections.defaultdict(set) + for rec in records: + if "episode_id" in rec and "transcript_id" in rec: + episodes[rec["episode_id"]].add(rec["transcript_id"]) + for episode, transcripts in episodes.items(): + if len(transcripts) > 1: + yield ( + "episode_id_is_per_transcript", + f"{episode} spans {len(transcripts)} transcripts", + ) + + steps: Dict[Tuple[str, str], collections.Counter] = collections.defaultdict( + collections.Counter + ) + for rec in records: + steps[(rec["session_id"], rec["transcript_id"])][rec["step_index"]] += 1 + for key, counter in steps.items(): + for step, n in counter.items(): + if n > 1: + yield "unique_step_index", f"{key[1][:16]} step {step} appears {n} times" + + for rec in records: + rid = rec["record_id"] + + for prior in rec["state"]["recent_steps"]: + if prior["step_index"] >= rec["step_index"]: + yield "no_lookahead_in_state", f"{rid} sees step {prior['step_index']}" + break + + used = {c["tool"] for c in rec["action"]["calls"]} + available = set(rec["tools_available"]) + if used and used == available: + yield "tools_available_does_not_leak", f"{rid} advertises exactly {sorted(used)}" + missing = used - available + if missing: + yield "tools_available_is_a_superset", f"{rid} missing {sorted(missing)}" + + for c in rec["action"]["calls"]: + if _hash_args(c["arguments"]) != c["arg_hash"]: + yield "arg_hash_matches_shipped_args", f"{rid} tool {c['tool']}" + break + + body = {k: v for k, v in rec.items() if k != "record_sha256"} + digest = hashlib.sha256( + json.dumps(body, sort_keys=True, ensure_ascii=False).encode("utf-8") + ).hexdigest() + if digest != rec.get("record_sha256"): + yield "record_sha256_verifies", rid + + if rec["scope"] == "subagent": + if rec["parent_session_id"] != rec["session_id"]: + yield "subagent_parent_is_partition_key", rid + elif rec["parent_session_id"] is not None: + yield "main_record_has_no_parent", rid + + if "reference_checks" not in rec: + yield "reference_checks_present", rid + + # Self-containment: everything a harness is shown must be usable without + # the private corpus. state.transcript_ref is provenance, not a + # dependency — a consumer who cannot read the transcripts must still be + # able to prompt and grade every record. + if not str(rec.get("goal", "")).strip(): + yield "harness_prompt_is_complete", f"{rid} empty goal" + elif not str(rec.get("episode_instruction", "")).strip(): + yield "harness_prompt_is_complete", f"{rid} empty episode_instruction" + elif not rec.get("tools_available"): + yield "harness_prompt_is_complete", f"{rid} no tools_available" + elif rec["state"].get("recent_steps") is None: + yield "harness_prompt_is_complete", f"{rid} no recent_steps" + + # An outcome label inside `state` would hand a harness the answer to + # "did this work?" before it acts. + leaked = [k for k in rec["state"] if "outcome" in k or "success" in k] + if leaked: + yield "outcome_not_visible_in_state", f"{rid} state has {leaked}" + + if rec["outcome"]["reference_quality"] == "errored": + if rec["grading_polarity"] != "avoid_reference": + yield "failed_reference_inverts_polarity", rid + elif rec["grading_polarity"] != "match_reference": + yield "succeeded_reference_matches_polarity", rid + + for obs in rec["observation"]: + for key in ("blob_ref", "file_content_ref", "original_file_ref"): + ref = obs.get(key) + if ref and not (dataset / "blobs" / f"{ref}.txt").is_file(): + yield "blob_refs_resolve", f"{rid} {key}={ref[:12]}" + for path, entry in rec["state"]["files_in_context"].items(): + if not (dataset / "blobs" / f"{entry['blob']}.txt").is_file(): + yield "blob_refs_resolve", f"{rid} files_in_context {path[:40]}" + + +#: Fields no record may omit. Everything downstream assumes they exist. +REQUIRED_FIELDS = ( + "record_id", + "schema_version", + "session_id", + "transcript_id", + "episode_id", + "message_id", + "scope", + "step_index", + "depth_index", + "partition", + "use_case", + "capability_axes", + "difficulty", + "tools_available", + "grading_polarity", + "reasoning", + "state", + "action", + "observation", + "outcome", + "reference_checks", + "episode_turn_class", + "episode_outcome", + "record_sha256", +) + +INVARIANTS = ( + "required_fields_present", + "unique_record_id", + "unique_step_index", + "episode_id_is_per_transcript", + "no_lookahead_in_state", + "tools_available_does_not_leak", + "tools_available_is_a_superset", + "arg_hash_matches_shipped_args", + "record_sha256_verifies", + "subagent_parent_is_partition_key", + "main_record_has_no_parent", + "reference_checks_present", + "harness_prompt_is_complete", + "outcome_not_visible_in_state", + "failed_reference_inverts_polarity", + "succeeded_reference_matches_polarity", + "blob_refs_resolve", +) + + +def audit(dataset: Path) -> Dict[str, Any]: + """Run every invariant. Returns a report; does not raise.""" + records = load_records(dataset) + counts: collections.Counter = collections.Counter() + examples: Dict[str, List[str]] = collections.defaultdict(list) + for name, detail in _violations(records, dataset): + counts[name] += 1 + if len(examples[name]) < 5: + examples[name].append(detail) + return { + "dataset": str(dataset), + "records": len(records), + "clean": not counts, + "invariants_checked": list(INVARIANTS), + "violations": dict(counts), + "examples": {k: v for k, v in examples.items()}, + } + + +def assert_clean(dataset: Path, log: Callable[[str], None] = print) -> None: + """Raise unless every invariant holds. + + Called at the end of a build: a dataset that violates these is worse than no + dataset, because it scores a harness against corrupted state and reports a + number anyway. + """ + report = audit(dataset) + if report["clean"]: + log( + f" audit clean — {len(INVARIANTS)} invariants over {report['records']} records" + ) + return + raise SystemExit( + "Dataset integrity audit FAILED:\n" + + json.dumps( + {"violations": report["violations"], "examples": report["examples"]}, + indent=2, + ) + + "\n\nFix the builder and rebuild. These breaches do not crash anything — " + "they produce a dataset that silently scores a harness against corrupted state." + ) + + +def main(argv: List[str] = None) -> int: + parser = argparse.ArgumentParser(description=__doc__.split("\n")[0]) + parser.add_argument( + "--dataset", + type=Path, + default=Path.home() / ".gaia" / "cache" / "factory" / "dataset", + ) + args = parser.parse_args(argv) + report = audit(args.dataset) + print(json.dumps(report, indent=2)) + return 0 if report["clean"] else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/src/gaia/factory/dataset/axes.py b/src/gaia/factory/dataset/axes.py new file mode 100644 index 0000000000..55f2f44401 --- /dev/null +++ b/src/gaia/factory/dataset/axes.py @@ -0,0 +1,164 @@ +# Copyright(C) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. +# SPDX-License-Identifier: MIT +"""Tag a decision point with the capabilities it probes. + +A harness designer needs to ask "how does mine do on error recovery +specifically", not "what is my average". Each axis below has a mechanical +detector and traces to a measured finding in the corpus analysis, so a record's +tags are reproducible and arguable rather than a matter of taste. + +A record carries every axis that fires — they are not mutually exclusive, and the +interesting records fire several. +""" + +import re +from typing import Any, Dict, List, Optional, Sequence + +#: Corpus median tool calls per human turn. Past it, the agent is sustaining a +#: plan rather than answering directly. +PLANNING_DEPTH = 8 + +#: Corpus max depth is 273 and p90 is 64; 32 is comfortably into the tail. +HARD_DEPTH = 32 + +#: p90 read result is ~32K chars, so a prior observation past 20K is a context +#: pressure the harness has to handle. +LARGE_OBSERVATION = 20_000 +HUGE_OBSERVATION = 50_000 + +_TRUNCATORS = re.compile( + r"\|\s*(?:head|tail)\b|\bhead\s+-\d|\btail\s+-\d|--max-count|\|\s*wc\b" +) +_REGEX_META = re.compile(r"[\\\[\]().*+?{}|^$]") +_VERIFIERS = re.compile( + r"\bpytest\b|\bunittest\b|\bnpm (?:run )?test\b|\bgo test\b" + r"|\bblack\b|\bisort\b|\bflake8\b|\bruff\b|\beslint\b|\blint\b" + r"|\bmake\b|\bnpm run build\b|\btsc\b|\bcargo build\b" + r"|git (?:diff|status)\b|gh pr checks\b" +) +_SCAFFOLD_LEADERS = frozenset({"cd", "export", "pwd", "source", "set"}) + +#: Every axis, in report order. Keeping the list explicit means a new axis has to +#: be added deliberately rather than appearing because some detector happened to +#: return a new string. +ALL_AXES: List[str] = [ + "tool_selection", + "argument_construction", + "error_recovery", + "context_management", + "state_reconstruction", + "parallelism", + "delegation", + "verification", + "multi_step_planning", + "stopping", +] + + +def _substantive_segments(call: Dict[str, Any]) -> int: + return sum( + 1 for s in call.get("shell_segments", []) if s.get("kind") == "substantive" + ) + + +def axes_for( + record: Dict[str, Any], + previous: Optional[Dict[str, Any]], + is_episode_final: bool, +) -> List[str]: + """Which capabilities this decision point probes. + + ``previous`` is the preceding decision point in the same transcript, needed + because error recovery is a property of the *transition*, not of the record. + """ + axes = {"tool_selection"} + calls: Sequence[Dict[str, Any]] = record["action"]["calls"] + observations: Sequence[Dict[str, Any]] = record["observation"] + + if record["action"]["width"] > 1: + axes.add("parallelism") + + if previous is not None and previous["outcome"]["reference_quality"] == "errored": + axes.add("error_recovery") + + if record["depth_index"] >= PLANNING_DEPTH: + axes.add("multi_step_planning") + + if is_episode_final: + axes.add("stopping") + + prior_big = previous is not None and any( + (o.get("chars") or 0) > LARGE_OBSERVATION for o in previous["observation"] + ) + if prior_big: + axes.add("context_management") + + for call in calls: + tool = call["tool"] + args = call.get("arguments", {}) + + if call["family"] == "delegate": + axes.add("delegation") + + if tool == "Bash": + command = str(args.get("command", "")) + if _substantive_segments(call) >= 2: + axes.add("argument_construction") + if _TRUNCATORS.search(command): + axes.add("context_management") + if _VERIFIERS.search(command): + axes.add("verification") + segments = call.get("shell_segments", []) + if segments and segments[0].get("leader") in _SCAFFOLD_LEADERS: + axes.add("state_reconstruction") + + elif tool in ("Grep", "Glob"): + pattern = str(args.get("pattern", "")) + if _REGEX_META.search(pattern): + axes.add("argument_construction") + + elif tool in ("Edit", "MultiEdit"): + if len(str(args.get("old_string", ""))) >= 200: + axes.add("argument_construction") + + elif tool == "Read" and args.get("offset") is not None: + axes.add("context_management") + + if any((o.get("chars") or 0) > LARGE_OBSERVATION for o in observations): + axes.add("context_management") + + return [a for a in ALL_AXES if a in axes] + + +def difficulty_for(record: Dict[str, Any], prior_failures_in_episode: int) -> str: + """Bucket a record's difficulty *for the reference harness*. + + Derived from position and outcome, which are the only difficulty signals the + corpus actually holds. This is not a measure of intrinsic task difficulty and + the datasheet says so — a trivial action taken at depth 40 lands in ``hard`` + because everything at depth 40 is expensive, not because the action was. + """ + depth = record["depth_index"] + width = record["action"]["width"] + biggest = ( + max((o.get("chars") or 0) for o in record["observation"]) + if record["observation"] + else 0 + ) + + if ( + depth >= HARD_DEPTH + or record["outcome"]["reference_quality"] == "errored" + or prior_failures_in_episode >= 2 + or biggest > HUGE_OBSERVATION + or width >= 4 + ): + return "hard" + if ( + depth < 4 + and width == 1 + and record["outcome"]["reference_quality"] == "succeeded" + and prior_failures_in_episode == 0 + ): + return "easy" + return "moderate" diff --git a/src/gaia/factory/dataset/build.py b/src/gaia/factory/dataset/build.py new file mode 100644 index 0000000000..c0eb61fa89 --- /dev/null +++ b/src/gaia/factory/dataset/build.py @@ -0,0 +1,1143 @@ +# Copyright(C) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. +# SPDX-License-Identifier: MIT +"""Build the step-level agentic evaluation dataset. + +One record is one *decision point*: an assistant inference call that dispatched at +least one tool. That is what a harness must reproduce — given everything known so +far, emit the next action — and it is the only unit that keeps width (did the +harness fan out?) and the reasoning preamble attached to the action they belong to. + +Two passes over the corpus. The first indexes every decision point cheaply so +quotas can be computed against real availability; the second materializes only the +selected records, with their bounded state and content-addressed blobs. Splitting +it avoids holding several gigabytes of file content in memory to sample 4,000 rows +out of 32,000. + +Everything written goes to the cache directory. Nothing derived from a transcript +belongs in a repository. +""" + +import argparse +import hashlib +import json +import os +import re +import shutil +import sys +from collections import Counter, defaultdict +from pathlib import Path +from typing import Any, Dict, List, Optional, Set, Tuple + +from gaia.factory.dataset import annotate as annotate_mod +from gaia.factory.dataset import audit as audit_mod +from gaia.factory.dataset import axes as axes_mod +from gaia.factory.dataset import partition as part +from gaia.factory.dataset import verify as verify_mod +from gaia.factory.dataset.extract import ( + DecisionPoint, + TranscriptScan, + iter_transcripts, + scan_transcript, +) +from gaia.factory.dataset.scrub import Scrubber, ScrubStats, is_pasted_third_party +from gaia.factory.dataset.verifiers import STATIC_CHECKS, run_static_checks +from gaia.factory.harvest.reader import _hash_args + +#: Inline observation text past this goes to a content-addressed blob. File +#: contents are median ~6 KB and p90 ~32 KB, so leaving them inline would make the +#: JSONL mostly file bodies and defeat streaming. +INLINE_LIMIT = 4000 + +#: How many prior decision points to materialize in full. Enough to decide the +#: next action; the exact prefix stays reachable through ``state.transcript_ref``. +RECENT_STEPS = 12 + +#: Quota shape, applied per (use_case, partition). Compresses the head so +#: doc_authoring's 4,933 decision points do not swamp eval_quality's 164, and +#: floors the tail so the rare use-cases stay evaluable. +QUOTA_FRACTION = 0.12 +QUOTA_FLOOR = 40 +QUOTA_CEILING = 220 + +#: Minimum records per capability axis inside a quota, so error_recovery does not +#: sit at its 3.4% natural base rate. +AXIS_FLOOR = 8 + +#: How much of the agent's accumulated vocabulary to carry on a record. Both caps +#: were far too tight at first: 30% of records hit a 60-binary limit, so binaries +#: the agent had demonstrably used looked unseen to the argument verifier. +KNOWN_PATHS_CAP = 800 +OBSERVED_BINARIES_CAP = 250 + +#: How many distinct transcripts must use a binary before it counts as installed +#: on this machine rather than as a one-off token from an inline script body. +ENV_BINARY_MIN_TRANSCRIPTS = 3 + +#: A path-shaped token inside a shell command. The agent learns most of its path +#: vocabulary from commands and search hits, not from Read arguments — collecting +#: only the latter left the median record knowing three paths. +#: +#: Both separators must be in the class. Written as ``[\/]`` the backslash is an +#: escape for ``/`` rather than a member, so the class matched forward slashes +#: only and every Windows path in a shell command was silently skipped — on a +#: corpus that is entirely Windows. +_PATH_TOKEN = re.compile(r"[A-Za-z0-9_.<>-]*[\\/][A-Za-z0-9_.<>/\\-]{2,}") + + +def _sha(text: str) -> str: + return hashlib.sha256(text.encode("utf-8", "replace")).hexdigest() + + +def _rank(record_id: str) -> str: + """Deterministic sampling order, independent of filesystem iteration order.""" + return hashlib.sha256(("rank:" + record_id).encode("utf-8")).hexdigest() + + +def load_snapshot( + snapshot: Path, +) -> Tuple[List[str], Dict[str, str], Dict[str, List[str]]]: + """Session ids from the frozen traces, plus primary and secondary labels.""" + traces = snapshot / "traces.jsonl" + labels_path = snapshot / "labels.txt" + if not traces.is_file(): + raise FileNotFoundError( + f"No traces.jsonl under {snapshot}.\n" + "Run the extractor first: python -m gaia.factory.harvest.scan\n" + "It writes traces.jsonl, intents.jsonl and stats.json to " + "~/.gaia/cache/factory/. Pass --snapshot only to build from a frozen " + "copy of those files." + ) + if not labels_path.is_file(): + raise FileNotFoundError( + f"No labels.txt under {snapshot}.\n" + "Use-case labels are required and are assigned by Claude Code, not by " + "a script — see .claude/skills/building-eval-dataset/SKILL.md, step 2. " + "It batches intents.jsonl and writes " + f"{snapshot / 'labels.txt'} as '<8-char-session-prefix> " + "'." + ) + session_ids: List[str] = [] + with traces.open(encoding="utf-8") as fh: + for line in fh: + line = line.strip() + if line: + session_ids.append(json.loads(line)["session_id"]) + primary: Dict[str, str] = {} + secondary: Dict[str, List[str]] = {} + with labels_path.open(encoding="utf-8") as fh: + for line in fh: + parts = line.split() + if len(parts) >= 2: + primary[parts[0]] = parts[1] + secondary[parts[0]] = parts[2].split(",") if len(parts) > 2 else [] + return session_ids, primary, secondary + + +def _use_case(session_id: str, primary: Dict[str, str]) -> str: + return primary.get(session_id[:8], "unlabelled") + + +def tools_available_for( + transcript_tools: Set[str], + core_tools: Set[str], + mcp_tools_by_server: Dict[str, Set[str]], +) -> List[str]: + """The tool set offered at a decision point, without leaking the answer. + + Naively this is "every tool this transcript used", but that is the union over + the transcript's *whole life* — including the tool the record is asking the + harness to choose. Measured on the first build it was severe: 30% of records + listed three tools or fewer and **4.6% listed exactly the tools used**, which + hands the answer over. Short subagent transcripts were the worst case. + + So the set is rebuilt from what is true rather than from what was used: + + * **Core tools are harness-constant.** Every non-MCP tool observed anywhere + in the corpus was available in every session — one Claude Code build spans + the whole period — so listing all of them leaks nothing. + * **MCP tools come by server.** An MCP server is connected for a session's + entire life, so knowing ``mcp__claudia__*`` is reachable at step 1 is a + fact about the session, not foresight. Naming every tool that server + exposes corpus-wide keeps the individual choice hidden. + """ + available = set(core_tools) + servers = { + t.split("__")[1] + for t in transcript_tools + if t.startswith("mcp__") and len(t.split("__")) > 2 + } + for server in servers: + available |= mcp_tools_by_server.get(server, set()) + return sorted(available) + + +class BlobStore: + """Content-addressed store for payloads too big to inline. + + Deduplicates for free, which matters: ~39% of reads in this corpus re-read a + file already read in the same session. + """ + + def __init__(self, root: Path) -> None: + self.root = root + self.root.mkdir(parents=True, exist_ok=True) + self.written: Set[str] = set() + self.bytes_written = 0 + + def put(self, text: str) -> str: + digest = _sha(text) + if digest not in self.written: + path = self.root / f"{digest}.txt" + data = text.encode("utf-8", "replace") + path.write_bytes(data) + self.written.add(digest) + self.bytes_written += len(data) + return digest + + def read(self, digest: str) -> Optional[str]: + path = self.root / f"{digest}.txt" + if not path.is_file(): + return None + return path.read_text(encoding="utf-8", errors="replace") + + +def _light_record( + point: DecisionPoint, + scan: TranscriptScan, + use_case: str, + partition_side: str, +) -> Dict[str, Any]: + """The minimum needed to tag axes and apply quotas, without any payload.""" + return { + "record_id": _sha(point.session_id + "|" + point.message_id)[:16], + "session_id": point.session_id, + "message_id": point.message_id, + "scope": point.scope, + "transcript_stem": scan.session_id, + "step_index": point.step_index, + "episode_index": point.episode_index, + "depth_index": point.depth_index, + "use_case": use_case, + "partition": partition_side, + "work_item": part.work_item(scan.project, scan.git_branch, point.session_id), + "action": { + "width": point.width, + "calls": [ + { + "tool": c.tool, + "family": c.family, + "arg_hash": c.arg_hash, + "shell_segments": c.shell_segments, + "arguments": c.arguments, + } + for c in point.calls + ], + }, + "observation": [{"chars": o.chars} for o in point.observations], + "outcome": {"reference_quality": point.reference_quality}, + } + + +def index_pass( + session_ids: List[str], + projects_root: Path, + primary: Dict[str, str], + index_path: Path, +) -> Dict[str, Any]: + """Pass 1 — walk every wanted transcript and index its decision points.""" + tool_keys: Dict[str, Counter] = defaultdict(Counter) + tool_calls: Counter = Counter() + tool_types: Dict[str, Dict[str, Counter]] = defaultdict( + lambda: defaultdict(Counter) + ) + tools_by_transcript: Dict[str, Set[str]] = defaultdict(set) + core_tools: Set[str] = set() + mcp_tools_by_server: Dict[str, Set[str]] = defaultdict(set) + binary_transcripts: Dict[str, Set[str]] = defaultdict(set) + stats = { + "sessions_requested": len(session_ids), + "sessions_on_disk": 0, + "sessions_with_decision_points": 0, + "transcripts_scanned": 0, + "decision_points": 0, + "with_preamble": 0, + "with_thinking_block": 0, + "with_recoverable_thinking": 0, + "skipped_lines": 0, + } + on_disk: Set[str] = set() + seen_sessions: Set[str] = set() + + with index_path.open("w", encoding="utf-8") as out: + for session_id, path, scope in iter_transcripts(session_ids, projects_root): + on_disk.add(session_id) + scan = scan_transcript(path, scope=scope) + if scan is None: + continue + seen_sessions.add(session_id) + stats["transcripts_scanned"] += 1 + stats["skipped_lines"] += scan.skipped_lines + use_case = _use_case(session_id, primary) + side = part.assign(session_id) + episode_last: Dict[int, int] = {} + for point in scan.decisions: + episode_last[point.episode_index] = point.step_index + for point in scan.decisions: + # Subagent decision points inherit the parent session's id so a + # delegated run can never land on the other side of the split. + point.session_id = session_id + light = _light_record(point, scan, use_case, side) + light["transcript_path"] = str(path) + light["is_episode_final"] = ( + point.step_index == episode_last[point.episode_index] + ) + light["has_preamble"] = bool(point.reasoning_text.strip()) + light["had_thinking_block"] = point.had_thinking_block + out.write(json.dumps(light, ensure_ascii=False) + "\n") + stats["decision_points"] += 1 + if light["has_preamble"]: + stats["with_preamble"] += 1 + if point.had_thinking_block: + stats["with_thinking_block"] += 1 + for call in point.calls: + tool_calls[call.tool] += 1 + tools_by_transcript[scan.session_id].add(call.tool) + parts = call.tool.split("__") + if call.tool.startswith("mcp__") and len(parts) > 2: + mcp_tools_by_server[parts[1]].add(call.tool) + else: + core_tools.add(call.tool) + for seg in call.shell_segments: + if seg["kind"] == "substantive": + binary_transcripts[seg["leader"]].add(scan.session_id) + for key, value in call.arguments.items(): + tool_keys[call.tool][key] += 1 + tool_types[call.tool][key][type(value).__name__] += 1 + + # Two different absences, deliberately not merged: Claude Code prunes old + # transcripts, and a session that made no tool call carries no procedure to + # evaluate. Reporting them as one number would blame pruning for both. + stats["sessions_on_disk"] = len(on_disk) + stats["sessions_pruned_from_disk"] = len(session_ids) - len(on_disk) + stats["sessions_with_decision_points"] = len(seen_sessions) + stats["sessions_on_disk_without_tool_calls"] = len(on_disk) - len(seen_sessions) + catalog = { + "basis": "induced_from_corpus_usage", + "why": ( + "Transcripts record tool names and arguments, never the schema block " + "sent to the API. The harness version field cannot substitute: all 399 " + "transcripts spanning six weeks stamp one version, which is not " + "credible as a per-session build. These schemas are induced from " + "observed usage and must not be read as captured schemas." + ), + "tools": { + tool: { + "calls": tool_calls[tool], + "arguments": { + key: { + "seen": n, + "fraction_of_calls": round(n / tool_calls[tool], 3), + "types": dict(tool_types[tool][key]), + } + for key, n in tool_keys[tool].most_common() + }, + } + for tool in sorted(tool_calls) + }, + } + return { + "stats": stats, + "catalog": catalog, + "tools_by_transcript": tools_by_transcript, + "core_tools": core_tools, + "mcp_tools_by_server": dict(mcp_tools_by_server), + # A binary seen across several unrelated transcripts demonstrably exists + # on this machine, so its first use in one session is not an invention. + # Same reasoning as tools_available: environment facts are not foresight. + "environment_binaries": { + name + for name, seen in binary_transcripts.items() + if len(seen) >= ENV_BINARY_MIN_TRANSCRIPTS + }, + } + + +def _quota(available: int) -> int: + return min( + available, + max(QUOTA_FLOOR, min(QUOTA_CEILING, round(QUOTA_FRACTION * available))), + ) + + +def select(index_path: Path) -> Tuple[Dict[str, Dict[str, Any]], Dict[str, Any]]: + """Pass 1.5 — stratified selection: use-case quota, then axis floors. + + Returns the chosen ``record_id`` values mapped to the axis and difficulty + tags computed here. Those tags depend on a record's *predecessor* and on + failures accumulated across its episode, so they can only be derived during + an ordered walk — recomputing them in pass 2, which visits transcripts in a + different grouping, would silently produce different tags. + """ + by_group: Dict[Tuple[str, str], List[Dict[str, Any]]] = defaultdict(list) + prev_by_transcript: Dict[str, Optional[Dict[str, Any]]] = {} + failures_in_episode: Dict[Tuple[str, int], int] = defaultdict(int) + + with index_path.open(encoding="utf-8") as fh: + for line in fh: + light = json.loads(line) + stem = light["transcript_path"] + previous = prev_by_transcript.get(stem) + light["capability_axes"] = axes_mod.axes_for( + light, previous, light["is_episode_final"] + ) + ep_key = (stem, light["episode_index"]) + light["difficulty"] = axes_mod.difficulty_for( + light, failures_in_episode[ep_key] + ) + if light["outcome"]["reference_quality"] == "errored": + failures_in_episode[ep_key] += 1 + prev_by_transcript[stem] = light + by_group[(light["use_case"], light["partition"])].append( + { + "record_id": light["record_id"], + "axes": light["capability_axes"], + "difficulty": light["difficulty"], + "quality": light["outcome"]["reference_quality"], + } + ) + + tags: Dict[str, Dict[str, Any]] = {} + plan: List[Dict[str, Any]] = [] + for (use_case, side), items in sorted(by_group.items()): + items.sort(key=lambda it: _rank(it["record_id"])) + quota = _quota(len(items)) + chosen: Set[str] = set() + # Axis floors first, so scarce capabilities are represented before the + # bulk fill consumes the quota with whatever is most common. + for axis in axes_mod.ALL_AXES: + have = sum( + 1 for it in items if it["record_id"] in chosen and axis in it["axes"] + ) + for it in items: + if have >= AXIS_FLOOR or len(chosen) >= quota: + break + if it["record_id"] not in chosen and axis in it["axes"]: + chosen.add(it["record_id"]) + have += 1 + for it in items: + if len(chosen) >= quota: + break + chosen.add(it["record_id"]) + for it in items: + if it["record_id"] in chosen: + tags[it["record_id"]] = { + "capability_axes": it["axes"], + "difficulty": it["difficulty"], + } + plan.append( + { + "use_case": use_case, + "partition": side, + "available": len(items), + "quota": quota, + "selected": len(chosen), + "short_of_floor": len(chosen) < QUOTA_FLOOR, + } + ) + return tags, { + "quota_rule": { + "fraction": QUOTA_FRACTION, + "floor": QUOTA_FLOOR, + "ceiling": QUOTA_CEILING, + "axis_floor": AXIS_FLOOR, + }, + "groups": plan, + } + + +class StateAccumulator: + """What the agent knew, rebuilt incrementally as a transcript is walked.""" + + def __init__(self) -> None: + self.known_paths: List[str] = [] + self._paths_seen: Set[str] = set() + self.binaries: Counter = Counter() + self.files_in_context: Dict[str, Dict[str, Any]] = {} + self.recent: List[Dict[str, Any]] = [] + self.episode_actions: Dict[int, List[Dict[str, Any]]] = defaultdict(list) + self.failures: Counter = Counter() + + def note_path(self, value: Any) -> None: + if isinstance(value, str) and value and ("/" in value or "\\" in value): + if value not in self._paths_seen and len(self._paths_seen) < 8000: + self._paths_seen.add(value) + self.known_paths.append(value) + + def note_paths_in_text(self, text: str, limit: int = 60) -> None: + """Harvest path-shaped tokens from a command or a search result.""" + if not text: + return + for match in _PATH_TOKEN.findall(text[:20000])[:limit]: + self.note_path(match.strip("\"'`,;:()[]{}")) + + def _entry(self, key: str, text: str, blobs: BlobStore, step: int) -> None: + self.files_in_context.pop(key, None) + self.files_in_context[key] = { + "blob": blobs.put(text), + "chars": len(text), + "seen_at_step": step, + } + + def _apply_edit(self, call, obs, blobs: BlobStore, scrubber: Scrubber) -> None: + """Replay a successful edit onto the held copy of the file. + + Without this the held copy stays as whatever the last Read returned, so + every later check against that file fails once the agent has edited it. + 94% of old_string failures were exactly this staleness, not a bad + argument. + """ + path = call.arguments.get("file_path") or obs.file_path + old = call.arguments.get("old_string") + new = call.arguments.get("new_string") + if not ( + isinstance(path, str) and isinstance(old, str) and isinstance(new, str) + ): + return + key = scrubber.text(path) + entry = self.files_in_context.get(key) + if not entry: + return + current = blobs.read(entry["blob"]) + scrubbed_old = scrubber.text(old) + if current is None or scrubbed_old not in current: + return + count = -1 if call.arguments.get("replace_all") else 1 + updated = current.replace(scrubbed_old, scrubber.text(new), count) + self._entry(key, updated, blobs, entry["seen_at_step"]) + + def _apply_write(self, call, obs, blobs: BlobStore, scrubber: Scrubber) -> None: + path = call.arguments.get("file_path") or obs.file_path + content = call.arguments.get("content") + if isinstance(path, str) and isinstance(content, str): + self._entry(scrubber.text(path), scrubber.text(content), blobs, -1) + + def absorb( + self, point: DecisionPoint, blobs: BlobStore, scrubber: Scrubber + ) -> None: + for call, obs in zip(point.calls, point.observations): + for key in ("file_path", "path", "notebook_path"): + self.note_path(call.arguments.get(key)) + if isinstance(call.arguments.get("command"), str): + self.note_paths_in_text(call.arguments["command"]) + for name in obs.filenames: + self.note_path(name) + if obs.ok and obs.text and call.family in ("search", "shell"): + self.note_paths_in_text(obs.text) + for seg in call.shell_segments: + self.binaries[seg["leader"]] += 1 + if obs.ok and call.tool in ("Edit", "MultiEdit"): + self._apply_edit(call, obs, blobs, scrubber) + if obs.ok and call.tool == "Write": + self._apply_write(call, obs, blobs, scrubber) + if obs.file_path and obs.file_content: + self.note_path(obs.file_path) + # Re-reading a file must refresh its position: ~39% of reads + # in this corpus are re-reads, and updating in place left the + # recency slice keeping stale entries over just-read ones. + self._entry( + scrubber.text(obs.file_path), + scrubber.text(obs.file_content), + blobs, + point.step_index, + ) + if obs.ok is False and obs.error_class: + self.failures[obs.error_class] += 1 + self.episode_actions[point.episode_index].append( + { + "tool": call.tool, + "arg_digest": scrubber.text(call.arg_digest[:200]), + "ok": obs.ok, + } + ) + self.recent.append( + { + "step_index": point.step_index, + "reasoning": scrubber.text(point.reasoning_text.strip()[:600]), + "calls": [ + {"tool": c.tool, "arguments": scrubber.value(c.arguments)} + for c in point.calls + ], + "outcomes": [ + { + "ok": o.ok, + "error_class": o.error_class, + "chars": o.chars, + "head": scrubber.text(" ".join(o.text.split())[:600]), + } + for o in point.observations + ], + } + ) + if len(self.recent) > RECENT_STEPS: + self.recent.pop(0) + + +def _materialize_observation( + obs: Any, blobs: BlobStore, scrubber: Scrubber, stats: ScrubStats +) -> Dict[str, Any]: + text = scrubber.text(obs.text or "", stats) + entry: Dict[str, Any] = { + "ok": obs.ok, + "error_class": obs.error_class, + "error_text": scrubber.text(obs.error_text, stats), + "chars": obs.chars, + "truncated": len(text) > INLINE_LIMIT, + "text": text[:INLINE_LIMIT], + "blob_ref": blobs.put(text) if len(text) > INLINE_LIMIT else None, + } + if obs.file_path: + entry["file_path"] = scrubber.text(obs.file_path, stats) + if obs.file_content: + entry["file_content_ref"] = blobs.put(scrubber.text(obs.file_content, stats)) + entry["file_content_chars"] = len(obs.file_content) + if obs.original_file: + entry["original_file_ref"] = blobs.put(scrubber.text(obs.original_file, stats)) + else: + # Absence is a size effect, not an empty file: Claude Code nulls + # originalFile above roughly 10 KB. + entry["original_file_ref"] = None + entry["original_file_basis"] = "unavailable_above_harness_size_cap" + if obs.structured_patch: + entry["structured_patch"] = scrubber.value(obs.structured_patch, stats) + return entry + + +def build_record( + point: DecisionPoint, + scan: TranscriptScan, + light: Dict[str, Any], + state: StateAccumulator, + blobs: BlobStore, + scrubber: Scrubber, + tools_available: List[str], + environment_binaries: Set[str], + use_case_secondary: List[str], + turn_annotations: List[Any], + episode_outcomes: Dict[int, Any], + previous_point: Any, + stats: ScrubStats, +) -> Dict[str, Any]: + """Assemble one full, scrubbed record.""" + _scrubbed_args = [scrubber.value(c.arguments, stats) for c in point.calls] + reasoning = point.reasoning_text.strip() + if reasoning: + status = "visible_preamble" + elif point.had_thinking_block: + status = "redacted_thinking_only" + else: + status = "none" + + record = { + "record_id": light["record_id"], + "schema_version": 2, + "session_id": point.session_id, + "message_id": point.message_id, + "scope": point.scope, + # session_id is the *parent* session throughout, so a delegated run can + # never land on the other side of the partition from its parent. That + # makes it useless for locating the file, so the transcript stem ships + # alongside it — a subagent lives in /subagents/.jsonl. + "transcript_id": scan.session_id, + "parent_session_id": point.session_id if point.scope == "subagent" else None, + "step_index": point.step_index, + # Keyed on the transcript, not the session: a parent and its subagents + # share one session_id, so keying on that gave 20 independent agent runs + # the same episode_id and merged them into one fake episode. + "episode_id": f"{scan.session_id}#{point.episode_index}", + "depth_index": point.depth_index, + "work_item": scrubber.text(light["work_item"], stats), + "partition": light["partition"], + "use_case": light["use_case"], + "use_case_secondary": use_case_secondary, + "capability_axes": light["capability_axes"], + "difficulty": light["difficulty"], + "goal": scrubber.free_text(scan.first_prompt, stats), + "episode_instruction": scrubber.free_text( + ( + scan.episode_prompts[point.episode_index] + if point.episode_index < len(scan.episode_prompts) + else scan.first_prompt + ), + stats, + ), + "tools_available": tools_available, + "tools_available_basis": "induced_from_corpus_usage", + "reasoning": { + "visible_text": scrubber.text(reasoning, stats), + "thinking": None, + "thinking_basis": "unavailable_encrypted_by_harness", + "status": status, + # The model's own reasoning is encrypted corpus-wide. This is a + # reconstruction of the decision *context* from observable state, and + # says so in its own basis field. + "inferred": annotate_mod.infer_reasoning( + point, + previous_point, + scrubber.free_text(scan.first_prompt, stats), + point.depth_index, + stated_preamble=scrubber.text(reasoning, stats), + ), + }, + "state": { + "transcript_ref": { + "session_id": point.session_id, + "transcript_id": scan.session_id, + "scope": point.scope, + "message_id": point.message_id, + "step_index": point.step_index, + }, + "cwd": scrubber.text(point.cwd, stats), + "git_branch": scrubber.text(point.git_branch, stats), + "recent_steps": state.recent[-RECENT_STEPS:], + "episode_actions": state.episode_actions[point.episode_index][-400:], + "known_paths": [ + scrubber.text(p, stats) for p in state.known_paths[-KNOWN_PATHS_CAP:] + ], + "observed_binaries": [ + scrubber.text(b, stats) + for b, _ in state.binaries.most_common(OBSERVED_BINARIES_CAP) + ], + # Present on the machine, evidenced corpus-wide. Distinct from + # observed_binaries, which is only what *this* run has reached for. + "environment_binaries": sorted(environment_binaries), + "files_in_context": dict(list(state.files_in_context.items())[-40:]), + "prior_failures": { + "count": sum(state.failures.values()), + "classes": dict(state.failures), + }, + }, + "action": { + "width": point.width, + "calls": [ + { + "tool": c.tool, + "family": c.family, + "arguments": _scrubbed_args[i], + # Hashed over the *scrubbed* arguments, so a consumer can + # recompute identity from what they were given. Hashing the + # originals left 58% of records carrying a hash that did not + # describe their own contents, and made two records shipping + # identical arguments look distinct. + "arg_hash": _hash_args(_scrubbed_args[i]), + "arg_digest": scrubber.text(c.arg_digest[:400], stats), + "shell_segments": [ + { + **seg, + "leader": scrubber.text(seg["leader"], stats), + "text": scrubber.text(seg["text"], stats), + } + for seg in c.shell_segments + ], + } + for i, c in enumerate(point.calls) + ], + }, + "observation": [ + _materialize_observation(o, blobs, scrubber, stats) + for o in point.observations + ], + "outcome": { + "any_error": point.any_error, + "error_classes": [ + o.error_class for o in point.observations if o.error_class + ], + "reference_quality": point.reference_quality, + }, + "repo": { + "project": scrubber.text(scan.project, stats), + "branch": scrubber.text(scan.git_branch, stats), + "commit": None, + "commit_basis": "unavailable", + "commit_verified": False, + }, + # Grading metadata, deliberately OUTSIDE `state`: it describes how the + # episode turned out, which a candidate harness must never see. Same + # status as `action` and `observation`. + "episode_turn_class": ( + turn_annotations[point.episode_index].to_dict() + if point.episode_index < len(turn_annotations) + else None + ), + "episode_outcome": ( + episode_outcomes[point.episode_index].to_dict() + if point.episode_index in episode_outcomes + else None + ), + "grading_polarity": ( + "avoid_reference" + if point.reference_quality == "errored" + else "match_reference" + ), + } + # Calibration: run the static checks over the reference's own action. Where + # the reference fails, the check cannot discriminate between harnesses on + # this record and the consumer drops it rather than scoring noise. + record["reference_checks"] = run_static_checks( + record["action"]["calls"], record, blobs.read + ) + return record + + +def materialize_pass( + session_ids: List[str], + projects_root: Path, + secondary: Dict[str, List[str]], + index_path: Path, + tags: Dict[str, Dict[str, Any]], + tools_by_transcript: Dict[str, Set[str]], + core_tools: Set[str], + mcp_tools_by_server: Dict[str, Set[str]], + environment_binaries: Set[str], + blobs: BlobStore, + scrubber: Scrubber, +) -> Tuple[List[Dict[str, Any]], ScrubStats, Dict[str, int]]: + """Pass 2 — rebuild the selected decision points with full state.""" + selected = set(tags) + light_by_id: Dict[str, Dict[str, Any]] = {} + with index_path.open(encoding="utf-8") as fh: + for line in fh: + light = json.loads(line) + if light["record_id"] in selected: + light.update(tags[light["record_id"]]) + light_by_id[light["record_id"]] = light + + records: List[Dict[str, Any]] = [] + stats = ScrubStats() + counters = {"dropped_pasted": 0, "materialized": 0, "missing_light": 0} + + for session_id, path, scope in iter_transcripts(session_ids, projects_root): + scan = scan_transcript(path, scope=scope) + if scan is None: + continue + # Every human turn, not just the opening one. The canonical case in this + # corpus opens with "summarize the transcript I'm about to paste" and + # pastes it in the *second* turn — checking only the first prompt finds + # nothing and ships the transcript. + turn_annotations = annotate_mod.annotate_turns(scan.episode_prompts) + by_episode: Dict[int, List[Any]] = defaultdict(list) + for decision in scan.decisions: + by_episode[decision.episode_index].append(decision) + episode_outcomes = { + idx: annotate_mod.infer_episode_outcome( + points, + turn_annotations[idx + 1] if idx + 1 < len(turn_annotations) else None, + ) + for idx, points in by_episode.items() + } + paste_reason = next( + ( + reason + for reason in ( + is_pasted_third_party(p) + for p in [scan.first_prompt, *scan.episode_prompts] + ) + if reason + ), + None, + ) + state = StateAccumulator() + previous_point = None + available = tools_available_for( + tools_by_transcript.get(scan.session_id, set()), + core_tools, + mcp_tools_by_server, + ) + for point in scan.decisions: + point.session_id = session_id + record_id = _sha(point.session_id + "|" + point.message_id)[:16] + if record_id in selected: + light = light_by_id.get(record_id) + if light is None: + counters["missing_light"] += 1 + elif paste_reason: + # A regex cannot anonymise arbitrary proper nouns, so records + # built on pasted third-party prose are dropped, not cleaned. + counters["dropped_pasted"] += 1 + stats.bump("dropped_pasted_third_party") + else: + records.append( + build_record( + point, + scan, + light, + state, + blobs, + scrubber, + available, + environment_binaries, + secondary.get(session_id[:8], []), + turn_annotations, + episode_outcomes, + previous_point, + stats, + ) + ) + counters["materialized"] += 1 + state.absorb(point, blobs, scrubber) + previous_point = point + return records, stats, counters + + +def write_dataset( + out_dir: Path, records: List[Dict[str, Any]], flagged: Set[str] +) -> Dict[str, int]: + counts = {part.ORACLE: 0, part.POOL: 0} + handles = {} + for side in (part.ORACLE, part.POOL): + (out_dir / side).mkdir(parents=True, exist_ok=True) + handles[side] = (out_dir / side / "records.jsonl").open("w", encoding="utf-8") + try: + for rec in records: + rec["near_dup_in_other_partition"] = rec["record_id"] in flagged + rec["record_sha256"] = _sha( + json.dumps(rec, sort_keys=True, ensure_ascii=False) + ) + handles[rec["partition"]].write(json.dumps(rec, ensure_ascii=False) + "\n") + counts[rec["partition"]] += 1 + finally: + for fh in handles.values(): + fh.close() + (out_dir / part.ORACLE / "README.md").write_text( + "# Held-out oracle — do not tune against this\n\n" + "Records here are selected by " + "`int(sha256(session_id)[:8], 16) % 100 < 30`, before anything was " + "generated, so a harness tuned on `pool/` has not seen these sessions.\n\n" + "Two honest limits:\n\n" + "- This is an **auto-mined** oracle, not a human-curated one. Treat it as a " + "regression detector, never as a release gate.\n" + "- 58% of sessions share a repository branch with a session in `pool/`. " + "Specific actions barely leak (0.9% overlap), but codebase familiarity " + "does. See `../contamination.json`.\n", + encoding="utf-8", + ) + return counts + + +def summarize(records: List[Dict[str, Any]]) -> Dict[str, Any]: + by_use_case: Dict[str, Counter] = defaultdict(Counter) + by_axis: Counter = Counter() + by_difficulty: Counter = Counter() + reasoning: Counter = Counter() + polarity: Counter = Counter() + obs_recoverable = 0 + obs_total = 0 + file_content = 0 + original_file = 0 + edit_calls = 0 + for rec in records: + by_use_case[rec["use_case"]][rec["partition"]] += 1 + for axis in rec["capability_axes"]: + by_axis[axis] += 1 + by_difficulty[rec["difficulty"]] += 1 + reasoning[rec["reasoning"]["status"]] += 1 + polarity[rec["grading_polarity"]] += 1 + for call, obs in zip(rec["action"]["calls"], rec["observation"]): + obs_total += 1 + if obs.get("text") or obs.get("blob_ref") or obs.get("file_content_ref"): + obs_recoverable += 1 + if obs.get("file_content_ref"): + file_content += 1 + if call["tool"] in ("Edit", "MultiEdit"): + edit_calls += 1 + if obs.get("original_file_ref"): + original_file += 1 + baseline = {name: {"applicable": 0, "passed": 0} for name in STATIC_CHECKS} + for rec in records: + for name, res in rec.get("reference_checks", {}).items(): + if res.get("applicable"): + baseline[name]["applicable"] += 1 + baseline[name]["passed"] += 1 if res.get("passed") else 0 + for name, b in baseline.items(): + b["pct_reference_passes"] = round( + 100.0 * b["passed"] / max(b["applicable"], 1), 1 + ) + return { + "records": len(records), + "reference_check_baseline": baseline, + "by_use_case": {k: dict(v) for k, v in sorted(by_use_case.items())}, + "by_capability_axis": dict(by_axis.most_common()), + "by_difficulty": dict(by_difficulty), + "reasoning_status": dict(reasoning), + "grading_polarity": dict(polarity), + "observations": { + "total_tool_calls": obs_total, + "with_recoverable_observation": obs_recoverable, + "pct_recoverable": round(100.0 * obs_recoverable / max(obs_total, 1), 1), + "with_file_content": file_content, + "edit_calls": edit_calls, + "edit_calls_with_original_file": original_file, + "pct_edits_with_original_file": round( + 100.0 * original_file / max(edit_calls, 1), 1 + ), + }, + } + + +def main(argv: Optional[List[str]] = None) -> int: + parser = argparse.ArgumentParser(description=__doc__.split("\n")[0]) + parser.add_argument( + "--snapshot", + type=Path, + default=Path.home() / ".gaia" / "cache" / "factory", + help=( + "Directory holding traces.jsonl and labels.txt. Defaults to where " + "harvest.scan writes. Point it at a frozen copy for reproducibility." + ), + ) + parser.add_argument( + "--projects", + type=Path, + default=Path.home() / ".claude" / "projects", + help="Raw Claude Code transcript root.", + ) + parser.add_argument( + "--out", + type=Path, + default=Path.home() / ".gaia" / "cache" / "factory" / "dataset", + help="Output directory. Must be outside every repository working tree.", + ) + parser.add_argument( + "--username", + default=os.environ.get("USERNAME") or os.environ.get("USER") or "", + help="Account name owning the corpus; required for path anonymisation.", + ) + parser.add_argument( + "--extra-name", + action="append", + default=[], + help=( + "An additional literal to redact (repeatable). Use for colleagues named " + "in prompts: a regex cannot infer which capitalised words are people, so " + "known names must be supplied rather than guessed." + ), + ) + args = parser.parse_args(argv) + + if (args.out / ".git").exists() or any( + (p / ".git").exists() for p in args.out.parents + ): + raise SystemExit( + f"Refusing to write dataset into a git working tree: {args.out}. " + "Nothing derived from a session corpus belongs in a repository. " + "Point --out at a path under ~/.gaia/cache/." + ) + + scrubber = Scrubber(username=args.username, extra_names=args.extra_name) + args.out.mkdir(parents=True, exist_ok=True) + work = args.out / ".work" + work.mkdir(exist_ok=True) + index_path = work / "index.jsonl" + + session_ids, primary, secondary = load_snapshot(args.snapshot) + print(f"[1/5] indexing {len(session_ids)} sessions from {args.snapshot} …") + indexed = index_pass(session_ids, args.projects, primary, index_path) + print( + f" {indexed['stats']['decision_points']} decision points from " + f"{indexed['stats']['sessions_with_decision_points']} sessions " + f"({indexed['stats']['sessions_pruned_from_disk']} pruned from disk, " + f"{indexed['stats']['sessions_on_disk_without_tool_calls']} with no tool calls)" + ) + + print("[2/5] selecting …") + tags, sampling = select(index_path) + print(f" selected {len(tags)} records") + + print("[3/5] materializing + scrubbing …") + blobs = BlobStore(args.out / "blobs") + records, scrub_stats, counters = materialize_pass( + session_ids, + args.projects, + secondary, + index_path, + tags, + indexed["tools_by_transcript"], + indexed["core_tools"], + indexed["mcp_tools_by_server"], + indexed["environment_binaries"], + blobs, + scrubber, + ) + print( + f" {counters['materialized']} materialized, " + f"{counters['dropped_pasted']} dropped as pasted third-party content" + ) + + print("[4/5] measuring contamination …") + contamination, flagged = part.measure_contamination(records) + audit = part.build_audit(session_ids) + + print("[5/5] writing …") + counts = write_dataset(args.out, records, flagged) + (args.out / "partition_audit.json").write_text( + json.dumps(audit.to_dict(), indent=2), encoding="utf-8" + ) + (args.out / "contamination.json").write_text( + json.dumps(contamination, indent=2), encoding="utf-8" + ) + (args.out / "tool_catalog.json").write_text( + json.dumps(indexed["catalog"], indent=2), encoding="utf-8" + ) + (args.out / "scrub_report.json").write_text( + json.dumps( + { + "rules_fired": dict(sorted(scrub_stats.counts.items())), + "records_dropped_pasted_third_party": counters["dropped_pasted"], + "note": ( + "Rule counts are replacements made, not distinct secrets. " + "A regex cannot remove arbitrary proper nouns; records built " + "on pasted third-party prose are dropped rather than cleaned." + ), + }, + indent=2, + ), + encoding="utf-8", + ) + manifest = { + "build": { + "snapshot": str(args.snapshot), + "projects_root": str(args.projects), + "schema_version": 1, + }, + "extraction": indexed["stats"], + "sampling": sampling, + "partition_counts": counts, + "summary": summarize(records), + "blobs": {"objects": len(blobs.written), "bytes": blobs.bytes_written}, + "honesty": [ + "No ground truth for task success: no transcript records whether the " + "human's goal was met. No verifier here claims otherwise.", + "The reference action is what one strong harness did, not what was " + "optimal. Divergence is divergence, never accuracy.", + "Extended thinking is encrypted corpus-wide, so reasoning-quality " + "grading is not supported and no LLM judge ships.", + "Tool schemas are induced from usage; the harness version field is " + "constant across the corpus and cannot date them.", + "Friction signals upstream of this dataset are regex proxies.", + "Cost figures in the source analysis are API-equivalent, not spend.", + "Session duration is unusable past the median.", + ], + } + (args.out / "MANIFEST.json").write_text( + json.dumps(manifest, indent=2), encoding="utf-8" + ) + shutil.rmtree(work, ignore_errors=True) + # A dataset that breaks its own invariants is worse than none: it scores a + # harness against corrupted state and reports a number anyway. + audit_mod.assert_clean(args.out) + verify_mod.assert_no_leaks(args.out, args.username) + print(json.dumps(manifest["summary"], indent=2)[:2000]) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/src/gaia/factory/dataset/commits.py b/src/gaia/factory/dataset/commits.py new file mode 100644 index 0000000000..6bbf21831f --- /dev/null +++ b/src/gaia/factory/dataset/commits.py @@ -0,0 +1,206 @@ +# Copyright(C) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. +# SPDX-License-Identifier: MIT +"""Measure whether a session's commit can be recovered — and proven right. + +A record references files at a point in history, so the obvious move is to read +them back with ``git show :``. Transcripts record ``gitBranch`` +and a timestamp but never a commit, so the commit has to be inferred with +``git rev-list -n1 --before= ``. + +**That inference produces confident wrong answers.** On this corpus it returned +the *same* commit for four different session branches, because those branches +were merged and the walk lands on a shared ancestor. So it is never trusted on +its own: a recovered commit counts only when content read back from it +byte-matches content the transcript independently holds for the same file. + +This module measures that verification rate. It writes nothing into the dataset +— the decision it informs is whether ``repo.commit`` is a usable field at all. +""" + +import subprocess +from dataclasses import dataclass, field +from pathlib import Path +from typing import Dict, List, Optional, Tuple + +from gaia.factory.dataset.extract import iter_transcripts, scan_transcript + +#: A git call that has not answered in this long is not going to. +_TIMEOUT = 20 + + +def _git(repo: Path, *args: str) -> Optional[str]: + """Run git, returning stdout or ``None``. Never raises on a git-level failure. + + A missing branch, a pruned worktree and a deleted repository are all expected + here — this module's whole job is to find out how often that happens — so + they are counted, not raised. A missing ``git`` binary is a different thing + and is allowed to surface. + """ + try: + proc = subprocess.run( + ["git", "-C", str(repo), *args], + capture_output=True, + text=True, + # Repository content is UTF-8; without this, Python decodes with the + # locale codec (cp1252 on Windows) and one undecodable byte kills the + # whole measurement rather than one file. + encoding="utf-8", + errors="replace", + timeout=_TIMEOUT, + check=False, + ) + except subprocess.TimeoutExpired: + return None + if proc.returncode != 0: + return None + return proc.stdout + + +def find_repo_root(cwd: str) -> Optional[Path]: + """Walk up from a recorded working directory to a git root that still exists.""" + if not cwd: + return None + path = Path(cwd) + for candidate in [path, *path.parents]: + if (candidate / ".git").exists(): + return candidate + return None + + +def infer_commit(repo: Path, branch: str, timestamp: str) -> Optional[str]: + """Best guess at the commit live at ``timestamp`` on ``branch``. Unverified.""" + if not branch or branch in ("HEAD", ""): + return None + out = _git(repo, "rev-list", "-n", "1", f"--before={timestamp}", branch) + if not out: + return None + commit = out.strip().splitlines() + return commit[0] if commit else None + + +def verify_commit( + repo: Path, commit: str, file_path: str, expected: str +) -> Optional[bool]: + """Does ``file_path`` at ``commit`` match what the transcript says was read? + + Returns ``None`` when the file is not in that commit at all — absence is not + disagreement, and counting it as a mismatch would understate the rate. + """ + try: + relative = Path(file_path).resolve().relative_to(repo.resolve()) + except (ValueError, OSError): + return None + out = _git(repo, "show", f"{commit}:{relative.as_posix()}") + if out is None: + return None + return out.replace("\r\n", "\n").strip() == expected.replace("\r\n", "\n").strip() + + +@dataclass +class RecoveryStats: + """How far commit recovery got, stage by stage.""" + + sessions: int = 0 + repo_root_found: int = 0 + repo_root_missing: int = 0 + branch_unusable: int = 0 + commit_inferred: int = 0 + verification_attempted: int = 0 + verified_match: int = 0 + verified_mismatch: int = 0 + file_absent_at_commit: int = 0 + no_transcript_content_to_check: int = 0 + distinct_commits: Dict[str, int] = field(default_factory=dict) + + def to_dict(self) -> Dict[str, object]: + attempted = max(self.verification_attempted, 1) + collisions = sum(1 for n in self.distinct_commits.values() if n > 1) + return { + "sessions_sampled": self.sessions, + "repo_root_found": self.repo_root_found, + "repo_root_missing": self.repo_root_missing, + "branch_unusable": self.branch_unusable, + "commit_inferred": self.commit_inferred, + "verification_attempted": self.verification_attempted, + "verified_match": self.verified_match, + "verified_mismatch": self.verified_mismatch, + "file_absent_at_commit": self.file_absent_at_commit, + "no_transcript_content_to_check": self.no_transcript_content_to_check, + "pct_verified_of_attempted": round( + 100.0 * self.verified_match / attempted, 1 + ), + "pct_verified_of_sessions": round( + 100.0 * self.verified_match / max(self.sessions, 1), 1 + ), + "commits_claimed_by_more_than_one_session": collisions, + "note": ( + "A commit counts as recovered only when content read back from it " + "byte-matches content the transcript independently holds. An " + "inferred-but-unverified commit is a confident guess: rev-list " + "--before returns a shared ancestor for merged branches, so " + "several sessions resolve to one commit." + ), + } + + +def measure_recovery( + session_ids: List[str], projects_root: Path, limit: int = 60 +) -> Tuple[RecoveryStats, List[Dict[str, object]]]: + """Attempt commit recovery for up to ``limit`` sessions and report the rate.""" + stats = RecoveryStats() + detail: List[Dict[str, object]] = [] + seen: set = set() + + for session_id, path, scope in iter_transcripts(session_ids, projects_root): + if scope != "main" or session_id in seen or len(seen) >= limit: + continue + seen.add(session_id) + scan = scan_transcript(path) + if scan is None: + continue + stats.sessions += 1 + + repo = find_repo_root(scan.cwd) + if repo is None: + stats.repo_root_missing += 1 + detail.append({"session": session_id[:8], "outcome": "repo_root_missing"}) + continue + stats.repo_root_found += 1 + + commit = infer_commit(repo, scan.git_branch, scan.started_at) + if not commit: + stats.branch_unusable += 1 + detail.append({"session": session_id[:8], "outcome": "branch_unusable"}) + continue + stats.commit_inferred += 1 + stats.distinct_commits[commit] = stats.distinct_commits.get(commit, 0) + 1 + + probe = None + for point in scan.decisions: + for obs in point.observations: + if obs.file_path and obs.file_content and obs.ok: + probe = (obs.file_path, obs.file_content) + break + if probe: + break + if probe is None: + stats.no_transcript_content_to_check += 1 + detail.append( + {"session": session_id[:8], "outcome": "nothing_to_verify_against"} + ) + continue + + stats.verification_attempted += 1 + result = verify_commit(repo, commit, probe[0], probe[1]) + if result is None: + stats.file_absent_at_commit += 1 + outcome = "file_absent_at_commit" + elif result: + stats.verified_match += 1 + outcome = "verified" + else: + stats.verified_mismatch += 1 + outcome = "mismatch" + detail.append({"session": session_id[:8], "outcome": outcome}) + + return stats, detail diff --git a/src/gaia/factory/dataset/extract.py b/src/gaia/factory/dataset/extract.py new file mode 100644 index 0000000000..ce28e5532f --- /dev/null +++ b/src/gaia/factory/dataset/extract.py @@ -0,0 +1,498 @@ +# Copyright(C) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. +# SPDX-License-Identifier: MIT +"""Turn a raw Claude Code transcript into decision points. + +``harvest.reader`` normalizes a session into ordered ``Step`` objects and is the +right tool for counting. It deliberately drops three things a step-level eval +dataset needs, all of which live only in the raw JSONL: + +* **The reasoning preamble.** Claude Code writes one JSONL record per *content + block*, so a message's ``text`` and its ``tool_use`` land in different records. + Grouping by ``message.id`` is the only way to see them together — a per-record + scan finds zero co-occurrence and concludes, wrongly, that no reasoning exists. +* **The observation payload.** ``toolUseResult`` carries the file content a + ``Read`` returned, a ``Bash`` command's stdout/stderr, and an ``Edit``'s + structured patch. That is the state the agent actually saw. +* **Episode structure.** Which human turn a step descends from, and how deep into + that turn's dependent chain it sits. + +This module adds those without re-implementing what ``reader`` already gets right: +tool families, full-argument identity hashing, and the ``isMeta`` filter. +""" + +import json +import re +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any, Dict, Iterator, List, Optional, Tuple + +from gaia.factory.harvest.analyze import classify_error +from gaia.factory.harvest.reader import _digest_args, _hash_args, tool_family + +#: Sequential shell separators. Pipes are excluded on purpose: ``grep x | head`` +#: spawns two processes but is one dependent dataflow producing one answer, and +#: counting it as two actions flatters both the width and depth axes. +_SEGMENT_SEPARATORS = (";", "&&", "||", "\n") + +#: Segment leaders that exist to rebuild state the runtime discarded, rather than +#: to do the work. 46% of raw segments in the corpus are these. +_SCAFFOLDING = frozenset({"cd", "echo", "pwd", "export", "set", "source", "."}) + +#: ``VAR=value`` opening a segment — a shell assignment, not a binary. +_ASSIGNMENT = re.compile(r"^([A-Za-z_][A-Za-z0-9_]*)=") + +#: A token a shell could actually execute: a bare command name, or a path to one. +#: Anything with a quote, paren or bracket in it is a fragment of a script body. +_BINARY_TOKEN = re.compile(r"^[A-Za-z0-9_][A-Za-z0-9_.+-]*$|^[./~][A-Za-z0-9_./+-]*$") + +#: ``< int: + return len(self.calls) + + @property + def any_error(self) -> bool: + return any(o.ok is False for o in self.observations) + + @property + def reference_quality(self) -> str: + """``errored`` flips grading polarity — see PLAN.md §6.3.""" + if any(o.ok is False for o in self.observations): + return "errored" + if all(o.ok is None for o in self.observations): + return "unresolved" + return "succeeded" + + +@dataclass +class TranscriptScan: + """Everything one transcript file yields.""" + + session_id: str + project: str + scope: str + cwd: str = "" + git_branch: str = "" + started_at: str = "" + first_prompt: str = "" + episode_prompts: List[str] = field(default_factory=list) + decisions: List[DecisionPoint] = field(default_factory=list) + skipped_lines: int = 0 + + +def split_shell_segments(command: str) -> List[Dict[str, str]]: + """Split a shell command on its *sequential* separators only. + + §6c of the corpus analysis: ``Bash`` is 54.6% of tool calls and carries a + median of 2 substantive segments, so a raw tool-call count understates + executed actions ~2.5x. The decision, though, is still one decision — the + model composed the whole pipeline in a single forward pass — so segments are + recorded as an attribute of the action rather than as separate records. + + Segments inside a heredoc or a quoted ``python -c`` body are marked + ``script_body``. 15.0% of shell commands in this corpus carry an inline + script, and splitting one on ``;``/newline as though it were shell yields + "binaries" like ``open(p`` and ``encoding="utf-8``. The source analysis + records the same ~7% residual noise; left in, it also made well-formed + commands look like they invoked programs that do not exist. + """ + if not command: + return [] + parts: List[str] = [command] + for sep in _SEGMENT_SEPARATORS: + nxt: List[str] = [] + for part in parts: + nxt.extend(part.split(sep)) + parts = nxt + + segments: List[Dict[str, str]] = [] + heredoc_terminator: Optional[str] = None + open_quote = False + + for raw in parts: + text = raw.strip() + if not text: + continue + + in_body = heredoc_terminator is not None or open_quote + if heredoc_terminator is not None and text.strip() == heredoc_terminator: + heredoc_terminator = None + segments.append( + {"leader": text[:80], "kind": "script_body", "text": text[:600]} + ) + continue + + leader = text.split()[0].lstrip("(").lstrip("{") + assignment = _ASSIGNMENT.match(leader) + if in_body: + kind = "script_body" + elif assignment: + # A leading VAR=value is an environment assignment, not a binary. + # Keeping the whole token put the assigned value — routinely an + # absolute path — into the binary vocabulary, unscrubbed. + leader = assignment.group(1) + kind = "scaffolding" + elif leader.startswith("#"): + kind = "control" + elif leader in _CONTROL: + kind = "control" + elif leader in _SCAFFOLDING: + kind = "scaffolding" + elif not _BINARY_TOKEN.match(leader): + # Not a token any shell could execute — a stray fragment of a script + # body the quote tracker did not catch. + kind = "script_body" + else: + kind = "substantive" + + if heredoc_terminator is None: + match = _HEREDOC.search(text) + if match: + heredoc_terminator = match.group(1) or match.group(2) or match.group(3) + open_quote = _has_unbalanced_quote(text) != open_quote + + segments.append({"leader": leader[:80], "kind": kind, "text": text[:600]}) + return segments + + +def _has_unbalanced_quote(text: str) -> bool: + """True when a segment leaves a quote open, so the next one is script body.""" + single = double = 0 + escaped = False + for char in text: + if escaped: + escaped = False + continue + if char == "\\": + escaped = True + elif char == "'" and double % 2 == 0: + single += 1 + elif char == '"' and single % 2 == 0: + double += 1 + return (single % 2 == 1) or (double % 2 == 1) + + +def _blocks(content: Any, want: str) -> Iterator[Dict[str, Any]]: + if not isinstance(content, list): + return + for block in content: + if isinstance(block, dict) and block.get("type") == want: + yield block + + +def _text_of(content: Any) -> str: + if isinstance(content, str): + return content + return "\n".join(b.get("text", "") for b in _blocks(content, "text")) + + +def _observation_from(block: Dict[str, Any], tool_result: Any) -> Observation: + """Build an Observation from a ``tool_result`` block plus ``toolUseResult``. + + The two carry different things: the block has the error flag and the text the + model saw; ``toolUseResult`` has the structured payload (file content, + stdout/stderr, patch hunks) that makes a record replayable. + """ + content = block.get("content") + if isinstance(content, str): + text, chars = content, len(content) + else: + parts = [str(b.get("text", "")) for b in _blocks(content, "text")] + text = "\n".join(parts) + chars = sum(len(p) for p in parts) + + flat = " ".join(text.split()) + ok = not block.get("is_error") + if "[Request interrupted" in flat: + ok = False + obs = Observation(ok=ok, chars=chars, text=text) + if not ok: + obs.error_class = classify_error(flat[:2000]) + obs.error_text = flat[:300] + + if isinstance(tool_result, dict): + stdout = tool_result.get("stdout") + stderr = tool_result.get("stderr") + if isinstance(stdout, str) and stdout and not obs.text: + obs.text = stdout + obs.chars = len(stdout) + if isinstance(stderr, str) and stderr and not obs.text: + obs.text = stderr + obs.chars = len(stderr) + file_block = tool_result.get("file") + if isinstance(file_block, dict): + obs.file_path = str(file_block.get("filePath") or "") + fc = file_block.get("content") + if isinstance(fc, str): + obs.file_content = fc + if not obs.file_path: + obs.file_path = str(tool_result.get("filePath") or "") + original = tool_result.get("originalFile") + # Claude Code nulls originalFile above roughly 10 KB, so its absence is + # a size effect, not an error — callers must treat it as unavailable + # rather than as "the file was empty". + if isinstance(original, str) and original: + obs.original_file = original + patch = tool_result.get("structuredPatch") + if isinstance(patch, list) and patch: + obs.structured_patch = patch + if not obs.text and isinstance(tool_result.get("content"), str): + obs.text = tool_result["content"] + obs.chars = len(obs.text) + names = tool_result.get("filenames") + if isinstance(names, list): + obs.filenames = [str(n) for n in names if isinstance(n, str)][:400] + return obs + + +def scan_transcript(path: Path, scope: str = "main") -> Optional[TranscriptScan]: + """Parse one transcript into ordered decision points. + + Returns ``None`` when the file has no tool-dispatching assistant message — + a pure question-and-answer session carries no procedure to evaluate. + """ + scan = TranscriptScan(session_id=path.stem, project=path.parent.name, scope=scope) + + records: List[Dict[str, Any]] = [] + with path.open("r", encoding="utf-8", errors="replace") as fh: + for line in fh: + line = line.strip() + if not line: + continue + try: + rec = json.loads(line) + except json.JSONDecodeError: + scan.skipped_lines += 1 + continue + if isinstance(rec, dict): + records.append(rec) + else: + scan.skipped_lines += 1 + + # Pass 1 — results, keyed by tool_use id. A result always arrives after its + # call, so this cannot be folded into the ordered walk below. + results: Dict[str, Observation] = {} + for rec in records: + if rec.get("type") != "user": + continue + message = rec.get("message") + if not isinstance(message, dict): + continue + for block in _blocks(message.get("content"), "tool_result"): + use_id = block.get("tool_use_id", "") + if use_id: + results[use_id] = _observation_from(block, rec.get("toolUseResult")) + + # Pass 2 — ordered walk building episodes and decision points. + grouped: Dict[str, DecisionPoint] = {} + episode_index = -1 + depth = 0 + step_index = 0 + + for rec in records: + if rec.get("cwd") and not scan.cwd: + scan.cwd = rec["cwd"] + if rec.get("gitBranch") and not scan.git_branch: + scan.git_branch = rec["gitBranch"] + if rec.get("timestamp") and not scan.started_at: + scan.started_at = rec["timestamp"] + + message = rec.get("message") + if not isinstance(message, dict): + continue + content = message.get("content") + rtype = rec.get("type") + + if rtype == "user": + if any(True for _ in _blocks(content, "tool_result")): + continue + # Harness-injected turns are not human intent; isMeta is the + # authoritative marker and a startswith("<") test misses most of them. + if rec.get("isMeta"): + continue + text = _text_of(content).strip() + if not text or text.startswith("<") or "[Request interrupted" in text: + continue + episode_index += 1 + depth = 0 + scan.episode_prompts.append(text) + if not scan.first_prompt: + scan.first_prompt = text + continue + + if rtype != "assistant" or not isinstance(content, list): + continue + + key = message.get("id") or rec.get("uuid") or "" + point = grouped.get(key) + if point is None: + point = DecisionPoint( + session_id=scan.session_id, + message_id=key, + scope=scope, + project=scan.project, + # step_index and depth_index are deliberately left at -1 and + # assigned when the point is appended below. Capturing them here + # takes the index the counter happens to hold when a message's + # *text* block arrives, and a message whose tool_use lands after + # another message's records then carries a stale index — two + # decision points end up sharing one step_index. + step_index=-1, + episode_index=max(episode_index, 0), + depth_index=-1, + cwd=rec.get("cwd", "") or scan.cwd, + git_branch=rec.get("gitBranch", "") or scan.git_branch, + timestamp=rec.get("timestamp", ""), + ) + grouped[key] = point + + for block in _blocks(content, "text"): + point.reasoning_text += block.get("text", "") + for block in _blocks(content, "thinking"): + point.had_thinking_block = True + # Extended thinking is encrypted corpus-wide: every block carries an + # empty string and a signature. Kept only as a flag. + for block in _blocks(content, "tool_use"): + args = block.get("input", {}) + if not isinstance(args, dict): + args = {"_raw": args} + call = ToolCall( + tool=block.get("name", "unknown"), + family=tool_family(block.get("name", "unknown")), + arguments=args, + arg_hash=_hash_args(args), + arg_digest=_digest_args(args), + ) + if call.tool == "Bash" and isinstance(args.get("command"), str): + call.shell_segments = split_shell_segments(args["command"]) + point.calls.append(call) + point.observations.append(results.get(block.get("id", ""), Observation())) + if len(point.calls) == 1: + point.step_index = step_index + point.depth_index = depth + point.episode_index = max(episode_index, 0) + scan.decisions.append(point) + step_index += 1 + depth += 1 + + if not scan.decisions: + return None + return scan + + +def iter_transcripts( + session_ids: List[str], projects_root: Path +) -> Iterator[Tuple[str, Path, str]]: + """Yield ``(session_id, path, scope)`` for wanted sessions and their subagents. + + A delegated run lives in ``//subagents/`` and a plain + ``*/*.jsonl`` glob misses it entirely — they carry 42.9% of all tool work in + this corpus, so missing them would silently halve the dataset. + """ + wanted = set(session_ids) + for path in sorted(projects_root.glob("*/*.jsonl")): + if path.stem not in wanted: + continue + yield path.stem, path, "main" + sub_dir = path.parent / path.stem / "subagents" + if sub_dir.is_dir(): + for sub_path in sorted(sub_dir.glob("*.jsonl")): + yield path.stem, sub_path, "subagent" diff --git a/src/gaia/factory/dataset/harness.py b/src/gaia/factory/dataset/harness.py new file mode 100644 index 0000000000..405805826b --- /dev/null +++ b/src/gaia/factory/dataset/harness.py @@ -0,0 +1,358 @@ +# Copyright(C) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. +# SPDX-License-Identifier: MIT +"""Present a record to an agent under test, and score what comes back. + +The one place that decides what a candidate harness may see. Keeping it here +rather than in each consumer's glue code means every agent is asked the same +question, and that no consumer accidentally hands over ``action`` — which would +make the whole exercise measure nothing. + +**Withheld, always:** ``action``, ``observation``, ``outcome``, +``reference_checks``, ``episode_outcome``, ``grading_polarity``. + +The trimmed view is the default. Full state averages ~8.5K tokens a record, +mostly file contents and a 12-step history the agent does not need to choose one +next action; trimming halves that with no loss of decision-relevant information. +""" + +import argparse +import hashlib +import json +import re +import sys +from collections import Counter, defaultdict +from pathlib import Path +from typing import Any, Callable, Dict, List, Optional, Sequence + +from gaia.factory.dataset.axes import ALL_AXES +from gaia.factory.dataset.verifiers import STATIC_CHECKS, grade + +#: Trimmed-view budgets. Chosen from measured token cost, not taste. +RECENT_STEPS_SHOWN = 6 +ARG_CHARS = 300 +HEAD_CHARS = 250 +KNOWN_PATHS_SHOWN = 120 +BINARIES_SHOWN = 60 +FILES_SHOWN = 15 + + +def prompt_view(record: Dict[str, Any], trim: bool = True) -> Dict[str, Any]: + """Exactly what an agent under test is allowed to see.""" + state = record["state"] + if not trim: + shown = state + else: + shown = { + "cwd": state["cwd"], + "git_branch": state["git_branch"], + "recent_steps": [ + { + "step_index": step["step_index"], + "reasoning": step["reasoning"][:ARG_CHARS], + "calls": [ + { + "tool": call["tool"], + "arguments": { + k: str(v)[:ARG_CHARS] + for k, v in call["arguments"].items() + }, + } + for call in step["calls"] + ], + "outcomes": [ + { + "ok": out["ok"], + "error_class": out["error_class"], + "chars": out["chars"], + "head": out["head"][:HEAD_CHARS], + } + for out in step["outcomes"] + ], + } + for step in state["recent_steps"][-RECENT_STEPS_SHOWN:] + ], + "episode_actions": state["episode_actions"][-40:], + "known_paths": state["known_paths"][-KNOWN_PATHS_SHOWN:], + "observed_binaries": state["observed_binaries"][:BINARIES_SHOWN], + "environment_binaries": state.get("environment_binaries", [])[ + :BINARIES_SHOWN + ], + # Sizes, not contents: the agent needs to know it holds the file, not + # to re-read it inside the prompt. + "files_in_context": { + k: {"chars": v["chars"], "seen_at_step": v["seen_at_step"]} + for k, v in list(state["files_in_context"].items())[-FILES_SHOWN:] + }, + "prior_failures": state["prior_failures"], + } + return { + "record_id": record["record_id"], + "goal": record["goal"], + "instruction": record["episode_instruction"], + "depth": record["depth_index"], + "tools_available": record["tools_available"], + "reasoning_so_far": record["reasoning"]["visible_text"], + "state": shown, + } + + +def select_pilot( + records: Sequence[Dict[str, Any]], n: int = 60, seed: str = "pilot" +) -> List[Dict[str, Any]]: + """A stratified slice: every capability axis represented, deterministically. + + Round-robins over the axes so the rare ones (``delegation`` at 4%, + ``error_recovery`` at 8%) are present rather than left to chance. + """ + by_axis: Dict[str, List[Dict[str, Any]]] = defaultdict(list) + for rec in records: + for axis in rec["capability_axes"]: + by_axis[axis].append(rec) + for axis in by_axis: + by_axis[axis].sort( + key=lambda r: hashlib.sha256((seed + r["record_id"]).encode()).hexdigest() + ) + chosen: Dict[str, Dict[str, Any]] = {} + cursors = {axis: 0 for axis in by_axis} + while len(chosen) < n: + progressed = False + for axis in ALL_AXES: + if len(chosen) >= n: + break + pool = by_axis.get(axis, []) + while cursors.get(axis, 0) < len(pool): + rec = pool[cursors[axis]] + cursors[axis] += 1 + if rec["record_id"] not in chosen: + chosen[rec["record_id"]] = rec + progressed = True + break + if not progressed: + break + return list(chosen.values()) + + +_JSON_BLOCK = re.compile(r"```(?:json)?\s*(.*?)```", re.S) + + +def parse_proposals(text: str) -> Dict[str, List[Dict[str, Any]]]: + """Pull ``{record_id: [calls]}`` out of an agent's reply. + + Tolerant of fenced blocks and prose around the JSON, because fighting an + agent's formatting is not the experiment. A reply that cannot be parsed + yields no entry, and the caller counts it as a non-answer rather than as a + wrong answer — they are different failures. + """ + candidates = _JSON_BLOCK.findall(text) or [text] + out: Dict[str, List[Dict[str, Any]]] = {} + for blob in candidates: + blob = blob.strip() + start = blob.find("[") + brace = blob.find("{") + if start < 0 or (0 <= brace < start): + start = brace + if start < 0: + continue + for end in range(len(blob), start, -1): + try: + parsed = json.loads(blob[start:end]) + except json.JSONDecodeError: + continue + items = parsed if isinstance(parsed, list) else [parsed] + for item in items: + if not isinstance(item, dict): + continue + rid = item.get("record_id") + calls = item.get("calls") or item.get("action") or [] + if isinstance(calls, dict): + calls = [calls] + if rid and isinstance(calls, list): + out[rid] = [ + c for c in calls if isinstance(c, dict) and c.get("tool") + ] + break + return out + + +PROPOSAL_INSTRUCTIONS = """\ +You are the agent under test. For each item below you are shown the state of a \ +coding agent partway through a task: the goal, the instruction, what it has done \ +recently, what it knows, and the tools it can use. + +Decide the SINGLE next action you would take. You may dispatch more than one tool \ +in the same step if you would genuinely run them together. + +Rules: +- Choose only from `tools_available`. +- Give complete, runnable arguments — a real command, a real path. +- Do not explain. Do not use any tools yourself. Just answer. + +Reply with ONE json array and nothing else: + +```json +[ + {"record_id": "", "calls": [{"tool": "Bash", "arguments": {"command": "..."}}]} +] +``` +""" + + +def score( + records: Sequence[Dict[str, Any]], + proposals: Dict[str, List[Dict[str, Any]]], + read_blob: Optional[Callable[[str], Optional[str]]] = None, +) -> Dict[str, Any]: + """Aggregate a run. Deliberately emits no single headline number.""" + by_axis: Dict[str, Counter] = defaultdict(Counter) + checks: Dict[str, Counter] = defaultdict(Counter) + totals: Counter = Counter() + per_record: List[Dict[str, Any]] = [] + + for rec in records: + rid = rec["record_id"] + calls = proposals.get(rid) + if not calls: + totals["no_answer"] += 1 + per_record.append({"record_id": rid, "answered": False}) + continue + totals["answered"] += 1 + result = grade(rec, calls, read_blob) + polarity = rec["grading_polarity"] + totals[f"{polarity}_n"] += 1 + if result["credited"]: + totals[f"{polarity}_credited"] += 1 + if polarity == "match_reference": + if result.get("tool_exact"): + totals["tool_exact"] += 1 + if result.get("tool_same_family"): + totals["tool_same_family"] += 1 + for axis in rec["capability_axes"]: + by_axis[axis]["n"] += 1 + by_axis[axis]["credited"] += bool(result["credited"]) + for name in STATIC_CHECKS: + check = result["checks"][name] + if check["applicable"] and result["checks_informative"][name]: + checks[name]["informative"] += 1 + checks[name]["passed"] += check["passed"] is True + per_record.append( + { + "record_id": rid, + "answered": True, + "polarity": polarity, + "credited": result["credited"], + "tool_exact": result.get("tool_exact"), + "proposed": [c.get("tool") for c in calls], + "reference": [c["tool"] for c in rec["action"]["calls"]], + } + ) + + def pct(num: int, den: int) -> Optional[float]: + return round(100.0 * num / den, 1) if den else None + + return { + "records": len(records), + "answered": totals["answered"], + "no_answer": totals["no_answer"], + "match_reference": { + "n": totals["match_reference_n"], + "credited_pct": pct( + totals["match_reference_credited"], totals["match_reference_n"] + ), + "tool_exact_pct": pct(totals["tool_exact"], totals["match_reference_n"]), + "tool_same_family_pct": pct( + totals["tool_same_family"], totals["match_reference_n"] + ), + }, + "avoid_reference": { + "n": totals["avoid_reference_n"], + "credited_pct": pct( + totals["avoid_reference_credited"], totals["avoid_reference_n"] + ), + }, + "checks": { + name: { + "informative": c["informative"], + "passed": c["passed"], + "pass_pct": pct(c["passed"], c["informative"]), + } + for name, c in checks.items() + }, + "by_axis": { + axis: { + "n": c["n"], + "credited": c["credited"], + "credited_pct": pct(c["credited"], c["n"]), + } + for axis, c in sorted(by_axis.items()) + }, + "per_record": per_record, + "caveats": [ + "Agreement, not accuracy: the reference is what one strong harness did.", + "match_reference and avoid_reference measure opposite things and are " + "never merged.", + "Checks are counted only where the reference itself passed them.", + "No claim about task success is made or implied.", + ], + } + + +def load_records(dataset: Path, partition: str = "pool") -> List[Dict[str, Any]]: + path = dataset / partition / "records.jsonl" + with path.open(encoding="utf-8") as fh: + return [json.loads(line) for line in fh if line.strip()] + + +def blob_reader(dataset: Path) -> Callable[[str], Optional[str]]: + def read(digest: str) -> Optional[str]: + path = dataset / "blobs" / f"{digest}.txt" + return ( + path.read_text(encoding="utf-8", errors="replace") + if path.is_file() + else None + ) + + return read + + +def main(argv: Optional[List[str]] = None) -> int: + parser = argparse.ArgumentParser( + description="Export prompt-only batches, or score replies." + ) + parser.add_argument( + "--dataset", type=Path, default=Path.home() / ".gaia/cache/factory/dataset" + ) + parser.add_argument("--partition", default="pool", choices=("pool", "oracle")) + parser.add_argument("--n", type=int, default=60) + parser.add_argument("--batch", type=int, default=6) + parser.add_argument( + "--out", type=Path, required=True, help="Directory for batch files." + ) + parser.add_argument("--full-state", action="store_true", help="~2x the tokens.") + args = parser.parse_args(argv) + + records = load_records(args.dataset, args.partition) + pilot = select_pilot(records, args.n) + args.out.mkdir(parents=True, exist_ok=True) + + (args.out / "selected.json").write_text( + json.dumps([r["record_id"] for r in pilot], indent=2), encoding="utf-8" + ) + batches = [pilot[i : i + args.batch] for i in range(0, len(pilot), args.batch)] + for i, batch in enumerate(batches): + payload = [prompt_view(r, trim=not args.full_state) for r in batch] + (args.out / f"batch_{i:02d}.json").write_text( + PROPOSAL_INSTRUCTIONS + "\n\n" + json.dumps(payload, indent=1), + encoding="utf-8", + ) + chars = sum( + (args.out / f"batch_{i:02d}.json").stat().st_size for i in range(len(batches)) + ) + print( + f"wrote {len(batches)} batches covering {len(pilot)} records to {args.out}\n" + f"~{chars // 4:,} input tokens total (excluding subagent system prompts)" + ) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/src/gaia/factory/dataset/partition.py b/src/gaia/factory/dataset/partition.py new file mode 100644 index 0000000000..bfa7a85242 --- /dev/null +++ b/src/gaia/factory/dataset/partition.py @@ -0,0 +1,171 @@ +# Copyright(C) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. +# SPDX-License-Identifier: MIT +"""Split sessions into a held-out oracle and a development pool. + +An LLM that writes the agent *and* its test data *and* grades the result certifies +nothing. Mining a scenario from session X and evaluating a harness tuned on +session X measures memorization, so the split has to exist before anything is +generated and has to be auditable afterwards. + +The rule is ``docs/plans/claude-session-harvest.md``'s: +``int(sha256(session_id)[:8], 16) % 100 < 30`` selects the oracle. Deterministic, +recomputable by anyone holding the session list, and independent of build order. + +**What it does not do**, measured rather than assumed: it does not stop a *work +item* (one branch in one repository) from straddling the split. On this corpus +58% of sessions share a work item with a session on the other side. The obvious +fix — assign whole work items — is unusable here because one work item is 17% of +the corpus and spans 12 use-cases, so honouring it destroys per-use-case balance +in both partitions. Instead every record carries its ``work_item`` key and +:func:`measure_contamination` quantifies the leak, so a consumer who wants the +stricter split can build it from shipped fields. +""" + +import hashlib +from dataclasses import dataclass +from typing import Any, Dict, Iterable, List, Optional, Sequence, Set, Tuple + +ORACLE = "oracle" +POOL = "pool" + +#: Percent of sessions routed to the held-out oracle. +ORACLE_SHARE = 30 + +#: Branch names that identify no particular unit of work. A session on detached +#: HEAD or on ``main`` shares nothing with the next one, so grouping them would +#: invent a work item that does not exist. +_ANONYMOUS_BRANCHES = frozenset({"", "HEAD", "main", "master", "detached"}) + + +def session_digest(session_id: str) -> str: + """The full hex digest; the audit file records it so the split is checkable.""" + return hashlib.sha256(session_id.encode("utf-8")).hexdigest() + + +def assign(session_id: str) -> str: + """``oracle`` or ``pool`` for one session id.""" + return ( + ORACLE if int(session_digest(session_id)[:8], 16) % 100 < ORACLE_SHARE else POOL + ) + + +def work_item(project: str, git_branch: Optional[str], session_id: str) -> str: + """A stable key for "the same piece of work", for contamination accounting.""" + branch = (git_branch or "").strip() + if branch in _ANONYMOUS_BRANCHES: + return "sess:" + session_id + return "wi:" + project + "\x00" + branch + + +@dataclass +class PartitionAudit: + """Recomputable evidence of how the split fell out.""" + + entries: List[Dict[str, Any]] + oracle_sessions: Set[str] + pool_sessions: Set[str] + + def to_dict(self) -> Dict[str, Any]: + return { + "rule": "int(sha256(session_id)[:8], 16) % 100 < 30 -> oracle", + "oracle_share_target_pct": ORACLE_SHARE, + "oracle_sessions": len(self.oracle_sessions), + "pool_sessions": len(self.pool_sessions), + "oracle_share_actual_pct": round( + 100.0 + * len(self.oracle_sessions) + / max(len(self.oracle_sessions) + len(self.pool_sessions), 1), + 2, + ), + "entries": self.entries, + } + + +def build_audit(session_ids: Iterable[str]) -> PartitionAudit: + entries: List[Dict[str, Any]] = [] + oracle: Set[str] = set() + pool: Set[str] = set() + for sid in sorted(session_ids): + digest = session_digest(sid) + bucket = int(digest[:8], 16) % 100 + side = ORACLE if bucket < ORACLE_SHARE else POOL + (oracle if side == ORACLE else pool).add(sid) + entries.append( + {"session_id": sid, "digest": digest, "bucket": bucket, "partition": side} + ) + return PartitionAudit(entries=entries, oracle_sessions=oracle, pool_sessions=pool) + + +def measure_contamination( + records: Sequence[Dict[str, Any]], +) -> Tuple[Dict[str, Any], Set[str]]: + """Quantify what leaks across the split, and name the records that do. + + Two independent measurements, because they answer different questions: + + * **Action overlap** — an ``(tool, arg_hash)`` pair present on both sides is a + literal answer a tuned harness could have memorised. Identity hashes the + full argument object; keying on one argument would merge different edits to + one file into a fake repeat and overstate this ~40x. + * **Work-item overlap** — sessions on both sides sharing a branch. Nothing + is memorised verbatim, but the codebase is familiar. + + Returns the report and the set of ``record_id`` values whose action appears on + the other side, so they can be flagged per record. + """ + oracle_actions: Dict[Tuple[str, str], List[str]] = {} + pool_actions: Set[Tuple[str, str]] = set() + oracle_items: Dict[str, Set[str]] = {} + pool_items: Dict[str, Set[str]] = {} + + for rec in records: + side = rec["partition"] + item = rec["work_item"] + (oracle_items if side == ORACLE else pool_items).setdefault(item, set()).add( + rec["session_id"] + ) + for call in rec["action"]["calls"]: + key = (call["tool"], call["arg_hash"]) + if side == ORACLE: + oracle_actions.setdefault(key, []).append(rec["record_id"]) + else: + pool_actions.add(key) + + overlapping = set(oracle_actions) & pool_actions + flagged: Set[str] = set() + for key in overlapping: + flagged.update(oracle_actions[key]) + + shared_items = set(oracle_items) & set(pool_items) + sessions_in_shared = sum( + len(oracle_items[i]) + len(pool_items[i]) for i in shared_items + ) + all_sessions = {r["session_id"] for r in records} + + report = { + "action_overlap": { + "oracle_distinct_actions": len(oracle_actions), + "pool_distinct_actions": len(pool_actions), + "overlapping_actions": len(overlapping), + "pct_of_oracle_actions": round( + 100.0 * len(overlapping) / max(len(oracle_actions), 1), 2 + ), + "oracle_records_flagged": len(flagged), + }, + "work_item_overlap": { + "work_items_spanning_partitions": len(shared_items), + "sessions_in_spanning_items": sessions_in_shared, + "pct_of_sessions": round( + 100.0 * sessions_in_shared / max(len(all_sessions), 1), 2 + ), + }, + "interpretation": ( + "Action overlap governs whether a tuned harness can have memorised a " + "literal answer, and is the number to quote for a step-level dataset. " + "Work-item overlap governs repository familiarity, which is real and " + "unfixable here without destroying per-use-case balance. Neither makes " + "this a release gate: an auto-mined oracle is a regression detector, " + "not a human-curated held-out set." + ), + } + return report, flagged diff --git a/src/gaia/factory/dataset/scrub.py b/src/gaia/factory/dataset/scrub.py new file mode 100644 index 0000000000..915f301489 --- /dev/null +++ b/src/gaia/factory/dataset/scrub.py @@ -0,0 +1,320 @@ +# Copyright(C) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. +# SPDX-License-Identifier: MIT +"""Strip identifying material out of anything derived from a session corpus. + +Transcripts carry absolute paths, the operator's username, branch names, hostnames, +credentials pasted into a prompt, and — worst for a regex — the names of people in a +meeting transcript somebody pasted in. Every string leaving the builder goes through +:func:`Scrubber.text`; :func:`Scrubber.value` walks nested structures. + +The rules below are deliberately aggressive: over-redacting a shell command costs a +record's usefulness, under-redacting costs a leak. What a regex fundamentally cannot +do is remove arbitrary proper nouns, so :func:`is_pasted_third_party` exists to drop +those records outright rather than pretend they were cleaned. +""" + +import re +from dataclasses import dataclass, field +from typing import Any, Dict, List, Optional, Pattern, Tuple + +# Ordered specific-before-generic: a GitHub PAT is also a long hex-ish blob, and a +# home directory is also an absolute path. First match wins, same as the error +# taxonomy in ``harvest``. +_SECRET_RULES: List[Tuple[str, Pattern[str], str]] = [ + ( + "secret_github", + re.compile(r"\bgh[pousr]_[A-Za-z0-9]{16,}\b"), + "", + ), + ( + "secret_github", + re.compile(r"\bgithub_pat_[A-Za-z0-9_]{20,}\b"), + "", + ), + ("secret_openai", re.compile(r"\bsk-[A-Za-z0-9_-]{20,}\b"), ""), + ( + "secret_anthropic", + re.compile(r"\bsk-ant-[A-Za-z0-9_-]{20,}\b"), + "", + ), + ("secret_aws", re.compile(r"\b(?:AKIA|ASIA)[0-9A-Z]{16}\b"), ""), + ( + "secret_slack", + re.compile(r"\bxox[baprs]-[A-Za-z0-9-]{10,}\b"), + "", + ), + ( + "secret_jwt", + re.compile( + r"\beyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\b" + ), + "", + ), + ( + "secret_bearer", + re.compile( + r"(?i)\b(bearer|token|api[_-]?key|password|passwd|secret)" + r"\s*[:=]\s*[\"']?([A-Za-z0-9_\-./+]{16,})[\"']?" + ), + r"\1=", + ), + ( + "secret_private_key", + re.compile( + r"-----BEGIN [A-Z ]*PRIVATE KEY-----.*?-----END [A-Z ]*PRIVATE KEY-----", + re.DOTALL, + ), + "", + ), +] + +# Meeting-transcript speaker lines, and *only* those. +# +# A bare ``Surname, Firstname`` rule was tried and withdrawn: it fired 9,722 +# times, almost all of it ordinary prose. "Multi-Agent Systems, Trust, & Outcome" +# became "Multi-Agent , , & Outcome", and an ISO working-group +# roster lost half its words. It damaged thousands of records to catch a threat +# that :func:`is_pasted_third_party` already handles by dropping the record +# whole. Names in web-fetched public documents are public attribution, not this +# corpus's private identity — the DATASHEET says so rather than the regex +# pretending otherwise. +_PERSON_RULES: List[Tuple[str, Pattern[str], str]] = [ + ( + "person_name", + re.compile( + r"^[ \t]*[A-Z][a-z]{1,20}(?:,)?\s+[A-Z][a-z]{1,20}\s+\d{1,2}:\d{2}\s*$", + re.MULTILINE, + ), + "