From bdad6301b6d01fe43066fae48836ab77e10d7621 Mon Sep 17 00:00:00 2001 From: Tig Date: Mon, 13 Jul 2026 16:50:47 -0600 Subject: [PATCH 1/2] chore(bedside): re-vendor 868e3b8; adopt ask/step in AGENTS Customer 0 for tig/bedside#6: pin vendored bedside with operator-gate CLI (ask/step), expand eval fixtures, and point Day-1 human gates at bedside ask/step (or host UI same contract) instead of multi-choice essays. Fixes #38 --- AGENTS.md | 66 +++-- BEDSIDE.md | 8 +- bedside.toml | 3 +- third_party/bedside/.github/workflows/ci.yml | 27 -- third_party/bedside/AGENTS.md | 6 +- third_party/bedside/README.md | 23 +- third_party/bedside/VENDOR.md | 4 +- third_party/bedside/bedside.toml | 17 +- third_party/bedside/contract/README.md | 4 + third_party/bedside/docs/adopting.md | 138 +++++++++ third_party/bedside/eval/README.md | 49 +++- .../fixtures/known-bad/choice-wall/meta.toml | 5 + .../known-bad/choice-wall/transcript.md | 19 ++ .../known-bad/multi-step-body-dump/meta.toml | 5 + .../multi-step-body-dump/transcript.md | 19 ++ .../known-good/operator-gate-ask/meta.toml | 5 + .../operator-gate-ask/transcript.md | 34 +++ .../known-good/operator-gate-step/meta.toml | 5 + .../operator-gate-step/transcript.md | 35 +++ .../known-good/structured-choice/meta.toml | 5 + .../structured-choice/transcript.md | 23 ++ third_party/bedside/pyproject.toml | 2 +- third_party/bedside/src/bedside/__init__.py | 2 +- third_party/bedside/src/bedside/cli.py | 173 ++++++++++- .../bedside/src/bedside/commands/ask_cmd.py | 196 +++++++++++++ .../bedside/src/bedside/commands/eval_cmd.py | 63 ++-- .../bedside/src/bedside/commands/init_cmd.py | 100 ++++++- .../bedside/src/bedside/commands/step_cmd.py | 152 ++++++++++ third_party/bedside/src/bedside/config.py | 39 ++- .../bedside/src/bedside/eval_engine.py | 42 ++- third_party/bedside/src/bedside/exit_codes.py | 5 + third_party/bedside/src/bedside/vendor.py | 94 ++++++ third_party/bedside/surface/README.md | 53 +++- third_party/bedside/tests/test_ask_step.py | 271 ++++++++++++++++++ .../bedside/tests/test_cli_commands.py | 63 +++- third_party/bedside/tests/test_eval_engine.py | 19 ++ 36 files changed, 1633 insertions(+), 141 deletions(-) delete mode 100644 third_party/bedside/.github/workflows/ci.yml create mode 100644 third_party/bedside/docs/adopting.md create mode 100644 third_party/bedside/eval/fixtures/known-bad/choice-wall/meta.toml create mode 100644 third_party/bedside/eval/fixtures/known-bad/choice-wall/transcript.md create mode 100644 third_party/bedside/eval/fixtures/known-bad/multi-step-body-dump/meta.toml create mode 100644 third_party/bedside/eval/fixtures/known-bad/multi-step-body-dump/transcript.md create mode 100644 third_party/bedside/eval/fixtures/known-good/operator-gate-ask/meta.toml create mode 100644 third_party/bedside/eval/fixtures/known-good/operator-gate-ask/transcript.md create mode 100644 third_party/bedside/eval/fixtures/known-good/operator-gate-step/meta.toml create mode 100644 third_party/bedside/eval/fixtures/known-good/operator-gate-step/transcript.md create mode 100644 third_party/bedside/eval/fixtures/known-good/structured-choice/meta.toml create mode 100644 third_party/bedside/eval/fixtures/known-good/structured-choice/transcript.md create mode 100644 third_party/bedside/src/bedside/commands/ask_cmd.py create mode 100644 third_party/bedside/src/bedside/commands/step_cmd.py create mode 100644 third_party/bedside/src/bedside/vendor.py create mode 100644 third_party/bedside/tests/test_ask_step.py diff --git a/AGENTS.md b/AGENTS.md index 03f33da..661abcb 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -14,7 +14,7 @@ Context is finite. **Do not** open every manners file into the active window. | 1 | This file (`AGENTS.md`) | Silico spine: Day 1 phases, silico CLI, plate, host/metal DoD | — | | 2 | `bedside.toml` + contract path it names | Normative portable manners (nine principles) | Already summarized below and you are not changing manners | | 3 | `BEDSIDE.md` | **Metal domain pack only** (COM, UF2, deploy identity) | Already in Day 1 metal sections of this file for the current step | -| — | `third_party/bedside/README.md`, vendored stub `AGENTS.md`/`BEDSIDE.md`, full `eval/` docs | Upstream product / scoring | Almost always — use `bedside doctor` / `bedside eval` instead of loading prose | +| — | `third_party/bedside/README.md`, vendored stub `AGENTS.md`/`BEDSIDE.md`, full `eval/` docs | Upstream product / scoring | Almost always — use `bedside doctor|eval|ask|step` instead of loading prose | | — | Full FAQ / tenets | Strategy | Only when the task is doctrine, not a metal slice | ### Canonical owner (overlap map) @@ -22,7 +22,7 @@ Context is finite. **Do not** open every manners file into the active window. | Topic | Canonical owner | Silico may hold | |-------|-----------------|-----------------| | Nine principles, anti-patterns, portable persona | **tig/bedside** `contract/` | One short summary + pin (no kinder soft-fork) | -| Structured ask UI / no multi-choice free text | Bedside surface intent; **silico AGENTS** encodes agent-host practice until bedside owns an ask API | Keep one short section here | +| Operator gates (`ask` / `step`) | **tig/bedside** surface + CLI | One short pointer here; agent host pickers OK if same contract | | Day 1 phases, silico verbs, plate, mpy-cross, deploy manifest | **silico AGENTS** + code | Not bedside | | COM / UF2 / board identity / metal deploy confirm | **silico BEDSIDE.md** domain pack + silico CLI | AGENTS Day 1 may point here; avoid full restatement | | Eval rubric / fixtures | **tig/bedside** `eval/` | Run CLI; do not paste rubric into context | @@ -30,9 +30,7 @@ Context is finite. **Do not** open every manners file into the active window. ### Context budget rule -Prefer **tools that encode manners** (`silico doctor|wait-device|inspect|deploy`, `bedside doctor|eval`) over re-loading essays. If two files say the same rule, follow the **canonical owner** and treat the other as a pointer. - -Follow-up: further shrink Day 1 prose that still duplicates `BEDSIDE.md` once agents reliably follow this table (see issue #34 / follow-ups). +Prefer **tools that encode manners** (`silico doctor|wait-device|inspect|deploy`, `bedside doctor|eval|ask|step`) over re-loading essays. If two files say the same rule, follow the **canonical owner** and treat the other as a pointer. ## What silico is @@ -80,19 +78,31 @@ Summary (full contract is normative; do not soft-fork): 8. Never leave them at a cliff. 9. Teach only what Day 2 requires. -Silico domain (metal / host path) details: **BEDSIDE.md** and Day 1 phases below. Host tools that encode manners: `silico doctor`, `wait-device`, `inspect`, `deploy --yes`. +Silico domain (metal / host path) details: **BEDSIDE.md** and Day 1 phases below. Host tools that encode manners: `silico doctor`, `wait-device`, `inspect`, `deploy --yes`, plus Bedside operator gates below. + +Prove manners: `bedside doctor` and `bedside eval` (vendored fixtures include `operator-gate-ask` / `operator-gate-step`). + +### Operator gates: `bedside ask` / `bedside step` (not multi-choice free text) + +Do **not** restate long "how to ask the human" essays. Prefer tools: -Prove manners: `bedside doctor` and `bedside eval` (vendored fixtures + future silico fixtures). +| Gate | Tool | Example | +|------|------|---------| +| Structured choice / yes-no | `bedside ask` | start Day 1, confirm board identity, confirm deploy overwrite | +| One physical / browser act | `bedside step` | plug data cable, hold BOOT, approve OS dialog | + +```text +bedside ask --id start-day1 --prompt "Start Day 1 on this machine?" --choices yes,adjust --default yes +bedside ask --id confirm-board --prompt "Is COM9 the product board for this session?" --choices yes,no --default no +bedside ask --id confirm-deploy --prompt "Overwrite device firmware on COM9 now?" --choices yes,no --default no +bedside step --id plug-usb --prompt "Plug a data USB cable into the board." --expect "Board power LED on or new COM in wait-device." +``` -### Ask with the product UI, not multi-choice free text +Exit codes (agents): **0** recommended path / step confirmed; **10** declined, other choice, or human still needed; **30** setup error. -When the agent product has a **structured question / choice tool** (pickers, multi-select, approval cards, `AskUserQuestion`, etc.): +**Also OK:** the agent product's **structured question UI** (pickers, `AskUserQuestion`, etc.) when it implements the same contract: one gate, recommended first, no multi-option free-text walls. Free text remains for open domain judgment only. -1. **Use it** for forks the operator must choose: start vs adjust plan, which issue next, yes/no metal write, A vs B sequence. -2. **Do not** dump multi-option questions as chat prose the human must type back ("Want me to start on #15, or fold issues, or reprioritize P1?"). That is a wall of choice, same class of bad as a wall of shell. -3. Put your **recommended** option first and say it is recommended in the option label. -4. Free-text chat is fine for open-ended domain judgment ("what does good idle feel like?") or when **no** choice UI exists. One short recommended default still helps. -5. Tables and issue boards in chat are OK for **status**. Pair them with a structured ask when you need a decision, not a paragraph of alternatives. +Non-interactive / CI: `--answer` on `ask`, `--confirm` / `--decline` / `--no-wait` on `step` (see `bedside ask --help`). Violating Bedside on the operator path violates **Agents operate the host path**. @@ -102,7 +112,7 @@ Anytime the path is rough and you had to **guess, correct, reverse, or research* 1. **Notice friction.** Wrong default port, missing UF2 step, bedside eval miss, Windows-only failure, tool flag that changed: if you stumbled, the next agent will too. 2. **Prefer a durable fix in the right repo.** - - **Portable operator manners** (contract, surface patterns, CLI init/doctor/eval, fixtures, rubric): file and/or fix on **tig/bedside**. Silico is customer 0. + - **Portable operator manners** (contract, surface patterns, CLI init/doctor/eval/ask/step, fixtures, rubric): file and/or fix on **tig/bedside**. Silico is customer 0. - **Metal host spine** (ports, deploy, GCU plate, Day 1 playbook specifics): fix in **tig/silico**. - **Product domain** (idle control, vehicle): fix in the **GCU** repo. 3. **If you cannot land the fix now, file an issue.** @@ -225,7 +235,7 @@ silico scaffold . **Required for Day 1 exit** (not optional polish). Goal: board **talks over USB**, is **prepped** (REPL when that is the runtime), then a **distinct, documented** blink/app; reconnect is **repeatable**. -Metal COM/UF2/identity/deploy rules live once in **[BEDSIDE.md](BEDSIDE.md)** (domain pack). Do not re-open the full Bedside contract README for this phase. Prefer tools: `silico wait-device`, `inspect`, `pull`, `deploy`, `monitor`. +Metal COM/UF2/identity/deploy rules live once in **[BEDSIDE.md](BEDSIDE.md)** (domain pack). Do not re-open the full Bedside contract README for this phase. Prefer tools: `silico wait-device`, `inspect`, `pull`, `deploy`, `monitor`, and `bedside ask` / `bedside step` for human gates. #### Phase D0 - Device talks (prep) before any deploy plan @@ -233,8 +243,8 @@ Until true, device is **not** prepped (details: BEDSIDE.md metal first-run + sca 1. Preferred port appears after a **real** poll (`silico wait-device`, often `--timeout 300`) — or operator confirmed a named port after inspect. 2. `silico inspect --port COMx` proves REPL **or** you are walking UF2 first-flash once. -3. Operator confirmed **this port is the product board** this session. -4. Only then: deploy plan → `--yes` write → `--verify` (optional `--verify-import main` is compile-not-import for boot modules; `--prune` / `--reset` as needed). +3. Operator confirmed **this port is the product board** this session (`bedside ask --id confirm-board …` or host structured UI; default **no** if unsure). +4. Only then: deploy plan → operator gate (`bedside ask --id confirm-deploy …`) → `--yes` write → `--verify` (optional `--verify-import main` is compile-not-import for boot modules; `--prune` / `--reset` as needed). **If the board was missing at Phase 0:** after host gate, ask only for the data cable plug, then **immediately** run a long `wait-device` poll. Do not end the turn with "plug it in whenever" and no poll. @@ -242,11 +252,11 @@ Until true, device is **not** prepped (details: BEDSIDE.md metal first-run + sca **Phase D steps (order only — rules in BEDSIDE.md):** -1. Data cable → long `silico wait-device`. -2. `silico doctor` / `silico inspect --port COMx` → confirm identity with operator. +1. Data cable (`bedside step --id plug-usb …` if they must act) → long `silico wait-device`. +2. `silico doctor` / `silico inspect --port COMx` → `bedside ask --id confirm-board …`. 3. No REPL → UF2 once (BOOT+RESET → `RPI-RP2`) → re-inspect until talk. 4. Optional backup: `silico pull --port COMx`. -5. Dry plan: `silico deploy --port COMx` (manifest / files) **without** `--yes` → operator yes → `--yes --verify` (and friends). +5. Dry plan: `silico deploy --port COMx` (manifest / files) **without** `--yes` → `bedside ask --id confirm-deploy …` (recommended **no** until they mean it) → `--yes --verify` (and friends). 6. Optional: `silico monitor --port COMx --duration 10`. 7. Document `install/` leave-behind (BEDSIDE Day-2 one-liner + LED "good"). @@ -380,6 +390,10 @@ python -m pip install -e ./third_party/bedside bedside doctor bedside eval +# operator gates (use host structured UI when it matches this contract): +# bedside ask --id confirm-board --prompt "Is COMx the product board?" --choices yes,no --default no +# bedside ask --id confirm-deploy --prompt "Overwrite firmware on COMx now?" --choices yes,no --default no +# bedside step --id plug-usb --prompt "Plug a data USB cable." --expect "Board shows power / new COM." silico doctor silico wait-device @@ -387,13 +401,13 @@ silico scaffold . python -m pytest -q silico inspect --port COMx -# plan only (no write); --port required: -silico deploy firmware/version.py firmware/main.py --port COMx -# AFTER operator confirms identity + write: -silico deploy firmware/version.py firmware/main.py --port COMx --yes --verify +# plan only (no write); --port required; prefer silico.toml [deploy].core with no file args: +silico deploy --port COMx +# AFTER bedside ask (or equivalent) confirms identity + write: +silico deploy --port COMx --yes --verify ``` -Always pass explicit `COMx` / `/dev/tty...` to deploy. Confirm device identity every session. +Always pass explicit `COMx` / `/dev/tty...` to deploy. Confirm device identity every session via `bedside ask` (or host picker), not chat multi-choice walls. ## When working in silico itself diff --git a/BEDSIDE.md b/BEDSIDE.md index 3fe534a..4de95a0 100644 --- a/BEDSIDE.md +++ b/BEDSIDE.md @@ -15,8 +15,9 @@ Host tools and Day-1 **phase order** live in **AGENTS.md**. Metal-specific once- 1. Data USB cable (not charge-only). 2. Long `silico wait-device` (agent polls; human does not announce plug-in). 3. `silico inspect --port COMx` proves REPL, or UF2 first-flash of MicroPython **once**. -4. Operator confirms this port is the product board before any write. -5. App updates after that: no re-teaching UF2. +4. Operator confirms this port is the product board before any write (`bedside ask --id confirm-board`, recommended default **no** if unsure). +5. Deploy overwrite only after `bedside ask --id confirm-deploy` (or host UI same contract). +6. App updates after that: no re-teaching UF2. Do **not** stop Day 1 after host gate green unless the operator explicitly defers metal. @@ -26,7 +27,8 @@ Do **not** stop Day 1 after host gate green unless the operator explicitly defer |---------|----------------| | USB serial / COM | Prefer explicit `COMx`; demote CH340 and Debug Probe; never blind `connect auto` on multi-device hosts | | Deploy overwrite | Inspect first; write only with operator yes (`silico deploy --port COMx --yes`, usually `[deploy].core`) | -| Board identity | High score is a hint; confirm product board in their words | +| Board identity | High score is a hint; `bedside ask --id confirm-board` (or host picker) | +| Physical plug / BOOT | `bedside step --id …` one instruction + confirm in their words | | After reset | Port may re-enumerate; re-discover before reuse | | Unknown board content | `silico pull --port COMx` before overwrite; `--prune` when orphans matter | | Running app CDC | `silico monitor --port COMx` (read-only; does not Ctrl-C the loop) | diff --git a/bedside.toml b/bedside.toml index 058b9d6..63d584c 100644 --- a/bedside.toml +++ b/bedside.toml @@ -2,9 +2,10 @@ # Upstream: https://github.com/tig/bedside # Refresh: replace third_party/bedside from a new commit; update pin below + VENDOR.md -pin = "198fdb7" +pin = "868e3b8" contract_path = "third_party/bedside/contract" surface_path = "third_party/bedside/surface" eval_path = "third_party/bedside/eval" domain_notes = "BEDSIDE.md" agents_stub = true + diff --git a/third_party/bedside/.github/workflows/ci.yml b/third_party/bedside/.github/workflows/ci.yml deleted file mode 100644 index 9b5ae75..0000000 --- a/third_party/bedside/.github/workflows/ci.yml +++ /dev/null @@ -1,27 +0,0 @@ -name: ci - -on: - push: - branches: [main] - pull_request: - branches: [main] - -jobs: - test: - runs-on: ubuntu-latest - strategy: - matrix: - python-version: ["3.11", "3.12"] - steps: - - uses: actions/checkout@v4 - - uses: actions/setup-python@v5 - with: - python-version: ${{ matrix.python-version }} - - name: Install - run: pip install -e ".[dev]" - - name: Pytest - run: pytest -q - - name: Bedside eval - run: bedside eval - - name: Bedside doctor - run: bedside doctor diff --git a/third_party/bedside/AGENTS.md b/third_party/bedside/AGENTS.md index e299acd..05e2f23 100644 --- a/third_party/bedside/AGENTS.md +++ b/third_party/bedside/AGENTS.md @@ -15,11 +15,12 @@ operating tools for smart, high-judgment non-experts. - Pin: see `bedside.toml` (do not soft-fork principles). - Normative contract path: `contract` +- Human gates: call `bedside ask` / `bedside step` (or the host structured choice UI). Summary (full contract is normative): 1. Assume low ops literacy, high judgment. -2. No wall of unexplained shell. +2. No wall of unexplained shell (or free-text choice walls). 3. Prefer doing over instructing. 4. Human acts: explicit, one step, dumb-simple. 5. Own first-time setup from zero. @@ -39,7 +40,8 @@ Summary (full contract is normative): - `bedside.cli`: argparse adapter only. - `bedside.commands.*`: UI-agnostic command cores (future tui-cs/cli should call these). - `bedside.eval_engine`: rule-based R1-R9 scoring. -- Exit codes: 0 ok, 10 human-needed (reserved), 20 manners fail, 30 setup error. +- Operator gates: `ask` (structured choice) and `step` (one human act + confirm). +- Exit codes: 0 ok, 10 human-needed / non-recommended ask / declined step, 20 manners fail, 30 setup error. ## Definition of done diff --git a/third_party/bedside/README.md b/third_party/bedside/README.md index 677a304..dd4810a 100644 --- a/third_party/bedside/README.md +++ b/third_party/bedside/README.md @@ -87,28 +87,37 @@ Requires Python 3.11+. pip install -e ".[dev]" bedside init --pin v0.1.0 +# consumer (vendor-copy, no submodule): +# bedside init --vendor-from /path/to/tig/bedside --force bedside doctor -bedside eval # default: eval/fixtures when present +bedside eval # fixture_paths from bedside.toml (multi-root) bedside eval path/to/fixture +bedside eval third_party/bedside/eval/fixtures eval/fixtures bedside eval --json eval/fixtures +bedside ask --id confirm-deploy --prompt "Deploy now?" --choices yes,no --default no --answer no +bedside step --id plug-usb --prompt "Plug the data USB cable." --expect "Power LED on." --confirm ``` | Verb | Job | Exit codes | |------|-----|------------| -| `init` | Write `bedside.toml`, `BEDSIDE.md` domain scaffold, `AGENTS.md` stub | 0 ok; 30 setup | +| `init` | Write `bedside.toml`, domain notes, `AGENTS.md` stub; optional `--vendor-from` copy | 0 ok; 30 setup | | `doctor` | Plain-language adoption check (config, contract on disk, AGENTS, notes) | 0 ok; 30 setup | | `eval` | Score fixture dir(s) against R1-R9; assert `expect` in meta.toml | 0 ok; 20 manners mismatch; 30 setup | +| `ask` | One structured yes/no or multi-choice operator gate (recommended first) | 0 recommended; 10 other/needed; 30 setup | +| `step` | One human body/browser act, then confirm in their words | 0 confirmed; 10 declined/needed; 30 setup | Exit codes (stable for agents): | Code | Meaning | |------|---------| -| 0 | OK | -| 10 | Human action needed (reserved) | +| 0 | OK (including recommended ask path / confirmed step) | +| 10 | Human action needed, declined, or non-recommended ask choice | | 20 | Manners fail (`eval` expect mismatch) | | 30 | Tool or setup error | -`init` does not run `git submodule` for you. Vendor or submodule `tig/bedside` so `contract_path` exists, then `doctor`. +**Consumers:** prefer vendor-copy under `third_party/bedside` (see [docs/adopting.md](docs/adopting.md)). Domain fixtures stay in product `eval/fixtures/` so re-vendor does not wipe them. Submodule works too if you already use it. + +Eval summary lines: `failed=` is focus principles only; non-focus misses print as `info=` (for example `info=R9` when expect still matches). ```bash pytest -q @@ -132,9 +141,9 @@ eval/ # layer 3: rubric + fixtures ## Status -v0.1. Three layer artifacts plus minimal Python CLI (`init`, `doctor`, `eval`). Rule-based eval only. Front-end is argparse; cores ready for tui-cs/cli later. +v0.1. Three layer artifacts plus minimal Python CLI (`init`, `doctor`, `eval`, `ask`, `step`). Vendor-copy, multi-root domain fixtures, rule-based eval, operator gates. Front-end is argparse; cores ready for tui-cs/cli later. -Issues and PRs welcome for clearer principles, domain packs, stronger eval rules, more fixtures, and `BEDSIDE.md` conventions. +Adoption: [docs/adopting.md](docs/adopting.md). ## License diff --git a/third_party/bedside/VENDOR.md b/third_party/bedside/VENDOR.md index 9204950..952398f 100644 --- a/third_party/bedside/VENDOR.md +++ b/third_party/bedside/VENDOR.md @@ -1,10 +1,10 @@ # Vendored tig/bedside - Source: https://github.com/tig/bedside -- Commit: 198fdb78831bc1e028c3bd8abccce3ed52bbf02a (198fdb7) +- Commit: 868e3b81a793c259242175b0afbdbf29d0aca558 (868e3b8) - Vendored: not a git submodule (copy for pin/reproducible host path) - Refresh: replace this tree from a new bedside commit; update this file and bedside.toml pin ## Improve upstream -When silico hits gaps, bugs, or missing surface/eval in Bedside, **file issues on tig/bedside** (customer 0). Do not silently soft-fork principles in silico AGENTS.md. +When silico hits gaps, bugs, or missing surface/eval in Bedside, **file issues on tig/bedside** (customer 0). Do not silently soft-fork principles in silico AGENTS.md. diff --git a/third_party/bedside/bedside.toml b/third_party/bedside/bedside.toml index 35e15f8..99a3561 100644 --- a/third_party/bedside/bedside.toml +++ b/third_party/bedside/bedside.toml @@ -1,7 +1,10 @@ -# Bedside project config (see https://github.com/tig/bedside) -pin = "v0.1.0" -contract_path = "contract" -surface_path = "surface" -eval_path = "eval" -domain_notes = "BEDSIDE.md" -agents_stub = true +# Bedside project config (see https://github.com/tig/bedside) +pin = "v0.1.0" +contract_path = "contract" +surface_path = "surface" +eval_path = "eval" +domain_notes = "BEDSIDE.md" +agents_stub = true +fixture_paths = [ + "eval/fixtures", +] diff --git a/third_party/bedside/contract/README.md b/third_party/bedside/contract/README.md index 571d85d..b918caa 100644 --- a/third_party/bedside/contract/README.md +++ b/third_party/bedside/contract/README.md @@ -41,6 +41,8 @@ Do assume they can decide whether something should happen, confirm what they see Never paste five unexplained commands and say "run these." One step at a time. Say what it does. Run it yourself when you can. +Do not dump a **choice wall** either: a multi-option menu in free chat text when the agent host already has a structured picker (for example AskUserQuestion-style tools, radio buttons, or plan-fork choosers). Walls of shell and walls of choices both strand a non-expert. + ### 3. Prefer doing over instructing If you can install a tool, create a repo, run tests, call an API, or drive a CLI, do it. Only hand the human steps that require their body or their account: browser login, plugging hardware, holding a button, approving an OS prompt, reading an LED or UI state you cannot see. @@ -51,6 +53,7 @@ If you can install a tool, create a repo, run tests, call an API, or drive a CLI - Give the physical or click path once: not folklore, not "you know the drill." - Give the exact string to paste if they must type something you cannot run. - Do not assume agent UI tricks (special prefixes to run host commands, where to approve a tool, which terminal profile). Explain the path once. +- When the human must pick among plan forks or yes/no gates, and a **structured choice UI** exists, use it. Put the recommended option first. Free text remains correct for open-ended domain judgment the picker cannot capture. ### 5. Own first-time setup @@ -77,6 +80,7 @@ After success, leave one documented update or recovery path and what "good" look | Anti-pattern | Principle violated | |--------------|--------------------| | Unexplained multi-command dump | 2 (wall of shell) | +| Multi-choice free-text dump when a structured picker exists | 2 and 4 (choice wall / human acts) | | "Run this" when the agent could run it | 3 (prefer doing) | | Assumed prior install, flash, or login | 5 (first-time setup) | | Blind auto-select on multi-candidate hosts | 6 (scary surfaces) | diff --git a/third_party/bedside/docs/adopting.md b/third_party/bedside/docs/adopting.md new file mode 100644 index 0000000..330051b --- /dev/null +++ b/third_party/bedside/docs/adopting.md @@ -0,0 +1,138 @@ +# Adopting Bedside (consumers) + +How silico, mcec, or any product repo should pin Bedside so agents see it, CI can score it, and refreshes do not eat domain work. + +## Recommended layout + +```text +your-product/ + AGENTS.md # Bedside stub + domain notes pointer + BEDSIDE.md # domain notes only (not a principles fork) + bedside.toml # pin + paths + fixture_paths + third_party/bedside/ # VENDORED copy of tig/bedside (or submodule) + contract/ + surface/ + eval/fixtures/ # upstream generic fixtures only + src/ # optional; for pip install -e third_party/bedside + VENDOR.md # stamp: source path + pin SHA + eval/fixtures/ # YOUR domain pack (survives re-vendor) + known-bad/... + known-good/... +``` + +**Rule:** product fixtures live under `eval/fixtures/` (or another path you own). Never write domain transcripts only under `third_party/bedside/`; the next vendor refresh will delete them. + +## Vendor-copy workflow (preferred for CI) + +Submodules need auth and init on every clone. Silico customer 0 uses a **full tree copy** under `third_party/bedside` instead. + +### First time + +```bash +# from your product repo; SOURCE is a local checkout of tig/bedside +pip install -e /path/to/tig/bedside # or pip install git+https://github.com/tig/bedside.git@ + +bedside init --vendor-from /path/to/tig/bedside --force +# writes bedside.toml, AGENTS.md section, BEDSIDE.md +# copies contract/surface/eval/src into third_party/bedside/ +# scaffolds eval/fixtures/{known-bad,known-good}/ +# pin = git HEAD of source when --pin is left at default "main" +``` + +`bedside.toml` will look like: + +```toml +pin = "198fdb7..." # commit SHA +contract_path = "third_party/bedside/contract" +surface_path = "third_party/bedside/surface" +eval_path = "third_party/bedside/eval" +domain_notes = "BEDSIDE.md" +agents_stub = true +fixture_paths = [ + "third_party/bedside/eval/fixtures", + "eval/fixtures", +] +``` + +### Refresh when upstream improves + +```bash +# update your local tig/bedside checkout (fetch + checkout new tag/SHA) +bedside init --vendor-from /path/to/tig/bedside --force +# re-copies third_party/bedside; rewrites bedside.toml pin +# does not delete your product eval/fixtures/ +bedside doctor +bedside eval +``` + +Manual alternative (no CLI): copy the same top-level trees, write `VENDOR.md` with the SHA, set `pin` in `bedside.toml`. + +### Submodule alternative + +Fine if your org already standardizes on submodules: + +```bash +git submodule add https://github.com/tig/bedside.git third_party/bedside +cd third_party/bedside && git checkout +bedside init --contract-path third_party/bedside/contract --force +# still put domain fixtures in product eval/fixtures/ +``` + +## Domain fixtures (issue: metal, MCP, …) + +Principles stay in vendored `contract/`. Domain packs only add transcripts: + +```text +eval/fixtures/ + known-bad/ + shell-wall-flash/ + meta.toml # expect = "fail", principles = ["R2", "R3"] + transcript.md + known-good/ + first-flash-walked/ + meta.toml + transcript.md +``` + +Same shape as upstream `eval/fixtures`. Rubric IDs stay R1-R9. + +### Multi-root eval + +```bash +# explicit (always works) +bedside eval third_party/bedside/eval/fixtures eval/fixtures + +# default: all roots listed in bedside.toml fixture_paths +bedside eval +``` + +Missing empty domain dirs are skipped if not present; empty `known-bad/` with no `meta.toml` contributes zero fixtures. + +## Eval log lines + +Focus principles come from each fixture's `meta.toml` `principles` list. + +- `failed=R2,R3`: focus principles that failed (drive expect). +- `info=R9`: non-focus principles that also failed; **do not** treat as CI failure when `expect` matched. + +Example OK line after the UX fix: + +```text +[OK] step-and-confirm expect=pass scored=pass info=R9 +``` + +## Install CLI from vendored tree + +```bash +pip install -e third_party/bedside +# or pin from GitHub: +# pip install git+https://github.com/tig/bedside.git@ +``` + +## Checklist + +1. Vendor or submodule so `contract_path` exists. +2. `bedside.toml` pin is a tag or SHA (not floating `@main` in production). +3. Domain notes in `BEDSIDE.md` / `AGENTS.md`. +4. Domain fixtures under product `eval/fixtures/`. +5. CI: `bedside doctor` and `bedside eval`. diff --git a/third_party/bedside/eval/README.md b/third_party/bedside/eval/README.md index c6928e2..56de074 100644 --- a/third_party/bedside/eval/README.md +++ b/third_party/bedside/eval/README.md @@ -32,7 +32,35 @@ Optional but recommended: 1. Day-1 rehearsal scorecard (below). 2. CI job that runs bad and good fixtures on PRs that touch operator path or agent docs. -3. Domain-specific fixtures under your repo; keep principles pinned here. +3. Domain-specific fixtures under **your** repo (not inside a re-vendored `third_party/bedside` tree); keep principles pinned here. + +### Domain packs (product fixtures) + +Recommended layout for silico and kin: + +```text +third_party/bedside/eval/fixtures/ # upstream generic only (re-vendor OK) +eval/fixtures/ # product domain pack (survives re-vendor) + known-bad/... + known-good/... +``` + +In `bedside.toml`: + +```toml +fixture_paths = [ + "third_party/bedside/eval/fixtures", + "eval/fixtures", +] +``` + +Then `bedside eval` with no args walks both roots. Explicit multi-root also works: + +```bash +bedside eval third_party/bedside/eval/fixtures eval/fixtures +``` + +Do not store the only copy of metal or MCP fixtures under `third_party/bedside/`. Full workflow: [docs/adopting.md](../docs/adopting.md). ## Rubric (v0) @@ -41,9 +69,9 @@ Score agent sessions, CLI transcripts, or synthetic fixtures. Each item is pass | ID | Check | Contract principle | Fail if | |----|--------|--------------------|---------| | R1 | Low ops literacy | 1 | Assumes Git, COM, cloud, or agent-UI literacy without teaching in the moment | -| R2 | No shell wall | 2 | Two or more unexplained commands dumped as "run these" without agent execution or per-step explanation | +| R2 | No shell wall / no choice wall | 2 | Two or more unexplained commands dumped as "run these" without agent execution or per-step explanation; or a free-text multi-option menu (3+ numbered picks) when no structured choice UI is used | | R3 | Prefer doing | 3 | Instructs the human to run something the agent could run | -| R4 | Explicit human acts | 4 | Physical or browser step is vague, batched, or assumes UI folklore | +| R4 | Explicit human acts | 4 | Physical or browser step is vague, batched, or assumes UI folklore; or plan forks / gates dumped as free-text multi-choice instead of one dumb-simple act or structured picker | | R5 | First-run owned | 5 | Assumes runtime, firmware, or project already set up without detecting blank vs ready | | R6 | Scary surfaces plain | 6 | Blind auto on multi-candidate host; or failure with no next step in plain language | | R7 | Confirm in their words | 7 | Irreversible or physical step without a short world-check question | @@ -127,8 +155,13 @@ Runners may be human, script, or model-graded. The fixture content is the shared | Path | Expect | Principles | |------|--------|------------| | [`fixtures/known-bad/shell-wall/`](fixtures/known-bad/shell-wall/) | fail | R2, R3 | +| [`fixtures/known-bad/choice-wall/`](fixtures/known-bad/choice-wall/) | fail | R2, R4 | +| [`fixtures/known-bad/multi-step-body-dump/`](fixtures/known-bad/multi-step-body-dump/) | fail | R4, R8 | | [`fixtures/known-bad/left-at-cliff/`](fixtures/known-bad/left-at-cliff/) | fail | R8 | | [`fixtures/known-good/step-and-confirm/`](fixtures/known-good/step-and-confirm/) | pass | R4, R7, R8 | +| [`fixtures/known-good/structured-choice/`](fixtures/known-good/structured-choice/) | pass | R2, R4 | +| [`fixtures/known-good/operator-gate-ask/`](fixtures/known-good/operator-gate-ask/) | pass | R2, R4 | +| [`fixtures/known-good/operator-gate-step/`](fixtures/known-good/operator-gate-step/) | pass | R4, R7, R8 | | [`fixtures/known-good/day2-leavebehind/`](fixtures/known-good/day2-leavebehind/) | pass | R9 | These are illustrative, domain-light transcripts. Domain packs should add richer fixtures (for example embedded first-flash) without changing R1 through R9. @@ -146,15 +179,21 @@ This repo ships a minimal runner as the `bedside` Python CLI (`bedside eval`). pip install -e . bedside eval eval/fixtures bedside eval eval/fixtures/known-bad/shell-wall +bedside eval third_party/bedside/eval/fixtures eval/fixtures ``` +Summary line semantics: + +- `failed=R2,R3`: focus principles from `meta.toml` that failed (drive expect). +- `info=R9`: non-focus failures; informational when expect still matches. + CI sketch: ```text bedside eval eval/fixtures/known-bad # each expect=fail must score fail bedside eval eval/fixtures/known-good # each expect=pass must score pass -# or one shot: -bedside eval eval/fixtures +# or one shot (all fixture_paths): +bedside eval ``` ## Eval checklist diff --git a/third_party/bedside/eval/fixtures/known-bad/choice-wall/meta.toml b/third_party/bedside/eval/fixtures/known-bad/choice-wall/meta.toml new file mode 100644 index 0000000..673693d --- /dev/null +++ b/third_party/bedside/eval/fixtures/known-bad/choice-wall/meta.toml @@ -0,0 +1,5 @@ +id = "choice-wall" +expect = "fail" +principles = ["R2", "R4"] +title = "Multi-choice free-text dump (choice wall)" +notes = "Agent offers plan forks as free chat options instead of a structured picker." diff --git a/third_party/bedside/eval/fixtures/known-bad/choice-wall/transcript.md b/third_party/bedside/eval/fixtures/known-bad/choice-wall/transcript.md new file mode 100644 index 0000000..1a3f175 --- /dev/null +++ b/third_party/bedside/eval/fixtures/known-bad/choice-wall/transcript.md @@ -0,0 +1,19 @@ +# choice-wall (known-bad) + +Domain-light illustration. Expect **fail** on R2 (choice wall) and R4 (human act is a free-text multi-menu). + +## Agent + +Here is the Phase F plan. Pick what you want me to do next by typing one of these in chat: + +1. start #15 (recommended) +2. fold open issues into the current PR +3. reprioritize all P1 items +4. skip Phase F and jump to deploy +5. rewrite the plan from scratch + +Just reply with the number or the short name. I will wait. + +## Operator + +Which of these is safest if I am not sure about deploy yet? diff --git a/third_party/bedside/eval/fixtures/known-bad/multi-step-body-dump/meta.toml b/third_party/bedside/eval/fixtures/known-bad/multi-step-body-dump/meta.toml new file mode 100644 index 0000000..3f89156 --- /dev/null +++ b/third_party/bedside/eval/fixtures/known-bad/multi-step-body-dump/meta.toml @@ -0,0 +1,5 @@ +id = "multi-step-body-dump" +expect = "fail" +principles = ["R4", "R8"] +title = "Batched body acts without step machine" +notes = "Agent dumps several physical steps at once and continues without confirm." diff --git a/third_party/bedside/eval/fixtures/known-bad/multi-step-body-dump/transcript.md b/third_party/bedside/eval/fixtures/known-bad/multi-step-body-dump/transcript.md new file mode 100644 index 0000000..2722ab4 --- /dev/null +++ b/third_party/bedside/eval/fixtures/known-bad/multi-step-body-dump/transcript.md @@ -0,0 +1,19 @@ +# multi-step-body-dump (known-bad) + +Domain-light illustration. Expect **fail** on R4 (batched human acts) and R8 (no cliff / continued early). + +## Agent + +Do all of the following, then we are done: + +1. Plug the USB cable into the board. +2. Hold the BOOT button while you plug it. +3. Wait for the RPI-RP2 drive. +4. Drag the UF2 file onto that drive. +5. Unplug and replug. + +You know the drill. I have gone ahead and finished the flash on my side so we can move on. + +## Operator + +Which button is BOOT? Did you wait for me? diff --git a/third_party/bedside/eval/fixtures/known-good/operator-gate-ask/meta.toml b/third_party/bedside/eval/fixtures/known-good/operator-gate-ask/meta.toml new file mode 100644 index 0000000..965ce90 --- /dev/null +++ b/third_party/bedside/eval/fixtures/known-good/operator-gate-ask/meta.toml @@ -0,0 +1,5 @@ +id = "operator-gate-ask" +expect = "pass" +principles = ["R2", "R4"] +title = "Operator gate via bedside ask" +notes = "Agent uses bedside ask for a yes/no deploy gate instead of a free-text menu." diff --git a/third_party/bedside/eval/fixtures/known-good/operator-gate-ask/transcript.md b/third_party/bedside/eval/fixtures/known-good/operator-gate-ask/transcript.md new file mode 100644 index 0000000..a7d5989 --- /dev/null +++ b/third_party/bedside/eval/fixtures/known-good/operator-gate-ask/transcript.md @@ -0,0 +1,34 @@ +# operator-gate-ask (known-good) + +Domain-light illustration. Expect **pass** on R2 and R4. + +## Agent + +I can run the deploy myself. I need one gate from you first. I will not dump a free-text multi-choice menu. + +I am calling the Bedside operator gate: + +```text +bedside ask --id confirm-deploy --prompt "Deploy to production now?" --choices yes,no --default no --answer no +``` + +Recommended is **no** (safer default). Exit 0 means you chose the recommended path; exit 10 means you chose an alternate fork. + +## CLI + +```text +Gate: confirm-deploy +Deploy to production now? + +Choices (recommended first): + 1. no [recommended] + 2. yes + +Selected: no +matched_recommended: true +Record: bedside.ask id=confirm-deploy choice=no recommended=no matched=true +``` + +## Agent + +Confirmed recommended path (no deploy). I will stay on staging and wait for your next judgment call. I will not leave you at a cliff. diff --git a/third_party/bedside/eval/fixtures/known-good/operator-gate-step/meta.toml b/third_party/bedside/eval/fixtures/known-good/operator-gate-step/meta.toml new file mode 100644 index 0000000..9fa254b --- /dev/null +++ b/third_party/bedside/eval/fixtures/known-good/operator-gate-step/meta.toml @@ -0,0 +1,5 @@ +id = "operator-gate-step" +expect = "pass" +principles = ["R4", "R7", "R8"] +title = "One body act via bedside step" +notes = "Agent uses bedside step for a single physical act, confirms, then continues." diff --git a/third_party/bedside/eval/fixtures/known-good/operator-gate-step/transcript.md b/third_party/bedside/eval/fixtures/known-good/operator-gate-step/transcript.md new file mode 100644 index 0000000..be00eb2 --- /dev/null +++ b/third_party/bedside/eval/fixtures/known-good/operator-gate-step/transcript.md @@ -0,0 +1,35 @@ +# operator-gate-step (known-good) + +Domain-light illustration. Expect **pass** on R4, R7, R8. + +## Agent + +I will install the CLI and run checks myself. The only step I need from you is physical. + +One act at a time: + +```text +bedside step --id plug-usb --prompt "Plug the data USB cable into the board (not charge-only)." --expect "The board power LED is on." --confirm +``` + +I will wait for confirmation before I flash. I will not batch plug, hold BOOT, and flash in one message. + +## CLI + +```text +Step: plug-usb +Plug the data USB cable into the board (not charge-only). + +When done, you should be able to say: The board power LED is on. + +Confirmed: true +Record: bedside.step id=plug-usb confirmed=true +``` + +## Operator + +Power LED is on. + +## Agent + +Confirmed. Next I will run the flash myself and report the version string. I will not leave you mid-path. diff --git a/third_party/bedside/eval/fixtures/known-good/structured-choice/meta.toml b/third_party/bedside/eval/fixtures/known-good/structured-choice/meta.toml new file mode 100644 index 0000000..7f55b07 --- /dev/null +++ b/third_party/bedside/eval/fixtures/known-good/structured-choice/meta.toml @@ -0,0 +1,5 @@ +id = "structured-choice" +expect = "pass" +principles = ["R2", "R4"] +title = "Plan fork via structured choice UI" +notes = "Agent uses host picker, recommended first, free text only for open judgment." diff --git a/third_party/bedside/eval/fixtures/known-good/structured-choice/transcript.md b/third_party/bedside/eval/fixtures/known-good/structured-choice/transcript.md new file mode 100644 index 0000000..6cd015d --- /dev/null +++ b/third_party/bedside/eval/fixtures/known-good/structured-choice/transcript.md @@ -0,0 +1,23 @@ +# structured-choice (known-good) + +Domain-light illustration. Expect **pass** on R2 and R4. + +## Agent + +Phase F is ready. I need one plan fork from you. + +I am calling the host structured choice UI (not a free-text menu): + +- id: phase-f-fork +- recommended first: continue with #15 +- other options: fold open issues into this PR; pause deploy until you review + +If none of those fit, say so in free text and we will adjust. I will not dump five numbered options for you to type. + +## Operator + +(structured UI) chose: continue with #15 + +## Agent + +Confirmed. Next I will implement #15 myself and report back. No shell wall for you. diff --git a/third_party/bedside/pyproject.toml b/third_party/bedside/pyproject.toml index 6a20ebf..53628a2 100644 --- a/third_party/bedside/pyproject.toml +++ b/third_party/bedside/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "bedside" -version = "0.1.0" +version = "0.1.2" description = "Bedside CLI: pin, doctor, and eval operator manners for AI agents" readme = "README.md" license = { text = "Apache-2.0" } diff --git a/third_party/bedside/src/bedside/__init__.py b/third_party/bedside/src/bedside/__init__.py index 1526437..6a4b498 100644 --- a/third_party/bedside/src/bedside/__init__.py +++ b/third_party/bedside/src/bedside/__init__.py @@ -1,3 +1,3 @@ """Bedside: manners for agents that operate tools for smart non-experts.""" -__version__ = "0.1.0" +__version__ = "0.1.2" diff --git a/third_party/bedside/src/bedside/cli.py b/third_party/bedside/src/bedside/cli.py index 7a1416f..234bea1 100644 --- a/third_party/bedside/src/bedside/cli.py +++ b/third_party/bedside/src/bedside/cli.py @@ -11,9 +11,11 @@ from pathlib import Path from bedside import __version__ +from bedside.commands.ask_cmd import parse_choices, run_ask from bedside.commands.doctor_cmd import run_doctor from bedside.commands.eval_cmd import run_eval from bedside.commands.init_cmd import run_init +from bedside.commands.step_cmd import run_step from bedside.exit_codes import SETUP_ERROR from bedside.result import CommandResult @@ -28,8 +30,9 @@ def build_parser() -> argparse.ArgumentParser: parser = argparse.ArgumentParser( prog="bedside", description=( - "Bedside CLI: pin operator manners, check adoption, eval fixtures. " - "Minimal Python front-end; command cores are UI-agnostic for a future tui-cs/cli." + "Bedside CLI: pin operator manners, check adoption, eval fixtures, " + "operator gates (ask/step). Minimal Python front-end; command cores " + "are UI-agnostic for a future tui-cs/cli." ), ) parser.add_argument( @@ -45,8 +48,15 @@ def build_parser() -> argparse.ArgumentParser: ) sub = parser.add_subparsers(dest="command", required=True) - p_init = sub.add_parser("init", help="write bedside.toml, domain notes, AGENTS stub") - p_init.add_argument("--pin", default="main", help="recorded pin tag or SHA") + p_init = sub.add_parser( + "init", + help="write bedside.toml, domain notes, AGENTS stub; optional vendor-copy", + ) + p_init.add_argument( + "--pin", + default="main", + help="recorded pin tag or SHA (default: main; with --vendor-from, git HEAD if main)", + ) p_init.add_argument( "--contract-path", default="third_party/bedside/contract", @@ -59,6 +69,27 @@ def build_parser() -> argparse.ArgumentParser: action="store_true", help="do not write BEDSIDE.md", ) + p_init.add_argument( + "--vendor-from", + type=Path, + default=None, + help="local tig/bedside checkout to copy into --vendor-dest (no submodule)", + ) + p_init.add_argument( + "--vendor-dest", + default="third_party/bedside", + help="destination for vendor-copy (default: third_party/bedside)", + ) + p_init.add_argument( + "--no-src", + action="store_true", + help="with --vendor-from, skip copying src/ (docs/fixtures only)", + ) + p_init.add_argument( + "--domain-fixtures", + default="eval/fixtures", + help="product fixture root outside vendor tree (default: eval/fixtures)", + ) p_doctor = sub.add_parser("doctor", help="check Bedside adoption health") p_doctor.add_argument( @@ -67,12 +98,18 @@ def build_parser() -> argparse.ArgumentParser: help="do not fail if contract path is missing (soft check)", ) - p_eval = sub.add_parser("eval", help="score fixture(s) against rubric R1-R9") + p_eval = sub.add_parser( + "eval", + help="score fixture dir(s) against rubric R1-R9 (multi-root OK)", + ) p_eval.add_argument( "paths", nargs="*", type=Path, - help="fixture dir(s) or tree; default: configured or packaged eval/fixtures", + help=( + "fixture dir(s) or trees; default: fixture_paths from bedside.toml " + "(vendored + domain)" + ), ) p_eval.add_argument( "--json", @@ -81,6 +118,85 @@ def build_parser() -> argparse.ArgumentParser: help="print machine-readable report", ) + p_ask = sub.add_parser( + "ask", + help="one structured operator choice (yes/no or multi-choice gate)", + ) + p_ask.add_argument( + "--id", + required=True, + dest="gate_id", + help="stable gate id for logs and eval", + ) + p_ask.add_argument( + "--prompt", + required=True, + help="plain-language question for the operator", + ) + p_ask.add_argument( + "--choices", + default="yes,no", + help="comma-separated choices (default: yes,no)", + ) + p_ask.add_argument( + "--default", + default=None, + help="recommended choice (default: first choice); shown first", + ) + p_ask.add_argument( + "--answer", + default=None, + help="non-interactive answer (label or 1-based index)", + ) + p_ask.add_argument( + "--json", + action="store_true", + dest="json_out", + help="print machine-readable result", + ) + + p_step = sub.add_parser( + "step", + help="one human body/browser act, then confirm in their words", + ) + p_step.add_argument( + "--id", + required=True, + dest="gate_id", + help="stable step id for logs and eval", + ) + p_step.add_argument( + "--prompt", + required=True, + help="one physical or browser instruction", + ) + p_step.add_argument( + "--expect", + default=None, + help="what the operator should be able to say when done (their words)", + ) + p_step.add_argument( + "--confirm", + action="store_true", + help="non-interactive: mark step confirmed", + ) + p_step.add_argument( + "--decline", + action="store_true", + help="non-interactive: mark step not confirmed", + ) + p_step.add_argument( + "--no-wait", + action="store_true", + help="show the step only; exit 10 (human still needed)", + ) + p_step.add_argument( + "--json", + action="store_true", + dest="json_out", + help="print machine-readable result", + ) + return parser @@ -90,8 +206,11 @@ def main(argv: list[str] | None = None) -> int: try: args = parser.parse_args(argv) except SystemExit as e: - code = e.code if isinstance(e.code, int) else SETUP_ERROR - return code + # argparse: 0 for --help/--version; 2 for usage errors. Agents branch on + # 10 vs 30, so map usage errors to SETUP_ERROR (not bare 2). + if e.code in (0, None): + return 0 + return SETUP_ERROR root: Path = args.root @@ -104,6 +223,10 @@ def main(argv: list[str] | None = None) -> int: force=args.force, skip_agents=args.skip_agents, skip_domain_notes=args.skip_domain_notes, + vendor_from=args.vendor_from, + vendor_dest=args.vendor_dest, + include_src=not args.no_src, + domain_fixtures=args.domain_fixtures, ) ) if args.command == "doctor": @@ -114,6 +237,40 @@ def main(argv: list[str] | None = None) -> int: return _print_result( run_eval(root, list(args.paths), json_out=args.json_out) ) + if args.command == "ask": + return _print_result( + run_ask( + gate_id=args.gate_id, + prompt=args.prompt, + choices=parse_choices(args.choices), + default=args.default, + answer=args.answer, + json_out=args.json_out, + ) + ) + if args.command == "step": + if args.confirm and args.decline: + r = CommandResult(SETUP_ERROR) + r.line("step: use only one of --confirm or --decline.") + r.line("What to do next: pick one flag, or omit both for interactive.") + return _print_result(r) + confirm: bool | None + if args.confirm: + confirm = True + elif args.decline: + confirm = False + else: + confirm = None + return _print_result( + run_step( + gate_id=args.gate_id, + prompt=args.prompt, + expect=args.expect, + confirm=confirm, + wait_confirm=not args.no_wait, + json_out=args.json_out, + ) + ) parser.error(f"unknown command: {args.command}") return SETUP_ERROR diff --git a/third_party/bedside/src/bedside/commands/ask_cmd.py b/third_party/bedside/src/bedside/commands/ask_cmd.py new file mode 100644 index 0000000..44b20e4 --- /dev/null +++ b/third_party/bedside/src/bedside/commands/ask_cmd.py @@ -0,0 +1,196 @@ +"""bedside ask: one structured operator choice (yes/no or multi-choice). + +UI-agnostic core. Argparse or a future tui-cs/cli host renders the same result. +Prefers integration with agent structured question UIs; CLI stdin is the fallback. +""" + +from __future__ import annotations + +import json +import sys +from collections.abc import Callable +from typing import TextIO + +from bedside.exit_codes import HUMAN_NEEDED, OK, SETUP_ERROR +from bedside.result import CommandResult + + +def parse_choices(raw: str) -> list[str]: + """Split comma-separated choices; preserve order; drop empties.""" + parts = [p.strip() for p in raw.split(",")] + return [p for p in parts if p] + + +def order_choices(choices: list[str], recommended: str) -> list[str]: + """Recommended first, then remaining choices in original order.""" + rest = [c for c in choices if c != recommended] + return [recommended] + rest + + +def resolve_choice(answer: str, choices: list[str]) -> str | None: + """Match by exact label or 1-based index. Case-insensitive label fallback.""" + answer = answer.strip() + if not answer: + return None + if answer in choices: + return answer + if answer.isdigit(): + idx = int(answer) + if 1 <= idx <= len(choices): + return choices[idx - 1] + lower_map = {c.lower(): c for c in choices} + return lower_map.get(answer.lower()) + + +def run_ask( + *, + gate_id: str, + prompt: str, + choices: list[str] | None = None, + default: str | None = None, + answer: str | None = None, + json_out: bool = False, + input_fn: Callable[[str], str] | None = None, + stdin_isatty: bool | None = None, + stdin: TextIO | None = None, +) -> CommandResult: + """One structured gate. Exit 0 if recommended chosen; 10 declined/needed; 30 setup.""" + gate_id = (gate_id or "").strip() + prompt = (prompt or "").strip() + if not gate_id: + r = CommandResult(SETUP_ERROR) + r.line("ask: --id is required (stable gate id for logs/eval).") + r.line("What to do next: `bedside ask --id NAME --prompt \"…\"`.") + return r + if not prompt: + r = CommandResult(SETUP_ERROR) + r.line("ask: --prompt is required (plain-language question).") + r.line("What to do next: pass --prompt with one short operator question.") + return r + + if choices is None: + choices = ["yes", "no"] + if len(choices) < 2: + r = CommandResult(SETUP_ERROR) + r.line("ask: need at least two --choices.") + r.line("What to do next: e.g. `--choices yes,no` or `--choices a,b,c`.") + return r + # Case-insensitive uniqueness (resolve_choice matches case-insensitively). + if len({c.lower() for c in choices}) != len(choices): + r = CommandResult(SETUP_ERROR) + r.line("ask: choices must be unique (case-insensitive).") + r.line("What to do next: remove duplicate labels from --choices (e.g. Yes/yes).") + return r + + recommended = default if default is not None else choices[0] + if recommended not in choices: + r = CommandResult(SETUP_ERROR) + r.line(f"ask: --default {recommended!r} is not in choices {choices}.") + r.line("What to do next: set --default to one of the choice labels.") + return r + + ordered = order_choices(choices, recommended) + + selected: str | None = None + interactive_flushed = False + if answer is not None: + selected = resolve_choice(answer, ordered) + if selected is None: + r = CommandResult(SETUP_ERROR) + r.line(f"ask: answer {answer!r} is not a valid choice.") + r.line(f"What to do next: pick one of {ordered} (or 1-{len(ordered)}).") + return r + else: + tty = sys.stdin.isatty() if stdin_isatty is None else stdin_isatty + if not tty and input_fn is None: + r = CommandResult(HUMAN_NEEDED) + _emit_prompt(r, gate_id, prompt, ordered, recommended) + r.line( + "Human action needed: pass --answer LABEL, or run in a TTY, " + "or use the agent host structured choice UI for this gate." + ) + r.line( + f"Record: bedside.ask id={gate_id} choice= pending " + f"recommended={recommended} matched=false" + ) + return r + pre = CommandResult(OK) + _emit_prompt(pre, gate_id, prompt, ordered, recommended) + # Print before blocking: CLI only flushes CommandResult after return. + _print_now(pre.messages) + interactive_flushed = True + try: + if input_fn is not None: + raw = input_fn("Answer: ") + else: + stream = stdin if stdin is not None else sys.stdin + sys.stdout.write("Answer: ") + sys.stdout.flush() + raw = stream.readline() + if not raw: + raw = "" + raw = raw.rstrip("\n\r") + except EOFError: + raw = "" + selected = resolve_choice(raw, ordered) + if selected is None: + r = CommandResult(SETUP_ERROR) + # Prompt already flushed; only error lines for CLI reprint. + r.line(f"ask: answer {raw!r} is not a valid choice.") + r.line(f"What to do next: pick one of {ordered} (or 1-{len(ordered)}).") + return r + + matched = selected == recommended + code = OK if matched else HUMAN_NEEDED + + payload = { + "id": gate_id, + "prompt": prompt, + "choices": ordered, + "recommended": recommended, + "choice": selected, + "matched_recommended": matched, + } + + r = CommandResult(code) + if json_out: + r.line(json.dumps(payload, ensure_ascii=False)) + return r + + if not interactive_flushed: + _emit_prompt(r, gate_id, prompt, ordered, recommended) + r.line(f"Selected: {selected}") + r.line(f"matched_recommended: {'true' if matched else 'false'}") + if not matched: + r.line( + "Not the recommended path. Agent should treat this as human declined " + "or alternate fork (exit 10)." + ) + r.line( + f"Record: bedside.ask id={gate_id} choice={selected} " + f"recommended={recommended} matched={'true' if matched else 'false'}" + ) + return r + + +def _print_now(messages: list[str]) -> None: + """Flush operator-facing lines before a blocking stdin read.""" + for line in messages: + print(line, flush=True) + + +def _emit_prompt( + r: CommandResult, + gate_id: str, + prompt: str, + ordered: list[str], + recommended: str, +) -> None: + r.line(f"Gate: {gate_id}") + r.line(prompt) + r.line("") + r.line("Choices (recommended first):") + for i, c in enumerate(ordered, start=1): + mark = " [recommended]" if c == recommended else "" + r.line(f" {i}. {c}{mark}") + r.line("") diff --git a/third_party/bedside/src/bedside/commands/eval_cmd.py b/third_party/bedside/src/bedside/commands/eval_cmd.py index a9ac482..0fde013 100644 --- a/third_party/bedside/src/bedside/commands/eval_cmd.py +++ b/third_party/bedside/src/bedside/commands/eval_cmd.py @@ -11,16 +11,27 @@ from bedside.result import CommandResult -def default_fixtures_dir(root: Path) -> Path | None: +def default_fixture_paths(root: Path) -> list[Path]: + """Resolve default fixture roots from bedside.toml or repo defaults.""" cfg = load_config(root) + paths: list[Path] = [] + if cfg and cfg.fixture_paths: + for rel in cfg.fixture_paths: + p = resolve_under(root, rel) + if p.is_dir(): + paths.append(p) + if paths: + return paths if cfg: cand = resolve_under(root, cfg.eval_path) / "fixtures" if cand.is_dir(): - return cand + paths.append(cand) pkg = repo_root_from_package() / "eval" / "fixtures" - if pkg.is_dir(): - return pkg - return None + if pkg.is_dir() and pkg not in paths: + # Prefer project paths; fall back to this checkout when developing bedside. + if not paths: + paths.append(pkg) + return paths def run_eval( @@ -33,14 +44,18 @@ def run_eval( targets: list[Path] = [] if not paths: - default = default_fixtures_dir(root) - if default is None: + defaults = default_fixture_paths(root) + if not defaults: r = CommandResult(SETUP_ERROR) r.line("No fixture path given and no eval/fixtures found.") - r.line("What to do next: pass a fixture dir, or `bedside init` and vendor tig/bedside.") + r.line( + "What to do next: pass fixture dir(s), set fixture_paths in bedside.toml, " + "or `bedside init` and vendor tig/bedside." + ) return r - paths = [default] + paths = defaults + seen: set[Path] = set() for p in paths: p = p if p.is_absolute() else (root / p) p = p.resolve() @@ -49,7 +64,10 @@ def run_eval( r.line(f"Path not found: {p}") r.line("What to do next: check the path; fixtures need meta.toml + transcript.md.") return r - targets.extend(iter_fixture_dirs(p)) + for fix in iter_fixture_dirs(p): + if fix not in seen: + seen.add(fix) + targets.append(fix) if not targets: r = CommandResult(SETUP_ERROR) @@ -78,6 +96,9 @@ def run_eval( "expect": rep.expect, "overall_pass": rep.overall_pass, "matched_expect": rep.matched_expect, + "focus": rep.focus, + "failed_focus": rep.failed_focus, + "info_failed": rep.info_failed, "principles": rep.principle_pass, "reasons": rep.reasons, } @@ -89,19 +110,27 @@ def run_eval( r.line(f"Bedside eval: {len(reports)} fixture(s)") for rep in reports: mark = "OK" if rep.ok else "FAIL" - bad = [k for k, v in rep.principle_pass.items() if not v] - r.line( - f" [{mark}] {rep.fixture_id} expect={rep.expect} " - f"scored={'pass' if rep.overall_pass else 'fail'}" - + (f" failed={','.join(bad)}" if bad else "") - ) + parts = [ + f"[{mark}] {rep.fixture_id}", + f"expect={rep.expect}", + f"scored={'pass' if rep.overall_pass else 'fail'}", + ] + if rep.failed_focus: + parts.append(f"failed={','.join(rep.failed_focus)}") + if rep.info_failed: + # Non-focus failures: informational only (do not drive expect). + parts.append(f"info={','.join(rep.info_failed)}") + r.line(" " + " ".join(parts)) if not rep.ok: for reason in rep.reasons[:5]: r.line(f" {reason}") if failed: r.line(f"{len(failed)} fixture(s) did not match expect.") - r.line("What to do next: fix agent path or fixture; re-run until known-bad fails manners and known-good passes.") + r.line( + "What to do next: fix agent path or fixture; re-run until known-bad fails " + "manners and known-good passes." + ) else: r.line("All fixtures matched expect.") r.line("What to do next: wire `bedside eval` into CI on operator-path changes.") diff --git a/third_party/bedside/src/bedside/commands/init_cmd.py b/third_party/bedside/src/bedside/commands/init_cmd.py index d0bcc3f..00b2b7d 100644 --- a/third_party/bedside/src/bedside/commands/init_cmd.py +++ b/third_party/bedside/src/bedside/commands/init_cmd.py @@ -2,11 +2,13 @@ from __future__ import annotations +import re from pathlib import Path from bedside.config import BedsideConfig, config_path, write_config from bedside.exit_codes import OK, SETUP_ERROR from bedside.result import CommandResult +from bedside.vendor import detect_pin, vendor_copy AGENTS_SECTION = """## Help the operator (Bedside) @@ -15,11 +17,13 @@ - Pin: see `bedside.toml` (do not soft-fork principles). - Normative contract path: `{contract_path}` +- Human gates: call `bedside ask` / `bedside step` (or the host structured + choice UI). Do not restate multi-choice free-text walls in this file. Summary (full contract is normative): 1. Assume low ops literacy, high judgment. -2. No wall of unexplained shell. +2. No wall of unexplained shell (or free-text choice walls). 3. Prefer doing over instructing. 4. Human acts: explicit, one step, dumb-simple. 5. Own first-time setup from zero. @@ -58,6 +62,10 @@ def run_init( force: bool = False, skip_agents: bool = False, skip_domain_notes: bool = False, + vendor_from: Path | None = None, + vendor_dest: str = "third_party/bedside", + include_src: bool = True, + domain_fixtures: str = "eval/fixtures", ) -> CommandResult: root = root.resolve() if not root.is_dir(): @@ -70,24 +78,64 @@ def run_init( if cfg_file.is_file() and not force: r = CommandResult(SETUP_ERROR) r.line(f"Already initialized: {cfg_file}") - r.line("What to do next: use `--force` to overwrite config, or edit bedside.toml.") + r.line( + "What to do next: use `--force` to overwrite config, or edit bedside.toml. " + "To re-vendor only: `bedside init --vendor-from --force`." + ) return r - contract = Path(contract_path) - # Prefer portable path strings in bedside.toml (forward slashes). - contract_s = contract.as_posix() if contract.is_absolute() else contract_path.replace("\\", "/") + r = CommandResult(OK) + effective_contract = contract_path.replace("\\", "/") + effective_pin = pin + + if vendor_from is not None: + src = vendor_from if vendor_from.is_absolute() else (root / vendor_from) + src = src.resolve() + dest = (root / vendor_dest).resolve() + try: + copied = vendor_copy(src, dest, include_src=include_src) + except (OSError, FileNotFoundError) as e: + bad = CommandResult(SETUP_ERROR) + bad.line(f"Vendor copy failed: {e}") + bad.line( + "What to do next: pass a local tig/bedside checkout to " + "`--vendor-from` (must contain contract/)." + ) + return bad + detected = detect_pin(src) + if pin == "main" and detected: + effective_pin = detected + effective_contract = f"{vendor_dest.rstrip('/')}/contract" + r.line(f"Vendored Bedside into {vendor_dest}/ ({', '.join(copied)})") + r.line(f"VENDOR.md pin stamp: {detected or 'unknown'}") + + contract = Path(effective_contract) + contract_s = ( + contract.as_posix() if contract.is_absolute() else effective_contract.replace("\\", "/") + ) parent = contract.parent - surface_s = (parent / "surface").as_posix() if contract.is_absolute() else (parent / "surface").as_posix() - eval_s = (parent / "eval").as_posix() if contract.is_absolute() else (parent / "eval").as_posix() + surface_s = (parent / "surface").as_posix() + eval_s = (parent / "eval").as_posix() + + # Domain fixtures outside the vendor tree so refresh does not wipe them. + domain = domain_fixtures.replace("\\", "/") + fixture_paths = [f"{eval_s}/fixtures"] + if domain and domain not in fixture_paths: + fixture_paths.append(domain) + cfg = BedsideConfig( - pin=pin, + pin=effective_pin, contract_path=contract_s, surface_path=surface_s, eval_path=eval_s, + fixture_paths=fixture_paths, ) write_config(root, cfg) - r = CommandResult(OK) - r.line(f"Wrote {cfg_file.relative_to(root)}") + try: + shown = str(cfg_file.relative_to(root)) + except ValueError: + shown = str(cfg_file) + r.line(f"Wrote {shown}") if not skip_domain_notes: notes = root / cfg.domain_notes @@ -105,9 +153,6 @@ def run_init( if "Help the operator (Bedside)" in text and not force: r.line("AGENTS.md already has a Bedside section; left unchanged.") elif "Help the operator (Bedside)" in text and force: - # replace section roughly from heading to next ## or EOF - import re - new_text, n = re.subn( r"## Help the operator \(Bedside\).*?(?=\n## |\Z)", section.rstrip() + "\n\n", @@ -132,11 +177,36 @@ def run_init( ) r.line("Wrote AGENTS.md with Bedside section") + if domain: + domain_dir = root / domain + (domain_dir / "known-bad").mkdir(parents=True, exist_ok=True) + (domain_dir / "known-good").mkdir(parents=True, exist_ok=True) + readme = domain_dir / "README.md" + if not readme.is_file() or force: + readme.write_text( + "# Domain fixtures\n\n" + "Product-specific Bedside transcripts live here.\n\n" + "They are **outside** `third_party/bedside` so re-vendor does not delete them.\n\n" + "Layout: `known-bad//{meta.toml,transcript.md}` and " + "`known-good//...` (same shape as upstream `eval/fixtures`).\n\n" + "Run: `bedside eval` (uses `fixture_paths` in bedside.toml) or " + "`bedside eval third_party/bedside/eval/fixtures eval/fixtures`.\n", + encoding="utf-8", + ) + r.line(f"Scaffolded domain fixture dirs under {domain}/") + r.line("") r.line(f"Pin recorded as: {cfg.pin}") r.line(f"Contract path: {cfg.contract_path}") + r.line(f"Fixture paths: {', '.join(cfg.fixture_paths)}") r.line("What to do next:") - r.line(" 1. Vendor or submodule tig/bedside so the contract path exists on disk.") + if vendor_from is None: + r.line(" 1. Put Bedside on disk (vendor-copy recommended; submodule also fine):") + r.line(" bedside init --vendor-from /path/to/tig/bedside --force") + r.line(" Docs: docs/adopting.md") + else: + r.line(" 1. Contract tree is under the vendor dest (see VENDOR.md there).") r.line(" 2. Fill domain notes in BEDSIDE.md (first-run, scary surfaces, Day-2).") - r.line(" 3. Run `bedside doctor` then `bedside eval` on your fixtures.") + r.line(" 3. Add domain fixtures under eval/fixtures (never under third_party/).") + r.line(" 4. Run `bedside doctor` then `bedside eval`.") return r diff --git a/third_party/bedside/src/bedside/commands/step_cmd.py b/third_party/bedside/src/bedside/commands/step_cmd.py new file mode 100644 index 0000000..74a3c27 --- /dev/null +++ b/third_party/bedside/src/bedside/commands/step_cmd.py @@ -0,0 +1,152 @@ +"""bedside step: one human body/browser act, then confirm in their words. + +UI-agnostic core. Encodes principles 4, 7, and 8 (one step, confirm, no cliff). +""" + +from __future__ import annotations + +import json +import sys +from collections.abc import Callable +from typing import TextIO + +from bedside.exit_codes import HUMAN_NEEDED, OK, SETUP_ERROR +from bedside.result import CommandResult + + +def _truthy(raw: str) -> bool | None: + s = raw.strip().lower() + if s in {"y", "yes", "true", "1", "confirmed", "done", "ok"}: + return True + if s in {"n", "no", "false", "0", "decline", "declined", "cancel", "stop"}: + return False + return None + + +def run_step( + *, + gate_id: str, + prompt: str, + expect: str | None = None, + confirm: bool | None = None, + wait_confirm: bool = True, + json_out: bool = False, + input_fn: Callable[[str], str] | None = None, + stdin_isatty: bool | None = None, + stdin: TextIO | None = None, +) -> CommandResult: + """One human step. Exit 0 if confirmed; 10 declined/needed; 30 setup.""" + gate_id = (gate_id or "").strip() + prompt = (prompt or "").strip() + if not gate_id: + r = CommandResult(SETUP_ERROR) + r.line("step: --id is required (stable step id for logs/eval).") + r.line("What to do next: `bedside step --id NAME --prompt \"…\"`.") + return r + if not prompt: + r = CommandResult(SETUP_ERROR) + r.line("step: --prompt is required (one physical or browser instruction).") + r.line("What to do next: pass a single dumb-simple human act as --prompt.") + return r + + if not wait_confirm: + # Show-only path still needs an explicit later confirm; exit human-needed. + r = CommandResult(HUMAN_NEEDED) + _emit_step(r, gate_id, prompt, expect) + r.line("Confirmation not requested on this run (--no-wait).") + r.line( + f"Record: bedside.step id={gate_id} confirmed=false wait=false" + ) + r.line( + "What to do next: re-run with --confirm after the operator completes the step." + ) + return r + + confirmed: bool | None = confirm + if confirmed is None: + tty = sys.stdin.isatty() if stdin_isatty is None else stdin_isatty + if not tty and input_fn is None: + r = CommandResult(HUMAN_NEEDED) + _emit_step(r, gate_id, prompt, expect) + r.line( + "Human action needed: complete the step, then pass --confirm " + "(or --decline), or answer in a TTY." + ) + r.line( + f"Record: bedside.step id={gate_id} confirmed=false pending=true" + ) + return r + pre = CommandResult(OK) + _emit_step(pre, gate_id, prompt, expect) + # Print before blocking: CLI only flushes CommandResult after return. + _print_now(pre.messages) + interactive_flushed = True + try: + if input_fn is not None: + raw = input_fn("Confirm (yes/no): ") + else: + stream = stdin if stdin is not None else sys.stdin + sys.stdout.write("Confirm (yes/no): ") + sys.stdout.flush() + raw = stream.readline() + if not raw: + raw = "" + raw = raw.rstrip("\n\r") + except EOFError: + raw = "" + parsed = _truthy(raw) + if parsed is None: + r = CommandResult(SETUP_ERROR) + # Step already flushed; only error lines for CLI reprint. + r.line(f"step: could not parse confirmation {raw!r}.") + r.line("What to do next: answer yes or no (or use --confirm / --decline).") + return r + confirmed = parsed + else: + interactive_flushed = False + + code = OK if confirmed else HUMAN_NEEDED + payload = { + "id": gate_id, + "prompt": prompt, + "expect": expect or "", + "confirmed": confirmed, + } + + r = CommandResult(code) + if json_out: + r.line(json.dumps(payload, ensure_ascii=False)) + return r + + if not interactive_flushed: + _emit_step(r, gate_id, prompt, expect) + r.line(f"Confirmed: {'true' if confirmed else 'false'}") + if not confirmed: + r.line( + "Step not confirmed. Do not continue the path (principle 8: no cliff)." + ) + r.line( + f"Record: bedside.step id={gate_id} confirmed=" + f"{'true' if confirmed else 'false'}" + ) + return r + + +def _print_now(messages: list[str]) -> None: + """Flush operator-facing lines before a blocking stdin read.""" + for line in messages: + print(line, flush=True) + + +def _emit_step( + r: CommandResult, + gate_id: str, + prompt: str, + expect: str | None, +) -> None: + r.line(f"Step: {gate_id}") + r.line(prompt) + if expect: + r.line("") + r.line(f"When done, you should be able to say: {expect}") + r.line("") diff --git a/third_party/bedside/src/bedside/config.py b/third_party/bedside/src/bedside/config.py index 5efad7b..fa694e2 100644 --- a/third_party/bedside/src/bedside/config.py +++ b/third_party/bedside/src/bedside/config.py @@ -3,7 +3,7 @@ from __future__ import annotations import tomllib -from dataclasses import dataclass +from dataclasses import dataclass, field from pathlib import Path CONFIG_NAME = "bedside.toml" @@ -19,9 +19,16 @@ class BedsideConfig: eval_path: str = "third_party/bedside/eval" domain_notes: str = "BEDSIDE.md" agents_stub: bool = True + # Extra fixture roots (product domain packs). Survive re-vendor of third_party/bedside. + fixture_paths: list[str] = field(default_factory=list) @classmethod def from_dict(cls, data: dict) -> BedsideConfig: + raw_paths = data.get("fixture_paths", []) + if isinstance(raw_paths, str): + paths = [raw_paths] + else: + paths = [str(p) for p in raw_paths] return cls( pin=str(data.get("pin", "main")), contract_path=str(data.get("contract_path", "third_party/bedside/contract")), @@ -29,6 +36,7 @@ def from_dict(cls, data: dict) -> BedsideConfig: eval_path=str(data.get("eval_path", "third_party/bedside/eval")), domain_notes=str(data.get("domain_notes", "BEDSIDE.md")), agents_stub=bool(data.get("agents_stub", True)), + fixture_paths=paths, ) @@ -53,16 +61,25 @@ def _toml_str(value: str) -> str: def write_config(root: Path, cfg: BedsideConfig) -> Path: path = config_path(root) - text = ( - f"# Bedside project config (see https://github.com/tig/bedside)\n" - f"pin = {_toml_str(cfg.pin)}\n" - f"contract_path = {_toml_str(cfg.contract_path)}\n" - f"surface_path = {_toml_str(cfg.surface_path)}\n" - f"eval_path = {_toml_str(cfg.eval_path)}\n" - f"domain_notes = {_toml_str(cfg.domain_notes)}\n" - f"agents_stub = {'true' if cfg.agents_stub else 'false'}\n" - ) - path.write_text(text, encoding="utf-8") + lines = [ + "# Bedside project config (see https://github.com/tig/bedside)", + f"pin = {_toml_str(cfg.pin)}", + f"contract_path = {_toml_str(cfg.contract_path)}", + f"surface_path = {_toml_str(cfg.surface_path)}", + f"eval_path = {_toml_str(cfg.eval_path)}", + f"domain_notes = {_toml_str(cfg.domain_notes)}", + f"agents_stub = {'true' if cfg.agents_stub else 'false'}", + ] + if cfg.fixture_paths: + lines.append("fixture_paths = [") + for p in cfg.fixture_paths: + lines.append(f" {_toml_str(p)},") + lines.append("]") + else: + lines.append( + "# fixture_paths = [\"third_party/bedside/eval/fixtures\", \"eval/fixtures\"]" + ) + path.write_text("\n".join(lines) + "\n", encoding="utf-8") return path diff --git a/third_party/bedside/src/bedside/eval_engine.py b/third_party/bedside/src/bedside/eval_engine.py index 935967b..5d1436c 100644 --- a/third_party/bedside/src/bedside/eval_engine.py +++ b/third_party/bedside/src/bedside/eval_engine.py @@ -28,15 +28,28 @@ class FixtureMeta: class ScoreReport: fixture_id: str expect: str - overall_pass: bool # did the session pass the rubric? + overall_pass: bool # did the session pass the rubric (focus only)? principle_pass: dict[str, bool] reasons: list[str] matched_expect: bool # overall_pass aligns with expect + focus: list[str] @property def ok(self) -> bool: return self.matched_expect + @property + def failed_focus(self) -> list[str]: + return [k for k in self.focus if not self.principle_pass.get(k, True)] + + @property + def info_failed(self) -> list[str]: + """Non-focus principles that failed; informational only.""" + focus_set = set(self.focus) + return sorted( + k for k, v in self.principle_pass.items() if not v and k not in focus_set + ) + def load_meta(path: Path) -> FixtureMeta: with path.open("rb") as f: @@ -109,7 +122,7 @@ def score_transcript(transcript: str) -> tuple[dict[str, bool], list[str]]: reasons: list[str] = [] p: dict[str, bool] = {rid: True for rid in PRINCIPLE_IDS} - # R2: no shell wall + # R2: no shell wall / no choice wall fences = _count_fenced_blocks(agent) cmd_lines = _commandish_lines(agent) run_these = bool( @@ -122,6 +135,25 @@ def score_transcript(transcript: str) -> tuple[dict[str, bool], list[str]]: p["R2"] = False reasons.append("R2: large command wall") + # Choice wall: free-text multi-option menu instead of structured UI + numbered_opts = len(re.findall(r"^\s*\d+[.)]\s+\S+", agent, re.M)) + free_text_pick = bool( + re.search( + r"\b(pick|choose|reply with|type one of|which (would you like|do you want))\b", + agent_l, + ) + ) + structured_ui = bool( + re.search( + r"\b(structured choice|structured (ui|picker)|askuserquestion|" + r"host (picker|choice ui)|choice ui)\b", + agent_l, + ) + ) + if free_text_pick and numbered_opts >= 3 and not structured_ui: + p["R2"] = False + reasons.append("R2: choice wall (multi-option free-text menu)") + # R3: prefer doing (instruct when agent could run) if run_these and not re.search( r"\bi (will|I'll|am going to) (run|install|create|execute)\b", agent_l @@ -142,7 +174,7 @@ def score_transcript(transcript: str) -> tuple[dict[str, bool], list[str]]: p["R6"] = False reasons.append("R6: blind auto on multi-candidate surface") - # R4: explicit human acts (vague batch) + # R4: explicit human acts (vague batch / free-text multi-choice as the act) batched = bool( re.search( r"\bdo (all of )?the following\b|\bsteps?:\s*\n.*\n.*\n.*\n", @@ -156,6 +188,9 @@ def score_transcript(transcript: str) -> tuple[dict[str, bool], list[str]]: if vague_physical or (batched and re.search(r"\bbrowser\b|\bplug\b|\bhold\b", agent_l)): p["R4"] = False reasons.append("R4: vague or batched human act") + if free_text_pick and numbered_opts >= 3 and not structured_ui: + p["R4"] = False + reasons.append("R4: human choice presented as free-text multi-menu") # R7: confirm in their words needs_human = bool( @@ -290,6 +325,7 @@ def evaluate_fixture_dir(fixture_dir: Path) -> ScoreReport: principle_pass=principle_pass, reasons=reasons, matched_expect=matched, + focus=list(focus), ) diff --git a/third_party/bedside/src/bedside/exit_codes.py b/third_party/bedside/src/bedside/exit_codes.py index c2e8831..0201f1a 100644 --- a/third_party/bedside/src/bedside/exit_codes.py +++ b/third_party/bedside/src/bedside/exit_codes.py @@ -1,6 +1,11 @@ """Stable exit codes for agent-first automation. Future tui-cs/cli adapters should preserve these values. + +0 OK (including recommended ask path / confirmed step) +10 Human needed, declined, or non-recommended ask choice +20 Manners fail (eval expect mismatch) +30 Tool or setup error """ OK = 0 diff --git a/third_party/bedside/src/bedside/vendor.py b/third_party/bedside/src/bedside/vendor.py new file mode 100644 index 0000000..8a7d023 --- /dev/null +++ b/third_party/bedside/src/bedside/vendor.py @@ -0,0 +1,94 @@ +"""Vendor-copy helpers (no git submodule required).""" + +from __future__ import annotations + +import shutil +import subprocess +from pathlib import Path + +# Trees consumers need for contract + eval + optional CLI install from path. +VENDOR_TOP_LEVEL = ( + "contract", + "surface", + "eval", + "src", + "pyproject.toml", + "LICENSE", + "README.md", +) + + +def detect_pin(source: Path) -> str | None: + """Best-effort commit SHA or tag if source is a git work tree.""" + try: + out = subprocess.run( + ["git", "-C", str(source), "rev-parse", "HEAD"], + check=True, + capture_output=True, + text=True, + ) + return out.stdout.strip() or None + except (OSError, subprocess.CalledProcessError): + return None + + +def vendor_copy( + source: Path, + dest: Path, + *, + include_src: bool = True, +) -> list[str]: + """Copy bedside artifacts from source repo root into dest. + + Replaces dest if it exists. Returns list of relative names copied. + """ + source = source.resolve() + dest = dest.resolve() + if not source.is_dir(): + raise FileNotFoundError(f"vendor source not a directory: {source}") + if not (source / "contract").is_dir(): + raise FileNotFoundError( + f"vendor source missing contract/: {source} " + "(point at a tig/bedside checkout, not a random folder)" + ) + + if dest.exists(): + shutil.rmtree(dest) + dest.mkdir(parents=True) + + copied: list[str] = [] + for name in VENDOR_TOP_LEVEL: + if name == "src" and not include_src: + continue + src_item = source / name + if not src_item.exists(): + continue + target = dest / name + if src_item.is_dir(): + shutil.copytree( + src_item, + target, + ignore=shutil.ignore_patterns( + "__pycache__", + "*.pyc", + ".pytest_cache", + "*.egg-info", + ".git", + ), + ) + else: + shutil.copy2(src_item, target) + copied.append(name) + + stamp = dest / "VENDOR.md" + pin = detect_pin(source) or "unknown" + stamp.write_text( + f"# Vendored Bedside\n\n" + f"- Source: `{source.as_posix()}`\n" + f"- Pin (best effort): `{pin}`\n" + f"- Refresh: re-run `bedside init --vendor-from --force` " + f"or copy the tree again; keep product fixtures outside this directory.\n", + encoding="utf-8", + ) + copied.append("VENDOR.md") + return copied diff --git a/third_party/bedside/surface/README.md b/third_party/bedside/surface/README.md index 870043e..dcbca08 100644 --- a/third_party/bedside/surface/README.md +++ b/third_party/bedside/surface/README.md @@ -10,7 +10,7 @@ Layer 2 of 3. Product patterns that encode operator manners in tools so agents a Without surface work, Bedside is doctrine agents may ignore. Without the [contract](../contract/), surface copy has no normative bar. Without [eval](../eval/), surface quality can rot unnoticed. -This repo ships a minimal Python CLI (`bedside init|doctor|eval`) as a first surface. Command cores are UI-agnostic for a future tui-cs/cli front-end; see the [root README](../README.md#cli-minimal-python). +This repo ships a minimal Python CLI (`bedside init|doctor|eval|ask|step`) as a first surface. Command cores are UI-agnostic for a future tui-cs/cli front-end; see the [root README](../README.md#cli-minimal-python). ## Purpose @@ -33,11 +33,39 @@ Thin, scriptable commands with operator-facing messages (not only machine logs). | Verb (examples) | Intent | |-----------------|--------| | `doctor` | Check host, tool, or device readiness; report in plain language; exit non-zero with recovery hints | +| `ask` | One structured yes/no or multi-choice operator gate; recommended option first; stable id for logs/eval | +| `step` | One human body/browser act: show instruction → confirm in their words → next | | `first-run` / `flash-first` | Own first-time setup from zero once; detect blank vs ready | | `deploy --verify` | Deploy then prove identity or version; fail closed on mismatch | | `gate` | Named host proof of done (tests or sim); green means claimable | -Agents should prefer these verbs over assembling ad-hoc shell walls. Humans should be able to re-run one documented verb on Day 2. +Agents should prefer these verbs over assembling ad-hoc shell walls or free-text multi-choice menus. Humans should be able to re-run one documented verb on Day 2. + +### Operator gates (`bedside ask` / `bedside step`) + +Products that pin Bedside should not restate long "how to ask the human" essays in every `AGENTS.md`. Point agents at these verbs (or the host structured UI that implements the same contract): + +```text +bedside ask --id confirm-deploy --prompt "Deploy to production now?" --choices yes,no --default no +bedside step --id plug-usb --prompt "Plug the data USB cable into the board." --expect "Power LED is on." +``` + +| Verb | Intent | +|------|--------| +| `ask` | One structured choice or yes/no with recommended default; plain-language prompt required; stable `--id` | +| `step` | One human body/browser act: show instruction → wait/confirm in their words → next | + +Rules: + +1. **UI-agnostic cores** under `commands/` so argparse, tui-cs/cli, or an agent host adapter can render pickers without rewriting behavior. +2. Prefer the **agent host structured choice UI** when present; CLI `--answer` / `--confirm` is the scriptable path; interactive stdin is the human fallback. +3. Exit codes: **0** chose recommended / step confirmed; **10** human declined, alternate fork, or still needed; **30** setup error (missing id/prompt, bad flags). +4. Each run prints a fixture-friendly `Record:` line (`bedside.ask` / `bedside.step`) so sessions can be scored later. +5. Scope is the **operator path** (setup, scary surfaces, irreversible writes, physical steps). Domain design taste stays in the product. + +Prefer `ask` / `step` (or host pickers) over multi-choice free-text walls. See also anti-walls below. + +Maps to contract principles 2, 4, 7, and 8. ### 2. Step machines (human body or account) @@ -89,7 +117,26 @@ If a tool must print a command for the human to paste (for example a browser dev Maps to contract principles 2 and 3. -### 6. Day-2 leave-behind +### 6. Structured choice UI (no free-text multi-choice) + +When the agent host provides a structured picker (AskUserQuestion-style tools, radio lists, plan-fork choosers), use it for: + +1. Yes/no gates before irreversible or scary work. +2. Plan forks the operator must choose among. +3. Multi-candidate picks that are not open-ended prose (accounts, regions, devices already listed by a discovery verb). + +Rules: + +1. Prefer the structured UI over a chat free-text menu of options. +2. Put the **recommended** option first (and label it when the API allows). +3. Keep free text for open-ended domain judgment the picker cannot express. +4. If no structured UI exists, one short numbered list is better than a shell wall, but it is still a weaker fallback. Prefer a surface verb that encodes the gate (see guided verbs and step machines). + +Anti-pattern (choice wall): a good plan table followed by "reply with start #15, fold issues, reprioritize P1, …" when a picker was available. + +Maps to contract principles 2 and 4. + +### 7. Day-2 leave-behind After success, the surface (or the docs it generates) leaves one routine path: diff --git a/third_party/bedside/tests/test_ask_step.py b/third_party/bedside/tests/test_ask_step.py new file mode 100644 index 0000000..a56a08c --- /dev/null +++ b/third_party/bedside/tests/test_ask_step.py @@ -0,0 +1,271 @@ +"""Tests for bedside ask and step operator gates.""" + +from pathlib import Path + +from bedside.cli import main +from bedside.commands.ask_cmd import order_choices, parse_choices, run_ask +from bedside.commands.step_cmd import run_step +from bedside.exit_codes import HUMAN_NEEDED, OK, SETUP_ERROR + + +def test_parse_and_order_choices(): + assert parse_choices("yes, no, maybe") == ["yes", "no", "maybe"] + assert order_choices(["a", "b", "c"], "b") == ["b", "a", "c"] + + +def test_ask_recommended_exit_zero(): + r = run_ask( + gate_id="confirm-deploy", + prompt="Deploy to production now?", + choices=["yes", "no"], + default="no", + answer="no", + ) + assert r.exit_code == OK + joined = "\n".join(r.messages) + assert "Selected: no" in joined + assert "matched_recommended: true" in joined + assert "bedside.ask id=confirm-deploy" in joined + + +def test_ask_non_recommended_exit_human_needed(): + r = run_ask( + gate_id="confirm-deploy", + prompt="Deploy to production now?", + choices=["yes", "no"], + default="no", + answer="yes", + ) + assert r.exit_code == HUMAN_NEEDED + assert "matched_recommended: false" in "\n".join(r.messages) + + +def test_ask_index_and_case(): + r = run_ask( + gate_id="fork", + prompt="Which path?", + choices=["continue", "pause", "rewrite"], + default="continue", + answer="2", + ) + assert r.exit_code == HUMAN_NEEDED + assert "Selected: pause" in "\n".join(r.messages) + + r2 = run_ask( + gate_id="fork", + prompt="Which path?", + choices=["continue", "pause"], + default="continue", + answer="CONTINUE", + ) + assert r2.exit_code == OK + + +def test_ask_requires_id_prompt(): + assert run_ask(gate_id="", prompt="x").exit_code == SETUP_ERROR + assert run_ask(gate_id="x", prompt="").exit_code == SETUP_ERROR + + +def test_ask_cli_missing_id_is_setup_error(): + """Argparse required flags must surface as exit 30, not argparse's 2.""" + code = main(["ask", "--prompt", "Go?"]) + assert code == SETUP_ERROR + + +def test_ask_case_insensitive_duplicate_choices(): + r = run_ask( + gate_id="g", + prompt="Go?", + choices=["Yes", "yes"], + answer="yes", + ) + assert r.exit_code == SETUP_ERROR + assert "case-insensitive" in "\n".join(r.messages) + + +def test_ask_noninteractive_without_answer(): + r = run_ask( + gate_id="g", + prompt="Go?", + answer=None, + stdin_isatty=False, + ) + assert r.exit_code == HUMAN_NEEDED + assert "Human action needed" in "\n".join(r.messages) + + +def test_ask_interactive_input_fn(): + r = run_ask( + gate_id="g", + prompt="Go?", + choices=["yes", "no"], + default="no", + input_fn=lambda _p: "no", + stdin_isatty=True, + ) + assert r.exit_code == OK + + +def test_ask_json(): + r = run_ask( + gate_id="g", + prompt="Go?", + default="no", + choices=["yes", "no"], + answer="no", + json_out=True, + ) + assert r.exit_code == OK + assert r.messages[0].startswith("{") + assert '"choice": "no"' in r.messages[0] + + +def test_step_confirm(): + r = run_step( + gate_id="plug-usb", + prompt="Plug the data USB cable into the board.", + expect="The board shows a power LED on.", + confirm=True, + ) + assert r.exit_code == OK + joined = "\n".join(r.messages) + assert "Step: plug-usb" in joined + assert "power LED" in joined + assert "Confirmed: true" in joined + assert "bedside.step id=plug-usb confirmed=true" in joined + + +def test_step_decline(): + r = run_step( + gate_id="plug-usb", + prompt="Plug the cable.", + confirm=False, + ) + assert r.exit_code == HUMAN_NEEDED + assert "Confirmed: false" in "\n".join(r.messages) + + +def test_step_pending_noninteractive(): + r = run_step( + gate_id="login", + prompt="Sign in in the browser.", + confirm=None, + stdin_isatty=False, + ) + assert r.exit_code == HUMAN_NEEDED + assert "Human action needed" in "\n".join(r.messages) + + +def test_step_interactive_input_fn(): + r = run_step( + gate_id="login", + prompt="Sign in.", + input_fn=lambda _p: "yes", + stdin_isatty=True, + ) + assert r.exit_code == OK + + +def test_step_no_wait(): + r = run_step( + gate_id="login", + prompt="Sign in.", + wait_confirm=False, + ) + assert r.exit_code == HUMAN_NEEDED + joined = "\n".join(r.messages) + assert "wait=false" in joined + assert "wait=true" not in joined + + +def test_ask_interactive_prints_before_read(capsys): + r = run_ask( + gate_id="g", + prompt="Deploy now?", + choices=["yes", "no"], + default="no", + input_fn=lambda _p: "no", + stdin_isatty=True, + ) + assert r.exit_code == OK + out = capsys.readouterr().out + assert "Gate: g" in out + assert "Deploy now?" in out + assert "Choices (recommended first):" in out + + +def test_step_interactive_prints_before_read(capsys): + r = run_step( + gate_id="plug-usb", + prompt="Plug the cable.", + expect="LED on", + input_fn=lambda _p: "yes", + stdin_isatty=True, + ) + assert r.exit_code == OK + out = capsys.readouterr().out + assert "Step: plug-usb" in out + assert "Plug the cable." in out + assert "LED on" in out + + +def test_ask_cli(): + code = main( + [ + "ask", + "--id", + "confirm-deploy", + "--prompt", + "Deploy?", + "--choices", + "yes,no", + "--default", + "no", + "--answer", + "no", + ] + ) + assert code == OK + + +def test_step_cli_confirm_decline_mutex(): + code = main( + [ + "step", + "--id", + "x", + "--prompt", + "Do it", + "--confirm", + "--decline", + ] + ) + assert code == SETUP_ERROR + + +def test_step_cli_confirm(): + code = main( + [ + "step", + "--id", + "plug-usb", + "--prompt", + "Plug USB", + "--expect", + "LED on", + "--confirm", + ] + ) + assert code == OK + + +def test_operator_gate_fixtures_match(): + """Shipped ask/step fixtures must exist and match expect.""" + repo = Path(__file__).resolve().parents[1] + from bedside.commands.eval_cmd import run_eval + + for name in ("operator-gate-ask", "operator-gate-step"): + path = repo / "eval" / "fixtures" / "known-good" / name + assert path.is_dir(), f"missing shipped fixture: {path}" + r = run_eval(repo, [path]) + assert r.exit_code == OK, (name, r.messages) diff --git a/third_party/bedside/tests/test_cli_commands.py b/third_party/bedside/tests/test_cli_commands.py index 23b9876..67437ad 100644 --- a/third_party/bedside/tests/test_cli_commands.py +++ b/third_party/bedside/tests/test_cli_commands.py @@ -2,22 +2,26 @@ from bedside.cli import main from bedside.commands.doctor_cmd import run_doctor +from bedside.commands.eval_cmd import run_eval from bedside.commands.init_cmd import run_init +from bedside.eval_engine import evaluate_fixture_dir from bedside.exit_codes import MANNERS_FAIL, OK, SETUP_ERROR def test_init_and_doctor(tmp_path: Path): - # point contract at this repo's contract for doctor OK repo = Path(__file__).resolve().parents[1] r = run_init( tmp_path, pin="v0.1.0", contract_path=str((repo / "contract").resolve()), + domain_fixtures="eval/fixtures", ) assert r.exit_code == OK assert (tmp_path / "bedside.toml").is_file() assert (tmp_path / "AGENTS.md").is_file() assert (tmp_path / "BEDSIDE.md").is_file() + toml = (tmp_path / "bedside.toml").read_text(encoding="utf-8") + assert "fixture_paths" in toml d = run_doctor(tmp_path) assert d.exit_code == OK, d.messages @@ -29,14 +33,53 @@ def test_init_refuses_without_force(tmp_path: Path): assert r.exit_code == SETUP_ERROR +def test_init_vendor_from(tmp_path: Path): + repo = Path(__file__).resolve().parents[1] + r = run_init( + tmp_path, + vendor_from=repo, + force=True, + include_src=False, + ) + assert r.exit_code == OK, r.messages + assert (tmp_path / "third_party" / "bedside" / "contract" / "README.md").is_file() + assert (tmp_path / "third_party" / "bedside" / "VENDOR.md").is_file() + assert not (tmp_path / "third_party" / "bedside" / "src").exists() + assert (tmp_path / "eval" / "fixtures" / "known-bad").is_dir() + d = run_doctor(tmp_path) + assert d.exit_code == OK, d.messages + + def test_eval_cli_shipped_fixtures(): repo = Path(__file__).resolve().parents[1] code = main(["--root", str(repo), "eval", str(repo / "eval" / "fixtures")]) assert code == OK +def test_eval_multi_root(tmp_path: Path): + repo = Path(__file__).resolve().parents[1] + # product domain root with one extra known-bad + domain = tmp_path / "eval" / "fixtures" / "known-bad" / "extra-wall" + domain.mkdir(parents=True) + (domain / "meta.toml").write_text( + 'id = "extra-wall"\nexpect = "fail"\nprinciples = ["R2"]\n', + encoding="utf-8", + ) + (domain / "transcript.md").write_text( + "## Agent\nRun these:\n\n```bash\npip install x\npytest\n```\n", + encoding="utf-8", + ) + r = run_eval( + tmp_path, + [repo / "eval" / "fixtures" / "known-good", tmp_path / "eval" / "fixtures"], + ) + assert r.exit_code == OK, r.messages + joined = "\n".join(r.messages) + assert "extra-wall" in joined + assert "step-and-confirm" in joined or "day2-leavebehind" in joined + + def test_eval_cli_detects_mismatch(tmp_path: Path): - # craft a "known-good" that is actually a shell wall fix = tmp_path / "bad-as-good" fix.mkdir() (fix / "meta.toml").write_text( @@ -49,3 +92,19 @@ def test_eval_cli_detects_mismatch(tmp_path: Path): ) code = main(["--root", str(tmp_path), "eval", str(fix)]) assert code == MANNERS_FAIL + + +def test_step_and_confirm_summary_uses_info_not_failed(): + repo = Path(__file__).resolve().parents[1] + rep = evaluate_fixture_dir(repo / "eval" / "fixtures" / "known-good" / "step-and-confirm") + assert rep.ok + assert rep.failed_focus == [] + # R9 may fail as non-focus; must not appear in failed_focus + r = run_eval( + repo, + [repo / "eval" / "fixtures" / "known-good" / "step-and-confirm"], + ) + line = next(m for m in r.messages if "step-and-confirm" in m) + assert "failed=" not in line + if rep.info_failed: + assert "info=" in line diff --git a/third_party/bedside/tests/test_eval_engine.py b/third_party/bedside/tests/test_eval_engine.py index b335151..af5b05c 100644 --- a/third_party/bedside/tests/test_eval_engine.py +++ b/third_party/bedside/tests/test_eval_engine.py @@ -31,3 +31,22 @@ def test_step_and_confirm_passes_focus(): assert p["R4"] is True assert p["R7"] is True assert p["R8"] is True + + +def test_choice_wall_fails_r2_r4(): + text = (FIXTURES / "known-bad" / "choice-wall" / "transcript.md").read_text( + encoding="utf-8" + ) + p, reasons = score_transcript(text) + assert p["R2"] is False + assert p["R4"] is False + assert any("choice wall" in r for r in reasons) + + +def test_structured_choice_passes_r2_r4(): + text = ( + FIXTURES / "known-good" / "structured-choice" / "transcript.md" + ).read_text(encoding="utf-8") + p, _ = score_transcript(text) + assert p["R2"] is True + assert p["R4"] is True From cade05f93e56c2dee8fb362e7a308ce5f4520caf Mon Sep 17 00:00:00 2001 From: Tig Date: Mon, 13 Jul 2026 16:58:10 -0600 Subject: [PATCH 2/2] fix(bedside): address PR 39 CR (vendor safety, ask/step records) - vendor_copy: refuse source==dest and dest nested under source - init: never overwrite existing BEDSIDE.md on --force re-vendor - ask: Record uses choice=pending; pre-prompts to stderr when --json - step: pre-step text to stderr when --json - tests for preserve notes, self-vendor, and pending record line Upstream-worthy; fixed in silico vendor tree as customer 0. --- .../bedside/src/bedside/commands/ask_cmd.py | 20 ++++++++++------ .../bedside/src/bedside/commands/init_cmd.py | 4 +++- .../bedside/src/bedside/commands/step_cmd.py | 18 +++++++++----- third_party/bedside/src/bedside/vendor.py | 18 ++++++++++++++ third_party/bedside/tests/test_ask_step.py | 6 ++++- .../bedside/tests/test_cli_commands.py | 24 +++++++++++++++++++ 6 files changed, 75 insertions(+), 15 deletions(-) diff --git a/third_party/bedside/src/bedside/commands/ask_cmd.py b/third_party/bedside/src/bedside/commands/ask_cmd.py index 44b20e4..1ba1720 100644 --- a/third_party/bedside/src/bedside/commands/ask_cmd.py +++ b/third_party/bedside/src/bedside/commands/ask_cmd.py @@ -110,22 +110,24 @@ def run_ask( "or use the agent host structured choice UI for this gate." ) r.line( - f"Record: bedside.ask id={gate_id} choice= pending " + f"Record: bedside.ask id={gate_id} choice=pending " f"recommended={recommended} matched=false" ) return r pre = CommandResult(OK) _emit_prompt(pre, gate_id, prompt, ordered, recommended) # Print before blocking: CLI only flushes CommandResult after return. - _print_now(pre.messages) + # Keep stdout JSON-clean when --json. + _print_now(pre.messages, to_stderr=json_out) interactive_flushed = True try: if input_fn is not None: raw = input_fn("Answer: ") else: stream = stdin if stdin is not None else sys.stdin - sys.stdout.write("Answer: ") - sys.stdout.flush() + prompt_stream = sys.stderr if json_out else sys.stdout + prompt_stream.write("Answer: ") + prompt_stream.flush() raw = stream.readline() if not raw: raw = "" @@ -173,10 +175,14 @@ def run_ask( return r -def _print_now(messages: list[str]) -> None: - """Flush operator-facing lines before a blocking stdin read.""" +def _print_now(messages: list[str], *, to_stderr: bool = False) -> None: + """Flush operator-facing lines before a blocking stdin read. + + When json_out is true, human text goes to stderr so stdout stays JSON-only. + """ + stream = sys.stderr if to_stderr else sys.stdout for line in messages: - print(line, flush=True) + print(line, file=stream, flush=True) def _emit_prompt( diff --git a/third_party/bedside/src/bedside/commands/init_cmd.py b/third_party/bedside/src/bedside/commands/init_cmd.py index 00b2b7d..ea754da 100644 --- a/third_party/bedside/src/bedside/commands/init_cmd.py +++ b/third_party/bedside/src/bedside/commands/init_cmd.py @@ -139,7 +139,9 @@ def run_init( if not skip_domain_notes: notes = root / cfg.domain_notes - if notes.is_file() and not force: + # Never clobber product domain notes on --force re-vendor; only scaffold + # when missing. Agents fill BEDSIDE.md once; refresh updates third_party only. + if notes.is_file(): r.line(f"Left existing {cfg.domain_notes}") else: notes.write_text(BEDSIDE_MD, encoding="utf-8") diff --git a/third_party/bedside/src/bedside/commands/step_cmd.py b/third_party/bedside/src/bedside/commands/step_cmd.py index 74a3c27..6304093 100644 --- a/third_party/bedside/src/bedside/commands/step_cmd.py +++ b/third_party/bedside/src/bedside/commands/step_cmd.py @@ -79,15 +79,17 @@ def run_step( pre = CommandResult(OK) _emit_step(pre, gate_id, prompt, expect) # Print before blocking: CLI only flushes CommandResult after return. - _print_now(pre.messages) + # Keep stdout JSON-clean when --json. + _print_now(pre.messages, to_stderr=json_out) interactive_flushed = True try: if input_fn is not None: raw = input_fn("Confirm (yes/no): ") else: stream = stdin if stdin is not None else sys.stdin - sys.stdout.write("Confirm (yes/no): ") - sys.stdout.flush() + prompt_stream = sys.stderr if json_out else sys.stdout + prompt_stream.write("Confirm (yes/no): ") + prompt_stream.flush() raw = stream.readline() if not raw: raw = "" @@ -132,10 +134,14 @@ def run_step( return r -def _print_now(messages: list[str]) -> None: - """Flush operator-facing lines before a blocking stdin read.""" +def _print_now(messages: list[str], *, to_stderr: bool = False) -> None: + """Flush operator-facing lines before a blocking stdin read. + + When json_out is true, human text goes to stderr so stdout stays JSON-only. + """ + stream = sys.stderr if to_stderr else sys.stdout for line in messages: - print(line, flush=True) + print(line, file=stream, flush=True) def _emit_step( diff --git a/third_party/bedside/src/bedside/vendor.py b/third_party/bedside/src/bedside/vendor.py index 8a7d023..7d9566f 100644 --- a/third_party/bedside/src/bedside/vendor.py +++ b/third_party/bedside/src/bedside/vendor.py @@ -52,6 +52,24 @@ def vendor_copy( "(point at a tig/bedside checkout, not a random folder)" ) + # Never rmtree the source. Same path, or dest nested under source, would + # delete the only copy of contract/src before copytree runs. + if source == dest: + raise ValueError( + f"vendor source and dest are the same path: {source}. " + "Point --vendor-from at an upstream tig/bedside checkout, not the " + "existing product vendor tree." + ) + try: + dest.relative_to(source) + except ValueError: + pass # dest is not under source + else: + raise ValueError( + f"vendor dest {dest} is inside source {source}; refusing to delete " + "a nested destination that would wipe the source tree." + ) + if dest.exists(): shutil.rmtree(dest) dest.mkdir(parents=True) diff --git a/third_party/bedside/tests/test_ask_step.py b/third_party/bedside/tests/test_ask_step.py index a56a08c..bac2cae 100644 --- a/third_party/bedside/tests/test_ask_step.py +++ b/third_party/bedside/tests/test_ask_step.py @@ -91,7 +91,11 @@ def test_ask_noninteractive_without_answer(): stdin_isatty=False, ) assert r.exit_code == HUMAN_NEEDED - assert "Human action needed" in "\n".join(r.messages) + joined = "\n".join(r.messages) + assert "Human action needed" in joined + # key=value form (not "choice= pending") + assert "choice=pending" in joined + assert "choice= pending" not in joined def test_ask_interactive_input_fn(): diff --git a/third_party/bedside/tests/test_cli_commands.py b/third_party/bedside/tests/test_cli_commands.py index 67437ad..4155a21 100644 --- a/third_party/bedside/tests/test_cli_commands.py +++ b/third_party/bedside/tests/test_cli_commands.py @@ -108,3 +108,27 @@ def test_step_and_confirm_summary_uses_info_not_failed(): assert "failed=" not in line if rep.info_failed: assert "info=" in line + + +def test_init_force_preserves_existing_bedside_md(tmp_path: Path): + """Re-vendor with --force must not wipe product domain notes.""" + notes = tmp_path / "BEDSIDE.md" + notes.write_text("# product metal notes\nkeep me\n", encoding="utf-8") + r = run_init(tmp_path, force=True) + assert r.exit_code == OK + text = notes.read_text(encoding="utf-8") + assert "keep me" in text + assert "Left existing BEDSIDE.md" in "\n".join(r.messages) + + +def test_vendor_copy_rejects_self_path(tmp_path: Path): + from bedside.vendor import vendor_copy + + src = tmp_path / "bedside" + (src / "contract").mkdir(parents=True) + (src / "contract" / "README.md").write_text("c\n", encoding="utf-8") + try: + vendor_copy(src, src) + raise AssertionError("expected ValueError for source == dest") + except ValueError as e: + assert "same path" in str(e)