From fbf13c68490ced44455624bade04af7877504e74 Mon Sep 17 00:00:00 2001 From: Tig Date: Mon, 13 Jul 2026 16:22:20 -0600 Subject: [PATCH] Document structured choice UI; score choice walls in eval. When agent hosts expose pickers, free-text multi-option menus are a choice wall (contract R2/R4). Surface pattern, known-bad/known-good fixtures, and rule heuristics make the anti-pattern falsifiable. Closes #5 --- contract/README.md | 4 +++ eval/README.md | 6 +++-- eval/fixtures/known-bad/choice-wall/meta.toml | 5 ++++ .../known-bad/choice-wall/transcript.md | 19 ++++++++++++++ .../known-good/structured-choice/meta.toml | 5 ++++ .../structured-choice/transcript.md | 23 ++++++++++++++++ src/bedside/eval_engine.py | 26 +++++++++++++++++-- surface/README.md | 21 ++++++++++++++- tests/test_eval_engine.py | 19 ++++++++++++++ 9 files changed, 123 insertions(+), 5 deletions(-) create mode 100644 eval/fixtures/known-bad/choice-wall/meta.toml create mode 100644 eval/fixtures/known-bad/choice-wall/transcript.md create mode 100644 eval/fixtures/known-good/structured-choice/meta.toml create mode 100644 eval/fixtures/known-good/structured-choice/transcript.md diff --git a/contract/README.md b/contract/README.md index 571d85d..b918caa 100644 --- a/contract/README.md +++ b/contract/README.md @@ -41,6 +41,8 @@ Do assume they can decide whether something should happen, confirm what they see Never paste five unexplained commands and say "run these." One step at a time. Say what it does. Run it yourself when you can. +Do not dump a **choice wall** either: a multi-option menu in free chat text when the agent host already has a structured picker (for example AskUserQuestion-style tools, radio buttons, or plan-fork choosers). Walls of shell and walls of choices both strand a non-expert. + ### 3. Prefer doing over instructing If you can install a tool, create a repo, run tests, call an API, or drive a CLI, do it. Only hand the human steps that require their body or their account: browser login, plugging hardware, holding a button, approving an OS prompt, reading an LED or UI state you cannot see. @@ -51,6 +53,7 @@ If you can install a tool, create a repo, run tests, call an API, or drive a CLI - Give the physical or click path once: not folklore, not "you know the drill." - Give the exact string to paste if they must type something you cannot run. - Do not assume agent UI tricks (special prefixes to run host commands, where to approve a tool, which terminal profile). Explain the path once. +- When the human must pick among plan forks or yes/no gates, and a **structured choice UI** exists, use it. Put the recommended option first. Free text remains correct for open-ended domain judgment the picker cannot capture. ### 5. Own first-time setup @@ -77,6 +80,7 @@ After success, leave one documented update or recovery path and what "good" look | Anti-pattern | Principle violated | |--------------|--------------------| | Unexplained multi-command dump | 2 (wall of shell) | +| Multi-choice free-text dump when a structured picker exists | 2 and 4 (choice wall / human acts) | | "Run this" when the agent could run it | 3 (prefer doing) | | Assumed prior install, flash, or login | 5 (first-time setup) | | Blind auto-select on multi-candidate hosts | 6 (scary surfaces) | diff --git a/eval/README.md b/eval/README.md index 0acab75..77f7cbc 100644 --- a/eval/README.md +++ b/eval/README.md @@ -69,9 +69,9 @@ Score agent sessions, CLI transcripts, or synthetic fixtures. Each item is pass | ID | Check | Contract principle | Fail if | |----|--------|--------------------|---------| | R1 | Low ops literacy | 1 | Assumes Git, COM, cloud, or agent-UI literacy without teaching in the moment | -| R2 | No shell wall | 2 | Two or more unexplained commands dumped as "run these" without agent execution or per-step explanation | +| R2 | No shell wall / no choice wall | 2 | Two or more unexplained commands dumped as "run these" without agent execution or per-step explanation; or a free-text multi-option menu (3+ numbered picks) when no structured choice UI is used | | R3 | Prefer doing | 3 | Instructs the human to run something the agent could run | -| R4 | Explicit human acts | 4 | Physical or browser step is vague, batched, or assumes UI folklore | +| R4 | Explicit human acts | 4 | Physical or browser step is vague, batched, or assumes UI folklore; or plan forks / gates dumped as free-text multi-choice instead of one dumb-simple act or structured picker | | R5 | First-run owned | 5 | Assumes runtime, firmware, or project already set up without detecting blank vs ready | | R6 | Scary surfaces plain | 6 | Blind auto on multi-candidate host; or failure with no next step in plain language | | R7 | Confirm in their words | 7 | Irreversible or physical step without a short world-check question | @@ -155,8 +155,10 @@ Runners may be human, script, or model-graded. The fixture content is the shared | Path | Expect | Principles | |------|--------|------------| | [`fixtures/known-bad/shell-wall/`](fixtures/known-bad/shell-wall/) | fail | R2, R3 | +| [`fixtures/known-bad/choice-wall/`](fixtures/known-bad/choice-wall/) | fail | R2, R4 | | [`fixtures/known-bad/left-at-cliff/`](fixtures/known-bad/left-at-cliff/) | fail | R8 | | [`fixtures/known-good/step-and-confirm/`](fixtures/known-good/step-and-confirm/) | pass | R4, R7, R8 | +| [`fixtures/known-good/structured-choice/`](fixtures/known-good/structured-choice/) | pass | R2, R4 | | [`fixtures/known-good/day2-leavebehind/`](fixtures/known-good/day2-leavebehind/) | pass | R9 | These are illustrative, domain-light transcripts. Domain packs should add richer fixtures (for example embedded first-flash) without changing R1 through R9. diff --git a/eval/fixtures/known-bad/choice-wall/meta.toml b/eval/fixtures/known-bad/choice-wall/meta.toml new file mode 100644 index 0000000..673693d --- /dev/null +++ b/eval/fixtures/known-bad/choice-wall/meta.toml @@ -0,0 +1,5 @@ +id = "choice-wall" +expect = "fail" +principles = ["R2", "R4"] +title = "Multi-choice free-text dump (choice wall)" +notes = "Agent offers plan forks as free chat options instead of a structured picker." diff --git a/eval/fixtures/known-bad/choice-wall/transcript.md b/eval/fixtures/known-bad/choice-wall/transcript.md new file mode 100644 index 0000000..1a3f175 --- /dev/null +++ b/eval/fixtures/known-bad/choice-wall/transcript.md @@ -0,0 +1,19 @@ +# choice-wall (known-bad) + +Domain-light illustration. Expect **fail** on R2 (choice wall) and R4 (human act is a free-text multi-menu). + +## Agent + +Here is the Phase F plan. Pick what you want me to do next by typing one of these in chat: + +1. start #15 (recommended) +2. fold open issues into the current PR +3. reprioritize all P1 items +4. skip Phase F and jump to deploy +5. rewrite the plan from scratch + +Just reply with the number or the short name. I will wait. + +## Operator + +Which of these is safest if I am not sure about deploy yet? diff --git a/eval/fixtures/known-good/structured-choice/meta.toml b/eval/fixtures/known-good/structured-choice/meta.toml new file mode 100644 index 0000000..7f55b07 --- /dev/null +++ b/eval/fixtures/known-good/structured-choice/meta.toml @@ -0,0 +1,5 @@ +id = "structured-choice" +expect = "pass" +principles = ["R2", "R4"] +title = "Plan fork via structured choice UI" +notes = "Agent uses host picker, recommended first, free text only for open judgment." diff --git a/eval/fixtures/known-good/structured-choice/transcript.md b/eval/fixtures/known-good/structured-choice/transcript.md new file mode 100644 index 0000000..6cd015d --- /dev/null +++ b/eval/fixtures/known-good/structured-choice/transcript.md @@ -0,0 +1,23 @@ +# structured-choice (known-good) + +Domain-light illustration. Expect **pass** on R2 and R4. + +## Agent + +Phase F is ready. I need one plan fork from you. + +I am calling the host structured choice UI (not a free-text menu): + +- id: phase-f-fork +- recommended first: continue with #15 +- other options: fold open issues into this PR; pause deploy until you review + +If none of those fit, say so in free text and we will adjust. I will not dump five numbered options for you to type. + +## Operator + +(structured UI) chose: continue with #15 + +## Agent + +Confirmed. Next I will implement #15 myself and report back. No shell wall for you. diff --git a/src/bedside/eval_engine.py b/src/bedside/eval_engine.py index de50980..5d1436c 100644 --- a/src/bedside/eval_engine.py +++ b/src/bedside/eval_engine.py @@ -122,7 +122,7 @@ def score_transcript(transcript: str) -> tuple[dict[str, bool], list[str]]: reasons: list[str] = [] p: dict[str, bool] = {rid: True for rid in PRINCIPLE_IDS} - # R2: no shell wall + # R2: no shell wall / no choice wall fences = _count_fenced_blocks(agent) cmd_lines = _commandish_lines(agent) run_these = bool( @@ -135,6 +135,25 @@ def score_transcript(transcript: str) -> tuple[dict[str, bool], list[str]]: p["R2"] = False reasons.append("R2: large command wall") + # Choice wall: free-text multi-option menu instead of structured UI + numbered_opts = len(re.findall(r"^\s*\d+[.)]\s+\S+", agent, re.M)) + free_text_pick = bool( + re.search( + r"\b(pick|choose|reply with|type one of|which (would you like|do you want))\b", + agent_l, + ) + ) + structured_ui = bool( + re.search( + r"\b(structured choice|structured (ui|picker)|askuserquestion|" + r"host (picker|choice ui)|choice ui)\b", + agent_l, + ) + ) + if free_text_pick and numbered_opts >= 3 and not structured_ui: + p["R2"] = False + reasons.append("R2: choice wall (multi-option free-text menu)") + # R3: prefer doing (instruct when agent could run) if run_these and not re.search( r"\bi (will|I'll|am going to) (run|install|create|execute)\b", agent_l @@ -155,7 +174,7 @@ def score_transcript(transcript: str) -> tuple[dict[str, bool], list[str]]: p["R6"] = False reasons.append("R6: blind auto on multi-candidate surface") - # R4: explicit human acts (vague batch) + # R4: explicit human acts (vague batch / free-text multi-choice as the act) batched = bool( re.search( r"\bdo (all of )?the following\b|\bsteps?:\s*\n.*\n.*\n.*\n", @@ -169,6 +188,9 @@ def score_transcript(transcript: str) -> tuple[dict[str, bool], list[str]]: if vague_physical or (batched and re.search(r"\bbrowser\b|\bplug\b|\bhold\b", agent_l)): p["R4"] = False reasons.append("R4: vague or batched human act") + if free_text_pick and numbered_opts >= 3 and not structured_ui: + p["R4"] = False + reasons.append("R4: human choice presented as free-text multi-menu") # R7: confirm in their words needs_human = bool( diff --git a/surface/README.md b/surface/README.md index 870043e..909b209 100644 --- a/surface/README.md +++ b/surface/README.md @@ -89,7 +89,26 @@ If a tool must print a command for the human to paste (for example a browser dev Maps to contract principles 2 and 3. -### 6. Day-2 leave-behind +### 6. Structured choice UI (no free-text multi-choice) + +When the agent host provides a structured picker (AskUserQuestion-style tools, radio lists, plan-fork choosers), use it for: + +1. Yes/no gates before irreversible or scary work. +2. Plan forks the operator must choose among. +3. Multi-candidate picks that are not open-ended prose (accounts, regions, devices already listed by a discovery verb). + +Rules: + +1. Prefer the structured UI over a chat free-text menu of options. +2. Put the **recommended** option first (and label it when the API allows). +3. Keep free text for open-ended domain judgment the picker cannot express. +4. If no structured UI exists, one short numbered list is better than a shell wall, but it is still a weaker fallback. Prefer a surface verb that encodes the gate (see guided verbs and step machines). + +Anti-pattern (choice wall): a good plan table followed by "reply with start #15, fold issues, reprioritize P1, …" when a picker was available. + +Maps to contract principles 2 and 4. + +### 7. Day-2 leave-behind After success, the surface (or the docs it generates) leaves one routine path: diff --git a/tests/test_eval_engine.py b/tests/test_eval_engine.py index b335151..af5b05c 100644 --- a/tests/test_eval_engine.py +++ b/tests/test_eval_engine.py @@ -31,3 +31,22 @@ def test_step_and_confirm_passes_focus(): assert p["R4"] is True assert p["R7"] is True assert p["R8"] is True + + +def test_choice_wall_fails_r2_r4(): + text = (FIXTURES / "known-bad" / "choice-wall" / "transcript.md").read_text( + encoding="utf-8" + ) + p, reasons = score_transcript(text) + assert p["R2"] is False + assert p["R4"] is False + assert any("choice wall" in r for r in reasons) + + +def test_structured_choice_passes_r2_r4(): + text = ( + FIXTURES / "known-good" / "structured-choice" / "transcript.md" + ).read_text(encoding="utf-8") + p, _ = score_transcript(text) + assert p["R2"] is True + assert p["R4"] is True