Skip to content

evals

evals #179

Workflow file for this run

name: evals
# Skillscope owns structure, selection, routing, and grading. OrchestrAI owns
# acquisition and execution of the default Linux/Windows Strix Halo sessions.
# The OrchestrAI controller receives only the selected skill/OS matrix and
# publishes normalized, allow-listed verdicts back to GitHub Actions.
on:
pull_request:
types: [opened, synchronize, reopened, labeled]
paths:
- "skills/**"
- ".github/workflows/evals.yml"
- ".github/orchestrai-config.json"
- ".github/scripts/build_orchestrai_plan.py"
- ".github/scripts/orchestrai_report.py"
- ".github/scripts/orchestrai_run.py"
- ".github/scripts/orchestrai_logs.py"
- ".github/scripts/orchestrai_verdict.py"
- ".github/scripts/test_orchestrai_evals.py"
- "**/*.md"
schedule:
- cron: "0 12 * * 1"
workflow_dispatch:
inputs:
mode:
description: "Which grader to run."
type: choice
options: [both, routing, behavioral, references]
default: both
skills:
description: "Comma-separated skill names (blank = every skill with a dataset)."
required: false
default: ""
only:
description: "Comma-separated case ids to run (routing only)."
required: false
default: ""
min_accuracy:
description: "Fail the routing run below this accuracy (0-1). 0 reports without gating."
required: false
default: "0"
orchestrai_mode:
description: "Use mock to validate the plan without allocating hardware."
type: choice
options: [live, mock]
default: live
orchestrai_os:
description: "Which Strix operating systems to run."
type: choice
options: [both, Linux, Windows]
default: both
concurrency:
# Unrelated label events are no-ops and use a unique group, so they cannot
# cancel or replace an in-flight graded run for the same pull request.
group: evals-${{ github.event.pull_request.number || github.ref }}-${{ github.event.action == 'labeled' && github.event.label.name != 'run_behavioral' && github.run_id || 'graded' }}
cancel-in-progress: ${{ github.event.action != 'labeled' || github.event.label.name == 'run_behavioral' }}
permissions:
contents: read
id-token: write
env:
SKILLSCOPE_REPOSITORY: amd/skillscope
SKILLSCOPE_VERSION: v0.1.3
ROUTING_ROOM: local-ai-use,local-ai-app-integration,serving-llms-on-instinct,tracelens-analysis-orchestrator,hyperloom-workload-optimizer
jobs:
discover:
name: ${{ github.event.action == 'labeled' && github.event.label.name != 'run_behavioral' && 'Ignore unrelated label (discover)' || 'Check dataset structure and select runs' }}
if: github.event.action != 'labeled' || github.event.label.name == 'run_behavioral'
runs-on: ubuntu-latest
permissions:
contents: read
outputs:
routing: ${{ steps.plan.outputs.routing }}
extended: ${{ steps.plan.outputs.extended }}
default: ${{ steps.plan.outputs.default }}
default_any: ${{ steps.plan.outputs.default_any }}
orchestrai: ${{ steps.plan.outputs.orchestrai }}
scoped: ${{ steps.plan.outputs.scoped }}
scoped_any: ${{ steps.plan.outputs.scoped_any }}
skipped: ${{ steps.plan.outputs.skipped }}
steps:
- name: Check out repository
uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6.1.0
with:
fetch-depth: 0
- name: Check out Skillscope
uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6.1.0
with:
repository: ${{ env.SKILLSCOPE_REPOSITORY }}
ref: ${{ env.SKILLSCOPE_VERSION }}
path: .skillscope-action
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0
with:
python-version: "3.12"
# Install the pinned checkout directly. The enterprise Actions policy
# does not allow the third-party setup-uv action used by the upstream
# Skillscope composite action.
- name: Install Skillscope
run: python -m pip install --disable-pip-version-check ./.skillscope-action
- name: Check every skill's structure
run: >-
python -m skillscope --repo . structural
--skills-dir 'skills/*'
--docs '*.md,docs/**/*.md,walkthroughs/*.md,staging/**/*.md'
--skill-files 'skill-card.md'
--skill-sections 'Description,Owner,License'
- name: Test OrchestrAI plan and trigger helpers
run: python3 .github/scripts/test_orchestrai_evals.py
- name: Work out how to select
id: how
shell: python
env:
EVENT: ${{ github.event_name }}
SKILLS: ${{ inputs.skills }}
LABELS: ${{ join(github.event.pull_request.labels.*.name, ',') }}
BASE: ${{ github.event.pull_request.base.sha }}
HEAD: ${{ github.event.pull_request.head.sha }}
ROUTING_ROOM_VALUE: ${{ env.ROUTING_ROOM }}
run: |
import os
import shlex
if os.environ["EVENT"] == "pull_request" and os.environ.get("BASE"):
args = [
"--since", os.environ["BASE"], os.environ.get("HEAD", ""),
"--labels", os.environ.get("LABELS", ""), "--no-extended",
]
else:
skills = os.environ.get("SKILLS", "").strip()
args = ["--names", skills] if skills else ["--all"]
args += ["--ignore-gates", "--no-extended"]
args += [
"--routing-room", os.environ["ROUTING_ROOM_VALUE"],
"--infra-paths", ".github/workflows/evals.yml,.github/orchestrai-config.json,.github/scripts/build_orchestrai_plan.py,.github/scripts/orchestrai_run.py,.github/scripts/orchestrai_verdict.py",
"--behavior-runner", '["orchestrai"]',
"--behavior-os", "Linux,Windows",
"--scoped-runner", '["self-hosted","Linux","X64"]',
"--scoped-gate", "enable_mi_ci",
"--scoped-environment", "behavioral-instinct",
]
line = "args=" + " ".join(shlex.quote(arg) for arg in args)
print(line)
with open(os.environ["GITHUB_OUTPUT"], "a", encoding="utf-8") as output:
output.write(line + "\n")
- name: Select what to run
id: select
shell: python
env:
SELECT_ARGS: ${{ steps.how.outputs.args }}
run: |
import os
import shlex
import subprocess
import sys
command = [
sys.executable, "-m", "skillscope", "--repo", ".", "select",
"--skills-dir", "skills/*", *shlex.split(os.environ["SELECT_ARGS"]),
]
completed = subprocess.run(command, text=True, capture_output=True)
sys.stderr.write(completed.stderr)
print(completed.stdout, end="")
if completed.returncode:
raise SystemExit(completed.returncode)
lines = [line for line in completed.stdout.splitlines() if line.strip()]
with open(os.environ["GITHUB_OUTPUT"], "a", encoding="utf-8") as output:
output.write("stdout=" + (lines[-1] if lines else "") + "\n")
- name: Apply requested mode and emit plan
id: plan
shell: python
env:
PLAN: ${{ steps.select.outputs.stdout }}
MODE: ${{ github.event_name == 'schedule' && 'references' || inputs.mode || 'both' }}
EVENT: ${{ github.event_name }}
ORCHESTRAI_OS: ${{ inputs.orchestrai_os || 'both' }}
run: |
import json
import os
plan = json.loads(os.environ["PLAN"])
mode = os.environ.get("MODE", "both")
if mode in ("behavioral", "behavior", "references"):
plan["routing"] = False
if mode in ("routing", "references"):
plan["default"] = []
plan["scoped"] = []
requested_os = os.environ.get("ORCHESTRAI_OS", "both")
if os.environ.get("EVENT") == "workflow_dispatch" and requested_os != "both":
plan["default"] = [
entry for entry in plan["default"]
if entry.get("os") == requested_os
]
# Keep the GitHub verdict matrix aligned with the plan builder's
# one-session-per-skill/OS identity, even if selection ever emits a
# duplicate entry.
seen = set()
unique_default = []
for entry in plan["default"]:
identity = (entry.get("skill"), entry.get("os"))
if identity not in seen:
seen.add(identity)
unique_default.append(entry)
plan["default"] = unique_default
orchestrai = []
for os_name in ("Linux", "Windows"):
entries = [
entry for entry in unique_default if entry.get("os") == os_name
]
if entries:
orchestrai.append({
"os": os_name,
"slug": os_name.lower(),
"matrix_json": json.dumps(entries, separators=(",", ":")),
})
lines = [
"routing=" + str(plan["routing"]).lower(),
"extended=" + ("--extended" if plan["extended"] else "--no-extended"),
"orchestrai=" + json.dumps(orchestrai, separators=(",", ":")),
]
for key in ("default", "scoped", "skipped"):
lines.append(key + "=" + json.dumps(plan[key], separators=(",", ":")))
for key in ("default", "scoped"):
lines.append(key + "_any=" + str(bool(plan[key])).lower())
print("\n".join(lines))
with open(os.environ["GITHUB_OUTPUT"], "a", encoding="utf-8") as output:
output.write("\n".join(lines) + "\n")
external-references:
name: ${{ github.event.action == 'labeled' && github.event.label.name != 'run_behavioral' && 'Ignore unrelated label (references)' || 'External references' }}
if: github.event.action != 'labeled' || github.event.label.name == 'run_behavioral'
runs-on: ubuntu-latest
permissions:
contents: read
steps:
- uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6.1.0
- name: Check out Skillscope
uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6.1.0
with:
repository: ${{ env.SKILLSCOPE_REPOSITORY }}
ref: ${{ env.SKILLSCOPE_VERSION }}
path: .skillscope-action
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0
with:
python-version: "3.12"
- name: Install Skillscope
run: python -m pip install --disable-pip-version-check ./.skillscope-action
- name: Fetch every external reference
run: >-
python -m skillscope --repo . structural
--skills-dir 'skills/*'
--external
--docs '*.md,docs/**/*.md,walkthroughs/*.md,staging/**/*.md'
--exclude-url '^https://openai\.com/codex/?$,^https://www\.amd\.com/en/resources/product-security\.html$'
routing:
name: Routing
needs: discover
if: needs.discover.outputs.routing == 'true'
# Routing exercises the hosted LLM gateway only. Use the control runner for
# its network path, while keeping direct Strix access exclusive to
# OrchestrAI-managed tests.
runs-on: ${{ vars.ORCHESTRAI_CONTROL_RUNNER || 'ubuntu-latest' }}
timeout-minutes: 40
steps:
- uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6.1.0
- name: Check out Skillscope
uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6.1.0
with:
repository: ${{ env.SKILLSCOPE_REPOSITORY }}
ref: ${{ env.SKILLSCOPE_VERSION }}
path: .skillscope-action
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0
with:
python-version: "3.12"
- name: Install Skillscope
run: python -m pip install --disable-pip-version-check ./.skillscope-action
- name: Resolve model credentials
shell: python
env:
API_KEY: ${{ secrets.ORCHESTR_API_KEY }}
API_BASE_URL: https://llm-api.amd.com/Anthropic
API_CUSTOM_HEADERS: |
Ocp-Apim-Subscription-Key: $API_KEY
user: a1_ucicd
SECRET_NAME: ORCHESTR_API_KEY
run: |
import subprocess
import sys
raise SystemExit(subprocess.run([sys.executable, ".skillscope-action/skillscope/credentials.py"]).returncode)
- uses: actions/setup-node@249970729cb0ef3589644e2896645e5dc5ba9c38 # v6
with:
node-version: "20"
- name: Install Claude Code
run: npm install -g @anthropic-ai/claude-code
- name: Run routing evals
env:
MIN_ACCURACY: ${{ inputs.min_accuracy || '0' }}
ONLY: ${{ inputs.only }}
EXTENDED_FLAG: ${{ needs.discover.outputs.extended }}
run: |
args=(
python -m skillscope --repo . routing
--skills-dir 'skills/*'
--routing-room "$ROUTING_ROOM"
--min-accuracy "$MIN_ACCURACY"
"$EXTENDED_FLAG"
--output routing-report.json
--keep-logs routing-logs
)
if [[ -n "$ONLY" ]]; then args+=(--only "$ONLY"); fi
"${args[@]}"
- name: Upload routing report
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: routing-report
path: |
routing-report.json
routing-logs/
if-no-files-found: warn
orchestrai-behavioral:
name: Behavioral (OrchestrAI Strix Halo — ${{ matrix.os }})
needs: discover
if: >-
needs.discover.outputs.default_any == 'true' &&
(github.event_name != 'pull_request' ||
(contains(github.event.pull_request.labels.*.name, 'run_behavioral') &&
github.event.pull_request.head.repo.full_name == github.repository))
runs-on: ${{ vars.ORCHESTRAI_CONTROL_RUNNER || 'ubuntu-latest' }}
timeout-minutes: 240
strategy:
fail-fast: false
matrix:
include: ${{ fromJSON(needs.discover.outputs.orchestrai) }}
steps:
- uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6.1.0
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0
with:
python-version: "3.12"
- name: Build OrchestrAI plan
env:
MATRIX_JSON: ${{ matrix.matrix_json }}
TARGET_REF: ${{ github.event.pull_request.head.ref || github.ref_name }}
# A PR must test the immutable commit GitHub associated with this
# check run. Non-PR events use the selected ref's triggering commit.
TARGET_SHA: ${{ github.event.pull_request.head.sha || github.sha }}
EXTENDED_FLAG: ${{ needs.discover.outputs.extended }}
# Fleet selectors are private infrastructure metadata. Store this
# JSON as a secret so Actions masks the value in the step's automatic
# environment preamble before our code starts.
DEVICE_TAGS_JSON: ${{ secrets.ORCHESTRAI_DEVICE_TAGS }}
# Same per-Ubuntu-release mapping used by Playbooks. Keep it as a
# secret here so its URLs are masked in this public repository's logs.
ORCHESTRAI_LINUX_DRIVER_SOURCES_JSON: ${{ secrets.ORCHESTRAI_LINUX_DRIVER_SOURCES_JSON }}
SOURCE_RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
run: >-
python3 .github/scripts/build_orchestrai_plan.py
--matrix-json "$MATRIX_JSON"
--device-tags-json "$DEVICE_TAGS_JSON"
--repository "${{ github.repository }}"
--ref "$TARGET_REF"
--sha "$TARGET_SHA"
--extended-flag="$EXTENDED_FLAG"
--run-id "${{ github.run_id }}"
--run-attempt "${{ github.run_attempt }}"
--source-run-url "$SOURCE_RUN_URL"
--output orchestrai-plan.json
- name: Acquire Strix machines and run behavioral evals
id: orchestrai
env:
PLAN_FILE: orchestrai-plan.json
MODE: ${{ github.event_name == 'workflow_dispatch' && inputs.orchestrai_mode || 'live' }}
POLL_TIMEOUT_SEC: "13800"
# Match Playbooks' five-minute continuous-unreachable allowance.
# A successful poll resets this budget.
POLL_ERROR_TIMEOUT_SEC: "300"
ORCHESTRAI_PORTAL_URL: ${{ secrets.ORCHESTRAI_PORTAL_URL }}
ORCHESTRAI_USER: ${{ secrets.ORCHESTRAI_USER }}
ORCHESTRAI_PASSWORD: ${{ secrets.ORCHESTRAI_PASSWORD }}
ORCHESTRAI_SPACE: ${{ secrets.ORCHESTRAI_SPACE }}
run: exec python3 .github/scripts/orchestrai_run.py
- name: Upload OrchestrAI result manifest
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: orchestrai-run-results-${{ matrix.slug }}
# The live controller snapshot can contain private pipeline topology
# (for example, stage names). Verdict jobs need only the normalized,
# allow-listed manifest, so keep the snapshot inside this ephemeral
# job workspace and out of public workflow artifacts.
path: orchestrai-results.json
if-no-files-found: warn
orchestrai-verdict:
name: Behavioral (${{ matrix.skill }} on ${{ matrix.os }})
needs: [discover, orchestrai-behavioral]
if: >-
!cancelled() &&
needs.discover.result == 'success' &&
needs.discover.outputs.default_any == 'true' &&
(github.event_name != 'pull_request' ||
(contains(github.event.pull_request.labels.*.name, 'run_behavioral') &&
github.event.pull_request.head.repo.full_name == github.repository))
runs-on: ${{ vars.ORCHESTRAI_WAIT_RUNNER || 'ubuntu-latest' }}
strategy:
fail-fast: false
matrix:
include: ${{ fromJSON(needs.discover.outputs.default) }}
steps:
- uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6.1.0
- name: Download OrchestrAI results
continue-on-error: true
uses: actions/download-artifact@37930b1c2abaa49bbe596cd826c3c89aef350131 # v7.0.0
with:
name: orchestrai-run-results-${{ matrix.os == 'Linux' && 'linux' || 'windows' }}
path: .orchestrai-report
- name: Resolve skill verdict
id: verdict
run: >-
python3 .github/scripts/orchestrai_verdict.py
--results .orchestrai-report/orchestrai-results.json
--skill "${{ matrix.skill }}"
--os "${{ matrix.os }}"
--output-dir test-results
- name: Upload skill result
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: test-results-${{ matrix.skill }}-${{ matrix.os }}
path: test-results/
if-no-files-found: warn
behavior-scoped:
name: Behavioral (${{ matrix.skill }} on ${{ matrix.os }}, ${{ matrix.environment }})
needs: discover
if: needs.discover.outputs.scoped_any == 'true'
runs-on: ${{ fromJSON(matrix.runner) }}
timeout-minutes: 45
permissions:
contents: read
id-token: write
environment:
name: ${{ matrix.environment }}
strategy:
fail-fast: false
matrix:
include: ${{ fromJSON(needs.discover.outputs.scoped) }}
steps:
- uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6.1.0
- name: Check out Skillscope
uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6.1.0
with:
repository: ${{ env.SKILLSCOPE_REPOSITORY }}
ref: ${{ env.SKILLSCOPE_VERSION }}
path: .skillscope-action
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0
with:
python-version: "3.12"
- name: Install Skillscope
run: python -m pip install --disable-pip-version-check ./.skillscope-action
- name: Resolve model credentials
shell: python
env:
API_KEY: ""
API_BASE_URL: ""
API_CUSTOM_HEADERS: ""
SECRET_NAME: ""
ENVIRONMENT: ${{ matrix.environment }}
FEDERATION_RULE_ID: fdrl_014u39zsJmsex5SkL1gWg7qo
FEDERATION_ORGANIZATION_ID: 9bdff596-1595-4140-8301-817e43ef8412
FEDERATION_SERVICE_ACCOUNT_ID: svac_016zcTqZ4XkFdjAJCag9jQC9
FEDERATION_WORKSPACE_ID: wrkspc_01EUPtU7mqFtoQowMAREPvoq
run: |
import subprocess
import sys
raise SystemExit(subprocess.run([sys.executable, ".skillscope-action/skillscope/credentials.py"]).returncode)
- uses: actions/setup-node@249970729cb0ef3589644e2896645e5dc5ba9c38 # v6
with:
node-version: "20"
- name: Install Claude Code
run: npm install -g @anthropic-ai/claude-code
- name: Run behavioral cases
env:
SKILL: ${{ matrix.skill }}
EXTENDED_FLAG: ${{ needs.discover.outputs.extended }}
run: >-
python -m skillscope --repo . behavioral
--skills-dir 'skills/*'
--skill "$SKILL"
"$EXTENDED_FLAG"
results:
name: ${{ github.event.action == 'labeled' && github.event.label.name != 'run_behavioral' && 'Ignore unrelated label (results)' || 'evals / results' }}
needs: [discover, routing, orchestrai-behavioral, orchestrai-verdict, behavior-scoped]
if: >-
always() &&
(github.event.action != 'labeled' || github.event.label.name == 'run_behavioral')
runs-on: ubuntu-latest
permissions:
contents: read
steps:
- uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6.1.0
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0
with:
python-version: "3.12"
- name: Download per-skill OrchestrAI results
if: >-
needs.discover.outputs.default_any == 'true' &&
(github.event_name != 'pull_request' ||
(contains(github.event.pull_request.labels.*.name, 'run_behavioral') &&
github.event.pull_request.head.repo.full_name == github.repository))
continue-on-error: true
uses: actions/download-artifact@37930b1c2abaa49bbe596cd826c3c89aef350131 # v7.0.0
with:
pattern: test-results-*
path: .orchestrai-results
- name: Publish aggregate OrchestrAI report
if: >-
needs.discover.outputs.default_any == 'true' &&
(github.event_name != 'pull_request' ||
(contains(github.event.pull_request.labels.*.name, 'run_behavioral') &&
github.event.pull_request.head.repo.full_name == github.repository))
env:
EXPECTED_JSON: ${{ needs.discover.outputs.default }}
ORCHESTRAI_CONTROLLER_RESULT: ${{ needs.orchestrai-behavioral.result }}
ORCHESTRAI_VERDICT_RESULT: ${{ needs.orchestrai-verdict.result }}
run: >-
python3 .github/scripts/orchestrai_report.py
--artifacts .orchestrai-results
--expected-json "$EXPECTED_JSON"
--output orchestrai-report.md
- name: Upload aggregate OrchestrAI report
if: >-
needs.discover.outputs.default_any == 'true' &&
(github.event_name != 'pull_request' ||
(contains(github.event.pull_request.labels.*.name, 'run_behavioral') &&
github.event.pull_request.head.repo.full_name == github.repository))
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: orchestrai-report
path: orchestrai-report.md
if-no-files-found: warn
- name: Verify eval results
shell: python
env:
EVENT: ${{ github.event_name }}
LABELS: ${{ join(github.event.pull_request.labels.*.name, ',') }}
HEAD_REPOSITORY: ${{ github.event.pull_request.head.repo.full_name }}
REPOSITORY: ${{ github.repository }}
DISCOVER: ${{ needs.discover.result }}
ROUTING: ${{ needs.routing.result }}
ORCHESTRAI: ${{ needs.orchestrai-behavioral.result }}
ORCHESTRAI_VERDICTS: ${{ needs.orchestrai-verdict.result }}
SCOPED: ${{ needs.behavior-scoped.result }}
ROUTING_WANTED: ${{ needs.discover.outputs.routing }}
BEHAVIOR_WANTED: ${{ needs.discover.outputs.default_any }}
SCOPED_WANTED: ${{ needs.discover.outputs.scoped_any }}
SKIPPED: ${{ needs.discover.outputs.skipped }}
run: |
import os
import sys
value = lambda name: os.environ.get(name, "")
labels = {label for label in value("LABELS").split(",") if label}
is_pull_request = value("EVENT") == "pull_request"
trusted_head = not is_pull_request or value("HEAD_REPOSITORY") == value("REPOSITORY")
orchestrai_authorized = not is_pull_request or (
"run_behavioral" in labels and trusted_head
)
print(f"discover: {value('DISCOVER')}")
print(f"routing: {value('ROUTING')} (requested: {value('ROUTING_WANTED')})")
print(f"orchestrai-behavioral:{value('ORCHESTRAI')} (requested: {value('BEHAVIOR_WANTED')}, authorized: {orchestrai_authorized})")
print(f"orchestrai-verdicts: {value('ORCHESTRAI_VERDICTS')}")
print(f"behavioral-scoped: {value('SCOPED')} (requested: {value('SCOPED_WANTED')})")
print(f"held back: {value('SKIPPED')}")
if value("DISCOVER") != "success":
sys.exit(f"The discover job did not succeed ({value('DISCOVER')}).")
skipped = value("SKIPPED").strip()
if skipped and skipped != "[]":
print(f"::warning::Scoped behavioral cases were held back by a missing gate: {skipped}")
failures = []
if value("ROUTING_WANTED") == "true" and value("ROUTING") != "success":
failures.append(f"Routing did not pass ({value('ROUTING')}).")
if value("BEHAVIOR_WANTED") == "true" and orchestrai_authorized and value("ORCHESTRAI") != "success":
failures.append(f"OrchestrAI infrastructure did not complete ({value('ORCHESTRAI')}).")
if (
value("BEHAVIOR_WANTED") == "true"
and orchestrai_authorized
and value("ORCHESTRAI") == "success"
and value("ORCHESTRAI_VERDICTS") != "success"
):
failures.append(f"One or more OrchestrAI skill verdicts did not pass ({value('ORCHESTRAI_VERDICTS')}).")
if value("BEHAVIOR_WANTED") == "true" and not orchestrai_authorized:
if is_pull_request and not trusted_head:
failures.append("Strix behavioral cases require a maintainer branch in amd/skills before hardware can be authorized.")
else:
failures.append("Strix behavioral cases require the run_behavioral label.")
if value("SCOPED_WANTED") == "true" and value("SCOPED") != "success":
failures.append(f"Scoped behavioral cases did not pass ({value('SCOPED')}).")
if failures:
sys.exit("\n".join(failures))
print("All requested and authorized evals passed.")