evals #179
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| name: evals | |
| # Skillscope owns structure, selection, routing, and grading. OrchestrAI owns | |
| # acquisition and execution of the default Linux/Windows Strix Halo sessions. | |
| # The OrchestrAI controller receives only the selected skill/OS matrix and | |
| # publishes normalized, allow-listed verdicts back to GitHub Actions. | |
| on: | |
| pull_request: | |
| types: [opened, synchronize, reopened, labeled] | |
| paths: | |
| - "skills/**" | |
| - ".github/workflows/evals.yml" | |
| - ".github/orchestrai-config.json" | |
| - ".github/scripts/build_orchestrai_plan.py" | |
| - ".github/scripts/orchestrai_report.py" | |
| - ".github/scripts/orchestrai_run.py" | |
| - ".github/scripts/orchestrai_logs.py" | |
| - ".github/scripts/orchestrai_verdict.py" | |
| - ".github/scripts/test_orchestrai_evals.py" | |
| - "**/*.md" | |
| schedule: | |
| - cron: "0 12 * * 1" | |
| workflow_dispatch: | |
| inputs: | |
| mode: | |
| description: "Which grader to run." | |
| type: choice | |
| options: [both, routing, behavioral, references] | |
| default: both | |
| skills: | |
| description: "Comma-separated skill names (blank = every skill with a dataset)." | |
| required: false | |
| default: "" | |
| only: | |
| description: "Comma-separated case ids to run (routing only)." | |
| required: false | |
| default: "" | |
| min_accuracy: | |
| description: "Fail the routing run below this accuracy (0-1). 0 reports without gating." | |
| required: false | |
| default: "0" | |
| orchestrai_mode: | |
| description: "Use mock to validate the plan without allocating hardware." | |
| type: choice | |
| options: [live, mock] | |
| default: live | |
| orchestrai_os: | |
| description: "Which Strix operating systems to run." | |
| type: choice | |
| options: [both, Linux, Windows] | |
| default: both | |
| concurrency: | |
| # Unrelated label events are no-ops and use a unique group, so they cannot | |
| # cancel or replace an in-flight graded run for the same pull request. | |
| group: evals-${{ github.event.pull_request.number || github.ref }}-${{ github.event.action == 'labeled' && github.event.label.name != 'run_behavioral' && github.run_id || 'graded' }} | |
| cancel-in-progress: ${{ github.event.action != 'labeled' || github.event.label.name == 'run_behavioral' }} | |
| permissions: | |
| contents: read | |
| id-token: write | |
| env: | |
| SKILLSCOPE_REPOSITORY: amd/skillscope | |
| SKILLSCOPE_VERSION: v0.1.3 | |
| ROUTING_ROOM: local-ai-use,local-ai-app-integration,serving-llms-on-instinct,tracelens-analysis-orchestrator,hyperloom-workload-optimizer | |
| jobs: | |
| discover: | |
| name: ${{ github.event.action == 'labeled' && github.event.label.name != 'run_behavioral' && 'Ignore unrelated label (discover)' || 'Check dataset structure and select runs' }} | |
| if: github.event.action != 'labeled' || github.event.label.name == 'run_behavioral' | |
| runs-on: ubuntu-latest | |
| permissions: | |
| contents: read | |
| outputs: | |
| routing: ${{ steps.plan.outputs.routing }} | |
| extended: ${{ steps.plan.outputs.extended }} | |
| default: ${{ steps.plan.outputs.default }} | |
| default_any: ${{ steps.plan.outputs.default_any }} | |
| orchestrai: ${{ steps.plan.outputs.orchestrai }} | |
| scoped: ${{ steps.plan.outputs.scoped }} | |
| scoped_any: ${{ steps.plan.outputs.scoped_any }} | |
| skipped: ${{ steps.plan.outputs.skipped }} | |
| steps: | |
| - name: Check out repository | |
| uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6.1.0 | |
| with: | |
| fetch-depth: 0 | |
| - name: Check out Skillscope | |
| uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6.1.0 | |
| with: | |
| repository: ${{ env.SKILLSCOPE_REPOSITORY }} | |
| ref: ${{ env.SKILLSCOPE_VERSION }} | |
| path: .skillscope-action | |
| - uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0 | |
| with: | |
| python-version: "3.12" | |
| # Install the pinned checkout directly. The enterprise Actions policy | |
| # does not allow the third-party setup-uv action used by the upstream | |
| # Skillscope composite action. | |
| - name: Install Skillscope | |
| run: python -m pip install --disable-pip-version-check ./.skillscope-action | |
| - name: Check every skill's structure | |
| run: >- | |
| python -m skillscope --repo . structural | |
| --skills-dir 'skills/*' | |
| --docs '*.md,docs/**/*.md,walkthroughs/*.md,staging/**/*.md' | |
| --skill-files 'skill-card.md' | |
| --skill-sections 'Description,Owner,License' | |
| - name: Test OrchestrAI plan and trigger helpers | |
| run: python3 .github/scripts/test_orchestrai_evals.py | |
| - name: Work out how to select | |
| id: how | |
| shell: python | |
| env: | |
| EVENT: ${{ github.event_name }} | |
| SKILLS: ${{ inputs.skills }} | |
| LABELS: ${{ join(github.event.pull_request.labels.*.name, ',') }} | |
| BASE: ${{ github.event.pull_request.base.sha }} | |
| HEAD: ${{ github.event.pull_request.head.sha }} | |
| ROUTING_ROOM_VALUE: ${{ env.ROUTING_ROOM }} | |
| run: | | |
| import os | |
| import shlex | |
| if os.environ["EVENT"] == "pull_request" and os.environ.get("BASE"): | |
| args = [ | |
| "--since", os.environ["BASE"], os.environ.get("HEAD", ""), | |
| "--labels", os.environ.get("LABELS", ""), "--no-extended", | |
| ] | |
| else: | |
| skills = os.environ.get("SKILLS", "").strip() | |
| args = ["--names", skills] if skills else ["--all"] | |
| args += ["--ignore-gates", "--no-extended"] | |
| args += [ | |
| "--routing-room", os.environ["ROUTING_ROOM_VALUE"], | |
| "--infra-paths", ".github/workflows/evals.yml,.github/orchestrai-config.json,.github/scripts/build_orchestrai_plan.py,.github/scripts/orchestrai_run.py,.github/scripts/orchestrai_verdict.py", | |
| "--behavior-runner", '["orchestrai"]', | |
| "--behavior-os", "Linux,Windows", | |
| "--scoped-runner", '["self-hosted","Linux","X64"]', | |
| "--scoped-gate", "enable_mi_ci", | |
| "--scoped-environment", "behavioral-instinct", | |
| ] | |
| line = "args=" + " ".join(shlex.quote(arg) for arg in args) | |
| print(line) | |
| with open(os.environ["GITHUB_OUTPUT"], "a", encoding="utf-8") as output: | |
| output.write(line + "\n") | |
| - name: Select what to run | |
| id: select | |
| shell: python | |
| env: | |
| SELECT_ARGS: ${{ steps.how.outputs.args }} | |
| run: | | |
| import os | |
| import shlex | |
| import subprocess | |
| import sys | |
| command = [ | |
| sys.executable, "-m", "skillscope", "--repo", ".", "select", | |
| "--skills-dir", "skills/*", *shlex.split(os.environ["SELECT_ARGS"]), | |
| ] | |
| completed = subprocess.run(command, text=True, capture_output=True) | |
| sys.stderr.write(completed.stderr) | |
| print(completed.stdout, end="") | |
| if completed.returncode: | |
| raise SystemExit(completed.returncode) | |
| lines = [line for line in completed.stdout.splitlines() if line.strip()] | |
| with open(os.environ["GITHUB_OUTPUT"], "a", encoding="utf-8") as output: | |
| output.write("stdout=" + (lines[-1] if lines else "") + "\n") | |
| - name: Apply requested mode and emit plan | |
| id: plan | |
| shell: python | |
| env: | |
| PLAN: ${{ steps.select.outputs.stdout }} | |
| MODE: ${{ github.event_name == 'schedule' && 'references' || inputs.mode || 'both' }} | |
| EVENT: ${{ github.event_name }} | |
| ORCHESTRAI_OS: ${{ inputs.orchestrai_os || 'both' }} | |
| run: | | |
| import json | |
| import os | |
| plan = json.loads(os.environ["PLAN"]) | |
| mode = os.environ.get("MODE", "both") | |
| if mode in ("behavioral", "behavior", "references"): | |
| plan["routing"] = False | |
| if mode in ("routing", "references"): | |
| plan["default"] = [] | |
| plan["scoped"] = [] | |
| requested_os = os.environ.get("ORCHESTRAI_OS", "both") | |
| if os.environ.get("EVENT") == "workflow_dispatch" and requested_os != "both": | |
| plan["default"] = [ | |
| entry for entry in plan["default"] | |
| if entry.get("os") == requested_os | |
| ] | |
| # Keep the GitHub verdict matrix aligned with the plan builder's | |
| # one-session-per-skill/OS identity, even if selection ever emits a | |
| # duplicate entry. | |
| seen = set() | |
| unique_default = [] | |
| for entry in plan["default"]: | |
| identity = (entry.get("skill"), entry.get("os")) | |
| if identity not in seen: | |
| seen.add(identity) | |
| unique_default.append(entry) | |
| plan["default"] = unique_default | |
| orchestrai = [] | |
| for os_name in ("Linux", "Windows"): | |
| entries = [ | |
| entry for entry in unique_default if entry.get("os") == os_name | |
| ] | |
| if entries: | |
| orchestrai.append({ | |
| "os": os_name, | |
| "slug": os_name.lower(), | |
| "matrix_json": json.dumps(entries, separators=(",", ":")), | |
| }) | |
| lines = [ | |
| "routing=" + str(plan["routing"]).lower(), | |
| "extended=" + ("--extended" if plan["extended"] else "--no-extended"), | |
| "orchestrai=" + json.dumps(orchestrai, separators=(",", ":")), | |
| ] | |
| for key in ("default", "scoped", "skipped"): | |
| lines.append(key + "=" + json.dumps(plan[key], separators=(",", ":"))) | |
| for key in ("default", "scoped"): | |
| lines.append(key + "_any=" + str(bool(plan[key])).lower()) | |
| print("\n".join(lines)) | |
| with open(os.environ["GITHUB_OUTPUT"], "a", encoding="utf-8") as output: | |
| output.write("\n".join(lines) + "\n") | |
| external-references: | |
| name: ${{ github.event.action == 'labeled' && github.event.label.name != 'run_behavioral' && 'Ignore unrelated label (references)' || 'External references' }} | |
| if: github.event.action != 'labeled' || github.event.label.name == 'run_behavioral' | |
| runs-on: ubuntu-latest | |
| permissions: | |
| contents: read | |
| steps: | |
| - uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6.1.0 | |
| - name: Check out Skillscope | |
| uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6.1.0 | |
| with: | |
| repository: ${{ env.SKILLSCOPE_REPOSITORY }} | |
| ref: ${{ env.SKILLSCOPE_VERSION }} | |
| path: .skillscope-action | |
| - uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0 | |
| with: | |
| python-version: "3.12" | |
| - name: Install Skillscope | |
| run: python -m pip install --disable-pip-version-check ./.skillscope-action | |
| - name: Fetch every external reference | |
| run: >- | |
| python -m skillscope --repo . structural | |
| --skills-dir 'skills/*' | |
| --external | |
| --docs '*.md,docs/**/*.md,walkthroughs/*.md,staging/**/*.md' | |
| --exclude-url '^https://openai\.com/codex/?$,^https://www\.amd\.com/en/resources/product-security\.html$' | |
| routing: | |
| name: Routing | |
| needs: discover | |
| if: needs.discover.outputs.routing == 'true' | |
| # Routing exercises the hosted LLM gateway only. Use the control runner for | |
| # its network path, while keeping direct Strix access exclusive to | |
| # OrchestrAI-managed tests. | |
| runs-on: ${{ vars.ORCHESTRAI_CONTROL_RUNNER || 'ubuntu-latest' }} | |
| timeout-minutes: 40 | |
| steps: | |
| - uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6.1.0 | |
| - name: Check out Skillscope | |
| uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6.1.0 | |
| with: | |
| repository: ${{ env.SKILLSCOPE_REPOSITORY }} | |
| ref: ${{ env.SKILLSCOPE_VERSION }} | |
| path: .skillscope-action | |
| - uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0 | |
| with: | |
| python-version: "3.12" | |
| - name: Install Skillscope | |
| run: python -m pip install --disable-pip-version-check ./.skillscope-action | |
| - name: Resolve model credentials | |
| shell: python | |
| env: | |
| API_KEY: ${{ secrets.ORCHESTR_API_KEY }} | |
| API_BASE_URL: https://llm-api.amd.com/Anthropic | |
| API_CUSTOM_HEADERS: | | |
| Ocp-Apim-Subscription-Key: $API_KEY | |
| user: a1_ucicd | |
| SECRET_NAME: ORCHESTR_API_KEY | |
| run: | | |
| import subprocess | |
| import sys | |
| raise SystemExit(subprocess.run([sys.executable, ".skillscope-action/skillscope/credentials.py"]).returncode) | |
| - uses: actions/setup-node@249970729cb0ef3589644e2896645e5dc5ba9c38 # v6 | |
| with: | |
| node-version: "20" | |
| - name: Install Claude Code | |
| run: npm install -g @anthropic-ai/claude-code | |
| - name: Run routing evals | |
| env: | |
| MIN_ACCURACY: ${{ inputs.min_accuracy || '0' }} | |
| ONLY: ${{ inputs.only }} | |
| EXTENDED_FLAG: ${{ needs.discover.outputs.extended }} | |
| run: | | |
| args=( | |
| python -m skillscope --repo . routing | |
| --skills-dir 'skills/*' | |
| --routing-room "$ROUTING_ROOM" | |
| --min-accuracy "$MIN_ACCURACY" | |
| "$EXTENDED_FLAG" | |
| --output routing-report.json | |
| --keep-logs routing-logs | |
| ) | |
| if [[ -n "$ONLY" ]]; then args+=(--only "$ONLY"); fi | |
| "${args[@]}" | |
| - name: Upload routing report | |
| if: always() | |
| uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 | |
| with: | |
| name: routing-report | |
| path: | | |
| routing-report.json | |
| routing-logs/ | |
| if-no-files-found: warn | |
| orchestrai-behavioral: | |
| name: Behavioral (OrchestrAI Strix Halo — ${{ matrix.os }}) | |
| needs: discover | |
| if: >- | |
| needs.discover.outputs.default_any == 'true' && | |
| (github.event_name != 'pull_request' || | |
| (contains(github.event.pull_request.labels.*.name, 'run_behavioral') && | |
| github.event.pull_request.head.repo.full_name == github.repository)) | |
| runs-on: ${{ vars.ORCHESTRAI_CONTROL_RUNNER || 'ubuntu-latest' }} | |
| timeout-minutes: 240 | |
| strategy: | |
| fail-fast: false | |
| matrix: | |
| include: ${{ fromJSON(needs.discover.outputs.orchestrai) }} | |
| steps: | |
| - uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6.1.0 | |
| - uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0 | |
| with: | |
| python-version: "3.12" | |
| - name: Build OrchestrAI plan | |
| env: | |
| MATRIX_JSON: ${{ matrix.matrix_json }} | |
| TARGET_REF: ${{ github.event.pull_request.head.ref || github.ref_name }} | |
| # A PR must test the immutable commit GitHub associated with this | |
| # check run. Non-PR events use the selected ref's triggering commit. | |
| TARGET_SHA: ${{ github.event.pull_request.head.sha || github.sha }} | |
| EXTENDED_FLAG: ${{ needs.discover.outputs.extended }} | |
| # Fleet selectors are private infrastructure metadata. Store this | |
| # JSON as a secret so Actions masks the value in the step's automatic | |
| # environment preamble before our code starts. | |
| DEVICE_TAGS_JSON: ${{ secrets.ORCHESTRAI_DEVICE_TAGS }} | |
| # Same per-Ubuntu-release mapping used by Playbooks. Keep it as a | |
| # secret here so its URLs are masked in this public repository's logs. | |
| ORCHESTRAI_LINUX_DRIVER_SOURCES_JSON: ${{ secrets.ORCHESTRAI_LINUX_DRIVER_SOURCES_JSON }} | |
| SOURCE_RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} | |
| run: >- | |
| python3 .github/scripts/build_orchestrai_plan.py | |
| --matrix-json "$MATRIX_JSON" | |
| --device-tags-json "$DEVICE_TAGS_JSON" | |
| --repository "${{ github.repository }}" | |
| --ref "$TARGET_REF" | |
| --sha "$TARGET_SHA" | |
| --extended-flag="$EXTENDED_FLAG" | |
| --run-id "${{ github.run_id }}" | |
| --run-attempt "${{ github.run_attempt }}" | |
| --source-run-url "$SOURCE_RUN_URL" | |
| --output orchestrai-plan.json | |
| - name: Acquire Strix machines and run behavioral evals | |
| id: orchestrai | |
| env: | |
| PLAN_FILE: orchestrai-plan.json | |
| MODE: ${{ github.event_name == 'workflow_dispatch' && inputs.orchestrai_mode || 'live' }} | |
| POLL_TIMEOUT_SEC: "13800" | |
| # Match Playbooks' five-minute continuous-unreachable allowance. | |
| # A successful poll resets this budget. | |
| POLL_ERROR_TIMEOUT_SEC: "300" | |
| ORCHESTRAI_PORTAL_URL: ${{ secrets.ORCHESTRAI_PORTAL_URL }} | |
| ORCHESTRAI_USER: ${{ secrets.ORCHESTRAI_USER }} | |
| ORCHESTRAI_PASSWORD: ${{ secrets.ORCHESTRAI_PASSWORD }} | |
| ORCHESTRAI_SPACE: ${{ secrets.ORCHESTRAI_SPACE }} | |
| run: exec python3 .github/scripts/orchestrai_run.py | |
| - name: Upload OrchestrAI result manifest | |
| if: always() | |
| uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 | |
| with: | |
| name: orchestrai-run-results-${{ matrix.slug }} | |
| # The live controller snapshot can contain private pipeline topology | |
| # (for example, stage names). Verdict jobs need only the normalized, | |
| # allow-listed manifest, so keep the snapshot inside this ephemeral | |
| # job workspace and out of public workflow artifacts. | |
| path: orchestrai-results.json | |
| if-no-files-found: warn | |
| orchestrai-verdict: | |
| name: Behavioral (${{ matrix.skill }} on ${{ matrix.os }}) | |
| needs: [discover, orchestrai-behavioral] | |
| if: >- | |
| !cancelled() && | |
| needs.discover.result == 'success' && | |
| needs.discover.outputs.default_any == 'true' && | |
| (github.event_name != 'pull_request' || | |
| (contains(github.event.pull_request.labels.*.name, 'run_behavioral') && | |
| github.event.pull_request.head.repo.full_name == github.repository)) | |
| runs-on: ${{ vars.ORCHESTRAI_WAIT_RUNNER || 'ubuntu-latest' }} | |
| strategy: | |
| fail-fast: false | |
| matrix: | |
| include: ${{ fromJSON(needs.discover.outputs.default) }} | |
| steps: | |
| - uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6.1.0 | |
| - name: Download OrchestrAI results | |
| continue-on-error: true | |
| uses: actions/download-artifact@37930b1c2abaa49bbe596cd826c3c89aef350131 # v7.0.0 | |
| with: | |
| name: orchestrai-run-results-${{ matrix.os == 'Linux' && 'linux' || 'windows' }} | |
| path: .orchestrai-report | |
| - name: Resolve skill verdict | |
| id: verdict | |
| run: >- | |
| python3 .github/scripts/orchestrai_verdict.py | |
| --results .orchestrai-report/orchestrai-results.json | |
| --skill "${{ matrix.skill }}" | |
| --os "${{ matrix.os }}" | |
| --output-dir test-results | |
| - name: Upload skill result | |
| if: always() | |
| uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 | |
| with: | |
| name: test-results-${{ matrix.skill }}-${{ matrix.os }} | |
| path: test-results/ | |
| if-no-files-found: warn | |
| behavior-scoped: | |
| name: Behavioral (${{ matrix.skill }} on ${{ matrix.os }}, ${{ matrix.environment }}) | |
| needs: discover | |
| if: needs.discover.outputs.scoped_any == 'true' | |
| runs-on: ${{ fromJSON(matrix.runner) }} | |
| timeout-minutes: 45 | |
| permissions: | |
| contents: read | |
| id-token: write | |
| environment: | |
| name: ${{ matrix.environment }} | |
| strategy: | |
| fail-fast: false | |
| matrix: | |
| include: ${{ fromJSON(needs.discover.outputs.scoped) }} | |
| steps: | |
| - uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6.1.0 | |
| - name: Check out Skillscope | |
| uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6.1.0 | |
| with: | |
| repository: ${{ env.SKILLSCOPE_REPOSITORY }} | |
| ref: ${{ env.SKILLSCOPE_VERSION }} | |
| path: .skillscope-action | |
| - uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0 | |
| with: | |
| python-version: "3.12" | |
| - name: Install Skillscope | |
| run: python -m pip install --disable-pip-version-check ./.skillscope-action | |
| - name: Resolve model credentials | |
| shell: python | |
| env: | |
| API_KEY: "" | |
| API_BASE_URL: "" | |
| API_CUSTOM_HEADERS: "" | |
| SECRET_NAME: "" | |
| ENVIRONMENT: ${{ matrix.environment }} | |
| FEDERATION_RULE_ID: fdrl_014u39zsJmsex5SkL1gWg7qo | |
| FEDERATION_ORGANIZATION_ID: 9bdff596-1595-4140-8301-817e43ef8412 | |
| FEDERATION_SERVICE_ACCOUNT_ID: svac_016zcTqZ4XkFdjAJCag9jQC9 | |
| FEDERATION_WORKSPACE_ID: wrkspc_01EUPtU7mqFtoQowMAREPvoq | |
| run: | | |
| import subprocess | |
| import sys | |
| raise SystemExit(subprocess.run([sys.executable, ".skillscope-action/skillscope/credentials.py"]).returncode) | |
| - uses: actions/setup-node@249970729cb0ef3589644e2896645e5dc5ba9c38 # v6 | |
| with: | |
| node-version: "20" | |
| - name: Install Claude Code | |
| run: npm install -g @anthropic-ai/claude-code | |
| - name: Run behavioral cases | |
| env: | |
| SKILL: ${{ matrix.skill }} | |
| EXTENDED_FLAG: ${{ needs.discover.outputs.extended }} | |
| run: >- | |
| python -m skillscope --repo . behavioral | |
| --skills-dir 'skills/*' | |
| --skill "$SKILL" | |
| "$EXTENDED_FLAG" | |
| results: | |
| name: ${{ github.event.action == 'labeled' && github.event.label.name != 'run_behavioral' && 'Ignore unrelated label (results)' || 'evals / results' }} | |
| needs: [discover, routing, orchestrai-behavioral, orchestrai-verdict, behavior-scoped] | |
| if: >- | |
| always() && | |
| (github.event.action != 'labeled' || github.event.label.name == 'run_behavioral') | |
| runs-on: ubuntu-latest | |
| permissions: | |
| contents: read | |
| steps: | |
| - uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6.1.0 | |
| - uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0 | |
| with: | |
| python-version: "3.12" | |
| - name: Download per-skill OrchestrAI results | |
| if: >- | |
| needs.discover.outputs.default_any == 'true' && | |
| (github.event_name != 'pull_request' || | |
| (contains(github.event.pull_request.labels.*.name, 'run_behavioral') && | |
| github.event.pull_request.head.repo.full_name == github.repository)) | |
| continue-on-error: true | |
| uses: actions/download-artifact@37930b1c2abaa49bbe596cd826c3c89aef350131 # v7.0.0 | |
| with: | |
| pattern: test-results-* | |
| path: .orchestrai-results | |
| - name: Publish aggregate OrchestrAI report | |
| if: >- | |
| needs.discover.outputs.default_any == 'true' && | |
| (github.event_name != 'pull_request' || | |
| (contains(github.event.pull_request.labels.*.name, 'run_behavioral') && | |
| github.event.pull_request.head.repo.full_name == github.repository)) | |
| env: | |
| EXPECTED_JSON: ${{ needs.discover.outputs.default }} | |
| ORCHESTRAI_CONTROLLER_RESULT: ${{ needs.orchestrai-behavioral.result }} | |
| ORCHESTRAI_VERDICT_RESULT: ${{ needs.orchestrai-verdict.result }} | |
| run: >- | |
| python3 .github/scripts/orchestrai_report.py | |
| --artifacts .orchestrai-results | |
| --expected-json "$EXPECTED_JSON" | |
| --output orchestrai-report.md | |
| - name: Upload aggregate OrchestrAI report | |
| if: >- | |
| needs.discover.outputs.default_any == 'true' && | |
| (github.event_name != 'pull_request' || | |
| (contains(github.event.pull_request.labels.*.name, 'run_behavioral') && | |
| github.event.pull_request.head.repo.full_name == github.repository)) | |
| uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 | |
| with: | |
| name: orchestrai-report | |
| path: orchestrai-report.md | |
| if-no-files-found: warn | |
| - name: Verify eval results | |
| shell: python | |
| env: | |
| EVENT: ${{ github.event_name }} | |
| LABELS: ${{ join(github.event.pull_request.labels.*.name, ',') }} | |
| HEAD_REPOSITORY: ${{ github.event.pull_request.head.repo.full_name }} | |
| REPOSITORY: ${{ github.repository }} | |
| DISCOVER: ${{ needs.discover.result }} | |
| ROUTING: ${{ needs.routing.result }} | |
| ORCHESTRAI: ${{ needs.orchestrai-behavioral.result }} | |
| ORCHESTRAI_VERDICTS: ${{ needs.orchestrai-verdict.result }} | |
| SCOPED: ${{ needs.behavior-scoped.result }} | |
| ROUTING_WANTED: ${{ needs.discover.outputs.routing }} | |
| BEHAVIOR_WANTED: ${{ needs.discover.outputs.default_any }} | |
| SCOPED_WANTED: ${{ needs.discover.outputs.scoped_any }} | |
| SKIPPED: ${{ needs.discover.outputs.skipped }} | |
| run: | | |
| import os | |
| import sys | |
| value = lambda name: os.environ.get(name, "") | |
| labels = {label for label in value("LABELS").split(",") if label} | |
| is_pull_request = value("EVENT") == "pull_request" | |
| trusted_head = not is_pull_request or value("HEAD_REPOSITORY") == value("REPOSITORY") | |
| orchestrai_authorized = not is_pull_request or ( | |
| "run_behavioral" in labels and trusted_head | |
| ) | |
| print(f"discover: {value('DISCOVER')}") | |
| print(f"routing: {value('ROUTING')} (requested: {value('ROUTING_WANTED')})") | |
| print(f"orchestrai-behavioral:{value('ORCHESTRAI')} (requested: {value('BEHAVIOR_WANTED')}, authorized: {orchestrai_authorized})") | |
| print(f"orchestrai-verdicts: {value('ORCHESTRAI_VERDICTS')}") | |
| print(f"behavioral-scoped: {value('SCOPED')} (requested: {value('SCOPED_WANTED')})") | |
| print(f"held back: {value('SKIPPED')}") | |
| if value("DISCOVER") != "success": | |
| sys.exit(f"The discover job did not succeed ({value('DISCOVER')}).") | |
| skipped = value("SKIPPED").strip() | |
| if skipped and skipped != "[]": | |
| print(f"::warning::Scoped behavioral cases were held back by a missing gate: {skipped}") | |
| failures = [] | |
| if value("ROUTING_WANTED") == "true" and value("ROUTING") != "success": | |
| failures.append(f"Routing did not pass ({value('ROUTING')}).") | |
| if value("BEHAVIOR_WANTED") == "true" and orchestrai_authorized and value("ORCHESTRAI") != "success": | |
| failures.append(f"OrchestrAI infrastructure did not complete ({value('ORCHESTRAI')}).") | |
| if ( | |
| value("BEHAVIOR_WANTED") == "true" | |
| and orchestrai_authorized | |
| and value("ORCHESTRAI") == "success" | |
| and value("ORCHESTRAI_VERDICTS") != "success" | |
| ): | |
| failures.append(f"One or more OrchestrAI skill verdicts did not pass ({value('ORCHESTRAI_VERDICTS')}).") | |
| if value("BEHAVIOR_WANTED") == "true" and not orchestrai_authorized: | |
| if is_pull_request and not trusted_head: | |
| failures.append("Strix behavioral cases require a maintainer branch in amd/skills before hardware can be authorized.") | |
| else: | |
| failures.append("Strix behavioral cases require the run_behavioral label.") | |
| if value("SCOPED_WANTED") == "true" and value("SCOPED") != "success": | |
| failures.append(f"Scoped behavioral cases did not pass ({value('SCOPED')}).") | |
| if failures: | |
| sys.exit("\n".join(failures)) | |
| print("All requested and authorized evals passed.") |