From 878e13e9decb061b2d3e9c9ea089a055d180524b Mon Sep 17 00:00:00 2001 From: OpenCode Date: Fri, 10 Jul 2026 15:46:52 +0700 Subject: [PATCH] feat: Phase W W4 benchmark challenge registry schema (BCR-) -- aggregates NCH+CMC+SCH verdicts into hardness grade A-D; REQUIRED_CHALLENGE_TYPES (3), VALID_CHALLENGE_VERDICTS (4), ChallengeEntry helper; verdict mapping from raw challenge outputs; grade A=all pass; dry_lab_only=True; 64 tests --- docs/research/NEXT_100_PR_MAP.md | 2 +- .../evidence/benchmark_challenge_registry.py | 235 +++++++++++ .../test_benchmark_challenge_registry.py | 374 ++++++++++++++++++ 3 files changed, 610 insertions(+), 1 deletion(-) create mode 100644 src/openamp_foundry/evidence/benchmark_challenge_registry.py create mode 100644 tests/evidence/test_benchmark_challenge_registry.py diff --git a/docs/research/NEXT_100_PR_MAP.md b/docs/research/NEXT_100_PR_MAP.md index 70346f42..a53ecd01 100644 --- a/docs/research/NEXT_100_PR_MAP.md +++ b/docs/research/NEXT_100_PR_MAP.md @@ -279,5 +279,5 @@ Make it machine-verifiable that the pipeline produces novel candidates that beat | W1 | Add novelty challenge harness schema (NCH-) (complete). — src/openamp_foundry/evidence/novelty_challenge_harness.py: VALID_NCH_VERDICTS (4: novel_batch/mixed_novelty/near_neighbor_dominated/challenge_not_run), VALID_REFERENCE_DATABASES (6), NEAR_NEIGHBOR_IDENTITY_THRESHOLD=0.80, NOVEL_BATCH_CEILING=0.20, NEAR_NEIGHBOR_DOMINATED_FLOOR=0.60; NCHCandidateResult helper; build() auto-computes is_near_neighbor from identity vs threshold, fraction, verdict; dry_lab_only=True enforced; 63 tests in tests/evidence/test_novelty_challenge_harness.py. | Batch-level novelty challenge: documents the fraction of top candidates with ≥80% sequence identity to a known AMP in a reference database (APD3/DRAMP/etc.); blocks novel_batch claim when near-neighbor fraction exceeds 20%; prevents pipeline from advancing near-copies of known AMPs under a novelty label. | C | | W2 | Add charge-matched challenge schema (CMC-) (complete). — src/openamp_foundry/evidence/charge_matched_challenge.py: VALID_CMC_VERDICTS (4: gap_meaningful/gap_marginal/gap_absent/challenge_not_run), VALID_CHARGE_BASELINE_METHODS (4), MEANINGFUL_GAP_THRESHOLD=0.05, MARGINAL_GAP_LOWER=0.02; auroc_gap auto-computed; verdict auto-derived; dry_lab_only=True enforced; 60 tests in tests/evidence/test_charge_matched_challenge.py. | Formally documents the charge-matched challenge: compares pipeline AUROC vs a charge-only baseline on the same candidate set; verdict controlled vocabulary (gap_meaningful/gap_marginal/gap_absent/not_run); blocks performance claims when the charge-only baseline explains the gap. | C | | W3 | Add similarity challenge harness schema (SCH-) (complete). — src/openamp_foundry/evidence/similarity_challenge_harness.py: VALID_SCH_VERDICTS (4: selection_adds_value/marginal_improvement/proximity_driven/challenge_not_run), VALID_SIMILARITY_METRICS (4), SELECTION_VALUE_GAP_THRESHOLD=0.10, MARGINAL_IMPROVEMENT_LOWER=0.03; SimilarityGroupStats helper; similarity_gap auto-computed; dry_lab_only=True enforced; 60 tests in tests/evidence/test_similarity_challenge_harness.py. | Documents whether pipeline-selected candidates are systematically more similar to known AMPs than random selection from the sequence space; flags selection bias from similarity clustering; prevents "novel panel" claim when selection is proximity-driven. | C | -| W4 | Add benchmark challenge registry schema (BCR-). | Machine-readable registry of which benchmark challenges (NCH/CMC/SCH) have been run and passed for a given pipeline version; aggregates challenge verdicts; overall hardness grade (A: all passed, B: most passed, C: some passed, D: none passed). | C | +| W4 | Add benchmark challenge registry schema (BCR-) (complete). — src/openamp_foundry/evidence/benchmark_challenge_registry.py: REQUIRED_CHALLENGE_TYPES (3: NCH/CMC/SCH), VALID_CHALLENGE_VERDICTS (4: pass/marginal/fail/not_run), VALID_BCR_HARDNESS_GRADES (A-D); ChallengeEntry helper; verdict mapping (novel_batch→pass, mixed_novelty→marginal, gap_meaningful→pass, etc.); grade A=all pass, B=all pass+marginal, C=some pass/marginal, D=all fail/not_run; dry_lab_only=True; 64 tests in tests/evidence/test_benchmark_challenge_registry.py. | Machine-readable registry of which benchmark challenges (NCH/CMC/SCH) have been run and passed for a given pipeline version; aggregates challenge verdicts; overall hardness grade (A: all passed, B: most passed, C: some passed, D: none passed). | C | | W5 | Add Phase W benchmark gate (WBG-). | Top-level gate asserting NCH + CMC + SCH + BCR all present; overall verdict: hardened/partially_hardened/not_hardened; closes Phase W; no batch-level performance claim is credible without passing this gate. | C | diff --git a/src/openamp_foundry/evidence/benchmark_challenge_registry.py b/src/openamp_foundry/evidence/benchmark_challenge_registry.py new file mode 100644 index 00000000..55c43112 --- /dev/null +++ b/src/openamp_foundry/evidence/benchmark_challenge_registry.py @@ -0,0 +1,235 @@ +"""BCR- benchmark challenge registry schema. + +Machine-readable registry of which benchmark challenges (NCH, CMC, SCH) have +been run and passed for a given pipeline version and batch. Aggregates the +per-challenge verdicts into an overall hardness grade. No batch-level +performance claim is credible without a passing BCR- registry entry. +""" + +from __future__ import annotations + +from dataclasses import dataclass + +REQUIRED_CHALLENGE_TYPES: tuple[str, ...] = ("NCH", "CMC", "SCH") + +VALID_CHALLENGE_VERDICTS: frozenset[str] = frozenset({ + "pass", + "marginal", + "fail", + "not_run", +}) + +VALID_BCR_HARDNESS_GRADES: frozenset[str] = frozenset({ + "A", + "B", + "C", + "D", +}) + +_PASSING_NCH_VERDICTS: frozenset[str] = frozenset({ + "novel_batch", +}) +_MARGINAL_NCH_VERDICTS: frozenset[str] = frozenset({ + "mixed_novelty", +}) + +_PASSING_CMC_VERDICTS: frozenset[str] = frozenset({ + "gap_meaningful", +}) +_MARGINAL_CMC_VERDICTS: frozenset[str] = frozenset({ + "gap_marginal", +}) + +_PASSING_SCH_VERDICTS: frozenset[str] = frozenset({ + "selection_adds_value", +}) +_MARGINAL_SCH_VERDICTS: frozenset[str] = frozenset({ + "marginal_improvement", +}) + + +@dataclass +class ChallengeEntry: + challenge_type: str + artifact_id: str + raw_verdict: str + challenge_verdict: str + + +@dataclass +class BenchmarkChallengeRegistry: + bcr_id: str + batch_id: str + pipeline_version: str + challenge_entries: list[ChallengeEntry] + n_challenges_required: int + n_challenges_passed: int + n_challenges_marginal: int + n_challenges_failed: int + hardness_grade: str + dry_lab_only: bool + limitations: list[str] + created_at: str + + +def _map_verdict(challenge_type: str, raw_verdict: str) -> str: + if challenge_type == "NCH": + if raw_verdict in _PASSING_NCH_VERDICTS: + return "pass" + if raw_verdict in _MARGINAL_NCH_VERDICTS: + return "marginal" + if raw_verdict == "challenge_not_run": + return "not_run" + return "fail" + if challenge_type == "CMC": + if raw_verdict in _PASSING_CMC_VERDICTS: + return "pass" + if raw_verdict in _MARGINAL_CMC_VERDICTS: + return "marginal" + if raw_verdict == "challenge_not_run": + return "not_run" + return "fail" + if challenge_type == "SCH": + if raw_verdict in _PASSING_SCH_VERDICTS: + return "pass" + if raw_verdict in _MARGINAL_SCH_VERDICTS: + return "marginal" + if raw_verdict == "challenge_not_run": + return "not_run" + return "fail" + return "not_run" + + +def _compute_hardness_grade( + n_pass: int, + n_marginal: int, + n_required: int, +) -> str: + if n_pass == n_required: + return "A" + if n_pass + n_marginal == n_required: + return "B" + if n_pass > 0 or n_marginal > 0: + return "C" + return "D" + + +def validate_benchmark_challenge_registry(bcr: BenchmarkChallengeRegistry) -> None: + if not bcr.bcr_id.startswith("BCR-"): + raise ValueError(f"bcr_id must start with 'BCR-': {bcr.bcr_id!r}") + if not bcr.batch_id: + raise ValueError("batch_id must be non-empty") + if not bcr.pipeline_version: + raise ValueError("pipeline_version must be non-empty") + seen_types: set[str] = set() + for entry in bcr.challenge_entries: + if entry.challenge_type not in REQUIRED_CHALLENGE_TYPES: + raise ValueError( + f"challenge_type {entry.challenge_type!r} not in REQUIRED_CHALLENGE_TYPES" + ) + if entry.challenge_type in seen_types: + raise ValueError( + f"duplicate challenge_type: {entry.challenge_type!r}" + ) + seen_types.add(entry.challenge_type) + if entry.challenge_verdict not in VALID_CHALLENGE_VERDICTS: + raise ValueError( + f"challenge_verdict {entry.challenge_verdict!r} not in VALID_CHALLENGE_VERDICTS" + ) + if bcr.n_challenges_required != len(REQUIRED_CHALLENGE_TYPES): + raise ValueError( + f"n_challenges_required must be {len(REQUIRED_CHALLENGE_TYPES)}" + ) + n_pass = sum(1 for e in bcr.challenge_entries if e.challenge_verdict == "pass") + n_marginal = sum(1 for e in bcr.challenge_entries if e.challenge_verdict == "marginal") + n_fail = sum( + 1 for e in bcr.challenge_entries + if e.challenge_verdict in ("fail", "not_run") + ) + if bcr.n_challenges_passed != n_pass: + raise ValueError("n_challenges_passed mismatch") + if bcr.n_challenges_marginal != n_marginal: + raise ValueError("n_challenges_marginal mismatch") + if bcr.n_challenges_failed != n_fail: + raise ValueError("n_challenges_failed mismatch") + if bcr.hardness_grade not in VALID_BCR_HARDNESS_GRADES: + raise ValueError( + f"hardness_grade {bcr.hardness_grade!r} not in VALID_BCR_HARDNESS_GRADES" + ) + if not bcr.dry_lab_only: + raise ValueError("dry_lab_only must be True") + if not bcr.limitations: + raise ValueError("limitations must be non-empty") + if not bcr.created_at: + raise ValueError("created_at must be non-empty") + + +def build_benchmark_challenge_registry( + *, + bcr_id: str, + batch_id: str, + pipeline_version: str, + nch_artifact_id: str = "", + nch_raw_verdict: str = "challenge_not_run", + cmc_artifact_id: str = "", + cmc_raw_verdict: str = "challenge_not_run", + sch_artifact_id: str = "", + sch_raw_verdict: str = "challenge_not_run", + limitations: list[str], + created_at: str, +) -> BenchmarkChallengeRegistry: + raw = { + "NCH": (nch_artifact_id, nch_raw_verdict), + "CMC": (cmc_artifact_id, cmc_raw_verdict), + "SCH": (sch_artifact_id, sch_raw_verdict), + } + entries = [ + ChallengeEntry( + challenge_type=ctype, + artifact_id=raw[ctype][0], + raw_verdict=raw[ctype][1], + challenge_verdict=_map_verdict(ctype, raw[ctype][1]), + ) + for ctype in REQUIRED_CHALLENGE_TYPES + ] + n_pass = sum(1 for e in entries if e.challenge_verdict == "pass") + n_marginal = sum(1 for e in entries if e.challenge_verdict == "marginal") + n_fail = sum(1 for e in entries if e.challenge_verdict in ("fail", "not_run")) + grade = _compute_hardness_grade(n_pass, n_marginal, len(REQUIRED_CHALLENGE_TYPES)) + bcr = BenchmarkChallengeRegistry( + bcr_id=bcr_id, + batch_id=batch_id, + pipeline_version=pipeline_version, + challenge_entries=entries, + n_challenges_required=len(REQUIRED_CHALLENGE_TYPES), + n_challenges_passed=n_pass, + n_challenges_marginal=n_marginal, + n_challenges_failed=n_fail, + hardness_grade=grade, + dry_lab_only=True, + limitations=limitations, + created_at=created_at, + ) + validate_benchmark_challenge_registry(bcr) + return bcr + + +def format_benchmark_challenge_registry(bcr: BenchmarkChallengeRegistry) -> str: + lines = [ + f"Benchmark Challenge Registry — {bcr.bcr_id}", + f"Batch: {bcr.batch_id} | Pipeline: {bcr.pipeline_version}", + f"Hardness grade: {bcr.hardness_grade} | " + f"Passed: {bcr.n_challenges_passed}/{bcr.n_challenges_required} " + f"Marginal: {bcr.n_challenges_marginal} Failed: {bcr.n_challenges_failed}", + "Challenges:", + ] + for entry in bcr.challenge_entries: + aid = entry.artifact_id if entry.artifact_id else "(none)" + lines.append( + f" {entry.challenge_type}: {entry.challenge_verdict.upper()}" + f" [{entry.raw_verdict}] {aid}" + ) + lines.append(f"Created: {bcr.created_at}") + lines.append(f"Limitations: {'; '.join(bcr.limitations)}") + lines.append(f"dry_lab_only: {bcr.dry_lab_only}") + return "\n".join(lines) diff --git a/tests/evidence/test_benchmark_challenge_registry.py b/tests/evidence/test_benchmark_challenge_registry.py new file mode 100644 index 00000000..5b894154 --- /dev/null +++ b/tests/evidence/test_benchmark_challenge_registry.py @@ -0,0 +1,374 @@ +"""Tests for BCR- benchmark challenge registry schema.""" + +import pytest +from openamp_foundry.evidence.benchmark_challenge_registry import ( + BenchmarkChallengeRegistry, + ChallengeEntry, + REQUIRED_CHALLENGE_TYPES, + VALID_CHALLENGE_VERDICTS, + VALID_BCR_HARDNESS_GRADES, + build_benchmark_challenge_registry, + format_benchmark_challenge_registry, + validate_benchmark_challenge_registry, +) + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + + +def _build(**kwargs): + defaults = dict( + bcr_id="BCR-001", + batch_id="BATCH-01", + pipeline_version="v1.0", + nch_artifact_id="NCH-001", + nch_raw_verdict="novel_batch", + cmc_artifact_id="CMC-001", + cmc_raw_verdict="gap_meaningful", + sch_artifact_id="SCH-001", + sch_raw_verdict="selection_adds_value", + limitations=["dry-lab only"], + created_at="2026-07-10", + ) + defaults.update(kwargs) + return build_benchmark_challenge_registry(**defaults) + + +# --------------------------------------------------------------------------- +# 1. Constants +# --------------------------------------------------------------------------- + + +def test_required_challenge_types_is_tuple(): + assert isinstance(REQUIRED_CHALLENGE_TYPES, tuple) + + +def test_required_challenge_types_contains_nch(): + assert "NCH" in REQUIRED_CHALLENGE_TYPES + + +def test_required_challenge_types_contains_cmc(): + assert "CMC" in REQUIRED_CHALLENGE_TYPES + + +def test_required_challenge_types_contains_sch(): + assert "SCH" in REQUIRED_CHALLENGE_TYPES + + +def test_required_challenge_types_count(): + assert len(REQUIRED_CHALLENGE_TYPES) == 3 + + +def test_valid_challenge_verdicts_is_frozenset(): + assert isinstance(VALID_CHALLENGE_VERDICTS, frozenset) + + +def test_valid_challenge_verdicts_contains_pass(): + assert "pass" in VALID_CHALLENGE_VERDICTS + + +def test_valid_challenge_verdicts_contains_marginal(): + assert "marginal" in VALID_CHALLENGE_VERDICTS + + +def test_valid_challenge_verdicts_contains_fail(): + assert "fail" in VALID_CHALLENGE_VERDICTS + + +def test_valid_challenge_verdicts_contains_not_run(): + assert "not_run" in VALID_CHALLENGE_VERDICTS + + +def test_valid_bcr_hardness_grades_is_frozenset(): + assert isinstance(VALID_BCR_HARDNESS_GRADES, frozenset) + + +def test_valid_bcr_hardness_grades_contains_a(): + assert "A" in VALID_BCR_HARDNESS_GRADES + + +def test_valid_bcr_hardness_grades_contains_b(): + assert "B" in VALID_BCR_HARDNESS_GRADES + + +def test_valid_bcr_hardness_grades_contains_c(): + assert "C" in VALID_BCR_HARDNESS_GRADES + + +def test_valid_bcr_hardness_grades_contains_d(): + assert "D" in VALID_BCR_HARDNESS_GRADES + + +# --------------------------------------------------------------------------- +# 2. build – happy paths +# --------------------------------------------------------------------------- + + +def test_build_returns_benchmark_challenge_registry(): + assert isinstance(_build(), BenchmarkChallengeRegistry) + + +def test_build_bcr_id_stored(): + assert _build().bcr_id == "BCR-001" + + +def test_build_batch_id_stored(): + assert _build().batch_id == "BATCH-01" + + +def test_build_pipeline_version_stored(): + assert _build().pipeline_version == "v1.0" + + +def test_build_dry_lab_only_true(): + assert _build().dry_lab_only is True + + +def test_build_n_challenges_required_is_3(): + assert _build().n_challenges_required == 3 + + +def test_build_all_pass_gives_grade_a(): + assert _build().hardness_grade == "A" + + +def test_build_all_pass_n_passed_3(): + assert _build().n_challenges_passed == 3 + + +def test_build_all_pass_n_marginal_0(): + assert _build().n_challenges_marginal == 0 + + +def test_build_all_pass_n_failed_0(): + assert _build().n_challenges_failed == 0 + + +def test_build_all_marginal_gives_grade_b(): + r = _build( + nch_raw_verdict="mixed_novelty", + cmc_raw_verdict="gap_marginal", + sch_raw_verdict="marginal_improvement", + ) + assert r.hardness_grade == "B" + + +def test_build_all_marginal_n_marginal_3(): + r = _build( + nch_raw_verdict="mixed_novelty", + cmc_raw_verdict="gap_marginal", + sch_raw_verdict="marginal_improvement", + ) + assert r.n_challenges_marginal == 3 + + +def test_build_mixed_pass_and_marginal_gives_grade_b(): + r = _build( + nch_raw_verdict="novel_batch", + cmc_raw_verdict="gap_marginal", + sch_raw_verdict="marginal_improvement", + ) + assert r.hardness_grade == "B" + + +def test_build_one_fail_gives_grade_c(): + r = _build( + nch_raw_verdict="near_neighbor_dominated", + cmc_raw_verdict="gap_meaningful", + sch_raw_verdict="selection_adds_value", + ) + assert r.hardness_grade == "C" + + +def test_build_all_fail_gives_grade_d(): + r = _build( + nch_raw_verdict="near_neighbor_dominated", + cmc_raw_verdict="gap_absent", + sch_raw_verdict="proximity_driven", + ) + assert r.hardness_grade == "D" + + +def test_build_not_run_verdicts_give_grade_d(): + r = _build() + r2 = _build( + nch_raw_verdict="challenge_not_run", + cmc_raw_verdict="challenge_not_run", + sch_raw_verdict="challenge_not_run", + ) + assert r2.hardness_grade == "D" + + +def test_build_challenge_entries_length(): + assert len(_build().challenge_entries) == 3 + + +def test_build_challenge_entries_are_challenge_entry(): + for e in _build().challenge_entries: + assert isinstance(e, ChallengeEntry) + + +def test_build_nch_entry_challenge_verdict_pass(): + entries = {e.challenge_type: e for e in _build().challenge_entries} + assert entries["NCH"].challenge_verdict == "pass" + + +def test_build_cmc_entry_challenge_verdict_pass(): + entries = {e.challenge_type: e for e in _build().challenge_entries} + assert entries["CMC"].challenge_verdict == "pass" + + +def test_build_sch_entry_challenge_verdict_pass(): + entries = {e.challenge_type: e for e in _build().challenge_entries} + assert entries["SCH"].challenge_verdict == "pass" + + +def test_build_nch_artifact_id_stored(): + entries = {e.challenge_type: e for e in _build().challenge_entries} + assert entries["NCH"].artifact_id == "NCH-001" + + +def test_build_cmc_artifact_id_stored(): + entries = {e.challenge_type: e for e in _build().challenge_entries} + assert entries["CMC"].artifact_id == "CMC-001" + + +def test_build_sch_artifact_id_stored(): + entries = {e.challenge_type: e for e in _build().challenge_entries} + assert entries["SCH"].artifact_id == "SCH-001" + + +def test_build_nch_mixed_novelty_maps_to_marginal(): + r = _build(nch_raw_verdict="mixed_novelty") + entries = {e.challenge_type: e for e in r.challenge_entries} + assert entries["NCH"].challenge_verdict == "marginal" + + +def test_build_cmc_gap_marginal_maps_to_marginal(): + r = _build(cmc_raw_verdict="gap_marginal") + entries = {e.challenge_type: e for e in r.challenge_entries} + assert entries["CMC"].challenge_verdict == "marginal" + + +def test_build_sch_marginal_improvement_maps_to_marginal(): + r = _build(sch_raw_verdict="marginal_improvement") + entries = {e.challenge_type: e for e in r.challenge_entries} + assert entries["SCH"].challenge_verdict == "marginal" + + +def test_build_nch_nn_dominated_maps_to_fail(): + r = _build(nch_raw_verdict="near_neighbor_dominated") + entries = {e.challenge_type: e for e in r.challenge_entries} + assert entries["NCH"].challenge_verdict == "fail" + + +def test_build_limitations_stored(): + assert _build().limitations == ["dry-lab only"] + + +def test_build_created_at_stored(): + assert _build().created_at == "2026-07-10" + + +# --------------------------------------------------------------------------- +# 3. validate – rejection cases +# --------------------------------------------------------------------------- + + +def test_validate_rejects_bad_bcr_id_prefix(): + with pytest.raises(ValueError, match="BCR-"): + _build(bcr_id="BAD-001") + + +def test_validate_rejects_empty_batch_id(): + with pytest.raises(ValueError): + _build(batch_id="") + + +def test_validate_rejects_empty_pipeline_version(): + with pytest.raises(ValueError): + _build(pipeline_version="") + + +def test_validate_rejects_invalid_hardness_grade(): + bcr = _build() + bcr.hardness_grade = "X" + with pytest.raises(ValueError, match="hardness_grade"): + validate_benchmark_challenge_registry(bcr) + + +def test_validate_rejects_n_challenges_required_mismatch(): + bcr = _build() + bcr.n_challenges_required = 5 + with pytest.raises(ValueError, match="n_challenges_required"): + validate_benchmark_challenge_registry(bcr) + + +def test_validate_rejects_n_challenges_passed_mismatch(): + bcr = _build() + bcr.n_challenges_passed = 99 + with pytest.raises(ValueError, match="n_challenges_passed"): + validate_benchmark_challenge_registry(bcr) + + +def test_validate_rejects_dry_lab_only_false(): + bcr = _build() + bcr.dry_lab_only = False + with pytest.raises(ValueError, match="dry_lab_only"): + validate_benchmark_challenge_registry(bcr) + + +def test_validate_rejects_empty_limitations(): + with pytest.raises(ValueError, match="limitations"): + _build(limitations=[]) + + +def test_validate_rejects_empty_created_at(): + with pytest.raises(ValueError): + _build(created_at="") + + +# --------------------------------------------------------------------------- +# 4. format +# --------------------------------------------------------------------------- + + +def test_format_contains_bcr_id(): + assert "BCR-001" in format_benchmark_challenge_registry(_build()) + + +def test_format_contains_batch_id(): + assert "BATCH-01" in format_benchmark_challenge_registry(_build()) + + +def test_format_contains_hardness_grade(): + assert "A" in format_benchmark_challenge_registry(_build()) + + +def test_format_contains_nch(): + assert "NCH" in format_benchmark_challenge_registry(_build()) + + +def test_format_contains_cmc(): + assert "CMC" in format_benchmark_challenge_registry(_build()) + + +def test_format_contains_sch(): + assert "SCH" in format_benchmark_challenge_registry(_build()) + + +def test_format_contains_pass_verdict(): + assert "PASS" in format_benchmark_challenge_registry(_build()) + + +def test_format_contains_limitations(): + assert "dry-lab only" in format_benchmark_challenge_registry(_build()) + + +def test_format_contains_dry_lab_only(): + assert "dry_lab_only: True" in format_benchmark_challenge_registry(_build()) + + +def test_format_is_string(): + assert isinstance(format_benchmark_challenge_registry(_build()), str)