diff --git a/docs/research/NEXT_100_PR_MAP.md b/docs/research/NEXT_100_PR_MAP.md index 74e9cd06..eccfec44 100644 --- a/docs/research/NEXT_100_PR_MAP.md +++ b/docs/research/NEXT_100_PR_MAP.md @@ -302,6 +302,6 @@ Track and publish structured comparisons between pipeline selections and cheap b |----|------|----------------|----------| | Y1 | Add cheap baseline comparison record schema (CBR-) (complete). — src/openamp_foundry/evidence/cheap_baseline_comparison_record.py: VALID_CBR_VERDICTS (4: pipeline_superior/tied/baseline_superior/insufficient_data), VALID_BASELINE_METHODS (5: charge_only_rank/length_only_rank/random_selection/charge_length_combined/hydrophobicity_only_rank), VALID_CBR_METRICS (4: auroc/hit_rate/top_k_precision/ndcg), SUPERIORITY_THRESHOLD=0.05, MIN_SAMPLE_SIZE=5; metric_delta auto-computed; verdict auto-derived; dry_lab_only=True; 62 tests. | Structured record: pipeline metric vs charge-only/random/length-only baseline; pre-registered threshold; verdict (pipeline_superior/tied/baseline_superior/insufficient_data). Forces every performance claim to cite the baseline it beat. | C | | Y2 | Add feature importance audit schema (FIA-) (complete). — src/openamp_foundry/evidence/feature_importance_audit.py: VALID_FIA_VERDICTS (5), VALID_FEATURE_IMPORTANCE_LEVELS (4), VALID_AUDIT_FEATURES (8), DOMINATION_THRESHOLD=0.80; importance_level auto-assigned; top_feature/charge_score/length_score auto-extracted; verdict: charge_dominated when charge_explains_fraction>=0.80; dry_lab_only=True; 50 tests. | Documents which features drove selections and whether charge/length alone explains the result; anti-cheap-explanation gate. | C | -| Y3 | Add selection diversity audit schema (SDA-). | Tracks sequence diversity of selected panel vs random draw; detects proximity-driven selection masquerading as discovery; required before any novelty claim. | C | +| Y3 | Add selection diversity audit schema (SDA-) (complete). — src/openamp_foundry/evidence/selection_diversity_audit.py: VALID_SDA_VERDICTS (4: diverse_panel/moderately_diverse/proximity_driven/insufficient_data), VALID_DIVERSITY_METRICS (4), DIVERSE_PANEL_THRESHOLD=0.10, PROXIMITY_DRIVEN_THRESHOLD=-0.05, MIN_PANEL_SIZE=3; diversity_delta auto-computed; verdict: diverse_panel (delta>=0.10), proximity_driven (delta<=-0.05); dry_lab_only=True; 46 tests. | Tracks sequence diversity of selected panel vs random draw; detects proximity-driven selection masquerading as discovery; required before any novelty claim. | C | | Y4 | Add pipeline maturity certificate schema (PMC-). | Aggregates CBR/FIA/SDA results into A/B/C/D maturity grade; anchors pre-registration; prevents retroactive interpretation of results. | C | | Y5 | Add Phase Y accountability gate (YAG-). | Top-level gate asserting CBR+FIA+SDA+PMC all present; verdict: accountability_verified/accountability_partial/accountability_not_established; closes Phase Y; no external pilot claim is credible without passing this gate. | C | diff --git a/src/openamp_foundry/evidence/selection_diversity_audit.py b/src/openamp_foundry/evidence/selection_diversity_audit.py new file mode 100644 index 00000000..3bd13ddb --- /dev/null +++ b/src/openamp_foundry/evidence/selection_diversity_audit.py @@ -0,0 +1,142 @@ +"""SDA- selection diversity audit schema. + +Tracks sequence diversity of selected candidate panel vs random draw from +the same pool. Detects proximity-driven selection masquerading as discovery. +Required before any novelty claim: if selected candidates cluster as tightly +as random, the selection is not adding diversity value. +""" + +from __future__ import annotations + +from dataclasses import dataclass + +VALID_SDA_VERDICTS: frozenset[str] = frozenset({ + "diverse_panel", + "moderately_diverse", + "proximity_driven", + "insufficient_data", +}) + +VALID_DIVERSITY_METRICS: frozenset[str] = frozenset({ + "mean_pairwise_identity", + "mean_pairwise_distance", + "clustering_coefficient", + "effective_sequence_count", +}) + +DIVERSE_PANEL_THRESHOLD: float = 0.10 +PROXIMITY_DRIVEN_THRESHOLD: float = -0.05 +MIN_PANEL_SIZE: int = 3 + + +@dataclass +class SelectionDiversityAudit: + sda_id: str + pipeline_version: str + diversity_metric: str + n_selected: int + panel_diversity_score: float + random_baseline_diversity_score: float + diversity_delta: float + sda_verdict: str + dry_lab_only: bool + limitations: list[str] + created_at: str + + +def validate_selection_diversity_audit(sda: SelectionDiversityAudit) -> None: + if not sda.sda_id.startswith("SDA-"): + raise ValueError(f"sda_id must start with 'SDA-': {sda.sda_id!r}") + if not sda.pipeline_version: + raise ValueError("pipeline_version must be non-empty") + if sda.diversity_metric not in VALID_DIVERSITY_METRICS: + raise ValueError( + f"diversity_metric {sda.diversity_metric!r} not in VALID_DIVERSITY_METRICS" + ) + if sda.n_selected < 0: + raise ValueError("n_selected must be non-negative") + expected_delta = round( + sda.panel_diversity_score - sda.random_baseline_diversity_score, 6 + ) + if abs(sda.diversity_delta - expected_delta) > 1e-5: + raise ValueError( + f"diversity_delta mismatch: expected {expected_delta}, got {sda.diversity_delta}" + ) + if sda.sda_verdict not in VALID_SDA_VERDICTS: + raise ValueError( + f"sda_verdict {sda.sda_verdict!r} not in VALID_SDA_VERDICTS" + ) + if not sda.dry_lab_only: + raise ValueError("dry_lab_only must be True") + if not sda.limitations: + raise ValueError("limitations must be non-empty") + if not sda.created_at: + raise ValueError("created_at must be non-empty") + + +def _compute_verdict( + n_selected: int, + delta: float, +) -> str: + if n_selected < MIN_PANEL_SIZE: + return "insufficient_data" + if delta >= DIVERSE_PANEL_THRESHOLD: + return "diverse_panel" + if delta <= PROXIMITY_DRIVEN_THRESHOLD: + return "proximity_driven" + return "moderately_diverse" + + +def build_selection_diversity_audit( + *, + sda_id: str, + pipeline_version: str, + diversity_metric: str, + n_selected: int, + panel_diversity_score: float, + random_baseline_diversity_score: float, + limitations: list[str], + created_at: str, +) -> SelectionDiversityAudit: + """Build a SelectionDiversityAudit. + + diversity_delta = panel_diversity_score - random_baseline_diversity_score (auto-computed). + For mean_pairwise_distance: higher score = more diverse. + For mean_pairwise_identity: lower score = more diverse (so delta>0 means less identity = more diversity). + Verdict: diverse_panel (delta>=0.10), proximity_driven (delta<=-0.05), else moderately_diverse. + """ + delta = round(panel_diversity_score - random_baseline_diversity_score, 6) + verdict = _compute_verdict(n_selected, delta) + sda = SelectionDiversityAudit( + sda_id=sda_id, + pipeline_version=pipeline_version, + diversity_metric=diversity_metric, + n_selected=n_selected, + panel_diversity_score=float(panel_diversity_score), + random_baseline_diversity_score=float(random_baseline_diversity_score), + diversity_delta=delta, + sda_verdict=verdict, + dry_lab_only=True, + limitations=limitations, + created_at=created_at, + ) + validate_selection_diversity_audit(sda) + return sda + + +def format_selection_diversity_audit(sda: SelectionDiversityAudit) -> str: + lines = [ + f"Selection Diversity Audit — {sda.sda_id}", + f"Pipeline: {sda.pipeline_version}", + f"Metric: {sda.diversity_metric} | Verdict: {sda.sda_verdict}", + f"Panel diversity: {sda.panel_diversity_score:.4f}", + f"Random baseline diversity: {sda.random_baseline_diversity_score:.4f}", + f"Delta: {sda.diversity_delta:+.4f} " + f"(diverse>=+{DIVERSE_PANEL_THRESHOLD}, " + f"proximity<={PROXIMITY_DRIVEN_THRESHOLD})", + f"Candidates selected: {sda.n_selected}", + f"Created: {sda.created_at}", + f"Limitations: {'; '.join(sda.limitations)}", + f"dry_lab_only: {sda.dry_lab_only}", + ] + return "\n".join(lines) diff --git a/tests/evidence/test_selection_diversity_audit.py b/tests/evidence/test_selection_diversity_audit.py new file mode 100644 index 00000000..b8069083 --- /dev/null +++ b/tests/evidence/test_selection_diversity_audit.py @@ -0,0 +1,262 @@ +"""Tests for SDA- selection diversity audit schema.""" + +import pytest +from openamp_foundry.evidence.selection_diversity_audit import ( + SelectionDiversityAudit, + VALID_SDA_VERDICTS, + VALID_DIVERSITY_METRICS, + DIVERSE_PANEL_THRESHOLD, + PROXIMITY_DRIVEN_THRESHOLD, + MIN_PANEL_SIZE, + build_selection_diversity_audit, + format_selection_diversity_audit, + validate_selection_diversity_audit, +) + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + + +def _build(**kwargs): + defaults = dict( + sda_id="SDA-001", + pipeline_version="v1.0", + diversity_metric="mean_pairwise_distance", + n_selected=10, + panel_diversity_score=0.70, + random_baseline_diversity_score=0.55, + limitations=["dry-lab only"], + created_at="2026-07-10", + ) + defaults.update(kwargs) + return build_selection_diversity_audit(**defaults) + + +# --------------------------------------------------------------------------- +# 1. Constants +# --------------------------------------------------------------------------- + + +def test_valid_sda_verdicts_is_frozenset(): + assert isinstance(VALID_SDA_VERDICTS, frozenset) + + +def test_valid_sda_verdicts_contains_diverse_panel(): + assert "diverse_panel" in VALID_SDA_VERDICTS + + +def test_valid_sda_verdicts_contains_moderately_diverse(): + assert "moderately_diverse" in VALID_SDA_VERDICTS + + +def test_valid_sda_verdicts_contains_proximity_driven(): + assert "proximity_driven" in VALID_SDA_VERDICTS + + +def test_valid_sda_verdicts_contains_insufficient_data(): + assert "insufficient_data" in VALID_SDA_VERDICTS + + +def test_valid_diversity_metrics_is_frozenset(): + assert isinstance(VALID_DIVERSITY_METRICS, frozenset) + + +def test_valid_diversity_metrics_contains_mean_pairwise_identity(): + assert "mean_pairwise_identity" in VALID_DIVERSITY_METRICS + + +def test_valid_diversity_metrics_contains_mean_pairwise_distance(): + assert "mean_pairwise_distance" in VALID_DIVERSITY_METRICS + + +def test_diverse_panel_threshold(): + assert DIVERSE_PANEL_THRESHOLD == 0.10 + + +def test_proximity_driven_threshold(): + assert PROXIMITY_DRIVEN_THRESHOLD == -0.05 + + +def test_min_panel_size(): + assert MIN_PANEL_SIZE == 3 + + +# --------------------------------------------------------------------------- +# 2. build – happy paths +# --------------------------------------------------------------------------- + + +def test_build_returns_selection_diversity_audit(): + assert isinstance(_build(), SelectionDiversityAudit) + + +def test_build_sda_id_stored(): + assert _build().sda_id == "SDA-001" + + +def test_build_pipeline_version_stored(): + assert _build().pipeline_version == "v1.0" + + +def test_build_dry_lab_only_true(): + assert _build().dry_lab_only is True + + +def test_build_diverse_panel_verdict(): + r = _build(panel_diversity_score=0.70, random_baseline_diversity_score=0.55) + assert r.sda_verdict == "diverse_panel" + + +def test_build_moderately_diverse_verdict(): + r = _build(panel_diversity_score=0.60, random_baseline_diversity_score=0.55) + assert r.sda_verdict == "moderately_diverse" + + +def test_build_proximity_driven_verdict(): + r = _build(panel_diversity_score=0.40, random_baseline_diversity_score=0.50) + assert r.sda_verdict == "proximity_driven" + + +def test_build_insufficient_data_small_panel(): + r = _build(n_selected=2) + assert r.sda_verdict == "insufficient_data" + + +def test_build_diversity_delta_auto_computed(): + r = _build(panel_diversity_score=0.70, random_baseline_diversity_score=0.55) + assert abs(r.diversity_delta - 0.15) < 1e-5 + + +def test_build_negative_delta(): + r = _build(panel_diversity_score=0.40, random_baseline_diversity_score=0.50) + assert r.diversity_delta < 0 + + +def test_build_diversity_metric_stored(): + assert _build().diversity_metric == "mean_pairwise_distance" + + +def test_build_n_selected_stored(): + assert _build().n_selected == 10 + + +def test_build_panel_score_stored(): + assert abs(_build().panel_diversity_score - 0.70) < 1e-6 + + +def test_build_baseline_score_stored(): + assert abs(_build().random_baseline_diversity_score - 0.55) < 1e-6 + + +def test_build_limitations_stored(): + assert _build().limitations == ["dry-lab only"] + + +def test_build_created_at_stored(): + assert _build().created_at == "2026-07-10" + + +def test_build_three_candidates_is_not_insufficient(): + r = _build(n_selected=3) + assert r.sda_verdict != "insufficient_data" + + +def test_build_mean_pairwise_identity_metric(): + r = _build(diversity_metric="mean_pairwise_identity") + assert r.diversity_metric == "mean_pairwise_identity" + + +def test_build_diverse_panel_at_exact_threshold(): + # delta = 0.10 → diverse_panel (>= comparison) + r = _build(panel_diversity_score=0.65, random_baseline_diversity_score=0.55) + assert r.sda_verdict == "diverse_panel" + + +# --------------------------------------------------------------------------- +# 3. validate – rejection cases +# --------------------------------------------------------------------------- + + +def test_validate_rejects_bad_sda_id_prefix(): + with pytest.raises(ValueError, match="SDA-"): + _build(sda_id="BAD-001") + + +def test_validate_rejects_empty_pipeline_version(): + with pytest.raises(ValueError): + _build(pipeline_version="") + + +def test_validate_rejects_invalid_diversity_metric(): + with pytest.raises(ValueError, match="diversity_metric"): + _build(diversity_metric="UNKNOWN") + + +def test_validate_rejects_negative_n_selected(): + with pytest.raises(ValueError, match="n_selected"): + _build(n_selected=-1) + + +def test_validate_rejects_diversity_delta_mismatch(): + sda = _build() + sda.diversity_delta = 99.0 + with pytest.raises(ValueError, match="diversity_delta"): + validate_selection_diversity_audit(sda) + + +def test_validate_rejects_invalid_sda_verdict(): + sda = _build() + sda.sda_verdict = "UNKNOWN" + with pytest.raises(ValueError, match="sda_verdict"): + validate_selection_diversity_audit(sda) + + +def test_validate_rejects_dry_lab_only_false(): + sda = _build() + sda.dry_lab_only = False + with pytest.raises(ValueError, match="dry_lab_only"): + validate_selection_diversity_audit(sda) + + +def test_validate_rejects_empty_limitations(): + with pytest.raises(ValueError, match="limitations"): + _build(limitations=[]) + + +def test_validate_rejects_empty_created_at(): + with pytest.raises(ValueError): + _build(created_at="") + + +# --------------------------------------------------------------------------- +# 4. format +# --------------------------------------------------------------------------- + + +def test_format_contains_sda_id(): + assert "SDA-001" in format_selection_diversity_audit(_build()) + + +def test_format_contains_pipeline_version(): + assert "v1.0" in format_selection_diversity_audit(_build()) + + +def test_format_contains_verdict(): + assert "diverse_panel" in format_selection_diversity_audit(_build()) + + +def test_format_contains_metric(): + assert "mean_pairwise_distance" in format_selection_diversity_audit(_build()) + + +def test_format_contains_limitations(): + assert "dry-lab only" in format_selection_diversity_audit(_build()) + + +def test_format_contains_dry_lab_only(): + assert "dry_lab_only: True" in format_selection_diversity_audit(_build()) + + +def test_format_is_string(): + assert isinstance(format_selection_diversity_audit(_build()), str) diff --git a/tests/test_test_count_regression.py b/tests/test_test_count_regression.py index 81689709..26022df3 100644 --- a/tests/test_test_count_regression.py +++ b/tests/test_test_count_regression.py @@ -4,7 +4,7 @@ import sys import math -BASELINE = 10610 +BASELINE = 10656 def test_test_count_regression():