diff --git a/src/openamp_foundry/evidence/evidence_completeness_index.py b/src/openamp_foundry/evidence/evidence_completeness_index.py new file mode 100644 index 00000000..aff82fb9 --- /dev/null +++ b/src/openamp_foundry/evidence/evidence_completeness_index.py @@ -0,0 +1,179 @@ +"""ECI- evidence completeness index schema. + +Top-level index of which schemas exist for each batch and family across all +phases. Aggregates PCC + CBA2 + BEG + SAT into a single completeness +snapshot. Closes Phase U. +""" + +from __future__ import annotations + +from dataclasses import dataclass + +AGGREGATED_SCHEMA_TYPES: frozenset[str] = frozenset({"PCC", "CBA2", "BEG", "SAT"}) + +VALID_ECI_GRADES: frozenset[str] = frozenset({"A", "B", "C", "D"}) + +# Grade thresholds: A=100%, B>=75%, C>=50%, D<50% +_GRADE_THRESHOLDS = [ + ("A", 1.0), + ("B", 0.75), + ("C", 0.5), +] + + +def _compute_grade(fraction: float) -> str: + for grade, threshold in _GRADE_THRESHOLDS: + if fraction >= threshold: + return grade + return "D" + + +@dataclass +class BatchSchemaPresence: + batch_id: str + pcc_id: str + pcc_present: bool + cba2_id: str + cba2_present: bool + beg_id: str + beg_present: bool + sat_id: str + sat_present: bool + n_schemas_present: int + n_schemas_required: int + completeness_fraction: float + batch_grade: str + + +@dataclass +class EvidenceCompletenessIndex: + eci_id: str + pipeline_version: str + batch_entries: list[BatchSchemaPresence] + n_batches: int + n_schemas_required_per_batch: int + total_schema_slots: int + total_schemas_present: int + overall_completeness_fraction: float + completeness_grade: str + dry_lab_only: bool + limitations: list[str] + created_at: str + + +def validate_evidence_completeness_index(eci: EvidenceCompletenessIndex) -> None: + if not eci.eci_id.startswith("ECI-"): + raise ValueError(f"eci_id must start with 'ECI-': {eci.eci_id!r}") + if not eci.pipeline_version: + raise ValueError("pipeline_version must be non-empty") + if eci.completeness_grade not in VALID_ECI_GRADES: + raise ValueError( + f"completeness_grade {eci.completeness_grade!r} not in VALID_ECI_GRADES" + ) + if not eci.dry_lab_only: + raise ValueError("dry_lab_only must be True") + if not eci.limitations: + raise ValueError("limitations must be non-empty") + if not eci.created_at: + raise ValueError("created_at must be non-empty") + n_req = len(AGGREGATED_SCHEMA_TYPES) + if eci.n_schemas_required_per_batch != n_req: + raise ValueError( + f"n_schemas_required_per_batch must be {n_req}, got {eci.n_schemas_required_per_batch}" + ) + if eci.n_batches != len(eci.batch_entries): + raise ValueError("n_batches must equal len(batch_entries)") + expected_slots = eci.n_batches * n_req + if eci.total_schema_slots != expected_slots: + raise ValueError( + f"total_schema_slots must be {expected_slots}, got {eci.total_schema_slots}" + ) + total_present = sum(e.n_schemas_present for e in eci.batch_entries) + if eci.total_schemas_present != total_present: + raise ValueError("total_schemas_present mismatch") + + +def build_evidence_completeness_index( + *, + eci_id: str, + pipeline_version: str, + batch_dicts: list[dict], + limitations: list[str], + created_at: str, +) -> EvidenceCompletenessIndex: + """Build an evidence completeness index. + + batch_dicts: list of dicts with keys: + batch_id, pcc_id, cba2_id, beg_id, sat_id + Each id is non-empty string if present, "" if absent. + """ + n_req = len(AGGREGATED_SCHEMA_TYPES) + entries: list[BatchSchemaPresence] = [] + for d in batch_dicts: + pcc_present = bool(d.get("pcc_id", "")) + cba2_present = bool(d.get("cba2_id", "")) + beg_present = bool(d.get("beg_id", "")) + sat_present = bool(d.get("sat_id", "")) + n_present = sum([pcc_present, cba2_present, beg_present, sat_present]) + fraction = n_present / n_req + entries.append(BatchSchemaPresence( + batch_id=d["batch_id"], + pcc_id=d.get("pcc_id", ""), + pcc_present=pcc_present, + cba2_id=d.get("cba2_id", ""), + cba2_present=cba2_present, + beg_id=d.get("beg_id", ""), + beg_present=beg_present, + sat_id=d.get("sat_id", ""), + sat_present=sat_present, + n_schemas_present=n_present, + n_schemas_required=n_req, + completeness_fraction=fraction, + batch_grade=_compute_grade(fraction), + )) + + n_batches = len(entries) + total_slots = n_batches * n_req + total_present = sum(e.n_schemas_present for e in entries) + overall_fraction = total_present / total_slots if total_slots > 0 else 0.0 + grade = _compute_grade(overall_fraction) + + eci = EvidenceCompletenessIndex( + eci_id=eci_id, + pipeline_version=pipeline_version, + batch_entries=entries, + n_batches=n_batches, + n_schemas_required_per_batch=n_req, + total_schema_slots=total_slots, + total_schemas_present=total_present, + overall_completeness_fraction=overall_fraction, + completeness_grade=grade, + dry_lab_only=True, + limitations=limitations, + created_at=created_at, + ) + validate_evidence_completeness_index(eci) + return eci + + +def format_evidence_completeness_index(eci: EvidenceCompletenessIndex) -> str: + lines = [ + f"Evidence Completeness Index — {eci.eci_id}", + f"Pipeline: {eci.pipeline_version}", + f"Grade: {eci.completeness_grade} | Fraction: {eci.overall_completeness_fraction:.2%}", + f"Batches: {eci.n_batches} | Schemas/batch: {eci.n_schemas_required_per_batch}", + f"Total slots: {eci.total_schema_slots} | Present: {eci.total_schemas_present}", + ] + if eci.batch_entries: + lines.append("Batch summary:") + for entry in eci.batch_entries: + lines.append( + f" {entry.batch_id}: grade={entry.batch_grade} " + f"({entry.n_schemas_present}/{entry.n_schemas_required}) " + f"PCC={entry.pcc_present} CBA2={entry.cba2_present} " + f"BEG={entry.beg_present} SAT={entry.sat_present}" + ) + lines.append(f"Created: {eci.created_at}") + lines.append(f"Limitations: {'; '.join(eci.limitations)}") + lines.append(f"dry_lab_only: {eci.dry_lab_only}") + return "\n".join(lines) diff --git a/tests/evidence/test_evidence_completeness_index.py b/tests/evidence/test_evidence_completeness_index.py new file mode 100644 index 00000000..fee99244 --- /dev/null +++ b/tests/evidence/test_evidence_completeness_index.py @@ -0,0 +1,282 @@ +"""Tests for ECI- evidence completeness index schema.""" + +import pytest +from openamp_foundry.evidence.evidence_completeness_index import ( + EvidenceCompletenessIndex, + BatchSchemaPresence, + AGGREGATED_SCHEMA_TYPES, + VALID_ECI_GRADES, + build_evidence_completeness_index, + format_evidence_completeness_index, + validate_evidence_completeness_index, +) + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + +_BATCH_ALL_PRESENT = { + "batch_id": "B1", + "pcc_id": "P1", + "cba2_id": "C1", + "beg_id": "G1", + "sat_id": "S1", +} +_BATCH_PARTIAL = { + "batch_id": "B2", + "pcc_id": "P2", + "cba2_id": "", + "beg_id": "G2", + "sat_id": "", +} +_BATCH_EMPTY = { + "batch_id": "B3", + "pcc_id": "", + "cba2_id": "", + "beg_id": "", + "sat_id": "", +} + + +def _build(**kwargs): + defaults = dict( + eci_id="ECI-001", + pipeline_version="v1.0", + batch_dicts=[_BATCH_ALL_PRESENT, _BATCH_PARTIAL], + limitations=["dry-lab only"], + created_at="2026-07-10", + ) + defaults.update(kwargs) + return build_evidence_completeness_index(**defaults) + + +# --------------------------------------------------------------------------- +# 1. Constants +# --------------------------------------------------------------------------- + + +def test_aggregated_schema_types_is_frozenset(): + assert isinstance(AGGREGATED_SCHEMA_TYPES, frozenset) + + +def test_aggregated_schema_types_contains_pcc(): + assert "PCC" in AGGREGATED_SCHEMA_TYPES + + +def test_aggregated_schema_types_contains_cba2(): + assert "CBA2" in AGGREGATED_SCHEMA_TYPES + + +def test_aggregated_schema_types_contains_beg(): + assert "BEG" in AGGREGATED_SCHEMA_TYPES + + +def test_aggregated_schema_types_contains_sat(): + assert "SAT" in AGGREGATED_SCHEMA_TYPES + + +def test_aggregated_schema_types_count(): + assert len(AGGREGATED_SCHEMA_TYPES) == 4 + + +def test_valid_eci_grades_is_frozenset(): + assert isinstance(VALID_ECI_GRADES, frozenset) + + +def test_valid_eci_grades_contains_a(): + assert "A" in VALID_ECI_GRADES + + +def test_valid_eci_grades_contains_b(): + assert "B" in VALID_ECI_GRADES + + +def test_valid_eci_grades_contains_c(): + assert "C" in VALID_ECI_GRADES + + +def test_valid_eci_grades_contains_d(): + assert "D" in VALID_ECI_GRADES + + +# --------------------------------------------------------------------------- +# 2. build – happy paths +# --------------------------------------------------------------------------- + + +def test_build_returns_evidence_completeness_index(): + assert isinstance(_build(), EvidenceCompletenessIndex) + + +def test_build_eci_id_stored(): + assert _build().eci_id == "ECI-001" + + +def test_build_pipeline_version_stored(): + assert _build().pipeline_version == "v1.0" + + +def test_build_dry_lab_only_true(): + assert _build().dry_lab_only is True + + +def test_build_n_batches_matches_input(): + r = _build() + assert r.n_batches == 2 + + +def test_build_n_schemas_required_per_batch_is_4(): + assert _build().n_schemas_required_per_batch == 4 + + +def test_build_total_schema_slots(): + r = _build() + assert r.total_schema_slots == r.n_batches * 4 + + +def test_build_total_schemas_present(): + r = _build() + assert r.total_schemas_present == 4 + 2 # B1=4, B2=2 + + +def test_build_grade_a_when_all_present(): + r = _build(batch_dicts=[_BATCH_ALL_PRESENT]) + assert r.completeness_grade == "A" + + +def test_build_grade_d_when_all_empty(): + r = _build(batch_dicts=[_BATCH_EMPTY]) + assert r.completeness_grade == "D" + + +def test_build_grade_c_when_half_present(): + r = _build(batch_dicts=[_BATCH_ALL_PRESENT, _BATCH_EMPTY]) + assert r.completeness_grade == "C" + + +def test_build_overall_fraction_1_when_all(): + r = _build(batch_dicts=[_BATCH_ALL_PRESENT]) + assert r.overall_completeness_fraction == 1.0 + + +def test_build_overall_fraction_0_when_empty(): + r = _build(batch_dicts=[_BATCH_EMPTY]) + assert r.overall_completeness_fraction == 0.0 + + +def test_build_empty_batch_list_fraction_zero(): + r = _build(batch_dicts=[]) + assert r.overall_completeness_fraction == 0.0 + + +def test_build_empty_batch_list_grade_d(): + r = _build(batch_dicts=[]) + assert r.completeness_grade == "D" + + +def test_build_batch_entries_are_batch_schema_presence(): + r = _build() + for entry in r.batch_entries: + assert isinstance(entry, BatchSchemaPresence) + + +def test_build_batch_pcc_present_when_id_given(): + r = _build(batch_dicts=[_BATCH_ALL_PRESENT]) + assert r.batch_entries[0].pcc_present is True + + +def test_build_batch_cba2_absent_when_empty_id(): + r = _build(batch_dicts=[_BATCH_PARTIAL]) + assert r.batch_entries[0].cba2_present is False + + +def test_build_batch_sat_absent_when_empty_id(): + r = _build(batch_dicts=[_BATCH_PARTIAL]) + assert r.batch_entries[0].sat_present is False + + +def test_build_batch_grade_a_for_full_batch(): + r = _build(batch_dicts=[_BATCH_ALL_PRESENT]) + assert r.batch_entries[0].batch_grade == "A" + + +def test_build_batch_grade_d_for_empty_batch(): + r = _build(batch_dicts=[_BATCH_EMPTY]) + assert r.batch_entries[0].batch_grade == "D" + + +def test_build_batch_n_schemas_present(): + r = _build(batch_dicts=[_BATCH_ALL_PRESENT]) + assert r.batch_entries[0].n_schemas_present == 4 + + +def test_build_limitations_stored(): + assert _build().limitations == ["dry-lab only"] + + +def test_build_created_at_stored(): + assert _build().created_at == "2026-07-10" + + +# --------------------------------------------------------------------------- +# 3. validate – rejection cases +# --------------------------------------------------------------------------- + + +def test_validate_rejects_bad_eci_id_prefix(): + with pytest.raises(ValueError, match="ECI-"): + _build(eci_id="BAD-001") + + +def test_validate_rejects_empty_pipeline_version(): + with pytest.raises(ValueError): + _build(pipeline_version="") + + +def test_validate_rejects_empty_limitations(): + with pytest.raises(ValueError, match="limitations"): + _build(limitations=[]) + + +def test_validate_rejects_empty_created_at(): + with pytest.raises(ValueError): + _build(created_at="") + + +# --------------------------------------------------------------------------- +# 4. format +# --------------------------------------------------------------------------- + + +def test_format_contains_eci_id(): + assert "ECI-001" in format_evidence_completeness_index(_build()) + + +def test_format_contains_pipeline_version(): + assert "v1.0" in format_evidence_completeness_index(_build()) + + +def test_format_contains_grade(): + r = _build(batch_dicts=[_BATCH_ALL_PRESENT]) + assert "Grade: A" in format_evidence_completeness_index(r) + + +def test_format_contains_batch_id(): + assert "B1" in format_evidence_completeness_index(_build()) + + +def test_format_contains_fraction(): + r = _build(batch_dicts=[_BATCH_ALL_PRESENT]) + assert "100.00%" in format_evidence_completeness_index(r) + + +def test_format_contains_limitations(): + assert "dry-lab only" in format_evidence_completeness_index(_build()) + + +def test_format_contains_dry_lab_only(): + assert "dry_lab_only: True" in format_evidence_completeness_index(_build()) + + +def test_format_is_string(): + assert isinstance(format_evidence_completeness_index(_build()), str) diff --git a/tests/test_test_count_regression.py b/tests/test_test_count_regression.py index d159cb82..24310a5b 100644 --- a/tests/test_test_count_regression.py +++ b/tests/test_test_count_regression.py @@ -4,7 +4,7 @@ import sys import math -BASELINE = 10080 +BASELINE = 10143 def test_test_count_regression():