Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion docs/research/NEXT_100_PR_MAP.md
Original file line number Diff line number Diff line change
Expand Up @@ -291,5 +291,5 @@ Track whether the pipeline actually improves across batches by capturing per-bat
| X1 | Add multi-batch learning record schema (MBL-) (complete). — src/openamp_foundry/evidence/multi_batch_learning_record.py: VALID_MBL_QUALITY_GRADES (5: A-D/N/A), VALID_BATCH_LEARNING_STATUSES (3), GRADE_A_HIT_RATE=0.40; hit_rate auto-computed from n_confirmed_hits/n_candidates_tested; grade auto-derived; no_wet_lab_data forces N/A grade; whr_ids list; dry_lab_only=True; 61 tests in tests/evidence/test_multi_batch_learning_record.py. | Per-batch snapshot of prediction quality (hit rate, AUROC, n_confirmed_hits) after wet-lab feedback; enables cross-batch comparison of whether the pipeline is learning; feeds into calibration improvement tracker. | C |
| X2 | Add calibration improvement tracker schema (CIT-) (complete). — src/openamp_foundry/evidence/calibration_improvement_tracker.py: VALID_CIT_TREND_DIRECTIONS (4: improving/stable/degrading/insufficient_data), VALID_CIT_SUMMARY_GRADES (5: A-D/N/A), MIN_BATCHES_FOR_TREND=2, IMPROVEMENT_THRESHOLD=0.05; BatchHitRateEntry helper; trend auto-computed from first→latest hit_rate delta; insufficient_data forces N/A grade; dry_lab_only=True; 49 tests. | Aggregates MBL records across batches; computes hit-rate trend direction (improving/stable/degrading/insufficient_data); minimum 2 batches required; flags when calibration is not producing measurable improvement. | C |
| X3 | Add learning progress report schema (LPR-) (complete). — src/openamp_foundry/evidence/learning_progress_report.py: VALID_LPR_VERDICTS (4: learning_confirmed/learning_inconclusive/no_learning_signal/insufficient_data), VALID_FEATURE_PREDICTIVITY (3: predictive/not_predictive/uncertain), VALID_FEATURE_CATEGORIES (8); FeatureLearningEntry helper; verdict auto-computed: learning_confirmed (n_pred>n_non), learning_inconclusive (equal+>0), no_learning_signal (n_pred<n_non), insufficient_data (no batches or no definitive features); dry_lab_only=True; 58 tests in tests/evidence/test_learning_progress_report.py. | Human-readable summary of what the pipeline has learned from all batches to date; references CIT- for trend data; includes which candidate features proved predictive vs not; links to calibration decision logs. | C |
| X4 | Add recalibration confidence certificate schema (RCC-). | Asserts with what confidence the current calibration weights are reliable based on cohort size, quality, and consistency across batches; A/B/C/D grade; prevents overconfident calibration claims. | C |
| X4 | Add recalibration confidence certificate schema (RCC-) (complete). — src/openamp_foundry/evidence/recalibration_confidence_certificate.py: VALID_RCC_GRADES (A-D), VALID_RCC_VERDICTS (4: high_confidence/moderate_confidence/low_confidence/insufficient_data), VALID_CONSISTENCY_RATINGS (4); BatchCohortEntry helper; grade A (≥4 batches+consistent), B (≥2 batches+consistent/moderately_consistent), D (insufficient_data); cross_batch_consistency auto-computed; total_cohort_size auto-summed; dry_lab_only=True; cit_id/lpr_id prefix-validated; 76 tests. | Asserts with what confidence the current calibration weights are reliable based on cohort size, quality, and consistency across batches; A/B/C/D grade; prevents overconfident calibration claims. | C |
| X5 | Add Phase X learning gate (XLG-). | Top-level gate asserting MBL + CIT + LPR + RCC all present; overall verdict: learning_verified/learning_in_progress/learning_not_started; closes Phase X; no calibration improvement claim is credible without passing this gate. | C |
232 changes: 232 additions & 0 deletions src/openamp_foundry/evidence/recalibration_confidence_certificate.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,232 @@
"""RCC- recalibration confidence certificate schema.

Asserts with what confidence the current calibration weights are reliable,
based on cohort size, quality consistency, and cross-batch agreement.
Prevents overconfident calibration claims when evidence is thin.
Gives a machine-checkable grade (A-D) that gates downstream claims.
"""

from __future__ import annotations

from dataclasses import dataclass

VALID_RCC_GRADES: frozenset[str] = frozenset({
"A",
"B",
"C",
"D",
})

VALID_RCC_VERDICTS: frozenset[str] = frozenset({
"high_confidence",
"moderate_confidence",
"low_confidence",
"insufficient_data",
})

VALID_CONSISTENCY_RATINGS: frozenset[str] = frozenset({
"consistent",
"moderately_consistent",
"inconsistent",
"unknown",
})

MIN_BATCHES_FOR_CONFIDENCE: int = 2
HIGH_CONFIDENCE_MIN_BATCHES: int = 4
MODERATE_CONFIDENCE_MIN_BATCHES: int = 2
MIN_COHORT_SIZE_PER_BATCH: int = 5

GRADE_A_MIN_BATCHES: int = 4
GRADE_B_MIN_BATCHES: int = 2
GRADE_C_MIN_BATCHES: int = 1


@dataclass
class BatchCohortEntry:
batch_id: str
mbl_id: str
cohort_size: int
quality_grade: str


@dataclass
class RecalibrationConfidenceCertificate:
rcc_id: str
pipeline_version: str
cit_id: str
lpr_id: str
batch_cohort_entries: list[BatchCohortEntry]
n_batches_assessed: int
total_cohort_size: int
cross_batch_consistency: str
rcc_grade: str
rcc_verdict: str
confidence_rationale: str
dry_lab_only: bool
limitations: list[str]
created_at: str


def validate_recalibration_confidence_certificate(
rcc: RecalibrationConfidenceCertificate,
) -> None:
if not rcc.rcc_id.startswith("RCC-"):
raise ValueError(f"rcc_id must start with 'RCC-': {rcc.rcc_id!r}")
if not rcc.pipeline_version:
raise ValueError("pipeline_version must be non-empty")
if not rcc.cit_id.startswith("CIT-"):
raise ValueError(f"cit_id must start with 'CIT-': {rcc.cit_id!r}")
if not rcc.lpr_id.startswith("LPR-"):
raise ValueError(f"lpr_id must start with 'LPR-': {rcc.lpr_id!r}")
for entry in rcc.batch_cohort_entries:
if entry.cohort_size < 0:
raise ValueError(
f"cohort_size must be non-negative for batch {entry.batch_id!r}"
)
if rcc.n_batches_assessed != len(rcc.batch_cohort_entries):
raise ValueError("n_batches_assessed must equal len(batch_cohort_entries)")
expected_total = sum(e.cohort_size for e in rcc.batch_cohort_entries)
if rcc.total_cohort_size != expected_total:
raise ValueError(
f"total_cohort_size mismatch: expected {expected_total}, got {rcc.total_cohort_size}"
)
if rcc.cross_batch_consistency not in VALID_CONSISTENCY_RATINGS:
raise ValueError(
f"cross_batch_consistency {rcc.cross_batch_consistency!r} not in VALID_CONSISTENCY_RATINGS"
)
if rcc.rcc_grade not in VALID_RCC_GRADES:
raise ValueError(
f"rcc_grade {rcc.rcc_grade!r} not in VALID_RCC_GRADES"
)
if rcc.rcc_verdict not in VALID_RCC_VERDICTS:
raise ValueError(
f"rcc_verdict {rcc.rcc_verdict!r} not in VALID_RCC_VERDICTS"
)
if not rcc.confidence_rationale:
raise ValueError("confidence_rationale must be non-empty")
if not rcc.dry_lab_only:
raise ValueError("dry_lab_only must be True")
if not rcc.limitations:
raise ValueError("limitations must be non-empty")
if not rcc.created_at:
raise ValueError("created_at must be non-empty")


def _compute_consistency(
entries: list[BatchCohortEntry],
) -> str:
if len(entries) < MIN_BATCHES_FOR_CONFIDENCE:
return "unknown"
grades = [e.quality_grade for e in entries if e.quality_grade != "N/A"]
if not grades:
return "unknown"
unique = set(grades)
if len(unique) == 1:
return "consistent"
if len(unique) == 2:
return "moderately_consistent"
return "inconsistent"


def _compute_grade_and_verdict(
n_batches: int,
consistency: str,
total_cohort_size: int,
) -> tuple[str, str]:
if n_batches == 0:
return "D", "insufficient_data"
if n_batches < MIN_BATCHES_FOR_CONFIDENCE:
return "D", "insufficient_data"
if n_batches >= GRADE_A_MIN_BATCHES and consistency == "consistent":
return "A", "high_confidence"
if n_batches >= GRADE_B_MIN_BATCHES and consistency in ("consistent", "moderately_consistent"):
return "B", "moderate_confidence"
if n_batches >= GRADE_C_MIN_BATCHES and consistency != "inconsistent":
return "C", "low_confidence"
return "D", "low_confidence"


def _build_rationale(
n_batches: int,
consistency: str,
total_cohort_size: int,
grade: str,
) -> str:
return (
f"Grade {grade}: {n_batches} batch(es) assessed, "
f"total cohort size {total_cohort_size}, "
f"cross-batch consistency={consistency}."
)


def build_recalibration_confidence_certificate(
*,
rcc_id: str,
pipeline_version: str,
cit_id: str,
lpr_id: str,
batch_cohort_entry_dicts: list[dict],
limitations: list[str],
created_at: str,
) -> RecalibrationConfidenceCertificate:
"""Build a RecalibrationConfidenceCertificate.

batch_cohort_entry_dicts: list of dicts with keys:
batch_id, mbl_id, cohort_size, quality_grade
"""
entries = [
BatchCohortEntry(
batch_id=d["batch_id"],
mbl_id=d["mbl_id"],
cohort_size=int(d["cohort_size"]),
quality_grade=d["quality_grade"],
)
for d in batch_cohort_entry_dicts
]
total = sum(e.cohort_size for e in entries)
consistency = _compute_consistency(entries)
grade, verdict = _compute_grade_and_verdict(len(entries), consistency, total)
rationale = _build_rationale(len(entries), consistency, total, grade)
rcc = RecalibrationConfidenceCertificate(
rcc_id=rcc_id,
pipeline_version=pipeline_version,
cit_id=cit_id,
lpr_id=lpr_id,
batch_cohort_entries=entries,
n_batches_assessed=len(entries),
total_cohort_size=total,
cross_batch_consistency=consistency,
rcc_grade=grade,
rcc_verdict=verdict,
confidence_rationale=rationale,
dry_lab_only=True,
limitations=limitations,
created_at=created_at,
)
validate_recalibration_confidence_certificate(rcc)
return rcc


def format_recalibration_confidence_certificate(
rcc: RecalibrationConfidenceCertificate,
) -> str:
lines = [
f"Recalibration Confidence Certificate — {rcc.rcc_id}",
f"Pipeline: {rcc.pipeline_version} | CIT: {rcc.cit_id} | LPR: {rcc.lpr_id}",
f"Grade: {rcc.rcc_grade} | Verdict: {rcc.rcc_verdict}",
f"Batches assessed: {rcc.n_batches_assessed} | "
f"Total cohort: {rcc.total_cohort_size} | "
f"Consistency: {rcc.cross_batch_consistency}",
f"Rationale: {rcc.confidence_rationale}",
]
if rcc.batch_cohort_entries:
lines.append("Batch cohorts:")
for entry in rcc.batch_cohort_entries:
lines.append(
f" {entry.batch_id} ({entry.mbl_id}): "
f"cohort={entry.cohort_size} grade={entry.quality_grade}"
)
lines.append(f"Created: {rcc.created_at}")
lines.append(f"Limitations: {'; '.join(rcc.limitations)}")
lines.append(f"dry_lab_only: {rcc.dry_lab_only}")
return "\n".join(lines)
Loading
Loading