Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion docs/research/NEXT_100_PR_MAP.md
Original file line number Diff line number Diff line change
Expand Up @@ -99,7 +99,7 @@ Make qualified external review easier and safer.
| E5 | Add non-protocol pilot pre-registration schema (complete). — evidence/pilot_preregistration.py: non-protocol pilot pre-registration schema (PPR-) freezing selection logic before batch release; tests/evidence/test_pilot_preregistration.py + test_pilot_preregistration_schema.py. | Freezes selection logic. | C/D |
| E6 | Add packet generator CLI (complete). — scripts/generate_review_packet.py: generates skeleton external review packet JSON; make generate-review-packet target; validates against schemas/external_review_packet.schema.json; dry_lab_only_attestation=True enforced. | Reduces manual packaging errors. | C/D |
| E7 | Add packet validator CLI (complete). — src/openamp_foundry/cli/commands/validate_packet.py: load_packet_from_json() reads ERP- JSON from disk; validate_packet_file() returns {valid, violations, packet_id, error}; _run_validate_packet() prints PASS/FAIL with violations; 45 tests in tests/cli/test_validate_packet.py. | Review readiness becomes testable. | C/D |
| E8 | Add release-summary generator that strips restricted fields. | Safer public summaries. | D |
| E8 | Add release-summary generator that strips restricted fields. | Safer public summaries. | D | DONE |
| E9 | Add domain review outcome schema (complete). | Structured expert verdict on a PEP with controlled taxonomy of domains and outcomes; closes ESC→RVQ→DRO review chain. | B/C |
| E10 | Add expert-review example with mock/toy candidates only (complete). | ERP- schema: 14 fields, 16 validation rules, mock candidate ID prefix enforcement (MOCK-/TOY-/EXAMPLE-/DEMO-/TEST-), is_example_data=True and dry_lab_only=True enforced; CI-checkable template cannot accidentally leak real candidates. | B/C |

Expand Down
181 changes: 181 additions & 0 deletions src/openamp_foundry/evidence/release_summary.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,181 @@
"""RSM- release-summary schema and generator.

Produces a public-safe summary of what was released by stripping restricted
fields before any external distribution. Every external share of pipeline
outputs must go through this generator — no ad-hoc field omission.

Restricted fields list what must NOT appear in public summaries:
individual candidate IDs, internal batch references, rejection reasons,
raw sequence data, reviewer identities, and internal restriction text.

Makes the stripping step machine-checkable rather than relying on human
memory of what is sensitive.
"""

from __future__ import annotations

from dataclasses import dataclass

RESTRICTED_FIELDS: tuple[str, ...] = (
"erp_id",
"rejection_reason",
"restrictions",
"candidate_ids",
"reviewer_ids",
"raw_sequences",
"batch_ids",
"internal_notes",
"collaborator_names",
)

VALID_RSM_RELEASE_SCOPES: frozenset[str] = frozenset({
"academic_collaboration",
"public_preprint",
"internal_only",
"restricted_partner",
})

VALID_RSM_DECISIONS: frozenset[str] = frozenset({
"authorized",
"rejected",
"pending_review",
})


@dataclass
class ReleaseSummary:
rsm_id: str
srd_id: str
pipeline_version: str
release_scope: str
release_decision: str
candidate_count: int
safety_checks_summary: list[str]
public_notes: str
limitations_summary: str
restricted_fields_stripped: bool
dry_lab_only: bool
created_at: str


def validate_release_summary(rsm: ReleaseSummary) -> None:
if not rsm.rsm_id.startswith("RSM-"):
raise ValueError(f"rsm_id must start with 'RSM-': {rsm.rsm_id!r}")
if not rsm.srd_id.startswith("SRD-"):
raise ValueError(f"srd_id must start with 'SRD-': {rsm.srd_id!r}")
if not rsm.pipeline_version:
raise ValueError("pipeline_version must be non-empty")
if rsm.release_scope not in VALID_RSM_RELEASE_SCOPES:
raise ValueError(
f"release_scope {rsm.release_scope!r} not in VALID_RSM_RELEASE_SCOPES"
)
if rsm.release_decision not in VALID_RSM_DECISIONS:
raise ValueError(
f"release_decision {rsm.release_decision!r} not in VALID_RSM_DECISIONS"
)
if rsm.candidate_count < 0:
raise ValueError("candidate_count must be non-negative")
if not rsm.restricted_fields_stripped:
raise ValueError("restricted_fields_stripped must be True")
if not rsm.dry_lab_only:
raise ValueError("dry_lab_only must be True")
if not rsm.limitations_summary:
raise ValueError("limitations_summary must be non-empty")
if not rsm.created_at:
raise ValueError("created_at must be non-empty")


def build_release_summary(
*,
rsm_id: str,
srd_id: str,
pipeline_version: str,
release_scope: str,
release_decision: str,
candidate_count: int,
safety_checks_summary: list[str],
public_notes: str = "",
limitations_summary: str,
created_at: str,
) -> ReleaseSummary:
"""Build a ReleaseSummary with restricted_fields_stripped and dry_lab_only always True."""
rsm = ReleaseSummary(
rsm_id=rsm_id,
srd_id=srd_id,
pipeline_version=pipeline_version,
release_scope=release_scope,
release_decision=release_decision,
candidate_count=candidate_count,
safety_checks_summary=list(safety_checks_summary),
public_notes=public_notes,
limitations_summary=limitations_summary,
restricted_fields_stripped=True,
dry_lab_only=True,
created_at=created_at,
)
validate_release_summary(rsm)
return rsm


def strip_restricted_fields(source: dict) -> dict:
"""Return a copy of *source* with all RESTRICTED_FIELDS removed."""
return {k: v for k, v in source.items() if k not in RESTRICTED_FIELDS}


def generate_release_summary(
*,
rsm_id: str,
srd_id: str,
pipeline_version: str,
release_scope: str,
release_decision: str,
candidate_count: int,
safety_checks_summary: list[str],
limitations_summary: str,
public_notes: str = "",
created_at: str,
source_fields: dict | None = None,
) -> ReleaseSummary:
"""Generate a public-safe ReleaseSummary, stripping any restricted fields
found in *source_fields* before building.

*source_fields* is optional raw data that will be validated for absence of
restricted fields — raising ValueError if any restricted key is present.
This makes accidental leakage detectable at generation time.
"""
if source_fields is not None:
leaked = [k for k in RESTRICTED_FIELDS if k in source_fields]
if leaked:
raise ValueError(
f"Restricted fields present in source_fields, must be stripped first: {leaked}"
)
return build_release_summary(
rsm_id=rsm_id,
srd_id=srd_id,
pipeline_version=pipeline_version,
release_scope=release_scope,
release_decision=release_decision,
candidate_count=candidate_count,
safety_checks_summary=safety_checks_summary,
public_notes=public_notes,
limitations_summary=limitations_summary,
created_at=created_at,
)


def format_release_summary(rsm: ReleaseSummary) -> str:
lines = [
f"Release Summary — {rsm.rsm_id}",
f"Authorization: {rsm.srd_id} | Pipeline: {rsm.pipeline_version}",
f"Decision: {rsm.release_decision} | Scope: {rsm.release_scope}",
f"Candidates in summary: {rsm.candidate_count}",
f"Restricted fields stripped: {rsm.restricted_fields_stripped}",
]
if rsm.safety_checks_summary:
lines.append(f"Safety checks: {'; '.join(rsm.safety_checks_summary)}")
if rsm.public_notes:
lines.append(f"Notes: {rsm.public_notes}")
lines.append(f"Limitations: {rsm.limitations_summary}")
lines.append(f"Created: {rsm.created_at}")
lines.append(f"dry_lab_only: {rsm.dry_lab_only}")
return "\n".join(lines)
Loading
Loading