Skip to content

Commit 68c2906

Browse files
author
OpenCode
committed
feat: Phase AA AA2 determinism check record -- DCR- schema records same-step re-run comparison; verdict deterministic/nondeterministic/single_run_only; output hash comparison; blocks nondeterministic steps (#xxxx)
1 parent 674f9c6 commit 68c2906

4 files changed

Lines changed: 540 additions & 1 deletion

File tree

‎docs/research/NEXT_100_PR_MAP.md‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -325,3 +325,4 @@ Machine-verifiable proof that every pipeline run carries the fields needed for e
325325
| PR | Task | Why it matters | Review class |
326326
|---:|---|---|---|
327327
| AA1 | Add run manifest completeness schema (RMC-). | Validates pipeline run output manifests carry all 9 required reproducibility fields (commit_hash/config_hash/input_hash/seed/version/command/timestamp); verdict complete/incomplete/partial; blocks release of runs missing reproducibility metadata. | C |
328+
| AA2 | Add determinism check record schema (DCR-). | Records result of running same pipeline step twice; verdict deterministic/nondeterministic/single_run_only/seed_dependent; run1/run2 output hash comparison; blocks releasing nondeterministic steps for external review. | C |
Lines changed: 211 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,211 @@
1+
"""DCR- Determinism check record schema.
2+
3+
Records the result of running the same pipeline step twice and comparing
4+
outputs. A pipeline step is not trustworthy for external review until it
5+
passes a determinism check.
6+
Verdict: deterministic / nondeterministic / single_run_only / seed_dependent.
7+
"""
8+
9+
from __future__ import annotations
10+
11+
from dataclasses import dataclass
12+
13+
VALID_DCR_VERDICTS: frozenset[str] = frozenset({
14+
"deterministic",
15+
"nondeterministic",
16+
"single_run_only",
17+
"seed_dependent",
18+
})
19+
20+
VALID_DCR_STEP_TYPES: frozenset[str] = frozenset({
21+
"scoring",
22+
"ranking",
23+
"embedding",
24+
"filtering",
25+
"clustering",
26+
"selection",
27+
"evidence_generation",
28+
})
29+
30+
VALID_DCR_COMPARISON_METHODS: frozenset[str] = frozenset({
31+
"exact_match",
32+
"rank_correlation",
33+
"score_delta",
34+
"set_equality",
35+
})
36+
37+
38+
@dataclass
39+
class DeterminismCheckRecord:
40+
dcr_id: str
41+
pipeline_version: str
42+
step_type: str
43+
step_id: str
44+
random_seed: str
45+
run1_output_hash: str
46+
run2_output_hash: str
47+
comparison_method: str
48+
outputs_match: bool
49+
verdict: str
50+
similarity_score: float
51+
n_runs: int
52+
limitations: list[str]
53+
created_at: str
54+
dry_lab_only: bool = True
55+
56+
57+
def _compute_outputs_match(
58+
n_runs: int,
59+
run2_output_hash: str,
60+
run1_output_hash: str,
61+
) -> bool:
62+
return (
63+
n_runs >= 2
64+
and run2_output_hash != ""
65+
and run1_output_hash == run2_output_hash
66+
)
67+
68+
69+
def _compute_verdict(
70+
n_runs: int,
71+
run2_output_hash: str,
72+
run1_output_hash: str,
73+
outputs_match: bool,
74+
override_verdict: str | None = None,
75+
) -> str:
76+
if override_verdict is not None and override_verdict in VALID_DCR_VERDICTS:
77+
return override_verdict
78+
if n_runs == 1 or run2_output_hash == "":
79+
return "single_run_only"
80+
if outputs_match:
81+
return "deterministic"
82+
return "nondeterministic"
83+
84+
85+
def build_determinism_check_record(
86+
*,
87+
dcr_id: str,
88+
pipeline_version: str,
89+
step_type: str,
90+
step_id: str,
91+
random_seed: str,
92+
run1_output_hash: str,
93+
run2_output_hash: str = "",
94+
comparison_method: str,
95+
n_runs: int,
96+
similarity_score: float,
97+
limitations: list[str],
98+
created_at: str,
99+
verdict: str | None = None,
100+
) -> DeterminismCheckRecord:
101+
outputs_match = _compute_outputs_match(n_runs, run2_output_hash, run1_output_hash)
102+
computed_verdict = _compute_verdict(
103+
n_runs, run2_output_hash, run1_output_hash, outputs_match, override_verdict=verdict
104+
)
105+
dcr = DeterminismCheckRecord(
106+
dcr_id=dcr_id,
107+
pipeline_version=pipeline_version,
108+
step_type=step_type,
109+
step_id=step_id,
110+
random_seed=random_seed,
111+
run1_output_hash=run1_output_hash,
112+
run2_output_hash=run2_output_hash,
113+
comparison_method=comparison_method,
114+
outputs_match=outputs_match,
115+
verdict=computed_verdict,
116+
similarity_score=similarity_score,
117+
n_runs=n_runs,
118+
dry_lab_only=True,
119+
limitations=limitations,
120+
created_at=created_at,
121+
)
122+
validate_determinism_check_record(dcr)
123+
return dcr
124+
125+
126+
def validate_determinism_check_record(dcr: DeterminismCheckRecord) -> None:
127+
if not dcr.dcr_id.startswith("DCR-"):
128+
raise ValueError(f"dcr_id must start with 'DCR-': {dcr.dcr_id!r}")
129+
if not dcr.pipeline_version:
130+
raise ValueError("pipeline_version must be non-empty")
131+
if dcr.step_type not in VALID_DCR_STEP_TYPES:
132+
raise ValueError(
133+
f"step_type {dcr.step_type!r} not in VALID_DCR_STEP_TYPES"
134+
)
135+
if not dcr.step_id:
136+
raise ValueError("step_id must be non-empty")
137+
if not dcr.random_seed:
138+
raise ValueError("random_seed must be non-empty")
139+
if not dcr.run1_output_hash:
140+
raise ValueError("run1_output_hash must be non-empty")
141+
if dcr.comparison_method not in VALID_DCR_COMPARISON_METHODS:
142+
raise ValueError(
143+
f"comparison_method {dcr.comparison_method!r} not in "
144+
f"VALID_DCR_COMPARISON_METHODS"
145+
)
146+
if dcr.n_runs not in (1, 2):
147+
raise ValueError(f"n_runs must be 1 or 2, got {dcr.n_runs}")
148+
if dcr.n_runs == 2 and not dcr.run2_output_hash:
149+
raise ValueError("n_runs=2 requires non-empty run2_output_hash")
150+
if dcr.n_runs == 1 and dcr.run2_output_hash != "":
151+
raise ValueError("n_runs=1 requires run2_output_hash to be empty")
152+
expected_outputs_match = _compute_outputs_match(
153+
dcr.n_runs, dcr.run2_output_hash, dcr.run1_output_hash
154+
)
155+
if dcr.outputs_match != expected_outputs_match:
156+
raise ValueError(
157+
f"outputs_match {dcr.outputs_match} inconsistent with "
158+
f"n_runs={dcr.n_runs}, run1={dcr.run1_output_hash!r}, "
159+
f"run2={dcr.run2_output_hash!r}"
160+
)
161+
if dcr.verdict not in VALID_DCR_VERDICTS:
162+
raise ValueError(f"verdict {dcr.verdict!r} not in VALID_DCR_VERDICTS")
163+
if dcr.verdict == "single_run_only" and dcr.n_runs != 1:
164+
raise ValueError(
165+
"verdict single_run_only requires n_runs=1"
166+
)
167+
if dcr.verdict == "deterministic" and not dcr.outputs_match:
168+
raise ValueError(
169+
"verdict deterministic requires outputs_match=True"
170+
)
171+
if dcr.verdict == "nondeterministic" and dcr.outputs_match:
172+
raise ValueError(
173+
"verdict nondeterministic requires outputs_match=False"
174+
)
175+
if dcr.verdict == "seed_dependent":
176+
if dcr.n_runs != 2:
177+
raise ValueError("seed_dependent verdict requires n_runs=2")
178+
if not dcr.run2_output_hash:
179+
raise ValueError("seed_dependent verdict requires non-empty run2_output_hash")
180+
if not (-1.0 <= dcr.similarity_score <= 1.0):
181+
raise ValueError(
182+
f"similarity_score must be in [-1.0, 1.0], got {dcr.similarity_score}"
183+
)
184+
if not dcr.dry_lab_only:
185+
raise ValueError("dry_lab_only must be True")
186+
if not dcr.limitations:
187+
raise ValueError("limitations must be non-empty")
188+
if not dcr.created_at:
189+
raise ValueError("created_at must be non-empty")
190+
191+
192+
def format_determinism_check_record(dcr: DeterminismCheckRecord) -> str:
193+
lines = [
194+
f"Determinism Check Record — {dcr.dcr_id}",
195+
f"Pipeline: {dcr.pipeline_version}",
196+
f"Step: {dcr.step_type} / {dcr.step_id}",
197+
f"Seed: {dcr.random_seed}",
198+
f"Comparison: {dcr.comparison_method}",
199+
f"Runs: {dcr.n_runs}",
200+
f"Outputs match: {dcr.outputs_match}",
201+
f"Run 1 hash: {dcr.run1_output_hash}",
202+
]
203+
if dcr.n_runs == 2 and dcr.run2_output_hash:
204+
lines.append(f"Run 2 hash: {dcr.run2_output_hash}")
205+
lines.append(f"Similarity: {dcr.similarity_score}")
206+
lines.append(f"Verdict: {dcr.verdict}")
207+
if dcr.limitations:
208+
lines.append(f"Limitations: {'; '.join(dcr.limitations)}")
209+
lines.append(f"Created: {dcr.created_at}")
210+
lines.append(f"dry_lab_only: {dcr.dry_lab_only}")
211+
return "\n".join(lines)

0 commit comments

Comments
 (0)