Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
27 commits
Select commit Hold shift + click to select a range
97115c2
Add control-kernel contracts
min9lin9 Jul 18, 2026
960637a
Add control-kernel validators
min9lin9 Jul 18, 2026
bab8e0f
Add fail-closed control evaluation
min9lin9 Jul 18, 2026
6803bcd
Add control-kernel audit CLI
min9lin9 Jul 18, 2026
8ecbbca
Add synthetic GP control fixture
min9lin9 Jul 18, 2026
a78feb3
Test control-kernel policy gates
min9lin9 Jul 18, 2026
2450da0
Test mandatory audit and decision seals
min9lin9 Jul 18, 2026
0b2b831
Route control-kernel CLI
min9lin9 Jul 18, 2026
a250c3f
Document control-kernel commands
min9lin9 Jul 18, 2026
8ef0444
Update package inventory for control kernel
min9lin9 Jul 18, 2026
361d6dd
Update package inventory contract
min9lin9 Jul 18, 2026
e37251c
Add temporary control-kernel CI diagnostics
min9lin9 Jul 18, 2026
e7a7977
Align diagnostics with full CI checkout
min9lin9 Jul 18, 2026
a87a4d6
Make replay check ordering deterministic
min9lin9 Jul 18, 2026
369ec30
Make replay run ordering deterministic
min9lin9 Jul 18, 2026
a29bc40
Stabilize service readiness evidence ordering
min9lin9 Jul 18, 2026
20e2424
Refresh deterministic service readiness baseline
min9lin9 Jul 18, 2026
4455ffe
Separate published and candidate release expectations
min9lin9 Jul 18, 2026
6384519
Refresh deterministic service readiness document
min9lin9 Jul 18, 2026
4403866
Refresh candidate package dry-run baseline
min9lin9 Jul 18, 2026
8587e96
Remove temporary control-kernel diagnostics
min9lin9 Jul 18, 2026
e28a388
Block seals for ineligible control runs
min9lin9 Jul 18, 2026
0946d3e
Test that blocked runs cannot be sealed
min9lin9 Jul 18, 2026
c92e03f
Test CLI seal refusal for blocked runs
min9lin9 Jul 18, 2026
eea9b28
Normalize nonsemantic pack size fields in readiness baseline
min9lin9 Jul 18, 2026
d622105
Verify seal integrity before storing or reusing
min9lin9 Jul 18, 2026
fe5c31d
Test seal-file integrity before persistence and reuse
min9lin9 Jul 18, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions docs/SERVICE_READINESS.md
Original file line number Diff line number Diff line change
Expand Up @@ -6,8 +6,8 @@ Status: ready

- service-loop: pass - docs/SERVICE_LOOP.md
- onboarding: pass - docs/ONBOARDING.md
- official-docs-coverage: pass - fixtures/replay/gajae-code/official-docs.json, fixtures/replay/awesome-codex-subagents/official-docs.json, fixtures/replay/kimi-agent-swarm-skill/official-docs.json
- external-replay: pass - fixtures/replay/gajae-code/replay.json, fixtures/replay/awesome-codex-subagents/replay.json, fixtures/replay/kimi-agent-swarm-skill/replay.json
- official-docs-coverage: pass - fixtures/replay/awesome-codex-subagents/official-docs.json, fixtures/replay/gajae-code/official-docs.json, fixtures/replay/kimi-agent-swarm-skill/official-docs.json
- external-replay: pass - fixtures/replay/awesome-codex-subagents/replay.json, fixtures/replay/gajae-code/replay.json, fixtures/replay/kimi-agent-swarm-skill/replay.json
- handoff-validation: pass - fixtures/handoffs/low.json, fixtures/handoffs/medium.json, fixtures/handoffs/high.json
- service-acceptance-gates: pass - fixtures/service-readiness/gates.json
- field-evidence: pass - evidence/field-readiness/oss-run-1
Expand Down
218 changes: 218 additions & 0 deletions fixtures/control-kernel/gp-screening-shadow-v0.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,218 @@
{
"schemaVersion": "boulder.control.fixture.v1",
"policy": {
"schemaVersion": "boulder.control.policy.v1",
"id": "gp-screening-shadow",
"version": "0.1.0",
"blockingSeverities": [
"critical",
"major"
],
"hardFailures": [
{
"id": "mandate-exclusion",
"signal": "mandate.exclusion.missed",
"severity": "critical",
"description": "A hard mandate exclusion was missed or suppressed."
},
{
"id": "conflict-escalation",
"signal": "conflict.escalation.missed",
"severity": "critical",
"description": "A required conflict or related-party escalation was missed."
},
{
"id": "unsupported-material-claim",
"signal": "evidence.material-claim.unsupported",
"severity": "major",
"description": "A material recommendation claim lacks admissible evidence."
}
],
"metricRules": [
{
"metricId": "decision-accuracy",
"operator": "gte",
"threshold": 0.75,
"description": "Decision agreement or adjudicated quality must meet the pilot floor."
},
{
"metricId": "evidence-coverage",
"operator": "gte",
"threshold": 0.9,
"description": "Material claims must be covered by admissible evidence."
},
{
"metricId": "risk-recall",
"operator": "gte",
"threshold": 0.8,
"description": "Known material risks must be surfaced."
}
]
},
"evidenceManifest": {
"schemaVersion": "boulder.control.evidence-manifest.v1",
"manifestId": "synthetic-screening-001-manifest",
"caseId": "synthetic-screening-001",
"evidenceCutoffAt": "2025-01-15T00:00:00.000Z",
"generatedAt": "2026-07-18T14:00:00.000Z",
"entries": [
{
"evidenceId": "cim-v1",
"sha256": "1111111111111111111111111111111111111111111111111111111111111111",
"sourceType": "document",
"classification": "confidential",
"sourceVersion": "v1",
"observedAt": "2025-01-10T00:00:00.000Z",
"sourceUriHash": "2222222222222222222222222222222222222222222222222222222222222222"
},
{
"evidenceId": "mandate-v3",
"sha256": "3333333333333333333333333333333333333333333333333333333333333333",
"sourceType": "database",
"classification": "restricted",
"sourceVersion": "v3",
"observedAt": "2025-01-01T00:00:00.000Z",
"sourceUriHash": "4444444444444444444444444444444444444444444444444444444444444444"
}
]
},
"runs": {
"pass": {
"schemaVersion": "boulder.control.run-event.v1",
"caseId": "synthetic-screening-001",
"taskId": "screening-decision-package",
"agentId": "underwriting-agent",
"agentVersion": "0.1.0",
"profileId": "gp-screening-shadow",
"profileVersion": "0.1.0",
"parentRunId": null,
"startedAt": "2026-07-18T14:01:00.000Z",
"completedAt": "2026-07-18T14:02:00.000Z",
"evidenceCutoffAt": "2025-01-15T00:00:00.000Z",
"evidenceManifestHash": "39c90fcde10b441df6cdf1c81999a0eda124a72a8946f812b817ca58f22b6ac4",
"policyId": "gp-screening-shadow",
"policyVersion": "0.1.0",
"policyHash": "9d82eaf6062bc15d597a374f791b64405be8ccb06379dcc9c4722dbb80b451ea",
"promptVersion": "screening-v0.1",
"model": {
"provider": "test",
"name": "synthetic-screening-model",
"version": "1.0.0"
},
"toolCalls": [
{
"toolId": "evidence-index",
"toolVersion": "0.1.0",
"action": "read",
"status": "completed",
"inputHash": "5555555555555555555555555555555555555555555555555555555555555555",
"outputHash": "6666666666666666666666666666666666666666666666666666666666666666"
}
],
"artifactHashes": [
"7777777777777777777777777777777777777777777777777777777777777777"
],
"status": "completed",
"runId": "synthetic-pass-run",
"idempotencyKey": "synthetic-pass-run-v1",
"metrics": {
"decision-accuracy": 0.83,
"evidence-coverage": 0.96,
"risk-recall": 0.88
},
"hardFailureSignals": []
},
"hardFailure": {
"schemaVersion": "boulder.control.run-event.v1",
"caseId": "synthetic-screening-001",
"taskId": "screening-decision-package",
"agentId": "underwriting-agent",
"agentVersion": "0.1.0",
"profileId": "gp-screening-shadow",
"profileVersion": "0.1.0",
"parentRunId": null,
"startedAt": "2026-07-18T14:01:00.000Z",
"completedAt": "2026-07-18T14:02:00.000Z",
"evidenceCutoffAt": "2025-01-15T00:00:00.000Z",
"evidenceManifestHash": "39c90fcde10b441df6cdf1c81999a0eda124a72a8946f812b817ca58f22b6ac4",
"policyId": "gp-screening-shadow",
"policyVersion": "0.1.0",
"policyHash": "9d82eaf6062bc15d597a374f791b64405be8ccb06379dcc9c4722dbb80b451ea",
"promptVersion": "screening-v0.1",
"model": {
"provider": "test",
"name": "synthetic-screening-model",
"version": "1.0.0"
},
"toolCalls": [
{
"toolId": "evidence-index",
"toolVersion": "0.1.0",
"action": "read",
"status": "completed",
"inputHash": "5555555555555555555555555555555555555555555555555555555555555555",
"outputHash": "6666666666666666666666666666666666666666666666666666666666666666"
}
],
"artifactHashes": [
"7777777777777777777777777777777777777777777777777777777777777777"
],
"status": "completed",
"runId": "synthetic-hard-failure-run",
"idempotencyKey": "synthetic-hard-failure-run-v1",
"metrics": {
"decision-accuracy": 1.0,
"evidence-coverage": 1.0,
"risk-recall": 1.0
},
"hardFailureSignals": [
"mandate.exclusion.missed"
]
},
"metricFailure": {
"schemaVersion": "boulder.control.run-event.v1",
"caseId": "synthetic-screening-001",
"taskId": "screening-decision-package",
"agentId": "underwriting-agent",
"agentVersion": "0.1.0",
"profileId": "gp-screening-shadow",
"profileVersion": "0.1.0",
"parentRunId": null,
"startedAt": "2026-07-18T14:01:00.000Z",
"completedAt": "2026-07-18T14:02:00.000Z",
"evidenceCutoffAt": "2025-01-15T00:00:00.000Z",
"evidenceManifestHash": "39c90fcde10b441df6cdf1c81999a0eda124a72a8946f812b817ca58f22b6ac4",
"policyId": "gp-screening-shadow",
"policyVersion": "0.1.0",
"policyHash": "9d82eaf6062bc15d597a374f791b64405be8ccb06379dcc9c4722dbb80b451ea",
"promptVersion": "screening-v0.1",
"model": {
"provider": "test",
"name": "synthetic-screening-model",
"version": "1.0.0"
},
"toolCalls": [
{
"toolId": "evidence-index",
"toolVersion": "0.1.0",
"action": "read",
"status": "completed",
"inputHash": "5555555555555555555555555555555555555555555555555555555555555555",
"outputHash": "6666666666666666666666666666666666666666666666666666666666666666"
}
],
"artifactHashes": [
"7777777777777777777777777777777777777777777777777777777777777777"
],
"status": "completed",
"runId": "synthetic-metric-failure-run",
"idempotencyKey": "synthetic-metric-failure-run-v1",
"metrics": {
"decision-accuracy": 0.83,
"evidence-coverage": 0.96,
"risk-recall": 0.6
},
"hardFailureSignals": []
}
}
}
13 changes: 9 additions & 4 deletions fixtures/package-inventory/packaged-files.v0.json
Original file line number Diff line number Diff line change
@@ -1,11 +1,11 @@
{
"schemaVersion": "packaged-files.v0",
"totalUniqueFiles": 186,
"totalPackedFiles": 187,
"totalUniqueFiles": 191,
"totalPackedFiles": 192,
"classes": [
{
"class": "runtime",
"count": 68,
"count": 72,
"files": [
"bin/boulder.js",
"bin/boulder.ts",
Expand All @@ -22,6 +22,10 @@
"src/cli-options.ts",
"src/cli-run-recording.ts",
"src/cli.ts",
"src/control-kernel-command.ts",
"src/control-kernel-types.ts",
"src/control-kernel-validation.ts",
"src/control-kernel.ts",
"src/executor-adapters.ts",
"src/executors.ts",
"src/export.ts",
Expand Down Expand Up @@ -176,12 +180,13 @@
},
{
"class": "fixture",
"count": 24,
"count": 25,
"files": [
"fixtures/benchmarks/mcp-server.json",
"fixtures/benchmarks/python-package.json",
"fixtures/benchmarks/typescript-library.json",
"fixtures/capabilities/codex-installed.json",
"fixtures/control-kernel/gp-screening-shadow-v0.json",
"fixtures/docs/doc-registry.v0.json",
"fixtures/handoffs/high.json",
"fixtures/handoffs/low.json",
Expand Down
4 changes: 4 additions & 0 deletions src/cli-format.ts
Original file line number Diff line number Diff line change
Expand Up @@ -55,6 +55,10 @@ export function printHelp(): void {
" boulder replay-check [--cwd path] [--json]",
" boulder replay-run [--cwd path] --dry-run [--json]",
" boulder doctor [--cwd path] [--json]",
" boulder control record --event path [--cwd path] [--json]",
" boulder control evaluate --event path --manifest path --policy path [--cwd path] [--json]",
" boulder control seal --event path --manifest path --policy path [--cwd path] [--json]",
" boulder control verify-seal --seal path --event path --manifest path --policy path [--cwd path] [--json]",
" boulder record field-readiness --run-id id --evidence path [--cwd path] [--json]",
" boulder export [--cwd path] [--force]",
"",
Expand Down
5 changes: 5 additions & 0 deletions src/cli.ts
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,7 @@ import { runCapabilityCommand } from "./capability-command";
import { formatLines, prettyJson, printHelp } from "./cli-format";
import { runOperationalCommand } from "./cli-ops-command";
import { optionValue, parseOptions, valueAfter } from "./cli-options";
import { runControlKernelCommand } from "./control-kernel-command";
import { UnsafeGeneratedWritePathError, writeGeneratedText } from "./fs";
import { exportHarness } from "./export";
import { runHandoffCommand } from "./handoff-command";
Expand Down Expand Up @@ -92,6 +93,10 @@ async function runMain(args: string[]): Promise<void> {
await runRunsCommand(parsed.commandArgs, args, options.cwd, options.json);
return;
}
if (command === "control") {
await runControlKernelCommand(args, { cwd: options.cwd, json: options.json });
return;
}
if (command === "bootstrap" && args.includes("interview")) {
const report = buildBootstrapInterview(optionValue(args, "--task"));
if (options.json) {
Expand Down
Loading
Loading