Repository navigation
Expand file tree
/
Copy pathMakefile
More file actions
535 lines (488 loc) · 25.4 KB
/
Copy pathMakefile
File metadata and controls
535 lines (488 loc) · 25.4 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
SHELL := /bin/bash
# Default configuration
# The scheme for the CLI package is typically "osaurus-cli" (the package name)
SCHEME_CLI := osaurus-cli
SCHEME_APP := osaurus
CONFIG := Release
PROJECT := App/osaurus.xcodeproj
WORKSPACE := osaurus.xcworkspace
DERIVED := build/DerivedData
XCODEBUILD_FLAGS ?=
.PHONY: help cli app install-cli serve status test ci-test computer-use-evidence clean wa-helper wa-helper-release bench-setup bench-ingest bench-ingest-chunks bench-run bench evals-prep evals evals-verbose evals-report evals-all evals-all-verbose evals-all-report evals-deterministic evals-capture-screen evals-loop evals-matrix evals-diff evals-contribute evals-compat evals-pr-report evals-pr-report-baseline evals-watcher-report evals-scoreboard
help:
@echo "Targets:"
@echo " cli Build CLI ($(SCHEME_CLI)) into $(DERIVED)"
@echo " app Build app ($(SCHEME_APP)) and embed CLI"
@echo " install-cli Install/update /usr/local/bin/osaurus symlink"
@echo " serve Build CLI and start server (use PORT=XXXX, EXPOSE=1)"
@echo " status Check if server is running"
@echo " bench-setup Clone EasyLocomo + apply patches + install deps"
@echo " bench-ingest Full LOCOMO ingestion (LLM extraction + chunks)"
@echo " bench-ingest-chunks Fast chunk-only backfill (no LLM, ~minutes)"
@echo " bench-run Run LOCOMO benchmark only (skip ingestion)"
@echo " bench Full ingest + run LOCOMO benchmark"
@echo " evals Run one OsaurusEvals suite (MODEL=, FILTER=, EVALS_SUITE=)"
@echo " evals-verbose Same as 'evals' plus per-case raw LLM response (debugging prompt iter)"
@echo " evals-report Same as 'evals' but also writes JSON to EVALS_OUT (build/evals.json)"
@echo " evals-all Run every suite under Packages/OsaurusEvals/Suites/* (MODEL=, FILTER=)"
@echo " evals-all-verbose Same as 'evals-all' plus per-case raw LLM response"
@echo " evals-all-report Same as 'evals-all' but writes per-suite JSON to EVALS_OUT_DIR (build/evals/)"
@echo " evals-deterministic Run the token-free suites with the floors gate (CI-safe, no model)"
@echo " evals-capture-screen Capture a real app's screen context into a (gitignored) fixture (APP=, OUT=)"
@echo " evals-loop Optimization loop: run all suites per model + scoreboard + diff (MODELS=, BASELINE=, RECORD=1 LABEL= to commit reports/SNAPSHOT+history)"
@echo " evals-matrix Cross-model scoreboard from a reports dir (DIR=, HISTORY= LABEL= to append a trend row)"
@echo " evals-diff All-domain before/after diff (BASELINE=, CURRENT=)"
@echo " evals-contribute Crowdsource: run one model on your Mac -> reports/community/<file>.json (MODEL=, PR=1 auto-PRs; see COMMUNITY_EVALS.md)"
@echo " evals-compat Fold reports/community/* into the COMPATIBILITY.md leaderboard (COMPAT_DIR=)"
@echo " evals-pr-report Generate local+frontier eval artifact bundle for PR review"
@echo " evals-pr-report-baseline Same as evals-pr-report plus BASELINE_DIR comparison"
@echo " evals-watcher-report Store a watcher eval report bundle and refresh its scoreboard"
@echo " evals-scoreboard Aggregate stored eval report bundles into scoreboard artifacts"
@echo " test Run OsaurusCore package tests via 'swift test'"
@echo " evals-test Run the OsaurusEvals harness unit tests (deterministic, token-free)"
@echo " ci-test Reproduce the CI test-core job locally (xcodebuild + xcbeautify)"
@echo " computer-use-evidence Run local Computer Use proof lane into build/computer-use-evidence/"
@echo " wa-helper Build the WhatsApp bridge helper (osaurus-wa) into build/ (needs Go)"
@echo " wa-helper-release Package build/osaurus-wa-macos.zip and print pin digests"
@echo " clean Remove DerivedData build output"
cli:
@echo "Building CLI ($(SCHEME_CLI))…"
xcodebuild -workspace $(WORKSPACE) -scheme $(SCHEME_CLI) -configuration $(CONFIG) -derivedDataPath $(DERIVED) build -quiet $(XCODEBUILD_FLAGS)
app: cli
@echo "Building app ($(SCHEME_APP))…"
xcodebuild -workspace $(WORKSPACE) -scheme $(SCHEME_APP) -configuration $(CONFIG) -derivedDataPath $(DERIVED) build -quiet $(XCODEBUILD_FLAGS)
@echo "Embedding CLI into App Bundle (Helpers)…"
# Copy osaurus-cli to osaurus.app/Contents/Helpers/osaurus
mkdir -p "$(DERIVED)/Build/Products/$(CONFIG)/osaurus.app/Contents/Helpers"
cp "$(DERIVED)/Build/Products/$(CONFIG)/osaurus-cli" "$(DERIVED)/Build/Products/$(CONFIG)/osaurus.app/Contents/Helpers/osaurus"
chmod +x "$(DERIVED)/Build/Products/$(CONFIG)/osaurus.app/Contents/Helpers/osaurus"
@echo "Building plugin host helper (osaurus-plugin-host)…"
xcodebuild -workspace $(WORKSPACE) -scheme osaurus-plugin-host -configuration $(CONFIG) -derivedDataPath $(DERIVED) build -quiet $(XCODEBUILD_FLAGS)
@echo "Embedding plugin host helper into App Bundle (Helpers)…"
# Killable out-of-process native plugin host (see PluginProcessHost.swift).
cp "$(DERIVED)/Build/Products/$(CONFIG)/osaurus-plugin-host" "$(DERIVED)/Build/Products/$(CONFIG)/osaurus.app/Contents/Helpers/osaurus-plugin-host"
chmod +x "$(DERIVED)/Build/Products/$(CONFIG)/osaurus.app/Contents/Helpers/osaurus-plugin-host"
@echo "Bundling sandbox kernel (Resources/SandboxRuntime)…"
./scripts/build/fetch_sandbox_kernel.sh "$(DERIVED)/Build/Products/$(CONFIG)/osaurus.app/Contents/Resources/SandboxRuntime"
install-cli: cli
@echo "Installing CLI symlink…"
./scripts/release/install_cli_symlink.sh --dev
serve: install-cli
@echo "Starting Osaurus server…"
@if [[ -n "$(PORT)" ]]; then \
ARGS="$$ARGS --port $(PORT)"; \
fi; \
if [[ "$(EXPOSE)" == "1" ]]; then \
ARGS="$$ARGS --expose"; \
fi; \
osaurus serve $$ARGS
status:
osaurus status
# WhatsApp Web bridge helper (whatsmeow). Dev lane: build locally and point
# the app at it with OSAURUS_WA_PATH (DEBUG builds only), mirroring the imsg
# helper's OSAURUS_IMSG_PATH override. Release distribution uses a pinned
# download manifest instead (scripts/build/wa-helper-manifest.json).
wa-helper:
@command -v go >/dev/null 2>&1 || { \
echo "Go toolchain not found. Install with: brew install go"; \
exit 1; \
}
@echo "Building osaurus-wa (WhatsApp bridge helper)…"
@mkdir -p build
cd helpers/osaurus-wa && go build -trimpath -o ../../build/osaurus-wa .
@echo "Built build/osaurus-wa ($$(build/osaurus-wa version))"
# Reproducible release archive for the pinned-helper download lane. Produces
# build/osaurus-wa-macos.zip plus the SHA-256 digests to copy into
# scripts/build/wa-helper-manifest.json and WhatsAppRuntimeAssets.swift when
# rotating pins (see docs/CHANNEL_RELEASE_RUNBOOK_WHATSAPP.md).
wa-helper-release: wa-helper
@echo "Packaging osaurus-wa release archive…"
@rm -f build/osaurus-wa-macos.zip
cd build && /usr/bin/ditto -c -k osaurus-wa osaurus-wa-macos.zip
@echo ""
@echo "executableSHA256: $$(shasum -a 256 build/osaurus-wa | cut -d' ' -f1)"
@echo "archiveSHA256: $$(shasum -a 256 build/osaurus-wa-macos.zip | cut -d' ' -f1)"
@echo ""
@echo "Upload build/osaurus-wa-macos.zip to the wa-helper-v$$(build/osaurus-wa version) release tag,"
@echo "then pin both digests in scripts/build/wa-helper-manifest.json and"
@echo "Packages/OsaurusCore/Services/WhatsApp/WhatsAppRuntimeAssets.swift."
test:
@echo "Running OsaurusCore tests…"
swift test --package-path Packages/OsaurusCore
# Harness unit tests for the evals package itself (fixture decode, scoring,
# regression lab, judge resolution). Deterministic and token-free — no LLM
# calls — so this is safe for CI, unlike the eval suites themselves.
evals-test:
@echo "Running OsaurusEvals harness tests…"
OSAURUS_DISABLE_KEYCHAIN_FOR_TESTS=1 swift test --package-path Packages/OsaurusEvals
# Mirrors the CI `test-core` execution policy: one xctest worker, the same
# timeout allowances, xcbeautify pipe, and xcresult bundle. Run this locally
# to reproduce a failed CI run without Xcode's parallel-worker starvation.
# After it finishes (pass or fail) you can `open build/Tests.xcresult` to
# get the same Test Navigator UI as Xcode.
ci-test:
@command -v xcbeautify >/dev/null 2>&1 || { \
echo "xcbeautify not found. Install with: brew install xcbeautify"; \
exit 1; \
}
@mkdir -p build
@rm -rf build/Tests.xcresult
@set -o pipefail; xcodebuild test \
-workspace osaurus.xcworkspace \
-scheme OsaurusCoreTests \
-resultBundlePath build/Tests.xcresult \
-quiet \
-skipPackagePluginValidation \
-skipMacroValidation \
-enableCodeCoverage NO \
-parallel-testing-enabled NO \
-parallel-testing-worker-count 1 \
-maximum-parallel-testing-workers 1 \
-test-timeouts-enabled YES \
-default-test-execution-time-allowance 180 \
-maximum-test-execution-time-allowance 300 \
COMPILER_INDEX_STORE_ENABLE=NO \
SWIFT_COMPILATION_MODE=incremental \
| xcbeautify --renderer terminal
@echo ""
@echo "Done. Inspect failures with: open build/Tests.xcresult"
computer-use-evidence:
@OUT_DIR="$(OUT_DIR)" RUN_EVALS="$(RUN_EVALS)" MODEL="$(MODEL)" STRICT="$(STRICT)" \
bash scripts/evals/computer-use-evidence.sh
## ── LOCOMO Benchmark ──────────────────────────────────────────────
BENCH_MODEL ?= openrouter/google/gemini-2.5-flash
BENCH_BASE_URL ?= http://localhost:1337
BENCH_BATCH ?= 20
EASYLOCOMO_REPO ?= https://github.com/playeriv65/EasyLocomo.git
EASYLOCOMO_DIR := benchmarks/EasyLocomo
BENCH_PYTHON := $(EASYLOCOMO_DIR)/.venv/bin/python
# Inference benchmark (TTFT / prefill / decode) against the running server.
# Usage: make bench-mlx [BENCH_MLX_ARGS="--model <id> --runs 5"]
BENCH_MLX_ARGS ?=
bench-mlx: install-cli
@echo "Running osaurus bench…"
osaurus bench $(BENCH_MLX_ARGS)
bench-setup:
@echo "Setting up EasyLocomo benchmark…"
@if [ ! -d "$(EASYLOCOMO_DIR)/.git" ]; then \
mkdir -p benchmarks && \
git clone $(EASYLOCOMO_REPO) $(EASYLOCOMO_DIR); \
else \
echo "EasyLocomo already cloned."; \
fi
@echo "Applying Osaurus patches…"
cd $(EASYLOCOMO_DIR) && git checkout -- . && git apply ../../scripts/benchmark/easylocomo.patch
@echo "Installing Python dependencies…"
cd $(EASYLOCOMO_DIR) && python -m venv .venv && .venv/bin/pip install -q -r requirements.txt
@echo "Done. Run 'make bench-ingest' then 'make bench-run'."
bench-ingest:
@echo "Ingesting LOCOMO conversations into Osaurus memory…"
$(BENCH_PYTHON) scripts/benchmark/ingest_locomo.py --base-url $(BENCH_BASE_URL)
bench-ingest-chunks:
@echo "Backfilling LOCOMO conversation chunks (no LLM, fast)…"
$(BENCH_PYTHON) scripts/benchmark/ingest_locomo.py --base-url $(BENCH_BASE_URL) --chunks-only --delay 0
bench-run:
@echo "Running LOCOMO benchmark (model=$(BENCH_MODEL), no-context, batch=$(BENCH_BATCH))…"
cd $(EASYLOCOMO_DIR) && .venv/bin/python run_evaluation.py \
--model $(BENCH_MODEL) \
--no-context \
--overwrite \
--batch-size $(BENCH_BATCH)
bench: bench-ingest bench-run
## ── OsaurusEvals (off-CI behaviour evals) ────────────────────────
# Override on the command line, e.g.
# make evals MODEL=foundation
# make evals MODEL=openai/gpt-4o-mini FILTER=browser
# make evals-report EVALS_OUT=reports/today.json
# Default model is `auto` (whatever ChatConfigurationStore is set to);
# see Packages/OsaurusEvals/README.md for the full --model grammar.
EVALS_ROOT := Packages/OsaurusEvals/Suites
EVALS_SUITE ?= $(EVALS_ROOT)/CapabilitySearch
EVALS_OUT ?= build/evals.json
EVALS_OUT_DIR ?= build/evals
# Floors gate (Config/floors.json) is on by default: per-suite pass-rate
# floors apply only to the deterministic suites listed in the file, and
# per-case recall floors apply only when the suite contains that domain,
# so the flag is a no-op for everything else. Disable with
# `make evals EVALS_FLOOR_FLAG=`.
EVALS_FLOOR_FLAG ?= --fail-on-floor
LOCAL_MODEL ?= foundation
FRONTIER_MODEL ?= openai/gpt-4o-mini
EVALS_PR_REPORT_OUT ?= build/evals/pr-report/$(shell date -u +%Y%m%dT%H%M%SZ)
EVALS_PR_REPORT_OUT := $(EVALS_PR_REPORT_OUT)
# The SwiftPM eval runner has no Info.plist, so `ModelManifest` cannot read a
# host version and refuses manifest-bearing bundles (`osaurus.json` with
# `required_osaurus_version`, e.g. the default Raptor model). Evals stand in
# for the latest released app version from the appcast; pin another with
# `make evals EVALS_HOST_VERSION=0.26.0`.
EVALS_HOST_VERSION ?= $(shell sed -n 's:.*<sparkle\:shortVersionString>\(.*\)</sparkle\:shortVersionString>.*:\1:p' docs/appcast.xml | head -1)
evals evals-verbose evals-report evals-all evals-all-verbose evals-all-report evals-deterministic \
evals-capture-screen evals-loop evals-matrix evals-diff evals-compat evals-pr-report \
evals-pr-report-baseline evals-scoreboard: export OSAURUS_HOST_VERSION = $(EVALS_HOST_VERSION)
EVALS_WATCHER_CHANNEL ?= main
EVALS_WATCHER_OUT ?= build/evals/watcher
EVALS_REPORT_PRESET ?= local-frontier
EVALS_WATCHER_ARTIFACT_ID ?=
EVALS_MAX_REGRESSIONS ?= 0
EVALS_SCOREBOARD_ROOT ?= $(EVALS_WATCHER_OUT)/$(EVALS_WATCHER_CHANNEL)
EVALS_SCOREBOARD_OUT ?= build/evals/scoreboard/$(shell date -u +%Y%m%dT%H%M%SZ)
EVALS_SCOREBOARD_OUT := $(EVALS_SCOREBOARD_OUT)
ifeq ($(strip $(EVALS_FROM_REPORTS)$(PLAN_ONLY)),)
EVALS_WATCHER_PREP := evals-prep
endif
# Auto-discovered list of every subdirectory under Suites/. Adding a new
# `Suites/MyDomain/` automatically picks it up here — no Makefile edit
# required when a new suite lands.
EVALS_ALL_SUITES := $(sort $(dir $(wildcard $(EVALS_ROOT)/*/)))
# Provision local assets the SwiftPM eval CLI can't self-provision: the
# MLX metallib (colocated beside the osaurus-evals binary) and the
# potion-base-4M embedder (Hugging Face cache). Idempotent; every evals*
# target depends on it so `make evals` works on a clean checkout. Skip
# with `make evals OSAURUS_EVALS_SKIP_PREP=1` if you've prepped manually.
evals-prep:
@if [ "$(OSAURUS_EVALS_SKIP_PREP)" != "1" ]; then \
bash scripts/evals/prepare-evals-env.sh; \
fi
evals: evals-prep
@echo "Running OsaurusEvals against $(EVALS_SUITE)…"
swift run --package-path Packages/OsaurusEvals osaurus-evals run \
--suite $(EVALS_SUITE) \
$(EVALS_FLOOR_FLAG) \
$(if $(MODEL),--model $(MODEL),) \
$(if $(FILTER),--filter $(FILTER),)
# Discover installed bundles using the same Core scanner as the app; each
# image-capable bundle runs the strict real-media suite in an isolated process.
VISION_EVALS_OUT ?= build/evals/vision-$(shell date -u +%Y%m%dT%H%M%SZ)
VISION_EVALS_OUT := $(VISION_EVALS_OUT)
.PHONY: evals-vision-installed
evals-vision-installed: evals-prep
swift build --package-path Packages/OsaurusEvals --product osaurus-evals
bash scripts/live-proof/run-installed-vision-evals.sh \
Packages/OsaurusEvals/.build/debug/osaurus-evals "$(VISION_EVALS_OUT)"
evals-verbose: evals-prep
@echo "Running OsaurusEvals (verbose) against $(EVALS_SUITE)…"
swift run --package-path Packages/OsaurusEvals osaurus-evals run \
--suite $(EVALS_SUITE) \
--verbose \
$(EVALS_FLOOR_FLAG) \
$(if $(MODEL),--model $(MODEL),) \
$(if $(FILTER),--filter $(FILTER),)
evals-report: evals-prep
@mkdir -p $(dir $(EVALS_OUT))
swift run --package-path Packages/OsaurusEvals osaurus-evals run \
--suite $(EVALS_SUITE) \
$(EVALS_FLOOR_FLAG) \
$(if $(MODEL),--model $(MODEL),) \
$(if $(FILTER),--filter $(FILTER),) \
--out $(EVALS_OUT)
@echo "Wrote $(EVALS_OUT)"
# Run every suite directory under $(EVALS_ROOT). The CLI exits 1 on any
# failed/errored case, so we run each suite independently (don't `set -e`)
# and aggregate exit codes so a single failure doesn't mask later suites.
# Final exit is non-zero if ANY suite failed.
evals-all: evals-prep
@echo "Discovered suites: $(notdir $(patsubst %/,%,$(EVALS_ALL_SUITES)))"
@rc=0; for suite in $(EVALS_ALL_SUITES); do \
echo ""; \
echo "── $$suite ──"; \
swift run --package-path Packages/OsaurusEvals osaurus-evals run \
--suite $$suite \
$(EVALS_FLOOR_FLAG) \
$(if $(MODEL),--model $(MODEL),) \
$(if $(FILTER),--filter $(FILTER),) \
|| rc=$$?; \
done; \
exit $$rc
evals-all-verbose: evals-prep
@echo "Discovered suites: $(notdir $(patsubst %/,%,$(EVALS_ALL_SUITES)))"
@rc=0; for suite in $(EVALS_ALL_SUITES); do \
echo ""; \
echo "── $$suite ──"; \
swift run --package-path Packages/OsaurusEvals osaurus-evals run \
--suite $$suite \
--verbose \
$(EVALS_FLOOR_FLAG) \
$(if $(MODEL),--model $(MODEL),) \
$(if $(FILTER),--filter $(FILTER),) \
|| rc=$$?; \
done; \
exit $$rc
# Writes one JSON report per suite under $(EVALS_OUT_DIR), named after
# the suite directory. Useful for CI dashboards / cross-run diffing.
evals-all-report: evals-prep
@mkdir -p $(EVALS_OUT_DIR)
@rc=0; for suite in $(EVALS_ALL_SUITES); do \
name=$$(basename $$suite); \
out="$(EVALS_OUT_DIR)/$$name.json"; \
echo ""; \
echo "── $$suite → $$out ──"; \
swift run --package-path Packages/OsaurusEvals osaurus-evals run \
--suite $$suite \
$(EVALS_FLOOR_FLAG) \
$(if $(MODEL),--model $(MODEL),) \
$(if $(FILTER),--filter $(FILTER),) \
--out $$out \
|| rc=$$?; \
done; \
echo ""; \
echo "Wrote per-suite reports to $(EVALS_OUT_DIR)/"; \
exit $$rc
# Deterministic token-free suites: pure-data scorers with no model load, no
# embedder, no network — every row is a code contract, so any failure is a
# regression (floors.json pins their pass rate at 1.0). Safe on hosted CI
# runners; the CI `test-evals` job runs this after the harness unit tests.
# No `evals-prep` dependency: these lanes never touch MLX or the embedder.
#
# SINGLE SOURCE OF TRUTH: `Config/floors.json#suitePassRates`. This list is
# derived from it, never hand-edited — a hand-maintained copy silently ships a
# thinner CI lane than the floor gate enforces the moment someone adds a suite
# to one file and forgets the other (issue #2266). Add or remove a
# deterministic suite by editing floors.json alone.
# `scripts/live-proof/assert-eval-floors-makefile-sync.sh` fails if this
# derivation is replaced by a literal list or if a declared suite has no
# `Suites/<name>/` directory.
EVALS_FLOORS_JSON := Packages/OsaurusEvals/Config/floors.json
EVALS_DETERMINISTIC_SUITES := $(shell jq -r '.suitePassRates | keys_unsorted[]' $(EVALS_FLOORS_JSON))
# Machine-readable echo of the derived list, so the sync guard can assert the
# derivation actually expands (a jq typo yields an EMPTY list, and the loop
# below would then "pass" by running nothing).
.PHONY: print-evals-deterministic-suites
print-evals-deterministic-suites:
@echo $(EVALS_DETERMINISTIC_SUITES)
#
# Model-free by construction: `OSAURUS_EVALS_SCRIPTED_ONLY=1` makes mixed
# suites (ComputerUseLoop) SKIP their live model-driven cases and score only
# the scripted rows, and `OSAURUS_EVALS_DISABLE_WARMUP=1` stops the runner from
# warming whatever model the local ChatConfiguration happens to point at —
# neither lane may load a local model.
evals-deterministic:
@rc=0; for name in $(EVALS_DETERMINISTIC_SUITES); do \
echo ""; \
echo "── $(EVALS_ROOT)/$$name ──"; \
OSAURUS_EVALS_SCRIPTED_ONLY=1 OSAURUS_EVALS_DISABLE_WARMUP=1 \
swift run --package-path Packages/OsaurusEvals osaurus-evals run \
--suite $(EVALS_ROOT)/$$name \
--fail-on-floor \
$(if $(FILTER),--filter $(FILTER),) \
|| rc=$$?; \
done; \
exit $$rc
# Capture a real app's screen context into a ScreenContextFixture JSON for the
# `screen_context` eval suite. Local-only: needs Accessibility permission for
# the process running it (grant your terminal in System Settings → Privacy &
# Security → Accessibility). Defaults to the frontmost app and a timestamped
# file under the gitignored Fixtures/ScreenContext/local/ dir. RENDER=1 also
# prints the exact injected block (the fast capture→diagnose loop).
# make evals-capture-screen
# make evals-capture-screen APP=Xcode RENDER=1
# make evals-capture-screen APP=Safari OUT=/tmp/safari.json
evals-capture-screen:
@swift run --package-path Packages/OsaurusEvals osaurus-evals capture-screen \
$(if $(APP),--app "$(APP)",) \
$(if $(OUT),--out $(OUT),) \
$(if $(RENDER),--render,)
# Optimization-loop backbone: prep → run every suite per model into a
# timestamped dir → cross-model matrix (scoreboard) → optional diff vs a
# saved baseline. The maintainer pipeline; see
# scripts/evals/optimization-loop.sh for env overrides (MODELS=, BASELINE=,
# FILTER=, STRICT=, EVALS_REPEAT=, PARALLEL_REMOTE=).
# make evals-loop
# make evals-loop MODELS="foundation qwen3-4b xai/grok-4.3" BASELINE=build/evals/loop/<prev>
# RECORD=1 LABEL="qwen fix" make evals-loop # also refresh committed reports/SNAPSHOT + history
# make evals-loop EVALS_REPEAT=3 # 3 trials per case; flaky rows marked, diff flake-aware
evals-loop:
@MODELS="$(MODELS)" BASELINE="$(BASELINE)" FILTER="$(FILTER)" STRICT="$(STRICT)" \
RECORD="$(RECORD)" LABEL="$(LABEL)" \
EVALS_REPEAT="$(EVALS_REPEAT)" PARALLEL_REMOTE="$(PARALLEL_REMOTE)" \
bash scripts/evals/optimization-loop.sh
# Cross-model scoreboard from an existing dir of *.json reports. Point
# MATRIX_OUT/MATRIX_MD at reports/SNAPSHOT.{json,md} and HISTORY at
# reports/history.jsonl to refresh the committed scoreboard by hand.
# make evals-matrix DIR=build/evals/loop/latest
evals-matrix:
@swift run --package-path Packages/OsaurusEvals osaurus-evals matrix $(DIR) \
$(if $(MATRIX_OUT),--out $(MATRIX_OUT),) \
$(if $(MATRIX_MD),--markdown $(MATRIX_MD),) \
$(if $(HISTORY),--history $(HISTORY),) \
$(if $(LABEL),--label "$(LABEL)",)
# All-domain before/after diff between two report dirs/files.
# make evals-diff BASELINE=build/evals/loop/<prev> CURRENT=build/evals/loop/latest
evals-diff:
@swift run --package-path Packages/OsaurusEvals osaurus-evals diff $(BASELINE) $(CURRENT) \
$(if $(DIFF_OUT),--out $(DIFF_OUT),) \
$(if $(DIFF_MD),--markdown $(DIFF_MD),) \
$(if $(STRICT),--fail-on-regression,)
# Crowdsource model compatibility: run the per-model LLM suites for ONE model on
# your hardware and emit a single contribution file under reports/community/.
# Export a strong judge key (e.g. XAI_API_KEY) or JUDGE_MODEL to avoid a
# self-judged (weaker) run. PR=1 auto-submits (branch -> push -> gh pr create).
# Contributor guide: COMMUNITY_EVALS.md.
# MODEL=mlx-community/Qwen3-4B-4bit make evals-contribute
# PR=1 MODEL=mlx-community/Qwen3-4B-4bit make evals-contribute
evals-contribute:
@MODEL="$(MODEL)" PR="$(PR)" bash scripts/evals/contribute.sh $(MODEL)
# Fold every contribution under reports/community/ into the committed
# COMPATIBILITY.{md,json} leaderboard. Run VALIDATE=1 for the PR gate (verify
# each contribution decodes and carries provenance) without rebuilding.
# make evals-compat
# VALIDATE=1 make evals-compat
COMPAT_DIR ?= reports/community
evals-compat:
@swift run --package-path Packages/OsaurusEvals osaurus-evals compat $(COMPAT_DIR) \
$(if $(VALIDATE),--validate,--out reports/COMPATIBILITY.json --markdown reports/COMPATIBILITY.md)
# Review-oriented artifact bundle for PRs that affect agent-loop behavior.
# Defaults to the required local+frontier lanes and AgentLoop,
# AgentLoopFrontier, and Subagent suites. SandboxFrontier is opt-in because it
# needs host sandbox prerequisites.
evals-pr-report: evals-prep
@mkdir -p "$(EVALS_PR_REPORT_OUT)"
swift run --package-path Packages/OsaurusEvals osaurus-evals report \
--local-model "$(LOCAL_MODEL)" \
--frontier-model "$(FRONTIER_MODEL)" \
--out-dir "$(EVALS_PR_REPORT_OUT)" \
$(if $(JUDGE_MODEL),--judge-model "$(JUDGE_MODEL)",) \
$(if $(FILTER),--filter "$(FILTER)",) \
$(if $(INCLUDE_SANDBOX_FRONTIER),--include-sandbox-frontier,)
@echo "Wrote eval PR report to $(EVALS_PR_REPORT_OUT)"
evals-pr-report-baseline: evals-prep
@if [[ -z "$(BASELINE_DIR)" ]]; then \
echo "BASELINE_DIR is required, e.g. make evals-pr-report-baseline BASELINE_DIR=build/evals/main-report"; \
exit 2; \
fi
@mkdir -p "$(EVALS_PR_REPORT_OUT)"
swift run --package-path Packages/OsaurusEvals osaurus-evals report \
--local-model "$(LOCAL_MODEL)" \
--frontier-model "$(FRONTIER_MODEL)" \
--baseline "$(BASELINE_DIR)" \
--out-dir "$(EVALS_PR_REPORT_OUT)" \
$(if $(JUDGE_MODEL),--judge-model "$(JUDGE_MODEL)",) \
$(if $(FILTER),--filter "$(FILTER)",) \
$(if $(INCLUDE_SANDBOX_FRONTIER),--include-sandbox-frontier,)
@echo "Wrote eval PR report with baseline comparison to $(EVALS_PR_REPORT_OUT)"
evals-watcher-report: $(EVALS_WATCHER_PREP)
scripts/evals/eval-watcher-report.sh \
--channel "$(EVALS_WATCHER_CHANNEL)" \
--out-root "$(EVALS_WATCHER_OUT)" \
$(if $(EVALS_WATCHER_ARTIFACT_ID),--artifact-id "$(EVALS_WATCHER_ARTIFACT_ID)",) \
--preset "$(EVALS_REPORT_PRESET)" \
--local-model "$(LOCAL_MODEL)" \
--frontier-model "$(FRONTIER_MODEL)" \
--max-regressions "$(EVALS_MAX_REGRESSIONS)" \
$(if $(BASELINE_DIR),--baseline "$(BASELINE_DIR)",) \
$(if $(JUDGE_MODEL),--judge-model "$(JUDGE_MODEL)",) \
$(if $(FILTER),--filter "$(FILTER)",) \
$(if $(INCLUDE_SANDBOX_FRONTIER),--include-sandbox-frontier,) \
$(if $(EVALS_FROM_REPORTS),--from-reports "$(EVALS_FROM_REPORTS)",) \
$(if $(OSAURUS_EVALS_STARTUP_TIMEOUT_SECONDS),--startup-timeout "$(OSAURUS_EVALS_STARTUP_TIMEOUT_SECONDS)",) \
$(if $(PLAN_ONLY),--plan-only,)
evals-scoreboard:
@mkdir -p "$(EVALS_SCOREBOARD_OUT)"
swift run --package-path Packages/OsaurusEvals osaurus-evals scoreboard \
--reports-root "$(EVALS_SCOREBOARD_ROOT)" \
--out-dir "$(EVALS_SCOREBOARD_OUT)" \
--max-regressions "$(EVALS_MAX_REGRESSIONS)"
@echo "Wrote eval scoreboard to $(EVALS_SCOREBOARD_OUT)"
## ── Housekeeping ─────────────────────────────────────────────────
clean:
rm -rf $(DERIVED)
@echo "Cleaned $(DERIVED)"