Skip to content

docs(kernels): add kernel framework docs and the LLVM interop design #57

docs(kernels): add kernel framework docs and the LLVM interop design

docs(kernels): add kernel framework docs and the LLVM interop design #57

Workflow file for this run

name: LLM Deploy v1.0
# CI for the Qwen3-0.6B-on-RISC-V project (docs/llm-deploy-v1.0/).
#
# Separate from ci.yml on purpose: that one tests the ScratchV compiler, this
# one tests the deployment project. A failure here should not read as "the
# compiler is broken".
#
# The PR job executes small numeric gates. A separate manual/nightly job
# downloads and validates the pinned full-model release with ORT.
#
# docs:interfaces docs:risks
# probe:qemu-matmul probe:small-transformer probe:qwen3-onnx
# numeric:ir-qwen3-small
#
# Later weeks' gates (numeric:ir-full-qwen3 at W3, numeric:qemu-full-qwen3 at
# W4, e2e:qemu-paris at W5, release:gate at W8) are NOT stubbed here. They get
# added when their week arrives, from the plan's §4.3 table.
on:
pull_request:
paths:
- 'docs/llm-deploy-v1.0/**'
- 'probes/**'
- 'scripts/check_llm_deploy_docs.py'
- 'scripts/check_w1_repro_env.py'
- 'export_qwen3_onnx.py'
- 'scratchv/**'
- 'requirements/**'
- 'tests/test_qwen3_small*.py'
- 'tests/test_tensor*.py'
- 'tests/test_riscv_tensor*.py'
- 'tests/test_optimizer*.py'
- 'tests/qwen_artifacts.py'
- 'tests/test_qwen_artifact_selection.py'
- 'tests/test_w1_qwen3*.py'
- 'tests/test_w1_repro_env.py'
- '.github/workflows/llm-deploy.yml'
schedule:
# Nightly, off the hour: 19:23 UTC = 03:23 北京时间.
- cron: '23 19 * * *'
workflow_dispatch:
permissions:
contents: read
env:
# Delivered numeric gates are mandatory: missing scripts/tools are failures.
PROBE_MATMUL: probes/w1_matmul_4x4/run.py
PROBE_TRANSFORMER: probes/w1_tiny_transformer/run.py
PROBE_QWEN3_EXPORT: probes/w1_qwen3_export/run.py
PROBE_QWEN3_SMALL: probes/w2_qwen3_small/run.py
QWEN3_PY: output/qwen3-probe-venv/bin/python
jobs:
llm-deploy:
name: "LLM Deploy v1.0 / W1"
runs-on: ubuntu-latest
timeout-minutes: 60
steps:
- uses: actions/checkout@v4
with:
lfs: true
- uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5
with:
python-version: "3.12"
# ── 前置:RISC-V 工具链 + Python 依赖 ──────────────────────────────
- name: Install RISC-V toolchain and QEMU
run: |
set -euo pipefail
if ! command -v clang >/dev/null || ! command -v ld.lld >/dev/null || ! command -v qemu-riscv32 >/dev/null || ! command -v qemu-system-riscv64 >/dev/null; then
sudo apt-get update
sudo apt-get install --no-install-recommends -y clang lld qemu-user qemu-system-misc
fi
missing=""
for tool in clang ld.lld qemu-riscv32 qemu-system-riscv64; do
command -v "$tool" >/dev/null 2>&1 || missing="$missing $tool"
done
if [ -n "$missing" ]; then
echo "::error::missing tools:$missing"
exit 1
fi
clang --version | head -1
qemu-riscv32 --version | head -1
qemu-system-riscv64 --version | head -1
- name: Install Python dependencies
run: |
python3 -m pip install --upgrade pip
python3 -m pip install -e ".[all]"
python3 -m pip install numpy "pytest==9.1.1"
python3 -m pip install "ziglang==0.14.1"
echo "SCRATCHV_CC=$(python3 -c 'from pathlib import Path; import ziglang; print(Path(ziglang.__file__).parent / "zig")')" >> "$GITHUB_ENV"
# ── docs:interfaces / docs:risks ──────────────────────────────────
- name: "docs:interfaces + docs:risks"
id: docs
run: python3 scripts/check_llm_deploy_docs.py --week W1
# ── probe:qemu-matmul ─────────────────────────────────────────────
- name: "probe:qemu-matmul"
id: matmul
run: |
set -euo pipefail
python3 "$PROBE_MATMUL"
- name: Tensor compiler and runtime regressions
run: |
python3 -m pytest tests/test_tensor_c_codegen.py tests/test_tensor_compiler.py tests/test_riscv_tensor_runtime.py tests/test_optimizer_numeric_semantics.py tests/test_qwen_artifact_selection.py tests/test_w1_qwen3_export.py tests/test_qwen3_small_riscv_gate.py tests/test_w1_repro_env.py -q
- name: Repeated Linux timeout cleanup regression
timeout-minutes: 5
run: |
set -euo pipefail
mkdir -p output/process-cleanup-repeats
for attempt in $(seq 1 20); do
python3 -B -m pytest \
tests/test_riscv_tensor_runtime.py::test_timeout_terminates_only_owned_process_tree \
-q -p no:cacheprovider \
--basetemp "output/process-cleanup-repeats/temp-$attempt" \
--junit-xml="output/process-cleanup-repeats/attempt-$attempt.xml"
done
# ── probe:small-transformer ───────────────────────────────────────
- name: "probe:small-transformer"
id: transformer
run: |
set -euo pipefail
python3 "$PROBE_TRANSFORMER" --model probes/w1_tiny_transformer/out/tiny_transformer_2l.onnx
# The real architecture probe uses random weights, needs no checkpoint
# download, and runs on PRs. Keep its export stack isolated from W1 and
# the generic compiler tests. Install the CPU wheel before requirements
# so pip never resolves torch from PyPI with CUDA dependencies.
- name: Install pinned CPU Qwen3 probe environment
run: |
set -euo pipefail
python3 -m venv output/qwen3-probe-venv
"$QWEN3_PY" -m pip install --upgrade pip
"$QWEN3_PY" -m pip install "torch==2.7.1" --index-url https://download.pytorch.org/whl/cpu
"$QWEN3_PY" -m pip install -r requirements/qwen3-small-probe.txt
"$QWEN3_PY" -m pip install --no-deps -e .
"$QWEN3_PY" -m pip check
"$QWEN3_PY" -c "import torch; assert torch.__version__ == '2.7.1+cpu' and torch.version.cuda is None, torch.__version__"
- name: W1 reproduction environment preflight
run: |
set -euo pipefail
"$QWEN3_PY" -X utf8 scripts/check_w1_repro_env.py \
--require-clean --output-dir output/w1-repro-env
- name: "numeric:ir-qwen3-small"
id: qwen3_small
timeout-minutes: 15
env:
PYTHONHASHSEED: "0"
OMP_NUM_THREADS: "1"
OPENBLAS_NUM_THREADS: "1"
MKL_NUM_THREADS: "1"
HF_HUB_OFFLINE: "1"
TRANSFORMERS_OFFLINE: "1"
run: |
set -euo pipefail
"$QWEN3_PY" "$PROBE_QWEN3_SMALL" --output-dir output/qwen3-small
- name: Explicit Qwen3 artifact optimization integration
id: qwen3_artifact_tests
env:
SCRATCHV_QWEN_ARTIFACT_DIR: output/qwen3-small
OMP_NUM_THREADS: "1"
OPENBLAS_NUM_THREADS: "1"
MKL_NUM_THREADS: "1"
run: |
set -euo pipefail
"$QWEN3_PY" -B -m pytest \
tests/test_optimizer_numeric_semantics.py::test_real_qwen_artifact_optimization_matches_ort \
-q -rs -o junit_family=xunit1 --junit-xml=output/qwen3-small/optimization-tests.xml
"$QWEN3_PY" - <<'PY'
import xml.etree.ElementTree as ET
cases = ET.parse('output/qwen3-small/optimization-tests.xml').getroot().findall('.//testcase')
expected = {f'test_real_qwen_artifact_optimization_matches_ort[{level}]' for level in ('none', 'basic', 'all')}
assert len(cases) == 3 and {case.get('name') for case in cases} == expected
assert not any(case.find(tag) is not None for case in cases for tag in ('skipped', 'failure', 'error'))
print('Qwen artifact integration: 3 passed, 0 skipped')
PY
- name: Qwen3 small probe regressions
id: qwen3_small_tests
env:
HF_HUB_OFFLINE: "1"
TRANSFORMERS_OFFLINE: "1"
run: |
set -euo pipefail
"$QWEN3_PY" -m pytest \
tests/test_qwen3_small_probe.py \
tests/test_qwen3_small_model.py \
tests/test_qwen3_small_gate.py -q \
--junit-xml=output/qwen3-small/tests.xml
- name: "numeric:qemu-qwen3-small"
id: qwen3_riscv
timeout-minutes: 25
env:
OMP_NUM_THREADS: "1"
OPENBLAS_NUM_THREADS: "1"
run: |
set -euo pipefail
"$QWEN3_PY" probes/w2_qwen3_small/riscv.py \
--model-dir output/qwen3-small --output-dir output/qwen3-riscv
- name: Upload RISC-V tensor evidence
if: always()
uses: actions/upload-artifact@v4
with:
name: riscv-tensor-probe-output
path: |
output/qemu-matmul/
output/qwen3-riscv/
output/process-cleanup-repeats/*.xml
if-no-files-found: ignore
- name: Summarize RISC-V numeric gate
if: always()
env:
PROBE_OUTCOME: ${{ steps.qwen3_riscv.outcome }}
run: |
{
echo "## numeric:qemu-qwen3-small"
echo "Outcome: $PROBE_OUTCOME"
if [ -f output/qwen3-riscv/report.md ]; then
cat output/qwen3-riscv/report.md
fi
} >> "$GITHUB_STEP_SUMMARY"
- name: Summarize Qwen3 small probe
if: always()
env:
PROBE_OUTCOME: ${{ steps.qwen3_small.outcome }}
TEST_OUTCOME: ${{ steps.qwen3_small_tests.outcome }}
ARTIFACT_TEST_OUTCOME: ${{ steps.qwen3_artifact_tests.outcome }}
run: |
{
echo "## numeric:ir-qwen3-small"
echo "Probe: $PROBE_OUTCOME; regression tests: $TEST_OUTCOME"
echo "Explicit artifact integration (requires 3 passed / 0 skipped): $ARTIFACT_TEST_OUTCOME"
echo ""
if [ -f output/qwen3-small/report.md ]; then
cat output/qwen3-small/report.md
else
echo "No report was produced; inspect the dependency/probe step logs."
fi
} >> "$GITHUB_STEP_SUMMARY"
- name: Upload Qwen3 small probe artifacts
if: always()
uses: actions/upload-artifact@v4
with:
name: qwen3-small-probe-output
path: output/qwen3-small/
if-no-files-found: ignore
- name: Upload W1 reproduction environment
if: always()
uses: actions/upload-artifact@v4
with:
name: w1-reproduction-environment
path: output/w1-repro-env/
if-no-files-found: ignore
- name: Upload probe artifacts
if: always()
uses: actions/upload-artifact@v4
with:
name: w1-probe-output
path: probes/**/out/
if-no-files-found: ignore
# ── 状态汇总 ──────────────────────────────────────────────────────
- name: W1 交付物状态
if: always()
env:
DOCS_OUTCOME: ${{ steps.docs.outcome }}
MATMUL_OUTCOME: ${{ steps.matmul.outcome }}
TRANSFORMER_OUTCOME: ${{ steps.transformer.outcome }}
QWEN_IR_OUTCOME: ${{ steps.qwen3_small.outcome }}
QWEN_RISCV_OUTCOME: ${{ steps.qwen3_riscv.outcome }}
run: |
{
echo "## W1 / small Qwen3 execution status"
echo ""
echo "| Gate | Actual step outcome |"
echo "|---|---|"
echo "| Document structure | $DOCS_OUTCOME |"
echo "| QEMU MatMul | $MATMUL_OUTCOME |"
echo "| Synthetic Transformer IR | $TRANSFORMER_OUTCOME |"
echo "| Real Qwen3 structure IR | $QWEN_IR_OUTCOME |"
echo "| Real Qwen3 structure QEMU | $QWEN_RISCV_OUTCOME |"
echo ""
echo "Full-model ORT validation runs in the separate manual/nightly job."
echo "Document checks do not certify team interface approval or second-person reproduction."
} >> "$GITHUB_STEP_SUMMARY"
full-qwen3-onnx:
name: "W1 / full Qwen3 ONNX release validation"
# This downloads 1.24 GB and executes the 2.38 GB FP32 model. It is not
# counted as a passing gate on PRs where it did not execute.
if: github.event_name != 'pull_request'
runs-on: ubuntu-latest
timeout-minutes: 60
steps:
- uses: actions/checkout@v4
with:
persist-credentials: false
- uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5
with:
python-version: "3.12"
- name: Install pinned full-model validation dependencies
run: |
python3 -m pip install "numpy==2.2.6" "onnx==1.18.0" "onnxruntime==1.22.1" "protobuf==5.29.5"
python3 -m pip check
- name: "probe:qwen3-onnx"
id: full_model
env:
OMP_NUM_THREADS: "2"
OPENBLAS_NUM_THREADS: "2"
run: |
set -euo pipefail
python3 "$PROBE_QWEN3_EXPORT" --mode download \
--model-dir output/qwen3-full-model \
--output-dir output/qwen3-full-probe --threads 2
- name: Summarize full-model gate
if: always()
env:
PROBE_OUTCOME: ${{ steps.full_model.outcome }}
run: |
{
echo "## probe:qwen3-onnx (complete FP32 release, ORT CPU)"
echo "Outcome: $PROBE_OUTCOME"
if [ -f output/qwen3-full-probe/report.md ]; then
cat output/qwen3-full-probe/report.md
else
echo "No report produced; inspect the failed step."
fi
echo "This validates the complete ONNX artifact, not ScratchV full-model execution."
} >> "$GITHUB_STEP_SUMMARY"
- name: Upload full-model validation evidence
if: always()
uses: actions/upload-artifact@v4
with:
name: qwen3-full-onnx-probe-output
path: output/qwen3-full-probe/
if-no-files-found: ignore