Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
24 commits
Select commit Hold shift + click to select a range
3d7f2ac
ci: add cross-platform docker and hardware workflows
haoruilee Jun 26, 2026
07954dd
fix(ci): address 6 review findings in cross-platform CI
haoruilee Jun 27, 2026
4d13535
fix(ci): resolve all blocking and quality gaps from review
haoruilee Jun 27, 2026
2085ccd
fix(ci): drop cache-to from build-pr job (fork PR cache writes fail)
haoruilee Jun 27, 2026
3ef3ba9
fix(ci): add strict gpu preflight safeguards
haoruilee Jun 28, 2026
2610c62
fix(ci): handle runpod create failures
haoruilee Jun 28, 2026
f3c5230
fix(ci): use runpod ssh key for pods
haoruilee Jun 28, 2026
5c6f18d
fix(ci): create runpod workspace before clone
haoruilee Jun 28, 2026
8924cd3
fix(ci): harden runpod pip installs
haoruilee Jun 28, 2026
a21a928
fix(ci): explicitly build cuda extension on runpod
haoruilee Jun 28, 2026
7a03545
fix(ci): make flashinfer install optional for gpu core tests
haoruilee Jun 28, 2026
daf021b
fix(ci): wait for runpod ssh readiness
haoruilee Jun 28, 2026
337585e
fix(cuda): correct sm90 tma barrier sequencing
haoruilee Jun 28, 2026
1524d2e
fix(ci): retry transient runpod ssh disconnects
haoruilee Jun 28, 2026
55a2ac2
fix(cuda): pass sm90 tensor maps as raw addresses
haoruilee Jun 28, 2026
43c6b0f
fix(cuda): use grid-constant addresses for sm90 tensor maps
haoruilee Jun 28, 2026
df29724
fix(cuda): store sm90 tensor maps in device memory
haoruilee Jun 28, 2026
e9ceeb8
fix(cuda): use cuda12 tma tile modifier
haoruilee Jun 28, 2026
53f30d6
fix(cuda): preserve shared address state for sm90 tma
haoruilee Jun 28, 2026
3ebf596
fix(cuda): use cuda barriers for sm90 tma
haoruilee Jun 28, 2026
78a023a
fix(cuda): force global state for sm90 tensor maps
haoruilee Jun 28, 2026
263e116
fix(cuda): compile sm90 tma with cluster shared state
haoruilee Jun 28, 2026
734ea0f
fix(cuda): fall back from sm90 logp for fp32 logits
haoruilee Jun 28, 2026
d270e99
fix(cuda): expose generic logp api from sm90 wrapper
haoruilee Jun 28, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions .github/actionlint.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,3 @@
self-hosted-runner:
labels:
- rocm
77 changes: 69 additions & 8 deletions .github/workflows/build-ci-image.yml
Original file line number Diff line number Diff line change
@@ -1,29 +1,86 @@
name: Build CI Image (NVIDIA/CUDA)
name: Build CI Images

on:
push:
branches: [ main ]
paths:
- 'docker/Dockerfile.cuda'
- 'docker/**'
- 'pyproject.toml'
- 'setup.py'
- 'requirements*.txt'
- 'csrc/**'
- '.github/workflows/build-ci-image.yml'
pull_request:
branches: [ main ]
paths:
- 'docker/**'
- 'pyproject.toml'
- 'setup.py'
- 'requirements*.txt'
- 'csrc/**'
- '.github/workflows/build-ci-image.yml'
workflow_dispatch:

jobs:
build-pr:
if: github.event_name == 'pull_request'
runs-on: ubuntu-latest
permissions:
contents: read
strategy:
fail-fast: false
matrix:
include:
- backend: cuda
dockerfile: docker/Dockerfile.cuda
tag: cuda
- backend: rocm
dockerfile: docker/Dockerfile.rocm
tag: rocm
steps:
- uses: actions/checkout@v4

- name: Set lower case image name
run: |
REPO="${GITHUB_REPOSITORY,,}"
echo "IMAGE=ghcr.io/$REPO/rl-kernel-ci" >> "$GITHUB_ENV"

- name: Set up Docker Buildx
uses: docker/setup-buildx-action@v3

- name: Build image
uses: docker/build-push-action@v6
with:
context: .
file: ${{ matrix.dockerfile }}
platforms: linux/amd64
push: false
tags: ${{ env.IMAGE }}:${{ matrix.tag }}
cache-from: type=gha,scope=ci-image-${{ matrix.backend }}

build-and-push:
if: github.event_name != 'pull_request'
runs-on: ubuntu-latest
permissions:
contents: read
packages: write
strategy:
fail-fast: false
matrix:
include:
- backend: cuda
dockerfile: docker/Dockerfile.cuda
tag: cuda
- backend: rocm
dockerfile: docker/Dockerfile.rocm
tag: rocm
steps:
- uses: actions/checkout@v4

- name: Set lower case image name
run: |
REPO="${GITHUB_REPOSITORY,,}"
echo "IMAGE=ghcr.io/$REPO/rl-kernel-ci:cuda" >> "$GITHUB_ENV"
echo "IMAGE=ghcr.io/$REPO/rl-kernel-ci" >> "$GITHUB_ENV"

- name: Log in to GitHub Container Registry
uses: docker/login-action@v3
Expand All @@ -32,12 +89,16 @@ jobs:
username: ${{ github.actor }}
password: ${{ secrets.GITHUB_TOKEN }}

- name: Build and push
- name: Set up Docker Buildx
uses: docker/setup-buildx-action@v3

- name: Build and push image
uses: docker/build-push-action@v6
with:
context: .
file: docker/Dockerfile.cuda
file: ${{ matrix.dockerfile }}
platforms: linux/amd64
push: true
tags: ${{ env.IMAGE }}
cache-from: type=gha
cache-to: type=gha,mode=max
tags: ${{ env.IMAGE }}:${{ matrix.tag }}
cache-from: type=gha,scope=ci-image-${{ matrix.backend }}
cache-to: type=gha,mode=max,scope=ci-image-${{ matrix.backend }}
8 changes: 7 additions & 1 deletion .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -61,13 +61,19 @@ jobs:
run: |
python -m pip install --upgrade pip
pip install torch --index-url https://download.pytorch.org/whl/cpu
pip install -e ".[dev]"
pip install -e ".[test]"

- name: Run Mocked Hardware Discovery Tests
run: |
python -m pytest rl_engine/tests/test_dispatch.py -v
python -m pytest tests/test_kernel_registry.py -q
PYTEST_DISABLE_PLUGIN_AUTOLOAD=1 python -m pytest tests/test_attention_correctness.py -q -rs

- name: Run CPU-only Operator Tests
run: |
python -m pytest tests/test_linear_logp.py -q -rs -k "native"
python -m pytest tests/test_cpu_hal.py -q

docs:
runs-on: ubuntu-latest
steps:
Expand Down
56 changes: 51 additions & 5 deletions .github/workflows/gpu-ci.yml
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
name: GPU CI
name: Hardware GPU CI

on:
pull_request_target:
Expand All @@ -10,21 +10,41 @@ on:
- 'pyproject.toml'
- 'setup.py'
- 'requirements*.txt'
- 'docker/Dockerfile.cuda'
- 'docker/**'
- 'ci/run_gpu_ci.sh'
- 'ci/run_rocm_ci.sh'
- '.github/workflows/gpu-ci.yml'
types: [ opened, synchronize, reopened, labeled ]

concurrency:
group: gpu-ci-serial
cancel-in-progress: false
group: hardware-gpu-ci-${{ github.event.pull_request.number }}
cancel-in-progress: true

permissions:
contents: read

jobs:
gpu-tests:
cuda-runpod:
if: contains(github.event.pull_request.labels.*.name, 'needs-gpu-ci')
runs-on: ubuntu-latest
timeout-minutes: 60
strategy:
fail-fast: false
max-parallel: 1
matrix:
include:
- name: cuda-a4000-tp2
gpu_id: NVIDIA RTX A4000
gpu_count: 2
fallback_gpu_id: NVIDIA A40
fallback_gpu_count: 1
torch_cuda_arch_list: "8.6"
- name: cuda-a40-tp1
gpu_id: NVIDIA A40
gpu_count: 1
fallback_gpu_id: NVIDIA RTX A4000
fallback_gpu_count: 1
torch_cuda_arch_list: "8.6"
steps:
- name: Checkout secure orchestrator script from base branch
uses: actions/checkout@v4
Expand All @@ -51,6 +71,32 @@ jobs:
- name: Run GPU tests on RunPod
env:
RUNPOD_API_KEY: ${{ secrets.RUNPOD_API_KEY }}
RUNPOD_GPU_ID: ${{ matrix.gpu_id }}
RUNPOD_GPU_COUNT: ${{ matrix.gpu_count }}
RUNPOD_FALLBACK_GPU_ID: ${{ matrix.fallback_gpu_id }}
RUNPOD_FALLBACK_GPU_COUNT: ${{ matrix.fallback_gpu_count }}
RUNPOD_PROFILE_NAME: ${{ matrix.name }}
TORCH_CUDA_ARCH_LIST: ${{ matrix.torch_cuda_arch_list }}
PR_REPO_URL: ${{ github.event.pull_request.head.repo.clone_url }}
PR_SHA: ${{ github.event.pull_request.head.sha }}
run: bash ci/run_gpu_ci.sh

rocm-self-hosted:
if: contains(github.event.pull_request.labels.*.name, 'needs-rocm-ci')
runs-on: [ self-hosted, linux, x64, rocm ]
timeout-minutes: 90
steps:
# Check out only the base branch so fork-controlled code never runs on
# the self-hosted runner with host privileges. run_rocm_ci.sh clones the
# PR code into an isolated /tmp workspace at test time.
- name: Checkout secure CI scripts from base branch
uses: actions/checkout@v4
with:
ref: ${{ github.event.pull_request.base.sha }}

- name: Run ROCm hardware tests
env:
RL_KERNEL_ROCM_ATTN_BACKEND: sdpa
PR_REPO_URL: ${{ github.event.pull_request.head.repo.clone_url }}
PR_SHA: ${{ github.event.pull_request.head.sha }}
run: bash ci/run_rocm_ci.sh
22 changes: 20 additions & 2 deletions benchmarks/profiler.py
Original file line number Diff line number Diff line change
Expand Up @@ -83,6 +83,16 @@ class GPUProfiler:

@staticmethod
def get_target_info(device_index: int = 0) -> GPUTargetInfo:
if device_index is not None and device_index < 0:
return GPUTargetInfo(
name="CPU",
architecture="unknown",
total_memory_gb=0.0,
driver_version="N/A",
backend="cpu",
compute_capability=None,
device_index=-1,
)
if not torch.cuda.is_available():
return GPUTargetInfo(
name="CPU",
Expand All @@ -94,7 +104,7 @@ def get_target_info(device_index: int = 0) -> GPUTargetInfo:
device_index=-1,
)

if device_index is None or device_index < 0:
if device_index is None:
device_index = torch.cuda.current_device()
device = torch.device(f"cuda:{device_index}")
name = torch.cuda.get_device_name(device)
Expand Down Expand Up @@ -554,7 +564,15 @@ def _fused_logp_fn(
generator=generator,
)
op = kernel_registry.get_op("logp")
return lambda: op.apply_fp32(logits, token_ids)
backend_name = op.__class__.__name__
if not backend_name.startswith("FusedLogp"):
raise RuntimeError(
"logp-fused requested, but kernel dispatch selected "
f"{backend_name}. Build the CUDA extension and verify strict fused dispatch first."
)
if hasattr(op, "apply_fp32"):
return lambda: op.apply_fp32(logits, token_ids)
return lambda: op(logits, token_ids).float()


def _sampling_fn(
Expand Down
Loading
Loading