Skip to content

trunk-merge/pr-110846/0087bae5-aeeb-4fcb-b689-cf3c751addf8 #352798

trunk-merge/pr-110846/0087bae5-aeeb-4fcb-b689-cf3c751addf8

trunk-merge/pr-110846/0087bae5-aeeb-4fcb-b689-cf3c751addf8 #352798

Workflow file for this run

name: Rust CI
on:
workflow_dispatch:
push:
branches: [master, main]
pull_request:
concurrency:
group: ${{ github.workflow }}-${{ github.head_ref || github.ref }}
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
env:
CARGO_TERM_COLOR: always
UV_HTTP_TIMEOUT: 120
RUNS_ON_INTERNAL_PR: ${{ github.event_name != 'pull_request' || github.event.pull_request.head.repo.fork == false }}
jobs:
# Learning mode: nothing reads these outputs, so a recommendation cannot change what runs.
# Fork and Dependabot runs get no secret; the queue always runs everything anyway.
# The step fails open, because this telemetry must not turn a passing workflow red.
dynamic-ci-filter:
name: Trunk Dynamic CI (Learning Mode)
if: >-
github.event_name == 'pull_request' &&
github.repository == 'PostHog/posthog' &&
github.event.pull_request.head.repo.full_name == github.repository &&
github.actor != 'dependabot[bot]' &&
!startsWith(github.head_ref, 'trunk-merge/')
runs-on: ubuntu-24.04
timeout-minutes: 5
permissions: {}
steps:
- name: Ask Trunk which jobs this diff needs
continue-on-error: true
uses: trunk-io/dynamic-ci@7e3af9331e8ebdfe0c71ba7d0ff6b7424dde8c57 # v1
with:
token: ${{ secrets.TRUNK_API_TOKEN }}
# Job to decide if we should run rust ci
# See .github/actions/paths-filter/README.md for filter semantics
changes:
runs-on: ubuntu-24.04
timeout-minutes: 5
if: github.repository == 'PostHog/posthog'
name: Determine need to run Rust checks
permissions:
contents: read
pull-requests: read
# Set job outputs to values from filter step
outputs:
rust: ${{ steps.filter.outputs.rust || 'true' }}
steps:
# For pull requests it's not necessary to checkout the code, but we
# also want this to run on master so we need to checkout
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
clean: false
sparse-checkout: .github/actions/paths-filter
sparse-checkout-cone-mode: false
- uses: actions/create-github-app-token@1b10c78c7865c340bc4f6099eb2f838309f1e8c3 # v3.1.1
id: app-token
if: github.event_name == 'pull_request' && github.event.pull_request.head.repo.full_name == github.repository
with:
client-id: ${{ vars.GH_APP_POSTHOG_PATHS_FILTER_APP_ID }}
private-key: ${{ secrets.GH_APP_POSTHOG_PATHS_FILTER_PRIVATE_KEY }}
- uses: ./.github/actions/paths-filter
id: filter
if: github.event_name != 'push' # Run all tests on master push
with:
token: ${{ steps.app-token.outputs.token || github.token }}
filters: |
rust:
# Avoid running rust tests for irrelevant changes
- 'rust/**'
- 'proto/**'
- '.github/workflows/ci-rust.yml'
- '.github/actions/setup-uv/**'
- '.github/workflows/rust.yml'
- '.github/workflows/rust-docker-build.yml'
- '.github/actions/setup-protoc/**'
- '.github/actions/setup-sccache/**'
- 'posthog/management/commands/setup_test_environment.py'
- 'posthog/migrations/**'
- 'ee/migrations/**'
- 'docker-compose.dev.yml'
# Hypercache contract: Python serializer changes may break Rust deserialization
- 'posthog/api/feature_flag.py'
- 'posthog/models/feature_flag/flags_cache.py'
- 'posthog/models/feature_flag/feature_flag.py'
- 'rust/feature-flags/tests/fixtures/hypercache_contract.json'
# Compute affected crates and test shards. Only runs when Rust changes
# are detected, keeping the lightweight `changes` job fast for non-Rust PRs.
affected:
name: Compute affected Rust crates
needs: changes
if: needs.changes.outputs.rust == 'true'
runs-on: ubuntu-24.04
timeout-minutes: 10
permissions:
contents: read
outputs:
matrix: ${{ steps.shards.outputs.matrix }}
affected_crates: ${{ steps.affected.outputs.crates }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
fetch-depth: 0
filter: blob:none
sparse-checkout: |
rust/
proto/
.github/rust-images.yml
.github/actions/rust-compute-affected/
sparse-checkout-cone-mode: false
clean: false
- name: Install rust
uses: dtolnay/rust-toolchain@3c5f7ea28cd621ae0bf5283f0e981fb97b8a7af9
with:
toolchain: 1.91.1
- name: Compute affected crates
id: affected
uses: ./.github/actions/rust-compute-affected
with:
event-name: ${{ github.event_name }}
pr-base-sha: ${{ github.event.pull_request.base.sha }}
push-before-sha: ${{ github.event.before }}
rebuild-all: ${{ github.event_name == 'workflow_dispatch' && 'true' || 'false' }}
# Bin-pack affected packages into balanced test shards.
- name: Compute test shards
id: shards
env:
AFFECTED_CRATES: ${{ steps.affected.outputs.crates }}
shell: python3 {0}
run: |
import json, math, os
affected = set(json.loads(os.environ.get("AFFECTED_CRATES", "[]")))
packages = {
"affected-services": 37,
"assignment-coordination": 5,
"batch-import-worker": 69,
"capture": 102,
"capture-load-gen": 38,
"capture-logs": 87,
"capture-apm-metrics": 5,
"cohort-core": 22,
"cohort-event-shuffler": 25,
"cohort-seeder": 77,
"cohort-stream-processor": 146,
"common-alloc": 5,
"common-cache": 6,
"common-compression": 5,
"common-continuous-profiling": 9,
"common-cookieless": 103,
"common-database": 35,
"common-dns": 18,
"common-geoip": 5,
"common-hypercache": 36,
"common-ingestion-warnings": 59,
"common-kafka": 5,
"common-liveness": 5,
"common-metrics": 5,
"common-posthog": 18,
"common-profiler": 59,
"common-redis": 12,
"common-s3": 5,
"common-temporal": 12,
"common-types": 16,
"cymbal": 236,
"cymbal-proto": 5,
"deltalite-core": 5,
"deltalite-python": 5,
"embedding-worker": 90,
"feature-flags": 241,
"flags-consumer": 23,
"health": 5,
"hogql_parser_rs": 7,
"hogvm": 5,
"hypercache-server": 40,
"ingestion-consumer": 20,
"ingestion-control-plane": 41,
"pgcollector": 45,
"pgapi": 40,
"k8s-awareness": 116,
"kafka-assigner": 175,
"kafka-assigner-proto": 5,
"kafka-deduplicator": 129,
"keyed-stash": 5,
"lifecycle": 5,
"limiters": 5,
"personhog-common": 7,
"personhog-coordination": 230,
"personhog-identity": 80,
"personhog-leader": 113,
"personhog-proto": 9,
"personhog-replica": 37,
"personhog-router": 114,
"personhog-stateright": 173,
"personhog-test-harness": 33,
"personhog-writer": 63,
"posthog-replay-anonymizer": 147,
"posthog-symbol-data": 15,
"property-defs-rs": 74,
"property-vals-rs": 70,
"replay-anonymizer-node": 5,
"serve-metrics": 5,
"usage-ingestion": 70,
"usage-ingestion-proto": 5,
}
# Filter to only affected crates when selective builds are active.
# Unknown crates (not in packages dict) get a default weight so new
# services are never silently dropped from CI.
DEFAULT_WEIGHT = 60
if affected:
unknown = affected - packages.keys()
if unknown:
print(f"WARNING: affected crates missing from weight table (using {DEFAULT_WEIGHT}s default): {', '.join(sorted(unknown))}")
print("Add them to the packages dict in ci-rust.yml with measured weights.")
for crate in unknown:
packages[crate] = DEFAULT_WEIGHT
packages = {k: v for k, v in packages.items() if k in affected}
print(f"Selective build: testing {len(packages)} of affected crates")
else:
print("Full build: testing all crates")
if not packages:
with open(os.environ["GITHUB_OUTPUT"], "a") as f:
f.write('matrix={"include":[]}\n')
raise SystemExit(0)
# Weights are approximate marginal durations (compile + test) in seconds,
# fitted by non-negative least squares over 2 weeks of shard job
# durations (job time = fixed overhead + sum of member weights;
# last updated: 2026-08-18).
# TARGET_MINUTES controls max wall-clock per shard; shard count is derived.
TARGET_MINUTES = 8
target_seconds = TARGET_MINUTES * 60
total = sum(packages.values())
num_shards = max(1, math.ceil(total / target_seconds))
# Greedy bin-packing: largest first into lightest bucket
buckets = [[] for _ in range(num_shards)]
times = [0] * num_shards
for pkg, t in sorted(packages.items(), key=lambda x: -x[1]):
lightest = min(range(num_shards), key=lambda i: times[i])
buckets[lightest].append(pkg)
times[lightest] += t
matrix = {
"include": [
{"packages": " ".join(sorted(b))}
for b in buckets
if b
]
}
for i, b in enumerate(buckets):
if b:
print(f"Shard {i + 1} ({times[i]}s): {', '.join(sorted(b))}")
print(f"\nTotal: {total}s across {num_shards} shards (target ≤{target_seconds}s/shard)")
with open(os.environ["GITHUB_OUTPUT"], "a") as f:
f.write(f"matrix={json.dumps(matrix)}\n")
build:
name: Build Rust (${{ matrix.packages }})
strategy:
matrix: ${{ fromJSON(needs.affected.outputs.matrix) }}
needs: [changes, affected]
if: needs.changes.outputs.rust == 'true'
runs-on: depot-ubuntu-22.04-4
timeout-minutes: 20
permissions:
contents: read
defaults:
run:
working-directory: rust
steps:
# Checkout project code
# Use sparse checkout to only select files in rust directory
# Turning off cone mode ensures that files in the project root are not included during checkout
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
sparse-checkout: |
rust/
proto/
.github/actions/setup-protoc/
.github/actions/setup-sccache/
sparse-checkout-cone-mode: false
clean: false
- name: Install system OpenSSL
run: |
sudo rm -f /etc/apt/sources.list.d/*twingate*
sudo apt-get update
sudo apt-get install -y libssl-dev pkg-config
# protoc and rustup both write $GITHUB_PATH, so they stay out of a
# `parallel:` block. Concurrent writes crash the GitHub Actions runner
# ("Collection was modified" or a missing-key error), and the runner
# fails the step even though the tool installed fine.
- name: Install protoc
uses: ./.github/actions/setup-protoc
- name: Install rust
uses: dtolnay/rust-toolchain@3c5f7ea28cd621ae0bf5283f0e981fb97b8a7af9
with:
toolchain: 1.91.1
- name: Set sccache WebDAV key prefix (Cargo.lock hash)
run: echo "SCCACHE_WEBDAV_KEY_PREFIX=cargo-${{ hashFiles('rust/Cargo.lock') }}" >> "$GITHUB_ENV"
- name: Install sccache
uses: ./.github/actions/setup-sccache
- name: Cache Rust dependencies
uses: Swatinem/rust-cache@779680da715d629ac1d338a641029a2f4372abb5 # v2.8.2
with:
shared-key: 'v2-rust-ci'
workspaces: rust
save-if: ${{ github.ref == 'refs/heads/master' }}
- name: Run cargo build
env:
# Use system OpenSSL instead of vendored to avoid assembly build issue (2026-01-13)
OPENSSL_NO_VENDOR: '1'
SHARD_PACKAGES: ${{ matrix.packages }}
run: |
crates=()
for c in $SHARD_PACKAGES; do crates+=(-p "$c"); done
echo "Building shard crates: ${crates[*]}"
cargo build "${crates[@]}" --locked --release
find target/release/ -maxdepth 1 -executable -type f -exec strip {} +
- name: Report sccache counters
if: always()
continue-on-error: true
run: |
if command -v sccache >/dev/null; then
# The full report includes the cache endpoint, so allow only numeric counters.
sccache --show-stats 2>/dev/null | awk '/^(Compile requests|Cache hits|Cache misses|Non-cacheable calls)[[:space:]]+[0-9]+$/ { print }'
fi
test:
name: Test Rust (${{ matrix.packages }})
strategy:
matrix: ${{ fromJSON(needs.affected.outputs.matrix) }}
needs: [changes, affected]
if: needs.changes.outputs.rust == 'true'
runs-on: depot-ubuntu-24.04-4
timeout-minutes: 20
permissions:
contents: read
env:
DOCKERHUB_USERNAME: ${{ vars.DOCKERHUB_USER }}
DOCKERHUB_TOKEN: ${{ secrets.DOCKERHUB_TOKEN }}
defaults:
run:
working-directory: rust
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
clean: false
- name: Clean up data directories with container permissions
working-directory: .
run: |
# Use docker to clean up files created by containers (from repo root, not rust/)
[ -d "data" ] && docker run --rm -v "$(pwd)/data:/data" alpine sh -c "rm -rf /data/seaweedfs /data/minio" || true
continue-on-error: true
- name: Log in to Docker Hub
continue-on-error: true
if: ${{ env.DOCKERHUB_USERNAME != '' && env.DOCKERHUB_TOKEN != '' }}
uses: docker/login-action@4907a6ddec9925e35a0a9e82d7399ccc52663121 # v4.1.0
with:
username: ${{ vars.DOCKERHUB_USER }}
password: ${{ secrets.DOCKERHUB_TOKEN }}
- name: Setup main repo and dependencies
env:
COMPOSE_FILE: ../docker-compose.dev.yml:../docker-compose.profiles.yml
# `replay` boots SeaweedFS (S3 API on :8333), required by the
# batch-import-worker temp-bucket integration tests, which fail
# rather than skip when the store is down in CI.
COMPOSE_PROFILES: etcd,replay
WAIT_FOR_DOCKER_LAUNCH_RETRY_DELAY: 10
run: |
../bin/ci-wait-for-docker launch --down
../bin/ci-wait-for-docker wait seaweedfs
echo "127.0.0.1 db redis7 kafka clickhouse clickhouse-coordinator objectstorage seaweedfs temporal" | sudo tee -a /etc/hosts
- name: Dump Kafka logs on failure
if: failure()
run: |
docker ps -a || true
docker logs --tail=500 rust-kafka-1 || true
docker inspect rust-kafka-1 || true
# please keep the tag version here in sync with rust-version in rust/*/Cargo.toml
- name: Install rust
uses: dtolnay/rust-toolchain@3c5f7ea28cd621ae0bf5283f0e981fb97b8a7af9
with:
toolchain: 1.91.1
- name: Install protoc
uses: ./.github/actions/setup-protoc
- name: Set sccache WebDAV key prefix (Cargo.lock hash)
run: echo "SCCACHE_WEBDAV_KEY_PREFIX=cargo-${{ hashFiles('rust/Cargo.lock') }}" >> "$GITHUB_ENV"
- name: Install sccache
uses: ./.github/actions/setup-sccache
- name: Mint setup-action GitHub token
id: setup-gh-token
if: github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository
continue-on-error: true
uses: actions/create-github-app-token@1b10c78c7865c340bc4f6099eb2f838309f1e8c3 # v3.1.1
with:
client-id: ${{ vars.GH_APP_POSTHOG_SETUP_ACTIONS_APP_ID }}
private-key: ${{ secrets.GH_APP_POSTHOG_SETUP_ACTIONS_PRIVATE_KEY }}
skip-token-revoke: true
- name: Set up Python
uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0
with:
python-version-file: 'pyproject.toml'
token: ${{ steps.setup-gh-token.outputs.token || github.token }}
- name: Install uv
id: setup-uv
uses: ./.github/actions/setup-uv
- name: Install SAML (python3-saml) dependencies
if: steps.setup-uv.outputs.cache-hit != 'true'
run: |
sudo rm -f /etc/apt/sources.list.d/*twingate*
sudo apt-get update
sudo apt-get install libxml2-dev libxmlsec1-dev libxmlsec1-openssl postgresql-client
- name: Install python dependencies
run: |
UV_PROJECT_ENVIRONMENT=$pythonLocation uv sync --frozen --dev --directory ..
- name: Install sqlx-cli
uses: ./.github/actions/setup-sqlx-cli
- name: Set up databases
env:
DEBUG: 'true'
TEST: 'true'
SECRET_KEY: 'abcdef' # unsafe - for testing only
DATABASE_URL: 'postgres://posthog:posthog@localhost:5432/posthog'
run: cd ../ && python manage.py setup_test_environment
- name: Run sqlx migrations
env:
DATABASE_URL: 'postgres://posthog:posthog@localhost:5432/posthog_persons'
run: |
sqlx migrate run --source persons_migrations/
DATABASE_URL='postgres://posthog:posthog@localhost:5432/test_posthog' sqlx migrate run --source behavioral_cohorts_migrations/
- name: Download MaxMind Database
run: |
cd ../ && ./bin/download-mmdb
- name: Install cargo-binstall
uses: cargo-bins/cargo-binstall@5cbf019d8cb9b9d5b086218c41458ea35d817691 # main
- name: Install cargo-nextest
# --force: Swatinem/rust-cache restores ~/.cargo/.crates2.json but not the
# binary itself, so without --force binstall can short-circuit ("already
# installed") and leave no `cargo nextest` binary.
run: cargo binstall --no-confirm --force cargo-nextest@0.9.140
- name: Run cargo test
id: run-tests
# continue-on-error so the quarantine gate below is the verdict — it passes when
# every failure is an already-quarantined flake, fails otherwise.
continue-on-error: true
env:
RUST_BACKTRACE: 1
# Set up dual database environment for feature-flags service
PERSONS_READ_DATABASE_URL: ${{ contains(format(' {0} ', matrix.packages), ' feature-flags ') && 'postgres://posthog:posthog@localhost:5432/posthog_persons' || '' }}
PERSONS_WRITE_DATABASE_URL: ${{ contains(format(' {0} ', matrix.packages), ' feature-flags ') && 'postgres://posthog:posthog@localhost:5432/posthog_persons' || '' }}
run: |
# nextest (instead of plain `cargo test`) emits JUnit XML for the Trunk quarantine
# gate below. It does not run doctests; those run in their own step, outside the
# gate, because they emit no JUnit and so could never be cleared as quarantined.
# --no-tests=pass matches `cargo test` semantics for packages/filters with no tests.
# Each run overwrites target/nextest/ci/junit.xml, so stash it per package —
# even on failure, so the failing results still reach Trunk.
run_nextest() {
local junit_name=$1
shift
local rc=0
cargo nextest run --profile ci --no-tests=pass "$@" || rc=$?
if [ -f target/nextest/ci/junit.xml ]; then
mv target/nextest/ci/junit.xml "junit-nextest-$junit_name.xml"
fi
return $rc
}
# Keep testing the remaining packages after a failure: the gate masks a shard
# only from its JUnit files, so every package must contribute its results —
# an early exit would let a masked failure skip the packages behind it.
overall_rc=0
for pkg in ${{ matrix.packages }}; do
echo "::group::Testing $pkg"
# Free cgroups, overlayfs, and network namespaces from previous testcontainers
docker system prune -f --volumes 2>/dev/null || true
extra_args=""
# Limit test threads for packages with sqlx pool exhaustion issues
if [ "$pkg" = "property-defs-rs" ]; then
extra_args="--test-threads=4"
fi
if [ "$pkg" = "cohort-seeder" ] && [ -f cohort-seeder/Cargo.toml ] && grep -q '^pg-test-support[[:space:]]*=' cohort-seeder/Cargo.toml; then
features="pg-test-support"
# ch-test-support arms the live-ClickHouse rebuild corpus against the
# compose stack's clickhouse; the grep keeps branches that predate the
# feature building.
if grep -q '^ch-test-support[[:space:]]*=' cohort-seeder/Cargo.toml; then
features="pg-test-support,ch-test-support"
fi
DATABASE_URL='postgres://posthog:posthog@localhost:5432/test_posthog' run_nextest "$pkg" -p "$pkg" --features "$features" $extra_args || overall_rc=1
else
run_nextest "$pkg" -p "$pkg" $extra_args || overall_rc=1
fi
# Error-tracking rate-limiter e2e tests are #[ignore]'d (they need a real Redis
# via testcontainers/Docker, absent in some environments). This job has Docker,
# so run them here, single-threaded to avoid concurrent-container flakes. The
# `rate_limiting` filter matches nothing on branches without these tests, so it
# stays safe for unrebased PRs.
if [ "$pkg" = "cymbal" ]; then
run_nextest cymbal-ignored -p cymbal -E 'test(rate_limiting)' --run-ignored only --test-threads=1 || overall_rc=1
fi
# usage-ingestion's e2e tests are #[ignore]'d (they need Kafka and ClickHouse).
# This job has both, on the suffixed schema `setup_test_environment` created.
# Single-threaded so the load test's throughput measurement stays clean. The
# Cargo.toml guard keeps this a no-op on branches predating the crate.
if [ "$pkg" = "usage-ingestion" ] && [ -f usage-ingestion/Cargo.toml ]; then
USAGE_INGESTION_E2E_CLICKHOUSE_DATABASE=posthog_test \
USAGE_INGESTION_E2E_TOPIC=clickhouse_billing_usage_records_test \
USAGE_INGESTION_E2E_LOAD_REQUESTS=2000 \
run_nextest usage-ingestion-ignored -p usage-ingestion --run-ignored only --test-threads=1 || overall_rc=1
fi
echo "::endgroup::"
done
exit $overall_rc
# Doctests fail the job directly: they emit no JUnit, so the quarantine gate
# below can neither see nor mask them.
- name: Run doctests
env:
RUST_BACKTRACE: 1
PERSONS_READ_DATABASE_URL: ${{ contains(format(' {0} ', matrix.packages), ' feature-flags ') && 'postgres://posthog:posthog@localhost:5432/posthog_persons' || '' }}
PERSONS_WRITE_DATABASE_URL: ${{ contains(format(' {0} ', matrix.packages), ' feature-flags ') && 'postgres://posthog:posthog@localhost:5432/posthog_persons' || '' }}
run: |
for pkg in ${{ matrix.packages }}; do
# nextest skips doctests; run them only for packages with a lib target
# (`cargo test --doc` errors on bin-only packages).
if cargo metadata --no-deps --format-version 1 | jq -e --arg p "$pkg" '[.packages[] | select(.name == $p) | .targets[] | select(.kind | index("lib"))] | length > 0' >/dev/null; then
cargo test --doc -p "$pkg"
fi
done
# Best-effort Trunk upload (continue-on-error); the "Fail on test failure" step below is
# the verdict, so a Trunk outage can't red a passing job. Internal PRs only (needs the
# secret).
# TRUNK_UPLOAD_ENABLED is the master kill-switch: when it is not 'true' this gate
# skips, so nothing uploads and no known flakes are quarantined. TRUNK_QUARANTINE_ENABLED
# splits the two — it defaults off, so set it to 'true' to let the verdict step below mask
# known flakes; otherwise any real failure reds the job (quarantining needs upload on too).
- name: Quarantine gate
id: quarantine_gate
continue-on-error: true
if: ${{ !cancelled() && env.RUNS_ON_INTERNAL_PR == 'true' && github.repository == 'PostHog/posthog' && github.actor != 'dependabot[bot]' && vars.TRUNK_UPLOAD_ENABLED == 'true' }}
uses: ./.github/actions/trunk-quarantine-gate
with:
junit-paths: rust/junit-nextest-*.xml
test-collection-id: TewkWR1w # rust
previous-step-outcome: ${{ steps.run-tests.outcome }}
token: ${{ secrets.TRUNK_API_TOKEN }}
# Verdict: red a real failure the gate didn't clear as a quarantined flake. != 'success'
# also covers the skipped gate on fork/Dependabot.
- name: Fail on test failure
if: ${{ !cancelled() && steps.run-tests.outcome == 'failure' && (vars.TRUNK_QUARANTINE_ENABLED != 'true' || steps.quarantine_gate.outcome != 'success') }}
shell: bash
run: exit 1
- name: Report sccache counters
if: always()
continue-on-error: true
run: |
if command -v sccache >/dev/null; then
# The full report includes the cache endpoint, so allow only numeric counters.
sccache --show-stats 2>/dev/null | awk '/^(Compile requests|Cache hits|Cache misses|Non-cacheable calls)[[:space:]]+[0-9]+$/ { print }'
fi
# Spawns a real personhog stack (replica, writer, leaders, leader-mode
# routers) against dockerized Postgres/Kafka/etcd and asserts that every
# write acked by the leader path is visible afterwards — in strong reads
# and in Postgres at the acked version — including under chaos: leader
# kill (lease-revoked and TTL-expiry), scale-up, graceful drain, zombie
# (SIGSTOP/SIGCONT), writer crash and lag, same-identity restart,
# coordinator failover, and a mid-handoff target kill. Not a required
# check while it soaks.
personhog-gate:
name: PersonHog e2e gate
needs: [changes, affected]
# Gated on the affected computation: a determinator package-rule
# marks the harness affected whenever a personhog service it spawns
# changes, so unrelated rust PRs skip the job entirely.
if: needs.changes.outputs.rust == 'true' && contains(needs.affected.outputs.affected_crates, 'personhog-test-harness')
runs-on: depot-ubuntu-24.04-4
timeout-minutes: 25
permissions:
contents: read
env:
DOCKERHUB_USERNAME: ${{ vars.DOCKERHUB_USER }}
DOCKERHUB_TOKEN: ${{ secrets.DOCKERHUB_TOKEN }}
defaults:
run:
working-directory: rust
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
clean: false
- name: Log in to Docker Hub
continue-on-error: true
if: ${{ env.DOCKERHUB_USERNAME != '' && env.DOCKERHUB_TOKEN != '' }}
uses: docker/login-action@4907a6ddec9925e35a0a9e82d7399ccc52663121 # v4.1.0
with:
username: ${{ vars.DOCKERHUB_USER }}
password: ${{ secrets.DOCKERHUB_TOKEN }}
- name: Start Postgres, Kafka, and etcd
env:
COMPOSE_FILE: ../docker-compose.dev.yml:../docker-compose.profiles.yml
COMPOSE_PROFILES: etcd
WAIT_FOR_DOCKER_LAUNCH_RETRY_DELAY: 10
run: |
../bin/ci-wait-for-docker launch --down db kafka etcd
../bin/ci-wait-for-docker wait --only db kafka etcd
echo "127.0.0.1 db kafka" | sudo tee -a /etc/hosts
# please keep the tag version here in sync with rust-version in rust/*/Cargo.toml
- name: Install rust
uses: dtolnay/rust-toolchain@3c5f7ea28cd621ae0bf5283f0e981fb97b8a7af9
with:
toolchain: 1.91.1
- name: Install protoc
uses: ./.github/actions/setup-protoc
- name: Set sccache WebDAV key prefix (Cargo.lock hash)
run: echo "SCCACHE_WEBDAV_KEY_PREFIX=cargo-${{ hashFiles('rust/Cargo.lock') }}" >> "$GITHUB_ENV"
- name: Install sccache
uses: ./.github/actions/setup-sccache
- name: Install sqlx-cli
uses: ./.github/actions/setup-sqlx-cli
- name: Create and migrate the persons database
env:
DATABASE_URL: 'postgres://posthog:posthog@localhost:5432/posthog_persons'
run: |
sqlx database create
sqlx migrate run --source persons_migrations/
- name: Build the personhog stack and harness
run: cargo build -p personhog-replica -p personhog-router -p personhog-leader -p personhog-writer -p personhog-identity -p personhog-test-harness
# Unit tests for the harness's own verification logic — a
# false-negative verifier would pass gates that lost data.
# Ignored tests need the etcd this job already started, so
# they run here rather than in the plain test job.
- name: Test the harness verification logic
run: cargo test -p personhog-test-harness -- --include-ignored
- name: Run the e2e gate
run: |
./target/debug/personhog-test-harness gate \
--leaders 2 --partitions 4 --persons 100 --concurrency 10 --duration 10s
# The identity-service lifecycle path: persons are created
# through GetOrCreatePersonsByDistinctIds (stub insert on the
# primary, then initial properties through the router to the
# owning leader), updated through the router, and finally
# deleted through the lifecycle saga, whose outcomes the gate
# asserts (all deleted, then all not_found on a re-delete
# under a fresh op id). Create acks are journaled and held to
# the same visibility invariant as update acks, through a
# leader kill and a scale-up. The identity service writes the
# real posthog_person table (its distinct id FKs require it),
# so the whole stack targets it here instead of the default
# personhog_person_tmp.
- name: Run the e2e gate creating persons via the identity service
run: |
./target/debug/personhog-test-harness gate \
--create-via-identity --pg-target-table posthog_person \
--leaders 3 --partitions 8 --persons 100 \
--concurrency 10 --duration 15s --kill-after 5s --scale-up-after 8s
# The merge saga under load and chaos: MergePersons calls against
# live persons while blast writes race each source's fence and
# each target's fold, through a leader kill and a scale-up. Two
# sources per call exercise the fold's request order. The wide
# persons make the flip repoint a thousand mappings per source.
# The gate asserts the folded survivor document, the merge
# event's $set and $set_once, that each source is gone (not-found
# reads, a tombstone above every acked version, no mapping left
# behind), and that the delete leg answers not_found for merged
# sources.
- name: Run the e2e gate merging persons under chaos (kill + scale-up)
run: |
./target/debug/personhog-test-harness gate \
--pg-target-table posthog_person \
--leaders 3 --partitions 8 --persons 200 \
--concurrency 10 --duration 15s --merge-concurrency 4 --merge-rate 5 \
--merge-sources 2 \
--merge-wide-persons 3 --merge-wide-distinct-ids 1000 --merge-wide-role source \
--kill-after 5s --scale-up-after 8s
# Regression gate for eviction under writer lag: the cache is
# sized far below the person count so dirty entries evict while
# the paused writer guarantees their PG rows are stale. Before
# the dirty-index recovery this run lost thousands of acked
# writes.
- name: Run the e2e gate under cache eviction pressure + writer lag
run: |
./target/debug/personhog-test-harness gate \
--leaders 3 --partitions 8 --persons 50 --concurrency 10 --duration 15s \
--cache-capacity 4096 --writer-pause-after 3s --writer-pause-duration 8s
- name: Run the e2e gate under chaos (kill + scale-up)
run: |
./target/debug/personhog-test-harness gate \
--leaders 3 --partitions 8 --persons 100 --concurrency 10 --duration 15s \
--kill-after 5s --scale-up-after 8s
- name: Run the e2e gate under chaos (drain + zombie + writer lag)
run: |
./target/debug/personhog-test-harness gate \
--leaders 3 --partitions 8 --persons 100 --concurrency 10 --duration 18s \
--shutdown-after 3s --zombie-after 8s --zombie-duration 6s \
--writer-pause-after 4s --writer-pause-duration 6s
# The freeze has to outlive the lease, or the woken pod still
# holds a valid claim and nothing is ever asked. 27s is the
# lowest TTL validate_fencing_timescales accepts, so the
# freeze runs past it. Rejection is asserted by consequence:
# an unfenced landing fails the verifier.
- name: Run the e2e gate under chaos (zombie + writer lag)
run: |
./target/debug/personhog-test-harness gate \
--leaders 3 --partitions 8 --persons 100 --concurrency 10 --duration 45s \
--leader-lease-ttl 27 \
--zombie-after 5s --zombie-duration 32s \
--writer-pause-after 4s --writer-pause-duration 6s
# A fenced write's budget rules out the short TTL this used
# to run, so the run is lengthened to keep the expiry inside.
- name: Run the e2e gate under chaos (TTL-expiry kill + restarts)
run: |
./target/debug/personhog-test-harness gate \
--leaders 3 --partitions 8 --persons 100 --concurrency 10 --duration 45s \
--leader-lease-ttl 27 --kill-after 4s --kill-fast false \
--writer-crash-after 9s --restart-after 12s
- name: Run the e2e gate under chaos (coordinator failover + mid-handoff kill)
run: |
./target/debug/personhog-test-harness gate \
--routers 3 --leaders 3 --partitions 8 --persons 100 --concurrency 10 \
--duration 18s --router-kill-after 4s --shutdown-after 8s --kill-handoff-target
- name: Run the e2e gate under chaos (graceful coordinator handover + drain)
run: |
./target/debug/personhog-test-harness gate \
--routers 3 --leaders 3 --partitions 8 --persons 100 --concurrency 10 \
--duration 15s --router-shutdown-after 4s --shutdown-after 8s
# A true coordinator crash: no lease is revoked, so the survivor
# is blind until the election and registration TTLs expire, and
# the drain issued inside that window must complete once they do.
- name: Run the e2e gate under chaos (coordinator crash, slow failover + drain)
run: |
./target/debug/personhog-test-harness gate \
--routers 3 --leaders 3 --partitions 8 --persons 100 --concurrency 10 \
--duration 18s --router-kill-after 4s --router-kill-fast false --shutdown-after 8s
- name: Dump service logs on failure
if: failure()
run: |
docker ps -a || true
docker logs --tail=200 rust-kafka-1 || true
docker logs --tail=200 rust-etcd-1 || true
- name: Upload harness logs on failure
if: failure()
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6.0.0
with:
name: personhog-test-harness-logs
path: rust/target/debug/harness-logs/
if-no-files-found: ignore
- name: Report sccache counters
if: always()
continue-on-error: true
run: |
if command -v sccache >/dev/null; then
# The full report includes the cache endpoint, so allow only numeric counters.
sccache --show-stats 2>/dev/null | awk '/^(Compile requests|Cache hits|Cache misses|Non-cacheable calls)[[:space:]]+[0-9]+$/ { print }'
fi
linting:
name: Lint Rust services (depot-ubuntu-22.04-4)
needs: changes
if: needs.changes.outputs.rust == 'true'
runs-on: depot-ubuntu-22.04-4
timeout-minutes: 15
permissions:
contents: read
defaults:
run:
working-directory: rust
steps:
# Checkout project code
# Use sparse checkout to only select files in rust directory
# Turning off cone mode ensures that files in the project root are not included during checkout
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
sparse-checkout: |
rust/
proto/
.github/actions/setup-protoc/
.github/actions/setup-sccache/
sparse-checkout-cone-mode: false
clean: false
- name: Install system OpenSSL
run: |
sudo rm -f /etc/apt/sources.list.d/*twingate*
sudo apt-get update
sudo apt-get install -y libssl-dev pkg-config
# protoc and rustup both write $GITHUB_PATH, so they stay out of a
# `parallel:` block. Concurrent writes crash the GitHub Actions runner
# ("Collection was modified" or a missing-key error), and the runner
# fails the step even though the tool installed fine.
- name: Install protoc
uses: ./.github/actions/setup-protoc
- name: Install rust
uses: dtolnay/rust-toolchain@3c5f7ea28cd621ae0bf5283f0e981fb97b8a7af9
with:
toolchain: 1.91.1
components: clippy,rustfmt
- name: Set sccache WebDAV key prefix (Cargo.lock hash)
run: echo "SCCACHE_WEBDAV_KEY_PREFIX=cargo-${{ hashFiles('rust/Cargo.lock') }}" >> "$GITHUB_ENV"
- name: Install sccache
uses: ./.github/actions/setup-sccache
- name: Cache Rust dependencies
uses: Swatinem/rust-cache@779680da715d629ac1d338a641029a2f4372abb5 # v2.8.2
with:
shared-key: 'v2-rust-ci'
workspaces: rust
save-if: ${{ github.ref == 'refs/heads/master' }}
- name: Check format
run: cargo fmt -- --check
- name: Run clippy
env:
# Use system OpenSSL instead of vendored to avoid assembly build issue (2026-01-13)
OPENSSL_NO_VENDOR: '1'
run: cargo clippy --all-targets --all-features -- -D warnings
- name: Run cargo check
env:
# Use system OpenSSL instead of vendored to avoid assembly build issue (2026-01-13)
OPENSSL_NO_VENDOR: '1'
run: cargo check --all-features
- name: Install cargo-binstall
uses: cargo-bins/cargo-binstall@5cbf019d8cb9b9d5b086218c41458ea35d817691 # main
- name: Install cargo-shear
# --force is required: Swatinem/rust-cache restores ~/.cargo/.crates2.json
# (which records cargo-shear as installed) but not the binary itself, so
# without --force binstall short-circuits ("already installed") and the
# next step fails with `error: no such command: shear`.
run: cargo binstall --no-confirm --force cargo-shear@1.1.12
- name: Run cargo shear
run: cargo shear
- name: Report sccache counters
if: always()
continue-on-error: true
run: |
if command -v sccache >/dev/null; then
# The full report includes the cache endpoint, so allow only numeric counters.
sccache --show-stats 2>/dev/null | awk '/^(Compile requests|Cache hits|Cache misses|Non-cacheable calls)[[:space:]]+[0-9]+$/ { print }'
fi
# Collate job - single required status check for branch protection.
# Individual jobs skip entirely when no Rust changes, saving runner time.
rust_tests:
needs: [changes, affected, build, test, linting]
name: Rust Tests Pass
runs-on: ubuntu-24.04
timeout-minutes: 5
if: ${{ !cancelled() }}
permissions: {}
steps:
- name: Check all Rust jobs
run: |
# Fail if change detection failed, affected computation failed, or either was cancelled (outputs are empty on cancel, so the exit below would report green).
if [[ "${{ needs.changes.result }}" != "success" && "${{ needs.changes.result }}" != "skipped" ]]; then
echo "Change detection did not succeed (result: ${{ needs.changes.result }})."
exit 1
fi
# Pass if no Rust changes detected (jobs were skipped)
if [[ "${{ needs.changes.outputs.rust }}" != "true" ]]; then
echo "Rust checks were skipped (no relevant changes)"
exit 0
fi
if [[ "${{ needs.affected.result }}" != "success" && "${{ needs.affected.result }}" != "skipped" ]]; then
echo "Affected crate computation did not succeed (result: ${{ needs.affected.result }})."
exit 1
fi
if [[ "${{ needs.build.result }}" != "success" && "${{ needs.build.result }}" != "skipped" ]]; then
echo "Rust build failed."
exit 1
fi
if [[ "${{ needs.test.result }}" != "success" && "${{ needs.test.result }}" != "skipped" ]]; then
echo "Rust tests failed."
exit 1
fi
if [[ "${{ needs.linting.result }}" != "success" && "${{ needs.linting.result }}" != "skipped" ]]; then
echo "Rust linting failed."
exit 1
fi
echo "All Rust checks passed."