Skip to content

Crawl Registry

Crawl Registry #17638

Workflow file for this run

name: Crawl Registry
concurrency:
group: crawl-pipeline
cancel-in-progress: false
on:
schedule:
- cron: '*/15 * * * *'
workflow_dispatch:
push:
branches:
- main
- package_control_channel
permissions:
contents: write
jobs:
crawl:
runs-on: ubuntu-latest
env:
RELEASE_TAG: crawler-status
GH_TOKEN: ${{ github.token }}
GITHUB_TOKEN: ${{ github.token }}
GITLAB_TOKEN: ${{ secrets.GITLAB_TOKEN }}
PRESTO_PRESTO_CRAWL: ${{ vars.PRESTO_PRESTO_CRAWL }}
steps:
- name: Checkout repository
uses: actions/checkout@v6
- name: Checkout registry worktree
run: |
git fetch origin the-registry
git worktree add --force -B the-registry ./.the-registry origin/the-registry
- name: Set up Python
uses: actions/setup-python@v6
with:
python-version: '3.13'
- name: Install uv
run: pip install uv
- name: Ensure wrk directory exists
run: mkdir -p ./wrk
# --------------------------------------------------------------------
# Freeze one run-level timestamp for the entire crawl job.
#
# Why:
# - We want all artifacts and logs from a single workflow run to agree
# on one exact point in time.
# - This avoids subtle drift where separate `date` calls differ by
# seconds and make later analysis harder.
#
# Consumers of this frozen timestamp:
# - scripts/crawl.py (run timestamp for crawl/update detection)
# - scripts/collect_logs.py (log entry timestamp fallback via NOW_TS)
#
# Notes:
# - We export via $GITHUB_ENV so NOW_TS is available to subsequent steps
# in this job.
# - Use epoch seconds (`date +%s`) to stay timezone-agnostic.
# --------------------------------------------------------------------
- name: Freeze run timestamp
run: echo "NOW_TS=$(date +%s)" >> "$GITHUB_ENV"
- name: Restore wrk cache
uses: actions/cache@v5
with:
path: |
./wrk
./.pypi-cache
key: wrk-cache-${{ github.run_id }}
fail-on-cache-miss: true
restore-keys: |
wrk-cache-
- name: Sync Python environment
run: uv sync --locked
- name: Generate registry
run: |
set -o pipefail
PYTHONUNBUFFERED=1 uv run -m scripts.generate_registry \
--seed ./.the-registry/registry.json \
-o ./wrk/registry.json \
2> >(tee registry.log >&2)
- name: Sync registry branch
run: bash ./.github/workflows/sync_registry_branch.sh ./.the-registry/ ./wrk/registry.json
- name: Run crawler
run: |
set -o pipefail
PYTHONUNBUFFERED=1 uv run -m scripts.crawl --limit 1500 \
--registry ./wrk/registry.json \
--workspace ./wrk/workspace.json \
--fetch-readmes ./wrk/readmes.json \
2>&1 | tee crawl.log
- name: Check if we should resolve the libraries
id: resolve_window
run: |
# ensure the file is there in case we actually don't run
touch lib.log
DATE=$(TZ=Europe/Berlin date +%Y-%m-%d)
HOUR=$(TZ=Europe/Berlin date +%H)
if [ "$HOUR" -ge 7 ] && [ "$HOUR" -lt 18 ]; then
RUN_STAMP="$DATE morning"
elif [ "$HOUR" -ge 18 ]; then
RUN_STAMP="$DATE evening"
else
RUN_STAMP="$DATE initial"
OUTSIDE_LIBRARY_WINDOW=true
fi
# ensure we run initially
if [ ! -f ./wrk/last-lib-run.txt ]; then
echo "run=true" >> $GITHUB_OUTPUT
echo "stamp=$RUN_STAMP" >> $GITHUB_OUTPUT
exit 0
fi
# otherwise, run once per Berlin-time library window
if [ "${OUTSIDE_LIBRARY_WINDOW:-false}" = true ]; then
echo "run=false" >> $GITHUB_OUTPUT
exit 0
fi
if grep -Fxq "$RUN_STAMP" ./wrk/last-lib-run.txt; then
echo "run=false" >> $GITHUB_OUTPUT
exit 0
fi
echo "run=true" >> $GITHUB_OUTPUT
echo "stamp=$RUN_STAMP" >> $GITHUB_OUTPUT
- name: Resolve libraries
if: steps.resolve_window.outputs.run == 'true'
run: |
set -o pipefail
PYTHONUNBUFFERED=1 uv run -m scripts.crawl_libraries \
--registry ./wrk/registry.json \
--workspace ./wrk/workspace.json \
--allowed-source https://raw.githubusercontent.com/packagecontrol/channel/refs/heads/main/repository.json \
--limit 150 \
1> >(tee lib.log)
echo "${{ steps.resolve_window.outputs.stamp }}" > ./wrk/last-lib-run.txt
- name: Generate channel
run: |
set -o pipefail
PYTHONUNBUFFERED=1 uv run -m scripts.generate_channel \
--registry ./wrk/registry.json \
--workspace ./wrk/workspace.json \
--berlin \
-o ./wrk/channel.json \
2>&1 | tee channel.log
- name: Update release notes
run: |
# Create or update the release
gh release view ${{ env.RELEASE_TAG }} > /dev/null || \
gh release create ${{ env.RELEASE_TAG }} \
--title "The Crawler Logs" \
--notes "..." \
--latest=false
echo "Uploading working dir files..."
gh release upload ${{ env.RELEASE_TAG }} ./wrk/channel.json --clobber
gh release upload ${{ env.RELEASE_TAG }} ./wrk/registry.json --clobber
gh release upload ${{ env.RELEASE_TAG }} ./wrk/workspace.json --clobber
gh release upload ${{ env.RELEASE_TAG }} ./wrk/readmes.json --clobber
DATE=$(TZ=Europe/Berlin date -d "@$NOW_TS" +"%B %d, %Y, %H:%M GMT%:::z" | sed -E 's/([+-])0/\1/')
REPO_URL="https://github.com/${{ github.repository }}/actions/runs/${{ github.run_id }}"
# Build new notes
{
echo "$DATE ([logs]($REPO_URL))"
echo ""
if [ -s registry.log ]; then
cat registry.log
echo ""
fi
# Insert a blank line before lines that are exactly '---'
awk '{ if ($0 == "---") print ""; print }' crawl.log
if [ -s lib.log ]; then
echo ""
echo "---"
cat lib.log
fi
echo ""
echo "---"
cat channel.log
} > notes.txt
echo "Updating release notes..."
gh release edit ${{ env.RELEASE_TAG }} --notes-file notes.txt
echo "Updating logs history..."
ASSET_NAMES="$(gh release view ${{ env.RELEASE_TAG }} --json assets --jq '.assets[].name')"
if printf '%s\n' "$ASSET_NAMES" | grep -Fxq "logs.json"; then
gh release download ${{ env.RELEASE_TAG }} \
--pattern "logs.json" \
--output ./wrk/logs.json \
--clobber
else
echo "No logs.json asset yet; starting fresh."
fi
uv run -m scripts.collect_logs \
--run-id "${{ github.run_id }}" \
--workspace ./wrk/workspace.json \
-o ./wrk/logs.json \
notes.txt
- name: Upload wrk backup
id: crawl-backup-step
uses: actions/upload-artifact@v7
with:
name: crawl-backup
path: wrk/
retention-days: 30
fetch_stats:
runs-on: ubuntu-latest
env:
RELEASE_TAG: crawler-status
GH_TOKEN: ${{ github.token }}
GITHUB_TOKEN: ${{ github.token }}
steps:
- name: Checkout repository
uses: actions/checkout@v6
- name: Set up Python
uses: actions/setup-python@v6
with:
python-version: '3.13'
- name: Install uv
run: pip install uv
- name: Ensure wrk directory exists
run: mkdir -p ./wrk
- name: Restore wrk cache
uses: actions/cache@v5
with:
path: ./wrk
key: stats-cache-${{ github.run_id }}
fail-on-cache-miss: true
restore-keys: |
stats-cache-
- name: Run script
run: uv run -m scripts.accumulate_stats --wd ./wrk
- name: Update stats
run: |
echo "Upload stats..."
gh release upload ${{ env.RELEASE_TAG }} ./wrk/stats.json --clobber
- name: Upload wrk backup
id: stats-backup-step
uses: actions/upload-artifact@v7
with:
name: stats-backup
path: wrk/
retention-days: 30