perf(epa_ppls): make the monthly refresh fit the runner's 3 h budget (#3)
Image rebuild (skip scrape) / build (push) Successful in 1h46m17s

Co-authored-by: claude <[email protected]>
This commit was merged in pull request #3.
This commit is contained in:
2026-09-01 16:40:48 -04:00
committed by claude
parent 0f296a0ec4
commit 98842d1ed6
316 changed files with 129846 additions and 84529 deletions
+29 -11
View File
@@ -5,11 +5,13 @@ name: Monthly corpus refresh
# reindex + image-push if the scrape produced no diff against the
# committed corpus.
#
# Bayer takes ~30 min; EPA PPLS takes ~7 h with row-crop +
# registrant filters. The whole monthly job is ~8-9 h end-to-end.
# If that's too long for the runner you can:
# - Run just one source: workflow_dispatch with sources="bayer"
# - Limit EPA at the scraper: edit the step to add "--limit 5000"
# Runtime budget: act_runner kills the job container at exactly 3 h
# (it runs as `/bin/sleep 10800`), which is what killed every run from
# 2026-06-01 through 2026-09-01. The EPA step is incremental and
# parallel so a steady-state refresh is ~15-20 min: cached not-row-crop
# verdicts cost no request, and a product already on disk is only
# re-downloaded when EPA reports a new label date. `timeout-minutes`
# below fails the job legibly before the container disappears.
on:
schedule:
@@ -42,6 +44,9 @@ env:
jobs:
refresh:
runs-on: docker
# Below act_runner's own 3 h container lifetime, so an overrun fails
# as a timeout instead of "container ... does not exist".
timeout-minutes: 170
container:
image: catthehacker/ubuntu:act-latest
steps:
@@ -80,9 +85,14 @@ jobs:
- name: Scrape EPA PPLS
if: ${{ inputs.sources == '' || contains(inputs.sources, 'epa_ppls') }}
# Row-crop + registrant filters keep this to ~16K PDFs / ~7h.
# Pass --no-row-crop-filter or --no-registrant-filter to broaden.
run: python -m scrape.runner --source epa_ppls --force
# Deliberately NOT --force: that re-downloaded all ~11.4K candidate
# registrations every month (~15-20 h of work) and never finished.
# Without it the run is incremental — one cheap API call per product,
# a PDF only when the label date actually moved — and the committed
# filter cache skips the ~7.3K non-row-crop products outright.
# Workers share one 5 req/sec ceiling, so this is faster without
# being ruder. Local full re-fetch: add --force --no-filter-cache.
run: python -m scrape.runner --source epa_ppls --workers 6
# ---- Commit corpus changes + retry-on-race -----------------
- name: Commit corpus changes (if any)
@@ -90,13 +100,21 @@ jobs:
run: |
git config user.name "crop-chem-docs-refresh"
git config user.email "[email protected]"
git add sources.json corpus
if git diff --cached --quiet; then
# The filter-verdict cache is committed so next month starts warm,
# but it changes on every run — only a real corpus diff may trigger
# the reindex + image build, so `changed` is decided on the corpus
# paths alone.
git add sources.json corpus scrape/state
if git diff --cached --quiet -- sources.json corpus; then
echo "no corpus changes — skipping reindex and image build"
echo "changed=false" >> "$GITHUB_OUTPUT"
else
echo "changed=true" >> "$GITHUB_OUTPUT"
fi
if git diff --cached --quiet; then
echo "nothing staged — no commit"
exit 0
fi
echo "changed=true" >> "$GITHUB_OUTPUT"
ts=$(date -u +"%Y-%m-%dT%H:%MZ")
n_bayer=$(find corpus/bayer -name '*.json' 2>/dev/null | wc -l)
n_epa=$(find corpus/epa_ppls -name '*.json' 2>/dev/null | wc -l)