feat(scrape): vision OCR of image-only doc pages (support matrices) (#5)

Co-authored-by: claude <[email protected]>
This commit was merged in pull request #5.
This commit is contained in:
2026-07-23 09:48:38 -04:00
committed by claude
parent caf562b7c3
commit 698196bd63
10 changed files with 407 additions and 6 deletions
+15
View File
@@ -37,6 +37,18 @@ env:
OLLAMA_URLS: http://192.168.0.2:11435,http://192.168.0.2:11436,http://192.168.0.125:11434,http://192.168.0.126:11434
EMBED_MODEL: nomic-embed-text
# Vision OCR of image-only pages (support/lifecycle matrices etc. — see
# scrape/vision.py). Uses the host's primary Ollama on :11434, the only one
# with a vision model; the embed pool above is nomic-only. qwen2.5vl:7b
# scored best on the release-schedule matrix (55-56/56, ~18s, GPU-resident,
# prompt-robust). Content-hash cached in corpus/.vision-cache/; VISION_MAX_NEW
# bounds NEW OCRs per run so the cache fills incrementally instead of a
# multi-hour first pass. Any failure degrades to the pre-vision behavior.
VISION_OCR: "1"
VISION_URL: http://192.168.0.2:11434
VISION_MODEL: qwen2.5vl:7b
VISION_MAX_NEW: "60"
PRODUCT_NAME: morpheus
jobs:
@@ -64,6 +76,9 @@ jobs:
run: |
python -m pip install -q --upgrade pip
python -m pip install -q -r requirements.txt
# Vision OCR deps (Pillow) — only the scrape step needs these; kept
# out of requirements.txt so they never bloat the server image.
python -m pip install -q -r requirements-vision.txt
# ---- Phase 1: scrape ---------------------------------------
- name: Refresh bundle catalog