fix(scrape): make vision OCR a per-bundle allowlist, not a corpus sweep (#6)
Co-authored-by: claude <[email protected]>
This commit was merged in pull request #6.
This commit is contained in:
@@ -0,0 +1,171 @@
|
||||
"""Build a human-inspectable report of every vision-OCR'd image.
|
||||
|
||||
Reads corpus/.vision-cache/*.json (written by scrape/vision.py) and emits a
|
||||
self-contained HTML page that puts each SOURCE IMAGE next to its OCR
|
||||
transcription, so a human can eyeball accuracy. Uncertain transcriptions
|
||||
(the two OCR samples disagreed) sort to the top. Images are downscaled and
|
||||
inlined as data URIs so the page is fully self-contained (no external
|
||||
requests) and the original full-res image is one click away.
|
||||
|
||||
Usage:
|
||||
python -m scripts.vision_report # -> vision_report.html
|
||||
python -m scripts.vision_report --out /tmp/r.html
|
||||
python -m scripts.vision_report --no-embed # link images, don't inline (tiny file)
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import base64
|
||||
import glob
|
||||
import html
|
||||
import io
|
||||
import json
|
||||
import re
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import requests
|
||||
|
||||
ROOT = Path(__file__).resolve().parent.parent
|
||||
CACHE_DIR = ROOT / "corpus" / ".vision-cache"
|
||||
CORPUS = ROOT / "corpus"
|
||||
UA = "morpheus-docs-mcp/0.1 (+https://git.jpaul.io/justin/morpheus-docs)"
|
||||
MAX_W = 1100 # downscale width for the inlined preview
|
||||
|
||||
|
||||
def _basename(url: str) -> str:
|
||||
return re.sub(r"\?.*$", "", url.rsplit("/", 1)[-1])
|
||||
|
||||
|
||||
def _pages_for(basename: str) -> list[str]:
|
||||
"""Which corpus pages reference this image (by filename), as bundle/page."""
|
||||
hits: list[str] = []
|
||||
for md in CORPUS.glob("*/*.md"):
|
||||
try:
|
||||
if basename in md.read_text():
|
||||
hits.append(f"{md.parent.name}/{md.stem}")
|
||||
except Exception:
|
||||
continue
|
||||
return sorted(hits)
|
||||
|
||||
|
||||
def _thumb(url: str, embed: bool) -> str | None:
|
||||
if not embed:
|
||||
return None
|
||||
try:
|
||||
data = requests.get(url, headers={"User-Agent": UA}, timeout=60).content
|
||||
from PIL import Image
|
||||
im = Image.open(io.BytesIO(data))
|
||||
if im.mode not in ("RGB", "L"):
|
||||
im = im.convert("RGB")
|
||||
if im.width > MAX_W:
|
||||
im = im.resize((MAX_W, round(im.height * MAX_W / im.width)), Image.LANCZOS)
|
||||
buf = io.BytesIO()
|
||||
im.save(buf, format="JPEG", quality=82)
|
||||
return "data:image/jpeg;base64," + base64.b64encode(buf.getvalue()).decode()
|
||||
except Exception as e:
|
||||
print(f" ! thumb failed for {url[:70]}: {e}", file=sys.stderr)
|
||||
return None
|
||||
|
||||
|
||||
def _md_table_to_html(md: str) -> str:
|
||||
"""Render a markdown table as an HTML table; fall back to <pre> otherwise."""
|
||||
rows = [ln for ln in md.splitlines() if ln.strip().startswith("|")]
|
||||
if len(rows) < 2:
|
||||
return f"<pre>{html.escape(md)}</pre>"
|
||||
def cells(line: str) -> list[str]:
|
||||
return [c.strip() for c in line.strip().strip("|").split("|")]
|
||||
head = cells(rows[0])
|
||||
body = [cells(r) for r in rows[2:]] # rows[1] is the |---| separator
|
||||
out = ["<table><thead><tr>"] + [f"<th>{html.escape(c)}</th>" for c in head] + ["</tr></thead><tbody>"]
|
||||
for r in body:
|
||||
out.append("<tr>" + "".join(f"<td>{html.escape(c)}</td>" for c in r) + "</tr>")
|
||||
out.append("</tbody></table>")
|
||||
return "".join(out)
|
||||
|
||||
|
||||
def build(embed: bool) -> str:
|
||||
entries = []
|
||||
for f in sorted(glob.glob(str(CACHE_DIR / "*.json"))):
|
||||
try:
|
||||
entries.append(json.load(open(f)))
|
||||
except Exception:
|
||||
continue
|
||||
withmd = [e for e in entries if e.get("markdown")]
|
||||
# uncertain first, then by source URL for stable ordering
|
||||
withmd.sort(key=lambda e: (bool(e.get("certain")), e.get("src", "")))
|
||||
|
||||
n_uncertain = sum(1 for e in withmd if not e.get("certain"))
|
||||
cards = []
|
||||
for i, e in enumerate(withmd, 1):
|
||||
src = e.get("src", "")
|
||||
bn = _basename(src)
|
||||
pages = _pages_for(bn)
|
||||
thumb = _thumb(src, embed)
|
||||
badge = ("<span class='ok'>certain</span>" if e.get("certain")
|
||||
else "<span class='warn'>UNCERTAIN — samples disagreed</span>")
|
||||
img_html = (f"<img src='{thumb}' alt='source image'>" if thumb
|
||||
else f"<div class='noimg'>[not inlined — open original]</div>")
|
||||
pages_html = " ".join(f"<code>{html.escape(p)}</code>" for p in pages) or "<em>(none found)</em>"
|
||||
cards.append(f"""
|
||||
<section class='card'>
|
||||
<div class='meta'>
|
||||
<span class='num'>#{i}</span> {badge}
|
||||
<span class='model'>{html.escape(e.get('model',''))}</span>
|
||||
<div class='pages'>pages: {pages_html}</div>
|
||||
<a class='orig' href='{html.escape(src)}' target='_blank' rel='noopener'>open original image ↗</a>
|
||||
</div>
|
||||
<div class='pair'>
|
||||
<div class='src'>{img_html}</div>
|
||||
<div class='ocr'>{_md_table_to_html(e.get('markdown',''))}</div>
|
||||
</div>
|
||||
</section>""")
|
||||
|
||||
style = """
|
||||
:root{color-scheme:light dark}
|
||||
body{font:14px/1.5 system-ui,sans-serif;margin:0;padding:24px;max-width:1400px;margin:auto}
|
||||
h1{font-size:20px} .summary{color:#666;margin-bottom:20px}
|
||||
.card{border:1px solid #8883;border-radius:10px;padding:16px;margin:0 0 20px}
|
||||
.meta{display:flex;flex-wrap:wrap;gap:10px;align-items:center;margin-bottom:12px;font-size:13px}
|
||||
.num{font-weight:700}
|
||||
.ok{background:#1a7f37;color:#fff;padding:1px 8px;border-radius:10px;font-size:12px}
|
||||
.warn{background:#b34700;color:#fff;padding:1px 8px;border-radius:10px;font-size:12px}
|
||||
.model{color:#888;font-family:monospace}
|
||||
.pages{flex-basis:100%;color:#777} .pages code{font-size:12px}
|
||||
.orig{margin-left:auto}
|
||||
.pair{display:grid;grid-template-columns:1fr 1fr;gap:16px;align-items:start}
|
||||
@media(max-width:900px){.pair{grid-template-columns:1fr}}
|
||||
.src img{max-width:100%;border:1px solid #8884;border-radius:6px}
|
||||
.noimg{color:#999;padding:40px;text-align:center;border:1px dashed #8884;border-radius:6px}
|
||||
table{border-collapse:collapse;font-size:12px;width:100%}
|
||||
th,td{border:1px solid #8884;padding:3px 7px;text-align:left}
|
||||
th{background:#8881}
|
||||
pre{white-space:pre-wrap;font-size:12px;background:#8881;padding:10px;border-radius:6px}
|
||||
"""
|
||||
return f"""<!doctype html><html><head><meta charset='utf-8'>
|
||||
<meta name='viewport' content='width=device-width,initial-scale=1'>
|
||||
<title>Vision OCR inspection report</title><style>{style}</style></head><body>
|
||||
<h1>Vision OCR inspection report</h1>
|
||||
<div class='summary'>{len(withmd)} transcribed image(s) — {n_uncertain} uncertain, {len(withmd)-n_uncertain} certain.
|
||||
Each source image (downscaled preview; click “open original” for full-res) is shown next to its OCR transcription.
|
||||
Uncertain items are listed first.</div>
|
||||
{''.join(cards)}
|
||||
</body></html>"""
|
||||
|
||||
|
||||
def main() -> int:
|
||||
p = argparse.ArgumentParser(description="Build the vision-OCR inspection report.")
|
||||
p.add_argument("--out", default=str(ROOT / "vision_report.html"))
|
||||
p.add_argument("--no-embed", action="store_true", help="link images instead of inlining them")
|
||||
args = p.parse_args()
|
||||
if not CACHE_DIR.exists():
|
||||
print(f"no cache dir at {CACHE_DIR} — run the scraper with VISION_OCR=1 first", file=sys.stderr)
|
||||
return 1
|
||||
html_doc = build(embed=not args.no_embed)
|
||||
Path(args.out).write_text(html_doc)
|
||||
print(f"wrote {args.out}", file=sys.stderr)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user