Files
morpheus-docs/scrape/bundles.py
T
claudeandClaude Opus 4.8 1d3e52a5ed feat(scrape): vision OCR of image-only doc pages (support matrices)
Some HPE pages carry information only inside an image. The Morpheus
release schedule / support-lifecycle matrix (sf000111242en_us) is the
case that surfaced this: the entire "which version ships when, when a
stream hits Maintenance/EOL" table is a JPEG, so html_to_md dropped
every value and retrieval could never answer "when does 8.x.x reach
end of life".

New scrape/vision.py (opt-in via VISION_OCR) transcribes qualifying
images with a local Ollama vision model and appends a labeled markdown
block to the page so the text is chunked, embedded and retrieved.

Model/prompt chosen against ground truth on the matrix (14x4 cells):
- qwen2.5vl:7b + a bare "transcribe the table" prompt: 55-56/56, ~18s,
  GPU-resident, and prompt-ROBUST (accurate without hand-tuning).
- gemma3:12b needs an exact "read row by row, keep cells aligned"
  prompt or it shifts a column by one row; qwen2.5vl:32b is no more
  accurate and 5x slower (CPU spill). Both documented in the code.
- Note: self-consistency (2 samples must agree) only catches RANDOM
  flakiness — a wrong read is stable across seeds — so every block also
  ships a "verify against the source image" caveat.

Reliability/cost:
- content-hash cache in corpus/.vision-cache/ (committed) — each unique
  image OCR'd once ever, shared across version bundles.
- VISION_MAX_NEW bounds NEW OCRs per run so the first pass can't balloon
  into hours; the cache fills incrementally. Deferred count is logged.
- every failure path degrades to the pre-vision behavior; never blocks
  the scrape.

Also:
- add the morpheus_release_schedule bundle (sf single-doc) + 2 eval
  golden queries for it.
- fetch_single_doc: title falls back to the bundle title (not docId)
  when a page has no <h1> (sf solution articles).
- Pillow in requirements-vision.txt (scrape-only; kept out of the
  server image), installed + VISION_* wired into refresh.yml.

Verified locally: matrix transcribes to the exact 14-row table; second
run is a cache hit (ocr=0).

Co-Authored-By: Claude Opus 4.8 <[email protected]>
Claude-Session: https://claude.ai/code/session_01LFowQzJu7k97QLCRDSAeh1
2026-07-23 09:35:00 -04:00

327 lines
14 KiB
Python

"""Discover Morpheus Enterprise doc bundles on HPE Support DocPortal and write bundles.json.
Mirrors hvm-docs/scrape/bundles.py — same portal, same API shape, same single-doc-blob
treatment for Release Notes, but pointing at the Morpheus Enterprise docId range.
For each bundle this script:
1. GETs /hpesc/public/api/document/{docId} → abstract HTML
2. GETs /hpesc/public/api/document/{docId}/toc → page tree (or 404 for single-doc)
3. Writes bundles.json at repo root with the schema PLAN.md Phase 1 documents.
QuickSpecs is a special case: lives at www.hpe.com (not support.hpe.com), gets the
html-file mode and is scraped via curl_cffi (see scrape/quickspecs.py).
"""
from __future__ import annotations
import argparse
import json
import os
import re
import sys
import time
from concurrent.futures import ThreadPoolExecutor
from dataclasses import dataclass, field
from pathlib import Path
from typing import Any
import requests
from bs4 import BeautifulSoup
API = "https://support.hpe.com/hpesc/public/api/document"
DOC_URL = "https://support.hpe.com/hpesc/public/docDisplay?docId={doc_id}"
UA = "morpheus-docs-mcp/0.1 (+https://git.jpaul.io/justin/morpheus-docs; [email protected])"
ROOT = Path(__file__).resolve().parent.parent
BUNDLES_JSON = ROOT / "bundles.json"
# --- Auto-discovery of new versions -------------------------------------
# The pinned BUNDLES list below is a known-good baseline. On top of it we
# sweep the HPE docId range for NEWLY published Morpheus Enterprise User
# Manuals / Release Notes (e.g. a 9.0.1 that lands after this file was last
# edited) so the weekly refresh picks them up without a code change. New
# versions always get higher docIds, so we scan from just below the lowest
# pinned bundle up to the highest pinned docId plus a forward window.
#
# Set DISCOVER_VERSIONS=0 to disable (reproducible/offline runs). Widen
# DISCOVER_WINDOW if HPE publishes a new version far beyond the newest pin.
DISCOVER = os.environ.get("DISCOVER_VERSIONS", "1") != "0"
DISCOVER_WINDOW = int(os.environ.get("DISCOVER_WINDOW", "500"))
DISCOVER_WORKERS = int(os.environ.get("DISCOVER_WORKERS", "12"))
_DOCID_NUM_RE = re.compile(r"sd0*(\d+)en_us")
_VERSION_RE = re.compile(r"v?(\d+\.\d+\.\d+)")
@dataclass
class BundleSpec:
slug: str
doc_id: str
title: str
version: str | None
product: str # e.g. "User Manual", "Release Notes", "QuickSpecs"
mode: str # "toc", "single", or "html-file"
platform: str | None = None
language: str = "en-US"
source_url: str | None = None # overrides the default support.hpe.com URL
# Pinned baseline bundles. docIds confirmed by probing the portal for
# `Morpheus Enterprise` matches in the abstract (8.1.x on 2026-05-22,
# 9.0.0 on 2026-07-15). This list is a known-good floor; _discover_enterprise()
# sweeps forward from here each run and appends any NEWER versions HPE has
# published, so a fresh 9.0.1 lands in the corpus without editing this file.
#
# Notes:
# - Morpheus Enterprise has User Manuals dating back to 8.0.10
# (sd00006774en_us, Sep 2025) but we ship the 8.1.x + 9.0.0 line.
# Add the 8.0.x bundles here if you need older versions in the corpus
# (they predate the discovery floor, so they won't auto-appear).
# - No dedicated Deployment Guide or Qualification Matrix for Morpheus
# Enterprise on HPE Support — the only QM (sd00006551en_us) covers
# HVM clusters managed by Morpheus, which lives in hvm-docs.
# - QuickSpecs lives on www.hpe.com (not support.hpe.com), uses the
# html-file scrape mode with curl_cffi Chrome impersonation.
BUNDLES: list[BundleSpec] = [
BundleSpec("morpheus_user_manual_8_1_0", "sd00007510en_us", "HPE Morpheus Enterprise Software Documentation", "8.1.0", "User Manual", "toc"),
BundleSpec("morpheus_user_manual_8_1_1", "sd00007621en_us", "HPE Morpheus Enterprise Software Documentation", "8.1.1", "User Manual", "toc"),
BundleSpec("morpheus_user_manual_8_1_2", "sd00007732en_us", "HPE Morpheus Enterprise Software Documentation", "8.1.2", "User Manual", "toc"),
BundleSpec("morpheus_release_notes_8_1_0", "sd00007496en_us", "HPE Morpheus Enterprise Software Release Notes", "8.1.0", "Release Notes", "single"),
BundleSpec("morpheus_release_notes_8_1_1", "sd00007610en_us", "HPE Morpheus Enterprise Software Release Notes", "8.1.1", "Release Notes", "single"),
BundleSpec("morpheus_release_notes_8_1_2", "sd00007733en_us", "HPE Morpheus Enterprise Software Release Notes", "8.1.2", "Release Notes", "single"),
# 9.0.0 (May 2026). Confirmed 2026-07-15 by probing the docId range;
# also auto-discovered by _discover_enterprise() below, but pinned here
# so it ships even if discovery is disabled or the portal is down.
BundleSpec("morpheus_user_manual_9_0_0", "sd00008014en_us", "HPE Morpheus Enterprise Software Documentation", "9.0.0", "User Manual", "toc"),
BundleSpec("morpheus_release_notes_9_0_0", "sd00008017en_us", "HPE Morpheus Enterprise Software Release Notes", "9.0.0", "Release Notes", "single"),
BundleSpec("morpheus_quickspecs", "a50009231enw", "HPE Morpheus Enterprise Software QuickSpecs",
"v1", "QuickSpecs", "html-file",
source_url="https://www.hpe.com/psnow/doc/a50009231enw"),
# Release schedule / support-lifecycle matrix. This `sf` solution article
# is a single page whose table (which version ships when; when a stream
# hits Maintenance/EOL) lives ENTIRELY inside a JPEG. Scraped as `single`;
# the matrix text comes from the VISION_OCR pass (see scrape/vision.py).
# version=None — it spans all major streams, not one release.
BundleSpec("morpheus_release_schedule", "sf000111242en_us", "HPE Morpheus Software Release Schedule",
None, "Release Schedule", "single"),
]
def _session() -> requests.Session:
s = requests.Session()
s.headers.update({"User-Agent": UA, "Accept": "application/json, text/html"})
return s
def _get(s: requests.Session, url: str, expect_json: bool = False, retries: int = 4) -> Any:
delay = 1.0
for attempt in range(retries):
r = s.get(url, timeout=30)
if r.status_code == 200:
return r.json() if expect_json else r.text
if r.status_code == 404:
return None
if r.status_code in (429, 500, 502, 503, 504):
time.sleep(delay)
delay *= 2
continue
r.raise_for_status()
raise RuntimeError(f"GET failed after {retries} retries: {url}")
def _count_toc(toc: list[dict] | None) -> tuple[int, str | None]:
if not toc:
return 0, None
landing = None
n = 0
def walk(nodes: list[dict] | None, depth: int) -> None:
nonlocal n, landing
for node in nodes or []:
link = node.get("topicLink")
if link:
n += 1
m = re.search(r"page=(GUID-[A-F0-9-]+)\.html", link)
if m and landing is None:
landing = m.group(1)
walk(node.get("children"), depth + 1)
walk(toc, 0)
return n, landing
def _parse_abstract(html: str) -> dict[str, str]:
soup = BeautifulSoup(html, "html.parser")
out: dict[str, str] = {}
h1 = soup.select_one("h1.title.topictitle1")
if h1:
out["title"] = h1.get_text(" ", strip=True)
desc = soup.select_one("div.desc")
if desc:
out["abstract"] = desc.get_text(" ", strip=True)
pub = soup.select_one("div.publishedDate")
if pub:
out["published"] = pub.get_text(" ", strip=True).replace("Published:", "").strip()
return out
def discover_bundle(s: requests.Session, spec: BundleSpec) -> dict[str, Any]:
# html-file bundles are static fixtures or live-fetched outside support.hpe.com.
if spec.mode == "html-file":
return {
"slug": spec.slug,
"doc_id": spec.doc_id,
"title": spec.title,
"version": spec.version,
"platform": spec.platform,
"product": spec.product,
"language": spec.language,
"page_count": 1,
"mode": "html-file",
"abstract": "",
"dates": {},
"landing_page": spec.doc_id,
"source_url": spec.source_url or f"https://www.hpe.com/psnow/doc/{spec.doc_id}",
}
abstract_html = _get(s, f"{API}/{spec.doc_id}", expect_json=False)
meta = _parse_abstract(abstract_html or "")
page_count: int
landing: str | None
if spec.mode == "toc":
toc = _get(s, f"{API}/{spec.doc_id}/toc", expect_json=True)
page_count, landing = _count_toc(toc)
if page_count == 0:
print(f" ! {spec.slug}: TOC empty — falling back to single-doc mode", file=sys.stderr)
spec.mode = "single"
page_count, landing = 1, spec.doc_id
else:
page_count, landing = 1, spec.doc_id
return {
"slug": spec.slug,
"doc_id": spec.doc_id,
"title": meta.get("title") or spec.title,
"version": spec.version,
"platform": spec.platform,
"product": spec.product,
"language": spec.language,
"page_count": page_count,
"mode": spec.mode,
"abstract": meta.get("abstract", ""),
"dates": {"Published": meta.get("published", "")},
"landing_page": landing,
"source_url": spec.source_url or DOC_URL.format(doc_id=spec.doc_id),
}
def _docid_num(doc_id: str) -> int | None:
m = _DOCID_NUM_RE.fullmatch(doc_id)
return int(m.group(1)) if m else None
def _classify_enterprise(title: str, abstract: str) -> tuple[str, str, str] | None:
"""Decide whether an abstract is a Morpheus Enterprise User Manual or
Release Notes and return (product, mode, version); else None.
Filters out the sibling products that share the HPE portal (Morpheus VM
Essentials → hvm-docs, Morpheus Central). The Release Notes abstract puts
the "Morpheus Enterprise" signal in the description (its <h1> is just
"vX.Y.Z Release Notes"), so we match against title + abstract combined."""
blob = f"{title} {abstract}"
if "Morpheus Enterprise" not in blob:
return None
if "VM Essentials" in blob or "Morpheus Central" in blob:
return None
m = _VERSION_RE.search(title) or _VERSION_RE.search(abstract)
if not m:
return None
version = m.group(1)
if "Release Notes" in blob or "Release notes" in blob:
return ("Release Notes", "single", version)
if "Documentation" in blob or "User Manual" in blob:
return ("User Manual", "toc", version)
return None
def _discover_enterprise(s: requests.Session, pinned: list[BundleSpec]) -> list[BundleSpec]:
"""Sweep the docId range for Morpheus Enterprise UM/RN bundles not already
pinned. Best-effort and per-id fault tolerant: a probe that errors just
yields nothing, and the caller falls back to the pinned list on any
top-level failure. The pinned bundles always ship regardless."""
nums = [n for n in (_docid_num(b.doc_id) for b in pinned) if n is not None]
if not nums:
return []
lo, hi = min(nums), max(nums) + DISCOVER_WINDOW
pinned_docids = {b.doc_id for b in pinned}
pinned_pv = {(b.product, b.version) for b in pinned}
def probe(n: int) -> BundleSpec | None:
doc_id = f"sd{n:08d}en_us"
if doc_id in pinned_docids:
return None
try:
html = _get(s, f"{API}/{doc_id}", expect_json=False)
except Exception:
return None
if not html:
return None
meta = _parse_abstract(html)
cls = _classify_enterprise(meta.get("title", ""), meta.get("abstract", ""))
if not cls:
return None
product, mode, version = cls
if (product, version) in pinned_pv:
return None
vslug = version.replace(".", "_")
if product == "Release Notes":
slug = f"morpheus_release_notes_{vslug}"
title = "HPE Morpheus Enterprise Software Release Notes"
else:
slug = f"morpheus_user_manual_{vslug}"
title = "HPE Morpheus Enterprise Software Documentation"
return BundleSpec(slug, doc_id, title, version, product, mode)
found: dict[tuple[str, str], BundleSpec] = {}
print(f" discovery: sweeping docId {lo}..{hi} for new Enterprise bundles", file=sys.stderr)
with ThreadPoolExecutor(max_workers=DISCOVER_WORKERS) as pool:
# pool.map preserves input (ascending docId) order, so the first hit
# for a (product, version) is the lowest docId. HPE mirrors some docs
# under several docIds; the lowest is the original and tends to be the
# most complete, so keep it and ignore later duplicates.
for spec in pool.map(probe, range(lo, hi + 1)):
if spec and (spec.product, spec.version) not in found:
found[(spec.product, spec.version)] = spec
return list(found.values())
def main() -> int:
p = argparse.ArgumentParser(description="Build bundles.json from BUNDLES list.")
p.add_argument("--out", default=str(BUNDLES_JSON))
p.add_argument("--no-discover", action="store_true",
help="skip the docId-range sweep; use the pinned BUNDLES list only")
args = p.parse_args()
s = _session()
specs = list(BUNDLES)
if DISCOVER and not args.no_discover:
try:
for d in sorted(_discover_enterprise(s, BUNDLES), key=lambda b: b.slug):
print(f" + discovered {d.slug} ({d.doc_id})", file=sys.stderr)
specs.append(d)
except Exception as e:
print(f" ! discovery failed ({e}); using pinned bundles only", file=sys.stderr)
out: list[dict[str, Any]] = []
for spec in specs:
print(f" • {spec.slug} ({spec.doc_id}) ...", file=sys.stderr)
out.append(discover_bundle(s, spec))
Path(args.out).write_text(json.dumps(out, indent=2) + "\n")
print(f"wrote {args.out}: {len(out)} bundles, {sum(b['page_count'] for b in out)} pages total", file=sys.stderr)
return 0
if __name__ == "__main__":
sys.exit(main())