Default search_docs to dense; make rerank opt-in (#17)
Co-authored-by: claude <[email protected]>
This commit was merged in pull request #17.
This commit is contained in:
@@ -17,7 +17,7 @@ once deployed.
|
||||
|
||||
| Tool | Use |
|
||||
|---|---|
|
||||
| `search_docs` | BM25-default search with optional version / platform / bundle filters; cross-encoder reranked when `RERANK_URL` is set |
|
||||
| `search_docs` | Dense-default search with optional version / platform / bundle filters; rerank only when `RERANK_ENABLED=true` |
|
||||
| `get_page` | Full markdown of one page with metadata header + source URL |
|
||||
| `list_versions` | Discover available versions, doc types, and bundle slugs |
|
||||
| `list_cluster` | Cross-version peers of a page (synthesized from same-GUID overlap) |
|
||||
@@ -51,19 +51,19 @@ peer mapping is free (no fuzzy matching needed).
|
||||
|
||||
## Retrieval
|
||||
|
||||
Eval against 22 hand-curated golden queries — see
|
||||
[`eval/results/baseline.md`](eval/results/baseline.md):
|
||||
Eval against 22 hand-curated golden queries. May 2026 baseline is
|
||||
[`eval/results/baseline.md`](eval/results/baseline.md); after nomic
|
||||
prefixes + heading-recursive chunking (2026-09-30, live index):
|
||||
|
||||
| Retriever | MRR | Recall@5 | nDCG@5 | latency |
|
||||
| Retriever | P@1 | MRR | Recall@5 | nDCG@5 |
|
||||
|---|---:|---:|---:|---:|
|
||||
| dense (Ollama nomic-embed-text) | 0.539 | 0.621 | 0.558 | 88 ms |
|
||||
| BM25 (SQLite FTS5) | 0.880 | 0.909 | 0.883 | 3 ms |
|
||||
| hybrid (dense + BM25 + RRF) | 0.692 | 0.818 | 0.713 | 69 ms |
|
||||
| **bm25 + jina-rerank** | **0.920** | **0.939** | **0.927** | 490 ms (CPU) / ~50 ms (GPU) |
|
||||
| **dense** (nomic prefixes) | **0.955** | **0.966** | **0.985** | **0.972** |
|
||||
| hybrid (dense + BM25 + RRF) | 0.955 | 0.961 | 0.955 | 0.955 |
|
||||
| BM25 (SQLite FTS5) | 0.864 | 0.882 | 0.909 | 0.886 |
|
||||
| bm25 + jina-rerank | 0.773 | 0.827 | 0.848 | 0.816 |
|
||||
|
||||
HPE docs use controlled vocabulary, so lexical match dominates; the
|
||||
cross-encoder cleans up the long tail. See PLAN.md Phase 7/8 for the
|
||||
reasoning.
|
||||
`search_docs` defaults to dense. Rerank is opt-in (`RERANK_ENABLED=true`)
|
||||
— it now hurts. Full table: [`eval/results/post-prefix.md`](eval/results/post-prefix.md).
|
||||
|
||||
## Architecture
|
||||
|
||||
|
||||
@@ -32,16 +32,16 @@ services:
|
||||
# add that hostname here. "*" disables the rebind check entirely.
|
||||
MCP_ALLOWED_HOSTS: "hvm-docs-mcp,localhost,127.0.0.1"
|
||||
|
||||
# Phase 6 — reranker sidecar (jina-reranker-v2-base via llama.cpp).
|
||||
# Phase 6 — reranker sidecar is wired but off. Eval 2026-09-30
|
||||
# (post nomic prefixes): dense MRR=0.966 vs bm25+rerank 0.827.
|
||||
# Set RERANK_ENABLED=true to turn the sidecar back on.
|
||||
RERANK_URL: http://hvm-rerank:8080
|
||||
RERANK_POOL: "200"
|
||||
RERANK_TIMEOUT: "30"
|
||||
RERANK_ENABLED: "false"
|
||||
|
||||
# Phase 8 — hybrid retrieval (BM25 + dense + RRF).
|
||||
# Eval on the HVM corpus (eval/results/baseline.md, 2026-05-22) shows
|
||||
# BM25-default + reranker beats hybrid on every metric (MRR 0.920 vs
|
||||
# 0.875). Leaving HYBRID_SEARCH off so search_docs runs BM25-first +
|
||||
# reranker; dense is the fallback when BM25 finds nothing.
|
||||
# Phase 8 — hybrid retrieval (BM25 + dense + RRF). Off: dense is
|
||||
# the default (eval/results/post-prefix.md). BM25 is fallback only.
|
||||
HYBRID_SEARCH: "false"
|
||||
|
||||
# Phase 10 — usage telemetry.
|
||||
|
||||
+21
-14
@@ -61,6 +61,9 @@ API_LESSONS_MD = Path(__file__).resolve().parent / "api_lessons.md"
|
||||
RERANK_URL = os.environ.get("RERANK_URL", "").rstrip("/") or None
|
||||
RERANK_POOL = int(os.environ.get("RERANK_POOL", "50"))
|
||||
RERANK_TIMEOUT = float(os.environ.get("RERANK_TIMEOUT", "30"))
|
||||
# Opt-in. Watchtower keeps the old container env (RERANK_URL is set in
|
||||
# live compose); default off so a code-only ship actually stops reranking.
|
||||
RERANK_ENABLED = os.environ.get("RERANK_ENABLED", "").lower() in ("true", "1", "yes", "on")
|
||||
|
||||
HYBRID_SEARCH = os.environ.get("HYBRID_SEARCH", "").lower() in ("true", "1", "yes", "on")
|
||||
RRF_K = int(os.environ.get("RRF_K", "60"))
|
||||
@@ -295,11 +298,10 @@ def search_docs(
|
||||
bm25_where = _where_for_bm25(version, platform, bundle_id)
|
||||
pool = max(k * 5, 50)
|
||||
|
||||
# Retrieval mode selection. Eval on this corpus (2026-05-22, 22 golden
|
||||
# queries) showed BM25 MRR=0.88 vs dense MRR=0.54 vs hybrid MRR=0.69 —
|
||||
# HPE structured docs use controlled vocabulary, so lexical match wins.
|
||||
# Dense is kept as fallback when BM25 has no tokens to chew on (e.g.
|
||||
# purely stopword queries). HYBRID_SEARCH=true forces RRF fusion.
|
||||
# Retrieval mode. Eval 2026-09-30 (22 queries, post nomic prefixes +
|
||||
# heading-recursive chunking): dense MRR=0.966 P@1=0.955 vs
|
||||
# bm25+rerank MRR=0.827 P@1=0.773. Default is dense. HYBRID_SEARCH
|
||||
# still forces RRF. Rerank is opt-in (RERANK_ENABLED) — it now hurts.
|
||||
bm = _bm25()
|
||||
docs: list[str] = []
|
||||
metas: list[dict] = []
|
||||
@@ -322,7 +324,18 @@ def search_docs(
|
||||
else "dense_only")
|
||||
retrieval_mode = "hybrid"
|
||||
except Exception as e:
|
||||
log.warning("hybrid failed, falling back to BM25→dense: %s", e)
|
||||
log.warning("hybrid failed, falling back to dense: %s", e)
|
||||
|
||||
if not docs:
|
||||
try:
|
||||
res = _query_dense(col, query, k, where)
|
||||
docs = (res.get("documents") or [[]])[0]
|
||||
metas = (res.get("metadatas") or [[]])[0]
|
||||
dists = (res.get("distances") or [[]])[0]
|
||||
retrieval_mode = "dense"
|
||||
top1_source = "dense_only"
|
||||
except Exception as e:
|
||||
log.warning("dense retrieval failed, falling back to BM25: %s", e)
|
||||
|
||||
if not docs and bm is not None:
|
||||
try:
|
||||
@@ -336,16 +349,10 @@ def search_docs(
|
||||
retrieval_mode = "bm25"
|
||||
top1_source = "bm25_only"
|
||||
except Exception as e:
|
||||
log.warning("BM25 retrieval failed, falling back to dense: %s", e)
|
||||
|
||||
if not docs:
|
||||
res = _query_dense(col, query, k, where)
|
||||
docs = (res.get("documents") or [[]])[0]
|
||||
metas = (res.get("metadatas") or [[]])[0]
|
||||
dists = (res.get("distances") or [[]])[0]
|
||||
log.warning("BM25 retrieval failed: %s", e)
|
||||
|
||||
reranker_fired = False
|
||||
if RERANK_URL and docs:
|
||||
if RERANK_URL and RERANK_ENABLED and docs:
|
||||
# Pull a deeper pool to give the reranker something to chew on.
|
||||
# We over-fetch up to RERANK_POOL chunks from whichever retriever
|
||||
# already won, then ask the reranker to pick the final top-k.
|
||||
|
||||
@@ -0,0 +1,21 @@
|
||||
# Retrieval eval — k=5
|
||||
|
||||
_22 hand-curated queries, generated 2026-09-30 02:40:07_
|
||||
|
||||
Live container after nomic prefixes + heading-recursive chunking
|
||||
(`hvm-docs` image `b200c5f76c56`, Watchtower 2026-09-30T02:36Z).
|
||||
Compared with [`baseline.md`](baseline.md) (2026-05-22, pre-prefix).
|
||||
|
||||
| Retriever | P@1 | MRR | Recall@5 | nDCG@5 | avg latency |
|
||||
| --- | ---: | ---: | ---: | ---: | ---: |
|
||||
| `dense` | 0.955 | 0.966 | 0.985 | 0.972 | 183ms |
|
||||
| `bm25` | 0.864 | 0.882 | 0.909 | 0.886 | 8ms |
|
||||
| `hybrid_rrf` | 0.955 | 0.961 | 0.955 | 0.955 | 113ms |
|
||||
| `bm25+rerank` | 0.773 | 0.827 | 0.848 | 0.816 | 162ms |
|
||||
| `hybrid_rrf+rerank` | 0.727 | 0.813 | 0.894 | 0.821 | 254ms |
|
||||
|
||||
Dense flipped from worst (May MRR 0.539) to best. Rerank now hurts.
|
||||
`search_docs` therefore defaults to dense; rerank is opt-in via
|
||||
`RERANK_ENABLED`.
|
||||
|
||||
Dense P@1 miss: `create a user account` (MRR 0.250).
|
||||
+1
-1
@@ -86,7 +86,7 @@ def main() -> int:
|
||||
expected = [(e["bundle_id"], e["page_id"]) for e in q["expected"]]
|
||||
dense_pages = dense_r.retrieve(q["query"], k=50)
|
||||
bm25_pages = bm25_r.retrieve(q["query"], k=50)
|
||||
ranked = bm25_pages or dense_pages # HVM default retrieval is BM25-first
|
||||
ranked = dense_pages or bm25_pages # HVM default retrieval is dense-first
|
||||
top1 = ranked[0] if ranked else None
|
||||
p1 = p_at_1(ranked, expected)
|
||||
rows.append({
|
||||
|
||||
Reference in New Issue
Block a user