From 579aea3b46db426949539d61f7ec5d0da6550a6d Mon Sep 17 00:00:00 2001 From: Justin Paul Date: Thu, 10 Sep 2026 21:47:46 -0400 Subject: [PATCH] fix(rerank): derive the doc cap from tokens (measured floor 1.47), cap the query MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ports the docs-mcp-template fix. Both rerank call sites here (_rerank_pool in docs_mcp/server.py and RerankedRetriever in rag/retrieval.py) truncated docs to a flat 2000 CHARACTERS as a stand-in for the reranker's 1024-TOKEN pair limit. jina-reranker-v2 is a BERT cross-encoder with bert.context_length=1024 and learned absolute position embeddings — 1024 is a hard ceiling, not a tunable — and llama.cpp 500s the ENTIRE batch if any one pair exceeds it, silently dropping that query to fused order. Measured floor for this corpus via {RERANK_URL}/tokenize: 1.47 chars/token (EPA/Bayer label prose), worst observed 997 tokens at the old 2000-char cap — under the ceiling alone, but over it once the query is prepended. The cap is now derived from RERANK_CTX_TOKENS / RERANK_CHARS_PER_TOKEN / a query reserve (1091 chars here), budgeting the PAIR to ~94% rather than exactly 1024. The query is now truncated too; previously only the document was, though it is the pair that must fit. Eval (hybrid+rerank, 35 golden queries, k=5, pool=50) — no regression: before MRR 0.667 Recall@5 0.643 nDCG@5 0.627 0 errors after MRR 0.667 Recall@5 0.643 nDCG@5 0.627 0 errors Note when re-testing in a running container: the image ships precompiled __pycache__/*.pyc and Python will load the STALE bytecode over a docker cp'd source edit. rm -rf /app//__pycache__ first or you measure the old code. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01AiYH8nxc6DgUTdwHnP9PEe --- docs_mcp/server.py | 40 +++++++++++++++++++++++++++++++++++----- rag/retrieval.py | 31 +++++++++++++++++++++++++++++-- 2 files changed, 64 insertions(+), 7 deletions(-) diff --git a/docs_mcp/server.py b/docs_mcp/server.py index 10c51fd..07ef814 100644 --- a/docs_mcp/server.py +++ b/docs_mcp/server.py @@ -57,6 +57,32 @@ RERANK_URL = os.environ.get("RERANK_URL", "").rstrip("/") or None RERANK_POOL = int(os.environ.get("RERANK_POOL", "50")) RERANK_TIMEOUT = float(os.environ.get("RERANK_TIMEOUT", "30")) +# --- Reranker input budget (see docs-mcp-template + retrieval-lore.md) ------ +# jina-reranker-v2 is a BERT cross-encoder with bert.context_length=1024 and +# LEARNED ABSOLUTE position embeddings: 1024 is a HARD ceiling, not something a +# bigger llama.cpp --ubatch-size can lift. llama.cpp 500s the ENTIRE batch if +# any (query, doc) pair exceeds it, so ONE oversized chunk silently drops that +# whole query to fused order — the same ~90%->~62% P@1 cliff as the sidecar +# being off the network, and just as quiet. +# +# A char cap is only a PROXY for tokens. Derive it from this corpus's measured +# chars-per-token FLOOR (tokenise real chunks via {RERANK_URL}/tokenize and +# take the min). Measured floor for this corpus: 1.47 (EPA/Bayer label text). +RERANK_CTX_TOKENS = int(os.environ.get("RERANK_CTX_TOKENS", "1024")) +RERANK_CHARS_PER_TOKEN = float(os.environ.get("RERANK_CHARS_PER_TOKEN", "1.45")) +RERANK_QUERY_MAX_CHARS = int(os.environ.get("RERANK_QUERY_MAX_CHARS", "300")) +# Margin covers [CLS]/[SEP] framing plus slack, because chars-per-token is an +# ESTIMATE from a sample. Budget to ~94%, never to exactly 1024. +_RERANK_MARGIN_TOKENS = int(os.environ.get("RERANK_MARGIN_TOKENS", "64")) +_RERANK_QUERY_TOKENS = int(RERANK_QUERY_MAX_CHARS / RERANK_CHARS_PER_TOKEN) + 1 +RERANK_DOC_MAX_CHARS = max( + 256, + int( + (RERANK_CTX_TOKENS - _RERANK_QUERY_TOKENS - _RERANK_MARGIN_TOKENS) + * RERANK_CHARS_PER_TOKEN + ), +) + HYBRID_SEARCH = os.environ.get("HYBRID_SEARCH", "").lower() in ("true", "1", "yes", "on") RRF_K = int(os.environ.get("RRF_K", "60")) @@ -366,16 +392,20 @@ def _rerank_pool( pool: list[tuple[str, dict, float]], ) -> list[tuple[str, dict, float]]: """Send (query, doc_text) pairs to a llama.cpp /v1/rerank endpoint - and reorder by relevance score. Truncates docs to 2000 chars (the - jina-reranker GGUF rejects the ENTIRE batch if any pair exceeds - n_ctx_train=1024; full text still goes back to the user).""" + and reorder by relevance score. Both sides are truncated to the derived + RERANK_* budget above — the pair, not the doc alone, must fit the + reranker's 1024-token ceiling, and one oversized pair 500s the whole + batch. Truncation is for SCORING ONLY; full text still goes to the user.""" import httpx - docs_truncated = [d[:2000] for d, _meta, _s in pool[:RERANK_POOL]] + docs_truncated = [ + d[:RERANK_DOC_MAX_CHARS] for d, _meta, _s in pool[:RERANK_POOL] + ] if not docs_truncated: return pool r = httpx.post( f"{RERANK_URL}/v1/rerank", - json={"query": query, "documents": docs_truncated}, + json={"query": query[:RERANK_QUERY_MAX_CHARS], + "documents": docs_truncated}, timeout=RERANK_TIMEOUT, ) r.raise_for_status() diff --git a/rag/retrieval.py b/rag/retrieval.py index bb4725e..436fff9 100644 --- a/rag/retrieval.py +++ b/rag/retrieval.py @@ -25,6 +25,32 @@ BM25_DB = Path(os.environ.get("BM25_DB", str(REPO_ROOT / "bm25" / "crop_chem_docs.db"))) COLLECTION = f"{os.environ.get('PRODUCT_NAME', 'crop_chem')}_docs" +# --- Reranker input budget (see docs-mcp-template + retrieval-lore.md) ------ +# jina-reranker-v2 is a BERT cross-encoder with bert.context_length=1024 and +# LEARNED ABSOLUTE position embeddings: 1024 is a HARD ceiling, not something a +# bigger llama.cpp --ubatch-size can lift. llama.cpp 500s the ENTIRE batch if +# any (query, doc) pair exceeds it, so ONE oversized chunk silently drops that +# whole query to fused order — the same ~90%->~62% P@1 cliff as the sidecar +# being off the network, and just as quiet. +# +# A char cap is only a PROXY for tokens. Derive it from this corpus's measured +# chars-per-token FLOOR (tokenise real chunks via {RERANK_URL}/tokenize and +# take the min). Measured floor for this corpus: 1.47 (EPA/Bayer label text). +RERANK_CTX_TOKENS = int(os.environ.get("RERANK_CTX_TOKENS", "1024")) +RERANK_CHARS_PER_TOKEN = float(os.environ.get("RERANK_CHARS_PER_TOKEN", "1.45")) +RERANK_QUERY_MAX_CHARS = int(os.environ.get("RERANK_QUERY_MAX_CHARS", "300")) +# Margin covers [CLS]/[SEP] framing plus slack, because chars-per-token is an +# ESTIMATE from a sample. Budget to ~94%, never to exactly 1024. +_RERANK_MARGIN_TOKENS = int(os.environ.get("RERANK_MARGIN_TOKENS", "64")) +_RERANK_QUERY_TOKENS = int(RERANK_QUERY_MAX_CHARS / RERANK_CHARS_PER_TOKEN) + 1 +RERANK_DOC_MAX_CHARS = max( + 256, + int( + (RERANK_CTX_TOKENS - _RERANK_QUERY_TOKENS - _RERANK_MARGIN_TOKENS) + * RERANK_CHARS_PER_TOKEN + ), +) + class Retriever(Protocol): name: str @@ -230,7 +256,7 @@ class RerankedRetriever: if not d: continue # Truncate to keep under the reranker's per-pair context limit - docs.append(d[:2000]) + docs.append(d[:RERANK_DOC_MAX_CHARS]) kept_pages.append((source, source_key)) if not docs: @@ -240,7 +266,8 @@ class RerankedRetriever: try: r = httpx.post( f"{self.rerank_url}/v1/rerank", - json={"query": query, "documents": docs}, + json={"query": query[:RERANK_QUERY_MAX_CHARS], + "documents": docs}, timeout=self.timeout, ) r.raise_for_status() -- 2.54.0