fix(rerank): derive the doc cap from tokens (measured floor 1.47), cap the query
Ports the docs-mcp-template fix. Both rerank call sites here (_rerank_pool in
docs_mcp/server.py and RerankedRetriever in rag/retrieval.py) truncated docs to
a flat 2000 CHARACTERS as a stand-in for the reranker's 1024-TOKEN pair limit.
jina-reranker-v2 is a BERT cross-encoder with bert.context_length=1024 and
learned absolute position embeddings — 1024 is a hard ceiling, not a tunable —
and llama.cpp 500s the ENTIRE batch if any one pair exceeds it, silently
dropping that query to fused order.
Measured floor for this corpus via {RERANK_URL}/tokenize: 1.47 chars/token
(EPA/Bayer label prose), worst observed 997 tokens at the old 2000-char cap —
under the ceiling alone, but over it once the query is prepended. The cap is
now derived from RERANK_CTX_TOKENS / RERANK_CHARS_PER_TOKEN / a query reserve
(1091 chars here), budgeting the PAIR to ~94% rather than exactly 1024.
The query is now truncated too; previously only the document was, though it is
the pair that must fit.
Eval (hybrid+rerank, 35 golden queries, k=5, pool=50) — no regression:
before MRR 0.667 Recall@5 0.643 nDCG@5 0.627 0 errors
after MRR 0.667 Recall@5 0.643 nDCG@5 0.627 0 errors
Note when re-testing in a running container: the image ships precompiled
__pycache__/*.pyc and Python will load the STALE bytecode over a docker cp'd
source edit. rm -rf /app/<pkg>/__pycache__ first or you measure the old code.
Co-Authored-By: Claude Opus 5 (1M context) <[email protected]>
Claude-Session: https://claude.ai/code/session_01AiYH8nxc6DgUTdwHnP9PEe
This commit is contained in:
+35
-5
@@ -57,6 +57,32 @@ RERANK_URL = os.environ.get("RERANK_URL", "").rstrip("/") or None
|
|||||||
RERANK_POOL = int(os.environ.get("RERANK_POOL", "50"))
|
RERANK_POOL = int(os.environ.get("RERANK_POOL", "50"))
|
||||||
RERANK_TIMEOUT = float(os.environ.get("RERANK_TIMEOUT", "30"))
|
RERANK_TIMEOUT = float(os.environ.get("RERANK_TIMEOUT", "30"))
|
||||||
|
|
||||||
|
# --- Reranker input budget (see docs-mcp-template + retrieval-lore.md) ------
|
||||||
|
# jina-reranker-v2 is a BERT cross-encoder with bert.context_length=1024 and
|
||||||
|
# LEARNED ABSOLUTE position embeddings: 1024 is a HARD ceiling, not something a
|
||||||
|
# bigger llama.cpp --ubatch-size can lift. llama.cpp 500s the ENTIRE batch if
|
||||||
|
# any (query, doc) pair exceeds it, so ONE oversized chunk silently drops that
|
||||||
|
# whole query to fused order — the same ~90%->~62% P@1 cliff as the sidecar
|
||||||
|
# being off the network, and just as quiet.
|
||||||
|
#
|
||||||
|
# A char cap is only a PROXY for tokens. Derive it from this corpus's measured
|
||||||
|
# chars-per-token FLOOR (tokenise real chunks via {RERANK_URL}/tokenize and
|
||||||
|
# take the min). Measured floor for this corpus: 1.47 (EPA/Bayer label text).
|
||||||
|
RERANK_CTX_TOKENS = int(os.environ.get("RERANK_CTX_TOKENS", "1024"))
|
||||||
|
RERANK_CHARS_PER_TOKEN = float(os.environ.get("RERANK_CHARS_PER_TOKEN", "1.45"))
|
||||||
|
RERANK_QUERY_MAX_CHARS = int(os.environ.get("RERANK_QUERY_MAX_CHARS", "300"))
|
||||||
|
# Margin covers [CLS]/[SEP] framing plus slack, because chars-per-token is an
|
||||||
|
# ESTIMATE from a sample. Budget to ~94%, never to exactly 1024.
|
||||||
|
_RERANK_MARGIN_TOKENS = int(os.environ.get("RERANK_MARGIN_TOKENS", "64"))
|
||||||
|
_RERANK_QUERY_TOKENS = int(RERANK_QUERY_MAX_CHARS / RERANK_CHARS_PER_TOKEN) + 1
|
||||||
|
RERANK_DOC_MAX_CHARS = max(
|
||||||
|
256,
|
||||||
|
int(
|
||||||
|
(RERANK_CTX_TOKENS - _RERANK_QUERY_TOKENS - _RERANK_MARGIN_TOKENS)
|
||||||
|
* RERANK_CHARS_PER_TOKEN
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
HYBRID_SEARCH = os.environ.get("HYBRID_SEARCH", "").lower() in ("true", "1", "yes", "on")
|
HYBRID_SEARCH = os.environ.get("HYBRID_SEARCH", "").lower() in ("true", "1", "yes", "on")
|
||||||
RRF_K = int(os.environ.get("RRF_K", "60"))
|
RRF_K = int(os.environ.get("RRF_K", "60"))
|
||||||
|
|
||||||
@@ -366,16 +392,20 @@ def _rerank_pool(
|
|||||||
pool: list[tuple[str, dict, float]],
|
pool: list[tuple[str, dict, float]],
|
||||||
) -> list[tuple[str, dict, float]]:
|
) -> list[tuple[str, dict, float]]:
|
||||||
"""Send (query, doc_text) pairs to a llama.cpp /v1/rerank endpoint
|
"""Send (query, doc_text) pairs to a llama.cpp /v1/rerank endpoint
|
||||||
and reorder by relevance score. Truncates docs to 2000 chars (the
|
and reorder by relevance score. Both sides are truncated to the derived
|
||||||
jina-reranker GGUF rejects the ENTIRE batch if any pair exceeds
|
RERANK_* budget above — the pair, not the doc alone, must fit the
|
||||||
n_ctx_train=1024; full text still goes back to the user)."""
|
reranker's 1024-token ceiling, and one oversized pair 500s the whole
|
||||||
|
batch. Truncation is for SCORING ONLY; full text still goes to the user."""
|
||||||
import httpx
|
import httpx
|
||||||
docs_truncated = [d[:2000] for d, _meta, _s in pool[:RERANK_POOL]]
|
docs_truncated = [
|
||||||
|
d[:RERANK_DOC_MAX_CHARS] for d, _meta, _s in pool[:RERANK_POOL]
|
||||||
|
]
|
||||||
if not docs_truncated:
|
if not docs_truncated:
|
||||||
return pool
|
return pool
|
||||||
r = httpx.post(
|
r = httpx.post(
|
||||||
f"{RERANK_URL}/v1/rerank",
|
f"{RERANK_URL}/v1/rerank",
|
||||||
json={"query": query, "documents": docs_truncated},
|
json={"query": query[:RERANK_QUERY_MAX_CHARS],
|
||||||
|
"documents": docs_truncated},
|
||||||
timeout=RERANK_TIMEOUT,
|
timeout=RERANK_TIMEOUT,
|
||||||
)
|
)
|
||||||
r.raise_for_status()
|
r.raise_for_status()
|
||||||
|
|||||||
+29
-2
@@ -25,6 +25,32 @@ BM25_DB = Path(os.environ.get("BM25_DB",
|
|||||||
str(REPO_ROOT / "bm25" / "crop_chem_docs.db")))
|
str(REPO_ROOT / "bm25" / "crop_chem_docs.db")))
|
||||||
COLLECTION = f"{os.environ.get('PRODUCT_NAME', 'crop_chem')}_docs"
|
COLLECTION = f"{os.environ.get('PRODUCT_NAME', 'crop_chem')}_docs"
|
||||||
|
|
||||||
|
# --- Reranker input budget (see docs-mcp-template + retrieval-lore.md) ------
|
||||||
|
# jina-reranker-v2 is a BERT cross-encoder with bert.context_length=1024 and
|
||||||
|
# LEARNED ABSOLUTE position embeddings: 1024 is a HARD ceiling, not something a
|
||||||
|
# bigger llama.cpp --ubatch-size can lift. llama.cpp 500s the ENTIRE batch if
|
||||||
|
# any (query, doc) pair exceeds it, so ONE oversized chunk silently drops that
|
||||||
|
# whole query to fused order — the same ~90%->~62% P@1 cliff as the sidecar
|
||||||
|
# being off the network, and just as quiet.
|
||||||
|
#
|
||||||
|
# A char cap is only a PROXY for tokens. Derive it from this corpus's measured
|
||||||
|
# chars-per-token FLOOR (tokenise real chunks via {RERANK_URL}/tokenize and
|
||||||
|
# take the min). Measured floor for this corpus: 1.47 (EPA/Bayer label text).
|
||||||
|
RERANK_CTX_TOKENS = int(os.environ.get("RERANK_CTX_TOKENS", "1024"))
|
||||||
|
RERANK_CHARS_PER_TOKEN = float(os.environ.get("RERANK_CHARS_PER_TOKEN", "1.45"))
|
||||||
|
RERANK_QUERY_MAX_CHARS = int(os.environ.get("RERANK_QUERY_MAX_CHARS", "300"))
|
||||||
|
# Margin covers [CLS]/[SEP] framing plus slack, because chars-per-token is an
|
||||||
|
# ESTIMATE from a sample. Budget to ~94%, never to exactly 1024.
|
||||||
|
_RERANK_MARGIN_TOKENS = int(os.environ.get("RERANK_MARGIN_TOKENS", "64"))
|
||||||
|
_RERANK_QUERY_TOKENS = int(RERANK_QUERY_MAX_CHARS / RERANK_CHARS_PER_TOKEN) + 1
|
||||||
|
RERANK_DOC_MAX_CHARS = max(
|
||||||
|
256,
|
||||||
|
int(
|
||||||
|
(RERANK_CTX_TOKENS - _RERANK_QUERY_TOKENS - _RERANK_MARGIN_TOKENS)
|
||||||
|
* RERANK_CHARS_PER_TOKEN
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
class Retriever(Protocol):
|
class Retriever(Protocol):
|
||||||
name: str
|
name: str
|
||||||
@@ -230,7 +256,7 @@ class RerankedRetriever:
|
|||||||
if not d:
|
if not d:
|
||||||
continue
|
continue
|
||||||
# Truncate to keep under the reranker's per-pair context limit
|
# Truncate to keep under the reranker's per-pair context limit
|
||||||
docs.append(d[:2000])
|
docs.append(d[:RERANK_DOC_MAX_CHARS])
|
||||||
kept_pages.append((source, source_key))
|
kept_pages.append((source, source_key))
|
||||||
|
|
||||||
if not docs:
|
if not docs:
|
||||||
@@ -240,7 +266,8 @@ class RerankedRetriever:
|
|||||||
try:
|
try:
|
||||||
r = httpx.post(
|
r = httpx.post(
|
||||||
f"{self.rerank_url}/v1/rerank",
|
f"{self.rerank_url}/v1/rerank",
|
||||||
json={"query": query, "documents": docs},
|
json={"query": query[:RERANK_QUERY_MAX_CHARS],
|
||||||
|
"documents": docs},
|
||||||
timeout=self.timeout,
|
timeout=self.timeout,
|
||||||
)
|
)
|
||||||
r.raise_for_status()
|
r.raise_for_status()
|
||||||
|
|||||||
Reference in New Issue
Block a user