fix(rerank): derive the doc cap from tokens (measured floor 1.47), cap the query (#9)
Image rebuild (skip scrape) / build (push) Successful in 1h30m30s

This commit was merged in pull request #9.
This commit is contained in:
2026-09-10 22:00:28 -04:00
parent 053884f9fb
commit 9008bd372c
2 changed files with 64 additions and 7 deletions
+35 -5
View File
@@ -57,6 +57,32 @@ RERANK_URL = os.environ.get("RERANK_URL", "").rstrip("/") or None
RERANK_POOL = int(os.environ.get("RERANK_POOL", "50"))
RERANK_TIMEOUT = float(os.environ.get("RERANK_TIMEOUT", "30"))
# --- Reranker input budget (see docs-mcp-template + retrieval-lore.md) ------
# jina-reranker-v2 is a BERT cross-encoder with bert.context_length=1024 and
# LEARNED ABSOLUTE position embeddings: 1024 is a HARD ceiling, not something a
# bigger llama.cpp --ubatch-size can lift. llama.cpp 500s the ENTIRE batch if
# any (query, doc) pair exceeds it, so ONE oversized chunk silently drops that
# whole query to fused order — the same ~90%->~62% P@1 cliff as the sidecar
# being off the network, and just as quiet.
#
# A char cap is only a PROXY for tokens. Derive it from this corpus's measured
# chars-per-token FLOOR (tokenise real chunks via {RERANK_URL}/tokenize and
# take the min). Measured floor for this corpus: 1.47 (EPA/Bayer label text).
RERANK_CTX_TOKENS = int(os.environ.get("RERANK_CTX_TOKENS", "1024"))
RERANK_CHARS_PER_TOKEN = float(os.environ.get("RERANK_CHARS_PER_TOKEN", "1.45"))
RERANK_QUERY_MAX_CHARS = int(os.environ.get("RERANK_QUERY_MAX_CHARS", "300"))
# Margin covers [CLS]/[SEP] framing plus slack, because chars-per-token is an
# ESTIMATE from a sample. Budget to ~94%, never to exactly 1024.
_RERANK_MARGIN_TOKENS = int(os.environ.get("RERANK_MARGIN_TOKENS", "64"))
_RERANK_QUERY_TOKENS = int(RERANK_QUERY_MAX_CHARS / RERANK_CHARS_PER_TOKEN) + 1
RERANK_DOC_MAX_CHARS = max(
256,
int(
(RERANK_CTX_TOKENS - _RERANK_QUERY_TOKENS - _RERANK_MARGIN_TOKENS)
* RERANK_CHARS_PER_TOKEN
),
)
HYBRID_SEARCH = os.environ.get("HYBRID_SEARCH", "").lower() in ("true", "1", "yes", "on")
RRF_K = int(os.environ.get("RRF_K", "60"))
@@ -366,16 +392,20 @@ def _rerank_pool(
pool: list[tuple[str, dict, float]],
) -> list[tuple[str, dict, float]]:
"""Send (query, doc_text) pairs to a llama.cpp /v1/rerank endpoint
and reorder by relevance score. Truncates docs to 2000 chars (the
jina-reranker GGUF rejects the ENTIRE batch if any pair exceeds
n_ctx_train=1024; full text still goes back to the user)."""
and reorder by relevance score. Both sides are truncated to the derived
RERANK_* budget above — the pair, not the doc alone, must fit the
reranker's 1024-token ceiling, and one oversized pair 500s the whole
batch. Truncation is for SCORING ONLY; full text still goes to the user."""
import httpx
docs_truncated = [d[:2000] for d, _meta, _s in pool[:RERANK_POOL]]
docs_truncated = [
d[:RERANK_DOC_MAX_CHARS] for d, _meta, _s in pool[:RERANK_POOL]
]
if not docs_truncated:
return pool
r = httpx.post(
f"{RERANK_URL}/v1/rerank",
json={"query": query, "documents": docs_truncated},
json={"query": query[:RERANK_QUERY_MAX_CHARS],
"documents": docs_truncated},
timeout=RERANK_TIMEOUT,
)
r.raise_for_status()