fix(rerank): derive the doc cap from tokens (measured floor 1.92), cap the query (#24)
Image rebuild (skip scrape) / build (push) Successful in 6m4s

This commit was merged in pull request #24.
This commit is contained in:
2026-09-10 22:00:30 -04:00
parent 0eb8ad0db6
commit 4de5aaa02f
2 changed files with 31 additions and 8 deletions
+30 -7
View File
@@ -293,13 +293,35 @@ def _rrf_fuse(rankings: list[list[str]], k: int = RRF_K) -> list[str]:
return sorted(scores, key=lambda d: scores[d], reverse=True)
# Per-doc character cap when sending to the reranker. jina-reranker-v2-base
# accepts up to ~1024 tokens PER QUERY+DOC PAIR (n_ctx_train) and rejects
# the WHOLE BATCH if any one pair exceeds it. Truncating each doc to
# ~2000 chars (≈ 500-700 tokens) leaves headroom for the query + chat
# template overhead. The truncation is reranking-only — full chunk text
# still goes back to the LLM caller.
RERANK_DOC_MAX_CHARS = 2000
# --- Reranker input budget (see docs-mcp-template + retrieval-lore.md) ------
# jina-reranker-v2 is a BERT cross-encoder with bert.context_length=1024 and
# LEARNED ABSOLUTE position embeddings: 1024 is a HARD ceiling, not something a
# bigger llama.cpp --ubatch-size can lift. llama.cpp 500s the ENTIRE batch if
# any (query, doc) pair exceeds it, so ONE oversized chunk silently drops that
# whole query to fused order — the same ~90%->~62% P@1 cliff as the sidecar
# being off the network, and just as quiet.
#
# A char cap is only a PROXY for tokens. Derive it from this corpus's measured
# chars-per-token FLOOR (tokenise real chunks via {RERANK_URL}/tokenize and
# take the min). Measured floor for this corpus: 1.92 — identifier-dense variety/trial
# tables tokenise far worse than prose. The old flat 2000-char cap produced docs
# of up to 1042 tokens here, i.e. over the ceiling before the query was even
# prepended; 16% of sampled chunks exceeded 1000 tokens.
RERANK_CTX_TOKENS = int(os.environ.get("RERANK_CTX_TOKENS", "1024"))
RERANK_CHARS_PER_TOKEN = float(os.environ.get("RERANK_CHARS_PER_TOKEN", "1.90"))
RERANK_QUERY_MAX_CHARS = int(os.environ.get("RERANK_QUERY_MAX_CHARS", "300"))
# Margin covers [CLS]/[SEP] framing plus slack, because chars-per-token is an
# ESTIMATE from a sample. Budget to ~94%, never to exactly 1024.
_RERANK_MARGIN_TOKENS = int(os.environ.get("RERANK_MARGIN_TOKENS", "64"))
_RERANK_QUERY_TOKENS = int(RERANK_QUERY_MAX_CHARS / RERANK_CHARS_PER_TOKEN) + 1
RERANK_DOC_MAX_CHARS = max(
256,
int(
(RERANK_CTX_TOKENS - _RERANK_QUERY_TOKENS - _RERANK_MARGIN_TOKENS)
* RERANK_CHARS_PER_TOKEN
),
)
def _rerank(query: str, candidates: list[tuple[str, str]]) -> list[str] | None:
@@ -329,6 +351,7 @@ def _rerank(query: str, candidates: list[tuple[str, str]]) -> list[str] | None:
# Truncate each doc to fit the per-pair token budget. jina-reranker
# rejects the entire batch on any oversize doc.
docs = [(text[:RERANK_DOC_MAX_CHARS] if text else "") for _cid, text in candidates]
query = query[:RERANK_QUERY_MAX_CHARS]
ids = [cid for cid, _ in candidates]
try: