diff --git a/docs_mcp/server.py b/docs_mcp/server.py index 3abcc66d..4b5428b8 100644 --- a/docs_mcp/server.py +++ b/docs_mcp/server.py @@ -293,13 +293,35 @@ def _rrf_fuse(rankings: list[list[str]], k: int = RRF_K) -> list[str]: return sorted(scores, key=lambda d: scores[d], reverse=True) -# Per-doc character cap when sending to the reranker. jina-reranker-v2-base -# accepts up to ~1024 tokens PER QUERY+DOC PAIR (n_ctx_train) and rejects -# the WHOLE BATCH if any one pair exceeds it. Truncating each doc to -# ~2000 chars (≈ 500-700 tokens) leaves headroom for the query + chat -# template overhead. The truncation is reranking-only — full chunk text -# still goes back to the LLM caller. -RERANK_DOC_MAX_CHARS = 2000 + +# --- Reranker input budget (see docs-mcp-template + retrieval-lore.md) ------ +# jina-reranker-v2 is a BERT cross-encoder with bert.context_length=1024 and +# LEARNED ABSOLUTE position embeddings: 1024 is a HARD ceiling, not something a +# bigger llama.cpp --ubatch-size can lift. llama.cpp 500s the ENTIRE batch if +# any (query, doc) pair exceeds it, so ONE oversized chunk silently drops that +# whole query to fused order — the same ~90%->~62% P@1 cliff as the sidecar +# being off the network, and just as quiet. +# +# A char cap is only a PROXY for tokens. Derive it from this corpus's measured +# chars-per-token FLOOR (tokenise real chunks via {RERANK_URL}/tokenize and +# take the min). Measured floor for this corpus: 1.92 — identifier-dense variety/trial +# tables tokenise far worse than prose. The old flat 2000-char cap produced docs +# of up to 1042 tokens here, i.e. over the ceiling before the query was even +# prepended; 16% of sampled chunks exceeded 1000 tokens. +RERANK_CTX_TOKENS = int(os.environ.get("RERANK_CTX_TOKENS", "1024")) +RERANK_CHARS_PER_TOKEN = float(os.environ.get("RERANK_CHARS_PER_TOKEN", "1.90")) +RERANK_QUERY_MAX_CHARS = int(os.environ.get("RERANK_QUERY_MAX_CHARS", "300")) +# Margin covers [CLS]/[SEP] framing plus slack, because chars-per-token is an +# ESTIMATE from a sample. Budget to ~94%, never to exactly 1024. +_RERANK_MARGIN_TOKENS = int(os.environ.get("RERANK_MARGIN_TOKENS", "64")) +_RERANK_QUERY_TOKENS = int(RERANK_QUERY_MAX_CHARS / RERANK_CHARS_PER_TOKEN) + 1 +RERANK_DOC_MAX_CHARS = max( + 256, + int( + (RERANK_CTX_TOKENS - _RERANK_QUERY_TOKENS - _RERANK_MARGIN_TOKENS) + * RERANK_CHARS_PER_TOKEN + ), +) def _rerank(query: str, candidates: list[tuple[str, str]]) -> list[str] | None: @@ -329,6 +351,7 @@ def _rerank(query: str, candidates: list[tuple[str, str]]) -> list[str] | None: # Truncate each doc to fit the per-pair token budget. jina-reranker # rejects the entire batch on any oversize doc. docs = [(text[:RERANK_DOC_MAX_CHARS] if text else "") for _cid, text in candidates] + query = query[:RERANK_QUERY_MAX_CHARS] ids = [cid for cid, _ in candidates] try: diff --git a/eval/retrievers.py b/eval/retrievers.py index 4aeebb83..d176254e 100644 --- a/eval/retrievers.py +++ b/eval/retrievers.py @@ -124,7 +124,7 @@ class HybridRerankRetriever: def __init__(self, collection, bm25, rerank_url: str, pool: int = 50, rerank_pool: int = 50, - rrf_k: int = 60, doc_max_chars: int = 2000, + rrf_k: int = 60, doc_max_chars: int = 1523, timeout: float = 30.0): self.col = collection self.bm25 = bm25