fix(rerank): derive the doc cap from tokens (measured floor 1.47), cap the query (#9)
Image rebuild (skip scrape) / build (push) Successful in 1h30m30s
Image rebuild (skip scrape) / build (push) Successful in 1h30m30s
This commit was merged in pull request #9.
This commit is contained in:
+29
-2
@@ -25,6 +25,32 @@ BM25_DB = Path(os.environ.get("BM25_DB",
|
||||
str(REPO_ROOT / "bm25" / "crop_chem_docs.db")))
|
||||
COLLECTION = f"{os.environ.get('PRODUCT_NAME', 'crop_chem')}_docs"
|
||||
|
||||
# --- Reranker input budget (see docs-mcp-template + retrieval-lore.md) ------
|
||||
# jina-reranker-v2 is a BERT cross-encoder with bert.context_length=1024 and
|
||||
# LEARNED ABSOLUTE position embeddings: 1024 is a HARD ceiling, not something a
|
||||
# bigger llama.cpp --ubatch-size can lift. llama.cpp 500s the ENTIRE batch if
|
||||
# any (query, doc) pair exceeds it, so ONE oversized chunk silently drops that
|
||||
# whole query to fused order — the same ~90%->~62% P@1 cliff as the sidecar
|
||||
# being off the network, and just as quiet.
|
||||
#
|
||||
# A char cap is only a PROXY for tokens. Derive it from this corpus's measured
|
||||
# chars-per-token FLOOR (tokenise real chunks via {RERANK_URL}/tokenize and
|
||||
# take the min). Measured floor for this corpus: 1.47 (EPA/Bayer label text).
|
||||
RERANK_CTX_TOKENS = int(os.environ.get("RERANK_CTX_TOKENS", "1024"))
|
||||
RERANK_CHARS_PER_TOKEN = float(os.environ.get("RERANK_CHARS_PER_TOKEN", "1.45"))
|
||||
RERANK_QUERY_MAX_CHARS = int(os.environ.get("RERANK_QUERY_MAX_CHARS", "300"))
|
||||
# Margin covers [CLS]/[SEP] framing plus slack, because chars-per-token is an
|
||||
# ESTIMATE from a sample. Budget to ~94%, never to exactly 1024.
|
||||
_RERANK_MARGIN_TOKENS = int(os.environ.get("RERANK_MARGIN_TOKENS", "64"))
|
||||
_RERANK_QUERY_TOKENS = int(RERANK_QUERY_MAX_CHARS / RERANK_CHARS_PER_TOKEN) + 1
|
||||
RERANK_DOC_MAX_CHARS = max(
|
||||
256,
|
||||
int(
|
||||
(RERANK_CTX_TOKENS - _RERANK_QUERY_TOKENS - _RERANK_MARGIN_TOKENS)
|
||||
* RERANK_CHARS_PER_TOKEN
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
class Retriever(Protocol):
|
||||
name: str
|
||||
@@ -230,7 +256,7 @@ class RerankedRetriever:
|
||||
if not d:
|
||||
continue
|
||||
# Truncate to keep under the reranker's per-pair context limit
|
||||
docs.append(d[:2000])
|
||||
docs.append(d[:RERANK_DOC_MAX_CHARS])
|
||||
kept_pages.append((source, source_key))
|
||||
|
||||
if not docs:
|
||||
@@ -240,7 +266,8 @@ class RerankedRetriever:
|
||||
try:
|
||||
r = httpx.post(
|
||||
f"{self.rerank_url}/v1/rerank",
|
||||
json={"query": query, "documents": docs},
|
||||
json={"query": query[:RERANK_QUERY_MAX_CHARS],
|
||||
"documents": docs},
|
||||
timeout=self.timeout,
|
||||
)
|
||||
r.raise_for_status()
|
||||
|
||||
Reference in New Issue
Block a user