mcp 2.x already on origin/main (#10). Does not change BM25-first search_docs — n=6 is too small to flip the default. - Numbered [1] citations via docs_mcp/format.py - Eval P@1 + JSONL sidecar + eval.pvalue + eval.trace - Heading-recursive chunker, keep chunk-0 and MAX_CHARS=4000 - Nomic prefixes at embed time only; stored text unprefixed Closes #12
44 lines
1.1 KiB
Python
44 lines
1.1 KiB
Python
"""Markdown formatters for MCP tool output.
|
|
|
|
No third-party imports — tests can run without mcp/httpx/chromadb.
|
|
Citation numbers are per-call (1-based, dense) and stateless.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
|
|
def format_search_hit(
|
|
n: int,
|
|
title: str,
|
|
url: str,
|
|
text: str,
|
|
extra: str = "",
|
|
) -> str:
|
|
"""One numbered hit. `n` is 1-based in final reranked/fused order."""
|
|
head = f"[{n}] **{title}**"
|
|
if url:
|
|
head += f" — {url}"
|
|
parts = [head]
|
|
if extra:
|
|
parts.append(extra)
|
|
body = (text or "").strip()
|
|
if body:
|
|
parts.append(body)
|
|
return "\n".join(parts)
|
|
|
|
|
|
def format_search_hits(
|
|
hits: list[tuple[str, str, str, str]],
|
|
) -> str:
|
|
"""Render hits as `[1] **title** — url` then text.
|
|
|
|
`hits` is a list of (title, url, text, extra) in display order.
|
|
Empty list → empty string (no invented citations).
|
|
"""
|
|
if not hits:
|
|
return ""
|
|
blocks = [
|
|
format_search_hit(n, title, url, text, extra)
|
|
for n, (title, url, text, extra) in enumerate(hits, start=1)
|
|
]
|
|
return "\n\n".join(blocks) + "\n"
|