mcp 2.x is already on origin/main. This ports the rest of docs-mcp-template #12–#16 without replacing HVM's BM25-first search_docs path. - Numbered [1] citations via docs_mcp/format.py - Eval P@1 + JSONL sidecar + eval.pvalue - Heading-recursive chunker, keep chunk-0 and MAX_CHARS=4000 - Nomic prefixes at embed time only; stored text unprefixed Closes #15
72 lines
2.5 KiB
Python
72 lines
2.5 KiB
Python
"""Heading-recursive chunker tests. No corpus, no Chroma."""
|
|
from __future__ import annotations
|
|
|
|
import unittest
|
|
|
|
import rag.chunk as chunk
|
|
|
|
|
|
META = {"bundle_id": "Admin.10.0", "title": "Admin Guide"}
|
|
|
|
|
|
def _bodies(text: str) -> list[str]:
|
|
return [c["text"] for c in chunk.chunks_from_page(text, "Page", META)]
|
|
|
|
|
|
class ChunkTests(unittest.TestCase):
|
|
def setUp(self) -> None:
|
|
self._saved = chunk.TARGET_CHARS
|
|
|
|
def tearDown(self) -> None:
|
|
chunk.TARGET_CHARS = self._saved
|
|
|
|
def test_chunk0_always_has_title_on_short_page(self) -> None:
|
|
page = "# Install\n\nJust one short paragraph about installing."
|
|
chunks = list(chunk.chunks_from_page(page, "Install", META))
|
|
self.assertGreaterEqual(len(chunks), 1)
|
|
self.assertEqual(chunks[0]["metadata"]["ordinal"], 0)
|
|
self.assertIn("# Admin Guide", chunks[0]["text"])
|
|
|
|
def test_sibling_h2_sections_do_not_mix(self) -> None:
|
|
chunk.TARGET_CHARS = 80
|
|
page = (
|
|
"# A\n\n"
|
|
"intro text for A\n\n"
|
|
"## A.1\n\n"
|
|
"alpha content lives here only\n\n"
|
|
"## A.2\n\n"
|
|
"beta content lives here only\n"
|
|
)
|
|
bodies = _bodies(page)
|
|
self.assertTrue(any("alpha content" in b and "beta content" not in b for b in bodies[1:]),
|
|
bodies)
|
|
self.assertTrue(any("beta content" in b and "alpha content" not in b for b in bodies[1:]),
|
|
bodies)
|
|
|
|
def test_giant_section_splits_on_paragraphs(self) -> None:
|
|
chunk.TARGET_CHARS = 40
|
|
paras = [f"Paragraph number {i} with enough words." for i in range(8)]
|
|
page = "# Giant\n\n" + "\n\n".join(paras)
|
|
bodies = _bodies(page)
|
|
# skip chunk 0
|
|
for b in bodies[1:]:
|
|
# may exceed by at most one paragraph (the one that filled the buf)
|
|
self.assertLessEqual(len(b), chunk.TARGET_CHARS + len(paras[0]) + 2, b)
|
|
|
|
def test_fenced_code_never_sliced(self) -> None:
|
|
chunk.TARGET_CHARS = 30
|
|
fence = "```\n" + ("x" * 80) + "\n```"
|
|
page = "# Code\n\nBefore.\n\n" + fence + "\n\nAfter."
|
|
bodies = _bodies(page)
|
|
joined = "\n".join(bodies)
|
|
self.assertIn(fence, joined)
|
|
# no body chunk should contain a half-fence
|
|
for b in bodies:
|
|
if "```" in b:
|
|
self.assertTrue(b.strip().startswith("```") or "```\n" in b)
|
|
self.assertGreaterEqual(b.count("```"), 2, b)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|