feat: port template upgrades (citations, eval, chunking, nomic prefixes)
mcp 2.x already on origin/main (#10). Does not change BM25-first search_docs — n=6 is too small to flip the default. - Numbered [1] citations via docs_mcp/format.py - Eval P@1 + JSONL sidecar + eval.pvalue + eval.trace - Heading-recursive chunker, keep chunk-0 and MAX_CHARS=4000 - Nomic prefixes at embed time only; stored text unprefixed Closes #12
This commit is contained in:
@@ -0,0 +1,71 @@
|
||||
"""Heading-recursive chunker tests. No corpus, no Chroma."""
|
||||
from __future__ import annotations
|
||||
|
||||
import unittest
|
||||
|
||||
import rag.chunk as chunk
|
||||
|
||||
|
||||
META = {"bundle_id": "Admin.10.0", "title": "Admin Guide"}
|
||||
|
||||
|
||||
def _bodies(text: str) -> list[str]:
|
||||
return [c["text"] for c in chunk.chunks_from_page(text, "Page", META)]
|
||||
|
||||
|
||||
class ChunkTests(unittest.TestCase):
|
||||
def setUp(self) -> None:
|
||||
self._saved = chunk.TARGET_CHARS
|
||||
|
||||
def tearDown(self) -> None:
|
||||
chunk.TARGET_CHARS = self._saved
|
||||
|
||||
def test_chunk0_always_has_title_on_short_page(self) -> None:
|
||||
page = "# Install\n\nJust one short paragraph about installing."
|
||||
chunks = list(chunk.chunks_from_page(page, "Install", META))
|
||||
self.assertGreaterEqual(len(chunks), 1)
|
||||
self.assertEqual(chunks[0]["metadata"]["ordinal"], 0)
|
||||
self.assertIn("# Admin Guide", chunks[0]["text"])
|
||||
|
||||
def test_sibling_h2_sections_do_not_mix(self) -> None:
|
||||
chunk.TARGET_CHARS = 80
|
||||
page = (
|
||||
"# A\n\n"
|
||||
"intro text for A\n\n"
|
||||
"## A.1\n\n"
|
||||
"alpha content lives here only\n\n"
|
||||
"## A.2\n\n"
|
||||
"beta content lives here only\n"
|
||||
)
|
||||
bodies = _bodies(page)
|
||||
self.assertTrue(any("alpha content" in b and "beta content" not in b for b in bodies[1:]),
|
||||
bodies)
|
||||
self.assertTrue(any("beta content" in b and "alpha content" not in b for b in bodies[1:]),
|
||||
bodies)
|
||||
|
||||
def test_giant_section_splits_on_paragraphs(self) -> None:
|
||||
chunk.TARGET_CHARS = 40
|
||||
paras = [f"Paragraph number {i} with enough words." for i in range(8)]
|
||||
page = "# Giant\n\n" + "\n\n".join(paras)
|
||||
bodies = _bodies(page)
|
||||
# skip chunk 0
|
||||
for b in bodies[1:]:
|
||||
# may exceed by at most one paragraph (the one that filled the buf)
|
||||
self.assertLessEqual(len(b), chunk.TARGET_CHARS + len(paras[0]) + 2, b)
|
||||
|
||||
def test_fenced_code_never_sliced(self) -> None:
|
||||
chunk.TARGET_CHARS = 30
|
||||
fence = "```\n" + ("x" * 80) + "\n```"
|
||||
page = "# Code\n\nBefore.\n\n" + fence + "\n\nAfter."
|
||||
bodies = _bodies(page)
|
||||
joined = "\n".join(bodies)
|
||||
self.assertIn(fence, joined)
|
||||
# no body chunk should contain a half-fence
|
||||
for b in bodies:
|
||||
if "```" in b:
|
||||
self.assertTrue(b.strip().startswith("```") or "```\n" in b)
|
||||
self.assertGreaterEqual(b.count("```"), 2, b)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
Reference in New Issue
Block a user