feat: citations, eval sidecar/pvalue/trace, nomic prefixes
mcp 2.x already on main. Does not replace the variety chunker (one chunk per variety is the anti-hallucination contract). - Numbered [1] citations on search_docs / search_trials - Eval JSONL sidecar, k-curve, eval.pvalue, eval.trace - Nomic prefixes at embed time only; stored text unprefixed Closes #25
This commit is contained in:
+5
-2
@@ -23,7 +23,7 @@ import chromadb
|
||||
from chromadb.config import Settings
|
||||
|
||||
from .chunk import chunks_from_variety, chunks_from_trial
|
||||
from .embeddings import embedding_function
|
||||
from .embeddings import EMBED_DOC_PREFIX, embed_texts, embedding_function
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(message)s")
|
||||
@@ -83,9 +83,12 @@ def upsert_to_chroma(records: list[dict]) -> int:
|
||||
total = 0
|
||||
for i in range(0, len(records), BATCH):
|
||||
chunk = records[i:i + BATCH]
|
||||
texts = [r["text"] for r in chunk]
|
||||
vectors = embed_texts(texts, prefix=EMBED_DOC_PREFIX, ef=embedding_function())
|
||||
col.upsert(
|
||||
ids=[r["id"] for r in chunk],
|
||||
documents=[r["text"] for r in chunk],
|
||||
documents=texts,
|
||||
embeddings=vectors,
|
||||
metadatas=[r["metadata"] for r in chunk],
|
||||
)
|
||||
total += len(chunk)
|
||||
|
||||
Reference in New Issue
Block a user