From e1fa9b2b05e2df46f8acbb874817fe39678383c6 Mon Sep 17 00:00:00 2001 From: Gerard Louis Recinto Date: Thu, 8 Oct 2026 02:17:48 -0700 Subject: [PATCH 1/4] grounded search over the docs: Hugging Face embeddings into a Joltrin vector store, with citations and I don't know --- .github/workflows/hf-search-example.yml | 56 +++++++ examples/hf_grounded_search/requirements.txt | 3 + examples/hf_grounded_search/search.py | 146 +++++++++++++++++++ examples/hf_grounded_search/test_search.py | 81 ++++++++++ 4 files changed, 286 insertions(+) create mode 100644 .github/workflows/hf-search-example.yml create mode 100644 examples/hf_grounded_search/requirements.txt create mode 100644 examples/hf_grounded_search/search.py create mode 100644 examples/hf_grounded_search/test_search.py diff --git a/.github/workflows/hf-search-example.yml b/.github/workflows/hf-search-example.yml new file mode 100644 index 000000000..cbc8839c6 --- /dev/null +++ b/.github/workflows/hf-search-example.yml @@ -0,0 +1,56 @@ +name: Hugging Face search example + +# Builds the Joltrin native library for Linux, installs CPU PyTorch and +# transformers, and runs examples/hf_grounded_search/test_search.py: real +# embeddings, a real Joltrin vector store, and questions with known answers. +# Not a required check, so it can use a path filter. +on: + pull_request: + paths: + - 'examples/hf_grounded_search/**' + - 'bindings/python/sop/**' + - 'bindings/main/**' + - 'README.md' + - 'docs/AGENT_PROTOCOLS.md' + - 'docs/AGENT_BARRIER_TESTS.md' + - 'docs/MCP_A2A_AND_VERIFICATION_ENGINE.md' + - 'docs/WHO_IS_IT_FOR.md' + - '.github/workflows/hf-search-example.yml' + +permissions: + contents: read + +jobs: + hf-search: + name: Grounded search answers and refuses correctly + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - uses: actions/setup-go@v7 + with: + go-version-file: 'go.mod' + cache: true + + - uses: actions/setup-python@v5 + with: + python-version: '3.12' + cache: 'pip' + cache-dependency-path: examples/hf_grounded_search/requirements.txt + + - name: Build the native library + run: CGO_ENABLED=1 go build -buildmode=c-shared -o bindings/python/sop/libjsondb_amd64linux.so ./bindings/main + + - name: Install CPU PyTorch and transformers + run: pip install -r examples/hf_grounded_search/requirements.txt --extra-index-url https://download.pytorch.org/whl/cpu + + - uses: actions/cache@v6 + with: + path: ~/.cache/huggingface + key: hf-all-MiniLM-L6-v2-1110a24 + + - name: Run the search test + working-directory: examples/hf_grounded_search + env: + PYTHONPATH: ${{ github.workspace }}/bindings/python + run: python test_search.py diff --git a/examples/hf_grounded_search/requirements.txt b/examples/hf_grounded_search/requirements.txt new file mode 100644 index 000000000..769523f42 --- /dev/null +++ b/examples/hf_grounded_search/requirements.txt @@ -0,0 +1,3 @@ +torch==2.14.1 +transformers +huggingface-hub diff --git a/examples/hf_grounded_search/search.py b/examples/hf_grounded_search/search.py new file mode 100644 index 000000000..402408720 --- /dev/null +++ b/examples/hf_grounded_search/search.py @@ -0,0 +1,146 @@ +"""Answer questions from Joltrin's own docs, with citations, or say "I don't know". + +What it does, end to end: + + 1. Split the docs into chunks, one citation (file#heading) per chunk. + 2. Turn each chunk into a 384-number vector with a Hugging Face model. + 3. Store the vectors in a Joltrin vector store, inside a transaction. + 4. For a question: embed it, ask the store for the nearest chunks, and + return them with their citations. If even the best match is not close + enough, return nothing and say so. + +Step 2 is written with transformers and torch directly, not with the +sentence-transformers wrapper, so each part is visible: tokenize, run the +model, average the token vectors while ignoring padding, normalize to length 1. + +This does retrieval only. Nothing here writes new text, so nothing here can +invent a sentence. The passages are the docs' own words. Giving them to a +language model as context is a separate step this example does not take. + +Run it from the repo root, after building the native library for your machine +(see bindings/python/README.md) and installing requirements.txt: + + PYTHONPATH=bindings/python python examples/hf_grounded_search/search.py "how long do lessons last?" +""" +import argparse +import re +import sys +import tempfile +from pathlib import Path + +import torch +import torch.nn.functional as F +from transformers import AutoModel, AutoTokenizer + +from sop import Context +from sop.ai import Database, DatabaseType, Item +from sop.database import DatabaseOptions + +REPO = Path(__file__).resolve().parents[2] +CORPUS = ["README.md", "docs/AGENT_PROTOCOLS.md", "docs/AGENT_BARRIER_TESTS.md", + "docs/MCP_A2A_AND_VERIFICATION_ENGINE.md", "docs/WHO_IS_IT_FOR.md"] + +# Pinned to one commit, so a changed model on the Hub cannot change the results. +MODEL = "sentence-transformers/all-MiniLM-L6-v2" +REVISION = "1110a243fdf4706b3f48f1d95db1a4f5529b4d41" + +CHUNK_WORDS = 150 +# A match scoring below this is treated as "not in the docs". It sits between the +# highest score of the questions about other topics (0.17) and the lowest score of +# the questions the docs answer (0.30) in test_search.py. That is 14 questions, so +# treat it as a starting point and re-measure on your own questions. +MIN_SCORE = 0.25 + + +def chunks(root: Path = REPO, files=CORPUS): + """Yield (citation, text). A chunk never crosses a heading.""" + for name in files: + heading, buf = "top", [] + + def flush(): + words = " ".join(buf).split() + for i in range(0, len(words), CHUNK_WORDS): + piece = " ".join(words[i:i + CHUNK_WORDS]) + if len(piece.split()) >= 8: + yield f"{name}#{heading}", piece + + in_code = False + for line in (root / name).read_text().splitlines(): + if line.startswith("```"): + in_code = not in_code + if not in_code and re.match(r"#{1,3} ", line): + yield from flush() + heading, buf = line.lstrip("# ").strip(), [] + else: + buf.append(line) + yield from flush() + + +class Embedder: + def __init__(self): + self.tok = AutoTokenizer.from_pretrained(MODEL, revision=REVISION) + self.model = AutoModel.from_pretrained(MODEL, revision=REVISION).eval() + + @torch.no_grad() + def __call__(self, texts, batch=32): + out = [] + for i in range(0, len(texts), batch): + # 1. Text to token ids. Short texts are padded so a batch is one tensor. + enc = self.tok(texts[i:i + batch], padding=True, truncation=True, max_length=256, return_tensors="pt") + # 2. One 384-number vector per token: shape (batch, tokens, 384). + tokens = self.model(**enc).last_hidden_state + # 3. Average the token vectors, but not the padding. The mask is 1 for + # real tokens and 0 for padding. + mask = enc["attention_mask"].unsqueeze(-1).float() + pooled = (tokens * mask).sum(dim=1) / mask.sum(dim=1).clamp(min=1e-9) + # 4. Length 1, so the store's similarity is the cosine of the angle. + out.extend(F.normalize(pooled, dim=1).tolist()) + return out + + +class Index: + def __init__(self, folder: str): + self.ctx = Context() + self.db = Database(DatabaseOptions(stores_folders=[folder], type=DatabaseType.Standalone)) + + def build(self, embed: Embedder, root: Path = REPO, files=CORPUS) -> int: + rows = list(chunks(root, files)) + vectors = embed([text for _, text in rows]) + items = [Item(id=f"{i}", vector=v, payload={"source": src, "text": text}) + for i, ((src, text), v) in enumerate(zip(rows, vectors))] + tx = self.db.begin_transaction(self.ctx) # all chunks land together or not at all + self.db.open_vector_store(self.ctx, tx, "docs").upsert_batch(self.ctx, items) + tx.commit(self.ctx) + return len(items) + + def ask(self, embed: Embedder, question: str, k: int = 3, min_score: float = MIN_SCORE): + """The k nearest chunks, best first, or [] when the best one is below min_score.""" + tx = self.db.begin_transaction(self.ctx) + hits = self.db.open_vector_store(self.ctx, tx, "docs").query(self.ctx, vector=embed([question])[0], k=k) + tx.commit(self.ctx) + if not hits or hits[0].score < min_score: + return [], (hits[0].score if hits else 0.0) + return hits, hits[0].score + + +def main(): + ap = argparse.ArgumentParser(description=__doc__.split("\n")[0]) + ap.add_argument("question") + ap.add_argument("-k", type=int, default=3) + ap.add_argument("--min-score", type=float, default=MIN_SCORE) + args = ap.parse_args() + + embed = Embedder() + index = Index(tempfile.mkdtemp(prefix="joltrin-hf-")) + n = index.build(embed) + hits, best = index.ask(embed, args.question, args.k, args.min_score) + if not hits: + print(f"I don't know. Nothing in {n} indexed passages is close enough (best score {best:.2f}, needs {args.min_score:.2f}).") + return 1 + for h in hits: + print(f"[{h.score:.2f}] {h.payload['source']}\n {h.payload['text'][:400]}\n") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/examples/hf_grounded_search/test_search.py b/examples/hf_grounded_search/test_search.py new file mode 100644 index 000000000..5f4e7c59b --- /dev/null +++ b/examples/hf_grounded_search/test_search.py @@ -0,0 +1,81 @@ +"""Checks the search against questions whose answers are known, and prints the scores. + + PYTHONPATH=bindings/python python examples/hf_grounded_search/test_search.py + +It asserts three things: + - the score the Joltrin store returns is the cosine similarity of the two vectors, + - each in-domain question finds its source file in the top 3, + - each question about something else gets "I don't know". +The last group is probes: questions near the docs' topic that the docs do not answer. +They are printed, not asserted, because a score threshold cannot tell "close to the +topic" from "answered by the docs", and the output shows where that breaks. +""" +import tempfile + +import torch + +import search + +IN_DOMAIN = [ + ("How long are remembered blocks kept before they stop applying?", {"AGENT_PROTOCOLS.md"}), + ("Which state must hold before a node can be drained?", {"AGENT_PROTOCOLS.md"}), + ("What do I set to serve my own runbooks instead of the example?", {"AGENT_PROTOCOLS.md"}), + ("Is the precedence check full LTL or CTL model checking?", {"AGENT_PROTOCOLS.md", "MCP_A2A_AND_VERIFICATION_ENGINE.md"}), + ("Who is this for if I run a platform or SRE team?", {"WHO_IS_IT_FOR.md"}), + ("Did the agents act on the feedback in a block?", {"AGENT_BARRIER_TESTS.md"}), + ("How do I add the library to a Go project?", {"README.md"}), + ("How do I register the server with Claude Code, Codex and the Gemini CLI?", {"AGENT_PROTOCOLS.md"}), +] +OUT_OF_DOMAIN = [ + "What is the capital of France?", + "How do I bake sourdough bread?", + "Which graphics card should I buy for gaming?", + "What is the weather in San Diego today?", + "Explain how photosynthesis works.", + "Who won the 2018 football World Cup?", +] +NEAR_DOMAIN_PROBES = [ + "How do I configure PostgreSQL streaming replication?", + "How do I write a Kubernetes liveness probe?", + "What is the difference between TCP and UDP?", +] + +embed = search.Embedder() +index = search.Index(tempfile.mkdtemp(prefix="joltrin-hf-test-")) +n = index.build(embed) +print(f"indexed {n} chunks with {search.MODEL}@{search.REVISION[:7]}") + +# 1. The store's score is the cosine similarity. Vectors have length 1, so that is a dot product. +question = "What do I set to serve my own runbooks instead of the example?" +q = embed([question]) +hits, best = index.ask(embed, question, k=1, min_score=0.0) +chunk_vec = torch.tensor(embed([hits[0].payload["text"]])[0]) +cosine = float(torch.dot(torch.tensor(q[0]), chunk_vec)) +assert abs(cosine - hits[0].score) < 0.01, f"store score {hits[0].score} is not the cosine {cosine}" +print(f"store score {hits[0].score:.4f} matches cosine {cosine:.4f}") + +# 2. In-domain questions find their source in the top 3. +low_in = 1.0 +for question, sources in IN_DOMAIN: + hits, best = index.ask(embed, question, k=3, min_score=0.0) + found = [h.payload["source"].split("#")[0].split("/")[-1] for h in hits] + low_in = min(low_in, best) + print(f" in {best:.2f} top3={found} {question}") + assert sources & set(found), f"{question!r} did not find {sources} in {found}" + +# 3. Questions about something else are refused, with room to spare. +high_out = 0.0 +for question in OUT_OF_DOMAIN: + hits, best = index.ask(embed, question) + high_out = max(high_out, best) + print(f" out {best:.2f} {'answered' if hits else 'refused '} {question}") + assert not hits, f"{question!r} should be refused, best score {best:.2f}" + +print(f"\nlowest in-domain best score {low_in:.2f}, highest out-of-domain {high_out:.2f}, threshold {search.MIN_SCORE:.2f}") +assert high_out < search.MIN_SCORE < low_in, "the threshold no longer separates the two groups" + +# Probes: printed only. +for question in NEAR_DOMAIN_PROBES: + hits, best = index.ask(embed, question) + print(f" near {best:.2f} {'ANSWERED' if hits else 'refused '} {question}") +print("ok") From e95f371efbd954b8755af7a302a4ba7080336056 Mon Sep 17 00:00:00 2001 From: Gerard Louis Recinto Date: Thu, 8 Oct 2026 02:18:12 -0700 Subject: [PATCH 2/4] list the grounded search example in the examples README --- examples/README.md | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/examples/README.md b/examples/README.md index fe2b0c968..49a03e369 100644 --- a/examples/README.md +++ b/examples/README.md @@ -44,6 +44,11 @@ Demonstrates a standalone in-memory KnowledgeBase workflow with nested categorie - **Key Feature**: `memory.KnowledgeBase` + `memory.NewStore` - **Why**: Gives you a minimal app you can run directly to see the high-level API in action. +### 6. Grounded search with Hugging Face embeddings (`hf_grounded_search`) +Embeds Joltrin's own docs with `all-MiniLM-L6-v2`, stores the vectors in a Joltrin vector store through the Python bindings, and answers questions with citations or says "I don't know". The embedding step is written with `transformers` and `torch` directly so each part is visible. +- **Key Feature**: `sop.ai` `upsert_batch` and `query`, with the model pinned to one Hub commit. +- **Check**: `PYTHONPATH=bindings/python python examples/hf_grounded_search/test_search.py` (needs the native library built for your machine) + --- ## ▶️ Running the Examples From 106a5b5ff64a8347fb0e36d368e4bd713232dc65 Mon Sep 17 00:00:00 2001 From: Gerard Louis Recinto Date: Thu, 8 Oct 2026 02:48:58 -0700 Subject: [PATCH 3/4] pin transformers and huggingface-hub to the versions the example was run with --- examples/hf_grounded_search/requirements.txt | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/examples/hf_grounded_search/requirements.txt b/examples/hf_grounded_search/requirements.txt index 769523f42..58c5d1c70 100644 --- a/examples/hf_grounded_search/requirements.txt +++ b/examples/hf_grounded_search/requirements.txt @@ -1,3 +1,3 @@ torch==2.14.1 -transformers -huggingface-hub +transformers==5.19.0 +huggingface-hub==1.33.0 From 25f21d184994651e6cc39fdb1cb2e160d2e2d3ae Mon Sep 17 00:00:00 2001 From: Gerard Louis Recinto Date: Thu, 8 Oct 2026 10:48:22 -0700 Subject: [PATCH 4/4] hf example: remove the temporary vector store folder when done --- examples/hf_grounded_search/search.py | 7 ++++--- examples/hf_grounded_search/test_search.py | 3 ++- 2 files changed, 6 insertions(+), 4 deletions(-) diff --git a/examples/hf_grounded_search/search.py b/examples/hf_grounded_search/search.py index 402408720..5ddae9ae3 100644 --- a/examples/hf_grounded_search/search.py +++ b/examples/hf_grounded_search/search.py @@ -131,9 +131,10 @@ def main(): args = ap.parse_args() embed = Embedder() - index = Index(tempfile.mkdtemp(prefix="joltrin-hf-")) - n = index.build(embed) - hits, best = index.ask(embed, args.question, args.k, args.min_score) + with tempfile.TemporaryDirectory(prefix="joltrin-hf-") as tmp: + index = Index(tmp) + n = index.build(embed) + hits, best = index.ask(embed, args.question, args.k, args.min_score) if not hits: print(f"I don't know. Nothing in {n} indexed passages is close enough (best score {best:.2f}, needs {args.min_score:.2f}).") return 1 diff --git a/examples/hf_grounded_search/test_search.py b/examples/hf_grounded_search/test_search.py index 5f4e7c59b..5463eb014 100644 --- a/examples/hf_grounded_search/test_search.py +++ b/examples/hf_grounded_search/test_search.py @@ -41,7 +41,8 @@ ] embed = search.Embedder() -index = search.Index(tempfile.mkdtemp(prefix="joltrin-hf-test-")) +tmp = tempfile.TemporaryDirectory(prefix="joltrin-hf-test-") # removed when the script exits +index = search.Index(tmp.name) n = index.build(embed) print(f"indexed {n} chunks with {search.MODEL}@{search.REVISION[:7]}")