Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
79 changes: 79 additions & 0 deletions examples/captains_view.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,79 @@
"""The captain's view — end-to-end demo of The Grand Quilt P0 stack.

Plays the real captured vessel stream (same fixtures as PR #1's tests)
through the emitter + Jev gate (stubbed backend so it runs offline), then
renders STATE.md — the one page a human reads in thirty seconds.

Run: python3 examples/captains_view.py
Needs: the fixture stream at ../git-agent/tests/fixtures or /tmp/git-agent/...
"""

from __future__ import annotations

import json
import sys
import tempfile
from pathlib import Path

sys.path.insert(0, str(Path(__file__).parent.parent / "src"))

from git_agent.jev_gate import JevGate, Judgment
from git_agent.projection import render_state_md
from git_agent.quilt_emit import QuiltEmitter

FIXTURE_CANDIDATES = [
Path(__file__).parent.parent / "tests" / "fixtures",
Path("/tmp/git-agent/tests/fixtures"),
]
EVENT_ORDER = ["session_start", "worklog", "task_completion", "promotion",
"fence", "skill", "snapshot", "heartbeat", "session_end"]


class DemoBackend:
"""Pretends to be Jev: scores worklog-style events high, vague ones low."""

def available(self):
return True

def decide_batch(self, state, questions, model="jev-latest"):
ev = json.loads(state)["vessel_event"] if isinstance(state, str) else state["vessel_event"]
etype = ev.get("type", "")
if etype == "worklog":
summary, target = str(ev.get("summary", "")), str(ev.get("target", ""))
concrete = any(k in summary for k in ("branch", "PR", "commit", "merge")) and bool(target.strip())
score = 0.93 if concrete else 0.08
elif etype == "promotion":
score = 0.9 if ev.get("from_stage") and ev.get("to_stage") else 0.2
else: # task_completion
score = 0.85 if "success" in ev else 0.3
return ([Judgment(kind="noul", value=score, confidence=score) for _ in questions],
{"latency_ms": 0.0, "questions": len(questions)})


def main():
fixtures = next((c for c in FIXTURE_CANDIDATES if c.exists()), None)
if fixtures is None:
print("fixtures not found — run tests/_capture_fixtures.py first")
sys.exit(1)

with tempfile.TemporaryDirectory() as td:
emitter = QuiltEmitter(wal_path=Path(td) / "quilt.jsonl")
gate = JevGate(DemoBackend(), threshold=0.5)

# a normal session, then a vague one that deserves flagging
for i, name in enumerate(EVENT_ORDER):
ev = json.loads((fixtures / f"{name}.json").read_text())
ev["event_id"] = f"evt-demo-{i:02d}"
gate.ingest(emitter, ev)
vague = json.loads((fixtures / "worklog.json").read_text())
vague.update(event_id="evt-demo-vague",
summary="did some stuff", target="")
gate.ingest(emitter, vague)

print(render_state_md(emitter))
print(f"WAL: {emitter.wal_path}")
print(f"verify: {emitter.verify()}")


if __name__ == "__main__":
main()
130 changes: 130 additions & 0 deletions src/git_agent/jev_gate.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,130 @@
"""
Jev judgment gate — the attention layer of The Grand Quilt (P0.2).

Deterministic validators check shape; this gate checks substance. Events that
make *claims about work* (worklog entries, task completions, promotions) are
judged by a decision-model backend (TypeSafe Jev by default) before they earn
unflagged WAL space. Below threshold, the claim does not enter the WAL as
work — an EFFECT {kind: "flagged"} line referencing it does, so the rejection
is legible, replayable, and never silent. The gate flags; it never blocks,
never drops, never crashes the loop.

Policy (from JEV session 16): aim nouls at the specific claim; trust the
score; let confidence route. Structural events (identity, heartbeat, session
bookkeeping) are not claims and are never substance-judged.

Offline: no backend / no key -> events pass unjudged, receipt
{"jev": "skipped"}. The gate degrades to a no-op, not a wall.
"""

from __future__ import annotations

from dataclasses import dataclass
from typing import Any, Dict, Optional

# Event types that make verifiable claims about work. Everything else is
# structural — presence-checked by the emitter, not substance-judged.
JUDGED_TYPES = ("worklog", "task_completion", "promotion")

DEFAULT_THRESHOLD = 0.5

_VERIFIABLE_WORK_Q = (
"Does this vessel event describe concrete, verifiable engineering work "
"(a named branch, commit, PR, repo, or artifact) — as opposed to vague "
"activity claims that no one could check?"
)

_JUDGED_STATE_FIELDS = ("type", "action", "target", "summary",
"outcome", "success", "from_stage", "to_stage")


@dataclass
class Judgment:
"""One typed answer from a decision-model backend."""
kind: str
value: Any
confidence: Optional[float] = None


@dataclass
class GateVerdict:
event_type: str
judged: bool
flagged: bool
score: Optional[float]
question: Optional[str]


class JevGate:
"""Substance gate over a QuiltEmitter.

Parameters
----------
backend:
Any object with ``available() -> bool`` and
``decide_batch(state, questions, model) -> (judgments, meta)`` where
each judgment has ``.value``. The production backend is
jev-quilt's TypeSafeBackend; pass None for offline mode.
threshold:
Scores below this are flagged (not blocked).
question:
The noul aimed at each judged event's specific claim (session 16:
specific nouls move on quality; global scores don't).
"""

def __init__(self, backend=None, threshold: float = DEFAULT_THRESHOLD,
question: str = _VERIFIABLE_WORK_Q):
self.backend = backend
self.threshold = threshold
self.question = question

def _online(self) -> bool:
return self.backend is not None and bool(self.backend.available())

def judge(self, event: Dict[str, Any]) -> GateVerdict:
"""Judge one event without writing anything."""
etype = event.get("type", "")
if etype not in JUDGED_TYPES or not self._online():
return GateVerdict(etype, judged=False, flagged=False,
score=None, question=None)
state = {"vessel_event": {k: event[k] for k in _JUDGED_STATE_FIELDS
if k in event}}
judgments, _ = self.backend.decide_batch(
state, [{"name": "substance", "type": "noul",
"instructions": self.question}])
score = float(judgments[0].value) if judgments else None
flagged = score is not None and score < self.threshold
return GateVerdict(etype, judged=True, flagged=flagged,
score=score, question=self.question)

def ingest(self, emitter, event: Dict[str, Any]) -> Dict[str, Any]:
"""Judge + persist one event. Returns the WAL line written.

Flagged claim -> EFFECT vessel/flags line referencing the original
event (the claim does not earn unflagged WAL space;
its rejection is the record).
Judged, passed -> normal emitter line + receipts.jev = score.
Not judged -> normal emitter line; receipts.jev = "skipped" if it
was a claim type judged offline, else no receipts.
"""
verdict = self.judge(event)
if verdict.flagged:
from .quilt_emit import _map
original_op = _map({**event, "event_id": "x", "timestamp": "t"})["op"]
return emitter.append_line(
"EFFECT", "vessel/flags",
{"kind": "flagged",
"original_type": event["type"],
"original_op": original_op,
"original_event_id": event["event_id"],
"score": verdict.score,
"threshold": self.threshold},
event["event_id"] + "/flagged",
event["timestamp"],
extra={"receipts": {"jev": verdict.score, "gate": "flagged"}})
receipts = None
if verdict.judged:
receipts = {"jev": verdict.score}
elif event.get("type") in JUDGED_TYPES:
receipts = {"jev": "skipped"}
return emitter.ingest(event, extra={"receipts": receipts} if receipts else None)
123 changes: 123 additions & 0 deletions src/git_agent/projection.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,123 @@
"""
The captain's VIEW — P0.3 of The Grand Quilt.

Given a vessel's quilt WAL, render the one page a human actually reads:
identity, career trajectory, worklog digest, judgment receipts, quality flags.
The last mile is rendering, not archaeology — thirty seconds of reading
instead of a diff nobody opened.

Two renderers:
render_state_json(emitter) -> dict (machine view; feeds tidepool/gauge)
render_state_md(emitter) -> str (human view; STATE.md)

Everything is derived from the WAL alone — replay it, render it, and the view
is exactly as trustworthy as the chain underneath it.
"""

from __future__ import annotations

from typing import Any, Dict, List

FLAG_CELL = "vessel/flags"


def _collect(emitter) -> Dict[str, Any]:
lines = emitter.wal()
identity = None
promotions: List = []
fences: List = []
skills: List = []
worklog_actions: List[str] = []
receipts = {"judged": 0, "scores": [], "skipped": 0}
flags = 0
stage = None

for line in lines:
op, cell, args = line.get("op"), line.get("cell"), line.get("args", {})
if op == "BIND" and cell == "vessel/identity":
identity = args
elif op == "BIND" and cell == "vessel/skills":
if args.get("skill") not in skills:
skills.append(args.get("skill"))
elif op == "LINK" and cell == "vessel/worklog":
worklog_actions.append(args.get("action", "?"))
elif op == "EFFECT":
kind = args.get("kind")
if kind == "promotion":
promotions.append([args.get("from_stage"), args.get("to_stage")])
stage = args.get("to_stage")
elif kind == "fence":
if args.get("fence_name") not in fences:
fences.append(args.get("fence_name"))
elif kind == "flagged":
flags += 1
elif op == "VIEW" and cell == "vessel/state":
stage = args.get("stage") or stage
rec = line.get("receipts", {}).get("jev")
if rec == "skipped":
receipts["skipped"] += 1
elif isinstance(rec, (int, float)):
receipts["judged"] += 1
receipts["scores"].append(rec)

return {"identity": identity, "stage": stage, "promotions": promotions,
"fences": fences, "skills": skills,
"worklog_count": len(worklog_actions),
"worklog_actions": worklog_actions,
"flags": flags, "receipts": receipts,
"lines": len(lines)}


def render_state_json(emitter) -> Dict[str, Any]:
c = _collect(emitter)
return {"identity": c["identity"], "stage": c["stage"],
"career": {"promotions": c["promotions"], "fences": c["fences"],
"skills": c["skills"]},
"worklog_count": c["worklog_count"],
"worklog_actions": c["worklog_actions"],
"flags": c["flags"], "receipts": c["receipts"],
"wal_lines": c["lines"]}


def render_state_md(emitter) -> str:
c = _collect(emitter)
if c["lines"] == 0:
return "# Vessel\n\n_no events recorded — this vessel has not spoken._\n"

ident = c["identity"] or {}
out = ["# Vessel — the captain's view", ""]
out.append("## Identity")
out.append(f"- **{ident.get('name', 'unnamed')}** — {ident.get('designation', 'unknown designation')}")
out.append(f"- version {ident.get('version', '?')} · domains: {', '.join(ident.get('domains', []) or ['?'])}")
out.append("")
out.append("## Career")
out.append(f"- stage: **{c['stage'] or 'unknown'}**")
if c["promotions"]:
trail = " → ".join(f"{a}->{b}" for a, b in c["promotions"])
out.append(f"- promotions: {trail}")
if c["fences"]:
out.append(f"- fences: {', '.join(c['fences'])}")
if c["skills"]:
out.append(f"- skills: {', '.join(c['skills'])}")
out.append("")
out.append("## Worklog")
if c["worklog_actions"]:
counts: Dict[str, int] = {}
for a in c["worklog_actions"]:
counts[a] = counts.get(a, 0) + 1
out.append(", ".join(f"{a} ×{n}" for a, n in counts.items()))
else:
out.append("_no work recorded yet._")
out.append("")
out.append("## Receipts")
r = c["receipts"]
if r["scores"]:
mean_s = sum(r["scores"]) / len(r["scores"])
out.append(f"- judged: {r['judged']} (mean {mean_s:.2f})")
else:
out.append(f"- judged: {r['judged']}")
out.append(f"- skipped (offline): {r['skipped']}")
out.append(f"- **FLAGGED: {c['flags']}**" if c["flags"] else "- flagged: 0")
out.append("")
out.append(f"_rendered from {c['lines']} WAL lines — as trustworthy as the chain._")
return "\n".join(out) + "\n"
Loading