diff --git a/.gitignore b/.gitignore index 18f10c5..c948980 100644 --- a/.gitignore +++ b/.gitignore @@ -3,3 +3,6 @@ __pycache__/ .venv/ *.db .codex-logger/ +*.egg-info/ +build/ +dist/ diff --git a/README.md b/README.md index b768a2a..5e6df89 100644 --- a/README.md +++ b/README.md @@ -37,16 +37,22 @@ Claude Code — would silently miss a large part of what a Codex agent does: `codex-logger` bypasses hooks entirely. It reads the append-only **rollout JSONL files** Codex already writes to -`~/.codex/sessions/YYYY/MM/DD/rollout-.jsonl`, which record the complete event -stream. That means it: +`~/.codex/sessions/YYYY/MM/DD/rollout-.jsonl`, giving you a queryable index of +sessions, turns, tool calls, messages, models, tokens, and subagent relationships. +That means it: -- captures **everything** — shell, `apply_patch`, `write_stdin`, MCP calls, - subagents — not just the shell subset hooks expose, +- covers the tools hooks don't — `apply_patch`, `write_stdin`, MCP calls, and + subagent activity, not just the shell subset, - works **retroactively** on sessions already on disk, with nothing installed into Codex and no change to your workflow, - runs **local-first** — a zero-config SQLite file by default, optional Postgres only if you want to. +It reports what Codex reports and doesn't invent what it doesn't: a tool call's +success/failure comes from the structured `*_end` event Codex emits for it +(`exec_command_end.exit_code`, `patch_apply_end.success`), and a call whose outcome +Codex never states is kept as `unknown` rather than guessed. + > Codex changes fast. The hook behavior above is what was observed on the version > noted, not a permanent claim — the point is that reading rollout files is robust > to whichever tools Codex routes through hooks in a given release. @@ -71,8 +77,13 @@ Each rollout event maps to normalized columns: | `session_meta` | session id, `parent_thread_id` (subagent → parent link), `cwd`, `originator` (e.g. `codex_vscode`), `cli_version`, subagent type (e.g. `guardian`) | | `turn_context` | active `model` per turn (`gpt-5.4-mini`, `codex-auto-review`, …) | | `event_msg / token_count` | cumulative session tokens + **per-turn** usage (input / cached / output / reasoning / total) | -| `response_item / function_call` + `function_call_output` | every tool call — `exec_command`, `apply_patch`, `write_stdin`, MCP — paired by `call_id`, with exit code + success/failure | -| `response_item / message` | user prompts + assistant text | +| `response_item / function_call` (+`_output`) | every tool call — `exec_command`, `apply_patch`, `write_stdin`, MCP — paired by `call_id`, stamped with its `turn_id` | +| `event_msg / *_end` | authoritative outcome per call: `exec_command_end.exit_code`, `patch_apply_end.success` (a non-shell call with no reported status stays `unknown`) | +| `response_item / message` | user prompts + assistant text (the `event_msg` `agent_message`/`user_message` events are the streamed duplicates — not re-ingested) | + +Because every tool call and message carries the `turn_id` it happened in, you can +ask turn-level questions — which prompt triggered the failed patch, which subagent +turn burned the most tokens, which model was active for a given MCP call. ## How it compares @@ -87,14 +98,19 @@ version noted above, and may change in future Codex releases. ## Quick start -Nothing to install for the default SQLite backend — it's pure stdlib. +Nothing to install for the default SQLite backend — it's pure stdlib. Run it from +a checkout, or install the `codex-logger` command: + +```bash +pipx install git+https://github.com/kkrlstrm/codex-logger # or: pip install -e . +``` ```bash -cd codex-logger -python3 -m codex_logger ingest # load all rollout files on disk -python3 -m codex_logger sessions # list recent sessions -python3 -m codex_logger stats # tokens by model + top tools -python3 -m codex_logger inspect # one session in detail (id prefix ok) +codex-logger ingest # load all rollout files on disk +codex-logger sessions # list recent sessions +codex-logger stats # tokens by model + top tools +codex-logger inspect # one session in detail (id prefix ok) +# equivalently, from a checkout without installing: python3 -m codex_logger ``` `sessions` gives you the recent history at a glance: @@ -118,21 +134,23 @@ time ...T19:29:48Z -> ...T19:51:43Z tokens in=182,140 cached=160,448 out=48,435 reasoning=27,051 total=257,626 turns 6 tool_calls 41 -tool calls: - 4 exec_command success {"cmd":"pytest -q","workdir":"~/projects/acme", ...} - 12 apply_patch success {"changes":{"src/app.py":{"update": ...}}} - 17 write_stdin success {"stdin":"y\n", ...} - 26 mcp.fetch success {"url":"https://api.example.com/...", ...} +tool calls: (seq · turn · tool · status · args) + 4 019xxab1 exec_command success {"cmd":"pytest -q", ...} + 12 019xxab1 apply_patch success {"changes":{"src/app.py": ...}} + 17 019xxcd2 apply_patch failure {"changes":{"src/db.py": ...}} + 26 019xxcd2 mcp.fetch success {"url":"https://api.example.com/...", ...} ``` -Illustrative session; paths and arguments elided. +Illustrative session; paths and arguments elided. Each call shows the turn it +ran in, and a status resolved from Codex's own `*_end` events. ## Commands ```bash -python3 -m codex_logger ingest [--watch] [--interval N] [--force] [--verbose] -python3 -m codex_logger sessions [--limit N] [--days N] -python3 -m codex_logger inspect -python3 -m codex_logger stats [--days N] +codex-logger ingest [--watch] [--interval N] [--force] [--verbose] +codex-logger sessions [--limit N] [--days N] +codex-logger inspect +codex-logger stats [--days N] +codex-logger install-launchd [--interval N] [--print] # macOS scheduling ``` `ingest` is incremental — unchanged files are skipped by size+mtime, so re-running @@ -162,8 +180,16 @@ ORDER BY failures DESC; SELECT subagent_type, COUNT(*) AS sessions, SUM(total_tokens) AS tokens FROM sessions GROUP BY subagent_type; + +-- Which turn triggered a failed patch (turn-level attribution) +SELECT session_id, turn_id, tool_name, status +FROM tool_calls +WHERE tool_name = 'apply_patch' AND status = 'failure'; ``` +On Postgres the tables are prefixed (`codex_sessions`, `codex_tool_calls`, …); +on SQLite they're bare as shown. The CLI handles the difference for you. + ## Storage Default: SQLite at `~/.codex-logger/codex.db` — zero-config, works immediately. @@ -171,33 +197,51 @@ Default: SQLite at `~/.codex-logger/codex.db` — zero-config, works immediately To co-locate with cc-logger's Postgres/Neon warehouse (one dashboard across Claude Code + Codex), point it at a `postgresql://` URL. Tables are prefixed `codex_*` and every row carries `source='codex'`, so a `UNION` view against the cc-logger tables -is trivial: +is trivial — and all four commands work against Postgres, not just `ingest`: ```bash export CODEX_LOGGER_DB="postgresql://…/neondb?sslmode=require" pip install 'psycopg[binary]' -python3 -m codex_logger ingest +codex-logger ingest +codex-logger stats # queries codex_* automatically ``` > The SQLite path is exercised end-to-end in the test suite. The Postgres backend -> mirrors the same schema but isn't yet covered by a live-DB test — that's the next -> hardening step. +> mirrors the same schema (and the CLI routes to the prefixed tables), but isn't yet +> covered by a live-DB integration test — that's the next hardening step. ## Schema `sessions`, `tool_calls`, `messages`, `turns`, plus `ingest_state` for incremental -bookkeeping. See [`codex_logger/store.py`](codex_logger/store.py) for the DDL. +bookkeeping. `tool_calls` and `messages` each carry a `turn_id`. See +[`codex_logger/store.py`](codex_logger/store.py) for the DDL. + +## Privacy & security + +`codex-logger` stores your local Codex history verbatim: prompts, assistant +messages, tool arguments, and command output — which can include file contents, +internal URLs, and secrets that scrolled through a terminal. Treat the database as +sensitive developer telemetry: + +- The default lives at `~/.codex-logger/codex.db` on your machine. Keep it there. +- Don't commit it to a repo or sync it to shared/cloud storage. +- If you point it at Postgres, use a private database with least-privilege access. + +Tool output is capped per row (200 KB) to bound runaway logs; a redaction mode for +prompts/outputs is a planned option, not yet implemented. -## Run it on a schedule (launchd) +## Run it on a schedule (launchd, macOS) ```bash -cp launchd/com.kaikarlstrom.codex-logger.plist ~/Library/LaunchAgents/ -launchctl load ~/Library/LaunchAgents/com.kaikarlstrom.codex-logger.plist -launchctl start com.kaikarlstrom.codex-logger # run now +codex-logger install-launchd --interval 300 # writes a plist wired to this machine ``` -Ingests changed rollout files every 5 minutes. Logs to -`~/Library/Logs/codex-logger.{out,err}.log`. +This generates `~/Library/LaunchAgents/com.codex-logger.ingest.plist` with your real +Python path and repo path filled in (use `--print` to review it first), then tells +you the `launchctl load` command to run. It ingests changed rollout files every +5 minutes — near-zero work when idle — logging to +`~/Library/Logs/codex-logger.{out,err}.log`. A hand-editable template lives in +[`launchd/`](launchd/). ## Tests @@ -206,9 +250,10 @@ python3 -m unittest discover -s tests -v ``` Tests run against synthetic Codex `0.140`-style rollout events and cover session -identity, model extraction, tool calls, exit status, token accounting, message -extraction, malformed-line tolerance, and idempotent SQLite writes. The parser is -fail-open — a malformed JSONL line is skipped, never fatal. +identity, model extraction, tool calls, **status resolved from `*_end` events**, +**per-turn attribution**, token accounting, message extraction, malformed-line +tolerance, and idempotent SQLite writes. The parser is fail-open — a malformed JSONL +line is skipped, never fatal. CI runs the suite on Python 3.10–3.13. ## Where this fits diff --git a/codex_logger/cli.py b/codex_logger/cli.py index 814c049..7f34b12 100644 --- a/codex_logger/cli.py +++ b/codex_logger/cli.py @@ -4,6 +4,7 @@ python3 -m codex_logger sessions [--limit N] [--days N] python3 -m codex_logger inspect python3 -m codex_logger stats [--days N] + python3 -m codex_logger install-launchd [--interval N] [--print] DB target: --db, or $CODEX_LOGGER_DB, else ~/.codex-logger/codex.db (SQLite). Pass a postgresql:// URL to co-locate with cc-logger. @@ -11,11 +12,14 @@ from __future__ import annotations import argparse +import os import sys from .ingest import DEFAULT_SESSIONS_DIR, ingest_once, watch from .store import open_store +LAUNCHD_LABEL = "com.codex-logger.ingest" + def _fmt_int(n): return f"{n:,}" if isinstance(n, int) else (n or "") @@ -41,7 +45,7 @@ def cmd_sessions(a): rows = store.query( f"""SELECT session_id, model, originator, subagent_type, num_tool_calls, total_tokens, started_at, cwd - FROM sessions {where} + FROM {store.table('sessions')} {where} ORDER BY started_at DESC LIMIT ?""", (*params, a.limit), ) @@ -62,7 +66,8 @@ def cmd_sessions(a): def cmd_inspect(a): store = open_store(a.db) rows = store.query( - "SELECT * FROM sessions WHERE session_id LIKE ? ORDER BY started_at DESC LIMIT 1", + f"SELECT * FROM {store.table('sessions')} WHERE session_id LIKE ? " + "ORDER BY started_at DESC LIMIT 1", (a.session + "%",), ) if not rows: @@ -85,14 +90,16 @@ def cmd_inspect(a): f"total={_fmt_int(s['total_tokens'])}") print(f"turns {s['num_turns']} tool_calls {s['num_tool_calls']}") calls = store.query( - "SELECT seq, tool_name, status, exit_code, arguments FROM tool_calls " + f"SELECT seq, turn_id, tool_name, status, exit_code, arguments " + f"FROM {store.table('tool_calls')} " "WHERE session_id=? ORDER BY seq", (s["session_id"],)) if calls: print("\ntool calls:") for c in calls: arg = (c["arguments"] or "").replace("\n", " ") - print(f" {c['seq']:>3} {(c['tool_name'] or '?'):16.16} " - f"{(c['status'] or ''):8.8} {arg[:90]}") + turn = (c["turn_id"] or "")[:8] + print(f" {c['seq']:>3} {turn:8.8} {(c['tool_name'] or '?'):16.16} " + f"{(c['status'] or ''):8.8} {arg[:78]}") store.close() @@ -107,19 +114,79 @@ def cmd_stats(a): for r in store.query( f"""SELECT model, COUNT(*) n, SUM(num_tool_calls) calls, SUM(total_tokens) tok - FROM sessions {where} GROUP BY model ORDER BY tok DESC""", tuple(params)): + FROM {store.table('sessions')} {where} + GROUP BY model ORDER BY tok DESC""", tuple(params)): print(f" {(r['model'] or '-'):20.20} {r['n']:>4} sessions " f"{_fmt_int(r['calls'] or 0):>7} calls {_fmt_int(r['tok'] or 0):>11} tok") print("\ntop tools:") for r in store.query( - """SELECT tool_name, COUNT(*) n, + f"""SELECT tool_name, COUNT(*) n, SUM(CASE WHEN status='failure' THEN 1 ELSE 0 END) fails - FROM tool_calls GROUP BY tool_name ORDER BY n DESC LIMIT 15"""): + FROM {store.table('tool_calls')} + GROUP BY tool_name ORDER BY n DESC LIMIT 15"""): print(f" {(r['tool_name'] or '?'):20.20} {r['n']:>6} " f"{r['fails'] or 0} failures") store.close() +def _render_plist(interval: int, db: str) -> str: + """A launchd plist wired to THIS machine — real interpreter + repo path, + not a hardcoded template. Runs `-m codex_logger ingest` from the repo root.""" + from xml.sax.saxutils import escape + repo_root = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) + home = os.path.expanduser("~") + return f""" + + + + Label + {LAUNCHD_LABEL} + ProgramArguments + + {escape(sys.executable)} + -m + codex_logger + ingest + + WorkingDirectory + {escape(repo_root)} + EnvironmentVariables + + CODEX_LOGGER_DB + {escape(db)} + + StartInterval + {int(interval)} + RunAtLoad + + StandardOutPath + {escape(home)}/Library/Logs/codex-logger.out.log + StandardErrorPath + {escape(home)}/Library/Logs/codex-logger.err.log + + +""" + + +def cmd_install_launchd(a): + db = a.db or os.environ.get("CODEX_LOGGER_DB", "") + plist = _render_plist(a.interval, db) + if a.print: + print(plist, end="") + return + dest_dir = os.path.expanduser("~/Library/LaunchAgents") + os.makedirs(dest_dir, exist_ok=True) + dest = os.path.join(dest_dir, f"{LAUNCHD_LABEL}.plist") + with open(dest, "w", encoding="utf-8") as f: + f.write(plist) + print(f"wrote {dest}") + print("load it with:") + print(f" launchctl unload {dest} 2>/dev/null") + print(f" launchctl load {dest}") + print(f" launchctl start {LAUNCHD_LABEL} # run once now") + + def main(argv=None): p = argparse.ArgumentParser(prog="codex-logger", description="Codex CLI telemetry from rollout files") @@ -149,6 +216,14 @@ def main(argv=None): pt.add_argument("--days", type=int) pt.set_defaults(func=cmd_stats) + pl = sub.add_parser("install-launchd", + help="generate a launchd plist wired to this machine") + pl.add_argument("--interval", type=int, default=300, + help="seconds between ingests (default 300)") + pl.add_argument("--print", action="store_true", + help="print the plist instead of writing it") + pl.set_defaults(func=cmd_install_launchd) + a = p.parse_args(argv) a.func(a) diff --git a/codex_logger/parse.py b/codex_logger/parse.py index cdddd9c..6f87029 100644 --- a/codex_logger/parse.py +++ b/codex_logger/parse.py @@ -13,6 +13,13 @@ Tool calls are function_call / function_call_output pairs matched by call_id and cover every tool (exec_command, apply_patch, shell, MCP) — not shell-only, which is the whole reason this reads the rollout file instead of PreToolUse hooks. +Success/failure is resolved from the structured `*_end` event_msg events Codex +emits per call (`exec_command_end.exit_code`, `patch_apply_end.success`); a tool +whose outcome Codex doesn't report is preserved as `unknown` rather than guessed. + +Assistant + user text is taken from `response_item`/`message` (the durable record). +The `event_msg`/`agent_message`/`user_message` events are the streamed duplicates of +that same text, so they are intentionally not re-ingested (no double-counting). Pure stdlib, fail-open: a malformed line is skipped, never raised. """ @@ -48,6 +55,7 @@ class ToolCall: output: Optional[str] = None exit_code: Optional[int] = None status: str = "pending" # pending | success | failure | unknown + turn_id: Optional[str] = None ts: Optional[str] = None seq: int = 0 @@ -57,6 +65,7 @@ class Message: role: str text: str phase: Optional[str] = None + turn_id: Optional[str] = None ts: Optional[str] = None seq: int = 0 @@ -166,6 +175,18 @@ def parse_session(path: str) -> Optional[Session]: turns: dict[str, Turn] = {} saw_any = False + def touch_call(cid: str) -> ToolCall: + """Get-or-create the ToolCall for call_id, stamping the active turn.""" + nonlocal seq + tc = calls.get(cid) + if tc is None: + seq += 1 + tc = ToolCall(call_id=cid, seq=seq, turn_id=last_turn_id) + calls[cid] = tc + elif tc.turn_id is None: + tc.turn_id = last_turn_id + return tc + for obj in iter_lines(path): saw_any = True typ = obj.get("type") @@ -226,13 +247,34 @@ def parse_session(path: str) -> Optional[Session]: t.output_tokens = last.get("output_tokens", 0) t.reasoning_tokens = last.get("reasoning_output_tokens", 0) t.total_tokens = last.get("total_tokens", 0) + elif pt == "exec_command_end": + # Authoritative shell outcome — a real exit_code, not scraped text. + cid = payload.get("call_id") + if cid: + tc = touch_call(cid) + ec = payload.get("exit_code") + if ec is not None: + tc.exit_code = ec + tc.status = "success" if ec == 0 else "failure" + if not tc.output: + tc.output = payload.get("aggregated_output") or tc.output + elif pt == "patch_apply_end": + # apply_patch reports a success bool (and a 'declined' status when + # a guard rejects it) — the only reliable signal for a patch call. + cid = payload.get("call_id") + if cid: + tc = touch_call(cid) + succ = payload.get("success") + if succ is not None: + tc.status = "success" if succ else "failure" + if not tc.output: + tc.output = payload.get("stderr") or payload.get("stdout") or tc.output elif typ == "response_item": pt = payload.get("type") if pt in ("function_call", "custom_tool_call"): cid = payload.get("call_id") or payload.get("id") or f"_noid_{seq}" - seq += 1 - tc = calls.setdefault(cid, ToolCall(call_id=cid, seq=seq)) + tc = touch_call(cid) tc.tool_name = payload.get("name") or tc.tool_name args = payload.get("arguments") if args is None and "input" in payload: @@ -246,14 +288,16 @@ def parse_session(path: str) -> Optional[Session]: out = payload.get("output") if isinstance(out, (dict, list)): out = json.dumps(out) - tc = calls.setdefault(cid, ToolCall(call_id=cid, seq=seq)) + tc = touch_call(cid) tc.output = out + # exit-code scrape is a fallback only — never downgrade a status + # already resolved by an exec_command_end / patch_apply_end event. code = _parse_exit_code(out or "") - tc.exit_code = code - if code is None: - tc.status = "unknown" - else: + if code is not None: + tc.exit_code = code tc.status = "success" if code == 0 else "failure" + elif tc.status == "pending": + tc.status = "unknown" elif pt == "message": role = payload.get("role") if role in ("assistant", "user"): @@ -261,8 +305,8 @@ def parse_session(path: str) -> Optional[Session]: if text.strip(): seq += 1 s.messages.append(Message( - role=role, text=text, - phase=payload.get("phase"), ts=ts, seq=seq, + role=role, text=text, phase=payload.get("phase"), + turn_id=last_turn_id, ts=ts, seq=seq, )) if not saw_any or not s.session_id: diff --git a/codex_logger/store.py b/codex_logger/store.py index 5a21dce..52ac67a 100644 --- a/codex_logger/store.py +++ b/codex_logger/store.py @@ -47,6 +47,7 @@ session_id TEXT, call_id TEXT, seq INTEGER, + turn_id TEXT, tool_name TEXT, arguments TEXT, output TEXT, @@ -58,6 +59,7 @@ CREATE TABLE IF NOT EXISTS messages ( session_id TEXT, seq INTEGER, + turn_id TEXT, role TEXT, phase TEXT, text TEXT, @@ -162,26 +164,27 @@ def upsert_session(self, s: Session): for tc in s.tool_calls: self.conn.execute( """INSERT INTO tool_calls( - session_id, call_id, seq, tool_name, arguments, output, - exit_code, status, ts) - VALUES(?,?,?,?,?,?,?,?,?) + session_id, call_id, seq, turn_id, tool_name, arguments, + output, exit_code, status, ts) + VALUES(?,?,?,?,?,?,?,?,?,?) ON CONFLICT(session_id, call_id) DO UPDATE SET - seq=excluded.seq, tool_name=excluded.tool_name, + seq=excluded.seq, turn_id=excluded.turn_id, + tool_name=excluded.tool_name, arguments=excluded.arguments, output=excluded.output, exit_code=excluded.exit_code, status=excluded.status, ts=excluded.ts""", - (s.session_id, tc.call_id, tc.seq, tc.tool_name, + (s.session_id, tc.call_id, tc.seq, tc.turn_id, tc.tool_name, _clip(tc.arguments), _clip(tc.output), tc.exit_code, tc.status, tc.ts), ) for m in s.messages: self.conn.execute( - """INSERT INTO messages(session_id, seq, role, phase, text, ts) - VALUES(?,?,?,?,?,?) + """INSERT INTO messages(session_id, seq, turn_id, role, phase, text, ts) + VALUES(?,?,?,?,?,?,?) ON CONFLICT(session_id, seq) DO UPDATE SET - role=excluded.role, phase=excluded.phase, - text=excluded.text, ts=excluded.ts""", - (s.session_id, m.seq, m.role, m.phase, _clip(m.text), m.ts), + turn_id=excluded.turn_id, role=excluded.role, + phase=excluded.phase, text=excluded.text, ts=excluded.ts""", + (s.session_id, m.seq, m.turn_id, m.role, m.phase, _clip(m.text), m.ts), ) for t in s.turns: self.conn.execute( @@ -200,6 +203,11 @@ def upsert_session(self, s: Session): t.total_tokens, t.ts), ) + def table(self, name: str) -> str: + """Physical table name for a logical one. SQLite uses bare names; + Postgres prefixes `codex_` to share a warehouse with cc-logger.""" + return name + def commit(self): self.conn.commit() diff --git a/codex_logger/store_pg.py b/codex_logger/store_pg.py index 67f9d9d..1aaa15d 100644 --- a/codex_logger/store_pg.py +++ b/codex_logger/store_pg.py @@ -35,12 +35,13 @@ ingested_at TIMESTAMPTZ DEFAULT now() ); CREATE TABLE IF NOT EXISTS codex_tool_calls ( - session_id TEXT, call_id TEXT, seq INTEGER, tool_name TEXT, + session_id TEXT, call_id TEXT, seq INTEGER, turn_id TEXT, tool_name TEXT, arguments TEXT, output TEXT, exit_code INTEGER, status TEXT, ts TIMESTAMPTZ, PRIMARY KEY (session_id, call_id) ); CREATE TABLE IF NOT EXISTS codex_messages ( - session_id TEXT, seq INTEGER, role TEXT, phase TEXT, text TEXT, ts TIMESTAMPTZ, + session_id TEXT, seq INTEGER, turn_id TEXT, role TEXT, phase TEXT, text TEXT, + ts TIMESTAMPTZ, PRIMARY KEY (session_id, seq) ); CREATE TABLE IF NOT EXISTS codex_turns ( @@ -106,20 +107,23 @@ def upsert_session(self, s: Session): s.total_tokens, s.rollout_path)) for tc in s.tool_calls: cur.execute( - """INSERT INTO codex_tool_calls(session_id,call_id,seq,tool_name,arguments, - output,exit_code,status,ts) VALUES(%s,%s,%s,%s,%s,%s,%s,%s,%s) + """INSERT INTO codex_tool_calls(session_id,call_id,seq,turn_id,tool_name, + arguments,output,exit_code,status,ts) + VALUES(%s,%s,%s,%s,%s,%s,%s,%s,%s,%s) ON CONFLICT(session_id,call_id) DO UPDATE SET seq=EXCLUDED.seq, - tool_name=EXCLUDED.tool_name, arguments=EXCLUDED.arguments, + turn_id=EXCLUDED.turn_id, tool_name=EXCLUDED.tool_name, + arguments=EXCLUDED.arguments, output=EXCLUDED.output, exit_code=EXCLUDED.exit_code, status=EXCLUDED.status, ts=EXCLUDED.ts""", - (s.session_id, tc.call_id, tc.seq, tc.tool_name, _clip(tc.arguments), - _clip(tc.output), tc.exit_code, tc.status, tc.ts)) + (s.session_id, tc.call_id, tc.seq, tc.turn_id, tc.tool_name, + _clip(tc.arguments), _clip(tc.output), tc.exit_code, tc.status, tc.ts)) for m in s.messages: cur.execute( - """INSERT INTO codex_messages(session_id,seq,role,phase,text,ts) - VALUES(%s,%s,%s,%s,%s,%s) ON CONFLICT(session_id,seq) DO UPDATE SET - role=EXCLUDED.role, phase=EXCLUDED.phase, text=EXCLUDED.text, ts=EXCLUDED.ts""", - (s.session_id, m.seq, m.role, m.phase, _clip(m.text), m.ts)) + """INSERT INTO codex_messages(session_id,seq,turn_id,role,phase,text,ts) + VALUES(%s,%s,%s,%s,%s,%s,%s) ON CONFLICT(session_id,seq) DO UPDATE SET + turn_id=EXCLUDED.turn_id, role=EXCLUDED.role, phase=EXCLUDED.phase, + text=EXCLUDED.text, ts=EXCLUDED.ts""", + (s.session_id, m.seq, m.turn_id, m.role, m.phase, _clip(m.text), m.ts)) for t in s.turns: cur.execute( """INSERT INTO codex_turns(session_id,turn_id,model,input_tokens, @@ -132,6 +136,11 @@ def upsert_session(self, s: Session): (s.session_id, t.turn_id, t.model, t.input_tokens, t.cached_input_tokens, t.output_tokens, t.reasoning_tokens, t.total_tokens, t.ts)) + def table(self, name: str) -> str: + """Physical table name — Postgres prefixes `codex_` so Codex telemetry + shares a warehouse with cc-logger without colliding on `sessions` etc.""" + return "codex_" + name + def commit(self): self.conn.commit() diff --git a/launchd/com.kaikarlstrom.codex-logger.plist b/launchd/com.codex-logger.ingest.plist similarity index 56% rename from launchd/com.kaikarlstrom.codex-logger.plist rename to launchd/com.codex-logger.ingest.plist index 4f6cd67..957b5a6 100644 --- a/launchd/com.kaikarlstrom.codex-logger.plist +++ b/launchd/com.codex-logger.ingest.plist @@ -1,28 +1,32 @@ + Label - com.kaikarlstrom.codex-logger + com.codex-logger.ingest ProgramArguments - /usr/bin/python3 + __PYTHON__ -m codex_logger ingest WorkingDirectory - /Users/kaikarlstrom/kai-gtm-agents/codex-logger + __REPO_ROOT__ EnvironmentVariables + co-locate with cc-logger's warehouse. --> CODEX_LOGGER_DB @@ -34,8 +38,8 @@ StandardOutPath - /Users/kaikarlstrom/Library/Logs/codex-logger.out.log + __HOME__/Library/Logs/codex-logger.out.log StandardErrorPath - /Users/kaikarlstrom/Library/Logs/codex-logger.err.log + __HOME__/Library/Logs/codex-logger.err.log diff --git a/tests/test_parse.py b/tests/test_parse.py index b86815f..fa91c71 100644 --- a/tests/test_parse.py +++ b/tests/test_parse.py @@ -40,7 +40,7 @@ {"timestamp": "2026-07-06T13:34:12.500Z", "type": "response_item", "payload": {"type": "function_call_output", "call_id": "call_B", "output": "Process exited with code 1"}}, - {"timestamp": "2026-07-06T13:34:13.000Z", "type": "event_msg", + {"timestamp": "2026-07-06T13:34:12.550Z", "type": "event_msg", "payload": {"type": "token_count", "info": { "total_token_usage": {"input_tokens": 100, "cached_input_tokens": 10, "output_tokens": 20, "reasoning_output_tokens": 5, @@ -48,6 +48,26 @@ "last_token_usage": {"input_tokens": 100, "cached_input_tokens": 10, "output_tokens": 20, "reasoning_output_tokens": 5, "total_tokens": 120}}}}, + # --- turn 2: outcomes come from structured *_end events, not scraped text --- + {"timestamp": "2026-07-06T13:34:12.600Z", "type": "event_msg", + "payload": {"type": "task_started", "turn_id": "turn-2"}}, + {"timestamp": "2026-07-06T13:34:12.650Z", "type": "response_item", + "payload": {"type": "function_call", "name": "exec_command", + "arguments": "{\"cmd\":\"false\"}", "call_id": "call_D"}}, + # exec_command_end carries a real exit_code; note there is NO + # function_call_output with a "Process exited" marker for this call. + {"timestamp": "2026-07-06T13:34:12.680Z", "type": "event_msg", + "payload": {"type": "exec_command_end", "call_id": "call_D", + "turn_id": "turn-2", "exit_code": 2, "status": "completed", + "aggregated_output": "boom\n"}}, + {"timestamp": "2026-07-06T13:34:12.700Z", "type": "response_item", + "payload": {"type": "function_call", "name": "apply_patch", + "arguments": "{\"input\":\"*** Begin Patch\"}", "call_id": "call_C"}}, + # apply_patch reports success via patch_apply_end; no exit code anywhere. + {"timestamp": "2026-07-06T13:34:12.750Z", "type": "event_msg", + "payload": {"type": "patch_apply_end", "call_id": "call_C", + "turn_id": "turn-2", "success": True, "status": "completed", + "stdout": "", "stderr": ""}}, ] @@ -88,6 +108,21 @@ def test_exit_code_and_status(self): self.assertEqual(by["call_B"].exit_code, 1) self.assertEqual(by["call_B"].status, "failure") + def test_status_from_structured_end_events(self): + # These outcomes are ONLY knowable from exec_command_end / patch_apply_end + # — the old exit-code scrape would have left both as "unknown". + by = {c.call_id: c for c in self.s.tool_calls} + self.assertEqual(by["call_D"].exit_code, 2) + self.assertEqual(by["call_D"].status, "failure") # exec_command_end + self.assertEqual(by["call_C"].status, "success") # patch_apply_end.success + + def test_turn_attribution(self): + by = {c.call_id: c for c in self.s.tool_calls} + self.assertEqual(by["call_A"].turn_id, "turn-1") + self.assertEqual(by["call_D"].turn_id, "turn-2") + self.assertEqual(by["call_C"].turn_id, "turn-2") + self.assertTrue(all(m.turn_id == "turn-1" for m in self.s.messages)) + def test_tokens(self): self.assertEqual(self.s.total_tokens, 120) self.assertEqual(self.s.reasoning_tokens, 5) @@ -119,9 +154,13 @@ def test_roundtrip_and_idempotent(self): store.upsert_session(s) # second time must not duplicate store.commit() rows = store.query("SELECT COUNT(*) c FROM tool_calls") - self.assertEqual(rows[0]["c"], 2) + self.assertEqual(rows[0]["c"], 4) srow = store.query("SELECT total_tokens FROM sessions")[0] self.assertEqual(srow["total_tokens"], 120) + # turn_id must survive the round-trip (turn-level attribution) + trow = store.query( + "SELECT turn_id FROM tool_calls WHERE call_id='call_C'")[0] + self.assertEqual(trow["turn_id"], "turn-2") store.close() finally: os.remove(dbpath)