diff --git a/README.md b/README.md index 02955cf..43e7bbf 100644 --- a/README.md +++ b/README.md @@ -35,18 +35,37 @@ extensions/ code/ syntax/ session-metrics/ + semantic-observer/ task/ terminal/ +packages/ + semantic-predicate/ skills/ prompts/ docs/ tsconfig.json ``` -Every runtime workspace now lives under `extensions/`; there is no separate `packages/` layer. `session-metrics` owns both the Pi extension and its offline CLI/analysis kernel. Multi-word extension directories use kebab-case, and the shared TypeScript configuration lives at the repository root. +`extensions/` remains the home of Pi runtime integration. `packages/` is reserved for code that is meaningful without Pi; both are root workspaces. The experimental `semantic-predicate` package lives under `packages/` so Jev/OpenRouter evaluation can be removed or reused without changing Pi runtime contracts. `session-metrics` continues to own both its Pi extension and offline CLI/analysis kernel. Multi-word extension directories use kebab-case, and the shared TypeScript configuration lives at the repository root. The repository extension exposes `context` and `code`, and transparently strengthens the built-in `edit` path for supported source files. The old standalone Astrolabe and BM25 tool surfaces are gone; their useful structural and lexical mechanisms are internal implementation details under `src/syntax` and `src/context`. -Additional independent utilities remain available through the extensions listed above. `ask` provides synchronous structured user decisions in the interactive TUI; offline session analysis is provided by the `session-metrics` CLI in `extensions/session-metrics`. +Additional independent utilities remain available through the extensions listed above. `ask` provides synchronous structured user decisions in the interactive TUI; offline session analysis is provided by the `session-metrics` CLI in `extensions/session-metrics`. `semantic-observer` is an experimental, explicitly invoked observer: the caller selects semantic judgments, while Pi Kit builds compact evidence from task state, tracked reads/context, actual workspace changes, and executed verification. It returns advisory probabilities without changing task, verification, or completion state. + + +## Experimental semantic observation + +`semantic-observer` treats Jev as a sensor, not an authority. It uses OpenRouter's Decisions API through the Pi-independent `packages/semantic-predicate` package and keeps thresholds or actions outside the model boundary. + +Context is assembled as evidence, not as a transcript: + +- Prefer runtime-captured primary evidence: the task goal and acceptance criteria, tracked mutations and the task-baseline Git diff, executed verification, and the successful `read`/`context` results the agent actually observed. +- Keep fields named and structured. Questions refer to the state fields they judge rather than relying on one opaque prompt. +- Give each judgment only the fields it needs. Questions that need different evidence are evaluated against separate minimal states; questions with the same state may be batched. +- Keep deterministic facts in code. Jev is for semantic judgments such as scope drift or whether verification meaningfully covers a change, not whether a check exists or how many files changed. +- Preserve probabilities. The observer does not turn Jev output into a pass/fail result; later policy may choose thresholds after the behavior has been measured. +- Do not feed broad session history, repository dumps, or previous Jev outputs back into later state by default. Add context only when it is evidence for the next judgment. + +The current observer accepts only an `observations` list (`scopeDrift`, `verificationGap`, `consistencyRisk`). Evidence payloads are not authored by the calling model. It is deliberately explicit-call and advisory while the experiment is being evaluated. See [`docs/architecture.md`](docs/architecture.md) for the design rationale and runtime contracts. diff --git a/bun.lock b/bun.lock index 6aeb735..ead0ebb 100644 --- a/bun.lock +++ b/bun.lock @@ -83,6 +83,18 @@ "typescript": "catalog:", }, }, + "extensions/semantic-observer": { + "name": "@halqme/semantic-observer", + "dependencies": { + "@earendil-works/pi-ai": "catalog:", + "@earendil-works/pi-coding-agent": "catalog:", + "@earendil-works/pi-tui": "catalog:", + }, + "devDependencies": { + "@types/node": "catalog:", + "typescript": "catalog:", + }, + }, "extensions/session-metrics": { "name": "@halqme/session-metrics", "bin": { @@ -122,6 +134,13 @@ "typescript": "catalog:", }, }, + "packages/semantic-predicate": { + "name": "@halqme/semantic-predicate", + "devDependencies": { + "@types/node": "catalog:", + "typescript": "catalog:", + }, + }, }, "catalog": { "@earendil-works/pi-agent-core": "^0.84.1", @@ -213,6 +232,10 @@ "@halqme/repository": ["@halqme/repository@workspace:extensions/repository"], + "@halqme/semantic-observer": ["@halqme/semantic-observer@workspace:extensions/semantic-observer"], + + "@halqme/semantic-predicate": ["@halqme/semantic-predicate@workspace:packages/semantic-predicate"], + "@halqme/session-metrics": ["@halqme/session-metrics@workspace:extensions/session-metrics"], "@halqme/task": ["@halqme/task@workspace:extensions/task"], diff --git a/docs/architecture.md b/docs/architecture.md index 4a1012b..8418487 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -45,6 +45,45 @@ This follows the centralized asynchronous isolated delegation pattern evaluated Stable behavior belongs in tools and runtime state. `AGENTS.md` therefore contains only repository invariants and development mechanics; tool-routing and workflow state are not encoded as an always-on prompt layer. This is consistent with the repository-context results in arXiv:2602.11988. + +## Experimental semantic observation + +`semantic-observer` is outside the mechanical authority path. Its outputs are observations only: they cannot mutate repository state, satisfy `verify`, or unlock `task.finish`. The Pi-facing extension adapts runtime evidence; `packages/semantic-predicate` owns the Pi-independent OpenRouter Decisions API client and typed Jev primitives. + +The context boundary is intentionally narrower than the model context window. Jev 1.13 degrades when state contains irrelevant detail, so the observer does not treat the current conversation or repository as a default context blob. Each semantic judgment declares the evidence it needs and receives a small structured state with named fields. Primary runtime or repository evidence is preferred over a model-authored narrative summary. + +The observer does not ask the calling model to summarize its own work. It projects existing Pi Kit runtime state instead. `task/evidence.ts` exposes the same side-effect-free packet used by `task.review_context`: task contract and latest checkpoint, resource provenance, workspace delta, and verification evidence. The semantic observer augments that packet with a bounded Git diff from the task's captured baseline and bounded excerpts from successful `read`/`context` tool results identified by their tracked tool-call IDs. + +The current context views are: + +```text +scopeDrift + task goal + acceptance + + current checkpoint (plan marked as hypothesis) + + tracked mutations + task-baseline diff + +verificationGap + task goal + acceptance + + tracked mutations + task-baseline diff + + executed verification only + +consistencyRisk + tracked mutations + task-baseline diff + + paths observed during the task + + bounded excerpts from the exact read/context results already seen +``` + +The caller supplies only which observation IDs to run. These are separate requests because their evidence sets differ. If future questions genuinely share the same state, they should be batched into one Decisions API request; Jev evaluates questions independently and batching avoids sending the same state repeatedly. + +This boundary follows four rules: + +1. **Filter before inference.** Retrieval and runtime state select evidence before Jev sees it. +2. **Semantic only.** Exact checks, counts, dates, presence tests, and arithmetic stay in code. +3. **Probabilities before policy.** Raw Noul probabilities or Choice/Score distributions are recorded first; thresholds and actions belong to deterministic policy outside the package. +4. **No ambient accumulation.** Session history, broad diffs, repository dumps, and prior semantic answers are not automatically carried forward. A second-stage request receives earlier output only when code needs that result to construct genuinely new state. + +The experiment is intentionally explicit-call. Automatic hooks, escalation, or review routing should be added only after session evidence shows which judgments are useful and how their probabilities calibrate on Pi Kit work. + ## Evaluation `session-metrics` reconstructs runtime behavior from Pi session JSONL without active instrumentation. In addition to generic tool/action metrics, it records the `context`, `code`, `task`, `delegate`, and `verify` surfaces and verification provenance so harness changes can be compared against historical trajectories. diff --git a/extensions/background-process/index.ts b/extensions/background-process/index.ts index c873b05..83a25c8 100644 --- a/extensions/background-process/index.ts +++ b/extensions/background-process/index.ts @@ -130,7 +130,8 @@ export default function backgroundProcessExtension(pi: ExtensionAPI): void { inspectRunning: Type.Optional( Type.Boolean({ default: false, - description: "For check only: include stdout/stderr while a process is pending or running.", + description: + "For check only: include stdout/stderr while a process is pending or running.", }), ), }), diff --git a/extensions/semantic-observer/evidence.test.ts b/extensions/semantic-observer/evidence.test.ts new file mode 100644 index 0000000..90818a8 --- /dev/null +++ b/extensions/semantic-observer/evidence.test.ts @@ -0,0 +1,122 @@ +import assert from "node:assert/strict"; +import test from "node:test"; + +import type { TaskEvidencePacket } from "../task/evidence.ts"; +import { projectObservationState } from "./evidence.ts"; + +const packet: TaskEvidencePacket = { + task: { + id: "task-1", + goal: "Update the parser only", + acceptance: ["Parser accepts the new syntax"], + status: "active", + latestCheckpoint: { + at: "2026-09-18T00:00:00.000Z", + summary: "Parser implementation is in progress", + plan: ["Rewrite unrelated renderer", "Update parser"], + completed: ["Located parser"], + }, + }, + resources: { + observed: ["src/parser.ts", "src/parser.test.ts"], + mutated: ["src/parser.ts"], + changedDuringTask: ["src/parser.ts"], + preexistingDirty: [], + timeline: [ + { + operation: "observe", + path: "src/parser.ts", + tool: "context", + action: "inspect", + toolCallId: "context-1", + }, + { + operation: "mutate", + path: "src/parser.ts", + tool: "code", + action: "edit", + toolCallId: "code-1", + }, + ], + coverage: { + observations: "explicit-tools", + mutations: "explicit-tools", + workspaceDelta: "git", + opaqueToolEffects: "not-attributed", + }, + }, + verification: [ + { + id: "verify-1", + taskId: "task-1", + provenance: "typecheck", + origin: "executed", + passed: true, + summary: "tsc --noEmit", + at: "2026-09-18T00:01:00.000Z", + }, + { + id: "verify-2", + taskId: "task-1", + provenance: "self_review", + origin: "reported", + passed: true, + summary: "looks good", + at: "2026-09-18T00:02:00.000Z", + }, + ], + workspace: { + baselineHead: "abc", + currentHead: "abc", + currentDirty: ["src/parser.ts"], + changedDuringTask: ["src/parser.ts"], + uncommittedTaskChanges: ["src/parser.ts"], + taskCommitRequired: true, + taskCommitPresent: false, + }, +}; + +test("scope drift state uses task authority and actual changes", () => { + const state = projectObservationState("scopeDrift", packet, { + diff: "@@ parser diff @@", + context: [], + }) as any; + + assert.equal(state.task.goal, "Update the parser only"); + assert.deepEqual(state.task.acceptance, ["Parser accepts the new syntax"]); + assert.equal(state.task.current_stage.summary, "Parser implementation is in progress"); + assert.equal(state.task.current_stage.plan_authority, "hypothesis"); + assert.deepEqual(state.changes.changed_paths, ["src/parser.ts"]); + assert.equal(state.changes.diff, "@@ parser diff @@"); +}); + +test("verification gap state includes only executed verification", () => { + const state = projectObservationState("verificationGap", packet, { + context: [], + }) as any; + + assert.equal(state.verification.length, 1); + assert.equal(state.verification[0].provenance, "typecheck"); + assert.equal(state.verification[0].summary, "tsc --noEmit"); +}); + +test("consistency state reuses observed repository evidence", () => { + const state = projectObservationState("consistencyRisk", packet, { + context: [ + { + tool: "context", + paths: ["src/parser.ts"], + text: "export function parse() {}", + }, + ], + }) as any; + + assert.deepEqual(state.repository_evidence.observed_paths, [ + "src/parser.ts", + "src/parser.test.ts", + ]); + assert.equal( + state.repository_evidence.excerpts[0].text, + "export function parse() {}", + ); +}); diff --git a/extensions/semantic-observer/evidence.ts b/extensions/semantic-observer/evidence.ts new file mode 100644 index 0000000..f5c9520 --- /dev/null +++ b/extensions/semantic-observer/evidence.ts @@ -0,0 +1,273 @@ +import { execFile } from "node:child_process"; +import { promisify } from "node:util"; +import type { ExtensionContext } from "@earendil-works/pi-coding-agent"; + +import { + taskEvidencePacket, + type TaskEvidencePacket, +} from "../task/evidence.ts"; +import type { JsonValue } from "../../packages/semantic-predicate/src/index.ts"; + +const exec = promisify(execFile); +const MAX_DIFF_CHARS = 12_000; +const MAX_CONTEXT_CHARS = 9_000; +const MAX_CONTEXT_ITEM_CHARS = 3_000; +const MAX_CONTEXT_ITEMS = 4; + +export type ObservationId = "scopeDrift" | "verificationGap" | "consistencyRisk"; + +export interface ContextExcerpt { + tool: string; + paths: string[]; + text: string; +} + +export interface ObservationRuntimeEvidence { + diff?: string; + context: ContextExcerpt[]; +} + +type RecordValue = Record; + +function record(value: unknown): RecordValue | undefined { + return value !== null && typeof value === "object" && !Array.isArray(value) + ? (value as RecordValue) + : undefined; +} + +function textContent(value: unknown): string { + if (typeof value === "string") return value; + if (!Array.isArray(value)) return ""; + return value + .map((block) => { + const item = record(block); + return item?.type === "text" && typeof item.text === "string" ? item.text : ""; + }) + .filter(Boolean) + .join("\n"); +} + +function clip(value: string, max: number): string { + if (value.length <= max) return value; + return `${value.slice(0, max)}\n…[truncated ${value.length - max} chars]`; +} + +function taskView(packet: TaskEvidencePacket): JsonValue { + const checkpoint = packet.task.latestCheckpoint; + return { + goal: packet.task.goal, + acceptance: packet.task.acceptance, + status: packet.task.status, + ...(checkpoint + ? { + current_stage: { + summary: checkpoint.summary, + ...(checkpoint.completed?.length ? { completed: checkpoint.completed } : {}), + ...(checkpoint.plan?.length + ? { + working_plan: checkpoint.plan, + plan_authority: "hypothesis", + } + : {}), + }, + } + : {}), + ...(packet.task.blocker ? { blocker: packet.task.blocker } : {}), + }; +} + +function changeView(packet: TaskEvidencePacket, runtime: ObservationRuntimeEvidence): JsonValue { + const mutations = packet.resources.timeline + .filter((event) => event.operation === "mutate") + .map((event) => ({ + path: event.path, + tool: event.tool, + ...(event.action ? { action: event.action } : {}), + })); + + return { + changed_paths: packet.resources.changedDuringTask, + recorded_mutations: mutations, + ...(runtime.diff ? { diff: runtime.diff } : {}), + provenance: { + workspace_delta: packet.resources.coverage.workspaceDelta, + explicit_mutation_tracking: packet.resources.coverage.mutations, + opaque_tool_effects: packet.resources.coverage.opaqueToolEffects, + }, + }; +} + +function verificationView(packet: TaskEvidencePacket): JsonValue { + return packet.verification + .filter((item) => item.origin === "executed") + .map((item) => ({ + provenance: item.provenance, + passed: item.passed, + summary: item.summary, + ...(item.detail ? { detail: clip(item.detail, 2_000) } : {}), + })); +} + +function repositoryEvidenceView( + packet: TaskEvidencePacket, + runtime: ObservationRuntimeEvidence, +): JsonValue { + return { + observed_paths: packet.resources.observed, + excerpts: runtime.context.map((item) => ({ + tool: item.tool, + paths: item.paths, + text: item.text, + })), + provenance: { + explicit_observation_tracking: packet.resources.coverage.observations, + note: "Excerpts are successful read/context tool results previously observed during this task.", + }, + }; +} + +export function projectObservationState( + observation: ObservationId, + packet: TaskEvidencePacket, + runtime: ObservationRuntimeEvidence, +): JsonValue { + if (observation === "scopeDrift") { + return { + task: taskView(packet), + changes: changeView(packet, runtime), + }; + } + + if (observation === "verificationGap") { + return { + task: taskView(packet), + changes: changeView(packet, runtime), + verification: verificationView(packet), + }; + } + + return { + changes: changeView(packet, runtime), + repository_evidence: repositoryEvidenceView(packet, runtime), + }; +} + +async function diffEvidence( + ctx: ExtensionContext, + packet: TaskEvidencePacket, +): Promise { + const paths = packet.resources.changedDuringTask; + const baseline = packet.workspace?.baselineHead; + if (!baseline || paths.length === 0) return undefined; + + try { + const { stdout } = await exec( + "git", + ["--no-pager", "diff", "--no-ext-diff", "--unified=2", baseline, "--", ...paths], + { + cwd: ctx.cwd, + encoding: "utf8", + maxBuffer: 2 * 1024 * 1024, + timeout: 10_000, + }, + ); + const diff = stdout.trim(); + return diff ? clip(diff, MAX_DIFF_CHARS) : undefined; + } catch { + return undefined; + } +} + +function toolResultText(entries: unknown[]): Map { + const results = new Map(); + + for (const candidate of entries) { + const entry = record(candidate); + if (entry?.type !== "message") continue; + const message = record(entry.message); + if (message?.role !== "toolResult" || message.isError === true) continue; + const id = typeof message.toolCallId === "string" ? message.toolCallId : undefined; + const tool = typeof message.toolName === "string" ? message.toolName : undefined; + if (!id || !tool) continue; + const text = textContent(message.content).trim(); + if (text) results.set(id, { tool, text }); + } + + return results; +} + +function contextEvidence( + ctx: ExtensionContext, + packet: TaskEvidencePacket, +): ContextExcerpt[] { + const results = toolResultText(ctx.sessionManager.getEntries()); + const changed = new Set(packet.resources.changedDuringTask); + const grouped = new Map< + string, + { tool: string; paths: Set; index: number; overlapsChange: boolean } + >(); + + packet.resources.timeline.forEach((event, index) => { + if (event.operation !== "observe" || (event.tool !== "read" && event.tool !== "context")) { + return; + } + const existing = grouped.get(event.toolCallId); + if (existing) { + existing.paths.add(event.path); + existing.overlapsChange ||= changed.has(event.path); + existing.index = index; + return; + } + grouped.set(event.toolCallId, { + tool: event.tool, + paths: new Set([event.path]), + index, + overlapsChange: changed.has(event.path), + }); + }); + + const candidates = [...grouped.entries()] + .map(([toolCallId, value]) => ({ toolCallId, ...value })) + .filter((item) => results.has(item.toolCallId)) + .sort((a, b) => { + if (a.overlapsChange !== b.overlapsChange) return a.overlapsChange ? -1 : 1; + return b.index - a.index; + }) + .slice(0, MAX_CONTEXT_ITEMS); + + let remaining = MAX_CONTEXT_CHARS; + const excerpts: ContextExcerpt[] = []; + for (const item of candidates) { + if (remaining <= 0) break; + const result = results.get(item.toolCallId); + if (!result) continue; + const text = clip(result.text, Math.min(MAX_CONTEXT_ITEM_CHARS, remaining)); + remaining -= text.length; + excerpts.push({ + tool: item.tool, + paths: [...item.paths].sort(), + text, + }); + } + return excerpts; +} + +export async function buildObservationState( + ctx: ExtensionContext, + observation: ObservationId, +): Promise { + const packet = await taskEvidencePacket(ctx); + if (!packet || (packet.task.status !== "active" && packet.task.status !== "blocked")) { + throw new Error("precondition: semantic_observe requires an active or blocked task."); + } + + const [diff, context] = await Promise.all([ + diffEvidence(ctx, packet), + Promise.resolve(contextEvidence(ctx, packet)), + ]); + + return projectObservationState(observation, packet, { + ...(diff ? { diff } : {}), + context, + }); +} diff --git a/extensions/semantic-observer/index.ts b/extensions/semantic-observer/index.ts new file mode 100644 index 0000000..12ffff5 --- /dev/null +++ b/extensions/semantic-observer/index.ts @@ -0,0 +1,136 @@ +import { Type } from "@earendil-works/pi-ai"; +import type { ExtensionAPI } from "@earendil-works/pi-coding-agent"; +import { + createOpenRouterSemanticEvaluator, + type JsonValue, + type NoulQuestion, + type SemanticEvaluator, +} from "../../packages/semantic-predicate/src/index.ts"; +import { buildObservationState, type ObservationId } from "./evidence.ts"; + +const noul = ( + instructions: string, + yes: string, + no: string, +): NoulQuestion => ({ + type: "noul", + instructions, + criteria: { + true: yes, + false: no, + }, +}); + +const QUESTIONS = { + scopeDrift: noul( + "Using `task` as the scope authority and `changes` as runtime evidence, has the work materially moved beyond the requested outcome or the smallest necessary implementation scope? Treat `task.current_stage.working_plan` as a hypothesis, not as authority to expand scope.", + "The changes add behavior, refactoring, dependencies, or scope that is not needed for the task goal or acceptance criteria.", + "The changes stay within the task goal and acceptance criteria, or are necessary to satisfy them.", + ), + verificationGap: noul( + "Given `task`, `changes`, and executed `verification`, is there a meaningful gap between what changed and what the executed verification demonstrates?", + "Important changed behavior or an important failure mode is not covered by the supplied executed verification.", + "The supplied executed verification is relevant evidence for the important changed behavior and failure modes.", + ), + consistencyRisk: noul( + "Given `changes` and the previously observed `repository_evidence`, do the changes appear inconsistent with relevant repository contracts, conventions, or related code?", + "The supplied repository evidence indicates a material inconsistency or likely integration mismatch.", + "The changes are consistent with the supplied repository evidence, or the evidence does not indicate a material mismatch.", + ), +} as const; + +type ObservationResult = { + probability: number; + model?: string; +}; + +async function evaluateObservation( + evaluate: SemanticEvaluator, + observation: ObservationId, + state: JsonValue, +): Promise { + if (observation === "scopeDrift") { + const response = await evaluate({ + state, + questions: { scope_drift: QUESTIONS.scopeDrift }, + }); + return { + probability: response.answers.scope_drift.noul, + ...(response.model ? { model: response.model } : {}), + }; + } + + if (observation === "verificationGap") { + const response = await evaluate({ + state, + questions: { verification_gap: QUESTIONS.verificationGap }, + }); + return { + probability: response.answers.verification_gap.noul, + ...(response.model ? { model: response.model } : {}), + }; + } + + const response = await evaluate({ + state, + questions: { consistency_risk: QUESTIONS.consistencyRisk }, + }); + return { + probability: response.answers.consistency_risk.noul, + ...(response.model ? { model: response.model } : {}), + }; +} + +export default function semanticObserverExtension(pi: ExtensionAPI): void { + pi.registerTool({ + name: "semantic_observe", + label: "Semantic Observe", + description: + "Run optional, non-authoritative Jev observations over evidence already captured by Pi Kit task, repository, mutation, and verification runtime state. The caller chooses observations, not the evidence payload.", + parameters: Type.Object({ + observations: Type.Array( + Type.Union([ + Type.Literal("scopeDrift"), + Type.Literal("verificationGap"), + Type.Literal("consistencyRisk"), + ]), + { + minItems: 1, + maxItems: 3, + uniqueItems: true, + description: "Semantic judgments to run against Pi Kit runtime evidence.", + }, + ), + }), + async execute(_toolCallId, params, _signal, _update, ctx) { + const apiKey = process.env.OPENROUTER_API_KEY; + if (!apiKey) { + throw new Error("semantic_observe requires OPENROUTER_API_KEY"); + } + + const evaluate = createOpenRouterSemanticEvaluator({ apiKey }); + const observations = Object.fromEntries( + await Promise.all( + params.observations.map(async (observation) => { + const state = await buildObservationState(ctx, observation); + return [observation, await evaluateObservation(evaluate, observation, state)] as const; + }), + ), + ); + + return { + content: [ + { + type: "text" as const, + text: JSON.stringify({ observations }, null, 2), + }, + ], + details: { + advisory: true, + evidenceSource: "pi-runtime", + observations: Object.keys(observations), + }, + }; + }, + }); +} diff --git a/extensions/semantic-observer/package.json b/extensions/semantic-observer/package.json new file mode 100644 index 0000000..53e065f --- /dev/null +++ b/extensions/semantic-observer/package.json @@ -0,0 +1,23 @@ +{ + "name": "@halqme/semantic-observer", + "private": true, + "type": "module", + "scripts": { + "check": "bun run typecheck && bun run test", + "typecheck": "tsc --noEmit", + "dev": "pi -e ./index.ts", + "test": "node --test" + }, + "dependencies": { + "@earendil-works/pi-ai": "catalog:", + "@earendil-works/pi-coding-agent": "catalog:", + "@earendil-works/pi-tui": "catalog:" + }, + "devDependencies": { + "@types/node": "catalog:", + "typescript": "catalog:" + }, + "engines": { + "node": ">=26.0.0" + } +} diff --git a/extensions/semantic-observer/tsconfig.json b/extensions/semantic-observer/tsconfig.json new file mode 100644 index 0000000..73b9679 --- /dev/null +++ b/extensions/semantic-observer/tsconfig.json @@ -0,0 +1,4 @@ +{ + "extends": "../../tsconfig.json", + "include": ["*.ts"] +} diff --git a/extensions/task/evidence.ts b/extensions/task/evidence.ts new file mode 100644 index 0000000..6665a41 --- /dev/null +++ b/extensions/task/evidence.ts @@ -0,0 +1,77 @@ +import type { ExtensionContext } from "@earendil-works/pi-coding-agent"; + +import { taskReviewResources, taskWorkspaceState, type TaskReviewResources, type TaskWorkspaceState } from "./resources.ts"; +import { customEntries, latestCustom, TASK_ENTRY, VERIFY_ENTRY } from "./shared.ts"; + +export interface TaskEvidenceCheckpoint { + at: string; + summary: string; + plan?: string[]; + observations?: string[]; + completed?: string[]; +} + +interface StoredTaskState { + id: string; + goal: string; + acceptance: string[]; + status: "active" | "blocked" | "done" | "stopped"; + checkpoints: TaskEvidenceCheckpoint[]; + blocker?: string; +} + +export interface TaskVerificationEvidence { + id: string; + taskId?: string; + provenance: string; + origin: "executed" | "reported"; + passed: boolean; + summary: string; + detail?: string; + reviewRequestId?: string; + at: string; +} + +export interface TaskEvidencePacket { + task: { + id: string; + goal: string; + acceptance: string[]; + status: StoredTaskState["status"]; + latestCheckpoint?: TaskEvidenceCheckpoint; + blocker?: string; + }; + resources: TaskReviewResources; + verification: TaskVerificationEvidence[]; + workspace?: TaskWorkspaceState; +} + +export async function taskEvidencePacket( + ctx: ExtensionContext, +): Promise { + const current = latestCustom(ctx, TASK_ENTRY); + if (!current) return undefined; + + const [resources, workspace] = await Promise.all([ + taskReviewResources(ctx, current.id), + taskWorkspaceState(ctx, current.id).catch(() => undefined), + ]); + const verification = customEntries(ctx, VERIFY_ENTRY).filter( + (item) => item.taskId === current.id, + ); + const latestCheckpoint = current.checkpoints.at(-1); + + return { + task: { + id: current.id, + goal: current.goal, + acceptance: current.acceptance, + status: current.status, + ...(latestCheckpoint ? { latestCheckpoint } : {}), + ...(current.blocker ? { blocker: current.blocker } : {}), + }, + resources, + verification, + ...(workspace ? { workspace } : {}), + }; +} diff --git a/extensions/task/resources.ts b/extensions/task/resources.ts index b0904e3..504f3dc 100644 --- a/extensions/task/resources.ts +++ b/extensions/task/resources.ts @@ -51,6 +51,7 @@ export interface TaskReviewResources { path: string; tool: string; action?: string; + toolCallId: string; assistantEntryId?: string; }>; coverage: { @@ -440,6 +441,7 @@ export async function taskReviewResources( path: event.path, tool: event.tool, ...(event.action ? { action: event.action } : {}), + toolCallId: event.toolCallId, ...(event.assistantEntryId ? { assistantEntryId: event.assistantEntryId } : {}), })), coverage: { diff --git a/extensions/task/runtime.ts b/extensions/task/runtime.ts index 7109235..385ce4a 100644 --- a/extensions/task/runtime.ts +++ b/extensions/task/runtime.ts @@ -9,11 +9,11 @@ import { captureWorkspaceBaseline, registerTaskResourceTracking, resolveProjectRoot, - taskReviewResources, taskWorkspaceRevision, taskWorkspaceState, WORKSPACE_ENTRY, } from "./resources.ts"; +import { taskEvidencePacket } from "./evidence.ts"; import { customEntries, jsonResult, @@ -371,11 +371,8 @@ export function registerTask(pi: ExtensionAPI): void { if (!current) throw new Error("precondition: No task state. Start a task first."); if (params.action === "review_context") { - const evidence = customEntries(ctx, VERIFY_ENTRY).filter( - (item) => item.taskId === current.id, - ); - const resources = await taskReviewResources(ctx, current.id); - const latestCheckpoint = current.checkpoints.at(-1); + const evidence = await taskEvidencePacket(ctx); + if (!evidence) throw new Error("precondition: No task evidence available."); const workspaceRevision = await taskWorkspaceRevision(ctx, current.id); const reviewRequest: TaskReviewRequest = { version: 1, @@ -386,16 +383,7 @@ export function registerTask(pi: ExtensionAPI): void { }; pi.appendEntry(REVIEW_ENTRY, reviewRequest); return jsonResult({ - task: { - id: current.id, - goal: current.goal, - acceptance: current.acceptance, - status: current.status, - ...(latestCheckpoint ? { latestCheckpoint } : {}), - ...(current.blocker ? { blocker: current.blocker } : {}), - }, - resources, - verification: evidence, + ...evidence, reviewRequest: { id: reviewRequest.id, at: reviewRequest.at }, }); } diff --git a/package.json b/package.json index 12d004c..697d545 100644 --- a/package.json +++ b/package.json @@ -5,7 +5,8 @@ "pi-package" ], "workspaces": [ - "extensions/*" + "extensions/*", + "packages/*" ], "scripts": { "hooks:install": "git config core.hooksPath .githooks", @@ -38,6 +39,7 @@ "./extensions/browser-inspector/index.ts", "./extensions/macos-talk/index.ts", "./extensions/session-metrics/index.ts", + "./extensions/semantic-observer/index.ts", "./extensions/terminal/index.ts" ], "skills": [ diff --git a/packages/semantic-predicate/package.json b/packages/semantic-predicate/package.json new file mode 100644 index 0000000..9b3e6e6 --- /dev/null +++ b/packages/semantic-predicate/package.json @@ -0,0 +1,21 @@ +{ + "name": "@halqme/semantic-predicate", + "private": true, + "type": "module", + "exports": { + ".": "./src/index.ts" + }, + "scripts": { + "check": "bun run typecheck && bun run test", + "typecheck": "tsc --noEmit", + "test": "node --test" + }, + "dependencies": {}, + "devDependencies": { + "@types/node": "catalog:", + "typescript": "catalog:" + }, + "engines": { + "node": ">=26.0.0" + } +} diff --git a/packages/semantic-predicate/src/index.test.ts b/packages/semantic-predicate/src/index.test.ts new file mode 100644 index 0000000..9d099e3 --- /dev/null +++ b/packages/semantic-predicate/src/index.test.ts @@ -0,0 +1,143 @@ +import assert from "node:assert/strict"; +import { describe, test } from "node:test"; +import { + createOpenRouterSemanticEvaluator, + parseSemanticDecisionResponse, +} from "./index.ts"; + +describe("parseSemanticDecisionResponse", () => { + test("parses a Noul without inventing a confidence field", () => { + const result = parseSemanticDecisionResponse( + { + model: "typesafe/jev-1.13", + answers: { + scope_drift: { + type: "noul", + noul: 0.82, + }, + }, + }, + { + scope_drift: { + type: "noul", + instructions: "Has the work moved outside the requested scope?", + }, + }, + ); + + assert.deepEqual(result.answers.scope_drift, { + type: "noul", + noul: 0.82, + }); + }); + + test("parses Choice distributions and confidence", () => { + const result = parseSemanticDecisionResponse( + { + answers: { + route: { + type: "choice", + choice: "review", + probabilities: { + continue: 0.25, + review: 0.75, + }, + confidence: 0.5, + }, + }, + }, + { + route: { + type: "choice", + instructions: "Which route fits the state?", + criteria: { + continue: "Continue normally.", + review: "Request review.", + }, + }, + }, + ); + + assert.equal(result.answers.route.choice, "review"); + assert.equal(result.answers.route.probabilities.review, 0.75); + }); + + test("rejects out-of-range Noul probabilities", () => { + assert.throws( + () => + parseSemanticDecisionResponse( + { + answers: { + gap: { + type: "noul", + noul: 1.2, + }, + }, + }, + { + gap: { + type: "noul", + instructions: "Is there a verification gap?", + }, + }, + ), + /between 0 and 1/, + ); + }); +}); + +describe("createOpenRouterSemanticEvaluator", () => { + test("uses the Decisions API with state and typed questions directly", async () => { + let requestUrl = ""; + let requestBody: unknown; + + const evaluate = createOpenRouterSemanticEvaluator({ + apiKey: "test-key", + fetch: async (input, init) => { + requestUrl = String(input); + requestBody = JSON.parse(String(init?.body)); + return new Response( + JSON.stringify({ + model: "typesafe/jev-1.13", + answers: { + scope_drift: { + type: "noul", + noul: 0.2, + }, + }, + usage: { + input_tokens: 42, + output_tokens: 3, + }, + }), + { + status: 200, + headers: { "Content-Type": "application/json" }, + }, + ); + }, + }); + + const state = { + request: "Only update the parser.", + changes: "Changed parser.ts.", + }; + + const questions = { + scope_drift: { + type: "noul" as const, + instructions: "Given `request` and `changes`, has the work moved outside the request?", + }, + }; + + const result = await evaluate({ state, questions }); + + assert.equal(requestUrl, "https://openrouter.ai/api/alpha/decisions"); + assert.deepEqual(requestBody, { + model: "typesafe/jev-1.13", + state, + questions, + }); + assert.equal(result.answers.scope_drift.noul, 0.2); + }); +}); diff --git a/packages/semantic-predicate/src/index.ts b/packages/semantic-predicate/src/index.ts new file mode 100644 index 0000000..0ad5bdf --- /dev/null +++ b/packages/semantic-predicate/src/index.ts @@ -0,0 +1,234 @@ +export type JsonValue = + | string + | number + | boolean + | null + | JsonValue[] + | { [key: string]: JsonValue }; + +export type NoulQuestion = { + type: "noul"; + instructions: JsonValue; + criteria?: { + true: JsonValue; + false: JsonValue; + }; +}; + +export type ChoiceQuestion = { + type: "choice"; + instructions: JsonValue; + criteria: Record; +}; + +export type ScoreQuestion = { + type: "score"; + instructions: JsonValue; + criteria: JsonValue[]; +}; + +export type SemanticQuestion = NoulQuestion | ChoiceQuestion | ScoreQuestion; +export type SemanticQuestionSet = Record; + +export type NoulAnswer = { + type: "noul"; + noul: number; +}; + +export type ChoiceAnswer = { + type: "choice"; + choice: string; + probabilities: Record; + confidence: number; +}; + +export type ScoreAnswer = { + type: "score"; + score: number; + legend: Record; + probabilities: Record; + confidence: number; +}; + +export type SemanticAnswer = NoulAnswer | ChoiceAnswer | ScoreAnswer; + +export type AnswerForQuestion = T extends NoulQuestion + ? NoulAnswer + : T extends ChoiceQuestion + ? ChoiceAnswer + : T extends ScoreQuestion + ? ScoreAnswer + : never; + +export type AnswersForQuestions = { + [K in keyof T]: AnswerForQuestion; +}; + +export type EvaluateInput = { + state: JsonValue; + questions: T; +}; + +export type SemanticDecisionResponse = { + model?: string; + answers: AnswersForQuestions; + usage?: { + input_tokens?: number; + output_tokens?: number; + [key: string]: JsonValue | undefined; + }; + [key: string]: unknown; +}; + +export type SemanticEvaluator = ( + input: EvaluateInput, +) => Promise>; + +export type OpenRouterEvaluatorOptions = { + apiKey: string; + model?: string; + endpoint?: string; + fetch?: typeof globalThis.fetch; + headers?: Record; +}; + +const probability = (value: unknown, field: string): number => { + if (typeof value !== "number" || !Number.isFinite(value)) { + throw new Error(`Invalid probability: ${field}`); + } + if (value < 0 || value > 1) { + throw new Error(`${field} must be between 0 and 1`); + } + return value; +}; + +const probabilities = (value: unknown, field: string): Record => { + if (!value || typeof value !== "object" || Array.isArray(value)) { + throw new Error(`Invalid probability distribution: ${field}`); + } + + return Object.fromEntries( + Object.entries(value).map(([key, candidate]) => [ + key, + probability(candidate, `${field}.${key}`), + ]), + ); +}; + +const parseAnswer = (raw: unknown, question: SemanticQuestion, id: string): SemanticAnswer => { + if (!raw || typeof raw !== "object" || Array.isArray(raw)) { + throw new Error(`Missing answer for question: ${id}`); + } + + const answer = raw as Record; + if (answer.type !== question.type) { + throw new Error(`Unexpected answer type for question: ${id}`); + } + + if (question.type === "noul") { + return { + type: "noul", + noul: probability(answer.noul, `${id}.noul`), + }; + } + + if (question.type === "choice") { + if (typeof answer.choice !== "string" || !(answer.choice in question.criteria)) { + throw new Error(`Invalid choice for question: ${id}`); + } + + return { + type: "choice", + choice: answer.choice, + probabilities: probabilities(answer.probabilities, `${id}.probabilities`), + confidence: probability(answer.confidence, `${id}.confidence`), + }; + } + + if (typeof answer.score !== "number" || !Number.isFinite(answer.score)) { + throw new Error(`Invalid score for question: ${id}`); + } + if (!answer.legend || typeof answer.legend !== "object" || Array.isArray(answer.legend)) { + throw new Error(`Invalid score legend for question: ${id}`); + } + + return { + type: "score", + score: answer.score, + legend: answer.legend as Record, + probabilities: probabilities(answer.probabilities, `${id}.probabilities`), + confidence: probability(answer.confidence, `${id}.confidence`), + }; +}; + +export const parseSemanticDecisionResponse = ( + raw: unknown, + questions: T, +): SemanticDecisionResponse => { + if (!raw || typeof raw !== "object" || Array.isArray(raw)) { + throw new Error("Invalid semantic decision response"); + } + + const response = raw as Record; + if (!response.answers || typeof response.answers !== "object" || Array.isArray(response.answers)) { + throw new Error("Semantic decision response is missing answers"); + } + + const rawAnswers = response.answers as Record; + const parsedAnswers: Record = {}; + + for (const [id, question] of Object.entries(questions)) { + parsedAnswers[id] = parseAnswer(rawAnswers[id], question, id); + } + + const parsed: SemanticDecisionResponse = { + answers: parsedAnswers as AnswersForQuestions, + }; + + if (typeof response.model === "string") { + parsed.model = response.model; + } + if (response.usage && typeof response.usage === "object") { + parsed.usage = response.usage as NonNullable["usage"]>; + } + + return parsed; +}; + +export const createOpenRouterSemanticEvaluator = ( + options: OpenRouterEvaluatorOptions, +): SemanticEvaluator => { + const endpoint = options.endpoint ?? "https://openrouter.ai/api/alpha/decisions"; + const model = options.model ?? "typesafe/jev-1.13"; + const fetchImpl = options.fetch ?? globalThis.fetch; + + return async ({ + state, + questions, + }: EvaluateInput): Promise> => { + if (Object.keys(questions).length === 0) { + return { model, answers: {} as AnswersForQuestions }; + } + + const response = await fetchImpl(endpoint, { + method: "POST", + headers: { + Authorization: `Bearer ${options.apiKey}`, + "Content-Type": "application/json", + ...options.headers, + }, + body: JSON.stringify({ + model, + state, + questions, + }), + }); + + if (!response.ok) { + const body = await response.text(); + throw new Error(`OpenRouter semantic evaluation failed (${response.status}): ${body}`); + } + + return parseSemanticDecisionResponse(await response.json(), questions); + }; +}; diff --git a/packages/semantic-predicate/tsconfig.json b/packages/semantic-predicate/tsconfig.json new file mode 100644 index 0000000..a5cb75c --- /dev/null +++ b/packages/semantic-predicate/tsconfig.json @@ -0,0 +1,4 @@ +{ + "extends": "../../tsconfig.json", + "include": ["src/**/*.ts"] +}