diff --git a/package.json b/package.json index e53e312..d8d6ec6 100644 --- a/package.json +++ b/package.json @@ -1,7 +1,7 @@ { "name": "@tangle-network/traces", "version": "0.11.5", - "description": "Point it at your coding-agent session traces (Claude Code, Codex, OpenCode, Gemini, Pi, …) and get failure-mode + efficiency findings. CLI + SDK over the @tangle-network/agent-eval analyst suite — observe live sessions, run your own analysts, redact, and upload to the Tangle Intelligence Platform.", + "description": "Point it at your coding-agent session traces (Claude Code, Codex, OpenCode, Gemini, Pi, \u2026) and get failure-mode + efficiency findings. CLI + SDK over the @tangle-network/agent-eval analyst suite \u2014 observe live sessions, run your own analysts, redact, and upload to the Tangle Intelligence Platform.", "type": "module", "license": "MIT", "repository": { @@ -45,7 +45,7 @@ "README.md" ], "engines": { - "node": ">=22" + "node": ">=22.13.0" }, "scripts": { "dev": "tsx src/cli.ts", @@ -59,10 +59,10 @@ "prepublishOnly": "pnpm check:source && pnpm build && pnpm check:package" }, "dependencies": { - "@tangle-network/agent-eval": "0.143.0", - "@tangle-network/agent-runtime": "0.126.0", + "@tangle-network/agent-eval": "0.145.3", + "@tangle-network/agent-runtime": "0.133.2", "@tangle-network/agent-trace-contract": "^1.0.2", - "@tangle-network/sandbox": "0.17.2" + "@tangle-network/sandbox": "0.21.1" }, "devDependencies": { "@types/node": "^22.0.0", diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 941c1d0..ee9dae2 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -14,17 +14,17 @@ importers: .: dependencies: '@tangle-network/agent-eval': - specifier: 0.143.0 - version: 0.143.0 + specifier: 0.145.3 + version: 0.145.3 '@tangle-network/agent-runtime': - specifier: 0.126.0 - version: 0.126.0(@tangle-network/agent-eval@0.143.0)(@tangle-network/agent-interface@0.43.0)(@tangle-network/sandbox@0.17.2) + specifier: 0.133.2 + version: 0.133.2(@tangle-network/agent-eval@0.145.3)(@tangle-network/agent-interface@0.47.0)(@tangle-network/sandbox@0.21.1) '@tangle-network/agent-trace-contract': specifier: ^1.0.2 version: 1.0.2 '@tangle-network/sandbox': - specifier: 0.17.2 - version: 0.17.2 + specifier: 0.21.1 + version: 0.21.1 devDependencies: '@types/node': specifier: ^22.0.0 @@ -495,47 +495,47 @@ packages: '@standard-schema/spec@1.1.0': resolution: {integrity: sha512-l2aFy5jALhniG5HgqrD6jXLi/rUWrKvqN/qJx6yoJsgKhblVd+iqqU4RCXavm/jPityDo5TCvKMnpjKnOriy0w==} - '@tangle-network/agent-core@0.4.33': - resolution: {integrity: sha512-if3DeIo4e2c9wELJjrWOu4SsKv3WnWdDrp+kezP6JrpiQ1DXValRRlN/aTppicFHepw67OD5JQI9i351Z/hvDQ==} + '@tangle-network/agent-core@0.6.1': + resolution: {integrity: sha512-vPpxnSmXpXA9NBjmbOzEnocMJBdskMTrueCepJnztQsO3Z7CKpPQShcJbA5XRv4KyH3KPF284vgMI5lojErvsw==} - '@tangle-network/agent-eval@0.143.0': - resolution: {integrity: sha512-Vx26rz0+qhb1QYd8YkSfoTgvdcoWtqAkR+4H7de4iu4mXKjLlefdWm5iRd4Nwqxgrd4R3Zrgm17N/b282mMCLQ==} + '@tangle-network/agent-eval@0.145.3': + resolution: {integrity: sha512-KPiYjw/SYLFF+mFZF2mjneUmU1Q6SQB6B1PiJRyw3UcnxNxdR8K6M+C7+YGoa/HsEVluhGkD25VPE3T1wTAAww==} engines: {node: '>=20'} hasBin: true - '@tangle-network/agent-interface@0.43.0': - resolution: {integrity: sha512-t35nGOJ3sWouHoUM/5A8eUmsR+8JcuRF6VFQbT9cPzxIMUV3NsFMJgYXPGmyIj65vtvs73O6Mvijp47FAAGbuQ==} + '@tangle-network/agent-interface@0.47.0': + resolution: {integrity: sha512-sCkr7FOT+SNut8EYtwh/x2JheI+Eer30V5IQVIrO8Fn6csC9pqlnLRuNDJh//pnk4+fsg4yLiiSW8/hL1l9e1w==} - '@tangle-network/agent-knowledge@7.0.8': - resolution: {integrity: sha512-FXA8KGCraUwX+2aNGWI8hZYAJ8f2T0iLBPbtzwbUoYl4t6h87munYoDQOh9YjarrAmkp6tcuGIGPKnw0FdCydg==} + '@tangle-network/agent-knowledge@7.2.4': + resolution: {integrity: sha512-xzDAH7YuMD3ijO4gnhvtWz92OvrQ5mlvjTco4tea5fBVkQJVZ1YfPVvgnJ0tTmJXSTe9hqpLiKUTZ5jqe9gnxg==} engines: {node: '>=20.19.0'} hasBin: true + peerDependencies: + '@tangle-network/agent-eval': '>=0.145.2 <0.146.0' + '@tangle-network/agent-interface': '>=0.47.0 <0.48.0' - '@tangle-network/agent-profile-materialize@0.10.2': - resolution: {integrity: sha512-u3MtUy8BD5odTWu/nK0b2XuieBSZMMx1kR9izwqIoR+c8pQaIRQiENNms9qeyQFLqqm7ewC1/Qq5oTtuihs/5g==} + '@tangle-network/agent-profile-materialize@0.14.0': + resolution: {integrity: sha512-LfuyhtrPFvjMvtmZoFoPQehk+8I2e3urAlozEzm9MGQAstkPhsv6lJ/M7mpJYSwD6UVm42Q2GhqgRASi0b8lww==} peerDependencies: - '@tangle-network/agent-interface': '>=0.38.0 <0.44.0' + '@tangle-network/agent-interface': '>=0.46.0 <0.48.0' - '@tangle-network/agent-runtime@0.126.0': - resolution: {integrity: sha512-JTsF13tdb0vv+Tb2Rzm9M2vm6izLno7bJ0Yk/VC0zDE63xllIia0tNKBj2ZGDAkbNFARSIz3wqGFZ/A00x52Yg==} + '@tangle-network/agent-runtime@0.133.2': + resolution: {integrity: sha512-dkqARkCxNEeofHOUiMV0+30Tdw4D2XGX0FErGeF525OeHVMOhfIH2CbbFIWZysHD1J8ZPBFaMnnj+4LztYjnDg==} engines: {node: '>=22.13.0'} hasBin: true peerDependencies: - '@tangle-network/agent-eval': '>=0.143.0 <0.144.0' - '@tangle-network/agent-interface': '>=0.43.0 <0.44.0' - '@tangle-network/sandbox': '>=0.17.2 <0.18.0' - playwright: ^1.40.0 + '@tangle-network/agent-eval': '>=0.145.2 <0.146.0' + '@tangle-network/agent-interface': '>=0.47.0 <0.48.0' + '@tangle-network/sandbox': '>=0.21.1 <0.22.0' peerDependenciesMeta: '@tangle-network/sandbox': optional: true - playwright: - optional: true '@tangle-network/agent-trace-contract@1.0.2': resolution: {integrity: sha512-v7uMh56jkEp4vckevEU9xKsIatbs5dqzGPp69dFLSSXUVit0RP6VD6EANMXVlTCUk+6wVKBLHJx23XspVCEiIA==} - '@tangle-network/sandbox@0.17.2': - resolution: {integrity: sha512-e+p/Uet2nVej3pLLCOoxeRN29t38QfWEWpcnnJsC2b0BNsQEPUg6xNlEgdeT9Eo0U6++QyR+GUrDovJ/houd2g==} + '@tangle-network/sandbox@0.21.1': + resolution: {integrity: sha512-xlqI9fxq9TLCOnmcxOU2XgOD5Ls6IOxLAbfa3SxVa/wUNnUCHtnwRHYteF7zUCwngVs6HR01+9dyChc0Q8m2tA==} peerDependencies: '@mastra/core': ^1.36.0 '@modelcontextprotocol/sdk': ^1.29.0 @@ -1441,50 +1441,51 @@ snapshots: '@standard-schema/spec@1.1.0': {} - '@tangle-network/agent-core@0.4.33': + '@tangle-network/agent-core@0.6.1': dependencies: - '@tangle-network/agent-interface': 0.43.0 + '@tangle-network/agent-interface': 0.47.0 zod: 4.4.3 - '@tangle-network/agent-eval@0.143.0': + '@tangle-network/agent-eval@0.145.3': dependencies: '@asteasolutions/zod-to-openapi': 9.1.0(zod@4.4.3) '@hono/node-server': 2.0.12(hono@4.12.32) - '@tangle-network/agent-core': 0.4.33 - '@tangle-network/agent-interface': 0.43.0 + '@tangle-network/agent-core': 0.6.1 + '@tangle-network/agent-interface': 0.47.0 '@tangle-network/agent-trace-contract': 1.0.2 hono: 4.12.32 linear-sum-assignment: 1.0.9 re2js: 2.8.6 zod: 4.4.3 - '@tangle-network/agent-interface@0.43.0': + '@tangle-network/agent-interface@0.47.0': dependencies: '@noble/hashes': 1.8.0 spdx-expression-parse: 5.0.0 zod: 4.4.3 - '@tangle-network/agent-knowledge@7.0.8': + '@tangle-network/agent-knowledge@7.2.4(@tangle-network/agent-eval@0.145.3)(@tangle-network/agent-interface@0.47.0)': dependencies: - '@tangle-network/agent-eval': 0.143.0 - '@tangle-network/agent-interface': 0.43.0 + '@tangle-network/agent-eval': 0.145.3 + '@tangle-network/agent-interface': 0.47.0 proper-lockfile: 4.1.2 zod: 4.4.3 - '@tangle-network/agent-profile-materialize@0.10.2(@tangle-network/agent-interface@0.43.0)': + '@tangle-network/agent-profile-materialize@0.14.0(@tangle-network/agent-interface@0.47.0)': dependencies: - '@tangle-network/agent-interface': 0.43.0 + '@tangle-network/agent-interface': 0.47.0 - '@tangle-network/agent-runtime@0.126.0(@tangle-network/agent-eval@0.143.0)(@tangle-network/agent-interface@0.43.0)(@tangle-network/sandbox@0.17.2)': + '@tangle-network/agent-runtime@0.133.2(@tangle-network/agent-eval@0.145.3)(@tangle-network/agent-interface@0.47.0)(@tangle-network/sandbox@0.21.1)': dependencies: - '@tangle-network/agent-eval': 0.143.0 - '@tangle-network/agent-interface': 0.43.0 - '@tangle-network/agent-knowledge': 7.0.8 - '@tangle-network/agent-profile-materialize': 0.10.2(@tangle-network/agent-interface@0.43.0) + '@tangle-network/agent-core': 0.6.1 + '@tangle-network/agent-eval': 0.145.3 + '@tangle-network/agent-interface': 0.47.0 + '@tangle-network/agent-knowledge': 7.2.4(@tangle-network/agent-eval@0.145.3)(@tangle-network/agent-interface@0.47.0) + '@tangle-network/agent-profile-materialize': 0.14.0(@tangle-network/agent-interface@0.47.0) '@tangle-network/agent-trace-contract': 1.0.2 tar-stream: 3.2.0 optionalDependencies: - '@tangle-network/sandbox': 0.17.2 + '@tangle-network/sandbox': 0.21.1 transitivePeerDependencies: - bare-abort-controller - bare-buffer @@ -1492,10 +1493,10 @@ snapshots: '@tangle-network/agent-trace-contract@1.0.2': {} - '@tangle-network/sandbox@0.17.2': + '@tangle-network/sandbox@0.21.1': dependencies: - '@tangle-network/agent-core': 0.4.33 - '@tangle-network/agent-interface': 0.43.0 + '@tangle-network/agent-core': 0.6.1 + '@tangle-network/agent-interface': 0.47.0 zod: 4.4.3 '@tybys/wasm-util@0.10.3': diff --git a/src/analyst-model-call.ts b/src/analyst-model-call.ts new file mode 100644 index 0000000..d90e1ef --- /dev/null +++ b/src/analyst-model-call.ts @@ -0,0 +1,111 @@ +import { + callLlm, + costReceiptFromLlm, + costReceiptFromLlmError, + LlmCallError, + type LlmCallRequest, +} from '@tangle-network/agent-eval' +import type { createDspyRlmTraceEngine } from '@tangle-network/agent-eval/analyst' + +/** + * The engine's own model-call seam. Derived from the factory rather than + * imported, because agent-eval does not export the contract type by name and a + * hand-written copy would silently drift from the version installed here. + */ +type ExternalOptimizerModelCall = NonNullable[0]['call']> + +/** + * How much of a provider error body is retained as execution evidence. Enough + * to name the cause, bounded because the body is attacker-influenced and the + * proxy holds the whole execution record in memory. + */ +export const PROVIDER_ERROR_BODY_LIMIT = 2_000 + +/** + * The CLI's owned execution path for one admitted analyst model call. + * + * agent-eval 0.144.0 stopped accepting provider credentials: the caller owns + * execution and returns a typed result plus a cost receipt for every admitted + * call. This path is one OpenAI-compatible HTTP call through agent-eval's own + * `callLlm`. + * + * The callback always resolves. Rejecting loses the execution record and fails + * the whole optimizer attempt. + */ +export function createAnalystModelCall(opts: { + apiKey: string + baseUrl: string +}): ExternalOptimizerModelCall { + const { apiKey, baseUrl } = opts + return async ({ request, callId, signal }) => { + try { + const req = structuredClone(request) as LlmCallRequest + // callId is the ledger's stable identity for this one paid call, so it + // is the provider idempotency key: callLlm retries transient failures, + // and without it a lost-but-billed response is charged twice. + const response = await callLlm(req, { + apiKey, + baseUrl, + signal, + idempotencyKey: callId, + }) + return { + succeeded: true, + response, + receipt: costReceiptFromLlm(response), + execution: { + callId, + baseUrl, + requestedModel: req.model, + servedModel: response.servedModel ?? null, + finishReason: response.finishReason ?? null, + durationMs: response.durationMs, + }, + } + } catch (error) { + const err = error instanceof Error ? error : new Error(String(error)) + // The callback must resolve, so an abort cannot reach the proxy as a + // thrown AbortError and can never take its 504 branch. Naming the class + // in the failure text is what lets the bridge tell a cancelled call apart + // from a transient one it should retry. + // + // Only the caller's signal proves a cancellation. `callLlm` aborts an + // internal controller to enforce its own per-attempt timeout, so a plain + // provider timeout also surfaces as an `AbortError` while this signal + // stays clear. Reading the error name here would mark that timeout + // uncancellable and stop the bridge retrying a call it should retry. + const aborted = signal.aborted + const message = aborted ? `AbortError: ${err.message}` : err.message + // LlmCallError carries the provider's own reason (bad key, context + // length, unknown model). Without it the operator sees only the HTTP + // status. It stays in the execution evidence and out of the log line, + // because a gateway can echo request headers into an error body. + const detail = + err instanceof LlmCallError + ? { status: err.status, body: err.body.slice(0, PROVIDER_ERROR_BODY_LIMIT) } + : {} + return { + succeeded: false, + error: message, + // costReceiptFromLlmError recovers the provider receipt when the + // response completed but violated the contract; otherwise usage and + // cost stay explicitly unknown. + receipt: costReceiptFromLlmError(err) ?? { + model: request.model, + inputTokens: 0, + outputTokens: 0, + costUnknown: true, + usageUnknown: true, + }, + execution: { + callId, + baseUrl, + requestedModel: request.model, + aborted, + error: message, + ...detail, + }, + } + } + } +} diff --git a/src/cli.ts b/src/cli.ts index ae819b4..cab88bd 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -89,6 +89,7 @@ import { import { fileRunContextSupervisorRunReader, isFileRunContextDir } from './supervisor-run-context.js' import { resolveRunWatchTarget, watchRunTarget } from './run-watch.js' import { createDspyRlmTraceEngine, type TraceAnalysisEngine } from '@tangle-network/agent-eval/analyst' +import { createAnalystModelCall } from './analyst-model-call.js' import type { OtlpSpan } from './otlp.js' import { serializeSpans, writeOtlpFile } from './otlp.js' import type { @@ -501,15 +502,19 @@ function buildAnalysisEngine(model: string, budgetUsd?: number): TraceAnalysisEn (tangleKey ? TANGLE_ROUTER_BASE_URL : 'https://api.openai.com/v1') const python = process.env.TRACES_PYTHON return createDspyRlmTraceEngine({ - apiKey, - baseUrl, + call: createAnalystModelCall({ apiKey, baseUrl }), + callRef: `traces-cli:${baseUrl}#${model}`, + recordExecution: (observation) => { + analystLog( + `[analyst] model call ${observation.sequence} ${observation.succeeded ? 'ok' : 'FAIL'} ${observation.model}`, + observation.succeeded ? undefined : { error: observation.error }, + ) + }, model, - // agent-eval 0.139.3's engine defaults are tuned below what real runs - // need. maxOutputTokens 4096 is under what current coding models emit for - // one findings array: glm-5.2 counts reasoning tokens in - // completion_tokens, the first oversized completion breaches its cost - // reservation, and the fail-closed ledger then refuses every later call - // in the run. + // Pinned, not defaulted: one findings array from a current coding model + // needs this cap, and glm-5.2 counts reasoning tokens in + // completion_tokens — a smaller cap breaches its cost reservation and the + // fail-closed ledger then refuses every later call in the run. maxOutputTokens: 16_384, // maxCostUsd defaults to $1 per analyst — a proxy-side ceiling separate // from --budget. With the larger token cap the per-call reservation grows diff --git a/tests/analyst-model-call.test.ts b/tests/analyst-model-call.test.ts new file mode 100644 index 0000000..5fba11f --- /dev/null +++ b/tests/analyst-model-call.test.ts @@ -0,0 +1,129 @@ +import { createServer, type IncomingMessage, type Server, type ServerResponse } from 'node:http' +import type { AddressInfo } from 'node:net' +import { afterEach, describe, expect, it } from 'vitest' +import { createAnalystModelCall, PROVIDER_ERROR_BODY_LIMIT } from '../src/analyst-model-call.js' + +type Handler = (req: IncomingMessage, res: ServerResponse) => void + +let server: Server | undefined + +async function startGateway(handler: Handler): Promise { + const created = createServer(handler) + server = created + await new Promise((resolve) => created.listen(0, '127.0.0.1', resolve)) + const { port } = created.address() as AddressInfo + return `http://127.0.0.1:${port}/v1` +} + +afterEach(async () => { + const running = server + server = undefined + if (running) await new Promise((resolve) => running.close(() => resolve())) +}) + +function chatRequest(model = 'test-model') { + return { + model, + messages: [{ role: 'user' as const, content: 'why did this run fail?' }], + } +} + +function callArgs(request: ReturnType, signal: AbortSignal) { + // Shape of one admitted call the loopback proxy hands to the execution owner. + return { request, callId: 'call-abc123', signal } as unknown as Parameters< + ReturnType + >[0] +} + +describe('analyst model call', () => { + it('returns a receipt and execution evidence, and sends callId as the idempotency key', async () => { + let seenIdempotencyKey: string | undefined + const baseUrl = await startGateway((req, res) => { + seenIdempotencyKey = req.headers['idempotency-key'] as string | undefined + res.writeHead(200, { 'content-type': 'application/json' }) + res.end( + JSON.stringify({ + id: 'chatcmpl-1', + object: 'chat.completion', + model: 'test-model', + choices: [{ index: 0, message: { role: 'assistant', content: 'ok' }, finish_reason: 'stop' }], + usage: { prompt_tokens: 11, completion_tokens: 7, total_tokens: 18 }, + }), + ) + }) + + const call = createAnalystModelCall({ apiKey: 'test-key', baseUrl }) + const result = await call(callArgs(chatRequest(), new AbortController().signal)) + + expect(result.succeeded).toBe(true) + if (!result.succeeded) return + expect(result.response.content).toBe('ok') + expect(result.receipt.inputTokens).toBe(11) + expect(result.receipt.outputTokens).toBe(7) + // callId must reach the provider so a retried-but-already-billed call is + // not charged twice. + expect(seenIdempotencyKey).toBe('call-abc123') + const execution = result.execution as Record + expect(execution.callId).toBe('call-abc123') + expect(execution.servedModel).toBe('test-model') + expect(execution.finishReason).toBe('stop') + }) + + it('keeps the provider reason in execution evidence on a non-2xx response', async () => { + const baseUrl = await startGateway((_req, res) => { + res.writeHead(400, { 'content-type': 'application/json' }) + res.end(JSON.stringify({ error: { message: 'model not found: test-model', type: 'invalid_request_error' } })) + }) + + const call = createAnalystModelCall({ apiKey: 'test-key', baseUrl }) + const result = await call(callArgs(chatRequest(), new AbortController().signal)) + + expect(result.succeeded).toBe(false) + if (result.succeeded) return + const execution = result.execution as Record + expect(execution.status).toBe(400) + expect(String(execution.body)).toContain('model not found: test-model') + expect(String(execution.body).length).toBeLessThanOrEqual(PROVIDER_ERROR_BODY_LIMIT) + expect(execution.aborted).toBe(false) + // A failed call must never read as a free, measured call. + expect(result.receipt.costUnknown).toBe(true) + expect(result.receipt.usageUnknown).toBe(true) + }) + + it('does not mark a provider timeout as a cancellation', async () => { + const baseUrl = await startGateway(() => { + // Never responds, so callLlm's own per-attempt timeout fires. + }) + + const call = createAnalystModelCall({ apiKey: 'test-key', baseUrl }) + // callLlm aborts an internal controller to enforce this timeout, so the + // error arrives as an AbortError even though nobody cancelled the call. + const request = { ...chatRequest(), timeoutMs: 250 } + const result = await call(callArgs(request, new AbortController().signal)) + + expect(result.succeeded).toBe(false) + if (result.succeeded) return + // The bridge retries a transient failure and gives up on a cancelled one, + // so a timeout must not carry the cancellation marker. + expect(result.error.startsWith('AbortError:')).toBe(false) + expect((result.execution as Record).aborted).toBe(false) + }) + + it('marks an aborted call so it is distinguishable from a transient failure', async () => { + const controller = new AbortController() + const baseUrl = await startGateway(() => { + // Never responds: the abort is the only way this call ends. + controller.abort() + }) + + const call = createAnalystModelCall({ apiKey: 'test-key', baseUrl }) + const result = await call(callArgs(chatRequest(), controller.signal)) + + expect(result.succeeded).toBe(false) + if (result.succeeded) return + expect(result.error.startsWith('AbortError:')).toBe(true) + const execution = result.execution as Record + expect(execution.aborted).toBe(true) + expect(execution.callId).toBe('call-abc123') + }) +})