Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -34,6 +34,14 @@ export default Sentry.withSentry(
return new Response(JSON.stringify(result));
}

if (url.pathname === '/evaluate') {
const result = await ai.run('typesafe/jev', {
state: 'Help! My payouts have been failing for 3 days.',
questions: { is_urgent: { type: 'noul', instructions: 'Does this convey urgency?' } },
});
return new Response(JSON.stringify(result));
}

if (url.pathname === '/stream') {
const stream = (await ai.run('@cf/meta/llama-3.1-8b-instruct', {
messages: [{ role: 'user', content: 'What is the capital of France?' }],
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -23,6 +23,17 @@ export class MockAi {
throw error;
}

if (model === 'typesafe/jev') {
return {
state: 'Completed',
result: {
model: 'jev-1.13.0',
answers: { is_urgent: { type: 'noul', noul: 0.97 } },
usage: { input_tokens: 426, output_tokens: 73 },
},
};
}

if (inputs?.stream === true) {
return createSseStream([
'{"response":"The capital "}',
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,7 @@ import {
GEN_AI_REQUEST_MAX_TOKENS,
GEN_AI_REQUEST_MODEL,
GEN_AI_REQUEST_TEMPERATURE,
GEN_AI_RESPONSE_MODEL,
GEN_AI_RESPONSE_STREAMING,
GEN_AI_RESPONSE_TEXT,
GEN_AI_USAGE_INPUT_TOKENS,
Expand Down Expand Up @@ -132,6 +133,52 @@ it('traces a streaming Workers AI text generation request', async ({ signal }) =
await runner.completed();
});

it('traces a TypeSafe Jev evaluation like the TypeSafe integration', async ({ signal }) => {
const runner = createRunner(__dirname)
.ignore('event')
.expect(envelope => {
const spans = getSpansFromEnvelope(envelope);
const segmentSpan = spans.find(span => span.is_segment);

const genAiSpans = spans.filter(span => getSpanOp(span)?.startsWith('gen_ai.'));
expect(genAiSpans).toHaveLength(1);

expect(genAiSpans[0]).toEqual(
expect.objectContaining({
name: 'evaluate typesafe/jev',
status: 'ok',
is_segment: false,
attributes: {
'sentry.origin': { value: 'auto.ai.cloudflare.workers_ai', type: 'string' },
'sentry.op': { value: 'gen_ai.evaluate', type: 'string' },
[GEN_AI_PROVIDER_NAME]: { value: 'cloudflare.workers_ai', type: 'string' },
[GEN_AI_OPERATION_NAME]: { value: 'evaluate', type: 'string' },
[GEN_AI_REQUEST_MODEL]: { value: 'typesafe/jev', type: 'string' },
[GEN_AI_RESPONSE_MODEL]: { value: 'jev-1.13.0', type: 'string' },
[GEN_AI_USAGE_INPUT_TOKENS]: { value: 426, type: 'integer' },
[GEN_AI_USAGE_OUTPUT_TOKENS]: { value: 73, type: 'integer' },
[GEN_AI_USAGE_TOTAL_TOKENS]: { value: 499, type: 'integer' },
// collect only output messages
[GEN_AI_OUTPUT_MESSAGES]: {
type: 'string',
value: '[{"type":"evaluation","answers":{"is_urgent":{"type":"noul","noul":0.97}}}]',
},
'sentry.is_localhost': { value: true, type: 'boolean' },
[SENTRY_TRACE_LIFECYCLE]: { value: 'stream', type: 'string' },
[SENTRY_SEGMENT_NAME]: { value: segmentSpan!.name, type: 'string' },
[SENTRY_SEGMENT_ID]: { value: segmentSpan!.span_id, type: 'string' },
[SENTRY_SDK_NAME]: { value: 'sentry.javascript.cloudflare', type: 'string' },
[SENTRY_SDK_VERSION]: { value: SDK_VERSION, type: 'string' },
[SEMANTIC_ATTRIBUTE_SENTRY_ENVIRONMENT]: { value: 'production', type: 'string' },
},
}),
);
})
.start(signal);
await runner.makeRequest('get', '/evaluate');
await runner.completed();
});

// The Workers AI integration deliberately does not call `captureException` itself.
// When a `run` call fails, the error must bubble up out of the fetch handler and be
// reported by the top-level Cloudflare instrumentation instead — so it shows up in
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -31,6 +31,13 @@ export default {
await env.AI.run('@cf/meta/llama-3.2-1b-instruct', { prompt: 'Say hi', max_tokens: 5 });
return Response.json({ traceId: spanContext?.traceId });
}
case '/test-workers-ai-jev': {
await env.AI.run('typesafe/jev', {
state: 'Help! My payouts have been failing for 3 days.',
questions: { is_urgent: { type: 'noul', instructions: 'Does this convey urgency?' } },
});
return Response.json({ traceId: spanContext?.traceId });
}
case '/test-span':
return Response.json({ spanId: spanContext?.spanId, traceId: spanContext?.traceId });
case '/test-workflow-sleep': {
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -93,3 +93,24 @@ test('Sends a Workers AI gen_ai span to Sentry', async () => {
'gen_ai.output.messages': expect.stringContaining('"role":"assistant"'),
});
});

test('Sends a Workers AI gen_ai.evaluate span for a TypeSafe Jev call to Sentry', async () => {
const { traceId }: { traceId: string } = JSON.parse(await fetchFromWorker(`${workerUrl}/test-workers-ai-jev`, 200));

console.log(`Polling for gen_ai.evaluate span: sentry trace view ${traceTarget(traceId)}`);

let spanId: string | undefined;
await expect
.poll(() => (spanId = findSpanInTrace(traceId, 'gen_ai.evaluate')?.event_id), EVENT_POLLING_OPTIONS)
.toBeDefined();

await expect
.poll(() => fetchSpanAttributes(traceId, spanId!), EVENT_POLLING_OPTIONS)
.toMatchObject({
'gen_ai.operation.name': 'evaluate',
'gen_ai.request.model': 'typesafe/jev',
'gen_ai.response.model': expect.stringMatching(/^jev-/),
'gen_ai.input.messages': expect.stringContaining('My payouts have been failing'),
'gen_ai.output.messages': expect.stringContaining('"is_urgent"'),
});
});
18 changes: 10 additions & 8 deletions packages/server-utils/src/ai/typesafe/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -49,17 +49,19 @@ function getRequestAttributes(
[GEN_AI_OPERATION_NAME]: 'evaluate',
[GEN_AI_PROVIDER_NAME]: TYPESAFE_PROVIDER_NAME,
...(model ? { [GEN_AI_REQUEST_MODEL]: model } : {}),
...(recordInputs
? {
[GEN_AI_INPUT_MESSAGES]: stringify([
{ type: 'evaluation', state: request.state, questions: request.questions },
]),
}
: {}),
...(recordInputs ? { [GEN_AI_INPUT_MESSAGES]: getEvaluationInputMessages(request) } : {}),
};
}

/** Add the response model, token usage and (optionally) the answers of a `systemOne` result. */
/** Serialize the `state` and `questions` of an evaluation request. Also used for TypeSafe models on Workers AI. */
export function getEvaluationInputMessages(request: Record<string, unknown>): string | undefined {
return stringify([{ type: 'evaluation', state: request.state, questions: request.questions }]);
}

/**
* Add the response model, token usage and (optionally) the answers of a `systemOne` result.
* Also used for TypeSafe models on Workers AI.
*/
export function addResponseAttributes(span: Span, result: unknown, recordOutputs: boolean): void {
if (!isObjectLike(result)) {
return;
Expand Down
4 changes: 2 additions & 2 deletions packages/server-utils/src/ai/workers-ai/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -89,7 +89,7 @@ function instrumentRun(
}

// The model did not actually return a stream — finalize the span eagerly.
addResponseAttributes(span, result, options.recordOutputs);
addResponseAttributes(span, result, options.recordOutputs, operationName);
span.end();
return result;
}, handleError);
Expand All @@ -105,7 +105,7 @@ function instrumentRun(

return originalResult.then(result => {
if (!returnsRawResponse) {
addResponseAttributes(span, result, options.recordOutputs);
addResponseAttributes(span, result, options.recordOutputs, operationName);
}
return result;
});
Expand Down
36 changes: 32 additions & 4 deletions packages/server-utils/src/ai/workers-ai/utils.ts
Original file line number Diff line number Diff line change
Expand Up @@ -16,10 +16,16 @@ import {
GEN_AI_SYSTEM_INSTRUCTIONS,
} from '@sentry/conventions/attributes';
import { GEN_AI_CHAT, GEN_AI_EMBEDDINGS } from '@sentry/conventions/op';
import { SEMANTIC_ATTRIBUTE_SENTRY_ORIGIN, stringify } from '@sentry/core';
import { isObjectLike, SEMANTIC_ATTRIBUTE_SENTRY_ORIGIN, stringify } from '@sentry/core';
import type { Span, SpanAttributeValue } from '@sentry/core';
import { GEN_AI_REQUEST_STREAM_ATTRIBUTE } from '../core/gen-ai-attributes';
import { extractSystemInstructions, setOutputMessagesAttribute, setTokenUsageAttributes } from '../core/utils';
import {
extractSystemInstructions,
getGenAiSpanOp,
setOutputMessagesAttribute,
setTokenUsageAttributes,
} from '../core/utils';
import { addResponseAttributes as addEvaluateResponseAttributes, getEvaluationInputMessages } from '../typesafe';
// Re-exported so `workers-ai/streaming.ts` keeps importing it from this module.
export { setOutputMessagesAttribute };
import { WORKERS_AI_ORIGIN, WORKERS_AI_PROVIDER_NAME } from './constants';
Expand All @@ -29,15 +35,20 @@ import type { WorkersAiInput, WorkersAiOutput } from './types';
* Determine the gen_ai operation name from the inputs passed to `AI.run`.
* Workers AI exposes a single `run` method, so we infer the operation from the input shape.
*/
export type WorkersAiOperationName = 'chat' | 'embeddings';
export type WorkersAiOperationName = 'chat' | 'embeddings' | 'evaluate';

export const WORKERS_AI_OPERATION_SPAN_OPS: Record<WorkersAiOperationName, string> = {
chat: GEN_AI_CHAT,
embeddings: GEN_AI_EMBEDDINGS,
evaluate: getGenAiSpanOp('evaluate'),
};

export function getOperationName(inputs: unknown): WorkersAiOperationName {
if (inputs && typeof inputs === 'object') {
// TypeSafe evaluation models (e.g. `typesafe/jev`)
if ('state' in inputs && 'questions' in inputs) {
return 'evaluate';
}
if ('messages' in inputs || 'prompt' in inputs) {
return 'chat';
}
Expand Down Expand Up @@ -102,6 +113,11 @@ export function addRequestAttributes(span: Span, inputs: unknown, operationName:
}
const params = inputs as WorkersAiInput;

if (operationName === 'evaluate') {
span.setAttribute(GEN_AI_INPUT_MESSAGES, getEvaluationInputMessages(inputs as Record<string, unknown>));
return;
}

// Store embeddings input on a separate attribute
if (operationName === 'embeddings') {
const text = params.text;
Expand Down Expand Up @@ -132,7 +148,19 @@ export function addRequestAttributes(span: Span, inputs: unknown, operationName:
/**
* Record the response attributes (token usage, response text, tool calls) on the span.
*/
export function addResponseAttributes(span: Span, result: unknown, recordOutputs: boolean): void {
export function addResponseAttributes(
span: Span,
result: unknown,
recordOutputs: boolean,
operationName?: WorkersAiOperationName,
): void {
if (operationName === 'evaluate') {
// The binding wraps the TypeSafe response as `{ state, result }`, but the docs show it unwrapped.
const body = isObjectLike(result) && isObjectLike(result.result) ? result.result : result;
addEvaluateResponseAttributes(span, body, recordOutputs);
return;
}

if (
!result ||
typeof result !== 'object' ||
Expand Down
43 changes: 43 additions & 0 deletions packages/server-utils/test/ai/lib/tracing/workers-ai.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -5,8 +5,12 @@ import {
GEN_AI_OUTPUT_MESSAGES,
GEN_AI_PROVIDER_NAME,
GEN_AI_REQUEST_MODEL,
GEN_AI_RESPONSE_MODEL,
GEN_AI_RESPONSE_TEXT,
GEN_AI_SYSTEM_INSTRUCTIONS,
GEN_AI_USAGE_INPUT_TOKENS,
GEN_AI_USAGE_OUTPUT_TOKENS,
GEN_AI_USAGE_TOTAL_TOKENS,
} from '@sentry/conventions/attributes';
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest';
import {
Expand Down Expand Up @@ -160,6 +164,45 @@ describe('instrumentWorkersAiClient', () => {
});
});

it('records TypeSafe models as evaluate spans, like the TypeSafe integration', async () => {
const client = new TestClient(
getDefaultTestClientOptions({ dsn: 'https://public@dsn.ingest.sentry.io/1337', tracesSampleRate: 1 }),
);
setCurrentClient(client);
client.init();
const endedSpans: Span[] = [];
client.on('spanEnd', span => endedSpans.push(span));

const questions = { is_urgent: { type: 'noul', instructions: 'Does this convey urgency?' } };
const answers = { is_urgent: { type: 'noul', noul: 0.97 } };
const ai = {
run: vi.fn().mockResolvedValue({
state: 'Completed',
result: { model: 'jev-1.13.0', answers, usage: { input_tokens: 426, output_tokens: 73 } },
}),
};

await instrumentWorkersAiClient(ai).run('typesafe/jev', { state: 'Help!', questions });

const span = spanToStaticSpanJSON(endedSpans[0]!);
expect(span.description).toBe('evaluate typesafe/jev');
expect(span.data).toEqual({
[SEMANTIC_ATTRIBUTE_SENTRY_ORIGIN]: 'auto.ai.cloudflare.workers_ai',
[SEMANTIC_ATTRIBUTE_SENTRY_OP]: 'gen_ai.evaluate',
[SEMANTIC_ATTRIBUTE_SENTRY_SAMPLE_RATE]: 1,
[SENTRY_SEGMENT_NAME_SOURCE]: 'custom',
[GEN_AI_PROVIDER_NAME]: 'cloudflare.workers_ai',
[GEN_AI_OPERATION_NAME]: 'evaluate',
[GEN_AI_REQUEST_MODEL]: 'typesafe/jev',
[GEN_AI_RESPONSE_MODEL]: 'jev-1.13.0',
[GEN_AI_USAGE_INPUT_TOKENS]: 426,
[GEN_AI_USAGE_OUTPUT_TOKENS]: 73,
[GEN_AI_USAGE_TOTAL_TOKENS]: 499,
[GEN_AI_INPUT_MESSAGES]: JSON.stringify([{ type: 'evaluation', state: 'Help!', questions }]),
[GEN_AI_OUTPUT_MESSAGES]: JSON.stringify([{ type: 'evaluation', answers }]),
});
});

describe('span names', () => {
function setupClient(traceLifecycle: 'static' | 'stream'): Span[] {
const client = new TestClient(
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,7 @@ import {
GEN_AI_REQUEST_TOP_K,
GEN_AI_REQUEST_TOP_P,
GEN_AI_OUTPUT_MESSAGES,
GEN_AI_RESPONSE_MODEL,
GEN_AI_RESPONSE_TEXT,
GEN_AI_RESPONSE_TOOL_CALLS,
GEN_AI_SYSTEM_INSTRUCTIONS,
Expand Down Expand Up @@ -59,6 +60,10 @@ describe('workers-ai utils', () => {
expect(getOperationName({ text: 'embed me' })).toBe('embeddings');
});

it('returns "evaluate" for TypeSafe state and questions inputs', () => {
expect(getOperationName({ state: 'Help!', questions: {} })).toBe('evaluate');
});

it('prefers "chat" when both messages and text are present', () => {
expect(getOperationName({ messages: [{ role: 'user', content: 'Hi' }], text: 'embed me' })).toBe('chat');
});
Expand Down Expand Up @@ -185,6 +190,29 @@ describe('workers-ai utils', () => {
});

describe('addResponseAttributes', () => {
it.each([
{ name: 'wrapped in { state, result }', wrap: (body: object) => ({ state: 'Completed', result: body }) },
{ name: 'unwrapped', wrap: (body: object) => body },
])('records evaluate responses $name', ({ wrap }) => {
const { span, attributes } = createMockSpan();
const answers = { is_urgent: { type: 'noul', noul: 0.95 } };

addResponseAttributes(
span,
wrap({ model: 'jev-1.13.0', answers, usage: { input_tokens: 366, output_tokens: 50 } }),
true,
'evaluate',
);

expect(attributes).toEqual({
[GEN_AI_RESPONSE_MODEL]: 'jev-1.13.0',
[GEN_AI_USAGE_INPUT_TOKENS]: 366,
[GEN_AI_USAGE_OUTPUT_TOKENS]: 50,
[GEN_AI_USAGE_TOTAL_TOKENS]: 416,
[GEN_AI_OUTPUT_MESSAGES]: JSON.stringify([{ type: 'evaluation', answers }]),
});
});

it('sets token usage and computes the total from input and output tokens', () => {
const { span, attributes } = createMockSpan();

Expand Down
Loading