Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -0,0 +1,48 @@
import * as Sentry from '@sentry/node';
import { Mastra } from '@mastra/core';
import { Classifier } from '@mastra/core/classifier';
import { Observability } from '@mastra/observability';
import { SentryMastraExporter } from '@sentry/node';

// Inlined `EvaluationModelV4` mock: the ESM/CJS runner copies only this file into its temp dir.
const evaluationModel = {
specificationVersion: 'v4',
provider: 'typesafe-ai.evaluation',
modelId: 'jev-latest',
supportedQuestionTypes: ['choice', 'score', 'boolean'],
async doEvaluate() {
throw new Error('evaluation failed');
},
};

async function run() {
const classifier = new Classifier({
id: 'request-classifier',
model: evaluationModel,
questions: { urgent: { type: 'boolean', instructions: 'Does this request need an immediate response?' } },
});

const mastra = new Mastra({
classifiers: { classifier },
logger: false,
observability: new Observability({
configs: {
default: {
serviceName: 'mastra-test',
exporters: [new SentryMastraExporter()],
},
},
}),
});

await Sentry.startSpan({ op: 'function', name: 'mastra-test' }, async () => {
await mastra
.getClassifier('classifier')
.evaluate({ state: { message: 'I was charged twice.' }, maxRetries: 0 })
.catch(() => undefined);
});

await mastra.observability.shutdown();
}

run();
Original file line number Diff line number Diff line change
@@ -0,0 +1,51 @@
import * as Sentry from '@sentry/node';
import { Mastra } from '@mastra/core';
import { Classifier } from '@mastra/core/classifier';
import { Observability } from '@mastra/observability';
import { SentryMastraExporter } from '@sentry/node';

// Inlined `EvaluationModelV4` mock: the ESM/CJS runner copies only this file into its temp dir.
const evaluationModel = {
specificationVersion: 'v4',
provider: 'typesafe-ai.evaluation',
modelId: 'jev-latest',
supportedQuestionTypes: ['choice', 'score', 'boolean'],
async doEvaluate() {
return {
answers: { urgent: { type: 'boolean', probability: 0.9 } },
usage: { inputTokens: 30, outputTokens: 2 },
warnings: [],
};
},
};

async function run() {
const classifier = new Classifier({
id: 'request-classifier',
model: evaluationModel,
questions: { urgent: { type: 'boolean', instructions: 'Does this request need an immediate response?' } },
});

const mastra = new Mastra({
classifiers: { classifier },
logger: false,
observability: new Observability({
configs: {
default: {
serviceName: 'mastra-test',
exporters: [new SentryMastraExporter()],
},
},
}),
});

await Sentry.startSpan({ op: 'function', name: 'mastra-test' }, async () => {
await mastra
.getClassifier('classifier')
.evaluate({ state: { message: 'I was charged twice.', apiKey: 'sk-test-123' } });
});

await mastra.observability.shutdown();
}

run();
117 changes: 109 additions & 8 deletions dev-packages/node-integration-tests/suites/tracing/mastra/test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,7 @@ import {
SENTRY_ORIGIN,
URL_FULL,
} from '@sentry/conventions/attributes';
import { GEN_AI_CHAT, GEN_AI_EVALUATE, GEN_AI_EXECUTE_TOOL, GEN_AI_INVOKE_AGENT } from '@sentry/conventions/op';
import { afterAll, expect } from 'vitest';
import { conditionalTest } from '../../../utils';
import { cleanupChildProcesses, createEsmAndCjsTests } from '../../../utils/runner';
Expand All @@ -43,6 +44,14 @@ const MASTRA_NESTING_DEPENDENCIES = {
},
};

// `Classifier` (the `@mastra/core/classifier` entry) ships in newer `@mastra/core` releases only.
const MASTRA_CLASSIFIER_DEPENDENCIES = {
additionalDependencies: {
'@mastra/core': '1.74.0',
'@mastra/observability': '1.18.3',
},
};

conditionalTest({ min: 22 })('Mastra integration', () => {
afterAll(() => {
cleanupChildProcesses();
Expand All @@ -63,7 +72,7 @@ conditionalTest({ min: 22 })('Mastra integration', () => {

const agentSpan = spans.find(span => span.name === 'invoke_agent weather_agent')!;
expect(agentSpan.status).toBe('ok');
expect(agentSpan.attributes[SENTRY_OP].value).toBe('gen_ai.invoke_agent');
expect(agentSpan.attributes[SENTRY_OP].value).toBe(GEN_AI_INVOKE_AGENT);
expect(agentSpan.attributes[SENTRY_ORIGIN].value).toBe('auto.ai.mastra');
expect(agentSpan.attributes[GEN_AI_OPERATION_NAME].value).toBe('invoke_agent');
expect(agentSpan.attributes[GEN_AI_AGENT_NAME].value).toBe('weather_agent');
Expand All @@ -75,7 +84,7 @@ conditionalTest({ min: 22 })('Mastra integration', () => {

const chatSpan = spans.find(span => span.name === 'chat gpt-4o-mini')!;
expect(chatSpan.status).toBe('ok');
expect(chatSpan.attributes[SENTRY_OP].value).toBe('gen_ai.chat');
expect(chatSpan.attributes[SENTRY_OP].value).toBe(GEN_AI_CHAT);
expect(chatSpan.attributes[SENTRY_ORIGIN].value).toBe('auto.ai.mastra');
expect(chatSpan.attributes[GEN_AI_OPERATION_NAME].value).toBe('chat');
expect(chatSpan.attributes[GEN_AI_REQUEST_MODEL].value).toBe('gpt-4o-mini');
Expand Down Expand Up @@ -157,7 +166,7 @@ conditionalTest({ min: 22 })('Mastra integration', () => {

const toolSpan = spans.find(span => span.name === 'execute_tool get_weather')!;
expect(toolSpan.status).toBe('ok');
expect(toolSpan.attributes[SENTRY_OP].value).toBe('gen_ai.execute_tool');
expect(toolSpan.attributes[SENTRY_OP].value).toBe(GEN_AI_EXECUTE_TOOL);
expect(toolSpan.attributes[SENTRY_ORIGIN].value).toBe('auto.ai.mastra');
expect(toolSpan.attributes[GEN_AI_OPERATION_NAME].value).toBe('execute_tool');
expect(toolSpan.attributes[GEN_AI_TOOL_NAME].value).toBe('get_weather');
Expand Down Expand Up @@ -196,7 +205,7 @@ conditionalTest({ min: 22 })('Mastra integration', () => {
expect(spans.map(span => span.name)).toEqual(['invoke_agent math_workflow']);

const workflowSpan = spans[0]!;
expect(workflowSpan.attributes[SENTRY_OP].value).toBe('gen_ai.invoke_agent');
expect(workflowSpan.attributes[SENTRY_OP].value).toBe(GEN_AI_INVOKE_AGENT);
expect(workflowSpan.attributes[GEN_AI_PIPELINE_NAME].value).toBe('math_workflow');
},
})
Expand All @@ -207,6 +216,98 @@ conditionalTest({ min: 22 })('Mastra integration', () => {
MASTRA_DEPENDENCIES,
);

createEsmAndCjsTests(
__dirname,
'scenario-classifier.mjs',
'instrument.mjs',
(createRunner, test) => {
test('maps a classifier evaluation to a gen_ai.evaluate span', async () => {
await createRunner()
.expect({
span: container => {
expect(container.items.find(span => span.is_segment && span.name === 'mastra-test')).toBeDefined();
const spans = container.items.filter(span => span.attributes[SENTRY_ORIGIN]?.value === 'auto.ai.mastra');
expect(spans.map(span => span.name)).toEqual(['evaluate jev-latest']);

const evaluateSpan = spans[0]!;
expect(evaluateSpan.attributes[SENTRY_OP].value).toBe(GEN_AI_EVALUATE);
expect(evaluateSpan.attributes[SENTRY_ORIGIN].value).toBe('auto.ai.mastra');
expect(evaluateSpan.attributes[GEN_AI_OPERATION_NAME].value).toBe('evaluate');
expect(evaluateSpan.attributes[GEN_AI_REQUEST_MODEL].value).toBe('jev-latest');
expect(evaluateSpan.attributes[GEN_AI_PROVIDER_NAME].value).toBe('typesafe-ai.evaluation');
expect(evaluateSpan.attributes[GEN_AI_USAGE_INPUT_TOKENS].value).toBe(30);
expect(evaluateSpan.attributes[GEN_AI_USAGE_OUTPUT_TOKENS].value).toBe(2);
expect(evaluateSpan.attributes[GEN_AI_USAGE_TOTAL_TOKENS].value).toBe(32);
expect(evaluateSpan.attributes[GEN_AI_INPUT_MESSAGES]).toBeUndefined();
expect(evaluateSpan.attributes[GEN_AI_OUTPUT_MESSAGES]).toBeUndefined();
},
})
.start()
.completed();
});
},
MASTRA_CLASSIFIER_DEPENDENCIES,
);

createEsmAndCjsTests(
__dirname,
'scenario-classifier.mjs',
'instrument-with-pii.mjs',
(createRunner, test) => {
test('records the evaluated state, questions and answers when genAI recording is on', async () => {
await createRunner()
.expect({
span: container => {
expect(container.items.find(span => span.is_segment && span.name === 'mastra-test')).toBeDefined();
const evaluateSpan = container.items.find(span => span.name === 'evaluate jev-latest')!;
expect(JSON.parse(evaluateSpan.attributes[GEN_AI_INPUT_MESSAGES].value)).toEqual([
{
type: 'evaluation',
state: { message: 'I was charged twice.', apiKey: '[REDACTED]' },
questions: {
urgent: { type: 'boolean', instructions: 'Does this request need an immediate response?' },
},
},
]);
expect(JSON.parse(evaluateSpan.attributes[GEN_AI_OUTPUT_MESSAGES].value)).toEqual([
{ type: 'evaluation', answers: { urgent: { type: 'boolean', probability: 0.9 } } },
]);
},
})
.start()
.completed();
});
},
MASTRA_CLASSIFIER_DEPENDENCIES,
);

createEsmAndCjsTests(
__dirname,
'scenario-classifier-error.mjs',
'instrument-with-pii.mjs',
(createRunner, test) => {
test('ends a failed classifier evaluation span with an error status', async () => {
await createRunner()
.ignore('event')
.expect({
span: container => {
expect(container.items.find(span => span.is_segment && span.name === 'mastra-test')).toBeDefined();
const spans = container.items.filter(span => span.attributes[SENTRY_ORIGIN]?.value === 'auto.ai.mastra');
expect(spans.map(span => span.name)).toEqual(['evaluate jev-latest']);

const evaluateSpan = spans[0]!;
expect(evaluateSpan.status).toBe('error');
expect(evaluateSpan.attributes[GEN_AI_INPUT_MESSAGES].value).toContain('I was charged twice.');
expect(evaluateSpan.attributes[GEN_AI_OUTPUT_MESSAGES]).toBeUndefined();
},
})
.start()
.completed();
});
},
MASTRA_CLASSIFIER_DEPENDENCIES,
);

createEsmAndCjsTests(
__dirname,
'scenario-auto.mjs',
Expand Down Expand Up @@ -302,14 +403,14 @@ conditionalTest({ min: 22 })('Mastra integration', () => {
);
expect(loadSpans.length).toBeGreaterThan(0);
const cacheGetParentIds = loadSpans.map(span => span.parent_span_id);
const chat = container.items.find(span => span.attributes[SENTRY_OP]?.value === 'gen_ai.chat')!;
const chat = container.items.find(span => span.attributes[SENTRY_OP]?.value === GEN_AI_CHAT)!;
expect(chat).toBeDefined();
const chatSpanId = chat.span_id;

const executeTool = container.items.find(
span => span.attributes[GEN_AI_TOOL_NAME]?.value === 'count_items',
)!;
expect(executeTool.attributes[SENTRY_OP]?.value).toBe('gen_ai.execute_tool');
expect(executeTool.attributes[SENTRY_OP]?.value).toBe(GEN_AI_EXECUTE_TOOL);
const executeToolSpanId = executeTool.span_id;

// The `executeWithContext` bridge makes the exporter spans active during the real work, so the
Expand Down Expand Up @@ -355,14 +456,14 @@ conditionalTest({ min: 22 })('Mastra integration', () => {
);
expect(loadSpans.length).toBeGreaterThan(0);
const cacheGetParentIds = loadSpans.map(span => span.parent_span_id);
const chat = container.items.find(span => span.attributes[SENTRY_OP]?.value === 'gen_ai.chat')!;
const chat = container.items.find(span => span.attributes[SENTRY_OP]?.value === GEN_AI_CHAT)!;
expect(chat).toBeDefined();
const chatSpanId = chat.span_id;

const executeTool = container.items.find(
span => span.attributes[GEN_AI_TOOL_NAME]?.value === 'count_items',
)!;
expect(executeTool.attributes[SENTRY_OP]?.value).toBe('gen_ai.execute_tool');
expect(executeTool.attributes[SENTRY_OP]?.value).toBe(GEN_AI_EXECUTE_TOOL);
const executeToolSpanId = executeTool.span_id;

expect(chatSpanId).toBeDefined();
Expand Down
10 changes: 10 additions & 0 deletions packages/server-utils/src/ai/core/utils.ts
Original file line number Diff line number Diff line change
Expand Up @@ -136,6 +136,16 @@ export function getTokenUsageAttributes(
return attributes;
}

/** Serialize the `state` and `questions` of an evaluation request (TypeSafe, Workers AI, Mastra classifiers). */
export function getEvaluationInputMessages(request: Record<string, unknown>): string | undefined {
return stringify([{ type: 'evaluation', state: request.state, questions: request.questions }]);
}

/** Serialize the `answers` of an evaluation result (TypeSafe, Workers AI, Mastra classifiers). */
export function getEvaluationOutputMessages(answers: unknown): string | undefined {
return stringify([{ type: 'evaluation', answers }]);
}

/** One assistant turn for {@link setOutputMessagesAttribute}. */
export interface GenAiOutputMessage {
/** The message's text content, already flattened out of any content-part array. */
Expand Down
58 changes: 58 additions & 0 deletions packages/server-utils/src/ai/mastra/classifier-evaluation.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,58 @@
import type { Span, SpanTimeInput } from '@sentry/core';
import type { MastraExportedSpan, MastraSpanOutputProcessor } from './types';

/**
* Links a `Classifier.evaluate()` call to the Sentry span the exporter opens for its
* `classifier_evaluation` span. Mastra's span carries no input or output, so the integration adds
* them from the call. Mastra ends its span before `evaluate()` returns the answers, so the exporter
* leaves the Sentry span open and the integration ends it once the call settles.
*/
export interface ClassifierEvaluationCall {
span?: Span;
/** Mastra's span and the processors of its observability instance, to filter the call's data. */
mastraSpan?: MastraExportedSpan;
spanOutputProcessors?: MastraSpanOutputProcessor[];
/** The recording options of the exporter that opened the span, so the call's data follows them. */
recordInputs?: boolean;
recordOutputs?: boolean;
/** Mastra's end time, set when Mastra ended its span before the call settled. */
endTime?: SpanTimeInput;
settled?: boolean;
}

// Mastra emits `span_started` synchronously inside `evaluate()`, before its first `await`, so at
// most one call is ever in this window.
let startingCall: ClassifierEvaluationCall | undefined;

export function setStartingClassifierEvaluation(call: ClassifierEvaluationCall | undefined): void {
startingCall = call;
}

export function takeStartingClassifierEvaluation(): ClassifierEvaluationCall | undefined {
const call = startingCall;
startingCall = undefined;
return call;
}

/**
* Run the call's data through Mastra's span output processors (for example its default
* `SensitiveDataFilter`), as Mastra does for the input and output of its own spans. Returns `undefined`
* when a processor drops the span or throws, so unfiltered data is never recorded.
*/
export function processClassifierEvaluationData(
call: ClassifierEvaluationCall,
data: { input?: unknown; output?: unknown },
): { input?: unknown; output?: unknown } | undefined {
let span: MastraExportedSpan | undefined = { ...call.mastraSpan, ...data } as MastraExportedSpan;
for (const processor of call.spanOutputProcessors ?? []) {
try {
span = processor.process(span);
} catch {
return undefined;
}
if (!span) {
return undefined;
}
}
return span;
}
9 changes: 8 additions & 1 deletion packages/server-utils/src/ai/mastra/constants.ts
Original file line number Diff line number Diff line change
@@ -1,4 +1,10 @@
import { GEN_AI_CHAT, GEN_AI_EMBEDDINGS, GEN_AI_EXECUTE_TOOL, GEN_AI_INVOKE_AGENT } from '@sentry/conventions/op';
import {
GEN_AI_CHAT,
GEN_AI_EMBEDDINGS,
GEN_AI_EVALUATE,
GEN_AI_EXECUTE_TOOL,
GEN_AI_INVOKE_AGENT,
} from '@sentry/conventions/op';
import type { MastraSpanType } from './types';

export const MASTRA_INTEGRATION_NAME = 'Mastra' as const;
Expand Down Expand Up @@ -36,6 +42,7 @@ export const SPAN_TYPE_OPS: Readonly<Record<string, { op: string; operationName:
provider_tool_call: { op: GEN_AI_EXECUTE_TOOL, operationName: 'execute_tool' },
client_tool_call: { op: GEN_AI_EXECUTE_TOOL, operationName: 'execute_tool' },
rag_embedding: { op: GEN_AI_EMBEDDINGS, operationName: 'embeddings' },
classifier_evaluation: { op: GEN_AI_EVALUATE, operationName: 'evaluate' },
};

export const TOOL_SPAN_TYPES: ReadonlySet<string> = new Set<MastraSpanType>([
Expand Down
Loading
Loading