Skip to content

Commit 602f245

Browse files
authored
fix(v10/ai): Normalize token usage and preserve cache breakdowns (#25126)
Backport of: #25074 ## Differences to the original PR - Applied the instrumentation and tests under `packages/core/src/tracing/` and `packages/core/test/lib/`, where they live in v10, instead of `packages/server-utils`. Adapted imports to v10's local helpers and attribute constants, retaining its truncation and recording APIs. All token-accounting changes and regression cases are preserved.
1 parent 527414e commit 602f245

17 files changed

Lines changed: 885 additions & 162 deletions

File tree

‎packages/core/src/tracing/ai/utils.ts‎

Lines changed: 43 additions & 52 deletions
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,11 @@
11
/**
22
* Shared utils for AI integrations (OpenAI, Anthropic, Verce.AI, etc.)
33
*/
4+
import {
5+
GEN_AI_USAGE_CACHE_CREATION_INPUT_TOKENS,
6+
GEN_AI_USAGE_CACHE_READ_INPUT_TOKENS,
7+
GEN_AI_USAGE_REASONING_OUTPUT_TOKENS,
8+
} from '@sentry/conventions/attributes';
49
import { getClient } from '../../currentScopes';
510
import { hasSpanStreamingEnabled } from '../spans/hasSpanStreamingEnabled';
611
import type { Span } from '../../types/span';
@@ -85,47 +90,37 @@ export function buildMethodPath(currentPath: string, prop: string): string {
8590
}
8691

8792
/**
88-
* Set token usage attributes
89-
* @param span - The span to add attributes to
90-
* @param promptTokens - The number of prompt tokens
91-
* @param completionTokens - The number of completion tokens
92-
* @param cachedInputTokens - The number of cached input tokens
93-
* @param cachedOutputTokens - The number of cached output tokens
93+
* Build token usage attributes. Input tokens include cache reads and writes, which are subsets.
9494
*/
95-
export function setTokenUsageAttributes(
96-
span: Span,
95+
export function getTokenUsageAttributes(
9796
promptTokens?: number,
9897
completionTokens?: number,
99-
cachedInputTokens?: number,
100-
cachedOutputTokens?: number,
101-
): void {
102-
if (promptTokens !== undefined) {
103-
span.setAttributes({
104-
[GEN_AI_USAGE_INPUT_TOKENS_ATTRIBUTE]: promptTokens,
105-
});
98+
cacheCreationInputTokens?: number,
99+
cacheReadInputTokens?: number,
100+
totalTokens?: number,
101+
): Record<string, number> {
102+
const attributes: Record<string, number> = {};
103+
104+
if (typeof promptTokens === 'number') {
105+
attributes[GEN_AI_USAGE_INPUT_TOKENS_ATTRIBUTE] = promptTokens;
106+
}
107+
if (typeof cacheCreationInputTokens === 'number') {
108+
attributes[GEN_AI_USAGE_CACHE_CREATION_INPUT_TOKENS] = cacheCreationInputTokens;
106109
}
107-
if (completionTokens !== undefined) {
108-
span.setAttributes({
109-
[GEN_AI_USAGE_OUTPUT_TOKENS_ATTRIBUTE]: completionTokens,
110-
});
110+
if (typeof cacheReadInputTokens === 'number') {
111+
attributes[GEN_AI_USAGE_CACHE_READ_INPUT_TOKENS] = cacheReadInputTokens;
111112
}
112-
if (
113-
promptTokens !== undefined ||
114-
completionTokens !== undefined ||
115-
cachedInputTokens !== undefined ||
116-
cachedOutputTokens !== undefined
117-
) {
118-
/**
119-
* Total input tokens in a request is the summation of `input_tokens`,
120-
* `cache_creation_input_tokens`, and `cache_read_input_tokens`.
121-
*/
122-
const totalTokens =
123-
(promptTokens ?? 0) + (completionTokens ?? 0) + (cachedInputTokens ?? 0) + (cachedOutputTokens ?? 0);
124-
125-
span.setAttributes({
126-
[GEN_AI_USAGE_TOTAL_TOKENS_ATTRIBUTE]: totalTokens,
127-
});
113+
if (typeof completionTokens === 'number') {
114+
attributes[GEN_AI_USAGE_OUTPUT_TOKENS_ATTRIBUTE] = completionTokens;
128115
}
116+
if (typeof totalTokens === 'number') {
117+
attributes[GEN_AI_USAGE_TOTAL_TOKENS_ATTRIBUTE] = totalTokens;
118+
} else if (typeof promptTokens === 'number' || typeof completionTokens === 'number') {
119+
attributes[GEN_AI_USAGE_TOTAL_TOKENS_ATTRIBUTE] =
120+
(typeof promptTokens === 'number' ? promptTokens : 0) +
121+
(typeof completionTokens === 'number' ? completionTokens : 0);
122+
}
123+
return attributes;
129124
}
130125

131126
export interface StreamResponseState {
@@ -139,6 +134,7 @@ export interface StreamResponseState {
139134
totalTokens?: number;
140135
cacheCreationInputTokens?: number;
141136
cacheReadInputTokens?: number;
137+
reasoningOutputTokens?: number;
142138
}
143139

144140
/**
@@ -157,23 +153,18 @@ export function endStreamSpan(span: Span, state: StreamResponseState, recordOutp
157153
if (state.responseId) attrs[GEN_AI_RESPONSE_ID_ATTRIBUTE] = state.responseId;
158154
if (state.responseModel) attrs[GEN_AI_RESPONSE_MODEL_ATTRIBUTE] = state.responseModel;
159155

160-
if (state.promptTokens !== undefined) attrs[GEN_AI_USAGE_INPUT_TOKENS_ATTRIBUTE] = state.promptTokens;
161-
if (state.completionTokens !== undefined) attrs[GEN_AI_USAGE_OUTPUT_TOKENS_ATTRIBUTE] = state.completionTokens;
162-
163-
// Use explicit total if provided (OpenAI, Google), otherwise compute from cache tokens (Anthropic)
164-
if (state.totalTokens !== undefined) {
165-
attrs[GEN_AI_USAGE_TOTAL_TOKENS_ATTRIBUTE] = state.totalTokens;
166-
} else if (
167-
state.promptTokens !== undefined ||
168-
state.completionTokens !== undefined ||
169-
state.cacheCreationInputTokens !== undefined ||
170-
state.cacheReadInputTokens !== undefined
171-
) {
172-
attrs[GEN_AI_USAGE_TOTAL_TOKENS_ATTRIBUTE] =
173-
(state.promptTokens ?? 0) +
174-
(state.completionTokens ?? 0) +
175-
(state.cacheCreationInputTokens ?? 0) +
176-
(state.cacheReadInputTokens ?? 0);
156+
Object.assign(
157+
attrs,
158+
getTokenUsageAttributes(
159+
state.promptTokens,
160+
state.completionTokens,
161+
state.cacheCreationInputTokens,
162+
state.cacheReadInputTokens,
163+
state.totalTokens,
164+
),
165+
);
166+
if (typeof state.reasoningOutputTokens === 'number') {
167+
attrs[GEN_AI_USAGE_REASONING_OUTPUT_TOKENS] = state.reasoningOutputTokens;
177168
}
178169

179170
if (state.finishReasons.length) {

‎packages/core/src/tracing/anthropic-ai/index.ts‎

Lines changed: 10 additions & 7 deletions
Original file line numberDiff line numberDiff line change
@@ -21,8 +21,8 @@ import {
2121
} from '../ai/gen-ai-attributes';
2222
import type { InstrumentedMethodEntry } from '../ai/utils';
2323
import {
24+
getTokenUsageAttributes,
2425
resolveAIRecordingOptions,
25-
setTokenUsageAttributes,
2626
shouldEnableTruncation,
2727
wrapPromiseWithMethods,
2828
} from '../ai/utils';
@@ -144,12 +144,15 @@ function addMetadataAttributes(span: Span, response: AnthropicAiResponse): void
144144
});
145145

146146
if ('usage' in response && response.usage) {
147-
setTokenUsageAttributes(
148-
span,
149-
response.usage.input_tokens,
150-
response.usage.output_tokens,
151-
response.usage.cache_creation_input_tokens,
152-
response.usage.cache_read_input_tokens,
147+
span.setAttributes(
148+
getTokenUsageAttributes(
149+
response.usage.input_tokens +
150+
(response.usage.cache_creation_input_tokens ?? 0) +
151+
(response.usage.cache_read_input_tokens ?? 0),
152+
response.usage.output_tokens,
153+
response.usage.cache_creation_input_tokens,
154+
response.usage.cache_read_input_tokens,
155+
),
153156
);
154157
}
155158
}

‎packages/core/src/tracing/anthropic-ai/streaming.ts‎

Lines changed: 6 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -84,7 +84,12 @@ function handleMessageMetadata(event: AnthropicAiStreamingEvent, state: Streamin
8484
if (message.model) state.responseModel = message.model;
8585

8686
if (message.usage) {
87-
if (typeof message.usage.input_tokens === 'number') state.promptTokens = message.usage.input_tokens;
87+
if (typeof message.usage.input_tokens === 'number') {
88+
state.promptTokens =
89+
message.usage.input_tokens +
90+
(message.usage.cache_creation_input_tokens ?? 0) +
91+
(message.usage.cache_read_input_tokens ?? 0);
92+
}
8893
if (typeof message.usage.cache_creation_input_tokens === 'number')
8994
state.cacheCreationInputTokens = message.usage.cache_creation_input_tokens;
9095
if (typeof message.usage.cache_read_input_tokens === 'number')

‎packages/core/src/tracing/anthropic-ai/types.ts‎

Lines changed: 2 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -55,8 +55,8 @@ type SuccessfulResponse = {
5555
usage?: {
5656
input_tokens: number;
5757
output_tokens: number;
58-
cache_creation_input_tokens: number;
59-
cache_read_input_tokens: number;
58+
cache_creation_input_tokens?: number;
59+
cache_read_input_tokens?: number;
6060
};
6161
error?: never; // This should help TypeScript infer the type correctly
6262
};

‎packages/core/src/tracing/google-genai/index.ts‎

Lines changed: 14 additions & 17 deletions
Original file line numberDiff line numberDiff line change
@@ -1,4 +1,5 @@
11
/* eslint-disable max-lines */
2+
import { GEN_AI_USAGE_REASONING_OUTPUT_TOKENS } from '@sentry/conventions/attributes';
23
import { SEMANTIC_ATTRIBUTE_SENTRY_ORIGIN } from '../../semanticAttributes';
34
import { SPAN_STATUS_ERROR } from '../../tracing';
45
import { startSpan, startSpanManual } from '../../tracing/trace';
@@ -22,14 +23,12 @@ import {
2223
GEN_AI_RESPONSE_TOOL_CALLS_ATTRIBUTE,
2324
GEN_AI_SYSTEM_ATTRIBUTE,
2425
GEN_AI_SYSTEM_INSTRUCTIONS_ATTRIBUTE,
25-
GEN_AI_USAGE_INPUT_TOKENS_ATTRIBUTE,
26-
GEN_AI_USAGE_OUTPUT_TOKENS_ATTRIBUTE,
27-
GEN_AI_USAGE_TOTAL_TOKENS_ATTRIBUTE,
2826
} from '../ai/gen-ai-attributes';
2927
import type { InstrumentedMethodEntry } from '../ai/utils';
3028
import { stringify } from '../../utils/string';
3129
import {
3230
buildMethodPath,
31+
getTokenUsageAttributes,
3332
extractSystemInstructions,
3433
getTruncatedJsonString,
3534
resolveAIRecordingOptions,
@@ -215,20 +214,18 @@ export function addResponseAttributes(span: Span, response: GoogleGenAIResponse,
215214
// Add usage metadata if present
216215
if (response.usageMetadata && typeof response.usageMetadata === 'object') {
217216
const usage = response.usageMetadata;
218-
if (typeof usage.promptTokenCount === 'number') {
219-
span.setAttributes({
220-
[GEN_AI_USAGE_INPUT_TOKENS_ATTRIBUTE]: usage.promptTokenCount,
221-
});
222-
}
223-
if (typeof usage.candidatesTokenCount === 'number') {
224-
span.setAttributes({
225-
[GEN_AI_USAGE_OUTPUT_TOKENS_ATTRIBUTE]: usage.candidatesTokenCount,
226-
});
227-
}
228-
if (typeof usage.totalTokenCount === 'number') {
229-
span.setAttributes({
230-
[GEN_AI_USAGE_TOTAL_TOKENS_ATTRIBUTE]: usage.totalTokenCount,
231-
});
217+
const hasOutput = typeof usage.candidatesTokenCount === 'number' || typeof usage.thoughtsTokenCount === 'number';
218+
span.setAttributes(
219+
getTokenUsageAttributes(
220+
usage.promptTokenCount,
221+
hasOutput ? (usage.candidatesTokenCount ?? 0) + (usage.thoughtsTokenCount ?? 0) : undefined,
222+
undefined,
223+
usage.cachedContentTokenCount,
224+
usage.totalTokenCount,
225+
),
226+
);
227+
if (typeof usage.thoughtsTokenCount === 'number') {
228+
span.setAttribute(GEN_AI_USAGE_REASONING_OUTPUT_TOKENS, usage.thoughtsTokenCount);
232229
}
233230
}
234231

‎packages/core/src/tracing/google-genai/streaming.ts‎

Lines changed: 9 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -20,8 +20,11 @@ interface StreamingState {
2020
promptTokens?: number;
2121
/** Number of completion/output tokens used. */
2222
completionTokens?: number;
23+
candidateTokens?: number;
24+
reasoningOutputTokens?: number;
2325
/** Number of total tokens used. */
2426
totalTokens?: number;
27+
cacheReadInputTokens?: number;
2528
/** Accumulated tool calls (finalized) */
2629
toolCalls: Array<Record<string, unknown>>;
2730
}
@@ -57,8 +60,13 @@ function handleResponseMetadata(chunk: GoogleGenAIResponse, state: StreamingStat
5760
const usage = chunk.usageMetadata;
5861
if (usage) {
5962
if (typeof usage.promptTokenCount === 'number') state.promptTokens = usage.promptTokenCount;
60-
if (typeof usage.candidatesTokenCount === 'number') state.completionTokens = usage.candidatesTokenCount;
63+
if (typeof usage.candidatesTokenCount === 'number') state.candidateTokens = usage.candidatesTokenCount;
64+
if (typeof usage.thoughtsTokenCount === 'number') state.reasoningOutputTokens = usage.thoughtsTokenCount;
65+
if (state.candidateTokens !== undefined || state.reasoningOutputTokens !== undefined) {
66+
state.completionTokens = (state.candidateTokens ?? 0) + (state.reasoningOutputTokens ?? 0);
67+
}
6168
if (typeof usage.totalTokenCount === 'number') state.totalTokens = usage.totalTokenCount;
69+
if (typeof usage.cachedContentTokenCount === 'number') state.cacheReadInputTokens = usage.cachedContentTokenCount;
6270
}
6371
}
6472

‎packages/core/src/tracing/langchain/types.ts‎

Lines changed: 15 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -59,12 +59,27 @@ export interface LangChainMessage {
5959
};
6060
role?: string;
6161
additional_kwargs?: Record<string, unknown>;
62+
usage_metadata?: {
63+
input_tokens?: number;
64+
output_tokens?: number;
65+
total_tokens?: number;
66+
input_token_details?: {
67+
cache_read?: number;
68+
cache_creation?: number;
69+
};
70+
};
6271
// LangChain serialized format
6372
lc?: number;
6473
id?: string[] | string;
6574
response_metadata?: {
75+
[key: string]: unknown;
6676
model_name?: string;
6777
finish_reason?: string;
78+
tokenUsage?: {
79+
promptTokens?: number;
80+
completionTokens?: number;
81+
totalTokens?: number;
82+
};
6883
};
6984
kwargs?: {
7085
[key: string]: unknown;

0 commit comments

Comments
 (0)