Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions .changeset/cache-write-tokens-docs.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,5 @@
---
'@core-ai/core-ai': patch
---

Clarify that `usage.inputTokenDetails.cacheWriteTokens` is populated when a provider reports cache writes, and is `0` otherwise.
5 changes: 5 additions & 0 deletions .changeset/openai-cache-write-tokens.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,5 @@
---
'@core-ai/openai': patch
---

Map `cache_write_tokens` into `usage.inputTokenDetails.cacheWriteTokens` for the Responses API and Chat Completions, including streaming usage chunks. `inputTokens` stays inclusive of cache reads and writes. Responses that omit the field still report `0`.
2 changes: 1 addition & 1 deletion packages/core-ai/src/types.ts
Original file line number Diff line number Diff line change
Expand Up @@ -377,7 +377,7 @@ export type ChatInputTokenDetails = {
cacheReadTokens: number;
/**
* Input tokens written to cache for future reuse. Subset of `inputTokens`.
* Only Anthropic reports this; other providers report `0`.
* Populated when the provider reports cache writes; otherwise `0`.
*/
cacheWriteTokens: number;
};
Expand Down
137 changes: 137 additions & 0 deletions packages/openai/src/chat-adapter.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -917,6 +917,63 @@ describe('mapGenerateResponse', () => {
});
});

it('should map cache write tokens from responses usage', () => {
const response = asResponse({
output: [
{
type: 'message',
role: 'assistant',
content: [{ type: 'output_text', text: 'Cached' }],
},
],
status: 'completed',
usage: {
input_tokens: 100,
output_tokens: 5,
input_tokens_details: {
cached_tokens: 40,
cache_write_tokens: 24,
},
output_tokens_details: { reasoning_tokens: 0 },
total_tokens: 105,
},
});

expect(mapGenerateResponse(response).usage).toEqual({
inputTokens: 100,
outputTokens: 5,
inputTokenDetails: {
cacheReadTokens: 40,
cacheWriteTokens: 24,
},
outputTokenDetails: {
reasoningTokens: 0,
},
});
});

it('should map a missing responses cache write field to zero', () => {
const response = asResponse({
output: [],
status: 'completed',
usage: {
input_tokens: 12,
output_tokens: 7,
input_tokens_details: { cached_tokens: 3 },
output_tokens_details: { reasoning_tokens: 2 },
total_tokens: 19,
},
});

const { usage } = mapGenerateResponse(response);

expect(usage.inputTokens).toBe(12);
expect(usage.inputTokenDetails).toEqual({
cacheReadTokens: 3,
cacheWriteTokens: 0,
});
});

it('should namespace encrypted reasoning under a wrapping provider id', () => {
const response = asResponse({
output: [
Expand Down Expand Up @@ -1263,6 +1320,86 @@ describe('transformStream', () => {
]);
});

it('should map cache write tokens from a completed response stream', async () => {
const stream = toAsyncIterable<ResponseStreamEvent>([
asStreamEvent({
type: 'response.completed',
response: asResponse({
output: [],
status: 'completed',
usage: {
input_tokens: 80,
output_tokens: 4,
input_tokens_details: {
cached_tokens: 20,
cache_write_tokens: 16,
},
output_tokens_details: { reasoning_tokens: 0 },
total_tokens: 84,
},
}),
}),
]);

const events = [];
for await (const event of transformStream(stream)) {
events.push(event);
}

expect(events).toEqual([
{
type: 'finish',
finishReason: 'stop',
usage: {
inputTokens: 80,
outputTokens: 4,
inputTokenDetails: {
cacheReadTokens: 20,
cacheWriteTokens: 16,
},
outputTokenDetails: {
reasoningTokens: 0,
},
},
},
]);
});

it('should map a missing streamed responses cache write field to zero', async () => {
const stream = toAsyncIterable<ResponseStreamEvent>([
asStreamEvent({
type: 'response.completed',
response: asResponse({
output: [],
status: 'completed',
usage: {
input_tokens: 4,
output_tokens: 2,
input_tokens_details: { cached_tokens: 1 },
output_tokens_details: { reasoning_tokens: 0 },
total_tokens: 6,
},
}),
}),
]);

const events = [];
for await (const event of transformStream(stream)) {
events.push(event);
}

expect(events.at(-1)).toMatchObject({
type: 'finish',
usage: {
inputTokens: 4,
inputTokenDetails: {
cacheReadTokens: 1,
cacheWriteTokens: 0,
},
},
});
});

it('should emit a single reasoning lifecycle across multiple summary parts', async () => {
const stream = toAsyncIterable<ResponseStreamEvent>([
asStreamEvent({
Expand Down
3 changes: 2 additions & 1 deletion packages/openai/src/chat-adapter.ts
Original file line number Diff line number Diff line change
Expand Up @@ -625,7 +625,8 @@ function mapUsage(usage: ResponseUsage | undefined): GenerateResult['usage'] {
outputTokens: usage?.output_tokens ?? 0,
inputTokenDetails: {
cacheReadTokens: usage?.input_tokens_details?.cached_tokens ?? 0,
cacheWriteTokens: 0,
cacheWriteTokens:
usage?.input_tokens_details?.cache_write_tokens ?? 0,
},
outputTokenDetails: {
...(reasoningTokens !== undefined ? { reasoningTokens } : {}),
Expand Down
132 changes: 132 additions & 0 deletions packages/openai/src/chat-completions/chat-adapter.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -757,6 +757,138 @@ describe('reasoning token accounting', () => {
});
});

describe('cache write tokens', () => {
type PromptTokenDetails = NonNullable<
NonNullable<ChatCompletion['usage']>['prompt_tokens_details']
>;

function createResponse(promptTokensDetails: PromptTokenDetails) {
return asChatCompletion({
choices: [
{
index: 0,
finish_reason: 'stop',
logprobs: null,
message: {
role: 'assistant',
content: 'answer',
refusal: null,
},
},
],
usage: {
prompt_tokens: 100,
completion_tokens: 8,
total_tokens: 108,
prompt_tokens_details: promptTokensDetails,
},
});
}

it('should map cache write tokens from prompt token details', () => {
const result = mapGenerateResponse(
createResponse({
cached_tokens: 40,
cache_write_tokens: 24,
})
);

expect(result.usage).toEqual({
inputTokens: 100,
outputTokens: 8,
inputTokenDetails: {
cacheReadTokens: 40,
cacheWriteTokens: 24,
},
outputTokenDetails: {},
});
});

it('should map a missing prompt cache write field to zero', () => {
const result = mapGenerateResponse(
createResponse({ cached_tokens: 40 })
);

expect(result.usage.inputTokens).toBe(100);
expect(result.usage.inputTokenDetails).toEqual({
cacheReadTokens: 40,
cacheWriteTokens: 0,
});
});

it('should map cache write tokens from the streaming usage chunk', async () => {
const events = await collectEvents(
transformStream(
toAsyncIterable([
asChunk({
choices: [
{
index: 0,
finish_reason: 'stop',
delta: { content: 'answer' },
},
],
usage: {
prompt_tokens: 90,
completion_tokens: 4,
total_tokens: 94,
prompt_tokens_details: {
cached_tokens: 30,
cache_write_tokens: 12,
},
},
}),
])
)
);

expect(events.at(-1)).toEqual({
type: 'finish',
finishReason: 'stop',
usage: {
inputTokens: 90,
outputTokens: 4,
inputTokenDetails: {
cacheReadTokens: 30,
cacheWriteTokens: 12,
},
outputTokenDetails: {},
},
});
});

it('should map a missing streaming cache write field to zero', async () => {
const events = await collectEvents(
transformStream(
toAsyncIterable([
asChunk({
choices: [],
usage: {
prompt_tokens: 90,
completion_tokens: 4,
total_tokens: 94,
prompt_tokens_details: {
cached_tokens: 30,
},
},
}),
])
)
);

expect(events.at(-1)).toMatchObject({
type: 'finish',
usage: {
inputTokens: 90,
inputTokenDetails: {
cacheReadTokens: 30,
cacheWriteTokens: 0,
},
},
});
});
});

describe('reasoning support', () => {
it('should fold reasoning parts into text content wrapped in <thinking> tags', () => {
const messages: Message[] = [
Expand Down
8 changes: 6 additions & 2 deletions packages/openai/src/chat-completions/chat-adapter.ts
Original file line number Diff line number Diff line change
Expand Up @@ -455,7 +455,9 @@ export function mapGenerateResponse(
inputTokenDetails: {
cacheReadTokens:
response.usage?.prompt_tokens_details?.cached_tokens ?? 0,
cacheWriteTokens: 0,
cacheWriteTokens:
response.usage?.prompt_tokens_details?.cache_write_tokens ??
0,
},
outputTokenDetails: {
...(reasoningTokens !== undefined ? { reasoningTokens } : {}),
Expand Down Expand Up @@ -590,7 +592,9 @@ export async function* transformStream(
inputTokenDetails: {
cacheReadTokens:
chunk.usage.prompt_tokens_details?.cached_tokens ?? 0,
cacheWriteTokens: 0,
cacheWriteTokens:
chunk.usage.prompt_tokens_details?.cache_write_tokens ??
0,
},
outputTokenDetails: {
...(reasoningTokens !== undefined
Expand Down
Loading