diff --git a/.changeset/cache-write-tokens-docs.md b/.changeset/cache-write-tokens-docs.md new file mode 100644 index 0000000..851e98b --- /dev/null +++ b/.changeset/cache-write-tokens-docs.md @@ -0,0 +1,5 @@ +--- +'@core-ai/core-ai': patch +--- + +Clarify that `usage.inputTokenDetails.cacheWriteTokens` is populated when a provider reports cache writes, and is `0` otherwise. diff --git a/.changeset/openai-cache-write-tokens.md b/.changeset/openai-cache-write-tokens.md new file mode 100644 index 0000000..069d7b4 --- /dev/null +++ b/.changeset/openai-cache-write-tokens.md @@ -0,0 +1,5 @@ +--- +'@core-ai/openai': patch +--- + +Map `cache_write_tokens` into `usage.inputTokenDetails.cacheWriteTokens` for the Responses API and Chat Completions, including streaming usage chunks. `inputTokens` stays inclusive of cache reads and writes. Responses that omit the field still report `0`. diff --git a/packages/core-ai/src/types.ts b/packages/core-ai/src/types.ts index 7c0f94f..1799a4d 100644 --- a/packages/core-ai/src/types.ts +++ b/packages/core-ai/src/types.ts @@ -377,7 +377,7 @@ export type ChatInputTokenDetails = { cacheReadTokens: number; /** * Input tokens written to cache for future reuse. Subset of `inputTokens`. - * Only Anthropic reports this; other providers report `0`. + * Populated when the provider reports cache writes; otherwise `0`. */ cacheWriteTokens: number; }; diff --git a/packages/openai/src/chat-adapter.test.ts b/packages/openai/src/chat-adapter.test.ts index 059ca52..ada23d9 100644 --- a/packages/openai/src/chat-adapter.test.ts +++ b/packages/openai/src/chat-adapter.test.ts @@ -917,6 +917,63 @@ describe('mapGenerateResponse', () => { }); }); + it('should map cache write tokens from responses usage', () => { + const response = asResponse({ + output: [ + { + type: 'message', + role: 'assistant', + content: [{ type: 'output_text', text: 'Cached' }], + }, + ], + status: 'completed', + usage: { + input_tokens: 100, + output_tokens: 5, + input_tokens_details: { + cached_tokens: 40, + cache_write_tokens: 24, + }, + output_tokens_details: { reasoning_tokens: 0 }, + total_tokens: 105, + }, + }); + + expect(mapGenerateResponse(response).usage).toEqual({ + inputTokens: 100, + outputTokens: 5, + inputTokenDetails: { + cacheReadTokens: 40, + cacheWriteTokens: 24, + }, + outputTokenDetails: { + reasoningTokens: 0, + }, + }); + }); + + it('should map a missing responses cache write field to zero', () => { + const response = asResponse({ + output: [], + status: 'completed', + usage: { + input_tokens: 12, + output_tokens: 7, + input_tokens_details: { cached_tokens: 3 }, + output_tokens_details: { reasoning_tokens: 2 }, + total_tokens: 19, + }, + }); + + const { usage } = mapGenerateResponse(response); + + expect(usage.inputTokens).toBe(12); + expect(usage.inputTokenDetails).toEqual({ + cacheReadTokens: 3, + cacheWriteTokens: 0, + }); + }); + it('should namespace encrypted reasoning under a wrapping provider id', () => { const response = asResponse({ output: [ @@ -1263,6 +1320,86 @@ describe('transformStream', () => { ]); }); + it('should map cache write tokens from a completed response stream', async () => { + const stream = toAsyncIterable([ + asStreamEvent({ + type: 'response.completed', + response: asResponse({ + output: [], + status: 'completed', + usage: { + input_tokens: 80, + output_tokens: 4, + input_tokens_details: { + cached_tokens: 20, + cache_write_tokens: 16, + }, + output_tokens_details: { reasoning_tokens: 0 }, + total_tokens: 84, + }, + }), + }), + ]); + + const events = []; + for await (const event of transformStream(stream)) { + events.push(event); + } + + expect(events).toEqual([ + { + type: 'finish', + finishReason: 'stop', + usage: { + inputTokens: 80, + outputTokens: 4, + inputTokenDetails: { + cacheReadTokens: 20, + cacheWriteTokens: 16, + }, + outputTokenDetails: { + reasoningTokens: 0, + }, + }, + }, + ]); + }); + + it('should map a missing streamed responses cache write field to zero', async () => { + const stream = toAsyncIterable([ + asStreamEvent({ + type: 'response.completed', + response: asResponse({ + output: [], + status: 'completed', + usage: { + input_tokens: 4, + output_tokens: 2, + input_tokens_details: { cached_tokens: 1 }, + output_tokens_details: { reasoning_tokens: 0 }, + total_tokens: 6, + }, + }), + }), + ]); + + const events = []; + for await (const event of transformStream(stream)) { + events.push(event); + } + + expect(events.at(-1)).toMatchObject({ + type: 'finish', + usage: { + inputTokens: 4, + inputTokenDetails: { + cacheReadTokens: 1, + cacheWriteTokens: 0, + }, + }, + }); + }); + it('should emit a single reasoning lifecycle across multiple summary parts', async () => { const stream = toAsyncIterable([ asStreamEvent({ diff --git a/packages/openai/src/chat-adapter.ts b/packages/openai/src/chat-adapter.ts index 422ad67..b406148 100644 --- a/packages/openai/src/chat-adapter.ts +++ b/packages/openai/src/chat-adapter.ts @@ -625,7 +625,8 @@ function mapUsage(usage: ResponseUsage | undefined): GenerateResult['usage'] { outputTokens: usage?.output_tokens ?? 0, inputTokenDetails: { cacheReadTokens: usage?.input_tokens_details?.cached_tokens ?? 0, - cacheWriteTokens: 0, + cacheWriteTokens: + usage?.input_tokens_details?.cache_write_tokens ?? 0, }, outputTokenDetails: { ...(reasoningTokens !== undefined ? { reasoningTokens } : {}), diff --git a/packages/openai/src/chat-completions/chat-adapter.test.ts b/packages/openai/src/chat-completions/chat-adapter.test.ts index 3d4c944..c6ab97a 100644 --- a/packages/openai/src/chat-completions/chat-adapter.test.ts +++ b/packages/openai/src/chat-completions/chat-adapter.test.ts @@ -757,6 +757,138 @@ describe('reasoning token accounting', () => { }); }); +describe('cache write tokens', () => { + type PromptTokenDetails = NonNullable< + NonNullable['prompt_tokens_details'] + >; + + function createResponse(promptTokensDetails: PromptTokenDetails) { + return asChatCompletion({ + choices: [ + { + index: 0, + finish_reason: 'stop', + logprobs: null, + message: { + role: 'assistant', + content: 'answer', + refusal: null, + }, + }, + ], + usage: { + prompt_tokens: 100, + completion_tokens: 8, + total_tokens: 108, + prompt_tokens_details: promptTokensDetails, + }, + }); + } + + it('should map cache write tokens from prompt token details', () => { + const result = mapGenerateResponse( + createResponse({ + cached_tokens: 40, + cache_write_tokens: 24, + }) + ); + + expect(result.usage).toEqual({ + inputTokens: 100, + outputTokens: 8, + inputTokenDetails: { + cacheReadTokens: 40, + cacheWriteTokens: 24, + }, + outputTokenDetails: {}, + }); + }); + + it('should map a missing prompt cache write field to zero', () => { + const result = mapGenerateResponse( + createResponse({ cached_tokens: 40 }) + ); + + expect(result.usage.inputTokens).toBe(100); + expect(result.usage.inputTokenDetails).toEqual({ + cacheReadTokens: 40, + cacheWriteTokens: 0, + }); + }); + + it('should map cache write tokens from the streaming usage chunk', async () => { + const events = await collectEvents( + transformStream( + toAsyncIterable([ + asChunk({ + choices: [ + { + index: 0, + finish_reason: 'stop', + delta: { content: 'answer' }, + }, + ], + usage: { + prompt_tokens: 90, + completion_tokens: 4, + total_tokens: 94, + prompt_tokens_details: { + cached_tokens: 30, + cache_write_tokens: 12, + }, + }, + }), + ]) + ) + ); + + expect(events.at(-1)).toEqual({ + type: 'finish', + finishReason: 'stop', + usage: { + inputTokens: 90, + outputTokens: 4, + inputTokenDetails: { + cacheReadTokens: 30, + cacheWriteTokens: 12, + }, + outputTokenDetails: {}, + }, + }); + }); + + it('should map a missing streaming cache write field to zero', async () => { + const events = await collectEvents( + transformStream( + toAsyncIterable([ + asChunk({ + choices: [], + usage: { + prompt_tokens: 90, + completion_tokens: 4, + total_tokens: 94, + prompt_tokens_details: { + cached_tokens: 30, + }, + }, + }), + ]) + ) + ); + + expect(events.at(-1)).toMatchObject({ + type: 'finish', + usage: { + inputTokens: 90, + inputTokenDetails: { + cacheReadTokens: 30, + cacheWriteTokens: 0, + }, + }, + }); + }); +}); + describe('reasoning support', () => { it('should fold reasoning parts into text content wrapped in tags', () => { const messages: Message[] = [ diff --git a/packages/openai/src/chat-completions/chat-adapter.ts b/packages/openai/src/chat-completions/chat-adapter.ts index 30f2dbd..3df0b95 100644 --- a/packages/openai/src/chat-completions/chat-adapter.ts +++ b/packages/openai/src/chat-completions/chat-adapter.ts @@ -455,7 +455,9 @@ export function mapGenerateResponse( inputTokenDetails: { cacheReadTokens: response.usage?.prompt_tokens_details?.cached_tokens ?? 0, - cacheWriteTokens: 0, + cacheWriteTokens: + response.usage?.prompt_tokens_details?.cache_write_tokens ?? + 0, }, outputTokenDetails: { ...(reasoningTokens !== undefined ? { reasoningTokens } : {}), @@ -590,7 +592,9 @@ export async function* transformStream( inputTokenDetails: { cacheReadTokens: chunk.usage.prompt_tokens_details?.cached_tokens ?? 0, - cacheWriteTokens: 0, + cacheWriteTokens: + chunk.usage.prompt_tokens_details?.cache_write_tokens ?? + 0, }, outputTokenDetails: { ...(reasoningTokens !== undefined