fix(ai): capture usage from choice.usage for non-standard OpenAI-compatible providers
Some OpenAI-compatible providers (e.g., Moonshot/Kimi) return usage data in chunk.choices[0].usage instead of the standard chunk.usage. Extract usage parsing into a helper and check choice.usage as fallback. closes #2017
This commit is contained in:
@@ -5,6 +5,7 @@
|
|||||||
### Fixed
|
### Fixed
|
||||||
|
|
||||||
- Fixed GitHub Copilot device-code login polling to respect OAuth slow-down intervals, wait before the first token poll, and include a clearer clock-drift hint in WSL/VM environments when repeated slow-downs lead to timeout.
|
- Fixed GitHub Copilot device-code login polling to respect OAuth slow-down intervals, wait before the first token poll, and include a clearer clock-drift hint in WSL/VM environments when repeated slow-downs lead to timeout.
|
||||||
|
- Fixed usage statistics not being captured for OpenAI-compatible providers that return usage in `choice.usage` instead of the standard `chunk.usage` (e.g., Moonshot/Kimi) ([#2017](https://github.com/badlogic/pi-mono/issues/2017))
|
||||||
|
|
||||||
## [0.57.1] - 2026-03-07
|
## [0.57.1] - 2026-03-07
|
||||||
|
|
||||||
|
|||||||
@@ -128,33 +128,18 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions", OpenA
|
|||||||
|
|
||||||
for await (const chunk of openaiStream) {
|
for await (const chunk of openaiStream) {
|
||||||
if (chunk.usage) {
|
if (chunk.usage) {
|
||||||
const cachedTokens = chunk.usage.prompt_tokens_details?.cached_tokens || 0;
|
output.usage = parseChunkUsage(chunk.usage, model);
|
||||||
const reasoningTokens = chunk.usage.completion_tokens_details?.reasoning_tokens || 0;
|
|
||||||
const input = (chunk.usage.prompt_tokens || 0) - cachedTokens;
|
|
||||||
const outputTokens = (chunk.usage.completion_tokens || 0) + reasoningTokens;
|
|
||||||
output.usage = {
|
|
||||||
// OpenAI includes cached tokens in prompt_tokens, so subtract to get non-cached input
|
|
||||||
input,
|
|
||||||
output: outputTokens,
|
|
||||||
cacheRead: cachedTokens,
|
|
||||||
cacheWrite: 0,
|
|
||||||
// Compute totalTokens ourselves since we add reasoning_tokens to output
|
|
||||||
// and some providers (e.g., Groq) don't include them in total_tokens
|
|
||||||
totalTokens: input + outputTokens + cachedTokens,
|
|
||||||
cost: {
|
|
||||||
input: 0,
|
|
||||||
output: 0,
|
|
||||||
cacheRead: 0,
|
|
||||||
cacheWrite: 0,
|
|
||||||
total: 0,
|
|
||||||
},
|
|
||||||
};
|
|
||||||
calculateCost(model, output.usage);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
const choice = chunk.choices?.[0];
|
const choice = chunk.choices?.[0];
|
||||||
if (!choice) continue;
|
if (!choice) continue;
|
||||||
|
|
||||||
|
// Fallback: some providers (e.g., Moonshot) return usage
|
||||||
|
// in choice.usage instead of the standard chunk.usage
|
||||||
|
if (!chunk.usage && (choice as any).usage) {
|
||||||
|
output.usage = parseChunkUsage((choice as any).usage, model);
|
||||||
|
}
|
||||||
|
|
||||||
if (choice.finish_reason) {
|
if (choice.finish_reason) {
|
||||||
output.stopReason = mapStopReason(choice.finish_reason);
|
output.stopReason = mapStopReason(choice.finish_reason);
|
||||||
}
|
}
|
||||||
@@ -717,6 +702,34 @@ function convertTools(
|
|||||||
}));
|
}));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
function parseChunkUsage(
|
||||||
|
rawUsage: {
|
||||||
|
prompt_tokens?: number;
|
||||||
|
completion_tokens?: number;
|
||||||
|
prompt_tokens_details?: { cached_tokens?: number };
|
||||||
|
completion_tokens_details?: { reasoning_tokens?: number };
|
||||||
|
},
|
||||||
|
model: Model<"openai-completions">,
|
||||||
|
): AssistantMessage["usage"] {
|
||||||
|
const cachedTokens = rawUsage.prompt_tokens_details?.cached_tokens || 0;
|
||||||
|
const reasoningTokens = rawUsage.completion_tokens_details?.reasoning_tokens || 0;
|
||||||
|
// OpenAI includes cached tokens in prompt_tokens, so subtract to get non-cached input
|
||||||
|
const input = (rawUsage.prompt_tokens || 0) - cachedTokens;
|
||||||
|
// Compute totalTokens ourselves since we add reasoning_tokens to output
|
||||||
|
// and some providers (e.g., Groq) don't include them in total_tokens
|
||||||
|
const outputTokens = (rawUsage.completion_tokens || 0) + reasoningTokens;
|
||||||
|
const usage: AssistantMessage["usage"] = {
|
||||||
|
input,
|
||||||
|
output: outputTokens,
|
||||||
|
cacheRead: cachedTokens,
|
||||||
|
cacheWrite: 0,
|
||||||
|
totalTokens: input + outputTokens + cachedTokens,
|
||||||
|
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||||
|
};
|
||||||
|
calculateCost(model, usage);
|
||||||
|
return usage;
|
||||||
|
}
|
||||||
|
|
||||||
function mapStopReason(reason: ChatCompletionChunk.Choice["finish_reason"]): StopReason {
|
function mapStopReason(reason: ChatCompletionChunk.Choice["finish_reason"]): StopReason {
|
||||||
if (reason === null) return "stop";
|
if (reason === null) return "stop";
|
||||||
switch (reason) {
|
switch (reason) {
|
||||||
|
|||||||
Reference in New Issue
Block a user