fix(console): openai usage normalization and tier threshold config (#47342)

Co-authored-by: Jack <jack@anoma.ly>
This commit is contained in:
黑墨水鱼 2026-09-05 11:36:32 +08:00 committed by GitHub
parent 70b4ca8c18
commit e2894562f8
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
5 changed files with 81 additions and 6 deletions

View file

@ -1023,7 +1023,8 @@ export async function handler(
modelInfo.costPeak && isPeakPricing(new Date())
? modelInfo.costPeak
: modelInfo.cost200K &&
inputTokens + (cacheReadTokens ?? 0) + (cacheWrite5mTokens ?? 0) + (cacheWrite1hTokens ?? 0) > 200_000
inputTokens + (cacheReadTokens ?? 0) + (cacheWrite5mTokens ?? 0) + (cacheWrite1hTokens ?? 0) >
modelInfo.cost200K.threshold
? modelInfo.cost200K
: modelInfo.cost

View file

@ -51,7 +51,10 @@ export const openaiHelper: ProviderHelper = ({ workspaceID }) => ({
const cacheReadTokens = usage.input_tokens_details?.cached_tokens ?? undefined
const cacheWriteTokens = usage.input_tokens_details?.cache_write_tokens ?? undefined
return {
inputTokens: inputTokens - (cacheReadTokens ?? 0),
// OpenAI's input_tokens includes both cached_tokens and cache_write_tokens;
// each is billed separately at its own rate. Clamp to zero so a provider
// reporting overlapping detail fields cannot drive input cost negative.
inputTokens: Math.max(0, inputTokens - (cacheReadTokens ?? 0) - (cacheWriteTokens ?? 0)),
outputTokens,
reasoningTokens,
cacheReadTokens,

View file

@ -26,7 +26,7 @@ describe("provider usage extraction", () => {
expect(providers.google.normalizeUsage(usage)).toEqual({
inputTokens: 6,
outputTokens: 3,
outputTokens: 5,
reasoningTokens: 2,
cacheReadTokens: 4,
cacheWrite5mTokens: undefined,
@ -42,7 +42,7 @@ describe("provider usage extraction", () => {
expect(providers.google.normalizeUsage(usageParser.retrieve())).toEqual({
inputTokens: 6,
outputTokens: 3,
outputTokens: 5,
reasoningTokens: 2,
cacheReadTokens: 4,
cacheWrite5mTokens: undefined,
@ -73,7 +73,23 @@ describe("provider usage extraction", () => {
)
expect(providers.openai.normalizeUsage(usageParser.retrieve())).toEqual({
inputTokens: 6,
inputTokens: 3,
outputTokens: 2,
reasoningTokens: undefined,
cacheReadTokens: 4,
cacheWrite5mTokens: 3,
cacheWrite1hTokens: undefined,
})
})
test("clamps input tokens when detail fields overlap", () => {
const usageParser = providers.openai.createUsageParser()
usageParser.parse(
'event: response.completed\ndata: {"response":{"usage":{"input_tokens":5,"input_tokens_details":{"cached_tokens":4,"cache_write_tokens":3},"output_tokens":2}}}',
)
expect(providers.openai.normalizeUsage(usageParser.retrieve())).toEqual({
inputTokens: 0,
outputTokens: 2,
reasoningTokens: undefined,
cacheReadTokens: 4,

View file

@ -19,11 +19,17 @@ export namespace ZenData {
cacheWrite1h: z.number().optional(),
})
// Long-context tier. The flip threshold defaults to 200_000 for backward
// compatibility with existing ZEN_MODELS secrets.
const ModelCostTierSchema = ModelCostSchema.extend({
threshold: z.number().default(200_000),
})
const ModelSchema = z.object({
name: z.string(),
cost: ModelCostSchema,
costMultiplier: z.number().default(1),
cost200K: ModelCostSchema.optional(),
cost200K: ModelCostTierSchema.optional(),
costPeak: ModelCostSchema.optional(),
allowAnonymous: z.boolean().optional(),
byokProvider: z.enum(["openai", "anthropic", "google"]).optional(),

View file

@ -0,0 +1,49 @@
import { describe, expect, test } from "bun:test"
import { ZenData } from "../src/model"
const base = {
zenModels: {
"gpt-5.6-sol": {
name: "GPT-5.6 Sol",
cost: { input: 2, output: 10 },
costMultiplier: 1,
providers: [{ id: "openai", model: "gpt-5.6-sol" }],
},
},
liteModels: {},
providers: { openai: { api: "https://api.openai.com/v1", apiKey: "test" } },
}
const entry = (data: ReturnType<typeof ZenData.validate>) => {
const value = data.zenModels["gpt-5.6-sol"]
return Array.isArray(value) ? value[0] : value
}
describe("ZenData cost200K threshold", () => {
test("defaults to 200_000 when not configured", () => {
const data = ZenData.validate({
...base,
zenModels: {
"gpt-5.6-sol": {
...base.zenModels["gpt-5.6-sol"],
// The secret arrives as parsed JSON in production, so untyped here too.
cost200K: JSON.parse('{"input":4,"output":15}'),
},
},
})
expect(entry(data).cost200K?.threshold).toBe(200_000)
})
test("accepts an explicit 272_000 threshold", () => {
const data = ZenData.validate({
...base,
zenModels: {
"gpt-5.6-sol": {
...base.zenModels["gpt-5.6-sol"],
cost200K: { input: 4, output: 15, threshold: 272_000 },
},
},
})
expect(entry(data).cost200K?.threshold).toBe(272_000)
})
})