fix(console): openai usage normalization and tier threshold config (#47342)
Co-authored-by: Jack <jack@anoma.ly>
This commit is contained in:
parent
70b4ca8c18
commit
e2894562f8
5 changed files with 81 additions and 6 deletions
|
|
@ -1023,7 +1023,8 @@ export async function handler(
|
|||
modelInfo.costPeak && isPeakPricing(new Date())
|
||||
? modelInfo.costPeak
|
||||
: modelInfo.cost200K &&
|
||||
inputTokens + (cacheReadTokens ?? 0) + (cacheWrite5mTokens ?? 0) + (cacheWrite1hTokens ?? 0) > 200_000
|
||||
inputTokens + (cacheReadTokens ?? 0) + (cacheWrite5mTokens ?? 0) + (cacheWrite1hTokens ?? 0) >
|
||||
modelInfo.cost200K.threshold
|
||||
? modelInfo.cost200K
|
||||
: modelInfo.cost
|
||||
|
||||
|
|
|
|||
|
|
@ -51,7 +51,10 @@ export const openaiHelper: ProviderHelper = ({ workspaceID }) => ({
|
|||
const cacheReadTokens = usage.input_tokens_details?.cached_tokens ?? undefined
|
||||
const cacheWriteTokens = usage.input_tokens_details?.cache_write_tokens ?? undefined
|
||||
return {
|
||||
inputTokens: inputTokens - (cacheReadTokens ?? 0),
|
||||
// OpenAI's input_tokens includes both cached_tokens and cache_write_tokens;
|
||||
// each is billed separately at its own rate. Clamp to zero so a provider
|
||||
// reporting overlapping detail fields cannot drive input cost negative.
|
||||
inputTokens: Math.max(0, inputTokens - (cacheReadTokens ?? 0) - (cacheWriteTokens ?? 0)),
|
||||
outputTokens,
|
||||
reasoningTokens,
|
||||
cacheReadTokens,
|
||||
|
|
|
|||
|
|
@ -26,7 +26,7 @@ describe("provider usage extraction", () => {
|
|||
|
||||
expect(providers.google.normalizeUsage(usage)).toEqual({
|
||||
inputTokens: 6,
|
||||
outputTokens: 3,
|
||||
outputTokens: 5,
|
||||
reasoningTokens: 2,
|
||||
cacheReadTokens: 4,
|
||||
cacheWrite5mTokens: undefined,
|
||||
|
|
@ -42,7 +42,7 @@ describe("provider usage extraction", () => {
|
|||
|
||||
expect(providers.google.normalizeUsage(usageParser.retrieve())).toEqual({
|
||||
inputTokens: 6,
|
||||
outputTokens: 3,
|
||||
outputTokens: 5,
|
||||
reasoningTokens: 2,
|
||||
cacheReadTokens: 4,
|
||||
cacheWrite5mTokens: undefined,
|
||||
|
|
@ -73,7 +73,23 @@ describe("provider usage extraction", () => {
|
|||
)
|
||||
|
||||
expect(providers.openai.normalizeUsage(usageParser.retrieve())).toEqual({
|
||||
inputTokens: 6,
|
||||
inputTokens: 3,
|
||||
outputTokens: 2,
|
||||
reasoningTokens: undefined,
|
||||
cacheReadTokens: 4,
|
||||
cacheWrite5mTokens: 3,
|
||||
cacheWrite1hTokens: undefined,
|
||||
})
|
||||
})
|
||||
|
||||
test("clamps input tokens when detail fields overlap", () => {
|
||||
const usageParser = providers.openai.createUsageParser()
|
||||
usageParser.parse(
|
||||
'event: response.completed\ndata: {"response":{"usage":{"input_tokens":5,"input_tokens_details":{"cached_tokens":4,"cache_write_tokens":3},"output_tokens":2}}}',
|
||||
)
|
||||
|
||||
expect(providers.openai.normalizeUsage(usageParser.retrieve())).toEqual({
|
||||
inputTokens: 0,
|
||||
outputTokens: 2,
|
||||
reasoningTokens: undefined,
|
||||
cacheReadTokens: 4,
|
||||
|
|
|
|||
|
|
@ -19,11 +19,17 @@ export namespace ZenData {
|
|||
cacheWrite1h: z.number().optional(),
|
||||
})
|
||||
|
||||
// Long-context tier. The flip threshold defaults to 200_000 for backward
|
||||
// compatibility with existing ZEN_MODELS secrets.
|
||||
const ModelCostTierSchema = ModelCostSchema.extend({
|
||||
threshold: z.number().default(200_000),
|
||||
})
|
||||
|
||||
const ModelSchema = z.object({
|
||||
name: z.string(),
|
||||
cost: ModelCostSchema,
|
||||
costMultiplier: z.number().default(1),
|
||||
cost200K: ModelCostSchema.optional(),
|
||||
cost200K: ModelCostTierSchema.optional(),
|
||||
costPeak: ModelCostSchema.optional(),
|
||||
allowAnonymous: z.boolean().optional(),
|
||||
byokProvider: z.enum(["openai", "anthropic", "google"]).optional(),
|
||||
|
|
|
|||
49
packages/console/core/test/model.test.ts
Normal file
49
packages/console/core/test/model.test.ts
Normal file
|
|
@ -0,0 +1,49 @@
|
|||
import { describe, expect, test } from "bun:test"
|
||||
import { ZenData } from "../src/model"
|
||||
|
||||
const base = {
|
||||
zenModels: {
|
||||
"gpt-5.6-sol": {
|
||||
name: "GPT-5.6 Sol",
|
||||
cost: { input: 2, output: 10 },
|
||||
costMultiplier: 1,
|
||||
providers: [{ id: "openai", model: "gpt-5.6-sol" }],
|
||||
},
|
||||
},
|
||||
liteModels: {},
|
||||
providers: { openai: { api: "https://api.openai.com/v1", apiKey: "test" } },
|
||||
}
|
||||
|
||||
const entry = (data: ReturnType<typeof ZenData.validate>) => {
|
||||
const value = data.zenModels["gpt-5.6-sol"]
|
||||
return Array.isArray(value) ? value[0] : value
|
||||
}
|
||||
|
||||
describe("ZenData cost200K threshold", () => {
|
||||
test("defaults to 200_000 when not configured", () => {
|
||||
const data = ZenData.validate({
|
||||
...base,
|
||||
zenModels: {
|
||||
"gpt-5.6-sol": {
|
||||
...base.zenModels["gpt-5.6-sol"],
|
||||
// The secret arrives as parsed JSON in production, so untyped here too.
|
||||
cost200K: JSON.parse('{"input":4,"output":15}'),
|
||||
},
|
||||
},
|
||||
})
|
||||
expect(entry(data).cost200K?.threshold).toBe(200_000)
|
||||
})
|
||||
|
||||
test("accepts an explicit 272_000 threshold", () => {
|
||||
const data = ZenData.validate({
|
||||
...base,
|
||||
zenModels: {
|
||||
"gpt-5.6-sol": {
|
||||
...base.zenModels["gpt-5.6-sol"],
|
||||
cost200K: { input: 4, output: 15, threshold: 272_000 },
|
||||
},
|
||||
},
|
||||
})
|
||||
expect(entry(data).cost200K?.threshold).toBe(272_000)
|
||||
})
|
||||
})
|
||||
Loading…
Reference in a new issue