fix(models): refresh model pricing

This commit is contained in:
Patrick Wozniak
2026-08-04 13:46:26 +02:00
parent 690c640d5b
commit 947915df55
10 changed files with 619 additions and 127 deletions
+83 -6
View File
@@ -13,13 +13,17 @@ import { describe, it } from "node:test"
import { calculateCommandCodeCost } from "../src/cost.ts"
import type { Usage } from "../src/types.ts"
interface CostTable {
interface CostRates {
input: number
output: number
cacheRead: number
cacheWrite: number
}
interface CostTable extends CostRates {
tiers?: Array<CostRates & { inputTokensAbove: number }>
}
const COST_FIXTURES: Record<string, CostTable> = {
"zero-cost-model": { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
"claude-sonnet-4-6": { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 3.75 },
@@ -29,7 +33,17 @@ const COST_FIXTURES: Record<string, CostTable> = {
cacheRead: 0.003625,
cacheWrite: 0,
},
"Qwen/Qwen3.7-Max": { input: 1.25, output: 3.75, cacheRead: 0.25, cacheWrite: 1.56 },
"Qwen/Qwen3.7-Max": { input: 2.5, output: 7.5, cacheRead: 0.5, cacheWrite: 3.13 },
"Qwen/Qwen3.7-Flash": {
input: 0.03,
output: 0.13,
cacheRead: 0.006,
cacheWrite: 0.038,
tiers: [
{ inputTokensAbove: 32_000, input: 0.1, output: 0.4, cacheRead: 0.02, cacheWrite: 0.125 },
{ inputTokensAbove: 256_000, input: 0.2, output: 0.8, cacheRead: 0.04, cacheWrite: 0.25 },
],
},
"gpt-5.5": { input: 5, output: 30, cacheRead: 0.5, cacheWrite: 0 },
}
@@ -51,6 +65,14 @@ function commandCodeModel(id: string, cost: CostTable) {
}
}
function assertClose(actual: number, expected: number) {
assert.ok(
Math.abs(actual - expected) <=
Number.EPSILON * Math.max(1, Math.abs(actual), Math.abs(expected)),
`expected ${actual} to be close to ${expected}`,
)
}
function freshUsage(tokens: (typeof USAGE_CASES)[number]): Usage {
return {
...tokens,
@@ -60,10 +82,20 @@ function freshUsage(tokens: (typeof USAGE_CASES)[number]): Usage {
}
function expectedCost(cost: CostTable, tokens: (typeof USAGE_CASES)[number]): Usage["cost"] {
const input = (cost.input / 1_000_000) * tokens.input
const output = (cost.output / 1_000_000) * tokens.output
const cacheRead = (cost.cacheRead / 1_000_000) * tokens.cacheRead
const cacheWrite = (cost.cacheWrite * tokens.cacheWrite) / 1_000_000
const inputTokens = tokens.input + tokens.cacheRead + tokens.cacheWrite
let rates: CostRates = cost
let matchedThreshold = -1
for (const tier of cost.tiers ?? []) {
if (inputTokens > tier.inputTokensAbove && tier.inputTokensAbove > matchedThreshold) {
rates = tier
matchedThreshold = tier.inputTokensAbove
}
}
const input = (rates.input / 1_000_000) * tokens.input
const output = (rates.output / 1_000_000) * tokens.output
const cacheRead = (rates.cacheRead / 1_000_000) * tokens.cacheRead
const cacheWrite = (rates.cacheWrite * tokens.cacheWrite) / 1_000_000
return {
input,
output,
@@ -91,6 +123,51 @@ describe("calculateCommandCodeCost()", () => {
}
})
it("applies the highest request-wide input tier above its threshold", () => {
const model = commandCodeModel("Qwen/Qwen3.7-Flash", COST_FIXTURES["Qwen/Qwen3.7-Flash"])
const atThreshold = freshUsage({
input: 32_000,
output: 1_000,
cacheRead: 0,
cacheWrite: 0,
})
calculateCommandCodeCost(model, atThreshold)
assertClose(atThreshold.cost.input, (0.03 * 32_000) / 1_000_000)
const aboveFirstTier = freshUsage({
input: 30_000,
output: 1_000,
cacheRead: 2_001,
cacheWrite: 0,
})
calculateCommandCodeCost(model, aboveFirstTier)
assertClose(aboveFirstTier.cost.input, (0.1 * 30_000) / 1_000_000)
assertClose(aboveFirstTier.cost.cacheRead, (0.02 * 2_001) / 1_000_000)
const aboveHighestTier = freshUsage({
input: 100_000,
output: 1_000,
cacheRead: 156_001,
cacheWrite: 0,
})
calculateCommandCodeCost(model, aboveHighestTier)
assertClose(aboveHighestTier.cost.input, (0.2 * 100_000) / 1_000_000)
assertClose(aboveHighestTier.cost.output, (0.8 * 1_000) / 1_000_000)
})
it("prices one-hour cache writes at twice the active input rate", () => {
const model = commandCodeModel("claude-sonnet-4-6", COST_FIXTURES["claude-sonnet-4-6"])
const usage = freshUsage({ input: 0, output: 0, cacheRead: 0, cacheWrite: 1_000 })
usage.cacheWrite1h = 400
calculateCommandCodeCost(model, usage)
const expectedShortWrite = (3.75 * 600) / 1_000_000
const expectedLongWrite = (3 * 2 * 400) / 1_000_000
assertClose(usage.cost.cacheWrite, expectedShortWrite + expectedLongWrite)
})
it("writes the total as the sum of all cost components", () => {
const model = commandCodeModel("claude-sonnet-4-6", COST_FIXTURES["claude-sonnet-4-6"])
const usage = freshUsage({ input: 1_000, output: 500, cacheRead: 10_000, cacheWrite: 2_000 })