Merge pull request #65 from ThomasByr/main

feat(models): add GLM5.3-Flash and Qwen3.8-Flash

# Conflicts:
#	CHANGELOG.md
This commit is contained in:
Patrick Wozniak
2026-09-01 23:27:37 +02:00
8 changed files with 86 additions and 21 deletions
+6 -2
View File
@@ -1,5 +1,5 @@
{
"fetchedAt": "2026-08-25T13:32:11.631Z",
"fetchedAt": "2026-08-28T09:27:52.554Z",
"source": "https://api.commandcode.ai/provider/v1/models",
"modelIds": [
"claude-sonnet-5",
@@ -24,6 +24,7 @@
"moonshotai/Kimi-K2.7-Code-Highspeed",
"moonshotai/Kimi-K2.6",
"moonshotai/Kimi-K2.5",
"z-ai/glm-5.3-flash",
"zai-org/GLM-5.3",
"zai-org/GLM-5.2",
"zai-org/GLM-5.2-Fast",
@@ -31,11 +32,14 @@
"zai-org/GLM-5",
"MiniMaxAI/MiniMax-M3",
"MiniMaxAI/MiniMax-M2.7",
"minimax/minimax-m3-free",
"minimax/minimax-m2.7-free",
"MiniMaxAI/MiniMax-M2.5",
"xiaomi/mimo-v2.5-pro",
"xiaomi/mimo-v2.5",
"Qwen/Qwen3.8-Max",
"Qwen/Qwen3.8-27B",
"Qwen/Qwen3.8-Flash",
"Qwen/Qwen3.7-Max",
"Qwen/Qwen3.7-Plus",
"Qwen/Qwen3.7-Flash",
@@ -44,6 +48,7 @@
"stepfun/Step-3.7-Flash",
"stepfun/Step-3.5-Flash",
"tencent/hy3-paid",
"tencent/hy4-preview",
"google/gemini-3.7-flash",
"google/gemini-3.6-flash",
"google/gemini-3.5-flash",
@@ -53,7 +58,6 @@
"nvidia/nemotron-3-ultra-550b-a55b",
"thinkingmachines/inkling",
"thinkingmachines/inkling-small",
"stealth/ox-alpha",
"poolside/laguna-s-2.1-free",
"meta/muse-spark-1.1",
"meta/muse-spark-1.2",
+6 -2
View File
@@ -1,5 +1,5 @@
{
"verifiedAt": "2026-08-25",
"verifiedAt": "2026-08-28",
"source": "https://commandcode.ai/docs/resources/pricing-limits",
"tierPolicy": "Use request-wide input tiers; the highest threshold exceeded by input plus cache tokens applies to the full request.",
"tiers": {
@@ -19,6 +19,7 @@
"moonshotai/Kimi-K2.7-Code-Highspeed": [1.9, 8, 0.38, 0],
"moonshotai/Kimi-K2.6": [0.95, 4, 0.16, 0],
"moonshotai/Kimi-K2.5": [0.6, 3, 0.1, 0],
"z-ai/glm-5.3-flash": [0.15, 0.5, 0.03, 0],
"zai-org/GLM-5.3": [1.4, 4.4, 0.26, 0],
"zai-org/GLM-5.2": [1.4, 4.4, 0.26, 0],
"zai-org/GLM-5.2-Fast": [3, 10.25, 0.5, 0],
@@ -31,6 +32,7 @@
"xiaomi/mimo-v2.5": [0.14, 0.28, 0.0028, 0],
"Qwen/Qwen3.8-Max": [2, 6, 0.25, 2.5],
"Qwen/Qwen3.8-27B": [0.4, 3, 0.04, 0],
"Qwen/Qwen3.8-Flash": [0.16, 0.47, 0.016, 0],
"Qwen/Qwen3.7-Max": [2.5, 7.5, 0.5, 3.13],
"Qwen/Qwen3.7-Plus": [0.4, 1.6, 0.08, 0.5],
"Qwen/Qwen3.7-Flash": [0.03, 0.13, 0.006, 0.038],
@@ -38,12 +40,14 @@
"Qwen/Qwen3.6-Plus": [0.5, 3, 0.1, 0],
"stepfun/Step-3.7-Flash": [0.2, 1.15, 0.04, 0],
"stepfun/Step-3.5-Flash": [0.1, 0.3, 0.02, 0],
"minimax/minimax-m3-free": [0, 0, 0, 0],
"minimax/minimax-m2.7-free": [0, 0, 0, 0],
"tencent/hy3-paid": [0.14, 0.58, 0.035, 0],
"tencent/hy4-preview": [0.834, 2.501, 0.042, 0],
"nvidia/nemotron-3-ultra-550b-a55b": [0.6, 2.4, 0.12, 0],
"thinkingmachines/inkling": [1, 4.05, 0.17, 0],
"thinkingmachines/inkling-small": [0.5, 1.2, 0.1, 0],
"poolside/laguna-s-2.1-free": [0, 0, 0, 0],
"stealth/ox-alpha": [0, 0, 0, 0],
"claude-sonnet-5": [2, 10, 0.2, 2.5],
"claude-sonnet-4-6": [3, 15, 0.3, 3.75],
"claude-fable-5": [10, 50, 1, 12.5],
+23 -5
View File
@@ -112,13 +112,15 @@ describe("commandCodeModelsFromApiResponse()", () => {
])
assert.deepEqual(inputModalitiesForModel("Qwen/Qwen3.8-27B"), ["text", "image"])
assert.deepEqual(inputModalitiesForModel("google/gemini-3.7-flash"), ["text", "image"])
assert.deepEqual(inputModalitiesForModel("stealth/ox-alpha"), ["text", "image"])
assert.deepEqual(inputModalitiesForModel("Qwen/Qwen3.8-Flash"), ["text", "image"])
assert.deepEqual(inputModalitiesForModel("z-ai/glm-5.3-flash"), ["text", "image"])
assert.deepEqual(inputModalitiesForModel("minimax/minimax-m3-free"), ["text", "image"])
assert.deepEqual(inputModalitiesForModel("deepseek/deepseek-v4-pro"), ["text"])
assert.deepEqual(inputModalitiesForModel("zai-org/GLM-5.3"), ["text"])
assert.deepEqual(inputModalitiesForModel("unknown-new-model"), ["text"])
assert.equal(modelSupportsImageInput("gpt-5.6-luna"), true)
assert.equal(modelSupportsImageInput("deepseek/deepseek-v4-flash-vision-exp"), true)
assert.equal(modelSupportsImageInput("stealth/ox-alpha"), true)
assert.equal(modelSupportsImageInput("z-ai/glm-5.3-flash"), true)
assert.equal(modelSupportsImageInput("deepseek/deepseek-v4-pro"), false)
assert.ok(Object.keys(MODEL_INPUT_MODALITIES).length > 0)
for (const modalities of Object.values(MODEL_INPUT_MODALITIES)) {
@@ -149,7 +151,7 @@ describe("commandCodeModelsFromApiResponse()", () => {
},
})
assert.equal(models[2]?.reasoning, false)
assert.equal(Object.keys(MODEL_REASONING).length, 48)
assert.equal(Object.keys(MODEL_REASONING).length, 50)
})
it("uses model-specific output limits from the CLI catalog", () => {
@@ -157,7 +159,7 @@ describe("commandCodeModelsFromApiResponse()", () => {
object: "list",
data: [
{ ...API_RESPONSE.data[0], id: "Qwen/Qwen3.8-27B", context_length: 262_144 },
{ ...API_RESPONSE.data[0], id: "stealth/ox-alpha", context_length: 1_048_576 },
{ ...API_RESPONSE.data[0], id: "z-ai/glm-5.3-flash", context_length: 1_048_576 },
{
...API_RESPONSE.data[0],
id: "poolside/laguna-s-2.1-free",
@@ -170,7 +172,7 @@ describe("commandCodeModelsFromApiResponse()", () => {
models.map(({ id, maxTokens }) => ({ id, maxTokens })),
[
{ id: "Qwen/Qwen3.8-27B", maxTokens: 32_768 },
{ id: "stealth/ox-alpha", maxTokens: 131_072 },
{ id: "z-ai/glm-5.3-flash", maxTokens: 131_072 },
{ id: "poolside/laguna-s-2.1-free", maxTokens: 32_768 },
],
)
@@ -217,6 +219,22 @@ describe("commandCodeModelsFromApiResponse()", () => {
xhigh: null,
max: "max",
})
assert.deepEqual(thinkingLevelMapForEfforts(MODEL_EFFORTS["Qwen/Qwen3.8-Flash"]), {
minimal: null,
low: "low",
medium: "medium",
high: null,
xhigh: "xhigh",
max: null,
})
assert.deepEqual(thinkingLevelMapForEfforts(MODEL_EFFORTS["z-ai/glm-5.3-flash"]), {
minimal: null,
low: "low",
medium: null,
high: "high",
xhigh: null,
max: "max",
})
assert.deepEqual(thinkingMetadataForModel("new-model-without-metadata"), undefined)
})
+25 -3
View File
@@ -27,7 +27,11 @@ const fixtureUrl = new URL("./fixtures/commandcode-model-ids.json", import.meta.
const fixture = JSON.parse(await readFile(fixtureUrl, "utf-8")) as ModelCatalogSnapshot
const pricingFixtureUrl = new URL("./fixtures/commandcode-pricing.json", import.meta.url)
const pricingFixture = JSON.parse(await readFile(pricingFixtureUrl, "utf-8")) as PricingSnapshot
const freeModels = new Set(["poolside/laguna-s-2.1-free", "stealth/ox-alpha"])
const freeModels = new Set([
"poolside/laguna-s-2.1-free",
"minimax/minimax-m3-free",
"minimax/minimax-m2.7-free",
])
function assertCost(
modelId: string,
@@ -50,7 +54,7 @@ function assertCost(
describe("MODEL_COSTS pricing overlay", () => {
it("covers the current Command Code model catalog snapshot", () => {
assert.equal(fixture.source, "https://api.commandcode.ai/provider/v1/models")
assert.match(fixture.fetchedAt, /^2026-08-25T/)
assert.match(fixture.fetchedAt, /^2026-08-28T/)
const catalogIds = [...fixture.modelIds].sort()
const pricedIds = Object.keys(MODEL_COSTS).sort()
@@ -144,6 +148,24 @@ describe("MODEL_COSTS pricing overlay", () => {
cacheRead: 0.04,
cacheWrite: 0,
})
assertCost("Qwen/Qwen3.8-Flash", {
input: 0.16,
output: 0.47,
cacheRead: 0.016,
cacheWrite: 0,
})
assertCost("z-ai/glm-5.3-flash", {
input: 0.15,
output: 0.5,
cacheRead: 0.03,
cacheWrite: 0,
})
assertCost("tencent/hy4-preview", {
input: 0.834,
output: 2.501,
cacheRead: 0.042,
cacheWrite: 0,
})
assertCost("google/gemini-3.7-flash", {
input: 0.75,
output: 3.75,
@@ -196,7 +218,7 @@ describe("MODEL_COSTS pricing overlay", () => {
it("tracks pricing provenance", () => {
assert.equal(PRICING_SOURCE_URL, "https://commandcode.ai/docs/resources/pricing-limits")
assert.equal(PRICING_LAST_VERIFIED, "2026-08-25")
assert.equal(PRICING_LAST_VERIFIED, "2026-08-28")
})
it("fails once temporary pricing needs review", () => {