test(models): assert catalog invariants instead of pinned model ids

The model catalog tests hard-coded specific model ids, the reasoning model
count, and output limits from command-code@1.32.2. Every upstream catalog
sync broke them, which made the daily catalog sync workflow fail before it
could open its PR.

Assert structural invariants over the generated catalog instead: image
models resolve to text+image, every effort entry has a reasoning flag,
reasoning without efforts yields an empty level map, output limits are
positive integers and clamp to the context length.

Closes #67

(cherry picked from commit 458e3a57892bb625a78fb5ded092627ba3cf2867)
This commit is contained in:
Patrick Wozniak
2026-09-01 23:28:06 +02:00
parent 09faff6e6e
commit 4d515c0d6b
+42 -55
View File
@@ -104,43 +104,48 @@ describe("commandCodeModelsFromApiResponse()", () => {
}) })
it(`uses the command-code@${COMMAND_CODE_CLI_VERSION} image capability catalog`, () => { it(`uses the command-code@${COMMAND_CODE_CLI_VERSION} image capability catalog`, () => {
assert.deepEqual(inputModalitiesForModel("gpt-5.6-luna"), ["text", "image"]) const imageModels = Object.keys(MODEL_INPUT_MODALITIES)
assert.deepEqual(inputModalitiesForModel("meta/muse-spark-1.2"), ["text", "image"]) assert.ok(imageModels.length > 0)
assert.deepEqual(inputModalitiesForModel("deepseek/deepseek-v4-flash-vision-exp"), [ for (const modelId of imageModels) {
"text", assert.deepEqual(MODEL_INPUT_MODALITIES[modelId], ["text", "image"], modelId)
"image", assert.deepEqual(inputModalitiesForModel(modelId), ["text", "image"], modelId)
]) assert.equal(modelSupportsImageInput(modelId), true, modelId)
assert.deepEqual(inputModalitiesForModel("Qwen/Qwen3.8-27B"), ["text", "image"])
assert.deepEqual(inputModalitiesForModel("google/gemini-3.7-flash"), ["text", "image"])
assert.deepEqual(inputModalitiesForModel("Qwen/Qwen3.8-Flash"), ["text", "image"])
assert.deepEqual(inputModalitiesForModel("z-ai/glm-5.3-flash"), ["text", "image"])
assert.deepEqual(inputModalitiesForModel("minimax/minimax-m3-free"), ["text", "image"])
assert.deepEqual(inputModalitiesForModel("deepseek/deepseek-v4-pro"), ["text"])
assert.deepEqual(inputModalitiesForModel("zai-org/GLM-5.3"), ["text"])
assert.deepEqual(inputModalitiesForModel("unknown-new-model"), ["text"])
assert.equal(modelSupportsImageInput("gpt-5.6-luna"), true)
assert.equal(modelSupportsImageInput("deepseek/deepseek-v4-flash-vision-exp"), true)
assert.equal(modelSupportsImageInput("z-ai/glm-5.3-flash"), true)
assert.equal(modelSupportsImageInput("deepseek/deepseek-v4-pro"), false)
assert.ok(Object.keys(MODEL_INPUT_MODALITIES).length > 0)
for (const modalities of Object.values(MODEL_INPUT_MODALITIES)) {
assert.deepEqual(modalities, ["text", "image"])
} }
const textOnlyModel = Object.keys(MODEL_REASONING).find(
(modelId) => !(modelId in MODEL_INPUT_MODALITIES),
)
assert.ok(textOnlyModel, "catalog should contain at least one text-only model")
assert.deepEqual(inputModalitiesForModel(textOnlyModel), ["text"])
assert.equal(modelSupportsImageInput(textOnlyModel), false)
assert.deepEqual(inputModalitiesForModel("unknown-new-model"), ["text"])
assert.equal(modelSupportsImageInput("unknown-new-model"), false)
}) })
it("tracks reasoning independently from selectable effort levels", () => { it("tracks reasoning independently from selectable effort levels", () => {
const reasoningModels = Object.keys(MODEL_REASONING)
const effortModels = Object.keys(MODEL_EFFORTS)
assert.ok(reasoningModels.length > 0)
assert.ok(effortModels.length > 0)
for (const modelId of effortModels) {
assert.equal(MODEL_REASONING[modelId], true, `${modelId} has efforts but no reasoning flag`)
}
const reasoningWithoutEfforts = reasoningModels.find((modelId) => !(modelId in MODEL_EFFORTS))
assert.ok(reasoningWithoutEfforts, "catalog should contain a reasoning model without efforts")
const models = commandCodeModelsFromApiResponse({ const models = commandCodeModelsFromApiResponse({
object: "list", object: "list",
data: [ data: [
{ ...API_RESPONSE.data[0], id: "deepseek/deepseek-v4-flash" }, { ...API_RESPONSE.data[0], id: effortModels[0] },
{ ...API_RESPONSE.data[0], id: "moonshotai/Kimi-K3" }, { ...API_RESPONSE.data[0], id: reasoningWithoutEfforts },
{ ...API_RESPONSE.data[0], id: "new-model-without-metadata" }, { ...API_RESPONSE.data[0], id: "new-model-without-metadata" },
], ],
}) })
assert.equal(models[0]?.reasoning, true) assert.equal(models[0]?.reasoning, true)
assert.equal(models[1]?.reasoning, true) assert.equal(models[1]?.reasoning, true)
assert.deepEqual(thinkingMetadataForModel("moonshotai/Kimi-K3"), { assert.deepEqual(thinkingMetadataForModel(reasoningWithoutEfforts), {
thinkingLevelMap: { thinkingLevelMap: {
minimal: null, minimal: null,
low: null, low: null,
@@ -151,32 +156,30 @@ describe("commandCodeModelsFromApiResponse()", () => {
}, },
}) })
assert.equal(models[2]?.reasoning, false) assert.equal(models[2]?.reasoning, false)
assert.equal(Object.keys(MODEL_REASONING).length, 50)
}) })
it("uses model-specific output limits from the CLI catalog", () => { it("uses model-specific output limits from the CLI catalog", () => {
const limitedModels = Object.entries(MODEL_MAX_OUTPUT_TOKENS)
assert.ok(limitedModels.length > 0)
for (const [modelId, limit] of limitedModels) {
assert.ok(Number.isInteger(limit) && limit > 0, `${modelId} has an invalid output limit`)
}
const [limitedId, limit] = limitedModels[0]!
const models = commandCodeModelsFromApiResponse({ const models = commandCodeModelsFromApiResponse({
object: "list", object: "list",
data: [ data: [
{ ...API_RESPONSE.data[0], id: "Qwen/Qwen3.8-27B", context_length: 262_144 }, { ...API_RESPONSE.data[0], id: limitedId, context_length: limit * 4 },
{ ...API_RESPONSE.data[0], id: "z-ai/glm-5.3-flash", context_length: 1_048_576 }, { ...API_RESPONSE.data[0], id: limitedId, context_length: Math.floor(limit / 2) },
{ { ...API_RESPONSE.data[0], id: "unknown-new-model", context_length: 256_000 },
...API_RESPONSE.data[0], { ...API_RESPONSE.data[0], id: "unknown-new-model", context_length: 8_192 },
id: "poolside/laguna-s-2.1-free",
context_length: 256_000,
},
], ],
}) })
assert.deepEqual( assert.deepEqual(
models.map(({ id, maxTokens }) => ({ id, maxTokens })), models.map(({ maxTokens }) => maxTokens),
[ [limit, Math.floor(limit / 2), 65_536, 8_192],
{ id: "Qwen/Qwen3.8-27B", maxTokens: 32_768 },
{ id: "z-ai/glm-5.3-flash", maxTokens: 131_072 },
{ id: "poolside/laguna-s-2.1-free", maxTokens: 32_768 },
],
) )
assert.equal(Object.keys(MODEL_MAX_OUTPUT_TOKENS).length, 3)
}) })
it(`uses the command-code@${COMMAND_CODE_CLI_VERSION} reasoning effort catalog`, () => { it(`uses the command-code@${COMMAND_CODE_CLI_VERSION} reasoning effort catalog`, () => {
@@ -219,22 +222,6 @@ describe("commandCodeModelsFromApiResponse()", () => {
xhigh: null, xhigh: null,
max: "max", max: "max",
}) })
assert.deepEqual(thinkingLevelMapForEfforts(MODEL_EFFORTS["Qwen/Qwen3.8-Flash"]), {
minimal: null,
low: "low",
medium: "medium",
high: null,
xhigh: "xhigh",
max: null,
})
assert.deepEqual(thinkingLevelMapForEfforts(MODEL_EFFORTS["z-ai/glm-5.3-flash"]), {
minimal: null,
low: "low",
medium: null,
high: "high",
xhigh: null,
max: "max",
})
assert.deepEqual(thinkingMetadataForModel("new-model-without-metadata"), undefined) assert.deepEqual(thinkingMetadataForModel("new-model-without-metadata"), undefined)
}) })