diff --git a/src/config.test.ts b/src/config.test.ts index 6c68801e1..ca139373c 100644 --- a/src/config.test.ts +++ b/src/config.test.ts @@ -1460,6 +1460,15 @@ describe("buildOpenAISource", () => { expect(source.baseURL).toBe("https://fp/v1"); }); + test("stays above the reasoning truncation floor", () => { + // Reasoning tokens consume max_output_tokens before any answer text is + // emitted. Measured on muse-spark-1.3-contributor, a 512-token cap at + // medium effort spent 397 tokens reasoning and returned 3 tokens of + // answer; 1024 was the lowest cap that answered on every rung. 4096 is + // the floor we will not drop below. See CL-7867. + expect(SOURCE_MAX_TOKENS).toBeGreaterThanOrEqual(4096); + }); + test("omits reasoning_effort when effort is absent", () => { const source = buildOpenAISource({ id: "fp", diff --git a/src/provider/reasoning-effort.test.ts b/src/provider/reasoning-effort.test.ts index 996d791cc..d658b227d 100644 --- a/src/provider/reasoning-effort.test.ts +++ b/src/provider/reasoning-effort.test.ts @@ -163,6 +163,40 @@ describe("supportedEfforts", () => { expect(supportedEfforts("glm-5.3")).toEqual(["low", "high", "max"]); expect(supportedEfforts("glm-5.3-flash")).toEqual(["low", "high", "max"]); }); + + // Five ids ship across two catalogs (packages/opencode-go and packages/zen) + // and nothing normalizes the model string before it reaches supportedEfforts, + // so every one of them has to land on the same ladder. + test.each([ + "muse-spark-1.3-contributor", + "muse-spark-1.2-contributor", + "muse-spark-1.3", + "muse-spark-1.2", + "muse-spark-1.3-contributor-free", + ])("Muse Spark id %s supports minimal through high", (model) => { + expect(supportedEfforts(model)).toEqual([ + "minimal", + "low", + "medium", + "high", + ]); + expect(defaultEffortForModel(model)).toBe("low"); + }); + + test("a model merely containing muse-spark is not matched", () => { + expect(supportedEfforts("not-muse-spark-1.3")).toEqual([ + "low", + "medium", + "high", + ]); + }); + + test("Muse Spark never offers none", () => { + // The Go gateway answers HTTP 400 on reasoning.effort: "none". + expect(supportedEfforts("muse-spark-1.3-contributor")).not.toContain( + "none", + ); + }); }); describe("validateEffort", () => { @@ -189,6 +223,13 @@ describe("validateEffort", () => { expect(validateEffort("grok-4.5", "xhigh").ok).toBe(false); }); + test("accepts minimal on Muse Spark and rejects none", () => { + expect(validateEffort("muse-spark-1.3-contributor", "minimal")).toEqual({ + ok: true, + }); + expect(validateEffort("muse-spark-1.3-contributor", "none").ok).toBe(false); + }); + test("rejects medium on glm-5.3 family", () => { expect(validateEffort("glm-5.3", "medium").ok).toBe(false); expect(validateEffort("glm-5.3-flash", "medium").ok).toBe(false); diff --git a/src/provider/reasoning-effort.ts b/src/provider/reasoning-effort.ts index 4fe0c0841..78a3a6359 100644 --- a/src/provider/reasoning-effort.ts +++ b/src/provider/reasoning-effort.ts @@ -61,6 +61,30 @@ const UNKNOWN_MODEL_EFFORTS: readonly ReasoningEffort[] = [ "high", ]; +// Muse Spark (Responses protocol) accepts minimal through high. Not `none` — +// the gateway rejects it with HTTP 400 on `reasoning.effort`. Measured on +// muse-spark-1.3-contributor and muse-spark-1.2-contributor via the Go +// endpoint and muse-spark-1.3-contributor-free via Zen: `minimal` returns 200 +// and `none` returns 400 on all three. See CL-7867. +const MUSE_SPARK_EFFORTS: readonly ReasoningEffort[] = [ + "minimal", + "low", + "medium", + "high", +]; + +// Matched by prefix, not by an id list. The family ships under five ids across +// two catalogs — `muse-spark-1.3-contributor` / `-1.2-contributor` in +// packages/opencode-go, and `muse-spark-1.3` / `-1.2` / +// `-1.3-contributor-free` in packages/zen — and nothing normalizes the model +// string before it reaches here. An exact list silently missed three of them +// and left the ladder at the unknown-model default. The `/^.../i` + `trim()` +// shape mirrors the grok/kimi prefix checks in +// src/subagent/provider-family.ts. +function isMuseSparkModel(model: string): boolean { + return /^muse-spark/i.test(model.trim()); +} + // grok-4.6 accepts xhigh; grok-4.5 and composer stay on the unknown-model subset. const GROK_46_EFFORTS: readonly ReasoningEffort[] = [ "low", @@ -143,6 +167,9 @@ export function supportedEfforts( if (GLM_53_MODELS.includes(model)) { return [...GLM_53_EFFORTS]; } + if (isMuseSparkModel(model)) { + return [...MUSE_SPARK_EFFORTS]; + } return [...UNKNOWN_MODEL_EFFORTS]; } @@ -196,7 +223,7 @@ export function cycleReasoningEffort( * (`defaultEffortForDirector`): this is what the prompt shows and what Shift+Tab * advances from when the operator has not picked a level. * - * Family table: grok* → high; glm-5.3* → max; Codex → medium; gpt-5.1 chat (`none` on the + * Family table: grok* → high; glm-5.3* → max; muse-spark* → low; Codex → medium; gpt-5.1 chat (`none` on the * ladder, not Codex) → none; gpt-5/gpt-6/o1/o3/o4 → medium. Unknown models with a * conservative rung set stay undefined so we do not invent a family default. */ @@ -210,6 +237,7 @@ export function defaultEffortForModel( supported.includes(desired) ? desired : undefined; if (model.startsWith("grok")) return pick("high"); if (GLM_53_MODELS.includes(model)) return pick("max"); + if (isMuseSparkModel(model)) return pick("low"); if (!isCodex && supported.includes("none")) return "none"; if (isCodex || isKnownOpenAIReasoningModel(model)) return pick("medium"); return undefined;