Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 9 additions & 0 deletions src/config.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -1460,6 +1460,15 @@ describe("buildOpenAISource", () => {
expect(source.baseURL).toBe("https://fp/v1");
});

test("stays above the reasoning truncation floor", () => {
// Reasoning tokens consume max_output_tokens before any answer text is
// emitted. Measured on muse-spark-1.3-contributor, a 512-token cap at
// medium effort spent 397 tokens reasoning and returned 3 tokens of
// answer; 1024 was the lowest cap that answered on every rung. 4096 is
// the floor we will not drop below. See CL-7867.
expect(SOURCE_MAX_TOKENS).toBeGreaterThanOrEqual(4096);
});

test("omits reasoning_effort when effort is absent", () => {
const source = buildOpenAISource({
id: "fp",
Expand Down
41 changes: 41 additions & 0 deletions src/provider/reasoning-effort.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -163,6 +163,40 @@ describe("supportedEfforts", () => {
expect(supportedEfforts("glm-5.3")).toEqual(["low", "high", "max"]);
expect(supportedEfforts("glm-5.3-flash")).toEqual(["low", "high", "max"]);
});

// Five ids ship across two catalogs (packages/opencode-go and packages/zen)
// and nothing normalizes the model string before it reaches supportedEfforts,
// so every one of them has to land on the same ladder.
test.each([
"muse-spark-1.3-contributor",
"muse-spark-1.2-contributor",
"muse-spark-1.3",
"muse-spark-1.2",
"muse-spark-1.3-contributor-free",
])("Muse Spark id %s supports minimal through high", (model) => {
expect(supportedEfforts(model)).toEqual([
"minimal",
"low",
"medium",
"high",
]);
expect(defaultEffortForModel(model)).toBe("low");
});

test("a model merely containing muse-spark is not matched", () => {
expect(supportedEfforts("not-muse-spark-1.3")).toEqual([
"low",
"medium",
"high",
]);
});

test("Muse Spark never offers none", () => {
// The Go gateway answers HTTP 400 on reasoning.effort: "none".
expect(supportedEfforts("muse-spark-1.3-contributor")).not.toContain(
"none",
);
});
});

describe("validateEffort", () => {
Expand All @@ -189,6 +223,13 @@ describe("validateEffort", () => {
expect(validateEffort("grok-4.5", "xhigh").ok).toBe(false);
});

test("accepts minimal on Muse Spark and rejects none", () => {
expect(validateEffort("muse-spark-1.3-contributor", "minimal")).toEqual({
ok: true,
});
expect(validateEffort("muse-spark-1.3-contributor", "none").ok).toBe(false);
});

test("rejects medium on glm-5.3 family", () => {
expect(validateEffort("glm-5.3", "medium").ok).toBe(false);
expect(validateEffort("glm-5.3-flash", "medium").ok).toBe(false);
Expand Down
30 changes: 29 additions & 1 deletion src/provider/reasoning-effort.ts
Original file line number Diff line number Diff line change
Expand Up @@ -61,6 +61,30 @@ const UNKNOWN_MODEL_EFFORTS: readonly ReasoningEffort[] = [
"high",
];

// Muse Spark (Responses protocol) accepts minimal through high. Not `none` —
// the gateway rejects it with HTTP 400 on `reasoning.effort`. Measured on
// muse-spark-1.3-contributor and muse-spark-1.2-contributor via the Go
// endpoint and muse-spark-1.3-contributor-free via Zen: `minimal` returns 200
// and `none` returns 400 on all three. See CL-7867.
const MUSE_SPARK_EFFORTS: readonly ReasoningEffort[] = [
"minimal",
"low",
"medium",
"high",
];

// Matched by prefix, not by an id list. The family ships under five ids across
// two catalogs — `muse-spark-1.3-contributor` / `-1.2-contributor` in
// packages/opencode-go, and `muse-spark-1.3` / `-1.2` /
// `-1.3-contributor-free` in packages/zen — and nothing normalizes the model
// string before it reaches here. An exact list silently missed three of them
// and left the ladder at the unknown-model default. The `/^.../i` + `trim()`
// shape mirrors the grok/kimi prefix checks in
// src/subagent/provider-family.ts.
function isMuseSparkModel(model: string): boolean {
return /^muse-spark/i.test(model.trim());
}

// grok-4.6 accepts xhigh; grok-4.5 and composer stay on the unknown-model subset.
const GROK_46_EFFORTS: readonly ReasoningEffort[] = [
"low",
Expand Down Expand Up @@ -143,6 +167,9 @@ export function supportedEfforts(
if (GLM_53_MODELS.includes(model)) {
return [...GLM_53_EFFORTS];
}
if (isMuseSparkModel(model)) {
return [...MUSE_SPARK_EFFORTS];
}
return [...UNKNOWN_MODEL_EFFORTS];
}

Expand Down Expand Up @@ -196,7 +223,7 @@ export function cycleReasoningEffort(
* (`defaultEffortForDirector`): this is what the prompt shows and what Shift+Tab
* advances from when the operator has not picked a level.
*
* Family table: grok* → high; glm-5.3* → max; Codex → medium; gpt-5.1 chat (`none` on the
* Family table: grok* → high; glm-5.3* → max; muse-spark* → low; Codex → medium; gpt-5.1 chat (`none` on the
* ladder, not Codex) → none; gpt-5/gpt-6/o1/o3/o4 → medium. Unknown models with a
* conservative rung set stay undefined so we do not invent a family default.
*/
Expand All @@ -210,6 +237,7 @@ export function defaultEffortForModel(
supported.includes(desired) ? desired : undefined;
if (model.startsWith("grok")) return pick("high");
if (GLM_53_MODELS.includes(model)) return pick("max");
if (isMuseSparkModel(model)) return pick("low");
if (!isCodex && supported.includes("none")) return "none";
if (isCodex || isKnownOpenAIReasoningModel(model)) return pick("medium");
return undefined;
Expand Down
Loading