Skip to content

Commit a89f8c3

Browse files
committed
Send max_completion_tokens to OpenAI reasoning models
First-party OpenAI reasoning models reject max_tokens, so the preset must send max_completion_tokens for those models. The requirement is declared per model on the catalog entry rather than inferred from name prefixes, and the quirk follows the first-party endpoint so relays serving the same model names keep max_tokens.
1 parent fcc150e commit a89f8c3

7 files changed

Lines changed: 154 additions & 2 deletions

File tree

‎packages/first-class-providers/src/index.ts‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1,5 +1,7 @@
11
export {
22
FIRST_CLASS_PROVIDERS,
3+
OPENAI_API_BASE_URL,
4+
OPENAI_API_MAX_COMPLETION_TOKENS_MODELS,
35
connectListProviders,
46
firstClassPathAsProvider,
57
firstClassProviderById,

‎packages/first-class-providers/src/providers.test.ts‎

Lines changed: 16 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -56,6 +56,22 @@ describe("FIRST_CLASS_PROVIDERS", () => {
5656
expect(api?.models).toContain(api?.defaultModel);
5757
});
5858

59+
test("OpenAI API path declares max_completion_tokens models explicitly", () => {
60+
const openai = firstClassProviderById("openai");
61+
const api = openai?.paths?.find((p) => p.id === "api");
62+
expect(api?.maxCompletionTokensModels).toEqual([
63+
"gpt-6-astra",
64+
"gpt-5.4",
65+
"gpt-5.4-mini",
66+
"o3",
67+
"o4-mini",
68+
]);
69+
for (const model of api?.maxCompletionTokensModels ?? []) {
70+
expect(api?.models).toContain(model);
71+
}
72+
expect(api?.maxCompletionTokensModels).not.toContain("gpt-4.1");
73+
});
74+
5975
test("OpenAI API and Zen catalogs include gpt-6-astra without changing defaults", () => {
6076
const openai = firstClassProviderById("openai");
6177
const api = openai?.paths?.find((p) => p.id === "api");

‎packages/first-class-providers/src/providers.ts‎

Lines changed: 22 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -17,6 +17,23 @@ const OPENAI_API_MODELS = [
1717
] as const;
1818
const OPENAI_API_DEFAULT = "gpt-5.4";
1919

20+
/** First-party OpenAI chat-completions endpoint for the API-key path below. */
21+
export const OPENAI_API_BASE_URL = "https://api.openai.com/v1";
22+
23+
/**
24+
* Preset models whose first-party endpoint rejects `max_tokens` and requires
25+
* `max_completion_tokens`. Explicit per-model list: adding a model here
26+
* declares its own requirement, never inferred from name prefixes. gpt-4.1
27+
* is non-reasoning and stays on `max_tokens`.
28+
*/
29+
export const OPENAI_API_MAX_COMPLETION_TOKENS_MODELS: readonly string[] = [
30+
"gpt-6-astra",
31+
"gpt-5.4",
32+
"gpt-5.4-mini",
33+
"o3",
34+
"o4-mini",
35+
];
36+
2037
/**
2138
* First-class providers shown in the models-surface Connect list.
2239
* Tier A order: dual-path OpenAI, OAuth xAI, Go/Zen, Z.AI, big three, Custom.
@@ -38,11 +55,12 @@ export const FIRST_CLASS_PROVIDERS: readonly FirstClassProviderDef[] = [
3855
id: "api",
3956
label: "OpenAI API — API key",
4057
auth: "api-key",
41-
baseURL: "https://api.openai.com/v1",
58+
baseURL: OPENAI_API_BASE_URL,
4259
models: OPENAI_API_MODELS,
4360
defaultModel: OPENAI_API_DEFAULT,
4461
authHint: "Paste your OpenAI API key (sk-...)",
4562
providerId: "openai",
63+
maxCompletionTokensModels: OPENAI_API_MAX_COMPLETION_TOKENS_MODELS,
4664
},
4765
],
4866
},
@@ -167,6 +185,9 @@ export function firstClassPathAsProvider(
167185
? { defaultModel: path.defaultModel }
168186
: {}),
169187
...(path.authHint !== undefined ? { authHint: path.authHint } : {}),
188+
...(path.maxCompletionTokensModels !== undefined
189+
? { maxCompletionTokensModels: path.maxCompletionTokensModels }
190+
: {}),
170191
...(def.anthropic === true ? { anthropic: true } : {}),
171192
...(def.opencodeGo === true ? { opencodeGo: true } : {}),
172193
...(def.billingProduct !== undefined

‎packages/first-class-providers/src/types.ts‎

Lines changed: 14 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -28,6 +28,15 @@ export interface FirstClassProviderPath {
2828
* e.g. "codex" for ChatGPT OAuth, "openai" for API key.
2929
*/
3030
providerId?: string;
31+
/**
32+
* Models on this path whose endpoint rejects `max_tokens` and requires
33+
* `max_completion_tokens` instead (first-party OpenAI reasoning models).
34+
* An explicit per-model list: adding a model here declares its own
35+
* requirement, never inferred from name prefixes. Relays serving the same
36+
* model names through other endpoints are unaffected — the quirk follows
37+
* this endpoint, not the bare model name.
38+
*/
39+
maxCompletionTokensModels?: readonly string[];
3140
}
3241

3342
export interface FirstClassProviderDef {
@@ -57,4 +66,9 @@ export interface FirstClassProviderDef {
5766
* api-key flow runs against that path's fields / providerId.
5867
*/
5968
paths?: readonly FirstClassProviderPath[];
69+
/**
70+
* Carried from a chooser path by firstClassPathAsProvider when the seeded
71+
* def originates from a path (see FirstClassProviderPath for semantics).
72+
*/
73+
maxCompletionTokensModels?: readonly string[];
6074
}

‎src/config/index.ts‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -209,6 +209,7 @@ export function buildOpenAISource(fields: {
209209
apiKey?: string;
210210
model: string;
211211
reasoningEffort?: ReasoningEffort;
212+
quirks?: Record<string, unknown>;
212213
}): InferenceSource {
213214
const overrides =
214215
fields.reasoningEffort !== undefined
@@ -226,6 +227,7 @@ export function buildOpenAISource(fields: {
226227
: KEYLESS_API_KEY,
227228
model: fields.model,
228229
defaults: { maxTokens: SOURCE_MAX_TOKENS, ...overrides },
230+
...(fields.quirks !== undefined ? { quirks: fields.quirks } : {}),
229231
};
230232
}
231233

‎src/config/inference-sources.test.ts‎

Lines changed: 71 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -12,6 +12,7 @@ import {
1212
setProviderContextWindowOverrides,
1313
} from "../provider/context-window.js";
1414
import { createOpenAICompatibleAdapter } from "../provider/openai-compatible-adapter.js";
15+
import { firstClassProviderById } from "../../packages/first-class-providers/src/index.js";
1516

1617
const WINDOW = 400_000;
1718

@@ -85,3 +86,73 @@ describe("contextWindow / maxTokens split (CL-7784)", () => {
8586
expect(contextWindowFor("fp:fp-large")).toBe(WINDOW);
8687
});
8788
});
89+
90+
describe("OpenAI reasoning max_completion_tokens quirk (CL-7785)", () => {
91+
const openaiApi = firstClassProviderById("openai")?.paths?.find(
92+
(p) => p.id === "api",
93+
);
94+
const presetModels = [...(openaiApi?.models ?? [])];
95+
const presetBaseURL = openaiApi?.baseURL ?? "";
96+
// A relay serving the same model names through the same adapter but taking
97+
// max_tokens (per the vendor adapter comment) — the quirk must not follow
98+
// the bare model name there.
99+
const RELAY_BASE_URL = "https://opencode.ai/zen/v1";
100+
101+
function wireBody(model: string, baseURL: string): Record<string, unknown> {
102+
const entryCatalog: ProviderCatalogEntry[] = [
103+
{ name: "openai", baseURL, apiKey: "test-key", models: [model] },
104+
];
105+
const source = buildInferenceSourceForRef(
106+
{ provider: "openai", model },
107+
{ sessionId: "sess-1", catalog: entryCatalog },
108+
undefined,
109+
);
110+
// Mirror the harness: it resolves the adapter with source.quirks.
111+
const adapter = createOpenAICompatibleAdapter(
112+
source as unknown as Parameters<typeof createOpenAICompatibleAdapter>[0],
113+
source?.quirks,
114+
);
115+
const messages = [
116+
{ role: "user", content: [{ type: "text", text: "hi" }] },
117+
] as unknown as ConversationTurn[];
118+
const built = adapter.buildRequest(messages, model, {
119+
maxTokens: source?.defaults?.maxTokens,
120+
} as InferenceOptions);
121+
return JSON.parse(built.body) as Record<string, unknown>;
122+
}
123+
124+
test("shipped preset declares an explicit per-model requirement", () => {
125+
expect(presetModels.length).toBeGreaterThan(0);
126+
expect(openaiApi?.maxCompletionTokensModels?.length).toBeGreaterThan(0);
127+
for (const model of openaiApi?.maxCompletionTokensModels ?? []) {
128+
expect(presetModels).toContain(model);
129+
}
130+
});
131+
132+
test("reasoning preset models emit max_completion_tokens, never max_tokens", () => {
133+
for (const model of openaiApi?.maxCompletionTokensModels ?? []) {
134+
const body = wireBody(model, presetBaseURL);
135+
expect(body["max_completion_tokens"]).toBe(SOURCE_MAX_TOKENS);
136+
expect("max_tokens" in body).toBe(false);
137+
}
138+
});
139+
140+
test("non-reasoning preset models keep max_tokens", () => {
141+
const declared = new Set(openaiApi?.maxCompletionTokensModels ?? []);
142+
const rest = presetModels.filter((m) => !declared.has(m));
143+
expect(rest.length).toBeGreaterThan(0);
144+
for (const model of rest) {
145+
const body = wireBody(model, presetBaseURL);
146+
expect(body["max_tokens"]).toBe(SOURCE_MAX_TOKENS);
147+
expect("max_completion_tokens" in body).toBe(false);
148+
}
149+
});
150+
151+
test("relay endpoint keeps max_tokens for every preset model", () => {
152+
for (const model of presetModels) {
153+
const body = wireBody(model, RELAY_BASE_URL);
154+
expect(body["max_tokens"]).toBe(SOURCE_MAX_TOKENS);
155+
expect("max_completion_tokens" in body).toBe(false);
156+
}
157+
});
158+
});

‎src/config/inference-sources.ts‎

Lines changed: 27 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -9,7 +9,11 @@ import {
99
buildXaiSource,
1010
type ProviderCatalogEntry,
1111
} from "./index.js";
12-
import type { Settings } from "./settings.js";
12+
import {
13+
OPENAI_API_BASE_URL,
14+
OPENAI_API_MAX_COMPLETION_TOKENS_MODELS,
15+
} from "../../packages/first-class-providers/src/index.js";
16+
import { normalizeOpenAICompatibleBaseURL, type Settings } from "./settings.js";
1317
import {
1418
resolveSessionEffort,
1519
type ReasoningEffort,
@@ -36,6 +40,26 @@ function catalogEntry(
3640
return catalog.find((e) => e.name === provider);
3741
}
3842

43+
// First-party OpenAI reasoning models reject `max_tokens` and require
44+
// `max_completion_tokens`. The requirement is declared per model on the
45+
// first-class OpenAI API-key preset — never inferred from name prefixes —
46+
// and the quirk attaches to the source actually in use: it follows the
47+
// first-party endpoint, so relays serving the same model names through the
48+
// same adapter keep `max_tokens`.
49+
function openAISourceQuirks(
50+
baseURL: string,
51+
model: string,
52+
): Record<string, unknown> | undefined {
53+
const normalized = normalizeOpenAICompatibleBaseURL(baseURL);
54+
if (normalized !== normalizeOpenAICompatibleBaseURL(OPENAI_API_BASE_URL)) {
55+
return undefined;
56+
}
57+
if (!OPENAI_API_MAX_COMPLETION_TOKENS_MODELS.includes(model)) {
58+
return undefined;
59+
}
60+
return { maxTokensField: "max_completion_tokens" };
61+
}
62+
3963
export function buildInferenceSourceForRef(
4064
ref: ProviderRef,
4165
ctx: BuildSourceContext,
@@ -124,6 +148,7 @@ export function buildInferenceSourceForRef(
124148
});
125149
}
126150

151+
const quirks = openAISourceQuirks(baseURL, ref.model);
127152
return buildOpenAISource({
128153
id: ref.provider,
129154
baseURL,
@@ -134,6 +159,7 @@ export function buildInferenceSourceForRef(
134159
: {}),
135160
model: ref.model,
136161
...(effort !== undefined ? { reasoningEffort: effort } : {}),
162+
...(quirks !== undefined ? { quirks } : {}),
137163
});
138164
}
139165

0 commit comments

Comments
 (0)