-
Notifications
You must be signed in to change notification settings - Fork 13
Expand file tree
/
Copy pathmodels.js
More file actions
253 lines (235 loc) · 10.7 KB
/
Copy pathmodels.js
File metadata and controls
253 lines (235 loc) · 10.7 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
/**
* The model catalog makefaster offers after the provider picker: up to five
* models per provider, ranked by intelligence — plus the hosted provider's own
* short list, which is ranked by nothing because the server decides what is on
* it (see HOSTED_MODELS).
*
* Ranking source of truth is the CursorBench 3.2 snapshot (captured
* 2026-07-16) that jjcm/bb-plugin-autorouter carries in `benchmarks.ts` as
* `CURSOR_BENCHMARKS`:
* https://github.com/jjcm/bb-plugin-autorouter/blob/main/benchmarks.ts
* That file scores family x reasoning-level pairs; FAMILY_BEST below keeps the
* best score per family, which is what "most intelligent" means here. To
* refresh the ranking, re-read that file and update FAMILY_BEST.
*
* The three providers do not share a model-id namespace, and none of the ids
* here is invented:
*
* Cursor ids are *variants*: family plus reasoning effort plus an
* optional `-fast` twin, e.g. `claude-fable-5-thinking-medium`.
* They come from `cursor-agent --list-models`, and the picker
* intersects this catalog with that live list when it can run it.
* Claude Code ids carry no effort suffix — reasoning is a separate field —
* and some carry a bracketed context parameter that is part of
* the id, e.g. `claude-opus-4-8[1m]`. This is bb's curated
* catalog (plugins/provider-claude-code/src/model-catalog-data.ts),
* which is exactly five rows.
* Codex plain family ids from the app-server's `model/list`.
*
* A family the snapshot does not score is never given a made-up number: it
* sorts after every scored model and says so. Codex only has four scored
* families, so its fifth slot is filled from the live list or left empty.
*/
/**
* The hosted provider's models — the ones the makefaster.dev proxy will spend
* its OpenRouter credential on. The set is the server's, not this catalog's,
* so these ids must match `AllowedModels` in backend/internal/inference. None
* is in the CursorBench snapshot, so none carries a score.
*
* The ids are sent verbatim. The labels are for the picker. A single entry is
* selected without a prompt; if the list grows, the first entry is the default.
* Union Alpha is the free option.
*/
export const HOSTED_MODELS = Object.freeze([
Object.freeze({
id: "stealth/union-alpha",
label: "Union Alpha (free)",
detail: "free on OpenRouter via makefaster.dev",
}),
]);
/**
* Resolve a `--model` value for the hosted provider. Unlike the agent CLIs,
* there is no passthrough: the proxy serves exactly these, so an id it does
* not serve is a mistake to report rather than something to forward and have
* refused mid-run. Matching is case-insensitive; the id that comes back is the
* canonical one to send.
*
* @returns {{id: string, label: string, detail: string}|null} null when not offered
*/
export function resolveHostedModel(modelId) {
const wanted = String(modelId ?? "").trim().toLowerCase();
if (wanted === "") return null;
return HOSTED_MODELS.find((model) => model.id.toLowerCase() === wanted) ?? null;
}
/** Best CursorBench 3.2 score per model family, and the effort it came from. */
export const FAMILY_BEST = new Map([
["claude-fable-5", { score: 70.5, reasoning: "max" }],
["gpt-5.6-sol", { score: 67.2, reasoning: "max" }],
["gpt-5.6-terra", { score: 64.9, reasoning: "max" }],
["claude-opus-4-8", { score: 62.3, reasoning: "max" }],
["claude-sonnet-5", { score: 61.5, reasoning: "max" }],
["gpt-5.6-luna", { score: 61.1, reasoning: "max" }],
// grok-4.5 scores in the snapshot already carry the requested 10% reduction.
["grok-4.5", { score: 60.03, reasoning: "high" }],
["gpt-5.5", { score: 58.4, reasoning: "high" }],
["composer-2.5", { score: 56.1, reasoning: "medium" }],
]);
/**
* Longest-match family lookup, so `claude-opus-4-8-thinking-medium` does not
* resolve through a shorter family that happens to be a substring.
*/
export function benchmarkFamily(modelId) {
const id = String(modelId).toLowerCase();
let best = null;
for (const family of FAMILY_BEST.keys()) {
if (id.includes(family) && (best === null || family.length > best.length)) best = family;
}
return best;
}
export function benchmarkScore(modelId) {
const family = benchmarkFamily(modelId);
return family === null ? null : FAMILY_BEST.get(family).score;
}
/**
* Candidate models per provider, in each CLI's own order. Ranking happens in
* modelsForProvider(); this list only decides what is on offer.
*/
const CANDIDATES = {
// Cursor routes every family in the snapshot. bb's primary variants pin
// medium effort, so these are the snapshot's top five at medium thinking.
// Opus 5 and Grok 4.6 are newer than the snapshot: they are offered as
// unscored candidates rather than ranked on a score they do not have.
cursor: [
{ id: "claude-fable-5-thinking-medium", label: "Claude Fable 5 (thinking)" },
{ id: "gpt-5.6-sol-medium", label: "GPT-5.6 Sol" },
{ id: "gpt-5.6-terra-medium", label: "GPT-5.6 Terra" },
{ id: "claude-opus-4-8-thinking-medium", label: "Claude Opus 4.8 (thinking)" },
{ id: "claude-sonnet-5-thinking-medium", label: "Claude Sonnet 5 (thinking)" },
{ id: "claude-opus-5-thinking-medium", label: "Claude Opus 5 (thinking)" },
{ id: "cursor-grok-4.6-medium", label: "Cursor Grok 4.6" },
],
// bb's curated Claude Code catalog, exactly five rows. Opus 5 is bb's default
// but is not in the snapshot, so it cannot be ranked above Opus 4.8 here.
claude: [
{ id: "claude-fable-5", label: "Fable 5" },
{ id: "claude-opus-5[1m]", label: "Opus 5 (1M)" },
{ id: "claude-opus-4-8[1m]", label: "Opus 4.8 (1M)" },
{ id: "claude-sonnet-5", label: "Sonnet 5" },
{ id: "claude-opus-4-7[1m]", label: "Opus 4.7 (1M)" },
],
// Only four OpenAI families are scored. A fifth is added from the live
// `model/list` when one is available — never invented.
codex: [
{ id: "gpt-5.6-sol", label: "GPT-5.6 Sol" },
{ id: "gpt-5.6-terra", label: "GPT-5.6 Terra" },
{ id: "gpt-5.6-luna", label: "GPT-5.6 Luna" },
{ id: "gpt-5.5", label: "GPT-5.5" },
],
};
export const MAX_RECOMMENDATIONS = 5;
/**
* The score ranks the *family*, not this exact id. The snapshot scores
* family x effort pairs and FAMILY_BEST keeps each family's best, so saying
* "@ max" next to an id that pins medium would misread as this variant's own
* score.
*/
function describe(model) {
if (model.score === null) return "not in the CursorBench 3.2 snapshot";
return `CursorBench ${model.score} — best for ${model.family}`;
}
function annotate(candidate, order) {
const family = benchmarkFamily(candidate.id);
const best = family === null ? null : FAMILY_BEST.get(family);
const model = {
id: candidate.id,
label: candidate.label,
family,
score: best === null ? null : best.score,
reasoning: best === null ? null : best.reasoning,
order,
};
return { ...model, detail: describe(model) };
}
function byIntelligence(a, b) {
if (a.score === null && b.score === null) return a.order - b.order;
if (a.score === null) return 1;
if (b.score === null) return -1;
return b.score - a.score || a.order - b.order;
}
/**
* The provider's recommendations, most intelligent first.
*
* `live` is what the CLI itself reports, when makefaster could ask. Given one,
* the catalog is reconciled against it rather than trusted blindly: ids the CLI
* does not list are dropped (an account without Fable access should not be
* offered Fable), and for a provider with fewer than five scored families the
* remaining slots are filled from the live list. Without one, the static
* catalog stands.
*
* @param {"cursor"|"claude"|"codex"} providerKey
* @param {{live?: Array<{id: string, displayName?: string}>|null}} [options]
* @returns {Array<{id: string, label: string, score: number|null, family: string|null, reasoning: string|null, detail: string}>}
*/
export function modelsForProvider(providerKey, { live = null } = {}) {
const candidates = CANDIDATES[providerKey] || [];
if (candidates.length === 0) return [];
const liveIds = Array.isArray(live) && live.length > 0 ? new Map(live.map((entry) => [entry.id.toLowerCase(), entry])) : null;
let offered = candidates.map(annotate);
if (liveIds) {
offered = offered.filter((model) => liveIds.has(model.id.toLowerCase()));
// Fill from the live list only when the curated candidates cannot fill five,
// preferring scored families so a fill is still an intelligence ranking.
if (offered.length < MAX_RECOMMENDATIONS) {
const known = new Set(offered.map((model) => model.id.toLowerCase()));
const extras = [...liveIds.values()]
.filter((entry) => !known.has(entry.id.toLowerCase()))
.map((entry, index) => annotate({ id: entry.id, label: entry.displayName || entry.id }, candidates.length + index))
.sort(byIntelligence);
offered = [...offered, ...extras.slice(0, MAX_RECOMMENDATIONS - offered.length)];
}
}
return offered
.sort(byIntelligence)
.slice(0, MAX_RECOMMENDATIONS)
.map(({ order, ...model }) => model);
}
/**
* The picker's rows: the model name and nothing else. The id slug, the
* CursorBench score, and the "best for" copy stay out of the list — the row a
* user scans is the name, and the id is confirmed right after the choice.
*
* @param {Array<{label: string}>} models
* @returns {Array<{label: string}>}
*/
export function modelPickerOptions(models) {
return models.map((model) => ({ label: model.label }));
}
/** The top-intelligence model for a provider — the picker's default highlight. */
export function defaultModelFor(providerKey, options) {
return modelsForProvider(providerKey, options)[0] ?? null;
}
/**
* Resolve a `--model` value. A catalog id (case-insensitive) comes back with its
* label and score; anything else passes through untouched, so a model released
* after this snapshot still works.
*/
export function resolveModel(providerKey, modelId, options) {
const wanted = String(modelId).trim();
if (wanted === "") return null;
const known = modelsForProvider(providerKey, options).find((model) => model.id.toLowerCase() === wanted.toLowerCase());
if (known) return known;
const { order, ...model } = annotate({ id: wanted, label: wanted }, 0);
return { ...model, passthrough: true };
}
/** Parse `cursor-agent --list-models` output: one `id - Display Name` per line. */
export function parseCursorModelList(output) {
const models = [];
for (const line of String(output).split(/\r?\n/)) {
const match = /^\s*([A-Za-z0-9][\w.\-[\]]*)\s+-\s+(.+?)\s*$/.exec(line);
if (!match) continue;
const [, id, displayName] = match;
if (id === "auto") continue; // a router, not a model to rank
models.push({ id, displayName });
}
return models;
}