Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
20 changes: 12 additions & 8 deletions src/agent/optimize.ts
Original file line number Diff line number Diff line change
Expand Up @@ -53,14 +53,18 @@ const MODEL_MAX_OUTPUT: Record<string, number> = {
'openai/gpt-5.6-sol': 128_000,
'openai/gpt-5.6-terra': 128_000,
'openai/gpt-5.6-luna': 128_000,
// gpt-5.5 / gpt-5.4 stay at 32768. The catalog claims 128000 for both and
// that claim is UNREFUTED — an earlier attempt to verify it was invalid:
// both SDKs reject max_tokens > 100000 client-side (blockrun_llm
// validation.py:233, blockrun-llm-ts validation.ts:74), so the "rejections"
// collected at 128000 never reached a provider. Their real ceilings are
// unknown. Raising them needs a probe path that bypasses the SDK guard.
'openai/gpt-5.5': 32_768,
'openai/gpt-5.4': 32_768,
// 128000 is measured, not read off the catalog. The earlier probe that
// seemed to refute it was invalid — both SDKs rejected max_tokens > 100000
// client-side, so nothing reached a provider. Re-probed 2026-07-21 with that
// guard bypassed: both accept 128000, as did every model advertising a
// ceiling above 100000. The catalog was right; the measurement was broken.
//
// Franklin does not run that SDK guard on either request path — the agent
// loop and the proxy both use raw fetch and import only payment helpers from
// @blockrun/llm — so these values are independent of whether the SDK-side
// fix lands. (blockrun-llm#27 / blockrun-llm-ts#15, open at time of writing.)
'openai/gpt-5.5': 128_000,
'openai/gpt-5.4': 128_000,
'openai/gpt-5.4-mini': 128_000,
'openai/gpt-5.4-nano': 32_768,
// Probed accepted at 65536 through the live gateway (2026-07-21) — a real
Expand Down
25 changes: 13 additions & 12 deletions test/local.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -6710,21 +6710,22 @@ test('anthropic max output matches the documented ceilings', async () => {
assert.equal(getMaxOutputTokens('anthropic/claude-sonnet-4.6'), 128_000);
});

test('openai ceilings: one probed, two still unverified', async () => {
// gpt-5-mini was raised to 65536 on a real accepted request through the live
// gateway (2026-07-21). It is a FLOOR, not a proven ceiling — 65536 is also
// ESCALATED_MAX_TOKENS, so nothing here could tell a higher true limit apart.
test('openai ceilings are measured, not read off the catalog', async () => {
// All three are now probed against the live gateway.
//
// gpt-5.5 / gpt-5.4 stay at 32768. The catalog claims 128000 and that claim
// is UNREFUTED. An attempt to verify it was invalid: both SDKs reject
// max_tokens > 100000 client-side (blockrun_llm validation.py:233,
// blockrun-llm-ts validation.ts:74), so probes at 128000 never reached a
// provider — the "maximum: 100000" in those errors is an SDK constant, not
// an upstream response. Raising these needs a probe that bypasses the guard.
// gpt-5-mini 65536 accepted (probe was below the SDK guard, so it went out)
// gpt-5.5 128000 accepted, 2026-07-21, guard bypassed
// gpt-5.4 128000 accepted, same run
//
// An earlier pass recorded these as REJECTED at 128000 and concluded the
// catalog was wrong. That was measurement error: both SDKs threw on
// max_tokens > 100000 client-side, so 19 "rejections" never left the
// process. The uniform limit across three different providers was the tell.
// Fixed upstream in blockrun-llm#27 and blockrun-llm-ts#15.
const { getMaxOutputTokens } = await import('../dist/agent/optimize.js');
assert.equal(getMaxOutputTokens('openai/gpt-5.5'), 128_000);
assert.equal(getMaxOutputTokens('openai/gpt-5.4'), 128_000);
assert.equal(getMaxOutputTokens('openai/gpt-5-mini'), 65_536);
assert.equal(getMaxOutputTokens('openai/gpt-5.5'), 32_768);
assert.equal(getMaxOutputTokens('openai/gpt-5.4'), 32_768);
});

// ─── gateway catalog as a GAP-FILLER, never an override ───────────────────
Expand Down
Loading