Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -118,6 +118,7 @@ export function AddModelDialog(props: {
copy={copy}
modelId={trimmedId}
customDefaultApiProtocol={props.defaultApiProtocol}
providerType={props.providerType}
declared={profile}
limitsConflict={limitsConflict}
onChange={(patch) => setProfile((current) => ({ ...current, ...patch }))}
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -24,7 +24,9 @@ import {
MODEL_API_PROTOCOL_LABELS,
MODEL_API_PROTOCOLS,
type ModelApiProtocol,
type ProviderType,
} from '@maka/core/llm-connections';
import { providerAcceptsOutputTokenLimit } from '@maka/core/provider-registry';
import {
DECLARABLE_RELAY_THINKING_LEVELS,
THINKING_LEVELS,
Expand All @@ -42,6 +44,8 @@ export function CapabilityEditor(props: {
/** Set only on a custom connection, which alone takes wire and thinking declarations. */
customDefaultApiProtocol?: ModelApiProtocol;
declared: ModelOverride | undefined;
/** The connection's provider; one that rejects any output-token limit gets a read-only field. */
providerType: ProviderType;
contextWindowInput: string;
contextWindowInputInvalid: boolean;
numericInputs?: Partial<Record<'inputLimit' | 'compactionThreshold' | 'maxOutputTokens', string>>;
Expand Down Expand Up @@ -182,6 +186,8 @@ export function CapabilityEditor(props: {
{(['inputLimit', 'compactionThreshold', 'maxOutputTokens'] as const).map((field) => {
const input = props.numericInputs?.[field] ?? String(declared?.[field] ?? '');
const invalid = input.trim() !== '' && parseContextWindowInput(input) === null;
const unsupported =
field === 'maxOutputTokens' && !providerAcceptsOutputTokenLimit(props.providerType);
return (
<TextInput
size="sm"
Expand All @@ -191,11 +197,14 @@ export function CapabilityEditor(props: {
labelTooltip={copy[`${field}Help`]}
value={input}
onChange={(value) => props.onNumericInput(field, value)}
isDisabled={props.disabled}
isDisabled={props.disabled || unsupported}
{...(unsupported ? { disabledMessage: copy.maxOutputTokensUnsupported } : {})}
hasClear
placeholder={field === 'inputLimit' && props.defaultInputLimit !== undefined ? String(props.defaultInputLimit) : field === 'maxOutputTokens' ? '8192 / 8K' : '128000 / 128K / 1M'}
status={
invalid ? { type: 'error', message: copy.contextWindowInputInvalid } : field === 'inputLimit' && props.limitsConflict ? { type: 'error', message: copy.modelLimitsConflict } : undefined
unsupported && input.trim() !== ''
? { type: 'warning', message: copy.maxOutputTokensUnsupported }
: invalid ? { type: 'error', message: copy.contextWindowInputInvalid } : field === 'inputLimit' && props.limitsConflict ? { type: 'error', message: copy.modelLimitsConflict } : undefined
}
/>
);
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -69,6 +69,7 @@ const zhCapabilitiesCopy = {
compactionThresholdHelp: '达到此 token 数时压缩上下文。留空则不主动压缩。',
maxOutputTokens: '输出上限',
maxOutputTokensHelp: '单次回复的输出 token 上限,含思考。留空自动设置。',
maxOutputTokensUnsupported: 'ChatGPT 订阅(Codex)不接受输出上限,Maka 不会发送此设置。',
fastMode: 'Fast 模式',
fastModeHelp: '选择更快的服务档位,可能产生额外费用。',
fastAuto: '自动',
Expand Down Expand Up @@ -111,6 +112,7 @@ const zhTwCapabilitiesCopy = {
compactionThresholdHelp: '達到此 token 數時壓縮上下文。留空則不主動壓縮。',
maxOutputTokens: '輸出上限',
maxOutputTokensHelp: '單次回覆的輸出 token 上限,含思考。留空自動設定。',
maxOutputTokensUnsupported: 'ChatGPT 訂閱(Codex)不接受輸出上限,Maka 不會送出此設定。',
fastMode: 'Fast 模式',
fastModeHelp: '選擇更快的服務檔位,可能產生額外費用。',
fastAuto: '自動',
Expand Down Expand Up @@ -153,6 +155,7 @@ const enCapabilitiesCopy = {
compactionThresholdHelp: 'Compact at this token count. Leave empty to disable proactive compaction.',
maxOutputTokens: 'Maximum output',
maxOutputTokensHelp: 'Output token budget per reply, including thinking. Leave empty for automatic limits.',
maxOutputTokensUnsupported: 'The ChatGPT subscription (Codex) does not accept an output limit, so Maka does not send this setting.',
fastMode: 'Fast mode',
fastModeHelp: 'Use the faster service tier. Additional charges may apply.',
fastAuto: 'Auto',
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -714,6 +714,7 @@ function ConnectionDetailInner(props: ConnectionDetailProps) {
copy={copy}
modelId={editingModelId}
customDefaultApiProtocol={connection.defaultApiProtocol}
providerType={connection.providerType}
numericInputs={numericInputs}
onNumericInput={(field, input) => {
setEditingRow((current) => ({ ...(typeof current === 'object' && current ? current : {}), model: editingModelId, numericInputs: { ...numericInputs, [field]: input } }));
Expand Down
13 changes: 13 additions & 0 deletions packages/core/src/__tests__/provider-catalog-contract.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -32,6 +32,7 @@ import {
CATALOG_PROVIDER_TYPES,
PROVIDER_REGISTRY,
isRetiredProvider,
providerAcceptsOutputTokenLimit,
providerFallbackModelIds,
} from '../provider-registry.js';
import { buildConnectionModelCatalogEntries } from '../model-catalog.js';
Expand Down Expand Up @@ -331,3 +332,15 @@ describe('provider catalog contract — fallback lifecycle', () => {
assert.deepEqual(regressed, []);
});
});

describe('providerAcceptsOutputTokenLimit', () => {
it('refuses an output limit only where the backend rejects max_output_tokens', () => {
// The ChatGPT Codex backend answers max_output_tokens with HTTP 400.
assert.equal(providerAcceptsOutputTokenLimit('openai-codex'), false);
for (const providerType of ['openai', 'anthropic', 'google', 'kimi-coding-plan'] as const) {
assert.equal(providerAcceptsOutputTokenLimit(providerType), true, providerType);
}
// An unknown provider keeps today's behaviour.
assert.equal(providerAcceptsOutputTokenLimit('not-a-provider'), true);
});
});
10 changes: 10 additions & 0 deletions packages/core/src/provider-registry.ts
Original file line number Diff line number Diff line change
Expand Up @@ -1720,5 +1720,15 @@ export function isRetiredProvider(providerType: string): boolean {
return providerDefaultsOf(providerType)?.retired === true;
}

/**
* Whether a request on this provider may carry an output-token limit. The
* ChatGPT Codex backend answers `max_output_tokens` with HTTP 400
* "Unsupported parameter", so no limit can be honoured there: Runtime must not
* send one, and settings must not offer one.
*/
export function providerAcceptsOutputTokenLimit(providerType: string): boolean {
return providerDefaultsOf(providerType)?.runtimeAdapter.kind !== 'openai-codex';
}

export const CATALOG_PROVIDER_TYPES = providerTypesByOrder('catalogOrder');
export const RECOMMENDED_PROVIDER_TYPES = providerTypesByOrder('recommendedOrder');
87 changes: 87 additions & 0 deletions packages/runtime/src/__tests__/model-adapter.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -19,7 +19,9 @@

import assert from 'node:assert/strict';
import { describe, test } from 'node:test';
import type { LanguageModelV4StreamPart } from '@ai-sdk/provider';
import { RetryError } from 'ai';
import { convertArrayToReadableStream, MockLanguageModelV4 } from 'ai/test';

import { ModelAdapter, normalizeAiSdkUsage } from '../model-adapter.js';
import type { ModelStreamEvent } from '../model-protocol.js';
Expand Down Expand Up @@ -52,6 +54,91 @@ describe('ModelAdapter stream and error normalization', () => {
assert.equal(adapter.maxOutputTokensForInput(192_000), 8_000);
});

test('sends no output limit on a Codex subscription, even a configured one', () => {
const adapterFor = (providerType: 'openai' | 'openai-codex') =>
new ModelAdapter({
connection: {
slug: providerType,
providerType,
defaultModel: 'gpt-6-astra',
modelOverrides: { 'gpt-6-astra': { maxOutputTokens: 4_096 } },
},
apiKey: 'token',
modelId: 'gpt-6-astra',
modelFactory: () => ({}),
newId: idGenerator(),
now: monotonicClock(),
});

// The Codex backend rejects max_output_tokens, so a turn carrying the
// configured limit would fail with HTTP 400 on every request.
const codex = adapterFor('openai-codex');
assert.equal(codex.acceptsOutputTokenLimit(), false);
assert.equal(codex.maxOutputTokens(), undefined);
assert.equal(codex.maxOutputTokensForInput(100_000), undefined);

const openai = adapterFor('openai');
assert.equal(openai.acceptsOutputTokenLimit(), true);
assert.equal(openai.maxOutputTokens(), 4_096);
});

test('startStream sends a Codex subscription request no output limit from any source', async () => {
const sentLimit = async (
providerType: 'openai' | 'openai-codex',
callerLimit: number | undefined,
) => {
const model = new MockLanguageModelV4({
doStream: {
stream: convertArrayToReadableStream([
{ type: 'stream-start', warnings: [] },
{
type: 'finish',
finishReason: { unified: 'stop', raw: 'stop' },
usage: {
inputTokens: { total: 1, noCache: 1, cacheRead: 0, cacheWrite: 0 },
outputTokens: { total: 1, text: 1, reasoning: 0 },
},
},
] satisfies LanguageModelV4StreamPart[]),
},
});
const adapter = new ModelAdapter({
connection: {
slug: providerType,
providerType,
defaultModel: 'gpt-6-astra',
modelOverrides: { 'gpt-6-astra': { maxOutputTokens: 4_096 } },
},
apiKey: 'token',
modelId: 'gpt-6-astra',
modelFactory: () => model,
newId: idGenerator(),
now: monotonicClock(),
});
const result = await adapter.startStream({
model,
messages: [{ role: 'user', content: 'hi' }],
tools: {},
activeTools: [],
abortSignal: new AbortController().signal,
repairToolCall: async () => null,
onStreamActivity: () => {},
...(callerLimit === undefined ? {} : { maxOutputTokens: callerLimit }),
});
for await (const _event of result.events) {
// Drain so the provider call settles.
}
return model.doStreamCalls[0]?.maxOutputTokens;
};

// The configured limit, and a caller-supplied cap such as overflow
// recovery's 8K, would both be rejected by the Codex backend with HTTP 400.
assert.equal(await sentLimit('openai-codex', undefined), undefined);
assert.equal(await sentLimit('openai-codex', 8_000), undefined);
assert.equal(await sentLimit('openai', undefined), 4_096);
assert.equal(await sentLimit('openai', 8_000), 8_000);
});

test('forwards the stable Session identity to the model factory', () => {
let observedSessionId: string | undefined;
const model = {};
Expand Down
18 changes: 18 additions & 0 deletions packages/runtime/src/__tests__/overflow-reactive-recovery.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -1258,6 +1258,24 @@ describe('reactive overflow recovery in the streaming backend', () => {
assert.equal(fixture.model.doStreamCalls[2]?.maxOutputTokens, 8_000);
});

test('sends no output cap on a Codex subscription retry after overflow recovery', async () => {
// Elsewhere recovery caps the retry at 8K; the Codex backend rejects any
// max_output_tokens, so the recovered request would fail with HTTP 400.
const fixture = buildReactiveFixture({
script: ['tool', 'overflow', 'done'],
bigPriors: true,
modelMaxOutputTokens: 128_000,
providerNative: true,
});
await runTurn(fixture);

assert.equal(complete(fixture)?.stopReason, 'end_turn');
assert.equal(fixture.model.doStreamCalls.length, 3);
for (const call of fixture.model.doStreamCalls) {
assert.equal(call?.maxOutputTokens, undefined);
}
});

test('drops hydrated images before the single overflow retry without rerunning tools', async () => {
const fixture = buildReactiveFixture({
script: ['tool', 'overflow', 'done'],
Expand Down
31 changes: 23 additions & 8 deletions packages/runtime/src/model-adapter.ts
Original file line number Diff line number Diff line change
Expand Up @@ -25,6 +25,7 @@ import {
type RuntimeExecutionConnection,
} from '@maka/core/llm-connections';
import { lookupModelMetadata } from '@maka/core/model-metadata';
import { providerAcceptsOutputTokenLimit } from '@maka/core/provider-registry';
import type { CacheMissInputSource } from '@maka/core/usage-stats/types';
import { rawFinishReasonString } from './model-protocol.js';
import type {
Expand Down Expand Up @@ -217,7 +218,17 @@ export class ModelAdapter {
});
}

/**
* Whether a request to this connection may carry an output-token limit at
* all. When it may not, no limit is sent: neither a configured per-model
* limit nor the context-recovery cap.
*/
acceptsOutputTokenLimit(): boolean {
return providerAcceptsOutputTokenLimit(this.input.connection.providerType);
}

maxOutputTokens(): number | undefined {
if (!this.acceptsOutputTokenLimit()) return undefined;

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P1: This guard does not protect the actual streamed request. startStream() at lines 288-295 uses input.maxOutputTokens ?? selectedModelMaxOutputTokens(...) instead of this guarded method. When AiSdkTurn omits the limit for openai-codex, that fallback recalculates the configured override and passes it to streamText (line 382). On this head, a Codex adapter with a 4096 override reports maxOutputTokensForInput() === undefined, but a direct startStream() probe observes doStream.maxOutputTokens === 4096. The streamed Codex request therefore still carries the backend-rejected limit. Gate this fallback as well, and test the final doStream/wire request rather than only the helper return value.

Copy link
Copy Markdown
Member Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Confirmed, thanks. When its caller passed no limit, ModelAdapter.startStream fell back to selectedModelMaxOutputTokens(...) directly. The guard in maxOutputTokens() / maxOutputTokensForInput() never saw that path, so a Codex turn still sent the configured 4096.

Fixed in cad717b. The check now sits at the one place a main-turn limit reaches the wire: startStream sends no maxOutputTokens when acceptsOutputTokenLimit() is false, whether the value came from the caller or from the model configuration. Because of that, overflow recovery needs no special case, so I reverted the ai-sdk-turn.ts change; this PR no longer touches that file.

Tests:

  • New model-adapter case that goes through startStream into a MockLanguageModelV4. It asserts what reaches doStream: on openai-codex with a 4096 override, nothing, both with no caller limit and with a caller-supplied 8000. On openai, 4096 and 8000 respectively.
  • The existing Codex overflow-recovery case now passes through the adapter alone.
  • Both fail when the gate is removed, and pass with it.
  • model-adapter 38/38, overflow-reactive-recovery 60/60, mid-turn-capacity-backend 68/68; runtime tsc passes.

On the red CI: I can't read the job log from here (the log download is blocked on my network), so thanks for naming opencli-chrome.test.js. That test covers the Windows Chrome Web Store launch path, which this PR does not touch. Other unrelated PRs went red around the same time, so I'll see whether it reproduces on this push.

On #5723: agreed with your reading. Neither PR alone fixes both call paths.

return selectedModelMaxOutputTokens(
this.input.connection,
this.input.modelId,
Expand Down Expand Up @@ -274,14 +285,18 @@ export class ModelAdapter {
wrapLanguageModel: (input: Record<string, unknown>) => unknown;
};

const maxOutputTokens =
input.maxOutputTokens ??
selectedModelMaxOutputTokens(
this.input.connection,
this.input.modelId,
this.input.providerOptions,
this.runtime,
);
// The one place a main-turn output limit reaches the wire. A provider that
// rejects any limit gets none, whether it came from the caller (overflow
// recovery, a resumed request) or from the configured model limit.
const maxOutputTokens = this.acceptsOutputTokenLimit()
? (input.maxOutputTokens ??
selectedModelMaxOutputTokens(
this.input.connection,
this.input.modelId,
this.input.providerOptions,
this.runtime,
))
: undefined;
let settleAccounting: ((outcome: ModelStepOutcome) => Promise<void>) | undefined;
const terminalModel = withProviderFinishBoundary(input.model, wrapLanguageModel);
const trackedModel = input.providerRequestTracker
Expand Down