fix(mcplocal): learn which models reject sampling params #128

Merged
michal merged 1 commits from deploy/bounded-failures into main 2026-08-25 23:12:08 +00:00
2 changed files with 101 additions and 3 deletions
Showing only changes of commit bc7eb5a0ad - Show all commits

View File

@@ -30,6 +30,15 @@ function familyOf(model: string): string | null {
return null;
}
/**
* A 400 naming a sampling parameter, e.g.
* "`temperature` is deprecated for this model."
*/
function isSamplingRejection(err: unknown): boolean {
const msg = err instanceof Error ? err.message : String(err);
return msg.includes('HTTP 400') && /`?(temperature|top_p|top_k)`?/.test(msg);
}
/** Resolutions are cached this long before the models endpoint is consulted again. */
const MODEL_CACHE_TTL_MS = Number(process.env['MCPCTL_ANTHROPIC_MODEL_TTL_MS']) || 12 * 60 * 60 * 1000;
@@ -40,6 +49,17 @@ export class AnthropicProvider implements LlmProvider {
readonly name = 'anthropic';
/** Shared across instances: the model list is account-wide, not per-provider. */
private static readonly modelCache = new Map<string, { id: string; expiresAt: number }>();
/**
* Models that rejected a sampling parameter, learned at runtime.
*
* Anthropic removed `temperature`/`top_p`/`top_k` on the current generation
* (Opus 5, Sonnet 5, Opus 4.7/4.8, Fable 5) — sending one is a hard 400. A
* hardcoded list of which models accept it would rot exactly the way the
* pinned model ids did, so learn it from the API instead: the first call
* retries without the parameter and remembers, and every later call omits it
* up front.
*/
private static readonly rejectsSampling = new Set<string>();
private apiKey: string;
private defaultModel: string;
@@ -66,7 +86,9 @@ export class AnthropicProvider implements LlmProvider {
if (systemMessages.length > 0) {
body.system = systemMessages.map((m) => m.content).join('\n');
}
if (options.temperature !== undefined) body.temperature = options.temperature;
if (options.temperature !== undefined && !AnthropicProvider.rejectsSampling.has(model)) {
body.temperature = options.temperature;
}
if (options.tools && options.tools.length > 0) {
body.tools = options.tools.map((t) => ({
@@ -76,8 +98,19 @@ export class AnthropicProvider implements LlmProvider {
}));
}
const response = await this.request(body, options.signal);
return parseAnthropicResponse(response);
try {
return parseAnthropicResponse(await this.request(body, options.signal));
} catch (err) {
// Learn-and-retry rather than fail: a model that has dropped sampling
// support should cost one wasted call, once, not every call forever.
if (body.temperature !== undefined && isSamplingRejection(err)) {
AnthropicProvider.rejectsSampling.add(model);
delete body.temperature;
process.stderr.write(`[anthropic] ${model} rejects sampling params — retrying without\n`);
return parseAnthropicResponse(await this.request(body, options.signal));
}
throw err;
}
}
/**

View File

@@ -81,3 +81,68 @@ describe('Anthropic model resolution', () => {
await expect(providerWith(MODELS).listModels()).resolves.toContain('claude-opus-5');
});
});
describe('sampling-parameter rejection', () => {
afterEach(() => {
(AnthropicProvider as unknown as { rejectsSampling: Set<string> }).rejectsSampling.clear();
});
/** Rejects `temperature` exactly as the current Anthropic models do. */
function samplingStrictProvider(): { provider: AnthropicProvider; bodies: Array<Record<string, unknown>> } {
const provider = new AnthropicProvider({ apiKey: 'sk-ant-api-test' });
const bodies: Array<Record<string, unknown>> = [];
(provider as unknown as { request: (b: unknown) => Promise<unknown> }).request = async (b) => {
const body = b as Record<string, unknown>;
bodies.push({ ...body });
if (body.temperature !== undefined) {
throw new Error(
'Anthropic HTTP 400: {"type":"error","error":{"type":"invalid_request_error",'
+ '"message":"`temperature` is deprecated for this model."}}',
);
}
return { content: [{ type: 'text', text: 'ok' }], stop_reason: 'end_turn' };
};
return { provider, bodies };
}
it('retries without temperature when the model rejects it', async () => {
vi.spyOn(process.stderr, 'write').mockImplementation(() => true);
const { provider, bodies } = samplingStrictProvider();
const result = await provider.complete({
model: 'claude-opus-5',
messages: [{ role: 'user', content: 'hi' }],
temperature: 0,
});
expect(result.content).toBe('ok');
expect(bodies).toHaveLength(2);
expect(bodies[0]!.temperature).toBe(0);
expect(bodies[1]!.temperature).toBeUndefined();
});
it('remembers, so the wasted call happens once and not forever', async () => {
vi.spyOn(process.stderr, 'write').mockImplementation(() => true);
const { provider, bodies } = samplingStrictProvider();
const req = { model: 'claude-opus-5', messages: [{ role: 'user' as const, content: 'hi' }], temperature: 0 };
await provider.complete(req);
await provider.complete(req);
await provider.complete(req);
// 2 for the first call (reject + retry), then 1 each — not 2 each.
expect(bodies).toHaveLength(4);
});
it('does not swallow unrelated 400s', async () => {
const provider = new AnthropicProvider({ apiKey: 'sk-ant-api-test' });
(provider as unknown as { request: () => Promise<unknown> }).request = async () => {
throw new Error('Anthropic HTTP 400: {"error":{"message":"max_tokens is required"}}');
};
await expect(provider.complete({
model: 'claude-opus-5',
messages: [{ role: 'user', content: 'hi' }],
temperature: 0,
})).rejects.toThrow(/max_tokens/);
});
});