diff --git a/docs/src/content/docs/docs/QuickAddAPI.md b/docs/src/content/docs/docs/QuickAddAPI.md index 7ac0ad4d2..6c6d06431 100644 --- a/docs/src/content/docs/docs/QuickAddAPI.md +++ b/docs/src/content/docs/docs/QuickAddAPI.md @@ -695,6 +695,9 @@ const { text, steps, toolCalls } = await agent.generate({ - `system` - system prompt (defaults to your AI Assistant default system prompt). - `tools` - an object map of tool name → tool (from `ai.tool()` and/or `ai.tools.*`). - `toolChoice` - `"auto"` (default) | `"none"` | `"required"` | `{ type: "tool", toolName }`. + Some models can't be forced to call a tool: Claude Opus 5.5 and Claude Fable 5.1 reject + `"required"` and `{ type: "tool", toolName }`, and QuickAdd's error says so. Use + `"auto"` and say in the prompt when the tool applies, or pass a `schema` for a fixed JSON shape. - `stopWhen` - one or more stop conditions from `ai.stepCountIs(n)` / `ai.hasToolCall(name)`. - `maxSteps` - step budget (default 20, hard cap 100). Sugar for `stopWhen: ai.stepCountIs(n)`. - `maxOutputTokens`, `modelOptions` - passed to the provider. @@ -788,15 +791,16 @@ GPT-4o-class), Anthropic Claude 4.x, and Gemini 3.x; it can be combined with too that do not support schema-constrained output (e.g. legacy OpenAI chat models) reject the request outright with a provider error - use a current model rather than expecting a best-effort fallback. -:::note[OpenAI reasoning models (GPT-5.x, o-series)] -These accept only the default `temperature` (omit it from `modelOptions`), and QuickAdd -automatically sends `maxOutputTokens` as `max_completion_tokens` for them. The agent's default -path sets neither, so `quickAddApi.ai.agent({ model: "gpt-5" })` works as-is. +:::note[OpenAI reasoning models (GPT-5.x, GPT-6, o-series)] +These accept only the default `temperature` (omit it from `modelOptions`). The agent sends no +`temperature` unless you set one, so `quickAddApi.ai.agent({ model: "gpt-6-luna" })` works as-is. -GPT-5.6 and GPT-6 models reason by default, and OpenAI's Chat Completions API rejects function -tools for them unless reasoning is off. When a tool turn is rejected for that reason, QuickAdd -retries it once with `reasoning_effort: "none"`. If you set `reasoning_effort` yourself in -`modelOptions`, QuickAdd keeps it and shows the provider's error instead. +Agent turns to OpenAI's own API (`https://api.openai.com/v1`) use the Responses API, which lets +reasoning models such as GPT-5.6 and GPT-6 call tools with reasoning on. Other OpenAI-compatible +providers use Chat Completions. `modelOptions` keep their Chat Completions names either way: +QuickAdd sends `reasoning_effort` as `reasoning.effort` and `max_tokens` as `max_output_tokens` +on the Responses API, and sends `maxOutputTokens` as `max_completion_tokens` to reasoning models +on Chat Completions. ::: ### `getModels(): string[]` diff --git a/src/ai/OpenAIRequest.sampling.test.ts b/src/ai/OpenAIRequest.sampling.test.ts index c91707d03..ae32a4910 100644 --- a/src/ai/OpenAIRequest.sampling.test.ts +++ b/src/ai/OpenAIRequest.sampling.test.ts @@ -266,7 +266,18 @@ describe("sampling parameter recovery (chat/tool path)", () => { getModelProviderMock.mockReturnValue(openaiProvider); requestUrlMock .mockReturnValueOnce(Promise.resolve(unsupportedParamFailure("top_p"))) - .mockReturnValueOnce(Promise.resolve(openaiSuccess("chat ok"))); + // Chat turns on api.openai.com use the Responses API. + .mockReturnValueOnce( + Promise.resolve({ + status: 200, + json: Promise.resolve({ + id: "resp_1", + status: "completed", + output: [{ type: "message", content: [{ type: "output_text", text: "chat ok" }] }], + usage: { input_tokens: 1, output_tokens: 1, total_tokens: 2 }, + }), + }), + ); const model: Model = { name: "o4-mini", maxTokens: 200000 }; const res = await chatRequest(makeApp(), "sk", model, currentProvider(), { diff --git a/src/ai/OpenAIRequest.toolReasoning.test.ts b/src/ai/OpenAIRequest.toolReasoning.test.ts deleted file mode 100644 index f5c476d96..000000000 --- a/src/ai/OpenAIRequest.toolReasoning.test.ts +++ /dev/null @@ -1,166 +0,0 @@ -import { beforeEach, describe, expect, it } from "vitest"; -import type { AIProvider, Model } from "./Provider"; -import type { NormalizedChatRequest } from "./tools/NormalizedTools"; - -import { storeState, mocks, makeApp } from "../../tests/helpers/ai/requestHarness"; - -const { requestUrlMock, noticeMock } = mocks; - -const { chatRequest } = await import("./OpenAIRequest"); - -const openaiProvider: AIProvider = { - name: "OpenAI", - endpoint: "https://api.openai.com/v1", - kind: "openai", - apiKey: "sk", - models: [], - modelSource: "modelsDev", -}; - -const gpt6: Model = { - name: "gpt-6-sol", - maxTokens: 1_050_000, - maxOutputTokens: 128_000, - supportsTemperature: false, -}; - -// Exact live error for gpt-6-sol with function tools on /v1/chat/completions -// (2026-09-26); gpt-5.6-* and the other gpt-6-* models return the same text. -function toolsWhileReasoningFailure(model = "gpt-6-sol") { - return { - status: 400, - json: { - error: { - message: `Function tools with reasoning_effort are not supported for ${model} in /v1/chat/completions. To use function tools, use /v1/responses or set reasoning_effort to 'none'.`, - type: "invalid_request_error", - param: null, - code: null, - }, - }, - }; -} - -function toolCallSuccess() { - return { - status: 200, - json: Promise.resolve({ - id: "1", - model: "gpt-6-sol", - choices: [ - { - finish_reason: "tool_calls", - index: 0, - message: { - role: "assistant", - content: null, - tool_calls: [ - { - id: "call_1", - type: "function", - function: { name: "get_weather", arguments: '{"city":"Paris"}' }, - }, - ], - }, - }, - ], - usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 }, - created: 0, - }), - }; -} - -function toolRequest( - modelParams: Record = {}, -): NormalizedChatRequest { - return { - messages: [{ role: "user", content: "Weather in Paris?" }], - modelParams, - tools: [ - { - name: "get_weather", - description: "Get weather", - parameters: { - type: "object", - properties: { city: { type: "string" } }, - required: ["city"], - }, - }, - ], - toolChoice: "auto", - }; -} - -function sentBody(callIndex: number): Record { - return JSON.parse(requestUrlMock.mock.calls[callIndex][0].body as string); -} - -beforeEach(() => { - requestUrlMock.mockReset(); - noticeMock.mockReset(); - storeState.disableOnlineFeatures = false; -}); - -describe("function tools on models that reason by default", () => { - it("retries once with reasoning_effort 'none' when the model rejects tools while reasoning", async () => { - requestUrlMock - .mockReturnValueOnce(Promise.resolve(toolsWhileReasoningFailure())) - .mockReturnValueOnce(Promise.resolve(toolCallSuccess())); - - const res = await chatRequest(makeApp(), "sk", gpt6, openaiProvider, toolRequest()); - - expect(res.toolCalls?.map((call) => call.name)).toEqual(["get_weather"]); - expect(requestUrlMock).toHaveBeenCalledTimes(2); - expect(sentBody(0).reasoning_effort).toBeUndefined(); - expect(sentBody(1).reasoning_effort).toBe("none"); - expect(sentBody(1).tools).toEqual(sentBody(0).tools); - }); - - it("keeps a reasoning effort the caller chose and surfaces the error", async () => { - requestUrlMock.mockReturnValueOnce( - Promise.resolve(toolsWhileReasoningFailure()), - ); - - await expect( - chatRequest( - makeApp(), - "sk", - gpt6, - openaiProvider, - toolRequest({ reasoning_effort: "high" }), - ), - ).rejects.toThrow(/reasoning_effort/); - expect(requestUrlMock).toHaveBeenCalledTimes(1); - }); - - it("does not retry other 400s", async () => { - requestUrlMock.mockReturnValueOnce( - Promise.resolve({ - status: 400, - json: { - error: { - message: "Invalid schema for function 'get_weather'.", - type: "invalid_request_error", - }, - }, - }), - ); - - await expect( - chatRequest(makeApp(), "sk", gpt6, openaiProvider, toolRequest()), - ).rejects.toThrow(/Invalid schema/); - expect(requestUrlMock).toHaveBeenCalledTimes(1); - }); - - it("does not add reasoning_effort to a request without tools", async () => { - requestUrlMock.mockReturnValueOnce( - Promise.resolve(toolsWhileReasoningFailure()), - ); - - await expect( - chatRequest(makeApp(), "sk", gpt6, openaiProvider, { - messages: [{ role: "user", content: "hi" }], - }), - ).rejects.toThrow(); - expect(requestUrlMock).toHaveBeenCalledTimes(1); - }); -}); diff --git a/src/ai/OpenAIRequest.toolTurns.test.ts b/src/ai/OpenAIRequest.toolTurns.test.ts new file mode 100644 index 000000000..8ad421f35 --- /dev/null +++ b/src/ai/OpenAIRequest.toolTurns.test.ts @@ -0,0 +1,239 @@ +import { beforeEach, describe, expect, it } from "vitest"; +import type { AIProvider, Model } from "./Provider"; +import type { NormalizedChatRequest } from "./tools/NormalizedTools"; + +import { storeState, mocks, makeApp } from "../../tests/helpers/ai/requestHarness"; + +const { requestUrlMock, noticeMock } = mocks; + +const { chatRequest } = await import("./OpenAIRequest"); + +function openaiCompatible(endpoint: string, name = "OpenAI"): AIProvider { + return { name, endpoint, kind: "openai", apiKey: "sk", models: [], modelSource: "modelsDev" }; +} + +const gpt6: Model = { + name: "gpt-6-luna", + maxTokens: 1_050_000, + maxOutputTokens: 128_000, + supportsTemperature: false, +}; + +const weatherTool = { + name: "get_weather", + description: "Get weather", + parameters: { + type: "object" as const, + properties: { city: { type: "string" as const } }, + required: ["city"], + }, +}; + +function toolRequest(): NormalizedChatRequest { + return { + messages: [ + { role: "system", content: "Use tools." }, + { role: "user", content: "Weather in Paris?" }, + ], + tools: [weatherTool], + toolChoice: "auto", + }; +} + +// Shape of a live /v1/responses reply from a reasoning model (2026-09-26): +// a reasoning item with encrypted_content, then a function_call whose item +// `id` (fc_…) differs from the `call_id` the result must reference. +const responsesToolCallOutput = [ + { id: "rs_1", type: "reasoning", summary: [], encrypted_content: "opaque-blob" }, + { + id: "fc_1", + type: "function_call", + status: "completed", + call_id: "call_abc", + name: "get_weather", + arguments: '{"city":"Paris"}', + }, +]; + +function ok(json: unknown) { + return Promise.resolve({ status: 200, json: Promise.resolve(json) }); +} + +function sent(callIndex: number): { url: string; body: Record } { + const arg = requestUrlMock.mock.calls[callIndex][0]; + return { url: arg.url as string, body: JSON.parse(arg.body as string) }; +} + +beforeEach(() => { + requestUrlMock.mockReset(); + noticeMock.mockReset(); + storeState.disableOnlineFeatures = false; +}); + +describe("OpenAI tool turns use the Responses API on api.openai.com", () => { + it("sends a tool turn to /v1/responses and returns the call under its call_id", async () => { + requestUrlMock.mockReturnValueOnce( + ok({ + id: "resp_1", + status: "completed", + output: responsesToolCallOutput, + usage: { input_tokens: 10, output_tokens: 5, total_tokens: 15 }, + }), + ); + + const res = await chatRequest( + makeApp(), "sk", gpt6, openaiCompatible("https://api.openai.com/v1"), toolRequest(), + ); + + const { url, body } = sent(0); + expect(url).toBe("https://api.openai.com/v1/responses"); + expect(body.messages).toBeUndefined(); + expect(body.reasoning_effort).toBeUndefined(); + expect(body.store).toBe(false); + expect(body.tools).toEqual([ + { type: "function", name: "get_weather", description: "Get weather", parameters: weatherTool.parameters, strict: false }, + ]); + expect(res.toolCalls).toEqual([ + { id: "call_abc", name: "get_weather", args: { city: "Paris" }, rawArgs: '{"city":"Paris"}' }, + ]); + expect(res.normalizedStopReason).toBe("tool_calls"); + expect(res.usage).toEqual({ promptTokens: 10, completionTokens: 5, totalTokens: 15 }); + expect(res.providerRaw).toEqual(responsesToolCallOutput); + }); + + it("echoes the previous turn's output items and answers with function_call_output", async () => { + requestUrlMock.mockReturnValueOnce( + ok({ + id: "resp_2", + status: "completed", + output: [ + { id: "msg_1", type: "message", role: "assistant", content: [{ type: "output_text", text: "Sunny, 21°C." }] }, + ], + usage: { input_tokens: 1, output_tokens: 1, total_tokens: 2 }, + }), + ); + + const req = toolRequest(); + req.messages.push( + { + role: "assistant", + content: "", + toolCalls: [{ id: "call_abc", name: "get_weather", args: { city: "Paris" } }], + providerRaw: responsesToolCallOutput, + }, + { role: "tool", results: [{ toolCallId: "call_abc", name: "get_weather", content: "sunny" }] }, + ); + const res = await chatRequest( + makeApp(), "sk", gpt6, openaiCompatible("https://api.openai.com/v1"), req, + ); + + expect(sent(0).body.input).toEqual([ + { role: "system", content: "Use tools." }, + { role: "user", content: "Weather in Paris?" }, + ...responsesToolCallOutput, + { type: "function_call_output", call_id: "call_abc", output: "sunny" }, + ]); + expect(res.content).toBe("Sunny, 21°C."); + expect(res.normalizedStopReason).toBe("stop"); + }); + + it("does not retry with reasoning_effort when OpenAI rejects a tool turn", async () => { + requestUrlMock.mockReturnValueOnce( + Promise.resolve({ + status: 400, + json: { error: { message: "Invalid schema for function 'get_weather'.", type: "invalid_request_error" } }, + }), + ); + + await expect( + chatRequest(makeApp(), "sk", gpt6, openaiCompatible("https://api.openai.com/v1"), toolRequest()), + ).rejects.toThrow(/Invalid schema/); + expect(requestUrlMock).toHaveBeenCalledTimes(1); + }); +}); + +describe("OpenAI-compatible endpoints keep Chat Completions", () => { + it.each([ + ["a third-party endpoint", "https://api.groq.com/openai/v1"], + ["a proxy named OpenAI", "https://llm-proxy.example/v1"], + ["a lookalike host", "https://api.openai.com.evil.example/v1"], + ])("%s", async (_label, endpoint) => { + requestUrlMock.mockReturnValueOnce( + ok({ + id: "1", + model: "m", + choices: [ + { + finish_reason: "tool_calls", + index: 0, + message: { + role: "assistant", + content: null, + tool_calls: [{ id: "call_1", type: "function", function: { name: "get_weather", arguments: '{"city":"Paris"}' } }], + }, + }, + ], + usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 }, + }), + ); + + const res = await chatRequest(makeApp(), "sk", gpt6, openaiCompatible(endpoint), toolRequest()); + + const { url, body } = sent(0); + expect(url).toBe(`${endpoint}/chat/completions`); + expect(Array.isArray(body.messages)).toBe(true); + expect(body.input).toBeUndefined(); + expect(res.toolCalls?.map((call) => call.id)).toEqual(["call_1"]); + }); +}); + +describe("Anthropic models that reject forced tool use", () => { + const anthropic: AIProvider = { + name: "Anthropic", + endpoint: "https://api.anthropic.com", + kind: "anthropic", + apiKey: "sk", + models: [], + modelSource: "modelsDev", + }; + const opus55: Model = { name: "claude-opus-5-5", maxTokens: 1_000_000, maxOutputTokens: 128_000 }; + + function anthropic400(message: string) { + return Promise.resolve({ + status: 400, + json: { type: "error", error: { type: "invalid_request_error", message } }, + }); + } + + it("explains the documented 400 and how to fix the call", async () => { + // Exact text from https://platform.claude.com/docs/en/api/errors#forced-tool-use-not-supported + requestUrlMock.mockReturnValueOnce( + anthropic400('tool_choice: type "tool" and "any" are not supported for this model.'), + ); + + const error = await chatRequest(makeApp(), "sk", opus55, anthropic, { + ...toolRequest(), + toolChoice: "required", + }).catch((e: Error) => e); + + expect(sent(0).body.tool_choice).toEqual({ type: "any" }); + expect(requestUrlMock).toHaveBeenCalledTimes(1); + expect(String(error)).toContain( + 'claude-opus-5-5 can\'t be forced to call a tool, so toolChoice "required" and named tools don\'t work with it. Use toolChoice "auto"', + ); + }); + + it("leaves other tool_choice errors alone", async () => { + requestUrlMock.mockReturnValueOnce( + anthropic400("tool_choice.name: Tool 'missing' not found in tools."), + ); + + const error = await chatRequest(makeApp(), "sk", opus55, anthropic, { + ...toolRequest(), + toolChoice: { name: "missing" }, + }).catch((e: Error) => e); + + expect(String(error)).toContain("not found in tools"); + expect(String(error)).not.toContain("can't be forced"); + }); +}); diff --git a/src/ai/OpenAIRequest.ts b/src/ai/OpenAIRequest.ts index b17c2afc0..69b5b2cd8 100644 --- a/src/ai/OpenAIRequest.ts +++ b/src/ai/OpenAIRequest.ts @@ -16,7 +16,7 @@ import { import { preventCursorChange } from "./preventCursorChange"; import { reportError } from "../utils/errorUtils"; import type { AIProvider, Model } from "./Provider"; -import { getProviderKind } from "./Provider"; +import { getChatWire, getProviderKind } from "./Provider"; import type { NormalizedChatRequest } from "./tools/NormalizedTools"; import { buildChatBody, @@ -24,7 +24,10 @@ import { } from "./tools/providerToolMapping"; import { log } from "src/logger/logManager"; import { estimateTokenCount } from "./tokenEstimator"; -import { classifyProviderError } from "./providerErrors"; +import { + classifyProviderError, + isForcedToolChoiceUnsupportedError, +} from "./providerErrors"; export type { CommonResponse, AnthropicContentBlock, AnthropicResponse, GeminiResponse } from "./providerRequest"; @@ -172,7 +175,7 @@ export async function chatRequest( ); } - const kind = getProviderKind(modelProvider); + const wire = getChatWire(modelProvider); // Same sampling safety as the single-prompt path: drop params the model's // metadata marks unsupported, and keep the sent set for the reactive retry. const effectiveRequest: NormalizedChatRequest = { @@ -183,7 +186,7 @@ export async function chatRequest( effectiveRequest.modelParams ?? {}, ); const body = buildChatBody( - kind, + wire, model.name, effectiveRequest, anthropicMaxTokens(model), @@ -206,35 +209,19 @@ export async function chatRequest( }); try { - const send = (body: Record) => dispatchProviderRequest>({ - kind, apiKey, provider: modelProvider, model, body, + const dispatch = (body: Record) => dispatchProviderRequest>({ + kind: wire, apiKey, provider: modelProvider, model, body, afterRequest: afterRequestCallback, }); - const dispatch = async (body: Record) => { - try { - return await send(body); - } catch (error) { - const retryBody = toolReasoningRetryBody( - kind, - body, - (error as { message?: string }).message ?? String(error), - ); - if (!retryBody) throw error; - log.logMessage( - `[AI Chat ${requestLogId}] ${model.name} rejected function tools while reasoning; retrying with reasoning_effort "none".`, - ); - return send(retryBody); - } - }; const json = await retrySampling( () => dispatch(body), - () => dispatch(buildChatBody(kind, model.name, { + () => dispatch(buildChatBody(wire, model.name, { ...effectiveRequest, modelParams: stripSamplingParams(effectiveRequest.modelParams ?? {}), }, anthropicMaxTokens(model))), { sentKeys: samplingKeysSent, model, provider: modelProvider, logPrefix: `AI Chat ${requestLogId}` }, ); - const parsed = parseChatResponse(kind, json); + const parsed = parseChatResponse(wire, json); const durationMs = Date.now() - requestStart; finishAIRequestLogEntry(requestLogId, { status: "success", @@ -267,8 +254,11 @@ export async function chatRequest( // Report the wrapper, not the bare cause: its message names the provider, and // `reportError` reports a failure once (#1601), so reporting the cause first // would suppress the more informative message at every layer above. + const guidance = isForcedToolChoiceUnsupportedError(error) + ? ` ${model.name} can't be forced to call a tool, so toolChoice "required" and named tools don't work with it. Use toolChoice "auto" (the default) and say in the prompt when to call the tool, or pass a schema to get a fixed JSON shape.` + : ""; const failure = new Error( - `Error while making request to ${modelProvider.name}: ${errorMessage}`, + `Error while making request to ${modelProvider.name}: ${errorMessage}${guidance}`, { cause: error }, ); reportError(failure); @@ -276,30 +266,6 @@ export async function chatRequest( } } -// "Function tools with reasoning_effort are not supported for gpt-6-sol in -// /v1/chat/completions. To use function tools, use /v1/responses or set -// reasoning_effort to 'none'." (verified live 2026-09-26 for the gpt-6 and -// gpt-5.6 families, which reason by default; gpt-5.5 and older default to none). -const TOOLS_NEED_NO_REASONING_RE = - /function tools with reasoning_effort are not supported[\s\S]*reasoning_effort to 'none'/i; - -/** - * The body to retry a Chat Completions tool request with when the model - * rejected function tools because it reasons by default, or null when the - * error is anything else or the caller already chose a reasoning effort. - */ -export function toolReasoningRetryBody( - kind: ReturnType, - body: Record, - errorText: string, -): Record | null { - if (kind !== "openai") return null; - if (!Array.isArray(body.tools) || body.tools.length === 0) return null; - if (body.reasoning_effort !== undefined) return null; - if (!TOOLS_NEED_NO_REASONING_RE.test(errorText)) return null; - return { ...body, reasoning_effort: "none" }; -} - async function retrySampling(attempt: () => Promise, retry: () => Promise, context: { sentKeys: ReturnType; model: Model; diff --git a/src/ai/Provider.test.ts b/src/ai/Provider.test.ts index e2cf68c1d..5fb2ef93e 100644 --- a/src/ai/Provider.test.ts +++ b/src/ai/Provider.test.ts @@ -5,11 +5,32 @@ import { DefaultProviders, activeModelRef, ensureProviderIds, + getChatWire, getProviderKind, slugifyProviderId, uniqueProviderId, } from "./Provider"; +describe("getChatWire", () => { + it("uses the Responses API only for OpenAI's own endpoint", () => { + expect(getChatWire({ kind: "openai", endpoint: "https://api.openai.com/v1" })).toBe("openai-responses"); + // Legacy providers without an explicit kind infer it first. + expect(getChatWire({ name: "OpenAI", endpoint: "https://api.openai.com/v1" })).toBe("openai-responses"); + }); + + it("keeps Chat Completions for OpenAI-compatible endpoints and lookalike hosts", () => { + expect(getChatWire({ kind: "openai", endpoint: "https://openrouter.ai/api/v1" })).toBe("openai"); + expect(getChatWire({ name: "OpenAI", endpoint: "https://proxy.example/v1" })).toBe("openai"); + expect(getChatWire({ kind: "openai", endpoint: "https://api.openai.com.evil.example/v1" })).toBe("openai"); + expect(getChatWire({ kind: "openai", endpoint: "https://evil.example/?h=api.openai.com" })).toBe("openai"); + }); + + it("leaves Anthropic and Gemini alone, even on an OpenAI host", () => { + expect(getChatWire({ kind: "anthropic", endpoint: "https://api.openai.com/v1" })).toBe("anthropic"); + expect(getChatWire({ kind: "gemini", endpoint: "https://generativelanguage.googleapis.com" })).toBe("gemini"); + }); +}); + describe("getProviderKind", () => { it("prefers an explicit kind", () => { expect(getProviderKind({ kind: "anthropic", name: "Whatever" })).toBe("anthropic"); diff --git a/src/ai/Provider.ts b/src/ai/Provider.ts index d12319319..aad527bbe 100644 --- a/src/ai/Provider.ts +++ b/src/ai/Provider.ts @@ -60,6 +60,27 @@ export function getProviderKind(provider: { return "openai"; } +/** + * The request format a tool/agent turn is sent in. OpenAI's own API gets the + * Responses API: its newest models (gpt-6-*, gpt-5.6-*) reason by default and + * reject function tools on Chat Completions (verified live 2026-09-26: "use + * /v1/responses or set reasoning_effort to 'none'"). OpenAI-compatible + * third-party endpoints keep Chat Completions, the format they all implement. + */ +export type ChatWire = ProviderKind | "openai-responses"; + +export function getChatWire(provider: { + kind?: ProviderKind; + name?: string; + endpoint?: string; +}): ChatWire { + const kind = getProviderKind(provider); + if (kind === "openai" && endpointHost(provider.endpoint) === "api.openai.com") { + return "openai-responses"; + } + return kind; +} + /** Lowercased hostname of an endpoint, or "" if it can't be parsed (scheme optional). */ function endpointHost(endpoint?: string): string { const raw = (endpoint ?? "").trim(); diff --git a/src/ai/providerErrors.test.ts b/src/ai/providerErrors.test.ts index d03733883..c0f70fa24 100644 --- a/src/ai/providerErrors.test.ts +++ b/src/ai/providerErrors.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it } from "vitest"; import { buildProviderError, classifyProviderError, + isForcedToolChoiceUnsupportedError, isLikelyContextLimitError, } from "./providerErrors"; @@ -165,3 +166,32 @@ describe("buildProviderError", () => { expect(err.status).toBe(503); }); }); + +describe("isForcedToolChoiceUnsupportedError", () => { + it("matches Anthropic's documented forced-tool-use 400 as built by buildProviderError", () => { + const error = buildProviderError("Anthropic", { + status: 400, + json: { + type: "error", + error: { + type: "invalid_request_error", + message: 'tool_choice: type "tool" and "any" are not supported for this model.', + }, + }, + }); + expect(isForcedToolChoiceUnsupportedError(error)).toBe(true); + }); + + it("ignores other tool_choice errors", () => { + expect( + isForcedToolChoiceUnsupportedError( + new Error("tool_choice.name: Tool 'x' not found in tools."), + ), + ).toBe(false); + expect( + isForcedToolChoiceUnsupportedError( + new Error("Thinking may not be enabled when tool_choice forces tool use."), + ), + ).toBe(false); + }); +}); diff --git a/src/ai/providerErrors.ts b/src/ai/providerErrors.ts index b081c4296..a5b975f6e 100644 --- a/src/ai/providerErrors.ts +++ b/src/ai/providerErrors.ts @@ -148,6 +148,22 @@ export function isLikelyContextLimitError(error: unknown): boolean { return classifyProviderError(error) === "input_context"; } +// Anthropic's documented 400 for models that don't support forced tool use +// (Claude Opus 5.5, Fable 5.1, Mythos 5.1 as of 2026-09): +// tool_choice: type "tool" and "any" are not supported for this model. +// Neither Anthropic's Models API capabilities nor models.dev expose this as +// model metadata, so the error itself is the signal; matching it (rather +// than a list of model ids) also covers models that adopt the rule later. +const FORCED_TOOL_CHOICE_UNSUPPORTED_RE = + /tool_choice:\s*type\s*"tool"\s*and\s*"any"\s*are not supported/i; + +/** True when the provider rejected a forced tool_choice (`required` / a named tool). */ +export function isForcedToolChoiceUnsupportedError(error: unknown): boolean { + return FORCED_TOOL_CHOICE_UNSUPPORTED_RE.test( + collectErrorMessages(error).join(" "), + ); +} + export interface NormalizedProviderError extends Error { status: number; providerCode?: string; diff --git a/src/ai/providerRequest.ts b/src/ai/providerRequest.ts index 88bfd87fa..85960a5e0 100644 --- a/src/ai/providerRequest.ts +++ b/src/ai/providerRequest.ts @@ -1,6 +1,6 @@ import { requestUrl } from "obsidian"; import type { OpenAIModelParameters } from "./OpenAIModelParameters"; -import type { AIProvider, Model } from "./Provider"; +import type { AIProvider, ChatWire, Model } from "./Provider"; import type { ProviderKind } from "./tools/providerToolMapping"; import type { NormalizedStopReason, NormalizedToolCall } from "./tools/NormalizedTools"; import { buildProviderError } from "./providerErrors"; @@ -17,7 +17,7 @@ type RequestContext = { export async function dispatchProviderRequest({ kind, apiKey, provider, model, body, afterRequest, }: RequestContext & { - kind: ProviderKind; + kind: ChatWire; body: Record; }): Promise { const headers: Record = { "Content-Type": "application/json" }; @@ -32,6 +32,10 @@ export async function dispatchProviderRequest({ path = `/v1beta/models/${encodeURIComponent(model.name)}:generateContent`; headers["x-goog-api-key"] = apiKey; break; + case "openai-responses": + path = "/responses"; + headers.Authorization = `Bearer ${apiKey}`; + break; default: path = "/chat/completions"; headers.Authorization = `Bearer ${apiKey}`; diff --git a/src/ai/tools/openai.e2e.test.ts b/src/ai/tools/openai.e2e.test.ts index b291c3fd3..85a690bb3 100644 --- a/src/ai/tools/openai.e2e.test.ts +++ b/src/ai/tools/openai.e2e.test.ts @@ -3,40 +3,55 @@ * pure modules (providerToolMapping.buildChatBody/parseChatResponse + runToolLoop + * jsonSchemaValidator) end-to-end — the layer the prototype could only mock. * + * Two wires, matching getChatWire: OpenAI's own endpoint gets the Responses API + * (/v1/responses), and OpenAI-compatible endpoints get Chat Completions, exercised + * here against OpenAI's own Chat Completions as the reference implementation. + * * Skipped unless OPENAI_API_KEY is set, so the normal suite + CI never hit the network: * OPENAI_API_KEY=$(op read "op://Agent Secrets/OpenAI API Key/credential") \ * npx vitest run src/ai/tools/openai.e2e.test.ts --config vitest.config.mts */ import { describe } from "vitest"; import { liveWireCases } from "../../../tests/helpers/ai/liveWireCases"; +import type { ChatWire } from "../Provider"; import { buildChatBody, parseChatResponse } from "./providerToolMapping"; import type { NormalizedChatRequest } from "./NormalizedTools"; const KEY = process.env.OPENAI_API_KEY; -// Current-generation default (GPT-5.x). The bare request path sends no max_tokens and -// no temperature, so it is wire-compatible with GPT-5.x reasoning models (which reject -// `max_tokens` in favour of `max_completion_tokens` and only accept the default -// temperature). Override with OPENAI_E2E_MODEL to target a specific model. -const MODEL = process.env.OPENAI_E2E_MODEL ?? "gpt-5-mini"; -const URL = "https://api.openai.com/v1/chat/completions"; +// Responses API default: a current reasoning model. gpt-6-* and gpt-5.6-* reason by +// default, which Chat Completions rejects alongside function tools; the Responses +// API accepts both. Override with OPENAI_E2E_MODEL to target a specific model. +const MODEL = process.env.OPENAI_E2E_MODEL ?? "gpt-6-luna"; +// Chat Completions default: a model that accepts function tools there as-is. +// Override with OPENAI_E2E_CHAT_MODEL. +const CHAT_MODEL = process.env.OPENAI_E2E_CHAT_MODEL ?? "gpt-5-mini"; -async function openaiDispatch(req: NormalizedChatRequest) { - const body = buildChatBody("openai", MODEL, req); - const res = await fetch(URL, { - method: "POST", - headers: { - "Content-Type": "application/json", - Authorization: `Bearer ${KEY}`, - }, - body: JSON.stringify(body), - }); - const json = (await res.json()) as Record; - if (!res.ok) throw new Error(`OpenAI ${res.status}: ${JSON.stringify(json)}`); - return parseChatResponse("openai", json); +function dispatcher(wire: ChatWire, model: string, path: string) { + return async (req: NormalizedChatRequest) => { + const body = buildChatBody(wire, model, req); + const res = await fetch(`https://api.openai.com/v1${path}`, { + method: "POST", + headers: { + "Content-Type": "application/json", + Authorization: `Bearer ${KEY}`, + }, + body: JSON.stringify(body), + }); + const json = (await res.json()) as Record; + if (!res.ok) throw new Error(`OpenAI ${res.status}: ${JSON.stringify(json)}`); + return parseChatResponse(wire, json); + }; } -describe.skipIf(!KEY)("OpenAI live wire (e2e)", () => { - liveWireCases(openaiDispatch, { +describe.skipIf(!KEY)(`OpenAI Responses API live wire (e2e, ${MODEL})`, () => { + liveWireCases(dispatcher("openai-responses", MODEL, "/responses"), { + toolLoop: "runs a real tool-calling loop end-to-end", + structured: "returns schema-constrained structured output", + }, 60000); +}); + +describe.skipIf(!KEY)(`OpenAI Chat Completions live wire (e2e, ${CHAT_MODEL})`, () => { + liveWireCases(dispatcher("openai", CHAT_MODEL, "/chat/completions"), { toolLoop: "runs a real tool-calling loop end-to-end", structured: "returns schema-constrained structured output", }, 60000); diff --git a/src/ai/tools/providerToolMapping.test.ts b/src/ai/tools/providerToolMapping.test.ts index 1c41bdd68..84b7878c5 100644 --- a/src/ai/tools/providerToolMapping.test.ts +++ b/src/ai/tools/providerToolMapping.test.ts @@ -275,3 +275,134 @@ describe("injectStrictObjectSchema", () => { }); }); }); + +describe("OpenAI Responses mapping (api.openai.com)", () => { + it("minimal body: model, input, stateless storage, encrypted reasoning", () => { + const body = buildChatBody("openai-responses", "gpt-6-luna", { + messages: [{ role: "user", content: "hi" }], + }); + expect(body).toEqual({ + model: "gpt-6-luna", + input: [{ role: "user", content: "hi" }], + store: false, + include: ["reasoning.encrypted_content"], + }); + }); + + it("renames Chat Completions modelOptions and passes the rest through", () => { + const body = buildChatBody("openai-responses", "gpt-6-luna", { + messages: [{ role: "user", content: "hi" }], + modelParams: { reasoning_effort: "high", max_tokens: 300, top_p: 0.5 } as never, + }); + expect(body.reasoning).toEqual({ effort: "high" }); + expect(body.max_output_tokens).toBe(300); + expect(body.top_p).toBe(0.5); + expect(body).not.toHaveProperty("reasoning_effort"); + expect(body).not.toHaveProperty("max_tokens"); + }); + + it("maxOutputTokens wins over a max_tokens model option", () => { + const body = buildChatBody("openai-responses", "gpt-6-luna", { + messages: [{ role: "user", content: "hi" }], + maxOutputTokens: 50, + modelParams: { max_tokens: 300 } as never, + }); + expect(body.max_output_tokens).toBe(50); + }); + + it("flat tools (strict schema injected), named tool_choice, parallel flag", () => { + const body = buildChatBody("openai-responses", "gpt-6-luna", { + messages: [{ role: "user", content: "make a note" }], + tools: [{ ...tool, strict: true }], + toolChoice: { name: "create_note" }, + disableParallel: true, + }); + expect(body.tools).toEqual([ + { + type: "function", + name: "create_note", + description: "Create a note", + parameters: { ...tool.parameters, additionalProperties: false }, + strict: true, + }, + ]); + expect(body.tool_choice).toEqual({ type: "function", name: "create_note" }); + expect(body.parallel_tool_calls).toBe(false); + }); + + it("structured output goes under text.format with the strict schema", () => { + const schema = { type: "object" as const, properties: { title: { type: "string" as const } } }; + const body = buildChatBody("openai-responses", "gpt-6-luna", { + messages: [{ role: "user", content: "x" }], + responseFormat: { schema, name: "meta" }, + }); + expect(body.text).toEqual({ + format: { + type: "json_schema", + name: "meta", + strict: true, + schema: { ...schema, required: ["title"], additionalProperties: false }, + }, + }); + expect(body).not.toHaveProperty("response_format"); + }); + + it("rebuilds function_call items when there is no provider output to echo", () => { + const body = buildChatBody("openai-responses", "gpt-6-luna", { + messages: [ + { role: "user", content: "q" }, + { + role: "assistant", + content: "Checking.", + toolCalls: [{ id: "call_1", name: "create_note", args: { path: "a.md" } }], + }, + { role: "tool", results: [{ toolCallId: "call_1", name: "create_note", content: "no", isError: true }] }, + ], + tools: [tool], + }); + expect(body.input).toEqual([ + { role: "user", content: "q" }, + { role: "assistant", content: "Checking." }, + { type: "function_call", call_id: "call_1", name: "create_note", arguments: '{"path":"a.md"}' }, + { type: "function_call_output", call_id: "call_1", output: "ERROR: no" }, + ]); + }); + + it("an output cut off by max_output_tokens is 'length', and reasoning text is not content", () => { + const parsed = parseChatResponse("openai-responses", { + status: "incomplete", + incomplete_details: { reason: "max_output_tokens" }, + output: [ + { type: "reasoning", summary: [{ type: "summary_text", text: "thinking" }] }, + { type: "message", content: [{ type: "output_text", text: "Once upon" }, { type: "refusal", refusal: "x" }] }, + ], + usage: { input_tokens: 3, output_tokens: 16, total_tokens: 19 }, + }); + expect(parsed.content).toBe("Once upon"); + expect(parsed.normalizedStopReason).toBe("length"); + expect(parsed.rawStopReason).toBe("max_output_tokens"); + }); + + it("returns a refusal's explanation instead of an empty answer", () => { + const parsed = parseChatResponse("openai-responses", { + status: "completed", + output: [ + { type: "message", content: [{ type: "refusal", refusal: "I can't help with that." }] }, + ], + }); + expect(parsed.content).toBe("I can't help with that."); + expect(parsed.normalizedStopReason).toBe("other"); + expect(parsed.rawStopReason).toBe("refusal"); + }); + + it("flags unparseable function_call arguments instead of throwing", () => { + const parsed = parseChatResponse("openai-responses", { + status: "completed", + output: [{ type: "function_call", id: "fc_1", call_id: "call_9", name: "create_note", arguments: "{oops" }], + }); + expect(parsed.toolCalls).toEqual([ + { id: "call_9", name: "create_note", args: null, rawArgs: "{oops", parseError: true }, + ]); + expect(parsed.normalizedStopReason).toBe("tool_calls"); + }); +}); diff --git a/src/ai/tools/providerToolMapping.ts b/src/ai/tools/providerToolMapping.ts index 05ec4dd34..93a112ef4 100644 --- a/src/ai/tools/providerToolMapping.ts +++ b/src/ai/tools/providerToolMapping.ts @@ -22,6 +22,7 @@ import type { NormalizedToolChoice, NormalizedToolDefinition, } from "./NormalizedTools"; +import type { ChatWire } from "../Provider"; export type ProviderKind = "openai" | "anthropic" | "gemini"; @@ -237,6 +238,160 @@ function parseOpenAIToolCall(tc: OpenAIToolCallRaw): NormalizedToolCall { return { id: tc.id, name: tc.function.name, args: rec }; } +// =========================================================================== +// OpenAI Responses API (/v1/responses) — OpenAI's own endpoint only +// =========================================================================== +// Same normalized request as Chat Completions, different shape: tools are flat +// ({type,name,...} with no `function` wrapper), tool calls and their results are +// top-level input items linked by `call_id`, and structured output lives under +// `text.format`. Requests are stateless (`store: false`), so reasoning items only +// survive between turns as `encrypted_content`: we ask for it and echo the +// previous turn's output items back verbatim (providerRaw), which verified live +// keeps multi-turn tool loops working on reasoning models. +function responsesInput(messages: NormalizedMessage[]): Body[] { + const out: Body[] = []; + for (const m of messages) { + if (m.role === "system" || m.role === "user") { + out.push({ role: m.role, content: m.content }); + } else if (m.role === "assistant") { + if (Array.isArray(m.providerRaw)) { + out.push(...(m.providerRaw as Body[])); + continue; + } + if (m.content) out.push({ role: "assistant", content: m.content }); + for (const c of m.toolCalls ?? []) { + out.push({ + type: "function_call", + call_id: c.id, + name: c.name, + arguments: c.rawArgs ?? JSON.stringify(c.args ?? {}), + }); + } + } else { + for (const r of m.results) { + out.push({ + type: "function_call_output", + call_id: r.toolCallId, + output: r.isError ? `ERROR: ${r.content}` : r.content, + }); + } + } + } + return out; +} + +function responsesToolChoice(choice: NormalizedToolChoice): unknown { + if (typeof choice === "string") return choice; + return { type: "function", name: choice.name }; +} + +function buildOpenAIResponsesBody(modelName: string, req: NormalizedChatRequest): Body { + // modelOptions is written for Chat Completions; move the fields the + // Responses API names differently and pass the rest through unchanged. + const { + reasoning_effort: reasoningEffort, + max_tokens: legacyMaxTokens, + max_completion_tokens: legacyMaxCompletionTokens, + ...params + } = (req.modelParams ?? {}) as Body; + const body: Body = { + model: modelName, + ...params, + input: responsesInput(req.messages), + store: false, + include: ["reasoning.encrypted_content"], + }; + if (reasoningEffort !== undefined) body.reasoning = { effort: reasoningEffort }; + const maxOutputTokens = + req.maxOutputTokens ?? legacyMaxCompletionTokens ?? legacyMaxTokens; + if (maxOutputTokens !== undefined) body.max_output_tokens = maxOutputTokens; + if (req.tools && req.tools.length > 0) { + body.tools = req.tools.map((t) => ({ + type: "function", + name: t.name, + description: t.description, + parameters: t.strict ? injectStrictObjectSchema(t.parameters) : t.parameters, + // Explicit either way: keep Chat Completions' non-strict default. + strict: t.strict === true, + })); + if (req.toolChoice) body.tool_choice = responsesToolChoice(req.toolChoice); + if (req.disableParallel) body.parallel_tool_calls = false; + } + if (req.responseFormat) { + const strict = req.responseFormat.strict ?? true; + body.text = { + format: req.responseFormat.schema + ? { + type: "json_schema", + name: req.responseFormat.name ?? "response", + strict, + schema: strict + ? injectStrictObjectSchema(req.responseFormat.schema) + : req.responseFormat.schema, + } + : { type: "json_object" }, + }; + } + return body; +} + +interface ResponsesOutputItemRaw { + type: string; + call_id?: string; + name?: string; + arguments?: unknown; + content?: Array<{ type: string; text?: string; refusal?: string }>; +} +function parseOpenAIResponsesResponse(json: Record): ParsedChatResult { + const output = (json.output as ResponsesOutputItemRaw[] | undefined) ?? []; + const toolCalls = output + .filter((item) => item.type === "function_call") + .map((item) => + parseOpenAIToolCall({ + id: item.call_id ?? "", + function: { name: item.name ?? "", arguments: item.arguments }, + }), + ); + const parts = output + .filter((item) => item.type === "message") + .flatMap((item) => item.content ?? []); + const text = parts + .filter((part) => part.type === "output_text") + .map((part) => part.text ?? "") + .join(""); + // A safety refusal arrives as a `refusal` part instead of output_text. Return + // its explanation rather than an empty answer, and don't call it a normal stop. + const refusal = parts + .filter((part) => part.type === "refusal") + .map((part) => part.refusal ?? "") + .join(""); + const refused = !text && refusal.length > 0; + const status = String(json.status ?? ""); + const incompleteReason = String( + (json.incomplete_details as { reason?: string } | null | undefined)?.reason ?? "", + ); + const usage = (json.usage as Record) ?? {}; + return { + content: refused ? refusal : text, + toolCalls, + normalizedStopReason: + toolCalls.length > 0 + ? "tool_calls" + : incompleteReason === "max_output_tokens" + ? "length" + : status === "completed" && !refused + ? "stop" + : "other", + rawStopReason: refused ? "refusal" : incompleteReason || status, + usage: { + promptTokens: usage.input_tokens ?? 0, + completionTokens: usage.output_tokens ?? 0, + totalTokens: usage.total_tokens ?? 0, + }, + providerRaw: output, + }; +} + // =========================================================================== // Anthropic // =========================================================================== @@ -481,30 +636,34 @@ function parseGeminiResponse(json: Record): ParsedChatResult { // Dispatch // =========================================================================== export function buildChatBody( - kind: ProviderKind, + wire: ChatWire, modelName: string, req: NormalizedChatRequest, anthropicDefaultMaxTokens = 8192, ): Body { - switch (kind) { + switch (wire) { case "anthropic": return buildAnthropicBody(modelName, req, anthropicDefaultMaxTokens); case "gemini": return buildGeminiBody(modelName, req); + case "openai-responses": + return buildOpenAIResponsesBody(modelName, req); default: return buildOpenAIBody(modelName, req); } } export function parseChatResponse( - kind: ProviderKind, + wire: ChatWire, json: Record, ): ParsedChatResult { - switch (kind) { + switch (wire) { case "anthropic": return parseAnthropicResponse(json); case "gemini": return parseGeminiResponse(json); + case "openai-responses": + return parseOpenAIResponsesResponse(json); default: return parseOpenAIResponse(json); }