diff --git a/CHANGELOG.md b/CHANGELOG.md index ecb075f81..4b3638d43 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,8 @@ ## Unreleased +- Fix gateway image-generation deployment errors: infer the built-in tool only for the direct OpenAI API; gateways can opt in with model `imageGeneration` and `extraHeaders`. + ## 0.161.2 - List the new OpenAI `gpt-6-sol`/`gpt-6-luna` models for ChatGPT OAuth accounts by bumping the Codex client version the backend gates on. diff --git a/docs/config.json b/docs/config.json index 01b324cbf..7a7dd9146 100644 --- a/docs/config.json +++ b/docs/config.json @@ -868,6 +868,11 @@ "markdownDescription": "Whether this model accepts image input (vision). Useful for custom/local models models.dev doesn't know about.", "default": false }, + "imageGeneration": { + "type": "boolean", + "description": "Enable the built-in image_generation tool on compatible Responses API endpoints. When omitted, enabled only for capable models on the direct OpenAI API, not gateways or Copilot. Azure-backed gateways may require x-ms-oai-image-generation-deployment in extraHeaders. Does not affect image input.", + "markdownDescription": "Enable the built-in `image_generation` tool on compatible Responses API endpoints. When omitted, enabled only for capable models on the direct OpenAI API, not gateways or Copilot. Azure-backed gateways may require `x-ms-oai-image-generation-deployment` in `extraHeaders`. Does not affect image input." + }, "limit": { "type": "object", "description": "Override the model's token limits (overrides models.dev). Useful for local models or to cap a known model's context window.", diff --git a/docs/troubleshooting.md b/docs/troubleshooting.md index 86bbc11b6..24e1a76a7 100644 --- a/docs/troubleshooting.md +++ b/docs/troubleshooting.md @@ -99,6 +99,25 @@ One way to workaround that is to start the editor from your terminal. launchctl setenv ANTHROPIC_API_KEY "your-key-here" ``` +## Image-generation deployment errors + +An error such as `imagegen deployment must be provided through header: x-ms-oai-image-generation-deployment` means the endpoint rejected the Responses API `image_generation` tool because no image deployment was selected. A gateway serving an OpenAI chat model does not necessarily support OpenAI's built-in image-generation tool. + +ECA automatically enables this tool only for capable models on the direct OpenAI API. Gateways and Copilot require explicit opt-in. To disable it, set `"imageGeneration": false` in `providers..models.`. This does not disable image input or ordinary function/MCP tools. + +If your gateway supports image generation and requires the deployment header, configure that model with: + +```json +{ + "imageGeneration": true, + "extraHeaders": { + "x-ms-oai-image-generation-deployment": "your-image-deployment-name" + } +} +``` + +Use the actual image-generation deployment name supplied by your gateway administrator, not the chat model name. The header can also be set in provider-level `extraHeaders`; model-level values take precedence. ECA cannot infer or provision the deployment. + ## Ask for help You can ask for help via chat [here](https://clojurians.slack.com/archives/C093426FPUG) diff --git a/src/eca/llm_api.clj b/src/eca/llm_api.clj index a01dca68d..38df718aa 100644 --- a/src/eca/llm_api.clj +++ b/src/eca/llm_api.clj @@ -290,7 +290,6 @@ supports-image? (:image-input? model-capabilities) web-search (:web-search model-capabilities) mid-conversation-system? (:mid-conversation-system? model-capabilities) - image-generation (:image-generation? model-capabilities) max-output-tokens (:max-output-tokens model-capabilities) provider-config (get-in config [:providers provider]) model-config (get-in provider-config [:models model]) @@ -317,6 +316,16 @@ reasoning-history (or (:reasoningHistory model-config) :all) [auth-type api-key] (llm-util/provider-api-key provider provider-auth config) api-url (llm-util/provider-api-url provider config) + ;; Model-name inference cannot establish gateway-side tool support. + ;; Keep explicit opt-in/out, but infer only for the direct OpenAI API. + image-generation (if-some [enabled (:imageGeneration model-config)] + enabled + (boolean + (and (:image-generation? model-capabilities) + (not= "github-copilot" provider) + (not (and (= "openai" provider) (= :auth/oauth auth-type))) + api-url + (re-matches #"(?i)https://api\.openai\.com(?::443)?(?:/.*)?" api-url)))) ;; Flatten {:static :dynamic} instructions map into a single string for non-Anthropic providers flat-instructions (if (map? instructions) (f.prompt/instructions->str instructions) instructions) anthropic-opts {:model real-model diff --git a/src/eca/models.clj b/src/eca/models.clj index 68795bfd9..07dd9b7ce 100644 --- a/src/eca/models.clj +++ b/src/eca/models.clj @@ -682,6 +682,10 @@ ;; the provider adapter that discovered it. :provider-data (not-empty (:discovered-provider-data model-config)) :image-input? (:imageInput model-config) + ;; Built-in Responses `image_generation` server-tool opt-out/opt-in; + ;; gateways (e.g. Azure-backed) may reject the tool even when the + ;; model name matches the OpenAI catalog entry that allows it. + :image-generation? (:imageGeneration model-config) :limit (not-empty limit-overrides) :max-output-tokens output-override :input-token-cost (cost-per-1m->per-token (:input cost)) diff --git a/test/eca/llm_api_test.clj b/test/eca/llm_api_test.clj index cd1eaed2b..4f9a37baa 100644 --- a/test/eca/llm_api_test.clj +++ b/test/eca/llm_api_test.clj @@ -473,28 +473,85 @@ (is (= {:effort "medium" :summary "auto"} (get-in @captured* [:extra-payload :reasoning]))) (is (nil? (get-in @captured* [:extra-payload :reasoning_effort]))))))) -(deftest prompt-passes-image-generation-to-openai-handler-test - (testing "openai branch forwards :image-generation true to create-response! when capability is on" - (let [captured* (atom nil)] - (with-redefs [llm-providers.openai/create-response! - (fn [opts _callbacks] (reset! captured* opts) :ok)] - (#'eca.llm-api/prompt! - {:provider "openai" - :model "gpt-5.2" - :model-capabilities {:tools true - :reason? false - :web-search false - :image-generation? true - :model-name "gpt-5.2"} - :user-messages [{:role "user" :content [{:type :text :text "hi"}]}] - :past-messages [] - :tools [] - :provider-auth {:api-key "test-key"} - :config {:providers {"openai" {:url "https://api.openai.com" :key "test-key"}}} - :sync? false})) - (is (= true (:image-generation @captured*)) - "openai handler should receive :image-generation true"))) - +(deftest prompt-image-generation-provider-boundaries-test + (doseq [{:keys [label provider url model-config catalog-image? expected-image?] + :or {provider "openai" + url "https://api.openai.com" + model-config {} + catalog-image? true}} + [{:label "direct OpenAI retains image generation" :expected-image? true} + {:label "custom provider using direct OpenAI" :provider "custom" :expected-image? true} + {:label "unknown model does not gain image generation" :catalog-image? false :expected-image? false} + {:label "explicit opt-out overrides direct OpenAI capability" + :model-config {:imageGeneration false} :expected-image? false} + {:label "gateway does not inherit server tools from the model name" + :provider "gateway" :url "https://gateway.example.com" :expected-image? false} + {:label "overriding the built-in OpenAI URL also disables inference" + :url "https://gateway.example.com" :expected-image? false} + {:label "Copilot Responses does not inherit image generation" + :provider "github-copilot" :url "https://api.githubcopilot.com" :expected-image? false} + {:label "an OpenAI-looking gateway hostname is not the direct API" + :url "https://api.openai.com.gateway.example.com" :expected-image? false} + {:label "gateway opt-in supports a configured deployment and unknown model" + :provider "gateway" :url "https://gateway.example.com" :catalog-image? false + :model-config {:imageGeneration true + :extraHeaders {"x-ms-oai-image-generation-deployment" "image-deployment"}} + :expected-image? true} + {:label "a deployment header alone does not override an explicit opt-out" + :provider "gateway" :url "https://gateway.example.com" + :model-config {:imageGeneration false + :extraHeaders {"x-ms-oai-image-generation-deployment" "image-deployment"}} + :expected-image? false}]] + (testing label + (let [requests* (atom []) + messages* (atom []) + errors* (atom []) + tool {:full-name "eca__directory_tree" :description "list" :parameters {:type "object"}} + user-message {:role "user" :content [{:type :text :text "hello"} + {:type :image :media-type "image/png" :base64 "AAA"}]}] + (with-redefs [http/post + (fn [_ opts] + (swap! requests* conj (update opts :body #(json/parse-string % true))) + {:status 200 + :body (java.io.ByteArrayInputStream. + (.getBytes + (str "event: response.completed\ndata: " + (json/generate-string + {:response {:status "completed" + :usage {:input_tokens 1 :output_tokens 1} + :output (if (= 1 (count @requests*)) + [{:type "function_call" :id "item-1" :call_id "call-1" + :name "eca__directory_tree" :arguments "{}"}] + [])}}) + "\n\n") + java.nio.charset.StandardCharsets/UTF_8))})] + (#'llm-api/prompt! + {:provider provider :model "gpt-5.2" + :model-capabilities {:api :openai-responses :tools true :web-search true + :image-generation? catalog-image? :image-input? true} + :instructions "test" :user-messages [user-message] :past-messages [] :tools [tool] + :config {:providers {provider {:api "openai-responses" :url url :key "test-key" + :models {"gpt-5.2" model-config}}}} + :on-message-received #(swap! messages* conj %) + :on-error #(swap! errors* conj %) + :on-usage-updated identity :on-prepare-tool-call identity + :on-tools-called (fn [_] + {:new-messages [user-message + {:role "tool_call_output" + :content {:id "call-1" :output {:contents [{:type :text :text "result"}]}}}] + :tools [tool]})})) + (is (empty? @errors*)) + (is (= [{:type :finish :finish-reason "completed"}] @messages*)) + (is (= 2 (count @requests*))) + (doseq [request @requests*] + (is (= (cond-> ["function" "web_search"] expected-image? (conj "image_generation")) + (mapv :type (get-in request [:body :tools])))) + (is (= (get-in model-config [:extraHeaders "x-ms-oai-image-generation-deployment"]) + (get-in request [:headers "x-ms-oai-image-generation-deployment"])))) + (is (= {:type "input_image" :image_url "data:image/png;base64,AAA"} + (get-in (first @requests*) [:body :input 0 :content 1]))))))) + +(deftest prompt-sanitizes-openai-messages-test (testing "openai branch strips internal top-level message fields before reaching handler" (let [captured* (atom nil)] (with-redefs [llm-providers.openai/create-response! @@ -573,28 +630,7 @@ (is (= 2 (count @seen-bodies*))) (is (= [{:role "assistant" :content [{:type :text :text "after tool"}]}] - (:input (second @seen-bodies*)))))) - - (testing "openai branch forwards :image-generation false (or nil) when capability is off" - (let [captured* (atom nil)] - (with-redefs [llm-providers.openai/create-response! - (fn [opts _callbacks] (reset! captured* opts) :ok)] - (#'eca.llm-api/prompt! - {:provider "openai" - :model "gpt-4-legacy" - :model-capabilities {:tools true - :reason? false - :web-search false - :image-generation? false - :model-name "gpt-4-legacy"} - :user-messages [{:role "user" :content [{:type :text :text "hi"}]}] - :past-messages [] - :tools [] - :provider-auth {:api-key "test-key"} - :config {:providers {"openai" {:url "https://api.openai.com" :key "test-key"}}} - :sync? false})) - (is (not (true? (:image-generation @captured*))) - "openai handler should NOT receive :image-generation true when capability is off")))) + (:input (second @seen-bodies*))))))) (deftest prompt-forwards-codex-decision-inputs-only-for-openai-test (let [base-opts {:model "gpt-5.6-sol"