From ab180b155ff67eb0e29442a01f3ffe43587c9e32 Mon Sep 17 00:00:00 2001 From: Ryan Schmukler Date: Fri, 9 Oct 2026 15:59:07 -0400 Subject: [PATCH 1/2] fix: recover Anthropic streams failing mid-response with transient SSE errors Classify api_error, overloaded_error and timeout_error SSE error events as overloaded so they are retried before content, or auto-continued once the response has started, instead of ending the prompt. --- CHANGELOG.md | 1 + src/eca/llm_providers/anthropic.clj | 8 +++++- test/eca/llm_providers/anthropic_test.clj | 35 ++++++++++++++++++++++- 3 files changed, 42 insertions(+), 2 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 0df44e53e..3f0314fcb 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -3,6 +3,7 @@ ## Unreleased - Fix "Waiting for tool call approval" progress being replaced by "Generating" when another tool call finishes while one still waits for approval. eca-emacs#334 +- Retry or auto-continue Anthropic streams that fail mid-response with an `api_error`, `overloaded_error` or `timeout_error` SSE event, instead of ending the prompt. ## 0.163.1 diff --git a/src/eca/llm_providers/anthropic.clj b/src/eca/llm_providers/anthropic.clj index c11b0fe52..00fd264ac 100644 --- a/src/eca/llm_providers/anthropic.clj +++ b/src/eca/llm_providers/anthropic.clj @@ -250,6 +250,10 @@ :message (llm-util/connection-error-message e)}))) @response*)) +(def ^:private transient-stream-error-types + "Anthropic SSE `error` event types equivalent to transient 5xx statuses." + #{"api_error" "overloaded_error" "timeout_error"}) + (defn ^:private request-with-retry! "Retries one exact post-tool request, never the tool execution that built it. Defer retry decisions until the failed stream and its watchdog are closed." @@ -713,7 +717,9 @@ :premature? true}) (throw (ex-info "Stream ended without completion" {:error/type :premature-stop})))) - "error" (on-error {:message (format "\nAnthropic error response: %s" (:error data))}) + "error" (on-error (cond-> {:message (format "\nAnthropic error response: %s" (:error data))} + (contains? transient-stream-error-types (-> data :error :type)) + (assoc :error/type :overloaded))) nil))))] (base-request! {:rid (llm-util/gen-rid) diff --git a/test/eca/llm_providers/anthropic_test.clj b/test/eca/llm_providers/anthropic_test.clj index f99c4b245..19c7348fe 100644 --- a/test/eca/llm_providers/anthropic_test.clj +++ b/test/eca/llm_providers/anthropic_test.clj @@ -265,12 +265,45 @@ (is (= 3 (count requests))) (is (= 1 tools-called) "events after the SSE error are not dispatched") (is (apply = (rest requests))) - (is (= [:rate-limited] (mapv #(get-in % [:classified :error/type]) retries))) + (is (= [:overloaded] (mapv #(get-in % [:classified :error/type]) retries))) (is (nil? (get-in retries [0 :error-data :exception])) "do not replace the SSE error with the unwind exception") (is (empty? errors)) (is (= [:text :finish] (mapv :type messages))) (is (= [2 3 1] closed)))) +(deftest post-tool-sse-transient-error-test + (doseq [error-type ["api_error" "overloaded_error" "timeout_error"]] + (testing (str error-type " before content is retried as overloaded") + (let [{:keys [requests errors retries messages]} + (post-tool-scenario! + {:child-response (fn [n respond] + (if (= 2 n) + (respond [["error" {:error {:type error-type :message "read: operation timed out"}}]]) + (respond final-events)))})] + (is (= 3 (count requests))) + (is (= [:overloaded] (mapv #(get-in % [:classified :error/type]) retries))) + (is (empty? errors)) + (is (= [:text :finish] (mapv :type messages))))) + (testing (str error-type " after reasoning started surfaces as overloaded for chat recovery") + (let [{:keys [requests errors retries]} + (post-tool-scenario! + {:child-response (fn [_ respond] + (respond [(first final-events) + ["content_block_start" {:index 0 :content_block {:type "thinking"}}] + ["error" {:error {:type error-type :message "read: operation timed out"}}]]))})] + (is (= 2 (count requests))) + (is (empty? retries)) + (is (= [:overloaded] (mapv :error/type errors))))))) + +(deftest post-tool-sse-non-transient-error-test + (let [{:keys [requests errors retries]} + (post-tool-scenario! + {:child-response (fn [_ respond] + (respond [["error" {:error {:type "invalid_request_error" :message "bad request"}}]]))})] + (is (= 2 (count requests))) + (is (empty? retries)) + (is (= [nil] (mapv :error/type errors))))) + (deftest post-tool-rate-limit-delay-test (let [{:keys [requests tools-called errors retries sleeps]} (post-tool-scenario! From 849fe10b1bf535fe6a39285a8f424d10f069a4ee Mon Sep 17 00:00:00 2001 From: Ryan Schmukler Date: Fri, 9 Oct 2026 16:18:43 -0400 Subject: [PATCH 2/2] fix: refill auto-continue budget after a completed step maxAutoContinues now limits consecutive recoveries instead of recoveries per user turn. The count resets once a response completes and its tools run, so long agentic turns aren't stopped by transient failures spread across many steps. --- CHANGELOG.md | 1 + docs/config/network.md | 5 +++-- src/eca/features/chat.clj | 14 ++++++++++---- test/eca/features/chat_test.clj | 34 +++++++++++++++++++++++++++++++-- 4 files changed, 46 insertions(+), 8 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 3f0314fcb..12c6edfcf 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,7 @@ ## Unreleased +- Refill the `maxAutoContinues` recovery budget once a response completes and its tools run, so long turns aren't stopped by failures spread across many steps. - Fix "Waiting for tool call approval" progress being replaced by "Generating" when another tool call finishes while one still waits for approval. eca-emacs#334 - Retry or auto-continue Anthropic streams that fail mid-response with an `api_error`, `overloaded_error` or `timeout_error` SSE event, instead of ending the prompt. diff --git a/docs/config/network.md b/docs/config/network.md index 8ca650cb3..f437f81fd 100644 --- a/docs/config/network.md +++ b/docs/config/network.md @@ -59,8 +59,9 @@ tools. Once output has started, or request retries are exhausted, ECA falls back chat-level recovery when safe. Chat-level recovery is limited by `providers..retry.maxAutoContinues` -(default `3`) **per user turn**, not per connection. Truncated-response continuations -share this budget. Progress shows the recovery count, and the terminal error explains +(default `3`) **consecutive recoveries**, not per connection; the budget refills once a +response completes and its tools run. Truncated-response continuations share this +budget. Progress shows the recovery count, and the terminal error explains when recovery is exhausted or disabled (`maxAutoContinues: 0`). A new user message starts a fresh budget. See [Retry Policy and Rules](models.md#retry-policy-and-rules). diff --git a/src/eca/features/chat.clj b/src/eca/features/chat.clj index eaef24ae8..9d1762d1c 100644 --- a/src/eca/features/chat.clj +++ b/src/eca/features/chat.clj @@ -1074,6 +1074,8 @@ provider-auth (get-in @db* [:auth provider]) all-tools (f.tools/all-tools chat-id agent @db* config {:full-model full-model}) auto-continue-limit (provider-max-auto-continues config provider) + ;; Consecutive recoveries; reset when a response completes and its tools run. + auto-continue-count* (atom (:auto-continue-count chat-ctx 0)) received-msgs* (atom "") reasonings* (atom {}) server-tool-times* (atom {}) @@ -1285,7 +1287,8 @@ " Try rephrasing or switching to a different model.")}) (swap! db* update-in [:chats chat-id] dissoc :auto-compacting? :compacting?) (lifecycle/finish-chat-prompt-stopped! :idle chat-ctx)) - :finish (let [response-text @received-msgs* + :finish (let [chat-ctx (assoc chat-ctx :auto-continue-count @auto-continue-count*) + response-text @received-msgs* stopping? (identical? :stopping (get-in @db* [:chats chat-id :status]))] (when-not (string/blank? response-text) (add-to-history! {:role "assistant" @@ -1344,7 +1347,9 @@ :on-tools-called (tc/on-tools-called! (assoc chat-ctx :continue-fn (fn [tc-all-tools tc-user-messages] - (let [continue-turn! (fn [] + (reset! auto-continue-count* 0) + (let [chat-ctx (assoc chat-ctx :auto-continue-count 0) + continue-turn! (fn [] (consume-steer-message! chat-id db* chat-ctx add-to-history!) (consume-pending-job-notifications! chat-id db* add-to-history!) {:tools tc-all-tools @@ -1523,7 +1528,8 @@ {:name resolved-name})) nil))) :on-error (fn [{:keys [message exception idle-timeout?] :as error-data}] - (let [{error-type :error/type} (llm-providers.errors/classify-error error-data) + (let [chat-ctx (assoc chat-ctx :auto-continue-count @auto-continue-count*) + {error-type :error/type} (llm-providers.errors/classify-error error-data) db @db* ;; A dead shared connection makes every stacked tool-continuation ;; request fail: only the first error belongs to this prompt, later @@ -1743,7 +1749,7 @@ (str "\n\n" (or message (str "Error: " (or (ex-message exception) (.getName (class exception))))) (case recovery-blocked-reason :disabled "\nAutomatic recovery is disabled for this provider (retry.maxAutoContinues: 0)." - :limit-reached (format "\nAutomatic recovery limit reached (%d/%d for this turn). Send a new message to continue." + :limit-reached (format "\nAutomatic recovery limit reached (%d/%d in a row). Send a new message to continue." auto-continue-count auto-continue-limit) nil) (when-let [resets-at (:rate-limit-resets-at error-data)] diff --git a/test/eca/features/chat_test.clj b/test/eca/features/chat_test.clj index 90af7e8dc..9da15f0e6 100644 --- a/test/eca/features/chat_test.clj +++ b/test/eca/features/chat_test.clj @@ -435,7 +435,7 @@ :text #(and (string/includes? % "bad_record_mac") (string/includes? % (if (zero? limit) "Automatic recovery is disabled" - (format "Automatic recovery limit reached (%d/%d for this turn)" limit limit))))}}])} + (format "Automatic recovery limit reached (%d/%d in a row)" limit limit))))}}])} (h/messages))))))) (deftest truncated-response-shares-recovery-budget-test @@ -460,7 +460,37 @@ (m/embeds [{:role :system :content {:type :progress :text #"Response interrupted.*recovery 1/1"}} {:role :system - :content {:type :text :text #"(?s).*Automatic recovery limit reached \(1/1 for this turn\).*"}}])} + :content {:type :text :text #"(?s).*Automatic recovery limit reached \(1/1 in a row\).*"}}])} + (h/messages))))) + +(deftest recovery-budget-resets-after-completed-step-test + (h/config! {:providers {"openai" {:retry {:maxAutoContinues 1}}}}) + (let [attempts* (atom 0) + exception (javax.net.ssl.SSLException. "Received fatal alert: bad_record_mac") + fail! (fn [on-error] + (on-error {:exception exception + :message (llm-util/connection-error-message exception)})) + {:keys [chat-id]} + (prompt! + {:message "Keep working"} + {:all-tools-mock (constantly [{:name "list_allowed_directories" :full-name "eca__list_allowed_directories" :server {:name "eca"}}]) + :call-tool-mock (constantly {:error false :contents [{:type :text :text "ok"}]}) + :api-mock (fn [{:keys [on-first-response-received on-message-received on-prepare-tool-call on-tools-called on-error]}] + (let [attempt (swap! attempts* inc)] + (on-first-response-received) + (on-message-received {:type :text :text "Partial"}) + (case attempt + 1 (fail! on-error) + 2 (do + (on-prepare-tool-call {:id "call-1" :full-name "eca__list_allowed_directories" :arguments-text ""}) + (on-tools-called [{:id "call-1" :full-name "eca__list_allowed_directories" :arguments {}}]) + (fail! on-error)) + (on-message-received {:type :finish}))))})] + (is (= 3 @attempts*) "a completed step refills the budget, so the second failure is recovered") + (is (nil? (get-in (h/db) [:chats chat-id :prompt-error]))) + (is (match? {:chat-content-received + (m/embeds [{:role :system :content {:type :progress :text #"recovery 1/1"}} + {:role :system :content {:type :progress :text #"recovery 1/1"}}])} (h/messages))))) (deftest stream-error-rejects-preparing-tool-calls-test