Add configurable OpenAI request timeout (#3304)

* Add configurable OpenAI request timeout

* Address AI timeout review feedback

---------

Co-authored-by: Jonathan Kaiser <jaysbeekay@users.noreply.github.com>
This commit is contained in:
jaysbeekay
2026-09-02 03:16:53 +02:00
committed by GitHub
co-authored by Jonathan Kaiser
parent 9714612bb1
commit 3a928a0faf
11 changed files with 107 additions and 18 deletions
@@ -8,6 +8,7 @@ class Settings::HostingsController < ApplicationController
llm_context_window: 256,
llm_max_response_tokens: 64,
llm_max_items_per_call: 1,
openai_request_timeout: Provider::Openai::MIN_REQUEST_TIMEOUT,
ai_response_timeout: Chat::MIN_RESPONSE_TIMEOUT.to_i
}.freeze
@@ -270,9 +271,10 @@ class Settings::HostingsController < ApplicationController
end
private
# Strong parameters for the self-hosting settings form.
def hosting_params
return ActionController::Parameters.new unless params.key?(:setting)
params.require(:setting).permit(:onboarding_state, :require_email_confirmation, :invite_only_default_family_id, :brand_fetch_client_id, :brand_fetch_high_res_logos, :twelve_data_api_key, :tiingo_api_key, :eodhd_api_key, :alpha_vantage_api_key, :tinkoff_invest_api_key, :rentcast_api_key, :realie_api_key, :openai_access_token, :openai_uri_base, :openai_model, :openai_json_mode, :anthropic_access_token, :anthropic_base_url, :anthropic_model, :llm_provider, :llm_context_window, :llm_max_response_tokens, :llm_max_items_per_call, :ai_response_timeout, :exchange_rate_provider, :securities_provider, :syncs_include_pending, :auto_sync_enabled, :auto_sync_time, :external_assistant_url, :external_assistant_token, :external_assistant_agent_id, securities_providers: [])
params.require(:setting).permit(:onboarding_state, :require_email_confirmation, :invite_only_default_family_id, :brand_fetch_client_id, :brand_fetch_high_res_logos, :twelve_data_api_key, :tiingo_api_key, :eodhd_api_key, :alpha_vantage_api_key, :tinkoff_invest_api_key, :rentcast_api_key, :realie_api_key, :openai_access_token, :openai_uri_base, :openai_model, :openai_json_mode, :anthropic_access_token, :anthropic_base_url, :anthropic_model, :llm_provider, :llm_context_window, :llm_max_response_tokens, :llm_max_items_per_call, :openai_request_timeout, :ai_response_timeout, :exchange_rate_provider, :securities_provider, :syncs_include_pending, :auto_sync_enabled, :auto_sync_time, :external_assistant_url, :external_assistant_token, :external_assistant_agent_id, securities_providers: [])
end
def update_assistant_type
+2 -1
View File
@@ -339,11 +339,12 @@ class AiHealth
ENV["OPENAI_ACCESS_TOKEN"].presence || Setting.openai_access_token
end
# Reports the timeout used by normal LLM requests for the selected provider.
def request_timeout(provider)
if provider == :anthropic
ENV.fetch("ANTHROPIC_REQUEST_TIMEOUT", 600).to_i
else
ENV.fetch("OPENAI_REQUEST_TIMEOUT", 60).to_i
Provider::Openai.request_timeout
end
end
+14 -1
View File
@@ -5,6 +5,8 @@ class Provider::Openai < Provider
Error = Class.new(Provider::Error)
DEFAULT_MODEL = "gpt-4.1".freeze
DEFAULT_REQUEST_TIMEOUT = 60
MIN_REQUEST_TIMEOUT = 1
SUPPORTED_MODELS = %w[gpt-4 gpt-5 o1 o3].freeze
VISION_CAPABLE_MODEL_PREFIXES = %w[gpt-4o gpt-4-turbo gpt-4.1 gpt-5 o1 o3].freeze
@@ -18,12 +20,23 @@ class Provider::Openai < Provider
ENV["OPENAI_ACCESS_TOKEN"].present? || Setting.openai_access_token.present?
end
# Effective per-request HTTP timeout for OpenAI-compatible calls.
# Precedence matches other self-hosting settings: ENV > Setting > default.
def self.request_timeout
configured = ENV["OPENAI_REQUEST_TIMEOUT"].to_s.strip.to_i
configured = Setting.openai_request_timeout.to_i unless configured.positive?
return DEFAULT_REQUEST_TIMEOUT unless configured.positive?
[ configured, MIN_REQUEST_TIMEOUT ].max
end
# Builds a client that uses the effective request timeout for every OpenAI call.
def initialize(access_token, uri_base: nil, model: nil)
client_options = { access_token: access_token }
llm_uri_base = uri_base.presence
llm_model = model.presence
client_options[:uri_base] = llm_uri_base if llm_uri_base.present?
client_options[:request_timeout] = ENV.fetch("OPENAI_REQUEST_TIMEOUT", 60).to_i
client_options[:request_timeout] = self.class.request_timeout
@client = ::OpenAI::Client.new(**client_options)
@uri_base = llm_uri_base
+1
View File
@@ -10,6 +10,7 @@ class Setting < RailsSettings::Base
field :openai_uri_base, type: :string, default: ENV["OPENAI_URI_BASE"]
field :openai_model, type: :string, default: ENV["OPENAI_MODEL"]
field :openai_json_mode, type: :string, default: ENV["LLM_JSON_MODE"]
field :openai_request_timeout, type: :integer, default: ENV["OPENAI_REQUEST_TIMEOUT"]&.to_i
field :anthropic_access_token, type: :string, default: ENV["ANTHROPIC_ACCESS_TOKEN"].presence || ENV["ANTHROPIC_API_KEY"].presence
field :anthropic_model, type: :string, default: ENV["ANTHROPIC_MODEL"]
field :anthropic_base_url, type: :string, default: ENV["ANTHROPIC_BASE_URL"]
+2 -1
View File
@@ -7,10 +7,11 @@
# OpenAI manages chunking, embedding, and retrieval; we simply upload files
# and issue search queries.
class VectorStore::Openai < VectorStore::Base
# Builds a vector-store client with the same OpenAI request timeout as chat and batch calls.
def initialize(access_token:, uri_base: nil)
client_options = { access_token: access_token }
client_options[:uri_base] = uri_base if uri_base.present?
client_options[:request_timeout] = ENV.fetch("OPENAI_REQUEST_TIMEOUT", 60).to_i
client_options[:request_timeout] = Provider::Openai.request_timeout
@client = ::OpenAI::Client.new(**client_options)
end
@@ -103,8 +103,17 @@
</div>
<div class="pt-4 border-t border-secondary">
<h3 class="font-medium mb-1"><%= t(".ai_response_timeout_heading") %></h3>
<p class="text-xs text-secondary mb-3"><%= t(".ai_response_timeout_description") %></p>
<h3 class="font-medium mb-1"><%= t(".timeout_heading") %></h3>
<p class="text-xs text-secondary mb-3"><%= t(".timeout_description") %></p>
<%= form.number_field :openai_request_timeout,
label: t(".openai_request_timeout_label"),
placeholder: Provider::Openai::DEFAULT_REQUEST_TIMEOUT.to_s,
value: ENV["OPENAI_REQUEST_TIMEOUT"].present? ? Provider::Openai.request_timeout : Setting.openai_request_timeout,
min: Provider::Openai::MIN_REQUEST_TIMEOUT,
disabled: ENV["OPENAI_REQUEST_TIMEOUT"].present?,
data: { "auto-submit-form-target": "auto" } %>
<p class="text-xs text-secondary mt-1 mb-3"><%= t(".openai_request_timeout_help") %></p>
<%= form.number_field :ai_response_timeout,
label: t(".ai_response_timeout_label"),
@@ -143,10 +143,12 @@ en:
max_response_tokens_help: "Tokens reserved for the model's reply. Default: 512. Lower to free up room for longer history."
max_items_per_call_label: Max Items Per Batch (Optional)
max_items_per_call_help: "Upper bound for auto-categorize / merchant detection batches. Default: 25. Larger batches are auto-sliced to fit the context window."
ai_response_timeout_heading: Chat Response Timeout
ai_response_timeout_description: How long the chat waits for the assistant before showing a "no response" error. Raise this if you run a local model on slow hardware — responses from OpenAI-compatible providers are not streamed, and tool-call rounds display nothing, so the chat can sit on "Thinking…" for the whole turn.
timeout_heading: Timeouts
timeout_description: Controls how long OpenAI-compatible model calls may run before Sure gives up. Raise these if you run a local model on slow hardware.
openai_request_timeout_label: Request Timeout in Seconds (Optional)
openai_request_timeout_help: "Default: 60. Applies to each OpenAI-compatible HTTP request, including auto-categorization, merchant detection, chat, PDF processing, and vector-store calls. OPENAI_REQUEST_TIMEOUT overrides this field."
ai_response_timeout_label: Response Timeout in Seconds (Optional)
ai_response_timeout_help: "Default: 90. This covers the whole turn, so size it as a sum: (1 + ASSISTANT_MAX_TOOL_CALL_ITERATIONS) × OPENAI_REQUEST_TIMEOUT, plus tool execution and queue time. Lowering ASSISTANT_MAX_TOOL_CALL_ITERATIONS reduces how large this needs to be, and is often the better lever on slow hardware."
ai_response_timeout_help: "Default: 90. Applies only to chat's whole-turn watchdog, so size it as a sum: (1 + ASSISTANT_MAX_TOOL_CALL_ITERATIONS) × request timeout, plus tool execution and queue time. AI_RESPONSE_TIMEOUT overrides this field."
title: OpenAI
yahoo_finance_settings:
title: Yahoo Finance
@@ -143,10 +143,12 @@ es:
max_response_tokens_help: 'Tokens reservados para la respuesta del modelo. Predeterminado: 512. Bájalo para dejar sitio a un historial más largo.'
max_items_per_call_label: Elementos máximos por lote (opcional)
max_items_per_call_help: 'Límite superior para los lotes de categorización automática / detección de comercios. Predeterminado: 25. Los lotes más grandes se dividen automáticamente para caber en la ventana de contexto.'
ai_response_timeout_heading: Tiempo de espera de respuesta del chat
ai_response_timeout_description: Cuánto tiempo espera el chat al asistente antes de mostrar un error de "sin respuesta". Auméntalo si ejecutas un modelo local en hardware lento; las respuestas de proveedores compatibles con OpenAI no se transmiten por streaming, y las rondas de llamadas a herramientas no muestran nada, así que el chat puede quedarse en "Pensando..." durante todo el turno.
timeout_heading: Tiempos de espera
timeout_description: Controla cuánto tiempo pueden ejecutarse las llamadas a modelos compatibles con OpenAI antes de que Sure las cancele. Auméntalos si ejecutas un modelo local en hardware lento.
openai_request_timeout_label: Tiempo de espera de solicitud en segundos (opcional)
openai_request_timeout_help: "Predeterminado: 60. Se aplica a cada solicitud HTTP compatible con OpenAI, incluida la categorización automática, la detección de comercios, el chat, el procesamiento de PDF y las llamadas al almacén vectorial. OPENAI_REQUEST_TIMEOUT sobrescribe este campo."
ai_response_timeout_label: Tiempo de espera de respuesta en segundos (opcional)
ai_response_timeout_help: 'Predeterminado: 90. Cubre todo el turno, así que dimensiona este valor como una suma: (1 + ASSISTANT_MAX_TOOL_CALL_ITERATIONS) × OPENAI_REQUEST_TIMEOUT, más el tiempo de ejecución de herramientas y cola. Reducir ASSISTANT_MAX_TOOL_CALL_ITERATIONS disminuye cuánto debe crecer este valor y suele ser la mejor palanca en hardware lento.'
ai_response_timeout_help: "Predeterminado: 90. Solo se aplica al monitor de turno completo del chat, así que dimensiona este valor como una suma: (1 + ASSISTANT_MAX_TOOL_CALL_ITERATIONS) × tiempo de espera de solicitud, más el tiempo de ejecución de herramientas y cola. AI_RESPONSE_TIMEOUT sobrescribe este campo."
title: OpenAI
yahoo_finance_settings:
title: Yahoo Finance
+3 -3
View File
@@ -220,7 +220,7 @@ OPENAI_MODEL=llama3.1:13b
# have enough prompt budget for categories + schemas before transaction rows are added.
LLM_CONTEXT_WINDOW=8192
# Slow local models often need a longer HTTP timeout once the prompt budget issue is fixed.
# Slow local models often need a longer per-request HTTP timeout once the prompt budget issue is fixed.
OPENAI_REQUEST_TIMEOUT=180
# Chained tool calls per turn. Each iteration is another call to the model, so
@@ -245,7 +245,7 @@ AI_DEBUG_MODE=true
- The `OPENAI_ACCESS_TOKEN` can be any non-empty value (Ollama ignores it)
- If you don't set a model, chats will fail with a validation error
- Auto-categorization uses a conservative default `LLM_CONTEXT_WINDOW=2048`, so large category lists or schemas can exhaust the prompt budget before any transactions are sent
- If requests start timing out after raising `LLM_CONTEXT_WINDOW`, increase `OPENAI_REQUEST_TIMEOUT` too; these are separate limits
- If requests start timing out after raising `LLM_CONTEXT_WINDOW`, increase `OPENAI_REQUEST_TIMEOUT` too; these are separate limits. You can also set this in **Settings → Self-Hosting → OpenAI → Request Timeout** when the environment variable is not configured.
- Responses from custom providers are **not streamed** — the chat shows "Thinking…" until the entire reply is generated, and a turn that chains tool calls stays there through every round, since tool-call responses have no text to display. If the chat errors while your model is clearly still working, raise `AI_RESPONSE_TIMEOUT` or lower `ASSISTANT_MAX_TOOL_CALL_ITERATIONS`; `OPENAI_REQUEST_TIMEOUT` alone will not help. `AI_RESPONSE_TIMEOUT` has to cover the whole turn, so size it as `(1 + ASSISTANT_MAX_TOOL_CALL_ITERATIONS) × OPENAI_REQUEST_TIMEOUT` plus tool execution and queue wait — a sum, not simply a larger number than the per-call limit
### Docker Compose Example
@@ -1181,7 +1181,7 @@ Keeping the full eight iterations at 300s per call would instead need `9 × 300
If `AI_RESPONSE_TIMEOUT` ends up below what the turn actually takes, you get a generic "no response" instead of the specific timeout error, and the job keeps running and burning tokens after the chat has given up.
`AI_RESPONSE_TIMEOUT` can also be set at **Settings → Self-Hosting → OpenAI → Chat Response Timeout**, which takes effect without a restart. The environment variable wins if both are set. The minimum accepted value is `30`.
`OPENAI_REQUEST_TIMEOUT` and `AI_RESPONSE_TIMEOUT` can also be set at **Settings → Self-Hosting → OpenAI → Timeouts**, which takes effect without a restart when the corresponding environment variable is not configured. Environment variables win over the settings fields. The minimum accepted chat response timeout is `30`.
Restart `web` and `worker` after changing the environment variables, and make sure your Docker Compose file forwards them into the containers.
@@ -27,7 +27,7 @@ class Settings::HostingsControllerTest < ActionDispatch::IntegrationTest
teardown do
# These tests persist global Setting.* values; reset them so state can't
# leak into later (order-dependent) tests.
%i[anthropic_access_token anthropic_base_url anthropic_model llm_provider twelve_data_api_key openai_access_token external_assistant_token rentcast_api_key realie_api_key].each do |key|
%i[anthropic_access_token anthropic_base_url anthropic_model llm_provider twelve_data_api_key openai_access_token openai_request_timeout ai_response_timeout external_assistant_token rentcast_api_key realie_api_key].each do |key|
Setting.public_send("#{key}=", nil)
end
end
@@ -550,28 +550,38 @@ class Settings::HostingsControllerTest < ActionDispatch::IntegrationTest
patch settings_hosting_url, params: { setting: {
llm_context_window: "4096",
llm_max_response_tokens: "1024",
llm_max_items_per_call: "40"
llm_max_items_per_call: "40",
openai_request_timeout: "180",
ai_response_timeout: "240"
} }
assert_redirected_to settings_hosting_url
assert_equal 4096, Setting.llm_context_window
assert_equal 1024, Setting.llm_max_response_tokens
assert_equal 40, Setting.llm_max_items_per_call
assert_equal 180, Setting.openai_request_timeout
assert_equal 240, Setting.ai_response_timeout
patch settings_hosting_url, params: { setting: {
llm_context_window: "",
llm_max_response_tokens: "",
llm_max_items_per_call: ""
llm_max_items_per_call: "",
openai_request_timeout: "",
ai_response_timeout: ""
} }
assert_nil Setting.llm_context_window
assert_nil Setting.llm_max_response_tokens
assert_nil Setting.llm_max_items_per_call
assert_nil Setting.openai_request_timeout
assert_nil Setting.ai_response_timeout
end
ensure
Setting.llm_context_window = nil
Setting.llm_max_response_tokens = nil
Setting.llm_max_items_per_call = nil
Setting.openai_request_timeout = nil
Setting.ai_response_timeout = nil
end
test "rejects llm budget below field minimum" do
@@ -593,11 +603,33 @@ class Settings::HostingsControllerTest < ActionDispatch::IntegrationTest
assert_response :unprocessable_entity
assert_match(/must be a whole number/, flash[:alert])
assert_nil Setting.llm_max_items_per_call
patch settings_hosting_url, params: { setting: { openai_request_timeout: "0" } }
assert_response :unprocessable_entity
assert_match(/must be a whole number/, flash[:alert])
assert_nil Setting.openai_request_timeout
end
ensure
Setting.llm_context_window = nil
Setting.llm_max_response_tokens = nil
Setting.llm_max_items_per_call = nil
Setting.openai_request_timeout = nil
end
test "shows environment backed OpenAI request timeout when field is disabled" do
with_self_hosting do
Setting.openai_request_timeout = 180
with_env_overrides("OPENAI_REQUEST_TIMEOUT" => "300") do
get settings_hosting_url
assert_response :success
assert_select "input[name='setting[openai_request_timeout]'][value='300'][disabled='disabled']"
end
end
ensure
Setting.openai_request_timeout = nil
end
test "can clear data only when admin" do
+26
View File
@@ -8,6 +8,32 @@ class Provider::OpenaiTest < ActiveSupport::TestCase
@subject_model = "gpt-4.1"
end
test "request_timeout uses ENV then Setting then default" do
Setting.stubs(:openai_request_timeout).returns(nil)
with_env_overrides("OPENAI_REQUEST_TIMEOUT" => nil) do
assert_equal Provider::Openai::DEFAULT_REQUEST_TIMEOUT, Provider::Openai.request_timeout
end
Setting.stubs(:openai_request_timeout).returns(180)
with_env_overrides("OPENAI_REQUEST_TIMEOUT" => nil) do
assert_equal 180, Provider::Openai.request_timeout
end
Setting.stubs(:openai_request_timeout).returns(180)
with_env_overrides("OPENAI_REQUEST_TIMEOUT" => "300") do
assert_equal 300, Provider::Openai.request_timeout
end
end
test "request_timeout is passed to OpenAI client" do
with_env_overrides("OPENAI_REQUEST_TIMEOUT" => nil) do
Setting.stubs(:openai_request_timeout).returns(180)
::OpenAI::Client.expects(:new).with(access_token: "test-token", request_timeout: 180).returns(mock)
Provider::Openai.new("test-token")
end
end
test "openai errors are automatically raised" do
VCR.use_cassette("openai/chat/error") do
response = @openai.chat_response("Test", model: "invalid-model-that-will-trigger-api-error")