Add live AI checks to system health (#3155)

* Add live AI checks to system health

Give super admins a dedicated AI status view with bounded liveness probes for LLMs, vector stores, pgvector, and embedding endpoints. Record sanitized failures in both the system debug log and Rails logger, and document the recommended local configuration.\n\nCloses #3145

* Fix AI health CI checks

* Address AI health review feedback

* Correct Ollama model preload guidance

* Distinguish OpenAI-compatible providers

* Make Ollama startup readiness explicit

* Recognize Cloudflare AI endpoints
This commit is contained in:
Juan José Mata
2026-08-24 22:41:08 +02:00
committed by GitHub
parent c26b16d36f
commit fd6f4ff078
24 changed files with 1773 additions and 120 deletions
+11
View File
@@ -64,6 +64,17 @@ OPENAI_URI_BASE=
# OPENAI_SUPPORTS_RESPONSES_ENDPOINT= # Override Responses-API vs chat.completions routing
# LLM_JSON_MODE= # auto | strict | json_object | none
# Optional: document-search vector store. Hosted OpenAI is selected by default
# when OPENAI_ACCESS_TOKEN is configured. For a local OpenAI-compatible LLM,
# use pgvector plus a separate OpenAI-compatible embeddings endpoint.
# VECTOR_STORE_PROVIDER=pgvector # openai | pgvector | qdrant (scaffolded only)
# EMBEDDING_URI_BASE=http://ollama:11434/v1
# EMBEDDING_MODEL=mxbai-embed-large
# EMBEDDING_DIMENSIONS=1024 # Must match the embedding model
# EMBEDDING_ACCESS_TOKEN= # Optional; falls back to OPENAI_ACCESS_TOKEN
# AI_HEALTH_PROBE_TIMEOUT=5 # Per-request timeout for admin live checks
# AI_HEALTH_PROBE_CACHE_TTL=60 # Cache live results and deduplicate failure logs
# Optional: External AI Assistant — delegates chat to a remote AI agent
# instead of calling LLMs directly. The agent calls back to Sure's /mcp endpoint.
# See docs/hosting/ai.md for full details.
+11
View File
@@ -52,6 +52,17 @@ OPENAI_MODEL =
# OPENAI_SUPPORTS_RESPONSES_ENDPOINT = # true to force Responses API on custom providers
# LLM_JSON_MODE = # auto | strict | json_object | none
# Document-search vector store. Local OpenAI-compatible chat endpoints usually
# do not implement OpenAI's hosted /v1/vector_stores API, so use pgvector and a
# separate embeddings endpoint when testing local AI.
# VECTOR_STORE_PROVIDER = pgvector
# EMBEDDING_URI_BASE = http://host.docker.internal:11434/v1
# EMBEDDING_MODEL = mxbai-embed-large
# EMBEDDING_DIMENSIONS = 1024 # Must match the embedding model
# EMBEDDING_ACCESS_TOKEN = # Optional; falls back to OPENAI_ACCESS_TOKEN
# AI_HEALTH_PROBE_TIMEOUT = 5 # Per-request timeout for admin live checks
# AI_HEALTH_PROBE_CACHE_TTL = 60 # Cache live results and deduplicate failure logs
# (example: LM Studio/Docker config) OpenAI-compatible API endpoint config
# OPENAI_URI_BASE = http://host.docker.internal:1234/
# OPENAI_MODEL = qwen/qwen3-vl-4b
+1
View File
@@ -3,6 +3,7 @@
testid: testid,
DS__tabs_session_key_value: session_key,
DS__tabs_url_param_key_value: url_param_key,
DS__tabs_navigate_on_change_value: navigate_on_change,
DS__tabs_nav_btn_active_class: active_btn_classes,
DS__tabs_nav_btn_inactive_class: inactive_btn_classes
} do %>
+3 -2
View File
@@ -47,12 +47,13 @@ class DS::Tabs < DesignSystemComponent
}
}
attr_reader :active_tab, :url_param_key, :session_key, :variant, :testid
attr_reader :active_tab, :url_param_key, :session_key, :variant, :testid, :navigate_on_change
def initialize(active_tab:, url_param_key: nil, session_key: nil, variant: :default, active_btn_classes: "", inactive_btn_classes: "", testid: nil)
def initialize(active_tab:, url_param_key: nil, session_key: nil, navigate_on_change: false, variant: :default, active_btn_classes: "", inactive_btn_classes: "", testid: nil)
@active_tab = active_tab
@url_param_key = url_param_key
@session_key = session_key
@navigate_on_change = navigate_on_change
@variant = variant.to_sym
@active_btn_classes = active_btn_classes
@inactive_btn_classes = inactive_btn_classes
+11 -1
View File
@@ -4,7 +4,11 @@ import { Controller } from "@hotwired/stimulus";
export default class extends Controller {
static classes = ["navBtnActive", "navBtnInactive"];
static targets = ["panel", "navBtn"];
static values = { sessionKey: String, urlParamKey: String };
static values = {
sessionKey: String,
urlParamKey: String,
navigateOnChange: Boolean,
};
show(e) {
const btn = e.target.closest("button");
@@ -37,6 +41,12 @@ export default class extends Controller {
if (this.urlParamKeyValue) {
const url = new URL(window.location.href);
url.searchParams.set(this.urlParamKeyValue, selectedTabId);
if (this.navigateOnChangeValue) {
window.location.assign(url.toString());
return;
}
window.history.replaceState({}, "", url);
}
@@ -10,6 +10,11 @@ module Admin
def show
SidekiqHealth.expire_cache!
@health = SidekiqHealth.new
ai_tab = params[:tab] == "ai"
@ai_health = AiHealth.new(
run_probes: ai_tab,
force_probes: ai_tab && params[:refresh_ai_health] == "1"
)
end
end
end
+310
View File
@@ -0,0 +1,310 @@
# frozen_string_literal: true
require "uri"
# Snapshot of AI configuration and bounded, non-destructive liveness checks for
# operators diagnosing chat, PDF import, and document-search failures.
class AiHealth
OPENAI_DEFAULT_ENDPOINT = "https://api.openai.com/v1".freeze
ANTHROPIC_DEFAULT_ENDPOINT = "https://api.anthropic.com".freeze
OPENAI_COMPATIBLE_PROVIDER_DOMAINS = {
openrouter: %w[openrouter.ai],
together: %w[together.ai together.xyz],
kilo: %w[kilo.ai],
cloudflare: %w[api.cloudflare.com gateway.ai.cloudflare.com]
}.freeze
attr_reader :selected_llm_provider, :effective_llm_provider, :llm_model,
:llm_endpoint, :llm_request_timeout, :openai_endpoint, :vector_store_adapter,
:embedding_endpoint, :embedding_model, :embedding_dimensions,
:pgvector_extension_available, :pgvector_extension_enabled,
:pgvector_table_available, :qdrant_endpoint, :llm_probe,
:vector_store_probe, :embedding_probe
def initialize(run_probes: true, force_probes: false)
@run_probes = run_probes
@force_probes = force_probes
load_llm_status
load_vector_store_status
load_probes
end
def openai_credentials_configured?
@openai_credentials_configured
end
def anthropic_credentials_configured?
@anthropic_credentials_configured
end
def llm_configured?
@llm_provider.present?
end
def llm_status
return :not_configured unless llm_configured?
llm_probe.status
end
def llm_fallback?
@effective_llm_protocol.present? && @effective_llm_protocol != @selected_llm_protocol
end
def openai_compatible_endpoint?
@openai_custom_endpoint
end
def pdf_processing_supported?
@pdf_processing_supported == true
end
def pdf_processing_status
return :unavailable unless llm_configured?
pdf_processing_supported? ? :supported : :unsupported
end
def vector_store_configured?
@vector_store_configured
end
def vector_store_status
return :missing if vector_store_adapter.nil?
return :scaffolded if vector_store_adapter == :qdrant
return :not_checked unless run_probes?
return :failing if vector_store_probe.failing?
return :not_configured unless vector_store_configured?
return :failing unless vector_store_probe.passing?
return :failing if vector_store_adapter == :pgvector && !embedding_probe.passing?
:passing
end
def openai_vector_store_uses_custom_endpoint?
vector_store_adapter == :openai && @openai_custom_endpoint
end
def last_checked_at
[ llm_probe, vector_store_probe, embedding_probe ].filter_map(&:checked_at).max
end
def self.redact_endpoint(value)
return if value.blank?
uri = URI.parse(value.to_s)
uri.user = nil if uri.respond_to?(:user=)
uri.password = nil if uri.respond_to?(:password=)
uri.query = nil if uri.respond_to?(:query=)
uri.fragment = nil if uri.respond_to?(:fragment=)
uri.to_s
rescue URI::InvalidURIError
value.to_s
.sub(%r{\A([^:]+://)[^/@]+@}, "\\1")
.split(/[?#]/, 2)
.first
end
private
def load_llm_status
@selected_llm_protocol = normalized_llm_provider(Setting.llm_provider)
@openai_custom_endpoint = openai_uri_base.present? && !hosted_openai_endpoint?(openai_uri_base)
@selected_llm_provider = selected_provider_name(@selected_llm_protocol)
@openai_credentials_configured = safely(false) { Provider::Openai.configured? }
@anthropic_credentials_configured = safely(false) { Provider::Anthropic.configured? }
@llm_provider = safely(nil) { Provider::Registry.preferred_llm_provider }
@effective_llm_protocol = protocol_name(@llm_provider)
@effective_llm_provider = effective_provider_name(@effective_llm_protocol)
provider_for_details = @effective_llm_protocol || @selected_llm_protocol
@llm_model = effective_model(provider_for_details)
@llm_endpoint = endpoint(provider_for_details)
@llm_request_timeout = request_timeout(provider_for_details)
@pdf_processing_supported = safely(false) do
@llm_provider&.supports_pdf_processing?(model: llm_model)
end
@openai_endpoint = redact_endpoint(openai_uri_base.presence || OPENAI_DEFAULT_ENDPOINT)
@llm_access_token = access_token(@effective_llm_protocol)
@llm_raw_endpoint = raw_endpoint(@effective_llm_protocol).presence || default_endpoint(@effective_llm_protocol)
end
def load_vector_store_status
@vector_store_adapter = safely(nil) { VectorStore::Registry.adapter_name }
@vector_store_configured = safely(false) { VectorStore.configured? }
case vector_store_adapter
when :pgvector
load_pgvector_status
@embedding_model = VectorStore.embedding_model
@embedding_dimensions = VectorStore.embedding_dimensions
@embedding_raw_endpoint = VectorStore.embedding_uri_base
@embedding_endpoint = redact_endpoint(@embedding_raw_endpoint)
@embedding_access_token = VectorStore.embedding_access_token
when :qdrant
@qdrant_endpoint = redact_endpoint(ENV.fetch("QDRANT_URL", "http://localhost:6333"))
end
end
def load_probes
@llm_probe = llm_configured? ? Probe.not_checked : Probe.not_configured
@vector_store_probe = vector_store_adapter.present? ? Probe.not_checked : Probe.not_configured
@embedding_probe = vector_store_adapter == :pgvector ? Probe.not_checked : Probe.not_configured
return unless run_probes?
probe = Probe.new(force: @force_probes)
if llm_configured?
@llm_probe = probe.llm(
provider: @effective_llm_protocol,
endpoint: @llm_raw_endpoint,
access_token: @llm_access_token,
model: llm_model
)
end
case vector_store_adapter
when :openai
if vector_store_configured?
@vector_store_probe = probe.openai_vector_store(
endpoint: openai_uri_base.presence || OPENAI_DEFAULT_ENDPOINT,
access_token: openai_access_token
)
end
when :pgvector
@vector_store_probe = probe.pgvector
@embedding_probe = probe.embedding(
endpoint: @embedding_raw_endpoint,
access_token: @embedding_access_token,
model: embedding_model,
dimensions: embedding_dimensions
)
end
end
def run_probes?
@run_probes
end
def load_pgvector_status
connection = ActiveRecord::Base.connection
@pgvector_table_available = connection.table_exists?(VectorStore::Pgvector::TABLE_NAME)
@pgvector_extension_enabled = connection.extension_enabled?("vector")
@pgvector_extension_available = @pgvector_extension_enabled || connection.select_value(
"SELECT 1 FROM pg_available_extensions WHERE name = 'vector' LIMIT 1"
).present?
rescue StandardError
@pgvector_table_available = false
@pgvector_extension_enabled = false
@pgvector_extension_available = false
end
def normalized_llm_provider(value)
value.to_s == "anthropic" ? :anthropic : :openai
end
def protocol_name(provider)
case provider
when Provider::Openai then :openai
when Provider::Anthropic then :anthropic
end
end
def selected_provider_name(protocol)
protocol == :openai && @openai_custom_endpoint ? :openai_compatible : protocol
end
def effective_provider_name(protocol)
return protocol unless protocol == :openai
return :openai unless @openai_custom_endpoint
openai_compatible_provider_name(openai_uri_base)
end
def openai_compatible_provider_name(value)
uri = URI.parse(value.to_s)
host = uri.host.to_s.downcase
return :ollama if ollama_endpoint?(uri, host)
OPENAI_COMPATIBLE_PROVIDER_DOMAINS.each do |provider, domains|
return provider if domains.any? { |domain| host == domain || host.end_with?(".#{domain}") }
end
:custom_openai_compatible
rescue URI::InvalidURIError
:custom_openai_compatible
end
def ollama_endpoint?(uri, host)
uri.port == 11_434 || host == "ollama" || host.end_with?(".ollama")
end
def effective_model(provider)
case provider
when :anthropic then Provider::Anthropic.effective_model
else Provider::Openai.effective_model
end
end
def endpoint(provider)
value = raw_endpoint(provider)
redact_endpoint(value.presence || default_endpoint(provider))
end
def default_endpoint(provider)
provider == :anthropic ? ANTHROPIC_DEFAULT_ENDPOINT : OPENAI_DEFAULT_ENDPOINT
end
def raw_endpoint(provider)
provider == :anthropic ? anthropic_base_url : openai_uri_base
end
def access_token(provider)
if provider == :anthropic
ENV["ANTHROPIC_ACCESS_TOKEN"].presence ||
ENV["ANTHROPIC_API_KEY"].presence ||
Setting.anthropic_access_token
else
openai_access_token
end
end
def openai_access_token
ENV["OPENAI_ACCESS_TOKEN"].presence || Setting.openai_access_token
end
def request_timeout(provider)
if provider == :anthropic
ENV.fetch("ANTHROPIC_REQUEST_TIMEOUT", 600).to_i
else
ENV.fetch("OPENAI_REQUEST_TIMEOUT", 60).to_i
end
end
def openai_uri_base
ENV["OPENAI_URI_BASE"].presence || Setting.openai_uri_base
end
def anthropic_base_url
ENV["ANTHROPIC_BASE_URL"].presence || Setting.anthropic_base_url
end
def hosted_openai_endpoint?(value)
uri = URI.parse(value.to_s)
normalized_path = uri.path.to_s.sub(%r{/+\z}, "")
uri.scheme == "https" && uri.host == "api.openai.com" && uri.port == 443 && normalized_path.in?([ "", "/v1" ])
rescue URI::InvalidURIError
false
end
def redact_endpoint(value)
self.class.redact_endpoint(value)
end
def safely(fallback)
yield
rescue StandardError
fallback
end
end
+235
View File
@@ -0,0 +1,235 @@
# frozen_string_literal: true
require "digest"
class AiHealth
# Performs bounded, non-destructive checks against the exact services used
# by Sure. Results are cached briefly because this runs in an admin request,
# and a failed check is recorded once per cache fill in both operator-facing
# debug logs and the application log.
class Probe
CACHE_NAMESPACE = "ai_health/probes/v1"
DEFAULT_CACHE_TTL = 60.seconds
DEFAULT_TIMEOUT = 5
EMBEDDING_TEST_INPUT = "Sure AI health check"
Result = Data.define(:status, :checked_at, :failure_code, :http_status) do
def passing?
status == :passing
end
def failing?
status == :failing
end
end
class Failure < StandardError
attr_reader :failure_code
def initialize(failure_code)
@failure_code = failure_code
super(failure_code.to_s)
end
end
def self.not_checked
Result.new(status: :not_checked, checked_at: nil, failure_code: nil, http_status: nil)
end
def self.not_configured
Result.new(status: :not_configured, checked_at: nil, failure_code: nil, http_status: nil)
end
def initialize(force: false, cache: Rails.cache)
@force = force
@cache = cache
end
def llm(provider:, endpoint:, access_token:, model:)
run(
component: "llm",
provider_key: provider,
endpoint: endpoint,
model: model,
credential: access_token
) do
model_available = case provider
when :openai
response = openai_client(access_token:, endpoint:).models.list
openai_model_ids(response).include?(model)
when :anthropic
model_info = anthropic_client(access_token:, endpoint:).models.retrieve(model)
model_info.respond_to?(:id) && model_info.id.present?
else
raise Failure, :unsupported_provider
end
raise Failure, :model_not_available unless model_available
end
end
def openai_vector_store(endpoint:, access_token:)
run(
component: "vector_store",
provider_key: :openai,
endpoint: endpoint,
credential: access_token
) do
response = openai_client(access_token:, endpoint:).vector_stores.list(parameters: { limit: 1 })
raise Failure, :invalid_response unless response.is_a?(Hash) && response["data"].is_a?(Array)
end
end
def pgvector(connection: ActiveRecord::Base.connection)
run(component: "vector_store", provider_key: :pgvector) do
raise Failure, :extension_not_enabled unless connection.extension_enabled?("vector")
raise Failure, :table_not_found unless connection.table_exists?(VectorStore::Pgvector::TABLE_NAME)
table = connection.quote_table_name(VectorStore::Pgvector::TABLE_NAME)
connection.select_value("SELECT 1 FROM #{table} LIMIT 1")
end
end
def embedding(endpoint:, access_token:, model:, dimensions:)
run(
component: "embedding",
provider_key: :openai_compatible,
endpoint: endpoint,
model: model,
credential: access_token,
dimensions: dimensions
) do
response = embedding_client(endpoint:, access_token:).post("embeddings") do |request|
request.body = { model: model, input: EMBEDDING_TEST_INPUT }
end
vector = response.body.dig("data", 0, "embedding") if response.body.is_a?(Hash)
raise Failure, :invalid_response unless vector.is_a?(Array)
raise Failure, :dimensions_mismatch unless vector.length == dimensions
end
end
private
attr_reader :cache, :force
def run(component:, provider_key:, endpoint: nil, model: nil, credential: nil, dimensions: nil)
key = cache_key(component:, provider_key:, endpoint:, model:, credential:, dimensions:)
cache.delete(key) if force
result = nil
cache.fetch(key, expires_in: cache_ttl) do
result = perform(component:, provider_key:, endpoint:, model:) { yield }
end
rescue StandardError => error
record_cache_failure(error)
result || perform(component:, provider_key:, endpoint:, model:) { yield }
end
def perform(component:, provider_key:, endpoint:, model:)
yield
Result.new(status: :passing, checked_at: Time.current, failure_code: nil, http_status: nil)
rescue StandardError => error
result = Result.new(
status: :failing,
checked_at: Time.current,
failure_code: failure_code(error),
http_status: http_status(error)
)
record_failure(component:, provider_key:, endpoint:, model:, error:, result:)
result
end
def openai_client(access_token:, endpoint:)
options = { access_token: access_token, request_timeout: timeout }
options[:uri_base] = endpoint if endpoint.present?
::OpenAI::Client.new(**options)
end
def anthropic_client(access_token:, endpoint:)
options = { api_key: access_token, max_retries: 0, timeout: timeout }
options[:base_url] = endpoint if endpoint.present?
::Anthropic::Client.new(**options)
end
def embedding_client(endpoint:, access_token:)
Faraday.new(url: endpoint) do |faraday|
faraday.request :json
faraday.response :json
faraday.response :raise_error
faraday.headers["Authorization"] = "Bearer #{access_token}" if access_token.present?
faraday.options.timeout = timeout
faraday.options.open_timeout = [ timeout, 3 ].min
end
end
def openai_model_ids(response)
raise Failure, :invalid_response unless response.is_a?(Hash) && response["data"].is_a?(Array)
response["data"].filter_map { |item| item["id"] || item[:id] }
end
def cache_key(component:, provider_key:, endpoint:, model:, credential:, dimensions:)
fingerprint = Digest::SHA256.hexdigest(
[ component, provider_key, endpoint, model, credential, dimensions ].join("\0")
)
"#{CACHE_NAMESPACE}/#{fingerprint}"
end
def cache_ttl
seconds = ENV.fetch("AI_HEALTH_PROBE_CACHE_TTL", DEFAULT_CACHE_TTL.to_i).to_i
seconds.positive? ? seconds.seconds : DEFAULT_CACHE_TTL
end
def timeout
seconds = ENV.fetch("AI_HEALTH_PROBE_TIMEOUT", DEFAULT_TIMEOUT).to_i
seconds.positive? ? seconds : DEFAULT_TIMEOUT
end
def failure_code(error)
return error.failure_code if error.respond_to?(:failure_code)
error.is_a?(Faraday::TimeoutError) ? :timeout : :request_failed
end
def http_status(error)
response = error.respond_to?(:response) ? error.response : nil
return response[:status] || response["status"] if response.is_a?(Hash)
error.status if error.respond_to?(:status)
end
def record_failure(component:, provider_key:, endpoint:, model:, error:, result:)
message = "AI health #{component.tr('_', ' ')} liveness probe failed"
metadata = {
component: component,
endpoint: AiHealth.redact_endpoint(endpoint),
model: model,
failure_code: result.failure_code,
exception_class: error.class.name,
http_status: result.http_status
}.compact
Rails.logger.error("#{message}: #{metadata.to_json}")
DebugLogEntry.capture(
category: "ai_health",
level: "error",
message: message,
source: self.class.name,
provider_key: provider_key.to_s,
metadata: metadata
)
end
def record_cache_failure(error)
message = "AI health probe cache failed"
Rails.logger.warn("#{message}: #{error.class.name}")
DebugLogEntry.capture(
category: "ai_health",
level: "warn",
message: message,
source: self.class.name,
metadata: { exception_class: error.class.name }
)
end
end
end
+20
View File
@@ -4,6 +4,10 @@ module VectorStore
Response = Data.define(:success?, :data, :error)
DEFAULT_EMBEDDING_MODEL = "mxbai-embed-large".freeze
DEFAULT_EMBEDDING_DIMENSIONS = 1024
DEFAULT_EMBEDDING_URI_BASE = "https://api.openai.com/v1/".freeze
def self.adapter
Registry.adapter
end
@@ -11,4 +15,20 @@ module VectorStore
def self.configured?
Registry.configured?
end
def self.embedding_model
ENV.fetch("EMBEDDING_MODEL", DEFAULT_EMBEDDING_MODEL)
end
def self.embedding_dimensions
ENV.fetch("EMBEDDING_DIMENSIONS", DEFAULT_EMBEDDING_DIMENSIONS).to_i
end
def self.embedding_uri_base
ENV["EMBEDDING_URI_BASE"].presence || ENV["OPENAI_URI_BASE"].presence || DEFAULT_EMBEDDING_URI_BASE
end
def self.embedding_access_token
ENV["EMBEDDING_ACCESS_TOKEN"].presence || ENV["OPENAI_ACCESS_TOKEN"].presence
end
end
+4 -4
View File
@@ -135,18 +135,18 @@ module VectorStore::Embeddable
end
def embedding_model
ENV.fetch("EMBEDDING_MODEL", "nomic-embed-text")
VectorStore.embedding_model
end
def embedding_dimensions
ENV.fetch("EMBEDDING_DIMENSIONS", "1024").to_i
VectorStore.embedding_dimensions
end
def embedding_uri_base
ENV["EMBEDDING_URI_BASE"].presence || ENV["OPENAI_URI_BASE"].presence || "https://api.openai.com/v1/"
VectorStore.embedding_uri_base
end
def embedding_access_token
ENV["EMBEDDING_ACCESS_TOKEN"].presence || ENV["OPENAI_ACCESS_TOKEN"].presence
VectorStore.embedding_access_token
end
end
@@ -0,0 +1,261 @@
<div class="space-y-4" data-testid="ai-status">
<div class="flex flex-wrap items-center justify-between gap-3">
<p class="text-sm text-secondary">
<% if ai_health.last_checked_at %>
<%= t("admin.system_health.show.ai.last_checked", time_ago: time_ago_in_words(ai_health.last_checked_at)) %>
<% else %>
<%= t("admin.system_health.show.ai.not_checked_help") %>
<% end %>
</p>
<%= render DS::Button.new(
href: admin_system_health_path(tab: "ai", refresh_ai_health: "1"),
method: :get,
variant: :outline,
size: :sm,
icon: "refresh-cw",
text: t("admin.system_health.show.ai.run_checks")
) %>
</div>
<% if ai_health.llm_status == :failing %>
<%= render DS::Alert.new(
title: t("admin.system_health.show.ai.alerts.llm_probe_failed.title"),
message: t("admin.system_health.show.ai.alerts.llm_probe_failed.message"),
variant: :error
) %>
<% end %>
<% case ai_health.vector_store_status %>
<% when :failing %>
<% if ai_health.openai_vector_store_uses_custom_endpoint? %>
<%= render DS::Alert.new(title: t("admin.system_health.show.ai.alerts.custom_openai.title"), variant: :error) do %>
<p><%= t("admin.system_health.show.ai.alerts.custom_openai.message") %></p>
<% end %>
<% else %>
<%= render DS::Alert.new(
title: t("admin.system_health.show.ai.alerts.vector_probe_failed.title"),
message: t("admin.system_health.show.ai.alerts.vector_probe_failed.message"),
variant: :error
) %>
<% end %>
<% when :scaffolded %>
<%= render DS::Alert.new(
title: t("admin.system_health.show.ai.alerts.qdrant.title"),
message: t("admin.system_health.show.ai.alerts.qdrant.message"),
variant: :warning
) %>
<% when :missing %>
<%= render DS::Alert.new(
title: t("admin.system_health.show.ai.alerts.missing_vector_store.title"),
message: t("admin.system_health.show.ai.alerts.missing_vector_store.message"),
variant: :warning
) %>
<% when :not_configured %>
<%= render DS::Alert.new(
title: t("admin.system_health.show.ai.alerts.unavailable_vector_store.title"),
message: t("admin.system_health.show.ai.alerts.unavailable_vector_store.message"),
variant: :warning
) %>
<% end %>
<%= render DS::Card.new(class: "gap-4") do %>
<% llm_tone = { passing: :success, failing: :error, not_checked: :neutral, not_configured: :warning }.fetch(ai_health.llm_status) %>
<% llm_icon = { passing: "circle-check", failing: "circle-x", not_checked: "circle-help", not_configured: "triangle-alert" }.fetch(ai_health.llm_status) %>
<div class="flex flex-wrap items-start justify-between gap-3">
<div>
<h2 class="text-lg font-semibold text-primary"><%= t("admin.system_health.show.ai.llm.title") %></h2>
<p class="mt-1 text-sm text-secondary"><%= t("admin.system_health.show.ai.llm.description") %></p>
</div>
<%= render DS::Pill.new(
label: t("admin.system_health.show.ai.probe_statuses.#{ai_health.llm_status}"),
tone: llm_tone,
marker: false,
icon: llm_icon
) %>
</div>
<dl class="grid grid-cols-1 gap-4 text-sm sm:grid-cols-2 lg:grid-cols-3">
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.ai.labels.selected_provider") %></dt>
<dd class="mt-1 font-medium text-primary" data-testid="selected-llm-provider"><%= t("admin.system_health.show.ai.providers.#{ai_health.selected_llm_provider}") %></dd>
</div>
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.ai.labels.effective_provider") %></dt>
<dd class="mt-1 font-medium text-primary" data-testid="effective-llm-provider">
<% if ai_health.effective_llm_provider %>
<%= t("admin.system_health.show.ai.providers.#{ai_health.effective_llm_provider}") %>
<% if ai_health.llm_fallback? %>
<span class="text-warning"><%= t("admin.system_health.show.ai.values.fallback") %></span>
<% end %>
<% else %>
<span class="text-warning"><%= t("admin.system_health.show.ai.values.not_configured") %></span>
<% end %>
</dd>
</div>
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.ai.labels.model") %></dt>
<dd class="mt-1 break-all font-mono text-primary"><%= ai_health.llm_model.presence || t("admin.system_health.show.ai.values.not_set") %></dd>
</div>
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.ai.labels.endpoint") %></dt>
<dd class="mt-1 break-all font-mono text-primary"><%= ai_health.llm_endpoint %></dd>
</div>
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.ai.labels.pdf_processing") %></dt>
<dd class="mt-1 font-medium <%= ai_health.pdf_processing_supported? ? "text-success" : "text-warning" %>">
<%= t("admin.system_health.show.ai.pdf_statuses.#{ai_health.pdf_processing_status}") %>
</dd>
</div>
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.ai.labels.request_timeout") %></dt>
<dd class="mt-1 font-medium text-primary"><%= t("admin.system_health.show.values.seconds", seconds: ai_health.llm_request_timeout) %></dd>
</div>
<div>
<dt class="text-secondary">
<%= t("admin.system_health.show.ai.labels.#{ai_health.openai_compatible_endpoint? ? :openai_compatible_credentials : :openai_credentials}") %>
</dt>
<dd class="mt-1 font-medium text-primary">
<%= t("admin.system_health.show.ai.values.#{ai_health.openai_credentials_configured? ? :configured : :not_configured}") %>
</dd>
</div>
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.ai.labels.anthropic_credentials") %></dt>
<dd class="mt-1 font-medium text-primary">
<%= t("admin.system_health.show.ai.values.#{ai_health.anthropic_credentials_configured? ? :configured : :not_configured}") %>
</dd>
</div>
<% if ai_health.llm_probe.failure_code %>
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.ai.labels.failure_reason") %></dt>
<dd class="mt-1 font-medium text-destructive"><%= t("admin.system_health.show.ai.failure_codes.#{ai_health.llm_probe.failure_code}") %></dd>
</div>
<% end %>
</dl>
<% end %>
<%= render DS::Card.new(class: "gap-4") do %>
<% vector_status = ai_health.vector_store_status %>
<% vector_tone = { passing: :success, failing: :error, not_checked: :neutral, not_configured: :warning, scaffolded: :warning, missing: :warning }.fetch(vector_status) %>
<% vector_icon = vector_status == :passing ? "circle-check" : (vector_status == :failing ? "circle-x" : "triangle-alert") %>
<div class="flex flex-wrap items-start justify-between gap-3">
<div>
<h2 class="text-lg font-semibold text-primary"><%= t("admin.system_health.show.ai.vector_store.title") %></h2>
<p class="mt-1 text-sm text-secondary"><%= t("admin.system_health.show.ai.vector_store.description") %></p>
</div>
<%= render DS::Pill.new(
label: t("admin.system_health.show.ai.vector_statuses.#{vector_status}"),
tone: vector_tone,
marker: false,
icon: vector_icon
) %>
</div>
<dl class="grid grid-cols-1 gap-4 text-sm sm:grid-cols-2 lg:grid-cols-3">
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.ai.labels.adapter") %></dt>
<dd class="mt-1 font-medium text-primary">
<% if ai_health.vector_store_adapter %>
<%= t("admin.system_health.show.ai.adapters.#{ai_health.vector_store_adapter}") %>
<% else %>
<span class="text-warning"><%= t("admin.system_health.show.ai.values.not_configured") %></span>
<% end %>
</dd>
</div>
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.ai.labels.adapter_available") %></dt>
<dd class="mt-1 font-medium <%= ai_health.vector_store_configured? ? "text-success" : "text-warning" %>">
<%= t("admin.system_health.show.ai.values.#{ai_health.vector_store_configured? ? :yes : :no}") %>
</dd>
</div>
<% case ai_health.vector_store_adapter %>
<% when :openai %>
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.ai.labels.endpoint") %></dt>
<dd class="mt-1 break-all font-mono text-primary"><%= ai_health.openai_endpoint %></dd>
</div>
<% if ai_health.vector_store_probe.failure_code %>
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.ai.labels.failure_reason") %></dt>
<dd class="mt-1 font-medium text-destructive"><%= t("admin.system_health.show.ai.failure_codes.#{ai_health.vector_store_probe.failure_code}") %></dd>
</div>
<% end %>
<% when :pgvector %>
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.ai.labels.storage_probe") %></dt>
<dd class="mt-1">
<%= render DS::Pill.new(
label: t("admin.system_health.show.ai.probe_statuses.#{ai_health.vector_store_probe.status}"),
tone: ai_health.vector_store_probe.passing? ? :success : (ai_health.vector_store_probe.failing? ? :error : :neutral),
marker: false
) %>
</dd>
</div>
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.ai.labels.embedding_probe") %></dt>
<dd class="mt-1">
<%= render DS::Pill.new(
label: t("admin.system_health.show.ai.probe_statuses.#{ai_health.embedding_probe.status}"),
tone: ai_health.embedding_probe.passing? ? :success : (ai_health.embedding_probe.failing? ? :error : :neutral),
marker: false
) %>
</dd>
</div>
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.ai.labels.pgvector_extension") %></dt>
<dd class="mt-1 font-medium text-primary">
<% if ai_health.pgvector_extension_enabled %>
<span class="text-success"><%= t("admin.system_health.show.ai.values.enabled") %></span>
<% elsif ai_health.pgvector_extension_available %>
<span class="text-warning"><%= t("admin.system_health.show.ai.values.available_not_enabled") %></span>
<% else %>
<span class="text-warning"><%= t("admin.system_health.show.ai.values.not_available") %></span>
<% end %>
</dd>
</div>
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.ai.labels.pgvector_table") %></dt>
<dd class="mt-1 font-medium <%= ai_health.pgvector_table_available ? "text-success" : "text-warning" %>">
<%= t("admin.system_health.show.ai.values.#{ai_health.pgvector_table_available ? :available : :not_found}") %>
</dd>
</div>
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.ai.labels.embedding_model") %></dt>
<dd class="mt-1 break-all font-mono text-primary"><%= ai_health.embedding_model.presence || t("admin.system_health.show.ai.values.not_set") %></dd>
</div>
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.ai.labels.embedding_endpoint") %></dt>
<dd class="mt-1 break-all font-mono text-primary"><%= ai_health.embedding_endpoint %></dd>
</div>
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.ai.labels.embedding_dimensions") %></dt>
<dd class="mt-1 font-mono text-primary"><%= ai_health.embedding_dimensions %></dd>
</div>
<% [ ai_health.vector_store_probe, ai_health.embedding_probe ].filter_map(&:failure_code).uniq.each do |failure_code| %>
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.ai.labels.failure_reason") %></dt>
<dd class="mt-1 font-medium text-destructive"><%= t("admin.system_health.show.ai.failure_codes.#{failure_code}") %></dd>
</div>
<% end %>
<% when :qdrant %>
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.ai.labels.endpoint") %></dt>
<dd class="mt-1 break-all font-mono text-primary"><%= ai_health.qdrant_endpoint %></dd>
</div>
<% end %>
</dl>
<% end %>
<%= render DS::Alert.new(title: t("admin.system_health.show.ai.local_setup.title"), variant: :info) do %>
<p><%= t("admin.system_health.show.ai.local_setup.description") %></p>
<p>
<code class="rounded bg-container-inset px-1.5 py-0.5">VECTOR_STORE_PROVIDER=pgvector</code>
<span aria-hidden="true">·</span>
<code class="rounded bg-container-inset px-1.5 py-0.5">EMBEDDING_URI_BASE</code>
<span aria-hidden="true">·</span>
<code class="rounded bg-container-inset px-1.5 py-0.5">EMBEDDING_MODEL</code>
<span aria-hidden="true">·</span>
<code class="rounded bg-container-inset px-1.5 py-0.5">EMBEDDING_DIMENSIONS</code>
</p>
<% end %>
</div>
@@ -0,0 +1,103 @@
<div class="space-y-4">
<% unless health.healthy? %>
<%= render DS::Alert.new(
title: t("admin.system_health.show.alert.title"),
message: t("shared.sidekiq_health_banner.reasons.#{health.reason}"),
variant: :warning,
live: :polite
) %>
<% end %>
<div class="bg-container rounded-xl shadow-border-xs p-4">
<h2 class="text-lg font-semibold mb-1"><%= t("admin.system_health.show.status_section_title") %></h2>
<p class="text-sm text-secondary mb-4"><%= t("admin.system_health.show.status_section_description") %></p>
<dl class="grid grid-cols-2 md:grid-cols-4 gap-4 text-sm">
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.labels.status") %></dt>
<dd class="text-primary font-medium">
<% if health.healthy? %>
<span class="text-success"><%= t("admin.system_health.show.values.healthy") %></span>
<% else %>
<span class="text-warning"><%= t("admin.system_health.show.values.unhealthy") %></span>
<% end %>
</dd>
</div>
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.labels.processes") %></dt>
<dd class="text-primary font-medium"><%= health.processes_count %></dd>
</div>
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.labels.last_heartbeat") %></dt>
<dd class="text-primary font-medium">
<% if health.last_heartbeat_at %>
<%= t("admin.system_health.show.values.time_ago", time_ago: time_ago_in_words(health.last_heartbeat_at)) %>
<% else %>
<%= t("admin.system_health.show.values.never") %>
<% end %>
</dd>
</div>
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.labels.max_queue_latency") %></dt>
<dd class="text-primary font-medium">
<%= t("admin.system_health.show.values.seconds", seconds: number_with_precision(health.max_queue_latency, precision: 1)) %>
</dd>
</div>
</dl>
</div>
<div class="bg-container rounded-xl shadow-border-xs p-4">
<h2 class="text-lg font-semibold mb-1"><%= t("admin.system_health.show.counters_section_title") %></h2>
<p class="text-sm text-secondary mb-4"><%= t("admin.system_health.show.counters_section_description") %></p>
<dl class="grid grid-cols-2 md:grid-cols-4 gap-4 text-sm">
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.labels.enqueued") %></dt>
<dd class="text-primary font-medium"><%= number_with_delimiter(health.enqueued_count) %></dd>
</div>
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.labels.retries") %></dt>
<dd class="text-primary font-medium"><%= number_with_delimiter(health.retry_count) %></dd>
</div>
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.labels.failed") %></dt>
<dd class="text-primary font-medium"><%= number_with_delimiter(health.failed_count) %></dd>
</div>
<div>
<dt class="text-secondary"><%= t("admin.system_health.show.labels.processed_total") %></dt>
<dd class="text-primary font-medium"><%= number_with_delimiter(health.processed_count) %></dd>
</div>
</dl>
</div>
<div class="bg-container rounded-xl shadow-border-xs p-4">
<h2 class="text-lg font-semibold mb-1"><%= t("admin.system_health.show.queues_section_title") %></h2>
<p class="text-sm text-secondary mb-4"><%= t("admin.system_health.show.queues_section_description") %></p>
<% breakdown = health.queue_breakdown %>
<% if breakdown.empty? %>
<p class="text-sm text-secondary"><%= t("admin.system_health.show.values.no_queues") %></p>
<% else %>
<table class="w-full text-sm">
<thead>
<tr class="text-left text-secondary border-b border-primary">
<th class="py-2 font-medium"><%= t("admin.system_health.show.labels.queue") %></th>
<th class="py-2 font-medium"><%= t("admin.system_health.show.labels.size") %></th>
<th class="py-2 font-medium"><%= t("admin.system_health.show.labels.latency") %></th>
</tr>
</thead>
<tbody class="divide-y divide-primary">
<% breakdown.each do |name, size, latency| %>
<tr>
<td class="py-2 text-primary"><%= name %></td>
<td class="py-2 text-primary"><%= number_with_delimiter(size) %></td>
<td class="py-2 text-primary">
<%= t("admin.system_health.show.values.seconds", seconds: number_with_precision(latency, precision: 1)) %>
</td>
</tr>
<% end %>
</tbody>
</table>
<% end %>
</div>
</div>
+16 -100
View File
@@ -1,105 +1,21 @@
<%= content_for :page_title, t(".title") %>
<div class="space-y-4">
<% unless @health.healthy? %>
<%= render DS::Alert.new(
title: t(".alert.title"),
message: t("shared.sidekiq_health_banner.reasons.#{@health.reason}"),
variant: :warning,
live: :polite
) %>
<%= render DS::Tabs.new(
active_tab: params[:tab].presence_in(%w[background_jobs ai]) || "background_jobs",
url_param_key: "tab",
navigate_on_change: true,
testid: "system-health-tabs"
) do |tabs| %>
<% tabs.with_nav do |nav| %>
<% nav.with_btn(id: "background_jobs", label: t(".tabs.background_jobs")) %>
<% nav.with_btn(id: "ai", label: t(".tabs.ai")) %>
<% end %>
<div class="bg-container rounded-xl shadow-border-xs p-4">
<h2 class="text-lg font-semibold mb-1"><%= t(".status_section_title") %></h2>
<p class="text-sm text-secondary mb-4"><%= t(".status_section_description") %></p>
<% tabs.with_panel(tab_id: "background_jobs") do %>
<%= render "background_jobs", health: @health %>
<% end %>
<dl class="grid grid-cols-2 md:grid-cols-4 gap-4 text-sm">
<div>
<dt class="text-secondary"><%= t(".labels.status") %></dt>
<dd class="text-primary font-medium">
<% if @health.healthy? %>
<span class="text-success"><%= t(".values.healthy") %></span>
<% else %>
<span class="text-warning"><%= t(".values.unhealthy") %></span>
<% end %>
</dd>
</div>
<div>
<dt class="text-secondary"><%= t(".labels.processes") %></dt>
<dd class="text-primary font-medium"><%= @health.processes_count %></dd>
</div>
<div>
<dt class="text-secondary"><%= t(".labels.last_heartbeat") %></dt>
<dd class="text-primary font-medium">
<% if @health.last_heartbeat_at %>
<%= t(".values.time_ago", time_ago: time_ago_in_words(@health.last_heartbeat_at)) %>
<% else %>
<%= t(".values.never") %>
<% end %>
</dd>
</div>
<div>
<dt class="text-secondary"><%= t(".labels.max_queue_latency") %></dt>
<dd class="text-primary font-medium">
<%= t(".values.seconds", seconds: number_with_precision(@health.max_queue_latency, precision: 1)) %>
</dd>
</div>
</dl>
</div>
<div class="bg-container rounded-xl shadow-border-xs p-4">
<h2 class="text-lg font-semibold mb-1"><%= t(".counters_section_title") %></h2>
<p class="text-sm text-secondary mb-4"><%= t(".counters_section_description") %></p>
<dl class="grid grid-cols-2 md:grid-cols-4 gap-4 text-sm">
<div>
<dt class="text-secondary"><%= t(".labels.enqueued") %></dt>
<dd class="text-primary font-medium"><%= number_with_delimiter(@health.enqueued_count) %></dd>
</div>
<div>
<dt class="text-secondary"><%= t(".labels.retries") %></dt>
<dd class="text-primary font-medium"><%= number_with_delimiter(@health.retry_count) %></dd>
</div>
<div>
<dt class="text-secondary"><%= t(".labels.failed") %></dt>
<dd class="text-primary font-medium"><%= number_with_delimiter(@health.failed_count) %></dd>
</div>
<div>
<dt class="text-secondary"><%= t(".labels.processed_total") %></dt>
<dd class="text-primary font-medium"><%= number_with_delimiter(@health.processed_count) %></dd>
</div>
</dl>
</div>
<div class="bg-container rounded-xl shadow-border-xs p-4">
<h2 class="text-lg font-semibold mb-1"><%= t(".queues_section_title") %></h2>
<p class="text-sm text-secondary mb-4"><%= t(".queues_section_description") %></p>
<% breakdown = @health.queue_breakdown %>
<% if breakdown.empty? %>
<p class="text-sm text-secondary"><%= t(".values.no_queues") %></p>
<% else %>
<table class="w-full text-sm">
<thead>
<tr class="text-left text-secondary border-b border-primary">
<th class="py-2 font-medium"><%= t(".labels.queue") %></th>
<th class="py-2 font-medium"><%= t(".labels.size") %></th>
<th class="py-2 font-medium"><%= t(".labels.latency") %></th>
</tr>
</thead>
<tbody class="divide-y divide-primary">
<% breakdown.each do |name, size, latency| %>
<tr>
<td class="py-2 text-primary"><%= name %></td>
<td class="py-2 text-primary"><%= number_with_delimiter(size) %></td>
<td class="py-2 text-primary">
<%= t(".values.seconds", seconds: number_with_precision(latency, precision: 1)) %>
</td>
</tr>
<% end %>
</tbody>
</table>
<% end %>
</div>
</div>
<% tabs.with_panel(tab_id: "ai") do %>
<%= render "ai_status", ai_health: @ai_health %>
<% end %>
<% end %>
+17 -4
View File
@@ -114,7 +114,8 @@ x-rails-env: &rails_env
AI_RESPONSE_TIMEOUT: ${AI_RESPONSE_TIMEOUT:-1200}
# Vector store — pgvector keeps all data local (requires pgvector/pgvector Docker image for db)
VECTOR_STORE_PROVIDER: pgvector
EMBEDDING_MODEL: nomic-embed-text
EMBEDDING_URI_BASE: http://ollama:11434/v1
EMBEDDING_MODEL: mxbai-embed-large
EMBEDDING_DIMENSIONS: "1024"
# NOTE: enabling OpenAI will incur costs when you use AI-related features in the app (chat, rules). Make sure you have set appropriate spend limits on your account before adding this.
# OPENAI_ACCESS_TOKEN: ${OPENAI_ACCESS_TOKEN}
@@ -182,7 +183,14 @@ services:
networks:
- sure_net
# Note: You still have to download models manually using the ollama CLI or via Open WebUI
# Ollama's OLLAMA_MODELS setting controls the model storage directory; it is
# not a preload list. Start Ollama, then pull each model once. The ollama
# volume below keeps the downloaded models across container restarts:
#
# docker compose -f compose.example.ai.yml --profile local-ai up -d --wait ollama
# docker compose -f compose.example.ai.yml exec ollama ollama pull deepseek-r1:8b
# docker compose -f compose.example.ai.yml exec ollama ollama pull llama3.1:8b
# docker compose -f compose.example.ai.yml exec ollama ollama pull mxbai-embed-large
ollama:
profiles:
- local-ai
@@ -196,7 +204,12 @@ services:
- "11434:11434"
environment:
- OLLAMA_KEEP_ALIVE=1h
- OLLAMA_MODELS=deepseek-r1:8b,llama3.1:8b,nomic-embed-text # Pre-load model on startup, you can change this to your preferred model
healthcheck:
test: ["CMD", "ollama", "list"]
interval: 5s
timeout: 5s
retries: 12
start_period: 5s
networks:
- sure_net
# Recommended: Enable GPU support
@@ -338,7 +351,7 @@ services:
- sure_net
db:
image: pgvector/pgvector:pg16
image: pgvector/pgvector:pg16-trixie
restart: unless-stopped
volumes:
- postgres-data:/var/lib/postgresql/data
@@ -4,6 +4,9 @@ en:
system_health:
show:
title: System health
tabs:
background_jobs: Background jobs
ai: AI status
alert:
title: Background jobs aren't running
status_section_title: Sidekiq status
@@ -31,3 +34,106 @@ en:
never: Never
seconds: "%{seconds}s"
no_queues: "No queues are registered. The worker may not be running."
ai:
run_checks: Run checks again
last_checked: "Live checks completed %{time_ago} ago. Results are cached briefly."
not_checked_help: Open the AI status URL directly or run the checks to verify service liveness.
alerts:
custom_openai:
title: The hosted vector-store check failed at this custom endpoint
message: The configured endpoint did not pass the /v1/vector_stores liveness check. Most local and OpenAI-compatible endpoints do not implement that API. Chat and PDF processing may still work; use pgvector with a separate embeddings endpoint for the usual local setup. The failure was recorded in Settings → Debug logs and Rails.logger.
llm_probe_failed:
title: The LLM live check failed
message: Sure could not verify both the configured endpoint and model. The failure was recorded in Settings → Debug logs and Rails.logger.
vector_probe_failed:
title: The document-search live check failed
message: Sure could not verify the selected vector store and every required embedding service. The failure was recorded in Settings → Debug logs and Rails.logger.
qdrant:
title: Qdrant support is not implemented yet
message: The Qdrant adapter is scaffolded, but document upload and search operations currently fail. Select OpenAI or pgvector for document search.
missing_vector_store:
title: No vector store is configured
message: Uploaded documents cannot be indexed or searched until a vector-store adapter and its requirements are configured.
unavailable_vector_store:
title: The selected vector store is unavailable
message: Check the adapter requirements below. For pgvector, the database must provide the vector extension or an existing vector_store_chunks table.
llm:
title: LLM and PDF processing
description: The live check queries the provider's models API and verifies that the configured model is available. Credential values are never displayed.
vector_store:
title: Vector store and document search
description: The live check queries the hosted vector-store API, or tests the pgvector table and creates one short embedding without storing it.
labels:
selected_provider: Selected provider
effective_provider: Effective provider
model: Model
endpoint: Endpoint
pdf_processing: PDF processing
request_timeout: Request timeout
openai_credentials: OpenAI credentials
openai_compatible_credentials: OpenAI-compatible API credentials
anthropic_credentials: Anthropic credentials
adapter: Adapter
adapter_available: Adapter available
pgvector_extension: PostgreSQL vector extension
pgvector_table: Vector chunks table
embedding_model: Embedding model
embedding_endpoint: Embedding endpoint
embedding_dimensions: Embedding dimensions
storage_probe: pgvector storage check
embedding_probe: Embedding endpoint check
failure_reason: Failure reason
providers:
openai: OpenAI
openai_compatible: OpenAI-compatible
anthropic: Anthropic
ollama: Ollama
openrouter: OpenRouter
together: Together
kilo: Kilo
cloudflare: Cloudflare
custom_openai_compatible: Custom endpoint
adapters:
openai: OpenAI hosted vector store
pgvector: pgvector
qdrant: Qdrant
values:
configured: Configured
not_configured: Not configured
fallback: "(fallback)"
not_set: Not set
"yes": "Yes"
"no": "No"
enabled: Enabled
available_not_enabled: Available, not enabled
not_available: Not available
available: Available
not_found: Not found
pdf_statuses:
supported: Supported
unsupported: Not supported by the effective provider/model
unavailable: Unavailable until an LLM provider is configured
probe_statuses:
passing: Live check passed
failing: Live check failed
not_checked: Not checked
not_configured: Not configured
vector_statuses:
passing: Live checks passed
failing: Live check failed
not_checked: Not checked
not_configured: Not configured
scaffolded: Scaffolded
missing: Not configured
failure_codes:
model_not_available: The configured model was not returned by the provider
invalid_response: The service returned an unexpected response
dimensions_mismatch: The embedding vector dimensions do not match the configured dimensions
extension_not_enabled: The PostgreSQL vector extension is not enabled
table_not_found: The vector_store_chunks table was not found
timeout: The service did not respond before the probe timeout
request_failed: The service request failed
unsupported_provider: The provider does not support this check
local_setup:
title: Recommended local setup
description: Run the LLM through your local OpenAI-compatible endpoint, store document vectors in pgvector, and point the embedding settings at an OpenAI-compatible embeddings endpoint. The PostgreSQL image must include the vector extension, and embedding dimensions must match the selected model.
+68 -8
View File
@@ -1309,8 +1309,8 @@ OPENAI_ACCESS_TOKEN=sk-proj-...
Use PostgreSQL's pgvector extension for fully local document search. All data stays on your infrastructure.
**Requirements:**
- Use the `pgvector/pgvector:pg16` Docker image instead of `postgres:16` (drop-in replacement)
- An embedding model served via an OpenAI-compatible `/v1/embeddings` endpoint (e.g. Ollama with `nomic-embed-text`)
- Use the `pgvector/pgvector:pg16-trixie` Docker image instead of `postgres:16` (drop-in replacement)
- An embedding model served via an OpenAI-compatible `/v1/embeddings` endpoint (e.g. Ollama with `mxbai-embed-large`)
- Run the migration with `VECTOR_STORE_PROVIDER=pgvector` to create the `vector_store_chunks` table
```bash
@@ -1318,22 +1318,55 @@ Use PostgreSQL's pgvector extension for fully local document search. All data st
VECTOR_STORE_PROVIDER=pgvector
# Embedding model configuration
EMBEDDING_MODEL=nomic-embed-text # Default: nomic-embed-text
EMBEDDING_MODEL=mxbai-embed-large # Default: mxbai-embed-large
EMBEDDING_DIMENSIONS=1024 # Default: 1024 (must match your model)
EMBEDDING_URI_BASE=http://ollama:11434/v1 # Falls back to OPENAI_URI_BASE if not set
EMBEDDING_ACCESS_TOKEN= # Falls back to OPENAI_ACCESS_TOKEN if not set
```
Sure enables the `vector` extension when it first provisions the chunks table,
provided the database user has permission. If the AI status page reports that
the extension is available but not enabled and automatic provisioning cannot
enable it, connect as the PostgreSQL superuser and run:
```sql
CREATE EXTENSION vector;
```
The LLM and embedding endpoints are independent. A common fully local setup is
an OpenAI-compatible chat model through `OPENAI_URI_BASE`, pgvector for storage,
and an embedding model through `EMBEDDING_URI_BASE`. Make sure
`EMBEDDING_DIMENSIONS` matches the selected embedding model (for example,
`mxbai-embed-large` uses 1024 dimensions).
If you are using Ollama (as in `compose.example.ai.yml`), pull the embedding model:
```bash
docker compose exec ollama ollama pull nomic-embed-text
docker compose -f compose.example.ai.yml --profile local-ai up -d --wait ollama
docker compose exec ollama ollama pull mxbai-embed-large
```
> [!WARNING]
> Do not change `EMBEDDING_MODEL` for an existing pgvector index without
> rebuilding it. Vectors created by different models are not comparable, even
> when they have the same dimensions. Back up the database and the source
> documents, then remove the existing documents from Sure. If the new model has
> different dimensions, drop the now-empty chunks table so Sure can recreate it
> with the new vector size:
>
> ```bash
> docker compose -f compose.example.ai.yml exec web bin/rails runner \
> 'ActiveRecord::Base.connection_pool.with_connection { |connection| connection.drop_table(VectorStore::Pgvector::TABLE_NAME, if_exists: true) }'
> ```
>
> Change the embedding settings and restart Sure. Confirm that **System health
> → AI status** reports the new model and dimensions, then upload the source
> documents again. This recreates every embedding with only the new model.
##### Qdrant (Self-Hosted)
> [!CAUTION]
> Only `OpenAI` has been implemented!
> Qdrant is not implemented yet. Use OpenAI or pgvector for document search.
Use a dedicated Qdrant vector database:
@@ -1369,7 +1402,33 @@ volumes:
#### Verifying the Configuration
You can check whether a vector store is properly configured from the Rails console:
Super admins can open **System health → AI status** at
`/admin/system_health?tab=ai`. Opening that URL runs bounded, non-destructive
live checks against the effective configuration:
- OpenAI-compatible and Anthropic providers must return the configured model
from their models API.
- The hosted OpenAI vector-store adapter must answer a list request without
creating or changing a store.
- The pgvector adapter must have its extension enabled, its chunks table
present, and successfully execute a query.
- A pgvector embedding endpoint must create one short test embedding, and the
returned vector must match `EMBEDDING_DIMENSIONS`. The test vector is not
stored.
When `OPENAI_URI_BASE` points outside OpenAI's hosted API, the page labels the
selected provider **OpenAI-compatible** and identifies a known effective
provider from the endpoint (for example Ollama, OpenRouter, Together, Kilo, or
Cloudflare Workers AI/AI Gateway). Unrecognized services are shown as **Custom
endpoint**.
Results are cached for 60 seconds by default. **Run checks again** bypasses the
cache. Set `AI_HEALTH_PROBE_TIMEOUT` to change the default five-second request
timeout and `AI_HEALTH_PROBE_CACHE_TTL` to change the cache duration. Failed
checks are written as system-wide entries in **Settings → Debug logs** and to
`Rails.logger`; endpoints are redacted and credentials are never included.
You can also check the adapter from the Rails console:
```ruby
VectorStore.configured? # => true / false
@@ -1386,7 +1445,8 @@ The following file extensions are supported for document upload and search:
#### Privacy Notes
- **OpenAI backend:** Document content is sent to OpenAI's API for indexing and search. The same privacy considerations as the AI chat apply.
- **Pgvector / Qdrant backends:** All data stays on your infrastructure. No external API calls are made for document search.
- **Pgvector backend:** Stored chunks stay in PostgreSQL. Text is still sent to the configured embedding endpoint, which may be local or remote.
- **Qdrant backend:** The adapter is currently scaffolded and cannot upload or search documents.
### Multi-Model Setup
@@ -1432,4 +1492,4 @@ For issues with AI features:
---
**Last Updated:** March 2026
**Last Updated:** August 2026
+21
View File
@@ -0,0 +1,21 @@
require "test_helper"
class DS::TabsTest < ViewComponent::TestCase
test "can navigate to the selected tab for server-backed content" do
render_inline(DS::Tabs.new(
active_tab: "background_jobs",
url_param_key: "tab",
navigate_on_change: true
)) do |tabs|
tabs.with_nav do |nav|
nav.with_btn(id: "background_jobs", label: "Background jobs")
nav.with_btn(id: "ai", label: "AI status")
end
tabs.with_panel(tab_id: "background_jobs") { "Jobs" }
tabs.with_panel(tab_id: "ai") { "AI" }
end
assert_selector "[data-ds--tabs-url-param-key-value='tab']"
assert_selector "[data-ds--tabs-navigate-on-change-value='true']"
end
end
@@ -1,6 +1,29 @@
require "test_helper"
class Admin::SystemHealthControllerTest < ActionDispatch::IntegrationTest
AI_ENVIRONMENT = %w[
OPENAI_ACCESS_TOKEN OPENAI_URI_BASE OPENAI_MODEL OPENAI_REQUEST_TIMEOUT
OPENAI_SUPPORTS_PDF_PROCESSING ANTHROPIC_ACCESS_TOKEN ANTHROPIC_API_KEY
ANTHROPIC_BASE_URL ANTHROPIC_MODEL ANTHROPIC_REQUEST_TIMEOUT
VECTOR_STORE_PROVIDER EMBEDDING_URI_BASE EMBEDDING_MODEL
EMBEDDING_DIMENSIONS EMBEDDING_ACCESS_TOKEN QDRANT_URL QDRANT_API_KEY
AI_HEALTH_PROBE_TIMEOUT AI_HEALTH_PROBE_CACHE_TTL
].index_with(nil).freeze
setup do
Setting.stubs(:llm_provider).returns("openai")
Setting.stubs(:openai_access_token).returns(nil)
Setting.stubs(:openai_uri_base).returns(nil)
Setting.stubs(:openai_model).returns(nil)
Setting.stubs(:anthropic_access_token).returns(nil)
Setting.stubs(:anthropic_base_url).returns(nil)
Setting.stubs(:anthropic_model).returns(nil)
AiHealth::Probe.any_instance.stubs(:llm).returns(probe_result(:passing))
AiHealth::Probe.any_instance.stubs(:openai_vector_store).returns(probe_result(:passing))
AiHealth::Probe.any_instance.stubs(:pgvector).returns(probe_result(:passing))
AiHealth::Probe.any_instance.stubs(:embedding).returns(probe_result(:passing))
end
test "super admin can view the system health page" do
sign_in users(:sure_support_staff)
SidekiqHealth.any_instance.stubs(:healthy?).returns(true)
@@ -18,6 +41,8 @@ class Admin::SystemHealthControllerTest < ActionDispatch::IntegrationTest
assert_response :success
assert_match(/Sidekiq status/, response.body)
assert_match(/Healthy/, response.body)
assert_select "button[role='tab']", text: "AI status"
assert_select "[data-ds--tabs-navigate-on-change-value='true']"
end
test "renders degraded state with reason when Sidekiq is unhealthy" do
@@ -53,4 +78,188 @@ class Admin::SystemHealthControllerTest < ActionDispatch::IntegrationTest
assert_redirected_to new_session_path
end
test "AI status reports the default OpenAI LLM and hosted vector store" do
sign_in users(:sure_support_staff)
stub_healthy_sidekiq
with_ai_environment("OPENAI_ACCESS_TOKEN" => "sk-secret-openai") do
get admin_system_health_url(tab: "ai")
end
assert_response :success
assert_select "button[role='tab'][aria-selected='true']", text: "AI status"
assert_match(/LLM and PDF processing/, response.body)
assert_select "[data-testid='selected-llm-provider']", text: "OpenAI"
assert_select "[data-testid='effective-llm-provider']", text: "OpenAI"
assert_match(/gpt-4\.1/, response.body)
assert_match(%r{https://api\.openai\.com/v1}, response.body)
assert_match(/OpenAI hosted vector store/, response.body)
assert_match(/Live check passed/, response.body)
assert_match(/Live checks passed/, response.body)
assert_match(/Supported/, response.body)
assert_no_match(/sk-secret-openai/, response.body)
end
test "background jobs tab does not run AI probes" do
sign_in users(:sure_support_staff)
stub_healthy_sidekiq
AiHealth::Probe.any_instance.expects(:llm).never
AiHealth::Probe.any_instance.expects(:openai_vector_store).never
with_ai_environment("OPENAI_ACCESS_TOKEN" => "sk-secret-openai") do
get admin_system_health_url
end
assert_response :success
assert_match(/Not checked/, response.body)
assert_no_match(/sk-secret-openai/, response.body)
end
test "AI status warns when a custom OpenAI endpoint is paired with the hosted vector store" do
sign_in users(:sure_support_staff)
stub_healthy_sidekiq
with_ai_environment(
"OPENAI_ACCESS_TOKEN" => "local-token",
"OPENAI_URI_BASE" => credentialed_url(
scheme: "http",
host: "ollama",
port: 11_434,
path: "/v1",
user: "operator",
password: "uri-secret",
query: "api_key=query-secret"
),
"OPENAI_MODEL" => "qwen3:8b"
) do
AiHealth::Probe.any_instance.stubs(:openai_vector_store).returns(
probe_result(:failing, failure_code: :request_failed, http_status: 404)
)
get admin_system_health_url(tab: "ai")
end
assert_response :success
assert_select "[data-testid='selected-llm-provider']", text: "OpenAI-compatible"
assert_select "[data-testid='effective-llm-provider']", text: "Ollama"
assert_match(/OpenAI-compatible API credentials/, response.body)
assert_match(%r{http://ollama:11434/v1}, response.body)
assert_match(%r{did not pass the /v1/vector_stores liveness check}, response.body)
assert_match(/use pgvector with a separate embeddings endpoint/, response.body)
assert_match(/Live check failed/, response.body)
assert_no_match(/local-token|uri-secret|query-secret/, response.body)
end
test "AI status reports Anthropic with an available pgvector store" do
sign_in users(:sure_support_staff)
stub_healthy_sidekiq
Setting.stubs(:llm_provider).returns("anthropic")
connection = stub("connection")
connection.stubs(:table_exists?).with(VectorStore::Pgvector::TABLE_NAME).returns(true)
connection.stubs(:extension_enabled?).with("vector").returns(true)
ActiveRecord::Base.stubs(:connection).returns(connection)
VectorStore.expects(:embedding_access_token).returns("runtime-embedding-token")
AiHealth::Probe.any_instance.expects(:embedding).with(
endpoint: "http://ollama:11434/v1",
access_token: "runtime-embedding-token",
model: "mxbai-embed-large",
dimensions: 1024
).returns(probe_result(:passing))
with_ai_environment(
"ANTHROPIC_ACCESS_TOKEN" => "anthropic-secret",
"ANTHROPIC_MODEL" => "claude-sonnet-4-6",
"EMBEDDING_URI_BASE" => "http://ollama:11434/v1",
"EMBEDDING_MODEL" => "mxbai-embed-large",
"EMBEDDING_DIMENSIONS" => "1024"
) do
get admin_system_health_url(tab: "ai")
end
assert_response :success
assert_match(/Anthropic/, response.body)
assert_match(/pgvector/, response.body)
assert_match(/PostgreSQL vector extension/, response.body)
assert_match(/mxbai-embed-large/, response.body)
assert_match(%r{http://ollama:11434/v1}, response.body)
assert_match(/Live checks passed/, response.body)
assert_no_match(/anthropic-secret/, response.body)
end
test "AI status explains when no vector store is configured" do
sign_in users(:sure_support_staff)
stub_healthy_sidekiq
with_ai_environment do
get admin_system_health_url(tab: "ai")
end
assert_response :success
assert_match(/No vector store is configured/, response.body)
assert_match(/Uploaded documents cannot be indexed or searched/, response.body)
end
test "AI status marks Qdrant as scaffolded and redacts its URL" do
sign_in users(:sure_support_staff)
stub_healthy_sidekiq
with_ai_environment(
"VECTOR_STORE_PROVIDER" => "qdrant",
"QDRANT_URL" => credentialed_url(
scheme: "https",
host: "qdrant.example.test",
port: 6333,
user: "admin",
password: "qdrant-secret",
query: "api_key=query-secret"
),
"QDRANT_API_KEY" => "header-secret"
) do
get admin_system_health_url(tab: "ai")
end
assert_response :success
assert_match(/Qdrant support is not implemented yet/, response.body)
assert_match(/Scaffolded/, response.body)
assert_match(%r{https://qdrant\.example\.test:6333}, response.body)
assert_no_match(/qdrant-secret|query-secret|header-secret/, response.body)
end
private
def credentialed_url(scheme:, host:, port:, user:, password:, path: nil, query: nil)
URI::Generic.build(
scheme: scheme,
userinfo: "#{user}:#{password}",
host: host,
port: port,
path: path,
query: query
).to_s
end
def probe_result(status, failure_code: nil, http_status: nil)
AiHealth::Probe::Result.new(
status: status,
checked_at: status.in?([ :passing, :failing ]) ? Time.current : nil,
failure_code: failure_code,
http_status: http_status
)
end
def with_ai_environment(overrides = {}, &block)
ClimateControl.modify(AI_ENVIRONMENT.merge(overrides), &block)
end
def stub_healthy_sidekiq
SidekiqHealth.any_instance.stubs(:healthy?).returns(true)
SidekiqHealth.any_instance.stubs(:processes_count).returns(1)
SidekiqHealth.any_instance.stubs(:last_heartbeat_at).returns(Time.current)
SidekiqHealth.any_instance.stubs(:max_queue_latency).returns(0.0)
SidekiqHealth.any_instance.stubs(:enqueued_count).returns(0)
SidekiqHealth.any_instance.stubs(:retry_count).returns(0)
SidekiqHealth.any_instance.stubs(:failed_count).returns(0)
SidekiqHealth.any_instance.stubs(:processed_count).returns(42)
SidekiqHealth.any_instance.stubs(:queue_breakdown).returns([ [ "default", 0, 0.0 ] ])
end
end
+198
View File
@@ -0,0 +1,198 @@
require "test_helper"
class AiHealth::ProbeTest < ActiveSupport::TestCase
setup do
@cache = ActiveSupport::Cache::MemoryStore.new
@probe = AiHealth::Probe.new(cache: @cache)
end
test "OpenAI LLM probe calls the models endpoint and verifies the configured model" do
request = stub_request(:get, "http://ollama.example.test:11434/v1/models")
.with(headers: { "Authorization" => "Bearer local-token" })
.to_return(
status: 200,
headers: { "Content-Type" => "application/json" },
body: { data: [ { id: "qwen3:8b" } ] }.to_json
)
result = @probe.llm(
provider: :openai,
endpoint: "http://ollama.example.test:11434/v1",
access_token: "local-token",
model: "qwen3:8b"
)
assert result.passing?
assert result.checked_at
assert_requested request
end
test "Anthropic LLM probe calls the models endpoint and verifies the configured model" do
model_info = Struct.new(:id).new("claude-sonnet-4-6")
models = mock("models")
models.expects(:retrieve).with("claude-sonnet-4-6").returns(model_info)
client = mock("anthropic_client")
client.expects(:models).returns(models)
@probe.stubs(:anthropic_client).returns(client)
result = @probe.llm(
provider: :anthropic,
endpoint: "https://api.anthropic.com",
access_token: "anthropic-token",
model: "claude-sonnet-4-6"
)
assert result.passing?
end
test "failed LLM probe writes a system-wide debug entry and Rails log without secrets" do
models = stub(list: { "data" => [ { "id" => "another-model" } ] })
@probe.stubs(:openai_client).returns(stub(models: models))
endpoint = URI::HTTP.build(
userinfo: "operator:uri-secret",
host: "ollama",
port: 11_434,
path: "/v1",
query: "api_key=query-secret"
).to_s
Rails.logger.expects(:error).with do |message|
message.include?("AI health llm liveness probe failed") &&
!message.include?("secret-token") &&
!message.include?("uri-secret") &&
!message.include?("query-secret")
end
assert_difference -> { DebugLogEntry.where(category: "ai_health").count }, 1 do
@result = @probe.llm(
provider: :openai,
endpoint: endpoint,
access_token: "secret-token",
model: "missing-model"
)
end
assert @result.failing?
assert_equal :model_not_available, @result.failure_code
entry = DebugLogEntry.where(category: "ai_health")
.where("metadata ->> 'model' = ?", "missing-model")
.order(:id)
.last
assert_equal "error", entry.level
assert_equal "AiHealth::Probe", entry.source
assert_equal "openai", entry.provider_key
assert_nil entry.family
assert_nil entry.account
assert_equal "http://ollama:11434/v1", entry.metadata.fetch("endpoint")
assert_equal "model_not_available", entry.metadata.fetch("failure_code")
assert_no_match(/secret-token|uri-secret|query-secret/, entry.metadata.to_json)
end
test "probe results are cached to avoid repeated requests and failure logs" do
models = mock("models")
models.expects(:list).once.returns({ "data" => [ { "id" => "gpt-4.1" } ] })
client = stub(models: models)
@probe.stubs(:openai_client).returns(client)
2.times do
result = @probe.llm(
provider: :openai,
endpoint: "https://api.openai.com/v1",
access_token: "token",
model: "gpt-4.1"
)
assert result.passing?
end
end
test "forced probe bypasses the cached result" do
models = mock("models")
models.expects(:list).twice.returns({ "data" => [ { "id" => "gpt-4.1" } ] })
client = stub(models: models)
@probe.stubs(:openai_client).returns(client)
arguments = {
provider: :openai,
endpoint: "https://api.openai.com/v1",
access_token: "token",
model: "gpt-4.1"
}
@probe.llm(**arguments)
forced_probe = AiHealth::Probe.new(cache: @cache, force: true)
forced_probe.stubs(:openai_client).returns(client)
assert forced_probe.llm(**arguments).passing?
end
test "hosted vector-store probe calls the non-destructive list endpoint" do
request = stub_request(:get, "https://api.openai.example.test/v1/vector_stores")
.with(query: { limit: 1 }, headers: { "Authorization" => "Bearer token" })
.to_return(
status: 200,
headers: { "Content-Type" => "application/json" },
body: { data: [] }.to_json
)
result = @probe.openai_vector_store(
endpoint: "https://api.openai.example.test/v1",
access_token: "token"
)
assert result.passing?
assert_requested request
end
test "pgvector probe verifies the extension, table, and a real query" do
connection = mock("connection")
connection.expects(:extension_enabled?).with("vector").returns(true)
connection.expects(:table_exists?).with("vector_store_chunks").returns(true)
connection.expects(:quote_table_name).with("vector_store_chunks").returns(%("vector_store_chunks"))
connection.expects(:select_value).with('SELECT 1 FROM "vector_store_chunks" LIMIT 1').returns(nil)
assert @probe.pgvector(connection: connection).passing?
end
test "embedding probe sends a small request and verifies dimensions" do
request = stub_request(:post, "http://ollama.example.test:11434/v1/embeddings")
.with(
body: {
model: "nomic-embed-text",
input: AiHealth::Probe::EMBEDDING_TEST_INPUT
}
)
.to_return(
status: 200,
headers: { "Content-Type" => "application/json" },
body: { data: [ { embedding: [ 0.1, 0.2, 0.3 ] } ] }.to_json
)
result = @probe.embedding(
endpoint: "http://ollama.example.test:11434/v1",
access_token: nil,
model: "nomic-embed-text",
dimensions: 3
)
assert result.passing?
assert_requested request
end
test "embedding probe fails when returned dimensions do not match configuration" do
response = Struct.new(:body).new({ "data" => [ { "embedding" => [ 0.1, 0.2 ] } ] })
client = stub
client.stubs(:post).yields(Struct.new(:body).new).returns(response)
@probe.stubs(:embedding_client).returns(client)
Rails.logger.stubs(:error)
DebugLogEntry.stubs(:capture)
result = @probe.embedding(
endpoint: "http://ollama:11434/v1",
access_token: nil,
model: "nomic-embed-text",
dimensions: 3
)
assert result.failing?
assert_equal :dimensions_mismatch, result.failure_code
end
end
+58
View File
@@ -0,0 +1,58 @@
require "test_helper"
class AiHealthTest < ActiveSupport::TestCase
AI_ENVIRONMENT = %w[
OPENAI_ACCESS_TOKEN OPENAI_URI_BASE OPENAI_MODEL
ANTHROPIC_ACCESS_TOKEN ANTHROPIC_API_KEY VECTOR_STORE_PROVIDER
].index_with(nil).freeze
setup do
Setting.stubs(:llm_provider).returns("openai")
Setting.stubs(:openai_access_token).returns(nil)
Setting.stubs(:openai_uri_base).returns(nil)
Setting.stubs(:openai_model).returns(nil)
Setting.stubs(:anthropic_access_token).returns(nil)
end
test "native OpenAI remains distinct from OpenAI-compatible providers" do
with_openai_endpoint(nil) do |health|
assert_equal :openai, health.selected_llm_provider
assert_equal :openai, health.effective_llm_provider
assert_not health.openai_compatible_endpoint?
end
end
test "identifies known OpenAI-compatible providers from their endpoints" do
{
"http://ollama:11434/v1" => :ollama,
"http://127.0.0.1:11434/v1" => :ollama,
"https://openrouter.ai/api/v1" => :openrouter,
"https://api.together.ai/v1" => :together,
"https://api.kilo.ai/api/gateway" => :kilo,
"https://api.cloudflare.com/client/v4/accounts/account-id/ai/v1" => :cloudflare,
"https://gateway.ai.cloudflare.com/v1/account-id/gateway-id/compat" => :cloudflare,
"https://models.example.test/v1" => :custom_openai_compatible
}.each do |endpoint, provider|
with_openai_endpoint(endpoint) do |health|
assert_equal :openai_compatible, health.selected_llm_provider, endpoint
assert_equal provider, health.effective_llm_provider, endpoint
assert health.openai_compatible_endpoint?, endpoint
assert_not health.llm_fallback?, endpoint
end
end
end
private
def with_openai_endpoint(endpoint)
ClimateControl.modify(
AI_ENVIRONMENT.merge(
"OPENAI_ACCESS_TOKEN" => "test-token",
"OPENAI_URI_BASE" => endpoint,
"OPENAI_MODEL" => endpoint.present? ? "test-model" : nil,
"VECTOR_STORE_PROVIDER" => "qdrant"
)
) do
yield AiHealth.new(run_probes: false)
end
end
end
+15 -1
View File
@@ -4,7 +4,9 @@ class VectorStore::EmbeddableTest < ActiveSupport::TestCase
class EmbeddableHost
include VectorStore::Embeddable
# Expose private methods for testing
public :extract_text, :chunk_text, :embed, :embed_batch
public :extract_text, :chunk_text, :embed, :embed_batch,
:embedding_model, :embedding_dimensions, :embedding_uri_base,
:embedding_access_token
end
setup do
@@ -138,6 +140,18 @@ class VectorStore::EmbeddableTest < ActiveSupport::TestCase
assert_raises(VectorStore::Error) { @host.embed("test text") }
end
test "embedding configuration delegates to the shared runtime configuration" do
VectorStore.expects(:embedding_model).returns("model")
VectorStore.expects(:embedding_dimensions).returns(3)
VectorStore.expects(:embedding_uri_base).returns("https://embeddings.example.test/v1")
VectorStore.expects(:embedding_access_token).returns("token")
assert_equal "model", @host.embedding_model
assert_equal 3, @host.embedding_dimensions
assert_equal "https://embeddings.example.test/v1", @host.embedding_uri_base
assert_equal "token", @host.embedding_access_token
end
# --- embed_batch ---
test "embed_batch processes texts and returns ordered vectors" do
+34
View File
@@ -0,0 +1,34 @@
require "test_helper"
class VectorStoreTest < ActiveSupport::TestCase
EMBEDDING_ENVIRONMENT = %w[
EMBEDDING_MODEL EMBEDDING_DIMENSIONS EMBEDDING_URI_BASE
EMBEDDING_ACCESS_TOKEN OPENAI_URI_BASE OPENAI_ACCESS_TOKEN
].index_with(nil).freeze
test "embedding defaults use a matching model and vector width" do
ClimateControl.modify(EMBEDDING_ENVIRONMENT) do
assert_equal "mxbai-embed-large", VectorStore.embedding_model
assert_equal 1024, VectorStore.embedding_dimensions
end
end
test "embedding credentials match runtime environment precedence" do
ClimateControl.modify(EMBEDDING_ENVIRONMENT.merge(
"EMBEDDING_ACCESS_TOKEN" => "embedding-token",
"OPENAI_ACCESS_TOKEN" => "openai-token"
)) do
assert_equal "embedding-token", VectorStore.embedding_access_token
end
ClimateControl.modify(EMBEDDING_ENVIRONMENT.merge(
"OPENAI_ACCESS_TOKEN" => "openai-token"
)) do
assert_equal "openai-token", VectorStore.embedding_access_token
end
ClimateControl.modify(EMBEDDING_ENVIRONMENT) do
assert_nil VectorStore.embedding_access_token
end
end
end
+54
View File
@@ -0,0 +1,54 @@
require "application_system_test_case"
class Admin::SystemHealthTest < ApplicationSystemTestCase
setup do
sign_in users(:sure_support_staff)
Setting.stubs(:llm_provider).returns("openai")
Setting.stubs(:openai_access_token).returns(nil)
Setting.stubs(:openai_uri_base).returns(nil)
Setting.stubs(:openai_model).returns(nil)
stub_healthy_sidekiq
AiHealth::Probe.any_instance.stubs(:llm).returns(probe_result(:passing))
AiHealth::Probe.any_instance.stubs(:openai_vector_store).returns(probe_result(:passing))
end
test "selecting AI status runs live probes" do
ClimateControl.modify(
"OPENAI_ACCESS_TOKEN" => "test-token",
"OPENAI_URI_BASE" => nil,
"OPENAI_MODEL" => nil,
"VECTOR_STORE_PROVIDER" => nil
) do
visit admin_system_health_path
click_button "AI status"
assert_current_path admin_system_health_path(tab: "ai")
assert_selector "button[role='tab'][aria-selected='true']", text: "AI status"
assert_text "Live check passed"
assert_text "Live checks passed"
end
end
private
def probe_result(status)
AiHealth::Probe::Result.new(
status: status,
checked_at: Time.current,
failure_code: nil,
http_status: nil
)
end
def stub_healthy_sidekiq
SidekiqHealth.any_instance.stubs(:healthy?).returns(true)
SidekiqHealth.any_instance.stubs(:processes_count).returns(1)
SidekiqHealth.any_instance.stubs(:last_heartbeat_at).returns(Time.current)
SidekiqHealth.any_instance.stubs(:max_queue_latency).returns(0.0)
SidekiqHealth.any_instance.stubs(:enqueued_count).returns(0)
SidekiqHealth.any_instance.stubs(:retry_count).returns(0)
SidekiqHealth.any_instance.stubs(:failed_count).returns(0)
SidekiqHealth.any_instance.stubs(:processed_count).returns(42)
SidekiqHealth.any_instance.stubs(:queue_breakdown).returns([ [ "default", 0, 0.0 ] ])
end
end
+2
View File
@@ -52,6 +52,8 @@ class SettingsTest < ApplicationSystemTestCase
test "can update self hosting settings" do
sign_in users(:sure_support_staff)
Rails.application.config.app_mode.stubs(:self_hosted?).returns(true)
Provider::Registry.stubs(:get_provider).with(:openai).returns(nil)
Provider::Registry.stubs(:get_provider).with(:anthropic).returns(nil)
Provider::Registry.stubs(:get_provider).with(:twelve_data).returns(nil)
Provider::Registry.stubs(:get_provider).with(:yahoo_finance).returns(nil)
Provider::Registry.stubs(:get_provider).with(:rentcast).returns(nil)