No introduction found. Create it?
litellm from bjw-s-labs/charts is more popular with 15 repositories.
Install with:
helm repo add litellm oci://ghcr.io/berriai/litellm-helm
helm install litellm litellm/litellm -f values.yamlSee examples from other people.
| Name | Repo | Stars | Version | Timestamp |
|---|---|---|---|---|
| litellm | mirceanton/home-ops | 122 | 5.2.1 | a day ago |
| litellm | ToaHartor/maisonneux | 43 | 1.100.1 | 20 days ago |
See the most popular values for this chart:
| Key | Types |
|---|---|
| string | |
controllers.litellm.containers.app.image.repository (10) ghcr.io/berriai/litellm | string |
controllers.litellm.containers.app.image.tag (10) v1.83.14-stable@sha256:d6401c001f90f3bab4bb23c5fd6d9302a7df58999ec6fa9e3f175e1f5f26544b | string |
| boolean | |
| boolean | |
| number | |
| string | |
| number | |
| number | |
| number | |
| number | |
| boolean | |
| boolean | |
| number | |
| string | |
| number | |
| number | |
| number | |
| number | |
| boolean | |
| number | |
| number | |
| string | |
| number | |
| number | |
| number | |
| boolean | |
| string | |
| string | |
| string | |
controllers.litellm.containers.app.args[] (9) - --config=/app/config.yaml | string |
| string | |
| string | |
| string | |
controllers.litellm.containers.app.env.REDIS_HOST (6) litellm-dragonfly.ai.svc.cluster.local | string |
| number, string | |
controllers.litellm.containers.app.env.GENERIC_AUTHORIZATION_ENDPOINT (5) https://sso.${SECRET_DOMAIN}/application/o/authorize/ | string |
controllers.litellm.containers.app.env.GENERIC_SCOPE (5) openid email profile litellm_role | string |
controllers.litellm.containers.app.env.GENERIC_TOKEN_ENDPOINT (5) https://sso.${SECRET_DOMAIN}/application/o/token/ | string |
controllers.litellm.containers.app.env.GENERIC_USERINFO_ENDPOINT (5) https://sso.${SECRET_DOMAIN}/application/o/userinfo/ | string |
controllers.litellm.containers.app.env.PROXY_BASE_URL (5) https://litellm.${SECRET_DOMAIN} | string |
| string | |
| string | |
controllers.litellm.containers.app.env.DATABASE_URL (1) postgresql://litellm@pooler-rw.database.svc:5432/litellm?sslmode=require&sslidentity=/var/run/secrets/prisma/client-identity.p12&sslcert=/var/run/secrets/root-ca/ca.crt | string |
| string | |
| string | |
| string | |
| string | |
| string | |
| string | |
| string | |
| string | |
| string | |
| string | |
| string | |
controllers.litellm.containers.app.env.CHATGPT_TOKEN_DIR (1) /var/lib/litellm/chatgpt | string |
| string | |
| string | |
| string | |
| string | |
| string | |
| string | |
| string | |
controllers.litellm.containers.app.env.GENERIC_USER_ID_ATTRIBUTE (1) preferred_username | string |
| string | |
| string | |
| string | |
| string | |
| string | |
controllers.litellm.containers.app.env.OPENCODE_GO_API_BASE (1) https://opencode.ai/zen/go/v1 | string |
controllers.litellm.containers.app.env.OPENCODE_ZEN_API_BASE (1) https://opencode.ai/zen/v1 | string |
| string | |
| string | |
controllers.litellm.containers.app.env.OTEL_ENDPOINT (1) http://alloy-gateway.observability.svc.cluster.local:4317 | string |
| string | |
| string | |
| string | |
| string | |
| string | |
| string | |
| string | |
| string | |
controllers.litellm.containers.app.env.REDIS_URL (1) redis://dragonfly.database.svc.cluster.local:6379/11 | string |
| string | |
| string | |
controllers.litellm.containers.app.env.YOUTUBE_ANALYTICS_MCP_TOKEN.valueFrom.secretKeyRef.key (1) YOUTUBE_ANALYTICS_MCP_TOKEN | string |
| string | |
| boolean | |
| string | |
| boolean | |
| number | |
| boolean | |
| number | |
controllers.litellm.containers.exporter.args[] (1) - --no-collector.wal | string |
| string | |
| string | |
controllers.litellm.containers.exporter.env.DATA_SOURCE_URI (1) litellm-db-rw.ai.svc.cluster.local:5432/litellm?sslmode=require | string |
| string | |
| string | |
| string | |
| string | |
controllers.litellm.containers.exporter.env.PG_EXPORTER_EXTEND_QUERY_PATH (1) /etc/pg-exporter/queries.yaml | string |
controllers.litellm.containers.exporter.image.repository (1) quay.io/prometheuscommunity/postgres-exporter | string |
controllers.litellm.containers.exporter.image.tag (1) v0.20.1@sha256:ac5ec343104fae0e2d84a27bb8d69b38430a11910c5382cad85d478d2bab713e | string |
| boolean | |
| boolean | |
| string | |
| number | |
| number | |
| number | |
| boolean | |
| boolean | |
| string | |
| number | |
| number | |
| number | |
| string | |
| string | |
| string | |
| boolean | |
| string | |
| boolean | |
| number | |
| string | |
| string | |
| string | |
| string | |
| string | |
| string | |
controllers.litellm.containers.ollama.env.TZ (1) America/Chicago | string |
controllers.litellm.containers.ollama.image.repository (1) docker.io/ollama/ollama | string |
controllers.litellm.containers.ollama.image.tag (1) 0.34.4@sha256:8262851b2846b87c649eddf3e76beb270c52f4d1bc94559f47efde16b0841551 | string |
| boolean | |
| boolean | |
| number | |
| string | |
| number | |
| number | |
| number | |
| boolean | |
| boolean | |
| number | |
| string | |
| number | |
| number | |
| number | |
| string | |
| string | |
| string | |
| string | |
controllers.litellm.containers.speaches.env.TZ (1) America/Chicago | string |
| string | |
| string | |
controllers.litellm.containers.speaches.image.repository (1) ghcr.io/speaches-ai/speaches | string |
controllers.litellm.containers.speaches.image.tag (1) 0.8.3-cpu@sha256:21e3df06d842fb7802ab470dd77c25f0e8c0d22950e8d8c6ae886e851af53ef8 | string |
| boolean | |
| boolean | |
| number | |
| string | |
| number | |
| number | |
| number | |
| boolean | |
| boolean | |
| number | |
| string | |
| number | |
| number | |
| number | |
| string | |
| string | |
| string | |
controllers.litellm.strategy (6) RollingUpdate | string |
| number | |
| string | |
controllers.litellm.initContainers.01-init-db.image.repository (1) ghcr.io/home-operations/postgres-init | string |
| string | |
controllers.litellm.initContainers.gen-prisma-cert.command[] (1) - /bin/sh | string |
controllers.litellm.initContainers.gen-prisma-cert.image.repository (1) docker.io/library/alpine | string |
| string | |
controllers.litellm.initContainers.install-pillow-deps.command[] (1) - /app/.venv/bin/python | string |
controllers.litellm.initContainers.install-pillow-deps.image.repository (1) ghcr.io/berriai/litellm | string |
controllers.litellm.initContainers.install-pillow-deps.image.tag (1) v1.83.14-stable@sha256:d6401c001f90f3bab4bb23c5fd6d9302a7df58999ec6fa9e3f175e1f5f26544b | string |
controllers.litellm.initContainers.model-import.command[] (1) - /bin/sh | string |
controllers.litellm.initContainers.model-import.image.repository (1) docker.io/ollama/ollama | string |
controllers.litellm.initContainers.model-import.image.tag (1) 0.34.4@sha256:8262851b2846b87c649eddf3e76beb270c52f4d1bc94559f47efde16b0841551 | string |
| string | |
| string | |
| string | |
controllers.litellm.initContainers.wait-for-pg.args[] (1) - postgresql | string |
| string | |
| string | |
| string | |
controllers.litellm.initContainers.wait-for-pg.image.repository (1) ghcr.io/wait4x/wait4x | string |
| string | |
| string | |
| string | |
| string | |
| string | |
| number | |
controllers.litellm.pod.topologySpreadConstraints[].topologyKey (1) kubernetes.io/hostname | string |
| string | |
| number | |
| number | |
| number | |
| number | |
| number | |
| number | |
service.app.controller (5) litellm | string |
service.port (1) ${GATEWAY_SVC_PORT} | string |
persistence.config.type (9) configMap | string |
persistence.config.name (8) litellm-configmap | string |
persistence.config.globalMounts[].path (7) /app/config.yaml | string |
| boolean | |
| string | |
persistence.config.advancedMounts.litellm.app[].path (2) /app/config.yaml | string |
| boolean | |
| string | |
| string | |
| string | |
persistence.cache.type (3) emptyDir | string |
persistence.chatgpt-auth.advancedMounts.litellm.app[].path (1) /var/lib/litellm/chatgpt | string |
persistence.chatgpt-auth.existingClaim (1) {{ .Release.Name }} | string |
| string | |
| boolean | |
persistence.exporter-queries.name (1) litellm-exporter-queries | string |
| string | |
persistence.models.accessMode (1) ReadWriteOnce | string |
| string | |
| string | |
persistence.models.advancedMounts.litellm.speaches[].path (1) /home/ubuntu/.cache/huggingface | string |
persistence.models.forceRename (1) litellm-models | string |
| string | |
persistence.models.storageClass (1) openebs-hostpath | string |
persistence.models.type (1) persistentVolumeClaim | string |
| string | |
| string | |
persistence.pillow.type (1) emptyDir | string |
persistence.prisma-certs.advancedMounts.litellm.app[].path (1) /var/run/secrets/prisma | string |
persistence.prisma-certs.advancedMounts.litellm.gen-prisma-cert[].path (1) /var/run/secrets/prisma | string |
| string | |
persistence.runtime-patches.advancedMounts.litellm.app[].path (1) /app/.venv/lib/python3.13/site-packages/sitecustomize.py | string |
| boolean | |
| string | |
persistence.runtime-patches.name (1) litellm-runtime-patches | string |
persistence.runtime-patches.type (1) configMap | string |
| string | |
persistence.tmp.type (1) emptyDir | string |
| string | |
| string | |
persistence.tmpfs.type (1) emptyDir | string |
route.app.hostnames[] (6) - litellm.${DOMAIN_NAME} | string |
route.app.parentRefs[].name (6) envoy-internal | string |
| string | |
| string | |
| string, number | |
| string | |
| string | |
route.home.hostnames[] (1) - llm.home.mirceanton.com | string |
route.home.parentRefs[].name (1) envoy-internal | string |
route.home.parentRefs[].namespace (1) network-system | string |
| boolean | |
db.database (1) ${PSQL_DB_NAME} | string |
| string | |
db.secret.name (1) ${PSQL_DB_USER}-db-creds | string |
db.secret.passwordKey (1) password | string |
db.secret.usernameKey (1) username | string |
| boolean | |
| string | |
masterkeySecretKey (3) masterkey | string |
masterkeySecretName (3) litellm-env | string |
proxy_config.general_settings.master_key (3) os.environ/PROXY_MASTER_KEY | string |
| string | |
proxy_config.model_list[].litellm_params.model (3) github_copilot/* | string |
proxy_config.model_list[].litellm_params.extra_headers.anthropic-beta (2) context-1m-2025-08-07 | string |
proxy_config.model_list[].litellm_params.api_base (1) http://qwen38.ai.svc.cluster.local:8080/v1 | string |
proxy_config.model_list[].litellm_params.api_key (1) sk-no-key-required | string |
| boolean | |
proxy_config.model_list[].model_info.cache_creation_input_token_cost (3) 0 | number |
proxy_config.model_list[].model_info.cache_read_input_token_cost (3) 0 | number |
| number | |
| number | |
| number | |
| number | |
| boolean | |
| boolean | |
proxy_config.model_list[].model_name (3) github_copilot/* | string |
proxy_config.litellm_settings.callbacks[] (2) - websearch_interception | string |
| boolean | |
| boolean | |
| boolean | |
| string | |
| string | |
proxy_config.router_settings.default_fallbacks[] (2) - claude-sonnet-5 | string |
proxy_config.search_tools[].litellm_params.api_key (2) os.environ/TAVILY_API_KEY | string |
| string | |
proxy_config.search_tools[].search_tool_name (2) tavily-search | string |
environmentSecrets[] (2) - litellm-env | string |
envVars.GITHUB_COPILOT_TOKEN_DIR (2) /cache/copilot | string |
| string | |
envVars.DOCS_URL (1) /docs | string |
envVars.GENERIC_AUTHORIZATION_ENDPOINT (1) https://${auth_subdomain}.${main_domain_websecure}/application/o/authorize/ | string |
envVars.GENERIC_SCOPE (1) openid profile email litellm_role | string |
envVars.GENERIC_TOKEN_ENDPOINT (1) https://${auth_subdomain}.${main_domain_websecure}/application/o/token/ | string |
envVars.GENERIC_USER_ROLE_ATTRIBUTE (1) litellm_role | string |
envVars.GENERIC_USERINFO_ENDPOINT (1) https://${auth_subdomain}.${main_domain_websecure}/application/o/userinfo/ | string |
envVars.PROXY_BASE_URL (1) https://${SUBDOMAIN}.${DOMAIN_WEBSECURE} | string |
envVars.PROXY_LOGOUT_URL (1) https://${auth_subdomain}.${main_domain_websecure}/application/o/${APP}/end-session/ | string |
envVars.REDIS_URL (1) redis://dragonfly-cluster.dragonfly.svc.cluster.local:6379/3 | string |
| string | |
image.repository (2) ghcr.io/berriai/litellm | string |
| boolean | |
| string | |
| string | |
| string | |
| string | |
| string | |
| boolean | |
| string | |
| boolean | |
volumeMounts[].mountPath (2) /cache | string |
volumeMounts[].name (2) litellm-cache | string |
volumeMounts[].subPath (2) access-token | string |
volumes[].name (2) litellm-cache | string |
volumes[].secret.items[].key (2) GITHUB_COPILOT_ACCESS_TOKEN | string |
volumes[].secret.items[].path (2) access-token | string |
volumes[].secret.secretName (2) litellm-env | string |
args[] (1) - --config | string |
configMaps.config.data."config.yaml" (1) model_list:
# --- primary: DGX Spark (spark-ab23) over the LAN ---
# Two engines on the Spark, both OpenAI-compatible at /v1 and both
# without auth of their own: vLLM (:8000) serves gpt-oss, Ollama
# (:11434) serves the vision and dense models. Each ignores the
# bearer token, so api_key is a placeholder, not a secret. LiteLLM
# is the authentication boundary in front of them; they are
# reachable by anything on VLAN 10.
#
# Addressed by IP, deliberately, after the name stopped resolving.
#
# This was http://spark-ab23.${SECRET_DOMAIN}:11434/v1 and that
# worked when it was set. Hours later the same lookup returned
# NODATA from the UDM (NOERROR, no A record) while the BARE
# `spark-ab23` still answered 10.0.10.237 — i.e. UniFi's automatic
# DHCP-hostname registration survived but the explicit
# derekjacobs.dev record did not. Every Hindsight call failed as
# `litellm.InternalServerError: Connection error` until this change.
#
# The bare name is not a usable substitute: with ndots:5 the
# cluster search path swallows a single-label name, and it resolves
# only as `spark-ab23.` with a trailing dot, which not every HTTP
# client normalises.
#
# Using the IP is SAFE now in a way it was not originally. The
# first version of this file used 10.0.10.237 and carried a warning
# that it was a DHCP lease which could move silently. It is now a
# fixed assignment on the UDM, so that failure mode is gone.
#
# Switch back to the name once the A record is restored and
# verified resolving FROM A POD, not just from the Spark — the
# Spark resolves names the cluster cannot.
#
# Served by vLLM on :8000 (spark-setup, roles/vllm), not Ollama,
# since 2026-09. vLLM pages its KV cache per request instead of
# preallocating 8 fixed 32k slots, so there is no slot cap to
# queue behind: measured on an idle Spark, 226 vs 153 tok/s
# aggregate at 8 concurrent, and 385 at 16 where Ollama queued
# requests 26s for a slot. It is ~25% slower for a LONE request
# (46 vs 61 tok/s) -- the trade taken deliberately.
#
# Like Ollama, vLLM runs with no --api-key: the placeholder below
# is sent and ignored, and LiteLLM stays the auth boundary.
#
# num_retries: ~4.7% of replayed reflect tool calls (3/64) came back
# 500 "unexpected tokens remaining in message header" -- gpt-oss
# intermittently emits a malformed harmony tool-call header that
# vLLM's parser rejects. The same request succeeds on retry, and
# hindsight's own retries cover only its background worker jobs, so
# without this an interactive reflect surfaces the 500. Two retries
# take it to ~1 in 10,000.
- model_name: spark-gpt-oss-20b
litellm_params:
model: openai/gpt-oss-20b
api_base: http://10.0.10.237:8000/v1
api_key: unused-no-auth
num_retries: 2
# --- long-context alias for hindsight reflect ---
# The SAME vLLM instance as spark-gpt-oss-20b, kept as a separate
# model_name only so hindsight's reflect config and virtual key
# need no change. It used to be a second, hand-made ollama on
# :11435 (one slot, 98k window, ~17 GiB for a second copy of the
# weights), because reflect prompts on the flux-talos bank are
# 55-65k tokens and the main 32k window silently kept only the tail.
# vLLM's single instance serves the full 131,072-token window to
# any request: an ~87k-token prompt measured 18s to first token
# alongside short traffic.
#
# It no longer has ONE slot, so the old "nothing else may point
# here" rule is gone -- but keep it reflect-only anyway, so the
# two can be split again without touching other consumers.
- model_name: spark-gpt-oss-20b-long
litellm_params:
model: openai/gpt-oss-20b
api_base: http://10.0.10.237:8000/v1
api_key: unused-no-auth
# Reflect is the tool-calling stage; see num_retries above.
num_retries: 2
# --- audio (spark-setup `audio` role) ---
# Two plain Docker containers on the Spark, NOT Ollama, so they use
# their own ports and the `mode` below tells LiteLLM which OpenAI
# audio route each one serves. Like Ollama they have no auth of their
# own; LiteLLM is the boundary. Round trip verified on the box: TTS ->
# WAV -> STT returned the synthesised sentence.
#
# Speech-to-text: parakeet.cpp (NVIDIA Parakeet TDT 0.6B v3,
# multilingual). An 11 s clip transcribed word-for-word in ~0.3 s warm.
# This is the model Mealie's "audio" provider slot should point at.
# The `openai/` prefix + api_base makes LiteLLM speak the OpenAI wire
# format to it; the part after `openai/` is only what LiteLLM sends as
# `model`, which parakeet accepts and ignores.
- model_name: spark-stt
litellm_params:
model: openai/tdt-0.6b-v3
api_base: http://10.0.10.237:8081/v1
api_key: unused-no-auth
model_info:
mode: audio_transcription
# Text-to-speech: Qwen3-TTS 1.7B voice-clone. The `voice` field selects
# an entry in voices.json on the Spark (a reference clip + its exact
# transcript), which is how character voices are added -- see the
# spark-setup README. ~7 s warm for a ~5 s sentence, so this is fine
# for notifications and read-aloud, not for real-time conversation.
#
# ⚠️ Streams the WAV, so the header's length field is bogus (a 4.8 s
# clip reports ~89,000 s). Players cope; strict parsers may not. Ask for
# response_format=mp3 (encoded after generation, so the header is right).
- model_name: spark-tts
litellm_params:
model: openai/tts-1
api_base: http://10.0.10.237:8082/v1
api_key: unused-no-auth
model_info:
mode: audio_speech
# --- removed 2026-09-29: spark-qwen25-14b (qwen2.5:14b-instruct) ---
# A dense non-reasoning alternative to gpt-oss that nothing used: it
# was reflect's escape hatch until reflect needed a ~92k window, and it
# was never faster (26 tok/s vs gpt-oss's 63 -- dense loses to MoE on
# a bandwidth-bound box). Removed for the memory hazard below.
#
# ⚠️ MEMORY HAZARD, still true for anything served by Ollama. It runs
# with OLLAMA_KEEP_ALIVE=-1, so ANY model touched through this proxy
# stays resident forever, and its KV cache is slots x context per
# model: one evaluation request against the dense 14B pinned 73 GiB
# at 4 slots and took the box to 103 of 121 GiB, with hindsight
# operations timing out into PERMANENT failures. With vLLM now
# holding a fixed ~36 GiB, the same load would reach earlyoom, whose
# largest preferred target is vLLM. Before adding a dense model here,
# size slots x context x its per-token KV against what is free.
# --- vision ---
# The ONLY model here that can see an image. The gpt-oss entries above
# are text-only, which is why karakeep's image slot and
# homebox-companion's photo item-detection were dead after the
# Azure migration -- gpt-4.1-mini and gpt-5-mini were both
# vision-capable and nothing on the Spark replaced that.
#
# Measured on this box at OLLAMA_NUM_PARALLEL=8, not estimated:
# 20 GiB resident alongside gpt-oss's 13 GiB (54 of 121 GiB used,
# 67 free, when gpt-oss still ran in Ollama -- it is on vLLM now),
# 17.6s cold including model load, then 1.3-2.7s warm.
# It read two lines of rendered text exactly, including a currency
# amount, and identified two shapes and their colours.
#
# It is DENSE, unlike gpt-oss -- see the memory hazard above. At 7B
# that is affordable; it would not be at 14B+.
#
# ⚠️ Images must be sent as a base64 `data:` URI. Ollama does NOT
# fetch remote http(s) image URLs -- its own docs list "Image URL"
# as unsupported -- so a consumer that passes a link instead of
# bytes gets a failure that looks like a model problem. Verified
# working here through this exact OpenAI-compatible shape.
- model_name: spark-qwen25-vl-7b
litellm_params:
model: openai/qwen2.5vl:7b
api_base: http://10.0.10.237:11434/v1
api_key: unused-ollama-accepts-any
# There is deliberately NO fallback entry here, and no
# router_settings.fallbacks block.
#
# An Azure `gpt-5-mini` member was configured initially, but both
# the Azure and Anthropic credit pools are exhausted, so it returned
# 401 on every call. A fallback pointing at a dead credential is
# worse than no fallback at all: LiteLLM spends a failed round-trip
# on it and then buries the real cause inside a nested error
# ("Connection error.. Error doing the fallback:
# AuthenticationError.."), so an outage of the Spark reads as an
# auth problem somewhere else. Failing cleanly against a single
# backend is the honest behaviour, and reachability is alerted on
# separately by the `DGX Spark (vLLM)` (gpt-oss) and
# `DGX Spark (Ollama)` (vision, dense) gatus endpoints.
#
# To restore a fallback, add a second model_list member and a
# router_settings.fallbacks block. Any OpenAI-compatible provider
# works -- OpenRouter's free tier is the likely candidate -- and it
# needs exactly one new OpenBao field wired through the
# ExternalSecret. Do not re-add a member without first confirming
# its credential actually authenticates.
litellm_settings:
drop_params: true
# Response cache in the shared dragonfly, OPT-IN ONLY (default_off).
# A request is cached only when the caller sends
# `cache: {"use-cache": true}`; everything else, including all of
# hindsight's retain/consolidation/reflect traffic, is never cached.
# That is deliberate: those prompts are generated from evolving memory
# with sampling on, so a blanket cache would return a stale or
# identical "creative" answer. Turning it on per request is safe;
# turning it on globally would be a behaviour change nobody asked for.
#
# `namespace` prefixes every key so this tenant is separable inside a
# dragonfly shared with authentik, nextcloud, grafana and others. It
# runs --cluster_mode=emulated, where SELECT-based db isolation is not
# reliable, so a key prefix is the isolation mechanism (same as
# tracearr's REDIS_PREFIX).
cache: true
cache_params:
type: redis
host: dragonfly.database.svc.cluster.local
port: 6379
namespace: litellm.cache
mode: default_off
# Shared state that must NOT be per-process: tpm/rpm limits, each virtual
# key's max_parallel_requests, spend counters. Without this every worker
# enforces its own copy, so a key capped at 2 parallel requests can
# actually run 2 per worker, defeating the Spark slot budget the caps
# exist to enforce, and spend can overshoot. This is what the UI's
# "No Redis configured" banner is about; it clears once it is set.
general_settings:
coordination_redis:
host: dragonfly.database.svc.cluster.local
port: 6379
# Keep only the last hour of spend-log rows. This is a workaround for
# a collision, not a storage decision.
#
# LiteLLM keys each SpendLogs row on the provider's response id and
# silently drops a row whose id already exists (its own code notes
# "SpendLogs does not allow duplicate request_id"). Ollama makes ids
# like `chatcmpl-335` from a range of about 1,000, so once the table
# holds ~1,000 rows almost every new call collides and is not logged.
# Measured 2026-09-23: 1,645 rows with ids spanning 0-998, and 7 of 8
# probe calls returned an id already in the table and produced no row,
# so store_prompts_in_spend_logs was capturing ~1% of traffic. It was
# NOT caused by running two replicas: the same loss showed on the
# single pod before that change.
#
# A short window keeps the table small enough that collisions are
# rare (roughly 40 rows/hour here, so a few percent). The interval
# defaults to "1d", which would let a day of rows build up between
# cleanups and defeat the point, so it is set explicitly. Daily spend
# aggregates and key/team totals live in other tables and are
# unaffected. It also stops stored prompts, mental models included,
# from accumulating in Postgres indefinitely, which the unset default
# would do.
#
# Not a fix: prompt history is only ever the last hour, and a burst
# can still collide. Hindsight's own /llm-requests log has unique ids
# and is the reliable record for its calls.
maximum_spend_logs_retention_period: "1h"
maximum_spend_logs_retention_interval: "30m"
# The router's own Redis: cooldowns and usage-based routing state. Kept
# as a separate setting because LiteLLM treats it as a separate client.
router_settings:
redis_host: dragonfly.database.svc.cluster.local
redis_port: 6379 | string |
configMaps.litellm-config.data."config.yaml" (1) model_list:
- model_name: qwen3-1.7b-nothink
litellm_params:
# Generic OpenAI-compatible route (not ollama_chat) so
# logprobs/top_logprobs pass through spec-faithfully.
model: openai/qwen3-1.7b-nothink
api_base: http://127.0.0.1:11434/v1
api_key: ollama # required by the openai provider; unused
# OpenRouter cloud models ride the same proxy/key — one
# endpoint for local + cloud. Key comes from hermes-secret.
- model_name: deepseek-v4.1-flash
litellm_params:
model: openrouter/deepseek/deepseek-v4.1-flash
api_key: os.environ/OPENROUTER_API_KEY
# Audio via the in-pod speaches sidecar. The request's `voice`
# param passes straight through to piper (e.g.
# en_US-glados-medium to match the wyoming-piper voice).
- model_name: whisper
litellm_params:
model: openai/Systran/faster-whisper-small
api_base: http://127.0.0.1:8000/v1
api_key: speaches # required by the openai provider; unused
model_info:
mode: audio_transcription
- model_name: piper
litellm_params:
model: openai/rhasspy/piper-voices
api_base: http://127.0.0.1:8000/v1
api_key: speaches # required by the openai provider; unused
model_info:
mode: audio_speech
litellm_settings:
# Don't 400 on params a local backend doesn't implement.
drop_params: true
num_retries: 2
general_settings:
master_key: os.environ/LITELLM_MASTER_KEY
| string |
configMaps.litellm-config.forceRename (1) litellm-config | string |
| boolean | |
| string | |
| number | |
| string | |
| number | |
| boolean | |
| number | |
| string | |
| number | |
extraEnvVars[].name (1) GENERIC_CLIENT_ID | string |
extraEnvVars[].valueFrom.secretKeyRef.key (1) clientID | string |
extraEnvVars[].valueFrom.secretKeyRef.name (1) ${APP}-oidc-authentik-application | string |
fullnameOverride (1) litellm | string |
| boolean | |
ingress.enabled (1) false | boolean |
redis.enabled (1) false | boolean |
| number | |
| string | |
| string | |
| string |