# ============================================================================ # Hindsight — helm values (Phase C). Consumed by the ArgoCD Application source 1 # via `helm.valueFiles: ["$values/values.yaml"]` (openviking multi-source pattern). # # Design decisions (all verified against chart v0.9.1 + rendered output): # - Chart is the single source for the app (api, control-plane, services, # probes, ingress). We do NOT hand-roll Deployments/Services. # - Postgres is EXTERNAL (separate Deployment in this dir, firecrawl pattern) # => postgresql.enabled: false, external.* points at hindsight-postgres:5432. # - Secrets come from 1Password via ExternalSecret => existingSecret: # hindsight-credentials. The chart does envFrom(secretRef) so # HINDSIGHT_API_LLM_API_KEY / HINDSIGHT_API_MCP_AUTH_TOKEN are injected # automatically; POSTGRES_PASSWORD is a secretKeyRef that K8s expands into # HINDSIGHT_API_DATABASE_URL (verified with a live envFrom test pod). # - LLM is the Nous free-tier inference endpoint # (https://inference-api.nousresearch.com/v1), model # `upstage/solar-pro4:free` (tool-calling, verified reflect). Fallback # (documented, NOT deployed): `stepfun/step-3.7-flash:free`. API key via existingSecret # envFrom (hindsight-credentials / HINDSIGHT_API_LLM_API_KEY), sourced from # 1Password `nous` item per decision 4. # - Ingress is driven through the chart's NATIVE ingress template (approved # plan: "Ingress driven through values.yaml"). api.service.port=8888, # controlPlane.service.port=3000. # - Image tag defaults to .Values.version (root) when api.image.tag is unset, # so version: "0.9.1" pins the API image to 0.9.1. # ============================================================================ version: "0.9.1" # ---------------------------------------------------------------------------- # External PostgreSQL (chart's bundled postgresql is disabled). # password is the K8s env expansion `$(POSTGRES_PASSWORD)` — the chart defines # POSTGRES_PASSWORD as a secretKeyRef (hindsight-credentials / postgres-password) # earlier in the same container, so K8s substitutes it at container start. # ---------------------------------------------------------------------------- postgresql: enabled: false external: host: hindsight-postgres port: 5432 username: hindsight database: hindsight password: $(POSTGRES_PASSWORD) # ExternalSecret (from 1Password) that the chart injects via envFrom(secretRef). # Keys it must expose: postgres-password, HINDSIGHT_API_LLM_API_KEY, # HINDSIGHT_API_MCP_AUTH_TOKEN. See externalsecret.yaml in this dir. existingSecret: hindsight-credentials # ---------------------------------------------------------------------------- # API container environment (explicit env entries; the chart renders this map # to individual env vars). LLM points at the Nous free-tier inference endpoint # (https://inference-api.nousresearch.com/v1), model upstage/solar-pro4:free # (tool-calling, verified retain/recall). stepfun/step-3.7-flash was swapped out # because it rejects Hindsight's tagged fact-extraction (BadRequestError 400 # 'missing tags'); solar-pro4 is the previously-verified-good Nous model for the # extract/retain path (tasks t_e3375410 / t_d0dffc3d). # HINDSIGHT_API_LLM_API_KEY is NOT set here — it comes from the existingSecret # via envFrom (1Password `nous` item). # ---------------------------------------------------------------------------- api: env: # Restore (2026-08-29, t_e0e6f7ca): astro-orbiter back online; move LLM back # to local Qwen3.8-27B-Q4_K_M on llama-swap. Nous free tier returns 400 # 'missing tags' on Hindsight structured fact-extraction (retain broken). HINDSIGHT_API_LLM_BASE_URL: "http://astro-orbiter:8001/v1" HINDSIGHT_API_LLM_PROVIDER: "openai" HINDSIGHT_API_LLM_MODEL: "Qwen3.8-27B-Q4_K_M" # --- t_d7f8cd65: fix 502s on the serial astro-orbiter node --- # astro-orbiter is a single llama-swap process (serial: 1 generate at a # time, ctx 64K). Hindsight's default LLM concurrency is 32, so a retain # burst hits the node with N parallel calls -> the node rejects/times out # the extras -> hindsight-api surfaces APITimeoutError as 502. astro-orbiter # is the ONLY LLM endpoint (all ops route there), so cap the whole pool to # 1 and pin retain to 1 as well. The upstream chart exposes these as native # semaphore config (HINDSIGHT_API_*_MAX_CONCURRENT); no code change needed. HINDSIGHT_API_LLM_MAX_CONCURRENT: "1" HINDSIGHT_API_RETAIN_LLM_MAX_CONCURRENT: "1" # Client + per-request timeout. Default is 120s; a 29K-token retain runs # ~29s and under load a single long retain can reach ~90s. Raise to 600s to # cover the longest round-trip so the serial call never times out the client # (keep >= ingress proxy-read-timeout below). Per-op retain timeout pins the # retain path explicitly; the global timeout covers reflect/consolidation. HINDSIGHT_API_LLM_TIMEOUT: "600" HINDSIGHT_API_RETAIN_LLM_TIMEOUT: "600" # ---------------------------------------------------------------------------- # Ingress via the chart's native template. # /health,/v1,/mcp,/ext -> api:8888 (longest-prefix wins in nginx) # / -> controlPlane:3000 # TLS secret hindsight-tls provisioned by the letsencrypt-prod issuer. # ---------------------------------------------------------------------------- ingress: enabled: true className: "nginx" annotations: cert-manager.io/cluster-issuer: "letsencrypt-prod" # Raised read/send timeout so a slow agentic reflect / long single retain # (up to ~90s under load on the serial astro-orbiter node; client timeout # is 600s per t_d7f8cd65) can complete before nginx cuts the connection. # Raised 300 -> 600 (t_d7f8cd65) to cover the longest retain round-trip. nginx.ingress.kubernetes.io/proxy-read-timeout: "600" nginx.ingress.kubernetes.io/proxy-send-timeout: "600" hosts: - host: cosmic-rewind.local.mk-labs.cloud paths: - path: /health pathType: Prefix service: api - path: /v1 pathType: Prefix service: api - path: /mcp pathType: Prefix service: api - path: /ext pathType: Prefix service: api - path: / pathType: Prefix service: controlPlane tls: - hosts: - cosmic-rewind.local.mk-labs.cloud secretName: hindsight-tls