# ============================================================================ # Hindsight — helm values (Phase C). Consumed by the ArgoCD Application source 1 # via `helm.valueFiles: ["$values/values.yaml"]` (openviking multi-source pattern). # # Design decisions (all verified against chart v0.9.1 + rendered output): # - Chart is the single source for the app (api, control-plane, services, # probes, ingress). We do NOT hand-roll Deployments/Services. # - Postgres is EXTERNAL (separate Deployment in this dir, firecrawl pattern) # => postgresql.enabled: false, external.* points at hindsight-postgres:5432. # - Secrets come from 1Password via ExternalSecret => existingSecret: # hindsight-credentials. The chart does envFrom(secretRef) so # HINDSIGHT_API_LLM_API_KEY / HINDSIGHT_API_MCP_AUTH_TOKEN are injected # automatically; POSTGRES_PASSWORD is a secretKeyRef that K8s expands into # HINDSIGHT_API_DATABASE_URL (verified with a live envFrom test pod). # - LLM is the Nous free-tier inference endpoint # (https://inference-api.nousresearch.com/v1), model # `upstage/solar-pro4:free` (tool-calling, verified reflect). Fallback # (documented, NOT deployed): `stepfun/step-3.7-flash:free`. API key via existingSecret # envFrom (hindsight-credentials / HINDSIGHT_API_LLM_API_KEY), sourced from # 1Password `nous` item per decision 4. # - Ingress is driven through the chart's NATIVE ingress template (approved # plan: "Ingress driven through values.yaml"). api.service.port=8888, # controlPlane.service.port=3000. # - Image tag defaults to .Values.version (root) when api.image.tag is unset, # so version: "0.9.1" pins the API image to 0.9.1. # ============================================================================ version: "0.9.1" # ---------------------------------------------------------------------------- # External PostgreSQL (chart's bundled postgresql is disabled). # password is the K8s env expansion `$(POSTGRES_PASSWORD)` — the chart defines # POSTGRES_PASSWORD as a secretKeyRef (hindsight-credentials / postgres-password) # earlier in the same container, so K8s substitutes it at container start. # ---------------------------------------------------------------------------- postgresql: enabled: false external: host: hindsight-postgres port: 5432 username: hindsight database: hindsight password: $(POSTGRES_PASSWORD) # ExternalSecret (from 1Password) that the chart injects via envFrom(secretRef). # Keys it must expose: postgres-password, HINDSIGHT_API_LLM_API_KEY, # HINDSIGHT_API_MCP_AUTH_TOKEN. See externalsecret.yaml in this dir. existingSecret: hindsight-credentials # ---------------------------------------------------------------------------- # API container environment (explicit env entries; the chart renders this map # to individual env vars). LLM points at the Nous free-tier inference endpoint # (https://inference-api.nousresearch.com/v1), model upstage/solar-pro4:free # (tool-calling, verified retain/recall). stepfun/step-3.7-flash was swapped out # because it rejects Hindsight's tagged fact-extraction (BadRequestError 400 # 'missing tags'); solar-pro4 is the previously-verified-good Nous model for the # extract/retain path (tasks t_e3375410 / t_d0dffc3d). # HINDSIGHT_API_LLM_API_KEY is NOT set here — it comes from the existingSecret # via envFrom (1Password `nous` item). # ---------------------------------------------------------------------------- api: env: # CUT OVER to vLLM (t_5508360a, 2026-08-31): dashboard explicitly # approved "stop and disable llama-swap and start vLLM" as a breaking # change. llama-swap is now stopped+disabled on astro-orbiter; vLLM # was the permanent, boot-persistent replacement (originally # Qwen2.5-32B-Instruct-AWQ). This exact config (base URL, model name, # retry-safe low concurrency) was validated end-to-end in t_e6facb19's # shadow window (health, /v1/chat/completions, live hindsight_retain+ # recall round-trip) before that task reverted it pending this # decision -- now re-applied for real. See roles/deploy-vllm/README.md # "Critical architectural finding" + "Consumer cutover status" for the # full history. # # MODEL SWAP (t_r1d32b_swap, 2026-09-01): astro-orbiter's vLLM primary # model changed from Qwen2.5-32B-Instruct-AWQ to # DeepSeek-R1-Distill-Qwen-32B-AWQ (single-model deployment; nomic-embed # and Qwen3-8B-AWQ both disabled on that host). Superseded below. # # MODEL SWAP #2 (t_gemma4_swap, 2026-09-01): DeepSeek-R1-Distill-Qwen-32B # retired after confirming its tool_choice="auto" reliability is a # known, documented DeepSeek-R1-distillation limitation (upstream # GitHub-confirmed: trained on pure reasoning traces, no function- # calling data) -- not relevant to Hindsight's pure-text extraction # use case, but disqualifying for agent-facing Hermes profiles, which # drove the swap. Replaced with Gemma 4 26B A4B (Google, Apache 2.0, # US-origin). Same endpoint (http://astro-orbiter:8000/v1), same API # key -- only the served model name changed. Gemma 4 does NOT emit a # reasoning trace by default (confirmed live) -- simpler completion # parsing than DeepSeek-R1's always-on blocks. HINDSIGHT_API_LLM_BASE_URL: "http://astro-orbiter:8000/v1" HINDSIGHT_API_LLM_PROVIDER: "openai" HINDSIGHT_API_LLM_MODEL: "Gemma-4-26B-A4B-it-AWQ" # vLLM's max_model_len is now 65536 (up from DeepSeek's 32768, up from # the original 8192 role default). Gemma 4's native context is 256K; # 65536 is astro-orbiter's configured ceiling, comfortably above # Hermes's 64K floor. Completion cap left at 4096 pending live # verification -- Gemma 4 doesn't burn tokens on unwanted reasoning # traces the way DeepSeek-R1 did, so 4096 should have MORE effective # headroom for the actual extraction output than it did before. HINDSIGHT_API_RETAIN_MAX_COMPLETION_TOKENS: "4096" # DO NOT set HINDSIGHT_API_EMBEDDINGS_* here (t_e6facb19, 2026-08-31 # attempted this, reverted after a production incident — see below). # # DISCOVERY: Hindsight's embeddings provider was NEVER pointed at # astro-orbiter. It defaults to "local" (bundled sentence-transformers, # BAAI/bge-small-en-v1.5, 384 dimensions) whenever # HINDSIGHT_API_EMBEDDINGS_PROVIDER is unset — verified via # `kubectl exec ... env | grep -i embed` showing NO # HINDSIGHT_API_EMBEDDINGS_* vars in the live pod, despite this file's # LLM section referencing astro-orbiter for years. The nomic-embed- # text-v1.5 model documented across mk-labs skills as "Hindsight's # embedding model" was OpenViking's embedding model, not Hindsight's. # # INCIDENT: pointing HINDSIGHT_API_EMBEDDINGS_PROVIDER at vLLM's # nomic-embed-text-v1.5 (768 dimensions) crash-looped hindsight-api on # rollout: `RuntimeError: Cannot change embedding dimension from 384 to # 768: memory_units table contains 1289 rows with embeddings.` The # migration path (`ensure_embedding_dimension` in migrations.py) refuses # a live dimension change without either re-embedding everything or # deleting all existing memory_units rows across every bank (jarvis, # hermes, war-machine, and ~18 other agent banks) — a destructive, # irreversible operation requiring explicit human approval, not # something to do as a side effect of an infra migration task. Reverted # immediately; Hindsight keeps its bundled local embedder (384-dim, # unchanged, zero data risk) until a deliberate, approved re-embedding # migration is planned as its own task. # --- t_d7f8cd65: fix 502s on the serial astro-orbiter node --- # astro-orbiter is a single llama-swap process (serial: 1 generate at a # time, ctx 64K). Hindsight's default LLM concurrency is 32, so a retain # burst hits the node with N parallel calls -> the node rejects/times out # the extras -> hindsight-api surfaces APITimeoutError as 502. astro-orbiter # is the ONLY LLM endpoint (all ops route there), so cap the whole pool to # 1 and pin retain to 1 as well. The upstream chart exposes these as native # semaphore config (HINDSIGHT_API_*_MAX_CONCURRENT); no code change needed. # Kept at 1 post-cutover: vLLM's single-process-per-model design is also # effectively serial for a single generative model instance under this # GPU's VRAM budget (KV cache sized tight against the 24GB card at # max_model_len=65536 for Gemma-4-26B-A4B-it-AWQ, t_gemma4_swap # 2026-09-01 — previously 32768 for DeepSeek-R1-Distill-Qwen-32B-AWQ, # previously 8192 for Qwen2.5-32B-Instruct-AWQ). HINDSIGHT_API_LLM_MAX_CONCURRENT: "1" HINDSIGHT_API_RETAIN_LLM_MAX_CONCURRENT: "1" # Client + per-request timeout. Default is 120s; a 29K-token retain runs # ~29s and under load a single long retain can reach ~90s. Raise to 600s to # cover the longest round-trip so the serial call never times out the client # (keep >= ingress proxy-read-timeout below). Per-op retain timeout pins the # retain path explicitly; the global timeout covers reflect/consolidation. HINDSIGHT_API_LLM_TIMEOUT: "600" HINDSIGHT_API_RETAIN_LLM_TIMEOUT: "600" # ---------------------------------------------------------------------------- # Ingress via the chart's native template. # /health,/v1,/mcp,/ext -> api:8888 (longest-prefix wins in nginx) # / -> controlPlane:3000 # TLS secret hindsight-tls provisioned by the letsencrypt-prod issuer. # ---------------------------------------------------------------------------- ingress: enabled: true className: "nginx" annotations: cert-manager.io/cluster-issuer: "letsencrypt-prod" # Raised read/send timeout so a slow agentic reflect / long single retain # (up to ~90s under load on the serial astro-orbiter node; client timeout # is 600s per t_d7f8cd65) can complete before nginx cuts the connection. # Raised 300 -> 600 (t_d7f8cd65) to cover the longest retain round-trip. nginx.ingress.kubernetes.io/proxy-read-timeout: "600" nginx.ingress.kubernetes.io/proxy-send-timeout: "600" hosts: - host: cosmic-rewind.local.mk-labs.cloud paths: - path: /health pathType: Prefix service: api - path: /v1 pathType: Prefix service: api - path: /mcp pathType: Prefix service: api - path: /ext pathType: Prefix service: api - path: / pathType: Prefix service: controlPlane tls: - hosts: - cosmic-rewind.local.mk-labs.cloud secretName: hindsight-tls