diff --git a/cluster/platform/openviking/values.yaml b/cluster/platform/openviking/values.yaml index 6f89988..3913b37 100644 --- a/cluster/platform/openviking/values.yaml +++ b/cluster/platform/openviking/values.yaml @@ -156,7 +156,19 @@ config: # estimator is not the same tokenizer llama.cpp uses to count context, so token # counts won't match 1:1 between the two. Approved by Ryan as lowest-risk fix # (option 1 of 3) vs. touching the astro-orbiter serving stack further. - max_input_tokens: 1536 + # + # ROOT CAUSE (2026-08-15 incident): 1536 was still not low enough. Observed + # llama.cpp actual n_prompt_tokens vs. OpenViking's own max_input_tokens=1536 + # estimate ratio ranged 1.35x-1.86x across real ingested chunks (see homelab + # re-ingest circuit-breaker errors, e.g. estimate 1536 -> actual 2860 tokens, + # 2124, 2088, 2066... all > 2048 n_ctx ceiling). OpenViking's estimator + # (likely a chars/4 or similar heuristic) undercounts vs. llama.cpp's real + # BPE/wordpiece tokenizer for this corpus's content (dense code/config + # snippets tokenize denser than the estimator assumes). Lowering to 1536 alone + # does not hold for all chunks; using worst-observed ratio (1.86x) with margin, + # 2048 / 1.86 ~= 1100, rounded down further for safety across untested + # corpora -> 1024. + max_input_tokens: 1024 # ============================================================================ # VLM / Summarization configuration (L0/L1/L2 generation) @@ -169,7 +181,12 @@ config: vlm: api_base: "http://astro-orbiter:8002/v1" api_key: "${OPENVIKING_VLM_API_KEY}" # Placeholder: "local-llama" or similar - model: "llama3.1-8b" # Confirm exact alias from astro-orbiter /v1/models before applying + # Fixed 2026-08-15: "llama3.1-8b" does not exist on astro-orbiter's /v1/models + # (caused every summarization call to fail with 400 model not found, endless + # circuit-breaker retries). Actual served model id/alias confirmed via + # /home/hermes/git/homelab/ansible/playbooks/day2_add_nomic_embed.yml and + # day2_per_model_ctx_size.yml: "Meta-Llama-3.1-8B-Instruct-Q4_K_M". + model: "Meta-Llama-3.1-8B-Instruct-Q4_K_M" provider: "openai" temperature: 0.0 max_retries: 2