--- # ------------------------------------------------------------------------------ # FILE: ansible/host_vars/astro_orbiter/vars.yml # HOST: astro-orbiter (10.1.71.130) # ROLE: llama.cpp LLM inference host — Ryzen 7 5800XT / RTX 3090 (ATX rebuild, # 2026-08-04). Superseded the prior AMD RX 5700 / Ollama config below; # drive was transplanted into new hardware, not reinstalled. # ------------------------------------------------------------------------------ ansible_host: 10.1.71.130 ansible_user: jarvis ansible_ssh_private_key_file: ~/.ssh/id_jarvis ansible_become: true # LVM root expansion — xlarge template uses sda3 partition, standard VG/LV names common_expand_root_lvm: true common_root_pv: /dev/sda3 common_root_vg: ubuntu-vg common_root_lv: ubuntu-lv # --- Staged GGUF models for the llama.cpp router (:8002) --------------------- # Data-driven list consumed by roles/llm-inference-multimodel tasks/models.yml # (loop -> tasks/stage_model.yml). Each entry is idempotently staged into # /opt/models: stat + EXACT-size check vs HF manifest; skip (no download, no # restart) when present + size matches. Source repos are public bartowski GGUFs # on HuggingFace (no auth). A router restart is notified ONLY when a new GGUF # is actually downloaded. # Added 2026-08-12 (War Machine): codify Phi-3.5-mini-instruct-Q8_0 and # Meta-Llama-3.1-8B-Instruct-Q4_K_M as router models alongside the production # Qwen3.6-35B-A3B-UD-Q4_K_S. The live files were already present/correct on # astro-orbiter; this pass codifies them. Future adds = append to this list. # Router --models-max override for astro-orbiter. # Default in defaults/main.yml is 1 (conservative). Bumped to 4 on 2026-08-12 # (t_33acbb2e) so the router can keep more than one GGUF resident on-demand # and LRU-evict when needed. # # VRAM NOTE (t_33acbb2e, updated t_55c164f5, updated t_34b96e83, updated t_f5f7e9ad, updated t_441470b9): # With models-max=4 and all 5 GGUFs registered, worst case is all 5 loaded simultaneously: # Qwen3.8-27B Q4_K_M: ~23.3GB (weights ~17.1GB + KV ~6.2GB @ 128K ctx, q4_0) ← UPDATED # Phi-3.5-mini-instruct Q8_0: ~4.3GB (weights ~3.8GB + KV ~0.5GB @ 32K ctx) # Meta-Llama-3.1-8B Q4_K_M: ~5.6GB (weights ~4.6GB + KV ~0.2GB @ 8K ctx) # Qwen2.5-Coder-14B Q4_K_M: ~9.0GB (weights ~8.4GB + KV ~0.6GB @ 16K ctx) # nomic-embed-text-v1.5 Q4_K_M: ~0.09GB (~84MB, embedding only — no KV cache) # Total worst-case: ~42.3GB >> 24GB RTX 3090 # # OOM RISK: Full co-residency is impossible on 24GB. LRU eviction prevents this # in practice: models-max=4 means the router can REGISTER 5 models but only keeps # up to 4 LOADED simultaneously — the router will evict the LRU model when a new # one is needed. nomic-embed-text-v1.5 is pinned via sleep-idle-seconds=-1 and # load-on-startup=true but it uses only ~84MB, so it never meaningfully changes # the budget. In single-user homelab operation, only one generative model is active # at a time alongside the always-resident embedding model. # Qwen3.8-27B alone uses ~23.1GB (weights+KV); co-residency with Coder (~9GB) = ~32GB > 24GB. # LRU eviction handles this automatically — the router evicts the idle model before # loading the new one. Ryan should be aware this means model-switching always incurs # a ~30-60s cold-load latency when switching between Qwen3.8-27B and any other model. # Proceeding to models-max=4 as instructed; flagged for Ryan's attention. # Router --models-max override for astro-orbiter. # UPDATED (t_f5f7e9ad, 2026-08-16): Set to 2 because Qwen3.8-27B-Q4_K_M # uses 17,804 MiB at 65536 ctx. Only nomic-embed (558MB, pinned) and ONE # generative model can be resident simultaneously. Co-residency of Qwen3.8 # with any auxiliary model (Phi 8.3GB, Llama 5.9GB, Coder 9GB) exceeds 24GB. # models-max=2: slot 1 = nomic-embed (pinned, always loaded), slot 2 = LRU # generative model (Qwen3.8 primary, cold-loaded on first request ~30-60s; # auxiliary models evict it on demand, and vice versa). # NOTE: Qwen3.8 does NOT have load-on-startup — it loads on first request. # This avoids an LRU eviction race with nomic-embed at startup. # UPDATED (t_72646029, 2026-08-17): CPU offload for Coder + Llama changes the # constraint. Coder and Llama now use CPU inference (n-gpu-layers=0). GPU-resident # VRAM: Qwen3.8 (~20,302 MiB at 128K ctx) + nomic-embed (558 MiB, pinned) plus the # CUDA-context buffers llama.cpp 6ea215d allocates for the CPU models (~1.4-1.7GB # each) = ~24,004 MiB steady-state, below the 24,576 MiB physical limit. # models-max raised to 4: nomic (slot 1, pinned) + Qwen3.8 (slot 2, GPU) + # Llama (slot 3, CPU) + Coder (slot 4, CPU). Phi (GPU, ~8.3GB) can still be # requested but evicts Qwen3.8 due to VRAM constraint. models-max=4 # is required so CPU-offloaded models count as loaded without evicting Qwen3.8. llm_router_models_max: 4 llm_staged_models: - filename: "Phi-3.5-mini-instruct-Q8_0.gguf" url: "https://huggingface.co/bartowski/Phi-3.5-mini-instruct-GGUF/resolve/main/Phi-3.5-mini-instruct-Q8_0.gguf" size_bytes: 4061222688 source_repo: "bartowski/Phi-3.5-mini-instruct-GGUF" - filename: "Meta-Llama-3.1-8B-Instruct-Q4_K_M.gguf" url: "https://huggingface.co/bartowski/Meta-Llama-3.1-8B-Instruct-GGUF/resolve/main/Meta-Llama-3.1-8B-Instruct-Q4_K_M.gguf" size_bytes: 4920739232 source_repo: "bartowski/Meta-Llama-3.1-8B-Instruct-GGUF" - filename: "Qwen2.5-Coder-14B-Instruct-Q4_K_M.gguf" url: "https://huggingface.co/bartowski/Qwen2.5-Coder-14B-Instruct-GGUF/resolve/main/Qwen2.5-Coder-14B-Instruct-Q4_K_M.gguf" size_bytes: 8988111072 source_repo: "bartowski/Qwen2.5-Coder-14B-Instruct-GGUF" - filename: "nomic-embed-text-v1.5-Q4_K_M.gguf" url: "https://huggingface.co/nomic-ai/nomic-embed-text-v1.5-GGUF/resolve/main/nomic-embed-text-v1.5.Q4_K_M.gguf" size_bytes: 84106624 source_repo: "nomic-ai/nomic-embed-text-v1.5-GGUF"