|
|
|
@@ -34,26 +34,27 @@ common_root_lv: ubuntu-lv
|
|
|
|
# (t_33acbb2e) so the router can keep more than one GGUF resident on-demand
|
|
|
|
# (t_33acbb2e) so the router can keep more than one GGUF resident on-demand
|
|
|
|
# and LRU-evict when needed.
|
|
|
|
# and LRU-evict when needed.
|
|
|
|
#
|
|
|
|
#
|
|
|
|
# VRAM NOTE (t_33acbb2e, updated t_55c164f5, updated t_34b96e83, updated t_f5f7e9ad, updated t_441470b9):
|
|
|
|
# VRAM NOTE (t_33acbb2e, updated t_55c164f5, updated t_34b96e83, updated t_f5f7e9ad, updated t_441470b9, updated t_c5cef2b2):
|
|
|
|
# With models-max=4 and all 5 GGUFs registered, worst case is all 5 loaded simultaneously:
|
|
|
|
# With models-max=4 and all 6 GGUFs registered, worst case is all 6 loaded simultaneously:
|
|
|
|
# Qwen3.8-27B Q4_K_M: ~23.3GB (weights ~17.1GB + KV ~6.2GB @ 128K ctx, q4_0) ← UPDATED
|
|
|
|
# Qwen3.8-27B Q4_K_M: ~20.0GB (weights ~17.1GB + KV ~2.9GB @ 65536 ctx, q4_0) ← CORRECTED (ctx rolled back from 128K to 65536, t_c9fed26c 2026-08-18)
|
|
|
|
# Phi-3.5-mini-instruct Q8_0: ~4.3GB (weights ~3.8GB + KV ~0.5GB @ 32K ctx)
|
|
|
|
# Phi-3.5-mini-instruct Q8_0: ~4.3GB (weights ~3.8GB + KV ~0.5GB @ 32K ctx)
|
|
|
|
# Meta-Llama-3.1-8B Q4_K_M: ~5.6GB (weights ~4.6GB + KV ~0.2GB @ 8K ctx)
|
|
|
|
# Meta-Llama-3.1-8B Q4_K_M: ~5.6GB (weights ~4.6GB + KV ~0.2GB @ 8K ctx)
|
|
|
|
# Qwen2.5-Coder-14B Q4_K_M: ~9.0GB (weights ~8.4GB + KV ~0.6GB @ 16K ctx)
|
|
|
|
# Qwen2.5-Coder-14B Q4_K_M: ~9.0GB (weights ~8.4GB + KV ~0.6GB @ 16K ctx)
|
|
|
|
# nomic-embed-text-v1.5 Q4_K_M: ~0.09GB (~84MB, embedding only — no KV cache)
|
|
|
|
# nomic-embed-text-v1.5 Q4_K_M: ~0.09GB (~84MB, embedding only — no KV cache)
|
|
|
|
# Total worst-case: ~42.3GB >> 24GB RTX 3090
|
|
|
|
# Qwen3-8B Q4_K_M: ~5.5GB (weights ~4.68GB + KV ~0.5GB @ 32K ctx, q4_0)
|
|
|
|
|
|
|
|
# Total worst-case: ~44.5GB >> 24GB RTX 3090
|
|
|
|
#
|
|
|
|
#
|
|
|
|
# OOM RISK: Full co-residency is impossible on 24GB. LRU eviction prevents this
|
|
|
|
# OOM RISK: Full co-residency is impossible on 24GB. LRU eviction prevents this
|
|
|
|
# in practice: models-max=4 means the router can REGISTER 5 models but only keeps
|
|
|
|
# in practice: models-max=4 means the router can REGISTER 6 models but only keeps
|
|
|
|
# up to 4 LOADED simultaneously — the router will evict the LRU model when a new
|
|
|
|
# up to 4 LOADED simultaneously — the router will evict the LRU model when a new
|
|
|
|
# one is needed. nomic-embed-text-v1.5 is pinned via sleep-idle-seconds=-1 and
|
|
|
|
# one is needed. nomic-embed-text-v1.5 is pinned via sleep-idle-seconds=-1 and
|
|
|
|
# load-on-startup=true but it uses only ~84MB, so it never meaningfully changes
|
|
|
|
# load-on-startup=true but it uses only ~84MB, so it never meaningfully changes
|
|
|
|
# the budget. In single-user homelab operation, only one generative model is active
|
|
|
|
# the budget. In single-user homelab operation, only one generative model is active
|
|
|
|
# at a time alongside the always-resident embedding model.
|
|
|
|
# at a time alongside the always-resident embedding model.
|
|
|
|
# Qwen3.8-27B alone uses ~23.1GB (weights+KV); co-residency with Coder (~9GB) = ~32GB > 24GB.
|
|
|
|
# Qwen3.8-27B alone uses ~17,804 MiB (weights+KV @ 65536 ctx); co-residency
|
|
|
|
# LRU eviction handles this automatically — the router evicts the idle model before
|
|
|
|
# with Coder (~9GB) = ~27GB > 24GB. LRU eviction handles this automatically.
|
|
|
|
# loading the new one. Ryan should be aware this means model-switching always incurs
|
|
|
|
# Ryan should be aware this means model-switching always incurs a ~30-60s
|
|
|
|
# a ~30-60s cold-load latency when switching between Qwen3.8-27B and any other model.
|
|
|
|
# cold-load latency when switching between Qwen3.8-27B and any other model.
|
|
|
|
# Proceeding to models-max=4 as instructed; flagged for Ryan's attention.
|
|
|
|
# Proceeding to models-max=4 as instructed; flagged for Ryan's attention.
|
|
|
|
# Router --models-max override for astro-orbiter.
|
|
|
|
# Router --models-max override for astro-orbiter.
|
|
|
|
# UPDATED (t_f5f7e9ad, 2026-08-16): Set to 2 because Qwen3.8-27B-Q4_K_M
|
|
|
|
# UPDATED (t_f5f7e9ad, 2026-08-16): Set to 2 because Qwen3.8-27B-Q4_K_M
|
|
|
|
@@ -67,13 +68,16 @@ common_root_lv: ubuntu-lv
|
|
|
|
# This avoids an LRU eviction race with nomic-embed at startup.
|
|
|
|
# This avoids an LRU eviction race with nomic-embed at startup.
|
|
|
|
# UPDATED (t_72646029, 2026-08-17): CPU offload for Coder + Llama changes the
|
|
|
|
# UPDATED (t_72646029, 2026-08-17): CPU offload for Coder + Llama changes the
|
|
|
|
# constraint. Coder and Llama now use CPU inference (n-gpu-layers=0). GPU-resident
|
|
|
|
# constraint. Coder and Llama now use CPU inference (n-gpu-layers=0). GPU-resident
|
|
|
|
# VRAM: Qwen3.8 (~20,302 MiB at 128K ctx) + nomic-embed (558 MiB, pinned) plus the
|
|
|
|
# VRAM: Qwen3.8 (~17,804 MiB at 65536 ctx) + nomic-embed (558 MiB, pinned) plus
|
|
|
|
# CUDA-context buffers llama.cpp 6ea215d allocates for the CPU models (~1.4-1.7GB
|
|
|
|
# the CUDA-context buffers llama.cpp 6ea215d allocates for the CPU models (~1.4-1.7GB
|
|
|
|
# each) = ~24,004 MiB steady-state, below the 24,576 MiB physical limit.
|
|
|
|
# each) = ~20,004 MiB steady-state, below the 24,576 MiB physical limit.
|
|
|
|
|
|
|
|
# CORRECTED (t_c5cef2b2, 2026-08-19): ctx-size was rolled back from 131072 to 65536
|
|
|
|
|
|
|
|
# (t_c9fed26c 2026-08-18). Qwen3.8 VRAM at 65536: 17,804 MiB (not 20,302 MiB).
|
|
|
|
# models-max raised to 4: nomic (slot 1, pinned) + Qwen3.8 (slot 2, GPU) +
|
|
|
|
# models-max raised to 4: nomic (slot 1, pinned) + Qwen3.8 (slot 2, GPU) +
|
|
|
|
# Llama (slot 3, CPU) + Coder (slot 4, CPU). Phi (GPU, ~8.3GB) can still be
|
|
|
|
# Llama (slot 3, CPU) + Coder (slot 4, CPU). Phi (GPU, ~8.3GB) and new
|
|
|
|
# requested but evicts Qwen3.8 due to VRAM constraint. models-max=4
|
|
|
|
# Qwen3-8B (GPU, ~5.5GB) can also be requested but evict Qwen3.8 due to VRAM.
|
|
|
|
# is required so CPU-offloaded models count as loaded without evicting Qwen3.8.
|
|
|
|
# models-max=4 is required so CPU-offloaded models count as loaded without
|
|
|
|
|
|
|
|
# evicting Qwen3.8.
|
|
|
|
llm_router_models_max: 4
|
|
|
|
llm_router_models_max: 4
|
|
|
|
|
|
|
|
|
|
|
|
llm_staged_models:
|
|
|
|
llm_staged_models:
|
|
|
|
@@ -93,4 +97,15 @@ llm_staged_models:
|
|
|
|
url: "https://huggingface.co/nomic-ai/nomic-embed-text-v1.5-GGUF/resolve/main/nomic-embed-text-v1.5.Q4_K_M.gguf"
|
|
|
|
url: "https://huggingface.co/nomic-ai/nomic-embed-text-v1.5-GGUF/resolve/main/nomic-embed-text-v1.5.Q4_K_M.gguf"
|
|
|
|
size_bytes: 84106624
|
|
|
|
size_bytes: 84106624
|
|
|
|
source_repo: "nomic-ai/nomic-embed-text-v1.5-GGUF"
|
|
|
|
source_repo: "nomic-ai/nomic-embed-text-v1.5-GGUF"
|
|
|
|
|
|
|
|
# Added t_c5cef2b2 (2026-08-19, War Machine): Qwen3-8B dense 8B model for
|
|
|
|
|
|
|
|
# aux tasks (routing, rewriting, structured extraction, tool-call construction).
|
|
|
|
|
|
|
|
# Source: bartowski/Qwen_Qwen3-8B-GGUF (public, no auth). HF filename is
|
|
|
|
|
|
|
|
# Qwen_Qwen3-8B-Q4_K_M.gguf; stored locally as Qwen3-8B-Q4_K_M.gguf.
|
|
|
|
|
|
|
|
# Exact size verified from HF manifest (content-length): 5,027,784,224 bytes.
|
|
|
|
|
|
|
|
# VRAM: ~4.68GB weights + ~0.5GB KV @ 32K ctx (q4_0) ≈ 5.2GB total.
|
|
|
|
|
|
|
|
# Thinking mode ON by default; use /no_think for latency-sensitive aux tasks.
|
|
|
|
|
|
|
|
- filename: "Qwen3-8B-Q4_K_M.gguf"
|
|
|
|
|
|
|
|
url: "https://huggingface.co/bartowski/Qwen_Qwen3-8B-GGUF/resolve/main/Qwen_Qwen3-8B-Q4_K_M.gguf"
|
|
|
|
|
|
|
|
size_bytes: 5027784224
|
|
|
|
|
|
|
|
source_repo: "bartowski/Qwen_Qwen3-8B-GGUF"
|
|
|
|
|
|
|
|
|
|
|
|
|