Files
homelab/ansible/roles/llm-inference-multimodel/tasks/verify.yml
Hermes Agent service account d1f97ad5ac Phase 2 revised: consolidate astro-orbiter to single Qwen2.5-14B-1M model (port 8002)
- Retire llama-server-aux (Phi-4, 8000) and llama-server-toolcall (Mistral-Small-24B, 8001): stopped, disabled, unit files removed from host and Ansible role
- Promote llama-server-qwen (Qwen2.5-14B-Instruct-1M, port 8002) to sole production model, serving both friday and war-machine Hermes profiles
- Verified live: n_ctx=65536/n_ctx_train=1010000, and tool_calls response via /v1/chat/completions probe (no hallucination)
- Deleted superseded GGUF weights (phi-4, mistral-small, orphaned base-Qwen, gemma-2-27b) from astro-orbiter, ~45GB reclaimed
- Updated friday and war-machine Hermes profile configs (model + compression + skills_hub aux) to point at 10.1.71.130:8002
- Ryan explicitly accepted single-model tradeoffs for both profiles
2026-08-06 11:42:34 -05:00

117 lines
4.8 KiB
YAML

---
# ------------------------------------------------------------------------------
# FILE: roles/llm-inference-multimodel/tasks/verify.yml
# DESCRIPTION: Phase 4 (REVISED 2026-08-06) — consolidated deployment.
# Only llama-server-qwen (Qwen2.5-14B-Instruct-1M, port 8002) is
# started/enabled here now. The prior aux (Phi-4, port 8000) and
# toolcall (Mistral-Small-24B, port 8001) start/smoke-test tasks
# were removed along with those services — see git log for the
# previous version of this file if a rollback needs them.
#
# This is the ONLY phase that actually starts the qwen service
# (systemd.yml deliberately does not). Enabling happens here too,
# so a reboot brings it back.
# ------------------------------------------------------------------------------
- name: Gather service facts (systemd unit inventory) — ensure available even if discover.yml's tag wasn't selected
ansible.builtin.service_facts:
when: llm_existing_gemma_unit_found is not defined
- name: Determine whether a systemd unit matching the existing Gemma service exists (if not already known from discover.yml)
ansible.builtin.set_fact:
llm_existing_gemma_unit_found: "{{ (llm_existing_gemma_service_name_guess + '.service') in ansible_facts.services }}"
when: llm_existing_gemma_unit_found is not defined
- name: Stop pre-existing Gemma llama-server before starting new instances (avoid double VRAM usage / OOM)
ansible.builtin.systemd:
name: "{{ llm_existing_gemma_service_name_guess }}"
state: stopped
become: true
when: llm_existing_gemma_unit_found | default(false)
- name: Enable llama-server-qwen and start/restart based on Phase 2 unit-content change
ansible.builtin.systemd:
name: "{{ llm_qwen_service_name }}"
state: "{{ 'restarted' if (llm_qwen_unit_deployed.changed | default(false)) else 'started' }}"
enabled: true
daemon_reload: true
become: true
when: llm_qwen_service_enabled | default(false)
- name: Wait for Qwen instance API to become available
ansible.builtin.uri:
url: "http://{{ llm_bind_address }}:{{ llm_qwen_port }}/health"
status_code: 200
register: llm_qwen_health
retries: 24
delay: 10
until: llm_qwen_health.status == 200
when: llm_qwen_service_enabled | default(false)
- name: Smoke-test — Qwen instance model listing + n_ctx verification
ansible.builtin.uri:
url: "http://{{ llm_bind_address }}:{{ llm_qwen_port }}/v1/models"
status_code: 200
return_content: true
register: llm_qwen_models
when: llm_qwen_service_enabled | default(false)
- name: Report Qwen instance served model + verified n_ctx
ansible.builtin.debug:
msg:
- "Qwen (:{{ llm_qwen_port }}) serving: {{ llm_qwen_models.json.data | map(attribute='id') | list }}"
- "Verified n_ctx (must be >= 64000, not just requested): {{ llm_qwen_models.json.data | map(attribute='meta') | map(attribute='n_ctx') | list }}"
when: llm_qwen_service_enabled | default(false)
- name: Basic tool-calling smoke test — Qwen instance (this is the sole production model for both profiles)
ansible.builtin.uri:
url: "http://{{ llm_bind_address }}:{{ llm_qwen_port }}/v1/chat/completions"
method: POST
body_format: json
body:
model: "{{ llm_qwen_model_id }}"
messages:
- role: user
content: "What is the weather in Chicago?"
tools:
- type: function
function:
name: get_weather
description: Get weather for a city
parameters:
type: object
properties:
city:
type: string
required:
- city
status_code: 200
return_content: true
register: llm_qwen_toolcall_smoke
when: llm_qwen_service_enabled | default(false)
- name: Check GPU VRAM usage after Qwen instance is running
ansible.builtin.command:
cmd: nvidia-smi --query-gpu=memory.used,memory.total,utilization.gpu --format=csv,noheader
register: llm_post_start_vram
changed_when: false
- name: Report VRAM usage
ansible.builtin.debug:
msg:
- "Measured (nvidia-smi): {{ llm_post_start_vram.stdout }}"
- "Qwen2.5-14B-Instruct-1M expected footprint: ~{{ llm_qwen_expected_vram_gb }}GB. Ports 8000/8001 are retired and no longer consume VRAM."
- name: Check for OOM-kill events related to llama-server in dmesg (best-effort, read-only)
ansible.builtin.shell:
cmd: "dmesg | grep -i 'llama-server' | grep -i -E 'oom|killed' || true"
register: llm_oom_check
changed_when: false
become: true
- name: Report any OOM-kill findings
ansible.builtin.debug:
msg: >-
{{ llm_oom_check.stdout if llm_oom_check.stdout | length > 0
else 'No OOM-kill events found for llama-server in dmesg.' }}