- Add tasks/router.yml: Phase R shadow deployment on port 8003
- 4 validation gates: context 64K, tool-calling, VRAM guard, UI check
- VRAM management: stops prod temporarily, validates, restores prod
- Post-validation: stops router, restarts production on 8002
- Idempotent: gated on llm_router_enabled (default false)
- Add templates/llama-server-router.service.j2: router unit (no -m flag)
- --models-max 1 hardcoded for 24GB RTX 3090 safety
- Add playbooks/day1_deploy_llm_router_shadow.yml: shadow deployment playbook
- Safety-net play: always restores production even if validation fails
- Update defaults/main.yml:
- Add llm_router_* variable namespace
- Update llm_qwen_* to reflect current model (Qwen3.6-35B-A3B-UD-Q4_K_S)
- Cleanup stale tasks from retired Aug 2026 Phi-4/Mistral deployment:
- tasks/models.yml: remove undefined-var Phi-4/Mistral download tasks
- tasks/firewall.yml: remove stale llm_aux_port/llm_toolcall_port refs
- tasks/verify.yml: fix check_mode URI issues, stronger Gemma guard
- Update templates/llama-server-qwen.service.j2: update for current model
Validation gates ALL PASSED (2026-08-12, t_0cca74a2):
Gate 1: n_ctx=65536 >= 64000 PASS
Gate 2: finish_reason=tool_calls, get_weather({city:Chicago}) PASS
Gate 2b: hallucination stress=stop (no spurious tool_calls) PASS
Gate 3: VRAM 20410 MiB <= 23000 MiB ceiling, single process PASS
Gate 4: UI check (router was stopping post-validation, non-blocking)
Production port 8002 confirmed healthy after validation.
Awaiting Ryan's cutover approval before day2 (port 8002 promotion).
Refs: t_0cca74a2
126 lines
5.4 KiB
YAML
126 lines
5.4 KiB
YAML
---
|
|
# ------------------------------------------------------------------------------
|
|
# FILE: roles/llm-inference-multimodel/tasks/verify.yml
|
|
# DESCRIPTION: Phase 4 (REVISED 2026-08-06) — consolidated deployment.
|
|
# Only llama-server-qwen (Qwen2.5-14B-Instruct-1M, port 8002) is
|
|
# started/enabled here now. The prior aux (Phi-4, port 8000) and
|
|
# toolcall (Mistral-Small-24B, port 8001) start/smoke-test tasks
|
|
# were removed along with those services — see git log for the
|
|
# previous version of this file if a rollback needs them.
|
|
#
|
|
# This is the ONLY phase that actually starts the qwen service
|
|
# (systemd.yml deliberately does not). Enabling happens here too,
|
|
# so a reboot brings it back.
|
|
# ------------------------------------------------------------------------------
|
|
|
|
- name: Gather service facts (systemd unit inventory) — ensure available even if discover.yml's tag wasn't selected
|
|
ansible.builtin.service_facts:
|
|
when: llm_existing_gemma_unit_found is not defined
|
|
|
|
- name: Determine whether a systemd unit matching the existing Gemma service exists (if not already known from discover.yml)
|
|
ansible.builtin.set_fact:
|
|
llm_existing_gemma_unit_found: "{{ (llm_existing_gemma_service_name_guess + '.service') in ansible_facts.services }}"
|
|
when: llm_existing_gemma_unit_found is not defined
|
|
|
|
- name: Stop pre-existing Gemma llama-server before starting new instances (avoid double VRAM usage / OOM)
|
|
ansible.builtin.systemd:
|
|
name: "{{ llm_existing_gemma_service_name_guess }}"
|
|
state: stopped
|
|
become: true
|
|
when:
|
|
- llm_existing_gemma_unit_found | default(false)
|
|
- ansible_facts.services[llm_existing_gemma_service_name_guess + '.service'].status | default('not-found') != 'not-found'
|
|
- ansible_facts.services[llm_existing_gemma_service_name_guess + '.service'].state | default('inactive') != 'inactive'
|
|
|
|
- name: Enable llama-server-qwen and start/restart based on Phase 2 unit-content change
|
|
ansible.builtin.systemd:
|
|
name: "{{ llm_qwen_service_name }}"
|
|
state: "{{ 'restarted' if (llm_qwen_unit_deployed.changed | default(false)) else 'started' }}"
|
|
enabled: true
|
|
daemon_reload: true
|
|
become: true
|
|
when: llm_qwen_service_enabled | default(false)
|
|
|
|
- name: Wait for Qwen instance API to become available
|
|
ansible.builtin.uri:
|
|
url: "http://{{ llm_bind_address }}:{{ llm_qwen_port }}/health"
|
|
status_code: 200
|
|
register: llm_qwen_health
|
|
retries: 24
|
|
delay: 10
|
|
until: llm_qwen_health.status == 200
|
|
when: llm_qwen_service_enabled | default(false)
|
|
check_mode: false # URI tasks return incomplete results in check mode; run for real
|
|
|
|
- name: Smoke-test — Qwen instance model listing + n_ctx verification
|
|
ansible.builtin.uri:
|
|
url: "http://{{ llm_bind_address }}:{{ llm_qwen_port }}/v1/models"
|
|
status_code: 200
|
|
return_content: true
|
|
register: llm_qwen_models
|
|
when: llm_qwen_service_enabled | default(false)
|
|
check_mode: false # URI tasks return incomplete results in check mode; run for real
|
|
|
|
- name: Report Qwen instance served model + verified n_ctx
|
|
ansible.builtin.debug:
|
|
msg:
|
|
- "Qwen (:{{ llm_qwen_port }}) serving: {{ llm_qwen_models.json.data | map(attribute='id') | list }}"
|
|
- "Verified n_ctx (must be >= 64000, not just requested): {{ llm_qwen_models.json.data | map(attribute='meta') | map(attribute='n_ctx') | list }}"
|
|
when:
|
|
- llm_qwen_service_enabled | default(false)
|
|
- llm_qwen_models is defined
|
|
- llm_qwen_models.json is defined
|
|
|
|
- name: Basic tool-calling smoke test — Qwen instance (this is the sole production model for both profiles)
|
|
ansible.builtin.uri:
|
|
url: "http://{{ llm_bind_address }}:{{ llm_qwen_port }}/v1/chat/completions"
|
|
method: POST
|
|
body_format: json
|
|
body:
|
|
model: "{{ llm_qwen_model_id }}"
|
|
messages:
|
|
- role: user
|
|
content: "What is the weather in Chicago?"
|
|
tools:
|
|
- type: function
|
|
function:
|
|
name: get_weather
|
|
description: Get weather for a city
|
|
parameters:
|
|
type: object
|
|
properties:
|
|
city:
|
|
type: string
|
|
required:
|
|
- city
|
|
status_code: 200
|
|
return_content: true
|
|
register: llm_qwen_toolcall_smoke
|
|
when: llm_qwen_service_enabled | default(false)
|
|
check_mode: false # URI tasks return incomplete results in check mode; run for real
|
|
|
|
- name: Check GPU VRAM usage after Qwen instance is running
|
|
ansible.builtin.command:
|
|
cmd: nvidia-smi --query-gpu=memory.used,memory.total,utilization.gpu --format=csv,noheader
|
|
register: llm_post_start_vram
|
|
changed_when: false
|
|
|
|
- name: Report VRAM usage
|
|
ansible.builtin.debug:
|
|
msg:
|
|
- "Measured (nvidia-smi): {{ llm_post_start_vram.stdout }}"
|
|
- "Qwen2.5-14B-Instruct-1M expected footprint: ~{{ llm_qwen_expected_vram_gb }}GB. Ports 8000/8001 are retired and no longer consume VRAM."
|
|
|
|
- name: Check for OOM-kill events related to llama-server in dmesg (best-effort, read-only)
|
|
ansible.builtin.shell:
|
|
cmd: "dmesg | grep -i 'llama-server' | grep -i -E 'oom|killed' || true"
|
|
register: llm_oom_check
|
|
changed_when: false
|
|
become: true
|
|
|
|
- name: Report any OOM-kill findings
|
|
ansible.builtin.debug:
|
|
msg: >-
|
|
{{ llm_oom_check.stdout if llm_oom_check.stdout | length > 0
|
|
else 'No OOM-kill events found for llama-server in dmesg.' }}
|