--- # ------------------------------------------------------------------------------ # FILE: roles/llm-inference-multimodel/tasks/verify.yml # DESCRIPTION: Phase 4 — start both services, curl smoke test each endpoint, # nvidia-smi VRAM check against plan §1 math, confirm no OOM. # # This is the ONLY phase that actually starts the services # (systemd.yml deliberately does not). Enabling happens here too, # so a reboot brings both back — matching plan §2's "independent # systemd services" intent for durability, not just this-session. # ------------------------------------------------------------------------------ - name: Enable llama-server-aux and start/restart based on Phase 2 unit-content change ansible.builtin.systemd: name: "{{ llm_aux_service_name }}" state: "{{ 'restarted' if (llm_aux_unit_deployed.changed | default(false)) else 'started' }}" enabled: true daemon_reload: true become: true - name: Enable llama-server-toolcall and start/restart based on Phase 2 unit-content change ansible.builtin.systemd: name: "{{ llm_toolcall_service_name }}" state: "{{ 'restarted' if (llm_toolcall_unit_deployed.changed | default(false)) else 'started' }}" enabled: true daemon_reload: true become: true - name: Wait for aux instance API to become available (model load may take a couple minutes) ansible.builtin.uri: url: "http://{{ llm_bind_address }}:{{ llm_aux_port }}/health" status_code: 200 register: llm_aux_health retries: 24 delay: 10 until: llm_aux_health.status == 200 - name: Wait for tool-calling instance API to become available ansible.builtin.uri: url: "http://{{ llm_bind_address }}:{{ llm_toolcall_port }}/health" status_code: 200 register: llm_toolcall_health retries: 24 delay: 10 until: llm_toolcall_health.status == 200 - name: Smoke-test — aux instance model listing ansible.builtin.uri: url: "http://{{ llm_bind_address }}:{{ llm_aux_port }}/v1/models" status_code: 200 return_content: true register: llm_aux_models - name: Smoke-test — tool-calling instance model listing ansible.builtin.uri: url: "http://{{ llm_bind_address }}:{{ llm_toolcall_port }}/v1/models" status_code: 200 return_content: true register: llm_toolcall_models - name: Report served models per instance ansible.builtin.debug: msg: - "Aux (:{{ llm_aux_port }}) serving: {{ llm_aux_models.json.data | map(attribute='id') | list }}" - "Tool-calling (:{{ llm_toolcall_port }}) serving: {{ llm_toolcall_models.json.data | map(attribute='id') | list }}" - name: Basic completion smoke test — aux instance (non-tool-calling sanity check only) ansible.builtin.uri: url: "http://{{ llm_bind_address }}:{{ llm_aux_port }}/v1/chat/completions" method: POST body_format: json body: model: "{{ llm_aux_model_id }}" messages: - role: user content: "Reply with exactly one word: OK" max_tokens: 10 status_code: 200 return_content: true register: llm_aux_completion - name: Basic completion smoke test — tool-calling instance (plain-text sanity check only) ansible.builtin.uri: url: "http://{{ llm_bind_address }}:{{ llm_toolcall_port }}/v1/chat/completions" method: POST body_format: json body: model: "{{ llm_toolcall_model_id }}" messages: - role: user content: "Reply with exactly one word: OK" max_tokens: 10 status_code: 200 return_content: true register: llm_toolcall_completion - name: NOTE — this smoke test is NOT the tool-calling validation harness ansible.builtin.debug: msg: >- Both endpoints respond to basic completions. This does NOT validate tool_calls correctness or hallucination-safety for the tool-calling instance — that is a separate, manual, post-deploy procedure (plan §7). See references/tool-calling-validation.sh (copied from the llm-inference-homelab skill) and docs/validation-log.md in this role. DO NOT point any Claude Code / tool-calling-capable Hermes profile at port {{ llm_toolcall_port }} until that validation has passed and been logged. - name: Check GPU VRAM usage after both instances are running ansible.builtin.command: cmd: nvidia-smi --query-gpu=memory.used,memory.total,utilization.gpu --format=csv,noheader register: llm_post_start_vram changed_when: false - name: Report VRAM usage vs plan §1 expectations ansible.builtin.debug: msg: - "Measured (nvidia-smi): {{ llm_post_start_vram.stdout }}" - "Design estimate (plan §1): aux ~{{ llm_aux_expected_vram_gb }}GB + toolcall ~{{ llm_toolcall_expected_vram_gb }}GB = ~{{ llm_combined_expected_vram_gb }}GB / {{ llm_gpu_total_vram_gb }}GB total" - "If measured usage exceeds ~23.5GB or is within ~0.5GB of the 24GB card limit, treat as the OOM-risk trigger condition from plan §6 — do not leave both services running unattended without confirming headroom." - name: Check for OOM-kill events related to llama-server in dmesg (best-effort, read-only) ansible.builtin.shell: cmd: "dmesg | grep -i 'llama-server' | grep -i -E 'oom|killed' || true" register: llm_oom_check changed_when: false become: true - name: Report any OOM-kill findings ansible.builtin.debug: msg: >- {{ llm_oom_check.stdout if llm_oom_check.stdout | length > 0 else 'No OOM-kill events found for llama-server in dmesg.' }}