Files
homelab/ansible/roles/llm-inference-multimodel/tasks/discover.yml
2026-08-05 15:53:31 -05:00

88 lines
4.3 KiB
YAML

---
# ------------------------------------------------------------------------------
# FILE: roles/llm-inference-multimodel/tasks/discover.yml
# DESCRIPTION: Phase 0 — READ-ONLY fact gathering on how the existing Gemma
# llama-server is actually managed on astro-orbiter TODAY.
#
# Per plan §0/§4: "Service management: unverified — plan requires
# confirming systemd unit exists before touching anything.
# Do not assume." This file performs that confirmation. It makes
# NO changes to the host — no `state: present/started/stopped`,
# no file writes, no service actions. Every task here is either
# a `_facts` module, a `command`/`shell` in check-safe read mode,
# or a `stat`.
#
# Outcomes recorded as facts for later phases/for a human to read
# in the play recap — this file does not branch role behavior
# on the result (that would be over-engineering a role meant to
# run once); it surfaces what's true so a human confirms before
# Phase 2 proceeds.
# ------------------------------------------------------------------------------
- name: Gather service facts (systemd unit inventory)
ansible.builtin.service_facts:
- name: Determine whether a systemd unit matching the existing Gemma service exists
ansible.builtin.set_fact:
llm_existing_gemma_unit_found: "{{ (llm_existing_gemma_service_name_guess + '.service') in ansible_facts.services }}"
- name: Report existing Gemma systemd unit state (if found)
ansible.builtin.debug:
msg: >-
Existing unit '{{ llm_existing_gemma_service_name_guess }}.service' found:
state={{ ansible_facts.services[llm_existing_gemma_service_name_guess + '.service'].state | default('unknown') }},
status={{ ansible_facts.services[llm_existing_gemma_service_name_guess + '.service'].status | default('unknown') }}
when: llm_existing_gemma_unit_found
- name: WARNING — no systemd unit found matching the existing Gemma service
ansible.builtin.debug:
msg: >-
No systemd unit named '{{ llm_existing_gemma_service_name_guess }}.service'
was found via service_facts. This means the current single-model
llama-server is likely run some other way (manual nohup, screen/tmux,
or a differently-named unit). DO NOT PROCEED to Phase 2 assuming a
clean rollback target exists. Before continuing: (1) check for any
running llama-server process via `ansible -m command -a "pgrep -fa
llama-server"`, (2) if found running ad hoc, codify it as a proper
systemd unit FIRST (reusing roles/llm-inference's existing
llama-server.service.j2 pattern) so plan §6's rollback story
("systemctl start llama-server-gemma to fully revert") is real and
not aspirational. This is a human decision point, not something this
role auto-remediates.
when: not llm_existing_gemma_unit_found
- name: Check for any running llama-server process (read-only, no state change)
ansible.builtin.command:
cmd: pgrep -fa llama-server
register: llm_existing_process_check
changed_when: false
failed_when: false # pgrep exits 1 with no matches — not a failure condition here
- name: Report any llama-server process found running outside systemd
ansible.builtin.debug:
msg: "Running llama-server process(es): {{ llm_existing_process_check.stdout_lines }}"
when: llm_existing_process_check.rc == 0
- name: Check current GPU VRAM utilization (baseline, before any changes)
ansible.builtin.command:
cmd: nvidia-smi --query-gpu=memory.used,memory.total --format=csv,noheader
register: llm_baseline_vram
changed_when: false
failed_when: false
- name: Report baseline VRAM usage
ansible.builtin.debug:
msg: "Baseline GPU VRAM (before this role's changes): {{ llm_baseline_vram.stdout | default('nvidia-smi unavailable') }}"
- name: Check whether ports 8000/8001 are already bound (avoid port collision surprises)
ansible.builtin.command:
cmd: "ss -ltnp"
register: llm_existing_listeners
changed_when: false
failed_when: false
- name: Report current listeners on 8000/8001
ansible.builtin.debug:
msg: "{{ llm_existing_listeners.stdout_lines | select('search', ':(8000|8001)\\s') | list }}"
when: llm_existing_listeners.rc == 0