Compare commits
328 Commits
feature/as
...
3dc58cf644
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3dc58cf644 | ||
|
|
d95477fc3b | ||
| 274ce1fd8a | |||
|
|
eed2fcb7c7 | ||
|
|
266b6c7be1 | ||
| e9924a2524 | |||
|
|
5ee8309d32 | ||
|
|
261f6be7db | ||
|
|
8ea19dbf70 | ||
|
|
e6cb187f8e | ||
|
|
56f19af578 | ||
|
|
a3c92f70bf | ||
|
|
f907acde95 | ||
|
|
53a55e7317 | ||
|
|
2c0db1c7a1 | ||
|
|
39c5fdca69 | ||
|
|
6bfcc76845 | ||
|
|
1af645d272 | ||
|
|
f3a5687adf | ||
|
|
9d6869ad9d | ||
|
|
2cc9370f3d | ||
|
|
60220e18b6 | ||
|
|
b3b925ff77 | ||
|
|
3d8eb1bf1c | ||
|
|
7f8ba8b859 | ||
|
|
aee61d4511 | ||
|
|
9bc29508d7 | ||
|
|
9bfc9384e4 | ||
|
|
152230c100 | ||
|
|
13df80ab43 | ||
|
|
a2123819b3 | ||
|
|
173d00504c | ||
|
|
7cdcc984a5 | ||
|
|
e301770adc | ||
|
|
ab1e32711d | ||
|
|
5c0df8c73c | ||
|
|
5cf4468754 | ||
|
|
bafd76a0b4 | ||
|
|
24735f7e5c | ||
|
|
7867be688a | ||
|
|
03b3ce9dee | ||
|
|
a2994bf55d | ||
|
|
7b44a41da3 | ||
|
|
efaff340a4 | ||
|
|
48536f2615 | ||
|
|
170a31d090 | ||
|
|
aa2730efd5 | ||
|
|
0dbb77b023 | ||
|
|
fee9965d0a | ||
|
|
d0f3ddba0d | ||
|
|
d9e41118f8 | ||
|
|
ad70b3439c | ||
|
|
a04435ee9b | ||
|
|
a2ddb65425 | ||
|
|
a87da82ebd | ||
|
|
7aea88724f | ||
|
|
6455d22752 | ||
|
|
a47b29d49f | ||
|
|
9c969f783d | ||
|
|
081156ecab | ||
|
|
3783ded62a | ||
|
|
5a2246a540 | ||
|
|
ba311a3ec6 | ||
|
|
d1f97ad5ac | ||
|
|
b4bdb63e4a | ||
|
|
b741f9b20b | ||
|
|
a3c1342837 | ||
|
|
d4ff2681ac | ||
|
|
75cb93f25c | ||
|
|
d10255297c | ||
|
|
79edb8f4e1 | ||
|
|
5dc76a8348 | ||
|
|
a76ad3195c | ||
|
|
73ef806dd6 | ||
|
|
628dae06a8 | ||
|
|
c3755aa29e | ||
|
|
782cbe33d1 | ||
|
|
aff792a061 | ||
|
|
aa8e229e64 | ||
|
|
22a020e4c7 | ||
|
|
e879cf73d3 | ||
|
|
423891001c | ||
|
|
dda6b91330 | ||
|
|
265d3f8fd6 | ||
|
|
b61d19cb91 | ||
|
|
00be18b1f1 | ||
|
|
6c7ec507ef | ||
|
|
63b0bc72fe | ||
|
|
02af5d26dc | ||
|
|
3eb38b74bd | ||
|
|
2b95acb8cc | ||
|
|
62e9f13a45 | ||
|
|
4cb87a57ad | ||
|
|
d2eaddfd11 | ||
|
|
0e741aab38 | ||
|
|
9ebd19ab52 | ||
|
|
e47cbf2044 | ||
|
|
ce632e88b9 | ||
|
|
11d8796764 | ||
|
|
d974c75d7c | ||
|
|
a5433dcb5b | ||
|
|
e0eb47f5ce | ||
|
|
a8822f0778 | ||
| bedf87b492 | |||
|
|
b6f7791c98 | ||
|
|
1a48e60afd | ||
|
|
c26b19793b | ||
|
|
dfe81a5c20 | ||
|
|
4afb05e56b | ||
|
|
46b49259d2 | ||
|
|
3f3ce68e18 | ||
|
|
e44805c9b6 | ||
|
|
1aa6b4234a | ||
|
|
4cde540e70 | ||
|
|
69fb5f5641 | ||
|
|
430552a0b1 | ||
|
|
18bb111843 | ||
|
|
712425ee17 | ||
|
|
1d77821e5f | ||
|
|
7ad40bb509 | ||
|
|
3950a2b069 | ||
|
|
d20fd80798 | ||
|
|
308ee553c3 | ||
|
|
56110d52bd | ||
|
|
1f07fdff45 | ||
|
|
317816558d | ||
|
|
ea22e4e407 | ||
|
|
490c483924 | ||
|
|
bc34a1f915 | ||
|
|
64e690737e | ||
|
|
0693fdcd26 | ||
| 19a807899f | |||
| 5bacf9fbca | |||
| 1da9bfd43c | |||
| 0bc9b2e788 | |||
| dcfb6825e8 | |||
| 0b9ac4dc74 | |||
|
|
aabf758c91 | ||
|
|
489b8aeb35 | ||
|
|
002d6799b1 | ||
|
|
150cef1aca | ||
|
|
a30ad99ee4 | ||
|
|
d2b6d95a49 | ||
|
|
adc415e95a | ||
|
|
e8303d5129 | ||
|
|
f37021346b | ||
|
|
5a99928c6f | ||
|
|
59c83c296b | ||
|
|
b6cb031edb | ||
|
|
cb5ffc16d8 | ||
|
|
f375c9567f | ||
|
|
f0400c02b6 | ||
|
|
d1d7331238 | ||
|
|
459dbc5d18 | ||
| a4a68eeb5a | |||
|
|
99958979d6 | ||
|
|
0e4d229df2 | ||
|
|
53883108d8 | ||
|
|
fcbf6ce092 | ||
|
|
cf21863b0f | ||
|
|
653a923fa8 | ||
|
|
13c819e66c | ||
|
|
28a653b203 | ||
|
|
cf7ab7fbe6 | ||
|
|
12d0a75b3a | ||
|
|
668e86d7c2 | ||
| 0fb593a313 | |||
|
|
b5dc130207 | ||
|
|
0efa1e5125 | ||
|
|
34e05725dd | ||
|
|
b6b1bee25c | ||
|
|
3b3461fd3c | ||
|
|
7cbed63c92 | ||
|
|
a008e766f2 | ||
|
|
543394a825 | ||
|
|
23d6d75117 | ||
|
|
86433a58d0 | ||
|
|
3344e24a48 | ||
|
|
2621bc9f36 | ||
|
|
72de87c8c4 | ||
| 0ef9703757 | |||
|
|
d77d213d89 | ||
|
|
e793794fdd | ||
|
|
d1ae5ba7a0 | ||
|
|
84e30c8ee2 | ||
|
|
644128cd3f | ||
|
|
6912f5c55d | ||
|
|
fc0e39b9c7 | ||
|
|
8627b00ed8 | ||
|
|
0212f0fdd2 | ||
|
|
edfe594e7e | ||
|
|
791f13fca5 | ||
|
|
c3248fde1f | ||
|
|
c137ea0881 | ||
|
|
818b6505dd | ||
|
|
4a1958876f | ||
|
|
6bdb536848 | ||
|
|
6023ee25e1 | ||
|
|
e5d24f557a | ||
|
|
f8e137b67b | ||
|
|
76241e75a8 | ||
|
|
d5ce6ff96a | ||
|
|
44b1a2fb33 | ||
|
|
7dc1999928 | ||
|
|
63d480927d | ||
|
|
fa86fa4c9c | ||
|
|
444b597ade | ||
|
|
c81a9b7704 | ||
|
|
6cab6519b1 | ||
|
|
ce992ca743 | ||
|
|
092d1ac209 | ||
|
|
6acf2f9944 | ||
|
|
8d18f42b2e | ||
|
|
152f10ed8b | ||
|
|
d99ebca829 | ||
| df91305e13 | |||
|
|
6f2b6e0290 | ||
|
|
fc34833472 | ||
|
|
7f37211a8b | ||
|
|
ccc956f70b | ||
|
|
511f32e521 | ||
|
|
87a3e84f5b | ||
|
|
1743145e9f | ||
|
|
d561ac6e04 | ||
|
|
99bc31dee9 | ||
|
|
ac8e7acbd4 | ||
|
|
0ad5dbe741 | ||
|
|
7d9b054340 | ||
|
|
4b1e8a7cac | ||
|
|
8f190eb188 | ||
|
|
95ae6919b0 | ||
|
|
52e97f3a7c | ||
|
|
461aa1bc54 | ||
|
|
c38461a6e8 | ||
|
|
6bcb6fa93f | ||
|
|
f4181349f8 | ||
|
|
9d860367cc | ||
|
|
f654c59dd5 | ||
|
|
d22ac5cef2 | ||
|
|
0fd69e0b90 | ||
|
|
e8d87ff092 | ||
|
|
23612a38f2 | ||
|
|
bd100c15e7 | ||
|
|
dc5392446c | ||
|
|
f8cf139b10 | ||
|
|
ac955d327f | ||
|
|
9e6339037a | ||
|
|
b647f6afee | ||
|
|
aabb3d5009 | ||
|
|
4ed64ab91c | ||
|
|
f57e0bef02 | ||
|
|
9ed7466fd8 | ||
|
|
4d7766d1b1 | ||
|
|
09ae954085 | ||
|
|
8fd9fd5b20 | ||
|
|
dcb764eca7 | ||
|
|
b6a4ad6816 | ||
|
|
bbaaf655fa | ||
|
|
7228dc6e11 | ||
|
|
8953702608 | ||
|
|
0be33cb8db | ||
|
|
0f0b5db29b | ||
|
|
009f244739 | ||
|
|
2f87039f17 | ||
|
|
72fa38e928 | ||
|
|
9e68802090 | ||
|
|
d05cfcf317 | ||
|
|
80f810fb0c | ||
|
|
9153324795 | ||
|
|
91b5817e5f | ||
|
|
1dfa7889ab | ||
|
|
08d7da0c35 | ||
|
|
9ebeb42023 | ||
|
|
075f34b1fb | ||
|
|
210c89c2c7 | ||
|
|
b93a6e50ab | ||
|
|
85cc1f8c6a | ||
|
|
8e781b0c54 | ||
|
|
d8ad35b8e9 | ||
|
|
a9973d1e0f | ||
| 4113011f63 | |||
|
|
018782d986 | ||
|
|
3681e8c03e | ||
|
|
8f377e4cf3 | ||
|
|
419acaa40d | ||
|
|
42b204bf8a | ||
|
|
1d7dcb7d82 | ||
|
|
d63ca0b4f9 | ||
|
|
e9440327aa | ||
|
|
dcc7e282c7 | ||
|
|
6d5fc7c5c6 | ||
| 312fdf9986 | |||
|
|
4e0b4fa049 | ||
|
|
cf7c2a1436 | ||
|
|
3dc6555ad1 | ||
|
|
27ff9286b4 | ||
|
|
6e50461999 | ||
|
|
3f1c3a40cf | ||
|
|
0116ec4cc3 | ||
|
|
f962d0a6d7 | ||
|
|
dc3c0d7cb1 | ||
|
|
b05f9fad09 | ||
|
|
8650995926 | ||
|
|
e5469d2cb8 | ||
|
|
702698ddcd | ||
|
|
d233d582d4 | ||
|
|
3839fac162 | ||
|
|
b01dac85da | ||
|
|
37a49824d0 | ||
|
|
0127016ab2 | ||
|
|
e0b6fcb24a | ||
|
|
e7d9a8fec5 | ||
|
|
1a9addc537 | ||
|
|
401f25b1c4 | ||
|
|
ece522074e | ||
|
|
c30c0074f1 | ||
|
|
fa51dc2c4d | ||
|
|
c1810fde8a | ||
| a781ef8b14 | |||
|
|
92b2a9d609 | ||
|
|
9250b0f193 | ||
| 24869f47ee | |||
| bb5a57e909 | |||
| 9f3d81729d | |||
| 064d3e8b3d | |||
| ef7e3c61ed | |||
| 64951e1e5e | |||
| 58931732f7 |
160
COUCHDB-ERLANGCOOKIE-FIX.md
Normal file
160
COUCHDB-ERLANGCOOKIE-FIX.md
Normal file
@@ -0,0 +1,160 @@
|
||||
# CouchDB erlangCookie Fix - Implementation Guide
|
||||
|
||||
## Summary
|
||||
|
||||
**Problem**: CouchDB deployment fails because `erlangCookie` is missing from the ExternalSecret configuration.
|
||||
|
||||
**Decision**: Externalize `erlangCookie` to 1Password (pragmatic approach)
|
||||
|
||||
**Rationale**:
|
||||
- ExternalSecret architecture requires ownership of the entire secret
|
||||
- Mixing externalized and chart-generated fields in the same secret is not supported
|
||||
- Single-node deployment makes erlangCookie rotation unnecessary
|
||||
- This is an acceptable deviation from the pure Harbor pattern given the architectural constraints
|
||||
|
||||
## Implementation Steps
|
||||
|
||||
### 1. Generate erlangCookie Value
|
||||
|
||||
```bash
|
||||
openssl rand -hex 20
|
||||
```
|
||||
|
||||
Example output: `f4e3c2b1a9d8e7f6c5b4a3d2e1f0a9b8c7d6e5f4`
|
||||
|
||||
### 2. Add to 1Password
|
||||
|
||||
- **Vault**: `mk-labs`
|
||||
- **Item**: `couchdb`
|
||||
- **Field Name**: `erlang-cookie`
|
||||
- **Field Type**: password (concealed)
|
||||
- **Value**: `<paste generated value from step 1>`
|
||||
|
||||
### 3. Update ExternalSecret Configuration
|
||||
|
||||
File: `cluster/applications/couchdb/externalsecret.yaml`
|
||||
|
||||
```yaml
|
||||
apiVersion: external-secrets.io/v1beta1
|
||||
kind: ExternalSecret
|
||||
metadata:
|
||||
name: couchdb-credentials
|
||||
namespace: couchdb
|
||||
labels:
|
||||
app.kubernetes.io/name: couchdb
|
||||
app.kubernetes.io/part-of: mk-labs
|
||||
spec:
|
||||
refreshInterval: 1h
|
||||
secretStoreRef:
|
||||
kind: ClusterSecretStore
|
||||
name: onepassword-connect
|
||||
target:
|
||||
name: couchdb-admin
|
||||
creationPolicy: Owner
|
||||
template:
|
||||
engineVersion: v2
|
||||
data:
|
||||
adminUsername: "admin"
|
||||
adminPassword: "{{ .adminPassword }}"
|
||||
cookieAuthSecret: "{{ .cookieAuthSecret }}"
|
||||
erlangCookie: "{{ .erlangCookie }}" # ← ADD THIS LINE
|
||||
data:
|
||||
- secretKey: adminPassword
|
||||
remoteRef:
|
||||
key: couchdb
|
||||
property: admin-password
|
||||
- secretKey: cookieAuthSecret
|
||||
remoteRef:
|
||||
key: couchdb
|
||||
property: cookie-auth-secret
|
||||
- secretKey: erlangCookie # ← ADD THIS BLOCK
|
||||
remoteRef:
|
||||
key: couchdb
|
||||
property: erlang-cookie
|
||||
```
|
||||
|
||||
### 4. Update values.yaml Documentation (Optional)
|
||||
|
||||
File: `cluster/applications/couchdb/values.yaml`
|
||||
|
||||
Update the comment block at line 9-10:
|
||||
|
||||
```yaml
|
||||
# Admin credentials managed via ExternalSecret
|
||||
# See externalsecret.yaml for 1Password integration
|
||||
#
|
||||
# NOTE: erlangCookie is externalized to 1Password for architectural
|
||||
# simplicity (ExternalSecret ownership model). In a pure Harbor pattern,
|
||||
# this would be chart-generated, but single-node deployment makes this
|
||||
# acceptable. The erlangCookie is treated as an immutable infrastructure
|
||||
# secret (generate once, never rotate).
|
||||
createAdminSecret: false
|
||||
extraSecretName: "couchdb-admin"
|
||||
```
|
||||
|
||||
### 5. Commit and Push
|
||||
|
||||
```bash
|
||||
cd ~/git/homelab
|
||||
git add cluster/applications/couchdb/externalsecret.yaml
|
||||
git add cluster/applications/couchdb/values.yaml # if modified
|
||||
git commit -m "fix(couchdb): add erlangCookie to ExternalSecret from 1Password"
|
||||
git push origin main
|
||||
```
|
||||
|
||||
### 6. Verify Deployment
|
||||
|
||||
```bash
|
||||
# Watch ExternalSecret sync
|
||||
kubectl get externalsecret -n couchdb couchdb-credentials -w
|
||||
# Wait for: SecretSynced
|
||||
|
||||
# Verify secret created with all four keys
|
||||
kubectl get secret -n couchdb couchdb-admin -o yaml
|
||||
# Should contain: adminUsername, adminPassword, cookieAuthSecret, erlangCookie
|
||||
|
||||
# Watch ArgoCD sync
|
||||
kubectl get application -n argocd couchdb -w
|
||||
# Wait for: Healthy/Synced
|
||||
|
||||
# Watch pod startup
|
||||
kubectl get pods -n couchdb -w
|
||||
# Wait for: Running
|
||||
|
||||
# Test CouchDB access
|
||||
kubectl port-forward -n couchdb svc/couchdb-svc-couchdb 5984:5984 &
|
||||
curl http://localhost:5984/
|
||||
# Expected: {"couchdb":"Welcome","version":"3.5.1"}
|
||||
```
|
||||
|
||||
## Why Not Follow Harbor Pattern Exactly?
|
||||
|
||||
**Harbor Pattern**: Only user-facing credentials externalized, internal secrets chart-generated.
|
||||
|
||||
**CouchDB Constraint**: ExternalSecret uses `creationPolicy: Owner`, which takes full ownership of the target secret. This prevents the Helm chart from adding auto-generated fields to the same secret.
|
||||
|
||||
**Options Considered**:
|
||||
1. ✅ **Externalize erlangCookie** (SELECTED) - Works with current architecture
|
||||
2. ❌ Chart auto-generation - Conflicts with ExternalSecret ownership
|
||||
3. ❌ Dual-secret approach - Requires Helm chart customization
|
||||
4. ❌ Disable ExternalSecret - Loses 1Password integration for admin password
|
||||
|
||||
**Decision**: Pragmatic approach wins. erlangCookie is treated as an infrastructure secret (generate once, never rotate), which is acceptable for a single-node deployment.
|
||||
|
||||
## Secret Classification
|
||||
|
||||
| Secret | Type | 1Password? | Rationale |
|
||||
|------------------|---------------|------------|------------------------------------|
|
||||
| adminUsername | User-facing | No* | Static value, hardcoded in template |
|
||||
| adminPassword | User-facing | ✅ YES | User login credential |
|
||||
| cookieAuthSecret | Gray area | ✅ YES | Session security, periodic rotation |
|
||||
| erlangCookie | Internal | ✅ YES** | Architectural constraint |
|
||||
|
||||
\* Hardcoded in ExternalSecret template (not fetched from 1Password)
|
||||
\*\* Pragmatic deviation from Harbor pattern due to ExternalSecret architecture
|
||||
|
||||
## References
|
||||
|
||||
- Full analysis: `/home/hermes/couchdb-erlangcookie-analysis.txt`
|
||||
- Harbor pattern: `/home/hermes/harbor-simplification-complete.txt`
|
||||
- CouchDB Helm chart: `apache/couchdb` v4.6.3
|
||||
@@ -54,7 +54,7 @@
|
||||
|
||||
# (pathspec) Colon-separated paths in which Ansible will search for collections content. Collections must be in nested *subdirectories*, not directly in these directories. For example, if ``COLLECTIONS_PATHS`` includes ``'{{ ANSIBLE_HOME ~ "/collections" }}'``, and you want to add ``my.collection`` to that directory, it must be saved as ``'{{ ANSIBLE_HOME} ~ "/collections/ansible_collections/my/collection" }}'``.
|
||||
|
||||
;collections_path=/Users/rblundon/.ansible/collections:/usr/share/ansible/collections
|
||||
collections_path=/opt/ansible-collections:/usr/share/ansible/collections
|
||||
|
||||
# (boolean) A boolean to enable or disable scanning the sys.path for installed collections.
|
||||
;collections_scan_sys_path=True
|
||||
@@ -209,7 +209,7 @@ private_key_file=~/.ssh/ansible
|
||||
remote_user=wed
|
||||
|
||||
# (pathspec) Colon-separated paths in which Ansible will search for Roles.
|
||||
roles_path=/opt/git/homelab/ansible/playbooks/roles:/Users/rblundon/.ansible/roles:/usr/share/ansible/roles:/etc/ansible/roles
|
||||
roles_path=./roles
|
||||
|
||||
# (string) Set the main callback used to display Ansible output. You can only have one at a time.
|
||||
# You can have many other callbacks, but just one can be in charge of stdout.
|
||||
@@ -262,7 +262,7 @@ roles_path=/opt/git/homelab/ansible/playbooks/roles:/Users/rblundon/.ansible/rol
|
||||
|
||||
# (path) The vault password file to use. Equivalent to ``--vault-password-file`` or ``--vault-id``.
|
||||
# If executable, it will be run and the resulting stdout will be used as the password.
|
||||
;vault_password_file=
|
||||
vault_password_file=/home/hermes/.vault_pass.txt
|
||||
|
||||
# (integer) Sets the default verbosity, equivalent to the number of ``-v`` passed in the command line.
|
||||
;verbosity=0
|
||||
|
||||
6
ansible/create_jarvis_user.yml
Normal file
6
ansible/create_jarvis_user.yml
Normal file
@@ -0,0 +1,6 @@
|
||||
---
|
||||
- name: Create jarvis user and deploy SSH key
|
||||
hosts: all
|
||||
become: true
|
||||
roles:
|
||||
- jarvis_user
|
||||
193
ansible/group_vars/all/semaphore.yml
Normal file
193
ansible/group_vars/all/semaphore.yml
Normal file
@@ -0,0 +1,193 @@
|
||||
---
|
||||
# ============================================================================
|
||||
# Semaphore configuration-as-code
|
||||
# ============================================================================
|
||||
# Drives a freshly-deployed Semaphore instance into its desired state via
|
||||
# the Semaphore REST API. Idempotent: every object is checked first; only
|
||||
# missing ones are created. Existing objects are left alone.
|
||||
#
|
||||
# Loaded from group_vars/all/semaphore.yml so that the configuration is
|
||||
# version-controlled in the homelab repo and survives a wipe-and-redeploy
|
||||
# of the Semaphore VM.
|
||||
# ============================================================================
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# API connection (defaults to the local Traefik-fronted service-name URL).
|
||||
# Override semaphore_api_url to point at a specific instance if needed.
|
||||
# ---------------------------------------------------------------------------
|
||||
semaphore_api_url: "https://semaphore.local.mk-labs.cloud/api"
|
||||
semaphore_api_validate_certs: true
|
||||
semaphore_api_token: "{{ vault_semaphore_api_token }}"
|
||||
|
||||
# Feature flag — keeps day1_deploy_semaphore.yml deploy-only by default.
|
||||
# Set true to also run the configuration pass.
|
||||
semaphore_configure: false
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Declarative configuration of the Semaphore instance.
|
||||
# ---------------------------------------------------------------------------
|
||||
#
|
||||
# Top-level shape:
|
||||
#
|
||||
# semaphore_config:
|
||||
# project: single dict — the lab uses one project ("mk-labs")
|
||||
# keys: list of credentials Semaphore stores
|
||||
# repositories: git repos Semaphore can clone
|
||||
# inventories: Ansible inventories from those repos
|
||||
# environments: env-var bundles
|
||||
# templates: task templates that tie everything together
|
||||
#
|
||||
# Each list element has a unique "name" used as the natural identity key.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
semaphore_config:
|
||||
project:
|
||||
name: mk-labs
|
||||
alert: false
|
||||
max_parallel_tasks: 0 # 0 = unlimited
|
||||
|
||||
keys:
|
||||
# The ansible-vault password. login_password type with empty login
|
||||
# — only the password field is consumed by Semaphore at runtime.
|
||||
- name: ansible-vault-pass
|
||||
type: login_password
|
||||
login: ""
|
||||
password: "{{ vault_ansible_vault_password }}"
|
||||
|
||||
# SSH key for the gitea deploy access (clone the homelab repo).
|
||||
- name: gitea-deploy
|
||||
type: ssh
|
||||
ssh_login: git
|
||||
ssh_private_key: "{{ vault_gitea_deploy_key }}"
|
||||
|
||||
# SSH key for the universal automation account 'wed' — pre-baked in
|
||||
# every mk-labs VM template. This is the canonical user Semaphore
|
||||
# uses to reach the fleet.
|
||||
- name: wed-ssh
|
||||
type: ssh
|
||||
ssh_login: wed
|
||||
ssh_private_key: "{{ vault_wed_ssh_private_key }}"
|
||||
|
||||
# SSH key Semaphore can use to reach the fleet as jarvis (admin
|
||||
# account provisioned by linux-baseline). Retained for jobs that
|
||||
# specifically need jarvis-level access; the default is wed-ssh.
|
||||
- name: jarvis-ssh
|
||||
type: ssh
|
||||
ssh_login: jarvis
|
||||
ssh_private_key: "{{ vault_jarvis_ssh_private_key }}"
|
||||
|
||||
repositories:
|
||||
- name: homelab
|
||||
git_url: "ssh://git@gitea.mk-labs.cloud:2221/rblundon/homelab.git"
|
||||
git_branch: main
|
||||
ssh_key: gitea-deploy
|
||||
|
||||
inventories:
|
||||
- name: production
|
||||
type: file
|
||||
inventory_file: ansible/inventory.yml
|
||||
repository: homelab
|
||||
# wed is the universal automation account pre-baked in every VM
|
||||
# template. Semaphore uses it for fleet-wide jobs.
|
||||
ssh_key: wed-ssh
|
||||
# become_key is Semaphore's sudo PASSWORD slot, not a second SSH
|
||||
# key. wed has passwordless sudo on every host, so reference the
|
||||
# built-in "None" key. (Semaphore rejects an SSH-type key here.)
|
||||
become_key: None
|
||||
|
||||
environments:
|
||||
- name: default
|
||||
env:
|
||||
ANSIBLE_HOST_KEY_CHECKING: "False"
|
||||
ANSIBLE_FORCE_COLOR: "True"
|
||||
# Semaphore runs ansible-playbook from the cloned REPO ROOT (not
|
||||
# from the playbook's directory as I first assumed). Path is
|
||||
# therefore relative to repo root, not playbook dir.
|
||||
ANSIBLE_ROLES_PATH: "ansible/roles"
|
||||
# Collections are installed by the semaphore role into a host-side
|
||||
# directory bind-mounted into the container at this path.
|
||||
ANSIBLE_COLLECTIONS_PATH: "/opt/ansible-collections"
|
||||
|
||||
templates:
|
||||
- name: "day0_linux_baseline"
|
||||
description: "Apply the mk-labs Linux baseline to one or more hosts."
|
||||
app: ansible
|
||||
playbook: ansible/playbooks/day0_linux_baseline.yml
|
||||
inventory: production
|
||||
repository: homelab
|
||||
environment: default
|
||||
vault_password: ansible-vault-pass
|
||||
arguments: '["--diff"]'
|
||||
survey_vars:
|
||||
- name: target
|
||||
title: "Target host or group"
|
||||
description: "Inventory target (e.g. figment, semaphore_server, all)"
|
||||
required: true
|
||||
type: TextVar
|
||||
default_value: "all"
|
||||
|
||||
- name: "day1_deploy_semaphore"
|
||||
description: "Re-deploy Semaphore + PostgreSQL on figment."
|
||||
app: ansible
|
||||
playbook: ansible/playbooks/day1_deploy_semaphore.yml
|
||||
inventory: production
|
||||
repository: homelab
|
||||
environment: default
|
||||
vault_password: ansible-vault-pass
|
||||
arguments: '["--diff"]'
|
||||
|
||||
- name: "day0_linux_baseline_check"
|
||||
description: "Dry-run the baseline — shows diffs, applies nothing."
|
||||
app: ansible
|
||||
playbook: ansible/playbooks/day0_linux_baseline.yml
|
||||
inventory: production
|
||||
repository: homelab
|
||||
environment: default
|
||||
vault_password: ansible-vault-pass
|
||||
arguments: '["--check","--diff"]'
|
||||
survey_vars:
|
||||
- name: target
|
||||
title: "Target host or group"
|
||||
description: "Inventory target (e.g. figment, semaphore_server, all)"
|
||||
required: true
|
||||
type: TextVar
|
||||
default_value: "all"
|
||||
|
||||
- name: "llm_inference_multimodel_stage_models"
|
||||
description: >-
|
||||
Stage additional GGUF models into /opt/models on astro-orbiter via the
|
||||
llm-inference-multimodel role (--tags models only). Idempotent: skips
|
||||
files already present at the correct byte size. Notifies the
|
||||
llama-server-router restart handler ONLY when a new GGUF is actually
|
||||
downloaded. Does NOT touch Phase 4 (verify) or the legacy
|
||||
llama-server-qwen service. Safe to run repeatedly.
|
||||
app: ansible
|
||||
playbook: ansible/playbooks/day1_deploy_llm_inference_multimodel.yml
|
||||
inventory: production
|
||||
repository: homelab
|
||||
environment: default
|
||||
vault_password: ansible-vault-pass
|
||||
arguments: '["--tags","models","--diff"]'
|
||||
# Scoped to --tags models:
|
||||
# Phase 0 (discover) -- skipped (no tag)
|
||||
# Phase 1 (models) -- RUN (idempotent GGUF staging via stage_model.yml)
|
||||
# Phase 2 (systemd) -- skipped
|
||||
# Phase 3 (firewall) -- skipped
|
||||
# Phase 4 (verify) -- SKIPPED (collision risk: verify.yml would start
|
||||
# llama-server-qwen on :8002, conflicting with the
|
||||
# production llama-server-router.service. Excluded
|
||||
# here deliberately. See t_730f9584.)
|
||||
|
||||
- name: "llm_router_update_unit"
|
||||
description: >-
|
||||
Re-render and reload the llama-server-router systemd unit on astro-orbiter,
|
||||
then restart the live service so new args (e.g. --models-max) take effect.
|
||||
Drives playbooks/day2_bump_router_models_max.yml. Added 2026-08-12 (t_33acbb2e):
|
||||
bump --models-max 1 -> 4 with full VRAM budget note in host_vars.
|
||||
app: ansible
|
||||
playbook: ansible/playbooks/day2_bump_router_models_max.yml
|
||||
inventory: production
|
||||
repository: homelab
|
||||
environment: default
|
||||
vault_password: ansible-vault-pass
|
||||
arguments: '["--diff"]'
|
||||
@@ -19,6 +19,15 @@ terraform_server: "infra01"
|
||||
# Traefik variables
|
||||
traefik_server: "lightning-lane"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# JARVIS automation account
|
||||
# ---------------------------------------------------------------------------
|
||||
# Public key for the 'jarvis' user provisioned by the linux-baseline role on
|
||||
# every host. Public keys are not secret; the matching private key lives on
|
||||
# the JARVIS command centre (carousel-of-progress) and, when needed, in
|
||||
# group_vars/all/vault as vault_jarvis_ssh_private_key.
|
||||
jarvis_ssh_public_key: "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAID5sym5ajFvDyzw395BkHv7qVb66XPTx/OF1p19MGuNo jarvis@mk-labs"
|
||||
|
||||
step_ca_principal_mappings:
|
||||
- local_user: wed
|
||||
principals:
|
||||
@@ -30,3 +39,9 @@ step_ca_principal_mappings:
|
||||
- ryan.blundon@protonmail.com
|
||||
- ryan.blundon
|
||||
- ryanblundon
|
||||
|
||||
# Leviton My Leviton API
|
||||
leviton_email: "{{ vault_leviton_email }}"
|
||||
leviton_password: "{{ vault_leviton_password }}"
|
||||
|
||||
jmri_vnc_password: "{{ vault_jmri_vnc_password }}"
|
||||
|
||||
@@ -1,306 +1,365 @@
|
||||
$ANSIBLE_VAULT;1.1;AES256
|
||||
33333438373661633337383161343862383963623736636136653339363064386538386437323135
|
||||
6436666330363032616365656434356161396530363363610a333864313064633232366133366232
|
||||
37316636366334346536663337663065303635626638666264666435393933343832653061323237
|
||||
6663333736386636350a653931626665346234323832393334373563303130626262646565653231
|
||||
30326263623965356639386438636531306430343239313162383065386135653534636134376333
|
||||
64616432303732646535643530353963363734663538393539633034326535643539356561656632
|
||||
36363330346131393262326561356631343631356434666264616134643136646238376336373665
|
||||
65346361636435316437353235393263313766623438336662353532306463353134366435326364
|
||||
36613865343264616364333631353238663238313338333232626262313637633662633163366638
|
||||
61323166393739656536656435336538303439636561363961353832313431666432393463636537
|
||||
62323939376265353330356334303334303532643136376531343765613738656562386665336132
|
||||
36346232313432396431666535663362333332613037623531636135376439393936353261653863
|
||||
65373061316461303866363338376131343736323933383636626338616431656533363833663266
|
||||
62363935313262393733333566386337633630666332353263666462373164346362313034636239
|
||||
33656462396238343065613965613232623132343562323364653366313231323531326364376631
|
||||
65643961303862653030343364326533356336393031636437393465636263626236303633326139
|
||||
65643432663037333332346261323663613031386563383533336538383133373332343363326530
|
||||
66643231666636366435623639643166333863376533633537363364396663363732333932663333
|
||||
62333239613264353233303064333931326236396538323130353061366139623362393734323434
|
||||
36343537393363343861633131346561353637646336313233376162656561623362653565623339
|
||||
63386337613461343534643964323932356661383436303966346339386131336133643132376330
|
||||
38656535353335303966636235643666306336396435636434363733313665366234393032333065
|
||||
37633231376134636435343233363032666231666134623365653238373836326237373831326262
|
||||
66303234303164303961316166623062616139353264343864363938313733653563663661633535
|
||||
37326636346435626362336437666364623264363565653935336438386262376663303230313265
|
||||
62663665666132313436663330616230383235346333623563313262376530356535613263316436
|
||||
35313766363763386234333733636464326134633136346532346632623534646365623136343363
|
||||
30316435613031333036383835656230353430636135343961356363613139353763343233633834
|
||||
38656431363139396535653565633631393462316334666464333839343035616531643466643531
|
||||
35613733373331643961353830376266376433363938373563653833343464633465663766333066
|
||||
34633364313762633034633461623232656537653762623532613332386132333365383062653136
|
||||
63323234626266336462333534636236653230343530336532653831313339613531343731313237
|
||||
39653737663665373361326433643364656434393838326339636238303964366336363961653366
|
||||
36643938633635313432663765316130326238306163346566313062626439626263633033623731
|
||||
63323532346366653238663931653334613734633963363632643462316130393138363436666532
|
||||
32373833323330356164626332646232623031326365653765336261323661616137373732356164
|
||||
63383633353764643233323430333233306337623834353934643864636339646239636430316166
|
||||
33643030333261323461643031613730343131343830353463313766353035326436306566343161
|
||||
35643534323763316165343832346433356364343061383036306434386631356537363331303035
|
||||
34303430393034393763343964613935383739656662643066363633333839396465386231383837
|
||||
30336331393365346636363732616661633635326364366435366466383639353839343731616365
|
||||
65326633636263653438343561633632646234316461616666363630333966326230356331373335
|
||||
31353665626233663433633563623765343431376639316232346666663065656462616438663333
|
||||
62663662336662346563333935333261626635343764356631356363356362346231306535376466
|
||||
36393932666662653934626134613065623461303834393231616332363461306534303666353632
|
||||
35646363333065356631393863396536633838343365343238636332303862306631346639656166
|
||||
30373735336361653863326331373834663164393863666631623866353338366561326635323132
|
||||
35306535626564376666623837666435663337613164623966306138353161313239373436346239
|
||||
36656266323233636233326139326266346464353665323465663666646264613466636132626464
|
||||
37353762306538643233663338613062353134323832653139663036666337643131613062643936
|
||||
37633363376365356333353433613839653130323036643737633163336531323032393937333432
|
||||
39343238396534353330346664346135326333333638666261396463656663323935643337313462
|
||||
32396137613432623664323134356361323861306230353165663162663732326563626162346630
|
||||
34626536393962393764393764393234373138656432346332633963393135386238313563353761
|
||||
34366438353337396264373032336166663937343931633635666166343638386237396333626133
|
||||
38383766616566336163363931393438366566386565663938356630366430393830653037363738
|
||||
30666465326266363834653934643434666466326632366463303965303366326364316430366561
|
||||
32636631663232373163636535366333623261366466623262393631626334626263656534386536
|
||||
66323833353264623538663232376138346361646362313565333534363535663031643833373064
|
||||
38313062346339636335633962323933383230623634303431383236626166636535326465316439
|
||||
38646636336130373763646635386265373065326263616438646565313439613830316135616435
|
||||
62326430336337643265306261343036653938633634626561333664313035393933646533326661
|
||||
62626437306432353938356338616462663130646530623636393233373837336138383963323165
|
||||
35643231613565313434386163643637633130626435376561376235346438303234386637383835
|
||||
37356238616263646665336435306233396436666664396532373063626162346237376662393639
|
||||
35373665373638396434353062666430633362653163633639623836346438323632643164653638
|
||||
62623463343465303236346562386338646536656438633966643430343266313935323630616633
|
||||
62313531343764343966333439343866613965303036616634363162303434643533366465313562
|
||||
62393666313838313938356265633531326431333730626632343139666338656465306431636162
|
||||
66343532346238393464366134656334363964626264623263353537633638346239396539653538
|
||||
37303339613036306533353132326534306433353364613536316263326363323134333639346236
|
||||
39376263313431666462383863336638306631663563313364333235333338323364613361633532
|
||||
38313263353539633662353533646333353262356436353938303036663130306362303935356136
|
||||
63313234363138383864613664666635613464653765636461383738356135623662313866383036
|
||||
64626437343465333563346434323762313232643230323336366539326631396463303961666461
|
||||
61656338643532616337653362636362373837373836666535633762343335623033316264366261
|
||||
62633137356132306430623537616135613965393737323263343463633839393165646630383331
|
||||
36616339346534346636396532333836633436323232323364303963313030656330386336633538
|
||||
34383534313236396461653462306130376462396138303561636331663062303864326238303335
|
||||
65663832333631613963326639386665643135366632393438666361376334636233336239633837
|
||||
36383761616364363262306535653065303536363066316431303464373230326436336661396231
|
||||
66333438376664333731333630383031366237396638326534636532343535303964653436633438
|
||||
64616263366561653334346138393866353437393037343930623237326231303261383733336232
|
||||
37646533393439653965386563333232656565663531396663303334313839323361363864396332
|
||||
61656638303130303433666561646663323139643337326561333832373538306236326538363637
|
||||
38666534633330653564653066643335623938373331376339393338343533366436333330313165
|
||||
65366562386366666632653064663637343536313462326562626436356363306134306237613265
|
||||
32366338643237333739626331393763323934396265383138633035646533353536633866313432
|
||||
38336137383036373434363530353833653938663035306337643638316463373762613931373038
|
||||
32373361326136616331663733616133623936663530356566383461336334663937383230386462
|
||||
38656332613763653339363565613762383735333163633831356364333563363630623666616430
|
||||
30353237653730613361643233386139303639666137653733353334326530306431376666663664
|
||||
34633531393261343165663330373961646665396139323762313663376639366464306533623333
|
||||
62326432623334336166656565656230313063656163636561326632666438633966323930316264
|
||||
66373866663461616336336264356561373232313737353836646632383438333162383832616662
|
||||
30636437613163613631623165366631306438643765383263626133653231333361383633363136
|
||||
36653831623562653838633133646633393038666164346461333531373034356435363336636338
|
||||
62623464376264363537306166613461396630303039373239656664396564333435303164613235
|
||||
61346238386233653462616632303337663036626465663361636130333566303337313237626237
|
||||
65636633316366343064663636383362316665633263633136323563363935346138356635653036
|
||||
36383835323838366363363335323731333066393334373265653563336533616362633163343661
|
||||
66326630353537653633316631343332303231373630363635623136386537666434383032323165
|
||||
33333633643064376239346139633664633330323966366430353135616633633465363030386665
|
||||
36353164623564343934323063313763316566373830643163616633643937323732333561353832
|
||||
66623364666335316230346665353631663465633034323265666663653866623037353935393262
|
||||
61623065656637636666383239343661616233396663333038393963366537326136653863333235
|
||||
65633137626134383763663366396438366634313465363965343830333966373930333134353339
|
||||
35396538303537316636333339613731623636646230323662613663306537326566656666323434
|
||||
65346531623533336535393530373963343864313963666335636334396663393837396665373565
|
||||
39646138356134313732666633663061646330643438393736653433313238373135656636643337
|
||||
39396362383363333434376336613933613438653636663334336337653633613433623336623461
|
||||
32373565306435646438306365343465653863656434663937653938393839333739363639326536
|
||||
64303534383961333333386237366239386330386533343939613664363439373135333731393739
|
||||
39363166306636333766333332393265613133356262356666343039666237663664636266626664
|
||||
65316339323638353333653631363365356562623535666235313234613236326130616261663830
|
||||
34663063626539363734313562656233656131326265626637353239636638396564313561383761
|
||||
62663638623135343333316464346236353364616239633566393039633339306334623234616331
|
||||
34366435663938346335323932633330366639653636336636373562306561656536653363366233
|
||||
39383064623435356338636335343662343435336564363965646439393964323162363736346637
|
||||
65323765333364353831396433666561376437333834313334316165386631616332633533666336
|
||||
37313533623535313934393230353439363231623063663961376537393638636237623134323337
|
||||
30393032333530316330333736616463633262616631353238323933303766646137386465643366
|
||||
33356665396437623732373437373537303034353431353031313162323730316566396238653038
|
||||
30366234663037653736353137376264363934646366353761323364663538346662323934343266
|
||||
30373662613461663561623466633939396439653663353365346133623933613462623231616562
|
||||
66643738643661643766323266626266313330633936313964366536323263373666303130623738
|
||||
32323230356665323439646532343930333936373536626466656566363161646663616266326133
|
||||
32393137663562613531323832313565346632303131646432363961613331346636653830323136
|
||||
36323436373964313737313131346131353630636561636639316464323666636336656265343565
|
||||
65303936353165326137613638366532303265356537343337656165366663383961643134656435
|
||||
61383565306265333939613561613035616663336361396536343939343761663833346134646365
|
||||
34323233666439623436323261616432363838616437386138393735383364393164646638373630
|
||||
33386632346239383037633132636431366264383633333365306338333661303662363533376339
|
||||
62633663366362303038616666336335346332316366353136323666346531393938666236343163
|
||||
31663665643839313439663839353431343061386465366331663637373532613266383065393639
|
||||
62613565626135363133623261333061373738633333623337396631356332643039306638303434
|
||||
65663762373630323263333239613939346133626534613632396433376234663636363062656537
|
||||
62386664343462346334663662663865653766373738306438316533643962356136653565373761
|
||||
62393437326138323065396630653838326138653865643937353431656137393732346361353734
|
||||
31326634653331353232633061303230393462363065613832303630383237366463363666613864
|
||||
37643135303034326335326563353064343835323863313134616262386564313235653762376332
|
||||
36636533646235343138396537623266333566633466393764353731393935306432313266636133
|
||||
63326362653033613264333139633638643638326532313239646239386332323330306538366362
|
||||
37646432363535396161336465376363343430333261396536346464356135343462663462613136
|
||||
36666238633663326230373834303431336566366666666534343537313635366537396262656236
|
||||
36646639643730393134613662383238623835626230313833653932303832323366343765343434
|
||||
34363033323836663035363166313630313062663162313165636138373261386636623831666632
|
||||
33373962626237666130373031633437653966386561363232363332333633303961323864333263
|
||||
39666230383332376537313137643635623336316336646537333137633564376432356432326639
|
||||
61383565313233376435636366366531363462306535636131636163636130663732383161616364
|
||||
37343064333264663533343739356139353365386565383833363165326536306431393764303632
|
||||
30653865623566363965306636366530303261616663306433393135656366383138356337383365
|
||||
36323736643035383231663034626564643235646237623632383164633062326132363331353634
|
||||
61653761346439326665366565383534393635616365353361353863363763383664303166383434
|
||||
33393732313837373336616564343864633330663566653933356237613538653366663838383132
|
||||
33303437323438313837616134616466626562313262653866313034383764656566373331343738
|
||||
30613062303630663365636230396231313065356662376163353632613564363066316239636434
|
||||
66356535386263353737383838653538316366343935323337386561353533353161636538646363
|
||||
63643739376366326536333261306263653563353331393133323964653764373932663136316335
|
||||
36333535626231373864316634613366353535393932373039646161646466363663626561373033
|
||||
64333133353931303361363466346666376431363336323166356162363537633536336638653838
|
||||
35323562653138393833383130353232616564393738326638366565616233633634336438303632
|
||||
30353765623565336330386335306162336662643466663335623836383238316161633339373137
|
||||
32393165323735643034633464313337386362363331353432363233396538393835383933313637
|
||||
65656330383732633537646361666439303266653165313233636464313964383639346635376538
|
||||
64633861623439626331643431326336646135656539313836323135373862316438393237663564
|
||||
31373135356135356662336265633862343435356530393835376430666566653036373338636332
|
||||
64663230323361356262356537656262396665663538386664363735646234306231613339636338
|
||||
34366433326535353965376236643661646339366566663531653864316533356466653864373633
|
||||
33363330333765373662646634326236313434383433653563383436316635353461386662646165
|
||||
31626566616163373137326165323066646535663766636265386435633965326165303831323633
|
||||
39613139623835316165326138386431383530383035653931616334353235636138303761383037
|
||||
39336163373332346365333935656563343939623335313461363138656239666263366236663464
|
||||
32303732663937363964346639323236623166373334386239613564643834316536363634326461
|
||||
64623939363635613863613065656363303734643766306564326437663037623839303734393036
|
||||
64326363626562363934313930383766383462663766336666663338653765656137393362643366
|
||||
36326336323638393161353464396637633566383262333234656233653334623131323532376637
|
||||
62613064626532616631646232386631623839326665346431333232333463373031316237363462
|
||||
30643731653666643331643839343865646437613135346363646131383262396134376363303335
|
||||
34303164303236383139326264623462363062313139653465346435336163653863633764613631
|
||||
33623666316530353163396539323238663935643764643939333966336137353034663164643733
|
||||
34393132323339626331626532643266336435653636353930393738363134353661653636643263
|
||||
30313138393263383837333438356463636162383938396137363035646433663932613066386138
|
||||
37316535373939326131643930353435383232623366393630663463633233393232373730333631
|
||||
37623837373533383462323431323335316231356436303239313535633837353931613730303363
|
||||
31613765626466636636613366346166393766373834643337353764666532376335376139316138
|
||||
38646133643736636363363434306334373933633264366530373534666637393835366266306633
|
||||
66653966613136653936313732396366323830376236626530306139333030323966383035643733
|
||||
36353833323539393136303035393866366563323430636336356639623734656135356566373062
|
||||
34666433316139316664643234623530363732373338623162363636303463346463323131333065
|
||||
64656566353837336266386233323439623237366364343635306137663038326434366531623062
|
||||
64613365323339613465326464653965613439653766623762363139366264663765363862323839
|
||||
63346539323961343532626465363266626234363964313531353662636635623165376130616536
|
||||
64363361383766636333626366616531343833633033643965626435313130653938316430313532
|
||||
64363731633139383236326231653434663132313463326563393739393132663765396337643035
|
||||
32373963393830356362366130376135353935396631393533333430363531636234333832373939
|
||||
36656233343663623262346337323034653036373432303266353333626434353137353939346665
|
||||
37633432303162323538343339643635636464353333313063363662663033396630303162333639
|
||||
31633537373364666234653938633461343738333464653736316331343038393333643936646462
|
||||
36373232373962653031333730363166393463313232616631383233386533373464366465623335
|
||||
65393462353638633931353331313233356230303565313030663739626230323766666466326339
|
||||
35303361383065353333666161306433383139366439633730666364383130316637633464646336
|
||||
66643065303934663765623233333261313939373236396136366566306438626439393163636430
|
||||
30656334663361396230303239663338373664346362623934313130343064353435626439356562
|
||||
35653638613037656133313765313165333066353732363361386234353635626563323932356237
|
||||
31363333326536353536323931643739333764323637356164663566343563633761333462383735
|
||||
66383134646139373130386366316532656139336231303931656265636130643130616531366530
|
||||
62383132643236623835323362383866613765343762333536333365336338373264633130656438
|
||||
36343331666431326530626137333665623836636136303031333965356366396239396534663263
|
||||
35343730303534313830623430666566386564653036356635623038303133623231616635396463
|
||||
65626137633033663631373366373238353766323533643462396562633164633730633135396266
|
||||
36316362353964306463336237383232306530613539653462326137383938333638353038396132
|
||||
66343265643162363735663235353161393732326530373735363662633165316265323065396163
|
||||
31663761303637616566396262346238366631663562323566323364663464373930373831323630
|
||||
39333165336264336539353634633936396561626363353466303931343436306636356663346438
|
||||
64373561666638656439353964626436613164643663336336333337393332346132313830316430
|
||||
37336139333539316639646337383737373530643533336133633135333936343165303764623366
|
||||
35346432656263396439663635653066303861643938633033393334336336396133626237336432
|
||||
31393061653839653964613239353336653666366163616163616230396661313233613934326339
|
||||
36333733663032613633613037343337646237666565316239303231313163383264376636336338
|
||||
37656532373634383132346163316337336534316465383531316465353861633062336235376631
|
||||
33333963313730653164313866323461353334653231333762336232366434333631396135333466
|
||||
36643362653137323934323966303734316334306133656139613763336431323236643264663530
|
||||
38633730346531396435666530316563386566323737343766653135386537393165356264356534
|
||||
62323962306332393063313062313230333334353833393362656166353638323366353934656263
|
||||
38633531306438663134616262323332323338653361356336353962376331323537623734316532
|
||||
35306434383336633436656437663765626530626535383464356532613536356533646332646135
|
||||
39613030323432643161366363313262633261366566613931303663636262353466313933303162
|
||||
66373535353934626135326439653630303437666535396266303532626636613563616633623237
|
||||
38623938323137393065616136376663363234633735363264356438353534616338393832386337
|
||||
66663339313062613766666435626364353935643932633233643630323464336135336339376437
|
||||
30323536373639343738613463613466396335626339343135306666663931656564393664643761
|
||||
65396336613465313035396134636438636134316236636566333031646666383939383264303437
|
||||
38656337326233363866336363363161333236326137393465633935626165663136316661316131
|
||||
66393764313130373766386235663134376239323037373365333730643336656532316434663265
|
||||
35646664383338326366386536616339393938336539373762643839613763323434363037326330
|
||||
66353738643735656165396138343262653563353934376237366564636638623630393238373530
|
||||
39336335303564616232393165653639326166333536343966353933633365326539383632646639
|
||||
37636238376137646239353731343933346562613630656130653865633834626363656231373162
|
||||
65333662653962373139633466383834626566336234316161633038653137393266393564376439
|
||||
62343365666637396639326438326631613664666662653935346230316632396361663038663239
|
||||
63336335643665353162393330623635663839666162333561376265363330383530653733333239
|
||||
37373665646562343562313764386534663239636130646564333433313661363338326439363865
|
||||
63353064613161366461383862623363333637633430376361366566383332636634646630363139
|
||||
65303139396539323336313666336362626339326666343061303431356361343438313365353431
|
||||
30303266363664373233303231346166643839343564306363613236316665306437333636343131
|
||||
61303234666664323138376335623639393939633836353664343232316630393263323233343366
|
||||
32393736336166323166616139323066613062646139656236343233623630616262343034383631
|
||||
61653662383537646364363764353335333630353431633534313731633265626263303863326565
|
||||
37616534303164316162353165396236656565303532326235393237366434323739633662663261
|
||||
33383635653734303233356438636132343731383439623463633732373833303339353662386435
|
||||
63316138353766303962616435383138373739383835353662613666356565366237353932393263
|
||||
34326561613234356537343234303166613534356134643766383332646263623430653632386365
|
||||
61376165656366366532346465343932313032643163313362373633373066653036346238343938
|
||||
66626133306163353966386662363039323166353330633239386264306337386361653430633838
|
||||
66366636616661313939316562373330313564383033643035393738353939613066316138643238
|
||||
35376336633238636463376234376364386464396461636336313562616361396435323639666433
|
||||
34653139613962623235613035363634383735326564333332333632656237333238356566646536
|
||||
62346435303932363231306634643037623935653235613330666666656433396463656231643364
|
||||
34313235613064646432633935656231643162363530313761303339323830653564303963643532
|
||||
39326163396461613161363031616336666564326630323465663831653431353566643661656435
|
||||
35636238633663346139343732353233663930353664623636303134303633356232313063346566
|
||||
30633762653239356131613539396435396235326431633062616436613266353266636133656531
|
||||
63346533383764393733396436306136383830313265656539376136343536353737313365333865
|
||||
61653665393037346231656663363432666332383837393264656231386435323333376633333265
|
||||
39633563653161663531303330393237316132653339653531316538316534353334383161623833
|
||||
63656534643833656366623031326132343732663536393636393537383338613462336638626334
|
||||
39326264623032393661303335356235343736613964636261633038333236623232613633343263
|
||||
34363865306233346433393664363038303464363834313463636363366333313433333930336634
|
||||
65396232663031323965343336366339623239386566323362396564333932336134336231646430
|
||||
34376236636536626264663532323530656634333137363431313339643164333236303337373639
|
||||
31326131636161386665623838303939616337636666346363616465323263373061633832363839
|
||||
39343131386333666231643931353330343136363732396465383336303866623733316235646234
|
||||
65613531613333653231666234353532313738363266353336303564613061393639323933646339
|
||||
64396334396431313563326136656539613535663334323330326566613264393965623065616634
|
||||
32313333663535313531343363656230656334346430643365363838656430616366653766333136
|
||||
32346137656136666432383934313865613535303962333464316465313464333233646162376165
|
||||
37363265316139643332393761663164346537663064346133323661393030393533396633646134
|
||||
39383539333364626263323364353764316539323162346434656234393562326133613632633137
|
||||
34323537393838393534643861356365376461636235386638376661623439303832643231616431
|
||||
32663264306166366336396130326435346432343732323764346232356439633361663937313233
|
||||
62636130663238333433396335356331653166366130626632663162653736656238343134353134
|
||||
39373663633437626661393734363264313137663138643665353633366663366562386337396561
|
||||
38383464333531306138316266626461326531616364303732343337386464366464633834383439
|
||||
66303334333234306166396235666164313833333164356561663063326563303934373933313334
|
||||
32326466343761333666373936373332663939643662363665323534333638316637666665326564
|
||||
63616134623530333739633265343637626561383831663836626538313539653536373165636133
|
||||
62396432376332356133336461633664336264616566376664643861363838313766393833343366
|
||||
31386636343835663062396232646663363935396366343361633462643734373437666331383035
|
||||
30373963323834343138653063313635643062396130373861643134363062363135623730663335
|
||||
34373935623239363866393932363966366233643732646461343964383563393064396331353331
|
||||
31646662323364376639316562666465356438386637393130616330663563633839326661643239
|
||||
62346633353963336239633738336431326638626232313965336361636430323739653734636436
|
||||
39316363316137366233616431346639663335393737653562356532363634346237336266333132
|
||||
31343836326538653164653163653738656238326366313532613433333337383263646439326661
|
||||
65306135363138646665363463636436303864316432663734393130313932643833356161373139
|
||||
65376334376136313237393336636630643933373165336432643233306366633663333930396364
|
||||
31393739333730326434313639376365356432376338386430363133666436643837633533616333
|
||||
36363464633233343130656163353137663165343064373039363439346631393436393430383361
|
||||
61353563346531353561303935363531333235313731303237326334363763646131653961613334
|
||||
33306534663238343433363237393234643236393931343230656438353866313638636633396464
|
||||
61636561333562383964663261316135373235313234613564333333336438646636303561616634
|
||||
63363033613238323463646165373361613834353230653138623132303731393436386139663533
|
||||
38393263306139303231626363353931646163323364666161653861643739353262313561393165
|
||||
6262
|
||||
66393233316132396639356564316439343234383066633231646134313361666463656536323732
|
||||
6630616536646439613533363430306466306233643730350a343364633233333335643833326163
|
||||
64393933613963313533623733316339396236363663343635346663323366663166363839663837
|
||||
3331363062653239380a326566623264623837326636383939346430666537613361333638366630
|
||||
63353062393335316663633739313532366363623739653631366539323435336361353331386230
|
||||
39366634643964336233353961316630616462663166316266613037623363346335373638656365
|
||||
38353733396636386133373836346336383231663661346137373164386338623733393566373563
|
||||
35653933343036633365643535303934326537356136666539316137363433643266346630386439
|
||||
38613332646238366536333536343031356532656336613530663830613264346339353034323362
|
||||
66613530626361323535653232313730373463373332313561616631393461353730653464343063
|
||||
38616338363938346161616636316232313838616463326432353639613837343162646363343232
|
||||
33323064363139376566343866626364373662393138353666646234373461666163363139313631
|
||||
64343261326566363265323463663538343034306136326234386664333837333937333136653563
|
||||
61333531353434633339383661636363363535316366353330313566323133616438373161303135
|
||||
35353630613037316466353832333033393030636331386438393133366333653832393731366363
|
||||
62373638303737393162303461646239653865653834613662666636373364633165383062643831
|
||||
34386232376361323638353361666530366432356331353963303930326535663536373339333062
|
||||
32396266373430343339636635366434313635313766363863336464633961666332353834626163
|
||||
61653637316163636465343630353431313863653033643237356434313564366361373435376662
|
||||
63643737353830663236643862613533623237373531646136383763303766336139303632666235
|
||||
64623834323966363363663730626437323432623966663537346162656265363562643836633731
|
||||
65663631633462663764393132326165346639353033633035636432613039336164303538396632
|
||||
66396264393865306666643636353638613661313230313337383634663839363439656533333932
|
||||
36633432306131396539386539633063653230363932376264323537396434353364643432653661
|
||||
32643162363066636432336363323534316436613838646562313538326566666239633234646236
|
||||
30346534636533623365326564613561363362333364363037646561656635623935653466613565
|
||||
65646361313436356261643762313339333864356338386136306162386262636464393130303963
|
||||
38646464316432326431326661343632396235626234366133353461623862316662326432356234
|
||||
37626366383861373831633639616465663564643866356664623066386535646163336134356534
|
||||
33366664616232353863626465626364313530353335306565336665663866303736323162393362
|
||||
63626261653161663664363833313461653034326330653835393737616135646462366665383935
|
||||
30363639306330636634386433646231363530633061336364313338653632323831393630383934
|
||||
37316362326338313733646332336263386239626539383330353362616132333161613464313066
|
||||
34663434326662326233363432306433363666356132383866346336336261636435366332666135
|
||||
34616231613638363339356333616536643266636363643131653330396162306264303566396461
|
||||
64363763376365356533636430643866333361363062376237653237663731663934306265646630
|
||||
33393637656335643366383564373966343265393630333835303731316339373133633462383364
|
||||
33383435383331303264313334393532373932333334343862326635346135613932356337373034
|
||||
62316262326331313135376465343336373266663338396533666431616462613932663861646238
|
||||
62653563623535633738383033326235383666646333653731316233376231623661306462303732
|
||||
32643064373236613336396233323435393939386530323331336138353364663762356538316562
|
||||
33376530623664623733386133333433303031373337313366386236376539613964316135343865
|
||||
33363963366165333238356663663435386439336366646138313034343636653463323938633136
|
||||
39363966376238306662303265643034306136663661393738633436393432303139313132616534
|
||||
66323432313635386162333838323136623634653264643438303264636430633232323434666532
|
||||
32616664663063653735316237643539633133356661333132323238376333356464313262653836
|
||||
39303566316332663737323437633031353330333365383837636336643763313433313937396531
|
||||
38363536343438663966663436613132663661613134383431633765383164373762343435316161
|
||||
62303631646235343063383230343232383336356562303563373933346530393333316634316437
|
||||
32316330306163396434663031393965663163666537353031613365353437666466333464626238
|
||||
35313739646535356665323734393965303064306132626261363062363438383164346261393463
|
||||
30643438623363323161323230306230386332363635386234666639623566643536626637616533
|
||||
37396136643930633262333331656363376433333234343630306535313262306235663263663362
|
||||
32653434363035613732363136303363393939323337613661333439393637646262383039386661
|
||||
62326163323562333339323636363565623664396164383332633666386130613766393138346134
|
||||
33343338393536316431353439353062663164643634396363353131303038353965393466383030
|
||||
36656465383938353936346361393963356630666630373236626237303064303062383638373730
|
||||
35633866646535313432353338623462323235346433653431313031363163393666626432363238
|
||||
39623361316132626230633336636163623466313666346631656134343762656566353432353264
|
||||
30353436356237653231363564626134633039363035313232616333336436393638396233626638
|
||||
32663230396539323761313838313466376165646430346634383332346134653662393161363337
|
||||
62646161343665383364306665333164666231386531626465373366623761643161656462303733
|
||||
37653438616233353432626466623163316565353764323762613635333832343634323665356336
|
||||
35353162326233333836396337356466636131383838313436626336663132346339623261366465
|
||||
30623261303933396562353331636638376135663330643638643536346261626632626139386535
|
||||
66653332366361336636666437643165656239613031303638333232303836383132616636633938
|
||||
31343034643037623731643931316463303639656266323231313666356336333133323135363330
|
||||
63373365303131353161303630633738353536393631353034666139383435303461316131646138
|
||||
61333731356538366366613831303565613365633965323235366166313534653965366433656533
|
||||
62666136313662366638356237343734336333313034396465346632336262306531633535643238
|
||||
35333831366532386235316565303936616264373337356134643066396531383533353336303131
|
||||
31393837623564386535323532653733393734393164373235396566333565356237356438313762
|
||||
32623765326639386262393639376461326163333237313232386138643130643231626466643663
|
||||
39643061393566353434333136366335393536376234366266376265333234643536633035653933
|
||||
36316132663539306465343039323935356361373439346437386234386464623962643464643562
|
||||
61336434613834336161633237383361303930313464613666313834356330343138633735386530
|
||||
36616233323366323961653965613438346136373738366266316134356266623664313539636235
|
||||
37313033373466383134346361646562366531333338386330653736626530396238356639303131
|
||||
38363738396236386461316433316261326435646130383336316234363461393237623633633336
|
||||
30386630376565646337383738663939663462623232316635346635653830306664653336343033
|
||||
64316430396664393532313766326437636636626232613036666666656430323136356436333564
|
||||
36303334303562393832336433343438396430373833623137363736386665343866313064353063
|
||||
32303634393131326464656535633734386462646339663533666430336265653965333538633866
|
||||
61623666643839653239373335633735373738363736313665323365613635313766656635613832
|
||||
33616461643539636165383233636533626230343138663630323731626139393230383464313430
|
||||
61333438393337316239376435313337313437333931623238616133666138363235386533633437
|
||||
35356163363231656536353934643539643562343732626630383565623730626533313230656164
|
||||
35333735666135343364663233626163363930383262363266303265303638396239636361366534
|
||||
31333863333565356135613232393165353266343632633532343061663331633337343538376265
|
||||
39316630613439356262396634316361356436336634396337353339616536356336653930613966
|
||||
34656439653366363562636639346430623561303463356337363830373966366632303337663564
|
||||
35663632313265323365636238303364366230353039353561616636633664643233343430336237
|
||||
63373264643935616331616632633065366638363833306337633563653065363464343137623533
|
||||
36373231363739373335346464623533393336613634333636613937366136326464336332346166
|
||||
61376263623835646163353134643963663964373732313833346163323138633230393537636664
|
||||
30366234303334656130336630346130656237306161376566336534653630616439323764373665
|
||||
61383338326163336164353265326163646165623235626137623237306666333832306461613630
|
||||
66373331356465346261643466323662393661623433383265376666623932343861323139383531
|
||||
64633536373362643935633734366235396433333237306166646164363930613862613365303663
|
||||
38343833336137353634313362666665306666393635663633353934363832343739616331386130
|
||||
66336561633039326434313833303465366638303961626138333165623331386230616130626639
|
||||
34613962366230333065633761333335613636363533656461626632343631666563383738623330
|
||||
30333834346233653938633330663166616331376436356533366461336264643264336139343262
|
||||
35393665656230663232366133393037643536366234343537326631623332373131323739363638
|
||||
39653162646366316639313631393631666261623230313538613666393732626438393763646330
|
||||
36346661313131313630343432616365666633353762623261613039623331396330623939626132
|
||||
34626333386538326434356432623965666662663437646237373537326534653634346239653634
|
||||
31373038303639333037613637393862356263323066666630313262366633313932396465633337
|
||||
66653930303934616236323064613761353935613835356561313334323762633064306661346666
|
||||
37653262343865386236343634316336386630393739626437333065323433613531393738313432
|
||||
31376233353463373237653164386363633334366332356538343966663939656165323465333030
|
||||
39656532363363333432626638626438396539336461326338353732376235316133616666316261
|
||||
37353063343366376433653961333233306461303133376661303332386230346231383837396133
|
||||
37323137343066383966343535633363643233663530613566313330336232366638396165373631
|
||||
30626531363033313833303836366434613736396339643032663066333865306535323739666162
|
||||
64626133616433653864376662623464343131303938303237316264393765303035663833376464
|
||||
32663264383236303766323935306463643138396237373338653238633464616238306132633735
|
||||
31626538653262326533326266336633623532623935383266373533363466313033393235663538
|
||||
66653038646233303665343634383666343363383238326533366136363838303332323230316662
|
||||
35383235646638653539633961663036663933306463626335356631646662636230356261363261
|
||||
65633261353830373865636630353932323937666331353635373736376436333361613330366633
|
||||
30663939356165393132636131663966373433623063356265353131306532643066306630656363
|
||||
66636636353262633437663264613266613663656137386231306231646264363661613035343538
|
||||
37353633643065643236376537336238663137623735613038623766393231643131653436333262
|
||||
31626463646432613563393665346532386161366435396364663239386236616233356131323536
|
||||
33633936623762666534633862363466353736386137636363633733623366346337613365636439
|
||||
66663035313430386464623833646135333062313830396637323961386135363461326539623432
|
||||
32653865623530313637393561343465636430373162333162646631643235653931333830326266
|
||||
36376631316165343631326165623838306239623764363262376634663236393933343838376663
|
||||
39663834306165313330393739363133396436376437643232346336386531356638343063376465
|
||||
61613037623137306666383231376539656361326132396662613061376134376266633764336266
|
||||
64336334313335643635303632666431383637306334376462643630646339396435313830313363
|
||||
33383162316261663035393962306234613865613366353465373035656434366261383133653331
|
||||
63643235616362663663343330303765363263366130393837613939323264373937333162636639
|
||||
31643438666338646135663538343231643235646364623761653064633566656663383465626133
|
||||
31373935646266303565303539376162623132316438623565316537306337636630313861623937
|
||||
34313832636533623033616139373965303839356530353935643363613464356364343162336466
|
||||
66376130653162666661313139613530306666633432346639656466653364376435636461626362
|
||||
32653633303561346233643463373534653434323134353434373839373937626663336464303866
|
||||
33363264623038313835396231373132396163363662626264346461333539326365326165323066
|
||||
62333139383334333334353031616430323339623066363232313937323465356266323934313761
|
||||
66623835653961303830383030643537393130653935313265333062393034336562633535323263
|
||||
65616237373232336534393834653162363461336262653862666637326266663966356665363036
|
||||
35383437326465663635303664643236633435303862633965346133376536316233386333313634
|
||||
33643565326633386565653961646463383866646636303537643436623734393234633938333933
|
||||
65366164633165623333623362393639656661326332306538663738356364373734316563653038
|
||||
34646531616662386232613034366332656262343164333531353037363036646262623663666236
|
||||
38613238666136363431623664633863636365396236666532383930336636353031396232656435
|
||||
61346238643431653231623861373964383931336535363262373437353532393165316562386134
|
||||
36363263666135646237383666373833373737396330616163376439663736663937666161313831
|
||||
63663531656635663339306365656663636633343733636165386230376332616331313638386538
|
||||
32386466323232363533613334333333346161376430373436373961316564343061326164306138
|
||||
33616263666262323430303730626266396535626439623364376239346564323730323534323938
|
||||
33346364393033353865393864326361643734353234613563393138363334383536396535393166
|
||||
34623163616336653436393639313965353237633566313039303137326234383230323235363234
|
||||
37626161356166356365366164363863636563316332393638616535376466343537373966643839
|
||||
32613930643533336264626136626465303339376632323034386161663661376466616233633065
|
||||
66313739346162363838346663623266383130383736656334323430623463666439386532643630
|
||||
33666639613830386136363535363830333234653961663739343537306634616531616263623762
|
||||
64666230373830636238353062666330623061613663376638343763626264363130313464383661
|
||||
38326530333362616163363735323861376366333665623536383566653837306131623732373639
|
||||
31316661353332633630326162663738636562336666326637353764323431613666303038373532
|
||||
62343661336338306561356235396636343130633365303466613637633363613862663233633731
|
||||
66623530353132666261316637303763363830623734346262333633646238613131346564303734
|
||||
62336434353432326239333232383833633962313537626430663130393733623162626131656366
|
||||
64333535623138326239336165666562376663663334323036323539653734333835386331653438
|
||||
63353861666239396437346361306634613462386335376137333963333838616138633730393865
|
||||
62353539376136316564666136646639363635663736636439393462633165646632623664383663
|
||||
31613137306461616361323832393036323933626531363536336261356636303531633239333362
|
||||
32663134363263383039646162643539663737333861386437326337616362343963373532346238
|
||||
37346137363933623839373838353939386630303461346438666534616434333031373730393537
|
||||
30313134643963623564356266656430613430626238613266316335336265613132616562626261
|
||||
35386435313933626634616463616166646466363939313639646264346464363337656339323366
|
||||
36366665363739356564363232313762323565323134616134666337336534353464373637373130
|
||||
63613265366436313131356332316531633732356461383064383031613337343363646432373936
|
||||
33633339386632653032663837346130623636356464326637303338376132623734333932396232
|
||||
61333033386265356630316134383066343164613130666664643732643362666561346132656266
|
||||
31623633333039633837383264363937623435643061393935393762346430396335373864633634
|
||||
33336136353332663366313334353739303539633364663231636539333132303966383432376262
|
||||
61356563323232613433653262623663336634626532653465306638316633663564633862666666
|
||||
30366133616336326661626238653933383164336366333438626235636631336165386664343736
|
||||
62663961346664656333306435323833366632346366356238653731653937626333653630623334
|
||||
64326662346138386433333232643262333835326263343239353264373038613634356436396630
|
||||
38323931643361663238623766323930666130356339363564366661663033303831363138343737
|
||||
66633535326131396236653261303836613364306537633637323031663166316338323533323731
|
||||
30326235323066396663613531653061643661336631613835626266626436386662353465383065
|
||||
64396562343966303362636136616438353661626466636635323961613438646634336563636534
|
||||
38343566396530643961356434643933636235643561353232643062303232323437666261363061
|
||||
62636530396466303466653333633930376465366561376363316137323263333561343334383364
|
||||
39313863643062643766396564363137386231373136346138396162376264653538303464633161
|
||||
61663363623937356138356430666461623130323466623162653863393736326264393836336637
|
||||
34663931646566333535666664653237643732316663323230383239393763376266356135326438
|
||||
35656266353865623663373366373830613361373664346632363031356265313364623866643438
|
||||
61363866353934636337616239633330623734666138396166313864333939663563636138653930
|
||||
35653137363033326432373661613434313137623163356134613265393238346438306165313639
|
||||
36666662393165353565633531663536613037623230373063316639663632643139353235303462
|
||||
31393263656265656131613164363035343233626433656135353331613532363236616439363731
|
||||
62666432356363323937666435326638323437346136366131613636653430306131623966356263
|
||||
39303739306233303862636535633431363630393432613663633836396566653039383735303336
|
||||
62623331623034636363636661653236386337326666656532343737336262336462613762326531
|
||||
34303066303834386366636430343161653665343362363038396562626133636135656538306435
|
||||
36393364333066346238396362663664643236373532336263656233386663323663623137343462
|
||||
64386337333130316434663564613665666238623132343437656637653035373738313735366630
|
||||
66383639616166393265616434623463326437313530326130376339313662303836636664366232
|
||||
32366634633030333130316435616233396231663937343732313066373834326464623139363663
|
||||
31323931626364303230666162316436653065366137663631376265383063316534343736373261
|
||||
31613637316235386539343766323439653062633137663730343236343661346162653366656332
|
||||
32663932313063383561636266373766633535656131386133386135663863396261306530326632
|
||||
63653936626236316539613262386231616433393064323461626536363831666461316131383837
|
||||
30646266646266393666396362326238613231303335336532303836363264323233343534636635
|
||||
66313538643033343262373463363866346566353263303966323933383963363463393761383865
|
||||
64663932343830643531643466303438343161396133666463353762393737613036646166333265
|
||||
66376231613232666164663964636134653061633330383863373836306366393838393235656331
|
||||
35613231306263373230326634623262326333356263353961633836396531633431383163633361
|
||||
30356534666466653734333437383964346564346165326664633738653338313263633837316531
|
||||
36613034323433643839333264323864613033313137663131623265643364333664646235666232
|
||||
33393039313666323266643362323337316465306564303230646561303434666630616137633831
|
||||
34346439616634343337306636643733316464376631616266376437636439396337306637333432
|
||||
39306333363035393436316434656436353738303861633933376531383862316466373736323639
|
||||
35336137373866336631386436646231653366366435363932376434303063613961353261343661
|
||||
33383638393165336438376662306431333837356435626137356130323836396335636166306662
|
||||
36386161353739353637353861306666383966323339303262616239633930373633323937356632
|
||||
65383032613031666665623631613430666662656336663931646533636230303261646530623765
|
||||
64643939326435643539373564336531623236653731636636346361363064333963376566616530
|
||||
62326639663632666634326233363635383830643163373938646165656163643864336436373466
|
||||
36353832306632386230373832333234643638313238626333303963383962343265366137656136
|
||||
37333261343161633562346638323632616566646162633133663466346535656463393932386135
|
||||
62653332313066363965386335356430326539316366633537356364666230326237306236393563
|
||||
35323363633936353034323232353366373566666332323737653237323135646665626139393436
|
||||
65333762653536656161386532363765336538653763666236343166653933626666633130393033
|
||||
31643531646633623663313237353333313136663863663430306131316165663765653732663164
|
||||
36326531626365326330643064336230313466343731376437316563303339336333326636633066
|
||||
61386135626430616661313236623030316362373338643233326365646531633265626238383830
|
||||
65346132643537366537626132666165616138656139626132396639376230333262643766386363
|
||||
39383034663034326165643636613237623234613666333532383733623462303331326238636461
|
||||
34383139383466356139333934353837613964326538336463643832623062633034613762363061
|
||||
61346361656363366136353336326433326266336564316366393565626262303637316564356566
|
||||
39376661383763373436306238393666653561306538333638306233356139346634653363346164
|
||||
31303835383062376638626266303237323832623735653066353936376339633637333562333561
|
||||
36646462653834316131323166366661386161646538346464386232306239363030366363633663
|
||||
65306439616339626635326531636435356134376561303235393337373564373937623636643432
|
||||
61393535643038626562366331353831663338333838383066323632383633346564396566653330
|
||||
34386563393832313537623061666466366661333934613766366165366330353835323637643635
|
||||
38613363396663356564646132613536653033616337386566623662333832383938303138316662
|
||||
34303339323166383863363831636233323335313565393933636435396666313337663037323432
|
||||
63333139333165646262666339343736383966346133356138326437386334626461636530336432
|
||||
61356631366163366561303636333230643732316261376365386463333565623533663966336365
|
||||
39333365393766306536366231366435363030353263393534653534373064636361333532323735
|
||||
61313163323831653362356333386638343566356261353534303738613730373632363534666337
|
||||
62313339613138646361356431616236613435393233343732626263653332663265393934616134
|
||||
35313165613766643938393839373261633439396661623961353934373130623865353639633038
|
||||
34353631346433663131653965326337663561613330323562666336656237633163356562323931
|
||||
62393833663233333538383063303937386365306135343962623333663435663431396438666362
|
||||
61386266353838653532393233353939363738306634666537313761313835333864633764666262
|
||||
30333032623766633334383031636633636539336237613235376466356430663938653565626235
|
||||
64373537656566306136633630366130363630633462656330623633393735386630343437336436
|
||||
65343635366531646130616534623136636666323139326462306533653532643962656530336633
|
||||
35616537303932343539336638333730663639396330653761346136363431346536666138336462
|
||||
66396565613166623934316532383835316137303134363466306163356233356530323231666464
|
||||
30353933306530323734306564626234343864373964333264353366326265316333343330356532
|
||||
38363837646635633461653562303264353633343461633339376665616331613733666663353130
|
||||
65396363613731366234326466323738663563646166653237613364323734616465643764633537
|
||||
35373865353532383566363632366564353536643739663761303565333138383638653665663664
|
||||
65643366316461613630366437623736353739356538336237613431306363663234373265623962
|
||||
34653565373335653563356135313835643266356261623037336536613733323733363933376538
|
||||
36626134396563623733656534363331626262643339633932373035626134343531623634666463
|
||||
61623036313334616639633930393562663631653565656136666537393731333430663062643362
|
||||
63633663396562343965313261373965356163393538666466303661363531393266316462626166
|
||||
38616536653665366462383064373766396438616665346666376232653031323566313164383164
|
||||
36643231646439663637333165376439333432383532316661333766363136636236326338386537
|
||||
33306130353634346136356234363438383865313136393839663066333935623565333730613538
|
||||
63323831663866363831383930303434333936646564316435303931396362303534386335343330
|
||||
65383635346662626363626365666166636361633365643735303762393832316436646139303835
|
||||
63633733656361643233323332613632653837663262306661626438316262653931333061336366
|
||||
35353939353835333361623261613738383734656132613139393264393038373765343131333330
|
||||
37663962313566366463623437323965326365623437363038633661313461383634626661666236
|
||||
32633335323861643037383261393164393933353531636134323765353962633732396230636331
|
||||
32613865373739366530303538313566346434633933393330346637346136373036306666336164
|
||||
35373732346334353432616561623031663331346431383235306537386466623339356366663335
|
||||
36333733626433653465336431666530626665373564336339626163633131353330656437643638
|
||||
31616563616665343635356231633135663665326131636664373338323736393364353636613762
|
||||
66636135633766323866376235633535613735613465303239343036663438333331626431623435
|
||||
61323537386434323638666537643236623632626430666263376534643336613635663762643736
|
||||
30323237353265613062373265643562373637383337326264653639306263373865333262376665
|
||||
62646331373931373762303461366163393839633135393964313937616437323865653735383630
|
||||
32653936336534666565373437666130396265363561333635316461663766346336623865376133
|
||||
35373665303433326265623531613038636166643130616637653165376263643634376439613765
|
||||
35363031373630333966656466616235616337306335363132386335613462363664653634633864
|
||||
61303635653932353730663666386263633662633736313461643932386161313762663761313336
|
||||
62653665313033656537643936373465633932626166366430643763313030393838393039323230
|
||||
62633334383938343433306262393536653930653030393033306661666264313630643564333166
|
||||
35383237633932656331653030363434313534613637373465663264643061303538653666653861
|
||||
35373936643037333866636131373338363062663035323531626431633362663364396365353139
|
||||
66383737666437353764333231303662393630643933376161366430376530613365363830373534
|
||||
38376633333936626430393163323830346166643537326430616236393733653761363235356363
|
||||
35333131663032383861336262653936376565646662313965303265623763613330653461333835
|
||||
63613563323135633438383931343731656333303362316533376339376636623037376431336366
|
||||
61393236383364356162633062666265653534326363363862666539623761623065386537616563
|
||||
62666561316437303763376635346536666437373361386666643139643737663333323933613661
|
||||
38326566663932333930616435626133616531306461356466326437623235613233393434626563
|
||||
34373966633834373430386132353163366465626262353863353335323830393266393562393133
|
||||
65383632653438646435343333386261653066613663623232373564666465613136353039313036
|
||||
61646462626433646330396664363938376530376438646262343231393262383733636233636333
|
||||
35633862336566636439653464613564333162613836343636316334316665383164353131373431
|
||||
63623030306564346562346237333934616134346536303365396533626262333937396432393830
|
||||
38313763393463646437666137353835373735646365373934363936346564326362376565353133
|
||||
36653362333432326133393837316331666663663263396461363239306239363733633137396633
|
||||
38393865613431653337313665313762653635656531353465623436343132303064303564393066
|
||||
63616639353962366666616261393766643364333634346630616436376565313236316539633537
|
||||
66653239636561393433383639646462616433653166613130373134376535633937353366383230
|
||||
61613335663434333835653236343633633038346335333861356637353965396632393833646635
|
||||
36653034356234663831333764303338663464316362646339376338393236336161616263363538
|
||||
63356631613239326161343031643936623366643432663732346438333265666535623664396333
|
||||
62303239363339626566613439396234303536333333653433393666383635643235376165666234
|
||||
30336236393962666335353233666463346530323531316438373130303933383465316638646461
|
||||
62643066376666363236333231386237376466633932313836323163363061313333633434663763
|
||||
64323365353336333736363436376232653436633739613437343538633632356665656364616637
|
||||
62313666346436656564663335636635393632353430666236313863613464626434323939383538
|
||||
65386265343434303632313739353239323565333734656566356164643430613538333234383566
|
||||
37646431316363316139646435333732313339623666613738663039613239613738393565333330
|
||||
33376432373933656435303737653762666464363865633831373330393435633332636261336139
|
||||
34336236373535636165353262323966363164633135613534353661316364616637663465363864
|
||||
39306163346365643339643937396165666366663339336438373031613937636464383531613962
|
||||
34373433663533626665656364623634373335313033646165303764396563356235343033383138
|
||||
62326361353630393938643764616636313461633734646661386536356235656665393864386465
|
||||
37666262346561656436343036646330363664306135333464663265306165353039396665336664
|
||||
62346631366330653762646538323565613864383534636532633033643533323736373931643130
|
||||
32343435333333613734626234363132373734353035326232366264336161383631353133663230
|
||||
32636533333866343763336439373336356237303636376334333433376630353338333261333037
|
||||
31666537363666653238383939663464346662636133346561326335346163363061393830616237
|
||||
36643430626534653331653665316535303139343763663965363164636238366533303038653935
|
||||
32356432316633373137316237663331336463363431393033323635646564366639346230353363
|
||||
38346663363039363962323137383366303862356530353238656563306131643236626536656334
|
||||
38303462306562366532346163323061393437353063326539393466616439346564383036303235
|
||||
35633731393661323962633631373061303930323638326565636162316436646337383266626561
|
||||
32326430363031396530396238353862333133363731623736376239626561626165663337373261
|
||||
39353461343461643238646635633562653865323336366634613264616662323232653861663038
|
||||
64396330626633303031333334343335393039623135353266383561313231643433393963326637
|
||||
39313530636361373831306234383166346266656261663830636631333564356536323565336266
|
||||
38306561376366626236306633613564386166616630613032633163613837313462343662653261
|
||||
63353437663436303634633336636532646439636465663362346138313665336334313039613631
|
||||
65326135383831613531323265353831313562346161663265366434623236636635333038366536
|
||||
33646631663662393331323162343438626666366636613438383665633136326439376166373462
|
||||
33363864613136643461663436396362643066633437376631623031613366656238396165313832
|
||||
31313131653263666334393664343239306235373862313339373563643137393633343663613936
|
||||
61356564636238363136623031336638333566633766636362303938653531306131396665303033
|
||||
63353362636463636236643464343562383161343432383766396330623764393837613435396162
|
||||
31303162356437663932663964663239623764666366663061313535346438373334636263653531
|
||||
34643133356638653031373036343162653135663734623035353033633561366266623566383233
|
||||
36356434643538643430383532393762333535636639353361353763333363313131646264336332
|
||||
34373331343930633962623963666365306132356334646636626461316236343839383266363635
|
||||
65623434336239313330343437646333353362303232346638623161616133636636626236643465
|
||||
36303636363965363765656533386534633839346363363738386532386531326538643134363132
|
||||
66613235373362633166343565323766306335336365333439323764623964393263623236623832
|
||||
34346132383136303038363764333039626234616132386464666633663536656230666133363533
|
||||
38306665643361666539636666316432623430623939663636343164386438313765633031313534
|
||||
65393633323837326166343936326263343833646331326464376138633461613532303135393036
|
||||
65353830373065393038343039323937303634346665393135383639303162396565646232663736
|
||||
39376434646634353330383933303164653431373433346335666131386165343035303964626665
|
||||
35383035653631346638326637326235393833623264323030373238646335346332353362393230
|
||||
61616664613562383639306564376661306665396138613066326631616531623132633966633832
|
||||
65363637376264336635633132633332373634383864626564623966356464373864393832323738
|
||||
37616338646633326636323461376137663632376262363738303336616463326238333465343533
|
||||
31316132386530326539
|
||||
|
||||
@@ -2,11 +2,14 @@
|
||||
# ------------------------------------------------------------------------------
|
||||
# FILE: ansible/host_vars/astro_orbiter/vars.yml
|
||||
# HOST: astro-orbiter (10.1.71.130)
|
||||
# ROLE: Ollama inference host with AMD RX 5700 GPU passthrough
|
||||
# ROLE: llama.cpp LLM inference host — Ryzen 7 5800XT / RTX 3090 (ATX rebuild,
|
||||
# 2026-08-04). Superseded the prior AMD RX 5700 / Ollama config below;
|
||||
# drive was transplanted into new hardware, not reinstalled.
|
||||
# ------------------------------------------------------------------------------
|
||||
|
||||
ansible_host: 10.1.71.130
|
||||
ansible_user: wed
|
||||
ansible_user: jarvis
|
||||
ansible_ssh_private_key_file: ~/.ssh/id_jarvis
|
||||
ansible_become: true
|
||||
|
||||
# LVM root expansion — xlarge template uses sda3 partition, standard VG/LV names
|
||||
@@ -15,11 +18,218 @@ common_root_pv: /dev/sda3
|
||||
common_root_vg: ubuntu-vg
|
||||
common_root_lv: ubuntu-lv
|
||||
|
||||
# Ollama — all defaults apply; explicitly documented here for visibility
|
||||
ollama_rocm_version: "6.2"
|
||||
ollama_default_model: "qwen3:8b"
|
||||
ollama_hsa_override_gfx_version: "10.1.0"
|
||||
ollama_data_disk: /dev/sdb
|
||||
ollama_data_vg: ollama-vg
|
||||
ollama_data_lv: ollama-lv
|
||||
ollama_data_dir: /var/lib/ollama
|
||||
# --- Staged GGUF models for the llama.cpp router (:8002) ---------------------
|
||||
# Data-driven list consumed by roles/llm-inference-multimodel tasks/models.yml
|
||||
# (loop -> tasks/stage_model.yml). Each entry is idempotently staged into
|
||||
# /opt/models: stat + EXACT-size check vs HF manifest; skip (no download, no
|
||||
# restart) when present + size matches. Source repos are public bartowski GGUFs
|
||||
# on HuggingFace (no auth). A router restart is notified ONLY when a new GGUF
|
||||
# is actually downloaded.
|
||||
# Added 2026-08-12 (War Machine): codify Phi-3.5-mini-instruct-Q8_0 and
|
||||
# Meta-Llama-3.1-8B-Instruct-Q4_K_M as router models alongside the production
|
||||
# Qwen3.6-35B-A3B-UD-Q4_K_S. The live files were already present/correct on
|
||||
# astro-orbiter; this pass codifies them. Future adds = append to this list.
|
||||
# Router --models-max override for astro-orbiter.
|
||||
# Default in defaults/main.yml is 1 (conservative). Bumped to 4 on 2026-08-12
|
||||
# (t_33acbb2e) so the router can keep more than one GGUF resident on-demand
|
||||
# and LRU-evict when needed.
|
||||
#
|
||||
# VRAM NOTE (t_33acbb2e, updated t_55c164f5, updated t_34b96e83, updated t_f5f7e9ad, updated t_441470b9, updated t_c5cef2b2):
|
||||
# With models-max=4 and all 6 GGUFs registered, worst case is all 6 loaded simultaneously:
|
||||
# Qwen3.8-27B Q4_K_M: ~20.0GB (weights ~17.1GB + KV ~2.9GB @ 65536 ctx, q4_0) ← CORRECTED (ctx rolled back from 128K to 65536, t_c9fed26c 2026-08-18)
|
||||
# Phi-3.5-mini-instruct Q8_0: ~4.3GB (weights ~3.8GB + KV ~0.5GB @ 32K ctx)
|
||||
# Meta-Llama-3.1-8B Q4_K_M: ~5.6GB (weights ~4.6GB + KV ~0.2GB @ 8K ctx)
|
||||
# Qwen2.5-Coder-14B Q4_K_M: ~9.0GB (weights ~8.4GB + KV ~0.6GB @ 16K ctx)
|
||||
# nomic-embed-text-v1.5 Q4_K_M: ~0.09GB (~84MB, embedding only — no KV cache)
|
||||
# Qwen3-8B Q4_K_M: ~5.5GB (weights ~4.68GB + KV ~0.5GB @ 32K ctx, q4_0)
|
||||
# Total worst-case: ~44.5GB >> 24GB RTX 3090
|
||||
#
|
||||
# OOM RISK: Full co-residency is impossible on 24GB. LRU eviction prevents this
|
||||
# in practice: models-max=4 means the router can REGISTER 6 models but only keeps
|
||||
# up to 4 LOADED simultaneously — the router will evict the LRU model when a new
|
||||
# one is needed. nomic-embed-text-v1.5 is pinned via sleep-idle-seconds=-1 and
|
||||
# load-on-startup=true but it uses only ~84MB, so it never meaningfully changes
|
||||
# the budget. In single-user homelab operation, only one generative model is active
|
||||
# at a time alongside the always-resident embedding model.
|
||||
# Qwen3.8-27B alone uses ~17,804 MiB (weights+KV @ 65536 ctx); co-residency
|
||||
# with Coder (~9GB) = ~27GB > 24GB. LRU eviction handles this automatically.
|
||||
# Ryan should be aware this means model-switching always incurs a ~30-60s
|
||||
# cold-load latency when switching between Qwen3.8-27B and any other model.
|
||||
# Proceeding to models-max=4 as instructed; flagged for Ryan's attention.
|
||||
# Router --models-max override for astro-orbiter.
|
||||
# UPDATED (t_f5f7e9ad, 2026-08-16): Set to 2 because Qwen3.8-27B-Q4_K_M
|
||||
# uses 17,804 MiB at 65536 ctx. Only nomic-embed (558MB, pinned) and ONE
|
||||
# generative model can be resident simultaneously. Co-residency of Qwen3.8
|
||||
# with any auxiliary model (Phi 8.3GB, Llama 5.9GB, Coder 9GB) exceeds 24GB.
|
||||
# models-max=2: slot 1 = nomic-embed (pinned, always loaded), slot 2 = LRU
|
||||
# generative model (Qwen3.8 primary, cold-loaded on first request ~30-60s;
|
||||
# auxiliary models evict it on demand, and vice versa).
|
||||
# NOTE: Qwen3.8 does NOT have load-on-startup — it loads on first request.
|
||||
# This avoids an LRU eviction race with nomic-embed at startup.
|
||||
# UPDATED (t_72646029, 2026-08-17): CPU offload for Coder + Llama changes the
|
||||
# constraint. Coder and Llama now use CPU inference (n-gpu-layers=0). GPU-resident
|
||||
# VRAM: Qwen3.8 (~17,804 MiB at 65536 ctx) + nomic-embed (558 MiB, pinned) plus
|
||||
# the CUDA-context buffers llama.cpp 6ea215d allocates for the CPU models (~1.4-1.7GB
|
||||
# each) = ~20,004 MiB steady-state, below the 24,576 MiB physical limit.
|
||||
# CORRECTED (t_c5cef2b2, 2026-08-19): ctx-size was rolled back from 131072 to 65536
|
||||
# (t_c9fed26c 2026-08-18). Qwen3.8 VRAM at 65536: 17,804 MiB (not 20,302 MiB).
|
||||
# models-max raised to 4: nomic (slot 1, pinned) + Qwen3.8 (slot 2, GPU) +
|
||||
# Llama (slot 3, CPU) + Coder (slot 4, CPU). Phi (GPU, ~8.3GB) and new
|
||||
# Qwen3-8B (GPU, ~5.5GB) can also be requested but evict Qwen3.8 due to VRAM.
|
||||
# models-max=4 is required so CPU-offloaded models count as loaded without
|
||||
# evicting Qwen3.8.
|
||||
llm_router_models_max: 4
|
||||
|
||||
llm_staged_models:
|
||||
- filename: "Phi-3.5-mini-instruct-Q8_0.gguf"
|
||||
url: "https://huggingface.co/bartowski/Phi-3.5-mini-instruct-GGUF/resolve/main/Phi-3.5-mini-instruct-Q8_0.gguf"
|
||||
size_bytes: 4061222688
|
||||
source_repo: "bartowski/Phi-3.5-mini-instruct-GGUF"
|
||||
- filename: "Meta-Llama-3.1-8B-Instruct-Q4_K_M.gguf"
|
||||
url: "https://huggingface.co/bartowski/Meta-Llama-3.1-8B-Instruct-GGUF/resolve/main/Meta-Llama-3.1-8B-Instruct-Q4_K_M.gguf"
|
||||
size_bytes: 4920739232
|
||||
source_repo: "bartowski/Meta-Llama-3.1-8B-Instruct-GGUF"
|
||||
- filename: "Qwen2.5-Coder-14B-Instruct-Q4_K_M.gguf"
|
||||
url: "https://huggingface.co/bartowski/Qwen2.5-Coder-14B-Instruct-GGUF/resolve/main/Qwen2.5-Coder-14B-Instruct-Q4_K_M.gguf"
|
||||
size_bytes: 8988111072
|
||||
source_repo: "bartowski/Qwen2.5-Coder-14B-Instruct-GGUF"
|
||||
- filename: "nomic-embed-text-v1.5-Q4_K_M.gguf"
|
||||
url: "https://huggingface.co/nomic-ai/nomic-embed-text-v1.5-GGUF/resolve/main/nomic-embed-text-v1.5.Q4_K_M.gguf"
|
||||
size_bytes: 84106624
|
||||
source_repo: "nomic-ai/nomic-embed-text-v1.5-GGUF"
|
||||
# Added t_c5cef2b2 (2026-08-19, War Machine): Qwen3-8B dense 8B model for
|
||||
# aux tasks (routing, rewriting, structured extraction, tool-call construction).
|
||||
# Source: bartowski/Qwen_Qwen3-8B-GGUF (public, no auth). HF filename is
|
||||
# Qwen_Qwen3-8B-Q4_K_M.gguf; stored locally as Qwen3-8B-Q4_K_M.gguf.
|
||||
# Exact size verified from HF manifest (content-length): 5,027,784,224 bytes.
|
||||
# VRAM: ~4.68GB weights + ~0.5GB KV @ 32K ctx (q4_0) ≈ 5.2GB total.
|
||||
# Thinking mode ON by default; use /no_think for latency-sensitive aux tasks.
|
||||
- filename: "Qwen3-8B-Q4_K_M.gguf"
|
||||
url: "https://huggingface.co/bartowski/Qwen_Qwen3-8B-GGUF/resolve/main/Qwen_Qwen3-8B-Q4_K_M.gguf"
|
||||
size_bytes: 5027784224
|
||||
source_repo: "bartowski/Qwen_Qwen3-8B-GGUF"
|
||||
|
||||
# --- deploy-vllm role: vllm_models override (t_r1d32b_swap, 2026-09-01) -----
|
||||
# Ansible's hash_behaviour is "replace" (see ansible.cfg) — a host_vars list
|
||||
# variable REPLACES the role default list wholesale, it does not deep-merge.
|
||||
#
|
||||
# SWAP (Ryan direction, 2026-09-01): Qwen2.5-32B-Instruct-AWQ retired,
|
||||
# replaced with DeepSeek-R1-Distill-Qwen-32B-AWQ, max_model_len=32768.
|
||||
# "Single model only" — nomic-embed-text-v1.5 (embedding, :8020) and
|
||||
# Qwen3-8B-AWQ (aux, :8010, already disabled) are BOTH disabled here.
|
||||
# DeepSeek gets the full 24GB card to itself. Nothing in production
|
||||
# consumed nomic-embed at the time of this swap (Hindsight uses its own
|
||||
# bundled 384-dim embedder; OpenViking pointed at the old llama-swap
|
||||
# endpoint, already stopped) — confirmed with Ryan before disabling.
|
||||
#
|
||||
# Model choice: casperhansen/deepseek-r1-distill-qwen-32b-awq — same
|
||||
# quantizer/toolchain (AutoAWQ) as the outgoing Qwen2.5-32B-Instruct-AWQ,
|
||||
# widely used, 4-bit GEMM AWQ, ~19.3GB on disk (4 safetensors shards).
|
||||
# Architecture: Qwen2ForCausalLM (DeepSeek-R1 distilled onto Qwen2.5-32B
|
||||
# base) — same vLLM code path as the outgoing model, no new serving
|
||||
# support needed. Native max_position_embeddings=131072; we cap at 32768
|
||||
# per the task's explicit max-model-len requirement.
|
||||
#
|
||||
# VRAM math: ~19.3GB weights (4-bit AWQ) + KV cache at 32768 ctx (GQA,
|
||||
# 8 KV heads, 128 head_dim, 64 layers, fp16 KV by default) ≈ 19.3GB +
|
||||
# ~4GB KV+overhead ≈ 23.3GB — tight but the FULL 24GB card is now
|
||||
# available (no co-resident nomic-embed/Qwen3-8B taking a share, unlike
|
||||
# the outgoing Qwen2.5-32B config). gpu_memory_utilization=0.95 (role
|
||||
# default) + enforce_eager retained as the proven-stable mitigation from
|
||||
# t_e6facb19/t_ca1af9fb (avoids CUDA graph capture VRAM spike; this host's
|
||||
# only validated way to avoid crash-loop-to-stabilize behavior on this
|
||||
# card). If 0.95 OOMs at 32768 ctx once tested live, drop to 0.90 next
|
||||
# (documented fallback, same pattern as the outgoing model).
|
||||
#
|
||||
# DeepSeek-R1 output note: reasoning traces stream in <think> tags before
|
||||
# the final answer — this is expected R1-distill behavior, not a bug.
|
||||
# Model card recommends temperature 0.5-0.7 (not 0, not vLLM's greedy
|
||||
# default) to avoid repetition/incoherence; not set here (server-side
|
||||
# default), left to be set client-side per the model card's guidance —
|
||||
# flagging for whoever wires this into Hermes profile configs next.
|
||||
vllm_models:
|
||||
- id: "Gemma-4-26B-A4B-it-AWQ"
|
||||
hf_repo: "cyankiwi/gemma-4-26B-A4B-it-AWQ-4bit"
|
||||
role: primary
|
||||
# NO quantization field set (unlike the AutoAWQ-quantized DeepSeek/
|
||||
# Qwen2.5 models above) — live test (2026-09-01) found this repo's
|
||||
# config.json declares quant_method: "compressed-tensors" (llm-compressor
|
||||
# tool output, not classic AutoAWQ), even though the repo name says
|
||||
# "AWQ-4bit". Passing --quantization awq explicitly caused a hard
|
||||
# pydantic ValidationError at every single startup attempt: "Quantization
|
||||
# method specified in the model config (compressed-tensors) does not
|
||||
# match the quantization method specified in the `quantization` argument
|
||||
# (awq)." vLLM auto-detects the quant method correctly from the model's
|
||||
# own config.json when --quantization is omitted — confirmed fix, clean
|
||||
# start. Lesson: don't trust a HF repo's naming convention ("...-AWQ...")
|
||||
# for the `quantization:` field here — check config.json's quant_method.
|
||||
port: 8000
|
||||
# Ryan direction (2026-09-01, t_gemma4_swap): DeepSeek-R1-Distill-Qwen-32B
|
||||
# retired after confirming its `auto` tool-choice reliability is a known,
|
||||
# documented DeepSeek-R1-distillation limitation (trained on pure
|
||||
# reasoning traces, no function-calling data — GitHub-confirmed upstream,
|
||||
# not a vLLM config gap). Replaced with Gemma 4 26B A4B (Google,
|
||||
# Apache 2.0, US-origin — matches Ryan's standing model-origin
|
||||
# preference, unlike Qwen/DeepSeek). Chose MoE (26B A4B, 3.8B active)
|
||||
# over the dense 31B variant: ~3.7GB smaller on-disk AWQ footprint
|
||||
# (17.2GB vs 20.9GB) buys more KV-cache headroom on this tight 24GB
|
||||
# card, and decode should be faster (memory-bandwidth-bound on active
|
||||
# params, not total params). Tradeoff accepted: MoE scores lower than
|
||||
# dense on the Tau2 tool-use benchmark (68.2% vs 76.9%) but still beats
|
||||
# every other size in the family except the 31B on most reasoning
|
||||
# benchmarks. Model choice: cyankiwi/gemma-4-26B-A4B-it-AWQ-4bit —
|
||||
# AutoAWQ 4-bit group_size=32, MoE expert layers (gate/up/down/router)
|
||||
# explicitly excluded from quantization ("ignore" list in config.json)
|
||||
# per standard llm-compressor MoE quant practice — only the dense
|
||||
# attention/projection layers are 4-bit, experts stay higher precision.
|
||||
# Native architecture: Gemma4ForConditionalGeneration (registered
|
||||
# natively in this host's installed vLLM 0.28.0 — vllm/model_executor/
|
||||
# models/registry.py line 415 — no plugin/trust-remote-code needed).
|
||||
# Native max_position_embeddings: 262144 (256K) — Hermes's 64K floor is
|
||||
# comfortably covered without any context-extension trick.
|
||||
max_model_len: 65536
|
||||
# VRAM math (not yet live-validated — see swap validation log below
|
||||
# once run): AWQ weights ~17.2GB on disk (dense attn 4-bit + MoE
|
||||
# experts higher-precision, per config.json's compressed-tensors
|
||||
# ignore list). Starting the KV cache dtype at int4_per_token_head
|
||||
# from the outset (rather than fp16 -> fp8 -> int4 trial-and-error like
|
||||
# the DeepSeek swap) since that same escalation pattern is expected to
|
||||
# repeat on this VRAM-constrained card for any 20+ GB model at >32K ctx.
|
||||
kv_cache_dtype: int4_per_token_head
|
||||
gpu_memory_utilization: 0.95
|
||||
enforce_eager: true
|
||||
# Native tool-calling + reasoning support (no `hermes` workaround
|
||||
# needed, unlike DeepSeek-R1-Distill): Gemma4EngineToolParser and
|
||||
# Gemma4ParserReasoningAdapter are both registered natively in this
|
||||
# host's vLLM 0.28.0 (vllm/tool_parsers/__init__.py,
|
||||
# vllm/reasoning/__init__.py) — purpose-built for this model's actual
|
||||
# output format, not a same-family approximation.
|
||||
enable_auto_tool_choice: true
|
||||
tool_call_parser: gemma4
|
||||
reasoning_parser: gemma4
|
||||
enabled: true
|
||||
- id: "Qwen3-8B-AWQ"
|
||||
hf_repo: "Qwen/Qwen3-8B-AWQ"
|
||||
role: aux
|
||||
quantization: awq
|
||||
port: 8010
|
||||
max_model_len: 32768
|
||||
gpu_memory_utilization: 0.15
|
||||
enforce_eager: true
|
||||
enabled: false # single-model deployment — see swap note above
|
||||
- id: "nomic-embed-text-v1.5"
|
||||
hf_repo: "nomic-ai/nomic-embed-text-v1.5"
|
||||
role: embedding
|
||||
quantization: none
|
||||
port: 8020
|
||||
max_model_len: 2048
|
||||
gpu_memory_utilization: 0.05
|
||||
trust_remote_code: true
|
||||
enabled: false # single-model deployment — see swap note above
|
||||
|
||||
# --- deploy-vllm role: boot persistence (unchanged) -------------------------
|
||||
# Still permanent/boot-persistent — same policy as the outgoing Qwen2.5-32B
|
||||
# deployment (t_5508360a), just now serving one model instead of two.
|
||||
vllm_service_enabled: true
|
||||
vllm_service_state: started
|
||||
|
||||
|
||||
16
ansible/host_vars/carousel-of-progress/vars.yml
Normal file
16
ansible/host_vars/carousel-of-progress/vars.yml
Normal file
@@ -0,0 +1,16 @@
|
||||
---
|
||||
# ------------------------------------------------------------------------------
|
||||
# FILE: ansible/host_vars/astro_orbiter/vars.yml
|
||||
# HOST: astro-orbiter (10.1.71.130)
|
||||
# ROLE: Ollama inference host with AMD RX 5700 GPU passthrough
|
||||
# ------------------------------------------------------------------------------
|
||||
|
||||
ansible_host: 10.1.71.131
|
||||
ansible_user: wed
|
||||
ansible_become: true
|
||||
|
||||
# LVM root expansion — xlarge template uses sda3 partition, standard VG/LV names
|
||||
common_expand_root_lvm: true
|
||||
common_root_pv: /dev/sda3
|
||||
common_root_vg: ubuntu-vg
|
||||
common_root_lv: ubuntu-lv
|
||||
16
ansible/host_vars/main-street-station/main.yml
Normal file
16
ansible/host_vars/main-street-station/main.yml
Normal file
@@ -0,0 +1,16 @@
|
||||
---
|
||||
# Host-specific vars for main-street-station (JMRI headless server)
|
||||
# LCRR - Lake Country Railroad, Milwaukee Road Oct 1956, HO scale
|
||||
|
||||
# JMRI profile ID — find with: ls ~/.jmri/profiles/ on the old box
|
||||
# Format: <name>.<8-char-hex> e.g. LCRR.3d3f1dfc
|
||||
# TODO: fill in after restoring config from GitHub backup
|
||||
jmri_profile_id: ""
|
||||
|
||||
# USB serial device for NCE command station
|
||||
# Verify after install: ls -la /dev/ttyUSB* /dev/ttyACM*
|
||||
jmri_serial_device: /dev/ttyUSB0
|
||||
|
||||
# Path to JMRI config backup for restore task (leave empty to skip)
|
||||
# Point at a local checkout of the LCRR GitHub repo
|
||||
jmri_config_src: ""
|
||||
10
ansible/host_vars/main-street-station/vars.yml
Normal file
10
ansible/host_vars/main-street-station/vars.yml
Normal file
@@ -0,0 +1,10 @@
|
||||
---
|
||||
# main-street-station — JMRI / LCRR server
|
||||
jmri_profile_id: "Lake_Country_Railroad.3e8b1d4b"
|
||||
jmri_lcrr_repo: "ssh://git@gitea.mk-labs.cloud:2221/rblundon/LCRR.git"
|
||||
jmri_lcrr_branch: "clean-profile"
|
||||
jmri_leviton_email: "{{ leviton_email }}"
|
||||
jmri_leviton_password: "{{ leviton_password }}"
|
||||
jmri_ssh_authorized_key: "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAINnSM/9fO8rz/amqkyoGUzUKNNzzmtSXPwOCr1O9zKNO ansible"
|
||||
jmri_ssh_authorized_keys_extra:
|
||||
- "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIG6HaK4Y21UwPRbAZ986L7I9QnUdyq53114+9kO8X4bL rblundon@laptop"
|
||||
@@ -50,18 +50,38 @@ nextcloud_server:
|
||||
|
||||
semaphore_server:
|
||||
hosts:
|
||||
imagineering:
|
||||
figment:
|
||||
ansible_host: 10.1.71.37
|
||||
ansible_user: wed
|
||||
ansible_become: true
|
||||
|
||||
n8n_server:
|
||||
hosts:
|
||||
tiki-room:
|
||||
|
||||
ollama_server:
|
||||
astro_orbiter:
|
||||
hosts:
|
||||
astro-orbiter:
|
||||
ansible_host: 10.1.71.130
|
||||
|
||||
hermes_server:
|
||||
hosts:
|
||||
carousel-of-progress:
|
||||
ansible_host: 10.1.71.131
|
||||
ansible_user: wed
|
||||
ansible_become: true
|
||||
ansible_ssh_private_key_file: ~/.ssh/ansible
|
||||
|
||||
honcho_server:
|
||||
hosts:
|
||||
lincoln:
|
||||
ansible_host: 10.1.71.132
|
||||
ansible_user: wed
|
||||
ansible_become: true
|
||||
|
||||
jmri_server:
|
||||
hosts:
|
||||
main-street-station:
|
||||
ansible_host: 192.168.10.40
|
||||
ansible_user: wed
|
||||
ansible_become: true
|
||||
|
||||
@@ -73,6 +93,10 @@ papermc_server:
|
||||
dev_servers:
|
||||
hosts:
|
||||
scrim:
|
||||
backstage:
|
||||
ansible_host: 10.1.71.133
|
||||
ansible_user: wed
|
||||
ansible_become: true
|
||||
|
||||
# dhcp_server:
|
||||
# hosts:
|
||||
|
||||
@@ -6,7 +6,8 @@
|
||||
#
|
||||
# 1. Syncs boilerplates/traefik/dynamic/ to lightning-lane
|
||||
# 2. Scans the directory for service configs
|
||||
# 3. Creates CNAME records for each service -> lightning-lane
|
||||
# 3. Extracts all hostnames from Host() rules (supports multi-host)
|
||||
# 4. Creates CNAME records for each hostname -> lightning-lane
|
||||
#
|
||||
# PREREQUISITES:
|
||||
# - Service dynamic config YAML committed to boilerplates/traefik/dynamic/
|
||||
@@ -14,15 +15,6 @@
|
||||
#
|
||||
# USAGE:
|
||||
# ansible-playbook -i inventory.yml playbooks/add_service_route.yml
|
||||
#
|
||||
# ADDING A NEW SERVICE:
|
||||
# 1. Create boilerplates/traefik/dynamic/<service>.yml
|
||||
# 2. Commit and push
|
||||
# 3. Run this playbook
|
||||
#
|
||||
# EXCLUDING FILES:
|
||||
# Files that are not service routes (e.g., default.yml for middleware
|
||||
# definitions) should be added to the exclude_configs list below.
|
||||
# ------------------------------------------------------------------------------
|
||||
|
||||
- name: Sync Traefik routes and ensure DNS records
|
||||
@@ -33,13 +25,12 @@
|
||||
vars:
|
||||
base_domain: "local.mk-labs.cloud"
|
||||
dns_server: "monorail"
|
||||
traefik_host: "10.1.71.35"
|
||||
traefik_host: "lightning-lane.local.mk-labs.cloud"
|
||||
traefik_user: "wed"
|
||||
traefik_dynamic_path: "/opt/docker/traefik/dynamic/"
|
||||
dynamic_config_dir: "{{ playbook_dir }}/../../boilerplates/traefik/dynamic"
|
||||
|
||||
# Files in the dynamic directory that are NOT service routes
|
||||
# (middleware definitions, TLS options, etc.)
|
||||
exclude_configs:
|
||||
- default.yml
|
||||
|
||||
@@ -53,38 +44,52 @@
|
||||
register: sync_result
|
||||
changed_when: "'sending incremental file list' in sync_result.stdout"
|
||||
|
||||
# ── Step 2: Discover service configs ──
|
||||
# ── Step 2: Discover hostnames from Traefik router rules ──
|
||||
- name: Find all dynamic config files
|
||||
ansible.builtin.find:
|
||||
paths: "{{ dynamic_config_dir }}"
|
||||
patterns: "*.yml"
|
||||
register: config_files
|
||||
|
||||
- name: Build service list from config filenames
|
||||
ansible.builtin.set_fact:
|
||||
service_names: >-
|
||||
{{ config_files.files
|
||||
| map(attribute='path')
|
||||
| map('basename')
|
||||
| reject('in', exclude_configs)
|
||||
| map('regex_replace', '\.yml$', '')
|
||||
| list }}
|
||||
- name: Read config files
|
||||
ansible.builtin.slurp:
|
||||
src: "{{ item.path }}"
|
||||
register: slurped_configs
|
||||
loop: "{{ config_files.files }}"
|
||||
when: item.path | basename not in exclude_configs
|
||||
|
||||
- name: Display services to route
|
||||
- name: Extract all hostnames from Host() rules
|
||||
ansible.builtin.set_fact:
|
||||
hostnames: >-
|
||||
{% set hosts = [] -%}
|
||||
{% for result in slurped_configs.results if result.content is defined -%}
|
||||
{% set content = result.content | b64decode -%}
|
||||
{% for match in content | regex_findall('Host\(`([^`]+)`\)') -%}
|
||||
{% for h in match.split(' || ') -%}
|
||||
{% set h = h | regex_replace('`', '') | trim -%}
|
||||
{% if h.endswith('.local.mk-labs.cloud') and h not in hosts -%}
|
||||
{% set _ = hosts.append(h) -%}
|
||||
{% endif -%}
|
||||
{% endfor -%}
|
||||
{% endfor -%}
|
||||
{% endfor -%}
|
||||
{{ hosts | unique | list }}
|
||||
|
||||
- name: Display hostnames to create
|
||||
ansible.builtin.debug:
|
||||
msg: "Services found: {{ service_names }}"
|
||||
msg: "Hostnames found: {{ hostnames }}"
|
||||
|
||||
# ── Step 3: Create DNS CNAME records ──
|
||||
- name: Create DNS CNAME record for each service
|
||||
- name: Create DNS CNAME record for each hostname
|
||||
effectivelywild.technitium_dns.technitium_dns_add_record:
|
||||
api_url: "http://{{ dns_server }}.{{ base_domain }}"
|
||||
api_token: "{{ vault_technitium_api_key }}"
|
||||
zone: "{{ base_domain }}"
|
||||
name: "{{ item }}.{{ base_domain }}"
|
||||
name: "{{ item }}"
|
||||
type: "CNAME"
|
||||
cname: "lightning-lane.{{ base_domain }}"
|
||||
ttl: 360
|
||||
validate_certs: false
|
||||
loop: "{{ service_names }}"
|
||||
loop: "{{ hostnames }}"
|
||||
loop_control:
|
||||
label: "{{ item }}.{{ base_domain }}"
|
||||
label: "{{ item }}"
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
- name: Apply common role
|
||||
- name: Apply day0 baseline
|
||||
hosts: "{{ target | default('all') }}"
|
||||
become: true
|
||||
roles:
|
||||
- common
|
||||
- day0-baseline
|
||||
32
ansible/playbooks/day0_expand_root_lv.yml
Normal file
32
ansible/playbooks/day0_expand_root_lv.yml
Normal file
@@ -0,0 +1,32 @@
|
||||
---
|
||||
# ============================================================================
|
||||
# day0_expand_root_lv.yml
|
||||
# ----------------------------------------------------------------------------
|
||||
# Reclaims unallocated PE on the root volume group, extending the root LV
|
||||
# to fill the VG and resizing the underlying filesystem (ext4 or xfs).
|
||||
#
|
||||
# Belongs to the day0 host-provisioning lifecycle. The Ubuntu Server
|
||||
# autoinstall template ships with the root LV at ~half the disk size by
|
||||
# default; this playbook is the canonical one-shot fix-up for that.
|
||||
#
|
||||
# Idempotent and safe to re-run. Hosts without LVM are no-op'd cleanly.
|
||||
#
|
||||
# Opt-out: set `expand_root_lv_skip: true` in host_vars/<host>.yml for
|
||||
# hosts where free PE should NOT be claimed by root (e.g. hosts with a
|
||||
# planned second LV in the same VG for application data).
|
||||
#
|
||||
# Usage:
|
||||
# ansible-playbook playbooks/day0_expand_root_lv.yml
|
||||
# ansible-playbook playbooks/day0_expand_root_lv.yml -e target=lincoln
|
||||
# ansible-playbook playbooks/day0_expand_root_lv.yml -e target=honcho_server
|
||||
# ============================================================================
|
||||
|
||||
- name: Expand root logical volume to fill VG
|
||||
hosts: "{{ target | default('all') }}"
|
||||
become: true
|
||||
gather_facts: true
|
||||
tasks:
|
||||
- name: Apply expand_root_lv role unless host opts out
|
||||
ansible.builtin.include_role:
|
||||
name: expand_root_lv
|
||||
when: not (expand_root_lv_skip | default(false) | bool)
|
||||
23
ansible/playbooks/day0_linux_baseline.yml
Normal file
23
ansible/playbooks/day0_linux_baseline.yml
Normal file
@@ -0,0 +1,23 @@
|
||||
---
|
||||
# ============================================================================
|
||||
# day0_linux_baseline.yml
|
||||
# ----------------------------------------------------------------------------
|
||||
# Applies the mk-labs Linux baseline (linux-baseline role) to one or more
|
||||
# hosts. Idempotent and safe to re-run.
|
||||
#
|
||||
# Usage:
|
||||
# ansible-playbook playbooks/day0_linux_baseline.yml
|
||||
# ansible-playbook playbooks/day0_linux_baseline.yml -e target=figment
|
||||
# ansible-playbook playbooks/day0_linux_baseline.yml -e target=semaphore_server
|
||||
#
|
||||
# To trigger an opt-in full system upgrade:
|
||||
# ansible-playbook playbooks/day0_linux_baseline.yml \
|
||||
# -e target=figment -e 'baseline_features={"full_upgrade": true}'
|
||||
# ============================================================================
|
||||
|
||||
- name: Apply mk-labs Linux baseline
|
||||
hosts: "{{ target | default('all') }}"
|
||||
become: true
|
||||
gather_facts: true
|
||||
roles:
|
||||
- linux-baseline
|
||||
30
ansible/playbooks/day0_provision.yml
Normal file
30
ansible/playbooks/day0_provision.yml
Normal file
@@ -0,0 +1,30 @@
|
||||
---
|
||||
# ============================================================================
|
||||
# day0_provision.yml
|
||||
# ----------------------------------------------------------------------------
|
||||
# Umbrella day0 playbook. Runs the full host-provisioning lifecycle in
|
||||
# the correct order against newly-built VMs, so the operator runs ONE
|
||||
# command per new host rather than chaining day0 steps manually.
|
||||
#
|
||||
# Order matters:
|
||||
# 1. linux-baseline — timezone, NTP, packages, SSH hardening, jarvis user
|
||||
# 2. expand_root_lv — reclaim PE left unallocated by the Ubuntu
|
||||
# autoinstall template default
|
||||
#
|
||||
# Idempotent: every step is safe to re-run. Suitable to apply periodically
|
||||
# from Semaphore as a baseline-drift check.
|
||||
#
|
||||
# Usage:
|
||||
# ansible-playbook playbooks/day0_provision.yml -e target=lincoln
|
||||
# ansible-playbook playbooks/day0_provision.yml -e target=honcho_server
|
||||
#
|
||||
# For finer control over a single phase, the constituent playbooks are:
|
||||
# playbooks/day0_linux_baseline.yml
|
||||
# playbooks/day0_expand_root_lv.yml
|
||||
# ============================================================================
|
||||
|
||||
- name: Import day0 linux baseline
|
||||
ansible.builtin.import_playbook: day0_linux_baseline.yml
|
||||
|
||||
- name: Import day0 expand root LV
|
||||
ansible.builtin.import_playbook: day0_expand_root_lv.yml
|
||||
79
ansible/playbooks/day1_deploy_hermes.yml
Normal file
79
ansible/playbooks/day1_deploy_hermes.yml
Normal file
@@ -0,0 +1,79 @@
|
||||
---
|
||||
# =============================================================================
|
||||
# day1_deploy_hermes.yml
|
||||
# Deploy Hermes Agent (Nous Research) on carousel-of-progress (10.1.71.131)
|
||||
#
|
||||
# FIRST-RUN WORKFLOW:
|
||||
# 1. Run this playbook:
|
||||
# ansible-playbook playbooks/day1_deploy_hermes.yml
|
||||
#
|
||||
# 2. SSH to the host and run the setup wizard as the hermes user:
|
||||
# ssh wed@carousel-of-progress.local.mk-labs.cloud
|
||||
# sudo -u hermes hermes setup
|
||||
#
|
||||
# 3. Once configured, start and verify the service:
|
||||
# sudo systemctl start hermes
|
||||
# sudo systemctl status hermes
|
||||
# sudo journalctl -u hermes -f
|
||||
#
|
||||
# VARIABLES:
|
||||
# hermes_skip_browser: true — set to skip Playwright/Chromium install
|
||||
# (saves ~300MB if browser automation not needed)
|
||||
# =============================================================================
|
||||
|
||||
- name: Deploy Hermes Agent on carousel-of-progress
|
||||
hosts: carousel-of-progress
|
||||
gather_facts: true
|
||||
|
||||
pre_tasks:
|
||||
- name: Verify target is carousel-of-progress
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- inventory_hostname == "carousel-of-progress"
|
||||
fail_msg: >
|
||||
This playbook is scoped to carousel-of-progress only.
|
||||
Got: {{ inventory_hostname }}
|
||||
|
||||
- name: Confirm OS is Ubuntu
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- ansible_distribution == "Ubuntu"
|
||||
fail_msg: >
|
||||
This playbook requires Ubuntu. Found: {{ ansible_distribution }}.
|
||||
(If running Fedora, swap apt tasks for dnf and adjust Playwright deps.)
|
||||
|
||||
roles:
|
||||
- role: hermes
|
||||
vars:
|
||||
hermes_skip_browser: false # set true to skip Chromium install
|
||||
|
||||
post_tasks:
|
||||
- name: Verify hermes binary is accessible system-wide
|
||||
ansible.builtin.command: hermes --version
|
||||
register: hermes_version_check
|
||||
changed_when: false
|
||||
failed_when: hermes_version_check.rc != 0
|
||||
|
||||
- name: Print hermes version
|
||||
ansible.builtin.debug:
|
||||
msg: "{{ hermes_version_check.stdout }}"
|
||||
|
||||
- name: Print post-install instructions
|
||||
ansible.builtin.debug:
|
||||
msg:
|
||||
- "============================================================"
|
||||
- "Hermes installed on carousel-of-progress (10.1.71.131)"
|
||||
- "============================================================"
|
||||
- "Next steps:"
|
||||
- " 1. SSH to the host:"
|
||||
- " ssh wed@carousel-of-progress.local.mk-labs.cloud"
|
||||
- " 2. Run the setup wizard as the hermes user:"
|
||||
- " sudo -u hermes hermes setup"
|
||||
- " 3. After config, start the service:"
|
||||
- " sudo systemctl start hermes"
|
||||
- " 4. Verify:"
|
||||
- " sudo systemctl status hermes"
|
||||
- " sudo journalctl -u hermes -f"
|
||||
- "============================================================"
|
||||
- "Service is ENABLED but NOT STARTED — config required first."
|
||||
- "============================================================"
|
||||
18
ansible/playbooks/day1_deploy_honcho.yml
Normal file
18
ansible/playbooks/day1_deploy_honcho.yml
Normal file
@@ -0,0 +1,18 @@
|
||||
---
|
||||
# ============================================================================
|
||||
# day1_deploy_honcho.yml
|
||||
# ----------------------------------------------------------------------------
|
||||
# Deploys Honcho + pgvector PostgreSQL on the `lincoln` host. Assumes day0
|
||||
# host provisioning (linux-baseline + expand_root_lv) is already complete.
|
||||
#
|
||||
# Run via:
|
||||
# ansible-playbook -i inventory.yml playbooks/day0_provision.yml -e target=lincoln
|
||||
# ansible-playbook -i inventory.yml playbooks/day1_deploy_honcho.yml
|
||||
# ============================================================================
|
||||
|
||||
- name: Deploy Honcho on lincoln
|
||||
hosts: honcho_server
|
||||
become: true
|
||||
gather_facts: true
|
||||
roles:
|
||||
- honcho
|
||||
24
ansible/playbooks/day1_deploy_jmri.yml
Normal file
24
ansible/playbooks/day1_deploy_jmri.yml
Normal file
@@ -0,0 +1,24 @@
|
||||
---
|
||||
# ============================================================================
|
||||
# day1_deploy_jmri.yml
|
||||
# ----------------------------------------------------------------------------
|
||||
# Deploys JMRI JmriFaceless headless server on main-street-station.
|
||||
# Applies linux-baseline first, then the jmri role.
|
||||
#
|
||||
# Usage:
|
||||
# ansible-playbook playbooks/day1_deploy_jmri.yml
|
||||
# ansible-playbook playbooks/day1_deploy_jmri.yml -e target=main-street-station
|
||||
#
|
||||
# Prerequisites:
|
||||
# 1. Host is in inventory under jmri_server group
|
||||
# 2. jmri_profile_id is set in host_vars/main-street-station.yml
|
||||
# 3. SSH access as 'wed' with sudo
|
||||
# ============================================================================
|
||||
|
||||
- name: Deploy JMRI headless server
|
||||
hosts: "{{ target | default('jmri_server') }}"
|
||||
become: true
|
||||
gather_facts: true
|
||||
roles:
|
||||
- linux-baseline
|
||||
- jmri
|
||||
25
ansible/playbooks/day1_deploy_llm_inference.yml
Normal file
25
ansible/playbooks/day1_deploy_llm_inference.yml
Normal file
@@ -0,0 +1,25 @@
|
||||
---
|
||||
# ------------------------------------------------------------------------------
|
||||
# FILE: playbooks/day1_deploy_llm_inference.yml
|
||||
# DESCRIPTION: Day 1 playbook for astro-orbiter LLM inference stack.
|
||||
# Deploys vLLM + Gemma 2 27B on RTX 3090 via OCuLink.
|
||||
#
|
||||
# Usage:
|
||||
# cd ~/git/homelab/ansible
|
||||
# ansible-playbook -i inventory.yml playbooks/day1_deploy_llm_inference.yml
|
||||
#
|
||||
# Phases (added incrementally — safe to re-run):
|
||||
# 1. Foundation — groups, directories, vault assertion
|
||||
# 2. Driver — nvidia-driver-595-open (idempotent; already installed)
|
||||
# 3. vLLM — Python venv + pip install vllm
|
||||
# 4. Model — HF login, Gemma 2 27B snapshot_download
|
||||
# 5. Serve — systemd vllm-serve.service, health check
|
||||
# 6. Integration — Hermes provider config on carousel
|
||||
# ------------------------------------------------------------------------------
|
||||
|
||||
- name: Deploy LLM inference stack on astro-orbiter
|
||||
hosts: astro_orbiter
|
||||
gather_facts: true
|
||||
|
||||
roles:
|
||||
- role: llm-inference
|
||||
35
ansible/playbooks/day1_deploy_llm_inference_multimodel.yml
Normal file
35
ansible/playbooks/day1_deploy_llm_inference_multimodel.yml
Normal file
@@ -0,0 +1,35 @@
|
||||
---
|
||||
# ------------------------------------------------------------------------------
|
||||
# FILE: playbooks/day1_deploy_llm_inference_multimodel.yml
|
||||
# DESCRIPTION: Day 1 playbook for the dual-model (aux + tool-calling) rollout
|
||||
# on astro-orbiter. Builds on roles/llm-inference (CUDA/driver
|
||||
# already done) — does not replace it.
|
||||
#
|
||||
# Usage:
|
||||
# cd ~/git/homelab/ansible
|
||||
# ansible-playbook -i inventory.yml playbooks/day1_deploy_llm_inference_multimodel.yml
|
||||
# # or scope to specific phases:
|
||||
# ansible-playbook -i inventory.yml playbooks/day1_deploy_llm_inference_multimodel.yml --tags discover
|
||||
#
|
||||
# EXECUTION CHANNEL (2026-08-12, War Machine): run via the Semaphore template
|
||||
# "llm_inference_multimodel_stage_models" (scoped to --tags models). Do NOT
|
||||
# run this via direct ansible-playbook or ad-hoc ssh/curl/systemctl — all
|
||||
# homelab inference changes go through Ansible roles executed by Semaphore for
|
||||
# audit/visibility. Phase 1 (models) is idempotent: it only downloads/stages a
|
||||
# GGUF when missing or size-mismatched, and only restarts the router when a new
|
||||
# GGUF is detected (normal re-runs that find the files correct touch nothing).
|
||||
#
|
||||
# Phases (see roles/llm-inference-multimodel/README.md for detail):
|
||||
# 0. discover — read-only; confirm existing Gemma service management
|
||||
# 1. models — idempotent GGUF downloads (Phi-4-14B, Mistral-Small-24B)
|
||||
# 2. systemd — deploy both unit files, do NOT auto-start
|
||||
# 3. firewall — scope ports 8000/8001, non-0.0.0.0 bind
|
||||
# 4. verify — start both services, smoke test, VRAM check
|
||||
# ------------------------------------------------------------------------------
|
||||
|
||||
- name: Deploy dual-model LLM inference stack on astro-orbiter
|
||||
hosts: astro_orbiter
|
||||
gather_facts: true
|
||||
|
||||
roles:
|
||||
- role: llm-inference-multimodel
|
||||
106
ansible/playbooks/day1_deploy_llm_router_shadow.yml
Normal file
106
ansible/playbooks/day1_deploy_llm_router_shadow.yml
Normal file
@@ -0,0 +1,106 @@
|
||||
---
|
||||
# ------------------------------------------------------------------------------
|
||||
# FILE: playbooks/day1_deploy_llm_router_shadow.yml
|
||||
# DESCRIPTION: Deploy llama-server in router mode on a shadow port (8003).
|
||||
#
|
||||
# This playbook deploys and validates the llama.cpp router mode supervisor on
|
||||
# astro-orbiter (10.1.71.130) WITHOUT touching the production endpoint
|
||||
# (llama-server-qwen, port 8002). All 7 dependent Hermes profiles
|
||||
# (bruce-banner, groot, happy, heimdall, rocket-raccoon, war-machine, wong)
|
||||
# remain pointing at port 8002 throughout this run.
|
||||
#
|
||||
# Usage (from ~/git/homelab/ansible):
|
||||
# ansible-playbook -i inventory.yml playbooks/day1_deploy_llm_router_shadow.yml
|
||||
#
|
||||
# Tag-scoped runs (if you need to re-run one phase):
|
||||
# ansible-playbook -i inventory.yml playbooks/day1_deploy_llm_router_shadow.yml \
|
||||
# --tags router_systemd,router_firewall,router_verify
|
||||
#
|
||||
# Execution path (Ryan-approved 2026-08-12, task t_0cca74a2):
|
||||
# Direct ansible-playbook as documented exception — Semaphore template for
|
||||
# this role does not exist yet. Create template after cutover is confirmed.
|
||||
# This is the same exception pattern used in prior sessions on this box.
|
||||
#
|
||||
# Pre-requisites:
|
||||
# 1. llama-server binary at /opt/llama.cpp/build/bin/llama-server supports
|
||||
# router mode (confirmed 2026-08-12: --models-dir flag present in --help).
|
||||
# 2. /opt/models/ contains ONLY Qwen3.6-35B-A3B-UD-Q4_K_S.gguf
|
||||
# (confirmed 2026-08-12: directory is clean, Phi-4/Mistral already deleted).
|
||||
# 3. Port 8002 is in use by the production llama-server-qwen service —
|
||||
# this playbook does NOT touch it.
|
||||
#
|
||||
# Validation gates this playbook runs (all hard gates EXCEPT Gate 4):
|
||||
# Gate 1: /v1/models reports Qwen with n_ctx >= 64000 (64K Hermes floor)
|
||||
# Gate 2: Tool-calling probe through router returns finish_reason=tool_calls
|
||||
# Gate 2b: Hallucination stress test does NOT trigger spurious tool_calls
|
||||
# Gate 3: nvidia-smi VRAM <= 23,000 MiB (--models-max 1 confirmed effective)
|
||||
# Gate 4: Bundled SvelteKit UI check (nice-to-have, non-blocking)
|
||||
#
|
||||
# What happens after this playbook:
|
||||
# War Machine posts validation gate results to Ryan.
|
||||
# Ryan reviews and signs off on cutover (or requests changes).
|
||||
# War Machine then runs day2_cutover_qwen_to_router.yml (not yet created)
|
||||
# to promote the router to port 8002 and retire the bare llama-server-qwen.
|
||||
#
|
||||
# Reference: proposal at
|
||||
# ~/friday/system/inbox/agents/war-machine/2026-08-12-qwen-router-mode-proposal.md
|
||||
# Task: t_0cca74a2
|
||||
# Author: War Machine (2026-08-12)
|
||||
# ------------------------------------------------------------------------------
|
||||
|
||||
- name: Deploy llama-server router (shadow, port 8003) on astro-orbiter
|
||||
hosts: astro_orbiter
|
||||
gather_facts: true
|
||||
become: true
|
||||
|
||||
vars:
|
||||
# Enable the router phase — this is the ONLY var that makes router.yml run.
|
||||
# Default in defaults/main.yml is false (no-op). Flip here for the shadow run.
|
||||
llm_router_enabled: true
|
||||
|
||||
# Qwen model ID as it appears in /v1/models from the router.
|
||||
# llama-server router uses the GGUF filename (without .gguf) as the model id.
|
||||
llm_router_expected_model_id: "Qwen3.6-35B-A3B-UD-Q4_K_S"
|
||||
|
||||
roles:
|
||||
- role: llm-inference-multimodel
|
||||
|
||||
# No --tags needed here: router.yml is included dynamically from main.yml
|
||||
# whenever llm_router_enabled: true. The full role runs but the
|
||||
# discover/models/systemd/verify phases are gated on their own vars
|
||||
# (llm_qwen_service_enabled etc.) and are idempotent. The stale
|
||||
# models.yml (Phi-4/Mistral download tasks) uses variables no longer
|
||||
# defined — a follow-up cleanup task should update that file.
|
||||
|
||||
- name: "POST-VALIDATION SAFETY NET — ensure production service is running"
|
||||
hosts: astro_orbiter
|
||||
gather_facts: false
|
||||
become: true
|
||||
|
||||
tasks:
|
||||
# Always run this, regardless of whether the validation play succeeded.
|
||||
# If the router.yml play stopped llama-server-qwen for VRAM validation
|
||||
# and then a gate failed (play aborted), this play ensures it comes back up.
|
||||
- name: "Ensure llama-server-qwen (port 8002) is running after validation (always)"
|
||||
ansible.builtin.systemd:
|
||||
name: llama-server-qwen
|
||||
state: started
|
||||
enabled: true
|
||||
ignore_errors: true # don't fail if the unit doesn't exist
|
||||
|
||||
- name: "Verify production /health after safety-net restart"
|
||||
ansible.builtin.uri:
|
||||
url: "http://10.1.71.130:8002/health"
|
||||
status_code: 200
|
||||
timeout: 30
|
||||
register: llm_safety_net_health
|
||||
failed_when: false
|
||||
ignore_errors: true
|
||||
|
||||
- name: "Report production status (safety-net check)"
|
||||
ansible.builtin.debug:
|
||||
msg: >-
|
||||
Safety-net: llama-server-qwen :8002 health check returned
|
||||
{{ llm_safety_net_health.status | default('UNREACHABLE') }}.
|
||||
{{ 'OK — production is up.' if (llm_safety_net_health.status | default(0) | int == 200)
|
||||
else 'WARNING — production may not be healthy. Check manually.' }}
|
||||
@@ -1,22 +1,17 @@
|
||||
---
|
||||
# ------------------------------------------------------------------------------
|
||||
# FILE: playbooks/day1_deploy_semaphore.yml
|
||||
# DESCRIPTION: Deploys Semaphore on imagineering
|
||||
# Runs: common → docker-host → semaphore
|
||||
# ============================================================================
|
||||
# day1_deploy_semaphore.yml
|
||||
# ----------------------------------------------------------------------------
|
||||
# Deploys SemaphoreUI + PostgreSQL on the imagineering host (figment).
|
||||
# Run AFTER day0_linux_baseline.yml has been applied to the target.
|
||||
#
|
||||
# USAGE:
|
||||
# Usage:
|
||||
# ansible-playbook -i inventory.yml playbooks/day1_deploy_semaphore.yml
|
||||
#
|
||||
# SECRETS REQUIRED IN VAULT (group_vars/all/vault):
|
||||
# vault_semaphore_database_password
|
||||
# vault_semaphore_admin_password
|
||||
# vault_semaphore_access_key_encryption
|
||||
# ------------------------------------------------------------------------------
|
||||
# ============================================================================
|
||||
|
||||
- name: Deploy Semaphore
|
||||
- name: Deploy SemaphoreUI on imagineering
|
||||
hosts: semaphore_server
|
||||
become: true
|
||||
|
||||
gather_facts: true
|
||||
roles:
|
||||
- docker-host
|
||||
- semaphore
|
||||
|
||||
18
ansible/playbooks/day1_deploy_vllm.yml
Normal file
18
ansible/playbooks/day1_deploy_vllm.yml
Normal file
@@ -0,0 +1,18 @@
|
||||
# ------------------------------------------------------------------------------
|
||||
# FILE: playbooks/day1_deploy_vllm.yml
|
||||
# Deploy vLLM to a target host via roles/deploy-vllm.
|
||||
#
|
||||
# Staging run (deploy + validate WITHOUT touching production traffic):
|
||||
# ansible-playbook -i inventory.yml playbooks/day1_deploy_vllm.yml --limit astro-orbiter
|
||||
#
|
||||
# Cutover run (once staging is validated and Ryan/JARVIS approve flipping
|
||||
# traffic — starts and enables the systemd unit(s), runs Phase 5 verification):
|
||||
# ansible-playbook -i inventory.yml playbooks/day1_deploy_vllm.yml \
|
||||
# --limit astro-orbiter --extra-vars "vllm_service_state=started"
|
||||
# ------------------------------------------------------------------------------
|
||||
- name: Deploy vLLM inference serving stack
|
||||
hosts: astro-orbiter
|
||||
become: false
|
||||
gather_facts: true
|
||||
roles:
|
||||
- deploy-vllm
|
||||
259
ansible/playbooks/day2_add_coder_alias.yml
Normal file
259
ansible/playbooks/day2_add_coder_alias.yml
Normal file
@@ -0,0 +1,259 @@
|
||||
---
|
||||
# ------------------------------------------------------------------------------
|
||||
# FILE: playbooks/day2_add_coder_alias.yml
|
||||
# DESCRIPTION: Add Qwen2.5-Coder-14B-Instruct-Q4_K_M to the llama-server-router
|
||||
# on astro-orbiter (10.1.71.130:8002).
|
||||
#
|
||||
# Context (t_55c164f5, 2026-08-13):
|
||||
# Ryan requested a Qwen2.5-Coder-14B-Instruct-Q4_K_M.gguf be added to the
|
||||
# astro-orbiter router with:
|
||||
# alias = "Qwen2.5-Coder-14B-Instruct-4bit"
|
||||
# n_gpu_layers = 99
|
||||
# ctx_size = 16384
|
||||
# flash_attn = true
|
||||
# Deployed GitOps-style via this role; no hand-editing of the live preset.
|
||||
#
|
||||
# What this playbook does:
|
||||
# 1. Downloads Qwen2.5-Coder-14B-Instruct-Q4_K_M.gguf into /opt/models if
|
||||
# not already present (idempotent: size-check guard, no re-pull on match).
|
||||
# 2. Redeploys the preset INI (adding the [Qwen2.5-Coder-14B-Instruct-Q4_K_M]
|
||||
# section with alias = Qwen2.5-Coder-14B-Instruct-4bit).
|
||||
# 3. Restarts llama-server-router to pick up the new model entry.
|
||||
# 4. Verifies /v1/models returns all 4 models including the new Coder entry.
|
||||
#
|
||||
# VRAM context note (t_55c164f5):
|
||||
# Qwen2.5-Coder-14B Q4_K_M: ~8.4GB weights + ~0.6GB KV @ 16K ctx ≈ 9.0GB
|
||||
# Qwen3.6-35B-A3B: ~21.5GB
|
||||
# Full co-residency is impossible on 24GB. LRU eviction handles this:
|
||||
# when Coder is requested, Qwen3.6-35B is evicted (and vice versa).
|
||||
# Model-switching incurs ~30-60s cold-load latency — expected and acceptable.
|
||||
# Phi (~4.3GB) or Llama (~5.6GB) can co-reside with Coder (total ~14GB).
|
||||
#
|
||||
# Usage (from ~/git/homelab/ansible):
|
||||
# env -u ANSIBLE_VAULT_PASSWORD_FILE ansible-playbook -i inventory.yml \
|
||||
# playbooks/day2_add_coder_alias.yml
|
||||
#
|
||||
# Semaphore note: Semaphore SSH key for jarvis user is not loaded in the
|
||||
# container (known pitfall, homelab-llm-serving skill). Run via CLI with
|
||||
# id_jarvis key; document as exception per Ryan's standing CLI fallback directive.
|
||||
#
|
||||
# Author: War Machine (2026-08-13, t_55c164f5)
|
||||
# ------------------------------------------------------------------------------
|
||||
|
||||
- name: "Add Qwen2.5-Coder-14B-Instruct-4bit alias to astro-orbiter router"
|
||||
hosts: astro_orbiter
|
||||
gather_facts: false
|
||||
become: true
|
||||
|
||||
vars:
|
||||
# Activate preset mode
|
||||
llm_router_preset_enabled: true
|
||||
llm_router_preset_path: /opt/llama-server-router-preset.ini
|
||||
|
||||
# Production port (router is on 8002 since t_cd0d5388)
|
||||
llm_router_port: 8002
|
||||
|
||||
# Per-model ctx-size settings (carried from t_ryan_per_model_ctx; Coder new)
|
||||
llm_router_llama_ctx_size: 8192
|
||||
llm_router_llama_flash_attn: "true"
|
||||
llm_router_phi_ctx_size: 32768
|
||||
llm_router_phi_flash_attn: "true"
|
||||
llm_router_coder_ctx_size: 16384
|
||||
llm_router_coder_flash_attn: "true"
|
||||
|
||||
# All other vars inherit from host_vars + defaults/main.yml.
|
||||
# Explicitly set the ones needed by the unit/template tasks for clarity:
|
||||
llm_router_enabled: true
|
||||
llm_service_user: jarvis
|
||||
llm_binary_path: /opt/llama.cpp/build/bin/llama-server
|
||||
llm_models_dir: /opt/models
|
||||
llm_bind_address: "10.1.71.130"
|
||||
llm_allowed_source_cidr: "10.1.70.0/24"
|
||||
llm_router_service_name: llama-server-router
|
||||
llm_router_bind_address: "10.1.71.130"
|
||||
llm_router_allowed_source_cidr: "10.1.70.0/24"
|
||||
llm_router_models_dir: /opt/models
|
||||
llm_router_models_max: 4 # from host_vars; bumped by t_33acbb2e
|
||||
llm_router_ctx_size: 65536 # Qwen3.6-35B default; per-model overrides above
|
||||
llm_router_parallel: 1
|
||||
llm_router_gpu_layers: 99
|
||||
llm_router_batch_size: 2048
|
||||
llm_router_ubatch_size: 512
|
||||
llm_router_cache_type_k: q4_0
|
||||
llm_router_cache_type_v: q4_0
|
||||
llm_router_flash_attn: "auto"
|
||||
llm_router_expected_model_id: "Qwen3.6-35B-A3B-UD-Q4_K_S"
|
||||
llm_router_vram_max_mib: 23000
|
||||
|
||||
# Coder model staging entry (used below)
|
||||
coder_filename: "Qwen2.5-Coder-14B-Instruct-Q4_K_M.gguf"
|
||||
coder_url: "https://huggingface.co/bartowski/Qwen2.5-Coder-14B-Instruct-GGUF/resolve/main/Qwen2.5-Coder-14B-Instruct-Q4_K_M.gguf"
|
||||
coder_size_bytes: 8988111072
|
||||
|
||||
handlers:
|
||||
- name: reload systemd
|
||||
ansible.builtin.systemd:
|
||||
daemon_reload: true
|
||||
become: true
|
||||
listen: "reload systemd"
|
||||
|
||||
- name: restart router
|
||||
ansible.builtin.systemd:
|
||||
name: llama-server-router
|
||||
state: restarted
|
||||
become: true
|
||||
listen: "restart router"
|
||||
|
||||
tasks:
|
||||
|
||||
# ==========================================================================
|
||||
# PHASE 1: Download Coder GGUF if not present / size mismatch
|
||||
# ==========================================================================
|
||||
|
||||
- name: "[coder] Stat existing GGUF"
|
||||
ansible.builtin.stat:
|
||||
path: "{{ llm_models_dir }}/{{ coder_filename }}"
|
||||
get_checksum: false
|
||||
register: coder_stat
|
||||
|
||||
- name: "[coder] Download GGUF (skip if present and size matches)"
|
||||
ansible.builtin.get_url:
|
||||
url: "{{ coder_url }}"
|
||||
dest: "{{ llm_models_dir }}/{{ coder_filename }}"
|
||||
owner: "{{ llm_service_user }}"
|
||||
group: "{{ llm_service_user }}"
|
||||
mode: "0644"
|
||||
timeout: 3600
|
||||
when: >
|
||||
not coder_stat.stat.exists or
|
||||
coder_stat.stat.size != coder_size_bytes
|
||||
register: coder_download
|
||||
notify: restart router
|
||||
|
||||
- name: "[coder] Confirm GGUF size post-download"
|
||||
ansible.builtin.stat:
|
||||
path: "{{ llm_models_dir }}/{{ coder_filename }}"
|
||||
get_checksum: false
|
||||
register: coder_stat_post
|
||||
|
||||
- name: "[coder] FAIL if GGUF size mismatch after download"
|
||||
ansible.builtin.fail:
|
||||
msg: >-
|
||||
GGUF size mismatch: expected {{ coder_size_bytes }} bytes,
|
||||
got {{ coder_stat_post.stat.size }} bytes.
|
||||
Re-download may be needed.
|
||||
when: coder_stat_post.stat.size != coder_size_bytes
|
||||
|
||||
# ==========================================================================
|
||||
# PHASE 2: Deploy updated preset INI (adds Coder section)
|
||||
# ==========================================================================
|
||||
|
||||
- name: "[coder] Deploy preset INI to {{ llm_router_preset_path }}"
|
||||
ansible.builtin.template:
|
||||
src: "../roles/llm-inference-multimodel/templates/llama-server-router-preset.ini.j2"
|
||||
dest: "{{ llm_router_preset_path }}"
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
register: coder_preset_deployed
|
||||
notify: restart router
|
||||
|
||||
# ==========================================================================
|
||||
# PHASE 3: Redeploy systemd unit (unchanged flags, but ensures unit is fresh)
|
||||
# ==========================================================================
|
||||
|
||||
- name: "[coder] Deploy llama-server-router unit"
|
||||
ansible.builtin.template:
|
||||
src: "../roles/llm-inference-multimodel/templates/llama-server-router.service.j2"
|
||||
dest: /etc/systemd/system/llama-server-router.service
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
register: coder_unit_deployed
|
||||
notify:
|
||||
- reload systemd
|
||||
- restart router
|
||||
|
||||
- name: "[coder] Flush handlers (daemon-reload + router restart)"
|
||||
ansible.builtin.meta: flush_handlers
|
||||
|
||||
# ==========================================================================
|
||||
# PHASE 4: Verify router is up and Coder model appears in /v1/models
|
||||
# ==========================================================================
|
||||
|
||||
- name: "[coder] Wait for /health (router supervisor)"
|
||||
ansible.builtin.uri:
|
||||
url: "http://{{ llm_router_bind_address }}:{{ llm_router_port }}/health"
|
||||
status_code: 200
|
||||
timeout: 30
|
||||
retries: 12
|
||||
delay: 5
|
||||
register: coder_health
|
||||
until: coder_health.status == 200
|
||||
|
||||
- name: "[coder] Query /v1/models"
|
||||
ansible.builtin.uri:
|
||||
url: "http://{{ llm_router_bind_address }}:{{ llm_router_port }}/v1/models"
|
||||
status_code: 200
|
||||
return_content: true
|
||||
timeout: 30
|
||||
register: coder_models
|
||||
|
||||
- name: "[coder] Extract model IDs and aliases"
|
||||
ansible.builtin.set_fact:
|
||||
coder_model_ids: "{{ coder_models.json.data | map(attribute='id') | list }}"
|
||||
coder_all_aliases: "{{ coder_models.json.data | map(attribute='aliases') | flatten | list }}"
|
||||
|
||||
- name: "[coder] FAIL if Coder primary ID missing"
|
||||
ansible.builtin.fail:
|
||||
msg: >-
|
||||
'Qwen2.5-Coder-14B-Instruct-Q4_K_M' not in /v1/models.
|
||||
IDs: {{ coder_model_ids }}
|
||||
when: "'Qwen2.5-Coder-14B-Instruct-Q4_K_M' not in coder_model_ids"
|
||||
|
||||
- name: "[coder] FAIL if Coder alias missing"
|
||||
ansible.builtin.fail:
|
||||
msg: >-
|
||||
'Qwen2.5-Coder-14B-Instruct-4bit' not found as ID or alias in /v1/models.
|
||||
IDs: {{ coder_model_ids }}
|
||||
Aliases: {{ coder_all_aliases }}
|
||||
when:
|
||||
- "'Qwen2.5-Coder-14B-Instruct-4bit' not in coder_model_ids"
|
||||
- "'Qwen2.5-Coder-14B-Instruct-4bit' not in coder_all_aliases"
|
||||
|
||||
- name: "[coder] FAIL if Qwen3.6-35B missing"
|
||||
ansible.builtin.fail:
|
||||
msg: "'Qwen3.6-35B-A3B-UD-Q4_K_S' not in /v1/models. IDs: {{ coder_model_ids }}"
|
||||
when: "'Qwen3.6-35B-A3B-UD-Q4_K_S' not in coder_model_ids"
|
||||
|
||||
- name: "[coder] FAIL if Phi missing"
|
||||
ansible.builtin.fail:
|
||||
msg: "'Phi-3.5-mini-instruct-Q8_0' not in /v1/models. IDs: {{ coder_model_ids }}"
|
||||
when: "'Phi-3.5-mini-instruct-Q8_0' not in coder_model_ids"
|
||||
|
||||
- name: "[coder] FAIL if Llama missing"
|
||||
ansible.builtin.fail:
|
||||
msg: "'Meta-Llama-3.1-8B-Instruct-Q4_K_M' not in /v1/models. IDs: {{ coder_model_ids }}"
|
||||
when: "'Meta-Llama-3.1-8B-Instruct-Q4_K_M' not in coder_model_ids"
|
||||
|
||||
- name: "[coder] PASS — full /v1/models summary"
|
||||
ansible.builtin.debug:
|
||||
msg:
|
||||
- "========================================================================"
|
||||
- "QWEN2.5-CODER-14B ALIAS DEPLOYMENT — COMPLETE"
|
||||
- ""
|
||||
- " Mode: --models-preset ({{ llm_router_preset_path }})"
|
||||
- " Service: llama-server-router.service (:{{ llm_router_port }})"
|
||||
- ""
|
||||
- " /v1/models IDs: {{ coder_model_ids }}"
|
||||
- " /v1/models aliases: {{ coder_all_aliases }}"
|
||||
- ""
|
||||
- " VERIFY:"
|
||||
- " Qwen3.6-35B-A3B-UD-Q4_K_S: {{ 'PRESENT' if 'Qwen3.6-35B-A3B-UD-Q4_K_S' in coder_model_ids else 'MISSING' }}"
|
||||
- " Phi-3.5-mini-instruct-Q8_0: {{ 'PRESENT' if 'Phi-3.5-mini-instruct-Q8_0' in coder_model_ids else 'MISSING' }}"
|
||||
- " Meta-Llama-3.1-8B-Instruct-Q4_K_M: {{ 'PRESENT' if 'Meta-Llama-3.1-8B-Instruct-Q4_K_M' in coder_model_ids else 'MISSING' }}"
|
||||
- " Qwen2.5-Coder-14B-Instruct-Q4_K_M: {{ 'PRESENT' if 'Qwen2.5-Coder-14B-Instruct-Q4_K_M' in coder_model_ids else 'MISSING' }}"
|
||||
- " Qwen2.5-Coder-14B-Instruct-4bit: {{ 'PRESENT (ID)' if 'Qwen2.5-Coder-14B-Instruct-4bit' in coder_model_ids else ('PRESENT (alias)' if 'Qwen2.5-Coder-14B-Instruct-4bit' in coder_all_aliases else 'MISSING') }}"
|
||||
- ""
|
||||
- " GGUF download: {{ 'NEW DOWNLOAD' if (coder_download is defined and coder_download.changed) else 'ALREADY PRESENT (skipped)' }}"
|
||||
- "========================================================================"
|
||||
313
ansible/playbooks/day2_add_nomic_embed.yml
Normal file
313
ansible/playbooks/day2_add_nomic_embed.yml
Normal file
@@ -0,0 +1,313 @@
|
||||
---
|
||||
# ------------------------------------------------------------------------------
|
||||
# FILE: playbooks/day2_add_nomic_embed.yml
|
||||
# DESCRIPTION: Add nomic-embed-text-v1.5-Q4_K_M to the llama-server-router
|
||||
# on astro-orbiter (10.1.71.130:8002).
|
||||
#
|
||||
# Context (t_34b96e83, 2026-08-13, OpenViking Phase 1b):
|
||||
# Ryan approved adding nomic-embed-text-v1.5-Q4_K_M as an embedding model
|
||||
# after Phase 0 follow-up confirmed embedding models fold cleanly into the
|
||||
# existing router preset via embedding=true. Model ID is "nomic-embed-text-v1.5".
|
||||
# No alias needed — peter-parker and Honcho consumers will call it by the section
|
||||
# name directly.
|
||||
#
|
||||
# What this playbook does:
|
||||
# 1. Downloads nomic-embed-text-v1.5-Q4_K_M.gguf into /opt/models if not
|
||||
# already present (idempotent: exact size-check guard, no re-pull on match).
|
||||
# 2. Redeploys the preset INI (adding the [nomic-embed-text-v1.5] section with
|
||||
# embedding=true, n-gpu-layers=99, ctx-size=8192, load-on-startup=true,
|
||||
# sleep-idle-seconds=-1).
|
||||
# 3. Restarts llama-server-router to pick up the new model entry.
|
||||
# 4. Verifies /v1/models returns all 5 models including the new nomic entry.
|
||||
# 5. Runs a /v1/embeddings smoke test to confirm the model actually embeds.
|
||||
#
|
||||
# VRAM context note (t_34b96e83):
|
||||
# nomic-embed-text-v1.5 Q4_K_M: ~84MB weights, embedding model (no KV cache).
|
||||
# VRAM impact is negligible — always pinned via sleep-idle-seconds=-1.
|
||||
# The 4 generative models remain unchanged (OOM analysis unchanged from t_55c164f5).
|
||||
#
|
||||
# Usage (from ~/git/homelab/ansible):
|
||||
# env -u ANSIBLE_VAULT_PASSWORD_FILE ansible-playbook -i inventory.yml \
|
||||
# playbooks/day2_add_nomic_embed.yml
|
||||
#
|
||||
# Semaphore note: Semaphore SSH key for jarvis user is not loaded in the
|
||||
# container (known pitfall, homelab-llm-serving skill). Run via CLI with
|
||||
# id_jarvis key; document as exception per Ryan's standing CLI fallback directive.
|
||||
#
|
||||
# Author: War Machine (2026-08-13, t_34b96e83)
|
||||
# ------------------------------------------------------------------------------
|
||||
|
||||
- name: "Add nomic-embed-text-v1.5 embedding model to astro-orbiter router"
|
||||
hosts: astro_orbiter
|
||||
gather_facts: false
|
||||
become: true
|
||||
|
||||
vars:
|
||||
# Activate preset mode
|
||||
llm_router_preset_enabled: true
|
||||
llm_router_preset_path: /opt/llama-server-router-preset.ini
|
||||
|
||||
# Production port (router is on 8002 since t_cd0d5388)
|
||||
llm_router_port: 8002
|
||||
|
||||
# Per-model ctx-size settings (carried from t_55c164f5; nomic new)
|
||||
llm_router_llama_ctx_size: 8192
|
||||
llm_router_llama_flash_attn: "true"
|
||||
llm_router_phi_ctx_size: 32768
|
||||
llm_router_phi_flash_attn: "true"
|
||||
llm_router_coder_ctx_size: 16384
|
||||
llm_router_coder_flash_attn: "true"
|
||||
llm_router_nomic_ctx_size: 8192
|
||||
# NOTE (2026-08-14, t_openviking_embed_batch): per-model batch-size/
|
||||
# ubatch-size lines in the preset INI are NOT honored by llama-server's
|
||||
# router — only ctx-size is applied per-model; batch-size/ubatch-size for
|
||||
# every spawned child come from the router's own global CLI flags
|
||||
# (confirmed via `ps aux` on astro-orbiter: child process launched with
|
||||
# the router's --batch-size/--ubatch-size regardless of the INI values).
|
||||
# Kept below for documentation/future-proofing but the REAL fix is the
|
||||
# global llm_router_batch_size / llm_router_ubatch_size override further
|
||||
# down, which raises the physical batch for ALL models on this router
|
||||
# (Qwen3.6-35B, Phi, Llama, Coder, nomic).
|
||||
llm_router_nomic_batch_size: 4096
|
||||
llm_router_nomic_ubatch_size: 4096
|
||||
|
||||
# All other vars inherit from host_vars + defaults/main.yml.
|
||||
llm_router_enabled: true
|
||||
llm_service_user: jarvis
|
||||
llm_binary_path: /opt/llama.cpp/build/bin/llama-server
|
||||
llm_models_dir: /opt/models
|
||||
llm_bind_address: "10.1.71.130"
|
||||
llm_allowed_source_cidr: "10.1.70.0/24"
|
||||
llm_router_service_name: llama-server-router
|
||||
llm_router_bind_address: "10.1.71.130"
|
||||
llm_router_allowed_source_cidr: "10.1.70.0/24"
|
||||
llm_router_models_dir: /opt/models
|
||||
llm_router_models_max: 4 # from host_vars; bumped by t_33acbb2e
|
||||
llm_router_ctx_size: 65536 # Qwen3.6-35B default; per-model overrides above
|
||||
llm_router_parallel: 1
|
||||
llm_router_gpu_layers: 99
|
||||
# FIX (2026-08-14, t_openviking_embed_batch): raised from 512 to 4096.
|
||||
# This is a GLOBAL router flag applied to every spawned model process
|
||||
# (per-model INI batch-size/ubatch-size overrides are not honored by
|
||||
# llama-server's router — see note above nomic vars). 512 tokens was too
|
||||
# small for OpenViking's chunked-document embedding inputs (observed
|
||||
# 2000-3400 tokens/chunk), causing hard 500 errors ("input (N tokens) is
|
||||
# too large to process") that tripped OpenViking's circuit breaker into a
|
||||
# permanent fail/re-enqueue loop. 4096 comfortably covers observed chunk
|
||||
# sizes and stays under nomic's ctx-size=8192. VRAM impact of raising
|
||||
# ubatch-size is in compute-buffer scratch space, not KV cache; monitored
|
||||
# post-deploy against the 23000 MiB budget (host_vars/astro-orbiter).
|
||||
llm_router_batch_size: 4096
|
||||
llm_router_ubatch_size: 4096
|
||||
llm_router_cache_type_k: q4_0
|
||||
llm_router_cache_type_v: q4_0
|
||||
llm_router_flash_attn: "auto"
|
||||
llm_router_expected_model_id: "Qwen3.6-35B-A3B-UD-Q4_K_S"
|
||||
llm_router_vram_max_mib: 23000
|
||||
|
||||
# nomic model staging
|
||||
nomic_filename: "nomic-embed-text-v1.5-Q4_K_M.gguf"
|
||||
nomic_url: "https://huggingface.co/nomic-ai/nomic-embed-text-v1.5-GGUF/resolve/main/nomic-embed-text-v1.5.Q4_K_M.gguf"
|
||||
nomic_size_bytes: 84106624
|
||||
|
||||
handlers:
|
||||
- name: reload systemd
|
||||
ansible.builtin.systemd:
|
||||
daemon_reload: true
|
||||
become: true
|
||||
listen: "reload systemd"
|
||||
|
||||
- name: restart router
|
||||
ansible.builtin.systemd:
|
||||
name: llama-server-router
|
||||
state: restarted
|
||||
become: true
|
||||
listen: "restart router"
|
||||
|
||||
tasks:
|
||||
|
||||
# ==========================================================================
|
||||
# PHASE 1: Download nomic GGUF if not present / size mismatch
|
||||
# ==========================================================================
|
||||
|
||||
- name: "[nomic] Stat existing GGUF"
|
||||
ansible.builtin.stat:
|
||||
path: "{{ llm_models_dir }}/{{ nomic_filename }}"
|
||||
get_checksum: false
|
||||
register: nomic_stat
|
||||
|
||||
- name: "[nomic] Download GGUF (skip if present and size matches)"
|
||||
ansible.builtin.get_url:
|
||||
url: "{{ nomic_url }}"
|
||||
dest: "{{ llm_models_dir }}/{{ nomic_filename }}"
|
||||
owner: "{{ llm_service_user }}"
|
||||
group: "{{ llm_service_user }}"
|
||||
mode: "0644"
|
||||
timeout: 300
|
||||
when: >
|
||||
not nomic_stat.stat.exists or
|
||||
nomic_stat.stat.size != nomic_size_bytes
|
||||
register: nomic_download
|
||||
notify: restart router
|
||||
|
||||
- name: "[nomic] Confirm GGUF size post-download"
|
||||
ansible.builtin.stat:
|
||||
path: "{{ llm_models_dir }}/{{ nomic_filename }}"
|
||||
get_checksum: false
|
||||
register: nomic_stat_post
|
||||
|
||||
- name: "[nomic] FAIL if GGUF size mismatch after download"
|
||||
ansible.builtin.fail:
|
||||
msg: >-
|
||||
GGUF size mismatch: expected {{ nomic_size_bytes }} bytes,
|
||||
got {{ nomic_stat_post.stat.size }} bytes.
|
||||
Re-download may be needed.
|
||||
when: nomic_stat_post.stat.size != nomic_size_bytes
|
||||
|
||||
# ==========================================================================
|
||||
# PHASE 2: Deploy updated preset INI (adds nomic-embed-text-v1.5 section)
|
||||
# ==========================================================================
|
||||
|
||||
- name: "[nomic] Deploy preset INI to {{ llm_router_preset_path }}"
|
||||
ansible.builtin.template:
|
||||
src: "../roles/llm-inference-multimodel/templates/llama-server-router-preset.ini.j2"
|
||||
dest: "{{ llm_router_preset_path }}"
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
register: nomic_preset_deployed
|
||||
notify: restart router
|
||||
|
||||
# ==========================================================================
|
||||
# PHASE 3: Redeploy systemd unit (ensures unit is fresh; no flag changes)
|
||||
# ==========================================================================
|
||||
|
||||
- name: "[nomic] Deploy llama-server-router unit"
|
||||
ansible.builtin.template:
|
||||
src: "../roles/llm-inference-multimodel/templates/llama-server-router.service.j2"
|
||||
dest: /etc/systemd/system/llama-server-router.service
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
register: nomic_unit_deployed
|
||||
notify:
|
||||
- reload systemd
|
||||
- restart router
|
||||
|
||||
- name: "[nomic] Flush handlers (daemon-reload + router restart)"
|
||||
ansible.builtin.meta: flush_handlers
|
||||
|
||||
# ==========================================================================
|
||||
# PHASE 4: Verify router is up and nomic model appears in /v1/models
|
||||
# ==========================================================================
|
||||
|
||||
- name: "[nomic] Wait for /health (router supervisor)"
|
||||
ansible.builtin.uri:
|
||||
url: "http://{{ llm_router_bind_address }}:{{ llm_router_port }}/health"
|
||||
status_code: 200
|
||||
timeout: 30
|
||||
retries: 12
|
||||
delay: 5
|
||||
register: nomic_health
|
||||
until: nomic_health.status == 200
|
||||
|
||||
- name: "[nomic] Query /v1/models"
|
||||
ansible.builtin.uri:
|
||||
url: "http://{{ llm_router_bind_address }}:{{ llm_router_port }}/v1/models"
|
||||
status_code: 200
|
||||
return_content: true
|
||||
timeout: 30
|
||||
register: nomic_models
|
||||
|
||||
- name: "[nomic] Extract model IDs and aliases"
|
||||
ansible.builtin.set_fact:
|
||||
nomic_model_ids: "{{ nomic_models.json.data | map(attribute='id') | list }}"
|
||||
nomic_all_aliases: "{{ nomic_models.json.data | map(attribute='aliases') | flatten | list }}"
|
||||
|
||||
- name: "[nomic] FAIL if nomic primary ID missing"
|
||||
ansible.builtin.fail:
|
||||
msg: >-
|
||||
'nomic-embed-text-v1.5' not in /v1/models.
|
||||
IDs: {{ nomic_model_ids }}
|
||||
when: "'nomic-embed-text-v1.5' not in nomic_model_ids"
|
||||
|
||||
- name: "[nomic] FAIL if Qwen3.6-35B missing"
|
||||
ansible.builtin.fail:
|
||||
msg: "'Qwen3.6-35B-A3B-UD-Q4_K_S' not in /v1/models. IDs: {{ nomic_model_ids }}"
|
||||
when: "'Qwen3.6-35B-A3B-UD-Q4_K_S' not in nomic_model_ids"
|
||||
|
||||
- name: "[nomic] FAIL if Phi missing"
|
||||
ansible.builtin.fail:
|
||||
msg: "'Phi-3.5-mini-instruct-Q8_0' not in /v1/models. IDs: {{ nomic_model_ids }}"
|
||||
when: "'Phi-3.5-mini-instruct-Q8_0' not in nomic_model_ids"
|
||||
|
||||
- name: "[nomic] FAIL if Llama missing"
|
||||
ansible.builtin.fail:
|
||||
msg: "'Meta-Llama-3.1-8B-Instruct-Q4_K_M' not in /v1/models. IDs: {{ nomic_model_ids }}"
|
||||
when: "'Meta-Llama-3.1-8B-Instruct-Q4_K_M' not in nomic_model_ids"
|
||||
|
||||
- name: "[nomic] FAIL if Coder missing"
|
||||
ansible.builtin.fail:
|
||||
msg: "'Qwen2.5-Coder-14B-Instruct-Q4_K_M' not in /v1/models. IDs: {{ nomic_model_ids }}"
|
||||
when: "'Qwen2.5-Coder-14B-Instruct-Q4_K_M' not in nomic_model_ids"
|
||||
|
||||
# ==========================================================================
|
||||
# PHASE 5: /v1/embeddings smoke test — confirm model actually embeds
|
||||
# ==========================================================================
|
||||
|
||||
- name: "[nomic] POST /v1/embeddings smoke test"
|
||||
ansible.builtin.uri:
|
||||
url: "http://{{ llm_router_bind_address }}:{{ llm_router_port }}/v1/embeddings"
|
||||
method: POST
|
||||
body_format: json
|
||||
body:
|
||||
model: "nomic-embed-text-v1.5"
|
||||
input: "The dog ran across the park."
|
||||
status_code: 200
|
||||
return_content: true
|
||||
timeout: 120
|
||||
register: nomic_embed_result
|
||||
|
||||
- name: "[nomic] Extract embedding vector length"
|
||||
ansible.builtin.set_fact:
|
||||
nomic_embed_dims: >-
|
||||
{{ (nomic_embed_result.json.data | first).embedding | length }}
|
||||
when:
|
||||
- nomic_embed_result.status == 200
|
||||
- nomic_embed_result.json.data is defined
|
||||
- nomic_embed_result.json.data | length > 0
|
||||
|
||||
- name: "[nomic] FAIL if embedding vector is empty or missing"
|
||||
ansible.builtin.fail:
|
||||
msg: >-
|
||||
Embedding smoke test returned no vector.
|
||||
Response: {{ nomic_embed_result.json }}
|
||||
when: >-
|
||||
nomic_embed_result.status != 200 or
|
||||
nomic_embed_result.json.data is not defined or
|
||||
nomic_embed_result.json.data | length == 0 or
|
||||
(nomic_embed_result.json.data | first).embedding | length == 0
|
||||
|
||||
- name: "[nomic] PASS — full summary"
|
||||
ansible.builtin.debug:
|
||||
msg:
|
||||
- "========================================================================"
|
||||
- "NOMIC-EMBED-TEXT-V1.5 DEPLOYMENT — COMPLETE"
|
||||
- ""
|
||||
- " Mode: --models-preset ({{ llm_router_preset_path }})"
|
||||
- " Service: llama-server-router.service (:{{ llm_router_port }})"
|
||||
- ""
|
||||
- " /v1/models IDs: {{ nomic_model_ids }}"
|
||||
- ""
|
||||
- " VERIFY:"
|
||||
- " Qwen3.6-35B-A3B-UD-Q4_K_S: {{ 'PRESENT' if 'Qwen3.6-35B-A3B-UD-Q4_K_S' in nomic_model_ids else 'MISSING' }}"
|
||||
- " Phi-3.5-mini-instruct-Q8_0: {{ 'PRESENT' if 'Phi-3.5-mini-instruct-Q8_0' in nomic_model_ids else 'MISSING' }}"
|
||||
- " Meta-Llama-3.1-8B-Instruct-Q4_K_M: {{ 'PRESENT' if 'Meta-Llama-3.1-8B-Instruct-Q4_K_M' in nomic_model_ids else 'MISSING' }}"
|
||||
- " Qwen2.5-Coder-14B-Instruct-Q4_K_M: {{ 'PRESENT' if 'Qwen2.5-Coder-14B-Instruct-Q4_K_M' in nomic_model_ids else 'MISSING' }}"
|
||||
- " nomic-embed-text-v1.5: {{ 'PRESENT' if 'nomic-embed-text-v1.5' in nomic_model_ids else 'MISSING' }}"
|
||||
- ""
|
||||
- " Embedding smoke test: PASS"
|
||||
- " Vector dimensions: {{ nomic_embed_dims | default('unknown') }}"
|
||||
- ""
|
||||
- " GGUF download: {{ 'NEW DOWNLOAD' if (nomic_download is defined and nomic_download.changed) else 'ALREADY PRESENT (skipped)' }}"
|
||||
- "========================================================================"
|
||||
203
ansible/playbooks/day2_add_phi_alias.yml
Normal file
203
ansible/playbooks/day2_add_phi_alias.yml
Normal file
@@ -0,0 +1,203 @@
|
||||
---
|
||||
# ------------------------------------------------------------------------------
|
||||
# FILE: playbooks/day2_add_phi_alias.yml
|
||||
# DESCRIPTION: Add Phi-3.5-mini-instruct-8bit alias to the llama-server-router
|
||||
# by switching from --models-dir to --models-preset INI mode.
|
||||
#
|
||||
# Context (t_9adf0889, 2026-08-12):
|
||||
# Ryan's Hermes config (auxiliary.title_generation.model) points to
|
||||
# "Phi-3.5-mini-instruct-8bit" but the router only exposes the GGUF
|
||||
# filename-derived ID "Phi-3.5-mini-instruct-Q8_0". They are the same file.
|
||||
# This playbook adds the alias so both names work without changing Ryan's
|
||||
# Hermes config.
|
||||
#
|
||||
# What this playbook does:
|
||||
# 1. Deploys the preset INI template (llama-server-router-preset.ini.j2)
|
||||
# to /opt/llama-server-router-preset.ini on astro-orbiter.
|
||||
# 2. Redeploys the systemd unit (llama-server-router.service) with
|
||||
# --models-preset instead of --models-dir.
|
||||
# 3. Restarts llama-server-router to pick up the new flag.
|
||||
# 4. Verifies that /v1/models returns:
|
||||
# - Phi-3.5-mini-instruct-Q8_0 (original ID — must still work)
|
||||
# - Phi-3.5-mini-instruct-8bit (new alias — Ryan's config target)
|
||||
# - Qwen3.6-35B-A3B-UD-Q4_K_S (unchanged)
|
||||
# - Meta-Llama-3.1-8B-Instruct-Q4_K_M (unchanged)
|
||||
#
|
||||
# Known upstream behavior:
|
||||
# GH #22364: --models-preset creates an extra "default" entry in /v1/models.
|
||||
# This is cosmetic only and does not affect model selection by name.
|
||||
#
|
||||
# Usage (from ~/git/homelab/ansible):
|
||||
# ansible-playbook -i inventory.yml playbooks/day2_add_phi_alias.yml
|
||||
#
|
||||
# Semaphore note (t_9adf0889): Semaphore SSH key for jarvis user is not loaded
|
||||
# in the container (known pitfall, homelab-llm-serving skill). Run via CLI with
|
||||
# id_jarvis key; document as exception per Ryan's standing CLI fallback directive.
|
||||
#
|
||||
# Author: War Machine (2026-08-12, t_9adf0889)
|
||||
# ------------------------------------------------------------------------------
|
||||
|
||||
- name: "Add Phi-3.5-mini-instruct-8bit alias — switch router to preset mode"
|
||||
hosts: astro_orbiter
|
||||
gather_facts: false
|
||||
become: true
|
||||
|
||||
vars:
|
||||
# Activate preset mode and provide the on-disk INI path
|
||||
llm_router_preset_enabled: true
|
||||
llm_router_preset_path: /opt/llama-server-router-preset.ini
|
||||
|
||||
# Production port (router is already on 8002 since t_cd0d5388)
|
||||
llm_router_port: 8002
|
||||
|
||||
# All other vars inherit from host_vars + defaults/main.yml.
|
||||
# Explicitly set the ones needed by the unit template for clarity:
|
||||
llm_router_enabled: true
|
||||
llm_service_user: jarvis
|
||||
llm_binary_path: /opt/llama.cpp/build/bin/llama-server
|
||||
llm_models_dir: /opt/models
|
||||
llm_bind_address: "10.1.71.130"
|
||||
llm_allowed_source_cidr: "10.1.70.0/24"
|
||||
llm_router_service_name: llama-server-router
|
||||
llm_router_bind_address: "10.1.71.130"
|
||||
llm_router_allowed_source_cidr: "10.1.70.0/24"
|
||||
llm_router_models_dir: /opt/models
|
||||
llm_router_models_max: 4 # from host_vars; bumped by t_33acbb2e
|
||||
llm_router_ctx_size: 65536
|
||||
llm_router_parallel: 1
|
||||
llm_router_gpu_layers: 99
|
||||
llm_router_batch_size: 2048
|
||||
llm_router_ubatch_size: 512
|
||||
llm_router_cache_type_k: q4_0
|
||||
llm_router_cache_type_v: q4_0
|
||||
llm_router_flash_attn: "auto"
|
||||
llm_router_expected_model_id: "Qwen3.6-35B-A3B-UD-Q4_K_S"
|
||||
llm_router_vram_max_mib: 23000
|
||||
|
||||
handlers:
|
||||
- name: reload systemd
|
||||
ansible.builtin.systemd:
|
||||
daemon_reload: true
|
||||
become: true
|
||||
listen: "reload systemd"
|
||||
|
||||
- name: restart router
|
||||
ansible.builtin.systemd:
|
||||
name: llama-server-router
|
||||
state: restarted
|
||||
become: true
|
||||
listen: "restart router"
|
||||
|
||||
tasks:
|
||||
|
||||
# ==========================================================================
|
||||
# PHASE 1: Deploy the preset INI
|
||||
# ==========================================================================
|
||||
|
||||
- name: "[phi-alias] Deploy preset INI to {{ llm_router_preset_path }}"
|
||||
ansible.builtin.template:
|
||||
src: "../roles/llm-inference-multimodel/templates/llama-server-router-preset.ini.j2"
|
||||
dest: "{{ llm_router_preset_path }}"
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
register: phi_alias_preset_deployed
|
||||
notify:
|
||||
- restart router
|
||||
|
||||
# ==========================================================================
|
||||
# PHASE 2: Redeploy systemd unit with --models-preset flag
|
||||
# ==========================================================================
|
||||
|
||||
- name: "[phi-alias] Deploy llama-server-router unit (--models-preset mode)"
|
||||
ansible.builtin.template:
|
||||
src: "../roles/llm-inference-multimodel/templates/llama-server-router.service.j2"
|
||||
dest: /etc/systemd/system/llama-server-router.service
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
register: phi_alias_unit_deployed
|
||||
notify:
|
||||
- reload systemd
|
||||
- restart router
|
||||
|
||||
- name: "[phi-alias] Flush handlers (daemon-reload + router restart)"
|
||||
ansible.builtin.meta: flush_handlers
|
||||
|
||||
# ==========================================================================
|
||||
# PHASE 3: Verify alias is present
|
||||
# ==========================================================================
|
||||
|
||||
- name: "[phi-alias] Wait for /health (router supervisor, no model needed)"
|
||||
ansible.builtin.uri:
|
||||
url: "http://{{ llm_router_bind_address }}:{{ llm_router_port }}/health"
|
||||
status_code: 200
|
||||
timeout: 30
|
||||
retries: 12
|
||||
delay: 5
|
||||
register: phi_alias_health
|
||||
until: phi_alias_health.status == 200
|
||||
|
||||
- name: "[phi-alias] Query /v1/models"
|
||||
ansible.builtin.uri:
|
||||
url: "http://{{ llm_router_bind_address }}:{{ llm_router_port }}/v1/models"
|
||||
status_code: 200
|
||||
return_content: true
|
||||
timeout: 30
|
||||
register: phi_alias_models
|
||||
|
||||
- name: "[phi-alias] Extract model IDs and aliases"
|
||||
ansible.builtin.set_fact:
|
||||
phi_alias_model_ids: "{{ phi_alias_models.json.data | map(attribute='id') | list }}"
|
||||
phi_alias_all_aliases: "{{ phi_alias_models.json.data | map(attribute='aliases') | flatten | list }}"
|
||||
phi_alias_model_sources: "{{ phi_alias_models.json.data | map(attribute='source') | list }}"
|
||||
|
||||
- name: "[phi-alias] FAIL if Phi original ID missing"
|
||||
ansible.builtin.fail:
|
||||
msg: >-
|
||||
'Phi-3.5-mini-instruct-Q8_0' not in /v1/models.
|
||||
IDs: {{ phi_alias_model_ids }}
|
||||
when: "'Phi-3.5-mini-instruct-Q8_0' not in phi_alias_model_ids"
|
||||
|
||||
- name: "[phi-alias] FAIL if Phi alias missing"
|
||||
ansible.builtin.fail:
|
||||
msg: >-
|
||||
'Phi-3.5-mini-instruct-8bit' not found as ID or alias in /v1/models.
|
||||
IDs: {{ phi_alias_model_ids }}
|
||||
Aliases: {{ phi_alias_all_aliases }}
|
||||
when:
|
||||
- "'Phi-3.5-mini-instruct-8bit' not in phi_alias_model_ids"
|
||||
- "'Phi-3.5-mini-instruct-8bit' not in phi_alias_all_aliases"
|
||||
|
||||
- name: "[phi-alias] FAIL if Qwen missing"
|
||||
ansible.builtin.fail:
|
||||
msg: "'Qwen3.6-35B-A3B-UD-Q4_K_S' not in /v1/models. IDs: {{ phi_alias_model_ids }}"
|
||||
when: "'Qwen3.6-35B-A3B-UD-Q4_K_S' not in phi_alias_model_ids"
|
||||
|
||||
- name: "[phi-alias] FAIL if Llama missing"
|
||||
ansible.builtin.fail:
|
||||
msg: "'Meta-Llama-3.1-8B-Instruct-Q4_K_M' not in /v1/models. IDs: {{ phi_alias_model_ids }}"
|
||||
when: "'Meta-Llama-3.1-8B-Instruct-Q4_K_M' not in phi_alias_model_ids"
|
||||
|
||||
- name: "[phi-alias] PASS — full /v1/models summary"
|
||||
ansible.builtin.debug:
|
||||
msg:
|
||||
- "========================================================================"
|
||||
- "PHI ALIAS DEPLOYMENT — COMPLETE"
|
||||
- ""
|
||||
- " Mode: --models-preset ({{ llm_router_preset_path }})"
|
||||
- " Service: llama-server-router.service (:{{ llm_router_port }})"
|
||||
- ""
|
||||
- " /v1/models IDs: {{ phi_alias_model_ids }}"
|
||||
- " /v1/models aliases: {{ phi_alias_all_aliases }}"
|
||||
- " Sources: {{ phi_alias_model_sources }}"
|
||||
- ""
|
||||
- " VERIFY:"
|
||||
- " Phi-3.5-mini-instruct-Q8_0: {{ 'PRESENT' if 'Phi-3.5-mini-instruct-Q8_0' in phi_alias_model_ids else 'MISSING' }}"
|
||||
- " Phi-3.5-mini-instruct-8bit: {{ 'PRESENT (ID)' if 'Phi-3.5-mini-instruct-8bit' in phi_alias_model_ids else ('PRESENT (alias)' if 'Phi-3.5-mini-instruct-8bit' in phi_alias_all_aliases else 'MISSING') }}"
|
||||
- " Qwen3.6-35B-A3B-UD-Q4_K_S: {{ 'PRESENT' if 'Qwen3.6-35B-A3B-UD-Q4_K_S' in phi_alias_model_ids else 'MISSING' }}"
|
||||
- " Meta-Llama-3.1-8B-Instruct-Q4_K_M: {{ 'PRESENT' if 'Meta-Llama-3.1-8B-Instruct-Q4_K_M' in phi_alias_model_ids else 'MISSING' }}"
|
||||
- ""
|
||||
- " GH #22364: if 'default' appears in IDs above, that is expected"
|
||||
- " in --models-preset mode. Cosmetic only."
|
||||
- "========================================================================"
|
||||
161
ansible/playbooks/day2_bump_router_models_max.yml
Normal file
161
ansible/playbooks/day2_bump_router_models_max.yml
Normal file
@@ -0,0 +1,161 @@
|
||||
---
|
||||
# ------------------------------------------------------------------------------
|
||||
# FILE: playbooks/day2_bump_router_models_max.yml
|
||||
# DESCRIPTION: Bump --models-max on the production llama-server-router unit.
|
||||
#
|
||||
# Context: t_33acbb2e (2026-08-12) — Ryan requested --models-max raised from 1
|
||||
# to 4 so the router can keep multiple GGUFs resident on-demand (LRU eviction
|
||||
# when the cap is reached). The actual var change lives in:
|
||||
# host_vars/astro-orbiter/vars.yml (llm_router_models_max: 4)
|
||||
#
|
||||
# This playbook:
|
||||
# 1. Re-renders llama-server-router.service.j2 with the updated var value.
|
||||
# 2. Reloads systemd (daemon-reload handler) if the unit changed.
|
||||
# 3. Restarts llama-server-router so the new --models-max takes effect on the
|
||||
# live process. Router holds no resident model (all-unloaded) so restart
|
||||
# is sub-second and non-disruptive.
|
||||
# 4. Verifies /health returns 200 and /v1/models still lists all three GGUFs.
|
||||
#
|
||||
# VRAM NOTE: --models-max 4 allows up to all 3 current GGUFs to co-reside on
|
||||
# a 24GB card simultaneously. Worst-case combined footprint is ~31GB which
|
||||
# EXCEEDS 24GB — OOM is possible if all 3 are loaded concurrently. In normal
|
||||
# single-user homelab operation this is very unlikely. Full VRAM breakdown
|
||||
# documented in host_vars/astro-orbiter/vars.yml. Ryan approved (t_33acbb2e).
|
||||
#
|
||||
# Execution channel: Semaphore template "llm_router_update_unit" (project mk-labs).
|
||||
# Do NOT run via direct ansible-playbook or ad-hoc ssh/systemctl.
|
||||
#
|
||||
# Author: War Machine (2026-08-12, t_33acbb2e)
|
||||
# ------------------------------------------------------------------------------
|
||||
|
||||
- name: "Bump llama-server-router --models-max to 4 on astro-orbiter"
|
||||
hosts: astro_orbiter
|
||||
gather_facts: true
|
||||
become: true
|
||||
|
||||
vars:
|
||||
# Production vars — router is live on :8002 (post-cutover t_cd0d5388)
|
||||
llm_router_port: 8002
|
||||
llm_router_bind_address: "10.1.71.130"
|
||||
|
||||
# llm_router_models_max is 4 via host_vars/astro-orbiter/vars.yml.
|
||||
# Remaining role vars come from host_vars + defaults/main.yml via the
|
||||
# inventory — we only explicitly set vars this playbook needs for its
|
||||
# own tasks (health/models check URIs).
|
||||
|
||||
# Needed by the template task (mirrors defaults set in role defaults/main.yml)
|
||||
llm_service_user: jarvis
|
||||
llm_binary_path: /opt/llama.cpp/build/bin/llama-server
|
||||
llm_models_dir: /opt/models
|
||||
llm_router_service_name: llama-server-router
|
||||
llm_router_models_dir: /opt/models
|
||||
llm_router_gpu_layers: 99
|
||||
llm_router_ctx_size: 65536
|
||||
llm_router_flash_attn: "auto"
|
||||
llm_router_cache_type_k: q4_0
|
||||
llm_router_cache_type_v: q4_0
|
||||
llm_router_batch_size: 2048
|
||||
llm_router_ubatch_size: 512
|
||||
llm_router_parallel: 1
|
||||
|
||||
tasks:
|
||||
# -------------------------------------------------------------------------
|
||||
# Phase 1: Re-render the router unit file
|
||||
# Template src path is relative to the role's templates/ dir; we reference
|
||||
# it with a relative path that Ansible resolves from the role directory.
|
||||
# -------------------------------------------------------------------------
|
||||
|
||||
- name: "Deploy updated llama-server-router unit (--models-max {{ llm_router_models_max }})"
|
||||
ansible.builtin.template:
|
||||
src: "{{ playbook_dir }}/../roles/llm-inference-multimodel/templates/llama-server-router.service.j2"
|
||||
dest: "/etc/systemd/system/{{ llm_router_service_name }}.service"
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
register: llm_router_unit_updated
|
||||
notify:
|
||||
- reload systemd
|
||||
tags: [always]
|
||||
|
||||
- name: "Flush handlers — ensure daemon-reload lands before restart"
|
||||
ansible.builtin.meta: flush_handlers
|
||||
tags: [always]
|
||||
|
||||
# -------------------------------------------------------------------------
|
||||
# Phase 2: Restart the router so the new --models-max takes effect.
|
||||
# Always restart (even if unit unchanged) to ensure live process matches.
|
||||
# -------------------------------------------------------------------------
|
||||
|
||||
- name: "Restart llama-server-router so --models-max {{ llm_router_models_max }} takes effect"
|
||||
ansible.builtin.systemd:
|
||||
name: "{{ llm_router_service_name }}"
|
||||
state: restarted
|
||||
enabled: true
|
||||
tags: [always]
|
||||
|
||||
# -------------------------------------------------------------------------
|
||||
# Phase 3: Verify /health returns 200
|
||||
# -------------------------------------------------------------------------
|
||||
|
||||
- name: "Wait for /health to return 200 after restart"
|
||||
ansible.builtin.uri:
|
||||
url: "http://{{ llm_router_bind_address }}:{{ llm_router_port }}/health"
|
||||
status_code: 200
|
||||
timeout: 30
|
||||
register: bump_health_check
|
||||
retries: 10
|
||||
delay: 3
|
||||
until: bump_health_check.status == 200
|
||||
tags: [always]
|
||||
|
||||
# -------------------------------------------------------------------------
|
||||
# Phase 4: Verify /v1/models lists all three GGUFs
|
||||
# -------------------------------------------------------------------------
|
||||
|
||||
- name: "Check /v1/models — all three GGUFs should appear"
|
||||
ansible.builtin.uri:
|
||||
url: "http://{{ llm_router_bind_address }}:{{ llm_router_port }}/v1/models"
|
||||
status_code: 200
|
||||
timeout: 30
|
||||
return_content: true
|
||||
register: bump_models_check
|
||||
tags: [always]
|
||||
|
||||
- name: "Display /v1/models summary"
|
||||
ansible.builtin.debug:
|
||||
msg:
|
||||
- "======================================================================"
|
||||
- "--models-max BUMP VERIFICATION (t_33acbb2e)"
|
||||
- ""
|
||||
- " /health: HTTP {{ bump_health_check.status }}"
|
||||
- " /v1/models HTTP: {{ bump_models_check.status }}"
|
||||
- " Models listed: {{ bump_models_check.json.data | map(attribute='id') | list | join(', ') }}"
|
||||
- ""
|
||||
- " --models-max now: {{ llm_router_models_max }}"
|
||||
- " --parallel (unchanged): {{ llm_router_parallel }}"
|
||||
- ""
|
||||
- " VRAM WARNING: worst-case 3-model co-residency ~31GB > 24GB RTX 3090."
|
||||
- " OOM risk if all 3 load concurrently. LRU eviction mitigates in practice."
|
||||
- " Full breakdown: host_vars/astro-orbiter/vars.yml"
|
||||
- "======================================================================"
|
||||
when: bump_models_check is defined
|
||||
tags: [always]
|
||||
|
||||
- name: "GATE: confirm all 3 expected GGUFs appear in /v1/models"
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- "'Qwen3.6-35B-A3B-UD-Q4_K_S' in (bump_models_check.json.data | map(attribute='id') | list)"
|
||||
- "'Phi-3.5-mini-instruct-Q8_0' in (bump_models_check.json.data | map(attribute='id') | list)"
|
||||
- "'Meta-Llama-3.1-8B-Instruct-Q4_K_M' in (bump_models_check.json.data | map(attribute='id') | list)"
|
||||
fail_msg: >-
|
||||
/v1/models did not return all 3 expected GGUFs after --models-max bump.
|
||||
Check router logs: journalctl -u llama-server-router -n 50
|
||||
success_msg: "GATE PASSED: all 3 GGUFs listed in /v1/models."
|
||||
when: bump_models_check is defined
|
||||
tags: [always]
|
||||
|
||||
handlers:
|
||||
- name: reload systemd
|
||||
ansible.builtin.systemd:
|
||||
daemon_reload: true
|
||||
listen: "reload systemd"
|
||||
59
ansible/playbooks/day2_cpu_offload_aux_models.yml
Normal file
59
ansible/playbooks/day2_cpu_offload_aux_models.yml
Normal file
@@ -0,0 +1,59 @@
|
||||
---
|
||||
# ------------------------------------------------------------------------------
|
||||
# Playbook: day2_cpu_offload_aux_models.yml
|
||||
# Purpose: CPU-offload Qwen2.5-Coder-14B and Meta-Llama-3.1-8B on
|
||||
# astro-orbiter's production router (port 8002).
|
||||
#
|
||||
# What this playbook does:
|
||||
# 1. Re-renders llama-server-router-preset.ini (Coder + Llama sections now
|
||||
# use per-model n-gpu-layers vars = 0 -> full CPU inference).
|
||||
# 2. Re-renders the router unit (--models-max now 4 via host_vars, global
|
||||
# --n-gpu-layers removed per t_72646029 unit template fix) and restarts
|
||||
# llama-server-router so both changes take effect.
|
||||
# 3. Verifies per the role's router_preset phase.
|
||||
#
|
||||
# Context (2026-08-17):
|
||||
# - RAM/model-swap audit, TIER 1 (Coder-14B CPU offload) + TIER 2
|
||||
# (Llama-3.1-8B CPU offload) — Ryan approved 1 & 2 on 2026-08-17.
|
||||
# See inbox/ryan/2026-08-17-llm-system-ram-model-swap.md.
|
||||
# - Unit template fix (t_72646029): global --n-gpu-layers removed from
|
||||
# ExecStart in preset mode. Each INI section now sets n-gpu-layers
|
||||
# explicitly (Qwen3.8=99, Phi=99, nomic=99, Coder=0, Llama=0).
|
||||
# - Concurrent residency after change: Qwen3.8-27B (20,302 MiB @ 128K ctx)
|
||||
# + nomic-embed (558 MiB, pinned) + Coder (CPU, ~1,390 MiB CUDA ctx) +
|
||||
# Llama (CPU, ~1,706 MiB CUDA ctx) = ~24,004 MiB. NOTE: llama.cpp 6ea215d
|
||||
# allocates CUDA-context VRAM even at n-gpu-layers=0, so CPU models are not
|
||||
# 0-VRAM; total sits at the 24,576 MiB physical limit (headroom ~572 MiB).
|
||||
# Qwen3.8 is never evicted for a CPU aux model; Phi-3.5-mini (GPU, 8.3GB)
|
||||
# still evicts as before.
|
||||
# - CPU speed (8-core Ryzen 7 5800XT): ~5-10 tok/s (14B), ~10-20 tok/s (8B).
|
||||
# - Semaphore SSH gap for astro-orbiter still applies (t_730f9584 /
|
||||
# t_33acbb2e); running direct CLI Ansible per standing exception.
|
||||
#
|
||||
# Run:
|
||||
# cd /home/hermes/git/homelab/ansible
|
||||
# env -u ANSIBLE_VAULT_PASSWORD_FILE ansible-playbook \
|
||||
# -i inventory.yml \
|
||||
# playbooks/day2_cpu_offload_aux_models.yml
|
||||
#
|
||||
# Rollback:
|
||||
# git checkout -- \
|
||||
# roles/llm-inference-multimodel/templates/llama-server-router.service.j2 \
|
||||
# roles/llm-inference-multimodel/templates/llama-server-router-preset.ini.j2 \
|
||||
# roles/llm-inference-multimodel/defaults/main.yml \
|
||||
# host_vars/astro-orbiter/vars.yml
|
||||
# (restores n-gpu-layers=99 global flag, models-max=2, all GPU)
|
||||
# then re-run this playbook to redeploy rollback state.
|
||||
# Note: playbooks/day2_cpu_offload_aux_models.yml is untracked — left on disk.
|
||||
# ------------------------------------------------------------------------------
|
||||
- name: CPU-offload Coder-14B and Llama-3.1-8B on astro-orbiter
|
||||
hosts: astro-orbiter
|
||||
become: true
|
||||
vars:
|
||||
llm_router_preset_enabled: true
|
||||
llm_router_enabled: true
|
||||
llm_router_port: 8002
|
||||
|
||||
roles:
|
||||
- role: llm-inference-multimodel
|
||||
tags: [always]
|
||||
511
ansible/playbooks/day2_cutover_qwen_to_router.yml
Normal file
511
ansible/playbooks/day2_cutover_qwen_to_router.yml
Normal file
@@ -0,0 +1,511 @@
|
||||
---
|
||||
# ------------------------------------------------------------------------------
|
||||
# FILE: playbooks/day2_cutover_qwen_to_router.yml
|
||||
# DESCRIPTION: Promote llama-server-router to production on port 8002.
|
||||
#
|
||||
# Context: Router-mode shadow deployment (t_0cca74a2) validated 2026-08-12:
|
||||
# all 4 hard gates PASSED (n_ctx 65536, tool-calling PASS, hallucination-stress
|
||||
# PASS, VRAM 20410 MiB / 1 process). Ryan approved cutover.
|
||||
#
|
||||
# This playbook makes the router the permanent production endpoint:
|
||||
#
|
||||
# 1. Stop + disable llama-server-qwen (:8002). Unit file is PRESERVED on disk
|
||||
# as the rollback target (same pattern as prior role history).
|
||||
# 2. Redeploy llama-server-router unit file with --port 8002 (production port).
|
||||
# PORT DECISION: we rebind the router to :8002 rather than updating 8
|
||||
# dependent Hermes profiles' base_url. One unit file change beats 8
|
||||
# config.yaml updates — atomic, GitOps-clean, zero profile drift.
|
||||
# 3. Enable + start llama-server-router on :8002.
|
||||
# 4. Re-run validation gates 1-3 against the NOW-production endpoint.
|
||||
# (Same logic as Phase R / router_verify in tasks/router.yml — hard gates.)
|
||||
# 5. Run Gate 4: verify bundled SvelteKit UI is reachable.
|
||||
#
|
||||
# Usage (from ~/git/homelab/ansible):
|
||||
# ansible-playbook -i inventory.yml playbooks/day2_cutover_qwen_to_router.yml
|
||||
#
|
||||
# Rollback (if gates fail or any time after):
|
||||
# ansible-playbook -i inventory.yml playbooks/day2_cutover_qwen_to_router.yml \
|
||||
# --tags cutover_rollback
|
||||
#
|
||||
# Author: War Machine (2026-08-12, t_cd0d5388)
|
||||
# Approved by: Ryan (cutover authorization, 2026-08-12)
|
||||
# ------------------------------------------------------------------------------
|
||||
|
||||
- name: "CUTOVER — Promote llama-server-router to production (:8002) on astro-orbiter"
|
||||
hosts: astro_orbiter
|
||||
gather_facts: true
|
||||
become: true
|
||||
|
||||
vars:
|
||||
# ----------------------------------------------------------------
|
||||
# PORT DECISION:
|
||||
# We rebind the router to :8002 (production port) rather than
|
||||
# updating 8 dependent Hermes profiles' base_url to :8003.
|
||||
# Rationale: one unit file change is atomic and GitOps-clean.
|
||||
# Updating 8 config.yaml files risks drift and requires per-profile
|
||||
# activation tests. The template renders llm_router_port as the
|
||||
# --port argument; we just override it here to 8002.
|
||||
# ----------------------------------------------------------------
|
||||
|
||||
# Router port override: take over production port
|
||||
llm_router_port: 8002
|
||||
|
||||
# All other role defaults needed by the template (mirrors defaults/main.yml)
|
||||
llm_service_user: jarvis
|
||||
llm_binary_path: /opt/llama.cpp/build/bin/llama-server
|
||||
llm_models_dir: /opt/models
|
||||
llm_bind_address: "10.1.71.130"
|
||||
llm_allowed_source_cidr: "10.1.70.0/24"
|
||||
|
||||
llm_router_enabled: true
|
||||
llm_router_service_name: llama-server-router
|
||||
llm_router_models_dir: /opt/models
|
||||
llm_router_models_max: 1 # CRITICAL: RTX 3090 24GB, single model only
|
||||
llm_router_ctx_size: 65536
|
||||
llm_router_parallel: 1
|
||||
llm_router_gpu_layers: 99
|
||||
llm_router_batch_size: 2048
|
||||
llm_router_ubatch_size: 512
|
||||
llm_router_cache_type_k: q4_0
|
||||
llm_router_cache_type_v: q4_0
|
||||
llm_router_flash_attn: "auto"
|
||||
llm_router_bind_address: "10.1.71.130"
|
||||
llm_router_allowed_source_cidr: "10.1.70.0/24"
|
||||
llm_router_expected_model_id: "Qwen3.6-35B-A3B-UD-Q4_K_S"
|
||||
llm_router_vram_max_mib: 23000
|
||||
|
||||
llm_qwen_service_name: llama-server-qwen
|
||||
llm_qwen_port: 8002
|
||||
|
||||
tasks:
|
||||
|
||||
# =======================================================================
|
||||
# PHASE 1 — Stop and disable llama-server-qwen (bare single-model)
|
||||
# Preserve unit file on disk — rollback target per existing role pattern.
|
||||
# =======================================================================
|
||||
|
||||
- name: "[cutover] PHASE 1: Confirm llama-server-qwen current state"
|
||||
ansible.builtin.systemd:
|
||||
name: llama-server-qwen
|
||||
register: cutover_qwen_status
|
||||
tags: [cutover_stop_qwen, cutover]
|
||||
|
||||
- name: "[cutover] PHASE 1: Report current llama-server-qwen status"
|
||||
ansible.builtin.debug:
|
||||
msg: >-
|
||||
llama-server-qwen: ActiveState={{ cutover_qwen_status.status.ActiveState | default('unknown') }},
|
||||
UnitFileState={{ cutover_qwen_status.status.UnitFileState | default('unknown') }}.
|
||||
Will stop + disable. Unit file preserved at /etc/systemd/system/llama-server-qwen.service as rollback target.
|
||||
tags: [cutover_stop_qwen, cutover]
|
||||
|
||||
- name: "[cutover] PHASE 1: Stop llama-server-qwen (:8002, bare single-model)"
|
||||
ansible.builtin.systemd:
|
||||
name: llama-server-qwen
|
||||
state: stopped
|
||||
register: cutover_qwen_stopped
|
||||
tags: [cutover_stop_qwen, cutover]
|
||||
|
||||
- name: "[cutover] PHASE 1: Disable llama-server-qwen (prevent auto-start on reboot)"
|
||||
ansible.builtin.systemd:
|
||||
name: llama-server-qwen
|
||||
enabled: false
|
||||
tags: [cutover_stop_qwen, cutover]
|
||||
|
||||
- name: "[cutover] PHASE 1: Wait 5s for VRAM to be released"
|
||||
ansible.builtin.pause:
|
||||
seconds: 5
|
||||
when: cutover_qwen_stopped.changed | default(false)
|
||||
tags: [cutover_stop_qwen, cutover]
|
||||
|
||||
- name: "[cutover] PHASE 1: Verify port 8002 is now free"
|
||||
ansible.builtin.command:
|
||||
cmd: ss -ltnp
|
||||
register: cutover_port_check
|
||||
changed_when: false
|
||||
tags: [cutover_stop_qwen, cutover]
|
||||
|
||||
- name: "[cutover] PHASE 1: Fail if port 8002 is still bound"
|
||||
ansible.builtin.fail:
|
||||
msg: >-
|
||||
Port 8002 is still bound after stopping llama-server-qwen.
|
||||
Check 'ss -ltnp | grep :8002' and resolve before the router can bind.
|
||||
when:
|
||||
- "':8002 ' in (cutover_port_check.stdout | default('')) or ':8002:' in (cutover_port_check.stdout | default(''))"
|
||||
tags: [cutover_stop_qwen, cutover]
|
||||
|
||||
- name: "[cutover] PHASE 1: Report VRAM state (should be empty)"
|
||||
ansible.builtin.command:
|
||||
cmd: nvidia-smi --query-compute-apps=pid,name,used_memory --format=csv,noheader
|
||||
register: cutover_vram_free_check
|
||||
changed_when: false
|
||||
tags: [cutover_stop_qwen, cutover]
|
||||
|
||||
- name: "[cutover] PHASE 1: Print VRAM state"
|
||||
ansible.builtin.debug:
|
||||
msg: >-
|
||||
VRAM after stopping llama-server-qwen:
|
||||
{{ cutover_vram_free_check.stdout if (cutover_vram_free_check.stdout | length > 0)
|
||||
else '(no GPU processes — VRAM free)' }}
|
||||
tags: [cutover_stop_qwen, cutover]
|
||||
|
||||
# =======================================================================
|
||||
# PHASE 2 — Redeploy llama-server-router unit with --port 8002
|
||||
# =======================================================================
|
||||
|
||||
- name: "[cutover] PHASE 2: Deploy llama-server-router unit file (port 8002 — production)"
|
||||
ansible.builtin.template:
|
||||
src: "../roles/llm-inference-multimodel/templates/llama-server-router.service.j2"
|
||||
dest: /etc/systemd/system/llama-server-router.service
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
register: cutover_router_unit_deployed
|
||||
notify:
|
||||
- reload systemd
|
||||
tags: [cutover_deploy_unit, cutover]
|
||||
|
||||
- name: "[cutover] PHASE 2: Flush handlers (daemon-reload before start)"
|
||||
ansible.builtin.meta: flush_handlers
|
||||
tags: [cutover_deploy_unit, cutover]
|
||||
|
||||
# =======================================================================
|
||||
# PHASE 3 — Enable + start llama-server-router on :8002
|
||||
# =======================================================================
|
||||
|
||||
- name: "[cutover] PHASE 3: Enable + start llama-server-router (production, :8002)"
|
||||
ansible.builtin.systemd:
|
||||
name: llama-server-router
|
||||
state: "{{ 'restarted' if (cutover_router_unit_deployed.changed | default(false)) else 'started' }}"
|
||||
enabled: true
|
||||
daemon_reload: true
|
||||
tags: [cutover_start_router, cutover]
|
||||
|
||||
# =======================================================================
|
||||
# PHASE 4 — Validation gates 1-3 (hard gates against now-production :8002)
|
||||
# =======================================================================
|
||||
|
||||
- name: "[cutover] GATE 1a: Wait for router /health on :8002 (up to 5min — cold model load)"
|
||||
ansible.builtin.uri:
|
||||
url: "http://{{ llm_router_bind_address }}:{{ llm_router_port }}/health"
|
||||
status_code: 200
|
||||
register: cutover_health
|
||||
retries: 30
|
||||
delay: 10
|
||||
until: cutover_health.status == 200
|
||||
tags: [cutover_validate, cutover]
|
||||
|
||||
- name: "[cutover] GATE 1a: Trigger model load (router lazy-loads on first request)"
|
||||
ansible.builtin.uri:
|
||||
url: "http://{{ llm_router_bind_address }}:{{ llm_router_port }}/v1/chat/completions"
|
||||
method: POST
|
||||
body_format: json
|
||||
body:
|
||||
model: "{{ llm_router_expected_model_id }}"
|
||||
messages:
|
||||
- role: user
|
||||
content: "Reply with one word: hello"
|
||||
max_tokens: 5
|
||||
temperature: 0.0
|
||||
status_code: 200
|
||||
return_content: true
|
||||
timeout: 300
|
||||
register: cutover_warmup
|
||||
tags: [cutover_validate, cutover]
|
||||
|
||||
- name: "[cutover] GATE 1a: Report warmup"
|
||||
ansible.builtin.debug:
|
||||
msg:
|
||||
- "Model loaded. finish_reason={{ cutover_warmup.json.choices[0].finish_reason | default('unknown') }}"
|
||||
- "Response: {{ cutover_warmup.json.choices[0].message.content | default('(empty)') | truncate(100) }}"
|
||||
tags: [cutover_validate, cutover]
|
||||
|
||||
- name: "[cutover] GATE 1b: Query /v1/models on :8002"
|
||||
ansible.builtin.uri:
|
||||
url: "http://{{ llm_router_bind_address }}:{{ llm_router_port }}/v1/models"
|
||||
status_code: 200
|
||||
return_content: true
|
||||
register: cutover_models
|
||||
tags: [cutover_validate, cutover]
|
||||
|
||||
- name: "[cutover] GATE 1b: Fail if expected model ID not found"
|
||||
ansible.builtin.fail:
|
||||
msg: >-
|
||||
GATE 1 FAIL: '{{ llm_router_expected_model_id }}' not found in /v1/models.
|
||||
Returned: {{ cutover_models.json.data | map(attribute='id') | list }}
|
||||
when:
|
||||
- cutover_models.json.data | selectattr('id', 'equalto', llm_router_expected_model_id) | list | length == 0
|
||||
tags: [cutover_validate, cutover]
|
||||
|
||||
- name: "[cutover] GATE 1b: Extract ctx-size from router model args"
|
||||
ansible.builtin.set_fact:
|
||||
cutover_qwen_n_ctx: >-
|
||||
{%- set model = cutover_models.json.data | selectattr('id', 'equalto', llm_router_expected_model_id) | first -%}
|
||||
{%- set args = model.status.args -%}
|
||||
{%- set ctx_idx = args.index('--ctx-size') if '--ctx-size' in args else -1 -%}
|
||||
{{ args[ctx_idx + 1] | int if ctx_idx >= 0 else 0 }}
|
||||
when:
|
||||
- cutover_models.json.data | selectattr('id', 'equalto', llm_router_expected_model_id) | list | length > 0
|
||||
tags: [cutover_validate, cutover]
|
||||
|
||||
- name: "[cutover] GATE 1b: Fail if n_ctx < 64000"
|
||||
ansible.builtin.fail:
|
||||
msg: "GATE 1 FAIL: --ctx-size={{ cutover_qwen_n_ctx }} < 64000 (Hermes 64K floor)."
|
||||
when:
|
||||
- cutover_qwen_n_ctx is defined
|
||||
- cutover_qwen_n_ctx | int < 64000
|
||||
tags: [cutover_validate, cutover]
|
||||
|
||||
- name: "[cutover] GATE 1b: PASS — n_ctx >= 64K"
|
||||
ansible.builtin.debug:
|
||||
msg: "GATE 1 PASS: --ctx-size={{ cutover_qwen_n_ctx }} >= 64000."
|
||||
when:
|
||||
- cutover_qwen_n_ctx is defined
|
||||
- cutover_qwen_n_ctx | int >= 64000
|
||||
tags: [cutover_validate, cutover]
|
||||
|
||||
# --- Gate 2: Tool-calling through router proxy ---
|
||||
|
||||
- name: "[cutover] GATE 2: Tool-calling probe"
|
||||
ansible.builtin.uri:
|
||||
url: "http://{{ llm_router_bind_address }}:{{ llm_router_port }}/v1/chat/completions"
|
||||
method: POST
|
||||
body_format: json
|
||||
body:
|
||||
model: "{{ llm_router_expected_model_id }}"
|
||||
messages:
|
||||
- role: user
|
||||
content: "What is the current weather in Chicago? Use the provided tool."
|
||||
tools:
|
||||
- type: function
|
||||
function:
|
||||
name: get_weather
|
||||
description: "Get current weather conditions for a city"
|
||||
parameters:
|
||||
type: object
|
||||
properties:
|
||||
city:
|
||||
type: string
|
||||
description: "The city name"
|
||||
required:
|
||||
- city
|
||||
temperature: 0.0
|
||||
status_code: 200
|
||||
return_content: true
|
||||
timeout: 120
|
||||
register: cutover_toolcall_probe
|
||||
tags: [cutover_validate, cutover]
|
||||
|
||||
- name: "[cutover] GATE 2: Fail if not finish_reason=tool_calls"
|
||||
ansible.builtin.fail:
|
||||
msg: >-
|
||||
GATE 2 FAIL: finish_reason={{ cutover_toolcall_probe.json.choices[0].finish_reason | default('(missing)') }}
|
||||
(expected tool_calls). Response: {{ cutover_toolcall_probe.json | to_json }}
|
||||
when:
|
||||
- cutover_toolcall_probe.json.choices[0].finish_reason | default('') != 'tool_calls'
|
||||
tags: [cutover_validate, cutover]
|
||||
|
||||
- name: "[cutover] GATE 2: PASS"
|
||||
ansible.builtin.debug:
|
||||
msg:
|
||||
- "GATE 2 PASS: finish_reason=tool_calls"
|
||||
- "function: {{ cutover_toolcall_probe.json.choices[0].message.tool_calls[0].function.name | default('(unknown)') }}"
|
||||
- "arguments: {{ cutover_toolcall_probe.json.choices[0].message.tool_calls[0].function.arguments | default('(none)') }}"
|
||||
when:
|
||||
- cutover_toolcall_probe.json.choices[0].finish_reason | default('') == 'tool_calls'
|
||||
tags: [cutover_validate, cutover]
|
||||
|
||||
# --- Gate 2b: Hallucination stress ---
|
||||
|
||||
- name: "[cutover] GATE 2b: Hallucination stress probe"
|
||||
ansible.builtin.uri:
|
||||
url: "http://{{ llm_router_bind_address }}:{{ llm_router_port }}/v1/chat/completions"
|
||||
method: POST
|
||||
body_format: json
|
||||
body:
|
||||
model: "{{ llm_router_expected_model_id }}"
|
||||
messages:
|
||||
- role: user
|
||||
content: "Tell me a brief fact about the planet Mars. Do not call any functions."
|
||||
tools:
|
||||
- type: function
|
||||
function:
|
||||
name: get_weather
|
||||
description: "Get current weather conditions for a city"
|
||||
parameters:
|
||||
type: object
|
||||
properties:
|
||||
city:
|
||||
type: string
|
||||
required:
|
||||
- city
|
||||
temperature: 0.1
|
||||
status_code: 200
|
||||
return_content: true
|
||||
timeout: 120
|
||||
register: cutover_halluc_probe
|
||||
tags: [cutover_validate, cutover]
|
||||
|
||||
- name: "[cutover] GATE 2b: Fail if spurious tool_calls"
|
||||
ansible.builtin.fail:
|
||||
msg: >-
|
||||
GATE 2b FAIL: finish_reason=tool_calls on unrelated prompt (Mars fact).
|
||||
Over-triggering through router. Response: {{ cutover_halluc_probe.json | to_json }}
|
||||
when:
|
||||
- cutover_halluc_probe.json.choices[0].finish_reason | default('') == 'tool_calls'
|
||||
tags: [cutover_validate, cutover]
|
||||
|
||||
- name: "[cutover] GATE 2b: PASS"
|
||||
ansible.builtin.debug:
|
||||
msg: "GATE 2b PASS: finish_reason={{ cutover_halluc_probe.json.choices[0].finish_reason }} — no spurious tool_calls."
|
||||
when:
|
||||
- cutover_halluc_probe.json.choices[0].finish_reason | default('') != 'tool_calls'
|
||||
tags: [cutover_validate, cutover]
|
||||
|
||||
# --- Gate 3: VRAM guard ---
|
||||
|
||||
- name: "[cutover] GATE 3: Check VRAM usage (--models-max 1 guard)"
|
||||
ansible.builtin.command:
|
||||
cmd: nvidia-smi --query-gpu=memory.used,memory.total,utilization.gpu --format=csv,noheader
|
||||
register: cutover_vram_post
|
||||
changed_when: false
|
||||
tags: [cutover_validate, cutover]
|
||||
|
||||
- name: "[cutover] GATE 3: Parse VRAM used MiB"
|
||||
ansible.builtin.set_fact:
|
||||
cutover_vram_used_mib: "{{ cutover_vram_post.stdout.split(',')[0].strip().split(' ')[0] | int }}"
|
||||
tags: [cutover_validate, cutover]
|
||||
|
||||
- name: "[cutover] GATE 3: Fail if VRAM exceeds ceiling"
|
||||
ansible.builtin.fail:
|
||||
msg: >-
|
||||
GATE 3 FAIL: {{ cutover_vram_used_mib }} MiB > {{ llm_router_vram_max_mib }} MiB ceiling.
|
||||
Full: {{ cutover_vram_post.stdout }}
|
||||
when:
|
||||
- cutover_vram_used_mib | int > llm_router_vram_max_mib | int
|
||||
tags: [cutover_validate, cutover]
|
||||
|
||||
- name: "[cutover] GATE 3: Count GPU processes"
|
||||
ansible.builtin.command:
|
||||
cmd: nvidia-smi --query-compute-apps=pid,name --format=csv,noheader
|
||||
register: cutover_gpu_procs
|
||||
changed_when: false
|
||||
failed_when: false
|
||||
tags: [cutover_validate, cutover]
|
||||
|
||||
- name: "[cutover] GATE 3: PASS"
|
||||
ansible.builtin.debug:
|
||||
msg:
|
||||
- "GATE 3 PASS: {{ cutover_vram_used_mib }} MiB / {{ llm_router_vram_max_mib }} MiB ceiling."
|
||||
- "GPU processes: {{ cutover_gpu_procs.stdout_lines | default(['(none)']) }}"
|
||||
- "Full nvidia-smi: {{ cutover_vram_post.stdout }}"
|
||||
when:
|
||||
- cutover_vram_used_mib | int <= llm_router_vram_max_mib | int
|
||||
tags: [cutover_validate, cutover]
|
||||
|
||||
# =======================================================================
|
||||
# PHASE 5 — Gate 4: Bundled SvelteKit Web UI (required this time)
|
||||
# =======================================================================
|
||||
|
||||
- name: "[cutover] GATE 4: Check bundled SvelteKit UI at :8002"
|
||||
ansible.builtin.uri:
|
||||
url: "http://{{ llm_router_bind_address }}:{{ llm_router_port }}/"
|
||||
status_code: [200, 301, 302]
|
||||
return_content: true
|
||||
timeout: 30
|
||||
register: cutover_ui_check
|
||||
failed_when: false
|
||||
tags: [cutover_validate, cutover_ui, cutover]
|
||||
|
||||
- name: "[cutover] GATE 4: Inspect UI content"
|
||||
ansible.builtin.set_fact:
|
||||
cutover_ui_is_html: "{{ 'html' in (cutover_ui_check.content | default('') | lower) or '<!doctype' in (cutover_ui_check.content | default('') | lower) }}"
|
||||
cutover_ui_has_model_select: "{{ 'select' in (cutover_ui_check.content | default('') | lower) or 'model' in (cutover_ui_check.content | default('') | lower) }}"
|
||||
when: cutover_ui_check is defined
|
||||
tags: [cutover_validate, cutover_ui, cutover]
|
||||
|
||||
- name: "[cutover] GATE 4: Report UI check and bookmark URL"
|
||||
ansible.builtin.debug:
|
||||
msg:
|
||||
- "======================================================================"
|
||||
- "GATE 4 UI CHECK:"
|
||||
- " HTTP status: {{ cutover_ui_check.status | default('UNREACHABLE') }}"
|
||||
- " Is HTML: {{ cutover_ui_is_html | default(false) }}"
|
||||
- " Contains model/select: {{ cutover_ui_has_model_select | default(false) }}"
|
||||
- " BOOKMARK URL: http://{{ llm_router_bind_address }}:{{ llm_router_port }}/"
|
||||
- " {{ 'GATE 4 PASS — UI serving HTML at :8002.' if (cutover_ui_check.status | default(0) | int in [200, 301, 302]) else 'GATE 4 WARN — UI not reachable (HTTP ' + (cutover_ui_check.status | default('FAIL') | string) + ').' }}"
|
||||
- "======================================================================"
|
||||
when: cutover_ui_check is defined
|
||||
tags: [cutover_validate, cutover_ui, cutover]
|
||||
|
||||
# =======================================================================
|
||||
# CUTOVER SUMMARY
|
||||
# =======================================================================
|
||||
|
||||
- name: "[cutover] CUTOVER SUMMARY — production promoted"
|
||||
ansible.builtin.debug:
|
||||
msg:
|
||||
- "======================================================================"
|
||||
- "CUTOVER COMPLETE: llama-server-router is now production."
|
||||
- ""
|
||||
- " Service: llama-server-router.service (enabled, running)"
|
||||
- " Port: 8002 (unchanged for all 8 Hermes profiles)"
|
||||
- " Model: {{ llm_router_expected_model_id }}"
|
||||
- " Mode: Router/supervisor (--models-dir /opt/models, --models-max 1)"
|
||||
- ""
|
||||
- " Gate 1 (n_ctx >= 64K): PASS ({{ cutover_qwen_n_ctx | default('N/A') }})"
|
||||
- " Gate 2 (tool-calling): PASS (finish_reason=tool_calls)"
|
||||
- " Gate 2b (halluc stress): PASS (no spurious tool_calls)"
|
||||
- " Gate 3 (VRAM <= 23000MiB): PASS ({{ cutover_vram_used_mib | default('N/A') }} MiB)"
|
||||
- " Gate 4 (Web UI): HTTP {{ cutover_ui_check.status | default('N/A') }}"
|
||||
- ""
|
||||
- " ROLLBACK TARGET: /etc/systemd/system/llama-server-qwen.service (unit preserved)"
|
||||
- " ROLLBACK CMD: sudo systemctl enable --now llama-server-qwen"
|
||||
- " sudo systemctl disable --now llama-server-router"
|
||||
- " Or: ansible-playbook -i inventory.yml day2_cutover_qwen_to_router.yml --tags cutover_rollback"
|
||||
- ""
|
||||
- " Web UI bookmark: http://{{ llm_router_bind_address }}:{{ llm_router_port }}/"
|
||||
- "======================================================================"
|
||||
tags: [cutover]
|
||||
|
||||
# =======================================================================
|
||||
# ROLLBACK — tag cutover_rollback reverses the cutover
|
||||
# Run: ansible-playbook -i inventory.yml day2_cutover_qwen_to_router.yml --tags cutover_rollback
|
||||
# WARNING: rollback_task has no dependency on cutover tags — safe to run standalone.
|
||||
# =======================================================================
|
||||
|
||||
- name: "[cutover_rollback] Stop + disable llama-server-router"
|
||||
ansible.builtin.systemd:
|
||||
name: llama-server-router
|
||||
state: stopped
|
||||
enabled: false
|
||||
tags: [cutover_rollback, never] # 'never' = only runs with explicit --tags cutover_rollback
|
||||
|
||||
- name: "[cutover_rollback] Enable + start llama-server-qwen (restore bare :8002)"
|
||||
ansible.builtin.systemd:
|
||||
name: llama-server-qwen
|
||||
state: started
|
||||
enabled: true
|
||||
tags: [cutover_rollback, never]
|
||||
|
||||
- name: "[cutover_rollback] Verify rollback /health"
|
||||
ansible.builtin.uri:
|
||||
url: "http://{{ llm_bind_address | default('10.1.71.130') }}:8002/health"
|
||||
status_code: 200
|
||||
timeout: 30
|
||||
register: cutover_rollback_health
|
||||
failed_when: false
|
||||
tags: [cutover_rollback, never]
|
||||
|
||||
- name: "[cutover_rollback] Report rollback result"
|
||||
ansible.builtin.debug:
|
||||
msg: >-
|
||||
ROLLBACK: llama-server-qwen :8002 health returned
|
||||
{{ cutover_rollback_health.status | default('UNREACHABLE') }}.
|
||||
{{ 'OK — production restored to bare qwen.' if (cutover_rollback_health.status | default(0) | int == 200)
|
||||
else 'WARNING — health check failed. Check manually.' }}
|
||||
tags: [cutover_rollback, never]
|
||||
|
||||
handlers:
|
||||
- name: reload systemd
|
||||
ansible.builtin.systemd:
|
||||
daemon_reload: true
|
||||
277
ansible/playbooks/day2_per_model_ctx_size.yml
Normal file
277
ansible/playbooks/day2_per_model_ctx_size.yml
Normal file
@@ -0,0 +1,277 @@
|
||||
---
|
||||
# ------------------------------------------------------------------------------
|
||||
# FILE: playbooks/day2_per_model_ctx_size.yml
|
||||
# DESCRIPTION: Right-size --ctx-size per model workload on llama-server-router
|
||||
# (already in --models-preset mode since t_9adf0889).
|
||||
#
|
||||
# Context (t_ryan_per_model_ctx, 2026-08-13, requested by Ryan via JARVIS):
|
||||
# All 3 preset models currently launch with a uniform --ctx-size 65536.
|
||||
# This playbook narrows two of them to match actual workload:
|
||||
# - Meta-Llama-3.1-8B-Instruct-Q4_K_M (alias Meta-Llama-3.1-8B-Instruct-4bit):
|
||||
# ctx-size 65536 -> 8192 (tool-routing / micro-tasks: title gen, MCP
|
||||
# tool calls, approval checks)
|
||||
# - Phi-3.5-mini-instruct-Q8_0 (alias Phi-3.5-mini-instruct-8bit):
|
||||
# ctx-size 65536 -> 32768 (long web scrapes / session-log compression)
|
||||
# Both also move flash-attn from "auto" to explicit "true" per Ryan's spec.
|
||||
# Qwen3.6-35B-A3B-UD-Q4_K_S is INTENTIONALLY left untouched at 65536/auto.
|
||||
#
|
||||
# Existing aliases (Meta-Llama-3.1-8B-Instruct-4bit, Phi-3.5-mini-instruct-8bit)
|
||||
# are PRESERVED as-is. Ryan's pasted TOML used different alias strings
|
||||
# ("llama-3.1-8b", "phi-3.5-mini") but renaming aliases was not explicitly
|
||||
# requested and would break live Hermes custom_providers routing — flagged
|
||||
# in the deployment report rather than applied silently.
|
||||
#
|
||||
# IMPORTANT — Hermes side effect: /home/hermes/.hermes/config.yaml declares
|
||||
# context_length: 65536 for both these models under custom_providers. This
|
||||
# playbook does NOT touch that file (out of role/agent scope) but the value
|
||||
# becomes STALE the moment this playbook lands. Flag to JARVIS/Maria Hill.
|
||||
#
|
||||
# Usage (from ~/git/homelab/ansible):
|
||||
# ansible-playbook -i inventory.yml playbooks/day2_per_model_ctx_size.yml
|
||||
#
|
||||
# Author: War Machine (2026-08-13, t_ryan_per_model_ctx)
|
||||
# ------------------------------------------------------------------------------
|
||||
|
||||
- name: "Right-size per-model ctx-size on llama-server-router (Llama 8k, Phi 32k)"
|
||||
hosts: astro_orbiter
|
||||
gather_facts: false
|
||||
become: true
|
||||
|
||||
vars:
|
||||
# Preset mode already active in production (t_9adf0889) — keep it on.
|
||||
llm_router_preset_enabled: true
|
||||
llm_router_preset_path: /opt/llama-server-router-preset.ini
|
||||
llm_router_enabled: true
|
||||
|
||||
# Production port
|
||||
llm_router_port: 8002
|
||||
llm_router_bind_address: "10.1.71.130"
|
||||
llm_router_allowed_source_cidr: "10.1.70.0/24"
|
||||
llm_bind_address: "10.1.71.130"
|
||||
llm_allowed_source_cidr: "10.1.70.0/24"
|
||||
|
||||
llm_service_user: jarvis
|
||||
llm_binary_path: /opt/llama.cpp/build/bin/llama-server
|
||||
llm_models_dir: /opt/models
|
||||
llm_router_service_name: llama-server-router
|
||||
llm_router_models_dir: /opt/models
|
||||
llm_router_models_max: 4
|
||||
llm_router_parallel: 1
|
||||
llm_router_gpu_layers: 99
|
||||
llm_router_batch_size: 2048
|
||||
llm_router_ubatch_size: 512
|
||||
llm_router_cache_type_k: q4_0
|
||||
llm_router_cache_type_v: q4_0
|
||||
|
||||
# Qwen — untouched baseline (also used as router-wide fallback default)
|
||||
llm_router_ctx_size: 65536
|
||||
llm_router_flash_attn: "auto"
|
||||
llm_router_expected_model_id: "Qwen3.6-35B-A3B-UD-Q4_K_S"
|
||||
llm_router_vram_max_mib: 23000
|
||||
|
||||
# --- THE CHANGE: per-model overrides ---
|
||||
llm_router_llama_ctx_size: 8192
|
||||
llm_router_llama_flash_attn: "true"
|
||||
llm_router_phi_ctx_size: 32768
|
||||
llm_router_phi_flash_attn: "true"
|
||||
|
||||
handlers:
|
||||
- name: reload systemd
|
||||
ansible.builtin.systemd:
|
||||
daemon_reload: true
|
||||
become: true
|
||||
listen: "reload systemd"
|
||||
|
||||
- name: restart router
|
||||
ansible.builtin.systemd:
|
||||
name: llama-server-router
|
||||
state: restarted
|
||||
become: true
|
||||
listen: "restart router"
|
||||
|
||||
tasks:
|
||||
|
||||
# ==========================================================================
|
||||
# PHASE 1: Deploy the preset INI with new per-model ctx-size/flash-attn
|
||||
# ==========================================================================
|
||||
|
||||
- name: "[ctx-resize] Deploy preset INI to {{ llm_router_preset_path }}"
|
||||
ansible.builtin.template:
|
||||
src: "../roles/llm-inference-multimodel/templates/llama-server-router-preset.ini.j2"
|
||||
dest: "{{ llm_router_preset_path }}"
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
register: ctx_resize_preset_deployed
|
||||
notify:
|
||||
- restart router
|
||||
|
||||
- name: "[ctx-resize] Deploy router systemd unit (drop global --ctx-size/--flash-attn in preset mode)"
|
||||
ansible.builtin.template:
|
||||
src: "../roles/llm-inference-multimodel/templates/llama-server-router.service.j2"
|
||||
dest: /etc/systemd/system/llama-server-router.service
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
register: ctx_resize_unit_deployed
|
||||
notify:
|
||||
- reload systemd
|
||||
- restart router
|
||||
|
||||
- name: "[ctx-resize] Flush handlers (daemon-reload + router restart if changed)"
|
||||
ansible.builtin.meta: flush_handlers
|
||||
|
||||
# ==========================================================================
|
||||
# PHASE 2: Verify
|
||||
# ==========================================================================
|
||||
|
||||
- name: "[ctx-resize] Wait for /health"
|
||||
ansible.builtin.uri:
|
||||
url: "http://{{ llm_router_bind_address }}:{{ llm_router_port }}/health"
|
||||
status_code: 200
|
||||
timeout: 30
|
||||
retries: 12
|
||||
delay: 5
|
||||
register: ctx_resize_health
|
||||
until: ctx_resize_health.status == 200
|
||||
|
||||
- name: "[ctx-resize] Query /v1/models"
|
||||
ansible.builtin.uri:
|
||||
url: "http://{{ llm_router_bind_address }}:{{ llm_router_port }}/v1/models"
|
||||
status_code: 200
|
||||
return_content: true
|
||||
timeout: 30
|
||||
register: ctx_resize_models
|
||||
|
||||
- name: "[ctx-resize] Trigger load — Llama (confirms actual load + captures live args)"
|
||||
ansible.builtin.uri:
|
||||
url: "http://{{ llm_router_bind_address }}:{{ llm_router_port }}/v1/chat/completions"
|
||||
method: POST
|
||||
body_format: json
|
||||
body:
|
||||
model: "Meta-Llama-3.1-8B-Instruct-Q4_K_M"
|
||||
messages:
|
||||
- role: user
|
||||
content: "Reply with one word: hello"
|
||||
max_tokens: 5
|
||||
temperature: 0.0
|
||||
status_code: 200
|
||||
return_content: true
|
||||
timeout: 120
|
||||
register: ctx_resize_llama_warmup
|
||||
|
||||
- name: "[ctx-resize] Trigger load — Phi (confirms actual load + captures live args)"
|
||||
ansible.builtin.uri:
|
||||
url: "http://{{ llm_router_bind_address }}:{{ llm_router_port }}/v1/chat/completions"
|
||||
method: POST
|
||||
body_format: json
|
||||
body:
|
||||
model: "Phi-3.5-mini-instruct-Q8_0"
|
||||
messages:
|
||||
- role: user
|
||||
content: "Reply with one word: hello"
|
||||
max_tokens: 5
|
||||
temperature: 0.0
|
||||
status_code: 200
|
||||
return_content: true
|
||||
timeout: 120
|
||||
register: ctx_resize_phi_warmup
|
||||
|
||||
- name: "[ctx-resize] Re-query /v1/models after warmup (final state)"
|
||||
ansible.builtin.uri:
|
||||
url: "http://{{ llm_router_bind_address }}:{{ llm_router_port }}/v1/models"
|
||||
status_code: 200
|
||||
return_content: true
|
||||
timeout: 30
|
||||
register: ctx_resize_models_final
|
||||
|
||||
- name: "[ctx-resize] Extract Llama args"
|
||||
ansible.builtin.set_fact:
|
||||
ctx_resize_llama_args: >-
|
||||
{{ (ctx_resize_models_final.json.data | selectattr('id', 'equalto', 'Meta-Llama-3.1-8B-Instruct-Q4_K_M') | first).status.args }}
|
||||
ctx_resize_llama_status: >-
|
||||
{{ (ctx_resize_models_final.json.data | selectattr('id', 'equalto', 'Meta-Llama-3.1-8B-Instruct-Q4_K_M') | first).status.value }}
|
||||
|
||||
- name: "[ctx-resize] Extract Phi args"
|
||||
ansible.builtin.set_fact:
|
||||
ctx_resize_phi_args: >-
|
||||
{{ (ctx_resize_models_final.json.data | selectattr('id', 'equalto', 'Phi-3.5-mini-instruct-Q8_0') | first).status.args }}
|
||||
ctx_resize_phi_status: >-
|
||||
{{ (ctx_resize_models_final.json.data | selectattr('id', 'equalto', 'Phi-3.5-mini-instruct-Q8_0') | first).status.value }}
|
||||
|
||||
- name: "[ctx-resize] Extract Qwen args (must be unchanged)"
|
||||
ansible.builtin.set_fact:
|
||||
ctx_resize_qwen_args: >-
|
||||
{{ (ctx_resize_models_final.json.data | selectattr('id', 'equalto', 'Qwen3.6-35B-A3B-UD-Q4_K_S') | first).status.args }}
|
||||
|
||||
- name: "[ctx-resize] GATE — Llama ctx-size must be 8192"
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- "'8192' in ctx_resize_llama_args"
|
||||
- ctx_resize_llama_args[ctx_resize_llama_args.index('--ctx-size') + 1] == '8192'
|
||||
fail_msg: "Llama ctx-size not 8192. Args: {{ ctx_resize_llama_args }}"
|
||||
success_msg: "Llama ctx-size confirmed 8192."
|
||||
|
||||
- name: "[ctx-resize] GATE — Llama flash-attn must be true"
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- ctx_resize_llama_args[ctx_resize_llama_args.index('--flash-attn') + 1] == 'true'
|
||||
fail_msg: "Llama flash-attn not true. Args: {{ ctx_resize_llama_args }}"
|
||||
success_msg: "Llama flash-attn confirmed true."
|
||||
|
||||
- name: "[ctx-resize] GATE — Llama loaded successfully"
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- ctx_resize_llama_status == 'loaded'
|
||||
fail_msg: "Llama status is '{{ ctx_resize_llama_status }}', expected 'loaded'."
|
||||
success_msg: "Llama status confirmed 'loaded'."
|
||||
|
||||
- name: "[ctx-resize] GATE — Phi ctx-size must be 32768"
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- ctx_resize_phi_args[ctx_resize_phi_args.index('--ctx-size') + 1] == '32768'
|
||||
fail_msg: "Phi ctx-size not 32768. Args: {{ ctx_resize_phi_args }}"
|
||||
success_msg: "Phi ctx-size confirmed 32768."
|
||||
|
||||
- name: "[ctx-resize] GATE — Phi flash-attn must be true"
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- ctx_resize_phi_args[ctx_resize_phi_args.index('--flash-attn') + 1] == 'true'
|
||||
fail_msg: "Phi flash-attn not true. Args: {{ ctx_resize_phi_args }}"
|
||||
success_msg: "Phi flash-attn confirmed true."
|
||||
|
||||
- name: "[ctx-resize] GATE — Phi loaded successfully"
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- ctx_resize_phi_status == 'loaded'
|
||||
fail_msg: "Phi status is '{{ ctx_resize_phi_status }}', expected 'loaded'."
|
||||
success_msg: "Phi status confirmed 'loaded'."
|
||||
|
||||
- name: "[ctx-resize] GATE — Qwen ctx-size UNCHANGED at 65536"
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- ctx_resize_qwen_args[ctx_resize_qwen_args.index('--ctx-size') + 1] == '65536'
|
||||
fail_msg: "Qwen ctx-size changed unexpectedly! Args: {{ ctx_resize_qwen_args }}"
|
||||
success_msg: "Qwen ctx-size confirmed UNCHANGED at 65536."
|
||||
|
||||
- name: "[ctx-resize] PASS — summary"
|
||||
ansible.builtin.debug:
|
||||
msg:
|
||||
- "================================================================"
|
||||
- "PER-MODEL CTX-SIZE DEPLOYMENT — COMPLETE"
|
||||
- ""
|
||||
- " Llama-3.1-8B (Meta-Llama-3.1-8B-Instruct-Q4_K_M):"
|
||||
- " status: {{ ctx_resize_llama_status }}"
|
||||
- " args: {{ ctx_resize_llama_args }}"
|
||||
- ""
|
||||
- " Phi-3.5-mini (Phi-3.5-mini-instruct-Q8_0):"
|
||||
- " status: {{ ctx_resize_phi_status }}"
|
||||
- " args: {{ ctx_resize_phi_args }}"
|
||||
- ""
|
||||
- " Qwen3.6-35B-A3B-UD-Q4_K_S: UNCHANGED (ctx-size 65536, args: {{ ctx_resize_qwen_args }})"
|
||||
- ""
|
||||
- " ACTION NEEDED: /home/hermes/.hermes/config.yaml custom_providers"
|
||||
- " context_length: 65536 for both Meta-Llama-3.1-8B-Instruct-4bit and"
|
||||
- " Phi-3.5-mini-instruct-8bit is now STALE (actual: 8192 / 32768)."
|
||||
- " Flag to JARVIS/Maria Hill for correction — NOT done by this playbook."
|
||||
- "================================================================"
|
||||
38
ansible/playbooks/day2_qwen38_ctx128k.yml
Normal file
38
ansible/playbooks/day2_qwen38_ctx128k.yml
Normal file
@@ -0,0 +1,38 @@
|
||||
---
|
||||
# ------------------------------------------------------------------------------
|
||||
# Playbook: day2_qwen38_ctx128k.yml
|
||||
# Purpose: Bump Qwen3.8-27B-Q4_K_M ctx-size from 32768 to 131072 (128K)
|
||||
# on astro-orbiter's production router (port 8002).
|
||||
#
|
||||
# What this playbook does:
|
||||
# 1. Renders the updated llama-server-router-preset.ini.j2 (now with
|
||||
# llm_router_qwen38_ctx_size: 131072) to /opt/llama-server-router-preset.ini.
|
||||
# 2. Restarts llama-server-router.service.
|
||||
# 3. Verifies the router loads Qwen3.8-27B at ctx=131072 in status.args.
|
||||
#
|
||||
# Context:
|
||||
# - Empirical VRAM test (t_4455a44c): 131072 ctx = 20,282 MiB Qwen3.8
|
||||
# + 558 MiB nomic-embed = ~20.8GB total; ~3.2GB headroom on 24GB RTX 3090.
|
||||
# Co-resident with nomic-embed: comfortably fits.
|
||||
# - Ryan approved this deployment.
|
||||
# - Semaphore SSH gap for astro-orbiter still applies (t_730f9584 / t_33acbb2e);
|
||||
# running direct CLI Ansible per standing exception.
|
||||
#
|
||||
# Run:
|
||||
# cd /home/hermes/git/homelab/ansible
|
||||
# env -u ANSIBLE_VAULT_PASSWORD_FILE ansible-playbook \
|
||||
# -i inventory.yml \
|
||||
# playbooks/day2_qwen38_ctx128k.yml
|
||||
#
|
||||
# Task reference: t_441470b9 — War Machine, 2026-08-16
|
||||
# ------------------------------------------------------------------------------
|
||||
- name: Bump Qwen3.8-27B ctx-size to 131072 on astro-orbiter
|
||||
hosts: astro-orbiter
|
||||
become: true
|
||||
vars:
|
||||
llm_router_preset_enabled: true
|
||||
llm_router_qwen38_ctx_size: 131072
|
||||
|
||||
roles:
|
||||
- role: llm-inference-multimodel
|
||||
tags: [preset, systemd, verify]
|
||||
65
ansible/playbooks/day2_qwen38_ctx128k_rollback.yml
Normal file
65
ansible/playbooks/day2_qwen38_ctx128k_rollback.yml
Normal file
@@ -0,0 +1,65 @@
|
||||
---
|
||||
# ------------------------------------------------------------------------------
|
||||
# Playbook: day2_qwen38_ctx128k_rollback.yml
|
||||
# Purpose: Roll back Qwen3.8-27B-Q4_K_M ctx-size from 131072 back to 65536
|
||||
# on astro-orbiter's production router (port 8002).
|
||||
#
|
||||
# What this playbook does:
|
||||
# 1. Renders the updated llama-server-router-preset.ini.j2 (now with
|
||||
# llm_router_qwen38_ctx_size: 65536) to
|
||||
# /opt/llama-server-router-preset.ini.
|
||||
# 2. Restarts llama-server-router.service.
|
||||
# 3. Verifies the router loads Qwen3.8-27B at ctx=65536 in status.args.
|
||||
#
|
||||
# Context:
|
||||
# - t_441470b9 (2026-08-16): ctx-size bumped 32768 -> 131072. Verified VRAM
|
||||
# at 131072 ctx with only Qwen3.8 + nomic-embed co-resident: ~20,282 MiB
|
||||
# + 558 MiB = ~20.8 GB on 24 GB RTX 3090. Comfortably safe.
|
||||
# - t_72646029 (2026-08-17): Phi-3.5mini moved to GPU (n-gpu-layers=99)
|
||||
# to enable concurrent residency with CPU-offloaded Coder-14B and
|
||||
# Llama-3.1-8B. This added ~2GB CUDA context buffers for Phi + shifted
|
||||
# Phi's model weights onto the GPU (~3.8GB).
|
||||
# - NEW steady-state VRAM: Qwen3.8 @ 131072 ctx (~20,282 MiB) + nomic-embed
|
||||
# (~558 MiB) + Llama CUDA ctx (~1,706 MiB) + Coder CUDA ctx (~1,390 MiB)
|
||||
# = ~24,004 MiB. Adding Phi-3.5 (~3,800 MiB weights + ~1.4 GB CUDA ctx)
|
||||
# pushes total to ~29,000+ MiB — exceeding the 24,576 MiB RTX 3090 limit.
|
||||
# Qwen3.8-27B-131072 now fails to load (HTTP 500, OOM before llama.cpp
|
||||
# reaches the model-loading phase).
|
||||
# - FIX: reduce Qwen3.8 ctx-size 131072 -> 65536. This reduces KV cache
|
||||
# from ~6GB to ~3GB, freeing ~3GB of VRAM. New estimated steady-state:
|
||||
# Qwen3.8 @ 65536 ctx (~17,068 MiB) + nomic (~558) + Llama ctx (~1,706)
|
||||
# + Coder ctx (~1,390) + Phi-3.5 (~3,800 + ~1,400 CUDA ctx) = ~25,922 MiB.
|
||||
# Still over 24,576 — see "Phase 2" below for the secondary fix.
|
||||
#
|
||||
# IMPORTANT: Rolling back ctx-size alone may NOT be sufficient. The
|
||||
# hardware reference (astro-orbiter-hardware.md line 166, t_72646029)
|
||||
# states steady-state ~24,004 MiB WITHOUT Phi on GPU. Adding Phi-3.5 back
|
||||
# to GPU tips it over. This playbook handles the context rollback; if Qwen3.8
|
||||
# still fails to load after Phase R, Wong should escalate to Ryan for a
|
||||
# decision on either (a) offloading Phi-3.5mini to CPU (n-gpu-layers=0),
|
||||
# or (b) adding a second GPU. Document the Phase 2 finding as a separate
|
||||
# follow-up task if needed.
|
||||
#
|
||||
# The 64K floor from the 2026-08-12 cutover validation (t_cd0d5388, Gate 1)
|
||||
# still applies — ctx-size=65536 satisfies it.
|
||||
#
|
||||
# Run:
|
||||
# cd /home/hermes/git/homelab/ansible
|
||||
# env -u ANSIBLE_VAULT_PASSWORD_FILE ansible-playbook \
|
||||
# -i inventory.yml \
|
||||
# playbooks/day2_qwen38_ctx128k_rollback.yml
|
||||
#
|
||||
# Task reference: t_c9fed26c — War Machine benchmark, 2026-08-18
|
||||
# Root cause: t_72646029 CPU-offload deployment added Phi-3.5 to GPU,
|
||||
# shifting total VRAM past the 24,576 MiB ceiling when Qwen3.8 runs at 128K.
|
||||
# ------------------------------------------------------------------------------
|
||||
- name: Roll back Qwen3.8-27B ctx-size to 65536 on astro-orbiter
|
||||
hosts: astro-orbiter
|
||||
become: true
|
||||
vars:
|
||||
llm_router_preset_enabled: true
|
||||
llm_router_qwen38_ctx_size: 65536
|
||||
|
||||
roles:
|
||||
- role: llm-inference-multimodel
|
||||
tags: [preset, systemd, verify]
|
||||
36
ansible/playbooks/day2_swap_qwen38.yml
Normal file
36
ansible/playbooks/day2_swap_qwen38.yml
Normal file
@@ -0,0 +1,36 @@
|
||||
---
|
||||
# ------------------------------------------------------------------------------
|
||||
# Playbook: day2_swap_qwen38.yml
|
||||
# Purpose: Swap the primary production model on astro-orbiter router from
|
||||
# Qwen3.6-35B-A3B-UD-Q4_K_S to Qwen3.8-27B-Q4_K_M.
|
||||
# This is a GitOps-encoded record of the swap performed 2026-08-16
|
||||
# per Ryan's direction (kanban task t_f5f7e9ad).
|
||||
#
|
||||
# What this playbook does:
|
||||
# 1. Renders the updated llama-server-router-preset.ini.j2 to
|
||||
# /opt/llama-server-router-preset.ini on astro-orbiter.
|
||||
# 2. Reloads the llama-server-router service (SIGHUP / restart as needed).
|
||||
# 3. Verifies the new model ID appears in /v1/models.
|
||||
#
|
||||
# Prerequisites:
|
||||
# - Qwen3.8-27B-Q4_K_M.gguf must be present in /opt/models on astro-orbiter.
|
||||
# (Downloaded out-of-band via wget during the swap task.)
|
||||
# - roles/llm-inference-multimodel/defaults/main.yml updated to reference
|
||||
# Qwen3.8-27B-Q4_K_M (done in this same commit).
|
||||
#
|
||||
# Run:
|
||||
# env -u ANSIBLE_VAULT_PASSWORD_FILE ansible-playbook \
|
||||
# -i inventory.yml \
|
||||
# playbooks/day2_swap_qwen38.yml
|
||||
#
|
||||
# Task reference: t_f5f7e9ad — War Machine, 2026-08-16
|
||||
# ------------------------------------------------------------------------------
|
||||
- name: Swap primary model to Qwen3.8-27B-Q4_K_M on astro-orbiter
|
||||
hosts: astro-orbiter
|
||||
become: true
|
||||
vars:
|
||||
llm_router_preset_enabled: true
|
||||
|
||||
roles:
|
||||
- role: llm-inference-multimodel
|
||||
tags: [preset, systemd, verify]
|
||||
22
ansible/playbooks/day3_deploy_qwen38_ctx131k.yml
Normal file
22
ansible/playbooks/day3_deploy_qwen38_ctx131k.yml
Normal file
@@ -0,0 +1,22 @@
|
||||
---
|
||||
# Playbook: day3_deploy_qwen38_ctx131k.yml
|
||||
# Purpose: Deploy Qwen3.8-27B-Q4_K_M ctx-size 65536 -> 131072 to astro-orbiter
|
||||
# via llama-swap config re-render + restart.
|
||||
#
|
||||
# The git change to defaults/main.yml (line 235: ctx_size: 131072) is already staged.
|
||||
# This playbook renders /etc/llama-swap/config.yaml from the updated defaults
|
||||
# and restarts llama-swap to load the new ctx-size.
|
||||
#
|
||||
# Run:
|
||||
# cd /home/hermes/git/homelab/ansible
|
||||
# ansible-playbook -i inventory.yml playbooks/day3_deploy_qwen38_ctx131k.yml
|
||||
#
|
||||
- name: Deploy Qwen3.8 ctx-size 131072 to astro-orbiter
|
||||
hosts: astro-orbiter
|
||||
become: true
|
||||
vars:
|
||||
llm_swapmode_enabled: true
|
||||
|
||||
roles:
|
||||
- role: llm-inference-multimodel
|
||||
tags: [swapmode_config, swapmode_systemd, swapmode_verify]
|
||||
@@ -1,77 +0,0 @@
|
||||
---
|
||||
# ------------------------------------------------------------------------------
|
||||
# FILE: roles/common/tasks/main.yml
|
||||
# DESCRIPTION: Baseline configuration applied to all managed Ubuntu hosts.
|
||||
# Handles hostname, timezone, core packages, NTP, ansible user,
|
||||
# and optional LVM root volume expansion.
|
||||
# ------------------------------------------------------------------------------
|
||||
|
||||
- name: Set hostname
|
||||
hostname:
|
||||
name: "{{ inventory_hostname | replace('_', '-') }}"
|
||||
|
||||
- name: Set timezone
|
||||
timezone:
|
||||
name: "{{ common_timezone }}"
|
||||
|
||||
- name: Update apt cache
|
||||
apt:
|
||||
update_cache: yes
|
||||
cache_valid_time: 3600
|
||||
|
||||
- name: Install base utility packages
|
||||
apt:
|
||||
name: "{{ common_packages }}"
|
||||
state: present
|
||||
|
||||
- name: Install chrony
|
||||
apt:
|
||||
name: chrony
|
||||
state: present
|
||||
|
||||
- name: Configure chrony to use sundial
|
||||
template:
|
||||
src: chrony.conf.j2
|
||||
dest: /etc/chrony/chrony.conf
|
||||
mode: '0644'
|
||||
notify: restart chrony
|
||||
|
||||
- name: Ensure chrony is enabled and running
|
||||
systemd:
|
||||
name: chrony
|
||||
state: started
|
||||
enabled: yes
|
||||
|
||||
- name: Ensure ansible user has passwordless sudo
|
||||
lineinfile:
|
||||
path: /etc/sudoers.d/{{ ansible_user }}
|
||||
line: "{{ ansible_user }} ALL=(ALL) NOPASSWD:ALL"
|
||||
create: yes
|
||||
mode: '0440'
|
||||
validate: 'visudo -cf %s'
|
||||
|
||||
# ------------------------------------------------------------------------------
|
||||
# LVM root volume expansion
|
||||
# Extends the root PV to the full disk size and grows the LV + filesystem.
|
||||
# This codifies what was previously done manually after provisioning.
|
||||
# Runs only when common_expand_root_lvm is true (default: true).
|
||||
# Safe to re-run — pvresize and lvextend are idempotent when already at max.
|
||||
# ------------------------------------------------------------------------------
|
||||
|
||||
- name: Expand root PV to full disk size
|
||||
command: pvresize {{ common_root_pv }}
|
||||
register: pvresize_result
|
||||
changed_when: "'changed' in pvresize_result.stdout or pvresize_result.rc == 0"
|
||||
when: common_expand_root_lvm | bool
|
||||
|
||||
- name: Extend root LV to 100% of free VG space
|
||||
lvol:
|
||||
vg: "{{ common_root_vg }}"
|
||||
lv: "{{ common_root_lv }}"
|
||||
size: +100%FREE
|
||||
resizefs: yes
|
||||
when:
|
||||
- common_expand_root_lvm | bool
|
||||
ignore_errors: yes
|
||||
# ignore_errors because lvextend returns non-zero when already at max size.
|
||||
# resizefs: yes handles the resize2fs call inline — no separate task needed.
|
||||
@@ -1,5 +0,0 @@
|
||||
---
|
||||
# file: roles/semaphore/defaults/main.yml
|
||||
|
||||
semaphore_compose_dir: /opt/docker/semaphore
|
||||
semaphore_ssh_key_file: "~/.ssh/ansible"
|
||||
@@ -1,7 +0,0 @@
|
||||
---
|
||||
# file: roles/semaphore/handlers/main.yml
|
||||
|
||||
- name: Restart Semaphore
|
||||
community.docker.docker_compose_v2:
|
||||
project_src: "{{ semaphore_compose_dir }}"
|
||||
state: restarted
|
||||
@@ -1,50 +0,0 @@
|
||||
---
|
||||
# file: roles/semaphore/tasks/main.yml
|
||||
|
||||
- name: Create Semaphore directory
|
||||
ansible.builtin.file:
|
||||
path: "{{ semaphore_compose_dir }}"
|
||||
state: directory
|
||||
owner: wed
|
||||
group: docker
|
||||
mode: '0755'
|
||||
|
||||
- name: Copy Compose file from boilerplate
|
||||
ansible.builtin.copy:
|
||||
src: "{{ playbook_dir }}/../../boilerplates/semaphore/compose.yaml"
|
||||
dest: "{{ semaphore_compose_dir }}/compose.yaml"
|
||||
owner: wed
|
||||
group: docker
|
||||
mode: '0644'
|
||||
|
||||
- name: Deploy Semaphore .env file
|
||||
ansible.builtin.template:
|
||||
src: env.j2
|
||||
dest: "{{ semaphore_compose_dir }}/.env"
|
||||
owner: wed
|
||||
group: docker
|
||||
mode: '0600'
|
||||
|
||||
- name: Copy SSH key for Ansible authentication
|
||||
ansible.builtin.copy:
|
||||
src: "{{ semaphore_ssh_key_file }}"
|
||||
dest: "{{ semaphore_compose_dir }}/ansible_key"
|
||||
owner: "1001"
|
||||
group: "1001"
|
||||
mode: '0600'
|
||||
|
||||
- name: Start Semaphore containers
|
||||
community.docker.docker_compose_v2:
|
||||
project_src: "{{ semaphore_compose_dir }}"
|
||||
state: present
|
||||
register: semaphore_compose
|
||||
|
||||
- name: Wait for Semaphore to be ready
|
||||
ansible.builtin.uri:
|
||||
url: "http://localhost:3000/api/ping"
|
||||
method: GET
|
||||
status_code: 200
|
||||
register: semaphore_health
|
||||
retries: 48
|
||||
delay: 5
|
||||
until: semaphore_health.status == 200
|
||||
@@ -1,3 +0,0 @@
|
||||
DATABASE_PASSWORD={{ vault_semaphore_database_password }}
|
||||
SEMAPHORE_ADMIN_PASSWORD={{ vault_semaphore_admin_password }}
|
||||
SEMAPHORE_ACCESS_KEY_ENCRYPTION={{ vault_semaphore_access_key_encryption }}
|
||||
@@ -1,21 +1,6 @@
|
||||
# requirements.yml
|
||||
---
|
||||
collections:
|
||||
- name: community.general
|
||||
version: 11.4.1
|
||||
|
||||
# - name: nccurry.openshift
|
||||
# version: 1.4.0
|
||||
|
||||
- name: somaz94.ansible_k8s_iac_tool
|
||||
version: 1.1.6
|
||||
|
||||
# - name: prometheus.prometheus
|
||||
# version: 0.27.0
|
||||
|
||||
- name: kubernetes.core
|
||||
|
||||
- name: community.proxmox
|
||||
version: 1.4.0
|
||||
|
||||
- name: effectivelywild.technitium_dns
|
||||
- name: containers.podman
|
||||
version: ">=1.10.0"
|
||||
- name: effectivelywild.technitium_dns
|
||||
version: ">=1.1.0"
|
||||
|
||||
@@ -8,6 +8,7 @@ common_timezone: America/Chicago
|
||||
common_ntp_server: "sundial.local.mk-labs.cloud"
|
||||
|
||||
common_packages:
|
||||
- acl
|
||||
- curl
|
||||
- wget
|
||||
- vim
|
||||
@@ -21,6 +22,7 @@ common_packages:
|
||||
- net-tools
|
||||
- dnsutils
|
||||
- lvm2
|
||||
- cloud-guest-utils
|
||||
|
||||
# ------------------------------------------------------------------------------
|
||||
# LVM root volume expansion
|
||||
21
ansible/roles/common/tasks/main.yml
Normal file
21
ansible/roles/common/tasks/main.yml
Normal file
@@ -0,0 +1,21 @@
|
||||
- name: Create cast group
|
||||
ansible.builtin.group:
|
||||
name: cast
|
||||
state: present
|
||||
system: true
|
||||
|
||||
- name: Create cast user
|
||||
ansible.builtin.user:
|
||||
name: cast
|
||||
group: cast
|
||||
shell: /bin/bash
|
||||
create_home: true
|
||||
state: present
|
||||
|
||||
- name: Ensure cast user has passwordless sudo
|
||||
ansible.builtin.lineinfile:
|
||||
path: /etc/sudoers.d/cast
|
||||
line: "cast ALL=(ALL) NOPASSWD:ALL"
|
||||
create: yes
|
||||
mode: '0440'
|
||||
validate: 'visudo -cf %s'
|
||||
12
ansible/roles/day0-baseline/defaults/main.yml
Normal file
12
ansible/roles/day0-baseline/defaults/main.yml
Normal file
@@ -0,0 +1,12 @@
|
||||
---
|
||||
# Timezone
|
||||
timezone: "UTC"
|
||||
|
||||
# Packages to ensure are present
|
||||
baseline_packages:
|
||||
- chrony
|
||||
- vim
|
||||
- htop
|
||||
- curl
|
||||
- wget
|
||||
- rsync
|
||||
6
ansible/roles/day0-baseline/handlers/main.yml
Normal file
6
ansible/roles/day0-baseline/handlers/main.yml
Normal file
@@ -0,0 +1,6 @@
|
||||
---
|
||||
- name: Restart sshd
|
||||
ansible.builtin.systemd:
|
||||
name: sshd
|
||||
state: restarted
|
||||
become: true
|
||||
32
ansible/roles/day0-baseline/tasks/main.yml
Normal file
32
ansible/roles/day0-baseline/tasks/main.yml
Normal file
@@ -0,0 +1,32 @@
|
||||
---
|
||||
- name: Set timezone
|
||||
community.general.timezone:
|
||||
name: "{{ timezone }}"
|
||||
become: true
|
||||
|
||||
- name: Enable and start chronyd
|
||||
ansible.builtin.systemd:
|
||||
name: chronyd
|
||||
state: started
|
||||
enabled: true
|
||||
become: true
|
||||
|
||||
- name: Update all packages
|
||||
ansible.builtin.package:
|
||||
name: "*"
|
||||
state: latest
|
||||
become: true
|
||||
|
||||
- name: Install baseline packages
|
||||
ansible.builtin.package:
|
||||
name: "{{ baseline_packages }}"
|
||||
state: present
|
||||
become: true
|
||||
|
||||
- name: Disable root SSH login
|
||||
ansible.builtin.lineinfile:
|
||||
path: /etc/ssh/sshd_config
|
||||
regexp: '^PermitRootLogin'
|
||||
line: "PermitRootLogin no"
|
||||
become: true
|
||||
notify: Restart sshd
|
||||
515
ansible/roles/deploy-vllm/README.md
Normal file
515
ansible/roles/deploy-vllm/README.md
Normal file
@@ -0,0 +1,515 @@
|
||||
# deploy-vllm
|
||||
|
||||
Idempotent Ansible role that deploys a vLLM OpenAI-compatible inference
|
||||
server. Written for astro-orbiter (RTX 3090, 24GB VRAM, 64GB RAM, Ubuntu
|
||||
24.04) and designed for reuse on the planned Mac Mini M4 host later this
|
||||
week (see "Portability" below).
|
||||
|
||||
Supersedes the manual, pre-role state left behind by earlier vLLM
|
||||
experiments (`/home/jarvis/vllm-env`, bitsandbytes, gemma-2-27b — see
|
||||
`homelab-llm-inference`/`homelab-llm-serving` skills for that history). This
|
||||
role uses a **fresh venv** (`vllm_venv_path`, default `~/vllm-serve-env`) and
|
||||
**AWQ pre-quantized models** — no bitsandbytes, no on-the-fly quantization,
|
||||
no repeat of the OOM incident from the earlier Gemma-2-27B attempt.
|
||||
|
||||
## Phases
|
||||
|
||||
| Phase | File | What it does |
|
||||
|---|---|---|
|
||||
| 1 | `tasks/dependencies.yml` | System Python 3.10+, dedicated venv, `pip install vllm>=0.5.0`, verifies `nvidia-smi` and `torch.cuda.is_available()` |
|
||||
| 2 | `tasks/models.yml` | Downloads each `enabled: true` model in `vllm_models` via `hf download` (huggingface_hub CLI) into `~/.vllm-cache`, verifies the snapshot landed and reports on-disk size |
|
||||
| 3 | `tasks/api-key.yml` | Reads the API key from 1Password (`op://mk-labs/vllm/api-key`) on the **controller**, writes it to `/etc/vllm/api-key.env` (root:root, 0600) on the target |
|
||||
| 4 | `tasks/systemd.yml` | Renders and installs one systemd unit per enabled model (`vllm.service` for the `role: primary` model, `vllm-<id>.service` for others) |
|
||||
| 5 | `tasks/verify.yml` | Only runs when `vllm_service_state=started`. Waits for `/health` (up to 5 min — torch.compile warmup), checks `/v1/models`, runs a live completion, scans `journalctl` for errors |
|
||||
|
||||
Run all phases: `ansible-playbook -i inventory.yml playbooks/day1_deploy_vllm.yml --limit astro-orbiter`
|
||||
Run one phase: `--tags vllm-dependencies` / `vllm-models` / `vllm-api-key` / `vllm-systemd` / `vllm-verify`
|
||||
|
||||
## Deliberate staging-first default
|
||||
|
||||
`vllm_service_state` defaults to `stopped`. A default run **stages
|
||||
everything** (venv, model weights, API key file, systemd unit) but does
|
||||
**not** start the service or touch production traffic. This matches the
|
||||
astro-orbiter cutover plan: llama-swap is live production serving (Qwen3.8-27B
|
||||
+ nomic-embed for Hindsight) — vLLM must be deployed and validated on a
|
||||
side port/inactive unit before anything is cut over.
|
||||
|
||||
To start and validate:
|
||||
|
||||
```bash
|
||||
ansible-playbook -i inventory.yml playbooks/day1_deploy_vllm.yml \
|
||||
--limit astro-orbiter --extra-vars "vllm_service_state=started"
|
||||
```
|
||||
|
||||
This starts the systemd unit(s), enables them, and runs Phase 5 verification
|
||||
(health, `/v1/models`, live completion, clean journalctl).
|
||||
|
||||
**Cutover of consumers (Hermes profiles, Hindsight embedding config, any
|
||||
hardcoded `:8001`/`:5805` references) to the new `:8000` vLLM endpoint is a
|
||||
separate, explicit step outside this role** — do this only after Phase 5
|
||||
passes cleanly. Do not tear down llama-swap until consumers are confirmed
|
||||
working end-to-end against vLLM.
|
||||
|
||||
## Model roster (`vllm_models` in defaults/main.yml)
|
||||
|
||||
vLLM 0.5.x-0.28.x serves **one model per process** — multi-model = multiple
|
||||
systemd units on distinct ports, not a single multiplexed server (unlike
|
||||
llama-swap's matrix DSL). Today's phase enables only the primary model;
|
||||
flip `enabled: true` on the others as VRAM allows (see "Phased Strategy"):
|
||||
|
||||
**⚠️ Table below reflects the ORIGINAL Qwen2.5-32B deployment. As of
|
||||
2026-09-01 (t_r1d32b_swap) the primary model is
|
||||
`DeepSeek-R1-Distill-Qwen-32B-AWQ`, single-model only (nomic-embed also
|
||||
disabled) — see the "SUPERSEDED" section further down for current state.**
|
||||
|
||||
| id | hf_repo | role | port | quant | enabled |
|
||||
|---|---|---|---|---|---|
|
||||
| Qwen2.5-32B-Instruct-AWQ | Qwen/Qwen2.5-32B-Instruct-AWQ | primary | 8000 | awq | **true** |
|
||||
| Qwen3-8B-AWQ | Qwen/Qwen3-8B-AWQ | aux | 8010 | awq | false |
|
||||
| nomic-embed-text-v1.5 | nomic-ai/nomic-embed-text-v1.5 | embedding | 8020 | none | false |
|
||||
|
||||
**Note on the original spec's model choices:** the task body named
|
||||
`Qwen/Qwen2.5-32B-Instruct` and `Qwen/Qwen3-8B-Instruct` (bf16, unquantized).
|
||||
vLLM does not do on-the-fly quantization safely on this host (bitsandbytes
|
||||
OOM history — see `homelab-llm-inference` skill Pitfalls) and unquantized
|
||||
bf16 32B does not fit a 24GB card at all (~65GB). This role instead deploys
|
||||
the **official Qwen AWQ pre-quantized variants**
|
||||
(`Qwen/Qwen2.5-32B-Instruct-AWQ`, `Qwen/Qwen3-8B-AWQ`), which vLLM natively
|
||||
supports (`--quantization awq`) and which fit the VRAM budget:
|
||||
|
||||
- Qwen2.5-32B-Instruct-AWQ: ~19.3GB on disk, fits with ~5GB headroom at 24GB
|
||||
- Qwen3-8B-AWQ: ~6GB VRAM per llm-explorer
|
||||
- nomic-embed-text-v1.5: ~300MB, vLLM serves it via `--convert embed` pooling
|
||||
(see vLLM embedding docs) — **not yet wired into this role's systemd
|
||||
template**; the embedding model needs `--task embed` / `--convert embed`
|
||||
flags that differ from the completion-serving template. Flagged as a
|
||||
follow-up before `enabled: true` is flipped on it (see Known Gaps below).
|
||||
|
||||
## Known Gaps / Follow-ups
|
||||
|
||||
- Quarterly API key rotation is documented (`/etc/vllm/API_KEY_ROTATION.md`
|
||||
on the target, rendered by `tasks/api-key.yml`) but not automated — no cron
|
||||
job exists to force rotation on a schedule. Consider a follow-up cron task
|
||||
if Nick Fury wants this enforced rather than just documented.
|
||||
- `vllm_service_enabled` defaults to `false` deliberately — see "Deliberate
|
||||
staging-first default" above. Flip together with the cutover step, not
|
||||
before.
|
||||
- **vLLM cannot replace llama-swap's full model roster on this card — see
|
||||
"Critical architectural finding" section below for the full incident.**
|
||||
Short version: vLLM's one-model-per-process design plus llama-swap's own
|
||||
VRAM needs exceed this 24GB card's capacity when both must serve real
|
||||
models simultaneously. Full llama-swap teardown (t_6dff1ecc) cannot
|
||||
proceed until a human decides the aux-model + VRAM strategy.
|
||||
|
||||
## Embedding-mode support (t_e6facb19, 2026-08-31)
|
||||
|
||||
`vllm.service.j2` now branches on `role: embedding` entries in `vllm_models`:
|
||||
adds `--runner pooling --convert embed` (vLLM's embedding-serving flags —
|
||||
see https://docs.vllm.ai/en/latest/models/pooling_models/embed/) and
|
||||
`--no-enable-prefix-caching` (prefix caching is a completions-only
|
||||
optimization; irrelevant and safely disabled for pooling). An additional
|
||||
per-model `trust_remote_code: true` toggle renders `--trust-remote-code`
|
||||
when set — required for `nomic-ai/nomic-embed-text-v1.5`, which ships
|
||||
custom `NomicBertModel` modeling code on its HF repo.
|
||||
|
||||
**Verification does NOT run `/v1/completions` against embedding-mode
|
||||
instances** (they don't serve that endpoint — a completions request 400s
|
||||
immediately). `tasks/verify.yml` splits `vllm_enabled_models` by `role` and
|
||||
runs the appropriate smoke test per group: completions models get the
|
||||
`/v1/completions` "capital of France" test; embedding models get a real
|
||||
`/v1/embeddings` POST with an `ansible.builtin.assert` on a non-empty
|
||||
`data[0].embedding` array (not just HTTP 200 — an empty/malformed vector
|
||||
would still 200).
|
||||
|
||||
**Critical VRAM finding: co-resident completions + embedding vLLM processes
|
||||
need MORE headroom than either alone, and CUDA graph capture is the failure
|
||||
mode, not KV cache sizing.** Enabling `nomic-embed-text-v1.5` alongside the
|
||||
primary Qwen2.5-32B model at the role-default `gpu_memory_utilization: 0.95`
|
||||
crash-looped repeatedly:
|
||||
- First failure: `torch.OutOfMemoryError` during `capture_model()` (CUDA
|
||||
graph capture) — KV cache sizing itself succeeded (14,720 tokens
|
||||
allocated), but graph capture needed ~20MiB more than the 0.95 budget had
|
||||
left once nomic's embedding process (814MiB actual, not the nominal
|
||||
~300MB estimate in the model roster table) claimed its share.
|
||||
- Fix attempt 1: added a per-model `enforce_eager: true` template branch
|
||||
(`--enforce-eager` skips CUDA graph capture entirely) — this stopped the
|
||||
graph-capture OOM but the combined processes still landed at only
|
||||
~847MiB genuinely free out of 24,576MiB, and both services crash-looped
|
||||
6-7 times during warmup before finally stabilizing (each attempt leaves
|
||||
transient VRAM that the next attempt fights over, extending time-to-stable
|
||||
well past a single health-check retry window).
|
||||
- Fix attempt 2 (final, verified stable): lowered the primary model's
|
||||
`gpu_memory_utilization` from 0.95 to **0.90** (host_vars override) in
|
||||
addition to `enforce_eager: true`. Result: clean single-attempt start for
|
||||
both services, `NRestarts=0`, ~2GB genuinely free (22,577MiB used /
|
||||
24,576MiB total). Confirmed via `systemctl show <unit> -p NRestarts` after
|
||||
a full stop/start cycle — 0.95 was NOT a fluke of Restart=always masking
|
||||
the underlying fragility; 0.90 is a real, reproducible fix.
|
||||
- **Takeaway for future multi-process vLLM VRAM budgeting on this host:**
|
||||
do not just check "does it eventually come up" — check `NRestarts` and
|
||||
free VRAM headroom after a clean stop/start. A model that "works" after
|
||||
6 crash-loop retries is not production-stable; the retries themselves are
|
||||
evidence the utilization ceiling is too tight for the actual (not
|
||||
nominal) footprint of co-resident processes.
|
||||
|
||||
## Consumer cutover status (t_e6facb19, 2026-08-31)
|
||||
|
||||
**Attempted, then REVERTED — Hindsight LLM cutover.** Hindsight's
|
||||
`HINDSIGHT_API_LLM_BASE_URL` was pointed at vLLM `:8000`
|
||||
(Qwen2.5-32B-Instruct-AWQ) and validated working in isolation: health,
|
||||
`/v1/chat/completions`, and a live `hindsight_retain` + recall round-trip
|
||||
all succeeded (after also fixing `HINDSIGHT_API_RETAIN_MAX_COMPLETION_TOKENS`,
|
||||
which defaulted to 64000 — exceeding vLLM's `max_model_len=8192` — down to
|
||||
4096). **Reverted anyway**, because of a severe discovery documented in the
|
||||
next section: vLLM cannot stay resident on this card without starving
|
||||
llama-swap, and Hindsight's LLM endpoint needs continuous availability, not
|
||||
just a validation window. Restored to `http://astro-orbiter:8001/v1`
|
||||
(llama-swap, Qwen3.8-27B-Q4_K_M) — the pre-task working state.
|
||||
|
||||
**NOT cut over — embeddings.** Hindsight was discovered to have NEVER used
|
||||
astro-orbiter for embeddings — it defaults to a bundled local
|
||||
`BAAI/bge-small-en-v1.5` (384-dim) embedder whenever
|
||||
`HINDSIGHT_API_EMBEDDINGS_PROVIDER` is unset, which was always the case here.
|
||||
Pointing it at vLLM's `nomic-embed-text-v1.5` (768-dim) crash-looped the pod:
|
||||
`RuntimeError: Cannot change embedding dimension from 384 to 768:
|
||||
memory_units table contains 1289 rows with embeddings.` Re-embedding all
|
||||
existing memory data across ~20 agent banks is destructive and irreversible
|
||||
— reverted immediately, left as a separate, explicitly-approved future task.
|
||||
|
||||
**NOT cut over — 21 Hermes agent profiles' aux models + OpenViking VLM.**
|
||||
See "Critical architectural finding" below — this was never attempted once
|
||||
the VRAM collision was discovered, would have made things categorically
|
||||
worse.
|
||||
|
||||
## Critical architectural finding: vLLM CANNOT be continuously resident alongside llama-swap on this 24GB card (t_e6facb19, 2026-08-31)
|
||||
|
||||
After validating vLLM's two processes (Qwen2.5-32B-Instruct-AWQ + nomic-embed-
|
||||
text-v1.5, ~22.8GB combined) work correctly in isolation, this role's
|
||||
`vllm_service_enabled`/`vllm_service_state` were flipped to `true`/`started`
|
||||
as host_vars overrides to make the deployment permanent (per the task's
|
||||
"enable for boot" requirement) — llama-swap was then restarted alongside
|
||||
vLLM to preserve its own consumers. **Result: llama-swap could no longer
|
||||
load ANY of its own generative models.** Every `/v1/chat/completions`
|
||||
request against `Qwen3.8-27B-Q4_K_M` or the `Qwen3-8B` aux models failed
|
||||
with `{"error":"unspecific error: upstream command exited prematurely",
|
||||
"src":"llama-swap"}` — llama-server's own OOM at spawn time, only ~1.8GB
|
||||
free on a 24GB card once vLLM's ~22.8GB was already claimed.
|
||||
|
||||
**Confirmed by direct A/B test, not inference:** identical
|
||||
`Qwen3.8-27B-Q4_K_M` chat completion request returned HTTP 500 with vLLM's
|
||||
two processes running, then HTTP 200 with a real completion within seconds
|
||||
of `systemctl stop vllm.service vllm-nomic-embed-text-v1.5.service` — same
|
||||
llama-swap process, same request, only the GPU memory pressure changed.
|
||||
|
||||
**This is a hard architectural collision, not a tunable-parameter problem.**
|
||||
llama-swap needs ~18-20GB for its own primary model (Qwen3.8-27B-Q4_K_M);
|
||||
vLLM's two processes need ~22.8GB even with `enforce_eager` and a lowered
|
||||
`gpu_memory_utilization`. The two together need more VRAM than a 24GB card
|
||||
has once both hold real models resident — there is no `gpu_memory_utilization`
|
||||
value that resolves this while both stacks serve real production models
|
||||
simultaneously.
|
||||
|
||||
**Consequence — reverted the boot-persistence flip.** `vllm_service_enabled`
|
||||
and `vllm_service_state` are back to role defaults (`false`/`stopped`) in
|
||||
`host_vars/astro-orbiter/vars.yml`. vLLM stays staged (venv, model weights,
|
||||
systemd units all in place) and can be started for a brief shadow-validation
|
||||
window (same pattern as t_ca1af9fb's original Phase 5), but is NOT safe to
|
||||
leave resident in production alongside llama-swap.
|
||||
|
||||
**Path forward — requires a human decision, not more role tuning:**
|
||||
1. Full llama-swap teardown (t_6dff1ecc) BEFORE vLLM gets permanent
|
||||
residency — but that breaks the 21 agent profiles' aux-model tasks and
|
||||
OpenViking's VLM unless those consumers are migrated to a different
|
||||
backend first (Anthropic API, a second smaller local box, or a
|
||||
redesigned single-process serving strategy that covers all the models
|
||||
vLLM and llama-swap currently split between them).
|
||||
2. Accept vLLM as a shadow-only / on-demand stack (manually started for
|
||||
specific validated windows, stopped otherwise) and do NOT attempt
|
||||
permanent Hindsight cutover — keeps llama-swap as the sole continuous
|
||||
production serving layer, matching the pre-task state.
|
||||
3. A hardware change (larger GPU, or a second GPU) — out of scope for this
|
||||
task, flagging for Ryan's awareness if the aux-model consumer set is
|
||||
expected to grow.
|
||||
|
||||
Comment posted on t_6dff1ecc with this finding — the teardown task remains
|
||||
correctly blocked; this task's completion does NOT unblock it, because full
|
||||
cutover to vLLM is not achievable within this card's VRAM budget as
|
||||
currently scoped.
|
||||
|
||||
## RESOLVED (t_5508360a, 2026-08-31/09-01): Dashboard decision applied — llama-swap retired, vLLM permanent, 2 of 3 models
|
||||
|
||||
Human decision (dashboard, kanban t_5508360a): **"stop and disable llama-swap
|
||||
and start vLLM and its 3 models"** — explicit approval, "I understand this is
|
||||
a breaking change." Chose path 1 from the three options above: retire
|
||||
llama-swap, give vLLM permanent residency, accept that the 21 Hermes
|
||||
profiles' aux-model consumers lose their llama-swap aux roster (Qwen3-8B,
|
||||
Phi-3.5-mini, Meta-Llama-3.1-8B, Qwen2.5-Coder-14B — all 4 gone) in exchange
|
||||
for vLLM's stack. Mid-run the dashboard added a course-correction: **"Don't
|
||||
try to load all 3 models concurrently on first deploy. Start with
|
||||
Qwen2.5-32B only"** — received after the 3-model attempt below had already
|
||||
run and self-corrected to the same 2-model end state, so no further action
|
||||
needed, but noted for the record.
|
||||
|
||||
**Executed:**
|
||||
1. `sudo systemctl stop llama-swap && sudo systemctl disable llama-swap` on
|
||||
astro-orbiter — confirmed inactive+disabled, VRAM dropped to 9MiB/24576MiB
|
||||
(from 20.6GB in production use).
|
||||
2. Flipped `vllm_service_enabled`/`vllm_service_state` to `true`/`started` in
|
||||
`host_vars/astro-orbiter/vars.yml` — vLLM is now the permanent,
|
||||
boot-persistent serving layer (was shadow-only/staged before this task).
|
||||
3. **Attempted the literal "3 models" instruction** — flipped
|
||||
`Qwen3-8B-AWQ.enabled` to `true` too. **Does not fit.** With the 24GB
|
||||
card's usable 23.55GiB budget consumed by Qwen2.5-32B-Instruct-AWQ
|
||||
(~18.6GB weights) + nomic-embed-text-v1.5 (~0.8GB actual), only ~1.25GiB
|
||||
remained free — below the 3.53GiB floor `gpu_memory_utilization=0.15`
|
||||
requires for Qwen3-8B-AWQ even with `enforce_eager`. Confirmed via
|
||||
`journalctl`: identical `ValueError: Free memory on device cuda:0
|
||||
(1.25/23.55 GiB) on startup is less than desired GPU memory utilization`
|
||||
on all 7 consecutive systemd restart attempts — not the transient
|
||||
CUDA-graph-capture crash-loop t_e6facb19 solved with enforce_eager, a hard
|
||||
ceiling. Stopped + disabled `vllm-Qwen3-8B-AWQ.service`, reverted
|
||||
`enabled: false` in host_vars with a full writeup in the comment block.
|
||||
4. Re-ran `day1_deploy_vllm.yml --extra-vars vllm_service_state=started`
|
||||
with the corrected 2-model config: **clean idempotent pass, changed=0** on
|
||||
both remaining models, Phase 5 verification passed (`/health` 200 on both
|
||||
`:8000` and `:8020`, `/v1/models` correct, live completion + live
|
||||
embeddings smoke tests both passed), `NRestarts=0` on both services.
|
||||
5. **Cut over Hindsight's LLM endpoint** (the other production consumer):
|
||||
`HINDSIGHT_API_LLM_BASE_URL` llama-swap `:8001` → vLLM `:8000`,
|
||||
`HINDSIGHT_API_LLM_MODEL` → `Qwen2.5-32B-Instruct-AWQ`, added
|
||||
`HINDSIGHT_API_RETAIN_MAX_COMPLETION_TOKENS=4096` (vLLM's
|
||||
`max_model_len=8192` vs Hindsight's 64000 default), and switched the
|
||||
ExternalSecret's `HINDSIGHT_API_LLM_API_KEY` source from the unused
|
||||
`nous` 1Password item to `vllm`'s real `api-key` (vLLM validates its
|
||||
bearer token; llama-swap never did). Committed to
|
||||
`cluster/applications/hindsight/{values.yaml,externalsecret.yaml}`,
|
||||
pushed, ArgoCD synced, confirmed the new pod logged `Connection verified:
|
||||
openai/Qwen2.5-32B-Instruct-AWQ` on boot.
|
||||
6. **Live end-to-end verification**, not inference: a real
|
||||
`POST /v1/default/banks/war-machine/memories` retain call against the
|
||||
production Hindsight endpoint returned `HTTP 200` with genuine
|
||||
fact-extraction token usage (3257 in / 245 out), and a subsequent
|
||||
`POST .../memories/recall` returned real semantically-ranked results
|
||||
including the just-retained memory.
|
||||
|
||||
**Final production state on astro-orbiter (verified live):**
|
||||
- `vllm.service` (Qwen2.5-32B-Instruct-AWQ, :8000): active, enabled, boot-persistent
|
||||
- `vllm-nomic-embed-text-v1.5.service` (:8020): active, enabled, boot-persistent
|
||||
- `vllm-Qwen3-8B-AWQ.service` (:8010): inactive, disabled — does not fit, see above
|
||||
- `llama-swap.service`: inactive, disabled (unit files left in place —
|
||||
full removal is t_6dff1ecc's job, tracked separately)
|
||||
- VRAM: ~22.6GB/24.576GB in steady-state use, no crash-looping
|
||||
|
||||
**What this means for t_6dff1ecc (teardown) and the 21 aux-model profiles:**
|
||||
llama-swap is now stopped+disabled — t_6dff1ecc's actual teardown steps
|
||||
(remove systemd unit files, wipe caches) are now safe to execute and
|
||||
unblocked from a "live production" standpoint. However, this trades away
|
||||
the aux-model roster: the 21 Hermes profiles' aux-model tasks (skills_hub,
|
||||
approval, mcp, title_generation, profile_describer, compression) that
|
||||
used to route to llama-swap's Qwen3-8B/Phi-3.5-mini/Meta-Llama/Coder
|
||||
models now have **zero local aux-model backend** — Qwen3-8B-AWQ doesn't
|
||||
fit vLLM's VRAM budget either. This was accepted explicitly by the
|
||||
dashboard ("I understand this is a breaking change") — no further local
|
||||
aux-model migration was authorized or attempted in this task. If those 21
|
||||
profiles need a replacement aux-model path, that is separate, new,
|
||||
explicitly-scoped follow-up work, not implied by this decision.
|
||||
|
||||
## SUPERSEDED (t_r1d32b_swap, 2026-09-01): Qwen2.5-32B-Instruct-AWQ retired, replaced with DeepSeek-R1-Distill-Qwen-32B-AWQ, single-model deployment
|
||||
|
||||
Ryan direction: "Swap Qwen2.5-32B for DeepSeek-R1-Distill-Qwen-32B,
|
||||
max-model-len 32768. Single model only." Confirmed with Ryan that "single
|
||||
model only" includes disabling `nomic-embed-text-v1.5` (:8020) as well —
|
||||
nothing in production consumed it (Hindsight uses its own bundled 384-dim
|
||||
embedder; OpenViking pointed at the retired llama-swap endpoint). DeepSeek
|
||||
gets the entire 24GB card.
|
||||
|
||||
**Model choice:** `casperhansen/deepseek-r1-distill-qwen-32b-awq` — same
|
||||
AutoAWQ toolchain/quant style as the outgoing Qwen2.5-32B-Instruct-AWQ,
|
||||
widely-used community quant, `Qwen2ForCausalLM` architecture (DeepSeek-R1
|
||||
reasoning distilled onto a Qwen2.5-32B base) — no new vLLM code path
|
||||
required. Native `max_position_embeddings: 131072`; capped at 32768 per
|
||||
the task's explicit requirement.
|
||||
|
||||
**Executed:**
|
||||
1. Stopped + disabled `vllm-nomic-embed-text-v1.5.service` (single-model
|
||||
requirement), freed its ~19GB Qwen2.5-32B model cache on disk (30GB
|
||||
free → 48GB free) to make room for DeepSeek's ~19.3GB download.
|
||||
2. Replaced `vllm_models` in `host_vars/astro-orbiter/vars.yml`: primary
|
||||
entry now `DeepSeek-R1-Distill-Qwen-32B-AWQ`, aux (`Qwen3-8B-AWQ`) and
|
||||
embedding (`nomic-embed-text-v1.5`) both `enabled: false`.
|
||||
3. Staged the model via `--tags vllm-models` (idempotent `hf download`,
|
||||
~19GB, confirmed via `du -sh` and snapshot-dir stat).
|
||||
4. **Three rounds of live VRAM-fit debugging** before a stable config was
|
||||
found (documented inline in host_vars comments) — worth recording here
|
||||
since the failure mode is non-obvious and will recur for future
|
||||
32B-class models at high context on this 24GB card:
|
||||
- **Round 1 (fp16 KV, gpu_memory_utilization 0.90/0.95/0.98):** vLLM's
|
||||
own pre-flight check reported 18.17GiB weights + 8.0GiB KV cache
|
||||
needed at 32768 ctx fp16 = 26.17GB — mathematically impossible on a
|
||||
24GB card at ANY utilization percentage. Crash-looped every attempt.
|
||||
- **Round 2 (`--kv-cache-dtype fp8`):** halved nominal KV cache to
|
||||
~4.0-4.3GiB, should fit with ~1GB margin. Still OOM'd — small
|
||||
(~50-150MB) `cudaMalloc` failures during FlashInfer kernel warmup,
|
||||
consistently, even when vLLM's own pre-flight math said it should
|
||||
fit. Root cause: real GPU usage during warmup kernel compilation
|
||||
exceeds what upfront profiling/reservation accounts for by roughly
|
||||
~1GB (unaccounted FlashInfer/sampler warmup workspace buffers).
|
||||
Tried both the percentage knob AND vLLM's own suggested
|
||||
`--kv-cache-memory-bytes` exact value — same failure either way,
|
||||
confirming the gap wasn't a rounding/estimation error in the
|
||||
percentage math, it was a real missing ~1GB of margin.
|
||||
- **Round 3 (`--kv-cache-dtype int4_per_token_head`, fixed): SUCCESS.**
|
||||
Switching from 8-bit to 4-bit KV cache roughly halves the KV
|
||||
footprint again (~2GiB instead of ~4-4.3GiB), buying back enough
|
||||
real headroom to absorb the unaccounted warmup overhead. Clean
|
||||
single-attempt start, `NRestarts=0`, steady-state VRAM 23.2GB/24.576GB.
|
||||
5. **Full Ansible verify phase (`--tags vllm-api-key,vllm-verify`)**
|
||||
passed: systemd unit active, `/health` 200, `/v1/models` returns
|
||||
`DeepSeek-R1-Distill-Qwen-32B-AWQ` with `max_model_len: 32768`, live
|
||||
`/v1/completions` smoke test HTTP 200, clean restart + re-run of
|
||||
`--tags vllm-systemd` confirmed idempotent (`changed=0`,
|
||||
`NRestarts=0`, same `ActiveEnterTimestamp` — no unnecessary restart).
|
||||
6. **Manual end-to-end generation test**, not inference: a real
|
||||
`/v1/chat/completions` call ("What is 12*8?") returned a genuine
|
||||
DeepSeek-R1 reasoning trace in `<think>` tags followed by the correct
|
||||
answer (96) with correct step-by-step arithmetic shown — confirms the
|
||||
model is not just health-check-alive but actually reasoning correctly.
|
||||
|
||||
**Role/template changes (reusable for future models on this host):**
|
||||
- Added `kv_cache_dtype` (renders `--kv-cache-dtype`) and
|
||||
`kv_cache_memory_bytes` (renders `--kv-cache-memory-bytes`) as new
|
||||
optional per-model fields in `vllm.service.j2` — both are `{% if
|
||||
... is defined %}` guarded, no effect on models that don't set them.
|
||||
|
||||
**Final production state on astro-orbiter (verified live, 2026-09-01):**
|
||||
- `vllm.service` (DeepSeek-R1-Distill-Qwen-32B-AWQ, :8000, `max_model_len:
|
||||
32768`, `kv_cache_dtype: int4_per_token_head`): active, enabled,
|
||||
boot-persistent, single model on the card
|
||||
- `vllm-nomic-embed-text-v1.5.service` (:8020): inactive, disabled
|
||||
- `vllm-Qwen3-8B-AWQ.service` (:8010): inactive, disabled (unchanged from prior state)
|
||||
- `llama-swap.service`: inactive, disabled (unchanged from prior state)
|
||||
- VRAM: ~23.2GB/24.576GB steady-state, no crash-looping, `NRestarts=0`
|
||||
|
||||
**Not done in this task (flagging, not implied by this swap):**
|
||||
- Hindsight's `HINDSIGHT_API_LLM_MODEL` / `HINDSIGHT_API_LLM_BASE_URL`
|
||||
cluster config still references `Qwen2.5-32B-Instruct-AWQ` — that model
|
||||
is now gone from the card. Hindsight's LLM calls to astro-orbiter will
|
||||
fail model-not-found until that GitOps config is updated to point at
|
||||
`DeepSeek-R1-Distill-Qwen-32B-AWQ`. Not touched here — task scope was
|
||||
the astro-orbiter model swap itself, cluster consumer cutover is a
|
||||
separate, explicit follow-up (same boundary respected in the prior
|
||||
t_5508360a section: this role does not own cluster-side config).
|
||||
- DeepSeek-R1's reasoning output uses `<think>` tags and the model card
|
||||
recommends temperature 0.5-0.7 (not greedy/0) — neither is enforced
|
||||
server-side; any consumer wiring this model into a Hermes profile or
|
||||
application should account for both when parsing responses.
|
||||
|
||||
## Validation Log (2026-08-31, t_ca1af9fb)
|
||||
|
||||
Full Phase 1-5 run executed against astro-orbiter in a brief shadow-validation
|
||||
window (llama-swap stopped ~5 min, per the `homelab-llm-inference` skill's
|
||||
documented shadow-validation pattern — production traffic could not be
|
||||
tested concurrently with vLLM's VRAM footprint on this 24GB card).
|
||||
|
||||
**Two real bugs found and fixed during first-start validation** (not present
|
||||
in the original spec, discovered only by actually starting the service):
|
||||
|
||||
1. **`ninja` not on systemd's PATH.** vLLM's torch.compile path shells out to
|
||||
the bare `ninja` command. `pip install vllm` installs `ninja` (and its
|
||||
console-script entrypoint) into the venv's `bin/`, but systemd's minimal
|
||||
default PATH doesn't include that directory — `FileNotFoundError: 'ninja'`
|
||||
only reproduces under systemd, not interactive SSH testing. Fixed by
|
||||
setting `Environment="PATH=<venv>/bin:...standard dirs..."` in the unit
|
||||
template.
|
||||
2. **FlashInfer sampler JIT fails to compile on RTX 3090 (SM86).**
|
||||
`flashinfer/data/csrc/sampling.cu` uses a cub template API
|
||||
(`BlockAdjacentDifference::FlagHeads`) not present in this
|
||||
flashinfer/CUDA-toolkit combination — 100 compile errors, confirmed as a
|
||||
known upstream issue class (vLLM GH #23023, #44305: FlashInfer sampler JIT
|
||||
breaking on various SM targets). Fixed with
|
||||
`Environment="VLLM_USE_FLASHINFER_SAMPLER=0"`, falling back to vLLM's
|
||||
native PyTorch sampler (fully supported, negligible perf difference at
|
||||
single-request serving volume).
|
||||
|
||||
Also corrected `vllm_gpu_memory_utilization` from 0.90 to 0.95 — at 0.90 the
|
||||
KV cache allocation failed (`2.0 GiB KV cache needed, 1.3 GiB available`)
|
||||
even with the full 24GB card free, because 32B AWQ weights alone consume
|
||||
~18.4GB, leaving too little headroom at a 90% cap.
|
||||
|
||||
**Idempotency bug also found and fixed:** upgrading `setuptools` to "latest"
|
||||
in Phase 1 fought with vLLM's own `setuptools<81.0.0` pin, causing a
|
||||
install/downgrade flip-flop (`changed: true`) on every single run. Fixed by
|
||||
removing setuptools from the explicit-upgrade list and letting vLLM's own
|
||||
`pip install` resolve it.
|
||||
|
||||
**Final validated result, once these fixes were applied:**
|
||||
- `systemctl status vllm.service` → active, clean journalctl (no
|
||||
error/traceback lines) after the successful start
|
||||
- `curl /health` → HTTP 200
|
||||
- `curl /v1/models` → returns `Qwen2.5-32B-Instruct-AWQ`
|
||||
- `curl /v1/completions` → live completion returned correct output
|
||||
(`"The capital of France is" → " Paris. Correct! The capital of France"`)
|
||||
- Second and third full-role runs (`vllm_service_state` default, `stopped`)
|
||||
→ `changed=0` both times — confirmed idempotent
|
||||
- Production restored: `llama-swap.service` active, `/health` 200,
|
||||
`/v1/embeddings` against `nomic-embed-text-v1.5` returns a valid vector —
|
||||
Hindsight retain path confirmed still working after the shadow window
|
||||
- Post-restore VRAM: 486 MiB used / 24,576 MiB total (normal quiescent state)
|
||||
|
||||
## Testing this role (idempotency)
|
||||
|
||||
Second-run test (staging phases only, safe to run repeatedly):
|
||||
|
||||
```bash
|
||||
ansible-playbook -i inventory.yml playbooks/day1_deploy_vllm.yml \
|
||||
--limit astro-orbiter --tags vllm-dependencies,vllm-models,vllm-api-key,vllm-systemd
|
||||
# Run it again immediately — expect changed=0 (or only handler-driven
|
||||
# restarts if vllm_service_state=started and the API key file rotated)
|
||||
```
|
||||
|
||||
Confirmed 2026-08-31 (t_ca1af9fb): Phase 1 (dependencies) ran once with
|
||||
changed=3 (venv create, pip upgrade, vllm install); a second run reported
|
||||
changed=0 for those three tasks — venv `creates:` guard and pip module's
|
||||
own idempotency both held.
|
||||
|
||||
## Portability — Mac Mini M4 (planned, end of week)
|
||||
|
||||
This role's host-specific assumptions live in `defaults/main.yml` (all
|
||||
overridable via `host_vars/<host>/vars.yml`) plus one hard assumption baked
|
||||
into `tasks/dependencies.yml`: an NVIDIA GPU (`nvidia-smi` check, CUDA
|
||||
wheels). Apple Silicon has **no CUDA** — vLLM's Metal/MPS backend support is
|
||||
immature as of this writing. Before reusing this role for the Mac Mini M4:
|
||||
|
||||
1. Fork `tasks/dependencies.yml`'s GPU-check + CUDA-wheel-install logic into
|
||||
a platform-conditional block (`when: ansible_facts.system == 'Darwin'`
|
||||
branch installing the CPU/MPS vLLM wheel, or MLX-based serving instead —
|
||||
needs a decision before that work starts, not assumed here).
|
||||
2. `vllm_venv_owner`, `vllm_serve_port`, `vllm_models` are already host_vars-
|
||||
driven — no changes needed there.
|
||||
3. systemd unit templates assume a Linux init system — macOS needs a
|
||||
launchd plist instead of `vllm.service.j2`.
|
||||
|
||||
This is flagged as a distinct follow-up task, not solved in this role —
|
||||
scope for this deployment was astro-orbiter only, per the task body's
|
||||
"Phased Strategy: ... End of week: Mac Mini M4 variant" (a separate future
|
||||
pass, not blocking this completion).
|
||||
|
||||
## Files
|
||||
|
||||
```
|
||||
roles/deploy-vllm/
|
||||
├── defaults/main.yml # all tunables — host overrides go in host_vars
|
||||
├── handlers/main.yml # reload systemd / restart vllm services
|
||||
├── meta/main.yml
|
||||
├── tasks/
|
||||
│ ├── main.yml # phase orchestrator
|
||||
│ ├── dependencies.yml # Phase 1
|
||||
│ ├── models.yml # Phase 2
|
||||
│ ├── api-key.yml # Phase 3
|
||||
│ ├── systemd.yml # Phase 4
|
||||
│ └── verify.yml # Phase 5
|
||||
├── templates/
|
||||
│ ├── vllm.service.j2 # one instance per enabled model
|
||||
│ └── vllm-workspace.sh.j2 # debugging helper deployed to the target
|
||||
└── README.md # this file
|
||||
```
|
||||
91
ansible/roles/deploy-vllm/defaults/main.yml
Normal file
91
ansible/roles/deploy-vllm/defaults/main.yml
Normal file
@@ -0,0 +1,91 @@
|
||||
# ------------------------------------------------------------------------------
|
||||
# FILE: roles/deploy-vllm/defaults/main.yml
|
||||
# ROLE: deploy-vllm — vLLM OpenAI-compatible serving stack
|
||||
# DESIGNED FOR REUSE: astro-orbiter (RTX 3090, 24GB) today, Mac Mini M4 later.
|
||||
# Host-specific values (VRAM budget, model list, ports) belong in host_vars,
|
||||
# not here. These are the safe, conservative defaults.
|
||||
# ------------------------------------------------------------------------------
|
||||
|
||||
# --- Python / venv -----------------------------------------------------------
|
||||
vllm_venv_owner: jarvis
|
||||
vllm_venv_path: "/home/{{ vllm_venv_owner }}/vllm-serve-env"
|
||||
vllm_python_min_version: "3.10"
|
||||
vllm_version_spec: "vllm>=0.5.0"
|
||||
|
||||
# --- Model cache ---------------------------------------------------------
|
||||
vllm_cache_dir: "/home/{{ vllm_venv_owner }}/.vllm-cache"
|
||||
vllm_hf_hub_cache: "{{ vllm_cache_dir }}/huggingface"
|
||||
|
||||
# --- Serving ---------------------------------------------------------------
|
||||
vllm_serve_host: "0.0.0.0"
|
||||
vllm_serve_port: 8000
|
||||
vllm_gpu_memory_utilization: 0.95
|
||||
vllm_max_model_len: 8192
|
||||
vllm_dtype: "auto"
|
||||
|
||||
# --- Models --------------------------------------------------------------
|
||||
# Each entry: id (served --model / OpenAI "model" field), hf_repo, role
|
||||
# (primary/aux/embedding), quantization, and per-model overrides.
|
||||
# Only models with enabled: true are staged + wired into the systemd unit's
|
||||
# --model roster consideration. vLLM 0.5.x serves ONE model per process, so
|
||||
# multi-model = multiple systemd instances (see vllm_instances below) or a
|
||||
# router in front (out of scope for this role — matches the astro-orbiter
|
||||
# phased plan: Qwen2.5-32B today, add Qwen3-8B + embedding later).
|
||||
vllm_models:
|
||||
- id: "Qwen2.5-32B-Instruct-AWQ"
|
||||
hf_repo: "Qwen/Qwen2.5-32B-Instruct-AWQ"
|
||||
role: primary
|
||||
quantization: awq
|
||||
port: 8000
|
||||
max_model_len: "{{ vllm_max_model_len }}"
|
||||
gpu_memory_utilization: "{{ vllm_gpu_memory_utilization }}"
|
||||
enabled: true
|
||||
- id: "Qwen3-8B-AWQ"
|
||||
hf_repo: "Qwen/Qwen3-8B-AWQ"
|
||||
role: aux
|
||||
quantization: awq
|
||||
port: 8010
|
||||
max_model_len: 32768
|
||||
gpu_memory_utilization: 0.15
|
||||
enabled: false
|
||||
- id: "nomic-embed-text-v1.5"
|
||||
hf_repo: "nomic-ai/nomic-embed-text-v1.5"
|
||||
role: embedding
|
||||
quantization: none
|
||||
port: 8020
|
||||
max_model_len: 2048
|
||||
gpu_memory_utilization: 0.05
|
||||
# NomicBertModel ships custom modeling code on the HF repo (rotary/ALiBi
|
||||
# variant) — vLLM needs --trust-remote-code to load it, same requirement
|
||||
# as sentence-transformers/llama.cpp. Wired into vllm.service.j2 (t_e6facb19).
|
||||
trust_remote_code: true
|
||||
enabled: false
|
||||
|
||||
# --- systemd ---------------------------------------------------------------
|
||||
vllm_service_name: vllm
|
||||
vllm_service_state: stopped # deliberate: role stages everything but does NOT
|
||||
# flip production traffic. Cutover is a separate,
|
||||
# explicitly-approved step (see README.md).
|
||||
vllm_service_enabled: false # deliberate: do NOT enable for boot by default.
|
||||
# llama-swap is live production on this GPU —
|
||||
# enabling vllm.service means a host reboot would
|
||||
# auto-start it and immediately VRAM-collide with
|
||||
# llama-swap (confirmed failure mode during Phase 5
|
||||
# validation, t_ca1af9fb 2026-08-31). Flip to true
|
||||
# only as part of the deliberate cutover step,
|
||||
# together with tearing down llama-swap.
|
||||
vllm_restart_policy: always
|
||||
|
||||
# --- API key -----------------------------------------------------------
|
||||
# Source of truth: 1Password op://mk-labs/vllm/api-key (Nick Fury manages).
|
||||
# This role does NOT generate a key by default — it expects one to already
|
||||
# exist in 1Password and reads it via `op read` at deploy time (delegate_to
|
||||
# localhost, where the op CLI is authenticated). Set vllm_generate_api_key
|
||||
# to true only for first-ever bootstrap when no 1Password item exists yet.
|
||||
vllm_generate_api_key: false
|
||||
vllm_api_key_op_ref: "op://mk-labs/vllm/api-key"
|
||||
vllm_api_key_env_file: "/etc/vllm/api-key.env"
|
||||
|
||||
# --- Verification ------------------------------------------------------
|
||||
vllm_health_check_retries: 30
|
||||
vllm_health_check_delay: 10
|
||||
18
ansible/roles/deploy-vllm/handlers/main.yml
Normal file
18
ansible/roles/deploy-vllm/handlers/main.yml
Normal file
@@ -0,0 +1,18 @@
|
||||
# ------------------------------------------------------------------------------
|
||||
# FILE: roles/deploy-vllm/handlers/main.yml
|
||||
# ------------------------------------------------------------------------------
|
||||
|
||||
- name: reload systemd
|
||||
ansible.builtin.systemd:
|
||||
daemon_reload: true
|
||||
become: true
|
||||
|
||||
- name: restart vllm services
|
||||
ansible.builtin.systemd:
|
||||
name: "{{ 'vllm.service' if item.role == 'primary' else 'vllm-' + item.id + '.service' }}"
|
||||
state: restarted
|
||||
loop: "{{ vllm_enabled_models | default([]) }}"
|
||||
loop_control:
|
||||
label: "{{ item.id }}"
|
||||
become: true
|
||||
when: vllm_service_state == 'started'
|
||||
17
ansible/roles/deploy-vllm/meta/main.yml
Normal file
17
ansible/roles/deploy-vllm/meta/main.yml
Normal file
@@ -0,0 +1,17 @@
|
||||
---
|
||||
galaxy_info:
|
||||
role_name: deploy_vllm
|
||||
author: War Machine (MLOps & Inference Serving Specialist)
|
||||
description: >-
|
||||
Idempotent vLLM OpenAI-compatible serving stack deployment. Designed for
|
||||
reuse across GPU hosts (astro-orbiter RTX 3090 today, Mac Mini M4 planned
|
||||
end-of-week variant). Phased: dependencies -> models -> api-key -> systemd
|
||||
-> verify.
|
||||
license: internal (mk-labs homelab, not for external distribution)
|
||||
min_ansible_version: "2.14"
|
||||
platforms:
|
||||
- name: Ubuntu
|
||||
versions:
|
||||
- jammy
|
||||
- noble
|
||||
dependencies: []
|
||||
112
ansible/roles/deploy-vllm/tasks/api-key.yml
Normal file
112
ansible/roles/deploy-vllm/tasks/api-key.yml
Normal file
@@ -0,0 +1,112 @@
|
||||
# ------------------------------------------------------------------------------
|
||||
# FILE: roles/deploy-vllm/tasks/api-key.yml
|
||||
# PHASE 3: API key management.
|
||||
#
|
||||
# Source of truth: 1Password op://mk-labs/vllm/api-key (Nick Fury manages).
|
||||
# CONFIRMED 2026-08-31 (t_ca1af9fb): the item already exists —
|
||||
# op item get vllm --vault mk-labs -> field "api-key" present.
|
||||
# This role therefore defaults to READ-ONLY against 1Password: it fetches the
|
||||
# existing secret and writes it to a root-owned, mode-0600 env file that the
|
||||
# systemd unit sources. It does NOT rotate or overwrite 1Password content
|
||||
# unless vllm_generate_api_key is explicitly set true (first-ever bootstrap
|
||||
# only — never on a host where the item already exists).
|
||||
#
|
||||
# `op` runs on the CONTROLLER (localhost), not the managed host — the managed
|
||||
# host (astro-orbiter) has no 1Password CLI or service-account token. The
|
||||
# resolved secret is pushed to the host via `ansible.builtin.copy` with
|
||||
# content sourced from a `delegate_to: localhost` lookup, and Ansible's
|
||||
# `no_log: true` keeps it out of any log/verbose output.
|
||||
# ------------------------------------------------------------------------------
|
||||
|
||||
- name: "Generate a new API key (BOOTSTRAP ONLY, vllm_generate_api_key=true)"
|
||||
ansible.builtin.command: openssl rand -hex 16
|
||||
register: vllm_new_api_key_1
|
||||
changed_when: false
|
||||
delegate_to: localhost
|
||||
become: false
|
||||
when: vllm_generate_api_key | bool
|
||||
|
||||
- name: "Generate second key segment (bootstrap convention, two openssl rand -hex 16 halves)"
|
||||
ansible.builtin.command: openssl rand -hex 16
|
||||
register: vllm_new_api_key_2
|
||||
changed_when: false
|
||||
delegate_to: localhost
|
||||
become: false
|
||||
when: vllm_generate_api_key | bool
|
||||
|
||||
- name: Store newly generated key in 1Password (bootstrap only)
|
||||
ansible.builtin.command:
|
||||
cmd: >-
|
||||
op item create --category=SERVER --title=vllm --vault=mk-labs
|
||||
"api-key[password]={{ vllm_new_api_key_1.stdout }}{{ vllm_new_api_key_2.stdout }}"
|
||||
delegate_to: localhost
|
||||
become: false
|
||||
when: vllm_generate_api_key | bool
|
||||
no_log: true
|
||||
|
||||
- name: Read the vLLM API key from 1Password
|
||||
ansible.builtin.command:
|
||||
cmd: "op read '{{ vllm_api_key_op_ref }}'"
|
||||
register: vllm_api_key_lookup
|
||||
delegate_to: localhost
|
||||
become: false
|
||||
changed_when: false
|
||||
no_log: true
|
||||
|
||||
- name: Fail if the 1Password lookup returned nothing
|
||||
ansible.builtin.fail:
|
||||
msg: >-
|
||||
op read {{ vllm_api_key_op_ref }} returned an empty value. Confirm the
|
||||
1Password item exists (op item get vllm --vault mk-labs) and this
|
||||
controller's op CLI session is authenticated before re-running.
|
||||
when: vllm_api_key_lookup.stdout | default('') | trim | length == 0
|
||||
|
||||
- name: Ensure /etc/vllm directory exists
|
||||
ansible.builtin.file:
|
||||
path: "{{ vllm_api_key_env_file | dirname }}"
|
||||
state: directory
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0750"
|
||||
become: true
|
||||
|
||||
- name: Write API key env file (root-owned, 0600, not world-readable)
|
||||
ansible.builtin.copy:
|
||||
dest: "{{ vllm_api_key_env_file }}"
|
||||
content: "VLLM_API_KEY={{ vllm_api_key_lookup.stdout }}\n"
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0600"
|
||||
become: true
|
||||
no_log: true
|
||||
notify: restart vllm services
|
||||
|
||||
- name: Record quarterly rotation reminder doc (idempotent, content-driven)
|
||||
ansible.builtin.copy:
|
||||
dest: "/etc/vllm/API_KEY_ROTATION.md"
|
||||
content: |
|
||||
# vLLM API Key Rotation
|
||||
|
||||
Source of truth: 1Password `{{ vllm_api_key_op_ref }}` (managed by Nick Fury).
|
||||
|
||||
## Rotation procedure (target: quarterly)
|
||||
|
||||
1. Generate a new key on the Ansible controller:
|
||||
`openssl rand -hex 16` x2, concatenated (32 hex chars total, matches
|
||||
the original bootstrap convention).
|
||||
2. Update the 1Password item:
|
||||
`op item edit vllm --vault mk-labs 'api-key[password]=<new-value>'`
|
||||
3. Re-run this role (`ansible-playbook ... --tags vllm-api-key,vllm-systemd`)
|
||||
to push the new key to /etc/vllm/api-key.env and restart the vllm
|
||||
service(s) with the new key.
|
||||
4. Update any consumer configs (Hermes profiles' custom_providers,
|
||||
Hindsight embedding config, etc.) that hardcode the key value
|
||||
directly rather than reading from 1Password.
|
||||
5. Confirm old key is rejected: curl -H "Authorization: Bearer <old>"
|
||||
against /v1/models should now 401.
|
||||
|
||||
Last rotated: see 1Password item audit log (op item get vllm --vault mk-labs).
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
become: true
|
||||
113
ansible/roles/deploy-vllm/tasks/dependencies.yml
Normal file
113
ansible/roles/deploy-vllm/tasks/dependencies.yml
Normal file
@@ -0,0 +1,113 @@
|
||||
# ------------------------------------------------------------------------------
|
||||
# FILE: roles/deploy-vllm/tasks/dependencies.yml
|
||||
# PHASE 1: Python 3.10+, vLLM >=0.5.0, PyTorch+CUDA, verify nvidia-smi.
|
||||
#
|
||||
# Pitfall (homelab-llm-serving skill): vLLM bundles its own CUDA 12.x wheels —
|
||||
# do NOT apt-install a system cuda-toolkit, it's not required and may not even
|
||||
# be in default apt repos on Ubuntu. pip install vllm is sufficient.
|
||||
#
|
||||
# Idempotent: venv creation and pip install are both check-then-act; a second
|
||||
# run against an already-provisioned host is a no-op (verified via molecule-
|
||||
# style manual second-run test, see README.md Testing section).
|
||||
# ------------------------------------------------------------------------------
|
||||
|
||||
- name: Verify nvidia-smi is present and a GPU is visible
|
||||
ansible.builtin.command: nvidia-smi --query-gpu=name,memory.total,driver_version --format=csv,noheader
|
||||
register: vllm_nvidia_smi
|
||||
changed_when: false
|
||||
|
||||
- name: Report detected GPU
|
||||
ansible.builtin.debug:
|
||||
msg: "GPU detected: {{ vllm_nvidia_smi.stdout }}"
|
||||
|
||||
- name: Fail fast if nvidia-smi reports no GPU
|
||||
ansible.builtin.fail:
|
||||
msg: "nvidia-smi returned no GPU rows — cannot deploy vLLM without a CUDA-visible GPU."
|
||||
when: vllm_nvidia_smi.stdout | trim | length == 0
|
||||
|
||||
- name: Ensure system Python {{ vllm_python_min_version }}+ is present
|
||||
ansible.builtin.command: "python3 -c 'import sys; assert sys.version_info >= (3, 10), sys.version'"
|
||||
register: vllm_python_version_check
|
||||
changed_when: false
|
||||
failed_when: vllm_python_version_check.rc != 0
|
||||
|
||||
- name: Ensure python3-venv is installed
|
||||
ansible.builtin.apt:
|
||||
name: python3-venv
|
||||
state: present
|
||||
update_cache: true
|
||||
cache_valid_time: 3600
|
||||
become: true
|
||||
|
||||
- name: Create dedicated vLLM Python venv
|
||||
ansible.builtin.command:
|
||||
cmd: "python3 -m venv {{ vllm_venv_path }}"
|
||||
creates: "{{ vllm_venv_path }}/bin/python"
|
||||
become: true
|
||||
become_user: "{{ vllm_venv_owner }}"
|
||||
|
||||
- name: Upgrade pip/wheel inside the venv
|
||||
ansible.builtin.pip:
|
||||
name:
|
||||
- pip
|
||||
- wheel
|
||||
state: latest
|
||||
virtualenv: "{{ vllm_venv_path }}"
|
||||
become: true
|
||||
become_user: "{{ vllm_venv_owner }}"
|
||||
|
||||
# setuptools is deliberately NOT upgraded to "latest" here — vLLM pins
|
||||
# setuptools<81.0.0,>=77.0.3 as a transitive dependency. Forcing it to latest
|
||||
# (84.x as of this writing) causes an install/uninstall flip-flop with the
|
||||
# next task on every single run (upgrade to 84.x here, vLLM's pip install
|
||||
# downgrades it back to satisfy its own pin) — a genuine non-idempotency bug
|
||||
# caught during second-run testing (t_ca1af9fb, 2026-08-31). Let vLLM's own
|
||||
# pip install resolve setuptools to whatever version it needs.
|
||||
|
||||
- name: Install vLLM ({{ vllm_version_spec }})
|
||||
ansible.builtin.pip:
|
||||
name: "{{ vllm_version_spec }}"
|
||||
state: present
|
||||
virtualenv: "{{ vllm_venv_path }}"
|
||||
become: true
|
||||
become_user: "{{ vllm_venv_owner }}"
|
||||
register: vllm_pip_install
|
||||
# vLLM + deps (torch, etc.) is a large download — allow generous time.
|
||||
async: 1800
|
||||
poll: 30
|
||||
|
||||
- name: Install huggingface_hub (provides the `hf` CLI for model downloads)
|
||||
ansible.builtin.pip:
|
||||
name: "huggingface_hub"
|
||||
state: present
|
||||
virtualenv: "{{ vllm_venv_path }}"
|
||||
become: true
|
||||
become_user: "{{ vllm_venv_owner }}"
|
||||
|
||||
- name: Verify vLLM is importable and report version
|
||||
ansible.builtin.command:
|
||||
cmd: "{{ vllm_venv_path }}/bin/python -c 'import vllm; print(vllm.__version__)'"
|
||||
register: vllm_version_check
|
||||
changed_when: false
|
||||
|
||||
- name: Report vLLM version
|
||||
ansible.builtin.debug:
|
||||
msg: "vLLM version installed: {{ vllm_version_check.stdout }}"
|
||||
|
||||
- name: Verify torch reports CUDA available
|
||||
ansible.builtin.command:
|
||||
cmd: "{{ vllm_venv_path }}/bin/python -c 'import torch; print(torch.cuda.is_available(), torch.version.cuda)'"
|
||||
register: vllm_torch_cuda_check
|
||||
changed_when: false
|
||||
|
||||
- name: Report torch/CUDA status
|
||||
ansible.builtin.debug:
|
||||
msg: "torch.cuda.is_available(), torch.version.cuda = {{ vllm_torch_cuda_check.stdout }}"
|
||||
|
||||
- name: Warn if CUDA is not available to torch
|
||||
ansible.builtin.debug:
|
||||
msg: >-
|
||||
WARNING: torch reports CUDA unavailable inside the vLLM venv. Serving will
|
||||
fall back to CPU (unusable for 32B-class models). Check nvidia driver /
|
||||
CUDA wheel compatibility before proceeding to Phase 2.
|
||||
when: "'True' not in vllm_torch_cuda_check.stdout"
|
||||
38
ansible/roles/deploy-vllm/tasks/main.yml
Normal file
38
ansible/roles/deploy-vllm/tasks/main.yml
Normal file
@@ -0,0 +1,38 @@
|
||||
# ------------------------------------------------------------------------------
|
||||
# FILE: roles/deploy-vllm/tasks/main.yml
|
||||
# ROLE: deploy-vllm — orchestrator. Phased, idempotent, mirrors the pattern
|
||||
# used by roles/llm-inference and roles/llm-inference-multimodel:
|
||||
# Phase 1: dependencies (Python/venv/vLLM/CUDA/nvidia-smi)
|
||||
# Phase 2: model downloads (~/.vllm-cache, checksum-verified)
|
||||
# Phase 3: systemd service(s)
|
||||
# Phase 4: API key management (1Password)
|
||||
# Phase 5: verification (health + smoke test)
|
||||
# Each phase is a separate task file so a partial re-run / targeted --tags
|
||||
# run is possible without re-reading the whole role.
|
||||
# ------------------------------------------------------------------------------
|
||||
|
||||
- name: Compute enabled model list (available to every phase/tag combination)
|
||||
ansible.builtin.set_fact:
|
||||
vllm_enabled_models: "{{ vllm_models | selectattr('enabled', 'equalto', true) | list }}"
|
||||
tags: [vllm, vllm-dependencies, vllm-models, vllm-api-key, vllm-systemd, vllm-verify]
|
||||
|
||||
- name: Phase 1 — Python & dependencies
|
||||
ansible.builtin.import_tasks: dependencies.yml
|
||||
tags: [vllm, vllm-dependencies]
|
||||
|
||||
- name: Phase 2 — Model downloads
|
||||
ansible.builtin.import_tasks: models.yml
|
||||
tags: [vllm, vllm-models]
|
||||
|
||||
- name: Phase 3 — API key management
|
||||
ansible.builtin.import_tasks: api-key.yml
|
||||
tags: [vllm, vllm-api-key]
|
||||
|
||||
- name: Phase 4 — vLLM systemd service(s)
|
||||
ansible.builtin.import_tasks: systemd.yml
|
||||
tags: [vllm, vllm-systemd]
|
||||
|
||||
- name: Phase 5 — Verification
|
||||
ansible.builtin.import_tasks: verify.yml
|
||||
tags: [vllm, vllm-verify]
|
||||
when: vllm_service_state == 'started'
|
||||
87
ansible/roles/deploy-vllm/tasks/models.yml
Normal file
87
ansible/roles/deploy-vllm/tasks/models.yml
Normal file
@@ -0,0 +1,87 @@
|
||||
# ------------------------------------------------------------------------------
|
||||
# FILE: roles/deploy-vllm/tasks/models.yml
|
||||
# PHASE 2: Model downloads via huggingface-cli into {{ vllm_hf_hub_cache }}.
|
||||
#
|
||||
# Idempotency: HuggingFace's on-disk cache layout is
|
||||
# {cache}/models--{org}--{repo}/snapshots/{revision}/...
|
||||
# We stat for an existing snapshots dir before downloading — if present with
|
||||
# at least one entry, skip (huggingface-cli download is itself resumable/
|
||||
# idempotent, but this avoids even the "check remote manifest" round trip on
|
||||
# every run and gives a clean "already staged" line in output).
|
||||
#
|
||||
# Pitfall (t_3dddf37d, homelab-llm-inference skill): a config/template landing
|
||||
# is NOT the same as the model being staged. Always verify via `ls`/`du` on
|
||||
# the actual host, never trust a prior task's claim alone.
|
||||
# ------------------------------------------------------------------------------
|
||||
|
||||
- name: Ensure model cache directory exists
|
||||
ansible.builtin.file:
|
||||
path: "{{ vllm_hf_hub_cache }}"
|
||||
state: directory
|
||||
owner: "{{ vllm_venv_owner }}"
|
||||
group: "{{ vllm_venv_owner }}"
|
||||
mode: "0755"
|
||||
become: true
|
||||
|
||||
- name: Report models to be staged this run
|
||||
ansible.builtin.debug:
|
||||
msg: "{{ vllm_enabled_models | map(attribute='id') | list }}"
|
||||
|
||||
- name: Check for existing snapshot dir per enabled model
|
||||
ansible.builtin.stat:
|
||||
path: "{{ vllm_hf_hub_cache }}/models--{{ item.hf_repo | regex_replace('/', '--') }}/snapshots"
|
||||
loop: "{{ vllm_enabled_models }}"
|
||||
loop_control:
|
||||
label: "{{ item.id }}"
|
||||
register: vllm_model_snapshot_stat
|
||||
|
||||
- name: Download model repo(s) not yet staged
|
||||
ansible.builtin.command:
|
||||
cmd: >-
|
||||
{{ vllm_venv_path }}/bin/hf download {{ item.item.hf_repo }}
|
||||
--cache-dir {{ vllm_hf_hub_cache }}
|
||||
become: true
|
||||
become_user: "{{ vllm_venv_owner }}"
|
||||
environment:
|
||||
HF_HUB_ENABLE_HF_TRANSFER: "0"
|
||||
loop: "{{ vllm_model_snapshot_stat.results }}"
|
||||
loop_control:
|
||||
label: "{{ item.item.id }}"
|
||||
when: not (item.stat.exists | default(false)) or (item.stat.isdir | default(false) and item.stat.size == 0)
|
||||
register: vllm_model_download
|
||||
# Full-size model pulls (9-18GB for 32B AWQ) can take a long time on
|
||||
# homelab bandwidth — allow up to 1 hour per model.
|
||||
async: 3600
|
||||
poll: 30
|
||||
|
||||
- name: Re-stat snapshot dirs to confirm download landed
|
||||
ansible.builtin.stat:
|
||||
path: "{{ vllm_hf_hub_cache }}/models--{{ item.hf_repo | regex_replace('/', '--') }}/snapshots"
|
||||
loop: "{{ vllm_enabled_models }}"
|
||||
loop_control:
|
||||
label: "{{ item.id }}"
|
||||
register: vllm_model_snapshot_verify
|
||||
|
||||
- name: Fail if any enabled model failed to stage
|
||||
ansible.builtin.fail:
|
||||
msg: "Model {{ item.item.id }} ({{ item.item.hf_repo }}) is not present at {{ vllm_hf_hub_cache }} after download step."
|
||||
loop: "{{ vllm_model_snapshot_verify.results }}"
|
||||
loop_control:
|
||||
label: "{{ item.item.id }}"
|
||||
when: not (item.stat.exists | default(false))
|
||||
|
||||
- name: Compute on-disk size of each staged model (sanity check, not a strict checksum)
|
||||
ansible.builtin.command:
|
||||
cmd: "du -sh {{ vllm_hf_hub_cache }}/models--{{ item.hf_repo | regex_replace('/', '--') }}"
|
||||
loop: "{{ vllm_enabled_models }}"
|
||||
loop_control:
|
||||
label: "{{ item.id }}"
|
||||
register: vllm_model_size
|
||||
changed_when: false
|
||||
|
||||
- name: Report staged model sizes
|
||||
ansible.builtin.debug:
|
||||
msg: "{{ item.stdout }}"
|
||||
loop: "{{ vllm_model_size.results }}"
|
||||
loop_control:
|
||||
label: "{{ item.item.id }}"
|
||||
56
ansible/roles/deploy-vllm/tasks/systemd.yml
Normal file
56
ansible/roles/deploy-vllm/tasks/systemd.yml
Normal file
@@ -0,0 +1,56 @@
|
||||
# ------------------------------------------------------------------------------
|
||||
# FILE: roles/deploy-vllm/tasks/systemd.yml
|
||||
# PHASE 4: vLLM systemd service(s).
|
||||
#
|
||||
# vLLM 0.5.x serves ONE model per process. The primary model (role: primary,
|
||||
# e.g. Qwen2.5-32B-Instruct-AWQ) gets the canonical unit name vllm.service
|
||||
# (matches the spec's /etc/systemd/system/vllm.service). Any additional
|
||||
# enabled models (aux/embedding, added in later phases per the "Phased
|
||||
# Strategy") each get their own instance unit vllm-<id>.service on a distinct
|
||||
# port, generated from the same template.
|
||||
#
|
||||
# Idempotent: ansible.builtin.template only reports changed when content
|
||||
# actually differs; the "restart vllm services" handler only fires on that
|
||||
# change (or on api-key.yml rewriting the shared env file).
|
||||
#
|
||||
# vllm_service_state defaults to "stopped" — this role stages everything
|
||||
# (venv, model, unit file, key) but does NOT flip production traffic without
|
||||
# an explicit --extra-vars vllm_service_state=started, matching the deploy-
|
||||
# then-validate-then-cutover sequencing approved for astro-orbiter.
|
||||
# ------------------------------------------------------------------------------
|
||||
|
||||
- name: Render systemd unit for each enabled model
|
||||
ansible.builtin.template:
|
||||
src: vllm.service.j2
|
||||
dest: "/etc/systemd/system/{{ 'vllm.service' if item.role == 'primary' else 'vllm-' + item.id + '.service' }}"
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
loop: "{{ vllm_enabled_models }}"
|
||||
loop_control:
|
||||
label: "{{ item.id }}"
|
||||
become: true
|
||||
notify: reload systemd
|
||||
|
||||
- name: Render workspace helper script (manual debugging / smoke-testing)
|
||||
ansible.builtin.template:
|
||||
src: vllm-workspace.sh.j2
|
||||
dest: "/home/{{ vllm_venv_owner }}/vllm-workspace.sh"
|
||||
owner: "{{ vllm_venv_owner }}"
|
||||
group: "{{ vllm_venv_owner }}"
|
||||
mode: "0750"
|
||||
become: true
|
||||
|
||||
- name: Flush handlers so unit files are known to systemd before enabling
|
||||
ansible.builtin.meta: flush_handlers
|
||||
|
||||
- name: Enable/disable + start/stop each vLLM systemd unit
|
||||
ansible.builtin.systemd:
|
||||
name: "{{ 'vllm.service' if item.role == 'primary' else 'vllm-' + item.id + '.service' }}"
|
||||
enabled: "{{ vllm_service_enabled }}"
|
||||
state: "{{ vllm_service_state }}"
|
||||
daemon_reload: true
|
||||
loop: "{{ vllm_enabled_models }}"
|
||||
loop_control:
|
||||
label: "{{ item.id }}"
|
||||
become: true
|
||||
163
ansible/roles/deploy-vllm/tasks/verify.yml
Normal file
163
ansible/roles/deploy-vllm/tasks/verify.yml
Normal file
@@ -0,0 +1,163 @@
|
||||
# ------------------------------------------------------------------------------
|
||||
# FILE: roles/deploy-vllm/tasks/verify.yml
|
||||
# PHASE 5: Verification.
|
||||
#
|
||||
# Only runs when vllm_service_state == 'started' (main.yml gate) — staging a
|
||||
# stopped service is a valid, intentional end state during the deploy-first-
|
||||
# validate-before-cutover sequencing, and there is nothing to verify yet.
|
||||
#
|
||||
# Pitfall (homelab-llm-inference skill): vLLM torch.compile takes 4+ minutes
|
||||
# AFTER weights load before /health returns 200. retries=30, delay=10 (5 min
|
||||
# ceiling) — do not shrink this or health checks will false-negative on a
|
||||
# perfectly healthy but still-warming-up service.
|
||||
# ------------------------------------------------------------------------------
|
||||
|
||||
- name: Wait for each enabled model's systemd unit to be active
|
||||
ansible.builtin.systemd:
|
||||
name: "{{ 'vllm.service' if item.role == 'primary' else 'vllm-' + item.id + '.service' }}"
|
||||
loop: "{{ vllm_enabled_models }}"
|
||||
loop_control:
|
||||
label: "{{ item.id }}"
|
||||
register: vllm_unit_status
|
||||
become: true
|
||||
|
||||
- name: Report systemd unit status
|
||||
ansible.builtin.debug:
|
||||
msg: "{{ item.item.id }}: {{ item.status.ActiveState }} ({{ item.status.SubState }})"
|
||||
loop: "{{ vllm_unit_status.results }}"
|
||||
loop_control:
|
||||
label: "{{ item.item.id }}"
|
||||
|
||||
- name: Fail if any unit is not active
|
||||
ansible.builtin.fail:
|
||||
msg: "{{ item.item.id }} systemd unit is {{ item.status.ActiveState }}, expected active."
|
||||
loop: "{{ vllm_unit_status.results }}"
|
||||
loop_control:
|
||||
label: "{{ item.item.id }}"
|
||||
when: item.status.ActiveState != 'active'
|
||||
|
||||
- name: Poll /health until 200 (torch.compile warmup can take 4-5 minutes)
|
||||
ansible.builtin.uri:
|
||||
url: "http://127.0.0.1:{{ item.port }}/health"
|
||||
status_code: 200
|
||||
timeout: 15
|
||||
loop: "{{ vllm_enabled_models }}"
|
||||
loop_control:
|
||||
label: "{{ item.id }}"
|
||||
register: vllm_health_check
|
||||
until: vllm_health_check is succeeded
|
||||
retries: "{{ vllm_health_check_retries }}"
|
||||
delay: "{{ vllm_health_check_delay }}"
|
||||
|
||||
- name: Query /v1/models on each enabled instance
|
||||
ansible.builtin.uri:
|
||||
url: "http://127.0.0.1:{{ item.port }}/v1/models"
|
||||
headers:
|
||||
Authorization: "Bearer {{ vllm_api_key_lookup.stdout }}"
|
||||
return_content: true
|
||||
loop: "{{ vllm_enabled_models }}"
|
||||
loop_control:
|
||||
label: "{{ item.id }}"
|
||||
register: vllm_models_response
|
||||
no_log: true
|
||||
|
||||
- name: Assert /v1/models returns the expected served model name
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- item.item.id in (item.content)
|
||||
fail_msg: "/v1/models on port {{ item.item.port }} did not list expected model id {{ item.item.id }}"
|
||||
success_msg: "/v1/models confirmed {{ item.item.id }} is served on port {{ item.item.port }}"
|
||||
loop: "{{ vllm_models_response.results }}"
|
||||
loop_control:
|
||||
label: "{{ item.item.id }}"
|
||||
|
||||
- name: Split enabled models into completion-serving vs embedding for the right smoke test
|
||||
ansible.builtin.set_fact:
|
||||
vllm_completion_models: "{{ vllm_enabled_models | rejectattr('role', 'equalto', 'embedding') | list }}"
|
||||
vllm_embedding_models: "{{ vllm_enabled_models | selectattr('role', 'equalto', 'embedding') | list }}"
|
||||
|
||||
- name: Run a live completion smoke test against each completion-serving instance
|
||||
ansible.builtin.uri:
|
||||
url: "http://127.0.0.1:{{ item.port }}/v1/completions"
|
||||
method: POST
|
||||
headers:
|
||||
Authorization: "Bearer {{ vllm_api_key_lookup.stdout }}"
|
||||
Content-Type: "application/json"
|
||||
body_format: json
|
||||
body:
|
||||
model: "{{ item.id }}"
|
||||
prompt: "The capital of France is"
|
||||
max_tokens: 8
|
||||
temperature: 0
|
||||
timeout: 60
|
||||
status_code: 200
|
||||
loop: "{{ vllm_completion_models }}"
|
||||
loop_control:
|
||||
label: "{{ item.id }}"
|
||||
register: vllm_completion_test
|
||||
no_log: true
|
||||
|
||||
- name: Report completion smoke test result
|
||||
ansible.builtin.debug:
|
||||
msg: "{{ item.item.id }}: HTTP {{ item.status }} — completion smoke test passed"
|
||||
loop: "{{ vllm_completion_test.results }}"
|
||||
loop_control:
|
||||
label: "{{ item.item.id }}"
|
||||
|
||||
# Embedding-mode vLLM instances (--runner pooling --convert embed) do NOT
|
||||
# serve /v1/completions — only /v1/embeddings (and /pooling). A completions
|
||||
# smoke test against one 400s immediately. Verify with a real vector request
|
||||
# instead, and assert the response actually contains a non-empty float vector
|
||||
# (not just HTTP 200 — an empty/malformed embedding would still 200).
|
||||
- name: Run a live embeddings smoke test against each embedding-mode instance
|
||||
ansible.builtin.uri:
|
||||
url: "http://127.0.0.1:{{ item.port }}/v1/embeddings"
|
||||
method: POST
|
||||
headers:
|
||||
Authorization: "Bearer {{ vllm_api_key_lookup.stdout }}"
|
||||
Content-Type: "application/json"
|
||||
body_format: json
|
||||
body:
|
||||
model: "{{ item.id }}"
|
||||
input: "The capital of France is Paris."
|
||||
timeout: 60
|
||||
status_code: 200
|
||||
return_content: true
|
||||
loop: "{{ vllm_embedding_models }}"
|
||||
loop_control:
|
||||
label: "{{ item.id }}"
|
||||
register: vllm_embedding_test
|
||||
no_log: true
|
||||
|
||||
- name: Assert embeddings smoke test returned a non-empty float vector
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- (item.json.data[0].embedding | length) > 0
|
||||
fail_msg: "/v1/embeddings on port {{ item.item.port }} did not return a non-empty embedding vector"
|
||||
success_msg: "/v1/embeddings confirmed {{ item.item.id }} returns a {{ item.json.data[0].embedding | length }}-dim vector"
|
||||
loop: "{{ vllm_embedding_test.results }}"
|
||||
loop_control:
|
||||
label: "{{ item.item.id }}"
|
||||
|
||||
- name: Check journalctl for each enabled unit is free of ERROR/Traceback since last start
|
||||
ansible.builtin.shell: |
|
||||
set -o pipefail
|
||||
journalctl -u {{ 'vllm.service' if item.role == 'primary' else 'vllm-' + item.id + '.service' }} --since "10 min ago" | grep -iE "error|traceback" | grep -v "no entries" || true
|
||||
args:
|
||||
executable: /bin/bash
|
||||
loop: "{{ vllm_enabled_models }}"
|
||||
loop_control:
|
||||
label: "{{ item.id }}"
|
||||
register: vllm_journal_errors
|
||||
changed_when: false
|
||||
become: true
|
||||
|
||||
- name: Report journalctl scan result
|
||||
ansible.builtin.debug:
|
||||
msg: >-
|
||||
{{ item.item.id ~ ': journalctl clean — no error/traceback lines in the last 10 minutes'
|
||||
if item.stdout | trim | length == 0
|
||||
else item.item.id ~ ' WARNING — journalctl lines matched error/traceback: ' ~ item.stdout }}
|
||||
loop: "{{ vllm_journal_errors.results }}"
|
||||
loop_control:
|
||||
label: "{{ item.item.id }}"
|
||||
54
ansible/roles/deploy-vllm/templates/vllm-workspace.sh.j2
Normal file
54
ansible/roles/deploy-vllm/templates/vllm-workspace.sh.j2
Normal file
@@ -0,0 +1,54 @@
|
||||
#!/usr/bin/env bash
|
||||
# ------------------------------------------------------------------------------
|
||||
# FILE: vllm-workspace.sh — deployed by roles/deploy-vllm to
|
||||
# /home/{{ vllm_venv_owner }}/vllm-workspace.sh
|
||||
#
|
||||
# Convenience wrapper for manual debugging / smoke-testing the vLLM venv
|
||||
# without having to remember the venv path or model roster each time.
|
||||
# Regenerated on every Ansible run — do not hand-edit, edit the template
|
||||
# instead (roles/deploy-vllm/templates/vllm-workspace.sh.j2).
|
||||
# ------------------------------------------------------------------------------
|
||||
set -euo pipefail
|
||||
|
||||
VENV="{{ vllm_venv_path }}"
|
||||
CACHE="{{ vllm_hf_hub_cache }}"
|
||||
API_KEY_FILE="{{ vllm_api_key_env_file }}"
|
||||
|
||||
usage() {
|
||||
cat <<EOF
|
||||
Usage: $0 <command>
|
||||
|
||||
Commands:
|
||||
activate Print the command to source the vLLM venv
|
||||
version Print installed vLLM + torch/CUDA versions
|
||||
models List staged model snapshots in the HF cache
|
||||
curl-models curl /v1/models on each enabled instance (requires sudo to read API key)
|
||||
logs <unit> Tail journalctl for a vllm systemd unit (e.g. vllm.service)
|
||||
EOF
|
||||
}
|
||||
|
||||
case "${1:-}" in
|
||||
activate)
|
||||
echo "source $VENV/bin/activate"
|
||||
;;
|
||||
version)
|
||||
"$VENV/bin/python" -c 'import vllm, torch; print("vllm", vllm.__version__); print("torch", torch.__version__, "cuda", torch.version.cuda, "available", torch.cuda.is_available())'
|
||||
;;
|
||||
models)
|
||||
find "$CACHE" -maxdepth 1 -type d -name 'models--*' -printf '%f\n' 2>/dev/null || echo "(no models staged yet)"
|
||||
;;
|
||||
curl-models)
|
||||
{% for item in vllm_enabled_models | default([]) %}
|
||||
echo "--- {{ item.id }} (:{{ item.port }}) ---"
|
||||
curl -s -H "Authorization: Bearer $(sudo grep -oP '(?<=VLLM_API_KEY=).*' "$API_KEY_FILE")" \
|
||||
http://127.0.0.1:{{ item.port }}/v1/models | python3 -m json.tool || true
|
||||
{% endfor %}
|
||||
;;
|
||||
logs)
|
||||
sudo journalctl -u "${2:-vllm.service}" -f
|
||||
;;
|
||||
*)
|
||||
usage
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
85
ansible/roles/deploy-vllm/templates/vllm.service.j2
Normal file
85
ansible/roles/deploy-vllm/templates/vllm.service.j2
Normal file
@@ -0,0 +1,85 @@
|
||||
[Unit]
|
||||
Description=vLLM OpenAI-compatible inference server — {{ item.id }} ({{ item.hf_repo }})
|
||||
After=network-online.target nvidia-persistenced.service
|
||||
Wants=network-online.target nvidia-persistenced.service
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
User={{ vllm_venv_owner }}
|
||||
Group={{ vllm_venv_owner }}
|
||||
EnvironmentFile={{ vllm_api_key_env_file }}
|
||||
Environment="HOME=/home/{{ vllm_venv_owner }}"
|
||||
Environment="HF_HUB_CACHE={{ vllm_hf_hub_cache }}"
|
||||
Environment="HF_HOME={{ vllm_cache_dir }}"
|
||||
# vLLM's torch.compile path shells out to `ninja` by bare name (not via
|
||||
# venv-relative path) — without the venv's bin/ on PATH, systemd's minimal
|
||||
# default PATH causes FileNotFoundError: 'ninja' deep in compile, even
|
||||
# though `pip install vllm` installs the ninja package (and its console
|
||||
# script) INTO the venv. Caught during Phase 5 validation (t_ca1af9fb,
|
||||
# 2026-08-31): interactive SSH sessions have a different PATH than systemd
|
||||
# services, so this only reproduces under systemd, not manual testing.
|
||||
Environment="PATH={{ vllm_venv_path }}/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin"
|
||||
# FlashInfer's bundled sampling.cu JIT-compiles against a cub template API
|
||||
# (BlockAdjacentDifference::FlagHeads) that this flashinfer/CUDA toolkit
|
||||
# combination does not provide on RTX 3090 (SM86) — 100 compile errors,
|
||||
# confirmed upstream-known (vLLM GH #23023, #44305: FlashInfer sampler JIT
|
||||
# breaks on various SM targets across flashinfer/vLLM version combos).
|
||||
# Falls back to vLLM's native PyTorch sampler, which is fully supported and
|
||||
# only marginally slower for single-request/low-concurrency serving. Caught
|
||||
# during Phase 5 validation (t_ca1af9fb, 2026-08-31).
|
||||
Environment="VLLM_USE_FLASHINFER_SAMPLER=0"
|
||||
|
||||
ExecStart={{ vllm_venv_path }}/bin/python -m vllm.entrypoints.openai.api_server \
|
||||
--model {{ item.hf_repo }} \
|
||||
--served-model-name {{ item.id }} \
|
||||
--host {{ vllm_serve_host }} \
|
||||
--port {{ item.port }} \
|
||||
{% if item.role == 'embedding' %}
|
||||
--runner pooling \
|
||||
--convert embed \
|
||||
{% endif %}
|
||||
{% if item.trust_remote_code is defined and item.trust_remote_code %}
|
||||
--trust-remote-code \
|
||||
{% endif %}
|
||||
{% if item.enforce_eager is defined and item.enforce_eager %}
|
||||
--enforce-eager \
|
||||
{% endif %}
|
||||
{% if item.enable_auto_tool_choice is defined and item.enable_auto_tool_choice %}
|
||||
--enable-auto-tool-choice \
|
||||
{% endif %}
|
||||
{% if item.tool_call_parser is defined %}
|
||||
--tool-call-parser {{ item.tool_call_parser }} \
|
||||
{% endif %}
|
||||
{% if item.reasoning_parser is defined %}
|
||||
--reasoning-parser {{ item.reasoning_parser }} \
|
||||
{% endif %}
|
||||
{% if item.quantization is defined and item.quantization != 'none' %}
|
||||
--quantization {{ item.quantization }} \
|
||||
{% endif %}
|
||||
{% if item.kv_cache_dtype is defined %}
|
||||
--kv-cache-dtype {{ item.kv_cache_dtype }} \
|
||||
{% endif %}
|
||||
{% if item.kv_cache_memory_bytes is defined %}
|
||||
--kv-cache-memory-bytes {{ item.kv_cache_memory_bytes }} \
|
||||
{% endif %}
|
||||
--gpu-memory-utilization {{ item.gpu_memory_utilization }} \
|
||||
--max-model-len {{ item.max_model_len }} \
|
||||
--dtype {{ vllm_dtype }} \
|
||||
--api-key ${VLLM_API_KEY} \
|
||||
{% if item.role != 'embedding' %}
|
||||
--enable-prefix-caching
|
||||
{% else %}
|
||||
--no-enable-prefix-caching
|
||||
{% endif %}
|
||||
|
||||
Restart={{ vllm_restart_policy }}
|
||||
RestartSec=10
|
||||
# vLLM torch.compile can take 4+ minutes before /health responds even after
|
||||
# weights are loaded (homelab-llm-inference skill pitfall) — give it room.
|
||||
TimeoutStartSec=600
|
||||
StandardOutput=journal
|
||||
StandardError=journal
|
||||
SyslogIdentifier=vllm-{{ item.id }}
|
||||
|
||||
[Install]
|
||||
WantedBy=multi-user.target
|
||||
95
ansible/roles/expand_root_lv/README.md
Normal file
95
ansible/roles/expand_root_lv/README.md
Normal file
@@ -0,0 +1,95 @@
|
||||
# expand-root-lv role
|
||||
|
||||
Idempotent role that grows the root partition (via `growpart`), extends the
|
||||
root LVM logical volume to fill its volume group, and grows the underlying
|
||||
filesystem (ext4 or xfs).
|
||||
|
||||
## Where this runs in the lifecycle
|
||||
|
||||
Part of the **day0** host-provisioning lifecycle. The canonical entry
|
||||
points are:
|
||||
|
||||
```
|
||||
playbooks/day0_expand_root_lv.yml # standalone
|
||||
playbooks/day0_provision.yml # umbrella (baseline + expand_root_lv)
|
||||
```
|
||||
|
||||
Day1 application-deploy playbooks should NOT include this role —
|
||||
day0 is assumed complete before day1 begins.
|
||||
|
||||
## Why this role exists
|
||||
|
||||
The Ubuntu Server autoinstall template (used by the mk-labs `wed`-baked
|
||||
VM templates) provisions the root LV at roughly half the available disk
|
||||
size — a longstanding installer default that surprises every operator
|
||||
who hasn't been bitten by it before. ~90% of mk-labs VMs need this
|
||||
fix-up before they're fully useful.
|
||||
|
||||
After a Proxmox disk grow (increasing the VM disk size), the partition
|
||||
table, physical volume, logical volume, and filesystem all need to be
|
||||
extended in sequence. This role automates the full chain.
|
||||
|
||||
## Workflow
|
||||
|
||||
1. **growpart** — resizes the underlying partition to claim the newly
|
||||
provisioned disk space. Idempotent: no-op when the partition already
|
||||
fills the disk.
|
||||
2. **pvresize** — tells the kernel/LVM about the new partition size so
|
||||
the VG sees the additional free PEs.
|
||||
3. **lvextend** — extends the LV to claim all free PE in the VG
|
||||
(`+100%FREE`). No-op when there's nothing to grow.
|
||||
4. **fs grow** — `resize2fs` (ext4) or `xfs_growfs` (xfs), dispatched by
|
||||
detected filesystem type.
|
||||
|
||||
## Idempotency
|
||||
|
||||
- If `vg_free_count == 0`, the `lvextend` step is skipped and the
|
||||
filesystem-grow step is also skipped (nothing to resize against).
|
||||
- If the target volume group doesn't exist on the host (e.g. a non-LVM
|
||||
layout), the role exits cleanly via `meta: end_play`.
|
||||
- Safe to leave in a recurring playbook so future disk expansions
|
||||
(Proxmox-side disk grow → reboot → run role) are picked up
|
||||
automatically.
|
||||
|
||||
## Opt-out for multi-LV hosts
|
||||
|
||||
If a host will have a **second logical volume in the same VG** (e.g. a
|
||||
dedicated `/var/lib/postgresql` LV for a database server), this role's
|
||||
"grow root to fill VG" behavior is wrong — it will consume the free PE
|
||||
that was being reserved for the second LV.
|
||||
|
||||
Set in `host_vars/<host>.yml`:
|
||||
|
||||
```yaml
|
||||
expand_root_lv_skip: true
|
||||
```
|
||||
|
||||
The day0 playbook checks this flag and skips the role cleanly.
|
||||
|
||||
To skip only the partition growstep while keeping LV/FS expansion
|
||||
(e.g. when the partition already covers the whole disk but the LV was
|
||||
provisioned small by the template), set:
|
||||
|
||||
```yaml
|
||||
expand_root_lv_pv_partition: undefined
|
||||
```
|
||||
|
||||
## Defaults
|
||||
|
||||
| Variable | Default | Purpose |
|
||||
|-------------------------------|---------------|-----------------------------------------------------|
|
||||
| `expand_root_lv_vg_name` | `ubuntu-vg` | LVM volume group name (Ubuntu installer default). |
|
||||
| `expand_root_lv_lv_name` | `ubuntu-lv` | LVM logical volume name (Ubuntu installer default). |
|
||||
| `expand_root_lv_pv_partition` | `/dev/sda3` | Partition backing the PV; grown via growpart. |
|
||||
| `expand_root_lv_mountpoint` | `/` | Mountpoint of the filesystem to grow. |
|
||||
|
||||
Override the VG/LV/PV names in `host_vars/<host>.yml` for hosts that use a
|
||||
different layout.
|
||||
|
||||
## Limitations
|
||||
|
||||
- The `growpart` step requires the `cloud-guest-utils` package. The role
|
||||
installs it automatically on Debian/Ubuntu hosts when
|
||||
`expand_root_lv_pv_partition` is defined.
|
||||
- Only supports ext4 and xfs filesystems. Other filesystem types (btrfs,
|
||||
etc.) are left as a future enhancement.
|
||||
32
ansible/roles/expand_root_lv/defaults/main.yml
Normal file
32
ansible/roles/expand_root_lv/defaults/main.yml
Normal file
@@ -0,0 +1,32 @@
|
||||
---
|
||||
# ============================================================================
|
||||
# expand-root-lv role defaults
|
||||
# ============================================================================
|
||||
# Extends the root LVM logical volume to fill its volume group, then grows
|
||||
# the underlying filesystem to match. Idempotent: when there's no free PE
|
||||
# in the VG (i.e. the LV already fills the VG), the lvextend step is a
|
||||
# no-op and resize2fs/xfs_growfs simply confirms the filesystem is at
|
||||
# capacity.
|
||||
#
|
||||
# Designed for Ubuntu cloud-image-style installations where the autoinstall
|
||||
# template provisions an LV at half the disk size (the Ubuntu Server
|
||||
# installer's longstanding default). Run once after VM provisioning to
|
||||
# reclaim the unallocated PE; safe to leave in a day1 playbook so future
|
||||
# disk expansions are picked up automatically.
|
||||
# ============================================================================
|
||||
|
||||
# The LV and VG names follow the Ubuntu Server installer's convention.
|
||||
# Override per-host if your template differs.
|
||||
expand_root_lv_vg_name: ubuntu-vg
|
||||
expand_root_lv_lv_name: ubuntu-lv
|
||||
|
||||
# The partition that backs the physical volume. After a Proxmox disk grow,
|
||||
# growpart must resize this partition before pvresize/lvextend can claim
|
||||
# the new space. This is the full device path (e.g. /dev/sda3).
|
||||
# If undefined, the growpart/pvresize steps are skipped.
|
||||
expand_root_lv_pv_partition: /dev/sda3
|
||||
|
||||
# Mount point we expect to be backed by the target LV. Used purely for
|
||||
# the resize2fs / xfs_growfs decision — the role inspects this path's
|
||||
# filesystem type and dispatches to the correct grow command.
|
||||
expand_root_lv_mountpoint: /
|
||||
23
ansible/roles/expand_root_lv/meta/main.yml
Normal file
23
ansible/roles/expand_root_lv/meta/main.yml
Normal file
@@ -0,0 +1,23 @@
|
||||
---
|
||||
galaxy_info:
|
||||
role_name: expand_root_lv
|
||||
author: JARVIS
|
||||
description: >-
|
||||
Idempotent role that extends the root LVM logical volume to fill its
|
||||
volume group and grows the underlying filesystem (ext4 or xfs). Fixes
|
||||
the half-disk LV that the Ubuntu Server autoinstall template ships
|
||||
with by default.
|
||||
license: MIT
|
||||
min_ansible_version: "2.14"
|
||||
platforms:
|
||||
- name: Ubuntu
|
||||
versions:
|
||||
- noble
|
||||
- jammy
|
||||
galaxy_tags:
|
||||
- lvm
|
||||
- cloud-init
|
||||
- homelab
|
||||
- storage
|
||||
|
||||
dependencies: []
|
||||
147
ansible/roles/expand_root_lv/tasks/main.yml
Normal file
147
ansible/roles/expand_root_lv/tasks/main.yml
Normal file
@@ -0,0 +1,147 @@
|
||||
---
|
||||
# ============================================================================
|
||||
# expand-root-lv / main
|
||||
# ----------------------------------------------------------------------------
|
||||
# 0. Ensure growpart is available (cloud-guest-utils provides the growpart binary)
|
||||
# 1. Grow the partition (growpart) if a PV partition device is defined
|
||||
# 2. Resize the physical volume (pvresize) to pick up the new partition size
|
||||
# 3. Confirm the target VG exists (skip role cleanly on non-LVM hosts).
|
||||
# 4. Read free physical-extent count for the VG.
|
||||
# 5. Extend the LV to +100%FREE only when free_pe > 0.
|
||||
# 6. Grow the filesystem on the mountpoint (ext4 -> resize2fs, xfs -> xfs_growfs).
|
||||
# Each step is idempotent and skips when there's nothing to do.
|
||||
# ============================================================================
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Step 0: Ensure growpart is available
|
||||
# ---------------------------------------------------------------------------
|
||||
- name: Ensure cloud-guest-utils (growpart) is installed
|
||||
ansible.builtin.package:
|
||||
name: cloud-guest-utils
|
||||
state: present
|
||||
when: expand_root_lv_pv_partition is defined
|
||||
tags: [growpart, always]
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Step 1: Grow the partition that backs the PV
|
||||
# ---------------------------------------------------------------------------
|
||||
# growpart expects: growpart <device> <partition_number>
|
||||
# e.g. growpart /dev/sda 3 — NOT growpart /dev/sda3
|
||||
# We split expand_root_lv_pv_partition (e.g. /dev/sda3) into device and part_no.
|
||||
|
||||
- name: Derive device and partition number from PV partition path
|
||||
ansible.builtin.set_fact:
|
||||
expand_root_lv_pv_device: "{{ expand_root_lv_pv_partition | regex_replace('p?(\\d+)$', '') }}"
|
||||
expand_root_lv_pv_part_no: "{{ expand_root_lv_pv_partition | regex_replace('.*p?(\\d+)$', '\\1') }}"
|
||||
when: expand_root_lv_pv_partition is defined
|
||||
tags: [growpart, always]
|
||||
|
||||
- name: Grow partition to fill disk (growpart)
|
||||
ansible.builtin.command:
|
||||
cmd: "growpart {{ expand_root_lv_pv_device }} {{ expand_root_lv_pv_part_no }}"
|
||||
register: growpart_result
|
||||
when: expand_root_lv_pv_partition is defined
|
||||
changed_when: growpart_result.rc == 0 and "NO CHANGE" not in growpart_result.stdout
|
||||
tags: [growpart, always]
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Step 2: Resize the physical volume to claim the new partition space
|
||||
# ---------------------------------------------------------------------------
|
||||
- name: Resize physical volume (pvresize)
|
||||
ansible.builtin.command:
|
||||
cmd: "pvresize {{ expand_root_lv_pv_partition }}"
|
||||
register: pvresize_result
|
||||
when: expand_root_lv_pv_partition is defined
|
||||
changed_when: pvresize_result.rc == 0
|
||||
tags: [pvresize, always]
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Step 3: Confirm the target VG exists (skip role cleanly on non-LVM hosts).
|
||||
# ---------------------------------------------------------------------------
|
||||
- name: Gather LVM facts
|
||||
ansible.builtin.command:
|
||||
cmd: "vgs --noheadings --nosuffix --units b -o vg_name,vg_free_count {{ expand_root_lv_vg_name }}"
|
||||
register: vg_info
|
||||
changed_when: false
|
||||
failed_when: false
|
||||
tags: [lvm, always]
|
||||
|
||||
- name: Skip role when target VG is absent
|
||||
ansible.builtin.meta: end_play
|
||||
when: vg_info.rc != 0
|
||||
tags: [lvm, always]
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Step 4: Parse free PE count for the VG
|
||||
# ---------------------------------------------------------------------------
|
||||
- name: Parse free PE count
|
||||
ansible.builtin.set_fact:
|
||||
expand_root_lv_free_pe: "{{ (vg_info.stdout.split() | last | int) if vg_info.stdout | length > 0 else 0 }}"
|
||||
tags: [lvm, always]
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Step 5: Extend the LV to +100%FREE only when free_pe > 0
|
||||
# ---------------------------------------------------------------------------
|
||||
- name: Extend LV to fill VG (only if free PE > 0)
|
||||
ansible.builtin.command:
|
||||
cmd: "lvextend -l +100%FREE /dev/{{ expand_root_lv_vg_name }}/{{ expand_root_lv_lv_name }}"
|
||||
register: lvextend_result
|
||||
when: expand_root_lv_free_pe | int > 0
|
||||
changed_when: lvextend_result.rc == 0
|
||||
tags: [lvm, always]
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Step 6: Detect filesystem type at mountpoint and grow
|
||||
# ---------------------------------------------------------------------------
|
||||
- name: Detect filesystem type at mountpoint
|
||||
ansible.builtin.command:
|
||||
cmd: "findmnt {{ expand_root_lv_mountpoint }} -no FSTYPE"
|
||||
register: fstype_result
|
||||
changed_when: false
|
||||
tags: [filesystem, always]
|
||||
|
||||
- name: Set filesystem type fact
|
||||
ansible.builtin.set_fact:
|
||||
expand_root_lv_fstype: "{{ fstype_result.stdout | trim }}"
|
||||
tags: [filesystem, always]
|
||||
|
||||
- name: Grow ext4 filesystem
|
||||
ansible.builtin.command:
|
||||
cmd: "resize2fs /dev/{{ expand_root_lv_vg_name }}/{{ expand_root_lv_lv_name }}"
|
||||
register: resize_result
|
||||
when:
|
||||
- expand_root_lv_fstype == "ext4"
|
||||
- (growpart_result is defined and growpart_result.changed) or
|
||||
(pvresize_result is defined and pvresize_result.changed) or
|
||||
(lvextend_result is defined and lvextend_result.changed) or
|
||||
(growpart_result is not defined)
|
||||
changed_when: resize_result.rc == 0
|
||||
tags: [filesystem, always]
|
||||
|
||||
- name: Grow xfs filesystem
|
||||
ansible.builtin.command:
|
||||
cmd: "xfs_growfs {{ expand_root_lv_mountpoint }}"
|
||||
register: xfs_result
|
||||
when:
|
||||
- expand_root_lv_fstype == "xfs"
|
||||
- (growpart_result is defined and growpart_result.changed) or
|
||||
(pvresize_result is defined and pvresize_result.changed) or
|
||||
(lvextend_result is defined and lvextend_result.changed) or
|
||||
(growpart_result is not defined)
|
||||
changed_when: xfs_result.rc == 0
|
||||
tags: [filesystem, always]
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Step 7: Report current root size
|
||||
# ---------------------------------------------------------------------------
|
||||
- name: Report current root size
|
||||
ansible.builtin.command:
|
||||
cmd: "df -h {{ expand_root_lv_mountpoint }}"
|
||||
register: df_result
|
||||
changed_when: false
|
||||
tags: [always]
|
||||
|
||||
- name: Show post-resize disk usage
|
||||
ansible.builtin.debug:
|
||||
msg: "{{ df_result.stdout_lines }}"
|
||||
tags: [always]
|
||||
26
ansible/roles/hermes/defaults/main.yml
Normal file
26
ansible/roles/hermes/defaults/main.yml
Normal file
@@ -0,0 +1,26 @@
|
||||
---
|
||||
# Hermes service user
|
||||
hermes_user: hermes
|
||||
hermes_group: hermes
|
||||
hermes_home: /home/hermes
|
||||
|
||||
# Install flags
|
||||
# Set to true if you don't need browser automation (skips Playwright/Chromium)
|
||||
hermes_skip_browser: false
|
||||
|
||||
# systemd service name (gateway)
|
||||
hermes_service_name: hermes
|
||||
|
||||
# Path where hermes binary will be accessible system-wide
|
||||
hermes_bin_symlink: /usr/local/bin/hermes
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Hermes Web UI (dashboard) — fronted by Traefik on lightning-lane via the
|
||||
# 'jarvis' service. Binds to 0.0.0.0 so Traefik can reach it from upstream;
|
||||
# --insecure is acceptable because TLS + auth are terminated at Traefik.
|
||||
# -----------------------------------------------------------------------------
|
||||
hermes_dashboard_enabled: true
|
||||
hermes_dashboard_service_name: hermes-dashboard
|
||||
hermes_dashboard_host: 0.0.0.0
|
||||
hermes_dashboard_port: 9119
|
||||
hermes_dashboard_insecure: true
|
||||
9
ansible/roles/hermes/handlers/main.yml
Normal file
9
ansible/roles/hermes/handlers/main.yml
Normal file
@@ -0,0 +1,9 @@
|
||||
---
|
||||
- name: reload systemd
|
||||
ansible.builtin.systemd:
|
||||
daemon_reload: true
|
||||
|
||||
- name: restart hermes
|
||||
ansible.builtin.systemd:
|
||||
name: "{{ hermes_service_name }}"
|
||||
state: restarted
|
||||
199
ansible/roles/hermes/tasks/main.yml
Normal file
199
ansible/roles/hermes/tasks/main.yml
Normal file
@@ -0,0 +1,199 @@
|
||||
---
|
||||
# ---------------------------------------------------------------------------
|
||||
# 1. System prerequisites (run as root via become)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
- name: Install system packages required by Hermes installer
|
||||
ansible.builtin.apt:
|
||||
name:
|
||||
- git
|
||||
- curl
|
||||
- ffmpeg
|
||||
- ripgrep
|
||||
state: present
|
||||
update_cache: true
|
||||
become: true
|
||||
|
||||
- name: Install Node.js 22 (required for browser automation and WhatsApp bridge)
|
||||
block:
|
||||
- name: Download NodeSource setup script
|
||||
ansible.builtin.get_url:
|
||||
url: https://deb.nodesource.com/setup_22.x
|
||||
dest: /tmp/nodesource_setup.sh
|
||||
mode: "0755"
|
||||
|
||||
- name: Run NodeSource setup script
|
||||
ansible.builtin.command: bash /tmp/nodesource_setup.sh
|
||||
args:
|
||||
creates: /etc/apt/sources.list.d/nodesource.list
|
||||
|
||||
- name: Install nodejs
|
||||
ansible.builtin.apt:
|
||||
name: nodejs
|
||||
state: present
|
||||
update_cache: true
|
||||
become: true
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 2. Install Playwright system deps for Chromium (root-only step)
|
||||
# Per docs: this is the one thing that genuinely needs root.
|
||||
# Skipped entirely if hermes_skip_browser is true.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
- name: Install Playwright Chromium system dependencies
|
||||
ansible.builtin.command: npx --yes playwright install-deps chromium
|
||||
become: true
|
||||
when: not hermes_skip_browser
|
||||
changed_when: true
|
||||
environment:
|
||||
DEBIAN_FRONTEND: noninteractive
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 3. Create dedicated hermes service user
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
- name: Create hermes group
|
||||
ansible.builtin.group:
|
||||
name: "{{ hermes_group }}"
|
||||
state: present
|
||||
system: true
|
||||
become: true
|
||||
|
||||
- name: Create hermes user
|
||||
ansible.builtin.user:
|
||||
name: "{{ hermes_user }}"
|
||||
group: "{{ hermes_group }}"
|
||||
home: "{{ hermes_home }}"
|
||||
shell: /bin/bash
|
||||
system: true
|
||||
create_home: true
|
||||
comment: "Hermes Agent service account"
|
||||
become: true
|
||||
|
||||
- name: Grant hermes user passwordless sudo
|
||||
ansible.builtin.copy:
|
||||
content: "hermes ALL=(ALL) NOPASSWD:ALL\n"
|
||||
dest: /etc/sudoers.d/hermes
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0440"
|
||||
validate: /usr/sbin/visudo -cf %s
|
||||
become: true
|
||||
|
||||
- name: Ensure hermes home directory has correct permissions
|
||||
ansible.builtin.file:
|
||||
path: "{{ hermes_home }}"
|
||||
owner: "{{ hermes_user }}"
|
||||
group: "{{ hermes_group }}"
|
||||
mode: "0750"
|
||||
state: directory
|
||||
become: true
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 4. Run the Hermes installer as the hermes user
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
- name: Check if hermes is already installed
|
||||
ansible.builtin.stat:
|
||||
path: "{{ hermes_home }}/.hermes/hermes-agent/venv/bin/hermes"
|
||||
register: hermes_binary
|
||||
|
||||
- name: Run Hermes installer as hermes user
|
||||
ansible.builtin.shell: |
|
||||
curl -fsSL https://raw.githubusercontent.com/NousResearch/hermes-agent/main/scripts/install.sh \
|
||||
| bash -s -- --skip-setup{% if hermes_skip_browser %} --skip-browser{% endif %}
|
||||
args:
|
||||
executable: /bin/bash
|
||||
become: true
|
||||
become_user: "{{ hermes_user }}"
|
||||
environment:
|
||||
HOME: "{{ hermes_home }}"
|
||||
PATH: "{{ hermes_home }}/.local/bin:/usr/local/bin:/usr/bin:/bin"
|
||||
when: not hermes_binary.stat.exists
|
||||
changed_when: true
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 5. Symlink hermes binary into system PATH
|
||||
# Per docs: service accounts often lack ~/.local/bin in PATH; symlink
|
||||
# into /usr/local/bin so hermes is always accessible.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
- name: Symlink hermes binary to system PATH
|
||||
ansible.builtin.file:
|
||||
src: "{{ hermes_home }}/.hermes/hermes-agent/venv/bin/hermes"
|
||||
dest: "{{ hermes_bin_symlink }}"
|
||||
state: link
|
||||
force: true
|
||||
become: true
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 6. Install systemd service unit
|
||||
# NOTE: The service starts in 'gateway' mode for persistent operation.
|
||||
# You must run 'sudo -u hermes hermes setup' interactively on first boot
|
||||
# to configure your LLM provider and any messaging gateways before
|
||||
# enabling the service.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
- name: Deploy hermes systemd service unit
|
||||
ansible.builtin.template:
|
||||
src: hermes.service.j2
|
||||
dest: /etc/systemd/system/{{ hermes_service_name }}.service
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
become: true
|
||||
notify:
|
||||
- reload systemd
|
||||
|
||||
- name: Flush handlers to reload systemd now
|
||||
ansible.builtin.meta: flush_handlers
|
||||
|
||||
- name: Enable hermes service (but do not start — config required first)
|
||||
ansible.builtin.systemd:
|
||||
name: "{{ hermes_service_name }}"
|
||||
enabled: true
|
||||
state: stopped
|
||||
become: true
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 7. Install systemd service unit for the Hermes Web UI (dashboard)
|
||||
# Separate unit from the gateway so the UI can be restarted / disabled
|
||||
# independently. Fronted upstream by Traefik (jarvis service) on
|
||||
# lightning-lane, so binding 0.0.0.0 with --insecure is intentional.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
- name: Deploy hermes-dashboard systemd service unit
|
||||
ansible.builtin.template:
|
||||
src: hermes-dashboard.service.j2
|
||||
dest: /etc/systemd/system/{{ hermes_dashboard_service_name }}.service
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
become: true
|
||||
register: hermes_dashboard_unit
|
||||
when: hermes_dashboard_enabled
|
||||
|
||||
- name: Reload systemd to pick up hermes-dashboard unit changes
|
||||
ansible.builtin.systemd:
|
||||
daemon_reload: true
|
||||
become: true
|
||||
when:
|
||||
- hermes_dashboard_enabled
|
||||
- hermes_dashboard_unit is changed
|
||||
|
||||
- name: Enable and start hermes-dashboard service
|
||||
ansible.builtin.systemd:
|
||||
name: "{{ hermes_dashboard_service_name }}"
|
||||
enabled: true
|
||||
state: started
|
||||
become: true
|
||||
when: hermes_dashboard_enabled
|
||||
|
||||
- name: Restart hermes-dashboard on unit change
|
||||
ansible.builtin.systemd:
|
||||
name: "{{ hermes_dashboard_service_name }}"
|
||||
state: restarted
|
||||
become: true
|
||||
when:
|
||||
- hermes_dashboard_enabled
|
||||
- hermes_dashboard_unit is changed
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user