feat(llm): Flotten-Katalog + modell-bewusstes LLM-Routing (Stage C)

ARIA nutzt lokale LLMs auf mehreren ai-boxen; jede Box (llama-swap) kann
mehrere Modelle fahren. Auswahl nach MODELL, Box wird automatisch gewaehlt.

- xtts/llm-adapter/adapter.py: fragt beim Connect llama-swap GET /v1/models ab
  und meldet die Modell-Liste in worker_hello (models:[...]), Fallback [LLM_MODEL].
- bridge/aria_bridge.py: Worker-Registry speichert models[]; _pick_worker(service,
  model=) beruecksichtigt nur Boxen, die das Modell fahren koennen (nachsichtig:
  keine → None → Broadcast/Claude-Fallback); Round-Robin je service+model
  verteilt mehrere Projekte auf mehrere Boxen; _local_llm reicht das Modell durch;
  _worker_list traegt models[].
- diagnostic/server.js: workers-Map + workerList um models[] erweitert.
- diagnostic/index.html: Compute-Flotte zeigt bei llm die Modell-Liste; das
  "Lokales Modell"-Dropdown wird LIVE aus den angemeldeten Boxen gebaut
  (Vereinigung + kuratierte Namen aus local_models.json, "· N Box(en)"/"offline"),
  darunter eine kompakte LLM-Box-Liste (Node→Modelle→Health). Speisung aus dem
  vorhandenen worker_update-Broadcast.

Kein Brain-/App-Eingriff. Routing greift nur bei aktivem Lokal-Schalter.
Deploy: diagnostic + bridge + llm-Boxen neu bauen.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
2026-09-19 02:49:30 +02:00
co-authored by Claude Opus 4.8
parent 99e4b63a06
commit 39426dc70d
4 changed files with 114 additions and 26 deletions
+66 -14
View File
@@ -602,9 +602,13 @@
<div style="font-size:10px;color:#FFD60A;margin:0 0 4px 0;line-height:1.5;">
Beim ersten Wechsel zu einem Modell lädt die AI-Box das GGUF (mehrere GB) —
die <strong>erste Antwort dauert dann länger</strong>, danach ist es gecacht.
Liste kommt aus <code>/shared/config/local_models.json</code> (Keys = xtts/llama-swap/config.yaml).
Verfügbare Modelle kommen <strong>live</strong> von den angemeldeten LLM-Boxen;
Namen aus <code>/shared/config/local_models.json</code>.
</div>
<!-- LLM-Flotte: welche Box faehrt welche Modelle (live via RVS) -->
<div id="local-llm-fleet" style="font-size:11px;color:#8888AA;margin:6px 0 0 0;"></div>
<div id="local-llm-status" style="font-size:11px;color:#6a6a88;margin-top:8px;padding-top:8px;border-top:1px solid #2a2a3a;min-height:14px;"></div>
</div>
</div>
@@ -2006,7 +2010,7 @@
}
if (msg.type === 'sat_update') { satellites = msg.satellites || []; renderSatellites(); return; }
if (msg.type === 'worker_update') { workers = msg.workers || []; renderWorkers(); return; }
if (msg.type === 'worker_update') { workers = msg.workers || []; renderWorkers(); if (typeof refreshLocalLlmModelChoices === 'function') refreshLocalLlmModelChoices(); return; }
if (msg.type === 'sat_devices') {
if (msg.satellite) { satDevices[msg.satellite] = { devices: msg.devices || [], location: msg.location, ts: Date.now() }; }
satScanning = null;
@@ -4265,10 +4269,13 @@
const meta = WORKER_SVC_META[w.service] || { icon: '⚙️', label: w.service };
const dot = !w.online ? '#666' : (w.busy ? '#FFB020' : '#3FFF3F');
const stat = !w.online ? 'offline' : (w.busy ? 'beschaeftigt' : 'frei');
// llm-Boxen koennen mehrere Modelle fahren (llama-swap) → Liste zeigen.
const modelText = (w.service === 'llm' && Array.isArray(w.models) && w.models.length)
? w.models.join(', ') : (w.model || '');
return '<div style="display:flex;align-items:center;gap:8px;padding:4px 0;">' +
'<span style="width:8px;height:8px;border-radius:50%;background:' + dot + ';display:inline-block;"></span>' +
'<span>' + meta.icon + ' <b>' + escapeHtml(meta.label) + '</b></span>' +
'<span style="color:#8888AA;">' + escapeHtml(w.model || '') + '</span>' +
'<span style="color:#8888AA;">' + escapeHtml(modelText) + '</span>' +
(w.gpus ? '<span style="color:#8888AA;">GPU ' + escapeHtml(w.gpus) + '</span>' : '') +
'<span style="margin-left:auto;color:' + dot + ';">' + stat + '</span>' +
'</div>';
@@ -6825,18 +6832,66 @@
el.textContent = m && m.description ? m.description : '';
}
async function loadLocalModelList() {
// Kuratierte Namen/Beschreibungen laden; die tatsaechliche Verfuegbarkeit
// kommt live aus der Flotte (refreshLocalLlmModelChoices).
try {
const r = await fetch('/api/local-models-list');
const j = await r.json();
_localModelsCache = (j && j.models) || [];
} catch (e) { _localModelsCache = []; }
const sel = document.getElementById('local-llm-model');
if (sel) {
sel.innerHTML = _localModelsCache.map(m =>
`<option value="${m.id}">${m.display_name || m.id}</option>`).join('') || '<option value="">(keine)</option>';
if (_localModelsCache.some(m => m.id === _currentLocalModel)) sel.value = _currentLocalModel;
updateLocalModelDesc();
refreshLocalLlmModelChoices();
}
// Zaehlt pro Modell die online LLM-Boxen, die es fahren koennen.
function fleetLlmModelCounts() {
const counts = {};
for (const w of workers) {
if (w.service !== 'llm' || !w.online) continue;
const list = (Array.isArray(w.models) && w.models.length) ? w.models : (w.model ? [w.model] : []);
for (const id of list) counts[id] = (counts[id] || 0) + 1;
}
return counts;
}
// Baut das Modell-Dropdown aus der Vereinigung von live-Flotte + kuratierter
// Liste; zeigt pro Modell "· N Box(en)" bzw. "· offline". Auswahl bleibt
// erhalten (auch wenn das gewaehlte Modell gerade keine online Box hat).
function refreshLocalLlmModelChoices() {
const sel = document.getElementById('local-llm-model');
if (!sel) return;
const counts = fleetLlmModelCounts();
const nameOf = id => { const m = _localModelsCache.find(x => x.id === id); return (m && m.display_name) || id; };
const ids = new Set();
Object.keys(counts).forEach(id => ids.add(id));
_localModelsCache.forEach(m => ids.add(m.id));
if (_currentLocalModel) ids.add(_currentLocalModel);
const order = Array.from(ids).sort();
sel.innerHTML = order.map(id => {
const n = counts[id] || 0;
const avail = n > 0 ? ` · ${n} Box${n > 1 ? 'en' : ''}` : ' · offline';
return `<option value="${id}">${escapeHtml(nameOf(id))}${avail}</option>`;
}).join('') || '<option value="">(keine)</option>';
if (order.includes(_currentLocalModel)) sel.value = _currentLocalModel;
updateLocalModelDesc();
renderLocalLlmFleet();
}
// Kompakte Box-Liste unter dem Dropdown: Node → Modelle → Health.
function renderLocalLlmFleet() {
const box = document.getElementById('local-llm-fleet');
if (!box) return;
const llm = workers.filter(w => w.service === 'llm');
if (!llm.length) { box.innerHTML = '<span style="color:#6a6a88;">Keine LLM-Box angemeldet. Starte eine Box mit <code>COMPOSE_PROFILES=llm</code>.</span>'; return; }
const byNode = {};
for (const w of llm) { (byNode[w.node || '?'] = byNode[w.node || '?'] || []).push(w); }
box.innerHTML = '<div style="color:#AAB;margin-bottom:2px;">LLM-Boxen:</div>' + Object.keys(byNode).sort().map(node => {
return byNode[node].map(w => {
const dot = !w.online ? '#666' : (w.busy ? '#FFB020' : '#3FFF3F');
const models = (Array.isArray(w.models) && w.models.length) ? w.models.join(', ') : (w.model || '—');
return '<div style="display:flex;align-items:center;gap:6px;padding:2px 0;">' +
'<span style="width:7px;height:7px;border-radius:50%;background:' + dot + ';display:inline-block;"></span>' +
'<span>🖥️ ' + escapeHtml(node) + '</span>' +
'<span style="color:#8888AA;">' + escapeHtml(models) + '</span>' +
'</div>';
}).join('');
}).join('');
}
async function loadLocalLlmConfig() {
try {
@@ -6849,11 +6904,8 @@
if (ol) ol.checked = !!c.localOnly;
if (tv) tv.value = (c.toolVariant === 'full') ? 'full' : 'slim';
_currentLocalModel = c.localLlmModel || 'qwen3-8b';
const sel = document.getElementById('local-llm-model');
if (sel && _localModelsCache.some(m => m.id === _currentLocalModel)) {
sel.value = _currentLocalModel;
updateLocalModelDesc();
}
// Dropdown neu aufbauen (Flotte+kuratiert) und die gespeicherte Auswahl setzen.
refreshLocalLlmModelChoices();
setLocalLlmStatus(c);
} catch (e) { /* still */ }
}