Compare commits
118
Commits
a49c022308
..
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
0612395bbf | ||
|
|
16b6fbeaf6 | ||
|
|
ea0c7e21f8 | ||
|
|
f9cc10433a | ||
|
|
8b1555f926 | ||
|
|
02303c35ee | ||
|
|
c2472a7958 | ||
|
|
5b952812ee | ||
|
|
512e24bb6f | ||
|
|
af1a3bf386 | ||
|
|
76e8b8e04c | ||
|
|
6918ddca61 | ||
|
|
318bb47590 | ||
|
|
348e9783aa | ||
|
|
1c7e07cb67 | ||
|
|
778d75ae61 | ||
|
|
f3e3aedc6b | ||
|
|
2d2c182a0c | ||
|
|
831fbfab58 | ||
|
|
a0dedd8c97 | ||
|
|
8633840f94 | ||
|
|
45167b7bb5 | ||
|
|
66781d8d90 | ||
|
|
5c25d6abeb | ||
|
|
05c6c7687a | ||
|
|
0f122a1ad7 | ||
|
|
a773832bca | ||
|
|
17a60e5d3e | ||
|
|
182351e022 | ||
|
|
ec78eb8efe | ||
|
|
f2ead1242f | ||
|
|
e75f1eeb6a | ||
|
|
2bd7c747d9 | ||
|
|
d0694cb637 | ||
|
|
8f313eb68b | ||
|
|
954ad9ade8 | ||
|
|
75fa574a85 | ||
|
|
a3d7933949 | ||
|
|
626281dead | ||
|
|
49ea172617 | ||
|
|
a3650bb0e4 | ||
|
|
514ba44e51 | ||
|
|
617dddf3d8 | ||
|
|
88faf57253 | ||
|
|
41714f7cf1 | ||
|
|
93401d42d1 | ||
|
|
c4e658446d | ||
|
|
3693157210 | ||
|
|
64670cdd12 | ||
|
|
c82616ebbd | ||
|
|
b226e1da11 | ||
|
|
0b7ed241b4 | ||
|
|
a2b7e3a48d | ||
|
|
7b72149671 | ||
|
|
e9439dbccb | ||
|
|
219de091d2 | ||
|
|
5992a7e441 | ||
|
|
07ccf05429 | ||
|
|
4ab0e68245 | ||
|
|
9636a702d3 | ||
|
|
5f09e2bca3 | ||
|
|
ebe0e8065f | ||
|
|
9d2c07d8d1 | ||
|
|
9aae5af6a9 | ||
|
|
a8ff73f93d | ||
|
|
0e9adeee5c | ||
|
|
6addb2f8fe | ||
|
|
9bdfb7193e | ||
|
|
9e78d75149 | ||
|
|
517c993ac8 | ||
|
|
c1bd13687d | ||
|
|
d9bb7239c6 | ||
|
|
0265aabb5e | ||
|
|
17bc50b847 | ||
|
|
1f2be4299d | ||
|
|
0ca8a82013 | ||
|
|
7bc3f827d0 | ||
|
|
e7da9cf9c4 | ||
|
|
353fd98d3f | ||
|
|
0350dd33c8 | ||
|
|
7b956c6606 | ||
|
|
36f04f83ff | ||
|
|
01df26e6df | ||
|
|
f226c91973 | ||
|
|
3324d39d50 | ||
|
|
fff2e7df34 | ||
|
|
0aac114142 | ||
|
|
ba60f793fb | ||
|
|
a7c2f07361 | ||
|
|
ccf6dd84fb | ||
|
|
75675ed3aa | ||
|
|
73fee27e90 | ||
|
|
03021a6787 | ||
|
|
b6a5d7029f | ||
|
|
3a8202d2de | ||
|
|
7bbb75481c | ||
|
|
ee6c4f34db | ||
|
|
75daadf72d | ||
|
|
37aaa90239 | ||
|
|
03e6d784b6 | ||
|
|
762f1a2dd9 | ||
|
|
49f6b26ab8 | ||
|
|
c340b9d683 | ||
|
|
a0429fc91e | ||
|
|
174d6d643d | ||
|
|
fd6ba73f59 | ||
|
|
f5b22253b2 | ||
|
|
761f4c8903 | ||
|
|
091a1b7755 | ||
|
|
e2b1eced3c | ||
|
|
fe804fa40e | ||
|
|
18eb94e942 | ||
|
|
74c7ea0a2d | ||
|
|
648e3b04fd | ||
|
|
2ad8c2f245 | ||
|
|
85d190e98c | ||
|
|
70705269fd | ||
|
|
2aef0347ae |
+1
-1
@@ -235,7 +235,7 @@ Nur **App neu bauen** (kein Backend). APK 0.2.2.0.
|
|||||||
|
|
||||||
## [0.2.0.4 – 0.2.0.5] — 2026-07-11 — Plan B: Lokales LLM („Gemini-Feeling")
|
## [0.2.0.4 – 0.2.0.5] — 2026-07-11 — Plan B: Lokales LLM („Gemini-Feeling")
|
||||||
|
|
||||||
Ein kleines, schnelles Modell (**Qwen3 8B** via llama.cpp/llama-swap auf der Gamebox-GPU) übernimmt einfache Turns in **<1 s**; alles Schwere/Technische/Werkzeug-artige reicht ein Router automatisch an **Claude** weiter. Ziel: schnelle Antworten ohne die Claude-Max-Subscription aufzugeben.
|
Ein kleines, schnelles Modell (**Qwen3 8B** via llama.cpp/llama-swap auf der AI-Box-GPU) übernimmt einfache Turns in **<1 s**; alles Schwere/Technische/Werkzeug-artige reicht ein Router automatisch an **Claude** weiter. Ziel: schnelle Antworten ohne die Claude-Max-Subscription aufzugeben.
|
||||||
|
|
||||||
### Hinzugefügt
|
### Hinzugefügt
|
||||||
|
|
||||||
|
|||||||
@@ -35,18 +35,17 @@ ARIA hat zwei Rollen:
|
|||||||
│ WebSocket Tunnel │ WebSocket Tunnel
|
│ WebSocket Tunnel │ WebSocket Tunnel
|
||||||
▼ ▼
|
▼ ▼
|
||||||
┌─────────────────────────────────┐
|
┌─────────────────────────────────┐
|
||||||
│ Gamebox (Windows + WSL2) │
|
│ Compute-Node(s) (NVIDIA GPU) │
|
||||||
│ RTX 3060, Docker Desktop │
|
│ beliebig viele, je per .env │
|
||||||
|
│ konfiguriert (COMPOSE_PROFILES) │
|
||||||
│ ┌──────────────────────────┐ │
|
│ ┌──────────────────────────┐ │
|
||||||
│ │ aria-f5tts-bridge │ │
|
│ │ aria-voxtral-bridge │ │ Profil: voxtral (Default-STT)
|
||||||
│ │ F5-TTS Voice Cloning │ │
|
│ │ aria-f5tts-bridge │ │ Profil: f5tts (TTS)
|
||||||
│ │ PCM-Streaming an die App │ │
|
│ │ aria-llm-adapter+swap │ │ Profil: llm (lokales LLM)
|
||||||
│ ├──────────────────────────┤ │
|
│ │ aria-whisper-bridge │ │ Profil: whisper (STT-Fallback)
|
||||||
│ │ aria-whisper-bridge │ │
|
|
||||||
│ │ Faster-Whisper CUDA │ │
|
|
||||||
│ │ STT in fast-Echtzeit │ │
|
|
||||||
│ └──────────────────────────┘ │
|
│ └──────────────────────────┘ │
|
||||||
│ Beide teilen ./voices Volume │
|
│ Aufteilbar: 1 Node pro Dienst │
|
||||||
|
│ ODER All-in-One. GPU per *_GPU. │
|
||||||
│ xtts/docker-compose.yml │
|
│ xtts/docker-compose.yml │
|
||||||
└─────────────────────────────────┘
|
└─────────────────────────────────┘
|
||||||
┌─────────────────────────────────────────────────────────┐
|
┌─────────────────────────────────────────────────────────┐
|
||||||
@@ -95,12 +94,16 @@ ARIA hat zwei Rollen:
|
|||||||
|-----|----|-----|
|
|-----|----|-----|
|
||||||
| RVS | Rechenzentrum | `cd rvs && docker compose up -d` |
|
| RVS | Rechenzentrum | `cd rvs && docker compose up -d` |
|
||||||
| ARIA Brain/Bridge/Diagnostic | Debian 13 VM | `./init.sh && ./aria-setup.sh && docker compose up -d` |
|
| ARIA Brain/Bridge/Diagnostic | Debian 13 VM | `./init.sh && ./aria-setup.sh && docker compose up -d` |
|
||||||
| Gamebox-Stack (F5-TTS + Whisper) | Gamebox (GPU) | `cd xtts && docker compose up -d` |
|
| Compute-Node(s) (STT/TTS/LLM) | 1..n GPU-Rechner | `cd xtts && cp .env.example .env && docker compose up -d` |
|
||||||
| Satellit(en) 🛰️ (optional) | Fremdes Netz (Büro …) | `cd satellite && cp .env.example .env && docker compose up -d --build` |
|
| Satellit(en) 🛰️ (optional) | Fremdes Netz (Büro …) | `cd satellite && cp .env.example .env && docker compose up -d --build` |
|
||||||
| Android App | Stefans Handy | APK installieren (Auto-Update via RVS) |
|
| Android App | Stefans Handy | APK installieren (Auto-Update via RVS) |
|
||||||
|
|
||||||
> Der Gamebox-Stack ist optional: ohne ihn faellt STT auf lokales Whisper (CPU,
|
> Compute-Nodes sind optional: ohne sie faellt STT auf lokales Whisper (CPU,
|
||||||
> langsamer) zurueck; TTS bleibt aus (ARIA antwortet dann nur als Text).
|
> langsamer) zurueck; TTS bleibt aus (ARIA antwortet dann nur als Text).
|
||||||
|
> Jeder Node startet per `COMPOSE_PROFILES` in seiner `.env` nur die Dienste,
|
||||||
|
> die er anbieten soll (`voxtral`/`f5tts`/`llm`/`whisper`) — so laesst sich der
|
||||||
|
> GPU-Stack auf mehrere Maschinen verteilen (STT-Box, TTS-Box, LLM-Box) oder
|
||||||
|
> als All-in-One auf einer Kiste fahren (`voxtral,f5tts,llm`).
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
@@ -135,8 +138,8 @@ RVS_PORT=443
|
|||||||
RVS_TLS=true
|
RVS_TLS=true
|
||||||
RVS_TLS_FALLBACK=true
|
RVS_TLS_FALLBACK=true
|
||||||
|
|
||||||
# Pairing-Token: Verbindet App, Bridge, Diagnostic und Gamebox im gleichen RVS-Room
|
# Pairing-Token: Verbindet App, Bridge, Diagnostic und Compute-Nodes im gleichen RVS-Room
|
||||||
# MUSS auf allen Geraeten identisch sein (ARIA-VM, Gaming-PC, App)
|
# MUSS auf allen Geraeten identisch sein (ARIA-VM, Compute-Nodes, App)
|
||||||
RVS_TOKEN= # ./generate-token.sh
|
RVS_TOKEN= # ./generate-token.sh
|
||||||
```
|
```
|
||||||
|
|
||||||
@@ -304,28 +307,28 @@ Danach wird der Proxy gepatcht:
|
|||||||
|
|
||||||
## Voice Bridge
|
## Voice Bridge
|
||||||
|
|
||||||
Die Bridge verbindet die Android App mit ARIA und orchestriert die GPU-Services
|
Die Bridge verbindet die Android App mit ARIA und orchestriert die GPU-Dienste
|
||||||
auf der Gamebox.
|
auf den Compute-Nodes.
|
||||||
|
|
||||||
**Nachrichtenfluss:**
|
**Nachrichtenfluss:**
|
||||||
```
|
```
|
||||||
Text: App → RVS → Bridge → aria-brain (HTTP)
|
Text: App → RVS → Bridge → aria-brain (HTTP)
|
||||||
Audio: App → RVS → Bridge → stt_request (RVS) → whisper-bridge (Gamebox)
|
Audio: App → RVS → STT-Node (voxtral/whisper) direkt (Streaming)
|
||||||
→ stt_response → Bridge → aria-brain
|
→ stt_endpoint/stt_stream_done → Bridge → aria-brain
|
||||||
Fallback bei Timeout: lokales faster-whisper (CPU)
|
Fallback bei Timeout: lokales faster-whisper (CPU)
|
||||||
Datei: App → RVS → Bridge → /shared/uploads/ → aria-brain (mit Pfad)
|
Datei: App → RVS → Bridge → /shared/uploads/ → aria-brain (mit Pfad)
|
||||||
|
|
||||||
aria-brain → Antwort → Bridge → RVS → App
|
aria-brain → Antwort → Bridge → RVS → App
|
||||||
→ xtts_request (RVS) → f5tts-bridge
|
→ xtts_request (RVS) → f5tts-Node
|
||||||
→ audio_pcm Stream → RVS → App AudioTrack
|
→ audio_pcm Stream → RVS → App AudioTrack
|
||||||
```
|
```
|
||||||
|
|
||||||
### Features
|
### Features
|
||||||
|
|
||||||
- **STT primaer remote**: aria-bridge sendet `stt_request` an die Gamebox-Whisper
|
- **STT primaer remote**: die App streamt Audio direkt an einen STT-Node
|
||||||
(faster-whisper CUDA, fast Echtzeit). 45s Timeout, dann Fallback auf lokales
|
(Voxtral-3B default, faster-whisper Fallback-Profil), fast Echtzeit. Timeout →
|
||||||
CPU-Whisper. Modell-Wahl in Diagnostic, Hot-Swap via config-Broadcast.
|
Fallback auf lokales CPU-Whisper. Modell-Wahl in Diagnostic, Hot-Swap via config.
|
||||||
- **TTS via F5-TTS**: aria-f5tts-bridge auf der Gamebox. Voice Cloning mit
|
- **TTS via F5-TTS**: aria-f5tts-bridge auf einem Compute-Node. Voice Cloning mit
|
||||||
Referenz-Audio + automatisch transkribiertem Referenz-Text.
|
Referenz-Audio + automatisch transkribiertem Referenz-Text.
|
||||||
- **Text-Cleanup**: `<voice>...</voice>` Tag bevorzugt; Markdown, Code,
|
- **Text-Cleanup**: `<voice>...</voice>` Tag bevorzugt; Markdown, Code,
|
||||||
Einheiten und URLs werden TTS-gerecht aufbereitet. Dezimalzahlen werden
|
Einheiten und URLs werden TTS-gerecht aufbereitet. Dezimalzahlen werden
|
||||||
@@ -486,7 +489,7 @@ Erreichbar unter `http://<VM-IP>:3001`. Teilt das Netzwerk mit der Bridge.
|
|||||||
- **Disk-Voll Banner** mit copy-baren Cleanup-Befehlen (safe + aggressiv)
|
- **Disk-Voll Banner** mit copy-baren Cleanup-Befehlen (safe + aggressiv)
|
||||||
- **Token/Call-Metrics**: pro Claude-Call ein Eintrag in `/data/metrics.jsonl` mit ts + Token-Schaetzung. Gehirn-Tab zeigt 1h/5h/24h/30d-Aggregat plus Progress-Bar gegen Plan-Limit (Pro / Max 5x / Max 20x / Custom). Warn-Schwelle 80%, kritisch 90%.
|
- **Token/Call-Metrics**: pro Claude-Call ein Eintrag in `/data/metrics.jsonl` mit ts + Token-Schaetzung. Gehirn-Tab zeigt 1h/5h/24h/30d-Aggregat plus Progress-Bar gegen Plan-Limit (Pro / Max 5x / Max 20x / Custom). Warn-Schwelle 80%, kritisch 90%.
|
||||||
- **Voice Cloning**: Audio-Samples hochladen, Whisper transkribiert den Ref-Text automatisch
|
- **Voice Cloning**: Audio-Samples hochladen, Whisper transkribiert den Ref-Text automatisch
|
||||||
- **Voice Export/Import**: einzelne Stimmen als `.tar.gz` zwischen Gameboxen mitnehmen
|
- **Voice Export/Import**: einzelne Stimmen als `.tar.gz` zwischen Compute-Nodes mitnehmen
|
||||||
- **Settings Export/Import**: `voice_config.json` + `highlight_triggers.json` als JSON-Bundle
|
- **Settings Export/Import**: `voice_config.json` + `highlight_triggers.json` als JSON-Bundle
|
||||||
- **Claude Login**: Browser-Terminal zum Einloggen in den Proxy
|
- **Claude Login**: Browser-Terminal zum Einloggen in den Proxy
|
||||||
- **ARIA Live**: read-only Mirror der Claude-Code-Session — alle Tool-Calls + Inputs + Outputs live in einer Monospace-Liste, farbcodiert. **Persistenz**: jeder `agent_stream`-Event wird parallel in `/shared/logs/agent_stream.jsonl` (soft-cap 50 MB) geschrieben, Live-View laedt beim Tab-Oeffnen / Page-Reload die letzten 200 Eintraege — Browser-Standby wirft nichts mehr weg. Plus ⛔ **Not-Aus**-Button der per RVS einen `cancel_request` mit `hard:true` ausloest → aria-bridge ruft den proxy-internen `/cancel-all` Side-Channel → alle Claude-Subprocesses werden sofort gekillt
|
- **ARIA Live**: read-only Mirror der Claude-Code-Session — alle Tool-Calls + Inputs + Outputs live in einer Monospace-Liste, farbcodiert. **Persistenz**: jeder `agent_stream`-Event wird parallel in `/shared/logs/agent_stream.jsonl` (soft-cap 50 MB) geschrieben, Live-View laedt beim Tab-Oeffnen / Page-Reload die letzten 200 Eintraege — Browser-Standby wirft nichts mehr weg. Plus ⛔ **Not-Aus**-Button der per RVS einen `cancel_request` mit `hard:true` ausloest → aria-bridge ruft den proxy-internen `/cancel-all` Side-Channel → alle Claude-Subprocesses werden sofort gekillt
|
||||||
@@ -515,7 +518,7 @@ Erreichbar unter `http://<VM-IP>:3001`. Teilt das Netzwerk mit der Bridge.
|
|||||||
- **Wake-Word waehrend TTS**: Du kannst "Computer" sagen waehrend ARIA noch redet — AcousticEchoCanceler verhindert dass ARIAs eigene Stimme das Wake-Word triggert
|
- **Wake-Word waehrend TTS**: Du kannst "Computer" sagen waehrend ARIA noch redet — AcousticEchoCanceler verhindert dass ARIAs eigene Stimme das Wake-Word triggert
|
||||||
- **Anruf-Pause + Auto-Resume**: TTS verstummt bei klassischem Anruf oder VoIP-Call (WhatsApp/Signal/Discord). Nach dem Auflegen geht ARIA von der **genauen Stelle** weiter wo sie unterbrochen wurde — die App misst die Position vom Wiedergabe-Anfang und nutzt den WAV-Cache der Antwort
|
- **Anruf-Pause + Auto-Resume**: TTS verstummt bei klassischem Anruf oder VoIP-Call (WhatsApp/Signal/Discord). Nach dem Auflegen geht ARIA von der **genauen Stelle** weiter wo sie unterbrochen wurde — die App misst die Position vom Wiedergabe-Anfang und nutzt den WAV-Cache der Antwort
|
||||||
- **Speech Gate**: Aufnahme wird verworfen wenn keine Sprache erkannt
|
- **Speech Gate**: Aufnahme wird verworfen wenn keine Sprache erkannt
|
||||||
- **STT (Speech-to-Text)**: 16kHz mono → Bridge → Gamebox-Whisper (CUDA) → Text im Chat. Fast in Echtzeit.
|
- **STT (Speech-to-Text)**: 16kHz mono → STT-Node (Voxtral-3B, CUDA) → Text im Chat. Fast in Echtzeit.
|
||||||
- **"ARIA denkt..." Indicator**: Zeigt live den Status vom Core (Denken, Tool, Schreiben) + Abbrechen-Button
|
- **"ARIA denkt..." Indicator**: Zeigt live den Status vom Core (Denken, Tool, Schreiben) + Abbrechen-Button
|
||||||
- **TTS-Wiedergabe**: F5-TTS PCM-Streaming direkt in AudioTrack mit konfigurierbarem Pre-Roll-Buffer (1.0–6.0s, Default 3.5s) gegen Gaps bei Render-Pausen
|
- **TTS-Wiedergabe**: F5-TTS PCM-Streaming direkt in AudioTrack mit konfigurierbarem Pre-Roll-Buffer (1.0–6.0s, Default 3.5s) gegen Gaps bei Render-Pausen
|
||||||
- **Audio-Pause**: Andere Apps (Spotify, YouTube etc.) pausieren komplett waehrend ARIA spricht und kommen erst wieder nach echtem Wiedergabe-Ende
|
- **Audio-Pause**: Andere Apps (Spotify, YouTube etc.) pausieren komplett waehrend ARIA spricht und kommen erst wieder nach echtem Wiedergabe-Ende
|
||||||
@@ -655,7 +658,7 @@ Der Update-Flow:
|
|||||||
App (Mikrofon) → AAC/MP4 Aufnahme → Base64 → RVS → Bridge
|
App (Mikrofon) → AAC/MP4 Aufnahme → Base64 → RVS → Bridge
|
||||||
Bridge: FFmpeg (16kHz PCM) → Whisper STT → Text → aria-brain
|
Bridge: FFmpeg (16kHz PCM) → Whisper STT → Text → aria-brain
|
||||||
Bridge: STT-Ergebnis → RVS → App (Placeholder wird durch transkribierten Text ersetzt)
|
Bridge: STT-Ergebnis → RVS → App (Placeholder wird durch transkribierten Text ersetzt)
|
||||||
aria-brain → Antwort → Bridge → F5-TTS (Gaming-PC) → PCM-Stream → RVS → App
|
aria-brain → Antwort → Bridge → F5-TTS (Compute-Node) → PCM-Stream → RVS → App
|
||||||
App: AudioTrack MODE_STREAM (nahtlos), Cache als WAV pro Message
|
App: AudioTrack MODE_STREAM (nahtlos), Cache als WAV pro Message
|
||||||
```
|
```
|
||||||
|
|
||||||
@@ -797,36 +800,57 @@ cp ARIA-v0.0.3.0.apk ~/ARIA-AGENT/rvs/updates/
|
|||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
## Gamebox-Stack — F5-TTS + Whisper (GPU-Services)
|
## Compute-Nodes — STT / TTS / LLM (GPU-Dienste)
|
||||||
|
|
||||||
Laeuft auf einem separaten Rechner mit NVIDIA GPU (z.B. Gaming-PC mit RTX 3060).
|
Die GPU-Dienste laufen auf einem oder mehreren separaten Rechnern mit NVIDIA GPU.
|
||||||
Verbindet sich ueber RVS mit der ARIA-Infrastruktur — kein VPN noetig, funktioniert
|
Jeder **Compute-Node** verbindet sich ueber RVS mit der ARIA-Infrastruktur — kein
|
||||||
ueber verschiedene Netze hinweg.
|
VPN noetig, funktioniert ueber verschiedene Netze hinweg. Frueher war das *eine*
|
||||||
|
feste „AI-Box"; jetzt sind es beliebig viele Nodes, jeder per `.env` konfiguriert.
|
||||||
|
|
||||||
### Architektur
|
### Dienste & Profile
|
||||||
|
|
||||||
|
Jeder Dienst haengt an einem Compose-Profil. Ein Node startet ueber
|
||||||
|
`COMPOSE_PROFILES` (in seiner `.env`) nur die Profile, die er anbieten soll:
|
||||||
|
|
||||||
|
| Profil | Container | Rolle |
|
||||||
|
|-----------|------------------------------|-------|
|
||||||
|
| `voxtral` | aria-voxtral-bridge | Default-STT (Voxtral-Mini-3B, ~9 GB) |
|
||||||
|
| `whisper` | aria-whisper-bridge | STT-Fallback (faster-whisper CUDA) |
|
||||||
|
| `f5tts` | aria-f5tts-bridge | TTS (F5-TTS Voice Cloning) |
|
||||||
|
| `llm` | aria-llama-swap + llm-adapter| Lokales LLM (llama-swap, OpenAI-kompat.) |
|
||||||
|
|
||||||
|
### Architektur (Aufteilung auf mehrere Nodes)
|
||||||
|
|
||||||
```
|
```
|
||||||
Gamebox (Windows, RTX 3060, Docker Desktop + WSL2)
|
STT-Box COMPOSE_PROFILES=voxtral VOXTRAL_GPU=0
|
||||||
├── aria-f5tts-bridge F5-TTS Voice Cloning + RVS-Relay
|
TTS-Box COMPOSE_PROFILES=f5tts F5TTS_GPU=0
|
||||||
│ Hoert auf xtts_request, streamt audio_pcm
|
LLM-Box COMPOSE_PROFILES=llm LLM_GPU=0
|
||||||
├── aria-whisper-bridge faster-whisper auf CUDA (float16)
|
-- kleine Karte: Whisper (klein) statt Voxtral, neben F5-TTS --
|
||||||
│ Hoert auf stt_request, antwortet mit stt_response
|
Klein-Box COMPOSE_PROFILES=whisper,f5tts WHISPER_GPU=0 F5TTS_GPU=0
|
||||||
└── ./voices/ Geteilt zwischen beiden:
|
── oder All-in-One ──
|
||||||
{name}.wav — Referenz-Audio (~6-10s)
|
AI-Box COMPOSE_PROFILES=voxtral,f5tts,llm VOXTRAL_GPU=1 F5TTS_GPU=0 LLM_GPU=0
|
||||||
{name}.txt — Referenz-Text (auto via Whisper)
|
|
||||||
|
|
||||||
↕ RVS (Rechenzentrum, WebSocket Relay)
|
↕ RVS (Rechenzentrum, WebSocket Relay)
|
||||||
|
|
||||||
ARIA-VM
|
ARIA-VM
|
||||||
└── aria-bridge: STT primaer remote (45s Timeout, dann lokaler CPU-Fallback)
|
└── aria-bridge: orchestriert TTS/LLM (xtts_request/llm_request),
|
||||||
TTS via xtts_request → audio_pcm Stream
|
lauscht passiv auf den STT-Stream App↔STT-Node.
|
||||||
|
STT-Timeout → lokaler CPU-Whisper-Fallback.
|
||||||
```
|
```
|
||||||
|
|
||||||
### Voraussetzungen
|
> STT: pro Node **genau einen** — Voxtral-3B (~9 GB, beste Qualitaet) *oder*
|
||||||
|
> Whisper (klein, passt neben F5-TTS auf eine GPU mit wenig VRAM). Beide zusammen
|
||||||
|
> beantworten dieselbe Anfrage doppelt.
|
||||||
|
|
||||||
- Docker Desktop mit WSL2 (Windows) oder Docker mit NVIDIA Runtime (Linux)
|
Die STT-Node teilt sich das `./voices/`-Volume mit F5-TTS nur, wenn beide auf
|
||||||
- NVIDIA Container Toolkit
|
demselben Node laufen (Referenz-Text-Transkription beim Voice-Upload). Auf
|
||||||
- GPU mit mindestens 6GB VRAM (Whisper-large + F5-TTS gemeinsam)
|
getrennten Nodes transkribiert F5-TTS ueber den STT-Node via RVS.
|
||||||
|
|
||||||
|
### Voraussetzungen (pro Node)
|
||||||
|
|
||||||
|
- Docker + **NVIDIA Container Toolkit** (registriert die `nvidia`-Runtime — die
|
||||||
|
Compose nutzt `runtime: nvidia` + `NVIDIA_VISIBLE_DEVICES`).
|
||||||
|
- Genug VRAM fuer die gewaehlten Profile (Voxtral-3B ~9 GB, F5-TTS ~1 GB, LLM je Modell).
|
||||||
- **Gleicher RVS_TOKEN wie auf der ARIA-VM!**
|
- **Gleicher RVS_TOKEN wie auf der ARIA-VM!**
|
||||||
|
|
||||||
### Setup
|
### Setup
|
||||||
@@ -834,13 +858,17 @@ ARIA-VM
|
|||||||
```bash
|
```bash
|
||||||
cd xtts
|
cd xtts
|
||||||
cp .env.example .env
|
cp .env.example .env
|
||||||
# .env mit RVS-Verbindungsdaten fuellen (gleicher Token wie ARIA-VM!)
|
# .env anpassen:
|
||||||
|
# COMPOSE_PROFILES → welche Dienste dieser Node fahren soll
|
||||||
|
# NODE_NAME → Name des Rechners (erscheint in Diagnostic + Logs)
|
||||||
|
# *_GPU → welche Grafikkarte pro Dienst (NVIDIA_VISIBLE_DEVICES)
|
||||||
|
# RVS_* → gleiche Verbindungsdaten wie die ARIA-VM
|
||||||
docker compose up -d
|
docker compose up -d
|
||||||
# Erster Start laedt die Modelle (Whisper ~1-3GB je nach Groesse, F5-TTS ~1GB)
|
# Erster Start laedt die Modelle der aktiven Profile (Voxtral ~9GB, F5-TTS ~1GB)
|
||||||
```
|
```
|
||||||
|
|
||||||
Die Modelle werden in den Volumes `f5tts-models` und `whisper-models` gecacht
|
Die Modelle liegen im Bind-Mount `./hf-cache/` (bzw. `./models/` fuer LLM-GGUFs)
|
||||||
und muessen nur einmal geladen werden.
|
und muessen pro Node nur einmal geladen werden.
|
||||||
|
|
||||||
### Features
|
### Features
|
||||||
|
|
||||||
@@ -862,7 +890,7 @@ In der Diagnostic unter Einstellungen → Sprachausgabe:
|
|||||||
- **TTS aktiv**: Global An/Aus
|
- **TTS aktiv**: Global An/Aus
|
||||||
- **F5-TTS Stimme**: Default oder gecloned (Maia etc.)
|
- **F5-TTS Stimme**: Default oder gecloned (Maia etc.)
|
||||||
|
|
||||||
> F5-TTS ist die einzige Engine — wenn die Gamebox offline ist, bleibt ARIA stumm.
|
> F5-TTS ist die einzige Engine — wenn kein f5tts-Node online ist, bleibt ARIA stumm.
|
||||||
> Chat-Antworten kommen weiter an (nur kein Audio).
|
> Chat-Antworten kommen weiter an (nur kein Audio).
|
||||||
|
|
||||||
### Stimme klonen
|
### Stimme klonen
|
||||||
@@ -1024,7 +1052,7 @@ docker exec aria-brain curl localhost:8080/memory/stats
|
|||||||
- [x] Pre-Roll-Buffer einstellbar in App-Settings
|
- [x] Pre-Roll-Buffer einstellbar in App-Settings
|
||||||
- [x] Decimal-zu-Worte fuer TTS + generisches Acronym-Buchstabieren
|
- [x] Decimal-zu-Worte fuer TTS + generisches Acronym-Buchstabieren
|
||||||
- [x] voice_preload/voice_ready: visueller Status-Indikator beim Stimmen-Wechsel
|
- [x] voice_preload/voice_ready: visueller Status-Indikator beim Stimmen-Wechsel
|
||||||
- [x] Whisper STT auf die Gamebox ausgelagert (CUDA float16, fast Echtzeit)
|
- [x] Whisper STT auf die AI-Box ausgelagert (CUDA float16, fast Echtzeit)
|
||||||
- [x] **F5-TTS ersetzt XTTS** — bessere Voice-Cloning-Qualitaet, Whisper-auto-transkribierter Referenz-Text
|
- [x] **F5-TTS ersetzt XTTS** — bessere Voice-Cloning-Qualitaet, Whisper-auto-transkribierter Referenz-Text
|
||||||
- [x] Audio-Pause statt Ducking (TRANSIENT statt MAY_DUCK) + release-Timing fix
|
- [x] Audio-Pause statt Ducking (TRANSIENT statt MAY_DUCK) + release-Timing fix
|
||||||
- [x] VAD-Stille-Toleranz einstellbar (1-8s) + adaptive Mikro-Baseline + Max-Aufnahme einstellbar (1-30 min)
|
- [x] VAD-Stille-Toleranz einstellbar (1-8s) + adaptive Mikro-Baseline + Max-Aufnahme einstellbar (1-30 min)
|
||||||
@@ -1079,7 +1107,7 @@ docker exec aria-brain curl localhost:8080/memory/stats
|
|||||||
- [x] App: Chat-Suche mit Next/Prev Navigation statt Filter
|
- [x] App: Chat-Suche mit Next/Prev Navigation statt Filter
|
||||||
- [x] Token/Call-Metrics + Subscription-Quota-Tracking (Pro / Max 5x / Max 20x / Custom)
|
- [x] Token/Call-Metrics + Subscription-Quota-Tracking (Pro / Max 5x / Max 20x / Custom)
|
||||||
- [x] Datei-Manager Multi-Select: Bulk-Download als ZIP + Bulk-Delete (Diagnostic + App)
|
- [x] Datei-Manager Multi-Select: Bulk-Download als ZIP + Bulk-Delete (Diagnostic + App)
|
||||||
- [x] **FLUX.1 Bildgenerierung**: eigener `flux-bridge`-Container auf der Gamebox (analog xtts/whisper) mit Hot-Swap zwischen FLUX.1-dev (Quali) und FLUX.1-schnell (Tempo). Default-Modell + Raw-/Switch-Keywords + HuggingFace-Token in Diagnostic-UI verwaltet, automatischer Pipeline-Reload bei Modell-Wechsel. ARIA bekommt `flux_generate`-Tool, Output landet als `/shared/uploads/aria_generated_<ts>.png` und wird via `[FILE: ...]`-Marker als Anhang-Bubble in App + Diagnostic gerendert. Download-Status (mehrere GB) sichtbar als 🎉-Toast wenn fertig
|
- [x] **FLUX.1 Bildgenerierung**: eigener `flux-bridge`-Container auf der AI-Box (analog xtts/whisper) mit Hot-Swap zwischen FLUX.1-dev (Quali) und FLUX.1-schnell (Tempo). Default-Modell + Raw-/Switch-Keywords + HuggingFace-Token in Diagnostic-UI verwaltet, automatischer Pipeline-Reload bei Modell-Wechsel. ARIA bekommt `flux_generate`-Tool, Output landet als `/shared/uploads/aria_generated_<ts>.png` und wird via `[FILE: ...]`-Marker als Anhang-Bubble in App + Diagnostic gerendert. Download-Status (mehrere GB) sichtbar als 🎉-Toast wenn fertig
|
||||||
- [x] **ARIA Live (Diagnostic) + Not-Aus**: read-only Mirror der Claude-Code-Session ersetzt den SSH-Tab. Tool-Calls + Inputs + Outputs (truncated 4 KB) live, farbcodiert. Roter ⛔ Not-Aus-Button schickt `cancel_request` mit `hard:true` → Bridge ruft den proxy-internen `/cancel-all` Side-Channel (Port 3457) → alle Claude-Subprocesses sofort tot. Plus: Idle-Watchdog im Proxy (20 min Inaktivitaet → Subprocess-Kill) + httpx-Timeout-Split im Brain (connect 10s / read 24h) damit lange Pentests durchlaufen
|
- [x] **ARIA Live (Diagnostic) + Not-Aus**: read-only Mirror der Claude-Code-Session ersetzt den SSH-Tab. Tool-Calls + Inputs + Outputs (truncated 4 KB) live, farbcodiert. Roter ⛔ Not-Aus-Button schickt `cancel_request` mit `hard:true` → Bridge ruft den proxy-internen `/cancel-all` Side-Channel (Port 3457) → alle Claude-Subprocesses sofort tot. Plus: Idle-Watchdog im Proxy (20 min Inaktivitaet → Subprocess-Kill) + httpx-Timeout-Split im Brain (connect 10s / read 24h) damit lange Pentests durchlaufen
|
||||||
- [x] **OAuth2-Pipeline ueber RVS-Callback**: Caddy mit Let's Encrypt vor dem RVS, HTTP-Route `/oauth/callback/{service}` broadcastet als `oauth_callback`-WS-Message, aria-bridge forwarded an Brain, Token landet in `/shared/config/oauth_tokens.json` (mode 0600). ARIAs `oauth_register_provider`-Tool legt neue Provider on-demand an (URLs/scopes, nicht Credentials). Diagnostic + App haben beide Provider-Verwaltung inklusive Custom-Provider-Anlage
|
- [x] **OAuth2-Pipeline ueber RVS-Callback**: Caddy mit Let's Encrypt vor dem RVS, HTTP-Route `/oauth/callback/{service}` broadcastet als `oauth_callback`-WS-Message, aria-bridge forwarded an Brain, Token landet in `/shared/config/oauth_tokens.json` (mode 0600). ARIAs `oauth_register_provider`-Tool legt neue Provider on-demand an (URLs/scopes, nicht Credentials). Diagnostic + App haben beide Provider-Verwaltung inklusive Custom-Provider-Anlage
|
||||||
- [x] **Skill-Mgmt-Tools fuer ARIA**: `skill_update` (Code/README/pip_packages mit venv-Rebuild) + `skill_delete` — verhindert Skill-Friedhof mit `-v2`/`-fixed`-Suffixen. Plus App-seitiger SkillBrowser (Run + Live-Output + Logs der letzten 20 Runs) in Settings → 🛠️ Skills
|
- [x] **Skill-Mgmt-Tools fuer ARIA**: `skill_update` (Code/README/pip_packages mit venv-Rebuild) + `skill_delete` — verhindert Skill-Friedhof mit `-v2`/`-fixed`-Suffixen. Plus App-seitiger SkillBrowser (Run + Live-Output + Logs der letzten 20 Runs) in Settings → 🛠️ Skills
|
||||||
|
|||||||
@@ -79,8 +79,8 @@ android {
|
|||||||
applicationId "com.ariacockpit"
|
applicationId "com.ariacockpit"
|
||||||
minSdkVersion rootProject.ext.minSdkVersion
|
minSdkVersion rootProject.ext.minSdkVersion
|
||||||
targetSdkVersion rootProject.ext.targetSdkVersion
|
targetSdkVersion rootProject.ext.targetSdkVersion
|
||||||
versionCode 20208
|
versionCode 20408
|
||||||
versionName "0.2.2.8"
|
versionName "0.2.4.8"
|
||||||
// Fallback fuer Libraries mit Product Flavors
|
// Fallback fuer Libraries mit Product Flavors
|
||||||
missingDimensionStrategy 'react-native-camera', 'general'
|
missingDimensionStrategy 'react-native-camera', 'general'
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -59,7 +59,12 @@ class OpenWakeWordModule(reactContext: ReactApplicationContext) : ReactContextBa
|
|||||||
// Trigger eingestuft werden kann. Folge: App pausiert beim Oeffnen die Musik,
|
// Trigger eingestuft werden kann. Folge: App pausiert beim Oeffnen die Musik,
|
||||||
// weil der False-Positive die AudioFocus-Switch-Logik anwirft (Stefan-Bug 06/2026).
|
// weil der False-Positive die AudioFocus-Switch-Logik anwirft (Stefan-Bug 06/2026).
|
||||||
// Loesung: in dieser Phase keine Detections an JS weiterleiten.
|
// Loesung: in dieser Phase keine Detections an JS weiterleiten.
|
||||||
private const val STARTUP_SUPPRESSION_MS = 1500L
|
private const val STARTUP_SUPPRESSION_MS = 600L
|
||||||
|
// PCM-Ringpuffer fuer die Wake-Wort-Bestaetigung: letzte 2s roh (16kHz
|
||||||
|
// mono s16). Bei einer Erkennung wird der Vor-Trigger-Schnipsel an JS
|
||||||
|
// gereicht und dort von Voxtral verifiziert (gegen Musik-Fehltrigger).
|
||||||
|
private const val PCM_RING_SAMPLES = 32000 // 2.0s @ 16kHz
|
||||||
|
private const val PRE_TRIGGER_SAMPLES = 24000 // 1.5s Schnipsel an JS
|
||||||
}
|
}
|
||||||
|
|
||||||
private val env: OrtEnvironment = OrtEnvironment.getEnvironment()
|
private val env: OrtEnvironment = OrtEnvironment.getEnvironment()
|
||||||
@@ -106,6 +111,13 @@ class OpenWakeWordModule(reactContext: ReactApplicationContext) : ReactContextBa
|
|||||||
private val embBuffer: ArrayDeque<FloatArray> = ArrayDeque(32) // Ringpuffer letzter Embeddings
|
private val embBuffer: ArrayDeque<FloatArray> = ArrayDeque(32) // Ringpuffer letzter Embeddings
|
||||||
private var consecutiveAboveThreshold: Int = 0
|
private var consecutiveAboveThreshold: Int = 0
|
||||||
private var lastDetectionMs: Long = 0L
|
private var lastDetectionMs: Long = 0L
|
||||||
|
// Roh-PCM-Ringpuffer (letzte ~2s) fuer die Wake-Wort-Bestaetigung. Bei einer
|
||||||
|
// Erkennung wird der Vor-Trigger-Schnipsel base64-kodiert an JS gereicht und
|
||||||
|
// dort von Voxtral verifiziert ("war das wirklich 'Computer' oder Musik?").
|
||||||
|
private val pcmRing = ShortArray(PCM_RING_SAMPLES)
|
||||||
|
private var pcmRingPos = 0
|
||||||
|
private var pcmRingFilled = false
|
||||||
|
private val pcmRingLock = Any()
|
||||||
// Zeitpunkt des letzten startRecording — fuer STARTUP_SUPPRESSION_MS-Fenster
|
// Zeitpunkt des letzten startRecording — fuer STARTUP_SUPPRESSION_MS-Fenster
|
||||||
private var recordingStartedMs: Long = 0L
|
private var recordingStartedMs: Long = 0L
|
||||||
|
|
||||||
@@ -430,6 +442,36 @@ class OpenWakeWordModule(reactContext: ReactApplicationContext) : ReactContextBa
|
|||||||
embBuffer.clear()
|
embBuffer.clear()
|
||||||
consecutiveAboveThreshold = 0
|
consecutiveAboveThreshold = 0
|
||||||
lastDetectionMs = 0L
|
lastDetectionMs = 0L
|
||||||
|
// PCM-Ring frisch: sonst koennte Alt-Audio aus dem vorigen Arm-Zyklus
|
||||||
|
// in den Bestaetigungs-Schnipsel bluten.
|
||||||
|
synchronized(pcmRingLock) { pcmRingPos = 0; pcmRingFilled = false }
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Letzte ~1.5s Roh-PCM aus dem Ringpuffer als Base64 (s16le, 16kHz mono),
|
||||||
|
* fuer die Voxtral-Wake-Bestaetigung. null wenn noch zu wenig Audio da ist
|
||||||
|
* oder das Kodieren scheitert (dann macht JS fail-open weiter wie bisher). */
|
||||||
|
private fun snapshotPreTrigger(): String? {
|
||||||
|
val out: ByteArray
|
||||||
|
synchronized(pcmRingLock) {
|
||||||
|
val available = if (pcmRingFilled) PCM_RING_SAMPLES else pcmRingPos
|
||||||
|
val n = if (available < PRE_TRIGGER_SAMPLES) available else PRE_TRIGGER_SAMPLES
|
||||||
|
if (n <= 0) return null
|
||||||
|
out = ByteArray(n * 2)
|
||||||
|
var idx = (pcmRingPos - n + PCM_RING_SAMPLES) % PCM_RING_SAMPLES
|
||||||
|
for (i in 0 until n) {
|
||||||
|
val s = pcmRing[idx].toInt()
|
||||||
|
out[i * 2] = (s and 0xFF).toByte()
|
||||||
|
out[i * 2 + 1] = ((s shr 8) and 0xFF).toByte()
|
||||||
|
idx += 1
|
||||||
|
if (idx >= PCM_RING_SAMPLES) idx = 0
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return try {
|
||||||
|
android.util.Base64.encodeToString(out, android.util.Base64.NO_WRAP)
|
||||||
|
} catch (e: Exception) {
|
||||||
|
Log.w(TAG, "snapshotPreTrigger base64 fehlgeschlagen: ${e.message}")
|
||||||
|
null
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
private fun emitDetected() {
|
private fun emitDetected() {
|
||||||
@@ -438,8 +480,10 @@ class OpenWakeWordModule(reactContext: ReactApplicationContext) : ReactContextBa
|
|||||||
Log.i(TAG, "Wake-Word emit unterdrueckt (sinceStart=${sinceStart}ms < ${STARTUP_SUPPRESSION_MS}ms — Mikro-Spin-up-Spike)")
|
Log.i(TAG, "Wake-Word emit unterdrueckt (sinceStart=${sinceStart}ms < ${STARTUP_SUPPRESSION_MS}ms — Mikro-Spin-up-Spike)")
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
val preTriggerB64 = snapshotPreTrigger()
|
||||||
val params = com.facebook.react.bridge.Arguments.createMap().apply {
|
val params = com.facebook.react.bridge.Arguments.createMap().apply {
|
||||||
putString("model", modelName)
|
putString("model", modelName)
|
||||||
|
if (preTriggerB64 != null) putString("preTriggerPcm", preTriggerB64)
|
||||||
}
|
}
|
||||||
try {
|
try {
|
||||||
reactApplicationContext
|
reactApplicationContext
|
||||||
@@ -466,6 +510,14 @@ class OpenWakeWordModule(reactContext: ReactApplicationContext) : ReactContextBa
|
|||||||
read += n
|
read += n
|
||||||
}
|
}
|
||||||
if (!running.get()) break
|
if (!running.get()) break
|
||||||
|
// Chunk in den PCM-Ringpuffer schreiben (fuer Wake-Wort-Bestaetigung).
|
||||||
|
synchronized(pcmRingLock) {
|
||||||
|
for (i in 0 until CHUNK_SAMPLES) {
|
||||||
|
pcmRing[pcmRingPos] = buf[i]
|
||||||
|
pcmRingPos += 1
|
||||||
|
if (pcmRingPos >= PCM_RING_SAMPLES) { pcmRingPos = 0; pcmRingFilled = true }
|
||||||
|
}
|
||||||
|
}
|
||||||
try {
|
try {
|
||||||
processChunk(buf)
|
processChunk(buf)
|
||||||
} catch (e: Exception) {
|
} catch (e: Exception) {
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "aria-cockpit",
|
"name": "aria-cockpit",
|
||||||
"version": "0.2.2.8",
|
"version": "0.2.4.8",
|
||||||
"private": true,
|
"private": true,
|
||||||
"scripts": {
|
"scripts": {
|
||||||
"android": "react-native run-android",
|
"android": "react-native run-android",
|
||||||
|
|||||||
@@ -35,7 +35,7 @@ import MemoryBrowser from '../components/MemoryBrowser';
|
|||||||
import ErrorBoundary from '../components/ErrorBoundary';
|
import ErrorBoundary from '../components/ErrorBoundary';
|
||||||
import rvs, { RVSMessage, ConnectionState } from '../services/rvs';
|
import rvs, { RVSMessage, ConnectionState } from '../services/rvs';
|
||||||
import audioService from '../services/audio';
|
import audioService from '../services/audio';
|
||||||
import wakeWordService, { loadPassiveListenMs } from '../services/wakeword';
|
import wakeWordService from '../services/wakeword';
|
||||||
import ProjectsBrowser from '../components/ProjectsBrowser';
|
import ProjectsBrowser from '../components/ProjectsBrowser';
|
||||||
import brainApi, { Project as BrainProject } from '../services/brainApi';
|
import brainApi, { Project as BrainProject } from '../services/brainApi';
|
||||||
import projectFocus from '../services/projectFocus';
|
import projectFocus from '../services/projectFocus';
|
||||||
@@ -50,7 +50,7 @@ import VoiceButton from '../components/VoiceButton';
|
|||||||
import FileUpload, { FileData } from '../components/FileUpload';
|
import FileUpload, { FileData } from '../components/FileUpload';
|
||||||
import CameraUpload, { PhotoData } from '../components/CameraUpload';
|
import CameraUpload, { PhotoData } from '../components/CameraUpload';
|
||||||
import MessageText from '../components/MessageText';
|
import MessageText from '../components/MessageText';
|
||||||
import { loadConvWindowMs, loadTtsSpeed, TTS_SPEED_DEFAULT, loadSttEndpointMs } from '../services/audio';
|
import { loadTtsSpeed, TTS_SPEED_DEFAULT, loadSttEndpointMs, loadMaxRecordingMs, loadBargeInEnabled } from '../services/audio';
|
||||||
import Geolocation from '@react-native-community/geolocation';
|
import Geolocation from '@react-native-community/geolocation';
|
||||||
|
|
||||||
// --- Typen ---
|
// --- Typen ---
|
||||||
@@ -93,6 +93,9 @@ interface ChatMessage {
|
|||||||
* gespiegelt damit wir die EXAKT richtige Placeholder-Bubble ersetzen,
|
* gespiegelt damit wir die EXAKT richtige Placeholder-Bubble ersetzen,
|
||||||
* auch wenn mehrere Aufnahmen parallel offen sind. */
|
* auch wenn mehrere Aufnahmen parallel offen sind. */
|
||||||
audioRequestId?: string;
|
audioRequestId?: string;
|
||||||
|
/** Laenge der Sprachaufnahme in Sekunden (aus dem stt_endpoint) — fuer die
|
||||||
|
* Dauer-Anzeige an der Voice-Bubble. */
|
||||||
|
durationS?: number;
|
||||||
/** Skill-Created-Bubble: ARIA hat einen neuen Skill angelegt */
|
/** Skill-Created-Bubble: ARIA hat einen neuen Skill angelegt */
|
||||||
skillCreated?: {
|
skillCreated?: {
|
||||||
name: string;
|
name: string;
|
||||||
@@ -184,6 +187,14 @@ function stripSystemHints(text: string): string {
|
|||||||
}
|
}
|
||||||
return out;
|
return out;
|
||||||
}
|
}
|
||||||
|
/** Sekunden → "M:SS" fuer die Sprachnachricht-Dauer. */
|
||||||
|
function formatDur(sec: number): string {
|
||||||
|
const s = Math.max(0, Math.round(sec));
|
||||||
|
const m = Math.floor(s / 60);
|
||||||
|
const r = s % 60;
|
||||||
|
return `${m}:${r.toString().padStart(2, '0')}`;
|
||||||
|
}
|
||||||
|
|
||||||
const DEFAULT_ATTACHMENT_DIR = `${RNFS.DocumentDirectoryPath}/chat_attachments`;
|
const DEFAULT_ATTACHMENT_DIR = `${RNFS.DocumentDirectoryPath}/chat_attachments`;
|
||||||
const STORAGE_PATH_KEY = 'aria_attachment_storage_path';
|
const STORAGE_PATH_KEY = 'aria_attachment_storage_path';
|
||||||
|
|
||||||
@@ -346,7 +357,7 @@ const ChatScreen: React.FC = () => {
|
|||||||
// folgende identische Events (z.B. zwei 'thinking' hintereinander) den
|
// folgende identische Events (z.B. zwei 'thinking' hintereinander) den
|
||||||
// Stream zumuellen. Eigentlich seltener Fall, aber billig zu pruefen.
|
// Stream zumuellen. Eigentlich seltener Fall, aber billig zu pruefen.
|
||||||
const lastThoughtKeyRef = useRef<string>('');
|
const lastThoughtKeyRef = useRef<string>('');
|
||||||
// Service-Status (Gamebox: F5-TTS / Whisper Lade-Status) + Banner-Sichtbarkeit
|
// Service-Status (AI-Box: F5-TTS / Whisper Lade-Status) + Banner-Sichtbarkeit
|
||||||
const [serviceStatus, setServiceStatus] = useState<Record<string, {state: string, model?: string, loadSeconds?: number, error?: string, downloading?: boolean, freshlyDownloaded?: boolean}>>({});
|
const [serviceStatus, setServiceStatus] = useState<Record<string, {state: string, model?: string, loadSeconds?: number, error?: string, downloading?: boolean, freshlyDownloaded?: boolean}>>({});
|
||||||
const [serviceBannerDismissed, setServiceBannerDismissed] = useState(false);
|
const [serviceBannerDismissed, setServiceBannerDismissed] = useState(false);
|
||||||
// Gerätelokale TTS-Config: globaler Toggle (aus Settings) + temporäres Muten (Mund-Button)
|
// Gerätelokale TTS-Config: globaler Toggle (aus Settings) + temporäres Muten (Mund-Button)
|
||||||
@@ -373,6 +384,16 @@ const ChatScreen: React.FC = () => {
|
|||||||
// stoppen? Kommt als 'converse' in der Chat-Payload; onPlaybackFinished liest
|
// stoppen? Kommt als 'converse' in der Chat-Payload; onPlaybackFinished liest
|
||||||
// es. Default true (Konversation). false = Einzelaktion/Skill-Antwort.
|
// es. Default true (Konversation). false = Einzelaktion/Skill-Antwort.
|
||||||
const converseRef = useRef<boolean>(true);
|
const converseRef = useRef<boolean>(true);
|
||||||
|
// Passiv-Lausch-Fenster (Weiterreden nach ARIAs Antwort): Umgebungsgeraeusch
|
||||||
|
// (Musik/TV) darf das Fenster NICHT vorzeitig beenden. Bei einem no-speech-
|
||||||
|
// Endpoint (Silero verwirft Musik) wird — solange die Stille-Toleranz ab
|
||||||
|
// Fenster-Oeffnung noch laeuft — nochmal gelauscht statt sofort aufs Wake-Word
|
||||||
|
// zurueckzufallen. Start-Zeit + Re-Listen-Zaehler + Budget hier gemerkt.
|
||||||
|
const passiveListenStartRef = useRef<number>(0);
|
||||||
|
const passiveReListenCountRef = useRef<number>(0);
|
||||||
|
const passiveToleranceRef = useRef<number>(5000);
|
||||||
|
// Barge-in erlaubt? Default false = Halb-Duplex (waehrend TTS kein Mikro).
|
||||||
|
const bargeInEnabledRef = useRef<boolean>(false);
|
||||||
|
|
||||||
const flatListRef = useRef<FlatList>(null);
|
const flatListRef = useRef<FlatList>(null);
|
||||||
const messageIdCounter = useRef(0);
|
const messageIdCounter = useRef(0);
|
||||||
@@ -651,6 +672,7 @@ const ChatScreen: React.FC = () => {
|
|||||||
const voice = await AsyncStorage.getItem('aria_xtts_voice');
|
const voice = await AsyncStorage.getItem('aria_xtts_voice');
|
||||||
localXttsVoiceRef.current = voice || '';
|
localXttsVoiceRef.current = voice || '';
|
||||||
ttsSpeedRef.current = await loadTtsSpeed();
|
ttsSpeedRef.current = await loadTtsSpeed();
|
||||||
|
bargeInEnabledRef.current = await loadBargeInEnabled();
|
||||||
const gps = await AsyncStorage.getItem('aria_gps_enabled');
|
const gps = await AsyncStorage.getItem('aria_gps_enabled');
|
||||||
setGpsEnabled(gps === 'true');
|
setGpsEnabled(gps === 'true');
|
||||||
const hints = await AsyncStorage.getItem('aria_show_hints');
|
const hints = await AsyncStorage.getItem('aria_show_hints');
|
||||||
@@ -1387,13 +1409,61 @@ const ChatScreen: React.FC = () => {
|
|||||||
// Fallback mehr: die Bridge schickt speak zuverlaessig mit.
|
// Fallback mehr: die Bridge schickt speak zuverlaessig mit.
|
||||||
// Merken ob nach dem Vorlesen 30s weiterlauschen (Gespraech) oder direkt
|
// Merken ob nach dem Vorlesen 30s weiterlauschen (Gespraech) oder direkt
|
||||||
// stoppen — onPlaybackFinished liest converseRef. Default true.
|
// stoppen — onPlaybackFinished liest converseRef. Default true.
|
||||||
converseRef.current = (message.payload as any).converse !== false;
|
// Passiv-Lauschen (30s) NUR wenn das Brain explizit converse:true schickt.
|
||||||
|
// Vorher default true → jeder Befehl (auch "Spiele Spotify" mit gesproche-
|
||||||
|
// ner Bestaetigung) landete im 30s-Fenster. Jetzt: einzelne Befehle enden
|
||||||
|
// sofort (zurueck aufs Wake-Word), nur echte Gespraeche lauschen weiter.
|
||||||
|
converseRef.current = (message.payload as any).converse === true;
|
||||||
|
const _wakeOff = (message.payload as any).wake_off === true;
|
||||||
|
const _wakeOn = (message.payload as any).wake_on === true;
|
||||||
const _isSilent = (message.payload as any).speak === false;
|
const _isSilent = (message.payload as any).speak === false;
|
||||||
if (_isSilent && wakeWordService.isConversing()) {
|
if (_wakeOn) {
|
||||||
// Klarer Steuerbefehl (Liedersteuerung etc.) = KEINE Konversation →
|
// "Wake-Word an" per Text/Aufnahme-Button → Listener wieder starten
|
||||||
// STOP: direkt zurueck aufs Wake-Word. Kein Gong, keine Aufnahme,
|
// (gleicher Weg wie toggleWakeWord-on). Geht auch wenn das Ohr taub war,
|
||||||
// kein 30s-Fenster (skipPassive=true).
|
// weil der Befehl NICHT ueber "Computer" kam.
|
||||||
wakeWordService.endConversation(true).catch(() => {});
|
(async () => {
|
||||||
|
try {
|
||||||
|
const started = await wakeWordService.start();
|
||||||
|
setWakeWordActive(started);
|
||||||
|
console.log('[Chat] Wake-Word per Befehl AN gestartet:', started);
|
||||||
|
} catch (e) {
|
||||||
|
console.warn('[Chat] Wake-Word AN fehlgeschlagen:', e);
|
||||||
|
}
|
||||||
|
})();
|
||||||
|
} else if (_wakeOff) {
|
||||||
|
// ARIA hat "Wake-Word aus" per Sprache bekommen → Listener KOMPLETT
|
||||||
|
// stoppen (Mikro frei, echte Ruhe). Gleicher Weg wie der Ohr-Button
|
||||||
|
// (toggleWakeWord-off). Wieder-An nur ueber den Button (dann taub).
|
||||||
|
(async () => {
|
||||||
|
try {
|
||||||
|
if (audioService.isStreamingRecording()) {
|
||||||
|
await audioService.cancelStreamingRecording('wake-off-voice');
|
||||||
|
} else {
|
||||||
|
await audioService.stopRecording();
|
||||||
|
}
|
||||||
|
} catch {}
|
||||||
|
try { await wakeWordService.stop(); } catch {}
|
||||||
|
setWakeWordActive(false);
|
||||||
|
console.log('[Chat] Wake-Word per Sprachbefehl AUS — Ohr-Button zum Wieder-Anmachen');
|
||||||
|
})();
|
||||||
|
} else if (_isSilent) {
|
||||||
|
// Steuerbefehl (speak=false) ist ausgefuehrt und wird NICHT vorgelesen.
|
||||||
|
// Ohne TTS feuert onPlaybackFinished nie — der Mikro-/Konversations-
|
||||||
|
// Lifecycle muss hier selbst weitergeschaltet werden, sonst haengt das Ohr.
|
||||||
|
if (converseRef.current) {
|
||||||
|
// Befehlskette laeuft WEITER ([[WEITER]]): Mikro NICHT schliessen,
|
||||||
|
// sondern das passive Lausch-Fenster oeffnen (endConversation(false)),
|
||||||
|
// damit der naechste Kettenbefehl direkt gesprochen werden kann. ARIA
|
||||||
|
// haelt bewusst offen, bis sie [[ENDE]] (converse=false) schickt.
|
||||||
|
if (wakeWordService.isConversing()) {
|
||||||
|
wakeWordService.endConversation(false).catch(() => {});
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
// Einzelbefehl / [[ENDE]] → ARIA "drueckt selbst Stop": jede offene
|
||||||
|
// Aufnahme schliessen + zurueck aufs Wake-Word, egal in welchem Zustand
|
||||||
|
// (conversing, passives Lauschen ODER offene Streaming-Aufnahme).
|
||||||
|
ariaStopRecording('silent-command').catch(() => {});
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1521,7 +1591,7 @@ const ChatScreen: React.FC = () => {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Gamebox-Bridges (f5tts/whisper/flux) melden Lade-Status — Banner oben.
|
// AI-Box-Bridges (f5tts/whisper/flux) melden Lade-Status — Banner oben.
|
||||||
// Toast bei Download-Ende: erstmaliger HF-Download (mehrere GB) → User
|
// Toast bei Download-Ende: erstmaliger HF-Download (mehrere GB) → User
|
||||||
// soll wissen dass er Bilder/Stimmen jetzt nutzen kann ohne in den
|
// soll wissen dass er Bilder/Stimmen jetzt nutzen kann ohne in den
|
||||||
// Banner gucken zu muessen.
|
// Banner gucken zu muessen.
|
||||||
@@ -1628,8 +1698,12 @@ const ChatScreen: React.FC = () => {
|
|||||||
// Im Hintergrund gibt's kein Multi-Turn → direkt re-armen (skipPassive).
|
// Im Hintergrund gibt's kein Multi-Turn → direkt re-armen (skipPassive).
|
||||||
// converse=false (Skill-/Einzelantwort, z.B. 'was laeuft gerade') → auch
|
// converse=false (Skill-/Einzelantwort, z.B. 'was laeuft gerade') → auch
|
||||||
// ohne 30s: vorlesen + direkt zurueck aufs Wake-Word.
|
// ohne 30s: vorlesen + direkt zurueck aufs Wake-Word.
|
||||||
|
// Im Hintergrund normalerweise direkt re-armen (kein Multi-Turn) — ABER
|
||||||
|
// wenn Hintergrund-Wake bewusst AN ist, will der User auch im Hintergrund
|
||||||
|
// ein Gespraech fuehren, also den Konversationsmodus offen halten.
|
||||||
const bg = AppState.currentState !== 'active';
|
const bg = AppState.currentState !== 'active';
|
||||||
wakeWordService.endConversation(bg || !converseRef.current).catch(() => {});
|
const bgForcesArmed = bg && !wakeWordService.isBgWakeEnabled();
|
||||||
|
wakeWordService.endConversation(bgForcesArmed || !converseRef.current).catch(() => {});
|
||||||
});
|
});
|
||||||
return () => unsubPlayback();
|
return () => unsubPlayback();
|
||||||
}, []);
|
}, []);
|
||||||
@@ -1648,7 +1722,11 @@ const ChatScreen: React.FC = () => {
|
|||||||
rememberMyRequest(audioRequestId);
|
rememberMyRequest(audioRequestId);
|
||||||
const wasInterrupted = interruptAriaIfBusy();
|
const wasInterrupted = interruptAriaIfBusy();
|
||||||
const location = await getCurrentLocation();
|
const location = await getCurrentLocation();
|
||||||
const windowMs = await loadConvWindowMs();
|
// EIN Wert regiert: die Stille-Toleranz. Sie gilt sowohl als Pause WÄHREND
|
||||||
|
// des Redens (endpointMs) ALS AUCH als "wenn du nicht anfängst zu reden,
|
||||||
|
// ist Schluss" (noSpeechTimeoutMs). Kein separates 30s-Konversationsfenster
|
||||||
|
// mehr — Stefans Modell: sagst du nichts, greift der Stille-Wert.
|
||||||
|
const sttEndpointMs = await loadSttEndpointMs();
|
||||||
|
|
||||||
const userMsg: ChatMessage = {
|
const userMsg: ChatMessage = {
|
||||||
id: nextId(),
|
id: nextId(),
|
||||||
@@ -1666,9 +1744,11 @@ const ChatScreen: React.FC = () => {
|
|||||||
speed: ttsSpeedRef.current,
|
speed: ttsSpeedRef.current,
|
||||||
interrupted: wasInterrupted,
|
interrupted: wasInterrupted,
|
||||||
location: location || null,
|
location: location || null,
|
||||||
noSpeechTimeoutMs: windowMs,
|
noSpeechTimeoutMs: sttEndpointMs,
|
||||||
endpointMs: await loadSttEndpointMs(),
|
endpointMs: sttEndpointMs,
|
||||||
hardCapMs: 60000,
|
// Notbremse 5 min (nicht 1 min) — der Stille-Endpoint beendet normale
|
||||||
|
// Turns eh sofort; der Cap darf lange Diktate nicht mitten drin kappen.
|
||||||
|
hardCapMs: await loadMaxRecordingMs(),
|
||||||
projectId: focusedProjectIdRef.current,
|
projectId: focusedProjectIdRef.current,
|
||||||
});
|
});
|
||||||
import('../services/logger').then(m => m.reportAppDebug('wake.cb', `startStreamingRecording returned ok=${ok}`)).catch(()=>{});
|
import('../services/logger').then(m => m.reportAppDebug('wake.cb', `startStreamingRecording returned ok=${ok}`)).catch(()=>{});
|
||||||
@@ -1696,6 +1776,13 @@ const ChatScreen: React.FC = () => {
|
|||||||
if (ev.text && ev.text.trim()) {
|
if (ev.text && ev.text.trim()) {
|
||||||
console.log('[Chat] STT-Endpoint: %r (reason=%s, %dms, %.1fs Audio)',
|
console.log('[Chat] STT-Endpoint: %r (reason=%s, %dms, %.1fs Audio)',
|
||||||
ev.text.slice(0, 80), ev.reason, ev.sttMs, ev.durationS);
|
ev.text.slice(0, 80), ev.reason, ev.sttMs, ev.durationS);
|
||||||
|
// Aufnahme-Dauer an die passende Voice-Bubble haengen (Anzeige). Der
|
||||||
|
// spaetere STT-Text-Update spreadet die Message, die Dauer bleibt.
|
||||||
|
if (ev.audioRequestId && typeof ev.durationS === 'number' && ev.durationS > 0) {
|
||||||
|
const dur = ev.durationS;
|
||||||
|
setMessages(prev => prev.map(m =>
|
||||||
|
m.audioRequestId === ev.audioRequestId ? { ...m, durationS: dur } : m));
|
||||||
|
}
|
||||||
// Wenn passive lauschend: User hat tatsaechlich was gesagt → uebergang
|
// Wenn passive lauschend: User hat tatsaechlich was gesagt → uebergang
|
||||||
// zu 'conversing' damit der normale Flow greift (TTS, resume, etc.)
|
// zu 'conversing' damit der normale Flow greift (TTS, resume, etc.)
|
||||||
if (wakeWordService.getState() === 'listening') {
|
if (wakeWordService.getState() === 'listening') {
|
||||||
@@ -1716,12 +1803,24 @@ const ChatScreen: React.FC = () => {
|
|||||||
!(m.audioRequestId === ev.audioRequestId
|
!(m.audioRequestId === ev.audioRequestId
|
||||||
&& m.text.includes('Spracheingabe wird verarbeitet'))));
|
&& m.text.includes('Spracheingabe wird verarbeitet'))));
|
||||||
}
|
}
|
||||||
// Bei Passive-Listen + speaker_mismatch oder no-speech: erneut passiv
|
// Passiv-Lauschen: leeres Endpoint (no-speech / Silero-Musik / speaker_
|
||||||
// lauschen (Timer im wakeword-service laeuft weiter, regelt das Ende).
|
// mismatch). NICHT sofort beenden — solange die Stille-Toleranz ab
|
||||||
// Sonst endConversation wie bisher.
|
// Fenster-Oeffnung noch laeuft, nochmal lauschen. So killt Umgebungs-
|
||||||
|
// musik das Weiterreden nicht: die Musik wird verworfen, das Fenster
|
||||||
|
// bleibt bis zur Toleranz offen, du kannst innerhalb reden. Erst wenn
|
||||||
|
// die Toleranz wirklich um ist (oder zu viele Runden) → aufs Wake-Word.
|
||||||
if (wakeWordService.getState() === 'listening') {
|
if (wakeWordService.getState() === 'listening') {
|
||||||
console.log('[Chat] Passive-Listen: leeres Endpoint — naechste passive Aufnahme');
|
const elapsed = Date.now() - passiveListenStartRef.current;
|
||||||
startPassiveStreamingRecording();
|
const budget = passiveToleranceRef.current || 5000;
|
||||||
|
if (elapsed < budget && passiveReListenCountRef.current < 15) {
|
||||||
|
passiveReListenCountRef.current += 1;
|
||||||
|
console.log('[Chat] Passive-Listen: leeres Endpoint (%s) — Umgebung, re-listen (%dms/%dms, #%d)',
|
||||||
|
ev.reason, elapsed, budget, passiveReListenCountRef.current);
|
||||||
|
startPassiveStreamingRecording();
|
||||||
|
} else {
|
||||||
|
console.log('[Chat] Passive-Listen: Stille-Toleranz aufgebraucht (%dms) — Ende, zurueck aufs Wake-Word', elapsed);
|
||||||
|
wakeWordService.exitPassiveListening('timeout').catch(() => {});
|
||||||
|
}
|
||||||
} else {
|
} else {
|
||||||
wakeWordService.endConversation();
|
wakeWordService.endConversation();
|
||||||
if (!wakeWordService.isActive()) setWakeWordActive(false);
|
if (!wakeWordService.isActive()) setWakeWordActive(false);
|
||||||
@@ -1733,8 +1832,14 @@ const ChatScreen: React.FC = () => {
|
|||||||
// geschaltet (nach endConversation). Wir starten eine streaming-Aufnahme
|
// geschaltet (nach endConversation). Wir starten eine streaming-Aufnahme
|
||||||
// OHNE User-Bubble + ohne wake-ready-Sound. Speaker-ID-Gating in der
|
// OHNE User-Bubble + ohne wake-ready-Sound. Speaker-ID-Gating in der
|
||||||
// Whisper-Bridge filtert fremde Stimmen weg.
|
// Whisper-Bridge filtert fremde Stimmen weg.
|
||||||
const unsubPassive = wakeWordService.onPassiveListen(() => {
|
const unsubPassive = wakeWordService.onPassiveListen(async () => {
|
||||||
console.log('[Chat] Passive-Listen aktiviert — starte stille Streaming-Aufnahme');
|
console.log('[Chat] Passive-Listen aktiviert — starte stille Streaming-Aufnahme');
|
||||||
|
// Fenster NEU geoeffnet (nach ARIAs Antwort): Budget-Uhr + Re-Listen-Zaehler
|
||||||
|
// zuruecksetzen. Re-Listen ruft startPassiveStreamingRecording direkt (nicht
|
||||||
|
// ueber diesen Callback), also bleibt der Startzeitpunkt erhalten.
|
||||||
|
passiveListenStartRef.current = Date.now();
|
||||||
|
passiveReListenCountRef.current = 0;
|
||||||
|
passiveToleranceRef.current = await loadSttEndpointMs();
|
||||||
startPassiveStreamingRecording();
|
startPassiveStreamingRecording();
|
||||||
});
|
});
|
||||||
|
|
||||||
@@ -1751,7 +1856,7 @@ const ChatScreen: React.FC = () => {
|
|||||||
const audioRequestId = `audio_${Date.now()}_${Math.floor(Math.random() * 100000)}`;
|
const audioRequestId = `audio_${Date.now()}_${Math.floor(Math.random() * 100000)}`;
|
||||||
rememberMyRequest(audioRequestId);
|
rememberMyRequest(audioRequestId);
|
||||||
const location = await getCurrentLocation();
|
const location = await getCurrentLocation();
|
||||||
const windowMs = await loadConvWindowMs();
|
const sttEndpointMs = await loadSttEndpointMs(); // ein Wert für Pause + No-Speech
|
||||||
|
|
||||||
const userMsg: ChatMessage = {
|
const userMsg: ChatMessage = {
|
||||||
id: nextId(),
|
id: nextId(),
|
||||||
@@ -1769,9 +1874,10 @@ const ChatScreen: React.FC = () => {
|
|||||||
speed: ttsSpeedRef.current,
|
speed: ttsSpeedRef.current,
|
||||||
interrupted: true, // Barge-In → Brain weiss "User hat unterbrochen"
|
interrupted: true, // Barge-In → Brain weiss "User hat unterbrochen"
|
||||||
location: location || null,
|
location: location || null,
|
||||||
noSpeechTimeoutMs: windowMs,
|
noSpeechTimeoutMs: sttEndpointMs,
|
||||||
endpointMs: await loadSttEndpointMs(),
|
endpointMs: sttEndpointMs,
|
||||||
hardCapMs: 60000,
|
// Notbremse 5 min (s.o.) — lange Diktate nicht bei 1 min abschneiden.
|
||||||
|
hardCapMs: await loadMaxRecordingMs(),
|
||||||
projectId: focusedProjectIdRef.current,
|
projectId: focusedProjectIdRef.current,
|
||||||
});
|
});
|
||||||
if (ok) {
|
if (ok) {
|
||||||
@@ -1789,7 +1895,9 @@ const ChatScreen: React.FC = () => {
|
|||||||
// Prozess nicht killt wenn die App im Hintergrund ist.
|
// Prozess nicht killt wenn die App im Hintergrund ist.
|
||||||
const unsubTtsStart = audioService.onPlaybackStarted(() => {
|
const unsubTtsStart = audioService.onPlaybackStarted(() => {
|
||||||
acquireBackgroundAudio('tts').catch(() => {});
|
acquireBackgroundAudio('tts').catch(() => {});
|
||||||
if (wakeWordService.isConversing() && wakeWordService.hasWakeWord()) {
|
// Barge-Listening (Mikro waehrend TTS) NUR im Barge-in-Modus. Default aus =
|
||||||
|
// Halb-Duplex: ARIA spricht ungestoert zu Ende, dann erst geht das Mikro auf.
|
||||||
|
if (bargeInEnabledRef.current && wakeWordService.isConversing() && wakeWordService.hasWakeWord()) {
|
||||||
wakeWordService.startBargeListening().catch(() => {});
|
wakeWordService.startBargeListening().catch(() => {});
|
||||||
}
|
}
|
||||||
});
|
});
|
||||||
@@ -1828,16 +1936,20 @@ const ChatScreen: React.FC = () => {
|
|||||||
const audioRequestId = `audio_passive_${Date.now()}_${Math.floor(Math.random() * 100000)}`;
|
const audioRequestId = `audio_passive_${Date.now()}_${Math.floor(Math.random() * 100000)}`;
|
||||||
rememberMyRequest(audioRequestId);
|
rememberMyRequest(audioRequestId);
|
||||||
const location = await getCurrentLocation();
|
const location = await getCurrentLocation();
|
||||||
const passiveMs = await loadPassiveListenMs();
|
// Kein 30s-Passiv-Fenster mehr: nach ARIAs Antwort geht das Mikro auf, und
|
||||||
|
// fängst du nicht innerhalb der Stille-Toleranz an zu reden, ist Schluss →
|
||||||
|
// zurück aufs Wake-Word. Derselbe Wert wie die Pause-Toleranz beim Reden.
|
||||||
|
const sttEndpointMs = await loadSttEndpointMs();
|
||||||
const { ok } = await audioService.startStreamingRecording({
|
const { ok } = await audioService.startStreamingRecording({
|
||||||
audioRequestId,
|
audioRequestId,
|
||||||
voice: localXttsVoiceRef.current,
|
voice: localXttsVoiceRef.current,
|
||||||
speed: ttsSpeedRef.current,
|
speed: ttsSpeedRef.current,
|
||||||
interrupted: false,
|
interrupted: false,
|
||||||
location: location || null,
|
location: location || null,
|
||||||
noSpeechTimeoutMs: Math.min(passiveMs, 30000),
|
noSpeechTimeoutMs: sttEndpointMs,
|
||||||
endpointMs: await loadSttEndpointMs(),
|
endpointMs: sttEndpointMs,
|
||||||
hardCapMs: Math.max(passiveMs + 5000, 35000),
|
// Lange Antworten nicht kappen (früher 35s → schnitt langes Reden ab).
|
||||||
|
hardCapMs: await loadMaxRecordingMs(),
|
||||||
projectId: focusedProjectIdRef.current,
|
projectId: focusedProjectIdRef.current,
|
||||||
});
|
});
|
||||||
if (!ok) {
|
if (!ok) {
|
||||||
@@ -2177,7 +2289,7 @@ const ChatScreen: React.FC = () => {
|
|||||||
|
|
||||||
// Aufraeumen von "verarbeitet"-Placeholder die nie ein STT-Result bekommen
|
// Aufraeumen von "verarbeitet"-Placeholder die nie ein STT-Result bekommen
|
||||||
// haben (leere Aufnahme, Wake-Word-Echo, STT-Fehler etc). Timeout skaliert
|
// haben (leere Aufnahme, Wake-Word-Echo, STT-Fehler etc). Timeout skaliert
|
||||||
// mit der Aufnahmedauer — Whisper braucht auf der Gamebox grob real-time/5,
|
// mit der Aufnahmedauer — Whisper braucht auf der AI-Box grob real-time/5,
|
||||||
// plus Bridge-Roundtrip + Network. Formel: 60s Buffer + 1x Aufnahmedauer.
|
// plus Bridge-Roundtrip + Network. Formel: 60s Buffer + 1x Aufnahmedauer.
|
||||||
// Bei 5min Aufnahme = 6 min Wait, bei 5s Aufnahme = 65s. Sicher genug damit
|
// Bei 5min Aufnahme = 6 min Wait, bei 5s Aufnahme = 65s. Sicher genug damit
|
||||||
// langsame STTs nicht versehentlich aufgeraeumt werden.
|
// langsame STTs nicht versehentlich aufgeraeumt werden.
|
||||||
@@ -2210,15 +2322,16 @@ const ChatScreen: React.FC = () => {
|
|||||||
advanceQueue(pid);
|
advanceQueue(pid);
|
||||||
}, [advanceQueue]);
|
}, [advanceQueue]);
|
||||||
|
|
||||||
// Queue-Modus („immer anstellen"): eine neue Sprachnachricht bricht ARIAs
|
// Nimmt der User das Mikro waehrend ARIA SPRICHT, ist das ein echter Interrupt:
|
||||||
// laufende Arbeit NICHT mehr ab. Sie wird — wie Text — angestellt und laeuft
|
// TTS stoppen UND die laufende Brain-Antwort abbrechen (cancel_request). Sonst
|
||||||
// serialisiert (der Brain-Lock pro Projekt reiht /chat-/audio-Turns auf).
|
// produziert das Brain weiter TTS, die ins offene Mikro laeuft → genau der
|
||||||
// Nur das TTS wird akustisch gestoppt, damit das Mikro ARIAs eigene Stimme
|
// "Mischmasch" (ARIA antwortet weiter waehrend ich rede). Fuer bewusstes
|
||||||
// nicht mithoert. Explizites Abbrechen laeuft ueber den Stop-Button
|
// Nicht-Abbrechen gibt es weiterhin den separaten Zwischenruf-Button (📣).
|
||||||
// (cancelRequest). Rueckgabe = false, weil kein Barge-In/Interrupt mehr.
|
|
||||||
const interruptAriaIfBusy = useCallback(() => {
|
const interruptAriaIfBusy = useCallback(() => {
|
||||||
if (audioService.isPlayingAudio()) {
|
if (audioService.isPlayingAudio()) {
|
||||||
audioService.haltAllPlayback('user startet Aufnahme (Queue-Modus, kein Abbruch)');
|
audioService.haltAllPlayback('user startet Aufnahme — Interrupt');
|
||||||
|
rvs.send('cancel_request' as any, { hard: true, source: 'voice-interrupt' });
|
||||||
|
return true;
|
||||||
}
|
}
|
||||||
return false;
|
return false;
|
||||||
}, []);
|
}, []);
|
||||||
@@ -2255,7 +2368,7 @@ const ChatScreen: React.FC = () => {
|
|||||||
// die Session auch app-seitig haben wir +2s Toleranz.
|
// die Session auch app-seitig haben wir +2s Toleranz.
|
||||||
noSpeechTimeoutMs: 0,
|
noSpeechTimeoutMs: 0,
|
||||||
endpointMs: await loadSttEndpointMs(),
|
endpointMs: await loadSttEndpointMs(),
|
||||||
hardCapMs: 300000,
|
hardCapMs: await loadMaxRecordingMs(),
|
||||||
projectId: focusedProjectIdRef.current,
|
projectId: focusedProjectIdRef.current,
|
||||||
});
|
});
|
||||||
if (!ok) {
|
if (!ok) {
|
||||||
@@ -2267,11 +2380,50 @@ const ChatScreen: React.FC = () => {
|
|||||||
return true;
|
return true;
|
||||||
}, [getCurrentLocation, interruptAriaIfBusy, scheduleStaleAudioCleanup]);
|
}, [getCurrentLocation, interruptAriaIfBusy, scheduleStaleAudioCleanup]);
|
||||||
|
|
||||||
|
// ARIA schliesst die Aufnahme SELBST — das programmatische Gegenstueck zum
|
||||||
|
// Stop-Button. Aufgerufen nach einem stillen Steuerbefehl (speak=false): der
|
||||||
|
// Befehl ist ausgefuehrt, ARIA hat die Rueckinfo (Skill-Ergebnis) und antwortet
|
||||||
|
// NICHT vorgelesen. Weil ohne TTS kein onPlaybackFinished kommt, muss der
|
||||||
|
// Aufnahme-/Konversations-Zustand hier aktiv aufgeraeumt werden, sonst bleibt
|
||||||
|
// das Ohr haengen bzw. das Aufnahme-Fenster laeuft leer weiter (Stefans
|
||||||
|
// Reproduktion: "spotify play" und das Mikro wartet trotzdem 30s).
|
||||||
|
// Unterschied zum manuellen Stop: der verwirft NICHT, sondern finalisiert die
|
||||||
|
// Aufnahme (User will seinen Satz verarbeitet haben) — hier ist der Befehl
|
||||||
|
// schon durch, ein evtl. offenes Folge-Fenster wird verworfen.
|
||||||
|
const ariaStopRecording = useCallback(async (reason: string): Promise<void> => {
|
||||||
|
converseRef.current = false;
|
||||||
|
// 1) Passiv-Lauschen: sauber beenden (cancelt den Stream selbst, startet
|
||||||
|
// KEINE neue passive Aufnahme).
|
||||||
|
if (wakeWordService.getState() === 'listening') {
|
||||||
|
await wakeWordService.exitPassiveListening('manual').catch(() => {});
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
// 2) Noch offene Streaming-Aufnahme (aktiv / Barge-In) verwerfen.
|
||||||
|
if (audioService.isStreamingRecording()) {
|
||||||
|
await audioService.cancelStreamingRecording(reason).catch(() => {});
|
||||||
|
}
|
||||||
|
// 3) Konversation beenden → zurueck aufs Wake-Word (skipPassive: kein 30s-Fenster).
|
||||||
|
if (wakeWordService.isConversing()) {
|
||||||
|
await wakeWordService.endConversation(true).catch(() => {});
|
||||||
|
} else if (!wakeWordService.isActive()) {
|
||||||
|
setWakeWordActive(false);
|
||||||
|
}
|
||||||
|
}, []);
|
||||||
|
|
||||||
// Manueller Aufnahme-Knopf — Stop. Sendet stt_stream_end an Whisper, die
|
// Manueller Aufnahme-Knopf — Stop. Sendet stt_stream_end an Whisper, die
|
||||||
// dann ihrerseits den finalen Text als stt_endpoint emittiert. aria-bridge
|
// dann ihrerseits den finalen Text als stt_endpoint emittiert. aria-bridge
|
||||||
// forwarded direkt an Brain. Im wake-word-conversing-Fall zusaetzlich
|
// forwarded direkt an Brain. Im wake-word-conversing-Fall zusaetzlich
|
||||||
// endConversation: User hat explizit gestoppt → kein Multi-Turn-Resume.
|
// endConversation: User hat explizit gestoppt → kein Multi-Turn-Resume.
|
||||||
const handleVoiceButtonStop = useCallback(async (): Promise<void> => {
|
const handleVoiceButtonStop = useCallback(async (): Promise<void> => {
|
||||||
|
// Manueller Stop = endgueltig: auch die NACH der Antwort kommende
|
||||||
|
// onPlaybackFinished darf kein 30s-Passiv-Fenster mehr oeffnen.
|
||||||
|
converseRef.current = false;
|
||||||
|
// Stop = ALLES beenden, vorhersehbar. Spricht ARIA gerade, hart stoppen +
|
||||||
|
// laufende Brain-Antwort abbrechen (sonst "sagt sie ihren letzten Satz").
|
||||||
|
if (audioService.isPlayingAudio()) {
|
||||||
|
audioService.haltAllPlayback('user stop');
|
||||||
|
rvs.send('cancel_request' as any, { hard: true, source: 'voice-stop' });
|
||||||
|
}
|
||||||
// Stop WAEHREND des passiven 30s-Lauschens ('listening'): sauber beenden
|
// Stop WAEHREND des passiven 30s-Lauschens ('listening'): sauber beenden
|
||||||
// (zurueck aufs Wake-Word), NICHT den passiven Stream neu starten.
|
// (zurueck aufs Wake-Word), NICHT den passiven Stream neu starten.
|
||||||
// exitPassiveListening cancelt den Stream selbst (via _freeMic) → es feuert
|
// exitPassiveListening cancelt den Stream selbst (via _freeMic) → es feuert
|
||||||
@@ -2371,6 +2523,10 @@ const ChatScreen: React.FC = () => {
|
|||||||
size: file.size,
|
size: file.size,
|
||||||
base64,
|
base64,
|
||||||
projectId: activePid,
|
projectId: activePid,
|
||||||
|
// Korrelation: dieselbe clientMsgId wie der Text, damit die Bridge
|
||||||
|
// die Datei genau DIESER Nachricht zuordnet — auch wenn Files (fire-
|
||||||
|
// and-forget) und Text (ACK-getrackt, bei Queue verzoegert) desyncen.
|
||||||
|
...(cmid && { clientMsgId: cmid }),
|
||||||
...(isPhoto && file.width && { width: file.width, height: file.height }),
|
...(isPhoto && file.width && { width: file.width, height: file.height }),
|
||||||
...(location && { location }),
|
...(location && { location }),
|
||||||
});
|
});
|
||||||
@@ -2440,6 +2596,24 @@ const ChatScreen: React.FC = () => {
|
|||||||
}
|
}
|
||||||
}, [inputText, pendingAttachments, sendPendingAttachments, getCtxState, setCtxState, setCtxQueue, actuallySend]);
|
}, [inputText, pendingAttachments, sendPendingAttachments, getCtxState, setCtxState, setCtxQueue, actuallySend]);
|
||||||
|
|
||||||
|
// Zwischenruf: waehrend ARIA arbeitet eine Korrektur MITTEN in den laufenden
|
||||||
|
// Turn schieben — NICHT in die Queue, KEIN Abbruch. Geht als 'interject' ueber
|
||||||
|
// RVS an die Bridge → Proxy → laufender Subprozess (greift es an der naechsten
|
||||||
|
// Tool-Grenze auf). Sichtbar nur, wenn der aktive Kontext gerade arbeitet.
|
||||||
|
const sendInterject = useCallback(() => {
|
||||||
|
const text = inputText.trim();
|
||||||
|
if (!text) return;
|
||||||
|
const activePid = focusedProjectIdRef.current;
|
||||||
|
rvs.send('interject' as any, { projectId: activePid, text });
|
||||||
|
// Lokale Bubble zur Rueckmeldung (laeuft NICHT durch Send/Queue).
|
||||||
|
setMessages(prev => capMessages([...prev, {
|
||||||
|
id: nextId(), sender: 'user', text: `📣 Zwischenruf: ${text}`,
|
||||||
|
timestamp: Date.now(), projectId: activePid,
|
||||||
|
}]));
|
||||||
|
projectDraftsRef.current = { ...projectDraftsRef.current, [activePid]: '' };
|
||||||
|
setInputText('');
|
||||||
|
}, [inputText]);
|
||||||
|
|
||||||
// --- Rendering ---
|
// --- Rendering ---
|
||||||
|
|
||||||
const renderMessage = ({ item }: { item: ChatMessage }) => {
|
const renderMessage = ({ item }: { item: ChatMessage }) => {
|
||||||
@@ -2625,6 +2799,16 @@ const ChatScreen: React.FC = () => {
|
|||||||
{att.serverPath ? '(tippen zum Laden)' : '(nicht verfuegbar)'}
|
{att.serverPath ? '(tippen zum Laden)' : '(nicht verfuegbar)'}
|
||||||
</Text>
|
</Text>
|
||||||
</TouchableOpacity>
|
</TouchableOpacity>
|
||||||
|
) : att.type === 'audio' ? (
|
||||||
|
<View style={styles.attachmentFile}>
|
||||||
|
<Text style={styles.attachmentFileIcon}>{'🎙'}</Text>
|
||||||
|
<Text style={styles.attachmentFileName} numberOfLines={1}>
|
||||||
|
{att.name || 'Sprachaufnahme'}
|
||||||
|
</Text>
|
||||||
|
{typeof item.durationS === 'number' && item.durationS > 0 ? (
|
||||||
|
<Text style={styles.attachmentFileSize}>{formatDur(item.durationS)}</Text>
|
||||||
|
) : null}
|
||||||
|
</View>
|
||||||
) : (
|
) : (
|
||||||
<TouchableOpacity
|
<TouchableOpacity
|
||||||
style={styles.attachmentFile}
|
style={styles.attachmentFile}
|
||||||
@@ -2876,7 +3060,7 @@ const ChatScreen: React.FC = () => {
|
|||||||
</TouchableOpacity>
|
</TouchableOpacity>
|
||||||
</View>
|
</View>
|
||||||
|
|
||||||
{/* Service-Status Banner (Gamebox: F5-TTS / Whisper Lade-Status) */}
|
{/* Service-Status Banner (AI-Box: F5-TTS / Whisper Lade-Status) */}
|
||||||
{(() => {
|
{(() => {
|
||||||
const entries = Object.entries(serviceStatus);
|
const entries = Object.entries(serviceStatus);
|
||||||
if (entries.length === 0 || serviceBannerDismissed) return null;
|
if (entries.length === 0 || serviceBannerDismissed) return null;
|
||||||
@@ -3214,9 +3398,19 @@ const ChatScreen: React.FC = () => {
|
|||||||
|
|
||||||
{/* Senden oder Sprache */}
|
{/* Senden oder Sprache */}
|
||||||
{inputText.trim() || pendingAttachments.length > 0 ? (
|
{inputText.trim() || pendingAttachments.length > 0 ? (
|
||||||
<TouchableOpacity style={styles.sendButton} onPress={sendTextMessage}>
|
<>
|
||||||
<Text style={styles.sendIcon}>{'\u2B06\uFE0F'}</Text>
|
{/* Zwischenruf: nur wenn ARIA im aktiven Kontext gerade arbeitet und
|
||||||
</TouchableOpacity>
|
Text da ist. Schiebt die Korrektur in den laufenden Turn statt
|
||||||
|
sie anzustellen. */}
|
||||||
|
{inputText.trim() && (agentActivityByCtx[focusedProjectId]?.activity || 'idle') !== 'idle' ? (
|
||||||
|
<TouchableOpacity style={styles.interjectButton} onPress={sendInterject} accessibilityLabel="Zwischenruf">
|
||||||
|
<Text style={styles.interjectIcon}>{'\uD83D\uDCE3'}</Text>
|
||||||
|
</TouchableOpacity>
|
||||||
|
) : null}
|
||||||
|
<TouchableOpacity style={styles.sendButton} onPress={sendTextMessage}>
|
||||||
|
<Text style={styles.sendIcon}>{'\u2B06\uFE0F'}</Text>
|
||||||
|
</TouchableOpacity>
|
||||||
|
</>
|
||||||
) : (
|
) : (
|
||||||
<>
|
<>
|
||||||
<VoiceButton
|
<VoiceButton
|
||||||
@@ -3734,6 +3928,18 @@ const styles = StyleSheet.create({
|
|||||||
sendIcon: {
|
sendIcon: {
|
||||||
fontSize: 18,
|
fontSize: 18,
|
||||||
},
|
},
|
||||||
|
interjectButton: {
|
||||||
|
width: 40,
|
||||||
|
height: 40,
|
||||||
|
borderRadius: 20,
|
||||||
|
backgroundColor: '#FF9500', // orange — Zwischenruf, klar vom blauen Senden getrennt
|
||||||
|
alignItems: 'center',
|
||||||
|
justifyContent: 'center',
|
||||||
|
marginRight: 6,
|
||||||
|
},
|
||||||
|
interjectIcon: {
|
||||||
|
fontSize: 18,
|
||||||
|
},
|
||||||
wakeWordBtn: {
|
wakeWordBtn: {
|
||||||
width: 32,
|
width: 32,
|
||||||
height: 32,
|
height: 32,
|
||||||
|
|||||||
@@ -63,14 +63,16 @@ import {
|
|||||||
VAD_SILENCE_MIN_SEC,
|
VAD_SILENCE_MIN_SEC,
|
||||||
VAD_SILENCE_MAX_SEC,
|
VAD_SILENCE_MAX_SEC,
|
||||||
VAD_SILENCE_STORAGE_KEY,
|
VAD_SILENCE_STORAGE_KEY,
|
||||||
CONV_WINDOW_DEFAULT_SEC,
|
STT_ENDPOINT_DEFAULT_MS,
|
||||||
CONV_WINDOW_MIN_SEC,
|
STT_ENDPOINT_MIN_MS,
|
||||||
CONV_WINDOW_MAX_SEC,
|
STT_ENDPOINT_MAX_MS,
|
||||||
CONV_WINDOW_STORAGE_KEY,
|
STT_ENDPOINT_STORAGE_KEY,
|
||||||
MAX_RECORDING_DEFAULT_SEC,
|
MAX_RECORDING_DEFAULT_SEC,
|
||||||
MAX_RECORDING_MIN_SEC,
|
MAX_RECORDING_MIN_SEC,
|
||||||
MAX_RECORDING_MAX_SEC,
|
MAX_RECORDING_MAX_SEC,
|
||||||
MAX_RECORDING_STORAGE_KEY,
|
MAX_RECORDING_STORAGE_KEY,
|
||||||
|
loadBargeInEnabled,
|
||||||
|
saveBargeInEnabled,
|
||||||
VAD_SILENCE_DB_DEFAULT,
|
VAD_SILENCE_DB_DEFAULT,
|
||||||
VAD_SILENCE_DB_MIN,
|
VAD_SILENCE_DB_MIN,
|
||||||
VAD_SILENCE_DB_MAX,
|
VAD_SILENCE_DB_MAX,
|
||||||
@@ -110,9 +112,10 @@ import wakeWordService, {
|
|||||||
WAKE_THRESHOLD_MAX,
|
WAKE_THRESHOLD_MAX,
|
||||||
loadWakeThreshold,
|
loadWakeThreshold,
|
||||||
saveWakeThreshold,
|
saveWakeThreshold,
|
||||||
PASSIVE_LISTEN_DEFAULT_MS,
|
loadBgWakeEnabled,
|
||||||
loadPassiveListenMs,
|
saveBgWakeEnabled,
|
||||||
savePassiveListenMs,
|
loadWakeConfirmEnabled,
|
||||||
|
saveWakeConfirmEnabled,
|
||||||
} from '../services/wakeword';
|
} from '../services/wakeword';
|
||||||
import ModeSelector from '../components/ModeSelector';
|
import ModeSelector from '../components/ModeSelector';
|
||||||
import QRScanner from '../components/QRScanner';
|
import QRScanner from '../components/QRScanner';
|
||||||
@@ -195,8 +198,12 @@ const SettingsScreen: React.FC = () => {
|
|||||||
const [ttsEnabled, setTtsEnabled] = useState(true);
|
const [ttsEnabled, setTtsEnabled] = useState(true);
|
||||||
const [ttsPrerollSec, setTtsPrerollSec] = useState<number>(TTS_PREROLL_DEFAULT_SEC);
|
const [ttsPrerollSec, setTtsPrerollSec] = useState<number>(TTS_PREROLL_DEFAULT_SEC);
|
||||||
const [vadSilenceSec, setVadSilenceSec] = useState<number>(VAD_SILENCE_DEFAULT_SEC);
|
const [vadSilenceSec, setVadSilenceSec] = useState<number>(VAD_SILENCE_DEFAULT_SEC);
|
||||||
const [convWindowSec, setConvWindowSec] = useState<number>(CONV_WINDOW_DEFAULT_SEC);
|
// Aktive Streaming-Pausen-Toleranz (STT_ENDPOINT) — der "Stille-Toleranz"-Regler
|
||||||
|
// steuert jetzt DIESEN Wert (der alte vadSilenceSec war der tote Legacy-dB-Pfad).
|
||||||
|
const [sttEndpointSec, setSttEndpointSec] = useState<number>(STT_ENDPOINT_DEFAULT_MS / 1000);
|
||||||
const [maxRecordingSec, setMaxRecordingSec] = useState<number>(MAX_RECORDING_DEFAULT_SEC);
|
const [maxRecordingSec, setMaxRecordingSec] = useState<number>(MAX_RECORDING_DEFAULT_SEC);
|
||||||
|
// Barge-in: ARIA waehrend ihrer Antwort unterbrechen duerfen. Default aus (Halb-Duplex).
|
||||||
|
const [bargeIn, setBargeIn] = useState<boolean>(false);
|
||||||
// null = automatisch (adaptive Baseline), sonst manueller dB-Override
|
// null = automatisch (adaptive Baseline), sonst manueller dB-Override
|
||||||
const [vadSilenceDb, setVadSilenceDb] = useState<number | null>(null);
|
const [vadSilenceDb, setVadSilenceDb] = useState<number | null>(null);
|
||||||
const [showVadInfo, setShowVadInfo] = useState(false);
|
const [showVadInfo, setShowVadInfo] = useState(false);
|
||||||
@@ -209,7 +216,10 @@ const SettingsScreen: React.FC = () => {
|
|||||||
const [wakeStatus, setWakeStatus] = useState<string>('');
|
const [wakeStatus, setWakeStatus] = useState<string>('');
|
||||||
const [wakeReadySound, setWakeReadySound] = useState<boolean>(true);
|
const [wakeReadySound, setWakeReadySound] = useState<boolean>(true);
|
||||||
const [wakeThreshold, setWakeThreshold] = useState<number>(WAKE_THRESHOLD_DEFAULT);
|
const [wakeThreshold, setWakeThreshold] = useState<number>(WAKE_THRESHOLD_DEFAULT);
|
||||||
const [passiveSec, setPassiveSec] = useState<number>(Math.round(PASSIVE_LISTEN_DEFAULT_MS / 1000));
|
// Hintergrund-Wake: auch bei gesperrtem Bildschirm auf das Wake-Wort hoeren. Default aus.
|
||||||
|
const [bgWake, setBgWake] = useState<boolean>(false);
|
||||||
|
// Wake-Wort per Voxtral bestaetigen (gegen Musik-Fehltrigger). Default aus.
|
||||||
|
const [wakeConfirm, setWakeConfirm] = useState<boolean>(false);
|
||||||
const [editingPath, setEditingPath] = useState(false);
|
const [editingPath, setEditingPath] = useState(false);
|
||||||
const [xttsVoice, setXttsVoice] = useState('');
|
const [xttsVoice, setXttsVoice] = useState('');
|
||||||
const [loadingVoice, setLoadingVoice] = useState<string | null>(null);
|
const [loadingVoice, setLoadingVoice] = useState<string | null>(null);
|
||||||
@@ -298,11 +308,11 @@ const SettingsScreen: React.FC = () => {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
});
|
});
|
||||||
AsyncStorage.getItem(CONV_WINDOW_STORAGE_KEY).then(saved => {
|
AsyncStorage.getItem(STT_ENDPOINT_STORAGE_KEY).then(saved => {
|
||||||
if (saved != null) {
|
if (saved != null) {
|
||||||
const n = parseFloat(saved);
|
const n = parseInt(saved, 10);
|
||||||
if (isFinite(n) && n >= CONV_WINDOW_MIN_SEC && n <= CONV_WINDOW_MAX_SEC) {
|
if (isFinite(n) && n >= STT_ENDPOINT_MIN_MS && n <= STT_ENDPOINT_MAX_MS) {
|
||||||
setConvWindowSec(n);
|
setSttEndpointSec(n / 1000);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
});
|
});
|
||||||
@@ -314,6 +324,7 @@ const SettingsScreen: React.FC = () => {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
});
|
});
|
||||||
|
loadBargeInEnabled().then(setBargeIn).catch(() => {});
|
||||||
AsyncStorage.getItem(VAD_SILENCE_DB_OVERRIDE_KEY).then(saved => {
|
AsyncStorage.getItem(VAD_SILENCE_DB_OVERRIDE_KEY).then(saved => {
|
||||||
if (saved != null && saved !== '') {
|
if (saved != null && saved !== '') {
|
||||||
const n = parseFloat(saved);
|
const n = parseFloat(saved);
|
||||||
@@ -333,7 +344,8 @@ const SettingsScreen: React.FC = () => {
|
|||||||
});
|
});
|
||||||
isWakeReadySoundEnabled().then(setWakeReadySound);
|
isWakeReadySoundEnabled().then(setWakeReadySound);
|
||||||
loadWakeThreshold().then(setWakeThreshold).catch(() => {});
|
loadWakeThreshold().then(setWakeThreshold).catch(() => {});
|
||||||
loadPassiveListenMs().then(ms => setPassiveSec(Math.round(ms / 1000))).catch(() => {});
|
loadBgWakeEnabled().then(setBgWake).catch(() => {});
|
||||||
|
loadWakeConfirmEnabled().then(setWakeConfirm).catch(() => {});
|
||||||
updateService.getApkCacheSize().then(setApkCacheInfo).catch(() => {});
|
updateService.getApkCacheSize().then(setApkCacheInfo).catch(() => {});
|
||||||
audioService.getTtsCacheSize().then(setTtsCacheInfo).catch(() => {});
|
audioService.getTtsCacheSize().then(setTtsCacheInfo).catch(() => {});
|
||||||
AsyncStorage.getItem('aria_xtts_voice').then(saved => {
|
AsyncStorage.getItem('aria_xtts_voice').then(saved => {
|
||||||
@@ -1651,72 +1663,56 @@ const SettingsScreen: React.FC = () => {
|
|||||||
{currentSection === 'voice_input' && (<>
|
{currentSection === 'voice_input' && (<>
|
||||||
<Text style={styles.sectionTitle}>Spracheingabe</Text>
|
<Text style={styles.sectionTitle}>Spracheingabe</Text>
|
||||||
<View style={styles.card}>
|
<View style={styles.card}>
|
||||||
<Text style={styles.toggleLabel}>Stille-Toleranz</Text>
|
<View style={styles.toggleRow}>
|
||||||
|
<View style={styles.toggleInfo}>
|
||||||
|
<Text style={styles.toggleLabel}>Barge-in (unterbrechen)</Text>
|
||||||
|
<Text style={styles.toggleHint}>
|
||||||
|
AUS (empfohlen): ARIA spricht ihre Antwort ZU ENDE, dann geht das
|
||||||
|
Mikro auf — sauber, kein Selbst-Echo, du hoerst sie ganz. AN: du
|
||||||
|
kannst sie waehrend des Sprechens per Wake-Wort unterbrechen.
|
||||||
|
</Text>
|
||||||
|
</View>
|
||||||
|
<Switch
|
||||||
|
value={bargeIn}
|
||||||
|
onValueChange={(v) => { setBargeIn(v); saveBargeInEnabled(v).catch(() => {}); }}
|
||||||
|
trackColor={{ false: '#2A2A3E', true: '#0096FF' }}
|
||||||
|
thumbColor={bargeIn ? '#FFFFFF' : '#666680'}
|
||||||
|
/>
|
||||||
|
</View>
|
||||||
|
|
||||||
|
<Text style={[styles.toggleLabel, {marginTop: 20}]}>Stille-Toleranz</Text>
|
||||||
<Text style={styles.toggleHint}>
|
<Text style={styles.toggleHint}>
|
||||||
Wie lange du eine Sprechpause machen darfst, bevor die Aufnahme
|
Wie lange du eine Sprechpause machen darfst, bevor die Aufnahme
|
||||||
automatisch beendet und gesendet wird. Hoeher = mehr Zeit zum
|
automatisch beendet und gesendet wird. Hoeher = mehr Zeit zum
|
||||||
Nachdenken; niedriger = schnelleres Senden.
|
Nachdenken (z.B. im Auto); niedriger = schnelleres Senden.
|
||||||
Default: {VAD_SILENCE_DEFAULT_SEC.toFixed(1)}s.
|
Default: {(STT_ENDPOINT_DEFAULT_MS / 1000).toFixed(1)}s.
|
||||||
</Text>
|
</Text>
|
||||||
<View style={styles.prerollRow}>
|
<View style={styles.prerollRow}>
|
||||||
<TouchableOpacity
|
<TouchableOpacity
|
||||||
style={styles.prerollButton}
|
style={styles.prerollButton}
|
||||||
onPress={() => {
|
onPress={() => {
|
||||||
const next = Math.max(VAD_SILENCE_MIN_SEC, Math.round((vadSilenceSec - 0.5) * 10) / 10);
|
const next = Math.max(STT_ENDPOINT_MIN_MS / 1000, Math.round((sttEndpointSec - 0.5) * 10) / 10);
|
||||||
setVadSilenceSec(next);
|
setSttEndpointSec(next);
|
||||||
AsyncStorage.setItem(VAD_SILENCE_STORAGE_KEY, String(next));
|
AsyncStorage.setItem(STT_ENDPOINT_STORAGE_KEY, String(Math.round(next * 1000)));
|
||||||
}}
|
}}
|
||||||
disabled={vadSilenceSec <= VAD_SILENCE_MIN_SEC}
|
disabled={sttEndpointSec <= STT_ENDPOINT_MIN_MS / 1000}
|
||||||
>
|
>
|
||||||
<Text style={styles.prerollButtonText}>−0.5</Text>
|
<Text style={styles.prerollButtonText}>−0.5</Text>
|
||||||
</TouchableOpacity>
|
</TouchableOpacity>
|
||||||
<Text style={styles.prerollValue}>{vadSilenceSec.toFixed(1)} s</Text>
|
<Text style={styles.prerollValue}>{sttEndpointSec.toFixed(1)} s</Text>
|
||||||
<TouchableOpacity
|
<TouchableOpacity
|
||||||
style={styles.prerollButton}
|
style={styles.prerollButton}
|
||||||
onPress={() => {
|
onPress={() => {
|
||||||
const next = Math.min(VAD_SILENCE_MAX_SEC, Math.round((vadSilenceSec + 0.5) * 10) / 10);
|
const next = Math.min(STT_ENDPOINT_MAX_MS / 1000, Math.round((sttEndpointSec + 0.5) * 10) / 10);
|
||||||
setVadSilenceSec(next);
|
setSttEndpointSec(next);
|
||||||
AsyncStorage.setItem(VAD_SILENCE_STORAGE_KEY, String(next));
|
AsyncStorage.setItem(STT_ENDPOINT_STORAGE_KEY, String(Math.round(next * 1000)));
|
||||||
}}
|
}}
|
||||||
disabled={vadSilenceSec >= VAD_SILENCE_MAX_SEC}
|
disabled={sttEndpointSec >= STT_ENDPOINT_MAX_MS / 1000}
|
||||||
>
|
>
|
||||||
<Text style={styles.prerollButtonText}>+0.5</Text>
|
<Text style={styles.prerollButtonText}>+0.5</Text>
|
||||||
</TouchableOpacity>
|
</TouchableOpacity>
|
||||||
</View>
|
</View>
|
||||||
|
|
||||||
<Text style={[styles.toggleLabel, {marginTop: 24}]}>Konversations-Fenster</Text>
|
|
||||||
<Text style={styles.toggleHint}>
|
|
||||||
Im Gespraechsmodus (Ohr-Button): nach ARIA's Antwort hast du so lange
|
|
||||||
Zeit, weiter zu sprechen, bevor die Konversation automatisch beendet wird.
|
|
||||||
Sprichst du nichts → Mikrofon zu.
|
|
||||||
Default: {CONV_WINDOW_DEFAULT_SEC.toFixed(1)}s.
|
|
||||||
</Text>
|
|
||||||
<View style={styles.prerollRow}>
|
|
||||||
<TouchableOpacity
|
|
||||||
style={styles.prerollButton}
|
|
||||||
onPress={() => {
|
|
||||||
const next = Math.max(CONV_WINDOW_MIN_SEC, Math.round((convWindowSec - 1) * 10) / 10);
|
|
||||||
setConvWindowSec(next);
|
|
||||||
AsyncStorage.setItem(CONV_WINDOW_STORAGE_KEY, String(next));
|
|
||||||
}}
|
|
||||||
disabled={convWindowSec <= CONV_WINDOW_MIN_SEC}
|
|
||||||
>
|
|
||||||
<Text style={styles.prerollButtonText}>−1</Text>
|
|
||||||
</TouchableOpacity>
|
|
||||||
<Text style={styles.prerollValue}>{convWindowSec.toFixed(0)} s</Text>
|
|
||||||
<TouchableOpacity
|
|
||||||
style={styles.prerollButton}
|
|
||||||
onPress={() => {
|
|
||||||
const next = Math.min(CONV_WINDOW_MAX_SEC, Math.round((convWindowSec + 1) * 10) / 10);
|
|
||||||
setConvWindowSec(next);
|
|
||||||
AsyncStorage.setItem(CONV_WINDOW_STORAGE_KEY, String(next));
|
|
||||||
}}
|
|
||||||
disabled={convWindowSec >= CONV_WINDOW_MAX_SEC}
|
|
||||||
>
|
|
||||||
<Text style={styles.prerollButtonText}>+1</Text>
|
|
||||||
</TouchableOpacity>
|
|
||||||
</View>
|
|
||||||
|
|
||||||
<Text style={[styles.toggleLabel, {marginTop: 24}]}>Maximale Aufnahmedauer</Text>
|
<Text style={[styles.toggleLabel, {marginTop: 24}]}>Maximale Aufnahmedauer</Text>
|
||||||
<Text style={styles.toggleHint}>
|
<Text style={styles.toggleHint}>
|
||||||
Notbremse: nach so vielen Minuten wird die Aufnahme automatisch beendet,
|
Notbremse: nach so vielen Minuten wird die Aufnahme automatisch beendet,
|
||||||
@@ -1749,56 +1745,10 @@ const SettingsScreen: React.FC = () => {
|
|||||||
</TouchableOpacity>
|
</TouchableOpacity>
|
||||||
</View>
|
</View>
|
||||||
|
|
||||||
<View style={{flexDirection: 'row', alignItems: 'center', marginTop: 24, gap: 8}}>
|
{/* "Stille-Pegel (dB)"-Regler entfernt: der aktive Streaming-STT nutzt
|
||||||
<Text style={styles.toggleLabel}>Stille-Pegel (dB)</Text>
|
einen adaptiven Rausch-Boden (automatisch), ein manueller dB-Wert war
|
||||||
<TouchableOpacity onPress={() => setShowVadInfo(true)} style={styles.infoBtn}>
|
wirkungslos. Rauschen-als-Wort verhindert das STT-Modell selbst
|
||||||
<Text style={styles.infoBtnText}>i</Text>
|
(no_speech_prob-Filter), nicht die dB-Schwelle. */}
|
||||||
</TouchableOpacity>
|
|
||||||
</View>
|
|
||||||
<Text style={styles.toggleHint}>
|
|
||||||
Welcher Mikro-Pegel als "Stille" gilt. Standard: automatisch (Baseline aus
|
|
||||||
den ersten 500ms). Manuell setzen wenn Auto nicht zuverlaessig greift.
|
|
||||||
</Text>
|
|
||||||
<View style={styles.prerollRow}>
|
|
||||||
<TouchableOpacity
|
|
||||||
style={styles.prerollButton}
|
|
||||||
onPress={() => {
|
|
||||||
const next = vadSilenceDb == null
|
|
||||||
? VAD_SILENCE_DB_DEFAULT - 1
|
|
||||||
: Math.max(VAD_SILENCE_DB_MIN, vadSilenceDb - 1);
|
|
||||||
setVadSilenceDb(next);
|
|
||||||
AsyncStorage.setItem(VAD_SILENCE_DB_OVERRIDE_KEY, String(next));
|
|
||||||
}}
|
|
||||||
>
|
|
||||||
<Text style={styles.prerollButtonText}>−1</Text>
|
|
||||||
</TouchableOpacity>
|
|
||||||
<Text style={styles.prerollValue}>
|
|
||||||
{vadSilenceDb == null ? 'auto' : `${vadSilenceDb} dB`}
|
|
||||||
</Text>
|
|
||||||
<TouchableOpacity
|
|
||||||
style={styles.prerollButton}
|
|
||||||
onPress={() => {
|
|
||||||
const next = vadSilenceDb == null
|
|
||||||
? VAD_SILENCE_DB_DEFAULT + 1
|
|
||||||
: Math.min(VAD_SILENCE_DB_MAX, vadSilenceDb + 1);
|
|
||||||
setVadSilenceDb(next);
|
|
||||||
AsyncStorage.setItem(VAD_SILENCE_DB_OVERRIDE_KEY, String(next));
|
|
||||||
}}
|
|
||||||
>
|
|
||||||
<Text style={styles.prerollButtonText}>+1</Text>
|
|
||||||
</TouchableOpacity>
|
|
||||||
</View>
|
|
||||||
{vadSilenceDb != null && (
|
|
||||||
<TouchableOpacity
|
|
||||||
onPress={() => {
|
|
||||||
setVadSilenceDb(null);
|
|
||||||
AsyncStorage.removeItem(VAD_SILENCE_DB_OVERRIDE_KEY);
|
|
||||||
}}
|
|
||||||
style={{alignSelf: 'center', marginTop: 8, paddingVertical: 6, paddingHorizontal: 12}}
|
|
||||||
>
|
|
||||||
<Text style={{color: '#0096FF', fontSize: 13}}>↻ Auf automatisch zuruecksetzen</Text>
|
|
||||||
</TouchableOpacity>
|
|
||||||
)}
|
|
||||||
</View>
|
</View>
|
||||||
|
|
||||||
<Modal
|
<Modal
|
||||||
@@ -1954,38 +1904,58 @@ const SettingsScreen: React.FC = () => {
|
|||||||
/>
|
/>
|
||||||
</View>
|
</View>
|
||||||
|
|
||||||
<Text style={[styles.toggleLabel, {marginTop: 20}]}>Weiterreden-Fenster (Gespraech)</Text>
|
<View style={[styles.toggleRow, {marginTop: 20, borderTopWidth: 1, borderTopColor: '#1E1E2E', paddingTop: 16}]}>
|
||||||
<Text style={styles.toggleHint}>
|
<View style={styles.toggleInfo}>
|
||||||
Nach einer gesprochenen ARIA-Antwort kannst du so lange einfach
|
<Text style={styles.toggleLabel}>Auch bei gesperrtem Bildschirm zuhören</Text>
|
||||||
weiterreden — ohne Wake-Word — bevor zurueck aufs Wake-Word geschaltet
|
<Text style={styles.toggleHint}>
|
||||||
wird. Reine Steuerbefehle (z.B. „nächster Titel") beenden sofort.
|
AUS (empfohlen): das Wake-Wort greift nur, wenn die App offen ist —
|
||||||
Default: {Math.round(PASSIVE_LISTEN_DEFAULT_MS / 1000)}s.
|
im Hintergrund sind die meisten „Trigger" Fehlalarme (TV, Husten).
|
||||||
</Text>
|
AN: ARIA hört auch bei gesperrtem Bildschirm / im Hintergrund auf
|
||||||
<View style={styles.prerollRow}>
|
„{KEYWORD_LABELS[wakeKeyword as keyof typeof KEYWORD_LABELS] || wakeKeyword}" — mehr Fehlauslöser möglich.
|
||||||
<TouchableOpacity
|
</Text>
|
||||||
style={styles.prerollButton}
|
</View>
|
||||||
onPress={() => {
|
<Switch
|
||||||
const next = Math.max(10, passiveSec - 5);
|
value={bgWake}
|
||||||
setPassiveSec(next);
|
onValueChange={(val) => {
|
||||||
savePassiveListenMs(next * 1000);
|
setBgWake(val);
|
||||||
|
saveBgWakeEnabled(val).catch(() => {});
|
||||||
|
wakeWordService.setBgWakeEnabled(val);
|
||||||
}}
|
}}
|
||||||
disabled={passiveSec <= 10}
|
trackColor={{ false: '#2A2A3E', true: '#0096FF' }}
|
||||||
>
|
thumbColor={bgWake ? '#FFFFFF' : '#666680'}
|
||||||
<Text style={styles.prerollButtonText}>−5</Text>
|
/>
|
||||||
</TouchableOpacity>
|
|
||||||
<Text style={styles.prerollValue}>{passiveSec} s</Text>
|
|
||||||
<TouchableOpacity
|
|
||||||
style={styles.prerollButton}
|
|
||||||
onPress={() => {
|
|
||||||
const next = Math.min(60, passiveSec + 5);
|
|
||||||
setPassiveSec(next);
|
|
||||||
savePassiveListenMs(next * 1000);
|
|
||||||
}}
|
|
||||||
disabled={passiveSec >= 60}
|
|
||||||
>
|
|
||||||
<Text style={styles.prerollButtonText}>+5</Text>
|
|
||||||
</TouchableOpacity>
|
|
||||||
</View>
|
</View>
|
||||||
|
|
||||||
|
<View style={[styles.toggleRow, {marginTop: 20, borderTopWidth: 1, borderTopColor: '#1E1E2E', paddingTop: 16}]}>
|
||||||
|
<View style={styles.toggleInfo}>
|
||||||
|
<Text style={styles.toggleLabel}>Wake-Wort per Voxtral bestätigen</Text>
|
||||||
|
<Text style={styles.toggleHint}>
|
||||||
|
Gegen Musik-Fehltrigger: nach „{KEYWORD_LABELS[wakeKeyword as keyof typeof KEYWORD_LABELS] || wakeKeyword}"
|
||||||
|
prüft Voxtral kurz nach, ob's wirklich das Wake-Wort war (oder nur
|
||||||
|
Musik/TV) — erst dann Gong + Mikro. Kostet ~0,5–1 s Extra vor dem
|
||||||
|
Gong. Empfohlen zusammen mit Hintergrund-Zuhören.
|
||||||
|
</Text>
|
||||||
|
</View>
|
||||||
|
<Switch
|
||||||
|
value={wakeConfirm}
|
||||||
|
onValueChange={(val) => {
|
||||||
|
setWakeConfirm(val);
|
||||||
|
saveWakeConfirmEnabled(val).catch(() => {});
|
||||||
|
wakeWordService.setWakeConfirmEnabled(val);
|
||||||
|
}}
|
||||||
|
trackColor={{ false: '#2A2A3E', true: '#0096FF' }}
|
||||||
|
thumbColor={wakeConfirm ? '#FFFFFF' : '#666680'}
|
||||||
|
/>
|
||||||
|
</View>
|
||||||
|
|
||||||
|
<Text style={[styles.toggleLabel, {marginTop: 20}]}>Weiterreden nach der Antwort</Text>
|
||||||
|
<Text style={styles.toggleHint}>
|
||||||
|
Nach einer gesprochenen ARIA-Antwort geht das Mikro auf — du kannst ohne
|
||||||
|
Wake-Word weiterreden. Fängst du nicht innerhalb der „Stille-Toleranz"
|
||||||
|
(Sektion Spracheingabe) an, geht's zurück aufs Wake-Word. Reine
|
||||||
|
Steuerbefehle beenden sofort. Ein separates Zeitfenster gibt es nicht
|
||||||
|
mehr — es zählt überall derselbe Stille-Wert.
|
||||||
|
</Text>
|
||||||
</View>
|
</View>
|
||||||
</>)}
|
</>)}
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,104 @@
|
|||||||
|
/**
|
||||||
|
* ariaView — Empfaenger der von ARIA komponierten RAEUMLICHEN Ansichten (M1).
|
||||||
|
*
|
||||||
|
* Fluss: ARIA ruft im Brain `present_view` → Brain-Event `aria_view` → Bridge →
|
||||||
|
* RVS `aria_view` → hier gepuffert → WorkspaceCanvas rendert Orb + Karten, die
|
||||||
|
* auf der Flaeche materialisieren.
|
||||||
|
*
|
||||||
|
* Der Service haelt pro Projekt die AKTUELLE View-Spec, damit eine spaet
|
||||||
|
* gemountete Canvas-Kachel sofort den Ist-Stand bekommt. Muster wie
|
||||||
|
* services/codeFile.ts (Singleton, rvs.onMessage).
|
||||||
|
*
|
||||||
|
* Die Karten-Typen sind bewusst offen (string), damit spaetere Renderer (vnc,
|
||||||
|
* chart, file …) ohne Service-Aenderung dazukommen. Der jeweilige Client-Renderer
|
||||||
|
* entscheidet, was er mit einem unbekannten Typ macht (i.d.R. ignorieren).
|
||||||
|
*/
|
||||||
|
|
||||||
|
import rvs, { RVSMessage } from './rvs';
|
||||||
|
|
||||||
|
export type OrbState = 'idle' | 'listening' | 'thinking' | 'speaking' | 'working';
|
||||||
|
|
||||||
|
export interface ViewMarker {
|
||||||
|
lat: number;
|
||||||
|
lon: number;
|
||||||
|
label?: string;
|
||||||
|
}
|
||||||
|
|
||||||
|
export interface ViewCard {
|
||||||
|
type: 'text' | 'image' | 'map' | 'code' | 'list' | string;
|
||||||
|
title?: string;
|
||||||
|
md?: string; // text/list
|
||||||
|
src?: string; // image
|
||||||
|
markers?: ViewMarker[]; // map
|
||||||
|
path?: string; // code
|
||||||
|
lang?: string; // code
|
||||||
|
// Zukuenftige Kartenfelder ohne Service-Aenderung:
|
||||||
|
[k: string]: any;
|
||||||
|
}
|
||||||
|
|
||||||
|
export interface ViewSpec {
|
||||||
|
cards: ViewCard[];
|
||||||
|
orb?: OrbState;
|
||||||
|
title?: string;
|
||||||
|
}
|
||||||
|
|
||||||
|
export interface AriaView {
|
||||||
|
projectId: string;
|
||||||
|
view: ViewSpec;
|
||||||
|
clientMsgId?: string;
|
||||||
|
ts: number;
|
||||||
|
}
|
||||||
|
|
||||||
|
type ViewSub = (v: AriaView) => void;
|
||||||
|
|
||||||
|
class AriaViewService {
|
||||||
|
private views = new Map<string, AriaView>();
|
||||||
|
private subs: ViewSub[] = [];
|
||||||
|
|
||||||
|
constructor() {
|
||||||
|
rvs.onMessage((m) => this.onMessage(m));
|
||||||
|
}
|
||||||
|
|
||||||
|
private onMessage(m: RVSMessage): void {
|
||||||
|
if (m.type !== 'aria_view') return;
|
||||||
|
const p = (m.payload || {}) as any;
|
||||||
|
const raw = (p.view || {}) as any;
|
||||||
|
const cards: ViewCard[] = Array.isArray(raw.cards) ? raw.cards : [];
|
||||||
|
if (cards.length === 0) return; // leere Ansicht ignorieren
|
||||||
|
const view: ViewSpec = {
|
||||||
|
cards,
|
||||||
|
orb: raw.orb || 'speaking',
|
||||||
|
title: raw.title || '',
|
||||||
|
};
|
||||||
|
const projectId: string = p.projectId || '';
|
||||||
|
const entry: AriaView = {
|
||||||
|
projectId,
|
||||||
|
view,
|
||||||
|
clientMsgId: p.clientMsgId || '',
|
||||||
|
ts: Date.now(),
|
||||||
|
};
|
||||||
|
this.views.set(projectId, entry);
|
||||||
|
this.subs.forEach((cb) => {
|
||||||
|
try { cb(entry); } catch {}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Aktuelle Ansicht eines Projekts (leer = Hauptchat). */
|
||||||
|
getView(projectId: string): AriaView | undefined {
|
||||||
|
return this.views.get(projectId || '');
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Registriert einen Listener fuer neue Ansichten. */
|
||||||
|
subscribe(cb: ViewSub): () => void {
|
||||||
|
this.subs.push(cb);
|
||||||
|
return () => { this.subs = this.subs.filter((s) => s !== cb); };
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Ansicht eines Projekts verwerfen (z.B. wenn der User sie wegwischt). */
|
||||||
|
clear(projectId: string): void {
|
||||||
|
this.views.delete(projectId || '');
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
const ariaView = new AriaViewService();
|
||||||
|
export default ariaView;
|
||||||
+106
-27
@@ -143,23 +143,100 @@ export const VAD_SILENCE_MIN_SEC = 1.0;
|
|||||||
export const VAD_SILENCE_MAX_SEC = 8.0;
|
export const VAD_SILENCE_MAX_SEC = 8.0;
|
||||||
export const VAD_SILENCE_STORAGE_KEY = 'aria_vad_silence_sec';
|
export const VAD_SILENCE_STORAGE_KEY = 'aria_vad_silence_sec';
|
||||||
|
|
||||||
// Konversations-Fenster (in Sekunden) — nach ARIA's Antwort hat der User so
|
|
||||||
// lange Zeit, im Gespraechsmodus weiter zu sprechen, ohne dass die Konversation
|
|
||||||
// beendet wird. Sprichst du im Fenster nichts → Konversation aus.
|
|
||||||
export const CONV_WINDOW_DEFAULT_SEC = 8.0;
|
|
||||||
export const CONV_WINDOW_MIN_SEC = 3.0;
|
|
||||||
export const CONV_WINDOW_MAX_SEC = 20.0;
|
|
||||||
export const CONV_WINDOW_STORAGE_KEY = 'aria_conv_window_sec';
|
|
||||||
|
|
||||||
// STT-Endpoint (ms Stille bis "fertig gesprochen"). Zu kurz = schneidet mitten
|
// STT-Endpoint (ms Stille bis "fertig gesprochen"). Zu kurz = schneidet mitten
|
||||||
// im Satz ab, besonders im Auto wo man mit Pausen spricht (Reproduktion: die
|
// im Satz ab, besonders im Auto oder wenn man zum Nachdenken pausiert. 1500 war
|
||||||
// 11.8s-Frage wurde bei "…ohne dass ein" gekappt). 1500 war zu aggressiv;
|
// zu aggressiv; 2400 default, bis 8s hoch stellbar (Denkpausen). In den Settings
|
||||||
// 2400 default, im Auto ggf. hoeher. Konfigurierbar in den Settings.
|
// unter "Stille-Toleranz" konfigurierbar.
|
||||||
export const STT_ENDPOINT_DEFAULT_MS = 2400;
|
export const STT_ENDPOINT_DEFAULT_MS = 2400;
|
||||||
export const STT_ENDPOINT_MIN_MS = 1000;
|
export const STT_ENDPOINT_MIN_MS = 1000;
|
||||||
export const STT_ENDPOINT_MAX_MS = 4000;
|
export const STT_ENDPOINT_MAX_MS = 8000; // bis 8s: genug Zeit zum Ueberlegen
|
||||||
export const STT_ENDPOINT_STORAGE_KEY = 'aria_stt_endpoint_ms';
|
export const STT_ENDPOINT_STORAGE_KEY = 'aria_stt_endpoint_ms';
|
||||||
|
|
||||||
|
// Barge-in-Modus: darf man ARIA waehrend ihrer TTS-Antwort unterbrechen (reden)?
|
||||||
|
// Default AUS = sauberes Halb-Duplex (ARIA spricht aus, DANN oeffnet das Mikro —
|
||||||
|
// kein Selbst-Echo, kein Mischmasch). AN = waehrend TTS auf Wake-Wort lauschen.
|
||||||
|
export const BARGE_IN_STORAGE_KEY = 'aria_barge_in_enabled';
|
||||||
|
|
||||||
|
export async function loadBargeInEnabled(): Promise<boolean> {
|
||||||
|
try {
|
||||||
|
return (await AsyncStorage.getItem(BARGE_IN_STORAGE_KEY)) === 'true';
|
||||||
|
} catch {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
export async function saveBargeInEnabled(enabled: boolean): Promise<void> {
|
||||||
|
try {
|
||||||
|
await AsyncStorage.setItem(BARGE_IN_STORAGE_KEY, String(enabled));
|
||||||
|
} catch {}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** One-Shot-Transkription eines PCM-Schnipsels (base64, s16le 16kHz mono) via
|
||||||
|
* Voxtral — fuer die Wake-Wort-Bestaetigung. Schickt stt_transcribe_blob und
|
||||||
|
* wartet auf stt_transcribe_result (matching requestId) mit Timeout.
|
||||||
|
* Rueckgabe: Text (evtl. '') bei Antwort, oder null bei Timeout/Fehler →
|
||||||
|
* Aufrufer macht dann fail-open (Wake normal durchlassen). */
|
||||||
|
export async function transcribeBlob(pcmBase64: string, timeoutMs = 2500): Promise<string | null> {
|
||||||
|
if (!pcmBase64) return null;
|
||||||
|
const requestId = `wakeverify_${Date.now()}_${Math.floor(Math.random() * 100000)}`;
|
||||||
|
return new Promise<string | null>((resolve) => {
|
||||||
|
let done = false;
|
||||||
|
let unsub: (() => void) | null = null;
|
||||||
|
const timer = setTimeout(() => finish(null), timeoutMs);
|
||||||
|
function finish(val: string | null) {
|
||||||
|
if (done) return;
|
||||||
|
done = true;
|
||||||
|
try { unsub && unsub(); } catch {}
|
||||||
|
clearTimeout(timer);
|
||||||
|
resolve(val);
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
unsub = rvs.onMessage((msg: any) => {
|
||||||
|
if (msg?.type !== 'stt_transcribe_result') return;
|
||||||
|
const p = (msg as any).payload || {};
|
||||||
|
if (String(p.requestId || '') !== requestId) return;
|
||||||
|
finish(typeof p.text === 'string' ? p.text : '');
|
||||||
|
});
|
||||||
|
rvs.send('stt_transcribe_blob' as any, { requestId, pcm: pcmBase64, language: 'de' });
|
||||||
|
} catch {
|
||||||
|
finish(null);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Fragt die Bridge vor dem Aufnahme-Stream, welche STT-Instanz adressiert
|
||||||
|
* werden soll (Redundanz ueber mehrere STT-Nodes/Apps). Schickt
|
||||||
|
* stt_lease_request, wartet kurz auf stt_lease (matching requestId).
|
||||||
|
* Rueckgabe: instanceId (z.B. "voxtral@box-a") oder '' bei Timeout/keine
|
||||||
|
* Instanz — dann streamt die App wie bisher an ALLE (Broadcast, Single-Node
|
||||||
|
* unveraendert). Bewusst kurzer Timeout, damit die Aufnahme nie haengt. */
|
||||||
|
export async function requestSttLease(timeoutMs = 250): Promise<string> {
|
||||||
|
const requestId = `sttlease_${Date.now()}_${Math.floor(Math.random() * 100000)}`;
|
||||||
|
return new Promise<string>((resolve) => {
|
||||||
|
let done = false;
|
||||||
|
let unsub: (() => void) | null = null;
|
||||||
|
const timer = setTimeout(() => finish(''), timeoutMs);
|
||||||
|
function finish(val: string) {
|
||||||
|
if (done) return;
|
||||||
|
done = true;
|
||||||
|
try { unsub && unsub(); } catch {}
|
||||||
|
clearTimeout(timer);
|
||||||
|
resolve(val);
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
unsub = rvs.onMessage((msg: any) => {
|
||||||
|
if (msg?.type !== 'stt_lease') return;
|
||||||
|
const p = (msg as any).payload || {};
|
||||||
|
if (String(p.requestId || '') !== requestId) return;
|
||||||
|
finish(typeof p.instanceId === 'string' ? p.instanceId : '');
|
||||||
|
});
|
||||||
|
rvs.send('stt_lease_request' as any, { requestId });
|
||||||
|
} catch {
|
||||||
|
finish('');
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
export async function loadSttEndpointMs(): Promise<number> {
|
export async function loadSttEndpointMs(): Promise<number> {
|
||||||
try {
|
try {
|
||||||
const raw = await AsyncStorage.getItem(STT_ENDPOINT_STORAGE_KEY);
|
const raw = await AsyncStorage.getItem(STT_ENDPOINT_STORAGE_KEY);
|
||||||
@@ -189,18 +266,6 @@ export async function loadTtsSpeed(): Promise<number> {
|
|||||||
return TTS_SPEED_DEFAULT;
|
return TTS_SPEED_DEFAULT;
|
||||||
}
|
}
|
||||||
|
|
||||||
export async function loadConvWindowMs(): Promise<number> {
|
|
||||||
try {
|
|
||||||
const raw = await AsyncStorage.getItem(CONV_WINDOW_STORAGE_KEY);
|
|
||||||
if (raw != null) {
|
|
||||||
const n = parseFloat(raw);
|
|
||||||
if (isFinite(n) && n >= CONV_WINDOW_MIN_SEC && n <= CONV_WINDOW_MAX_SEC) {
|
|
||||||
return Math.round(n * 1000);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
} catch {}
|
|
||||||
return Math.round(CONV_WINDOW_DEFAULT_SEC * 1000);
|
|
||||||
}
|
|
||||||
|
|
||||||
async function loadVadSilenceMs(): Promise<number> {
|
async function loadVadSilenceMs(): Promise<number> {
|
||||||
try {
|
try {
|
||||||
@@ -351,6 +416,9 @@ class AudioService {
|
|||||||
// lich Chunks einer alten Session in eine neue mischen.
|
// lich Chunks einer alten Session in eine neue mischen.
|
||||||
private streamRequestId: string = '';
|
private streamRequestId: string = '';
|
||||||
private streamAudioRequestId: string = '';
|
private streamAudioRequestId: string = '';
|
||||||
|
// Adressierte STT-Instanz fuer diesen Stream (Redundanz-Routing). '' =
|
||||||
|
// Broadcast an alle STT-Nodes (Single-Node / kein Lease = wie bisher).
|
||||||
|
private streamTargetInstance: string = '';
|
||||||
// Latch: ist endpointListeners fuer den aktuellen Session-Cycle schon gefeuert
|
// Latch: ist endpointListeners fuer den aktuellen Session-Cycle schon gefeuert
|
||||||
// worden? Wird auf false gesetzt beim startStreamingRecording, auf true beim
|
// worden? Wird auf false gesetzt beim startStreamingRecording, auf true beim
|
||||||
// ersten Endpoint (egal ob via RVS oder Fallback). Verhindert Doppel-Fires.
|
// ersten Endpoint (egal ob via RVS oder Fallback). Verhindert Doppel-Fires.
|
||||||
@@ -1094,6 +1162,15 @@ class AudioService {
|
|||||||
const requestId = `sttstr_${Date.now()}_${Math.floor(Math.random() * 100000)}`;
|
const requestId = `sttstr_${Date.now()}_${Math.floor(Math.random() * 100000)}`;
|
||||||
this.streamRequestId = requestId;
|
this.streamRequestId = requestId;
|
||||||
this.streamAudioRequestId = opts.audioRequestId || '';
|
this.streamAudioRequestId = opts.audioRequestId || '';
|
||||||
|
// Redundanz-Routing: freie STT-Instanz leasen BEVOR Chunks fliessen, damit
|
||||||
|
// start + alle Chunks + end dieselbe Instanz adressieren. Kurzer Timeout →
|
||||||
|
// '' (Broadcast) falls keine Instanz/keine Antwort. Nie blockierend genug
|
||||||
|
// um die Aufnahme spuerbar zu verzoegern.
|
||||||
|
try {
|
||||||
|
this.streamTargetInstance = await requestSttLease();
|
||||||
|
} catch {
|
||||||
|
this.streamTargetInstance = '';
|
||||||
|
}
|
||||||
this.streamGotPartial = false;
|
this.streamGotPartial = false;
|
||||||
this.streamEndpointFired = false;
|
this.streamEndpointFired = false;
|
||||||
this.recordingStartTime = Date.now();
|
this.recordingStartTime = Date.now();
|
||||||
@@ -1114,6 +1191,7 @@ class AudioService {
|
|||||||
requestId: sessionId,
|
requestId: sessionId,
|
||||||
pcm: String(e?.pcm || ''),
|
pcm: String(e?.pcm || ''),
|
||||||
seq: Number(e?.seq || 0),
|
seq: Number(e?.seq || 0),
|
||||||
|
targetInstance: this.streamTargetInstance,
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
this.streamPcmErrorSub = emitter.addListener('PcmStreamError', (e: any) => {
|
this.streamPcmErrorSub = emitter.addListener('PcmStreamError', (e: any) => {
|
||||||
@@ -1145,10 +1223,11 @@ class AudioService {
|
|||||||
speed: typeof opts.speed === 'number' ? opts.speed : 1.0,
|
speed: typeof opts.speed === 'number' ? opts.speed : 1.0,
|
||||||
interrupted: !!opts.interrupted,
|
interrupted: !!opts.interrupted,
|
||||||
location: opts.location || null,
|
location: opts.location || null,
|
||||||
endpointMs: typeof opts.endpointMs === 'number' ? opts.endpointMs : 1500,
|
endpointMs: typeof opts.endpointMs === 'number' ? opts.endpointMs : STT_ENDPOINT_DEFAULT_MS,
|
||||||
hardCapMs: typeof opts.hardCapMs === 'number' ? opts.hardCapMs : 60000,
|
hardCapMs: typeof opts.hardCapMs === 'number' ? opts.hardCapMs : 60000,
|
||||||
sampleRate: 16000,
|
sampleRate: 16000,
|
||||||
projectId: opts.projectId || '',
|
projectId: opts.projectId || '',
|
||||||
|
targetInstance: this.streamTargetInstance,
|
||||||
});
|
});
|
||||||
|
|
||||||
// No-Speech-Watchdog — ersetzt den alten VAD-noSpeechTimer.
|
// No-Speech-Watchdog — ersetzt den alten VAD-noSpeechTimer.
|
||||||
@@ -1198,7 +1277,7 @@ class AudioService {
|
|||||||
if (!reqId) return;
|
if (!reqId) return;
|
||||||
const audioReqId = this.streamAudioRequestId;
|
const audioReqId = this.streamAudioRequestId;
|
||||||
try {
|
try {
|
||||||
rvs.send('stt_stream_end' as any, { requestId: reqId, reason });
|
rvs.send('stt_stream_end' as any, { requestId: reqId, reason, targetInstance: this.streamTargetInstance });
|
||||||
} catch (e) {
|
} catch (e) {
|
||||||
console.warn('[Audio] stt_stream_end senden fehlgeschlagen:', e);
|
console.warn('[Audio] stt_stream_end senden fehlgeschlagen:', e);
|
||||||
}
|
}
|
||||||
@@ -1233,7 +1312,7 @@ class AudioService {
|
|||||||
if (!reqId) return;
|
if (!reqId) return;
|
||||||
const audioReqId = this.streamAudioRequestId;
|
const audioReqId = this.streamAudioRequestId;
|
||||||
try {
|
try {
|
||||||
rvs.send('stt_stream_end' as any, { requestId: reqId, reason: `cancel:${reason}` });
|
rvs.send('stt_stream_end' as any, { requestId: reqId, reason: `cancel:${reason}`, targetInstance: this.streamTargetInstance });
|
||||||
} catch {}
|
} catch {}
|
||||||
this._cleanupStreamLocal(`cancel:${reason}`);
|
this._cleanupStreamLocal(`cancel:${reason}`);
|
||||||
// Listener feuern damit ChatScreen reagieren kann (endConversation etc.)
|
// Listener feuern damit ChatScreen reagieren kann (endConversation etc.)
|
||||||
|
|||||||
@@ -145,6 +145,17 @@ class GpsTrackingService {
|
|||||||
// liefert im Hintergrund keine Updates (nur Heartbeat sendet alte Werte).
|
// liefert im Hintergrund keine Updates (nur Heartbeat sendet alte Werte).
|
||||||
const bgEnabled = await isBackgroundGpsEnabled();
|
const bgEnabled = await isBackgroundGpsEnabled();
|
||||||
if (bgEnabled) {
|
if (bgEnabled) {
|
||||||
|
// Ohne ACCESS_BACKGROUND_LOCATION liefert watchPosition im Hintergrund
|
||||||
|
// NICHTS (Android 10+) → der Foreground-Service allein bringt nichts, und
|
||||||
|
// genau der Fall "Ankunft waehrend der Fahrt, Screen aus" faellt durch.
|
||||||
|
// Deshalb erst die Permission sicherstellen (oeffnet ggf. die Android-
|
||||||
|
// Settings fuer "Immer erlauben"), DANN den Location-Foreground-Service
|
||||||
|
// hochziehen — der haelt den Prozess wach, sodass watchPosition + der
|
||||||
|
// 60s-Heartbeat auch unter Doze weiterlaufen.
|
||||||
|
const bgOk = await ensureBackgroundLocationPermission();
|
||||||
|
if (!bgOk) {
|
||||||
|
console.warn('[gps-track] Background-Permission fehlt — Tracking nur im Vordergrund zuverlaessig');
|
||||||
|
}
|
||||||
try { await acquireBackgroundAudio('location'); } catch {}
|
try { await acquireBackgroundAudio('location'); } catch {}
|
||||||
}
|
}
|
||||||
try {
|
try {
|
||||||
|
|||||||
@@ -7,7 +7,7 @@
|
|||||||
*/
|
*/
|
||||||
|
|
||||||
import AsyncStorage from '@react-native-async-storage/async-storage';
|
import AsyncStorage from '@react-native-async-storage/async-storage';
|
||||||
import { Platform, DeviceEventEmitter } from 'react-native';
|
import { Platform, DeviceEventEmitter, AppState } from 'react-native';
|
||||||
import rvs from './rvs';
|
import rvs from './rvs';
|
||||||
|
|
||||||
// Lokales Event damit die SettingsScreen Live Logs / Events Tabs
|
// Lokales Event damit die SettingsScreen Live Logs / Events Tabs
|
||||||
@@ -38,6 +38,23 @@ const noop = () => {};
|
|||||||
let _verbose = true;
|
let _verbose = true;
|
||||||
let _debugLogsToBridge = false;
|
let _debugLogsToBridge = false;
|
||||||
|
|
||||||
|
// ─── Crash-Kontext ohne adb ─────────────────────────────────────────
|
||||||
|
// Ein RUN_MARKER bleibt gesetzt, solange die App AKTIV laeuft; bei sauberem
|
||||||
|
// Wechsel in den Hintergrund wird er geloescht. Ist er beim naechsten Start
|
||||||
|
// noch da, ist der vorige Lauf unsauber gestorben (nativer Crash/OOM — der
|
||||||
|
// schreibt KEINEN JS-Fehler, taucht also sonst nirgends auf). Wir melden das
|
||||||
|
// dann via RVS mit dem letzten Breadcrumb (was die App zuletzt tat).
|
||||||
|
const RUN_MARKER_KEY = 'aria_run_marker';
|
||||||
|
const BREADCRUMB_KEY = 'aria_last_breadcrumb';
|
||||||
|
let _breadcrumb: { ts: number; scope: string; message: string } = { ts: 0, scope: '', message: '' };
|
||||||
|
let _breadcrumbDirty = false;
|
||||||
|
|
||||||
|
/** Letzte App-Aktivitaet merken — Crash-Kontext fuer den naechsten Boot. */
|
||||||
|
export function noteBreadcrumb(scope: string, message: string): void {
|
||||||
|
_breadcrumb = { ts: Date.now(), scope: scope || '', message: String(message || '').slice(0, 120) };
|
||||||
|
_breadcrumbDirty = true;
|
||||||
|
}
|
||||||
|
|
||||||
function applyState(): void {
|
function applyState(): void {
|
||||||
console.log = _verbose ? originalLog : noop;
|
console.log = _verbose ? originalLog : noop;
|
||||||
}
|
}
|
||||||
@@ -53,6 +70,45 @@ export async function initLogger(): Promise<void> {
|
|||||||
_debugLogsToBridge = d === 'true'; // default: false
|
_debugLogsToBridge = d === 'true'; // default: false
|
||||||
} catch {}
|
} catch {}
|
||||||
applyState();
|
applyState();
|
||||||
|
await _initCrashDetection();
|
||||||
|
}
|
||||||
|
|
||||||
|
// Native-Crash-Erkennung (ohne adb) — siehe RUN_MARKER-Kommentar oben.
|
||||||
|
async function _initCrashDetection(): Promise<void> {
|
||||||
|
try {
|
||||||
|
const marker = await AsyncStorage.getItem(RUN_MARKER_KEY);
|
||||||
|
if (marker) {
|
||||||
|
let bc: any = {};
|
||||||
|
try { bc = JSON.parse((await AsyncStorage.getItem(BREADCRUMB_KEY)) || '{}'); } catch {}
|
||||||
|
const gap = bc && bc.ts ? Math.round((Date.now() - bc.ts) / 1000) : -1;
|
||||||
|
// Verzoegert melden — RVS ist beim Boot oft noch nicht verbunden.
|
||||||
|
setTimeout(() => {
|
||||||
|
reportAppError({
|
||||||
|
scope: 'app.crash-detected',
|
||||||
|
level: 'warn',
|
||||||
|
message: `Voriger Lauf ohne sauberes Shutdown beendet (nativer Crash/OOM?). `
|
||||||
|
+ `Letzte Aktivitaet: [${(bc && bc.scope) || '?'}] ${(bc && bc.message) || '?'}`
|
||||||
|
+ (gap >= 0 ? ` (vor ~${gap}s)` : ''),
|
||||||
|
});
|
||||||
|
}, 6000);
|
||||||
|
}
|
||||||
|
await AsyncStorage.setItem(RUN_MARKER_KEY, String(Date.now()));
|
||||||
|
} catch {}
|
||||||
|
// Breadcrumb throttled persistieren (alle 5s, nur wenn geaendert).
|
||||||
|
setInterval(() => {
|
||||||
|
if (_breadcrumbDirty) {
|
||||||
|
_breadcrumbDirty = false;
|
||||||
|
AsyncStorage.setItem(BREADCRUMB_KEY, JSON.stringify(_breadcrumb)).catch(() => {});
|
||||||
|
}
|
||||||
|
}, 5000);
|
||||||
|
// Sauberer Hintergrund-Wechsel → Marker weg (kein Crash). Rueckkehr → wieder
|
||||||
|
// scharf. So melden nur echte Aktiv-Crashes, kein normales Backgrounden.
|
||||||
|
try {
|
||||||
|
AppState.addEventListener('change', (s) => {
|
||||||
|
if (s === 'background') AsyncStorage.removeItem(RUN_MARKER_KEY).catch(() => {});
|
||||||
|
else if (s === 'active') AsyncStorage.setItem(RUN_MARKER_KEY, String(Date.now())).catch(() => {});
|
||||||
|
});
|
||||||
|
} catch {}
|
||||||
}
|
}
|
||||||
|
|
||||||
export function isVerboseLogging(): boolean {
|
export function isVerboseLogging(): boolean {
|
||||||
@@ -94,6 +150,7 @@ let _reportingInstalled = false;
|
|||||||
/** Schickt einen App-Fehler via RVS an die Bridge. */
|
/** Schickt einen App-Fehler via RVS an die Bridge. */
|
||||||
export function reportAppError(ev: AppErrorEvent): void {
|
export function reportAppError(ev: AppErrorEvent): void {
|
||||||
const ts = Date.now();
|
const ts = Date.now();
|
||||||
|
noteBreadcrumb(ev.scope, ev.message);
|
||||||
try {
|
try {
|
||||||
rvs.send('app_log' as any, {
|
rvs.send('app_log' as any, {
|
||||||
ts,
|
ts,
|
||||||
@@ -128,6 +185,9 @@ export function reportAppError(ev: AppErrorEvent): void {
|
|||||||
* Default aus damit Mama-Modus keine Disk-Schreiblast hat. Error-Reports
|
* Default aus damit Mama-Modus keine Disk-Schreiblast hat. Error-Reports
|
||||||
* (reportAppError) gehen weiterhin IMMER durch. */
|
* (reportAppError) gehen weiterhin IMMER durch. */
|
||||||
export function reportAppDebug(scope: string, message: string): void {
|
export function reportAppDebug(scope: string, message: string): void {
|
||||||
|
// Breadcrumb IMMER aktualisieren (auch wenn Debug-Logs-an-Bridge aus ist) —
|
||||||
|
// fuer den Crash-Kontext beim naechsten Boot.
|
||||||
|
noteBreadcrumb(scope, message);
|
||||||
if (!_debugLogsToBridge) return;
|
if (!_debugLogsToBridge) return;
|
||||||
const ts = Date.now();
|
const ts = Date.now();
|
||||||
const trimmed = String(message).slice(0, 2000);
|
const trimmed = String(message).slice(0, 2000);
|
||||||
|
|||||||
@@ -30,34 +30,22 @@ type PassiveListenCallback = () => void;
|
|||||||
|
|
||||||
export type WakeWordState = 'off' | 'armed' | 'conversing' | 'listening';
|
export type WakeWordState = 'off' | 'armed' | 'conversing' | 'listening';
|
||||||
|
|
||||||
/** Default-Dauer fuer den Passive-Listen-Modus nach einer Konversation —
|
/** Reine HANG-Notbremse fuer den Passive-Listen-Modus. Das echte Ende regelt IMMER
|
||||||
* in dem Fenster braucht's kein Wake-Word, Speaker-ID-Filter haelt
|
* die passive Aufnahme selbst: Stille-Toleranz (User pausiert), No-Speech (User
|
||||||
* fremde Stimmen raus (TV, Familie). 30s default; konfigurierbar. */
|
* sagt gar nichts) oder Hard-Cap (max. Aufnahmedauer, ~5min) → ChatScreen ruft
|
||||||
export const PASSIVE_LISTEN_DEFAULT_MS = 30_000;
|
* dann exitPassiveListening. Dieser Timer darf aktives Reden NIE abschneiden —
|
||||||
export const PASSIVE_LISTEN_STORAGE_KEY = 'aria_passive_listen_ms';
|
* deshalb LÄNGER als der Hard-Cap (nur falls ein Endpoint-Event mal verloren geht
|
||||||
|
* und der State sonst ewig 'listening' bliebe). Das alte 30s-Fenster, das lange
|
||||||
export async function loadPassiveListenMs(): Promise<number> {
|
* Antworten mitten im Satz kappte, ist damit raus. */
|
||||||
try {
|
const PASSIVE_BACKSTOP_MS = 10 * 60_000;
|
||||||
const raw = await AsyncStorage.getItem(PASSIVE_LISTEN_STORAGE_KEY);
|
|
||||||
if (raw) {
|
|
||||||
const n = parseInt(raw, 10);
|
|
||||||
if (isFinite(n) && n >= 0 && n <= 120_000) return n;
|
|
||||||
}
|
|
||||||
} catch {}
|
|
||||||
return PASSIVE_LISTEN_DEFAULT_MS;
|
|
||||||
}
|
|
||||||
|
|
||||||
export async function savePassiveListenMs(ms: number): Promise<void> {
|
|
||||||
await AsyncStorage.setItem(PASSIVE_LISTEN_STORAGE_KEY, String(ms));
|
|
||||||
}
|
|
||||||
|
|
||||||
export const WAKE_KEYWORD_STORAGE = 'aria_wake_keyword';
|
export const WAKE_KEYWORD_STORAGE = 'aria_wake_keyword';
|
||||||
|
|
||||||
// Wake-Word-Empfindlichkeit (openWakeWord-Threshold). Hoeher = strenger =
|
// Wake-Word-Empfindlichkeit (openWakeWord-Threshold). Hoeher = strenger =
|
||||||
// weniger Fehlauslösung (z.B. durch Musik/Radio ueber die Auto-Lautsprecher,
|
// weniger Fehlauslösung, aber man muss deutlicher/lauter sprechen (fuehlt sich
|
||||||
// die das Mikro mithoert — der App-Echo-Canceler kann nur ARIAs eigenes TTS
|
// "traege" an). Fehlausloeser werden ueber Speaker-ID (E3) ohnehin verworfen,
|
||||||
// rausrechnen, NICHT Spotify). Default 0.6 (war 0.5). 0..1.
|
// deshalb darf der Default empfindlicher sein. 0.45 (war 0.6/0.5). 0..1.
|
||||||
export const WAKE_THRESHOLD_DEFAULT = 0.6;
|
export const WAKE_THRESHOLD_DEFAULT = 0.45;
|
||||||
export const WAKE_THRESHOLD_MIN = 0.3;
|
export const WAKE_THRESHOLD_MIN = 0.3;
|
||||||
export const WAKE_THRESHOLD_MAX = 0.9;
|
export const WAKE_THRESHOLD_MAX = 0.9;
|
||||||
export const WAKE_THRESHOLD_STORAGE_KEY = 'aria_wake_threshold';
|
export const WAKE_THRESHOLD_STORAGE_KEY = 'aria_wake_threshold';
|
||||||
@@ -77,6 +65,49 @@ export async function saveWakeThreshold(v: number): Promise<void> {
|
|||||||
await AsyncStorage.setItem(WAKE_THRESHOLD_STORAGE_KEY, String(v));
|
await AsyncStorage.setItem(WAKE_THRESHOLD_STORAGE_KEY, String(v));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Hintergrund-Wake: darf das Wake-Wort auch triggern, wenn die App im
|
||||||
|
// Hintergrund / der Bildschirm gesperrt ist? Default AUS — im Hintergrund
|
||||||
|
// sind die meisten „Trigger" Fehlalarme (TV, Husten, AudioFocus-Spikes).
|
||||||
|
// AN = auch bei gesperrtem Bildschirm zuhoeren. Die native Erkennung laeuft
|
||||||
|
// ohnehin durch (Foreground-Service + Wake-Locks) — dieser Schalter oeffnet
|
||||||
|
// nur das JS-Gate in onWakeDetected.
|
||||||
|
export const BG_WAKE_STORAGE_KEY = 'aria_bg_wake_enabled';
|
||||||
|
|
||||||
|
export async function loadBgWakeEnabled(): Promise<boolean> {
|
||||||
|
try {
|
||||||
|
return (await AsyncStorage.getItem(BG_WAKE_STORAGE_KEY)) === 'true';
|
||||||
|
} catch {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
export async function saveBgWakeEnabled(enabled: boolean): Promise<void> {
|
||||||
|
try {
|
||||||
|
await AsyncStorage.setItem(BG_WAKE_STORAGE_KEY, String(enabled));
|
||||||
|
} catch {}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Wake-Wort-Bestaetigung: nach einem openWakeWord-Trigger den Vor-Trigger-Audio
|
||||||
|
// von Voxtral gegenpruefen lassen ("war das wirklich 'Computer' oder Musik?").
|
||||||
|
// Killt Musik-Fehltrigger (z.B. Pet Shop Boys), kostet ~0.5-1s Extra-Latenz pro
|
||||||
|
// Wake. Default AUS (opt-in), fail-open. Braucht das native preTriggerPcm im
|
||||||
|
// Event (neueres APK) — ohne das macht die App normal weiter.
|
||||||
|
export const WAKE_CONFIRM_STORAGE_KEY = 'aria_wake_confirm_enabled';
|
||||||
|
|
||||||
|
export async function loadWakeConfirmEnabled(): Promise<boolean> {
|
||||||
|
try {
|
||||||
|
return (await AsyncStorage.getItem(WAKE_CONFIRM_STORAGE_KEY)) === 'true';
|
||||||
|
} catch {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
export async function saveWakeConfirmEnabled(enabled: boolean): Promise<void> {
|
||||||
|
try {
|
||||||
|
await AsyncStorage.setItem(WAKE_CONFIRM_STORAGE_KEY, String(enabled));
|
||||||
|
} catch {}
|
||||||
|
}
|
||||||
|
|
||||||
/** Verfuegbare Wake-Words — entsprechen den .onnx Dateien in
|
/** Verfuegbare Wake-Words — entsprechen den .onnx Dateien in
|
||||||
* android/app/src/main/assets/openwakeword/. Custom-Keywords (eigenes
|
* android/app/src/main/assets/openwakeword/. Custom-Keywords (eigenes
|
||||||
* Training via openwakeword Notebook) muessen aktuell als Asset eingebaut
|
* Training via openwakeword Notebook) muessen aktuell als Asset eingebaut
|
||||||
@@ -103,7 +134,9 @@ export const KEYWORD_LABELS: Record<WakeKeyword, string> = {
|
|||||||
// Detection-Tuning. Threshold ist ueber die Settings konfigurierbar
|
// Detection-Tuning. Threshold ist ueber die Settings konfigurierbar
|
||||||
// (loadWakeThreshold) — der Wert hier ist nur der Fallback.
|
// (loadWakeThreshold) — der Wert hier ist nur der Fallback.
|
||||||
const DEFAULT_THRESHOLD = WAKE_THRESHOLD_DEFAULT;
|
const DEFAULT_THRESHOLD = WAKE_THRESHOLD_DEFAULT;
|
||||||
const DEFAULT_PATIENCE = 2;
|
// patience=1 statt 2: nur EIN Frame ueber Threshold noetig → deutlich schneller.
|
||||||
|
// Speaker-ID filtert Fehlausloeser, also ist das vertretbar.
|
||||||
|
const DEFAULT_PATIENCE = 1;
|
||||||
const DEFAULT_DEBOUNCE_MS = 1500;
|
const DEFAULT_DEBOUNCE_MS = 1500;
|
||||||
|
|
||||||
interface OpenWakeWordModule {
|
interface OpenWakeWordModule {
|
||||||
@@ -143,6 +176,13 @@ class WakeWordService {
|
|||||||
* Hintergrund-Detections sind quasi immer false-positives (TV, Husten,
|
* Hintergrund-Detections sind quasi immer false-positives (TV, Husten,
|
||||||
* AudioFocus-Switch beim Wechsel zu Musik etc.). */
|
* AudioFocus-Switch beim Wechsel zu Musik etc.). */
|
||||||
private inBackground: boolean = false;
|
private inBackground: boolean = false;
|
||||||
|
/** Wenn true: Wake-Wort triggert auch im Hintergrund / bei gesperrtem
|
||||||
|
* Bildschirm. Default false. Wird beim Arm aus AsyncStorage geladen und
|
||||||
|
* bei Aenderung in den Einstellungen via setBgWakeEnabled() aktualisiert. */
|
||||||
|
private bgWakeEnabled: boolean = false;
|
||||||
|
/** Wake-Wort per Voxtral bestaetigen (gegen Musik-Fehltrigger)? Default false.
|
||||||
|
* Wird beim Arm geladen + per setWakeConfirmEnabled aus den Einstellungen. */
|
||||||
|
private wakeConfirmEnabled: boolean = false;
|
||||||
/** Re-Entry-Guard fuer onWakeDetected: native kann mehrere
|
/** Re-Entry-Guard fuer onWakeDetected: native kann mehrere
|
||||||
* WakeWordDetected-Events emitten BEVOR OpenWakeWord.stop() in JS
|
* WakeWordDetected-Events emitten BEVOR OpenWakeWord.stop() in JS
|
||||||
* resolved (Bridge-Queue + Doze-Backlog). Mit dem Flag wird das zweite
|
* resolved (Bridge-Queue + Doze-Backlog). Mit dem Flag wird das zweite
|
||||||
@@ -150,8 +190,9 @@ class WakeWordService {
|
|||||||
* Ausnahme: bargeListening → Barge-In ist ein legitimer neuer Trigger
|
* Ausnahme: bargeListening → Barge-In ist ein legitimer neuer Trigger
|
||||||
* waehrend ARIA noch redet, NICHT vom Guard blockieren. */
|
* waehrend ARIA noch redet, NICHT vom Guard blockieren. */
|
||||||
private detectionInProgress: boolean = false;
|
private detectionInProgress: boolean = false;
|
||||||
/** Passive-Listen-Timer: feuert nach PASSIVE_LISTEN_MS ohne Stefan-Speech,
|
/** Passive-Listen-Backstop-Timer: Notbremse (PASSIVE_BACKSTOP_MS). Normal endet
|
||||||
* beendet den listening-State und geht zurueck zu armed. */
|
* das Fenster ueber die Stille-Toleranz der Aufnahme; feuert dieser Timer
|
||||||
|
* trotzdem, zurueck zu armed. */
|
||||||
private passiveListenTimer: ReturnType<typeof setTimeout> | null = null;
|
private passiveListenTimer: ReturnType<typeof setTimeout> | null = null;
|
||||||
/** Callbacks fuer den Eintritt in Passive-Listen — ChatScreen startet
|
/** Callbacks fuer den Eintritt in Passive-Listen — ChatScreen startet
|
||||||
* hier eine streaming-Aufnahme OHNE User-Bubble (passiv lauschen). */
|
* hier eine streaming-Aufnahme OHNE User-Bubble (passiv lauschen). */
|
||||||
@@ -223,15 +264,20 @@ class WakeWordService {
|
|||||||
this.initInProgress = (async () => {
|
this.initInProgress = (async () => {
|
||||||
try {
|
try {
|
||||||
const threshold = await loadWakeThreshold();
|
const threshold = await loadWakeThreshold();
|
||||||
console.log('[WakeWord] init mit threshold=%s', threshold);
|
this.bgWakeEnabled = await loadBgWakeEnabled();
|
||||||
|
this.wakeConfirmEnabled = await loadWakeConfirmEnabled();
|
||||||
|
console.log('[WakeWord] init mit threshold=%s, bgWake=%s, confirm=%s',
|
||||||
|
threshold, this.bgWakeEnabled, this.wakeConfirmEnabled);
|
||||||
await OpenWakeWord.init(this.keyword, threshold, DEFAULT_PATIENCE, DEFAULT_DEBOUNCE_MS);
|
await OpenWakeWord.init(this.keyword, threshold, DEFAULT_PATIENCE, DEFAULT_DEBOUNCE_MS);
|
||||||
// Subscribe nur einmal
|
// Subscribe nur einmal
|
||||||
if (!this.eventSub) {
|
if (!this.eventSub) {
|
||||||
const emitter = new NativeEventEmitter(NativeModules.OpenWakeWord);
|
const emitter = new NativeEventEmitter(NativeModules.OpenWakeWord);
|
||||||
this.eventSub = emitter.addListener('WakeWordDetected', () => {
|
this.eventSub = emitter.addListener('WakeWordDetected', (payload: any) => {
|
||||||
console.log('[WakeWord] Native Detection-Event empfangen');
|
console.log('[WakeWord] Native Detection-Event empfangen');
|
||||||
this.onWakeDetected().catch(err =>
|
// payload.preTriggerPcm (base64 s16le 16kHz) fuer die Bestaetigung —
|
||||||
console.warn('[WakeWord] onWakeDetected crashed:', err));
|
// nur in neueren APKs vorhanden; ohne = fail-open (kein Verify).
|
||||||
|
this.onWakeDetected(payload && payload.preTriggerPcm ? String(payload.preTriggerPcm) : null)
|
||||||
|
.catch(err => console.warn('[WakeWord] onWakeDetected crashed:', err));
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
this.nativeReady = true;
|
this.nativeReady = true;
|
||||||
@@ -310,7 +356,7 @@ class WakeWordService {
|
|||||||
/** Cooldown setzen — alle Wake-Word-Detections in den naechsten ms ignorieren.
|
/** Cooldown setzen — alle Wake-Word-Detections in den naechsten ms ignorieren.
|
||||||
* Wird beim App-Resume gerufen weil AppState-Wechsel Audio-Spikes erzeugen
|
* Wird beim App-Resume gerufen weil AppState-Wechsel Audio-Spikes erzeugen
|
||||||
* die openWakeWord faelschlich als Trigger interpretiert. */
|
* die openWakeWord faelschlich als Trigger interpretiert. */
|
||||||
setResumeCooldown(ms: number = 1500): void {
|
setResumeCooldown(ms: number = 500): void {
|
||||||
this.cooldownUntilMs = Date.now() + ms;
|
this.cooldownUntilMs = Date.now() + ms;
|
||||||
console.log('[WakeWord] Cooldown aktiv fuer %dms', ms);
|
console.log('[WakeWord] Cooldown aktiv fuer %dms', ms);
|
||||||
}
|
}
|
||||||
@@ -320,23 +366,45 @@ class WakeWordService {
|
|||||||
* was als „Wake-Word" reinkommt ist Husten/TV/AudioFocus-Switch. */
|
* was als „Wake-Word" reinkommt ist Husten/TV/AudioFocus-Switch. */
|
||||||
setBackground(): void {
|
setBackground(): void {
|
||||||
this.inBackground = true;
|
this.inBackground = true;
|
||||||
console.log('[WakeWord] App im Hintergrund — Detections gesperrt');
|
console.log('[WakeWord] App im Hintergrund — Detections %s',
|
||||||
|
this.bgWakeEnabled ? 'AKTIV (Hintergrund-Wake an)' : 'gesperrt');
|
||||||
}
|
}
|
||||||
|
|
||||||
/** App im Vordergrund: Detections wieder freigeben, plus 3s Cooldown
|
/** Hintergrund-Wake ein/aus schalten (aus den Einstellungen). */
|
||||||
* als Schutz gegen den AudioFocus-/AudioTrack-Spike der direkt nach
|
setBgWakeEnabled(enabled: boolean): void {
|
||||||
* dem Resume kommt. Ersetzt das alte setResumeCooldown(3000)-Pattern. */
|
this.bgWakeEnabled = enabled;
|
||||||
|
console.log('[WakeWord] Hintergrund-Wake = %s', enabled);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Wake-Wort-Bestaetigung (Voxtral) ein/aus (aus den Einstellungen). */
|
||||||
|
setWakeConfirmEnabled(enabled: boolean): void {
|
||||||
|
this.wakeConfirmEnabled = enabled;
|
||||||
|
console.log('[WakeWord] Wake-Bestaetigung = %s', enabled);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Ist Hintergrund-Wake an? Steuert u.a. ob der Konversationsmodus auch im
|
||||||
|
* Hintergrund weiterlaeuft (sonst: im Hintergrund direkt zurueck aufs Wake-Word). */
|
||||||
|
isBgWakeEnabled(): boolean {
|
||||||
|
return this.bgWakeEnabled;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** App im Vordergrund: Detections wieder freigeben, plus kurzer Cooldown
|
||||||
|
* als Schutz gegen den AudioFocus-/AudioTrack-Spike direkt nach dem Resume.
|
||||||
|
* 1s statt 3s — 3s hat sich "traege" angefuehlt (Trigger direkt nach dem
|
||||||
|
* App-Oeffnen wurden verschluckt). */
|
||||||
setForeground(): void {
|
setForeground(): void {
|
||||||
this.inBackground = false;
|
this.inBackground = false;
|
||||||
this.cooldownUntilMs = Date.now() + 3000;
|
this.cooldownUntilMs = Date.now() + 1000;
|
||||||
console.log('[WakeWord] App im Vordergrund — Cooldown 3s aktiv');
|
console.log('[WakeWord] App im Vordergrund — Cooldown 1s aktiv');
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Wake-Word getriggert: Native-Modul pausieren, Konversation starten. */
|
/** Wake-Word getriggert: Native-Modul pausieren, Konversation starten.
|
||||||
private async onWakeDetected(): Promise<void> {
|
* preTriggerPcm: base64 s16le 16kHz Vor-Trigger-Audio fuer die Bestaetigung
|
||||||
if (this.inBackground) {
|
* (null = nicht verfuegbar → keine Bestaetigung, normal weiter). */
|
||||||
console.log('[WakeWord] Trigger ignoriert (App im Hintergrund)');
|
private async onWakeDetected(preTriggerPcm: string | null = null): Promise<void> {
|
||||||
import('./logger').then(m => m.reportAppDebug('wake.detect', 'ignored: app in background')).catch(()=>{});
|
if (this.inBackground && !this.bgWakeEnabled) {
|
||||||
|
console.log('[WakeWord] Trigger ignoriert (App im Hintergrund, Hintergrund-Wake aus)');
|
||||||
|
import('./logger').then(m => m.reportAppDebug('wake.detect', 'ignored: app in background (bg-wake off)')).catch(()=>{});
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
// Re-Entry-Guard: blocken wenn ein Detection-Zyklus schon laeuft.
|
// Re-Entry-Guard: blocken wenn ein Detection-Zyklus schon laeuft.
|
||||||
@@ -379,6 +447,22 @@ class WakeWordService {
|
|||||||
// Kein erneutes setState — wir bleiben in 'conversing'.
|
// Kein erneutes setState — wir bleiben in 'conversing'.
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
// Wake-Wort-Bestaetigung (gegen Musik-Fehltrigger): den Vor-Trigger-Schnipsel
|
||||||
|
// von Voxtral gegenpruefen. Bestaetigt → weiter (Gong + Mikro). Verworfen
|
||||||
|
// (Musik/Rauschen, kein "Computer") → kein Dialog, kein Gong, re-arm. Fail-
|
||||||
|
// open: ohne PCM / bei Timeout/Fehler laeuft es normal durch.
|
||||||
|
if (this.wakeConfirmEnabled && preTriggerPcm) {
|
||||||
|
const confirmed = await this.confirmWake(preTriggerPcm);
|
||||||
|
if (!confirmed) {
|
||||||
|
this.detectionInProgress = false;
|
||||||
|
if (this.nativeReady && OpenWakeWord) {
|
||||||
|
try { await OpenWakeWord.start(); } catch (e) {
|
||||||
|
console.warn('[WakeWord] re-arm nach verworfener Bestaetigung failed:', e);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
this.setState('conversing');
|
this.setState('conversing');
|
||||||
// Direkt feuern — KEIN setTimeout. Im Hintergrund (Display aus) parkt
|
// Direkt feuern — KEIN setTimeout. Im Hintergrund (Display aus) parkt
|
||||||
// Android den JS-Thread; ein setTimeout(200ms) kann dann Minuten lang
|
// Android den JS-Thread; ein setTimeout(200ms) kann dann Minuten lang
|
||||||
@@ -392,6 +476,34 @@ class WakeWordService {
|
|||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/** Voxtral-Bestaetigung des Vor-Trigger-Schnipsels. true = Wake-Wort erkannt
|
||||||
|
* (oder fail-open bei Timeout/Fehler), false = Musik/Rauschen → verwerfen. */
|
||||||
|
private async confirmWake(pcm: string): Promise<boolean> {
|
||||||
|
try {
|
||||||
|
const audio = await import('./audio');
|
||||||
|
const text = await audio.transcribeBlob(pcm);
|
||||||
|
if (text === null) {
|
||||||
|
console.log('[WakeWord] Bestaetigung: Timeout/Fehler → fail-open (durchlassen)');
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
const norm = text.toLowerCase();
|
||||||
|
// Distinktive Wake-Wort-Bestandteile (>= 4 Zeichen; 'hey' o.ae. rausfiltern,
|
||||||
|
// taucht sonst in Song-Texten auf und wuerde faelschlich bestaetigen).
|
||||||
|
const kwWords = this.keyword.toLowerCase().replace(/_/g, ' ')
|
||||||
|
.split(/\s+/).filter(w => w.length >= 4);
|
||||||
|
if (kwWords.length === 0) return true; // zu kurzes Keyword → nicht pruefbar
|
||||||
|
const ok = kwWords.some(w => norm.includes(w));
|
||||||
|
console.log('[WakeWord] Bestaetigung: text=%o kw=%o → %s',
|
||||||
|
text, kwWords, ok ? 'BESTAETIGT' : 'verworfen (Musik-FP?)');
|
||||||
|
import('./logger').then(m => m.reportAppDebug('wake.confirm',
|
||||||
|
`text="${text.slice(0, 40)}" kw=${kwWords.join('|')} → ${ok ? 'ok' : 'reject'}`)).catch(() => {});
|
||||||
|
return ok;
|
||||||
|
} catch (e) {
|
||||||
|
console.warn('[WakeWord] confirmWake err → fail-open:', e);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/** Wake-Word PARALLEL zur TTS-Wiedergabe lauschen lassen — User kann
|
/** Wake-Word PARALLEL zur TTS-Wiedergabe lauschen lassen — User kann
|
||||||
* "Computer" sagen waehrend ARIA noch redet, AcousticEchoCanceler im
|
* "Computer" sagen waehrend ARIA noch redet, AcousticEchoCanceler im
|
||||||
* Native-Modul verhindert dass ARIAs eigene Stimme triggert.
|
* Native-Modul verhindert dass ARIAs eigene Stimme triggert.
|
||||||
@@ -486,13 +598,12 @@ class WakeWordService {
|
|||||||
import('./logger').then(m => m.reportAppDebug('wake.end',
|
import('./logger').then(m => m.reportAppDebug('wake.end',
|
||||||
`endConversation called, wasBarge=${wasBarge}, nativeReady=${this.nativeReady}`)).catch(()=>{});
|
`endConversation called, wasBarge=${wasBarge}, nativeReady=${this.nativeReady}`)).catch(()=>{});
|
||||||
|
|
||||||
// Passive-Listen aktiv? Dann nicht direkt zu armed — passive lauschen
|
// Kein skipPassive? Dann EIN Stille-Fenster zum Weiterreden (kein Wake-Word
|
||||||
// fuer N Sekunden, dann erst Wake-Word wieder aktivieren. Speaker-ID
|
// noetig). Das echte Ende regelt die Stille-Toleranz der passiven Aufnahme;
|
||||||
// (Phase 3) filtert fremde Stimmen weg, der User kann ohne erneute
|
// der Backstop-Timer ist nur die Notbremse. Der User kann ohne erneute
|
||||||
// Anrede weitersprechen.
|
// Anrede weitersprechen; sagt er nichts → zurueck aufs Wake-Word.
|
||||||
const passiveMs = await loadPassiveListenMs();
|
if (!skipPassive && this.nativeReady) {
|
||||||
if (!skipPassive && passiveMs > 0 && this.nativeReady) {
|
this.enterPassiveListening(PASSIVE_BACKSTOP_MS);
|
||||||
this.enterPassiveListening(passiveMs);
|
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -534,10 +645,10 @@ class WakeWordService {
|
|||||||
this.cancelPassiveListenTimer();
|
this.cancelPassiveListenTimer();
|
||||||
this.setState('listening');
|
this.setState('listening');
|
||||||
const seconds = Math.round(durationMs / 1000);
|
const seconds = Math.round(durationMs / 1000);
|
||||||
console.log('[WakeWord] Passive-Listen aktiv (%ds) — Speaker-ID gefiltert', seconds);
|
console.log('[WakeWord] Passive-Listen aktiv (Backstop %ds) — Speaker-ID gefiltert', seconds);
|
||||||
import('./logger').then(m => m.reportAppDebug('wake.passive',
|
import('./logger').then(m => m.reportAppDebug('wake.passive',
|
||||||
`entered listening for ${seconds}s, cb-count=${this.passiveListenCallbacks.length}`)).catch(()=>{});
|
`entered listening (backstop ${seconds}s), cb-count=${this.passiveListenCallbacks.length}`)).catch(()=>{});
|
||||||
ToastAndroid.show(`🎧 ${seconds}s lauscht — sprich einfach weiter`, ToastAndroid.SHORT);
|
ToastAndroid.show('🎧 sprich einfach weiter', ToastAndroid.SHORT);
|
||||||
this.passiveListenTimer = setTimeout(() => {
|
this.passiveListenTimer = setTimeout(() => {
|
||||||
this.passiveListenTimer = null;
|
this.passiveListenTimer = null;
|
||||||
this.exitPassiveListening('timeout').catch(() => {});
|
this.exitPassiveListening('timeout').catch(() => {});
|
||||||
|
|||||||
@@ -0,0 +1,175 @@
|
|||||||
|
/**
|
||||||
|
* AriaViewCanvas — die pannbare Flaeche, auf der ARIAs komponierte Ansicht
|
||||||
|
* (aria_view) MATERIALISIERT: Orb oben, darunter die Karten. Erscheint als
|
||||||
|
* Overlay ueber dem Chat, sobald ARIA present_view aufruft ("sag was → Orb denkt
|
||||||
|
* → Karte fliegt rein"). Der erste, greifbare Vorgeschmack aufs generative
|
||||||
|
* Cockpit (M1).
|
||||||
|
*
|
||||||
|
* Bedienung (NoMachine-Prinzip): 2-Finger halten + schieben bewegt die Welt,
|
||||||
|
* Pinch zoomt. Ein-Finger-Touch geht an die Karten durch (Scrollen). Die Welt
|
||||||
|
* traegt gerenderte/gestreamte Inhalte — interaktive native Panels rasten
|
||||||
|
* spaeter bei Scale 1 ein (Chat bleibt separat darunter).
|
||||||
|
*
|
||||||
|
* Geraete-agnostisch gehalten: liest nur die ViewSpec, damit ein spaeterer Web-/
|
||||||
|
* AR-Renderer dieselbe Spec konsumieren kann.
|
||||||
|
*/
|
||||||
|
|
||||||
|
import React from 'react';
|
||||||
|
import { StyleSheet, Text, TouchableOpacity, View } from 'react-native';
|
||||||
|
import Animated, {
|
||||||
|
FadeInDown,
|
||||||
|
useAnimatedStyle,
|
||||||
|
useSharedValue,
|
||||||
|
withTiming,
|
||||||
|
} from 'react-native-reanimated';
|
||||||
|
import { Gesture, GestureDetector } from 'react-native-gesture-handler';
|
||||||
|
import { ViewSpec } from '../services/ariaView';
|
||||||
|
import Orb from './Orb';
|
||||||
|
import CardView from './CardView';
|
||||||
|
|
||||||
|
const MIN_SCALE = 0.5;
|
||||||
|
const MAX_SCALE = 3;
|
||||||
|
|
||||||
|
interface Props {
|
||||||
|
view: ViewSpec;
|
||||||
|
onClose: () => void;
|
||||||
|
}
|
||||||
|
|
||||||
|
const AriaViewCanvas: React.FC<Props> = ({ view, onClose }) => {
|
||||||
|
const tx = useSharedValue(0);
|
||||||
|
const ty = useSharedValue(0);
|
||||||
|
const scale = useSharedValue(1);
|
||||||
|
const savedTx = useSharedValue(0);
|
||||||
|
const savedTy = useSharedValue(0);
|
||||||
|
const savedScale = useSharedValue(1);
|
||||||
|
|
||||||
|
const pan = Gesture.Pan()
|
||||||
|
.minPointers(2)
|
||||||
|
.maxPointers(2)
|
||||||
|
.onUpdate((e) => {
|
||||||
|
tx.value = savedTx.value + e.translationX;
|
||||||
|
ty.value = savedTy.value + e.translationY;
|
||||||
|
})
|
||||||
|
.onEnd(() => {
|
||||||
|
savedTx.value = tx.value;
|
||||||
|
savedTy.value = ty.value;
|
||||||
|
});
|
||||||
|
|
||||||
|
const pinch = Gesture.Pinch()
|
||||||
|
.onUpdate((e) => {
|
||||||
|
const next = savedScale.value * e.scale;
|
||||||
|
scale.value = Math.max(MIN_SCALE, Math.min(MAX_SCALE, next));
|
||||||
|
})
|
||||||
|
.onEnd(() => {
|
||||||
|
savedScale.value = scale.value;
|
||||||
|
});
|
||||||
|
|
||||||
|
const composed = Gesture.Simultaneous(pan, pinch);
|
||||||
|
|
||||||
|
const worldStyle = useAnimatedStyle(() => ({
|
||||||
|
transform: [
|
||||||
|
{ translateX: tx.value },
|
||||||
|
{ translateY: ty.value },
|
||||||
|
{ scale: scale.value },
|
||||||
|
],
|
||||||
|
}));
|
||||||
|
|
||||||
|
const resetCamera = () => {
|
||||||
|
tx.value = withTiming(0);
|
||||||
|
ty.value = withTiming(0);
|
||||||
|
scale.value = withTiming(1);
|
||||||
|
savedTx.value = 0;
|
||||||
|
savedTy.value = 0;
|
||||||
|
savedScale.value = 1;
|
||||||
|
};
|
||||||
|
|
||||||
|
const cards = Array.isArray(view.cards) ? view.cards : [];
|
||||||
|
|
||||||
|
return (
|
||||||
|
<View style={styles.overlay}>
|
||||||
|
<GestureDetector gesture={composed}>
|
||||||
|
<Animated.View style={[styles.world, worldStyle]}>
|
||||||
|
<View style={styles.orbWrap}>
|
||||||
|
<Orb state={view.orb} size={110} />
|
||||||
|
</View>
|
||||||
|
{!!view.title && <Text style={styles.worldTitle}>{view.title}</Text>}
|
||||||
|
<View style={styles.cards}>
|
||||||
|
{cards.map((c, i) => (
|
||||||
|
<Animated.View
|
||||||
|
key={i}
|
||||||
|
entering={FadeInDown.duration(420).delay(120 + i * 90)}
|
||||||
|
>
|
||||||
|
<CardView card={c} />
|
||||||
|
</Animated.View>
|
||||||
|
))}
|
||||||
|
</View>
|
||||||
|
</Animated.View>
|
||||||
|
</GestureDetector>
|
||||||
|
|
||||||
|
{/* Steuerung — ausserhalb des Transforms, immer bei Scale 1 bedienbar */}
|
||||||
|
<View style={styles.topBar} pointerEvents="box-none">
|
||||||
|
<TouchableOpacity style={styles.iconBtn} onPress={resetCamera}>
|
||||||
|
<Text style={styles.icon}>⤢</Text>
|
||||||
|
</TouchableOpacity>
|
||||||
|
<TouchableOpacity style={styles.iconBtn} onPress={onClose}>
|
||||||
|
<Text style={styles.icon}>✕</Text>
|
||||||
|
</TouchableOpacity>
|
||||||
|
</View>
|
||||||
|
<View style={styles.hintWrap} pointerEvents="none">
|
||||||
|
<Text style={styles.hint}>2 Finger: schieben · Pinch: zoomen</Text>
|
||||||
|
</View>
|
||||||
|
</View>
|
||||||
|
);
|
||||||
|
};
|
||||||
|
|
||||||
|
const styles = StyleSheet.create({
|
||||||
|
overlay: {
|
||||||
|
...StyleSheet.absoluteFillObject,
|
||||||
|
backgroundColor: 'rgba(6,6,16,0.94)',
|
||||||
|
zIndex: 50,
|
||||||
|
},
|
||||||
|
world: {
|
||||||
|
...StyleSheet.absoluteFillObject,
|
||||||
|
alignItems: 'center',
|
||||||
|
paddingTop: 48,
|
||||||
|
paddingHorizontal: 18,
|
||||||
|
},
|
||||||
|
orbWrap: { marginTop: 8, marginBottom: 6 },
|
||||||
|
worldTitle: {
|
||||||
|
color: '#C9C9FF',
|
||||||
|
fontSize: 18,
|
||||||
|
fontWeight: '700',
|
||||||
|
marginBottom: 4,
|
||||||
|
textAlign: 'center',
|
||||||
|
},
|
||||||
|
cards: { width: '100%', maxWidth: 560 },
|
||||||
|
topBar: {
|
||||||
|
position: 'absolute',
|
||||||
|
top: 10,
|
||||||
|
right: 12,
|
||||||
|
flexDirection: 'row',
|
||||||
|
},
|
||||||
|
iconBtn: {
|
||||||
|
width: 40,
|
||||||
|
height: 40,
|
||||||
|
borderRadius: 20,
|
||||||
|
marginLeft: 10,
|
||||||
|
alignItems: 'center',
|
||||||
|
justifyContent: 'center',
|
||||||
|
backgroundColor: 'rgba(30,30,60,0.9)',
|
||||||
|
borderWidth: 1,
|
||||||
|
borderColor: 'rgba(123,92,255,0.4)',
|
||||||
|
},
|
||||||
|
icon: { color: '#C9C9FF', fontSize: 18 },
|
||||||
|
hintWrap: {
|
||||||
|
position: 'absolute',
|
||||||
|
bottom: 14,
|
||||||
|
alignSelf: 'center',
|
||||||
|
},
|
||||||
|
hint: {
|
||||||
|
color: '#6A6A90',
|
||||||
|
fontSize: 12,
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
export default AriaViewCanvas;
|
||||||
@@ -0,0 +1,114 @@
|
|||||||
|
/**
|
||||||
|
* CardView — rendert EINE Karte einer aria_view-Spec (M1). Schaltet nach
|
||||||
|
* card.type auf den passenden Renderer. Unbekannte Typen werden als Text-
|
||||||
|
* Fallback gezeigt (nie crashen).
|
||||||
|
*
|
||||||
|
* Bewusst dependency-leicht (v1): Markdown wird als Klartext dargestellt, Map
|
||||||
|
* als Marker-Liste (kein Karten-Lib), Code als Monospace-Block. Spaeter koennen
|
||||||
|
* einzelne Renderer aufgebohrt werden, ohne die Spec/den Fluss zu aendern.
|
||||||
|
*/
|
||||||
|
|
||||||
|
import React from 'react';
|
||||||
|
import { Image, ScrollView, StyleSheet, Text, View } from 'react-native';
|
||||||
|
import { ViewCard, ViewMarker } from '../services/ariaView';
|
||||||
|
|
||||||
|
const ImageBody: React.FC<{ src?: string }> = ({ src }) => {
|
||||||
|
const isUrl = !!src && /^https?:\/\//i.test(src);
|
||||||
|
if (isUrl) {
|
||||||
|
return <Image source={{ uri: src }} style={styles.image} resizeMode="contain" />;
|
||||||
|
}
|
||||||
|
return <Text style={styles.muted}>🖼️ {src || '(kein Bild)'}</Text>;
|
||||||
|
};
|
||||||
|
|
||||||
|
const ListBody: React.FC<{ md?: string }> = ({ md }) => {
|
||||||
|
const lines = (md || '')
|
||||||
|
.split('\n')
|
||||||
|
.map((l) => l.replace(/^\s*[-*•]\s?/, '').trim())
|
||||||
|
.filter(Boolean);
|
||||||
|
if (lines.length === 0) return <Text style={styles.muted}>(leer)</Text>;
|
||||||
|
return (
|
||||||
|
<View>
|
||||||
|
{lines.map((l, i) => (
|
||||||
|
<View key={i} style={styles.listRow}>
|
||||||
|
<Text style={styles.bullet}>•</Text>
|
||||||
|
<Text style={styles.text}>{l}</Text>
|
||||||
|
</View>
|
||||||
|
))}
|
||||||
|
</View>
|
||||||
|
);
|
||||||
|
};
|
||||||
|
|
||||||
|
const MapBody: React.FC<{ markers?: ViewMarker[] }> = ({ markers }) => {
|
||||||
|
const ms = Array.isArray(markers) ? markers : [];
|
||||||
|
return (
|
||||||
|
<View style={styles.map}>
|
||||||
|
<Text style={styles.mapHint}>🗺️ Karte ({ms.length} Orte)</Text>
|
||||||
|
{ms.map((m, i) => (
|
||||||
|
<Text key={i} style={styles.text}>
|
||||||
|
📍 {m.label || `${m.lat?.toFixed?.(4)}, ${m.lon?.toFixed?.(4)}`}
|
||||||
|
</Text>
|
||||||
|
))}
|
||||||
|
</View>
|
||||||
|
);
|
||||||
|
};
|
||||||
|
|
||||||
|
const CodeBody: React.FC<{ md?: string; path?: string; lang?: string }> = ({ md, path, lang }) => (
|
||||||
|
<View>
|
||||||
|
{(path || lang) && (
|
||||||
|
<Text style={styles.codeCaption}>
|
||||||
|
{path || ''}{lang ? ` · ${lang}` : ''}
|
||||||
|
</Text>
|
||||||
|
)}
|
||||||
|
<ScrollView horizontal style={styles.codeScroll}>
|
||||||
|
<Text style={styles.code}>{md || ''}</Text>
|
||||||
|
</ScrollView>
|
||||||
|
</View>
|
||||||
|
);
|
||||||
|
|
||||||
|
const CardView: React.FC<{ card: ViewCard }> = ({ card }) => {
|
||||||
|
return (
|
||||||
|
<View style={styles.card}>
|
||||||
|
{!!card.title && <Text style={styles.cardTitle}>{card.title}</Text>}
|
||||||
|
{card.type === 'image' ? (
|
||||||
|
<ImageBody src={card.src} />
|
||||||
|
) : card.type === 'list' ? (
|
||||||
|
<ListBody md={card.md} />
|
||||||
|
) : card.type === 'map' ? (
|
||||||
|
<MapBody markers={card.markers} />
|
||||||
|
) : card.type === 'code' ? (
|
||||||
|
<CodeBody md={card.md} path={card.path} lang={card.lang} />
|
||||||
|
) : (
|
||||||
|
<Text style={styles.text}>{card.md || ''}</Text>
|
||||||
|
)}
|
||||||
|
</View>
|
||||||
|
);
|
||||||
|
};
|
||||||
|
|
||||||
|
const styles = StyleSheet.create({
|
||||||
|
card: {
|
||||||
|
backgroundColor: 'rgba(18,18,42,0.92)',
|
||||||
|
borderColor: 'rgba(123,92,255,0.35)',
|
||||||
|
borderWidth: 1,
|
||||||
|
borderRadius: 14,
|
||||||
|
padding: 14,
|
||||||
|
marginVertical: 8,
|
||||||
|
shadowColor: '#7B5CFF',
|
||||||
|
shadowOpacity: 0.25,
|
||||||
|
shadowRadius: 12,
|
||||||
|
shadowOffset: { width: 0, height: 2 },
|
||||||
|
elevation: 6,
|
||||||
|
},
|
||||||
|
cardTitle: { color: '#C9C9FF', fontSize: 15, fontWeight: '700', marginBottom: 8 },
|
||||||
|
text: { color: '#E6E6F0', fontSize: 14, lineHeight: 20, flexShrink: 1 },
|
||||||
|
muted: { color: '#8A8AB0', fontSize: 13, fontStyle: 'italic' },
|
||||||
|
image: { width: '100%', height: 200, borderRadius: 8, backgroundColor: '#0D0D1A' },
|
||||||
|
listRow: { flexDirection: 'row', alignItems: 'flex-start', marginVertical: 2 },
|
||||||
|
bullet: { color: '#7B5CFF', marginRight: 8, fontSize: 14, lineHeight: 20 },
|
||||||
|
map: { backgroundColor: '#0D0D1A', borderRadius: 8, padding: 10 },
|
||||||
|
mapHint: { color: '#00B4D8', fontSize: 13, fontWeight: '600', marginBottom: 6 },
|
||||||
|
codeCaption: { color: '#8A8AB0', fontSize: 12, marginBottom: 6 },
|
||||||
|
codeScroll: { backgroundColor: '#0A0A14', borderRadius: 8, padding: 10 },
|
||||||
|
code: { color: '#B9F5C9', fontFamily: 'monospace', fontSize: 12.5, lineHeight: 18 },
|
||||||
|
});
|
||||||
|
|
||||||
|
export default React.memo(CardView);
|
||||||
@@ -0,0 +1,106 @@
|
|||||||
|
/**
|
||||||
|
* Orb — ARIAs Praesenz-Avatar (M1). Zeigt ihren Zustand (idle/listening/
|
||||||
|
* thinking/speaking/working) als pulsierender Leucht-Kern und ist das
|
||||||
|
* verbindende Element ueber alle Oberflaechen (App/Web/spaeter Brille).
|
||||||
|
*
|
||||||
|
* Reine Optik, keine Logik — der Zustand kommt von aussen (aria_view.orb bzw.
|
||||||
|
* spaeter direkt von Audio/Wake-Word-Signalen). Dependency-leicht: nur
|
||||||
|
* reanimated (schon installiert), kein SVG/Gradient noetig.
|
||||||
|
*/
|
||||||
|
|
||||||
|
import React, { useEffect } from 'react';
|
||||||
|
import { StyleSheet, View } from 'react-native';
|
||||||
|
import Animated, {
|
||||||
|
Easing,
|
||||||
|
cancelAnimation,
|
||||||
|
useAnimatedStyle,
|
||||||
|
useSharedValue,
|
||||||
|
withRepeat,
|
||||||
|
withTiming,
|
||||||
|
} from 'react-native-reanimated';
|
||||||
|
import { OrbState } from '../services/ariaView';
|
||||||
|
|
||||||
|
const COLORS: Record<OrbState, string> = {
|
||||||
|
idle: '#3A6EA5',
|
||||||
|
listening: '#00B4D8',
|
||||||
|
thinking: '#7B5CFF',
|
||||||
|
speaking: '#34C759',
|
||||||
|
working: '#FF9500',
|
||||||
|
};
|
||||||
|
|
||||||
|
interface Props {
|
||||||
|
state?: OrbState;
|
||||||
|
size?: number;
|
||||||
|
}
|
||||||
|
|
||||||
|
const Orb: React.FC<Props> = ({ state = 'idle', size = 120 }) => {
|
||||||
|
const pulse = useSharedValue(1);
|
||||||
|
|
||||||
|
useEffect(() => {
|
||||||
|
const fast = state === 'thinking' || state === 'working';
|
||||||
|
cancelAnimation(pulse);
|
||||||
|
pulse.value = 1;
|
||||||
|
pulse.value = withRepeat(
|
||||||
|
withTiming(fast ? 1.14 : 1.07, {
|
||||||
|
duration: fast ? 620 : 1500,
|
||||||
|
easing: Easing.inOut(Easing.ease),
|
||||||
|
}),
|
||||||
|
-1,
|
||||||
|
true,
|
||||||
|
);
|
||||||
|
return () => cancelAnimation(pulse);
|
||||||
|
}, [state, pulse]);
|
||||||
|
|
||||||
|
const animStyle = useAnimatedStyle(() => ({ transform: [{ scale: pulse.value }] }));
|
||||||
|
const color = COLORS[state] || COLORS.idle;
|
||||||
|
|
||||||
|
return (
|
||||||
|
<View style={[styles.wrap, { width: size, height: size }]}>
|
||||||
|
<Animated.View
|
||||||
|
style={[
|
||||||
|
styles.glow,
|
||||||
|
{ width: size, height: size, borderRadius: size / 2, backgroundColor: color },
|
||||||
|
animStyle,
|
||||||
|
]}
|
||||||
|
/>
|
||||||
|
<Animated.View
|
||||||
|
style={[
|
||||||
|
styles.ring,
|
||||||
|
{
|
||||||
|
width: size * 0.72,
|
||||||
|
height: size * 0.72,
|
||||||
|
borderRadius: size * 0.36,
|
||||||
|
borderColor: color,
|
||||||
|
},
|
||||||
|
animStyle,
|
||||||
|
]}
|
||||||
|
/>
|
||||||
|
<View
|
||||||
|
style={[
|
||||||
|
styles.core,
|
||||||
|
{
|
||||||
|
width: size * 0.44,
|
||||||
|
height: size * 0.44,
|
||||||
|
borderRadius: size * 0.22,
|
||||||
|
backgroundColor: color,
|
||||||
|
shadowColor: color,
|
||||||
|
},
|
||||||
|
]}
|
||||||
|
/>
|
||||||
|
</View>
|
||||||
|
);
|
||||||
|
};
|
||||||
|
|
||||||
|
const styles = StyleSheet.create({
|
||||||
|
wrap: { alignItems: 'center', justifyContent: 'center' },
|
||||||
|
glow: { position: 'absolute', opacity: 0.22 },
|
||||||
|
ring: { position: 'absolute', borderWidth: 2, opacity: 0.55 },
|
||||||
|
core: {
|
||||||
|
shadowOpacity: 0.9,
|
||||||
|
shadowRadius: 16,
|
||||||
|
shadowOffset: { width: 0, height: 0 },
|
||||||
|
elevation: 12,
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
export default React.memo(Orb);
|
||||||
@@ -9,6 +9,7 @@
|
|||||||
*/
|
*/
|
||||||
|
|
||||||
import React, { useEffect, useMemo, useState } from 'react';
|
import React, { useEffect, useMemo, useState } from 'react';
|
||||||
|
import { View } from 'react-native';
|
||||||
import projectFocus, { FocusSnapshot } from '../services/projectFocus';
|
import projectFocus, { FocusSnapshot } from '../services/projectFocus';
|
||||||
import codeFile from '../services/codeFile';
|
import codeFile from '../services/codeFile';
|
||||||
import brainApi from '../services/brainApi';
|
import brainApi from '../services/brainApi';
|
||||||
@@ -16,6 +17,8 @@ import viewMode, { ViewModeValue } from '../services/viewMode';
|
|||||||
import ChatScreen from '../screens/ChatScreen';
|
import ChatScreen from '../screens/ChatScreen';
|
||||||
import { TileId } from './layout';
|
import { TileId } from './layout';
|
||||||
import WorkspaceDeck from './WorkspaceDeck';
|
import WorkspaceDeck from './WorkspaceDeck';
|
||||||
|
import ariaView, { AriaView } from '../services/ariaView';
|
||||||
|
import AriaViewCanvas from './AriaViewCanvas';
|
||||||
|
|
||||||
const COCKPIT_PANELS: TileId[] = ['chat', 'files', 'editor', 'vnc'];
|
const COCKPIT_PANELS: TileId[] = ['chat', 'files', 'editor', 'vnc'];
|
||||||
|
|
||||||
@@ -24,12 +27,21 @@ const WorkspaceScreen: React.FC = () => {
|
|||||||
const [focus, setFocus] = useState<FocusSnapshot>(projectFocus.get());
|
const [focus, setFocus] = useState<FocusSnapshot>(projectFocus.get());
|
||||||
const [hasCode, setHasCode] = useState(false);
|
const [hasCode, setHasCode] = useState(false);
|
||||||
const [hasDesktop, setHasDesktop] = useState(false);
|
const [hasDesktop, setHasDesktop] = useState(false);
|
||||||
|
const [view, setView] = useState<AriaView | undefined>(undefined);
|
||||||
|
|
||||||
useEffect(() => viewMode.subscribe(setMode), []);
|
useEffect(() => viewMode.subscribe(setMode), []);
|
||||||
useEffect(() => projectFocus.subscribe(setFocus), []);
|
useEffect(() => projectFocus.subscribe(setFocus), []);
|
||||||
|
|
||||||
const pid = focus.focusedProjectId;
|
const pid = focus.focusedProjectId;
|
||||||
|
|
||||||
|
// aria_view: ARIAs komponierte Ansicht fuers fokussierte Projekt spiegeln.
|
||||||
|
useEffect(() => {
|
||||||
|
setView(ariaView.getView(pid));
|
||||||
|
return ariaView.subscribe((v) => {
|
||||||
|
if ((v.projectId || '') === (pid || '')) setView(v);
|
||||||
|
});
|
||||||
|
}, [pid]);
|
||||||
|
|
||||||
// Code-Signal: hat der Spiegel schon Dateien fuer dieses Projekt?
|
// Code-Signal: hat der Spiegel schon Dateien fuer dieses Projekt?
|
||||||
useEffect(() => {
|
useEffect(() => {
|
||||||
setHasCode(codeFile.getFiles(pid).length > 0);
|
setHasCode(codeFile.getFiles(pid).length > 0);
|
||||||
@@ -59,13 +71,32 @@ const WorkspaceScreen: React.FC = () => {
|
|||||||
vnc: hasDesktop ? '#34C759' : undefined,
|
vnc: hasDesktop ? '#34C759' : undefined,
|
||||||
} as Partial<Record<TileId, string>>), [hasCode, hasDesktop]);
|
} as Partial<Record<TileId, string>>), [hasCode, hasDesktop]);
|
||||||
|
|
||||||
// Kompakt-Ansicht: klassischer Vollbild-Chat, exakt wie vor dem Umbau.
|
// Kompakt-Ansicht: klassischer Vollbild-Chat; Cockpit: Workbench mit Dock.
|
||||||
if (mode === 'compact') {
|
const content =
|
||||||
return <ChatScreen />;
|
mode === 'compact' ? (
|
||||||
}
|
<ChatScreen />
|
||||||
|
) : (
|
||||||
|
<WorkspaceDeck projectId={pid} panels={COCKPIT_PANELS} badges={badges} />
|
||||||
|
);
|
||||||
|
|
||||||
// Cockpit: Workbench mit Dock.
|
// Generative Flaeche als Overlay, sobald ARIA fuer dieses Projekt eine Ansicht
|
||||||
return <WorkspaceDeck projectId={pid} panels={COCKPIT_PANELS} badges={badges} />;
|
// komponiert hat (present_view → aria_view). Chat/Cockpit bleiben darunter.
|
||||||
|
const showView = !!view && (view.projectId || '') === (pid || '');
|
||||||
|
|
||||||
|
return (
|
||||||
|
<View style={{ flex: 1 }}>
|
||||||
|
{content}
|
||||||
|
{showView && view && (
|
||||||
|
<AriaViewCanvas
|
||||||
|
view={view.view}
|
||||||
|
onClose={() => {
|
||||||
|
ariaView.clear(pid);
|
||||||
|
setView(undefined);
|
||||||
|
}}
|
||||||
|
/>
|
||||||
|
)}
|
||||||
|
</View>
|
||||||
|
);
|
||||||
};
|
};
|
||||||
|
|
||||||
export default WorkspaceScreen;
|
export default WorkspaceScreen;
|
||||||
|
|||||||
+404
-23
@@ -132,6 +132,34 @@ WEB_SEARCH_TOOL = {
|
|||||||
|
|
||||||
# Meta-Tool: ARIA kann selbst neue Skills bauen
|
# Meta-Tool: ARIA kann selbst neue Skills bauen
|
||||||
META_TOOLS = [
|
META_TOOLS = [
|
||||||
|
{
|
||||||
|
"type": "function",
|
||||||
|
"function": {
|
||||||
|
"name": "ear_control",
|
||||||
|
"description": (
|
||||||
|
"Schaltet dein Ohr (den Wake-Word-Listener) an oder aus. Nutze das, "
|
||||||
|
"wenn Stefan will dass du aufhoerst zuzuhoeren — EGAL wie er es "
|
||||||
|
"formuliert: 'leg dich schlafen', 'geh schlafen', 'Ohr aus', 'gute "
|
||||||
|
"Nacht', 'mach mal Pause vom Zuhoeren', 'ich geh ins Bett, du kannst "
|
||||||
|
"aus' → action='off'. Wenn er dich wieder aktivieren will ('Ohr an', "
|
||||||
|
"'wach auf', 'hoer wieder zu') → action='on'. Reiner Steuerbefehl: "
|
||||||
|
"die App stoppt/startet den Listener, du bestaetigst nur kurz (wird "
|
||||||
|
"nicht vorgelesen). Nach 'off' geht Wieder-An per 'Ohr an' (Text/"
|
||||||
|
"Aufnahme-Knopf) oder App-Button."
|
||||||
|
),
|
||||||
|
"parameters": {
|
||||||
|
"type": "object",
|
||||||
|
"properties": {
|
||||||
|
"action": {
|
||||||
|
"type": "string",
|
||||||
|
"enum": ["off", "on"],
|
||||||
|
"description": "off = Ohr aus (nicht mehr zuhoeren), on = Ohr wieder an",
|
||||||
|
},
|
||||||
|
},
|
||||||
|
"required": ["action"],
|
||||||
|
},
|
||||||
|
},
|
||||||
|
},
|
||||||
{
|
{
|
||||||
"type": "function",
|
"type": "function",
|
||||||
"function": {
|
"function": {
|
||||||
@@ -627,10 +655,11 @@ META_TOOLS = [
|
|||||||
"name": "request_location_tracking",
|
"name": "request_location_tracking",
|
||||||
"description": (
|
"description": (
|
||||||
"Bittet die App, das kontinuierliche GPS-Tracking zu aktivieren oder zu "
|
"Bittet die App, das kontinuierliche GPS-Tracking zu aktivieren oder zu "
|
||||||
"deaktivieren. Default ist AUS (Akku-Schutz). Nutze das wenn du einen "
|
"deaktivieren. Default ist AUS (Akku-Schutz). HINWEIS: Beim Anlegen/Loeschen "
|
||||||
"GPS-basierten Watcher anlegst (z.B. `near(...)`), sonst hat die App "
|
"eines Standort-Watchers (`near/entered_near/left_near`) schaltet das System "
|
||||||
"veraltete Position und der Watcher feuert nie. Auch wieder ausschalten "
|
"das Tracking bereits AUTOMATISCH mit an bzw. aus — dieses Tool brauchst du "
|
||||||
"wenn der letzte GPS-Watcher geloescht wurde."
|
"dafuer nicht mehr. Nutze es nur fuer manuelle/explizite Faelle (z.B. Tracking "
|
||||||
|
"kurz einschalten ohne Watcher)."
|
||||||
),
|
),
|
||||||
"parameters": {
|
"parameters": {
|
||||||
"type": "object",
|
"type": "object",
|
||||||
@@ -642,6 +671,48 @@ META_TOOLS = [
|
|||||||
},
|
},
|
||||||
},
|
},
|
||||||
},
|
},
|
||||||
|
{
|
||||||
|
"type": "function",
|
||||||
|
"function": {
|
||||||
|
"name": "present_view",
|
||||||
|
"description": (
|
||||||
|
"Komponiere eine RAEUMLICHE Ansicht, die auf ARIAs Flaeche materialisiert "
|
||||||
|
"(Orb + Karten). Nutze das, wenn eine visuelle Anordnung besser ist als reiner "
|
||||||
|
"Text — z.B. eine Akte/Datei zeigen (Bild links, Text rechts, Karte unten), "
|
||||||
|
"einen Vergleich, eine Liste, eine Karte mit Orten, oder Code. NICHT fuer "
|
||||||
|
"normale Gespraechsantworten. Die Karten erscheinen ZUSAETZLICH zu deiner "
|
||||||
|
"(kurzen) Sprachantwort — halte die Antwort dann knapp, das Visuelle traegt."
|
||||||
|
),
|
||||||
|
"parameters": {
|
||||||
|
"type": "object",
|
||||||
|
"properties": {
|
||||||
|
"cards": {
|
||||||
|
"type": "array",
|
||||||
|
"description": "Die Karten der Ansicht, in sinnvoller Anordnung.",
|
||||||
|
"items": {
|
||||||
|
"type": "object",
|
||||||
|
"properties": {
|
||||||
|
"type": {"type": "string", "enum": ["text", "image", "map", "code", "list"],
|
||||||
|
"description": "Kartentyp"},
|
||||||
|
"title": {"type": "string", "description": "Optionaler Kartentitel"},
|
||||||
|
"md": {"type": "string", "description": "text/list: Markdown-Inhalt"},
|
||||||
|
"src": {"type": "string", "description": "image: Bild-URL oder /shared-Pfad"},
|
||||||
|
"markers": {"type": "array", "items": {"type": "object"},
|
||||||
|
"description": "map: Liste [{lat, lon, label}]"},
|
||||||
|
"path": {"type": "string", "description": "code: Datei-Pfad im Projekt"},
|
||||||
|
"lang": {"type": "string", "description": "code: Sprache fuers Highlighting"},
|
||||||
|
},
|
||||||
|
"required": ["type"],
|
||||||
|
},
|
||||||
|
},
|
||||||
|
"orb": {"type": "string", "enum": ["idle", "speaking", "working"],
|
||||||
|
"description": "Orb-Zustand nach dem Rendern (default: speaking)"},
|
||||||
|
"title": {"type": "string", "description": "Optionaler Titel der ganzen Ansicht"},
|
||||||
|
},
|
||||||
|
"required": ["cards"],
|
||||||
|
},
|
||||||
|
},
|
||||||
|
},
|
||||||
{
|
{
|
||||||
"type": "function",
|
"type": "function",
|
||||||
"function": {
|
"function": {
|
||||||
@@ -788,7 +859,7 @@ META_TOOLS = [
|
|||||||
"function": {
|
"function": {
|
||||||
"name": "flux_generate",
|
"name": "flux_generate",
|
||||||
"description": (
|
"description": (
|
||||||
"Generiere ein Bild aus einem Text-Prompt via FLUX auf der Gamebox-GPU. "
|
"Generiere ein Bild aus einem Text-Prompt via FLUX auf der AI-Box-GPU. "
|
||||||
"Brauchbar fuer 'mal mir ein X', 'wie sieht ein Y aus?', Mockups, "
|
"Brauchbar fuer 'mal mir ein X', 'wie sieht ein Y aus?', Mockups, "
|
||||||
"Konzept-Skizzen, Memes. Render dauert 20-90s — kuendige es Stefan "
|
"Konzept-Skizzen, Memes. Render dauert 20-90s — kuendige es Stefan "
|
||||||
"kurz an, dann ist er nicht ueberrascht.\n\n"
|
"kurz an, dann ist er nicht ueberrascht.\n\n"
|
||||||
@@ -1113,9 +1184,11 @@ META_TOOLS = [
|
|||||||
"description": (
|
"description": (
|
||||||
"Zeigt welche ARIA-Satelliten (Aussenposten-Container in FREMDEN Netzen, "
|
"Zeigt welche ARIA-Satelliten (Aussenposten-Container in FREMDEN Netzen, "
|
||||||
"z.B. 'Buero') gerade ONLINE sind und was sie koennen (discover / "
|
"z.B. 'Buero') gerade ONLINE sind und was sie koennen (discover / "
|
||||||
"dial.launch / wol / http). Nutze das ZUERST, wenn Stefan etwas 'im "
|
"dial.launch / wol / http). Nutze das ZUERST, wenn es um ein Geraet/einen "
|
||||||
"Buero' / 'im Netz X' / 'auf dem <Geraet> dort' machen will — so weisst "
|
"Host in einem Netz geht, auf dem Du nicht direkt sitzt (Zuhause, Buero, "
|
||||||
"Du welche Netze erreichbar sind."
|
"jede private IP wie 192.168.x) — z.B. Drucker-Fuellstand, NAS, Smart-TV. "
|
||||||
|
"Rufe es IMMER auf, BEVOR Du sagst ein Geraet sei nicht erreichbar — ein "
|
||||||
|
"Satellit im Zielnetz ist der Weg hinein."
|
||||||
),
|
),
|
||||||
"parameters": {"type": "object", "properties": {}},
|
"parameters": {"type": "object", "properties": {}},
|
||||||
},
|
},
|
||||||
@@ -1126,8 +1199,11 @@ META_TOOLS = [
|
|||||||
"name": "satellite_devices",
|
"name": "satellite_devices",
|
||||||
"description": (
|
"description": (
|
||||||
"Listet die Geraete im Netz eines Satelliten (Fire TV, Chromecast, "
|
"Listet die Geraete im Netz eines Satelliten (Fire TV, Chromecast, "
|
||||||
"Smart-TVs, Drucker, NAS, Hosts ...). Nutze es um herauszufinden welches "
|
"Smart-TVs, Drucker, Switches, Router, APs, NAS, Hosts ...). Nutze es um "
|
||||||
"Geraet gemeint ist, BEVOR Du satellite_command aufrufst."
|
"herauszufinden welches Geraet gemeint ist, BEVOR Du satellite_command "
|
||||||
|
"aufrufst. SNMP-faehige Geraete tragen ein 'snmp'-Feld (Name, Beschreibung, "
|
||||||
|
"Standort, Uptime) und einen praeziseren 'type' (switch/router/nas ...) — "
|
||||||
|
"gute Quelle um Netz-Hardware zu erklaeren."
|
||||||
),
|
),
|
||||||
"parameters": {
|
"parameters": {
|
||||||
"type": "object",
|
"type": "object",
|
||||||
@@ -1147,8 +1223,28 @@ META_TOOLS = [
|
|||||||
"Beispiel YouTube-Video auf Fire TV: action='dial.launch', "
|
"Beispiel YouTube-Video auf Fire TV: action='dial.launch', "
|
||||||
"device='Fire TV', params={'app':'YouTube','v':'<videoId>'}. Weitere "
|
"device='Fire TV', params={'app':'YouTube','v':'<videoId>'}. Weitere "
|
||||||
"Aktionen: 'wol' (params={'mac':'...'}) zum Aufwecken, "
|
"Aktionen: 'wol' (params={'mac':'...'}) zum Aufwecken, "
|
||||||
"'http.get'/'http.post' (params={'url':'...'}) fuer lokale Webhooks. "
|
"'snmp.printer' (params={'ip':'<drucker-ip>'}) — BEVORZUGT fuer Drucker-"
|
||||||
"Geht nur, wenn der Satellit Steuerung erlaubt (siehe satellite_list)."
|
"Tinte/Toner: liest die Fuellstaende zuverlaessig als Prozent aus der "
|
||||||
|
"Printer-MIB (kein HTML-Scrapen). 'snmp.ports' (params={'ip':'...'}) — "
|
||||||
|
"Switch/Router-Interfaces: aktive/freie Ports + Linkspeed (beantwortet "
|
||||||
|
"'sind noch Ports frei'). 'snmp.info' (params={'ip':'...'}) — Modell, "
|
||||||
|
"Seriennummer, INSTALLIERTE Firmware/Software-Version (ob ein Update "
|
||||||
|
"existiert, weiss SNMP NICHT). 'snmp.get'/'snmp.walk' "
|
||||||
|
"(params={'ip':'...','oid':'...','community':'public'}) fuer beliebige "
|
||||||
|
"SNMP-Werte. 'fritzbox.info' / 'fritzbox.hosts' (params={'ip':'<fritzbox>'}) "
|
||||||
|
"— Internetverbindung/Datenrate/externe IP bzw. verbundene Geraete (braucht "
|
||||||
|
"hinterlegten FritzBox-Login). Fuer Geraete mit hinterlegten Zugangsdaten "
|
||||||
|
"(community/v3/Login) nutzt der Satellit diese automatisch — Du musst keine "
|
||||||
|
"Passwoerter mitgeben. "
|
||||||
|
"'http.get'/'http.post' (params={'url':'...'}) fuer lokale Webhooks UND "
|
||||||
|
"um Geraete-Statusseiten zu lesen (Fallback fuer Tinte, NAS ...). Bei grossen "
|
||||||
|
"Seiten NICHT blind paginieren: setze params['contains'] (String oder "
|
||||||
|
"Liste) — dann kommen nur Zeilen zurueck, die einen der Begriffe enthalten "
|
||||||
|
"(z.B. contains=['ink','toner','cyan','magenta','yellow','black','%'] fuer "
|
||||||
|
"Tinte). Zusaetzlich moeglich: params['offset'] und params['max_chars'] "
|
||||||
|
"(Default 20000). Die Antwort meldet total_chars + truncated, damit Du "
|
||||||
|
"siehst, ob noch mehr da ist. Geht nur, wenn der Satellit Steuerung "
|
||||||
|
"erlaubt (siehe satellite_list)."
|
||||||
),
|
),
|
||||||
"parameters": {
|
"parameters": {
|
||||||
"type": "object",
|
"type": "object",
|
||||||
@@ -1304,6 +1400,154 @@ def _extract_await_marker(text: str) -> tuple:
|
|||||||
return text, False
|
return text, False
|
||||||
|
|
||||||
|
|
||||||
|
# ── Sprach-/Gespraechs-Steuermarker (ARIA deklariert die Phase SELBST) ──
|
||||||
|
#
|
||||||
|
# Voice-First: ARIA erkennt aus dem Text, ob Stefan einen BEFEHL gibt (etwas tun)
|
||||||
|
# oder eine FRAGE stellt (etwas wissen), und ob das Gespraech/eine Befehlskette
|
||||||
|
# weiterlaeuft oder endet. Sie haengt dazu Marker ans Ende ihrer Antwort — genau
|
||||||
|
# wie [[AWAIT]], und sie werden ebenso entfernt (nicht angezeigt/vorgelesen/in
|
||||||
|
# History). Der Marker ist AUTORITATIV — er ueberschreibt das Skill-Manifest-Flag,
|
||||||
|
# denn dasselbe Skill (z.B. VM-/GUI-Steuerung) ist mal Befehl, mal Auskunft; nur
|
||||||
|
# ARIA weiss aus dem Kontext, was gerade gemeint ist.
|
||||||
|
#
|
||||||
|
# [[STUMM]] -> reiner Steuerbefehl: NICHT vorlesen (speak=false). Allein =
|
||||||
|
# Einzelbefehl → danach zurueck aufs Wake-Word (converse=false).
|
||||||
|
# [[WEITER]] -> Konversation/Befehlskette laeuft weiter: Mikro offen halten
|
||||||
|
# (converse=true) — kein erneutes "Computer" noetig.
|
||||||
|
# [[ENDE]] -> Konversation/Kette beenden: zurueck aufs Wake-Word (converse=false).
|
||||||
|
_SILENT_MARKER_RE = re.compile(r"\[\[\s*STUMM\s*\]\]", re.IGNORECASE)
|
||||||
|
_CONT_MARKER_RE = re.compile(r"\[\[\s*WEITER\s*\]\]", re.IGNORECASE)
|
||||||
|
_END_MARKER_RE = re.compile(r"\[\[\s*ENDE\s*\]\]", re.IGNORECASE)
|
||||||
|
|
||||||
|
|
||||||
|
def _extract_flow_markers(text: str) -> tuple:
|
||||||
|
"""Zieht [[STUMM]]/[[WEITER]]/[[ENDE]] aus dem finalen Text.
|
||||||
|
Gibt (clean_text, speak_override, converse_override) zurueck; ein Override ist
|
||||||
|
None, wenn der jeweilige Marker fehlt (dann gilt Default/Skill-Flag).
|
||||||
|
Regeln: [[STUMM]] alleine = Einzelbefehl → auch converse=false (Mikro zu),
|
||||||
|
ausser [[WEITER]] haelt es explizit offen. [[ENDE]] gewinnt gegen [[WEITER]]."""
|
||||||
|
if not text:
|
||||||
|
return text, None, None
|
||||||
|
speak_ov = None
|
||||||
|
conv_ov = None
|
||||||
|
if _SILENT_MARKER_RE.search(text):
|
||||||
|
speak_ov = False
|
||||||
|
text = _SILENT_MARKER_RE.sub("", text)
|
||||||
|
if _END_MARKER_RE.search(text):
|
||||||
|
conv_ov = False
|
||||||
|
text = _END_MARKER_RE.sub("", text)
|
||||||
|
if _CONT_MARKER_RE.search(text):
|
||||||
|
# [[ENDE]] hat Vorrang — widerspruechliche Marker → beenden.
|
||||||
|
if conv_ov is None:
|
||||||
|
conv_ov = True
|
||||||
|
text = _CONT_MARKER_RE.sub("", text)
|
||||||
|
# Stiller Einzelbefehl ohne explizites Weiterlauschen → Mikro zu.
|
||||||
|
if speak_ov is False and conv_ov is None:
|
||||||
|
conv_ov = False
|
||||||
|
return text.strip(), speak_ov, conv_ov
|
||||||
|
|
||||||
|
|
||||||
|
# Explizite "Konversation beenden"-Phrasen vom USER — deterministisch, NICHT auf
|
||||||
|
# ARIAs [[ENDE]]-Marker angewiesen. Stefan will "Konversation Ende" o.ae. als
|
||||||
|
# festen Trigger: danach zurueck aufs Wake-Word, egal was ARIA sonst tut. Eine in
|
||||||
|
# derselben Nachricht enthaltene Frage beantwortet sie normal (wird vorgelesen),
|
||||||
|
# aber converse wird auf false gezwungen. Nomen + Ende-Wort in EINEM Satzteil
|
||||||
|
# ([^.!?]{0,15}) in beliebiger Reihenfolge; "befehls?kette" damit "Lieferkette"
|
||||||
|
# o.ae. nicht faelschlich matcht.
|
||||||
|
_CONV_NOUN = r"(?:konversation|gespr[aä]ch|befehls?kette)"
|
||||||
|
_CONV_END_VERB = r"(?:ende|beend\w*|aus|stop\w*|schluss)"
|
||||||
|
_END_CONVERSATION_RE = re.compile(
|
||||||
|
rf"\b{_CONV_NOUN}\b[^.!?]{{0,15}}\b{_CONV_END_VERB}\b"
|
||||||
|
rf"|\b(?:beend\w*|schlie(?:ß|ss)\w*)\b[^.!?]{{0,15}}\b{_CONV_NOUN}\b",
|
||||||
|
re.IGNORECASE,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _user_wants_conversation_end(text: str) -> bool:
|
||||||
|
"""True, wenn der User in dieser Nachricht explizit die Konversation/Kette
|
||||||
|
beenden will (deterministisch, unabhaengig vom LLM-Marker)."""
|
||||||
|
if not text:
|
||||||
|
return False
|
||||||
|
return bool(_END_CONVERSATION_RE.search(_strip_leading_hint_blocks(text)))
|
||||||
|
|
||||||
|
|
||||||
|
# Gegenstueck zu _END: expliziter "Konversation OFFEN halten / fortfuehren"-Wunsch.
|
||||||
|
# Wichtig fuer BEFEHLE die den Fast-Path treffen: "spiel Spotify ab ABER Konversation
|
||||||
|
# fortfuehren" — der Fast-Path (Regex) versteht den Satz-Rest nicht und wuerde mit
|
||||||
|
# converse=false schliessen. Dieser Detektor erzwingt converse=true, auch am
|
||||||
|
# Fast-Path, egal was das Skill-Manifest sagt. Nomen+Verb in einem Satzteil, plus
|
||||||
|
# "weiter reden/sprechen" ohne Nomen.
|
||||||
|
_CONT_VERB = (r"(?:fortf[uü]hr\w*|fortsetz\w*|weiterf[uü]hr\w*|weiter\s*mach\w*|"
|
||||||
|
r"weiter\b|fort\b|offen\s+(?:halten|lassen)|nicht\s+beenden|weiterlauf\w*)")
|
||||||
|
_CONTINUE_CONVERSATION_RE = re.compile(
|
||||||
|
rf"\b{_CONV_NOUN}\b[^.!?]{{0,20}}\b{_CONT_VERB}"
|
||||||
|
rf"|\b{_CONT_VERB}[^.!?]{{0,20}}\b{_CONV_NOUN}\b"
|
||||||
|
rf"|\bweiter\s*(?:reden|sprechen|quatschen|plaudern|labern)\b",
|
||||||
|
re.IGNORECASE,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _user_wants_conversation_continue(text: str) -> bool:
|
||||||
|
"""True, wenn der User explizit weiter im Gespraech bleiben will (converse=true
|
||||||
|
erzwingen — auch bei einem Fast-Path-Befehl). [[ENDE]]/_wants_end hat Vorrang."""
|
||||||
|
if not text:
|
||||||
|
return False
|
||||||
|
return bool(_CONTINUE_CONVERSATION_RE.search(_strip_leading_hint_blocks(text)))
|
||||||
|
|
||||||
|
|
||||||
|
# ── Wake-Word AUS per Sprachbefehl ──────────────────────────────────────────
|
||||||
|
# "Wake-Word aus", "mach das Ohr aus", "hoer auf zuzuhoeren", "geh schlafen" …
|
||||||
|
# → die App stoppt den Wake-Word-Listener KOMPLETT (Mikro frei, echte Ruhe).
|
||||||
|
# Wieder-An geht nur ueber den App-Button (bewusst, weil dann taub). Reiner
|
||||||
|
# Steuerbefehl: still (speak=false), kein Weiterlauschen (converse=false).
|
||||||
|
_WAKE_NOUN = r"(?:wake[\s-]?word|wakeword|ohr(?:en)?|mikro(?:fon)?|zuh[oö]r\w*|lausch\w*)"
|
||||||
|
_WAKE_OFF_VERB = (r"(?:aus(?:schalten|stellen)?|abschalten|abstellen|deaktivier\w*|"
|
||||||
|
r"beenden|beende|stopp?\w*|schlafen|ruhe)")
|
||||||
|
_WAKE_OFF_RE = re.compile(
|
||||||
|
rf"\b{_WAKE_NOUN}\b[^.!?]{{0,20}}\b{_WAKE_OFF_VERB}\b"
|
||||||
|
rf"|\b(?:beende|beend\w*|deaktivier\w*|stopp?e?)\b[^.!?]{{0,20}}\b{_WAKE_NOUN}\b"
|
||||||
|
rf"|h[oö]r\s+auf\s+(?:zu\s*)?(?:zu)?(?:h[oö]r|lausch)\w*"
|
||||||
|
rf"|h[oö]r\s+nicht\s+mehr\s+zu"
|
||||||
|
# Schlaf-Idiome NUR als klare Imperative an ARIA ('geh/leg dich schlafen',
|
||||||
|
# 'leg dich aufs Ohr/hin'). Mehrdeutige Gruesse ('gute Nacht', 'schlaf gut')
|
||||||
|
# bewusst NICHT — das sind Verabschiedungen, kein Ohr-Aus-Befehl; die faengt
|
||||||
|
# ARIA per ear_control aus dem Kontext ab, wenn's wirklich gemeint ist.
|
||||||
|
rf"|(?:geh|leg\s+dich)\s+schlafen"
|
||||||
|
rf"|leg\s+dich\s+(?:aufs?\s+ohr|hin)",
|
||||||
|
re.IGNORECASE,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _user_wants_wake_off(text: str) -> bool:
|
||||||
|
"""True, wenn der User den Wake-Word-Listener per Sprache abschalten will."""
|
||||||
|
if not text:
|
||||||
|
return False
|
||||||
|
return bool(_WAKE_OFF_RE.search(_strip_leading_hint_blocks(text)))
|
||||||
|
|
||||||
|
|
||||||
|
# ── Wake-Word AN per Befehl (Text ODER manueller Aufnahme-Button) ────────────
|
||||||
|
# Gegenstueck zu wake_off. Das "Wieder-An" per Stimme geht NICHT ueber "Computer"
|
||||||
|
# (das hoert ja nicht mehr), aber ueber eine Text-Nachricht oder den manuellen
|
||||||
|
# Aufnahme-Button — beide laufen unabhaengig vom Wake-Word-Listener. Die App
|
||||||
|
# startet den Listener dann wieder (wakeWordService.start()).
|
||||||
|
_WAKE_ON_VERB = r"(?:an(?:schalten|machen|stellen)?|einschalten|aktivier\w*|starte\w*|reaktivier\w*)"
|
||||||
|
_WAKE_ON_RE = re.compile(
|
||||||
|
rf"\b{_WAKE_NOUN}\b[^.!?]{{0,20}}\b{_WAKE_ON_VERB}\b"
|
||||||
|
rf"|\b(?:aktivier\w*|reaktivier\w*|starte\w*)\b[^.!?]{{0,15}}\b{_WAKE_NOUN}\b"
|
||||||
|
rf"|h[oö]r\s+(?:mir\s+)?wieder\s+zu"
|
||||||
|
rf"|(?:wieder|erneut)\s+(?:zu\s*)?(?:h[oö]r|lausch)\w*"
|
||||||
|
rf"|wach\s+auf|aufwachen",
|
||||||
|
re.IGNORECASE,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _user_wants_wake_on(text: str) -> bool:
|
||||||
|
"""True, wenn der User den Wake-Word-Listener per Befehl wieder anschalten will."""
|
||||||
|
if not text:
|
||||||
|
return False
|
||||||
|
return bool(_WAKE_ON_RE.search(_strip_leading_hint_blocks(text)))
|
||||||
|
|
||||||
|
|
||||||
def _normalize_for_fast_match(text: str) -> str:
|
def _normalize_for_fast_match(text: str) -> str:
|
||||||
norm = _strip_leading_hint_blocks(text).lower()
|
norm = _strip_leading_hint_blocks(text).lower()
|
||||||
norm = _fold_umlauts(norm)
|
norm = _fold_umlauts(norm)
|
||||||
@@ -1355,6 +1599,17 @@ def _skill_to_tool(s: dict) -> dict:
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
# Standort-Funktionen in Watcher-Conditions — brauchen aktives GPS-Tracking.
|
||||||
|
_LOCATION_FUNCS = ("near(", "entered_near(", "left_near(")
|
||||||
|
|
||||||
|
|
||||||
|
def _condition_needs_location(condition: str) -> bool:
|
||||||
|
"""True, wenn die Watcher-Condition eine Standort-Funktion nutzt und damit
|
||||||
|
laufende GPS-Updates der App braucht."""
|
||||||
|
c = condition or ""
|
||||||
|
return any(fn in c for fn in _LOCATION_FUNCS)
|
||||||
|
|
||||||
|
|
||||||
class Agent:
|
class Agent:
|
||||||
# Mindest-Score den ein Cold-Memory-Treffer haben muss um in den
|
# Mindest-Score den ein Cold-Memory-Treffer haben muss um in den
|
||||||
# System-Prompt aufgenommen zu werden. Unter dieser Schwelle ist's
|
# System-Prompt aufgenommen zu werden. Unter dieser Schwelle ist's
|
||||||
@@ -1488,9 +1743,12 @@ class Agent:
|
|||||||
return False
|
return False
|
||||||
|
|
||||||
# Kuratierte Tool-Auswahl fuers lokale Tier (B1b): web_search (local-only,
|
# Kuratierte Tool-Auswahl fuers lokale Tier (B1b): web_search (local-only,
|
||||||
# SearXNG) + memory_search/trigger_timer (aus META_TOOLS) + Spotify-Skill.
|
# SearXNG) + memory_search/trigger_timer (aus META_TOOLS) + Spotify-Skill +
|
||||||
# Bewusst klein (Speed + Sicherheit); alles andere → Claude.
|
# die Satelliten-Tools (Augen/Haende in fremden Netzen — auch das lokale Tier
|
||||||
_LOCAL_TOOL_NAMES = {"memory_search", "trigger_timer"}
|
# muss ein Geraet im Heim-/Buero-Netz erreichen koennen, statt zu fabulieren).
|
||||||
|
# Sonst bewusst klein (Speed + Sicherheit); alles andere → Claude.
|
||||||
|
_LOCAL_TOOL_NAMES = {"memory_search", "trigger_timer",
|
||||||
|
"satellite_list", "satellite_devices", "satellite_command"}
|
||||||
|
|
||||||
def _build_local_tools(self) -> list:
|
def _build_local_tools(self) -> list:
|
||||||
tools = [WEB_SEARCH_TOOL]
|
tools = [WEB_SEARCH_TOOL]
|
||||||
@@ -1701,6 +1959,14 @@ class Agent:
|
|||||||
if not user_message:
|
if not user_message:
|
||||||
raise ValueError("Leere Nachricht")
|
raise ValueError("Leere Nachricht")
|
||||||
|
|
||||||
|
# Explizite Gespraechs-Steuerung vom USER (deterministisch, an JEDEM Return
|
||||||
|
# angewendet — auch am Fast-Path, den die LLM-Marker nicht erreichen):
|
||||||
|
# _wants_end → converse=false ("Konversation Ende")
|
||||||
|
# _wants_continue → converse=true ("... aber Konversation fortfuehren")
|
||||||
|
# End hat Vorrang bei Widerspruch.
|
||||||
|
_wants_end = _user_wants_conversation_end(user_message)
|
||||||
|
_wants_continue = (not _wants_end) and _user_wants_conversation_continue(user_message)
|
||||||
|
|
||||||
# Events vom letzten Turn weglassen
|
# Events vom letzten Turn weglassen
|
||||||
self._pending_events = []
|
self._pending_events = []
|
||||||
|
|
||||||
@@ -1708,6 +1974,27 @@ class Agent:
|
|||||||
active_project_id = (project_id or "").strip()
|
active_project_id = (project_id or "").strip()
|
||||||
active_project = projects_mod.get_project(active_project_id) if active_project_id else None
|
active_project = projects_mod.get_project(active_project_id) if active_project_id else None
|
||||||
|
|
||||||
|
# Wake-Word AUS per Sprache: reiner Steuerbefehl, KEIN Claude/Fast-Path
|
||||||
|
# noetig. Still (speak=false) + kein Weiterlauschen (converse=false). Das
|
||||||
|
# tatsaechliche Stoppen des Listeners macht die App anhand von wake_off in
|
||||||
|
# der Antwort (ChatOut) — hier signalisieren wir es nur. Wieder-An: App-Button.
|
||||||
|
if _user_wants_wake_off(user_message):
|
||||||
|
reply = "Ohr aus. Sag 'Wake-Word an' (per Text oder Aufnahme-Knopf) oder tipp den Ohr-Button, wenn ich wieder lauschen soll. 🔇"
|
||||||
|
self.conversation.add("user", user_message, source=source,
|
||||||
|
project_id=active_project_id)
|
||||||
|
self.conversation.add("assistant", reply, project_id=active_project_id)
|
||||||
|
logger.info("[wake-off] Sprachbefehl erkannt — App stoppt Listener")
|
||||||
|
return reply, "wake-off", False, False, False
|
||||||
|
|
||||||
|
# Wake-Word AN per Befehl (Text/Aufnahme-Button): App startet den Listener.
|
||||||
|
if _user_wants_wake_on(user_message):
|
||||||
|
reply = "Ohr wieder an — ich lausche auf 'Computer'. 👂"
|
||||||
|
self.conversation.add("user", user_message, source=source,
|
||||||
|
project_id=active_project_id)
|
||||||
|
self.conversation.add("assistant", reply, project_id=active_project_id)
|
||||||
|
logger.info("[wake-on] Befehl erkannt — App startet Listener")
|
||||||
|
return reply, "wake-on", False, False, False
|
||||||
|
|
||||||
# Fast-Path: einfache "reines Steuern"-Commands ueberspringen Claude komplett.
|
# Fast-Path: einfache "reines Steuern"-Commands ueberspringen Claude komplett.
|
||||||
# Jeder Skill kann in seinem Manifest fast_patterns deklarieren — das Brain
|
# Jeder Skill kann in seinem Manifest fast_patterns deklarieren — das Brain
|
||||||
# iteriert hier ueber alle aktiven Skills und matched. Spart 5-10s Latenz.
|
# iteriert hier ueber alle aktiven Skills und matched. Spart 5-10s Latenz.
|
||||||
@@ -1728,6 +2015,10 @@ class Agent:
|
|||||||
speak = bool(getattr(self, "_fast_path_speak", False))
|
speak = bool(getattr(self, "_fast_path_speak", False))
|
||||||
# converse folgt dem Skill (Manifest/Output) — nicht mehr generell False.
|
# converse folgt dem Skill (Manifest/Output) — nicht mehr generell False.
|
||||||
converse = bool(getattr(self, "_fast_path_converse", False))
|
converse = bool(getattr(self, "_fast_path_converse", False))
|
||||||
|
if _wants_end:
|
||||||
|
converse = False
|
||||||
|
elif _wants_continue:
|
||||||
|
converse = True
|
||||||
# Fast-Path = reiner Steuerbefehl, nie eine Rueckfrage → awaiting=False.
|
# Fast-Path = reiner Steuerbefehl, nie eine Rueckfrage → awaiting=False.
|
||||||
return fast_reply, "fast-path", speak, converse, False
|
return fast_reply, "fast-path", speak, converse, False
|
||||||
|
|
||||||
@@ -1747,6 +2038,10 @@ class Agent:
|
|||||||
# dem Skill (bzw. Default: Info/Gespraech = vorlesen + 30s).
|
# dem Skill (bzw. Default: Info/Gespraech = vorlesen + 30s).
|
||||||
speak = getattr(self, "_local_turn_speak", True)
|
speak = getattr(self, "_local_turn_speak", True)
|
||||||
converse = getattr(self, "_local_turn_converse", True)
|
converse = getattr(self, "_local_turn_converse", True)
|
||||||
|
if _wants_end:
|
||||||
|
converse = False
|
||||||
|
elif _wants_continue:
|
||||||
|
converse = True
|
||||||
# Local ist tool-loses Reden; blockierende Rueckfragen macht Claude.
|
# Local ist tool-loses Reden; blockierende Rueckfragen macht Claude.
|
||||||
return local_reply, "local", speak, converse, False
|
return local_reply, "local", speak, converse, False
|
||||||
|
|
||||||
@@ -1764,6 +2059,15 @@ class Agent:
|
|||||||
logger.warning("Cold-Search fehlgeschlagen: %s", exc)
|
logger.warning("Cold-Search fehlgeschlagen: %s", exc)
|
||||||
cold = []
|
cold = []
|
||||||
|
|
||||||
|
# 3b. Titel-Index des kalten Gedaechtnisses — ARIA sieht WAS sie an
|
||||||
|
# Nachschlage-Wissen hat (Zugangsdaten, Infra, Projekte) und holt es via
|
||||||
|
# memory_search, statt Stefan danach zu fragen. Nur Titel = billig.
|
||||||
|
try:
|
||||||
|
memory_index = self.store.list_index_titles()
|
||||||
|
except Exception as exc:
|
||||||
|
logger.warning("Titel-Index laden fehlgeschlagen: %s", exc)
|
||||||
|
memory_index = []
|
||||||
|
|
||||||
# 4. Aktive Skills holen + Tool-Liste bauen
|
# 4. Aktive Skills holen + Tool-Liste bauen
|
||||||
all_skills = skills_mod.list_skills(active_only=False)
|
all_skills = skills_mod.list_skills(active_only=False)
|
||||||
active_skills = [s for s in all_skills if s.get("active", True)]
|
active_skills = [s for s in all_skills if s.get("active", True)]
|
||||||
@@ -1786,7 +2090,8 @@ class Agent:
|
|||||||
oauth_port = os.environ.get("RVS_PORT_PUBLIC", os.environ.get("RVS_PORT", "443")).strip()
|
oauth_port = os.environ.get("RVS_PORT_PUBLIC", os.environ.get("RVS_PORT", "443")).strip()
|
||||||
oauth_tls = os.environ.get("RVS_TLS", "true").strip().lower() != "false"
|
oauth_tls = os.environ.get("RVS_TLS", "true").strip().lower() != "false"
|
||||||
|
|
||||||
system_prompt = build_system_prompt(hot, cold, skills=all_skills,
|
system_prompt = build_system_prompt(hot, cold, memory_index=memory_index,
|
||||||
|
skills=all_skills,
|
||||||
triggers=all_triggers,
|
triggers=all_triggers,
|
||||||
condition_vars=condition_vars,
|
condition_vars=condition_vars,
|
||||||
condition_funcs=condition_funcs,
|
condition_funcs=condition_funcs,
|
||||||
@@ -1888,6 +2193,7 @@ class Agent:
|
|||||||
# Konversation bleiben gesprochen.
|
# Konversation bleiben gesprochen.
|
||||||
self._claude_turn_speak = True
|
self._claude_turn_speak = True
|
||||||
self._claude_turn_converse = True # Default Gespraech; run_*-Skill setzt es
|
self._claude_turn_converse = True # Default Gespraech; run_*-Skill setzt es
|
||||||
|
self._ear_control = None # ear_control-Tool: 'off'|'on' oder None
|
||||||
try:
|
try:
|
||||||
for iteration in range(self.MAX_TOOL_ITERATIONS):
|
for iteration in range(self.MAX_TOOL_ITERATIONS):
|
||||||
result = self.proxy.chat_full(messages, tools=tools,
|
result = self.proxy.chat_full(messages, tools=tools,
|
||||||
@@ -1979,16 +2285,38 @@ class Agent:
|
|||||||
# Rueckfrage-Marker aus dem finalen Text ziehen (vor History/Return, damit
|
# Rueckfrage-Marker aus dem finalen Text ziehen (vor History/Return, damit
|
||||||
# er nicht angezeigt/vorgelesen wird und nicht die Conversation vergiftet).
|
# er nicht angezeigt/vorgelesen wird und nicht die Conversation vergiftet).
|
||||||
final_reply, awaiting_reply = _extract_await_marker(final_reply)
|
final_reply, awaiting_reply = _extract_await_marker(final_reply)
|
||||||
|
# ARIAs Phasen-Marker ([[STUMM]]/[[WEITER]]/[[ENDE]]) ziehen — VOR History,
|
||||||
|
# damit sie nicht angezeigt/vorgelesen/gespeichert werden.
|
||||||
|
final_reply, _speak_ov, _conv_ov = _extract_flow_markers(final_reply)
|
||||||
|
|
||||||
# 7. Assistant-Turn (final reply) in die Conversation
|
# 7. Assistant-Turn (final reply) in die Conversation
|
||||||
self.conversation.add("assistant", final_reply,
|
self.conversation.add("assistant", final_reply,
|
||||||
project_id=active_project_id)
|
project_id=active_project_id)
|
||||||
# speak/converse folgen dem ausgefuehrten Skill (sonst Default: Gespraech);
|
# speak/converse folgen dem ausgefuehrten Skill (sonst Default: Gespraech).
|
||||||
|
# ARIAs Phasen-Marker sind AUTORITATIV: sie kennt aus dem Text den Unter-
|
||||||
|
# schied Befehl/Frage und Kette/Ende, den das Skill-Manifest nicht kennt.
|
||||||
|
speak = bool(getattr(self, "_claude_turn_speak", True))
|
||||||
|
converse = bool(getattr(self, "_claude_turn_converse", True))
|
||||||
|
if _speak_ov is not None:
|
||||||
|
speak = _speak_ov
|
||||||
|
if _conv_ov is not None:
|
||||||
|
converse = _conv_ov
|
||||||
|
# Explizite User-Woerter gewinnen ueber Marker/Manifest: "Konversation
|
||||||
|
# beenden" → zu; "... fortfuehren" → offen halten. End hat Vorrang.
|
||||||
|
if _wants_end:
|
||||||
|
converse = False
|
||||||
|
elif _wants_continue:
|
||||||
|
converse = True
|
||||||
|
# ear_control-Tool aufgerufen → Ohr schalten. answered_by=wake-off/wake-on
|
||||||
|
# nutzt die bestehende Plumbing (main.py → wake_off/wake_on → Bridge → App).
|
||||||
|
# Reiner Steuerbefehl: still + kein Weiterlauschen.
|
||||||
|
answered_by = "claude"
|
||||||
|
if self._ear_control == "off":
|
||||||
|
answered_by, speak, converse = "wake-off", False, False
|
||||||
|
elif self._ear_control == "on":
|
||||||
|
answered_by, speak, converse = "wake-on", False, False
|
||||||
# awaiting_reply = ARIA stellt eine blockierende Rueckfrage (Queue pausiert).
|
# awaiting_reply = ARIA stellt eine blockierende Rueckfrage (Queue pausiert).
|
||||||
return (final_reply, "claude",
|
return (final_reply, answered_by, speak, converse, awaiting_reply)
|
||||||
bool(getattr(self, "_claude_turn_speak", True)),
|
|
||||||
bool(getattr(self, "_claude_turn_converse", True)),
|
|
||||||
awaiting_reply)
|
|
||||||
|
|
||||||
# ── Tool-Dispatcher ───────────────────────────────────────
|
# ── Tool-Dispatcher ───────────────────────────────────────
|
||||||
|
|
||||||
@@ -2372,13 +2700,41 @@ class Agent:
|
|||||||
"trigger": {"name": t["name"], "type": "watcher",
|
"trigger": {"name": t["name"], "type": "watcher",
|
||||||
"condition": t["condition"], "message": t["message"]},
|
"condition": t["condition"], "message": t["message"]},
|
||||||
})
|
})
|
||||||
return f"OK — Watcher '{t['name']}' angelegt: feuert wenn '{t['condition']}'."
|
# GPS-Kopplung: ein Standort-Watcher (near/entered_near/left_near)
|
||||||
|
# ist tot, wenn die App kein Tracking sendet. Frueher musste ARIA
|
||||||
|
# separat request_location_tracking aufrufen — wurde oft vergessen,
|
||||||
|
# dann feuerte der Trigger nie. Jetzt automatisch mit-anschalten.
|
||||||
|
extra = ""
|
||||||
|
if _condition_needs_location(t["condition"]):
|
||||||
|
self._pending_events.append({
|
||||||
|
"type": "location_tracking",
|
||||||
|
"on": True,
|
||||||
|
"reason": f"watcher:{t['name']}",
|
||||||
|
})
|
||||||
|
extra = " GPS-Tracking automatisch aktiviert."
|
||||||
|
return f"OK — Watcher '{t['name']}' angelegt: feuert wenn '{t['condition']}'.{extra}"
|
||||||
if name == "trigger_cancel":
|
if name == "trigger_cancel":
|
||||||
try:
|
try:
|
||||||
triggers_mod.delete(arguments["name"])
|
triggers_mod.delete(arguments["name"])
|
||||||
return f"OK — Trigger '{arguments['name']}' geloescht."
|
|
||||||
except ValueError as e:
|
except ValueError as e:
|
||||||
return f"FEHLER: {e}"
|
return f"FEHLER: {e}"
|
||||||
|
# GPS-Kopplung (Gegenstueck): existiert kein Standort-Watcher mehr,
|
||||||
|
# Tracking wieder ausschalten (Akku schonen).
|
||||||
|
extra = ""
|
||||||
|
remaining = triggers_mod.list_triggers(active_only=False)
|
||||||
|
still_needs_gps = any(
|
||||||
|
r.get("type") == "watcher"
|
||||||
|
and _condition_needs_location(r.get("condition") or "")
|
||||||
|
for r in remaining
|
||||||
|
)
|
||||||
|
if not still_needs_gps:
|
||||||
|
self._pending_events.append({
|
||||||
|
"type": "location_tracking",
|
||||||
|
"on": False,
|
||||||
|
"reason": "no-location-watchers",
|
||||||
|
})
|
||||||
|
extra = " GPS-Tracking deaktiviert (kein Standort-Watcher mehr)."
|
||||||
|
return f"OK — Trigger '{arguments['name']}' geloescht.{extra}"
|
||||||
if name == "request_location_tracking":
|
if name == "request_location_tracking":
|
||||||
on = bool(arguments.get("on", False))
|
on = bool(arguments.get("on", False))
|
||||||
reason = (arguments.get("reason") or "").strip()
|
reason = (arguments.get("reason") or "").strip()
|
||||||
@@ -2388,6 +2744,22 @@ class Agent:
|
|||||||
"reason": reason,
|
"reason": reason,
|
||||||
})
|
})
|
||||||
return f"OK — Tracking-Request gesendet (on={on}). App wird in Kuerze umschalten."
|
return f"OK — Tracking-Request gesendet (on={on}). App wird in Kuerze umschalten."
|
||||||
|
if name == "present_view":
|
||||||
|
cards = arguments.get("cards") or []
|
||||||
|
if not isinstance(cards, list) or not cards:
|
||||||
|
return "FEHLER: present_view braucht mindestens eine Karte in 'cards'."
|
||||||
|
view = {
|
||||||
|
"cards": cards,
|
||||||
|
"orb": arguments.get("orb") or "speaking",
|
||||||
|
"title": arguments.get("title") or "",
|
||||||
|
}
|
||||||
|
self._pending_events.append({
|
||||||
|
"type": "aria_view",
|
||||||
|
"view": view,
|
||||||
|
"project_id": project_id or "",
|
||||||
|
})
|
||||||
|
return (f"OK — Ansicht mit {len(cards)} Karte(n) an die Flaeche geschickt "
|
||||||
|
f"(Kontext: {project_id or 'Hauptchat'}).")
|
||||||
if name == "trigger_list":
|
if name == "trigger_list":
|
||||||
items = triggers_mod.list_triggers(active_only=False)
|
items = triggers_mod.list_triggers(active_only=False)
|
||||||
if not items:
|
if not items:
|
||||||
@@ -2539,6 +2911,15 @@ class Agent:
|
|||||||
f"WICHTIG: Schreibe in deiner Antwort an Stefan den Pfad EXAKT als "
|
f"WICHTIG: Schreibe in deiner Antwort an Stefan den Pfad EXAKT als "
|
||||||
f"Marker: [FILE: {result['path']}] — dann zeigt die App das Bild inline."
|
f"Marker: [FILE: {result['path']}] — dann zeigt die App das Bild inline."
|
||||||
)
|
)
|
||||||
|
if name == "ear_control":
|
||||||
|
action = (arguments.get("action") or "").strip().lower()
|
||||||
|
if action not in ("off", "on"):
|
||||||
|
return "FEHLER: action muss 'off' oder 'on' sein."
|
||||||
|
self._ear_control = action
|
||||||
|
logger.info("[ear_control] action=%s — App schaltet den Listener", action)
|
||||||
|
return ("Ohr wird ausgeschaltet — Wieder-An per 'Ohr an' (Text/"
|
||||||
|
"Aufnahme-Knopf) oder App-Button." if action == "off"
|
||||||
|
else "Ohr wird wieder eingeschaltet.")
|
||||||
if name == "memory_search":
|
if name == "memory_search":
|
||||||
query = (arguments.get("query") or "").strip()
|
query = (arguments.get("query") or "").strip()
|
||||||
if not query:
|
if not query:
|
||||||
|
|||||||
@@ -0,0 +1,90 @@
|
|||||||
|
"""Einmaliger Backfill: weist bestehenden Memory-Punkten ein `scope`
|
||||||
|
(system | personal) zu. Sicher & reversibel — Stefan kann pro Eintrag in der
|
||||||
|
Diagnostic-UI umschalten. Idempotent: laeuft mehrfach ohne Schaden.
|
||||||
|
|
||||||
|
Heuristik (datengetrieben aus dem realen Bestand):
|
||||||
|
- type=preference / fact / conversation / reminder -> personal
|
||||||
|
- source in (seed, auto-feedback) -> system
|
||||||
|
- type=identity -> system
|
||||||
|
- type in (rule, tool, skill) und category in SYSTEM_CATS -> system
|
||||||
|
- sonst -> personal (sicher: nichts leakt)
|
||||||
|
|
||||||
|
Aufruf im Brain-Container:
|
||||||
|
docker exec aria-brain python3 /app/backfill_scope.py # dry-run
|
||||||
|
docker exec aria-brain python3 /app/backfill_scope.py --apply # schreibt
|
||||||
|
"""
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
from collections import Counter
|
||||||
|
|
||||||
|
from qdrant_client import QdrantClient
|
||||||
|
from qdrant_client.http import models as qm
|
||||||
|
|
||||||
|
COLLECTION = "aria_memory"
|
||||||
|
SYSTEM_CATS = {
|
||||||
|
"sicherheit", "arbeitsweise", "architektur", "ehrlichkeit", "verhalten",
|
||||||
|
"voice", "skills", "freigaben", "infrastruktur", "persoenlichkeit",
|
||||||
|
"pentest", "ausgabe",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def compute_scope(pl: dict) -> str:
|
||||||
|
typ = pl.get("type")
|
||||||
|
src = pl.get("source")
|
||||||
|
cat = (pl.get("category") or "").lower()
|
||||||
|
if typ == "preference":
|
||||||
|
return "personal"
|
||||||
|
if typ in ("fact", "conversation", "reminder"):
|
||||||
|
return "personal"
|
||||||
|
if src in ("seed", "auto-feedback"):
|
||||||
|
return "system"
|
||||||
|
if typ == "identity":
|
||||||
|
return "system"
|
||||||
|
if typ in ("rule", "tool", "skill") and cat in SYSTEM_CATS:
|
||||||
|
return "system"
|
||||||
|
return "personal"
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
apply = "--apply" in sys.argv
|
||||||
|
force = "--force" in sys.argv # auch schon gesetzte scopes ueberschreiben
|
||||||
|
c = QdrantClient(
|
||||||
|
host=os.environ.get("QDRANT_HOST", "aria-qdrant"),
|
||||||
|
port=int(os.environ.get("QDRANT_PORT", "6333")),
|
||||||
|
)
|
||||||
|
pts, _ = c.scroll(collection_name=COLLECTION, limit=5000,
|
||||||
|
with_payload=True, with_vectors=False)
|
||||||
|
|
||||||
|
per_scope: dict[str, list] = {"system": [], "personal": []}
|
||||||
|
pinned_examples = Counter()
|
||||||
|
skipped = 0
|
||||||
|
for p in pts:
|
||||||
|
pl = p.payload or {}
|
||||||
|
if pl.get("scope") in ("system", "personal") and not force:
|
||||||
|
skipped += 1
|
||||||
|
continue
|
||||||
|
scope = compute_scope(pl)
|
||||||
|
per_scope[scope].append(p.id)
|
||||||
|
if pl.get("pinned"):
|
||||||
|
pinned_examples[(scope, pl.get("source"), pl.get("type"),
|
||||||
|
pl.get("category"))] += 1
|
||||||
|
|
||||||
|
print(f"total={len(pts)} skipped(already set)={skipped}")
|
||||||
|
print(f"-> system={len(per_scope['system'])} personal={len(per_scope['personal'])}")
|
||||||
|
print("pinned split (scope, source, type, category):")
|
||||||
|
for k, v in sorted(pinned_examples.items()):
|
||||||
|
print(" ", k, v)
|
||||||
|
|
||||||
|
if not apply:
|
||||||
|
print("\nDRY-RUN — nichts geschrieben. Mit --apply ausfuehren.")
|
||||||
|
return
|
||||||
|
|
||||||
|
for scope, ids in per_scope.items():
|
||||||
|
if not ids:
|
||||||
|
continue
|
||||||
|
c.set_payload(collection_name=COLLECTION, payload={"scope": scope}, points=ids)
|
||||||
|
print(f"\nAPPLIED: system={len(per_scope['system'])} personal={len(per_scope['personal'])}")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -149,14 +149,23 @@ async def _fire(trigger: dict, agent_factory) -> None:
|
|||||||
)
|
)
|
||||||
|
|
||||||
try:
|
try:
|
||||||
agent = agent_factory()
|
# WICHTIG: agent.chat() ist ein SYNCHRONER, blockierender Aufruf (Proxy-
|
||||||
reply, _, _, _, _ = agent.chat(prompt, source="trigger")
|
# HTTP mit bis zu 24h Read-Timeout). NIEMALS direkt im async-Loop —
|
||||||
events = agent.pop_events()
|
# sonst friert ein einziger getriggerter Turn den GESAMTEN Brain ein
|
||||||
|
# (kein /health, kein weiterer Request). Wie der /chat-Pfad in den
|
||||||
|
# Executor auslagern, damit der Event-Loop frei bleibt.
|
||||||
|
loop = asyncio.get_running_loop()
|
||||||
|
|
||||||
|
def _run_turn():
|
||||||
|
a = agent_factory()
|
||||||
|
rep, *_rest = a.chat(prompt, source="trigger")
|
||||||
|
return rep, a.pop_events()
|
||||||
|
|
||||||
|
reply, events = await loop.run_in_executor(None, _run_turn)
|
||||||
logger.info("[trigger] %s gefeuert → ARIA-Reply: %s", name, reply[:80])
|
logger.info("[trigger] %s gefeuert → ARIA-Reply: %s", name, reply[:80])
|
||||||
triggers_mod.append_log(name, {"event": "reply", "text": reply[:500]})
|
triggers_mod.append_log(name, {"event": "reply", "text": reply[:500]})
|
||||||
# Reply an die Bridge pushen, damit App + Diagnostic + TTS sie kriegen.
|
# Reply an die Bridge pushen, damit App + Diagnostic + TTS sie kriegen.
|
||||||
# Ohne diesen Push wuerde die Antwort nur im Brain-Log landen.
|
# Ohne diesen Push wuerde die Antwort nur im Brain-Log landen.
|
||||||
loop = asyncio.get_event_loop()
|
|
||||||
await loop.run_in_executor(None, _push_to_bridge, reply, name, ttype, events)
|
await loop.run_in_executor(None, _push_to_bridge, reply, name, ttype, events)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.exception("Trigger %s feuern fehlgeschlagen: %s", name, e)
|
logger.exception("Trigger %s feuern fehlgeschlagen: %s", name, e)
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
"""
|
"""
|
||||||
Local-LLM-Client (Plan B) — Brain-Seite.
|
Local-LLM-Client (Plan B) — Brain-Seite.
|
||||||
|
|
||||||
Ruft das schnelle lokale LLM (Qwen3 auf der Gamebox) ueber die Bridge:
|
Ruft das schnelle lokale LLM (Qwen3 auf der AI-Box) ueber die Bridge:
|
||||||
Brain → HTTP /internal/local-llm → Bridge → RVS → llm-adapter → llama.cpp
|
Brain → HTTP /internal/local-llm → Bridge → RVS → llm-adapter → llama.cpp
|
||||||
|
|
||||||
Analog zum Claude-`proxy_client`, nur ueber die Bridge (die ist der RVS-Client;
|
Analog zum Claude-`proxy_client`, nur ueber die Bridge (die ist der RVS-Client;
|
||||||
|
|||||||
+52
-13
@@ -190,6 +190,7 @@ class MemoryIn(BaseModel):
|
|||||||
pinned: bool = False
|
pinned: bool = False
|
||||||
category: str = ""
|
category: str = ""
|
||||||
source: str = "manual"
|
source: str = "manual"
|
||||||
|
scope: str = "personal" # system | personal — steuert Bootstrap-Export
|
||||||
tags: List[str] = Field(default_factory=list)
|
tags: List[str] = Field(default_factory=list)
|
||||||
conversation_id: Optional[str] = None
|
conversation_id: Optional[str] = None
|
||||||
# Vorhandene Anhang-Metadaten beim Save mitgeben (i.d.R. werden Anhaenge
|
# Vorhandene Anhang-Metadaten beim Save mitgeben (i.d.R. werden Anhaenge
|
||||||
@@ -203,6 +204,7 @@ class MemoryUpdate(BaseModel):
|
|||||||
content: Optional[str] = None
|
content: Optional[str] = None
|
||||||
pinned: Optional[bool] = None
|
pinned: Optional[bool] = None
|
||||||
category: Optional[str] = None
|
category: Optional[str] = None
|
||||||
|
scope: Optional[str] = None # system | personal
|
||||||
tags: Optional[List[str]] = None
|
tags: Optional[List[str]] = None
|
||||||
|
|
||||||
|
|
||||||
@@ -214,6 +216,7 @@ class MemoryOut(BaseModel):
|
|||||||
pinned: bool
|
pinned: bool
|
||||||
category: str
|
category: str
|
||||||
source: str
|
source: str
|
||||||
|
scope: str = "personal"
|
||||||
tags: List[str]
|
tags: List[str]
|
||||||
created_at: str
|
created_at: str
|
||||||
updated_at: str
|
updated_at: str
|
||||||
@@ -328,6 +331,7 @@ def memory_save(body: MemoryIn):
|
|||||||
pinned=body.pinned,
|
pinned=body.pinned,
|
||||||
category=body.category,
|
category=body.category,
|
||||||
source=body.source,
|
source=body.source,
|
||||||
|
scope=body.scope,
|
||||||
tags=body.tags,
|
tags=body.tags,
|
||||||
conversation_id=body.conversation_id,
|
conversation_id=body.conversation_id,
|
||||||
attachments=body.attachments or [],
|
attachments=body.attachments or [],
|
||||||
@@ -353,6 +357,8 @@ def memory_update(point_id: str, body: MemoryUpdate):
|
|||||||
existing.pinned = body.pinned
|
existing.pinned = body.pinned
|
||||||
if body.category is not None:
|
if body.category is not None:
|
||||||
existing.category = body.category
|
existing.category = body.category
|
||||||
|
if body.scope is not None:
|
||||||
|
existing.scope = body.scope
|
||||||
if body.tags is not None:
|
if body.tags is not None:
|
||||||
existing.tags = body.tags
|
existing.tags = body.tags
|
||||||
|
|
||||||
@@ -537,12 +543,23 @@ def memory_import_files():
|
|||||||
# Wiederherstellen einer schlanken ARIA nach Wipe.
|
# Wiederherstellen einer schlanken ARIA nach Wipe.
|
||||||
|
|
||||||
@app.get("/memory/export-bootstrap")
|
@app.get("/memory/export-bootstrap")
|
||||||
def memory_export_bootstrap():
|
def memory_export_bootstrap(scope: str = "system"):
|
||||||
"""Gibt alle pinned Memories als JSON zurueck — fuer Browser-Download."""
|
"""Gibt pinned Memories als JSON zurueck — fuer Browser-Download.
|
||||||
|
|
||||||
|
scope='system' → nur generische Regeln (fuer ein frisches System),
|
||||||
|
scope='personal' → nur Stefan-spezifisches (Name, Zugangsdaten, Projekte),
|
||||||
|
scope='all' → alles pinned (Vollbackup).
|
||||||
|
Default 'system', damit man nicht versehentlich Persoenliches teilt."""
|
||||||
s = store()
|
s = store()
|
||||||
pinned = s.list_pinned()
|
if scope == "all":
|
||||||
|
pinned = s.list_pinned()
|
||||||
|
elif scope in ("system", "personal"):
|
||||||
|
pinned = s.list_pinned_by_scope(scope)
|
||||||
|
else:
|
||||||
|
raise HTTPException(400, f"Ungueltiger scope: {scope}")
|
||||||
return {
|
return {
|
||||||
"version": 1,
|
"version": 2,
|
||||||
|
"scope": scope,
|
||||||
"exported_at": __import__("datetime").datetime.now(
|
"exported_at": __import__("datetime").datetime.now(
|
||||||
__import__("datetime").timezone.utc
|
__import__("datetime").timezone.utc
|
||||||
).isoformat(),
|
).isoformat(),
|
||||||
@@ -555,6 +572,7 @@ def memory_export_bootstrap():
|
|||||||
"pinned": True,
|
"pinned": True,
|
||||||
"category": p.category,
|
"category": p.category,
|
||||||
"source": p.source,
|
"source": p.source,
|
||||||
|
"scope": p.scope,
|
||||||
"tags": p.tags,
|
"tags": p.tags,
|
||||||
}
|
}
|
||||||
for p in pinned
|
for p in pinned
|
||||||
@@ -564,13 +582,18 @@ def memory_export_bootstrap():
|
|||||||
|
|
||||||
class BootstrapBundle(BaseModel):
|
class BootstrapBundle(BaseModel):
|
||||||
version: int = 1
|
version: int = 1
|
||||||
|
scope: Optional[str] = None # system | personal | all (aus dem Export)
|
||||||
memories: List[dict]
|
memories: List[dict]
|
||||||
|
|
||||||
|
|
||||||
@app.post("/memory/import-bootstrap")
|
@app.post("/memory/import-bootstrap")
|
||||||
def memory_import_bootstrap(body: BootstrapBundle):
|
def memory_import_bootstrap(body: BootstrapBundle):
|
||||||
"""Loescht alle pinned Memories und importiert die im Bundle.
|
"""Importiert ein Bootstrap-Bundle scope-sicher.
|
||||||
Cold Memory (unpinned) bleibt unangetastet.
|
|
||||||
|
Es werden NUR die aktuell pinned Punkte geloescht, deren scope zum Import
|
||||||
|
gehoert — ein System-Import laesst also die persoenlichen pinned Memories
|
||||||
|
(Name, Zugangsdaten) unangetastet und umgekehrt. Bei einem 'all'-Bundle
|
||||||
|
(Vollbackup) werden alle pinned ersetzt.
|
||||||
|
|
||||||
Wenn keine Memories im Bundle: nur loeschen ist NICHT erlaubt — der
|
Wenn keine Memories im Bundle: nur loeschen ist NICHT erlaubt — der
|
||||||
Caller soll erst exportieren und dann importieren.
|
Caller soll erst exportieren und dann importieren.
|
||||||
@@ -580,23 +603,31 @@ def memory_import_bootstrap(body: BootstrapBundle):
|
|||||||
|
|
||||||
s = store()
|
s = store()
|
||||||
e = embedder()
|
e = embedder()
|
||||||
|
|
||||||
# Alle aktuell pinned Punkte loeschen
|
|
||||||
from qdrant_client.http import models as qm
|
from qdrant_client.http import models as qm
|
||||||
from memory.vector_store import COLLECTION
|
from memory.vector_store import COLLECTION
|
||||||
|
|
||||||
|
# Scope bestimmen: explizit aus dem Bundle, sonst aus den memories ableiten.
|
||||||
|
bundle_scope = body.scope
|
||||||
|
if bundle_scope not in ("system", "personal", "all"):
|
||||||
|
scopes_in_mems = {m.get("scope", "personal") for m in body.memories}
|
||||||
|
bundle_scope = scopes_in_mems.pop() if len(scopes_in_mems) == 1 else "all"
|
||||||
|
|
||||||
|
# Nur die pinned Punkte des betroffenen scope loeschen.
|
||||||
|
del_must = [qm.FieldCondition(key="pinned", match=qm.MatchValue(value=True))]
|
||||||
|
if bundle_scope in ("system", "personal"):
|
||||||
|
del_must.append(qm.FieldCondition(key="scope", match=qm.MatchValue(value=bundle_scope)))
|
||||||
s.client.delete(
|
s.client.delete(
|
||||||
collection_name=COLLECTION,
|
collection_name=COLLECTION,
|
||||||
points_selector=qm.FilterSelector(filter=qm.Filter(must=[
|
points_selector=qm.FilterSelector(filter=qm.Filter(must=del_must)),
|
||||||
qm.FieldCondition(key="pinned", match=qm.MatchValue(value=True))
|
|
||||||
])),
|
|
||||||
)
|
)
|
||||||
|
|
||||||
# Neue Punkte einspeisen
|
# Neue Punkte einspeisen — scope pro memory (Fallback: bundle_scope bzw. personal).
|
||||||
created = 0
|
created = 0
|
||||||
for m in body.memories:
|
for m in body.memories:
|
||||||
content = (m.get("content") or "").strip()
|
content = (m.get("content") or "").strip()
|
||||||
if not content:
|
if not content:
|
||||||
continue
|
continue
|
||||||
|
mscope = m.get("scope") or (bundle_scope if bundle_scope != "all" else "personal")
|
||||||
point = MemoryPoint(
|
point = MemoryPoint(
|
||||||
id="",
|
id="",
|
||||||
type=m.get("type", "fact"),
|
type=m.get("type", "fact"),
|
||||||
@@ -605,13 +636,14 @@ def memory_import_bootstrap(body: BootstrapBundle):
|
|||||||
pinned=True,
|
pinned=True,
|
||||||
category=m.get("category", ""),
|
category=m.get("category", ""),
|
||||||
source=m.get("source", "bootstrap-import"),
|
source=m.get("source", "bootstrap-import"),
|
||||||
|
scope=mscope,
|
||||||
tags=list(m.get("tags", [])),
|
tags=list(m.get("tags", [])),
|
||||||
)
|
)
|
||||||
vec = e.embed(content)
|
vec = e.embed(content)
|
||||||
s.upsert(point, vec)
|
s.upsert(point, vec)
|
||||||
created += 1
|
created += 1
|
||||||
|
|
||||||
return {"created": created, "deleted_previous_pinned": True}
|
return {"created": created, "scope": bundle_scope, "deleted_previous_pinned": True}
|
||||||
|
|
||||||
|
|
||||||
# ─── Conversation-Loop ──────────────────────────────────────────────
|
# ─── Conversation-Loop ──────────────────────────────────────────────
|
||||||
@@ -644,6 +676,11 @@ class ChatOut(BaseModel):
|
|||||||
# Task fertig ist)? Dann pausiert die App die Projekt-Queue und leitet die
|
# Task fertig ist)? Dann pausiert die App die Projekt-Queue und leitet die
|
||||||
# naechste Eingabe als Antwort weiter, statt sie als neuen Auftrag anzustellen.
|
# naechste Eingabe als Antwort weiter, statt sie als neuen Auftrag anzustellen.
|
||||||
awaiting_reply: bool = False
|
awaiting_reply: bool = False
|
||||||
|
# Der User hat per Sprache "Wake-Word aus" gesagt → die App stoppt den
|
||||||
|
# Wake-Word-Listener komplett (Mikro frei).
|
||||||
|
wake_off: bool = False
|
||||||
|
# "Wake-Word an" per Befehl (Text/Aufnahme-Button) → App startet den Listener.
|
||||||
|
wake_on: bool = False
|
||||||
# Echo der project_id die dieser Turn hatte. Bridge nutzt sie damit die
|
# Echo der project_id die dieser Turn hatte. Bridge nutzt sie damit die
|
||||||
# ausgehende Chat-Bubble sauber getaggt in der richtigen Thread-Bahn der
|
# ausgehende Chat-Bubble sauber getaggt in der richtigen Thread-Bahn der
|
||||||
# UI landet.
|
# UI landet.
|
||||||
@@ -753,6 +790,8 @@ async def chat(body: ChatIn, background: BackgroundTasks):
|
|||||||
speak=speak,
|
speak=speak,
|
||||||
converse=converse,
|
converse=converse,
|
||||||
awaiting_reply=awaiting_reply,
|
awaiting_reply=awaiting_reply,
|
||||||
|
wake_off=(answered_by == "wake-off"),
|
||||||
|
wake_on=(answered_by == "wake-on"),
|
||||||
)
|
)
|
||||||
finally:
|
finally:
|
||||||
_project_pending[pid] = [
|
_project_pending[pid] = [
|
||||||
|
|||||||
@@ -11,6 +11,10 @@ Punkt-Schema (Payload):
|
|||||||
content — eigentlicher Text (wird embedded)
|
content — eigentlicher Text (wird embedded)
|
||||||
pinned — bool, True = Hot Memory (immer in Prompt)
|
pinned — bool, True = Hot Memory (immer in Prompt)
|
||||||
source — import | conversation | manual
|
source — import | conversation | manual
|
||||||
|
scope — system | personal. system = generische Regeln, die JEDER
|
||||||
|
braucht, der das System aufsetzt (Sicherheit, Ehrlichkeit,
|
||||||
|
Skill-Regeln). personal = Stefan-spezifisch (Name, Zugangs-
|
||||||
|
daten, Projekte). Steuert den getrennten Bootstrap-Export.
|
||||||
tags — Liste von Strings
|
tags — Liste von Strings
|
||||||
created_at, updated_at — ISO-Strings
|
created_at, updated_at — ISO-Strings
|
||||||
conversation_id — optional, nur fuer type=conversation
|
conversation_id — optional, nur fuer type=conversation
|
||||||
@@ -55,6 +59,7 @@ class MemoryPoint:
|
|||||||
pinned: bool = False
|
pinned: bool = False
|
||||||
category: str = ""
|
category: str = ""
|
||||||
source: str = "manual"
|
source: str = "manual"
|
||||||
|
scope: str = "personal" # system | personal — steuert Bootstrap-Export
|
||||||
tags: List[str] = field(default_factory=list)
|
tags: List[str] = field(default_factory=list)
|
||||||
created_at: str = ""
|
created_at: str = ""
|
||||||
updated_at: str = ""
|
updated_at: str = ""
|
||||||
@@ -74,6 +79,7 @@ class MemoryPoint:
|
|||||||
"pinned": self.pinned,
|
"pinned": self.pinned,
|
||||||
"category": self.category,
|
"category": self.category,
|
||||||
"source": self.source,
|
"source": self.source,
|
||||||
|
"scope": self.scope,
|
||||||
"tags": self.tags,
|
"tags": self.tags,
|
||||||
"created_at": self.created_at,
|
"created_at": self.created_at,
|
||||||
"updated_at": self.updated_at,
|
"updated_at": self.updated_at,
|
||||||
@@ -94,6 +100,7 @@ class MemoryPoint:
|
|||||||
pinned=payload.get("pinned", False),
|
pinned=payload.get("pinned", False),
|
||||||
category=payload.get("category", ""),
|
category=payload.get("category", ""),
|
||||||
source=payload.get("source", "manual"),
|
source=payload.get("source", "manual"),
|
||||||
|
scope=payload.get("scope", "personal"),
|
||||||
tags=payload.get("tags", []),
|
tags=payload.get("tags", []),
|
||||||
created_at=payload.get("created_at", ""),
|
created_at=payload.get("created_at", ""),
|
||||||
updated_at=payload.get("updated_at", ""),
|
updated_at=payload.get("updated_at", ""),
|
||||||
@@ -120,14 +127,23 @@ class VectorStore:
|
|||||||
collection_name=COLLECTION,
|
collection_name=COLLECTION,
|
||||||
vectors_config=qm.VectorParams(size=VECTOR_DIM, distance=qm.Distance.COSINE),
|
vectors_config=qm.VectorParams(size=VECTOR_DIM, distance=qm.Distance.COSINE),
|
||||||
)
|
)
|
||||||
# Indexe fuer typische Filter-Felder
|
# Indexe fuer typische Filter-Felder — idempotent, laeuft auch auf
|
||||||
for field_name in ("type", "pinned", "category", "source", "migration_key"):
|
# einer bestehenden Collection (fuer neu hinzugekommene Felder wie scope).
|
||||||
|
self._ensure_indexes()
|
||||||
|
|
||||||
|
def _ensure_indexes(self):
|
||||||
|
for field_name in ("type", "pinned", "category", "source", "scope", "migration_key"):
|
||||||
|
schema = (qm.PayloadSchemaType.BOOL if field_name == "pinned"
|
||||||
|
else qm.PayloadSchemaType.KEYWORD)
|
||||||
|
try:
|
||||||
self.client.create_payload_index(
|
self.client.create_payload_index(
|
||||||
collection_name=COLLECTION,
|
collection_name=COLLECTION,
|
||||||
field_name=field_name,
|
field_name=field_name,
|
||||||
field_schema=qm.PayloadSchemaType.KEYWORD if field_name != "pinned"
|
field_schema=schema,
|
||||||
else qm.PayloadSchemaType.BOOL,
|
|
||||||
)
|
)
|
||||||
|
except Exception:
|
||||||
|
# Index existiert bereits — kein Problem.
|
||||||
|
pass
|
||||||
|
|
||||||
# ─── Schreib-Operationen ─────────────────────────────────────────
|
# ─── Schreib-Operationen ─────────────────────────────────────────
|
||||||
|
|
||||||
@@ -164,6 +180,38 @@ class VectorStore:
|
|||||||
qm.FieldCondition(key="pinned", match=qm.MatchValue(value=True))
|
qm.FieldCondition(key="pinned", match=qm.MatchValue(value=True))
|
||||||
]))
|
]))
|
||||||
|
|
||||||
|
def list_pinned_by_scope(self, scope: str) -> List[MemoryPoint]:
|
||||||
|
"""Alle pinned Punkte eines scope (system | personal). Fuer den
|
||||||
|
getrennten Bootstrap-Export."""
|
||||||
|
return self._scroll(filter=qm.Filter(must=[
|
||||||
|
qm.FieldCondition(key="pinned", match=qm.MatchValue(value=True)),
|
||||||
|
qm.FieldCondition(key="scope", match=qm.MatchValue(value=scope)),
|
||||||
|
]))
|
||||||
|
|
||||||
|
def list_index_titles(self, limit: int = 500) -> List[MemoryPoint]:
|
||||||
|
"""Leichtgewichtiger Titel-Index des kalten Gedaechtnisses fuer den
|
||||||
|
System-Prompt: ARIA sieht WAS sie an Nachschlage-Wissen hat (Zugangs-
|
||||||
|
daten, Infrastruktur, Projekte) und holt den Inhalt bei Bedarf via
|
||||||
|
memory_search — statt Stefan nach etwas zu fragen, das schon da ist.
|
||||||
|
|
||||||
|
Bewusst NUR die deliberat gespeicherten Punkte:
|
||||||
|
- nicht pinned (die sind eh schon voll im Prompt),
|
||||||
|
- kein type=conversation (Chat-Mitschnitte),
|
||||||
|
- kein source=distilled (die 100e auto-destillierten Gespraechs-
|
||||||
|
Fakten — die traegt das semantische Auto-Retrieval, sie hier
|
||||||
|
als Titel zu listen wuerde nur Kontext fressen).
|
||||||
|
So bleibt der Index klein (Dutzende statt Hunderte Zeilen)."""
|
||||||
|
return self._scroll(
|
||||||
|
filter=qm.Filter(
|
||||||
|
must_not=[
|
||||||
|
qm.FieldCondition(key="pinned", match=qm.MatchValue(value=True)),
|
||||||
|
qm.FieldCondition(key="type", match=qm.MatchValue(value="conversation")),
|
||||||
|
qm.FieldCondition(key="source", match=qm.MatchValue(value="distilled")),
|
||||||
|
]
|
||||||
|
),
|
||||||
|
limit=limit,
|
||||||
|
)
|
||||||
|
|
||||||
def list_by_type(self, type_: str, limit: int = 100) -> List[MemoryPoint]:
|
def list_by_type(self, type_: str, limit: int = 100) -> List[MemoryPoint]:
|
||||||
return self._scroll(
|
return self._scroll(
|
||||||
filter=qm.Filter(must=[
|
filter=qm.Filter(must=[
|
||||||
|
|||||||
@@ -252,6 +252,7 @@ def _parse_user_md(md: str, source_file: str) -> List[MemoryPoint]:
|
|||||||
type_="preference", title=f"User: {btitle}",
|
type_="preference", title=f"User: {btitle}",
|
||||||
content=btext, category="allgemein",
|
content=btext, category="allgemein",
|
||||||
migration_key=f"{source_file}/general-{idx}",
|
migration_key=f"{source_file}/general-{idx}",
|
||||||
|
scope="personal",
|
||||||
))
|
))
|
||||||
else:
|
else:
|
||||||
cat_key = re.sub(r"[^a-z0-9]+", "-", title.lower()).strip("-") or "allgemein"
|
cat_key = re.sub(r"[^a-z0-9]+", "-", title.lower()).strip("-") or "allgemein"
|
||||||
@@ -259,6 +260,7 @@ def _parse_user_md(md: str, source_file: str) -> List[MemoryPoint]:
|
|||||||
type_="preference", title=title,
|
type_="preference", title=title,
|
||||||
content=content, category=cat_key,
|
content=content, category=cat_key,
|
||||||
migration_key=f"{source_file}/{cat_key}",
|
migration_key=f"{source_file}/{cat_key}",
|
||||||
|
scope="personal",
|
||||||
))
|
))
|
||||||
return points
|
return points
|
||||||
|
|
||||||
@@ -283,7 +285,11 @@ def _mk(
|
|||||||
migration_key: str,
|
migration_key: str,
|
||||||
pinned: bool = True,
|
pinned: bool = True,
|
||||||
category: str = "",
|
category: str = "",
|
||||||
|
scope: str = "system",
|
||||||
) -> MemoryPoint:
|
) -> MemoryPoint:
|
||||||
|
# scope-Default 'system': AGENT.md + TOOLING.md beschreiben ARIA selbst
|
||||||
|
# (Identitaet, Sicherheit, Architektur) — das braucht jedes System.
|
||||||
|
# USER.md-Praeferenzen sind personal und uebergeben scope='personal'.
|
||||||
p = MemoryPoint(
|
p = MemoryPoint(
|
||||||
id="",
|
id="",
|
||||||
type=type_,
|
type=type_,
|
||||||
@@ -292,6 +298,7 @@ def _mk(
|
|||||||
pinned=pinned,
|
pinned=pinned,
|
||||||
category=category,
|
category=category,
|
||||||
source="import",
|
source="import",
|
||||||
|
scope=scope,
|
||||||
tags=[],
|
tags=[],
|
||||||
)
|
)
|
||||||
# migration_key wird ueber Payload-Index angesprochen — in to_payload manuell anhaengen
|
# migration_key wird ueber Payload-Index angesprochen — in to_payload manuell anhaengen
|
||||||
|
|||||||
+83
-1
@@ -162,6 +162,53 @@ def build_time_section() -> str:
|
|||||||
]
|
]
|
||||||
return "\n".join(lines)
|
return "\n".join(lines)
|
||||||
|
|
||||||
|
def build_voice_flow_section() -> str:
|
||||||
|
"""Sprach-/Gespraechssteuerung: ARIA erkennt AUS DEM TEXT die Phase (Befehl vs.
|
||||||
|
Frage, Kette vs. Ende) und deklariert sie per Marker — wie [[AWAIT]]. Die
|
||||||
|
Marker werden im Brain entfernt (nie angezeigt/vorgelesen)."""
|
||||||
|
return "\n".join([
|
||||||
|
"## Sprach- & Gespraechssteuerung (Voice-First — du entscheidest die Phase)",
|
||||||
|
"Stefan spricht meist mit dir. DU erkennst aus dem Text, was gerade Phase "
|
||||||
|
"ist — niemand raet das fuer dich. Dazu haengst du EINEN Marker (bei Bedarf "
|
||||||
|
"zwei) ganz ans ENDE deiner Antwort. Sie werden entfernt: nicht angezeigt, "
|
||||||
|
"nicht vorgelesen, nicht gespeichert — genau wie `[[AWAIT]]`.",
|
||||||
|
"",
|
||||||
|
"- `[[STUMM]]` → Deine Antwort ist ein reiner **Steuerbefehl** (du hast etwas "
|
||||||
|
"GETAN: Musik, VNC oeffnen, einen Menuepunkt klicken, Licht …). Sie wird "
|
||||||
|
"NICHT vorgelesen; der kurze Bestaetigungstext steht nur in der Bubble. "
|
||||||
|
"Setz das IMMER, wenn Stefan dir einen Befehl gibt statt eine Frage stellt — "
|
||||||
|
"AUCH wenn du den Befehl ueber ein Skill/Tool ausfuehrst (nicht nur beim "
|
||||||
|
"Fast-Path). `[[STUMM]]` ALLEIN = Einzelbefehl → danach direkt zurueck aufs "
|
||||||
|
"Wake-Word.",
|
||||||
|
"- `[[WEITER]]` → Das Gespraech bzw. eine **Befehlskette** laeuft weiter: das "
|
||||||
|
"Mikro bleibt offen, du wartest auf die naechste Eingabe (kein erneutes "
|
||||||
|
"\"Computer\" noetig). Setz das, wenn Stefan eine Kette ankuendigt ('ich geb "
|
||||||
|
"dir gleich mehrere Befehle', 'wir machen das jetzt Schritt fuer Schritt') "
|
||||||
|
"oder das Gespraech klar weitergeht.",
|
||||||
|
"- `[[ENDE]]` → Konversation/Kette ist zu Ende: zurueck aufs Wake-Word. Setz "
|
||||||
|
"das, wenn Stefan schliesst ('das war's', 'Konversation Ende', 'Befehlskette "
|
||||||
|
"Ende', 'danke, fertig'). Stellt er in DERSELBEN Nachricht noch eine Frage, "
|
||||||
|
"beantworte sie normal (OHNE `[[STUMM]]`, wird also vorgelesen) UND haeng "
|
||||||
|
"`[[ENDE]]` an.",
|
||||||
|
"",
|
||||||
|
"Regeln:",
|
||||||
|
"- Befehl (etwas TUN) → `[[STUMM]]`. Frage (etwas WISSEN / plaudern) → normal, "
|
||||||
|
"ohne Marker (wird vorgelesen).",
|
||||||
|
"- Befehlskette: JEDER Schritt `[[STUMM]] [[WEITER]]` (stumm arbeiten, Mikro "
|
||||||
|
"offen), bis Stefan die Kette beendet → letzter Turn `[[ENDE]]`.",
|
||||||
|
"- Ohne Marker = normales Gespraech: du wirst vorgelesen und ich lausche "
|
||||||
|
"danach kurz weiter (Stefan kann einfach antworten, ohne 'Computer').",
|
||||||
|
"- Nie widerspruechlich: `[[ENDE]]` schlaegt `[[WEITER]]`.",
|
||||||
|
"",
|
||||||
|
"**Ohr aus/an:** Wenn Stefan will dass du aufhoerst zuzuhoeren — egal wie "
|
||||||
|
"formuliert ('leg dich schlafen', 'geh schlafen', 'Ohr aus', 'gute Nacht', "
|
||||||
|
"'mach Pause vom Zuhoeren') — ruf das Tool `ear_control(action='off')`. "
|
||||||
|
"Wieder aktivieren ('Ohr an', 'wach auf', 'hoer wieder zu') → "
|
||||||
|
"`ear_control(action='on')`. Die App stoppt/startet dann den Listener; du "
|
||||||
|
"bestaetigst nur kurz.",
|
||||||
|
])
|
||||||
|
|
||||||
|
|
||||||
TYPE_HEADINGS = {
|
TYPE_HEADINGS = {
|
||||||
"identity": "## Wer du bist",
|
"identity": "## Wer du bist",
|
||||||
"rule": "## Sicherheitsregeln & Prinzipien",
|
"rule": "## Sicherheitsregeln & Prinzipien",
|
||||||
@@ -260,6 +307,36 @@ def build_cold_memory_section(matches: List[MemoryPoint]) -> str:
|
|||||||
return "\n".join(lines)
|
return "\n".join(lines)
|
||||||
|
|
||||||
|
|
||||||
|
def build_memory_index_section(index_titles: List[MemoryPoint]) -> str:
|
||||||
|
"""Titel-Index des kalten Gedaechtnisses: ARIA sieht WELCHES Nachschlage-
|
||||||
|
Wissen sie hat (nur Titel, kein Inhalt = billig), damit sie den Inhalt via
|
||||||
|
memory_search holt statt Stefan nach etwas zu fragen, das schon da ist.
|
||||||
|
Nach Kategorie gruppiert; Conversation-Logs + auto-destillierte Fakten sind
|
||||||
|
bereits ausgefiltert (siehe list_index_titles)."""
|
||||||
|
if not index_titles:
|
||||||
|
return ""
|
||||||
|
grouped: dict[str, List[MemoryPoint]] = {}
|
||||||
|
for p in index_titles:
|
||||||
|
key = (p.category or p.type or "sonstiges").strip() or "sonstiges"
|
||||||
|
grouped.setdefault(key, []).append(p)
|
||||||
|
|
||||||
|
lines = [
|
||||||
|
"## Was in deinem Gedaechtnis liegt (per memory_search abrufbar)",
|
||||||
|
"Diese Eintraege hast DU gespeichert — hier nur die Titel, nicht der "
|
||||||
|
"Inhalt. Wenn einer zur Aufgabe passt, hol den Inhalt mit `memory_search` "
|
||||||
|
"(Titel oder Stichwort). **Frag Stefan NICHT nach etwas, das hier steht** "
|
||||||
|
"(Zugangsdaten, Server/Hosts, Projekt-Stand, Konfig) — erst nachsehen.",
|
||||||
|
"",
|
||||||
|
]
|
||||||
|
for cat in sorted(grouped.keys()):
|
||||||
|
items = grouped[cat]
|
||||||
|
lines.append(f"### {cat}")
|
||||||
|
for p in items:
|
||||||
|
lines.append(f"- {p.title}")
|
||||||
|
lines.append("")
|
||||||
|
return "\n".join(lines).strip()
|
||||||
|
|
||||||
|
|
||||||
def build_skills_section(skills: List[dict]) -> str:
|
def build_skills_section(skills: List[dict]) -> str:
|
||||||
"""Listet alle Skills (aktiv + deaktiviert) damit ARIA weiss was es gibt
|
"""Listet alle Skills (aktiv + deaktiviert) damit ARIA weiss was es gibt
|
||||||
und keine doppelt baut. Plus klare Schwelle wann ein Skill sich lohnt."""
|
und keine doppelt baut. Plus klare Schwelle wann ein Skill sich lohnt."""
|
||||||
@@ -450,6 +527,7 @@ def build_flux_section(flux_config: dict) -> str:
|
|||||||
def build_system_prompt(
|
def build_system_prompt(
|
||||||
pinned: List[MemoryPoint],
|
pinned: List[MemoryPoint],
|
||||||
cold: List[MemoryPoint] | None = None,
|
cold: List[MemoryPoint] | None = None,
|
||||||
|
memory_index: List[MemoryPoint] | None = None,
|
||||||
skills: List[dict] | None = None,
|
skills: List[dict] | None = None,
|
||||||
triggers: List[dict] | None = None,
|
triggers: List[dict] | None = None,
|
||||||
condition_vars: List[dict] | None = None,
|
condition_vars: List[dict] | None = None,
|
||||||
@@ -463,7 +541,8 @@ def build_system_prompt(
|
|||||||
"""Kompletter System-Prompt: Hot + Cold + Skills + Triggers + FLUX + OAuth."""
|
"""Kompletter System-Prompt: Hot + Cold + Skills + Triggers + FLUX + OAuth."""
|
||||||
# Identitaets-Anker IMMER zuerst — vor allen Memories/Sektionen, damit die
|
# Identitaets-Anker IMMER zuerst — vor allen Memories/Sektionen, damit die
|
||||||
# ARIA-Rolle auch in Projekten mit injection-artigem Inhalt (Pentest) haelt.
|
# ARIA-Rolle auch in Projekten mit injection-artigem Inhalt (Pentest) haelt.
|
||||||
parts = [IDENTITY_ANCHOR, "", build_hot_memory_section(pinned), "", build_time_section()]
|
parts = [IDENTITY_ANCHOR, "", build_hot_memory_section(pinned), "", build_time_section(),
|
||||||
|
"", build_voice_flow_section()]
|
||||||
if skills:
|
if skills:
|
||||||
parts.append("")
|
parts.append("")
|
||||||
parts.append(build_skills_section(skills))
|
parts.append(build_skills_section(skills))
|
||||||
@@ -482,6 +561,9 @@ def build_system_prompt(
|
|||||||
callback_host=oauth_callback_host,
|
callback_host=oauth_callback_host,
|
||||||
callback_port=oauth_callback_port,
|
callback_port=oauth_callback_port,
|
||||||
callback_tls=oauth_callback_tls))
|
callback_tls=oauth_callback_tls))
|
||||||
|
if memory_index:
|
||||||
|
parts.append("")
|
||||||
|
parts.append(build_memory_index_section(memory_index))
|
||||||
if cold:
|
if cold:
|
||||||
parts.append("")
|
parts.append("")
|
||||||
parts.append(build_cold_memory_section(cold))
|
parts.append(build_cold_memory_section(cold))
|
||||||
|
|||||||
@@ -475,13 +475,32 @@ SEED_RULES: List[dict] = [
|
|||||||
"zeigt UND steuerbar macht. So erreichst Du Netze, in denen der Haupt-Stack "
|
"zeigt UND steuerbar macht. So erreichst Du Netze, in denen der Haupt-Stack "
|
||||||
"NICHT steht.\n"
|
"NICHT steht.\n"
|
||||||
"\n"
|
"\n"
|
||||||
"Wenn Stefan etwas 'im Buero' / 'im Netz X' / 'auf dem <Geraet> dort' will:\n"
|
"WICHTIG — nicht nur bei 'im Buero'/'im Netz X': IMMER wenn eine Anfrage ein "
|
||||||
|
"GERAET oder einen HOST betrifft, das/der in einem Netz lebt, auf dem Du NICHT "
|
||||||
|
"direkt sitzt (Stefans Zuhause, Buero, Werkstatt, irgendein LAN mit privater "
|
||||||
|
"IP wie 192.168.x/10.x) — z.B. Drucker-Fuellstand, NAS, Smart-TV, ein Host per "
|
||||||
|
"IP — dann pruefe ZUERST `satellite_list`, ob ein Satellit dieses Netz abdeckt. "
|
||||||
|
"Sage NIEMALS 'da komm ich nicht ran' / 'bin nicht im Netz', BEVOR Du "
|
||||||
|
"`satellite_list` aufgerufen hast — ein Satellit im Zielnetz ist genau der Weg "
|
||||||
|
"hinein. Nur wenn wirklich keiner online ist, ist 'erreiche ich nicht' korrekt.\n"
|
||||||
|
"\n"
|
||||||
|
"Ablauf:\n"
|
||||||
" 1. `satellite_list` — welche Satelliten/Netze sind online + was koennen sie.\n"
|
" 1. `satellite_list` — welche Satelliten/Netze sind online + was koennen sie.\n"
|
||||||
" 2. `satellite_devices(satellite='Buero')` — welche Geraete gibt es dort "
|
" 2. `satellite_devices(satellite='Buero')` — welche Geraete gibt es dort "
|
||||||
"(Fire TV, Chromecast, Smart-TVs, Drucker, NAS ...). Nutze es um das "
|
"(Fire TV, Chromecast, Smart-TVs, Drucker, NAS ...). Nutze es um das "
|
||||||
"gemeinte Geraet zu finden, BEVOR Du steuerst.\n"
|
"gemeinte Geraet zu finden, BEVOR Du steuerst.\n"
|
||||||
" 3. `satellite_command(...)` — Aktion ausfuehren.\n"
|
" 3. `satellite_command(...)` — Aktion ausfuehren.\n"
|
||||||
"\n"
|
"\n"
|
||||||
|
"Beispiel 'Patronenstand vom Drucker zuhause': satellite_list -> Satellit im "
|
||||||
|
"Heimnetz online? -> BEVORZUGT satellite_command(satellite='<Heim>', "
|
||||||
|
"action='snmp.printer', params={'ip':'<drucker-ip>'}) -> liefert supplies mit "
|
||||||
|
"name + percent je Patrone (BK/C/M/Y), zuverlaessig aus der Printer-MIB. "
|
||||||
|
"NUR falls SNMP nichts liefert, als Fallback die HTML-Statusseite: "
|
||||||
|
"action='http.get', params={'url':'http://<drucker-ip>/general/status.html', "
|
||||||
|
"'contains':['ink','toner','cyan','magenta','yellow','black','%']} — der "
|
||||||
|
"contains-Filter zieht die relevanten Zeilen (sonst wird der Body bei "
|
||||||
|
"max_chars, Default 20000, abgeschnitten). NICHT aus dem Gedaechtnis raten.\n"
|
||||||
|
"\n"
|
||||||
"Beispiel 'spiel YouTube-Video auf dem Buero-Stick':\n"
|
"Beispiel 'spiel YouTube-Video auf dem Buero-Stick':\n"
|
||||||
" satellite_command(satellite='Buero', device='Fire TV', "
|
" satellite_command(satellite='Buero', device='Fire TV', "
|
||||||
"action='dial.launch', params={'app':'YouTube','v':'<videoId>'})\n"
|
"action='dial.launch', params={'app':'YouTube','v':'<videoId>'})\n"
|
||||||
@@ -915,6 +934,7 @@ def apply(store: VectorStore, embedder: Embedder) -> dict:
|
|||||||
"pinned": True,
|
"pinned": True,
|
||||||
"category": rule.get("category", ""),
|
"category": rule.get("category", ""),
|
||||||
"source": "seed",
|
"source": "seed",
|
||||||
|
"scope": "system",
|
||||||
"tags": [],
|
"tags": [],
|
||||||
"created_at": now,
|
"created_at": now,
|
||||||
"updated_at": now,
|
"updated_at": now,
|
||||||
|
|||||||
+297
-58
@@ -2,7 +2,7 @@
|
|||||||
ARIA Voice Bridge — Hauptmodul.
|
ARIA Voice Bridge — Hauptmodul.
|
||||||
|
|
||||||
Verbindet die Android App (via RVS) mit ARIA-Core. Spracheingabe laeuft
|
Verbindet die Android App (via RVS) mit ARIA-Core. Spracheingabe laeuft
|
||||||
ueber die whisper-bridge (Gamebox, faster-whisper auf CUDA), Sprachausgabe
|
ueber die whisper-bridge (AI-Box, faster-whisper auf CUDA), Sprachausgabe
|
||||||
ueber die f5tts-bridge (Voice Cloning, satzweises PCM-Streaming).
|
ueber die f5tts-bridge (Voice Cloning, satzweises PCM-Streaming).
|
||||||
|
|
||||||
Nachrichtenfluss:
|
Nachrichtenfluss:
|
||||||
@@ -479,7 +479,7 @@ class STTEngine:
|
|||||||
Erkannter Text oder leerer String.
|
Erkannter Text oder leerer String.
|
||||||
"""
|
"""
|
||||||
if self.model is None:
|
if self.model is None:
|
||||||
# Lazy-Load: normalerweise laeuft STT remote auf der Gamebox.
|
# Lazy-Load: normalerweise laeuft STT remote auf der AI-Box.
|
||||||
# Erst wenn das Fallback hier zuschlaegt, laden wir lokal.
|
# Erst wenn das Fallback hier zuschlaegt, laden wir lokal.
|
||||||
logger.info("Lokales Whisper-Fallback — Modell wird nachgeladen...")
|
logger.info("Lokales Whisper-Fallback — Modell wird nachgeladen...")
|
||||||
try:
|
try:
|
||||||
@@ -654,7 +654,7 @@ class ARIABridge:
|
|||||||
self._seen_client_msg_ids: "OrderedDict[str, float]" = OrderedDict()
|
self._seen_client_msg_ids: "OrderedDict[str, float]" = OrderedDict()
|
||||||
self._SEEN_CLIENT_MSG_LIMIT = 200
|
self._SEEN_CLIENT_MSG_LIMIT = 200
|
||||||
|
|
||||||
# Komponenten (TTS: F5-TTS remote auf der Gamebox, lokales TTS wurde entfernt)
|
# Komponenten (TTS: F5-TTS remote auf der AI-Box, lokales TTS wurde entfernt)
|
||||||
self.tts_enabled = True
|
self.tts_enabled = True
|
||||||
self.xtts_voice = ""
|
self.xtts_voice = ""
|
||||||
self._f5tts_config: dict = {}
|
self._f5tts_config: dict = {}
|
||||||
@@ -686,7 +686,7 @@ class ARIABridge:
|
|||||||
except (TypeError, ValueError):
|
except (TypeError, ValueError):
|
||||||
self._persistent_xtts_speed = None
|
self._persistent_xtts_speed = None
|
||||||
# F5-TTS-Felder aufsammeln (werden spaeter via RVS rebroadcastet,
|
# F5-TTS-Felder aufsammeln (werden spaeter via RVS rebroadcastet,
|
||||||
# damit die f5tts-bridge auf der Gamebox die Settings auch nach
|
# damit die f5tts-bridge auf der AI-Box die Settings auch nach
|
||||||
# Restart wiederbekommt — sonst stuende sie auf Hard-Defaults)
|
# Restart wiederbekommt — sonst stuende sie auf Hard-Defaults)
|
||||||
for k in ("f5ttsModel", "f5ttsCkptFile", "f5ttsVocabFile",
|
for k in ("f5ttsModel", "f5ttsCkptFile", "f5ttsVocabFile",
|
||||||
"f5ttsCfgStrength", "f5ttsNfeStep"):
|
"f5ttsCfgStrength", "f5ttsNfeStep"):
|
||||||
@@ -746,7 +746,7 @@ class ARIABridge:
|
|||||||
# Gleiche Logik fuer die Wiedergabegeschwindigkeit (F5-TTS speed-Param,
|
# Gleiche Logik fuer die Wiedergabegeschwindigkeit (F5-TTS speed-Param,
|
||||||
# App-Setting aria_tts_speed, 1.0 = normal).
|
# App-Setting aria_tts_speed, 1.0 = normal).
|
||||||
self._next_speed_override: Optional[float] = None
|
self._next_speed_override: Optional[float] = None
|
||||||
# STT-Requests die aktuell auf Antwort von der whisper-bridge (Gamebox) warten.
|
# STT-Requests die aktuell auf Antwort von der whisper-bridge (AI-Box) warten.
|
||||||
# requestId → Future mit dem Text (oder None bei Fehler).
|
# requestId → Future mit dem Text (oder None bei Fehler).
|
||||||
self._pending_stt: dict[str, asyncio.Future] = {}
|
self._pending_stt: dict[str, asyncio.Future] = {}
|
||||||
# whisper-bridge service_status: True wenn ready, False/None wenn loading/unbekannt.
|
# whisper-bridge service_status: True wenn ready, False/None wenn loading/unbekannt.
|
||||||
@@ -766,7 +766,12 @@ class ARIABridge:
|
|||||||
# requestId → Future (sat_devices / sat_result), analog _pending_flux.
|
# requestId → Future (sat_devices / sat_result), analog _pending_flux.
|
||||||
self._satellites: dict[str, dict] = {}
|
self._satellites: dict[str, dict] = {}
|
||||||
self._pending_sat: dict[str, asyncio.Future] = {}
|
self._pending_sat: dict[str, asyncio.Future] = {}
|
||||||
# FLUX-Render-Requests die aktuell auf Antwort der flux-bridge (Gamebox) warten.
|
# Compute-Fleet: GPU-Worker (voxtral/whisper/f5tts/llm) melden sich per
|
||||||
|
# worker_hello, halten sich per worker_ping frisch. instanceId →
|
||||||
|
# {service, node, gpus, model, busy, last_seen}. Genutzt fuer die
|
||||||
|
# Diagnostic-Flotten-Anzeige und (Stage 3) targetInstance-Routing.
|
||||||
|
self._workers: dict[str, dict] = {}
|
||||||
|
# FLUX-Render-Requests die aktuell auf Antwort der flux-bridge (AI-Box) warten.
|
||||||
# requestId → Future mit dem flux_response-Payload (oder None bei Fehler).
|
# requestId → Future mit dem flux_response-Payload (oder None bei Fehler).
|
||||||
self._pending_flux: dict[str, asyncio.Future] = {}
|
self._pending_flux: dict[str, asyncio.Future] = {}
|
||||||
# flux-bridge service_status: True wenn ready. Render-Timeouts werden
|
# flux-bridge service_status: True wenn ready. Render-Timeouts werden
|
||||||
@@ -774,7 +779,7 @@ class ARIABridge:
|
|||||||
self._remote_flux_ready: bool = False
|
self._remote_flux_ready: bool = False
|
||||||
# Lokales LLM (Plan B): requestId → Future mit dem llm_response-Payload.
|
# Lokales LLM (Plan B): requestId → Future mit dem llm_response-Payload.
|
||||||
# Analog zu _pending_flux — Brain ruft /internal/local-llm, wir relayen
|
# Analog zu _pending_flux — Brain ruft /internal/local-llm, wir relayen
|
||||||
# llm_request via RVS an den llm-adapter (Gamebox) und warten auf
|
# llm_request via RVS an den llm-adapter (AI-Box) und warten auf
|
||||||
# llm_response.
|
# llm_response.
|
||||||
self._pending_llm: dict[str, asyncio.Future] = {}
|
self._pending_llm: dict[str, asyncio.Future] = {}
|
||||||
# User-Message-Counter fuer Auto-Compact. Bei zu langer Konversation
|
# User-Message-Counter fuer Auto-Compact. Bei zu langer Konversation
|
||||||
@@ -790,7 +795,8 @@ class ARIABridge:
|
|||||||
# Anfrage an aria-core. Sonst antwortet ARIA zweimal (einmal "warte auf
|
# Anfrage an aria-core. Sonst antwortet ARIA zweimal (einmal "warte auf
|
||||||
# Anweisung" beim file, einmal auf den Chat-Text).
|
# Anweisung" beim file, einmal auf den Chat-Text).
|
||||||
# Liste von Tuples: (file_path, name, file_type, size_kb, width, height)
|
# Liste von Tuples: (file_path, name, file_type, size_kb, width, height)
|
||||||
self._pending_files: list[tuple[str, str, str, int, int, int]] = []
|
# (file_path, name, type, kb, width, height, clientMsgId)
|
||||||
|
self._pending_files: list[tuple[str, str, str, int, int, int, str]] = []
|
||||||
self._pending_files_flush_task: Optional[asyncio.Task] = None
|
self._pending_files_flush_task: Optional[asyncio.Task] = None
|
||||||
# Projekt-Kontext der gerade gepufferten Anhaenge (aus dem file-Upload).
|
# Projekt-Kontext der gerade gepufferten Anhaenge (aus dem file-Upload).
|
||||||
# Wird beim Flush an send_to_core gegeben, damit Anhaenge im richtigen
|
# Wird beim Flush an send_to_core gegeben, damit Anhaenge im richtigen
|
||||||
@@ -809,7 +815,7 @@ class ARIABridge:
|
|||||||
logger.info("ARIA Voice Bridge startet...")
|
logger.info("ARIA Voice Bridge startet...")
|
||||||
logger.info("=" * 50)
|
logger.info("=" * 50)
|
||||||
|
|
||||||
# STT wird standardmaessig von der whisper-bridge (Gamebox) erledigt.
|
# STT wird standardmaessig von der whisper-bridge (AI-Box) erledigt.
|
||||||
# Lokales Whisper ist nur Fallback und wird lazy geladen wenn remote nicht
|
# Lokales Whisper ist nur Fallback und wird lazy geladen wenn remote nicht
|
||||||
# antwortet. Das spart RAM auf der VM und Startup-Zeit.
|
# antwortet. Das spart RAM auf der VM und Startup-Zeit.
|
||||||
|
|
||||||
@@ -1621,6 +1627,11 @@ class ARIABridge:
|
|||||||
# die Projekt-Queue und leitet die naechste Eingabe als Antwort auf
|
# die Projekt-Queue und leitet die naechste Eingabe als Antwort auf
|
||||||
# DIESE Rueckfrage weiter, statt sie als neuen Auftrag anzustellen.
|
# DIESE Rueckfrage weiter, statt sie als neuen Auftrag anzustellen.
|
||||||
"awaiting_reply": bool(payload.get("awaiting_reply", False)) if isinstance(payload, dict) else False,
|
"awaiting_reply": bool(payload.get("awaiting_reply", False)) if isinstance(payload, dict) else False,
|
||||||
|
# User hat "Wake-Word aus" gesagt → App stoppt den Listener komplett
|
||||||
|
# (Mikro frei).
|
||||||
|
"wake_off": bool(payload.get("wake_off", False)) if isinstance(payload, dict) else False,
|
||||||
|
# "Wake-Word an" (Text/Aufnahme-Button) → App startet den Listener.
|
||||||
|
"wake_on": bool(payload.get("wake_on", False)) if isinstance(payload, dict) else False,
|
||||||
},
|
},
|
||||||
"timestamp": int(asyncio.get_event_loop().time() * 1000),
|
"timestamp": int(asyncio.get_event_loop().time() * 1000),
|
||||||
})
|
})
|
||||||
@@ -1673,20 +1684,27 @@ class ARIABridge:
|
|||||||
if len(self._xtts_request_to_message) > 100:
|
if len(self._xtts_request_to_message) > 100:
|
||||||
oldest = next(iter(self._xtts_request_to_message))
|
oldest = next(iter(self._xtts_request_to_message))
|
||||||
self._xtts_request_to_message.pop(oldest, None)
|
self._xtts_request_to_message.pop(oldest, None)
|
||||||
|
# Redundanz: freie f5tts-Instanz waehlen und gezielt adressieren.
|
||||||
|
# None (keine Instanz bekannt / alle offline) → kein targetInstance,
|
||||||
|
# Broadcast wie bisher (Single-Node laeuft unveraendert).
|
||||||
|
tts_target = self._pick_worker("f5tts")
|
||||||
|
tts_payload = {
|
||||||
|
"text": tts_text,
|
||||||
|
"voice": xtts_voice,
|
||||||
|
"speed": xtts_speed,
|
||||||
|
"language": "de",
|
||||||
|
"requestId": xtts_request_id,
|
||||||
|
"messageId": message_id,
|
||||||
|
}
|
||||||
|
if tts_target:
|
||||||
|
tts_payload["targetInstance"] = tts_target
|
||||||
await self._send_to_rvs({
|
await self._send_to_rvs({
|
||||||
"type": "xtts_request",
|
"type": "xtts_request",
|
||||||
"payload": {
|
"payload": tts_payload,
|
||||||
"text": tts_text,
|
|
||||||
"voice": xtts_voice,
|
|
||||||
"speed": xtts_speed,
|
|
||||||
"language": "de",
|
|
||||||
"requestId": xtts_request_id,
|
|
||||||
"messageId": message_id,
|
|
||||||
},
|
|
||||||
"timestamp": int(asyncio.get_event_loop().time() * 1000),
|
"timestamp": int(asyncio.get_event_loop().time() * 1000),
|
||||||
})
|
})
|
||||||
logger.info("[core] XTTS-Request gesendet (voice=%s, speed=%.2fx): '%s'",
|
logger.info("[core] XTTS-Request gesendet (voice=%s, speed=%.2fx, target=%s): '%s'",
|
||||||
xtts_voice or "default", xtts_speed, tts_text[:60])
|
xtts_voice or "default", xtts_speed, tts_target or "(broadcast)", tts_text[:60])
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.error("[core] XTTS-Request fehlgeschlagen: %s — kein Audio", e)
|
logger.error("[core] XTTS-Request fehlgeschlagen: %s — kein Audio", e)
|
||||||
|
|
||||||
@@ -1741,7 +1759,7 @@ class ARIABridge:
|
|||||||
"""Broadcastet die aktuelle voice_config.json einmalig nach RVS-Connect.
|
"""Broadcastet die aktuelle voice_config.json einmalig nach RVS-Connect.
|
||||||
|
|
||||||
Damit bekommen frisch verbundene Bridges (insbesondere die f5tts-bridge
|
Damit bekommen frisch verbundene Bridges (insbesondere die f5tts-bridge
|
||||||
auf der Gamebox nach Container-Restart) die zuletzt in Diagnostic
|
auf der AI-Box nach Container-Restart) die zuletzt in Diagnostic
|
||||||
gewaehlten Settings — ohne dass der User in Diagnostic was klicken muss.
|
gewaehlten Settings — ohne dass der User in Diagnostic was klicken muss.
|
||||||
"""
|
"""
|
||||||
try:
|
try:
|
||||||
@@ -1839,16 +1857,16 @@ class ARIABridge:
|
|||||||
return " ".join(parts) + " " + text
|
return " ".join(parts) + " " + text
|
||||||
return text
|
return text
|
||||||
|
|
||||||
def _build_pending_files_message(self, user_text: str) -> str:
|
def _build_pending_files_message(self, user_text: str, files: list) -> str:
|
||||||
"""Baut eine Anweisung an aria-core aus den gepufferten Files + optionalem
|
"""Baut eine Anweisung an aria-core aus den uebergebenen Files + optionalem
|
||||||
User-Text. user_text leer → 'warte auf Anweisung'-Variante."""
|
User-Text. user_text leer → 'warte auf Anweisung'-Variante."""
|
||||||
parts: list[str] = []
|
parts: list[str] = []
|
||||||
for fp, name, ftype, kb, w, h in self._pending_files:
|
for fp, name, ftype, kb, w, h, _cmid in files:
|
||||||
dim = f" {w}x{h}px" if (w and h) else ""
|
dim = f" {w}x{h}px" if (w and h) else ""
|
||||||
kind = "Bild" if ftype.startswith("image/") else "Datei"
|
kind = "Bild" if ftype.startswith("image/") else "Datei"
|
||||||
parts.append(f"- {kind}: {name}{dim} ({ftype}, {kb}KB) liegt unter {fp}")
|
parts.append(f"- {kind}: {name}{dim} ({ftype}, {kb}KB) liegt unter {fp}")
|
||||||
files_summary = "\n".join(parts)
|
files_summary = "\n".join(parts)
|
||||||
n = len(self._pending_files)
|
n = len(files)
|
||||||
anhang = "Anhang" if n == 1 else "Anhaenge"
|
anhang = "Anhang" if n == 1 else "Anhaenge"
|
||||||
if user_text:
|
if user_text:
|
||||||
return (f"Stefan hat dir {n} {anhang} geschickt:\n{files_summary}\n\n"
|
return (f"Stefan hat dir {n} {anhang} geschickt:\n{files_summary}\n\n"
|
||||||
@@ -1857,15 +1875,16 @@ class ARIABridge:
|
|||||||
f"Warte auf seine Anweisung was du damit tun sollst.")
|
f"Warte auf seine Anweisung was du damit tun sollst.")
|
||||||
|
|
||||||
async def _flush_pending_files_after(self, delay: float) -> None:
|
async def _flush_pending_files_after(self, delay: float) -> None:
|
||||||
"""Wenn nach `delay`s kein chat-Text gekommen ist: Files alleine an
|
"""Wenn nach `delay`s kein chat-Text gekommen ist: alle noch gepufferten
|
||||||
aria-core senden ('warte auf Anweisung'-Variante)."""
|
Files alleine an aria-core senden ('warte auf Anweisung'-Variante)."""
|
||||||
try:
|
try:
|
||||||
await asyncio.sleep(delay)
|
await asyncio.sleep(delay)
|
||||||
except asyncio.CancelledError:
|
except asyncio.CancelledError:
|
||||||
return
|
return
|
||||||
if not self._pending_files:
|
if not self._pending_files:
|
||||||
return
|
return
|
||||||
text = self._build_pending_files_message("")
|
files = self._pending_files
|
||||||
|
text = self._build_pending_files_message("", files)
|
||||||
self._pending_files = []
|
self._pending_files = []
|
||||||
self._pending_files_flush_task = None
|
self._pending_files_flush_task = None
|
||||||
pid = self._pending_files_project_id
|
pid = self._pending_files_project_id
|
||||||
@@ -1873,23 +1892,48 @@ class ARIABridge:
|
|||||||
await self.send_to_core(text, source="app-file", project_id=pid)
|
await self.send_to_core(text, source="app-file", project_id=pid)
|
||||||
|
|
||||||
async def _flush_pending_files_with_text(self, user_text: str,
|
async def _flush_pending_files_with_text(self, user_text: str,
|
||||||
project_id: str = "") -> bool:
|
project_id: str = "",
|
||||||
|
client_msg_id: str = "") -> bool:
|
||||||
"""Wenn ein chat-Text reinkommt waehrend Files gepuffert sind:
|
"""Wenn ein chat-Text reinkommt waehrend Files gepuffert sind:
|
||||||
Files + Text zu einer einzigen aria-core-Nachricht mergen.
|
Files + Text zu einer einzigen aria-core-Nachricht mergen.
|
||||||
Returns True wenn gemerged wurde (Caller soll dann nicht nochmal senden).
|
Returns True wenn gemerged wurde (Caller soll dann nicht nochmal senden).
|
||||||
|
|
||||||
|
KORRELATION (Fix Queue-Bug): Files tragen dieselbe clientMsgId wie ihr
|
||||||
|
Text. Bei einer Queue gehen Files (fire-and-forget) und Text (ACK-
|
||||||
|
getrackt, ggf. verzoegert) auseinander — ohne Korrelation landeten die
|
||||||
|
Bilder beim falschen Text. Wir mergen darum NUR die Files mit passender
|
||||||
|
cmid; der Rest bleibt gepuffert fuer seine eigene Nachricht. Fallback
|
||||||
|
(Legacy-App ohne cmid an Files, oder cmid ohne Treffer): altes Verhalten
|
||||||
|
(alle Files mit diesem Text), damit nie ein Bild verloren geht.
|
||||||
|
|
||||||
project_id: Projekt-Kontext aus dem chat-Payload (der sichtbare Focus
|
project_id: Projekt-Kontext aus dem chat-Payload (der sichtbare Focus
|
||||||
beim Absenden). Faellt auf den beim File-Upload gemerkten Kontext
|
beim Absenden). Faellt auf den beim File-Upload gemerkten Kontext zurueck.
|
||||||
zurueck, damit Anhaenge im richtigen Projekt landen statt im Hauptchat."""
|
"""
|
||||||
if not self._pending_files:
|
if not self._pending_files:
|
||||||
return False
|
return False
|
||||||
|
cmid = (client_msg_id or "").strip()
|
||||||
|
matching = [f for f in self._pending_files if cmid and f[6] == cmid]
|
||||||
|
if not matching:
|
||||||
|
# Kein cmid-Treffer → altes Verhalten: alle gepufferten Files mergen.
|
||||||
|
matching = list(self._pending_files)
|
||||||
|
remaining: list = []
|
||||||
|
else:
|
||||||
|
remaining = [f for f in self._pending_files if f not in matching]
|
||||||
|
|
||||||
|
text = self._build_pending_files_message(user_text, matching)
|
||||||
|
self._pending_files = remaining
|
||||||
|
pid = (project_id or "").strip() or self._pending_files_project_id
|
||||||
|
# Flush-Timer neu setzen wenn noch Files anderer Nachrichten warten,
|
||||||
|
# sonst zuruecksetzen.
|
||||||
if self._pending_files_flush_task and not self._pending_files_flush_task.done():
|
if self._pending_files_flush_task and not self._pending_files_flush_task.done():
|
||||||
self._pending_files_flush_task.cancel()
|
self._pending_files_flush_task.cancel()
|
||||||
self._pending_files_flush_task = None
|
self._pending_files_flush_task = None
|
||||||
text = self._build_pending_files_message(user_text)
|
if remaining:
|
||||||
self._pending_files = []
|
self._pending_files_flush_task = asyncio.create_task(
|
||||||
pid = (project_id or "").strip() or self._pending_files_project_id
|
self._flush_pending_files_after(self._PENDING_FILES_WINDOW_SEC)
|
||||||
self._pending_files_project_id = ""
|
)
|
||||||
|
else:
|
||||||
|
self._pending_files_project_id = ""
|
||||||
# create_task statt await — sonst blockt der RVS-recv-Loop bis Brain
|
# create_task statt await — sonst blockt der RVS-recv-Loop bis Brain
|
||||||
# fertig ist (siehe chat-handler oben).
|
# fertig ist (siehe chat-handler oben).
|
||||||
asyncio.create_task(self.send_to_core(text, source="app-file+chat", project_id=pid))
|
asyncio.create_task(self.send_to_core(text, source="app-file+chat", project_id=pid))
|
||||||
@@ -1936,10 +1980,13 @@ class ARIABridge:
|
|||||||
url, data=payload, method="POST",
|
url, data=payload, method="POST",
|
||||||
headers={"Content-Type": "application/json"},
|
headers={"Content-Type": "application/json"},
|
||||||
)
|
)
|
||||||
# 20 Min Timeout — lange Multi-Tool-Workflows (Karten,
|
# Timeout MUSS zum Proxy passen (der laesst ARIA bis 24h
|
||||||
# PDFs, viele curl-Calls) brauchen das. 5 Min waren chronisch
|
# rechnen). 1200s (20 Min) war zu knapp: lange Software-Dev-
|
||||||
# zu knapp und haben ARIA mitten in der Arbeit gekappt.
|
# Turns dauern laenger → die Bridge gab auf, die Antwort ging
|
||||||
with urllib.request.urlopen(req, timeout=1200) as resp:
|
# verloren (kein Bubble), der Kontext blieb auf 'running'
|
||||||
|
# haengen. Jetzt 24h (env BRAIN_CHAT_TIMEOUT_SEC).
|
||||||
|
_chat_timeout = float(os.environ.get("BRAIN_CHAT_TIMEOUT_SEC", "86400"))
|
||||||
|
with urllib.request.urlopen(req, timeout=_chat_timeout) as resp:
|
||||||
return resp.status, resp.read().decode("utf-8", errors="ignore")
|
return resp.status, resp.read().decode("utf-8", errors="ignore")
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
return None, str(exc)
|
return None, str(exc)
|
||||||
@@ -1987,6 +2034,10 @@ class ARIABridge:
|
|||||||
# Stellt ARIA eine blockierende Rueckfrage? Dann pausiert die App die
|
# Stellt ARIA eine blockierende Rueckfrage? Dann pausiert die App die
|
||||||
# Projekt-Queue und leitet die naechste Eingabe als Antwort weiter.
|
# Projekt-Queue und leitet die naechste Eingabe als Antwort weiter.
|
||||||
awaiting_reply = bool(data.get("awaiting_reply", False))
|
awaiting_reply = bool(data.get("awaiting_reply", False))
|
||||||
|
# User hat per Sprache "Wake-Word aus" gesagt → App stoppt den Listener.
|
||||||
|
wake_off = bool(data.get("wake_off", False))
|
||||||
|
# "Wake-Word an" (Text/Aufnahme-Button) → App startet den Listener wieder.
|
||||||
|
wake_on = bool(data.get("wake_on", False))
|
||||||
|
|
||||||
# Side-Channel-Events VOR der Chat-Bubble broadcasten (z.B. skill_created)
|
# Side-Channel-Events VOR der Chat-Bubble broadcasten (z.B. skill_created)
|
||||||
# damit sie in der UI vor der Reply auftauchen
|
# damit sie in der UI vor der Reply auftauchen
|
||||||
@@ -2050,6 +2101,22 @@ class ARIABridge:
|
|||||||
proj = event.get("project") or {}
|
proj = event.get("project") or {}
|
||||||
logger.info("[brain] Projekt %s: %s (id=%s)",
|
logger.info("[brain] Projekt %s: %s (id=%s)",
|
||||||
event.get("action") or "?", proj.get("name"), proj.get("id"))
|
event.get("action") or "?", proj.get("name"), proj.get("id"))
|
||||||
|
elif etype == "aria_view":
|
||||||
|
# M1: ARIA hat via present_view eine raeumliche Ansicht komponiert.
|
||||||
|
# View-Spec (Orb + Karten) + Projekt-Kontext an App/Web/Diagnostic;
|
||||||
|
# deren Renderer materialisieren die Karten auf der Flaeche.
|
||||||
|
view = event.get("view") or {}
|
||||||
|
await self._send_to_rvs({
|
||||||
|
"type": "aria_view",
|
||||||
|
"payload": {
|
||||||
|
"view": view,
|
||||||
|
"projectId": event.get("project_id") or "",
|
||||||
|
"clientMsgId": client_msg_id or "",
|
||||||
|
},
|
||||||
|
"timestamp": int(asyncio.get_event_loop().time() * 1000),
|
||||||
|
})
|
||||||
|
logger.info("[brain] ARIA hat eine Ansicht geschickt: %d Karte(n), orb=%s",
|
||||||
|
len(view.get("cards") or []), view.get("orb"))
|
||||||
|
|
||||||
# _process_core_response uebernimmt alles weitere:
|
# _process_core_response uebernimmt alles weitere:
|
||||||
# File-Marker extrahieren + broadcasten, NO_REPLY-Check, Chat-
|
# File-Marker extrahieren + broadcasten, NO_REPLY-Check, Chat-
|
||||||
@@ -2062,7 +2129,9 @@ class ARIABridge:
|
|||||||
"answeredBy": answered_by,
|
"answeredBy": answered_by,
|
||||||
"speak": speak,
|
"speak": speak,
|
||||||
"converse": converse,
|
"converse": converse,
|
||||||
"awaiting_reply": awaiting_reply})
|
"awaiting_reply": awaiting_reply,
|
||||||
|
"wake_off": wake_off,
|
||||||
|
"wake_on": wake_on})
|
||||||
except Exception:
|
except Exception:
|
||||||
logger.exception("[brain] _process_core_response Fehler")
|
logger.exception("[brain] _process_core_response Fehler")
|
||||||
await self._emit_activity("idle", "", project_id=project_id)
|
await self._emit_activity("idle", "", project_id=project_id)
|
||||||
@@ -2133,7 +2202,7 @@ class ARIABridge:
|
|||||||
await self._broadcast_current_mode()
|
await self._broadcast_current_mode()
|
||||||
|
|
||||||
# Persistierte Voice-Config broadcasten — die f5tts-bridge auf
|
# Persistierte Voice-Config broadcasten — die f5tts-bridge auf
|
||||||
# der Gamebox bekommt damit nach Restart die zuletzt in
|
# der AI-Box bekommt damit nach Restart die zuletzt in
|
||||||
# Diagnostic gewaehlten Settings wieder (sonst stuende sie auf
|
# Diagnostic gewaehlten Settings wieder (sonst stuende sie auf
|
||||||
# ihren Hard-Defaults).
|
# ihren Hard-Defaults).
|
||||||
asyncio.create_task(self._broadcast_persisted_config())
|
asyncio.create_task(self._broadcast_persisted_config())
|
||||||
@@ -2364,7 +2433,8 @@ class ARIABridge:
|
|||||||
# gesendet), mergen wir sie zu einer einzigen Anfrage statt
|
# gesendet), mergen wir sie zu einer einzigen Anfrage statt
|
||||||
# zwei separater send_to_core-Calls.
|
# zwei separater send_to_core-Calls.
|
||||||
merged = await self._flush_pending_files_with_text(
|
merged = await self._flush_pending_files_with_text(
|
||||||
text, project_id=str(payload.get("projectId") or ""))
|
text, project_id=str(payload.get("projectId") or ""),
|
||||||
|
client_msg_id=client_msg_id or "")
|
||||||
if merged:
|
if merged:
|
||||||
logger.info("[rvs] App-Chat (mit Anhaengen) project=%s: '%s'",
|
logger.info("[rvs] App-Chat (mit Anhaengen) project=%s: '%s'",
|
||||||
str(payload.get("projectId") or "") or "(main)", text[:80])
|
str(payload.get("projectId") or "") or "(main)", text[:80])
|
||||||
@@ -2406,6 +2476,20 @@ class ARIABridge:
|
|||||||
await self._emit_activity("idle", "", project_id=cancel_pid)
|
await self._emit_activity("idle", "", project_id=cancel_pid)
|
||||||
return
|
return
|
||||||
|
|
||||||
|
if msg_type == "interject":
|
||||||
|
# Zwischenruf: waehrend eines laufenden Turns eine Korrektur
|
||||||
|
# reinschieben — KEIN Abbruch, keine Queue. Geht an den Proxy-
|
||||||
|
# internen /interject, der die Message in den laufenden Subprozess
|
||||||
|
# des Kontexts schreibt (claude greift sie an der naechsten Tool-
|
||||||
|
# Grenze auf).
|
||||||
|
interject_pid = str(payload.get("projectId") or "")
|
||||||
|
interject_text = str(payload.get("text") or "")
|
||||||
|
logger.info("[rvs] Zwischenruf project=%s: '%s'",
|
||||||
|
interject_pid or "(main)", interject_text[:80])
|
||||||
|
if interject_text.strip():
|
||||||
|
await self._interject_proxy_for_project(interject_pid, interject_text)
|
||||||
|
return
|
||||||
|
|
||||||
elif msg_type == "audio_pcm":
|
elif msg_type == "audio_pcm":
|
||||||
# Audio-PCM geht direkt von XTTS-Bridge an die App.
|
# Audio-PCM geht direkt von XTTS-Bridge an die App.
|
||||||
# Die aria-bridge darf es NICHT rebroadcasten — sonst bekommt die App
|
# Die aria-bridge darf es NICHT rebroadcasten — sonst bekommt die App
|
||||||
@@ -2479,7 +2563,7 @@ class ARIABridge:
|
|||||||
elif msg_type == "config":
|
elif msg_type == "config":
|
||||||
# Konfiguration von App/Diagnostic empfangen + persistent speichern.
|
# Konfiguration von App/Diagnostic empfangen + persistent speichern.
|
||||||
# Felder die nicht direkt zur aria-bridge gehoeren (f5tts*) werden
|
# Felder die nicht direkt zur aria-bridge gehoeren (f5tts*) werden
|
||||||
# nur persistiert; die f5tts-bridge auf der Gamebox empfaengt den
|
# nur persistiert; die f5tts-bridge auf der AI-Box empfaengt den
|
||||||
# gleichen RVS-Broadcast und reagiert selber.
|
# gleichen RVS-Broadcast und reagiert selber.
|
||||||
changed = False
|
changed = False
|
||||||
if "ttsEnabled" in payload:
|
if "ttsEnabled" in payload:
|
||||||
@@ -2503,7 +2587,7 @@ class ARIABridge:
|
|||||||
new_model = payload["whisperModel"]
|
new_model = payload["whisperModel"]
|
||||||
allowed = {"tiny", "base", "small", "medium", "large-v3"}
|
allowed = {"tiny", "base", "small", "medium", "large-v3"}
|
||||||
if new_model in allowed and new_model != self.stt_engine.model_size:
|
if new_model in allowed and new_model != self.stt_engine.model_size:
|
||||||
logger.info("[rvs] Whisper-Modell → %s (nur Config; Modell laedt Gamebox)",
|
logger.info("[rvs] Whisper-Modell → %s (nur Config; Modell laedt AI-Box)",
|
||||||
new_model)
|
new_model)
|
||||||
self.stt_engine.model_size = new_model
|
self.stt_engine.model_size = new_model
|
||||||
self.stt_engine.model = None
|
self.stt_engine.model = None
|
||||||
@@ -2658,8 +2742,11 @@ class ARIABridge:
|
|||||||
logger.warning("[rvs] Bild-Resize fehlgeschlagen (%s) — Original wird genutzt: %s",
|
logger.warning("[rvs] Bild-Resize fehlgeschlagen (%s) — Original wird genutzt: %s",
|
||||||
file_name, e)
|
file_name, e)
|
||||||
|
|
||||||
# In Pending-Queue + Flush-Timer (anti-spam Buffering)
|
# In Pending-Queue + Flush-Timer (anti-spam Buffering).
|
||||||
self._pending_files.append((file_path, file_name, file_type, size_kb, int(width or 0), int(height or 0)))
|
# clientMsgId mitpuffern → spaeterer Text-Flush ordnet die Datei
|
||||||
|
# genau SEINER Nachricht zu (Queue-Korrelation, s. _flush_*).
|
||||||
|
file_cmid = str(payload.get("clientMsgId") or "")
|
||||||
|
self._pending_files.append((file_path, file_name, file_type, size_kb, int(width or 0), int(height or 0), file_cmid))
|
||||||
if self._pending_files_flush_task and not self._pending_files_flush_task.done():
|
if self._pending_files_flush_task and not self._pending_files_flush_task.done():
|
||||||
self._pending_files_flush_task.cancel()
|
self._pending_files_flush_task.cancel()
|
||||||
self._pending_files_flush_task = asyncio.create_task(
|
self._pending_files_flush_task = asyncio.create_task(
|
||||||
@@ -2719,8 +2806,18 @@ class ARIABridge:
|
|||||||
"message": payload.get("message", ""),
|
"message": payload.get("message", ""),
|
||||||
"stack": payload.get("stack", ""),
|
"stack": payload.get("stack", ""),
|
||||||
}
|
}
|
||||||
with (log_dir / "app.log").open("a", encoding="utf-8") as f:
|
log_path = log_dir / "app.log"
|
||||||
|
with log_path.open("a", encoding="utf-8") as f:
|
||||||
f.write(json.dumps(line, ensure_ascii=False) + "\n")
|
f.write(json.dumps(line, ensure_ascii=False) + "\n")
|
||||||
|
# Rotation: app.log waechst sonst unbegrenzt (jede App-Log-Zeile
|
||||||
|
# haengt an). Bei >5 MB die letzten 2000 Zeilen behalten.
|
||||||
|
try:
|
||||||
|
if log_path.stat().st_size > 5 * 1024 * 1024:
|
||||||
|
tail = log_path.read_text(encoding="utf-8",
|
||||||
|
errors="ignore").splitlines()[-2000:]
|
||||||
|
log_path.write_text("\n".join(tail) + "\n", encoding="utf-8")
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
logger.info("[app-log] %s %s: %s",
|
logger.info("[app-log] %s %s: %s",
|
||||||
line["level"], line["scope"], line["message"][:120])
|
line["level"], line["scope"], line["message"][:120])
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
@@ -3361,7 +3458,7 @@ class ARIABridge:
|
|||||||
return
|
return
|
||||||
|
|
||||||
elif msg_type == "llm_response":
|
elif msg_type == "llm_response":
|
||||||
# Antwort des llm-adapter (Gamebox) auf unseren llm_request.
|
# Antwort des llm-adapter (AI-Box) auf unseren llm_request.
|
||||||
request_id = payload.get("requestId", "")
|
request_id = payload.get("requestId", "")
|
||||||
future = self._pending_llm.get(request_id)
|
future = self._pending_llm.get(request_id)
|
||||||
if future is None or future.done():
|
if future is None or future.done():
|
||||||
@@ -3370,7 +3467,7 @@ class ARIABridge:
|
|||||||
return
|
return
|
||||||
|
|
||||||
elif msg_type == "service_status":
|
elif msg_type == "service_status":
|
||||||
# Gamebox-Bridges (whisper / f5tts / flux) melden ihren Lade-Status.
|
# AI-Box-Bridges (whisper / f5tts / flux) melden ihren Lade-Status.
|
||||||
# Wir nutzen das fuer den dynamischen STT-Timeout: solange whisper
|
# Wir nutzen das fuer den dynamischen STT-Timeout: solange whisper
|
||||||
# im 'loading' steckt, geben wir der Bridge mehr Zeit (Modell-Download
|
# im 'loading' steckt, geben wir der Bridge mehr Zeit (Modell-Download
|
||||||
# kann 1-2 Min dauern), statt nach 45s lokal zu fallbacken.
|
# kann 1-2 Min dauern), statt nach 45s lokal zu fallbacken.
|
||||||
@@ -3453,6 +3550,62 @@ class ARIABridge:
|
|||||||
self._satellites[sid]["caps"], self._satellites[sid]["control"])
|
self._satellites[sid]["caps"], self._satellites[sid]["control"])
|
||||||
return
|
return
|
||||||
|
|
||||||
|
elif msg_type == "worker_hello":
|
||||||
|
iid = (payload.get("instanceId") or "").strip()
|
||||||
|
if iid:
|
||||||
|
prev = self._workers.get(iid, {})
|
||||||
|
_models = payload.get("models")
|
||||||
|
self._workers[iid] = {
|
||||||
|
"instanceId": iid,
|
||||||
|
"service": payload.get("service") or "",
|
||||||
|
"node": payload.get("node") or "",
|
||||||
|
"gpus": payload.get("gpus") or "",
|
||||||
|
"model": payload.get("model") or "",
|
||||||
|
# models: welche Modelle die Box fahren kann (llm/llama-swap).
|
||||||
|
# Fallback auf [model] fuer alte Adapter ohne models-Feld.
|
||||||
|
"models": [m for m in _models if m] if isinstance(_models, list)
|
||||||
|
else ([payload.get("model")] if payload.get("model") else []),
|
||||||
|
"busy": bool(prev.get("busy", False)),
|
||||||
|
"last_seen": time.time(),
|
||||||
|
}
|
||||||
|
logger.info("[worker] online: %s (service=%s node=%s gpus=%s model=%s)",
|
||||||
|
iid, self._workers[iid]["service"], self._workers[iid]["node"],
|
||||||
|
self._workers[iid]["gpus"] or "?", self._workers[iid]["model"] or "?")
|
||||||
|
return
|
||||||
|
|
||||||
|
elif msg_type == "stt_lease_request":
|
||||||
|
# Die App fragt vor dem Aufnahme-Stream, welche STT-Instanz sie
|
||||||
|
# adressieren soll (Redundanz ueber mehrere Apps/Nodes). Wir waehlen
|
||||||
|
# eine freie STT-Instanz und antworten per stt_lease. Ist keine
|
||||||
|
# Instanz bekannt (instanceId leer), streamt die App wie bisher an
|
||||||
|
# ALLE (Broadcast) — Single-Node bleibt unveraendert.
|
||||||
|
req_id = (payload.get("requestId") or "").strip()
|
||||||
|
iid = self._pick_stt_worker() or ""
|
||||||
|
await self._send_to_rvs({
|
||||||
|
"type": "stt_lease",
|
||||||
|
"payload": {"requestId": req_id, "instanceId": iid},
|
||||||
|
"timestamp": int(time.time() * 1000),
|
||||||
|
})
|
||||||
|
logger.info("[stt-lease] req=%s → %s", req_id[:8] if req_id else "?",
|
||||||
|
iid or "(broadcast)")
|
||||||
|
return
|
||||||
|
|
||||||
|
elif msg_type == "worker_ping":
|
||||||
|
iid = (payload.get("instanceId") or "").strip()
|
||||||
|
if iid:
|
||||||
|
w = self._workers.get(iid)
|
||||||
|
if w is None:
|
||||||
|
# Ping ohne vorheriges hello (Bridge-Neustart) → Minimal-Eintrag,
|
||||||
|
# service aus der instanceId ableiten (Form: "service@node").
|
||||||
|
svc = iid.split("@", 1)[0]
|
||||||
|
w = self._workers[iid] = {
|
||||||
|
"instanceId": iid, "service": svc, "node": "", "gpus": "",
|
||||||
|
"model": "", "busy": False, "last_seen": 0.0,
|
||||||
|
}
|
||||||
|
w["busy"] = bool(payload.get("busy", False))
|
||||||
|
w["last_seen"] = time.time()
|
||||||
|
return
|
||||||
|
|
||||||
elif msg_type in ("sat_devices", "sat_result"):
|
elif msg_type in ("sat_devices", "sat_result"):
|
||||||
req_id = payload.get("requestId", "")
|
req_id = payload.get("requestId", "")
|
||||||
future = self._pending_sat.get(req_id)
|
future = self._pending_sat.get(req_id)
|
||||||
@@ -3478,11 +3631,11 @@ class ARIABridge:
|
|||||||
else:
|
else:
|
||||||
logger.debug("[rvs] Unbekannter Typ: %s", msg_type)
|
logger.debug("[rvs] Unbekannter Typ: %s", msg_type)
|
||||||
|
|
||||||
# STT-Orchestrierung: zuerst Remote (Gamebox), Fallback lokal.
|
# STT-Orchestrierung: zuerst Remote (AI-Box), Fallback lokal.
|
||||||
# Zwei Timeouts:
|
# Zwei Timeouts:
|
||||||
# ready=True → 45s reicht selbst fuer lange Audios
|
# ready=True → 45s reicht selbst fuer lange Audios
|
||||||
# ready=False → 300s, weil das Modell evtl. noch heruntergeladen wird
|
# ready=False → 300s, weil das Modell evtl. noch heruntergeladen wird
|
||||||
# (large-v3 ~3GB, kann auf der Gamebox 1-2 Min dauern).
|
# (large-v3 ~3GB, kann auf der AI-Box 1-2 Min dauern).
|
||||||
_STT_REMOTE_TIMEOUT_READY_S = 45.0
|
_STT_REMOTE_TIMEOUT_READY_S = 45.0
|
||||||
_STT_REMOTE_TIMEOUT_LOADING_S = 300.0
|
_STT_REMOTE_TIMEOUT_LOADING_S = 300.0
|
||||||
|
|
||||||
@@ -3835,15 +3988,15 @@ class ARIABridge:
|
|||||||
_FLUX_TIMEOUT_LOADING_S = 900.0 # 15 min beim allerersten Mal (Modell-Download)
|
_FLUX_TIMEOUT_LOADING_S = 900.0 # 15 min beim allerersten Mal (Modell-Download)
|
||||||
|
|
||||||
# ── Local-LLM-Roundtrip: Brain → Bridge → RVS → llm-adapter → zurueck ──
|
# ── Local-LLM-Roundtrip: Brain → Bridge → RVS → llm-adapter → zurueck ──
|
||||||
# Qwen3 auf der Gamebox antwortet auf kurze Turns in <1 s. Grosszuegiger
|
# Qwen3 auf der AI-Box antwortet auf kurze Turns in <1 s. Grosszuegiger
|
||||||
# Timeout deckt Kaltstart / laengere Antworten / Netz-Jitter (Gamebox@home)
|
# Timeout deckt Kaltstart / laengere Antworten / Netz-Jitter (AI-Box@home)
|
||||||
# ab. Bei Timeout faellt der Router im Brain per Escalation auf Claude.
|
# ab. Bei Timeout faellt der Router im Brain per Escalation auf Claude.
|
||||||
_LLM_TIMEOUT_S = 30.0
|
_LLM_TIMEOUT_S = 30.0
|
||||||
|
|
||||||
async def _local_llm(self, messages: list, max_tokens: int = 512,
|
async def _local_llm(self, messages: list, max_tokens: int = 512,
|
||||||
temperature: float = 0.7, stop=None, tools=None,
|
temperature: float = 0.7, stop=None, tools=None,
|
||||||
model=None) -> dict:
|
model=None) -> dict:
|
||||||
"""Schickt einen llm_request an den llm-adapter (Gamebox), wartet auf
|
"""Schickt einen llm_request an den llm-adapter (AI-Box), wartet auf
|
||||||
llm_response. tools (B1b) werden durchgereicht; tool_calls kommen zurueck.
|
llm_response. tools (B1b) werden durchgereicht; tool_calls kommen zurueck.
|
||||||
Rueckgabe: {ok, content, tool_calls, model, elapsedMs} oder {ok:False, error}."""
|
Rueckgabe: {ok, content, tool_calls, model, elapsedMs} oder {ok:False, error}."""
|
||||||
if self.ws_rvs is None:
|
if self.ws_rvs is None:
|
||||||
@@ -3868,8 +4021,15 @@ class ARIABridge:
|
|||||||
req_payload["tools"] = tools
|
req_payload["tools"] = tools
|
||||||
if model:
|
if model:
|
||||||
req_payload["model"] = model
|
req_payload["model"] = model
|
||||||
logger.info("[rvs] llm_request → llm-adapter (id=%s, msgs=%d, max_tokens=%d, tools=%d, model=%s)",
|
# Redundanz/Multitasking: freie llm-Instanz gezielt adressieren, die
|
||||||
request_id[:8], len(messages), max_tokens, len(tools) if tools else 0, model or "-")
|
# das gewaehlte Modell fahren kann; None → Broadcast wie bisher.
|
||||||
|
# Mehrere Boxen mit demselben Modell → Round-Robin (pro Projekt verteilt).
|
||||||
|
llm_target = self._pick_worker("llm", model=model or None)
|
||||||
|
if llm_target:
|
||||||
|
req_payload["targetInstance"] = llm_target
|
||||||
|
logger.info("[rvs] llm_request → llm-adapter (id=%s, msgs=%d, max_tokens=%d, tools=%d, model=%s, target=%s)",
|
||||||
|
request_id[:8], len(messages), max_tokens, len(tools) if tools else 0,
|
||||||
|
model or "-", llm_target or "(broadcast)")
|
||||||
ok = await self._send_to_rvs({
|
ok = await self._send_to_rvs({
|
||||||
"type": "llm_request",
|
"type": "llm_request",
|
||||||
"payload": req_payload,
|
"payload": req_payload,
|
||||||
@@ -3880,7 +4040,7 @@ class ARIABridge:
|
|||||||
try:
|
try:
|
||||||
result = await asyncio.wait_for(future, timeout=self._LLM_TIMEOUT_S)
|
result = await asyncio.wait_for(future, timeout=self._LLM_TIMEOUT_S)
|
||||||
except asyncio.TimeoutError:
|
except asyncio.TimeoutError:
|
||||||
return {"ok": False, "error": f"Timeout ({self._LLM_TIMEOUT_S:.0f}s) — Gamebox nicht erreichbar?"}
|
return {"ok": False, "error": f"Timeout ({self._LLM_TIMEOUT_S:.0f}s) — AI-Box nicht erreichbar?"}
|
||||||
if not isinstance(result, dict) or not result.get("ok"):
|
if not isinstance(result, dict) or not result.get("ok"):
|
||||||
err = (result or {}).get("error") if isinstance(result, dict) else "leeres Resultat"
|
err = (result or {}).get("error") if isinstance(result, dict) else "leeres Resultat"
|
||||||
return {"ok": False, "error": err or "llm-adapter Fehler"}
|
return {"ok": False, "error": err or "llm-adapter Fehler"}
|
||||||
@@ -4101,6 +4261,28 @@ class ARIABridge:
|
|||||||
logger.info("[cancel] proxy /cancel project=%s: %s %s",
|
logger.info("[cancel] proxy /cancel project=%s: %s %s",
|
||||||
project_id or "(main)", status, body)
|
project_id or "(main)", status, body)
|
||||||
|
|
||||||
|
async def _interject_proxy_for_project(self, project_id: str, text: str) -> None:
|
||||||
|
"""Zwischenruf: schiebt eine User-Message in den laufenden Turn dieses
|
||||||
|
Kontexts (proxy-internes /interject) — ohne Abbruch. claude greift sie
|
||||||
|
an der naechsten Tool-Grenze auf."""
|
||||||
|
url = os.environ.get("PROXY_INTERNAL_URL", "http://aria-proxy:3457") + "/interject"
|
||||||
|
data = json.dumps({"projectId": project_id or "", "text": text}).encode("utf-8")
|
||||||
|
|
||||||
|
def _do_request():
|
||||||
|
try:
|
||||||
|
req = urllib.request.Request(
|
||||||
|
url, method="POST", data=data,
|
||||||
|
headers={"Content-Type": "application/json"},
|
||||||
|
)
|
||||||
|
with urllib.request.urlopen(req, timeout=3) as resp:
|
||||||
|
return resp.status, resp.read().decode("utf-8", "ignore")[:200]
|
||||||
|
except Exception as e:
|
||||||
|
return f"error: {e}", ""
|
||||||
|
|
||||||
|
status, body = await asyncio.get_event_loop().run_in_executor(None, _do_request)
|
||||||
|
logger.info("[interject] proxy /interject project=%s: %s %s",
|
||||||
|
project_id or "(main)", status, body)
|
||||||
|
|
||||||
async def _emit_activity(self, activity: str, tool: str = "", force: bool = False,
|
async def _emit_activity(self, activity: str, tool: str = "", force: bool = False,
|
||||||
project_id: str = "") -> None:
|
project_id: str = "") -> None:
|
||||||
"""Sendet agent_activity an die App — nur wenn sich der State geaendert hat.
|
"""Sendet agent_activity an die App — nur wenn sich der State geaendert hat.
|
||||||
@@ -4338,6 +4520,9 @@ class ARIABridge:
|
|||||||
elif method == "POST" and path == "/internal/satellite-list":
|
elif method == "POST" and path == "/internal/satellite-list":
|
||||||
# Brain fragt: welche Satelliten/Netze sind online + Capabilities.
|
# Brain fragt: welche Satelliten/Netze sind online + Capabilities.
|
||||||
await _send_response(writer, 200, {"ok": True, "satellites": self._satellite_list()})
|
await _send_response(writer, 200, {"ok": True, "satellites": self._satellite_list()})
|
||||||
|
elif method in ("GET", "POST") and path == "/internal/worker-list":
|
||||||
|
# Diagnostic/Brain fragt: welche Compute-Worker sind online (Flotte).
|
||||||
|
await _send_response(writer, 200, {"ok": True, "workers": self._worker_list()})
|
||||||
elif method == "POST" and path == "/internal/satellite":
|
elif method == "POST" and path == "/internal/satellite":
|
||||||
# Brain-Tool: Discovery oder Command an einen Satelliten.
|
# Brain-Tool: Discovery oder Command an einen Satelliten.
|
||||||
# body: {op:'discover'|'command', satellite, device?, action?, params?}
|
# body: {op:'discover'|'command', satellite, device?, action?, params?}
|
||||||
@@ -4358,7 +4543,7 @@ class ARIABridge:
|
|||||||
await _send_response(writer, 200, result)
|
await _send_response(writer, 200, result)
|
||||||
elif method == "POST" and path == "/internal/flux-generate":
|
elif method == "POST" and path == "/internal/flux-generate":
|
||||||
# Vom Brain (flux_generate-Tool) gefeuert. Wir routen den
|
# Vom Brain (flux_generate-Tool) gefeuert. Wir routen den
|
||||||
# Render-Request via RVS an die flux-bridge (Gamebox),
|
# Render-Request via RVS an die flux-bridge (AI-Box),
|
||||||
# warten synchron auf die PNG-Antwort, speichern sie nach
|
# warten synchron auf die PNG-Antwort, speichern sie nach
|
||||||
# /shared/uploads/ und melden Pfad + Render-Stats zurueck.
|
# /shared/uploads/ und melden Pfad + Render-Stats zurueck.
|
||||||
# Brain referenziert das Bild dann mit [FILE:]-Marker in
|
# Brain referenziert das Bild dann mit [FILE:]-Marker in
|
||||||
@@ -4395,7 +4580,7 @@ class ARIABridge:
|
|||||||
await _send_response(writer, status, result)
|
await _send_response(writer, status, result)
|
||||||
elif method == "POST" and path == "/internal/local-llm":
|
elif method == "POST" and path == "/internal/local-llm":
|
||||||
# Vom Brain (Router / Testchat) gefeuert. Wir relayen den
|
# Vom Brain (Router / Testchat) gefeuert. Wir relayen den
|
||||||
# Chat-Request via RVS an den llm-adapter (Gamebox Qwen3),
|
# Chat-Request via RVS an den llm-adapter (AI-Box Qwen3),
|
||||||
# warten synchron auf llm_response und geben content zurueck.
|
# warten synchron auf llm_response und geben content zurueck.
|
||||||
try:
|
try:
|
||||||
data = json.loads(body.decode("utf-8", "ignore"))
|
data = json.loads(body.decode("utf-8", "ignore"))
|
||||||
@@ -4587,6 +4772,60 @@ class ARIABridge:
|
|||||||
})
|
})
|
||||||
return out
|
return out
|
||||||
|
|
||||||
|
# worker_ping kommt alle ~10s; nach 35s ohne Ping gilt ein Worker als offline.
|
||||||
|
WORKER_OFFLINE_S = 35
|
||||||
|
|
||||||
|
def _worker_list(self) -> list[dict]:
|
||||||
|
"""Bekannte Compute-Worker (Flotte). online = kuerzlich per Ping gesehen."""
|
||||||
|
now = time.time()
|
||||||
|
out = []
|
||||||
|
for w in self._workers.values():
|
||||||
|
out.append({
|
||||||
|
"instanceId": w["instanceId"], "service": w.get("service") or "",
|
||||||
|
"node": w.get("node") or "", "gpus": w.get("gpus") or "",
|
||||||
|
"model": w.get("model") or "", "models": w.get("models") or [],
|
||||||
|
"busy": bool(w.get("busy")),
|
||||||
|
"online": (now - w.get("last_seen", 0)) < self.WORKER_OFFLINE_S,
|
||||||
|
})
|
||||||
|
return out
|
||||||
|
|
||||||
|
def _pick_worker(self, service: str, model: Optional[str] = None) -> Optional[str]:
|
||||||
|
"""Waehlt eine online, moeglichst freie Instanz des Diensts (Round-Robin
|
||||||
|
ueber die freien). Gibt die instanceId oder None. Fuer Stage-3-Routing
|
||||||
|
(targetInstance).
|
||||||
|
|
||||||
|
model: wenn gesetzt (nur llm sinnvoll), kommen nur Boxen in Frage, die das
|
||||||
|
Modell fahren koennen (models-Liste oder legacy model-Feld). Meldet KEINE
|
||||||
|
Box das Modell → None (nachsichtig: Aufrufer faellt auf Broadcast zurueck)."""
|
||||||
|
now = time.time()
|
||||||
|
online = [w for w in self._workers.values()
|
||||||
|
if w.get("service") == service
|
||||||
|
and (now - w.get("last_seen", 0)) < self.WORKER_OFFLINE_S]
|
||||||
|
if model:
|
||||||
|
online = [w for w in online
|
||||||
|
if model in (w.get("models") or [])
|
||||||
|
or w.get("model") == model]
|
||||||
|
if not online:
|
||||||
|
return None
|
||||||
|
free = [w for w in online if not w.get("busy")]
|
||||||
|
pool = free or online # alle busy → trotzdem eine nehmen (least-bad)
|
||||||
|
# Round-Robin: rotierender Zeiger pro Dienst(+Modell).
|
||||||
|
rr_key = f"{service}:{model}" if model else service
|
||||||
|
rr = getattr(self, "_worker_rr", None)
|
||||||
|
if rr is None:
|
||||||
|
rr = self._worker_rr = {}
|
||||||
|
idx = rr.get(rr_key, 0) % len(pool)
|
||||||
|
rr[rr_key] = idx + 1
|
||||||
|
chosen = pool[idx]
|
||||||
|
chosen["busy"] = True # optimistisch, bis der naechste Ping korrigiert
|
||||||
|
return chosen["instanceId"]
|
||||||
|
|
||||||
|
def _pick_stt_worker(self) -> Optional[str]:
|
||||||
|
"""Waehlt eine STT-Instanz fuer ein App-Lease. Voxtral (Default-STT) hat
|
||||||
|
Vorrang, Whisper ist der Fallback. None → keine online (App streamt dann
|
||||||
|
ohne targetInstance = heutiges Broadcast-Verhalten)."""
|
||||||
|
return self._pick_worker("voxtral") or self._pick_worker("whisper")
|
||||||
|
|
||||||
async def _satellite_request(self, op: str, satellite: str = "",
|
async def _satellite_request(self, op: str, satellite: str = "",
|
||||||
device: str = "", action: str = "",
|
device: str = "", action: str = "",
|
||||||
params: Optional[dict] = None,
|
params: Optional[dict] = None,
|
||||||
|
|||||||
+923
-264
File diff suppressed because it is too large
Load Diff
+355
-9
@@ -349,6 +349,66 @@ function loadLocalModels() {
|
|||||||
return DEFAULT_LOCAL_MODELS;
|
return DEFAULT_LOCAL_MODELS;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ── LLM-Modell-Katalog (Stage D): herunterladbare GGUF-Modelle ───────
|
||||||
|
// /shared/config/llm_catalog.json — kuratierte Liste guter GGUF-Modelle plus
|
||||||
|
// per HuggingFace-Refresh nachgeladene. Der llm-adapter zieht ein Modell via
|
||||||
|
// -hf beim ersten Load. { id(key), hfRepo, quant, sizeGB, description, source }.
|
||||||
|
const LLM_CATALOG_FILE = "/shared/config/llm_catalog.json";
|
||||||
|
const DEFAULT_LLM_CATALOG = [
|
||||||
|
{ id: "qwen3-8b", hfRepo: "Qwen/Qwen3-8B-GGUF", quant: "Q4_K_M", ctx: 8192, sizeGB: 6, description: "Bestes Tool-Calling, passt auf 12 GB.", source: "curated" },
|
||||||
|
{ id: "qwen3-4b", hfRepo: "Qwen/Qwen3-4B-GGUF", quant: "Q4_K_M", ctx: 8192, sizeGB: 3, description: "Kleiner + flotter, etwas schwaecher.", source: "curated" },
|
||||||
|
{ id: "qwen3-14b", hfRepo: "Qwen/Qwen3-14B-GGUF", quant: "Q4_K_M", ctx: 8192, sizeGB: 10, description: "Staerker, braucht mehr VRAM (~16 GB).", source: "curated" },
|
||||||
|
{ id: "llama-3.1-8b", hfRepo: "bartowski/Meta-Llama-3.1-8B-Instruct-GGUF", quant: "Q4_K_M", ctx: 8192, sizeGB: 5, description: "Llama 3.1 8B Instruct.", source: "curated" },
|
||||||
|
{ id: "mistral-small-3", hfRepo: "bartowski/Mistral-Small-24B-Instruct-2501-GGUF", quant: "Q4_K_M", ctx: 8192, sizeGB: 14, description: "Mistral Small 24B — stark, viel VRAM.", source: "curated" },
|
||||||
|
{ id: "gemma-2-9b", hfRepo: "bartowski/gemma-2-9b-it-GGUF", quant: "Q4_K_M", ctx: 8192, sizeGB: 6, description: "Google Gemma 2 9B Instruct.", source: "curated" },
|
||||||
|
];
|
||||||
|
function loadLlmCatalog() {
|
||||||
|
try {
|
||||||
|
const arr = JSON.parse(fs.readFileSync(LLM_CATALOG_FILE, "utf-8"));
|
||||||
|
if (Array.isArray(arr) && arr.length && arr.every(m => m && typeof m.id === "string")) return arr;
|
||||||
|
} catch {}
|
||||||
|
try {
|
||||||
|
fs.mkdirSync("/shared/config", { recursive: true });
|
||||||
|
fs.writeFileSync(LLM_CATALOG_FILE, JSON.stringify(DEFAULT_LLM_CATALOG, null, 2));
|
||||||
|
} catch {}
|
||||||
|
return DEFAULT_LLM_CATALOG;
|
||||||
|
}
|
||||||
|
function saveLlmCatalog(arr) {
|
||||||
|
try {
|
||||||
|
fs.mkdirSync("/shared/config", { recursive: true });
|
||||||
|
const tmp = LLM_CATALOG_FILE + ".tmp";
|
||||||
|
fs.writeFileSync(tmp, JSON.stringify(arr, null, 2));
|
||||||
|
fs.renameSync(tmp, LLM_CATALOG_FILE);
|
||||||
|
return true;
|
||||||
|
} catch (e) { log("warn", "llm", `Katalog speichern fehlgeschlagen: ${e.message}`); return false; }
|
||||||
|
}
|
||||||
|
function slugModelId(repo) {
|
||||||
|
return String(repo).toLowerCase().replace(/^.*\//, "").replace(/-gguf$/,"").replace(/[^a-z0-9]+/g, "-").replace(/^-+|-+$/g, "") || "model";
|
||||||
|
}
|
||||||
|
// Holt populaere GGUF-Modelle von der HuggingFace-API und merged sie in den
|
||||||
|
// Katalog (kuratierte Eintraege + Beschreibungen bleiben erhalten).
|
||||||
|
async function refreshLlmCatalogFromHF() {
|
||||||
|
const url = "https://huggingface.co/api/models?search=GGUF&sort=downloads&direction=-1&limit=40";
|
||||||
|
const r = await fetch(url, { headers: { "User-Agent": "aria-diagnostic" } });
|
||||||
|
if (!r.ok) throw new Error(`HF API ${r.status}`);
|
||||||
|
const list = await r.json();
|
||||||
|
const existing = loadLlmCatalog();
|
||||||
|
const byId = new Map(existing.map(m => [m.id, m]));
|
||||||
|
let added = 0;
|
||||||
|
for (const m of (Array.isArray(list) ? list : [])) {
|
||||||
|
const repo = m.id || m.modelId;
|
||||||
|
if (!repo || !/gguf/i.test(repo)) continue;
|
||||||
|
const id = slugModelId(repo);
|
||||||
|
if (byId.has(id)) continue; // kuratierte/vorhandene nicht ueberschreiben
|
||||||
|
const entry = { id, hfRepo: repo, quant: "Q4_K_M", ctx: 8192, sizeGB: 0,
|
||||||
|
description: `HuggingFace · ${(m.downloads || 0).toLocaleString("de")} Downloads`, source: "hf" };
|
||||||
|
byId.set(id, entry); added++;
|
||||||
|
}
|
||||||
|
const merged = Array.from(byId.values());
|
||||||
|
saveLlmCatalog(merged);
|
||||||
|
return { models: merged, added };
|
||||||
|
}
|
||||||
|
|
||||||
// ── File-Project-Manifest ───────────────────────────────────────────
|
// ── File-Project-Manifest ───────────────────────────────────────────
|
||||||
// Jeder Eintrag map[absoluter_pfad] = project_id (leer = Hauptchat).
|
// Jeder Eintrag map[absoluter_pfad] = project_id (leer = Hauptchat).
|
||||||
// Wird vom files-list-Endpoint + files-set-project gepflegt.
|
// Wird vom files-list-Endpoint + files-set-project gepflegt.
|
||||||
@@ -490,6 +550,82 @@ function broadcastSatellites() {
|
|||||||
broadcast({ type: "sat_update", satellites: satelliteList() });
|
broadcast({ type: "sat_update", satellites: satelliteList() });
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ── Compute-Fleet: GPU-Worker (voxtral/whisper/f5tts/llm) ──────────
|
||||||
|
// Worker melden sich per worker_hello + halten sich per worker_ping (busy) frisch.
|
||||||
|
const workers = new Map(); // instanceId → {instanceId, service, node, gpus, model, busy, last_seen}
|
||||||
|
const WORKER_OFFLINE_MS = 35000; // ping ~10s; nach 35s ohne Ping = offline
|
||||||
|
|
||||||
|
function workerList() {
|
||||||
|
const now = Date.now();
|
||||||
|
return Array.from(workers.values()).map(w => ({
|
||||||
|
instanceId: w.instanceId, service: w.service, node: w.node,
|
||||||
|
gpus: w.gpus, model: w.model, models: w.models || [], busy: !!w.busy,
|
||||||
|
online: (now - (w.last_seen || 0)) < WORKER_OFFLINE_MS,
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
|
||||||
|
function broadcastWorkers() {
|
||||||
|
broadcast({ type: "worker_update", workers: workerList() });
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── Voice-Flotte: zentraler Stimmen-Store + Auto-Provisioning ──────
|
||||||
|
// Der Diagnostic-Server ist der "Stimmen-Bibliothekar": Stimmen liegen
|
||||||
|
// zentral als /shared/voices/{name}.tar.gz (das f5tts-Export-Artefakt =
|
||||||
|
// wav+txt). Meldet sich eine f5tts-Box (worker_hello), gleichen wir ab und
|
||||||
|
// schieben ihr fehlende Stimmen (xtts_import_voice, targetInstance) bzw. ziehen
|
||||||
|
// bei ihr vorhandene, zentral fehlende Stimmen (xtts_export_voice) in den Store.
|
||||||
|
const CENTRAL_VOICES_DIR = "/shared/voices";
|
||||||
|
const centralExportPending = new Map(); // requestId -> name (unsere eigenen Export-Anfragen)
|
||||||
|
|
||||||
|
function ensureCentralVoicesDir() {
|
||||||
|
try { fs.mkdirSync(CENTRAL_VOICES_DIR, { recursive: true }); } catch (_) {}
|
||||||
|
}
|
||||||
|
function centralVoiceNames() {
|
||||||
|
ensureCentralVoicesDir();
|
||||||
|
try {
|
||||||
|
return fs.readdirSync(CENTRAL_VOICES_DIR)
|
||||||
|
.filter(f => f.endsWith(".tar.gz"))
|
||||||
|
.map(f => f.slice(0, -7));
|
||||||
|
} catch (_) { return []; }
|
||||||
|
}
|
||||||
|
function readCentralVoiceB64(name) {
|
||||||
|
try { return fs.readFileSync(`${CENTRAL_VOICES_DIR}/${name}.tar.gz`).toString("base64"); }
|
||||||
|
catch (_) { return null; }
|
||||||
|
}
|
||||||
|
function writeCentralVoice(name, dataB64) {
|
||||||
|
ensureCentralVoicesDir();
|
||||||
|
try { fs.writeFileSync(`${CENTRAL_VOICES_DIR}/${name}.tar.gz`, Buffer.from(dataB64, "base64")); return true; }
|
||||||
|
catch (e) { log("warn", "voice", `zentral schreiben ${name} fehlgeschlagen: ${e.message}`); return false; }
|
||||||
|
}
|
||||||
|
function deleteCentralVoice(name) {
|
||||||
|
try { fs.unlinkSync(`${CENTRAL_VOICES_DIR}/${name}.tar.gz`); log("info", "voice", `zentral geloescht: ${name}`); }
|
||||||
|
catch (_) {}
|
||||||
|
}
|
||||||
|
function requestCentralExport(name) {
|
||||||
|
// Broadcast-Export-Anfrage; die Box, die die Stimme hat, antwortet. Wir
|
||||||
|
// korrelieren die Antwort ueber requestId (nur unsere eigenen verarbeiten).
|
||||||
|
const requestId = "central_" + Date.now() + "_" + Math.random().toString(36).slice(2, 8);
|
||||||
|
centralExportPending.set(requestId, name);
|
||||||
|
setTimeout(() => centralExportPending.delete(requestId), 30000);
|
||||||
|
sendToRVS_raw({ type: "xtts_export_voice", payload: { name, requestId }, timestamp: Date.now() });
|
||||||
|
}
|
||||||
|
function provisionVoiceToInstance(name, instanceId) {
|
||||||
|
const data = readCentralVoiceB64(name);
|
||||||
|
if (!data) return;
|
||||||
|
sendToRVS_raw({ type: "xtts_import_voice",
|
||||||
|
payload: { name, data, targetInstance: instanceId }, timestamp: Date.now() });
|
||||||
|
log("info", "voice", `provisioniere '${name}' → ${instanceId}`);
|
||||||
|
}
|
||||||
|
// Abgleich beim worker_hello einer f5tts-Box: push (zentral→Box) + pull (Box→zentral, seed).
|
||||||
|
function reconcileVoices(instanceId, boxVoices) {
|
||||||
|
const central = centralVoiceNames();
|
||||||
|
const boxSet = new Set(Array.isArray(boxVoices) ? boxVoices : []);
|
||||||
|
const centralSet = new Set(central);
|
||||||
|
for (const name of central) if (!boxSet.has(name)) provisionVoiceToInstance(name, instanceId);
|
||||||
|
for (const name of boxSet) if (!centralSet.has(name)) requestCentralExport(name);
|
||||||
|
log("info", "voice", `reconcile ${instanceId}: box=${boxSet.size} central=${central.length}`);
|
||||||
|
}
|
||||||
|
|
||||||
// ── OpenClaw Gateway Verbindung ─────────────────────────
|
// ── OpenClaw Gateway Verbindung ─────────────────────────
|
||||||
|
|
||||||
async function connectGateway() {
|
async function connectGateway() {
|
||||||
@@ -935,6 +1071,66 @@ function connectRVS(forcePlain) {
|
|||||||
if (p.satellite && satellites.has(p.satellite)) satellites.get(p.satellite).last_seen = Date.now();
|
if (p.satellite && satellites.has(p.satellite)) satellites.get(p.satellite).last_seen = Date.now();
|
||||||
broadcast({ type: "sat_devices", satellite: p.satellite || "",
|
broadcast({ type: "sat_devices", satellite: p.satellite || "",
|
||||||
location: p.location || "", devices: p.devices || [] });
|
location: p.location || "", devices: p.devices || [] });
|
||||||
|
} else if (msg.type === "sat_creds_list_result" || msg.type === "sat_creds_result") {
|
||||||
|
// Credential-Store-Antworten eines Satelliten → an den Browser.
|
||||||
|
broadcast({ type: msg.type, payload: msg.payload || {} });
|
||||||
|
} else if (msg.type === "worker_hello") {
|
||||||
|
// Ein Compute-Worker (GPU-Dienst) meldet sich mit seiner Identitaet.
|
||||||
|
const p = msg.payload || {};
|
||||||
|
if (p.instanceId) {
|
||||||
|
const prev = workers.get(p.instanceId) || {};
|
||||||
|
// War die Box vor diesem hello schon frisch gesehen? worker_hello wird
|
||||||
|
// jetzt alle ~30s wiederholt — Reconcile nur beim ERSTEN/erneuten
|
||||||
|
// Auftauchen, nicht bei jedem Resend.
|
||||||
|
const wasFresh = prev.last_seen && (Date.now() - prev.last_seen < WORKER_OFFLINE_MS);
|
||||||
|
workers.set(p.instanceId, {
|
||||||
|
instanceId: p.instanceId, service: p.service || "",
|
||||||
|
node: p.node || "", gpus: p.gpus || "", model: p.model || "",
|
||||||
|
models: Array.isArray(p.models) ? p.models : (p.model ? [p.model] : []),
|
||||||
|
busy: !!prev.busy, last_seen: Date.now(),
|
||||||
|
});
|
||||||
|
broadcastWorkers();
|
||||||
|
// Voice-Flotte: f5tts-Box NEU online → Stimmen abgleichen/provisionieren.
|
||||||
|
if ((p.service || "") === "f5tts" && !wasFresh) reconcileVoices(p.instanceId, p.voices);
|
||||||
|
}
|
||||||
|
} else if (msg.type === "worker_ping") {
|
||||||
|
// Heartbeat eines Workers (traegt busy-Status).
|
||||||
|
const p = msg.payload || {};
|
||||||
|
if (p.instanceId) {
|
||||||
|
let w = workers.get(p.instanceId);
|
||||||
|
if (!w) {
|
||||||
|
const svc = String(p.instanceId).split("@")[0];
|
||||||
|
w = { instanceId: p.instanceId, service: svc, node: "", gpus: "", model: "", models: [], busy: false, last_seen: 0 };
|
||||||
|
workers.set(p.instanceId, w);
|
||||||
|
}
|
||||||
|
w.busy = !!p.busy;
|
||||||
|
w.last_seen = Date.now();
|
||||||
|
broadcastWorkers();
|
||||||
|
}
|
||||||
|
} else if (msg.type === "xtts_voice_saved") {
|
||||||
|
// Neue Stimme (App- ODER Diagnostic-Upload, via RVS-Broadcast) → zentral
|
||||||
|
// sichern. Online-Boxen haben sie durch den voice_upload-Broadcast schon;
|
||||||
|
// der zentrale Store macht sie persistent + fuer spaeter joinende Boxen
|
||||||
|
// verfuegbar (die holt dann reconcileVoices ab).
|
||||||
|
const p = msg.payload || {};
|
||||||
|
if (p.name && !p.error) {
|
||||||
|
log("info", "voice", `Stimme '${p.name}' gespeichert → zentraler Ingest`);
|
||||||
|
requestCentralExport(p.name);
|
||||||
|
}
|
||||||
|
} else if (msg.type === "xtts_voice_exported") {
|
||||||
|
// Antwort auf eine UNSERER zentralen Export-Anfragen (requestId-Match) →
|
||||||
|
// in den zentralen Store schreiben. Browser-initiierte Exports tragen
|
||||||
|
// keinen centralExportPending-requestId und werden hier ignoriert.
|
||||||
|
const p = msg.payload || {};
|
||||||
|
const rid = p.requestId || "";
|
||||||
|
if (rid && centralExportPending.has(rid)) {
|
||||||
|
centralExportPending.delete(rid);
|
||||||
|
if (p.ok && p.name && p.data) writeCentralVoice(p.name, p.data);
|
||||||
|
}
|
||||||
|
} else if (msg.type === "xtts_delete_voice") {
|
||||||
|
// App-initiierter Delete (Broadcast) → zentrale Kopie mitloeschen.
|
||||||
|
const p = msg.payload || {};
|
||||||
|
if (p.name) deleteCentralVoice(p.name);
|
||||||
} else if (msg.type === "agent_activity") {
|
} else if (msg.type === "agent_activity") {
|
||||||
// Bridge meldet "ARIA denkt/schreibt/tool" oder "idle" — an Browser
|
// Bridge meldet "ARIA denkt/schreibt/tool" oder "idle" — an Browser
|
||||||
// weiterreichen, damit der Thinking-Indikator im Chat erscheint.
|
// weiterreichen, damit der Thinking-Indikator im Chat erscheint.
|
||||||
@@ -981,7 +1177,7 @@ function connectRVS(forcePlain) {
|
|||||||
}
|
}
|
||||||
broadcast({ type: "voice_ready", payload: msg.payload });
|
broadcast({ type: "voice_ready", payload: msg.payload });
|
||||||
} else if (msg.type === "service_status") {
|
} else if (msg.type === "service_status") {
|
||||||
// Gamebox-Bridges (f5tts/whisper) melden ihren Lade-Status —
|
// AI-Box-Bridges (f5tts/whisper) melden ihren Lade-Status —
|
||||||
// an Browser durchreichen fuer das Banner unten rechts
|
// an Browser durchreichen fuer das Banner unten rechts
|
||||||
const svc = msg.payload?.service || "?";
|
const svc = msg.payload?.service || "?";
|
||||||
const state = msg.payload?.state || "?";
|
const state = msg.payload?.state || "?";
|
||||||
@@ -996,6 +1192,12 @@ function connectRVS(forcePlain) {
|
|||||||
log("info", "rvs", `service_status ${svc} ${state}${model ? ` (${model})` : ""}`);
|
log("info", "rvs", `service_status ${svc} ${state}${model ? ` (${model})` : ""}`);
|
||||||
}
|
}
|
||||||
broadcast({ type: "service_status", payload: msg.payload });
|
broadcast({ type: "service_status", payload: msg.payload });
|
||||||
|
} else if (msg.type === "llm_provision_result") {
|
||||||
|
// Ergebnis eines Modell-Downloads/Aktivierens → an Browser (Katalog-Status).
|
||||||
|
broadcast({ type: "llm_provision_result", payload: msg.payload || {} });
|
||||||
|
} else if (msg.type === "node_stats" || msg.type === "node_stats_history" || msg.type === "node_stats_reset_done") {
|
||||||
|
// Auslastungs-Monitor (Stage E): Box-Antworten an die Browser durchreichen.
|
||||||
|
broadcast({ type: msg.type, payload: msg.payload || {} });
|
||||||
} else if (msg.type === "audio_pcm" && msg.payload && _previewPending.size > 0) {
|
} else if (msg.type === "audio_pcm" && msg.payload && _previewPending.size > 0) {
|
||||||
// PCM-Chunks einer laufenden Voice-Preview — sammeln + WAV bauen
|
// PCM-Chunks einer laufenden Voice-Preview — sammeln + WAV bauen
|
||||||
_handlePreviewChunk(msg.payload);
|
_handlePreviewChunk(msg.payload);
|
||||||
@@ -1040,7 +1242,7 @@ function connectRVS(forcePlain) {
|
|||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
function sendToRVS_withResponse(sendType, sendPayload, expectType, clientWs) {
|
function sendToRVS_withResponse(sendType, sendPayload, expectType, clientWs, timeoutMs = 15000) {
|
||||||
if (!RVS_HOST || !RVS_TOKEN) return;
|
if (!RVS_HOST || !RVS_TOKEN) return;
|
||||||
const proto = RVS_TLS === "true" ? "wss" : "ws";
|
const proto = RVS_TLS === "true" ? "wss" : "ws";
|
||||||
const url = `${proto}://${RVS_HOST}:${RVS_PORT}?token=${RVS_TOKEN}`;
|
const url = `${proto}://${RVS_HOST}:${RVS_PORT}?token=${RVS_TOKEN}`;
|
||||||
@@ -1048,7 +1250,7 @@ function sendToRVS_withResponse(sendType, sendPayload, expectType, clientWs) {
|
|||||||
const timeout = setTimeout(() => {
|
const timeout = setTimeout(() => {
|
||||||
try { freshWs.close(); } catch (_) {}
|
try { freshWs.close(); } catch (_) {}
|
||||||
clientWs.send(JSON.stringify({ type: expectType, payload: { voices: [], error: "Timeout" }, timestamp: Date.now() }));
|
clientWs.send(JSON.stringify({ type: expectType, payload: { voices: [], error: "Timeout" }, timestamp: Date.now() }));
|
||||||
}, 15000);
|
}, timeoutMs);
|
||||||
freshWs.on("open", () => {
|
freshWs.on("open", () => {
|
||||||
freshWs.send(JSON.stringify({ type: sendType, payload: sendPayload, timestamp: Date.now() }));
|
freshWs.send(JSON.stringify({ type: sendType, payload: sendPayload, timestamp: Date.now() }));
|
||||||
});
|
});
|
||||||
@@ -1559,6 +1761,70 @@ function dockerExec(containerName, cmd) {
|
|||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// POST gegen die Docker-Daemon-API (via gemountetem Socket). Fuer prune-
|
||||||
|
// Endpoints — die geben SpaceReclaimed (Bytes) zurueck.
|
||||||
|
function dockerApiPost(apiPath) {
|
||||||
|
return new Promise((resolve, reject) => {
|
||||||
|
const req = http.request({
|
||||||
|
socketPath: "/var/run/docker.sock",
|
||||||
|
path: apiPath,
|
||||||
|
method: "POST",
|
||||||
|
headers: { "Content-Type": "application/json", "Content-Length": 0 },
|
||||||
|
}, (res) => {
|
||||||
|
let data = "";
|
||||||
|
res.on("data", (c) => data += c);
|
||||||
|
res.on("end", () => {
|
||||||
|
if (res.statusCode >= 200 && res.statusCode < 300) {
|
||||||
|
try { resolve(JSON.parse(data || "{}")); } catch { resolve({}); }
|
||||||
|
} else {
|
||||||
|
reject(new Error(`Docker API ${apiPath}: HTTP ${res.statusCode} — ${String(data).slice(0, 200)}`));
|
||||||
|
}
|
||||||
|
});
|
||||||
|
});
|
||||||
|
req.on("error", reject);
|
||||||
|
req.end();
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
// "Sicher aufraeumen": Build-Cache + ungenutzte Images prunen — OHNE Volumes
|
||||||
|
// (keine Daten weg). "aggressive": zusaetzlich gestoppte Container + ungenutzte
|
||||||
|
// Volumes (kann Daten kosten → nur auf ausdrueckliche Wahl). Fuehrt es WIRKLICH
|
||||||
|
// aus (frueher kopierte der Button nur den Befehl in die Zwischenablage).
|
||||||
|
async function handleDiskCleanup(clientWs, variant) {
|
||||||
|
const aggressive = variant === "aggressive";
|
||||||
|
const send = (o) => { try { clientWs.send(JSON.stringify(o)); } catch (_) {} };
|
||||||
|
send({ type: "disk_cleanup", status: "running", variant });
|
||||||
|
log("warn", "server", `Disk-Cleanup gestartet (${aggressive ? "aggressive" : "safe"})`);
|
||||||
|
try {
|
||||||
|
let reclaimed = 0;
|
||||||
|
const steps = [];
|
||||||
|
const bp = await dockerApiPost("/build/prune?all=true");
|
||||||
|
reclaimed += (bp.SpaceReclaimed || 0);
|
||||||
|
steps.push("Build-Cache");
|
||||||
|
// dangling=false → ALLE ungenutzten Images (nicht nur dangling).
|
||||||
|
// Docker-API-Filterformat: map[string][]string.
|
||||||
|
const imgFilter = encodeURIComponent(JSON.stringify({ dangling: ["false"] }));
|
||||||
|
const ip = await dockerApiPost("/images/prune?filters=" + imgFilter);
|
||||||
|
reclaimed += (ip.SpaceReclaimed || 0);
|
||||||
|
steps.push("ungenutzte Images");
|
||||||
|
if (aggressive) {
|
||||||
|
const cp = await dockerApiPost("/containers/prune");
|
||||||
|
reclaimed += (cp.SpaceReclaimed || 0);
|
||||||
|
steps.push("gestoppte Container");
|
||||||
|
const vp = await dockerApiPost("/volumes/prune");
|
||||||
|
reclaimed += (vp.SpaceReclaimed || 0);
|
||||||
|
steps.push("ungenutzte Volumes");
|
||||||
|
}
|
||||||
|
const mb = (reclaimed / (1024 * 1024));
|
||||||
|
const freed = mb >= 1024 ? (mb / 1024).toFixed(2) + " GB" : mb.toFixed(0) + " MB";
|
||||||
|
log("info", "server", `Disk-Cleanup fertig: ${freed} frei (${steps.join(", ")})`);
|
||||||
|
send({ type: "disk_cleanup", status: "done", variant, reclaimedBytes: reclaimed, freed, steps });
|
||||||
|
} catch (err) {
|
||||||
|
log("error", "server", `Disk-Cleanup fehlgeschlagen: ${err.message}`);
|
||||||
|
send({ type: "disk_cleanup", status: "error", variant, error: String(err && err.message || err) });
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// ── Hilfsfunktionen ─────────────────────────────────────
|
// ── Hilfsfunktionen ─────────────────────────────────────
|
||||||
|
|
||||||
function waitForMessage(ws, timeoutMs) {
|
function waitForMessage(ws, timeoutMs) {
|
||||||
@@ -1674,6 +1940,20 @@ const server = http.createServer((req, res) => {
|
|||||||
} else if (req.url === "/api/local-models-list" && req.method === "GET") {
|
} else if (req.url === "/api/local-models-list" && req.method === "GET") {
|
||||||
res.writeHead(200, { "Content-Type": "application/json" });
|
res.writeHead(200, { "Content-Type": "application/json" });
|
||||||
res.end(JSON.stringify({ ok: true, models: loadLocalModels() }));
|
res.end(JSON.stringify({ ok: true, models: loadLocalModels() }));
|
||||||
|
} else if (req.url === "/api/llm-catalog" && req.method === "GET") {
|
||||||
|
res.writeHead(200, { "Content-Type": "application/json" });
|
||||||
|
res.end(JSON.stringify({ ok: true, models: loadLlmCatalog() }));
|
||||||
|
} else if (req.url === "/api/llm-catalog/refresh" && req.method === "POST") {
|
||||||
|
refreshLlmCatalogFromHF()
|
||||||
|
.then(r => {
|
||||||
|
res.writeHead(200, { "Content-Type": "application/json" });
|
||||||
|
res.end(JSON.stringify({ ok: true, models: r.models, added: r.added }));
|
||||||
|
log("info", "llm", `LLM-Katalog von HuggingFace aktualisiert: +${r.added} Modelle`);
|
||||||
|
})
|
||||||
|
.catch(err => {
|
||||||
|
res.writeHead(200, { "Content-Type": "application/json" });
|
||||||
|
res.end(JSON.stringify({ ok: false, error: err.message, models: loadLlmCatalog() }));
|
||||||
|
});
|
||||||
} else if (req.url === "/api/local-llm-config" && req.method === "GET") {
|
} else if (req.url === "/api/local-llm-config" && req.method === "GET") {
|
||||||
res.writeHead(200, { "Content-Type": "application/json" });
|
res.writeHead(200, { "Content-Type": "application/json" });
|
||||||
res.end(JSON.stringify(readLocalLlmConfig()));
|
res.end(JSON.stringify(readLocalLlmConfig()));
|
||||||
@@ -2441,6 +2721,8 @@ wss.on("connection", (ws) => {
|
|||||||
if (currentDiskStatus) ws.send(JSON.stringify(currentDiskStatus));
|
if (currentDiskStatus) ws.send(JSON.stringify(currentDiskStatus));
|
||||||
// Aktuell bekannte Satelliten mitgeben (RVS replayt sat_hello nicht).
|
// Aktuell bekannte Satelliten mitgeben (RVS replayt sat_hello nicht).
|
||||||
ws.send(JSON.stringify({ type: "sat_update", satellites: satelliteList() }));
|
ws.send(JSON.stringify({ type: "sat_update", satellites: satelliteList() }));
|
||||||
|
// Aktuell bekannte Compute-Worker mitgeben (RVS replayt worker_hello nicht).
|
||||||
|
ws.send(JSON.stringify({ type: "worker_update", workers: workerList() }));
|
||||||
|
|
||||||
ws.on("message", (raw) => {
|
ws.on("message", (raw) => {
|
||||||
try {
|
try {
|
||||||
@@ -2455,6 +2737,16 @@ wss.on("connection", (ws) => {
|
|||||||
} else if (msg.action === "test_rvs") {
|
} else if (msg.action === "test_rvs") {
|
||||||
traceStart("RVS", msg.text || "aria lebst du noch?");
|
traceStart("RVS", msg.text || "aria lebst du noch?");
|
||||||
sendToRVS(msg.text || "aria lebst du noch?", true, msg.projectId || "");
|
sendToRVS(msg.text || "aria lebst du noch?", true, msg.projectId || "");
|
||||||
|
} else if (msg.action === "interject") {
|
||||||
|
// Zwischenruf: in den laufenden Turn schieben (kein Abbruch, keine
|
||||||
|
// Queue) → RVS interject → Bridge → Proxy /interject.
|
||||||
|
const t = String(msg.text || "");
|
||||||
|
if (t.trim()) {
|
||||||
|
sendToRVS_raw({ type: "interject", payload: { projectId: msg.projectId || "", text: t }, timestamp: Date.now() });
|
||||||
|
log("info", "server", "Zwischenruf an RVS (project=" + (msg.projectId || "(main)") + "): " + t.slice(0, 60));
|
||||||
|
}
|
||||||
|
} else if (msg.action === "disk_cleanup") {
|
||||||
|
handleDiskCleanup(ws, msg.variant === "aggressive" ? "aggressive" : "safe");
|
||||||
} else if (msg.action === "reconnect_gateway") {
|
} else if (msg.action === "reconnect_gateway") {
|
||||||
connectGateway();
|
connectGateway();
|
||||||
} else if (msg.action === "reconnect_rvs") {
|
} else if (msg.action === "reconnect_rvs") {
|
||||||
@@ -2462,11 +2754,28 @@ wss.on("connection", (ws) => {
|
|||||||
} else if (msg.action === "sat_list") {
|
} else if (msg.action === "sat_list") {
|
||||||
// Browser will die aktuelle Satelliten-Liste.
|
// Browser will die aktuelle Satelliten-Liste.
|
||||||
ws.send(JSON.stringify({ type: "sat_update", satellites: satelliteList() }));
|
ws.send(JSON.stringify({ type: "sat_update", satellites: satelliteList() }));
|
||||||
|
} else if (msg.action === "worker_list") {
|
||||||
|
// Browser will die aktuelle Compute-Flotte.
|
||||||
|
ws.send(JSON.stringify({ type: "worker_update", workers: workerList() }));
|
||||||
} else if (msg.action === "sat_discover") {
|
} else if (msg.action === "sat_discover") {
|
||||||
// Browser triggert einen Geraete-Scan auf einem Satelliten.
|
// Browser triggert einen Geraete-Scan auf einem Satelliten.
|
||||||
sendToRVS_raw({ type: "sat_discover",
|
sendToRVS_raw({ type: "sat_discover",
|
||||||
payload: { satellite: msg.satellite || "", force: true },
|
payload: { satellite: msg.satellite || "", force: true },
|
||||||
timestamp: Date.now() });
|
timestamp: Date.now() });
|
||||||
|
} else if (msg.action === "sat_creds_list") {
|
||||||
|
sendToRVS_raw({ type: "sat_creds_list",
|
||||||
|
payload: { satellite: msg.satellite || "", requestId: "dc_" + Date.now() },
|
||||||
|
timestamp: Date.now() });
|
||||||
|
} else if (msg.action === "sat_creds_set") {
|
||||||
|
sendToRVS_raw({ type: "sat_creds_set",
|
||||||
|
payload: { satellite: msg.satellite || "", ip: msg.ip || "",
|
||||||
|
creds: msg.creds || {}, requestId: "dc_" + Date.now() },
|
||||||
|
timestamp: Date.now() });
|
||||||
|
} else if (msg.action === "sat_creds_delete") {
|
||||||
|
sendToRVS_raw({ type: "sat_creds_delete",
|
||||||
|
payload: { satellite: msg.satellite || "", ip: msg.ip || "",
|
||||||
|
type: msg.credType || "", requestId: "dc_" + Date.now() },
|
||||||
|
timestamp: Date.now() });
|
||||||
} else if (msg.action === "test_proxy") {
|
} else if (msg.action === "test_proxy") {
|
||||||
testProxy(msg.text);
|
testProxy(msg.text);
|
||||||
} else if (msg.action === "check_proxy_auth") {
|
} else if (msg.action === "check_proxy_auth") {
|
||||||
@@ -2494,19 +2803,23 @@ wss.on("connection", (ws) => {
|
|||||||
// Datei von Diagnostic an Bridge via RVS senden
|
// Datei von Diagnostic an Bridge via RVS senden
|
||||||
sendToRVS_raw({
|
sendToRVS_raw({
|
||||||
type: "file",
|
type: "file",
|
||||||
payload: { name: msg.name, type: msg.type, size: msg.size, base64: msg.base64 },
|
payload: { name: msg.name, type: msg.type, size: msg.size, base64: msg.base64, projectId: msg.projectId || "" },
|
||||||
timestamp: Date.now(),
|
timestamp: Date.now(),
|
||||||
});
|
});
|
||||||
log("info", "server", `Datei gesendet: ${msg.name} (${msg.type})`);
|
log("info", "server", `Datei gesendet: ${msg.name} (${msg.type})`);
|
||||||
} else if (msg.action === "cancel_request") {
|
} else if (msg.action === "cancel_request") {
|
||||||
// Laufende Anfrage abbrechen — doctor --fix beendet stuck runs
|
// Laufende Anfrage abbrechen — ECHTER Cancel: RVS cancel_request (hard)
|
||||||
log("warn", "server", "Anfrage abgebrochen — fuehre doctor --fix aus");
|
// an die Bridge, die den Proxy-/cancel-all Side-Channel anruft und den
|
||||||
|
// laufenden claude-Subprozess killt. Das alte `openclaw doctor --fix`
|
||||||
|
// zielte auf den Container aria-core, den es nicht mehr gibt — es
|
||||||
|
// beendete den Run nie (ARIA lief munter weiter).
|
||||||
|
log("warn", "server", "Anfrage abgebrochen — cancel_request (hard) an Bridge/Proxy");
|
||||||
pendingMessageTime = 0;
|
pendingMessageTime = 0;
|
||||||
watchdogWarned = false;
|
watchdogWarned = false;
|
||||||
watchdogFixAttempted = false;
|
watchdogFixAttempted = false;
|
||||||
if (traceActive) traceEnd(false, "Vom Benutzer abgebrochen");
|
if (traceActive) traceEnd(false, "Vom Benutzer abgebrochen");
|
||||||
broadcast({ type: "agent_activity", activity: "idle" });
|
broadcast({ type: "agent_activity", activity: "idle" });
|
||||||
dockerExec("aria-core", "openclaw doctor --fix 2>/dev/null || true").catch(() => {});
|
sendToRVS_raw({ type: "cancel_request", payload: { hard: true, source: "diagnostic-cancel" }, timestamp: Date.now() });
|
||||||
} else if (msg.action === "aria_panic_stop") {
|
} else if (msg.action === "aria_panic_stop") {
|
||||||
// NOT-AUS aus ARIA-Live-View: lokales /api/cancel UND Hard-Kill via
|
// NOT-AUS aus ARIA-Live-View: lokales /api/cancel UND Hard-Kill via
|
||||||
// Bridge (die wiederum den Proxy-Side-Channel /cancel-all anruft).
|
// Bridge (die wiederum den Proxy-Side-Channel /cancel-all anruft).
|
||||||
@@ -2533,9 +2846,11 @@ wss.on("connection", (ws) => {
|
|||||||
// tar.gz (base64) an XTTS-Bridge schicken — die packt aus
|
// tar.gz (base64) an XTTS-Bridge schicken — die packt aus
|
||||||
sendToRVS_withResponse("xtts_import_voice", { name: msg.name, data: msg.data }, "xtts_voice_imported", ws);
|
sendToRVS_withResponse("xtts_import_voice", { name: msg.name, data: msg.data }, "xtts_voice_imported", ws);
|
||||||
} else if (msg.action === "xtts_delete_voice") {
|
} else if (msg.action === "xtts_delete_voice") {
|
||||||
// Weiterleiten an XTTS-Bridge, die antwortet mit neuer Liste
|
// Weiterleiten an alle f5tts-Boxen (Broadcast) + zentrale Kopie loeschen.
|
||||||
|
// (Der eigene Broadcast kommt nicht zu uns zurueck, daher hier direkt.)
|
||||||
sendToRVS_raw({ type: "xtts_delete_voice", payload: { name: msg.name }, timestamp: Date.now() });
|
sendToRVS_raw({ type: "xtts_delete_voice", payload: { name: msg.name }, timestamp: Date.now() });
|
||||||
log("info", "server", `Voice-Delete '${msg.name}' an XTTS-Bridge gesendet`);
|
if (msg.name) deleteCentralVoice(msg.name);
|
||||||
|
log("info", "server", `Voice-Delete '${msg.name}' an f5tts-Boxen + zentral geloescht`);
|
||||||
} else if (msg.action === "delete_chat_message") {
|
} else if (msg.action === "delete_chat_message") {
|
||||||
// Bubble loeschen — Bridge raeumt chat_backup.jsonl + Brain-conversation
|
// Bubble loeschen — Bridge raeumt chat_backup.jsonl + Brain-conversation
|
||||||
// + broadcastet chat_message_deleted via RVS.
|
// + broadcastet chat_message_deleted via RVS.
|
||||||
@@ -2603,6 +2918,12 @@ wss.on("connection", (ws) => {
|
|||||||
const t = parseFloat(msg.voiceIdThreshold);
|
const t = parseFloat(msg.voiceIdThreshold);
|
||||||
if (t >= 0.0 && t <= 1.0) voiceConfig.voiceIdThreshold = t;
|
if (t >= 0.0 && t <= 1.0) voiceConfig.voiceIdThreshold = t;
|
||||||
}
|
}
|
||||||
|
// Speaker-ID Gating an/aus ("nur meine Stimme"). Default aus (fail-open) —
|
||||||
|
// bewusster Schalter. voxtral/whisper-bridge lesen voiceIdEnabled aus dem
|
||||||
|
// config-Broadcast; aus = gar keine Pruefung.
|
||||||
|
if (msg.voiceIdEnabled !== undefined) {
|
||||||
|
voiceConfig.voiceIdEnabled = !!msg.voiceIdEnabled;
|
||||||
|
}
|
||||||
try {
|
try {
|
||||||
fs.mkdirSync("/shared/config", { recursive: true });
|
fs.mkdirSync("/shared/config", { recursive: true });
|
||||||
fs.writeFileSync("/shared/config/voice_config.json", JSON.stringify(voiceConfig, null, 2));
|
fs.writeFileSync("/shared/config/voice_config.json", JSON.stringify(voiceConfig, null, 2));
|
||||||
@@ -2618,6 +2939,31 @@ wss.on("connection", (ws) => {
|
|||||||
// Sessions- und Brain-File-Viewer entfernt — Sessions sind raus, Memory
|
// Sessions- und Brain-File-Viewer entfernt — Sessions sind raus, Memory
|
||||||
// laeuft jetzt komplett ueber die Vector-DB im aria-brain (siehe Gehirn-Tab).
|
// laeuft jetzt komplett ueber die Vector-DB im aria-brain (siehe Gehirn-Tab).
|
||||||
// restart_session kommt weiter rein, weil der Watchdog ihn manchmal triggert.
|
// restart_session kommt weiter rein, weil der Watchdog ihn manchmal triggert.
|
||||||
|
} else if (msg.action === "llm_provision_model") {
|
||||||
|
// Modell auf eine bestimmte LLM-Box laden/aktivieren (Stage D).
|
||||||
|
sendToRVS_raw({ type: "llm_provision_model", payload: {
|
||||||
|
targetInstance: msg.targetInstance || "",
|
||||||
|
key: msg.key, hfRepo: msg.hfRepo, quant: msg.quant, ctx: msg.ctx, ngl: msg.ngl,
|
||||||
|
}, timestamp: Date.now() });
|
||||||
|
log("info", "llm", `provision '${msg.key}' (${msg.hfRepo}) → ${msg.targetInstance || "?"}`);
|
||||||
|
} else if (msg.action === "llm_remove_model") {
|
||||||
|
sendToRVS_raw({ type: "llm_remove_model", payload: {
|
||||||
|
targetInstance: msg.targetInstance || "", key: msg.key }, timestamp: Date.now() });
|
||||||
|
log("info", "llm", `remove '${msg.key}' → ${msg.targetInstance || "?"}`);
|
||||||
|
} else if (msg.action === "llm_test") {
|
||||||
|
// Test-Chat: kurze Nachricht direkt ans lokale LLM (llm_request/llm_response).
|
||||||
|
const reqId = "diagtest_" + Date.now();
|
||||||
|
sendToRVS_withResponse("llm_request", {
|
||||||
|
requestId: reqId,
|
||||||
|
messages: [{ role: "user", content: String(msg.text || "Sag kurz Hallo.") }],
|
||||||
|
max_tokens: 256, temperature: 0.5,
|
||||||
|
model: msg.model || "", targetInstance: msg.targetInstance || "",
|
||||||
|
}, "llm_response", ws, 120000); // 2min: erster Modell-Swap laedt das GGUF kalt (mehrere GB) — 15s reichen dann nicht
|
||||||
|
log("info", "llm", `Test-Chat → ${msg.model || "?"} @ ${msg.targetInstance || "(broadcast)"}`);
|
||||||
|
} else if (msg.action === "node_stats_stream_start" || msg.action === "node_stats_stream_stop"
|
||||||
|
|| msg.action === "node_stats_history_request" || msg.action === "node_stats_reset") {
|
||||||
|
// Auslastungs-Monitor (Stage E): an die Box (targetInstance) durchreichen.
|
||||||
|
sendToRVS_raw({ type: msg.action, payload: { targetInstance: msg.targetInstance || "" }, timestamp: Date.now() });
|
||||||
} else if (msg.action === "restart_session") {
|
} else if (msg.action === "restart_session") {
|
||||||
handleRestartSession(ws);
|
handleRestartSession(ws);
|
||||||
// ── Einstellungen ──
|
// ── Einstellungen ──
|
||||||
|
|||||||
+1
-5
@@ -11,11 +11,7 @@ services:
|
|||||||
npm install -g @anthropic-ai/claude-code claude-max-api-proxy &&
|
npm install -g @anthropic-ai/claude-code claude-max-api-proxy &&
|
||||||
DIST=$$(find /usr/local/lib -path '*/claude-max-api-proxy/dist' -type d | head -1) &&
|
DIST=$$(find /usr/local/lib -path '*/claude-max-api-proxy/dist' -type d | head -1) &&
|
||||||
sed -i 's/startServer({ port })/startServer({ port, host: process.env.HOST || \"127.0.0.1\" })/' $$DIST/server/standalone.js &&
|
sed -i 's/startServer({ port })/startServer({ port, host: process.env.HOST || \"127.0.0.1\" })/' $$DIST/server/standalone.js &&
|
||||||
sed -i 's/\"--no-session-persistence\",/\"--no-session-persistence\",\"--dangerously-skip-permissions\",/' $$DIST/subprocess/manager.js &&
|
cp /proxy-patches/manager.js $$DIST/subprocess/manager.js &&
|
||||||
sed -i 's/\"--dangerously-skip-permissions\",/\"--dangerously-skip-permissions\",\"--system-prompt\",options.systemPrompt,/' $$DIST/subprocess/manager.js &&
|
|
||||||
sed -i 's/const DEFAULT_TIMEOUT = 300000;/const DEFAULT_TIMEOUT = 86400000;/' $$DIST/subprocess/manager.js &&
|
|
||||||
sed -i '/prompt, \\/\\/ Pass prompt as argument/d' $$DIST/subprocess/manager.js &&
|
|
||||||
sed -i 's|this\\.process\\.stdin?\\.end();|this.process.stdin?.end(prompt);|' $$DIST/subprocess/manager.js &&
|
|
||||||
cp /proxy-patches/openai-to-cli.js $$DIST/adapter/openai-to-cli.js &&
|
cp /proxy-patches/openai-to-cli.js $$DIST/adapter/openai-to-cli.js &&
|
||||||
cp /proxy-patches/cli-to-openai.js $$DIST/adapter/cli-to-openai.js &&
|
cp /proxy-patches/cli-to-openai.js $$DIST/adapter/cli-to-openai.js &&
|
||||||
cp /proxy-patches/routes.js $$DIST/server/routes.js &&
|
cp /proxy-patches/routes.js $$DIST/server/routes.js &&
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
# FLUX.1-dev Bildgenerierung — Architektur & Stand
|
# FLUX.1-dev Bildgenerierung — Architektur & Stand
|
||||||
|
|
||||||
Ergaenzung des ARIA-Agent-Stacks um native Text-to-Image-Generierung via
|
Ergaenzung des ARIA-Agent-Stacks um native Text-to-Image-Generierung via
|
||||||
FLUX.1-dev auf der Gamebox. Folgt dem **gleichen Pattern wie f5tts / whisper**:
|
FLUX.1-dev auf der AI-Box. Folgt dem **gleichen Pattern wie f5tts / whisper**:
|
||||||
ein eigener Container auf dem Gaming-PC, der sich selbst per WebSocket zum
|
ein eigener Container auf dem Gaming-PC, der sich selbst per WebSocket zum
|
||||||
RVS verbindet und auf seinen Request-Typ lauscht.
|
RVS verbindet und auf seinen Request-Typ lauscht.
|
||||||
|
|
||||||
@@ -23,7 +23,7 @@ aria-bridge ── send_to_core ──▶ aria-brain
|
|||||||
RVS
|
RVS
|
||||||
│ fanout
|
│ fanout
|
||||||
▼
|
▼
|
||||||
flux-bridge (Gamebox)
|
flux-bridge (AI-Box)
|
||||||
│ FluxPipeline.from_pretrained(...)
|
│ FluxPipeline.from_pretrained(...)
|
||||||
│ pipeline(prompt, width, height, steps, guidance).images[0]
|
│ pipeline(prompt, width, height, steps, guidance).images[0]
|
||||||
│ PIL → PNG → base64
|
│ PIL → PNG → base64
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
# Plan B — Lokaler LLM-Router (Gamebox) neben Claude
|
# Plan B — Lokaler LLM-Router (AI-Box) neben Claude
|
||||||
|
|
||||||
**Ziel:** „Gemini-Feeling" für den Alltag, ohne die Claude-Max-Subscription
|
**Ziel:** „Gemini-Feeling" für den Alltag, ohne die Claude-Max-Subscription
|
||||||
aufzugeben. Ein schnelles lokales LLM beantwortet die einfachen ~80 % der Turns
|
aufzugeben. Ein schnelles lokales LLM beantwortet die einfachen ~80 % der Turns
|
||||||
@@ -11,7 +11,7 @@ Gemessen (10.07.2026): CLI-Round-trip über den Claude-Max-Proxy hat einen
|
|||||||
**harten Boden von ~3,5 s** (Subprozess-Start pro Turn). Streaming-API würde das
|
**harten Boden von ~3,5 s** (Subprozess-Start pro Turn). Streaming-API würde das
|
||||||
brechen, kostet aber API-Geld → verliert die Max-Subscription. Ein lokales
|
brechen, kostet aber API-Geld → verliert die Max-Subscription. Ein lokales
|
||||||
LLM für die einfachen Turns umgeht den 3,5-s-Boden komplett und ist **gratis**
|
LLM für die einfachen Turns umgeht den 3,5-s-Boden komplett und ist **gratis**
|
||||||
(läuft auf vorhandener Gamebox-GPU). Echtes Speech-to-Speech-Duplex (Gemini
|
(läuft auf vorhandener AI-Box-GPU). Echtes Speech-to-Speech-Duplex (Gemini
|
||||||
Live nativ) ist mit einem Text-Modell als Hirn prinzipiell nicht drin.
|
Live nativ) ist mit einem Text-Modell als Hirn prinzipiell nicht drin.
|
||||||
|
|
||||||
## Modell & Serving (entschieden)
|
## Modell & Serving (entschieden)
|
||||||
@@ -19,17 +19,17 @@ Live nativ) ist mit einem Text-Modell als Hirn prinzipiell nicht drin.
|
|||||||
- **Modell:** Qwen3 8B, GGUF **Q4_K_M** (~6 GB). Bestes Tool-Calling der 7/8B-
|
- **Modell:** Qwen3 8B, GGUF **Q4_K_M** (~6 GB). Bestes Tool-Calling der 7/8B-
|
||||||
Klasse, solides Deutsch, Apache-2.0. Alt.: Mistral Small 3 7B (schneller,
|
Klasse, solides Deutsch, Apache-2.0. Alt.: Mistral Small 3 7B (schneller,
|
||||||
weniger Tool-Calling).
|
weniger Tool-Calling).
|
||||||
- **Serving:** **llama.cpp `llama-server`** im Docker-Container auf der Gamebox
|
- **Serving:** **llama.cpp `llama-server`** im Docker-Container auf der AI-Box
|
||||||
(kein Ollama nötig — nativer OpenAI-kompatibler `/v1/chat/completions`).
|
(kein Ollama nötig — nativer OpenAI-kompatibler `/v1/chat/completions`).
|
||||||
- **VRAM-Budget:** 12-GB-Karte, Whisper-small (~1–2 GB) + F5-TTS (~1–2 GB) →
|
- **VRAM-Budget:** 12-GB-Karte, Whisper-small (~1–2 GB) + F5-TTS (~1–2 GB) →
|
||||||
~8–9 GB frei → passt. (FLUX ist auf 12 GB eh raus.)
|
~8–9 GB frei → passt. (FLUX ist auf 12 GB eh raus.)
|
||||||
|
|
||||||
## Anbindung: über den RVS, wie TTS/STT (kein IP-Pflegen)
|
## Anbindung: über den RVS, wie TTS/STT (kein IP-Pflegen)
|
||||||
|
|
||||||
Die Gamebox ist ein anderer Host als das Brain. Statt direktem HTTP (IP/Port/
|
Die AI-Box ist ein anderer Host als das Brain. Statt direktem HTTP (IP/Port/
|
||||||
Firewall) läuft das LLM **über den RVS-Token-Room**, exakt wie Whisper/F5-TTS:
|
Firewall) läuft das LLM **über den RVS-Token-Room**, exakt wie Whisper/F5-TTS:
|
||||||
|
|
||||||
- llama.cpp hört nur auf localhost der Gamebox.
|
- llama.cpp hört nur auf localhost der AI-Box.
|
||||||
- Ein **dünner RVS-Adapter** daneben (Vorbild: whisper-/xtts-Bridge) verbindet
|
- Ein **dünner RVS-Adapter** daneben (Vorbild: whisper-/xtts-Bridge) verbindet
|
||||||
sich mit dem RVS-Token, lauscht auf `llm_request`, ruft lokal llama-server,
|
sich mit dem RVS-Token, lauscht auf `llm_request`, ruft lokal llama-server,
|
||||||
schickt `llm_response` (korreliert per requestId) zurück.
|
schickt `llm_response` (korreliert per requestId) zurück.
|
||||||
@@ -115,7 +115,7 @@ mit Ziel lokal.
|
|||||||
|
|
||||||
## Phasen
|
## Phasen
|
||||||
|
|
||||||
- **B0 — Infra:** llama.cpp-Container + RVS-Adapter auf der Gamebox,
|
- **B0 — Infra:** llama.cpp-Container + RVS-Adapter auf der AI-Box,
|
||||||
`ALLOWED_TYPES`, `local_llm_chat()` im Brain. Isoliert testen („sag hallo").
|
`ALLOWED_TYPES`, `local_llm_chat()` im Brain. Isoliert testen („sag hallo").
|
||||||
- **B1 — Router + lokale Tools:** Heuristik Tier-1/2 + Escalation, schlanke
|
- **B1 — Router + lokale Tools:** Heuristik Tier-1/2 + Escalation, schlanke
|
||||||
Persona lokal, **kuratierte Tool-Auswahl lokal** (Adapter/Bridge/Brain-Tool-
|
Persona lokal, **kuratierte Tool-Auswahl lokal** (Adapter/Bridge/Brain-Tool-
|
||||||
@@ -207,8 +207,8 @@ Zerfaellt in zwei Teile:
|
|||||||
(Docker-Socket) + Controller mit Placement-Policy + Reconciliation +
|
(Docker-Socket) + Controller mit Placement-Policy + Reconciliation +
|
||||||
Broadcast-Kollisions-Vermeidung (nicht 2× dieselbe Faehigkeit). = Mini-Nomad.
|
Broadcast-Kollisions-Vermeidung (nicht 2× dieselbe Faehigkeit). = Mini-Nomad.
|
||||||
|
|
||||||
**Empfehlung:** Fuer 2 Gameboxen NICHT bauen — statische Platzierung reicht
|
**Empfehlung:** Fuer 2 AI-Boxen NICHT bauen — statische Platzierung reicht
|
||||||
(Gamebox1=LLM, Gamebox2=Voice). Dynamisches Laden/Entladen zum VRAM-Freimachen
|
(AI-Box1=LLM, AI-Box2=Voice). Dynamisches Laden/Entladen zum VRAM-Freimachen
|
||||||
deckt `llama-swap` innerhalb eines Hosts (B0.5). Waechst die Flotte: erst den
|
deckt `llama-swap` innerhalb eines Hosts (B0.5). Waechst die Flotte: erst den
|
||||||
billigen Heartbeat-Teil; fuer echte Orchestrierung Docker Swarm / Nomad nehmen
|
billigen Heartbeat-Teil; fuer echte Orchestrierung Docker Swarm / Nomad nehmen
|
||||||
statt selbst einen Scheduler zu bauen.
|
statt selbst einen Scheduler zu bauen.
|
||||||
@@ -227,7 +227,7 @@ lohnt nicht):
|
|||||||
einen Heartbeat via RVS (Host, GPU-Util, VRAM frei/belegt, laufende
|
einen Heartbeat via RVS (Host, GPU-Util, VRAM frei/belegt, laufende
|
||||||
GPU-Container). Diagnostic zeigt pro Host VRAM-Balken + Dienste + „Host X hat
|
GPU-Container). Diagnostic zeigt pro Host VRAM-Balken + Dienste + „Host X hat
|
||||||
N GB frei". Kein Start/Stop, nur Sicht + Hinweis wohin verschiebbar.
|
N GB frei". Kein Start/Stop, nur Sicht + Hinweis wohin verschiebbar.
|
||||||
- **Zukunft (Gamebox3, 4×3060 = 48 GB):** neuer Host, eigenes Profil, `up` →
|
- **Zukunft (AI-Box3, 4×3060 = 48 GB):** neuer Host, eigenes Profil, `up` →
|
||||||
erscheint im Dashboard; grosses lokales LLM oder FLUX-Vollausbau dorthin.
|
erscheint im Dashboard; grosses lokales LLM oder FLUX-Vollausbau dorthin.
|
||||||
Ohne Orchestrator.
|
Ohne Orchestrator.
|
||||||
|
|
||||||
|
|||||||
+1
-1
@@ -1,6 +1,6 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
"""
|
"""
|
||||||
ARIA FLUX-Bridge — laeuft auf der Gamebox (RTX 3060).
|
ARIA FLUX-Bridge — laeuft auf der AI-Box (RTX 3060).
|
||||||
|
|
||||||
Empfaengt flux_request via RVS → FLUX.1-dev/-schnell auf GPU → sendet
|
Empfaengt flux_request via RVS → FLUX.1-dev/-schnell auf GPU → sendet
|
||||||
flux_response mit base64-PNG zurueck an die aria-bridge. Diese speichert
|
flux_response mit base64-PNG zurueck an die aria-bridge. Diese speichert
|
||||||
|
|||||||
@@ -183,9 +183,9 @@ Wichtige Mechanismen:
|
|||||||
- [x] Decimal-zu-Worte fuer TTS (0.1 → null komma eins, mit IP-Schutz-Lookahead)
|
- [x] Decimal-zu-Worte fuer TTS (0.1 → null komma eins, mit IP-Schutz-Lookahead)
|
||||||
- [x] Generic Acronym-Buchstabieren (XTTS → X T T S, USB → U S B, ueber expliziter Liste)
|
- [x] Generic Acronym-Buchstabieren (XTTS → X T T S, USB → U S B, ueber expliziter Liste)
|
||||||
- [x] voice_preload/voice_ready: Stille Mini-Render bei Voice-Wechsel + Toast/Status "bereit"
|
- [x] voice_preload/voice_ready: Stille Mini-Render bei Voice-Wechsel + Toast/Status "bereit"
|
||||||
- [x] Whisper STT auf die Gamebox ausgelagert (faster-whisper CUDA, float16) — neuer aria-whisper-bridge Container
|
- [x] Whisper STT auf die AI-Box ausgelagert (faster-whisper CUDA, float16) — neuer aria-whisper-bridge Container
|
||||||
- [x] aria-bridge: STT primaer remote (Gamebox), Fallback lokal nach 45s Timeout
|
- [x] aria-bridge: STT primaer remote (AI-Box), Fallback lokal nach 45s Timeout
|
||||||
- [x] Whisper-Modell hot-swap auf Gamebox via config-Broadcast aus Diagnostic
|
- [x] Whisper-Modell hot-swap auf AI-Box via config-Broadcast aus Diagnostic
|
||||||
- [x] **F5-TTS ersetzt XTTS komplett** — neuer aria-f5tts-bridge Container, Voice Cloning, satzweises Streaming
|
- [x] **F5-TTS ersetzt XTTS komplett** — neuer aria-f5tts-bridge Container, Voice Cloning, satzweises Streaming
|
||||||
- [x] Voice-Upload mit Whisper-Auto-Transkription — User muss keinen Referenz-Text eintippen
|
- [x] Voice-Upload mit Whisper-Auto-Transkription — User muss keinen Referenz-Text eintippen
|
||||||
- [x] Audio-Pause statt Ducking: Spotify/YouTube pausieren komplett waehrend TTS (TRANSIENT statt MAY_DUCK)
|
- [x] Audio-Pause statt Ducking: Spotify/YouTube pausieren komplett waehrend TTS (TRANSIENT statt MAY_DUCK)
|
||||||
@@ -334,7 +334,7 @@ Skills mit Tool-Use.
|
|||||||
|
|
||||||
- [x] Datei-Manager (Diagnostic + App-Modal): /shared/uploads/ verwalten, Multi-Select + Select-All + Bulk-Download als ZIP + Bulk-Delete
|
- [x] Datei-Manager (Diagnostic + App-Modal): /shared/uploads/ verwalten, Multi-Select + Select-All + Bulk-Download als ZIP + Bulk-Delete
|
||||||
- [x] Wipe-All-Button (Memory + Stimmen + Settings)
|
- [x] Wipe-All-Button (Memory + Stimmen + Settings)
|
||||||
- [x] Voice Export/Import pro Stimme (Diagnostic + XTTS-Bridge auf Gamebox)
|
- [x] Voice Export/Import pro Stimme (Diagnostic + XTTS-Bridge auf AI-Box)
|
||||||
- [x] F5/Whisper-Settings als JSON-Bundle Export/Import
|
- [x] F5/Whisper-Settings als JSON-Bundle Export/Import
|
||||||
- [x] App Chat-Suche umgebaut: Highlight + Next/Prev statt Filter
|
- [x] App Chat-Suche umgebaut: Highlight + Next/Prev statt Filter
|
||||||
- [x] App Pinch-Zoom in Bildern rewriten (Multi-Touch-Race-Bugs)
|
- [x] App Pinch-Zoom in Bildern rewriten (Multi-Touch-Race-Bugs)
|
||||||
@@ -399,7 +399,7 @@ Skills mit Tool-Use.
|
|||||||
### Architektur
|
### Architektur
|
||||||
- [ ] Diagnostic: System-Info Tab (Container-Status, Disk, RAM, CPU)
|
- [ ] Diagnostic: System-Info Tab (Container-Status, Disk, RAM, CPU)
|
||||||
- [ ] RVS Zombie-Connections endgueltig loesen
|
- [ ] RVS Zombie-Connections endgueltig loesen
|
||||||
- [ ] Gamebox: kleine Web-Oberflaeche fuer Credentials/Server-Config oder zentral aus Diagnostic per RVS push
|
- [ ] AI-Box: kleine Web-Oberflaeche fuer Credentials/Server-Config oder zentral aus Diagnostic per RVS push
|
||||||
- [ ] Erste Skills bauen lassen (yt-dlp, pdf-extract, image-resize, etc.) — durch normale Anfragen, ARIA legt sie selbst an
|
- [ ] Erste Skills bauen lassen (yt-dlp, pdf-extract, image-resize, etc.) — durch normale Anfragen, ARIA legt sie selbst an
|
||||||
- [ ] Heartbeat (periodische Selbst-Checks)
|
- [ ] Heartbeat (periodische Selbst-Checks)
|
||||||
- [ ] Lokales LLM als Waechter (Triage vor Claude-Call)
|
- [ ] Lokales LLM als Waechter (Triage vor Claude-Call)
|
||||||
|
|||||||
@@ -0,0 +1,256 @@
|
|||||||
|
/**
|
||||||
|
* Claude Code CLI Subprocess Manager — ARIA-Patch
|
||||||
|
*
|
||||||
|
* Basis: claude-max-api-proxy dist/subprocess/manager.js, plus die bisher per
|
||||||
|
* sed in docker-compose.yml eingespielten Anpassungen (dangerously-skip-
|
||||||
|
* permissions, system-prompt, 24h-Timeout, Prompt via stdin) — hier fest im
|
||||||
|
* File, damit der groessere Zwischenruf-Umbau nicht per sed gefrickelt werden
|
||||||
|
* muss. Wird per `cp` ueber die npm-Version gelegt (siehe docker-compose.yml).
|
||||||
|
*
|
||||||
|
* ZWISCHENRUF (interject): Statt den Prompt als Text zu schreiben und stdin
|
||||||
|
* sofort zu schliessen (--print/text), laeuft claude jetzt im
|
||||||
|
* `--input-format stream-json`-Modus. Der initiale Prompt geht als
|
||||||
|
* stream-json User-Message rein, stdin bleibt OFFEN — so kann waehrend des
|
||||||
|
* laufenden Turns per sendMessage() eine weitere User-Message reingeschoben
|
||||||
|
* werden, die claude an der naechsten Tool-Grenze aufgreift (kein Abbruch).
|
||||||
|
* Bei 'result' (Turn fertig) wird stdin geschlossen, damit claude sauber
|
||||||
|
* beendet und die HTTP-Response (in routes.js an 'close' gebunden) rausgeht.
|
||||||
|
*/
|
||||||
|
import { spawn } from "child_process";
|
||||||
|
import { EventEmitter } from "events";
|
||||||
|
import { isAssistantMessage, isResultMessage, isContentDelta } from "../types/claude-cli.js";
|
||||||
|
const DEFAULT_TIMEOUT = 86400000; // 24h — lange Agent-Loops (Pentests etc.)
|
||||||
|
export class ClaudeSubprocess extends EventEmitter {
|
||||||
|
process = null;
|
||||||
|
buffer = "";
|
||||||
|
timeoutId = null;
|
||||||
|
isKilled = false;
|
||||||
|
_stdinClosed = false;
|
||||||
|
/**
|
||||||
|
* Start the Claude CLI subprocess with the given prompt
|
||||||
|
*/
|
||||||
|
async start(prompt, options) {
|
||||||
|
const args = this.buildArgs(prompt, options);
|
||||||
|
const timeout = options.timeout || DEFAULT_TIMEOUT;
|
||||||
|
return new Promise((resolve, reject) => {
|
||||||
|
try {
|
||||||
|
// Use spawn() for security - no shell interpretation
|
||||||
|
this.process = spawn("claude", args, {
|
||||||
|
cwd: options.cwd || process.cwd(),
|
||||||
|
env: { ...process.env },
|
||||||
|
stdio: ["pipe", "pipe", "pipe"],
|
||||||
|
});
|
||||||
|
// Set timeout
|
||||||
|
this.timeoutId = setTimeout(() => {
|
||||||
|
if (!this.isKilled) {
|
||||||
|
this.isKilled = true;
|
||||||
|
this.process?.kill("SIGTERM");
|
||||||
|
this.emit("error", new Error(`Request timed out after ${timeout}ms`));
|
||||||
|
}
|
||||||
|
}, timeout);
|
||||||
|
// Handle spawn errors (e.g., claude not found)
|
||||||
|
this.process.on("error", (err) => {
|
||||||
|
this.clearTimeout();
|
||||||
|
if (err.message.includes("ENOENT")) {
|
||||||
|
reject(new Error("Claude CLI not found. Install with: npm install -g @anthropic-ai/claude-code"));
|
||||||
|
}
|
||||||
|
else {
|
||||||
|
reject(err);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
// stdin BLEIBT OFFEN: initialen Prompt als stream-json User-
|
||||||
|
// Message schreiben; spaetere Zwischenrufe kommen via
|
||||||
|
// sendMessage(). Geschlossen wird bei 'result' (s. processBuffer).
|
||||||
|
this._writeUserMessage(prompt);
|
||||||
|
// Falls stdin (z.B. EPIPE) frueh stirbt: nicht crashen.
|
||||||
|
this.process.stdin?.on("error", () => {});
|
||||||
|
console.error(`[Subprocess] Process spawned with PID: ${this.process.pid}`);
|
||||||
|
// Parse JSON stream from stdout
|
||||||
|
this.process.stdout?.on("data", (chunk) => {
|
||||||
|
const data = chunk.toString();
|
||||||
|
console.error(`[Subprocess] Received ${data.length} bytes of stdout`);
|
||||||
|
this.buffer += data;
|
||||||
|
this.processBuffer();
|
||||||
|
});
|
||||||
|
// Capture stderr for debugging
|
||||||
|
this.process.stderr?.on("data", (chunk) => {
|
||||||
|
const errorText = chunk.toString().trim();
|
||||||
|
if (errorText) {
|
||||||
|
// Don't emit as error unless it's actually an error
|
||||||
|
// Claude CLI may write debug info to stderr
|
||||||
|
console.error("[Subprocess stderr]:", errorText.slice(0, 200));
|
||||||
|
}
|
||||||
|
});
|
||||||
|
// Handle process close
|
||||||
|
this.process.on("close", (code) => {
|
||||||
|
console.error(`[Subprocess] Process closed with code: ${code}`);
|
||||||
|
this.clearTimeout();
|
||||||
|
// Process any remaining buffer
|
||||||
|
if (this.buffer.trim()) {
|
||||||
|
this.processBuffer();
|
||||||
|
}
|
||||||
|
this.emit("close", code);
|
||||||
|
});
|
||||||
|
// Resolve immediately since we're streaming
|
||||||
|
resolve();
|
||||||
|
}
|
||||||
|
catch (err) {
|
||||||
|
this.clearTimeout();
|
||||||
|
reject(err);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
/**
|
||||||
|
* Build CLI arguments array
|
||||||
|
*/
|
||||||
|
buildArgs(prompt, options) {
|
||||||
|
const args = [
|
||||||
|
"--print", // Non-interactive mode
|
||||||
|
"--output-format",
|
||||||
|
"stream-json", // JSON streaming output
|
||||||
|
"--verbose", // Required for stream-json
|
||||||
|
"--include-partial-messages", // Enable streaming chunks
|
||||||
|
"--input-format",
|
||||||
|
"stream-json", // ARIA: User-Messages via stdin (Zwischenruf)
|
||||||
|
"--model",
|
||||||
|
options.model, // Model alias (opus/sonnet/haiku)
|
||||||
|
"--no-session-persistence", "--dangerously-skip-permissions", "--system-prompt", options.systemPrompt, "--safe-mode",
|
||||||
|
];
|
||||||
|
if (options.sessionId) {
|
||||||
|
args.push("--session-id", options.sessionId);
|
||||||
|
}
|
||||||
|
return args;
|
||||||
|
}
|
||||||
|
/**
|
||||||
|
* Eine User-Message im stream-json-Input-Format an stdin schreiben.
|
||||||
|
* Genutzt fuer den initialen Prompt UND fuer Zwischenrufe (sendMessage).
|
||||||
|
*/
|
||||||
|
_writeUserMessage(text) {
|
||||||
|
const p = this.process;
|
||||||
|
if (!p || !p.stdin || p.stdin.destroyed || this._stdinClosed)
|
||||||
|
return false;
|
||||||
|
try {
|
||||||
|
p.stdin.write(JSON.stringify({ type: "user", message: { role: "user", content: String(text) } }) + "\n");
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
catch (_) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
/**
|
||||||
|
* Zwischenruf: waehrend eines laufenden Turns eine weitere User-Message
|
||||||
|
* reinschieben. claude greift sie an der naechsten Tool-Grenze auf, ohne
|
||||||
|
* den Turn abzubrechen. Kein Effekt, wenn stdin schon geschlossen ist
|
||||||
|
* (Turn praktisch fertig) — dann ist der Zwischenruf schlicht zu spaet.
|
||||||
|
*/
|
||||||
|
sendMessage(text) {
|
||||||
|
return this._writeUserMessage(text);
|
||||||
|
}
|
||||||
|
/**
|
||||||
|
* stdin schliessen → claude beendet den stream-json-Input und exit't.
|
||||||
|
*/
|
||||||
|
_closeStdin() {
|
||||||
|
if (this._stdinClosed)
|
||||||
|
return;
|
||||||
|
this._stdinClosed = true;
|
||||||
|
try {
|
||||||
|
this.process?.stdin?.end();
|
||||||
|
}
|
||||||
|
catch (_) { }
|
||||||
|
}
|
||||||
|
/**
|
||||||
|
* Process the buffer and emit parsed messages
|
||||||
|
*/
|
||||||
|
processBuffer() {
|
||||||
|
const lines = this.buffer.split("\n");
|
||||||
|
this.buffer = lines.pop() || ""; // Keep incomplete line
|
||||||
|
for (const line of lines) {
|
||||||
|
const trimmed = line.trim();
|
||||||
|
if (!trimmed)
|
||||||
|
continue;
|
||||||
|
try {
|
||||||
|
const message = JSON.parse(trimmed);
|
||||||
|
this.emit("message", message);
|
||||||
|
if (isContentDelta(message)) {
|
||||||
|
// Emit content delta for streaming
|
||||||
|
this.emit("content_delta", message);
|
||||||
|
}
|
||||||
|
else if (isAssistantMessage(message)) {
|
||||||
|
this.emit("assistant", message);
|
||||||
|
}
|
||||||
|
else if (isResultMessage(message)) {
|
||||||
|
this.emit("result", message);
|
||||||
|
// Turn fertig → stdin schliessen, sonst wartet claude im
|
||||||
|
// stream-json-Input auf weitere Messages und der Prozess
|
||||||
|
// (und damit die HTTP-Response) haengt fuer immer.
|
||||||
|
this._closeStdin();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
catch {
|
||||||
|
// Non-JSON output, emit as raw
|
||||||
|
this.emit("raw", trimmed);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
/**
|
||||||
|
* Clear the timeout timer
|
||||||
|
*/
|
||||||
|
clearTimeout() {
|
||||||
|
if (this.timeoutId) {
|
||||||
|
clearTimeout(this.timeoutId);
|
||||||
|
this.timeoutId = null;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
/**
|
||||||
|
* Kill the subprocess
|
||||||
|
*/
|
||||||
|
kill(signal = "SIGTERM") {
|
||||||
|
if (!this.isKilled && this.process) {
|
||||||
|
this.isKilled = true;
|
||||||
|
this.clearTimeout();
|
||||||
|
this.process.kill(signal);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
/**
|
||||||
|
* Check if the process is still running
|
||||||
|
*/
|
||||||
|
isRunning() {
|
||||||
|
return this.process !== null && !this.isKilled && this.process.exitCode === null;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
/**
|
||||||
|
* Verify that Claude CLI is installed and accessible
|
||||||
|
*/
|
||||||
|
export async function verifyClaude() {
|
||||||
|
return new Promise((resolve) => {
|
||||||
|
const proc = spawn("claude", ["--version"], { stdio: "pipe" });
|
||||||
|
let output = "";
|
||||||
|
proc.stdout?.on("data", (chunk) => {
|
||||||
|
output += chunk.toString();
|
||||||
|
});
|
||||||
|
proc.on("error", () => {
|
||||||
|
resolve({
|
||||||
|
ok: false,
|
||||||
|
error: "Claude CLI not found. Install with: npm install -g @anthropic-ai/claude-code",
|
||||||
|
});
|
||||||
|
});
|
||||||
|
proc.on("close", (code) => {
|
||||||
|
if (code === 0) {
|
||||||
|
resolve({ ok: true, version: output.trim() });
|
||||||
|
}
|
||||||
|
else {
|
||||||
|
resolve({
|
||||||
|
ok: false,
|
||||||
|
error: "Claude CLI returned non-zero exit code",
|
||||||
|
});
|
||||||
|
}
|
||||||
|
});
|
||||||
|
});
|
||||||
|
}
|
||||||
|
/**
|
||||||
|
* Check if Claude CLI is authenticated
|
||||||
|
*/
|
||||||
|
export async function verifyAuth() {
|
||||||
|
return { ok: true };
|
||||||
|
}
|
||||||
|
//# sourceMappingURL=manager.js.map
|
||||||
@@ -25,6 +25,13 @@ const MODEL_MAP = {
|
|||||||
"opus": "opus",
|
"opus": "opus",
|
||||||
"sonnet": "sonnet",
|
"sonnet": "sonnet",
|
||||||
"haiku": "haiku",
|
"haiku": "haiku",
|
||||||
|
"fable": "fable",
|
||||||
|
"claude-fable-5": "fable",
|
||||||
|
"claude-code-cli/fable": "fable",
|
||||||
|
// Volle aktuelle IDs (falls die App/Diagnostic sie mal direkt setzt)
|
||||||
|
"claude-opus-5": "opus",
|
||||||
|
"claude-sonnet-5": "sonnet",
|
||||||
|
"claude-haiku-4-5": "haiku",
|
||||||
};
|
};
|
||||||
|
|
||||||
export function extractModel(model) {
|
export function extractModel(model) {
|
||||||
|
|||||||
+46
-4
@@ -514,12 +514,18 @@ async function handleNonStreamingResponse(res, subprocess, cliInput, requestId)
|
|||||||
// Datei, greifen die eingebauten Defaults; die Datei wird dann einmalig mit
|
// Datei, greifen die eingebauten Defaults; die Datei wird dann einmalig mit
|
||||||
// diesen Defaults angelegt, damit es was zu editieren gibt.
|
// diesen Defaults angelegt, damit es was zu editieren gibt.
|
||||||
const MODELS_FILE = process.env.ARIA_MODELS_FILE || "/shared/config/models.json";
|
const MODELS_FILE = process.env.ARIA_MODELS_FILE || "/shared/config/models.json";
|
||||||
|
// Tier-Aliase als id (opus/sonnet/haiku/fable) — die CLI loest sie automatisch
|
||||||
|
// auf die AKTUELLE Version des Tiers auf (Stand 2026-07: fable→Fable 5,
|
||||||
|
// opus→Opus 5, sonnet→Sonnet 5, haiku→Haiku 4.5). So bleibt die Liste
|
||||||
|
// versions-robust; die display_name-Texte nur bei Tier-Wechsel anpassen.
|
||||||
const DEFAULT_MODELS = [
|
const DEFAULT_MODELS = [
|
||||||
{ id: "claude-sonnet-4", tier: "sonnet", display_name: "Sonnet (aktuell: Sonnet 5)",
|
{ id: "fable", tier: "fable", display_name: "Fable (aktuell: Fable 5)",
|
||||||
|
description: "Staerkstes Modell — fuer die haertesten Aufgaben (Software-Entwicklung, lange Agent-Laeufe)." },
|
||||||
|
{ id: "opus", tier: "opus", display_name: "Opus (aktuell: Opus 5)",
|
||||||
|
description: "Sehr schlau, schneller als Fable — fuer schwere/lange Aufgaben." },
|
||||||
|
{ id: "sonnet", tier: "sonnet", display_name: "Sonnet (aktuell: Sonnet 5)",
|
||||||
description: "Schnell & gut — Standard fuer den Alltag." },
|
description: "Schnell & gut — Standard fuer den Alltag." },
|
||||||
{ id: "claude-opus-4", tier: "opus", display_name: "Opus (aktuell: Opus 4.8)",
|
{ id: "haiku", tier: "haiku", display_name: "Haiku (aktuell: Haiku 4.5)",
|
||||||
description: "Langsamer, aber am schlausten — fuer schwere/lange Aufgaben." },
|
|
||||||
{ id: "claude-haiku-4", tier: "haiku", display_name: "Haiku (aktuell: Haiku 4.5)",
|
|
||||||
description: "Sehr schnell & guenstig, kleinerer Kontext — fuer einfache Tasks." },
|
description: "Sehr schnell & guenstig, kleinerer Kontext — fuer einfache Tasks." },
|
||||||
];
|
];
|
||||||
|
|
||||||
@@ -621,6 +627,28 @@ function _cancelByProject(projectId) {
|
|||||||
return { killed, requestIds: ids, projectId: pid };
|
return { killed, requestIds: ids, projectId: pid };
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Zwischenruf: schiebt eine User-Message in den/die laufenden Subprozess(e)
|
||||||
|
// eines Kontexts, OHNE sie zu killen. claude greift sie an der naechsten Tool-
|
||||||
|
// Grenze auf (stream-json-Input, s. manager.js). Kein Treffer / stdin schon
|
||||||
|
// zu (Turn quasi fertig) → delivered=0.
|
||||||
|
function _interjectByProject(projectId, text) {
|
||||||
|
const pid = String(projectId || "");
|
||||||
|
const ids = [];
|
||||||
|
let delivered = 0;
|
||||||
|
for (const [id, entry] of Array.from(_activeSubprocesses)) {
|
||||||
|
if (entry.projectId !== pid) continue;
|
||||||
|
try {
|
||||||
|
if (typeof entry.subprocess.sendMessage === "function" && entry.subprocess.sendMessage(text)) {
|
||||||
|
delivered++;
|
||||||
|
ids.push(id);
|
||||||
|
}
|
||||||
|
} catch (e) {
|
||||||
|
console.error("[aria-interject] sendMessage failed for", id, e?.message);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return { delivered, requestIds: ids, projectId: pid };
|
||||||
|
}
|
||||||
|
|
||||||
try {
|
try {
|
||||||
const internalServer = http.createServer((req, res) => {
|
const internalServer = http.createServer((req, res) => {
|
||||||
if (req.method === "POST" && req.url === "/cancel-all") {
|
if (req.method === "POST" && req.url === "/cancel-all") {
|
||||||
@@ -645,6 +673,20 @@ try {
|
|||||||
});
|
});
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
if (req.method === "POST" && req.url === "/interject") {
|
||||||
|
// Body: {projectId, text}. Zwischenruf in den laufenden Turn.
|
||||||
|
let raw = "";
|
||||||
|
req.on("data", (c) => { raw += c; if (raw.length > 65536) req.destroy(); });
|
||||||
|
req.on("end", () => {
|
||||||
|
let projectId = "", text = "";
|
||||||
|
try { const b = JSON.parse(raw || "{}"); projectId = String(b.projectId || ""); text = String(b.text || ""); } catch (_) {}
|
||||||
|
const result = text ? _interjectByProject(projectId, text) : { delivered: 0, requestIds: [], projectId };
|
||||||
|
console.warn("[aria-interject] /interject project=%s — delivered %d", projectId || "(main)", result.delivered);
|
||||||
|
res.writeHead(200, { "Content-Type": "application/json" });
|
||||||
|
res.end(JSON.stringify({ ok: true, ...result }));
|
||||||
|
});
|
||||||
|
return;
|
||||||
|
}
|
||||||
if (req.method === "GET" && req.url === "/health") {
|
if (req.method === "GET" && req.url === "/health") {
|
||||||
res.writeHead(200, { "Content-Type": "application/json" });
|
res.writeHead(200, { "Content-Type": "application/json" });
|
||||||
res.end(JSON.stringify({ ok: true, active: _activeSubprocesses.size }));
|
res.end(JSON.stringify({ ok: true, active: _activeSubprocesses.size }));
|
||||||
|
|||||||
+20
-2
@@ -17,7 +17,7 @@ const ALLOWED_TYPES = new Set([
|
|||||||
"file_request", "file_response", "file_saved", "stt_result", "config", "tts_request",
|
"file_request", "file_response", "file_saved", "stt_result", "config", "tts_request",
|
||||||
"xtts_request", "xtts_response", "xtts_list_voices", "xtts_voices_list", "voice_upload", "xtts_voice_saved",
|
"xtts_request", "xtts_response", "xtts_list_voices", "xtts_voices_list", "voice_upload", "xtts_voice_saved",
|
||||||
"update_check", "update_available", "update_download", "update_data",
|
"update_check", "update_available", "update_download", "update_data",
|
||||||
"agent_activity", "cancel_request",
|
"agent_activity", "cancel_request", "interject",
|
||||||
"audio_pcm",
|
"audio_pcm",
|
||||||
"file_from_aria",
|
"file_from_aria",
|
||||||
"container_restart",
|
"container_restart",
|
||||||
@@ -63,19 +63,37 @@ const ALLOWED_TYPES = new Set([
|
|||||||
"agent_stream",
|
"agent_stream",
|
||||||
"oauth_callback",
|
"oauth_callback",
|
||||||
// Lokales LLM (Plan B) — Router im Brain schickt einfache Turns an das
|
// Lokales LLM (Plan B) — Router im Brain schickt einfache Turns an das
|
||||||
// Qwen3 auf der Gamebox (via Bridge → RVS → llm-adapter → llama.cpp).
|
// Qwen3 auf der AI-Box (via Bridge → RVS → llm-adapter → llama.cpp).
|
||||||
// llm_partial ist fuer B2 (Token-Streaming) reserviert, noch ungenutzt.
|
// llm_partial ist fuer B2 (Token-Streaming) reserviert, noch ungenutzt.
|
||||||
"llm_request", "llm_response", "llm_partial",
|
"llm_request", "llm_response", "llm_partial",
|
||||||
// Workspace-Desktop (Code-Projekte): Live-Code-Editor (CodeMirror in der App)
|
// Workspace-Desktop (Code-Projekte): Live-Code-Editor (CodeMirror in der App)
|
||||||
// spiegelt ARIAs Datei-Writes, und QEMU-VNC wird als RFB-Bytes durch RVS
|
// spiegelt ARIAs Datei-Writes, und QEMU-VNC wird als RFB-Bytes durch RVS
|
||||||
// getunnelt (Base64-in-JSON wie audio_pcm — kein Binaer-Handling noetig).
|
// getunnelt (Base64-in-JSON wie audio_pcm — kein Binaer-Handling noetig).
|
||||||
"code_file", "code_file_edit",
|
"code_file", "code_file_edit",
|
||||||
|
// M1 Generatives Cockpit: ARIA komponiert via present_view eine View-Spec
|
||||||
|
// (Orb + Karten), die App/Web/Diagnostic mit ihrem jeweiligen Renderer
|
||||||
|
// materialisieren. Brain → Bridge → RVS → Clients.
|
||||||
|
"aria_view",
|
||||||
"check_desktop", "desktop_status",
|
"check_desktop", "desktop_status",
|
||||||
"vnc_open", "vnc_close", "vnc_data", "vnc_input",
|
"vnc_open", "vnc_close", "vnc_data", "vnc_input",
|
||||||
// Satelliten (Info-/Gateway-Aussenposten in fremden Netzen): melden sich mit
|
// Satelliten (Info-/Gateway-Aussenposten in fremden Netzen): melden sich mit
|
||||||
// sat_hello, liefern Geraete-Inventar (sat_devices) auf sat_discover und
|
// sat_hello, liefern Geraete-Inventar (sat_devices) auf sat_discover und
|
||||||
// fuehren Aktionen aus (sat_command → sat_result).
|
// fuehren Aktionen aus (sat_command → sat_result).
|
||||||
"sat_hello", "sat_discover", "sat_devices", "sat_command", "sat_result",
|
"sat_hello", "sat_discover", "sat_devices", "sat_command", "sat_result",
|
||||||
|
// Satelliten-Credential-Store: Diagnostic legt pro Geraet Zugangsdaten ab
|
||||||
|
// (SNMP/HTTP/FritzBox), der Satellit speichert sie verschluesselt.
|
||||||
|
"sat_creds_set", "sat_creds_delete", "sat_creds_list",
|
||||||
|
"sat_creds_result", "sat_creds_list_result",
|
||||||
|
// Compute-Flotte (AI-Boxen): Worker (f5tts/whisper/voxtral/llm-adapter) melden
|
||||||
|
// sich per worker_hello an und pingen per worker_ping; der Diagnostic-Server
|
||||||
|
// aggregiert das und broadcastet worker_update/worker_list an die Browser-UI.
|
||||||
|
// node_stats_* speisen den Auslastungs-Monitor (live nvidia-smi + Historie).
|
||||||
|
// OHNE diese Typen verwirft der RVS die Meldungen an der Allow-List (Z. 312),
|
||||||
|
// und die Box bleibt in der Flotte unsichtbar, obwohl sie sendet.
|
||||||
|
"worker_hello", "worker_ping", "worker_update", "worker_list",
|
||||||
|
"node_stats", "node_stats_stream_start", "node_stats_stream_stop",
|
||||||
|
"node_stats_history_request", "node_stats_history",
|
||||||
|
"node_stats_reset", "node_stats_reset_done",
|
||||||
]);
|
]);
|
||||||
|
|
||||||
// Token-Raum: token -> { clients: Set<ws> }
|
// Token-Raum: token -> { clients: Set<ws> }
|
||||||
|
|||||||
+31
-2
@@ -24,9 +24,38 @@ CONTROL_ENABLED=true
|
|||||||
# Erlaubte Steuer-Aktionen (kommagetrennt). Alles andere wird abgelehnt.
|
# Erlaubte Steuer-Aktionen (kommagetrennt). Alles andere wird abgelehnt.
|
||||||
# dial.launch App-Launch via DIAL (z.B. YouTube-Video auf Fire TV / Smart-TV)
|
# dial.launch App-Launch via DIAL (z.B. YouTube-Video auf Fire TV / Smart-TV)
|
||||||
# wol Wake-on-LAN (Geraet per MAC aufwecken)
|
# wol Wake-on-LAN (Geraet per MAC aufwecken)
|
||||||
# http.get generischer HTTP-GET (z.B. lokale IoT-Webhooks)
|
# http.get generischer HTTP-GET (z.B. lokale IoT-Webhooks, Statusseiten)
|
||||||
# http.post generischer HTTP-POST
|
# http.post generischer HTTP-POST
|
||||||
CONTROL_ALLOWLIST=dial.launch,wol,http.get
|
# snmp.get einzelner SNMP-Wert (params: ip, oid)
|
||||||
|
# snmp.walk SNMP-Teilbaum (params: ip, oid)
|
||||||
|
# snmp.printer Drucker-Fuellstaende (Tinte/Toner) aus der Printer-MIB (params: ip)
|
||||||
|
# snmp.ports Switch/Router-Interfaces: aktive/freie Ports (params: ip)
|
||||||
|
# snmp.info Modell/Seriennummer/Firmware-Version (params: ip)
|
||||||
|
# fritzbox.info FritzBox: Verbindung/Datenrate/externe IP (TR-064, braucht Login)
|
||||||
|
# fritzbox.hosts FritzBox: verbundene Geraete (TR-064, braucht Login)
|
||||||
|
CONTROL_ALLOWLIST=dial.launch,wol,http.get,snmp.get,snmp.walk,snmp.printer,snmp.ports,snmp.info,fritzbox.info,fritzbox.hosts
|
||||||
|
|
||||||
|
# ─── Credential-Store (optional) ───────────────────────────────────
|
||||||
|
# Pro Geraet koennen im Diagnostic Zugangsdaten hinterlegt werden (SNMP-Community/
|
||||||
|
# v3, HTTP-Basic, FritzBox-Login). Der Satellit speichert sie VERSCHLUESSELT im
|
||||||
|
# Bind-Volume ./data. Der Schluessel wird beim ersten Start automatisch erzeugt
|
||||||
|
# (./data/creds.key) — oder hier fest vorgeben (Fernet-Key, base64):
|
||||||
|
# CREDS_KEY=
|
||||||
|
|
||||||
|
# ─── SNMP (optional) ───────────────────────────────────────────────
|
||||||
|
# Defaults fuer die snmp.*-Aktionen; pro Request per params ueberschreibbar.
|
||||||
|
SNMP_COMMUNITY=public # Drucker/Switches antworten meist auf 'public'
|
||||||
|
SNMP_VERSION=2c # 1 | 2c
|
||||||
|
SNMP_TIMEOUT_SEC=5
|
||||||
|
# Discovery-Anreicherung: jedes entdeckte Geraet wird beim Scan kurz per SNMP
|
||||||
|
# nach Name/Beschreibung/Standort/Uptime gefragt (Switches, Router, APs, NAS ...).
|
||||||
|
SNMP_DISCOVERY=true
|
||||||
|
SNMP_DISCOVERY_CONCURRENCY=16 # parallele SNMP-Abfragen pro Scan
|
||||||
|
SNMP_DISCOVERY_TIMEOUT=2 # Timeout je Geraet (s) — Nicht-SNMP-Hosts fallen schnell raus
|
||||||
|
|
||||||
|
# ─── HTTP (optional) ───────────────────────────────────────────────
|
||||||
|
HTTP_TIMEOUT_SEC=10 # Timeout fuer http.get/http.post
|
||||||
|
HTTP_MAX_CHARS=20000 # Default-Body-Ausschnitt (offset/max_chars pro Request)
|
||||||
|
|
||||||
# ─── Discovery-Tuning (optional) ───────────────────────────────────
|
# ─── Discovery-Tuning (optional) ───────────────────────────────────
|
||||||
SCAN_INTERVAL_SEC=300 # Hintergrund-Rescan-Intervall
|
SCAN_INTERVAL_SEC=300 # Hintergrund-Rescan-Intervall
|
||||||
|
|||||||
@@ -0,0 +1,3 @@
|
|||||||
|
.env
|
||||||
|
# Verschluesselter Credential-Store + Schluessel (nie einchecken!)
|
||||||
|
data/
|
||||||
@@ -5,6 +5,12 @@ FROM python:3.12-slim
|
|||||||
|
|
||||||
WORKDIR /app
|
WORKDIR /app
|
||||||
|
|
||||||
|
# net-snmp-CLI (snmpget/snmpwalk) fuer die snmp.*-Aktionen — z.B. Drucker-
|
||||||
|
# Tintenstaende zuverlaessig aus der Printer-MIB statt HTML zu scrapen.
|
||||||
|
RUN apt-get update \
|
||||||
|
&& apt-get install -y --no-install-recommends snmp \
|
||||||
|
&& rm -rf /var/lib/apt/lists/*
|
||||||
|
|
||||||
COPY requirements.txt .
|
COPY requirements.txt .
|
||||||
RUN pip install --no-cache-dir -r requirements.txt
|
RUN pip install --no-cache-dir -r requirements.txt
|
||||||
|
|
||||||
|
|||||||
@@ -21,3 +21,6 @@ services:
|
|||||||
network_mode: host
|
network_mode: host
|
||||||
env_file: .env
|
env_file: .env
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
|
volumes:
|
||||||
|
# Persistenter Credential-Store (verschluesselt) + Schluesseldatei.
|
||||||
|
- ./data:/data
|
||||||
|
|||||||
@@ -1,3 +1,4 @@
|
|||||||
websockets>=12.0
|
websockets>=12.0
|
||||||
zeroconf>=0.131.0
|
zeroconf>=0.131.0
|
||||||
requests>=2.31.0
|
requests>=2.31.0
|
||||||
|
cryptography>=42.0 # Verschluesselung des Geraete-Credential-Stores (Fernet)
|
||||||
|
|||||||
+629
-6
@@ -35,6 +35,7 @@ import re
|
|||||||
import socket
|
import socket
|
||||||
import struct
|
import struct
|
||||||
import time
|
import time
|
||||||
|
from pathlib import Path
|
||||||
from typing import Optional
|
from typing import Optional
|
||||||
|
|
||||||
import websockets
|
import websockets
|
||||||
@@ -113,7 +114,9 @@ SATELLITE_LOCATION = (os.environ.get("SATELLITE_LOCATION") or SATELLITE_ID).stri
|
|||||||
CONTROL_ENABLED = _env_bool("CONTROL_ENABLED", False)
|
CONTROL_ENABLED = _env_bool("CONTROL_ENABLED", False)
|
||||||
CONTROL_ALLOWLIST = [
|
CONTROL_ALLOWLIST = [
|
||||||
a.strip() for a in
|
a.strip() for a in
|
||||||
os.environ.get("CONTROL_ALLOWLIST", "dial.launch,wol,http.get").split(",")
|
os.environ.get("CONTROL_ALLOWLIST",
|
||||||
|
"dial.launch,wol,http.get,snmp.get,snmp.walk,snmp.printer,"
|
||||||
|
"snmp.ports,snmp.info,fritzbox.info,fritzbox.hosts").split(",")
|
||||||
if a.strip()
|
if a.strip()
|
||||||
]
|
]
|
||||||
|
|
||||||
@@ -121,6 +124,114 @@ SCAN_INTERVAL_SEC = int(os.environ.get("SCAN_INTERVAL_SEC", "300") or "300")
|
|||||||
DISCOVER_TIMEOUT_SEC = float(os.environ.get("DISCOVER_TIMEOUT_SEC", "6") or "6")
|
DISCOVER_TIMEOUT_SEC = float(os.environ.get("DISCOVER_TIMEOUT_SEC", "6") or "6")
|
||||||
DEVICE_CACHE_TTL_SEC = int(os.environ.get("DEVICE_CACHE_TTL_SEC", "120") or "120")
|
DEVICE_CACHE_TTL_SEC = int(os.environ.get("DEVICE_CACHE_TTL_SEC", "120") or "120")
|
||||||
|
|
||||||
|
# http.get/http.post: Body-Ausschnitt. Default grosszuegig (ganze Statusseiten
|
||||||
|
# passen), mit hartem Deckel gegen Riesen-Payloads durchs RVS. offset/max_chars
|
||||||
|
# pro Request ueberschreibbar; contains-Filter zieht nur relevante Zeilen.
|
||||||
|
HTTP_TIMEOUT_SEC = float(os.environ.get("HTTP_TIMEOUT_SEC", "10") or "10")
|
||||||
|
HTTP_MAX_CHARS = int(os.environ.get("HTTP_MAX_CHARS", "20000") or "20000")
|
||||||
|
HTTP_MAX_CHARS_HARD = int(os.environ.get("HTTP_MAX_CHARS_HARD", "200000") or "200000")
|
||||||
|
|
||||||
|
# SNMP (net-snmp-CLI): Default-Community/Version + Timeout. Drucker antworten
|
||||||
|
# i.d.R. auf community 'public', v2c.
|
||||||
|
SNMP_COMMUNITY = os.environ.get("SNMP_COMMUNITY", "public") or "public"
|
||||||
|
SNMP_VERSION = os.environ.get("SNMP_VERSION", "2c") or "2c"
|
||||||
|
SNMP_TIMEOUT_SEC = float(os.environ.get("SNMP_TIMEOUT_SEC", "5") or "5")
|
||||||
|
# Printer-MIB (RFC 3805) prtMarkerSuppliesEntry-Spalten (numerisch, ohne MIB-Files):
|
||||||
|
SNMP_SUPPLY_DESC = "1.3.6.1.2.1.43.11.1.1.6.1" # Beschreibung (z.B. "Black Ink")
|
||||||
|
SNMP_SUPPLY_MAX = "1.3.6.1.2.1.43.11.1.1.8.1" # Max-Kapazitaet
|
||||||
|
SNMP_SUPPLY_LVL = "1.3.6.1.2.1.43.11.1.1.9.1" # aktueller Fuellstand
|
||||||
|
|
||||||
|
# SNMP-Anreicherung bei der Discovery: jedes entdeckte Geraet mit IP wird kurz
|
||||||
|
# nach seiner System-Group (RFC 1213) gefragt. Switches/Router/APs/NAS geben so
|
||||||
|
# Name, Beschreibung, Standort & Uptime preis -> im Inventar (satellite_devices)
|
||||||
|
# sichtbar. Abschaltbar; kurzer Timeout + parallel, damit der Scan flott bleibt.
|
||||||
|
SNMP_DISCOVERY = _env_bool("SNMP_DISCOVERY", True)
|
||||||
|
SNMP_DISCOVERY_CONCURRENCY = int(os.environ.get("SNMP_DISCOVERY_CONCURRENCY", "16") or "16")
|
||||||
|
SNMP_DISCOVERY_TIMEOUT = float(os.environ.get("SNMP_DISCOVERY_TIMEOUT", "2") or "2")
|
||||||
|
# System-Group (RFC 1213) .0-Instanzen:
|
||||||
|
SNMP_SYS_OIDS = {
|
||||||
|
"descr": "1.3.6.1.2.1.1.1.0", # sysDescr
|
||||||
|
"objectid": "1.3.6.1.2.1.1.2.0", # sysObjectID
|
||||||
|
"uptime": "1.3.6.1.2.1.1.3.0", # sysUpTime
|
||||||
|
"contact": "1.3.6.1.2.1.1.4.0", # sysContact
|
||||||
|
"name": "1.3.6.1.2.1.1.5.0", # sysName
|
||||||
|
"location": "1.3.6.1.2.1.1.6.0", # sysLocation
|
||||||
|
}
|
||||||
|
|
||||||
|
# ─── Geraete-Credential-Store (verschluesselt, pro IP) ─────────────
|
||||||
|
# Diagnostic legt via sat_creds_set pro Geraet Zugangsdaten ab (SNMP-Community/
|
||||||
|
# v3, HTTP-Basic, FritzBox-Login). Der Satellit nutzt sie automatisch bei snmp.*/
|
||||||
|
# http/fritzbox. Persistiert verschluesselt (Fernet) in einem Bind-Volume.
|
||||||
|
CREDS_PATH = os.environ.get("CREDS_PATH", "/data/credentials.json.enc")
|
||||||
|
CREDS_KEY_PATH = os.environ.get("CREDS_KEY_PATH", "/data/creds.key")
|
||||||
|
_CREDS: dict = {} # {ip: {snmp:{...}, http:{...}, fritzbox:{...}}}
|
||||||
|
_creds_fernet = None # Fernet-Instanz (lazy)
|
||||||
|
|
||||||
|
|
||||||
|
def _creds_cipher():
|
||||||
|
"""Fernet-Instanz; Schluessel aus CREDS_KEY (env) oder Schluesseldatei im
|
||||||
|
Volume (wird beim ersten Start erzeugt, 0600)."""
|
||||||
|
global _creds_fernet
|
||||||
|
if _creds_fernet is not None:
|
||||||
|
return _creds_fernet
|
||||||
|
from cryptography.fernet import Fernet
|
||||||
|
key = os.environ.get("CREDS_KEY", "").strip().encode() or None
|
||||||
|
if not key:
|
||||||
|
kp = Path(CREDS_KEY_PATH)
|
||||||
|
if kp.exists():
|
||||||
|
key = kp.read_bytes().strip()
|
||||||
|
else:
|
||||||
|
key = Fernet.generate_key()
|
||||||
|
kp.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
kp.write_bytes(key)
|
||||||
|
try:
|
||||||
|
os.chmod(kp, 0o600)
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
logger.info("[creds] neuer Verschluesselungs-Schluessel erzeugt: %s", CREDS_KEY_PATH)
|
||||||
|
_creds_fernet = Fernet(key)
|
||||||
|
return _creds_fernet
|
||||||
|
|
||||||
|
|
||||||
|
def _creds_load() -> None:
|
||||||
|
global _CREDS
|
||||||
|
p = Path(CREDS_PATH)
|
||||||
|
if not p.exists():
|
||||||
|
_CREDS = {}
|
||||||
|
return
|
||||||
|
try:
|
||||||
|
blob = p.read_bytes()
|
||||||
|
raw = _creds_cipher().decrypt(blob)
|
||||||
|
_CREDS = json.loads(raw.decode("utf-8")) or {}
|
||||||
|
logger.info("[creds] %d Geraete-Eintraege geladen", len(_CREDS))
|
||||||
|
except Exception as exc:
|
||||||
|
logger.warning("[creds] laden fehlgeschlagen (%s) — starte leer", exc)
|
||||||
|
_CREDS = {}
|
||||||
|
|
||||||
|
|
||||||
|
def _creds_save() -> None:
|
||||||
|
p = Path(CREDS_PATH)
|
||||||
|
p.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
blob = _creds_cipher().encrypt(json.dumps(_CREDS).encode("utf-8"))
|
||||||
|
p.write_bytes(blob)
|
||||||
|
try:
|
||||||
|
os.chmod(p, 0o600)
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
def _creds_for(ip: str) -> dict:
|
||||||
|
return _CREDS.get((ip or "").strip(), {}) if ip else {}
|
||||||
|
|
||||||
|
|
||||||
|
def _creds_public_summary() -> list:
|
||||||
|
"""Fuer sat_creds_list: welche Geraete welche Cred-Typen haben — OHNE Secrets."""
|
||||||
|
out = []
|
||||||
|
for ip, entry in sorted(_CREDS.items()):
|
||||||
|
types = [t for t in ("snmp", "http", "fritzbox") if entry.get(t)]
|
||||||
|
out.append({"ip": ip, "types": types})
|
||||||
|
return out
|
||||||
|
|
||||||
HEARTBEAT_SEC = 25
|
HEARTBEAT_SEC = 25
|
||||||
|
|
||||||
# mDNS-Servicetypen, die fuer ARIA interessant sind.
|
# mDNS-Servicetypen, die fuer ARIA interessant sind.
|
||||||
@@ -420,6 +531,16 @@ async def _control(action: str, params: dict, devices: list[dict]) -> dict:
|
|||||||
return await loop.run_in_executor(None, _do_wol, params)
|
return await loop.run_in_executor(None, _do_wol, params)
|
||||||
if action in ("http.get", "http.post"):
|
if action in ("http.get", "http.post"):
|
||||||
return await loop.run_in_executor(None, _do_http, action, params)
|
return await loop.run_in_executor(None, _do_http, action, params)
|
||||||
|
if action in ("snmp.get", "snmp.walk"):
|
||||||
|
return await loop.run_in_executor(None, _do_snmp, action, params)
|
||||||
|
if action == "snmp.printer":
|
||||||
|
return await loop.run_in_executor(None, _do_snmp_printer, params)
|
||||||
|
if action == "snmp.ports":
|
||||||
|
return await loop.run_in_executor(None, _do_snmp_ports, params)
|
||||||
|
if action == "snmp.info":
|
||||||
|
return await loop.run_in_executor(None, _do_snmp_info, params)
|
||||||
|
if action in ("fritzbox.info", "fritzbox.hosts"):
|
||||||
|
return await loop.run_in_executor(None, _do_fritzbox, action, params)
|
||||||
return {"ok": False, "error": f"Aktion '{action}' nicht implementiert."}
|
return {"ok": False, "error": f"Aktion '{action}' nicht implementiert."}
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
return {"ok": False, "error": f"{action} fehlgeschlagen: {exc}"}
|
return {"ok": False, "error": f"{action} fehlgeschlagen: {exc}"}
|
||||||
@@ -472,14 +593,424 @@ def _do_wol(params: dict) -> dict:
|
|||||||
|
|
||||||
|
|
||||||
def _do_http(action: str, params: dict) -> dict:
|
def _do_http(action: str, params: dict) -> dict:
|
||||||
|
"""HTTP-GET/POST vom Satelliten aus (lokale Webhooks, Geraete-Statusseiten …).
|
||||||
|
|
||||||
|
params:
|
||||||
|
url Pflicht (http/https).
|
||||||
|
body/headers optional (POST).
|
||||||
|
offset ab welchem Zeichen der Body zurueckgegeben wird (Default 0).
|
||||||
|
max_chars wie viele Zeichen max. (Default HTTP_MAX_CHARS, hart gedeckelt).
|
||||||
|
contains String oder Liste: nur Zeilen, die (case-insensitive) einen der
|
||||||
|
Begriffe enthalten, werden zurueckgegeben. Ideal um aus einer
|
||||||
|
grossen Statusseite nur die relevanten Werte (z.B. Tinte) zu
|
||||||
|
ziehen, ohne die ganze Seite zu paginieren.
|
||||||
|
Antwort enthaelt total_chars + truncated, damit der Aufrufer weiss, ob noch
|
||||||
|
mehr da ist."""
|
||||||
import requests
|
import requests
|
||||||
url = params.get("url") or ""
|
url = params.get("url") or ""
|
||||||
if not url.startswith(("http://", "https://")):
|
if not url.startswith(("http://", "https://")):
|
||||||
return {"ok": False, "error": "url (http/https) erforderlich."}
|
return {"ok": False, "error": "url (http/https) erforderlich."}
|
||||||
method = "GET" if action == "http.get" else "POST"
|
method = "GET" if action == "http.get" else "POST"
|
||||||
|
# HTTP-Basic-Auth: explizite params > gespeicherte http-Creds fuer den Host.
|
||||||
|
auth = None
|
||||||
|
hcreds = {}
|
||||||
|
try:
|
||||||
|
from urllib.parse import urlparse
|
||||||
|
host = urlparse(url).hostname or ""
|
||||||
|
hcreds = _creds_for(host).get("http", {})
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
user = params.get("user") or hcreds.get("user")
|
||||||
|
pw = params.get("pass") or params.get("password") or hcreds.get("pass")
|
||||||
|
if user or pw: # Benutzer optional — manche Geraete nutzen Password-only-Basic-Auth
|
||||||
|
auth = (str(user or ""), str(pw or ""))
|
||||||
r = requests.request(method, url, data=params.get("body"),
|
r = requests.request(method, url, data=params.get("body"),
|
||||||
headers=params.get("headers"), timeout=6)
|
headers=params.get("headers"), auth=auth,
|
||||||
return {"ok": True, "result": {"status": r.status_code, "body": r.text[:2000]}}
|
timeout=HTTP_TIMEOUT_SEC)
|
||||||
|
text = r.text
|
||||||
|
total = len(text)
|
||||||
|
|
||||||
|
contains = params.get("contains")
|
||||||
|
if contains:
|
||||||
|
terms = [contains] if isinstance(contains, str) else list(contains)
|
||||||
|
terms = [str(t).lower() for t in terms if str(t).strip()]
|
||||||
|
if terms:
|
||||||
|
lines = [ln for ln in text.splitlines()
|
||||||
|
if any(t in ln.lower() for t in terms)]
|
||||||
|
text = "\n".join(lines)
|
||||||
|
|
||||||
|
try:
|
||||||
|
offset = max(0, int(params.get("offset", 0)))
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
offset = 0
|
||||||
|
try:
|
||||||
|
max_chars = int(params.get("max_chars", HTTP_MAX_CHARS))
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
max_chars = HTTP_MAX_CHARS
|
||||||
|
max_chars = max(1, min(max_chars, HTTP_MAX_CHARS_HARD))
|
||||||
|
|
||||||
|
body = text[offset:offset + max_chars]
|
||||||
|
returned_end = offset + len(body)
|
||||||
|
truncated = returned_end < len(text)
|
||||||
|
return {"ok": True, "result": {
|
||||||
|
"status": r.status_code,
|
||||||
|
"body": body,
|
||||||
|
"total_chars": total, # Groesse der Roh-Antwort
|
||||||
|
"filtered": bool(contains), # contains-Filter aktiv?
|
||||||
|
"offset": offset,
|
||||||
|
"returned_chars": len(body),
|
||||||
|
"truncated": truncated, # noch mehr Text nach diesem Ausschnitt?
|
||||||
|
}}
|
||||||
|
|
||||||
|
|
||||||
|
def _snmp_run(args: list, timeout: float) -> tuple:
|
||||||
|
"""Fuehrt ein net-snmp-CLI-Tool aus. Gibt (ok, stdout|fehlertext)."""
|
||||||
|
import subprocess
|
||||||
|
try:
|
||||||
|
r = subprocess.run(args, capture_output=True, text=True, timeout=timeout)
|
||||||
|
except FileNotFoundError:
|
||||||
|
return False, "snmp-Tools fehlen im Container (Paket 'snmp' im Dockerfile)."
|
||||||
|
except subprocess.TimeoutExpired:
|
||||||
|
return False, "SNMP-Timeout — Geraet antwortet nicht (community/version/IP pruefen)."
|
||||||
|
if r.returncode != 0:
|
||||||
|
return False, (r.stderr or r.stdout or "SNMP-Fehler").strip()[:200]
|
||||||
|
return True, r.stdout
|
||||||
|
|
||||||
|
|
||||||
|
def _snmp_base_args(params: dict, ip: str = "") -> list:
|
||||||
|
"""Version/Community bzw. v3-Auth. Prioritaet: explizite params > gespeicherte
|
||||||
|
Creds fuer die IP > globale Defaults. OHNE -t/-r (haengt der Aufrufer an)."""
|
||||||
|
creds = _creds_for(ip).get("snmp", {}) if ip else {}
|
||||||
|
version = str(params.get("version") or creds.get("version") or SNMP_VERSION)
|
||||||
|
if version == "3":
|
||||||
|
v3 = creds.get("v3", {}) or {}
|
||||||
|
user = str(params.get("user") or v3.get("user") or "")
|
||||||
|
level = str(params.get("level") or v3.get("level") or "authPriv")
|
||||||
|
args = ["-v", "3", "-u", user, "-l", level]
|
||||||
|
ap = params.get("authProto") or v3.get("authProto")
|
||||||
|
ak = params.get("authKey") or v3.get("authKey")
|
||||||
|
pp = params.get("privProto") or v3.get("privProto")
|
||||||
|
pk = params.get("privKey") or v3.get("privKey")
|
||||||
|
if ap and ak:
|
||||||
|
args += ["-a", str(ap), "-A", str(ak)]
|
||||||
|
if pp and pk:
|
||||||
|
args += ["-x", str(pp), "-X", str(pk)]
|
||||||
|
return args
|
||||||
|
community = str(params.get("community") or creds.get("community") or SNMP_COMMUNITY)
|
||||||
|
return ["-v", version, "-c", community]
|
||||||
|
|
||||||
|
|
||||||
|
def _snmp_target(params: dict) -> str:
|
||||||
|
return (params.get("ip") or params.get("host") or params.get("device") or "").strip()
|
||||||
|
|
||||||
|
|
||||||
|
def _do_snmp(action: str, params: dict) -> dict:
|
||||||
|
"""Generisches snmp.get / snmp.walk.
|
||||||
|
params: {ip|host, oid, community?='public', version?='2c'}."""
|
||||||
|
ip = _snmp_target(params)
|
||||||
|
if not ip:
|
||||||
|
return {"ok": False, "error": "ip/host erforderlich."}
|
||||||
|
oid = str(params.get("oid") or "").strip()
|
||||||
|
if not oid:
|
||||||
|
return {"ok": False, "error": "oid erforderlich (z.B. 1.3.6.1.2.1.1.5.0 fuer sysName)."}
|
||||||
|
tool = "snmpwalk" if action == "snmp.walk" else "snmpget"
|
||||||
|
# -OQ: OID = Wert, ohne Typannotation; numerische OIDs brauchen keine MIB-Files.
|
||||||
|
args = [tool, "-OQ", *_snmp_base_args(params, ip), "-t", "2", "-r", "1", ip, oid]
|
||||||
|
ok, out = _snmp_run(args, SNMP_TIMEOUT_SEC)
|
||||||
|
if not ok:
|
||||||
|
return {"ok": False, "error": out}
|
||||||
|
lines = [ln.strip() for ln in out.splitlines() if ln.strip()]
|
||||||
|
return {"ok": True, "result": {"ip": ip, "oid": oid, "lines": lines[:200]}}
|
||||||
|
|
||||||
|
|
||||||
|
def _snmp_walk_values(ip: str, base: list, oid: str) -> list:
|
||||||
|
"""snmpwalk -Oqv (nur Werte, in OID-Index-Reihenfolge)."""
|
||||||
|
ok, out = _snmp_run(["snmpwalk", "-Oqv", *base, ip, oid], SNMP_TIMEOUT_SEC)
|
||||||
|
if not ok:
|
||||||
|
return []
|
||||||
|
return [ln.strip().strip('"') for ln in out.splitlines() if ln.strip()]
|
||||||
|
|
||||||
|
|
||||||
|
def _do_snmp_printer(params: dict) -> dict:
|
||||||
|
"""Komfort: liest die Verbrauchsmaterialien (Tinte/Toner) aus der Printer-MIB
|
||||||
|
und rechnet Fuellstaende in Prozent. params: {ip|host, community?, version?}."""
|
||||||
|
ip = _snmp_target(params)
|
||||||
|
if not ip:
|
||||||
|
return {"ok": False, "error": "ip/host erforderlich."}
|
||||||
|
base = [*_snmp_base_args(params, ip), "-t", "2", "-r", "1"]
|
||||||
|
descs = _snmp_walk_values(ip, base, SNMP_SUPPLY_DESC)
|
||||||
|
if not descs:
|
||||||
|
return {"ok": False, "error":
|
||||||
|
"Keine Printer-MIB-Daten (Geraet unterstuetzt kein SNMP, falsche "
|
||||||
|
"community/version, oder es ist kein Drucker)."}
|
||||||
|
lvls = _snmp_walk_values(ip, base, SNMP_SUPPLY_LVL)
|
||||||
|
maxs = _snmp_walk_values(ip, base, SNMP_SUPPLY_MAX)
|
||||||
|
supplies = []
|
||||||
|
for i, name in enumerate(descs):
|
||||||
|
lvl = _to_int(lvls[i]) if i < len(lvls) else None
|
||||||
|
mx = _to_int(maxs[i]) if i < len(maxs) else None
|
||||||
|
percent = None
|
||||||
|
if lvl is not None and mx and mx > 0 and lvl >= 0:
|
||||||
|
percent = round(lvl / mx * 100)
|
||||||
|
elif lvl == -3:
|
||||||
|
percent = "vorhanden (Stand unbekannt)" # RFC: some remaining
|
||||||
|
elif lvl in (-1, -2):
|
||||||
|
percent = "unbekannt"
|
||||||
|
supplies.append({"name": name, "level": lvl, "max": mx, "percent": percent})
|
||||||
|
return {"ok": True, "result": {"ip": ip, "supplies": supplies}}
|
||||||
|
|
||||||
|
|
||||||
|
# ifTable (RFC 1213) Spalten:
|
||||||
|
_IF_DESCR = "1.3.6.1.2.1.2.2.1.2"
|
||||||
|
_IF_TYPE = "1.3.6.1.2.1.2.2.1.3"
|
||||||
|
_IF_SPEED = "1.3.6.1.2.1.2.2.1.5"
|
||||||
|
_IF_ADMIN = "1.3.6.1.2.1.2.2.1.7" # up(1) down(2)
|
||||||
|
_IF_OPER = "1.3.6.1.2.1.2.2.1.8" # up(1) down(2) ...
|
||||||
|
_IF_ALIAS = "1.3.6.1.2.1.31.1.1.1.18" # ifAlias (ifXTable, optional)
|
||||||
|
|
||||||
|
|
||||||
|
def _do_snmp_ports(params: dict) -> dict:
|
||||||
|
"""Interface-Uebersicht eines Switches/Routers: welche Ports sind aktiv (Link),
|
||||||
|
welche frei. params: {ip|host, community?/v3?}. ethernetCsmacd(6)=echte Ports;
|
||||||
|
Loopback/VLAN etc. werden als 'other' markiert, nicht als freier Port gezaehlt."""
|
||||||
|
ip = _snmp_target(params)
|
||||||
|
if not ip:
|
||||||
|
return {"ok": False, "error": "ip/host erforderlich."}
|
||||||
|
base = [*_snmp_base_args(params, ip), "-t", "2", "-r", "1"]
|
||||||
|
descr = _snmp_walk_values(ip, base, _IF_DESCR)
|
||||||
|
if not descr:
|
||||||
|
return {"ok": False, "error":
|
||||||
|
"Keine Interface-Daten (kein SNMP / falsche Credentials / kein Switch)."}
|
||||||
|
types = _snmp_walk_values(ip, base, _IF_TYPE)
|
||||||
|
opers = _snmp_walk_values(ip, base, _IF_OPER)
|
||||||
|
admins = _snmp_walk_values(ip, base, _IF_ADMIN)
|
||||||
|
speeds = _snmp_walk_values(ip, base, _IF_SPEED)
|
||||||
|
aliases = _snmp_walk_values(ip, base, _IF_ALIAS)
|
||||||
|
ports = []
|
||||||
|
up = down_free = disabled = 0
|
||||||
|
for i, name in enumerate(descr):
|
||||||
|
itype = _to_int(types[i]) if i < len(types) else None
|
||||||
|
oper = _to_int(opers[i]) if i < len(opers) else None
|
||||||
|
admin = _to_int(admins[i]) if i < len(admins) else None
|
||||||
|
speed = _to_int(speeds[i]) if i < len(speeds) else None
|
||||||
|
is_eth = (itype == 6) # ethernetCsmacd
|
||||||
|
state = ("up" if oper == 1 else
|
||||||
|
"disabled" if admin == 2 else "down")
|
||||||
|
if is_eth:
|
||||||
|
if state == "up":
|
||||||
|
up += 1
|
||||||
|
elif state == "disabled":
|
||||||
|
disabled += 1
|
||||||
|
else:
|
||||||
|
down_free += 1
|
||||||
|
ports.append({
|
||||||
|
"name": name.strip('"'),
|
||||||
|
"alias": (aliases[i].strip('"') if i < len(aliases) else ""),
|
||||||
|
"physical": is_eth,
|
||||||
|
"state": state,
|
||||||
|
"speedMbps": round(speed / 1_000_000) if speed else None,
|
||||||
|
})
|
||||||
|
return {"ok": True, "result": {
|
||||||
|
"ip": ip,
|
||||||
|
"summary": {"physical_ports": up + down_free + disabled,
|
||||||
|
"up": up, "free": down_free, "disabled": disabled},
|
||||||
|
"ports": ports,
|
||||||
|
}}
|
||||||
|
|
||||||
|
|
||||||
|
# entPhysicalTable (RFC 4133) — Modell/Serie/Firmware:
|
||||||
|
_ENT_MODEL = "1.3.6.1.2.1.47.1.1.1.1.13" # entPhysicalModelName
|
||||||
|
_ENT_SERIAL = "1.3.6.1.2.1.47.1.1.1.1.11" # entPhysicalSerialNum
|
||||||
|
_ENT_SWREV = "1.3.6.1.2.1.47.1.1.1.1.10" # entPhysicalSoftwareRev
|
||||||
|
_ENT_FWREV = "1.3.6.1.2.1.47.1.1.1.1.9" # entPhysicalFirmwareRev
|
||||||
|
|
||||||
|
|
||||||
|
def _do_snmp_info(params: dict) -> dict:
|
||||||
|
"""Geraeteinfo: sysName/sysDescr + (falls vorhanden) Modell, Seriennummer,
|
||||||
|
Firmware-/Software-Version aus der Entity-MIB. Sagt die INSTALLIERTE Version —
|
||||||
|
ob ein Update existiert, weiss SNMP nicht (Hersteller-Sache)."""
|
||||||
|
ip = _snmp_target(params)
|
||||||
|
if not ip:
|
||||||
|
return {"ok": False, "error": "ip/host erforderlich."}
|
||||||
|
sysinfo = _snmp_system(ip, str(params.get("community") or ""), str(params.get("version") or ""))
|
||||||
|
base = [*_snmp_base_args(params, ip), "-t", "2", "-r", "1"]
|
||||||
|
|
||||||
|
def _first(oid):
|
||||||
|
vals = [v for v in _snmp_walk_values(ip, base, oid)
|
||||||
|
if v and "No Such" not in v]
|
||||||
|
return vals[0] if vals else None
|
||||||
|
|
||||||
|
result = {
|
||||||
|
"ip": ip,
|
||||||
|
"name": (sysinfo or {}).get("name"),
|
||||||
|
"descr": (sysinfo or {}).get("descr"),
|
||||||
|
"location": (sysinfo or {}).get("location"),
|
||||||
|
"uptime": (sysinfo or {}).get("uptime"),
|
||||||
|
"model": _first(_ENT_MODEL),
|
||||||
|
"serial": _first(_ENT_SERIAL),
|
||||||
|
"firmware": _first(_ENT_FWREV) or _first(_ENT_SWREV),
|
||||||
|
}
|
||||||
|
if not any(result[k] for k in ("name", "descr", "model", "firmware")):
|
||||||
|
return {"ok": False, "error": "Kein SNMP / keine verwertbaren Infos."}
|
||||||
|
return {"ok": True, "result": result}
|
||||||
|
|
||||||
|
|
||||||
|
# ─── FritzBox (TR-064) ─────────────────────────────────────────────
|
||||||
|
# TR-064 ist SOAP+Digest-Auth — zu fummelig fuer on-the-fly http.post, daher ein
|
||||||
|
# schlanker Reader. Braucht FritzBox-Login (Credential-Store, Typ 'fritzbox').
|
||||||
|
|
||||||
|
def _tr064(ip: str, user: str, pw: str, service: str, control: str,
|
||||||
|
action: str, args: Optional[dict] = None) -> dict:
|
||||||
|
"""Ein TR-064-SOAP-Call. Gibt {ok, fields|error}. fields = alle <NewX>-Tags."""
|
||||||
|
import requests
|
||||||
|
from requests.auth import HTTPDigestAuth
|
||||||
|
body = "".join(f"<{k}>{v}</{k}>" for k, v in (args or {}).items())
|
||||||
|
envelope = (
|
||||||
|
'<?xml version="1.0"?>'
|
||||||
|
'<s:Envelope xmlns:s="http://schemas.xmlsoap.org/soap/envelope/" '
|
||||||
|
's:encodingStyle="http://schemas.xmlsoap.org/soap/encoding/"><s:Body>'
|
||||||
|
f'<u:{action} xmlns:u="{service}">{body}</u:{action}>'
|
||||||
|
'</s:Body></s:Envelope>'
|
||||||
|
)
|
||||||
|
url = f"http://{ip}:49000{control}"
|
||||||
|
try:
|
||||||
|
r = requests.post(url, data=envelope.encode("utf-8"),
|
||||||
|
headers={"Content-Type": 'text/xml; charset="utf-8"',
|
||||||
|
"SOAPAction": f"{service}#{action}"},
|
||||||
|
auth=HTTPDigestAuth(user, pw), timeout=HTTP_TIMEOUT_SEC)
|
||||||
|
except Exception as exc:
|
||||||
|
return {"ok": False, "error": f"TR-064 nicht erreichbar: {exc}"}
|
||||||
|
if r.status_code == 401:
|
||||||
|
return {"ok": False, "error": "TR-064 Auth fehlgeschlagen (FritzBox-Login pruefen)."}
|
||||||
|
if r.status_code != 200:
|
||||||
|
return {"ok": False, "error": f"TR-064 HTTP {r.status_code}"}
|
||||||
|
fields = {m.group(1): m.group(2) for m in
|
||||||
|
re.finditer(r"<(New[^>/]+)>(.*?)</\1>", r.text, re.DOTALL)}
|
||||||
|
return {"ok": True, "fields": fields}
|
||||||
|
|
||||||
|
|
||||||
|
def _do_fritzbox(action: str, params: dict) -> dict:
|
||||||
|
"""fritzbox.info -> Modell/Firmware/Verbindung/externe IP/Datenrate.
|
||||||
|
fritzbox.hosts -> Liste der bekannten Geraete (Name/IP/MAC/aktiv)."""
|
||||||
|
ip = _snmp_target(params)
|
||||||
|
if not ip:
|
||||||
|
return {"ok": False, "error": "ip/host erforderlich."}
|
||||||
|
fb = _creds_for(ip).get("fritzbox", {})
|
||||||
|
user = str(params.get("user") or fb.get("user") or "")
|
||||||
|
pw = str(params.get("pass") or params.get("password") or fb.get("pass") or "")
|
||||||
|
if not pw:
|
||||||
|
return {"ok": False, "error":
|
||||||
|
"Kein FritzBox-Login hinterlegt. In der Geraeteliste Credentials "
|
||||||
|
"(Typ 'fritzbox') fuer diese IP setzen."}
|
||||||
|
|
||||||
|
if action == "fritzbox.hosts":
|
||||||
|
p = _tr064(ip, user, pw, "urn:dslforum-org:service:Hosts:1",
|
||||||
|
"/upnp/control/hosts", "X_AVM-DE_GetHostListPath")
|
||||||
|
if not p.get("ok"):
|
||||||
|
return p
|
||||||
|
path = p["fields"].get("NewX_AVM-DE_HostListPath", "")
|
||||||
|
if not path:
|
||||||
|
return {"ok": False, "error": "FritzBox lieferte keinen Host-Listen-Pfad."}
|
||||||
|
import requests
|
||||||
|
from requests.auth import HTTPDigestAuth
|
||||||
|
try:
|
||||||
|
r = requests.get(f"http://{ip}:49000{path}",
|
||||||
|
auth=HTTPDigestAuth(user, pw), timeout=HTTP_TIMEOUT_SEC)
|
||||||
|
except Exception as exc:
|
||||||
|
return {"ok": False, "error": f"Host-Liste nicht abrufbar: {exc}"}
|
||||||
|
hosts = []
|
||||||
|
for item in re.finditer(r"<Item>(.*?)</Item>", r.text, re.DOTALL):
|
||||||
|
blk = item.group(1)
|
||||||
|
|
||||||
|
def _t(tag):
|
||||||
|
m = re.search(rf"<{tag}>(.*?)</{tag}>", blk, re.DOTALL)
|
||||||
|
return m.group(1) if m else ""
|
||||||
|
hosts.append({"name": _t("HostName"), "ip": _t("IPAddress"),
|
||||||
|
"mac": _t("MACAddress"),
|
||||||
|
"active": _t("Active") in ("1", "true")})
|
||||||
|
return {"ok": True, "result": {"ip": ip, "count": len(hosts), "hosts": hosts}}
|
||||||
|
|
||||||
|
# fritzbox.info (Default): mehrere Services, Teil-Fehler tolerieren.
|
||||||
|
info = {"ip": ip}
|
||||||
|
dev = _tr064(ip, user, pw, "urn:dslforum-org:service:DeviceInfo:1",
|
||||||
|
"/upnp/control/deviceinfo", "GetInfo")
|
||||||
|
if dev.get("ok"):
|
||||||
|
f = dev["fields"]
|
||||||
|
info.update({"model": f.get("NewModelName"), "firmware": f.get("NewSoftwareVersion"),
|
||||||
|
"serial": f.get("NewSerialNumber"), "uptime_s": _to_int(f.get("NewUpTime"))})
|
||||||
|
st = _tr064(ip, user, pw, "urn:dslforum-org:service:WANIPConnection:1",
|
||||||
|
"/upnp/control/wanipconnection1", "GetStatusInfo")
|
||||||
|
if st.get("ok"):
|
||||||
|
info["connection"] = st["fields"].get("NewConnectionStatus")
|
||||||
|
info["connection_uptime_s"] = _to_int(st["fields"].get("NewUptime"))
|
||||||
|
ext = _tr064(ip, user, pw, "urn:dslforum-org:service:WANIPConnection:1",
|
||||||
|
"/upnp/control/wanipconnection1", "GetExternalIPAddress")
|
||||||
|
if ext.get("ok"):
|
||||||
|
info["external_ip"] = ext["fields"].get("NewExternalIPAddress")
|
||||||
|
link = _tr064(ip, user, pw, "urn:dslforum-org:service:WANCommonInterfaceConfig:1",
|
||||||
|
"/upnp/control/wancommonifconfig1", "GetCommonLinkProperties")
|
||||||
|
if link.get("ok"):
|
||||||
|
f = link["fields"]
|
||||||
|
dn = _to_int(f.get("NewLayer1DownstreamMaxBitRate"))
|
||||||
|
upr = _to_int(f.get("NewLayer1UpstreamMaxBitRate"))
|
||||||
|
info["downstream_mbit"] = round(dn / 1_000_000, 1) if dn else None
|
||||||
|
info["upstream_mbit"] = round(upr / 1_000_000, 1) if upr else None
|
||||||
|
info["physical_link"] = f.get("NewPhysicalLinkStatus")
|
||||||
|
if len(info) == 1:
|
||||||
|
return {"ok": False, "error":
|
||||||
|
"FritzBox antwortet nicht auf TR-064 (Login/Rechte pruefen; TR-064 in "
|
||||||
|
"der FritzBox unter Heimnetz > Netzwerkeinstellungen aktivieren)."}
|
||||||
|
return {"ok": True, "result": info}
|
||||||
|
|
||||||
|
|
||||||
|
def _to_int(s: str):
|
||||||
|
try:
|
||||||
|
return int(str(s).strip())
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _snmp_system(ip: str, community: str = "", version: str = "") -> Optional[dict]:
|
||||||
|
"""Fragt die SNMP-System-Group eines Hosts ab (ein snmpget, alle 6 OIDs).
|
||||||
|
Gibt {descr,name,contact,location,uptime,objectid} oder None (kein SNMP).
|
||||||
|
Nutzt gespeicherte Creds fuer die IP; kurzer Timeout, keine Retries ->
|
||||||
|
Nicht-SNMP-Hosts scheitern schnell."""
|
||||||
|
params = {}
|
||||||
|
if community:
|
||||||
|
params["community"] = community
|
||||||
|
if version:
|
||||||
|
params["version"] = version
|
||||||
|
base = [*_snmp_base_args(params, ip), "-t", "1", "-r", "0"]
|
||||||
|
keys = list(SNMP_SYS_OIDS.keys())
|
||||||
|
ok, out = _snmp_run(["snmpget", "-Oqv", *base, ip, *SNMP_SYS_OIDS.values()],
|
||||||
|
SNMP_DISCOVERY_TIMEOUT)
|
||||||
|
if not ok:
|
||||||
|
return None
|
||||||
|
vals = out.splitlines()
|
||||||
|
info = {}
|
||||||
|
for k, v in zip(keys, vals):
|
||||||
|
v = (v or "").strip().strip('"')
|
||||||
|
if v and "No Such" not in v and "No more" not in v:
|
||||||
|
info[k] = v
|
||||||
|
return info or None
|
||||||
|
|
||||||
|
|
||||||
|
def _snmp_kind(descr: str) -> str:
|
||||||
|
"""Grobe Geraeteklasse aus sysDescr (fuer type im Inventar)."""
|
||||||
|
d = (descr or "").lower()
|
||||||
|
if any(k in d for k in ("switch", "catalyst", "procurve", "aruba", "powerconnect")):
|
||||||
|
return "switch"
|
||||||
|
if any(k in d for k in ("router", "mikrotik", "routeros", "edgeos", "openwrt", "pfsense", "fritz!box")):
|
||||||
|
return "router"
|
||||||
|
if any(k in d for k in ("access point", "accesspoint", "unifi", "wifi", "wlan")):
|
||||||
|
return "access-point"
|
||||||
|
if any(k in d for k in ("printer", "laserjet", "officejet", "brother", "epson", "kyocera")):
|
||||||
|
return "printer"
|
||||||
|
if any(k in d for k in ("nas", "synology", "qnap", "truenas", "diskstation")):
|
||||||
|
return "nas"
|
||||||
|
if any(k in d for k in ("ups", "usv", "smart-ups")):
|
||||||
|
return "ups"
|
||||||
|
return ""
|
||||||
|
|
||||||
|
|
||||||
# ─── Helpers ────────────────────────────────────────────────────────
|
# ─── Helpers ────────────────────────────────────────────────────────
|
||||||
@@ -531,13 +1062,42 @@ class Satellite:
|
|||||||
ssdp = await loop.run_in_executor(None, _discover_ssdp, DISCOVER_TIMEOUT_SEC)
|
ssdp = await loop.run_in_executor(None, _discover_ssdp, DISCOVER_TIMEOUT_SEC)
|
||||||
arp = await loop.run_in_executor(None, _discover_arp)
|
arp = await loop.run_in_executor(None, _discover_arp)
|
||||||
self._devices = _merge_devices(mdns, ssdp, arp)
|
self._devices = _merge_devices(mdns, ssdp, arp)
|
||||||
|
if SNMP_DISCOVERY:
|
||||||
|
await self._enrich_snmp(self._devices)
|
||||||
self._devices_ts = time.time()
|
self._devices_ts = time.time()
|
||||||
logger.info("[scan] %d Geraete (mdns=%d ssdp=%d arp=%d)",
|
n_snmp = sum(1 for d in self._devices if d.get("snmpCapable"))
|
||||||
len(self._devices), len(mdns), len(ssdp), len(arp))
|
logger.info("[scan] %d Geraete (mdns=%d ssdp=%d arp=%d, snmp=%d)",
|
||||||
|
len(self._devices), len(mdns), len(ssdp), len(arp), n_snmp)
|
||||||
finally:
|
finally:
|
||||||
self._scanning = False
|
self._scanning = False
|
||||||
return self._devices
|
return self._devices
|
||||||
|
|
||||||
|
async def _enrich_snmp(self, devices: list) -> None:
|
||||||
|
"""Fragt jedes Geraet mit IP parallel per SNMP-System-Group ab und haengt
|
||||||
|
die Infos an. Verbessert Name/Typ, wenn bisher nur eine IP bekannt war."""
|
||||||
|
loop = asyncio.get_event_loop()
|
||||||
|
sem = asyncio.Semaphore(SNMP_DISCOVERY_CONCURRENCY)
|
||||||
|
|
||||||
|
async def _one(dev: dict) -> None:
|
||||||
|
ip = dev.get("ip") or ""
|
||||||
|
if not ip:
|
||||||
|
return
|
||||||
|
async with sem:
|
||||||
|
info = await loop.run_in_executor(None, _snmp_system, ip)
|
||||||
|
if not info:
|
||||||
|
return
|
||||||
|
dev["snmp"] = info
|
||||||
|
dev["snmpCapable"] = True
|
||||||
|
# Name aufwerten, wenn er bisher nur die IP/leer war.
|
||||||
|
if info.get("name") and dev.get("name", "") in ("", ip):
|
||||||
|
dev["name"] = info["name"]
|
||||||
|
# Typ aufwerten, wenn bisher generisch (host/leer).
|
||||||
|
kind = _snmp_kind(info.get("descr", ""))
|
||||||
|
if kind and dev.get("type", "") in ("", "host"):
|
||||||
|
dev["type"] = kind
|
||||||
|
|
||||||
|
await asyncio.gather(*(_one(d) for d in devices))
|
||||||
|
|
||||||
async def _send(self, message: dict) -> None:
|
async def _send(self, message: dict) -> None:
|
||||||
if self.ws is None:
|
if self.ws is None:
|
||||||
return
|
return
|
||||||
@@ -597,7 +1157,12 @@ class Satellite:
|
|||||||
params = payload.get("params") or {}
|
params = payload.get("params") or {}
|
||||||
if payload.get("device") and "device" not in params:
|
if payload.get("device") and "device" not in params:
|
||||||
params["device"] = payload.get("device")
|
params["device"] = payload.get("device")
|
||||||
devices = await self._scan()
|
# Geraeteliste nur scannen, wenn die Aktion sie wirklich braucht
|
||||||
|
# (dial.launch loest ein Geraet auf, oder es wurde ein device-Ref
|
||||||
|
# mitgegeben). http.get/http.post/wol arbeiten direkt mit url/mac —
|
||||||
|
# ein voller LAN-Scan davor kostete nur unnoetig viele Sekunden.
|
||||||
|
needs_devices = action == "dial.launch" or bool(params.get("device"))
|
||||||
|
devices = await self._scan() if needs_devices else self._devices
|
||||||
result = await _control(action, params, devices)
|
result = await _control(action, params, devices)
|
||||||
await self._send({
|
await self._send({
|
||||||
"type": "sat_result",
|
"type": "sat_result",
|
||||||
@@ -605,6 +1170,63 @@ class Satellite:
|
|||||||
"timestamp": int(time.time() * 1000),
|
"timestamp": int(time.time() * 1000),
|
||||||
})
|
})
|
||||||
|
|
||||||
|
elif mtype == "sat_creds_set":
|
||||||
|
if not self._for_me(payload):
|
||||||
|
return
|
||||||
|
ip = (payload.get("ip") or "").strip()
|
||||||
|
creds = payload.get("creds") or {}
|
||||||
|
ok = False
|
||||||
|
if ip and isinstance(creds, dict):
|
||||||
|
entry = _CREDS.setdefault(ip, {})
|
||||||
|
for t in ("snmp", "http", "fritzbox"):
|
||||||
|
if t in creds:
|
||||||
|
if creds[t]: # leeres Objekt = Typ loeschen
|
||||||
|
entry[t] = creds[t]
|
||||||
|
else:
|
||||||
|
entry.pop(t, None)
|
||||||
|
if not entry:
|
||||||
|
_CREDS.pop(ip, None)
|
||||||
|
try:
|
||||||
|
_creds_save()
|
||||||
|
ok = True
|
||||||
|
except Exception as exc:
|
||||||
|
logger.warning("[creds] speichern fehlgeschlagen: %s", exc)
|
||||||
|
await self._send({"type": "sat_creds_result",
|
||||||
|
"payload": {"requestId": payload.get("requestId", ""),
|
||||||
|
"satellite": SATELLITE_ID, "ok": ok, "ip": ip},
|
||||||
|
"timestamp": int(time.time() * 1000)})
|
||||||
|
|
||||||
|
elif mtype == "sat_creds_delete":
|
||||||
|
if not self._for_me(payload):
|
||||||
|
return
|
||||||
|
ip = (payload.get("ip") or "").strip()
|
||||||
|
ctype = (payload.get("type") or "").strip()
|
||||||
|
if ip in _CREDS:
|
||||||
|
if ctype:
|
||||||
|
_CREDS[ip].pop(ctype, None)
|
||||||
|
if not _CREDS[ip]:
|
||||||
|
_CREDS.pop(ip, None)
|
||||||
|
else:
|
||||||
|
_CREDS.pop(ip, None)
|
||||||
|
try:
|
||||||
|
_creds_save()
|
||||||
|
except Exception as exc:
|
||||||
|
logger.warning("[creds] speichern fehlgeschlagen: %s", exc)
|
||||||
|
await self._send({"type": "sat_creds_result",
|
||||||
|
"payload": {"requestId": payload.get("requestId", ""),
|
||||||
|
"satellite": SATELLITE_ID, "ok": True, "ip": ip},
|
||||||
|
"timestamp": int(time.time() * 1000)})
|
||||||
|
|
||||||
|
elif mtype == "sat_creds_list":
|
||||||
|
if not self._for_me(payload):
|
||||||
|
return
|
||||||
|
await self._send({"type": "sat_creds_list_result",
|
||||||
|
"payload": {"requestId": payload.get("requestId", ""),
|
||||||
|
"satellite": SATELLITE_ID,
|
||||||
|
"location": SATELLITE_LOCATION,
|
||||||
|
"items": _creds_public_summary()},
|
||||||
|
"timestamp": int(time.time() * 1000)})
|
||||||
|
|
||||||
async def _periodic_scan(self) -> None:
|
async def _periodic_scan(self) -> None:
|
||||||
while True:
|
while True:
|
||||||
try:
|
try:
|
||||||
@@ -626,6 +1248,7 @@ class Satellite:
|
|||||||
if not RVS_HOST or not RVS_TOKEN:
|
if not RVS_HOST or not RVS_TOKEN:
|
||||||
logger.error("RVS_HOST und RVS_TOKEN sind Pflicht (siehe .env.example).")
|
logger.error("RVS_HOST und RVS_TOKEN sind Pflicht (siehe .env.example).")
|
||||||
return
|
return
|
||||||
|
_creds_load()
|
||||||
asyncio.create_task(self._periodic_scan())
|
asyncio.create_task(self._periodic_scan())
|
||||||
backoff = 1
|
backoff = 1
|
||||||
while True:
|
while True:
|
||||||
|
|||||||
+43
-3
@@ -1,11 +1,51 @@
|
|||||||
# ════════════════════════════════════════════════
|
# ════════════════════════════════════════════════
|
||||||
# ARIA XTTS v2 — Konfiguration
|
# ARIA Compute-Node — Konfiguration
|
||||||
# Kopieren nach .env und anpassen
|
# Kopieren nach .env und anpassen (pro Worker-Node eine eigene .env)
|
||||||
# ════════════════════════════════════════════════
|
# ════════════════════════════════════════════════
|
||||||
|
|
||||||
# RVS Verbindung (gleiche Daten wie auf der ARIA-VM)
|
# ─── Welche Dienste startet DIESER Node? ──────────
|
||||||
|
# Komma-getrennt aus: voxtral, whisper, f5tts, llm
|
||||||
|
# Nur die aufgefuehrten Dienste starten bei `docker compose up -d`.
|
||||||
|
# nur STT-Box (Voxtral) → COMPOSE_PROFILES=voxtral
|
||||||
|
# nur STT-Box (Whisper) → COMPOSE_PROFILES=whisper
|
||||||
|
# nur TTS-Box → COMPOSE_PROFILES=f5tts
|
||||||
|
# nur LLM-Box → COMPOSE_PROFILES=llm
|
||||||
|
# Kleine Karte (wenig VRAM): Whisper + F5-TTS auf EINER GPU
|
||||||
|
# → COMPOSE_PROFILES=whisper,f5tts
|
||||||
|
# All-in-One, grosse Karte → COMPOSE_PROFILES=voxtral,f5tts,llm
|
||||||
|
# Hinweis: voxtral UND whisper zusammen NICHT sinnvoll — beide sind STT und
|
||||||
|
# wuerden dieselbe Anfrage doppelt beantworten. Pro Node genau EINEN STT waehlen:
|
||||||
|
# Voxtral-3B (~9 GB, beste Qualitaet) ODER Whisper (klein, passt neben F5-TTS).
|
||||||
|
COMPOSE_PROFILES=whisper,f5tts
|
||||||
|
|
||||||
|
# ─── Node-Name ────────────────────────────────────
|
||||||
|
# Freier Name dieses Rechners. Erscheint in Diagnostic (Flotten-Anzeige) und
|
||||||
|
# in den Logs, und bildet die Instanz-ID der Dienste (z.B. f5tts@ai-box).
|
||||||
|
NODE_NAME=ai-box
|
||||||
|
|
||||||
|
# ─── GPU-Zuordnung pro Dienst ─────────────────────
|
||||||
|
# Setzt NVIDIA_VISIBLE_DEVICES fuer den jeweiligen Container.
|
||||||
|
# Einzelne Karte → "0" oder "1"; mehrere Karten → "0,1".
|
||||||
|
# Braucht das NVIDIA Container Toolkit (registriert die `nvidia`-Runtime).
|
||||||
|
# Nur die GPUs der aktiven Profile (COMPOSE_PROFILES) sind ueberhaupt relevant.
|
||||||
|
WHISPER_GPU=0 # kleines STT → passt neben F5-TTS auf EINE Karte
|
||||||
|
F5TTS_GPU=0 # TTS ist klein → gleiche Karte wie Whisper
|
||||||
|
VOXTRAL_GPU=1 # STT-3B ~9 GB → nur auf einer groesseren Karte (z.B. 12 GB)
|
||||||
|
LLM_GPU=0 # lokales LLM (teilt sich ggf. die Karte mit F5-TTS)
|
||||||
|
|
||||||
|
# ─── RVS-Verbindung ───────────────────────────────────────
|
||||||
|
# WICHTIG: Host, Port UND Token muessen EXAKT mit dem ARIA-Stack (Bridge/
|
||||||
|
# Diagnostic) uebereinstimmen — das RVS gruppiert Clients pro Server+Token in
|
||||||
|
# einen Raum. Bei abweichendem Port/Host landet die Box in einem ANDEREN Raum
|
||||||
|
# und taucht nicht in der Compute-Flotte auf.
|
||||||
RVS_HOST=mobil.hacker-net.de
|
RVS_HOST=mobil.hacker-net.de
|
||||||
RVS_PORT=444
|
RVS_PORT=444
|
||||||
RVS_TLS=true
|
RVS_TLS=true
|
||||||
RVS_TLS_FALLBACK=true
|
RVS_TLS_FALLBACK=true
|
||||||
RVS_TOKEN=dein_token_hier
|
RVS_TOKEN=dein_token_hier
|
||||||
|
|
||||||
|
# ─── Optional ─────────────────────────────────────
|
||||||
|
# HF_TOKEN= # nur falls ein HF-gated Modell (z.B. Voxtral) geladen wird
|
||||||
|
# WHISPER_MODEL=small # tiny|base|small|medium|large-v3 (Hot-Swap via Diagnostic)
|
||||||
|
# WHISPER_LANGUAGE=de
|
||||||
|
# LLM_MODEL=qwen3-8b # Key aus llama-swap/config.yaml
|
||||||
|
|||||||
@@ -0,0 +1,68 @@
|
|||||||
|
# xtts — AI-Box (Compute-Node) Setup & Bootstrap
|
||||||
|
|
||||||
|
Der `xtts`-Stack sind ARIAs GPU-Dienste (STT: Voxtral/Whisper · TTS: F5-TTS ·
|
||||||
|
lokales LLM). Er läuft auf einer oder mehreren **AI-Boxen** — jede per `.env`
|
||||||
|
konfiguriert (`COMPOSE_PROFILES`, GPU-Zuordnung). Details zu Profilen/GPU-Wahl
|
||||||
|
stehen im Haupt-README (Abschnitt „Compute-Nodes").
|
||||||
|
|
||||||
|
`bootstrap.sh` macht aus einem frisch installierten **Debian Trixie** (headless,
|
||||||
|
nur SSH) einen startklaren GPU-Host für diesen Stack.
|
||||||
|
|
||||||
|
## Was das Script tut
|
||||||
|
|
||||||
|
1. Basis-Pakete (curl, gnupg, git …)
|
||||||
|
2. `contrib non-free non-free-firmware` aktivieren (Trixie-**deb822**-Format berücksichtigt)
|
||||||
|
3. **NVIDIA-Treiber** installieren (`nvidia-driver` + firmware)
|
||||||
|
4. **Docker** Engine + Compose-Plugin
|
||||||
|
5. **NVIDIA Container Toolkit** + Docker-Runtime auf NVIDIA konfigurieren
|
||||||
|
6. `xtts/.env` aus `.env.example` anlegen (RVS_TOKEN optional gleich setzen)
|
||||||
|
7. **GPU-im-Container-Test** (`docker run --gpus all … nvidia-smi`)
|
||||||
|
8. optional (`--up`): den `xtts`-Stack hochziehen
|
||||||
|
|
||||||
|
Alles **idempotent** — mehrfach ausführbar.
|
||||||
|
|
||||||
|
## Ablauf
|
||||||
|
|
||||||
|
```bash
|
||||||
|
git clone <repo> ARIA-AGENT
|
||||||
|
cd ARIA-AGENT/xtts
|
||||||
|
|
||||||
|
sudo ./bootstrap.sh
|
||||||
|
# → Wenn der Treiber frisch installiert wurde: einmal neu starten, dann nochmal:
|
||||||
|
sudo reboot
|
||||||
|
# … nach dem Boot:
|
||||||
|
cd ARIA-AGENT/xtts
|
||||||
|
sudo ./bootstrap.sh --up --token <DEIN_RVS_TOKEN>
|
||||||
|
```
|
||||||
|
|
||||||
|
Der Treiber-Reboot ist normal (Kernel-Modul wird erst beim Boot geladen). Beim
|
||||||
|
zweiten Lauf überspringt das Script alles Erledigte und macht nur noch den
|
||||||
|
GPU-Test + Stack-Start.
|
||||||
|
|
||||||
|
## Optionen
|
||||||
|
|
||||||
|
| Option | Wirkung |
|
||||||
|
|---|---|
|
||||||
|
| `--up` | am Ende `docker compose up -d --build` (Default-Profil) |
|
||||||
|
| `--token <TOK>` | `RVS_TOKEN` in `xtts/.env` eintragen (auch via `RVS_TOKEN=…` env) |
|
||||||
|
| `--rvs-host <H>` | `RVS_HOST` setzen |
|
||||||
|
|
||||||
|
## Wichtig
|
||||||
|
|
||||||
|
- **Stimm-Daten** (nicht in git): falls von der alten Box noch vorhanden,
|
||||||
|
`xtts/voice-id/` (Speaker-Fingerprint) + `xtts/voices/` (F5-Referenz) herkopieren.
|
||||||
|
Sonst egal — in der App neu anlegen: Stimme neu enrollen (ohne Fingerprint läuft
|
||||||
|
die Speaker-ID fail-open, alles geht durch) + F5-Voice-Referenz neu hochladen.
|
||||||
|
- **Voxtral bleibt aus** auf der 3060 (braucht ≥16 GB VRAM). Das Default-Profil
|
||||||
|
fährt Whisper (mit dem M0.1-Fix) + F5-TTS + lokales LLM. Voxtral erst mit der
|
||||||
|
24-GB-Karte: `docker compose stop whisper-bridge && docker compose --profile voxtral up -d --build`.
|
||||||
|
- **Erster Start lädt Modelle** (mehrere GB via HuggingFace nach `xtts/hf-cache`
|
||||||
|
+ `xtts/models`) — genug Platz (1 TB NVMe ✓) und etwas Geduld.
|
||||||
|
|
||||||
|
## Verifizieren
|
||||||
|
|
||||||
|
```bash
|
||||||
|
nvidia-smi # Host sieht die GPU
|
||||||
|
docker run --rm --gpus all nvidia/cuda:12.4.0-base-ubuntu22.04 nvidia-smi # Container auch
|
||||||
|
docker logs -f aria-whisper-bridge # "RVS verbunden" + service_status ready
|
||||||
|
```
|
||||||
Executable
+277
@@ -0,0 +1,277 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
#
|
||||||
|
# ARIA KI-Box (gpubox) Bootstrap — frisches Debian Trixie → startklarer
|
||||||
|
# GPU-Satelliten-Host fuer den xtts-Stack (Voxtral/Whisper STT, F5-TTS, lokales LLM).
|
||||||
|
#
|
||||||
|
# Ablauf nach `git clone`:
|
||||||
|
# cd ARIA-AGENT/xtts
|
||||||
|
# sudo ./bootstrap.sh # richtet Treiber + Docker + NVIDIA-Toolkit ein
|
||||||
|
# # (falls Treiber frisch installiert: einmal `sudo reboot`, dann Script erneut)
|
||||||
|
# sudo ./bootstrap.sh --up # dazu: xtts-Stack (Whisper+F5+LLM) hochziehen
|
||||||
|
#
|
||||||
|
# Optionen:
|
||||||
|
# --up am Ende den xtts-Stack starten (Default-Profil, OHNE voxtral)
|
||||||
|
# --upgrade-driver NVIDIA-Treiber aus trixie-backports (modernes CUDA fuer Voxtral).
|
||||||
|
# Danach REBOOT noetig. Default-Bootstrap laesst 550 unangetastet.
|
||||||
|
# --token <TOK> RVS_TOKEN in xtts/.env eintragen (alternativ: env RVS_TOKEN=...)
|
||||||
|
# --rvs-host <H> RVS_HOST setzen (Default aus .env.example)
|
||||||
|
#
|
||||||
|
# IDEMPOTENT: bereits erledigte Schritte werden uebersprungen. Nach dem
|
||||||
|
# Treiber-Reboot einfach nochmal ausfuehren — der Rest laeuft dann durch.
|
||||||
|
#
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
# ── CLI ──
|
||||||
|
DO_UP=0
|
||||||
|
DO_UPGRADE_DRIVER=0
|
||||||
|
RVS_TOKEN_ARG="${RVS_TOKEN:-}"
|
||||||
|
RVS_HOST_ARG="${RVS_HOST:-}"
|
||||||
|
while [[ $# -gt 0 ]]; do
|
||||||
|
case "$1" in
|
||||||
|
--up) DO_UP=1; shift ;;
|
||||||
|
--upgrade-driver) DO_UPGRADE_DRIVER=1; shift ;;
|
||||||
|
--token) RVS_TOKEN_ARG="${2:-}"; shift 2 ;;
|
||||||
|
--rvs-host) RVS_HOST_ARG="${2:-}"; shift 2 ;;
|
||||||
|
-h|--help) grep '^#' "$0" | sed 's/^# \{0,1\}//'; exit 0 ;;
|
||||||
|
*) echo "Unbekannte Option: $1"; exit 1 ;;
|
||||||
|
esac
|
||||||
|
done
|
||||||
|
|
||||||
|
# ── Log-Helfer ──
|
||||||
|
c_g="\033[1;32m"; c_y="\033[1;33m"; c_r="\033[1;31m"; c_b="\033[1;34m"; c_0="\033[0m"
|
||||||
|
STEP=0
|
||||||
|
step() { STEP=$((STEP+1)); echo -e "\n${c_b}[${STEP}] $*${c_0}"; }
|
||||||
|
ok() { echo -e " ${c_g}✓${c_0} $*"; }
|
||||||
|
warn() { echo -e " ${c_y}!${c_0} $*"; }
|
||||||
|
die() { echo -e "${c_r}✗ $*${c_0}" >&2; exit 1; }
|
||||||
|
|
||||||
|
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||||
|
REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
|
||||||
|
XTTS_DIR="$REPO_ROOT/xtts"
|
||||||
|
|
||||||
|
# ── Preflight ──
|
||||||
|
[[ "$(id -u)" -eq 0 ]] || die "Bitte als root ausfuehren (sudo ./bootstrap.sh)."
|
||||||
|
command -v apt-get >/dev/null || die "Kein apt-get — dieses Script ist fuer Debian/Trixie."
|
||||||
|
export DEBIAN_FRONTEND=noninteractive
|
||||||
|
|
||||||
|
echo -e "${c_b}=== ARIA KI-Box Bootstrap ===${c_0}"
|
||||||
|
echo "Repo: $REPO_ROOT"
|
||||||
|
echo "xtts: $XTTS_DIR"
|
||||||
|
|
||||||
|
# ── 1. Basis-Pakete ──
|
||||||
|
step "Basis-Pakete"
|
||||||
|
apt-get update -qq
|
||||||
|
apt-get install -y -qq ca-certificates curl gnupg git lsb-release >/dev/null
|
||||||
|
ok "ca-certificates, curl, gnupg, git"
|
||||||
|
|
||||||
|
# ── 2. non-free Repos aktivieren (Trixie deb822 + Legacy) ──
|
||||||
|
step "APT-Komponenten (contrib non-free non-free-firmware)"
|
||||||
|
# Ergaenzt die drei Komponenten in JEDER Components:-Zeile einer deb822-Datei —
|
||||||
|
# pro Zeile nur was fehlt, reihenfolge-robust, idempotent. Die Adressen pruefen
|
||||||
|
# ganze Woerter (non-free-firmware zaehlt NICHT als non-free).
|
||||||
|
add_components() {
|
||||||
|
sed -i -E '/^[Cc]omponents:/{
|
||||||
|
/(^|[[:space:]])contrib([[:space:]]|$)/!s/$/ contrib/
|
||||||
|
/(^|[[:space:]])non-free([[:space:]]|$)/!s/$/ non-free/
|
||||||
|
/(^|[[:space:]])non-free-firmware([[:space:]]|$)/!s/$/ non-free-firmware/
|
||||||
|
}' "$1"
|
||||||
|
}
|
||||||
|
ENABLED_ANY=0
|
||||||
|
shopt -s nullglob
|
||||||
|
for f in /etc/apt/sources.list.d/*.sources; do
|
||||||
|
grep -qE '^[Cc]omponents:' "$f" || continue
|
||||||
|
b="$(md5sum "$f")"; add_components "$f"; a="$(md5sum "$f")"
|
||||||
|
if [[ "$b" != "$a" ]]; then ENABLED_ANY=1; ok "aktualisiert: $(basename "$f")"; fi
|
||||||
|
done
|
||||||
|
shopt -u nullglob
|
||||||
|
# Legacy /etc/apt/sources.list (deb-Zeilen)
|
||||||
|
if [[ -f /etc/apt/sources.list ]] && grep -qE '^deb ' /etc/apt/sources.list; then
|
||||||
|
if ! grep -qE '^deb .*[[:space:]]non-free([[:space:]]|$)' /etc/apt/sources.list; then
|
||||||
|
sed -i -E '/^deb .*debian/ s/$/ contrib non-free non-free-firmware/' /etc/apt/sources.list
|
||||||
|
ENABLED_ANY=1; ok "aktualisiert: sources.list"
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
[[ $ENABLED_ANY -eq 0 ]] && ok "non-free schon aktiv"
|
||||||
|
apt-get update -qq # immer neu einlesen, damit der Kandidat sicher da ist
|
||||||
|
|
||||||
|
# ── 2b. (opt-in) Treiber-Upgrade via trixie-backports — fuer Voxtral/modernes CUDA ──
|
||||||
|
# Der Trixie-Standardtreiber (550, CUDA 12.4) ist zu alt fuer den modernen Stack
|
||||||
|
# (torch 2.13, vLLM). Backports bringt einen neueren, apt-verwalteten Treiber.
|
||||||
|
# NUR mit --upgrade-driver, damit ein laufendes Setup nicht ungewollt angefasst wird.
|
||||||
|
if [[ "$DO_UPGRADE_DRIVER" -eq 1 ]]; then
|
||||||
|
step "NVIDIA-Treiber-Upgrade (trixie-backports)"
|
||||||
|
BP_FILE="/etc/apt/sources.list.d/backports.sources"
|
||||||
|
if ! grep -rqs "trixie-backports" /etc/apt/sources.list /etc/apt/sources.list.d/ 2>/dev/null; then
|
||||||
|
cat > "$BP_FILE" <<'EOF'
|
||||||
|
Types: deb
|
||||||
|
URIs: http://deb.debian.org/debian
|
||||||
|
Suites: trixie-backports
|
||||||
|
Components: main contrib non-free non-free-firmware
|
||||||
|
Signed-By: /usr/share/keyrings/debian-archive-keyring.gpg
|
||||||
|
EOF
|
||||||
|
ok "trixie-backports hinzugefuegt"
|
||||||
|
else
|
||||||
|
ok "trixie-backports bereits aktiv"
|
||||||
|
fi
|
||||||
|
apt-get update
|
||||||
|
apt-get install -y linux-headers-amd64 || true
|
||||||
|
apt-get install -y "linux-headers-$(uname -r)" || true
|
||||||
|
if apt-get install -y -t trixie-backports nvidia-driver; then
|
||||||
|
dkms autoinstall >/dev/null 2>&1 || true
|
||||||
|
NEWV="$(dpkg-query -W -f='${Version}' nvidia-driver 2>/dev/null || echo '?')"
|
||||||
|
ok "nvidia-driver aus backports installiert: ${NEWV}"
|
||||||
|
echo
|
||||||
|
echo -e "${c_y}==> REBOOT noetig, damit der neue Treiber laedt:${c_0}"
|
||||||
|
echo -e "${c_y} sudo reboot && danach: cd xtts && sudo ./bootstrap.sh --up --token <TOKEN>${c_0}"
|
||||||
|
echo -e "${c_y} (nvidia-smi zeigt dann die neue Version + CUDA-Level)${c_0}"
|
||||||
|
exit 0
|
||||||
|
else
|
||||||
|
warn "backports-Install fehlgeschlagen — 550er bleibt aktiv."
|
||||||
|
warn "Alternative fuer den neuesten Treiber: NVIDIAs CUDA-Repo fuer debian13"
|
||||||
|
warn " (developer.download.nvidia.com/compute/cuda/repos/debian13/x86_64) → Paket 'cuda-drivers'."
|
||||||
|
die "Treiber-Upgrade nicht moeglich — siehe oben."
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
|
||||||
|
# ── 3. NVIDIA-Treiber ──
|
||||||
|
step "NVIDIA-Treiber"
|
||||||
|
if nvidia-smi >/dev/null 2>&1; then
|
||||||
|
ok "Treiber aktiv: $(nvidia-smi --query-gpu=name --format=csv,noheader | paste -sd', ')"
|
||||||
|
DRIVER_ACTIVE=1
|
||||||
|
else
|
||||||
|
if dpkg -l | grep -q '^ii nvidia-driver '; then
|
||||||
|
warn "Treiber installiert, aber nvidia-smi antwortet nicht → REBOOT noetig."
|
||||||
|
DRIVER_ACTIVE=0
|
||||||
|
else
|
||||||
|
# KEIN separater Kandidaten-Check — die apt-cache-Ausgabe ist locale-/pipe-
|
||||||
|
# fragil (hat faelschlich "kein Kandidat" gemeldet). Der Install IST der Test:
|
||||||
|
# direkt installieren; schlaegt er fehl, einmal volles apt-get update + Retry,
|
||||||
|
# dann erst mit Diagnose abbrechen.
|
||||||
|
# Kernel-Header ZUERST — sonst ueberspringt DKMS den Modulbau ("No kernel
|
||||||
|
# headers were found") und nvidia-smi kann spaeter nicht mit dem Treiber reden.
|
||||||
|
warn "Kernel-Header + nvidia-driver installieren (DKMS-Build, dauert)…"
|
||||||
|
apt-get install -y linux-headers-amd64 || true
|
||||||
|
apt-get install -y "linux-headers-$(uname -r)" || true
|
||||||
|
if ! apt-get install -y nvidia-driver firmware-misc-nonfree; then
|
||||||
|
warn "Install fehlgeschlagen — volles 'apt-get update' + zweiter Versuch…"
|
||||||
|
apt-get update
|
||||||
|
if ! apt-get install -y nvidia-driver firmware-misc-nonfree; then
|
||||||
|
echo
|
||||||
|
warn "nvidia-driver liess sich nicht installieren. Aktive Quellen:"
|
||||||
|
grep -rhE '^[Cc]omponents:' /etc/apt/sources.list.d/*.sources 2>/dev/null | sed 's/^/ /' || true
|
||||||
|
grep -E '^deb ' /etc/apt/sources.list 2>/dev/null | sed 's/^/ /' || true
|
||||||
|
die "Pruefe 'apt-cache policy nvidia-driver' + Netz/non-free."
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
# Falls nvidia-kernel-dkms schon (ohne Header) installiert war: Modul jetzt bauen.
|
||||||
|
dkms autoinstall >/dev/null 2>&1 || true
|
||||||
|
ok "nvidia-driver + Kernel-Modul installiert"
|
||||||
|
DRIVER_ACTIVE=0
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
|
||||||
|
# ── 4. Docker Engine + Compose-Plugin ──
|
||||||
|
step "Docker"
|
||||||
|
if command -v docker >/dev/null 2>&1; then
|
||||||
|
ok "Docker vorhanden: $(docker --version)"
|
||||||
|
else
|
||||||
|
warn "Installiere Docker (offizielles get.docker.com)…"
|
||||||
|
curl -fsSL https://get.docker.com | sh >/dev/null
|
||||||
|
systemctl enable --now docker >/dev/null 2>&1 || true
|
||||||
|
ok "Docker installiert: $(docker --version)"
|
||||||
|
fi
|
||||||
|
if docker compose version >/dev/null 2>&1; then
|
||||||
|
ok "Compose-Plugin: $(docker compose version | head -1)"
|
||||||
|
else
|
||||||
|
warn "Compose-Plugin fehlt — installiere docker-compose-plugin…"
|
||||||
|
apt-get install -y -qq docker-compose-plugin >/dev/null || \
|
||||||
|
warn "Konnte docker-compose-plugin nicht via apt holen — get.docker.com bringt es normalerweise mit."
|
||||||
|
fi
|
||||||
|
|
||||||
|
# ── 5. NVIDIA Container Toolkit ──
|
||||||
|
step "NVIDIA Container Toolkit"
|
||||||
|
NCT_LIST="/etc/apt/sources.list.d/nvidia-container-toolkit.list"
|
||||||
|
NCT_KEY="/usr/share/keyrings/nvidia-container-toolkit-keyring.gpg"
|
||||||
|
if ! dpkg -l | grep -q '^ii nvidia-container-toolkit '; then
|
||||||
|
[[ -f "$NCT_KEY" ]] || curl -fsSL https://nvidia.github.io/libnvidia-container/gpgkey \
|
||||||
|
| gpg --dearmor -o "$NCT_KEY"
|
||||||
|
if [[ ! -f "$NCT_LIST" ]]; then
|
||||||
|
curl -fsSL https://nvidia.github.io/libnvidia-container/stable/deb/nvidia-container-toolkit.list \
|
||||||
|
| sed "s#deb https://#deb [signed-by=${NCT_KEY}] https://#g" > "$NCT_LIST"
|
||||||
|
fi
|
||||||
|
apt-get update -qq
|
||||||
|
apt-get install -y -qq nvidia-container-toolkit >/dev/null
|
||||||
|
ok "nvidia-container-toolkit installiert"
|
||||||
|
else
|
||||||
|
ok "nvidia-container-toolkit vorhanden"
|
||||||
|
fi
|
||||||
|
# Docker-Runtime auf NVIDIA konfigurieren (idempotent)
|
||||||
|
if ! grep -q '"nvidia"' /etc/docker/daemon.json 2>/dev/null; then
|
||||||
|
nvidia-ctk runtime configure --runtime=docker >/dev/null
|
||||||
|
systemctl restart docker
|
||||||
|
ok "Docker-Runtime auf NVIDIA konfiguriert + Docker neugestartet"
|
||||||
|
else
|
||||||
|
ok "Docker-Runtime bereits NVIDIA-konfiguriert"
|
||||||
|
fi
|
||||||
|
|
||||||
|
# ── 6. xtts/.env vorbereiten ──
|
||||||
|
step "xtts/.env"
|
||||||
|
if [[ -f "$XTTS_DIR/.env" ]]; then
|
||||||
|
ok ".env existiert bereits (unangetastet)"
|
||||||
|
else
|
||||||
|
cp "$XTTS_DIR/.env.example" "$XTTS_DIR/.env"
|
||||||
|
ok ".env aus .env.example erstellt"
|
||||||
|
fi
|
||||||
|
if [[ -n "$RVS_TOKEN_ARG" ]]; then
|
||||||
|
sed -i -E "s#^RVS_TOKEN=.*#RVS_TOKEN=${RVS_TOKEN_ARG}#" "$XTTS_DIR/.env"
|
||||||
|
ok "RVS_TOKEN eingetragen"
|
||||||
|
fi
|
||||||
|
if [[ -n "$RVS_HOST_ARG" ]]; then
|
||||||
|
sed -i -E "s#^RVS_HOST=.*#RVS_HOST=${RVS_HOST_ARG}#" "$XTTS_DIR/.env"
|
||||||
|
ok "RVS_HOST=${RVS_HOST_ARG} eingetragen"
|
||||||
|
fi
|
||||||
|
if grep -q '^RVS_TOKEN=dein_token_hier' "$XTTS_DIR/.env"; then
|
||||||
|
warn "RVS_TOKEN ist noch der Platzhalter — vor dem Start setzen:"
|
||||||
|
warn " nano $XTTS_DIR/.env (oder: sudo ./bootstrap.sh --token <TOKEN>)"
|
||||||
|
fi
|
||||||
|
warn "Stimm-Daten (nicht in git): falls von der alten Box noch vorhanden, xtts/voice-id/"
|
||||||
|
warn " + xtts/voices/ herkopieren. Sonst egal — in der App neu anlegen:"
|
||||||
|
warn " Sprache neu enrollen (Speaker-ID, sonst fail-open) + F5-Referenz neu hochladen."
|
||||||
|
|
||||||
|
# ── 7. GPU-im-Container verifizieren ──
|
||||||
|
step "GPU-im-Container Test"
|
||||||
|
if [[ "${DRIVER_ACTIVE:-0}" -eq 1 ]]; then
|
||||||
|
if docker run --rm --gpus all nvidia/cuda:12.4.0-base-ubuntu22.04 nvidia-smi >/dev/null 2>&1; then
|
||||||
|
ok "Docker sieht die GPU — KI-Box ist einsatzbereit."
|
||||||
|
GPU_READY=1
|
||||||
|
else
|
||||||
|
warn "Host-Treiber ok, aber Container sieht die GPU nicht — Toolkit/Runtime pruefen."
|
||||||
|
GPU_READY=0
|
||||||
|
fi
|
||||||
|
else
|
||||||
|
warn "Treiber noch nicht aktiv → Test uebersprungen."
|
||||||
|
echo
|
||||||
|
echo -e "${c_y}==> REBOOT noetig, dann Script erneut ausfuehren:${c_0}"
|
||||||
|
echo -e "${c_y} sudo reboot && (nach dem Boot) sudo ./bootstrap.sh${c_0}"
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
# ── 8. Optional: xtts-Stack hochziehen ──
|
||||||
|
if [[ $DO_UP -eq 1 && "${GPU_READY:-0}" -eq 1 ]]; then
|
||||||
|
step "xtts-Stack starten (Default-Profil: Whisper + F5 + LLM — OHNE voxtral)"
|
||||||
|
warn "Erster Start laedt Modelle (mehrere GB via HuggingFace) — kann dauern."
|
||||||
|
( cd "$XTTS_DIR" && docker compose up -d --build )
|
||||||
|
ok "Stack laeuft. Logs: docker logs -f aria-whisper-bridge"
|
||||||
|
echo
|
||||||
|
echo " voxtral (STT via Voxtral) braucht >=16 GB VRAM → erst mit der 24-GB-Karte:"
|
||||||
|
echo " cd $XTTS_DIR && docker compose stop whisper-bridge && docker compose --profile voxtral up -d --build"
|
||||||
|
fi
|
||||||
|
|
||||||
|
# ── Abschluss ──
|
||||||
|
echo
|
||||||
|
echo -e "${c_g}=== Fertig. KI-Box startklar. ===${c_0}"
|
||||||
|
if [[ $DO_UP -eq 0 ]]; then
|
||||||
|
echo "Naechster Schritt — Stack starten:"
|
||||||
|
echo " cd $XTTS_DIR && docker compose up -d --build"
|
||||||
|
echo "oder direkt: sudo ./bootstrap.sh --up"
|
||||||
|
fi
|
||||||
+94
-59
@@ -1,45 +1,50 @@
|
|||||||
# ════════════════════════════════════════════════
|
# ════════════════════════════════════════════════
|
||||||
# ARIA Gamebox Stack — GPU F5-TTS + Whisper STT
|
# ARIA Compute-Node — GPU-Dienste (STT / TTS / LLM)
|
||||||
# Laeuft auf dem Gaming-PC (RTX 3060)
|
|
||||||
# Verbindet sich zum RVS fuer TTS/STT-Requests
|
|
||||||
#
|
#
|
||||||
# FLUX-Bildgenerierung liegt im /flux Verzeichnis im Repo-Root —
|
# KEINE feste "AI-Box" mehr: dieser Stack laeuft auf beliebig vielen
|
||||||
# eigener Compose-Stack, kann auch auf einer anderen Maschine laufen.
|
# Worker-Nodes. Jeder Node startet ueber COMPOSE_PROFILES nur die Dienste,
|
||||||
|
# die er anbieten soll, und pinnt sie per *_GPU auf bestimmte Grafikkarten.
|
||||||
|
# Alles verbindet sich ueber RVS mit der ARIA-Infrastruktur — kein VPN.
|
||||||
|
#
|
||||||
|
# Beispiele (in der jeweiligen .env):
|
||||||
|
# STT-Box → COMPOSE_PROFILES=voxtral
|
||||||
|
# TTS-Box → COMPOSE_PROFILES=f5tts
|
||||||
|
# LLM-Box → COMPOSE_PROFILES=llm
|
||||||
|
# All-in-One → COMPOSE_PROFILES=voxtral,f5tts,llm (die alte AI-Box)
|
||||||
|
#
|
||||||
|
# FLUX-Bildgenerierung liegt im /flux Verzeichnis — eigener Stack.
|
||||||
# ════════════════════════════════════════════════
|
# ════════════════════════════════════════════════
|
||||||
#
|
#
|
||||||
# Voraussetzungen:
|
# Voraussetzungen:
|
||||||
# - Docker Desktop mit WSL2
|
# - Docker + NVIDIA Container Toolkit (registriert die `nvidia`-Runtime)
|
||||||
# - NVIDIA Container Toolkit
|
# - .env mit RVS-Verbindungsdaten, COMPOSE_PROFILES, NODE_NAME, *_GPU
|
||||||
# - .env mit RVS-Verbindungsdaten
|
|
||||||
#
|
#
|
||||||
# Start: docker compose up -d
|
# Start: docker compose up -d (liest COMPOSE_PROFILES aus der .env)
|
||||||
# ════════════════════════════════════════════════
|
# ════════════════════════════════════════════════
|
||||||
|
|
||||||
services:
|
services:
|
||||||
|
|
||||||
# ─── F5-TTS Bridge (GPU) ──────────────────────
|
# ─── F5-TTS Bridge (GPU) ──────────────────────
|
||||||
# Ersetzt den frueheren XTTS-Stack. Empfaengt xtts_request via RVS,
|
# Empfaengt xtts_request via RVS, rendert via F5-TTS mit Voice-Cloning,
|
||||||
# rendert via F5-TTS mit Voice-Cloning, streamt PCM an die App.
|
# streamt PCM an die App. Voice-Upload: speichert WAV und laesst eine
|
||||||
# Voice-Upload: speichert WAV und laesst whisper-bridge den Referenz-
|
# STT-Bridge den Referenztext transkribieren — der User tippt nichts.
|
||||||
# text transkribieren — der User muss nichts eintippen.
|
|
||||||
f5tts-bridge:
|
f5tts-bridge:
|
||||||
build: ./f5tts
|
build: ./f5tts
|
||||||
container_name: aria-f5tts-bridge
|
container_name: aria-f5tts-bridge
|
||||||
deploy:
|
profiles: ["f5tts"] # startet nur mit COMPOSE_PROFILES=…f5tts…
|
||||||
resources:
|
runtime: nvidia
|
||||||
reservations:
|
|
||||||
devices:
|
|
||||||
- driver: nvidia
|
|
||||||
count: 1
|
|
||||||
capabilities: [gpu]
|
|
||||||
volumes:
|
volumes:
|
||||||
- ./voices:/voices # WAV + TXT Referenz
|
- ./voices:/voices # WAV + TXT Referenz
|
||||||
- ./hf-cache:/root/.cache/huggingface # HF-Cache als Bind-Mount.
|
- ./hf-cache:/root/.cache/huggingface # HF-Cache als Bind-Mount.
|
||||||
# Direkt sichtbar im xtts/hf-cache/,
|
# Direkt sichtbar im xtts/hf-cache/,
|
||||||
# einfach manuell zu loeschen, kein
|
# einfach manuell zu loeschen, kein
|
||||||
# Docker-Desktop .vhdx Bloat.
|
# Docker-Desktop .vhdx Bloat.
|
||||||
# Wird mit whisper-bridge geteilt.
|
# Wird mit STT-Bridges geteilt.
|
||||||
environment:
|
environment:
|
||||||
|
# GPU-Wahl: NVIDIA_VISIBLE_DEVICES (mehrere via "0,1"). Default GPU 0.
|
||||||
|
- NVIDIA_VISIBLE_DEVICES=${F5TTS_GPU:-0}
|
||||||
|
- NVIDIA_DRIVER_CAPABILITIES=compute,utility
|
||||||
|
- NODE_NAME=${NODE_NAME:-node} # erscheint in Diagnostic + Logs
|
||||||
# Bootstrap-only — alle anderen F5-TTS-Settings (Modell, cfg_strength,
|
# Bootstrap-only — alle anderen F5-TTS-Settings (Modell, cfg_strength,
|
||||||
# nfe_step, Custom-Checkpoint) kommen ueber Diagnostic via RVS-config.
|
# nfe_step, Custom-Checkpoint) kommen ueber Diagnostic via RVS-config.
|
||||||
- RVS_HOST=${RVS_HOST}
|
- RVS_HOST=${RVS_HOST}
|
||||||
@@ -51,26 +56,22 @@ services:
|
|||||||
- VOICES_DIR=/voices
|
- VOICES_DIR=/voices
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
|
|
||||||
# ─── Whisper STT (GPU) ────────────────────────
|
# ─── Whisper STT (GPU) — opt-in Fallback-STT ──
|
||||||
# Faster-Whisper auf der Gamebox statt auf der VM (CPU) —
|
# Faster-Whisper auf CUDA. Verbindet sich selbst per WebSocket an den RVS
|
||||||
# deutlich schneller. Verbindet sich selbst per WebSocket an
|
# und nimmt stt_request / stt_stream_* Nachrichten entgegen. Zusaetzlich
|
||||||
# den RVS und nimmt dort stt_request Nachrichten der aria-bridge
|
# nutzt die f5tts-bridge Whisper intern fuer die Referenz-Transkription bei
|
||||||
# entgegen, antwortet mit stt_response. Zusaetzlich nutzt die
|
# Voice-Uploads. Modell-Hot-Swap via Diagnostic (config-Broadcast).
|
||||||
# f5tts-bridge Whisper intern fuer die Referenz-Transkription bei
|
# Nur EINEN STT-Provider pro Node laufen lassen (voxtral ODER whisper) —
|
||||||
# Voice-Uploads. Laedt das Modell beim Start vor; auf Config-
|
# sonst beantworten beide dieselbe Anfrage doppelt.
|
||||||
# Broadcasts (Diagnostic → whisperModel) wird zur Laufzeit hot-
|
|
||||||
# swapped.
|
|
||||||
whisper-bridge:
|
whisper-bridge:
|
||||||
build: ./whisper
|
build: ./whisper
|
||||||
container_name: aria-whisper-bridge
|
container_name: aria-whisper-bridge
|
||||||
deploy:
|
profiles: ["whisper"] # startet nur mit COMPOSE_PROFILES=…whisper…
|
||||||
resources:
|
runtime: nvidia
|
||||||
reservations:
|
|
||||||
devices:
|
|
||||||
- driver: nvidia
|
|
||||||
count: 1
|
|
||||||
capabilities: [gpu]
|
|
||||||
environment:
|
environment:
|
||||||
|
- NVIDIA_VISIBLE_DEVICES=${WHISPER_GPU:-1}
|
||||||
|
- NVIDIA_DRIVER_CAPABILITIES=compute,utility
|
||||||
|
- NODE_NAME=${NODE_NAME:-node}
|
||||||
- RVS_HOST=${RVS_HOST}
|
- RVS_HOST=${RVS_HOST}
|
||||||
- RVS_PORT=${RVS_PORT:-443}
|
- RVS_PORT=${RVS_PORT:-443}
|
||||||
- RVS_TLS=${RVS_TLS:-true}
|
- RVS_TLS=${RVS_TLS:-true}
|
||||||
@@ -90,43 +91,74 @@ services:
|
|||||||
# Container-Restarts.
|
# Container-Restarts.
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
|
|
||||||
# ─── Lokales LLM (Plan B, B0.5) — llama-swap (GPU) ────────────
|
# ─── Voxtral STT-3B (Transformers, GPU) — DEFAULT-STT ─────────
|
||||||
# llama-swap laedt/swappt mehrere Modelle on-demand (nur eins passt gleich-
|
# Voxtral-Mini-3B-2507 (~9 GB bf16). Laeuft auf Treiber 550/CUDA 12.4
|
||||||
# zeitig in die 12 GB). Welches geladen wird, bestimmt das `model`-Feld im
|
# (torch cu124, KEIN Treiber-Upgrade noetig). Whisper ist der opt-in
|
||||||
# Request — das Brain schickt es aus local_llm.json mit. Erster Load eines
|
# Fallback — immer nur EINEN STT-Provider pro Node aktiv haben.
|
||||||
# Modells zieht das GGUF via -hf von HF (Cache unter /models, persistent).
|
voxtral-bridge:
|
||||||
|
build: ./voxtral
|
||||||
|
container_name: aria-voxtral-bridge
|
||||||
|
profiles: ["voxtral"] # startet nur mit COMPOSE_PROFILES=…voxtral…
|
||||||
|
runtime: nvidia
|
||||||
|
volumes:
|
||||||
|
- ./hf-cache:/root/.cache/huggingface # gleicher Modell-Cache wie whisper/f5
|
||||||
|
- ./voice-id:/voice-id # Speaker-Fingerprint (wie whisper)
|
||||||
|
environment:
|
||||||
|
- NVIDIA_VISIBLE_DEVICES=${VOXTRAL_GPU:-1} # STT-3B ~9 GB → 12-GB-Karte
|
||||||
|
- NVIDIA_DRIVER_CAPABILITIES=compute,utility
|
||||||
|
- NODE_NAME=${NODE_NAME:-node}
|
||||||
|
- RVS_HOST=${RVS_HOST}
|
||||||
|
- RVS_PORT=${RVS_PORT:-443}
|
||||||
|
- RVS_TLS=${RVS_TLS:-true}
|
||||||
|
- RVS_TLS_FALLBACK=${RVS_TLS_FALLBACK:-true}
|
||||||
|
- RVS_TOKEN=${RVS_TOKEN}
|
||||||
|
- VOXTRAL_MODEL=mistralai/Voxtral-Mini-3B-2507
|
||||||
|
- VOXTRAL_LANGUAGE=${WHISPER_LANGUAGE:-de}
|
||||||
|
- HUGGING_FACE_HUB_TOKEN=${HF_TOKEN:-} # falls das Modell HF-gated ist
|
||||||
|
- PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True # weniger VRAM-Fragmentierung
|
||||||
|
restart: unless-stopped
|
||||||
|
|
||||||
|
# ─── Lokales LLM — llama-swap (GPU) ───────────
|
||||||
|
# llama-swap laedt/swappt mehrere Modelle on-demand. Welches geladen wird,
|
||||||
|
# bestimmt das `model`-Feld im Request (Brain schickt es aus local_llm.json).
|
||||||
|
# Erster Load zieht das GGUF via -hf von HF (Cache unter /models, persistent).
|
||||||
# OpenAI-kompatibel auf :8080, nur im Compose-Netz; die Bruecke macht der
|
# OpenAI-kompatibel auf :8080, nur im Compose-Netz; die Bruecke macht der
|
||||||
# llm-adapter. Modell-Liste: ./llama-swap/config.yaml.
|
# llm-adapter. Die Modell-Liste erzeugt der llm-adapter dynamisch aus
|
||||||
#
|
# ./llama-swap/config.yaml (Basis) + Registry → /models/llama-swap.config.yaml.
|
||||||
# BLIND GEBAUT (kein Gamebox-Test hier): beim ersten Start
|
|
||||||
# `docker logs -f aria-llama-swap` pruefen. Image bundelt llama-server.
|
|
||||||
llama-swap:
|
llama-swap:
|
||||||
image: ghcr.io/mostlygeek/llama-swap:unified-cuda
|
image: ghcr.io/mostlygeek/llama-swap:unified-cuda
|
||||||
container_name: aria-llama-swap
|
container_name: aria-llama-swap
|
||||||
deploy:
|
profiles: ["llm"] # startet nur mit COMPOSE_PROFILES=…llm…
|
||||||
resources:
|
runtime: nvidia
|
||||||
reservations:
|
|
||||||
devices:
|
|
||||||
- driver: nvidia
|
|
||||||
count: 1
|
|
||||||
capabilities: [gpu]
|
|
||||||
volumes:
|
volumes:
|
||||||
- ./models:/models # HF-Download-Cache (persistent)
|
- ./models:/models # HF-Cache + generierte Config
|
||||||
- ./llama-swap/config.yaml:/app/config.yaml:ro # Modell-Liste
|
|
||||||
environment:
|
environment:
|
||||||
|
- NVIDIA_VISIBLE_DEVICES=${LLM_GPU:-0}
|
||||||
|
- NVIDIA_DRIVER_CAPABILITIES=compute,utility
|
||||||
- LLAMA_CACHE=/models # llama-server legt -hf-Downloads hier ab
|
- LLAMA_CACHE=/models # llama-server legt -hf-Downloads hier ab
|
||||||
command: ["--config", "/app/config.yaml", "--listen", "0.0.0.0:8080"]
|
# Liest die vom llm-adapter generierte Config. Beim allerersten Boot faengt
|
||||||
|
# restart: unless-stopped die Reihenfolge ab, bis der Adapter sie geschrieben hat.
|
||||||
|
command: ["--config", "/models/llama-swap.config.yaml", "--listen", "0.0.0.0:8080"]
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
|
|
||||||
# ─── Local-LLM-Adapter — RVS <-> llama.cpp (Plan B, B0) ──────
|
# ─── Local-LLM-Adapter — RVS <-> llama.cpp ────
|
||||||
# Verbindet sich per Token an den RVS (wie f5tts/whisper), nimmt
|
# Verbindet sich per Token an den RVS (wie f5tts/whisper), nimmt llm_request
|
||||||
# llm_request entgegen, ruft llama.cpp lokal, antwortet llm_response.
|
# entgegen, ruft llama.cpp lokal, antwortet llm_response. Verwaltet ausserdem
|
||||||
|
# llama-swaps Config (Modelle hinzufuegen/entfernen via llm_provision_model).
|
||||||
llm-adapter:
|
llm-adapter:
|
||||||
build: ./llm-adapter
|
build: ./llm-adapter
|
||||||
container_name: aria-llm-adapter
|
container_name: aria-llm-adapter
|
||||||
|
profiles: ["llm"]
|
||||||
|
runtime: nvidia # nur fuer nvidia-smi (Auslastungs-Monitor) — kein Compute
|
||||||
depends_on:
|
depends_on:
|
||||||
- llama-swap
|
- llama-swap
|
||||||
|
volumes:
|
||||||
|
- ./models:/models # generierte Config + Registry + Cache
|
||||||
|
- ./llama-swap:/llamaswap:ro # Basis-Template (config.yaml)
|
||||||
environment:
|
environment:
|
||||||
|
- NVIDIA_VISIBLE_DEVICES=all # alle Karten sichtbar (nur nvidia-smi)
|
||||||
|
- NVIDIA_DRIVER_CAPABILITIES=utility # utility = nvidia-smi, KEIN VRAM/Compute
|
||||||
|
- NODE_NAME=${NODE_NAME:-node}
|
||||||
- RVS_HOST=${RVS_HOST}
|
- RVS_HOST=${RVS_HOST}
|
||||||
- RVS_PORT=${RVS_PORT:-443}
|
- RVS_PORT=${RVS_PORT:-443}
|
||||||
- RVS_TLS=${RVS_TLS:-true}
|
- RVS_TLS=${RVS_TLS:-true}
|
||||||
@@ -134,6 +166,9 @@ services:
|
|||||||
- RVS_TOKEN=${RVS_TOKEN}
|
- RVS_TOKEN=${RVS_TOKEN}
|
||||||
- LLAMA_URL=http://llama-swap:8080
|
- LLAMA_URL=http://llama-swap:8080
|
||||||
- LLM_MODEL=${LLM_MODEL:-qwen3-8b}
|
- LLM_MODEL=${LLM_MODEL:-qwen3-8b}
|
||||||
|
- LLAMA_BASE_CONFIG=/llamaswap/config.yaml
|
||||||
|
- LLAMA_GEN_CONFIG=/models/llama-swap.config.yaml
|
||||||
|
- LLM_REGISTRY=/models/aria_models.json
|
||||||
# Erster Load eines Modells kann ein GGUF ziehen (mehrere GB) — grosszuegig.
|
# Erster Load eines Modells kann ein GGUF ziehen (mehrere GB) — grosszuegig.
|
||||||
- LLM_TIMEOUT_SEC=${LLM_TIMEOUT_SEC:-600}
|
- LLM_TIMEOUT_SEC=${LLM_TIMEOUT_SEC:-600}
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
|
|||||||
@@ -9,13 +9,18 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
|
|||||||
|
|
||||||
WORKDIR /app
|
WORKDIR /app
|
||||||
|
|
||||||
# PyTorch CUDA-Wheels zuerst (f5-tts zieht sonst CPU-only Torch rein)
|
# torch FEST auf cu124 (Treiber 550 = CUDA 12.4). 2.6.0 ist der NEUESTE cu124-
|
||||||
RUN pip3 install --no-cache-dir torch==2.3.1 torchaudio==2.3.1 \
|
# Build — torch 2.7+ gibt es nur noch fuer cu126+, was 550 nicht unterstuetzt
|
||||||
--index-url https://download.pytorch.org/whl/cu121
|
# ("NVIDIA driver too old, found 12040"). Der Constraint haelt f5-tts davon ab,
|
||||||
|
# torch beim Dependency-Aufloesen wieder auf eine zu neue CUDA-Version zu ziehen.
|
||||||
|
RUN pip3 install --no-cache-dir torch==2.6.0 torchaudio==2.6.0 \
|
||||||
|
--index-url https://download.pytorch.org/whl/cu124
|
||||||
|
|
||||||
COPY requirements.txt .
|
COPY requirements.txt .
|
||||||
RUN pip3 install --no-cache-dir -r requirements.txt
|
RUN printf 'torch==2.6.0\ntorchaudio==2.6.0\n' > /tmp/torch-constraint.txt && \
|
||||||
|
pip3 install --no-cache-dir -c /tmp/torch-constraint.txt -r requirements.txt
|
||||||
|
|
||||||
|
COPY node_stats.py .
|
||||||
COPY bridge.py .
|
COPY bridge.py .
|
||||||
|
|
||||||
CMD ["python3", "bridge.py"]
|
CMD ["python3", "bridge.py"]
|
||||||
|
|||||||
+87
-8
@@ -1,6 +1,6 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
"""
|
"""
|
||||||
ARIA F5-TTS Bridge — laeuft auf der Gamebox (RTX 3060).
|
ARIA F5-TTS Bridge — laeuft auf der AI-Box (RTX 3060).
|
||||||
|
|
||||||
Empfaengt xtts_request via RVS → F5-TTS Voice Cloning auf GPU → streamt
|
Empfaengt xtts_request via RVS → F5-TTS Voice Cloning auf GPU → streamt
|
||||||
16-bit PCM Chunks als audio_pcm Nachrichten zurueck an die App.
|
16-bit PCM Chunks als audio_pcm Nachrichten zurueck an die App.
|
||||||
@@ -60,6 +60,24 @@ RVS_TOKEN = os.getenv("RVS_TOKEN", "").strip()
|
|||||||
# f5ttsCkptFile, f5ttsVocabFile, f5ttsCfgStrength, f5ttsNfeStep).
|
# f5ttsCkptFile, f5ttsVocabFile, f5ttsCfgStrength, f5ttsNfeStep).
|
||||||
F5TTS_DEVICE = os.getenv("F5TTS_DEVICE", "cuda") # nur Bootstrap
|
F5TTS_DEVICE = os.getenv("F5TTS_DEVICE", "cuda") # nur Bootstrap
|
||||||
|
|
||||||
|
# ── Compute-Fleet: Worker-Identitaet & Registrierung ──────────────
|
||||||
|
# Meldet sich bei der aria-bridge (worker_hello) + periodischer worker_ping.
|
||||||
|
NODE_NAME = os.getenv("NODE_NAME", "node").strip() or "node"
|
||||||
|
GPU_IDS = os.getenv("NVIDIA_VISIBLE_DEVICES", "").strip()
|
||||||
|
WORKER_SERVICE = "f5tts"
|
||||||
|
INSTANCE_ID = f"{WORKER_SERVICE}@{NODE_NAME}"
|
||||||
|
WORKER_PING_INTERVAL_S = int(os.getenv("WORKER_PING_INTERVAL_S", "10"))
|
||||||
|
# Empfangs-Watchdog: kommt in RX_STALE_S kein Broadcast rein (ein echter Raum hat
|
||||||
|
# staendig Traffic, z.B. sat_hello alle 25s / Brain-Polling), gilt die Verbindung
|
||||||
|
# als halb-tot (Caddy pongt die WS-Pings selbst) -> Zwangs-Reconnect.
|
||||||
|
RX_STALE_S = int(os.getenv("RX_STALE_S", "60"))
|
||||||
|
_tts_busy = False # True waehrend eine Synthese laeuft (busy-Report im ping)
|
||||||
|
|
||||||
|
# ── Auslastungs-Monitor (Stage E) ──────────────────────────
|
||||||
|
import node_stats
|
||||||
|
STATS_PATH = os.getenv("STATS_PATH", f"/root/.cache/huggingface/aria_stats_{WORKER_SERVICE}.json")
|
||||||
|
_stats = node_stats.NodeStats(INSTANCE_ID, NODE_NAME, STATS_PATH, logger=logger)
|
||||||
|
|
||||||
DEFAULT_F5TTS_MODEL = "F5TTS_v1_Base"
|
DEFAULT_F5TTS_MODEL = "F5TTS_v1_Base"
|
||||||
DEFAULT_F5TTS_CKPT_FILE = "" # leer = Default-Checkpoint von HF
|
DEFAULT_F5TTS_CKPT_FILE = "" # leer = Default-Checkpoint von HF
|
||||||
DEFAULT_F5TTS_VOCAB_FILE = "" # leer = Default-Vocab vom Modell
|
DEFAULT_F5TTS_VOCAB_FILE = "" # leer = Default-Vocab vom Modell
|
||||||
@@ -378,7 +396,7 @@ async def _send(ws, mtype: str, payload: dict) -> None:
|
|||||||
# ──────────────────────────────────────────────────────────────
|
# ──────────────────────────────────────────────────────────────
|
||||||
# DEBUG-LOG ueber RVS → /shared/logs/app.log
|
# DEBUG-LOG ueber RVS → /shared/logs/app.log
|
||||||
#
|
#
|
||||||
# Gleiches Pattern wie in whisper-bridge: Stefan's Gamebox ist
|
# Gleiches Pattern wie in whisper-bridge: Stefan's AI-Box ist
|
||||||
# Windows (kein SSH), in Zukunft koennten whisper + f5tts auf
|
# Windows (kein SSH), in Zukunft koennten whisper + f5tts auf
|
||||||
# unterschiedlichen Hosts laufen. Logs ueber RVS heisst: ein Pfad.
|
# unterschiedlichen Hosts laufen. Logs ueber RVS heisst: ein Pfad.
|
||||||
#
|
#
|
||||||
@@ -460,13 +478,16 @@ _tts_queue: asyncio.Queue[tuple] = asyncio.Queue()
|
|||||||
|
|
||||||
async def _tts_worker(ws, runner: F5Runner) -> None:
|
async def _tts_worker(ws, runner: F5Runner) -> None:
|
||||||
"""Serialisiert Synthesen — GPU kann sonst OOM gehen."""
|
"""Serialisiert Synthesen — GPU kann sonst OOM gehen."""
|
||||||
|
global _tts_busy
|
||||||
while True:
|
while True:
|
||||||
text, voice, request_id, message_id, language, speed = await _tts_queue.get()
|
text, voice, request_id, message_id, language, speed = await _tts_queue.get()
|
||||||
|
_tts_busy = True
|
||||||
try:
|
try:
|
||||||
await _do_tts(ws, runner, text, voice, request_id, message_id, language, speed)
|
await _do_tts(ws, runner, text, voice, request_id, message_id, language, speed)
|
||||||
except Exception:
|
except Exception:
|
||||||
logger.exception("TTS-Worker Fehler")
|
logger.exception("TTS-Worker Fehler")
|
||||||
finally:
|
finally:
|
||||||
|
_tts_busy = False
|
||||||
_tts_queue.task_done()
|
_tts_queue.task_done()
|
||||||
|
|
||||||
|
|
||||||
@@ -663,6 +684,18 @@ async def handle_voice_upload(ws, payload: dict) -> None:
|
|||||||
await _send(ws, "xtts_voice_saved", {"name": name, "error": str(e)[:200]})
|
await _send(ws, "xtts_voice_saved", {"name": name, "error": str(e)[:200]})
|
||||||
|
|
||||||
|
|
||||||
|
def _local_voice_names() -> list:
|
||||||
|
"""Namen der lokal vorhandenen Nutzer-Stimmen (ohne den Box-lokalen
|
||||||
|
default_ref-Fallback) — fuer die Provisioning-Reconciliation im worker_hello."""
|
||||||
|
names = []
|
||||||
|
if VOICES_DIR.exists():
|
||||||
|
for wav in sorted(VOICES_DIR.glob("*.wav")):
|
||||||
|
if wav.stem == "default_ref":
|
||||||
|
continue
|
||||||
|
names.append(wav.stem)
|
||||||
|
return names
|
||||||
|
|
||||||
|
|
||||||
async def handle_list_voices(ws) -> None:
|
async def handle_list_voices(ws) -> None:
|
||||||
try:
|
try:
|
||||||
voices = []
|
voices = []
|
||||||
@@ -699,13 +732,16 @@ async def handle_delete_voice(ws, payload: dict) -> None:
|
|||||||
async def handle_export_voice(ws, payload: dict) -> None:
|
async def handle_export_voice(ws, payload: dict) -> None:
|
||||||
"""Packt eine Stimme (.wav + .txt) als tar.gz und sendet sie base64 zurueck."""
|
"""Packt eine Stimme (.wav + .txt) als tar.gz und sendet sie base64 zurueck."""
|
||||||
name = (payload.get("name") or "").strip()
|
name = (payload.get("name") or "").strip()
|
||||||
|
# requestId aus dem Request zuruueckspiegeln — der Diagnostic-Bibliothekar
|
||||||
|
# korreliert damit zentrale Exports gegen browser-initiierte.
|
||||||
|
req_id = payload.get("requestId", "") or ""
|
||||||
if not name:
|
if not name:
|
||||||
await _send(ws, "xtts_voice_exported", {"ok": False, "error": "name fehlt"})
|
await _send(ws, "xtts_voice_exported", {"ok": False, "requestId": req_id, "error": "name fehlt"})
|
||||||
return
|
return
|
||||||
try:
|
try:
|
||||||
wav, txt = voice_paths(name)
|
wav, txt = voice_paths(name)
|
||||||
if not wav.exists():
|
if not wav.exists():
|
||||||
await _send(ws, "xtts_voice_exported", {"ok": False, "name": name, "error": "Stimme nicht gefunden"})
|
await _send(ws, "xtts_voice_exported", {"ok": False, "requestId": req_id, "name": name, "error": "Stimme nicht gefunden"})
|
||||||
return
|
return
|
||||||
import io, tarfile
|
import io, tarfile
|
||||||
buf = io.BytesIO()
|
buf = io.BytesIO()
|
||||||
@@ -715,10 +751,10 @@ async def handle_export_voice(ws, payload: dict) -> None:
|
|||||||
tar.add(txt, arcname=txt.name)
|
tar.add(txt, arcname=txt.name)
|
||||||
data = base64.b64encode(buf.getvalue()).decode("ascii")
|
data = base64.b64encode(buf.getvalue()).decode("ascii")
|
||||||
logger.info("Voice exportiert: %s (%d KB tar.gz)", name, len(buf.getvalue()) // 1024)
|
logger.info("Voice exportiert: %s (%d KB tar.gz)", name, len(buf.getvalue()) // 1024)
|
||||||
await _send(ws, "xtts_voice_exported", {"ok": True, "name": name, "data": data})
|
await _send(ws, "xtts_voice_exported", {"ok": True, "requestId": req_id, "name": name, "data": data})
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.exception("handle_export_voice Fehler")
|
logger.exception("handle_export_voice Fehler")
|
||||||
await _send(ws, "xtts_voice_exported", {"ok": False, "name": name, "error": str(e)[:200]})
|
await _send(ws, "xtts_voice_exported", {"ok": False, "requestId": req_id, "name": name, "error": str(e)[:200]})
|
||||||
|
|
||||||
|
|
||||||
async def handle_import_voice(ws, payload: dict) -> None:
|
async def handle_import_voice(ws, payload: dict) -> None:
|
||||||
@@ -808,6 +844,31 @@ async def _broadcast_status(ws, state: str, **extra) -> None:
|
|||||||
await _send(ws, "service_status", payload)
|
await _send(ws, "service_status", payload)
|
||||||
|
|
||||||
|
|
||||||
|
async def _worker_register(ws, *, model: str = "", busy_fn=None) -> None:
|
||||||
|
"""Meldet diesen Worker bei der aria-bridge an (worker_hello) und haelt die
|
||||||
|
Flotten-Registry per periodischem worker_ping frisch. worker_hello wird
|
||||||
|
zusaetzlich alle ~30s WIEDERHOLT (wie der Satellit sat_hello), damit auch ein
|
||||||
|
neu gestartetes Diagnostic/Bridge uns lernt — RVS spielt hellos nicht nach."""
|
||||||
|
def _hello():
|
||||||
|
return {"instanceId": INSTANCE_ID, "service": WORKER_SERVICE,
|
||||||
|
"node": NODE_NAME, "gpus": GPU_IDS, "model": model,
|
||||||
|
"voices": _local_voice_names()}
|
||||||
|
try:
|
||||||
|
await _send(ws, "worker_hello", _hello())
|
||||||
|
n = 0
|
||||||
|
while True:
|
||||||
|
await asyncio.sleep(WORKER_PING_INTERVAL_S)
|
||||||
|
n += 1
|
||||||
|
busy = bool(busy_fn()) if busy_fn else False
|
||||||
|
await _send(ws, "worker_ping", {"instanceId": INSTANCE_ID, "busy": busy})
|
||||||
|
if n % 3 == 0:
|
||||||
|
await _send(ws, "worker_hello", _hello())
|
||||||
|
except asyncio.CancelledError:
|
||||||
|
raise
|
||||||
|
except Exception:
|
||||||
|
return # Socket tot → still beenden; run_loop reconnectet + startet neu
|
||||||
|
|
||||||
|
|
||||||
async def run_loop(runner: F5Runner) -> None:
|
async def run_loop(runner: F5Runner) -> None:
|
||||||
use_tls = RVS_TLS
|
use_tls = RVS_TLS
|
||||||
retry_s = 2
|
retry_s = 2
|
||||||
@@ -816,7 +877,7 @@ async def run_loop(runner: F5Runner) -> None:
|
|||||||
|
|
||||||
while True:
|
while True:
|
||||||
scheme = "wss" if use_tls else "ws"
|
scheme = "wss" if use_tls else "ws"
|
||||||
url = f"{scheme}://{RVS_HOST}:{RVS_PORT}/ws?token={RVS_TOKEN}"
|
url = f"{scheme}://{RVS_HOST}:{RVS_PORT}?token={RVS_TOKEN}"
|
||||||
masked = url.replace(RVS_TOKEN, "***") if RVS_TOKEN else url
|
masked = url.replace(RVS_TOKEN, "***") if RVS_TOKEN else url
|
||||||
|
|
||||||
try:
|
try:
|
||||||
@@ -855,15 +916,31 @@ async def run_loop(runner: F5Runner) -> None:
|
|||||||
|
|
||||||
# TTS-Worker fuer diese Verbindung starten
|
# TTS-Worker fuer diese Verbindung starten
|
||||||
worker = asyncio.create_task(_tts_worker(ws, runner))
|
worker = asyncio.create_task(_tts_worker(ws, runner))
|
||||||
|
ping_task = asyncio.create_task(_worker_register(
|
||||||
|
ws, model=runner.model_id,
|
||||||
|
busy_fn=lambda: _tts_busy or not _tts_queue.empty()))
|
||||||
|
|
||||||
try:
|
try:
|
||||||
async for raw in ws:
|
while True:
|
||||||
|
try:
|
||||||
|
raw = await asyncio.wait_for(ws.recv(), timeout=RX_STALE_S)
|
||||||
|
except asyncio.TimeoutError:
|
||||||
|
logger.warning("Kein RVS-Traffic seit %ds — Verbindung halb-tot, reconnect", RX_STALE_S)
|
||||||
|
raise ConnectionError("rvs-stale")
|
||||||
try:
|
try:
|
||||||
msg = json.loads(raw)
|
msg = json.loads(raw)
|
||||||
except Exception:
|
except Exception:
|
||||||
continue
|
continue
|
||||||
mtype = msg.get("type", "")
|
mtype = msg.get("type", "")
|
||||||
payload = msg.get("payload", {}) or {}
|
payload = msg.get("payload", {}) or {}
|
||||||
|
# Redundanz-Routing: gezielt an eine andere Instanz
|
||||||
|
# adressiert → ignorieren. Ohne targetInstance → wie bisher.
|
||||||
|
tgt = payload.get("targetInstance")
|
||||||
|
if tgt and tgt != INSTANCE_ID:
|
||||||
|
continue
|
||||||
|
# Auslastungs-Monitor (node_stats_*) abfangen.
|
||||||
|
if await _stats.handle(ws, mtype, payload, _send):
|
||||||
|
continue
|
||||||
|
|
||||||
if mtype == "xtts_request":
|
if mtype == "xtts_request":
|
||||||
try:
|
try:
|
||||||
@@ -958,6 +1035,7 @@ async def run_loop(runner: F5Runner) -> None:
|
|||||||
_last_diag_voice = ""
|
_last_diag_voice = ""
|
||||||
finally:
|
finally:
|
||||||
worker.cancel()
|
worker.cancel()
|
||||||
|
ping_task.cancel()
|
||||||
try:
|
try:
|
||||||
await worker
|
await worker
|
||||||
except asyncio.CancelledError:
|
except asyncio.CancelledError:
|
||||||
@@ -985,6 +1063,7 @@ async def main() -> None:
|
|||||||
sys.exit(1)
|
sys.exit(1)
|
||||||
VOICES_DIR.mkdir(parents=True, exist_ok=True)
|
VOICES_DIR.mkdir(parents=True, exist_ok=True)
|
||||||
runner = F5Runner()
|
runner = F5Runner()
|
||||||
|
asyncio.create_task(_stats.run_sampler()) # Auslastungs-Sampler (Stage E)
|
||||||
await run_loop(runner)
|
await run_loop(runner)
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,167 @@
|
|||||||
|
"""
|
||||||
|
ARIA Node-Stats — Auslastungs-Monitor pro Box (GPU + optional Tokens).
|
||||||
|
|
||||||
|
Identische Kopie in jedem Worker-Build-Context (f5tts/whisper/voxtral/llm-adapter),
|
||||||
|
weil jeder Worker ein eigener Docker-Build-Context ist.
|
||||||
|
|
||||||
|
Aufgaben:
|
||||||
|
- Sampler-Loop (alle SAMPLE_SEC): nvidia-smi-Auslastung + Token-Delta → Ringpuffer
|
||||||
|
(persistent als JSON auf der Box). Laeuft unabhaengig vom Modal.
|
||||||
|
- Live-Stream: bei node_stats_stream_start jede Sekunde rohes nvidia-smi + Werte
|
||||||
|
senden (bis stop / Auto-Timeout).
|
||||||
|
- History-Request + Reset (Besen).
|
||||||
|
|
||||||
|
Reicht `handle(ws, mtype, payload)` in die Worker-Message-Loop ein; gibt True
|
||||||
|
zurueck, wenn die Nachricht eine node_stats_*-Nachricht war.
|
||||||
|
"""
|
||||||
|
import asyncio
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import time
|
||||||
|
|
||||||
|
SAMPLE_SEC = int(os.getenv("STATS_SAMPLE_SEC", "15"))
|
||||||
|
HISTORY_CAP = int(os.getenv("STATS_HISTORY_CAP", "500")) # ~2h bei 15s
|
||||||
|
STREAM_MAX_SEC = int(os.getenv("STATS_STREAM_MAX_SEC", "300"))
|
||||||
|
|
||||||
|
|
||||||
|
async def _run_cmd(*args, timeout=8) -> str:
|
||||||
|
"""Fuehrt ein Kommando aus, gibt stdout (str) zurueck; '' bei Fehler."""
|
||||||
|
try:
|
||||||
|
proc = await asyncio.create_subprocess_exec(
|
||||||
|
*args,
|
||||||
|
stdout=asyncio.subprocess.PIPE,
|
||||||
|
stderr=asyncio.subprocess.DEVNULL,
|
||||||
|
)
|
||||||
|
out, _ = await asyncio.wait_for(proc.communicate(), timeout=timeout)
|
||||||
|
return (out or b"").decode("utf-8", "replace")
|
||||||
|
except Exception:
|
||||||
|
return ""
|
||||||
|
|
||||||
|
|
||||||
|
class NodeStats:
|
||||||
|
def __init__(self, instance_id: str, node_name: str, history_path: str,
|
||||||
|
token_getter=None, logger=None):
|
||||||
|
self.instance_id = instance_id
|
||||||
|
self.node_name = node_name
|
||||||
|
self.history_path = history_path
|
||||||
|
self.token_getter = token_getter # callable -> kumulative Token-Zahl (oder None)
|
||||||
|
self.log = logger
|
||||||
|
self.samples = self._load()
|
||||||
|
self._last_tokens = self._tokens_now()
|
||||||
|
self._stream_task = None
|
||||||
|
|
||||||
|
# ── Persistenz ──────────────────────────────────────────
|
||||||
|
def _load(self) -> list:
|
||||||
|
try:
|
||||||
|
with open(self.history_path) as f:
|
||||||
|
data = json.load(f)
|
||||||
|
return data if isinstance(data, list) else []
|
||||||
|
except Exception:
|
||||||
|
return []
|
||||||
|
|
||||||
|
def _persist(self) -> None:
|
||||||
|
try:
|
||||||
|
os.makedirs(os.path.dirname(self.history_path) or ".", exist_ok=True)
|
||||||
|
tmp = self.history_path + ".tmp"
|
||||||
|
with open(tmp, "w") as f:
|
||||||
|
json.dump(self.samples[-HISTORY_CAP:], f)
|
||||||
|
os.replace(tmp, self.history_path)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
def _tokens_now(self) -> int:
|
||||||
|
try:
|
||||||
|
return int(self.token_getter()) if self.token_getter else 0
|
||||||
|
except Exception:
|
||||||
|
return 0
|
||||||
|
|
||||||
|
# ── nvidia-smi ──────────────────────────────────────────
|
||||||
|
async def _query_gpu(self) -> dict:
|
||||||
|
"""Aggregierte GPU-Werte ueber alle sichtbaren Karten."""
|
||||||
|
out = await _run_cmd(
|
||||||
|
"nvidia-smi",
|
||||||
|
"--query-gpu=utilization.gpu,memory.used,memory.total",
|
||||||
|
"--format=csv,noheader,nounits")
|
||||||
|
utils, used, total = [], 0, 0
|
||||||
|
for line in out.strip().splitlines():
|
||||||
|
parts = [p.strip() for p in line.split(",")]
|
||||||
|
if len(parts) < 3:
|
||||||
|
continue
|
||||||
|
try:
|
||||||
|
utils.append(float(parts[0]))
|
||||||
|
used += float(parts[1])
|
||||||
|
total += float(parts[2])
|
||||||
|
except ValueError:
|
||||||
|
continue
|
||||||
|
gpu = round(sum(utils) / len(utils), 1) if utils else 0.0
|
||||||
|
return {"gpu": gpu, "memUsed": int(used), "memTotal": int(total)}
|
||||||
|
|
||||||
|
async def _nvidia_smi_text(self) -> str:
|
||||||
|
txt = await _run_cmd("nvidia-smi")
|
||||||
|
return txt or "nvidia-smi nicht verfuegbar"
|
||||||
|
|
||||||
|
# ── Sampler (Verlauf) ───────────────────────────────────
|
||||||
|
async def run_sampler(self) -> None:
|
||||||
|
while True:
|
||||||
|
try:
|
||||||
|
g = await self._query_gpu()
|
||||||
|
now_tok = self._tokens_now()
|
||||||
|
dtok = max(0, now_tok - self._last_tokens)
|
||||||
|
self._last_tokens = now_tok
|
||||||
|
self.samples.append({
|
||||||
|
"ts": int(time.time()),
|
||||||
|
"gpu": g["gpu"], "memUsed": g["memUsed"],
|
||||||
|
"memTotal": g["memTotal"], "tokens": dtok,
|
||||||
|
})
|
||||||
|
if len(self.samples) > HISTORY_CAP:
|
||||||
|
self.samples = self.samples[-HISTORY_CAP:]
|
||||||
|
self._persist()
|
||||||
|
except Exception as e:
|
||||||
|
if self.log:
|
||||||
|
self.log.debug("node_stats sample fehlgeschlagen: %s", e)
|
||||||
|
await asyncio.sleep(SAMPLE_SEC)
|
||||||
|
|
||||||
|
# ── Live-Stream ─────────────────────────────────────────
|
||||||
|
async def _stream(self, ws, send) -> None:
|
||||||
|
t0 = time.time()
|
||||||
|
try:
|
||||||
|
while time.time() - t0 < STREAM_MAX_SEC:
|
||||||
|
g = await self._query_gpu()
|
||||||
|
smi = await self._nvidia_smi_text()
|
||||||
|
await send(ws, "node_stats", {
|
||||||
|
"instanceId": self.instance_id, "node": self.node_name,
|
||||||
|
"nvidiaSmi": smi, **g,
|
||||||
|
})
|
||||||
|
await asyncio.sleep(1)
|
||||||
|
except asyncio.CancelledError:
|
||||||
|
raise
|
||||||
|
except Exception:
|
||||||
|
return
|
||||||
|
|
||||||
|
# ── Dispatch ────────────────────────────────────────────
|
||||||
|
async def handle(self, ws, mtype: str, payload: dict, send) -> bool:
|
||||||
|
if mtype == "node_stats_stream_start":
|
||||||
|
if self._stream_task and not self._stream_task.done():
|
||||||
|
self._stream_task.cancel()
|
||||||
|
self._stream_task = asyncio.create_task(self._stream(ws, send))
|
||||||
|
return True
|
||||||
|
if mtype == "node_stats_stream_stop":
|
||||||
|
if self._stream_task:
|
||||||
|
self._stream_task.cancel()
|
||||||
|
self._stream_task = None
|
||||||
|
return True
|
||||||
|
if mtype == "node_stats_history_request":
|
||||||
|
await send(ws, "node_stats_history", {
|
||||||
|
"instanceId": self.instance_id, "node": self.node_name,
|
||||||
|
"samples": self.samples[-HISTORY_CAP:],
|
||||||
|
"tokenCapable": self.token_getter is not None,
|
||||||
|
"sampleSec": SAMPLE_SEC,
|
||||||
|
})
|
||||||
|
return True
|
||||||
|
if mtype == "node_stats_reset":
|
||||||
|
self.samples = []
|
||||||
|
self._persist()
|
||||||
|
await send(ws, "node_stats_reset_done",
|
||||||
|
{"instanceId": self.instance_id, "node": self.node_name})
|
||||||
|
return True
|
||||||
|
return False
|
||||||
@@ -3,6 +3,7 @@ FROM python:3.11-slim
|
|||||||
WORKDIR /app
|
WORKDIR /app
|
||||||
COPY requirements.txt .
|
COPY requirements.txt .
|
||||||
RUN pip install --no-cache-dir -r requirements.txt
|
RUN pip install --no-cache-dir -r requirements.txt
|
||||||
|
COPY node_stats.py .
|
||||||
COPY adapter.py .
|
COPY adapter.py .
|
||||||
|
|
||||||
CMD ["python", "-u", "adapter.py"]
|
CMD ["python", "-u", "adapter.py"]
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
# Local-LLM-Adapter (Gamebox) — Plan B, Phase B0
|
# Local-LLM-Adapter (AI-Box) — Plan B, Phase B0
|
||||||
|
|
||||||
Bringt ein lokales, schnelles LLM (Qwen3 8B) auf die Gamebox und haengt es
|
Bringt ein lokales, schnelles LLM (Qwen3 8B) auf die AI-Box und haengt es
|
||||||
per RVS an ARIA — fuer die einfachen ~80 % der Turns (<1 s), waehrend Claude
|
per RVS an ARIA — fuer die einfachen ~80 % der Turns (<1 s), waehrend Claude
|
||||||
das Tiefen-Hirn bleibt. Siehe `docs/plan-local-llm-router.md` im Repo-Root.
|
das Tiefen-Hirn bleibt. Siehe `docs/plan-local-llm-router.md` im Repo-Root.
|
||||||
|
|
||||||
@@ -17,7 +17,7 @@ das Tiefen-Hirn bleibt. Siehe `docs/plan-local-llm-router.md` im Repo-Root.
|
|||||||
cached es unter `xtts/models/` (Bind-Mount → kein Re-Download bei Restart).
|
cached es unter `xtts/models/` (Bind-Mount → kein Re-Download bei Restart).
|
||||||
Default: **Qwen3 8B, Q4_K_M** aus dem offiziellen Repo `Qwen/Qwen3-8B-GGUF`.
|
Default: **Qwen3 8B, Q4_K_M** aus dem offiziellen Repo `Qwen/Qwen3-8B-GGUF`.
|
||||||
|
|
||||||
Modell/Quant wechseln = in der `.env` der Gamebox setzen (kein Code):
|
Modell/Quant wechseln = in der `.env` der AI-Box setzen (kein Code):
|
||||||
|
|
||||||
```
|
```
|
||||||
LLM_HF_REPO=Qwen/Qwen3-8B-GGUF # HF-Repo
|
LLM_HF_REPO=Qwen/Qwen3-8B-GGUF # HF-Repo
|
||||||
@@ -35,7 +35,7 @@ umstellen + Container neu — Ein-Zeilen-Wechsel, kein Code.
|
|||||||
Modelle) ist ein geplanter Folge-Baustein via `llama-swap` — siehe
|
Modelle) ist ein geplanter Folge-Baustein via `llama-swap` — siehe
|
||||||
`docs/plan-local-llm-router.md`.
|
`docs/plan-local-llm-router.md`.
|
||||||
|
|
||||||
## Start (auf der Gamebox)
|
## Start (auf der AI-Box)
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
cd xtts
|
cd xtts
|
||||||
|
|||||||
+260
-10
@@ -1,5 +1,5 @@
|
|||||||
"""
|
"""
|
||||||
ARIA Local-LLM-Adapter (Gamebox) — Plan B, Phase B0.
|
ARIA Local-LLM-Adapter (AI-Box) — Plan B, Phase B0.
|
||||||
|
|
||||||
Bruecke zwischen RVS und dem lokalen llama.cpp-Server. Spiegelt das Muster der
|
Bruecke zwischen RVS und dem lokalen llama.cpp-Server. Spiegelt das Muster der
|
||||||
whisper-bridge: verbindet sich per WebSocket mit dem RVS (Token-Room, TLS mit
|
whisper-bridge: verbindet sich per WebSocket mit dem RVS (Token-Room, TLS mit
|
||||||
@@ -7,7 +7,7 @@ ws-Fallback, Reconnect-Backoff), lauscht auf `llm_request` und ruft den lokalen
|
|||||||
llama.cpp-`/v1/chat/completions`-Endpoint (OpenAI-kompatibel), antwortet mit
|
llama.cpp-`/v1/chat/completions`-Endpoint (OpenAI-kompatibel), antwortet mit
|
||||||
`llm_response` (korreliert per requestId).
|
`llm_response` (korreliert per requestId).
|
||||||
|
|
||||||
Topologie: Gamebox steht zuhause, ARIA im RZ — die Kommunikation laeuft ueber
|
Topologie: AI-Box steht zuhause, ARIA im RZ — die Kommunikation laeuft ueber
|
||||||
den RVS (wie TTS/STT), keine IPs zu pflegen. Nur URL + Token.
|
den RVS (wie TTS/STT), keine IPs zu pflegen. Nur URL + Token.
|
||||||
|
|
||||||
Env:
|
Env:
|
||||||
@@ -47,6 +47,19 @@ RVS_TOKEN = os.getenv("RVS_TOKEN", "").strip()
|
|||||||
LLAMA_URL = os.getenv("LLAMA_URL", "http://llama:8081").rstrip("/")
|
LLAMA_URL = os.getenv("LLAMA_URL", "http://llama:8081").rstrip("/")
|
||||||
LLM_MODEL = os.getenv("LLM_MODEL", "qwen3-8b")
|
LLM_MODEL = os.getenv("LLM_MODEL", "qwen3-8b")
|
||||||
LLM_TIMEOUT_SEC = float(os.getenv("LLM_TIMEOUT_SEC", "60"))
|
LLM_TIMEOUT_SEC = float(os.getenv("LLM_TIMEOUT_SEC", "60"))
|
||||||
|
|
||||||
|
# ── Compute-Fleet: Worker-Identitaet & Registrierung ──────────────
|
||||||
|
# Meldet sich bei der aria-bridge (worker_hello) + periodischer worker_ping.
|
||||||
|
NODE_NAME = os.getenv("NODE_NAME", "node").strip() or "node"
|
||||||
|
GPU_IDS = os.getenv("NVIDIA_VISIBLE_DEVICES", "").strip()
|
||||||
|
WORKER_SERVICE = "llm"
|
||||||
|
INSTANCE_ID = f"{WORKER_SERVICE}@{NODE_NAME}"
|
||||||
|
WORKER_PING_INTERVAL_S = int(os.getenv("WORKER_PING_INTERVAL_S", "10"))
|
||||||
|
# Empfangs-Watchdog: kommt in RX_STALE_S kein Broadcast rein (ein echter Raum hat
|
||||||
|
# staendig Traffic, z.B. sat_hello alle 25s / Brain-Polling), gilt die Verbindung
|
||||||
|
# als halb-tot (Caddy pongt die WS-Pings selbst) -> Zwangs-Reconnect.
|
||||||
|
RX_STALE_S = int(os.getenv("RX_STALE_S", "60"))
|
||||||
|
_inflight = 0 # laufende llm_requests (busy-Report im ping)
|
||||||
# Qwen3 hat Thinking-Mode default AN — dann verbraet es Tokens in einem
|
# Qwen3 hat Thinking-Mode default AN — dann verbraet es Tokens in einem
|
||||||
# <think>-Block und liefert (bei kleinem max_tokens) leeren/abgeschnittenen
|
# <think>-Block und liefert (bei kleinem max_tokens) leeren/abgeschnittenen
|
||||||
# content, ausserdem 3x langsamer. ARIAs schnelles Tier will KEIN Grübeln
|
# content, ausserdem 3x langsamer. ARIAs schnelles Tier will KEIN Grübeln
|
||||||
@@ -56,6 +69,94 @@ LLM_TIMEOUT_SEC = float(os.getenv("LLM_TIMEOUT_SEC", "60"))
|
|||||||
# empfindlich reagiert: LLM_DISABLE_THINKING=false setzen.
|
# empfindlich reagiert: LLM_DISABLE_THINKING=false setzen.
|
||||||
LLM_DISABLE_THINKING = os.getenv("LLM_DISABLE_THINKING", "true").lower() == "true"
|
LLM_DISABLE_THINKING = os.getenv("LLM_DISABLE_THINKING", "true").lower() == "true"
|
||||||
|
|
||||||
|
# ── Modell-Verwaltung (Stage D): Adapter besitzt llama-swaps Config ──
|
||||||
|
# llama-swap liest die GENERIERTE Config (beschreibbar, im /models-Bind). Wir
|
||||||
|
# erzeugen sie aus dem Basis-Template (kuratierte Defaults) + der persistenten
|
||||||
|
# Box-Registry (per Diagnostic hinzugefuegte Modelle). So werden neue Modelle
|
||||||
|
# ohne Image-Rebuild waehlbar.
|
||||||
|
import yaml # pyyaml
|
||||||
|
BASE_CONFIG_PATH = os.getenv("LLAMA_BASE_CONFIG", "/llamaswap/config.yaml")
|
||||||
|
GEN_CONFIG_PATH = os.getenv("LLAMA_GEN_CONFIG", "/models/llama-swap.config.yaml")
|
||||||
|
REGISTRY_PATH = os.getenv("LLM_REGISTRY", "/models/aria_models.json")
|
||||||
|
|
||||||
|
# ── Auslastungs-Monitor (Stage E) ──────────────────────────
|
||||||
|
import node_stats
|
||||||
|
STATS_PATH = os.getenv("STATS_PATH", "/models/aria_stats.json")
|
||||||
|
_total_tokens = 0 # kumulativ, fuer den Token-Graph
|
||||||
|
_stats = node_stats.NodeStats(INSTANCE_ID, NODE_NAME, STATS_PATH,
|
||||||
|
token_getter=lambda: _total_tokens, logger=logger)
|
||||||
|
|
||||||
|
|
||||||
|
def _load_registry() -> list:
|
||||||
|
try:
|
||||||
|
with open(REGISTRY_PATH) as f:
|
||||||
|
data = json.load(f)
|
||||||
|
return data if isinstance(data, list) else []
|
||||||
|
except Exception:
|
||||||
|
return []
|
||||||
|
|
||||||
|
|
||||||
|
def _save_registry(reg: list) -> None:
|
||||||
|
try:
|
||||||
|
tmp = REGISTRY_PATH + ".tmp"
|
||||||
|
with open(tmp, "w") as f:
|
||||||
|
json.dump(reg, f, indent=2)
|
||||||
|
os.replace(tmp, REGISTRY_PATH)
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning("Registry speichern fehlgeschlagen: %s", e)
|
||||||
|
|
||||||
|
|
||||||
|
def _generate_config() -> int:
|
||||||
|
"""Schreibt die llama-swap-Config aus Basis-Template + Registry. Gibt die
|
||||||
|
Anzahl Modelle zurueck. Idempotent, bei jeder Aenderung + beim Start."""
|
||||||
|
base = {}
|
||||||
|
try:
|
||||||
|
with open(BASE_CONFIG_PATH) as f:
|
||||||
|
base = yaml.safe_load(f) or {}
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning("Basis-Template %s nicht lesbar (%s)", BASE_CONFIG_PATH, e)
|
||||||
|
models = dict(base.get("models") or {})
|
||||||
|
for e in _load_registry():
|
||||||
|
key = (e.get("key") or "").strip()
|
||||||
|
repo = (e.get("hfRepo") or "").strip()
|
||||||
|
if not key or not repo:
|
||||||
|
continue
|
||||||
|
quant = (e.get("quant") or "Q4_K_M").strip()
|
||||||
|
ctx = int(e.get("ctx") or 8192)
|
||||||
|
ngl = int(e.get("ngl") or 99)
|
||||||
|
models[key] = {
|
||||||
|
"cmd": (f"llama-server --port ${{PORT}} --host 127.0.0.1\n"
|
||||||
|
f"-hf {repo}:{quant}\n-ngl {ngl} -c {ctx} --jinja"),
|
||||||
|
"ttl": 3600,
|
||||||
|
}
|
||||||
|
out = dict(base)
|
||||||
|
out["models"] = models
|
||||||
|
try:
|
||||||
|
os.makedirs(os.path.dirname(GEN_CONFIG_PATH), exist_ok=True)
|
||||||
|
tmp = GEN_CONFIG_PATH + ".tmp"
|
||||||
|
with open(tmp, "w") as f:
|
||||||
|
yaml.safe_dump(out, f, sort_keys=False, default_flow_style=False)
|
||||||
|
os.replace(tmp, GEN_CONFIG_PATH)
|
||||||
|
logger.info("llama-swap-Config generiert: %d Modelle → %s", len(models), GEN_CONFIG_PATH)
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("Config schreiben fehlgeschlagen: %s", e)
|
||||||
|
return len(models)
|
||||||
|
|
||||||
|
|
||||||
|
async def _reload_llama() -> None:
|
||||||
|
"""Stoesst llama-swap-Reload an. Viele Builds watchen die Config-Datei ohnehin;
|
||||||
|
zusaetzlich versuchen wir bekannte Reload-Endpunkte (Fehler ignoriert)."""
|
||||||
|
for path in ("/api/config/reload", "/reload"):
|
||||||
|
try:
|
||||||
|
async with httpx.AsyncClient(timeout=10) as c:
|
||||||
|
r = await c.post(f"{LLAMA_URL}{path}")
|
||||||
|
if r.status_code < 400:
|
||||||
|
logger.info("llama-swap reload via %s", path)
|
||||||
|
return
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
logger.info("llama-swap reload: kein Endpoint — verlasse mich auf File-Watch")
|
||||||
|
|
||||||
|
|
||||||
async def _send(ws, mtype: str, payload: dict) -> None:
|
async def _send(ws, mtype: str, payload: dict) -> None:
|
||||||
try:
|
try:
|
||||||
@@ -99,11 +200,17 @@ async def _call_llama(messages: list, *, max_tokens: int, temperature: float,
|
|||||||
r.raise_for_status()
|
r.raise_for_status()
|
||||||
data = r.json()
|
data = r.json()
|
||||||
msg = (data.get("choices") or [{}])[0].get("message", {}) or {}
|
msg = (data.get("choices") or [{}])[0].get("message", {}) or {}
|
||||||
|
usage = data.get("usage") or {}
|
||||||
|
try:
|
||||||
|
global _total_tokens
|
||||||
|
_total_tokens += int(usage.get("total_tokens") or 0)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
return {
|
return {
|
||||||
"ok": True,
|
"ok": True,
|
||||||
"content": msg.get("content") or "",
|
"content": msg.get("content") or "",
|
||||||
"tool_calls": msg.get("tool_calls") or None,
|
"tool_calls": msg.get("tool_calls") or None,
|
||||||
"usage": data.get("usage"),
|
"usage": usage,
|
||||||
}
|
}
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.warning("llama.cpp-Call fehlgeschlagen: %s", e)
|
logger.warning("llama.cpp-Call fehlgeschlagen: %s", e)
|
||||||
@@ -124,7 +231,63 @@ async def _emit_llm_status(ws, state: str, model: str, **extra) -> None:
|
|||||||
{"service": "llm", "state": state, "model": model, **extra})
|
{"service": "llm", "state": state, "model": model, **extra})
|
||||||
|
|
||||||
|
|
||||||
|
async def _fetch_available_models() -> list:
|
||||||
|
"""Fragt llama-swap ab, welche Modelle diese Box fahren kann (GET /v1/models,
|
||||||
|
OpenAI-kompatibel → {data:[{id},...]}). Das sind die config.yaml-Keys.
|
||||||
|
Defensiv: bei Fehler Fallback auf [LLM_MODEL]."""
|
||||||
|
try:
|
||||||
|
async with httpx.AsyncClient(timeout=10) as client:
|
||||||
|
r = await client.get(f"{LLAMA_URL}/v1/models")
|
||||||
|
r.raise_for_status()
|
||||||
|
data = r.json()
|
||||||
|
ids = [m.get("id") for m in (data.get("data") or []) if m.get("id")]
|
||||||
|
return ids or [LLM_MODEL]
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning("llama-swap /v1/models nicht abfragbar (%s) — Fallback [%s]", e, LLM_MODEL)
|
||||||
|
return [LLM_MODEL]
|
||||||
|
|
||||||
|
|
||||||
|
async def _announce(ws) -> None:
|
||||||
|
"""Sendet ein frisches worker_hello mit der aktuellen Modell-Liste (nach
|
||||||
|
Provision/Remove aufrufen, damit Bridge+Diagnostic das neue Modell lernen)."""
|
||||||
|
models = await _fetch_available_models()
|
||||||
|
await _send(ws, "worker_hello", {
|
||||||
|
"instanceId": INSTANCE_ID, "service": WORKER_SERVICE,
|
||||||
|
"node": NODE_NAME, "gpus": GPU_IDS, "model": LLM_MODEL,
|
||||||
|
"models": models, # welche Modelle diese Box fahren kann (llama-swap-Keys)
|
||||||
|
})
|
||||||
|
logger.info("worker_hello: models=%s", models)
|
||||||
|
|
||||||
|
|
||||||
|
async def _worker_register(ws) -> None:
|
||||||
|
"""Meldet diesen Worker bei der aria-bridge an (worker_hello) und haelt die
|
||||||
|
Flotten-Registry per periodischem worker_ping (mit busy-Status) frisch."""
|
||||||
|
try:
|
||||||
|
await _announce(ws)
|
||||||
|
n = 0
|
||||||
|
while True:
|
||||||
|
await asyncio.sleep(WORKER_PING_INTERVAL_S)
|
||||||
|
n += 1
|
||||||
|
await _send(ws, "worker_ping",
|
||||||
|
{"instanceId": INSTANCE_ID, "busy": _inflight > 0})
|
||||||
|
if n % 3 == 0: # ~30s worker_hello wiederholen (wie der Satellit) →
|
||||||
|
await _announce(ws) # auch neu gestartetes Diagnostic/Bridge lernt uns
|
||||||
|
except asyncio.CancelledError:
|
||||||
|
raise
|
||||||
|
except Exception:
|
||||||
|
return # Socket tot → still beenden; _run reconnectet + startet neu
|
||||||
|
|
||||||
|
|
||||||
async def _handle_llm_request(ws, payload: dict) -> None:
|
async def _handle_llm_request(ws, payload: dict) -> None:
|
||||||
|
global _last_model, _inflight
|
||||||
|
_inflight += 1
|
||||||
|
try:
|
||||||
|
await _do_llm_request(ws, payload)
|
||||||
|
finally:
|
||||||
|
_inflight -= 1
|
||||||
|
|
||||||
|
|
||||||
|
async def _do_llm_request(ws, payload: dict) -> None:
|
||||||
global _last_model
|
global _last_model
|
||||||
req_id = payload.get("requestId", "")
|
req_id = payload.get("requestId", "")
|
||||||
messages = payload.get("messages") or []
|
messages = payload.get("messages") or []
|
||||||
@@ -177,6 +340,61 @@ async def _handle_llm_request(ws, payload: dict) -> None:
|
|||||||
})
|
})
|
||||||
|
|
||||||
|
|
||||||
|
async def _handle_provision(ws, payload: dict) -> None:
|
||||||
|
"""Fuegt ein Modell hinzu: Registry+Config schreiben, reload, dann Warmup
|
||||||
|
(zieht das GGUF via -hf beim ersten Load). Meldet die neue Modell-Liste."""
|
||||||
|
key = (payload.get("key") or "").strip()
|
||||||
|
repo = (payload.get("hfRepo") or "").strip()
|
||||||
|
if not key or not repo:
|
||||||
|
await _send(ws, "llm_provision_result",
|
||||||
|
{"instanceId": INSTANCE_ID, "key": key, "ok": False, "error": "key/hfRepo fehlt"})
|
||||||
|
return
|
||||||
|
entry = {
|
||||||
|
"key": key, "hfRepo": repo,
|
||||||
|
"quant": (payload.get("quant") or "Q4_K_M").strip(),
|
||||||
|
"ctx": int(payload.get("ctx") or 8192),
|
||||||
|
"ngl": int(payload.get("ngl") or 99),
|
||||||
|
}
|
||||||
|
reg = [e for e in _load_registry() if e.get("key") != key]
|
||||||
|
reg.append(entry)
|
||||||
|
_save_registry(reg)
|
||||||
|
_generate_config()
|
||||||
|
await _reload_llama()
|
||||||
|
await _announce(ws) # Bridge/Diagnostic lernen das neue Modell
|
||||||
|
# Warmup: Mini-Request → llama-swap laedt/zieht das Modell (Fortschritt via
|
||||||
|
# service_status loading→ready, freshlyDownloaded).
|
||||||
|
await _emit_llm_status(ws, "loading", key)
|
||||||
|
t0 = time.time()
|
||||||
|
res = await _call_llama([{"role": "user", "content": "hi"}],
|
||||||
|
max_tokens=1, temperature=0.0, stop=None, model=key)
|
||||||
|
dt = time.time() - t0
|
||||||
|
if res.get("ok"):
|
||||||
|
_ready_models.add(key)
|
||||||
|
await _emit_llm_status(ws, "ready", key, loadSeconds=round(dt, 1),
|
||||||
|
freshlyDownloaded=dt > 25)
|
||||||
|
else:
|
||||||
|
await _emit_llm_status(ws, "error", key, error=(res.get("error") or "")[:160])
|
||||||
|
await _send(ws, "llm_provision_result",
|
||||||
|
{"instanceId": INSTANCE_ID, "key": key, "ok": res.get("ok", False),
|
||||||
|
"error": res.get("error"), "elapsedMs": int(dt * 1000)})
|
||||||
|
logger.info("provision %s (%s) → ok=%s %.1fs", key, repo, res.get("ok"), dt)
|
||||||
|
|
||||||
|
|
||||||
|
async def _handle_remove(ws, payload: dict) -> None:
|
||||||
|
"""Entfernt ein Modell aus Registry+Config (GGUF bleibt im Cache)."""
|
||||||
|
key = (payload.get("key") or "").strip()
|
||||||
|
if not key:
|
||||||
|
return
|
||||||
|
reg = [e for e in _load_registry() if e.get("key") != key]
|
||||||
|
_save_registry(reg)
|
||||||
|
_generate_config()
|
||||||
|
await _reload_llama()
|
||||||
|
await _announce(ws)
|
||||||
|
await _send(ws, "llm_provision_result",
|
||||||
|
{"instanceId": INSTANCE_ID, "key": key, "ok": True, "removed": True})
|
||||||
|
logger.info("removed model %s", key)
|
||||||
|
|
||||||
|
|
||||||
async def _run() -> None:
|
async def _run() -> None:
|
||||||
if not RVS_HOST:
|
if not RVS_HOST:
|
||||||
logger.error("RVS_HOST nicht gesetzt — Abbruch")
|
logger.error("RVS_HOST nicht gesetzt — Abbruch")
|
||||||
@@ -185,13 +403,21 @@ async def _run() -> None:
|
|||||||
logger.error("RVS_TOKEN nicht gesetzt — Abbruch")
|
logger.error("RVS_TOKEN nicht gesetzt — Abbruch")
|
||||||
return
|
return
|
||||||
|
|
||||||
|
# llama-swap-Config aus Basis-Template + Registry erzeugen, BEVOR llama-swap
|
||||||
|
# sie braucht (llama-swap restart: unless-stopped faengt die Erst-Boot-
|
||||||
|
# Reihenfolge ab, falls es kurz vor uns startet).
|
||||||
|
_generate_config()
|
||||||
|
|
||||||
|
# Auslastungs-Sampler (GPU + Tokens) laeuft unabhaengig vom RVS.
|
||||||
|
asyncio.create_task(_stats.run_sampler())
|
||||||
|
|
||||||
use_tls = RVS_TLS
|
use_tls = RVS_TLS
|
||||||
retry_s = 2
|
retry_s = 2
|
||||||
tls_fallback_tried = False
|
tls_fallback_tried = False
|
||||||
|
|
||||||
while True:
|
while True:
|
||||||
scheme = "wss" if use_tls else "ws"
|
scheme = "wss" if use_tls else "ws"
|
||||||
url = f"{scheme}://{RVS_HOST}:{RVS_PORT}/ws?token={RVS_TOKEN}"
|
url = f"{scheme}://{RVS_HOST}:{RVS_PORT}?token={RVS_TOKEN}"
|
||||||
masked = url.replace(RVS_TOKEN, "***") if RVS_TOKEN else url
|
masked = url.replace(RVS_TOKEN, "***") if RVS_TOKEN else url
|
||||||
try:
|
try:
|
||||||
logger.info("Verbinde zu RVS: %s (llama=%s)", masked, LLAMA_URL)
|
logger.info("Verbinde zu RVS: %s (llama=%s)", masked, LLAMA_URL)
|
||||||
@@ -201,19 +427,43 @@ async def _run() -> None:
|
|||||||
logger.info("RVS verbunden — llm-adapter online")
|
logger.info("RVS verbunden — llm-adapter online")
|
||||||
retry_s = 2
|
retry_s = 2
|
||||||
tls_fallback_tried = False
|
tls_fallback_tried = False
|
||||||
async for raw in ws:
|
ping_task = asyncio.create_task(_worker_register(ws))
|
||||||
|
while True:
|
||||||
|
try:
|
||||||
|
raw = await asyncio.wait_for(ws.recv(), timeout=RX_STALE_S)
|
||||||
|
except asyncio.TimeoutError:
|
||||||
|
logger.warning("Kein RVS-Traffic seit %ds — Verbindung halb-tot, reconnect", RX_STALE_S)
|
||||||
|
raise ConnectionError("rvs-stale")
|
||||||
try:
|
try:
|
||||||
msg = json.loads(raw)
|
msg = json.loads(raw)
|
||||||
except Exception:
|
except Exception:
|
||||||
continue
|
continue
|
||||||
if msg.get("type") != "llm_request":
|
mtype = msg.get("type")
|
||||||
continue
|
|
||||||
payload = msg.get("payload", {}) or {}
|
payload = msg.get("payload", {}) or {}
|
||||||
# Jede Anfrage nebenlaeufig — llama.cpp serialisiert intern,
|
# Redundanz-Routing: gezielt an eine andere Instanz adressiert
|
||||||
# aber wir blockieren so nicht den Empfang weiterer Messages.
|
# → ignorieren. Ohne targetInstance → wie bisher (jeder nimmt).
|
||||||
asyncio.create_task(_handle_llm_request(ws, payload))
|
tgt = payload.get("targetInstance")
|
||||||
|
if tgt and tgt != INSTANCE_ID:
|
||||||
|
continue
|
||||||
|
# Auslastungs-Monitor (node_stats_*) abfangen.
|
||||||
|
if await _stats.handle(ws, mtype, payload, _send):
|
||||||
|
continue
|
||||||
|
if mtype not in ("llm_request", "llm_provision_model", "llm_remove_model"):
|
||||||
|
continue
|
||||||
|
if mtype == "llm_provision_model":
|
||||||
|
asyncio.create_task(_handle_provision(ws, payload))
|
||||||
|
elif mtype == "llm_remove_model":
|
||||||
|
asyncio.create_task(_handle_remove(ws, payload))
|
||||||
|
else:
|
||||||
|
# Jede Anfrage nebenlaeufig — llama.cpp serialisiert intern,
|
||||||
|
# aber wir blockieren so nicht den Empfang weiterer Messages.
|
||||||
|
asyncio.create_task(_handle_llm_request(ws, payload))
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.warning("RVS-Verbindung verloren/fehlgeschlagen: %s", e)
|
logger.warning("RVS-Verbindung verloren/fehlgeschlagen: %s", e)
|
||||||
|
try:
|
||||||
|
ping_task.cancel()
|
||||||
|
except NameError:
|
||||||
|
pass
|
||||||
if use_tls and RVS_TLS_FALLBACK and not tls_fallback_tried:
|
if use_tls and RVS_TLS_FALLBACK and not tls_fallback_tried:
|
||||||
tls_fallback_tried = True
|
tls_fallback_tried = True
|
||||||
use_tls = False
|
use_tls = False
|
||||||
|
|||||||
@@ -0,0 +1,167 @@
|
|||||||
|
"""
|
||||||
|
ARIA Node-Stats — Auslastungs-Monitor pro Box (GPU + optional Tokens).
|
||||||
|
|
||||||
|
Identische Kopie in jedem Worker-Build-Context (f5tts/whisper/voxtral/llm-adapter),
|
||||||
|
weil jeder Worker ein eigener Docker-Build-Context ist.
|
||||||
|
|
||||||
|
Aufgaben:
|
||||||
|
- Sampler-Loop (alle SAMPLE_SEC): nvidia-smi-Auslastung + Token-Delta → Ringpuffer
|
||||||
|
(persistent als JSON auf der Box). Laeuft unabhaengig vom Modal.
|
||||||
|
- Live-Stream: bei node_stats_stream_start jede Sekunde rohes nvidia-smi + Werte
|
||||||
|
senden (bis stop / Auto-Timeout).
|
||||||
|
- History-Request + Reset (Besen).
|
||||||
|
|
||||||
|
Reicht `handle(ws, mtype, payload)` in die Worker-Message-Loop ein; gibt True
|
||||||
|
zurueck, wenn die Nachricht eine node_stats_*-Nachricht war.
|
||||||
|
"""
|
||||||
|
import asyncio
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import time
|
||||||
|
|
||||||
|
SAMPLE_SEC = int(os.getenv("STATS_SAMPLE_SEC", "15"))
|
||||||
|
HISTORY_CAP = int(os.getenv("STATS_HISTORY_CAP", "500")) # ~2h bei 15s
|
||||||
|
STREAM_MAX_SEC = int(os.getenv("STATS_STREAM_MAX_SEC", "300"))
|
||||||
|
|
||||||
|
|
||||||
|
async def _run_cmd(*args, timeout=8) -> str:
|
||||||
|
"""Fuehrt ein Kommando aus, gibt stdout (str) zurueck; '' bei Fehler."""
|
||||||
|
try:
|
||||||
|
proc = await asyncio.create_subprocess_exec(
|
||||||
|
*args,
|
||||||
|
stdout=asyncio.subprocess.PIPE,
|
||||||
|
stderr=asyncio.subprocess.DEVNULL,
|
||||||
|
)
|
||||||
|
out, _ = await asyncio.wait_for(proc.communicate(), timeout=timeout)
|
||||||
|
return (out or b"").decode("utf-8", "replace")
|
||||||
|
except Exception:
|
||||||
|
return ""
|
||||||
|
|
||||||
|
|
||||||
|
class NodeStats:
|
||||||
|
def __init__(self, instance_id: str, node_name: str, history_path: str,
|
||||||
|
token_getter=None, logger=None):
|
||||||
|
self.instance_id = instance_id
|
||||||
|
self.node_name = node_name
|
||||||
|
self.history_path = history_path
|
||||||
|
self.token_getter = token_getter # callable -> kumulative Token-Zahl (oder None)
|
||||||
|
self.log = logger
|
||||||
|
self.samples = self._load()
|
||||||
|
self._last_tokens = self._tokens_now()
|
||||||
|
self._stream_task = None
|
||||||
|
|
||||||
|
# ── Persistenz ──────────────────────────────────────────
|
||||||
|
def _load(self) -> list:
|
||||||
|
try:
|
||||||
|
with open(self.history_path) as f:
|
||||||
|
data = json.load(f)
|
||||||
|
return data if isinstance(data, list) else []
|
||||||
|
except Exception:
|
||||||
|
return []
|
||||||
|
|
||||||
|
def _persist(self) -> None:
|
||||||
|
try:
|
||||||
|
os.makedirs(os.path.dirname(self.history_path) or ".", exist_ok=True)
|
||||||
|
tmp = self.history_path + ".tmp"
|
||||||
|
with open(tmp, "w") as f:
|
||||||
|
json.dump(self.samples[-HISTORY_CAP:], f)
|
||||||
|
os.replace(tmp, self.history_path)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
def _tokens_now(self) -> int:
|
||||||
|
try:
|
||||||
|
return int(self.token_getter()) if self.token_getter else 0
|
||||||
|
except Exception:
|
||||||
|
return 0
|
||||||
|
|
||||||
|
# ── nvidia-smi ──────────────────────────────────────────
|
||||||
|
async def _query_gpu(self) -> dict:
|
||||||
|
"""Aggregierte GPU-Werte ueber alle sichtbaren Karten."""
|
||||||
|
out = await _run_cmd(
|
||||||
|
"nvidia-smi",
|
||||||
|
"--query-gpu=utilization.gpu,memory.used,memory.total",
|
||||||
|
"--format=csv,noheader,nounits")
|
||||||
|
utils, used, total = [], 0, 0
|
||||||
|
for line in out.strip().splitlines():
|
||||||
|
parts = [p.strip() for p in line.split(",")]
|
||||||
|
if len(parts) < 3:
|
||||||
|
continue
|
||||||
|
try:
|
||||||
|
utils.append(float(parts[0]))
|
||||||
|
used += float(parts[1])
|
||||||
|
total += float(parts[2])
|
||||||
|
except ValueError:
|
||||||
|
continue
|
||||||
|
gpu = round(sum(utils) / len(utils), 1) if utils else 0.0
|
||||||
|
return {"gpu": gpu, "memUsed": int(used), "memTotal": int(total)}
|
||||||
|
|
||||||
|
async def _nvidia_smi_text(self) -> str:
|
||||||
|
txt = await _run_cmd("nvidia-smi")
|
||||||
|
return txt or "nvidia-smi nicht verfuegbar"
|
||||||
|
|
||||||
|
# ── Sampler (Verlauf) ───────────────────────────────────
|
||||||
|
async def run_sampler(self) -> None:
|
||||||
|
while True:
|
||||||
|
try:
|
||||||
|
g = await self._query_gpu()
|
||||||
|
now_tok = self._tokens_now()
|
||||||
|
dtok = max(0, now_tok - self._last_tokens)
|
||||||
|
self._last_tokens = now_tok
|
||||||
|
self.samples.append({
|
||||||
|
"ts": int(time.time()),
|
||||||
|
"gpu": g["gpu"], "memUsed": g["memUsed"],
|
||||||
|
"memTotal": g["memTotal"], "tokens": dtok,
|
||||||
|
})
|
||||||
|
if len(self.samples) > HISTORY_CAP:
|
||||||
|
self.samples = self.samples[-HISTORY_CAP:]
|
||||||
|
self._persist()
|
||||||
|
except Exception as e:
|
||||||
|
if self.log:
|
||||||
|
self.log.debug("node_stats sample fehlgeschlagen: %s", e)
|
||||||
|
await asyncio.sleep(SAMPLE_SEC)
|
||||||
|
|
||||||
|
# ── Live-Stream ─────────────────────────────────────────
|
||||||
|
async def _stream(self, ws, send) -> None:
|
||||||
|
t0 = time.time()
|
||||||
|
try:
|
||||||
|
while time.time() - t0 < STREAM_MAX_SEC:
|
||||||
|
g = await self._query_gpu()
|
||||||
|
smi = await self._nvidia_smi_text()
|
||||||
|
await send(ws, "node_stats", {
|
||||||
|
"instanceId": self.instance_id, "node": self.node_name,
|
||||||
|
"nvidiaSmi": smi, **g,
|
||||||
|
})
|
||||||
|
await asyncio.sleep(1)
|
||||||
|
except asyncio.CancelledError:
|
||||||
|
raise
|
||||||
|
except Exception:
|
||||||
|
return
|
||||||
|
|
||||||
|
# ── Dispatch ────────────────────────────────────────────
|
||||||
|
async def handle(self, ws, mtype: str, payload: dict, send) -> bool:
|
||||||
|
if mtype == "node_stats_stream_start":
|
||||||
|
if self._stream_task and not self._stream_task.done():
|
||||||
|
self._stream_task.cancel()
|
||||||
|
self._stream_task = asyncio.create_task(self._stream(ws, send))
|
||||||
|
return True
|
||||||
|
if mtype == "node_stats_stream_stop":
|
||||||
|
if self._stream_task:
|
||||||
|
self._stream_task.cancel()
|
||||||
|
self._stream_task = None
|
||||||
|
return True
|
||||||
|
if mtype == "node_stats_history_request":
|
||||||
|
await send(ws, "node_stats_history", {
|
||||||
|
"instanceId": self.instance_id, "node": self.node_name,
|
||||||
|
"samples": self.samples[-HISTORY_CAP:],
|
||||||
|
"tokenCapable": self.token_getter is not None,
|
||||||
|
"sampleSec": SAMPLE_SEC,
|
||||||
|
})
|
||||||
|
return True
|
||||||
|
if mtype == "node_stats_reset":
|
||||||
|
self.samples = []
|
||||||
|
self._persist()
|
||||||
|
await send(ws, "node_stats_reset_done",
|
||||||
|
{"instanceId": self.instance_id, "node": self.node_name})
|
||||||
|
return True
|
||||||
|
return False
|
||||||
@@ -1,2 +1,3 @@
|
|||||||
websockets>=12.0
|
websockets>=12.0
|
||||||
httpx>=0.27.0
|
httpx>=0.27.0
|
||||||
|
pyyaml>=6.0
|
||||||
|
|||||||
@@ -0,0 +1,26 @@
|
|||||||
|
# Voxtral-STT-3B Bridge (Transformers). Laeuft auf Treiber 550/CUDA 12.4 via
|
||||||
|
# torch cu124 — KEIN Treiber-Upgrade noetig (gleicher Trick wie f5tts).
|
||||||
|
# Modell: Voxtral-Mini-3B-2507 (~9 GB in bf16) → passt auf die 12-GB-Karte (GPU 1).
|
||||||
|
FROM nvidia/cuda:12.2.2-cudnn8-runtime-ubuntu22.04
|
||||||
|
|
||||||
|
ENV DEBIAN_FRONTEND=noninteractive
|
||||||
|
ENV PYTHONUNBUFFERED=1
|
||||||
|
WORKDIR /app
|
||||||
|
|
||||||
|
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||||
|
python3 python3-pip ffmpeg git \
|
||||||
|
&& rm -rf /var/lib/apt/lists/*
|
||||||
|
|
||||||
|
# torch FEST auf cu124 (Treiber 550 = CUDA 12.4). 2.6.0 ist der neueste cu124-Build;
|
||||||
|
# torch 2.7+ gibt es nur fuer cu126+ und braeuchte einen neueren Treiber.
|
||||||
|
RUN pip3 install --no-cache-dir torch==2.6.0 torchaudio==2.6.0 \
|
||||||
|
--index-url https://download.pytorch.org/whl/cu124
|
||||||
|
|
||||||
|
COPY requirements.txt .
|
||||||
|
# Constraint haelt transformers/mistral-common davon ab, torch wieder hochzuziehen.
|
||||||
|
RUN printf 'torch==2.6.0\ntorchaudio==2.6.0\n' > /tmp/torch-constraint.txt && \
|
||||||
|
pip3 install --no-cache-dir -c /tmp/torch-constraint.txt -r requirements.txt
|
||||||
|
|
||||||
|
COPY bridge.py speaker_id.py node_stats.py ./
|
||||||
|
|
||||||
|
CMD ["python3", "bridge.py"]
|
||||||
@@ -0,0 +1,47 @@
|
|||||||
|
# Voxtral-STT-3B-Satellit (Transformers)
|
||||||
|
|
||||||
|
Streaming-STT via **Voxtral-Mini-3B-2507** (Mistral, Apache 2.0) über
|
||||||
|
🤗 Transformers. Ersetzt whisper als STT — genauer, und **ohne Treiber-Upgrade**:
|
||||||
|
läuft auf dem Trixie-Standardtreiber (550/CUDA 12.4) via **torch 2.6.0+cu124**
|
||||||
|
(derselbe Trick wie bei f5tts).
|
||||||
|
|
||||||
|
- Modell ~9 GB (bf16) → **GPU 1** (die 12-GB-Karte; per Compose gepinnt).
|
||||||
|
- **F5-TTS + LLM** bleiben auf GPU 0 (8 GB).
|
||||||
|
- Arbeitsweise = chunked wie whisper: live PCM → alle ~1 s transkribieren
|
||||||
|
(Partials) → **adaptiver Endpointer** (Rausch-Boden-VAD + semantische
|
||||||
|
Stagnation, aus M0.1) feuert `stt_endpoint`. RVS-Protokoll identisch → drop-in.
|
||||||
|
|
||||||
|
## Starten (Profil `voxtral`)
|
||||||
|
|
||||||
|
```bash
|
||||||
|
cd xtts
|
||||||
|
docker compose stop whisper-bridge # sonst beantworten beide stt_*
|
||||||
|
docker compose --profile voxtral up -d --build
|
||||||
|
docker logs -f aria-voxtral-bridge # Modell laedt (mehrere GB), dann "RVS verbunden"
|
||||||
|
```
|
||||||
|
Zurück zu whisper: `docker compose --profile voxtral down && docker compose up -d whisper-bridge`.
|
||||||
|
|
||||||
|
## ⚠️ Auf echter Hardware verifizieren (blind gebaut)
|
||||||
|
|
||||||
|
1. **Transformers-API.** Die exakte Voxtral-Transkriptions-API ist in `bridge.py`
|
||||||
|
in **einer** Methode gekapselt (`VoxtralRunner._transcribe_blocking`), modelliert
|
||||||
|
nach der HF-Modelcard (`apply_transcription_request` → `generate` → `batch_decode`).
|
||||||
|
Beim ersten Lauf gegen die Modelcard prüfen und dort anpassen.
|
||||||
|
2. **HF-Gating.** Ist `Voxtral-Mini-3B-2507` gated, `HF_TOKEN` in `xtts/.env` setzen
|
||||||
|
(wird als `HUGGING_FACE_HUB_TOKEN` durchgereicht).
|
||||||
|
3. **VRAM/Tempo.** 3B in bf16 ~9 GB auf der 12-GB-Karte — Rest fürs KV-Cache. Ist die
|
||||||
|
Partial-Transkription (alle 1 s) zu schwer, `STREAM_TRANSCRIBE_INTERVAL_MS` hochsetzen.
|
||||||
|
4. **torch-Konflikt.** Falls `transformers`/`mistral-common` beim Build torch>2.6
|
||||||
|
erzwingen, meldet der Constraint einen Konflikt → dann brauchen wir doch das
|
||||||
|
Treiber-Upgrade (`bootstrap.sh --upgrade-driver` via NVIDIA-CUDA-Repo) + cu126-torch.
|
||||||
|
|
||||||
|
## Protokoll (RVS, identisch zu whisper — drop-in)
|
||||||
|
Rein: `stt_stream_start`, `stt_audio_chunk` (16 kHz mono s16le, base64), `stt_stream_end`.
|
||||||
|
Raus: `stt_partial`, `stt_endpoint`, `stt_stream_done`.
|
||||||
|
|
||||||
|
## TTS
|
||||||
|
Bleibt **F5-TTS** (klingt gut, passt auf GPU 0). Voxtral-TTS bräuchte ~24 GB — separates Thema.
|
||||||
|
|
||||||
|
## Quellen
|
||||||
|
- Modell: https://huggingface.co/mistralai/Voxtral-Mini-3B-2507
|
||||||
|
- Transformers-Nutzung: HF-Modelcard (Voxtral) + `mistral-common`
|
||||||
@@ -0,0 +1,888 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""
|
||||||
|
ARIA Voxtral-STT-3B Bridge (Transformers) — Ersatz fuer whisper.
|
||||||
|
|
||||||
|
Laeuft auf Treiber 550/CUDA 12.4 via torch cu124 (kein Treiber-Upgrade noetig).
|
||||||
|
Modell: Voxtral-Mini-3B-2507 (bf16, ~9 GB) → GPU 1 (12 GB, per Compose gepinnt).
|
||||||
|
|
||||||
|
Arbeitsweise = Zwilling der whisper-Bridge: App schickt live PCM-Chunks; wir
|
||||||
|
transkribieren alle ~STREAM_TRANSCRIBE_INTERVAL_MS auf dem Ringbuffer (Partials)
|
||||||
|
und feuern stt_endpoint, sobald der ADAPTIVE Endpointer (Rausch-Boden-VAD +
|
||||||
|
semantische Stagnation, aus M0.1) "fertig" sagt. RVS-Wire-Protokoll identisch zu
|
||||||
|
whisper → drop-in (die App merkt nur bessere Genauigkeit).
|
||||||
|
|
||||||
|
⚠️ VERIFY-ON-FIRST-RUN: Die exakte Transformers-Transkriptions-API von Voxtral
|
||||||
|
(apply_transcription_request / generate / decode) ist unten in EINER Methode
|
||||||
|
(VoxtralRunner._transcribe_blocking) gekapselt und nach dem HF-Modelcard-Muster
|
||||||
|
modelliert. Beim ersten echten Lauf gegen die Voxtral-Modelcard pruefen und dort
|
||||||
|
anpassen. Alles andere (RVS, Endpointer) ist bewaehrt.
|
||||||
|
|
||||||
|
Env:
|
||||||
|
RVS_HOST, RVS_PORT, RVS_TLS, RVS_TLS_FALLBACK, RVS_TOKEN
|
||||||
|
VOXTRAL_MODEL Default: mistralai/Voxtral-Mini-3B-2507
|
||||||
|
VOXTRAL_LANGUAGE Default: de
|
||||||
|
VOXTRAL_DEVICE Default: cuda
|
||||||
|
STREAM_TRANSCRIBE_INTERVAL_MS Default 1000 (3B ist schwerer als whisper-small)
|
||||||
|
"""
|
||||||
|
import asyncio
|
||||||
|
import base64
|
||||||
|
import json
|
||||||
|
import logging
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import tempfile
|
||||||
|
import time
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from typing import Optional
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
|
import soundfile as sf
|
||||||
|
import websockets
|
||||||
|
|
||||||
|
import speaker_id # Speaker-ID (nur Stefans Stimme) — portiert aus der whisper-Bridge
|
||||||
|
|
||||||
|
logging.basicConfig(
|
||||||
|
level=logging.INFO,
|
||||||
|
format="%(asctime)s [%(levelname)s] %(message)s",
|
||||||
|
datefmt="%H:%M:%S",
|
||||||
|
)
|
||||||
|
logger = logging.getLogger("voxtral-bridge")
|
||||||
|
|
||||||
|
RVS_HOST = os.getenv("RVS_HOST", "").strip()
|
||||||
|
RVS_PORT = int(os.getenv("RVS_PORT", "443"))
|
||||||
|
RVS_TLS = os.getenv("RVS_TLS", "true").lower() == "true"
|
||||||
|
RVS_TLS_FALLBACK = os.getenv("RVS_TLS_FALLBACK", "true").lower() == "true"
|
||||||
|
RVS_TOKEN = os.getenv("RVS_TOKEN", "").strip()
|
||||||
|
|
||||||
|
VOXTRAL_MODEL = os.getenv("VOXTRAL_MODEL", "mistralai/Voxtral-Mini-3B-2507")
|
||||||
|
VOXTRAL_LANGUAGE = os.getenv("VOXTRAL_LANGUAGE", "de")
|
||||||
|
VOXTRAL_DEVICE = os.getenv("VOXTRAL_DEVICE", "cuda")
|
||||||
|
|
||||||
|
# ── Compute-Fleet: Worker-Identitaet & Registrierung ──────────────
|
||||||
|
# Jeder Node meldet sich bei der aria-bridge (worker_hello) und haelt die
|
||||||
|
# Registry per periodischem worker_ping frisch. INSTANCE_ID adressiert diesen
|
||||||
|
# Worker bei Redundanz (targetInstance-Routing, Stage 3).
|
||||||
|
NODE_NAME = os.getenv("NODE_NAME", "node").strip() or "node"
|
||||||
|
GPU_IDS = os.getenv("NVIDIA_VISIBLE_DEVICES", "").strip()
|
||||||
|
WORKER_SERVICE = "voxtral"
|
||||||
|
INSTANCE_ID = f"{WORKER_SERVICE}@{NODE_NAME}"
|
||||||
|
WORKER_PING_INTERVAL_S = int(os.getenv("WORKER_PING_INTERVAL_S", "10"))
|
||||||
|
# Empfangs-Watchdog: kommt in RX_STALE_S kein Broadcast rein (ein echter Raum hat
|
||||||
|
# staendig Traffic, z.B. sat_hello alle 25s / Brain-Polling), gilt die Verbindung
|
||||||
|
# als halb-tot (Caddy pongt die WS-Pings selbst) -> Zwangs-Reconnect.
|
||||||
|
RX_STALE_S = int(os.getenv("RX_STALE_S", "60"))
|
||||||
|
|
||||||
|
# ── Auslastungs-Monitor (Stage E) ──────────────────────────
|
||||||
|
import node_stats
|
||||||
|
STATS_PATH = os.getenv("STATS_PATH", f"/root/.cache/huggingface/aria_stats_{WORKER_SERVICE}.json")
|
||||||
|
_stats = node_stats.NodeStats(INSTANCE_ID, NODE_NAME, STATS_PATH, logger=logger)
|
||||||
|
|
||||||
|
STREAM_TRANSCRIBE_INTERVAL_MS = int(os.getenv("STREAM_TRANSCRIBE_INTERVAL_MS", "1000"))
|
||||||
|
STREAM_DEFAULT_ENDPOINT_MS = 2400
|
||||||
|
STREAM_DEFAULT_HARD_CAP_MS = 300000
|
||||||
|
STREAM_MIN_AUDIO_MS = 600
|
||||||
|
STREAM_SPEAKER_CHECK_MS = 1500 # ab so viel Audio einmalig Speaker-ID pruefen
|
||||||
|
STREAM_SESSION_TTL_S = 120
|
||||||
|
STREAM_ENERGY_WINDOW_MS = 300
|
||||||
|
STREAM_SEMANTIC_BACKUP_FACTOR = 2.0
|
||||||
|
# Adaptiver Voice-Schwellwert (M0.1): relativ zum gemessenen Rausch-Boden.
|
||||||
|
STREAM_VOICE_FACTOR = 2.5
|
||||||
|
STREAM_VOICE_RMS_MIN = 0.005
|
||||||
|
STREAM_VOICE_RMS_MAX = 0.020
|
||||||
|
# Mindest-Stimme (in ~200ms-Endpointer-Frames), ab der eine Aufnahme ueberhaupt
|
||||||
|
# als Sprache gilt. Darunter = Stille / kurzer Geraeusch-Blip → KEIN Transkript
|
||||||
|
# (Voxtral halluziniert aus Fast-Nichts sonst einen Fuellsatz). 2 ≈ 400ms.
|
||||||
|
STREAM_MIN_VOICED_FRAMES = int(os.getenv("STREAM_MIN_VOICED_FRAMES", "2"))
|
||||||
|
|
||||||
|
# Halluzinations-Filter (2. Netz NACH der Transkription). Der voiced_frames-Guard
|
||||||
|
# oben faengt die reine Stille; hier kommt das "borderline"-Band dazu: wenn wenig
|
||||||
|
# echte Stimme da war UND das Transkript ein bekanntes Voxtral-Silence-Artefakt
|
||||||
|
# ist (Untertitel-Credits, Staedte-/Geo-Fakten "Flaeche von X km2"), ist es fast
|
||||||
|
# sicher ein Phantom aus Fast-Nichts → verwerfen. Gegated auf wenig voiced_frames,
|
||||||
|
# damit eine ECHTE Geografie-Frage (die hat normale Stimm-Energie) durchgeht.
|
||||||
|
STREAM_HALLUC_GUARD_FRAMES = int(os.getenv("STREAM_HALLUC_GUARD_FRAMES",
|
||||||
|
str(STREAM_MIN_VOICED_FRAMES * 4))) # ~1.6s
|
||||||
|
_HALLUCINATION_RE = re.compile(
|
||||||
|
r"untertitel"
|
||||||
|
r"|amara\.org"
|
||||||
|
r"|vielen\s+dank\s+f[uü]r'?s?\s+(zuschauen|zusehen|zuh[oö]ren)"
|
||||||
|
r"|bis\s+zum\s+n[aä]chsten\s+mal"
|
||||||
|
r"|abonnier"
|
||||||
|
r"|fl[aä]che\s+von\s+[\d.,]+\s*(km|quadratkilometer)"
|
||||||
|
r"|[\d.,]+\s*(km²|quadratkilometern?|einwohnern?)\b",
|
||||||
|
re.IGNORECASE,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Kollabiert unmittelbar wiederholte Phrasen (Voxtral-Repetition-Loop) auf EINE
|
||||||
|
# Kopie. Zweites Netz hinter no_repeat_ngram in der Generation. Phrase 5-80 Zeichen,
|
||||||
|
# 3+ mal hintereinander → eine. Kurze legitime Doppelungen ('ja ja', 'sehr sehr')
|
||||||
|
# bleiben (Unit < 5 Zeichen bzw. < 3 Wiederholungen).
|
||||||
|
_REPEAT_RE = re.compile(r"(.{5,80}?)(?:\s*\1){2,}", re.IGNORECASE | re.DOTALL)
|
||||||
|
|
||||||
|
|
||||||
|
def _collapse_repetitions(text: str) -> str:
|
||||||
|
if not text:
|
||||||
|
return text
|
||||||
|
out = text
|
||||||
|
for _ in range(3): # mehrfach fuer verschachtelte/ungleiche Loops
|
||||||
|
new = _REPEAT_RE.sub(r"\1", out)
|
||||||
|
if new == out:
|
||||||
|
break
|
||||||
|
out = new
|
||||||
|
return out.strip()
|
||||||
|
|
||||||
|
|
||||||
|
# ── Silero VAD: echte Sprach-Erkennung VOR dem Transkribieren ──────────────
|
||||||
|
# Der Muster-Filter oben kennt nur spezifische Artefakte. Generische Phantome
|
||||||
|
# ("Ich bin ein guter Mann" aus Fast-Stille) kann ein Text-Regex nicht fangen —
|
||||||
|
# aber ein VAD schon, weil es am AUDIO entscheidet, nicht am Text. Silero trennt
|
||||||
|
# Sprache zuverlaessig von Stille / Rauschen / MUSIK. Kein Speech-Segment →
|
||||||
|
# no-speech → nicht transkribieren → kein Phantom (und Musik/Instrumental fliegt
|
||||||
|
# gleich mit raus).
|
||||||
|
# FAIL-OPEN: klappt das VAD nicht (Import/Load/Inferenz), wird trotzdem normal
|
||||||
|
# transkribiert. Die STT darf NIE komplett sterben (Speaker-ID-Lektion).
|
||||||
|
SILERO_VAD_ENABLED = os.getenv("SILERO_VAD_ENABLED", "true").lower() in ("1", "true", "yes")
|
||||||
|
SILERO_VAD_THRESHOLD = float(os.getenv("SILERO_VAD_THRESHOLD", "0.5"))
|
||||||
|
SILERO_MIN_SPEECH_MS = int(os.getenv("SILERO_MIN_SPEECH_MS", "150"))
|
||||||
|
SILERO_PAD_MS = int(os.getenv("SILERO_PAD_MS", "200"))
|
||||||
|
# Speech-Endpoint gegen laute Umgebungsmusik: der RMS-Stille-Endpoint feuert bei
|
||||||
|
# durchgehender Musik NIE (Energie bleibt oben). Deshalb waehrend lauter Phasen
|
||||||
|
# periodisch (alle X ms) mit Silero pruefen, ob im letzten endpoint_ms-Fenster
|
||||||
|
# ueberhaupt noch Sprache ist — wenn nicht (nur Musik/Stille), Turn beenden.
|
||||||
|
STREAM_SPEECH_ENDPOINT_CHECK_MS = int(os.getenv("STREAM_SPEECH_ENDPOINT_CHECK_MS", "700"))
|
||||||
|
|
||||||
|
_vad_state = {"model": None, "get_ts": None, "failed": False}
|
||||||
|
|
||||||
|
|
||||||
|
def _speech_segments(audio_f32):
|
||||||
|
"""Silero-VAD-Sprachsegmente (Liste von {start,end} Sample-Indizes) im
|
||||||
|
16kHz-float32-Audio. Rueckgabe:
|
||||||
|
[] → kein Speech (Stille/Rauschen/Musik) → Phantom-Verdacht, verwerfen.
|
||||||
|
[...] → Speech vorhanden.
|
||||||
|
None → VAD nicht verfuegbar → fail-open (Aufrufer transkribiert normal)."""
|
||||||
|
if not SILERO_VAD_ENABLED or _vad_state["failed"]:
|
||||||
|
return None
|
||||||
|
if _vad_state["model"] is None:
|
||||||
|
try:
|
||||||
|
from silero_vad import load_silero_vad, get_speech_timestamps
|
||||||
|
_vad_state["model"] = load_silero_vad()
|
||||||
|
_vad_state["get_ts"] = get_speech_timestamps
|
||||||
|
logger.info("Silero VAD geladen (threshold=%.2f, min_speech=%dms)",
|
||||||
|
SILERO_VAD_THRESHOLD, SILERO_MIN_SPEECH_MS)
|
||||||
|
except Exception:
|
||||||
|
logger.exception("Silero VAD Laden fehlgeschlagen — dauerhaft aus (fail-open)")
|
||||||
|
_vad_state["failed"] = True
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
import torch as _torch
|
||||||
|
segs = _vad_state["get_ts"](
|
||||||
|
_torch.from_numpy(audio_f32), _vad_state["model"],
|
||||||
|
sampling_rate=16000, threshold=SILERO_VAD_THRESHOLD,
|
||||||
|
min_speech_duration_ms=SILERO_MIN_SPEECH_MS,
|
||||||
|
)
|
||||||
|
return segs or []
|
||||||
|
except Exception:
|
||||||
|
logger.exception("Silero VAD Inferenz fehlgeschlagen — dieser Turn fail-open")
|
||||||
|
return None
|
||||||
|
|
||||||
|
# Speaker-ID Gating global an/aus. DEFAULT AUS (fail-open) — die "nur meine Stimme"-
|
||||||
|
# Pruefung ist ein BEWUSSTER Schalter, kein Automatismus: ein einziger schlechter
|
||||||
|
# Enroll darf nie die ganze STT lahmlegen (genau das ist passiert). Wird per config-
|
||||||
|
# Broadcast (voiceIdEnabled, aus dem Diagnostic) zur Laufzeit gesetzt. Kann per ENV
|
||||||
|
# vorbelegt werden.
|
||||||
|
SPEAKER_ID_ENABLED = os.getenv("VOICE_ID_ENABLED", "false").lower() in ("1", "true", "yes")
|
||||||
|
|
||||||
|
|
||||||
|
def _set_speaker_id_enabled(val: bool) -> None:
|
||||||
|
global SPEAKER_ID_ENABLED
|
||||||
|
SPEAKER_ID_ENABLED = bool(val)
|
||||||
|
|
||||||
|
|
||||||
|
def pcm_s16le_to_float32(data: bytes) -> np.ndarray:
|
||||||
|
if not data:
|
||||||
|
return np.zeros(0, dtype=np.float32)
|
||||||
|
return np.frombuffer(data, dtype=np.int16).astype(np.float32) / 32768.0
|
||||||
|
|
||||||
|
|
||||||
|
async def _send(ws, mtype: str, payload: dict) -> None:
|
||||||
|
try:
|
||||||
|
await ws.send(json.dumps({
|
||||||
|
"type": mtype, "payload": payload, "timestamp": int(time.time() * 1000),
|
||||||
|
}))
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning("RVS-Send fehlgeschlagen (%s): %s", mtype, e)
|
||||||
|
|
||||||
|
|
||||||
|
class VoxtralRunner:
|
||||||
|
"""Haelt das Voxtral-Modell (Transformers). transcribe() blockiert → aus dem
|
||||||
|
Event-Loop via run_in_executor aufrufen. Ein Lock serialisiert GPU-Zugriffe."""
|
||||||
|
|
||||||
|
def __init__(self) -> None:
|
||||||
|
self.model = None
|
||||||
|
self.processor = None
|
||||||
|
self._lock = asyncio.Lock()
|
||||||
|
|
||||||
|
def load(self) -> None:
|
||||||
|
import torch
|
||||||
|
from transformers import AutoProcessor, VoxtralForConditionalGeneration
|
||||||
|
t0 = time.time()
|
||||||
|
logger.info("Lade Voxtral '%s' (device=%s, bf16)…", VOXTRAL_MODEL, VOXTRAL_DEVICE)
|
||||||
|
self.processor = AutoProcessor.from_pretrained(VOXTRAL_MODEL)
|
||||||
|
self.model = VoxtralForConditionalGeneration.from_pretrained(
|
||||||
|
VOXTRAL_MODEL, torch_dtype=torch.bfloat16, device_map=VOXTRAL_DEVICE,
|
||||||
|
)
|
||||||
|
logger.info("Voxtral geladen in %.1fs", time.time() - t0)
|
||||||
|
|
||||||
|
def _transcribe_blocking(self, audio_f32: np.ndarray, language: str) -> str:
|
||||||
|
import torch
|
||||||
|
proc, model = self.processor, self.model
|
||||||
|
if proc is None or model is None or audio_f32.size == 0:
|
||||||
|
return ""
|
||||||
|
# VoxtralProcessor verlangt bei rohen Arrays ein 'format'. Robuster:
|
||||||
|
# in ein temp-WAV schreiben und den PFAD uebergeben — der Processor liest
|
||||||
|
# Format + Samplerate selbst, kein 'format'-Argument noetig.
|
||||||
|
wav_path = None
|
||||||
|
try:
|
||||||
|
with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as tf:
|
||||||
|
wav_path = tf.name
|
||||||
|
sf.write(wav_path, audio_f32, 16000, subtype="PCM_16")
|
||||||
|
inputs = proc.apply_transcription_request(
|
||||||
|
language=language, audio=wav_path, model_id=VOXTRAL_MODEL,
|
||||||
|
)
|
||||||
|
inputs = inputs.to(VOXTRAL_DEVICE, dtype=torch.bfloat16)
|
||||||
|
with torch.no_grad():
|
||||||
|
# hoch genug fuer lange Diktate (stoppt eh am EOS); 512 hat
|
||||||
|
# mehrminutige Aufnahmen abgeschnitten.
|
||||||
|
# Repetition-Bremse: Voxtral kippt bei Stille/Rauschen am Ende
|
||||||
|
# gern in eine Schleife und wiederholt einen Satz zig-mal
|
||||||
|
# ("Vergiss das, das ist nur... Vergiss das, das ist nur..."
|
||||||
|
# x15). no_repeat_ngram_size=4 laesst die ERSTE echte Nennung
|
||||||
|
# durch, verbietet aber die exakte 4-Gramm-Wiederholung → Loop
|
||||||
|
# bricht ab; repetition_penalty daempft zusaetzlich. Beides mild,
|
||||||
|
# damit normale Sprache (auch mal ein doppeltes Wort) unberuehrt
|
||||||
|
# bleibt.
|
||||||
|
outputs = model.generate(
|
||||||
|
**inputs,
|
||||||
|
max_new_tokens=4096,
|
||||||
|
no_repeat_ngram_size=4,
|
||||||
|
repetition_penalty=1.15,
|
||||||
|
)
|
||||||
|
trimmed = outputs[:, inputs.input_ids.shape[1]:]
|
||||||
|
text = proc.batch_decode(trimmed, skip_special_tokens=True)
|
||||||
|
return (text[0] if text else "").strip()
|
||||||
|
finally:
|
||||||
|
if wav_path:
|
||||||
|
try:
|
||||||
|
os.unlink(wav_path)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
async def transcribe(self, audio_f32: np.ndarray, language: str) -> str:
|
||||||
|
loop = asyncio.get_running_loop()
|
||||||
|
async with self._lock:
|
||||||
|
return await loop.run_in_executor(None, self._transcribe_blocking, audio_f32, language)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class StreamSession:
|
||||||
|
request_id: str
|
||||||
|
audio_request_id: str
|
||||||
|
language: str
|
||||||
|
endpoint_ms: int
|
||||||
|
hard_cap_ms: int
|
||||||
|
voice: str = ""
|
||||||
|
speed: float = 1.0
|
||||||
|
interrupted: bool = False
|
||||||
|
location: Optional[dict] = None
|
||||||
|
sample_rate: int = 16000
|
||||||
|
voice_factor: float = STREAM_VOICE_FACTOR
|
||||||
|
voice_rms_min: float = STREAM_VOICE_RMS_MIN
|
||||||
|
voice_rms_max: float = STREAM_VOICE_RMS_MAX
|
||||||
|
pcm_buffer: bytearray = field(default_factory=bytearray)
|
||||||
|
started_at: float = field(default_factory=time.time)
|
||||||
|
last_chunk_at: float = field(default_factory=time.time)
|
||||||
|
last_partial: str = ""
|
||||||
|
last_growth_at: float = 0.0
|
||||||
|
last_transcribe_at: float = 0.0
|
||||||
|
last_voice_at: float = 0.0
|
||||||
|
last_speech_check_at: float = 0.0 # Drossel fuer den Silero-Speech-Endpoint
|
||||||
|
noise_floor: float = 0.0
|
||||||
|
closed: bool = False
|
||||||
|
endpoint_sent: bool = False
|
||||||
|
# Einmaliges "Sprache erkannt"-Signal an die App gesendet? Voxtral schickt
|
||||||
|
# keine Live-Partials, aber der App-No-Speech-Watchdog wartet auf ein
|
||||||
|
# stt_partial, um "der User redet" zu erkennen — sonst cancelt er mitten im
|
||||||
|
# Satz. Wir feuern EIN leeres stt_partial beim ersten Voice-Frame.
|
||||||
|
speech_signaled: bool = False
|
||||||
|
# Anzahl Endpointer-Frames (~200ms) mit echter Stimme. Gate gegen Halluzination
|
||||||
|
# aus Stille/Blips: unter STREAM_MIN_VOICED_FRAMES wird nicht transkribiert.
|
||||||
|
voiced_frames: int = 0
|
||||||
|
# Speaker-ID Gating (einmalig auf die ersten ~1.5s der Aufnahme)
|
||||||
|
speaker_checked: bool = False
|
||||||
|
speaker_match: Optional[bool] = None
|
||||||
|
speaker_similarity: float = 0.0
|
||||||
|
|
||||||
|
|
||||||
|
class SessionManager:
|
||||||
|
def __init__(self, runner: VoxtralRunner) -> None:
|
||||||
|
self.runner = runner
|
||||||
|
self._sessions: dict[str, StreamSession] = {}
|
||||||
|
self._ws = None
|
||||||
|
|
||||||
|
def attach_ws(self, ws) -> None:
|
||||||
|
self._ws = ws
|
||||||
|
|
||||||
|
def start_session(self, payload: dict) -> None:
|
||||||
|
rid = (payload.get("requestId") or "").strip()
|
||||||
|
if not rid:
|
||||||
|
return
|
||||||
|
try:
|
||||||
|
endpoint_ms = int(payload.get("endpointMs") or STREAM_DEFAULT_ENDPOINT_MS)
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
endpoint_ms = STREAM_DEFAULT_ENDPOINT_MS
|
||||||
|
try:
|
||||||
|
hard_cap_ms = int(payload.get("hardCapMs") or STREAM_DEFAULT_HARD_CAP_MS)
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
hard_cap_ms = STREAM_DEFAULT_HARD_CAP_MS
|
||||||
|
try:
|
||||||
|
voice_factor = float(payload.get("voiceFactor") or STREAM_VOICE_FACTOR)
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
voice_factor = STREAM_VOICE_FACTOR
|
||||||
|
self._sessions[rid] = StreamSession(
|
||||||
|
request_id=rid,
|
||||||
|
audio_request_id=payload.get("audioRequestId", "") or "",
|
||||||
|
language=payload.get("language") or VOXTRAL_LANGUAGE,
|
||||||
|
endpoint_ms=endpoint_ms,
|
||||||
|
hard_cap_ms=hard_cap_ms,
|
||||||
|
voice=payload.get("voice", "") or "",
|
||||||
|
speed=float(payload.get("speed") or 1.0),
|
||||||
|
voice_factor=voice_factor,
|
||||||
|
interrupted=bool(payload.get("interrupted", False)),
|
||||||
|
location=payload.get("location") or None,
|
||||||
|
sample_rate=int(payload.get("sampleRate") or 16000),
|
||||||
|
)
|
||||||
|
logger.info("Voxtral-Session offen: id=%s lang=%s endpointMs=%d",
|
||||||
|
rid[:8], self._sessions[rid].language, endpoint_ms)
|
||||||
|
|
||||||
|
def feed_chunk(self, payload: dict) -> bool:
|
||||||
|
sess = self._sessions.get(payload.get("requestId", ""))
|
||||||
|
if sess is None or sess.closed:
|
||||||
|
return False
|
||||||
|
pcm_b64 = payload.get("pcm", "")
|
||||||
|
if pcm_b64:
|
||||||
|
try:
|
||||||
|
sess.pcm_buffer.extend(base64.b64decode(pcm_b64))
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
sess.last_chunk_at = time.time()
|
||||||
|
return True
|
||||||
|
|
||||||
|
def end_session(self, request_id: str) -> None:
|
||||||
|
sess = self._sessions.get(request_id)
|
||||||
|
if sess is not None:
|
||||||
|
sess.closed = True
|
||||||
|
|
||||||
|
def drop(self, request_id: str) -> None:
|
||||||
|
self._sessions.pop(request_id, None)
|
||||||
|
|
||||||
|
# ── Endpointer (adaptiv, M0.1) ──
|
||||||
|
def _buffer_ms(self, sess: StreamSession) -> float:
|
||||||
|
samples = len(sess.pcm_buffer) // 2
|
||||||
|
return (samples / sess.sample_rate) * 1000.0 if samples else 0.0
|
||||||
|
|
||||||
|
def _tail_rms(self, sess: StreamSession) -> float:
|
||||||
|
win = int(sess.sample_rate * STREAM_ENERGY_WINDOW_MS / 1000) * 2
|
||||||
|
if win <= 0:
|
||||||
|
return 0.0
|
||||||
|
tail = sess.pcm_buffer[-win:]
|
||||||
|
if len(tail) < 2:
|
||||||
|
return 0.0
|
||||||
|
arr = pcm_s16le_to_float32(bytes(tail))
|
||||||
|
return float(np.sqrt(np.mean(arr * arr))) if arr.size else 0.0
|
||||||
|
|
||||||
|
def _voice_threshold(self, sess: StreamSession) -> float:
|
||||||
|
nf = sess.noise_floor
|
||||||
|
if nf <= 0.0:
|
||||||
|
return sess.voice_rms_min
|
||||||
|
return min(max(nf * sess.voice_factor, sess.voice_rms_min), sess.voice_rms_max)
|
||||||
|
|
||||||
|
def _update_noise_floor(self, sess: StreamSession, rms: float) -> None:
|
||||||
|
nf = sess.noise_floor
|
||||||
|
if nf <= 0.0:
|
||||||
|
sess.noise_floor = rms
|
||||||
|
elif rms < nf:
|
||||||
|
sess.noise_floor = 0.90 * nf + 0.10 * rms
|
||||||
|
else:
|
||||||
|
sess.noise_floor = 0.98 * nf + 0.02 * rms
|
||||||
|
|
||||||
|
async def _check_speaker(self, sess: StreamSession) -> None:
|
||||||
|
"""Einmalig: erste ~1.5s → Embedding → Vergleich mit Fingerprint.
|
||||||
|
Ohne Fingerprint fail-open (match=True). Bei Mismatch: Session beenden."""
|
||||||
|
sess.speaker_checked = True
|
||||||
|
# Schalter aus (Default) → gar keine Pruefung, alles durchlassen.
|
||||||
|
if not SPEAKER_ID_ENABLED:
|
||||||
|
sess.speaker_match = True
|
||||||
|
return
|
||||||
|
head = bytes(sess.pcm_buffer[: STREAM_SPEAKER_CHECK_MS * 32])
|
||||||
|
if len(head) < speaker_id.MIN_SAMPLE_BYTES:
|
||||||
|
sess.speaker_match = True
|
||||||
|
return
|
||||||
|
try:
|
||||||
|
loop = asyncio.get_running_loop()
|
||||||
|
is_match, sim = await loop.run_in_executor(None, speaker_id.verify, head)
|
||||||
|
except Exception as exc:
|
||||||
|
logger.warning("Stream %s: speaker-check crashed (%s) — fail-open",
|
||||||
|
sess.request_id[:8], exc)
|
||||||
|
sess.speaker_match = True
|
||||||
|
return
|
||||||
|
sess.speaker_match = is_match
|
||||||
|
sess.speaker_similarity = sim
|
||||||
|
logger.info("Stream %s: speaker-check sim=%.2f → %s (thr=%.2f)",
|
||||||
|
sess.request_id[:8], sim, "MATCH" if is_match else "REJECT",
|
||||||
|
speaker_id.DEFAULT_THRESHOLD)
|
||||||
|
if not is_match:
|
||||||
|
await self._finalize_speaker_mismatch(sess, sim)
|
||||||
|
|
||||||
|
async def _finalize_speaker_mismatch(self, sess: StreamSession, similarity: float) -> None:
|
||||||
|
"""Fremde Stimme: synthetisches leeres stt_endpoint (reason=speaker_mismatch),
|
||||||
|
Session droppen — kein Voxtral-Transcribe, kein Brain-Call."""
|
||||||
|
if sess.endpoint_sent:
|
||||||
|
return
|
||||||
|
sess.endpoint_sent = True
|
||||||
|
duration_s = self._buffer_ms(sess) / 1000.0
|
||||||
|
logger.info("Stream %s: speaker-mismatch (sim=%.2f) — DROP nach %.1fs",
|
||||||
|
sess.request_id[:8], similarity, duration_s)
|
||||||
|
if self._ws is not None:
|
||||||
|
payload = {
|
||||||
|
"requestId": sess.request_id,
|
||||||
|
"audioRequestId": sess.audio_request_id,
|
||||||
|
"text": "", "reason": "speaker_mismatch",
|
||||||
|
"durationS": duration_s, "sttMs": 0,
|
||||||
|
"voice": sess.voice, "speed": sess.speed,
|
||||||
|
"interrupted": sess.interrupted,
|
||||||
|
"speakerSimilarity": float(similarity),
|
||||||
|
}
|
||||||
|
if sess.location:
|
||||||
|
payload["location"] = sess.location
|
||||||
|
await _send(self._ws, "stt_endpoint", payload)
|
||||||
|
await _send(self._ws, "stt_stream_done", {
|
||||||
|
"requestId": sess.request_id,
|
||||||
|
"audioRequestId": sess.audio_request_id,
|
||||||
|
"text": "", "reason": "speaker_mismatch",
|
||||||
|
})
|
||||||
|
self.drop(sess.request_id)
|
||||||
|
|
||||||
|
async def run_endpointer(self) -> None:
|
||||||
|
logger.info("Voxtral-Endpointer gestartet (adaptiver VAD, interval=%dms)",
|
||||||
|
STREAM_TRANSCRIBE_INTERVAL_MS)
|
||||||
|
while True:
|
||||||
|
await asyncio.sleep(0.2)
|
||||||
|
now = time.time()
|
||||||
|
for sid, sess in list(self._sessions.items()):
|
||||||
|
try:
|
||||||
|
await self._tick(sess, now)
|
||||||
|
except Exception:
|
||||||
|
logger.exception("Tick crashed (session=%s)", sid[:8])
|
||||||
|
for sid, sess in list(self._sessions.items()):
|
||||||
|
if now - sess.last_chunk_at > STREAM_SESSION_TTL_S:
|
||||||
|
logger.info("Stream %s: TTL — drop", sid[:8])
|
||||||
|
self.drop(sid)
|
||||||
|
|
||||||
|
async def _tick(self, sess: StreamSession, now: float) -> None:
|
||||||
|
if sess.endpoint_sent:
|
||||||
|
return
|
||||||
|
if (now - sess.started_at) * 1000.0 > sess.hard_cap_ms and not sess.closed:
|
||||||
|
await self._finalize(sess, "hardcap")
|
||||||
|
return
|
||||||
|
if sess.closed:
|
||||||
|
await self._finalize(sess, "stream_end")
|
||||||
|
return
|
||||||
|
if self._buffer_ms(sess) < STREAM_MIN_AUDIO_MS:
|
||||||
|
return
|
||||||
|
# Speaker-ID einmalig: ist es Stefans Stimme? Fremde → Session verwerfen
|
||||||
|
# (kein Transcribe, kein Brain-Call). Ohne Enrollment fail-open.
|
||||||
|
if not sess.speaker_checked and self._buffer_ms(sess) >= STREAM_SPEAKER_CHECK_MS:
|
||||||
|
await self._check_speaker(sess)
|
||||||
|
if sess.speaker_match is False:
|
||||||
|
return
|
||||||
|
# Adaptive akustische Sprach-Aktivitaet (M0.1). KEINE Live-Partials mehr:
|
||||||
|
# Voxtral-3B transkribiert den ganzen WACHSENDEN Buffer und braucht dafuer
|
||||||
|
# bei langen Aufnahmen 5-6 s — zu langsam fuer Live-Text, UND diese Latenz
|
||||||
|
# hat den semantischen Endpoint faelschlich ausgeloest (Partial-Latenz >
|
||||||
|
# Timeout → willkuerliche Abbrueche nach 20-40 s). Deshalb: Turn-Ende rein
|
||||||
|
# AKUSTISCH, transkribiert wird nur EINMAL im _finalize.
|
||||||
|
rms = self._tail_rms(sess)
|
||||||
|
if rms >= self._voice_threshold(sess):
|
||||||
|
sess.last_voice_at = now
|
||||||
|
sess.voiced_frames += 1
|
||||||
|
# Einmalig der App melden, dass Sprache begonnen hat — aber ERST ab genug
|
||||||
|
# echter Stimme (>= STREAM_MIN_VOICED_FRAMES). Ein einzelner Geraeusch-
|
||||||
|
# Blip darf den No-Speech-Watchdog NICHT loeschen, sonst transkribiert
|
||||||
|
# Voxtral das Fast-Nichts und HALLUZINIERT einen Phantom-Satz. Ohne Live-
|
||||||
|
# Partials wuerde der Watchdog die Aufnahme sonst am Konversationsfenster
|
||||||
|
# canceln, obwohl der User redet ("beendet nach ~4s"-Repro). Leeres
|
||||||
|
# stt_partial: App setzt streamGotPartial=true + loescht den Watchdog.
|
||||||
|
# Nach der Speaker-ID-Pruefung (oben) → fremde Stimmen signalisieren NICHT.
|
||||||
|
if (not sess.speech_signaled and self._ws is not None
|
||||||
|
and sess.voiced_frames >= STREAM_MIN_VOICED_FRAMES):
|
||||||
|
sess.speech_signaled = True
|
||||||
|
await _send(self._ws, "stt_partial", {
|
||||||
|
"requestId": sess.request_id,
|
||||||
|
"audioRequestId": sess.audio_request_id,
|
||||||
|
"text": "",
|
||||||
|
})
|
||||||
|
# Speech-Endpoint gegen laute Umgebungsmusik: es ist gerade laut (rms
|
||||||
|
# ueber Schwelle) — aber ist es Sprache oder Musik? Der RMS-Endpoint
|
||||||
|
# unten wuerde bei Musik NIE feuern (last_voice_at bleibt frisch).
|
||||||
|
# Deshalb gedrosselt mit Silero das letzte endpoint_ms-Fenster pruefen:
|
||||||
|
# KEINE Sprache drin (nur Musik) → Turn ist zu Ende. Real gesprochene
|
||||||
|
# Turns haben Sprache im Fenster → laufen weiter.
|
||||||
|
if (self._buffer_ms(sess) >= sess.endpoint_ms
|
||||||
|
and (now - sess.last_speech_check_at) * 1000.0 >= STREAM_SPEECH_ENDPOINT_CHECK_MS):
|
||||||
|
sess.last_speech_check_at = now
|
||||||
|
try:
|
||||||
|
tail_bytes = int(sess.endpoint_ms / 1000.0 * sess.sample_rate) * 2
|
||||||
|
tail = pcm_s16le_to_float32(bytes(sess.pcm_buffer[-tail_bytes:]))
|
||||||
|
segs = _speech_segments(tail)
|
||||||
|
if segs is not None and len(segs) == 0:
|
||||||
|
logger.info("Stream %s: Speech-Endpoint (keine Sprache im letzten %dms — Musik/Stille) → finalize",
|
||||||
|
sess.request_id[:8], sess.endpoint_ms)
|
||||||
|
await self._finalize(sess, "endpoint")
|
||||||
|
return
|
||||||
|
except Exception:
|
||||||
|
logger.exception("Speech-Endpoint-Check fehlgeschlagen — ignoriert")
|
||||||
|
else:
|
||||||
|
self._update_noise_floor(sess, rms)
|
||||||
|
# No-Speech-Timeout: wurde die GANZE Zeit KEINE Stimme erkannt
|
||||||
|
# (last_voice_at==0), feuert der normale Endpoint unten NIE — der braucht
|
||||||
|
# last_voice_at>0. Ohne das bleibt ein reines Stille-Fenster offen bis
|
||||||
|
# Hardcap/manuellem Stop → genau Stefans Repro: "die Stille-Ende wird nie
|
||||||
|
# erreicht, stop ich selbst ist es weg". Nach endpoint_ms Stille ab Start
|
||||||
|
# schliessen wir das Fenster selbst als no-speech (leer, lautlos, zurueck
|
||||||
|
# aufs Wake-Word). voiced_frames==0 → _finalize verwirft ohne Transkript,
|
||||||
|
# also KEIN Phantom.
|
||||||
|
if sess.last_voice_at == 0 and (now - sess.started_at) * 1000.0 >= sess.endpoint_ms:
|
||||||
|
await self._finalize(sess, "no_speech")
|
||||||
|
return
|
||||||
|
# Endpoint: hat der User schon gesprochen UND ist es seit endpoint_ms still?
|
||||||
|
if sess.last_voice_at > 0 and (now - sess.last_voice_at) * 1000.0 >= sess.endpoint_ms:
|
||||||
|
await self._finalize(sess, "endpoint")
|
||||||
|
|
||||||
|
async def _emit_no_speech(self, sess: "StreamSession", reason_label: str) -> None:
|
||||||
|
"""Leeres no-speech-Endpoint senden + Session droppen (kein Transkript).
|
||||||
|
App re-armt still, zurueck aufs Wake-Word."""
|
||||||
|
if self._ws is not None:
|
||||||
|
payload = {"requestId": sess.request_id,
|
||||||
|
"audioRequestId": sess.audio_request_id,
|
||||||
|
"text": "", "reason": reason_label,
|
||||||
|
"durationS": 0.0, "sttMs": 0}
|
||||||
|
await _send(self._ws, "stt_endpoint", payload)
|
||||||
|
await _send(self._ws, "stt_stream_done", {
|
||||||
|
"requestId": sess.request_id,
|
||||||
|
"audioRequestId": sess.audio_request_id,
|
||||||
|
"text": "", "reason": reason_label})
|
||||||
|
self.drop(sess.request_id)
|
||||||
|
|
||||||
|
async def _finalize(self, sess: StreamSession, reason: str) -> None:
|
||||||
|
if sess.endpoint_sent:
|
||||||
|
return
|
||||||
|
sess.endpoint_sent = True
|
||||||
|
# Halluzinations-Guard: zu wenig echte Stimme (Stille / kurzer Blip im
|
||||||
|
# Passiv-/Wake-Fenster) → NICHT transkribieren. Voxtral (wie Whisper) baut
|
||||||
|
# aus Fast-Nichts gern einen Fuellsatz ("Die Stadt hat eine Flaeche von
|
||||||
|
# 1,5 km2"), der dann als PHANTOM-Nachricht ans Brain geht und das Gespraech
|
||||||
|
# entgleisen laesst (Stefans Repro: "kam Nachricht von mir, obwohl ich
|
||||||
|
# nichts sagte"). Leeres Endpoint = no-speech → App re-armt still.
|
||||||
|
#
|
||||||
|
# WICHTIG (aus dem ai-box-Log gelernt): die Phantome kommen mit
|
||||||
|
# reason=stream_end — Passiv-/Wake-Fenster enden AUCH per stream_end, wenn
|
||||||
|
# sie auf Stille zumachen. stream_end ist also NICHT gleich "manueller Stop".
|
||||||
|
# Deshalb greift der Guard jetzt auch bei stream_end, aber mit niedrigerer
|
||||||
|
# Schwelle (voiced==0 = gar keine Stimme), damit ein kurzes bewusstes Wort
|
||||||
|
# ('ja', 'stopp') am Aufnahme-Button noch durchgeht, echte Stille aber nicht.
|
||||||
|
_min_voiced = STREAM_MIN_VOICED_FRAMES if reason != "stream_end" else 1
|
||||||
|
if sess.voiced_frames < _min_voiced:
|
||||||
|
logger.info("Stream %s: no-speech (voiced_frames=%d<%d, reason=%s) — leeres Endpoint",
|
||||||
|
sess.request_id[:8], sess.voiced_frames, _min_voiced, reason)
|
||||||
|
if self._ws is not None:
|
||||||
|
nospeech = {"requestId": sess.request_id,
|
||||||
|
"audioRequestId": sess.audio_request_id,
|
||||||
|
"text": "", "reason": f"no_speech:{reason}",
|
||||||
|
"durationS": 0.0, "sttMs": 0}
|
||||||
|
await _send(self._ws, "stt_endpoint", nospeech)
|
||||||
|
await _send(self._ws, "stt_stream_done", {
|
||||||
|
"requestId": sess.request_id,
|
||||||
|
"audioRequestId": sess.audio_request_id,
|
||||||
|
"text": "", "reason": f"no_speech:{reason}"})
|
||||||
|
self.drop(sess.request_id)
|
||||||
|
return
|
||||||
|
audio = pcm_s16le_to_float32(bytes(sess.pcm_buffer))
|
||||||
|
|
||||||
|
# Silero VAD: ist ueberhaupt echte Sprache im Audio? Das entscheidet am
|
||||||
|
# AUDIO, nicht am Text — faengt also generische Phantome ("Ich bin ein
|
||||||
|
# guter Mann") UND Musik/Rauschen, die der Muster-Filter nicht kennt.
|
||||||
|
# Kein Speech-Segment → no-speech, gar nicht erst transkribieren.
|
||||||
|
# fail-open: segs=None (VAD nicht verfuegbar) → normal weiter.
|
||||||
|
segs = _speech_segments(audio)
|
||||||
|
if segs is not None and len(segs) == 0:
|
||||||
|
logger.info("Stream %s: Silero VAD — keine Sprache (%.1fs, reason=%s) → no-speech",
|
||||||
|
sess.request_id[:8], audio.size / 16000.0, reason)
|
||||||
|
await self._emit_no_speech(sess, f"vad_no_speech:{reason}")
|
||||||
|
return
|
||||||
|
if segs:
|
||||||
|
# Auf die Sprach-Spanne trimmen (Stille-Raender weg → Voxtral
|
||||||
|
# halluziniert an den Enden weniger). Kleiner Pad gegen abgeschnittene
|
||||||
|
# leise Wort-Anfaenge/-Enden.
|
||||||
|
pad = int(SILERO_PAD_MS / 1000.0 * 16000)
|
||||||
|
s0 = max(0, segs[0]["start"] - pad)
|
||||||
|
s1 = min(int(audio.size), segs[-1]["end"] + pad)
|
||||||
|
if s1 > s0 and (s1 - s0) < audio.size:
|
||||||
|
audio = audio[s0:s1]
|
||||||
|
|
||||||
|
t0 = time.time()
|
||||||
|
try:
|
||||||
|
final_text = (await self.runner.transcribe(audio, sess.language)).strip()
|
||||||
|
except Exception:
|
||||||
|
logger.exception("Stream %s: Final-Transcribe crashed", sess.request_id[:8])
|
||||||
|
final_text = sess.last_partial
|
||||||
|
stt_ms = int((time.time() - t0) * 1000)
|
||||||
|
duration_s = audio.size / 16000.0
|
||||||
|
# Repetition-Loop einkassieren, falls trotz no_repeat_ngram was durchkam.
|
||||||
|
_collapsed = _collapse_repetitions(final_text)
|
||||||
|
if _collapsed != final_text:
|
||||||
|
logger.info("Stream %s: Repetition-Loop kollabiert (%d→%d Zeichen)",
|
||||||
|
sess.request_id[:8], len(final_text), len(_collapsed))
|
||||||
|
final_text = _collapsed
|
||||||
|
logger.info("Stream %s: FINAL (reason=%s, %.1fs, %dms): %r",
|
||||||
|
sess.request_id[:8], reason, duration_s, stt_ms, final_text[:120])
|
||||||
|
|
||||||
|
# Halluzinations-Filter (2. Netz): leeres/Artefakt-Transkript im borderline-
|
||||||
|
# Band → als no-speech verwerfen statt ein Phantom ("Die Stadt hat eine
|
||||||
|
# Flaeche von 1,5 km2") ans Brain zu schicken. Gilt fuer ALLE reasons inkl.
|
||||||
|
# stream_end (dort kamen die realen Phantome!) — aber das borderline-Band
|
||||||
|
# (wenig voiced_frames) schuetzt echte, klar gesprochene Eingaben: eine echte
|
||||||
|
# Geografie-FRAGE hat normale Stimm-Energie (voiced_frames >> Schwelle) und
|
||||||
|
# geht durch; das Phantom aus Stille hat ~0 und wird verworfen. Ein leeres
|
||||||
|
# Transkript wird immer verworfen (nichts gesagt = nichts senden).
|
||||||
|
_clean = final_text.strip(" .,!?…-\t\n\r")
|
||||||
|
_borderline = sess.voiced_frames < STREAM_HALLUC_GUARD_FRAMES
|
||||||
|
_is_phantom = (not _clean) or (_borderline and bool(_HALLUCINATION_RE.search(final_text)))
|
||||||
|
if _is_phantom:
|
||||||
|
logger.info("Stream %s: Halluzination verworfen (voiced_frames=%d<%d, %.1fs, text=%r)",
|
||||||
|
sess.request_id[:8], sess.voiced_frames, STREAM_HALLUC_GUARD_FRAMES,
|
||||||
|
duration_s, final_text[:80])
|
||||||
|
if self._ws is not None:
|
||||||
|
nospeech = {"requestId": sess.request_id,
|
||||||
|
"audioRequestId": sess.audio_request_id,
|
||||||
|
"text": "", "reason": f"hallucination:{reason}",
|
||||||
|
"durationS": 0.0, "sttMs": stt_ms}
|
||||||
|
await _send(self._ws, "stt_endpoint", nospeech)
|
||||||
|
await _send(self._ws, "stt_stream_done", {
|
||||||
|
"requestId": sess.request_id,
|
||||||
|
"audioRequestId": sess.audio_request_id,
|
||||||
|
"text": "", "reason": f"hallucination:{reason}"})
|
||||||
|
self.drop(sess.request_id)
|
||||||
|
return
|
||||||
|
|
||||||
|
if self._ws is not None:
|
||||||
|
payload = {
|
||||||
|
"requestId": sess.request_id,
|
||||||
|
"audioRequestId": sess.audio_request_id,
|
||||||
|
"text": final_text,
|
||||||
|
"reason": reason,
|
||||||
|
"durationS": duration_s,
|
||||||
|
"sttMs": stt_ms,
|
||||||
|
"voice": sess.voice,
|
||||||
|
"speed": sess.speed,
|
||||||
|
"interrupted": sess.interrupted,
|
||||||
|
}
|
||||||
|
if sess.location:
|
||||||
|
payload["location"] = sess.location
|
||||||
|
await _send(self._ws, "stt_endpoint", payload)
|
||||||
|
await _send(self._ws, "stt_stream_done", {
|
||||||
|
"requestId": sess.request_id,
|
||||||
|
"audioRequestId": sess.audio_request_id,
|
||||||
|
"text": final_text,
|
||||||
|
"reason": reason,
|
||||||
|
})
|
||||||
|
self.drop(sess.request_id)
|
||||||
|
|
||||||
|
|
||||||
|
async def _broadcast_status(ws, state: str, **extra) -> None:
|
||||||
|
payload = {"service": "voxtral", "state": state}
|
||||||
|
payload.update(extra)
|
||||||
|
await _send(ws, "service_status", payload)
|
||||||
|
|
||||||
|
|
||||||
|
async def _worker_register(ws, *, model: str = "", busy_fn=None) -> None:
|
||||||
|
"""Meldet diesen Worker bei der aria-bridge an (worker_hello) und haelt die
|
||||||
|
Flotten-Registry per periodischem worker_ping (mit busy-Status) frisch."""
|
||||||
|
def _hello():
|
||||||
|
return {"instanceId": INSTANCE_ID, "service": WORKER_SERVICE,
|
||||||
|
"node": NODE_NAME, "gpus": GPU_IDS, "model": model}
|
||||||
|
try:
|
||||||
|
await _send(ws, "worker_hello", _hello())
|
||||||
|
n = 0
|
||||||
|
while True:
|
||||||
|
await asyncio.sleep(WORKER_PING_INTERVAL_S)
|
||||||
|
n += 1
|
||||||
|
busy = bool(busy_fn()) if busy_fn else False
|
||||||
|
await _send(ws, "worker_ping", {"instanceId": INSTANCE_ID, "busy": busy})
|
||||||
|
if n % 3 == 0: # ~30s worker_hello wiederholen (wie der Satellit)
|
||||||
|
await _send(ws, "worker_hello", _hello())
|
||||||
|
except asyncio.CancelledError:
|
||||||
|
raise
|
||||||
|
except Exception:
|
||||||
|
return # Socket tot → still beenden; run_loop reconnectet + startet neu
|
||||||
|
|
||||||
|
|
||||||
|
async def run_loop(sessions: SessionManager) -> None:
|
||||||
|
use_tls = RVS_TLS
|
||||||
|
retry_s = 2
|
||||||
|
tls_fallback_tried = False
|
||||||
|
while True:
|
||||||
|
scheme = "wss" if use_tls else "ws"
|
||||||
|
url = f"{scheme}://{RVS_HOST}:{RVS_PORT}?token={RVS_TOKEN}"
|
||||||
|
masked = url.replace(RVS_TOKEN, "***") if RVS_TOKEN else url
|
||||||
|
try:
|
||||||
|
logger.info("Verbinde zu RVS: %s", masked)
|
||||||
|
async with websockets.connect(url, ping_interval=20, ping_timeout=10,
|
||||||
|
max_size=50 * 1024 * 1024) as ws:
|
||||||
|
logger.info("RVS verbunden")
|
||||||
|
retry_s = 2
|
||||||
|
tls_fallback_tried = False
|
||||||
|
sessions.attach_ws(ws)
|
||||||
|
await _broadcast_status(ws, "ready", model=VOXTRAL_MODEL)
|
||||||
|
await _send(ws, "config_request", {"service": "voxtral"})
|
||||||
|
ping_task = asyncio.create_task(_worker_register(
|
||||||
|
ws, model=VOXTRAL_MODEL,
|
||||||
|
busy_fn=lambda: bool(sessions._sessions)))
|
||||||
|
while True:
|
||||||
|
try:
|
||||||
|
raw = await asyncio.wait_for(ws.recv(), timeout=RX_STALE_S)
|
||||||
|
except asyncio.TimeoutError:
|
||||||
|
logger.warning("Kein RVS-Traffic seit %ds — Verbindung halb-tot, reconnect", RX_STALE_S)
|
||||||
|
raise ConnectionError("rvs-stale")
|
||||||
|
try:
|
||||||
|
msg = json.loads(raw)
|
||||||
|
except Exception:
|
||||||
|
continue
|
||||||
|
mtype = msg.get("type", "")
|
||||||
|
payload = msg.get("payload", {}) or {}
|
||||||
|
# Redundanz-Routing: ist die Anfrage gezielt an eine andere
|
||||||
|
# Instanz adressiert, ignorieren. Ohne targetInstance (Feld
|
||||||
|
# fehlt) → wie bisher, jeder Worker nimmt sie an.
|
||||||
|
tgt = payload.get("targetInstance")
|
||||||
|
if tgt and tgt != INSTANCE_ID:
|
||||||
|
continue
|
||||||
|
# Auslastungs-Monitor (node_stats_*) abfangen.
|
||||||
|
if await _stats.handle(ws, mtype, payload, _send):
|
||||||
|
continue
|
||||||
|
if mtype == "stt_stream_start":
|
||||||
|
sessions.start_session(payload)
|
||||||
|
elif mtype == "stt_audio_chunk":
|
||||||
|
sessions.feed_chunk(payload)
|
||||||
|
elif mtype == "stt_stream_end":
|
||||||
|
sessions.end_session(payload.get("requestId", ""))
|
||||||
|
elif mtype == "stt_transcribe_blob":
|
||||||
|
# One-Shot-Transkription eines PCM-Schnipsels (kein Live-
|
||||||
|
# Stream) — fuer die Wake-Wort-Bestaetigung: die App schickt
|
||||||
|
# den Vor-Trigger-Audio, wir sagen was gesagt wurde, die App
|
||||||
|
# prueft ob "Computer" drin ist. Silero vorgeschaltet:
|
||||||
|
# Musik/Rauschen → leerer Text (nicht bestaetigt).
|
||||||
|
req_id = payload.get("requestId", "")
|
||||||
|
try:
|
||||||
|
pcm = base64.b64decode(payload.get("pcm", ""))
|
||||||
|
audio = pcm_s16le_to_float32(pcm)
|
||||||
|
segs = _speech_segments(audio)
|
||||||
|
if segs is not None and len(segs) == 0:
|
||||||
|
text = ""
|
||||||
|
else:
|
||||||
|
text = (await sessions.runner.transcribe(
|
||||||
|
audio, payload.get("language", "de"))).strip()
|
||||||
|
logger.info("stt_transcribe_blob (%.1fs) → %r",
|
||||||
|
audio.size / 16000.0, text[:60])
|
||||||
|
await _send(ws, "stt_transcribe_result",
|
||||||
|
{"requestId": req_id, "text": text})
|
||||||
|
except Exception as exc:
|
||||||
|
logger.warning("stt_transcribe_blob fehlgeschlagen: %s", exc)
|
||||||
|
await _send(ws, "stt_transcribe_result",
|
||||||
|
{"requestId": req_id, "text": "", "error": str(exc)[:200]})
|
||||||
|
elif mtype == "voice_id_status_request":
|
||||||
|
req_id = payload.get("requestId", "")
|
||||||
|
try:
|
||||||
|
status = speaker_id.status()
|
||||||
|
await _send(ws, "voice_id_status_response",
|
||||||
|
{"requestId": req_id, "ok": True, **status})
|
||||||
|
except Exception as exc:
|
||||||
|
await _send(ws, "voice_id_status_response",
|
||||||
|
{"requestId": req_id, "ok": False, "error": str(exc)[:200]})
|
||||||
|
elif mtype == "voice_id_enroll_request":
|
||||||
|
req_id = payload.get("requestId", "")
|
||||||
|
samples = payload.get("samples") or []
|
||||||
|
logger.info("voice_id_enroll_request: %d Samples (id=%s)", len(samples), req_id[:8])
|
||||||
|
try:
|
||||||
|
result = await asyncio.get_running_loop().run_in_executor(
|
||||||
|
None, speaker_id.enroll_from_samples, samples)
|
||||||
|
await _send(ws, "voice_id_enroll_response", {
|
||||||
|
"requestId": req_id, "ok": True,
|
||||||
|
"sample_count": result.get("sample_count", 0),
|
||||||
|
"rejected": result.get("rejected", []),
|
||||||
|
"updated_at": result.get("updated_at"),
|
||||||
|
"embedding_dim": result.get("embedding_dim"),
|
||||||
|
})
|
||||||
|
except Exception as exc:
|
||||||
|
logger.warning("voice_id_enroll failed: %s", exc)
|
||||||
|
await _send(ws, "voice_id_enroll_response",
|
||||||
|
{"requestId": req_id, "ok": False, "error": str(exc)[:300]})
|
||||||
|
elif mtype == "voice_id_delete_request":
|
||||||
|
req_id = payload.get("requestId", "")
|
||||||
|
removed = speaker_id.delete_fingerprint()
|
||||||
|
await _send(ws, "voice_id_delete_response",
|
||||||
|
{"requestId": req_id, "ok": True, "removed": removed})
|
||||||
|
elif mtype == "config":
|
||||||
|
if "voiceIdThreshold" in payload:
|
||||||
|
try:
|
||||||
|
t = float(payload.get("voiceIdThreshold", 0.5))
|
||||||
|
if 0.0 <= t <= 1.0:
|
||||||
|
speaker_id.DEFAULT_THRESHOLD = t
|
||||||
|
logger.info("[speaker-id] threshold gesetzt: %.2f", t)
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
pass
|
||||||
|
if "voiceIdEnabled" in payload:
|
||||||
|
_set_speaker_id_enabled(payload.get("voiceIdEnabled"))
|
||||||
|
logger.info("[speaker-id] Gating %s (voiceIdEnabled)",
|
||||||
|
"AN" if SPEAKER_ID_ENABLED else "AUS")
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning("RVS-Verbindung verloren: %s — retry in %ds", e, retry_s)
|
||||||
|
try:
|
||||||
|
ping_task.cancel()
|
||||||
|
except NameError:
|
||||||
|
pass
|
||||||
|
if use_tls and RVS_TLS_FALLBACK and not tls_fallback_tried:
|
||||||
|
use_tls = False
|
||||||
|
tls_fallback_tried = True
|
||||||
|
continue
|
||||||
|
await asyncio.sleep(retry_s)
|
||||||
|
retry_s = min(retry_s * 2, 30)
|
||||||
|
use_tls = RVS_TLS
|
||||||
|
|
||||||
|
|
||||||
|
async def main() -> None:
|
||||||
|
if not RVS_HOST or not RVS_TOKEN:
|
||||||
|
logger.error("RVS_HOST/RVS_TOKEN fehlen — .env pruefen. Abbruch.")
|
||||||
|
return
|
||||||
|
runner = VoxtralRunner()
|
||||||
|
loop = asyncio.get_running_loop()
|
||||||
|
await loop.run_in_executor(None, runner.load) # Modell laden (blockierend)
|
||||||
|
sessions = SessionManager(runner)
|
||||||
|
logger.info("Voxtral-Bridge startet — Modell=%s", VOXTRAL_MODEL)
|
||||||
|
asyncio.create_task(_stats.run_sampler()) # Auslastungs-Sampler (Stage E)
|
||||||
|
await asyncio.gather(run_loop(sessions), sessions.run_endpointer())
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
try:
|
||||||
|
asyncio.run(main())
|
||||||
|
except KeyboardInterrupt:
|
||||||
|
pass
|
||||||
@@ -0,0 +1,167 @@
|
|||||||
|
"""
|
||||||
|
ARIA Node-Stats — Auslastungs-Monitor pro Box (GPU + optional Tokens).
|
||||||
|
|
||||||
|
Identische Kopie in jedem Worker-Build-Context (f5tts/whisper/voxtral/llm-adapter),
|
||||||
|
weil jeder Worker ein eigener Docker-Build-Context ist.
|
||||||
|
|
||||||
|
Aufgaben:
|
||||||
|
- Sampler-Loop (alle SAMPLE_SEC): nvidia-smi-Auslastung + Token-Delta → Ringpuffer
|
||||||
|
(persistent als JSON auf der Box). Laeuft unabhaengig vom Modal.
|
||||||
|
- Live-Stream: bei node_stats_stream_start jede Sekunde rohes nvidia-smi + Werte
|
||||||
|
senden (bis stop / Auto-Timeout).
|
||||||
|
- History-Request + Reset (Besen).
|
||||||
|
|
||||||
|
Reicht `handle(ws, mtype, payload)` in die Worker-Message-Loop ein; gibt True
|
||||||
|
zurueck, wenn die Nachricht eine node_stats_*-Nachricht war.
|
||||||
|
"""
|
||||||
|
import asyncio
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import time
|
||||||
|
|
||||||
|
SAMPLE_SEC = int(os.getenv("STATS_SAMPLE_SEC", "15"))
|
||||||
|
HISTORY_CAP = int(os.getenv("STATS_HISTORY_CAP", "500")) # ~2h bei 15s
|
||||||
|
STREAM_MAX_SEC = int(os.getenv("STATS_STREAM_MAX_SEC", "300"))
|
||||||
|
|
||||||
|
|
||||||
|
async def _run_cmd(*args, timeout=8) -> str:
|
||||||
|
"""Fuehrt ein Kommando aus, gibt stdout (str) zurueck; '' bei Fehler."""
|
||||||
|
try:
|
||||||
|
proc = await asyncio.create_subprocess_exec(
|
||||||
|
*args,
|
||||||
|
stdout=asyncio.subprocess.PIPE,
|
||||||
|
stderr=asyncio.subprocess.DEVNULL,
|
||||||
|
)
|
||||||
|
out, _ = await asyncio.wait_for(proc.communicate(), timeout=timeout)
|
||||||
|
return (out or b"").decode("utf-8", "replace")
|
||||||
|
except Exception:
|
||||||
|
return ""
|
||||||
|
|
||||||
|
|
||||||
|
class NodeStats:
|
||||||
|
def __init__(self, instance_id: str, node_name: str, history_path: str,
|
||||||
|
token_getter=None, logger=None):
|
||||||
|
self.instance_id = instance_id
|
||||||
|
self.node_name = node_name
|
||||||
|
self.history_path = history_path
|
||||||
|
self.token_getter = token_getter # callable -> kumulative Token-Zahl (oder None)
|
||||||
|
self.log = logger
|
||||||
|
self.samples = self._load()
|
||||||
|
self._last_tokens = self._tokens_now()
|
||||||
|
self._stream_task = None
|
||||||
|
|
||||||
|
# ── Persistenz ──────────────────────────────────────────
|
||||||
|
def _load(self) -> list:
|
||||||
|
try:
|
||||||
|
with open(self.history_path) as f:
|
||||||
|
data = json.load(f)
|
||||||
|
return data if isinstance(data, list) else []
|
||||||
|
except Exception:
|
||||||
|
return []
|
||||||
|
|
||||||
|
def _persist(self) -> None:
|
||||||
|
try:
|
||||||
|
os.makedirs(os.path.dirname(self.history_path) or ".", exist_ok=True)
|
||||||
|
tmp = self.history_path + ".tmp"
|
||||||
|
with open(tmp, "w") as f:
|
||||||
|
json.dump(self.samples[-HISTORY_CAP:], f)
|
||||||
|
os.replace(tmp, self.history_path)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
def _tokens_now(self) -> int:
|
||||||
|
try:
|
||||||
|
return int(self.token_getter()) if self.token_getter else 0
|
||||||
|
except Exception:
|
||||||
|
return 0
|
||||||
|
|
||||||
|
# ── nvidia-smi ──────────────────────────────────────────
|
||||||
|
async def _query_gpu(self) -> dict:
|
||||||
|
"""Aggregierte GPU-Werte ueber alle sichtbaren Karten."""
|
||||||
|
out = await _run_cmd(
|
||||||
|
"nvidia-smi",
|
||||||
|
"--query-gpu=utilization.gpu,memory.used,memory.total",
|
||||||
|
"--format=csv,noheader,nounits")
|
||||||
|
utils, used, total = [], 0, 0
|
||||||
|
for line in out.strip().splitlines():
|
||||||
|
parts = [p.strip() for p in line.split(",")]
|
||||||
|
if len(parts) < 3:
|
||||||
|
continue
|
||||||
|
try:
|
||||||
|
utils.append(float(parts[0]))
|
||||||
|
used += float(parts[1])
|
||||||
|
total += float(parts[2])
|
||||||
|
except ValueError:
|
||||||
|
continue
|
||||||
|
gpu = round(sum(utils) / len(utils), 1) if utils else 0.0
|
||||||
|
return {"gpu": gpu, "memUsed": int(used), "memTotal": int(total)}
|
||||||
|
|
||||||
|
async def _nvidia_smi_text(self) -> str:
|
||||||
|
txt = await _run_cmd("nvidia-smi")
|
||||||
|
return txt or "nvidia-smi nicht verfuegbar"
|
||||||
|
|
||||||
|
# ── Sampler (Verlauf) ───────────────────────────────────
|
||||||
|
async def run_sampler(self) -> None:
|
||||||
|
while True:
|
||||||
|
try:
|
||||||
|
g = await self._query_gpu()
|
||||||
|
now_tok = self._tokens_now()
|
||||||
|
dtok = max(0, now_tok - self._last_tokens)
|
||||||
|
self._last_tokens = now_tok
|
||||||
|
self.samples.append({
|
||||||
|
"ts": int(time.time()),
|
||||||
|
"gpu": g["gpu"], "memUsed": g["memUsed"],
|
||||||
|
"memTotal": g["memTotal"], "tokens": dtok,
|
||||||
|
})
|
||||||
|
if len(self.samples) > HISTORY_CAP:
|
||||||
|
self.samples = self.samples[-HISTORY_CAP:]
|
||||||
|
self._persist()
|
||||||
|
except Exception as e:
|
||||||
|
if self.log:
|
||||||
|
self.log.debug("node_stats sample fehlgeschlagen: %s", e)
|
||||||
|
await asyncio.sleep(SAMPLE_SEC)
|
||||||
|
|
||||||
|
# ── Live-Stream ─────────────────────────────────────────
|
||||||
|
async def _stream(self, ws, send) -> None:
|
||||||
|
t0 = time.time()
|
||||||
|
try:
|
||||||
|
while time.time() - t0 < STREAM_MAX_SEC:
|
||||||
|
g = await self._query_gpu()
|
||||||
|
smi = await self._nvidia_smi_text()
|
||||||
|
await send(ws, "node_stats", {
|
||||||
|
"instanceId": self.instance_id, "node": self.node_name,
|
||||||
|
"nvidiaSmi": smi, **g,
|
||||||
|
})
|
||||||
|
await asyncio.sleep(1)
|
||||||
|
except asyncio.CancelledError:
|
||||||
|
raise
|
||||||
|
except Exception:
|
||||||
|
return
|
||||||
|
|
||||||
|
# ── Dispatch ────────────────────────────────────────────
|
||||||
|
async def handle(self, ws, mtype: str, payload: dict, send) -> bool:
|
||||||
|
if mtype == "node_stats_stream_start":
|
||||||
|
if self._stream_task and not self._stream_task.done():
|
||||||
|
self._stream_task.cancel()
|
||||||
|
self._stream_task = asyncio.create_task(self._stream(ws, send))
|
||||||
|
return True
|
||||||
|
if mtype == "node_stats_stream_stop":
|
||||||
|
if self._stream_task:
|
||||||
|
self._stream_task.cancel()
|
||||||
|
self._stream_task = None
|
||||||
|
return True
|
||||||
|
if mtype == "node_stats_history_request":
|
||||||
|
await send(ws, "node_stats_history", {
|
||||||
|
"instanceId": self.instance_id, "node": self.node_name,
|
||||||
|
"samples": self.samples[-HISTORY_CAP:],
|
||||||
|
"tokenCapable": self.token_getter is not None,
|
||||||
|
"sampleSec": SAMPLE_SEC,
|
||||||
|
})
|
||||||
|
return True
|
||||||
|
if mtype == "node_stats_reset":
|
||||||
|
self.samples = []
|
||||||
|
self._persist()
|
||||||
|
await send(ws, "node_stats_reset_done",
|
||||||
|
{"instanceId": self.instance_id, "node": self.node_name})
|
||||||
|
return True
|
||||||
|
return False
|
||||||
@@ -0,0 +1,11 @@
|
|||||||
|
# Voxtral-3B via Transformers. torch/torchaudio kommen cu124-gepinnt aus dem
|
||||||
|
# Dockerfile (nicht hier, sonst zieht pip das Default-CUDA-Wheel).
|
||||||
|
transformers>=4.54
|
||||||
|
mistral-common[audio]>=1.8.1
|
||||||
|
accelerate>=0.30
|
||||||
|
speechbrain>=1.0 # Speaker-ID (ECAPA-TDNN) — nur Stefans Stimme
|
||||||
|
silero-vad>=5.1 # neuronales VAD: echte Sprache vs Stille/Rauschen/Musik
|
||||||
|
soundfile>=0.12
|
||||||
|
librosa>=0.10 # VoxtralProcessor.load_audio_as nutzt librosa zum WAV-Laden
|
||||||
|
numpy>=1.24
|
||||||
|
websockets>=12.0
|
||||||
@@ -0,0 +1,272 @@
|
|||||||
|
"""
|
||||||
|
Speaker-ID Backend fuer ARIAs Stimmen-Erkennung.
|
||||||
|
|
||||||
|
Nutzt SpeechBrain ECAPA-TDNN (192-dim Embeddings, auf VoxCeleb-1+2 trainiert).
|
||||||
|
Fingerprint = gemittelter, L2-normalisierter Embedding-Vektor aus N
|
||||||
|
Enrollment-Samples. Verify: cosine_similarity(neue_aufnahme, fingerprint).
|
||||||
|
|
||||||
|
Persistenz: /voice-id/fingerprint.json (Float-Liste + Metadaten).
|
||||||
|
Modell-Cache: /root/.cache/huggingface/ (Bind-Mount mit f5tts geteilt).
|
||||||
|
|
||||||
|
Verhalten OHNE Enrollment (kein Fingerprint vorhanden):
|
||||||
|
verify() → (True, 0.0) — Fail-open, damit Speaker-ID-Gating den
|
||||||
|
ungeenrollten Brain-Pfad nicht versehentlich blockiert.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import base64
|
||||||
|
import json
|
||||||
|
import logging
|
||||||
|
import os
|
||||||
|
import time
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Optional
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
VOICE_ID_DIR = Path(os.environ.get("VOICE_ID_DIR", "/voice-id"))
|
||||||
|
FINGERPRINT_FILE = VOICE_ID_DIR / "fingerprint.json"
|
||||||
|
|
||||||
|
# Cosine-Threshold: 0.5 ist konservativ (wenig false-positives), 0.3 ist
|
||||||
|
# locker (mehr Treffer auch bei Nebengeraeuschen). Stefan kann's per
|
||||||
|
# Diagnostic-Setting feintunen.
|
||||||
|
DEFAULT_THRESHOLD = 0.5
|
||||||
|
|
||||||
|
# Minimal-Sample-Laenge fuer ein verlaessliches Embedding (~1s @ 16kHz int16 = 32000 bytes)
|
||||||
|
MIN_SAMPLE_BYTES = 32000
|
||||||
|
|
||||||
|
_model = None
|
||||||
|
|
||||||
|
|
||||||
|
def _ensure_loaded():
|
||||||
|
"""Lazy-Load des ECAPA-TDNN. Holt das Modell beim ersten Aufruf von HF;
|
||||||
|
danach cached im HF-Cache-Volume. Erste Init: ~30s download + load,
|
||||||
|
danach <1s warm. Wirft bei Fehler — Caller muss catchen + fail-open."""
|
||||||
|
global _model
|
||||||
|
if _model is not None:
|
||||||
|
return _model
|
||||||
|
import torch
|
||||||
|
from speechbrain.inference.speaker import EncoderClassifier
|
||||||
|
device = "cuda" if torch.cuda.is_available() else "cpu"
|
||||||
|
logger.info("[speaker-id] loading ECAPA-TDNN on %s ...", device)
|
||||||
|
_model = EncoderClassifier.from_hparams(
|
||||||
|
source="speechbrain/spkrec-ecapa-voxceleb",
|
||||||
|
savedir="/root/.cache/huggingface/speechbrain-ecapa",
|
||||||
|
run_opts={"device": device},
|
||||||
|
)
|
||||||
|
logger.info("[speaker-id] model ready (device=%s)", device)
|
||||||
|
return _model
|
||||||
|
|
||||||
|
|
||||||
|
def _decode_compressed_to_pcm(audio_bytes: bytes) -> bytes:
|
||||||
|
"""Dekodiert komprimiertes Audio (MP4/M4A/AAC vom Android-Recorder) via ffmpeg
|
||||||
|
(im Container vorhanden) auf rohes 16kHz mono int16 LE PCM. Input geht ueber
|
||||||
|
eine Temp-Datei (nicht Pipe): Androids MediaRecorder legt das moov-Atom ans
|
||||||
|
ENDE, das braucht seekbaren Input, sonst 'moov atom not found'."""
|
||||||
|
import os
|
||||||
|
import subprocess
|
||||||
|
import tempfile
|
||||||
|
tmp = None
|
||||||
|
try:
|
||||||
|
with tempfile.NamedTemporaryFile(suffix=".mp4", delete=False) as tf:
|
||||||
|
tf.write(audio_bytes)
|
||||||
|
tmp = tf.name
|
||||||
|
proc = subprocess.run(
|
||||||
|
["ffmpeg", "-hide_banner", "-loglevel", "error", "-i", tmp,
|
||||||
|
"-f", "s16le", "-ac", "1", "-ar", "16000", "pipe:1"],
|
||||||
|
stdout=subprocess.PIPE, stderr=subprocess.PIPE,
|
||||||
|
)
|
||||||
|
if proc.returncode != 0 or not proc.stdout:
|
||||||
|
raise ValueError(
|
||||||
|
f"ffmpeg decode failed: {proc.stderr.decode('utf-8', 'ignore')[:200]}")
|
||||||
|
return proc.stdout
|
||||||
|
finally:
|
||||||
|
if tmp:
|
||||||
|
try:
|
||||||
|
os.unlink(tmp)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
def _normalize_audio_bytes(audio_bytes: bytes) -> bytes:
|
||||||
|
"""Akzeptiert rohes 16kHz int16 LE PCM, eine WAV-Datei (RIFF/WAVE) ODER einen
|
||||||
|
komprimierten MP4/M4A/AAC-Container (Android-Recorder). WAV → Header strippen +
|
||||||
|
Format validieren; MP4/AAC → via ffmpeg dekodieren. Ergebnis: rohes PCM."""
|
||||||
|
if (len(audio_bytes) >= 44
|
||||||
|
and audio_bytes[:4] == b"RIFF"
|
||||||
|
and audio_bytes[8:12] == b"WAVE"):
|
||||||
|
import io
|
||||||
|
import wave
|
||||||
|
with wave.open(io.BytesIO(audio_bytes), "rb") as wav:
|
||||||
|
sr = wav.getframerate()
|
||||||
|
ch = wav.getnchannels()
|
||||||
|
sw = wav.getsampwidth()
|
||||||
|
if sr != 16000:
|
||||||
|
raise ValueError(f"WAV-Samplerate {sr} != 16000")
|
||||||
|
if ch != 1:
|
||||||
|
raise ValueError(f"WAV-Kanalzahl {ch} != 1 (mono erwartet)")
|
||||||
|
if sw != 2:
|
||||||
|
raise ValueError(f"WAV-Sampleweite {sw} != 2 (int16 erwartet)")
|
||||||
|
return wav.readframes(wav.getnframes())
|
||||||
|
# MP4/M4A/AAC-Container: Android-AAC-Recorder legt 'ftyp' bei Offset 4 an.
|
||||||
|
if len(audio_bytes) >= 12 and audio_bytes[4:8] == b"ftyp":
|
||||||
|
return _decode_compressed_to_pcm(audio_bytes)
|
||||||
|
return audio_bytes
|
||||||
|
|
||||||
|
|
||||||
|
def _audio_bytes_to_tensor(audio_bytes: bytes):
|
||||||
|
"""int16 LE PCM (16kHz mono) → Torch-Tensor (1, N), normalisiert auf [-1, 1].
|
||||||
|
WAV wird vorher auf rohes PCM reduziert (Header strippen)."""
|
||||||
|
import torch
|
||||||
|
raw = _normalize_audio_bytes(audio_bytes)
|
||||||
|
arr = np.frombuffer(raw, dtype=np.int16).astype(np.float32) / 32768.0
|
||||||
|
return torch.from_numpy(arr).unsqueeze(0)
|
||||||
|
|
||||||
|
|
||||||
|
def embed(audio_bytes: bytes) -> np.ndarray:
|
||||||
|
"""Berechnet das Speaker-Embedding fuer einen Audio-Chunk.
|
||||||
|
Erwartet 16kHz int16 LE PCM Mono. Returns 192-dim numpy float32."""
|
||||||
|
import torch
|
||||||
|
model = _ensure_loaded()
|
||||||
|
wav = _audio_bytes_to_tensor(audio_bytes)
|
||||||
|
with torch.no_grad():
|
||||||
|
emb = model.encode_batch(wav)
|
||||||
|
return emb.squeeze().cpu().numpy().astype(np.float32)
|
||||||
|
|
||||||
|
|
||||||
|
def cosine_similarity(a: np.ndarray, b: np.ndarray) -> float:
|
||||||
|
"""Kosinus-Aehnlichkeit zwischen zwei 1D-Vektoren, Range [-1, 1].
|
||||||
|
Hoeher = aehnlicher. Bei normalisierten Vektoren ist das gleich dem Skalarprodukt."""
|
||||||
|
na = np.linalg.norm(a)
|
||||||
|
nb = np.linalg.norm(b)
|
||||||
|
if na < 1e-9 or nb < 1e-9:
|
||||||
|
return 0.0
|
||||||
|
return float(np.dot(a, b) / (na * nb))
|
||||||
|
|
||||||
|
|
||||||
|
def save_fingerprint(embeddings: list[np.ndarray], sample_durations_s: list[float]) -> dict:
|
||||||
|
"""Mittelt + L2-normalisiert die Embeddings und schreibt sie nach
|
||||||
|
FINGERPRINT_FILE. Returns das gespeicherte Dict."""
|
||||||
|
if not embeddings:
|
||||||
|
raise ValueError("Keine Embeddings zum Speichern")
|
||||||
|
VOICE_ID_DIR.mkdir(parents=True, exist_ok=True)
|
||||||
|
stacked = np.stack(embeddings)
|
||||||
|
mean = stacked.mean(axis=0)
|
||||||
|
mean = mean / max(np.linalg.norm(mean), 1e-9)
|
||||||
|
data = {
|
||||||
|
"version": 1,
|
||||||
|
"embedding": mean.tolist(),
|
||||||
|
"embedding_dim": int(mean.shape[0]),
|
||||||
|
"sample_count": len(embeddings),
|
||||||
|
"sample_durations_s": [float(s) for s in sample_durations_s],
|
||||||
|
"updated_at": int(time.time()),
|
||||||
|
}
|
||||||
|
FINGERPRINT_FILE.write_text(json.dumps(data, indent=2), encoding="utf-8")
|
||||||
|
logger.info("[speaker-id] fingerprint gespeichert: %d Samples, dim=%d, total_s=%.1f",
|
||||||
|
len(embeddings), mean.shape[0], sum(sample_durations_s))
|
||||||
|
return data
|
||||||
|
|
||||||
|
|
||||||
|
def load_fingerprint() -> Optional[dict]:
|
||||||
|
"""Returns das Fingerprint-Dict oder None wenn noch nicht enrolled."""
|
||||||
|
if not FINGERPRINT_FILE.exists():
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
return json.loads(FINGERPRINT_FILE.read_text(encoding="utf-8"))
|
||||||
|
except Exception as exc:
|
||||||
|
logger.warning("[speaker-id] fingerprint laden fehlgeschlagen: %s", exc)
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def delete_fingerprint() -> bool:
|
||||||
|
"""Loescht den Fingerprint (z.B. fuer Re-Enrollment). True wenn was weg ist."""
|
||||||
|
if FINGERPRINT_FILE.exists():
|
||||||
|
FINGERPRINT_FILE.unlink()
|
||||||
|
logger.info("[speaker-id] fingerprint geloescht")
|
||||||
|
return True
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def verify(audio_bytes: bytes, threshold: Optional[float] = None) -> tuple[bool, float]:
|
||||||
|
"""Returns (is_match, similarity).
|
||||||
|
|
||||||
|
Wenn threshold=None: nutzt den Modul-Default (DEFAULT_THRESHOLD) — der wird
|
||||||
|
vom config-Broadcast zur Laufzeit auf den Diagnostic-Slider-Wert gesetzt.
|
||||||
|
Default-Arg-Bindung waere zur Def-Zeit, also bewusst None statt direkt.
|
||||||
|
|
||||||
|
Fail-open: wenn kein Fingerprint vorhanden ist oder das Embedding-Modell
|
||||||
|
crasht, returnt (True, 0.0) — kein Filtering. Sonst wuerde ein kaputter
|
||||||
|
Speaker-ID-Service die ganze Aufnahme blockieren."""
|
||||||
|
if threshold is None:
|
||||||
|
threshold = DEFAULT_THRESHOLD
|
||||||
|
fp = load_fingerprint()
|
||||||
|
if fp is None:
|
||||||
|
return True, 0.0
|
||||||
|
if len(audio_bytes) < MIN_SAMPLE_BYTES:
|
||||||
|
# Zu wenig Audio fuer ein verlaessliches Embedding → durchlassen
|
||||||
|
return True, 0.0
|
||||||
|
try:
|
||||||
|
saved_emb = np.array(fp["embedding"], dtype=np.float32)
|
||||||
|
new_emb = embed(audio_bytes)
|
||||||
|
except Exception as exc:
|
||||||
|
logger.warning("[speaker-id] verify embed failed: %s — fail-open", exc)
|
||||||
|
return True, 0.0
|
||||||
|
sim = cosine_similarity(new_emb, saved_emb)
|
||||||
|
return sim >= threshold, sim
|
||||||
|
|
||||||
|
|
||||||
|
def status() -> dict:
|
||||||
|
"""Status-Snapshot fuer die App / Diagnostic."""
|
||||||
|
fp = load_fingerprint()
|
||||||
|
return {
|
||||||
|
"enrolled": fp is not None,
|
||||||
|
"sample_count": fp.get("sample_count", 0) if fp else 0,
|
||||||
|
"sample_durations_s": fp.get("sample_durations_s", []) if fp else [],
|
||||||
|
"updated_at": fp.get("updated_at") if fp else None,
|
||||||
|
"embedding_dim": fp.get("embedding_dim") if fp else None,
|
||||||
|
"default_threshold": DEFAULT_THRESHOLD,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def enroll_from_samples(samples_b64: list[str]) -> dict:
|
||||||
|
"""Verarbeitet base64-Samples (16kHz int16 LE PCM Mono) zu einem neuen
|
||||||
|
Fingerprint. Returns Status-Dict. Wirft ValueError wenn nichts brauchbar ist."""
|
||||||
|
if not samples_b64:
|
||||||
|
raise ValueError("Keine Samples uebergeben")
|
||||||
|
embeddings: list[np.ndarray] = []
|
||||||
|
durations: list[float] = []
|
||||||
|
rejected: list[dict] = []
|
||||||
|
for idx, s in enumerate(samples_b64):
|
||||||
|
try:
|
||||||
|
raw = base64.b64decode(s)
|
||||||
|
except Exception as exc:
|
||||||
|
rejected.append({"index": idx, "reason": f"base64: {exc}"})
|
||||||
|
continue
|
||||||
|
# Erst dekodieren (WAV/MP4/AAC → rohes PCM), DANN Laenge pruefen: der
|
||||||
|
# Android-Recorder liefert komprimiertes MP4, dessen Byte-Laenge nichts
|
||||||
|
# ueber die Dauer sagt (4s AAC < 32KB → faelschlich "zu kurz").
|
||||||
|
try:
|
||||||
|
pcm = _normalize_audio_bytes(raw)
|
||||||
|
except Exception as exc:
|
||||||
|
rejected.append({"index": idx, "reason": f"decode: {exc}"})
|
||||||
|
continue
|
||||||
|
if len(pcm) < MIN_SAMPLE_BYTES:
|
||||||
|
rejected.append({"index": idx, "reason": f"zu kurz ({len(pcm)} bytes PCM)"})
|
||||||
|
continue
|
||||||
|
try:
|
||||||
|
emb = embed(pcm)
|
||||||
|
embeddings.append(emb)
|
||||||
|
durations.append(len(pcm) / 2 / 16000.0)
|
||||||
|
except Exception as exc:
|
||||||
|
rejected.append({"index": idx, "reason": f"embed: {exc}"})
|
||||||
|
if not embeddings:
|
||||||
|
raise ValueError(
|
||||||
|
f"Keine Samples konnten verarbeitet werden ({len(rejected)} rejected). "
|
||||||
|
f"Details: {rejected[:3]}"
|
||||||
|
)
|
||||||
|
fingerprint = save_fingerprint(embeddings, durations)
|
||||||
|
fingerprint["rejected"] = rejected
|
||||||
|
return fingerprint
|
||||||
@@ -17,6 +17,6 @@ RUN pip3 install --no-cache-dir torch==2.3.1 torchaudio==2.3.1 \
|
|||||||
COPY requirements.txt .
|
COPY requirements.txt .
|
||||||
RUN pip3 install --no-cache-dir -r requirements.txt
|
RUN pip3 install --no-cache-dir -r requirements.txt
|
||||||
|
|
||||||
COPY bridge.py speaker_id.py ./
|
COPY bridge.py speaker_id.py node_stats.py ./
|
||||||
|
|
||||||
CMD ["python3", "bridge.py"]
|
CMD ["python3", "bridge.py"]
|
||||||
|
|||||||
+135
-7
@@ -1,6 +1,6 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
"""
|
"""
|
||||||
ARIA Whisper Bridge — laeuft auf der Gamebox (RTX 3060).
|
ARIA Whisper Bridge — laeuft auf der AI-Box (RTX 3060).
|
||||||
|
|
||||||
Zwei Modi:
|
Zwei Modi:
|
||||||
|
|
||||||
@@ -59,6 +59,23 @@ WHISPER_DEVICE = os.getenv("WHISPER_DEVICE", "cuda")
|
|||||||
WHISPER_COMPUTE_TYPE = os.getenv("WHISPER_COMPUTE_TYPE", "float16")
|
WHISPER_COMPUTE_TYPE = os.getenv("WHISPER_COMPUTE_TYPE", "float16")
|
||||||
WHISPER_LANGUAGE = os.getenv("WHISPER_LANGUAGE", "de")
|
WHISPER_LANGUAGE = os.getenv("WHISPER_LANGUAGE", "de")
|
||||||
|
|
||||||
|
# ── Compute-Fleet: Worker-Identitaet & Registrierung ──────────────
|
||||||
|
# Meldet sich bei der aria-bridge (worker_hello) + periodischer worker_ping.
|
||||||
|
NODE_NAME = os.getenv("NODE_NAME", "node").strip() or "node"
|
||||||
|
GPU_IDS = os.getenv("NVIDIA_VISIBLE_DEVICES", "").strip()
|
||||||
|
WORKER_SERVICE = "whisper"
|
||||||
|
INSTANCE_ID = f"{WORKER_SERVICE}@{NODE_NAME}"
|
||||||
|
WORKER_PING_INTERVAL_S = int(os.getenv("WORKER_PING_INTERVAL_S", "10"))
|
||||||
|
# Empfangs-Watchdog: kommt in RX_STALE_S kein Broadcast rein (ein echter Raum hat
|
||||||
|
# staendig Traffic, z.B. sat_hello alle 25s / Brain-Polling), gilt die Verbindung
|
||||||
|
# als halb-tot (Caddy pongt die WS-Pings selbst) -> Zwangs-Reconnect.
|
||||||
|
RX_STALE_S = int(os.getenv("RX_STALE_S", "60"))
|
||||||
|
|
||||||
|
# ── Auslastungs-Monitor (Stage E) ──────────────────────────
|
||||||
|
import node_stats
|
||||||
|
STATS_PATH = os.getenv("STATS_PATH", f"/root/.cache/huggingface/aria_stats_{WORKER_SERVICE}.json")
|
||||||
|
_stats = node_stats.NodeStats(INSTANCE_ID, NODE_NAME, STATS_PATH, logger=logger)
|
||||||
|
|
||||||
ALLOWED_MODELS = {"tiny", "base", "small", "medium", "large-v3"}
|
ALLOWED_MODELS = {"tiny", "base", "small", "medium", "large-v3"}
|
||||||
|
|
||||||
# Streaming-Parameter (Defaults — koennen pro Session vom App-Payload ueberschrieben werden)
|
# Streaming-Parameter (Defaults — koennen pro Session vom App-Payload ueberschrieben werden)
|
||||||
@@ -75,7 +92,27 @@ STREAM_SESSION_TTL_S = 120 # tote Sessions nach 2 min aufraeumen
|
|||||||
# (Whisper oszilliert/halluziniert). Echte akustische Stille ist das robuste
|
# (Whisper oszilliert/halluziniert). Echte akustische Stille ist das robuste
|
||||||
# „User hat aufgehoert"-Signal.
|
# „User hat aufgehoert"-Signal.
|
||||||
STREAM_ENERGY_WINDOW_MS = 300 # RMS ueber die letzten 300ms Audio messen
|
STREAM_ENERGY_WINDOW_MS = 300 # RMS ueber die letzten 300ms Audio messen
|
||||||
STREAM_VOICE_RMS_THRESHOLD = 0.012 # RMS darueber = Sprache (haelt Session am Leben)
|
# --- Adaptiver Voice-Schwellwert (ersetzt die fixe 0.012-Grenze) ---
|
||||||
|
# Problem (Repro dokumentiert in audio.ts): eine FESTE RMS-Grenze schneidet
|
||||||
|
# leises/entferntes Sprechen faelschlich als „Stille" (Handy weiter weg vom Mund,
|
||||||
|
# ruhig im Auto, kurze Sprech-Pause) → Cut mitten im Satz. Loesung: die Grenze
|
||||||
|
# relativ zum gemessenen Rausch-Boden der Session fuehren. Sprache =
|
||||||
|
# noise_floor * Faktor, geklammert auf [MIN, MAX]. Bei noch ungelerntem Boden
|
||||||
|
# gilt MIN → sensibel, lieber nicht abschneiden.
|
||||||
|
STREAM_VOICE_FACTOR = 2.5 # Sprache = noise_floor * Faktor
|
||||||
|
STREAM_VOICE_RMS_MIN = 0.005 # Untergrenze (stiller Raum: nicht auf 0 kollabieren)
|
||||||
|
STREAM_VOICE_RMS_MAX = 0.020 # Obergrenze (lautes Auto: Sprache nie ganz aussperren)
|
||||||
|
STREAM_VOICE_RMS_THRESHOLD = 0.012 # Legacy-Konstante (nicht mehr im Cut-Pfad genutzt)
|
||||||
|
|
||||||
|
# Speaker-ID Gating global an/aus. DEFAULT AUS (fail-open) — bewusster Schalter
|
||||||
|
# ("nur meine Stimme"), kein Automatismus: ein schlechter Enroll darf nie die STT
|
||||||
|
# lahmlegen. Wird per config-Broadcast (voiceIdEnabled) zur Laufzeit gesetzt.
|
||||||
|
SPEAKER_ID_ENABLED = os.getenv("VOICE_ID_ENABLED", "false").lower() in ("1", "true", "yes")
|
||||||
|
|
||||||
|
|
||||||
|
def _set_speaker_id_enabled(val: bool) -> None:
|
||||||
|
global SPEAKER_ID_ENABLED
|
||||||
|
SPEAKER_ID_ENABLED = bool(val)
|
||||||
# Rein-semantischer Backstop: wenn die Energie NIE faellt (laute Umgebung,
|
# Rein-semantischer Backstop: wenn die Energie NIE faellt (laute Umgebung,
|
||||||
# z.B. Auto), endpointen wir trotzdem — aber erst nach diesem Faktor x
|
# z.B. Auto), endpointen wir trotzdem — aber erst nach diesem Faktor x
|
||||||
# endpoint_ms, damit normales Sprechen mit Pausen nicht abgeschnitten wird.
|
# endpoint_ms, damit normales Sprechen mit Pausen nicht abgeschnitten wird.
|
||||||
@@ -264,7 +301,7 @@ async def _send(ws, mtype: str, payload: dict) -> None:
|
|||||||
# ──────────────────────────────────────────────────────────────
|
# ──────────────────────────────────────────────────────────────
|
||||||
# DEBUG-LOG ueber RVS → /shared/logs/app.log
|
# DEBUG-LOG ueber RVS → /shared/logs/app.log
|
||||||
#
|
#
|
||||||
# Stefan's Gamebox ist Windows, kein SSH → wir brauchen Whisper-Bridge-
|
# Stefan's AI-Box ist Windows, kein SSH → wir brauchen Whisper-Bridge-
|
||||||
# Logs ueber den gleichen Pfad wie die App: app_log-Messages via RVS,
|
# Logs ueber den gleichen Pfad wie die App: app_log-Messages via RVS,
|
||||||
# aria-bridge schreibt sie in /shared/logs/app.log. Diagnostic / App-
|
# aria-bridge schreibt sie in /shared/logs/app.log. Diagnostic / App-
|
||||||
# Logs-Tab zeigen sie dann mit platform="whisper".
|
# Logs-Tab zeigen sie dann mit platform="whisper".
|
||||||
@@ -323,6 +360,10 @@ class StreamSession:
|
|||||||
last_growth_at: float = 0.0
|
last_growth_at: float = 0.0
|
||||||
last_transcribe_at: float = 0.0
|
last_transcribe_at: float = 0.0
|
||||||
last_voice_at: float = 0.0 # letzter Tick mit akustischer Sprach-Energie
|
last_voice_at: float = 0.0 # letzter Tick mit akustischer Sprach-Energie
|
||||||
|
noise_floor: float = 0.0 # adaptiver Rausch-Boden (0.0 = noch ungelernt)
|
||||||
|
voice_factor: float = STREAM_VOICE_FACTOR # per-Session konfigurierbar (Payload voiceFactor)
|
||||||
|
voice_rms_min: float = STREAM_VOICE_RMS_MIN
|
||||||
|
voice_rms_max: float = STREAM_VOICE_RMS_MAX
|
||||||
closed: bool = False # nach stream_end gesetzt
|
closed: bool = False # nach stream_end gesetzt
|
||||||
endpoint_sent: bool = False # Endpoint nur einmal feuern
|
endpoint_sent: bool = False # Endpoint nur einmal feuern
|
||||||
# Speaker-ID Gating: bei aktiviertem Fingerprint pruefen wir die ersten
|
# Speaker-ID Gating: bei aktiviertem Fingerprint pruefen wir die ersten
|
||||||
@@ -372,6 +413,10 @@ class SessionManager:
|
|||||||
speed = float(payload.get("speed") or 1.0)
|
speed = float(payload.get("speed") or 1.0)
|
||||||
except (TypeError, ValueError):
|
except (TypeError, ValueError):
|
||||||
speed = 1.0
|
speed = 1.0
|
||||||
|
try:
|
||||||
|
voice_factor = float(payload.get("voiceFactor") or STREAM_VOICE_FACTOR)
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
voice_factor = STREAM_VOICE_FACTOR
|
||||||
session = StreamSession(
|
session = StreamSession(
|
||||||
request_id=request_id,
|
request_id=request_id,
|
||||||
audio_request_id=payload.get("audioRequestId", "") or "",
|
audio_request_id=payload.get("audioRequestId", "") or "",
|
||||||
@@ -381,6 +426,7 @@ class SessionManager:
|
|||||||
hard_cap_ms=hard_cap_ms,
|
hard_cap_ms=hard_cap_ms,
|
||||||
voice=payload.get("voice", "") or "",
|
voice=payload.get("voice", "") or "",
|
||||||
speed=speed,
|
speed=speed,
|
||||||
|
voice_factor=voice_factor,
|
||||||
interrupted=bool(payload.get("interrupted", False)),
|
interrupted=bool(payload.get("interrupted", False)),
|
||||||
location=payload.get("location") or None,
|
location=payload.get("location") or None,
|
||||||
sample_rate=int(payload.get("sampleRate") or 16000),
|
sample_rate=int(payload.get("sampleRate") or 16000),
|
||||||
@@ -448,6 +494,11 @@ class SessionManager:
|
|||||||
Ohne Fingerprint → fail-open (match=True). Bei mismatch wird die
|
Ohne Fingerprint → fail-open (match=True). Bei mismatch wird die
|
||||||
Session sofort beendet mit synthetischem stt_endpoint."""
|
Session sofort beendet mit synthetischem stt_endpoint."""
|
||||||
sess.speaker_checked = True
|
sess.speaker_checked = True
|
||||||
|
# Schalter aus (Default) → gar keine Pruefung, alles durchlassen.
|
||||||
|
if not SPEAKER_ID_ENABLED:
|
||||||
|
sess.speaker_match = True
|
||||||
|
sess.speaker_similarity = 0.0
|
||||||
|
return
|
||||||
# Erste ~1.5s aus dem Buffer entnehmen (16kHz * 2 byte/sample = 32 bytes/ms)
|
# Erste ~1.5s aus dem Buffer entnehmen (16kHz * 2 byte/sample = 32 bytes/ms)
|
||||||
head_bytes = bytes(sess.pcm_buffer[: STREAM_SPEAKER_CHECK_MS * 32])
|
head_bytes = bytes(sess.pcm_buffer[: STREAM_SPEAKER_CHECK_MS * 32])
|
||||||
if len(head_bytes) < speaker_id.MIN_SAMPLE_BYTES:
|
if len(head_bytes) < speaker_id.MIN_SAMPLE_BYTES:
|
||||||
@@ -549,8 +600,16 @@ class SessionManager:
|
|||||||
# Akustische Sprach-Aktivitaet JEDEN Tick (~200ms) messen — unabhaengig
|
# Akustische Sprach-Aktivitaet JEDEN Tick (~200ms) messen — unabhaengig
|
||||||
# vom Transcribe-Throttle. Solange wirklich gesprochen wird, bleibt die
|
# vom Transcribe-Throttle. Solange wirklich gesprochen wird, bleibt die
|
||||||
# Session am Leben, auch wenn Whisper gerade keinen neuen Text liefert.
|
# Session am Leben, auch wenn Whisper gerade keinen neuen Text liefert.
|
||||||
if self._tail_rms(sess) >= STREAM_VOICE_RMS_THRESHOLD:
|
# ADAPTIV: der Schwellwert richtet sich nach dem gemessenen Rausch-Boden
|
||||||
|
# (fast-down/slow-up), damit leises/entferntes Sprechen nicht faelschlich
|
||||||
|
# als Stille gilt und der Satz mitten drin abgeschnitten wird.
|
||||||
|
rms = self._tail_rms(sess)
|
||||||
|
if rms >= self._voice_threshold(sess):
|
||||||
sess.last_voice_at = now
|
sess.last_voice_at = now
|
||||||
|
else:
|
||||||
|
# Rausch-Boden NUR aus Nicht-Sprache lernen — waehrend Sprache
|
||||||
|
# einfrieren, sonst wandert die Grenze hoch und sperrt Sprache aus.
|
||||||
|
self._update_noise_floor(sess, rms)
|
||||||
|
|
||||||
# Endpoint-Entscheidung JEDEN Tick, sobald ueberhaupt Text erkannt wurde:
|
# Endpoint-Entscheidung JEDEN Tick, sobald ueberhaupt Text erkannt wurde:
|
||||||
# (a) akustisch: seit endpoint_ms keine Sprach-Energie mehr → User ist
|
# (a) akustisch: seit endpoint_ms keine Sprach-Energie mehr → User ist
|
||||||
@@ -620,6 +679,26 @@ class SessionManager:
|
|||||||
return 0.0
|
return 0.0
|
||||||
return (samples / sess.sample_rate) * 1000.0
|
return (samples / sess.sample_rate) * 1000.0
|
||||||
|
|
||||||
|
def _voice_threshold(self, sess: StreamSession) -> float:
|
||||||
|
"""Adaptiver Voice-Schwellwert = Rausch-Boden * Faktor, geklammert auf
|
||||||
|
[min, max]. Bei noch ungelerntem Boden (0.0) → Untergrenze: sensibel,
|
||||||
|
lieber nicht abschneiden (das war der eigentliche Cutoff-Bug)."""
|
||||||
|
nf = sess.noise_floor
|
||||||
|
if nf <= 0.0:
|
||||||
|
return sess.voice_rms_min
|
||||||
|
return min(max(nf * sess.voice_factor, sess.voice_rms_min), sess.voice_rms_max)
|
||||||
|
|
||||||
|
def _update_noise_floor(self, sess: StreamSession, rms: float) -> None:
|
||||||
|
"""Rausch-Boden nachfuehren: schnell runter (neue, leisere Stille),
|
||||||
|
langsam rauf (Umgebung wird lauter). NUR mit Nicht-Sprache aufrufen."""
|
||||||
|
nf = sess.noise_floor
|
||||||
|
if nf <= 0.0:
|
||||||
|
sess.noise_floor = rms
|
||||||
|
elif rms < nf:
|
||||||
|
sess.noise_floor = 0.90 * nf + 0.10 * rms
|
||||||
|
else:
|
||||||
|
sess.noise_floor = 0.98 * nf + 0.02 * rms
|
||||||
|
|
||||||
def _tail_rms(self, sess: StreamSession) -> float:
|
def _tail_rms(self, sess: StreamSession) -> float:
|
||||||
"""RMS-Energie der letzten STREAM_ENERGY_WINDOW_MS des Audio-Buffers.
|
"""RMS-Energie der letzten STREAM_ENERGY_WINDOW_MS des Audio-Buffers.
|
||||||
Dient als akustisches „redet noch / ist still"-Signal."""
|
Dient als akustisches „redet noch / ist still"-Signal."""
|
||||||
@@ -760,6 +839,30 @@ async def _broadcast_status(ws, state: str, **extra) -> None:
|
|||||||
await _send(ws, "service_status", payload)
|
await _send(ws, "service_status", payload)
|
||||||
|
|
||||||
|
|
||||||
|
async def _worker_register(ws, *, model: str = "", busy_fn=None) -> None:
|
||||||
|
"""Meldet diesen Worker bei der aria-bridge an (worker_hello) und haelt die
|
||||||
|
Flotten-Registry per periodischem worker_ping frisch. worker_hello wird alle
|
||||||
|
~30s WIEDERHOLT (wie der Satellit), damit ein neu gestartetes Diagnostic/
|
||||||
|
Bridge uns lernt — RVS spielt hellos nicht nach."""
|
||||||
|
def _hello():
|
||||||
|
return {"instanceId": INSTANCE_ID, "service": WORKER_SERVICE,
|
||||||
|
"node": NODE_NAME, "gpus": GPU_IDS, "model": model}
|
||||||
|
try:
|
||||||
|
await _send(ws, "worker_hello", _hello())
|
||||||
|
n = 0
|
||||||
|
while True:
|
||||||
|
await asyncio.sleep(WORKER_PING_INTERVAL_S)
|
||||||
|
n += 1
|
||||||
|
busy = bool(busy_fn()) if busy_fn else False
|
||||||
|
await _send(ws, "worker_ping", {"instanceId": INSTANCE_ID, "busy": busy})
|
||||||
|
if n % 3 == 0:
|
||||||
|
await _send(ws, "worker_hello", _hello())
|
||||||
|
except asyncio.CancelledError:
|
||||||
|
raise
|
||||||
|
except Exception:
|
||||||
|
return # Socket tot → still beenden; run_loop reconnectet + startet neu
|
||||||
|
|
||||||
|
|
||||||
# ──────────────────────────────────────────────────────────────
|
# ──────────────────────────────────────────────────────────────
|
||||||
# WS-LOOP
|
# WS-LOOP
|
||||||
# ──────────────────────────────────────────────────────────────
|
# ──────────────────────────────────────────────────────────────
|
||||||
@@ -771,7 +874,7 @@ async def run_loop(runner: WhisperRunner, sessions: SessionManager) -> None:
|
|||||||
|
|
||||||
while True:
|
while True:
|
||||||
scheme = "wss" if use_tls else "ws"
|
scheme = "wss" if use_tls else "ws"
|
||||||
url = f"{scheme}://{RVS_HOST}:{RVS_PORT}/ws?token={RVS_TOKEN}"
|
url = f"{scheme}://{RVS_HOST}:{RVS_PORT}?token={RVS_TOKEN}"
|
||||||
masked = url.replace(RVS_TOKEN, "***") if RVS_TOKEN else url
|
masked = url.replace(RVS_TOKEN, "***") if RVS_TOKEN else url
|
||||||
try:
|
try:
|
||||||
logger.info("Verbinde zu RVS: %s", masked)
|
logger.info("Verbinde zu RVS: %s", masked)
|
||||||
@@ -793,21 +896,37 @@ async def run_loop(runner: WhisperRunner, sessions: SessionManager) -> None:
|
|||||||
logger.info("Initial: sende config_request an aria-bridge")
|
logger.info("Initial: sende config_request an aria-bridge")
|
||||||
await _send(ws, "config_request", {"service": "whisper"})
|
await _send(ws, "config_request", {"service": "whisper"})
|
||||||
# Startup-Marker — App-Logs zeigen damit ob Streaming-Code
|
# Startup-Marker — App-Logs zeigen damit ob Streaming-Code
|
||||||
# ueberhaupt aktiv ist (Stefan baut auf Gamebox via PS,
|
# ueberhaupt aktiv ist (Stefan baut auf AI-Box via PS,
|
||||||
# Build/Restart kann unbeabsichtigt alte Version weiterfahren).
|
# Build/Restart kann unbeabsichtigt alte Version weiterfahren).
|
||||||
await _debug_log(ws, "boot",
|
await _debug_log(ws, "boot",
|
||||||
"whisper-bridge online — streaming-mode ENABLED, debug-log ON")
|
"whisper-bridge online — streaming-mode ENABLED, debug-log ON")
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.exception("Initial-Handshake crashed: %s", e)
|
logger.exception("Initial-Handshake crashed: %s", e)
|
||||||
asyncio.create_task(_initial_handshake())
|
asyncio.create_task(_initial_handshake())
|
||||||
|
ping_task = asyncio.create_task(_worker_register(
|
||||||
|
ws, model=(runner.model_size or WHISPER_MODEL),
|
||||||
|
busy_fn=lambda: bool(sessions._sessions)))
|
||||||
|
|
||||||
async for raw in ws:
|
while True:
|
||||||
|
try:
|
||||||
|
raw = await asyncio.wait_for(ws.recv(), timeout=RX_STALE_S)
|
||||||
|
except asyncio.TimeoutError:
|
||||||
|
logger.warning("Kein RVS-Traffic seit %ds — Verbindung halb-tot, reconnect", RX_STALE_S)
|
||||||
|
raise ConnectionError("rvs-stale")
|
||||||
try:
|
try:
|
||||||
msg = json.loads(raw)
|
msg = json.loads(raw)
|
||||||
except Exception:
|
except Exception:
|
||||||
continue
|
continue
|
||||||
mtype = msg.get("type", "")
|
mtype = msg.get("type", "")
|
||||||
payload = msg.get("payload", {}) or {}
|
payload = msg.get("payload", {}) or {}
|
||||||
|
# Redundanz-Routing: gezielt an eine andere Instanz adressiert
|
||||||
|
# → ignorieren. Ohne targetInstance → wie bisher.
|
||||||
|
tgt = payload.get("targetInstance")
|
||||||
|
if tgt and tgt != INSTANCE_ID:
|
||||||
|
continue
|
||||||
|
# Auslastungs-Monitor (node_stats_*) abfangen.
|
||||||
|
if await _stats.handle(ws, mtype, payload, _send):
|
||||||
|
continue
|
||||||
|
|
||||||
if mtype == "stt_request":
|
if mtype == "stt_request":
|
||||||
req_id = payload.get("requestId", "?")
|
req_id = payload.get("requestId", "?")
|
||||||
@@ -928,6 +1047,10 @@ async def run_loop(runner: WhisperRunner, sessions: SessionManager) -> None:
|
|||||||
logger.info("[speaker-id] threshold gesetzt: %.2f", t)
|
logger.info("[speaker-id] threshold gesetzt: %.2f", t)
|
||||||
except (TypeError, ValueError):
|
except (TypeError, ValueError):
|
||||||
pass
|
pass
|
||||||
|
if "voiceIdEnabled" in payload:
|
||||||
|
_set_speaker_id_enabled(payload.get("voiceIdEnabled"))
|
||||||
|
logger.info("[speaker-id] Gating %s (voiceIdEnabled)",
|
||||||
|
"AN" if SPEAKER_ID_ENABLED else "AUS")
|
||||||
if "whisperDebugLog" in payload:
|
if "whisperDebugLog" in payload:
|
||||||
global _DEBUG_LOG_TO_BRIDGE
|
global _DEBUG_LOG_TO_BRIDGE
|
||||||
old = _DEBUG_LOG_TO_BRIDGE
|
old = _DEBUG_LOG_TO_BRIDGE
|
||||||
@@ -972,6 +1095,10 @@ async def run_loop(runner: WhisperRunner, sessions: SessionManager) -> None:
|
|||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.warning("Verbindung verloren: %s", e)
|
logger.warning("Verbindung verloren: %s", e)
|
||||||
sessions.detach_ws()
|
sessions.detach_ws()
|
||||||
|
try:
|
||||||
|
ping_task.cancel()
|
||||||
|
except NameError:
|
||||||
|
pass
|
||||||
if use_tls and RVS_TLS_FALLBACK and not tls_fallback_tried:
|
if use_tls and RVS_TLS_FALLBACK and not tls_fallback_tried:
|
||||||
logger.info("TLS-Verbindung fehlgeschlagen — Fallback auf ws://")
|
logger.info("TLS-Verbindung fehlgeschlagen — Fallback auf ws://")
|
||||||
use_tls = False
|
use_tls = False
|
||||||
@@ -992,6 +1119,7 @@ async def main() -> None:
|
|||||||
# Endpointer-Loop nebenbei laufen lassen — er pruefst _ws is None und
|
# Endpointer-Loop nebenbei laufen lassen — er pruefst _ws is None und
|
||||||
# schlaeft solange das nicht gesetzt ist.
|
# schlaeft solange das nicht gesetzt ist.
|
||||||
asyncio.create_task(sessions.run_endpointer())
|
asyncio.create_task(sessions.run_endpointer())
|
||||||
|
asyncio.create_task(_stats.run_sampler()) # Auslastungs-Sampler (Stage E)
|
||||||
await run_loop(runner, sessions)
|
await run_loop(runner, sessions)
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,167 @@
|
|||||||
|
"""
|
||||||
|
ARIA Node-Stats — Auslastungs-Monitor pro Box (GPU + optional Tokens).
|
||||||
|
|
||||||
|
Identische Kopie in jedem Worker-Build-Context (f5tts/whisper/voxtral/llm-adapter),
|
||||||
|
weil jeder Worker ein eigener Docker-Build-Context ist.
|
||||||
|
|
||||||
|
Aufgaben:
|
||||||
|
- Sampler-Loop (alle SAMPLE_SEC): nvidia-smi-Auslastung + Token-Delta → Ringpuffer
|
||||||
|
(persistent als JSON auf der Box). Laeuft unabhaengig vom Modal.
|
||||||
|
- Live-Stream: bei node_stats_stream_start jede Sekunde rohes nvidia-smi + Werte
|
||||||
|
senden (bis stop / Auto-Timeout).
|
||||||
|
- History-Request + Reset (Besen).
|
||||||
|
|
||||||
|
Reicht `handle(ws, mtype, payload)` in die Worker-Message-Loop ein; gibt True
|
||||||
|
zurueck, wenn die Nachricht eine node_stats_*-Nachricht war.
|
||||||
|
"""
|
||||||
|
import asyncio
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import time
|
||||||
|
|
||||||
|
SAMPLE_SEC = int(os.getenv("STATS_SAMPLE_SEC", "15"))
|
||||||
|
HISTORY_CAP = int(os.getenv("STATS_HISTORY_CAP", "500")) # ~2h bei 15s
|
||||||
|
STREAM_MAX_SEC = int(os.getenv("STATS_STREAM_MAX_SEC", "300"))
|
||||||
|
|
||||||
|
|
||||||
|
async def _run_cmd(*args, timeout=8) -> str:
|
||||||
|
"""Fuehrt ein Kommando aus, gibt stdout (str) zurueck; '' bei Fehler."""
|
||||||
|
try:
|
||||||
|
proc = await asyncio.create_subprocess_exec(
|
||||||
|
*args,
|
||||||
|
stdout=asyncio.subprocess.PIPE,
|
||||||
|
stderr=asyncio.subprocess.DEVNULL,
|
||||||
|
)
|
||||||
|
out, _ = await asyncio.wait_for(proc.communicate(), timeout=timeout)
|
||||||
|
return (out or b"").decode("utf-8", "replace")
|
||||||
|
except Exception:
|
||||||
|
return ""
|
||||||
|
|
||||||
|
|
||||||
|
class NodeStats:
|
||||||
|
def __init__(self, instance_id: str, node_name: str, history_path: str,
|
||||||
|
token_getter=None, logger=None):
|
||||||
|
self.instance_id = instance_id
|
||||||
|
self.node_name = node_name
|
||||||
|
self.history_path = history_path
|
||||||
|
self.token_getter = token_getter # callable -> kumulative Token-Zahl (oder None)
|
||||||
|
self.log = logger
|
||||||
|
self.samples = self._load()
|
||||||
|
self._last_tokens = self._tokens_now()
|
||||||
|
self._stream_task = None
|
||||||
|
|
||||||
|
# ── Persistenz ──────────────────────────────────────────
|
||||||
|
def _load(self) -> list:
|
||||||
|
try:
|
||||||
|
with open(self.history_path) as f:
|
||||||
|
data = json.load(f)
|
||||||
|
return data if isinstance(data, list) else []
|
||||||
|
except Exception:
|
||||||
|
return []
|
||||||
|
|
||||||
|
def _persist(self) -> None:
|
||||||
|
try:
|
||||||
|
os.makedirs(os.path.dirname(self.history_path) or ".", exist_ok=True)
|
||||||
|
tmp = self.history_path + ".tmp"
|
||||||
|
with open(tmp, "w") as f:
|
||||||
|
json.dump(self.samples[-HISTORY_CAP:], f)
|
||||||
|
os.replace(tmp, self.history_path)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
def _tokens_now(self) -> int:
|
||||||
|
try:
|
||||||
|
return int(self.token_getter()) if self.token_getter else 0
|
||||||
|
except Exception:
|
||||||
|
return 0
|
||||||
|
|
||||||
|
# ── nvidia-smi ──────────────────────────────────────────
|
||||||
|
async def _query_gpu(self) -> dict:
|
||||||
|
"""Aggregierte GPU-Werte ueber alle sichtbaren Karten."""
|
||||||
|
out = await _run_cmd(
|
||||||
|
"nvidia-smi",
|
||||||
|
"--query-gpu=utilization.gpu,memory.used,memory.total",
|
||||||
|
"--format=csv,noheader,nounits")
|
||||||
|
utils, used, total = [], 0, 0
|
||||||
|
for line in out.strip().splitlines():
|
||||||
|
parts = [p.strip() for p in line.split(",")]
|
||||||
|
if len(parts) < 3:
|
||||||
|
continue
|
||||||
|
try:
|
||||||
|
utils.append(float(parts[0]))
|
||||||
|
used += float(parts[1])
|
||||||
|
total += float(parts[2])
|
||||||
|
except ValueError:
|
||||||
|
continue
|
||||||
|
gpu = round(sum(utils) / len(utils), 1) if utils else 0.0
|
||||||
|
return {"gpu": gpu, "memUsed": int(used), "memTotal": int(total)}
|
||||||
|
|
||||||
|
async def _nvidia_smi_text(self) -> str:
|
||||||
|
txt = await _run_cmd("nvidia-smi")
|
||||||
|
return txt or "nvidia-smi nicht verfuegbar"
|
||||||
|
|
||||||
|
# ── Sampler (Verlauf) ───────────────────────────────────
|
||||||
|
async def run_sampler(self) -> None:
|
||||||
|
while True:
|
||||||
|
try:
|
||||||
|
g = await self._query_gpu()
|
||||||
|
now_tok = self._tokens_now()
|
||||||
|
dtok = max(0, now_tok - self._last_tokens)
|
||||||
|
self._last_tokens = now_tok
|
||||||
|
self.samples.append({
|
||||||
|
"ts": int(time.time()),
|
||||||
|
"gpu": g["gpu"], "memUsed": g["memUsed"],
|
||||||
|
"memTotal": g["memTotal"], "tokens": dtok,
|
||||||
|
})
|
||||||
|
if len(self.samples) > HISTORY_CAP:
|
||||||
|
self.samples = self.samples[-HISTORY_CAP:]
|
||||||
|
self._persist()
|
||||||
|
except Exception as e:
|
||||||
|
if self.log:
|
||||||
|
self.log.debug("node_stats sample fehlgeschlagen: %s", e)
|
||||||
|
await asyncio.sleep(SAMPLE_SEC)
|
||||||
|
|
||||||
|
# ── Live-Stream ─────────────────────────────────────────
|
||||||
|
async def _stream(self, ws, send) -> None:
|
||||||
|
t0 = time.time()
|
||||||
|
try:
|
||||||
|
while time.time() - t0 < STREAM_MAX_SEC:
|
||||||
|
g = await self._query_gpu()
|
||||||
|
smi = await self._nvidia_smi_text()
|
||||||
|
await send(ws, "node_stats", {
|
||||||
|
"instanceId": self.instance_id, "node": self.node_name,
|
||||||
|
"nvidiaSmi": smi, **g,
|
||||||
|
})
|
||||||
|
await asyncio.sleep(1)
|
||||||
|
except asyncio.CancelledError:
|
||||||
|
raise
|
||||||
|
except Exception:
|
||||||
|
return
|
||||||
|
|
||||||
|
# ── Dispatch ────────────────────────────────────────────
|
||||||
|
async def handle(self, ws, mtype: str, payload: dict, send) -> bool:
|
||||||
|
if mtype == "node_stats_stream_start":
|
||||||
|
if self._stream_task and not self._stream_task.done():
|
||||||
|
self._stream_task.cancel()
|
||||||
|
self._stream_task = asyncio.create_task(self._stream(ws, send))
|
||||||
|
return True
|
||||||
|
if mtype == "node_stats_stream_stop":
|
||||||
|
if self._stream_task:
|
||||||
|
self._stream_task.cancel()
|
||||||
|
self._stream_task = None
|
||||||
|
return True
|
||||||
|
if mtype == "node_stats_history_request":
|
||||||
|
await send(ws, "node_stats_history", {
|
||||||
|
"instanceId": self.instance_id, "node": self.node_name,
|
||||||
|
"samples": self.samples[-HISTORY_CAP:],
|
||||||
|
"tokenCapable": self.token_getter is not None,
|
||||||
|
"sampleSec": SAMPLE_SEC,
|
||||||
|
})
|
||||||
|
return True
|
||||||
|
if mtype == "node_stats_reset":
|
||||||
|
self.samples = []
|
||||||
|
self._persist()
|
||||||
|
await send(ws, "node_stats_reset_done",
|
||||||
|
{"instanceId": self.instance_id, "node": self.node_name})
|
||||||
|
return True
|
||||||
|
return False
|
||||||
@@ -61,10 +61,40 @@ def _ensure_loaded():
|
|||||||
return _model
|
return _model
|
||||||
|
|
||||||
|
|
||||||
|
def _decode_compressed_to_pcm(audio_bytes: bytes) -> bytes:
|
||||||
|
"""Dekodiert komprimiertes Audio (MP4/M4A/AAC vom Android-Recorder) via ffmpeg
|
||||||
|
(im Container vorhanden) auf rohes 16kHz mono int16 LE PCM. Input geht ueber
|
||||||
|
eine Temp-Datei (nicht Pipe): Androids MediaRecorder legt das moov-Atom ans
|
||||||
|
ENDE, das braucht seekbaren Input, sonst 'moov atom not found'."""
|
||||||
|
import os
|
||||||
|
import subprocess
|
||||||
|
import tempfile
|
||||||
|
tmp = None
|
||||||
|
try:
|
||||||
|
with tempfile.NamedTemporaryFile(suffix=".mp4", delete=False) as tf:
|
||||||
|
tf.write(audio_bytes)
|
||||||
|
tmp = tf.name
|
||||||
|
proc = subprocess.run(
|
||||||
|
["ffmpeg", "-hide_banner", "-loglevel", "error", "-i", tmp,
|
||||||
|
"-f", "s16le", "-ac", "1", "-ar", "16000", "pipe:1"],
|
||||||
|
stdout=subprocess.PIPE, stderr=subprocess.PIPE,
|
||||||
|
)
|
||||||
|
if proc.returncode != 0 or not proc.stdout:
|
||||||
|
raise ValueError(
|
||||||
|
f"ffmpeg decode failed: {proc.stderr.decode('utf-8', 'ignore')[:200]}")
|
||||||
|
return proc.stdout
|
||||||
|
finally:
|
||||||
|
if tmp:
|
||||||
|
try:
|
||||||
|
os.unlink(tmp)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
def _normalize_audio_bytes(audio_bytes: bytes) -> bytes:
|
def _normalize_audio_bytes(audio_bytes: bytes) -> bytes:
|
||||||
"""Akzeptiert entweder rohes 16kHz int16 LE PCM ODER eine WAV-Datei (RIFF/WAVE).
|
"""Akzeptiert rohes 16kHz int16 LE PCM, eine WAV-Datei (RIFF/WAVE) ODER einen
|
||||||
Bei WAV wird der Header gestrippt + Format validiert (16kHz / mono / int16).
|
komprimierten MP4/M4A/AAC-Container (Android-Recorder). WAV → Header strippen +
|
||||||
Ergebnis: rohes PCM."""
|
Format validieren; MP4/AAC → via ffmpeg dekodieren. Ergebnis: rohes PCM."""
|
||||||
if (len(audio_bytes) >= 44
|
if (len(audio_bytes) >= 44
|
||||||
and audio_bytes[:4] == b"RIFF"
|
and audio_bytes[:4] == b"RIFF"
|
||||||
and audio_bytes[8:12] == b"WAVE"):
|
and audio_bytes[8:12] == b"WAVE"):
|
||||||
@@ -81,6 +111,9 @@ def _normalize_audio_bytes(audio_bytes: bytes) -> bytes:
|
|||||||
if sw != 2:
|
if sw != 2:
|
||||||
raise ValueError(f"WAV-Sampleweite {sw} != 2 (int16 erwartet)")
|
raise ValueError(f"WAV-Sampleweite {sw} != 2 (int16 erwartet)")
|
||||||
return wav.readframes(wav.getnframes())
|
return wav.readframes(wav.getnframes())
|
||||||
|
# MP4/M4A/AAC-Container: Android-AAC-Recorder legt 'ftyp' bei Offset 4 an.
|
||||||
|
if len(audio_bytes) >= 12 and audio_bytes[4:8] == b"ftyp":
|
||||||
|
return _decode_compressed_to_pcm(audio_bytes)
|
||||||
return audio_bytes
|
return audio_bytes
|
||||||
|
|
||||||
|
|
||||||
@@ -212,13 +245,21 @@ def enroll_from_samples(samples_b64: list[str]) -> dict:
|
|||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
rejected.append({"index": idx, "reason": f"base64: {exc}"})
|
rejected.append({"index": idx, "reason": f"base64: {exc}"})
|
||||||
continue
|
continue
|
||||||
if len(raw) < MIN_SAMPLE_BYTES:
|
# Erst dekodieren (WAV/MP4/AAC → rohes PCM), DANN Laenge pruefen: der
|
||||||
rejected.append({"index": idx, "reason": f"zu kurz ({len(raw)} bytes)"})
|
# Android-Recorder liefert komprimiertes MP4, dessen Byte-Laenge nichts
|
||||||
|
# ueber die Dauer sagt (4s AAC < 32KB → faelschlich "zu kurz").
|
||||||
|
try:
|
||||||
|
pcm = _normalize_audio_bytes(raw)
|
||||||
|
except Exception as exc:
|
||||||
|
rejected.append({"index": idx, "reason": f"decode: {exc}"})
|
||||||
|
continue
|
||||||
|
if len(pcm) < MIN_SAMPLE_BYTES:
|
||||||
|
rejected.append({"index": idx, "reason": f"zu kurz ({len(pcm)} bytes PCM)"})
|
||||||
continue
|
continue
|
||||||
try:
|
try:
|
||||||
emb = embed(raw)
|
emb = embed(pcm)
|
||||||
embeddings.append(emb)
|
embeddings.append(emb)
|
||||||
durations.append(len(raw) / 2 / 16000.0)
|
durations.append(len(pcm) / 2 / 16000.0)
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
rejected.append({"index": idx, "reason": f"embed: {exc}"})
|
rejected.append({"index": idx, "reason": f"embed: {exc}"})
|
||||||
if not embeddings:
|
if not embeddings:
|
||||||
|
|||||||
Reference in New Issue
Block a user