mirror of
https://github.com/Jeuners/astra-local-voice.git
synced 2026-09-09 15:02:35 +02:00
feat: initial commit of astra local voice agent
Lokaler deutscher Sprachagent für Apple Silicon: Pipecat-Pipeline mit Nemotron-ASR (MLX), Qwen über natives Ollama /api/chat, und Pocket TTS. Loopback-only WebRTC-Server mit Origin/Host-Härtung. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01LVgSHNHdRx3UNTBodFmhRA
This commit is contained in:
commit
a42b3e8d73
17 changed files with 1436 additions and 0 deletions
80
astra/core.py
Normal file
80
astra/core.py
Normal file
|
|
@ -0,0 +1,80 @@
|
|||
"""Configuration and pure request policy, independent of audio hardware."""
|
||||
|
||||
import os
|
||||
from dataclasses import dataclass
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Settings:
|
||||
model: str = "qwen3.5:latest"
|
||||
ollama_url: str = "http://127.0.0.1:11434"
|
||||
stt_model: str = "mlx-community/nemotron-3.5-asr-streaming-0.6b-8bit"
|
||||
tts_language: str = "german"
|
||||
voice: str = "alba"
|
||||
port: int = 7860
|
||||
context_tokens: int = 4096
|
||||
|
||||
@classmethod
|
||||
def from_env(cls):
|
||||
return cls(
|
||||
model=os.getenv("ASTRA_MODEL", cls.model),
|
||||
ollama_url=os.getenv("ASTRA_OLLAMA_URL", cls.ollama_url).rstrip("/"),
|
||||
stt_model=os.getenv("ASTRA_STT_MODEL", cls.stt_model),
|
||||
tts_language=os.getenv("ASTRA_TTS_LANGUAGE", cls.tts_language),
|
||||
voice=os.getenv("ASTRA_VOICE", cls.voice),
|
||||
port=int(os.getenv("ASTRA_PORT", cls.port)),
|
||||
)
|
||||
|
||||
|
||||
SYSTEM_PROMPT = (
|
||||
"Du bist Astra, ein freundlicher deutschsprachiger Gesprächsassistent. "
|
||||
"Antworte natürlich und knapp, normalerweise in ein bis drei kurzen Sätzen. "
|
||||
"Deine Antwort wird vorgelesen: kein Markdown, keine Sternchen, keine Listen. "
|
||||
"Sprich Zahlen und Abkürzungen verständlich aus. Stelle bei Bedarf eine kurze Rückfrage. "
|
||||
"Du hast keine Werkzeuge, keinen Internetzugang und keinen Zugriff auf Dateien oder Apps. "
|
||||
"Behaupte nicht, Aktionen ausgeführt zu haben."
|
||||
)
|
||||
|
||||
|
||||
def trim_messages(messages: list[dict], max_chars: int = 10000) -> list[dict]:
|
||||
"""Retain recent whole turns within a conservative context character budget."""
|
||||
system = [dict(m) for m in messages if m["role"] == "system"][:1]
|
||||
if system:
|
||||
system[0]["content"] = system[0]["content"][: max_chars // 2]
|
||||
budget = max_chars - sum(len(m["content"]) for m in system)
|
||||
turns = []
|
||||
for message in reversed(messages):
|
||||
if message["role"] not in ("user", "assistant"):
|
||||
continue
|
||||
content = message.get("content")
|
||||
if not isinstance(content, str) or not content:
|
||||
continue
|
||||
if len(content) > budget:
|
||||
if not turns:
|
||||
turns.append({"role": message["role"], "content": content[-budget:]})
|
||||
break
|
||||
turns.append({"role": message["role"], "content": content})
|
||||
budget -= len(content)
|
||||
turns.reverse()
|
||||
while turns and turns[0]["role"] != "user":
|
||||
turns.pop(0)
|
||||
return system + turns
|
||||
|
||||
|
||||
def build_request(settings: Settings, messages: list[dict]) -> dict:
|
||||
return {
|
||||
"model": settings.model,
|
||||
"messages": trim_messages(messages),
|
||||
"think": False,
|
||||
"stream": True,
|
||||
"keep_alive": -1,
|
||||
"options": {
|
||||
"num_ctx": settings.context_tokens,
|
||||
"num_predict": 256,
|
||||
"temperature": 0.6,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def local_origin_allowed(origin: str, port: int = 7860) -> bool:
|
||||
return origin in {f"http://localhost:{port}", f"http://127.0.0.1:{port}"}
|
||||
Loading…
Add table
Add a link
Reference in a new issue