mirror of
https://github.com/Jeuners/astra-vision.git
synced 2026-09-09 15:02:35 +02:00
Zwei neue, sauber getrennte Fähigkeiten, jede in ihrem eigenen Modul: - astra/comfyui.py: async HTTP-Client für einen lokalen ComfyUI-Server (z-image-turbo-Workflow), kennt nichts von der Pipeline. - astra/documents.py: PDF-Textextraktion via pypdf, keine Netzwerkzugriffe. - astra/tools.py: verdrahtet generate_image als natives Ollama-Tool-Call — Qwen 3.5 unterstützt Tools und Vision bereits nativ laut `ollama show`. Dafür wurde astra/services.py so erweitert, dass NativeOllamaService Ollamas native tool_calls im Streaming-Response erkennt und als ChatCompletionChunk-Deltas an Pipecats bereits vorhandene, generische Function-Calling-Maschinerie (_process_context/run_function_calls) weiterreicht — die musste dafür nicht angefasst werden. trim_messages in core.py bewahrt jetzt Tool-Roundtrips und Bild-Anhänge vollständig statt sie auf role/content zu reduzieren. Neuer Upload-Button im UI (Bild oder PDF, während eines laufenden Gesprächs): PDFs gehen als Text, Bilder als Base64 über Qwens Vision in den Gesprächskontext ein. Generierte Bilder werden über /api/media/<id> ausgeliefert und per Datenkanal im Transkript angezeigt. Kompletter Function-Calling-Roundtrip end-to-end gegen echtes Ollama und echtes ComfyUI verifiziert (Modell ruft generate_image korrekt auf, Bild wird erzeugt und im media_store abgelegt). Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01LVgSHNHdRx3UNTBodFmhRA
65 lines
2.3 KiB
Python
65 lines
2.3 KiB
Python
"""Async client for a local ComfyUI server. Knows nothing about the voice pipeline."""
|
|
|
|
import asyncio
|
|
import json
|
|
import random
|
|
import time
|
|
from pathlib import Path
|
|
|
|
import httpx
|
|
|
|
WORKFLOW_PATH = Path(__file__).resolve().parent / "comfyui_workflow.json"
|
|
SAVE_NODE = "9"
|
|
PROMPT_NODE = "57:27"
|
|
SIZE_NODE = "57:13"
|
|
SEED_NODE = "57:3"
|
|
|
|
|
|
class ComfyUIError(Exception):
|
|
"""Raised when ComfyUI is unreachable or fails to produce an image."""
|
|
|
|
|
|
async def generate_image(
|
|
endpoint: str,
|
|
prompt: str,
|
|
*,
|
|
width: int = 1024,
|
|
height: int = 1024,
|
|
timeout: float = 120.0,
|
|
) -> bytes:
|
|
"""Generate one PNG via the z-image-turbo workflow and return its raw bytes."""
|
|
workflow = json.loads(WORKFLOW_PATH.read_text())
|
|
workflow[PROMPT_NODE]["inputs"]["text"] = prompt
|
|
workflow[SIZE_NODE]["inputs"]["width"] = width
|
|
workflow[SIZE_NODE]["inputs"]["height"] = height
|
|
workflow[SEED_NODE]["inputs"]["seed"] = random.randint(0, 2**32 - 1)
|
|
|
|
async with httpx.AsyncClient(timeout=timeout) as client:
|
|
try:
|
|
response = await client.post(f"{endpoint}/prompt", json={"prompt": workflow})
|
|
response.raise_for_status()
|
|
except httpx.HTTPError as exc:
|
|
raise ComfyUIError(f"ComfyUI ist nicht erreichbar: {exc}") from exc
|
|
prompt_id = response.json()["prompt_id"]
|
|
|
|
deadline = time.monotonic() + timeout
|
|
while time.monotonic() < deadline:
|
|
history = await client.get(f"{endpoint}/history/{prompt_id}")
|
|
history.raise_for_status()
|
|
data = history.json().get(prompt_id)
|
|
if data:
|
|
images = data.get("outputs", {}).get(SAVE_NODE, {}).get("images", [])
|
|
if images:
|
|
image = images[0]
|
|
view = await client.get(
|
|
f"{endpoint}/view",
|
|
params={
|
|
"filename": image["filename"],
|
|
"subfolder": image["subfolder"],
|
|
"type": image["type"],
|
|
},
|
|
)
|
|
view.raise_for_status()
|
|
return view.content
|
|
await asyncio.sleep(1)
|
|
raise ComfyUIError("Zeitüberschreitung beim Warten auf das generierte Bild.")
|