Files
LLM-chatinterface/chat.py
T
2026-08-30 19:03:53 +02:00

135 lines
3.8 KiB
Python

#!/usr/bin/env python3
"""Liest eine als Chat formatierte Markdown-Datei, schickt sie an einen lokalen
OpenAI-kompatiblen Endpunkt und haengt die Antwort als neuen Eintrag an.
Dateiformat (Rollen: system / assistant / user):
[system]
foo bar lorem ipsum
[assistant]
okok verstanden
[user]
Textnachricht
Die Datei ist die einzige Quelle der Wahrheit. Dieses Modul stellt die
Bausteine bereit, die auch das TUel (tui.py) verwendet.
"""
import os
import re
import sys
from pathlib import Path
from typing import Iterator
from openai import OpenAI
BASE_URL = "http://192.168.161.4:9161/v1"
MODEL = "glm-5.2-colibri"
MAX_TOKENS = 1024
API_KEY = os.environ.get("COLIBRI_API_KEY", "not-needed")
HEADER_RE = re.compile(r"^\[(system|assistant|user)\]\s*$", re.IGNORECASE)
TRAILING_EMPTY_RE = re.compile(r"\n*\[(?:system|assistant|user)\][ \t]*\s*$", re.IGNORECASE)
_client: OpenAI | None = None
def client() -> OpenAI:
"""Lazy angelegter OpenAI-Client. Key aus COLIBRI_API_KEY, sonst Dummy
(der lokale Endpunkt braucht keinen)."""
global _client
if _client is None:
_client = OpenAI(base_url=BASE_URL, api_key=API_KEY)
return _client
def parse_chat(text: str) -> list[dict]:
"""Zerlegt den Dateiinhalt in eine Liste von {role, content}-Nachrichten."""
messages: list[dict] = []
role: str | None = None
buf: list[str] = []
def flush() -> None:
if role is not None:
messages.append({"role": role, "content": "\n".join(buf).strip()})
for line in text.splitlines():
m = HEADER_RE.match(line)
if m:
flush()
role = m.group(1).lower()
buf = []
elif role is not None:
buf.append(line)
flush()
return [msg for msg in messages if msg["content"]]
def load_messages(path: str | Path) -> list[dict]:
path = Path(path)
if not path.exists():
return []
return parse_chat(path.read_text(encoding="utf-8"))
def append_block(path: str | Path, role: str, content: str) -> None:
"""Haengt einen neuen [role]-Block an die Datei an (Datei = Quelle der Wahrheit)."""
path = Path(path)
existing = path.read_text(encoding="utf-8") if path.exists() else ""
# einen evtl. am Ende stehenden leeren Block (nur Header) verwerfen
existing = TRAILING_EMPTY_RE.sub("", existing).rstrip("\n") if existing else ""
prefix = f"{existing}\n\n" if existing else ""
path.write_text(f"{prefix}[{role}]\n{content.strip()}\n", encoding="utf-8")
def stream_reply(messages: list[dict]) -> Iterator[str]:
"""Streamt die Antwort tokenweise. Rueckgabewert (StopIteration.value) ist
der finish_reason ("stop", "length", ...)."""
stream = client().chat.completions.create(
model=MODEL,
messages=messages,
max_tokens=MAX_TOKENS,
stream=True,
)
finish: str | None = None
for chunk in stream:
if not chunk.choices:
continue
choice = chunk.choices[0]
if choice.delta and choice.delta.content:
yield choice.delta.content
if choice.finish_reason:
finish = choice.finish_reason
return finish
def main() -> None:
path = Path(sys.argv[1] if len(sys.argv) > 1 else "chat.md")
messages = load_messages(path)
if not messages:
sys.exit(f"Keine Nachrichten in {path} gefunden.")
parts: list[str] = []
gen = stream_reply(messages)
finish: str | None = None
try:
while True:
piece = next(gen)
parts.append(piece)
print(piece, end="", flush=True)
except StopIteration as stop:
finish = stop.value
print()
if finish == "length":
print(f"[Warnung] Antwort bei MAX_TOKENS={MAX_TOKENS} abgeschnitten.", file=sys.stderr)
append_block(path, "assistant", "".join(parts))
if __name__ == "__main__":
main()