#!/usr/bin/env python3 """Liest eine als Chat formatierte Markdown-Datei, schickt sie an einen lokalen OpenAI-kompatiblen Endpunkt und haengt die Antwort als neuen Eintrag an. Dateiformat (Rollen: system / assistant / user): [system] foo bar lorem ipsum [assistant] okok verstanden [user] Textnachricht Die Datei ist die einzige Quelle der Wahrheit. Dieses Modul stellt die Bausteine bereit, die auch das TUel (tui.py) verwendet. """ import os import re import sys from pathlib import Path from typing import Iterator from openai import OpenAI BASE_URL = "http://192.168.161.4:9161/v1" MODEL = "glm-5.2-colibri" MAX_TOKENS = 1024 API_KEY = os.environ.get("COLIBRI_API_KEY", "not-needed") HEADER_RE = re.compile(r"^\[(system|assistant|user)\]\s*$", re.IGNORECASE) TRAILING_EMPTY_RE = re.compile(r"\n*\[(?:system|assistant|user)\][ \t]*\s*$", re.IGNORECASE) _client: OpenAI | None = None def client() -> OpenAI: """Lazy angelegter OpenAI-Client. Key aus COLIBRI_API_KEY, sonst Dummy (der lokale Endpunkt braucht keinen).""" global _client if _client is None: _client = OpenAI(base_url=BASE_URL, api_key=API_KEY) return _client def parse_chat(text: str) -> list[dict]: """Zerlegt den Dateiinhalt in eine Liste von {role, content}-Nachrichten.""" messages: list[dict] = [] role: str | None = None buf: list[str] = [] def flush() -> None: if role is not None: messages.append({"role": role, "content": "\n".join(buf).strip()}) for line in text.splitlines(): m = HEADER_RE.match(line) if m: flush() role = m.group(1).lower() buf = [] elif role is not None: buf.append(line) flush() return [msg for msg in messages if msg["content"]] def load_messages(path: str | Path) -> list[dict]: path = Path(path) if not path.exists(): return [] return parse_chat(path.read_text(encoding="utf-8")) def append_block(path: str | Path, role: str, content: str) -> None: """Haengt einen neuen [role]-Block an die Datei an (Datei = Quelle der Wahrheit).""" path = Path(path) existing = path.read_text(encoding="utf-8") if path.exists() else "" # einen evtl. am Ende stehenden leeren Block (nur Header) verwerfen existing = TRAILING_EMPTY_RE.sub("", existing).rstrip("\n") if existing else "" prefix = f"{existing}\n\n" if existing else "" path.write_text(f"{prefix}[{role}]\n{content.strip()}\n", encoding="utf-8") def stream_reply(messages: list[dict]) -> Iterator[str]: """Streamt die Antwort tokenweise. Rueckgabewert (StopIteration.value) ist der finish_reason ("stop", "length", ...).""" stream = client().chat.completions.create( model=MODEL, messages=messages, max_tokens=MAX_TOKENS, stream=True, ) finish: str | None = None for chunk in stream: if not chunk.choices: continue choice = chunk.choices[0] if choice.delta and choice.delta.content: yield choice.delta.content if choice.finish_reason: finish = choice.finish_reason return finish def main() -> None: path = Path(sys.argv[1] if len(sys.argv) > 1 else "chat.md") messages = load_messages(path) if not messages: sys.exit(f"Keine Nachrichten in {path} gefunden.") parts: list[str] = [] gen = stream_reply(messages) finish: str | None = None try: while True: piece = next(gen) parts.append(piece) print(piece, end="", flush=True) except StopIteration as stop: finish = stop.value print() if finish == "length": print(f"[Warnung] Antwort bei MAX_TOKENS={MAX_TOKENS} abgeschnitten.", file=sys.stderr) append_block(path, "assistant", "".join(parts)) if __name__ == "__main__": main()