#!/usr/bin/env python """07 -- gr.ChatInterface against a local, OpenAI-compatible LLM endpoint. Talks to a model server already running on this host; nothing leaves the box. Only the standard library is used for HTTP, so no extra wheels are needed. Point it at whichever server you run, for example: LLM_BASE_URL=http://127.0.0.1:11437/v1 LLM_MODEL=qwen3.8-27b python 07_chat_local_llm.py LLM_BASE_URL=http://127.0.0.1:11550/v1 LLM_MODEL=deepseek-r1:14b python 07_chat_local_llm.py With no server reachable it falls back to a local echo responder, so the UI still starts and the streaming path can be exercised. Compatibility note: gradio 4 defaults its chat history to the legacy "tuples" format and needs type="messages" to use the OpenAI-style dict format that gradio 5+ uses unconditionally. _CHAT_KW below hides that difference. """ import json import os import urllib.error import urllib.request import gradio as gr from _common import GRADIO_MAJOR, launch BASE_URL = os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11437/v1").rstrip("/") MODEL = os.environ.get("LLM_MODEL", "local-model") TIMEOUT = float(os.environ.get("LLM_TIMEOUT", "120")) # gradio 4 needs to be told to use the messages format; gradio 5+ has no such # argument because messages is the only format it supports. _CHAT_KW = {"type": "messages"} if GRADIO_MAJOR < 5 else {} def _post_stream(payload): """POST to /chat/completions and yield content deltas (SSE).""" req = urllib.request.Request( f"{BASE_URL}/chat/completions", data=json.dumps(payload).encode(), headers={"Content-Type": "application/json"}, method="POST", ) with urllib.request.urlopen(req, timeout=TIMEOUT) as resp: for raw in resp: line = raw.decode("utf-8", "replace").strip() if not line.startswith("data:"): continue body = line[5:].strip() if body == "[DONE]": return try: chunk = json.loads(body) except json.JSONDecodeError: continue for choice in chunk.get("choices", []): piece = (choice.get("delta") or {}).get("content") if piece: yield piece def respond(message, history, system_prompt, temperature): """ChatInterface handler. history is a list of {role, content} dicts.""" messages = [] if system_prompt: messages.append({"role": "system", "content": system_prompt}) for turn in history or []: # Be tolerant of either history format so the template survives a # gradio upgrade or a copy-paste into a gradio 4 app. if isinstance(turn, dict): messages.append({"role": turn["role"], "content": turn["content"]}) elif isinstance(turn, (list, tuple)) and len(turn) == 2: user, bot = turn if user: messages.append({"role": "user", "content": user}) if bot: messages.append({"role": "assistant", "content": bot}) messages.append({"role": "user", "content": message}) payload = {"model": MODEL, "messages": messages, "stream": True, "temperature": float(temperature)} acc = "" try: for piece in _post_stream(payload): acc += piece yield acc except (urllib.error.URLError, OSError, TimeoutError) as exc: if acc: # partial answer then a broken stream yield acc + f"\n\n_[stream interrupted: {exc}]_" return yield (f"**No model server at `{BASE_URL}`** ({exc}).\n\n" f"Falling back to echo: you said _{message}_") demo = gr.ChatInterface( fn=respond, additional_inputs=[ gr.Textbox(value="You are a concise assistant.", label="System prompt"), gr.Slider(0.0, 1.5, value=0.7, step=0.05, label="Temperature"), ], title="Local LLM chat (offline)", description=f"Endpoint `{BASE_URL}`, model `{MODEL}`.", **_CHAT_KW, ) if __name__ == "__main__": launch(demo)