GitHub

Example: AI chat

Streaming replies with a Stop button and Markdown, from Anthropic, an OpenAI-compatible server or a demo model.

  • 180 lines of Python
  • 1 page
  • 1 server function
  • 985 B page JS (gzip)
This example needs its server (database, sessions or server functions). Run it locally:
Terminal
pyweb dev examples/ai-chat/app.pyweb

#Source

examples/ai-chat/app.pyweb
"""AI chat: a streaming server function, a Stop button and Markdown answers.

Pick a model with environment variables (the key never reaches the browser):

    ANTHROPIC_API_KEY=...                      Anthropic (AI_MODEL defaults to claude-sonnet-5-5)
    OPENAI_API_KEY=... AI_MODEL=...            OpenAI, or any OpenAI-compatible server
    OPENAI_BASE_URL=http://localhost:11434/v1  (Ollama, vLLM, LM Studio, ...; key optional there)

With none of them set, a small demo model answers so the app works offline.
"""

import json
import os
import time
import urllib.error
import urllib.request

from pyweb import App, Markdown, RPCError, server

app = App(title="AI chat", stylesheets=["/static/app.css"])

SYSTEM = "You are a helpful assistant. Answer in Markdown."
MAX_MESSAGES = 20
MAX_CHARS = 4000


def provider():
    if os.environ.get("ANTHROPIC_API_KEY"):
        return "anthropic"
    if os.environ.get("OPENAI_API_KEY") or os.environ.get("OPENAI_BASE_URL"):
        return "openai"
    return "demo"


def describe():
    name = provider()
    if name == "demo":
        return "Demo model (set ANTHROPIC_API_KEY or OPENAI_API_KEY for a real one)"
    model = os.environ.get("AI_MODEL") or ("claude-sonnet-5-5" if name == "anthropic" else "?")
    return f"{name}: {model}"


def clean(messages):
    """The conversation as the model APIs want it, checked and trimmed."""
    out = []
    for m in messages[-MAX_MESSAGES:]:
        if not isinstance(m, dict) or m.get("role") not in ("user", "assistant"):
            raise RPCError("validation_error", "messages must be {role, content} objects")
        out.append({"role": m["role"], "content": str(m.get("content") or "")[:MAX_CHARS]})
    if not out or out[-1]["role"] != "user":
        raise RPCError("validation_error", "the last message must be from the user")
    return out


def post_lines(url, headers, body):
    """POST JSON and yield the response's lines as they arrive (closing the connection stops the model)."""
    req = urllib.request.Request(url, data=json.dumps(body).encode(), method="POST",
                                 headers={"Content-Type": "application/json", **headers})
    try:
        with urllib.request.urlopen(req, timeout=60) as resp:
            for raw in resp:
                yield raw.decode("utf-8", "replace").strip()
    except urllib.error.HTTPError as exc:
        detail = exc.read().decode("utf-8", "replace")[:300]
        raise RPCError("unavailable", f"the model API answered {exc.code}: {detail}") from None
    except OSError as exc:
        raise RPCError("unavailable", f"couldn't reach the model API: {exc}") from None


def anthropic(messages):
    body = {"model": os.environ.get("AI_MODEL", "claude-sonnet-5-5"), "max_tokens": 2048,
            "system": SYSTEM, "messages": messages, "stream": True}
    headers = {"x-api-key": os.environ["ANTHROPIC_API_KEY"], "anthropic-version": "2023-06-01"}
    for line in post_lines("https://api.anthropic.com/v1/messages", headers, body):
        if not line.startswith("data:"):
            continue
        event = json.loads(line[5:])
        if event.get("type") == "content_block_delta" and event["delta"].get("type") == "text_delta":
            yield event["delta"]["text"]
        elif event.get("type") == "error":
            raise RPCError("unavailable", event["error"].get("message", "model error"))


def openai(messages):
    model = os.environ.get("AI_MODEL")
    if not model:
        raise RPCError("unavailable", "set AI_MODEL to the model name to use")
    base = os.environ.get("OPENAI_BASE_URL", "https://api.openai.com/v1").rstrip("/")
    key = os.environ.get("OPENAI_API_KEY")
    headers = {"Authorization": f"Bearer {key}"} if key else {}
    body = {"model": model, "stream": True, "messages": [{"role": "system", "content": SYSTEM}] + messages}
    for line in post_lines(base + "/chat/completions", headers, body):
        if not line.startswith("data:") or line == "data: [DONE]":
            continue
        choices = json.loads(line[5:]).get("choices") or [{}]
        text = (choices[0].get("delta") or {}).get("content")
        if text:
            yield text


def demo(messages):
    question = messages[-1]["content"]
    answer = (f"You asked: **{question.strip()[:200]}**\n\n"
              "I'm the built-in demo model, so I can't really answer, but this reply shows what the app does:\n\n"
              "- it **streams**: each word arrives as the server `yield`s it;\n"
              "- **Stop** cancels the request, which closes the generator on the server;\n"
              "- replies are *Markdown*, rendered safely as they arrive.\n\n"
              "Set `ANTHROPIC_API_KEY` or `OPENAI_API_KEY` and restart to talk to a real model.")
    for word in answer.split(" "):
        time.sleep(0.03)
        yield word + " "


@server
def reply(messages: list):
    """Stream the assistant's next message, a piece at a time."""
    history = clean(messages)
    model = {"anthropic": anthropic, "openai": openai}.get(provider(), demo)
    yield from model(history)


@app.page("/", title="AI chat")
def Chat():
    model_name = describe()
    messages = []
    draft = ""
    busy = False
    stream = None
    error = ""

    async def send():
        text = draft.strip()
        if not text or busy:
            return
        draft = ""
        error = ""
        messages.append({"role": "user", "content": text})
        messages.append({"role": "assistant", "content": ""})
        busy = True
        stream = reply(messages[:-1])
        try:
            async for piece in stream:
                messages[-1]["content"] += piece
        except RPCError as e:
            error = str(e)
        busy = False

    def stop():
        if stream:
            stream.cancel()

    def on_unmount():
        stop()

    <main>
        <h1>Chat</h1>
        <p class="muted">{model_name}</p>
        <div class="chat">
            for m in messages:
                <div class={"bubble " + m["role"]}>
                    if m["role"] == "assistant":
                        if m["content"]:
                            <Markdown text={m["content"]} />
                        else:
                            <span class="typing">Thinking…</span>
                    else:
                        {m["content"]}
                </div>
            if not messages:
                <p class="empty">Ask something to start.</p>
        </div>
        <p class="error">{error}</p>
        <form onsubmit={send}>
            <input id="prompt" bind={draft} placeholder="Message" autocomplete="off" />
            if busy:
                <button id="stop" type="button" onclick={stop}>Stop</button>
            else:
                <button id="send">Send</button>
        </form>
    </main>

#What runs where

Output of pyweb inspect: every page variable, handler and server function, where it runs and why.

Output
page Chat  route=/
  server   model_name     computed per request on the server; value sent because browser code reads it
  browser  messages       reactive state: .append() in send(); literal initial value
  browser  draft          reactive state: bound to an input (line 174); literal initial value
  browser  busy           reactive state: assigned in send(); literal initial value
  browser  stream         reactive state: assigned in send(); literal initial value
  browser  error          reactive state: assigned in send(); literal initial value
  browser  send           event handler (compiled to JavaScript)
  browser  stop           event handler (compiled to JavaScript)
  browser  on_unmount     event handler (compiled to JavaScript)
rpc POST /__pyweb/rpc/reply  (messages: list) -> Any

#Generated JavaScript

The browser modules the compiler wrote (before minification). They import the shared runtime.

JavaScript
// Chat.js
import { h as $h, t as $t, dyn as $dyn, list as $list, when as $when, signal as $signal, computed as $computed, mount as $mount, onMount as $onMount, py as $py, rpc as $rpc, subscribe as $subscribe, slot as $slot, onCleanup as $onCleanup, live as $live } from "./runtime.js";
import { markdown as $markdown } from "./markdown.js";
function Chat($s) {
  const model_name = $s["model_name"];
  const messages = $signal("messages" in $s ? $s["messages"] : []);
  const draft = $signal("draft" in $s ? $s["draft"] : "");
  const busy = $signal("busy" in $s ? $s["busy"] : false);
  const stream = $signal("stream" in $s ? $s["stream"] : null);
  const error = $signal("error" in $s ? $s["error"] : "");
  async function send() {
    let e, piece, text;
    text = $py.m(draft(), "strip");
    if ((!$py.truth(text) || $py.truth(busy()))) {
      return;
    }
    draft("");
    error("");
    $py.mut(messages, [], ($v) => $py.m($v, "append", {"role": "user", "content": text}));
    $py.mut(messages, [], ($v) => $py.m($v, "append", {"role": "assistant", "content": ""}));
    busy(true);
    stream($rpc.stream("reply", {"messages": $py.slice(messages(), null, (-1), null)}));
    try {
      for await (piece of $py.aiter(stream())) {
        { const $k1 = [(-1), "content"]; $py.setp(messages, $k1, $py.add($py.getp(messages.peek(), $k1), piece)); }
      }
    } catch ($err) {
      if ($py.exc($err, ["RPCError"])) {
        e = $err;
        error($py.str(e));
      } else { throw $err; }
    }
    busy(false);
  }
  function stop() {
    if ($py.truth(stream())) {
      stream().cancel();
    }
  }
  function on_unmount() {
    stop();
  }
  $onCleanup(on_unmount);
  return [$h("main", null, () => [
      $h("h1", null, () => [$t("Chat")]),
      $h("p", {"class": "muted"}, () => [$t($py.text(model_name))]),
      $h("div", {"class": "chat"}, () => [
        $list(() => messages(), (m) => [$h("div", {"class": $py.add("bubble ", $py.at(m, "role"))}, () => [$when(() => ($py.at(m, "role") === "assistant"), () => [$when(() => $py.truth($py.at(m, "content")), () => [$markdown($py.at(m, "content"), null)], () => [$h("span", {"class": "typing"}, () => [$t("Thinking\u2026")])])], () => [$t($py.text($py.at(m, "content")))])])]),
        $when(() => !$py.truth(messages()), () => [$h("p", {"class": "empty"}, () => [$t("Ask something to start.")])])
      ]),
      $h("p", {"class": "error"}, () => [$dyn(() => error())]),
      $h("form", {"onsubmit": send}, () => [
        $h("input", {"id": "prompt", "$bind": draft, "placeholder": "Message", "autocomplete": "off"}),
        $when(() => $py.truth(busy()), () => [$h("button", {"id": "stop", "type": "button", "onclick": stop}, () => [$t("Stop")])], () => [$h("button", {"id": "send"}, () => [$t("Send")])])
      ])
    ])];
}
$mount("Chat", Chat);
Edit this page on GitHub