commit d1b512ff778fe1ed2e2cd867ea78b25cab1b72fb Author: bdeb1337 Date: Mon Oct 5 07:10:07 2026 +0200 feat: localchat, a Go + Templ + HTMX chat app for local Gemma models Streams replies over SSE from any OpenAI-compatible server (oMLX, llama.cpp, Ollama, ...). Single binary that can install itself as an OS service; docker compose bundles llama.cpp + Gemma 4 E2B. diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 0000000..69ecd3a --- /dev/null +++ b/.dockerignore @@ -0,0 +1,6 @@ +bin/ +docs/ +scripts/ +.env +.git +*.md diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..de5c00c --- /dev/null +++ b/.env.example @@ -0,0 +1,14 @@ +# Where your OpenAI-compatible model server lives (llama.cpp, oMLX, Ollama, LM Studio, ...) +LLM_BASE_URL=http://127.0.0.1:8080/v1 +# Leave empty to use the first model the server lists +LLM_MODEL= +# Only if your server requires one +LLM_API_KEY= + +# Make it yours +LOCALCHAT_TITLE=localchat +# SYSTEM_PROMPT="You are a patient Go tutor. Keep answers short and show small code examples." + +# LOCALCHAT_ADDR=127.0.0.1:3000 +# MAX_HISTORY=20 +# LLM_TIMEOUT=5m diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..b236934 --- /dev/null +++ b/.gitignore @@ -0,0 +1,6 @@ +bin/ +.env +*.local +dist/ +dev/ +.DS_Store diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000..f4d8637 --- /dev/null +++ b/Dockerfile @@ -0,0 +1,17 @@ +# syntax=docker/dockerfile:1 +FROM golang:1.27-alpine AS build +WORKDIR /src +COPY go.mod go.sum ./ +RUN go mod download +COPY . . +ARG VERSION=dev +# Templ output (*_templ.go) is committed, so no generate step is needed here. +RUN CGO_ENABLED=0 go build -trimpath -ldflags "-s -w -X main.version=${VERSION}" -o /localchat ./cmd/localchat + +FROM gcr.io/distroless/static-debian12:nonroot +COPY --from=build /localchat /localchat +ENV LOCALCHAT_ADDR=0.0.0.0:3000 +EXPOSE 3000 +HEALTHCHECK --interval=30s --timeout=5s CMD ["/localchat", "healthcheck"] +ENTRYPOINT ["/localchat"] +CMD ["serve"] diff --git a/Makefile b/Makefile new file mode 100644 index 0000000..db8d6c7 --- /dev/null +++ b/Makefile @@ -0,0 +1,34 @@ +VERSION ?= $(shell git describe --tags --always --dirty 2>/dev/null || echo dev) +LDFLAGS := -s -w -X main.version=$(VERSION) +PLATFORMS := linux/amd64 linux/arm64 darwin/arm64 darwin/amd64 windows/amd64 windows/arm64 + +.PHONY: generate build run test e2e dist up down + +generate: ## regenerate Templ components + go tool templ generate + +build: generate ## build ./bin/localchat for this machine + go build -trimpath -ldflags "$(LDFLAGS)" -o bin/localchat ./cmd/localchat + +run: build ## run in the foreground (reads ./.env) + ./bin/localchat --debug + +test: ## unit + handler tests (no model needed) + go test -race ./... + +e2e: ## browser test against a running instance + real model + uv run scripts/e2e.py + +dist: generate ## cross-compile for every platform into ./dist + @for p in $(PLATFORMS); do \ + os=$${p%/*}; arch=$${p#*/}; ext=$$( [ $$os = windows ] && echo .exe ); \ + echo " $$os/$$arch"; \ + CGO_ENABLED=0 GOOS=$$os GOARCH=$$arch go build -trimpath -ldflags "$(LDFLAGS)" \ + -o dist/localchat-$$os-$$arch$$ext ./cmd/localchat || exit 1; \ + done + +up: ## start app + llama.cpp + Gemma in containers + docker compose up -d --build + +down: + docker compose down diff --git a/README.md b/README.md new file mode 100644 index 0000000..ea07b08 --- /dev/null +++ b/README.md @@ -0,0 +1,125 @@ +# localchat + +A small, private chat app for a **local, open-weight AI model**: **Go + Templ + HTMX** on top of any OpenAI-compatible model server. It defaults to **Gemma 4 E2B**, which runs on a normal laptop. + +- 🔒 **Private:** prompts and replies never leave your machine. +- ✈️ **Works offline** once the model is downloaded. +- 💸 **Free to run:** no API keys, no per-token billing. +- 🔁 **Model-agnostic:** oMLX, llama.cpp, Ollama, LM Studio, Lemonade, vLLM… change one env var. +- 📦 **One ~9 MB binary** for Linux, macOS and Windows. It can install itself as a background service. +- 🐳 **Or `docker compose up`** for the app plus llama.cpp plus Gemma in one go. + +![screenshot](docs/screenshot.png) + +## Quick start + +### Option A: containers (easiest) + +```sh +docker compose up -d # first run downloads Gemma 4 E2B (~3 GB) +open http://localhost:3000 +``` + +To use a different model, set `LLM_HF_REPO=unsloth/gemma-4-E4B-it-GGUF:Q4_K_M docker compose up -d`. + +### Option B: binary + a model server you already run + +```sh +make build # or grab a binary from `make dist` +cp .env.example .env # point it at your model server +./bin/localchat # → http://127.0.0.1:3000 +``` + +Common `LLM_BASE_URL` values: + +| Server | `LLM_BASE_URL` | +|---|---| +| llama.cpp `llama-server` | `http://127.0.0.1:8080/v1` | +| oMLX (Apple Silicon) | `http://127.0.0.1:8000/v1` | +| Ollama | `http://127.0.0.1:11434/v1` | +| LM Studio | `http://127.0.0.1:1234/v1` | +| Lemonade (AMD) | `http://127.0.0.1:13305/api/v1` | + +For example, to get a model running quickly with llama.cpp: + +```sh +brew install llama.cpp # or a release from github.com/ggml-org/llama.cpp +llama-server -hf unsloth/gemma-4-E2B-it-GGUF:Q4_K_M --no-mmproj +``` + +### Option C: run it as a background service + +The binary registers itself with the OS service manager (systemd, launchd or the Windows Service Control Manager): + +```sh +./bin/localchat install --user # per-user (launchd agent / systemd --user); drop --user for system-wide (needs sudo/admin) +./bin/localchat start --user +./bin/localchat status --user +./bin/localchat stop --user && ./bin/localchat uninstall --user +``` + +The service gets the absolute path of your `.env` (or `--env-file`), so it runs with the same settings you tested with. + +## Configuration + +Set these as environment variables or in a `.env` file. Real environment variables take precedence over the file. + +| Variable | Default | | +|---|---|---| +| `LLM_BASE_URL` | `http://127.0.0.1:8000/v1` | OpenAI-compatible API root | +| `LLM_MODEL` | *(first model the server lists)* | model id | +| `LLM_API_KEY` | | bearer token, if your server needs one | +| `SYSTEM_PROMPT` | friendly, concise assistant | give it a personality or a purpose | +| `LOCALCHAT_TITLE` | `localchat` | name in the header and tab | +| `LOCALCHAT_ADDR` | `127.0.0.1:3000` | listen address (`0.0.0.0:3000` to share on your LAN) | +| `MAX_HISTORY` | `20` | past messages sent as context | +| `LLM_TIMEOUT` | `5m` | maximum time for one reply | + +## How it works + +``` +Browser ── HTMX + SSE extension + │ POST /chat → returns the user bubble + an empty reply bubble + │ GET /chat/stream/{id} ← server-sent events: "token" (append) … "done" (swap in Markdown) +Go (net/http + Templ) + │ POST /v1/chat/completions {stream: true} +Model server (llama.cpp / oMLX / Ollama / …) ── Gemma 4 E2B +``` + +1. The form posts with `hx-post`. The server stores the message and returns two Templ fragments: your message, and an assistant bubble with `sse-connect="/chat/stream/{id}"`. +2. The SSE handler claims that reply, sends the conversation to the model, and forwards each chunk as an HTML-escaped `token` event. HTMX appends each one (`hx-swap="beforeend"`), so the reply types out live. +3. When the model finishes, a `done` event replaces the whole bubble with the reply rendered as Markdown (goldmark, with raw HTML stripped). That also removes the `sse-connect` element, which closes the stream. +4. Browsers automatically reconnect an EventSource. A reconnect for a reply that is already claimed or finished gets the final state instead of a second generation. + +There is no JavaScript framework and no build step. The only JS is htmx, its SSE extension and about 40 lines of UX glue. All of it is embedded in the binary, so the app works fully offline. + +## Project layout + +``` +cmd/localchat/ entry point, CLI + service install/start/stop +internal/config/ env + .env loading +internal/llm/ tiny OpenAI-compatible streaming client (stdlib only) +internal/chat/ in-memory conversations, one per browser session +internal/web/ routes, SSE streaming, embedded static assets +internal/web/views/ Templ components (*.templ → generated *_templ.go) +scripts/e2e.py Playwright browser test against a real model +compose.yaml, Dockerfile +``` + +## Development + +```sh +make test # unit + handler tests with a fake model server, -race +make run # build + run with ./.env +make e2e # headless browser test against the running app + real model +make dist # cross-compile linux/darwin/windows × amd64/arm64 +``` + +After editing a `.templ` file, run `make generate` (or `go tool templ generate --watch`). The generated `*_templ.go` files are committed, so `go build` and Docker builds don't need Templ installed. + +## Good to know + +- Conversations live in memory and are lost on restart. That's on purpose for privacy, and simple enough to swap for SQLite. +- There is no authentication. It listens on localhost by default. Put a reverse proxy with auth in front before exposing it. +- Speed in Docker depends on how many CPUs the Docker VM gets. With Docker Desktop's 1-CPU setting, Gemma E2B takes about 20 s to start replying, so give it more cores. On a Mac, running the model natively (oMLX or `llama-server`, which use the GPU via Metal) is far faster: about 1.5 s to the first token on an M3. +- GPU in Docker depends on the platform. The compose file runs on CPU everywhere, which E2B handles fine. On Linux with an NVIDIA or AMD GPU, use the `server-cuda` or `server-rocm` llama.cpp image tags. diff --git a/cmd/localchat/main.go b/cmd/localchat/main.go new file mode 100644 index 0000000..45ff18f --- /dev/null +++ b/cmd/localchat/main.go @@ -0,0 +1,201 @@ +// Command localchat is a chat UI for a local, open-weight model. +// +// It runs in the foreground, or registers itself as a background service +// with the OS (systemd on Linux, launchd on macOS, the Service Control +// Manager on Windows). +package main + +import ( + "context" + "errors" + "flag" + "fmt" + "log/slog" + "net" + "net/http" + "os" + "path/filepath" + "time" + + "github.com/kardianos/service" + + "git.b0b.be/bdeb/localchat/internal/config" + "git.b0b.be/bdeb/localchat/internal/web" +) + +var version = "dev" // set with -ldflags "-X main.version=..." + +const usage = `localchat %s: chat with a local AI model in your browser. + +Usage: + localchat [command] [flags] + +Commands: + serve run in the foreground (default) + install register as a background service that starts on boot/login + uninstall remove the service + start start the installed service + stop stop the installed service + restart restart the installed service + status show whether the service is running + healthcheck exit 0 if the server answers (for container health checks) + version print the version + +Flags: +` + +func main() { + if err := run(os.Args[1:]); err != nil { + fmt.Fprintln(os.Stderr, "localchat:", err) + os.Exit(1) + } +} + +func run(args []string) error { + cmd := "serve" + if len(args) > 0 && args[0] != "" && args[0][0] != '-' { + cmd, args = args[0], args[1:] + } + + fs := flag.NewFlagSet("localchat", flag.ContinueOnError) + envFile := fs.String("env-file", "", "load settings from this KEY=VALUE file (default: ./.env if present)") + user := fs.Bool("user", false, "install as a per-user service (launchd agent / systemd --user) instead of system-wide") + debug := fs.Bool("debug", false, "verbose logging") + fs.Usage = func() { + fmt.Fprintf(fs.Output(), usage, version) + fs.PrintDefaults() + } + if err := fs.Parse(args); err != nil { + if errors.Is(err, flag.ErrHelp) { + return nil + } + return err + } + + if cmd == "version" { + fmt.Println(version) + return nil + } + + if *envFile == "" { + if _, err := os.Stat(".env"); err == nil { + *envFile = ".env" + } + } + if *envFile != "" { + abs, err := filepath.Abs(*envFile) + if err != nil { + return err + } + *envFile = abs + if err := config.LoadEnvFile(abs); err != nil { + return fmt.Errorf("env file: %w", err) + } + } + + level := slog.LevelInfo + if *debug { + level = slog.LevelDebug + } + log := slog.New(slog.NewTextHandler(os.Stderr, &slog.HandlerOptions{Level: level})) + + // The service manager re-runs us as "localchat serve --env-file ", + // so the installed service sees the same settings as this invocation. + svcArgs := []string{"serve"} + if *envFile != "" { + svcArgs = append(svcArgs, "--env-file", *envFile) + } + wd, _ := os.Getwd() + prg := &program{log: log} + svc, err := service.New(prg, &service.Config{ + Name: "localchat", + DisplayName: "localchat", + Description: "Chat with a local AI model in your browser.", + Arguments: svcArgs, + WorkingDirectory: wd, + Option: service.KeyValue{"UserService": *user}, + }) + if err != nil { + return err + } + + switch cmd { + case "healthcheck": + return healthcheck() + case "serve": + return svc.Run() // handles Ctrl+C in a terminal and stop requests from the OS + case "status": + st, err := svc.Status() + if err != nil { + return err + } + fmt.Println(map[service.Status]string{ + service.StatusRunning: "running", + service.StatusStopped: "stopped", + }[st]) + return nil + case "install", "uninstall", "start", "stop", "restart": + if err := service.Control(svc, cmd); err != nil { + return err + } + fmt.Printf("localchat: %s ok\n", cmd) + return nil + default: + fs.Usage() + return fmt.Errorf("unknown command %q", cmd) + } +} + +func healthcheck() error { + cfg, err := config.Load() + if err != nil { + return err + } + host, port, err := net.SplitHostPort(cfg.Addr) + if err != nil { + return err + } + if host == "" || host == "0.0.0.0" || host == "::" { + host = "127.0.0.1" + } + c := &http.Client{Timeout: 3 * time.Second} + resp, err := c.Get("http://" + net.JoinHostPort(host, port) + "/healthz") + if err != nil { + return err + } + resp.Body.Close() + if resp.StatusCode != http.StatusOK { + return fmt.Errorf("healthz: %s", resp.Status) + } + return nil +} + +// program adapts the web server to the service lifecycle. +type program struct { + log *slog.Logger + cancel context.CancelFunc + done chan struct{} +} + +func (p *program) Start(service.Service) error { + cfg, err := config.Load() + if err != nil { + return err + } + ctx, cancel := context.WithCancel(context.Background()) + p.cancel, p.done = cancel, make(chan struct{}) + go func() { + defer close(p.done) + if err := web.New(cfg, p.log).Run(ctx); err != nil { + p.log.Error("server stopped", "err", err) + os.Exit(1) + } + }() + return nil +} + +func (p *program) Stop(service.Service) error { + p.cancel() + <-p.done + return nil +} diff --git a/compose.yaml b/compose.yaml new file mode 100644 index 0000000..9fe0aba --- /dev/null +++ b/compose.yaml @@ -0,0 +1,44 @@ +# localchat + a local Gemma 4 E2B model served by llama.cpp. +# +# docker compose up -d # first start downloads the model (~3 GB) +# open http://localhost:3000 +# +# Swap the model with LLM_HF_REPO, e.g. LLM_HF_REPO=unsloth/gemma-4-E4B-it-GGUF:Q4_K_M +services: + llm: + image: ghcr.io/ggml-org/llama.cpp:server + command: + - -hf + - ${LLM_HF_REPO:-unsloth/gemma-4-E2B-it-GGUF:Q4_K_M} + - --no-mmproj # text-only chat; skip the vision projector download + - --ctx-size + - "8192" + - --port + - "8080" + environment: + LLAMA_CACHE: /models + volumes: + - models:/models + healthcheck: + test: ["CMD", "curl", "-fs", "http://localhost:8080/health"] + interval: 10s + timeout: 5s + start_period: 15m # first boot downloads the model + restart: unless-stopped + + app: + build: . + image: localchat:latest + ports: + - "3000:3000" + environment: + LLM_BASE_URL: http://llm:8080/v1 + LOCALCHAT_TITLE: ${LOCALCHAT_TITLE:-localchat} + SYSTEM_PROMPT: ${SYSTEM_PROMPT:-} + depends_on: + llm: + condition: service_healthy + restart: unless-stopped + +volumes: + models: diff --git a/docs/screenshot-dark.png b/docs/screenshot-dark.png new file mode 100644 index 0000000..90681ed Binary files /dev/null and b/docs/screenshot-dark.png differ diff --git a/docs/screenshot.png b/docs/screenshot.png new file mode 100644 index 0000000..efd04b6 Binary files /dev/null and b/docs/screenshot.png differ diff --git a/go.mod b/go.mod new file mode 100644 index 0000000..dbc31d3 --- /dev/null +++ b/go.mod @@ -0,0 +1,28 @@ +module git.b0b.be/bdeb/localchat + +go 1.27.1 + +require ( + github.com/a-h/templ v0.3.1070 + github.com/kardianos/service v1.3.0 + github.com/yuin/goldmark v1.8.6 +) + +require ( + github.com/a-h/parse v0.0.0-20250122154542-74294addb73e // indirect + github.com/andybalholm/brotli v1.2.6 // indirect + github.com/cenkalti/backoff/v4 v4.3.0 // indirect + github.com/cli/browser v1.3.0 // indirect + github.com/fatih/color v1.19.0 // indirect + github.com/fsnotify/fsnotify v1.10.1 // indirect + github.com/mattn/go-colorable v0.1.15 // indirect + github.com/mattn/go-isatty v0.0.24 // indirect + github.com/natefinch/atomic v1.0.1 // indirect + golang.org/x/mod v0.41.0 // indirect + golang.org/x/net v0.59.0 // indirect + golang.org/x/sync v0.23.0 // indirect + golang.org/x/sys v0.48.0 // indirect + golang.org/x/tools v0.51.0 // indirect +) + +tool github.com/a-h/templ/cmd/templ diff --git a/go.sum b/go.sum new file mode 100644 index 0000000..311a8c1 --- /dev/null +++ b/go.sum @@ -0,0 +1,46 @@ +github.com/a-h/parse v0.0.0-20250122154542-74294addb73e h1:HjVbSQHy+dnlS6C3XajZ69NYAb5jbGNfHanvm1+iYlo= +github.com/a-h/parse v0.0.0-20250122154542-74294addb73e/go.mod h1:3mnrkvGpurZ4ZrTDbYU84xhwXW2TjTKShSwjRi2ihfQ= +github.com/a-h/templ v0.3.1070 h1:3iCOXHXqFXa99OQzgkUOLJbNzv/QjFGLwKZkBAg+ADA= +github.com/a-h/templ v0.3.1070/go.mod h1:iIqB3g2RHBe/WlnPwa+Bvhh1TKXHWLRnBD9NWHOLm78= +github.com/andybalholm/brotli v1.2.6 h1:ftYnfj6usCp+UGV5kSJ3+chpMQgU+gJf/AxsUQ52REI= +github.com/andybalholm/brotli v1.2.6/go.mod h1:rzTDkvFWvIrjDXZHkuS16NPggd91W3kUSvPlQ1pLaKY= +github.com/cenkalti/backoff/v4 v4.3.0 h1:MyRJ/UdXutAwSAT+s3wNd7MfTIcy71VQueUuFK343L8= +github.com/cenkalti/backoff/v4 v4.3.0/go.mod h1:Y3VNntkOUPxTVeUxJ/G5vcM//AlwfmyYozVcomhLiZE= +github.com/cli/browser v1.3.0 h1:LejqCrpWr+1pRqmEPDGnTZOjsMe7sehifLynZJuqJpo= +github.com/cli/browser v1.3.0/go.mod h1:HH8s+fOAxjhQoBUAsKuPCbqUuxZDhQ2/aD+SzsEfBTk= +github.com/davecgh/go-spew v1.1.1 h1:vj9j/u1bqnvCEfJOwUhtlOARqs3+rkHYY13jYWTU97c= +github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38= +github.com/fatih/color v1.19.0 h1:Zp3PiM21/9Ld6FzSKyL5c/BULoe/ONr9KlbYVOfG8+w= +github.com/fatih/color v1.19.0/go.mod h1:zNk67I0ZUT1bEGsSGyCZYZNrHuTkJJB+r6Q9VuMi0LE= +github.com/fsnotify/fsnotify v1.10.1 h1:b0/UzAf9yR5rhf3RPm9gf3ehBPpf0oZKIjtpKrx59Ho= +github.com/fsnotify/fsnotify v1.10.1/go.mod h1:TLheqan6HD6GBK6PrDWyDPBaEV8LspOxvPSjC+bVfgo= +github.com/google/go-cmp v0.7.0 h1:wk8382ETsv4JYUZwIsn6YpYiWiBsYLSJiTsyBybVuN8= +github.com/google/go-cmp v0.7.0/go.mod h1:pXiqmnSA92OHEEa9HXL2W4E7lf9JzCmGVUdgjX3N/iU= +github.com/kardianos/service v1.3.0 h1:/LGy+xPP2TM+GLTiCZ2di7cy0Jd/qrawlTUfqKYFdTI= +github.com/kardianos/service v1.3.0/go.mod h1:E4V9ufUuY82F7Ztlu1eN9VXWIQxg8NoLQlmFe0MtrXc= +github.com/mattn/go-colorable v0.1.15 h1:+u9SLTRGnXv73cEsnsmoZBom+dMU88B2M0aDcWy0/jY= +github.com/mattn/go-colorable v0.1.15/go.mod h1:6LmQG8QLFO4G5z1gPvYEzlUgJ2wF+stgPZH1UqBm1s8= +github.com/mattn/go-isatty v0.0.24 h1:tGZZoVgT/KiqK1c8ocVLeDS8BSWMRd47J3Lbz7vsReI= +github.com/mattn/go-isatty v0.0.24/go.mod h1:nMCL3Zebbrt45jsMDgnfIwz6ydEQApk5oEI3HqDio6A= +github.com/natefinch/atomic v1.0.1 h1:ZPYKxkqQOx3KZ+RsbnP/YsgvxWQPGxjC0oBt2AhwV0A= +github.com/natefinch/atomic v1.0.1/go.mod h1:N/D/ELrljoqDyT3rZrsUmtsuzvHkeB/wWjHV22AZRbM= +github.com/pmezard/go-difflib v1.0.0 h1:4DBwDE0NGyQoBHbLQYPwSUPoCMWR5BEzIk/f1lZbAQM= +github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4= +github.com/stretchr/testify v1.10.0 h1:Xv5erBjTwe/5IxqUQTdXv5kgmIvbHo3QQyRwhJsOfJA= +github.com/stretchr/testify v1.10.0/go.mod h1:r2ic/lqez/lEtzL7wO/rwa5dbSLXVDPFyf8C91i36aY= +github.com/xyproto/randomstring v1.0.5 h1:YtlWPoRdgMu3NZtP45drfy1GKoojuR7hmRcnhZqKjWU= +github.com/xyproto/randomstring v1.0.5/go.mod h1:rgmS5DeNXLivK7YprL0pY+lTuhNQW3iGxZ18UQApw/E= +github.com/yuin/goldmark v1.8.6 h1:d0VcaP1sx9GkFVkoW+KtggpGi2KZ965i14b0+bDQST4= +github.com/yuin/goldmark v1.8.6/go.mod h1:ip/1k0VRfGynBgxOz0yCqHrbZXhcjxyuS66Brc7iBKg= +golang.org/x/mod v0.41.0 h1:qJmnOUb4YB+FsEuM3HcWucdZASCPGhsX6uljO6pog0c= +golang.org/x/mod v0.41.0/go.mod h1:Ek9pY8RKWXwsWvd3rQiHYtMqkjSUV+s1Rj7j4H5Ur6o= +golang.org/x/net v0.59.0 h1:5zfYln+w5XCxwrnMMJPufRgNoXEaGxl0wo5GqPXyues= +golang.org/x/net v0.59.0/go.mod h1:2DA/G1UfVbCpQPeWTmMPGY7Cs2PkBkwu743bVX5PIVg= +golang.org/x/sync v0.23.0 h1:KameEIfc1IkluZyXWLn39Wd4tURc6GbCiISGiZm2bQk= +golang.org/x/sync v0.23.0/go.mod h1:sUUOizhqBxiL6pEWpqNLUiaJn1ShEbZ6BBqskPbjZm0= +golang.org/x/sys v0.48.0 h1:bbX/i/6MgT9BVLM9RT1thmxL04yeTAhbEz4SyadbXoo= +golang.org/x/sys v0.48.0/go.mod h1:hNLxWAXmnKAxqDtdwIYC4bM9oQPEecfsnNMuSxOs3og= +golang.org/x/tools v0.51.0 h1:k4Xc/1Om9jwkBJBo4NVLMSARBoWtK10mx+W5BnXCeAI= +golang.org/x/tools v0.51.0/go.mod h1:9eEncMayCV6zRMGhR5eZEC2iBx98qWcF1HZ9Z7wJOoA= +gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA= +gopkg.in/yaml.v3 v3.0.1/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= diff --git a/internal/chat/store.go b/internal/chat/store.go new file mode 100644 index 0000000..d6b075f --- /dev/null +++ b/internal/chat/store.go @@ -0,0 +1,179 @@ +// Package chat keeps conversations in memory, one per browser session. +package chat + +import ( + "crypto/rand" + "encoding/hex" + "sync" + "time" + + "git.b0b.be/bdeb/localchat/internal/llm" +) + +// State tracks an assistant reply through its lifecycle. +type State int + +const ( + Pending State = iota // created, nobody is generating it yet + Streaming // a stream handler owns it + Done // finished (successfully or with Err set) +) + +// Message is a chat turn shown in the UI. +type Message struct { + ID string + Role string // "user" or "assistant" + Content string + State State + Err string +} + +// Conversation is the history of one session. +type Conversation struct { + mu sync.Mutex + messages []*Message + lastUsed time.Time +} + +// Store maps session ids to conversations. +type Store struct { + mu sync.Mutex + convs map[string]*Conversation +} + +// NewStore returns an empty store. +func NewStore() *Store { + return &Store{convs: make(map[string]*Conversation)} +} + +// NewID returns a random hex id, suitable for sessions and messages. +func NewID() string { + b := make([]byte, 16) + rand.Read(b) + return hex.EncodeToString(b) +} + +// Get returns the conversation for a session, creating it if needed. +func (s *Store) Get(session string) *Conversation { + s.mu.Lock() + defer s.mu.Unlock() + c, ok := s.convs[session] + if !ok { + c = &Conversation{} + s.convs[session] = c + } + c.mu.Lock() + c.lastUsed = time.Now() + c.mu.Unlock() + return c +} + +// Reset forgets a session's conversation. +func (s *Store) Reset(session string) { + s.mu.Lock() + defer s.mu.Unlock() + delete(s.convs, session) +} + +// Prune drops conversations idle for longer than maxIdle. +func (s *Store) Prune(maxIdle time.Duration) { + cutoff := time.Now().Add(-maxIdle) + s.mu.Lock() + defer s.mu.Unlock() + for id, c := range s.convs { + c.mu.Lock() + idle := c.lastUsed.Before(cutoff) + c.mu.Unlock() + if idle { + delete(s.convs, id) + } + } +} + +// Messages returns a snapshot of the conversation. +func (c *Conversation) Messages() []Message { + c.mu.Lock() + defer c.mu.Unlock() + out := make([]Message, len(c.messages)) + for i, m := range c.messages { + out[i] = *m + } + return out +} + +// Ask appends a user message plus a pending assistant reply and returns both. +func (c *Conversation) Ask(text string) (user, reply Message) { + c.mu.Lock() + defer c.mu.Unlock() + u := &Message{ID: NewID(), Role: "user", Content: text, State: Done} + r := &Message{ID: NewID(), Role: "assistant", State: Pending} + c.messages = append(c.messages, u, r) + return *u, *r +} + +// Claim marks a pending reply as streaming and returns the context to send +// to the model: the system prompt plus up to maxHistory finished messages +// before the reply. ok is false if the reply does not exist or someone else +// already claimed it; msg then holds its current state (if it exists). +func (c *Conversation) Claim(id, systemPrompt string, maxHistory int) (prompt []llm.Message, msg Message, ok bool) { + c.mu.Lock() + defer c.mu.Unlock() + idx := c.index(id) + if idx < 0 { + return nil, Message{}, false + } + m := c.messages[idx] + if m.State != Pending { + return nil, *m, false + } + m.State = Streaming + + var history []llm.Message + for _, h := range c.messages[:idx] { + if h.State == Done && h.Err == "" && h.Content != "" { + history = append(history, llm.Message{Role: h.Role, Content: h.Content}) + } + } + if len(history) > maxHistory { + history = history[len(history)-maxHistory:] + } + if systemPrompt != "" { + prompt = append(prompt, llm.Message{Role: "system", Content: systemPrompt}) + } + return append(prompt, history...), *m, true +} + +// Append adds streamed text to a reply. +func (c *Conversation) Append(id, delta string) { + c.mu.Lock() + defer c.mu.Unlock() + if i := c.index(id); i >= 0 { + c.messages[i].Content += delta + } +} + +// Finish marks a reply as done, recording err if generation failed, and +// returns its final state. +func (c *Conversation) Finish(id string, err error) Message { + c.mu.Lock() + defer c.mu.Unlock() + i := c.index(id) + if i < 0 { + return Message{ID: id, Role: "assistant", State: Done} + } + m := c.messages[i] + m.State = Done + if err != nil { + m.Err = err.Error() + } + return *m +} + +func (c *Conversation) index(id string) int { + for i, m := range c.messages { + if m.ID == id { + return i + } + } + return -1 +} diff --git a/internal/config/config.go b/internal/config/config.go new file mode 100644 index 0000000..d4d36c6 --- /dev/null +++ b/internal/config/config.go @@ -0,0 +1,95 @@ +// Package config loads localchat settings from the environment and an +// optional .env file. +package config + +import ( + "bufio" + "fmt" + "os" + "strconv" + "strings" + "time" +) + +// DefaultSystemPrompt is used when SYSTEM_PROMPT is not set. +const DefaultSystemPrompt = "You are a friendly, concise assistant running fully offline on the user's own machine. Answer clearly. Use Markdown when it helps." + +// Config holds all runtime settings. +type Config struct { + Addr string // LOCALCHAT_ADDR: where the web UI listens + BaseURL string // LLM_BASE_URL: OpenAI-compatible API root, e.g. http://127.0.0.1:8000/v1 + APIKey string // LLM_API_KEY: optional bearer token for the model server + Model string // LLM_MODEL: model id; empty means "first model the server lists" + SystemPrompt string // SYSTEM_PROMPT + Title string // LOCALCHAT_TITLE: name shown in the UI + MaxHistory int // MAX_HISTORY: messages sent back to the model as context + Timeout time.Duration // LLM_TIMEOUT: upper bound for a single reply +} + +// Load reads the configuration from environment variables, applying defaults. +func Load() (Config, error) { + c := Config{ + Addr: env("LOCALCHAT_ADDR", "127.0.0.1:3000"), + BaseURL: strings.TrimRight(env("LLM_BASE_URL", "http://127.0.0.1:8000/v1"), "/"), + APIKey: os.Getenv("LLM_API_KEY"), + Model: os.Getenv("LLM_MODEL"), + SystemPrompt: env("SYSTEM_PROMPT", DefaultSystemPrompt), + Title: env("LOCALCHAT_TITLE", "localchat"), + MaxHistory: 20, + Timeout: 5 * time.Minute, + } + if v := os.Getenv("MAX_HISTORY"); v != "" { + n, err := strconv.Atoi(v) + if err != nil || n < 1 { + return c, fmt.Errorf("MAX_HISTORY must be a positive integer, got %q", v) + } + c.MaxHistory = n + } + if v := os.Getenv("LLM_TIMEOUT"); v != "" { + d, err := time.ParseDuration(v) + if err != nil { + return c, fmt.Errorf("LLM_TIMEOUT: %w", err) + } + c.Timeout = d + } + return c, nil +} + +func env(key, fallback string) string { + if v, ok := os.LookupEnv(key); ok && v != "" { + return v + } + return fallback +} + +// LoadEnvFile sets variables from a KEY=VALUE file. Variables that are +// already set in the environment win, so the real environment can override +// the file. Blank lines and lines starting with # are ignored; values may be +// wrapped in single or double quotes. +func LoadEnvFile(path string) error { + f, err := os.Open(path) + if err != nil { + return err + } + defer f.Close() + + sc := bufio.NewScanner(f) + for n := 1; sc.Scan(); n++ { + line := strings.TrimSpace(sc.Text()) + if line == "" || strings.HasPrefix(line, "#") { + continue + } + key, val, ok := strings.Cut(strings.TrimPrefix(line, "export "), "=") + if !ok { + return fmt.Errorf("%s:%d: expected KEY=VALUE", path, n) + } + key, val = strings.TrimSpace(key), strings.TrimSpace(val) + if len(val) >= 2 && (val[0] == '"' || val[0] == '\'') && val[len(val)-1] == val[0] { + val = val[1 : len(val)-1] + } + if _, set := os.LookupEnv(key); !set { + os.Setenv(key, val) + } + } + return sc.Err() +} diff --git a/internal/config/config_test.go b/internal/config/config_test.go new file mode 100644 index 0000000..5e0da96 --- /dev/null +++ b/internal/config/config_test.go @@ -0,0 +1,45 @@ +package config + +import ( + "os" + "path/filepath" + "testing" +) + +func TestLoadEnvFile(t *testing.T) { + path := filepath.Join(t.TempDir(), ".env") + os.WriteFile(path, []byte(` +# comment +LLM_MODEL=from-file +export LOCALCHAT_TITLE="My Chat" +SYSTEM_PROMPT='Talk like a pirate = arr' +LLM_BASE_URL=http://file/v1/ +`), 0o600) + + t.Setenv("LLM_BASE_URL", "http://env/v1") // real env wins over the file + for _, k := range []string{"LLM_MODEL", "LOCALCHAT_TITLE", "SYSTEM_PROMPT"} { + t.Setenv(k, "") + os.Unsetenv(k) + } + + if err := LoadEnvFile(path); err != nil { + t.Fatal(err) + } + c, err := Load() + if err != nil { + t.Fatal(err) + } + if c.Model != "from-file" || c.Title != "My Chat" || c.SystemPrompt != "Talk like a pirate = arr" { + t.Fatalf("unexpected config: %+v", c) + } + if c.BaseURL != "http://env/v1" { + t.Fatalf("env should override file, got %q", c.BaseURL) + } +} + +func TestLoadValidates(t *testing.T) { + t.Setenv("MAX_HISTORY", "zero") + if _, err := Load(); err == nil { + t.Fatal("want error for bad MAX_HISTORY") + } +} diff --git a/internal/llm/client.go b/internal/llm/client.go new file mode 100644 index 0000000..3bcdbd1 --- /dev/null +++ b/internal/llm/client.go @@ -0,0 +1,175 @@ +// Package llm is a minimal client for OpenAI-compatible chat APIs, as served +// by llama.cpp, oMLX, Ollama, LM Studio, Lemonade, vLLM and friends. +package llm + +import ( + "bufio" + "bytes" + "context" + "encoding/json" + "errors" + "fmt" + "io" + "net/http" + "strings" +) + +// Message is one chat turn. +type Message struct { + Role string `json:"role"` // "system", "user" or "assistant" + Content string `json:"content"` +} + +// Client talks to a single OpenAI-compatible endpoint. +type Client struct { + BaseURL string // e.g. http://127.0.0.1:8000/v1 + APIKey string + Model string // empty: use the first model the server lists + HTTP *http.Client +} + +func (c *Client) httpClient() *http.Client { + if c.HTTP != nil { + return c.HTTP + } + return http.DefaultClient +} + +func (c *Client) newRequest(ctx context.Context, method, path string, body io.Reader) (*http.Request, error) { + req, err := http.NewRequestWithContext(ctx, method, c.BaseURL+path, body) + if err != nil { + return nil, err + } + if body != nil { + req.Header.Set("Content-Type", "application/json") + } + if c.APIKey != "" { + req.Header.Set("Authorization", "Bearer "+c.APIKey) + } + return req, nil +} + +// Models lists the model ids the server offers. It doubles as a health check. +func (c *Client) Models(ctx context.Context) ([]string, error) { + req, err := c.newRequest(ctx, http.MethodGet, "/models", nil) + if err != nil { + return nil, err + } + resp, err := c.httpClient().Do(req) + if err != nil { + return nil, err + } + defer resp.Body.Close() + if err := checkStatus(resp); err != nil { + return nil, err + } + var out struct { + Data []struct { + ID string `json:"id"` + } `json:"data"` + } + if err := json.NewDecoder(resp.Body).Decode(&out); err != nil { + return nil, fmt.Errorf("decode models: %w", err) + } + ids := make([]string, len(out.Data)) + for i, m := range out.Data { + ids[i] = m.ID + } + return ids, nil +} + +// ResolveModel returns the configured model, or the first one the server +// lists when none is configured. +func (c *Client) ResolveModel(ctx context.Context) (string, error) { + if c.Model != "" { + return c.Model, nil + } + ids, err := c.Models(ctx) + if err != nil { + return "", err + } + if len(ids) == 0 { + return "", errors.New("model server lists no models; set LLM_MODEL") + } + return ids[0], nil +} + +// Stream sends the conversation and calls onDelta for every chunk of the +// reply as it arrives. It returns when the reply is complete, the context is +// cancelled, or onDelta returns an error. +func (c *Client) Stream(ctx context.Context, msgs []Message, onDelta func(string) error) error { + model, err := c.ResolveModel(ctx) + if err != nil { + return err + } + body, err := json.Marshal(map[string]any{ + "model": model, + "messages": msgs, + "stream": true, + }) + if err != nil { + return err + } + req, err := c.newRequest(ctx, http.MethodPost, "/chat/completions", bytes.NewReader(body)) + if err != nil { + return err + } + req.Header.Set("Accept", "text/event-stream") + resp, err := c.httpClient().Do(req) + if err != nil { + return err + } + defer resp.Body.Close() + if err := checkStatus(resp); err != nil { + return err + } + + sc := bufio.NewScanner(resp.Body) + sc.Buffer(make([]byte, 64*1024), 1024*1024) + for sc.Scan() { + data, ok := strings.CutPrefix(sc.Text(), "data:") + if !ok { + continue // blank separators, comments, "event:" lines + } + data = strings.TrimSpace(data) + if data == "[DONE]" { + return nil + } + var chunk struct { + Choices []struct { + Delta struct { + Content string `json:"content"` + } `json:"delta"` + } `json:"choices"` + Error *struct { + Message string `json:"message"` + } `json:"error"` + } + if err := json.Unmarshal([]byte(data), &chunk); err != nil { + return fmt.Errorf("decode stream chunk: %w", err) + } + if chunk.Error != nil { + return fmt.Errorf("model server: %s", chunk.Error.Message) + } + for _, ch := range chunk.Choices { + if ch.Delta.Content == "" { + continue + } + if err := onDelta(ch.Delta.Content); err != nil { + return err + } + } + } + if err := sc.Err(); err != nil { + return err + } + return nil // stream ended without [DONE]; treat what we got as the reply +} + +func checkStatus(resp *http.Response) error { + if resp.StatusCode < 300 { + return nil + } + msg, _ := io.ReadAll(io.LimitReader(resp.Body, 2048)) + return fmt.Errorf("model server returned %s: %s", resp.Status, bytes.TrimSpace(msg)) +} diff --git a/internal/llm/client_test.go b/internal/llm/client_test.go new file mode 100644 index 0000000..b61fad4 --- /dev/null +++ b/internal/llm/client_test.go @@ -0,0 +1,107 @@ +package llm + +import ( + "context" + "encoding/json" + "fmt" + "net/http" + "net/http/httptest" + "strings" + "testing" +) + +// fakeServer mimics an OpenAI-compatible server that streams the given chunks. +func fakeServer(t *testing.T, chunks []string) *httptest.Server { + t.Helper() + return httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + switch r.URL.Path { + case "/v1/models": + fmt.Fprint(w, `{"data":[{"id":"tiny-model"},{"id":"other"}]}`) + case "/v1/chat/completions": + if got := r.Header.Get("Authorization"); got != "Bearer secret" { + http.Error(w, `{"error":"bad key"}`, http.StatusUnauthorized) + return + } + var req struct { + Model string `json:"model"` + Stream bool `json:"stream"` + Msgs []Message `json:"messages"` + } + if err := json.NewDecoder(r.Body).Decode(&req); err != nil { + t.Errorf("decode request: %v", err) + } + if req.Model != "tiny-model" || !req.Stream || len(req.Msgs) != 1 { + t.Errorf("unexpected request: %+v", req) + } + w.Header().Set("Content-Type", "text/event-stream") + fmt.Fprint(w, ": keep-alive comment\n\n") + fmt.Fprint(w, `data: {"choices":[{"delta":{"role":"assistant"}}]}`+"\n\n") + for _, c := range chunks { + b, _ := json.Marshal(c) + fmt.Fprintf(w, `data: {"choices":[{"delta":{"content":%s}}]}`+"\n\n", b) + } + fmt.Fprint(w, "data: [DONE]\n\n") + default: + http.NotFound(w, r) + } + })) +} + +func TestStream(t *testing.T) { + srv := fakeServer(t, []string{"Hel", "lo", "\n", "world"}) + defer srv.Close() + + c := &Client{BaseURL: srv.URL + "/v1", APIKey: "secret"} // no model: auto-pick first + var got strings.Builder + err := c.Stream(context.Background(), []Message{{Role: "user", Content: "hi"}}, func(d string) error { + got.WriteString(d) + return nil + }) + if err != nil { + t.Fatal(err) + } + if got.String() != "Hello\nworld" { + t.Fatalf("got %q", got.String()) + } +} + +func TestStreamHTTPError(t *testing.T) { + srv := fakeServer(t, nil) + defer srv.Close() + + c := &Client{BaseURL: srv.URL + "/v1", APIKey: "wrong", Model: "tiny-model"} + err := c.Stream(context.Background(), []Message{{Role: "user", Content: "hi"}}, func(string) error { return nil }) + if err == nil || !strings.Contains(err.Error(), "401") { + t.Fatalf("want 401 error, got %v", err) + } +} + +func TestStreamCallbackErrorStops(t *testing.T) { + srv := fakeServer(t, []string{"a", "b", "c"}) + defer srv.Close() + + c := &Client{BaseURL: srv.URL + "/v1", APIKey: "secret"} + stop := fmt.Errorf("stop") + n := 0 + err := c.Stream(context.Background(), []Message{{Role: "user", Content: "hi"}}, func(string) error { + n++ + return stop + }) + if err != stop || n != 1 { + t.Fatalf("err=%v n=%d", err, n) + } +} + +func TestResolveModel(t *testing.T) { + srv := fakeServer(t, nil) + defer srv.Close() + + c := &Client{BaseURL: srv.URL + "/v1"} + if m, err := c.ResolveModel(context.Background()); err != nil || m != "tiny-model" { + t.Fatalf("auto: %q %v", m, err) + } + c.Model = "pinned" + if m, _ := c.ResolveModel(context.Background()); m != "pinned" { + t.Fatalf("pinned: %q", m) + } +} diff --git a/internal/web/server.go b/internal/web/server.go new file mode 100644 index 0000000..dbcc07a --- /dev/null +++ b/internal/web/server.go @@ -0,0 +1,241 @@ +// Package web serves the chat UI: Templ pages, HTMX fragments and an SSE +// endpoint that streams model replies token by token. +package web + +import ( + "bytes" + "context" + "embed" + "errors" + "fmt" + "log/slog" + "net/http" + "strings" + "time" + + "github.com/a-h/templ" + + "git.b0b.be/bdeb/localchat/internal/chat" + "git.b0b.be/bdeb/localchat/internal/config" + "git.b0b.be/bdeb/localchat/internal/llm" + "git.b0b.be/bdeb/localchat/internal/web/views" +) + +//go:embed static +var staticFS embed.FS + +const ( + sessionCookie = "localchat_sid" + maxMessageLen = 32 << 10 +) + +// Server is the chat web app. +type Server struct { + cfg config.Config + llm *llm.Client + store *chat.Store + log *slog.Logger +} + +// New wires a server from config. +func New(cfg config.Config, log *slog.Logger) *Server { + return &Server{ + cfg: cfg, + llm: &llm.Client{ + BaseURL: cfg.BaseURL, + APIKey: cfg.APIKey, + Model: cfg.Model, + HTTP: &http.Client{}, // per-request deadlines come from contexts + }, + store: chat.NewStore(), + log: log, + } +} + +// Handler returns the app's routes. +func (s *Server) Handler() http.Handler { + mux := http.NewServeMux() + mux.Handle("GET /static/", http.FileServerFS(staticFS)) + mux.HandleFunc("GET /{$}", s.index) + mux.HandleFunc("GET /health", s.health) + mux.HandleFunc("GET /healthz", func(w http.ResponseWriter, _ *http.Request) { w.Write([]byte("ok")) }) + mux.HandleFunc("POST /chat", s.send) + mux.HandleFunc("POST /chat/reset", s.reset) + mux.HandleFunc("GET /chat/stream/{id}", s.stream) + return s.withSession(mux) +} + +// Run serves until ctx is cancelled, then shuts down gracefully. +func (s *Server) Run(ctx context.Context) error { + srv := &http.Server{ + Addr: s.cfg.Addr, + Handler: s.Handler(), + ReadHeaderTimeout: 10 * time.Second, + // No WriteTimeout: replies stream for as long as the model talks. + // LLM_TIMEOUT bounds each reply instead. + } + + go func() { + t := time.NewTicker(10 * time.Minute) + defer t.Stop() + for { + select { + case <-ctx.Done(): + return + case <-t.C: + s.store.Prune(24 * time.Hour) + } + } + }() + + errc := make(chan error, 1) + go func() { errc <- srv.ListenAndServe() }() + s.log.Info("localchat listening", "url", "http://"+s.cfg.Addr, "llm", s.cfg.BaseURL, "model", s.cfg.Model) + + select { + case err := <-errc: + return err + case <-ctx.Done(): + shutdownCtx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + defer cancel() + return srv.Shutdown(shutdownCtx) + } +} + +type ctxKey struct{} + +// withSession gives every browser a random session cookie. +func (s *Server) withSession(next http.Handler) http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + sid := "" + if c, err := r.Cookie(sessionCookie); err == nil && len(c.Value) == 32 { + sid = c.Value + } else { + sid = chat.NewID() + http.SetCookie(w, &http.Cookie{ + Name: sessionCookie, + Value: sid, + Path: "/", + HttpOnly: true, + SameSite: http.SameSiteLaxMode, + }) + } + next.ServeHTTP(w, r.WithContext(context.WithValue(r.Context(), ctxKey{}, sid))) + }) +} + +func session(r *http.Request) string { + sid, _ := r.Context().Value(ctxKey{}).(string) + return sid +} + +func (s *Server) render(w http.ResponseWriter, r *http.Request, c templ.Component) { + w.Header().Set("Content-Type", "text/html; charset=utf-8") + if err := c.Render(r.Context(), w); err != nil { + s.log.Error("render", "err", err) + } +} + +func (s *Server) index(w http.ResponseWriter, r *http.Request) { + msgs := s.store.Get(session(r)).Messages() + s.render(w, r, views.Page(s.cfg.Title, msgs)) +} + +func (s *Server) health(w http.ResponseWriter, r *http.Request) { + ctx, cancel := context.WithTimeout(r.Context(), 3*time.Second) + defer cancel() + model, err := s.llm.ResolveModel(ctx) + if err == nil && s.cfg.Model != "" { + _, err = s.llm.Models(ctx) // configured model: still check the server is up + } + h := views.HealthInfo{OK: err == nil, Model: model} + if err != nil { + h.Err = err.Error() + } + s.render(w, r, views.Health(h)) +} + +func (s *Server) send(w http.ResponseWriter, r *http.Request) { + r.Body = http.MaxBytesReader(w, r.Body, maxMessageLen) + text := strings.TrimSpace(r.FormValue("message")) + if text == "" { + http.Error(w, "message is empty or too long", http.StatusBadRequest) + return + } + user, reply := s.store.Get(session(r)).Ask(text) + s.render(w, r, views.Exchange(user, reply)) +} + +func (s *Server) reset(w http.ResponseWriter, r *http.Request) { + s.store.Reset(session(r)) + s.render(w, r, views.Messages(nil)) +} + +// stream generates a reply and pushes it to the browser as server-sent +// events. EventSource reconnects automatically, so a reply that is already +// claimed or finished is answered with its current state instead of being +// generated twice. +func (s *Server) stream(w http.ResponseWriter, r *http.Request) { + id := r.PathValue("id") + conv := s.store.Get(session(r)) + + w.Header().Set("Content-Type", "text/event-stream") + w.Header().Set("Cache-Control", "no-cache") + w.Header().Set("X-Accel-Buffering", "no") // don't let reverse proxies buffer + sse := &sseWriter{w: w, rc: http.NewResponseController(w)} + + prompt, msg, ok := conv.Claim(id, s.cfg.SystemPrompt, s.cfg.MaxHistory) + if !ok { + if msg.ID == "" { + msg = chat.Message{ID: id, Role: "assistant", State: chat.Done, + Err: "This reply was lost, probably because the server restarted. Please ask again."} + } + sse.component(r.Context(), "done", views.MessageView(msg)) + return + } + + ctx, cancel := context.WithTimeout(r.Context(), s.cfg.Timeout) + defer cancel() + start := time.Now() + err := s.llm.Stream(ctx, prompt, func(delta string) error { + conv.Append(id, delta) + return sse.component(ctx, "token", views.Token(delta)) + }) + + if r.Context().Err() != nil { // browser went away; keep what we have + conv.Finish(id, errors.New("interrupted")) + return + } + if errors.Is(err, context.DeadlineExceeded) { + err = fmt.Errorf("reply took longer than %s and was cut off", s.cfg.Timeout) + } + if err != nil { + s.log.Warn("generation failed", "err", err) + } else { + s.log.Debug("reply done", "id", id, "took", time.Since(start)) + } + sse.component(r.Context(), "done", views.MessageView(conv.Finish(id, err))) +} + +type sseWriter struct { + w http.ResponseWriter + rc *http.ResponseController +} + +// component sends one rendered Templ component as a named SSE event. +func (s *sseWriter) component(ctx context.Context, event string, c templ.Component) error { + var buf bytes.Buffer + if err := c.Render(ctx, &buf); err != nil { + return err + } + var b strings.Builder + b.WriteString("event: " + event + "\n") + for _, line := range strings.Split(buf.String(), "\n") { + b.WriteString("data: " + line + "\n") + } + b.WriteString("\n") + if _, err := s.w.Write([]byte(b.String())); err != nil { + return err + } + return s.rc.Flush() +} diff --git a/internal/web/server_test.go b/internal/web/server_test.go new file mode 100644 index 0000000..84e14fe --- /dev/null +++ b/internal/web/server_test.go @@ -0,0 +1,201 @@ +package web + +import ( + "encoding/json" + "fmt" + "io" + "log/slog" + "net/http" + "net/http/cookiejar" + "net/http/httptest" + "net/url" + "regexp" + "strings" + "testing" + "time" + + "git.b0b.be/bdeb/localchat/internal/config" +) + +// fakeLLM streams back "echo: " and records what it got. +func fakeLLM(t *testing.T, seen *[][]map[string]string) *httptest.Server { + return httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if r.URL.Path == "/v1/models" { + fmt.Fprint(w, `{"data":[{"id":"fake"}]}`) + return + } + var req struct { + Messages []map[string]string `json:"messages"` + } + json.NewDecoder(r.Body).Decode(&req) + *seen = append(*seen, req.Messages) + last := req.Messages[len(req.Messages)-1]["content"] + w.Header().Set("Content-Type", "text/event-stream") + for _, part := range []string{"echo: ", last, "\n\n**bold** "} { + b, _ := json.Marshal(part) + fmt.Fprintf(w, "data: {\"choices\":[{\"delta\":{\"content\":%s}}]}\n\n", b) + } + fmt.Fprint(w, "data: [DONE]\n\n") + })) +} + +func newTestApp(t *testing.T, llmURL string) (*httptest.Server, *http.Client) { + t.Helper() + cfg := config.Config{ + BaseURL: llmURL + "/v1", + SystemPrompt: "be nice", + Title: "test", + MaxHistory: 20, + Timeout: 10 * time.Second, + } + app := httptest.NewServer(New(cfg, slog.New(slog.NewTextHandler(io.Discard, nil))).Handler()) + t.Cleanup(app.Close) + jar, _ := cookiejar.New(nil) + return app, &http.Client{Jar: jar} +} + +var streamURL = regexp.MustCompile(`sse-connect="(/chat/stream/[0-9a-f]+)"`) + +func get(t *testing.T, c *http.Client, u string) string { + t.Helper() + resp, err := c.Get(u) + if err != nil { + t.Fatal(err) + } + defer resp.Body.Close() + b, _ := io.ReadAll(resp.Body) + return string(b) +} + +// ask posts a message, follows the SSE stream and returns the raw events. +func ask(t *testing.T, app *httptest.Server, c *http.Client, msg string) string { + t.Helper() + resp, err := c.PostForm(app.URL+"/chat", url.Values{"message": {msg}}) + if err != nil { + t.Fatal(err) + } + body, _ := io.ReadAll(resp.Body) + resp.Body.Close() + m := streamURL.FindStringSubmatch(string(body)) + if m == nil { + t.Fatalf("no stream url in %s", body) + } + return get(t, c, app.URL+m[1]) +} + +func TestChatFlow(t *testing.T) { + var seen [][]map[string]string + llm := fakeLLM(t, &seen) + defer llm.Close() + app, c := newTestApp(t, llm.URL) + + if page := get(t, c, app.URL+"/"); !strings.Contains(page, `id="composer"`) { + t.Fatal("index page missing composer") + } + if h := get(t, c, app.URL+"/health"); !strings.Contains(h, "dot ok") || !strings.Contains(h, "fake") { + t.Fatalf("health: %s", h) + } + + events := ask(t, app, c, "hi there") + for _, want := range []string{ + "event: token\ndata: echo: ", + "hi <b>there</b>", // user text is escaped in tokens + "event: done", + "bold", // final reply is rendered Markdown + "", // ... with raw HTML from the model dropped + } { + if !strings.Contains(events, want) { + t.Errorf("stream missing %q:\n%s", want, events) + } + } + if strings.Contains(events, " + + + + +
+

{ title }

+ @Health(HealthInfo{Checking: true}) +
+ +
+
+
+ @Messages(msgs) +
+
+ + +
+ + +} + +// Messages renders a whole conversation, or the empty state. +templ Messages(msgs []chat.Message) { + if len(msgs) == 0 { +
+

Hi! 👋

+

Everything you type stays on this machine. The model runs locally.

+
+ } + for _, m := range msgs { + if m.Role == "assistant" && m.State != chat.Done { + @StreamingReply(m) + } else { + @MessageView(m) + } + } +} + +// Exchange is returned after sending: the user's message plus a reply that +// streams in over server-sent events. +templ Exchange(user, reply chat.Message) { + @MessageView(user) + @StreamingReply(reply) +} + +// MessageView renders a finished message. +templ MessageView(m chat.Message) { +
+
+ if m.Role == "assistant" { +
+ @Markdown(m.Content) +
+ if m.Err != "" { +

⚠️ { m.Err }

+ } + } else { +
{ m.Content }
+ } +
+
+} + +// StreamingReply opens an SSE connection. "token" events append text as it +// arrives; the final "done" event swaps in the rendered Markdown and closes +// the connection. +templ StreamingReply(m chat.Message) { +
+
+
+
+
+} + +// Token is one streamed chunk of text. +templ Token(text string) { + { text } +} + +// HealthInfo describes the model server status shown in the header. +type HealthInfo struct { + Checking bool + OK bool + Model string + Err string +} + +templ Health(h HealthInfo) { +
+ if h.Checking { + connecting… + } else if h.OK { + { h.Model } + } else { + model offline + } +
+} diff --git a/internal/web/views/views_templ.go b/internal/web/views/views_templ.go new file mode 100644 index 0000000..ac43507 --- /dev/null +++ b/internal/web/views/views_templ.go @@ -0,0 +1,474 @@ +// Code generated by templ - DO NOT EDIT. + +// templ: version: v0.3.1070 +package views + +//lint:file-ignore SA4006 This context is only used if a nested component is present. + +import "github.com/a-h/templ" +import templruntime "github.com/a-h/templ/runtime" + +import "git.b0b.be/bdeb/localchat/internal/chat" + +// Page is the full chat screen. +func Page(title string, msgs []chat.Message) templ.Component { + return templruntime.GeneratedTemplate(func(templ_7745c5c3_Input templruntime.GeneratedComponentInput) (templ_7745c5c3_Err error) { + templ_7745c5c3_W, ctx := templ_7745c5c3_Input.Writer, templ_7745c5c3_Input.Context + if templ_7745c5c3_CtxErr := ctx.Err(); templ_7745c5c3_CtxErr != nil { + return templ_7745c5c3_CtxErr + } + templ_7745c5c3_Buffer, templ_7745c5c3_IsBuffer := templruntime.GetBuffer(templ_7745c5c3_W) + if !templ_7745c5c3_IsBuffer { + defer func() { + templ_7745c5c3_BufErr := templruntime.ReleaseBuffer(templ_7745c5c3_Buffer) + if templ_7745c5c3_Err == nil { + templ_7745c5c3_Err = templ_7745c5c3_BufErr + } + }() + } + ctx = templ.InitializeContext(ctx) + templ_7745c5c3_Var1 := templ.GetChildren(ctx) + if templ_7745c5c3_Var1 == nil { + templ_7745c5c3_Var1 = templ.NopComponent + } + ctx = templ.ClearChildren(ctx) + templ_7745c5c3_Err = templruntime.WriteString(templ_7745c5c3_Buffer, 1, "") + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + var templ_7745c5c3_Var2 string + templ_7745c5c3_Var2, templ_7745c5c3_Err = templ.JoinStringErrs(title) + if templ_7745c5c3_Err != nil { + return templ.Error{Err: templ_7745c5c3_Err, FileName: `internal/web/views/views.templ`, Line: 12, Col: 17} + } + _, templ_7745c5c3_Err = templ_7745c5c3_Buffer.WriteString(templ.EscapeString(templ_7745c5c3_Var2)) + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + templ_7745c5c3_Err = templruntime.WriteString(templ_7745c5c3_Buffer, 2, "

") + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + var templ_7745c5c3_Var3 string + templ_7745c5c3_Var3, templ_7745c5c3_Err = templ.JoinStringErrs(title) + if templ_7745c5c3_Err != nil { + return templ.Error{Err: templ_7745c5c3_Err, FileName: `internal/web/views/views.templ`, Line: 21, Col: 15} + } + _, templ_7745c5c3_Err = templ_7745c5c3_Buffer.WriteString(templ.EscapeString(templ_7745c5c3_Var3)) + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + templ_7745c5c3_Err = templruntime.WriteString(templ_7745c5c3_Buffer, 3, "

") + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + templ_7745c5c3_Err = Health(HealthInfo{Checking: true}).Render(ctx, templ_7745c5c3_Buffer) + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + templ_7745c5c3_Err = templruntime.WriteString(templ_7745c5c3_Buffer, 4, "
") + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + templ_7745c5c3_Err = Messages(msgs).Render(ctx, templ_7745c5c3_Buffer) + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + templ_7745c5c3_Err = templruntime.WriteString(templ_7745c5c3_Buffer, 5, "
") + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + return nil + }) +} + +// Messages renders a whole conversation, or the empty state. +func Messages(msgs []chat.Message) templ.Component { + return templruntime.GeneratedTemplate(func(templ_7745c5c3_Input templruntime.GeneratedComponentInput) (templ_7745c5c3_Err error) { + templ_7745c5c3_W, ctx := templ_7745c5c3_Input.Writer, templ_7745c5c3_Input.Context + if templ_7745c5c3_CtxErr := ctx.Err(); templ_7745c5c3_CtxErr != nil { + return templ_7745c5c3_CtxErr + } + templ_7745c5c3_Buffer, templ_7745c5c3_IsBuffer := templruntime.GetBuffer(templ_7745c5c3_W) + if !templ_7745c5c3_IsBuffer { + defer func() { + templ_7745c5c3_BufErr := templruntime.ReleaseBuffer(templ_7745c5c3_Buffer) + if templ_7745c5c3_Err == nil { + templ_7745c5c3_Err = templ_7745c5c3_BufErr + } + }() + } + ctx = templ.InitializeContext(ctx) + templ_7745c5c3_Var4 := templ.GetChildren(ctx) + if templ_7745c5c3_Var4 == nil { + templ_7745c5c3_Var4 = templ.NopComponent + } + ctx = templ.ClearChildren(ctx) + if len(msgs) == 0 { + templ_7745c5c3_Err = templruntime.WriteString(templ_7745c5c3_Buffer, 6, "

Hi! 👋

Everything you type stays on this machine. The model runs locally.

") + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + } + for _, m := range msgs { + if m.Role == "assistant" && m.State != chat.Done { + templ_7745c5c3_Err = StreamingReply(m).Render(ctx, templ_7745c5c3_Buffer) + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + } else { + templ_7745c5c3_Err = MessageView(m).Render(ctx, templ_7745c5c3_Buffer) + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + } + } + return nil + }) +} + +// Exchange is returned after sending: the user's message plus a reply that +// streams in over server-sent events. +func Exchange(user, reply chat.Message) templ.Component { + return templruntime.GeneratedTemplate(func(templ_7745c5c3_Input templruntime.GeneratedComponentInput) (templ_7745c5c3_Err error) { + templ_7745c5c3_W, ctx := templ_7745c5c3_Input.Writer, templ_7745c5c3_Input.Context + if templ_7745c5c3_CtxErr := ctx.Err(); templ_7745c5c3_CtxErr != nil { + return templ_7745c5c3_CtxErr + } + templ_7745c5c3_Buffer, templ_7745c5c3_IsBuffer := templruntime.GetBuffer(templ_7745c5c3_W) + if !templ_7745c5c3_IsBuffer { + defer func() { + templ_7745c5c3_BufErr := templruntime.ReleaseBuffer(templ_7745c5c3_Buffer) + if templ_7745c5c3_Err == nil { + templ_7745c5c3_Err = templ_7745c5c3_BufErr + } + }() + } + ctx = templ.InitializeContext(ctx) + templ_7745c5c3_Var5 := templ.GetChildren(ctx) + if templ_7745c5c3_Var5 == nil { + templ_7745c5c3_Var5 = templ.NopComponent + } + ctx = templ.ClearChildren(ctx) + templ_7745c5c3_Err = MessageView(user).Render(ctx, templ_7745c5c3_Buffer) + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + templ_7745c5c3_Err = StreamingReply(reply).Render(ctx, templ_7745c5c3_Buffer) + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + return nil + }) +} + +// MessageView renders a finished message. +func MessageView(m chat.Message) templ.Component { + return templruntime.GeneratedTemplate(func(templ_7745c5c3_Input templruntime.GeneratedComponentInput) (templ_7745c5c3_Err error) { + templ_7745c5c3_W, ctx := templ_7745c5c3_Input.Writer, templ_7745c5c3_Input.Context + if templ_7745c5c3_CtxErr := ctx.Err(); templ_7745c5c3_CtxErr != nil { + return templ_7745c5c3_CtxErr + } + templ_7745c5c3_Buffer, templ_7745c5c3_IsBuffer := templruntime.GetBuffer(templ_7745c5c3_W) + if !templ_7745c5c3_IsBuffer { + defer func() { + templ_7745c5c3_BufErr := templruntime.ReleaseBuffer(templ_7745c5c3_Buffer) + if templ_7745c5c3_Err == nil { + templ_7745c5c3_Err = templ_7745c5c3_BufErr + } + }() + } + ctx = templ.InitializeContext(ctx) + templ_7745c5c3_Var6 := templ.GetChildren(ctx) + if templ_7745c5c3_Var6 == nil { + templ_7745c5c3_Var6 = templ.NopComponent + } + ctx = templ.ClearChildren(ctx) + var templ_7745c5c3_Var7 = []any{"msg", m.Role} + templ_7745c5c3_Err = templ.RenderCSSItems(ctx, templ_7745c5c3_Buffer, templ_7745c5c3_Var7...) + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + templ_7745c5c3_Err = templruntime.WriteString(templ_7745c5c3_Buffer, 7, "
") + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + if m.Role == "assistant" { + templ_7745c5c3_Err = templruntime.WriteString(templ_7745c5c3_Buffer, 10, "
") + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + templ_7745c5c3_Err = Markdown(m.Content).Render(ctx, templ_7745c5c3_Buffer) + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + templ_7745c5c3_Err = templruntime.WriteString(templ_7745c5c3_Buffer, 11, "
") + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + if m.Err != "" { + templ_7745c5c3_Err = templruntime.WriteString(templ_7745c5c3_Buffer, 12, "

⚠️ ") + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + var templ_7745c5c3_Var10 string + templ_7745c5c3_Var10, templ_7745c5c3_Err = templ.JoinStringErrs(m.Err) + if templ_7745c5c3_Err != nil { + return templ.Error{Err: templ_7745c5c3_Err, FileName: `internal/web/views/views.templ`, Line: 71, Col: 36} + } + _, templ_7745c5c3_Err = templ_7745c5c3_Buffer.WriteString(templ.EscapeString(templ_7745c5c3_Var10)) + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + templ_7745c5c3_Err = templruntime.WriteString(templ_7745c5c3_Buffer, 13, "

") + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + } + } else { + templ_7745c5c3_Err = templruntime.WriteString(templ_7745c5c3_Buffer, 14, "
") + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + var templ_7745c5c3_Var11 string + templ_7745c5c3_Var11, templ_7745c5c3_Err = templ.JoinStringErrs(m.Content) + if templ_7745c5c3_Err != nil { + return templ.Error{Err: templ_7745c5c3_Err, FileName: `internal/web/views/views.templ`, Line: 74, Col: 42} + } + _, templ_7745c5c3_Err = templ_7745c5c3_Buffer.WriteString(templ.EscapeString(templ_7745c5c3_Var11)) + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + templ_7745c5c3_Err = templruntime.WriteString(templ_7745c5c3_Buffer, 15, "
") + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + } + templ_7745c5c3_Err = templruntime.WriteString(templ_7745c5c3_Buffer, 16, "
") + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + return nil + }) +} + +// StreamingReply opens an SSE connection. "token" events append text as it +// arrives; the final "done" event swaps in the rendered Markdown and closes +// the connection. +func StreamingReply(m chat.Message) templ.Component { + return templruntime.GeneratedTemplate(func(templ_7745c5c3_Input templruntime.GeneratedComponentInput) (templ_7745c5c3_Err error) { + templ_7745c5c3_W, ctx := templ_7745c5c3_Input.Writer, templ_7745c5c3_Input.Context + if templ_7745c5c3_CtxErr := ctx.Err(); templ_7745c5c3_CtxErr != nil { + return templ_7745c5c3_CtxErr + } + templ_7745c5c3_Buffer, templ_7745c5c3_IsBuffer := templruntime.GetBuffer(templ_7745c5c3_W) + if !templ_7745c5c3_IsBuffer { + defer func() { + templ_7745c5c3_BufErr := templruntime.ReleaseBuffer(templ_7745c5c3_Buffer) + if templ_7745c5c3_Err == nil { + templ_7745c5c3_Err = templ_7745c5c3_BufErr + } + }() + } + ctx = templ.InitializeContext(ctx) + templ_7745c5c3_Var12 := templ.GetChildren(ctx) + if templ_7745c5c3_Var12 == nil { + templ_7745c5c3_Var12 = templ.NopComponent + } + ctx = templ.ClearChildren(ctx) + templ_7745c5c3_Err = templruntime.WriteString(templ_7745c5c3_Buffer, 17, "
") + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + return nil + }) +} + +// Token is one streamed chunk of text. +func Token(text string) templ.Component { + return templruntime.GeneratedTemplate(func(templ_7745c5c3_Input templruntime.GeneratedComponentInput) (templ_7745c5c3_Err error) { + templ_7745c5c3_W, ctx := templ_7745c5c3_Input.Writer, templ_7745c5c3_Input.Context + if templ_7745c5c3_CtxErr := ctx.Err(); templ_7745c5c3_CtxErr != nil { + return templ_7745c5c3_CtxErr + } + templ_7745c5c3_Buffer, templ_7745c5c3_IsBuffer := templruntime.GetBuffer(templ_7745c5c3_W) + if !templ_7745c5c3_IsBuffer { + defer func() { + templ_7745c5c3_BufErr := templruntime.ReleaseBuffer(templ_7745c5c3_Buffer) + if templ_7745c5c3_Err == nil { + templ_7745c5c3_Err = templ_7745c5c3_BufErr + } + }() + } + ctx = templ.InitializeContext(ctx) + templ_7745c5c3_Var15 := templ.GetChildren(ctx) + if templ_7745c5c3_Var15 == nil { + templ_7745c5c3_Var15 = templ.NopComponent + } + ctx = templ.ClearChildren(ctx) + templ_7745c5c3_Err = templruntime.WriteString(templ_7745c5c3_Buffer, 20, "") + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + var templ_7745c5c3_Var16 string + templ_7745c5c3_Var16, templ_7745c5c3_Err = templ.JoinStringErrs(text) + if templ_7745c5c3_Err != nil { + return templ.Error{Err: templ_7745c5c3_Err, FileName: `internal/web/views/views.templ`, Line: 101, Col: 13} + } + _, templ_7745c5c3_Err = templ_7745c5c3_Buffer.WriteString(templ.EscapeString(templ_7745c5c3_Var16)) + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + templ_7745c5c3_Err = templruntime.WriteString(templ_7745c5c3_Buffer, 21, "") + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + return nil + }) +} + +// HealthInfo describes the model server status shown in the header. +type HealthInfo struct { + Checking bool + OK bool + Model string + Err string +} + +func Health(h HealthInfo) templ.Component { + return templruntime.GeneratedTemplate(func(templ_7745c5c3_Input templruntime.GeneratedComponentInput) (templ_7745c5c3_Err error) { + templ_7745c5c3_W, ctx := templ_7745c5c3_Input.Writer, templ_7745c5c3_Input.Context + if templ_7745c5c3_CtxErr := ctx.Err(); templ_7745c5c3_CtxErr != nil { + return templ_7745c5c3_CtxErr + } + templ_7745c5c3_Buffer, templ_7745c5c3_IsBuffer := templruntime.GetBuffer(templ_7745c5c3_W) + if !templ_7745c5c3_IsBuffer { + defer func() { + templ_7745c5c3_BufErr := templruntime.ReleaseBuffer(templ_7745c5c3_Buffer) + if templ_7745c5c3_Err == nil { + templ_7745c5c3_Err = templ_7745c5c3_BufErr + } + }() + } + ctx = templ.InitializeContext(ctx) + templ_7745c5c3_Var17 := templ.GetChildren(ctx) + if templ_7745c5c3_Var17 == nil { + templ_7745c5c3_Var17 = templ.NopComponent + } + ctx = templ.ClearChildren(ctx) + templ_7745c5c3_Err = templruntime.WriteString(templ_7745c5c3_Buffer, 22, "
") + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + if h.Checking { + templ_7745c5c3_Err = templruntime.WriteString(templ_7745c5c3_Buffer, 24, " connecting…") + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + } else if h.OK { + templ_7745c5c3_Err = templruntime.WriteString(templ_7745c5c3_Buffer, 25, " ") + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + var templ_7745c5c3_Var19 string + templ_7745c5c3_Var19, templ_7745c5c3_Err = templ.JoinStringErrs(h.Model) + if templ_7745c5c3_Err != nil { + return templ.Error{Err: templ_7745c5c3_Err, FileName: `internal/web/views/views.templ`, Line: 117, Col: 41} + } + _, templ_7745c5c3_Err = templ_7745c5c3_Buffer.WriteString(templ.EscapeString(templ_7745c5c3_Var19)) + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + } else { + templ_7745c5c3_Err = templruntime.WriteString(templ_7745c5c3_Buffer, 26, " model offline") + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + } + templ_7745c5c3_Err = templruntime.WriteString(templ_7745c5c3_Buffer, 28, "
") + if templ_7745c5c3_Err != nil { + return templ_7745c5c3_Err + } + return nil + }) +} + +var _ = templruntime.GeneratedTemplate diff --git a/scripts/e2e.py b/scripts/e2e.py new file mode 100644 index 0000000..ca781f7 --- /dev/null +++ b/scripts/e2e.py @@ -0,0 +1,70 @@ +# /// script +# requires-python = ">=3.11" +# dependencies = ["playwright"] +# /// +"""Browser end-to-end test: loads the UI, sends a message with Enter, waits +for the streamed reply to finish rendering, and saves screenshots. + + uv run scripts/e2e.py [--url http://127.0.0.1:3000] [--channel msedge] +""" +import argparse, sys, time +from playwright.sync_api import sync_playwright, expect + +p = argparse.ArgumentParser() +p.add_argument("--url", default="http://127.0.0.1:3000") +p.add_argument("--channel", default="msedge", help="installed browser: chrome, msedge, ...") +p.add_argument("--out", default="docs") +p.add_argument("--dark", action="store_true") +p.add_argument("--prompt", default="Give me 3 short tips for learning Go, as a numbered list. Include one tiny code snippet.") +args = p.parse_args() + +with sync_playwright() as pw: + browser = pw.chromium.launch(channel=args.channel, headless=True) + page = browser.new_page(viewport={"width": 900, "height": 760}, + color_scheme="dark" if args.dark else "light") + errors = [] + page.on("pageerror", lambda e: errors.append(str(e))) + page.goto(args.url) + + expect(page.locator("#health .dot.ok")).to_be_visible(timeout=10_000) + expect(page.locator(".empty")).to_be_visible() + + box = page.locator("#composer textarea") + box.fill(args.prompt) + t0 = time.time() + box.press("Enter") + + # user bubble shows up and the input clears + expect(page.locator(".msg.user")).to_have_count(1) + expect(box).to_have_value("") + expect(page.locator(".empty")).to_be_hidden() + + # tokens stream in (plain spans) ... + page.wait_for_selector(".streaming span", timeout=60_000) + first = time.time() - t0 + # ... then get replaced by rendered Markdown + page.wait_for_selector(".msg.assistant .markdown", timeout=180_000) + total = time.time() - t0 + page.wait_for_timeout(300) + assert page.locator(".msg.assistant [sse-connect]").count() == 0, "SSE element not cleaned up" + if "numbered list" in args.prompt: + assert page.locator(".msg.assistant ol li").count() >= 3, "expected a numbered list" + if "code" in args.prompt: + assert page.locator(".msg.assistant pre code").count() >= 1, "expected a code block" + page.screenshot(path=f"{args.out}/screenshot{'-dark' if args.dark else ''}.png") + + # reload: history persists for the session + page.reload() + expect(page.locator(".msg")).to_have_count(2) + + # new chat clears it + page.get_by_role("button", name="New chat").click() + expect(page.locator(".msg")).to_have_count(0) + expect(page.locator(".empty")).to_be_visible() + + browser.close() + +if errors: + print("JS errors:", errors) + sys.exit(1) +print(f"OK: first token {first:.1f}s, full reply {total:.1f}s")