Streams replies over SSE from any OpenAI-compatible server (oMLX, llama.cpp, Ollama, ...). Single binary that can install itself as an OS service; docker compose bundles llama.cpp + Gemma 4 E2B.
45 lines
1.1 KiB
YAML
45 lines
1.1 KiB
YAML
# localchat + a local Gemma 4 E2B model served by llama.cpp.
|
|
#
|
|
# docker compose up -d # first start downloads the model (~3 GB)
|
|
# open http://localhost:3000
|
|
#
|
|
# Swap the model with LLM_HF_REPO, e.g. LLM_HF_REPO=unsloth/gemma-4-E4B-it-GGUF:Q4_K_M
|
|
services:
|
|
llm:
|
|
image: ghcr.io/ggml-org/llama.cpp:server
|
|
command:
|
|
- -hf
|
|
- ${LLM_HF_REPO:-unsloth/gemma-4-E2B-it-GGUF:Q4_K_M}
|
|
- --no-mmproj # text-only chat; skip the vision projector download
|
|
- --ctx-size
|
|
- "8192"
|
|
- --port
|
|
- "8080"
|
|
environment:
|
|
LLAMA_CACHE: /models
|
|
volumes:
|
|
- models:/models
|
|
healthcheck:
|
|
test: ["CMD", "curl", "-fs", "http://localhost:8080/health"]
|
|
interval: 10s
|
|
timeout: 5s
|
|
start_period: 15m # first boot downloads the model
|
|
restart: unless-stopped
|
|
|
|
app:
|
|
build: .
|
|
image: localchat:latest
|
|
ports:
|
|
- "3000:3000"
|
|
environment:
|
|
LLM_BASE_URL: http://llm:8080/v1
|
|
LOCALCHAT_TITLE: ${LOCALCHAT_TITLE:-localchat}
|
|
SYSTEM_PROMPT: ${SYSTEM_PROMPT:-}
|
|
depends_on:
|
|
llm:
|
|
condition: service_healthy
|
|
restart: unless-stopped
|
|
|
|
volumes:
|
|
models:
|