# localchat + a local Gemma 4 E2B model served by llama.cpp. # # docker compose up -d # first start downloads the model (~3 GB) # open http://localhost:3000 # # Swap the model with LLM_HF_REPO, e.g. LLM_HF_REPO=unsloth/gemma-4-E4B-it-GGUF:Q4_K_M services: llm: image: ghcr.io/ggml-org/llama.cpp:server command: - -hf - ${LLM_HF_REPO:-unsloth/gemma-4-E2B-it-GGUF:Q4_K_M} - --no-mmproj # text-only chat; skip the vision projector download - --ctx-size - "8192" - --port - "8080" environment: LLAMA_CACHE: /models volumes: - models:/models healthcheck: test: ["CMD", "curl", "-fs", "http://localhost:8080/health"] interval: 10s timeout: 5s start_period: 15m # first boot downloads the model restart: unless-stopped app: build: . image: localchat:latest ports: - "3000:3000" environment: LLM_BASE_URL: http://llm:8080/v1 LOCALCHAT_TITLE: ${LOCALCHAT_TITLE:-localchat} SYSTEM_PROMPT: ${SYSTEM_PROMPT:-} depends_on: llm: condition: service_healthy restart: unless-stopped volumes: models: