From 717a57ff433c5617b5f1ebe0e90f696f7e16f037 Mon Sep 17 00:00:00 2001 From: sebaplaza Date: Fri, 5 Jun 2026 12:12:43 +0200 Subject: [PATCH] feat(docker): add container image for MCP server and CLI Add a multi-stage, uv-based Dockerfile producing a CPU image (with the extract + spellcheck extras), plus a CUDA variant (Dockerfile.gpu) for onnxruntime-gpu accelerated embeddings. A single flexible entrypoint dispatches to the MCP stdio server (default) or the mempalace CLI. All state -- palace, config, and the lazily downloaded embedding model -- persists under /data via HOME, runs as a non-root user, and is exposed as a volume. Also add a docker-compose.yml for convenience, a GHCR publish workflow, and a Docker section in the README. --- .dockerignore | 55 +++++++++++++++ .github/workflows/docker-publish.yml | 61 ++++++++++++++++ Dockerfile | 100 +++++++++++++++++++++++++++ Dockerfile.gpu | 77 +++++++++++++++++++++ README.md | 37 ++++++++++ docker-compose.yml | 30 ++++++++ docker-entrypoint.sh | 26 +++++++ 7 files changed, 386 insertions(+) create mode 100644 .dockerignore create mode 100644 .github/workflows/docker-publish.yml create mode 100644 Dockerfile create mode 100644 Dockerfile.gpu create mode 100644 docker-compose.yml create mode 100644 docker-entrypoint.sh diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 0000000..544641e --- /dev/null +++ b/.dockerignore @@ -0,0 +1,55 @@ +# Keep the build context lean — only what `uv sync` + the package need. + +# VCS +.git/ +.gitignore + +# Python build artifacts / caches +*.egg-info/ +dist/ +build/ +__pycache__/ +*.pyc +.pytest_cache/ +.mypy_cache/ +.ruff_cache/ +htmlcov/ +.coverage +coverage.xml + +# Virtual environments (rebuilt inside the image) +.venv/ +venv/ + +# Local config / secrets +.env +.env.* +.envrc +mempal.yaml + +# Editor / OS cruft +.idea/ +.vscode/ +*.swp +*.swo +*~ +.DS_Store +Thumbs.db + +# Repo material not needed at runtime +.devcontainer/ +.github/ +.agents/ +.claude/ +.codex/ +.codex-plugin/ +.claude-plugin/ +benchmarks/ +docs/ +website/ +landing/ +assets/ +examples/ +tests/ +*.md +!README.md diff --git a/.github/workflows/docker-publish.yml b/.github/workflows/docker-publish.yml new file mode 100644 index 0000000..8526801 --- /dev/null +++ b/.github/workflows/docker-publish.yml @@ -0,0 +1,61 @@ +name: Docker + +on: + push: + branches: [main] + tags: ["v*"] + pull_request: + branches: [main, develop] + workflow_dispatch: + +env: + REGISTRY: ghcr.io + IMAGE_NAME: ${{ github.repository }} + +jobs: + build: + runs-on: ubuntu-latest + permissions: + contents: read + packages: write + steps: + - uses: actions/checkout@v6 + + - name: Set up Docker Buildx + uses: docker/setup-buildx-action@v3 + + # Only authenticate + push for in-repo events. Fork PRs lack the + # packages:write token, so they build (to validate the Dockerfile) but + # do not push. + - name: Log in to GHCR + if: github.event_name != 'pull_request' + uses: docker/login-action@v3 + with: + registry: ${{ env.REGISTRY }} + username: ${{ github.actor }} + password: ${{ secrets.GITHUB_TOKEN }} + + - name: Extract metadata + id: meta + uses: docker/metadata-action@v5 + with: + images: ${{ env.REGISTRY }}/${{ env.IMAGE_NAME }} + # latest -> default branch; semver tags -> released versions. + tags: | + type=ref,event=branch + type=ref,event=pr + type=semver,pattern={{version}} + type=semver,pattern={{major}}.{{minor}} + type=raw,value=latest,enable={{is_default_branch}} + + - name: Build and push + uses: docker/build-push-action@v6 + with: + context: . + file: ./Dockerfile + platforms: linux/amd64 + push: ${{ github.event_name != 'pull_request' }} + tags: ${{ steps.meta.outputs.tags }} + labels: ${{ steps.meta.outputs.labels }} + cache-from: type=gha + cache-to: type=gha,mode=max diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000..7e61bbb --- /dev/null +++ b/Dockerfile @@ -0,0 +1,100 @@ +# syntax=docker/dockerfile:1.7 + +# MemPalace — CPU image. +# +# Multi-stage build using uv (the project ships a uv.lock, so we install from +# the frozen lockfile for reproducible images). The default runtime is the MCP +# server over stdio; the CLI is reachable through the same entrypoint. +# +# Build: +# docker build -t mempalace . +# docker build -t mempalace --build-arg EXTRAS="extract,spellcheck" . +# +# Run (MCP server over stdio, palace persisted on the host): +# docker run -i --rm -v mempalace-data:/data mempalace +# +# Run (CLI): +# docker run --rm -v mempalace-data:/data mempalace search "why GraphQL" +# +# GPU acceleration lives in Dockerfile.gpu (it needs a CUDA base image). + +ARG PYTHON_VERSION=3.12 + +# --- builder ---------------------------------------------------------------- +FROM python:${PYTHON_VERSION}-slim AS builder + +# uv: fast, lockfile-driven installer. Pinned by digest-less tag for clarity; +# bump deliberately. +COPY --from=ghcr.io/astral-sh/uv:0.5 /uv /uvx /bin/ + +# Some transitive deps (grpcio, onnxruntime, tokenizers) ship manylinux wheels +# for cp312, but keep a compiler around so a missing wheel degrades to a source +# build instead of failing the image. Dropped from the final stage. +RUN apt-get update \ + && apt-get install -y --no-install-recommends build-essential \ + && rm -rf /var/lib/apt/lists/* + +ENV UV_COMPILE_BYTECODE=1 \ + UV_LINK_MODE=copy \ + UV_PYTHON_DOWNLOADS=never + +WORKDIR /app + +# Optional extras baked into the image. CPU-safe by default; GPU is a separate +# Dockerfile. Pass a comma-separated list, e.g. EXTRAS="extract,spellcheck". +ARG EXTRAS="extract,spellcheck" + +# Layer 1: dependencies only (no project) — cached across source changes. +RUN --mount=type=cache,target=/root/.cache/uv \ + --mount=type=bind,source=pyproject.toml,target=pyproject.toml \ + --mount=type=bind,source=uv.lock,target=uv.lock \ + --mount=type=bind,source=README.md,target=README.md \ + set -e; \ + flags=""; \ + for e in $(echo "${EXTRAS}" | tr ',' ' '); do flags="${flags} --extra ${e}"; done; \ + uv sync --frozen --no-install-project --no-dev ${flags} + +# Layer 2: the project itself. --no-editable installs mempalace into the venv's +# site-packages (instead of an .pth pointing at /app), so the runtime stage can +# copy only /app/.venv and drop the source tree. +COPY . /app +RUN --mount=type=cache,target=/root/.cache/uv \ + set -e; \ + flags=""; \ + for e in $(echo "${EXTRAS}" | tr ',' ' '); do flags="${flags} --extra ${e}"; done; \ + uv sync --frozen --no-dev --no-editable ${flags} + +# --- runtime ---------------------------------------------------------------- +FROM python:${PYTHON_VERSION}-slim AS runtime + +LABEL org.opencontainers.image.title="MemPalace" \ + org.opencontainers.image.description="Local-first AI memory — verbatim storage, MCP server + CLI." \ + org.opencontainers.image.source="https://github.com/MemPalace/mempalace" \ + org.opencontainers.image.licenses="MIT" + +# /data is the single persistence root: HOME points here, so the palace +# (~/.mempalace/palace), config (~/.mempalace), and the HuggingFace model +# cache (~/.cache/huggingface, ~300 MB, lazy-downloaded on first use) all +# land under one mountable volume. +ENV HOME=/data \ + PATH="/app/.venv/bin:${PATH}" \ + PYTHONUNBUFFERED=1 \ + PYTHONDONTWRITEBYTECODE=1 + +# Non-root user owning the data volume. +RUN groupadd --gid 1000 mempalace \ + && useradd --uid 1000 --gid 1000 --home-dir /data --create-home mempalace + +WORKDIR /app + +# The resolved virtualenv from the builder — no build toolchain in this layer. +COPY --from=builder --chown=mempalace:mempalace /app/.venv /app/.venv +COPY --chown=mempalace:mempalace docker-entrypoint.sh /usr/local/bin/docker-entrypoint.sh +RUN chmod +x /usr/local/bin/docker-entrypoint.sh + +USER mempalace +VOLUME ["/data"] + +# Default to the MCP server; `docker run` it with `-i` for stdio JSON-RPC. +ENTRYPOINT ["docker-entrypoint.sh"] +CMD ["mcp"] diff --git a/Dockerfile.gpu b/Dockerfile.gpu new file mode 100644 index 0000000..2c59a74 --- /dev/null +++ b/Dockerfile.gpu @@ -0,0 +1,77 @@ +# syntax=docker/dockerfile:1.7 + +# MemPalace — GPU (NVIDIA CUDA) image. +# +# The `gpu` extra pulls onnxruntime-gpu, which needs CUDA + cuDNN shared +# libraries at runtime, so this variant builds on an nvidia/cuda base instead +# of python:slim. It is single-stage on purpose: the venv is created with a +# uv-managed interpreter, and copying that interpreter cleanly across build +# stages is more trouble than the size saving is worth for a GPU image. +# +# Build: +# docker build -f Dockerfile.gpu -t mempalace:gpu . +# +# Run (requires the NVIDIA Container Toolkit on the host): +# docker run -i --rm --gpus all \ +# -e MEMPALACE_EMBEDDING_DEVICE=cuda \ +# -v mempalace-data:/data mempalace:gpu +# +# NOTE: onnxruntime-gpu ties itself to a CUDA major version. If embeddings +# fail to load on the GPU, align CUDA_IMAGE below with the CUDA release that +# the resolved onnxruntime-gpu wheel targets (see its release notes), then +# rebuild. + +ARG CUDA_IMAGE=nvidia/cuda:12.6.3-cudnn-runtime-ubuntu22.04 +FROM ${CUDA_IMAGE} + +LABEL org.opencontainers.image.title="MemPalace (GPU)" \ + org.opencontainers.image.description="Local-first AI memory with CUDA-accelerated embeddings." \ + org.opencontainers.image.source="https://github.com/MemPalace/mempalace" \ + org.opencontainers.image.licenses="MIT" + +COPY --from=ghcr.io/astral-sh/uv:0.5 /uv /uvx /bin/ + +# Minimal toolchain for any source-built wheels; ca-certificates for the +# lazy HuggingFace model download on first use. +RUN apt-get update \ + && apt-get install -y --no-install-recommends build-essential ca-certificates \ + && rm -rf /var/lib/apt/lists/* + +ARG PYTHON_VERSION=3.12 +ARG EXTRAS="extract,spellcheck,gpu" + +ENV HOME=/data \ + PATH="/app/.venv/bin:${PATH}" \ + UV_COMPILE_BYTECODE=1 \ + UV_LINK_MODE=copy \ + UV_PYTHON_INSTALL_DIR=/opt/uv/python \ + UV_PYTHON_PREFERENCE=only-managed \ + PYTHONUNBUFFERED=1 \ + PYTHONDONTWRITEBYTECODE=1 \ + MEMPALACE_EMBEDDING_DEVICE=cuda + +WORKDIR /app + +# uv-managed interpreter pinned to PYTHON_VERSION, installed to a stable path +# so the venv stays valid at runtime. +RUN uv python install ${PYTHON_VERSION} + +COPY . /app +RUN --mount=type=cache,target=/root/.cache/uv \ + set -e; \ + flags=""; \ + for e in $(echo "${EXTRAS}" | tr ',' ' '); do flags="${flags} --extra ${e}"; done; \ + uv sync --frozen --no-dev --no-editable --python ${PYTHON_VERSION} ${flags} + +RUN groupadd --gid 1000 mempalace \ + && useradd --uid 1000 --gid 1000 --home-dir /data --create-home mempalace \ + && chown -R mempalace:mempalace /app /data + +COPY --chown=mempalace:mempalace docker-entrypoint.sh /usr/local/bin/docker-entrypoint.sh +RUN chmod +x /usr/local/bin/docker-entrypoint.sh + +USER mempalace +VOLUME ["/data"] + +ENTRYPOINT ["docker-entrypoint.sh"] +CMD ["mcp"] diff --git a/README.md b/README.md index 48208e4..ad8efbb 100644 --- a/README.md +++ b/README.md @@ -69,6 +69,43 @@ python -m venv .venv && source .venv/bin/activate pip install mempalace ``` +### Docker + +A container image is also available for running the MCP server or the CLI +without a local Python toolchain. Everything persists under `/data` (palace, +config, and the cached embedding model), so mount a volume there. + +```bash +# Build the image (CPU; bundles the `extract` + `spellcheck` extras) +docker build -t mempalace . + +# MCP server over stdio — note the `-i` flag (JSON-RPC needs stdin) +docker run -i --rm -v mempalace-data:/data mempalace + +# Run any CLI command instead +docker run --rm -v mempalace-data:/data mempalace mine /work +docker run --rm -v mempalace-data:/data mempalace search "why GraphQL" +``` + +Wire it into an MCP client (e.g. Claude Code) as a stdio server: + +```json +{ + "mcpServers": { + "mempalace": { + "command": "docker", + "args": ["run", "-i", "--rm", "-v", "mempalace-data:/data", "mempalace"] + } + } +} +``` + +`docker compose run --rm mcp` works too (see `docker-compose.yml`). For +CUDA-accelerated embeddings, build the GPU variant with +`docker build -f Dockerfile.gpu -t mempalace:gpu .` and run it with +`--gpus all`. Customise the bundled extras at build time, e.g. +`docker build --build-arg EXTRAS="extract,spellcheck" -t mempalace .`. + ## Quickstart ```bash diff --git a/docker-compose.yml b/docker-compose.yml new file mode 100644 index 0000000..cb6887f --- /dev/null +++ b/docker-compose.yml @@ -0,0 +1,30 @@ +# MemPalace via Docker Compose. +# +# The MCP server speaks JSON-RPC over stdio, so the `mcp` service is meant to +# be driven interactively (`docker compose run`), not left running detached: +# +# docker compose build +# docker compose run --rm mcp # MCP server over stdio +# docker compose run --rm mcp cli search "GraphQL" # one-off CLI command +# +# The named volume `mempalace-data` is mounted at /data and holds the palace, +# config, and the cached embedding model across runs. + +services: + mcp: + build: + context: . + dockerfile: Dockerfile + image: mempalace:local + # stdio transport: keep STDIN open and allocate no TTY (raw JSON-RPC). + stdin_open: true + tty: false + volumes: + - mempalace-data:/data + environment: + # Override the palace location, embedding model, etc. here if needed. + # - MEMPALACE_EMBEDDING_MODEL=minilm + MEMPALACE_PALACE_PATH: /data/.mempalace/palace + +volumes: + mempalace-data: diff --git a/docker-entrypoint.sh b/docker-entrypoint.sh new file mode 100644 index 0000000..79102fd --- /dev/null +++ b/docker-entrypoint.sh @@ -0,0 +1,26 @@ +#!/usr/bin/env sh +# Flexible entrypoint: pick the MCP server or the CLI from the first argument. +# +# docker run -i mempalace -> MCP server over stdio (default) +# docker run -i mempalace mcp -> MCP server over stdio (explicit) +# docker run mempalace cli search "query" -> CLI passthrough (explicit) +# docker run mempalace search "query" -> CLI passthrough (implicit) +# +# `mcp` and `cli` are dispatch keywords; anything else is forwarded to the +# `mempalace` CLI verbatim so subcommands like `mine`, `search`, `wake-up` +# work without ceremony. +set -e + +case "${1:-mcp}" in + mcp) + shift 2>/dev/null || true + exec mempalace-mcp "$@" + ;; + cli) + shift + exec mempalace "$@" + ;; + *) + exec mempalace "$@" + ;; +esac