# openplate + openplate-inference: your own AI, on your own hardware. # # mkdir -p ~/openplate && cd ~/openplate # curl -O https://raw.githubusercontent.com/LowCarbCheck/openplate/main/docker/topologies/compose.inference.yml # # # One key, used twice: the inference service accepts it, and the app # # hands it to every browser. Generate it here, on this machine. # echo "INFERENCE_API_KEY=opk_$(openssl rand -hex 24)" >> .env # # # The two URLs a BROWSER will use. Replace 192.168.1.20 with this # # machine's address, or with the names your reverse proxy serves. # echo "PUBLIC_APP_URL=http://192.168.1.20:3000" >> .env # echo "PUBLIC_INFERENCE_URL=http://192.168.1.20:8300/v1" >> .env # # docker compose -f compose.inference.yml up -d # # Keep this file in a folder that lasts, like ~/openplate above, and run all of # that from there. Compose treats the compose file's OWN directory as the # project directory, so a `.env` beside this file is the one it reads. # # The first start downloads about 2 GiB of weights before the first scan can # run. `logs -f inference` shows the progress; http://:8300/readyz # answers 200 once the model is loaded. # # Plate photos work over plain http://. Installing the app needs a secure # page: https://, or http://localhost on this machine. See the HTTPS section # of apps/app/docs/self-hosting.md. Once the app is on https://, the inference URL must # be https:// too, or the browser blocks the call from the secure page. # # docker compose -f compose.inference.yml logs -f inference # weight download + model load + ready # docker compose -f compose.inference.yml logs -f openplate # # What you get: an openplate instance whose users tap ONE button to use this # instance's own AI -- no provider account, no API key of their own, no photo # leaving your network. Both images are published to GHCR; nothing here builds # from source. There is still no database and no secret on the app side. # # "README" below always means openplate-inference's README: # https://github.com/LowCarbCheck/openplate/tree/main/apps/inference # # -- THE ONE THING PEOPLE GET WRONG ---------------------------------------- # `DEFAULT_INFERENCE_BASE_URL` must be a URL a BROWSER on the user's phone or # laptop can open. The photo goes from the device straight to the inference # endpoint -- openplate's server is never in the loop, which is what keeps the # scan private. So `http://inference:8300/v1` (the container hostname) does NOT # work, even though the two containers can talk to each other that way. Use the # LAN address of this host, or a hostname on your reverse proxy / tailnet. # Fixes the project name, so containers and the inference-models volume are # named after the stack rather than after whatever directory the file sits in. name: openplate-with-inference services: # -- The inference service ------------------------------------------------- inference: image: ghcr.io/lowcarbcheck/openplate-inference:latest restart: unless-stopped ports: # Published because the browser calls it directly. Bind to 0.0.0.0 (as # here) for LAN access; use "127.0.0.1:8300:8300" if a reverse proxy on # this host is the only thing that should reach it. - '${INFERENCE_PORT:-8300}:8300' volumes: # Weights land here on first boot (~2.0 GiB for lite, ~5.8 GiB for # quality) and are verified by sha256 on every start. Named, so # `docker compose down` does not throw the download away. - inference-models:/models environment: # lite | lite-apache | quality, see README "Hardware & measured latency". # external runs no model here and uses your runtime, see below. MODEL_PROFILE: ${MODEL_PROFILE:-lite} # The key callers must present. Set INFERENCE_API_KEY in .env (see the # top of this file); the placeholder below only keeps a first trial # booting. It must match DEFAULT_INFERENCE_API_KEY on the app. API_KEYS: ${INFERENCE_API_KEY:-opk_CHANGE_ME} # Self-host has no latency ceiling. 0 = never shed load for being slow. LATENCY_CEILING_MS: ${LATENCY_CEILING_MS:-0} # Scans in flight at once. It also sets llama.cpp's slot count; it does # NOT add CPU threads, the slots share LLAMA_THREADS. CONCURRENCY: ${CONCURRENCY:-2} # CPU threads for the model. Empty means every core but two (nproc - 2), # which leaves room for the service and the OS. LLAMA_THREADS: ${LLAMA_THREADS:-} # Where macros come from. fdc = the bundled offline USDA-derived dataset. # See README "Food data" before switching to `off` (ODbL) or `lcc`. FOOD_SOURCE: ${FOOD_SOURCE:-fdc} # -- Everything else the service reads. Empty means the default, and # openplate-inference's .env.example explains each one. # # Your own runtime, with MODEL_PROFILE=external (see below). No trailing # /v1, and the address must resolve from INSIDE this container. MODEL_RUNTIME_URL: ${MODEL_RUNTIME_URL:-} MODEL_RUNTIME_API_KEY: ${MODEL_RUNTIME_API_KEY:-} # The model id sent to the runtime. Empty means openplate-plate-1. vLLM # needs its exact served model name. MODEL_ID: ${MODEL_ID:-} # The waiting line, the bound on one completion call, the requests per # key per minute, the largest decoded image, and the downscale target. MAX_QUEUE_DEPTH: ${MAX_QUEUE_DEPTH:-8} RUNTIME_COMPLETION_TIMEOUT_MS: ${RUNTIME_COMPLETION_TIMEOUT_MS:-600000} RATE_LIMIT_RPM: ${RATE_LIMIT_RPM:-60} MAX_IMAGE_BYTES: ${MAX_IMAGE_BYTES:-8388608} IMAGE_MAX_LONG_EDGE: ${IMAGE_MAX_LONG_EDGE:-896} # debug, info, warn or error. LOG_LEVEL: ${LOG_LEVEL:-info} # What the service reports as its profile. Empty follows MODEL_PROFILE. PROFILE: ${PROFILE:-} # Each is read only by its own FOOD_SOURCE. Empty FDC_DATASET_PATH is the # bundled extract. Empty URLs are https://lowcarbcheck.org for lcc and # https://world.openfoodfacts.org for off. FDC_DATASET_PATH: ${FDC_DATASET_PATH:-} LCC_API_URL: ${LCC_API_URL:-} LCC_API_KEY: ${LCC_API_KEY:-} OFF_API_URL: ${OFF_API_URL:-} # A second runtime serving /v1/embeddings, for hybrid retrieval. Empty # means lexical retrieval only. EMBEDDING_RUNTIME_URL: ${EMBEDDING_RUNTIME_URL:-} EMBEDDING_RUNTIME_API_KEY: ${EMBEDDING_RUNTIME_API_KEY:-} # The bundled llama-server: its loopback port, the context per slot, the # layers put on the GPU (empty means detect), and extra flags passed on # as written. RUNTIME_PORT: ${RUNTIME_PORT:-8080} CONTEXT_SIZE: ${CONTEXT_SIZE:-8192} GPU_LAYERS: ${GPU_LAYERS:-} LLAMA_EXTRA_ARGS: ${LLAMA_EXTRA_ARGS:-} # A mirror tried before Hugging Face for the first weight download. WEIGHTS_MIRROR_BASE: ${WEIGHTS_MIRROR_BASE:-} # -- GPU (the `quality` profile) -- uncomment BOTH of these and switch the # image to the -cuda tag. The entrypoint detects the GPU and offloads every # layer automatically; there is no flag to set. # deploy: # resources: # reservations: # devices: # - driver: nvidia # count: all # capabilities: [gpu] # -- ALREADY RUNNING vLLM / llama.cpp / OLLAMA? ------------------------ # Set these in `.env` and this container downloads no weights and starts # no second model; it just turns your runtime into a plate scanner. You can # drop the `volumes:` block and the `inference-models` volume entirely. # # Read README "Bring your own runtime" FIRST: your runtime must enforce # grammar-constrained decoding, and there is a one-line curl there that # tells you whether yours does. vLLM's CPU build does NOT (it crashes). # # MODEL_PROFILE=external # # Must resolve from INSIDE this container, and no trailing /v1. # # `localhost` here means this container, not your host. Use the LAN # # address, or host.docker.internal on Docker Desktop. # MODEL_RUNTIME_URL=http://your-runtime.lan:8000 # # vLLM requires an EXACT match with its served model name. # MODEL_ID=your-served-model-name # # Only if your runtime is behind auth (`vllm serve --api-key ...`). # # Separate from INFERENCE_API_KEY, which callers present to THIS service. # MODEL_RUNTIME_API_KEY=sk_your_runtime_key # # Match your runtime's real slot count: # # llama.cpp --parallel N | vLLM --max-num-seqs N | OLLAMA_NUM_PARALLEL # CONCURRENCY=2 # # REQUIRED with external mode, and the one edit to this file it needs. The # image bakes in a 60-minute health start_period, sized for a first-boot # weight download. External mode has no download, so without this override # a wrong MODEL_RUNTIME_URL stays hidden for an hour instead of surfacing # in seconds. # healthcheck: # start_period: 30s # -- The app --------------------------------------------------------------- # Stateless. No accounts, no personal data, no database: your diary lives in # the browser. There is no secret to configure here -- that is the design, not # an omission (see apps/app/.adr/0006-the-app-server-holds-no-accounts.md). openplate: image: ghcr.io/lowcarbcheck/openplate:latest restart: unless-stopped ports: # Published on EVERY network interface. Behind a reverse proxy on this # machine, write '127.0.0.1:3000:3000' so only the proxy can reach it. - '3000:3000' depends_on: - inference # To serve legal pages, uncomment this and set CONTENT_DIR in `.env`: # CONTENT_DIR=/srv/openplate/content # volumes: # - ./content:/srv/openplate/content:ro environment: NODE_ENV: production PORT: 3000 # The URL a browser uses to reach THIS app: PUBLIC_APP_URL in .env. # Behind a reverse proxy, that is the public https:// address, not the # container port. APP_URL: ${PUBLIC_APP_URL:-http://openplate.example.lan:3000} # ON by default -- queries the public LowCarbCheck food database for # curated nutrition data (food NAMES only, never photos; fails open on # outages). An EMPTY string disables it entirely so no food names ever # leave your machine. FOOD_DB_API_URL: ${FOOD_DB_API_URL-https://lowcarbcheck.org} # Optional free key for that database. Empty is the shared anonymous # allowance; set one if more than one person scans on this instance. FOOD_DB_API_KEY: ${FOOD_DB_API_KEY:-} # "true" passes foods people save from an AI answer on to LowCarbCheck as # proposals. Needs FOOD_DB_API_KEY. Empty means off. FOOD_DB_BACKFILL: ${FOOD_DB_BACKFILL:-} # The most LowCarbCheck calls this server makes in one UTC day. Empty means # the default, 3200. FOOD_DB_DAILY_CALL_LIMIT: ${FOOD_DB_DAILY_CALL_LIMIT:-} # Number of reverse proxies in front of this container. Behind one proxy # keep 1: React Router compares the browser Origin against the host it # thinks it serves, and without the proxy's X-Forwarded-* headers form # posts fail. With NO proxy set TRUST_PROXY=0 in .env: 1 would let any # visitor fake their address in X-Forwarded-For and dodge the # per-address limit on food lookups. 2 = Cloudflare in front of one. TRUST_PROXY: ${TRUST_PROXY:-1} # -- The instance preset: this is what turns into openplate's one-tap # "This openplate provides its own AI" card, on the AI settings page # and on the scan screen. Leave these unset and nothing renders -- # bring-your-own-key stays the only path. # # PUBLIC_INFERENCE_URL in .env: a BROWSER-reachable address for the # inference container. NOT http://inference:8300. Note the /v1 suffix. DEFAULT_INFERENCE_BASE_URL: ${PUBLIC_INFERENCE_URL:-http://openplate.example.lan:8300/v1} # The same INFERENCE_API_KEY as API_KEYS above. # # ! THIS KEY IS PUBLIC. It is embedded in the HTML every browser loads, so # anyone who can open your openplate can read it with view-source. That # is fine for a household or a tailnet. It is NOT fine on an instance # open to the internet without a VPN or auth proxy in front of it -- in # that case leave this unset and let people paste the key themselves. DEFAULT_INFERENCE_API_KEY: ${INFERENCE_API_KEY:-opk_CHANGE_ME} DEFAULT_INFERENCE_MODEL: ${DEFAULT_INFERENCE_MODEL:-openplate-plate-1} # -- Everything else the app reads. Empty means the default. # # The address of openplate-core, one a BROWSER can reach # (the same name compose.core.yml uses). Empty means no sync. CORE_URL: ${PUBLIC_SYNC_URL:-} SYNC_SERVER_URL: ${SYNC_SERVER_URL:-} # open (the default) or managed, which needs a core server. See # apps/app/docs/configuration.md, Managed instances. INSTANCE_MODE: ${INSTANCE_MODE:-open} # The language a first-time visitor sees: en, de, fr, it, es or tr. # Empty means en. DEFAULT_UI_LANGUAGE: ${DEFAULT_UI_LANGUAGE:-} # "off" disables the six-hourly request to openplate.de for the newest version # and the project's daily count of asks. Empty means on. UPDATE_CHECK: ${UPDATE_CHECK:-} # Closes this instance: an https:// address where its people went. # Every page then names it. Empty means open as usual. MOVED_TO_URL: ${MOVED_TO_URL:-} # Extra origins the browser may call, space separated. Empty adds nothing. CSP_CONNECT_EXTRA: ${CSP_CONNECT_EXTRA:-} # debug, info, warn or error, for the inference service as well. LOG_LEVEL: ${LOG_LEVEL:-info} # Which published reference values the Nutrients screen quotes: dge (the # default), efsa or us. NUTRIENT_REFERENCE_BASIS: ${NUTRIENT_REFERENCE_BASIS:-dge} # Matomo analytics, off unless the first two are set together. The level # is pageviews, product (the default when empty) or research, and only # with the pair: a level on its own stops the boot. MATOMO_URL: ${MATOMO_URL:-} MATOMO_SITE_ID: ${MATOMO_SITE_ID:-} MATOMO_EVENT_LEVEL: ${MATOMO_EVENT_LEVEL:-} # A newsletter form on the landing page, off unless both are set. NEWSLETTER_SUBSCRIBE_URL: ${NEWSLETTER_SUBSCRIBE_URL:-} NEWSLETTER_TURNSTILE_SITE_KEY: ${NEWSLETTER_TURNSTILE_SITE_KEY:-} # The folder of legal pages, mounted read-only. Set it to the container # path of the volume line above. Empty means no legal pages. CONTENT_DIR: ${CONTENT_DIR:-} # The address the server binds to inside the container. Leave it empty: # the published port reaches only a server on every interface. HOST: ${HOST:-} healthcheck: # A shell line with no quotes and no brackets, run by the image's busybox # wget, so Docker Compose, podman-compose and Quadlet all run it alike. test: ['CMD-SHELL', 'wget -q -O /dev/null http://127.0.0.1:3000/healthcheck'] interval: 30s timeout: 5s start_period: 30s retries: 3 volumes: inference-models: driver: local