# Copy to .env and edit. Every value has the same default in compose.yaml. # Where the checkpoint lives on the host (scripts/download-model.sh fills it). MODELS_DIR=./models # Directory under MODELS_DIR that holds the checkpoint. MODEL_NAME=dgemma # FlashInfer JIT and torch compile cache; keep it between rebuilds. CACHE_DIR=./cache # Served canvas in tokens. A read only pays for its own width, so wide is # cheap; the width bounds the answer template and a thought's block size. CANVAS=256 # Generation block width by load: [low, high, width] over running plus # waiting requests. 256 is best single-stream, 64 at 32 concurrent. No width # may exceed CANVAS; empty uses CANVAS for every block. CANVAS_SCHEDULE=[[1, 2, 256], [3, 6, 128], [7, 32, 64]] MAX_SEQS=32 MAX_MODEL_LEN=4096 # Fraction of the box's memory vLLM may plan for. 0.40 leaves room on a # 121 GB Spark for the sampler's transients; do not raise it next to another # model. GPU_UTIL=0.40 ATTN=TRITON_ATTN PORT=8010 STRUCTURED_PORT=8011 # Nonzero adds an HTTPS listener on this port with a self-signed certificate # (kept in the cache dir). Browsers only open a webcam on https or localhost. TLS_PORT=0 # Appended to vllm serve. --async-scheduling is always on. EXTRA_ARGS= # The perf branches' server side: reads over the labels only, and a fixed # sample count as one request. MAX_SAMPLES caps samples in both. CONSTRAINED=1 ENGINE_SAMPLES=1 MAX_SAMPLES=32 # gemma4 strips the empty thought block from plain chat and parses tool # calls. Empty turns either off. REASONING_PARSER=gemma4 TOOL_CALL_PARSER=gemma4 # KV pool in GiB. 2 covers a few 4k prompts; see the 128k profile below. KV_CACHE_GB=2 # Empty keeps vLLM's default prefill chunk. MAX_NUM_BATCHED_TOKENS= # The entrypoint refuses to start with less than this free after weights and # the start-up transient. HEADROOM_GB=12 # fp32 [MAX_SEQS x CANVAS, vocab] copies the start-up budget allows for. TRANSIENT_COPIES=2 # Empty = no cap (upstream behaviour). 0.85 caps each worker at that fraction # of device memory, so an overshoot fails the request instead of the host. TORCH_MEM_FRACTION= # 1 serves a playground page at http://:8011/ (JSON request, image # file or webcam frames), and /walk and /cube. Off by default. TEST_PAGE= # When set, POST routes on the structured port need "Authorization: Bearer ". # GET /health and the playground page stay open; the page has a field for the key. API_KEY= # 128k-context profile for long documents. Uncomment together. A 128k # request takes about 1.7 GiB of KV on this model, so 24 GiB holds thirteen # at once; see README for what a long state costs in time. #MAX_MODEL_LEN=131072 #KV_CACHE_GB=24 #GPU_UTIL=0.45