# Copy to .env and fill in. .env is git-ignored. # Every commented-out line shows the default. See README.md > Configuration. # --- Required ----------------------------------------------------------------- MISTRAL_API_KEY=your-mistral-api-key # --- Docker compose ----------------------------------------------------------- # Host folders mounted at /data (inputs) and /config (work folder). DATA_PATH=./data CONFIG_PATH=./config # linuxserver.io conventions: user and group that own the data folder, the # permission mask for new files (002: group-writable), and the time zone. PUID=1000 PGID=1000 UMASK=022 TZ=Etc/UTC # --- Inputs and folders ------------------------------------------------------- # stacks/ splits large scans into documents; scanner/ takes one document per # file (e.g. a scanner saving to a network share). Each has inbox/, output/, # archive/ and failed/ below its root. # STACKS_ENABLED=true # SCANNER_ENABLED=true # DATA_DIR=/data # Text for splitting and naming: mistral (Mistral OCR, paid, splits somewhat # better) or tesseract (the text layer ocrmypdf adds anyway, free, no wait). # See README > Choosing the text source. # STACKS_TEXT_SOURCE=mistral # Default for the scanner: tesseract (mistral if OCRMYPDF_ENABLED=false). # SCANNER_TEXT_SOURCE=tesseract # STACKS_DIR=/data/stacks # SCANNER_DIR=/data/scanner # WORK_DIR=/config # Days after which a processed file's work folder (searchable copy, OCR text, # plan.json) is deleted. rebuild works until then. 0 keeps everything. # WORK_RETENTION_DAYS=30 # Temporary page images; default: $WORK_DIR/tmp, on disk rather than in RAM. # TMPDIR= # A file must stay unchanged this long (seconds) before it is picked up. # STABLE_SECONDS=60 # POLL_INTERVAL=30 # --- Paperless-ngx input ----------------------------------------------------- # Setting PAPERLESS_URL enables a third inbox (paperless/inbox/). Its files get # the Tesseract text layer and are uploaded to Paperless-ngx, which names and # tags them itself. # PAPERLESS_URL=http://paperless:8000 # PAPERLESS_TOKEN=your-paperless-api-token # PAPERLESS_DIR=/data/paperless # Comma-separated tag ids to add on upload. # PAPERLESS_TAGS= # PAPERLESS_MAX_WAIT_MINUTES=30 # mistral: replace the document's content in Paperless with Mistral OCR's text # (tables kept as Markdown). tesseract: keep the text layer's plain text. # PAPERLESS_TEXT_SOURCE=tesseract # A second Paperless input (paperless-2/inbox/) that uploads with its own # token, so its documents belong to another Paperless user. Without # PAPERLESS_2_URL it uses the same instance as PAPERLESS_URL. # PAPERLESS_2_TOKEN=api-token-of-the-second-user # PAPERLESS_2_URL= # PAPERLESS_2_DIR=/data/paperless-2 # PAPERLESS_2_TAGS= # PAPERLESS_2_TEXT_SOURCE=tesseract # Make every tag, correspondent or document type ownerless, so all Paperless # users see and may change it. Checked every PAPERLESS_SHARE_TAGS_MINUTES. # Needs a token that may change other users' objects (a superuser's). Tag # names in PAPERLESS_SHARE_TAGS_READONLY keep their owner and are only # visible to everyone else. # PAPERLESS_SHARE_TAGS=false # PAPERLESS_SHARE_CORRESPONDENTS=false # PAPERLESS_SHARE_DOCUMENT_TYPES=false # PAPERLESS_SHARE_TAGS_MINUTES=1 # PAPERLESS_SHARE_TAGS_READONLY=ai-processed # --- Mistral ------------------------------------------------------------------ # MISTRAL_LLM_MODEL=mistral-large-latest # MISTRAL_OCR_MODEL=mistral-ocr-latest # Requests per second across all workers. Your per-model limit is shown in # Mistral's console under API > Limits; 0 disables the throttle. # MISTRAL_MAX_RPS=1 # MISTRAL_API_BASE=https://api.mistral.ai/v1 # MISTRAL_TIMEOUT=300 # --- OCR ---------------------------------------------------------------------- # batch: Mistral's batch API at half price; a job takes minutes, not seconds. # direct: full price, immediate. # OCR_MODE=batch # BATCH_POLL_SECONDS=15 # A job still running after this long is cancelled; its chunks are retried directly. # BATCH_MAX_WAIT_HOURS=24 # --- Spending limit ----------------------------------------------------------- # When Mistral refuses the account (spending limit, quota, rejected key), # processing pauses and one file is retried as a probe this often. # PAUSE_RETRY_MINUTES=30 # --- Naming ------------------------------------------------------------------- # Language for titles and summaries. Default: each document's own language. # TITLE_LANGUAGE=English # {title} is required, {date} is YYYY-MM-DD. # FILENAME_PATTERN={title} {date} # NO_DATE_LABEL=undated # Text per document sent for naming; longer documents are shortened in the middle. # METADATA_MAX_CHARS=24000 # --- Splitting and blank pages ------------------------------------------------ # Pages per model request, and how far each window moves on. # BOUNDARY_WINDOW=12 # BOUNDARY_STEP=6 # Documents with a split decision below this confidence are flagged in review.md. # REVIEW_CONFIDENCE=0.75 # DROP_BLANK_PAGES=true # Pages with less visible ink than this (percent of the page) count as blank. # BLANK_MAX_INK_PERCENT=0.2 # Pages without images and with at most this many characters count as blank. # BLANK_MAX_CHARS=15 # --- Text layer (ocrmypdf / Tesseract) ---------------------------------------- # OCRMYPDF_ENABLED=true # Tesseract languages, joined with +. deu, eng and osd are built in; others # (e.g. fra, ita, spa) are downloaded once at startup into /config/tessdata. # OCRMYPDF_LANGUAGES=deu+eng # Download source for those, e.g. a local mirror of tessdata_best 4.1.0. # TESSDATA_URL=https://github.com/tesseract-ocr/tessdata_best/raw/4.1.0 # Parallel OCR pages. auto: as many as the container's memory allows # (1 GB base + 0.75 GB per page, at most one per CPU core). Less memory makes # processing slower, not fail. Shared by all inputs; scanner and Paperless # files go ahead of stacks. # OCRMYPDF_JOBS=auto # Images above this resolution are downsampled before OCR (0 = never). # OCRMYPDF_MAX_IMAGE_DPI=600 # OCRMYPDF_EXTRA_ARGS= # Limits, so one odd page can't exhaust the server. Peak memory is about # OCRMYPDF_JOBS x 16 bytes x OCRMYPDF_MAX_OCR_MPIXELS (megapixels). # OCRMYPDF_MAX_OCR_MPIXELS=50 # OCRMYPDF_PAGE_TIMEOUT=300 # OCRMYPDF_FILE_TIMEOUT_MINUTES=120 # OCRMYPDF_SKIP_BIG_MPIXELS=200 # --- Queue webhook ------------------------------------------------------------ # POSTs queue counts (JSON, no file names) on every change and as a heartbeat, # e.g. to a Home Assistant webhook trigger. Empty disables it. # QUEUE_WEBHOOK_URL=http://homeassistant.local:8123/api/webhook/your-random-id # QUEUE_WEBHOOK_CHECK_SECONDS=10 # QUEUE_WEBHOOK_HEARTBEAT_SECONDS=300 # --- Throughput and logging --------------------------------------------------- # OCR_CHUNK_PAGES=50 # OCR_CONCURRENCY=3 # LLM_CONCURRENCY=4 # LOG_LEVEL=INFO