Files
catalogue_backend/.env.example
2026-09-29 16:42:49 +05:30

441 lines
21 KiB
Plaintext

# Copy this file to .env and fill in your own values.
# Nothing here is a real credential.
#
# The host-side values below are for running the backend directly with uvicorn
# (python run_project.py). When the backend runs INSIDE Docker via the root
# docker-compose.yml, four of them are overridden by compose and you do not
# need to change them here:
#
# DB_HOST -> postgres (service name)
# DB_PORT -> 5432
# DB_PASSWORD -> POSTGRES_PASSWORD from the root .env
# OLLAMA_BASE_URL -> http://ollama:11434
#
# The reason is worth internalising: inside a container, `localhost` is the
# container itself, not the host and not a sibling container. A container-bound
# localhost URL points the backend at its own empty ports. Containers reach
# each other by service name over the compose network instead.
# --- Ports (container only) -----------------------------------------------
# The image serves 3000 and 8000 at once - 3000 because that is what Dokploy
# routes a domain to, 8000 because the README, the vite dev proxy and
# docker-compose all use it. Serving both means the container works whichever
# one the platform points at.
#
# PORTS the comma-separated pair to bind. PORT pins a single port instead and
# takes precedence, e.g. PORT=8080 serves only 8080.
#
# Neither affects running uvicorn directly for local development.
# PORTS=3000,8000
# PORT=8080
# --- Persistence (IMPORTANT in Docker) ------------------------------------
# The three directories the app WRITES to at runtime. The defaults point inside
# the repo/image and are right for local development; in a container each one
# needs a volume mounted on it, or a redeploy throws away everything written
# since the last build:
#
# SEED_CATALOG_DIR products added via POST /api/user/products/add and
# /upload-file are appended to the JSON files here
# MODEL_ARTIFACTS_DIR *.joblib bundles written by the training endpoints
# DATA_DIR catalogs saved by the ingestion pipeline
#
# Mount these two paths in Dokploy (SEED_CATALOG_DIR sits inside DATA_DIR, so
# one mount covers both):
#
# /app/data
# /app/app/intelligence/artifacts
#
# Either a named volume or a bind mount works. The image keeps read-only copies
# of the bundled seed catalogs and pre-trained models at /app/.bundled, and the
# app restores whatever a freshly-mounted directory is missing on startup
# without overwriting anything already there - so a bind mount, which starts
# empty and would otherwise hide them, is safe.
#
# Leave all three unset unless the writable data belongs somewhere else.
# DATA_DIR=/app/data
# SEED_CATALOG_DIR=/app/data/seed_catalogs
# MODEL_ARTIFACTS_DIR=/app/app/intelligence/artifacts
# --- CORS (REQUIRED when the API is on its own domain) --------------------
# Comma-separated list of the exact browser origins allowed to call this API.
# Scheme and host both matter; no trailing slash, and no wildcard - the
# frontend sends an Authorization header, and browsers refuse a credentialed
# cross-origin request whose Allow-Origin is "*" (app/main.py logs and turns
# credentials off if it sees one, so a wildcard silently breaks every call).
#
# Production - the React app is served from catalogue.nearle.ai.in and calls
# the API at mcp.nearle.ai.in, so that frontend origin must be listed:
#
# API_CORS_ORIGINS=https://catalogue.nearle.ai.in
#
# Add http://localhost:5173 alongside it if you point a local Vite dev server
# at the deployed API. Server-to-server callers (MCP clients, scripts) are not
# affected by any of this - CORS is a browser rule; they use X-API-Key.
API_CORS_ORIGINS=http://localhost:5173,http://127.0.0.1:5173
# --- Authentication (REQUIRED) -------------------------------------------
# The backend will not start without these while AUTH_ENABLED=true. Generate
# all four lines, plus sign-in passwords, with:
#
# python scripts/make_auth_secrets.py
#
# They guard the 18 write/compute endpoints - catalog generation, ML training,
# the upload endpoints and chat. CORS is not a substitute: browsers enforce it,
# curl ignores it entirely.
#
# AUTH_ENABLED=false turns every guard off and makes the whole API open again.
# It exists so a fresh checkout runs before you have generated secrets. Never
# set it false on a host reachable from the internet.
AUTH_ENABLED=true
# Signs access tokens. Changing it signs everybody out, which is how you revoke
# every issued token at once. Use a DIFFERENT value in production from the one
# on your laptop - a secret that has been on a dev machine is not a secret.
AUTH_SECRET_KEY=
# Only PBKDF2 digests are stored, never passwords. `make_auth_secrets.py`
# prints the password once and the hash to paste here; it cannot be reversed,
# so rerun the script to change a password.
AUTH_ADMIN_USERNAME=admin
AUTH_ADMIN_PASSWORD_HASH=
AUTH_USER_USERNAME=user
AUTH_USER_PASSWORD_HASH=
# Token lifetime in minutes. 12h by default: one sign-in per working day, and
# a leaked token expires by itself.
AUTH_TOKEN_TTL_MINUTES=720
# Failed-login throttle, per username+IP. Stops the login endpoint being an
# unlimited password oracle once it is on the internet.
AUTH_MAX_LOGIN_ATTEMPTS=10
AUTH_LOCKOUT_SECONDS=300
# Local development only: accept ANY password at /api/auth/login. The username
# still picks the role (`admin` -> admin pages, `user` -> user pages), and the
# token issued is a normal one, so every other guard behaves normally. Unlike
# AUTH_ENABLED=false it leaves the login page working - it just stops checking
# the password. Anyone who can reach the port becomes admin: keep it false here.
AUTH_ALLOW_ANY_LOGIN=true
# Machine consumers of api.<domain> - scripts, partner integrations, your own
# backends. Format: name:role:secret, comma-separated. Role is one of:
#
# admin everything, including system/init and model training
# user catalog write access (add products, upload inventory)
# uploader ONE verb: POST /api/uploads/catalog. Nothing else - it cannot
# read the catalog, cannot see another caller's submissions, and
# cannot cancel or resume anything. This is the role to issue to
# an outside party who needs to send you spreadsheets.
#
# Callers send the secret as an X-API-Key header.
#
# One entry per consumer, always: a shared key cannot be revoked for one caller
# without breaking every other. Mint them with:
# python scripts/make_auth_secrets.py --api-key partner-x:user
# python scripts/make_auth_secrets.py --api-key catalog-drop:uploader
#
# NAME THE KEY FOR ITS FUNCTION, NOT THE PERSON HOLDING IT. /api/health is
# public and reports {name, role, fingerprint} for every configured key. The
# secret is never exposed, but the NAME is - so `catalog-drop:uploader:...` is
# right and `priya-laptop:uploader:...` publishes a colleague's name to anyone
# who curls the health endpoint.
#
# Leave empty if only the web app calls the API - it signs in through
# /api/auth/login instead, and a key nobody needs is only risk.
API_KEYS=
# ---------------------------------------------------------------------------
# Catalog batch ingestion
# ---------------------------------------------------------------------------
# Used by both spreadsheet ingestion routes - the admin one
# (POST /api/admin/catalog-batch/ingest) and the API-client one
# (POST /api/uploads/catalog). Every ceiling is enforced in the application,
# not at the proxy: in production the caller reaches mcp.nearle.ai.in directly,
# so neither nginx's client_max_body_size nor Caddy's request_body cap is in
# front of these endpoints.
#
# Per-file limits are fixed in code at 10MB / 2000 rows, matching the
# single-file upload path. These bound the BATCH on top of that.
BATCH_MAX_FILES=20
BATCH_MAX_TOTAL_BYTES=52428800
BATCH_MAX_TOTAL_ROWS=20000
# Where staged uploads live. Under DATA_DIR because that path is already a
# declared volume, which is what lets a batch survive a container restart.
# BATCH_UPLOAD_DIR=/app/data/batch_uploads
# Batches allowed to wait behind the one running. One worker thread runs a
# single batch at a time; past this depth the endpoints answer 429 rather than
# accepting work they have no intention of starting soon. This queue is the
# only thing bounding what an uploader key can cost in CPU - raise it with care
# on a one-vCPU host.
BATCH_QUEUE_MAX=4
# Staged files are deleted this many days after the batch was created.
BATCH_RETENTION_DAYS=7
# Deliberately false. A batch a restart cut short is marked "interrupted" and
# waits for someone to press Resume. Auto-resuming means a container stuck in a
# restart loop re-runs the heaviest work in the app on every boot, which is how
# a slow start becomes an unrecoverable spiral.
#
# Worth knowing alongside UPLOAD_AUTORUN below: with uploads running unattended,
# a redeploy in the middle of one leaves that batch "interrupted" and waiting
# for a human. It is the one place manual intervention comes back.
BATCH_AUTO_RESUME=false
# ---------------------------------------------------------------------------
# Unattended ingestion
# ---------------------------------------------------------------------------
# Whether POST /api/uploads/catalog runs the pipeline on arrival (true) or parks
# the files in the admin review inbox for someone to start by hand (false).
#
# READ THIS BEFORE CHANGING IT. That endpoint takes NO credential - it was
# opened on purpose so colleagues could send spreadsheets without one being
# issued to them. With autorun on, "anyone who can reach this host" and "anyone
# who can write to the live catalogue" are the same set of people, and an ingest
# is an upsert with no undo.
#
# What still bounds it is throughput, not identity: the per-request ceilings
# above, and BATCH_QUEUE_MAX behind a single worker. A sender can occupy the
# ingestion worker; they cannot multiply it.
#
# Set false and the review inbox comes back with no code change - the INBOX_*
# ceilings below apply only on that path.
UPLOAD_AUTORUN=true
# How an auto-started run behaves. Deliberately NOT accepted from the request:
# the sender is anonymous, and letting an anonymous caller switch on the
# expensive outbound stages is the one thing this endpoint must not allow.
#
# Images on, because a product landing without one is the failure this endpoint
# exists to avoid. Stage 6 is the slowest stage and reaches the network, but
# only one batch runs at a time, so nothing else competes with it.
#
# LLM on, though in production it currently does nothing: USE_OLLAMA is false
# there, so the LLM call returns immediately without a request and the row keeps
# its blank description. Set true so the pipeline is already right for the day
# an Ollama server is reachable.
#
# If you DO set USE_OLLAMA=true, make sure something is actually listening on
# OLLAMA_BASE_URL. An unreachable server costs a 5s probe, and while that probe
# is cached per 30s rather than paid per row, a reachable-but-slow model is
# billed per row at OLLAMA_TIMEOUT_SECONDS.
UPLOAD_AUTORUN_FETCH_IMAGES=true
UPLOAD_AUTORUN_USE_LLM=true
USE_OLLAMA=true
OLLAMA_BASE_URL=http://localhost:11434
OLLAMA_MODEL_NAME=qwen2.5:1.5b
USE_EMBEDDINGS=true
EMBEDDINGS_MODEL=sentence-transformers/all-MiniLM-L6-v2
USE_PGVECTOR=true
DB_HOST=localhost
DB_PORT=5432
DB_NAME=pgvector
DB_USER=postgres
DB_PASSWORD=changeme
# How often (seconds) to reconcile brand_* tables against data/seed_catalogs/.
#
# Brands are discovered from the database, not from a list: every brand_* table
# is enumerated on each request, so a new one shows up in the catalog with no
# registration step. This interval covers the other direction - mirroring a
# table created directly in the database into a seed catalog file - which
# otherwise only happens at startup or via POST /api/system/brand-sync.
#
# A sweep with nothing to do is one COUNT(*) per brand table and writes nothing.
# Raise it if the database is remote and the chatter matters; 0 turns the loop
# off entirely, leaving only the startup run and the manual endpoint.
BRAND_SYNC_INTERVAL_SECONDS=300
# --- Active brands (development working set) --------------------------------
# Comma-separated. BLANK OR UNSET = every brand is active (the production
# default). Setting it narrows the catalog, RAG, search, MCP, analytics,
# nutrition and every Dagster asset to these brands in one place - nothing is
# deleted, the other brand_* tables just stop being discovered.
# Names resolve through the brand aliases, so "Tata" activates
# brand_hindustan_unilever exactly as ingesting Tata products would.
# NOTE: this is a whitelist, so anything omitted is invisible in browse, search,
# suggestions, chat and the MCP tools. If you narrow it, include "Own Products"
# - that is the bucket unbranded commodities (dal, sugar, salt, spices) are
# filed under, and it is a normal brand table as far as every read path is
# concerned.
#ACTIVE_BRANDS=Amul,Cadbury,Hindustan Unilever,Own Products
# ---------------------------------------------------------------------------
# Brand discovery (a brand NAME -> the 11-stage pipeline)
# ---------------------------------------------------------------------------
# POST /api/admin/brand-discovery/preview finds a brand's products, and
# /ingest stages the ones an admin approved as an ordinary catalog batch.
# Every value below has a working default; none of these need to be set.
#
# Open Food Facts is the primary source and the language model is the
# supplement. OFF returns real products with real barcodes and pack sizes;
# the default OLLAMA_MODEL_NAME (qwen2.5:1.5b) will invent plausible ones, and
# nothing downstream can tell a well-formed fiction from a real product. Turn
# BRAND_DISCOVERY_USE_OFF off and the result rests on the model alone.
#
# NOTE: discovering a brand that is not in ACTIVE_BRANDS writes a complete
# catalog that no endpoint can read. The ingest route refuses with a 409 and
# names the line to add here; it is a config change plus a restart, never a
# re-ingest.
#BRAND_DISCOVERY_USE_OFF=true
#BRAND_DISCOVERY_USE_LLM=true
#BRAND_DISCOVERY_MAX_PRODUCTS=200
# Pack sizes kept per product when only the language model offers any. Stage 6
# runs an image search per exploded row, so this multiplies the slowest stage.
#BRAND_DISCOVERY_MAX_SIZES=3
# Wall-clock ceiling on the language-model half, checked between prompts. Open
# Food Facts runs first and is never subject to it.
#BRAND_DISCOVERY_DEADLINE_SECONDS=300
# Brand Discovery's optional "Web & retail listings" source, for brands Open
# Food Facts does not carry. Reads search-engine results that point at retailer
# product pages (Blinkit, Zepto, Instamart, BigBasket, JioMart, Amazon,
# Flipkart...) - never the pages themselves, never a language model. Uses the
# Google Custom Search API when USE_GOOGLE_CSE=true, else DuckDuckGo.
#WEB_DISCOVERY_ENABLED=true
#WEB_DISCOVERY_MAX_QUERIES=80
#WEB_DISCOVERY_PAUSE_SECONDS=2.0
#WEB_DISCOVERY_RESULTS_PER_QUERY=20
#WEB_DISCOVERY_CACHE_DAYS=7
#WEB_DISCOVERY_DIR=/app/data/cache/web_discovery
USE_S3=true
S3_ACCESS_KEY=your-do-spaces-key
S3_SECRET_KEY=your-do-spaces-secret
S3_ENDPOINT=https://your-space.region.digitaloceanspaces.com
S3_BUCKET=your-bucket
S3_REGION=region
# Optional - leave USE_GOOGLE_CSE=false if you don't have a Google CSE key
USE_GOOGLE_CSE=false
GOOGLE_API_KEY=
GOOGLE_CSE_ID=
USE_DDG_IMAGES=true
USE_OPEN_FACTS=true
USE_WIKIMEDIA=true
USE_PLAYWRIGHT_FALLBACK=true
MIN_IMAGE_BYTES=3000
# img_vector: a MobileNetV3-Small embedding (1024 floats, L2-normalised) of
# each product's primary image, stored on its brand table and computed on a
# background thread after every catalog write (app/services/image_vector.py,
# model in app/services/image_embedder.py). Existing rows are filled by
# `python -m scripts.backfill_image_vectors --all --apply`. Set false to stop
# the API process from downloading images at all.
ENABLE_IMAGE_VECTORS=true
# Where the .tflite lives. The default is inside app/ (shipped with the image
# and never volume-mounted); only override to point at a different file.
#IMAGE_EMBED_MODEL_PATH=/app/app/services/models/mobilenet/mobilenet_v3_small_embedder.tflite
#IMAGE_EMBED_NUM_THREADS=2
# Identify a product from a phone photo (POST /api/search/identify). A phone
# photo scores ~0.63 on img_vector against the catalog's render of the same
# pack, so below IMAGE_IDENTIFY_MIN_IMAGE_SCORE the label text decides: the
# client's OCR `text` if it sent one, else what the server reads off the photo
# with rapidocr (models ship in the wheel; see requirements-ocr.txt). Set
# ENABLE_SERVER_OCR=false to rely on client text only.
ENABLE_SERVER_OCR=true
#OCR_MIN_CONFIDENCE=0.5
#OCR_MAX_SIDE_PX=1280
#OCR_NUM_THREADS=2
#IMAGE_IDENTIFY_MIN_IMAGE_SCORE=0.70
#IMAGE_IDENTIFY_MIN_TEXT_SCORE=0.60
# An image match is only "confirmed" (match_confidence) when it also leads the
# best different photo by this much.
#IMAGE_SEARCH_MIN_MARGIN=0.05
# Diagnosis only: save every image-search request as JSON for
# `python -m scripts.replay_image_query`. Blank = off; keeps the newest N.
#IMAGE_SEARCH_CAPTURE_DIR=/app/data/image_search_requests
#IMAGE_SEARCH_CAPTURE_MAX=200
# Capture-to-catalog: when identify cannot confirm a photo, read the label and -
# for a brand already in the catalog - add the product through the 11-stage
# pipeline in the background (validation_status=needs_review). The response
# carries a provisional card and a job id to poll at
# GET /api/search/identify/jobs/{id}. OFF by default: it is the only way a
# public route writes to a brand table. See app/services/capture_discovery.py.
ENABLE_CAPTURE_DISCOVERY=false
#CAPTURE_DIR=/app/data/captures
#CAPTURE_QUEUE_MAX=8
#CAPTURE_MAX_PER_CLIENT_PER_HOUR=20
# The API's public base URL. Set it to let the colleague's photo become the
# product image when the web search finds none; blank = never.
#CAPTURE_PUBLIC_BASE_URL=https://api.example.com
#CAPTURE_RETAIL_CHECK=true
#CAPTURE_USE_LLM=true
# Product SKU: try a live web search for a real marketplace product ID
# (Amazon ASIN, Flipkart PID, etc.) before falling back to an internal SKU.
# Set to false to always generate internal SKUs only (faster, offline-safe).
ENABLE_SKU_WEB_LOOKUP=true
# Deterministic product validation (see app/services/product_validator.py).
# Final gate applied to every catalog row before it's kept; rows scoring
# below VALIDATION_REJECT_THRESHOLD are dropped (with reasons recorded),
# rows below VALIDATION_REVIEW_THRESHOLD are kept but flagged for human review.
ENABLE_PRODUCT_VALIDATION=true
VALIDATION_REJECT_THRESHOLD=0.35
VALIDATION_REVIEW_THRESHOLD=0.70
# Per-pack-size product images (see docs/PER_VARIANT_IMAGES.md). When true,
# each size variant of a product (e.g. 100g/200g/500g) gets its own
# size-targeted image search/upload instead of sharing one generic image
# set. Falls back to the old shared behaviour automatically whenever no
# size-specific match is found, so this is safe to leave on. Set to
# "false" to fully restore the old shared-image behaviour.
ENABLE_PER_VARIANT_IMAGES=true
# Max image candidates fetched per size variant (kept smaller than the
# ~24 used for the general product-level search, since a per-size search
# only needs a handful of genuinely matching candidates).
PER_VARIANT_IMAGE_MAX_RESULTS=10
# Barcode Retrieval & Product Enrichment (see
# app/services/enrichment/barcode/ and docs/BARCODE_ENRICHMENT.md).
#
# INLINE, per product, during ingestion. Defaults FALSE in settings.py and this
# file used to ship `true`, which disagreed with the code for as long as both
# existed - anyone copying .env.example got a very different pipeline from
# anyone relying on the defaults. It is false here now to match.
#
# Leave it false unless you know the upload is small: it costs one search
# request per product against an endpoint capped at 10 requests/minute, so a
# 200-row sheet is twenty minutes of held request. ENRICH_BARCODES_ON_UPLOAD
# below is the cheap path and is on by default.
ENABLE_BARCODE_LOOKUP=false
# BULK, per brand, after the upload settles. Fetches each brand's whole Open
# Food Facts catalogue (~5 requests) and matches offline, on the enrichment
# job's own thread. Runs before the nutrition phase, because a barcode turns a
# 0.32-confidence name lookup into a 0.95-confidence exact one.
ENRICH_BARCODES_ON_UPLOAD=true
BARCODE_LOOKUP_TIMEOUT_SECONDS=10
# 30 days, in seconds
BARCODE_LOOKUP_CACHE_TTL_SECONDS=2592000
BARCODE_LOOKUP_MAX_CONCURRENCY=5
BARCODE_COUNTRY_TAG=india
# GS1 India tier - leave both blank to skip (no free public API exists at
# time of writing; the cascade falls through to Open Food Facts). Fill in
# if/when real GS1 India ("Verified by GS1") API access is provisioned.
GS1_INDIA_API_BASE_URL=
GS1_INDIA_API_KEY=
# UPCItemDB - works out of the box on the free trial tier (no key needed,
# rate limited ~100 req/day). Set to switch to the paid production endpoint.
UPC_DATABASE_API_KEY=
# Lowest-trust fallback tier: DuckDuckGo search + manufacturer page text
# scan for a labelled barcode (uses the same `ddgs` dependency as SKU
# lookup). Set to false to skip this tier.
ENABLE_MANUFACTURER_SITE_LOOKUP=true