# Copy this file to .env and fill in your own values. # Nothing here is a real credential. # # The host-side values below are for running the backend directly with uvicorn # (python run_project.py). When the backend runs INSIDE Docker via the root # docker-compose.yml, four of them are overridden by compose and you do not # need to change them here: # # DB_HOST -> postgres (service name) # DB_PORT -> 5432 # DB_PASSWORD -> POSTGRES_PASSWORD from the root .env # OLLAMA_BASE_URL -> http://ollama:11434 # # The reason is worth internalising: inside a container, `localhost` is the # container itself, not the host and not a sibling container. A container-bound # localhost URL points the backend at its own empty ports. Containers reach # each other by service name over the compose network instead. # --- Ports (container only) ----------------------------------------------- # The image serves 3000 and 8000 at once - 3000 because that is what Dokploy # routes a domain to, 8000 because the README, the vite dev proxy and # docker-compose all use it. Serving both means the container works whichever # one the platform points at. # # PORTS the comma-separated pair to bind. PORT pins a single port instead and # takes precedence, e.g. PORT=8080 serves only 8080. # # Neither affects running uvicorn directly for local development. # PORTS=3000,8000 # PORT=8080 # --- Persistence (IMPORTANT in Docker) ------------------------------------ # The three directories the app WRITES to at runtime. The defaults point inside # the repo/image and are right for local development; in a container each one # needs a volume mounted on it, or a redeploy throws away everything written # since the last build: # # SEED_CATALOG_DIR products added via POST /api/user/products/add and # /upload-file are appended to the JSON files here # MODEL_ARTIFACTS_DIR *.joblib bundles written by the training endpoints # DATA_DIR catalogs saved by the ingestion pipeline # # Mount these two paths in Dokploy (SEED_CATALOG_DIR sits inside DATA_DIR, so # one mount covers both): # # /app/data # /app/app/intelligence/artifacts # # Either a named volume or a bind mount works. The image keeps read-only copies # of the bundled seed catalogs and pre-trained models at /app/.bundled, and the # app restores whatever a freshly-mounted directory is missing on startup # without overwriting anything already there - so a bind mount, which starts # empty and would otherwise hide them, is safe. # # Leave all three unset unless the writable data belongs somewhere else. # DATA_DIR=/app/data # SEED_CATALOG_DIR=/app/data/seed_catalogs # MODEL_ARTIFACTS_DIR=/app/app/intelligence/artifacts # --- CORS (REQUIRED when the API is on its own domain) -------------------- # Comma-separated list of the exact browser origins allowed to call this API. # Scheme and host both matter; no trailing slash, and no wildcard - the # frontend sends an Authorization header, and browsers refuse a credentialed # cross-origin request whose Allow-Origin is "*" (app/main.py logs and turns # credentials off if it sees one, so a wildcard silently breaks every call). # # Production - the React app is served from catalogue.nearle.ai.in and calls # the API at mcp.nearle.ai.in, so that frontend origin must be listed: # # API_CORS_ORIGINS=https://catalogue.nearle.ai.in # # Add http://localhost:5173 alongside it if you point a local Vite dev server # at the deployed API. Server-to-server callers (MCP clients, scripts) are not # affected by any of this - CORS is a browser rule; they use X-API-Key. API_CORS_ORIGINS=http://localhost:5173,http://127.0.0.1:5173 # --- Authentication (REQUIRED) ------------------------------------------- # The backend will not start without these while AUTH_ENABLED=true. Generate # all four lines, plus sign-in passwords, with: # # python scripts/make_auth_secrets.py # # They guard the 18 write/compute endpoints - catalog generation, ML training, # the upload endpoints and chat. CORS is not a substitute: browsers enforce it, # curl ignores it entirely. # # AUTH_ENABLED=false turns every guard off and makes the whole API open again. # It exists so a fresh checkout runs before you have generated secrets. Never # set it false on a host reachable from the internet. AUTH_ENABLED=true # Signs access tokens. Changing it signs everybody out, which is how you revoke # every issued token at once. Use a DIFFERENT value in production from the one # on your laptop - a secret that has been on a dev machine is not a secret. AUTH_SECRET_KEY= # Only PBKDF2 digests are stored, never passwords. `make_auth_secrets.py` # prints the password once and the hash to paste here; it cannot be reversed, # so rerun the script to change a password. AUTH_ADMIN_USERNAME=admin AUTH_ADMIN_PASSWORD_HASH= AUTH_USER_USERNAME=user AUTH_USER_PASSWORD_HASH= # Token lifetime in minutes. 12h by default: one sign-in per working day, and # a leaked token expires by itself. AUTH_TOKEN_TTL_MINUTES=720 # Failed-login throttle, per username+IP. Stops the login endpoint being an # unlimited password oracle once it is on the internet. AUTH_MAX_LOGIN_ATTEMPTS=10 AUTH_LOCKOUT_SECONDS=300 # Local development only: accept ANY password at /api/auth/login. The username # still picks the role (`admin` -> admin pages, `user` -> user pages), and the # token issued is a normal one, so every other guard behaves normally. Unlike # AUTH_ENABLED=false it leaves the login page working - it just stops checking # the password. Anyone who can reach the port becomes admin: keep it false here. AUTH_ALLOW_ANY_LOGIN=true # Machine consumers of api. - scripts, partner integrations, your own # backends. Format: name:role:secret, comma-separated. Role is one of: # # admin everything, including system/init and model training # user catalog write access (add products, upload inventory) # uploader ONE verb: POST /api/uploads/catalog. Nothing else - it cannot # read the catalog, cannot see another caller's submissions, and # cannot cancel or resume anything. This is the role to issue to # an outside party who needs to send you spreadsheets. # # Callers send the secret as an X-API-Key header. # # One entry per consumer, always: a shared key cannot be revoked for one caller # without breaking every other. Mint them with: # python scripts/make_auth_secrets.py --api-key partner-x:user # python scripts/make_auth_secrets.py --api-key catalog-drop:uploader # # NAME THE KEY FOR ITS FUNCTION, NOT THE PERSON HOLDING IT. /api/health is # public and reports {name, role, fingerprint} for every configured key. The # secret is never exposed, but the NAME is - so `catalog-drop:uploader:...` is # right and `priya-laptop:uploader:...` publishes a colleague's name to anyone # who curls the health endpoint. # # Leave empty if only the web app calls the API - it signs in through # /api/auth/login instead, and a key nobody needs is only risk. API_KEYS= # --------------------------------------------------------------------------- # Catalog batch ingestion # --------------------------------------------------------------------------- # Used by both spreadsheet ingestion routes - the admin one # (POST /api/admin/catalog-batch/ingest) and the API-client one # (POST /api/uploads/catalog). Every ceiling is enforced in the application, # not at the proxy: in production the caller reaches mcp.nearle.ai.in directly, # so neither nginx's client_max_body_size nor Caddy's request_body cap is in # front of these endpoints. # # Per-file limits are fixed in code at 10MB / 2000 rows, matching the # single-file upload path. These bound the BATCH on top of that. BATCH_MAX_FILES=20 BATCH_MAX_TOTAL_BYTES=52428800 BATCH_MAX_TOTAL_ROWS=20000 # Where staged uploads live. Under DATA_DIR because that path is already a # declared volume, which is what lets a batch survive a container restart. # BATCH_UPLOAD_DIR=/app/data/batch_uploads # Batches allowed to wait behind the one running. One worker thread runs a # single batch at a time; past this depth the endpoints answer 429 rather than # accepting work they have no intention of starting soon. This queue is the # only thing bounding what an uploader key can cost in CPU - raise it with care # on a one-vCPU host. BATCH_QUEUE_MAX=4 # Staged files are deleted this many days after the batch was created. BATCH_RETENTION_DAYS=7 # Deliberately false. A batch a restart cut short is marked "interrupted" and # waits for someone to press Resume. Auto-resuming means a container stuck in a # restart loop re-runs the heaviest work in the app on every boot, which is how # a slow start becomes an unrecoverable spiral. BATCH_AUTO_RESUME=false USE_OLLAMA=true OLLAMA_BASE_URL=http://localhost:11434 OLLAMA_MODEL_NAME=qwen2.5:1.5b USE_EMBEDDINGS=true EMBEDDINGS_MODEL=sentence-transformers/all-MiniLM-L6-v2 USE_PGVECTOR=true DB_HOST=localhost DB_PORT=5432 DB_NAME=pgvector DB_USER=postgres DB_PASSWORD=changeme # How often (seconds) to reconcile brand_* tables against data/seed_catalogs/. # # Brands are discovered from the database, not from a list: every brand_* table # is enumerated on each request, so a new one shows up in the catalog with no # registration step. This interval covers the other direction - mirroring a # table created directly in the database into a seed catalog file - which # otherwise only happens at startup or via POST /api/system/brand-sync. # # A sweep with nothing to do is one COUNT(*) per brand table and writes nothing. # Raise it if the database is remote and the chatter matters; 0 turns the loop # off entirely, leaving only the startup run and the manual endpoint. BRAND_SYNC_INTERVAL_SECONDS=300 # --- Active brands (development working set) -------------------------------- # Comma-separated. BLANK OR UNSET = every brand is active (the production # default). Setting it narrows the catalog, RAG, search, MCP, analytics, # nutrition and every Dagster asset to these brands in one place - nothing is # deleted, the other brand_* tables just stop being discovered. # Names resolve through the brand aliases, so "Tata" activates # brand_hindustan_unilever exactly as ingesting Tata products would. #ACTIVE_BRANDS=Amul,Cadbury,Hindustan Unilever USE_S3=true S3_ACCESS_KEY=your-do-spaces-key S3_SECRET_KEY=your-do-spaces-secret S3_ENDPOINT=https://your-space.region.digitaloceanspaces.com S3_BUCKET=your-bucket S3_REGION=region # Optional - leave USE_GOOGLE_CSE=false if you don't have a Google CSE key USE_GOOGLE_CSE=false GOOGLE_API_KEY= GOOGLE_CSE_ID= USE_DDG_IMAGES=true USE_OPEN_FACTS=true USE_WIKIMEDIA=true USE_PLAYWRIGHT_FALLBACK=true MIN_IMAGE_BYTES=3000 # Product SKU: try a live web search for a real marketplace product ID # (Amazon ASIN, Flipkart PID, etc.) before falling back to an internal SKU. # Set to false to always generate internal SKUs only (faster, offline-safe). ENABLE_SKU_WEB_LOOKUP=true # Deterministic product validation (see app/services/product_validator.py). # Final gate applied to every catalog row before it's kept; rows scoring # below VALIDATION_REJECT_THRESHOLD are dropped (with reasons recorded), # rows below VALIDATION_REVIEW_THRESHOLD are kept but flagged for human review. ENABLE_PRODUCT_VALIDATION=true VALIDATION_REJECT_THRESHOLD=0.35 VALIDATION_REVIEW_THRESHOLD=0.70 # Per-pack-size product images (see docs/PER_VARIANT_IMAGES.md). When true, # each size variant of a product (e.g. 100g/200g/500g) gets its own # size-targeted image search/upload instead of sharing one generic image # set. Falls back to the old shared behaviour automatically whenever no # size-specific match is found, so this is safe to leave on. Set to # "false" to fully restore the old shared-image behaviour. ENABLE_PER_VARIANT_IMAGES=true # Max image candidates fetched per size variant (kept smaller than the # ~24 used for the general product-level search, since a per-size search # only needs a handful of genuinely matching candidates). PER_VARIANT_IMAGE_MAX_RESULTS=10 # Barcode Retrieval & Product Enrichment (see # app/services/enrichment/barcode/ and docs/BARCODE_ENRICHMENT.md). Set to # false to disable barcode lookup entirely (rows are stored with # barcode=NULL, barcode_lookup_status="disabled"). ENABLE_BARCODE_LOOKUP=true BARCODE_LOOKUP_TIMEOUT_SECONDS=10 # 30 days, in seconds BARCODE_LOOKUP_CACHE_TTL_SECONDS=2592000 BARCODE_LOOKUP_MAX_CONCURRENCY=5 BARCODE_COUNTRY_TAG=india # GS1 India tier - leave both blank to skip (no free public API exists at # time of writing; the cascade falls through to Open Food Facts). Fill in # if/when real GS1 India ("Verified by GS1") API access is provisioned. GS1_INDIA_API_BASE_URL= GS1_INDIA_API_KEY= # UPCItemDB - works out of the box on the free trial tier (no key needed, # rate limited ~100 req/day). Set to switch to the paid production endpoint. UPC_DATABASE_API_KEY= # Lowest-trust fallback tier: DuckDuckGo search + manufacturer page text # scan for a labelled barcode (uses the same `ddgs` dependency as SKU # lookup). Set to false to skip this tier. ENABLE_MANUFACTURER_SITE_LOOKUP=true