393 lines
18 KiB
Plaintext
393 lines
18 KiB
Plaintext
# Copy this file to .env and fill in your own values.
|
|
# Nothing here is a real credential.
|
|
#
|
|
# The host-side values below are for running the backend directly with uvicorn
|
|
# (python run_project.py). When the backend runs INSIDE Docker via the root
|
|
# docker-compose.yml, four of them are overridden by compose and you do not
|
|
# need to change them here:
|
|
#
|
|
# DB_HOST -> postgres (service name)
|
|
# DB_PORT -> 5432
|
|
# DB_PASSWORD -> POSTGRES_PASSWORD from the root .env
|
|
# OLLAMA_BASE_URL -> http://ollama:11434
|
|
#
|
|
# The reason is worth internalising: inside a container, `localhost` is the
|
|
# container itself, not the host and not a sibling container. A container-bound
|
|
# localhost URL points the backend at its own empty ports. Containers reach
|
|
# each other by service name over the compose network instead.
|
|
|
|
# --- Ports (container only) -----------------------------------------------
|
|
# The image serves 3000 and 8000 at once - 3000 because that is what Dokploy
|
|
# routes a domain to, 8000 because the README, the vite dev proxy and
|
|
# docker-compose all use it. Serving both means the container works whichever
|
|
# one the platform points at.
|
|
#
|
|
# PORTS the comma-separated pair to bind. PORT pins a single port instead and
|
|
# takes precedence, e.g. PORT=8080 serves only 8080.
|
|
#
|
|
# Neither affects running uvicorn directly for local development.
|
|
# PORTS=3000,8000
|
|
# PORT=8080
|
|
|
|
# --- Persistence (IMPORTANT in Docker) ------------------------------------
|
|
# The three directories the app WRITES to at runtime. The defaults point inside
|
|
# the repo/image and are right for local development; in a container each one
|
|
# needs a volume mounted on it, or a redeploy throws away everything written
|
|
# since the last build:
|
|
#
|
|
# SEED_CATALOG_DIR products added via POST /api/user/products/add and
|
|
# /upload-file are appended to the JSON files here
|
|
# MODEL_ARTIFACTS_DIR *.joblib bundles written by the training endpoints
|
|
# DATA_DIR catalogs saved by the ingestion pipeline
|
|
#
|
|
# Mount these two paths in Dokploy (SEED_CATALOG_DIR sits inside DATA_DIR, so
|
|
# one mount covers both):
|
|
#
|
|
# /app/data
|
|
# /app/app/intelligence/artifacts
|
|
#
|
|
# Either a named volume or a bind mount works. The image keeps read-only copies
|
|
# of the bundled seed catalogs and pre-trained models at /app/.bundled, and the
|
|
# app restores whatever a freshly-mounted directory is missing on startup
|
|
# without overwriting anything already there - so a bind mount, which starts
|
|
# empty and would otherwise hide them, is safe.
|
|
#
|
|
# Leave all three unset unless the writable data belongs somewhere else.
|
|
# DATA_DIR=/app/data
|
|
# SEED_CATALOG_DIR=/app/data/seed_catalogs
|
|
# MODEL_ARTIFACTS_DIR=/app/app/intelligence/artifacts
|
|
|
|
# --- CORS (REQUIRED when the API is on its own domain) --------------------
|
|
# Comma-separated list of the exact browser origins allowed to call this API.
|
|
# Scheme and host both matter; no trailing slash, and no wildcard - the
|
|
# frontend sends an Authorization header, and browsers refuse a credentialed
|
|
# cross-origin request whose Allow-Origin is "*" (app/main.py logs and turns
|
|
# credentials off if it sees one, so a wildcard silently breaks every call).
|
|
#
|
|
# Production - the React app is served from catalogue.nearle.ai.in and calls
|
|
# the API at mcp.nearle.ai.in, so that frontend origin must be listed:
|
|
#
|
|
# API_CORS_ORIGINS=https://catalogue.nearle.ai.in
|
|
#
|
|
# Add http://localhost:5173 alongside it if you point a local Vite dev server
|
|
# at the deployed API. Server-to-server callers (MCP clients, scripts) are not
|
|
# affected by any of this - CORS is a browser rule; they use X-API-Key.
|
|
API_CORS_ORIGINS=http://localhost:5173,http://127.0.0.1:5173
|
|
|
|
# --- Authentication (REQUIRED) -------------------------------------------
|
|
# The backend will not start without these while AUTH_ENABLED=true. Generate
|
|
# all four lines, plus sign-in passwords, with:
|
|
#
|
|
# python scripts/make_auth_secrets.py
|
|
#
|
|
# They guard the 18 write/compute endpoints - catalog generation, ML training,
|
|
# the upload endpoints and chat. CORS is not a substitute: browsers enforce it,
|
|
# curl ignores it entirely.
|
|
#
|
|
# AUTH_ENABLED=false turns every guard off and makes the whole API open again.
|
|
# It exists so a fresh checkout runs before you have generated secrets. Never
|
|
# set it false on a host reachable from the internet.
|
|
AUTH_ENABLED=true
|
|
|
|
# Signs access tokens. Changing it signs everybody out, which is how you revoke
|
|
# every issued token at once. Use a DIFFERENT value in production from the one
|
|
# on your laptop - a secret that has been on a dev machine is not a secret.
|
|
AUTH_SECRET_KEY=
|
|
|
|
# Only PBKDF2 digests are stored, never passwords. `make_auth_secrets.py`
|
|
# prints the password once and the hash to paste here; it cannot be reversed,
|
|
# so rerun the script to change a password.
|
|
AUTH_ADMIN_USERNAME=admin
|
|
AUTH_ADMIN_PASSWORD_HASH=
|
|
AUTH_USER_USERNAME=user
|
|
AUTH_USER_PASSWORD_HASH=
|
|
|
|
# Token lifetime in minutes. 12h by default: one sign-in per working day, and
|
|
# a leaked token expires by itself.
|
|
AUTH_TOKEN_TTL_MINUTES=720
|
|
|
|
# Failed-login throttle, per username+IP. Stops the login endpoint being an
|
|
# unlimited password oracle once it is on the internet.
|
|
AUTH_MAX_LOGIN_ATTEMPTS=10
|
|
AUTH_LOCKOUT_SECONDS=300
|
|
|
|
# Local development only: accept ANY password at /api/auth/login. The username
|
|
# still picks the role (`admin` -> admin pages, `user` -> user pages), and the
|
|
# token issued is a normal one, so every other guard behaves normally. Unlike
|
|
# AUTH_ENABLED=false it leaves the login page working - it just stops checking
|
|
# the password. Anyone who can reach the port becomes admin: keep it false here.
|
|
AUTH_ALLOW_ANY_LOGIN=true
|
|
|
|
# Machine consumers of api.<domain> - scripts, partner integrations, your own
|
|
# backends. Format: name:role:secret, comma-separated. Role is one of:
|
|
#
|
|
# admin everything, including system/init and model training
|
|
# user catalog write access (add products, upload inventory)
|
|
# uploader ONE verb: POST /api/uploads/catalog. Nothing else - it cannot
|
|
# read the catalog, cannot see another caller's submissions, and
|
|
# cannot cancel or resume anything. This is the role to issue to
|
|
# an outside party who needs to send you spreadsheets.
|
|
#
|
|
# Callers send the secret as an X-API-Key header.
|
|
#
|
|
# One entry per consumer, always: a shared key cannot be revoked for one caller
|
|
# without breaking every other. Mint them with:
|
|
# python scripts/make_auth_secrets.py --api-key partner-x:user
|
|
# python scripts/make_auth_secrets.py --api-key catalog-drop:uploader
|
|
#
|
|
# NAME THE KEY FOR ITS FUNCTION, NOT THE PERSON HOLDING IT. /api/health is
|
|
# public and reports {name, role, fingerprint} for every configured key. The
|
|
# secret is never exposed, but the NAME is - so `catalog-drop:uploader:...` is
|
|
# right and `priya-laptop:uploader:...` publishes a colleague's name to anyone
|
|
# who curls the health endpoint.
|
|
#
|
|
# Leave empty if only the web app calls the API - it signs in through
|
|
# /api/auth/login instead, and a key nobody needs is only risk.
|
|
API_KEYS=
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Catalog batch ingestion
|
|
# ---------------------------------------------------------------------------
|
|
# Used by both spreadsheet ingestion routes - the admin one
|
|
# (POST /api/admin/catalog-batch/ingest) and the API-client one
|
|
# (POST /api/uploads/catalog). Every ceiling is enforced in the application,
|
|
# not at the proxy: in production the caller reaches mcp.nearle.ai.in directly,
|
|
# so neither nginx's client_max_body_size nor Caddy's request_body cap is in
|
|
# front of these endpoints.
|
|
#
|
|
# Per-file limits are fixed in code at 10MB / 2000 rows, matching the
|
|
# single-file upload path. These bound the BATCH on top of that.
|
|
BATCH_MAX_FILES=20
|
|
BATCH_MAX_TOTAL_BYTES=52428800
|
|
BATCH_MAX_TOTAL_ROWS=20000
|
|
|
|
# Where staged uploads live. Under DATA_DIR because that path is already a
|
|
# declared volume, which is what lets a batch survive a container restart.
|
|
# BATCH_UPLOAD_DIR=/app/data/batch_uploads
|
|
|
|
# Batches allowed to wait behind the one running. One worker thread runs a
|
|
# single batch at a time; past this depth the endpoints answer 429 rather than
|
|
# accepting work they have no intention of starting soon. This queue is the
|
|
# only thing bounding what an uploader key can cost in CPU - raise it with care
|
|
# on a one-vCPU host.
|
|
BATCH_QUEUE_MAX=4
|
|
|
|
# Staged files are deleted this many days after the batch was created.
|
|
BATCH_RETENTION_DAYS=7
|
|
|
|
# Deliberately false. A batch a restart cut short is marked "interrupted" and
|
|
# waits for someone to press Resume. Auto-resuming means a container stuck in a
|
|
# restart loop re-runs the heaviest work in the app on every boot, which is how
|
|
# a slow start becomes an unrecoverable spiral.
|
|
#
|
|
# Worth knowing alongside UPLOAD_AUTORUN below: with uploads running unattended,
|
|
# a redeploy in the middle of one leaves that batch "interrupted" and waiting
|
|
# for a human. It is the one place manual intervention comes back.
|
|
BATCH_AUTO_RESUME=false
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Unattended ingestion
|
|
# ---------------------------------------------------------------------------
|
|
# Whether POST /api/uploads/catalog runs the pipeline on arrival (true) or parks
|
|
# the files in the admin review inbox for someone to start by hand (false).
|
|
#
|
|
# READ THIS BEFORE CHANGING IT. That endpoint takes NO credential - it was
|
|
# opened on purpose so colleagues could send spreadsheets without one being
|
|
# issued to them. With autorun on, "anyone who can reach this host" and "anyone
|
|
# who can write to the live catalogue" are the same set of people, and an ingest
|
|
# is an upsert with no undo.
|
|
#
|
|
# What still bounds it is throughput, not identity: the per-request ceilings
|
|
# above, and BATCH_QUEUE_MAX behind a single worker. A sender can occupy the
|
|
# ingestion worker; they cannot multiply it.
|
|
#
|
|
# Set false and the review inbox comes back with no code change - the INBOX_*
|
|
# ceilings below apply only on that path.
|
|
UPLOAD_AUTORUN=true
|
|
|
|
# How an auto-started run behaves. Deliberately NOT accepted from the request:
|
|
# the sender is anonymous, and letting an anonymous caller switch on the
|
|
# expensive outbound stages is the one thing this endpoint must not allow.
|
|
#
|
|
# Images on, because a product landing without one is the failure this endpoint
|
|
# exists to avoid. Stage 6 is the slowest stage and reaches the network, but
|
|
# only one batch runs at a time, so nothing else competes with it.
|
|
#
|
|
# LLM on, though in production it currently does nothing: USE_OLLAMA is false
|
|
# there, so the LLM call returns immediately without a request and the row keeps
|
|
# its blank description. Set true so the pipeline is already right for the day
|
|
# an Ollama server is reachable.
|
|
#
|
|
# If you DO set USE_OLLAMA=true, make sure something is actually listening on
|
|
# OLLAMA_BASE_URL. An unreachable server costs a 5s probe, and while that probe
|
|
# is cached per 30s rather than paid per row, a reachable-but-slow model is
|
|
# billed per row at OLLAMA_TIMEOUT_SECONDS.
|
|
UPLOAD_AUTORUN_FETCH_IMAGES=true
|
|
UPLOAD_AUTORUN_USE_LLM=true
|
|
|
|
USE_OLLAMA=true
|
|
OLLAMA_BASE_URL=http://localhost:11434
|
|
OLLAMA_MODEL_NAME=qwen2.5:1.5b
|
|
|
|
USE_EMBEDDINGS=true
|
|
EMBEDDINGS_MODEL=sentence-transformers/all-MiniLM-L6-v2
|
|
|
|
USE_PGVECTOR=true
|
|
DB_HOST=localhost
|
|
DB_PORT=5432
|
|
DB_NAME=pgvector
|
|
DB_USER=postgres
|
|
DB_PASSWORD=changeme
|
|
|
|
# How often (seconds) to reconcile brand_* tables against data/seed_catalogs/.
|
|
#
|
|
# Brands are discovered from the database, not from a list: every brand_* table
|
|
# is enumerated on each request, so a new one shows up in the catalog with no
|
|
# registration step. This interval covers the other direction - mirroring a
|
|
# table created directly in the database into a seed catalog file - which
|
|
# otherwise only happens at startup or via POST /api/system/brand-sync.
|
|
#
|
|
# A sweep with nothing to do is one COUNT(*) per brand table and writes nothing.
|
|
# Raise it if the database is remote and the chatter matters; 0 turns the loop
|
|
# off entirely, leaving only the startup run and the manual endpoint.
|
|
BRAND_SYNC_INTERVAL_SECONDS=300
|
|
|
|
# --- Active brands (development working set) --------------------------------
|
|
# Comma-separated. BLANK OR UNSET = every brand is active (the production
|
|
# default). Setting it narrows the catalog, RAG, search, MCP, analytics,
|
|
# nutrition and every Dagster asset to these brands in one place - nothing is
|
|
# deleted, the other brand_* tables just stop being discovered.
|
|
# Names resolve through the brand aliases, so "Tata" activates
|
|
# brand_hindustan_unilever exactly as ingesting Tata products would.
|
|
# NOTE: this is a whitelist, so anything omitted is invisible in browse, search,
|
|
# suggestions, chat and the MCP tools. If you narrow it, include "Own Products"
|
|
# - that is the bucket unbranded commodities (dal, sugar, salt, spices) are
|
|
# filed under, and it is a normal brand table as far as every read path is
|
|
# concerned.
|
|
#ACTIVE_BRANDS=Amul,Cadbury,Hindustan Unilever,Own Products
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Brand discovery (a brand NAME -> the 11-stage pipeline)
|
|
# ---------------------------------------------------------------------------
|
|
# POST /api/admin/brand-discovery/preview finds a brand's products, and
|
|
# /ingest stages the ones an admin approved as an ordinary catalog batch.
|
|
# Every value below has a working default; none of these need to be set.
|
|
#
|
|
# Open Food Facts is the primary source and the language model is the
|
|
# supplement. OFF returns real products with real barcodes and pack sizes;
|
|
# the default OLLAMA_MODEL_NAME (qwen2.5:1.5b) will invent plausible ones, and
|
|
# nothing downstream can tell a well-formed fiction from a real product. Turn
|
|
# BRAND_DISCOVERY_USE_OFF off and the result rests on the model alone.
|
|
#
|
|
# NOTE: discovering a brand that is not in ACTIVE_BRANDS writes a complete
|
|
# catalog that no endpoint can read. The ingest route refuses with a 409 and
|
|
# names the line to add here; it is a config change plus a restart, never a
|
|
# re-ingest.
|
|
#BRAND_DISCOVERY_USE_OFF=true
|
|
#BRAND_DISCOVERY_USE_LLM=true
|
|
#BRAND_DISCOVERY_MAX_PRODUCTS=200
|
|
# Pack sizes kept per product when only the language model offers any. Stage 6
|
|
# runs an image search per exploded row, so this multiplies the slowest stage.
|
|
#BRAND_DISCOVERY_MAX_SIZES=3
|
|
# Wall-clock ceiling on the language-model half, checked between prompts. Open
|
|
# Food Facts runs first and is never subject to it.
|
|
#BRAND_DISCOVERY_DEADLINE_SECONDS=300
|
|
|
|
|
|
USE_S3=true
|
|
S3_ACCESS_KEY=your-do-spaces-key
|
|
S3_SECRET_KEY=your-do-spaces-secret
|
|
S3_ENDPOINT=https://your-space.region.digitaloceanspaces.com
|
|
S3_BUCKET=your-bucket
|
|
S3_REGION=region
|
|
|
|
# Optional - leave USE_GOOGLE_CSE=false if you don't have a Google CSE key
|
|
USE_GOOGLE_CSE=false
|
|
GOOGLE_API_KEY=
|
|
GOOGLE_CSE_ID=
|
|
|
|
USE_DDG_IMAGES=true
|
|
USE_OPEN_FACTS=true
|
|
USE_WIKIMEDIA=true
|
|
USE_PLAYWRIGHT_FALLBACK=true
|
|
|
|
MIN_IMAGE_BYTES=3000
|
|
|
|
# img_vector: a MobileNetV3-Small embedding (1024 floats, L2-normalised) of
|
|
# each product's primary image, stored on its brand table and computed on a
|
|
# background thread after every catalog write (app/services/image_vector.py,
|
|
# model in app/services/image_embedder.py). Existing rows are filled by
|
|
# `python -m scripts.backfill_image_vectors --all --apply`. Set false to stop
|
|
# the API process from downloading images at all.
|
|
ENABLE_IMAGE_VECTORS=true
|
|
# Where the .tflite lives. The default is inside app/ (shipped with the image
|
|
# and never volume-mounted); only override to point at a different file.
|
|
#IMAGE_EMBED_MODEL_PATH=/app/app/services/models/mobilenet/mobilenet_v3_small_embedder.tflite
|
|
#IMAGE_EMBED_NUM_THREADS=2
|
|
|
|
# Product SKU: try a live web search for a real marketplace product ID
|
|
# (Amazon ASIN, Flipkart PID, etc.) before falling back to an internal SKU.
|
|
# Set to false to always generate internal SKUs only (faster, offline-safe).
|
|
ENABLE_SKU_WEB_LOOKUP=true
|
|
|
|
# Deterministic product validation (see app/services/product_validator.py).
|
|
# Final gate applied to every catalog row before it's kept; rows scoring
|
|
# below VALIDATION_REJECT_THRESHOLD are dropped (with reasons recorded),
|
|
# rows below VALIDATION_REVIEW_THRESHOLD are kept but flagged for human review.
|
|
ENABLE_PRODUCT_VALIDATION=true
|
|
VALIDATION_REJECT_THRESHOLD=0.35
|
|
VALIDATION_REVIEW_THRESHOLD=0.70
|
|
|
|
# Per-pack-size product images (see docs/PER_VARIANT_IMAGES.md). When true,
|
|
# each size variant of a product (e.g. 100g/200g/500g) gets its own
|
|
# size-targeted image search/upload instead of sharing one generic image
|
|
# set. Falls back to the old shared behaviour automatically whenever no
|
|
# size-specific match is found, so this is safe to leave on. Set to
|
|
# "false" to fully restore the old shared-image behaviour.
|
|
ENABLE_PER_VARIANT_IMAGES=true
|
|
# Max image candidates fetched per size variant (kept smaller than the
|
|
# ~24 used for the general product-level search, since a per-size search
|
|
# only needs a handful of genuinely matching candidates).
|
|
PER_VARIANT_IMAGE_MAX_RESULTS=10
|
|
|
|
# Barcode Retrieval & Product Enrichment (see
|
|
# app/services/enrichment/barcode/ and docs/BARCODE_ENRICHMENT.md).
|
|
#
|
|
# INLINE, per product, during ingestion. Defaults FALSE in settings.py and this
|
|
# file used to ship `true`, which disagreed with the code for as long as both
|
|
# existed - anyone copying .env.example got a very different pipeline from
|
|
# anyone relying on the defaults. It is false here now to match.
|
|
#
|
|
# Leave it false unless you know the upload is small: it costs one search
|
|
# request per product against an endpoint capped at 10 requests/minute, so a
|
|
# 200-row sheet is twenty minutes of held request. ENRICH_BARCODES_ON_UPLOAD
|
|
# below is the cheap path and is on by default.
|
|
ENABLE_BARCODE_LOOKUP=false
|
|
|
|
# BULK, per brand, after the upload settles. Fetches each brand's whole Open
|
|
# Food Facts catalogue (~5 requests) and matches offline, on the enrichment
|
|
# job's own thread. Runs before the nutrition phase, because a barcode turns a
|
|
# 0.32-confidence name lookup into a 0.95-confidence exact one.
|
|
ENRICH_BARCODES_ON_UPLOAD=true
|
|
BARCODE_LOOKUP_TIMEOUT_SECONDS=10
|
|
# 30 days, in seconds
|
|
BARCODE_LOOKUP_CACHE_TTL_SECONDS=2592000
|
|
BARCODE_LOOKUP_MAX_CONCURRENCY=5
|
|
BARCODE_COUNTRY_TAG=india
|
|
|
|
# GS1 India tier - leave both blank to skip (no free public API exists at
|
|
# time of writing; the cascade falls through to Open Food Facts). Fill in
|
|
# if/when real GS1 India ("Verified by GS1") API access is provisioned.
|
|
GS1_INDIA_API_BASE_URL=
|
|
GS1_INDIA_API_KEY=
|
|
|
|
# UPCItemDB - works out of the box on the free trial tier (no key needed,
|
|
# rate limited ~100 req/day). Set to switch to the paid production endpoint.
|
|
UPC_DATABASE_API_KEY=
|
|
|
|
# Lowest-trust fallback tier: DuckDuckGo search + manufacturer page text
|
|
# scan for a labelled barcode (uses the same `ddgs` dependency as SKU
|
|
# lookup). Set to false to skip this tier.
|
|
ENABLE_MANUFACTURER_SITE_LOOKUP=true
|