""" Centralized configuration for the AI Product Catalog + RAG backend. SECURITY NOTE -------------- The previous version of this project had a serious problem: `settings.py` hard-coded a *live* database host, port and password as Python literal fallbacks (`os.getenv("DB_HOST", "")`, etc.). That means the real production credentials shipped inside the source code itself - in every copy, every zip export, and every git commit - regardless of whether a `.env` file was present. This rewrite removes every hard-coded secret. Every credential (DB_PASSWORD, S3 keys, Google API key, ...) is read ONLY from the environment (via a local `.env` file, loaded through python-dotenv, or real OS/container environment variables). Non-secret values (ports, feature flags, model names) keep sane, publicly-safe defaults so the project still boots out of the box for local development. If a secret-shaped variable is required for a feature that is enabled (e.g. `USE_PGVECTOR=true` but no `DB_PASSWORD` set), we raise a clear `RuntimeError` at settings-load time instead of silently connecting with an empty/placeholder password. Fail loudly, not insecurely. """ from __future__ import annotations import os from pathlib import Path try: from dotenv import load_dotenv # backend/.env (one level up from this file: app/infrastructure/settings.py) _env_path = Path(__file__).resolve().parents[2] / ".env" load_dotenv(_env_path) except ImportError: # python-dotenv not installed - fall back to whatever is already in the # process environment (e.g. set by the shell, Docker, systemd, CI, etc.) pass def _bool(name: str, default: str) -> bool: return os.getenv(name, default).strip().lower() in {"1", "true", "yes"} def _require(name: str, *, feature_flag: str) -> str: """Read a required secret. Raises if missing and the owning feature is enabled.""" value = os.getenv(name) if not value: raise RuntimeError( f"Missing required environment variable '{name}'. It is required because " f"'{feature_flag}' is enabled. Set it in backend/.env (copy from " f".env.example) or disable the feature by setting {feature_flag}=false." ) return value # --------------------------------------------------------------------------- # Writable data directories (persistence) # --------------------------------------------------------------------------- # Everything the running app WRITES lives under one of these three paths. They # are settings rather than hard-coded paths because in a container they must be # mounted on a volume - otherwise every product added through the UI and every # retrained model is discarded the next time the image is redeployed. # # DATA_DIR generated catalogs (catalog_engine.save_catalog) # SEED_CATALOG_DIR per-brand JSON catalogs, appended to by # POST /api/user/products/add and /upload-file # MODEL_ARTIFACTS_DIR *.joblib bundles written by the training endpoints # # See BUNDLED_ASSETS_DIR below for how the read-only copies shipped inside the # image get into these directories the first time a volume is mounted. _BACKEND_ROOT = Path(__file__).resolve().parents[2] def _dir(name: str, default: Path) -> Path: raw = os.getenv(name, "").strip() return Path(raw).expanduser() if raw else default DATA_DIR = _dir("DATA_DIR", _BACKEND_ROOT / "data") SEED_CATALOG_DIR = _dir("SEED_CATALOG_DIR", DATA_DIR / "seed_catalogs") MODEL_ARTIFACTS_DIR = _dir( "MODEL_ARTIFACTS_DIR", _BACKEND_ROOT / "app" / "intelligence" / "artifacts" ) # Pristine copies of the bundled seed catalogs and pre-trained models, placed # here by the Dockerfile at a path that is never itself mounted over. # # This exists because the two ways of mounting a volume behave differently, and # the difference is silent. Docker copies the image's content into a *named* # volume the first time it is used, but a *bind* mount starts empty and simply # hides whatever the image had at that path. Mounting a bind mount on /app/data # would therefore leave the app with no seed catalogs at all: the next product # added would write a fresh JSON file containing only that one product, and the # ML endpoints would report no trained models. # # So the image keeps a second, unmounted copy, and restore_bundled_assets() # (app/infrastructure/persistence.py) fills in whatever the writable directory # is missing at startup. Empty/absent outside Docker, where nothing is mounted # and the defaults above already point at the real files. BUNDLED_ASSETS_DIR = _dir("BUNDLED_ASSETS_DIR", Path("/app/.bundled")) # --------------------------------------------------------------------------- # Ollama (local LLM) # --------------------------------------------------------------------------- USE_OLLAMA = _bool("USE_OLLAMA", "true") OLLAMA_BASE_URL = os.getenv("OLLAMA_BASE_URL", "http://localhost:11434") OLLAMA_MODEL_NAME = os.getenv("OLLAMA_MODEL_NAME", "qwen2.5:1.5b") # Per-request generation timeout (seconds). Small CPU-only models on modest # hardware (e.g. 8GB RAM, no GPU) can take a while for longer RAG contexts. OLLAMA_TIMEOUT_SECONDS = int(os.getenv("OLLAMA_TIMEOUT_SECONDS", "120")) # --------------------------------------------------------------------------- # Embeddings (sentence-transformers, CPU-friendly) # --------------------------------------------------------------------------- USE_EMBEDDINGS = _bool("USE_EMBEDDINGS", "true") EMBEDDINGS_MODEL = os.getenv("EMBEDDINGS_MODEL", "sentence-transformers/all-MiniLM-L6-v2") EMBEDDINGS_DIM = int(os.getenv("EMBEDDINGS_DIM", "384")) # --------------------------------------------------------------------------- # Postgres / pgvector # --------------------------------------------------------------------------- USE_PGVECTOR = _bool("USE_PGVECTOR", "true") DB_HOST = os.getenv("DB_HOST", "localhost") DB_PORT = os.getenv("DB_PORT", "5432") DB_NAME = os.getenv("DB_NAME", "pgvector") DB_USER = os.getenv("DB_USER", "postgres") DB_PASSWORD = _require("DB_PASSWORD", feature_flag="USE_PGVECTOR") if USE_PGVECTOR else os.getenv("DB_PASSWORD", "") # How long to wait for the TCP connect before giving up. Matters more than it # looks: a host that DROPS packets (a firewall, a typo'd DB_HOST) otherwise # blocks until the OS timeout - about 130 seconds on Linux - and every request # that touches the database inherits that wait, including /api/health. A short # ceiling turns "the database is unreachable" into a fast, honest error instead # of a hung worker and a container the platform decides is unhealthy. DB_CONNECT_TIMEOUT_SECONDS = int(os.getenv("DB_CONNECT_TIMEOUT_SECONDS", "5")) DATABASE_URL = os.getenv( "DATABASE_URL", f"postgresql://{DB_USER}:{DB_PASSWORD}@{DB_HOST}:{DB_PORT}/{DB_NAME}", ) # How often to re-run the brand-table <-> seed-catalog reconcile after boot. # # The startup run alone only catches what existed at boot. A brand table created # directly in the database while the server is up - by hand, by a script, or by # another machine sharing this database - is not mirrored into the seed catalogs # until the next restart. This interval is what closes that window. # # reconcile_brand_catalogs() is idempotent and non-destructive, so a sweep that # finds nothing to do costs one COUNT(*) per brand table and writes nothing. # 0 disables the loop, leaving the startup run and POST /api/system/brand-sync. BRAND_SYNC_INTERVAL_SECONDS = int(os.getenv("BRAND_SYNC_INTERVAL_SECONDS", "300")) # --------------------------------------------------------------------------- # S3 / DigitalOcean Spaces (product image storage) - optional # --------------------------------------------------------------------------- USE_S3 = _bool("USE_S3", "false") S3_ACCESS_KEY = _require("S3_ACCESS_KEY", feature_flag="USE_S3") if USE_S3 else os.getenv("S3_ACCESS_KEY") S3_SECRET_KEY = _require("S3_SECRET_KEY", feature_flag="USE_S3") if USE_S3 else os.getenv("S3_SECRET_KEY") S3_ENDPOINT = _require("S3_ENDPOINT", feature_flag="USE_S3") if USE_S3 else os.getenv("S3_ENDPOINT") S3_BUCKET = _require("S3_BUCKET", feature_flag="USE_S3") if USE_S3 else os.getenv("S3_BUCKET") S3_REGION = os.getenv("S3_REGION", "sgp1") # --------------------------------------------------------------------------- # Google Custom Search (OPTIONAL image source - leave disabled if unset) # --------------------------------------------------------------------------- USE_GOOGLE_CSE = _bool("USE_GOOGLE_CSE", "false") GOOGLE_API_KEY = _require("GOOGLE_API_KEY", feature_flag="USE_GOOGLE_CSE") if USE_GOOGLE_CSE else os.getenv("GOOGLE_API_KEY") GOOGLE_CSE_ID = _require("GOOGLE_CSE_ID", feature_flag="USE_GOOGLE_CSE") if USE_GOOGLE_CSE else os.getenv("GOOGLE_CSE_ID") # --------------------------------------------------------------------------- # Open-source image sources (no API key needed for any of these three) # --------------------------------------------------------------------------- USE_DDG_IMAGES = _bool("USE_DDG_IMAGES", "true") USE_OPEN_FACTS = _bool("USE_OPEN_FACTS", "true") USE_WIKIMEDIA = _bool("USE_WIKIMEDIA", "true") # Last-resort headless-browser (Python Playwright) image fallback. Requires # `pip install playwright && playwright install chromium`; automatically # skipped (logged once) if that hasn't been done. USE_PLAYWRIGHT_FALLBACK = _bool("USE_PLAYWRIGHT_FALLBACK", "true") # Minimum byte size for a downloaded image to be accepted as "real" (filters # out 1x1 tracking pixels / broken placeholder images) MIN_IMAGE_BYTES = int(os.getenv("MIN_IMAGE_BYTES", "3000")) # --------------------------------------------------------------------------- # Store-catalog enrichment pipeline (app/core/store_catalog_pipeline.py) # --------------------------------------------------------------------------- # Stages 7-10 of the store-Excel ingestion pipeline. Ported from the sibling # Universal_Catalog_Barcode_Enrichment project along with the code that reads # them; the names are kept identical so the ported modules need no edits. # # The three network-touching flags default to FALSE here, unlike in the # sibling. This backend serves an interactive API on a shared 8GB host, and a # 2000-row upload with web lookups on would fire thousands of outbound # requests. Turn them on deliberately, per environment. ENABLE_SKU_WEB_LOOKUP = _bool("ENABLE_SKU_WEB_LOOKUP", "false") ENABLE_BARCODE_LOOKUP = _bool("ENABLE_BARCODE_LOOKUP", "false") ENABLE_MANUFACTURER_SITE_LOOKUP = _bool("ENABLE_MANUFACTURER_SITE_LOOKUP", "false") # Offline/deterministic stages - safe to leave on. ENABLE_HSN_GST_ENRICHMENT = _bool("ENABLE_HSN_GST_ENRICHMENT", "true") ENABLE_PRODUCT_VALIDATION = _bool("ENABLE_PRODUCT_VALIDATION", "true") # Validation gate thresholds: below REJECT the row is dropped, below REVIEW it # is stored but flagged `validation_status="review"`. VALIDATION_REJECT_THRESHOLD = float(os.getenv("VALIDATION_REJECT_THRESHOLD", "0.35")) VALIDATION_REVIEW_THRESHOLD = float(os.getenv("VALIDATION_REVIEW_THRESHOLD", "0.70")) # Pack-size explosion: how many size rows one uploaded product may become. MAX_VARIANTS_PER_PRODUCT = int(os.getenv("MAX_VARIANTS_PER_PRODUCT", "6")) ENABLE_PER_VARIANT_IMAGES = _bool("ENABLE_PER_VARIANT_IMAGES", "false") PER_VARIANT_IMAGE_MAX_RESULTS = int(os.getenv("PER_VARIANT_IMAGE_MAX_RESULTS", "10")) # Barcode lookup tuning. The cache TTL is long (30 days) because a GTIN for a # given pack size does not change, and negative results are cached too. BARCODE_LOOKUP_TIMEOUT_SECONDS = float(os.getenv("BARCODE_LOOKUP_TIMEOUT_SECONDS", "10")) BARCODE_LOOKUP_CACHE_TTL_SECONDS = float(os.getenv("BARCODE_LOOKUP_CACHE_TTL_SECONDS", str(30 * 24 * 3600))) BARCODE_LOOKUP_MAX_CONCURRENCY = int(os.getenv("BARCODE_LOOKUP_MAX_CONCURRENCY", "5")) BARCODE_COUNTRY_TAG = os.getenv("BARCODE_COUNTRY_TAG", "india") # Optional barcode source credentials. Each source disables itself when its # key is blank, so leaving these unset simply narrows the lookup cascade. GS1_INDIA_API_BASE_URL = os.getenv("GS1_INDIA_API_BASE_URL", "") GS1_INDIA_API_KEY = os.getenv("GS1_INDIA_API_KEY", "") UPC_DATABASE_API_KEY = os.getenv("UPC_DATABASE_API_KEY", "") # --------------------------------------------------------------------------- # HTTP client defaults # --------------------------------------------------------------------------- USER_AGENT = os.getenv("USER_AGENT", "CatalogBot/1.0 (+https://example.com)") REQUEST_TIMEOUT_SECONDS = int(os.getenv("REQUEST_TIMEOUT_SECONDS", "20")) # --------------------------------------------------------------------------- # FastAPI / web server # --------------------------------------------------------------------------- API_CORS_ORIGINS = [ origin.strip() for origin in os.getenv("API_CORS_ORIGINS", "http://localhost:5173,http://127.0.0.1:5173").split(",") if origin.strip() ] # --------------------------------------------------------------------------- # Authentication # --------------------------------------------------------------------------- # CORS above is not access control - browsers enforce it, and curl ignores it # entirely. These settings are what actually guards the write/compute endpoints # (catalog generation, ML training, uploads, chat). # # AUTH_ENABLED=false turns every guard off, restoring the old behaviour where # any caller could reach any endpoint. It exists so a fresh checkout still runs # without generating secrets first; app/infrastructure/security.py logs a # warning at import when it is off. Never deploy with it off. AUTH_ENABLED = _bool("AUTH_ENABLED", "true") # Signs and verifies access tokens. Changing it invalidates every issued token, # which is the intended way to force everyone to sign in again. Generate with: # python scripts/make_auth_secrets.py AUTH_SECRET_KEY = ( _require("AUTH_SECRET_KEY", feature_flag="AUTH_ENABLED") if AUTH_ENABLED else os.getenv("AUTH_SECRET_KEY", "") ) # How long an issued token stays valid. 12h by default: long enough that a # working day needs one sign-in, short enough that a leaked token expires. AUTH_TOKEN_TTL_MINUTES = int(os.getenv("AUTH_TOKEN_TTL_MINUTES", "720")) # The interactive accounts. Only PBKDF2 digests are stored - never a password. # `make_auth_secrets.py` prints the lines ready to paste. # # `admin` is required whenever auth is on: without it nobody could sign in. AUTH_ADMIN_USERNAME = os.getenv("AUTH_ADMIN_USERNAME", "admin") AUTH_ADMIN_PASSWORD_HASH = ( _require("AUTH_ADMIN_PASSWORD_HASH", feature_flag="AUTH_ENABLED") if AUTH_ENABLED else os.getenv("AUTH_ADMIN_PASSWORD_HASH", "") ) # The second `user` account is OPTIONAL, and left unset in this deployment. # An empty hash is how the account is switched off: auth.py builds its account # table from these values and omits any entry whose hash is blank, so there is # nothing to sign in to. Setting the hash again re-enables it with no code # change - which is exactly what the test suite does in tests/conftest.py. AUTH_USER_USERNAME = os.getenv("AUTH_USER_USERNAME", "user") AUTH_USER_PASSWORD_HASH = os.getenv("AUTH_USER_PASSWORD_HASH", "") # Failed-login throttle, applied per username+client-IP. Prevents an exposed # login endpoint from being a free password oracle. AUTH_MAX_LOGIN_ATTEMPTS = int(os.getenv("AUTH_MAX_LOGIN_ATTEMPTS", "10")) AUTH_LOCKOUT_SECONDS = int(os.getenv("AUTH_LOCKOUT_SECONDS", "300")) # Local-development escape hatch: accept ANY password at /api/auth/login, so a # developer who does not have the configured passwords to hand can still reach # the admin and user pages. The username still selects the role, and the token # issued is a normal signed one - so every downstream guard, /api/auth/me, and # the React route gating all behave exactly as they do in production. What is # skipped is only the password check. # # This is NOT the same as AUTH_ENABLED=false. That disables every guard *and* # makes /api/auth/login return 503, which breaks the login page outright. This # flag keeps the whole auth machinery running and unlocks just the front door. # # Anyone who can reach the API can sign in as admin while it is on. Keep it # false anywhere the port is reachable by someone you would not hand the admin # password to. AUTH_ALLOW_ANY_LOGIN = _bool("AUTH_ALLOW_ANY_LOGIN", "false") def _parse_api_keys(raw: str) -> dict: """ Parse ``API_KEYS`` - ``name:role:secret`` triples, comma-separated. Keyed by secret because that is what an inbound request presents. One entry per consumer is the point: a shared key cannot be revoked for one caller without breaking all of them. """ parsed: dict = {} for entry in raw.split(","): entry = entry.strip() if not entry: continue parts = entry.split(":") if len(parts) != 3: raise RuntimeError( f"Malformed API_KEYS entry {entry!r}. Expected 'name:role:secret', " f"comma-separated between entries." ) name, role, secret = (p.strip() for p in parts) if role not in {"admin", "user"}: raise RuntimeError( f"API_KEYS entry {name!r} has role {role!r}; expected 'admin' or 'user'." ) if not secret: raise RuntimeError(f"API_KEYS entry {name!r} has an empty secret.") parsed[secret] = (name, role) return parsed # Machine consumers of api.. Empty by default - browser sessions go # through /api/auth/login instead, and a key that nobody needs is only risk. API_KEYS = _parse_api_keys(os.getenv("API_KEYS", "")) # Default RAG behaviour RAG_DEFAULT_TOP_K = int(os.getenv("RAG_DEFAULT_TOP_K", "5")) RAG_MAX_TOP_K = int(os.getenv("RAG_MAX_TOP_K", "15")) RAG_MAX_CONTEXT_CHARS = int(os.getenv("RAG_MAX_CONTEXT_CHARS", "4000")) # Optional pgvector cosine-distance ceiling (0 = identical, 2 = opposite) # used to drop weak semantic matches before they reach the LLM. Left # unset (None) by default since the primary relevance guardrail is the # category-aware retrieval in rag_service.py; set RAG_MAX_DISTANCE (e.g. # "0.9") once you've inspected real distance scores for your embeddings # if you want an extra cutoff on top of that. _raw_max_distance = os.getenv("RAG_MAX_DISTANCE", "").strip() RAG_MAX_DISTANCE = float(_raw_max_distance) if _raw_max_distance else None