735 lines
39 KiB
Python
735 lines
39 KiB
Python
"""
|
|
Centralized configuration for the AI Product Catalog + RAG backend.
|
|
|
|
SECURITY NOTE
|
|
--------------
|
|
The previous version of this project had a serious problem: `settings.py`
|
|
hard-coded a *live* database host, port and password as Python literal
|
|
fallbacks (`os.getenv("DB_HOST", "<real ip>")`, etc.). That means the
|
|
real production credentials shipped inside the source code itself - in
|
|
every copy, every zip export, and every git commit - regardless of
|
|
whether a `.env` file was present.
|
|
|
|
This rewrite removes every hard-coded secret. Every credential
|
|
(DB_PASSWORD, S3 keys, Google API key, ...) is read ONLY from the
|
|
environment (via a local `.env` file, loaded through python-dotenv, or
|
|
real OS/container environment variables). Non-secret values (ports,
|
|
feature flags, model names) keep sane, publicly-safe defaults so the
|
|
project still boots out of the box for local development.
|
|
|
|
If a secret-shaped variable is required for a feature that is enabled
|
|
(e.g. `USE_PGVECTOR=true` but no `DB_PASSWORD` set), we raise a clear
|
|
`RuntimeError` at settings-load time instead of silently connecting
|
|
with an empty/placeholder password. Fail loudly, not insecurely.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import os
|
|
from pathlib import Path
|
|
|
|
# Snapshotted BEFORE load_dotenv, and that ordering is the entire point.
|
|
# load_dotenv() is called without override=True, so a variable already in the
|
|
# process environment silently beats the .env file and keeps beating it no
|
|
# matter how many times the file is corrected. That is not hypothetical here:
|
|
# the deployment platform injects its Environment tab into the container, so a
|
|
# stale value left in that tab overrides the credentials baked into the image
|
|
# (backend/Dockerfile copies .env.production to /app/.env) and the only symptom
|
|
# is a 401 that nothing explains. Comparing a name against this set answers
|
|
# "which of the two won?" - see config_source() below.
|
|
_PREEXISTING_ENV = frozenset(os.environ)
|
|
|
|
try:
|
|
from dotenv import load_dotenv
|
|
|
|
# backend/.env (one level up from this file: app/infrastructure/settings.py)
|
|
_env_path = Path(__file__).resolve().parents[2] / ".env"
|
|
load_dotenv(_env_path)
|
|
except ImportError:
|
|
# python-dotenv not installed - fall back to whatever is already in the
|
|
# process environment (e.g. set by the shell, Docker, systemd, CI, etc.)
|
|
pass
|
|
|
|
|
|
# Names whose raw value arrived wrapped in quotes or padded with whitespace.
|
|
# Recorded rather than merely fixed: stripping keeps the login working, but the
|
|
# only place the original shape is still visible is right here, before the value
|
|
# is normalised. A quoted hash is the signature of a value pasted into a web
|
|
# form, so surfacing it at startup is what stops the next person rediscovering
|
|
# it from a 401. See DB_PASSWORD in .env.production for the counter-case where
|
|
# the quotes ARE part of the secret - which is why this warns, and does not fail.
|
|
_ENV_NEEDED_CLEANUP = set()
|
|
|
|
|
|
def _clean(name: str, raw: str) -> str:
|
|
"""Strip surrounding quotes/whitespace off an env value, remembering if it mattered."""
|
|
cleaned = raw.strip().strip("'\"")
|
|
if cleaned != raw:
|
|
_ENV_NEEDED_CLEANUP.add(name)
|
|
return cleaned
|
|
|
|
|
|
def cleaned_env_names() -> list:
|
|
"""Which settings needed quote/whitespace stripping. Reported at startup."""
|
|
return sorted(_ENV_NEEDED_CLEANUP)
|
|
|
|
|
|
def config_source(name: str) -> str:
|
|
"""
|
|
Where a setting's value actually came from: the process environment, the
|
|
.env file, or this module's own default.
|
|
|
|
Reported at startup for the AUTH_* values (see app/main.py) so that an
|
|
override arriving from outside the image is visible in the logs instead of
|
|
being inferred from a failing login.
|
|
"""
|
|
if name in _PREEXISTING_ENV:
|
|
return "process-env"
|
|
if name in os.environ:
|
|
return "env-file"
|
|
return "default"
|
|
|
|
|
|
def _bool(name: str, default: str) -> bool:
|
|
return os.getenv(name, default).strip().lower() in {"1", "true", "yes"}
|
|
|
|
|
|
def _require(name: str, *, feature_flag: str) -> str:
|
|
"""Read a required secret. Raises if missing and the owning feature is enabled."""
|
|
value = os.getenv(name)
|
|
if not value:
|
|
raise RuntimeError(
|
|
f"Missing required environment variable '{name}'. It is required because "
|
|
f"'{feature_flag}' is enabled. Set it in backend/.env (copy from "
|
|
f".env.example) or disable the feature by setting {feature_flag}=false."
|
|
)
|
|
return value
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Writable data directories (persistence)
|
|
# ---------------------------------------------------------------------------
|
|
# Everything the running app WRITES lives under one of these three paths. They
|
|
# are settings rather than hard-coded paths because in a container they must be
|
|
# mounted on a volume - otherwise every product added through the UI and every
|
|
# retrained model is discarded the next time the image is redeployed.
|
|
#
|
|
# DATA_DIR generated catalogs (catalog_engine.save_catalog)
|
|
# SEED_CATALOG_DIR per-brand JSON catalogs, appended to by
|
|
# POST /api/user/products/add and /upload-file
|
|
# MODEL_ARTIFACTS_DIR *.joblib bundles written by the training endpoints
|
|
#
|
|
# See BUNDLED_ASSETS_DIR below for how the read-only copies shipped inside the
|
|
# image get into these directories the first time a volume is mounted.
|
|
_BACKEND_ROOT = Path(__file__).resolve().parents[2]
|
|
|
|
|
|
def _dir(name: str, default: Path) -> Path:
|
|
raw = os.getenv(name, "").strip()
|
|
return Path(raw).expanduser() if raw else default
|
|
|
|
|
|
DATA_DIR = _dir("DATA_DIR", _BACKEND_ROOT / "data")
|
|
SEED_CATALOG_DIR = _dir("SEED_CATALOG_DIR", DATA_DIR / "seed_catalogs")
|
|
MODEL_ARTIFACTS_DIR = _dir(
|
|
"MODEL_ARTIFACTS_DIR", _BACKEND_ROOT / "app" / "intelligence" / "artifacts"
|
|
)
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Batch catalog ingestion (multi-file upload -> the 11-stage pipeline)
|
|
# ---------------------------------------------------------------------------
|
|
# Staged uploads live under DATA_DIR because that path is already a declared
|
|
# volume (backend/Dockerfile). A batch that survives a container restart is the
|
|
# whole point of writing the files down instead of holding them in the worker
|
|
# thread the way the single-file path does.
|
|
#
|
|
# Every ceiling below is enforced in the application, not at the proxy. In
|
|
# production the browser calls mcp.nearle.ai.in directly, so neither nginx's
|
|
# client_max_body_size nor Caddy's request_body cap is in front of these
|
|
# endpoints - whatever Traefik defaults to is, and it is not ours to rely on.
|
|
BATCH_UPLOAD_DIR = _dir("BATCH_UPLOAD_DIR", DATA_DIR / "batch_uploads")
|
|
|
|
# Per-file limits stay at the single-upload values, 10MB / 2000 rows. They are
|
|
# NOT settings and are not read from here: each router declares its own
|
|
# MAX_UPLOAD_BYTES / MAX_UPLOAD_ROWS pair with the same two numbers - see
|
|
# routers/batch_catalog.py, routers/uploads.py and routers/user_products.py.
|
|
# Change one and you have changed one. The ceilings below bound the BATCH on
|
|
# top of whichever per-file pair applied.
|
|
BATCH_MAX_FILES = int(os.getenv("BATCH_MAX_FILES", "20"))
|
|
BATCH_MAX_TOTAL_BYTES = int(os.getenv("BATCH_MAX_TOTAL_BYTES", str(50 * 1024 * 1024)))
|
|
BATCH_MAX_TOTAL_ROWS = int(os.getenv("BATCH_MAX_TOTAL_ROWS", "20000"))
|
|
|
|
# Batches waiting behind the one running. Past this the endpoint returns 429
|
|
# rather than accepting work it has no intention of starting soon.
|
|
BATCH_QUEUE_MAX = int(os.getenv("BATCH_QUEUE_MAX", "4"))
|
|
|
|
# Staged files are deleted this many days after the batch was created. Without
|
|
# this the upload directory only grows, on a host whose disk is the scarcest
|
|
# resource it has.
|
|
BATCH_RETENTION_DAYS = int(os.getenv("BATCH_RETENTION_DAYS", "7"))
|
|
|
|
# --- Unattended ingestion --------------------------------------------------
|
|
# Whether POST /api/uploads/catalog runs the pipeline on arrival, or parks the
|
|
# files in the admin review inbox for someone to start by hand.
|
|
#
|
|
# READ THIS BEFORE CHANGING IT. That endpoint takes NO credential - it was
|
|
# opened deliberately so colleagues could send spreadsheets without one being
|
|
# issued to them. With autorun on, "anyone who can reach this host" and "anyone
|
|
# who can write to the live catalogue" become the same set of people, and an
|
|
# ingest is an upsert with no undo. That trade was made knowingly: the ask was
|
|
# for uploads to run without manual intervention, and a review queue that needs
|
|
# an admin to press a button is not that.
|
|
#
|
|
# What still bounds it: the per-request ceilings above (20 files / 50MB / 20k
|
|
# rows), and BATCH_QUEUE_MAX behind a single worker thread - so a sender can
|
|
# occupy the ingestion worker but cannot multiply it. Those cap throughput, not
|
|
# who. If that stops being an acceptable trade, set this to false and the review
|
|
# inbox comes back with no code change; everything it needs is still here.
|
|
UPLOAD_AUTORUN = _bool("UPLOAD_AUTORUN", "true")
|
|
|
|
# How an auto-started run behaves. Not accepted from the request: the sender is
|
|
# anonymous, and letting an anonymous caller turn on the expensive stages is the
|
|
# one thing the open endpoint must not allow.
|
|
#
|
|
# Images ON, because a product landing without one is the failure this endpoint
|
|
# exists to avoid - stage 6 is the slowest stage and reaches the network, but
|
|
# only one batch runs at a time so nothing else is competing with it.
|
|
#
|
|
# LLM ON. `use_llm` gates only description generation in stage_2_row_intake, and
|
|
# in production it is currently a no-op: USE_OLLAMA is false there, so
|
|
# ollama_service._ensure_client() returns on its first line without a request
|
|
# and the row simply keeps its blank description. It is set true so the pipeline
|
|
# is already configured correctly for the day an Ollama server exists.
|
|
#
|
|
# This default USED to be false, on the grounds that turning it on "costs a
|
|
# connection timeout per row". That was true, and it was about the OTHER branch
|
|
# of _ensure_client - USE_OLLAMA=true with nothing listening, which is any
|
|
# developer machine that has not run `ollama serve`. It is answered now by the
|
|
# TTL cache on that probe rather than by leaving the feature off: one probe per
|
|
# batch instead of one per row. Do not remove that cache and this default
|
|
# together without re-reading why both exist.
|
|
UPLOAD_AUTORUN_FETCH_IMAGES = _bool("UPLOAD_AUTORUN_FETCH_IMAGES", "true")
|
|
UPLOAD_AUTORUN_USE_LLM = _bool("UPLOAD_AUTORUN_USE_LLM", "true")
|
|
|
|
# --- Review inbox ----------------------------------------------------------
|
|
# The bound that applies only when UPLOAD_AUTORUN is false. Files then wait in
|
|
# the admin review inbox rather than being queued, which removes BATCH_QUEUE_MAX
|
|
# as the bound on that endpoint and leaves the volume as the only thing an
|
|
# anonymous sender can exhaust. These are that bound; past either, the endpoint
|
|
# answers 429 and stages nothing.
|
|
#
|
|
# Both count only files still AWAITING review. Starting or dismissing a drop
|
|
# releases its share immediately, and BATCH_RETENTION_DAYS reclaims whatever
|
|
# nobody ever looks at.
|
|
INBOX_MAX_PENDING_FILES = int(os.getenv("INBOX_MAX_PENDING_FILES", "200"))
|
|
INBOX_MAX_PENDING_BYTES = int(
|
|
os.getenv("INBOX_MAX_PENDING_BYTES", str(200 * 1024 * 1024))
|
|
)
|
|
|
|
# Deliberately false. A batch interrupted by a restart is marked "interrupted"
|
|
# and waits for someone to press Resume. Auto-resuming would mean a container
|
|
# stuck in a restart loop re-runs the heaviest work in the app on every boot,
|
|
# which is precisely how a slow start turns into an unrecoverable spiral.
|
|
BATCH_AUTO_RESUME = _bool("BATCH_AUTO_RESUME", "false")
|
|
|
|
# Pristine copies of the bundled seed catalogs and pre-trained models, placed
|
|
# here by the Dockerfile at a path that is never itself mounted over.
|
|
#
|
|
# This exists because the two ways of mounting a volume behave differently, and
|
|
# the difference is silent. Docker copies the image's content into a *named*
|
|
# volume the first time it is used, but a *bind* mount starts empty and simply
|
|
# hides whatever the image had at that path. Mounting a bind mount on /app/data
|
|
# would therefore leave the app with no seed catalogs at all: the next product
|
|
# added would write a fresh JSON file containing only that one product, and the
|
|
# ML endpoints would report no trained models.
|
|
#
|
|
# So the image keeps a second, unmounted copy, and restore_bundled_assets()
|
|
# (app/infrastructure/persistence.py) fills in whatever the writable directory
|
|
# is missing at startup. Empty/absent outside Docker, where nothing is mounted
|
|
# and the defaults above already point at the real files.
|
|
BUNDLED_ASSETS_DIR = _dir("BUNDLED_ASSETS_DIR", Path("/app/.bundled"))
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Ollama (local LLM)
|
|
# ---------------------------------------------------------------------------
|
|
USE_OLLAMA = _bool("USE_OLLAMA", "true")
|
|
OLLAMA_BASE_URL = os.getenv("OLLAMA_BASE_URL", "http://localhost:11434")
|
|
OLLAMA_MODEL_NAME = os.getenv("OLLAMA_MODEL_NAME", "qwen2.5:1.5b")
|
|
# Per-request generation timeout (seconds). Small CPU-only models on modest
|
|
# hardware (e.g. 8GB RAM, no GPU) can take a while for longer RAG contexts.
|
|
OLLAMA_TIMEOUT_SECONDS = int(os.getenv("OLLAMA_TIMEOUT_SECONDS", "120"))
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Embeddings (sentence-transformers, CPU-friendly)
|
|
# ---------------------------------------------------------------------------
|
|
USE_EMBEDDINGS = _bool("USE_EMBEDDINGS", "true")
|
|
EMBEDDINGS_MODEL = os.getenv("EMBEDDINGS_MODEL", "sentence-transformers/all-MiniLM-L6-v2")
|
|
EMBEDDINGS_DIM = int(os.getenv("EMBEDDINGS_DIM", "384"))
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Postgres / pgvector
|
|
# ---------------------------------------------------------------------------
|
|
USE_PGVECTOR = _bool("USE_PGVECTOR", "true")
|
|
DB_HOST = os.getenv("DB_HOST", "localhost")
|
|
DB_PORT = os.getenv("DB_PORT", "5432")
|
|
DB_NAME = os.getenv("DB_NAME", "pgvector")
|
|
DB_USER = os.getenv("DB_USER", "postgres")
|
|
DB_PASSWORD = _require("DB_PASSWORD", feature_flag="USE_PGVECTOR") if USE_PGVECTOR else os.getenv("DB_PASSWORD", "")
|
|
# How long to wait for the TCP connect before giving up. Matters more than it
|
|
# looks: a host that DROPS packets (a firewall, a typo'd DB_HOST) otherwise
|
|
# blocks until the OS timeout - about 130 seconds on Linux - and every request
|
|
# that touches the database inherits that wait, including /api/health. A short
|
|
# ceiling turns "the database is unreachable" into a fast, honest error instead
|
|
# of a hung worker and a container the platform decides is unhealthy.
|
|
DB_CONNECT_TIMEOUT_SECONDS = int(os.getenv("DB_CONNECT_TIMEOUT_SECONDS", "5"))
|
|
|
|
DATABASE_URL = os.getenv(
|
|
"DATABASE_URL",
|
|
f"postgresql://{DB_USER}:{DB_PASSWORD}@{DB_HOST}:{DB_PORT}/{DB_NAME}",
|
|
)
|
|
|
|
# How often to re-run the brand-table <-> seed-catalog reconcile after boot.
|
|
#
|
|
# The startup run alone only catches what existed at boot. A brand table created
|
|
# directly in the database while the server is up - by hand, by a script, or by
|
|
# another machine sharing this database - is not mirrored into the seed catalogs
|
|
# until the next restart. This interval is what closes that window.
|
|
#
|
|
# reconcile_brand_catalogs() is idempotent and non-destructive, so a sweep that
|
|
# finds nothing to do costs one COUNT(*) per brand table and writes nothing.
|
|
# 0 disables the loop, leaving the startup run and POST /api/system/brand-sync.
|
|
BRAND_SYNC_INTERVAL_SECONDS = int(os.getenv("BRAND_SYNC_INTERVAL_SECONDS", "300"))
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Active brands (development working set)
|
|
# ---------------------------------------------------------------------------
|
|
# Comma-separated brand names that the application and every pipeline operate
|
|
# on. BLANK OR UNSET MEANS EVERY BRAND IS ACTIVE - that is the backwards
|
|
# compatible default and the way to switch this feature off again.
|
|
#
|
|
# Nothing is deleted when this is set: the other brand_* tables and their
|
|
# embeddings stay in the database untouched, they simply stop being discovered.
|
|
# Going from 3 brands to 5, 10 or all of them is an edit to this one line.
|
|
#
|
|
# Names are resolved through resolve_parent_brand + _sanitize_name, the same
|
|
# two steps that pick a product's storage table, so "Tata" here activates
|
|
# brand_hindustan_unilever exactly as ingesting Tata products would.
|
|
# See app/services/active_brands.py.
|
|
ACTIVE_BRANDS = os.getenv("ACTIVE_BRANDS", "")
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Brand discovery (brand name -> the 11-stage pipeline)
|
|
# ---------------------------------------------------------------------------
|
|
# Discovery turns a brand NAME into rows the ordinary catalog pipeline ingests.
|
|
# See app/services/brand_discovery.py. Every value below has a working default,
|
|
# so the feature needs no configuration to run.
|
|
#
|
|
# Open Food Facts is the primary source and the language model is the
|
|
# supplement, not the reverse: OFF returns real products carrying a real GTIN,
|
|
# while the default OLLAMA_MODEL_NAME (qwen2.5:1.5b) invents plausible ones that
|
|
# nothing downstream can catch. Turning BRAND_DISCOVERY_USE_OFF off leaves the
|
|
# result resting on the model alone.
|
|
BRAND_DISCOVERY_USE_OFF = _bool("BRAND_DISCOVERY_USE_OFF", "true")
|
|
BRAND_DISCOVERY_USE_LLM = _bool("BRAND_DISCOVERY_USE_LLM", "true")
|
|
|
|
# Products per discovery run. One CSV row per product; pack-size explosion
|
|
# happens later in stage 4, so this is well inside the 2000-row per-file cap.
|
|
BRAND_DISCOVERY_MAX_PRODUCTS = int(os.getenv("BRAND_DISCOVERY_MAX_PRODUCTS", "200"))
|
|
|
|
# Pack sizes kept per product when only the language model offers any. Stage 6
|
|
# runs an image search per exploded row, so this multiplies the slowest part of
|
|
# the run; 3 keeps a large brand inside a sane wall-clock.
|
|
BRAND_DISCOVERY_MAX_SIZES = int(os.getenv("BRAND_DISCOVERY_MAX_SIZES", "3"))
|
|
|
|
# Wall-clock ceiling on the LLM half of a run, checked between prompts. Open
|
|
# Food Facts runs first and is never subject to it, so a run that hits this
|
|
# still returns the evidence-backed products.
|
|
BRAND_DISCOVERY_DEADLINE_SECONDS = float(
|
|
os.getenv("BRAND_DISCOVERY_DEADLINE_SECONDS", "300")
|
|
)
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# S3 / DigitalOcean Spaces (product image storage) - optional
|
|
# ---------------------------------------------------------------------------
|
|
USE_S3 = _bool("USE_S3", "false")
|
|
S3_ACCESS_KEY = _require("S3_ACCESS_KEY", feature_flag="USE_S3") if USE_S3 else os.getenv("S3_ACCESS_KEY")
|
|
S3_SECRET_KEY = _require("S3_SECRET_KEY", feature_flag="USE_S3") if USE_S3 else os.getenv("S3_SECRET_KEY")
|
|
S3_ENDPOINT = _require("S3_ENDPOINT", feature_flag="USE_S3") if USE_S3 else os.getenv("S3_ENDPOINT")
|
|
S3_BUCKET = _require("S3_BUCKET", feature_flag="USE_S3") if USE_S3 else os.getenv("S3_BUCKET")
|
|
S3_REGION = os.getenv("S3_REGION", "sgp1")
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Google Custom Search (OPTIONAL image source - leave disabled if unset)
|
|
# ---------------------------------------------------------------------------
|
|
USE_GOOGLE_CSE = _bool("USE_GOOGLE_CSE", "false")
|
|
GOOGLE_API_KEY = _require("GOOGLE_API_KEY", feature_flag="USE_GOOGLE_CSE") if USE_GOOGLE_CSE else os.getenv("GOOGLE_API_KEY")
|
|
GOOGLE_CSE_ID = _require("GOOGLE_CSE_ID", feature_flag="USE_GOOGLE_CSE") if USE_GOOGLE_CSE else os.getenv("GOOGLE_CSE_ID")
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Open-source image sources (no API key needed for any of these three)
|
|
# ---------------------------------------------------------------------------
|
|
USE_DDG_IMAGES = _bool("USE_DDG_IMAGES", "true")
|
|
USE_OPEN_FACTS = _bool("USE_OPEN_FACTS", "true")
|
|
USE_WIKIMEDIA = _bool("USE_WIKIMEDIA", "true")
|
|
|
|
# Last-resort headless-browser (Python Playwright) image fallback. Requires
|
|
# `pip install playwright && playwright install chromium`; automatically
|
|
# skipped (logged once) if that hasn't been done.
|
|
USE_PLAYWRIGHT_FALLBACK = _bool("USE_PLAYWRIGHT_FALLBACK", "true")
|
|
|
|
# Minimum byte size for a downloaded image to be accepted as "real" (filters
|
|
# out 1x1 tracking pixels / broken placeholder images)
|
|
MIN_IMAGE_BYTES = int(os.getenv("MIN_IMAGE_BYTES", "3000"))
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# img_vector - a MobileNetV3-Small image embedding (1024 floats, L2-normalised)
|
|
# of each product's primary image, stored on its brand table
|
|
# (app/services/image_vector.py owns the column, image_embedder.py the model)
|
|
# ---------------------------------------------------------------------------
|
|
# Computed off the request path by one bounded worker thread after every
|
|
# catalog write, and by scripts/backfill_image_vectors.py for existing rows.
|
|
# Default on: the work is one small download and one ~30ms inference per row
|
|
# just written, never on a request. tests/conftest.py pins it OFF so the
|
|
# suite never dials the database named in a developer's .env.
|
|
ENABLE_IMAGE_VECTORS = _bool("ENABLE_IMAGE_VECTORS", "true")
|
|
# A download past this many bytes is abandoned - a wrong URL to a video must
|
|
# not fill the container.
|
|
IMAGE_VECTOR_MAX_BYTES = int(os.getenv("IMAGE_VECTOR_MAX_BYTES", str(8 * 1024 * 1024)))
|
|
# Refused before decoding when the header claims more pixels than this
|
|
# (decompression-bomb guard; 25MP is well past any product photo, and the
|
|
# embedder decodes at full resolution - ~75MB of RGB at this cap).
|
|
IMAGE_VECTOR_MAX_PIXELS = int(os.getenv("IMAGE_VECTOR_MAX_PIXELS", "25000000"))
|
|
IMAGE_VECTOR_TIMEOUT_SECONDS = float(os.getenv("IMAGE_VECTOR_TIMEOUT_SECONDS", "15"))
|
|
# Minimum gap between two requests to the same image host.
|
|
IMAGE_VECTOR_HOST_PAUSE_SECONDS = float(os.getenv("IMAGE_VECTOR_HOST_PAUSE_SECONDS", "0.5"))
|
|
# Writes waiting for the worker; beyond this a write's rows are left for the
|
|
# backfill script rather than queued.
|
|
IMAGE_VECTOR_QUEUE_MAX = int(os.getenv("IMAGE_VECTOR_QUEUE_MAX", "64"))
|
|
# The TFLite embedder (mobilenet_v3_small_embedder.tflite, input [1,224,224,3]
|
|
# float32, output [1,1024]). Lives under app/, NOT data/: /app/data is a named
|
|
# volume on every deployment (see BUNDLED_ASSETS_DIR), and a file added to
|
|
# the image under a mounted path is invisible on any volume that already
|
|
# exists. app/ is copied into the image and never mounted.
|
|
IMAGE_EMBED_MODEL_PATH = _dir(
|
|
"IMAGE_EMBED_MODEL_PATH",
|
|
_BACKEND_ROOT / "app" / "services" / "models" / "mobilenet" / "mobilenet_v3_small_embedder.tflite",
|
|
)
|
|
# Intra-op threads for one inference. 2 on the prod host; inference itself is
|
|
# serialised by a lock (a TFLite interpreter is not thread-safe).
|
|
IMAGE_EMBED_NUM_THREADS = int(os.getenv("IMAGE_EMBED_NUM_THREADS", "2"))
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# USDA FoodData Central - nutrition for loose, unbranded commodities
|
|
# ---------------------------------------------------------------------------
|
|
# Open Food Facts catalogues packaged products and has no entry for a raw
|
|
# apple, which is why every fresh-produce row had no nutrition at all. USDA's
|
|
# Foundation Foods and SR Legacy datasets are laboratory analyses of raw
|
|
# commodities, published per 100 g of edible portion.
|
|
#
|
|
# NO KEY IS REQUIRED for normal operation: `nutrition_usda_service` reads a
|
|
# snapshot built from USDA's open bulk download, so an enrichment run makes zero
|
|
# outbound USDA calls. A key is only needed to look up an id the snapshot lacks,
|
|
# or to rebuild the snapshot from the live API.
|
|
#
|
|
# Plain os.getenv rather than `_require(..., feature_flag=...)` on purpose: a
|
|
# fresh checkout with no key must still import, or the whole test suite fails at
|
|
# collection time.
|
|
USE_USDA_FDC = _bool("USE_USDA_FDC", "true")
|
|
USDA_FDC_API_KEY = os.getenv("USDA_FDC_API_KEY", "").strip()
|
|
|
|
# Score newly uploaded products automatically, instead of waiting for somebody
|
|
# to remember to POST /api/admin/nutrition-intelligence/enrich. Runs as a
|
|
# background job AFTER the ingestion batch finishes, never inside it - see the
|
|
# comment at the submission site in `app/core/batch_ingest.py`.
|
|
#
|
|
# The off switch exists because this is the one part of ingestion that makes
|
|
# outbound calls per product: an operator loading a very large catalogue on a
|
|
# metered connection, or re-running an import they intend to score later in one
|
|
# controlled pass, needs a way to say "not now" without a code change.
|
|
AUTO_ENRICH_ON_UPLOAD = _bool("AUTO_ENRICH_ON_UPLOAD", "true")
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Store-catalog enrichment pipeline (app/core/store_catalog_pipeline.py)
|
|
# ---------------------------------------------------------------------------
|
|
# Stages 7-10 of the store-Excel ingestion pipeline. Ported from the sibling
|
|
# Universal_Catalog_Barcode_Enrichment project along with the code that reads
|
|
# them; the names are kept identical so the ported modules need no edits.
|
|
#
|
|
# The three network-touching flags default to FALSE here, unlike in the
|
|
# sibling. This backend serves an interactive API on a shared 8GB host, and a
|
|
# 2000-row upload with web lookups on would fire thousands of outbound
|
|
# requests. Turn them on deliberately, per environment.
|
|
ENABLE_SKU_WEB_LOOKUP = _bool("ENABLE_SKU_WEB_LOOKUP", "false")
|
|
ENABLE_BARCODE_LOOKUP = _bool("ENABLE_BARCODE_LOOKUP", "false")
|
|
ENABLE_MANUFACTURER_SITE_LOOKUP = _bool("ENABLE_MANUFACTURER_SITE_LOOKUP", "false")
|
|
|
|
# Barcodes AFTER the upload settles, in bulk, on the enrichment job's own
|
|
# thread. Default TRUE where ENABLE_BARCODE_LOOKUP above is false, and the
|
|
# difference is cost, not appetite for risk:
|
|
#
|
|
# ENABLE_BARCODE_LOOKUP = one search request PER PRODUCT, inline, against an
|
|
# endpoint capped at 10 requests/minute. A 200-row
|
|
# upload is twenty minutes of a held request.
|
|
# ENRICH_BARCODES_ON_UPLOAD = one corpus fetch PER BRAND (~5 requests total),
|
|
# matched offline, after the uploader has their
|
|
# result. Cost is per brand, not per row.
|
|
#
|
|
# It also has to run before the nutrition phase rather than beside it: a
|
|
# barcode makes the nutrition lookup exact (0.95) instead of fuzzy (0.32), and
|
|
# skip_if_verified means whichever lands first wins permanently.
|
|
ENRICH_BARCODES_ON_UPLOAD = _bool("ENRICH_BARCODES_ON_UPLOAD", "true")
|
|
|
|
# Offline/deterministic stages - safe to leave on.
|
|
ENABLE_HSN_GST_ENRICHMENT = _bool("ENABLE_HSN_GST_ENRICHMENT", "true")
|
|
ENABLE_PRODUCT_VALIDATION = _bool("ENABLE_PRODUCT_VALIDATION", "true")
|
|
|
|
# Validation gate thresholds: below REJECT the row is dropped, below REVIEW it
|
|
# is stored but flagged `validation_status="review"`.
|
|
VALIDATION_REJECT_THRESHOLD = float(os.getenv("VALIDATION_REJECT_THRESHOLD", "0.35"))
|
|
VALIDATION_REVIEW_THRESHOLD = float(os.getenv("VALIDATION_REVIEW_THRESHOLD", "0.70"))
|
|
|
|
# Pack-size explosion: how many size rows one uploaded product may become.
|
|
MAX_VARIANTS_PER_PRODUCT = int(os.getenv("MAX_VARIANTS_PER_PRODUCT", "6"))
|
|
ENABLE_PER_VARIANT_IMAGES = _bool("ENABLE_PER_VARIANT_IMAGES", "false")
|
|
PER_VARIANT_IMAGE_MAX_RESULTS = int(os.getenv("PER_VARIANT_IMAGE_MAX_RESULTS", "10"))
|
|
|
|
# Barcode lookup tuning. The cache TTL is long (30 days) because a GTIN for a
|
|
# given pack size does not change, and negative results are cached too.
|
|
BARCODE_LOOKUP_TIMEOUT_SECONDS = float(os.getenv("BARCODE_LOOKUP_TIMEOUT_SECONDS", "10"))
|
|
BARCODE_LOOKUP_CACHE_TTL_SECONDS = float(os.getenv("BARCODE_LOOKUP_CACHE_TTL_SECONDS", str(30 * 24 * 3600)))
|
|
BARCODE_LOOKUP_MAX_CONCURRENCY = int(os.getenv("BARCODE_LOOKUP_MAX_CONCURRENCY", "5"))
|
|
BARCODE_COUNTRY_TAG = os.getenv("BARCODE_COUNTRY_TAG", "india")
|
|
|
|
# How similar a candidate's product name must be to ours before its barcode is
|
|
# believed. Applies to BOTH directions: looking a barcode up from a name, and
|
|
# looking a product up from a barcode.
|
|
#
|
|
# RAISED FROM matching.py's OWN 0.45 DEFAULT, ON EVIDENCE. That default is a
|
|
# reasonable general floor, but by the time a candidate reaches this gate its
|
|
# brand and pack size have ALREADY been matched - so the name is the only thing
|
|
# left doing any discriminating, and it has to carry the whole decision.
|
|
#
|
|
# At 0.45 it did not. Scored across every catalogue barcode Open Food Facts
|
|
# knows, more than half the accepted matches were a different product:
|
|
#
|
|
# floor accepted wrong
|
|
# 0.45 15 8
|
|
# 0.70 7 2
|
|
# 0.78 2 0
|
|
#
|
|
# 0.761 "Tata Tea Gold 500g" -> "Tata Tea Gold Care" different
|
|
# 0.658 "MTR Masala 300g" -> "MTR Chana Masala" different
|
|
# 0.538 "Aachi Chicken Masala 50g" -> "Chicken Kabab/65 Masala" different
|
|
# 0.097 "Lion Dates Powder 100g" -> "PEPER NOTEN" different
|
|
#
|
|
# 0.78 is where the sample is clean, NOT where the yield is good, and it is a
|
|
# judgement rather than a separation: two products tie at 0.773 with opposite
|
|
# verdicts. It is set for precision because a WRONG barcode is worse than no
|
|
# barcode - it is an identifier other systems join on, and 33 of the 95 already
|
|
# in the catalogue are wrong, all of them accepted at the old floor.
|
|
#
|
|
# Lower it only with the yield/error numbers in front of you.
|
|
BARCODE_MIN_NAME_SIMILARITY = float(os.getenv("BARCODE_MIN_NAME_SIMILARITY", "0.78"))
|
|
|
|
# Optional barcode source credentials. Each source disables itself when its
|
|
# key is blank, so leaving these unset simply narrows the lookup cascade.
|
|
GS1_INDIA_API_BASE_URL = os.getenv("GS1_INDIA_API_BASE_URL", "")
|
|
GS1_INDIA_API_KEY = os.getenv("GS1_INDIA_API_KEY", "")
|
|
UPC_DATABASE_API_KEY = os.getenv("UPC_DATABASE_API_KEY", "")
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# HTTP client defaults
|
|
# ---------------------------------------------------------------------------
|
|
USER_AGENT = os.getenv("USER_AGENT", "CatalogBot/1.0 (+https://example.com)")
|
|
REQUEST_TIMEOUT_SECONDS = int(os.getenv("REQUEST_TIMEOUT_SECONDS", "20"))
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# FastAPI / web server
|
|
# ---------------------------------------------------------------------------
|
|
API_CORS_ORIGINS = [
|
|
origin.strip()
|
|
for origin in os.getenv("API_CORS_ORIGINS", "http://localhost:5173,http://127.0.0.1:5173").split(",")
|
|
if origin.strip()
|
|
]
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Authentication
|
|
# ---------------------------------------------------------------------------
|
|
# CORS above is not access control - browsers enforce it, and curl ignores it
|
|
# entirely. These settings are what actually guards the write/compute endpoints
|
|
# (catalog generation, ML training, uploads, chat).
|
|
#
|
|
# AUTH_ENABLED=false turns every guard off, restoring the old behaviour where
|
|
# any caller could reach any endpoint. It exists so a fresh checkout still runs
|
|
# without generating secrets first; app/infrastructure/security.py logs a
|
|
# warning at import when it is off. Never deploy with it off.
|
|
AUTH_ENABLED = _bool("AUTH_ENABLED", "true")
|
|
|
|
# Signs and verifies access tokens. Changing it invalidates every issued token,
|
|
# which is the intended way to force everyone to sign in again. Generate with:
|
|
# python scripts/make_auth_secrets.py
|
|
AUTH_SECRET_KEY = (
|
|
_require("AUTH_SECRET_KEY", feature_flag="AUTH_ENABLED")
|
|
if AUTH_ENABLED
|
|
else os.getenv("AUTH_SECRET_KEY", "")
|
|
)
|
|
|
|
# How long an issued token stays valid. 12h by default: long enough that a
|
|
# working day needs one sign-in, short enough that a leaked token expires.
|
|
AUTH_TOKEN_TTL_MINUTES = int(os.getenv("AUTH_TOKEN_TTL_MINUTES", "720"))
|
|
|
|
# The interactive accounts. Only PBKDF2 digests are stored - never a password.
|
|
# `make_auth_secrets.py` prints the lines ready to paste.
|
|
#
|
|
# `admin` is required whenever auth is on: without it nobody could sign in.
|
|
AUTH_ADMIN_USERNAME = _clean(
|
|
"AUTH_ADMIN_USERNAME", os.getenv("AUTH_ADMIN_USERNAME", "admin")
|
|
)
|
|
AUTH_ADMIN_PASSWORD_HASH = _clean(
|
|
"AUTH_ADMIN_PASSWORD_HASH",
|
|
(
|
|
_require("AUTH_ADMIN_PASSWORD_HASH", feature_flag="AUTH_ENABLED")
|
|
if AUTH_ENABLED
|
|
else os.getenv("AUTH_ADMIN_PASSWORD_HASH", "")
|
|
),
|
|
)
|
|
|
|
# The second `user` account is OPTIONAL, and left unset in this deployment.
|
|
# An empty hash is how the account is switched off: auth.py builds its account
|
|
# table from these values and omits any entry whose hash is blank, so there is
|
|
# nothing to sign in to. Setting the hash again re-enables it with no code
|
|
# change - which is exactly what the test suite does in tests/conftest.py.
|
|
AUTH_USER_USERNAME = _clean(
|
|
"AUTH_USER_USERNAME", os.getenv("AUTH_USER_USERNAME", "user")
|
|
)
|
|
AUTH_USER_PASSWORD_HASH = _clean(
|
|
"AUTH_USER_PASSWORD_HASH", os.getenv("AUTH_USER_PASSWORD_HASH", "")
|
|
)
|
|
|
|
# Failed-login throttle, applied per username+client-IP. Prevents an exposed
|
|
# login endpoint from being a free password oracle.
|
|
AUTH_MAX_LOGIN_ATTEMPTS = int(os.getenv("AUTH_MAX_LOGIN_ATTEMPTS", "10"))
|
|
AUTH_LOCKOUT_SECONDS = int(os.getenv("AUTH_LOCKOUT_SECONDS", "300"))
|
|
|
|
# Local-development escape hatch: accept ANY password at /api/auth/login, so a
|
|
# developer who does not have the configured passwords to hand can still reach
|
|
# the admin and user pages. The username still selects the role, and the token
|
|
# issued is a normal signed one - so every downstream guard, /api/auth/me, and
|
|
# the React route gating all behave exactly as they do in production. What is
|
|
# skipped is only the password check.
|
|
#
|
|
# This is NOT the same as AUTH_ENABLED=false. That disables every guard *and*
|
|
# makes /api/auth/login return 503, which breaks the login page outright. This
|
|
# flag keeps the whole auth machinery running and unlocks just the front door.
|
|
#
|
|
# Anyone who can reach the API can sign in as admin while it is on. Keep it
|
|
# false anywhere the port is reachable by someone you would not hand the admin
|
|
# password to.
|
|
AUTH_ALLOW_ANY_LOGIN = _bool("AUTH_ALLOW_ANY_LOGIN", "false")
|
|
|
|
|
|
# Shortest acceptable API key secret. token_urlsafe(32) yields 43 characters, so
|
|
# this rejects hand-typed values without rejecting anything the documented
|
|
# generator produces.
|
|
API_KEY_MIN_LENGTH = 32
|
|
|
|
|
|
def _parse_api_keys(raw: str) -> dict:
|
|
"""
|
|
Parse ``API_KEYS`` - ``name:role:secret`` triples, comma-separated.
|
|
|
|
Keyed by secret because that is what an inbound request presents. One entry
|
|
per consumer is the point: a shared key cannot be revoked for one caller
|
|
without breaking all of them.
|
|
|
|
Secrets must be at least API_KEY_MIN_LENGTH characters. That is not about
|
|
guessing the key over the network - the lockout and the network itself make
|
|
online brute force impractical - but about what /api/health publishes. It
|
|
reports a truncated digest of every configured key so a deployment can be
|
|
checked against the config it was built from, and a digest of a *raw* secret
|
|
is only safe when the secret is unguessable offline. An admin password hash
|
|
embeds a random salt, so its fingerprint discloses nothing; an API key has no
|
|
salt, and a hand-picked "changeme" would fall to a wordlist in seconds.
|
|
Generate one with: python -c "import secrets; print(secrets.token_urlsafe(32))"
|
|
"""
|
|
parsed: dict = {}
|
|
for entry in raw.split(","):
|
|
entry = entry.strip()
|
|
if not entry:
|
|
continue
|
|
parts = entry.split(":")
|
|
if len(parts) != 3:
|
|
raise RuntimeError(
|
|
f"Malformed API_KEYS entry {entry!r}. Expected 'name:role:secret', "
|
|
f"comma-separated between entries."
|
|
)
|
|
name, role, secret = (p.strip() for p in parts)
|
|
# MUST stay in step with ROLE_PERMISSIONS in app/infrastructure/security.py,
|
|
# which is the source of truth. It is duplicated rather than imported
|
|
# because security.py imports THIS module, so importing it back here
|
|
# would be a cycle. A role added there but not here is rejected at boot
|
|
# with the message below - loud, and before any request is served.
|
|
if role not in {"admin", "user", "uploader"}:
|
|
raise RuntimeError(
|
|
f"API_KEYS entry {name!r} has role {role!r}; expected 'admin', 'user' "
|
|
f"or 'uploader'."
|
|
)
|
|
if not secret:
|
|
raise RuntimeError(f"API_KEYS entry {name!r} has an empty secret.")
|
|
if len(secret) < API_KEY_MIN_LENGTH:
|
|
raise RuntimeError(
|
|
f"API_KEYS entry {name!r} has a {len(secret)}-character secret; at least "
|
|
f"{API_KEY_MIN_LENGTH} are required, because /api/health publishes a digest "
|
|
f"of it. Generate one with: "
|
|
f"python -c \"import secrets; print(secrets.token_urlsafe(32))\""
|
|
)
|
|
parsed[secret] = (name, role)
|
|
return parsed
|
|
|
|
|
|
# Machine consumers of api.<domain>. Empty by default - browser sessions go
|
|
# through /api/auth/login instead, and a key that nobody needs is only risk.
|
|
#
|
|
# NAME THE KEY FOR ITS FUNCTION, NOT THE PERSON HOLDING IT.
|
|
# /api/health is public and reports {name, role, fingerprint} for every
|
|
# configured key (describe_api_keys in security.py). The secret is never
|
|
# exposed, but the NAME is - so `catalog-drop:uploader:...` is right and
|
|
# `priya-laptop:uploader:...` publishes a colleague's name to anyone who
|
|
# curls the health endpoint.
|
|
API_KEYS = _parse_api_keys(os.getenv("API_KEYS", ""))
|
|
|
|
# Default RAG behaviour
|
|
RAG_DEFAULT_TOP_K = int(os.getenv("RAG_DEFAULT_TOP_K", "5"))
|
|
RAG_MAX_TOP_K = int(os.getenv("RAG_MAX_TOP_K", "15"))
|
|
RAG_MAX_CONTEXT_CHARS = int(os.getenv("RAG_MAX_CONTEXT_CHARS", "4000"))
|
|
|
|
# Optional pgvector cosine-distance ceiling (0 = identical, 2 = opposite)
|
|
# used to drop weak semantic matches before they reach the LLM. Left
|
|
# unset (None) by default since the primary relevance guardrail is the
|
|
# category-aware retrieval in rag_service.py; set RAG_MAX_DISTANCE (e.g.
|
|
# "0.9") once you've inspected real distance scores for your embeddings
|
|
# if you want an extra cutoff on top of that.
|
|
_raw_max_distance = os.getenv("RAG_MAX_DISTANCE", "").strip()
|
|
RAG_MAX_DISTANCE = float(_raw_max_distance) if _raw_max_distance else None
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Catalog search (GET /api/search) and suggest (GET /api/suggest)
|
|
# ---------------------------------------------------------------------------
|
|
# Deliberately separate from RAG_MAX_TOP_K above. That ceiling exists to protect
|
|
# the LLM prompt budget in /api/chat, and raising it would degrade every chat
|
|
# answer. /api/search feeds a product grid, which has no such budget - sharing
|
|
# the constant is what silently truncated every brand search to 15 products.
|
|
SEARCH_DEFAULT_TOP_K = int(os.getenv("SEARCH_DEFAULT_TOP_K", "60"))
|
|
# Mirrors the browse endpoints' le=100000 so a brand search can return exactly
|
|
# the same set as GET /api/brands/{brand}/products.
|
|
SEARCH_MAX_TOP_K = int(os.getenv("SEARCH_MAX_TOP_K", "100000"))
|
|
# Separate, much smaller ceiling for the hybrid path: semantic_search multiplies
|
|
# top_k by 5 per brand table when a category/price filter is present, and every
|
|
# read does SELECT * (which drags the vector(384) embedding column over the wire).
|
|
SEARCH_HYBRID_MAX_TOP_K = int(os.getenv("SEARCH_HYBRID_MAX_TOP_K", "100"))
|
|
SEARCH_LEXICAL_CANDIDATES = int(os.getenv("SEARCH_LEXICAL_CANDIDATES", "200"))
|
|
|
|
SUGGEST_DEFAULT_LIMIT = int(os.getenv("SUGGEST_DEFAULT_LIMIT", "8"))
|
|
SUGGEST_MIN_QUERY_LEN = int(os.getenv("SUGGEST_MIN_QUERY_LEN", "2"))
|
|
SUGGEST_FUZZY_MIN_RATIO = float(os.getenv("SUGGEST_FUZZY_MIN_RATIO", "0.72"))
|