Compare commits
76 Commits
ad7bb9250b
...
master
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
34c65a7cbd | ||
|
|
c0601b65fe | ||
|
|
c0489d89d6 | ||
|
|
6628207810 | ||
|
|
fe208f4715 | ||
|
|
f933ea10a1 | ||
|
|
bc786b1c49 | ||
|
|
d6296bd1f0 | ||
|
|
afa0bfa743 | ||
|
|
deae694a1f | ||
|
|
9c8dbf1759 | ||
|
|
ce4fa70dee | ||
|
|
e1a5962f82 | ||
|
|
d5a23f6456 | ||
|
|
10b24c6348 | ||
|
|
2749bee1a3 | ||
|
|
75dd3eb3ce | ||
|
|
c2d7f0ca1d | ||
|
|
d3709c1a4f | ||
|
|
7ac3571b59 | ||
|
|
0c36628d4b | ||
|
|
bb30edaffa | ||
|
|
8059120ce3 | ||
|
|
0c3e23fac4 | ||
|
|
1300d4c678 | ||
|
|
1c20bfc408 | ||
|
|
d5ec5755bf | ||
|
|
5399fea4cc | ||
|
|
1de518feed | ||
|
|
7b3fbc47b4 | ||
|
|
68bff007ea | ||
|
|
6c7a886659 | ||
|
|
183b65b3bd | ||
|
|
a6e78fee95 | ||
|
|
164ef90b31 | ||
|
|
3df2dc5991 | ||
|
|
3b1352b99d | ||
|
|
8f25842281 | ||
|
|
8394316907 | ||
|
|
998df898db | ||
|
|
27d53fa957 | ||
|
|
5aa2669f7d | ||
|
|
52f5d3be1d | ||
|
|
0e75d32f61 | ||
|
|
a54bd43f8b | ||
|
|
f698720ee2 | ||
|
|
a9ab1fcf71 | ||
|
|
1b347f91db | ||
|
|
2482fe43e3 | ||
|
|
bbeb859d05 | ||
|
|
b267dbab74 | ||
|
|
7bf8dc6922 | ||
|
|
fbb1356e47 | ||
|
|
e224043e26 | ||
|
|
f41973e1cf | ||
|
|
a7a2430694 | ||
|
|
7a4583372f | ||
|
|
691efef880 | ||
|
|
dba8d36175 | ||
| 5e373ea20c | |||
|
|
dce2405e50 | ||
|
|
c17e1c7e13 | ||
|
|
7153e12bdb | ||
|
|
94f43a326e | ||
|
|
c101f2c8ba | ||
|
|
1b56aae396 | ||
|
|
21c675cb55 | ||
|
|
86bb1ebd0a | ||
|
|
241fd237f8 | ||
|
|
ea0c00d68e | ||
|
|
5f66a797d8 | ||
|
|
8169cafd9c | ||
|
|
c078bced04 | ||
|
|
d81c3ea18e | ||
|
|
56fdb82d0a | ||
|
|
00216ae834 |
@@ -3,6 +3,15 @@ __pycache__
|
||||
*.pyc
|
||||
.pytest_cache
|
||||
*.log
|
||||
.env
|
||||
.git
|
||||
tests
|
||||
|
||||
# .env.production carries this deployment's configuration and the Dockerfile
|
||||
# copies it to /app/.env inside the image, so it must NOT be ignored here.
|
||||
# Dokploy's own .env (written from the Environment tab, empty when that tab is
|
||||
# blank) is ignored instead - it is what silently overwrote the committed one.
|
||||
#
|
||||
# The generated sign-in passwords must never be in the image.
|
||||
.env
|
||||
.env.local
|
||||
SIGNIN_PASSWORDS.txt
|
||||
|
||||
239
.env.example
239
.env.example
@@ -65,7 +65,7 @@
|
||||
# credentials off if it sees one, so a wildcard silently breaks every call).
|
||||
#
|
||||
# Production - the React app is served from catalogue.nearle.ai.in and calls
|
||||
# the API at mcp.catalogue.nearle.ai.in, so that frontend origin must be listed:
|
||||
# the API at mcp.nearle.ai.in, so that frontend origin must be listed:
|
||||
#
|
||||
# API_CORS_ORIGINS=https://catalogue.nearle.ai.in
|
||||
#
|
||||
@@ -116,20 +116,115 @@ AUTH_LOCKOUT_SECONDS=300
|
||||
# token issued is a normal one, so every other guard behaves normally. Unlike
|
||||
# AUTH_ENABLED=false it leaves the login page working - it just stops checking
|
||||
# the password. Anyone who can reach the port becomes admin: keep it false here.
|
||||
AUTH_ALLOW_ANY_LOGIN=false
|
||||
AUTH_ALLOW_ANY_LOGIN=true
|
||||
|
||||
# Machine consumers of api.<domain> - scripts, partner integrations, your own
|
||||
# backends. Format: name:role:secret, comma-separated, role is admin or user.
|
||||
# backends. Format: name:role:secret, comma-separated. Role is one of:
|
||||
#
|
||||
# admin everything, including system/init and model training
|
||||
# user catalog write access (add products, upload inventory)
|
||||
# uploader ONE verb: POST /api/uploads/catalog. Nothing else - it cannot
|
||||
# read the catalog, cannot see another caller's submissions, and
|
||||
# cannot cancel or resume anything. This is the role to issue to
|
||||
# an outside party who needs to send you spreadsheets.
|
||||
#
|
||||
# Callers send the secret as an X-API-Key header.
|
||||
#
|
||||
# One entry per consumer, always: a shared key cannot be revoked for one caller
|
||||
# without breaking every other. Mint them with:
|
||||
# python scripts/make_auth_secrets.py --api-key partner-x:user
|
||||
# python scripts/make_auth_secrets.py --api-key catalog-drop:uploader
|
||||
#
|
||||
# NAME THE KEY FOR ITS FUNCTION, NOT THE PERSON HOLDING IT. /api/health is
|
||||
# public and reports {name, role, fingerprint} for every configured key. The
|
||||
# secret is never exposed, but the NAME is - so `catalog-drop:uploader:...` is
|
||||
# right and `priya-laptop:uploader:...` publishes a colleague's name to anyone
|
||||
# who curls the health endpoint.
|
||||
#
|
||||
# Leave empty if only the web app calls the API - it signs in through
|
||||
# /api/auth/login instead, and a key nobody needs is only risk.
|
||||
API_KEYS=
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Catalog batch ingestion
|
||||
# ---------------------------------------------------------------------------
|
||||
# Used by both spreadsheet ingestion routes - the admin one
|
||||
# (POST /api/admin/catalog-batch/ingest) and the API-client one
|
||||
# (POST /api/uploads/catalog). Every ceiling is enforced in the application,
|
||||
# not at the proxy: in production the caller reaches mcp.nearle.ai.in directly,
|
||||
# so neither nginx's client_max_body_size nor Caddy's request_body cap is in
|
||||
# front of these endpoints.
|
||||
#
|
||||
# Per-file limits are fixed in code at 10MB / 2000 rows, matching the
|
||||
# single-file upload path. These bound the BATCH on top of that.
|
||||
BATCH_MAX_FILES=20
|
||||
BATCH_MAX_TOTAL_BYTES=52428800
|
||||
BATCH_MAX_TOTAL_ROWS=20000
|
||||
|
||||
# Where staged uploads live. Under DATA_DIR because that path is already a
|
||||
# declared volume, which is what lets a batch survive a container restart.
|
||||
# BATCH_UPLOAD_DIR=/app/data/batch_uploads
|
||||
|
||||
# Batches allowed to wait behind the one running. One worker thread runs a
|
||||
# single batch at a time; past this depth the endpoints answer 429 rather than
|
||||
# accepting work they have no intention of starting soon. This queue is the
|
||||
# only thing bounding what an uploader key can cost in CPU - raise it with care
|
||||
# on a one-vCPU host.
|
||||
BATCH_QUEUE_MAX=4
|
||||
|
||||
# Staged files are deleted this many days after the batch was created.
|
||||
BATCH_RETENTION_DAYS=7
|
||||
|
||||
# Deliberately false. A batch a restart cut short is marked "interrupted" and
|
||||
# waits for someone to press Resume. Auto-resuming means a container stuck in a
|
||||
# restart loop re-runs the heaviest work in the app on every boot, which is how
|
||||
# a slow start becomes an unrecoverable spiral.
|
||||
#
|
||||
# Worth knowing alongside UPLOAD_AUTORUN below: with uploads running unattended,
|
||||
# a redeploy in the middle of one leaves that batch "interrupted" and waiting
|
||||
# for a human. It is the one place manual intervention comes back.
|
||||
BATCH_AUTO_RESUME=false
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Unattended ingestion
|
||||
# ---------------------------------------------------------------------------
|
||||
# Whether POST /api/uploads/catalog runs the pipeline on arrival (true) or parks
|
||||
# the files in the admin review inbox for someone to start by hand (false).
|
||||
#
|
||||
# READ THIS BEFORE CHANGING IT. That endpoint takes NO credential - it was
|
||||
# opened on purpose so colleagues could send spreadsheets without one being
|
||||
# issued to them. With autorun on, "anyone who can reach this host" and "anyone
|
||||
# who can write to the live catalogue" are the same set of people, and an ingest
|
||||
# is an upsert with no undo.
|
||||
#
|
||||
# What still bounds it is throughput, not identity: the per-request ceilings
|
||||
# above, and BATCH_QUEUE_MAX behind a single worker. A sender can occupy the
|
||||
# ingestion worker; they cannot multiply it.
|
||||
#
|
||||
# Set false and the review inbox comes back with no code change - the INBOX_*
|
||||
# ceilings below apply only on that path.
|
||||
UPLOAD_AUTORUN=true
|
||||
|
||||
# How an auto-started run behaves. Deliberately NOT accepted from the request:
|
||||
# the sender is anonymous, and letting an anonymous caller switch on the
|
||||
# expensive outbound stages is the one thing this endpoint must not allow.
|
||||
#
|
||||
# Images on, because a product landing without one is the failure this endpoint
|
||||
# exists to avoid. Stage 6 is the slowest stage and reaches the network, but
|
||||
# only one batch runs at a time, so nothing else competes with it.
|
||||
#
|
||||
# LLM on, though in production it currently does nothing: USE_OLLAMA is false
|
||||
# there, so the LLM call returns immediately without a request and the row keeps
|
||||
# its blank description. Set true so the pipeline is already right for the day
|
||||
# an Ollama server is reachable.
|
||||
#
|
||||
# If you DO set USE_OLLAMA=true, make sure something is actually listening on
|
||||
# OLLAMA_BASE_URL. An unreachable server costs a 5s probe, and while that probe
|
||||
# is cached per 30s rather than paid per row, a reachable-but-slow model is
|
||||
# billed per row at OLLAMA_TIMEOUT_SECONDS.
|
||||
UPLOAD_AUTORUN_FETCH_IMAGES=true
|
||||
UPLOAD_AUTORUN_USE_LLM=true
|
||||
|
||||
USE_OLLAMA=true
|
||||
OLLAMA_BASE_URL=http://localhost:11434
|
||||
OLLAMA_MODEL_NAME=qwen2.5:1.5b
|
||||
@@ -144,6 +239,74 @@ DB_NAME=pgvector
|
||||
DB_USER=postgres
|
||||
DB_PASSWORD=changeme
|
||||
|
||||
# How often (seconds) to reconcile brand_* tables against data/seed_catalogs/.
|
||||
#
|
||||
# Brands are discovered from the database, not from a list: every brand_* table
|
||||
# is enumerated on each request, so a new one shows up in the catalog with no
|
||||
# registration step. This interval covers the other direction - mirroring a
|
||||
# table created directly in the database into a seed catalog file - which
|
||||
# otherwise only happens at startup or via POST /api/system/brand-sync.
|
||||
#
|
||||
# A sweep with nothing to do is one COUNT(*) per brand table and writes nothing.
|
||||
# Raise it if the database is remote and the chatter matters; 0 turns the loop
|
||||
# off entirely, leaving only the startup run and the manual endpoint.
|
||||
BRAND_SYNC_INTERVAL_SECONDS=300
|
||||
|
||||
# --- Active brands (development working set) --------------------------------
|
||||
# Comma-separated. BLANK OR UNSET = every brand is active (the production
|
||||
# default). Setting it narrows the catalog, RAG, search, MCP, analytics,
|
||||
# nutrition and every Dagster asset to these brands in one place - nothing is
|
||||
# deleted, the other brand_* tables just stop being discovered.
|
||||
# Names resolve through the brand aliases, so "Tata" activates
|
||||
# brand_hindustan_unilever exactly as ingesting Tata products would.
|
||||
# NOTE: this is a whitelist, so anything omitted is invisible in browse, search,
|
||||
# suggestions, chat and the MCP tools. If you narrow it, include "Own Products"
|
||||
# - that is the bucket unbranded commodities (dal, sugar, salt, spices) are
|
||||
# filed under, and it is a normal brand table as far as every read path is
|
||||
# concerned.
|
||||
#ACTIVE_BRANDS=Amul,Cadbury,Hindustan Unilever,Own Products
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Brand discovery (a brand NAME -> the 11-stage pipeline)
|
||||
# ---------------------------------------------------------------------------
|
||||
# POST /api/admin/brand-discovery/preview finds a brand's products, and
|
||||
# /ingest stages the ones an admin approved as an ordinary catalog batch.
|
||||
# Every value below has a working default; none of these need to be set.
|
||||
#
|
||||
# Open Food Facts is the primary source and the language model is the
|
||||
# supplement. OFF returns real products with real barcodes and pack sizes;
|
||||
# the default OLLAMA_MODEL_NAME (qwen2.5:1.5b) will invent plausible ones, and
|
||||
# nothing downstream can tell a well-formed fiction from a real product. Turn
|
||||
# BRAND_DISCOVERY_USE_OFF off and the result rests on the model alone.
|
||||
#
|
||||
# NOTE: discovering a brand that is not in ACTIVE_BRANDS writes a complete
|
||||
# catalog that no endpoint can read. The ingest route refuses with a 409 and
|
||||
# names the line to add here; it is a config change plus a restart, never a
|
||||
# re-ingest.
|
||||
#BRAND_DISCOVERY_USE_OFF=true
|
||||
#BRAND_DISCOVERY_USE_LLM=true
|
||||
#BRAND_DISCOVERY_MAX_PRODUCTS=200
|
||||
# Pack sizes kept per product when only the language model offers any. Stage 6
|
||||
# runs an image search per exploded row, so this multiplies the slowest stage.
|
||||
#BRAND_DISCOVERY_MAX_SIZES=3
|
||||
# Wall-clock ceiling on the language-model half, checked between prompts. Open
|
||||
# Food Facts runs first and is never subject to it.
|
||||
#BRAND_DISCOVERY_DEADLINE_SECONDS=300
|
||||
|
||||
# Brand Discovery's optional "Web & retail listings" source, for brands Open
|
||||
# Food Facts does not carry. Reads search-engine results that point at retailer
|
||||
# product pages (Blinkit, Zepto, Instamart, BigBasket, JioMart, Amazon,
|
||||
# Flipkart...) - never the pages themselves, never a language model. Uses the
|
||||
# Google Custom Search API when USE_GOOGLE_CSE=true, else DuckDuckGo.
|
||||
#WEB_DISCOVERY_ENABLED=true
|
||||
#WEB_DISCOVERY_MAX_QUERIES=80
|
||||
#WEB_DISCOVERY_PAUSE_SECONDS=2.0
|
||||
#WEB_DISCOVERY_RESULTS_PER_QUERY=20
|
||||
#WEB_DISCOVERY_CACHE_DAYS=7
|
||||
#WEB_DISCOVERY_DIR=/app/data/cache/web_discovery
|
||||
|
||||
|
||||
USE_S3=true
|
||||
S3_ACCESS_KEY=your-do-spaces-key
|
||||
S3_SECRET_KEY=your-do-spaces-secret
|
||||
@@ -163,6 +326,54 @@ USE_PLAYWRIGHT_FALLBACK=true
|
||||
|
||||
MIN_IMAGE_BYTES=3000
|
||||
|
||||
# img_vector: a MobileNetV3-Small embedding (1024 floats, L2-normalised) of
|
||||
# each product's primary image, stored on its brand table and computed on a
|
||||
# background thread after every catalog write (app/services/image_vector.py,
|
||||
# model in app/services/image_embedder.py). Existing rows are filled by
|
||||
# `python -m scripts.backfill_image_vectors --all --apply`. Set false to stop
|
||||
# the API process from downloading images at all.
|
||||
ENABLE_IMAGE_VECTORS=true
|
||||
# Where the .tflite lives. The default is inside app/ (shipped with the image
|
||||
# and never volume-mounted); only override to point at a different file.
|
||||
#IMAGE_EMBED_MODEL_PATH=/app/app/services/models/mobilenet/mobilenet_v3_small_embedder.tflite
|
||||
#IMAGE_EMBED_NUM_THREADS=2
|
||||
|
||||
# Identify a product from a phone photo (POST /api/search/identify). A phone
|
||||
# photo scores ~0.63 on img_vector against the catalog's render of the same
|
||||
# pack, so below IMAGE_IDENTIFY_MIN_IMAGE_SCORE the label text decides: the
|
||||
# client's OCR `text` if it sent one, else what the server reads off the photo
|
||||
# with rapidocr (models ship in the wheel; see requirements-ocr.txt). Set
|
||||
# ENABLE_SERVER_OCR=false to rely on client text only.
|
||||
ENABLE_SERVER_OCR=true
|
||||
#OCR_MIN_CONFIDENCE=0.5
|
||||
#OCR_MAX_SIDE_PX=1280
|
||||
#OCR_NUM_THREADS=2
|
||||
#IMAGE_IDENTIFY_MIN_IMAGE_SCORE=0.70
|
||||
#IMAGE_IDENTIFY_MIN_TEXT_SCORE=0.60
|
||||
# An image match is only "confirmed" (match_confidence) when it also leads the
|
||||
# best different photo by this much.
|
||||
#IMAGE_SEARCH_MIN_MARGIN=0.05
|
||||
# Diagnosis only: save every image-search request as JSON for
|
||||
# `python -m scripts.replay_image_query`. Blank = off; keeps the newest N.
|
||||
#IMAGE_SEARCH_CAPTURE_DIR=/app/data/image_search_requests
|
||||
#IMAGE_SEARCH_CAPTURE_MAX=200
|
||||
|
||||
# Capture-to-catalog: when identify cannot confirm a photo, read the label and -
|
||||
# for a brand already in the catalog - add the product through the 11-stage
|
||||
# pipeline in the background (validation_status=needs_review). The response
|
||||
# carries a provisional card and a job id to poll at
|
||||
# GET /api/search/identify/jobs/{id}. OFF by default: it is the only way a
|
||||
# public route writes to a brand table. See app/services/capture_discovery.py.
|
||||
ENABLE_CAPTURE_DISCOVERY=false
|
||||
#CAPTURE_DIR=/app/data/captures
|
||||
#CAPTURE_QUEUE_MAX=8
|
||||
#CAPTURE_MAX_PER_CLIENT_PER_HOUR=20
|
||||
# The API's public base URL. Set it to let the colleague's photo become the
|
||||
# product image when the web search finds none; blank = never.
|
||||
#CAPTURE_PUBLIC_BASE_URL=https://api.example.com
|
||||
#CAPTURE_RETAIL_CHECK=true
|
||||
#CAPTURE_USE_LLM=true
|
||||
|
||||
# Product SKU: try a live web search for a real marketplace product ID
|
||||
# (Amazon ASIN, Flipkart PID, etc.) before falling back to an internal SKU.
|
||||
# Set to false to always generate internal SKUs only (faster, offline-safe).
|
||||
@@ -189,10 +400,24 @@ ENABLE_PER_VARIANT_IMAGES=true
|
||||
PER_VARIANT_IMAGE_MAX_RESULTS=10
|
||||
|
||||
# Barcode Retrieval & Product Enrichment (see
|
||||
# app/services/enrichment/barcode/ and docs/BARCODE_ENRICHMENT.md). Set to
|
||||
# false to disable barcode lookup entirely (rows are stored with
|
||||
# barcode=NULL, barcode_lookup_status="disabled").
|
||||
ENABLE_BARCODE_LOOKUP=true
|
||||
# app/services/enrichment/barcode/ and docs/BARCODE_ENRICHMENT.md).
|
||||
#
|
||||
# INLINE, per product, during ingestion. Defaults FALSE in settings.py and this
|
||||
# file used to ship `true`, which disagreed with the code for as long as both
|
||||
# existed - anyone copying .env.example got a very different pipeline from
|
||||
# anyone relying on the defaults. It is false here now to match.
|
||||
#
|
||||
# Leave it false unless you know the upload is small: it costs one search
|
||||
# request per product against an endpoint capped at 10 requests/minute, so a
|
||||
# 200-row sheet is twenty minutes of held request. ENRICH_BARCODES_ON_UPLOAD
|
||||
# below is the cheap path and is on by default.
|
||||
ENABLE_BARCODE_LOOKUP=false
|
||||
|
||||
# BULK, per brand, after the upload settles. Fetches each brand's whole Open
|
||||
# Food Facts catalogue (~5 requests) and matches offline, on the enrichment
|
||||
# job's own thread. Runs before the nutrition phase, because a barcode turns a
|
||||
# 0.32-confidence name lookup into a 0.95-confidence exact one.
|
||||
ENRICH_BARCODES_ON_UPLOAD=true
|
||||
BARCODE_LOOKUP_TIMEOUT_SECONDS=10
|
||||
# 30 days, in seconds
|
||||
BARCODE_LOOKUP_CACHE_TTL_SECONDS=2592000
|
||||
|
||||
154
.env.production
Normal file
154
.env.production
Normal file
@@ -0,0 +1,154 @@
|
||||
# Deployment configuration for mcp.nearle.ai.in.
|
||||
#
|
||||
# Committed at the repo owner's instruction so the deploy does not depend on
|
||||
# re-entering config in the Dokploy UI. Everything needed to boot is here; no
|
||||
# environment variables are required in Dokploy any more.
|
||||
#
|
||||
# A real environment variable still overrides anything set here - settings.py
|
||||
# calls load_dotenv() without override=True, so the process environment wins.
|
||||
# That is the escape hatch for changing a value without a commit.
|
||||
#
|
||||
# WHAT IS IN THIS FILE: live database, S3 and Google credentials, and the key
|
||||
# that signs every access token. Anyone with read access to this repository has
|
||||
# all of it, and git history keeps it after any rotation.
|
||||
|
||||
# --- Ports -----------------------------------------------------------------
|
||||
# Dokploy routes the domain to 3000; 8000 is kept for the vite dev proxy and
|
||||
# docker-compose. serve.py binds both.
|
||||
PORTS=3000,8000
|
||||
|
||||
# --- CORS ------------------------------------------------------------------
|
||||
# The FRONTEND's origin, not this API's. Wrong value = the browser blocks every
|
||||
# response while the server logs healthy 200s - which is exactly what happened
|
||||
# here: this was set to catalogue.nearle.ai.in, but the domain Traefik actually
|
||||
# serves is spelled "catalouge". That host does not even resolve, so nothing
|
||||
# pointed at the mistake except a silently failing UI.
|
||||
#
|
||||
# Both spellings are listed so this keeps working if the typo is ever corrected
|
||||
# in Dokploy. Exact origins, never a wildcard: the app sends an Authorization
|
||||
# header, and browsers reject credentialed requests to a wildcard origin.
|
||||
#
|
||||
# app.nearledaily.com is the merchant console, which calls /api/nutrition/*
|
||||
# from the browser. Until it was listed here every one of those calls failed
|
||||
# as an opaque "Failed to fetch" while curl returned 200 - the server saw a
|
||||
# healthy request and the browser discarded the response. localhost:3100 is
|
||||
# that console in local development.
|
||||
API_CORS_ORIGINS=https://catalouge.nearle.ai.in,https://catalogue.nearle.ai.in,https://app.nearledaily.com,http://localhost:3100
|
||||
|
||||
# --- Authentication --------------------------------------------------------
|
||||
AUTH_ENABLED=true
|
||||
|
||||
# One interactive account: admin. The `user` account is disabled here by
|
||||
# leaving AUTH_USER_PASSWORD_HASH unset - auth.py omits any account whose
|
||||
# hash is empty, so only admin can sign in.
|
||||
#
|
||||
# AUTH_SECRET_KEY stays as generated for this deployment; rotating it would
|
||||
# invalidate every token already issued.
|
||||
# Sign-in password for the hash below: admin / admin123.
|
||||
AUTH_SECRET_KEY=4Kmyr4Cjf_kdUIq_4EGxo5vFHfCT5_uKVR3eouszB8Le6F0n45m7eDY94_KJoqSz
|
||||
AUTH_ADMIN_USERNAME=admin
|
||||
AUTH_ADMIN_PASSWORD_HASH=pbkdf2_sha256$600000$S28AccXqQnNNElilb0JFsg==$IGPLr56iqwkwbrM5skVoXBMDFEfpZIKUA7NiaM74cmk=
|
||||
# AUTH_USER_USERNAME=user
|
||||
# AUTH_USER_PASSWORD_HASH= (unset: the `user` account is disabled)
|
||||
|
||||
AUTH_TOKEN_TTL_MINUTES=720
|
||||
AUTH_MAX_LOGIN_ATTEMPTS=10
|
||||
AUTH_LOCKOUT_SECONDS=300
|
||||
|
||||
# MUST stay false here. The development .env has this true, where it is a
|
||||
# convenience: it skips the password check entirely, so any username signs in
|
||||
# and `admin` gets the admin pages. On a host published to the internet it means
|
||||
# anyone who finds mcp.nearle.ai.in signs in as admin by typing anything at all.
|
||||
AUTH_ALLOW_ANY_LOGIN=false
|
||||
|
||||
# Machine consumers. `name:role:secret` triples, comma-separated; keyed by the
|
||||
# secret, so deleting one entry revokes exactly one caller and leaves the rest
|
||||
# working. Role MUST be `admin` for the store-catalog / catalog-generate /
|
||||
# training routes: those guard with require_admin, which is a ROLE check, and
|
||||
# `admin` is a superuser - a key here unlocks every admin endpoint, not just
|
||||
# the one it was issued for.
|
||||
#
|
||||
# DELIBERATELY EMPTY HERE. The real value lives in the Dokploy Environment tab:
|
||||
#
|
||||
# 1. This file is committed. It already carries the DB password, S3 keys and
|
||||
# the token-signing secret; a per-consumer API key is the one credential
|
||||
# that gets issued and revoked often, and it does not belong in git.
|
||||
# 2. Dockerfile does `COPY .env.production .env`, so a value here is baked at
|
||||
# BUILD time - issuing or revoking a key would mean rebuilding an image
|
||||
# that installs CPU torch, which has already failed once on disk space.
|
||||
# settings.py calls load_dotenv() WITHOUT override=True, so the tab's value
|
||||
# wins and takes effect on a plain restart.
|
||||
#
|
||||
# Consequence: /api/health reports api_keys_source "process-env" for this one,
|
||||
# and that is correct here, not a warning. A value set below would be silently
|
||||
# ignored while the tab is populated - so leave it empty.
|
||||
API_KEYS=
|
||||
|
||||
# --- Postgres / pgvector ---------------------------------------------------
|
||||
# DB_NAME is not set in the development .env, so it falls back to settings.py's
|
||||
# default. Stated explicitly here so the deployment does not depend on that
|
||||
# default staying the same.
|
||||
USE_PGVECTOR=true
|
||||
DB_HOST=31.97.228.132
|
||||
DB_PORT=6054
|
||||
DB_NAME=pgvector
|
||||
DB_USER=admin
|
||||
# The single quotes are PART OF THE PASSWORD, not shell/dotenv syntax. The outer
|
||||
# double quotes are what dotenv strips, leaving 'Package@321#' including quotes.
|
||||
# Writing it bare as Package@321# is what made every connection fail with
|
||||
# "password authentication failed for user admin", which surfaces as
|
||||
# /api/health reporting "database": false and an empty catalog on every page -
|
||||
# the API looks healthy and the database looks empty. Do not "tidy" the quotes.
|
||||
DB_PASSWORD="'Package@321#'"
|
||||
|
||||
# --- Embeddings ------------------------------------------------------------
|
||||
USE_EMBEDDINGS=true
|
||||
EMBEDDINGS_MODEL=sentence-transformers/all-MiniLM-L6-v2
|
||||
EMBEDDINGS_DIM=384
|
||||
|
||||
# --- Ollama (local LLM, powers /api/chat) ----------------------------------
|
||||
# Off, because the development value (http://localhost:11434) cannot work from
|
||||
# inside a container: there, localhost is the container itself, not the VPS
|
||||
# host. Left on with nothing listening, /api/chat fails AND every healthcheck
|
||||
# takes ~3s longer, because the health handler probes Ollama with a 3s timeout.
|
||||
#
|
||||
# To enable: set USE_OLLAMA=true and point OLLAMA_BASE_URL at something the
|
||||
# container can actually reach - http://host.docker.internal:11434 with a
|
||||
# host-gateway mapping, the VPS's LAN IP, or an ollama service name.
|
||||
USE_OLLAMA=false
|
||||
OLLAMA_BASE_URL=http://host.docker.internal:11434
|
||||
OLLAMA_MODEL_NAME=qwen2.5:1.5b
|
||||
OLLAMA_TIMEOUT_SECONDS=120
|
||||
|
||||
# --- DigitalOcean Spaces (product image storage) ---------------------------
|
||||
USE_S3=true
|
||||
S3_ACCESS_KEY=DO801G8Q8JAZKF49U3WJ
|
||||
S3_SECRET_KEY=lBQExYfkVqH+ybmGVmQH5MkThBbrIohA/VQLgcPUvug
|
||||
S3_ENDPOINT=https://nearle.sgp1.digitaloceanspaces.com
|
||||
S3_BUCKET=nearle
|
||||
S3_REGION=sgp1
|
||||
|
||||
# --- Google Custom Search (optional image source) --------------------------
|
||||
USE_GOOGLE_CSE=true
|
||||
GOOGLE_API_KEY=AIzaSyBY4pIO_Fp5FCMqeVxDNcfalzdWNHJWVn0
|
||||
GOOGLE_CSE_ID=9745cbd96dd164562
|
||||
|
||||
# --- Open-source image sources (no key needed) -----------------------------
|
||||
USE_DDG_IMAGES=true
|
||||
USE_OPEN_FACTS=true
|
||||
USE_WIKIMEDIA=true
|
||||
# The Playwright browser binary is NOT installed in the image (see Dockerfile),
|
||||
# so this tier is skipped at runtime regardless. false stops it being attempted.
|
||||
USE_PLAYWRIGHT_FALLBACK=false
|
||||
|
||||
MIN_IMAGE_BYTES=3000
|
||||
|
||||
# --- Product validation ----------------------------------------------------
|
||||
ENABLE_PRODUCT_VALIDATION=true
|
||||
VALIDATION_REJECT_THRESHOLD=0.35
|
||||
VALIDATION_REVIEW_THRESHOLD=0.70
|
||||
|
||||
# --- RAG -------------------------------------------------------------------
|
||||
RAG_DEFAULT_TOP_K=5
|
||||
RAG_MAX_TOP_K=15
|
||||
RAG_MAX_CONTEXT_CHARS=4000
|
||||
23
.gitattributes
vendored
Normal file
23
.gitattributes
vendored
Normal file
@@ -0,0 +1,23 @@
|
||||
# Line endings, pinned so they do not depend on each developer's Git config.
|
||||
#
|
||||
# Git for Windows ships core.autocrlf=true in its system gitconfig, which
|
||||
# stores LF but checks out CRLF - the source of the "LF will be replaced by
|
||||
# CRLF" warnings on `git add`. Declaring the policy here overrides that for
|
||||
# everyone, so Windows, Linux and the Docker build all agree.
|
||||
#
|
||||
# The repository is already entirely LF (`git ls-files --eol` shows i/lf
|
||||
# across the board), so eol=lf pins the working tree to what the index
|
||||
# already holds and rewrites no committed content.
|
||||
* text=auto eol=lf
|
||||
|
||||
# Binaries must never be EOL-converted - a substituted byte corrupts them.
|
||||
# Git already auto-detects these; saying so explicitly means a future edit
|
||||
# to the wildcard above cannot silently start mangling them.
|
||||
*.joblib binary
|
||||
*.db binary
|
||||
*.tflite binary
|
||||
|
||||
# Windows-only scripts genuinely need CRLF. None are tracked in this repo
|
||||
# today (start_app.bat lives above it), but the rule belongs with the policy.
|
||||
*.bat text eol=crlf
|
||||
*.ps1 text eol=crlf
|
||||
44
.gitignore
vendored
44
.gitignore
vendored
@@ -1,6 +1,26 @@
|
||||
# Secrets - never commit
|
||||
# .env.production is committed deliberately, at the repo owner's instruction, so
|
||||
# the deployment does not depend on re-entering config in the Dokploy UI.
|
||||
#
|
||||
# It is NOT named .env, and that matters: Dokploy writes its own .env into the
|
||||
# build context from the service's Environment tab after cloning, so a committed
|
||||
# .env is silently replaced (with an empty file when that tab is blank).
|
||||
# The Dockerfile copies .env.production to /app/.env inside the image.
|
||||
#
|
||||
# The two database secrets are NOT in it - they are set as Dokploy environment
|
||||
# variables, which override the file (settings.py calls load_dotenv() without
|
||||
# override=True, so the process environment wins).
|
||||
#
|
||||
# The auth secrets ARE in it. AUTH_SECRET_KEY signs every access token, so
|
||||
# anyone with read access to this repository can mint an admin token, and git
|
||||
# history keeps it after any rotation. Regenerate with
|
||||
# `python scripts/make_auth_secrets.py` if that stops being acceptable.
|
||||
!.env.production
|
||||
.env
|
||||
|
||||
# Local overrides and the generated sign-in passwords stay out of git.
|
||||
.env.local
|
||||
SIGNIN_PASSWORDS.txt
|
||||
|
||||
# Python
|
||||
__pycache__/
|
||||
*.pyc
|
||||
@@ -16,3 +36,25 @@ venv/
|
||||
# OS
|
||||
.DS_Store
|
||||
Thumbs.db
|
||||
|
||||
# Dagster (orchestration layer, development only)
|
||||
# Run/event history, compute logs and the pickled asset outputs from the
|
||||
# filesystem IO manager. Regenerated on demand; nothing here is source.
|
||||
orchestration/.dagster_home/storage/
|
||||
orchestration/.dagster_home/history/
|
||||
orchestration/.dagster_home/logs/
|
||||
orchestration/.dagster_home/schedules/
|
||||
orchestration/.dagster_home/*.db
|
||||
orchestration/.dagster_home/*.db-*
|
||||
orchestration/.dagster_home/.telemetry/
|
||||
# dagster.yaml IS committed - it is the instance configuration.
|
||||
# .env.orchestration IS committed - it holds no secrets, only the local
|
||||
# database pin that keeps orchestrated writes off production.
|
||||
|
||||
# Spreadsheets sent to the catalog ingestion endpoints, plus their
|
||||
# manifests. This is operator data, not source: it is whatever a colleague
|
||||
# happened to send, it can contain a store's real pricing, and in a container
|
||||
# it lives on the /app/data volume rather than in the image. Committing it
|
||||
# would put customer files in the repository permanently.
|
||||
# BATCH_UPLOAD_DIR overrides the location; this covers the default.
|
||||
data/batch_uploads/
|
||||
|
||||
111
Dockerfile
111
Dockerfile
@@ -21,8 +21,82 @@ WORKDIR /app
|
||||
# image-search fallback (see requirements.txt); run `playwright install
|
||||
# chromium` in the container if you need that specific fallback tier.
|
||||
COPY requirements.txt .
|
||||
RUN python -m venv /opt/venv \
|
||||
&& /opt/venv/bin/pip install --no-cache-dir -r requirements.txt
|
||||
COPY requirements-ocr.txt .
|
||||
|
||||
# torch is installed FIRST, from PyTorch's CPU-only index, and that ordering is
|
||||
# the point. sentence-transformers pulls torch in transitively, and pip's
|
||||
# default wheel for Linux bundles the entire CUDA stack - cuBLAS, cuDNN, NVRTC,
|
||||
# Triton - because it cannot know the target has no GPU. Measured in this image
|
||||
# it was 2.7GB of nvidia/ plus 691MB of triton/, none of which can ever execute
|
||||
# on a CPU-only VPS, in a 9.2GB image on a 48GB disk shared with a dozen
|
||||
# services. A build here has already failed once on "no space left on device".
|
||||
#
|
||||
# Installing it up front means the requirements.txt pass below finds torch
|
||||
# already satisfied and leaves it alone. Keep the two installs in that order.
|
||||
#
|
||||
# These three steps are also deliberately SEPARATE `RUN` layers rather than one
|
||||
# chained command, and that is a deployment fix rather than tidiness. BuildKit
|
||||
# caches COMPLETED steps, so as a single chained RUN there was no resume point:
|
||||
# a build killed partway through torch redid the venv, the 191MB download and
|
||||
# the unpack from zero on every retry. A Dokploy deploy died exactly there -
|
||||
# "#8 CANCELED / failed to solve: Canceled: context canceled", with no pip
|
||||
# traceback and no exit code, which is BuildKit reporting that the process
|
||||
# driving the build went away, not that pip failed. `Installing collected
|
||||
# packages` unpacks torch to ~1.3GB while the 191MB wheel is still on disk, so
|
||||
# peak memory and peak disk land in the same second; DEPLOYMENT.md's Sizing
|
||||
# section already warned that under 4GB "the build itself will fail". Split,
|
||||
# the torch layer is banked the first time it succeeds and no later deploy
|
||||
# runs it at all.
|
||||
RUN python -m venv /opt/venv
|
||||
|
||||
# PINNED for the same reason. Unpinned, `pip install torch` took whatever the
|
||||
# CPU index served that day - 2.13.0+cpu on the run that failed - so the layer
|
||||
# below could never be trusted as cached and the image's size was set by a
|
||||
# moving target.
|
||||
#
|
||||
# 2.12.1 rather than the newest available, because it is the version this
|
||||
# project is actually developed and tested against - the working venv here runs
|
||||
# torch 2.12.1 with sentence-transformers 5.6.0 - rather than whatever shipped
|
||||
# most recently. The `+cpu` local version is part of the specifier: it is how
|
||||
# the wheel is named on this index, and a bare `torch==2.12.1` would not match.
|
||||
RUN /opt/venv/bin/pip install --no-cache-dir \
|
||||
--index-url https://download.pytorch.org/whl/cpu torch==2.12.1+cpu
|
||||
|
||||
RUN /opt/venv/bin/pip install --no-cache-dir -r requirements.txt
|
||||
|
||||
# The OCR engine goes in AFTER requirements.txt and WITHOUT its declared
|
||||
# dependencies. rapidocr's metadata asks for opencv-python (the GUI build);
|
||||
# letting pip honour that would unpack it over opencv-python-headless and the
|
||||
# result fails on libGL.so.1 in this image - taking the img_vector embedder
|
||||
# down with it. Its real dependencies are in requirements.txt already; its
|
||||
# PP-OCR models are inside the wheel, so nothing is downloaded at runtime.
|
||||
RUN /opt/venv/bin/pip install --no-cache-dir --no-deps -r requirements-ocr.txt
|
||||
|
||||
# Strip payload the running service can never execute. Doing this in the build
|
||||
# stage is what makes it count: the runtime stage copies /opt/venv as one layer,
|
||||
# so anything deleted after that COPY would still occupy space in the layer
|
||||
# below it. Deleting it here means the bytes are never in the final image.
|
||||
#
|
||||
# Measured on this image, the venv was 1.65GB, and this removes ~350MB of it:
|
||||
#
|
||||
# - Bundled test suites (~237MB, of which torch/test alone is 83MB). Every
|
||||
# scientific wheel ships its own; pandas/tests is 40MB. Matched on exactly
|
||||
# `tests`/`test` so numpy.testing and sklearn.utils._testing - which ARE
|
||||
# imported by library code at runtime - are left alone.
|
||||
# - torch/include (62MB): C++ headers, needed only to COMPILE an extension
|
||||
# against libtorch. Nothing here does; torch is used through Python.
|
||||
# - torch/bin (50MB): C++ gtest binaries (test_api is 15MB, test_jit 13.5MB)
|
||||
# plus a protoc. torch_shm_manager is the one real program in there - it
|
||||
# brokers shared-memory tensors between processes - so it is kept.
|
||||
#
|
||||
# Verified against this app rather than assumed: sympy IS pulled in by `import
|
||||
# sentence_transformers` (via torch.fx), so it stays despite being 80MB and
|
||||
# looking like a pure-math dependency nothing here would want.
|
||||
RUN set -eux; \
|
||||
SP=/opt/venv/lib/python3.11/site-packages; \
|
||||
find "$SP" -type d \( -name tests -o -name test \) -prune -exec rm -rf {} +; \
|
||||
rm -rf "$SP/torch/include"; \
|
||||
find "$SP/torch/bin" -type f ! -name torch_shm_manager -delete
|
||||
|
||||
|
||||
# ---- Runtime stage ----
|
||||
@@ -46,6 +120,23 @@ COPY scripts ./scripts
|
||||
COPY data ./data
|
||||
COPY serve.py .
|
||||
|
||||
# The deployment's configuration, landing at /app/.env because settings.py
|
||||
# resolves it from the backend root - app/infrastructure/settings.py takes
|
||||
# parents[2], which is /app here - so it must sit next to app/, not inside it.
|
||||
#
|
||||
# The source file is named .env.production, NOT .env, and that detail is the
|
||||
# whole point. Dokploy writes its own .env into the build context from the
|
||||
# service's Environment tab AFTER cloning the repository. With that tab empty it
|
||||
# writes an empty file, overwriting the committed one - so `COPY .env .` copied
|
||||
# a zero-byte file, the container started with no configuration at all, and the
|
||||
# platform reported only a Bad Gateway. The checkout showed it plainly: every
|
||||
# file timestamped 08:33, and .env alone at 08:34, 0 bytes.
|
||||
#
|
||||
# Dokploy does not manage .env.production, so it survives. Anything set in the
|
||||
# Environment tab still wins at runtime, because settings.py calls load_dotenv()
|
||||
# without override=True and the process environment takes precedence.
|
||||
COPY .env.production .env
|
||||
|
||||
# Pristine copies of everything the app also WRITES to, kept at a path that is
|
||||
# never mounted over.
|
||||
#
|
||||
@@ -81,14 +172,16 @@ VOLUME ["/app/data", "/app/app/intelligence/artifacts"]
|
||||
ENV PORTS=3000,8000
|
||||
EXPOSE 3000 8000
|
||||
|
||||
# Liveness only, and passes if EITHER port answers. /api/health always returns
|
||||
# 200 - it reports Postgres and Ollama in the body as "degraded" rather than
|
||||
# failing - which is deliberate: a check that went red whenever Postgres blinked
|
||||
# would have Dokploy restart a perfectly healthy API in a loop.
|
||||
# Liveness only, and passes if EITHER port answers.
|
||||
#
|
||||
# The 10s timeout is not padding: the handler probes Ollama over HTTP with a 3s
|
||||
# timeout of its own, so an unreachable Ollama makes every check take ~3s.
|
||||
# start-period covers first boot, where the venv is still cold.
|
||||
# It probes "/", which is served from memory, NOT /api/health, which dials
|
||||
# Postgres and Ollama. That is the whole point: the platform's response to a
|
||||
# failed healthcheck is to stop routing traffic, so this may only ask "is the
|
||||
# process still serving HTTP". Tying it to the database meant an unreachable
|
||||
# Postgres blocked the handler for the OS TCP timeout, the check timed out, the
|
||||
# container was marked unhealthy, and a perfectly healthy API returned Bad
|
||||
# Gateway on every route. Use /api/health to ask whether dependencies are up;
|
||||
# it reports them in the body and always answers 200.
|
||||
HEALTHCHECK --interval=30s --timeout=10s --start-period=40s --retries=3 \
|
||||
CMD ["python", "serve.py", "--healthcheck"]
|
||||
|
||||
|
||||
103
README.md
103
README.md
@@ -32,7 +32,7 @@ It does **not** start Postgres or Ollama for you - it reports them via
|
||||
cd backend
|
||||
python3 -m venv .venv
|
||||
source .venv/bin/activate # Windows: .venv\Scripts\activate
|
||||
pip install -r requirements.txt
|
||||
pip install -r requirements.txt -r requirements-dev.txt # -dev is pytest only
|
||||
|
||||
cp .env.example .env # then edit DB_PASSWORD etc.
|
||||
|
||||
@@ -52,6 +52,58 @@ uvicorn app.main:app --reload --port 8000
|
||||
Then open http://localhost:8000/docs for interactive API docs, or run the
|
||||
frontend (`../frontend/README.md`) to use the React UI.
|
||||
|
||||
## Active brands (development working set)
|
||||
|
||||
One setting decides which brands the application and every pipeline work on:
|
||||
|
||||
```
|
||||
# backend/.env
|
||||
ACTIVE_BRANDS=Amul,Cadbury,Hindustan Unilever
|
||||
```
|
||||
|
||||
**Blank or unset means every brand is active** - that is the production default
|
||||
and the way to turn the feature off.
|
||||
|
||||
Nothing is deleted when it is set. The other `brand_*` tables and their
|
||||
embeddings stay in Postgres untouched; they simply stop being discovered.
|
||||
Everything downstream inherits it, because the whole app funnels through two
|
||||
functions in `app/services/vector_store.py`
|
||||
(`list_available_brands` and `_list_brand_table_suffixes`):
|
||||
|
||||
```
|
||||
ACTIVE_BRANDS
|
||||
|
|
||||
+-- /api/brands, /api/products, /api/search, /api/suggest
|
||||
+-- RAG chat and the query_intent brand index
|
||||
+-- MCP tools
|
||||
+-- nutrition enrichment, store intelligence, analytics
|
||||
+-- the boot auto-seed and the 300s brand reconcile
|
||||
+-- every Dagster asset and partition
|
||||
```
|
||||
|
||||
Going from 3 brands to 5, 10 or all of them is an edit to this one line - no
|
||||
code changes. Catalogs for inactive brands live in
|
||||
`data/seed_catalogs/archive/`, which the loader still reads, so re-activating a
|
||||
brand does not require moving files back.
|
||||
|
||||
Names resolve through the brand aliases, so `ACTIVE_BRANDS=Tata` activates
|
||||
`brand_hindustan_unilever` - the same table an ingest of Tata products targets.
|
||||
|
||||
## Orchestration (Dagster)
|
||||
|
||||
Ingestion, enrichment, embedding and ML training can be run as a Dagster asset
|
||||
graph, with lineage, retries and run history. It is a **development tool** -
|
||||
FastAPI still serves every request and nothing in a request path touches it.
|
||||
|
||||
```
|
||||
pip install -r requirements-orchestration.txt
|
||||
DAGSTER_HOME="$(pwd)/orchestration/.dagster_home" dagster dev -m orchestration.definitions -p 3030
|
||||
```
|
||||
|
||||
See `orchestration/README.md`. Note that it pins itself to the **local**
|
||||
database and refuses to write to a remote one, because `backend/.env` points at
|
||||
production.
|
||||
|
||||
## Authentication
|
||||
|
||||
Reads are public; the 18 write/compute endpoints require a credential, enforced
|
||||
@@ -65,10 +117,52 @@ curl -X POST localhost:8000/api/auth/login \
|
||||
|
||||
Send it as `Authorization: Bearer <token>`, or use an `X-API-Key` from the
|
||||
`API_KEYS` setting for server-to-server callers. `admin` passes every
|
||||
permission check; `user` holds the product/store/inventory permissions.
|
||||
permission check; `user` holds the product/store/inventory permissions;
|
||||
`uploader` holds exactly one, `upload_catalog` (see below).
|
||||
`AUTH_ENABLED=false` disables all of it for local work — never in a deployment.
|
||||
See the Authentication section of `../DEPLOYMENT.md` for the full endpoint map.
|
||||
|
||||
## Catalog ingestion for API clients
|
||||
|
||||
An outside party can send spreadsheets straight into the catalog pipeline
|
||||
without an admin in the loop. Issue them an `uploader` key:
|
||||
|
||||
```bash
|
||||
python scripts/make_auth_secrets.py --api-key catalog-drop:uploader
|
||||
# -> API_KEYS=catalog-drop:uploader:<secret> (add to .env, redeploy)
|
||||
```
|
||||
|
||||
Name the key for its function, not its holder: `/api/health` publicly reports
|
||||
every key's name, role and fingerprint (never the secret).
|
||||
|
||||
They then POST files and poll the batch:
|
||||
|
||||
```bash
|
||||
curl -X POST https://mcp.nearle.ai.in/api/uploads/catalog \
|
||||
-H 'X-API-Key: <secret>' \
|
||||
-F 'files=@store-catalog.xlsx' -F 'files=@second-store.xlsx'
|
||||
# -> 202 {"batch_id": "...", "status": "queued", "files": [...], "message": "..."}
|
||||
|
||||
curl https://mcp.nearle.ai.in/api/uploads/catalog/<batch_id> -H 'X-API-Key: <secret>'
|
||||
# -> {"status": "running", "files_done": 1, "files": [{"stage_name": "...", ...}]}
|
||||
```
|
||||
|
||||
`.xlsx`, `.xls` and `.csv` are accepted, up to 10MB / 2000 rows per file and
|
||||
`BATCH_MAX_FILES` files per request. Each file runs the same 11 stages as the
|
||||
admin route (`app/core/store_catalog_pipeline.py`). A sheet that cannot be
|
||||
parsed — or that has no product-name column — is rejected during the request
|
||||
with a 400 naming the problem, so the sender finds out while they can still fix
|
||||
it; a bad file alongside good ones comes back in `files` as `status: "failed"`
|
||||
while the rest still run.
|
||||
|
||||
Two things this credential cannot do. It cannot see anything but its own
|
||||
submissions — every read is filtered by `submitted_by`, so it reaches neither
|
||||
the catalog nor another caller's batches — and it cannot multiply the work:
|
||||
all ingestion, from every source, goes through one worker thread behind a queue
|
||||
of `BATCH_QUEUE_MAX`, past which the endpoint answers 429. Cancel, resume and
|
||||
the full batch list stay on the admin router
|
||||
(`/api/admin/catalog-batch/...`, `require_admin`).
|
||||
|
||||
## MCP server
|
||||
|
||||
The catalog is exposed to AI clients over the Model Context Protocol at `/mcp`,
|
||||
@@ -92,7 +186,7 @@ long-running client has to refresh. Raise the TTL if that is a problem.
|
||||
"mcpServers": {
|
||||
"nearle-catalogue": {
|
||||
"type": "http",
|
||||
"url": "https://mcp.catalogue.nearle.ai.in/mcp",
|
||||
"url": "https://mcp.nearle.ai.in/mcp",
|
||||
"headers": { "Authorization": "Bearer <token>" }
|
||||
}
|
||||
}
|
||||
@@ -198,7 +292,8 @@ backend/
|
||||
├── cli/ingest_brand.py # CLI: ingest one brand end-to-end
|
||||
├── scripts/seed_sample_data.py # load bundled sample catalogs (no LLM needed)
|
||||
├── data/seed_catalogs/*.json # bundled sample catalogs (Parle, Cadbury, ...)
|
||||
└── requirements.txt
|
||||
├── requirements.txt
|
||||
└── requirements-dev.txt
|
||||
```
|
||||
|
||||
## Running tests
|
||||
|
||||
421
app/api/batch_common.py
Normal file
421
app/api/batch_common.py
Normal file
@@ -0,0 +1,421 @@
|
||||
"""Shared machinery for the two routes that start a catalog batch.
|
||||
|
||||
app/api/routers/batch_catalog.py POST /api/admin/catalog-batch/ingest
|
||||
app/api/routers/uploads.py POST /api/uploads/catalog
|
||||
|
||||
Both accept spreadsheets, both run the same 11 stages over them, and both hand
|
||||
back a batch id to poll. What differs is only who may call them and what the
|
||||
caller is allowed to see afterwards - so everything between "read the upload"
|
||||
and "queue the batch" lives here instead of being written twice and drifting.
|
||||
|
||||
WHY THE LIMITS ARE ARGUMENTS RATHER THAN IMPORTS
|
||||
------------------------------------------------
|
||||
`read_uploads` and `parse_all` take an `UploadLimits` instead of reading the
|
||||
settings themselves. Each router builds one from ITS OWN module globals, at
|
||||
call time, which is what keeps
|
||||
|
||||
monkeypatch.setattr(batch_catalog, "BATCH_MAX_FILES", 2)
|
||||
|
||||
working - the idiom the existing suite is written in. Had this module read the
|
||||
settings directly, those patches would become silently inert and a limit test
|
||||
that no longer exercises its limit would still pass.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import queue
|
||||
from dataclasses import dataclass
|
||||
from typing import List, Optional, Tuple
|
||||
|
||||
from fastapi import HTTPException, UploadFile
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
from app.api.batch_job_store import batch_job_store
|
||||
from app.core import batch_ingest, batch_worker
|
||||
from app.core import store_catalog_pipeline as pipeline
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# The worker cannot import the API layer without a cycle, so the wiring is done
|
||||
# once, here, at import. This module is imported by every router that can start
|
||||
# a batch, which is why it is the right place: whichever of them loads first,
|
||||
# the worker is configured before anything can be submitted to it.
|
||||
batch_worker.configure(
|
||||
on_change=batch_job_store.put,
|
||||
should_cancel=batch_job_store.is_cancelled,
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class UploadLimits:
|
||||
"""Ceilings for one submission. Per-file first, then per-batch."""
|
||||
|
||||
max_files: int
|
||||
max_file_bytes: int
|
||||
max_file_rows: int
|
||||
max_total_bytes: int
|
||||
max_total_rows: int
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Reading and parsing
|
||||
# ---------------------------------------------------------------------------
|
||||
async def read_uploads(
|
||||
files: List[UploadFile], limits: UploadLimits
|
||||
) -> List[Tuple[str, bytes]]:
|
||||
"""Read every upload into memory, enforcing the count and size ceilings.
|
||||
|
||||
Read here rather than in the worker because `UploadFile` is backed by a
|
||||
temporary file tied to the request: FastAPI closes it when the response is
|
||||
returned, so a background thread reaching for it later finds nothing. The
|
||||
bytes have to be taken while the request is still alive.
|
||||
"""
|
||||
if not files:
|
||||
raise HTTPException(status_code=400, detail="No files were uploaded.")
|
||||
if len(files) > limits.max_files:
|
||||
raise HTTPException(
|
||||
status_code=413,
|
||||
detail=(
|
||||
f"{len(files)} files exceeds the {limits.max_files}-file limit for one "
|
||||
f"batch. Split the drop and send it in two."
|
||||
),
|
||||
)
|
||||
|
||||
read: List[Tuple[str, bytes]] = []
|
||||
total = 0
|
||||
for upload in files:
|
||||
contents = await upload.read()
|
||||
name = upload.filename or "upload.xlsx"
|
||||
if not contents:
|
||||
# Recorded rather than raised - an empty file among nine good ones
|
||||
# is a fact about that file, not a reason to reject the drop.
|
||||
read.append((name, b""))
|
||||
continue
|
||||
if len(contents) > limits.max_file_bytes:
|
||||
raise HTTPException(
|
||||
status_code=413,
|
||||
detail=(
|
||||
f"'{name}' is larger than the "
|
||||
f"{limits.max_file_bytes // (1024 * 1024)}MB per-file limit."
|
||||
),
|
||||
)
|
||||
total += len(contents)
|
||||
if total > limits.max_total_bytes:
|
||||
raise HTTPException(
|
||||
status_code=413,
|
||||
detail=(
|
||||
f"The batch is larger than the "
|
||||
f"{limits.max_total_bytes // (1024 * 1024)}MB total limit."
|
||||
),
|
||||
)
|
||||
read.append((name, contents))
|
||||
return read
|
||||
|
||||
|
||||
def parse_all(read: List[Tuple[str, bytes]], limits: UploadLimits):
|
||||
"""Split the uploads into (valid, invalid) by trying to parse each one.
|
||||
|
||||
Parsing up front is what stops a batch transitioning straight to "failed" a
|
||||
second after it started: an unreadable sheet is rejected in the response the
|
||||
caller is still waiting on, not in a manifest they have to go and poll for.
|
||||
Applied per file, so one bad sheet does not condemn the others.
|
||||
"""
|
||||
valid: List[Tuple[str, bytes, int]] = []
|
||||
invalid: List[Tuple[str, str]] = []
|
||||
rows_total = 0
|
||||
|
||||
for name, contents in read:
|
||||
if not contents:
|
||||
invalid.append((name, "The file is empty."))
|
||||
continue
|
||||
try:
|
||||
df, mapping = pipeline.parse_spreadsheet(name, contents)
|
||||
except HTTPException as exc:
|
||||
# read_products_dataframe raises HTTPException for an unsupported
|
||||
# extension or a missing Excel reader; its message already names the
|
||||
# file and says what to do about it.
|
||||
invalid.append((name, str(exc.detail)))
|
||||
continue
|
||||
except Exception as exc: # noqa: BLE001 - an unreadable sheet is caller error
|
||||
invalid.append((name, f"Could not parse the file: {exc}"))
|
||||
continue
|
||||
|
||||
if df.empty:
|
||||
invalid.append((name, "The file has no data rows."))
|
||||
continue
|
||||
|
||||
# A sheet whose headers carry no product name is not a catalog, and
|
||||
# without this it is accepted with a 202 and then ingests nothing - the
|
||||
# worst possible answer, because it looks like success from every angle
|
||||
# the caller can see. `user_products.py` has always made this check on
|
||||
# its own upload path; the catalog paths did not, and an API client
|
||||
# sending the wrong export is the likeliest mistake there is.
|
||||
if "product_name" not in mapping.columns:
|
||||
recognised = ", ".join(sorted(mapping.columns)) or "none"
|
||||
invalid.append((
|
||||
name,
|
||||
f"No product name column was found. Headers read: "
|
||||
f"{', '.join(str(c) for c in df.columns)}. Recognised fields: "
|
||||
f"{recognised}.",
|
||||
))
|
||||
continue
|
||||
|
||||
if len(df) > limits.max_file_rows:
|
||||
invalid.append((
|
||||
name,
|
||||
f"{len(df)} rows exceeds the {limits.max_file_rows}-row per-file limit.",
|
||||
))
|
||||
continue
|
||||
|
||||
rows_total += int(len(df))
|
||||
if rows_total > limits.max_total_rows:
|
||||
raise HTTPException(
|
||||
status_code=413,
|
||||
detail=(
|
||||
f"The batch totals more than {limits.max_total_rows} rows. "
|
||||
f"Split it and send it in two."
|
||||
),
|
||||
)
|
||||
valid.append((name, contents, int(len(df))))
|
||||
|
||||
return valid, invalid, rows_total
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Response shape
|
||||
# ---------------------------------------------------------------------------
|
||||
class StageOut(BaseModel):
|
||||
"""One pipeline stage as it happened to one file.
|
||||
|
||||
Kept even in slim list responses: eleven of these per file is a few hundred
|
||||
bytes, unlike the `products` manifest slim exists to drop.
|
||||
"""
|
||||
|
||||
index: int
|
||||
name: str
|
||||
rows_done: int = 0
|
||||
rows_total: int = 0
|
||||
started_at: Optional[float] = None
|
||||
# None while the stage is still running.
|
||||
finished_at: Optional[float] = None
|
||||
|
||||
|
||||
class BatchFileOut(BaseModel):
|
||||
index: int
|
||||
filename: str
|
||||
status: str
|
||||
detail: Optional[str] = None
|
||||
# Set only on a file that was released out of the review inbox: the id of
|
||||
# the run that took it, which is how a sender gets from the drop id they
|
||||
# hold to the batch that carries their results.
|
||||
released_to: Optional[str] = None
|
||||
# The other direction: which drop this file came out of. A run can be
|
||||
# assembled from several drops, and this is the only exact way for a sender
|
||||
# to pick their own file out of one - `filename` is a coincidence, because
|
||||
# two senders can both upload products.csv.
|
||||
from_drop: Optional[str] = None
|
||||
stage_index: int = 0
|
||||
stage_name: str = ""
|
||||
total_stages: int = pipeline.TOTAL_STAGES
|
||||
rows_done: int = 0
|
||||
rows_total: int = 0
|
||||
# The stages this file has been through, so a FINISHED file can still show
|
||||
# its timeline. The scalars above only ever say where it is right now.
|
||||
stages: List[StageOut] = Field(default_factory=list)
|
||||
size_bytes: int = 0
|
||||
started_at: Optional[float] = None
|
||||
finished_at: Optional[float] = None
|
||||
result: Optional[dict] = None
|
||||
|
||||
|
||||
class BatchOut(BaseModel):
|
||||
batch_id: str
|
||||
status: str
|
||||
detail: Optional[str] = None
|
||||
submitted_by: Optional[str] = None
|
||||
created_at: float
|
||||
updated_at: float
|
||||
files_total: int
|
||||
files_done: int
|
||||
files_failed: int
|
||||
current_file: Optional[str] = None
|
||||
use_llm: bool
|
||||
fetch_images: bool
|
||||
# Which executor owns this batch - "inprocess" or "dagster". A dagster batch
|
||||
# sits queued until the orchestrator picks it up, which the UI has to be
|
||||
# able to say out loud rather than showing a run that looks stuck.
|
||||
runner: str = batch_ingest.RUNNER_INPROCESS
|
||||
# The nutrition-enrichment job queued when this batch finished, pollable at
|
||||
# GET /api/admin/nutrition-intelligence/jobs/{job_id}. Surfaced here so the
|
||||
# poll a client is already doing for the ingestion also reveals the scoring
|
||||
# that follows it, instead of leaving the client to guess that it happened.
|
||||
#
|
||||
# None while the batch is still running, when AUTO_ENRICH_ON_UPLOAD is off,
|
||||
# and for every batch that finished before this existed.
|
||||
nutrition_job_id: Optional[str] = None
|
||||
# The 11 stage names, in order, so a client can draw the whole pipeline
|
||||
# before a file has entered any of it. Served rather than duplicated in the
|
||||
# frontend so the two cannot drift when a stage is added.
|
||||
stage_names: List[str] = Field(default_factory=lambda: list(pipeline.STAGE_NAMES))
|
||||
totals: dict
|
||||
brands: List[str]
|
||||
files: List[BatchFileOut]
|
||||
|
||||
|
||||
def to_out(manifest: batch_ingest.BatchManifest, *, slim: bool = False) -> BatchOut:
|
||||
"""Render a manifest for the API.
|
||||
|
||||
`slim=True` drops the per-file `products` manifest, which is for LIST
|
||||
responses. That list can run to thousands of rows per file
|
||||
(store_catalog_pipeline.MAX_REPORTED_PRODUCTS), so twenty batches rendered
|
||||
in full is a multi-megabyte response to a request that only wanted to know
|
||||
what ran lately. The single-batch read keeps it - that is where a caller
|
||||
goes to reconcile a specific run.
|
||||
"""
|
||||
body = manifest.to_dict()
|
||||
files = []
|
||||
for entry in body["files"]:
|
||||
rendered = {key: entry[key] for key in BatchFileOut.model_fields if key in entry}
|
||||
if slim and isinstance(rendered.get("result"), dict):
|
||||
rendered["result"] = {
|
||||
k: v for k, v in rendered["result"].items() if k != "products"
|
||||
}
|
||||
files.append(BatchFileOut(**rendered))
|
||||
body["files"] = files
|
||||
return BatchOut(**{k: v for k, v in body.items() if k in BatchOut.model_fields})
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Staging and queueing
|
||||
# ---------------------------------------------------------------------------
|
||||
def stage_and_queue(
|
||||
valid: List[Tuple[str, bytes, int]],
|
||||
invalid: List[Tuple[str, str]],
|
||||
*,
|
||||
use_llm: bool,
|
||||
fetch_images: bool,
|
||||
submitted_by: Optional[str] = None,
|
||||
) -> Tuple[batch_ingest.BatchManifest, bool]:
|
||||
"""Write the files down, publish the batch, and try to start it.
|
||||
|
||||
Returns `(manifest, started)`. `started` is False only when the worker queue
|
||||
was full: the batch is staged and durable either way, and the caller decides
|
||||
what to say about it - an admin has a Resume button, an API client does not,
|
||||
and the two deserve different words for the same 429.
|
||||
"""
|
||||
manifest = batch_ingest.stage_batch(
|
||||
[(name, contents) for name, contents, _n in valid],
|
||||
use_llm=use_llm,
|
||||
fetch_images=fetch_images,
|
||||
invalid=invalid,
|
||||
submitted_by=submitted_by,
|
||||
)
|
||||
# stage_batch records size but not row counts; it never parsed the files.
|
||||
for entry, (_name, _contents, rows) in zip(manifest.files, valid):
|
||||
entry.rows_total = rows
|
||||
batch_ingest.write_manifest(manifest)
|
||||
batch_job_store.put(manifest)
|
||||
|
||||
try:
|
||||
batch_worker.submit(manifest.batch_id)
|
||||
except queue.Full:
|
||||
manifest.status = batch_ingest.QUEUED
|
||||
manifest.detail = (
|
||||
"The ingestion queue was full when this batch arrived. It is staged "
|
||||
"and can be started with Resume once the running batches finish."
|
||||
)
|
||||
batch_ingest.write_manifest(manifest)
|
||||
batch_job_store.put(manifest)
|
||||
return manifest, False
|
||||
|
||||
return manifest, True
|
||||
|
||||
|
||||
def stage_for_orchestrator(
|
||||
valid: List[Tuple[str, bytes, int]],
|
||||
invalid: List[Tuple[str, str]],
|
||||
*,
|
||||
use_llm: bool,
|
||||
fetch_images: bool,
|
||||
submitted_by: Optional[str] = None,
|
||||
) -> batch_ingest.BatchManifest:
|
||||
"""Publish a batch for Dagster to claim, and hand it to no one else.
|
||||
|
||||
Same staging as `stage_and_queue`, minus the `batch_worker.submit`. The
|
||||
batch is left `queued` and stamped `runner="dagster"`, which is what
|
||||
`_pick_batch_id` and `batch_upload_sensor` filter on; the in-process worker
|
||||
never scans for work, so leaving it unsubmitted is enough to keep the two
|
||||
executors off each other's batches.
|
||||
|
||||
A third function rather than a `runner=` argument on `stage_and_queue`, for
|
||||
the reason given in `stage_pending`: whether work starts here or somewhere
|
||||
else is not the kind of decision that should hang off a boolean anyone can
|
||||
flip later.
|
||||
|
||||
The batch sits queued until an orchestrator actually runs - which, if
|
||||
`dagster dev` is not up, is never. Callers are expected to say so rather
|
||||
than present it as a run in progress.
|
||||
"""
|
||||
manifest = batch_ingest.stage_batch(
|
||||
[(name, contents) for name, contents, _n in valid],
|
||||
use_llm=use_llm,
|
||||
fetch_images=fetch_images,
|
||||
invalid=invalid,
|
||||
submitted_by=submitted_by,
|
||||
)
|
||||
for entry, (_name, _contents, rows) in zip(manifest.files, valid):
|
||||
entry.rows_total = rows
|
||||
manifest.runner = batch_ingest.RUNNER_DAGSTER
|
||||
manifest.detail = (
|
||||
"Waiting for the Dagster orchestrator to pick this batch up."
|
||||
)
|
||||
batch_ingest.write_manifest(manifest)
|
||||
batch_job_store.put(manifest)
|
||||
return manifest
|
||||
|
||||
|
||||
def stage_pending(
|
||||
valid: List[Tuple[str, bytes, int]],
|
||||
invalid: List[Tuple[str, str]],
|
||||
*,
|
||||
submitted_by: Optional[str] = None,
|
||||
) -> batch_ingest.BatchManifest:
|
||||
"""Write the files down and stop. The review inbox half of the flow.
|
||||
|
||||
Deliberately a separate function rather than a `queue=False` flag on
|
||||
`stage_and_queue`: the admin routes must queue unconditionally, and a shared
|
||||
boolean is exactly the kind of default that gets inverted in a later edit and
|
||||
silently starts running work nobody approved.
|
||||
|
||||
`use_llm` / `fetch_images` are not taken here. They are run-time choices, and
|
||||
the person who makes them is the admin pressing Start in the inbox - not the
|
||||
colleague who dropped the file. They are supplied to `stage_and_queue` when
|
||||
the selected files become a real batch.
|
||||
"""
|
||||
manifest = batch_ingest.stage_batch(
|
||||
[(name, contents) for name, contents, _n in valid],
|
||||
invalid=invalid,
|
||||
submitted_by=submitted_by,
|
||||
)
|
||||
# stage_batch records size but not row counts; it never parsed the files.
|
||||
for entry, (_name, _contents, rows) in zip(manifest.files, valid):
|
||||
entry.rows_total = rows
|
||||
|
||||
manifest.status = batch_ingest.PENDING
|
||||
manifest.detail = "Waiting for review. Nothing runs until an admin starts it."
|
||||
batch_ingest.write_manifest(manifest)
|
||||
batch_job_store.put(manifest)
|
||||
return manifest
|
||||
|
||||
|
||||
def pending_submissions() -> List[batch_ingest.BatchManifest]:
|
||||
"""Every drop awaiting review, newest first.
|
||||
|
||||
Read from disk rather than the in-memory store: the store is populated by
|
||||
whatever this process has seen, and an inbox that empties itself on restart
|
||||
would look exactly like a colleague's files having been processed.
|
||||
"""
|
||||
return [
|
||||
m for m in batch_ingest.list_manifests()
|
||||
if m.status == batch_ingest.PENDING and m.files
|
||||
]
|
||||
85
app/api/batch_job_store.py
Normal file
85
app/api/batch_job_store.py
Normal file
@@ -0,0 +1,85 @@
|
||||
"""Live view of the batches this process knows about.
|
||||
|
||||
Same pattern and the same documented trade-offs as `job_store.py` and its
|
||||
siblings: a process-local dict behind a lock, not shared across uvicorn
|
||||
workers. Adding a broker for this would be operational weight the project has
|
||||
already decided against (see `job_store.py`).
|
||||
|
||||
WHAT IS DIFFERENT HERE, AND WHY IT STILL EARNS ITS PLACE
|
||||
--------------------------------------------------------
|
||||
Unlike the other job stores, this one is not the only record. `manifest.json`
|
||||
on the volume is the durable truth; this is a cache in front of it, and it
|
||||
exists for one reason: the UI polls every 3 seconds while row-level progress
|
||||
ticks many times a second. Serving those polls from memory keeps both the disk
|
||||
writes and the read path off the hot loop. A cache miss is not a 404 - `get()`
|
||||
falls back to reading the manifest, so a batch from before the last restart is
|
||||
still visible.
|
||||
|
||||
Cancellation lives here too rather than on disk. It is a request about the run
|
||||
in flight, and the run in flight is in this process.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import threading
|
||||
from typing import Dict, List, Optional
|
||||
|
||||
from app.core import batch_ingest
|
||||
|
||||
|
||||
class BatchJobStore:
|
||||
def __init__(self) -> None:
|
||||
self._batches: Dict[str, batch_ingest.BatchManifest] = {}
|
||||
self._cancelled: set = set()
|
||||
self._lock = threading.Lock()
|
||||
|
||||
def put(self, manifest: batch_ingest.BatchManifest) -> None:
|
||||
"""Record (or refresh) a batch. This is the `on_change` callback."""
|
||||
with self._lock:
|
||||
self._batches[manifest.batch_id] = manifest
|
||||
|
||||
def get(self, batch_id: str) -> Optional[batch_ingest.BatchManifest]:
|
||||
"""Live state if we have it, otherwise whatever is on disk."""
|
||||
with self._lock:
|
||||
cached = self._batches.get(batch_id)
|
||||
if cached is not None:
|
||||
return cached
|
||||
return batch_ingest.read_manifest(batch_id)
|
||||
|
||||
def recent(self, limit: int = 20) -> List[batch_ingest.BatchManifest]:
|
||||
"""Newest first, merging the live view over the on-disk one.
|
||||
|
||||
Reading the directory rather than only the cache means a restart does
|
||||
not make previous batches vanish from the list.
|
||||
"""
|
||||
with self._lock:
|
||||
live = dict(self._batches)
|
||||
merged: Dict[str, batch_ingest.BatchManifest] = {}
|
||||
for manifest in batch_ingest.list_manifests():
|
||||
merged[manifest.batch_id] = live.get(manifest.batch_id, manifest)
|
||||
for batch_id, manifest in live.items():
|
||||
merged.setdefault(batch_id, manifest)
|
||||
ordered = sorted(merged.values(), key=lambda m: m.created_at, reverse=True)
|
||||
return ordered[: max(1, limit)]
|
||||
|
||||
# -- cancellation --------------------------------------------------------
|
||||
def cancel(self, batch_id: str) -> None:
|
||||
"""Ask the worker to stop before it picks up the next file.
|
||||
|
||||
Nothing interrupts the file already running. Killing a pipeline halfway
|
||||
would leave some of its rows written and the rest not, with no record of
|
||||
where it stopped; letting the current file finish is both simpler and
|
||||
the only version with a defined outcome.
|
||||
"""
|
||||
with self._lock:
|
||||
self._cancelled.add(batch_id)
|
||||
|
||||
def is_cancelled(self, batch_id: str) -> bool:
|
||||
with self._lock:
|
||||
return batch_id in self._cancelled
|
||||
|
||||
def clear_cancel(self, batch_id: str) -> None:
|
||||
with self._lock:
|
||||
self._cancelled.discard(batch_id)
|
||||
|
||||
|
||||
batch_job_store = BatchJobStore()
|
||||
@@ -128,3 +128,56 @@ class NutritionEnrichmentJobOut(BaseModel):
|
||||
unavailable: Optional[int] = None
|
||||
duration_seconds: Optional[float] = None
|
||||
error: Optional[str] = None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Health-score listing (GET /api/nutrition/health-scores)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class HealthScoreItemOut(BaseModel):
|
||||
"""One scored consumable product.
|
||||
|
||||
`health_score` is REQUIRED here, against this module's all-Optional house
|
||||
style, and that is the point: the query filters on `health_score IS NOT
|
||||
NULL`, so a null arriving in this model means the query lost its guarantee.
|
||||
Declaring it required is what turns that into a loud failure instead of a
|
||||
blank cell in whatever dashboard is reading this.
|
||||
"""
|
||||
brand: str
|
||||
image_id: str
|
||||
product_name: Optional[str] = None
|
||||
category: Optional[str] = None
|
||||
|
||||
health_score: float
|
||||
nutrition_score: Optional[float] = None
|
||||
health_band: str # derived from health_score; see _health_band()
|
||||
scoring_version: Optional[str] = None
|
||||
|
||||
# Provenance travels with the number. A score is only as trustworthy as the
|
||||
# source it was computed from, and the consumer should be able to show that.
|
||||
data_status: Optional[str] = None
|
||||
data_source: Optional[str] = None
|
||||
source_url: Optional[str] = None
|
||||
|
||||
calories_kcal: Optional[float] = None
|
||||
protein_g: Optional[float] = None
|
||||
dietary_fiber_g: Optional[float] = None
|
||||
total_sugar_g: Optional[float] = None
|
||||
sodium_mg: Optional[float] = None
|
||||
|
||||
diet_tags: Optional[List[str]] = None
|
||||
allergens: Optional[List[str]] = None
|
||||
|
||||
model_config = {"extra": "ignore"}
|
||||
|
||||
|
||||
class HealthScoreListOut(BaseModel):
|
||||
"""Envelope matching the catalogue convention in `schemas.ProductListOut`
|
||||
rather than the bare-array convention of the older nutrition list
|
||||
endpoints - `total` is the whole reason this endpoint exists, since without
|
||||
it a client paging a thousand products cannot tell when it has finished."""
|
||||
total: int
|
||||
limit: int
|
||||
offset: int
|
||||
generated_at: str
|
||||
items: List[HealthScoreItemOut] = []
|
||||
|
||||
@@ -3,7 +3,7 @@ from __future__ import annotations
|
||||
|
||||
import io
|
||||
import logging
|
||||
from typing import Any, Dict, List, Optional
|
||||
from typing import List, Optional
|
||||
import pandas as pd
|
||||
from pydantic import BaseModel, Field
|
||||
from fastapi import APIRouter, Depends, File, HTTPException, UploadFile
|
||||
@@ -12,7 +12,6 @@ from app.api.deps import require_permission
|
||||
from app.infrastructure.settings import S3_BUCKET
|
||||
from app.services.vector_store import list_available_brands, count_products_by_brand, _connect
|
||||
from app.services.s3_service import s3_service
|
||||
from app.services import store_db
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
router = APIRouter(prefix="/admin/training", tags=["admin_train"])
|
||||
|
||||
@@ -34,6 +34,8 @@ from app.infrastructure.security import (
|
||||
ROLE_PERMISSIONS,
|
||||
Principal,
|
||||
create_access_token,
|
||||
hash_is_wellformed,
|
||||
password_hash_fingerprint,
|
||||
verify_password,
|
||||
)
|
||||
from app.infrastructure.settings import (
|
||||
@@ -45,6 +47,7 @@ from app.infrastructure.settings import (
|
||||
AUTH_MAX_LOGIN_ATTEMPTS,
|
||||
AUTH_USER_PASSWORD_HASH,
|
||||
AUTH_USER_USERNAME,
|
||||
config_source,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -94,21 +97,31 @@ def _accounts() -> Dict[str, dict]:
|
||||
Usernames are compared case-insensitively (matching what the login form
|
||||
sends), but the password is not touched - the previous version lowercased
|
||||
it before comparing, which silently shrank the effective keyspace.
|
||||
|
||||
An account with a blank password hash is omitted entirely rather than
|
||||
included with an unmatchable digest. Both spellings deny the login, but
|
||||
only omission keeps it out of the account table, so nothing downstream can
|
||||
treat it as a real account. This is how the optional `user` account is
|
||||
switched off: leave AUTH_USER_PASSWORD_HASH unset and only `admin` exists.
|
||||
"""
|
||||
return {
|
||||
accounts = {
|
||||
AUTH_ADMIN_USERNAME.lower(): {
|
||||
"password_hash": AUTH_ADMIN_PASSWORD_HASH,
|
||||
"role": "admin",
|
||||
"display_name": "System Administrator",
|
||||
"email": "admin@nutritionintel.com",
|
||||
},
|
||||
AUTH_USER_USERNAME.lower(): {
|
||||
}
|
||||
|
||||
if AUTH_USER_PASSWORD_HASH:
|
||||
accounts[AUTH_USER_USERNAME.lower()] = {
|
||||
"password_hash": AUTH_USER_PASSWORD_HASH,
|
||||
"role": "user",
|
||||
"display_name": "Product & Store Manager",
|
||||
"email": "user@nutritionintel.com",
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
return accounts
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -206,12 +219,67 @@ def login(payload: LoginRequest, request: Request) -> LoginResponse:
|
||||
# username and a bad password take the same time. Otherwise the response
|
||||
# latency alone enumerates valid usernames.
|
||||
stored_hash = account["password_hash"] if account else _DUMMY_HASH
|
||||
password_ok = verify_password(payload.password, stored_hash)
|
||||
|
||||
# A hash that does not parse can never match, and verify_password bails
|
||||
# out of one before doing any PBKDF2 work - measured here, 0.16ms against
|
||||
# 439ms for a real digest. That inverts the very property _DUMMY_HASH
|
||||
# exists to protect: an account whose configured hash is corrupt would
|
||||
# answer ~2700x faster than every other username, announcing which
|
||||
# account is broken to anyone with a stopwatch. So spend the same work
|
||||
# regardless; the result is a rejection either way.
|
||||
hash_usable = hash_is_wellformed(stored_hash)
|
||||
password_ok = verify_password(
|
||||
payload.password, stored_hash if hash_usable else _DUMMY_HASH
|
||||
)
|
||||
|
||||
if account is None or not password_ok:
|
||||
_record_failure(key)
|
||||
logger.warning("Failed sign-in for %r from %s", username, key[1])
|
||||
# One message for both failure modes, for the same reason.
|
||||
# The reason goes to the LOG, never to the caller - the response
|
||||
# below is byte-identical whichever of these it was, so nothing here
|
||||
# can be used to enumerate usernames. It is computed after both the
|
||||
# lookup and the PBKDF2 call above, so it adds no timing signal
|
||||
# either. Without it, a deployment whose configured hash or admin
|
||||
# username has drifted is indistinguishable from someone simply
|
||||
# typing the wrong password, and this is exactly how a production
|
||||
# sign-in outage stayed unexplained: the log said "Failed sign-in
|
||||
# for 'admin'" and nothing more.
|
||||
if account is None:
|
||||
logger.warning(
|
||||
"Failed sign-in for %r from %s: reason=unknown-username. "
|
||||
"Configured accounts: %s (AUTH_ADMIN_USERNAME source=%s).",
|
||||
username,
|
||||
key[1],
|
||||
", ".join(sorted(_accounts())),
|
||||
config_source("AUTH_ADMIN_USERNAME"),
|
||||
)
|
||||
elif not hash_usable:
|
||||
# ERROR, not WARNING: this is a broken deployment, not a bad
|
||||
# guess. No password can ever match, so every sign-in to this
|
||||
# account will 401 until the hash itself is replaced.
|
||||
logger.error(
|
||||
"Failed sign-in for %r from %s: reason=malformed-hash. The configured "
|
||||
"password hash does not parse as pbkdf2_sha256$<iterations>$<b64 salt>$"
|
||||
"<b64 digest> (fingerprint=%s, source=%s). Nobody can sign in to this "
|
||||
"account until it is regenerated with scripts/make_auth_secrets.py.",
|
||||
username,
|
||||
key[1],
|
||||
password_hash_fingerprint(stored_hash) or "(empty)",
|
||||
config_source("AUTH_ADMIN_PASSWORD_HASH"),
|
||||
)
|
||||
else:
|
||||
logger.warning(
|
||||
"Failed sign-in for %r from %s: reason=bad-password. The account exists "
|
||||
"and its hash parses (fingerprint=%s, source=%s); the password did not "
|
||||
"match. If this IS the password you deployed, then the running config "
|
||||
"carries a different hash than the file you are reading - compare that "
|
||||
"fingerprint against: python scripts/make_auth_secrets.py "
|
||||
"--fingerprint .env.production",
|
||||
username,
|
||||
key[1],
|
||||
password_hash_fingerprint(stored_hash),
|
||||
config_source("AUTH_ADMIN_PASSWORD_HASH"),
|
||||
)
|
||||
# One message for every failure mode, for the same reason.
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_401_UNAUTHORIZED,
|
||||
detail="Invalid username or password.",
|
||||
|
||||
575
app/api/routers/batch_catalog.py
Normal file
575
app/api/routers/batch_catalog.py
Normal file
@@ -0,0 +1,575 @@
|
||||
"""Admin endpoints for ingesting several store spreadsheets as one batch.
|
||||
|
||||
POST /api/admin/catalog-batch/preview - parse only, per file
|
||||
POST /api/admin/catalog-batch/ingest - 202 + batch_id
|
||||
GET /api/admin/catalog-batch/batches - recent batches
|
||||
GET /api/admin/catalog-batch/batches/{id} - poll one batch
|
||||
POST /api/admin/catalog-batch/batches/{id}/resume - after a restart
|
||||
POST /api/admin/catalog-batch/batches/{id}/cancel - stop the rest
|
||||
GET /api/admin/catalog-batch/inbox - drops awaiting review
|
||||
POST /api/admin/catalog-batch/from-inbox - run selected files
|
||||
POST /api/admin/catalog-batch/inbox/dismiss - discard selected files
|
||||
|
||||
These began as the multi-file sibling of a single-file admin uploader
|
||||
(`store_catalog.py` and its Store Catalog Ingestion tab). THAT UPLOADER AND ITS
|
||||
ROUTES HAVE SINCE BEEN DELETED - do not go looking for them - and this router is
|
||||
now the only admin way in. What made it the survivor is the unit of work: five
|
||||
files are one batch with one id, so the question a colleague actually asks -
|
||||
"did the drop land?" - has one answer rather than five.
|
||||
|
||||
WHY IMAGES DEFAULT OFF HERE AND THE LLM DEFAULTS ON
|
||||
---------------------------------------------------
|
||||
Image search for a single file is a considered trade. At twenty files it is
|
||||
thousands of outbound requests and a Playwright subprocess that can burn three
|
||||
minutes on its own - on a single-vCPU container that is also serving the API.
|
||||
So a batch opts IN to that stage; it does not opt out.
|
||||
|
||||
`use_llm` is different and defaults ON: it gates only the description written
|
||||
in stage 2 for rows the sheet left blank, which is the one field a shopper
|
||||
reads and a store almost never supplies. It is cheap to leave on because
|
||||
`ollama_service._ensure_client` caches its reachability probe per batch and
|
||||
`store_catalog_pipeline.LlmBreaker` stops calling after three consecutive
|
||||
misses, so with `USE_OLLAMA` false (production today) the cost is one warning
|
||||
per file and the rows fall back to a factual template.
|
||||
|
||||
Note that these defaults bind THIS router only. The open upload endpoint runs
|
||||
itself and takes its two flags from `UPLOAD_AUTORUN_FETCH_IMAGES` and
|
||||
`UPLOAD_AUTORUN_USE_LLM`, both true - see the section below.
|
||||
|
||||
THE OTHER WAY INTO THE SAME PIPELINE
|
||||
------------------------------------
|
||||
`app/api/routers/uploads.py` exposes ingestion to outside API clients under
|
||||
`upload_catalog` rather than `require_admin`. It stages and queues through the
|
||||
identical helpers (`app/api/batch_common.py`) and produces ordinary batches,
|
||||
so everything here - the list, resume, cancel - applies to those too. The only
|
||||
difference is that its reads are filtered to the caller's own submissions.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import queue
|
||||
from typing import Dict, List, Optional
|
||||
|
||||
from fastapi import APIRouter, Depends, File, HTTPException, UploadFile, status
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
from app.api import batch_common
|
||||
from app.api.batch_job_store import batch_job_store
|
||||
from app.api.deps import require_admin
|
||||
from app.core import batch_ingest, batch_worker
|
||||
from app.core import store_catalog_pipeline as pipeline
|
||||
from app.infrastructure.settings import (
|
||||
BATCH_MAX_FILES,
|
||||
BATCH_MAX_TOTAL_BYTES,
|
||||
BATCH_MAX_TOTAL_ROWS,
|
||||
)
|
||||
|
||||
router = APIRouter(prefix="/admin/catalog-batch", tags=["admin", "catalog"])
|
||||
|
||||
# Per-file ceilings. They bound ONE spreadsheet, and they hold whether it
|
||||
# arrived alone or with nineteen friends: a file too big to parse safely does
|
||||
# not become acceptable by being part of a batch. The batch-wide ceilings
|
||||
# (BATCH_MAX_*, imported above) then bound the set on top of these.
|
||||
MAX_UPLOAD_BYTES = 10 * 1024 * 1024
|
||||
MAX_UPLOAD_ROWS = 2000
|
||||
PREVIEW_ROWS = 10
|
||||
|
||||
# The response shape, the upload readers and the staging helper are shared with
|
||||
# app/api/routers/uploads.py - see app/api/batch_common.py, which also wires the
|
||||
# worker to the job store at import.
|
||||
BatchFileOut = batch_common.BatchFileOut
|
||||
BatchOut = batch_common.BatchOut
|
||||
_to_out = batch_common.to_out
|
||||
|
||||
|
||||
def _limits() -> batch_common.UploadLimits:
|
||||
"""Read at call time, from THIS module's globals - see batch_common."""
|
||||
return batch_common.UploadLimits(
|
||||
max_files=BATCH_MAX_FILES,
|
||||
max_file_bytes=MAX_UPLOAD_BYTES,
|
||||
max_file_rows=MAX_UPLOAD_ROWS,
|
||||
max_total_bytes=BATCH_MAX_TOTAL_BYTES,
|
||||
max_total_rows=BATCH_MAX_TOTAL_ROWS,
|
||||
)
|
||||
|
||||
|
||||
@router.post("/preview", dependencies=[Depends(require_admin)])
|
||||
async def preview_catalog_batch(files: List[UploadFile] = File(...)) -> dict:
|
||||
"""Parse every file and report how its columns were understood.
|
||||
|
||||
Nothing is staged and no batch is created. The mapping from a store's own
|
||||
headers onto catalog fields is a guess, and finding out that "Item" was read
|
||||
as the description after twenty files have been scraped is expensive.
|
||||
"""
|
||||
read = await batch_common.read_uploads(files, _limits())
|
||||
out = []
|
||||
for name, contents in read:
|
||||
if not contents:
|
||||
out.append({"filename": name, "ok": False, "error": "The file is empty."})
|
||||
continue
|
||||
try:
|
||||
df, mapping = pipeline.parse_spreadsheet(name, contents)
|
||||
except HTTPException as exc:
|
||||
out.append({"filename": name, "ok": False, "error": str(exc.detail)})
|
||||
continue
|
||||
except Exception as exc: # noqa: BLE001
|
||||
out.append({"filename": name, "ok": False,
|
||||
"error": f"Could not parse the file: {exc}"})
|
||||
continue
|
||||
|
||||
if df.empty:
|
||||
out.append({"filename": name, "ok": False, "error": "The file has no data rows."})
|
||||
continue
|
||||
|
||||
out.append({
|
||||
"filename": name,
|
||||
"ok": True,
|
||||
"rows_total": int(len(df)),
|
||||
"over_row_limit": bool(len(df) > MAX_UPLOAD_ROWS),
|
||||
"recognised_columns": {f: str(c) for f, c in mapping.columns.items()},
|
||||
"unrecognised_columns": mapping.unrecognised,
|
||||
"brand_column_present": "brand" in mapping.columns,
|
||||
"preview": df.head(PREVIEW_ROWS).fillna("").astype(str).to_dict(orient="records"),
|
||||
})
|
||||
|
||||
return {
|
||||
"files": out,
|
||||
"files_total": len(out),
|
||||
"files_ok": sum(1 for f in out if f.get("ok")),
|
||||
"rows_total": sum(int(f.get("rows_total") or 0) for f in out if f.get("ok")),
|
||||
"stages": list(pipeline.STAGE_NAMES),
|
||||
"limits": {
|
||||
"max_files": BATCH_MAX_FILES,
|
||||
"max_rows_per_file": MAX_UPLOAD_ROWS,
|
||||
"max_rows_total": BATCH_MAX_TOTAL_ROWS,
|
||||
"max_bytes_per_file": MAX_UPLOAD_BYTES,
|
||||
"max_bytes_total": BATCH_MAX_TOTAL_BYTES,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
@router.post("/ingest", status_code=status.HTTP_202_ACCEPTED,
|
||||
dependencies=[Depends(require_admin)])
|
||||
async def ingest_catalog_batch(
|
||||
files: List[UploadFile] = File(...),
|
||||
use_llm: bool = True,
|
||||
fetch_images: bool = False,
|
||||
) -> BatchOut:
|
||||
"""Stage the files, queue the batch, and return an id to poll.
|
||||
|
||||
Returns immediately. In production the browser reaches this through Traefik
|
||||
on a different host to the frontend, so anything that sat on the request
|
||||
path would be racing an idle timeout nobody here controls.
|
||||
"""
|
||||
limits = _limits()
|
||||
read = await batch_common.read_uploads(files, limits)
|
||||
valid, invalid, _rows = batch_common.parse_all(read, limits)
|
||||
|
||||
if not valid:
|
||||
detail = "; ".join(f"{name}: {reason}" for name, reason in invalid)
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=f"None of the uploaded files could be ingested. {detail}",
|
||||
)
|
||||
|
||||
manifest, started = batch_common.stage_and_queue(
|
||||
valid, invalid, use_llm=use_llm, fetch_images=fetch_images,
|
||||
)
|
||||
if not started:
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
detail=(
|
||||
"Too many batches are already queued. This one has been saved - "
|
||||
"press Resume on it once the current batch finishes."
|
||||
),
|
||||
)
|
||||
|
||||
return _to_out(manifest)
|
||||
|
||||
|
||||
@router.get("/batches", dependencies=[Depends(require_admin)])
|
||||
def list_catalog_batches(limit: int = 20) -> dict:
|
||||
"""Runs, newest first. Inbox drops are NOT runs and are excluded.
|
||||
|
||||
A `pending` submission has never been near the pipeline. Listing it here
|
||||
would put a row with no progress and no result in the Batch tab, next to
|
||||
real runs, and the Resume button beside it would be a lie.
|
||||
"""
|
||||
limit = max(1, min(limit, 100))
|
||||
# Over-fetch, then drop the pending ones, so filtering cannot return fewer
|
||||
# than `limit` runs just because the inbox happens to be busy.
|
||||
runs = [
|
||||
m for m in batch_job_store.recent(limit * 10)
|
||||
if m.status != batch_ingest.PENDING
|
||||
][:limit]
|
||||
return {"batches": [_to_out(m, slim=True) for m in runs]}
|
||||
|
||||
|
||||
@router.get("/batches/{batch_id}", dependencies=[Depends(require_admin)])
|
||||
def get_catalog_batch(batch_id: str) -> BatchOut:
|
||||
manifest = batch_job_store.get(batch_id)
|
||||
if not manifest:
|
||||
raise HTTPException(status_code=404, detail="Batch not found")
|
||||
return _to_out(manifest)
|
||||
|
||||
|
||||
@router.post("/batches/{batch_id}/resume", dependencies=[Depends(require_admin)])
|
||||
def resume_catalog_batch(batch_id: str) -> BatchOut:
|
||||
"""Re-queue a batch a restart cut short, or one that was queued behind a full queue.
|
||||
|
||||
Also the way out of a batch staged for Dagster that no orchestrator ever
|
||||
came for - the "run it here instead" button. Because this hands the batch to
|
||||
THIS container's worker, it also takes ownership: the runner is flipped to
|
||||
`inprocess` so Dagster will not claim a batch that is already running here.
|
||||
"""
|
||||
manifest = batch_ingest.read_manifest(batch_id)
|
||||
if not manifest:
|
||||
raise HTTPException(status_code=404, detail="Batch not found")
|
||||
|
||||
pending = [f for f in manifest.files if f.status == batch_ingest.QUEUED]
|
||||
if not pending:
|
||||
raise HTTPException(
|
||||
status_code=409,
|
||||
detail=f"Nothing left to run in this batch (status: {manifest.status}).",
|
||||
)
|
||||
|
||||
batch_job_store.clear_cancel(batch_id)
|
||||
manifest.status = batch_ingest.QUEUED
|
||||
manifest.detail = None
|
||||
manifest.runner = batch_ingest.RUNNER_INPROCESS
|
||||
batch_ingest.write_manifest(manifest)
|
||||
batch_job_store.put(manifest)
|
||||
|
||||
try:
|
||||
batch_worker.submit(batch_id)
|
||||
except queue.Full:
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
detail="Too many batches are already queued. Try again shortly.",
|
||||
)
|
||||
return _to_out(manifest)
|
||||
|
||||
|
||||
@router.post("/batches/{batch_id}/cancel", dependencies=[Depends(require_admin)])
|
||||
def cancel_catalog_batch(batch_id: str) -> BatchOut:
|
||||
"""Stop before the next file. The file already running is allowed to finish.
|
||||
|
||||
Interrupting a pipeline mid-file would leave some of its rows written and
|
||||
the rest not, with nothing recording where it stopped. Letting the current
|
||||
file complete is the only version of "cancel" with a defined outcome.
|
||||
"""
|
||||
manifest = batch_job_store.get(batch_id)
|
||||
if not manifest:
|
||||
raise HTTPException(status_code=404, detail="Batch not found")
|
||||
|
||||
batch_job_store.cancel(batch_id)
|
||||
|
||||
# A batch that has not started yet has no worker to notice the flag, so
|
||||
# cancel it here and be done.
|
||||
on_disk = batch_ingest.read_manifest(batch_id)
|
||||
if on_disk and on_disk.status in {batch_ingest.QUEUED, batch_ingest.INTERRUPTED}:
|
||||
for entry in on_disk.files:
|
||||
if entry.status == batch_ingest.QUEUED:
|
||||
entry.status = batch_ingest.CANCELLED
|
||||
entry.detail = "Cancelled before this file started."
|
||||
on_disk.settle()
|
||||
on_disk.detail = "Cancelled."
|
||||
batch_ingest.write_manifest(on_disk)
|
||||
batch_job_store.put(on_disk)
|
||||
return _to_out(on_disk)
|
||||
|
||||
manifest.detail = "Cancelling - the file currently running will finish first."
|
||||
batch_job_store.put(manifest)
|
||||
return _to_out(manifest)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The review inbox
|
||||
# ---------------------------------------------------------------------------
|
||||
# The other half of the flow in app/api/routers/uploads.py: a colleague posts
|
||||
# spreadsheets there with no credential, they land as `pending`, and nothing
|
||||
# runs until somebody here says so. These three routes are what the Inbox tab
|
||||
# in the admin UI calls (frontend/src/pages/InboxPanel.jsx).
|
||||
#
|
||||
# Selection is per FILE and crosses submissions on purpose. "Two sheets from
|
||||
# Monday's drop plus one from today, as one batch" is the request an operator
|
||||
# actually has, and it has no expression in a model where the unit is the drop.
|
||||
|
||||
|
||||
class InboxFileOut(BaseModel):
|
||||
"""One spreadsheet waiting for review."""
|
||||
|
||||
# "{batch_id}:{index}". Addressed by a compound id rather than a bare index
|
||||
# because the UI holds one flat selection set spanning every submission, and
|
||||
# an index alone is not unique across two of them.
|
||||
file_id: str
|
||||
filename: str
|
||||
rows_total: int = 0
|
||||
size_bytes: int = 0
|
||||
|
||||
|
||||
class InboxSubmissionOut(BaseModel):
|
||||
"""One drop: the files that arrived together, and who sent them."""
|
||||
|
||||
submission_id: str
|
||||
submitted_by: Optional[str] = None
|
||||
created_at: float
|
||||
files: List[InboxFileOut]
|
||||
|
||||
|
||||
class InboxOut(BaseModel):
|
||||
pending_count: int
|
||||
submissions: List[InboxSubmissionOut]
|
||||
|
||||
|
||||
class InboxSelection(BaseModel):
|
||||
"""The files an admin ticked."""
|
||||
|
||||
file_ids: List[str] = Field(default_factory=list)
|
||||
|
||||
|
||||
class InboxStartRequest(InboxSelection):
|
||||
# Chosen HERE, not by the sender - see the note on the POST handler in
|
||||
# uploads.py. These commit the host to outbound work, so the decision
|
||||
# belongs to the person who can see what the machine is already doing.
|
||||
# The LLM is on by default for the reason in the module docstring.
|
||||
use_llm: bool = True
|
||||
fetch_images: bool = False
|
||||
# Who runs it. "inprocess" is this container's worker thread and is the
|
||||
# default, so an existing client that never sends the field is unaffected.
|
||||
# "dagster" stages the batch and leaves it for the orchestrator to claim.
|
||||
runner: str = batch_ingest.RUNNER_INPROCESS
|
||||
|
||||
|
||||
class InboxDismissOut(BaseModel):
|
||||
dismissed: int
|
||||
|
||||
|
||||
def _retire(batch_id: str, indices: List[int], *, state: str,
|
||||
released_to: Optional[str] = None) -> int:
|
||||
"""Retire files and refresh the cached manifest. Returns how many.
|
||||
|
||||
The refresh is the part that is easy to forget and impossible to see:
|
||||
`batch_job_store.get` prefers its in-memory copy over the disk, so a drop
|
||||
retired without this keeps reporting `queued` to the sender polling it -
|
||||
the exact question this whole record exists to answer. batch_ingest does
|
||||
not know about the store (the store imports IT), so the refresh belongs
|
||||
here, where both are already in hand.
|
||||
"""
|
||||
retired = batch_ingest.retire_files(
|
||||
batch_id, indices, state=state, released_to=released_to,
|
||||
)
|
||||
if retired:
|
||||
updated = batch_ingest.read_manifest(batch_id)
|
||||
if updated:
|
||||
batch_job_store.put(updated)
|
||||
return retired
|
||||
|
||||
|
||||
def _parse_file_ids(file_ids: List[str]) -> Dict[str, List[int]]:
|
||||
"""Group "{batch_id}:{index}" into {batch_id: [index, ...]}.
|
||||
|
||||
Malformed ids are dropped rather than raising: the UI polls every five
|
||||
seconds and prunes its selection against what came back, so a tick can
|
||||
legitimately refer to a file another tab started a moment ago. Failing the
|
||||
whole request would let one stale checkbox block the rest.
|
||||
"""
|
||||
grouped: Dict[str, List[int]] = {}
|
||||
for raw in file_ids or []:
|
||||
batch_id, _, index = str(raw).partition(":")
|
||||
if not batch_id or not index.isdigit():
|
||||
continue
|
||||
grouped.setdefault(batch_id, []).append(int(index))
|
||||
return grouped
|
||||
|
||||
|
||||
@router.get("/inbox", dependencies=[Depends(require_admin)])
|
||||
def list_inbox() -> InboxOut:
|
||||
"""Every drop awaiting review, newest first."""
|
||||
submissions = []
|
||||
pending_count = 0
|
||||
for manifest in batch_common.pending_submissions():
|
||||
files = [
|
||||
InboxFileOut(
|
||||
file_id=f"{manifest.batch_id}:{entry.index}",
|
||||
filename=entry.filename,
|
||||
rows_total=entry.rows_total,
|
||||
size_bytes=entry.size_bytes,
|
||||
)
|
||||
# A file the sender's own upload already rejected is recorded on the
|
||||
# manifest so they can be told about it, but it has no bytes on disk
|
||||
# and cannot be run - so it is not offered for selection.
|
||||
for entry in manifest.files
|
||||
if entry.status != batch_ingest.FAILED and entry.stored_name
|
||||
]
|
||||
if not files:
|
||||
continue
|
||||
pending_count += len(files)
|
||||
submissions.append(
|
||||
InboxSubmissionOut(
|
||||
submission_id=manifest.batch_id,
|
||||
submitted_by=manifest.submitted_by,
|
||||
created_at=manifest.created_at,
|
||||
files=files,
|
||||
)
|
||||
)
|
||||
return InboxOut(pending_count=pending_count, submissions=submissions)
|
||||
|
||||
|
||||
def _stamp_origins(manifest, origins: List[str]) -> None:
|
||||
"""Record which drop each file in a freshly staged run came from.
|
||||
|
||||
Done after staging, by position, for the same reason `stage_and_queue`
|
||||
back-fills `rows_total` that way: the staging helpers take a
|
||||
(filename, bytes, rows) tuple shared with the uploads router, and widening
|
||||
it here would change a signature three callers depend on.
|
||||
|
||||
Safe by position because `from-inbox` stages with `invalid=[]`, so
|
||||
manifest.files is exactly `picked` in order.
|
||||
"""
|
||||
for entry, drop_id in zip(manifest.files, origins):
|
||||
entry.from_drop = drop_id
|
||||
batch_ingest.write_manifest(manifest)
|
||||
batch_job_store.put(manifest)
|
||||
|
||||
|
||||
@router.post("/from-inbox", status_code=status.HTTP_202_ACCEPTED,
|
||||
dependencies=[Depends(require_admin)])
|
||||
def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
|
||||
"""Take the selected files out of the inbox and run them as one batch.
|
||||
|
||||
The bytes are COPIED into a fresh batch rather than the pending manifest
|
||||
being promoted in place. Two reasons: a selection can span submissions, and
|
||||
there is no such thing as promoting two manifests into one; and the run gets
|
||||
its own id, so it has an identity distinct from the drop it came from -
|
||||
which is what the Batch tab lists and what Resume acts on.
|
||||
|
||||
The originals are removed afterwards, so the same sheet cannot be started
|
||||
twice from a stale checkbox in another tab.
|
||||
"""
|
||||
if request.runner not in batch_ingest.RUNNERS:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail="Unknown runner {!r}. Expected one of: {}.".format(
|
||||
request.runner, ", ".join(sorted(batch_ingest.RUNNERS))
|
||||
),
|
||||
)
|
||||
|
||||
grouped = _parse_file_ids(request.file_ids)
|
||||
if not grouped:
|
||||
raise HTTPException(status_code=400, detail="No files were selected.")
|
||||
|
||||
# (filename, bytes, rows) - the shape stage_and_queue takes, which is also
|
||||
# what parse_all returns, so nothing needs reparsing here.
|
||||
picked: List[tuple] = []
|
||||
senders: List[str] = []
|
||||
# The drop each picked file came out of, in the SAME ORDER as `picked`, so
|
||||
# it can be stamped onto the staged manifest below. Kept parallel rather
|
||||
# than folded into the tuple because that tuple shape is shared with
|
||||
# parse_all and with the uploads router.
|
||||
origins: List[str] = []
|
||||
for batch_id, indices in grouped.items():
|
||||
manifest = batch_ingest.read_manifest(batch_id)
|
||||
if not manifest or manifest.status != batch_ingest.PENDING:
|
||||
continue
|
||||
directory = batch_ingest.batch_dir(batch_id)
|
||||
for entry in manifest.files:
|
||||
if entry.index not in indices or not entry.stored_name:
|
||||
continue
|
||||
try:
|
||||
contents = (directory / entry.stored_name).read_bytes()
|
||||
except OSError:
|
||||
# Retention or a concurrent dismiss got there first. Skipping is
|
||||
# right: the file is genuinely gone, and the poll that follows
|
||||
# will show it has left the inbox.
|
||||
continue
|
||||
picked.append((entry.filename, contents, entry.rows_total))
|
||||
origins.append(batch_id)
|
||||
if manifest.submitted_by:
|
||||
senders.append(manifest.submitted_by)
|
||||
|
||||
if not picked:
|
||||
raise HTTPException(
|
||||
status_code=409,
|
||||
detail=(
|
||||
"None of those files are still waiting - they may have been "
|
||||
"started or dismissed already. Refresh the inbox."
|
||||
),
|
||||
)
|
||||
|
||||
# Preserved so the Batch tab can say where a run came from: one name when a
|
||||
# drop came from one colleague, a joined list when a batch was assembled
|
||||
# from several, which is exactly when the question gets asked.
|
||||
unique_senders = sorted(set(senders))
|
||||
submitted_by = ", ".join(unique_senders)[:120] if unique_senders else None
|
||||
|
||||
if request.runner == batch_ingest.RUNNER_DAGSTER:
|
||||
# Staged and left alone: Dagster claims it on its next sensor tick, or
|
||||
# from the Launchpad. Nothing here waits on that, and the batch is
|
||||
# durable either way.
|
||||
manifest = batch_common.stage_for_orchestrator(
|
||||
picked,
|
||||
[],
|
||||
use_llm=request.use_llm,
|
||||
fetch_images=request.fetch_images,
|
||||
submitted_by=submitted_by,
|
||||
)
|
||||
_stamp_origins(manifest, origins)
|
||||
else:
|
||||
manifest, started = batch_common.stage_and_queue(
|
||||
picked,
|
||||
[],
|
||||
use_llm=request.use_llm,
|
||||
fetch_images=request.fetch_images,
|
||||
submitted_by=submitted_by,
|
||||
)
|
||||
_stamp_origins(manifest, origins)
|
||||
if not started:
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
detail=(
|
||||
"Too many batches are already queued. These files have been "
|
||||
"taken out of the inbox and saved as batch "
|
||||
f"{manifest.batch_id} - press Resume on it once the current "
|
||||
"batch finishes."
|
||||
),
|
||||
)
|
||||
|
||||
# Only now, once the bytes are safely staged under a new id. Retiring them
|
||||
# first would lose the files outright if staging then failed.
|
||||
#
|
||||
# `released_to` is the whole point: the sender polls the drop id they were
|
||||
# given, and this is how they learn which run took their sheet and where to
|
||||
# follow it. Without it a release is indistinguishable from a deletion.
|
||||
for batch_id, indices in grouped.items():
|
||||
_retire(batch_id, indices, state=batch_ingest.RELEASED,
|
||||
released_to=manifest.batch_id)
|
||||
|
||||
return _to_out(manifest)
|
||||
|
||||
|
||||
@router.post("/inbox/dismiss", dependencies=[Depends(require_admin)])
|
||||
def dismiss_inbox_files(request: InboxSelection) -> InboxDismissOut:
|
||||
"""Discard the selected files. The bytes go with them.
|
||||
|
||||
Deliberately irreversible and deliberately unceremonious: this is the
|
||||
disposal path for a drop nobody wants, and on an endpoint anyone can post to
|
||||
it is the control that keeps the volume from filling with rejected sheets.
|
||||
|
||||
The file entry survives as a record reading `dismissed`, so the sender who
|
||||
polls their drop id is told they were declined rather than left staring at a
|
||||
404. Only the bytes are gone.
|
||||
"""
|
||||
grouped = _parse_file_ids(request.file_ids)
|
||||
if not grouped:
|
||||
raise HTTPException(status_code=400, detail="No files were selected.")
|
||||
|
||||
dismissed = 0
|
||||
for batch_id, indices in grouped.items():
|
||||
manifest = batch_ingest.read_manifest(batch_id)
|
||||
# Only ever the inbox. Without this check a crafted id would delete
|
||||
# files out of a batch that was mid-run.
|
||||
if not manifest or manifest.status != batch_ingest.PENDING:
|
||||
continue
|
||||
dismissed += _retire(batch_id, indices, state=batch_ingest.DISMISSED)
|
||||
|
||||
return InboxDismissOut(dismissed=dismissed)
|
||||
322
app/api/routers/brand_discovery.py
Normal file
322
app/api/routers/brand_discovery.py
Normal file
@@ -0,0 +1,322 @@
|
||||
"""Admin routes for brand discovery: a brand NAME into the 11-stage pipeline.
|
||||
|
||||
Two steps on purpose, mirroring `batch_catalog.py`'s preview/ingest split.
|
||||
|
||||
`/preview` discovers and returns; it writes nothing, anywhere. `/ingest` takes
|
||||
the rows the admin kept, renders them as a CSV, and hands the bytes to the same
|
||||
`batch_common.stage_and_queue` an uploaded spreadsheet goes through - so the
|
||||
batch manifest, the stage timeline, Resume, Cancel and the nutrition
|
||||
auto-enrichment that follows a batch all work here without a line of new code.
|
||||
|
||||
The gap between the two steps is the point. Discovery's language-model half can
|
||||
invent a product that nothing downstream is able to catch: a well-formed
|
||||
fiction resolves a category, gets a price band and an internal SKU, and clears
|
||||
`product_validator`'s "verified" threshold comfortably. `product_validator` was
|
||||
built to reject MALFORMED rows, not false ones. A person looking at the list is
|
||||
the check, so the list is shown before anything is written.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from fastapi import APIRouter, Depends, HTTPException, status
|
||||
from pydantic import BaseModel, Field
|
||||
from starlette.concurrency import run_in_threadpool
|
||||
|
||||
from app.api import batch_common
|
||||
from app.api.deps import require_admin
|
||||
from app.core import store_catalog_pipeline as pipeline
|
||||
from app.infrastructure.settings import (
|
||||
BATCH_MAX_FILES,
|
||||
BATCH_MAX_TOTAL_BYTES,
|
||||
BATCH_MAX_TOTAL_ROWS,
|
||||
BRAND_DISCOVERY_DEADLINE_SECONDS,
|
||||
BRAND_DISCOVERY_MAX_PRODUCTS,
|
||||
WEB_DISCOVERY_ENABLED,
|
||||
)
|
||||
from app.services import active_brands, brand_discovery
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
router = APIRouter(prefix="/admin/brand-discovery", tags=["admin", "catalog"])
|
||||
|
||||
# The same per-file ceilings the admin batch routes apply. Discovery emits one
|
||||
# CSV row per product and pack-size explosion happens later, inside stage 4, so
|
||||
# 200 products is 200 rows here - three orders of magnitude inside the limit.
|
||||
MAX_UPLOAD_BYTES = 10 * 1024 * 1024
|
||||
MAX_UPLOAD_ROWS = 2000
|
||||
|
||||
|
||||
def _limits() -> batch_common.UploadLimits:
|
||||
return batch_common.UploadLimits(
|
||||
max_files=BATCH_MAX_FILES,
|
||||
max_file_bytes=MAX_UPLOAD_BYTES,
|
||||
max_file_rows=MAX_UPLOAD_ROWS,
|
||||
max_total_bytes=BATCH_MAX_TOTAL_BYTES,
|
||||
max_total_rows=BATCH_MAX_TOTAL_ROWS,
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Request bodies
|
||||
# ---------------------------------------------------------------------------
|
||||
class DiscoveryPreviewRequest(BaseModel):
|
||||
brand: str
|
||||
max_products: int = Field(default=BRAND_DISCOVERY_MAX_PRODUCTS, ge=1, le=2000)
|
||||
use_openfacts: bool = True
|
||||
use_llm: bool = True
|
||||
# Ungrounded language-model rows are dropped rather than shown by default.
|
||||
# Turning this off is how an admin sees them - they arrive unticked.
|
||||
require_evidence: bool = True
|
||||
# Shorter than the service default: somebody is watching a spinner.
|
||||
deadline_seconds: float = Field(default=90.0, ge=0.0,
|
||||
le=BRAND_DISCOVERY_DEADLINE_SECONDS)
|
||||
refresh_corpus: bool = False
|
||||
# Web & retail listings. Reads a FINISHED web-discovery job (start one at
|
||||
# POST /web-jobs first); off by default, and off means the preview is
|
||||
# exactly what it was before this source existed.
|
||||
use_web: bool = False
|
||||
web_job_id: Optional[str] = None
|
||||
|
||||
|
||||
class DiscoveredProductIn(BaseModel):
|
||||
"""One row the admin kept. Mirrors `DiscoveredProduct`'s written fields.
|
||||
|
||||
Sent back rather than re-discovered so that what is ingested is exactly what
|
||||
was reviewed - a second discovery pass could legitimately return something
|
||||
different, and then the approval would have been of a different list.
|
||||
"""
|
||||
|
||||
product_name: str
|
||||
title: Optional[str] = None
|
||||
category: Optional[str] = None
|
||||
description: Optional[str] = None
|
||||
size_variants: List[str] = Field(default_factory=list)
|
||||
providers: List[str] = Field(default_factory=list)
|
||||
highlights: List[str] = Field(default_factory=list)
|
||||
nutrients: List[str] = Field(default_factory=list)
|
||||
fssai_license: Optional[str] = None
|
||||
barcode: Optional[str] = None
|
||||
image_url: Optional[str] = None
|
||||
# Web-found rows only: the listings that justified the product. Not
|
||||
# written by the pipeline; recorded afterwards into field_sources by
|
||||
# web_discovery.provenance, which also fills a blank price_range from them.
|
||||
listings: List[Dict[str, Any]] = Field(default_factory=list)
|
||||
price_range: Optional[str] = None
|
||||
retailer_count: int = 0
|
||||
# The line and variant the preview grouped this row under. Web rows only
|
||||
# need them sent back: their variant came from the retailer's brackets,
|
||||
# which the plain product name no longer carries.
|
||||
product_line: Optional[str] = None
|
||||
variant: Optional[str] = None
|
||||
|
||||
|
||||
class WebJobRequest(BaseModel):
|
||||
brand: str
|
||||
# Re-search even when a finished job for this brand is less than a day old.
|
||||
# Answers already cached are still reused, so this is cheap.
|
||||
refresh: bool = False
|
||||
|
||||
|
||||
class DiscoveryIngestRequest(BaseModel):
|
||||
brand: str
|
||||
products: List[DiscoveredProductIn]
|
||||
# Stage 2 only calls Ollama for a row whose description is blank, and
|
||||
# discovery leaves most of them blank on purpose (see brand_discovery's note
|
||||
# on the boilerplate generator). On an unreachable Ollama each such row
|
||||
# costs up to OLLAMA_TIMEOUT_SECONDS, so this is worth being able to turn
|
||||
# off for a large brand.
|
||||
use_llm: bool = True
|
||||
fetch_images: bool = True
|
||||
# Ingesting a brand outside ACTIVE_BRANDS writes rows nothing can read.
|
||||
# Refused unless the caller says they mean it - see the 409 below.
|
||||
acknowledge_inactive: bool = False
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Routes
|
||||
# ---------------------------------------------------------------------------
|
||||
@router.post("/preview", dependencies=[Depends(require_admin)])
|
||||
async def preview_brand_discovery(payload: DiscoveryPreviewRequest) -> Dict[str, Any]:
|
||||
"""Discover a brand's products and return them. Writes nothing.
|
||||
|
||||
Run on a worker thread: discovery does blocking HTTP to Open Food Facts and,
|
||||
when the language model is enabled, a series of blocking Ollama calls. On
|
||||
the event loop that would stall every other request for the duration.
|
||||
"""
|
||||
try:
|
||||
result = await run_in_threadpool(
|
||||
brand_discovery.discover_brand_products,
|
||||
payload.brand,
|
||||
max_products=payload.max_products,
|
||||
deadline_seconds=payload.deadline_seconds,
|
||||
use_openfacts=payload.use_openfacts,
|
||||
use_llm=payload.use_llm,
|
||||
require_evidence=payload.require_evidence,
|
||||
refresh_corpus=payload.refresh_corpus,
|
||||
use_web=payload.use_web and WEB_DISCOVERY_ENABLED,
|
||||
web_job_id=payload.web_job_id,
|
||||
)
|
||||
except ValueError as exc:
|
||||
raise HTTPException(status_code=400, detail=str(exc)) from exc
|
||||
except Exception as exc: # noqa: BLE001 - report the failure, do not 500
|
||||
logger.exception("Brand discovery failed for %r", payload.brand)
|
||||
raise HTTPException(
|
||||
status_code=502,
|
||||
detail=f"Discovery failed for {payload.brand!r}: {exc}",
|
||||
) from exc
|
||||
|
||||
body = result.as_dict()
|
||||
# Served rather than duplicated in the frontend, exactly as BatchOut does,
|
||||
# so the two cannot drift when a stage is added.
|
||||
body["stages"] = list(pipeline.STAGE_NAMES)
|
||||
if result.filtering_enabled and not result.brand_active:
|
||||
body["warnings"] = list(body.get("warnings") or []) + [_inactive_message(result)]
|
||||
return body
|
||||
|
||||
|
||||
@router.post("/ingest", status_code=status.HTTP_202_ACCEPTED,
|
||||
dependencies=[Depends(require_admin)])
|
||||
async def ingest_brand_discovery(payload: DiscoveryIngestRequest) -> batch_common.BatchOut:
|
||||
"""Stage the reviewed products as a catalog batch and return an id to poll."""
|
||||
brand = (payload.brand or "").strip()
|
||||
if not brand:
|
||||
raise HTTPException(status_code=400, detail="A brand name is required.")
|
||||
if not payload.products:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail="No products were selected, so there is nothing to ingest.",
|
||||
)
|
||||
|
||||
# A green run over an unreadable catalog is the failure this project refuses
|
||||
# to ship. The rows WOULD be written and fully enriched - nutrition
|
||||
# auto-enrichment runs with include_inactive=True - but /api/brands, search,
|
||||
# suggest and the category listing all filter the brand out, so the result
|
||||
# looks like nothing happened.
|
||||
if active_brands.filtering_enabled() and not active_brands.is_active_brand(brand):
|
||||
if not payload.acknowledge_inactive:
|
||||
raise HTTPException(status_code=409, detail=_inactive_detail(brand))
|
||||
|
||||
products = [
|
||||
brand_discovery.DiscoveredProduct(
|
||||
brand=brand,
|
||||
product_name=item.product_name,
|
||||
title=item.title or item.product_name,
|
||||
category=item.category or "",
|
||||
category_hint="",
|
||||
description=item.description or "",
|
||||
size_variants=list(item.size_variants),
|
||||
providers=list(item.providers),
|
||||
highlights=list(item.highlights),
|
||||
nutrients=list(item.nutrients),
|
||||
fssai_license=item.fssai_license,
|
||||
barcode=item.barcode,
|
||||
image_url=item.image_url,
|
||||
)
|
||||
for item in payload.products
|
||||
]
|
||||
|
||||
filename = brand_discovery.synthetic_filename(brand)
|
||||
contents = brand_discovery.rows_to_csv_bytes(products)
|
||||
|
||||
# ONE FILE, NEVER CHUNKED. stage 11 groups by brand per file and reads the
|
||||
# existing catalog per file, and its intra-file de-duplication
|
||||
# ({image_id: row}) is per file too - so the same image_id split across two
|
||||
# chunks would not be caught. A single file makes that de-duplication total.
|
||||
valid, invalid, _rows = batch_common.parse_all([(filename, contents)], _limits())
|
||||
if not valid:
|
||||
detail = "; ".join(f"{name}: {reason}" for name, reason in invalid)
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=f"The discovered products could not be staged. {detail}",
|
||||
)
|
||||
|
||||
manifest, started = batch_common.stage_and_queue(
|
||||
valid, invalid,
|
||||
use_llm=payload.use_llm,
|
||||
fetch_images=payload.fetch_images,
|
||||
submitted_by=f"brand-discovery: {brand}",
|
||||
)
|
||||
if not started:
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
detail=(
|
||||
"Too many batches are already queued. This one has been saved - "
|
||||
"press Resume on it once the current batch finishes."
|
||||
),
|
||||
)
|
||||
# Web-found rows: record their retailer listings once the batch has stored
|
||||
# them. Keyed by CSV position, which is how stage 11 reports source rows.
|
||||
web_entries = {
|
||||
index: {"listings": item.listings, "price_range": item.price_range or "",
|
||||
"retailer_count": item.retailer_count,
|
||||
"product_line": item.product_line or "", "variant": item.variant or ""}
|
||||
for index, item in enumerate(payload.products) if item.listings
|
||||
}
|
||||
if web_entries:
|
||||
from app.services.web_discovery import provenance
|
||||
provenance.watch(manifest.batch_id, brand, web_entries)
|
||||
return batch_common.to_out(manifest)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Web & retail listings - a background search the panel polls
|
||||
# ---------------------------------------------------------------------------
|
||||
def _require_web() -> None:
|
||||
if not WEB_DISCOVERY_ENABLED:
|
||||
raise HTTPException(status_code=404, detail="Web discovery is switched off (WEB_DISCOVERY_ENABLED).")
|
||||
|
||||
|
||||
@router.post("/web-jobs", status_code=status.HTTP_202_ACCEPTED,
|
||||
dependencies=[Depends(require_admin)])
|
||||
def start_web_job(payload: WebJobRequest) -> Dict[str, Any]:
|
||||
"""Start searching retailer listings for a brand, or return the job that is
|
||||
already running or finished within the last day. Poll GET /web-jobs/{id}."""
|
||||
_require_web()
|
||||
from app.services.web_discovery import jobs as web_jobs
|
||||
try:
|
||||
job = web_jobs.start_job(payload.brand, refresh=payload.refresh)
|
||||
except ValueError as exc:
|
||||
raise HTTPException(status_code=400, detail=str(exc)) from exc
|
||||
return job.summary()
|
||||
|
||||
|
||||
@router.get("/web-jobs/{job_id}", dependencies=[Depends(require_admin)])
|
||||
def get_web_job(job_id: str) -> Dict[str, Any]:
|
||||
_require_web()
|
||||
from app.services.web_discovery import jobs as web_jobs
|
||||
job = web_jobs.get_job(job_id)
|
||||
if job is None:
|
||||
raise HTTPException(status_code=404, detail="No web discovery job with that id.")
|
||||
return job.summary()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The ACTIVE_BRANDS message, in one place
|
||||
# ---------------------------------------------------------------------------
|
||||
def _env_line(brand: str) -> str:
|
||||
names = active_brands.active_display_names()
|
||||
return "ACTIVE_BRANDS=" + ",".join(list(names) + [brand])
|
||||
|
||||
|
||||
def _inactive_message(result: brand_discovery.DiscoveryResult) -> str:
|
||||
return (
|
||||
f"{result.brand} is not in ACTIVE_BRANDS, so these products would be "
|
||||
f"written to {result.table} and then filtered out of /api/brands, "
|
||||
f"search, suggest and the category listing. The rows would be complete "
|
||||
f"and correct, just unreadable. To make them visible, set "
|
||||
f"'{_env_line(result.brand)}' in backend/.env and restart the API - "
|
||||
f"settings are read once at import, so a restart is required."
|
||||
)
|
||||
|
||||
|
||||
def _inactive_detail(brand: str) -> str:
|
||||
return (
|
||||
f"{brand} is not in ACTIVE_BRANDS. Ingesting it now would write a "
|
||||
f"complete catalog that no endpoint can read. Either set "
|
||||
f"'{_env_line(brand)}' in backend/.env and restart the API first, or "
|
||||
f"re-send with acknowledge_inactive=true to stage the data anyway - "
|
||||
f"adding the brand later is a config change and a restart, not a "
|
||||
f"re-ingest."
|
||||
)
|
||||
@@ -1,12 +1,23 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import Optional
|
||||
|
||||
from fastapi import APIRouter, HTTPException, Query
|
||||
|
||||
from app.api.schemas import AllProductsOut, BrandsOut, CategoriesOut, ProductListOut, ProductOut
|
||||
from app.api.schemas import (
|
||||
AllProductsOut,
|
||||
BrandCardOut,
|
||||
BrandCardsOut,
|
||||
BrandsOut,
|
||||
CategoriesOut,
|
||||
ProductListOut,
|
||||
ProductOut,
|
||||
)
|
||||
from app.services.brand_registry import get_brand_logo
|
||||
from app.services.s3_service import s3_service
|
||||
from app.services.vector_store import (
|
||||
get_brand_overview,
|
||||
list_available_brands,
|
||||
list_categories_for_brand,
|
||||
get_products_by_brand,
|
||||
@@ -75,6 +86,18 @@ def _row_to_product_out(row: dict, fallback_brand: str) -> ProductOut:
|
||||
if fssai is not None:
|
||||
fssai = str(fssai).strip() or None
|
||||
|
||||
# NUMERIC arrives as decimal.Decimal, which Pydantic would coerce but JSON
|
||||
# would not. Converted here so 0.0 survives - `or None` would turn a real
|
||||
# score of zero into "not scored", and zero is a valid health score.
|
||||
def _score(key: str) -> Optional[float]:
|
||||
raw = row.get(key)
|
||||
if raw is None:
|
||||
return None
|
||||
try:
|
||||
return float(raw)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
return ProductOut(
|
||||
image_id=image_id,
|
||||
image_url=primary_url,
|
||||
@@ -97,6 +120,8 @@ def _row_to_product_out(row: dict, fallback_brand: str) -> ProductOut:
|
||||
selling_price=sp,
|
||||
barcode=bcd,
|
||||
barcode_type=bcd_type,
|
||||
nutrition_score=_score("nutrition_score"),
|
||||
health_score=_score("health_score"),
|
||||
)
|
||||
|
||||
|
||||
@@ -106,6 +131,66 @@ def get_brands() -> BrandsOut:
|
||||
return BrandsOut(brands=list_available_brands())
|
||||
|
||||
|
||||
def _initials(name: str) -> str:
|
||||
"""Monogram for the card's image fallback - there are no logo assets."""
|
||||
words = [w for w in re.split(r"[^A-Za-z0-9]+", name) if w]
|
||||
if not words:
|
||||
return "?"
|
||||
if len(words) == 1:
|
||||
return words[0][:2].upper()
|
||||
return (words[0][0] + words[1][0]).upper()
|
||||
|
||||
|
||||
# Declared before /brands/{brand}/... so "overview" is never read as a brand
|
||||
# name. The path-segment counts differ, so this is belt-and-braces.
|
||||
@router.get("/brands/overview", response_model=BrandCardsOut)
|
||||
def get_brand_cards(
|
||||
refresh: bool = Query(False, description="Bypass the short-lived overview cache"),
|
||||
) -> BrandCardsOut:
|
||||
"""Per-brand summaries for the home page card grid.
|
||||
|
||||
Additive: GET /brands keeps returning a plain list of names, which the
|
||||
sidebar and the admin/user pages rely on.
|
||||
"""
|
||||
rows = get_brand_overview(force_refresh=refresh)
|
||||
|
||||
cards = []
|
||||
for row in rows:
|
||||
name = row["display_name"]
|
||||
# A curated logo wins over a sampled product photo. It has to be first,
|
||||
# not a fallback: the brands that need one do not have an EMPTY
|
||||
# sample_image_url, they have a non-empty string that only fails when
|
||||
# the browser tries to fetch it. Ordering this second would leave them
|
||||
# exactly as broken as they are now.
|
||||
image_url = _clean_url(get_brand_logo(name)) or _clean_url(row.get("sample_image_url"))
|
||||
|
||||
if not image_url and s3_service.enabled and row.get("sample_image_id"):
|
||||
# LIST, don't construct. get_product_image_url() builds
|
||||
# .../{image_id}/image_000.jpg and returns it whether or not the
|
||||
# object exists, which is how four brands ended up serving URLs
|
||||
# that 404 on every request. get_product_image_urls() does a real
|
||||
# list_objects_v2, so an empty result means genuinely no image -
|
||||
# and it finds .jpeg where the constructor assumed .jpg.
|
||||
listed = s3_service.get_product_image_urls(name, row["sample_image_id"])
|
||||
image_url = _clean_url(listed[0]) if listed else None
|
||||
|
||||
cards.append(BrandCardOut(
|
||||
name=name,
|
||||
slug=row["suffix"],
|
||||
product_count=row["product_count"],
|
||||
category_count=row["category_count"],
|
||||
categories=list(row.get("categories") or []),
|
||||
image_url=image_url,
|
||||
initials=_initials(name),
|
||||
))
|
||||
|
||||
return BrandCardsOut(
|
||||
total_brands=len(cards),
|
||||
total_products=sum(c.product_count for c in cards),
|
||||
brands=cards,
|
||||
)
|
||||
|
||||
|
||||
@router.get("/brands/{brand}/categories", response_model=CategoriesOut)
|
||||
def get_brand_categories(brand: str) -> CategoriesOut:
|
||||
return CategoriesOut(brand=brand, categories=list_categories_for_brand(brand))
|
||||
|
||||
@@ -19,7 +19,20 @@ async def _run_job(job_id: str, brand: str, max_products: int) -> None:
|
||||
job_store.update(job_id, "running")
|
||||
try:
|
||||
summary = await ingest_brand(brand, max_products=max_products)
|
||||
job_store.update(job_id, "done", detail=f"{summary['total_products']} products ingested")
|
||||
# "Ingested" has to mean "in the database". The generation stages can all
|
||||
# succeed while the pgvector write fails, and reporting that as done is
|
||||
# how a run that stored nothing ends up looking successful in the UI.
|
||||
if summary.get("storage_error"):
|
||||
job_store.update(
|
||||
job_id,
|
||||
"failed",
|
||||
detail=(
|
||||
f"Generated {summary['total_products']} product(s) but storing them "
|
||||
f"failed, so none are in the catalog: {summary['storage_error']}"
|
||||
),
|
||||
)
|
||||
else:
|
||||
job_store.update(job_id, "done", detail=f"{summary['total_products']} products ingested")
|
||||
except Exception as e: # noqa: BLE001 - surface any failure to the UI
|
||||
logger.exception("Catalog ingestion job %s failed", job_id)
|
||||
job_store.update(job_id, "failed", detail=str(e))
|
||||
@@ -35,6 +48,20 @@ def generate_catalog(payload: CatalogGenerateRequest) -> CatalogJobOut:
|
||||
"""Kick off brand catalog ingestion (discovery -> images -> embeddings ->
|
||||
pgvector) as a background daemon thread and return immediately with a job id.
|
||||
|
||||
PREFER /api/admin/brand-discovery/* FOR NEW WORK. This route is unchanged
|
||||
and still supported, but it does not run the eleven stages in
|
||||
`app/core/store_catalog_pipeline.py` - no title validation, no pack-size
|
||||
explosion, no SKU resolution, no barcode, no HSN/GST, no validation gate -
|
||||
and it mints `image_id` with `s3_service.generate_image_id()`, which appends
|
||||
a random uuid4. Nothing it writes can ever match an existing row, so running
|
||||
it twice for one brand produces two catalogs. It also calls
|
||||
`upsert_brand_products(cleanup=True)`, which deletes every row not in the
|
||||
batch it just built.
|
||||
|
||||
The discovery routes do run all eleven stages, use a deterministic
|
||||
`image_id`, write with `cleanup=False`, and show the products for approval
|
||||
before anything is stored. See docs/BRAND_DISCOVERY.md.
|
||||
|
||||
NOTE: on an 8GB RAM / CPU-only machine, running ingestion (which loads
|
||||
the embeddings model and calls Ollama repeatedly) at the same time as
|
||||
heavy chat traffic will be slow. This is intended as an occasional
|
||||
|
||||
@@ -5,8 +5,12 @@ import logging
|
||||
import requests
|
||||
from fastapi import APIRouter
|
||||
|
||||
from app.api.schemas import HealthOut
|
||||
from app.infrastructure.settings import OLLAMA_BASE_URL, OLLAMA_MODEL_NAME, EMBEDDINGS_MODEL
|
||||
from app.api.schemas import AuthConfigOut, HealthOut, ImageVectorsOut, OcrOut
|
||||
from app.infrastructure.security import auth_config_summary
|
||||
from app.infrastructure.settings import (
|
||||
ENABLE_IMAGE_VECTORS, OLLAMA_BASE_URL, OLLAMA_MODEL_NAME, EMBEDDINGS_MODEL,
|
||||
)
|
||||
from app.services import image_embedder, ocr_service
|
||||
from app.services.vector_store import _connect # internal, but handy for a connectivity probe
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -35,7 +39,17 @@ def _check_ollama() -> bool:
|
||||
@router.get("/health", response_model=HealthOut)
|
||||
def health() -> HealthOut:
|
||||
"""Liveness/readiness probe used by the React app to show a banner when
|
||||
Postgres or Ollama aren't reachable, instead of failing silently."""
|
||||
Postgres or Ollama aren't reachable, instead of failing silently.
|
||||
|
||||
The `auth` block serves the same purpose for credentials that `database`
|
||||
does for Postgres: it makes a misconfiguration visible from outside the
|
||||
container. See AuthConfigOut for why it is not behind a token - a
|
||||
diagnostic for "nobody can sign in" cannot itself require signing in.
|
||||
|
||||
`status` deliberately does NOT go degraded on an auth problem: this
|
||||
endpoint gates container routing in some deployments, and taking a
|
||||
perfectly serving process out of rotation over a credential mismatch would
|
||||
replace a login failure with an outage."""
|
||||
db_ok = _check_database()
|
||||
ollama_ok = _check_ollama()
|
||||
return HealthOut(
|
||||
@@ -44,4 +58,12 @@ def health() -> HealthOut:
|
||||
ollama=ollama_ok,
|
||||
ollama_model=OLLAMA_MODEL_NAME,
|
||||
embeddings_model=EMBEDDINGS_MODEL,
|
||||
auth=AuthConfigOut(**auth_config_summary()),
|
||||
# Same idea as `auth`: img_vector staying NULL after a deploy has one
|
||||
# usual cause (the model file is not in the image), and it must be
|
||||
# visible from outside the container. status() never loads the model.
|
||||
image_vectors=ImageVectorsOut(enabled=ENABLE_IMAGE_VECTORS, **image_embedder.status()),
|
||||
# And for server-side OCR behind /search/identify: "ocr_unavailable"
|
||||
# in a response has one of three causes, and this names it.
|
||||
ocr=OcrOut(**ocr_service.status()),
|
||||
)
|
||||
|
||||
@@ -1,12 +1,17 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import List, Optional
|
||||
import csv
|
||||
import io
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Dict, Iterator, List, Optional
|
||||
|
||||
from fastapi import APIRouter, HTTPException, Query
|
||||
from fastapi.responses import StreamingResponse
|
||||
|
||||
from app.api.nutrition_schemas import (
|
||||
FullNutritionOut, HealthyAlternativeOut, NutritionInsightsOut,
|
||||
PersonalizedRecommendationOut, ProductListItemOut, SimilarProductOut,
|
||||
FullNutritionOut, HealthScoreItemOut, HealthScoreListOut,
|
||||
HealthyAlternativeOut, NutritionInsightsOut, PersonalizedRecommendationOut,
|
||||
ProductListItemOut, SimilarProductOut,
|
||||
)
|
||||
from app.intelligence import nutrition_recommendation, nutrition_similarity
|
||||
from app.services import nutrition_alternatives_service, nutrition_analytics_service, nutrition_db
|
||||
@@ -100,6 +105,139 @@ def filter_products(
|
||||
return [ProductListItemOut(**r) for r in results]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# GET Health Scores - every scored consumable product
|
||||
# ---------------------------------------------------------------------------
|
||||
#
|
||||
# Distinct from `/products` above, which exists to answer "show me the ten
|
||||
# highest-protein snacks". This one exists to answer "give me the health score
|
||||
# of every consumable product in the catalogue", and that needs three things
|
||||
# `/products` does not offer: a total so a client knows when it has finished
|
||||
# paging, a brand filter, and a one-request export.
|
||||
|
||||
# Derived from health_score alone, never stored - a stored band would be one
|
||||
# more thing that can fall out of step with the score beside it.
|
||||
#
|
||||
# The 60 boundary is not a fresh invention: `nutrition_db.store_healthy_
|
||||
# distribution()` already counts `health_score >= 60` as a "healthy product".
|
||||
# Picking a different number here would leave two parts of the same API
|
||||
# disagreeing about what healthy means.
|
||||
_HEALTH_BANDS = ((80.0, "excellent"), (60.0, "good"), (40.0, "fair"))
|
||||
|
||||
|
||||
def _health_band(score: float) -> str:
|
||||
for threshold, label in _HEALTH_BANDS:
|
||||
if score >= threshold:
|
||||
return label
|
||||
return "poor"
|
||||
|
||||
|
||||
def _to_item(row: Dict[str, Any]) -> HealthScoreItemOut:
|
||||
return HealthScoreItemOut(**row, health_band=_health_band(row["health_score"]))
|
||||
|
||||
|
||||
_CSV_COLUMNS = (
|
||||
"brand", "image_id", "product_name", "category",
|
||||
"health_score", "health_band", "nutrition_score", "scoring_version",
|
||||
"data_status", "data_source", "source_url",
|
||||
"calories_kcal", "protein_g", "dietary_fiber_g", "total_sugar_g", "sodium_mg",
|
||||
"diet_tags", "allergens",
|
||||
)
|
||||
|
||||
|
||||
def _csv_rows(rows: Iterator[Dict[str, Any]]) -> Iterator[str]:
|
||||
"""Yields the export a row at a time so neither the full result set nor the
|
||||
full response body is ever held in memory at once."""
|
||||
buffer = io.StringIO()
|
||||
writer = csv.writer(buffer, lineterminator=chr(10))
|
||||
|
||||
def flush() -> str:
|
||||
value = buffer.getvalue()
|
||||
buffer.seek(0)
|
||||
buffer.truncate(0)
|
||||
return value
|
||||
|
||||
writer.writerow(_CSV_COLUMNS)
|
||||
yield flush()
|
||||
|
||||
for row in rows:
|
||||
row = dict(row, health_band=_health_band(row["health_score"]))
|
||||
writer.writerow([
|
||||
# A Postgres TEXT[] would render as "['Vegan', 'Gluten Free']" via
|
||||
# str(); pipe-joining keeps the cell readable in a spreadsheet.
|
||||
"|".join(row[c] or []) if c in ("diet_tags", "allergens") else
|
||||
("" if row.get(c) is None else row[c])
|
||||
for c in _CSV_COLUMNS
|
||||
])
|
||||
yield flush()
|
||||
|
||||
|
||||
@router.get(
|
||||
"/health-scores",
|
||||
response_model=None,
|
||||
responses={200: {
|
||||
"model": HealthScoreListOut,
|
||||
"content": {"application/json": {}, "text/csv": {}},
|
||||
"description": "Paged JSON envelope, or the whole list as CSV with format=csv.",
|
||||
}},
|
||||
)
|
||||
def list_health_scores(
|
||||
brand: Optional[str] = Query(None, description="Exact brand match, e.g. 'Own Products'"),
|
||||
category: Optional[str] = Query(None, description="Substring match, case-insensitive"),
|
||||
min_score: Optional[float] = Query(None, ge=0, le=100),
|
||||
max_score: Optional[float] = Query(None, ge=0, le=100),
|
||||
include_unknown: bool = Query(
|
||||
False,
|
||||
description="Also return products whose edibility was never confirmed. "
|
||||
"Non-consumables are never returned either way.",
|
||||
),
|
||||
sort_by: str = Query("health_score", description="health_score|nutrition_score|protein|fiber|sugar|sodium|calcium|iron|vitamin_c|calories|product_name|brand|category"),
|
||||
order: str = Query("desc", pattern="^(asc|desc)$"),
|
||||
limit: int = Query(100, ge=1, le=500),
|
||||
offset: int = Query(0, ge=0),
|
||||
fmt: str = Query("json", alias="format", pattern="^(json|csv)$"),
|
||||
):
|
||||
"""Every consumable product that has a health score.
|
||||
|
||||
Only scored products are returned: a consumable whose nutrition could not be
|
||||
matched to a verified source has no score, and is absent rather than present
|
||||
with a null - `compute_scores` refuses to invent one, and this endpoint
|
||||
refuses to imply one.
|
||||
|
||||
Non-consumables (soap, shampoo, mosquito repellent) are excluded by the
|
||||
stored `edibility` verdict, as are products the classifier could not decide
|
||||
on. `include_unknown=true` opts the undecided ones back in; nothing opts a
|
||||
confirmed non-consumable in.
|
||||
|
||||
`format=csv` streams the entire matching set in one response and ignores
|
||||
`limit`/`offset`, which is the point of it.
|
||||
"""
|
||||
filters = {
|
||||
"brand": brand, "category": category,
|
||||
"min_score": min_score, "max_score": max_score,
|
||||
"include_unknown": include_unknown,
|
||||
}
|
||||
|
||||
if fmt == "csv":
|
||||
stamp = datetime.now(timezone.utc).strftime("%Y%m%d")
|
||||
return StreamingResponse(
|
||||
_csv_rows(nutrition_db.iter_health_scores(sort_by=sort_by, order=order, **filters)),
|
||||
media_type="text/csv; charset=utf-8",
|
||||
headers={"Content-Disposition": f'attachment; filename="health_scores_{stamp}.csv"'},
|
||||
)
|
||||
|
||||
rows = nutrition_db.query_health_scores(
|
||||
sort_by=sort_by, order=order, limit=limit, offset=offset, **filters)
|
||||
return HealthScoreListOut(
|
||||
# Counted with the SAME filters that produced `rows`, so the two cannot
|
||||
# describe different populations.
|
||||
total=nutrition_db.count_health_scores(**filters),
|
||||
limit=limit, offset=offset,
|
||||
generated_at=datetime.now(timezone.utc).isoformat(timespec="seconds"),
|
||||
items=[_to_item(r) for r in rows],
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# GET Nutrition Analytics (Feature 9)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -9,6 +9,7 @@ from app.api.background import run_in_background
|
||||
from app.api.deps import require_admin
|
||||
from app.api.nutrition_job_store import nutrition_job_store
|
||||
from app.services import nutrition_enrichment_service
|
||||
from app.services.nutrition_autoenrich import run_enrich_job
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
router = APIRouter(prefix="/admin/nutrition-intelligence", tags=["admin", "nutrition"])
|
||||
@@ -19,30 +20,13 @@ class EnrichRequest(BaseModel):
|
||||
generate_narrative: bool = True
|
||||
max_products: int | None = None
|
||||
|
||||
|
||||
def _run_enrich_job(job_id: str, skip_if_verified: bool, generate_narrative: bool, max_products: int | None) -> None:
|
||||
nutrition_job_store.update(job_id, status="running")
|
||||
|
||||
def progress_cb(done: int, total: int) -> None:
|
||||
nutrition_job_store.update(job_id, processed=done, total=total)
|
||||
|
||||
try:
|
||||
result = nutrition_enrichment_service.enrich_all_products(
|
||||
skip_if_verified=skip_if_verified, generate_narrative=generate_narrative,
|
||||
progress_cb=progress_cb, max_products=max_products,
|
||||
)
|
||||
nutrition_job_store.update(
|
||||
job_id, status="done", detail="Enrichment complete",
|
||||
result={
|
||||
"total_products": result.total_products, "verified": result.verified,
|
||||
"partial": result.partial, "unavailable": result.unavailable,
|
||||
"duration_seconds": result.duration_seconds, "error_count": len(result.errors),
|
||||
"errors": result.errors[:20],
|
||||
},
|
||||
)
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.exception("Nutrition enrichment job %s failed", job_id)
|
||||
nutrition_job_store.update(job_id, status="failed", detail=str(e))
|
||||
# `enrich_all_products` has accepted these three for a while; the API had no
|
||||
# way to pass them, so the endpoint could not do what `scripts/
|
||||
# enrich_nutrition.py` could. In particular there was no way to widen past
|
||||
# ACTIVE_BRANDS, which is what a catalogue-wide run needs.
|
||||
brands: list[str] | None = None
|
||||
include_inactive: bool = False
|
||||
categories: list[str] | None = None
|
||||
|
||||
|
||||
def _run_train_job(job_id: str) -> None:
|
||||
@@ -64,7 +48,15 @@ def enrich_nutrition(payload: EnrichRequest) -> dict:
|
||||
Equivalent to `python scripts/enrich_nutrition.py`."""
|
||||
job = nutrition_job_store.create("enrich")
|
||||
run_in_background(
|
||||
lambda: _run_enrich_job(job.job_id, payload.skip_if_verified, payload.generate_narrative, payload.max_products),
|
||||
lambda: run_enrich_job(
|
||||
job.job_id,
|
||||
skip_if_verified=payload.skip_if_verified,
|
||||
generate_narrative=payload.generate_narrative,
|
||||
max_products=payload.max_products,
|
||||
brands=payload.brands,
|
||||
include_inactive=payload.include_inactive,
|
||||
categories=payload.categories,
|
||||
),
|
||||
name=f"nutrition-enrich-{job.job_id[:8]}",
|
||||
)
|
||||
return {"job_id": job.job_id, "status": job.status}
|
||||
|
||||
@@ -2,28 +2,439 @@ from __future__ import annotations
|
||||
|
||||
from typing import Optional
|
||||
|
||||
from fastapi import APIRouter, Query
|
||||
from fastapi import APIRouter, Depends, File, Form, HTTPException, Query, Request, UploadFile
|
||||
from fastapi.responses import FileResponse, JSONResponse
|
||||
from starlette.concurrency import run_in_threadpool
|
||||
|
||||
from app.api.schemas import SearchOut, SourceProductOut
|
||||
from app.services.rag_service import retrieve
|
||||
from app.api.deps import require_admin
|
||||
from app.api.routers.brands import _row_to_product_out
|
||||
from app.api.schemas import (
|
||||
CaptureJobOut,
|
||||
IdentifyOut,
|
||||
ImageMatchOut,
|
||||
ImageSearchOut,
|
||||
ImageVectorSearchRequest,
|
||||
ProvisionalProductOut,
|
||||
SearchOut,
|
||||
SourceProductOut,
|
||||
)
|
||||
from app.infrastructure import settings
|
||||
from app.infrastructure.settings import (
|
||||
IMAGE_SEARCH_DEFAULT_MIN_SCORE,
|
||||
IMAGE_SEARCH_DEFAULT_TOP_K,
|
||||
IMAGE_SEARCH_MAX_TOP_K,
|
||||
IMAGE_VECTOR_MAX_BYTES,
|
||||
SEARCH_DEFAULT_TOP_K,
|
||||
SEARCH_MAX_TOP_K,
|
||||
)
|
||||
from app.services import capture_discovery, image_embedder, image_search_log
|
||||
from app.services.catalog_search import search_catalog
|
||||
from app.services.image_match import (
|
||||
ImageSearchResult,
|
||||
InvalidVectorError,
|
||||
normalise_vector,
|
||||
parse_vector_param,
|
||||
search_by_vector,
|
||||
)
|
||||
from app.services.product_identify import IdentifyResult, identify_product
|
||||
|
||||
router = APIRouter(tags=["search"])
|
||||
|
||||
|
||||
@router.get("/search", response_model=SearchOut)
|
||||
def semantic_search(
|
||||
def catalog_search_endpoint(
|
||||
q: str = Query(..., min_length=1, max_length=500, description="Free-text search query"),
|
||||
brand: Optional[str] = Query(None, description="Restrict search to a single brand"),
|
||||
category: Optional[str] = Query(None, description="Restrict search to a category"),
|
||||
top_k: int = Query(10, ge=1, le=50),
|
||||
top_k: int = Query(SEARCH_DEFAULT_TOP_K, ge=1, le=SEARCH_MAX_TOP_K),
|
||||
offset: int = Query(0, ge=0, description="Pagination offset (brand listings only)"),
|
||||
) -> SearchOut:
|
||||
"""Pure vector similarity search over the catalog - no LLM call, just
|
||||
pgvector ranking. This is what powers the instant search-as-you-type
|
||||
grid in the React 'Search' tab. For a conversational, LLM-generated
|
||||
answer use POST /api/chat instead."""
|
||||
results = retrieve(q, brand=brand, top_k=top_k, category=category)
|
||||
"""Catalog search for the React 'Browse & Search' grid - no LLM call.
|
||||
|
||||
Two behaviours, chosen from the query text:
|
||||
|
||||
* A bare brand name ("Amul", "Colgate", "coke") returns that brand's whole
|
||||
catalog - the same rows as GET /api/brands/{brand}/products.
|
||||
* Anything else ("Amul Butter", "low sugar biscuit") ranks name matches
|
||||
first, then semantically similar products.
|
||||
|
||||
This does NOT share rag_service.retrieve() with /api/chat: that path clamps
|
||||
results to RAG_MAX_TOP_K to protect the LLM prompt budget, which is why a
|
||||
brand search used to come back with only 15 products.
|
||||
|
||||
For a conversational, LLM-generated answer use POST /api/chat instead.
|
||||
"""
|
||||
result = search_catalog(q, brand=brand, category=category, limit=top_k, offset=offset)
|
||||
return SearchOut(
|
||||
query=q,
|
||||
brand=brand,
|
||||
results=[SourceProductOut(**r.to_dict()) for r in results],
|
||||
results=[SourceProductOut(**p.to_dict()) for p in result.products],
|
||||
total=result.total,
|
||||
limit=result.limit,
|
||||
offset=result.offset,
|
||||
match_mode=result.match_mode,
|
||||
detected_brand=result.detected_brand,
|
||||
detected_category=result.detected_category,
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Search by image
|
||||
# ---------------------------------------------------------------------------
|
||||
# Public like GET /search. Two ways in, one ranking: the Nearle app embeds
|
||||
# the cropped photo on-device with the same MobileNetV3 model that filled
|
||||
# img_vector and POSTs the 1024 floats; anything without the model POSTs the
|
||||
# photo and this API embeds it (bounded: IMAGE_VECTOR_MAX_BYTES, one
|
||||
# inference at a time behind the embedder's lock).
|
||||
|
||||
def _to_image_search_out(result: ImageSearchResult) -> ImageSearchOut:
|
||||
matches = []
|
||||
for row in result.rows:
|
||||
card = _row_to_product_out(row, row.get("brand") or "")
|
||||
matches.append(ImageMatchOut(
|
||||
**card.model_dump(),
|
||||
score=round(float(row["score"]), 4),
|
||||
text_overlap=float(row.get("text_overlap", 0.0)),
|
||||
))
|
||||
return ImageSearchOut(
|
||||
results=matches,
|
||||
total=len(matches),
|
||||
detected_brand=result.detected_brand,
|
||||
scoped_to_brand=result.scoped_to_brand,
|
||||
scope_fallback=result.scope_fallback,
|
||||
min_score=result.min_score,
|
||||
top_k=result.top_k,
|
||||
query_text=result.query_text,
|
||||
match_confidence=result.match_confidence if matches else "none",
|
||||
margin=None if result.margin is None else round(float(result.margin), 4),
|
||||
)
|
||||
|
||||
|
||||
def _identify_confidence(result: IdentifyResult) -> str:
|
||||
"""The ladder's verdict in match_confidence terms: confirmed exactly when
|
||||
product_identify.py's client rule says so."""
|
||||
if not result.search.rows:
|
||||
return "none"
|
||||
if capture_discovery.is_confirmed(result.matched_by, result.fallback_reason):
|
||||
return "confirmed"
|
||||
return "low"
|
||||
|
||||
|
||||
def _to_identify_out(result: IdentifyResult) -> IdentifyOut:
|
||||
fields = _to_image_search_out(result.search).model_dump()
|
||||
fields["match_confidence"] = _identify_confidence(result)
|
||||
if result.matched_by != "image_vector":
|
||||
fields["margin"] = None # text rows: a MiniLM score, no image margin
|
||||
return IdentifyOut(
|
||||
**fields,
|
||||
matched_by=result.matched_by,
|
||||
ocr_text=result.ocr_text,
|
||||
ocr_source=result.ocr_source,
|
||||
image_top_score=None if result.image_top_score is None else round(float(result.image_top_score), 4),
|
||||
fallback_reason=result.fallback_reason,
|
||||
)
|
||||
|
||||
|
||||
def _log_search(route: str, result: ImageSearchResult, *, vector, text, brand, category, top_k,
|
||||
photo: Optional[bytes] = None, text_fallback: Optional[bool] = None) -> None:
|
||||
image_search_log.record(
|
||||
route, vector=vector, text=text, brand=brand, category=category, top_k=top_k,
|
||||
rows=result.rows, detected_brand=result.detected_brand, scoped_to_brand=result.scoped_to_brand,
|
||||
match_confidence=result.match_confidence, margin=result.margin,
|
||||
matched_by="image_vector" if result.rows else "none", photo=photo, text_fallback=text_fallback,
|
||||
)
|
||||
|
||||
|
||||
def _log_identify(route: str, result: IdentifyResult, *, vector, text, brand, category, top_k,
|
||||
photo: Optional[bytes] = None, text_fallback: Optional[bool] = None) -> None:
|
||||
image_search_log.record(
|
||||
route, vector=vector, text=text, brand=brand, category=category, top_k=top_k,
|
||||
rows=result.search.rows, detected_brand=result.search.detected_brand,
|
||||
scoped_to_brand=result.search.scoped_to_brand, match_confidence=_identify_confidence(result),
|
||||
margin=result.search.margin if result.matched_by == "image_vector" else None,
|
||||
matched_by=result.matched_by, fallback_reason=result.fallback_reason,
|
||||
photo=photo, text_fallback=text_fallback,
|
||||
)
|
||||
|
||||
|
||||
@router.post("/search/image-vector", response_model=ImageSearchOut)
|
||||
def image_vector_search_endpoint(body: ImageVectorSearchRequest):
|
||||
"""Products that look like the photo whose embedding is `vector`.
|
||||
|
||||
`vector` is the 1024-float, L2-normalised MobileNetV3-Small embedding the
|
||||
app computes on-device. `score` on each result is cosine similarity
|
||||
(1 - pgvector distance). Optional `text` - the OCR read of the label -
|
||||
narrows the search to the brand it names and picks the right pack size
|
||||
among products that share one photo. An explicit `brand` is a hard
|
||||
filter; a brand recognised from `text` falls back to every brand when it
|
||||
finds nothing (`scope_fallback`).
|
||||
|
||||
With `text_fallback` the request runs the /search/identify ladder
|
||||
instead: when the image match cannot be confirmed (best score under
|
||||
IMAGE_IDENTIFY_MIN_IMAGE_SCORE, or another photo within
|
||||
IMAGE_SEARCH_MIN_MARGIN) the label `text` is resolved against the
|
||||
catalogue, and the response is an IdentifyOut (this response plus
|
||||
matched_by, fallback_reason, ...). Left out, it is on whenever `text` is
|
||||
sent; `false` keeps the image-only ranking, where text only breaks ties.
|
||||
"""
|
||||
ladder = body.text_fallback if body.text_fallback is not None else bool((body.text or "").strip())
|
||||
log_fields = dict(vector=body.vector, text=body.text, brand=body.brand, category=body.category,
|
||||
top_k=body.top_k, text_fallback=body.text_fallback)
|
||||
try:
|
||||
if ladder:
|
||||
identified = identify_product(
|
||||
vector=body.vector, image_bytes=None, text=body.text, brand=body.brand,
|
||||
category=body.category, top_k=body.top_k, min_score=body.min_score,
|
||||
)
|
||||
_log_identify("image-vector", identified, **log_fields)
|
||||
# A Response bypasses response_model, which is the point: the
|
||||
# image-only path keeps its declared ImageSearchOut contract.
|
||||
return JSONResponse(content=_to_identify_out(identified).model_dump(mode="json"))
|
||||
result = search_by_vector(
|
||||
body.vector, text=body.text, brand=body.brand, category=body.category,
|
||||
top_k=body.top_k, min_score=body.min_score,
|
||||
)
|
||||
except InvalidVectorError as exc:
|
||||
raise HTTPException(status_code=422, detail=str(exc))
|
||||
_log_search("image-vector", result, **log_fields)
|
||||
return _to_image_search_out(result)
|
||||
|
||||
|
||||
@router.get("/search/image-vector", response_model=ImageSearchOut)
|
||||
def image_vector_search_get_endpoint(
|
||||
vector: str = Query(
|
||||
..., min_length=1, max_length=16_000,
|
||||
description="The 1024-float embedding: comma-separated decimals, or urlsafe base64 of "
|
||||
"1024 little-endian float32 (recommended - 5.5 KB instead of 8-10 KB)",
|
||||
),
|
||||
text: Optional[str] = Query(None, max_length=500, description="OCR text read off the label"),
|
||||
brand: Optional[str] = Query(None, max_length=120, description="Restrict to one brand (no fallback)"),
|
||||
category: Optional[str] = Query(None, max_length=120),
|
||||
top_k: int = Query(IMAGE_SEARCH_DEFAULT_TOP_K, ge=1, le=IMAGE_SEARCH_MAX_TOP_K),
|
||||
min_score: float = Query(IMAGE_SEARCH_DEFAULT_MIN_SCORE, ge=-1.0, le=1.0),
|
||||
) -> ImageSearchOut:
|
||||
"""The POST above as a GET, for clients that can only pass a query string.
|
||||
|
||||
Same ranking, same response. The vector is 1024 floats, so the URL is
|
||||
5.5 KB as base64 or 8-10 KB comma-separated: fine through the API host
|
||||
(Traefik -> uvicorn, which serve.py gives 64 KB of request-line room),
|
||||
but the comma form exceeds the 8 KB nginx allows on the app domain. Use
|
||||
base64, or the POST, for anything that has to work everywhere.
|
||||
"""
|
||||
try:
|
||||
# Validated here, not only inside the service, so this route rejects a
|
||||
# short / NaN / zero vector exactly as the POST's schema does.
|
||||
values = normalise_vector(parse_vector_param(vector))
|
||||
result = search_by_vector(
|
||||
values, text=text, brand=brand, category=category, top_k=top_k, min_score=min_score,
|
||||
)
|
||||
except InvalidVectorError as exc:
|
||||
raise HTTPException(status_code=422, detail=str(exc))
|
||||
_log_search("image-vector:get", result, vector=values, text=text, brand=brand,
|
||||
category=category, top_k=top_k)
|
||||
return _to_image_search_out(result)
|
||||
|
||||
|
||||
@router.post("/search/image", response_model=ImageSearchOut)
|
||||
async def image_search_endpoint(
|
||||
file: UploadFile = File(..., description="The product photo (JPEG/PNG/WebP), ideally cropped to the pack"),
|
||||
text: Optional[str] = Form(None, max_length=500, description="OCR text read off the label"),
|
||||
brand: Optional[str] = Form(None, max_length=120),
|
||||
category: Optional[str] = Form(None, max_length=120),
|
||||
top_k: int = Form(IMAGE_SEARCH_DEFAULT_TOP_K, ge=1, le=IMAGE_SEARCH_MAX_TOP_K),
|
||||
min_score: float = Form(IMAGE_SEARCH_DEFAULT_MIN_SCORE, ge=-1.0, le=1.0),
|
||||
) -> ImageSearchOut:
|
||||
"""Same as /search/image-vector, but the API embeds the photo itself.
|
||||
|
||||
503 when this deployment has no embedding model (GET /api/health ->
|
||||
image_vectors.model_present says so); send a vector instead.
|
||||
"""
|
||||
content = await file.read()
|
||||
if not content:
|
||||
raise HTTPException(status_code=400, detail="The uploaded image is empty.")
|
||||
if len(content) > IMAGE_VECTOR_MAX_BYTES:
|
||||
raise HTTPException(
|
||||
status_code=413,
|
||||
detail=f"Image is {len(content) / 1_048_576:.1f} MB; the limit is "
|
||||
f"{IMAGE_VECTOR_MAX_BYTES // 1_048_576} MB. Crop or downscale it.",
|
||||
)
|
||||
if not await run_in_threadpool(image_embedder.available):
|
||||
raise HTTPException(
|
||||
status_code=503,
|
||||
detail="The image embedding model is not available on this deployment. "
|
||||
"Embed the photo client-side and POST the vector to /api/search/image-vector.",
|
||||
)
|
||||
vector = await run_in_threadpool(image_embedder.embedding_for_bytes, content)
|
||||
if vector is None:
|
||||
raise HTTPException(
|
||||
status_code=422,
|
||||
detail="Could not decode the image (unsupported format, corrupt data, or too many pixels).",
|
||||
)
|
||||
try:
|
||||
result = await run_in_threadpool(
|
||||
search_by_vector, vector, text=text, brand=brand, category=category,
|
||||
top_k=top_k, min_score=min_score,
|
||||
)
|
||||
except InvalidVectorError as exc: # cannot happen for a model output, but the route must not 500
|
||||
raise HTTPException(status_code=422, detail=str(exc))
|
||||
_log_search("image", result, vector=vector, text=text, brand=brand, category=category,
|
||||
top_k=top_k, photo=content)
|
||||
return _to_image_search_out(result)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Identify: image first, label text second
|
||||
# ---------------------------------------------------------------------------
|
||||
# A phone photo of a pack scores ~0.63 against the catalog's render of the
|
||||
# same pack, so /search/image alone cannot confirm a product. This route
|
||||
# runs the ladder in app/services/product_identify.py: the image match when
|
||||
# it clears IMAGE_IDENTIFY_MIN_IMAGE_SCORE, otherwise the label text - the
|
||||
# client's `text`, else what the server reads off the photo (ocr_service) -
|
||||
# resolved through the catalogue's text embeddings and product names.
|
||||
|
||||
@router.post("/search/identify", response_model=IdentifyOut)
|
||||
async def identify_endpoint(
|
||||
request: Request,
|
||||
file: UploadFile = File(..., description="The product photo (JPEG/PNG/WebP), ideally cropped to the pack"),
|
||||
text: Optional[str] = Form(None, max_length=500,
|
||||
description="OCR text read off the label; when absent the server reads it"),
|
||||
brand: Optional[str] = Form(None, max_length=120),
|
||||
category: Optional[str] = Form(None, max_length=120),
|
||||
top_k: int = Form(IMAGE_SEARCH_DEFAULT_TOP_K, ge=1, le=IMAGE_SEARCH_MAX_TOP_K),
|
||||
min_score: float = Form(IMAGE_SEARCH_DEFAULT_MIN_SCORE, ge=-1.0, le=1.0),
|
||||
) -> IdentifyOut:
|
||||
"""Which catalog product is in this photo.
|
||||
|
||||
`matched_by` says which rung answered and therefore which space each
|
||||
result's `score` is in; `fallback_reason` is set whenever the answer is
|
||||
a best effort rather than a confirmed match. Works text-only on a
|
||||
deployment without the image model (fallback_reason
|
||||
"image_embedder_unavailable"); 503 only when neither a vector nor server
|
||||
OCR is possible and no `text` was sent - GET /api/health -> image_vectors
|
||||
and ocr say which is missing.
|
||||
"""
|
||||
content = await file.read()
|
||||
if not content:
|
||||
raise HTTPException(status_code=400, detail="The uploaded image is empty.")
|
||||
if len(content) > IMAGE_VECTOR_MAX_BYTES:
|
||||
raise HTTPException(
|
||||
status_code=413,
|
||||
detail=f"Image is {len(content) / 1_048_576:.1f} MB; the limit is "
|
||||
f"{IMAGE_VECTOR_MAX_BYTES // 1_048_576} MB. Crop or downscale it.",
|
||||
)
|
||||
vector = None
|
||||
if await run_in_threadpool(image_embedder.available):
|
||||
vector = await run_in_threadpool(image_embedder.embedding_for_bytes, content)
|
||||
if vector is None:
|
||||
raise HTTPException(
|
||||
status_code=422,
|
||||
detail="Could not decode the image (unsupported format, corrupt data, or too many pixels).",
|
||||
)
|
||||
try:
|
||||
result = await run_in_threadpool(
|
||||
identify_product, vector=vector, image_bytes=content, text=text, brand=brand,
|
||||
category=category, top_k=top_k, min_score=min_score,
|
||||
)
|
||||
except InvalidVectorError as exc: # cannot happen for a model output, but the route must not 500
|
||||
raise HTTPException(status_code=422, detail=str(exc))
|
||||
if vector is None and result.fallback_reason == "ocr_unavailable":
|
||||
raise HTTPException(
|
||||
status_code=503,
|
||||
detail="Neither the image embedding model nor server-side OCR is available on this "
|
||||
"deployment. Send the label as `text`, or embed the photo client-side and POST "
|
||||
"the vector to /api/search/image-vector.",
|
||||
)
|
||||
_log_identify("identify", result, vector=vector, text=text, brand=brand, category=category,
|
||||
top_k=top_k, photo=content)
|
||||
out = _to_identify_out(result)
|
||||
if settings.ENABLE_CAPTURE_DISCOVERY and not capture_discovery.is_confirmed(
|
||||
result.matched_by, result.fallback_reason
|
||||
):
|
||||
outcome = await run_in_threadpool(
|
||||
capture_discovery.handle_miss,
|
||||
label_text=result.ocr_text, image_bytes=content, vector=vector,
|
||||
brand=brand, category=category,
|
||||
client=request.client.host if request.client else "unknown",
|
||||
)
|
||||
_apply_capture_outcome(out, outcome)
|
||||
return out
|
||||
|
||||
|
||||
def _apply_capture_outcome(out: IdentifyOut, outcome: capture_discovery.CaptureOutcome) -> None:
|
||||
"""Fold a capture-to-catalog decision into the identify response.
|
||||
|
||||
The low-confidence rows the ladder returned are dropped whenever discovery
|
||||
has an answer of its own: they are OTHER products (another Godrej line, a
|
||||
rival detergent), and showing them beside "adding it now" invites the
|
||||
colleague to pick the wrong one.
|
||||
"""
|
||||
out.discovery_status = outcome.status
|
||||
out.discovery_job_id = outcome.job_id
|
||||
out.discovery_message = outcome.message
|
||||
out.provisional = ProvisionalProductOut(**outcome.provisional) if outcome.provisional else None
|
||||
if outcome.status == capture_discovery.EXISTS and outcome.existing:
|
||||
row = outcome.existing
|
||||
card = _row_to_product_out(row, row.get("brand") or "")
|
||||
out.results = [ImageMatchOut(**card.model_dump(), score=1.0, text_overlap=1.0)]
|
||||
out.total = 1
|
||||
out.matched_by = capture_discovery.MATCHED_BY_LABEL_EXACT
|
||||
out.match_confidence = "confirmed"
|
||||
out.margin = None
|
||||
elif outcome.status == capture_discovery.PENDING:
|
||||
out.results = []
|
||||
out.total = 0
|
||||
out.matched_by = capture_discovery.MATCHED_BY_DISCOVERY
|
||||
out.match_confidence = "none"
|
||||
out.margin = None
|
||||
|
||||
|
||||
@router.get("/search/identify/jobs/{job_id}", response_model=CaptureJobOut)
|
||||
def capture_job_endpoint(job_id: str) -> CaptureJobOut:
|
||||
"""A capture-to-catalog job started by /search/identify.
|
||||
|
||||
Poll until `status` is terminal (done, rejected, failed, interrupted). On
|
||||
done, `product` is the stored catalog row - validation_status needs_review.
|
||||
"""
|
||||
job = capture_discovery.get_job(job_id)
|
||||
if job is None:
|
||||
raise HTTPException(status_code=404, detail="No capture job with that id.")
|
||||
return _to_capture_job_out(job)
|
||||
|
||||
|
||||
def _to_capture_job_out(job: capture_discovery.CaptureJob) -> CaptureJobOut:
|
||||
from app.services.vector_store import get_product_by_image_id
|
||||
|
||||
product = None
|
||||
validation_status = None
|
||||
if job.status == capture_discovery.DONE and job.image_id:
|
||||
row = get_product_by_image_id(job.parent, job.image_id)
|
||||
if row:
|
||||
product = _row_to_product_out(row, row.get("brand") or job.parent)
|
||||
validation_status = row.get("validation_status")
|
||||
return CaptureJobOut(
|
||||
job_id=job.job_id, status=job.status, created_at=job.created_at,
|
||||
updated_at=job.updated_at, provisional=job.provisional, product=product,
|
||||
image_id=job.image_id, disposition=job.disposition,
|
||||
validation_status=validation_status, retail_presence=job.retail_presence,
|
||||
photo_used_as_image=job.photo_used_as_image, detail=job.detail,
|
||||
warnings=job.warnings,
|
||||
)
|
||||
|
||||
|
||||
@router.get("/admin/captures", response_model=list[CaptureJobOut],
|
||||
dependencies=[Depends(require_admin)])
|
||||
def list_capture_jobs_endpoint(limit: int = Query(50, ge=1, le=200)) -> list[CaptureJobOut]:
|
||||
"""Recent capture-to-catalog jobs, newest first - the review list for
|
||||
products colleagues added from the field."""
|
||||
return [_to_capture_job_out(job) for job in capture_discovery.list_jobs(limit)]
|
||||
|
||||
|
||||
@router.get("/search/captures/{name}", include_in_schema=False)
|
||||
def capture_photo_endpoint(name: str):
|
||||
"""A colleague's capture photo - served only because a product with no web
|
||||
image uses it as its image (CAPTURE_PUBLIC_BASE_URL)."""
|
||||
found = capture_discovery.photo_path(name)
|
||||
if found is None:
|
||||
raise HTTPException(status_code=404, detail="No such photo.")
|
||||
path, media_type = found
|
||||
return FileResponse(path, media_type=media_type)
|
||||
|
||||
35
app/api/routers/suggest.py
Normal file
35
app/api/routers/suggest.py
Normal file
@@ -0,0 +1,35 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from fastapi import APIRouter, Query
|
||||
|
||||
from app.api.schemas import SuggestOut, SuggestionOut
|
||||
from app.infrastructure.settings import SUGGEST_DEFAULT_LIMIT
|
||||
from app.services.suggest_service import suggest as suggest_service
|
||||
|
||||
router = APIRouter(tags=["search"])
|
||||
|
||||
|
||||
@router.get("/suggest", response_model=SuggestOut)
|
||||
def search_suggest(
|
||||
q: str = Query(..., min_length=1, max_length=64, description="Partial search text"),
|
||||
limit: int = Query(SUGGEST_DEFAULT_LIMIT, ge=1, le=20),
|
||||
) -> SuggestOut:
|
||||
"""Autocomplete for the catalog search box: brand and category names.
|
||||
|
||||
Answers from in-process caches, so it is safe to call on every keystroke.
|
||||
Product names are deliberately not suggested - brand tables have no
|
||||
trigram index, so that would mean an unindexed scan per keystroke.
|
||||
|
||||
Public, matching GET /api/search.
|
||||
"""
|
||||
results = suggest_service(q, limit=limit)
|
||||
return SuggestOut(
|
||||
query=q,
|
||||
suggestions=[
|
||||
SuggestionOut(
|
||||
type=s.type, value=s.value, label=s.label,
|
||||
sublabel=s.sublabel, score=s.score,
|
||||
)
|
||||
for s in results
|
||||
],
|
||||
)
|
||||
@@ -3,7 +3,6 @@ from __future__ import annotations
|
||||
import logging
|
||||
import os
|
||||
import subprocess
|
||||
import threading
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict
|
||||
|
||||
@@ -48,6 +47,26 @@ def _run_background_auto_seed():
|
||||
logger.error("Background Auto-Init error: %s", e)
|
||||
|
||||
|
||||
def _run_background_brand_sync() -> None:
|
||||
"""Reconcile brand tables against data/seed_catalogs/ in both directions.
|
||||
|
||||
Complements the auto-seed above rather than replacing it: that one only
|
||||
fires against a completely empty database, so without this a brand table
|
||||
created after first boot never gets a seed file, and a seed file added
|
||||
after first boot is never loaded.
|
||||
"""
|
||||
try:
|
||||
from app.services.brand_sync import reconcile_brand_catalogs
|
||||
summary = reconcile_brand_catalogs()
|
||||
logger.info("🔁 Brand catalog reconcile: %s", summary)
|
||||
except Exception as e:
|
||||
logger.error("Brand catalog reconcile error: %s", e)
|
||||
|
||||
|
||||
class BrandSyncRequest(BaseModel):
|
||||
dry_run: bool = False
|
||||
|
||||
|
||||
@router.get("/system/status", response_model=SystemStatusOut)
|
||||
def get_system_status() -> SystemStatusOut:
|
||||
"""Return unified status of database, vector store, stores, and frontend build."""
|
||||
@@ -67,7 +86,13 @@ def get_system_status() -> SystemStatusOut:
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
ollama_ok = _ensure_client()
|
||||
# bool(), because _ensure_client() has three return values, not two: True and
|
||||
# False when Ollama is enabled and reachable/unreachable, and None when
|
||||
# USE_OLLAMA is false - it returns before it ever probes. ollama_connected is
|
||||
# typed bool, so that None failed response_model validation and turned the
|
||||
# whole endpoint into a 500 on exactly the configuration this deployment
|
||||
# runs (USE_OLLAMA=false). "Ollama is switched off" is not a server error.
|
||||
ollama_ok = bool(_ensure_client())
|
||||
dist_ok = FRONTEND_DIST.exists() and (FRONTEND_DIST / "index.html").exists()
|
||||
|
||||
return SystemStatusOut(
|
||||
@@ -89,3 +114,23 @@ def initialize_system(background_tasks: BackgroundTasks) -> Dict[str, Any]:
|
||||
"status": "started",
|
||||
"message": "Background initialization triggered. Check /api/system/status for progress.",
|
||||
}
|
||||
|
||||
|
||||
@router.post("/system/brand-sync", dependencies=[Depends(require_admin)])
|
||||
def sync_brand_catalogs(payload: BrandSyncRequest, background_tasks: BackgroundTasks) -> Dict[str, Any]:
|
||||
"""Reconcile brand tables with their seed catalog files.
|
||||
|
||||
`dry_run` answers inline - it is a handful of count queries and writes
|
||||
nothing, so it is safe to poke at. A real run is backgrounded because
|
||||
exporting a large brand serialises thousands of 384-float embeddings.
|
||||
"""
|
||||
from app.services.brand_sync import reconcile_brand_catalogs
|
||||
|
||||
if payload.dry_run:
|
||||
return {"status": "ok", "summary": reconcile_brand_catalogs(dry_run=True)}
|
||||
|
||||
background_tasks.add_task(_run_background_brand_sync)
|
||||
return {
|
||||
"status": "started",
|
||||
"message": "Brand catalog reconcile triggered. Check /api/system/status for progress.",
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
328
app/api/routers/uploads.py
Normal file
328
app/api/routers/uploads.py
Normal file
@@ -0,0 +1,328 @@
|
||||
"""The catalog ingestion API given to outside API users.
|
||||
|
||||
POST /api/uploads/catalog - send spreadsheets, the pipeline runs
|
||||
GET /api/uploads/catalog - the batches this caller has sent
|
||||
GET /api/uploads/catalog/{batch_id} - progress and result of one of them
|
||||
|
||||
This is deliberately its own router, with its own prefix, so that the difference
|
||||
between it and everything else in the app is visible in one screen rather than
|
||||
inferred from a decorator halfway down a 400-line admin module. Everything else
|
||||
that touches catalog data is `require_admin`; the POST here has no guard at all.
|
||||
|
||||
WHAT HAPPENS WHEN A FILE ARRIVES
|
||||
--------------------------------
|
||||
It is parsed during the request - while the caller is still on the phone - so
|
||||
an unusable sheet comes back as a 400 naming the problem rather than as a job
|
||||
that fails a minute later into a void. Then the bytes are staged to disk, a
|
||||
batch is queued, and the same 11-stage pipeline the admin routes use runs over
|
||||
them: `app/core/batch_ingest.py` -> `store_catalog_pipeline.run_pipeline`.
|
||||
|
||||
The response is a `batch_id`. Ingestion is far too slow to finish inside a
|
||||
request - it is thousands of rows through eleven stages - so the caller polls
|
||||
GET /api/uploads/catalog/{batch_id} until `status` leaves `queued`/`running`.
|
||||
|
||||
AN UNAUTHENTICATED POST NOW STARTS REAL WORK
|
||||
--------------------------------------------
|
||||
Be clear-eyed about what that means. This endpoint takes no credential, and
|
||||
`UPLOAD_AUTORUN` (default true) runs the pipeline the moment a file lands. So
|
||||
"anyone who can reach this host" and "anyone who can write to the live catalog"
|
||||
are the same set of people, and an ingest is an upsert with no undo.
|
||||
|
||||
That was chosen deliberately, over the alternative of issuing the sender an
|
||||
`uploader` API key and auto-running only credentialed requests. The requirement
|
||||
was uploads that run without manual intervention, and a review queue that needs
|
||||
an admin to press a button is not that.
|
||||
|
||||
What still bounds it is throughput, not identity: the per-request ceilings in
|
||||
`_limits()`, and `batch_worker` running a single batch at a time behind a queue
|
||||
of `BATCH_QUEUE_MAX`, past which this endpoint answers 429. A sender can occupy
|
||||
the ingestion worker - that is what it is for - but cannot multiply it, which is
|
||||
what matters on a one-vCPU host also serving the API and its healthcheck.
|
||||
|
||||
Setting `UPLOAD_AUTORUN=false` restores the review inbox, where files wait for
|
||||
an admin and the cost of an unwanted drop is disk rather than products. Both
|
||||
paths are live and both are tested; see `stage_pending` in batch_common.
|
||||
|
||||
WHAT THIS ENDPOINT STILL CANNOT DO
|
||||
----------------------------------
|
||||
See anyone else's data. Every read here is filtered by `submitted_by`, so a key
|
||||
sees the batches it sent and nothing else - not the catalog, not other callers'
|
||||
submissions, not the admin batch list. Nothing here can cancel, resume, or
|
||||
delete; those stay on the admin router.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import List, Optional
|
||||
|
||||
from fastapi import APIRouter, Depends, File, Form, HTTPException, UploadFile, status
|
||||
|
||||
from app.api import batch_common
|
||||
from app.api.batch_job_store import batch_job_store
|
||||
from app.api.deps import get_optional_principal, require_permission
|
||||
from app.core import batch_ingest
|
||||
from app.infrastructure.security import Principal
|
||||
from app.infrastructure.settings import (
|
||||
BATCH_MAX_FILES,
|
||||
BATCH_MAX_TOTAL_BYTES,
|
||||
BATCH_MAX_TOTAL_ROWS,
|
||||
INBOX_MAX_PENDING_BYTES,
|
||||
INBOX_MAX_PENDING_FILES,
|
||||
UPLOAD_AUTORUN,
|
||||
UPLOAD_AUTORUN_FETCH_IMAGES,
|
||||
UPLOAD_AUTORUN_USE_LLM,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
router = APIRouter(prefix="/uploads", tags=["uploads"])
|
||||
|
||||
# Identical to the admin batch path. A file that is too large for an admin to
|
||||
# upload is not somehow acceptable because a colleague sent it.
|
||||
MAX_UPLOAD_BYTES = 10 * 1024 * 1024
|
||||
MAX_UPLOAD_ROWS = 2000
|
||||
|
||||
|
||||
def _limits() -> batch_common.UploadLimits:
|
||||
"""Read at call time, from THIS module's globals - see batch_common."""
|
||||
return batch_common.UploadLimits(
|
||||
max_files=BATCH_MAX_FILES,
|
||||
max_file_bytes=MAX_UPLOAD_BYTES,
|
||||
max_file_rows=MAX_UPLOAD_ROWS,
|
||||
max_total_bytes=BATCH_MAX_TOTAL_BYTES,
|
||||
max_total_rows=BATCH_MAX_TOTAL_ROWS,
|
||||
)
|
||||
|
||||
|
||||
class CatalogUploadOut(batch_common.BatchOut):
|
||||
"""A started batch, plus a sentence a human can read without a schema."""
|
||||
|
||||
message: str
|
||||
|
||||
|
||||
def _owns(manifest: batch_ingest.BatchManifest, principal: Principal) -> bool:
|
||||
"""May this caller see this batch?
|
||||
|
||||
An admin sees everything - they already have the whole batch router. Anyone
|
||||
else sees only what their own credential sent, matched on the credential
|
||||
NAME, which is what `principal.username` is for an API key (see
|
||||
security.principal_for_api_key).
|
||||
"""
|
||||
if principal.role == "admin":
|
||||
return True
|
||||
return bool(manifest.submitted_by) and manifest.submitted_by == principal.username
|
||||
|
||||
|
||||
def _inbox_capacity_or_429(incoming_files: int, incoming_bytes: int) -> None:
|
||||
"""Refuse a drop that would push the review inbox past its ceiling.
|
||||
|
||||
This endpoint takes files from anyone, and it queues nothing, so
|
||||
BATCH_QUEUE_MAX - the bound that made an authenticated uploader safe - does
|
||||
not apply here. Unreviewed submissions accumulate on the volume until
|
||||
somebody acts on them, which on this host is the scarcest resource there is.
|
||||
|
||||
Counted over drops still awaiting review only, so starting or dismissing one
|
||||
frees its share at once.
|
||||
"""
|
||||
pending = batch_common.pending_submissions()
|
||||
files_now = sum(len(m.files) for m in pending)
|
||||
bytes_now = sum(f.size_bytes for m in pending for f in m.files)
|
||||
|
||||
if files_now + incoming_files > INBOX_MAX_PENDING_FILES:
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
detail=(
|
||||
f"The review inbox is full ({files_now} file(s) awaiting review, "
|
||||
f"limit {INBOX_MAX_PENDING_FILES}). Nothing was stored. Ask an "
|
||||
f"admin to clear the inbox, then resend."
|
||||
),
|
||||
)
|
||||
if bytes_now + incoming_bytes > INBOX_MAX_PENDING_BYTES:
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
detail=(
|
||||
f"The review inbox is full "
|
||||
f"({bytes_now // (1024 * 1024)}MB awaiting review, limit "
|
||||
f"{INBOX_MAX_PENDING_BYTES // (1024 * 1024)}MB). Nothing was "
|
||||
f"stored. Ask an admin to clear the inbox, then resend."
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
@router.post("/catalog", status_code=status.HTTP_202_ACCEPTED)
|
||||
async def ingest_catalog_files(
|
||||
files: List[UploadFile] = File(...),
|
||||
sender: Optional[str] = Form(None),
|
||||
principal: Optional[Principal] = Depends(get_optional_principal),
|
||||
) -> CatalogUploadOut:
|
||||
"""Accept spreadsheets and start the pipeline over them. No credential.
|
||||
|
||||
Returns 202 and a `batch_id` to poll. A drop where some files parse and some
|
||||
do not is a partial success, not a failure: the good ones are accepted and
|
||||
the bad ones come back in `files` as `status: "failed"` with the reason, so
|
||||
the caller knows exactly which sheet to fix and resend.
|
||||
|
||||
WHAT HAPPENS ON ARRIVAL depends on one setting, `UPLOAD_AUTORUN`:
|
||||
|
||||
true (default) the batch is queued and the 11 stages run immediately. The
|
||||
id returned IS the run id - poll it and watch `stages[]`.
|
||||
false the batch is left `pending` in the admin review inbox and
|
||||
nothing runs until someone presses Start. The id returned
|
||||
is a DROP id; the run gets a different one, reachable
|
||||
through the file's `released_to`.
|
||||
|
||||
Clients should not care which is configured: both return 202 and an id that
|
||||
`GET /api/uploads/catalog/{id}` understands. Only the number of hops differs.
|
||||
|
||||
`use_llm` and `fetch_images` are still NOT accepted from the request, and
|
||||
that has not changed with autorun - if anything it matters more. The caller
|
||||
is anonymous, and letting an anonymous caller switch on the expensive
|
||||
outbound stages is the one thing this endpoint must not allow. They come
|
||||
from `UPLOAD_AUTORUN_FETCH_IMAGES` / `UPLOAD_AUTORUN_USE_LLM` instead.
|
||||
|
||||
`sender` is a free-text label, not identity - it is whatever the caller
|
||||
typed. It exists so a run can be attributed to a person, and three
|
||||
colleagues all showing as "anonymous" is a batch list nobody can triage. A
|
||||
real credential, if one is presented, wins over it.
|
||||
"""
|
||||
limits = _limits()
|
||||
read = await batch_common.read_uploads(files, limits)
|
||||
valid, invalid, _rows = batch_common.parse_all(read, limits)
|
||||
|
||||
if not valid:
|
||||
detail = "; ".join(f"{name}: {reason}" for name, reason in invalid)
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=f"None of the uploaded files could be ingested. {detail}",
|
||||
)
|
||||
|
||||
# Only meaningful on the review-inbox path. Under autorun nothing ever
|
||||
# awaits review, so the count it guards is permanently zero and the check
|
||||
# could never fire - and a guard that cannot guard anything reads, to the
|
||||
# next person, like protection that is actually there.
|
||||
if not UPLOAD_AUTORUN:
|
||||
_inbox_capacity_or_429(
|
||||
incoming_files=len(valid),
|
||||
incoming_bytes=sum(len(contents) for _n, contents, _r in valid),
|
||||
)
|
||||
|
||||
# A presented credential still names the sender - `get_optional_principal`
|
||||
# returns None only when NO credential was sent, and still raises on one
|
||||
# that is present and wrong. `sender` is trusted for a label and nothing
|
||||
# else; it is truncated because it is rendered in the admin UI.
|
||||
submitted_by = (
|
||||
principal.username if principal
|
||||
else ((sender or "").strip()[:60] or "anonymous")
|
||||
)
|
||||
|
||||
if UPLOAD_AUTORUN:
|
||||
manifest, started = batch_common.stage_and_queue(
|
||||
valid,
|
||||
invalid,
|
||||
use_llm=UPLOAD_AUTORUN_USE_LLM,
|
||||
fetch_images=UPLOAD_AUTORUN_FETCH_IMAGES,
|
||||
submitted_by=submitted_by,
|
||||
)
|
||||
if not started:
|
||||
# Staged and durable, but not running, and this caller has no Resume
|
||||
# button - that lives on the admin router. So the honest instruction
|
||||
# is to send it again shortly. The id is named so an admin can find
|
||||
# and resume THIS batch instead, if the caller reports it.
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
detail=(
|
||||
f"Too many batches are already queued. Batch "
|
||||
f"{manifest.batch_id} has been saved but not started; retry "
|
||||
f"this upload shortly."
|
||||
),
|
||||
)
|
||||
logger.info(
|
||||
"Catalog batch %s queued: %d file(s), %d rejected, from %s%s",
|
||||
manifest.batch_id, len(valid), len(invalid), submitted_by,
|
||||
"" if principal else " (no credential)",
|
||||
)
|
||||
else:
|
||||
manifest = batch_common.stage_pending(
|
||||
valid,
|
||||
invalid,
|
||||
submitted_by=submitted_by,
|
||||
)
|
||||
logger.info(
|
||||
"Catalog drop %s received for review: %d file(s), %d rejected, from %s%s",
|
||||
manifest.batch_id, len(valid), len(invalid), submitted_by,
|
||||
"" if principal else " (no credential)",
|
||||
)
|
||||
|
||||
body = batch_common.to_out(manifest).model_dump()
|
||||
if UPLOAD_AUTORUN:
|
||||
message = (
|
||||
f"{len(valid)} file(s) accepted and queued for ingestion. "
|
||||
f"Poll GET /api/uploads/catalog/{manifest.batch_id} for progress."
|
||||
)
|
||||
if invalid:
|
||||
message = (
|
||||
f"{len(valid)} file(s) accepted and queued for ingestion. "
|
||||
f"{len(invalid)} could not be read - see 'files' for the reason "
|
||||
f"on each, and resend those."
|
||||
)
|
||||
else:
|
||||
message = (
|
||||
f"{len(valid)} file(s) received and waiting for review. Nothing runs "
|
||||
f"until an admin starts them. "
|
||||
f"Poll GET /api/uploads/catalog/{manifest.batch_id} for status."
|
||||
)
|
||||
if invalid:
|
||||
message = (
|
||||
f"{len(valid)} file(s) received and waiting for review. "
|
||||
f"{len(invalid)} could not be read - see 'files' for the reason on "
|
||||
f"each, and resend those."
|
||||
)
|
||||
return CatalogUploadOut(**body, message=message)
|
||||
|
||||
|
||||
@router.get("/catalog")
|
||||
def list_my_catalog_batches(
|
||||
limit: int = 20,
|
||||
principal: Principal = Depends(require_permission("upload_catalog")),
|
||||
) -> dict:
|
||||
"""The batches this credential has sent, newest first.
|
||||
|
||||
Still credentialed, unlike the single-batch read below. An anonymous LIST
|
||||
would hand any caller every other sender's drops in one request, which is
|
||||
a different thing entirely from letting someone check the id they hold.
|
||||
"""
|
||||
limit = max(1, min(limit, 100))
|
||||
# Over-fetch before filtering: `recent` orders by creation across every
|
||||
# caller, so taking `limit` first would return fewer than `limit` of this
|
||||
# caller's own - or none at all while another key is busy.
|
||||
mine = [
|
||||
m for m in batch_job_store.recent(limit * 10) if _owns(m, principal)
|
||||
][:limit]
|
||||
return {"batches": [batch_common.to_out(m, slim=True) for m in mine]}
|
||||
|
||||
|
||||
@router.get("/catalog/{batch_id}")
|
||||
def get_my_catalog_batch(
|
||||
batch_id: str,
|
||||
principal: Optional[Principal] = Depends(get_optional_principal),
|
||||
) -> batch_common.BatchOut:
|
||||
"""Progress and result for one drop, addressed by its id.
|
||||
|
||||
THE ID IS THE CREDENTIAL HERE, and it has to be: the sender needed no
|
||||
credential to post, so requiring one to read the result would leave them
|
||||
unable to find out what happened to their own file. `batch_id` is a
|
||||
`uuid4().hex` handed only to whoever submitted the drop - 128 bits, not
|
||||
enumerable - so holding it is the proof of having sent it.
|
||||
|
||||
A caller who DID present a credential is held to it, and sees only their own
|
||||
submissions. That is stricter than anonymous access to the same row, which
|
||||
is the right way round: a named key should not become a way to browse.
|
||||
|
||||
Either way an id you may not see is a 404, never a 403 - whether it exists
|
||||
is not the caller's business, so the two answers must be indistinguishable.
|
||||
"""
|
||||
manifest = batch_job_store.get(batch_id)
|
||||
if not manifest:
|
||||
raise HTTPException(status_code=404, detail="Batch not found")
|
||||
if principal is not None and not _owns(manifest, principal):
|
||||
raise HTTPException(status_code=404, detail="Batch not found")
|
||||
return batch_common.to_out(manifest)
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,9 +1,16 @@
|
||||
"""Pydantic request/response models for the FastAPI layer."""
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from typing import List, Optional
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
from pydantic import BaseModel, Field, field_validator
|
||||
|
||||
from app.infrastructure.settings import (
|
||||
IMAGE_SEARCH_DEFAULT_MIN_SCORE,
|
||||
IMAGE_SEARCH_DEFAULT_TOP_K,
|
||||
IMAGE_SEARCH_MAX_TOP_K,
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -32,6 +39,10 @@ class ProductOut(BaseModel):
|
||||
selling_price: Optional[float] = None
|
||||
barcode: Optional[str] = None
|
||||
barcode_type: Optional[str] = None
|
||||
# Mirrored from nutrition_insights onto the brand table by
|
||||
# nutrition_score_sync. None means "not scored yet", never "scored zero".
|
||||
nutrition_score: Optional[float] = None
|
||||
health_score: Optional[float] = None
|
||||
|
||||
|
||||
class SourceProductOut(BaseModel):
|
||||
@@ -56,6 +67,10 @@ class SourceProductOut(BaseModel):
|
||||
selling_price: Optional[float] = None
|
||||
barcode: Optional[str] = None
|
||||
barcode_type: Optional[str] = None
|
||||
# Mirrored from nutrition_insights onto the brand table by
|
||||
# nutrition_score_sync. None means "not scored yet", never "scored zero".
|
||||
nutrition_score: Optional[float] = None
|
||||
health_score: Optional[float] = None
|
||||
similarity: float
|
||||
|
||||
|
||||
@@ -63,12 +78,88 @@ class SourceProductOut(BaseModel):
|
||||
# Health
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class ApiKeyInfoOut(BaseModel):
|
||||
"""One configured machine consumer, named but never quoted.
|
||||
|
||||
`fingerprint` is a truncated digest of name+secret, not the secret. It exists
|
||||
so a caller who was issued a key can confirm THAT key is the one this
|
||||
deployment loaded - the question a 401 cannot answer, since an undeployed key
|
||||
and a wrong key fail identically.
|
||||
"""
|
||||
|
||||
name: str
|
||||
role: str
|
||||
fingerprint: str
|
||||
|
||||
|
||||
class AuthConfigOut(BaseModel):
|
||||
"""
|
||||
The effective auth configuration, reported by /api/health.
|
||||
|
||||
Unauthenticated on purpose. The failure this exists to diagnose is "nobody
|
||||
can sign in", so anything gated behind an admin token is unreachable
|
||||
exactly when it is needed. Nothing here is a secret: the admin username is
|
||||
already the documented one, allow_any_login=true is a fact an operator
|
||||
urgently needs (and an attacker discovers with a single login attempt
|
||||
anyway), and the fingerprint is a truncated hash of a salted digest, not a
|
||||
password. The API key block follows the same rule: it names which consumers
|
||||
are configured and fingerprints their keys, so a caller can tell an
|
||||
undeployed key from a rejected one, but it never renders a secret. What it buys is a one-command answer to "is this deployment
|
||||
running the config I think it is?" - compare the fingerprint here against
|
||||
the one printed by scripts/make_auth_secrets.py --fingerprint.
|
||||
"""
|
||||
|
||||
enabled: bool
|
||||
allow_any_login: bool
|
||||
admin_username: str
|
||||
password_hash_valid: bool
|
||||
password_hash_iterations: Optional[int] = None
|
||||
password_hash_fingerprint: str
|
||||
# "process-env" | "env-file" | "default" - which one actually won.
|
||||
admin_username_source: str
|
||||
password_hash_source: str
|
||||
# Machine consumers. Names and fingerprints only - the secrets themselves are
|
||||
# never rendered here, and _parse_api_keys enforces enough entropy that the
|
||||
# fingerprints do not give them away. Defaulted so a client of this schema
|
||||
# still validates against a deployment predating these fields.
|
||||
api_keys_count: int = 0
|
||||
api_keys: List[ApiKeyInfoOut] = Field(default_factory=list)
|
||||
api_keys_source: str = "default"
|
||||
|
||||
|
||||
class ImageVectorsOut(BaseModel):
|
||||
"""Why img_vector is (or is not) being filled. Reported without loading
|
||||
the model. `model_present=false` after a deploy means the .tflite was not
|
||||
shipped in the image - the one failure this feature absorbs silently."""
|
||||
enabled: bool = True
|
||||
model_path: str = ""
|
||||
model_present: bool = False
|
||||
runtime_importable: bool = False
|
||||
state: str = "unknown"
|
||||
|
||||
|
||||
class OcrOut(BaseModel):
|
||||
"""Whether POST /api/search/identify can read a label off a photo itself.
|
||||
Reported without loading the engine. `runtime_importable=false` after a
|
||||
deploy means the rapidocr wheel was not installed (requirements-ocr.txt);
|
||||
`onnxruntime_importable=false` means its engine was not."""
|
||||
enabled: bool = True
|
||||
runtime_importable: bool = False
|
||||
onnxruntime_importable: bool = False
|
||||
state: str = "unknown"
|
||||
|
||||
|
||||
class HealthOut(BaseModel):
|
||||
status: str
|
||||
database: bool
|
||||
ollama: bool
|
||||
ollama_model: str
|
||||
embeddings_model: str
|
||||
auth: AuthConfigOut
|
||||
# Defaulted so a client of this schema still validates against a
|
||||
# deployment predating the field.
|
||||
image_vectors: ImageVectorsOut = Field(default_factory=ImageVectorsOut)
|
||||
ocr: OcrOut = Field(default_factory=OcrOut)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -79,6 +170,27 @@ class BrandsOut(BaseModel):
|
||||
brands: List[str]
|
||||
|
||||
|
||||
class BrandCardOut(BaseModel):
|
||||
"""A brand as shown on the home page card grid.
|
||||
|
||||
`name` is the same string GET /api/brands returns, so the frontend can
|
||||
pass it straight back to /api/brands/{brand}/products.
|
||||
"""
|
||||
name: str
|
||||
slug: str
|
||||
product_count: int
|
||||
category_count: int
|
||||
categories: List[str] = Field(default_factory=list)
|
||||
image_url: Optional[str] = None
|
||||
initials: str
|
||||
|
||||
|
||||
class BrandCardsOut(BaseModel):
|
||||
total_brands: int
|
||||
total_products: int
|
||||
brands: List[BrandCardOut]
|
||||
|
||||
|
||||
class CategoriesOut(BaseModel):
|
||||
brand: str
|
||||
categories: List[str]
|
||||
@@ -105,8 +217,156 @@ class AllProductsOut(BaseModel):
|
||||
|
||||
class SearchOut(BaseModel):
|
||||
query: str
|
||||
# Echoes the *request* param, as it always has. The brand inferred from the
|
||||
# query text goes in `detected_brand` instead - repurposing this field would
|
||||
# break any consumer reading it as "the filter I sent".
|
||||
brand: Optional[str] = None
|
||||
results: List[SourceProductOut]
|
||||
# Exact in brand_catalog mode. None in hybrid mode: the merge happens in
|
||||
# Python across N brand tables after per-table LIMITs, so there is no cheap
|
||||
# exact count and inventing one would misreport how much was found.
|
||||
total: Optional[int] = None
|
||||
limit: int = 0
|
||||
offset: int = 0
|
||||
match_mode: str = "hybrid" # "brand_catalog" | "hybrid"
|
||||
detected_brand: Optional[str] = None
|
||||
detected_category: Optional[str] = None
|
||||
|
||||
|
||||
class SuggestionOut(BaseModel):
|
||||
"""One row in the search box's autocomplete dropdown."""
|
||||
type: str # "brand" | "category"
|
||||
value: str # what the search box / filter should use
|
||||
label: str # display text
|
||||
sublabel: Optional[str] = None # e.g. "128 products"
|
||||
score: float = 0.0
|
||||
|
||||
|
||||
class SuggestOut(BaseModel):
|
||||
query: str
|
||||
suggestions: List[SuggestionOut]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Image search (POST /api/search/image-vector, POST /api/search/image)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class ImageVectorSearchRequest(BaseModel):
|
||||
"""A phone photo's embedding, as the Nearle app computes it on-device."""
|
||||
vector: List[float] = Field(
|
||||
..., min_length=1024, max_length=1024,
|
||||
description="L2-normalised MobileNetV3-Small embedding, 1024 floats",
|
||||
)
|
||||
text: Optional[str] = Field(None, max_length=500, description="OCR text read off the label")
|
||||
brand: Optional[str] = Field(None, max_length=120, description="Restrict to one brand (no fallback)")
|
||||
category: Optional[str] = Field(None, max_length=120)
|
||||
top_k: int = Field(IMAGE_SEARCH_DEFAULT_TOP_K, ge=1, le=IMAGE_SEARCH_MAX_TOP_K)
|
||||
min_score: float = Field(IMAGE_SEARCH_DEFAULT_MIN_SCORE, ge=-1.0, le=1.0,
|
||||
description="Drop matches with cosine similarity below this")
|
||||
# With it, the route runs the identify ladder: when the best image score
|
||||
# is under IMAGE_IDENTIFY_MIN_IMAGE_SCORE (or another photo is within
|
||||
# IMAGE_SEARCH_MIN_MARGIN of it), `text` is resolved against the
|
||||
# catalogue instead, and the response is an IdentifyOut (ImageSearchOut
|
||||
# plus matched_by / fallback_reason / ...). Left out, it is ON whenever
|
||||
# `text` is sent: a label that names the product must not lose to a 0.5
|
||||
# cosine, which it did when text only broke exact ties. `false` keeps the
|
||||
# old image-only ranking.
|
||||
text_fallback: Optional[bool] = Field(
|
||||
None,
|
||||
description="Resolve `text` when the image match cannot be confirmed; the response then "
|
||||
"carries the IdentifyOut fields. Default: on when `text` is sent. "
|
||||
"false = image-only ranking, text breaks ties only",
|
||||
)
|
||||
|
||||
@field_validator("vector")
|
||||
@classmethod
|
||||
def _finite_and_nonzero(cls, v: List[float]) -> List[float]:
|
||||
if not all(math.isfinite(x) for x in v):
|
||||
raise ValueError("vector contains NaN or infinite values")
|
||||
if math.sqrt(sum(x * x for x in v)) < 1e-6:
|
||||
raise ValueError("vector is all zeros")
|
||||
return v
|
||||
|
||||
|
||||
class ImageMatchOut(ProductOut):
|
||||
"""One catalog product that looks like the photo: the product card plus
|
||||
how close it is. `score` is cosine similarity (1 - pgvector distance);
|
||||
`text_overlap` is the label-text tie-break weight, 0 when no text was sent."""
|
||||
score: float
|
||||
text_overlap: float = 0.0
|
||||
|
||||
|
||||
class ImageSearchOut(BaseModel):
|
||||
results: List[ImageMatchOut]
|
||||
total: int
|
||||
detected_brand: Optional[str] = None
|
||||
scoped_to_brand: bool = False
|
||||
scope_fallback: bool = False
|
||||
min_score: float = 0.0
|
||||
top_k: int = 0
|
||||
query_text: Optional[str] = None
|
||||
# "confirmed" only when the best match clears IMAGE_IDENTIFY_MIN_IMAGE_SCORE
|
||||
# AND leads the best different photo by IMAGE_SEARCH_MIN_MARGIN (on
|
||||
# /identify: when the ladder confirmed it). "low": show the results as a
|
||||
# list to pick from, never as the answer. "none": no results.
|
||||
match_confidence: str = "none"
|
||||
# Best score minus the best DIFFERENT photo's; None when there was no rival.
|
||||
# Image space only - None when the rows came from the text rung.
|
||||
margin: Optional[float] = None
|
||||
|
||||
|
||||
class IdentifyOut(ImageSearchOut):
|
||||
"""POST /search/identify: an ImageSearchOut plus which rung answered.
|
||||
|
||||
`matched_by` names the space each result's `score` is in: "image_vector"
|
||||
(cosine of the 1024-d photo embedding) or "text" (cosine of the 384-d
|
||||
MiniLM embedding of the label). `image_top_score` always carries the
|
||||
image side. The answer is confirmed when matched_by is "text", or
|
||||
"image_vector" with no `fallback_reason`; otherwise `fallback_reason`
|
||||
says why the best effort shown is unconfirmed (see product_identify.py).
|
||||
"""
|
||||
matched_by: str = "none" # "image_vector" | "text" | "label_exact"
|
||||
# | "discovery_pending" | "none"
|
||||
ocr_text: Optional[str] = None # the label text the ladder used
|
||||
ocr_source: Optional[str] = None # "client" | "server"
|
||||
image_top_score: Optional[float] = None
|
||||
fallback_reason: Optional[str] = None
|
||||
# Capture-to-catalog (ENABLE_CAPTURE_DISCOVERY). All None when the answer
|
||||
# was confirmed or the feature is off. See app/services/capture_discovery.py.
|
||||
discovery_status: Optional[str] = None # "pending" | "exists" | "needs_input" | "busy"
|
||||
discovery_job_id: Optional[str] = None # poll GET /api/search/identify/jobs/{id}
|
||||
discovery_message: Optional[str] = None # one line to show the colleague
|
||||
provisional: Optional["ProvisionalProductOut"] = None
|
||||
|
||||
|
||||
class ProvisionalProductOut(BaseModel):
|
||||
"""What the label says, before the pipeline has run. Not a catalog row."""
|
||||
brand: str
|
||||
product_name: str
|
||||
size: Optional[str] = None
|
||||
category: Optional[str] = None
|
||||
hsn_code: Optional[str] = None
|
||||
gst_percent: Optional[float] = None
|
||||
hsn_gst_needs_review: Optional[bool] = None
|
||||
visible_in_search: bool = True # False: brand is not in ACTIVE_BRANDS
|
||||
source: str = "label"
|
||||
|
||||
|
||||
class CaptureJobOut(BaseModel):
|
||||
"""GET /search/identify/jobs/{job_id}: one capture-to-catalog job."""
|
||||
job_id: str
|
||||
status: str # queued | running | done | rejected | failed | interrupted
|
||||
created_at: float
|
||||
updated_at: float
|
||||
provisional: ProvisionalProductOut
|
||||
product: Optional[ProductOut] = None # the stored row, once done
|
||||
image_id: Optional[str] = None
|
||||
disposition: Optional[str] = None # inserted | backfilled | unchanged
|
||||
validation_status: Optional[str] = None
|
||||
retail_presence: Optional[dict] = None
|
||||
photo_used_as_image: bool = False
|
||||
detail: Optional[str] = None
|
||||
warnings: List[str] = Field(default_factory=list)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -87,7 +87,13 @@ class SeedResponse(BaseModel):
|
||||
class TrainRequest(BaseModel):
|
||||
models: Optional[List[str]] = Field(
|
||||
default=None,
|
||||
description="Subset of models to (re)train: discount, trending, popularity, forecast, store_performance, purchase_propensity. Omit to train all.",
|
||||
description=(
|
||||
"Subset of models to (re)train. Any of: discount, trending, popularity, "
|
||||
"forecast, store_performance, purchase_propensity. Omit to train the "
|
||||
"models that are actually served (discount, trending, popularity) - the "
|
||||
"other three fit fine but nothing reads their output back, so ask for "
|
||||
"them by name."
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
|
||||
786
app/core/batch_ingest.py
Normal file
786
app/core/batch_ingest.py
Normal file
@@ -0,0 +1,786 @@
|
||||
"""
|
||||
Multi-file store spreadsheets -> the same 11-stage pipeline -> brand tables.
|
||||
|
||||
WHAT THIS ADDS OVER `store_catalog_pipeline`
|
||||
--------------------------------------------
|
||||
Nothing about the pipeline itself. `run_pipeline()` already turns one whole
|
||||
spreadsheet into catalog rows, and this module calls it unchanged, once per
|
||||
file. What is new is everything *around* a file:
|
||||
|
||||
* several files are one unit of work with one id, so "did the whole drop
|
||||
land?" has an answer;
|
||||
* the uploads are written to disk before any work starts, so a restart
|
||||
mid-batch loses nothing but time;
|
||||
* one file failing does not take the others with it.
|
||||
|
||||
WHY THE FILES GO TO DISK
|
||||
------------------------
|
||||
The obvious cheap design - read the upload into memory and hand the bytes to a
|
||||
daemon thread - is fine for one file and one operator watching it: if the
|
||||
process dies, they re-upload. (An earlier single-file admin route did exactly
|
||||
that; it has since been removed.) A five-file batch is a different proposition
|
||||
- the colleague who sent them is not sitting there, and
|
||||
silently losing the drop is worse than any amount of extra code. So the bytes
|
||||
are staged under BATCH_UPLOAD_DIR, which is on the container's declared volume,
|
||||
and a manifest records what state each file reached.
|
||||
|
||||
THIS MODULE IS THE SHARED CORE
|
||||
------------------------------
|
||||
Two callers, one implementation:
|
||||
|
||||
app/api/routers/batch_catalog.py -> production (worker thread)
|
||||
orchestration/assets/batch_catalog.py -> Dagster (development)
|
||||
|
||||
That is the same arrangement `orchestration/assets/catalog.py` already
|
||||
describes for the brand pipeline: "Wrapping rather than reimplementing is what
|
||||
stops the two paths drifting: a fix to a stage fixes both." Dagster is not
|
||||
deployed here and is not on any request path; it wraps these functions so the
|
||||
graph is inspectable and re-runnable locally, and production executes the very
|
||||
same code without paying for a daemon.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import time
|
||||
import uuid
|
||||
from dataclasses import asdict, dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Any, Callable, Dict, List, Optional, Tuple
|
||||
|
||||
from app.core import store_catalog_pipeline as pipeline
|
||||
from app.infrastructure.settings import (
|
||||
BATCH_AUTO_RESUME,
|
||||
BATCH_RETENTION_DAYS,
|
||||
BATCH_UPLOAD_DIR,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
MANIFEST_NAME = "manifest.json"
|
||||
|
||||
# File states. `cancelled` only ever applies to files that had not started.
|
||||
QUEUED = "queued"
|
||||
RUNNING = "running"
|
||||
DONE = "done"
|
||||
FAILED = "failed"
|
||||
CANCELLED = "cancelled"
|
||||
|
||||
# What became of a file that was sitting in the review inbox. Recorded on the
|
||||
# file rather than deleted with it, because the sender polls the id THEY were
|
||||
# given and has no other way to learn what happened: `released` carries the id
|
||||
# of the run that took it, `dismissed` says an admin declined it.
|
||||
RELEASED = "released"
|
||||
DISMISSED = "dismissed"
|
||||
|
||||
# Batch states. `partial` is not cosmetic: a batch where four of five files
|
||||
# landed must not read as a flat success, or nobody goes looking for the fifth.
|
||||
INTERRUPTED = "interrupted"
|
||||
PARTIAL = "partial"
|
||||
|
||||
# A drop sitting in the review inbox: staged on disk, deliberately NOT queued.
|
||||
# It is a batch state rather than a store of its own so that retention, the
|
||||
# manifest format and the job store all apply to it unchanged. Its FILES stay
|
||||
# `queued`, which is why `settle()` has to leave this status alone - see there.
|
||||
PENDING = "pending"
|
||||
|
||||
# A drop whose files have all been released or declined. It is kept as a record
|
||||
# for the sender to poll, holds no bytes, and ages out on normal retention.
|
||||
RETIRED = "retired"
|
||||
|
||||
TERMINAL_BATCH_STATES = {DONE, FAILED, PARTIAL, CANCELLED, RETIRED}
|
||||
|
||||
# Who runs a batch. See `BatchManifest.runner`.
|
||||
RUNNER_INPROCESS = "inprocess"
|
||||
RUNNER_DAGSTER = "dagster"
|
||||
RUNNERS = {RUNNER_INPROCESS, RUNNER_DAGSTER}
|
||||
|
||||
# `..`, separators and drive letters all stripped. UploadFile.filename is
|
||||
# attacker-controlled in the general case, and it is used to build a path.
|
||||
_UNSAFE = re.compile(r"[^A-Za-z0-9._-]+")
|
||||
|
||||
|
||||
def _safe_name(filename: str) -> str:
|
||||
"""A filename that cannot escape the batch directory.
|
||||
|
||||
Path components are discarded rather than escaped: nothing downstream needs
|
||||
the original directory, and `os.path.basename` alone is not enough here,
|
||||
because a Windows-authored name like "..\\evil.csv" keeps its backslash on
|
||||
a Linux container and basename leaves it untouched.
|
||||
"""
|
||||
base = str(filename or "upload.xlsx").replace("\\", "/").rsplit("/", 1)[-1]
|
||||
base = _UNSAFE.sub("_", base).lstrip(".") or "upload.xlsx"
|
||||
return base[:120]
|
||||
|
||||
|
||||
@dataclass
|
||||
class StageRecord:
|
||||
"""One of the 11 pipeline stages, as it happened to one file.
|
||||
|
||||
The scalar `stage_index`/`stage_name` fields below say where a file is *now*
|
||||
and are overwritten on every tick, so once a file finishes there is no trace
|
||||
of what it went through. This keeps that trace: a completed file can still
|
||||
show its whole timeline, which is the point of the orchestration view.
|
||||
|
||||
One record per stage INDEX, not per callback. Stages 8-11 run once per brand
|
||||
(`store_catalog_pipeline.run_pipeline` loops `for brand, rows in
|
||||
by_brand.items()`), so a three-brand sheet reports 8,9,10,11 three times
|
||||
over. Those fold into the same record - earliest start, latest finish,
|
||||
largest row counts - so a file always has at most eleven of these however
|
||||
many brands its rows land in.
|
||||
"""
|
||||
|
||||
index: int # 1-based, matching STAGE_NAMES
|
||||
name: str
|
||||
rows_done: int = 0
|
||||
rows_total: int = 0
|
||||
started_at: Optional[float] = None
|
||||
finished_at: Optional[float] = None
|
||||
|
||||
|
||||
@dataclass
|
||||
class BatchFile:
|
||||
"""One spreadsheet inside a batch, and how far it got."""
|
||||
|
||||
index: int
|
||||
filename: str # what the colleague called it, for display
|
||||
stored_name: str # what it is called on disk; "" once retired
|
||||
# The run that took this file out of the review inbox. On the FILE and not
|
||||
# the manifest because one drop can be released a few sheets at a time, into
|
||||
# different runs, and the sender needs to know which of theirs went where.
|
||||
released_to: Optional[str] = None
|
||||
# The REVERSE of released_to: on a file inside a RUN, the id of the drop it
|
||||
# was released from. Both directions are needed and they are not the same
|
||||
# question - released_to answers "where did my drop go?", from_drop answers
|
||||
# "whose file is this?".
|
||||
#
|
||||
# That second question is the one that can corrupt inventory. An admin may
|
||||
# assemble one run from several drops, and until this existed the only way
|
||||
# to narrow a run's manifest to your own file was to match on `filename` -
|
||||
# so two senders who both upload `products.csv` would price and shelve each
|
||||
# other's products, silently. Matching on this id is exact.
|
||||
#
|
||||
# None for a file uploaded straight into a run, which never sat in an inbox.
|
||||
from_drop: Optional[str] = None
|
||||
size_bytes: int = 0
|
||||
status: str = QUEUED
|
||||
detail: Optional[str] = None
|
||||
stage_index: int = 0 # 1-based; 0 while queued
|
||||
stage_name: str = ""
|
||||
total_stages: int = pipeline.TOTAL_STAGES
|
||||
rows_done: int = 0
|
||||
rows_total: int = 0
|
||||
# The stages this file has entered so far, in the order it entered them.
|
||||
# Empty while queued; eleven entries once the pipeline has run through.
|
||||
stages: List[StageRecord] = field(default_factory=list)
|
||||
result: Optional[Dict[str, Any]] = None
|
||||
started_at: Optional[float] = None
|
||||
finished_at: Optional[float] = None
|
||||
|
||||
|
||||
@dataclass
|
||||
class BatchManifest:
|
||||
"""The whole batch. Serialised to manifest.json verbatim."""
|
||||
|
||||
batch_id: str
|
||||
status: str = QUEUED
|
||||
use_llm: bool = False
|
||||
fetch_images: bool = False
|
||||
created_at: float = field(default_factory=time.time)
|
||||
updated_at: float = field(default_factory=time.time)
|
||||
detail: Optional[str] = None
|
||||
# Who sent the files, when they came through the review inbox. None for a
|
||||
# batch an admin uploaded directly. Carried so the Batch tab can say where
|
||||
# a run came from instead of leaving it to be guessed from filenames.
|
||||
submitted_by: Optional[str] = None
|
||||
# Which executor owns this batch: the API's worker thread, or Dagster.
|
||||
#
|
||||
# Both watch the same directory and both can run the same `run_batch`, and
|
||||
# until this field existed nothing arbitrated between them - enabling
|
||||
# `batch_upload_sensor` beside a running API meant both claimed every queued
|
||||
# batch and ingested it twice. The worker only ever runs what is explicitly
|
||||
# submitted to it, so the field is really a claim check for the Dagster
|
||||
# side: `_pick_batch_id` and the sensor ignore anything not marked "dagster".
|
||||
#
|
||||
# Defaults to INPROCESS so every manifest written before this existed, and
|
||||
# every batch an admin uploads directly, keeps behaving exactly as it did.
|
||||
runner: str = RUNNER_INPROCESS
|
||||
# The nutrition-enrichment job queued when this batch finished, pollable at
|
||||
# GET /api/admin/nutrition-intelligence/jobs/{job_id}. None when scoring is
|
||||
# switched off, when the batch produced no brands, or - importantly - for
|
||||
# every manifest written before this field existed, which is why `from_dict`
|
||||
# reads it with a default instead of requiring it.
|
||||
nutrition_job_id: Optional[str] = None
|
||||
files: List[BatchFile] = field(default_factory=list)
|
||||
|
||||
# -- derived, recomputed rather than stored, so they cannot drift ---------
|
||||
@property
|
||||
def files_total(self) -> int:
|
||||
return len(self.files)
|
||||
|
||||
@property
|
||||
def files_done(self) -> int:
|
||||
return sum(1 for f in self.files if f.status == DONE)
|
||||
|
||||
@property
|
||||
def files_failed(self) -> int:
|
||||
return sum(1 for f in self.files if f.status == FAILED)
|
||||
|
||||
@property
|
||||
def current_file(self) -> Optional[str]:
|
||||
for entry in self.files:
|
||||
if entry.status == RUNNING:
|
||||
return entry.filename
|
||||
return None
|
||||
|
||||
def totals(self) -> Dict[str, int]:
|
||||
"""Summed across every file that produced a result."""
|
||||
keys = ("rows_total", "products_built", "inserted", "backfilled",
|
||||
"skipped_existing", "rejected", "error_count")
|
||||
out = {key: 0 for key in keys}
|
||||
for entry in self.files:
|
||||
for key in keys:
|
||||
out[key] += int((entry.result or {}).get(key) or 0)
|
||||
return out
|
||||
|
||||
def brands(self) -> List[str]:
|
||||
seen = set()
|
||||
for entry in self.files:
|
||||
seen.update((entry.result or {}).get("brands") or [])
|
||||
return sorted(seen)
|
||||
|
||||
def settle(self) -> str:
|
||||
"""Recompute the batch status from its files. Returns the new status.
|
||||
|
||||
A `pending` submission is exempt. Its files are `queued` - they are
|
||||
genuinely waiting - so the rule below would promote the batch to
|
||||
`queued` and it would read as work already accepted, which is the one
|
||||
thing the review inbox exists to prevent.
|
||||
"""
|
||||
if self.status == PENDING:
|
||||
return self.status
|
||||
states = {f.status for f in self.files}
|
||||
if states & {QUEUED, RUNNING}:
|
||||
self.status = RUNNING if RUNNING in states else QUEUED
|
||||
elif self.files_done and self.files_failed:
|
||||
self.status = PARTIAL
|
||||
elif self.files_done:
|
||||
self.status = DONE
|
||||
elif self.files_failed:
|
||||
self.status = FAILED
|
||||
else:
|
||||
self.status = CANCELLED
|
||||
self.updated_at = time.time()
|
||||
return self.status
|
||||
|
||||
# -- serialisation -------------------------------------------------------
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return {
|
||||
"batch_id": self.batch_id,
|
||||
"status": self.status,
|
||||
"use_llm": self.use_llm,
|
||||
"fetch_images": self.fetch_images,
|
||||
"created_at": self.created_at,
|
||||
"updated_at": self.updated_at,
|
||||
"detail": self.detail,
|
||||
"submitted_by": self.submitted_by,
|
||||
"runner": self.runner,
|
||||
"nutrition_job_id": self.nutrition_job_id,
|
||||
"files_total": self.files_total,
|
||||
"files_done": self.files_done,
|
||||
"files_failed": self.files_failed,
|
||||
"current_file": self.current_file,
|
||||
"totals": self.totals(),
|
||||
"brands": self.brands(),
|
||||
"files": [asdict(f) for f in self.files],
|
||||
}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, raw: Dict[str, Any]) -> "BatchManifest":
|
||||
allowed = set(BatchFile.__dataclass_fields__)
|
||||
stage_fields = set(StageRecord.__dataclass_fields__)
|
||||
files = []
|
||||
for entry in raw.get("files") or []:
|
||||
fields_ = {k: v for k, v in entry.items() if k in allowed}
|
||||
# `asdict` flattened these to plain dicts on the way out; rebuild
|
||||
# them so callers get StageRecords whichever direction the manifest
|
||||
# came from (live object, or re-read off disk after a restart).
|
||||
fields_["stages"] = [
|
||||
StageRecord(**{k: v for k, v in stage.items() if k in stage_fields})
|
||||
for stage in (entry.get("stages") or [])
|
||||
]
|
||||
files.append(BatchFile(**fields_))
|
||||
return cls(
|
||||
batch_id=raw["batch_id"],
|
||||
status=raw.get("status", QUEUED),
|
||||
use_llm=bool(raw.get("use_llm", False)),
|
||||
fetch_images=bool(raw.get("fetch_images", False)),
|
||||
created_at=float(raw.get("created_at") or time.time()),
|
||||
updated_at=float(raw.get("updated_at") or time.time()),
|
||||
detail=raw.get("detail"),
|
||||
submitted_by=raw.get("submitted_by"),
|
||||
runner=raw.get("runner") or RUNNER_INPROCESS,
|
||||
nutrition_job_id=raw.get("nutrition_job_id"),
|
||||
files=files,
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Disk layout
|
||||
# ---------------------------------------------------------------------------
|
||||
def batch_root() -> Path:
|
||||
"""Read at call time, not import time, so tests can repoint the directory."""
|
||||
return Path(BATCH_UPLOAD_DIR)
|
||||
|
||||
|
||||
def batch_dir(batch_id: str) -> Path:
|
||||
# The id is generated here (uuid4), never taken from a request, but this is
|
||||
# still the function that turns it into a path - so it validates.
|
||||
if not re.fullmatch(r"[A-Za-z0-9_-]{1,64}", batch_id or ""):
|
||||
raise ValueError("Invalid batch id: {!r}".format(batch_id))
|
||||
return batch_root() / batch_id
|
||||
|
||||
|
||||
def manifest_path(batch_id: str) -> Path:
|
||||
return batch_dir(batch_id) / MANIFEST_NAME
|
||||
|
||||
|
||||
def write_manifest(manifest: BatchManifest) -> None:
|
||||
"""Write via a temp file and os.replace.
|
||||
|
||||
A half-written manifest.json is indistinguishable from a corrupt one on the
|
||||
next boot, and the recovery path reads every manifest it finds.
|
||||
"""
|
||||
target = manifest_path(manifest.batch_id)
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
tmp = target.with_name(MANIFEST_NAME + ".tmp")
|
||||
tmp.write_text(json.dumps(manifest.to_dict(), indent=2), encoding="utf-8")
|
||||
os.replace(tmp, target)
|
||||
|
||||
|
||||
def read_manifest(batch_id: str) -> Optional[BatchManifest]:
|
||||
try:
|
||||
path = manifest_path(batch_id)
|
||||
except ValueError:
|
||||
return None
|
||||
if not path.exists():
|
||||
return None
|
||||
try:
|
||||
return BatchManifest.from_dict(json.loads(path.read_text(encoding="utf-8")))
|
||||
except Exception as exc: # noqa: BLE001 - a bad manifest must not break a listing
|
||||
logger.warning("Ignoring unreadable manifest %s: %s", path, exc)
|
||||
return None
|
||||
|
||||
|
||||
def list_manifests() -> List[BatchManifest]:
|
||||
"""Every readable batch on disk, newest first."""
|
||||
root = batch_root()
|
||||
if not root.exists():
|
||||
return []
|
||||
found = []
|
||||
for child in sorted(root.iterdir()):
|
||||
if not child.is_dir():
|
||||
continue
|
||||
manifest = read_manifest(child.name)
|
||||
if manifest:
|
||||
found.append(manifest)
|
||||
return sorted(found, key=lambda m: m.created_at, reverse=True)
|
||||
|
||||
|
||||
def retire_files(batch_id: str, indices, *, state: str,
|
||||
released_to: Optional[str] = None) -> int:
|
||||
"""Take these files out of the review inbox. Returns how many.
|
||||
|
||||
Both halves of triage land here: `state=RELEASED` with the id of the run
|
||||
that took them, or `state=DISMISSED` when an admin declined them.
|
||||
|
||||
THE BYTES GO; THE RECORD STAYS. Deleting the manifest as well - which is
|
||||
what the first version did - left the sender polling the only id they were
|
||||
ever given and getting a 404, unable to tell "accepted and running" from
|
||||
"declined" from "lost". The file entry is therefore kept, marked, and for a
|
||||
release annotated with the run to follow. Disk, which is the thing an open
|
||||
endpoint can actually exhaust, is still reclaimed immediately.
|
||||
|
||||
Surviving files KEEP their original `index`. The inbox addresses a file as
|
||||
"{batch_id}:{index}" and the admin UI holds those ids in a selection set
|
||||
across polls, so renumbering would silently repoint a tick at another file.
|
||||
"""
|
||||
manifest = read_manifest(batch_id)
|
||||
if not manifest:
|
||||
return 0
|
||||
|
||||
wanted = {int(i) for i in indices}
|
||||
retired = 0
|
||||
for entry in manifest.files:
|
||||
if entry.index not in wanted or not entry.stored_name:
|
||||
continue
|
||||
try:
|
||||
(batch_dir(batch_id) / entry.stored_name).unlink(missing_ok=True)
|
||||
except OSError as exc:
|
||||
logger.warning(
|
||||
"Could not delete %s from drop %s: %s",
|
||||
entry.stored_name, batch_id, exc,
|
||||
)
|
||||
entry.stored_name = ""
|
||||
entry.status = state
|
||||
entry.released_to = released_to
|
||||
entry.finished_at = time.time()
|
||||
retired += 1
|
||||
|
||||
if not retired:
|
||||
return 0
|
||||
|
||||
# Nothing left for anyone to decide on, so it leaves the inbox. RETIRED is
|
||||
# terminal, which is also what makes purge_expired willing to reclaim the
|
||||
# directory once retention is up.
|
||||
if not any(f.status == QUEUED and f.stored_name for f in manifest.files):
|
||||
manifest.status = RETIRED
|
||||
manifest.detail = "Every file here has been started or dismissed."
|
||||
|
||||
manifest.updated_at = time.time()
|
||||
write_manifest(manifest)
|
||||
return retired
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Staging
|
||||
# ---------------------------------------------------------------------------
|
||||
def stage_batch(
|
||||
uploads: List[Tuple[str, bytes]],
|
||||
*,
|
||||
use_llm: bool = False,
|
||||
fetch_images: bool = False,
|
||||
invalid: Optional[List[Tuple[str, str]]] = None,
|
||||
submitted_by: Optional[str] = None,
|
||||
) -> BatchManifest:
|
||||
"""Write the uploads to disk and return the manifest describing them.
|
||||
|
||||
`invalid` carries files the caller already rejected (unparseable, empty).
|
||||
They are recorded as failed members of the batch rather than dropped: an
|
||||
operator who selected six files and sees five must be told what happened to
|
||||
the sixth, and the batch page is the only place they will look.
|
||||
|
||||
`submitted_by` is the credential name the files arrived under. It is what
|
||||
scopes an API client's view to its own batches, so it is set at staging
|
||||
time rather than patched on afterwards - a batch that existed for even a
|
||||
moment without an owner is a batch the ownership filter would hide from
|
||||
the only person entitled to see it.
|
||||
"""
|
||||
batch_id = uuid.uuid4().hex
|
||||
directory = batch_dir(batch_id)
|
||||
directory.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
manifest = BatchManifest(
|
||||
batch_id=batch_id,
|
||||
use_llm=use_llm,
|
||||
fetch_images=fetch_images,
|
||||
submitted_by=submitted_by,
|
||||
)
|
||||
|
||||
position = 0
|
||||
for filename, content in uploads:
|
||||
stored = "{:02d}_{}".format(position, _safe_name(filename))
|
||||
(directory / stored).write_bytes(content)
|
||||
manifest.files.append(
|
||||
BatchFile(
|
||||
index=position,
|
||||
filename=filename or stored,
|
||||
stored_name=stored,
|
||||
size_bytes=len(content),
|
||||
)
|
||||
)
|
||||
position += 1
|
||||
|
||||
for filename, reason in invalid or []:
|
||||
manifest.files.append(
|
||||
BatchFile(
|
||||
index=position,
|
||||
filename=filename or "(unnamed)",
|
||||
stored_name="",
|
||||
status=FAILED,
|
||||
detail=reason,
|
||||
finished_at=time.time(),
|
||||
)
|
||||
)
|
||||
position += 1
|
||||
|
||||
write_manifest(manifest)
|
||||
return manifest
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Execution
|
||||
# ---------------------------------------------------------------------------
|
||||
OnChange = Callable[["BatchManifest"], None]
|
||||
"""Called after every file-level transition, and on row progress.
|
||||
|
||||
The callback is what keeps the in-memory view the API serves in step with the
|
||||
run. It must be cheap - it fires on every progress tick - which is why the
|
||||
manifest is NOT written to disk from inside it.
|
||||
"""
|
||||
|
||||
|
||||
def _noop_change(manifest: "BatchManifest") -> None:
|
||||
return None
|
||||
|
||||
|
||||
# Row-level progress arrives many times a second. Persisting each one would be
|
||||
# hundreds of small writes per file, to record something nobody reads back from
|
||||
# disk anyway - the API answers polls from memory. Disk is for surviving a
|
||||
# restart, and a restart only needs to know which FILE was in flight.
|
||||
_PROGRESS_FLUSH_SECONDS = 5.0
|
||||
|
||||
|
||||
def _record_stage(entry: "BatchFile", index: int, name: str,
|
||||
done: int, total: int, now: float) -> None:
|
||||
"""Fold one progress tick into `entry.stages`.
|
||||
|
||||
Keyed by stage INDEX rather than appended, because the pipeline visits
|
||||
stages 8-11 once per brand in the sheet: a three-brand file reports
|
||||
8,9,10,11 three times over, with `rows_total` reset to that brand's group
|
||||
size each pass. Appending would produce twenty-three entries for eleven
|
||||
stages and a UI that appears to run backwards. Folding keeps the first
|
||||
`started_at`, extends `finished_at`, and takes the high-water mark of both
|
||||
row counts, so the record reads as "this stage, across the whole file".
|
||||
|
||||
Entering a stage closes every record before it. That is deliberate rather
|
||||
than closing only the immediately-previous one: stage 1 emits a single tick
|
||||
and stages 8-11 interleave, so "everything with a lower index is done" is
|
||||
the only rule that leaves no record permanently open.
|
||||
"""
|
||||
if index <= 0:
|
||||
return
|
||||
for stage in entry.stages:
|
||||
if stage.index < index and stage.finished_at is None:
|
||||
stage.finished_at = now
|
||||
|
||||
for stage in entry.stages:
|
||||
if stage.index == index:
|
||||
stage.rows_done = max(stage.rows_done, done)
|
||||
stage.rows_total = max(stage.rows_total, total)
|
||||
stage.finished_at = None if done < total else now
|
||||
return
|
||||
|
||||
entry.stages.append(StageRecord(
|
||||
index=index,
|
||||
name=name,
|
||||
rows_done=done,
|
||||
rows_total=total,
|
||||
started_at=now,
|
||||
finished_at=now if total and done >= total else None,
|
||||
))
|
||||
|
||||
|
||||
def _close_stages(entry: "BatchFile", now: float) -> None:
|
||||
"""Mark whatever is still open as finished, once the file itself is done."""
|
||||
for stage in entry.stages:
|
||||
if stage.finished_at is None:
|
||||
stage.finished_at = now
|
||||
|
||||
|
||||
def run_batch(
|
||||
batch_id: str,
|
||||
*,
|
||||
on_change: OnChange = _noop_change,
|
||||
should_cancel: Optional[Callable[[], bool]] = None,
|
||||
) -> BatchManifest:
|
||||
"""Run every queued file in the batch, in order, one at a time.
|
||||
|
||||
Files are independent. A file that raises is marked failed with the reason
|
||||
and the loop moves to the next one - the pipeline's own "a stage never
|
||||
raises" rule protects rows within a file, and this is the same idea one
|
||||
level up.
|
||||
"""
|
||||
manifest = read_manifest(batch_id)
|
||||
if manifest is None:
|
||||
raise FileNotFoundError("No manifest for batch {}".format(batch_id))
|
||||
|
||||
directory = batch_dir(batch_id)
|
||||
manifest.status = RUNNING
|
||||
manifest.detail = None
|
||||
manifest.updated_at = time.time()
|
||||
write_manifest(manifest)
|
||||
on_change(manifest)
|
||||
|
||||
for entry in manifest.files:
|
||||
if entry.status != QUEUED:
|
||||
continue
|
||||
|
||||
if should_cancel is not None and should_cancel():
|
||||
entry.status = CANCELLED
|
||||
entry.detail = "Cancelled before this file started."
|
||||
entry.finished_at = time.time()
|
||||
continue
|
||||
|
||||
entry.status = RUNNING
|
||||
entry.started_at = time.time()
|
||||
entry.stage_index = 0
|
||||
entry.stage_name = ""
|
||||
# A re-run of an interrupted file starts its timeline over rather than
|
||||
# appending to the one from the attempt that died.
|
||||
entry.stages = []
|
||||
write_manifest(manifest)
|
||||
on_change(manifest)
|
||||
|
||||
last_flush = [time.time()]
|
||||
|
||||
def progress(stage_index: int, stage_name: str, done: int, total: int,
|
||||
_entry: BatchFile = entry) -> None:
|
||||
now = time.time()
|
||||
_record_stage(_entry, stage_index, stage_name, done, total, now)
|
||||
_entry.stage_index = stage_index
|
||||
_entry.stage_name = stage_name
|
||||
_entry.rows_done = done
|
||||
_entry.rows_total = total
|
||||
manifest.updated_at = now
|
||||
on_change(manifest)
|
||||
if now - last_flush[0] >= _PROGRESS_FLUSH_SECONDS:
|
||||
last_flush[0] = now
|
||||
write_manifest(manifest)
|
||||
|
||||
try:
|
||||
content = (directory / entry.stored_name).read_bytes()
|
||||
result = pipeline.run_pipeline(
|
||||
entry.filename,
|
||||
content,
|
||||
progress=progress,
|
||||
use_llm=manifest.use_llm,
|
||||
fetch_images=manifest.fetch_images,
|
||||
)
|
||||
body = result.as_dict()
|
||||
entry.result = body
|
||||
if body.get("storage_error"):
|
||||
# Same judgement as the single-file path: rows were built but
|
||||
# none reached the database, and calling that success would
|
||||
# leave the operator believing the catalog changed.
|
||||
entry.status = FAILED
|
||||
entry.detail = (
|
||||
"Built {} row(s) but storing them failed: {}".format(
|
||||
body["products_built"], body["storage_error"]
|
||||
)
|
||||
)
|
||||
else:
|
||||
entry.status = DONE
|
||||
entry.detail = (
|
||||
"{} inserted, {} backfilled, {} unchanged, {} rejected".format(
|
||||
body["inserted"], body["backfilled"],
|
||||
body["skipped_existing"], body["rejected"],
|
||||
)
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001 - one bad file must not end the batch
|
||||
logger.exception("Batch %s: file %s failed", batch_id, entry.filename)
|
||||
entry.status = FAILED
|
||||
entry.detail = str(exc)
|
||||
|
||||
entry.finished_at = time.time()
|
||||
# The last stage never sees a "next stage" tick to close it, and a file
|
||||
# that raised leaves whichever stage it died in open.
|
||||
_close_stages(entry, entry.finished_at)
|
||||
write_manifest(manifest)
|
||||
on_change(manifest)
|
||||
|
||||
manifest.settle()
|
||||
|
||||
# Score what was just ingested. Submitted here, after the last file has
|
||||
# settled, rather than per file: one job for the whole batch means one
|
||||
# `skip_if_verified` sweep per brand instead of one per spreadsheet, and a
|
||||
# batch of twenty files for the same brand does not queue twenty jobs.
|
||||
#
|
||||
# This runs on the batch worker's thread, so it must not do the enrichment
|
||||
# itself - `submit_enrichment_for_brands` only starts a separate thread and
|
||||
# returns. It also swallows its own failures: the products ARE ingested by
|
||||
# this point, and a scoring step that could not start is not a reason to
|
||||
# tell the operator their catalogue import failed.
|
||||
try:
|
||||
from app.services.nutrition_autoenrich import submit_enrichment_for_brands
|
||||
|
||||
manifest.nutrition_job_id = submit_enrichment_for_brands(
|
||||
manifest.brands(), source="batch {}".format(batch_id))
|
||||
except Exception: # noqa: BLE001
|
||||
logger.exception("Batch %s: could not queue nutrition enrichment", batch_id)
|
||||
|
||||
write_manifest(manifest)
|
||||
on_change(manifest)
|
||||
return manifest
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Recovery and retention
|
||||
# ---------------------------------------------------------------------------
|
||||
def scan_interrupted() -> List[str]:
|
||||
"""Mark batches that a restart cut short. Returns the ids resumable now.
|
||||
|
||||
DELIBERATELY DOES NOT RE-RUN ANYTHING by default. A container caught in a
|
||||
restart loop would otherwise re-enter the heaviest work in the application
|
||||
on every boot, turning a slow start into an unrecoverable one. The files and
|
||||
the manifest are on the volume, so nothing is lost by waiting for a human to
|
||||
press Resume - and BATCH_AUTO_RESUME=true is there for a deployment that has
|
||||
earned the trust.
|
||||
"""
|
||||
resumable: List[str] = []
|
||||
for manifest in list_manifests():
|
||||
if manifest.status in TERMINAL_BATCH_STATES or manifest.status == INTERRUPTED:
|
||||
continue
|
||||
# A pending drop was never running, so a restart did not cut it short.
|
||||
# This sweep takes every non-terminal manifest, so without this line
|
||||
# every restart would relabel the whole review inbox "interrupted" and
|
||||
# offer an admin a Resume button for work nobody had started.
|
||||
if manifest.status == PENDING:
|
||||
continue
|
||||
for entry in manifest.files:
|
||||
if entry.status == RUNNING:
|
||||
entry.status = QUEUED
|
||||
entry.stage_index = 0
|
||||
entry.stage_name = ""
|
||||
entry.rows_done = 0
|
||||
# The timeline described an attempt that no longer counts; the
|
||||
# re-run builds a fresh one.
|
||||
entry.stages = []
|
||||
entry.detail = "Interrupted by a restart; queued again."
|
||||
manifest.status = INTERRUPTED
|
||||
manifest.detail = "Interrupted by a restart. Press Resume to continue."
|
||||
manifest.updated_at = time.time()
|
||||
write_manifest(manifest)
|
||||
resumable.append(manifest.batch_id)
|
||||
|
||||
if resumable and not BATCH_AUTO_RESUME:
|
||||
logger.info(
|
||||
"Found %d interrupted batch(es); leaving them for a manual resume "
|
||||
"(BATCH_AUTO_RESUME is false).", len(resumable),
|
||||
)
|
||||
return resumable
|
||||
|
||||
|
||||
def purge_expired(now: Optional[float] = None) -> List[str]:
|
||||
"""Delete staged files for batches older than BATCH_RETENTION_DAYS.
|
||||
|
||||
Called when the worker goes idle, never on a request path - deleting a few
|
||||
hundred megabytes should not be something an operator waits on.
|
||||
"""
|
||||
if BATCH_RETENTION_DAYS <= 0:
|
||||
return []
|
||||
cutoff = (now if now is not None else time.time()) - BATCH_RETENTION_DAYS * 86400
|
||||
removed: List[str] = []
|
||||
for manifest in list_manifests():
|
||||
if manifest.created_at >= cutoff:
|
||||
continue
|
||||
if manifest.status not in TERMINAL_BATCH_STATES and manifest.status != PENDING:
|
||||
# An old batch still queued is a bug somewhere, but deleting the
|
||||
# only copy of its input is not the way to find out.
|
||||
#
|
||||
# A PENDING drop is the exception, and the reason retention matters
|
||||
# now: /api/uploads/catalog takes files from anyone, and nothing
|
||||
# about an unreviewed submission ever reaches a terminal state. Left
|
||||
# out of this sweep it would occupy the volume permanently.
|
||||
continue
|
||||
try:
|
||||
shutil.rmtree(batch_dir(manifest.batch_id))
|
||||
removed.append(manifest.batch_id)
|
||||
except OSError as exc:
|
||||
logger.warning("Could not purge batch %s: %s", manifest.batch_id, exc)
|
||||
if removed:
|
||||
logger.info("Purged %d expired batch upload(s).", len(removed))
|
||||
return removed
|
||||
122
app/core/batch_worker.py
Normal file
122
app/core/batch_worker.py
Normal file
@@ -0,0 +1,122 @@
|
||||
"""One worker thread for every batch this process will ever run.
|
||||
|
||||
WHY A SINGLE BOUNDED WORKER, AND NOT `run_in_background`
|
||||
---------------------------------------------------------
|
||||
`app/api/background.py` starts a fresh daemon thread per job and returns. That
|
||||
is right for the jobs it serves - they are started by hand, one at a time, by
|
||||
an operator watching the result. It is the wrong shape for this feature.
|
||||
|
||||
A batch is up to twenty spreadsheets of two thousand rows. The deployment this
|
||||
runs on is a single container with one vCPU and no CPU limit, so nothing above
|
||||
stops two batches from interleaving; they would simply both run, at half speed
|
||||
each, while the API tries to answer requests and the health probe tries to get
|
||||
a socket. Two operators uploading at the same time is not an unusual event, it
|
||||
is a Tuesday.
|
||||
|
||||
So: one thread, one batch at a time, and a bounded queue in front. A second
|
||||
batch waits its turn instead of competing, and beyond `BATCH_QUEUE_MAX` the
|
||||
endpoint says 429 rather than accepting work it has no intention of starting.
|
||||
|
||||
THE THREAD IS STARTED LAZILY
|
||||
----------------------------
|
||||
Not at import, not in the lifespan startup. A container that never receives a
|
||||
batch pays nothing for this module beyond the import, which is why `submit()`
|
||||
is the only thing that can bring the worker to life. Boot cost on this host is
|
||||
already ~21s of imports and is the thing most worth not adding to.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import queue
|
||||
import threading
|
||||
from typing import Callable, Optional
|
||||
|
||||
from app.core import batch_ingest
|
||||
from app.infrastructure.settings import BATCH_QUEUE_MAX
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
QueueFull = queue.Full
|
||||
|
||||
_queue: "queue.Queue[str]" = queue.Queue(maxsize=max(1, BATCH_QUEUE_MAX))
|
||||
_worker: Optional[threading.Thread] = None
|
||||
_lock = threading.Lock()
|
||||
|
||||
# Set by the router at import so this module does not import the API layer -
|
||||
# app.api.batch_job_store already imports app.core.batch_ingest, and closing
|
||||
# that loop the other way would be a circular import at boot.
|
||||
_on_change: Optional[Callable[[batch_ingest.BatchManifest], None]] = None
|
||||
_should_cancel: Optional[Callable[[str], bool]] = None
|
||||
|
||||
|
||||
def configure(
|
||||
*,
|
||||
on_change: Callable[[batch_ingest.BatchManifest], None],
|
||||
should_cancel: Callable[[str], bool],
|
||||
) -> None:
|
||||
"""Wire the worker to the job store. Called once, by the router module."""
|
||||
global _on_change, _should_cancel
|
||||
_on_change = on_change
|
||||
_should_cancel = should_cancel
|
||||
|
||||
|
||||
def queue_depth() -> int:
|
||||
return _queue.qsize()
|
||||
|
||||
|
||||
def is_running() -> bool:
|
||||
return _worker is not None and _worker.is_alive()
|
||||
|
||||
|
||||
def submit(batch_id: str) -> None:
|
||||
"""Enqueue a batch and make sure the worker exists. Raises `queue.Full`.
|
||||
|
||||
`put_nowait` rather than `put`: blocking here would block the request
|
||||
handler, which is the one thing an endpoint that returns 202 must never do.
|
||||
"""
|
||||
_queue.put_nowait(batch_id)
|
||||
_ensure_worker()
|
||||
|
||||
|
||||
def _ensure_worker() -> None:
|
||||
global _worker
|
||||
with _lock:
|
||||
if _worker is not None and _worker.is_alive():
|
||||
return
|
||||
_worker = threading.Thread(target=_loop, name="catalog-batch-worker", daemon=True)
|
||||
_worker.start()
|
||||
|
||||
|
||||
def _loop() -> None:
|
||||
"""Drain the queue forever.
|
||||
|
||||
Every iteration is wrapped, because a worker that dies on one bad batch
|
||||
would leave every future batch queued behind a thread that is not there -
|
||||
a failure that looks, from the UI, exactly like a batch that is merely slow.
|
||||
"""
|
||||
while True:
|
||||
batch_id = _queue.get()
|
||||
try:
|
||||
_run_one(batch_id)
|
||||
except Exception: # noqa: BLE001 - see docstring
|
||||
logger.exception("Batch worker: unhandled error on batch %s", batch_id)
|
||||
finally:
|
||||
_queue.task_done()
|
||||
|
||||
if _queue.empty():
|
||||
# Retention runs when there is nothing waiting, so deleting old
|
||||
# uploads never delays a batch and never sits on a request path.
|
||||
try:
|
||||
batch_ingest.purge_expired()
|
||||
except Exception: # noqa: BLE001 - housekeeping must not kill the worker
|
||||
logger.exception("Batch worker: purge failed")
|
||||
|
||||
|
||||
def _run_one(batch_id: str) -> None:
|
||||
on_change = _on_change or (lambda manifest: None)
|
||||
cancelled = _should_cancel or (lambda _id: False)
|
||||
batch_ingest.run_batch(
|
||||
batch_id,
|
||||
on_change=on_change,
|
||||
should_cancel=lambda: cancelled(batch_id),
|
||||
)
|
||||
@@ -7,24 +7,28 @@ resort). The Node.js/Crawlee scraping layer that used to live here has
|
||||
been removed - see app/services/image_search.py and
|
||||
app/services/playwright_image_fallback.py for details.
|
||||
"""
|
||||
import asyncio
|
||||
import json
|
||||
import re
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Dict, List, Optional, Any
|
||||
from typing import Dict, List, Any
|
||||
import logging
|
||||
|
||||
# Add app directory to path
|
||||
sys.path.append(str(Path(__file__).parent.parent))
|
||||
|
||||
from app.services.ollama_service import fetch_brand_catalog_with_gemini, fetch_brand_catalog_exhaustive, fetch_product_details
|
||||
from app.services.image_search import find_all_image_urls, find_product_quantity_openfacts
|
||||
from app.services.image_search import find_all_image_urls
|
||||
from app.infrastructure.settings import DATA_DIR, USE_OLLAMA
|
||||
from app.services.embeddings_service import embed_texts
|
||||
from app.services.vector_store import ensure_brand_schema, upsert_brand_products, get_existing_product_image_id
|
||||
from app.services.s3_service import s3_service
|
||||
from app.services.brand_registry import resolve_parent_brand
|
||||
from app.services import image_corroboration
|
||||
from app.services import price_estimator
|
||||
from app.services import product_grounding
|
||||
from app.services.enrichment.barcode.sources import off_bulk
|
||||
from app.services.product_validator import validate_catalog
|
||||
from app.services.category_registry import detect_category_from_text, sanitize_category_language
|
||||
|
||||
# Configure logging
|
||||
@@ -41,10 +45,8 @@ def generate_product_highlights(product: Dict[str, Any], brand: str) -> List[str
|
||||
product = {}
|
||||
|
||||
highlights = []
|
||||
title = str(product.get('title', '')).strip()
|
||||
description = str(product.get('description', '')).strip()
|
||||
category = str(product.get('category', '')).strip()
|
||||
price_range = str(product.get('price_range', '')).strip()
|
||||
size_variants = product.get('size_variants', [])
|
||||
|
||||
# Ensure size_variants is a list
|
||||
@@ -89,7 +91,8 @@ def generate_product_highlights(product: Dict[str, Any], brand: str) -> List[str
|
||||
highlights.append(f"Available in {size} - {price}")
|
||||
elif size:
|
||||
highlights.append(f"Available in {size}")
|
||||
elif isinstance(v, str) and v.strip():
|
||||
elif isinstance(v, str) and v.strip() and v.strip().lower() != "standard":
|
||||
# "Standard" is the pipeline's no-size sentinel, not a pack size.
|
||||
highlights.append(f"Available in {v}")
|
||||
|
||||
# Quality indicators from description
|
||||
@@ -379,10 +382,41 @@ class ProductCatalogEngine:
|
||||
"""Select the best images from a list of URLs based on quality and relevance"""
|
||||
if not image_urls:
|
||||
return []
|
||||
|
||||
|
||||
# Words that distinguish THIS product from its brand-mates. The brand
|
||||
# itself is removed on purpose: every Britannia URL contains
|
||||
# "britannia", so it separates nothing - what tells Marie Gold from Good
|
||||
# Day is "marie"/"gold" vs "good"/"day". Pack sizes and short filler
|
||||
# words are dropped for the same reason.
|
||||
_brand_words = {w for w in re.split(r"[^a-z0-9]+", (brand or "").lower()) if w}
|
||||
_distinctive: List[str] = []
|
||||
for word in re.split(r"[^a-z0-9]+", (product_title or "").lower()):
|
||||
if (
|
||||
len(word) > 2
|
||||
and word not in _brand_words
|
||||
and word not in _distinctive
|
||||
and not re.fullmatch(r"\d+(?:kg|g|gm|gms|ml|l|ltr|pcs|n)?", word)
|
||||
):
|
||||
_distinctive.append(word)
|
||||
|
||||
# Earlier words identify a product more strongly than later ones: a
|
||||
# title runs brand -> sub-brand -> variant -> size, so in "Coca-Cola
|
||||
# Sprite Lemon 750ml" the word that separates this product from its
|
||||
# brand-mates is "sprite", and "lemon" is only a modifier. Without this
|
||||
# weighting a combo listing that merely shares the modifier
|
||||
# ("...coca-cola-750-ml-limca-soft-drink-lemon-lime...") ties with the
|
||||
# real Sprite image and then wins on domain reputation.
|
||||
_weights = {word: len(_distinctive) - i for i, word in enumerate(_distinctive)}
|
||||
|
||||
def _mentions_product(url_lower: str) -> int:
|
||||
"""How strongly the URL names THIS product, not merely its brand."""
|
||||
if not _distinctive:
|
||||
return 1 # nothing to distinguish by; don't penalise anything
|
||||
return sum(weight for word, weight in _weights.items() if word in url_lower)
|
||||
|
||||
# Filter and score images
|
||||
scored_images = []
|
||||
|
||||
|
||||
for url in image_urls:
|
||||
if not url or not url.startswith('http'):
|
||||
continue
|
||||
@@ -479,14 +513,41 @@ class ProductCatalogEngine:
|
||||
# Avoid problematic URLs
|
||||
if any(bad in url_lower for bad in ['encrypted-tbn', 'googleusercontent', 'data:', 'placeholder']):
|
||||
score -= 5
|
||||
|
||||
scored_images.append((score, url))
|
||||
|
||||
# Sort by score (highest first) and take top images
|
||||
scored_images.sort(key=lambda x: x[0], reverse=True)
|
||||
best_images = [url for score, url in scored_images[:max_images]]
|
||||
|
||||
logger.info(f"📸 Selected {len(best_images)} best images for {product_title}")
|
||||
|
||||
# Multi-product listings picture several products at once, so even
|
||||
# when they name this one the photo is not of it alone.
|
||||
if any(kw in url_lower for kw in ['combo', 'multipack', 'multi-pack', 'pack-of', 'packof', 'assorted', 'variety-pack']):
|
||||
score -= 8
|
||||
|
||||
scored_images.append((_mentions_product(url_lower), score, url))
|
||||
|
||||
# Sort by (does the URL name this product, then score).
|
||||
#
|
||||
# Relevance has to be a GATE, not another additive term. Domain
|
||||
# reputation is worth up to +20 here while a matching product word is
|
||||
# worth +1, so a BigBasket photo of a Limca combo (15 + 2 + 2 + 2 = 21)
|
||||
# outranked the correct Sprite image on a lesser domain (8 + 2 + 1 + 1 =
|
||||
# 12) - and index 0 is what becomes the product's `image_url`. That is
|
||||
# the "top image is not this product" bug. Ranking every URL that names
|
||||
# the product above every URL that doesn't makes the primary image
|
||||
# correct, while the existing scoring still orders each group.
|
||||
#
|
||||
# Non-matching URLs are kept as a tail rather than dropped: a product
|
||||
# whose distinctive words never appear in any URL (common for
|
||||
# CDN-hashed filenames) would otherwise end up with no images at all.
|
||||
scored_images.sort(key=lambda x: (x[0], x[1]), reverse=True)
|
||||
best_images = [url for _relevant, _score, url in scored_images[:max_images]]
|
||||
|
||||
matched = sum(1 for relevant, _s, _u in scored_images[:max_images] if relevant)
|
||||
logger.info(
|
||||
f"📸 Selected {len(best_images)} best images for {product_title} "
|
||||
f"({matched} naming the product)"
|
||||
)
|
||||
if best_images and not matched:
|
||||
logger.warning(
|
||||
f"📸 No candidate image URL names {product_title!r} - the primary "
|
||||
f"image may not be this product"
|
||||
)
|
||||
return best_images
|
||||
|
||||
def search_with_python(self, query: str, brand: str) -> List[str]:
|
||||
@@ -555,15 +616,43 @@ class ProductCatalogEngine:
|
||||
|
||||
# Step 2: Enhance each product with comprehensive image search
|
||||
enhanced_products = []
|
||||
|
||||
# Parallel to enhanced_products: did an external source corroborate
|
||||
# that each product exists? Collected here and handed to
|
||||
# validate_catalog at the end - see the Step 3 comment for why this
|
||||
# path had no validation gate at all until now.
|
||||
grounded_flags: List[bool] = []
|
||||
|
||||
# Cap products by max_products
|
||||
discovered_products = discovered_products[:max_products]
|
||||
|
||||
# Fetched ONCE for the whole brand, not once per product: this is a
|
||||
# disk-cached whole-catalogue fetch (1-5 requests per brand), which is
|
||||
# the entire reason it is used here instead of the per-product live
|
||||
# search that looks like the natural fit. See product_grounding's
|
||||
# docstring - that endpoint is rate-limited to 10 requests/minute and
|
||||
# was returning 503 when measured.
|
||||
try:
|
||||
brand_corpus = off_bulk.fetch_brand_corpus(brand)
|
||||
except Exception as e: # noqa: BLE001 - unreachable corpus is not a verdict
|
||||
logger.warning("Open*Facts corpus unavailable for %s: %s", brand, e)
|
||||
brand_corpus = None
|
||||
if brand_corpus is not None:
|
||||
logger.info("📚 Open*Facts corpus for %s: %d real products", brand, len(brand_corpus))
|
||||
|
||||
total_products_to_process = len(discovered_products)
|
||||
for i, product in enumerate(discovered_products):
|
||||
product_title = product.get('title', '')
|
||||
logger.info(f"🔍 Processing product {i+1}/{total_products_to_process}: {product_title}")
|
||||
|
||||
|
||||
# Does anything outside this process say this product exists?
|
||||
# The LLM that produced `product_title` is a 1.5B local model and
|
||||
# cannot be asked to check its own work.
|
||||
grounding = product_grounding.ground_product(
|
||||
brand, product_title, corpus=brand_corpus
|
||||
)
|
||||
if grounding.status == product_grounding.NOT_FOUND:
|
||||
logger.warning("⚠️ Ungrounded: %r - %s", product_title, grounding.note())
|
||||
|
||||
# Strip brand prefix from title if present (avoids redundant
|
||||
# "Cadbury Perk Cadbury Perk Crunch" style queries that confuse
|
||||
# image search APIs).
|
||||
@@ -614,7 +703,26 @@ class ProductCatalogEngine:
|
||||
# Select exactly 20 best images
|
||||
all_prioritized = prioritized_images + other_images
|
||||
final_images = self._select_best_images(all_prioritized, product_title, brand, max_images=20)
|
||||
|
||||
|
||||
# _select_best_images ORDERS; it does not decide whether any
|
||||
# candidate is this product. Applied BEFORE the S3 upload below so a
|
||||
# photo of somebody else is never copied into our own bucket, where
|
||||
# its origin stops being visible at all.
|
||||
#
|
||||
# Nothing is dropped - the whole list is still uploaded and stored,
|
||||
# so an operator can look. Only the promotion to `image_url` is
|
||||
# withheld, and only when no candidate corroborates the product.
|
||||
image_choice = image_corroboration.choose_primary(
|
||||
final_images, product_title, brand or ""
|
||||
)
|
||||
final_images = list(image_choice.ordered)
|
||||
primary_eligible = image_choice.primary is not None
|
||||
if final_images and not primary_eligible:
|
||||
logger.warning(
|
||||
"No corroborated image for %r (%s) - storing candidates but "
|
||||
"leaving image_url empty", product_title, image_choice.reason,
|
||||
)
|
||||
|
||||
# Check if product already exists in DB to avoid duplicates
|
||||
image_id_val = ""
|
||||
try:
|
||||
@@ -780,18 +888,19 @@ class ProductCatalogEngine:
|
||||
if not _variant_size_price_pairs:
|
||||
default_sizes = price_estimator.default_size_variants(category_value, product_title)
|
||||
|
||||
# Ground this in a real packaging size where possible:
|
||||
# Open Food/Beauty/Products Facts reports an actual
|
||||
# `quantity` field (e.g. "200 g", "1 l") for products it
|
||||
# has on file, which is far more trustworthy than the
|
||||
# category preset list (itself just a fallback for when
|
||||
# *nothing* else is known). If we have one, swap it in for
|
||||
# the closest preset rather than presenting a size that may
|
||||
# not actually exist for this exact product.
|
||||
try:
|
||||
real_qty = find_product_quantity_openfacts(product_title, brand)
|
||||
except Exception:
|
||||
real_qty = None
|
||||
# Ground this in a real packaging size where possible. The
|
||||
# quantity comes from the brand corpus fetched once above,
|
||||
# NOT from find_product_quantity_openfacts, which issues a
|
||||
# live per-product query against an endpoint rate-limited to
|
||||
# 10 requests/minute - see product_grounding's docstring.
|
||||
#
|
||||
# ONLY WHEN THE PRODUCT ITSELF IS CORROBORATED. A quantity
|
||||
# borrowed from a product that merely scored well is worse
|
||||
# than the preset it replaces: it makes a fabricated row look
|
||||
# MORE real by dressing it in a size that genuinely exists.
|
||||
# An ungrounded product keeps the honest category preset and
|
||||
# is flagged for review instead.
|
||||
real_qty = grounding.quantity if grounding.is_grounded else None
|
||||
if real_qty and real_qty not in default_sizes:
|
||||
default_sizes = [real_qty] + default_sizes[:2]
|
||||
|
||||
@@ -820,7 +929,7 @@ class ProductCatalogEngine:
|
||||
price_str = variant.split(' - ₹')[1]
|
||||
price_num = float(price_str)
|
||||
prices.append(price_num)
|
||||
except:
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
if prices:
|
||||
min_price = min(prices)
|
||||
@@ -859,10 +968,16 @@ class ProductCatalogEngine:
|
||||
# is only used as a last resort since small local models frequently
|
||||
# hallucinate image links that don't actually resolve to an image.
|
||||
all_image_urls = s3_uploaded_urls or final_images
|
||||
primary_image = all_image_urls[0] if all_image_urls else (
|
||||
# `primary_eligible` is the corroboration verdict computed above.
|
||||
# S3 preserves candidate order (image_000, image_001, ...), so
|
||||
# index 0 here is the same picture choose_primary judged.
|
||||
primary_image = (all_image_urls[0] if (all_image_urls and primary_eligible) else None) or (
|
||||
enriched_img if enriched_img and str(enriched_img).startswith('http') else None
|
||||
)
|
||||
if not primary_image and s3_service.enabled and image_id_val:
|
||||
# Only reachable when there was no usable candidate at all. Guarded
|
||||
# by `primary_eligible` too, because this constructs a URL to
|
||||
# image_000 - the very image corroboration just declined to promote.
|
||||
if not primary_image and primary_eligible and s3_service.enabled and image_id_val:
|
||||
primary_image = s3_service.get_product_image_url(brand, image_id_val)
|
||||
if not all_image_urls and primary_image:
|
||||
all_image_urls = [primary_image]
|
||||
@@ -902,8 +1017,36 @@ class ProductCatalogEngine:
|
||||
}
|
||||
|
||||
enhanced_products.append(enhanced_product)
|
||||
grounded_flags.append(grounding.is_grounded)
|
||||
logger.info(f"✅ Enhanced {product_title}: {len(final_images)} images")
|
||||
|
||||
|
||||
# Step 2.6: THE DETERMINISTIC VALIDATION GATE.
|
||||
#
|
||||
# This path did not have one. `validate_catalog` was called from the
|
||||
# spreadsheet pipeline and from the Dagster assets, but never from
|
||||
# here, so POST /api/catalog/generate wrote straight to the brand
|
||||
# table with nothing between the language model and the database -
|
||||
# while product_validator's own docstring claimed this call site
|
||||
# already existed. That claim has been corrected along with this fix.
|
||||
#
|
||||
# Rows are annotated and kept, not dropped: `validate_catalog` returns
|
||||
# "verified" and "needs_review" rows together, and A1 gave the brand
|
||||
# tables somewhere to record which is which. Only rows scoring below
|
||||
# the reject threshold are withheld.
|
||||
enhanced_products, rejected, summary = validate_catalog(
|
||||
enhanced_products, brand, grounded_flags=grounded_flags,
|
||||
# These rows came out of a 1.5B language model, so "well formed"
|
||||
# is not evidence of anything. A row nothing corroborates is kept
|
||||
# and scored, but capped at needs_review rather than presented as
|
||||
# verified.
|
||||
require_grounding=True,
|
||||
)
|
||||
if rejected:
|
||||
logger.warning(
|
||||
"🚫 %d of %d generated product(s) failed validation for %s",
|
||||
len(rejected), summary.get("total_evaluated", 0), brand,
|
||||
)
|
||||
|
||||
# Step 3: Generate final catalog
|
||||
catalog = {
|
||||
'brand': brand,
|
||||
@@ -918,7 +1061,7 @@ class ProductCatalogEngine:
|
||||
'products': enhanced_products
|
||||
}
|
||||
|
||||
logger.info(f"🎉 Catalog generation complete!")
|
||||
logger.info("🎉 Catalog generation complete!")
|
||||
logger.info(f"📊 Products: {catalog['total_products']}")
|
||||
logger.info(f"🖼️ Total images: {catalog['total_images']}")
|
||||
|
||||
@@ -940,11 +1083,17 @@ class ProductCatalogEngine:
|
||||
if 'image_id' not in p:
|
||||
id_source = p.get('product_name') or p.get('title', 'unknown_product')
|
||||
p['image_id'] = s3_service.generate_image_id(id_source)
|
||||
upsert_brand_products(brand, enhanced_products, cleanup=True)
|
||||
logger.info("🧠 Stored embeddings to pgvector")
|
||||
stored = upsert_brand_products(brand, enhanced_products, cleanup=True)
|
||||
logger.info("🧠 Stored %s product(s) with embeddings to pgvector", stored)
|
||||
except Exception as e:
|
||||
logger.warning(f"Vector storage skipped/failed: {e}")
|
||||
|
||||
# Stage 4 is where the catalog becomes readable by the app - every
|
||||
# API read goes to pgvector, not to the dict returned here. So a
|
||||
# failure at this stage means the run produced nothing the user can
|
||||
# see, and it has to travel back to the job status rather than being
|
||||
# logged and forgotten.
|
||||
logger.error("❌ Stage 4 (pgvector storage) failed for '%s': %s", brand, e)
|
||||
catalog['storage_error'] = str(e)
|
||||
|
||||
return catalog
|
||||
|
||||
def save_catalog(self, catalog: Dict[str, Any], filename: str = None) -> str:
|
||||
|
||||
@@ -8,6 +8,18 @@ product discovery (Ollama), per-product image search + S3 upload,
|
||||
pricing/description enrichment, embedding generation, and the pgvector
|
||||
upsert. This module exists only to give that pipeline one clear, reusable
|
||||
entry point and a consistent result shape for callers.
|
||||
|
||||
NOT THE ELEVEN-STAGE PIPELINE, and prefer `app/services/brand_discovery.py`
|
||||
for new work. "Full pipeline" above means this module's own sequence, not the
|
||||
eleven stages in `app/core/store_catalog_pipeline.py`: there is no title
|
||||
validation, no pack-size explosion, no SKU service, no barcode, no HSN/GST and
|
||||
no validation gate here, and `image_id` carries a random uuid4 suffix, so a
|
||||
second run for the same brand cannot match the first and duplicates it.
|
||||
|
||||
This path is unchanged and still works. It also rewrites the brand's seed
|
||||
catalog via `brand_sync.export_brand_to_seed_file` (below), which the discovery
|
||||
path deliberately does not - so the two are not drop-in replacements for each
|
||||
other. See docs/BRAND_DISCOVERY.md.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -32,12 +44,33 @@ async def ingest_brand(brand: str, max_products: int = 50) -> Dict[str, Any]:
|
||||
catalog = await catalog_engine.generate_catalog(brand=brand, max_products=max_products)
|
||||
duration = time.time() - start
|
||||
|
||||
# Mirror the freshly ingested rows into data/seed_catalogs/ so a brand that
|
||||
# arrived through the pipeline is reproducible from disk like the bundled
|
||||
# ones. Reading back from pgvector rather than reusing catalog["products"]
|
||||
# keeps the file honest about what was actually stored.
|
||||
#
|
||||
# Placed here rather than in the API router because this function is also
|
||||
# the CLI's entry point (cli/ingest_brand.py), and a failure to write the
|
||||
# file must not turn a successful ingest into a failed job.
|
||||
storage_error = catalog.get("storage_error")
|
||||
|
||||
# Nothing reached the database, so there is nothing to mirror out of it -
|
||||
# and running the export anyway would either write an empty file or leave a
|
||||
# stale one looking current.
|
||||
if not storage_error:
|
||||
try:
|
||||
from app.services.brand_sync import export_brand_to_seed_file
|
||||
export_brand_to_seed_file(brand)
|
||||
except Exception: # noqa: BLE001 - DB rows are already committed
|
||||
logger.warning("Seed-catalog export failed for %s (DB rows intact)", brand, exc_info=True)
|
||||
|
||||
summary = {
|
||||
"brand": brand,
|
||||
"total_products": catalog.get("total_products", 0),
|
||||
"total_images": catalog.get("total_images", 0),
|
||||
"duration_seconds": round(duration, 2),
|
||||
"engine_info": catalog.get("engine_info", {}),
|
||||
"storage_error": storage_error,
|
||||
}
|
||||
logger.info("Finished ingestion for brand=%s in %.2fs: %s products",
|
||||
brand, duration, summary["total_products"])
|
||||
|
||||
1321
app/core/store_catalog_pipeline.py
Normal file
1321
app/core/store_catalog_pipeline.py
Normal file
File diff suppressed because it is too large
Load Diff
@@ -29,15 +29,19 @@ import logging
|
||||
import secrets
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Dict, List, Optional
|
||||
from typing import Dict, List, Optional, Tuple
|
||||
|
||||
import jwt
|
||||
|
||||
from app.infrastructure.settings import (
|
||||
API_KEYS,
|
||||
AUTH_ADMIN_PASSWORD_HASH,
|
||||
AUTH_ADMIN_USERNAME,
|
||||
AUTH_ALLOW_ANY_LOGIN,
|
||||
AUTH_ENABLED,
|
||||
AUTH_SECRET_KEY,
|
||||
AUTH_TOKEN_TTL_MINUTES,
|
||||
config_source,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -68,6 +72,30 @@ ROLE_PERMISSIONS: Dict[str, List[str]] = {
|
||||
"view_nutrition_insights",
|
||||
"optimize_profits",
|
||||
],
|
||||
# An outside API client that may send spreadsheets for catalog ingestion and
|
||||
# do NOTHING else. One permission, deliberately.
|
||||
#
|
||||
# This role exists because API keys carry no per-key scoping:
|
||||
# principal_for_api_key() derives permissions entirely from the role, so
|
||||
# "upload-only" can only be expressed as a role. Reusing `user` would have
|
||||
# been less code and would also have handed an outside contributor
|
||||
# add_product, upload_batch_products and upload_store_inventory - real
|
||||
# write access to the catalog - to solve a problem that needed one verb.
|
||||
#
|
||||
# WHAT A LEAKED UPLOADER KEY COSTS. Real CPU: this permission starts the
|
||||
# 11-stage pipeline, which is the point of the endpoint. The bound is not
|
||||
# "this role cannot work" but "all ingestion, from every source, shares one
|
||||
# worker" - batch_worker runs a single batch at a time behind a queue of
|
||||
# BATCH_QUEUE_MAX, past which POST /api/uploads/catalog answers 429. So a
|
||||
# key can occupy the ingestion worker; it cannot multiply it, and it cannot
|
||||
# touch the request path the healthcheck reads.
|
||||
#
|
||||
# What it still cannot do: read the catalog, read another caller's
|
||||
# submissions (every read on that router is filtered by submitted_by), or
|
||||
# cancel, resume or delete anything.
|
||||
"uploader": [
|
||||
"upload_catalog",
|
||||
],
|
||||
}
|
||||
|
||||
VALID_ROLES = frozenset(ROLE_PERMISSIONS)
|
||||
@@ -117,6 +145,154 @@ def hash_password(password: str, *, iterations: int = _PBKDF2_ITERATIONS) -> str
|
||||
)
|
||||
|
||||
|
||||
def _parse_encoded_hash(encoded: str) -> Optional[Tuple[bytes, bytes, int]]:
|
||||
"""
|
||||
Split an encoded digest into ``(salt, digest, iterations)``, or None if it
|
||||
is not one.
|
||||
|
||||
One parser, three callers. `verify_password` needs the parts, while
|
||||
`hash_is_wellformed` and `describe_password_hash` need only the verdict -
|
||||
and a login failing because the *configured* hash is corrupt is a different
|
||||
incident from a wrong password, so the two must agree on what "corrupt"
|
||||
means. Two copies of this parse would eventually disagree.
|
||||
|
||||
Values arrive here straight from the environment, so a hash pasted into a
|
||||
deployment platform's form field as "pbkdf2_sha256$..." is unwrapped rather
|
||||
than rejected: the surrounding quotes are almost never intended as part of
|
||||
the secret, and the failure they cause otherwise is a silent 401.
|
||||
"""
|
||||
if not encoded:
|
||||
return None
|
||||
encoded = encoded.strip().strip("'\"")
|
||||
try:
|
||||
prefix, raw_iterations, raw_salt, raw_digest = encoded.split("$")
|
||||
if prefix != _PBKDF2_PREFIX:
|
||||
return None
|
||||
# validate=True so junk is rejected rather than silently discarded:
|
||||
# b64decode's default drops non-alphabet characters, which would let a
|
||||
# subtly corrupted hash decode to the wrong bytes and fail as a "wrong
|
||||
# password" instead of as the configuration error it is.
|
||||
# binascii.Error subclasses ValueError, so it is caught below.
|
||||
salt = base64.b64decode(raw_salt, validate=True)
|
||||
digest = base64.b64decode(raw_digest, validate=True)
|
||||
iterations = int(raw_iterations)
|
||||
except (ValueError, TypeError):
|
||||
return None
|
||||
# A structurally valid string that decodes to nothing is still unusable,
|
||||
# and PBKDF2 rejects a non-positive iteration count by raising.
|
||||
if not salt or not digest or iterations < 1:
|
||||
return None
|
||||
return salt, digest, iterations
|
||||
|
||||
|
||||
def hash_is_wellformed(encoded: str) -> bool:
|
||||
"""Whether a configured digest can be checked against at all.
|
||||
|
||||
Distinct from "does the password match": this asks whether the credential
|
||||
*store* is usable, which is a deployment fault rather than a sign-in one.
|
||||
"""
|
||||
return _parse_encoded_hash(encoded) is not None
|
||||
|
||||
|
||||
def password_hash_fingerprint(encoded: str) -> str:
|
||||
"""
|
||||
A short, non-reversible identifier for a configured digest.
|
||||
|
||||
Safe to log and to publish: it is a truncated SHA-256 of the *encoded
|
||||
digest*, and that digest already embeds a 16-byte random salt, so this says
|
||||
which credential is loaded without saying anything about the password
|
||||
behind it. It exists so a running deployment can be compared against the
|
||||
config it was supposed to have been built from - the failure this project
|
||||
actually hit - without moving a secret in order to do the comparison.
|
||||
"""
|
||||
if not encoded:
|
||||
return ""
|
||||
return hashlib.sha256(encoded.strip().strip("'\"").encode("utf-8")).hexdigest()[:12]
|
||||
|
||||
|
||||
def api_key_fingerprint(name: str, secret: str) -> str:
|
||||
"""
|
||||
A short, non-reversible identifier for a configured API key.
|
||||
|
||||
Same purpose as password_hash_fingerprint - say *which* credential is loaded
|
||||
without moving the credential - but the safety argument is different and
|
||||
worth stating. That function digests an encoded hash which already embeds a
|
||||
16-byte random salt. An API key has no salt, so the name is mixed in here to
|
||||
keep two consumers that were mistakenly issued the same secret from
|
||||
fingerprinting identically, and settings._parse_api_keys enforces a minimum
|
||||
secret length so the digest cannot be walked back with a wordlist.
|
||||
"""
|
||||
if not secret:
|
||||
return ""
|
||||
cleaned = secret.strip().strip("'\"")
|
||||
material = f"{name}:{cleaned}"
|
||||
return hashlib.sha256(material.encode("utf-8")).hexdigest()[:12]
|
||||
|
||||
|
||||
def describe_api_keys() -> List[Dict[str, object]]:
|
||||
"""Every configured key as {name, role, fingerprint}, sorted by name.
|
||||
|
||||
Sorted so two deployments' /api/health output can be diffed line for line;
|
||||
API_KEYS is keyed by secret, whose iteration order says nothing useful.
|
||||
"""
|
||||
return sorted(
|
||||
(
|
||||
{"name": name, "role": role, "fingerprint": api_key_fingerprint(name, secret)}
|
||||
for secret, (name, role) in API_KEYS.items()
|
||||
),
|
||||
key=lambda entry: entry["name"],
|
||||
)
|
||||
|
||||
|
||||
def describe_password_hash(encoded: str) -> Dict[str, object]:
|
||||
"""A loggable/publishable summary of a configured digest. Never its bytes."""
|
||||
parsed = _parse_encoded_hash(encoded)
|
||||
return {
|
||||
"valid": parsed is not None,
|
||||
"algorithm": _PBKDF2_PREFIX if parsed is not None else None,
|
||||
"iterations": parsed[2] if parsed is not None else None,
|
||||
"fingerprint": password_hash_fingerprint(encoded),
|
||||
}
|
||||
|
||||
|
||||
def auth_config_summary() -> Dict[str, object]:
|
||||
"""
|
||||
The effective authentication configuration, in a form safe to both log and
|
||||
publish. Contains no password and no hash - only the fingerprint.
|
||||
|
||||
This is deliberately one function with two callers (the startup log in
|
||||
app/main.py and GET /api/health), because its entire purpose is letting two
|
||||
*deployments* be compared, and that only works if both report the same
|
||||
fields computed the same way.
|
||||
|
||||
`*_source` is the field that earns this its keep. A value of "process-env"
|
||||
means the container's own environment supplied it and the .env file baked
|
||||
into the image was ignored - which is invisible from anywhere else, and is
|
||||
precisely how a corrected credential can keep failing after a redeploy.
|
||||
|
||||
The same argument is why the API keys are summarised here. backend/Dockerfile
|
||||
copies .env.production in at BUILD time, so a key added to that file and then
|
||||
merely restarted is not present in the running process - and from outside,
|
||||
an undeployed key is indistinguishable from a wrong one, because both are
|
||||
just a 401. Publishing the names and fingerprints answers "is my key on this
|
||||
deployment?" without anyone having to send the secret to find out.
|
||||
"""
|
||||
described = describe_password_hash(AUTH_ADMIN_PASSWORD_HASH)
|
||||
return {
|
||||
"enabled": AUTH_ENABLED,
|
||||
"allow_any_login": AUTH_ALLOW_ANY_LOGIN,
|
||||
"admin_username": AUTH_ADMIN_USERNAME,
|
||||
"password_hash_valid": bool(described["valid"]),
|
||||
"password_hash_iterations": described["iterations"],
|
||||
"password_hash_fingerprint": described["fingerprint"],
|
||||
"admin_username_source": config_source("AUTH_ADMIN_USERNAME"),
|
||||
"password_hash_source": config_source("AUTH_ADMIN_PASSWORD_HASH"),
|
||||
"api_keys_count": len(API_KEYS),
|
||||
"api_keys": describe_api_keys(),
|
||||
"api_keys_source": config_source("API_KEYS"),
|
||||
}
|
||||
|
||||
|
||||
def verify_password(password: str, encoded: str) -> bool:
|
||||
"""
|
||||
Check a password against an encoded digest.
|
||||
@@ -127,20 +303,15 @@ def verify_password(password: str, encoded: str) -> bool:
|
||||
"""
|
||||
if not encoded:
|
||||
return False
|
||||
try:
|
||||
prefix, raw_iterations, raw_salt, raw_digest = encoded.split("$")
|
||||
if prefix != _PBKDF2_PREFIX:
|
||||
return False
|
||||
expected = base64.b64decode(raw_salt), base64.b64decode(raw_digest)
|
||||
salt, digest = expected
|
||||
iterations = int(raw_iterations)
|
||||
except (ValueError, TypeError):
|
||||
parsed = _parse_encoded_hash(encoded)
|
||||
if parsed is None:
|
||||
logger.error(
|
||||
"A configured password hash is malformed and cannot be used. Regenerate "
|
||||
"it with: python scripts/make_auth_secrets.py"
|
||||
)
|
||||
return False
|
||||
|
||||
salt, digest, iterations = parsed
|
||||
candidate = hashlib.pbkdf2_hmac("sha256", password.encode("utf-8"), salt, iterations)
|
||||
return hmac.compare_digest(candidate, digest)
|
||||
|
||||
|
||||
@@ -27,6 +27,17 @@ from __future__ import annotations
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
# Snapshotted BEFORE load_dotenv, and that ordering is the entire point.
|
||||
# load_dotenv() is called without override=True, so a variable already in the
|
||||
# process environment silently beats the .env file and keeps beating it no
|
||||
# matter how many times the file is corrected. That is not hypothetical here:
|
||||
# the deployment platform injects its Environment tab into the container, so a
|
||||
# stale value left in that tab overrides the credentials baked into the image
|
||||
# (backend/Dockerfile copies .env.production to /app/.env) and the only symptom
|
||||
# is a 401 that nothing explains. Comparing a name against this set answers
|
||||
# "which of the two won?" - see config_source() below.
|
||||
_PREEXISTING_ENV = frozenset(os.environ)
|
||||
|
||||
try:
|
||||
from dotenv import load_dotenv
|
||||
|
||||
@@ -39,6 +50,45 @@ except ImportError:
|
||||
pass
|
||||
|
||||
|
||||
# Names whose raw value arrived wrapped in quotes or padded with whitespace.
|
||||
# Recorded rather than merely fixed: stripping keeps the login working, but the
|
||||
# only place the original shape is still visible is right here, before the value
|
||||
# is normalised. A quoted hash is the signature of a value pasted into a web
|
||||
# form, so surfacing it at startup is what stops the next person rediscovering
|
||||
# it from a 401. See DB_PASSWORD in .env.production for the counter-case where
|
||||
# the quotes ARE part of the secret - which is why this warns, and does not fail.
|
||||
_ENV_NEEDED_CLEANUP = set()
|
||||
|
||||
|
||||
def _clean(name: str, raw: str) -> str:
|
||||
"""Strip surrounding quotes/whitespace off an env value, remembering if it mattered."""
|
||||
cleaned = raw.strip().strip("'\"")
|
||||
if cleaned != raw:
|
||||
_ENV_NEEDED_CLEANUP.add(name)
|
||||
return cleaned
|
||||
|
||||
|
||||
def cleaned_env_names() -> list:
|
||||
"""Which settings needed quote/whitespace stripping. Reported at startup."""
|
||||
return sorted(_ENV_NEEDED_CLEANUP)
|
||||
|
||||
|
||||
def config_source(name: str) -> str:
|
||||
"""
|
||||
Where a setting's value actually came from: the process environment, the
|
||||
.env file, or this module's own default.
|
||||
|
||||
Reported at startup for the AUTH_* values (see app/main.py) so that an
|
||||
override arriving from outside the image is visible in the logs instead of
|
||||
being inferred from a failing login.
|
||||
"""
|
||||
if name in _PREEXISTING_ENV:
|
||||
return "process-env"
|
||||
if name in os.environ:
|
||||
return "env-file"
|
||||
return "default"
|
||||
|
||||
|
||||
def _bool(name: str, default: str) -> bool:
|
||||
return os.getenv(name, default).strip().lower() in {"1", "true", "yes"}
|
||||
|
||||
@@ -84,6 +134,103 @@ MODEL_ARTIFACTS_DIR = _dir(
|
||||
"MODEL_ARTIFACTS_DIR", _BACKEND_ROOT / "app" / "intelligence" / "artifacts"
|
||||
)
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Batch catalog ingestion (multi-file upload -> the 11-stage pipeline)
|
||||
# ---------------------------------------------------------------------------
|
||||
# Staged uploads live under DATA_DIR because that path is already a declared
|
||||
# volume (backend/Dockerfile). A batch that survives a container restart is the
|
||||
# whole point of writing the files down instead of holding them in the worker
|
||||
# thread the way the single-file path does.
|
||||
#
|
||||
# Every ceiling below is enforced in the application, not at the proxy. In
|
||||
# production the browser calls mcp.nearle.ai.in directly, so neither nginx's
|
||||
# client_max_body_size nor Caddy's request_body cap is in front of these
|
||||
# endpoints - whatever Traefik defaults to is, and it is not ours to rely on.
|
||||
BATCH_UPLOAD_DIR = _dir("BATCH_UPLOAD_DIR", DATA_DIR / "batch_uploads")
|
||||
|
||||
# Per-file limits stay at the single-upload values, 10MB / 2000 rows. They are
|
||||
# NOT settings and are not read from here: each router declares its own
|
||||
# MAX_UPLOAD_BYTES / MAX_UPLOAD_ROWS pair with the same two numbers - see
|
||||
# routers/batch_catalog.py, routers/uploads.py and routers/user_products.py.
|
||||
# Change one and you have changed one. The ceilings below bound the BATCH on
|
||||
# top of whichever per-file pair applied.
|
||||
BATCH_MAX_FILES = int(os.getenv("BATCH_MAX_FILES", "20"))
|
||||
BATCH_MAX_TOTAL_BYTES = int(os.getenv("BATCH_MAX_TOTAL_BYTES", str(50 * 1024 * 1024)))
|
||||
BATCH_MAX_TOTAL_ROWS = int(os.getenv("BATCH_MAX_TOTAL_ROWS", "20000"))
|
||||
|
||||
# Batches waiting behind the one running. Past this the endpoint returns 429
|
||||
# rather than accepting work it has no intention of starting soon.
|
||||
BATCH_QUEUE_MAX = int(os.getenv("BATCH_QUEUE_MAX", "4"))
|
||||
|
||||
# Staged files are deleted this many days after the batch was created. Without
|
||||
# this the upload directory only grows, on a host whose disk is the scarcest
|
||||
# resource it has.
|
||||
BATCH_RETENTION_DAYS = int(os.getenv("BATCH_RETENTION_DAYS", "7"))
|
||||
|
||||
# --- Unattended ingestion --------------------------------------------------
|
||||
# Whether POST /api/uploads/catalog runs the pipeline on arrival, or parks the
|
||||
# files in the admin review inbox for someone to start by hand.
|
||||
#
|
||||
# READ THIS BEFORE CHANGING IT. That endpoint takes NO credential - it was
|
||||
# opened deliberately so colleagues could send spreadsheets without one being
|
||||
# issued to them. With autorun on, "anyone who can reach this host" and "anyone
|
||||
# who can write to the live catalogue" become the same set of people, and an
|
||||
# ingest is an upsert with no undo. That trade was made knowingly: the ask was
|
||||
# for uploads to run without manual intervention, and a review queue that needs
|
||||
# an admin to press a button is not that.
|
||||
#
|
||||
# What still bounds it: the per-request ceilings above (20 files / 50MB / 20k
|
||||
# rows), and BATCH_QUEUE_MAX behind a single worker thread - so a sender can
|
||||
# occupy the ingestion worker but cannot multiply it. Those cap throughput, not
|
||||
# who. If that stops being an acceptable trade, set this to false and the review
|
||||
# inbox comes back with no code change; everything it needs is still here.
|
||||
UPLOAD_AUTORUN = _bool("UPLOAD_AUTORUN", "true")
|
||||
|
||||
# How an auto-started run behaves. Not accepted from the request: the sender is
|
||||
# anonymous, and letting an anonymous caller turn on the expensive stages is the
|
||||
# one thing the open endpoint must not allow.
|
||||
#
|
||||
# Images ON, because a product landing without one is the failure this endpoint
|
||||
# exists to avoid - stage 6 is the slowest stage and reaches the network, but
|
||||
# only one batch runs at a time so nothing else is competing with it.
|
||||
#
|
||||
# LLM ON. `use_llm` gates only description generation in stage_2_row_intake, and
|
||||
# in production it is currently a no-op: USE_OLLAMA is false there, so
|
||||
# ollama_service._ensure_client() returns on its first line without a request
|
||||
# and the row simply keeps its blank description. It is set true so the pipeline
|
||||
# is already configured correctly for the day an Ollama server exists.
|
||||
#
|
||||
# This default USED to be false, on the grounds that turning it on "costs a
|
||||
# connection timeout per row". That was true, and it was about the OTHER branch
|
||||
# of _ensure_client - USE_OLLAMA=true with nothing listening, which is any
|
||||
# developer machine that has not run `ollama serve`. It is answered now by the
|
||||
# TTL cache on that probe rather than by leaving the feature off: one probe per
|
||||
# batch instead of one per row. Do not remove that cache and this default
|
||||
# together without re-reading why both exist.
|
||||
UPLOAD_AUTORUN_FETCH_IMAGES = _bool("UPLOAD_AUTORUN_FETCH_IMAGES", "true")
|
||||
UPLOAD_AUTORUN_USE_LLM = _bool("UPLOAD_AUTORUN_USE_LLM", "true")
|
||||
|
||||
# --- Review inbox ----------------------------------------------------------
|
||||
# The bound that applies only when UPLOAD_AUTORUN is false. Files then wait in
|
||||
# the admin review inbox rather than being queued, which removes BATCH_QUEUE_MAX
|
||||
# as the bound on that endpoint and leaves the volume as the only thing an
|
||||
# anonymous sender can exhaust. These are that bound; past either, the endpoint
|
||||
# answers 429 and stages nothing.
|
||||
#
|
||||
# Both count only files still AWAITING review. Starting or dismissing a drop
|
||||
# releases its share immediately, and BATCH_RETENTION_DAYS reclaims whatever
|
||||
# nobody ever looks at.
|
||||
INBOX_MAX_PENDING_FILES = int(os.getenv("INBOX_MAX_PENDING_FILES", "200"))
|
||||
INBOX_MAX_PENDING_BYTES = int(
|
||||
os.getenv("INBOX_MAX_PENDING_BYTES", str(200 * 1024 * 1024))
|
||||
)
|
||||
|
||||
# Deliberately false. A batch interrupted by a restart is marked "interrupted"
|
||||
# and waits for someone to press Resume. Auto-resuming would mean a container
|
||||
# stuck in a restart loop re-runs the heaviest work in the app on every boot,
|
||||
# which is precisely how a slow start turns into an unrecoverable spiral.
|
||||
BATCH_AUTO_RESUME = _bool("BATCH_AUTO_RESUME", "false")
|
||||
|
||||
# Pristine copies of the bundled seed catalogs and pre-trained models, placed
|
||||
# here by the Dockerfile at a path that is never itself mounted over.
|
||||
#
|
||||
@@ -127,11 +274,98 @@ DB_PORT = os.getenv("DB_PORT", "5432")
|
||||
DB_NAME = os.getenv("DB_NAME", "pgvector")
|
||||
DB_USER = os.getenv("DB_USER", "postgres")
|
||||
DB_PASSWORD = _require("DB_PASSWORD", feature_flag="USE_PGVECTOR") if USE_PGVECTOR else os.getenv("DB_PASSWORD", "")
|
||||
# How long to wait for the TCP connect before giving up. Matters more than it
|
||||
# looks: a host that DROPS packets (a firewall, a typo'd DB_HOST) otherwise
|
||||
# blocks until the OS timeout - about 130 seconds on Linux - and every request
|
||||
# that touches the database inherits that wait, including /api/health. A short
|
||||
# ceiling turns "the database is unreachable" into a fast, honest error instead
|
||||
# of a hung worker and a container the platform decides is unhealthy.
|
||||
DB_CONNECT_TIMEOUT_SECONDS = int(os.getenv("DB_CONNECT_TIMEOUT_SECONDS", "5"))
|
||||
|
||||
DATABASE_URL = os.getenv(
|
||||
"DATABASE_URL",
|
||||
f"postgresql://{DB_USER}:{DB_PASSWORD}@{DB_HOST}:{DB_PORT}/{DB_NAME}",
|
||||
)
|
||||
|
||||
# How often to re-run the brand-table <-> seed-catalog reconcile after boot.
|
||||
#
|
||||
# The startup run alone only catches what existed at boot. A brand table created
|
||||
# directly in the database while the server is up - by hand, by a script, or by
|
||||
# another machine sharing this database - is not mirrored into the seed catalogs
|
||||
# until the next restart. This interval is what closes that window.
|
||||
#
|
||||
# reconcile_brand_catalogs() is idempotent and non-destructive, so a sweep that
|
||||
# finds nothing to do costs one COUNT(*) per brand table and writes nothing.
|
||||
# 0 disables the loop, leaving the startup run and POST /api/system/brand-sync.
|
||||
BRAND_SYNC_INTERVAL_SECONDS = int(os.getenv("BRAND_SYNC_INTERVAL_SECONDS", "300"))
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Active brands (development working set)
|
||||
# ---------------------------------------------------------------------------
|
||||
# Comma-separated brand names that the application and every pipeline operate
|
||||
# on. BLANK OR UNSET MEANS EVERY BRAND IS ACTIVE - that is the backwards
|
||||
# compatible default and the way to switch this feature off again.
|
||||
#
|
||||
# Nothing is deleted when this is set: the other brand_* tables and their
|
||||
# embeddings stay in the database untouched, they simply stop being discovered.
|
||||
# Going from 3 brands to 5, 10 or all of them is an edit to this one line.
|
||||
#
|
||||
# Names are resolved through resolve_parent_brand + _sanitize_name, the same
|
||||
# two steps that pick a product's storage table, so "Tata" here activates
|
||||
# brand_hindustan_unilever exactly as ingesting Tata products would.
|
||||
# See app/services/active_brands.py.
|
||||
ACTIVE_BRANDS = os.getenv("ACTIVE_BRANDS", "")
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Brand discovery (brand name -> the 11-stage pipeline)
|
||||
# ---------------------------------------------------------------------------
|
||||
# Discovery turns a brand NAME into rows the ordinary catalog pipeline ingests.
|
||||
# See app/services/brand_discovery.py. Every value below has a working default,
|
||||
# so the feature needs no configuration to run.
|
||||
#
|
||||
# Open Food Facts is the primary source and the language model is the
|
||||
# supplement, not the reverse: OFF returns real products carrying a real GTIN,
|
||||
# while the default OLLAMA_MODEL_NAME (qwen2.5:1.5b) invents plausible ones that
|
||||
# nothing downstream can catch. Turning BRAND_DISCOVERY_USE_OFF off leaves the
|
||||
# result resting on the model alone.
|
||||
BRAND_DISCOVERY_USE_OFF = _bool("BRAND_DISCOVERY_USE_OFF", "true")
|
||||
BRAND_DISCOVERY_USE_LLM = _bool("BRAND_DISCOVERY_USE_LLM", "true")
|
||||
|
||||
# Products per discovery run. One CSV row per product; pack-size explosion
|
||||
# happens later in stage 4, so this is well inside the 2000-row per-file cap.
|
||||
BRAND_DISCOVERY_MAX_PRODUCTS = int(os.getenv("BRAND_DISCOVERY_MAX_PRODUCTS", "200"))
|
||||
|
||||
# Pack sizes kept per product when only the language model offers any. Stage 6
|
||||
# runs an image search per exploded row, so this multiplies the slowest part of
|
||||
# the run; 3 keeps a large brand inside a sane wall-clock.
|
||||
BRAND_DISCOVERY_MAX_SIZES = int(os.getenv("BRAND_DISCOVERY_MAX_SIZES", "3"))
|
||||
|
||||
# Wall-clock ceiling on the LLM half of a run, checked between prompts. Open
|
||||
# Food Facts runs first and is never subject to it, so a run that hits this
|
||||
# still returns the evidence-backed products.
|
||||
BRAND_DISCOVERY_DEADLINE_SECONDS = float(
|
||||
os.getenv("BRAND_DISCOVERY_DEADLINE_SECONDS", "300")
|
||||
)
|
||||
|
||||
# Web & retail listing discovery (app/services/web_discovery/) - the optional
|
||||
# Brand Discovery source for brands Open Food Facts does not carry. It reads
|
||||
# ONLY search-engine results (title, URL, snippet) that point at Indian
|
||||
# retailers' product pages; it never fetches a retailer page and never asks a
|
||||
# language model. A product is kept only when a real listing names the brand
|
||||
# and states a pack size. Off in a request unless the admin ticks it.
|
||||
WEB_DISCOVERY_ENABLED = _bool("WEB_DISCOVERY_ENABLED", "true")
|
||||
# Searches per brand job. A parent brand fans out to its sub-brands (Reckitt ->
|
||||
# Dettol, Harpic, Lizol, ...) times the retailers, so this is the cost ceiling.
|
||||
WEB_DISCOVERY_MAX_QUERIES = int(os.getenv("WEB_DISCOVERY_MAX_QUERIES", "80"))
|
||||
# Seconds between two LIVE searches (cached ones are free). DuckDuckGo throttles
|
||||
# a burst; a throttled search is "could not ask", never "nothing found".
|
||||
WEB_DISCOVERY_PAUSE_SECONDS = float(os.getenv("WEB_DISCOVERY_PAUSE_SECONDS", "2.0"))
|
||||
WEB_DISCOVERY_RESULTS_PER_QUERY = int(os.getenv("WEB_DISCOVERY_RESULTS_PER_QUERY", "20"))
|
||||
# How long a search answer is reused. An EMPTY answer is kept one day only:
|
||||
# search reach is unstable, and a flaky empty page must not become a week-long fact.
|
||||
WEB_DISCOVERY_CACHE_DAYS = float(os.getenv("WEB_DISCOVERY_CACHE_DAYS", "7"))
|
||||
WEB_DISCOVERY_DIR = _dir("WEB_DISCOVERY_DIR", DATA_DIR / "cache" / "web_discovery")
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# S3 / DigitalOcean Spaces (product image storage) - optional
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -165,6 +399,258 @@ USE_PLAYWRIGHT_FALLBACK = _bool("USE_PLAYWRIGHT_FALLBACK", "true")
|
||||
# out 1x1 tracking pixels / broken placeholder images)
|
||||
MIN_IMAGE_BYTES = int(os.getenv("MIN_IMAGE_BYTES", "3000"))
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# img_vector - a MobileNetV3-Small image embedding (1024 floats, L2-normalised)
|
||||
# of each product's primary image, stored on its brand table
|
||||
# (app/services/image_vector.py owns the column, image_embedder.py the model)
|
||||
# ---------------------------------------------------------------------------
|
||||
# Computed off the request path by one bounded worker thread after every
|
||||
# catalog write, and by scripts/backfill_image_vectors.py for existing rows.
|
||||
# Default on: the work is one small download and one ~30ms inference per row
|
||||
# just written, never on a request. tests/conftest.py pins it OFF so the
|
||||
# suite never dials the database named in a developer's .env.
|
||||
ENABLE_IMAGE_VECTORS = _bool("ENABLE_IMAGE_VECTORS", "true")
|
||||
# A download past this many bytes is abandoned - a wrong URL to a video must
|
||||
# not fill the container.
|
||||
IMAGE_VECTOR_MAX_BYTES = int(os.getenv("IMAGE_VECTOR_MAX_BYTES", str(8 * 1024 * 1024)))
|
||||
# Refused before decoding when the header claims more pixels than this
|
||||
# (decompression-bomb guard; 25MP is well past any product photo, and the
|
||||
# embedder decodes at full resolution - ~75MB of RGB at this cap).
|
||||
IMAGE_VECTOR_MAX_PIXELS = int(os.getenv("IMAGE_VECTOR_MAX_PIXELS", "25000000"))
|
||||
IMAGE_VECTOR_TIMEOUT_SECONDS = float(os.getenv("IMAGE_VECTOR_TIMEOUT_SECONDS", "15"))
|
||||
# Minimum gap between two requests to the same image host.
|
||||
IMAGE_VECTOR_HOST_PAUSE_SECONDS = float(os.getenv("IMAGE_VECTOR_HOST_PAUSE_SECONDS", "0.5"))
|
||||
# Writes waiting for the worker; beyond this a write's rows are left for the
|
||||
# backfill script rather than queued.
|
||||
IMAGE_VECTOR_QUEUE_MAX = int(os.getenv("IMAGE_VECTOR_QUEUE_MAX", "64"))
|
||||
# The TFLite embedder (mobilenet_v3_small_embedder.tflite, input [1,224,224,3]
|
||||
# float32, output [1,1024]). Lives under app/, NOT data/: /app/data is a named
|
||||
# volume on every deployment (see BUNDLED_ASSETS_DIR), and a file added to
|
||||
# the image under a mounted path is invisible on any volume that already
|
||||
# exists. app/ is copied into the image and never mounted.
|
||||
IMAGE_EMBED_MODEL_PATH = _dir(
|
||||
"IMAGE_EMBED_MODEL_PATH",
|
||||
_BACKEND_ROOT / "app" / "services" / "models" / "mobilenet" / "mobilenet_v3_small_embedder.tflite",
|
||||
)
|
||||
# Intra-op threads for one inference. 2 on the prod host; inference itself is
|
||||
# serialised by a lock (a TFLite interpreter is not thread-safe).
|
||||
IMAGE_EMBED_NUM_THREADS = int(os.getenv("IMAGE_EMBED_NUM_THREADS", "2"))
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Image search - POST /api/search/image-vector and /api/search/image
|
||||
# (app/services/image_search.py). Public, read-only.
|
||||
# ---------------------------------------------------------------------------
|
||||
IMAGE_SEARCH_DEFAULT_TOP_K = int(os.getenv("IMAGE_SEARCH_DEFAULT_TOP_K", "10"))
|
||||
IMAGE_SEARCH_MAX_TOP_K = int(os.getenv("IMAGE_SEARCH_MAX_TOP_K", "50"))
|
||||
# 0.0 on purpose. A simulated phone photo of Marie Gold against its catalog
|
||||
# render scored 0.63; the app team's "0.7 means the same product" is a
|
||||
# client-side rule of thumb for phone-vs-phone, so the server does not
|
||||
# impose it - callers pass min_score when they want a floor.
|
||||
IMAGE_SEARCH_DEFAULT_MIN_SCORE = float(os.getenv("IMAGE_SEARCH_DEFAULT_MIN_SCORE", "0.0"))
|
||||
# Candidates fetched PER brand table before re-ranking (pack sizes of one
|
||||
# product share an image and tie, so more than top_k must come back), and
|
||||
# the floor for hnsw.ef_search on that query so the index does not drop them.
|
||||
IMAGE_SEARCH_MAX_FETCH_K = int(os.getenv("IMAGE_SEARCH_MAX_FETCH_K", "100"))
|
||||
# How far the best image match must lead the best DIFFERENT photo for the
|
||||
# answer to count as confirmed (`match_confidence`). A phone photo of a card on
|
||||
# a screen measured 0.507 against its own product and 0.419 against a
|
||||
# stranger; with glare that gap closes, and a flipped order is a wrong
|
||||
# product shown with confidence. Pack sizes sharing one photo tie exactly and
|
||||
# are never each other's competitor. Unmeasured start: tune with
|
||||
# scripts/eval_identify.py.
|
||||
IMAGE_SEARCH_MIN_MARGIN = float(os.getenv("IMAGE_SEARCH_MIN_MARGIN", "0.05"))
|
||||
# Diagnosis: when set, every image-search request (vector, text, brand and the
|
||||
# top matches) is written here as one JSON file, for
|
||||
# scripts/replay_image_query.py. Blank = off. At most IMAGE_SEARCH_CAPTURE_MAX
|
||||
# files are kept; the oldest go first. The routes are public, so leave it off
|
||||
# except while chasing a report.
|
||||
IMAGE_SEARCH_CAPTURE_DIR = os.getenv("IMAGE_SEARCH_CAPTURE_DIR", "").strip()
|
||||
IMAGE_SEARCH_CAPTURE_MAX = int(os.getenv("IMAGE_SEARCH_CAPTURE_MAX", "200"))
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Identify a product from a phone photo - POST /api/search/identify
|
||||
# (app/services/product_identify.py), with server-side OCR
|
||||
# (app/services/ocr_service.py) and the label-text resolver
|
||||
# (app/services/label_match.py). Public, read-only.
|
||||
# ---------------------------------------------------------------------------
|
||||
# A phone photo of a pack against the catalog's render of it scored 0.63 on
|
||||
# img_vector (feasibility test), so the image alone cannot confirm a product.
|
||||
# Below this floor the ladder falls through to the label text: whatever the
|
||||
# client OCR'd, else what the server reads off the photo itself.
|
||||
IMAGE_IDENTIFY_MIN_IMAGE_SCORE = float(os.getenv("IMAGE_IDENTIFY_MIN_IMAGE_SCORE", "0.70"))
|
||||
# The text side is a MiniLM cosine (1 - embedding <=> q) in a different space
|
||||
# from the image score. A nearest-neighbour query always returns SOMETHING, so
|
||||
# a text match counts as found only when the label shares a name/size token
|
||||
# with the row, or the cosine clears this floor.
|
||||
IMAGE_IDENTIFY_MIN_TEXT_SCORE = float(os.getenv("IMAGE_IDENTIFY_MIN_TEXT_SCORE", "0.60"))
|
||||
# Server-side OCR (rapidocr, PP-OCR models on onnxruntime CPU; models ship
|
||||
# inside the wheel, nothing is downloaded). Loaded lazily on the first photo
|
||||
# that needs it, never at boot; a missing wheel means "no server OCR" and
|
||||
# GET /api/health -> ocr says so. tests/conftest.py pins it OFF.
|
||||
ENABLE_SERVER_OCR = _bool("ENABLE_SERVER_OCR", "true")
|
||||
# rapidocr's Global.text_score: recognised lines below this confidence are
|
||||
# dropped before the label is assembled.
|
||||
OCR_MIN_CONFIDENCE = float(os.getenv("OCR_MIN_CONFIDENCE", "0.5"))
|
||||
# The photo is downscaled so its longer side is at most this before
|
||||
# detection. The latency knob: a 12MP capture takes ~3x longer than 1280px
|
||||
# and reads no better off a pack label.
|
||||
OCR_MAX_SIDE_PX = int(os.getenv("OCR_MAX_SIDE_PX", "1280"))
|
||||
# onnxruntime intra-op threads per session; the read itself is serialised by
|
||||
# a lock (the engine is not thread-safe).
|
||||
OCR_NUM_THREADS = int(os.getenv("OCR_NUM_THREADS", "2"))
|
||||
# The assembled label is capped here, on a word boundary. 500 is the limit of
|
||||
# the `text` field on the image-search routes, so client and server text are
|
||||
# bounded alike.
|
||||
OCR_MAX_CHARS = int(os.getenv("OCR_MAX_CHARS", "500"))
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Capture-to-catalog - what /search/identify does when the photo is a product
|
||||
# the catalog does not have (app/services/capture_discovery.py)
|
||||
# ---------------------------------------------------------------------------
|
||||
# OFF by default. When on, an unconfirmed identify reads the label, and - for a
|
||||
# brand we already know - queues the product through the 11-stage pipeline and
|
||||
# answers at once with a provisional card and a job id to poll. The row lands
|
||||
# with validation_status=needs_review. This is the only path by which a public
|
||||
# route WRITES to a brand table, which is why it is a switch.
|
||||
ENABLE_CAPTURE_DISCOVERY = _bool("ENABLE_CAPTURE_DISCOVERY", "false")
|
||||
# Photos and job records. Under DATA_DIR so a container keeps them on the volume.
|
||||
CAPTURE_DIR = _dir("CAPTURE_DIR", DATA_DIR / "captures")
|
||||
# Jobs waiting behind the one capture worker. Each runs all 11 stages - web
|
||||
# image search and a live retail lookup included - so a minute or more apiece.
|
||||
CAPTURE_QUEUE_MAX = int(os.getenv("CAPTURE_QUEUE_MAX", "8"))
|
||||
# New discovery jobs one client (by IP) may start per hour. The route is public.
|
||||
CAPTURE_MAX_PER_CLIENT_PER_HOUR = int(os.getenv("CAPTURE_MAX_PER_CLIENT_PER_HOUR", "20"))
|
||||
# Absolute base of THIS API as the outside world reaches it, e.g.
|
||||
# https://api.example.com. Needed only to show the colleague's own photo as the
|
||||
# product image when the web search found none: image URLs must be absolute
|
||||
# (vector_store._usable_image_url drops a bare path). Blank = never use it.
|
||||
CAPTURE_PUBLIC_BASE_URL = os.getenv("CAPTURE_PUBLIC_BASE_URL", "").strip()
|
||||
# A live retail-presence lookup (retail_presence.check_listing, live=True) for
|
||||
# each discovered product: the pack-size check that catches a misread label.
|
||||
CAPTURE_RETAIL_CHECK = _bool("CAPTURE_RETAIL_CHECK", "true")
|
||||
# Stage 2's LLM description. Off falls back to the factual template.
|
||||
CAPTURE_USE_LLM = _bool("CAPTURE_USE_LLM", "true")
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# USDA FoodData Central - nutrition for loose, unbranded commodities
|
||||
# ---------------------------------------------------------------------------
|
||||
# Open Food Facts catalogues packaged products and has no entry for a raw
|
||||
# apple, which is why every fresh-produce row had no nutrition at all. USDA's
|
||||
# Foundation Foods and SR Legacy datasets are laboratory analyses of raw
|
||||
# commodities, published per 100 g of edible portion.
|
||||
#
|
||||
# NO KEY IS REQUIRED for normal operation: `nutrition_usda_service` reads a
|
||||
# snapshot built from USDA's open bulk download, so an enrichment run makes zero
|
||||
# outbound USDA calls. A key is only needed to look up an id the snapshot lacks,
|
||||
# or to rebuild the snapshot from the live API.
|
||||
#
|
||||
# Plain os.getenv rather than `_require(..., feature_flag=...)` on purpose: a
|
||||
# fresh checkout with no key must still import, or the whole test suite fails at
|
||||
# collection time.
|
||||
USE_USDA_FDC = _bool("USE_USDA_FDC", "true")
|
||||
USDA_FDC_API_KEY = os.getenv("USDA_FDC_API_KEY", "").strip()
|
||||
|
||||
# Score newly uploaded products automatically, instead of waiting for somebody
|
||||
# to remember to POST /api/admin/nutrition-intelligence/enrich. Runs as a
|
||||
# background job AFTER the ingestion batch finishes, never inside it - see the
|
||||
# comment at the submission site in `app/core/batch_ingest.py`.
|
||||
#
|
||||
# The off switch exists because this is the one part of ingestion that makes
|
||||
# outbound calls per product: an operator loading a very large catalogue on a
|
||||
# metered connection, or re-running an import they intend to score later in one
|
||||
# controlled pass, needs a way to say "not now" without a code change.
|
||||
AUTO_ENRICH_ON_UPLOAD = _bool("AUTO_ENRICH_ON_UPLOAD", "true")
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Store-catalog enrichment pipeline (app/core/store_catalog_pipeline.py)
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stages 7-10 of the store-Excel ingestion pipeline. Ported from the sibling
|
||||
# Universal_Catalog_Barcode_Enrichment project along with the code that reads
|
||||
# them; the names are kept identical so the ported modules need no edits.
|
||||
#
|
||||
# The three network-touching flags default to FALSE here, unlike in the
|
||||
# sibling. This backend serves an interactive API on a shared 8GB host, and a
|
||||
# 2000-row upload with web lookups on would fire thousands of outbound
|
||||
# requests. Turn them on deliberately, per environment.
|
||||
ENABLE_SKU_WEB_LOOKUP = _bool("ENABLE_SKU_WEB_LOOKUP", "false")
|
||||
ENABLE_BARCODE_LOOKUP = _bool("ENABLE_BARCODE_LOOKUP", "false")
|
||||
ENABLE_MANUFACTURER_SITE_LOOKUP = _bool("ENABLE_MANUFACTURER_SITE_LOOKUP", "false")
|
||||
|
||||
# Barcodes AFTER the upload settles, in bulk, on the enrichment job's own
|
||||
# thread. Default TRUE where ENABLE_BARCODE_LOOKUP above is false, and the
|
||||
# difference is cost, not appetite for risk:
|
||||
#
|
||||
# ENABLE_BARCODE_LOOKUP = one search request PER PRODUCT, inline, against an
|
||||
# endpoint capped at 10 requests/minute. A 200-row
|
||||
# upload is twenty minutes of a held request.
|
||||
# ENRICH_BARCODES_ON_UPLOAD = one corpus fetch PER BRAND (~5 requests total),
|
||||
# matched offline, after the uploader has their
|
||||
# result. Cost is per brand, not per row.
|
||||
#
|
||||
# It also has to run before the nutrition phase rather than beside it: a
|
||||
# barcode makes the nutrition lookup exact (0.95) instead of fuzzy (0.32), and
|
||||
# skip_if_verified means whichever lands first wins permanently.
|
||||
ENRICH_BARCODES_ON_UPLOAD = _bool("ENRICH_BARCODES_ON_UPLOAD", "true")
|
||||
|
||||
# Offline/deterministic stages - safe to leave on.
|
||||
ENABLE_HSN_GST_ENRICHMENT = _bool("ENABLE_HSN_GST_ENRICHMENT", "true")
|
||||
ENABLE_PRODUCT_VALIDATION = _bool("ENABLE_PRODUCT_VALIDATION", "true")
|
||||
|
||||
# Validation gate thresholds: below REJECT the row is dropped, below REVIEW it
|
||||
# is stored but flagged `validation_status="review"`.
|
||||
VALIDATION_REJECT_THRESHOLD = float(os.getenv("VALIDATION_REJECT_THRESHOLD", "0.35"))
|
||||
VALIDATION_REVIEW_THRESHOLD = float(os.getenv("VALIDATION_REVIEW_THRESHOLD", "0.70"))
|
||||
|
||||
# Pack-size explosion: how many size rows one uploaded product may become.
|
||||
MAX_VARIANTS_PER_PRODUCT = int(os.getenv("MAX_VARIANTS_PER_PRODUCT", "6"))
|
||||
ENABLE_PER_VARIANT_IMAGES = _bool("ENABLE_PER_VARIANT_IMAGES", "false")
|
||||
PER_VARIANT_IMAGE_MAX_RESULTS = int(os.getenv("PER_VARIANT_IMAGE_MAX_RESULTS", "10"))
|
||||
|
||||
# Barcode lookup tuning. The cache TTL is long (30 days) because a GTIN for a
|
||||
# given pack size does not change, and negative results are cached too.
|
||||
BARCODE_LOOKUP_TIMEOUT_SECONDS = float(os.getenv("BARCODE_LOOKUP_TIMEOUT_SECONDS", "10"))
|
||||
BARCODE_LOOKUP_CACHE_TTL_SECONDS = float(os.getenv("BARCODE_LOOKUP_CACHE_TTL_SECONDS", str(30 * 24 * 3600)))
|
||||
BARCODE_LOOKUP_MAX_CONCURRENCY = int(os.getenv("BARCODE_LOOKUP_MAX_CONCURRENCY", "5"))
|
||||
BARCODE_COUNTRY_TAG = os.getenv("BARCODE_COUNTRY_TAG", "india")
|
||||
|
||||
# How similar a candidate's product name must be to ours before its barcode is
|
||||
# believed. Applies to BOTH directions: looking a barcode up from a name, and
|
||||
# looking a product up from a barcode.
|
||||
#
|
||||
# RAISED FROM matching.py's OWN 0.45 DEFAULT, ON EVIDENCE. That default is a
|
||||
# reasonable general floor, but by the time a candidate reaches this gate its
|
||||
# brand and pack size have ALREADY been matched - so the name is the only thing
|
||||
# left doing any discriminating, and it has to carry the whole decision.
|
||||
#
|
||||
# At 0.45 it did not. Scored across every catalogue barcode Open Food Facts
|
||||
# knows, more than half the accepted matches were a different product:
|
||||
#
|
||||
# floor accepted wrong
|
||||
# 0.45 15 8
|
||||
# 0.70 7 2
|
||||
# 0.78 2 0
|
||||
#
|
||||
# 0.761 "Tata Tea Gold 500g" -> "Tata Tea Gold Care" different
|
||||
# 0.658 "MTR Masala 300g" -> "MTR Chana Masala" different
|
||||
# 0.538 "Aachi Chicken Masala 50g" -> "Chicken Kabab/65 Masala" different
|
||||
# 0.097 "Lion Dates Powder 100g" -> "PEPER NOTEN" different
|
||||
#
|
||||
# 0.78 is where the sample is clean, NOT where the yield is good, and it is a
|
||||
# judgement rather than a separation: two products tie at 0.773 with opposite
|
||||
# verdicts. It is set for precision because a WRONG barcode is worse than no
|
||||
# barcode - it is an identifier other systems join on, and 33 of the 95 already
|
||||
# in the catalogue are wrong, all of them accepted at the old floor.
|
||||
#
|
||||
# Lower it only with the yield/error numbers in front of you.
|
||||
BARCODE_MIN_NAME_SIMILARITY = float(os.getenv("BARCODE_MIN_NAME_SIMILARITY", "0.78"))
|
||||
|
||||
# Optional barcode source credentials. Each source disables itself when its
|
||||
# key is blank, so leaving these unset simply narrows the lookup cascade.
|
||||
GS1_INDIA_API_BASE_URL = os.getenv("GS1_INDIA_API_BASE_URL", "")
|
||||
GS1_INDIA_API_KEY = os.getenv("GS1_INDIA_API_KEY", "")
|
||||
UPC_DATABASE_API_KEY = os.getenv("UPC_DATABASE_API_KEY", "")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# HTTP client defaults
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -206,19 +692,32 @@ AUTH_SECRET_KEY = (
|
||||
# working day needs one sign-in, short enough that a leaked token expires.
|
||||
AUTH_TOKEN_TTL_MINUTES = int(os.getenv("AUTH_TOKEN_TTL_MINUTES", "720"))
|
||||
|
||||
# The two interactive accounts. Only PBKDF2 digests are stored - never a
|
||||
# password. `make_auth_secrets.py` prints both lines ready to paste.
|
||||
AUTH_ADMIN_USERNAME = os.getenv("AUTH_ADMIN_USERNAME", "admin")
|
||||
AUTH_ADMIN_PASSWORD_HASH = (
|
||||
_require("AUTH_ADMIN_PASSWORD_HASH", feature_flag="AUTH_ENABLED")
|
||||
if AUTH_ENABLED
|
||||
else os.getenv("AUTH_ADMIN_PASSWORD_HASH", "")
|
||||
# The interactive accounts. Only PBKDF2 digests are stored - never a password.
|
||||
# `make_auth_secrets.py` prints the lines ready to paste.
|
||||
#
|
||||
# `admin` is required whenever auth is on: without it nobody could sign in.
|
||||
AUTH_ADMIN_USERNAME = _clean(
|
||||
"AUTH_ADMIN_USERNAME", os.getenv("AUTH_ADMIN_USERNAME", "admin")
|
||||
)
|
||||
AUTH_USER_USERNAME = os.getenv("AUTH_USER_USERNAME", "user")
|
||||
AUTH_USER_PASSWORD_HASH = (
|
||||
_require("AUTH_USER_PASSWORD_HASH", feature_flag="AUTH_ENABLED")
|
||||
if AUTH_ENABLED
|
||||
else os.getenv("AUTH_USER_PASSWORD_HASH", "")
|
||||
AUTH_ADMIN_PASSWORD_HASH = _clean(
|
||||
"AUTH_ADMIN_PASSWORD_HASH",
|
||||
(
|
||||
_require("AUTH_ADMIN_PASSWORD_HASH", feature_flag="AUTH_ENABLED")
|
||||
if AUTH_ENABLED
|
||||
else os.getenv("AUTH_ADMIN_PASSWORD_HASH", "")
|
||||
),
|
||||
)
|
||||
|
||||
# The second `user` account is OPTIONAL, and left unset in this deployment.
|
||||
# An empty hash is how the account is switched off: auth.py builds its account
|
||||
# table from these values and omits any entry whose hash is blank, so there is
|
||||
# nothing to sign in to. Setting the hash again re-enables it with no code
|
||||
# change - which is exactly what the test suite does in tests/conftest.py.
|
||||
AUTH_USER_USERNAME = _clean(
|
||||
"AUTH_USER_USERNAME", os.getenv("AUTH_USER_USERNAME", "user")
|
||||
)
|
||||
AUTH_USER_PASSWORD_HASH = _clean(
|
||||
"AUTH_USER_PASSWORD_HASH", os.getenv("AUTH_USER_PASSWORD_HASH", "")
|
||||
)
|
||||
|
||||
# Failed-login throttle, applied per username+client-IP. Prevents an exposed
|
||||
@@ -243,6 +742,12 @@ AUTH_LOCKOUT_SECONDS = int(os.getenv("AUTH_LOCKOUT_SECONDS", "300"))
|
||||
AUTH_ALLOW_ANY_LOGIN = _bool("AUTH_ALLOW_ANY_LOGIN", "false")
|
||||
|
||||
|
||||
# Shortest acceptable API key secret. token_urlsafe(32) yields 43 characters, so
|
||||
# this rejects hand-typed values without rejecting anything the documented
|
||||
# generator produces.
|
||||
API_KEY_MIN_LENGTH = 32
|
||||
|
||||
|
||||
def _parse_api_keys(raw: str) -> dict:
|
||||
"""
|
||||
Parse ``API_KEYS`` - ``name:role:secret`` triples, comma-separated.
|
||||
@@ -250,6 +755,16 @@ def _parse_api_keys(raw: str) -> dict:
|
||||
Keyed by secret because that is what an inbound request presents. One entry
|
||||
per consumer is the point: a shared key cannot be revoked for one caller
|
||||
without breaking all of them.
|
||||
|
||||
Secrets must be at least API_KEY_MIN_LENGTH characters. That is not about
|
||||
guessing the key over the network - the lockout and the network itself make
|
||||
online brute force impractical - but about what /api/health publishes. It
|
||||
reports a truncated digest of every configured key so a deployment can be
|
||||
checked against the config it was built from, and a digest of a *raw* secret
|
||||
is only safe when the secret is unguessable offline. An admin password hash
|
||||
embeds a random salt, so its fingerprint discloses nothing; an API key has no
|
||||
salt, and a hand-picked "changeme" would fall to a wordlist in seconds.
|
||||
Generate one with: python -c "import secrets; print(secrets.token_urlsafe(32))"
|
||||
"""
|
||||
parsed: dict = {}
|
||||
for entry in raw.split(","):
|
||||
@@ -263,18 +778,38 @@ def _parse_api_keys(raw: str) -> dict:
|
||||
f"comma-separated between entries."
|
||||
)
|
||||
name, role, secret = (p.strip() for p in parts)
|
||||
if role not in {"admin", "user"}:
|
||||
# MUST stay in step with ROLE_PERMISSIONS in app/infrastructure/security.py,
|
||||
# which is the source of truth. It is duplicated rather than imported
|
||||
# because security.py imports THIS module, so importing it back here
|
||||
# would be a cycle. A role added there but not here is rejected at boot
|
||||
# with the message below - loud, and before any request is served.
|
||||
if role not in {"admin", "user", "uploader"}:
|
||||
raise RuntimeError(
|
||||
f"API_KEYS entry {name!r} has role {role!r}; expected 'admin' or 'user'."
|
||||
f"API_KEYS entry {name!r} has role {role!r}; expected 'admin', 'user' "
|
||||
f"or 'uploader'."
|
||||
)
|
||||
if not secret:
|
||||
raise RuntimeError(f"API_KEYS entry {name!r} has an empty secret.")
|
||||
if len(secret) < API_KEY_MIN_LENGTH:
|
||||
raise RuntimeError(
|
||||
f"API_KEYS entry {name!r} has a {len(secret)}-character secret; at least "
|
||||
f"{API_KEY_MIN_LENGTH} are required, because /api/health publishes a digest "
|
||||
f"of it. Generate one with: "
|
||||
f"python -c \"import secrets; print(secrets.token_urlsafe(32))\""
|
||||
)
|
||||
parsed[secret] = (name, role)
|
||||
return parsed
|
||||
|
||||
|
||||
# Machine consumers of api.<domain>. Empty by default - browser sessions go
|
||||
# through /api/auth/login instead, and a key that nobody needs is only risk.
|
||||
#
|
||||
# NAME THE KEY FOR ITS FUNCTION, NOT THE PERSON HOLDING IT.
|
||||
# /api/health is public and reports {name, role, fingerprint} for every
|
||||
# configured key (describe_api_keys in security.py). The secret is never
|
||||
# exposed, but the NAME is - so `catalog-drop:uploader:...` is right and
|
||||
# `priya-laptop:uploader:...` publishes a colleague's name to anyone who
|
||||
# curls the health endpoint.
|
||||
API_KEYS = _parse_api_keys(os.getenv("API_KEYS", ""))
|
||||
|
||||
# Default RAG behaviour
|
||||
@@ -290,3 +825,24 @@ RAG_MAX_CONTEXT_CHARS = int(os.getenv("RAG_MAX_CONTEXT_CHARS", "4000"))
|
||||
# if you want an extra cutoff on top of that.
|
||||
_raw_max_distance = os.getenv("RAG_MAX_DISTANCE", "").strip()
|
||||
RAG_MAX_DISTANCE = float(_raw_max_distance) if _raw_max_distance else None
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Catalog search (GET /api/search) and suggest (GET /api/suggest)
|
||||
# ---------------------------------------------------------------------------
|
||||
# Deliberately separate from RAG_MAX_TOP_K above. That ceiling exists to protect
|
||||
# the LLM prompt budget in /api/chat, and raising it would degrade every chat
|
||||
# answer. /api/search feeds a product grid, which has no such budget - sharing
|
||||
# the constant is what silently truncated every brand search to 15 products.
|
||||
SEARCH_DEFAULT_TOP_K = int(os.getenv("SEARCH_DEFAULT_TOP_K", "60"))
|
||||
# Mirrors the browse endpoints' le=100000 so a brand search can return exactly
|
||||
# the same set as GET /api/brands/{brand}/products.
|
||||
SEARCH_MAX_TOP_K = int(os.getenv("SEARCH_MAX_TOP_K", "100000"))
|
||||
# Separate, much smaller ceiling for the hybrid path: semantic_search multiplies
|
||||
# top_k by 5 per brand table when a category/price filter is present, and every
|
||||
# read does SELECT * (which drags the vector(384) embedding column over the wire).
|
||||
SEARCH_HYBRID_MAX_TOP_K = int(os.getenv("SEARCH_HYBRID_MAX_TOP_K", "100"))
|
||||
SEARCH_LEXICAL_CANDIDATES = int(os.getenv("SEARCH_LEXICAL_CANDIDATES", "200"))
|
||||
|
||||
SUGGEST_DEFAULT_LIMIT = int(os.getenv("SUGGEST_DEFAULT_LIMIT", "8"))
|
||||
SUGGEST_MIN_QUERY_LEN = int(os.getenv("SUGGEST_MIN_QUERY_LEN", "2"))
|
||||
SUGGEST_FUZZY_MIN_RATIO = float(os.getenv("SUGGEST_FUZZY_MIN_RATIO", "0.72"))
|
||||
|
||||
24
app/intelligence/artifacts/archive/README.md
Normal file
24
app/intelligence/artifacts/archive/README.md
Normal file
@@ -0,0 +1,24 @@
|
||||
# Archived model artifacts
|
||||
|
||||
These three models still have **all of their training code, CLI flags and API
|
||||
options intact**. Only the pre-trained `.joblib` bundles were moved out of the
|
||||
loaded directory, because nothing in the application ever reads them back.
|
||||
|
||||
| Artifact | Why archived |
|
||||
|---|---|
|
||||
| `demand_forecast_model.joblib` | `train_forecast` fits it and writes the `demand_forecast` table, but `store_db.get_latest_demand_forecast()` has **zero callers** - no endpoint, service or MCP tool consumes the forecast. |
|
||||
| `store_performance_model.joblib` | Referenced only from `ml_training_service.train_store_performance`. No inference consumer. |
|
||||
| `purchase_propensity_model.joblib` | Referenced only from `ml_training_service.train_purchase_propensity`. No inference consumer. |
|
||||
|
||||
They are excluded from `train_all()`'s **default** set
|
||||
(`ml_training_service.PRODUCTION_MODELS`), not from the codebase. To rebuild one:
|
||||
|
||||
```
|
||||
python scripts/train_ml_models.py --models store_performance
|
||||
```
|
||||
|
||||
or `POST /api/admin/store-intelligence/train {"models": ["store_performance"]}`.
|
||||
|
||||
Training writes to `MODEL_ARTIFACTS_DIR` (the parent directory), so a retrain
|
||||
promotes the model back to production automatically - wire up an endpoint that
|
||||
reads it first, or it will simply sit there unread again.
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -17,7 +17,7 @@ business-rule signal.
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import Dict, List
|
||||
from typing import Dict
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
@@ -9,10 +9,9 @@ ML feature logic without a live Postgres instance.
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from datetime import date, datetime
|
||||
from typing import Dict, Iterable, List, Optional
|
||||
from datetime import date
|
||||
from typing import Dict, Iterable, Optional
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -26,7 +26,7 @@ individually; only the training data is pooled).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import date, timedelta
|
||||
from datetime import date
|
||||
from typing import List
|
||||
|
||||
import numpy as np
|
||||
|
||||
@@ -20,7 +20,6 @@ from __future__ import annotations
|
||||
import logging
|
||||
from typing import Any, Dict, List, Tuple
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from app.intelligence.model_utils import ModelBundle, load_bundle, save_bundle
|
||||
|
||||
@@ -22,7 +22,6 @@ from __future__ import annotations
|
||||
import logging
|
||||
from typing import Any, Dict, List
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from app.intelligence.model_utils import ModelBundle, load_bundle, save_bundle
|
||||
|
||||
@@ -22,9 +22,9 @@ repeatable ML training and for demos.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from dataclasses import dataclass
|
||||
from datetime import date, timedelta
|
||||
from typing import Dict, List, Optional, Sequence
|
||||
from typing import Dict, List, Sequence
|
||||
|
||||
import hashlib
|
||||
import numpy as np
|
||||
|
||||
@@ -18,10 +18,9 @@ propensity score in the analytics UI).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import date, timedelta
|
||||
from datetime import date
|
||||
from typing import List
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from app.intelligence import features as F
|
||||
|
||||
@@ -15,7 +15,6 @@ heavy tuning.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import List
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
@@ -17,8 +17,8 @@ is `predicted_score.sort_values(ascending=False)` on live order data.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import date, timedelta
|
||||
from typing import Dict, List, Literal, Optional
|
||||
from datetime import date
|
||||
from typing import Dict, List, Literal
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
189
app/main.py
189
app/main.py
@@ -14,17 +14,26 @@ import threading
|
||||
from contextlib import asynccontextmanager
|
||||
from pathlib import Path
|
||||
|
||||
from fastapi import FastAPI, HTTPException
|
||||
from fastapi import FastAPI, HTTPException, Request
|
||||
from fastapi.exceptions import RequestValidationError
|
||||
from fastapi.encoders import jsonable_encoder
|
||||
from fastapi.middleware.cors import CORSMiddleware
|
||||
from fastapi.staticfiles import StaticFiles
|
||||
from fastapi.responses import FileResponse
|
||||
from fastapi.responses import FileResponse, JSONResponse
|
||||
|
||||
from app.infrastructure.persistence import restore_bundled_assets
|
||||
from app.infrastructure.settings import API_CORS_ORIGINS
|
||||
from app.api.routers import health, brands, search, chat, catalog, system
|
||||
from app.infrastructure.settings import (
|
||||
API_CORS_ORIGINS,
|
||||
BATCH_AUTO_RESUME,
|
||||
BRAND_SYNC_INTERVAL_SECONDS,
|
||||
cleaned_env_names,
|
||||
)
|
||||
from app.infrastructure.security import auth_config_summary
|
||||
from app.api.routers import health, brands, search, suggest, chat, catalog, system
|
||||
from app.api.routers import stores, discounts, analytics as store_analytics, trending, recommendations, store_admin
|
||||
from app.api.routers import nutrition, nutrition_admin, upload
|
||||
from app.api.routers import auth, user_products, admin_train, mcp_info
|
||||
from app.api.routers import batch_catalog, uploads, brand_discovery
|
||||
from app.services.store_db import ensure_store_intelligence_schema
|
||||
from app.services.nutrition_db import ensure_nutrition_schema
|
||||
|
||||
@@ -46,6 +55,12 @@ except Exception as e: # pragma: no cover - depends on an optional dependency
|
||||
MCP_PATH = "/mcp"
|
||||
|
||||
|
||||
# Signals the background thread below to stop. Doubles as its sleep: waiting on
|
||||
# an Event rather than time.sleep() means shutdown is immediate instead of
|
||||
# blocking until the current BRAND_SYNC_INTERVAL_SECONDS elapses.
|
||||
_brand_sync_stop = threading.Event()
|
||||
|
||||
|
||||
@asynccontextmanager
|
||||
async def lifespan(_app: FastAPI):
|
||||
"""
|
||||
@@ -55,6 +70,10 @@ async def lifespan(_app: FastAPI):
|
||||
Postgres, which may be slow or briefly unreachable on a cold boot, and the
|
||||
server answering /api/health in under a second is what lets the container
|
||||
healthcheck pass while that settles.
|
||||
|
||||
That same thread then stays alive to re-run the brand reconcile on an
|
||||
interval, so a brand table that appears in the database after boot reaches
|
||||
the catalog without a restart.
|
||||
"""
|
||||
|
||||
# Runs before the thread below, and synchronously: the seed catalogs and
|
||||
@@ -66,15 +85,79 @@ async def lifespan(_app: FastAPI):
|
||||
except Exception as e:
|
||||
logger.warning("Could not restore bundled assets: %s", e)
|
||||
|
||||
# Cheap, and it must happen before anyone can press Resume: a batch that a
|
||||
# restart cut short is still marked "running" on disk, and until it is
|
||||
# reconciled the UI shows it as in flight with nothing behind it. This only
|
||||
# rewrites manifests - it deliberately starts no work. See
|
||||
# batch_ingest.scan_interrupted() for why auto-resume is not the default.
|
||||
try:
|
||||
from app.core.batch_ingest import scan_interrupted
|
||||
|
||||
interrupted = scan_interrupted()
|
||||
if interrupted:
|
||||
logger.info("Marked %d interrupted catalog batch(es).", len(interrupted))
|
||||
|
||||
# Retention, which otherwise only runs when the ingestion worker goes
|
||||
# idle. That was sufficient while every upload was queued work; it is
|
||||
# not now that /api/uploads/catalog stages drops for review and queues
|
||||
# nothing. An inbox nobody acts on would never start the worker, so the
|
||||
# sweep that reclaims abandoned uploads would never run either.
|
||||
try:
|
||||
from app.core.batch_ingest import purge_expired
|
||||
|
||||
purged = purge_expired()
|
||||
if purged:
|
||||
logger.info("Purged %d expired batch upload(s) at startup.", len(purged))
|
||||
except Exception as exc: # noqa: BLE001 - housekeeping must not block boot
|
||||
logger.warning("Could not purge expired batch uploads: %s", exc)
|
||||
if interrupted and BATCH_AUTO_RESUME:
|
||||
# Opt-in only. Importing the worker here rather than at module
|
||||
# scope keeps the queue and its thread out of a boot that never
|
||||
# needs them.
|
||||
from app.core import batch_worker
|
||||
|
||||
for batch_id in interrupted:
|
||||
try:
|
||||
batch_worker.submit(batch_id)
|
||||
except Exception as exc: # noqa: BLE001 - a full queue is not fatal
|
||||
logger.warning("Could not auto-resume batch %s: %s", batch_id, exc)
|
||||
except Exception as e:
|
||||
logger.warning("Could not reconcile interrupted catalog batches: %s", e)
|
||||
|
||||
def _async_init():
|
||||
try:
|
||||
ensure_store_intelligence_schema()
|
||||
ensure_nutrition_schema()
|
||||
from app.api.routers.system import _run_background_auto_seed
|
||||
from app.api.routers.system import _run_background_auto_seed, _run_background_brand_sync
|
||||
_run_background_auto_seed()
|
||||
# Runs after the auto-seed so a cold boot has already loaded the
|
||||
# bundled catalogs and this finds nothing to do. On a warm boot the
|
||||
# auto-seed no-ops and this is what picks up a brand table or seed
|
||||
# file that appeared since last time.
|
||||
_run_background_brand_sync()
|
||||
except Exception as e:
|
||||
logger.warning("Startup background init warning: %s", e)
|
||||
|
||||
# Everything above is a one-shot boot sweep. Keep going on an interval so
|
||||
# a brand table created directly in the database - bypassing the app, and
|
||||
# therefore bypassing the cache invalidation every write path does - still
|
||||
# reaches the catalog. Without this it waits for the next restart.
|
||||
if BRAND_SYNC_INTERVAL_SECONDS <= 0:
|
||||
logger.info("Periodic brand sync disabled (BRAND_SYNC_INTERVAL_SECONDS=0)")
|
||||
return
|
||||
|
||||
logger.info("Periodic brand sync every %ss", BRAND_SYNC_INTERVAL_SECONDS)
|
||||
while not _brand_sync_stop.wait(BRAND_SYNC_INTERVAL_SECONDS):
|
||||
try:
|
||||
# Imported per iteration for the same reason as above: a failure
|
||||
# to import must not be what kills the loop.
|
||||
from app.api.routers.system import _run_background_brand_sync
|
||||
_run_background_brand_sync()
|
||||
except Exception as e:
|
||||
# One bad sweep - an unreachable database, a malformed seed file -
|
||||
# must not end the loop, or recovery needs a restart again.
|
||||
logger.warning("Periodic brand sync warning: %s", e)
|
||||
|
||||
threading.Thread(target=_async_init, daemon=True).start()
|
||||
|
||||
# The mounted MCP app carries its own lifespan, which starts the session
|
||||
@@ -82,11 +165,17 @@ async def lifespan(_app: FastAPI):
|
||||
# that lifespan - only the outermost app's is executed - so without this
|
||||
# the endpoint exists, accepts a connection, and then fails on the first
|
||||
# message with a session manager that was never started.
|
||||
if _mcp_app is not None:
|
||||
async with _mcp_app.lifespan(_mcp_app):
|
||||
try:
|
||||
if _mcp_app is not None:
|
||||
async with _mcp_app.lifespan(_mcp_app):
|
||||
yield
|
||||
else:
|
||||
yield
|
||||
else:
|
||||
yield
|
||||
finally:
|
||||
# Releases the interval wait immediately. The thread is a daemon, so this
|
||||
# is not what lets the process exit - it is what stops a reconcile from
|
||||
# starting against a database the shutdown is already tearing down.
|
||||
_brand_sync_stop.set()
|
||||
|
||||
|
||||
app = FastAPI(
|
||||
@@ -126,6 +215,27 @@ app.add_middleware(
|
||||
allow_headers=["*"],
|
||||
)
|
||||
|
||||
|
||||
def _json_safe(value):
|
||||
"""Replace floats JSON cannot carry (NaN, +/-inf) so an error can be sent."""
|
||||
if isinstance(value, float) and (value != value or value in (float("inf"), float("-inf"))):
|
||||
return str(value)
|
||||
if isinstance(value, dict):
|
||||
return {k: _json_safe(v) for k, v in value.items()}
|
||||
if isinstance(value, (list, tuple)):
|
||||
return [_json_safe(v) for v in value]
|
||||
return value
|
||||
|
||||
|
||||
@app.exception_handler(RequestValidationError)
|
||||
async def _validation_error_as_422(request: Request, exc: RequestValidationError) -> JSONResponse:
|
||||
"""FastAPI's own 422 body echoes the rejected input. Python's JSON parser
|
||||
accepts `NaN` on the way in, pydantic rejects it, and the echo then fails
|
||||
to serialise - so a client that sent one NaN in a float field got a 500
|
||||
instead of the 422 that names the field. Found by the image-vector
|
||||
search, where the body is 1024 floats; applies to every route."""
|
||||
return JSONResponse(status_code=422, content={"detail": _json_safe(jsonable_encoder(exc.errors()))})
|
||||
|
||||
# A wrong origin list fails only in the browser, as an opaque "blocked by CORS"
|
||||
# with a perfectly healthy 200 in the server log - so state the effective list
|
||||
# at startup, where it can actually be compared against the frontend's URL.
|
||||
@@ -144,6 +254,61 @@ if API_CORS_ORIGINS and all(
|
||||
", ".join(API_CORS_ORIGINS),
|
||||
)
|
||||
|
||||
# The same argument as the CORS block above, for the other setting whose
|
||||
# misconfiguration is invisible from the outside. A wrong credential fails only
|
||||
# as "Invalid username or password.", which is indistinguishable from a user
|
||||
# mistyping - so state what the process actually loaded, at startup, where it
|
||||
# can be compared against the config the image was built from.
|
||||
#
|
||||
# No password and no hash is printed. `fingerprint` identifies WHICH credential
|
||||
# is loaded (see security.password_hash_fingerprint); `source` says whether it
|
||||
# came from the container's environment or from the .env file, which is the
|
||||
# only way to notice a deployment platform's Environment tab overriding the
|
||||
# image. Compare against: python scripts/make_auth_secrets.py --fingerprint
|
||||
_auth_cfg = auth_config_summary()
|
||||
logger.info(
|
||||
"Auth config: enabled=%s allow_any_login=%s admin_username=%r "
|
||||
"hash=%s/%s fingerprint=%s source=%s (username source=%s) "
|
||||
"api_keys=%d/%s %s",
|
||||
_auth_cfg["enabled"],
|
||||
_auth_cfg["allow_any_login"],
|
||||
_auth_cfg["admin_username"],
|
||||
"pbkdf2_sha256" if _auth_cfg["password_hash_valid"] else "INVALID",
|
||||
_auth_cfg["password_hash_iterations"],
|
||||
_auth_cfg["password_hash_fingerprint"] or "(none)",
|
||||
_auth_cfg["password_hash_source"],
|
||||
_auth_cfg["admin_username_source"],
|
||||
_auth_cfg["api_keys_count"],
|
||||
_auth_cfg["api_keys_source"],
|
||||
# Names, not secrets. A key added to .env.production but only restarted into
|
||||
# a running container never appears here - which is the whole point.
|
||||
[k["name"] for k in _auth_cfg["api_keys"]] or "(none)",
|
||||
)
|
||||
if cleaned_env_names():
|
||||
logger.warning(
|
||||
"These settings arrived wrapped in quotes or padded with whitespace and were "
|
||||
"cleaned before use: %s. Pasting into a deployment platform's Environment tab "
|
||||
"is the usual source. They work now, but the next value may not - store them "
|
||||
"unquoted.",
|
||||
", ".join(cleaned_env_names()),
|
||||
)
|
||||
if _auth_cfg["enabled"] and _auth_cfg["password_hash_source"] == "process-env":
|
||||
logger.warning(
|
||||
"AUTH_ADMIN_PASSWORD_HASH came from the process environment, which OVERRIDES "
|
||||
"the .env file (load_dotenv is called without override=True). Under "
|
||||
"docker-compose that is just `env_file:` and is expected. Under Dokploy it "
|
||||
"means the service's Environment tab is supplying this credential and the one "
|
||||
"baked into the image by `COPY .env.production .env` is being ignored - which "
|
||||
"is how a corrected password keeps failing after a redeploy."
|
||||
)
|
||||
if _auth_cfg["enabled"] and not _auth_cfg["password_hash_valid"]:
|
||||
logger.error(
|
||||
"AUTH_ADMIN_PASSWORD_HASH is not a usable PBKDF2 digest, so EVERY sign-in "
|
||||
"will return 401 no matter which password is typed. It came from %s. "
|
||||
"Regenerate it with: python scripts/make_auth_secrets.py",
|
||||
_auth_cfg["password_hash_source"],
|
||||
)
|
||||
|
||||
app.include_router(health.router, prefix="/api")
|
||||
app.include_router(auth.router, prefix="/api")
|
||||
app.include_router(user_products.router, prefix="/api")
|
||||
@@ -151,6 +316,7 @@ app.include_router(admin_train.router, prefix="/api")
|
||||
app.include_router(system.router, prefix="/api")
|
||||
app.include_router(brands.router, prefix="/api")
|
||||
app.include_router(search.router, prefix="/api")
|
||||
app.include_router(suggest.router, prefix="/api")
|
||||
app.include_router(chat.router, prefix="/api")
|
||||
app.include_router(catalog.router, prefix="/api")
|
||||
app.include_router(stores.router, prefix="/api")
|
||||
@@ -162,6 +328,9 @@ app.include_router(store_admin.router, prefix="/api")
|
||||
app.include_router(nutrition.router, prefix="/api")
|
||||
app.include_router(nutrition_admin.router, prefix="/api")
|
||||
app.include_router(upload.router, prefix="/api")
|
||||
app.include_router(batch_catalog.router, prefix="/api")
|
||||
app.include_router(uploads.router, prefix="/api")
|
||||
app.include_router(brand_discovery.router, prefix="/api")
|
||||
app.include_router(mcp_info.router, prefix="/api")
|
||||
|
||||
# MCP lives outside /api on purpose: it is a protocol endpoint for AI clients,
|
||||
@@ -175,7 +344,7 @@ if _mcp_app is not None:
|
||||
# Serve built frontend static files if dist exists (single-port unified
|
||||
# deployment). Off in the normal setup: the React app is served by its own
|
||||
# nginx on catalogue.nearle.ai.in and this API answers on
|
||||
# mcp.catalogue.nearle.ai.in, so no dist/ is present here and the JSON root
|
||||
# mcp.nearle.ai.in, so no dist/ is present here and the JSON root
|
||||
# handler at the bottom of this file is what responds to /.
|
||||
#
|
||||
# The candidates cover both repo layouts - the sibling checkout is named
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
MCP server exposing the catalog as tools for AI clients.
|
||||
|
||||
Mounted onto the FastAPI app at /mcp (see app/main.py), so it ships in the same
|
||||
container and answers on the same host - mcp.catalogue.nearle.ai.in/mcp.
|
||||
container and answers on the same host - mcp.nearle.ai.in/mcp.
|
||||
|
||||
Scope: reads only. Every tool here maps to a service function the REST API
|
||||
already exposes through a GET. None of the write or compute endpoints - catalog
|
||||
@@ -154,7 +154,8 @@ def _slim(product: Dict[str, Any]) -> Dict[str, Any]:
|
||||
return {
|
||||
k: v
|
||||
for k, v in product.items()
|
||||
if k not in {"embedding", "embedding_text", "distance"} and v not in (None, "", [], {})
|
||||
if k not in {"embedding", "embedding_text", "distance", "img_vector", "img_vector_src"}
|
||||
and v not in (None, "", [], {})
|
||||
}
|
||||
|
||||
|
||||
|
||||
130
app/services/active_brands.py
Normal file
130
app/services/active_brands.py
Normal file
@@ -0,0 +1,130 @@
|
||||
"""The single source of truth for which brands the application works on.
|
||||
|
||||
WHY THIS EXISTS
|
||||
---------------
|
||||
The dev dataset grew to 31 seed catalogs and 16 `brand_*` tables. Every
|
||||
"all brands" query fans out across every table, boot parses ~19MB of seed
|
||||
JSON, and 14 of those seed files have no table yet - so the next auto-seed
|
||||
would silently create 14 more. That is far more than an 8GB dev machine
|
||||
needs to exercise the features.
|
||||
|
||||
Rather than delete data, ONE setting decides which brands participate:
|
||||
|
||||
ACTIVE_BRANDS=Amul,Cadbury,Hindustan Unilever
|
||||
|
||||
and every read path, pipeline, RAG query, MCP tool and Dagster asset
|
||||
inherits it, because they all funnel through `vector_store`'s two brand
|
||||
discovery functions (`list_available_brands` and `_list_brand_table_suffixes`)
|
||||
plus `brand_sync.load_seed_catalogs`. Going back to 5, 10 or 25 brands is a
|
||||
one-line `.env` change, not a code edit.
|
||||
|
||||
THE EMPTY-MEANS-ALL CONTRACT
|
||||
----------------------------
|
||||
An unset or blank `ACTIVE_BRANDS` disables filtering entirely, so this module
|
||||
is a no-op on any deployment that does not opt in. That is deliberate:
|
||||
`.env.production` leaves it unset, so production keeps serving every brand
|
||||
while local development runs on three. It also means the feature can be
|
||||
switched off wholesale if it ever gets in the way.
|
||||
|
||||
NAMES ARE RESOLVED, NOT MATCHED
|
||||
-------------------------------
|
||||
Configured names go through `resolve_parent_brand` -> `_sanitize_name`, the
|
||||
same two steps that decide which table a product is stored in. So
|
||||
`ACTIVE_BRANDS=Tata` activates `brand_hindustan_unilever` (the "hul tata tea"
|
||||
alias claims it), exactly like an ingest of that brand would. Comparing raw
|
||||
strings here would have let the config and the storage layer disagree.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import FrozenSet, Iterable, List, Optional
|
||||
|
||||
from app.infrastructure.settings import ACTIVE_BRANDS as _RAW_ACTIVE_BRANDS
|
||||
from app.services.brand_registry import resolve_parent_brand
|
||||
|
||||
# Parsed once. `_CACHE` holds `(suffixes, display_names)`; `None` in slot 0
|
||||
# means "no filtering configured", which is different from "an empty set of
|
||||
# active brands" - the latter would hide the entire catalog.
|
||||
_CACHE: Optional[tuple] = None
|
||||
|
||||
|
||||
def _sanitize(name: str) -> str:
|
||||
"""Brand name -> table suffix.
|
||||
|
||||
Imported lazily from vector_store because vector_store imports THIS module
|
||||
at the top level; a top-level import here would be a cycle. query_intent
|
||||
already breaks the identical cycle the same way.
|
||||
"""
|
||||
from app.services.vector_store import _sanitize_name
|
||||
|
||||
return _sanitize_name(name)
|
||||
|
||||
|
||||
def _parse(raw: str) -> tuple:
|
||||
names = [part.strip() for part in (raw or "").split(",")]
|
||||
names = [n for n in names if n]
|
||||
if not names:
|
||||
return (None, [])
|
||||
|
||||
suffixes = []
|
||||
display = []
|
||||
for name in names:
|
||||
suffix = _sanitize(resolve_parent_brand(name))
|
||||
if not suffix or suffix in suffixes:
|
||||
continue
|
||||
suffixes.append(suffix)
|
||||
display.append(name)
|
||||
return (frozenset(suffixes), display)
|
||||
|
||||
|
||||
def _load() -> tuple:
|
||||
global _CACHE
|
||||
if _CACHE is None:
|
||||
_CACHE = _parse(_RAW_ACTIVE_BRANDS)
|
||||
return _CACHE
|
||||
|
||||
|
||||
def invalidate() -> None:
|
||||
"""Drop the parsed config. For tests that monkeypatch the raw setting."""
|
||||
global _CACHE
|
||||
_CACHE = None
|
||||
|
||||
|
||||
def filtering_enabled() -> bool:
|
||||
"""True when ACTIVE_BRANDS is set to a non-empty list."""
|
||||
return _load()[0] is not None
|
||||
|
||||
|
||||
def active_brand_suffixes() -> Optional[FrozenSet[str]]:
|
||||
"""Table suffixes of the active brands, or None when filtering is off."""
|
||||
return _load()[0]
|
||||
|
||||
|
||||
def is_active_suffix(suffix: str) -> bool:
|
||||
"""Whether a `brand_<suffix>` table participates. True for all when off."""
|
||||
active = _load()[0]
|
||||
return True if active is None else (suffix or "").lower() in active
|
||||
|
||||
|
||||
def filter_suffixes(suffixes: Iterable[str]) -> List[str]:
|
||||
"""Keep only the active suffixes, preserving the caller's order."""
|
||||
active = _load()[0]
|
||||
if active is None:
|
||||
return list(suffixes)
|
||||
return [s for s in suffixes if (s or "").lower() in active]
|
||||
|
||||
|
||||
def is_active_brand(brand: str) -> bool:
|
||||
"""Whether a brand NAME (in any alias form) resolves to an active table."""
|
||||
active = _load()[0]
|
||||
if active is None:
|
||||
return True
|
||||
return _sanitize(resolve_parent_brand(brand)) in active
|
||||
|
||||
|
||||
def active_display_names() -> List[str]:
|
||||
"""The configured names, verbatim, for logs and Dagster partition keys.
|
||||
|
||||
Empty when filtering is off - callers that need the real brand list in
|
||||
that case should ask `vector_store.list_available_brands()` instead.
|
||||
"""
|
||||
return list(_load()[1])
|
||||
@@ -1,6 +1,6 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Dict, List, Optional
|
||||
from typing import Dict, List
|
||||
|
||||
import pandas as pd
|
||||
|
||||
@@ -50,7 +50,6 @@ def product_dashboard(brand: str, image_id: str) -> Dict:
|
||||
|
||||
popularity = None
|
||||
if engagement_row:
|
||||
from app.intelligence import features as F
|
||||
feat_row = pd.DataFrame([{
|
||||
"views_norm": min(engagement_row["views"] / 10, 100),
|
||||
"wishlist_norm": min(engagement_row["wishlist_count"] / 2, 100),
|
||||
|
||||
1370
app/services/brand_discovery.py
Normal file
1370
app/services/brand_discovery.py
Normal file
File diff suppressed because it is too large
Load Diff
@@ -1,3 +1,7 @@
|
||||
import re
|
||||
from functools import lru_cache
|
||||
from typing import Dict, Optional
|
||||
|
||||
BRAND_ALIASES = {
|
||||
# Cadbury family
|
||||
"cadbury gems": "cadbury",
|
||||
@@ -248,11 +252,97 @@ BRAND_ALIASES = {
|
||||
"sunfeast bounce": "sunfeast",
|
||||
"sunfeast yippee": "sunfeast",
|
||||
"sunfeast cookies": "sunfeast",
|
||||
# SPELLING VARIANTS, not sub-brands.
|
||||
#
|
||||
# "Haldiram" and "Haldirams" were resolving to themselves, so nothing knew
|
||||
# they were one brand and uploads built brand_haldiram and brand_haldirams
|
||||
# side by side. The data was merged into the plural, which is the correct
|
||||
# name; THIS LINE is what stops the split reappearing on the next sheet
|
||||
# that spells it without the s. Removing it re-opens the bug.
|
||||
"haldiram": "haldirams",
|
||||
"haldiram's": "haldirams",
|
||||
# --- Regional South Indian brands -------------------------------------
|
||||
# Not in Open Food Facts (Udhaiyam 0 hits, Gopuram 0, Tenali Double Horse
|
||||
# 0), so their catalogues come from the brands' own storefronts - see
|
||||
# BRAND_STORE_DOMAINS below and app/services/brand_store.py.
|
||||
"udhayam": "udhaiyam",
|
||||
"udhaiyam dhall": "udhaiyam",
|
||||
"uthayam": "udhaiyam",
|
||||
"gopuram products": "gopuram",
|
||||
#
|
||||
# TENALI DOUBLE HORSE IS A DIFFERENT COMPANY FROM DOUBLE HORSE.
|
||||
#
|
||||
# Double Horse is Manjilas, in Kerala; Tenali Double Horse is an Andhra
|
||||
# rice and rava brand. They share a name and nothing else.
|
||||
#
|
||||
# This entry is LOAD-BEARING and must stay a DIRECT one. `_contains_word`
|
||||
# matches whole words in either direction, and "double horse" is a whole
|
||||
# word sequence inside "tenali double horse" - so the moment any
|
||||
# "double horse" key exists, the fallback loop folds the Andhra brand into
|
||||
# the Kerala one and both companies' products land in a single table.
|
||||
# `resolve_parent_brand` checks BRAND_ALIASES directly before it ever
|
||||
# reaches that loop, which is the only reason this works.
|
||||
#
|
||||
# BOTH names need a direct self-entry, because `_contains_word` is applied
|
||||
# in BOTH directions. Registering only the longer one moved the collision
|
||||
# rather than fixing it - measured: with only "tenali double horse" present,
|
||||
# resolve_parent_brand("Double Horse") returned "tenali double horse",
|
||||
# because "double horse" is a whole word sequence inside the alias key and
|
||||
# the loop matches `_contains_word(alias, key)` too. Kerala's brand was
|
||||
# swallowed by Andhra's instead of the other way round.
|
||||
#
|
||||
# A direct hit short-circuits before the loop, so each of these entries
|
||||
# protects the OTHER brand. Remove either one and the two merge.
|
||||
"tenali double horse": "tenali double horse",
|
||||
"tenali doublehorse": "tenali double horse",
|
||||
"double horse": "double horse",
|
||||
"doublehorse": "double horse",
|
||||
}
|
||||
|
||||
# A brand's own online shop, verified by a person.
|
||||
#
|
||||
# NOT resolved by search, on purpose. `retail_presence.resolve_brand_domain`
|
||||
# answers "Naga" with `cityofnagacebu.gov.ph` - the Philippine city - and a
|
||||
# wrong domain here does not cost one bad row, it imports a hundred of another
|
||||
# company's real products under this brand's name. `brand_store` treats a
|
||||
# domain from this map as already checked and applies its heuristic identity
|
||||
# guard only to domains that came from search.
|
||||
#
|
||||
# Add a brand here only after opening the site and confirming it is theirs.
|
||||
BRAND_STORE_DOMAINS: dict[str, str] = {
|
||||
"aachi": "aachifoods.com",
|
||||
"anil": "shop.theanilgroup.com",
|
||||
"double horse": "doublehorse.in",
|
||||
"gopuram": "gopuramproducts.com",
|
||||
"udhaiyam": "udhaiyamdhall.com",
|
||||
# Naga and Tenali Double Horse are deliberately ABSENT: their official
|
||||
# sites have not been confirmed. An absent brand simply has no store
|
||||
# catalogue, which is the safe outcome; a guessed one is not.
|
||||
}
|
||||
|
||||
|
||||
def get_brand_store_domain(brand: str) -> Optional[str]:
|
||||
"""The verified storefront for a brand, or None if we do not have one."""
|
||||
return BRAND_STORE_DOMAINS.get(resolve_parent_brand(brand).lower().strip())
|
||||
|
||||
DEFAULT_ALIASES = BRAND_ALIASES
|
||||
|
||||
|
||||
def _contains_word(haystack: str, needle: str) -> bool:
|
||||
"""True when `needle` occurs in `haystack` as a whole word.
|
||||
|
||||
Plain `in` would treat any fragment as a match, so a short brand name
|
||||
could be swallowed by an unrelated alias that merely contains those
|
||||
letters - e.g. "sun" inside "hul sunsilk". Anchoring both ends on a
|
||||
word boundary keeps the genuine multi-word hits ("tata" inside
|
||||
"hul tata tea") while dropping the fragment ones.
|
||||
"""
|
||||
if not needle:
|
||||
return False
|
||||
return re.search(r"(?<!\w)" + re.escape(needle) + r"(?!\w)", haystack) is not None
|
||||
|
||||
|
||||
@lru_cache(maxsize=1024)
|
||||
def resolve_parent_brand(brand: str) -> str:
|
||||
"""Return the parent (canonical) brand for storage purposes.
|
||||
|
||||
@@ -260,19 +350,118 @@ def resolve_parent_brand(brand: str) -> str:
|
||||
returns the parent brand name so that sub-brands share the same
|
||||
database table, S3 folder, and JSON file as their parent.
|
||||
|
||||
Falls back to fuzzy substring matching, then returns the input
|
||||
unchanged if no alias is known.
|
||||
Falls back to whole-word matching in either direction, then returns
|
||||
the input unchanged if no alias is known.
|
||||
|
||||
Cached because `_table_name()` in vector_store calls this on every
|
||||
query and the fallback loop scans all ~230 aliases. BRAND_ALIASES is
|
||||
a module constant that is never mutated, so the result is stable.
|
||||
"""
|
||||
key = brand.lower().strip()
|
||||
direct = BRAND_ALIASES.get(key)
|
||||
if direct:
|
||||
return direct
|
||||
for alias, parent in BRAND_ALIASES.items():
|
||||
if alias in key or key in alias:
|
||||
if _contains_word(key, alias) or _contains_word(alias, key):
|
||||
return parent
|
||||
prefix = _known_leading_brand(key)
|
||||
if prefix:
|
||||
return resolve_parent_brand(prefix)
|
||||
line = _known_leading_line(key)
|
||||
if line:
|
||||
return line
|
||||
parent = _unique_parent_prefix(key)
|
||||
if parent:
|
||||
return parent
|
||||
return brand
|
||||
|
||||
|
||||
def is_abbreviation(token: str, parent: str) -> bool:
|
||||
""""rb" for Reckitt Benckiser, "hul" for Hindustan Unilever, "jnj" for
|
||||
Johnson & Johnson. Short, not itself a word of the parent, and starting
|
||||
with the parent's first letter - which is what keeps "kit kat" (Nestle)
|
||||
from being read as the abbreviation "kit" plus a sub-brand "kat"."""
|
||||
parent_words = re.findall(r"[a-z0-9]+", (parent or "").lower())
|
||||
return (bool(parent_words) and 1 < len(token) <= 4 and token not in parent_words
|
||||
and token[0] == parent_words[0][0])
|
||||
|
||||
|
||||
# Line names that are also everyday words or another maker's name. Read as a
|
||||
# leading word they would misfile: "Laxmi Chilli Powder" is a spice brand, not
|
||||
# HUL; "Boost Energy Drink", "Finish Line", "Baby Oil" (any maker) likewise.
|
||||
_AMBIGUOUS_LINES = frozenset({"always", "boost", "finish", "wheel", "fairy", "ivory",
|
||||
"vanish", "laxmi", "whisper"})
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def _abbreviated_lines() -> dict:
|
||||
""""lizol" -> "reckitt benckiser", read off the alias "rb lizol".
|
||||
|
||||
Only line names of 5+ characters that exactly one parent registers: the
|
||||
bare "sun" off "hul sun" would otherwise send "Sun Pharma" to Unilever."""
|
||||
owners: dict = {}
|
||||
for alias, parent in BRAND_ALIASES.items():
|
||||
words = alias.split()
|
||||
if len(words) > 1 and is_abbreviation(words[0], parent):
|
||||
rest = " ".join(words[1:])
|
||||
if (len(rest.replace(" ", "")) >= 5 and rest not in _AMBIGUOUS_LINES
|
||||
and not rest.startswith("baby ")):
|
||||
owners.setdefault(rest, set()).add(parent)
|
||||
return {rest: next(iter(ps)) for rest, ps in owners.items() if len(ps) == 1}
|
||||
|
||||
|
||||
def _known_leading_line(key: str) -> Optional[str]:
|
||||
""""Lizol Floor Cleaner" -> "reckitt benckiser".
|
||||
|
||||
The registry spells Reckitt's lines with an abbreviation ("rb lizol"), so
|
||||
the bare line name is neither a key nor a parent and `_known_leading_brand`
|
||||
could not see it: a product typed with a line after it built a
|
||||
brand_lizol_floor_cleaner table. Reached only after every older rule failed."""
|
||||
lines = _abbreviated_lines()
|
||||
words = key.split()
|
||||
for n in range(len(words), 0, -1):
|
||||
parent = lines.get(" ".join(words[:n]))
|
||||
if parent:
|
||||
return parent
|
||||
return None
|
||||
|
||||
|
||||
def _unique_parent_prefix(key: str) -> Optional[str]:
|
||||
""""Reckitt" -> "reckitt benckiser": the typed words are the opening words
|
||||
of exactly one parent. Two parents sharing the opening ("tata ...") give
|
||||
nothing, and a single word must be 5+ characters."""
|
||||
words = key.split()
|
||||
if not words or (len(words) == 1 and len(words[0]) < 5):
|
||||
return None
|
||||
hits = {p for p in set(BRAND_ALIASES.values())
|
||||
if p.split()[:len(words)] == words and len(p.split()) > len(words)}
|
||||
return next(iter(hits)) if len(hits) == 1 else None
|
||||
|
||||
|
||||
def _known_leading_brand(key: str) -> Optional[str]:
|
||||
"""The longest leading run of words in `key` that is itself a known brand.
|
||||
|
||||
"Godrej Fab" matched nothing above: no alias contains it, and it contains
|
||||
no alias ("godrej no.1" and friends are all longer than "godrej"). So it
|
||||
fell through to the identity and a field capture would have built a
|
||||
brand_godrej_fab table beside brand_godrej. The same held for "Amul Taaza"
|
||||
and "Hindustan Unilever Rin" - a parent name followed by a product line
|
||||
nobody had registered.
|
||||
|
||||
Only reached when every rule above failed, so no input that resolves today
|
||||
changes its answer. The prefix must be a PARENT or an exact ALIAS KEY,
|
||||
never a fuzzy hit: letting "fab" through would send "Fab Detergent" to
|
||||
Parle via the alias "parle fab".
|
||||
"""
|
||||
words = key.split()
|
||||
known = set(BRAND_ALIASES) | set(BRAND_ALIASES.values())
|
||||
for n in range(len(words) - 1, 0, -1):
|
||||
candidate = " ".join(words[:n])
|
||||
if candidate in known:
|
||||
return candidate
|
||||
return None
|
||||
|
||||
|
||||
def get_known_sub_brands(brand: str) -> list[str]:
|
||||
"""Return known sub-brand/product names for a given brand from BRAND_ALIASES.
|
||||
|
||||
@@ -291,3 +480,124 @@ def get_known_sub_brands(brand: str) -> list[str]:
|
||||
known.append(rest)
|
||||
seen.add(rest)
|
||||
return known
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# FSSAI licences (stage 1 of the store-catalog pipeline)
|
||||
# ---------------------------------------------------------------------------
|
||||
# Ported from the sibling Universal_Catalog_Barcode_Enrichment project. Purely
|
||||
# additive: BRAND_ALIASES and resolve_parent_brand are untouched.
|
||||
#
|
||||
# Food brands only. A miss means "not a food brand" (P&G, Colgate-Palmolive,
|
||||
# J&J, Reckitt, Godrej) just as much as it means "unknown", so callers must
|
||||
# treat None as "leave the column empty", never as an error.
|
||||
FSSAI_LICENSES: Dict[str, str] = {
|
||||
"britannia": "10012022000103",
|
||||
"pepsico": "10012031000047",
|
||||
"amul": "10012021000243",
|
||||
"cadbury": "10014022002711",
|
||||
"hindustan unilever": "10012022000217",
|
||||
"nestle": "10012011000168",
|
||||
"itc": "10018042000305",
|
||||
"coca-cola": "10012042000424",
|
||||
"tata": "12414003000511",
|
||||
"parle": "10012022000046",
|
||||
"marico": "10012022000258",
|
||||
"dabur": "10012011000084",
|
||||
"hatsun": "10012042000071",
|
||||
"milky mist": "10017042003191",
|
||||
"aachi": "10014042000577",
|
||||
"sakthi": "10012042000300",
|
||||
"kaleesuwari": "10012042000302",
|
||||
"idhayam": "10012042000109",
|
||||
"cavinkare": "10013042000366",
|
||||
"naga": "10012042000192",
|
||||
"manna": "10014042000169",
|
||||
"grb": "10012042000148",
|
||||
"anil": "10012042000213",
|
||||
"lion dates": "10012042000244",
|
||||
"brooke bond": "10013022001897",
|
||||
"mother dairy": "10012011000015",
|
||||
# Keyed on BOTH spellings deliberately. get_fssai_license() resolves to the
|
||||
# canonical parent first, so once "haldiram" aliases to "haldirams" the
|
||||
# lookup arrives as "haldirams" - and with only the singular key here it
|
||||
# would return None and every future Haldiram row would ship with no
|
||||
# licence at all. A silent loss, since a blank licence is a legitimate
|
||||
# outcome elsewhere and nothing would flag it.
|
||||
"haldiram": "10012011000140",
|
||||
"haldirams": "10012011000140",
|
||||
"fortune": "10012021000071",
|
||||
"paper boat": "10012043000083",
|
||||
"bisk farm": "10012031000012",
|
||||
"mtr": "10012043000058",
|
||||
"everest": "10012022000526",
|
||||
"mdh": "10012011000062",
|
||||
}
|
||||
|
||||
|
||||
def get_fssai_license(brand: str) -> Optional[str]:
|
||||
"""Return the 14-digit FSSAI licence number for a food brand, or None.
|
||||
|
||||
Resolves to the canonical parent first, so sub-brands (e.g. 'hul lux')
|
||||
inherit the parent's licence and every row in a brand table agrees.
|
||||
"""
|
||||
canonical = resolve_parent_brand(brand).lower().strip()
|
||||
return FSSAI_LICENSES.get(canonical)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Brand logos
|
||||
# ---------------------------------------------------------------------------
|
||||
# A curated logo for the brand card, used IN PREFERENCE to a sampled product
|
||||
# photo. Same contract as FSSAI_LICENSES above: keyed on the canonical parent in
|
||||
# lowercase, read through an accessor that resolves aliases first.
|
||||
#
|
||||
# Why a curated map rather than picking a product image: for some brands there
|
||||
# is no usable product image at all. Everest, Haldirams, MDH and Naga each carry
|
||||
# nothing but URLs into an S3 bucket that 404s on every one, so the card renders
|
||||
# its initials monogram no matter which row is sampled.
|
||||
#
|
||||
# Deliberately sparse. A miss is the normal case and costs nothing - the card
|
||||
# falls through to a product image, then to the monogram. Only add a brand here
|
||||
# when the product-image path genuinely cannot serve it.
|
||||
#
|
||||
# Two rules for values, both load-bearing:
|
||||
# * https ONLY. The site is served over https and an http:// image is blocked
|
||||
# as mixed content - which is half of the bug this map was added to fix.
|
||||
# * The URL must be hot-linkable and stable. Wikimedia is used here because it
|
||||
# is both, and correctly licensed.
|
||||
# test_brand_registry.py asserts both, and that every key is its own canonical
|
||||
# parent - without that check a key the alias map rewrites is silently dead.
|
||||
BRAND_LOGOS: Dict[str, str] = {
|
||||
# Keyed on the PLURAL only. Unlike FSSAI_LICENSES, which keys both
|
||||
# spellings, a "haldiram" key here would be unreachable: get_brand_logo
|
||||
# resolves through resolve_parent_brand first, and that maps the singular
|
||||
# to "haldirams" before the lookup ever happens. Both spellings still work
|
||||
# for callers - the alias is what makes them work.
|
||||
"haldirams": (
|
||||
"https://upload.wikimedia.org/wikipedia/en/thumb/9/91/"
|
||||
"Haldiram%27s_2024_Logo.svg/500px-Haldiram%27s_2024_Logo.svg.png"
|
||||
),
|
||||
"mdh": "https://upload.wikimedia.org/wikipedia/commons/5/5b/MDH_spices_logo.png",
|
||||
|
||||
# --- Slots, deliberately empty ------------------------------------------
|
||||
# No free, stable logo source was found for these. Everest has a Wikipedia
|
||||
# article but no page image; the others have no Wikipedia or Wikimedia
|
||||
# Commons page at all. They are served by repaired product images instead
|
||||
# (scripts/repair_brand_images.py). Fill a slot in if you source a logo.
|
||||
#
|
||||
# "everest": "",
|
||||
# "naga": "",
|
||||
# "kaleesuwari": "",
|
||||
# "colin": "",
|
||||
}
|
||||
|
||||
|
||||
def get_brand_logo(brand: str) -> Optional[str]:
|
||||
"""Return the curated logo URL for a brand, or None.
|
||||
|
||||
None is the normal answer for almost every brand and means "use a product
|
||||
image instead", never an error - the same contract get_fssai_license has.
|
||||
"""
|
||||
canonical = resolve_parent_brand(brand).lower().strip()
|
||||
return BRAND_LOGOS.get(canonical)
|
||||
|
||||
574
app/services/brand_store.py
Normal file
574
app/services/brand_store.py
Normal file
@@ -0,0 +1,574 @@
|
||||
"""
|
||||
Read a brand's real catalogue from the brand's own shop.
|
||||
|
||||
WHY THIS FILE EXISTS
|
||||
--------------------
|
||||
`off_bulk.fetch_brand_corpus()` is the only thing in this codebase that can
|
||||
ENUMERATE a brand's products. Everything else - `retail_presence`,
|
||||
`sku_service`, `manufacturer_site` - is a per-product VERIFIER: it needs a
|
||||
product name as input and answers yes or no. So for a brand Open Food Facts has
|
||||
never heard of there is nothing to verify, and the catalogue is whatever
|
||||
`qwen2.5:1.5b` invents.
|
||||
|
||||
Measured 2026-09-10, Open Food Facts hits: Udhaiyam 0, Tenali Double Horse 0,
|
||||
Gopuram 0, Double Horse 14. (An earlier note here said Naga's two rows were a
|
||||
UK / Bangladeshi pickle brand - that was the old free-text `brands:Naga`
|
||||
Search-a-licious query matching "Mr Naga" and "Bombay Naga Jhal". Under the
|
||||
exact `brands_tags=naga` filter the v2 API returns two real Naga Limited rows,
|
||||
Sooji 500 g and Maida 500 g, both on the 890 GS1 prefix - see off_bulk.py.)
|
||||
Regional South Indian brands are still thin in that database.
|
||||
|
||||
They are, however, on their own shop, and modern storefronts publish their
|
||||
whole catalogue as structured JSON:
|
||||
|
||||
doublehorse.in Shopify 91 products in one request
|
||||
gopuramproducts.com WooCommerce 100+
|
||||
aachifoods.com Shopify 123 (Open Food Facts has 61)
|
||||
shop.theanilgroup.com Shopify 52 (Open Food Facts has 8)
|
||||
|
||||
and each record carries what the pipeline needs, from the manufacturer itself:
|
||||
|
||||
title Soan Papdi Ghee
|
||||
product_type Snacks vendor Aachifoods
|
||||
variants[0] title='200g' price=76.00 sku='8904209319340' grams=215
|
||||
images[0] https://cdn.shopify.com/.../ghee-soan-papdi.webp
|
||||
|
||||
That SKU passes `validators.validate_barcode` - a real EAN-13 on the Indian 890
|
||||
GS1 prefix. Name, category, pack size, price, images and a barcode, none of it
|
||||
guessed.
|
||||
|
||||
THE RISK THIS MODULE CARRIES, AND THE GUARD ON IT
|
||||
--------------------------------------------------
|
||||
Every other source here verifies one product at a time, so a mistake costs one
|
||||
bad row. This one ASSERTS a hundred products at once, so pointing it at the
|
||||
wrong site imports a hundred bad rows under a real brand's name - and they would
|
||||
look impeccable, because they are a real catalogue, just somebody else's.
|
||||
|
||||
That is not hypothetical. `retail_presence.resolve_brand_domain("Naga")`
|
||||
returns `cityofnagacebu.gov.ph`, the website of Naga City in the Philippines.
|
||||
|
||||
So: the domain comes from a human-curated map (`brand_registry.
|
||||
BRAND_STORE_DOMAINS`), and the catalogue is additionally checked against the
|
||||
brand before any of it is accepted - see `_store_identity_ok`. A catalogue that
|
||||
fails is rejected WHOLE. Half a foreign catalogue is not better than all of it.
|
||||
|
||||
STRUCTURED ENDPOINTS ONLY
|
||||
-------------------------
|
||||
Shopify's `products.json`, WooCommerce's Store API, and failing those a product
|
||||
sitemap plus each page's JSON-LD. No HTML scraping and no Playwright: these are
|
||||
documented, stable, paginated interfaces, and a brand that offers none of them
|
||||
is better reported as "no store catalogue" than guessed at.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional, Sequence
|
||||
|
||||
import requests
|
||||
|
||||
from app.services.category_units import COUNT_UNITS, VOLUME_UNITS, WEIGHT_UNITS, parse_unit
|
||||
from app.services.enrichment.barcode.matching import brand_matches
|
||||
from app.services.enrichment.barcode.retry import with_retry
|
||||
from app.services.enrichment.barcode.validators import validate_barcode
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# The same identifying User-Agent off_bulk uses, deliberately NOT the
|
||||
# browser-spoofing strings elsewhere in this codebase. These are small
|
||||
# companies' own websites rather than marketplaces, and a crawler reading them
|
||||
# should say who it is and how to be reached.
|
||||
USER_AGENT = "BrandCatalogRAG/1.0 (suriya@tenext.in)"
|
||||
|
||||
# Same figure off_bulk uses between pages.
|
||||
PAUSE_SECONDS = 2.0
|
||||
|
||||
_BACKEND_DIR = Path(__file__).resolve().parent.parent.parent
|
||||
CACHE_DIR = _BACKEND_DIR / "data" / "cache" / "brand_store"
|
||||
|
||||
TIMEOUT_SECONDS = 25
|
||||
SHOPIFY_PAGE_SIZE = 250
|
||||
WOO_PAGE_SIZE = 100
|
||||
MAX_PAGES = 12 # 3000 Shopify products; no FMCG brand here is close
|
||||
|
||||
# Units a pack size may legitimately carry - the same set brand_discovery uses,
|
||||
# imported from category_units rather than from brand_discovery, because
|
||||
# brand_discovery imports THIS module and the reverse would close a cycle.
|
||||
_SIZE_UNITS = WEIGHT_UNITS | VOLUME_UNITS | COUNT_UNITS
|
||||
|
||||
# Shopify's placeholder when a product has no real variants.
|
||||
_NO_VARIANT = {"default title", "default", ""}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# HTTP
|
||||
# ---------------------------------------------------------------------------
|
||||
@with_retry(max_attempts=3, min_wait=2.0, max_wait=10.0)
|
||||
def _get(url: str, params: Optional[Dict[str, Any]] = None) -> requests.Response:
|
||||
return requests.get(
|
||||
url, params=params or {},
|
||||
headers={"User-Agent": USER_AGENT, "Accept": "application/json, */*"},
|
||||
timeout=TIMEOUT_SECONDS,
|
||||
)
|
||||
|
||||
|
||||
def _json_or_none(resp: requests.Response) -> Optional[Any]:
|
||||
"""Parse a JSON body, tolerating a UTF-8 BOM.
|
||||
|
||||
WooCommerce really does serve one: gopuramproducts.com's Store API response
|
||||
begins with EF BB BF, and `resp.json()` raises
|
||||
"Unexpected UTF-8 BOM (decode using utf-8-sig)". `brand_sync._read_catalog`
|
||||
already reads seed files as utf-8-sig for the same reason.
|
||||
"""
|
||||
if resp.status_code != 200:
|
||||
return None
|
||||
content_type = (resp.headers.get("content-type") or "").lower()
|
||||
if "json" not in content_type:
|
||||
return None
|
||||
try:
|
||||
return json.loads(resp.content.decode("utf-8-sig"))
|
||||
except Exception as e: # noqa: BLE001 - a malformed body is "no catalogue"
|
||||
logger.debug("Unparseable JSON from %s: %s", resp.url, e)
|
||||
return None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Sizes
|
||||
# ---------------------------------------------------------------------------
|
||||
def canonical_size(raw: Optional[str]) -> Optional[str]:
|
||||
"""A pack size in the one form the pipeline hashes consistently.
|
||||
|
||||
Mirrors `brand_discovery._canonical_size` exactly - "200 g" and "200g" are
|
||||
the same pack, but `build_image_id` slugifies them differently and would
|
||||
make two permanent rows. Returns None for anything that is not a size: a
|
||||
bare count, "Default Title", a colour.
|
||||
"""
|
||||
if not raw or not str(raw).strip():
|
||||
return None
|
||||
value, unit = parse_unit(str(raw))
|
||||
if value is None or not unit or unit not in _SIZE_UNITS:
|
||||
return None
|
||||
number = str(int(value)) if float(value).is_integer() else str(value)
|
||||
return f"{number}{unit}"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Identity - the guard that matters
|
||||
# ---------------------------------------------------------------------------
|
||||
def _store_identity_ok(brand: str, domain: str, vendors: Sequence[str],
|
||||
aliases: Optional[Sequence[str]] = None) -> bool:
|
||||
"""Does this catalogue actually belong to `brand`?
|
||||
|
||||
Accepts on either signal, because neither alone is reliable:
|
||||
|
||||
- the DOMAIN names the brand (`doublehorse.in` for Double Horse), or
|
||||
- the catalogue's own `vendor` field names it (Shopify's "Aachifoods").
|
||||
|
||||
Some stores leave `vendor` as the shop's theme name or blank, so requiring
|
||||
it would reject good catalogues; and a brand can trade on a domain that
|
||||
does not contain its name, so requiring that would too.
|
||||
|
||||
What this DOES stop is the case it was written for: a catalogue fetched
|
||||
from a site belonging to neither the brand's domain nor its name, which is
|
||||
how `cityofnagacebu.gov.ph` would otherwise have become Naga's product
|
||||
list.
|
||||
"""
|
||||
aliases = list(aliases or [])
|
||||
stem = (domain or "").split(".")[0].lower()
|
||||
domain_words = set(re.split(r"[^a-z0-9]+", stem))
|
||||
brand_words = [w for w in re.split(r"[^a-z0-9]+", (brand or "").lower()) if len(w) > 2]
|
||||
if brand_words and all(w in domain_words for w in brand_words):
|
||||
return True
|
||||
|
||||
# PREFIX, NOT SUBSTRING. A brand's own domain is the brand followed by a
|
||||
# qualifier - "gopuramproducts", "udhaiyamdhall", "doublehorse". A domain
|
||||
# that merely CONTAINS the brand somewhere in the middle is a different
|
||||
# organisation that happens to share a word:
|
||||
#
|
||||
# "gopuramproducts".startswith("gopuram") -> True, correct
|
||||
# "cityofnagacebu".startswith("naga") -> False, correct
|
||||
# "cityofnagacebu" contains "naga" -> True, THE BUG
|
||||
#
|
||||
# This is the whole reason `_pick_brand_domain` handed back the website of
|
||||
# Naga City in the Philippines as a Tamil Nadu food brand's catalogue.
|
||||
if brand_words and stem.startswith("".join(brand_words)):
|
||||
return True
|
||||
for vendor in vendors:
|
||||
if vendor and brand_matches(vendor, brand, aliases):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Shopify
|
||||
# ---------------------------------------------------------------------------
|
||||
def fetch_shopify(domain: str) -> Optional[List[Dict[str, Any]]]:
|
||||
"""Raw Shopify products, or None when this is not a Shopify store."""
|
||||
out: List[Dict[str, Any]] = []
|
||||
for page in range(1, MAX_PAGES + 1):
|
||||
try:
|
||||
resp = _get(f"https://{domain}/products.json",
|
||||
{"limit": SHOPIFY_PAGE_SIZE, "page": page})
|
||||
except Exception as e: # noqa: BLE001 - unreachable is not a verdict
|
||||
logger.debug("Shopify fetch failed for %s p%d: %s", domain, page, e)
|
||||
return out or None
|
||||
data = _json_or_none(resp)
|
||||
if not isinstance(data, dict) or "products" not in data:
|
||||
return out or None
|
||||
products = data.get("products") or []
|
||||
if not products:
|
||||
break
|
||||
out.extend(products)
|
||||
if len(products) < SHOPIFY_PAGE_SIZE:
|
||||
break
|
||||
time.sleep(PAUSE_SECONDS)
|
||||
return out or None
|
||||
|
||||
|
||||
def _shopify_candidates(product: Dict[str, Any]) -> List[Dict[str, Any]]:
|
||||
title = str(product.get("title") or "").strip()
|
||||
if not title:
|
||||
return []
|
||||
category = str(product.get("product_type") or "").strip() or None
|
||||
images = [
|
||||
str(i.get("src")) for i in (product.get("images") or [])
|
||||
if isinstance(i, dict) and i.get("src")
|
||||
]
|
||||
variants = product.get("variants") or []
|
||||
|
||||
rows: List[Dict[str, Any]] = []
|
||||
for variant in variants:
|
||||
if not isinstance(variant, dict):
|
||||
continue
|
||||
raw_size = variant.get("title")
|
||||
if str(raw_size or "").strip().lower() in _NO_VARIANT:
|
||||
# No real variant axis. The size, if any, is in the product name,
|
||||
# and `_resolve_sizes` reads a title's own size first anyway.
|
||||
raw_size = variant.get("option1")
|
||||
size = canonical_size(raw_size)
|
||||
|
||||
# The SKU is a barcode only when it IS one. Shopify shops put anything
|
||||
# here - Aachi puts a real EAN-13 ("8904209319340"), Gopuram leaves it
|
||||
# blank, others use an internal code. `validate_barcode` decides, so a
|
||||
# shop's internal numbering never reaches the barcode column.
|
||||
sku = str(variant.get("sku") or "").strip()
|
||||
barcode = validate_barcode(sku) if sku else None
|
||||
if not barcode:
|
||||
raw_barcode = str(variant.get("barcode") or "").strip()
|
||||
barcode = validate_barcode(raw_barcode) if raw_barcode else None
|
||||
|
||||
rows.append({
|
||||
"title": title,
|
||||
"category": category,
|
||||
"size": size,
|
||||
"barcode": barcode,
|
||||
"price": _decimal_or_none(variant.get("price")),
|
||||
"image_url": images[0] if images else None,
|
||||
"image_urls": images,
|
||||
"vendor": str(product.get("vendor") or "").strip() or None,
|
||||
"source": "store",
|
||||
})
|
||||
if not rows:
|
||||
rows.append({
|
||||
"title": title, "category": category, "size": None, "barcode": None,
|
||||
"price": None, "image_url": images[0] if images else None,
|
||||
"image_urls": images,
|
||||
"vendor": str(product.get("vendor") or "").strip() or None,
|
||||
"source": "store",
|
||||
})
|
||||
return rows
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# WooCommerce Store API
|
||||
# ---------------------------------------------------------------------------
|
||||
def fetch_woocommerce(domain: str) -> Optional[List[Dict[str, Any]]]:
|
||||
"""Raw WooCommerce Store API products, or None when unavailable.
|
||||
|
||||
udhaiyamdhall.com is WooCommerce but answers this endpoint with 403, which
|
||||
is why the sitemap tier below exists.
|
||||
"""
|
||||
out: List[Dict[str, Any]] = []
|
||||
for page in range(1, MAX_PAGES + 1):
|
||||
try:
|
||||
resp = _get(f"https://{domain}/wp-json/wc/store/products",
|
||||
{"per_page": WOO_PAGE_SIZE, "page": page})
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.debug("Woo fetch failed for %s p%d: %s", domain, page, e)
|
||||
return out or None
|
||||
data = _json_or_none(resp)
|
||||
if not isinstance(data, list):
|
||||
return out or None
|
||||
if not data:
|
||||
break
|
||||
out.extend(data)
|
||||
if len(data) < WOO_PAGE_SIZE:
|
||||
break
|
||||
time.sleep(PAUSE_SECONDS)
|
||||
return out or None
|
||||
|
||||
|
||||
def _woo_price(prices: Dict[str, Any]) -> Optional[float]:
|
||||
"""WooCommerce reports MINOR UNITS.
|
||||
|
||||
gopuramproducts.com returns `price: "5500"` with `currency_minor_unit: 2`,
|
||||
which is Rs55.00 and not Rs5,500. Read straight, every price on the site is
|
||||
a hundred times too large - and `product_validator.validate_price_range`
|
||||
would then reject the row as an implausible price, so the failure would
|
||||
surface as "this brand's products are all wrong" rather than as a units bug.
|
||||
"""
|
||||
raw = prices.get("price")
|
||||
if raw is None or str(raw).strip() == "":
|
||||
return None
|
||||
try:
|
||||
minor = int(prices.get("currency_minor_unit", 2))
|
||||
return float(raw) / (10 ** minor)
|
||||
except Exception: # noqa: BLE001
|
||||
return None
|
||||
|
||||
|
||||
def _woo_candidates(product: Dict[str, Any]) -> List[Dict[str, Any]]:
|
||||
title = str(product.get("name") or "").strip()
|
||||
if not title:
|
||||
return []
|
||||
categories = [
|
||||
str(c.get("name")).strip() for c in (product.get("categories") or [])
|
||||
if isinstance(c, dict) and c.get("name")
|
||||
]
|
||||
images = [
|
||||
str(i.get("src")) for i in (product.get("images") or [])
|
||||
if isinstance(i, dict) and i.get("src")
|
||||
]
|
||||
sku = str(product.get("sku") or "").strip()
|
||||
return [{
|
||||
"title": title,
|
||||
"category": categories[0] if categories else None,
|
||||
"size": canonical_size(title),
|
||||
"barcode": validate_barcode(sku) if sku else None,
|
||||
"price": _woo_price(product.get("prices") or {}),
|
||||
"image_url": images[0] if images else None,
|
||||
"image_urls": images,
|
||||
"vendor": None,
|
||||
"source": "store",
|
||||
}]
|
||||
|
||||
|
||||
def _decimal_or_none(raw: Any) -> Optional[float]:
|
||||
if raw is None or str(raw).strip() == "":
|
||||
return None
|
||||
try:
|
||||
return float(raw)
|
||||
except Exception: # noqa: BLE001
|
||||
return None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tier 3 - product sitemap, for sites that publish no catalogue API
|
||||
# ---------------------------------------------------------------------------
|
||||
# udhaiyamdhall.com is WooCommerce but answers the Store API with 403, and its
|
||||
# product pages carry only a BreadcrumbList in JSON-LD - no Product node, no
|
||||
# og: tags. What it does have is `wp-sitemap-posts-product-1.xml` listing 32
|
||||
# products, each page with an <h1> and a gallery image.
|
||||
#
|
||||
# So this tier yields NAMES AND IMAGES, AND NOTHING ELSE. No price, no pack
|
||||
# size, no barcode - because the site does not state them, and a catalogue
|
||||
# source that guesses is the exact problem this whole area exists to fix. The
|
||||
# names are real and the photographs are the brand's own, which solves
|
||||
# enumeration and solves the image problem; sizes have to come from somewhere
|
||||
# that actually knows them.
|
||||
_SITEMAP_CANDIDATES = ("/sitemap.xml", "/wp-sitemap.xml", "/sitemap_index.xml")
|
||||
MAX_SITEMAP_PAGES = 200
|
||||
|
||||
|
||||
def _sitemap_locs(xml: str) -> List[str]:
|
||||
return re.findall(r"<loc>\s*(.*?)\s*</loc>", xml or "", re.IGNORECASE | re.DOTALL)
|
||||
|
||||
|
||||
def fetch_sitemap(domain: str) -> Optional[List[Dict[str, Any]]]:
|
||||
"""Product names and images from a site's product sitemap, or None."""
|
||||
try:
|
||||
from bs4 import BeautifulSoup
|
||||
except ImportError: # pragma: no cover - declared in requirements.txt
|
||||
logger.debug("Sitemap tier unavailable: beautifulsoup4 not installed")
|
||||
return None
|
||||
|
||||
product_urls: List[str] = []
|
||||
for path in _SITEMAP_CANDIDATES:
|
||||
try:
|
||||
resp = _get(f"https://{domain}{path}")
|
||||
except Exception: # noqa: BLE001
|
||||
continue
|
||||
if resp.status_code != 200 or "xml" not in (resp.headers.get("content-type") or ""):
|
||||
continue
|
||||
locs = _sitemap_locs(resp.text)
|
||||
# A sitemap index points at sub-sitemaps; take only the product one.
|
||||
for loc in locs:
|
||||
if "product" not in loc.lower() or "categor" in loc.lower():
|
||||
continue
|
||||
if loc.lower().endswith(".xml"):
|
||||
try:
|
||||
sub = _get(loc)
|
||||
except Exception: # noqa: BLE001
|
||||
continue
|
||||
if sub.status_code == 200:
|
||||
product_urls.extend(
|
||||
u for u in _sitemap_locs(sub.text) if "/product/" in u.lower()
|
||||
)
|
||||
elif "/product/" in loc.lower():
|
||||
product_urls.append(loc)
|
||||
if product_urls:
|
||||
break
|
||||
|
||||
product_urls = list(dict.fromkeys(product_urls))[:MAX_SITEMAP_PAGES]
|
||||
if not product_urls:
|
||||
return None
|
||||
|
||||
out: List[Dict[str, Any]] = []
|
||||
for url in product_urls:
|
||||
try:
|
||||
page = _get(url)
|
||||
except Exception: # noqa: BLE001 - one dead page cannot stop the sweep
|
||||
continue
|
||||
if page.status_code != 200:
|
||||
continue
|
||||
soup = BeautifulSoup(page.text, "lxml")
|
||||
heading = soup.find("h1")
|
||||
title = heading.get_text(strip=True) if heading else ""
|
||||
if not title:
|
||||
continue
|
||||
image = None
|
||||
node = soup.select_one(".woocommerce-product-gallery img, .wp-post-image, "
|
||||
"meta[property='og:image']")
|
||||
if node is not None:
|
||||
image = node.get("src") or node.get("data-src") or node.get("content")
|
||||
out.append({
|
||||
"title": title,
|
||||
"category": None,
|
||||
# Left None deliberately - this site states neither, and inventing
|
||||
# them is the failure mode, not the fallback.
|
||||
"size": None,
|
||||
"barcode": None,
|
||||
"price": None,
|
||||
"image_url": image,
|
||||
"image_urls": [image] if image else [],
|
||||
"vendor": None,
|
||||
"source": "store",
|
||||
})
|
||||
time.sleep(1.0)
|
||||
return out or None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Public entry point
|
||||
# ---------------------------------------------------------------------------
|
||||
def _cache_path(brand: str) -> Path:
|
||||
slug = re.sub(r"[^a-z0-9]+", "_", (brand or "").lower()).strip("_")
|
||||
return CACHE_DIR / f"{slug}.json"
|
||||
|
||||
|
||||
def _write_cache(path: Path, payload: Dict[str, Any]) -> None:
|
||||
"""Atomic write, the same .tmp + os.replace off_bulk uses - a half-written
|
||||
catalogue read by the next run is worse than no cache."""
|
||||
try:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
tmp = path.with_suffix(".tmp")
|
||||
tmp.write_text(json.dumps(payload, indent=2), encoding="utf-8")
|
||||
os.replace(tmp, path)
|
||||
except Exception as e: # noqa: BLE001 - caching is best effort
|
||||
logger.debug("Could not cache brand store catalogue at %s: %s", path, e)
|
||||
|
||||
|
||||
def read_cached(brand: str) -> Optional[List[Dict[str, Any]]]:
|
||||
"""The cached catalogue for a brand, or None. Never raises."""
|
||||
path = _cache_path(brand)
|
||||
if not path.exists():
|
||||
return None
|
||||
try:
|
||||
data = json.loads(path.read_text(encoding="utf-8-sig"))
|
||||
return list(data.get("products") or [])
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.debug("Unreadable brand store cache %s: %s", path, e)
|
||||
return None
|
||||
|
||||
|
||||
def fetch_store_catalogue(brand: str, domain: Optional[str], *,
|
||||
live: bool = False, refresh: bool = False,
|
||||
trusted_domain: bool = False,
|
||||
aliases: Optional[Sequence[str]] = None,
|
||||
) -> Optional[List[Dict[str, Any]]]:
|
||||
"""The brand's own catalogue as candidate dicts, or None if there isn't one.
|
||||
|
||||
`live` defaults to FALSE, matching `retail_presence.check_listing`: the
|
||||
runtime path reads the cache and the backfill script does the fetching, so
|
||||
a discovery preview never blocks on somebody's storefront being slow.
|
||||
|
||||
None means "no catalogue" - not reachable, not a supported platform, or
|
||||
rejected by the identity guard. An empty list is not returned; a store with
|
||||
zero products is indistinguishable from no store and is reported the same.
|
||||
"""
|
||||
if not refresh:
|
||||
cached = read_cached(brand)
|
||||
if cached is not None:
|
||||
return cached or None
|
||||
if not live or not domain:
|
||||
return None
|
||||
|
||||
raw = fetch_shopify(domain)
|
||||
platform = "shopify"
|
||||
to_candidates = _shopify_candidates
|
||||
if raw is None:
|
||||
raw = fetch_woocommerce(domain)
|
||||
platform = "woocommerce"
|
||||
to_candidates = _woo_candidates
|
||||
if raw is None:
|
||||
# Last tier: names and images off the product sitemap. See fetch_sitemap
|
||||
# for why it yields nothing else.
|
||||
raw = fetch_sitemap(domain)
|
||||
platform = "sitemap"
|
||||
to_candidates = lambda row: [row] # noqa: E731 - already candidates
|
||||
if raw is None:
|
||||
logger.info("No structured catalogue at %s for %s", domain, brand)
|
||||
return None
|
||||
|
||||
candidates: List[Dict[str, Any]] = []
|
||||
for product in raw:
|
||||
if isinstance(product, dict):
|
||||
candidates.extend(to_candidates(product))
|
||||
|
||||
vendors = sorted({c.get("vendor") for c in candidates if c.get("vendor")})
|
||||
# A CURATED DOMAIN IS ALREADY VERIFIED, BY A PERSON.
|
||||
#
|
||||
# The heuristic below cannot recognise every legitimate shape a brand's
|
||||
# domain takes - "theanilgroup.com" is Anil's, and neither starts with
|
||||
# "anil" nor contains it as a whole token - so applying it to a
|
||||
# hand-checked mapping would reject good catalogues for looking unusual.
|
||||
# It runs only on domains that came from automatic resolution, which is
|
||||
# the path that produced cityofnagacebu.gov.ph.
|
||||
if not trusted_domain and not _store_identity_ok(brand, domain, vendors, aliases):
|
||||
# REJECTED WHOLE, on purpose. This is the failure that would otherwise
|
||||
# put a hundred of somebody else's products under this brand's name,
|
||||
# each of them a real product and none of them theirs.
|
||||
logger.warning(
|
||||
"Rejecting catalogue at %s for brand %r: neither the domain nor "
|
||||
"its vendors (%s) identify this brand",
|
||||
domain, brand, ", ".join(vendors) or "none stated",
|
||||
)
|
||||
return None
|
||||
|
||||
for c in candidates:
|
||||
c.pop("vendor", None)
|
||||
|
||||
logger.info("%s: %d products from %s (%s)", brand, len(candidates), domain, platform)
|
||||
_write_cache(_cache_path(brand), {
|
||||
"brand": brand,
|
||||
"domain": domain,
|
||||
"platform": platform,
|
||||
"fetched_at": time.time(),
|
||||
"fetched_at_human": time.strftime("%Y-%m-%d %H:%M:%S"),
|
||||
"products": candidates,
|
||||
})
|
||||
return candidates or None
|
||||
564
app/services/brand_sync.py
Normal file
564
app/services/brand_sync.py
Normal file
@@ -0,0 +1,564 @@
|
||||
"""
|
||||
Keeps the `brand_*` Postgres tables and `data/seed_catalogs/*.json` in step.
|
||||
|
||||
A brand can enter the system from either side: `POST /api/user/products/add`
|
||||
and `POST /api/catalog/generate` write rows, while a catalog file may be
|
||||
dropped in by hand or shipped in the image. Neither side used to produce the
|
||||
other, so a brand ingested at runtime had no seed file, and a seed file added
|
||||
after first boot was never loaded (the startup auto-seed only runs against a
|
||||
completely empty database).
|
||||
|
||||
This module is the single place that knows the correspondence between the two,
|
||||
and `reconcile_brand_catalogs()` repairs it in both directions.
|
||||
|
||||
The correspondence is *not* the filename. `brand_catalog_p_and_g.json` holds
|
||||
`"brand": "p&g"`, which sanitises to table `brand_p_g`; `brand_catalog_tata.json`
|
||||
resolves through BRAND_ALIASES into `brand_hindustan_unilever`. Everything here
|
||||
therefore indexes files by their `brand` field, exactly as
|
||||
`scripts/seed_sample_data.py` does. Keying on filenames instead would split
|
||||
P&G across two files and let Tata clobber Hindustan Unilever.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
from collections import defaultdict
|
||||
from datetime import date, datetime
|
||||
from decimal import Decimal
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from app.services.brand_registry import resolve_parent_brand
|
||||
from app.services.active_brands import (
|
||||
active_brand_suffixes,
|
||||
is_active_brand,
|
||||
is_active_suffix,
|
||||
)
|
||||
from app.services.vector_store import (
|
||||
_connect,
|
||||
_list_brand_table_suffixes,
|
||||
_sanitize_name,
|
||||
display_name_for_suffix,
|
||||
ensure_brand_schema,
|
||||
get_products_by_brand,
|
||||
invalidate_brand_overview_cache,
|
||||
upsert_brand_products,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# app/services/brand_sync.py -> parents[2] is backend/
|
||||
SEED_DIR = Path(__file__).resolve().parents[2] / "data" / "seed_catalogs"
|
||||
|
||||
# Catalogs for brands outside ACTIVE_BRANDS live in this subdirectory. It is a
|
||||
# plain subfolder rather than a separate tree so the two stay side by side, and
|
||||
# it is NOT matched by SEED_DIR.glob("*.json") - which is what keeps the boot
|
||||
# auto-seed from parsing ~19MB and creating a table for every archived brand.
|
||||
ARCHIVE_DIR_NAME = "archive"
|
||||
|
||||
|
||||
def seed_catalog_paths(seed_dir: Path = SEED_DIR) -> List[Path]:
|
||||
"""Every seed catalog, active directory first, then the archive.
|
||||
|
||||
The archive is searched too, deliberately: re-activating a brand must be a
|
||||
one-line ACTIVE_BRANDS change, so where a file physically sits is
|
||||
presentation, not policy. Filtering by brand happens in load_seed_catalogs.
|
||||
"""
|
||||
if not seed_dir.exists():
|
||||
return []
|
||||
paths = sorted(seed_dir.glob("*.json"))
|
||||
archive = seed_dir / ARCHIVE_DIR_NAME
|
||||
if archive.is_dir():
|
||||
paths.extend(sorted(archive.glob("*.json")))
|
||||
return paths
|
||||
|
||||
# The column list written by upsert_brand_products, minus `embedding` (handled
|
||||
# separately because it is huge and optional). Keeping these in sync is what
|
||||
# makes an exported file round-trip back through the seeder without losing
|
||||
# hsn/price/barcode/sku data - the failing mode of scripts/export_seed_data.py.
|
||||
#
|
||||
# `nutrition_score` and `health_score` are the exception to the first sentence:
|
||||
# they are brand-table columns that upsert_brand_products deliberately does NOT
|
||||
# write (see the comment on its INSERT). They are exported so a catalog file
|
||||
# carries them, but they do NOT round-trip back into the database - re-seeding
|
||||
# ignores them, and app/services/nutrition_score_sync.py is what restores them
|
||||
# from nutrition_insights, which is their source of truth.
|
||||
#
|
||||
# `nutrients_per_100g` is in the same category as the two scores: mirrored from
|
||||
# nutrition_facts by nutrition_score_sync, exported for readers, never
|
||||
# round-tripped back in.
|
||||
#
|
||||
# The barcode identity/provenance columns and the HSN/GST figures below ARE
|
||||
# round-tripped. They were absent from this tuple for as long as they were
|
||||
# absent from the INSERT, which is why scripts/backfill_barcodes_from_off.py
|
||||
# refuses to call export_brand_to_seed_file() - exporting used to silently
|
||||
# strip the nine barcode keys off every product. Listing them here is what
|
||||
# makes that helper safe to use again.
|
||||
EXPORT_COLUMNS = (
|
||||
"product_name", "title", "description", "category", "image_id",
|
||||
"image_url", "image_urls", "price_range", "size_variants", "providers",
|
||||
"fssai_license", "product_sku", "sku_source", "hsn_code",
|
||||
"final_selling_price", "selling_price", "barcode", "barcode_type",
|
||||
"gtin", "ean13", "upc", "barcode_source", "barcode_verified",
|
||||
"barcode_lookup_status", "barcode_last_updated",
|
||||
"gst_percent", "tax_amount", "hsn_gst_needs_review",
|
||||
"highlights", "nutrients", "search_query", "field_sources",
|
||||
"product_line", "variant",
|
||||
# The validation verdict. Listed here for the same reason the barcode keys
|
||||
# are: an export that strips them turns a re-seed into a silent downgrade.
|
||||
# The upsert assigns these three plainly rather than COALESCEing them, so a
|
||||
# seed file that has lost them clears the verdict on every row it restores -
|
||||
# which is right for a writer that never validated, and wrong for this one,
|
||||
# whose whole job is to echo back rows that already were.
|
||||
"validation_status", "confidence_score", "validation_issues",
|
||||
"nutrition_score", "health_score", "nutrients_per_100g",
|
||||
)
|
||||
|
||||
|
||||
def brand_slug(brand: str) -> str:
|
||||
"""The table suffix a brand name resolves to (e.g. 'ITC' -> 'itc')."""
|
||||
return _sanitize_name(resolve_parent_brand(brand))
|
||||
|
||||
|
||||
def _read_catalog(path: Path) -> Optional[Dict[str, Any]]:
|
||||
"""Parse a seed catalog, returning None for anything that isn't one.
|
||||
|
||||
Files without a brand or products (notably `hsn_gst_master.json`) are not
|
||||
catalogs, and a corrupt file must not take down a startup reconcile.
|
||||
"""
|
||||
try:
|
||||
data = json.loads(path.read_text(encoding="utf-8-sig"))
|
||||
except Exception as e: # noqa: BLE001 - one bad file cannot break the sweep
|
||||
logger.warning("Could not read seed catalog %s: %s", path.name, e)
|
||||
return None
|
||||
if not isinstance(data, dict):
|
||||
return None
|
||||
products = data.get("products")
|
||||
if not isinstance(products, list) or not products:
|
||||
return None
|
||||
brand = data.get("brand") or (products[0].get("brand_name") if isinstance(products[0], dict) else None)
|
||||
if not brand:
|
||||
return None
|
||||
data["brand"] = brand
|
||||
return data
|
||||
|
||||
|
||||
def index_seed_files(seed_dir: Path = SEED_DIR) -> Dict[str, List[Path]]:
|
||||
"""Map each brand table suffix to the seed files that feed it.
|
||||
|
||||
More than one file can feed a suffix - tata.json and hindustan_unilever.json
|
||||
both land in brand_hindustan_unilever - which is why the value is a list.
|
||||
"""
|
||||
index: Dict[str, List[Path]] = defaultdict(list)
|
||||
if not seed_dir.exists():
|
||||
return {}
|
||||
for path in seed_catalog_paths(seed_dir):
|
||||
data = _read_catalog(path)
|
||||
if data is None:
|
||||
continue
|
||||
index[brand_slug(data["brand"])].append(path)
|
||||
return dict(index)
|
||||
|
||||
|
||||
# Resolving a brand to its file means parsing every catalog, and /batch-add
|
||||
# calls that once per product. The mapping only changes when a file is created
|
||||
# or removed, so it is cached and invalidated explicitly on write.
|
||||
_TARGET_CACHE: Dict[str, Path] = {}
|
||||
|
||||
|
||||
def invalidate_seed_index_cache() -> None:
|
||||
_TARGET_CACHE.clear()
|
||||
|
||||
|
||||
def canonical_seed_file(brand: str, index: Optional[Dict[str, List[Path]]] = None) -> Path:
|
||||
"""The file to write for `brand`.
|
||||
|
||||
Prefers a file whose own brand field sanitises to the same slug, so P&G
|
||||
products append to brand_catalog_p_and_g.json rather than creating an
|
||||
orphan brand_catalog_p_g.json, and Hindustan Unilever products never get
|
||||
written into brand_catalog_tata.json.
|
||||
"""
|
||||
slug = brand_slug(brand)
|
||||
if index is None:
|
||||
cached = _TARGET_CACHE.get(slug)
|
||||
if cached is not None:
|
||||
return cached
|
||||
index = index_seed_files()
|
||||
|
||||
resolved = SEED_DIR / f"brand_catalog_{slug}.json"
|
||||
candidates = index.get(slug, [])
|
||||
for path in candidates:
|
||||
data = _read_catalog(path)
|
||||
if data and _sanitize_name(data["brand"]) == slug:
|
||||
resolved = path
|
||||
break
|
||||
else:
|
||||
if candidates:
|
||||
resolved = candidates[0]
|
||||
|
||||
_TARGET_CACHE[slug] = resolved
|
||||
return resolved
|
||||
|
||||
|
||||
def upsert_products_into_catalog_file(brand: str, products: List[Dict[str, Any]]) -> Optional[Path]:
|
||||
"""Merge `products` into the brand's seed catalog, creating it if needed.
|
||||
|
||||
Products are matched on image_id or product_name, so re-adding an existing
|
||||
product updates it in place instead of duplicating. Embeddings are stripped
|
||||
(they are ~384 floats each and the file is meant to stay readable), and the
|
||||
write is atomic so a crash cannot leave a half-written catalog that then
|
||||
fails to parse on the next boot.
|
||||
"""
|
||||
if not products:
|
||||
return None
|
||||
|
||||
SEED_DIR.mkdir(parents=True, exist_ok=True)
|
||||
file_path = canonical_seed_file(brand)
|
||||
|
||||
if file_path.exists():
|
||||
try:
|
||||
data = json.loads(file_path.read_text(encoding="utf-8-sig"))
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.warning("Could not read existing catalog JSON %s: %s", file_path.name, e)
|
||||
data = {"brand": brand, "products": []}
|
||||
else:
|
||||
data = {
|
||||
"brand": brand.lower(),
|
||||
"search_query": f"{brand} products catalog",
|
||||
"generation_timestamp": str(Path(__file__).resolve()),
|
||||
"total_products": 0,
|
||||
"total_images": 0,
|
||||
"products": [],
|
||||
}
|
||||
|
||||
products_list = data.get("products") or []
|
||||
by_image_id = {
|
||||
p.get("image_id"): i for i, p in enumerate(products_list) if p.get("image_id")
|
||||
}
|
||||
by_name = {
|
||||
p.get("product_name"): i for i, p in enumerate(products_list) if p.get("product_name")
|
||||
}
|
||||
|
||||
for product in products:
|
||||
clean = {k: v for k, v in product.items() if k != "embedding"}
|
||||
idx = by_image_id.get(clean.get("image_id"))
|
||||
if idx is None:
|
||||
idx = by_name.get(clean.get("product_name"))
|
||||
if idx is None:
|
||||
products_list.append(clean)
|
||||
if clean.get("image_id"):
|
||||
by_image_id[clean["image_id"]] = len(products_list) - 1
|
||||
if clean.get("product_name"):
|
||||
by_name[clean["product_name"]] = len(products_list) - 1
|
||||
else:
|
||||
products_list[idx] = clean
|
||||
|
||||
data["products"] = products_list
|
||||
data["total_products"] = len(products_list)
|
||||
data["total_images"] = sum(len(p.get("image_urls") or []) for p in products_list)
|
||||
|
||||
payload = json.dumps(data, indent=2, ensure_ascii=False)
|
||||
tmp_path = file_path.with_suffix(".json.tmp")
|
||||
tmp_path.write_text(payload, encoding="utf-8")
|
||||
os.replace(tmp_path, file_path)
|
||||
|
||||
logger.info("✅ Updated JSON seed file '%s' (total products: %d)",
|
||||
file_path.name, data["total_products"])
|
||||
return file_path
|
||||
|
||||
|
||||
def _jsonable(value: Any) -> Any:
|
||||
"""Coerce a psycopg row value into something json.dumps accepts."""
|
||||
if isinstance(value, Decimal):
|
||||
return float(value)
|
||||
# `barcode_last_updated` is a TIMESTAMP column, so psycopg hands back a
|
||||
# datetime, which json.dumps refuses. Emitted as an ISO-8601 string rather
|
||||
# than an epoch float so the seed file stays human-readable; the DB write
|
||||
# path accepts either (see vector_store._epoch_to_timestamp).
|
||||
if isinstance(value, (datetime, date)):
|
||||
return value.isoformat()
|
||||
if isinstance(value, (list, tuple)):
|
||||
return [_jsonable(v) for v in value]
|
||||
# JSONB (field_sources, nutrients_per_100g) arrives as a dict; recurse so
|
||||
# a Decimal nested inside a nutrient block does not break the dump.
|
||||
if isinstance(value, dict):
|
||||
return {k: _jsonable(v) for k, v in value.items()}
|
||||
return value
|
||||
|
||||
|
||||
def export_brand_to_seed_file(brand: str, include_embeddings: bool = True) -> Optional[Path]:
|
||||
"""Write a brand's database rows out to its seed catalog, field-complete.
|
||||
|
||||
Deliberately not scripts/export_seed_data.py, whose _read_products emits
|
||||
only ~10 keys - running that would silently strip hsn_code, prices,
|
||||
barcodes, SKUs and FSSAI numbers out of the existing catalogs.
|
||||
|
||||
Embeddings are carried through when present because that is what lets
|
||||
seed_sample_data.py re-seed without invoking the embedding model.
|
||||
"""
|
||||
rows = get_products_by_brand(brand)
|
||||
if not rows:
|
||||
logger.info("Nothing to export for brand '%s' (no rows)", brand)
|
||||
return None
|
||||
|
||||
display = display_name_for_suffix(brand_slug(brand))
|
||||
products: List[Dict[str, Any]] = []
|
||||
for row in rows:
|
||||
product: Dict[str, Any] = {"brand": display, "brand_name": display}
|
||||
for col in EXPORT_COLUMNS:
|
||||
if col in row:
|
||||
product[col] = _jsonable(row[col])
|
||||
if include_embeddings and row.get("embedding") is not None:
|
||||
try:
|
||||
product["embedding"] = [float(x) for x in row["embedding"]]
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
products.append(product)
|
||||
|
||||
path = upsert_products_into_catalog_file(display, products)
|
||||
if path:
|
||||
logger.info("📤 Exported %d product(s) for '%s' -> %s", len(products), display, path.name)
|
||||
return path
|
||||
|
||||
|
||||
def load_seed_catalogs(seed_dir: Path = SEED_DIR,
|
||||
only: Optional[List[str]] = None) -> Dict[str, List[Dict[str, Any]]]:
|
||||
"""Read seed catalogs and group their products by resolved parent brand.
|
||||
|
||||
Grouping matters: several files can feed one table, and the seeder's
|
||||
stale-row cleanup deletes anything not in the batch it is given. Merging
|
||||
first is what stops tata.json and hindustan_unilever.json erasing each
|
||||
other.
|
||||
"""
|
||||
if not seed_dir.exists():
|
||||
logger.error("Seed directory not found: %s", seed_dir)
|
||||
return {}
|
||||
|
||||
files = seed_catalog_paths(seed_dir)
|
||||
if only:
|
||||
wanted = [w.lower() for w in only]
|
||||
files = [f for f in files if any(w in f.name.lower() for w in wanted)]
|
||||
|
||||
# `only` is an explicit request for named files, so it wins over the
|
||||
# ACTIVE_BRANDS filter - that is how an archived brand can still be
|
||||
# re-ingested deliberately without first editing the config.
|
||||
respect_active = not only
|
||||
|
||||
brand_products: Dict[str, List[Dict[str, Any]]] = defaultdict(list)
|
||||
skipped_inactive = 0
|
||||
for path in files:
|
||||
data = _read_catalog(path)
|
||||
if data is None:
|
||||
logger.warning("Skipping %s - no brand/products found", path.name)
|
||||
continue
|
||||
resolved = resolve_parent_brand(data["brand"])
|
||||
if respect_active and not is_active_brand(resolved):
|
||||
skipped_inactive += 1
|
||||
continue
|
||||
brand_products[resolved].extend(data["products"])
|
||||
logger.info("Read %d products from %s -> resolved brand '%s'",
|
||||
len(data["products"]), path.name, resolved)
|
||||
|
||||
if skipped_inactive:
|
||||
logger.info("Skipped %d seed catalog(s) outside ACTIVE_BRANDS", skipped_inactive)
|
||||
|
||||
return dict(brand_products)
|
||||
|
||||
|
||||
def load_brand_products(brand: str) -> List[Dict[str, Any]]:
|
||||
"""Every seed product belonging to `brand`, resolved properly.
|
||||
|
||||
Use this instead of `load_seed_catalogs(only=[brand])` when you have a
|
||||
BRAND NAME. `only=` is a case-insensitive substring match on the FILE NAME,
|
||||
which is the right thing for `seed_sample_data.py --only amul` but the
|
||||
wrong thing for a brand:
|
||||
|
||||
* "Hindustan Unilever" contains a space; the file is
|
||||
`brand_catalog_hindustan_unilever.json`. The substring never matches
|
||||
and you silently get zero products - no error, just an empty result.
|
||||
* Even with the slug, a filename match misses the other files that feed
|
||||
the same table: `brand_catalog_tata.json` holds 121 Hindustan Unilever
|
||||
products because the "hul tata tea" alias claims it.
|
||||
|
||||
Going through `index_seed_files()` fixes both: it is keyed by the resolved
|
||||
table suffix and its value is the full list of files feeding that table.
|
||||
Archived catalogs are included, so an explicitly requested brand is found
|
||||
whether or not it is currently active.
|
||||
"""
|
||||
index = index_seed_files()
|
||||
products: List[Dict[str, Any]] = []
|
||||
for path in index.get(brand_slug(brand), []):
|
||||
data = _read_catalog(path)
|
||||
if data is None:
|
||||
continue
|
||||
products.extend(data["products"])
|
||||
logger.info("Read %d products from %s for brand '%s'",
|
||||
len(data["products"]), path.name, brand)
|
||||
return products
|
||||
|
||||
|
||||
def seed_brands(brand_products: Dict[str, List[Dict[str, Any]]], cleanup: bool = True) -> int:
|
||||
"""Upsert grouped products into their brand tables. Returns the row count."""
|
||||
total = 0
|
||||
for resolved_brand, all_products in brand_products.items():
|
||||
logger.info("Seeding %d product(s) for brand '%s'", len(all_products), resolved_brand)
|
||||
table = ensure_brand_schema(resolved_brand)
|
||||
if not table:
|
||||
logger.error("Could not create/verify table for brand '%s' - is pgvector reachable?",
|
||||
resolved_brand)
|
||||
continue
|
||||
upsert_brand_products(resolved_brand, all_products, cleanup=cleanup)
|
||||
logger.info("Seeded %d products for brand '%s' (table=%s)",
|
||||
len(all_products), resolved_brand, table)
|
||||
total += len(all_products)
|
||||
return total
|
||||
|
||||
|
||||
def _db_brand_counts() -> Dict[str, int]:
|
||||
"""{table suffix: row count} for every brand_* table, on one connection."""
|
||||
conn = _connect()
|
||||
if not conn:
|
||||
return {}
|
||||
counts: Dict[str, int] = {}
|
||||
try:
|
||||
with conn, conn.cursor() as cur:
|
||||
for suffix in sorted(set(_list_brand_table_suffixes(cur))):
|
||||
try:
|
||||
cur.execute(f"SELECT COUNT(*) FROM brand_{suffix}")
|
||||
row = cur.fetchone()
|
||||
counts[suffix] = int(row[0]) if row else 0
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.warning("Could not count brand_%s: %s", suffix, e)
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.error("Failed to enumerate brand tables: %s", e)
|
||||
finally:
|
||||
conn.close()
|
||||
return counts
|
||||
|
||||
|
||||
def _detect_collisions(seed_dir: Path = SEED_DIR) -> List[Dict[str, Any]]:
|
||||
"""Distinct parent brands that sanitise onto one table (e.g. 'P&G' vs 'P G').
|
||||
|
||||
Grouped by *resolved parent*, not by the raw brand field: tata.json and
|
||||
hindustan_unilever.json share a table because BRAND_ALIASES deliberately
|
||||
merges them, which is not a collision. A genuine one is two unrelated
|
||||
parents whose names differ only in characters _sanitize_name strips.
|
||||
|
||||
Reported rather than repaired - renaming a live table is a one-way door,
|
||||
and nothing collides today. This is here so it surfaces the day it does.
|
||||
"""
|
||||
by_slug: Dict[str, set] = defaultdict(set)
|
||||
if not seed_dir.exists():
|
||||
return []
|
||||
for path in seed_catalog_paths(seed_dir):
|
||||
data = _read_catalog(path)
|
||||
if data is None:
|
||||
continue
|
||||
parent = resolve_parent_brand(data["brand"]).strip().lower()
|
||||
by_slug[_sanitize_name(parent)].add(parent)
|
||||
return [
|
||||
{"slug": slug, "brands": sorted(names)}
|
||||
for slug, names in sorted(by_slug.items()) if len(names) > 1
|
||||
]
|
||||
|
||||
|
||||
def reconcile_brand_catalogs(dry_run: bool = False) -> Dict[str, Any]:
|
||||
"""Repair the table <-> seed-file correspondence in both directions.
|
||||
|
||||
Scoped to ACTIVE_BRANDS when that is set: an archived brand is neither
|
||||
exported nor seeded, in either direction.
|
||||
|
||||
Idempotent and non-destructive:
|
||||
|
||||
* A populated table with no seed file gets one exported.
|
||||
* A seed file whose table is empty or absent gets seeded, with
|
||||
cleanup disabled - the table holds nothing this batch could be a
|
||||
partial view of, so there is no stale row to remove and no way to
|
||||
delete data by passing an incomplete set.
|
||||
* A file that already maps to a populated table is left alone. That
|
||||
rule is what stops brand_catalog_tata.json being overwritten with
|
||||
Hindustan Unilever's merged rows.
|
||||
"""
|
||||
# Files may have appeared on disk since the last resolve (that is half of
|
||||
# what this function exists to handle), so start from a cold index.
|
||||
invalidate_seed_index_cache()
|
||||
|
||||
db_counts = _db_brand_counts()
|
||||
file_index = index_seed_files()
|
||||
collisions = _detect_collisions()
|
||||
|
||||
# BOTH SIDES MUST BE NARROWED TO THE ACTIVE BRANDS, OR NEITHER.
|
||||
#
|
||||
# _db_brand_counts() goes through _list_brand_table_suffixes(), so under
|
||||
# ACTIVE_BRANDS it only sees the active tables. index_seed_files() reads the
|
||||
# archive directory too, deliberately, so that re-activating a brand needs
|
||||
# only a config change.
|
||||
#
|
||||
# Left mismatched, every archived brand looks like "a seed file whose table
|
||||
# is empty" and lands in `to_seed` - so the boot reconcile would re-seed all
|
||||
# 27 archived catalogs on the next restart, recreate their tables, and
|
||||
# silently undo the archiving. Verified: a dry run reported exactly that.
|
||||
if active_brand_suffixes() is not None:
|
||||
file_index = {
|
||||
slug: paths for slug, paths in file_index.items() if is_active_suffix(slug)
|
||||
}
|
||||
|
||||
to_export = sorted(
|
||||
suffix for suffix, count in db_counts.items()
|
||||
if count > 0 and suffix not in file_index
|
||||
)
|
||||
to_seed = sorted(
|
||||
slug for slug in file_index
|
||||
if db_counts.get(slug, 0) == 0
|
||||
)
|
||||
|
||||
summary: Dict[str, Any] = {
|
||||
"tables": len(db_counts),
|
||||
"files": len(file_index),
|
||||
"exported": [],
|
||||
"seeded": [],
|
||||
"collisions": collisions,
|
||||
"skipped": [],
|
||||
"dry_run": dry_run,
|
||||
}
|
||||
|
||||
if dry_run:
|
||||
summary["exported"] = [display_name_for_suffix(s) for s in to_export]
|
||||
summary["seeded"] = to_seed
|
||||
return summary
|
||||
|
||||
for suffix in to_export:
|
||||
display = display_name_for_suffix(suffix)
|
||||
try:
|
||||
path = export_brand_to_seed_file(display)
|
||||
if path:
|
||||
summary["exported"].append(display)
|
||||
except Exception as e: # noqa: BLE001 - one brand must not stop the sweep
|
||||
logger.error("Export failed for brand '%s': %s", display, e)
|
||||
summary["skipped"].append({"brand": display, "reason": str(e)})
|
||||
|
||||
for slug in to_seed:
|
||||
paths = file_index.get(slug, [])
|
||||
try:
|
||||
grouped = load_seed_catalogs(only=[p.name for p in paths])
|
||||
if not grouped:
|
||||
continue
|
||||
seed_brands(grouped, cleanup=False)
|
||||
summary["seeded"].append(slug)
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.error("Seeding failed for slug '%s': %s", slug, e)
|
||||
summary["skipped"].append({"brand": slug, "reason": str(e)})
|
||||
|
||||
if summary["exported"] or summary["seeded"]:
|
||||
invalidate_brand_overview_cache()
|
||||
try:
|
||||
from app.services.query_intent import invalidate_brand_mention_cache
|
||||
invalidate_brand_mention_cache()
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
|
||||
if collisions:
|
||||
logger.warning("Brand slug collisions detected: %s", collisions)
|
||||
|
||||
return summary
|
||||
748
app/services/capture_discovery.py
Normal file
748
app/services/capture_discovery.py
Normal file
@@ -0,0 +1,748 @@
|
||||
"""Capture-to-catalog: a photographed product the catalog does not have.
|
||||
|
||||
THE FLOW
|
||||
--------
|
||||
/search/identify runs its read-only ladder (product_identify.py). When that
|
||||
answer is NOT confirmed and ENABLE_CAPTURE_DISCOVERY is on, the router hands
|
||||
the result here:
|
||||
|
||||
label text ─► parse_label ─► brand, product name, pack size, category
|
||||
│ unreadable / unknown brand ─► status "needs_input" (never "not found")
|
||||
▼
|
||||
dedup: the same image_id already stored ─► status "exists" + that product
|
||||
the same product already queued ─► that job's id
|
||||
▼
|
||||
one-row CSV ─► capture worker ─► store_catalog_pipeline.run_pipeline
|
||||
(all 11 stages: LLM description, category, pack size,
|
||||
pricing, web images, SKU, barcode, HSN/GST, validation,
|
||||
embed + upsert)
|
||||
▼
|
||||
post-pass on the INSERTED row only:
|
||||
validation_status = needs_review (a rejection is left alone)
|
||||
field_sources.capture = {origin, capture id, retail presence verdict}
|
||||
no web image found ─► the colleague's photo becomes the image
|
||||
(only with CAPTURE_PUBLIC_BASE_URL set)
|
||||
|
||||
The caller gets a provisional card (built from the label, instantly) and a job
|
||||
id; GET /api/search/identify/jobs/{id} returns the stored product when done.
|
||||
|
||||
WHY THE PHOTO IS TRUSTED, AND WHAT IS NOT
|
||||
-----------------------------------------
|
||||
Most of this catalogue is LLM output nobody grounded. A photo taken in a shop
|
||||
is the opposite: first-hand evidence the product exists. What can still be
|
||||
wrong is the READING - OCR misses a letter, or the pack size. That is why the
|
||||
live retail check matters (retail_presence matches the pack size, not only the
|
||||
title) and why the row is needs_review rather than verified.
|
||||
|
||||
WHY A WORKER OF ITS OWN, NOT batch_worker
|
||||
-----------------------------------------
|
||||
The post-pass has to run after the pipeline stores the row, and the batch
|
||||
worker has no hook for that. The pipeline itself is reused unchanged -
|
||||
`run_pipeline` is documented as "called on a background daemon thread". One
|
||||
thread, one bounded queue, the same shape as image_vector's worker.
|
||||
|
||||
The row is written ONLY for a brand we already know (a registry parent or an
|
||||
existing brand table): an OCR misread must not be able to mint a brand table.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import queue
|
||||
import re
|
||||
import threading
|
||||
import time
|
||||
import uuid
|
||||
from dataclasses import asdict, dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Any, Callable, Dict, List, Optional, Tuple
|
||||
|
||||
from app.infrastructure import settings
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Job states.
|
||||
QUEUED = "queued"
|
||||
RUNNING = "running"
|
||||
DONE = "done"
|
||||
REJECTED = "rejected" # the validation gate refused the product
|
||||
FAILED = "failed"
|
||||
INTERRUPTED = "interrupted" # the process restarted while it was queued/running
|
||||
TERMINAL = {DONE, REJECTED, FAILED, INTERRUPTED}
|
||||
|
||||
# What the identify response says about discovery (IdentifyOut.discovery_status).
|
||||
PENDING = "pending" # a job was queued (or one already was)
|
||||
EXISTS = "exists" # the product is stored; the ladder just missed it
|
||||
NEEDS_INPUT = "needs_input" # the label could not be turned into a product
|
||||
BUSY = "busy" # rate limit or full queue; provisional card only
|
||||
|
||||
ORIGIN = "field_capture"
|
||||
MATCHED_BY_DISCOVERY = "discovery_pending"
|
||||
MATCHED_BY_LABEL_EXACT = "label_exact"
|
||||
|
||||
_JOB_ID = re.compile(r"^[0-9a-f]{32}$")
|
||||
_PHOTO_TYPES = {
|
||||
"jpg": ("image/jpeg", b"\xff\xd8\xff"),
|
||||
"png": ("image/png", b"\x89PNG"),
|
||||
"webp": ("image/webp", b"RIFF"),
|
||||
}
|
||||
# Words that finish a product name when they follow a category keyword:
|
||||
# "Detergent Powder", "Detergent Bar", "Dishwash Liquid".
|
||||
_FORM_WORDS = {"powder", "bar", "liquid", "gel", "cake", "pods", "matic", "paste", "spray", "soap"}
|
||||
_MAX_NAME_WORDS = 6
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Label -> product
|
||||
# ---------------------------------------------------------------------------
|
||||
@dataclass
|
||||
class LabelParse:
|
||||
brand: str = "" # the display form written to the CSV ("Godrej")
|
||||
parent: str = "" # resolve_parent_brand(brand) - the table key
|
||||
product_name: str = "" # without the pack size ("Godrej Fab Detergent Powder")
|
||||
size: str = "" # compact, normalised ("1kg"); "" when not on the label
|
||||
category: str = "" # "" leaves stage 3 to decide
|
||||
hsn_code: Optional[str] = None
|
||||
gst_percent: Optional[float] = None
|
||||
hsn_gst_needs_review: Optional[bool] = None
|
||||
label_text: str = ""
|
||||
problem: Optional[str] = None # set when the label is unusable; the message to show
|
||||
|
||||
@property
|
||||
def usable(self) -> bool:
|
||||
return self.problem is None
|
||||
|
||||
|
||||
def _compact_size(size: str) -> str:
|
||||
return re.sub(r"(\d)\s+(?=[a-z])", r"\1", size.strip().lower())
|
||||
|
||||
|
||||
def _known_brand_names() -> List[str]:
|
||||
"""Every name that identifies a brand on its own: registry parents plus the
|
||||
exact alias keys, longest first so "hindustan unilever" beats "hindustan"."""
|
||||
from app.services.brand_registry import BRAND_ALIASES
|
||||
|
||||
names = set(BRAND_ALIASES.values()) | set(BRAND_ALIASES)
|
||||
return sorted(names, key=len, reverse=True)
|
||||
|
||||
|
||||
def _find_brand(text: str) -> Optional[Tuple[str, int]]:
|
||||
"""(brand name as written on the label, its word position) or None.
|
||||
|
||||
Registry parents are tried before `extract_brand_mention` because that
|
||||
function's brand map is built from ACTIVE brands - Godrej is inactive in
|
||||
this deployment, so "Godrej fab" would come back empty from it.
|
||||
"""
|
||||
lower = text.lower()
|
||||
for name in _known_brand_names():
|
||||
match = re.search(r"(?<!\w)" + re.escape(name) + r"(?!\w)", lower)
|
||||
if match:
|
||||
return text[match.start():match.end()], len(lower[:match.start()].split())
|
||||
from app.services.query_intent import extract_brand_mention
|
||||
|
||||
mention = extract_brand_mention(text)
|
||||
if mention:
|
||||
match = re.search(r"(?<!\w)" + re.escape(mention.lower()) + r"(?!\w)", lower)
|
||||
position = len(lower[:match.start()].split()) if match else 0
|
||||
return mention, position
|
||||
# A sub-brand printed on its own: "Surf excel" is only known through the
|
||||
# alias "hul surf excel". Two-word runs only - a single word would let
|
||||
# "Fab" reach Parle through "parle fab", which is exactly the misroute the
|
||||
# registry's own leading-brand rule refuses.
|
||||
from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand
|
||||
|
||||
parents = set(BRAND_ALIASES.values())
|
||||
words = text.split()
|
||||
for i in range(len(words) - 1):
|
||||
pair = f"{words[i]} {words[i + 1]}"
|
||||
if not all(w.isalpha() for w in pair.split()):
|
||||
continue
|
||||
if resolve_parent_brand(pair) in parents:
|
||||
return pair, i
|
||||
return None
|
||||
|
||||
|
||||
def _name_window(words: List[str], start: int) -> List[str]:
|
||||
"""Up to _MAX_NAME_WORDS words from `start`, stopping at the first number."""
|
||||
out: List[str] = []
|
||||
for word in words[start:]:
|
||||
if any(ch.isdigit() for ch in word) or len(out) >= _MAX_NAME_WORDS:
|
||||
break
|
||||
out.append(word)
|
||||
return out
|
||||
|
||||
|
||||
def _cut_after_category(window: List[str], brand_words: int) -> Tuple[List[str], Optional[str]]:
|
||||
"""Trim marketing copy: end the name at its category keyword (plus one form
|
||||
word), so "Godrej fab Detergent Powder Superior cleaning" stops at "Powder".
|
||||
|
||||
Never inside the brand: "Dairy Milk" names Cadbury's chocolate, and its
|
||||
"milk" must not end "Dairy Milk Silk" at the brand as a Dairy product."""
|
||||
from app.services.category_registry import detect_category_from_text
|
||||
|
||||
for end in range(brand_words + 1, len(window) + 1):
|
||||
category = detect_category_from_text(" ".join(window[:end]), exact_only=True)
|
||||
if category:
|
||||
if end < len(window) and window[end].lower() in _FORM_WORDS:
|
||||
end += 1
|
||||
return window[:end], category
|
||||
return window, None
|
||||
|
||||
|
||||
def parse_label(text: Optional[str], *, brand_hint: Optional[str] = None,
|
||||
category_hint: Optional[str] = None) -> LabelParse:
|
||||
"""Turn OCR'd label text into a product. Pure apart from registry lookups."""
|
||||
from app.core.store_catalog_pipeline import _SIZE_IN_TITLE
|
||||
from app.services.brand_registry import resolve_parent_brand
|
||||
from app.services.category_registry import detect_category_from_text
|
||||
from app.services.category_units import fix_or_reject_size
|
||||
from app.services.enrichment.hsn_gst.models import resolve_hsn_gst
|
||||
from app.services.label_match import clean_label
|
||||
|
||||
raw = (text or "").strip()
|
||||
parse = LabelParse(label_text=raw)
|
||||
if not raw:
|
||||
parse.problem = ("The label could not be read. Retake the photo closer to the "
|
||||
"front of the pack, or type the product name.")
|
||||
return parse
|
||||
|
||||
cleaned = clean_label(raw) or raw
|
||||
words = cleaned.split()
|
||||
|
||||
if brand_hint and brand_hint.strip():
|
||||
brand = brand_hint.strip()
|
||||
found = _find_brand(cleaned)
|
||||
position = found[1] if found and resolve_parent_brand(found[0]) == resolve_parent_brand(brand) else None
|
||||
else:
|
||||
found = _find_brand(cleaned)
|
||||
if not found:
|
||||
parse.problem = ("No brand we know was found on the label. Type the brand and "
|
||||
"product name so it can be added.")
|
||||
return parse
|
||||
brand, position = found
|
||||
|
||||
if position is None:
|
||||
# The caller named the brand but it is not on the label text: the name
|
||||
# is the label's leading words, prefixed with the brand.
|
||||
tail = _name_window(words, 0)
|
||||
window = brand.split() + [w for w in tail if w.lower() not in brand.lower().split()]
|
||||
else:
|
||||
window = _name_window(words, position)
|
||||
brand_words = len(brand.split())
|
||||
window, category = _cut_after_category(window, brand_words)
|
||||
|
||||
if len(window) <= brand_words:
|
||||
parse.problem = (f"Only the brand ({brand}) could be read. Type the product name "
|
||||
"so it can be added.")
|
||||
parse.brand, parse.parent = brand, resolve_parent_brand(brand)
|
||||
return parse
|
||||
|
||||
name = " ".join(w if w.isupper() and len(w) > 3 else w.capitalize() for w in window)
|
||||
# The brand as the label wrote it, but in title case - "godrej" -> "Godrej".
|
||||
display_brand = " ".join(w.capitalize() if w.islower() else w for w in brand.split())
|
||||
|
||||
category = category or detect_category_from_text(cleaned, exact_only=True) or (category_hint or "")
|
||||
|
||||
size = ""
|
||||
size_match = _SIZE_IN_TITLE.search(cleaned)
|
||||
if size_match:
|
||||
fixed, _changed, _reason = fix_or_reject_size(_compact_size(size_match.group(0)), category, name)
|
||||
size = _compact_size(fixed) if fixed else ""
|
||||
|
||||
parse.brand = display_brand
|
||||
parse.parent = resolve_parent_brand(display_brand)
|
||||
parse.product_name = name
|
||||
parse.size = size
|
||||
parse.category = category
|
||||
if category:
|
||||
info = resolve_hsn_gst(category, name)
|
||||
if info is not None:
|
||||
parse.hsn_code = info.hsn_code
|
||||
parse.gst_percent = info.gst_percent
|
||||
parse.hsn_gst_needs_review = info.hsn_gst_needs_review
|
||||
return parse
|
||||
|
||||
|
||||
def is_known_brand(parent: str, table_exists: Optional[Callable[[str], bool]] = None) -> bool:
|
||||
"""D4: only a brand we already know may receive a row - a registry parent,
|
||||
or a brand table that already exists (active or not)."""
|
||||
from app.services.brand_registry import BRAND_ALIASES
|
||||
|
||||
key = (parent or "").strip().lower()
|
||||
if not key:
|
||||
return False
|
||||
if key in set(BRAND_ALIASES.values()):
|
||||
return True
|
||||
return bool((table_exists or _brand_table_exists)(parent))
|
||||
|
||||
|
||||
def _brand_table_exists(parent: str) -> bool:
|
||||
from app.services.vector_store import _connect, _table_exists, _table_name
|
||||
|
||||
conn = _connect()
|
||||
if conn is None:
|
||||
return False
|
||||
try:
|
||||
with conn.cursor() as cur:
|
||||
return bool(_table_exists(cur, _table_name(parent)))
|
||||
except Exception: # noqa: BLE001 - "unknown" is the safe answer
|
||||
return False
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
|
||||
def expected_image_id(parse: LabelParse) -> str:
|
||||
from app.core.store_catalog_pipeline import build_image_id
|
||||
|
||||
return build_image_id(parse.parent, parse.product_name, parse.size)
|
||||
|
||||
|
||||
def provisional_card(parse: LabelParse) -> Dict[str, Any]:
|
||||
"""What the colleague sees immediately, from the label alone."""
|
||||
from app.services import active_brands
|
||||
|
||||
return {
|
||||
"brand": parse.brand,
|
||||
"product_name": parse.product_name,
|
||||
"size": parse.size or None,
|
||||
"category": parse.category or None,
|
||||
"hsn_code": parse.hsn_code,
|
||||
"gst_percent": parse.gst_percent,
|
||||
"hsn_gst_needs_review": parse.hsn_gst_needs_review,
|
||||
# False means: the row will be written, but search filters the brand
|
||||
# out until it is added to ACTIVE_BRANDS.
|
||||
"visible_in_search": (not active_brands.filtering_enabled())
|
||||
or active_brands.is_active_brand(parse.parent),
|
||||
"source": "label",
|
||||
}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Jobs: durable JSON under CAPTURE_DIR/jobs
|
||||
# ---------------------------------------------------------------------------
|
||||
@dataclass
|
||||
class CaptureJob:
|
||||
job_id: str
|
||||
status: str
|
||||
created_at: float
|
||||
updated_at: float
|
||||
brand: str
|
||||
parent: str
|
||||
product_name: str
|
||||
size: str
|
||||
category: str
|
||||
label_text: str
|
||||
provisional: Dict[str, Any]
|
||||
photo_ext: Optional[str] = None
|
||||
submitted_by: Optional[str] = None
|
||||
image_id: Optional[str] = None
|
||||
disposition: Optional[str] = None # inserted | backfilled | unchanged
|
||||
retail_presence: Optional[Dict[str, Any]] = None
|
||||
photo_used_as_image: bool = False
|
||||
detail: Optional[str] = None
|
||||
warnings: List[str] = field(default_factory=list)
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return asdict(self)
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, data: Dict[str, Any]) -> "CaptureJob":
|
||||
known = {k: data[k] for k in cls.__dataclass_fields__ if k in data}
|
||||
return cls(**known)
|
||||
|
||||
|
||||
def _jobs_dir() -> Path:
|
||||
path = Path(settings.CAPTURE_DIR) / "jobs"
|
||||
path.mkdir(parents=True, exist_ok=True)
|
||||
return path
|
||||
|
||||
|
||||
def _photos_dir() -> Path:
|
||||
path = Path(settings.CAPTURE_DIR) / "photos"
|
||||
path.mkdir(parents=True, exist_ok=True)
|
||||
return path
|
||||
|
||||
|
||||
_state_lock = threading.Lock()
|
||||
_jobs: Dict[str, CaptureJob] = {}
|
||||
_live_ids: set = set() # queued or running in THIS process
|
||||
_inflight: Dict[str, str] = {} # expected image_id -> job_id
|
||||
|
||||
|
||||
def _save(job: CaptureJob) -> None:
|
||||
job.updated_at = time.time()
|
||||
path = _jobs_dir() / f"{job.job_id}.json"
|
||||
tmp = path.with_suffix(".tmp")
|
||||
tmp.write_text(json.dumps(job.to_dict(), ensure_ascii=False), encoding="utf-8")
|
||||
tmp.replace(path)
|
||||
with _state_lock:
|
||||
_jobs[job.job_id] = job
|
||||
|
||||
|
||||
def get_job(job_id: str) -> Optional[CaptureJob]:
|
||||
if not _JOB_ID.match(job_id or ""):
|
||||
return None
|
||||
with _state_lock:
|
||||
job = _jobs.get(job_id)
|
||||
live = job_id in _live_ids
|
||||
if job is None:
|
||||
path = _jobs_dir() / f"{job_id}.json"
|
||||
if not path.exists():
|
||||
return None
|
||||
try:
|
||||
job = CaptureJob.from_dict(json.loads(path.read_text(encoding="utf-8")))
|
||||
except Exception: # noqa: BLE001 - a corrupt record is a missing one
|
||||
logger.warning("capture: unreadable job record %s", path)
|
||||
return None
|
||||
if job.status not in TERMINAL and not live:
|
||||
# Queued or running according to disk, but no worker here owns it: the
|
||||
# process restarted. Say so rather than letting a client poll forever.
|
||||
job.status = INTERRUPTED
|
||||
job.detail = "The server restarted before this job finished. Capture the product again."
|
||||
_save(job)
|
||||
return job
|
||||
|
||||
|
||||
def list_jobs(limit: int = 50) -> List[CaptureJob]:
|
||||
paths = sorted(_jobs_dir().glob("*.json"), key=lambda p: p.stat().st_mtime, reverse=True)
|
||||
out = []
|
||||
for path in paths[:limit]:
|
||||
job = get_job(path.stem)
|
||||
if job is not None:
|
||||
out.append(job)
|
||||
return out
|
||||
|
||||
|
||||
def photo_path(name: str) -> Optional[Tuple[Path, str]]:
|
||||
"""(file, media type) for a stored capture photo named "<job id>.<ext>"."""
|
||||
stem, _, ext = (name or "").partition(".")
|
||||
if not _JOB_ID.match(stem) or ext not in _PHOTO_TYPES:
|
||||
return None
|
||||
path = _photos_dir() / f"{stem}.{ext}"
|
||||
return (path, _PHOTO_TYPES[ext][0]) if path.exists() else None
|
||||
|
||||
|
||||
def _photo_ext(image_bytes: bytes) -> Optional[str]:
|
||||
for ext, (_media, magic) in _PHOTO_TYPES.items():
|
||||
if image_bytes.startswith(magic):
|
||||
if ext == "webp" and image_bytes[8:12] != b"WEBP":
|
||||
continue
|
||||
return ext
|
||||
return None
|
||||
|
||||
|
||||
def _photo_public_url(job: CaptureJob) -> Optional[str]:
|
||||
base = settings.CAPTURE_PUBLIC_BASE_URL
|
||||
if not base or not job.photo_ext:
|
||||
return None
|
||||
return f"{base.rstrip('/')}/api/search/captures/{job.job_id}.{job.photo_ext}"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Rate limit (the route is public)
|
||||
# ---------------------------------------------------------------------------
|
||||
_starts: Dict[str, List[float]] = {}
|
||||
|
||||
|
||||
def _allow_start(client: str, now: Optional[float] = None) -> bool:
|
||||
now = time.time() if now is None else now
|
||||
limit = settings.CAPTURE_MAX_PER_CLIENT_PER_HOUR
|
||||
with _state_lock:
|
||||
recent = [t for t in _starts.get(client, []) if now - t < 3600]
|
||||
if len(recent) >= limit:
|
||||
_starts[client] = recent
|
||||
return False
|
||||
recent.append(now)
|
||||
_starts[client] = recent
|
||||
return True
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The miss handler the router calls
|
||||
# ---------------------------------------------------------------------------
|
||||
@dataclass
|
||||
class CaptureOutcome:
|
||||
status: str
|
||||
message: Optional[str] = None
|
||||
job_id: Optional[str] = None
|
||||
provisional: Optional[Dict[str, Any]] = None
|
||||
existing: Optional[Dict[str, Any]] = None # the stored row, for EXISTS
|
||||
|
||||
|
||||
def is_confirmed(matched_by: str, fallback_reason: Optional[str]) -> bool:
|
||||
"""The rule product_identify.py documents for clients."""
|
||||
return matched_by == "text" or (matched_by == "image_vector" and not fallback_reason)
|
||||
|
||||
|
||||
def handle_miss(
|
||||
*,
|
||||
label_text: Optional[str],
|
||||
image_bytes: Optional[bytes],
|
||||
vector: Optional[List[float]],
|
||||
brand: Optional[str],
|
||||
category: Optional[str],
|
||||
client: str,
|
||||
) -> CaptureOutcome:
|
||||
"""Decide what an unconfirmed identify turns into. Never raises."""
|
||||
try:
|
||||
return _handle_miss(label_text=label_text, image_bytes=image_bytes, vector=vector,
|
||||
brand=brand, category=category, client=client)
|
||||
except Exception: # noqa: BLE001 - discovery must never break identify
|
||||
logger.exception("capture: discovery failed; identify answered without it")
|
||||
return CaptureOutcome(status=NEEDS_INPUT,
|
||||
message="The product could not be added automatically. "
|
||||
"Type the brand and product name.")
|
||||
|
||||
|
||||
def _handle_miss(*, label_text, image_bytes, vector, brand, category, client) -> CaptureOutcome:
|
||||
from app.services.vector_store import get_product_by_image_id
|
||||
|
||||
parse = parse_label(label_text, brand_hint=brand, category_hint=category)
|
||||
if not parse.usable:
|
||||
return CaptureOutcome(status=NEEDS_INPUT, message=parse.problem)
|
||||
card = provisional_card(parse)
|
||||
if not is_known_brand(parse.parent):
|
||||
return CaptureOutcome(
|
||||
status=NEEDS_INPUT, provisional=card,
|
||||
message=(f"'{parse.brand}' is not a brand in the catalog yet, so it was not "
|
||||
"added automatically. Ask an admin to add the brand, or correct the "
|
||||
"brand name if the label was misread."),
|
||||
)
|
||||
|
||||
image_id = expected_image_id(parse)
|
||||
existing = get_product_by_image_id(parse.parent, image_id)
|
||||
if existing:
|
||||
return CaptureOutcome(status=EXISTS, provisional=card, existing=existing,
|
||||
message="This product is already in the catalog.")
|
||||
|
||||
with _state_lock:
|
||||
running = _inflight.get(image_id)
|
||||
if running:
|
||||
return CaptureOutcome(status=PENDING, job_id=running, provisional=card,
|
||||
message="This product is already being added.")
|
||||
|
||||
if not _allow_start(client):
|
||||
return CaptureOutcome(status=BUSY, provisional=card,
|
||||
message="Too many new products from this device in the last "
|
||||
"hour. The details above are read from the label.")
|
||||
|
||||
now = time.time()
|
||||
job = CaptureJob(
|
||||
job_id=uuid.uuid4().hex, status=QUEUED, created_at=now, updated_at=now,
|
||||
brand=parse.brand, parent=parse.parent, product_name=parse.product_name,
|
||||
size=parse.size, category=parse.category, label_text=parse.label_text,
|
||||
provisional=card, submitted_by=client, image_id=image_id,
|
||||
)
|
||||
if image_bytes:
|
||||
ext = _photo_ext(image_bytes)
|
||||
if ext:
|
||||
(_photos_dir() / f"{job.job_id}.{ext}").write_bytes(image_bytes)
|
||||
job.photo_ext = ext
|
||||
_save(job)
|
||||
|
||||
with _state_lock:
|
||||
_live_ids.add(job.job_id)
|
||||
_inflight[image_id] = job.job_id
|
||||
try:
|
||||
_ensure_worker()
|
||||
_queue.put_nowait((job.job_id, list(vector) if vector is not None else None))
|
||||
except queue.Full:
|
||||
with _state_lock:
|
||||
_live_ids.discard(job.job_id)
|
||||
_inflight.pop(image_id, None)
|
||||
job.status = FAILED
|
||||
job.detail = "The capture queue was full. Capture the product again in a few minutes."
|
||||
_save(job)
|
||||
return CaptureOutcome(status=BUSY, provisional=card,
|
||||
message="Many products are being added right now. The details "
|
||||
"above are read from the label; try again shortly.")
|
||||
return CaptureOutcome(status=PENDING, job_id=job.job_id, provisional=card,
|
||||
message="Not in the catalog yet - adding it now. Details will "
|
||||
"fill in within a few minutes.")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The worker
|
||||
# ---------------------------------------------------------------------------
|
||||
_queue: "queue.Queue[Tuple[str, Optional[List[float]]]]" = queue.Queue(
|
||||
maxsize=max(1, settings.CAPTURE_QUEUE_MAX))
|
||||
_worker: Optional[threading.Thread] = None
|
||||
_worker_lock = threading.Lock()
|
||||
|
||||
|
||||
def _ensure_worker() -> None:
|
||||
global _worker
|
||||
with _worker_lock:
|
||||
if _worker is None or not _worker.is_alive():
|
||||
_worker = threading.Thread(target=_loop, name="capture-discovery", daemon=True)
|
||||
_worker.start()
|
||||
|
||||
|
||||
def _loop() -> None:
|
||||
while True:
|
||||
job_id, vector = _queue.get()
|
||||
try:
|
||||
job = get_job(job_id)
|
||||
if job is not None:
|
||||
run_job(job, vector)
|
||||
except Exception: # noqa: BLE001 - one bad job must not kill the worker
|
||||
logger.exception("capture: job %s crashed", job_id)
|
||||
finally:
|
||||
with _state_lock:
|
||||
_live_ids.discard(job_id)
|
||||
for key, value in list(_inflight.items()):
|
||||
if value == job_id:
|
||||
del _inflight[key]
|
||||
_queue.task_done()
|
||||
|
||||
|
||||
def build_csv(job: CaptureJob) -> Tuple[str, bytes]:
|
||||
"""The one-row sheet the pipeline reads, through the same writer the
|
||||
brand-discovery ingest uses."""
|
||||
from app.services.brand_discovery import DiscoveredProduct, rows_to_csv_bytes
|
||||
from app.services.vector_store import _sanitize_name
|
||||
|
||||
product = DiscoveredProduct(
|
||||
brand=job.brand, product_name=job.product_name, title=job.product_name,
|
||||
category=job.category or "", category_hint="", description="",
|
||||
size_variants=[job.size] if job.size else [],
|
||||
)
|
||||
name = f"capture-{_sanitize_name(job.parent) or 'brand'}-{job.job_id[:8]}.csv"
|
||||
return name, rows_to_csv_bytes([product])
|
||||
|
||||
|
||||
def run_job(job: CaptureJob, vector: Optional[List[float]] = None) -> CaptureJob:
|
||||
"""Run the pipeline for one capture and mark what it stored. Synchronous."""
|
||||
from app.core import store_catalog_pipeline as pipeline
|
||||
|
||||
job.status = RUNNING
|
||||
_save(job)
|
||||
try:
|
||||
filename, content = build_csv(job)
|
||||
result = pipeline.run_pipeline(filename, content, use_llm=settings.CAPTURE_USE_LLM,
|
||||
fetch_images=True)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.exception("capture: pipeline failed for %s", job.product_name)
|
||||
job.status, job.detail = FAILED, f"The catalog pipeline failed: {exc}"
|
||||
_save(job)
|
||||
return job
|
||||
|
||||
job.warnings = list(result.warnings)[:20]
|
||||
if result.storage_error:
|
||||
job.status, job.detail = FAILED, f"Storing the product failed: {result.storage_error}"
|
||||
_save(job)
|
||||
return job
|
||||
if not result.products:
|
||||
reasons = "; ".join(r.get("reason", "") for r in result.rejections) or \
|
||||
"; ".join(e.error for e in result.errors)
|
||||
job.status = REJECTED
|
||||
job.detail = f"The product was not added: {reasons or 'it failed validation'}"
|
||||
_save(job)
|
||||
return job
|
||||
|
||||
stored = result.products[0]
|
||||
job.image_id = stored.get("image_id") or job.image_id
|
||||
job.disposition = stored.get("disposition")
|
||||
|
||||
if job.disposition == "inserted":
|
||||
job.retail_presence = _retail_check(job)
|
||||
fallback = _photo_public_url(job)
|
||||
applied_photo = mark_captured_row(job, fallback)
|
||||
job.photo_used_as_image = applied_photo
|
||||
if applied_photo and vector is not None:
|
||||
_store_photo_vector(job, vector, fallback)
|
||||
else:
|
||||
# The row existed after all (a race, or the label named a product the
|
||||
# ladder missed). It keeps whatever verdict it had - a field capture
|
||||
# never downgrades a row it did not create.
|
||||
job.detail = "The product was already in the catalog; nothing was changed."
|
||||
|
||||
job.status = DONE
|
||||
_save(job)
|
||||
return job
|
||||
|
||||
|
||||
def _retail_check(job: CaptureJob) -> Optional[Dict[str, Any]]:
|
||||
if not settings.CAPTURE_RETAIL_CHECK:
|
||||
return None
|
||||
try:
|
||||
from app.services.retail_presence import check_listing
|
||||
|
||||
evidence = check_listing(job.parent, job.product_name, job.size, live=True)
|
||||
return {
|
||||
"status": evidence.status,
|
||||
"retailer": evidence.retailer,
|
||||
"url": evidence.url,
|
||||
"matched_title": evidence.matched_title,
|
||||
"matched_size": evidence.matched_size,
|
||||
}
|
||||
except Exception as exc: # noqa: BLE001 - corroboration is best-effort
|
||||
logger.warning("capture: retail check failed for %s: %s", job.product_name, exc)
|
||||
return {"status": "unknown", "error": str(exc)}
|
||||
|
||||
|
||||
def mark_captured_row(job: CaptureJob, fallback_image_url: Optional[str]) -> bool:
|
||||
"""needs_review + capture provenance on the row this job INSERTED; the photo
|
||||
as the image only where the pipeline stored none. True when the photo was
|
||||
applied. A rejected verdict is never lifted to needs_review."""
|
||||
from psycopg.types.json import Json
|
||||
|
||||
from app.services.vector_store import _connect, _table_name
|
||||
|
||||
provenance = {"capture": {
|
||||
"origin": ORIGIN,
|
||||
"capture_id": job.job_id,
|
||||
"captured_at": job.created_at,
|
||||
"label_text": job.label_text[:500],
|
||||
"retail_presence": job.retail_presence,
|
||||
}}
|
||||
conn = _connect()
|
||||
if conn is None:
|
||||
job.warnings.append("could not mark the row needs_review: no database connection")
|
||||
return False
|
||||
table = _table_name(job.parent)
|
||||
try:
|
||||
with conn.cursor() as cur:
|
||||
cur.execute(
|
||||
f"UPDATE {table} SET "
|
||||
f"validation_status = CASE WHEN validation_status = 'rejected' "
|
||||
f"THEN validation_status ELSE 'needs_review' END, "
|
||||
f"field_sources = COALESCE(field_sources, '{{}}'::jsonb) || %s "
|
||||
f"WHERE image_id = %s",
|
||||
(Json(provenance), job.image_id),
|
||||
)
|
||||
applied = False
|
||||
if fallback_image_url:
|
||||
cur.execute(
|
||||
f"UPDATE {table} SET image_url = %s, image_urls = ARRAY[%s]::text[] "
|
||||
f"WHERE image_id = %s AND COALESCE(image_url, '') = '' "
|
||||
f"AND COALESCE(cardinality(image_urls), 0) = 0",
|
||||
(fallback_image_url, fallback_image_url, job.image_id),
|
||||
)
|
||||
applied = cur.rowcount > 0
|
||||
return applied
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.exception("capture: marking %s failed", job.image_id)
|
||||
job.warnings.append(f"could not mark the row needs_review: {exc}")
|
||||
return False
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
|
||||
def _store_photo_vector(job: CaptureJob, vector: List[float], src: str) -> None:
|
||||
"""The photo IS the primary image now, so its vector is the row's current
|
||||
one - img_vector_src names the URL those exact bytes are served from."""
|
||||
from app.services import image_vector
|
||||
from app.services.vector_store import _connect, _table_name
|
||||
|
||||
conn = _connect()
|
||||
if conn is None:
|
||||
return
|
||||
try:
|
||||
with conn.cursor() as cur:
|
||||
if image_vector.table_has_columns(cur, _table_name(job.parent)):
|
||||
image_vector.store_vector(cur, _table_name(job.parent), job.image_id, vector, src)
|
||||
except Exception: # noqa: BLE001 - the worker will compute it from the URL later
|
||||
logger.exception("capture: storing the photo vector for %s failed", job.image_id)
|
||||
finally:
|
||||
conn.close()
|
||||
240
app/services/catalog_search.py
Normal file
240
app/services/catalog_search.py
Normal file
@@ -0,0 +1,240 @@
|
||||
"""Search behind the catalog page's search box (GET /api/search).
|
||||
|
||||
Deliberately separate from rag_service.retrieve(), which serves /api/chat.
|
||||
retrieve() clamps results to RAG_MAX_TOP_K (15) to protect the LLM's prompt
|
||||
budget; sharing that path is what silently truncated every brand search to 15
|
||||
products even though the UI asked for more. A product grid has no prompt
|
||||
budget, so it gets its own entry point and its own ceiling.
|
||||
|
||||
Two shapes are handled differently, which is the whole point of the module:
|
||||
|
||||
"Amul" -> brand_catalog mode: the brand's entire listing, exactly the
|
||||
rows GET /api/brands/{brand}/products returns.
|
||||
"Amul Butter" -> hybrid mode: name matches first (so every size variant is
|
||||
present and ranked together), then semantically similar
|
||||
products.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from app.infrastructure.settings import (
|
||||
SEARCH_DEFAULT_TOP_K,
|
||||
SEARCH_HYBRID_MAX_TOP_K,
|
||||
SEARCH_LEXICAL_CANDIDATES,
|
||||
)
|
||||
from app.services.category_registry import detect_category_from_text
|
||||
from app.services.embeddings_service import embed_texts
|
||||
from app.services.query_intent import (
|
||||
classify_query,
|
||||
extract_attributes,
|
||||
extract_max_price,
|
||||
)
|
||||
from app.services.rag_service import (
|
||||
RetrievedProduct,
|
||||
filter_to_category,
|
||||
rerank_by_attributes,
|
||||
row_to_retrieved_product,
|
||||
)
|
||||
from app.services.vector_store import (
|
||||
_sanitize_name,
|
||||
count_products_by_brand,
|
||||
get_products_by_brand,
|
||||
lexical_search,
|
||||
semantic_search,
|
||||
text_search,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Fallback similarity per lexical tier, used only when the embedding model is
|
||||
# unavailable and a real cosine distance cannot be computed. Mirrors
|
||||
# RetrievedProduct.similarity == 1 - distance/2.
|
||||
_LEX_TIER_SCORE = {0: 1.00, 1: 0.92, 2: 0.80, 3: 0.70}
|
||||
|
||||
# Words that add nothing to a lexical name match.
|
||||
_TERM_STOP_WORDS = {
|
||||
"a", "an", "the", "of", "for", "with", "and", "in", "on",
|
||||
"show", "me", "find", "get", "all", "any", "please",
|
||||
}
|
||||
|
||||
|
||||
@dataclass
|
||||
class CatalogSearchResult:
|
||||
products: List[RetrievedProduct]
|
||||
match_mode: str # "brand_catalog" | "hybrid"
|
||||
detected_brand: Optional[str] = None
|
||||
detected_category: Optional[str] = None
|
||||
total: Optional[int] = None # exact only in brand_catalog mode
|
||||
limit: int = 0
|
||||
offset: int = 0
|
||||
|
||||
|
||||
def _row_key(row: Dict[str, Any]) -> tuple:
|
||||
"""Dedup key for a product row.
|
||||
|
||||
image_id is UNIQUE per brand table but not across them, so the brand has
|
||||
to be part of the key for a cross-brand merge to be correct.
|
||||
"""
|
||||
return (_sanitize_name(str(row.get("brand") or "")), str(row.get("image_id") or ""))
|
||||
|
||||
|
||||
def _tokenize(text: str) -> List[str]:
|
||||
return [t for t in text.lower().split() if t and t not in _TERM_STOP_WORDS]
|
||||
|
||||
|
||||
def _brand_catalog(
|
||||
brand: str, category: Optional[str], limit: int, offset: int
|
||||
) -> CatalogSearchResult:
|
||||
"""The brand's full listing - the same rows the Browse tab shows.
|
||||
|
||||
Category is NOT auto-detected here. detect_category_from_text() has a fuzzy
|
||||
fallback that misfires on brand names ("Colgate" scores as Chocolates,
|
||||
"Milky Mist" as Dairy), and applying that to a brand-only query filters most
|
||||
of the catalog away. Only an explicit category filter applies.
|
||||
"""
|
||||
rows = get_products_by_brand(brand, limit=limit, offset=offset, category=category)
|
||||
total = count_products_by_brand(brand, category=category)
|
||||
|
||||
for row in rows:
|
||||
row["brand"] = row.get("brand") or brand
|
||||
row["distance"] = 0.0
|
||||
|
||||
return CatalogSearchResult(
|
||||
products=[row_to_retrieved_product(r) for r in rows],
|
||||
match_mode="brand_catalog",
|
||||
detected_brand=brand,
|
||||
detected_category=category,
|
||||
total=total,
|
||||
limit=limit,
|
||||
offset=offset,
|
||||
)
|
||||
|
||||
|
||||
def _hybrid_rank(
|
||||
semantic_rows: List[Dict[str, Any]],
|
||||
lexical_rows: List[Dict[str, Any]],
|
||||
limit: int,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Merge both arms, name matches first, semantic similarity within each tier."""
|
||||
merged: Dict[tuple, Dict[str, Any]] = {}
|
||||
|
||||
# Semantic rows first: they carry a real cosine distance worth keeping.
|
||||
for row in semantic_rows:
|
||||
merged[_row_key(row)] = row
|
||||
|
||||
for row in lexical_rows:
|
||||
key = _row_key(row)
|
||||
existing = merged.get(key)
|
||||
if existing is not None:
|
||||
# Same product from both arms: keep the real distance, take the tier.
|
||||
existing["lex_tier"] = row.get("lex_tier", 3)
|
||||
else:
|
||||
if row.get("distance") is None:
|
||||
tier = row.get("lex_tier", 3)
|
||||
row["distance"] = 2.0 * (1.0 - _LEX_TIER_SCORE.get(tier, 0.7))
|
||||
merged[key] = row
|
||||
|
||||
ordered = sorted(
|
||||
merged.values(),
|
||||
key=lambda r: (
|
||||
r.get("lex_tier", 9), # any name match outranks semantic-only
|
||||
r.get("distance", 9.0), # then vector similarity within the tier
|
||||
len(str(r.get("product_name") or r.get("title") or "")),
|
||||
str(r.get("product_name") or ""),
|
||||
),
|
||||
)
|
||||
return ordered[:limit]
|
||||
|
||||
|
||||
def _hybrid(
|
||||
query: str,
|
||||
brand: Optional[str],
|
||||
category: Optional[str],
|
||||
residual: str,
|
||||
limit: int,
|
||||
) -> CatalogSearchResult:
|
||||
"""Name matches first, then semantically similar products."""
|
||||
# Detect the category from the residual, never the whole query: the brand
|
||||
# token itself can fuzzy-match a category keyword ("Colgate" -> Chocolates).
|
||||
target_category = category or detect_category_from_text(residual or query)
|
||||
max_price = extract_max_price(query)
|
||||
hybrid_limit = min(limit, SEARCH_HYBRID_MAX_TOP_K)
|
||||
|
||||
try:
|
||||
vectors = embed_texts([query])
|
||||
except Exception as e: # noqa: BLE001 - fall back to lexical-only
|
||||
logger.warning("Embedding model failed: %s. Lexical-only search.", e)
|
||||
vectors = None
|
||||
|
||||
query_embedding = vectors[0] if vectors else None
|
||||
|
||||
terms = _tokenize(residual or query)
|
||||
lexical_rows: List[Dict[str, Any]] = []
|
||||
if terms:
|
||||
lexical_rows = lexical_search(
|
||||
terms,
|
||||
brand=brand,
|
||||
limit=SEARCH_LEXICAL_CANDIDATES,
|
||||
category=target_category,
|
||||
max_price=max_price,
|
||||
query_embedding=query_embedding,
|
||||
exact_phrase=" ".join(terms),
|
||||
)
|
||||
lexical_rows = filter_to_category(lexical_rows, target_category)
|
||||
|
||||
semantic_rows: List[Dict[str, Any]] = []
|
||||
if query_embedding is not None:
|
||||
semantic_rows = semantic_search(
|
||||
query_embedding=query_embedding,
|
||||
brand=brand,
|
||||
top_k=hybrid_limit,
|
||||
category=target_category,
|
||||
max_price=max_price,
|
||||
)
|
||||
semantic_rows = filter_to_category(semantic_rows, target_category)
|
||||
|
||||
if not semantic_rows and not lexical_rows:
|
||||
semantic_rows = filter_to_category(
|
||||
text_search(query, brand=brand, top_k=hybrid_limit,
|
||||
category=target_category, max_price=max_price),
|
||||
target_category,
|
||||
)
|
||||
|
||||
rows = _hybrid_rank(semantic_rows, lexical_rows, limit)
|
||||
products = [row_to_retrieved_product(r) for r in rows]
|
||||
|
||||
attrs = extract_attributes(query)
|
||||
if attrs and products:
|
||||
products = rerank_by_attributes(products, attrs)
|
||||
|
||||
return CatalogSearchResult(
|
||||
products=products,
|
||||
match_mode="hybrid",
|
||||
detected_brand=brand,
|
||||
detected_category=target_category,
|
||||
total=None, # no cheap exact count across N tables; do not invent one
|
||||
limit=limit,
|
||||
offset=0,
|
||||
)
|
||||
|
||||
|
||||
def search_catalog(
|
||||
query: str,
|
||||
brand: Optional[str] = None,
|
||||
category: Optional[str] = None,
|
||||
limit: int = SEARCH_DEFAULT_TOP_K,
|
||||
offset: int = 0,
|
||||
) -> CatalogSearchResult:
|
||||
"""Entry point for GET /api/search."""
|
||||
shape = classify_query(query, explicit_brand=brand)
|
||||
# An explicit filter from the sidebar always wins over one inferred from the
|
||||
# query text - same precedence retrieve() uses.
|
||||
effective_brand = brand or shape.brand
|
||||
|
||||
if shape.kind == "brand_only" and effective_brand:
|
||||
return _brand_catalog(effective_brand, category, limit, offset)
|
||||
|
||||
return _hybrid(query, effective_brand, category, shape.residual, limit)
|
||||
@@ -53,7 +53,12 @@ from typing import Dict, List, Optional, Tuple
|
||||
# this category when sanitizing cross-category language
|
||||
# out of a generated description.
|
||||
CATEGORY_REGISTRY: List[Dict[str, object]] = [
|
||||
{"category": "Biscuits & Cookies", "keywords": ["biscuits", "biscuit", "biscit", "biskut", "cookies", "cookie"], "generic_term": "biscuit"},
|
||||
# "bikis" is here so that "Britannia Milk Bikis" resolves as the biscuit it
|
||||
# is. Without it the only keyword in that name is Dairy's "milk", which not
|
||||
# only mislabels the product but hands stage 4 a volume unit rulebook - the
|
||||
# exact "Britannia Milk Bikis - 200ml, 500ml, 1L" defect category_units.py
|
||||
# was written to stop.
|
||||
{"category": "Biscuits & Cookies", "keywords": ["biscuits", "biscuit", "biscit", "biskut", "cookies", "cookie", "bikis"], "generic_term": "biscuit"},
|
||||
{"category": "Rusk", "keywords": ["rusks", "rusk"], "generic_term": "rusk"},
|
||||
{"category": "Crackers", "keywords": ["crackers", "cracker", "saltine"], "generic_term": "cracker"},
|
||||
{"category": "Cakes & Muffins", "keywords": ["cakes", "cake", "muffins", "muffin"], "generic_term": "bakery item"},
|
||||
@@ -62,14 +67,55 @@ CATEGORY_REGISTRY: List[Dict[str, object]] = [
|
||||
{"category": "Candy & Confectionery", "keywords": ["candy", "candies", "toffee", "toffees", "lollipop", "lollipops", "confectionery", "mints", "chewing gum"], "generic_term": "candy"},
|
||||
{"category": "Snacks", "keywords": ["snacks", "snack", "chips", "namkeen", "wafers", "wafer", "kurkure", "lays"], "generic_term": "snack"},
|
||||
{"category": "Chocolates", "keywords": ["chocolates", "chocolate", "chocate", "choclate", "cocoa", "cadbury chocolate", "dairy milk"], "generic_term": "chocolate"},
|
||||
# Listed ahead of "Cooking Oils" so a drink is resolved before that entry's
|
||||
# very broad bare "oil" keyword gets a chance, and ahead of "Dairy" only in
|
||||
# keywords it does not share - "milk", "lassi" and "buttermilk" are
|
||||
# deliberately left to Dairy. Kept free of "soda" (baking soda is a staple,
|
||||
# not a drink) and of "tea"/"coffee" on their own (those are sold as leaves
|
||||
# and grounds far more often than as a drink).
|
||||
{"category": "Beverages", "keywords": ["beverages", "beverage", "soft drink", "soft drinks", "cold drink", "cold drinks", "carbonated", "aerated drink", "cola", "coke", "juice", "juices", "squash", "sharbat", "energy drink", "sports drink", "mineral water", "packaged drinking water", "lemonade", "iced tea", "thums up", "sprite", "fanta", "limca", "maaza", "pepsi", "mirinda"], "generic_term": "beverage"},
|
||||
{"category": "Cooking Oils", "keywords": ["cooking oil", "edible oil", "sunflower oil", "mustard oil", "vanaspati", "refined oil", "oil", "oils"], "generic_term": "cooking oil"},
|
||||
# Loose-commodity categories. The names are chosen to match entries that
|
||||
# already exist in category_units.CATEGORY_UNIT_TYPE and in
|
||||
# enrichment/hsn_gst/models.HSN_GST_TABLE, so a bag of dal picks up its unit
|
||||
# rulebook and its HSN code without either table needing a new key. Listed
|
||||
# before "Cooking Oils", whose bare "oil" keyword is greedy, and before
|
||||
# "Atta & Staples", whose "dal"/"pulses"/"rice" keywords would otherwise
|
||||
# swallow every pulse.
|
||||
{"category": "Pulses, Grains & Spices", "keywords": ["dal", "dhal", "daal", "toor dal", "urad dal", "moong dal", "masoor dal", "chana dal", "arhar", "lentils", "lentil", "rajma", "pulses", "pulse"], "generic_term": "pulse"},
|
||||
{"category": "Spices & Masalas", "keywords": ["spices", "spice", "masala", "masalas", "turmeric", "haldi", "chilli powder", "coriander powder", "cumin", "jeera", "peppercorn", "black pepper", "cardamom", "asafoetida", "hing", "tamarind"], "generic_term": "spice"},
|
||||
{"category": "Sugar & Jaggery", "keywords": ["sugar", "jaggery", "gur", "brown sugar", "cane sugar", "misri"], "generic_term": "sweetener"},
|
||||
{"category": "Salt & Staples", "keywords": ["salt", "rock salt", "sea salt", "table salt", "iodised salt", "iodized salt"], "generic_term": "salt"},
|
||||
{"category": "Atta & Staples", "keywords": ["atta", "wheat flour", "flour", "rice", "dal", "pulses", "staples", "suji", "maida"], "generic_term": "staple product"},
|
||||
{"category": "Dairy", "keywords": ["milk", "dairy", "cheese", "paneer", "panner", "paner", "paneerr", "curd", "yogurt", "butter", "ghee", "dahi"], "generic_term": "dairy product"},
|
||||
# ---- Loose, unbranded fresh goods -----------------------------------
|
||||
# Added for the produce a grocer sells by weight or by the piece. These
|
||||
# rows carry no brand, so they land in the Own Products table via
|
||||
# generic_products.is_unbranded(); the categories exist so that HSN/GST
|
||||
# resolution and the pack-size unit rules have something to key off,
|
||||
# rather than falling through to "General".
|
||||
#
|
||||
# Keyword lists stay SHORT here on purpose. Detection for these rows comes
|
||||
# from the commodity lexicon, which is far more specific; a broad keyword
|
||||
# such as "fresh" or "leaf" would pull branded products in through
|
||||
# keyword matching, which is the failure this whole area exists to avoid.
|
||||
{"category": "Fruits & Vegetables", "keywords": ["fruits", "vegetables", "vegetable", "fresh produce", "loose produce"], "generic_term": "fresh produce"},
|
||||
{"category": "Fresh Herbs & Greens", "keywords": ["herbs", "greens", "curry leaves", "coriander leaves", "mint leaves", "spinach", "keerai"], "generic_term": "fresh greens"},
|
||||
{"category": "Flowers", "keywords": ["flowers", "flower", "garland", "jasmine flower", "loose flowers"], "generic_term": "flowers"},
|
||||
{"category": "Fish & Seafood", "keywords": ["fish", "seafood", "prawns", "prawn", "shrimp"], "generic_term": "fresh fish"},
|
||||
{"category": "Eggs", "keywords": ["eggs", "egg", "country egg"], "generic_term": "eggs"},
|
||||
{"category": "Oral Care", "keywords": ["toothpaste", "toothbrush", "mouthwash", "paste"], "generic_term": "oral care product"},
|
||||
{"category": "Hair Care", "keywords": ["shampoo", "shampooo", "conditioner", "hair oil"], "generic_term": "hair care product"},
|
||||
{"category": "Bath Soap", "keywords": ["bath soap", "soap bar", "soap", "soaps"], "generic_term": "soap"},
|
||||
{"category": "Skin & Bath Care", "keywords": ["face wash", "body lotion", "skin cream", "moisturizer", "body wash", "cream", "lotion"], "generic_term": "skin care product"},
|
||||
{"category": "Household Cleaning", "keywords": ["detergent", "laundry", "dishwash", "floor cleaner", "handwash", "cleaner"], "generic_term": "cleaning product"},
|
||||
# Laundry has its own category because the rest of the system already
|
||||
# names it: title_validator maps "detergent" to it, and HSN_GST_TABLE gives
|
||||
# it 3402 at 18% with no review flag. Folding it into "Household Cleaning"
|
||||
# made every detergent land on that table's needs_review row instead.
|
||||
# "fab" and bare "wash" are deliberately NOT keywords - Parle Fab is a
|
||||
# biscuit, and "wash" is inside face wash and body wash.
|
||||
{"category": "Detergents & Fabric Care", "keywords": ["detergent", "detergents", "washing powder", "detergent powder", "detergent bar", "laundry", "fabric wash", "fabric care"], "generic_term": "detergent"},
|
||||
{"category": "Household Cleaning", "keywords": ["dishwash", "floor cleaner", "handwash", "cleaner"], "generic_term": "cleaning product"},
|
||||
{"category": "Fragrance & Deodorants", "keywords": ["deodorant", "deo spray", "perfume", "fragrance", "body spray", "deo"], "generic_term": "fragrance product"},
|
||||
{"category": "Household - Agarbatti", "keywords": ["agarbatti", "incense sticks", "incense stick"], "generic_term": "agarbatti"},
|
||||
{"category": "Household - Lamp Oil", "keywords": ["lamp oil"], "generic_term": "lamp oil"},
|
||||
@@ -98,9 +144,25 @@ QUERY_STOP_WORDS = {
|
||||
}
|
||||
|
||||
|
||||
def _find_matches(text: str) -> List[Tuple[str, str, int]]:
|
||||
def _find_matches(text: str, exact_only: bool = False) -> List[Tuple[str, str, int]]:
|
||||
"""Return (category, matched_keyword, keyword_length) for every keyword
|
||||
found as a whole word/phrase in `text` (case-insensitive)."""
|
||||
found as a whole word/phrase in `text` (case-insensitive).
|
||||
|
||||
`exact_only` suppresses the fuzzy fallback below. It exists because that
|
||||
fallback is tuned for USER QUERIES, where a near-miss is a typo worth
|
||||
recovering, and is actively harmful when the text is a PRODUCT TITLE, where
|
||||
a near-miss is a different product. The case that forced it:
|
||||
|
||||
detect_category_from_text("Colgate-Palmolive Palmolive Naturals")
|
||||
-> "Chocolates"
|
||||
|
||||
because "colgate" scores 0.8 against the misspelling keyword "choclate".
|
||||
A toothpaste brand classified as chocolate is harmless in a search box and
|
||||
is not harmless in `consumability`, which would then hand a bar of soap a
|
||||
health score. Callers deciding something about a specific product pass
|
||||
exact_only=True; the RAG/query path keeps the fuzzy behaviour it was
|
||||
written for.
|
||||
"""
|
||||
if not text:
|
||||
return []
|
||||
lower = text.lower()
|
||||
@@ -113,7 +175,7 @@ def _find_matches(text: str) -> List[Tuple[str, str, int]]:
|
||||
matches.append((category, kw, len(kw)))
|
||||
|
||||
# Fuzzy matching fallback if exact word search found nothing
|
||||
if not matches:
|
||||
if not matches and not exact_only:
|
||||
import difflib
|
||||
words = re.findall(r"\b[a-z]{4,}\b", lower)
|
||||
for entry in CATEGORY_REGISTRY:
|
||||
@@ -130,15 +192,19 @@ def _find_matches(text: str) -> List[Tuple[str, str, int]]:
|
||||
return matches
|
||||
|
||||
|
||||
def detect_category_from_text(text: str) -> Optional[str]:
|
||||
def detect_category_from_text(text: str, exact_only: bool = False) -> Optional[str]:
|
||||
"""Infer a single canonical category from free text (typically a user
|
||||
query), or None if no category-identifying keyword is present.
|
||||
|
||||
When multiple categories match, the one listed earliest in
|
||||
`CATEGORY_REGISTRY` wins (see module docstring); ties within that are
|
||||
broken by the longest matched keyword.
|
||||
|
||||
Pass `exact_only=True` when `text` is a product title rather than a user
|
||||
query - see `_find_matches` for the Colgate/chocolate case that makes the
|
||||
distinction matter.
|
||||
"""
|
||||
matches = _find_matches(text)
|
||||
matches = _find_matches(text, exact_only=exact_only)
|
||||
if not matches:
|
||||
return None
|
||||
|
||||
|
||||
341
app/services/category_units.py
Normal file
341
app/services/category_units.py
Normal file
@@ -0,0 +1,341 @@
|
||||
"""
|
||||
Category -> Allowed Pack-Size-Unit Rules
|
||||
==========================================
|
||||
|
||||
WHY THIS FILE EXISTS
|
||||
---------------------
|
||||
Reported bug: brand="Britannia" produced catalog rows like
|
||||
"Tiger 10cm", "Nutrichoice 12cm", "Marie Gold 15cm" and
|
||||
"Britannia Milk Bikis - 200ml, 500ml, 1L". The small local LLM
|
||||
(qwen2.5:1.5b) occasionally predicts a *dimension* unit (cm/inch/m) or the
|
||||
*wrong physical-state* unit (ml/L for a solid biscuit) instead of a
|
||||
plausible FMCG pack-size unit, because nothing anywhere in the pipeline
|
||||
ever told it - or checked afterwards - which units are even legal for a
|
||||
given product category.
|
||||
|
||||
Two existing point-fixes come close but don't cover this:
|
||||
|
||||
- `price_estimator._parse_size_to_grams()` parses a size string into a
|
||||
gram/ml-equivalent float for pricing math. It has an explicit fallback
|
||||
for *unrecognised* unit tokens (case: "cm" is not in its litre/kg/base
|
||||
unit sets): "leave the numeric value as-is rather than guessing". That
|
||||
is the correct, conservative choice for a *pricing* function - but it
|
||||
means "10cm" silently becomes 10.0 (treated as if it were 10 grams),
|
||||
which is exactly wrong for a *validation* function: a bogus unit must
|
||||
never be treated as an implicitly-valid one.
|
||||
- `price_estimator.normalize_size_unit()` swaps ml<->g labels for
|
||||
categories it already knows are exclusively solid or exclusively liquid
|
||||
- but it only recognises `ml/l` <-> `g/kg` confusion. It has no concept
|
||||
of "cm" at all, so a dimension unit passes through completely
|
||||
unexamined.
|
||||
|
||||
This module is the missing, single source of truth for "what pack-size
|
||||
units are even legal for this product category", used in two places:
|
||||
|
||||
1. GENERATION TIME (app/services/ollama_service.py): to tell the LLM, as
|
||||
part of the prompt, exactly which units are allowed for the category it
|
||||
is currently enumerating products for - so the wrong unit is far less
|
||||
likely to be generated in the first place.
|
||||
2. VALIDATION TIME (app/core/catalog_engine.py, app/services/
|
||||
product_validator.py): to reject/replace a generated size string whose
|
||||
unit contradicts its resolved category, deterministically and without
|
||||
an LLM call, as a defence-in-depth backstop for whatever still gets
|
||||
through the prompt-level fix (e.g. the `fetch_brand_catalog_with_gemini`
|
||||
fallback path, or a cached/registry-sourced product).
|
||||
|
||||
Per the project's own stated requirement: package size must be determined
|
||||
from the PRODUCT CATEGORY, and dimension units (cm/inch/m) must never be
|
||||
generated for an FMCG consumable unless the category genuinely represents
|
||||
a physical dimension (none currently in this catalog do).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import Optional
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Unit vocabularies
|
||||
# ---------------------------------------------------------------------------
|
||||
WEIGHT_UNITS: set[str] = {
|
||||
"g", "gm", "gms", "gram", "grams",
|
||||
"kg", "kgs", "kilo", "kilos", "kilogram", "kilograms",
|
||||
}
|
||||
VOLUME_UNITS: set[str] = {
|
||||
"ml", "mls", "millilitre", "millilitres", "milliliter", "milliliters",
|
||||
"l", "lt", "ltr", "ltrs", "litre", "litres", "liter", "liters",
|
||||
}
|
||||
# Count-based packs (stationery, tablets, diapers, agarbatti sticks, etc.) -
|
||||
# legitimate for a handful of non-food categories this catalog also covers.
|
||||
COUNT_UNITS: set[str] = {
|
||||
"pcs", "pc", "piece", "pieces", "unit", "units", "tablet", "tablets",
|
||||
"capsule", "capsules", "strip", "strips", "sheet", "sheets", "roll",
|
||||
"rolls", "stick", "sticks", "count", "ct", "pack", "packs",
|
||||
}
|
||||
# Physical-dimension units. These are NEVER a valid FMCG/personal-care pack
|
||||
# size - no biscuit, chocolate, milk, tea, coffee, or ice-cream product is
|
||||
# ever sold "by the centimetre". Kept as an explicit set (rather than "any
|
||||
# unit we don't recognise") so the reject reason can name the exact problem.
|
||||
DIMENSION_UNITS: set[str] = {
|
||||
"cm", "centimetre", "centimetres", "centimeter", "centimeters",
|
||||
"mm", "millimetre", "millimetres", "millimeter", "millimeters",
|
||||
"m", "metre", "metres", "meter", "meters",
|
||||
"inch", "inches", "in",
|
||||
"ft", "feet", "foot",
|
||||
}
|
||||
|
||||
_KNOWN_UNIT_TOKENS: set[str] = WEIGHT_UNITS | VOLUME_UNITS | COUNT_UNITS | DIMENSION_UNITS
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Category -> allowed unit TYPE ("weight" / "volume" / "weight_or_volume" /
|
||||
# "count" / "unknown"). Keyed by lowercase, human-readable category strings -
|
||||
# the same vocabulary produced by app.services.brand_registry.
|
||||
# SUB_BRAND_CATEGORY and used as `known_category`/`category_value` throughout
|
||||
# catalog_engine.py. This is deliberately the AUTHORITATIVE per-category
|
||||
# rulebook the user specified: Biscuits -> g/kg, Milk/Juices -> ml/L,
|
||||
# Chocolate -> g, Tea -> g, Coffee -> g, Ice Cream -> ml/L - extended to
|
||||
# every other category already present in this project so every catalog row
|
||||
# gets a consistent check, not just the categories in the bug report.
|
||||
# ---------------------------------------------------------------------------
|
||||
CATEGORY_UNIT_TYPE: dict[str, str] = {
|
||||
# --- Weight-only (solid FMCG) ---
|
||||
"biscuits & cookies": "weight",
|
||||
"crackers": "weight",
|
||||
"rusk": "weight",
|
||||
"cakes & muffins": "weight",
|
||||
"bakery & breads": "weight",
|
||||
"snacks": "weight",
|
||||
"chocolates": "weight",
|
||||
"candy & confectionery": "weight",
|
||||
"atta & staples": "weight",
|
||||
"spices & masalas": "weight",
|
||||
# Loose commodities, sold by weight in every case.
|
||||
"pulses, grains & spices": "weight",
|
||||
"sugar & jaggery": "weight",
|
||||
"pasta & noodles": "weight",
|
||||
"noodles & instant food": "weight",
|
||||
"breakfast cereal": "weight",
|
||||
"dry fruits & nuts": "weight",
|
||||
"health foods": "weight",
|
||||
"millets": "weight",
|
||||
"salt & staples": "weight",
|
||||
"tea & coffee": "weight",
|
||||
"tea": "weight",
|
||||
"coffee": "weight",
|
||||
"health drinks": "weight", # malted/powder drinks: Horlicks, Bournvita, Boost, Complan
|
||||
"bath soap": "weight",
|
||||
"stationery": "count",
|
||||
"household - agarbatti": "count",
|
||||
"feminine hygiene": "count",
|
||||
"baby care": "weight_or_volume",
|
||||
# --- Volume-only (liquid FMCG) ---
|
||||
"beverages": "volume",
|
||||
"juices": "volume",
|
||||
"food - soups & sauces": "volume",
|
||||
"household cleaning": "volume",
|
||||
"dishwash": "volume",
|
||||
"ice cream": "volume",
|
||||
"cooking oils": "volume",
|
||||
"wellness oils": "volume",
|
||||
"household - lamp oil": "volume",
|
||||
# --- Mixed weight-or-volume (category legitimately spans both solid and
|
||||
# liquid sub-products, e.g. "Dairy" covers milk (ml) AND
|
||||
# cheese/paneer/butter/curd (g)) ---
|
||||
"dairy": "weight_or_volume",
|
||||
"dairy - desserts": "weight_or_volume",
|
||||
"hair care": "weight_or_volume", # shampoo/conditioner (ml) vs hair oil/cream (g/ml)
|
||||
"skin & bath care": "weight_or_volume",
|
||||
"skin care": "weight_or_volume",
|
||||
"beauty care": "weight_or_volume",
|
||||
"fragrance & deodorants": "weight_or_volume",
|
||||
"men's grooming": "weight_or_volume",
|
||||
"detergents & fabric care": "weight_or_volume", # powder (kg) vs liquid (L)
|
||||
"oral care": "weight_or_volume", # toothpaste (g) vs mouthwash (ml)
|
||||
"personal care": "weight_or_volume",
|
||||
"personal care - mosquito repellent": "weight_or_volume",
|
||||
"mosquito repellent": "weight_or_volume",
|
||||
"air freshener": "weight_or_volume",
|
||||
"health care - cold & cough": "weight_or_volume",
|
||||
"health care - digestive": "weight_or_volume",
|
||||
"health care - ayurvedic": "weight_or_volume",
|
||||
"health care - antiseptic": "weight_or_volume",
|
||||
"health care - first aid": "count",
|
||||
"pickles & chutneys": "weight",
|
||||
"food - spreads": "weight",
|
||||
"food - mixes": "weight",
|
||||
"rice & pulses": "weight",
|
||||
"sweets": "weight",
|
||||
"ready to eat": "weight_or_volume",
|
||||
}
|
||||
|
||||
# No category in this FMCG/personal-care catalog is legitimately sold "by
|
||||
# the centimetre" - kept as an explicit, extensible override point (e.g. a
|
||||
# future "Home Textiles" or "Furnishings" category) rather than a hardcoded
|
||||
# blanket rule, per the requirement that dimension units are only allowed
|
||||
# "if the product category genuinely represents physical dimensions".
|
||||
DIMENSION_OK_CATEGORIES: set[str] = set()
|
||||
|
||||
|
||||
def get_unit_type(category: Optional[str]) -> str:
|
||||
"""Return the allowed unit TYPE for `category` ('weight', 'volume',
|
||||
'weight_or_volume', 'count', or 'unknown' if the category isn't in our
|
||||
rulebook - callers should treat 'unknown' permissively, not as a
|
||||
rejection, since it just means we have no opinion yet, not that
|
||||
anything is wrong)."""
|
||||
if not category:
|
||||
return "unknown"
|
||||
return CATEGORY_UNIT_TYPE.get(category.strip().lower(), "unknown")
|
||||
|
||||
|
||||
def get_allowed_units(category: Optional[str]) -> set[str]:
|
||||
"""Concrete set of unit tokens allowed for `category`."""
|
||||
unit_type = get_unit_type(category)
|
||||
if unit_type == "weight":
|
||||
return set(WEIGHT_UNITS)
|
||||
if unit_type == "volume":
|
||||
return set(VOLUME_UNITS)
|
||||
if unit_type == "weight_or_volume":
|
||||
return WEIGHT_UNITS | VOLUME_UNITS
|
||||
if unit_type == "count":
|
||||
return set(COUNT_UNITS)
|
||||
# unknown category: permissive - anything except a dimension unit
|
||||
return WEIGHT_UNITS | VOLUME_UNITS | COUNT_UNITS
|
||||
|
||||
|
||||
def allowed_units_hint(category: Optional[str]) -> str:
|
||||
"""Short, human-readable allowed-units phrase for `category`, suitable
|
||||
for embedding directly into an LLM prompt, e.g. 'grams (g) or
|
||||
kilograms (kg)'. Used by ollama_service.py to tell the model, per
|
||||
category, exactly which units it must use."""
|
||||
unit_type = get_unit_type(category)
|
||||
return {
|
||||
"weight": "grams (g) or kilograms (kg) ONLY",
|
||||
"volume": "millilitres (ml) or litres (L) ONLY",
|
||||
"weight_or_volume": "grams/kilograms (g/kg) for solid items or millilitres/litres (ml/L) for liquid items",
|
||||
"count": "a piece/unit count (e.g. '10 pcs', '1 pack')",
|
||||
"unknown": "grams (g), kilograms (kg), millilitres (ml), or litres (L) as appropriate",
|
||||
}[unit_type]
|
||||
|
||||
|
||||
def is_forbidden_dimension_unit(unit: Optional[str], category: Optional[str] = None) -> bool:
|
||||
"""True if `unit` is a physical-dimension unit (cm/inch/m/...) that is
|
||||
never valid for `category` (see DIMENSION_OK_CATEGORIES)."""
|
||||
if not unit:
|
||||
return False
|
||||
u = unit.strip().lower()
|
||||
if u not in DIMENSION_UNITS:
|
||||
return False
|
||||
if category and category.strip().lower() in DIMENSION_OK_CATEGORIES:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def parse_unit(size: Optional[str]) -> tuple[Optional[float], Optional[str]]:
|
||||
"""Extract (numeric_value, unit_token) from a size string like '200g',
|
||||
'1.5 L', '10cm'. Returns (None, None) if no leading number is found."""
|
||||
if not size:
|
||||
return None, None
|
||||
s = size.strip().lower()
|
||||
match = re.search(r"(\d+(?:\.\d+)?)\s*([a-z]*)", s)
|
||||
if not match or not match.group(1):
|
||||
return None, None
|
||||
value = float(match.group(1))
|
||||
unit = match.group(2).strip() or None
|
||||
return value, unit
|
||||
|
||||
|
||||
def validate_unit_for_category(size: Optional[str], category: Optional[str]) -> tuple[bool, Optional[str]]:
|
||||
"""Check whether `size`'s unit is legal for `category`.
|
||||
|
||||
Returns (is_valid, reason). `reason` is populated whenever is_valid is
|
||||
False, explaining exactly what was wrong (dimension unit vs.
|
||||
wrong-physical-state unit vs. unrecognised token) so callers can log or
|
||||
surface it in an audit trail.
|
||||
"""
|
||||
if not size or not size.strip():
|
||||
return False, "size is missing"
|
||||
value, unit = parse_unit(size)
|
||||
if value is None:
|
||||
# No parseable number at all (e.g. "Family Pack") - not this
|
||||
# function's concern; price_estimator's own fallback handles it.
|
||||
return True, None
|
||||
if not unit:
|
||||
# Bare number, no unit token (e.g. "200") - ambiguous but not a
|
||||
# unit-category contradiction per se; leave to size-plausibility
|
||||
# checks elsewhere.
|
||||
return True, None
|
||||
|
||||
if is_forbidden_dimension_unit(unit, category):
|
||||
return False, (
|
||||
f"unit '{unit}' is a physical dimension, not a valid FMCG pack-size "
|
||||
f"unit for category '{category}'"
|
||||
)
|
||||
|
||||
if unit not in _KNOWN_UNIT_TOKENS:
|
||||
# Unrecognised token (e.g. "pcs" variants we don't track, "x6",
|
||||
# brand-specific packaging words) - not a contradiction we can
|
||||
# prove, so don't reject.
|
||||
return True, None
|
||||
|
||||
allowed = get_allowed_units(category)
|
||||
if unit not in allowed:
|
||||
unit_type = get_unit_type(category)
|
||||
return False, (
|
||||
f"unit '{unit}' is not valid for category '{category}' "
|
||||
f"(expects {unit_type.replace('_', ' ')} units: {allowed_units_hint(category)})"
|
||||
)
|
||||
return True, None
|
||||
|
||||
|
||||
def fix_or_reject_size(
|
||||
size: Optional[str], category: Optional[str], product_title: str = ""
|
||||
) -> tuple[Optional[str], bool, Optional[str]]:
|
||||
"""Authoritative size-unit correction, called at both generation time
|
||||
and validation time.
|
||||
|
||||
Returns (corrected_size_or_None, was_changed, reason):
|
||||
- If `size`'s unit is already valid for `category`: (size, False, None).
|
||||
- If `size` uses the WRONG PHYSICAL STATE for `category` (e.g. '200ml'
|
||||
for a strictly-solid category) but the number itself is plausible: the
|
||||
unit label is swapped (numeric value preserved), same behaviour as
|
||||
price_estimator.normalize_size_unit(), since a mislabelled-but-
|
||||
plausible quantity is safely recoverable.
|
||||
- If `size` uses a DIMENSION unit (cm/inch/m) or the unit can't be
|
||||
meaningfully reinterpreted for the category: there is no safe
|
||||
conversion (10cm has no defensible gram-equivalent), so this returns
|
||||
(None, True, reason) - the caller MUST NOT keep the original value and
|
||||
should substitute a category-appropriate default size instead (see
|
||||
price_estimator.default_size_variants()), never silently reuse the
|
||||
rejected number under a new label.
|
||||
"""
|
||||
if not size or not size.strip():
|
||||
return size, False, None
|
||||
|
||||
value, unit = parse_unit(size)
|
||||
if value is None or not unit:
|
||||
return size, False, None
|
||||
|
||||
ok, reason = validate_unit_for_category(size, category)
|
||||
if ok:
|
||||
return size, False, None
|
||||
|
||||
if is_forbidden_dimension_unit(unit, category):
|
||||
# No defensible conversion exists - reject outright.
|
||||
return None, True, reason
|
||||
|
||||
# Wrong-physical-state-but-plausible-number case: swap the label,
|
||||
# preserving the numeric value, mirroring the existing
|
||||
# price_estimator.normalize_size_unit() behaviour for ml<->g.
|
||||
unit_type = get_unit_type(category)
|
||||
value_str = (
|
||||
str(int(value)) if float(value).is_integer() else str(value)
|
||||
)
|
||||
if unit_type == "weight" and unit in VOLUME_UNITS:
|
||||
corrected = f"{value_str}g" if unit in {"ml", "mls"} else f"{value_str}kg"
|
||||
return corrected, True, reason
|
||||
if unit_type == "volume" and unit in WEIGHT_UNITS:
|
||||
corrected = f"{value_str}ml" if unit in {"g", "gm", "gms", "gram", "grams"} else f"{value_str}L"
|
||||
return corrected, True, reason
|
||||
|
||||
# Anything else we can't confidently repair (e.g. a count-unit category
|
||||
# given a weight unit) - reject rather than guess.
|
||||
return None, True, reason
|
||||
416
app/services/consumability.py
Normal file
416
app/services/consumability.py
Normal file
@@ -0,0 +1,416 @@
|
||||
"""
|
||||
Is this product something a person eats or drinks?
|
||||
|
||||
WHY THIS FILE EXISTS
|
||||
--------------------
|
||||
The nutrition module's contract is that every number it publishes traces to a
|
||||
verified source (see `nutrition_data_service`'s docstring). That contract says
|
||||
nothing about products which have no nutrition panel *at all*, and the only
|
||||
guard that existed was a seventeen-word substring list:
|
||||
|
||||
nutrition_data_service.NON_FOOD_KEYWORDS = ("soap", "detergent", "shampoo", ...)
|
||||
|
||||
matched against `f"{title} {category}".lower()`. It missed every category name it
|
||||
was not literally spelled with. Measured against the live catalogue, these rows
|
||||
carry a health score today:
|
||||
|
||||
Colgate-Palmolive Palmolive Naturals General score present
|
||||
Cavinkare Nyle / Nature's Hair Care score present
|
||||
P&G Pantene Hair Care score present
|
||||
Godrej Hit Spray Personal Care - Mosquito 291 kcal (!)
|
||||
|
||||
A mosquito repellent with a calorie count is not a cosmetic defect. It is the
|
||||
nutrition module asserting a fact about a product it has no business describing,
|
||||
and a shopper has no way to tell that the number is meaningless.
|
||||
|
||||
Being a substring test, that list also has the opposite failure: "soap" matches
|
||||
*soapnut* (reetha), a real commodity. Matching here is whole-word.
|
||||
|
||||
WHY THREE STATES AND NOT A BOOLEAN
|
||||
----------------------------------
|
||||
`is_consumable` and `is_non_consumable` are NOT inverses, and that is the single
|
||||
most important thing in this file.
|
||||
|
||||
Two callers ask opposite questions of the same fact:
|
||||
|
||||
the WRITE gate - "may I attach nutrition to this?" unknown => NO
|
||||
the DELETE gate - "may I destroy this row?" unknown => NO
|
||||
|
||||
Collapsing them into one boolean makes whichever caller loses the coin-toss act
|
||||
destructively on a guess: a single `not is_consumable()` in the purge script
|
||||
would delete every row the classifier merely failed to recognise. So the engine
|
||||
returns `CONSUMABLE | NON_CONSUMABLE | UNKNOWN`, and both booleans are positive
|
||||
tests that return False for UNKNOWN.
|
||||
|
||||
WHY A LAYERED DECISION AND NOT ONE KEYWORD LIST
|
||||
-----------------------------------------------
|
||||
The catalogue's category strings are not one vocabulary. Four maps have drifted
|
||||
apart - CATEGORY_REGISTRY (31 names), HSN_GST_TABLE (60), CATEGORY_UNIT_TYPE
|
||||
(~70) and CATEGORY_TYPE_WORDS - and the live tables hold 66 distinct values
|
||||
including "General" (40 rows) and the bare strings "1".."5" (17 rows) left by a
|
||||
bad import. Any single list is stale the moment somebody adds a category.
|
||||
|
||||
1. An explicit per-category verdict, hand-set, for every string the live
|
||||
catalogue actually contains.
|
||||
2. Failing that, the HSN chapter the category resolves to. The tariff's own
|
||||
classification, already maintained here for tax purposes.
|
||||
3. Failing that, the product title: first a non-food brand or product-line
|
||||
name (see `_NON_FOOD_LINE_WORDS`), then the same commodity lexicon and
|
||||
category detector the ingestion pipeline uses.
|
||||
4. Failing that, UNKNOWN - which deletes nothing, and which the write paths
|
||||
still enrich, so every non-food line that reaches it is a leak.
|
||||
|
||||
WHY THE HSN RANGE IS NOT SIMPLY 01-24
|
||||
--------------------------------------
|
||||
"Chapters 1 to 24 are the food chapters" is the obvious rule and it is wrong at
|
||||
exactly the case this project cares about. Chapter 06 is live plants and cut
|
||||
flowers, and `Flowers` is a live category here with 14 rows that reach the same
|
||||
Own Products table as the vegetables. A naive range would score a jasmine
|
||||
garland. Excluded for the same reason: 05 (inedible animal products), 14
|
||||
(vegetable plaiting materials), 23 (animal feed) and 24 (tobacco - consumed, but
|
||||
it publishes no nutrition panel).
|
||||
|
||||
EDGE CASES, AND WHY THEY WENT THE WAY THEY DID
|
||||
----------------------------------------------
|
||||
Flowers False. Sold loose beside the vegetables and routed by the same
|
||||
produce lexicon, but a garland is not food.
|
||||
Oral Care False. Toothpaste goes in the mouth and is spat out; it carries
|
||||
no nutrition panel, and it is Open *Beauty* Facts that matches it.
|
||||
Health Care - False. A cough syrup is ingested but is regulated as a drug and
|
||||
Cold & Cough / publishes dosage, not nutrition. Scoring it would be the most
|
||||
Digestive dangerous error available here.
|
||||
Baby Care MIXED, so it defers to the title. The audit found 21 live rows
|
||||
Health Care - that are Nestle Cerelac, Nan Pro and Lactogen sitting beside a
|
||||
Ayurvedic bottle of baby oil; Dabur Chyawanprash sits beside cough syrup.
|
||||
Infant formula is among the most nutrition-labelled food sold in
|
||||
India, so a blanket "not food" here would have been a worse
|
||||
defect than the one this module fixes.
|
||||
Health Drinks True. Horlicks, Boost and Complan publish a real nutrition
|
||||
panel, notwithstanding the word "Health".
|
||||
Household - False, and listed explicitly so it can never be swept in by
|
||||
Lamp Oil "Cooking Oils" being True. See `_hsn_verdict` for the second
|
||||
guard on the same trap.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from enum import Enum
|
||||
from typing import Dict, Optional, Tuple
|
||||
|
||||
from app.services.category_registry import _normalize, detect_category_from_text
|
||||
|
||||
|
||||
class Edibility(str, Enum):
|
||||
CONSUMABLE = "consumable"
|
||||
NON_CONSUMABLE = "non_consumable"
|
||||
UNKNOWN = "unknown"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class EdibilityVerdict:
|
||||
edibility: Edibility
|
||||
reason: str # human-readable; the purge audit prints this
|
||||
signal: str # category_map | hsn_chapter | title_brand_line | title_lexicon | title_keyword | none
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 1. Explicit verdicts
|
||||
# ---------------------------------------------------------------------------
|
||||
# Keyed on the NORMALIZED category ("Pulses, Grains & Spices" -> "pulses grains
|
||||
# and spices") so a stored string differing only in punctuation or case still
|
||||
# lands here rather than falling through to the HSN guess. `_normalize` is
|
||||
# imported rather than reimplemented: a fifth normalizer that disagreed with the
|
||||
# other four is exactly how this area got into trouble.
|
||||
_CONSUMABLE_CATEGORIES: Tuple[str, ...] = (
|
||||
# Dairy and dairy-adjacent
|
||||
"Dairy", "Cheese", "Dairy - Desserts", "Ice Cream",
|
||||
# Drinks
|
||||
"Beverages", "Tea & Coffee", "Health Drinks", "Food & Beverages",
|
||||
# Fresh, loose goods sold by weight or by the piece
|
||||
"Fruits & Vegetables", "Fresh Herbs & Greens", "Fish & Seafood", "Eggs",
|
||||
# Bakery and biscuit
|
||||
"Biscuits & Cookies", "Biscuits", "Crackers", "Rusk", "Cakes & Muffins",
|
||||
"Bakery & Breads", "Breakfast Cereal",
|
||||
# Confectionery
|
||||
"Chocolates", "Candy & Confectionery",
|
||||
# Savoury
|
||||
"Snacks", "Namkeen", "Ready to Eat", "Noodles & Instant Food",
|
||||
"Pasta & Noodles",
|
||||
# Staples, pulses, spices
|
||||
"Atta & Staples", "Staples", "Flour & Grains", "Salt & Staples",
|
||||
"Sugar & Jaggery", "Pulses, Grains & Spices", "Spices & Masalas",
|
||||
"Cooking Oils", "Dry Fruits & Nuts", "Millets", "Rice & Pulses",
|
||||
# Prepared foods
|
||||
"Food - Mixes", "Food - Spreads", "Food - Soups & Sauces",
|
||||
"Pickles & Chutneys", "Health Foods", "Sweets",
|
||||
)
|
||||
|
||||
_NON_CONSUMABLE_CATEGORIES: Tuple[str, ...] = (
|
||||
# Personal care
|
||||
"Hair Care", "Skin Care", "Skin & Bath Care", "Bath Soap", "Beauty Care",
|
||||
"Oral Care", "Fragrance & Deodorants", "Men's Grooming",
|
||||
"Feminine Hygiene", "Personal Care",
|
||||
# Household
|
||||
"Detergents & Fabric Care", "Dishwash", "Household Cleaning",
|
||||
"Household - Agarbatti", "Household - Lamp Oil", "Household - Air Freshener",
|
||||
"Personal Care - Mosquito Repellent",
|
||||
# Ingested, but regulated as medicines: they publish dosage, not nutrition
|
||||
"Health Care - Cold & Cough", "Health Care - Digestive",
|
||||
"Health Care - Antiseptic", "Health Care - First Aid",
|
||||
# Not food, despite arriving through the produce lexicon
|
||||
"Flowers",
|
||||
)
|
||||
|
||||
# Categories holding BOTH food and non-food, where a single verdict is simply
|
||||
# wrong. Deferred to the title, exactly like an uninformative category.
|
||||
#
|
||||
# Found by running the purge audit before deleting anything - which is what that
|
||||
# dry run is for. "Baby Care" held 21 live rows, and they are Nestle Cerelac,
|
||||
# Nan Pro and Lactogen alongside a bottle of baby oil. Infant formula and baby
|
||||
# cereal are among the most heavily nutrition-labelled products sold in India;
|
||||
# refusing them a score would have been a worse defect than the one being fixed.
|
||||
#
|
||||
# "Health Care - Ayurvedic" is the same shape: Dabur Chyawanprash is eaten by
|
||||
# the spoonful and carries a nutrition panel, while a cough syrup does not.
|
||||
_MIXED_CATEGORIES = frozenset(
|
||||
{_normalize("Baby Care"), _normalize("Health Care - Ayurvedic")}
|
||||
)
|
||||
|
||||
CATEGORY_VERDICTS: Dict[str, Edibility] = {
|
||||
**{_normalize(c): Edibility.CONSUMABLE for c in _CONSUMABLE_CATEGORIES},
|
||||
**{_normalize(c): Edibility.NON_CONSUMABLE for c in _NON_CONSUMABLE_CATEGORIES},
|
||||
}
|
||||
|
||||
# Category strings carrying no information. Treated as MISSING (ask the title),
|
||||
# not as UNKNOWN (refuse) - 40 live rows sit in "General" and many are real food.
|
||||
# The bare numerics "1".."5" are import damage on 17 rows; repairing those is a
|
||||
# catalogue fix, not a consumability rule, so they are recognised and reported
|
||||
# rather than accommodated.
|
||||
_UNINFORMATIVE_CATEGORIES = frozenset(
|
||||
{"", "general", "uncategorized", "uncategorised", "other", "others",
|
||||
"misc", "miscellaneous", "unknown", "na", "none"}
|
||||
)
|
||||
|
||||
_JUNK_CATEGORY_RE = re.compile(r"^\d+$")
|
||||
|
||||
# HSN chapters that are food and drink, per the customs tariff. Deliberately NOT
|
||||
# `range(1, 25)` - see the module docstring for why 05, 06, 14, 23 and 24 are out.
|
||||
_HSN_FOOD_CHAPTERS = frozenset(
|
||||
{1, 2, 3, 4, 7, 8, 9, 10, 11, 12, 13, 15, 16, 17, 18, 19, 20, 21, 22}
|
||||
)
|
||||
# Cosmetics, soap, pharma, insecticide, razors, hygiene articles.
|
||||
_HSN_NON_FOOD_CHAPTERS = frozenset({5, 6, 14, 23, 24, 28, 29, 30, 33, 34, 38, 82, 96})
|
||||
|
||||
# Title words that settle an uninformative category on their own. Deliberately
|
||||
# short: the commodity lexicon and the category detector do the real work, and a
|
||||
# longer list here would start overriding them.
|
||||
_NON_FOOD_TITLE_WORDS = frozenset({
|
||||
"soap", "detergent", "shampoo", "conditioner", "toothpaste", "toothbrush",
|
||||
"mouthwash", "deodorant", "perfume", "cosmetic", "lipstick", "kajal",
|
||||
"talc", "lotion", "moisturizer", "moisturiser", "sunscreen", "facewash",
|
||||
"handwash", "sanitizer", "sanitiser", "diaper", "sanitary", "napkin",
|
||||
"razor", "shaving", "cleaner", "disinfectant", "phenyl", "bleach",
|
||||
"repellent", "mosquito", "agarbatti", "incense", "camphor", "matchbox",
|
||||
"battery", "bulb", "candle", "polish", "freshener", "dishwash",
|
||||
})
|
||||
|
||||
# Title words that positively identify FOOD, consulted before the non-food list.
|
||||
# These exist for the mixed categories above: a product line name is the only
|
||||
# signal a title like "Nestle Cerelac 125g" carries, since the commodity lexicon
|
||||
# knows nothing of it. Naming specific ranges is consistent with how this
|
||||
# codebase already resolves ambiguity (BRAND_ALIASES, and the "bikis" keyword
|
||||
# added to Biscuits & Cookies so Britannia Milk Bikis is not filed as Dairy).
|
||||
_FOOD_TITLE_WORDS = frozenset({
|
||||
# infant and toddler nutrition
|
||||
"cerelac", "lactogen", "nangrow", "nan", "farex", "dexolac", "nusobee",
|
||||
"formula", "infant", "weaning", "porridge", "cereal", "cereals",
|
||||
# ayurvedic preparations eaten as food
|
||||
"chyawanprash", "chyavanprash", "honey", "malt",
|
||||
})
|
||||
|
||||
# Brand and product-line names sold ONLY as non-food, consulted before the food
|
||||
# words above. These exist for the uninformative "General" category: a title like
|
||||
# "Colgate-Palmolive Palmolive Naturals 350g" names no article at all, so neither
|
||||
# word list nor the lexicon recognises it, the verdict is UNKNOWN - and the write
|
||||
# paths refuse only NON_CONSUMABLE, so it was scored. Checked before the food
|
||||
# words because Palmolive sells a "Milk & Honey" range and "honey" is food.
|
||||
#
|
||||
# Deliberately conservative: a name any food range also uses stays out, because
|
||||
# a hit here outranks every food word. Left out on purpose: "dove" (Mars
|
||||
# chocolate), "himalaya" (supplements), "parachute" (edible coconut oil),
|
||||
# "hit", "wheel", "tide" (ordinary words). Only reached when the category says
|
||||
# nothing, so a real category always wins.
|
||||
_NON_FOOD_LINE_WORDS = frozenset({
|
||||
# personal care
|
||||
"palmolive", "colgate", "lifebuoy", "lux", "dettol", "savlon", "santoor",
|
||||
"cinthol", "pears", "hamam", "medimix", "margo", "fiama", "vivel", "rexona",
|
||||
"pantene", "sunsilk", "nivea", "vaseline", "ponds", "closeup", "pepsodent",
|
||||
"sensodyne",
|
||||
# household
|
||||
"ariel", "surf", "rin", "vim", "harpic", "lizol", "odonil", "goodknight",
|
||||
"allout", "mortein",
|
||||
})
|
||||
|
||||
_WORD_RE = re.compile(r"[a-z0-9]+")
|
||||
|
||||
|
||||
def _words(text: str) -> frozenset:
|
||||
"""Whole words, lowercased. Whole-word matching is the point: the old
|
||||
substring gate classified *soapnut* (reetha) as a soap."""
|
||||
return frozenset(_WORD_RE.findall((text or "").lower()))
|
||||
|
||||
|
||||
def _hsn_verdict(category: str) -> Optional[Edibility]:
|
||||
"""The verdict implied by the HSN chapter this category maps to.
|
||||
|
||||
Reads `HSN_GST_TABLE` directly rather than calling `resolve_hsn_gst`, and
|
||||
that is deliberate. `resolve_hsn_gst` falls through to `_KEYWORD_FALLBACKS`,
|
||||
which scans `f"{product_title} {category}"` and contains a greedy
|
||||
`("oil", "1517")` entry - enough to classify "Household - Lamp Oil" as an
|
||||
edible oil. Reading the table means only an exact category name can match,
|
||||
so the fallbacks can never fire here at all.
|
||||
|
||||
Imported lazily: the hsn_gst package pulls in the enrichment machinery, and
|
||||
the nutrition path should not pay for that import on every call.
|
||||
"""
|
||||
from app.services.enrichment.hsn_gst.models import HSN_GST_TABLE
|
||||
|
||||
target = _normalize(category)
|
||||
for name, entry in HSN_GST_TABLE.items():
|
||||
if _normalize(name) != target:
|
||||
continue
|
||||
try:
|
||||
chapter = int(str(entry[0])[:2])
|
||||
except (TypeError, ValueError, IndexError):
|
||||
return None
|
||||
if chapter in _HSN_FOOD_CHAPTERS:
|
||||
return Edibility.CONSUMABLE
|
||||
if chapter in _HSN_NON_FOOD_CHAPTERS:
|
||||
return Edibility.NON_CONSUMABLE
|
||||
return None
|
||||
return None
|
||||
|
||||
|
||||
def classify_edibility(category: Optional[str], title: str = "") -> EdibilityVerdict:
|
||||
"""The full verdict, with the reason that produced it.
|
||||
|
||||
The reason string is what the purge audit prints, so a row's fate can be
|
||||
argued with rather than taken on trust.
|
||||
"""
|
||||
normalized = _normalize(category)
|
||||
informative = (
|
||||
normalized
|
||||
and normalized not in _UNINFORMATIVE_CATEGORIES
|
||||
and normalized not in _MIXED_CATEGORIES
|
||||
and not _JUNK_CATEGORY_RE.match(normalized)
|
||||
)
|
||||
|
||||
if informative:
|
||||
verdict = CATEGORY_VERDICTS.get(normalized)
|
||||
if verdict is not None:
|
||||
return EdibilityVerdict(
|
||||
verdict,
|
||||
f"category {category!r} is listed as {verdict.value}",
|
||||
"category_map",
|
||||
)
|
||||
|
||||
hsn = _hsn_verdict(category or "")
|
||||
if hsn is not None:
|
||||
return EdibilityVerdict(
|
||||
hsn,
|
||||
f"category {category!r} maps to an HSN chapter that is "
|
||||
+ ("food/beverage" if hsn is Edibility.CONSUMABLE else "not food"),
|
||||
"hsn_chapter",
|
||||
)
|
||||
|
||||
# The category told us nothing usable. Ask the title, through the same two
|
||||
# resolvers the ingestion pipeline uses, so a row classified here agrees
|
||||
# with the row the pipeline would have written.
|
||||
text = (title or "").strip()
|
||||
if text:
|
||||
words = _words(text)
|
||||
|
||||
line_hit = words & _NON_FOOD_LINE_WORDS
|
||||
if line_hit:
|
||||
return EdibilityVerdict(
|
||||
Edibility.NON_CONSUMABLE,
|
||||
f"title {title!r} names a non-food brand or line "
|
||||
f"({sorted(line_hit)[0]!r})",
|
||||
"title_brand_line",
|
||||
)
|
||||
|
||||
food_hit = words & _FOOD_TITLE_WORDS
|
||||
if food_hit:
|
||||
return EdibilityVerdict(
|
||||
Edibility.CONSUMABLE,
|
||||
f"title {title!r} names a food product ({sorted(food_hit)[0]!r})",
|
||||
"title_keyword",
|
||||
)
|
||||
|
||||
hit = words & _NON_FOOD_TITLE_WORDS
|
||||
if hit:
|
||||
return EdibilityVerdict(
|
||||
Edibility.NON_CONSUMABLE,
|
||||
f"title {title!r} names a non-food article ({sorted(hit)[0]!r})",
|
||||
"title_keyword",
|
||||
)
|
||||
|
||||
from app.services.generic_products import canonical_category
|
||||
|
||||
# `detect_category_from_text` is pinned to exact_only. Its fuzzy
|
||||
# fallback scores "colgate" at 0.8 against the misspelling keyword
|
||||
# "choclate", so without this a tube of toothpaste reads as Chocolates
|
||||
# and earns a health score. Verified: that is the live behaviour for
|
||||
# "Colgate-Palmolive Palmolive Naturals", whose category is "General".
|
||||
for resolver in (
|
||||
canonical_category,
|
||||
lambda t: detect_category_from_text(t, exact_only=True),
|
||||
):
|
||||
detected = resolver(text)
|
||||
if not detected:
|
||||
continue
|
||||
verdict = CATEGORY_VERDICTS.get(_normalize(detected))
|
||||
if verdict is not None:
|
||||
return EdibilityVerdict(
|
||||
verdict,
|
||||
f"title {title!r} reads as {detected!r}, which is {verdict.value}",
|
||||
"title_lexicon",
|
||||
)
|
||||
|
||||
return EdibilityVerdict(
|
||||
Edibility.UNKNOWN,
|
||||
f"neither category {category!r} nor title {title!r} identifies this "
|
||||
"product; refusing to guess",
|
||||
"none",
|
||||
)
|
||||
|
||||
|
||||
def is_consumable(category: Optional[str], title: str = "") -> bool:
|
||||
"""True only when this is positively something a person eats or drinks.
|
||||
|
||||
The WRITE gate. False for UNKNOWN, so an unrecognised product gets no
|
||||
nutrition row and no health score - the same asymmetry
|
||||
`nutrition_data_service` already chose when it returns
|
||||
`data_status="unavailable"` rather than zeroes.
|
||||
"""
|
||||
return classify_edibility(category, title).edibility is Edibility.CONSUMABLE
|
||||
|
||||
|
||||
def is_non_consumable(category: Optional[str], title: str = "") -> bool:
|
||||
"""True only when this is positively NOT food.
|
||||
|
||||
The DELETE gate, and deliberately not `not is_consumable(...)`. False for
|
||||
UNKNOWN, so the purge script can never destroy a row it merely failed to
|
||||
recognise. See the module docstring.
|
||||
"""
|
||||
return classify_edibility(category, title).edibility is Edibility.NON_CONSUMABLE
|
||||
|
||||
|
||||
def is_junk_category(category: Optional[str]) -> bool:
|
||||
"""True for the bare numeric category strings left by a bad import.
|
||||
|
||||
Reported by the purge audit so the 17 affected rows get repaired in the
|
||||
catalogue rather than worked around here.
|
||||
"""
|
||||
return bool(_JUNK_CATEGORY_RE.match(_normalize(category)))
|
||||
1222
app/services/data/produce_reference.json
Normal file
1222
app/services/data/produce_reference.json
Normal file
File diff suppressed because it is too large
Load Diff
53605
app/services/data/usda_snapshot.json
Normal file
53605
app/services/data/usda_snapshot.json
Normal file
File diff suppressed because it is too large
Load Diff
50
app/services/enrichment/__init__.py
Normal file
50
app/services/enrichment/__init__.py
Normal file
@@ -0,0 +1,50 @@
|
||||
"""
|
||||
Product Enrichment Service
|
||||
===========================
|
||||
|
||||
WHY THIS PACKAGE EXISTS
|
||||
------------------------
|
||||
Everything upstream of this package (ollama_service.py, product_validator.py,
|
||||
sku_service.py, image_search.py, ...) either GENERATES a product row or
|
||||
JUDGES whether an already-generated row is plausible. None of it ever goes
|
||||
out and fetches a piece of ground-truth data from an authoritative external
|
||||
source and attaches it to the row. That is what this package is for.
|
||||
|
||||
The first concrete enricher is barcode lookup (see `barcode/`): resolving a
|
||||
real, checksum-valid EAN-13/UPC/GTIN for a catalog row from trusted external
|
||||
sources, never from the LLM. The package is deliberately structured so this
|
||||
is ONE stage among what will eventually be several independent ones (HSN
|
||||
code, GST rate, nutrition, allergens, manufacturer details, ...) - see
|
||||
`base.py` for the `EnrichmentStage` contract every future stage implements,
|
||||
and `pipeline.py` for the orchestrator that runs them in sequence.
|
||||
|
||||
DESIGN PRINCIPLES (apply to every enrichment stage, present and future)
|
||||
-------------------------------------------------------------------------
|
||||
- NEVER FABRICATE. If a stage cannot find authentic, verifiable data, it
|
||||
writes NULL/None + a "not_found" status - never a guessed or LLM-
|
||||
generated value. This mirrors the whole point of `product_validator.py`
|
||||
and `sku_service.py`'s "real ID or clearly-labelled internal fallback"
|
||||
design, taken one step further: for barcodes there IS no safe synthetic
|
||||
fallback (a fabricated barcode is actively harmful - it can collide with
|
||||
a real product), so the fallback is always NULL, never a generated value.
|
||||
- NEVER RAISE. A single product's enrichment failure (timeout, malformed
|
||||
response, source outage) must never abort the batch or the pipeline run.
|
||||
Every stage catches its own errors and degrades to "not found" for that
|
||||
one row.
|
||||
- ADDITIVE. This package has no dependents before this change and does not
|
||||
modify the behaviour of any existing module; it is only ever imported
|
||||
from new call sites (see `app/core/catalog_engine.py`'s Step 2.6 and
|
||||
`app/services/vector_store.py`'s additive barcode columns).
|
||||
- INDEPENDENT STAGES. Each stage owns its own external calls, matching
|
||||
rules, caching, and DB columns. A future HSN/GST/nutrition stage does not
|
||||
need to know barcode lookup exists, and vice versa - see `pipeline.py`.
|
||||
"""
|
||||
from app.services.enrichment.base import EnrichmentStage, StageOutcome
|
||||
from app.services.enrichment.pipeline import EnrichmentPipeline, run_default_pipeline
|
||||
|
||||
__all__ = [
|
||||
"EnrichmentStage",
|
||||
"StageOutcome",
|
||||
"EnrichmentPipeline",
|
||||
"run_default_pipeline",
|
||||
]
|
||||
47
app/services/enrichment/barcode/__init__.py
Normal file
47
app/services/enrichment/barcode/__init__.py
Normal file
@@ -0,0 +1,47 @@
|
||||
"""
|
||||
Barcode Retrieval & Product Enrichment module.
|
||||
|
||||
Public entry points:
|
||||
lookup_barcode(brand, product_title, size, category="") -> BarcodeResult
|
||||
Synchronous single-product cascading lookup. Safe to call from a
|
||||
script, notebook, or the Streamlit UI's "look up a barcode" action.
|
||||
|
||||
batch_lookup_barcodes(products, brand) -> list[BarcodeResult]
|
||||
Async batch lookup used by the pipeline stage (see `stage.py`) and
|
||||
available directly for a CLI/manual batch job.
|
||||
|
||||
See the package's module docstrings for the full architecture:
|
||||
models.py - BarcodeCandidate / BarcodeResult / enums
|
||||
validators.py - EAN-13/GTIN/UPC checksum + format validation
|
||||
matching.py - exact product matching rules
|
||||
cache.py - local SQLite lookup cache
|
||||
retry.py - tenacity-based retry/backoff
|
||||
sources/ - the 4-tier cascading source registry
|
||||
service.py - orchestrates cache -> sources -> validate -> match
|
||||
stage.py - adapts the service to the generic EnrichmentStage
|
||||
contract used by app/services/enrichment/pipeline.py
|
||||
"""
|
||||
from typing import Any, Dict, List
|
||||
|
||||
from app.services.enrichment.barcode.models import BarcodeCandidate, BarcodeResult, BarcodeType, LookupStatus
|
||||
from app.services.enrichment.barcode.service import BarcodeLookupService, get_default_service
|
||||
|
||||
|
||||
def lookup_barcode(brand: str, product_title: str, size: str, category: str = "") -> BarcodeResult:
|
||||
return get_default_service().lookup_one(brand, product_title, size, category)
|
||||
|
||||
|
||||
async def batch_lookup_barcodes(products: List[Dict[str, Any]], brand: str) -> List[BarcodeResult]:
|
||||
return await get_default_service().batch_lookup(products, brand)
|
||||
|
||||
|
||||
__all__ = [
|
||||
"BarcodeCandidate",
|
||||
"BarcodeResult",
|
||||
"BarcodeType",
|
||||
"LookupStatus",
|
||||
"BarcodeLookupService",
|
||||
"get_default_service",
|
||||
"lookup_barcode",
|
||||
"batch_lookup_barcodes",
|
||||
]
|
||||
127
app/services/enrichment/barcode/cache.py
Normal file
127
app/services/enrichment/barcode/cache.py
Normal file
@@ -0,0 +1,127 @@
|
||||
"""
|
||||
Local lookup cache for barcode results.
|
||||
|
||||
WHY A SEPARATE SQLITE FILE (data/cache/barcode_lookup.db) INSTEAD OF
|
||||
POSTGRES
|
||||
---------------------------------------------------------------------
|
||||
This cache exists purely to avoid repeating the SAME external-API search
|
||||
(GS1/Open Food Facts/UPCItemDB/manufacturer site) twice for the same
|
||||
(brand, product, size) - see "Cache barcode lookups to avoid repeated API
|
||||
calls" / "Prevent duplicate barcode searches" (Performance Requirements).
|
||||
It is deliberately NOT the authoritative store (that is Postgres, via
|
||||
`vector_store.py`'s additive barcode columns) - it needs to work even when
|
||||
Postgres is unreachable/USE_PGVECTOR=false, and it needs to be cheap and
|
||||
local on an 8GB-RAM/CPU-only machine, so plain stdlib `sqlite3` (no new
|
||||
dependency, no server process) is the right tool here, following the same
|
||||
"file-backed, atomic-write" spirit as `sku_service.py`'s SKU sequence
|
||||
counter.
|
||||
|
||||
A row's cache key is a normalized (brand, product_title, size) triple, not
|
||||
the barcode itself (we're caching "what did we already look up", not "what
|
||||
maps to what").
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
import sqlite3
|
||||
import threading
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
from app.services.enrichment.barcode.models import BarcodeResult
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_DB_PATH = Path("data") / "cache" / "barcode_lookup.db"
|
||||
_lock = threading.Lock()
|
||||
_initialized = False
|
||||
|
||||
|
||||
def _normalize_key_part(text: Optional[str]) -> str:
|
||||
return re.sub(r"\s+", " ", (text or "").strip().lower())
|
||||
|
||||
|
||||
def cache_key(brand: str, product_title: str, size: str) -> str:
|
||||
return f"{_normalize_key_part(brand)}|{_normalize_key_part(product_title)}|{_normalize_key_part(size)}"
|
||||
|
||||
|
||||
def _connect() -> sqlite3.Connection:
|
||||
_DB_PATH.parent.mkdir(parents=True, exist_ok=True)
|
||||
conn = sqlite3.connect(str(_DB_PATH), timeout=10)
|
||||
conn.execute("PRAGMA journal_mode=WAL")
|
||||
return conn
|
||||
|
||||
|
||||
def _ensure_schema(conn: sqlite3.Connection) -> None:
|
||||
global _initialized
|
||||
if _initialized:
|
||||
return
|
||||
conn.execute(
|
||||
"""
|
||||
CREATE TABLE IF NOT EXISTS barcode_lookup_cache (
|
||||
cache_key TEXT PRIMARY KEY,
|
||||
brand TEXT,
|
||||
product_title TEXT,
|
||||
size TEXT,
|
||||
result_json TEXT NOT NULL,
|
||||
created_at REAL NOT NULL
|
||||
)
|
||||
"""
|
||||
)
|
||||
conn.commit()
|
||||
_initialized = True
|
||||
|
||||
|
||||
def get_cached(brand: str, product_title: str, size: str, ttl_seconds: float) -> Optional[BarcodeResult]:
|
||||
"""Returns a cached BarcodeResult if present and not older than
|
||||
`ttl_seconds`, else None. Never raises - a cache read failure is
|
||||
treated exactly like a cache miss."""
|
||||
key = cache_key(brand, product_title, size)
|
||||
try:
|
||||
with _lock, _connect() as conn:
|
||||
_ensure_schema(conn)
|
||||
row = conn.execute(
|
||||
"SELECT result_json, created_at FROM barcode_lookup_cache WHERE cache_key = ?",
|
||||
(key,),
|
||||
).fetchone()
|
||||
except Exception as e:
|
||||
logger.debug(f"Barcode cache read failed for '{key}': {e}")
|
||||
return None
|
||||
|
||||
if not row:
|
||||
return None
|
||||
result_json, created_at = row
|
||||
if ttl_seconds > 0 and (time.time() - created_at) > ttl_seconds:
|
||||
return None
|
||||
try:
|
||||
data = json.loads(result_json)
|
||||
return BarcodeResult(**data)
|
||||
except Exception as e:
|
||||
logger.debug(f"Barcode cache entry for '{key}' unreadable, treating as miss: {e}")
|
||||
return None
|
||||
|
||||
|
||||
def set_cached(brand: str, product_title: str, size: str, result: BarcodeResult) -> None:
|
||||
"""Best-effort write - a failure here never blocks the lookup itself,
|
||||
it just means this row won't benefit from caching next time."""
|
||||
key = cache_key(brand, product_title, size)
|
||||
try:
|
||||
payload = json.dumps(result.__dict__)
|
||||
with _lock, _connect() as conn:
|
||||
_ensure_schema(conn)
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO barcode_lookup_cache (cache_key, brand, product_title, size, result_json, created_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?)
|
||||
ON CONFLICT(cache_key) DO UPDATE SET
|
||||
result_json = excluded.result_json,
|
||||
created_at = excluded.created_at
|
||||
""",
|
||||
(key, brand, product_title, size, payload, time.time()),
|
||||
)
|
||||
conn.commit()
|
||||
except Exception as e:
|
||||
logger.debug(f"Barcode cache write failed for '{key}': {e}")
|
||||
117
app/services/enrichment/barcode/identity_stage.py
Normal file
117
app/services/enrichment/barcode/identity_stage.py
Normal file
@@ -0,0 +1,117 @@
|
||||
"""Derives the rest of a product's barcode identity from the barcode itself.
|
||||
|
||||
WHY THIS IS A SEPARATE STAGE FROM BarcodeEnrichmentStage
|
||||
--------------------------------------------------------
|
||||
That stage FINDS a barcode, needs the network, and is off by default
|
||||
(`ENABLE_BARCODE_LOOKUP`, see settings.py:420-423 for why). This one FINDS
|
||||
NOTHING. It takes a barcode the row already has - typed into the merchant's
|
||||
spreadsheet, seeded from a catalog, or just located by the cascade - and fills
|
||||
in the fields that are pure arithmetic on those digits:
|
||||
|
||||
barcode_type from the length (classify_barcode_type)
|
||||
gtin the validated digits (a GTIN is what a barcode encodes)
|
||||
ean13 zero-padded UPC-A (to_ean13)
|
||||
upc the digits, for UPC-A only
|
||||
|
||||
There is no lookup, no host, no rate limit and no failure mode beyond "these
|
||||
digits are not a valid GTIN", so it needs no settings flag and costs nothing.
|
||||
|
||||
THE FAILURE IT ADDRESSES
|
||||
Measured against production on 2026-09-08: `upc` was 0.0% filled, `ean13`
|
||||
6.4%, `gtin` 8.7% - against `barcode` at 18.4%. Every one of those could
|
||||
have been computed from the barcode already sitting in the same row. They
|
||||
were not, because the only code that produced them was inside the disabled
|
||||
network cascade, and the writer dropped them anyway.
|
||||
|
||||
WHY IT RUNS AFTER THE LOOKUP STAGE
|
||||
So it also normalises whatever the cascade just found. The cascade already
|
||||
validates, but a sheet-supplied barcode never passes through
|
||||
`validate_barcode` at all today - it goes straight from the spreadsheet to
|
||||
the database. This stage is the first thing that checks those digits.
|
||||
|
||||
WHAT IT WILL NOT DO
|
||||
It will not correct, reformat or delete `barcode`. If the digits fail
|
||||
checksum validation the stage returns NOTHING, leaving the merchant's value
|
||||
exactly as typed - `enrichment/base.py`'s merge guard would refuse to blank
|
||||
it anyway, and silently "fixing" a barcode a shop supplied would be worse
|
||||
than leaving it visibly wrong. The failure is recorded in `field_sources`
|
||||
so the coverage report can surface it.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import time
|
||||
from typing import Any, Dict
|
||||
|
||||
from app.services.enrichment.base import EnrichmentStage, StageOutcome
|
||||
from app.services.enrichment.barcode.models import BarcodeType
|
||||
from app.services.enrichment.barcode.validators import (
|
||||
classify_barcode_type,
|
||||
normalize_barcode,
|
||||
to_ean13,
|
||||
validate_barcode,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class BarcodeIdentityStage(EnrichmentStage):
|
||||
"""Offline, deterministic, additive. Never raises, never erases."""
|
||||
|
||||
name = "barcode_identity"
|
||||
|
||||
async def enrich_one(self, product: Dict[str, Any], brand: str) -> StageOutcome:
|
||||
raw = product.get("barcode")
|
||||
if not str(raw or "").strip():
|
||||
return StageOutcome(stage_name=self.name, fields={})
|
||||
|
||||
code = validate_barcode(raw)
|
||||
if not code:
|
||||
# Not a GTIN. Say so in the provenance rather than in the data, and
|
||||
# leave `barcode` untouched.
|
||||
digits = normalize_barcode(raw)
|
||||
reason = (f"{len(digits)} digits is not a GTIN-8/12/13/14 length"
|
||||
if digits else "no digits in the value")
|
||||
return StageOutcome(
|
||||
stage_name=self.name,
|
||||
fields={"field_sources": {"barcode": {
|
||||
"method": "unvalidated",
|
||||
"source": product.get("barcode_source") or "sheet",
|
||||
"note": f"failed checksum/format validation: {reason}",
|
||||
}}},
|
||||
error=f"barcode {raw!r} failed validation: {reason}",
|
||||
)
|
||||
|
||||
barcode_type = classify_barcode_type(code)
|
||||
fields: Dict[str, Any] = {
|
||||
"barcode": code, # normalised digits, same value
|
||||
"barcode_type": barcode_type.value,
|
||||
"gtin": code,
|
||||
"ean13": to_ean13(code), # None for GTIN-8, which is not a short EAN-13
|
||||
"upc": code if barcode_type is BarcodeType.UPC_A else None,
|
||||
}
|
||||
|
||||
# Only claim provenance we can stand behind. A barcode that arrived on
|
||||
# the sheet is the merchant's assertion, not ours, and is emphatically
|
||||
# not "verified" - that word is reserved for the cascade's
|
||||
# brand+size+name-matched result.
|
||||
if not str(product.get("barcode_source") or "").strip():
|
||||
fields["barcode_source"] = "sheet"
|
||||
fields["barcode_lookup_status"] = "sheet_validated"
|
||||
fields["barcode_verified"] = False
|
||||
fields["barcode_last_updated"] = time.time()
|
||||
|
||||
fields["field_sources"] = {
|
||||
"barcode": {
|
||||
"method": "sourced" if product.get("barcode_verified") else "asserted",
|
||||
"source": product.get("barcode_source") or "sheet",
|
||||
},
|
||||
# These four are arithmetic on the barcode, never a lookup. Calling
|
||||
# them "sourced" would overstate them.
|
||||
"gtin": {"method": "derived", "source": "validators.validate_barcode"},
|
||||
"ean13": {"method": "derived", "source": "validators.to_ean13"},
|
||||
"upc": {"method": "derived", "source": "validators.classify_barcode_type"},
|
||||
"barcode_type": {"method": "derived", "source": "validators.classify_barcode_type"},
|
||||
}
|
||||
|
||||
return StageOutcome(stage_name=self.name, fields=fields)
|
||||
242
app/services/enrichment/barcode/matching.py
Normal file
242
app/services/enrichment/barcode/matching.py
Normal file
@@ -0,0 +1,242 @@
|
||||
"""
|
||||
Exact-product matching rules for barcode candidates.
|
||||
|
||||
A source returning SOME EAN-13 for a product with a similar name is not
|
||||
good enough - the "Product Matching Rules" requirement is explicit that
|
||||
brand, product name, variant, and size must all match, and that a
|
||||
different pack size (or a differently-branded variant like "Sugar-Free")
|
||||
must be REJECTED rather than accepted as "close enough". This module is
|
||||
deliberately conservative: a candidate only passes if every check agrees;
|
||||
`is_valid_ean13`-style tricks are not enough to earn a false positive here
|
||||
the way they might in a fuzzy search feature.
|
||||
|
||||
Deliberately stdlib-only (`difflib`, already in the standard library) -
|
||||
this project explicitly removed `fuzzywuzzy`/`python-Levenshtein` as dead
|
||||
weight (see requirements.txt), so no new fuzzy-matching dependency is
|
||||
introduced here either.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from difflib import SequenceMatcher
|
||||
from typing import Iterable, Optional
|
||||
|
||||
from app.services.enrichment.barcode.models import BarcodeCandidate
|
||||
# quantity_utils rather than image_search: this repo's image_search.py has no
|
||||
# quantity helpers, and it is a working file the store-catalog work leaves alone.
|
||||
from app.services.quantity_utils import quantities_match
|
||||
|
||||
# Words that mark a genuinely DIFFERENT retail variant from the plain/base
|
||||
# product - if the candidate's title contains one of these and the
|
||||
# target product's own title/product_name does NOT, the candidate is
|
||||
# rejected outright regardless of how well the rest of the name matches
|
||||
# (this is exactly the "Marie Gold 200g" vs "Marie Gold Sugar-Free"
|
||||
# example from the spec). Kept as a small, explicit, reviewable list
|
||||
# rather than a fuzzy heuristic - false negatives (missing a real match)
|
||||
# are far cheaper here than false positives (storing the wrong product's
|
||||
# barcode).
|
||||
VARIANT_DISTINGUISHING_TERMS = {
|
||||
"sugar free", "sugarfree", "sugar-free", "no sugar", "zero sugar",
|
||||
"diet", "family pack", "value pack", "jumbo pack", "combo pack",
|
||||
"combo", "gift pack", "gift box", "mini pack", "party pack",
|
||||
"refill pack", "refill", "pouch pack", "twin pack", "saver pack",
|
||||
"economy pack", "jar", "tin", "pet jar",
|
||||
}
|
||||
|
||||
_STOPWORDS = {"the", "and", "of", "with", "for", "a", "an", "new", "pack", "india"}
|
||||
|
||||
|
||||
def _normalize(text: Optional[str]) -> str:
|
||||
return re.sub(r"[^a-z0-9\s]", " ", (text or "").lower()).strip()
|
||||
|
||||
|
||||
def _tokens(text: Optional[str]) -> set:
|
||||
return {t for t in _normalize(text).split() if t and t not in _STOPWORDS}
|
||||
|
||||
|
||||
def brand_matches(candidate_brand: str, target_brand: str, brand_aliases: Optional[Iterable[str]] = None) -> bool:
|
||||
"""True if the candidate's reported brand plausibly refers to the same
|
||||
brand as the target. Accepts an exact/substring match on the target
|
||||
brand name itself, or a match against any known alias (e.g. "HUL" for
|
||||
"Hindustan Unilever") supplied by the caller via
|
||||
`brand_registry.get_brand_alias_set()`."""
|
||||
cand = _normalize(candidate_brand)
|
||||
target = _normalize(target_brand)
|
||||
if not cand or not target:
|
||||
return False
|
||||
if target in cand or cand in target:
|
||||
return True
|
||||
for alias in (brand_aliases or []):
|
||||
alias_n = _normalize(alias)
|
||||
if alias_n and (alias_n in cand or cand in alias_n):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def size_matches(candidate_size: str, target_size: str, tolerance: float = 0.03) -> bool:
|
||||
"""Tight tolerance (3%, vs. the 15% used for image matching elsewhere
|
||||
in this project) - a barcode belongs to exactly one pack size, so
|
||||
"close enough" is not an acceptable bar the way it is for reusing a
|
||||
product photo. Falls back to a plain normalized-string equality check
|
||||
when neither string parses as a numeric quantity (e.g. count-based
|
||||
sizes like "10 tablets"), rather than silently treating unparsable
|
||||
sizes as a match.
|
||||
"""
|
||||
if not candidate_size or not target_size:
|
||||
return False
|
||||
if quantities_match(candidate_size, target_size, tolerance=tolerance):
|
||||
return True
|
||||
return _normalize(candidate_size) == _normalize(target_size)
|
||||
|
||||
|
||||
def has_conflicting_variant_terms(candidate_title: str, target_title: str) -> bool:
|
||||
"""True if the candidate's title names a distinguishing variant
|
||||
(sugar-free, family pack, ...) that the target product does NOT -
|
||||
which means the candidate is a real but DIFFERENT product, not the one
|
||||
being looked up."""
|
||||
cand_n = _normalize(candidate_title)
|
||||
target_n = _normalize(target_title)
|
||||
for term in VARIANT_DISTINGUISHING_TERMS:
|
||||
if term in cand_n and term not in target_n:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def name_similarity(candidate_title: str, target_title: str) -> float:
|
||||
"""0.0-1.0 token-overlap-weighted similarity. Used only as a secondary
|
||||
signal / diagnostic (`match_confidence`) - never as the sole gate for
|
||||
acceptance; see `is_match()`."""
|
||||
cand_tokens, target_tokens = _tokens(candidate_title), _tokens(target_title)
|
||||
if not cand_tokens or not target_tokens:
|
||||
return 0.0
|
||||
overlap = len(cand_tokens & target_tokens) / len(target_tokens)
|
||||
seq_ratio = SequenceMatcher(None, _normalize(candidate_title), _normalize(target_title)).ratio()
|
||||
return round((overlap * 0.6) + (seq_ratio * 0.4), 3)
|
||||
|
||||
|
||||
def name_is_contained(candidate_title: str, target_title: str,
|
||||
target_brand: str = "") -> bool:
|
||||
"""True when the candidate's name is our name with only brand/size removed.
|
||||
|
||||
WHY THIS EXISTS - measured, not theoretical
|
||||
Open Food Facts stores short product names. We store long ones. Running
|
||||
`backfill_nutrition_from_barcodes` over the catalog on 2026-09-08, 149
|
||||
of 300 barcoded rows were rejected as "found, wrong product" when the
|
||||
barcode had resolved perfectly:
|
||||
|
||||
"Nestle Munch 8.9g" -> OFF "Munch" similarity 0.332
|
||||
"Coca-Cola Maaza 750ml" -> OFF "Maaza" similarity 0.304
|
||||
"Cadbury Perk 22 g" -> OFF "Perk" similarity 0.302
|
||||
|
||||
`name_similarity` divides the token overlap by the TARGET's token
|
||||
count, so a one-token candidate against a three-token target cannot
|
||||
exceed ~0.33 however right it is.
|
||||
|
||||
WHY NOT JUST LOWER THE THRESHOLD
|
||||
Because the same run also correctly rejected:
|
||||
|
||||
"Pepsico Lays 1kg" -> OFF "Spanish tomato tango" 0.133
|
||||
"Coca-Cola Fanta 750ml" -> OFF "Orange" 0.089
|
||||
"Lion Dates Powder 100g" -> OFF "PEPER NOTEN" 0.097
|
||||
|
||||
Those sit BELOW the containment cases but a threshold low enough to
|
||||
admit 0.30 also admits them. The measured yield table at
|
||||
settings.py:449-477 raised this floor to 0.78 for exactly that reason.
|
||||
Containment separates the two groups on structure rather than on a
|
||||
number: "Munch" is every token of our name minus brand and size;
|
||||
"Orange" is not a subset of "Coca-Cola Fanta 750ml" at all.
|
||||
|
||||
THE RULE
|
||||
Every token of the candidate's name must appear in the target's, once
|
||||
brand tokens and size tokens are discounted, and the candidate must
|
||||
carry at least one token that is not the brand. A bare brand name
|
||||
("Colgate", "godrej" - both real OFF titles) therefore does NOT match,
|
||||
which matters because those would otherwise attach to every product of
|
||||
that brand.
|
||||
"""
|
||||
cand_tokens = _tokens(candidate_title)
|
||||
target_tokens = _tokens(target_title)
|
||||
if not cand_tokens or not target_tokens:
|
||||
return False
|
||||
|
||||
brand_tokens = _tokens(target_brand)
|
||||
# A candidate that is only the brand identifies a brand, not a product.
|
||||
if not (cand_tokens - brand_tokens):
|
||||
return False
|
||||
|
||||
# Size tokens are not identity: our title carries the pack size, OFF's
|
||||
# usually does not, and `size_matches` has already checked the size
|
||||
# separately by the time this is consulted.
|
||||
def _meaningful(tokens):
|
||||
return {t for t in tokens if not _SIZE_TOKEN_RE.fullmatch(t)}
|
||||
|
||||
return _meaningful(cand_tokens) <= _meaningful(target_tokens | brand_tokens)
|
||||
|
||||
|
||||
# A token that is purely a quantity ("750ml", "8", "9g", "1kg"). Size is
|
||||
# compared by `size_matches`, so it must not also decide name identity.
|
||||
_SIZE_TOKEN_RE = re.compile(r"\d+(?:\.\d+)?(?:g|kg|ml|l|mg|cl|oz|gm|ltr|pcs|n)?", re.I)
|
||||
|
||||
|
||||
def is_match(candidate: BarcodeCandidate, target_brand: str, target_title: str, target_size: str,
|
||||
brand_aliases: Optional[Iterable[str]] = None,
|
||||
min_name_similarity: float = 0.45,
|
||||
barcode_is_identity: bool = False) -> tuple[bool, float]:
|
||||
"""The combined gate a candidate must pass to be accepted:
|
||||
1. Brand matches (or overlaps a known alias).
|
||||
2. Pack size matches within a tight tolerance.
|
||||
3. No conflicting variant terms (family pack / sugar-free / ...).
|
||||
4. Product-name similarity clears a floor - catches the case where
|
||||
brand+size coincidentally match but it's a completely different
|
||||
product line from the same brand.
|
||||
Returns (matched, confidence) - confidence is diagnostic only, stored
|
||||
on the result for audit/QA but never used to override rule 1-3.
|
||||
|
||||
`barcode_is_identity` says the caller already knows WHICH product this is,
|
||||
because it looked the candidate up BY its GTIN rather than by searching.
|
||||
That changes what rules 2 and 4 are for: they stop being evidence of
|
||||
identity and become sanity checks against our barcode being on the wrong
|
||||
row. A sanity check cannot fail on information the source does not have, so
|
||||
under this flag:
|
||||
|
||||
* rule 2 (size) - a BLANK candidate size no longer vetoes. Open Food
|
||||
Facts leaves `quantity` null on a large share of records (57 of 146
|
||||
Amul hits), and `size_matches` returns False whenever either side is
|
||||
blank. A record with no quantity does not disagree with our pack size;
|
||||
it says nothing about it. A quantity that is PRESENT and different
|
||||
still vetoes - that is our barcode pointing at the wrong pack.
|
||||
* rule 4 (name) - see `name_is_contained`.
|
||||
|
||||
Rules 1 and 3 are unaffected: a different brand, or a "sugar free" the
|
||||
target does not have, still means a different product.
|
||||
|
||||
It defaults to False because every relaxation here is unsafe on the SEARCH
|
||||
path, where many candidates compete and name and size are the only things
|
||||
telling them apart - "Munch" with no size would match every Nestle product
|
||||
containing that word. Pass True only where a single candidate was fetched
|
||||
by barcode. Today that is `fetch_verified_nutrition_by_barcode` and
|
||||
`scripts/backfill_nutrition_from_barcodes`, and nothing else.
|
||||
|
||||
Measured on 2026-09-08: of 300 barcoded catalog rows, 149 were refused as
|
||||
"found, wrong product" with the barcode resolving perfectly. The name gate
|
||||
was the visible symptom, but the SIZE gate rejected most of them first.
|
||||
"""
|
||||
if not brand_matches(candidate.candidate_brand, target_brand, brand_aliases):
|
||||
return False, 0.0
|
||||
# A blank candidate size is missing information, not a disagreement - but
|
||||
# only when the barcode already established identity. On the search path a
|
||||
# sizeless candidate is genuinely unidentifiable and must still be refused.
|
||||
size_unknown = barcode_is_identity and not str(candidate.candidate_size or "").strip()
|
||||
if not size_unknown and not size_matches(candidate.candidate_size, target_size):
|
||||
return False, 0.0
|
||||
if has_conflicting_variant_terms(candidate.candidate_title, target_title):
|
||||
return False, 0.0
|
||||
|
||||
similarity = name_similarity(candidate.candidate_title, target_title)
|
||||
if similarity < min_name_similarity:
|
||||
if barcode_is_identity and name_is_contained(
|
||||
candidate.candidate_title, target_title, target_brand):
|
||||
return True, similarity
|
||||
return False, similarity
|
||||
|
||||
return True, similarity
|
||||
82
app/services/enrichment/barcode/models.py
Normal file
82
app/services/enrichment/barcode/models.py
Normal file
@@ -0,0 +1,82 @@
|
||||
"""Data shapes shared across the barcode lookup module."""
|
||||
from __future__ import annotations
|
||||
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from enum import Enum
|
||||
from typing import Optional
|
||||
|
||||
|
||||
class BarcodeType(str, Enum):
|
||||
EAN13 = "EAN-13"
|
||||
UPC_A = "UPC-A"
|
||||
GTIN14 = "GTIN-14"
|
||||
GTIN8 = "GTIN-8"
|
||||
UNKNOWN = "UNKNOWN"
|
||||
|
||||
|
||||
class LookupStatus(str, Enum):
|
||||
"""Mirrors the `barcode_lookup_status` DB column."""
|
||||
VERIFIED = "verified" # found + checksum-valid + matched product
|
||||
NOT_FOUND = "not_found" # every source exhausted, nothing matched
|
||||
INVALID_CANDIDATE = "invalid_candidate" # a candidate was found but failed
|
||||
# checksum/format validation or product matching - rejected, not stored
|
||||
ERROR = "error" # a source/network error prevented a full search
|
||||
DISABLED = "disabled" # barcode lookup turned off via settings
|
||||
CACHED = "cached" # served from the local lookup cache
|
||||
|
||||
|
||||
@dataclass
|
||||
class BarcodeCandidate:
|
||||
"""A raw, not-yet-validated candidate returned by one source."""
|
||||
barcode: str
|
||||
source_name: str # e.g. "Open Food Facts"
|
||||
candidate_title: str = "" # the source's own product title, for matching
|
||||
candidate_brand: str = ""
|
||||
candidate_size: str = ""
|
||||
candidate_countries: str = "" # raw country tag/string from the source, if any
|
||||
|
||||
|
||||
@dataclass
|
||||
class BarcodeResult:
|
||||
"""Final, caller-facing result. Field names match the DB columns in
|
||||
`vector_store.py` and the JSON export shape requested for the
|
||||
catalog (`barcode`, `barcode_type`, `barcode_source`, `barcode_verified`,
|
||||
...) 1:1, so `service.py` -> product dict -> DB row -> JSON export is a
|
||||
straight field copy with no renaming at any layer.
|
||||
"""
|
||||
barcode: Optional[str] = None
|
||||
barcode_type: Optional[str] = None
|
||||
gtin: Optional[str] = None
|
||||
ean13: Optional[str] = None
|
||||
upc: Optional[str] = None
|
||||
barcode_source: Optional[str] = None
|
||||
barcode_verified: bool = False
|
||||
barcode_lookup_status: str = LookupStatus.NOT_FOUND.value
|
||||
barcode_last_updated: float = field(default_factory=time.time)
|
||||
match_confidence: float = 0.0
|
||||
|
||||
@classmethod
|
||||
def null_result(cls, status: LookupStatus = LookupStatus.NOT_FOUND) -> "BarcodeResult":
|
||||
"""The required NULL shape: barcode=NULL, barcode_verified=false,
|
||||
barcode_source=NULL - used for every path where no authentic,
|
||||
matching barcode could be confirmed. Never construct a
|
||||
BarcodeResult with a barcode value outside of `service.py`'s
|
||||
validated-and-matched path.
|
||||
"""
|
||||
return cls(barcode_lookup_status=status.value)
|
||||
|
||||
def as_product_fields(self) -> dict:
|
||||
"""Flat dict merged directly into the catalog row - see
|
||||
`stage.py`."""
|
||||
return {
|
||||
"barcode": self.barcode,
|
||||
"barcode_type": self.barcode_type,
|
||||
"gtin": self.gtin,
|
||||
"ean13": self.ean13,
|
||||
"upc": self.upc,
|
||||
"barcode_source": self.barcode_source,
|
||||
"barcode_verified": self.barcode_verified,
|
||||
"barcode_lookup_status": self.barcode_lookup_status,
|
||||
"barcode_last_updated": self.barcode_last_updated,
|
||||
}
|
||||
49
app/services/enrichment/barcode/retry.py
Normal file
49
app/services/enrichment/barcode/retry.py
Normal file
@@ -0,0 +1,49 @@
|
||||
"""
|
||||
Configurable retry-with-exponential-backoff for barcode source calls.
|
||||
|
||||
Built on `tenacity`, declared in requirements.txt. The import below is at
|
||||
module scope and this module sits on app/main.py's import path, so tenacity
|
||||
is a hard startup dependency, not an optional extra - without it the
|
||||
container exits 1 before uvicorn binds. Only retries on genuinely
|
||||
transient failures (network/timeout errors); a malformed response or a
|
||||
"no results" outcome is not retried, since retrying those wastes the
|
||||
source's rate-limit budget for no benefit (relevant for UPCItemDB's
|
||||
trial-tier daily cap in particular).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
import requests
|
||||
from tenacity import (
|
||||
retry,
|
||||
retry_if_exception_type,
|
||||
stop_after_attempt,
|
||||
wait_exponential,
|
||||
before_sleep_log,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_RETRYABLE_EXCEPTIONS = (
|
||||
requests.exceptions.ConnectionError,
|
||||
requests.exceptions.Timeout,
|
||||
requests.exceptions.ChunkedEncodingError,
|
||||
)
|
||||
|
||||
|
||||
def with_retry(max_attempts: int = 3, min_wait: float = 1.0, max_wait: float = 8.0):
|
||||
"""Decorator factory: `max_attempts` total tries, exponential backoff
|
||||
between `min_wait` and `max_wait` seconds. Applied per-source (see
|
||||
`sources/*.py`), not globally, so one slow/unreliable source retrying
|
||||
doesn't compound delay across the whole cascade - a source that keeps
|
||||
failing simply falls through to the next tier faster than a shared
|
||||
global retry budget would allow.
|
||||
"""
|
||||
return retry(
|
||||
reraise=True,
|
||||
stop=stop_after_attempt(max_attempts),
|
||||
wait=wait_exponential(multiplier=min_wait, max=max_wait),
|
||||
retry=retry_if_exception_type(_RETRYABLE_EXCEPTIONS),
|
||||
before_sleep=before_sleep_log(logger, logging.DEBUG),
|
||||
)
|
||||
180
app/services/enrichment/barcode/service.py
Normal file
180
app/services/enrichment/barcode/service.py
Normal file
@@ -0,0 +1,180 @@
|
||||
"""
|
||||
BarcodeLookupService - the cascading lookup orchestrator.
|
||||
|
||||
This is the ONE place that decides "is this barcode good enough to store".
|
||||
Every candidate from every source, regardless of tier, passes through the
|
||||
exact same two gates before it can become a `BarcodeResult`:
|
||||
1. `validators.validate_barcode()` - checksum + format.
|
||||
2. `matching.is_match()` - brand + size + variant + name.
|
||||
A source being "trusted" (e.g. GS1 India) does not skip either gate -
|
||||
trust only affects ORDER (which tier is tried first), never whether
|
||||
validation is required.
|
||||
|
||||
Cascade behaviour (see module docstring in `sources/__init__.py` for the
|
||||
tier list): tiers are tried in order; the first tier that produces at
|
||||
least one validated, matching candidate wins and the search stops - "The
|
||||
system should stop searching as soon as a verified barcode is found."
|
||||
Every tier is independently wrapped so a source outage never prevents
|
||||
falling through to the next one, and if every tier is exhausted with
|
||||
nothing matching, the result is the required NULL shape
|
||||
(`BarcodeResult.null_result()`), never a guess.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from app.infrastructure.settings import (
|
||||
ENABLE_BARCODE_LOOKUP,
|
||||
BARCODE_LOOKUP_CACHE_TTL_SECONDS,
|
||||
BARCODE_LOOKUP_MAX_CONCURRENCY,
|
||||
BARCODE_MIN_NAME_SIMILARITY,
|
||||
)
|
||||
from app.services.enrichment.barcode import cache
|
||||
from app.services.enrichment.barcode.matching import is_match
|
||||
from app.services.enrichment.barcode.models import BarcodeCandidate, BarcodeResult, BarcodeType, LookupStatus
|
||||
from app.services.enrichment.barcode.sources import get_default_sources
|
||||
from app.services.enrichment.barcode.validators import classify_barcode_type, to_ean13, validate_barcode
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
try:
|
||||
from app.services.brand_registry import get_brand_alias_set
|
||||
except Exception: # pragma: no cover - keeps the service importable/testable
|
||||
# in isolation even if brand_registry ever fails to import.
|
||||
def get_brand_alias_set(_brand: str) -> set:
|
||||
return set()
|
||||
|
||||
|
||||
def _build_result(barcode: str, source_name: str, confidence: float) -> BarcodeResult:
|
||||
btype = classify_barcode_type(barcode)
|
||||
ean13 = to_ean13(barcode) if btype in (BarcodeType.UPC_A, BarcodeType.EAN13) else None
|
||||
return BarcodeResult(
|
||||
barcode=barcode,
|
||||
barcode_type=btype.value,
|
||||
gtin=barcode,
|
||||
ean13=ean13,
|
||||
upc=barcode if btype == BarcodeType.UPC_A else None,
|
||||
barcode_source=source_name,
|
||||
barcode_verified=True,
|
||||
barcode_lookup_status=LookupStatus.VERIFIED.value,
|
||||
match_confidence=confidence,
|
||||
)
|
||||
|
||||
|
||||
class BarcodeLookupService:
|
||||
def __init__(self, sources=None):
|
||||
self.sources = sources if sources is not None else get_default_sources()
|
||||
|
||||
def lookup_one(self, brand: str, product_title: str, size: str, category: str = "",
|
||||
use_cache: bool = True) -> BarcodeResult:
|
||||
"""Synchronous cascading lookup for a single (brand, product,
|
||||
size). Never raises - every failure path returns a NULL-shaped
|
||||
`BarcodeResult` with a descriptive `barcode_lookup_status`."""
|
||||
if not ENABLE_BARCODE_LOOKUP:
|
||||
return BarcodeResult.null_result(LookupStatus.DISABLED)
|
||||
|
||||
if not brand or not product_title or not size:
|
||||
logger.debug(f"Barcode lookup skipped - missing brand/title/size ('{brand}', '{product_title}', '{size}')")
|
||||
return BarcodeResult.null_result(LookupStatus.ERROR)
|
||||
|
||||
if use_cache:
|
||||
cached = cache.get_cached(brand, product_title, size, BARCODE_LOOKUP_CACHE_TTL_SECONDS)
|
||||
if cached is not None:
|
||||
result = cached
|
||||
result.barcode_lookup_status = (
|
||||
LookupStatus.CACHED.value if result.barcode_verified else result.barcode_lookup_status
|
||||
)
|
||||
return result
|
||||
|
||||
brand_aliases = self._safe_brand_aliases(brand)
|
||||
had_source_error = False
|
||||
|
||||
for source in self.sources:
|
||||
if not source.available:
|
||||
continue
|
||||
try:
|
||||
raw_candidates = source.search(brand, product_title, size, category)
|
||||
except Exception as e:
|
||||
# Sources are documented to never raise, but this is the
|
||||
# pipeline-wide safety net referenced in base.py.
|
||||
logger.warning(f"[{source.name}] barcode search raised unexpectedly: {e}")
|
||||
had_source_error = True
|
||||
continue
|
||||
|
||||
result = self._first_validated_match(raw_candidates, brand, product_title, size, brand_aliases)
|
||||
if result is not None:
|
||||
if use_cache:
|
||||
cache.set_cached(brand, product_title, size, result)
|
||||
return result
|
||||
|
||||
status = LookupStatus.ERROR if had_source_error else LookupStatus.NOT_FOUND
|
||||
result = BarcodeResult.null_result(status)
|
||||
if use_cache:
|
||||
# Cache negative results too (Performance Requirements: "Prevent
|
||||
# duplicate barcode searches") - a short TTL still applies, so a
|
||||
# transient "not found" doesn't permanently block a later re-check.
|
||||
cache.set_cached(brand, product_title, size, result)
|
||||
return result
|
||||
|
||||
@staticmethod
|
||||
def _safe_brand_aliases(brand: str) -> set:
|
||||
try:
|
||||
return get_brand_alias_set(brand)
|
||||
except Exception:
|
||||
return set()
|
||||
|
||||
@staticmethod
|
||||
def _first_validated_match(raw_candidates: List[BarcodeCandidate], brand: str, product_title: str,
|
||||
size: str, brand_aliases: set) -> Optional[BarcodeResult]:
|
||||
for candidate in raw_candidates:
|
||||
clean_barcode = validate_barcode(candidate.barcode)
|
||||
if not clean_barcode:
|
||||
continue # invalid checksum/length/format - never stored, not even flagged
|
||||
# The floor comes from settings, not from matching.py's own 0.45
|
||||
# default. By this point the candidate's brand and pack size have
|
||||
# already matched, so the name is the only thing still telling two
|
||||
# products apart - and at 0.45 more than half the accepted matches
|
||||
# in this catalogue were the wrong product. See the setting for the
|
||||
# measured yield/error table.
|
||||
matched, confidence = is_match(
|
||||
candidate, brand, product_title, size, brand_aliases,
|
||||
min_name_similarity=BARCODE_MIN_NAME_SIMILARITY,
|
||||
)
|
||||
if not matched:
|
||||
continue
|
||||
return _build_result(clean_barcode, candidate.source_name, confidence)
|
||||
return None
|
||||
|
||||
async def batch_lookup(self, products: List[Dict[str, Any]], brand: str) -> List[BarcodeResult]:
|
||||
"""Async batch entry point used by `stage.py`. Bounded concurrency
|
||||
(`BARCODE_LOOKUP_MAX_CONCURRENCY`) so a large brand catalog doesn't
|
||||
fire dozens of simultaneous requests at any one source, and results
|
||||
are returned in the SAME ORDER as `products` regardless of which
|
||||
finished first, so callers can zip() them back together safely."""
|
||||
semaphore = asyncio.Semaphore(max(1, BARCODE_LOOKUP_MAX_CONCURRENCY))
|
||||
|
||||
async def _one(product: Dict[str, Any]) -> BarcodeResult:
|
||||
async with semaphore:
|
||||
return await asyncio.to_thread(
|
||||
self.lookup_one,
|
||||
brand,
|
||||
product.get("title") or product.get("product_name") or "",
|
||||
product.get("size") or "",
|
||||
product.get("category") or "",
|
||||
)
|
||||
|
||||
return await asyncio.gather(*(_one(p) for p in products))
|
||||
|
||||
|
||||
# Module-level singleton - sources are stateless, so one shared instance is
|
||||
# fine for both the sync CLI/test path and the async pipeline stage.
|
||||
_default_service: Optional[BarcodeLookupService] = None
|
||||
|
||||
|
||||
def get_default_service() -> BarcodeLookupService:
|
||||
global _default_service
|
||||
if _default_service is None:
|
||||
_default_service = BarcodeLookupService()
|
||||
return _default_service
|
||||
35
app/services/enrichment/barcode/sources/__init__.py
Normal file
35
app/services/enrichment/barcode/sources/__init__.py
Normal file
@@ -0,0 +1,35 @@
|
||||
"""
|
||||
Tier-ordered source registry - this list IS the cascade order described in
|
||||
the spec: GS1 India -> Open Food Facts -> trusted barcode databases ->
|
||||
manufacturer website -> (exhausted -> NULL). `service.py` iterates this
|
||||
list in order and stops at the first source that produces a validated,
|
||||
matching candidate.
|
||||
"""
|
||||
from app.services.enrichment.barcode.sources.base import BarcodeSource
|
||||
from app.services.enrichment.barcode.sources.gs1_india import GS1IndiaSource
|
||||
from app.services.enrichment.barcode.sources.open_food_facts import OpenFoodFactsSource
|
||||
from app.services.enrichment.barcode.sources.upc_database import UPCDatabaseSource
|
||||
from app.services.enrichment.barcode.sources.manufacturer_site import ManufacturerSiteSource
|
||||
|
||||
|
||||
def get_default_sources() -> list[BarcodeSource]:
|
||||
"""Fresh instances each call - sources are stateless/cheap to
|
||||
construct, and this avoids any shared-mutable-state surprises across
|
||||
concurrent batch lookups."""
|
||||
sources = [
|
||||
GS1IndiaSource(),
|
||||
OpenFoodFactsSource(),
|
||||
UPCDatabaseSource(),
|
||||
ManufacturerSiteSource(),
|
||||
]
|
||||
return sorted(sources, key=lambda s: s.tier)
|
||||
|
||||
|
||||
__all__ = [
|
||||
"BarcodeSource",
|
||||
"GS1IndiaSource",
|
||||
"OpenFoodFactsSource",
|
||||
"UPCDatabaseSource",
|
||||
"ManufacturerSiteSource",
|
||||
"get_default_sources",
|
||||
]
|
||||
46
app/services/enrichment/barcode/sources/base.py
Normal file
46
app/services/enrichment/barcode/sources/base.py
Normal file
@@ -0,0 +1,46 @@
|
||||
"""
|
||||
Pluggable barcode source contract, following the same adapter-registry
|
||||
pattern already used for marketplace SKU resolution
|
||||
(app/services/sku_service.py's `_MARKETPLACE_PATTERNS`) and the Product
|
||||
Resolver's adapter registry (see the Catalog_Project's `resolve_parent_brand`
|
||||
family) - every source is a self-contained class with one job: given a
|
||||
brand/title/size, return zero or more raw candidates. It does NOT decide
|
||||
whether a candidate is a match (that's `matching.py`) or whether its
|
||||
barcode is valid (that's `validators.py`) - keeping those concerns
|
||||
separate is what lets `service.py` apply the SAME validation/matching gate
|
||||
uniformly regardless of which tier produced the candidate.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from abc import ABC, abstractmethod
|
||||
from typing import List
|
||||
|
||||
from app.services.enrichment.barcode.models import BarcodeCandidate
|
||||
|
||||
|
||||
class BarcodeSource(ABC):
|
||||
#: Human-readable label stored in `barcode_source` on a successful
|
||||
#: match, e.g. "Open Food Facts", "GS1 India".
|
||||
name: str = "Unknown Source"
|
||||
|
||||
#: Lower = tried first in the cascade (see sources/__init__.py).
|
||||
tier: int = 99
|
||||
|
||||
@property
|
||||
def available(self) -> bool:
|
||||
"""False when the source is unusable for reasons unrelated to any
|
||||
one query (missing API key, disabled via settings, dependency not
|
||||
installed, ...). The cascade skips unavailable sources entirely
|
||||
rather than querying and failing every time."""
|
||||
return True
|
||||
|
||||
@abstractmethod
|
||||
def search(self, brand: str, product_title: str, size: str, category: str = "") -> List[BarcodeCandidate]:
|
||||
"""Best-effort search. MUST NOT raise - catch internally and
|
||||
return [] on any failure (network error, malformed response,
|
||||
nothing found). `service.py` treats an empty list and an
|
||||
exception identically (fall through to the next tier), so
|
||||
swallowing the error here vs. letting it propagate makes no
|
||||
behavioural difference except robustness - always swallow it.
|
||||
"""
|
||||
raise NotImplementedError
|
||||
91
app/services/enrichment/barcode/sources/gs1_india.py
Normal file
91
app/services/enrichment/barcode/sources/gs1_india.py
Normal file
@@ -0,0 +1,91 @@
|
||||
"""
|
||||
Tier 1: GS1 India / GEPIR (the official Indian GTIN registry).
|
||||
|
||||
HONESTY NOTE - READ BEFORE WIRING UP CREDENTIALS
|
||||
--------------------------------------------------
|
||||
GS1 India does not currently publish a free, public, no-registration JSON
|
||||
API for GTIN lookup (unlike Open Food Facts). Their real product-data
|
||||
registry is accessed either through GEPIR (https://gepir.gs1.org/ - a
|
||||
human web UI with no documented public API) or through GS1 India's paid
|
||||
"Verified by GS1" data-as-a-service product, which requires a commercial
|
||||
account and an API key.
|
||||
|
||||
Rather than scrape GEPIR's web UI (fragile, likely against its terms, and
|
||||
exactly the kind of "pretend this is a real integration" shortcut this
|
||||
project's whole hallucination-prevention philosophy exists to avoid - see
|
||||
product_validator.py's module docstring), this adapter is a REAL,
|
||||
WORKING integration point that:
|
||||
- does nothing (returns [], `available=False`) when no credentials are
|
||||
configured, so the cascade cleanly falls through to Open Food Facts
|
||||
(Tier 2) - exactly the "if not found -> next step" behaviour the
|
||||
spec asks for, just starting from Tier 2 in practice until GS1 India
|
||||
API access is provisioned;
|
||||
- immediately becomes live the moment `GS1_INDIA_API_BASE_URL` and
|
||||
`GS1_INDIA_API_KEY` are set (see .env.example), with no code changes
|
||||
needed elsewhere - `service.py` already treats every tier uniformly.
|
||||
|
||||
If/when real GS1 India API access is provisioned, fill in the request
|
||||
shape in `search()` below to match that API's actual contract (endpoint
|
||||
path, auth header, response schema) - the surrounding plumbing
|
||||
(validation, matching, caching, retry) does not need to change.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import List
|
||||
|
||||
import requests
|
||||
|
||||
from app.infrastructure.settings import GS1_INDIA_API_BASE_URL, GS1_INDIA_API_KEY, BARCODE_LOOKUP_TIMEOUT_SECONDS
|
||||
from app.services.enrichment.barcode.models import BarcodeCandidate
|
||||
from app.services.enrichment.barcode.retry import with_retry
|
||||
from app.services.enrichment.barcode.sources.base import BarcodeSource
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class GS1IndiaSource(BarcodeSource):
|
||||
name = "GS1 India"
|
||||
tier = 1
|
||||
|
||||
@property
|
||||
def available(self) -> bool:
|
||||
return bool(GS1_INDIA_API_BASE_URL and GS1_INDIA_API_KEY)
|
||||
|
||||
def search(self, brand: str, product_title: str, size: str, category: str = "") -> List[BarcodeCandidate]:
|
||||
if not self.available:
|
||||
return []
|
||||
|
||||
@with_retry(max_attempts=3)
|
||||
def _call():
|
||||
return requests.get(
|
||||
f"{GS1_INDIA_API_BASE_URL.rstrip('/')}/search",
|
||||
params={"brand": brand, "product": product_title, "size": size, "country": "India"},
|
||||
headers={"Authorization": f"Bearer {GS1_INDIA_API_KEY}"},
|
||||
timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS,
|
||||
)
|
||||
|
||||
try:
|
||||
resp = _call()
|
||||
if resp.status_code != 200:
|
||||
logger.debug(f"GS1 India lookup non-200 ({resp.status_code}) for '{brand} {product_title} {size}'")
|
||||
return []
|
||||
data = resp.json()
|
||||
except Exception as e:
|
||||
logger.debug(f"GS1 India lookup failed for '{brand} {product_title} {size}': {e}")
|
||||
return []
|
||||
|
||||
candidates: List[BarcodeCandidate] = []
|
||||
for item in (data.get("results") or data.get("items") or []):
|
||||
gtin = item.get("gtin") or item.get("barcode") or item.get("code")
|
||||
if not gtin:
|
||||
continue
|
||||
candidates.append(BarcodeCandidate(
|
||||
barcode=str(gtin),
|
||||
source_name=self.name,
|
||||
candidate_title=item.get("productName") or item.get("title") or "",
|
||||
candidate_brand=item.get("brandName") or item.get("brand") or "",
|
||||
candidate_size=item.get("netContent") or item.get("size") or "",
|
||||
candidate_countries="in",
|
||||
))
|
||||
return candidates
|
||||
108
app/services/enrichment/barcode/sources/manufacturer_site.py
Normal file
108
app/services/enrichment/barcode/sources/manufacturer_site.py
Normal file
@@ -0,0 +1,108 @@
|
||||
"""
|
||||
Tier 4: manufacturer / official product page - last resort.
|
||||
|
||||
Follows the exact same "DuckDuckGo search -> read structure straight off
|
||||
the result, no page-scraping HTML parser" spirit as
|
||||
`app/services/sku_service.py`'s `find_website_product_id()`, except here
|
||||
we DO need to fetch the page text (a barcode isn't embedded in the result
|
||||
URL the way an Amazon ASIN is), so this is the one tier that performs a
|
||||
capped number of lightweight page fetches, each wrapped in its own
|
||||
try/except so one slow/broken page can never block the others.
|
||||
|
||||
This tier is intentionally the lowest-trust one: a number that merely sits
|
||||
near the word "barcode"/"EAN"/"UPC"/"GTIN" on a webpage is only a
|
||||
CANDIDATE. It still has to pass `validators.validate_barcode()` (checksum)
|
||||
and `matching.is_match()` (brand/size/name) in `service.py` exactly like
|
||||
every other tier's candidates - nothing here is trusted just because it
|
||||
came from what looks like an official page.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
from typing import List
|
||||
|
||||
import requests
|
||||
|
||||
from app.infrastructure.settings import BARCODE_LOOKUP_TIMEOUT_SECONDS, ENABLE_MANUFACTURER_SITE_LOOKUP
|
||||
from app.services.enrichment.barcode.models import BarcodeCandidate
|
||||
from app.services.enrichment.barcode.retry import with_retry
|
||||
from app.services.enrichment.barcode.sources.base import BarcodeSource
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_BROWSER_UA = (
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
|
||||
"(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
|
||||
)
|
||||
|
||||
# A barcode label followed, within a short distance, by an 8/12/13/14-digit
|
||||
# run. Intentionally permissive on the label (source pages phrase this
|
||||
# differently) and tight on the digit run (only plausible GTIN lengths) -
|
||||
# every hit is still just a CANDIDATE, checksum-validated afterwards.
|
||||
_BARCODE_NEAR_LABEL_RE = re.compile(
|
||||
r"(?:barcode|ean|upc|gtin)\D{0,15}(\d{8}|\d{12,14})",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_MAX_PAGES = 3
|
||||
|
||||
|
||||
class ManufacturerSiteSource(BarcodeSource):
|
||||
name = "Manufacturer Website"
|
||||
tier = 4
|
||||
|
||||
@property
|
||||
def available(self) -> bool:
|
||||
if not ENABLE_MANUFACTURER_SITE_LOOKUP:
|
||||
return False
|
||||
try:
|
||||
import ddgs # noqa: F401
|
||||
return True
|
||||
except ImportError:
|
||||
logger.debug("Manufacturer-site barcode lookup unavailable: 'ddgs' package not installed")
|
||||
return False
|
||||
|
||||
def search(self, brand: str, product_title: str, size: str, category: str = "") -> List[BarcodeCandidate]:
|
||||
if not self.available:
|
||||
return []
|
||||
|
||||
query = f"{brand} {product_title} {size} barcode EAN".strip()
|
||||
try:
|
||||
from ddgs import DDGS
|
||||
with DDGS(timeout=10) as ddgs:
|
||||
results = ddgs.text(query, region="in-en", safesearch="off", max_results=_MAX_PAGES)
|
||||
except Exception as e:
|
||||
logger.debug(f"Manufacturer-site search failed for '{query}': {e}")
|
||||
return []
|
||||
|
||||
candidates: List[BarcodeCandidate] = []
|
||||
for r in (results or [])[:_MAX_PAGES]:
|
||||
url = str(r.get("href") or r.get("url") or "")
|
||||
if not url.startswith("http"):
|
||||
continue
|
||||
for code in self._extract_barcodes(url):
|
||||
candidates.append(BarcodeCandidate(
|
||||
barcode=code,
|
||||
source_name=f"{self.name} ({url})",
|
||||
candidate_title=r.get("title") or product_title,
|
||||
candidate_brand=brand,
|
||||
candidate_size=size,
|
||||
candidate_countries="",
|
||||
))
|
||||
return candidates
|
||||
|
||||
def _extract_barcodes(self, url: str) -> List[str]:
|
||||
@with_retry(max_attempts=2)
|
||||
def _call():
|
||||
return requests.get(url, headers={"User-Agent": _BROWSER_UA}, timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS)
|
||||
|
||||
try:
|
||||
resp = _call()
|
||||
if resp.status_code != 200:
|
||||
return []
|
||||
text = resp.text[:200_000] # cap - this is a text scan, not a full-page render
|
||||
except Exception as e:
|
||||
logger.debug(f"Manufacturer page fetch failed for '{url}': {e}")
|
||||
return []
|
||||
|
||||
return [m.group(1) for m in _BARCODE_NEAR_LABEL_RE.finditer(text)]
|
||||
546
app/services/enrichment/barcode/sources/off_bulk.py
Normal file
546
app/services/enrichment/barcode/sources/off_bulk.py
Normal file
@@ -0,0 +1,546 @@
|
||||
"""
|
||||
Bulk Open Food Facts brand-corpus fetch + offline name matching.
|
||||
|
||||
WHY THIS EXISTS ALONGSIDE `open_food_facts.py`
|
||||
----------------------------------------------
|
||||
`sources/open_food_facts.py` searches OFF **once per product** against
|
||||
`/cgi/search.pl`. OFF rate-limits that endpoint to 10 requests/minute, which
|
||||
makes a 600-product backfill a multi-hour job, and `stage.py` records that the
|
||||
resulting cascade "misses far more often than it hits". That module is wired
|
||||
into the live enrichment pipeline and is deliberately NOT touched by this file.
|
||||
|
||||
This module inverts the problem: fetch a brand's **entire** OFF catalogue in one
|
||||
or two requests from the v2 search API, cache it on disk, then match every one
|
||||
of our products against that corpus offline. A whole-catalogue backfill costs
|
||||
~5 HTTP requests instead of ~600, and re-tuning the similarity threshold costs
|
||||
zero network because the corpus is cached.
|
||||
|
||||
`fetch_brand_corpus` is the one network function here and is shared by the
|
||||
backfill scripts, `product_grounding`, `post_ingest_barcodes` and
|
||||
`brand_discovery`. Everything else is a pure function so it can be unit-tested
|
||||
without network or database.
|
||||
|
||||
ENDPOINT NOTES (verified empirically, 2026-09-11)
|
||||
GET https://world.openfoodfacts.org/api/v2/search
|
||||
?brands_tags=amul
|
||||
&countries_tags=en:india
|
||||
&fields=code,product_name,quantity,...
|
||||
&page_size=100&page=1
|
||||
* `brands_tags` is an EXACT match on the brand's tag slug ("naga", not
|
||||
"Naga"), which is what a brand catalogue needs. The earlier Search-a-licious
|
||||
query `q=brands:Naga` was a free-text match: it returned "Mr Naga" and
|
||||
"Bombay Naga Jhal" - other companies - and MISSED the real Naga rows,
|
||||
because Search-a-licious reads a separate Elasticsearch index that was
|
||||
stale for them (barcode 8906011830068 was still filed brandless under
|
||||
Kuwait). The v2 API reads the live product database. Measured on the same
|
||||
day: Amul 216 products here vs 140 there; Naga 2 vs 0.
|
||||
* `page_count` in this API is the number of products ON THIS PAGE, not the
|
||||
number of pages, and `page_size` in the RESPONSE is what the server
|
||||
actually applied (a larger request is capped to 100 without complaint).
|
||||
Paginate from `count` / that served `page_size`.
|
||||
* OFF rate-limits all search endpoints to 10 requests/minute per IP.
|
||||
PAUSE_SECONDS keeps a multi-page brand under that.
|
||||
* `world.openfoodfacts.org` intermittently serves an HTML "Page temporarily
|
||||
unavailable" page with a 200 status, and plain 503s under load, so every
|
||||
response is status- and content-type-checked before parsing, and a failed
|
||||
fetch is never written to the cache as an empty corpus.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import math
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, Iterable, List, Optional, Sequence, Tuple
|
||||
|
||||
import requests
|
||||
|
||||
from app.infrastructure.settings import (
|
||||
BARCODE_COUNTRY_TAG,
|
||||
BARCODE_LOOKUP_TIMEOUT_SECONDS,
|
||||
)
|
||||
from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand
|
||||
from app.services.enrichment.barcode.matching import (
|
||||
has_conflicting_variant_terms,
|
||||
name_similarity,
|
||||
)
|
||||
from app.services.enrichment.barcode.models import BarcodeType
|
||||
from app.services.enrichment.barcode.retry import with_retry
|
||||
from app.services.enrichment.barcode.validators import (
|
||||
classify_barcode_type,
|
||||
to_ean13,
|
||||
validate_barcode,
|
||||
)
|
||||
from app.services.quantity_utils import quantities_match
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
SEARCH_URL = "https://world.openfoodfacts.org/api/v2/search"
|
||||
|
||||
# OFF's usage policy requires a contactable custom User-Agent; requests sent
|
||||
# with the default python-requests agent are treated as anonymous crawling.
|
||||
USER_AGENT = "BrandCatalogRAG/1.0 (suriya@tenext.in)"
|
||||
|
||||
FIELDS = "code,product_name,product_name_en,brands,quantity,countries_tags"
|
||||
PAGE_SIZE = 100 # the v2 API caps this at 100 and silently serves that
|
||||
MAX_PAGES = 25 # 2500 products per brand is far beyond any real brand
|
||||
PAUSE_SECONDS = 6.0 # 10 search requests/minute is OFF's published limit
|
||||
|
||||
# Bumped when the cache file's meaning changes. Files written before this key
|
||||
# existed came from the Search-a-licious endpoint; an EMPTY one of those is
|
||||
# more likely to be that endpoint's stale index (or a swallowed 503) than a
|
||||
# real absence, so it is re-fetched. A non-empty one is real data and is kept.
|
||||
CACHE_SCHEMA = 2
|
||||
|
||||
# backend/app/services/enrichment/barcode/sources/off_bulk.py -> backend/
|
||||
_BACKEND_DIR = Path(__file__).resolve().parents[5]
|
||||
CACHE_DIR = _BACKEND_DIR / "data" / "cache" / "off_brand_corpus"
|
||||
|
||||
|
||||
class OffUnavailable(requests.exceptions.ConnectionError):
|
||||
"""OFF answered but could not serve (429 / 5xx).
|
||||
|
||||
Subclasses ConnectionError so `retry.with_retry` treats it exactly like a
|
||||
dropped socket - a few seconds later the same request usually succeeds -
|
||||
without widening the retry set for every other barcode source.
|
||||
"""
|
||||
|
||||
|
||||
@dataclass
|
||||
class CorpusFetch:
|
||||
"""What `fetch_brand_corpus_result` learned.
|
||||
|
||||
`error` is None when OFF answered every page, even if it answered with
|
||||
nothing - that is a real "this brand is not on Open Food Facts". When
|
||||
`error` is set the fetch did not complete and NOTHING was cached, so a
|
||||
caller can say "could not be reached" instead of "has nothing".
|
||||
"""
|
||||
hits: List[Dict[str, Any]] = field(default_factory=list)
|
||||
error: Optional[str] = None
|
||||
from_cache: bool = False
|
||||
|
||||
# Pack sizes embedded in a product name ("Marie Gold 250g", "Butter 1L").
|
||||
# Mirrors the unit list in scripts/backfill_nutrition_from_barcodes.py, widened
|
||||
# with the count-based units this catalog also uses.
|
||||
_SIZE_RE = re.compile(
|
||||
r"\b\d+(?:[.,]\d+)?\s*"
|
||||
r"(?:kgs|kg|gms|gm|grams|gram|mg|g|mls|ml|litres|litre|ltrs|ltr|l|"
|
||||
r"pcs|pc|pieces|piece|nos|no)\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
# "(20)", "[6]" - pack-count noise seen in OFF titles such as
|
||||
# "Brit 50-50 maska chaska 40g (20)".
|
||||
_PAREN_RE = re.compile(r"[\(\[][^\)\]]*[\)\]]")
|
||||
|
||||
_NON_ALNUM_RE = re.compile(r"[^a-z0-9]+")
|
||||
|
||||
# GS1 prefix for barcodes issued in India. Every correct match observed in
|
||||
# testing starts with it; the false ones were Mondelez EU codes (7622...) and a
|
||||
# Japanese Maggi (4987...), i.e. the same product line sold in another market
|
||||
# with a different pack and a different code.
|
||||
INDIA_GS1_PREFIX = "890"
|
||||
|
||||
# Variant words that `matching.VARIANT_DISTINGUISHING_TERMS` does not carry but
|
||||
# which mark a different retail SKU in this catalog. Kept local rather than
|
||||
# added to matching.py, which the live enrichment pipeline shares.
|
||||
EXTRA_VARIANT_TERMS = {
|
||||
"minis", "mini", "plus", "max", "lite", "duo", "multipack", "sugarfree",
|
||||
}
|
||||
|
||||
# A token shared by this fraction of a brand's non-prefixed aliases is treated
|
||||
# as part of the brand's own name rather than a product name. For Hindustan
|
||||
# Unilever every alias is "hul <subbrand>", so "hul" clears the bar and is
|
||||
# stripped, while "lux"/"dove" appear once each and are preserved.
|
||||
_BRAND_TOKEN_SHARE = 0.6
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Fetch
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _cache_path(slug: str, cache_dir: Optional[Path] = None) -> Path:
|
||||
return (cache_dir or CACHE_DIR) / f"{slug}.json"
|
||||
|
||||
|
||||
def brand_tag_slug(brand: str) -> str:
|
||||
"""The brand as Open Food Facts tags it: lowercase, runs of anything that
|
||||
is not a letter or digit collapsed to one hyphen. "Hindustan Unilever" ->
|
||||
"hindustan-unilever", "P&G" -> "p-g", "Naga" -> "naga". OFF matches
|
||||
`brands_tags` case-insensitively, but sending the slug form is what its
|
||||
own site does and avoids depending on that."""
|
||||
return re.sub(r"[^a-z0-9]+", "-", (brand or "").lower()).strip("-")
|
||||
|
||||
|
||||
@with_retry(max_attempts=3, min_wait=2.0, max_wait=10.0)
|
||||
def _get_page(brand: str, country: Optional[str], page: int) -> Optional[Dict[str, Any]]:
|
||||
"""One page of OFF v2 search results.
|
||||
|
||||
Raises on transport errors and on 429 / 5xx (both retried by the
|
||||
decorator); returns None for any other response that is not parseable
|
||||
JSON - an HTML "temporarily unavailable" page, a 4xx, a truncated body.
|
||||
None means "this page FAILED", which the caller must keep distinct from a
|
||||
page that parsed fine and simply held no products.
|
||||
"""
|
||||
params: Dict[str, Any] = {
|
||||
"brands_tags": brand_tag_slug(brand),
|
||||
"fields": FIELDS,
|
||||
"page_size": PAGE_SIZE,
|
||||
"page": page,
|
||||
}
|
||||
if country:
|
||||
params["countries_tags"] = f"en:{country}"
|
||||
|
||||
resp = requests.get(
|
||||
SEARCH_URL,
|
||||
params=params,
|
||||
headers={"User-Agent": USER_AGENT, "Accept": "application/json"},
|
||||
timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS,
|
||||
)
|
||||
|
||||
if resp.status_code == 429 or resp.status_code >= 500:
|
||||
raise OffUnavailable(f"HTTP {resp.status_code} from Open Food Facts")
|
||||
if resp.status_code != 200:
|
||||
logger.warning("OFF search returned HTTP %s for brand %r page %s",
|
||||
resp.status_code, brand, page)
|
||||
return None
|
||||
if "json" not in (resp.headers.get("content-type") or "").lower():
|
||||
logger.warning("OFF search returned non-JSON (%s) for brand %r - "
|
||||
"the service is probably serving an error page",
|
||||
resp.headers.get("content-type"), brand)
|
||||
return None
|
||||
try:
|
||||
payload = resp.json()
|
||||
except ValueError as e:
|
||||
logger.warning("OFF search returned unparseable JSON for brand %r: %s", brand, e)
|
||||
return None
|
||||
return payload if isinstance(payload, dict) else None
|
||||
|
||||
|
||||
def _read_cache(path: Path, brand: str) -> Optional[List[Dict[str, Any]]]:
|
||||
"""Cached hits, or None when the cache must not be trusted: unreadable, or
|
||||
an empty file from before CACHE_SCHEMA existed (see that constant)."""
|
||||
try:
|
||||
cached = json.loads(path.read_text(encoding="utf-8"))
|
||||
except Exception as e: # noqa: BLE001 - a corrupt cache must not be fatal
|
||||
logger.warning(" Ignoring unreadable OFF cache %s: %s", path.name, e)
|
||||
return None
|
||||
hits = cached.get("hits") or []
|
||||
if not hits and cached.get("schema") != CACHE_SCHEMA:
|
||||
logger.info(" Ignoring empty pre-v%s OFF cache for %r - re-fetching",
|
||||
CACHE_SCHEMA, brand)
|
||||
return None
|
||||
logger.info(" OFF corpus for %r: %d product(s) (cached %s)",
|
||||
brand, len(hits), cached.get("fetched_at_human", "?"))
|
||||
return hits
|
||||
|
||||
|
||||
def fetch_brand_corpus_result(brand: str,
|
||||
country: Optional[str] = None,
|
||||
refresh: bool = False,
|
||||
cache_dir: Optional[Path] = None) -> CorpusFetch:
|
||||
"""Every OFF product for `brand`, from disk cache unless `refresh`, plus
|
||||
whether the fetch actually completed.
|
||||
|
||||
Hits with no usable product name are dropped here rather than at match time
|
||||
(6 of 146 Amul hits, 1 of 233 Britannia hits) - a nameless hit can never
|
||||
clear a name-similarity threshold, so carrying it forward only inflates the
|
||||
corpus.
|
||||
|
||||
A fetch that fails part-way returns what it got with `error` set and
|
||||
writes NO cache file. Caching a failure as `hits: []` is how "Open Food
|
||||
Facts has nothing for this brand" was being asserted for brands OFF had
|
||||
simply been too busy to answer about.
|
||||
"""
|
||||
country = BARCODE_COUNTRY_TAG if country is None else (country or None)
|
||||
slug = re.sub(r"[^a-z0-9]+", "_", brand.lower()).strip("_")
|
||||
path = _cache_path(slug, cache_dir)
|
||||
|
||||
if not refresh and path.exists():
|
||||
cached = _read_cache(path, brand)
|
||||
if cached is not None:
|
||||
return CorpusFetch(hits=cached, from_cache=True)
|
||||
|
||||
hits: List[Dict[str, Any]] = []
|
||||
error: Optional[str] = None
|
||||
page = 1
|
||||
total_pages = 1
|
||||
while page <= min(total_pages, MAX_PAGES):
|
||||
if page > 1:
|
||||
time.sleep(PAUSE_SECONDS)
|
||||
try:
|
||||
payload = _get_page(brand, country, page)
|
||||
except requests.exceptions.RequestException as e:
|
||||
# Retries exhausted (OffUnavailable is a ConnectionError too).
|
||||
error = str(e) or e.__class__.__name__
|
||||
break
|
||||
if payload is None:
|
||||
error = "Open Food Facts returned an unusable response"
|
||||
break
|
||||
batch = payload.get("products") or []
|
||||
hits.extend(h for h in batch if isinstance(h, dict) and (h.get("product_name") or "").strip())
|
||||
if page == 1:
|
||||
# `page_count` is products-on-this-page in the v2 API; the real
|
||||
# page total is `count` over the page size the SERVER applied -
|
||||
# it silently caps the requested size (250 asked, 100 served for
|
||||
# Amul), so dividing by PAGE_SIZE under-pages.
|
||||
count = payload.get("count") or 0
|
||||
served = payload.get("page_size") or len(batch) or PAGE_SIZE
|
||||
try:
|
||||
total_pages = max(1, math.ceil(int(count) / int(served)))
|
||||
except (TypeError, ValueError, ZeroDivisionError):
|
||||
total_pages = 1
|
||||
if not batch:
|
||||
break
|
||||
page += 1
|
||||
|
||||
if error:
|
||||
logger.warning(" OFF corpus for %r: fetch failed after %d usable product(s) - %s "
|
||||
"(nothing cached)", brand, len(hits), error)
|
||||
return CorpusFetch(hits=hits, error=error)
|
||||
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
tmp = path.with_suffix(".json.tmp")
|
||||
tmp.write_text(json.dumps({
|
||||
"schema": CACHE_SCHEMA,
|
||||
"endpoint": SEARCH_URL,
|
||||
"brand": brand,
|
||||
"brand_tag": brand_tag_slug(brand),
|
||||
"country": country,
|
||||
"fetched_at": time.time(),
|
||||
"fetched_at_human": time.strftime("%Y-%m-%d %H:%M:%S"),
|
||||
"hits": hits,
|
||||
}, indent=2, ensure_ascii=False), encoding="utf-8")
|
||||
os.replace(tmp, path)
|
||||
|
||||
logger.info(" OFF corpus for %r: %d usable product(s) fetched", brand, len(hits))
|
||||
return CorpusFetch(hits=hits)
|
||||
|
||||
|
||||
def fetch_brand_corpus(brand: str,
|
||||
country: Optional[str] = None,
|
||||
refresh: bool = False,
|
||||
cache_dir: Optional[Path] = None) -> List[Dict[str, Any]]:
|
||||
"""`fetch_brand_corpus_result(...).hits` - the list-only form every
|
||||
backfill and ingestion caller uses. Returns [] rather than raising when
|
||||
OFF is unreachable, so one bad brand does not abort a multi-brand run;
|
||||
callers that need to tell "unreachable" from "empty" use the result form.
|
||||
"""
|
||||
return fetch_brand_corpus_result(brand, country=country, refresh=refresh,
|
||||
cache_dir=cache_dir).hits
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Normalisation
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def brand_tokens(brand: str) -> set:
|
||||
"""Tokens that name the BRAND rather than the product, and so must be
|
||||
stripped from both sides before names are compared.
|
||||
|
||||
Includes the brand as given, its canonical parent, and any token shared by
|
||||
most of the brand's non-prefixed aliases. That last rule is what recovers
|
||||
"hul": every Hindustan Unilever alias is "hul <subbrand>", so "hul" is
|
||||
brand noise, while "lux" and "dove" each appear once and survive as real
|
||||
product names. Without it our mangled "Hindustan Unilever Hul Lux" could
|
||||
never match OFF's "Lux".
|
||||
"""
|
||||
canonical = resolve_parent_brand(brand)
|
||||
canonical_l = canonical.lower().strip()
|
||||
is_own_parent = canonical_l == brand.lower().strip()
|
||||
|
||||
# Only inherit anything from the canonical parent when the brand IS that
|
||||
# parent. Several brands are filed under an unrelated parent for storage
|
||||
# reasons - Tata's catalogue lives in brand_hindustan_unilever - and
|
||||
# treating that parent's name as brand noise would strip real words out of
|
||||
# Tata product names.
|
||||
tokens = set(_tokenize(brand))
|
||||
if is_own_parent:
|
||||
tokens |= set(_tokenize(canonical))
|
||||
|
||||
non_prefixed = [
|
||||
a for a, parent in BRAND_ALIASES.items()
|
||||
if parent.lower().strip() == canonical_l and not a.startswith(canonical_l)
|
||||
] if is_own_parent else []
|
||||
if non_prefixed:
|
||||
counts: Dict[str, int] = {}
|
||||
for alias in non_prefixed:
|
||||
for tok in set(_tokenize(alias)):
|
||||
counts[tok] = counts.get(tok, 0) + 1
|
||||
threshold = len(non_prefixed) * _BRAND_TOKEN_SHARE
|
||||
tokens |= {tok for tok, n in counts.items() if n >= threshold}
|
||||
|
||||
return {t for t in tokens if t}
|
||||
|
||||
|
||||
def _tokenize(text: Optional[str]) -> List[str]:
|
||||
return [t for t in _NON_ALNUM_RE.sub(" ", (text or "").lower()).split() if t]
|
||||
|
||||
|
||||
def strip_sizes(text: Optional[str]) -> str:
|
||||
"""Remove pack sizes and pack-count parentheticals from a product name."""
|
||||
cleaned = _PAREN_RE.sub(" ", (text or ""))
|
||||
cleaned = _SIZE_RE.sub(" ", cleaned)
|
||||
return re.sub(r"\s+", " ", cleaned).strip()
|
||||
|
||||
|
||||
def normalize_for_match(text: Optional[str], tokens_to_drop: Optional[Iterable[str]] = None) -> str:
|
||||
"""Lowercased, size-free, brand-free comparison key.
|
||||
|
||||
Returns "" when nothing survives, and callers MUST treat that as "no
|
||||
product identity" rather than falling back to the brand-bearing form. The
|
||||
catalog contains rows titled only by their brand and a size - "Amul 90g",
|
||||
"Amul 1kg" - and OFF contains an equally anonymous entry named just "Amul".
|
||||
With a fallback those two normalise to "amul" and match at 1.000, which
|
||||
confidently stamps a real barcode onto six products that have no identity
|
||||
in common beyond the brand. An empty key is the correct answer there.
|
||||
"""
|
||||
base = _tokenize(strip_sizes(text))
|
||||
if not base:
|
||||
return ""
|
||||
drop = {t.lower() for t in (tokens_to_drop or ())}
|
||||
return " ".join(t for t in base if t not in drop)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Barcode acceptance
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def accept_barcode(raw: Optional[str]) -> Optional[Tuple[str, str, Optional[str]]]:
|
||||
"""Enforce the "valid 8 or 13 digit code" rule.
|
||||
|
||||
Returns (code, barcode_type, source_upc) or None. A checksum-valid 12-digit
|
||||
UPC-A is widened to its EAN-13 form (a leading zero contributes nothing to
|
||||
the GTIN checksum, so the check digit is unchanged and still valid), and the
|
||||
original 12-digit form is handed back as `source_upc` so the catalog's `upc`
|
||||
field can record where the code came from. GTIN-14 is a shipping-carton
|
||||
code, not a retail one, and is rejected outright.
|
||||
"""
|
||||
code = validate_barcode(raw)
|
||||
if not code:
|
||||
return None
|
||||
if len(code) == 12:
|
||||
widened = to_ean13(code)
|
||||
if not widened or not validate_barcode(widened):
|
||||
return None
|
||||
return widened, BarcodeType.EAN13.value, code
|
||||
if len(code) in (8, 13):
|
||||
return code, classify_barcode_type(code).value, None
|
||||
return None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Matching
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class Candidate:
|
||||
"""One scored OFF hit for one of our title groups."""
|
||||
|
||||
__slots__ = ("barcode", "barcode_type", "source_upc", "off_name", "off_qty",
|
||||
"score", "size_bonus")
|
||||
|
||||
def __init__(self, barcode: str, barcode_type: str, source_upc: Optional[str],
|
||||
off_name: str, off_qty: Optional[str], score: float, size_bonus: int):
|
||||
self.barcode = barcode
|
||||
self.barcode_type = barcode_type
|
||||
self.source_upc = source_upc
|
||||
self.off_name = off_name
|
||||
self.off_qty = off_qty
|
||||
self.score = score
|
||||
self.size_bonus = size_bonus
|
||||
|
||||
@property
|
||||
def rank(self) -> tuple:
|
||||
"""Sort key, best first. Size agreement outranks a marginally better
|
||||
name score, and the barcode breaks ties so a rerun over an unchanged
|
||||
corpus always picks the same product."""
|
||||
return (-self.size_bonus, -self.score, self.barcode)
|
||||
|
||||
def __repr__(self) -> str: # pragma: no cover - debugging aid
|
||||
return f"<Candidate {self.barcode} {self.off_name!r} score={self.score}>"
|
||||
|
||||
|
||||
def symmetric_similarity(clean_candidate: str, clean_target: str) -> float:
|
||||
"""`name_similarity` in both directions, worst case wins.
|
||||
|
||||
`name_similarity` divides the token overlap by the TARGET's token count
|
||||
only, so an OFF name that is a superset of ours scores near-perfectly. That
|
||||
asymmetry is not theoretical - measured against the live corpus it accepted
|
||||
"Butter milk amul" for our "Amul Butter" at 0.882, and "Dairy Milk Silk
|
||||
Minis" for "Dairy Milk Silk" at 0.933. Scoring both directions and taking
|
||||
the minimum drops those to 0.418 and 0.783 while leaving genuine matches
|
||||
("Marie Gold" / "Marie Gold") at 1.000, because a true match is symmetric
|
||||
by construction.
|
||||
"""
|
||||
return min(
|
||||
name_similarity(clean_candidate, clean_target),
|
||||
name_similarity(clean_target, clean_candidate),
|
||||
)
|
||||
|
||||
|
||||
def has_extra_variant_conflict(candidate_title: str, target_title: str) -> bool:
|
||||
"""The `has_conflicting_variant_terms` rule over `EXTRA_VARIANT_TERMS`."""
|
||||
cand = set(_tokenize(candidate_title))
|
||||
target = set(_tokenize(target_title))
|
||||
return bool((cand & EXTRA_VARIANT_TERMS) - target)
|
||||
|
||||
|
||||
def score_candidates(off_hits: Sequence[Dict[str, Any]],
|
||||
our_title: str,
|
||||
our_sizes: Sequence[str],
|
||||
tokens_to_drop: Iterable[str],
|
||||
review_min: float,
|
||||
require_india_prefix: bool = True) -> List[Candidate]:
|
||||
"""Every acceptable OFF hit for one title group, best first.
|
||||
|
||||
Deliberately does NOT use `matching.is_match()`: its size gate returns False
|
||||
whenever either side's size is blank (matching.py:85-86), and 57 of 146 Amul
|
||||
hits carry `quantity: null`. That gate would reject the exact case this
|
||||
backfill exists to handle - our "Britannia Marie Gold 250g" against OFF's
|
||||
"Britannia Marie Gold". Size is used as a ranking bonus here instead of a
|
||||
veto. `has_conflicting_variant_terms` IS still applied, because a
|
||||
"Sugar Free" or "Family Pack" hit is a genuinely different retail product.
|
||||
"""
|
||||
drop = set(tokens_to_drop)
|
||||
clean_target = normalize_for_match(our_title, drop)
|
||||
if not clean_target:
|
||||
return []
|
||||
|
||||
out: List[Candidate] = []
|
||||
for hit in off_hits:
|
||||
off_name = (hit.get("product_name") or hit.get("product_name_en") or "").strip()
|
||||
if not off_name:
|
||||
continue
|
||||
if has_conflicting_variant_terms(off_name, our_title):
|
||||
continue
|
||||
if has_extra_variant_conflict(off_name, our_title):
|
||||
continue
|
||||
|
||||
accepted = accept_barcode(hit.get("code"))
|
||||
if not accepted:
|
||||
continue
|
||||
code, btype, source_upc = accepted
|
||||
if require_india_prefix and not code.startswith(INDIA_GS1_PREFIX):
|
||||
continue
|
||||
|
||||
clean_cand = normalize_for_match(off_name, drop)
|
||||
if not clean_cand:
|
||||
continue
|
||||
|
||||
score = symmetric_similarity(clean_cand, clean_target)
|
||||
if score < review_min:
|
||||
continue
|
||||
|
||||
off_qty = (hit.get("quantity") or "").strip() or None
|
||||
size_bonus = 1 if off_qty and any(
|
||||
quantities_match(off_qty, s, tolerance=0.03) for s in our_sizes if s
|
||||
) else 0
|
||||
|
||||
out.append(Candidate(code, btype, source_upc, off_name, off_qty, score, size_bonus))
|
||||
|
||||
out.sort(key=lambda c: c.rank)
|
||||
return out
|
||||
180
app/services/enrichment/barcode/sources/open_food_facts.py
Normal file
180
app/services/enrichment/barcode/sources/open_food_facts.py
Normal file
@@ -0,0 +1,180 @@
|
||||
"""
|
||||
Tier 2: Open Food Facts / Open Beauty Facts / Open Products Facts.
|
||||
|
||||
Real, free, no-API-key public search API - the same family of open,
|
||||
community-maintained product databases already used as the project's
|
||||
PRIMARY image source (see `app/services/image_search.py`'s
|
||||
`OPEN_FACTS_HOSTS` / `_query_openfacts`). Every entry is keyed by its real
|
||||
barcode (the `code` field IS the GTIN/EAN/UPC printed on the physical
|
||||
pack), which is exactly the ground-truth this module needs - unlike
|
||||
`image_search.py`, which only reads the image URLs off each entry, this
|
||||
adapter reads the `code` field itself.
|
||||
|
||||
Deliberately a separate, self-contained query function rather than
|
||||
importing `image_search._query_openfacts` (which is a private, underscore-
|
||||
prefixed helper): this adapter needs different response fields (`code`,
|
||||
`countries_tags`) and India-specific filtering that the image-search
|
||||
helper has no reason to carry. `quantities_match`/`parse_quantity_grams`
|
||||
ARE imported from `image_search.py` (public functions) for size
|
||||
comparison, so the size-matching logic itself is not duplicated - see
|
||||
`matching.py`.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import List, Optional
|
||||
|
||||
import requests
|
||||
|
||||
from app.infrastructure.settings import BARCODE_LOOKUP_TIMEOUT_SECONDS, BARCODE_COUNTRY_TAG
|
||||
from app.services.enrichment.barcode.models import BarcodeCandidate
|
||||
from app.services.enrichment.barcode.retry import with_retry
|
||||
from app.services.enrichment.barcode.sources.base import BarcodeSource
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_HOSTS = [
|
||||
"world.openfoodfacts.org",
|
||||
"world.openbeautyfacts.org",
|
||||
"world.openproductsfacts.org",
|
||||
]
|
||||
_BROWSER_UA = (
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
|
||||
"(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
|
||||
)
|
||||
_FIELDS = "code,product_name,brands,brands_tags,quantity,countries_tags"
|
||||
|
||||
# The reverse direction asks for more than the search does, because its caller
|
||||
# builds a whole nutrition record rather than just reading a barcode off the
|
||||
# entry. Still an explicit list and not everything: a bare product fetch returns
|
||||
# ~296 fields per item, most of them editing metadata nobody here reads.
|
||||
_PRODUCT_FIELDS = (
|
||||
"code,product_name,brands,brands_tags,quantity,countries_tags,"
|
||||
"nutriments,serving_quantity,serving_size,ingredients_text,"
|
||||
"nutriscore_grade,allergens_tags,labels_tags,"
|
||||
"ingredients_analysis_tags,categories_tags"
|
||||
)
|
||||
|
||||
|
||||
def fetch_product_by_barcode(code: str) -> Optional[dict]:
|
||||
"""The one OFF call in this project that is EXACT rather than a guess.
|
||||
|
||||
Every other Open*Facts call here - this module's own `search()`,
|
||||
`nutrition_data_service`, `image_search` - queries by brand and product
|
||||
name and then scores whatever comes back. That is why the OFF-sourced rows
|
||||
in `nutrition_facts` carry match confidences as low as 0.32. A barcode is
|
||||
the identifier printed on the pack, so `/api/v2/product/{code}` either
|
||||
returns that exact product or nothing at all.
|
||||
|
||||
Returns the product dict, or None when OFF has never seen the barcode -
|
||||
which is the ordinary outcome for about a third of ours, not an error. The
|
||||
caller still has to decide whether the record describes the product WE
|
||||
attached that barcode to; see `matching.is_match`. Measured on real
|
||||
catalogue rows, a quarter of the found records were a different product,
|
||||
because the stored barcode itself was wrong.
|
||||
|
||||
Cascades the same three hosts as the search: a household or beauty item
|
||||
lives in openbeautyfacts, not openfoodfacts, under the same code.
|
||||
"""
|
||||
code = (code or "").strip()
|
||||
if not code:
|
||||
return None
|
||||
|
||||
for host in _HOSTS:
|
||||
@with_retry(max_attempts=2)
|
||||
def _call(host=host):
|
||||
return requests.get(
|
||||
f"https://{host}/api/v2/product/{code}.json",
|
||||
params={"fields": _PRODUCT_FIELDS},
|
||||
headers={"User-Agent": _BROWSER_UA},
|
||||
timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS,
|
||||
)
|
||||
|
||||
try:
|
||||
resp = _call()
|
||||
# 404 is how OFF says "no such barcode here" - try the next host
|
||||
# rather than treating it as a failure.
|
||||
if resp.status_code != 200:
|
||||
continue
|
||||
body = resp.json()
|
||||
# status 1 = found, 0 = not found. The HTTP code alone is not
|
||||
# enough: OFF answers 200 with status 0 for an unknown barcode.
|
||||
if body.get("status") == 1 and body.get("product"):
|
||||
return body["product"]
|
||||
except Exception as e:
|
||||
logger.debug("Open*Facts product fetch failed on %s for %s: %s", host, code, e)
|
||||
continue
|
||||
return None
|
||||
|
||||
|
||||
class OpenFoodFactsSource(BarcodeSource):
|
||||
name = "Open Food Facts"
|
||||
tier = 2
|
||||
|
||||
def search(self, brand: str, product_title: str, size: str, category: str = "") -> List[BarcodeCandidate]:
|
||||
query = f"{brand or ''} {product_title or ''}".strip()
|
||||
if not query:
|
||||
return []
|
||||
|
||||
products = self._query(query)
|
||||
if not products and brand and product_title:
|
||||
# Combined query too specific (common for regional brand-name
|
||||
# variants) - retry title-only, same fallback image_search.py uses.
|
||||
products = self._query(product_title)
|
||||
|
||||
candidates: List[BarcodeCandidate] = []
|
||||
for item in products:
|
||||
code = str(item.get("code") or "").strip()
|
||||
if not code:
|
||||
continue
|
||||
countries = ",".join(item.get("countries_tags") or [])
|
||||
candidates.append(BarcodeCandidate(
|
||||
barcode=code,
|
||||
source_name=self.name,
|
||||
candidate_title=item.get("product_name") or "",
|
||||
candidate_brand=item.get("brands") or "",
|
||||
candidate_size=item.get("quantity") or "",
|
||||
candidate_countries=countries,
|
||||
))
|
||||
|
||||
# Soft India-market prioritisation: candidates whose countries_tags
|
||||
# mention the target market are tried first, but non-tagged/other-
|
||||
# market candidates are kept (not dropped) since many genuine
|
||||
# Indian FMCG entries simply have this field blank upstream - the
|
||||
# brand/size/name gate in matching.py is what actually decides
|
||||
# correctness, this only affects which validated match is found
|
||||
# (and therefore stops the cascade) first.
|
||||
if BARCODE_COUNTRY_TAG:
|
||||
tag = BARCODE_COUNTRY_TAG.lower()
|
||||
candidates.sort(key=lambda c: 0 if tag in c.candidate_countries.lower() else 1)
|
||||
return candidates
|
||||
|
||||
def _query(self, query: str) -> list:
|
||||
for host in _HOSTS:
|
||||
@with_retry(max_attempts=2)
|
||||
def _call(host=host):
|
||||
return requests.get(
|
||||
f"https://{host}/cgi/search.pl",
|
||||
params={
|
||||
"search_terms": query,
|
||||
"search_simple": 1,
|
||||
"action": "process",
|
||||
"json": 1,
|
||||
"page_size": 20,
|
||||
"fields": _FIELDS,
|
||||
},
|
||||
headers={"User-Agent": _BROWSER_UA},
|
||||
timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS,
|
||||
)
|
||||
|
||||
try:
|
||||
resp = _call()
|
||||
if resp.status_code != 200:
|
||||
continue
|
||||
products = resp.json().get("products", [])
|
||||
if products:
|
||||
return products
|
||||
except Exception as e:
|
||||
logger.debug(f"Open*Facts barcode lookup failed on {host} for '{query}': {e}")
|
||||
continue
|
||||
return []
|
||||
81
app/services/enrichment/barcode/sources/upc_database.py
Normal file
81
app/services/enrichment/barcode/sources/upc_database.py
Normal file
@@ -0,0 +1,81 @@
|
||||
"""
|
||||
Tier 3: "trusted barcode databases" - UPCItemDB.
|
||||
|
||||
Real, free (trial-tier, no API key required, rate-limited) reverse lookup:
|
||||
search by product name/brand keywords and get back candidate items with
|
||||
their own `upc`/`ean` fields. This is the general "trusted barcode
|
||||
database" tier the spec asks for as a fallback below GS1 India / Open
|
||||
Food Facts.
|
||||
|
||||
RATE LIMIT: UPCItemDB's free trial endpoint is capped (documented as
|
||||
~100 requests/day, ~1 request/second) - this is exactly why this tier
|
||||
sits BELOW Open Food Facts (unlimited, no key) in the cascade, and why
|
||||
`service.py` only calls a lower tier at all when every higher tier has
|
||||
already failed to produce a validated match, plus why the local cache
|
||||
(`cache.py`) matters most for this specific source. If a paid UPCItemDB
|
||||
key is available, set `UPC_DATABASE_API_KEY` (see .env.example) to switch
|
||||
to the production endpoint with a higher quota - the request shape below
|
||||
already supports both.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import List
|
||||
|
||||
import requests
|
||||
|
||||
from app.infrastructure.settings import BARCODE_LOOKUP_TIMEOUT_SECONDS, UPC_DATABASE_API_KEY
|
||||
from app.services.enrichment.barcode.models import BarcodeCandidate
|
||||
from app.services.enrichment.barcode.retry import with_retry
|
||||
from app.services.enrichment.barcode.sources.base import BarcodeSource
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_TRIAL_URL = "https://api.upcitemdb.com/prod/trial/search"
|
||||
_PROD_URL = "https://api.upcitemdb.com/prod/v1/search"
|
||||
|
||||
|
||||
class UPCDatabaseSource(BarcodeSource):
|
||||
name = "UPCItemDB"
|
||||
tier = 3
|
||||
|
||||
def search(self, brand: str, product_title: str, size: str, category: str = "") -> List[BarcodeCandidate]:
|
||||
query = f"{brand or ''} {product_title or ''} {size or ''}".strip()
|
||||
if not query:
|
||||
return []
|
||||
|
||||
url = _PROD_URL if UPC_DATABASE_API_KEY else _TRIAL_URL
|
||||
headers = {"Accept": "application/json"}
|
||||
if UPC_DATABASE_API_KEY:
|
||||
headers["user_key"] = UPC_DATABASE_API_KEY
|
||||
headers["key_type"] = "3scale"
|
||||
|
||||
@with_retry(max_attempts=2)
|
||||
def _call():
|
||||
return requests.get(url, params={"s": query, "type": "product"}, headers=headers,
|
||||
timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS)
|
||||
|
||||
try:
|
||||
resp = _call()
|
||||
if resp.status_code != 200:
|
||||
logger.debug(f"UPCItemDB non-200 ({resp.status_code}) for '{query}'")
|
||||
return []
|
||||
data = resp.json()
|
||||
except Exception as e:
|
||||
logger.debug(f"UPCItemDB lookup failed for '{query}': {e}")
|
||||
return []
|
||||
|
||||
candidates: List[BarcodeCandidate] = []
|
||||
for item in data.get("items", []):
|
||||
code = item.get("ean") or item.get("upc")
|
||||
if not code:
|
||||
continue
|
||||
candidates.append(BarcodeCandidate(
|
||||
barcode=str(code),
|
||||
source_name=self.name,
|
||||
candidate_title=item.get("title") or "",
|
||||
candidate_brand=item.get("brand") or "",
|
||||
candidate_size=item.get("size") or "",
|
||||
candidate_countries="",
|
||||
))
|
||||
return candidates
|
||||
55
app/services/enrichment/barcode/stage.py
Normal file
55
app/services/enrichment/barcode/stage.py
Normal file
@@ -0,0 +1,55 @@
|
||||
"""Adapts BarcodeLookupService to the EnrichmentStage contract so it can be
|
||||
registered in app/services/enrichment/pipeline.py's default pipeline."""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import Any, Dict
|
||||
|
||||
from app.infrastructure.settings import ENABLE_BARCODE_LOOKUP
|
||||
from app.services.enrichment.base import EnrichmentStage, StageOutcome
|
||||
from app.services.enrichment.barcode.service import get_default_service
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class BarcodeEnrichmentStage(EnrichmentStage):
|
||||
name = "barcode_lookup"
|
||||
|
||||
@property
|
||||
def enabled(self) -> bool:
|
||||
return ENABLE_BARCODE_LOOKUP
|
||||
|
||||
async def enrich_one(self, product: Dict[str, Any], brand: str) -> StageOutcome:
|
||||
# ALREADY HAS ONE - never overwrite, and never even look. The same
|
||||
# guard HsnGstEnrichmentStage carries, and this stage was the only one
|
||||
# missing it.
|
||||
#
|
||||
# Without it the damage was not "a worse barcode" but no barcode at
|
||||
# all: `as_product_fields()` always returns all nine keys, so a failed
|
||||
# lookup handed back {"barcode": None, ...} and `apply()` merged that
|
||||
# straight over whatever the shop had typed. Proven end to end - a
|
||||
# sheet sending 8901262010016 stored NULL. Since the cascade misses far
|
||||
# more often than it hits, switching ENABLE_BARCODE_LOOKUP on would
|
||||
# have destroyed more real barcodes than it found.
|
||||
#
|
||||
# A barcode the shop supplied is also better evidence than anything the
|
||||
# cascade can find: they are holding the pack. Skipping the lookup
|
||||
# saves the network call as well.
|
||||
if str(product.get("barcode") or "").strip():
|
||||
return StageOutcome(stage_name=self.name, fields={})
|
||||
|
||||
service = get_default_service()
|
||||
title = product.get("title") or product.get("product_name") or ""
|
||||
size = product.get("size") or ""
|
||||
category = product.get("category") or ""
|
||||
|
||||
try:
|
||||
import asyncio
|
||||
result = await asyncio.to_thread(service.lookup_one, brand, title, size, category)
|
||||
except Exception as e:
|
||||
logger.warning(f"Barcode enrichment failed for '{title}' {size}: {e}")
|
||||
from app.services.enrichment.barcode.models import BarcodeResult, LookupStatus
|
||||
result = BarcodeResult.null_result(LookupStatus.ERROR)
|
||||
|
||||
return StageOutcome(stage_name=self.name, fields=result.as_product_fields(),
|
||||
error=None if result.barcode_verified else result.barcode_lookup_status)
|
||||
99
app/services/enrichment/barcode/validators.py
Normal file
99
app/services/enrichment/barcode/validators.py
Normal file
@@ -0,0 +1,99 @@
|
||||
"""
|
||||
Barcode format + checksum validation.
|
||||
|
||||
Deliberately the LAST gate a candidate passes through before it's allowed
|
||||
into a `BarcodeResult` (see `service.py`), regardless of which source
|
||||
produced it or how much we otherwise trust that source - a source telling
|
||||
us "this is the barcode" is never sufficient on its own; the digits have to
|
||||
actually check out mathematically. This is what "Validate EAN-13 checksum"
|
||||
/ "Validate GTIN format" / "Reject invalid barcode lengths" / "Reject
|
||||
malformed barcode values" (Barcode Validation requirements) means in code.
|
||||
|
||||
No third-party dependency - GTIN/EAN/UPC-A all share one checksum
|
||||
algorithm (the classic "alternating 3/1 weights counted from the rightmost
|
||||
digit before the check digit"), so GTIN-8/12/13/14 are all validated by the
|
||||
same function; only the accepted LENGTH differs per barcode type.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import Optional
|
||||
|
||||
from app.services.enrichment.barcode.models import BarcodeType
|
||||
|
||||
_VALID_LENGTHS = {8: BarcodeType.GTIN8, 12: BarcodeType.UPC_A, 13: BarcodeType.EAN13, 14: BarcodeType.GTIN14}
|
||||
|
||||
# Digits only, no separators - callers are expected to call normalize_barcode()
|
||||
# first if the raw source string may contain spaces/hyphens.
|
||||
_DIGITS_ONLY_RE = re.compile(r"^\d+$")
|
||||
|
||||
|
||||
def normalize_barcode(raw: Optional[str]) -> Optional[str]:
|
||||
"""Strip everything but digits (spaces, hyphens, a stray 'EAN:' label,
|
||||
etc.). Returns None for empty/unusable input - never raises."""
|
||||
if not raw:
|
||||
return None
|
||||
digits = re.sub(r"\D", "", str(raw))
|
||||
return digits or None
|
||||
|
||||
|
||||
def gtin_check_digit(digits_without_check: str) -> int:
|
||||
"""Compute the correct check digit for a GTIN-8/12/13/14 payload
|
||||
(i.e. every digit EXCEPT the check digit itself), using the standard
|
||||
alternating 3/1 weighting counted from the rightmost digit."""
|
||||
total = 0
|
||||
for i, ch in enumerate(reversed(digits_without_check)):
|
||||
weight = 3 if i % 2 == 0 else 1
|
||||
total += int(ch) * weight
|
||||
return (10 - (total % 10)) % 10
|
||||
|
||||
|
||||
def has_valid_checksum(code: str) -> bool:
|
||||
"""True if `code`'s own last digit matches the checksum computed over
|
||||
the rest of it. `code` must already be digits-only."""
|
||||
if not code or not _DIGITS_ONLY_RE.match(code):
|
||||
return False
|
||||
body, check_digit = code[:-1], code[-1]
|
||||
try:
|
||||
expected = gtin_check_digit(body)
|
||||
except (ValueError, IndexError):
|
||||
return False
|
||||
return str(expected) == check_digit
|
||||
|
||||
|
||||
def classify_barcode_type(code: str) -> BarcodeType:
|
||||
"""Barcode type purely from its (already checksum-validated) length.
|
||||
UPC-A (12 digits) is the one ambiguous case worth calling out: it is
|
||||
numerically a GTIN-13 with a leading zero, but is reported as UPC-A
|
||||
here since that's the label the requesting spec/UI expects for a
|
||||
12-digit code."""
|
||||
return _VALID_LENGTHS.get(len(code), BarcodeType.UNKNOWN)
|
||||
|
||||
|
||||
def validate_barcode(raw: Optional[str]) -> Optional[str]:
|
||||
"""Single entry point: normalize, check length, check checksum.
|
||||
Returns the clean digits-only barcode string if and only if it is a
|
||||
genuinely valid GTIN-8/12/13/14, else None. This is the ONLY function
|
||||
other modules should call to decide "is this barcode good enough to
|
||||
store" - never inline a length/regex check elsewhere.
|
||||
"""
|
||||
code = normalize_barcode(raw)
|
||||
if not code:
|
||||
return None
|
||||
if len(code) not in _VALID_LENGTHS:
|
||||
return None
|
||||
if not has_valid_checksum(code):
|
||||
return None
|
||||
return code
|
||||
|
||||
|
||||
def to_ean13(code: str) -> Optional[str]:
|
||||
"""Zero-pad a valid UPC-A (12 digits) up to its equivalent EAN-13
|
||||
representation. GTIN-8 is intentionally NOT padded (an 8-digit GTIN is
|
||||
its own distinct symbology, not a truncated EAN-13) - returns None for
|
||||
anything that isn't 12 or 13 digits already."""
|
||||
if len(code) == 13:
|
||||
return code
|
||||
if len(code) == 12:
|
||||
return "0" + code
|
||||
return None
|
||||
122
app/services/enrichment/base.py
Normal file
122
app/services/enrichment/base.py
Normal file
@@ -0,0 +1,122 @@
|
||||
"""
|
||||
Base contract every enrichment stage implements.
|
||||
|
||||
A stage takes ONE already-validated catalog row (a plain dict, the same
|
||||
`enhanced_product` shape `catalog_engine.py` builds) plus the brand name,
|
||||
and returns the SAME dict with additional fields merged in - it never
|
||||
removes or renames a key it didn't add itself, and it never raises: any
|
||||
internal failure is caught and reported via `StageOutcome.error` instead.
|
||||
|
||||
To add a new enrichment stage later (HSN, GST, nutrition, allergens, ...):
|
||||
1. Subclass `EnrichmentStage`.
|
||||
2. Implement `async def enrich_one(product, brand) -> StageOutcome`.
|
||||
3. Register an instance in `pipeline.run_default_pipeline()` (or build
|
||||
a custom `EnrichmentPipeline([...])` for a one-off run).
|
||||
No other file needs to change - `catalog_engine.py`'s call site and
|
||||
`vector_store.py`'s upsert already iterate whatever keys are present on the
|
||||
product dict via `.get(...)`, so a new stage's fields flow through to the
|
||||
database/JSON export automatically the same way barcode fields do.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from abc import ABC, abstractmethod
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class StageOutcome:
|
||||
"""Result of running one stage on one product row."""
|
||||
stage_name: str
|
||||
fields: Dict[str, Any] = field(default_factory=dict)
|
||||
error: Optional[str] = None
|
||||
|
||||
@property
|
||||
def ok(self) -> bool:
|
||||
return self.error is None
|
||||
|
||||
|
||||
class EnrichmentStage(ABC):
|
||||
"""One independent, pluggable enrichment step.
|
||||
|
||||
`name` is used for logging/metrics only. `enabled` lets a stage report
|
||||
itself as switched off (e.g. via a settings flag) without the pipeline
|
||||
orchestrator needing to know why - it's simply skipped and every
|
||||
product passes through untouched.
|
||||
"""
|
||||
|
||||
name: str = "unnamed_stage"
|
||||
|
||||
@property
|
||||
def enabled(self) -> bool:
|
||||
return True
|
||||
|
||||
@abstractmethod
|
||||
async def enrich_one(self, product: Dict[str, Any], brand: str) -> StageOutcome:
|
||||
"""Enrich a single product row. MUST NOT raise - catch internally
|
||||
and return a StageOutcome with `error` set instead."""
|
||||
raise NotImplementedError
|
||||
|
||||
async def apply(self, product: Dict[str, Any], brand: str) -> Dict[str, Any]:
|
||||
"""Run this stage on `product` and merge the result in place.
|
||||
Never raises - a stage bug degrades to "no fields added", logged,
|
||||
rather than aborting the whole catalog row.
|
||||
"""
|
||||
try:
|
||||
outcome = await self.enrich_one(product, brand)
|
||||
except Exception as e: # last-resort safety net - stages should
|
||||
# already catch their own errors, but a pipeline-wide guarantee
|
||||
# of "never raises" is worth the redundancy here.
|
||||
logger.error(f"[{self.name}] unhandled exception enriching '{product.get('product_name')}': {e}")
|
||||
return product
|
||||
|
||||
# A STAGE MAY FILL A GAP OR CORRECT A VALUE. IT MAY NOT ERASE ONE.
|
||||
#
|
||||
# This was a plain `product.update(outcome.fields)`, and the barcode
|
||||
# stage returns a fixed nine-key dict whose values are all None when
|
||||
# the lookup finds nothing - so a miss silently replaced the barcode
|
||||
# the shop had typed with NULL. Verified end to end before this guard
|
||||
# existed: a sheet sending 8901262010016 stored None.
|
||||
#
|
||||
# The rule below is the narrowest one that stops it. A stage can still
|
||||
# overwrite a value with a DIFFERENT value, which is what correcting a
|
||||
# field means; it just cannot blank one out. Stages that must not
|
||||
# overwrite at all say so themselves by returning no fields - see
|
||||
# HsnGstEnrichmentStage and BarcodeEnrichmentStage.
|
||||
if outcome.fields:
|
||||
for key, value in outcome.fields.items():
|
||||
# `field_sources` ACCUMULATES; every other key is assigned.
|
||||
#
|
||||
# It is a map keyed by column name, and each stage knows the
|
||||
# provenance of only the columns it filled. Assigning it like
|
||||
# anything else would mean the last stage to run erases what
|
||||
# every earlier stage recorded - so the barcode stage's
|
||||
# provenance would vanish the moment the HSN stage ran, and
|
||||
# the coverage report would show values with no origin.
|
||||
#
|
||||
# A shallow merge is the right depth: each key's value is one
|
||||
# flat record about one column. This mirrors the `||` in
|
||||
# vector_store's ON CONFLICT clause, so the in-memory merge
|
||||
# and the database merge agree.
|
||||
if key == "field_sources" and isinstance(value, dict):
|
||||
merged = dict(product.get("field_sources") or {})
|
||||
merged.update(value)
|
||||
product["field_sources"] = merged
|
||||
continue
|
||||
|
||||
blank_incoming = value is None or (isinstance(value, str) and not value.strip())
|
||||
existing = product.get(key)
|
||||
held = existing is not None and not (isinstance(existing, str) and not existing.strip())
|
||||
if blank_incoming and held:
|
||||
logger.debug(
|
||||
"[%s] kept existing %s=%r rather than blanking it",
|
||||
self.name, key, existing,
|
||||
)
|
||||
continue
|
||||
product[key] = value
|
||||
if not outcome.ok:
|
||||
logger.debug(f"[{self.name}] {product.get('product_name')}: {outcome.error}")
|
||||
return product
|
||||
142
app/services/enrichment/catalog_consensus.py
Normal file
142
app/services/enrichment/catalog_consensus.py
Normal file
@@ -0,0 +1,142 @@
|
||||
"""Fills a blank field from what the brand's OWN rows already agree on.
|
||||
|
||||
WHY THIS EXISTS
|
||||
`fssai_license` was 69.3% filled on 2026-09-08, sourced entirely from a
|
||||
hardcoded 34-brand map (`brand_registry.FSSAI_LICENSES`). A brand outside
|
||||
that map got nothing - except on one path, which got something far worse.
|
||||
|
||||
`user_products._build_product_dict` read:
|
||||
|
||||
fssai_license = req.fssai_license or sample_existing.get(...) or "10012042000244"
|
||||
|
||||
That constant is LION DATES' real, registered FSSAI licence. Any brand with
|
||||
no sample row was stamped with it. This is not a cosmetic default: an FSSAI
|
||||
number identifies the food business legally answerable for the product, and
|
||||
inventing one attributes a stranger's regulatory liability to a product they
|
||||
never made. `scripts/merge_haldiram.py:36` exists because this already
|
||||
reached production once, on `brand_haldirams`.
|
||||
|
||||
The honest source for a blank licence is the brand's own catalog: 400
|
||||
Britannia rows carrying one licence is good evidence for the 401st. That is
|
||||
what this module reads.
|
||||
|
||||
THE RULE IT ENFORCES
|
||||
Propagate only from UNAMBIGUOUS agreement. If a brand's rows carry two
|
||||
different licences, one of them is already wrong and this module returns
|
||||
None rather than picking. A blank field is a gap; a confidently wrong
|
||||
regulatory identifier is a liability.
|
||||
|
||||
Nothing here invents a value. Every result is a value already present on a
|
||||
row of the same brand, which is why the provenance method is
|
||||
`catalog_consensus` and never `sourced`.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from collections import Counter
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# A single dissenting row should not veto 400 agreeing ones, but a genuine
|
||||
# split must. Set so that "399 of 400 agree" propagates and "60/40" does not.
|
||||
_MIN_AGREEMENT = 0.85
|
||||
|
||||
# Below this many populated rows there is no consensus to speak of, only a
|
||||
# coincidence. Two rows agreeing proves nothing about a third.
|
||||
_MIN_ROWS = 3
|
||||
|
||||
|
||||
def _modal(values: List[Any], min_agreement: float = _MIN_AGREEMENT,
|
||||
min_rows: int = _MIN_ROWS) -> Tuple[Optional[Any], Dict[str, Any]]:
|
||||
"""The one value the population agrees on, or None with the reason why."""
|
||||
populated = [v for v in values if v not in (None, "", [], {})]
|
||||
if len(populated) < min_rows:
|
||||
return None, {"reason": "too few populated rows", "rows": len(populated)}
|
||||
|
||||
# Lists (providers) are unhashable; compare them as ordered tuples.
|
||||
keyed = [tuple(v) if isinstance(v, list) else v for v in populated]
|
||||
counts = Counter(keyed)
|
||||
winner, hits = counts.most_common(1)[0]
|
||||
agreement = hits / len(keyed)
|
||||
if agreement < min_agreement:
|
||||
return None, {"reason": "no clear majority", "agreement": round(agreement, 3),
|
||||
"distinct": len(counts)}
|
||||
|
||||
return (list(winner) if isinstance(winner, tuple) else winner), {
|
||||
"agreement": round(agreement, 3), "rows": len(keyed)}
|
||||
|
||||
|
||||
def consensus_value(column: str, rows: List[Dict[str, Any]],
|
||||
min_agreement: float = _MIN_AGREEMENT
|
||||
) -> Tuple[Optional[Any], Dict[str, Any]]:
|
||||
"""The agreed value of `column` across `rows`, plus why it was or was not
|
||||
reached. Never raises: an unreadable row set yields (None, reason)."""
|
||||
try:
|
||||
return _modal([r.get(column) for r in rows], min_agreement=min_agreement)
|
||||
except Exception as e: # pragma: no cover - defensive
|
||||
logger.debug("consensus for %s failed: %s", column, e)
|
||||
return None, {"reason": f"error: {e}"}
|
||||
|
||||
|
||||
def consensus_rows(brand: str, columns: List[str], limit: int = 300) -> List[Dict[str, Any]]:
|
||||
"""Read only the columns consensus needs, for a brand's rows.
|
||||
|
||||
Deliberately NOT `get_products_by_brand`, which is `SELECT *` and therefore
|
||||
carries the 384-dimension embedding on every row. Measured against
|
||||
production: 244 Hindustan Unilever rows cost 3.0 MB that way, 4.7 KB of the
|
||||
7.2 KB per row being an embedding string nothing here looks at.
|
||||
|
||||
Two named columns bring the same read down to roughly 50 KB. On a backend
|
||||
container capped at 2560 MB that difference is not dangerous either way -
|
||||
it is just the difference between reading what is needed and reading
|
||||
everything, once per brand per upload.
|
||||
|
||||
Probes `information_schema` first, because the column set genuinely differs
|
||||
between brand tables and a missing column would otherwise raise.
|
||||
"""
|
||||
from app.services.vector_store import _connect, _sanitize_name
|
||||
|
||||
conn = _connect()
|
||||
if conn is None:
|
||||
return []
|
||||
table = f"brand_{_sanitize_name(brand)}"
|
||||
try:
|
||||
with conn.cursor() as cur:
|
||||
cur.execute(
|
||||
"SELECT column_name FROM information_schema.columns "
|
||||
"WHERE table_schema = 'public' AND table_name = %s",
|
||||
(table,),
|
||||
)
|
||||
present = {r[0] for r in cur.fetchall()}
|
||||
wanted = [c for c in columns if c in present]
|
||||
if not wanted:
|
||||
return []
|
||||
select = ", ".join(f'"{c}"' for c in wanted)
|
||||
cur.execute(f'SELECT {select} FROM "{table}" LIMIT %s', (limit,))
|
||||
return [dict(zip(wanted, row)) for row in cur.fetchall()]
|
||||
except Exception as e: # noqa: BLE001 - defaults are a nicety, not the write
|
||||
logger.debug("consensus read failed for %s: %s", brand, e)
|
||||
return []
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
|
||||
def fssai_for_brand(brand: str, rows: Optional[List[Dict[str, Any]]] = None) -> Tuple[Optional[str], str]:
|
||||
"""The licence to use for a new product of `brand`, and where it came from.
|
||||
|
||||
Order: the curated registry map, then the brand's own rows. Never a
|
||||
constant, never another brand's number.
|
||||
"""
|
||||
from app.services.brand_registry import get_fssai_license
|
||||
|
||||
mapped = get_fssai_license(brand)
|
||||
if mapped:
|
||||
return mapped, "brand_registry"
|
||||
|
||||
if rows:
|
||||
value, _why = consensus_value("fssai_license", rows)
|
||||
if value:
|
||||
return str(value), "catalog_consensus"
|
||||
|
||||
return None, "unknown"
|
||||
1
app/services/enrichment/content/__init__.py
Normal file
1
app/services/enrichment/content/__init__.py
Normal file
@@ -0,0 +1 @@
|
||||
"""Offline content enrichment - the display columns the store pipeline left blank."""
|
||||
112
app/services/enrichment/content/stage.py
Normal file
112
app/services/enrichment/content/stage.py
Normal file
@@ -0,0 +1,112 @@
|
||||
"""Fills `highlights` and `nutrients` for rows the store pipeline leaves empty.
|
||||
|
||||
THE FAILURE THIS ADDRESSES
|
||||
`catalog_engine.generate_product_highlights` and `generate_nutrients_info`
|
||||
have existed for a long time and `brand_discovery._build_product` calls
|
||||
both. The store-catalog pipeline never did: `_to_storage_row` simply passed
|
||||
whatever the sheet had through, so a colleague's upload - which carries
|
||||
neither column - landed `highlights=[]` and `nutrients=[]` on every row.
|
||||
|
||||
That is the whole reason those two columns look healthy in aggregate
|
||||
(95.3% / 69.7% on 2026-09-08) while being empty for exactly the rows this
|
||||
work is about.
|
||||
|
||||
WHAT IT WRITES, AND HOW HONESTLY
|
||||
`highlights` is marketing copy derived from fields we already hold - the
|
||||
category, the pack size, the brand. It is `derived`, never `sourced`.
|
||||
|
||||
`nutrients` is the display list. Where real per-100g figures exist,
|
||||
`nutrition_score_sync.sync_nutrients_to_brand_tables` renders them from
|
||||
`nutrition_facts` and overwrites whatever this stage wrote - that mirror is
|
||||
the better source and runs later. This stage only supplies the
|
||||
category-keyword fallback, flagged `estimated`, so a row is not blank while
|
||||
it waits for a nutrition lookup that may never succeed.
|
||||
|
||||
THE CONSUMABILITY GATE
|
||||
`generate_nutrients_info` works off category keywords, so a Hair Care row
|
||||
whose category or description happens to contain a matching word acquires
|
||||
entries like "Vitamin B Complex - Energy". Shampoo has no nutrients. This
|
||||
stage refuses to write the column at all for a non-consumable, which is the
|
||||
same gate `nutrition_data_service` applies on the lookup path and the same
|
||||
reason `purge_non_consumable_nutrition.py` had to exist.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import Any, Dict, List
|
||||
|
||||
from app.services.enrichment.base import EnrichmentStage, StageOutcome
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class ContentEnrichmentStage(EnrichmentStage):
|
||||
"""Offline, deterministic, fills blanks only. Never raises, never erases."""
|
||||
|
||||
name = "content"
|
||||
|
||||
async def enrich_one(self, product: Dict[str, Any], brand: str) -> StageOutcome:
|
||||
# Imported lazily: catalog_engine pulls in the image and LLM services,
|
||||
# and this stage runs inside ingestion where those are already loaded
|
||||
# but the enrichment package on its own should not require them.
|
||||
from app.core.catalog_engine import (
|
||||
generate_nutrients_info,
|
||||
generate_product_highlights,
|
||||
)
|
||||
from app.services.consumability import is_non_consumable
|
||||
|
||||
fields: Dict[str, Any] = {}
|
||||
sources: Dict[str, Any] = {}
|
||||
|
||||
title = product.get("title") or product.get("product_name") or ""
|
||||
category = product.get("category") or ""
|
||||
|
||||
if not _has_entries(product.get("highlights")):
|
||||
try:
|
||||
highlights = generate_product_highlights(product, brand)
|
||||
except Exception as e: # never abort a row
|
||||
logger.debug("highlight generation failed for %r: %s", title, e)
|
||||
highlights = []
|
||||
if highlights:
|
||||
fields["highlights"] = highlights
|
||||
sources["highlights"] = {"method": "derived",
|
||||
"source": "catalog_engine.generate_product_highlights"}
|
||||
|
||||
if not _has_entries(product.get("nutrients")):
|
||||
if is_non_consumable(category, title):
|
||||
# Not a gap - a column that cannot apply. Recording it stops
|
||||
# the coverage report counting shampoo as missing nutrition
|
||||
# forever, which is what makes someone eventually fabricate it.
|
||||
sources["nutrients"] = {"method": "not_applicable",
|
||||
"source": "non_consumable_product"}
|
||||
else:
|
||||
try:
|
||||
nutrients = generate_nutrients_info(product, brand)
|
||||
except Exception as e:
|
||||
logger.debug("nutrient generation failed for %r: %s", title, e)
|
||||
nutrients = []
|
||||
if nutrients:
|
||||
fields["nutrients"] = nutrients
|
||||
sources["nutrients"] = {
|
||||
"method": "estimated",
|
||||
"source": "catalog_engine.generate_nutrients_info",
|
||||
"note": "category keywords; replaced by real per-100g "
|
||||
"figures when a nutrition lookup succeeds",
|
||||
}
|
||||
|
||||
if sources:
|
||||
fields["field_sources"] = sources
|
||||
|
||||
return StageOutcome(stage_name=self.name, fields=fields)
|
||||
|
||||
|
||||
def _has_entries(value: Any) -> bool:
|
||||
"""True when the column already carries something worth keeping.
|
||||
|
||||
A list of empty strings counts as empty: the spreadsheet parser produces
|
||||
those from a column that exists but has no value in it, and treating one as
|
||||
"already filled" is how a row keeps `['']` forever.
|
||||
"""
|
||||
if not isinstance(value, (list, tuple)):
|
||||
return bool(value)
|
||||
return any(str(v).strip() for v in value)
|
||||
32
app/services/enrichment/hsn_gst/__init__.py
Normal file
32
app/services/enrichment/hsn_gst/__init__.py
Normal file
@@ -0,0 +1,32 @@
|
||||
"""
|
||||
HSN / GST & Pricing enrichment module.
|
||||
|
||||
Public entry points:
|
||||
resolve_hsn_gst(category, product_title="") -> HsnGstInfo
|
||||
Deterministic, offline category -> (HSN code, GST %, review flag)
|
||||
resolution, mirroring the shape already present in the coca-cola
|
||||
and milky_mist catalog exports (hsn_code / gst_percent /
|
||||
hsn_gst_needs_review / selling_price / tax_amount /
|
||||
final_selling_price / cost_price / profit_before_tax /
|
||||
profit_after_tax).
|
||||
|
||||
enrich_pricing_fields(product) -> dict
|
||||
Computes selling_price / tax_amount / final_selling_price (plus the
|
||||
always-null cost/profit fields) from a catalog row's price_range,
|
||||
using the same convention as the existing enriched brands: the
|
||||
selling price is the upper bound of the row's retail price range.
|
||||
|
||||
See the package's module docstrings for the architecture:
|
||||
models.py - HsnGstInfo result type + curated category -> HSN/GST table
|
||||
stage.py - adapts the resolver to the generic EnrichmentStage
|
||||
contract used by app/services/enrichment/pipeline.py
|
||||
"""
|
||||
from app.services.enrichment.hsn_gst.models import HsnGstInfo, resolve_hsn_gst, enrich_pricing_fields
|
||||
from app.services.enrichment.hsn_gst.stage import HsnGstEnrichmentStage
|
||||
|
||||
__all__ = [
|
||||
"HsnGstInfo",
|
||||
"resolve_hsn_gst",
|
||||
"enrich_pricing_fields",
|
||||
"HsnGstEnrichmentStage",
|
||||
]
|
||||
298
app/services/enrichment/hsn_gst/models.py
Normal file
298
app/services/enrichment/hsn_gst/models.py
Normal file
@@ -0,0 +1,298 @@
|
||||
"""
|
||||
Curated, deterministic HSN / GST lookup for catalog product categories.
|
||||
|
||||
WHY THIS IS NOT "MADE UP"
|
||||
--------------------------
|
||||
HSN (Harmonized System of Nomenclature) is the 4-8 digit goods-classification
|
||||
code used on every Indian tax invoice, and GST% is the Indian Goods &
|
||||
Services Tax rate applied to that HSN chapter. Both are public, legal,
|
||||
category-level facts - not per-SKU secrets. Every entry in `HSN_GST_TABLE`
|
||||
below is a *typical* classification for the product category as sold at
|
||||
retail (matching the values already present in the coca-cola / milky_mist
|
||||
catalog exports where those overlap, e.g. Beverages -> 2009, Dairy -> 0401,
|
||||
Dairy - Desserts -> 2105, Tea & Coffee -> 0902).
|
||||
|
||||
The catch: an HSN chapter can cover several GST rates depending on the exact
|
||||
item and how it is packaged, so a category-level guess can be wrong for a
|
||||
specific product. That is precisely what `HSN_GST_NEEDS_REVIEW` is for - it is
|
||||
True whenever the mapping is approximate (the default), and False only for the
|
||||
few categories whose retail classification is unambiguous. The flag exists so
|
||||
a human reviewer / the Streamlit UI can spot-check exactly these rows.
|
||||
|
||||
DESIGN PRINCIPLES
|
||||
------------------
|
||||
- DETERMINISTIC & OFFLINE. No LLM, no network, no randomness. The same
|
||||
category always yields the same (HSN, GST%, review) tuple, so re-runs and
|
||||
backfills are stable and diffable.
|
||||
- ADDITIVE. Only ever attaches new keys to a product row; never removes or
|
||||
rewrites an existing key (the stage skips rows that already carry these
|
||||
fields, so re-running the pipeline over an already-enriched catalog is a
|
||||
no-op).
|
||||
- SAFE DEFAULTS. Unknown/unrecognised categories get hsn_code=None and
|
||||
gst_percent=None with needs_review=True, rather than a fabricated code -
|
||||
a wrong HSN on a real tax document is worse than no HSN at all.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from typing import Dict, Optional, Tuple
|
||||
|
||||
|
||||
# (HSN code, GST %, needs_review). Review=True whenever the category can map
|
||||
# to more than one GST rate / HSN chapter at retail; False for unambiguous ones.
|
||||
HSN_GST_TABLE: Dict[str, Tuple[str, int, bool]] = {
|
||||
# ---- Food & beverage (matches the existing coca-cola / milky_mist exports) ----
|
||||
"Dairy": ("0401", 5, False),
|
||||
"Cheese": ("0406", 12, True),
|
||||
"Dairy - Desserts": ("2105", 5, False),
|
||||
"Ice Cream": ("2105", 18, True),
|
||||
"Beverages": ("2009", 5, False),
|
||||
"Tea & Coffee": ("0902", 5, False),
|
||||
"Food & Beverages": ("2106", 18, True),
|
||||
"Health Drinks": ("2202", 18, True),
|
||||
"Health Foods": ("2106", 18, True),
|
||||
"Breakfast Cereal": ("1904", 18, False),
|
||||
"Chocolates": ("1806", 18, False),
|
||||
"Candy & Confectionery": ("1704", 18, False),
|
||||
"Biscuits & Cookies": ("1905", 18, True),
|
||||
"Crackers": ("1905", 18, True),
|
||||
"Rusk": ("1905", 5, True),
|
||||
"Cakes & Muffins": ("1905", 18, True),
|
||||
"Bakery & Breads": ("1905", 5, False),
|
||||
"Snacks": ("1905", 18, True),
|
||||
"Noodles & Instant Food": ("1902", 18, False),
|
||||
"Pasta & Noodles": ("1902", 18, False),
|
||||
"Atta & Staples": ("1101", 5, False),
|
||||
"Salt & Staples": ("2501", 5, False),
|
||||
"Pulses, Grains & Spices": ("0713", 5, True),
|
||||
"Spices & Masalas": ("0910", 5, False),
|
||||
# Chapter 17: cane/beet sugar and jaggery, 5% for ordinary retail sugar.
|
||||
"Sugar & Jaggery": ("1701", 5, False),
|
||||
"Cooking Oils": ("1517", 5, False),
|
||||
# ---- Loose fresh goods, sold by weight or by the piece ---------------
|
||||
# Chapters 3, 4, 6, 7 and 8. Fresh, unprocessed produce is NIL-rated under
|
||||
# GST - it is not a reduced rate, it is exempt - so 0 here is the real
|
||||
# figure and not a placeholder. The moment any of these is branded and
|
||||
# packaged in a unit container the rate changes, but that product would
|
||||
# resolve to its brand's category rather than these.
|
||||
"Fruits & Vegetables": ("0709", 0, False),
|
||||
"Fresh Herbs & Greens": ("0709", 0, False),
|
||||
"Flowers": ("0603", 0, False),
|
||||
"Fish & Seafood": ("0302", 0, False),
|
||||
"Eggs": ("0407", 0, False),
|
||||
"Pickles & Chutneys": ("2001", 12, True),
|
||||
"Dry Fruits & Nuts": ("0801", 12, True),
|
||||
"Food - Spreads": ("2007", 12, True),
|
||||
"Food - Mixes": ("2106", 18, True),
|
||||
"Food - Soups & Sauces": ("2103", 12, False),
|
||||
# ---- Personal care & household ----
|
||||
"Hair Care": ("3305", 18, False),
|
||||
"Skin Care": ("3304", 18, False),
|
||||
"Skin & Bath Care": ("3307", 18, False),
|
||||
"Bath Soap": ("3401", 18, False),
|
||||
"Beauty Care": ("3304", 18, False),
|
||||
"Oral Care": ("3306", 18, False),
|
||||
"Fragrance & Deodorants": ("3303", 18, False),
|
||||
"Men's Grooming": ("8212", 18, True),
|
||||
"Baby Care": ("3304", 18, True),
|
||||
"Feminine Hygiene": ("9619", 12, False),
|
||||
"Personal Care": ("3307", 18, True),
|
||||
"Detergents & Fabric Care": ("3402", 18, False),
|
||||
"Dishwash": ("3402", 18, False),
|
||||
"Household Cleaning": ("3402", 18, True),
|
||||
"Household - Air Freshener": ("3307", 18, True),
|
||||
"Personal Care - Mosquito Repellent": ("3808", 18, True),
|
||||
# ---- Health care ----
|
||||
"Health Care - Cold & Cough": ("3004", 12, True),
|
||||
"Health Care - Antiseptic": ("3808", 18, True),
|
||||
"Health Care - Ayurvedic": ("3003", 12, True),
|
||||
"Health Care - Digestive": ("3004", 12, True),
|
||||
"Health Care - First Aid": ("3005", 12, True),
|
||||
}
|
||||
|
||||
# Keyword fallbacks so a product whose category label is slightly different
|
||||
# from the table (or missing entirely) still resolves to a sensible chapter.
|
||||
_KEYWORD_FALLBACKS: Tuple[Tuple[str, str, int, bool], ...] = (
|
||||
("toothpaste", "3306", 18, False),
|
||||
("shampoo", "3305", 18, False),
|
||||
("soap", "3401", 18, False),
|
||||
("detergent", "3402", 18, False),
|
||||
("biscuit", "1905", 18, True),
|
||||
("cookie", "1905", 18, True),
|
||||
("chocolate", "1806", 18, False),
|
||||
("candy", "1704", 18, False),
|
||||
("toffee", "1704", 18, False),
|
||||
("milk", "0401", 5, False),
|
||||
("paneer", "0401", 5, False),
|
||||
("curd", "0401", 5, False),
|
||||
("cheese", "0406", 12, True),
|
||||
("ice cream", "2105", 18, True),
|
||||
("juice", "2009", 5, False),
|
||||
("tea", "0902", 5, False),
|
||||
("coffee", "0902", 5, False),
|
||||
("health drink", "2202", 18, True),
|
||||
("chips", "1905", 18, True),
|
||||
("namkeen", "1905", 18, True),
|
||||
("bread", "1905", 5, False),
|
||||
("atta", "1101", 5, False),
|
||||
("flour", "1101", 5, False),
|
||||
("masala", "0910", 5, False),
|
||||
("spice", "0910", 5, False),
|
||||
("pasta", "1902", 18, False),
|
||||
("noodle", "1902", 18, False),
|
||||
("cooking oil", "1517", 5, False),
|
||||
("oil", "1517", 5, True),
|
||||
("pickle", "2001", 12, True),
|
||||
("jam", "2007", 12, True),
|
||||
("honey", "0409", 5, False),
|
||||
("raisin", "0806", 5, True),
|
||||
("dry fruit", "0801", 12, True),
|
||||
("cereal", "1904", 18, False),
|
||||
("deodorant", "3303", 18, False),
|
||||
("perfume", "3303", 18, False),
|
||||
("lipstick", "3304", 18, False),
|
||||
("makeup", "3304", 18, False),
|
||||
("face wash", "3307", 18, False),
|
||||
("lotion", "3307", 18, False),
|
||||
("sunscreen", "3304", 18, False),
|
||||
("razor", "8212", 18, True),
|
||||
("diaper", "9619", 12, False),
|
||||
("mosquito", "3808", 18, True),
|
||||
("repellent", "3808", 18, True),
|
||||
("air freshener", "3307", 18, True),
|
||||
("dishwash", "3402", 18, False),
|
||||
("floor cleaner", "3402", 18, True),
|
||||
("hand wash", "3401", 18, False),
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class HsnGstInfo:
|
||||
"""HSN / GST resolution for one product category, mirroring the fields
|
||||
stored on enriched catalog rows (coca-cola / milky_mist exports)."""
|
||||
hsn_code: Optional[str] = None
|
||||
gst_percent: Optional[int] = None
|
||||
hsn_gst_needs_review: bool = True
|
||||
|
||||
def as_fields(self) -> Dict[str, object]:
|
||||
return {
|
||||
"hsn_code": self.hsn_code,
|
||||
"gst_percent": self.gst_percent,
|
||||
"hsn_gst_needs_review": self.hsn_gst_needs_review,
|
||||
}
|
||||
|
||||
|
||||
_UNKNOWN = HsnGstInfo(hsn_code=None, gst_percent=None, hsn_gst_needs_review=True)
|
||||
|
||||
|
||||
def resolve_hsn_gst(category: Optional[str], product_title: str = "") -> HsnGstInfo:
|
||||
"""Resolve (HSN code, GST %, needs_review) for a product category.
|
||||
|
||||
Exact category matches in `HSN_GST_TABLE` win; otherwise the product
|
||||
title/description is scanned against `_KEYWORD_FALLBACKS`; otherwise
|
||||
returns the safe unknown shape (all-None, needs_review=True).
|
||||
"""
|
||||
cat = (category or "").strip()
|
||||
if cat:
|
||||
exact = HSN_GST_TABLE.get(cat)
|
||||
if exact:
|
||||
return HsnGstInfo(*exact)
|
||||
|
||||
haystack = f"{product_title or ''} {cat}".lower()
|
||||
for kw, hsn, gst, review in _KEYWORD_FALLBACKS:
|
||||
if kw in haystack:
|
||||
return HsnGstInfo(hsn_code=hsn, gst_percent=gst, hsn_gst_needs_review=review)
|
||||
|
||||
return _UNKNOWN
|
||||
|
||||
|
||||
# Matches a price range string of the form "₹12-14" (also tolerates
|
||||
# "Rs 12-14", "12 - 14", a single "₹12", or corrupted rupee symbols).
|
||||
_RANGE_RE = re.compile(r"₹?\s*(\d+(?:\.\d+)?)\s*[-–—]\s*(\d+(?:\.\d+)?)")
|
||||
_SINGLE_RE = re.compile(r"₹?\s*(\d+(?:\.\d+)?)")
|
||||
|
||||
|
||||
def _extract_selling_price(price_range: object) -> Optional[float]:
|
||||
"""Return the upper bound of a row's price range (the convention used by
|
||||
the existing coca-cola / milky_mist exports for `selling_price`), or the
|
||||
single value when the row only carries one price."""
|
||||
if price_range is None:
|
||||
return None
|
||||
text = str(price_range).replace(",", "").strip()
|
||||
if not text:
|
||||
return None
|
||||
m = _RANGE_RE.search(text)
|
||||
if m:
|
||||
try:
|
||||
return float(m.group(2))
|
||||
except ValueError:
|
||||
return None
|
||||
m = _SINGLE_RE.search(text)
|
||||
if m:
|
||||
try:
|
||||
return float(m.group(1))
|
||||
except ValueError:
|
||||
return None
|
||||
return None
|
||||
|
||||
|
||||
def _round2(value: Optional[float]) -> Optional[float]:
|
||||
if value is None:
|
||||
return None
|
||||
return round(value, 2)
|
||||
|
||||
|
||||
def _held_number(value: object) -> Optional[float]:
|
||||
"""A price the row already carries, or None for blank / unparseable."""
|
||||
if value is None or (isinstance(value, str) and not value.strip()):
|
||||
return None
|
||||
try:
|
||||
return float(value)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def enrich_pricing_fields(product: dict) -> Dict[str, object]:
|
||||
"""Compute the pricing fields added by this stage for one catalog row.
|
||||
|
||||
Convention (matches the coca-cola / milky_mist exports):
|
||||
selling_price = upper bound of the row's `price_range`
|
||||
tax_amount = selling_price * gst_percent / 100
|
||||
final_selling_price = selling_price + tax_amount
|
||||
`cost_price`, `profit_before_tax` and `profit_after_tax` are not
|
||||
derivable from the data this pipeline generates, so they stay None
|
||||
(exactly as the existing enriched exports store them).
|
||||
|
||||
A PRICE THE ROW ALREADY HOLDS WINS. A store sheet that says "Retail Price
|
||||
155" has stated a fact; the band derived from it is Rs143-167, and this
|
||||
stage used to read the band's ceiling back as "selling_price = 167" and
|
||||
hand that to `EnrichmentStage.apply`, which overwrites a held value with
|
||||
a different one (it only refuses to BLANK one). Held prices are therefore
|
||||
returned as None here so `apply` keeps them, the base for the tax is the
|
||||
held selling price, then the held final price, and only then the band.
|
||||
"""
|
||||
held_selling = _held_number(product.get("selling_price"))
|
||||
held_final = _held_number(product.get("final_selling_price"))
|
||||
if held_selling is not None:
|
||||
selling_price = held_selling
|
||||
elif held_final is not None:
|
||||
selling_price = held_final
|
||||
else:
|
||||
selling_price = _extract_selling_price(product.get("price_range"))
|
||||
|
||||
gst = product.get("gst_percent")
|
||||
tax_amount = None
|
||||
final_selling_price = None
|
||||
if selling_price is not None and isinstance(gst, (int, float)) and gst and gst > 0:
|
||||
tax_amount = _round2(selling_price * float(gst) / 100.0)
|
||||
final_selling_price = _round2(selling_price + tax_amount)
|
||||
|
||||
return {
|
||||
"selling_price": None if held_selling is not None else _round2(selling_price),
|
||||
"cost_price": None,
|
||||
"tax_amount": tax_amount,
|
||||
"final_selling_price": None if held_final is not None else final_selling_price,
|
||||
"profit_before_tax": None,
|
||||
"profit_after_tax": None,
|
||||
}
|
||||
48
app/services/enrichment/hsn_gst/stage.py
Normal file
48
app/services/enrichment/hsn_gst/stage.py
Normal file
@@ -0,0 +1,48 @@
|
||||
"""Adapts the HSN/GST resolver to the EnrichmentStage contract so it can be
|
||||
registered in app/services/enrichment/pipeline.py's default pipeline.
|
||||
|
||||
Unlike the barcode stage this needs no network access at all - HSN/GST are
|
||||
deterministic, category-level facts - so it is synchronous and effectively
|
||||
free. It is pure ADDITIVE: rows that already carry the hsn_code/gst_percent
|
||||
keys (e.g. re-running the pipeline over a previously-enriched catalog) are
|
||||
left untouched.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import Any, Dict
|
||||
|
||||
from app.infrastructure.settings import ENABLE_HSN_GST_ENRICHMENT
|
||||
from app.services.enrichment.base import EnrichmentStage, StageOutcome
|
||||
from app.services.enrichment.hsn_gst.models import enrich_pricing_fields, resolve_hsn_gst
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class HsnGstEnrichmentStage(EnrichmentStage):
|
||||
name = "hsn_gst"
|
||||
|
||||
@property
|
||||
def enabled(self) -> bool:
|
||||
return ENABLE_HSN_GST_ENRICHMENT
|
||||
|
||||
async def enrich_one(self, product: Dict[str, Any], brand: str) -> StageOutcome:
|
||||
# Already enriched (previous run / manual backfill) - never overwrite.
|
||||
if product.get("hsn_code") is not None or product.get("gst_percent") is not None:
|
||||
return StageOutcome(stage_name=self.name, fields={})
|
||||
|
||||
title = product.get("title") or product.get("product_name") or ""
|
||||
category = product.get("category") or ""
|
||||
|
||||
hsn_info = resolve_hsn_gst(category, title)
|
||||
fields: Dict[str, Any] = hsn_info.as_fields()
|
||||
# Pricing fields depend on gst_percent, which was just resolved
|
||||
# above, so hand the resolved rate to the pricing helper.
|
||||
fields.update(enrich_pricing_fields({**product, "gst_percent": fields["gst_percent"]}))
|
||||
|
||||
needs_review = fields["hsn_gst_needs_review"]
|
||||
return StageOutcome(
|
||||
stage_name=self.name,
|
||||
fields=fields,
|
||||
error=None if not needs_review else "category-level HSN/GST estimate - verify against the physical pack"
|
||||
)
|
||||
112
app/services/enrichment/pipeline.py
Normal file
112
app/services/enrichment/pipeline.py
Normal file
@@ -0,0 +1,112 @@
|
||||
"""
|
||||
Orchestrates a list of independent `EnrichmentStage`s over a batch of
|
||||
catalog rows, running each stage's per-row work concurrently (bounded by
|
||||
`max_concurrency`) and NEVER letting one row's failure affect any other
|
||||
row or stage.
|
||||
|
||||
`store_catalog_pipeline.stages_8_9_enrichment()` builds an EnrichmentPipeline
|
||||
and runs it between SKU resolution and the validation gate; see
|
||||
docs/BARCODE_ENRICHMENT.md for the pipeline-position rationale.
|
||||
|
||||
This docstring used to say `catalog_engine.py` called `run_default_pipeline()`
|
||||
at a "Step 2.6". It never did - that module does not import this one at all,
|
||||
so the brand-name generation path gets no barcode, HSN/GST or content
|
||||
enrichment. Step 2.6 there is now the validation gate only. Wiring enrichment
|
||||
into that path is a real and separate piece of work; do not read this comment
|
||||
as saying it is already done.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
from typing import Any, Dict, List, Optional, Sequence
|
||||
|
||||
from app.services.enrichment.base import EnrichmentStage
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class EnrichmentPipeline:
|
||||
def __init__(self, stages: Sequence[EnrichmentStage], max_concurrency: int = 5):
|
||||
self.stages = [s for s in stages if s.enabled]
|
||||
self.max_concurrency = max(1, max_concurrency)
|
||||
|
||||
async def run(self, products: List[Dict[str, Any]], brand: str) -> List[Dict[str, Any]]:
|
||||
if not self.stages or not products:
|
||||
return products
|
||||
|
||||
semaphore = asyncio.Semaphore(self.max_concurrency)
|
||||
|
||||
async def _run_row(product: Dict[str, Any]) -> Dict[str, Any]:
|
||||
async with semaphore:
|
||||
for stage in self.stages:
|
||||
product = await stage.apply(product, brand)
|
||||
return product
|
||||
|
||||
results = await asyncio.gather(*(_run_row(p) for p in products), return_exceptions=True)
|
||||
|
||||
final: List[Dict[str, Any]] = []
|
||||
for original, result in zip(products, results):
|
||||
if isinstance(result, Exception):
|
||||
logger.error(f"Enrichment pipeline failed for '{original.get('product_name')}': {result}")
|
||||
final.append(original)
|
||||
else:
|
||||
final.append(result)
|
||||
return final
|
||||
|
||||
|
||||
def _build_default_stages() -> List[EnrichmentStage]:
|
||||
"""Import stages lazily so importing this module never pulls in a
|
||||
stage's own dependencies (network clients, DB drivers, ...) unless a
|
||||
default pipeline is actually requested."""
|
||||
stages: List[EnrichmentStage] = []
|
||||
try:
|
||||
from app.services.enrichment.barcode.stage import BarcodeEnrichmentStage
|
||||
stages.append(BarcodeEnrichmentStage())
|
||||
except Exception as e:
|
||||
logger.error(f"Barcode enrichment stage unavailable: {e}")
|
||||
|
||||
# Runs AFTER the lookup so it normalises whatever that found, and runs at
|
||||
# all even when the lookup is disabled - which is the point. It derives
|
||||
# barcode_type/gtin/ean13/upc from a barcode the row already has, offline
|
||||
# and for free, so a sheet-supplied barcode finally gets validated and
|
||||
# expanded instead of going straight to the database unchecked.
|
||||
try:
|
||||
from app.services.enrichment.barcode.identity_stage import BarcodeIdentityStage
|
||||
stages.append(BarcodeIdentityStage())
|
||||
except Exception as e:
|
||||
logger.error(f"Barcode identity stage unavailable: {e}")
|
||||
|
||||
# HSN / GST & pricing enrichment (see app/services/enrichment/hsn_gst/) -
|
||||
# deterministic, offline, pure-additive. Runs AFTER the barcode stage so
|
||||
# every stored/exported row carries both sets of fields; a failure here
|
||||
# degrades a row to "no HSN/GST attached", never drops the row.
|
||||
try:
|
||||
from app.services.enrichment.hsn_gst.stage import HsnGstEnrichmentStage
|
||||
stages.append(HsnGstEnrichmentStage())
|
||||
except Exception as e:
|
||||
logger.error(f"HSN/GST enrichment stage unavailable: {e}")
|
||||
|
||||
# Offline display columns. Registered last so the nutrients fallback it
|
||||
# writes is the lowest-priority source: the real per-100g figures mirrored
|
||||
# by nutrition_score_sync overwrite it whenever a lookup succeeds.
|
||||
try:
|
||||
from app.services.enrichment.content.stage import ContentEnrichmentStage
|
||||
stages.append(ContentEnrichmentStage())
|
||||
except Exception as e:
|
||||
logger.error(f"Content enrichment stage unavailable: {e}")
|
||||
|
||||
return stages
|
||||
|
||||
|
||||
_default_pipeline: Optional[EnrichmentPipeline] = None
|
||||
|
||||
|
||||
async def run_default_pipeline(products: List[Dict[str, Any]], brand: str,
|
||||
max_concurrency: int = 5) -> List[Dict[str, Any]]:
|
||||
"""Convenience entry point used by catalog_engine.py: runs every
|
||||
currently-registered, enabled enrichment stage over `products`."""
|
||||
global _default_pipeline
|
||||
if _default_pipeline is None:
|
||||
_default_pipeline = EnrichmentPipeline(_build_default_stages(), max_concurrency=max_concurrency)
|
||||
return await _default_pipeline.run(products, brand)
|
||||
254
app/services/enrichment/post_ingest_barcodes.py
Normal file
254
app/services/enrichment/post_ingest_barcodes.py
Normal file
@@ -0,0 +1,254 @@
|
||||
"""Finds barcodes for freshly-ingested rows, in bulk, before nutrition runs.
|
||||
|
||||
WHY THIS RUNS BEFORE THE NUTRITION JOB, NOT ALONGSIDE IT
|
||||
--------------------------------------------------------
|
||||
Ordering here is a correctness property, not a preference.
|
||||
|
||||
`fetch_verified_nutrition_by_barcode` matches on the GTIN and returns at
|
||||
confidence 0.95. The name search it falls back to accepts at a minimum of 0.32.
|
||||
The nutrition job runs with `skip_if_verified=True`, so whichever path lands
|
||||
first WINS PERMANENTLY - a 0.32 name match blocks the 0.95 barcode match from
|
||||
ever being attempted. Two jobs racing would produce exactly that, silently, and
|
||||
the catalog would end up with the worse of two available answers.
|
||||
|
||||
So this is a phase inside the same job, ahead of the nutrition phases.
|
||||
|
||||
WHY BULK, NOT THE PER-PRODUCT CASCADE
|
||||
-------------------------------------
|
||||
The per-product search endpoint Open Food Facts exposes is capped at 10
|
||||
requests per minute. A 200-row upload is twenty minutes of waiting, which is
|
||||
why `ENABLE_BARCODE_LOOKUP` defaults false (settings.py:420-423) and why the
|
||||
inline stage stays off.
|
||||
|
||||
`off_bulk.fetch_brand_corpus` fetches a brand's ENTIRE Open Food Facts
|
||||
catalogue in about five requests and matches offline against it. A brand is a
|
||||
brand whether it has 3 rows or 300, so the cost is per brand, not per product.
|
||||
That is what makes barcode enrichment affordable on the shared host at all.
|
||||
|
||||
It also uses `off_bulk.score_candidates` rather than the live matcher, because
|
||||
that scorer already fixes two measured flaws: `matching.name_similarity` is
|
||||
asymmetric ("Butter milk amul" vs "Amul Butter" scores 0.882 one way and 0.418
|
||||
the other), and `matching.size_matches` vetoes any candidate with a blank size
|
||||
when 57 of 146 Amul OFF records have `quantity: null`.
|
||||
|
||||
THE THRESHOLD IS 0.88, NOT 0.78
|
||||
-------------------------------
|
||||
`BARCODE_MIN_NAME_SIMILARITY` (0.78) is the floor for the REVERSE direction,
|
||||
where a barcode has already established identity and the name is a sanity
|
||||
check. This is the forward direction: many candidates compete and the name
|
||||
carries the whole decision. `scripts/backfill_barcodes_from_off.py` measured
|
||||
0.88 as the safe auto-apply point and 0.70-0.88 as review-only, and this reuses
|
||||
that number rather than inventing one.
|
||||
|
||||
WHAT IT WRITES
|
||||
barcode, barcode_type, gtin, ean13, barcode_source, barcode_verified=False,
|
||||
barcode_lookup_status='name_matched', barcode_last_updated, and the
|
||||
field_sources record - via targeted UPDATEs that pin the row's current
|
||||
value, never via upsert_brand_products.
|
||||
|
||||
WHY NOT THE UPSERT
|
||||
`get_products_by_brand` is `SELECT *`, so a row's `embedding` comes back as
|
||||
a pgvector string, and `upsert_brand_products` only accepts a list - it
|
||||
would write NULL and destroy the embedding. Targeted UPDATEs also cannot
|
||||
clobber a concurrent write.
|
||||
|
||||
WHAT IT WILL NOT DO
|
||||
It never overwrites a barcode a row already holds. A merchant typing one in
|
||||
is holding the pack; nothing found by name similarity outranks that.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import time
|
||||
from typing import Any, Dict, Iterable, List, Optional, Tuple
|
||||
|
||||
from psycopg.types.json import Json
|
||||
|
||||
from app.services.enrichment.barcode.validators import (
|
||||
classify_barcode_type,
|
||||
to_ean13,
|
||||
validate_barcode,
|
||||
)
|
||||
from app.services.vector_store import _connect, _sanitize_name
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Matches scripts/backfill_barcodes_from_off.py, which measured it.
|
||||
DEFAULT_MIN_SIMILARITY = 0.88
|
||||
|
||||
BARCODE_SOURCE = "openfoodfacts_bulk (world.openfoodfacts.org/api/v2)"
|
||||
|
||||
|
||||
def _rows_needing_a_barcode(cur, table: str) -> List[Dict[str, Any]]:
|
||||
"""Rows with no usable barcode. Probes the column list first, because a
|
||||
table written before the schema migration may still lack the newer ones."""
|
||||
cur.execute(
|
||||
"SELECT column_name FROM information_schema.columns "
|
||||
"WHERE table_schema = 'public' AND table_name = %s",
|
||||
(table,),
|
||||
)
|
||||
present = {r[0] for r in cur.fetchall()}
|
||||
if not {"id", "product_name", "barcode"} <= present:
|
||||
return []
|
||||
|
||||
size = "size" if "size" in present else "NULL AS size"
|
||||
cur.execute(
|
||||
f'SELECT id, product_name, title, {size}, category, barcode '
|
||||
f'FROM "{table}" '
|
||||
f"WHERE barcode IS NULL OR btrim(barcode) = ''"
|
||||
)
|
||||
return [{"id": r[0], "product_name": r[1], "title": r[2], "size": r[3],
|
||||
"category": r[4], "barcode": r[5]} for r in cur.fetchall()]
|
||||
|
||||
|
||||
def enrich_brand_barcodes(brand: str, *, min_similarity: float = DEFAULT_MIN_SIMILARITY,
|
||||
dry_run: bool = False,
|
||||
progress_cb=None) -> Dict[str, int]:
|
||||
"""Fill blank barcodes for one brand from its Open Food Facts corpus.
|
||||
|
||||
Never raises: a brand whose corpus cannot be fetched reports zero and the
|
||||
caller moves to the next one. Enrichment is best-effort by contract.
|
||||
"""
|
||||
from app.services.enrichment.barcode.sources.off_bulk import (
|
||||
brand_tokens,
|
||||
fetch_brand_corpus,
|
||||
score_candidates,
|
||||
)
|
||||
|
||||
stats = {"candidates": 0, "matched": 0, "written": 0, "rejected": 0}
|
||||
|
||||
# The same refusal `store_catalog_pipeline.stages_8_9_enrichment` makes for
|
||||
# the inline stages, for the same reason: a barcode identifies a
|
||||
# manufactured article and the Own Products bucket is loose produce - an
|
||||
# apple, a bunch of coriander. There is no GTIN to find, and a name match
|
||||
# against some packaged product's corpus could only attach the wrong one.
|
||||
from app.services.generic_products import OWN_PRODUCTS_BRAND
|
||||
if brand == OWN_PRODUCTS_BRAND:
|
||||
return stats
|
||||
|
||||
table = f"brand_{_sanitize_name(brand)}"
|
||||
|
||||
conn = _connect()
|
||||
if conn is None:
|
||||
return stats
|
||||
|
||||
try:
|
||||
with conn.cursor() as cur:
|
||||
rows = _rows_needing_a_barcode(cur, table)
|
||||
if not rows:
|
||||
return stats
|
||||
stats["candidates"] = len(rows)
|
||||
|
||||
try:
|
||||
# Returns the hit LIST directly (the on-disk cache file wraps it in
|
||||
# a "hits" key; the function unwraps it). About five requests for a
|
||||
# whole brand, then served from disk on later runs.
|
||||
hits = fetch_brand_corpus(brand) or []
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.warning("OFF corpus unavailable for %s: %s", brand, e)
|
||||
return stats
|
||||
|
||||
if not hits:
|
||||
logger.info("Open Food Facts holds no India catalogue for %s", brand)
|
||||
return stats
|
||||
|
||||
# Stripped from both sides before names are compared, so "Hindustan
|
||||
# Unilever Hul Lux" reduces to "lux" on our side and matches OFF's
|
||||
# "Lux". Computed once per brand, not once per row.
|
||||
drop = brand_tokens(brand)
|
||||
|
||||
for index, row in enumerate(rows):
|
||||
if progress_cb:
|
||||
progress_cb(index, len(rows))
|
||||
|
||||
title = row.get("title") or row.get("product_name") or ""
|
||||
try:
|
||||
scored = score_candidates(
|
||||
hits, title, [row.get("size") or ""], drop,
|
||||
review_min=min_similarity,
|
||||
)
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.debug("scoring failed for %r: %s", title, e)
|
||||
continue
|
||||
|
||||
# score_candidates already drops anything under review_min, so the
|
||||
# first entry is the best acceptable one. The explicit re-check is
|
||||
# kept because the ordering contract is "best first", not "all
|
||||
# above the floor" - relying on the filter alone would silently
|
||||
# break if that ever changed.
|
||||
best = scored[0] if scored else None
|
||||
if not best or best.score < min_similarity:
|
||||
stats["rejected"] += 1
|
||||
continue
|
||||
|
||||
code = validate_barcode(getattr(best, "barcode", None))
|
||||
if not code:
|
||||
stats["rejected"] += 1
|
||||
continue
|
||||
|
||||
stats["matched"] += 1
|
||||
if dry_run:
|
||||
continue
|
||||
|
||||
kind = classify_barcode_type(code)
|
||||
sources = {
|
||||
"barcode": {"method": "sourced", "source": BARCODE_SOURCE,
|
||||
"confidence": round(float(best.score), 3),
|
||||
"note": "matched on name against the brand's OFF "
|
||||
"catalogue; not verified against the pack"},
|
||||
"gtin": {"method": "derived", "source": "validators.validate_barcode"},
|
||||
"ean13": {"method": "derived", "source": "validators.to_ean13"},
|
||||
}
|
||||
try:
|
||||
with conn.cursor() as cur:
|
||||
cur.execute(
|
||||
f'UPDATE "{table}" SET barcode = %s, barcode_type = %s, '
|
||||
f"gtin = %s, ean13 = %s, barcode_source = %s, "
|
||||
f"barcode_verified = FALSE, barcode_lookup_status = %s, "
|
||||
f"barcode_last_updated = NOW(), "
|
||||
f"field_sources = COALESCE(field_sources, '{{}}'::jsonb) "
|
||||
f" || %s::jsonb, "
|
||||
f"updated_at = CURRENT_TIMESTAMP "
|
||||
f"WHERE id = %s "
|
||||
f" AND (barcode IS NULL OR btrim(barcode) = '')",
|
||||
(code, kind.value, code, to_ean13(code), BARCODE_SOURCE,
|
||||
"name_matched", Json(sources), row["id"]),
|
||||
)
|
||||
stats["written"] += cur.rowcount
|
||||
conn.commit()
|
||||
except Exception as e: # noqa: BLE001
|
||||
conn.rollback()
|
||||
logger.warning("barcode write failed for %s id=%s: %s",
|
||||
table, row["id"], e)
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
return stats
|
||||
|
||||
|
||||
def enrich_barcodes_for_brands(brands: Iterable[str], *,
|
||||
min_similarity: float = DEFAULT_MIN_SIMILARITY,
|
||||
dry_run: bool = False,
|
||||
progress_cb=None) -> Dict[str, Dict[str, int]]:
|
||||
"""Run `enrich_brand_barcodes` over several brands, one at a time.
|
||||
|
||||
Deliberately sequential. `EnrichmentPipeline`'s five-way concurrency is for
|
||||
per-row work against a local corpus; firing five brand-corpus fetches at
|
||||
Open Food Facts at once is how a shared host earns a rate limit.
|
||||
"""
|
||||
out: Dict[str, Dict[str, int]] = {}
|
||||
names = [b.strip() for b in brands if b and b.strip()]
|
||||
for i, brand in enumerate(names):
|
||||
if progress_cb:
|
||||
progress_cb(i, len(names))
|
||||
try:
|
||||
stats = enrich_brand_barcodes(brand, min_similarity=min_similarity,
|
||||
dry_run=dry_run)
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.warning("barcode enrichment failed for %s: %s", brand, e)
|
||||
continue
|
||||
if stats.get("candidates"):
|
||||
out[brand] = stats
|
||||
logger.info("%s: %d without a barcode, %d matched, %d written",
|
||||
brand, stats["candidates"], stats["matched"], stats["written"])
|
||||
return out
|
||||
384
app/services/generic_products.py
Normal file
384
app/services/generic_products.py
Normal file
@@ -0,0 +1,384 @@
|
||||
"""
|
||||
Unbranded grocery commodities, and the one table they belong in.
|
||||
|
||||
WHY THIS FILE EXISTS
|
||||
--------------------
|
||||
A store sheet routinely lists items that simply have no brand: "Toor Dhal 1kg",
|
||||
"Sugar", "Salt 1kg", "Black Pepper 100g". Nothing rejected those rows - what
|
||||
happened was worse. `store_catalog_pipeline.infer_brand()` falls back to the
|
||||
first word of the name, and `brand_registry.resolve_parent_brand()` then does a
|
||||
bidirectional whole-word scan over ~230 aliases. Between them:
|
||||
|
||||
"Toor Dhal 1kg" -> brand "Toor" -> a junk table brand_toor
|
||||
"Sugar 1kg" -> brand "Sugar" -> a junk table brand_sugar
|
||||
"Salt 1kg" -> "colgate active salt" contains "salt"
|
||||
-> brand_colgate_palmolive
|
||||
"Milk 1L" -> "cadbury dairy milk" contains "milk"
|
||||
-> brand_cadbury
|
||||
"Butter 500g" -> "nestle butter" -> brand_nestle
|
||||
"Red Chilli Powder" -> "brooke bond red label" -> brand_brooke_bond
|
||||
|
||||
The junk tables are noise. The misroutes are real damage: generic groceries
|
||||
written into the live Amul, Nestle, Cadbury and HUL catalogs - and
|
||||
`stage_1_brand_and_fssai` then stamps that brand's FSSAI licence number onto the
|
||||
row, so unbranded chilli powder ships carrying Brooke Bond's real licence.
|
||||
|
||||
WHY A WORD LIST AND NOT "THE BRAND IS UNKNOWN"
|
||||
----------------------------------------------
|
||||
"Not in BRAND_ALIASES" is the obvious rule and it is wrong here. That map holds
|
||||
231 mostly-large FMCG names; Bikaji, Aachi, Idhayam, Naga and Lion Dates are all
|
||||
real brands absent from it. Treating unknown as unbranded would sweep every
|
||||
regional brand into one bucket.
|
||||
|
||||
So the test is POSITIVE and conservative: strip the pack size and the words that
|
||||
carry no brand signal, and require that *everything still standing* is a
|
||||
commodity noun. One unrecognised token means "this is a brand". Hence:
|
||||
|
||||
"Butter 500g" -> {butter} -> unbranded
|
||||
"Amul Butter 500g" -> {amul, butter} -> branded, unchanged
|
||||
"Aachi Sambar" -> {aachi, sambar} -> branded, unchanged
|
||||
|
||||
The lexicon doubles as a category map, so the same entry that identifies a
|
||||
commodity also says which canonical category it belongs to - a bag of dal should
|
||||
not have to go through keyword detection twice.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import Dict, Optional, Set
|
||||
|
||||
from app.services.category_units import parse_unit
|
||||
|
||||
# The single brand every unbranded product is filed under. `_sanitize_name`
|
||||
# turns this into the table `brand_own_products`, and `display_name_for_suffix`
|
||||
# turns that back into "Own Products" for the UI, so the round trip the frontend
|
||||
# depends on (card -> /api/brands/{brand}/products) is an identity.
|
||||
#
|
||||
# Deliberately NOT registered in BRAND_ALIASES: an alias overlapping these words
|
||||
# would let resolve_parent_brand hijack the table, which is the exact class of
|
||||
# bug this module exists to end.
|
||||
OWN_PRODUCTS_BRAND = "Own Products"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The lexicon: commodity term -> canonical category
|
||||
# ---------------------------------------------------------------------------
|
||||
# Categories are the canonical names from category_registry.ALL_CATEGORIES, so
|
||||
# HSN/GST resolution and the pack-size unit rulebook both key off them without
|
||||
# a second translation step.
|
||||
_PULSES = "Pulses, Grains & Spices"
|
||||
_STAPLES = "Atta & Staples"
|
||||
_SPICES = "Spices & Masalas"
|
||||
_SUGAR = "Sugar & Jaggery"
|
||||
_SALT = "Salt & Staples"
|
||||
_OILS = "Cooking Oils"
|
||||
_DAIRY = "Dairy"
|
||||
_BEVERAGE = "Beverages"
|
||||
# Fresh, loose goods. These are sold by weight or by the piece and carry no
|
||||
# brand at all, which is precisely why they were the worst offenders: before
|
||||
# these existed, "Apple" resolved to a brand called Apple and "Red Rose" was
|
||||
# whole-word matched into brand_brooke_bond, carrying Brooke Bond's FSSAI
|
||||
# licence onto a flower.
|
||||
_PRODUCE = "Fruits & Vegetables"
|
||||
_GREENS = "Fresh Herbs & Greens"
|
||||
_FLOWERS = "Flowers"
|
||||
_SEAFOOD = "Fish & Seafood"
|
||||
_EGGS = "Eggs"
|
||||
|
||||
COMMODITY_TERMS: Dict[str, str] = {}
|
||||
|
||||
|
||||
def _add(category: str, *terms: str) -> None:
|
||||
for term in terms:
|
||||
COMMODITY_TERMS[term] = category
|
||||
|
||||
|
||||
# Pulses and lentils. Indian sheets spell dal a dozen ways.
|
||||
_add(_PULSES,
|
||||
"dal", "dhal", "dhall", "daal", "dail", "pulse", "pulses", "lentil", "lentils",
|
||||
"toor", "tur", "arhar", "urad", "urid", "moong", "mung", "masoor", "masur",
|
||||
"chana", "channa", "gram", "rajma", "lobia", "kabuli", "peas", "matar",
|
||||
"soya", "soyabean")
|
||||
|
||||
# Grains, flours and other dry staples.
|
||||
_add(_STAPLES,
|
||||
"rice", "basmati", "sona", "masoori", "ponni", "idli", "sona masoori",
|
||||
"wheat", "atta", "maida", "sooji", "suji", "rava", "semolina", "besan",
|
||||
"poha", "aval", "ragi", "bajra", "jowar", "millet", "millets", "quinoa",
|
||||
"sabudana", "vermicelli", "corn", "oats", "flour")
|
||||
|
||||
# Sweeteners.
|
||||
_add(_SUGAR, "sugar", "jaggery", "gur", "misri", "honey", "sakkarai")
|
||||
|
||||
# Salt.
|
||||
_add(_SALT, "salt", "sendha", "iodised", "iodized")
|
||||
|
||||
# Spices, whole and ground.
|
||||
_add(_SPICES,
|
||||
"pepper", "peppercorn", "peppercorns", "turmeric", "haldi", "manjal",
|
||||
"chilli", "chili", "chillies", "chile", "mirchi", "coriander", "dhania",
|
||||
"cumin", "jeera", "mustard", "methi", "fenugreek", "cardamom", "elaichi",
|
||||
"clove", "cloves", "lavang", "cinnamon", "dalchini", "bay", "tejpatta",
|
||||
"asafoetida", "hing", "tamarind", "imli", "masala", "garam", "sambar",
|
||||
"rasam", "ajwain", "saunf", "fennel", "nutmeg", "mace", "star", "anise",
|
||||
"kalonji", "poppy", "khus")
|
||||
|
||||
# Edible oils. Bare "oil" currently resolves to Johnson & Johnson.
|
||||
_add(_OILS,
|
||||
"oil", "gingelly", "groundnut", "peanut", "sunflower", "sesame", "til",
|
||||
"coconut", "castor", "vanaspati")
|
||||
|
||||
# Loose dairy. These are the ones misrouting into real brand tables today.
|
||||
_add(_DAIRY,
|
||||
"milk", "butter", "ghee", "paneer", "curd", "dahi", "yogurt", "yoghurt",
|
||||
"cheese", "khoa", "khoya", "cream", "buttermilk", "lassi")
|
||||
|
||||
# Loose tea and coffee.
|
||||
_add(_BEVERAGE, "tea", "coffee", "chai")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Fresh produce
|
||||
# ---------------------------------------------------------------------------
|
||||
# WHY THIS BLOCK IS SAFE TO ADD.
|
||||
#
|
||||
# The all-tokens-must-be-commodities rule means every word added here makes the
|
||||
# test MORE permissive, so the risk is real brands collapsing into this bucket.
|
||||
# That was measured, not assumed, before these went in: all 1,414 products in
|
||||
# the live catalogue were reclassified with this list applied, and exactly one
|
||||
# changed - `24 Mantra Organic Moong Dal 500g`, which was ALREADY misfiled by
|
||||
# the `_strip_sizes` bug fixed below and has nothing to do with produce.
|
||||
#
|
||||
# The other check was against BRAND_ALIASES: of 113 candidate terms only "amla"
|
||||
# appears in an alias ("dabur amla"), and that is harmless because "dabur" is
|
||||
# not a commodity, so "Dabur Amla" keeps every token it needs to stay branded.
|
||||
# Bare "Amla" is fruit and belongs here.
|
||||
#
|
||||
# RE-RUN BOTH CHECKS BEFORE ADDING A WORD TO THIS BLOCK. See
|
||||
# tests/test_generic_products_produce.py, which encodes them.
|
||||
|
||||
# Fruit. Indian sheets mix English, Tamil and Hindi names freely.
|
||||
_add(_PRODUCE,
|
||||
"apple", "orange", "banana", "grape", "grapes", "mango", "pineapple",
|
||||
"papaya", "guava", "pomegranate", "watermelon", "muskmelon", "melon",
|
||||
"lemon", "lime", "mosambi", "sathukudi", "sapota", "chikoo", "jackfruit",
|
||||
"fig", "anjeer", "pear", "peach", "plum", "apricot", "cherry", "cherries",
|
||||
"strawberry", "blueberry", "kiwi", "litchi", "lychee", "amla", "gooseberry",
|
||||
"avocado", "dragonfruit", "rambutan", "mangosteen", "starfruit", "jamun",
|
||||
"ber", "plantain", "vazhaikkai")
|
||||
|
||||
# Vegetables. "gourd" covers the whole family once _PHRASES has collapsed the
|
||||
# two-word forms (bitter gourd, bottle gourd, snake gourd, ridge gourd).
|
||||
_add(_PRODUCE,
|
||||
"tomato", "potato", "onion", "carrot", "beetroot", "beet", "radish",
|
||||
"turnip", "cabbage", "cauliflower", "broccoli", "brinjal", "eggplant",
|
||||
"aubergine", "okra", "bhindi", "cucumber", "pumpkin", "gourd", "drumstick",
|
||||
"beans", "bean", "capsicum", "garlic", "ginger", "yam", "colocasia",
|
||||
"tapioca", "arbi", "zucchini", "chayote", "sweetcorn", "babycorn",
|
||||
"mushroom", "leek", "celery", "lettuce", "shallot", "springonion")
|
||||
|
||||
# Greens and fresh herbs, sold in bunches and never branded. NOTE that
|
||||
# "coriander" and "methi" are deliberately NOT here: they are already spices in
|
||||
# the block above, `_add` is last-wins, and re-adding them would silently move
|
||||
# dhania powder out of Spices & Masalas. The leaf forms are collapsed to
|
||||
# distinct tokens by _PHRASES instead.
|
||||
_add(_GREENS,
|
||||
"spinach", "palak", "amaranth", "keerai", "greens", "cilantro", "mint",
|
||||
"pudina", "curryleaves", "basil", "thulasi", "tulsi", "parsley", "dill",
|
||||
"sorrel", "moringa", "methileaves", "bunch")
|
||||
|
||||
# Flowers. Sold loose or by the metre of garland; the reason "Red Rose" used to
|
||||
# land in a tea catalogue.
|
||||
_add(_FLOWERS,
|
||||
"flower", "flowers", "rose", "jasmine", "malli", "lotus", "marigold",
|
||||
"samanthi", "chrysanthemum", "kanakambaram", "arali", "garland", "poo")
|
||||
|
||||
# Fish and seafood, sold fresh by weight.
|
||||
_add(_SEAFOOD,
|
||||
"fish", "prawn", "prawns", "shrimp", "crab", "squid", "tuna", "mackerel",
|
||||
"sardine", "pomfret", "seer", "vanjaram", "anchovy", "nethili", "sole",
|
||||
"tilapia", "salmon", "shellfish", "clam", "mussel")
|
||||
|
||||
# Eggs.
|
||||
_add(_EGGS, "egg", "eggs", "muttai", "quail")
|
||||
|
||||
# Stragglers found by running the produce base list through is_unbranded and
|
||||
# fixing every row it refused. Kept in one block so the next person adding to
|
||||
# the seed list knows where the tail ends up.
|
||||
_add(_PRODUCE, "dates", "custard", "dragonfruit", "ivy", "raw")
|
||||
_add(_GREENS, "agathi", "ponnanganni", "keerai")
|
||||
_add(_FLOWERS, "tuberose", "lily")
|
||||
|
||||
|
||||
# Words that describe a product without naming a brand. Stripped before the
|
||||
# all-tokens-are-commodities test, so "Organic Toor Dal Whole 1kg" still reads
|
||||
# as unbranded.
|
||||
QUALIFIERS: Set[str] = {
|
||||
# quality / provenance
|
||||
"organic", "premium", "fresh", "natural", "pure", "best", "quality",
|
||||
"grade", "select", "special", "classic", "regular", "standard", "economy",
|
||||
"value", "farm", "country", "desi", "local", "homemade", "traditional",
|
||||
# processing / form
|
||||
"whole", "half", "split", "raw", "roasted", "unroasted", "polished",
|
||||
"unpolished", "sortex", "cleaned", "washed", "refined", "filtered",
|
||||
"double", "single", "extra", "fine", "coarse", "powder", "powdered",
|
||||
"ground", "crushed", "flakes", "seeds", "seed", "granules", "crystal",
|
||||
"crystals", "cube", "cubes", "stick", "sticks", "dried", "dry",
|
||||
"slice", "slices", "sliced", "block", "grated", "shredded", "chopped",
|
||||
# colour / variety, which qualify a commodity rather than brand it
|
||||
"black", "white", "red", "green", "yellow", "brown", "long", "short",
|
||||
"small", "big", "large", "medium",
|
||||
# packaging / retail noise
|
||||
"pack", "packet", "packed", "loose", "bag", "pouch", "box", "tin", "jar",
|
||||
"bottle", "refill", "combo", "assorted", "mixed", "mix",
|
||||
# connectives
|
||||
"and", "with", "of", "the", "in", "for",
|
||||
# Form words for fresh goods. "Leaves" is the important one: without it
|
||||
# "Mint Leaves" keeps an unknown token and reads as a brand.
|
||||
"leaves", "leaf", "bunch", "sweet", "broad", "cluster", "full", "toned",
|
||||
"seedless", "ripe", "tender", "baby", "country", "hybrid", "nati",
|
||||
# Varietal names. A variety qualifies a commodity, it does not brand it:
|
||||
# an Alphonso mango is a mango. None of these appears in BRAND_ALIASES -
|
||||
# the produce test asserts that, so a future addition cannot smuggle a
|
||||
# real brand in through this list.
|
||||
"robusta", "yelakki", "nendran", "alphonso", "banganapalli", "totapuri",
|
||||
"malgova", "sindoora", "shimla", "ooty", "kashmiri",
|
||||
}
|
||||
|
||||
# Multi-word commodities collapsed to a single token before tokenising, so the
|
||||
# individual words do not have to stand alone in the lexicon.
|
||||
_PHRASES = {
|
||||
"rock salt": "salt",
|
||||
"sea salt": "salt",
|
||||
"table salt": "salt",
|
||||
"black pepper": "pepper",
|
||||
"white pepper": "pepper",
|
||||
"bengal gram": "chana",
|
||||
"green gram": "moong",
|
||||
"black gram": "urad",
|
||||
"horse gram": "chana",
|
||||
"red chilli": "chilli",
|
||||
"bay leaf": "bay",
|
||||
"star anise": "anise",
|
||||
"sona masoori": "rice",
|
||||
"wheat flour": "atta",
|
||||
"gram flour": "besan",
|
||||
"corn flour": "flour",
|
||||
"rice flour": "flour",
|
||||
"brown sugar": "sugar",
|
||||
"palm jaggery": "jaggery",
|
||||
"cane sugar": "sugar",
|
||||
# Fresh produce. The two-word gourds collapse onto "gourd" so the whole
|
||||
# family is one lexicon entry. The leaf forms get their OWN tokens rather
|
||||
# than reusing "coriander" / "methi": those are spices, _add is last-wins,
|
||||
# and re-adding them under a greens category would silently move dhania
|
||||
# powder out of Spices & Masalas.
|
||||
"bitter gourd": "gourd",
|
||||
"bottle gourd": "gourd",
|
||||
"snake gourd": "gourd",
|
||||
"ridge gourd": "gourd",
|
||||
"ash gourd": "gourd",
|
||||
"bitter guard": "gourd", # misspellings seen in real merchant data
|
||||
"bottle ground": "gourd",
|
||||
"lady finger": "okra",
|
||||
"ladies finger": "okra",
|
||||
"spring onion": "springonion",
|
||||
"spring onions": "springonion",
|
||||
"sweet potato": "potato",
|
||||
"curry leaves": "curryleaves",
|
||||
"curry leaf": "curryleaves",
|
||||
"coriander leaves": "cilantro",
|
||||
"methi leaves": "methileaves",
|
||||
"fenugreek leaves": "methileaves",
|
||||
"french beans": "beans",
|
||||
"cluster beans": "beans",
|
||||
"green peas": "peas",
|
||||
"baby corn": "babycorn",
|
||||
"sweet corn": "sweetcorn",
|
||||
"tender coconut": "coconut",
|
||||
"dragon fruit": "dragonfruit",
|
||||
"custard apple": "apple",
|
||||
"sweet lime": "mosambi",
|
||||
"ivy gourd": "gourd",
|
||||
"broad beans": "beans",
|
||||
"cluster bean": "beans",
|
||||
"quail egg": "egg",
|
||||
"spring garlic": "garlic",
|
||||
}
|
||||
|
||||
_WORD_RE = re.compile(r"[a-z]+")
|
||||
|
||||
|
||||
# Units a pack size is actually written in. The strip below is bounded to these
|
||||
# rather than to "any letters", because [a-z]* after a number ate the NEXT WORD:
|
||||
# "24 Mantra Organic Moong Dal" became "organic moong dal", the brand was
|
||||
# destroyed, and the row was then filed as an unbranded commodity. Every brand
|
||||
# whose name begins with a number hit this. Keep the list tight - a unit added
|
||||
# here is a word that can be deleted from a product name.
|
||||
_UNITS = (
|
||||
"kg|kgs|g|gm|gms|gram|grams|mg|ml|l|ltr|ltrs|litre|litres|liter|liters"
|
||||
"|pc|pcs|piece|pieces|pack|packs|pkt|n|no|nos|x|cm|mm|inch|dozen"
|
||||
)
|
||||
|
||||
_SIZE_RE = re.compile(
|
||||
# "1kg", "500 g", "1.5 L" - a number followed by a REAL unit, optionally
|
||||
# spaced - or a bare number, which is a quantity and never a brand.
|
||||
r"\b\d+(?:[.,]\d+)?\s*(?:" + _UNITS + r")\b"
|
||||
r"|\b\d+(?:[.,]\d+)?\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
def _strip_sizes(text: str) -> str:
|
||||
"""Remove pack sizes and bare numbers - they never name a brand."""
|
||||
return _SIZE_RE.sub(" ", text)
|
||||
|
||||
|
||||
def canonical_category(name: str) -> Optional[str]:
|
||||
"""The category implied by the commodity words in `name`, if any.
|
||||
|
||||
Returns the category of the FIRST commodity term found, scanning left to
|
||||
right, because an Indian product name leads with its head noun ("Toor Dhal",
|
||||
"Sugar", "Groundnut Oil").
|
||||
"""
|
||||
for token in _tokens(name):
|
||||
category = COMMODITY_TERMS.get(token)
|
||||
if category:
|
||||
return category
|
||||
return None
|
||||
|
||||
|
||||
def _tokens(name: str) -> list:
|
||||
text = (name or "").lower()
|
||||
text = text.replace("-", " ").replace("/", " ").replace("&", " ")
|
||||
for phrase, replacement in _PHRASES.items():
|
||||
text = text.replace(phrase, replacement)
|
||||
text = _strip_sizes(text)
|
||||
return _WORD_RE.findall(text)
|
||||
|
||||
|
||||
def is_unbranded(name: str, sheet_brand: Optional[str] = None,
|
||||
brand_column_supplied: bool = False) -> bool:
|
||||
"""True when `name` names a commodity rather than a branded product.
|
||||
|
||||
`sheet_brand` is whatever the spreadsheet's own Brand column said, and
|
||||
`brand_column_supplied` whether that column existed at all. A store that
|
||||
troubled itself to include the column and left the cell empty has said
|
||||
something explicit, and is believed.
|
||||
|
||||
Deliberately conservative: a single token that is not a known commodity or
|
||||
qualifier means the row keeps its normal brand resolution. Getting this
|
||||
wrong in the permissive direction would collapse real regional brands into
|
||||
one bucket, which is far harder to undo than a staple sitting in its own
|
||||
table.
|
||||
"""
|
||||
if sheet_brand and str(sheet_brand).strip():
|
||||
return False
|
||||
if brand_column_supplied:
|
||||
return True
|
||||
|
||||
tokens = _tokens(name)
|
||||
significant = [t for t in tokens if t not in QUALIFIERS and len(t) > 1]
|
||||
if not significant:
|
||||
return False
|
||||
return all(token in COMMODITY_TERMS for token in significant)
|
||||
564
app/services/image_corroboration.py
Normal file
564
app/services/image_corroboration.py
Normal file
@@ -0,0 +1,564 @@
|
||||
"""
|
||||
Does this image URL actually depict THIS product?
|
||||
|
||||
WHY THIS FILE EXISTS
|
||||
--------------------
|
||||
`image_search.validate_image_url_live` answers "does this URL serve real image
|
||||
bytes", which is a liveness question. Nothing answered the relevance question,
|
||||
so a press photo of a person named like the brand passed every check the
|
||||
ingestion path had: it is a real image, above the byte floor, on a reputable
|
||||
host. That is how the live catalogue came to illustrate "Anil Samba Rava" with
|
||||
a photograph of the actor Anil Kapoor.
|
||||
|
||||
The logic here is ported from `scripts/repair_brand_images.py`, which already
|
||||
knew how to reject that image class - its comments name the failures it was
|
||||
written for ("Aachi Kulambu Mix" returning a press photo of a politician,
|
||||
"MTR Dosa Mix" returning an anatomy plate). It lived in a script that imports
|
||||
settings, brand_registry, produce_reference and four private `vector_store`
|
||||
symbols including `_connect`, so no ingestion path could import it. Moving the
|
||||
pure predicates here is what lets stage 6 and the catalog engine use them.
|
||||
|
||||
THE ONE SEMANTIC CHANGE MADE DURING THE MOVE
|
||||
--------------------------------------------
|
||||
The original corroborated a URL against `words + brand_tokens`:
|
||||
|
||||
return any(w in lowered for w in words + _brand_tokens(brand))
|
||||
|
||||
That works for "Aachi Kulambu Mix" precisely because *Aachi is not a human
|
||||
name*. Anil is. For product "Anil Samba Rava" under brand "Anil" the token list
|
||||
contains "anil" twice over, and `Anil_Kapoor_2019.jpg` contains "anil", so the
|
||||
gate returned True and the celebrity photo was corroborated by the very token
|
||||
that made it wrong.
|
||||
|
||||
`names_product` here requires a DISTINCTIVE token instead - the title minus the
|
||||
brand minus the pack size, which is the same set
|
||||
`catalog_engine._select_best_images` already computes to rank candidates:
|
||||
|
||||
"Anil Samba Rava" -> {samba, rava} -> rejects Anil_Kapoor_2019.jpg
|
||||
"Anil Wheat Vermicelli"-> {wheat, vermicelli}
|
||||
|
||||
Brand tokens remain, but only as an explicit fallback for titles that have no
|
||||
distinctive words at all ("Amul 1kg"), and a match found that way is reported
|
||||
as `via_brand_only` so the caller can decline to treat it as corroboration.
|
||||
|
||||
WHAT THIS MODULE MAY AND MAY NOT DECIDE
|
||||
---------------------------------------
|
||||
It decides which image is shown for a product. It never decides whether the
|
||||
PRODUCT is real. `brand_discovery._evidence_for` deliberately refuses to treat
|
||||
image naming as product evidence, because CDN filenames are frequently opaque
|
||||
hashes and absence of a naming URL is weak evidence of absence. That reasoning
|
||||
is right and this module does not disturb it: images gate images, never
|
||||
products.
|
||||
|
||||
Dependencies are stdlib plus `requests` on purpose. `catalog_engine` imports
|
||||
this module, and `repair_brand_images` already reaches back into
|
||||
`catalog_engine._select_best_images`, so a module-scope import of
|
||||
`catalog_engine` here would close a cycle.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
import sqlite3
|
||||
import threading
|
||||
import time
|
||||
from contextlib import closing
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Iterable, List, Optional, Sequence
|
||||
from urllib.parse import urlparse
|
||||
|
||||
import requests
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_BROWSER_UA = "nearle-catalogue/1.0 (product image corroboration)"
|
||||
|
||||
# "own" and "products" are the BUCKET's name, not a brand's. Left in, the token
|
||||
# "products" corroborates any URL containing /images/products/ or
|
||||
# /cdn/shop/products/ - which is every Open*Facts and every Shopify path - so
|
||||
# `names_product` waved through 40+ images that named nothing about the item.
|
||||
# That is how openbeautyfacts cosmetics photos became the stored image for
|
||||
# Banana, Orange, Papaya, Guava, Lemon and twenty more.
|
||||
BUCKET_TOKENS = frozenset({"own", "products", "product"})
|
||||
|
||||
# A trailing pack size on a product name. Used to collapse "X 100g"/"X 500g"
|
||||
# onto one search, and to keep size digits out of the distinctive token set.
|
||||
SIZE_TAIL = re.compile(
|
||||
r"\s+\d+(?:\.\d+)?\s*(?:g|gm|gms|kg|ml|l|ltr|litre|liter|pcs|pc|n|no|nos)\b\.?\s*$",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
# A bare quantity token, e.g. "500g" or "2l". Mirrors the filter in
|
||||
# catalog_engine._select_best_images so the two agree on what a size looks like.
|
||||
_SIZE_TOKEN = re.compile(r"\d+(?:kg|g|gm|gms|ml|l|ltr|pcs|n)?", re.IGNORECASE)
|
||||
|
||||
# The barcode embedded in an Open*Facts image path, e.g.
|
||||
# /images/products/890/604/215/0067/front_en.4.400.jpg
|
||||
_OFF_BARCODE = re.compile(r"/images/products/((?:\d+/)+)")
|
||||
|
||||
# Open*Facts image hosts and the API host that can identify a barcode for each.
|
||||
#
|
||||
# The original checked `if "openfoodfacts.org" not in url: return True`, so
|
||||
# every openbeautyfacts and openproductsfacts image skipped the cross-check
|
||||
# entirely - exactly the non-food case, and exactly the two sibling databases
|
||||
# `image_search.OPEN_FACTS_HOSTS` queries. A food product got the check and a
|
||||
# toothpaste did not.
|
||||
_OFF_API_HOSTS = {
|
||||
"openfoodfacts": "world.openfoodfacts.org",
|
||||
"openbeautyfacts": "world.openbeautyfacts.org",
|
||||
"openproductsfacts": "world.openproductsfacts.org",
|
||||
}
|
||||
|
||||
# Hosts whose image paths are content-hashed or barcode-keyed, so a filename
|
||||
# that fails to name the product says nothing about the photo. These are the
|
||||
# hosts catalog_engine's own comment is about when it refuses to drop
|
||||
# uncorroborated candidates.
|
||||
OPAQUE_PATH_DOMAINS = (
|
||||
"bbassets.com", "bigbasket.com", "flixcart.com", "flipkart.com",
|
||||
"media-amazon.com", "amazon.in", "amazon.com", "jiomart.com",
|
||||
"zeptonow.com", "blinkit.com", "grofers.com", "cloudinary.com",
|
||||
"shopifycdn.com", "cdn.shopify.com", "akamaized.net", "cloudfront.net",
|
||||
"openfoodfacts.org", "openbeautyfacts.org", "openproductsfacts.org",
|
||||
)
|
||||
|
||||
# Hosts whose filenames are human-authored and descriptive. A filename that
|
||||
# fails to name the product here is real evidence that the photo is of
|
||||
# something else - and this is the host family the celebrity photo came from.
|
||||
DESCRIPTIVE_FILENAME_DOMAINS = (
|
||||
"wikimedia.org", "wikipedia.org", "wikimedia.commons",
|
||||
)
|
||||
|
||||
# Two or more Capitalised words joined by underscores, optionally with a year:
|
||||
# the Wikimedia Commons house style for a photograph OF A PERSON
|
||||
# ("Anil_Kapoor_2019.jpg"). Deliberately used only to DEMOTE, never to reject -
|
||||
# "Britannia_Good_Day.jpg" has the identical shape and is a perfectly good
|
||||
# product photo, so this signal orders candidates and is not allowed to
|
||||
# eliminate one.
|
||||
_PERSON_FILENAME = re.compile(
|
||||
r"^[A-Z][a-z]+(?:_[A-Z][a-z]+)+(?:_\d{4})?[^/]*\.(?:jpg|jpeg|png|webp)$"
|
||||
)
|
||||
_PERSON_CONTEXT = re.compile(r"_at_|_in_\d{4}|portrait|headshot", re.IGNORECASE)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tokens
|
||||
# ---------------------------------------------------------------------------
|
||||
def brand_tokens(brand: str) -> List[str]:
|
||||
"""Distinctive words of a brand name, bucket words removed."""
|
||||
return [
|
||||
w for w in re.split(r"[^a-z0-9]+", (brand or "").lower())
|
||||
if len(w) > 2 and w not in BUCKET_TOKENS
|
||||
]
|
||||
|
||||
|
||||
def distinctive_tokens(product_name: str, brand: str) -> List[str]:
|
||||
"""Words that separate THIS product from its brand-mates.
|
||||
|
||||
The brand is removed on purpose: every Britannia URL contains "britannia",
|
||||
so it separates nothing - what tells Marie Gold from Good Day is
|
||||
"marie"/"gold" vs "good"/"day". Pack sizes and short filler words go for
|
||||
the same reason. This mirrors catalog_engine._select_best_images so the
|
||||
ranking and the gate can never disagree about what identifies a product.
|
||||
"""
|
||||
brand_words = {w for w in re.split(r"[^a-z0-9]+", (brand or "").lower()) if w}
|
||||
out: List[str] = []
|
||||
for word in re.split(r"[^a-z0-9]+", (product_name or "").lower()):
|
||||
if (
|
||||
len(word) > 2
|
||||
and word not in brand_words
|
||||
and word not in BUCKET_TOKENS
|
||||
and word not in out
|
||||
and not _SIZE_TOKEN.fullmatch(word)
|
||||
):
|
||||
out.append(word)
|
||||
return out
|
||||
|
||||
|
||||
def _token_in(token: str, lowered_url: str) -> bool:
|
||||
"""Does the URL carry this word, allowing for the plural on either side?
|
||||
|
||||
The catalogue names "Aachi Appalams" and "Aachi Pickles"; the photo is
|
||||
`Aachi-Appalam-100-g-1.webp`. A whole-token substring test rejected the
|
||||
correct image for the plural and then promoted an opaque Amazon URL over
|
||||
it. The singular is tried as well - only for tokens long enough that
|
||||
stripping the "s" leaves a real word ("gems" -> "gem" is fine; "kgs" never
|
||||
gets here, size tokens are removed upstream).
|
||||
"""
|
||||
if token in lowered_url:
|
||||
return True
|
||||
if len(token) > 4 and token.endswith("s") and token[:-1] in lowered_url:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def search_key(product_name: str, brand: str) -> tuple:
|
||||
"""Collapse a trailing pack size so sizes of one product share a lookup.
|
||||
|
||||
Sharing across sizes is correct rather than merely cheap: it is the same
|
||||
product in a different pack, and stage 4's size explosion produces exactly
|
||||
these rows from a single source product.
|
||||
"""
|
||||
base = SIZE_TAIL.sub("", product_name or "").strip()
|
||||
return ((brand or "").lower(), (base or product_name or "").lower())
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Corroboration
|
||||
# ---------------------------------------------------------------------------
|
||||
@dataclass(frozen=True)
|
||||
class Corroboration:
|
||||
"""Why a URL was or was not accepted as depicting this product."""
|
||||
|
||||
corroborated: bool
|
||||
reason: str
|
||||
via_brand_only: bool = False
|
||||
|
||||
|
||||
def corroborate(url: str, product_name: str, brand: str) -> Corroboration:
|
||||
"""Does `url` name this product?
|
||||
|
||||
Distinctive tokens first. Brand tokens are consulted only when the title
|
||||
has no distinctive words of its own, and a match found that way is flagged
|
||||
`via_brand_only` - it is the weakest possible signal, and for a brand that
|
||||
is also a personal name it is the signal that produced the defect this
|
||||
module exists for.
|
||||
"""
|
||||
lowered = (url or "").lower()
|
||||
if not lowered:
|
||||
return Corroboration(False, "empty url")
|
||||
|
||||
distinctive = distinctive_tokens(product_name, brand)
|
||||
if distinctive:
|
||||
hit = next((t for t in distinctive if _token_in(t, lowered)), None)
|
||||
if hit:
|
||||
return Corroboration(True, f"url names {hit!r}")
|
||||
return Corroboration(
|
||||
False,
|
||||
"url names none of " + ", ".join(repr(t) for t in distinctive[:4]),
|
||||
)
|
||||
|
||||
# No distinctive words at all (e.g. "Amul 1kg"). Fall back to the brand,
|
||||
# and say so, so the caller can decide how much that is worth.
|
||||
tokens = brand_tokens(brand)
|
||||
hit = next((t for t in tokens if t in lowered), None)
|
||||
if hit:
|
||||
return Corroboration(True, f"url names brand {hit!r}", via_brand_only=True)
|
||||
return Corroboration(False, "url names neither the product nor the brand")
|
||||
|
||||
|
||||
def names_product(url: str, product_name: str, brand: str) -> bool:
|
||||
"""Boolean form of `corroborate`, for callers that only need the verdict."""
|
||||
return corroborate(url, product_name, brand).corroborated
|
||||
|
||||
|
||||
def looks_like_person_photo(url: str) -> bool:
|
||||
"""True when the FILENAME has the shape of a photograph of a person.
|
||||
|
||||
A demotion signal only - see `_PERSON_FILENAME` for why this must never
|
||||
reject on its own.
|
||||
"""
|
||||
name = urlparse(url or "").path.rsplit("/", 1)[-1]
|
||||
if not name:
|
||||
return False
|
||||
return bool(_PERSON_FILENAME.match(name)) or bool(_PERSON_CONTEXT.search(name))
|
||||
|
||||
|
||||
def has_opaque_path(url: str) -> bool:
|
||||
lowered = (url or "").lower()
|
||||
return any(d in lowered for d in OPAQUE_PATH_DOMAINS)
|
||||
|
||||
|
||||
def has_descriptive_filename(url: str) -> bool:
|
||||
lowered = (url or "").lower()
|
||||
return any(d in lowered for d in DESCRIPTIVE_FILENAME_DOMAINS)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Open*Facts barcode cross-check
|
||||
# ---------------------------------------------------------------------------
|
||||
# The barcode is embedded in the image path, so the product is one cheap lookup
|
||||
# away. If the Open*Facts record does not mention our brand, the image is
|
||||
# somebody else's product. Nothing else can catch this: the path is opaque
|
||||
# digits, so no amount of filename matching would help. OFF matches on NAME,
|
||||
# not brand - asked for "Aachi Pickles" it returned the front-of-pack photo for
|
||||
# a French "Ducros Green Pitted Olives", which validated perfectly happily
|
||||
# because it IS a real image.
|
||||
_DB_PATH = Path("data") / "cache" / "image_corroboration.db"
|
||||
_lock = threading.Lock()
|
||||
_initialized = False
|
||||
_CACHE_TTL_SECONDS = 30 * 24 * 3600
|
||||
|
||||
# Open*Facts allows 100 product reads a minute per IP. A brand ingestion or a
|
||||
# re-gate asks about every distinct barcode it meets, and the Dabur run issued
|
||||
# about a hundred in two minutes - the tail was throttled, and a throttled
|
||||
# lookup fails open, which is how a Kellogg's honey got past the gate for a
|
||||
# moment. Pacing the calls keeps the gate answering instead of guessing.
|
||||
_OFF_MIN_INTERVAL_SECONDS = 0.65
|
||||
_OFF_LOOKUP_ATTEMPTS = 2
|
||||
_OFF_RETRY_SLEEP_SECONDS = 3.0
|
||||
_off_last_call = 0.0
|
||||
_off_pace_lock = threading.Lock()
|
||||
|
||||
|
||||
def _pace_off_lookup() -> None:
|
||||
global _off_last_call
|
||||
with _off_pace_lock:
|
||||
wait = _OFF_MIN_INTERVAL_SECONDS - (time.monotonic() - _off_last_call)
|
||||
if wait > 0:
|
||||
time.sleep(wait)
|
||||
_off_last_call = time.monotonic()
|
||||
|
||||
|
||||
def _connect() -> sqlite3.Connection:
|
||||
_DB_PATH.parent.mkdir(parents=True, exist_ok=True)
|
||||
conn = sqlite3.connect(str(_DB_PATH), timeout=10)
|
||||
conn.execute("PRAGMA journal_mode=WAL")
|
||||
return conn
|
||||
|
||||
|
||||
def _ensure_schema(conn: sqlite3.Connection) -> None:
|
||||
global _initialized
|
||||
if _initialized:
|
||||
return
|
||||
conn.execute(
|
||||
"""
|
||||
CREATE TABLE IF NOT EXISTS openfacts_brand_check (
|
||||
cache_key TEXT PRIMARY KEY,
|
||||
verdict INTEGER NOT NULL,
|
||||
created_at REAL NOT NULL
|
||||
)
|
||||
"""
|
||||
)
|
||||
conn.commit()
|
||||
_initialized = True
|
||||
|
||||
|
||||
def _cache_get(key: str) -> Optional[bool]:
|
||||
# `closing`, because sqlite3's own context manager commits the transaction
|
||||
# and leaves the connection open. Leaked connections are finalized at GC,
|
||||
# which surfaces as an unraisable exception - and pytest.ini turns warnings
|
||||
# into errors, so a leak here fails the suite from an unrelated test.
|
||||
try:
|
||||
with _lock, closing(_connect()) as conn:
|
||||
_ensure_schema(conn)
|
||||
row = conn.execute(
|
||||
"SELECT verdict, created_at FROM openfacts_brand_check WHERE cache_key = ?",
|
||||
(key,),
|
||||
).fetchone()
|
||||
except Exception as e: # noqa: BLE001 - a cache failure is a cache miss
|
||||
logger.debug("image corroboration cache read failed for %s: %s", key, e)
|
||||
return None
|
||||
if not row:
|
||||
return None
|
||||
verdict, created_at = row
|
||||
if (time.time() - created_at) > _CACHE_TTL_SECONDS:
|
||||
return None
|
||||
return bool(verdict)
|
||||
|
||||
|
||||
def _cache_set(key: str, verdict: bool) -> None:
|
||||
try:
|
||||
with _lock, closing(_connect()) as conn:
|
||||
_ensure_schema(conn)
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO openfacts_brand_check (cache_key, verdict, created_at)
|
||||
VALUES (?, ?, ?)
|
||||
ON CONFLICT(cache_key) DO UPDATE SET
|
||||
verdict = excluded.verdict,
|
||||
created_at = excluded.created_at
|
||||
""",
|
||||
(key, 1 if verdict else 0, time.time()),
|
||||
)
|
||||
conn.commit()
|
||||
except Exception as e: # noqa: BLE001 - best effort only
|
||||
logger.debug("image corroboration cache write failed for %s: %s", key, e)
|
||||
|
||||
|
||||
def _api_host_for(url: str) -> Optional[str]:
|
||||
lowered = (url or "").lower()
|
||||
for marker, host in _OFF_API_HOSTS.items():
|
||||
if marker in lowered:
|
||||
return host
|
||||
return None
|
||||
|
||||
|
||||
def openfacts_product_matches_brand(url: str, brand: str, *, timeout: int = 10) -> bool:
|
||||
"""False ONLY when Open*Facts positively says this barcode is another brand.
|
||||
|
||||
Every other outcome - not an Open*Facts URL, no barcode in the path, no
|
||||
brand tokens, a network failure, an unparseable response - returns True.
|
||||
A lookup failure must never reject a good image; that asymmetry is the
|
||||
whole point, and it is the same discipline the realtime checks follow.
|
||||
"""
|
||||
api_host = _api_host_for(url)
|
||||
if not api_host:
|
||||
return True
|
||||
match = _OFF_BARCODE.search(url or "")
|
||||
if not match:
|
||||
return True
|
||||
barcode = match.group(1).replace("/", "")
|
||||
tokens = brand_tokens(brand)
|
||||
if not tokens:
|
||||
return True
|
||||
|
||||
# "v2:" - entries written before the fail-open verdicts stopped being
|
||||
# cached are ignored rather than trusted; they age out with the TTL.
|
||||
key = f"v2:{api_host}:{barcode}:{(brand or '').lower()}"
|
||||
cached = _cache_get(key)
|
||||
if cached is not None:
|
||||
return cached
|
||||
|
||||
# ONLY A REAL ANSWER IS CACHED. The fail-open True for a lookup that did
|
||||
# not happen - a 429 or 503 from Open*Facts, a non-JSON body - used to be
|
||||
# written to the cache too, for thirty days. A Dabur ingestion issued a
|
||||
# hundred lookups in two minutes, the tail of them were throttled, and
|
||||
# Kellogg's "Miel Pops", a Toblerone and a Nature Valley bar were filed as
|
||||
# Dabur products; "Dabur Honey 1kg" then showed the Kellogg's honey. A
|
||||
# throttled lookup still fails open for THIS call, but the next call asks
|
||||
# again.
|
||||
payload = None
|
||||
for attempt in range(_OFF_LOOKUP_ATTEMPTS):
|
||||
_pace_off_lookup()
|
||||
try:
|
||||
resp = requests.get(
|
||||
f"https://{api_host}/api/v2/product/{barcode}.json",
|
||||
params={"fields": "brands,product_name"},
|
||||
timeout=timeout,
|
||||
headers={"User-Agent": _BROWSER_UA},
|
||||
)
|
||||
except Exception: # noqa: BLE001 - a lookup failure must not reject a good image
|
||||
return True
|
||||
if resp.ok:
|
||||
try:
|
||||
payload = resp.json() or {}
|
||||
except ValueError:
|
||||
return True
|
||||
break
|
||||
if resp.status_code == 429 or resp.status_code >= 500:
|
||||
# Throttled or unwell. One paced retry is cheap and turns most of
|
||||
# these into a real answer; a second failure fails open, uncached.
|
||||
time.sleep(_OFF_RETRY_SLEEP_SECONDS * (attempt + 1))
|
||||
continue
|
||||
return True
|
||||
if payload is None:
|
||||
return True
|
||||
|
||||
product = payload.get("product") or {}
|
||||
if not product and payload.get("status") == 0:
|
||||
# Open*Facts positively says: no such barcode. Nothing to compare
|
||||
# against, and asking again will not change that - cache the open verdict.
|
||||
_cache_set(key, True)
|
||||
return True
|
||||
haystack = f"{product.get('brands') or ''} {product.get('product_name') or ''}".lower()
|
||||
if not haystack.strip():
|
||||
return True
|
||||
verdict = any(t in haystack for t in tokens)
|
||||
_cache_set(key, verdict)
|
||||
return verdict
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Which candidate may become the product's primary image
|
||||
# ---------------------------------------------------------------------------
|
||||
@dataclass
|
||||
class PrimaryChoice:
|
||||
"""The outcome of choosing a product's `image_url` from its candidates.
|
||||
|
||||
`ordered` is every candidate, best first. `eligible` is the subset that
|
||||
may be SHOWN - tiers 1 and 2 - and is what the store pipeline persists as
|
||||
`image_urls`. The two differ for a reason that was learned the hard way:
|
||||
tier 3 was kept in `image_urls` "for review, never promoted", but the
|
||||
product card falls back to `image_urls[0]` whenever `image_url` is empty
|
||||
and the product modal shows the whole list as a gallery. So for "Dabur
|
||||
Honey 1kg" the gate correctly withheld the primary - every candidate was
|
||||
another company's honey - and the UI displayed those very honeys anyway.
|
||||
A candidate the gate would not promote must not be stored where the UI
|
||||
will promote it.
|
||||
"""
|
||||
|
||||
primary: Optional[str]
|
||||
ordered: List[str]
|
||||
reason: str
|
||||
eligible: List[str] = field(default_factory=list)
|
||||
rejected: List[str] = field(default_factory=list)
|
||||
|
||||
|
||||
def choose_primary(
|
||||
urls: Sequence[str],
|
||||
product_name: str,
|
||||
brand: str,
|
||||
*,
|
||||
check_openfacts: bool = True,
|
||||
) -> PrimaryChoice:
|
||||
"""Pick the URL that may become `image_url`, and order the rest.
|
||||
|
||||
Three tiers, because the two existing opinions about failing closed are
|
||||
both right and the disagreement is domain-scoped:
|
||||
|
||||
1. The URL names the product. Eligible.
|
||||
2. The URL cannot name anything (a hashed retailer CDN path, an
|
||||
Open*Facts barcode path) - and for Open*Facts, the barcode's own
|
||||
record agrees about the brand. Eligible: this is the case
|
||||
catalog_engine's comment protects, where dropping uncorroborated
|
||||
candidates would leave real products with no image at all.
|
||||
3. The URL could have named the product and did not - a human-authored
|
||||
Commons filename, say - or an Open*Facts photo whose barcode belongs
|
||||
to another brand. Returned in `ordered` and `rejected`, never in
|
||||
`eligible`, and the store pipeline does not persist it (see
|
||||
PrimaryChoice for why "kept for review" was not safe).
|
||||
|
||||
Nothing eligible means `primary is None`. A blank image renders as the
|
||||
brand monogram, which is honest; another company's product is not, and it
|
||||
stays invisible until somebody recognises the photo.
|
||||
"""
|
||||
tier1: List[str] = []
|
||||
tier2: List[str] = []
|
||||
tier3: List[str] = []
|
||||
|
||||
for url in urls:
|
||||
if not url or not str(url).startswith("http"):
|
||||
continue
|
||||
verdict = corroborate(url, product_name, brand)
|
||||
if verdict.corroborated and not verdict.via_brand_only:
|
||||
tier1.append(url)
|
||||
elif verdict.via_brand_only and not looks_like_person_photo(url):
|
||||
# The title has NO distinctive words of its own - "Godrej 50ml",
|
||||
# "Lion Dates 100g", "Amul 1kg". There is no product identity to
|
||||
# match on, so a brand-token match is the best signal that exists
|
||||
# and withholding the image gains nothing: measured over the seed
|
||||
# catalogues, treating these as ineligible accounted for 14 of 45
|
||||
# withheld primaries, every one of them a brand's own product page.
|
||||
#
|
||||
# Person-shaped filenames are the exception, because a brand token
|
||||
# matching a personal name is the exact defect this module exists
|
||||
# for and a title with no distinctive words cannot contradict it.
|
||||
tier2.append(url)
|
||||
elif has_opaque_path(url) and not has_descriptive_filename(url):
|
||||
if check_openfacts and not openfacts_product_matches_brand(url, brand):
|
||||
tier3.append(url)
|
||||
else:
|
||||
tier2.append(url)
|
||||
else:
|
||||
tier3.append(url)
|
||||
|
||||
# Person-shaped filenames sink within their tier. They are never dropped,
|
||||
# so a product whose only images look like this still keeps them in
|
||||
# `image_urls` for a human to review.
|
||||
tier3.sort(key=looks_like_person_photo)
|
||||
|
||||
ordered = tier1 + tier2 + tier3
|
||||
eligible = tier1 + tier2
|
||||
if tier1:
|
||||
return PrimaryChoice(tier1[0], ordered, "url names the product",
|
||||
eligible=eligible, rejected=tier3)
|
||||
if tier2:
|
||||
return PrimaryChoice(tier2[0], ordered, "opaque path on a known image host",
|
||||
eligible=eligible, rejected=tier3)
|
||||
return PrimaryChoice(
|
||||
None,
|
||||
ordered,
|
||||
"no candidate names this product; primary withheld rather than guessed",
|
||||
eligible=eligible, rejected=tier3,
|
||||
)
|
||||
334
app/services/image_embedder.py
Normal file
334
app/services/image_embedder.py
Normal file
@@ -0,0 +1,334 @@
|
||||
"""The image embedding model behind `img_vector`: MobileNetV3-Small, TFLite.
|
||||
|
||||
WHAT IT PRODUCES
|
||||
----------------
|
||||
One 1024-float vector per image, L2-normalised, from
|
||||
`mobilenet_v3_small_embedder.tflite` (input [1, 224, 224, 3] float32, output
|
||||
[1, 1024]). It is the same model, runtime and post-processing a colleague uses
|
||||
for their product images, which is the whole point: two vectors from the same
|
||||
photo must agree, and `<=>` (cosine distance) between our rows and theirs must
|
||||
mean something.
|
||||
|
||||
THIS FILE OWNS TWO THINGS AND NOTHING ELSE
|
||||
------------------------------------------
|
||||
1. `preprocess()` - bytes to the input tensor. It is the ONLY place the
|
||||
resize / crop / scaling decisions live, because those decisions are what
|
||||
make the vectors comparable. It is a port of the Nearle Flutter app's
|
||||
OpenCV pipeline (centre crop, INTER_AREA to 224, RGB, 0..1) and uses
|
||||
OpenCV itself so the two agree to a few decimals.
|
||||
2. The interpreter - one per process, created lazily on first use, and every
|
||||
call into it serialised by one lock. A TFLite interpreter is not
|
||||
thread-safe, and this process has exactly two callers: the single
|
||||
image-vector worker thread and the backfill script's download pool.
|
||||
|
||||
Which column gets the vector, when, and for which rows is
|
||||
app/services/image_vector.py's business, not this file's.
|
||||
|
||||
FAILURE IS SILENT BY DESIGN
|
||||
---------------------------
|
||||
The runtime (`ai-edge-litert`) is imported here, not at module import, and
|
||||
the model file is opened here, not at boot. If either is missing, `available()`
|
||||
returns False after ONE warning, `embed()` returns None, and the catalog keeps
|
||||
writing rows with a NULL img_vector. A missing wheel or a model left out of an
|
||||
image must never turn into a failed product write.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import logging
|
||||
import threading
|
||||
import warnings
|
||||
from typing import Any, List, Optional
|
||||
|
||||
import numpy as np
|
||||
|
||||
from app.infrastructure.settings import (
|
||||
IMAGE_EMBED_MODEL_PATH,
|
||||
IMAGE_EMBED_NUM_THREADS,
|
||||
IMAGE_VECTOR_MAX_PIXELS,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
EMBED_DIM = 1024
|
||||
IMAGE_EMBED_INPUT_SIZE = 224
|
||||
_INPUT_SHAPE = (1, IMAGE_EMBED_INPUT_SIZE, IMAGE_EMBED_INPUT_SIZE, 3)
|
||||
_OUTPUT_SHAPE = (1, EMBED_DIM)
|
||||
|
||||
_lock = threading.Lock()
|
||||
_interpreter: Any = None
|
||||
_input_index: Optional[int] = None
|
||||
_output_index: Optional[int] = None
|
||||
_disabled_reason: Optional[str] = None
|
||||
_warned = False
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# bytes -> input tensor
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _flatten_alpha_to_bgr(image_bytes: bytes) -> Optional[np.ndarray]:
|
||||
"""Pillow decode for images WITH transparency: EXIF applied, composited on
|
||||
white, returned as uint8 BGR so the OpenCV steps below see exactly what
|
||||
`cv2.imread` would have produced from an opaque file."""
|
||||
from PIL import Image, ImageOps
|
||||
|
||||
im = Image.open(io.BytesIO(image_bytes))
|
||||
im = ImageOps.exif_transpose(im) or im
|
||||
rgba = im.convert("RGBA")
|
||||
white = Image.new("RGBA", rgba.size, (255, 255, 255, 255))
|
||||
rgb = Image.alpha_composite(white, rgba).convert("RGB")
|
||||
return np.ascontiguousarray(np.asarray(rgb, dtype=np.uint8)[:, :, ::-1])
|
||||
|
||||
|
||||
def decode_bgr(image_bytes: bytes) -> Optional[np.ndarray]:
|
||||
"""`image_bytes` as a full-resolution uint8 BGR array, or None.
|
||||
|
||||
Step 1 of `preprocess()`, on its own because the OCR service
|
||||
(app/services/ocr_service.py) needs the same decode - pixel-bomb guard,
|
||||
EXIF orientation, transparency flattened onto white - but the WHOLE frame:
|
||||
the 224px centre crop that follows here would throw away the label.
|
||||
|
||||
Never raises: an undecodable or oversized file is None (see preprocess).
|
||||
"""
|
||||
if not image_bytes:
|
||||
return None
|
||||
try:
|
||||
import cv2
|
||||
from PIL import Image
|
||||
|
||||
# Header only: the pixel-bomb guard and the transparency check both
|
||||
# come from the file's metadata, no decode yet.
|
||||
probe = Image.open(io.BytesIO(image_bytes))
|
||||
if probe.width * probe.height > IMAGE_VECTOR_MAX_PIXELS:
|
||||
logger.debug("image rejected: %dx%d exceeds pixel cap", probe.width, probe.height)
|
||||
return None
|
||||
has_alpha = "A" in probe.getbands() or "transparency" in probe.info
|
||||
|
||||
if has_alpha:
|
||||
bgr = _flatten_alpha_to_bgr(image_bytes)
|
||||
else:
|
||||
bgr = cv2.imdecode(np.frombuffer(image_bytes, dtype=np.uint8), cv2.IMREAD_COLOR)
|
||||
if bgr is None or bgr.ndim != 3 or bgr.shape[2] != 3:
|
||||
logger.debug("image rejected: OpenCV could not decode it to BGR")
|
||||
return None
|
||||
return bgr
|
||||
except Exception as exc: # noqa: BLE001 - see preprocess()
|
||||
logger.debug("image undecodable: %s", exc)
|
||||
return None
|
||||
|
||||
|
||||
def preprocess(image_bytes: bytes) -> Optional[np.ndarray]:
|
||||
"""Decode `image_bytes` into the model's input tensor, or None.
|
||||
|
||||
A line-for-line port of the Nearle Flutter app's preprocessing
|
||||
(core/services/image_embed/image_embedder.dart, opencv_dart) and of the
|
||||
colleague's Python reference for it. The steps, in their order:
|
||||
|
||||
1. read as BGR cv2.imdecode(..., IMREAD_COLOR)
|
||||
2. centre square crop side = min(w, h); x = (w - side) // 2; y = (h - side) // 2
|
||||
3. resize to 224x224 cv2.resize(..., interpolation=INTER_AREA)
|
||||
4. BGR -> RGB cv2.cvtColor(..., COLOR_BGR2RGB)
|
||||
5. uint8 -> float 0..1 astype(float32) / 255.0 (ONLY that - no mean,
|
||||
no std, no -1..1; the model rescales internally)
|
||||
6. batch dimension [1, 224, 224, 3], HWC
|
||||
|
||||
OpenCV is used for the decode and the resize rather than Pillow on
|
||||
purpose: INTER_AREA and Pillow's BOX filter are not the same filter, and
|
||||
the two JPEG decoders differ by a pixel here and there. Matching the app
|
||||
to 3-4 decimals is the requirement, so the app's library is the tool.
|
||||
|
||||
The two "backend-only extras" from the same spec: an image WITH
|
||||
transparency is flattened onto white before step 1 (IMREAD_COLOR would
|
||||
turn the transparent area black), and a greyscale image comes out of
|
||||
IMREAD_COLOR as three channels already. cv2.imdecode applies EXIF
|
||||
orientation like cv2.imread does, which is what the phone camera path
|
||||
relies on.
|
||||
|
||||
Never raises - pytest runs warnings as errors, and a
|
||||
DecompressionBombWarning is one of the things the broad except absorbs.
|
||||
"""
|
||||
bgr = decode_bgr(image_bytes) # 1
|
||||
if bgr is None:
|
||||
return None
|
||||
try:
|
||||
import cv2
|
||||
|
||||
h, w = bgr.shape[:2] # 2
|
||||
side = min(w, h)
|
||||
x, y = (w - side) // 2, (h - side) // 2
|
||||
sq = bgr[y:y + side, x:x + side]
|
||||
|
||||
size = (IMAGE_EMBED_INPUT_SIZE, IMAGE_EMBED_INPUT_SIZE)
|
||||
resized = cv2.resize(sq, size, interpolation=cv2.INTER_AREA) # 3
|
||||
rgb = cv2.cvtColor(resized, cv2.COLOR_BGR2RGB) # 4
|
||||
arr = (rgb.astype(np.float32) / 255.0)[None] # 5, 6
|
||||
if arr.shape != _INPUT_SHAPE:
|
||||
logger.debug("image rejected: tensor shape %s", arr.shape)
|
||||
return None
|
||||
return np.ascontiguousarray(arr)
|
||||
except Exception as exc: # noqa: BLE001 - see docstring
|
||||
logger.debug("image undecodable: %s", exc)
|
||||
return None
|
||||
|
||||
|
||||
def l2_normalize(vec: np.ndarray) -> Optional[np.ndarray]:
|
||||
"""Unit-length copy of `vec`, or None for a (near-)zero vector."""
|
||||
arr = np.asarray(vec, dtype=np.float32).reshape(-1)
|
||||
norm = float(np.linalg.norm(arr))
|
||||
if not np.isfinite(norm) or norm < 1e-12:
|
||||
return None
|
||||
return arr / norm
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# the interpreter
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _disable(reason: str) -> None:
|
||||
"""Record why the embedder is off and say so once. Caller holds _lock."""
|
||||
global _disabled_reason, _interpreter, _warned
|
||||
_disabled_reason = reason
|
||||
_interpreter = None
|
||||
if not _warned:
|
||||
_warned = True
|
||||
logger.warning("image embedder disabled: %s (img_vector stays NULL until fixed)", reason)
|
||||
|
||||
|
||||
def _load() -> None:
|
||||
"""Create and validate the interpreter. Caller holds _lock. Never raises."""
|
||||
global _interpreter, _input_index, _output_index
|
||||
path = IMAGE_EMBED_MODEL_PATH
|
||||
if not path.is_file():
|
||||
_disable(f"model file not found at {path}")
|
||||
return
|
||||
try:
|
||||
with warnings.catch_warnings():
|
||||
# The wheel's import-time deprecation chatter would become a hard
|
||||
# error under pytest's filterwarnings=error.
|
||||
warnings.simplefilter("ignore")
|
||||
from ai_edge_litert.interpreter import Interpreter
|
||||
except Exception as exc: # noqa: BLE001 - ImportError or a broken wheel
|
||||
_disable(f"ai-edge-litert is not importable ({exc})")
|
||||
return
|
||||
try:
|
||||
interp = Interpreter(model_path=str(path), num_threads=max(1, IMAGE_EMBED_NUM_THREADS))
|
||||
interp.allocate_tensors()
|
||||
inputs = interp.get_input_details()
|
||||
outputs = interp.get_output_details()
|
||||
except Exception as exc: # noqa: BLE001 - unreadable / malformed model
|
||||
_disable(f"could not load {path.name} ({exc})")
|
||||
return
|
||||
|
||||
if len(inputs) != 1 or tuple(int(d) for d in inputs[0]["shape"]) != _INPUT_SHAPE \
|
||||
or np.dtype(inputs[0]["dtype"]) != np.float32:
|
||||
got = [(list(i["shape"]), np.dtype(i["dtype"]).name) for i in inputs]
|
||||
_disable(f"{path.name} input is {got}, expected [1,224,224,3] float32")
|
||||
return
|
||||
if len(outputs) != 1 or tuple(int(d) for d in outputs[0]["shape"]) != _OUTPUT_SHAPE:
|
||||
got = [list(o["shape"]) for o in outputs]
|
||||
_disable(f"{path.name} output is {got}, expected [1,1024]")
|
||||
return
|
||||
|
||||
_interpreter = interp
|
||||
_input_index = int(inputs[0]["index"])
|
||||
_output_index = int(outputs[0]["index"])
|
||||
logger.info("image embedder ready: %s (%d thread(s))", path.name, IMAGE_EMBED_NUM_THREADS)
|
||||
|
||||
|
||||
def _ensure_loaded() -> bool:
|
||||
"""Caller holds _lock."""
|
||||
if _interpreter is None and _disabled_reason is None:
|
||||
_load()
|
||||
return _interpreter is not None
|
||||
|
||||
|
||||
def available() -> bool:
|
||||
"""True when the model can be used. Loads it on first call."""
|
||||
with _lock:
|
||||
return _ensure_loaded()
|
||||
|
||||
|
||||
def status() -> dict:
|
||||
"""Diagnostics for /api/health - NEVER loads the model.
|
||||
|
||||
Exists because the failure this module is built to absorb is silent by
|
||||
design: a container without the .tflite writes NULL vectors and says so
|
||||
once, in a log line nobody reads. This puts the same facts on the health
|
||||
endpoint, where "why are the vectors NULL after the deploy?" can be
|
||||
answered with one curl.
|
||||
"""
|
||||
path = IMAGE_EMBED_MODEL_PATH
|
||||
try:
|
||||
import importlib.util
|
||||
runtime = importlib.util.find_spec("ai_edge_litert") is not None
|
||||
except Exception: # noqa: BLE001 - a broken finder counts as "not importable"
|
||||
runtime = False
|
||||
with _lock:
|
||||
if _interpreter is not None:
|
||||
state = "ready"
|
||||
elif _disabled_reason:
|
||||
state = f"disabled: {_disabled_reason}"
|
||||
else:
|
||||
state = "not loaded yet (loads on first write)"
|
||||
return {
|
||||
"model_path": str(path),
|
||||
"model_present": path.is_file(),
|
||||
"runtime_importable": runtime,
|
||||
"state": state,
|
||||
}
|
||||
|
||||
|
||||
def describe() -> str:
|
||||
"""One line for script banners: where the model is and whether it works."""
|
||||
with _lock:
|
||||
ok = _ensure_loaded()
|
||||
if ok:
|
||||
return f"ready - {IMAGE_EMBED_MODEL_PATH} ({IMAGE_EMBED_NUM_THREADS} thread(s), {EMBED_DIM}-d)"
|
||||
return f"disabled - {_disabled_reason}"
|
||||
|
||||
|
||||
def embed(preprocessed: np.ndarray) -> Optional[np.ndarray]:
|
||||
"""Raw model output for one preprocessed tensor, as (1024,) float32, or None.
|
||||
|
||||
Serialised on the module lock: the interpreter's tensors are shared
|
||||
state, and two threads calling set_tensor/invoke at once corrupt both.
|
||||
"""
|
||||
if preprocessed is None or tuple(preprocessed.shape) != _INPUT_SHAPE:
|
||||
return None
|
||||
with _lock:
|
||||
if not _ensure_loaded():
|
||||
return None
|
||||
try:
|
||||
_interpreter.set_tensor(_input_index, np.asarray(preprocessed, dtype=np.float32))
|
||||
_interpreter.invoke()
|
||||
out = _interpreter.get_tensor(_output_index)
|
||||
return np.array(out, dtype=np.float32, copy=True).reshape(-1)[:EMBED_DIM]
|
||||
except Exception as exc: # noqa: BLE001 - one bad tensor must not stop the worker
|
||||
logger.debug("inference failed: %s", exc)
|
||||
return None
|
||||
|
||||
|
||||
def embedding_for_bytes(image_bytes: bytes) -> Optional[List[float]]:
|
||||
"""bytes -> preprocess -> embed -> L2-normalise -> 1024 floats, or None."""
|
||||
tensor = preprocess(image_bytes)
|
||||
if tensor is None:
|
||||
return None
|
||||
raw = embed(tensor)
|
||||
if raw is None or raw.shape != (EMBED_DIM,):
|
||||
return None
|
||||
unit = l2_normalize(raw)
|
||||
if unit is None:
|
||||
return None
|
||||
return [float(v) for v in unit]
|
||||
|
||||
|
||||
def _reset() -> None:
|
||||
"""Tests only: forget the interpreter and any recorded failure."""
|
||||
global _interpreter, _input_index, _output_index, _disabled_reason, _warned
|
||||
with _lock:
|
||||
_interpreter = None
|
||||
_input_index = None
|
||||
_output_index = None
|
||||
_disabled_reason = None
|
||||
_warned = False
|
||||
455
app/services/image_match.py
Normal file
455
app/services/image_match.py
Normal file
@@ -0,0 +1,455 @@
|
||||
"""Find catalog products from a phone photo: img_vector nearest-neighbour + label text.
|
||||
|
||||
(Named image_match, not image_search: app/services/image_search.py is the
|
||||
image DISCOVERY module that finds photos for products. This is the reverse.)
|
||||
|
||||
WHAT COMES IN
|
||||
-------------
|
||||
The Nearle app photographs a pack, crops it, embeds it on-device with the same
|
||||
MobileNetV3-Small model and preprocessing that filled `img_vector`
|
||||
(app/services/image_embedder.py), L2-normalises, and sends the 1024 floats -
|
||||
plus whatever OCR read off the label. Alternatively a client sends the photo
|
||||
and the API embeds it. Either way this module gets a unit vector and maybe
|
||||
some text.
|
||||
|
||||
HOW A MATCH IS SCORED
|
||||
---------------------
|
||||
`score = 1 - (img_vector <=> q)`: cosine similarity, since both sides are unit
|
||||
length. This is the app team's convention and is NOT
|
||||
`RetrievedProduct.similarity` (`1 - distance/2`, rag_service.py), which is
|
||||
the text search's. A photo of a pack against the catalog's render of it
|
||||
scored 0.63 in the feasibility test; the app doc's "0.7 = same product" is a
|
||||
phone-vs-phone rule of thumb, so the server's default floor is 0.
|
||||
|
||||
WHY TEXT IS PART OF IT
|
||||
----------------------
|
||||
Two reasons, both measured on the catalog:
|
||||
|
||||
* Every pack size of one product usually shares one photo, so "Marie Gold
|
||||
89g / 300g / 1kg" tie to the last decimal. The label text is the only thing
|
||||
that can pick the 300g row: size tokens are normalised ("300 g", "300gm",
|
||||
"300G" -> "300g") and weighted above plain words.
|
||||
* The brand name, when the OCR caught it, turns a 57-table scan into one
|
||||
table via `query_intent.extract_brand_mention`. If that scope finds nothing
|
||||
above `min_score` the search is retried unscoped and says so
|
||||
(`scope_fallback`), because OCR misreads happen. An EXPLICIT `brand` never
|
||||
falls back - the caller asked for a filter.
|
||||
|
||||
The ranking is deterministic on purpose (`rank_key`): tied rows are ordered
|
||||
by text overlap, then name, then image_id, so the same request always
|
||||
returns the same order and the tests can pin it.
|
||||
|
||||
Read-only: nothing here writes, and the pipeline stages are untouched.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import binascii
|
||||
import logging
|
||||
import math
|
||||
import re
|
||||
import struct
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any, Dict, List, Optional, Sequence, Set, Tuple
|
||||
|
||||
from app.infrastructure.settings import (
|
||||
IMAGE_IDENTIFY_MIN_IMAGE_SCORE,
|
||||
IMAGE_SEARCH_DEFAULT_MIN_SCORE,
|
||||
IMAGE_SEARCH_DEFAULT_TOP_K,
|
||||
IMAGE_SEARCH_MAX_FETCH_K,
|
||||
IMAGE_SEARCH_MAX_TOP_K,
|
||||
IMAGE_SEARCH_MIN_MARGIN,
|
||||
)
|
||||
from app.services.image_embedder import EMBED_DIM
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# The floor pgvector's hnsw needs to return LIMIT rows: ef_search < LIMIT
|
||||
# silently truncates the candidate list.
|
||||
_MIN_EF_SEARCH = 40
|
||||
|
||||
_SIZE_RE = re.compile(r"(\d+(?:\.\d+)?)\s*(kg|gms|gm|g|ml|ltr|litre|l|pcs|pc|n)\b", re.I)
|
||||
_UNIT_ALIAS = {"gm": "g", "gms": "g", "ltr": "l", "litre": "l", "pc": "pcs", "n": "pcs"}
|
||||
_WORD_RE = re.compile(r"[a-z0-9]+")
|
||||
_STOP = {
|
||||
"the", "a", "an", "of", "and", "with", "for", "pack", "new", "net", "wt",
|
||||
"weight", "mrp", "rs", "inr", "in", "by", "per", "no", "nos",
|
||||
}
|
||||
_SIZE_WEIGHT = 3.0
|
||||
_WORD_WEIGHT = 1.0
|
||||
|
||||
|
||||
class InvalidVectorError(ValueError):
|
||||
"""The query vector cannot be searched with: wrong length, NaN, or zero."""
|
||||
|
||||
|
||||
@dataclass
|
||||
class ImageSearchResult:
|
||||
rows: List[Dict[str, Any]] = field(default_factory=list) # each carries "score" and "text_overlap"
|
||||
detected_brand: Optional[str] = None # explicit brand, else the OCR-derived one
|
||||
scoped_to_brand: bool = False # the rows came from one brand table
|
||||
scope_fallback: bool = False # OCR scope was empty; retried unscoped
|
||||
min_score: float = 0.0
|
||||
query_text: Optional[str] = None
|
||||
top_k: int = 0
|
||||
match_confidence: str = "none" # "confirmed" | "low" | "none"
|
||||
margin: Optional[float] = None # top score minus the best different photo's
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# the query vector
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def normalise_vector(vector: Sequence[float]) -> List[float]:
|
||||
"""`vector` as a unit-length list of EMBED_DIM floats, or InvalidVectorError.
|
||||
|
||||
Already-unit input (|norm - 1| < 1e-3) is returned as-is so the app's own
|
||||
normalisation is not disturbed by float rounding; anything else is
|
||||
rescaled, because a client that forgot to normalise should still get the
|
||||
right neighbours rather than distances scaled by its norm.
|
||||
"""
|
||||
try:
|
||||
values = [float(v) for v in vector]
|
||||
except (TypeError, ValueError) as exc:
|
||||
raise InvalidVectorError(f"vector must be a list of numbers: {exc}") from None
|
||||
if len(values) != EMBED_DIM:
|
||||
raise InvalidVectorError(f"vector must have {EMBED_DIM} values, got {len(values)}")
|
||||
if not all(math.isfinite(v) for v in values):
|
||||
raise InvalidVectorError("vector contains NaN or infinite values")
|
||||
norm = math.sqrt(sum(v * v for v in values))
|
||||
if norm < 1e-6:
|
||||
raise InvalidVectorError("vector is all zeros")
|
||||
if abs(norm - 1.0) < 1e-3:
|
||||
return values
|
||||
return [v / norm for v in values]
|
||||
|
||||
|
||||
def encode_vector_b64(vector: Sequence[float]) -> str:
|
||||
"""The compact query-string form: urlsafe base64 of EMBED_DIM little-endian
|
||||
float32 values, no padding. 5,464 characters for 1024 floats, against
|
||||
8-10 KB for comma-separated decimals - the difference between fitting
|
||||
every proxy's request-line limit and not. Client-side equivalents are in
|
||||
docs/IMAGE_SEARCH_API.md."""
|
||||
values = [float(v) for v in vector]
|
||||
if len(values) != EMBED_DIM:
|
||||
raise InvalidVectorError(f"vector must have {EMBED_DIM} values, got {len(values)}")
|
||||
packed = struct.pack(f"<{EMBED_DIM}f", *values)
|
||||
return base64.urlsafe_b64encode(packed).decode("ascii").rstrip("=")
|
||||
|
||||
|
||||
def parse_vector_param(raw: str) -> List[float]:
|
||||
"""The `vector` query parameter of GET /search/image-vector, as floats.
|
||||
|
||||
Two encodings, told apart by the presence of a comma:
|
||||
|
||||
* comma-separated decimals - what a log line or a print() gives you;
|
||||
surrounding brackets, whitespace and newlines are ignored;
|
||||
* base64 (urlsafe or standard alphabet, padding optional) of exactly
|
||||
EMBED_DIM little-endian float32 - what `encode_vector_b64` produces.
|
||||
|
||||
Only the shape is checked here; length, NaN and zero-norm are
|
||||
`normalise_vector`'s job, so every route reports them the same way.
|
||||
"""
|
||||
text = (raw or "").strip()
|
||||
if text[:1] == "[" and text[-1:] == "]":
|
||||
text = text[1:-1]
|
||||
if not text.strip():
|
||||
raise InvalidVectorError("vector is empty")
|
||||
|
||||
if "," in text:
|
||||
values: List[float] = []
|
||||
for position, piece in enumerate(text.split(","), start=1):
|
||||
piece = piece.strip()
|
||||
if not piece:
|
||||
continue
|
||||
try:
|
||||
values.append(float(piece))
|
||||
except ValueError:
|
||||
raise InvalidVectorError(f"value {position} ({piece[:20]!r}) is not a number") from None
|
||||
return values
|
||||
|
||||
compact = "".join(text.split())
|
||||
padded = compact + "=" * (-len(compact) % 4)
|
||||
try:
|
||||
packed = base64.urlsafe_b64decode(padded.replace("+", "-").replace("/", "_"))
|
||||
except (ValueError, binascii.Error):
|
||||
raise InvalidVectorError(
|
||||
f"vector must be {EMBED_DIM} comma-separated numbers or base64 of {EMBED_DIM} float32 values"
|
||||
) from None
|
||||
if len(packed) != EMBED_DIM * 4:
|
||||
raise InvalidVectorError(
|
||||
f"base64 vector decodes to {len(packed)} bytes, expected {EMBED_DIM * 4} "
|
||||
f"({EMBED_DIM} little-endian float32)"
|
||||
)
|
||||
return list(struct.unpack(f"<{EMBED_DIM}f", packed))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# label text
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def tokens(text: Optional[str]) -> Tuple[Set[str], Set[str]]:
|
||||
"""(words, sizes) from label text.
|
||||
|
||||
Sizes are normalised so the OCR's "300 g", a sheet's "300gm" and a
|
||||
product name's "300G" all become "300g". Words are lowercase, alnum,
|
||||
at least two characters, minus stop-words and minus anything that was
|
||||
part of a size token (so "300" and "g" do not also count as words).
|
||||
"""
|
||||
if not text:
|
||||
return set(), set()
|
||||
lowered = text.lower()
|
||||
sizes: Set[str] = set()
|
||||
for num, unit in _SIZE_RE.findall(lowered):
|
||||
unit = _UNIT_ALIAS.get(unit, unit)
|
||||
num = num.rstrip("0").rstrip(".") if "." in num else num
|
||||
sizes.add(f"{num}{unit}")
|
||||
without_sizes = _SIZE_RE.sub(" ", lowered)
|
||||
words = {w for w in _WORD_RE.findall(without_sizes) if len(w) >= 2 and w not in _STOP}
|
||||
return words, sizes
|
||||
|
||||
|
||||
def _row_text(row: Dict[str, Any]) -> str:
|
||||
parts = [str(row.get("product_name") or ""), str(row.get("title") or "")]
|
||||
variants = row.get("size_variants") or []
|
||||
if isinstance(variants, (list, tuple)):
|
||||
parts.extend(str(v) for v in variants if v)
|
||||
return " ".join(parts)
|
||||
|
||||
|
||||
def text_overlap(words: Set[str], sizes: Set[str], row: Dict[str, Any]) -> float:
|
||||
"""How much of the label text this row's name accounts for.
|
||||
|
||||
3.0 per shared size token, 1.0 per shared word. Sizes weigh more because
|
||||
they are what separates the pack sizes of one product, which is the tie
|
||||
this exists to break; brand and product words match every sibling alike.
|
||||
"""
|
||||
if not words and not sizes:
|
||||
return 0.0
|
||||
row_words, row_sizes = tokens(_row_text(row))
|
||||
return _SIZE_WEIGHT * len(sizes & row_sizes) + _WORD_WEIGHT * len(words & row_words)
|
||||
|
||||
|
||||
def size_in_name(sizes: Set[str], row: Dict[str, Any]) -> bool:
|
||||
"""Whether a pack size the label printed is in the row's OWN name.
|
||||
|
||||
`text_overlap` also counts `size_variants`, and on some products every
|
||||
pack-size row carries the SAME list (all Amul Cream rows say
|
||||
["90g", "1kg"]), so a "1kg" label ties every sibling and the tie fell to
|
||||
alphabetical order - "Amul Cream 125 ml" was shown, confirmed, for a
|
||||
photo of the 1kg pack. The name is per row; this breaks that tie.
|
||||
"""
|
||||
if not sizes:
|
||||
return False
|
||||
name_sizes = tokens(" ".join([str(row.get("product_name") or ""), str(row.get("title") or "")]))[1]
|
||||
return bool(sizes & name_sizes)
|
||||
|
||||
|
||||
def rank_key(row: Dict[str, Any]) -> tuple:
|
||||
"""Best first. Score rounded to 3 dp so siblings sharing a photo tie."""
|
||||
return (
|
||||
-round(float(row.get("score", 0.0)), 3),
|
||||
-float(row.get("text_overlap", 0.0)),
|
||||
-int(bool(row.get("size_in_name", False))),
|
||||
str(row.get("product_name") or row.get("title") or "").lower(),
|
||||
str(row.get("image_id") or ""),
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# confidence
|
||||
# ---------------------------------------------------------------------------
|
||||
# Pack sizes that share one catalog photo score identically (to pgvector's
|
||||
# float precision); anything further apart than this is a different picture.
|
||||
_SAME_PHOTO_EPS = 5e-4
|
||||
|
||||
CONFIRMED = "confirmed"
|
||||
LOW = "low"
|
||||
NONE = "none"
|
||||
|
||||
|
||||
def image_margin(rows: Sequence[Dict[str, Any]]) -> Optional[float]:
|
||||
"""How far the best row leads the best row with a DIFFERENT photo.
|
||||
|
||||
`rows` are best first by score. Siblings sharing the winner's photo are
|
||||
skipped: they are the same product in another size, which the label text
|
||||
separates, not a rival. None when nothing else is in the list - no rival
|
||||
was seen, which is not the same as a clear lead, so the caller treats it
|
||||
as unconfirmable only when the score itself is low.
|
||||
"""
|
||||
if not rows:
|
||||
return None
|
||||
try:
|
||||
top = float(rows[0].get("score", 0.0))
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
for row in rows[1:]:
|
||||
try:
|
||||
score = float(row.get("score", 0.0))
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
if top - score > _SAME_PHOTO_EPS:
|
||||
return top - score
|
||||
return None
|
||||
|
||||
|
||||
def match_confidence(
|
||||
rows: Sequence[Dict[str, Any]],
|
||||
min_image_score: Optional[float] = None,
|
||||
min_margin: Optional[float] = None,
|
||||
) -> Tuple[str, Optional[float]]:
|
||||
"""("confirmed" | "low" | "none", margin) for image rows, best first.
|
||||
|
||||
Confirmed needs BOTH a score at the identify floor and a lead over the
|
||||
next different photo. A photo of a card on a phone screen scores ~0.5
|
||||
against its own product, and a stranger can sit within a few hundredths
|
||||
of it: shown as the answer, that is how "Dairy Milk Lickables" came back
|
||||
as someone else's product. A client should offer the list, not pick one,
|
||||
when this is not "confirmed". The thresholds default to this module's
|
||||
settings, read at call time so scripts/eval_identify.py can vary them.
|
||||
"""
|
||||
if min_image_score is None:
|
||||
min_image_score = IMAGE_IDENTIFY_MIN_IMAGE_SCORE
|
||||
if min_margin is None:
|
||||
min_margin = IMAGE_SEARCH_MIN_MARGIN
|
||||
if not rows:
|
||||
return NONE, None
|
||||
margin = image_margin(rows)
|
||||
try:
|
||||
top = float(rows[0].get("score", 0.0))
|
||||
except (TypeError, ValueError):
|
||||
return LOW, margin
|
||||
if top < min_image_score:
|
||||
return LOW, margin
|
||||
if margin is not None and margin < min_margin:
|
||||
return LOW, margin
|
||||
return CONFIRMED, margin
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# the search
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _candidates(
|
||||
vector: List[float],
|
||||
brand: Optional[str],
|
||||
category: Optional[str],
|
||||
fetch_k: int,
|
||||
ef_search: int,
|
||||
min_score: float,
|
||||
) -> List[Dict[str, Any]]:
|
||||
from app.services.vector_store import image_vector_search
|
||||
|
||||
rows = image_vector_search(vector, brand=brand, top_k=fetch_k, category=category, ef_search=ef_search)
|
||||
out: List[Dict[str, Any]] = []
|
||||
seen: Set[Tuple[str, str]] = set()
|
||||
for row in rows:
|
||||
try:
|
||||
score = 1.0 - float(row.get("distance"))
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
if score < min_score:
|
||||
continue
|
||||
key = (str(row.get("brand_table") or ""), str(row.get("image_id") or ""))
|
||||
if key in seen:
|
||||
continue
|
||||
seen.add(key)
|
||||
row["score"] = score
|
||||
out.append(row)
|
||||
return out
|
||||
|
||||
|
||||
def _hydrate(rows: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
|
||||
"""Swap the light candidate rows for full product rows, one read per table.
|
||||
|
||||
Ranking ran on ~100-byte rows (image_id, name, sizes, distance) because
|
||||
the unscoped search pulls candidates from every brand table; only the
|
||||
winners are worth a 7KB product card. `score` and `text_overlap` are
|
||||
carried over. A row whose card cannot be read keeps its light form, so
|
||||
a match is never lost to a hydration hiccup.
|
||||
"""
|
||||
from app.services.vector_store import fetch_products_by_image_ids
|
||||
|
||||
by_table: Dict[str, List[str]] = {}
|
||||
for row in rows:
|
||||
by_table.setdefault(str(row.get("brand_table") or ""), []).append(str(row.get("image_id") or ""))
|
||||
cards: Dict[Tuple[str, str], Dict[str, Any]] = {}
|
||||
for table, ids in by_table.items():
|
||||
if not table:
|
||||
continue
|
||||
for card in fetch_products_by_image_ids(table, ids):
|
||||
cards[(table, str(card.get("image_id") or ""))] = card
|
||||
|
||||
out: List[Dict[str, Any]] = []
|
||||
for row in rows:
|
||||
key = (str(row.get("brand_table") or ""), str(row.get("image_id") or ""))
|
||||
card = cards.get(key)
|
||||
if card is None:
|
||||
out.append(row)
|
||||
continue
|
||||
merged = dict(card)
|
||||
merged["brand"] = merged.get("brand") or row.get("brand")
|
||||
merged["brand_table"] = row.get("brand_table")
|
||||
merged["score"] = row["score"]
|
||||
merged["text_overlap"] = row.get("text_overlap", 0.0)
|
||||
out.append(merged)
|
||||
return out
|
||||
|
||||
|
||||
def search_by_vector(
|
||||
vector: Sequence[float],
|
||||
text: Optional[str] = None,
|
||||
brand: Optional[str] = None,
|
||||
category: Optional[str] = None,
|
||||
top_k: int = IMAGE_SEARCH_DEFAULT_TOP_K,
|
||||
min_score: float = IMAGE_SEARCH_DEFAULT_MIN_SCORE,
|
||||
) -> ImageSearchResult:
|
||||
"""The catalog rows most like `vector`, best first, at most `top_k`.
|
||||
|
||||
Raises InvalidVectorError for a vector that cannot be searched with.
|
||||
Everything else degrades to an empty result (no database, no vectors).
|
||||
"""
|
||||
unit = normalise_vector(vector)
|
||||
top_k = max(1, min(int(top_k), IMAGE_SEARCH_MAX_TOP_K))
|
||||
fetch_k = min(max(top_k * 3, 30), IMAGE_SEARCH_MAX_FETCH_K)
|
||||
ef_search = max(_MIN_EF_SEARCH, fetch_k)
|
||||
label = (text or "").strip() or None
|
||||
|
||||
explicit = (brand or "").strip() or None
|
||||
detected = explicit
|
||||
if detected is None and label:
|
||||
try:
|
||||
from app.services.query_intent import extract_brand_mention
|
||||
detected = extract_brand_mention(label)
|
||||
except Exception as exc: # noqa: BLE001 - brand detection is an optimisation
|
||||
logger.debug("brand detection skipped: %s", exc)
|
||||
detected = None
|
||||
|
||||
rows = _candidates(unit, detected, category, fetch_k, ef_search, min_score)
|
||||
scoped = detected is not None
|
||||
fallback = False
|
||||
if not rows and detected and not explicit:
|
||||
rows = _candidates(unit, None, category, fetch_k, ef_search, min_score)
|
||||
scoped, fallback = False, True
|
||||
|
||||
words, sizes = tokens(label)
|
||||
for row in rows:
|
||||
row["text_overlap"] = text_overlap(words, sizes, row)
|
||||
row["size_in_name"] = size_in_name(sizes, row)
|
||||
rows.sort(key=rank_key)
|
||||
# Judged on every candidate, not the top_k shown: with top_k=1 the rival
|
||||
# would otherwise never be seen.
|
||||
confidence, margin = match_confidence(rows)
|
||||
winners = _hydrate(rows[:top_k])
|
||||
|
||||
return ImageSearchResult(
|
||||
rows=winners,
|
||||
detected_brand=detected,
|
||||
scoped_to_brand=scoped,
|
||||
scope_fallback=fallback,
|
||||
min_score=min_score,
|
||||
query_text=label,
|
||||
top_k=top_k,
|
||||
match_confidence=confidence,
|
||||
margin=margin,
|
||||
)
|
||||
@@ -48,6 +48,7 @@ from typing import Optional, List
|
||||
import requests
|
||||
from urllib.parse import urlparse
|
||||
import json
|
||||
import re
|
||||
import logging
|
||||
import subprocess
|
||||
import sys
|
||||
@@ -113,11 +114,44 @@ def _query_openfacts(query: str, max_results: int) -> list:
|
||||
return []
|
||||
|
||||
|
||||
def _openfacts_product_is_brand(product: dict, brand: Optional[str]) -> bool:
|
||||
"""Does this Open*Facts record belong to `brand`?
|
||||
|
||||
Open*Facts' free-text search matches on the NAME, not the brand: asked for
|
||||
"Dabur Honey 1kg" it answered with a UK, a French, a Swiss and a Spanish
|
||||
honey, every one a real front-of-pack photo of somebody else's product.
|
||||
The record says who made it - `brands` / `brands_tags` - and that field
|
||||
used to be thrown away here, leaving a per-URL API round trip downstream
|
||||
(`image_corroboration.openfacts_product_matches_brand`) as the only thing
|
||||
standing between those photos and the catalogue. Check it at the source.
|
||||
|
||||
Fails OPEN on a record with no brand at all: an unlabelled record is not
|
||||
evidence of another brand, and the downstream gate still runs.
|
||||
"""
|
||||
if not brand:
|
||||
return True
|
||||
from app.services.image_corroboration import brand_tokens
|
||||
|
||||
tokens = brand_tokens(brand)
|
||||
if not tokens:
|
||||
return True
|
||||
tags = product.get("brands_tags") or []
|
||||
haystack = " ".join([str(product.get("brands") or "")] + [str(t) for t in tags]).lower()
|
||||
if not haystack.strip():
|
||||
return True
|
||||
return any(t in haystack for t in tokens)
|
||||
|
||||
|
||||
def find_images_openfacts(title: str, brand: Optional[str] = None, max_results: int = 20) -> list:
|
||||
"""Query the Open *Facts family of open product databases for real
|
||||
product photos. No API key required. Falls back from a brand+title
|
||||
query to a title-only query if the combined query is too specific to
|
||||
match anything (small/regional brand name variants are a common case)."""
|
||||
match anything (small/regional brand name variants are a common case).
|
||||
|
||||
Only records whose own `brands` field names our brand are used - see
|
||||
`_openfacts_product_is_brand`. The title-only fallback makes this filter
|
||||
load-bearing: "Honey 1kg" matches every honey on the site.
|
||||
"""
|
||||
if not USE_OPEN_FACTS:
|
||||
return []
|
||||
|
||||
@@ -125,12 +159,12 @@ def find_images_openfacts(title: str, brand: Optional[str] = None, max_results:
|
||||
if not query:
|
||||
return []
|
||||
|
||||
products = _query_openfacts(query, max_results)
|
||||
products = [p for p in _query_openfacts(query, max_results) if _openfacts_product_is_brand(p, brand)]
|
||||
if not products and brand and title:
|
||||
# Combined "brand + title" query found nothing - retry with just
|
||||
# the title, since Open*Facts' free-text search is exact-ish and
|
||||
# brand naming conventions vary (e.g. "Dettol" vs "Reckitt Dettol").
|
||||
products = _query_openfacts(title, max_results)
|
||||
products = [p for p in _query_openfacts(title, max_results) if _openfacts_product_is_brand(p, brand)]
|
||||
|
||||
urls: List[str] = []
|
||||
for product in products:
|
||||
@@ -162,7 +196,8 @@ def find_product_quantity_openfacts(title: str, brand: Optional[str] = None) ->
|
||||
# ---------------------------------------------------------------------------
|
||||
# 2. Wikimedia Commons - free/open media repository, no API key
|
||||
# ---------------------------------------------------------------------------
|
||||
def find_images_wikimedia(title: str, brand: Optional[str] = None, max_results: int = 20) -> list:
|
||||
def find_images_wikimedia(title: str, brand: Optional[str] = None, max_results: int = 20,
|
||||
produce: bool = False) -> list:
|
||||
"""Search Wikimedia Commons (the open media library behind Wikipedia)
|
||||
for product/packaging photos. Good secondary source for established
|
||||
brands; complements Open*Facts which leans more food/grocery."""
|
||||
@@ -180,6 +215,14 @@ def find_images_wikimedia(title: str, brand: Optional[str] = None, max_results:
|
||||
# MediaWiki's standard search syntax including the minus (-) prefix
|
||||
# for terms to exclude.
|
||||
exclusion_terms = "-car -vehicle -automotive -motorsport -racing -motorcycle -bike -people -person -portrait -animal -pet -dog -cat -bird -fish -landscape -nature -tour -travel -building -architecture -sport -game -flower -rose -floral -petal -bouquet -botanical -plant -garden -tree -herb"
|
||||
if produce:
|
||||
# Every one of those exclusions describes what a fruit, a vegetable
|
||||
# or a fish ACTUALLY IS. Searching Commons for a tomato while
|
||||
# excluding "-plant -garden -nature", or for mackerel while
|
||||
# excluding "-fish -animal", asks the index to rule out the answer.
|
||||
# The exclusions exist to keep a BRANDED search off unrelated
|
||||
# subjects, which is not the problem here.
|
||||
exclusion_terms = "-logo -advertisement -packaging"
|
||||
search_query = f"{query} {exclusion_terms} filetype:bitmap"
|
||||
resp = requests.get(
|
||||
"https://commons.wikimedia.org/w/api.php",
|
||||
@@ -402,6 +445,23 @@ def _dedupe(urls: list) -> list:
|
||||
# that image search APIs tend to confuse with non-product content.
|
||||
# Keyed by the ambiguous word (lowercase); value is the context term
|
||||
# to append to the search query.
|
||||
#
|
||||
# EVERY KEY HERE ASSUMES THE WORD IS A PACKAGED BRAND, NOT A COMMODITY.
|
||||
# The table was written for Cadbury Perk and Britannia Tiger, where "perk" and
|
||||
# "tiger" really do need steering away from perks and big cats. For the
|
||||
# Own Products bucket the word IS the literal item, so the hint inverts the
|
||||
# search: `"apple" -> "fruit juice"` is why a search for the fruit returned
|
||||
# juice cartons and milkshakes, and why the stored image for `Apple` was a
|
||||
# 2-litre juice bottle.
|
||||
#
|
||||
# Callers handling loose produce pass `produce=True`, which skips this table
|
||||
# entirely. Do not "fix" an entry by making it more specific - the whole table
|
||||
# is wrong in that context, not any one row of it.
|
||||
# Kept as a named constant rather than an inline literal: this pattern has to
|
||||
# survive being edited by tooling that mangles backslash escapes, and a silent
|
||||
# corruption here turns whole-word matching back into no matching at all.
|
||||
_WORD_BOUNDARY = chr(92) + "b"
|
||||
|
||||
_AMBIQUITY_HINTS: dict[str, str] = {
|
||||
"perk": "chocolate wafer",
|
||||
"crunch": "chocolate wafer",
|
||||
@@ -438,7 +498,8 @@ _AMBIQUITY_HINTS: dict[str, str] = {
|
||||
}
|
||||
|
||||
|
||||
def _clean_search_title(title: str, brand: str | None = None) -> str:
|
||||
def _clean_search_title(title: str, brand: str | None = None,
|
||||
produce: bool = False) -> str:
|
||||
"""Strip redundant brand prefix from title and add context hints for
|
||||
ambiguous product names that confuse image search APIs.
|
||||
|
||||
@@ -448,6 +509,9 @@ def _clean_search_title(title: str, brand: str | None = None) -> str:
|
||||
|
||||
_clean_search_title("Britannia Good Day Biscuits", "Britannia")
|
||||
-> "Good Day Biscuits"
|
||||
|
||||
`produce=True` suppresses the hint table entirely - see `_AMBIQUITY_HINTS`
|
||||
for why it is exactly backwards for a raw commodity.
|
||||
"""
|
||||
clean = (title or "").strip()
|
||||
if brand:
|
||||
@@ -458,10 +522,25 @@ def _clean_search_title(title: str, brand: str | None = None) -> str:
|
||||
if not clean:
|
||||
clean = (title or "").strip()
|
||||
|
||||
# Append context hint for any ambiguous word in the cleaned title
|
||||
if produce:
|
||||
return clean
|
||||
|
||||
# Append a context hint for an ambiguous word in the cleaned title.
|
||||
#
|
||||
# MATCHED AS A WHOLE WORD, not as a substring. The substring version pulled
|
||||
# hints into every title that merely CONTAINED a keyword:
|
||||
#
|
||||
# Pineapple -> "Pineapple fruit juice" (apple)
|
||||
# Custard Apple -> "Custard Apple fruit juice" (apple)
|
||||
# Buttermilk -> "Buttermilk chocolate dairy" (milk)
|
||||
# Baby Corn -> "Baby Corn snack flakes" (corn)
|
||||
# Eggs White -> "Eggs White toothpaste dental" (white)
|
||||
#
|
||||
# Multi-word keys ("good day", "5 star") still need a phrase search, so the
|
||||
# boundary is applied around the whole key rather than per token.
|
||||
clean_lower = clean.lower()
|
||||
for word, hint in _AMBIQUITY_HINTS.items():
|
||||
if word in clean_lower:
|
||||
if re.search(_WORD_BOUNDARY + re.escape(word) + _WORD_BOUNDARY, clean_lower):
|
||||
clean = f"{clean} {hint}"
|
||||
break
|
||||
|
||||
@@ -472,14 +551,16 @@ def _clean_search_title(title: str, brand: str | None = None) -> str:
|
||||
# Public API (same signatures as before, so catalog_engine.py and
|
||||
# downstream callers don't need to change)
|
||||
# ---------------------------------------------------------------------------
|
||||
def find_image_url(title: str, brand: Optional[str] = None, country_hint: Optional[str] = None) -> Optional[str]:
|
||||
def find_image_url(title: str, brand: Optional[str] = None, country_hint: Optional[str] = None,
|
||||
produce: bool = False) -> Optional[str]:
|
||||
"""Get a single working, validated image URL."""
|
||||
urls = find_all_image_urls(title, brand, country_hint)
|
||||
urls = find_all_image_urls(title, brand, country_hint, produce=produce)
|
||||
return urls[0] if urls else None
|
||||
|
||||
|
||||
def find_all_image_urls(title: str, brand: Optional[str] = None, country_hint: Optional[str] = None,
|
||||
validate: bool = True, max_results: int = 24) -> list:
|
||||
validate: bool = True, max_results: int = 24,
|
||||
produce: bool = False) -> list:
|
||||
"""Get working image URLs by trying multiple open-source sources in
|
||||
priority order (cheapest/most-reliable first), merging and validating
|
||||
results. Only escalates to the Playwright browser fallback if every
|
||||
@@ -488,11 +569,18 @@ def find_all_image_urls(title: str, brand: Optional[str] = None, country_hint: O
|
||||
|
||||
# Strip redundant brand prefix from title to avoid repetitive queries
|
||||
# like "Cadbury Perk Cadbury Perk Crunch".
|
||||
search_title = _clean_search_title(title, brand)
|
||||
search_title = _clean_search_title(title, brand, produce=produce)
|
||||
|
||||
candidates.extend(find_images_openfacts(search_title, brand, max_results))
|
||||
# Open*Facts is a PACKAGED-GOODS database - food, beauty and household
|
||||
# products photographed front-of-pack. Asked about loose produce it answers
|
||||
# with whatever carton or bottle mentions the word, which is how the live
|
||||
# catalogue ended up serving openbeautyfacts cosmetics photos for Banana,
|
||||
# Orange, Papaya, Guava, Lemon, Pear and twenty more. Skipped for produce.
|
||||
if not produce:
|
||||
candidates.extend(find_images_openfacts(search_title, brand, max_results))
|
||||
if len(candidates) < max_results:
|
||||
candidates.extend(find_images_wikimedia(search_title, brand, max_results))
|
||||
candidates.extend(find_images_wikimedia(search_title, brand, max_results,
|
||||
produce=produce))
|
||||
if len(candidates) < max_results:
|
||||
candidates.extend(find_all_image_urls_ddg(search_title, brand, max_results))
|
||||
if len(candidates) < max_results:
|
||||
|
||||
133
app/services/image_search_log.py
Normal file
133
app/services/image_search_log.py
Normal file
@@ -0,0 +1,133 @@
|
||||
"""One log line per search-by-photo request, and - when asked - the request itself.
|
||||
|
||||
WHY
|
||||
---
|
||||
A colleague photographed the "Cadbury Dairy Milk Lickables" card and got
|
||||
"Milk Toned"; "Aachi Sambar Powder 100g" came back as Sakthi's "Sambar powder
|
||||
50g". Replayed on the server, both photos rank the right product first, so
|
||||
the difference is in what the app SENDS - the vector (is its on-device model
|
||||
the same as ours?), whether any OCR `text` came with it, which route it hit.
|
||||
None of that was visible. This module makes it visible:
|
||||
|
||||
* `record()` logs, for every /search/image*, /search/identify request: the
|
||||
route, the text, brand, detected brand, a short hash and the norm of the
|
||||
vector, the top three (brand, name, score) and the lead over the best
|
||||
different photo. Always on; it is one INFO line.
|
||||
* With IMAGE_SEARCH_CAPTURE_DIR set, the whole request - the 1024 floats,
|
||||
the fields, the photo when there was one - is also written there, for
|
||||
`python -m scripts.replay_image_query`. Off by default: these routes are
|
||||
public. The newest IMAGE_SEARCH_CAPTURE_MAX requests are kept.
|
||||
|
||||
Never raises: a diagnostic must not fail a search.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import logging
|
||||
import math
|
||||
import struct
|
||||
import time
|
||||
import uuid
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional, Sequence
|
||||
|
||||
from app.infrastructure import settings
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_TEXT_LOG_CHARS = 120
|
||||
|
||||
|
||||
def vector_fingerprint(vector: Optional[Sequence[float]]) -> Optional[str]:
|
||||
"""10 hex characters identifying a vector: equal for the same float32 values."""
|
||||
if vector is None:
|
||||
return None
|
||||
packed = struct.pack(f"<{len(vector)}f", *[float(v) for v in vector])
|
||||
return hashlib.sha1(packed).hexdigest()[:10]
|
||||
|
||||
|
||||
def _top(rows: Sequence[Dict[str, Any]], n: int = 3) -> List[Dict[str, Any]]:
|
||||
out = []
|
||||
for row in list(rows)[:n]:
|
||||
try:
|
||||
score = round(float(row.get("score", 0.0)), 4)
|
||||
except (TypeError, ValueError):
|
||||
score = None
|
||||
out.append({
|
||||
"brand": row.get("brand") or row.get("brand_table"),
|
||||
"product_name": row.get("product_name") or row.get("title"),
|
||||
"image_id": row.get("image_id"),
|
||||
"score": score,
|
||||
})
|
||||
return out
|
||||
|
||||
|
||||
def _prune(folder: Path, keep: int) -> None:
|
||||
files = sorted(folder.glob("*.json"), key=lambda p: p.stat().st_mtime)
|
||||
for stale in files[:max(0, len(files) - keep)]:
|
||||
for sibling in folder.glob(stale.stem + ".*"):
|
||||
sibling.unlink(missing_ok=True)
|
||||
|
||||
|
||||
def record(
|
||||
route: str,
|
||||
*,
|
||||
vector: Optional[Sequence[float]],
|
||||
text: Optional[str],
|
||||
brand: Optional[str],
|
||||
category: Optional[str],
|
||||
top_k: int,
|
||||
rows: Sequence[Dict[str, Any]],
|
||||
detected_brand: Optional[str] = None,
|
||||
scoped_to_brand: bool = False,
|
||||
match_confidence: Optional[str] = None,
|
||||
margin: Optional[float] = None,
|
||||
matched_by: Optional[str] = None,
|
||||
fallback_reason: Optional[str] = None,
|
||||
text_fallback: Optional[bool] = None,
|
||||
photo: Optional[bytes] = None,
|
||||
) -> None:
|
||||
"""Log the request and its answer; save it when capture is on."""
|
||||
try:
|
||||
norm = math.sqrt(sum(float(v) * float(v) for v in vector)) if vector is not None else None
|
||||
top = _top(rows)
|
||||
logger.info(
|
||||
"[IMAGE_SEARCH] route=%s text=%r brand=%r detected=%r scoped=%s vec=%s norm=%s "
|
||||
"matched_by=%s reason=%s confidence=%s margin=%s top=%s",
|
||||
route, (text or "")[:_TEXT_LOG_CHARS] or None, brand, detected_brand, scoped_to_brand,
|
||||
vector_fingerprint(vector), None if norm is None else round(norm, 4),
|
||||
matched_by, fallback_reason, match_confidence,
|
||||
None if margin is None else round(margin, 4),
|
||||
[(t["brand"], t["product_name"], t["score"]) for t in top],
|
||||
)
|
||||
folder_name = settings.IMAGE_SEARCH_CAPTURE_DIR
|
||||
if not folder_name:
|
||||
return
|
||||
folder = Path(folder_name)
|
||||
folder.mkdir(parents=True, exist_ok=True)
|
||||
stem = time.strftime("%Y%m%dT%H%M%S") + "_" + uuid.uuid4().hex[:6]
|
||||
payload = {
|
||||
"route": route,
|
||||
"received_at": time.strftime("%Y-%m-%dT%H:%M:%S%z"),
|
||||
"request": {
|
||||
"vector": None if vector is None else [float(v) for v in vector],
|
||||
"text": text, "brand": brand, "category": category, "top_k": top_k,
|
||||
"text_fallback": text_fallback,
|
||||
},
|
||||
"vector_fingerprint": vector_fingerprint(vector),
|
||||
"answer": {
|
||||
"matched_by": matched_by, "fallback_reason": fallback_reason,
|
||||
"match_confidence": match_confidence, "margin": margin,
|
||||
"detected_brand": detected_brand, "scoped_to_brand": scoped_to_brand, "top": top,
|
||||
},
|
||||
"photo": None,
|
||||
}
|
||||
if photo:
|
||||
photo_path = folder / f"{stem}.img"
|
||||
photo_path.write_bytes(photo)
|
||||
payload["photo"] = photo_path.name
|
||||
(folder / f"{stem}.json").write_text(json.dumps(payload), encoding="utf-8")
|
||||
_prune(folder, max(1, settings.IMAGE_SEARCH_CAPTURE_MAX))
|
||||
except Exception as exc: # noqa: BLE001 - a diagnostic must never fail the search
|
||||
logger.warning("image search log skipped: %s", exc)
|
||||
423
app/services/image_vector.py
Normal file
423
app/services/image_vector.py
Normal file
@@ -0,0 +1,423 @@
|
||||
"""An image embedding of every product's PRIMARY image, stored on its brand table.
|
||||
|
||||
WHAT `img_vector` IS
|
||||
--------------------
|
||||
The image the product card shows - `image_url`, else `image_urls[0]`, which is
|
||||
exactly the order `frontend/src/components/ProductCard.jsx` walks - run through
|
||||
MobileNetV3-Small (`app/services/image_embedder.py`): 1024 floats, L2-normalised,
|
||||
stored as pgvector `vector(1024)` with an hnsw cosine index. It is a learned
|
||||
embedding, not a picture: two photos of the same product from different angles
|
||||
land near each other, and `<=>` between rows is a visual-similarity search. The
|
||||
model, runtime and post-processing are the same a colleague uses for their
|
||||
product images, so our vectors and theirs are directly comparable.
|
||||
|
||||
(Until 2026-09-17 this column held a 32x32 pixel thumbnail as `vector(3072)`.
|
||||
`vector_store._ensure_img_vector_type` drops and re-creates a column of the
|
||||
wrong dimension on the next write, discarding those vectors; the backfill
|
||||
script fills the new ones.)
|
||||
|
||||
`img_vector_src` records the URL the vector was computed from. That is what
|
||||
makes recomputation idempotent: a row is due when it has no vector, or when
|
||||
its current primary URL differs from the one the vector came from. A change
|
||||
to the model or its preprocessing is neither; `backfill_rows(force=True)` /
|
||||
`--force` recomputes every row for that.
|
||||
|
||||
WHY THIS MODULE, AND NOT THE UPSERT
|
||||
-----------------------------------
|
||||
`upsert_brand_products` never names these columns, for the reason documented
|
||||
next to its INSERT: every product dict that reaches it comes from a
|
||||
spreadsheet, the seed JSON or a SELECT round-trip, none of which can carry a
|
||||
vector, so naming the column there would only ever NULL it. The writes here
|
||||
are targeted UPDATEs by image_id, the same shape as nutrition_score_sync.
|
||||
|
||||
WHY A SINGLE BOUNDED WORKER
|
||||
---------------------------
|
||||
The upsert calls `schedule()`, which enqueues (brand, image_ids) and returns.
|
||||
One daemon thread drains the queue, so at most one image is being downloaded
|
||||
and decoded in the API process at any moment, a write never waits on a CDN,
|
||||
and an overflowing queue simply leaves rows for
|
||||
`scripts/backfill_image_vectors.py`. Same shape as app/core/batch_worker.py,
|
||||
for the same reasons.
|
||||
|
||||
The worker is also the only thread in the API process that runs the model,
|
||||
which matters because a TFLite interpreter is not thread-safe; the embedder
|
||||
serialises every call on its own lock regardless.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import queue
|
||||
import threading
|
||||
import time
|
||||
from typing import Any, Dict, Iterable, List, Mapping, Optional, Tuple
|
||||
from urllib.parse import urlparse
|
||||
|
||||
from app.infrastructure.settings import (
|
||||
ENABLE_IMAGE_VECTORS,
|
||||
IMAGE_VECTOR_HOST_PAUSE_SECONDS,
|
||||
IMAGE_VECTOR_MAX_BYTES,
|
||||
IMAGE_VECTOR_QUEUE_MAX,
|
||||
IMAGE_VECTOR_TIMEOUT_SECONDS,
|
||||
MIN_IMAGE_BYTES,
|
||||
)
|
||||
from app.services import image_embedder
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
IMG_VECTOR_DIM = image_embedder.EMBED_DIM # 1024
|
||||
IMG_VECTOR_COLUMNS = ("img_vector", "img_vector_src")
|
||||
|
||||
_IMAGE_EXTENSIONS = (".jpg", ".jpeg", ".png", ".webp", ".gif", ".bmp")
|
||||
_CHUNK = 64 * 1024
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Which image
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def primary_image_url(row: Mapping[str, Any]) -> Optional[str]:
|
||||
"""The URL the product card displays, or None if it shows nothing.
|
||||
|
||||
`image_url` first, else `image_urls[0]` - ProductCard.jsx builds its list in
|
||||
that order. Normalised through `_usable_image_url` so blanks, bare paths and
|
||||
hosts in `_DEAD_IMAGE_HOSTS` never cost a request. The string returned here
|
||||
is what goes into `img_vector_src`, so staleness is always judged against
|
||||
the same definition.
|
||||
"""
|
||||
from app.services.vector_store import _usable_image_url
|
||||
|
||||
candidate = _usable_image_url(row.get("image_url"))
|
||||
if candidate:
|
||||
return candidate
|
||||
urls = row.get("image_urls") or []
|
||||
if isinstance(urls, (list, tuple)) and urls:
|
||||
return _usable_image_url(urls[0])
|
||||
return None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Bytes -> vector
|
||||
# ---------------------------------------------------------------------------
|
||||
# The model, its preprocessing and its lock live in image_embedder.py; this
|
||||
# module only decides which bytes go in and where the result goes.
|
||||
|
||||
def to_pg(vector: Iterable[float]) -> str:
|
||||
"""The text form pgvector accepts, identical to what the upsert uses for `embedding`."""
|
||||
return "[" + ",".join(str(float(v)) for v in vector) + "]"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# URL -> bytes
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
_RATE_LIMIT_BACKOFF_SECONDS = 5.0
|
||||
_RATE_LIMIT_BACKOFF_CAP_SECONDS = 15.0
|
||||
|
||||
|
||||
def _retry_after_seconds(header: Optional[str]) -> float:
|
||||
"""Seconds to wait after a 429: the server's Retry-After if it is a plain
|
||||
number, capped so one hostile header cannot stall the worker."""
|
||||
try:
|
||||
seconds = float(header) if header else _RATE_LIMIT_BACKOFF_SECONDS
|
||||
except (TypeError, ValueError):
|
||||
seconds = _RATE_LIMIT_BACKOFF_SECONDS
|
||||
return max(0.0, min(seconds, _RATE_LIMIT_BACKOFF_CAP_SECONDS))
|
||||
|
||||
|
||||
def download_image_bytes(url: str, timeout: Optional[float] = None) -> Optional[bytes]:
|
||||
"""Fetch `url` as image bytes, or None.
|
||||
|
||||
Same two-step Referer strategy as `image_search.validate_image_url_live`:
|
||||
first with a same-site Referer (hotlink-protected CDNs), then with none
|
||||
(hosts that reject a Referer pretending to be same-site). Streams in
|
||||
chunks and gives up past IMAGE_VECTOR_MAX_BYTES, so a wrong URL to a video
|
||||
cannot fill memory; anything under MIN_IMAGE_BYTES is a placeholder.
|
||||
"""
|
||||
if not url or not url.startswith(("http://", "https://")):
|
||||
return None
|
||||
import requests
|
||||
from app.services.image_search import _BROWSER_UA
|
||||
|
||||
parsed = urlparse(url)
|
||||
same_site = f"{parsed.scheme}://{parsed.netloc}/" if parsed.netloc else None
|
||||
timeout = timeout or IMAGE_VECTOR_TIMEOUT_SECONDS
|
||||
|
||||
# Third attempt exists for one reason: a 429. Wikimedia served ~100 of the
|
||||
# Own Products images in a row and then throttled the rest; those URLs are
|
||||
# fine, the pace was not. Honour Retry-After (capped) and try once more.
|
||||
for referer in (same_site, None, same_site):
|
||||
headers = {"User-Agent": _BROWSER_UA, "Accept": "image/*,*/*;q=0.8"}
|
||||
if referer:
|
||||
headers["Referer"] = referer
|
||||
resp = None
|
||||
try:
|
||||
resp = requests.get(url, headers=headers, timeout=timeout, stream=True)
|
||||
if resp.status_code == 429:
|
||||
logger.debug("HTTP 429 for %s - backing off", url)
|
||||
time.sleep(_retry_after_seconds(resp.headers.get("retry-after")))
|
||||
continue
|
||||
if resp.status_code != 200:
|
||||
logger.debug("HTTP %s for %s", resp.status_code, url)
|
||||
continue
|
||||
content_type = (resp.headers.get("content-type") or "").lower()
|
||||
if "image" not in content_type:
|
||||
# A server that says text/html means it - typically a CDN that
|
||||
# now redirects every dead path to a homepage (uat.amul.com does
|
||||
# exactly this), and the .png in the URL is no evidence at all.
|
||||
# Only an unlabelled body (octet-stream, blank) gets the benefit
|
||||
# of the extension.
|
||||
if content_type.startswith("text/") or not parsed.path.lower().endswith(_IMAGE_EXTENSIONS):
|
||||
logger.debug("not an image (%s): %s", content_type, url)
|
||||
return None
|
||||
buf = bytearray()
|
||||
for chunk in resp.iter_content(chunk_size=_CHUNK):
|
||||
if not chunk:
|
||||
continue
|
||||
buf.extend(chunk)
|
||||
if len(buf) > IMAGE_VECTOR_MAX_BYTES:
|
||||
logger.debug("image over %d bytes, abandoned: %s", IMAGE_VECTOR_MAX_BYTES, url)
|
||||
return None
|
||||
if len(buf) < MIN_IMAGE_BYTES:
|
||||
logger.debug("image too small (%d bytes): %s", len(buf), url)
|
||||
return None
|
||||
return bytes(buf)
|
||||
except Exception as exc: # noqa: BLE001 - a bad host must not stop the batch
|
||||
logger.debug("download failed for %s: %s", url, exc)
|
||||
continue
|
||||
finally:
|
||||
if resp is not None:
|
||||
try:
|
||||
resp.close()
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
return None
|
||||
|
||||
|
||||
def vector_for_url(url: str, timeout: Optional[float] = None) -> Tuple[Optional[List[float]], str]:
|
||||
"""(L2-normalised 1024-float embedding or None, the URL it was computed from)."""
|
||||
data = download_image_bytes(url, timeout=timeout)
|
||||
if data is None:
|
||||
return None, url
|
||||
return image_embedder.embedding_for_bytes(data), url
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Table access
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def table_has_columns(cur, table_name: str) -> bool:
|
||||
"""True when the table carries both img_vector columns.
|
||||
|
||||
Every writer checks this first, so a table the migration has not reached -
|
||||
or one whose pgvector refused `vector(3072)` - is skipped, never errored.
|
||||
"""
|
||||
cur.execute(
|
||||
"SELECT column_name FROM information_schema.columns "
|
||||
"WHERE table_schema = 'public' AND table_name = %s AND column_name = ANY(%s)",
|
||||
(table_name, list(IMG_VECTOR_COLUMNS)),
|
||||
)
|
||||
present = {row[0] for row in cur.fetchall()}
|
||||
return all(col in present for col in IMG_VECTOR_COLUMNS)
|
||||
|
||||
|
||||
def rows_needing_vectors(
|
||||
cur,
|
||||
table_name: str,
|
||||
image_ids: Optional[List[str]] = None,
|
||||
*,
|
||||
recompute_stale: bool = True,
|
||||
limit: Optional[int] = None,
|
||||
force: bool = False,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Rows that have a primary image and no current vector for it.
|
||||
|
||||
"Current" means `img_vector_src` equals the primary URL; a row whose
|
||||
`image_url` was rewritten by a later upsert is stale and is included when
|
||||
`recompute_stale` is on. Rows with no usable primary are dropped here, so
|
||||
the caller never spends a request on them.
|
||||
|
||||
`force` takes every row with a primary image, current or not. That is the
|
||||
switch for a model or preprocessing change: the URL did not move, so
|
||||
nothing else would notice that every stored vector is now the wrong one.
|
||||
"""
|
||||
sql = (
|
||||
f"SELECT image_id, product_name, image_url, image_urls, img_vector_src, "
|
||||
f"(img_vector IS NULL) AS missing FROM {table_name}"
|
||||
)
|
||||
clauses: List[str] = []
|
||||
params: List[Any] = []
|
||||
if not force:
|
||||
clauses.append(
|
||||
"(img_vector IS NULL OR img_vector_src IS DISTINCT FROM "
|
||||
"COALESCE(NULLIF(image_url, ''), image_urls[1]))"
|
||||
)
|
||||
if image_ids is not None:
|
||||
clauses.append("image_id = ANY(%s)")
|
||||
params.append(list(image_ids))
|
||||
if clauses:
|
||||
sql += " WHERE " + " AND ".join(clauses)
|
||||
sql += " ORDER BY updated_at DESC"
|
||||
if limit is not None:
|
||||
sql += " LIMIT %s"
|
||||
params.append(int(limit))
|
||||
cur.execute(sql, params)
|
||||
colnames = [d[0] for d in cur.description]
|
||||
|
||||
out: List[Dict[str, Any]] = []
|
||||
for raw in cur.fetchall():
|
||||
row = dict(zip(colnames, raw))
|
||||
primary = primary_image_url(row)
|
||||
if not primary:
|
||||
continue
|
||||
if not force:
|
||||
if not row.get("missing") and not recompute_stale:
|
||||
continue
|
||||
if not row.get("missing") and row.get("img_vector_src") == primary:
|
||||
continue
|
||||
row["primary_url"] = primary
|
||||
out.append(row)
|
||||
return out
|
||||
|
||||
|
||||
def store_vector(cur, table_name: str, image_id: str, vector: List[float], src: str) -> None:
|
||||
"""Write one vector. Deliberately leaves `updated_at` alone: the catalog
|
||||
listing orders on it, and a background enrichment must not reshuffle it."""
|
||||
cur.execute(
|
||||
f"UPDATE {table_name} SET img_vector = %s::vector, img_vector_src = %s WHERE image_id = %s",
|
||||
(to_pg(vector), src, image_id),
|
||||
)
|
||||
|
||||
|
||||
def backfill_rows(
|
||||
brand: str,
|
||||
image_ids: Optional[List[str]] = None,
|
||||
*,
|
||||
recompute_stale: bool = True,
|
||||
limit: Optional[int] = None,
|
||||
dry_run: bool = False,
|
||||
pause_seconds: Optional[float] = None,
|
||||
force: bool = False,
|
||||
) -> Dict[str, int]:
|
||||
"""Compute and store vectors for one brand, sequentially. Never raises.
|
||||
|
||||
Returns counts: candidates, computed, failed, skipped (1 when the table
|
||||
has no img_vector columns yet, or the embedder is unavailable - in both
|
||||
cases nothing is downloaded). `pause_seconds` is the gap between two
|
||||
requests to the same host; `force` recomputes rows that already have a
|
||||
current vector.
|
||||
"""
|
||||
from app.services.vector_store import _connect, _table_name
|
||||
|
||||
totals = {"candidates": 0, "computed": 0, "failed": 0, "skipped": 0}
|
||||
conn = _connect()
|
||||
if conn is None:
|
||||
logger.warning("image vectors: no database connection for %s", brand)
|
||||
return totals
|
||||
pause = IMAGE_VECTOR_HOST_PAUSE_SECONDS if pause_seconds is None else pause_seconds
|
||||
table_name = _table_name(brand)
|
||||
last_by_host: Dict[str, float] = {}
|
||||
try:
|
||||
with conn.cursor() as cur:
|
||||
if not table_has_columns(cur, table_name):
|
||||
logger.info("image vectors: %s has no img_vector columns yet - skipped", table_name)
|
||||
totals["skipped"] = 1
|
||||
return totals
|
||||
if not image_embedder.available():
|
||||
# available() already warned once with the reason. Rows stay
|
||||
# NULL and are picked up by the next write or the script.
|
||||
logger.info("image vectors: embedder unavailable - %s left for later", table_name)
|
||||
totals["skipped"] = 1
|
||||
return totals
|
||||
rows = rows_needing_vectors(
|
||||
cur, table_name, image_ids, recompute_stale=recompute_stale, limit=limit, force=force
|
||||
)
|
||||
totals["candidates"] = len(rows)
|
||||
for row in rows:
|
||||
url = row["primary_url"]
|
||||
host = urlparse(url).netloc.lower()
|
||||
wait = last_by_host.get(host, 0.0) + pause - time.monotonic()
|
||||
if wait > 0:
|
||||
time.sleep(wait)
|
||||
last_by_host[host] = time.monotonic()
|
||||
|
||||
vector, src = vector_for_url(url)
|
||||
if vector is None:
|
||||
totals["failed"] += 1
|
||||
continue
|
||||
if not dry_run:
|
||||
store_vector(cur, table_name, row["image_id"], vector, src)
|
||||
totals["computed"] += 1
|
||||
except Exception: # noqa: BLE001 - enrichment must never surface as a failure
|
||||
logger.exception("image vectors: backfill failed for %s", brand)
|
||||
finally:
|
||||
conn.close()
|
||||
if totals["candidates"]:
|
||||
logger.info(
|
||||
"image vectors: %s - %d candidate(s), %d computed, %d failed%s",
|
||||
table_name, totals["candidates"], totals["computed"], totals["failed"],
|
||||
" (dry run)" if dry_run else "",
|
||||
)
|
||||
return totals
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The hook the upsert calls
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
_queue: "queue.Queue[Tuple[str, List[str]]]" = queue.Queue(maxsize=max(1, IMAGE_VECTOR_QUEUE_MAX))
|
||||
_worker: Optional[threading.Thread] = None
|
||||
_lock = threading.Lock()
|
||||
|
||||
|
||||
def queue_depth() -> int:
|
||||
return _queue.qsize()
|
||||
|
||||
|
||||
def is_running() -> bool:
|
||||
return _worker is not None and _worker.is_alive()
|
||||
|
||||
|
||||
def schedule(brand: str, image_ids: Iterable[str]) -> bool:
|
||||
"""Queue the rows just written. Returns True if queued.
|
||||
|
||||
Off when ENABLE_IMAGE_VECTORS is false. `put_nowait`, never `put`: this
|
||||
runs inside the write path and must not wait. A full queue is logged and
|
||||
dropped - the backfill script picks those rows up, because they are the
|
||||
ones with no vector.
|
||||
"""
|
||||
if not ENABLE_IMAGE_VECTORS:
|
||||
return False
|
||||
ids = [i for i in image_ids if i]
|
||||
if not ids:
|
||||
return False
|
||||
try:
|
||||
_queue.put_nowait((brand, ids))
|
||||
except queue.Full:
|
||||
logger.warning(
|
||||
"image vectors: queue full (%d), %d row(s) for %s left for the backfill script",
|
||||
_queue.maxsize, len(ids), brand,
|
||||
)
|
||||
return False
|
||||
_ensure_worker()
|
||||
return True
|
||||
|
||||
|
||||
def _ensure_worker() -> None:
|
||||
global _worker
|
||||
with _lock:
|
||||
if _worker is not None and _worker.is_alive():
|
||||
return
|
||||
_worker = threading.Thread(target=_loop, name="image-vector-worker", daemon=True)
|
||||
_worker.start()
|
||||
|
||||
|
||||
def _loop() -> None:
|
||||
"""Drain forever; one bad batch must not strand the ones behind it."""
|
||||
while True:
|
||||
brand, ids = _queue.get()
|
||||
try:
|
||||
backfill_rows(brand, ids, recompute_stale=True)
|
||||
except Exception: # noqa: BLE001 - see docstring
|
||||
logger.exception("image vectors: worker failed on %s", brand)
|
||||
finally:
|
||||
_queue.task_done()
|
||||
362
app/services/label_match.py
Normal file
362
app/services/label_match.py
Normal file
@@ -0,0 +1,362 @@
|
||||
"""Find catalog products from the TEXT on a pack: OCR label -> `embedding` + names.
|
||||
|
||||
WHERE THIS SITS
|
||||
---------------
|
||||
The second rung of POST /api/search/identify (app/services/product_identify.py).
|
||||
When the photo's img_vector cannot confirm a product - a phone photo scores
|
||||
~0.63 against the catalog's render of the same pack - the label text takes
|
||||
over: the client's OCR `text`, or what app/services/ocr_service.py read off
|
||||
the photo. This module turns that text into ranked catalog rows.
|
||||
|
||||
WHY NOT catalog_search.search_catalog()
|
||||
---------------------------------------
|
||||
GET /api/search is tuned for a human typing in a search box, and two of its
|
||||
habits are wrong for OCR output:
|
||||
|
||||
* it passes the FUZZY category detection through as a hard filter, and that
|
||||
detector scores "taste" as Oral Care (matches "paste") and "Colgate" as
|
||||
Chocolates - a misread word would delete the right product from the result;
|
||||
* its lexical arm ANDs every term, so one smudged word means zero rows.
|
||||
|
||||
So this module goes to the store functions directly, never filters on a
|
||||
category it inferred, and asks the lexical arm for only the two most
|
||||
distinctive words.
|
||||
|
||||
HOW A MATCH IS RANKED
|
||||
---------------------
|
||||
Both arms run against the brand the label names (query_intent's
|
||||
`extract_brand_mention`), else every active brand:
|
||||
|
||||
* semantic - `embed_texts([f"{brand} {label}"])` (the stored vector is of
|
||||
"{brand} {name} {category} {description}", store_catalog_pipeline.py, so
|
||||
the brand prefix keeps the query in-distribution) -> `semantic_search`;
|
||||
* lexical - `lexical_search` on the two longest non-brand, non-size words,
|
||||
which tolerates a misread elsewhere on the label.
|
||||
|
||||
Merged rows are ordered by `text_overlap` FIRST (image_match's scorer: 3 per
|
||||
shared size token, 1 per shared word - the size is what separates the 89g,
|
||||
300g and 1kg rows of one product), then by `text_extra` ascending (name
|
||||
words the label never said - what separates "Dairy Milk" from "Dairy Milk
|
||||
Fruit & Nut"), then by cosine. `score` on a text row is
|
||||
`1 - (embedding <=> q)` in MiniLM's 384-d space: a different space from the
|
||||
image score, and the response says which one it is (`matched_by`).
|
||||
|
||||
A nearest-neighbour query always returns something, so `is_confident()`
|
||||
draws the line: the best row shares a token with the label, or its cosine
|
||||
clears IMAGE_IDENTIFY_MIN_TEXT_SCORE. Below that the caller treats the text
|
||||
rung as a miss. Read-only: nothing here writes.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
from typing import Any, Dict, List, Optional, Set, Tuple
|
||||
|
||||
from app.infrastructure.settings import (
|
||||
IMAGE_IDENTIFY_MIN_TEXT_SCORE,
|
||||
IMAGE_SEARCH_DEFAULT_TOP_K,
|
||||
IMAGE_SEARCH_MAX_FETCH_K,
|
||||
IMAGE_SEARCH_MAX_TOP_K,
|
||||
OCR_MAX_CHARS,
|
||||
)
|
||||
from app.services.image_match import _STOP, ImageSearchResult, _row_text, size_in_name, text_overlap, tokens
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Words a pack prints that name no product: the nutrition panel, the
|
||||
# regulatory block, storage advice, the price line. Dropped before either
|
||||
# arm sees the label so "ENERGY 480 kcal PROTEIN 7g" does not become the
|
||||
# query. Lowercase; compared after image_match.tokens' own normalisation.
|
||||
_NOISE: Set[str] = {
|
||||
"nutrition", "nutritional", "information", "facts", "typical", "values", "value",
|
||||
"energy", "kcal", "kj", "calories", "protein", "carbohydrate", "carbohydrates",
|
||||
"carbs", "sugar", "sugars", "fat", "fats", "saturated", "trans", "sodium",
|
||||
"cholesterol", "fibre", "fiber", "dietary", "serving", "servings", "serve",
|
||||
"serves", "amount", "ingredients", "ingredient", "contains", "allergen",
|
||||
"allergens", "allergy", "advice", "mfg", "mfd", "manufactured", "manufacturer",
|
||||
"marketed", "packed", "packer", "lot", "batch", "best", "before", "expiry",
|
||||
"exp", "use", "date", "fssai", "lic", "license", "licence", "mrp", "price",
|
||||
"incl", "inclusive", "all", "taxes", "tax", "ltd", "pvt", "limited", "india",
|
||||
"www", "com", "http", "https", "customer", "care", "helpline", "email", "call",
|
||||
"store", "cool", "dry", "place", "keep", "away", "sunlight", "hygienic",
|
||||
"net", "wt", "weight", "qty", "quantity", "approx", "veg", "vegetarian",
|
||||
"non", "product", "code", "unit", "units",
|
||||
}
|
||||
# A nutrition-panel basis ("per 100 g", "per serve") carries a quantity that
|
||||
# is NOT the pack size; left in, "100g" would score 3 points of overlap and
|
||||
# hand the match to the 100g sibling. Removed as a span, before tokenising.
|
||||
_PER_BASIS_RE = re.compile(
|
||||
r"\bper\s*\d+(?:\.\d+)?\s*(?:kg|gms|gm|g|ml|ltr|litre|l)\b|\bper\s+serv\w*\b",
|
||||
re.I,
|
||||
)
|
||||
# A nutrient with its value ("Protein 7 g", "Sugars: 20.5g", "Energy 480
|
||||
# kcal") is also a quantity that is not the pack size. The name and the
|
||||
# number go together, so the pair is removed as one span.
|
||||
_NUTRIENT_VALUE_RE = re.compile(
|
||||
r"\b(?:energy|calories|protein|carbohydrates?|carbs|sugars?|added\s+sugars?|"
|
||||
r"total\s+fat|fat|saturated(?:\s+fat)?|trans(?:\s+fat)?|cholesterol|sodium|salt|"
|
||||
r"dietary\s+fibre|dietary\s+fiber|fibre|fiber|calcium|iron|potassium|vitamin\s*\w*)"
|
||||
r"\s*[:\-]?\s*(?:<\s*)?\d+(?:\.\d+)?\s*(?:kg|mg|mcg|g|kcal|kj|cal|ml|%)?\b",
|
||||
re.I,
|
||||
)
|
||||
# What is left of a nutrition line once the names are gone: values in units
|
||||
# no pack is sold in.
|
||||
_NON_PACK_VALUE_RE = re.compile(r"\b\d+(?:\.\d+)?\s*(?:mg|mcg|kcal|kj|cal)\b|\d+(?:\.\d+)?\s*%", re.I)
|
||||
_URL_RE = re.compile(r"(?:https?://|www\.)\S+|\b\S+\.(?:com|in|org|net)\b", re.I)
|
||||
_PRICE_RE = re.compile(r"(?:₹|rs\.?|inr)\s*\d+(?:[.,]\d+)?", re.I)
|
||||
_SIZE_RE = re.compile(r"^\d+(?:\.\d+)?(?:kg|gms|gm|g|ml|ltr|litre|l|pcs|pc|n)$", re.I)
|
||||
_WORD_OR_SIZE_RE = re.compile(r"\d+(?:\.\d+)?\s*(?:kg|gms|gm|g|ml|ltr|litre|l|pcs|pc|n)\b|[a-z0-9]+(?:[-'&][a-z0-9]+)*", re.I)
|
||||
|
||||
_LEXICAL_TERMS = 2
|
||||
_LONG_NUMBER = 6 # digits; an FSSAI licence is 14, a phone number 10, a barcode 8-13
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# the label
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def clean_label(text: Optional[str]) -> str:
|
||||
"""The label as a query: boilerplate out, sizes kept, repeats folded.
|
||||
|
||||
Spans first (a per-100g basis, a URL, a price), then tokens: a size such
|
||||
as "300 g" is kept whole (as "300 g" - image_match.tokens normalises it
|
||||
later), a word in _NOISE or shorter than two characters is dropped, a
|
||||
repeat keeps its first position. Capped at OCR_MAX_CHARS on a word
|
||||
boundary like the OCR output it usually is.
|
||||
"""
|
||||
if not text:
|
||||
return ""
|
||||
lowered = " ".join(str(text).split())
|
||||
lowered = _PER_BASIS_RE.sub(" ", lowered)
|
||||
lowered = _NUTRIENT_VALUE_RE.sub(" ", lowered)
|
||||
lowered = _NON_PACK_VALUE_RE.sub(" ", lowered)
|
||||
lowered = _URL_RE.sub(" ", lowered)
|
||||
lowered = _PRICE_RE.sub(" ", lowered)
|
||||
|
||||
kept: List[str] = []
|
||||
seen: Set[str] = set()
|
||||
for match in _WORD_OR_SIZE_RE.finditer(lowered):
|
||||
piece = match.group(0)
|
||||
key = piece.lower().replace(" ", "")
|
||||
if _SIZE_RE.match(key):
|
||||
pass # a size stays, whatever its length
|
||||
elif len(key) < 2 or key in _NOISE or key in _STOP:
|
||||
continue
|
||||
elif key.isdigit() and len(key) >= _LONG_NUMBER:
|
||||
continue # a licence, phone or barcode number names no product
|
||||
if key in seen:
|
||||
continue
|
||||
seen.add(key)
|
||||
kept.append(" ".join(piece.split()))
|
||||
|
||||
out = " ".join(kept)
|
||||
if len(out) > OCR_MAX_CHARS:
|
||||
cut = out.rfind(" ", 0, OCR_MAX_CHARS + 1)
|
||||
out = out[:cut if cut > 0 else OCR_MAX_CHARS].rstrip()
|
||||
return out
|
||||
|
||||
|
||||
def _lexical_terms(words: Set[str], brand: Optional[str]) -> List[str]:
|
||||
"""The two longest label words that are not the brand: length is a cheap
|
||||
proxy for distinctiveness ("britannia" > "gold" > "go"), and two ANDed
|
||||
terms survive one misread elsewhere on the label."""
|
||||
brand_words = set(tokens(brand)[0]) if brand else set()
|
||||
candidates = sorted((w for w in words if w not in brand_words and not w.isdigit()),
|
||||
key=lambda w: (-len(w), w))
|
||||
return candidates[:_LEXICAL_TERMS]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# ranking
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def text_extra(words: Set[str], row: Dict[str, Any]) -> int:
|
||||
"""How many words of the row's NAME the label does not account for.
|
||||
|
||||
`text_overlap` is recall - how much of the label the row explains - and
|
||||
on its own it cannot tell "Dairy Milk 50g" from "Dairy Milk Fruit & Nut
|
||||
50g": both explain every word of a "Dairy Milk 50 g" label. The variant
|
||||
carries words the label never said, and this counts them, so the plain
|
||||
product outranks its variants and a "sugar free" or "family pack" row
|
||||
does not win on a label that mentions neither. Only the name and the
|
||||
sizes are looked at, never the description.
|
||||
"""
|
||||
if not words:
|
||||
return 0
|
||||
row_words, _ = tokens(_row_text(row))
|
||||
return len(row_words - words)
|
||||
|
||||
|
||||
def label_rank_key(row: Dict[str, Any]) -> tuple:
|
||||
"""Best first: what the label says, what the row adds, then the vector.
|
||||
|
||||
Overlap outranks cosine because the size token is the only thing that
|
||||
tells the pack sizes of one product apart, and their descriptions - and
|
||||
so their vectors - are near-identical. Unexplained name words come next
|
||||
for the same reason (a variant's vector is as close as the original's).
|
||||
"""
|
||||
return (
|
||||
-float(row.get("text_overlap", 0.0)),
|
||||
-int(bool(row.get("size_in_name", False))), # the label's size in THIS row's name
|
||||
int(row.get("text_extra", 0)),
|
||||
-round(float(row.get("score", 0.0)), 3),
|
||||
str(row.get("product_name") or row.get("title") or "").lower(),
|
||||
str(row.get("image_id") or ""),
|
||||
)
|
||||
|
||||
|
||||
def is_confident(rows: List[Dict[str, Any]]) -> bool:
|
||||
"""Whether the best text row is a find rather than the nearest stranger."""
|
||||
if not rows:
|
||||
return False
|
||||
best = rows[0]
|
||||
try:
|
||||
overlap = float(best.get("text_overlap", 0.0))
|
||||
score = float(best.get("score", 0.0))
|
||||
except (TypeError, ValueError):
|
||||
return False
|
||||
return overlap >= 1.0 or score >= IMAGE_IDENTIFY_MIN_TEXT_SCORE
|
||||
|
||||
|
||||
def _normalise_row(row: Dict[str, Any], scoped_table: Optional[str]) -> Dict[str, Any]:
|
||||
"""Give a store row the keys image_match's rows have.
|
||||
|
||||
Brand tables carry no `brand` column, so semantic_search / lexical_search
|
||||
fill `brand` with the caller's brand when scoped and with the TABLE
|
||||
SUFFIX ("hindustan_unilever") when not; neither adds `brand_table`. The
|
||||
identify response is built by the same card builder as the image rows,
|
||||
so both get the display name and the table here.
|
||||
"""
|
||||
from app.services.vector_store import display_name_for_suffix
|
||||
|
||||
row = dict(row)
|
||||
raw_brand = str(row.get("brand") or "")
|
||||
if scoped_table:
|
||||
row["brand_table"] = scoped_table
|
||||
else:
|
||||
row["brand_table"] = f"brand_{raw_brand}" if raw_brand else ""
|
||||
row["brand"] = display_name_for_suffix(raw_brand) if raw_brand else raw_brand
|
||||
distance = row.get("distance")
|
||||
try:
|
||||
row["score"] = 1.0 - float(distance) if distance is not None else 0.0
|
||||
except (TypeError, ValueError):
|
||||
row["score"] = 0.0
|
||||
return row
|
||||
|
||||
|
||||
def _candidates(
|
||||
query_text: str,
|
||||
embedding: Optional[List[float]],
|
||||
words: Set[str],
|
||||
brand: Optional[str],
|
||||
category: Optional[str],
|
||||
fetch_k: int,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Both arms, merged on (brand_table, image_id); the semantic row wins a
|
||||
tie because it carries the real distance for the full query."""
|
||||
from app.services.vector_store import _table_name, lexical_search, semantic_search
|
||||
|
||||
scoped_table = _table_name(brand) if brand else None
|
||||
merged: Dict[Tuple[str, str], Dict[str, Any]] = {}
|
||||
|
||||
if embedding is not None:
|
||||
try:
|
||||
for row in semantic_search(query_embedding=embedding, brand=brand,
|
||||
top_k=fetch_k, category=category):
|
||||
norm = _normalise_row(row, scoped_table)
|
||||
merged.setdefault((norm["brand_table"], str(norm.get("image_id") or "")), norm)
|
||||
except Exception as exc: # noqa: BLE001 - one arm failing must not lose the other
|
||||
logger.warning("label semantic arm failed: %s", exc)
|
||||
|
||||
terms = _lexical_terms(words, brand)
|
||||
for attempt in (terms, terms[:1]):
|
||||
if not attempt:
|
||||
break
|
||||
try:
|
||||
rows = lexical_search(attempt, brand=brand, limit=fetch_k, category=category,
|
||||
query_embedding=embedding, exact_phrase=" ".join(attempt))
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.warning("label lexical arm failed: %s", exc)
|
||||
rows = []
|
||||
for row in rows:
|
||||
norm = _normalise_row(row, scoped_table)
|
||||
merged.setdefault((norm["brand_table"], str(norm.get("image_id") or "")), norm)
|
||||
if rows:
|
||||
break
|
||||
return list(merged.values())
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# the search
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def resolve_label(
|
||||
text: Optional[str],
|
||||
brand: Optional[str] = None,
|
||||
category: Optional[str] = None,
|
||||
top_k: int = IMAGE_SEARCH_DEFAULT_TOP_K,
|
||||
) -> ImageSearchResult:
|
||||
"""The catalog rows the label text names, best first, at most `top_k`.
|
||||
|
||||
Never raises: no text, no model and no database all give an empty result.
|
||||
`query_text` on the result is the cleaned label the search actually used.
|
||||
"""
|
||||
cleaned = clean_label(text)
|
||||
top_k = max(1, min(int(top_k), IMAGE_SEARCH_MAX_TOP_K))
|
||||
if not cleaned:
|
||||
return ImageSearchResult(query_text=None, top_k=top_k)
|
||||
fetch_k = min(max(top_k * 3, 30), IMAGE_SEARCH_MAX_FETCH_K)
|
||||
|
||||
explicit = (brand or "").strip() or None
|
||||
detected = explicit
|
||||
if detected is None:
|
||||
try:
|
||||
from app.services.query_intent import extract_brand_mention
|
||||
detected = extract_brand_mention(cleaned)
|
||||
except Exception as exc: # noqa: BLE001 - brand detection is an optimisation
|
||||
logger.debug("brand detection skipped: %s", exc)
|
||||
detected = None
|
||||
|
||||
query = cleaned
|
||||
if detected and detected.lower() not in cleaned.lower():
|
||||
query = f"{detected} {cleaned}"
|
||||
|
||||
embedding: Optional[List[float]] = None
|
||||
try:
|
||||
from app.services.embeddings_service import embed_texts
|
||||
vectors = embed_texts([query])
|
||||
embedding = list(vectors[0]) if vectors else None
|
||||
except Exception as exc: # noqa: BLE001 - lexical-only, like catalog_search does
|
||||
logger.warning("label embedding failed: %s. Lexical-only.", exc)
|
||||
embedding = None
|
||||
|
||||
words, sizes = tokens(cleaned)
|
||||
|
||||
def _run(scope: Optional[str]) -> List[Dict[str, Any]]:
|
||||
rows = _candidates(query, embedding, words, scope, category, fetch_k)
|
||||
for row in rows:
|
||||
row["text_overlap"] = text_overlap(words, sizes, row)
|
||||
row["text_extra"] = text_extra(words, row)
|
||||
row["size_in_name"] = size_in_name(sizes, row)
|
||||
rows.sort(key=label_rank_key)
|
||||
return rows
|
||||
|
||||
rows = _run(detected)
|
||||
scoped = detected is not None
|
||||
fallback = False
|
||||
if detected and not explicit and not is_confident(rows):
|
||||
# The OCR named a brand whose table does not hold this label - a
|
||||
# misread, or a sub-brand mapped to the wrong parent. Try everywhere.
|
||||
wider = _run(None)
|
||||
if is_confident(wider) or not rows:
|
||||
rows, scoped, fallback = wider, False, True
|
||||
|
||||
return ImageSearchResult(
|
||||
rows=rows[:top_k],
|
||||
detected_brand=detected,
|
||||
scoped_to_brand=scoped,
|
||||
scope_fallback=fallback,
|
||||
min_score=0.0,
|
||||
query_text=cleaned,
|
||||
top_k=top_k,
|
||||
)
|
||||
@@ -24,8 +24,25 @@ from app.services.discount_service import predict_discounts_for_store
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Every model this module knows how to train. Still the full set accepted by
|
||||
# scripts/train_ml_models.py and POST /api/admin/store-intelligence/train.
|
||||
ALL_MODELS = ["discount", "trending", "popularity", "forecast", "store_performance", "purchase_propensity"]
|
||||
|
||||
# What train_all() trains when the caller does not name a subset.
|
||||
#
|
||||
# `forecast`, `store_performance` and `purchase_propensity` are omitted
|
||||
# deliberately: all three fit cleanly, but nothing reads their output back.
|
||||
# store_db.get_latest_demand_forecast() has zero callers, and neither of the
|
||||
# other two has an inference consumer anywhere in the app or the MCP tools.
|
||||
# Training them by default spent roughly half of every retrain, and ~5MB of
|
||||
# artifacts, producing bundles no request would ever load.
|
||||
#
|
||||
# Their training code, CLI flags and API options are untouched - ask for one by
|
||||
# name (`--models store_performance`) and it trains and is written to
|
||||
# MODEL_ARTIFACTS_DIR exactly as before. Move a model back into this list once
|
||||
# something actually serves it.
|
||||
PRODUCTION_MODELS = ["discount", "trending", "popularity"]
|
||||
|
||||
|
||||
def _load_common_frames():
|
||||
order_items = store_db.get_order_items_df()
|
||||
@@ -144,7 +161,11 @@ def train_purchase_propensity(orders: pd.DataFrame) -> Dict:
|
||||
|
||||
|
||||
def train_all(models: Optional[List[str]] = None) -> Dict[str, Dict]:
|
||||
models = models or ALL_MODELS
|
||||
"""Train `models`, defaulting to the models that are actually served.
|
||||
|
||||
Pass an explicit list (including any of ALL_MODELS) to train more.
|
||||
"""
|
||||
models = models or PRODUCTION_MODELS
|
||||
order_items, orders, store_products = _load_common_frames()
|
||||
results: Dict[str, Dict] = {}
|
||||
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user