Image vector to product details

This commit is contained in:
sriram
2026-09-19 15:39:53 +05:30
parent bc786b1c49
commit f933ea10a1
20 changed files with 2391 additions and 27 deletions

View File

@@ -433,6 +433,42 @@ IMAGE_SEARCH_DEFAULT_MIN_SCORE = float(os.getenv("IMAGE_SEARCH_DEFAULT_MIN_SCORE
# the floor for hnsw.ef_search on that query so the index does not drop them.
IMAGE_SEARCH_MAX_FETCH_K = int(os.getenv("IMAGE_SEARCH_MAX_FETCH_K", "100"))
# ---------------------------------------------------------------------------
# Identify a product from a phone photo - POST /api/search/identify
# (app/services/product_identify.py), with server-side OCR
# (app/services/ocr_service.py) and the label-text resolver
# (app/services/label_match.py). Public, read-only.
# ---------------------------------------------------------------------------
# A phone photo of a pack against the catalog's render of it scored 0.63 on
# img_vector (feasibility test), so the image alone cannot confirm a product.
# Below this floor the ladder falls through to the label text: whatever the
# client OCR'd, else what the server reads off the photo itself.
IMAGE_IDENTIFY_MIN_IMAGE_SCORE = float(os.getenv("IMAGE_IDENTIFY_MIN_IMAGE_SCORE", "0.70"))
# The text side is a MiniLM cosine (1 - embedding <=> q) in a different space
# from the image score. A nearest-neighbour query always returns SOMETHING, so
# a text match counts as found only when the label shares a name/size token
# with the row, or the cosine clears this floor.
IMAGE_IDENTIFY_MIN_TEXT_SCORE = float(os.getenv("IMAGE_IDENTIFY_MIN_TEXT_SCORE", "0.60"))
# Server-side OCR (rapidocr, PP-OCR models on onnxruntime CPU; models ship
# inside the wheel, nothing is downloaded). Loaded lazily on the first photo
# that needs it, never at boot; a missing wheel means "no server OCR" and
# GET /api/health -> ocr says so. tests/conftest.py pins it OFF.
ENABLE_SERVER_OCR = _bool("ENABLE_SERVER_OCR", "true")
# rapidocr's Global.text_score: recognised lines below this confidence are
# dropped before the label is assembled.
OCR_MIN_CONFIDENCE = float(os.getenv("OCR_MIN_CONFIDENCE", "0.5"))
# The photo is downscaled so its longer side is at most this before
# detection. The latency knob: a 12MP capture takes ~3x longer than 1280px
# and reads no better off a pack label.
OCR_MAX_SIDE_PX = int(os.getenv("OCR_MAX_SIDE_PX", "1280"))
# onnxruntime intra-op threads per session; the read itself is serialised by
# a lock (the engine is not thread-safe).
OCR_NUM_THREADS = int(os.getenv("OCR_NUM_THREADS", "2"))
# The assembled label is capped here, on a word boundary. 500 is the limit of
# the `text` field on the image-search routes, so client and server text are
# bounded alike.
OCR_MAX_CHARS = int(os.getenv("OCR_MAX_CHARS", "500"))
# ---------------------------------------------------------------------------
# USDA FoodData Central - nutrition for loose, unbranded commodities
# ---------------------------------------------------------------------------