Image vector to product details
This commit is contained in:
@@ -433,6 +433,42 @@ IMAGE_SEARCH_DEFAULT_MIN_SCORE = float(os.getenv("IMAGE_SEARCH_DEFAULT_MIN_SCORE
|
||||
# the floor for hnsw.ef_search on that query so the index does not drop them.
|
||||
IMAGE_SEARCH_MAX_FETCH_K = int(os.getenv("IMAGE_SEARCH_MAX_FETCH_K", "100"))
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Identify a product from a phone photo - POST /api/search/identify
|
||||
# (app/services/product_identify.py), with server-side OCR
|
||||
# (app/services/ocr_service.py) and the label-text resolver
|
||||
# (app/services/label_match.py). Public, read-only.
|
||||
# ---------------------------------------------------------------------------
|
||||
# A phone photo of a pack against the catalog's render of it scored 0.63 on
|
||||
# img_vector (feasibility test), so the image alone cannot confirm a product.
|
||||
# Below this floor the ladder falls through to the label text: whatever the
|
||||
# client OCR'd, else what the server reads off the photo itself.
|
||||
IMAGE_IDENTIFY_MIN_IMAGE_SCORE = float(os.getenv("IMAGE_IDENTIFY_MIN_IMAGE_SCORE", "0.70"))
|
||||
# The text side is a MiniLM cosine (1 - embedding <=> q) in a different space
|
||||
# from the image score. A nearest-neighbour query always returns SOMETHING, so
|
||||
# a text match counts as found only when the label shares a name/size token
|
||||
# with the row, or the cosine clears this floor.
|
||||
IMAGE_IDENTIFY_MIN_TEXT_SCORE = float(os.getenv("IMAGE_IDENTIFY_MIN_TEXT_SCORE", "0.60"))
|
||||
# Server-side OCR (rapidocr, PP-OCR models on onnxruntime CPU; models ship
|
||||
# inside the wheel, nothing is downloaded). Loaded lazily on the first photo
|
||||
# that needs it, never at boot; a missing wheel means "no server OCR" and
|
||||
# GET /api/health -> ocr says so. tests/conftest.py pins it OFF.
|
||||
ENABLE_SERVER_OCR = _bool("ENABLE_SERVER_OCR", "true")
|
||||
# rapidocr's Global.text_score: recognised lines below this confidence are
|
||||
# dropped before the label is assembled.
|
||||
OCR_MIN_CONFIDENCE = float(os.getenv("OCR_MIN_CONFIDENCE", "0.5"))
|
||||
# The photo is downscaled so its longer side is at most this before
|
||||
# detection. The latency knob: a 12MP capture takes ~3x longer than 1280px
|
||||
# and reads no better off a pack label.
|
||||
OCR_MAX_SIDE_PX = int(os.getenv("OCR_MAX_SIDE_PX", "1280"))
|
||||
# onnxruntime intra-op threads per session; the read itself is serialised by
|
||||
# a lock (the engine is not thread-safe).
|
||||
OCR_NUM_THREADS = int(os.getenv("OCR_NUM_THREADS", "2"))
|
||||
# The assembled label is capped here, on a word boundary. 500 is the limit of
|
||||
# the `text` field on the image-search routes, so client and server text are
|
||||
# bounded alike.
|
||||
OCR_MAX_CHARS = int(os.getenv("OCR_MAX_CHARS", "500"))
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# USDA FoodData Central - nutrition for loose, unbranded commodities
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
Reference in New Issue
Block a user