Image vector to product details

This commit is contained in:
sriram
2026-09-19 15:39:53 +05:30
parent bc786b1c49
commit f933ea10a1
20 changed files with 2391 additions and 27 deletions

View File

@@ -5,12 +5,12 @@ import logging
import requests
from fastapi import APIRouter
from app.api.schemas import AuthConfigOut, HealthOut, ImageVectorsOut
from app.api.schemas import AuthConfigOut, HealthOut, ImageVectorsOut, OcrOut
from app.infrastructure.security import auth_config_summary
from app.infrastructure.settings import (
ENABLE_IMAGE_VECTORS, OLLAMA_BASE_URL, OLLAMA_MODEL_NAME, EMBEDDINGS_MODEL,
)
from app.services import image_embedder
from app.services import image_embedder, ocr_service
from app.services.vector_store import _connect # internal, but handy for a connectivity probe
logger = logging.getLogger(__name__)
@@ -63,4 +63,7 @@ def health() -> HealthOut:
# usual cause (the model file is not in the image), and it must be
# visible from outside the container. status() never loads the model.
image_vectors=ImageVectorsOut(enabled=ENABLE_IMAGE_VECTORS, **image_embedder.status()),
# And for server-side OCR behind /search/identify: "ocr_unavailable"
# in a response has one of three causes, and this names it.
ocr=OcrOut(**ocr_service.status()),
)

View File

@@ -3,10 +3,12 @@ from __future__ import annotations
from typing import Optional
from fastapi import APIRouter, File, Form, HTTPException, Query, UploadFile
from fastapi.responses import JSONResponse
from starlette.concurrency import run_in_threadpool
from app.api.routers.brands import _row_to_product_out
from app.api.schemas import (
IdentifyOut,
ImageMatchOut,
ImageSearchOut,
ImageVectorSearchRequest,
@@ -30,6 +32,7 @@ from app.services.image_match import (
parse_vector_param,
search_by_vector,
)
from app.services.product_identify import IdentifyResult, identify_product
router = APIRouter(tags=["search"])
@@ -101,8 +104,19 @@ def _to_image_search_out(result: ImageSearchResult) -> ImageSearchOut:
)
def _to_identify_out(result: IdentifyResult) -> IdentifyOut:
return IdentifyOut(
**_to_image_search_out(result.search).model_dump(),
matched_by=result.matched_by,
ocr_text=result.ocr_text,
ocr_source=result.ocr_source,
image_top_score=None if result.image_top_score is None else round(float(result.image_top_score), 4),
fallback_reason=result.fallback_reason,
)
@router.post("/search/image-vector", response_model=ImageSearchOut)
def image_vector_search_endpoint(body: ImageVectorSearchRequest) -> ImageSearchOut:
def image_vector_search_endpoint(body: ImageVectorSearchRequest):
"""Products that look like the photo whose embedding is `vector`.
`vector` is the 1024-float, L2-normalised MobileNetV3-Small embedding the
@@ -112,8 +126,22 @@ def image_vector_search_endpoint(body: ImageVectorSearchRequest) -> ImageSearchO
among products that share one photo. An explicit `brand` is a hard
filter; a brand recognised from `text` falls back to every brand when it
finds nothing (`scope_fallback`).
With `text_fallback: true` the request runs the /search/identify ladder
instead: when the best image score is under IMAGE_IDENTIFY_MIN_IMAGE_SCORE
the label `text` is resolved against the catalogue, and the response is
an IdentifyOut (this response plus matched_by, fallback_reason, ...).
Off by default so the app's existing calls are answered exactly as before.
"""
try:
if body.text_fallback:
identified = identify_product(
vector=body.vector, image_bytes=None, text=body.text, brand=body.brand,
category=body.category, top_k=body.top_k, min_score=body.min_score,
)
# A Response bypasses response_model, which is the point: the
# default path keeps its declared ImageSearchOut contract.
return JSONResponse(content=_to_identify_out(identified).model_dump(mode="json"))
result = search_by_vector(
body.vector, text=body.text, brand=body.brand, category=body.category,
top_k=body.top_k, min_score=body.min_score,
@@ -199,3 +227,67 @@ async def image_search_endpoint(
except InvalidVectorError as exc: # cannot happen for a model output, but the route must not 500
raise HTTPException(status_code=422, detail=str(exc))
return _to_image_search_out(result)
# ---------------------------------------------------------------------------
# Identify: image first, label text second
# ---------------------------------------------------------------------------
# A phone photo of a pack scores ~0.63 against the catalog's render of the
# same pack, so /search/image alone cannot confirm a product. This route
# runs the ladder in app/services/product_identify.py: the image match when
# it clears IMAGE_IDENTIFY_MIN_IMAGE_SCORE, otherwise the label text - the
# client's `text`, else what the server reads off the photo (ocr_service) -
# resolved through the catalogue's text embeddings and product names.
@router.post("/search/identify", response_model=IdentifyOut)
async def identify_endpoint(
file: UploadFile = File(..., description="The product photo (JPEG/PNG/WebP), ideally cropped to the pack"),
text: Optional[str] = Form(None, max_length=500,
description="OCR text read off the label; when absent the server reads it"),
brand: Optional[str] = Form(None, max_length=120),
category: Optional[str] = Form(None, max_length=120),
top_k: int = Form(IMAGE_SEARCH_DEFAULT_TOP_K, ge=1, le=IMAGE_SEARCH_MAX_TOP_K),
min_score: float = Form(IMAGE_SEARCH_DEFAULT_MIN_SCORE, ge=-1.0, le=1.0),
) -> IdentifyOut:
"""Which catalog product is in this photo.
`matched_by` says which rung answered and therefore which space each
result's `score` is in; `fallback_reason` is set whenever the answer is
a best effort rather than a confirmed match. Works text-only on a
deployment without the image model (fallback_reason
"image_embedder_unavailable"); 503 only when neither a vector nor server
OCR is possible and no `text` was sent - GET /api/health -> image_vectors
and ocr say which is missing.
"""
content = await file.read()
if not content:
raise HTTPException(status_code=400, detail="The uploaded image is empty.")
if len(content) > IMAGE_VECTOR_MAX_BYTES:
raise HTTPException(
status_code=413,
detail=f"Image is {len(content) / 1_048_576:.1f} MB; the limit is "
f"{IMAGE_VECTOR_MAX_BYTES // 1_048_576} MB. Crop or downscale it.",
)
vector = None
if await run_in_threadpool(image_embedder.available):
vector = await run_in_threadpool(image_embedder.embedding_for_bytes, content)
if vector is None:
raise HTTPException(
status_code=422,
detail="Could not decode the image (unsupported format, corrupt data, or too many pixels).",
)
try:
result = await run_in_threadpool(
identify_product, vector=vector, image_bytes=content, text=text, brand=brand,
category=category, top_k=top_k, min_score=min_score,
)
except InvalidVectorError as exc: # cannot happen for a model output, but the route must not 500
raise HTTPException(status_code=422, detail=str(exc))
if vector is None and result.fallback_reason == "ocr_unavailable":
raise HTTPException(
status_code=503,
detail="Neither the image embedding model nor server-side OCR is available on this "
"deployment. Send the label as `text`, or embed the photo client-side and POST "
"the vector to /api/search/image-vector.",
)
return _to_identify_out(result)