Image vector to product details

This commit is contained in:
sriram
2026-09-19 15:39:53 +05:30
parent bc786b1c49
commit f933ea10a1
20 changed files with 2391 additions and 27 deletions

View File

@@ -5,12 +5,12 @@ import logging
import requests
from fastapi import APIRouter
from app.api.schemas import AuthConfigOut, HealthOut, ImageVectorsOut
from app.api.schemas import AuthConfigOut, HealthOut, ImageVectorsOut, OcrOut
from app.infrastructure.security import auth_config_summary
from app.infrastructure.settings import (
ENABLE_IMAGE_VECTORS, OLLAMA_BASE_URL, OLLAMA_MODEL_NAME, EMBEDDINGS_MODEL,
)
from app.services import image_embedder
from app.services import image_embedder, ocr_service
from app.services.vector_store import _connect # internal, but handy for a connectivity probe
logger = logging.getLogger(__name__)
@@ -63,4 +63,7 @@ def health() -> HealthOut:
# usual cause (the model file is not in the image), and it must be
# visible from outside the container. status() never loads the model.
image_vectors=ImageVectorsOut(enabled=ENABLE_IMAGE_VECTORS, **image_embedder.status()),
# And for server-side OCR behind /search/identify: "ocr_unavailable"
# in a response has one of three causes, and this names it.
ocr=OcrOut(**ocr_service.status()),
)

View File

@@ -3,10 +3,12 @@ from __future__ import annotations
from typing import Optional
from fastapi import APIRouter, File, Form, HTTPException, Query, UploadFile
from fastapi.responses import JSONResponse
from starlette.concurrency import run_in_threadpool
from app.api.routers.brands import _row_to_product_out
from app.api.schemas import (
IdentifyOut,
ImageMatchOut,
ImageSearchOut,
ImageVectorSearchRequest,
@@ -30,6 +32,7 @@ from app.services.image_match import (
parse_vector_param,
search_by_vector,
)
from app.services.product_identify import IdentifyResult, identify_product
router = APIRouter(tags=["search"])
@@ -101,8 +104,19 @@ def _to_image_search_out(result: ImageSearchResult) -> ImageSearchOut:
)
def _to_identify_out(result: IdentifyResult) -> IdentifyOut:
return IdentifyOut(
**_to_image_search_out(result.search).model_dump(),
matched_by=result.matched_by,
ocr_text=result.ocr_text,
ocr_source=result.ocr_source,
image_top_score=None if result.image_top_score is None else round(float(result.image_top_score), 4),
fallback_reason=result.fallback_reason,
)
@router.post("/search/image-vector", response_model=ImageSearchOut)
def image_vector_search_endpoint(body: ImageVectorSearchRequest) -> ImageSearchOut:
def image_vector_search_endpoint(body: ImageVectorSearchRequest):
"""Products that look like the photo whose embedding is `vector`.
`vector` is the 1024-float, L2-normalised MobileNetV3-Small embedding the
@@ -112,8 +126,22 @@ def image_vector_search_endpoint(body: ImageVectorSearchRequest) -> ImageSearchO
among products that share one photo. An explicit `brand` is a hard
filter; a brand recognised from `text` falls back to every brand when it
finds nothing (`scope_fallback`).
With `text_fallback: true` the request runs the /search/identify ladder
instead: when the best image score is under IMAGE_IDENTIFY_MIN_IMAGE_SCORE
the label `text` is resolved against the catalogue, and the response is
an IdentifyOut (this response plus matched_by, fallback_reason, ...).
Off by default so the app's existing calls are answered exactly as before.
"""
try:
if body.text_fallback:
identified = identify_product(
vector=body.vector, image_bytes=None, text=body.text, brand=body.brand,
category=body.category, top_k=body.top_k, min_score=body.min_score,
)
# A Response bypasses response_model, which is the point: the
# default path keeps its declared ImageSearchOut contract.
return JSONResponse(content=_to_identify_out(identified).model_dump(mode="json"))
result = search_by_vector(
body.vector, text=body.text, brand=body.brand, category=body.category,
top_k=body.top_k, min_score=body.min_score,
@@ -199,3 +227,67 @@ async def image_search_endpoint(
except InvalidVectorError as exc: # cannot happen for a model output, but the route must not 500
raise HTTPException(status_code=422, detail=str(exc))
return _to_image_search_out(result)
# ---------------------------------------------------------------------------
# Identify: image first, label text second
# ---------------------------------------------------------------------------
# A phone photo of a pack scores ~0.63 against the catalog's render of the
# same pack, so /search/image alone cannot confirm a product. This route
# runs the ladder in app/services/product_identify.py: the image match when
# it clears IMAGE_IDENTIFY_MIN_IMAGE_SCORE, otherwise the label text - the
# client's `text`, else what the server reads off the photo (ocr_service) -
# resolved through the catalogue's text embeddings and product names.
@router.post("/search/identify", response_model=IdentifyOut)
async def identify_endpoint(
file: UploadFile = File(..., description="The product photo (JPEG/PNG/WebP), ideally cropped to the pack"),
text: Optional[str] = Form(None, max_length=500,
description="OCR text read off the label; when absent the server reads it"),
brand: Optional[str] = Form(None, max_length=120),
category: Optional[str] = Form(None, max_length=120),
top_k: int = Form(IMAGE_SEARCH_DEFAULT_TOP_K, ge=1, le=IMAGE_SEARCH_MAX_TOP_K),
min_score: float = Form(IMAGE_SEARCH_DEFAULT_MIN_SCORE, ge=-1.0, le=1.0),
) -> IdentifyOut:
"""Which catalog product is in this photo.
`matched_by` says which rung answered and therefore which space each
result's `score` is in; `fallback_reason` is set whenever the answer is
a best effort rather than a confirmed match. Works text-only on a
deployment without the image model (fallback_reason
"image_embedder_unavailable"); 503 only when neither a vector nor server
OCR is possible and no `text` was sent - GET /api/health -> image_vectors
and ocr say which is missing.
"""
content = await file.read()
if not content:
raise HTTPException(status_code=400, detail="The uploaded image is empty.")
if len(content) > IMAGE_VECTOR_MAX_BYTES:
raise HTTPException(
status_code=413,
detail=f"Image is {len(content) / 1_048_576:.1f} MB; the limit is "
f"{IMAGE_VECTOR_MAX_BYTES // 1_048_576} MB. Crop or downscale it.",
)
vector = None
if await run_in_threadpool(image_embedder.available):
vector = await run_in_threadpool(image_embedder.embedding_for_bytes, content)
if vector is None:
raise HTTPException(
status_code=422,
detail="Could not decode the image (unsupported format, corrupt data, or too many pixels).",
)
try:
result = await run_in_threadpool(
identify_product, vector=vector, image_bytes=content, text=text, brand=brand,
category=category, top_k=top_k, min_score=min_score,
)
except InvalidVectorError as exc: # cannot happen for a model output, but the route must not 500
raise HTTPException(status_code=422, detail=str(exc))
if vector is None and result.fallback_reason == "ocr_unavailable":
raise HTTPException(
status_code=503,
detail="Neither the image embedding model nor server-side OCR is available on this "
"deployment. Send the label as `text`, or embed the photo client-side and POST "
"the vector to /api/search/image-vector.",
)
return _to_identify_out(result)

View File

@@ -138,6 +138,17 @@ class ImageVectorsOut(BaseModel):
state: str = "unknown"
class OcrOut(BaseModel):
"""Whether POST /api/search/identify can read a label off a photo itself.
Reported without loading the engine. `runtime_importable=false` after a
deploy means the rapidocr wheel was not installed (requirements-ocr.txt);
`onnxruntime_importable=false` means its engine was not."""
enabled: bool = True
runtime_importable: bool = False
onnxruntime_importable: bool = False
state: str = "unknown"
class HealthOut(BaseModel):
status: str
database: bool
@@ -148,6 +159,7 @@ class HealthOut(BaseModel):
# Defaulted so a client of this schema still validates against a
# deployment predating the field.
image_vectors: ImageVectorsOut = Field(default_factory=ImageVectorsOut)
ocr: OcrOut = Field(default_factory=OcrOut)
# ---------------------------------------------------------------------------
@@ -251,6 +263,16 @@ class ImageVectorSearchRequest(BaseModel):
top_k: int = Field(IMAGE_SEARCH_DEFAULT_TOP_K, ge=1, le=IMAGE_SEARCH_MAX_TOP_K)
min_score: float = Field(IMAGE_SEARCH_DEFAULT_MIN_SCORE, ge=-1.0, le=1.0,
description="Drop matches with cosine similarity below this")
# Opt-in, so a client that never asked for it gets exactly the response
# it always did. With it, the route runs the identify ladder: when the
# best image score is under IMAGE_IDENTIFY_MIN_IMAGE_SCORE, `text` is
# resolved against the catalogue instead, and the response is an
# IdentifyOut (ImageSearchOut plus matched_by / fallback_reason / ...).
text_fallback: bool = Field(
False,
description="Fall back to resolving `text` when the image match is below the identify floor; "
"the response then carries the IdentifyOut fields",
)
@field_validator("vector")
@classmethod
@@ -281,6 +303,23 @@ class ImageSearchOut(BaseModel):
query_text: Optional[str] = None
class IdentifyOut(ImageSearchOut):
"""POST /search/identify: an ImageSearchOut plus which rung answered.
`matched_by` names the space each result's `score` is in: "image_vector"
(cosine of the 1024-d photo embedding) or "text" (cosine of the 384-d
MiniLM embedding of the label). `image_top_score` always carries the
image side. The answer is confirmed when matched_by is "text", or
"image_vector" with no `fallback_reason`; otherwise `fallback_reason`
says why the best effort shown is unconfirmed (see product_identify.py).
"""
matched_by: str = "none" # "image_vector" | "text" | "none"
ocr_text: Optional[str] = None # the label text the ladder used
ocr_source: Optional[str] = None # "client" | "server"
image_top_score: Optional[float] = None
fallback_reason: Optional[str] = None
# ---------------------------------------------------------------------------
# RAG chat
# ---------------------------------------------------------------------------