Image vector to product details
This commit is contained in:
@@ -5,12 +5,12 @@ import logging
|
||||
import requests
|
||||
from fastapi import APIRouter
|
||||
|
||||
from app.api.schemas import AuthConfigOut, HealthOut, ImageVectorsOut
|
||||
from app.api.schemas import AuthConfigOut, HealthOut, ImageVectorsOut, OcrOut
|
||||
from app.infrastructure.security import auth_config_summary
|
||||
from app.infrastructure.settings import (
|
||||
ENABLE_IMAGE_VECTORS, OLLAMA_BASE_URL, OLLAMA_MODEL_NAME, EMBEDDINGS_MODEL,
|
||||
)
|
||||
from app.services import image_embedder
|
||||
from app.services import image_embedder, ocr_service
|
||||
from app.services.vector_store import _connect # internal, but handy for a connectivity probe
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -63,4 +63,7 @@ def health() -> HealthOut:
|
||||
# usual cause (the model file is not in the image), and it must be
|
||||
# visible from outside the container. status() never loads the model.
|
||||
image_vectors=ImageVectorsOut(enabled=ENABLE_IMAGE_VECTORS, **image_embedder.status()),
|
||||
# And for server-side OCR behind /search/identify: "ocr_unavailable"
|
||||
# in a response has one of three causes, and this names it.
|
||||
ocr=OcrOut(**ocr_service.status()),
|
||||
)
|
||||
|
||||
@@ -3,10 +3,12 @@ from __future__ import annotations
|
||||
from typing import Optional
|
||||
|
||||
from fastapi import APIRouter, File, Form, HTTPException, Query, UploadFile
|
||||
from fastapi.responses import JSONResponse
|
||||
from starlette.concurrency import run_in_threadpool
|
||||
|
||||
from app.api.routers.brands import _row_to_product_out
|
||||
from app.api.schemas import (
|
||||
IdentifyOut,
|
||||
ImageMatchOut,
|
||||
ImageSearchOut,
|
||||
ImageVectorSearchRequest,
|
||||
@@ -30,6 +32,7 @@ from app.services.image_match import (
|
||||
parse_vector_param,
|
||||
search_by_vector,
|
||||
)
|
||||
from app.services.product_identify import IdentifyResult, identify_product
|
||||
|
||||
router = APIRouter(tags=["search"])
|
||||
|
||||
@@ -101,8 +104,19 @@ def _to_image_search_out(result: ImageSearchResult) -> ImageSearchOut:
|
||||
)
|
||||
|
||||
|
||||
def _to_identify_out(result: IdentifyResult) -> IdentifyOut:
|
||||
return IdentifyOut(
|
||||
**_to_image_search_out(result.search).model_dump(),
|
||||
matched_by=result.matched_by,
|
||||
ocr_text=result.ocr_text,
|
||||
ocr_source=result.ocr_source,
|
||||
image_top_score=None if result.image_top_score is None else round(float(result.image_top_score), 4),
|
||||
fallback_reason=result.fallback_reason,
|
||||
)
|
||||
|
||||
|
||||
@router.post("/search/image-vector", response_model=ImageSearchOut)
|
||||
def image_vector_search_endpoint(body: ImageVectorSearchRequest) -> ImageSearchOut:
|
||||
def image_vector_search_endpoint(body: ImageVectorSearchRequest):
|
||||
"""Products that look like the photo whose embedding is `vector`.
|
||||
|
||||
`vector` is the 1024-float, L2-normalised MobileNetV3-Small embedding the
|
||||
@@ -112,8 +126,22 @@ def image_vector_search_endpoint(body: ImageVectorSearchRequest) -> ImageSearchO
|
||||
among products that share one photo. An explicit `brand` is a hard
|
||||
filter; a brand recognised from `text` falls back to every brand when it
|
||||
finds nothing (`scope_fallback`).
|
||||
|
||||
With `text_fallback: true` the request runs the /search/identify ladder
|
||||
instead: when the best image score is under IMAGE_IDENTIFY_MIN_IMAGE_SCORE
|
||||
the label `text` is resolved against the catalogue, and the response is
|
||||
an IdentifyOut (this response plus matched_by, fallback_reason, ...).
|
||||
Off by default so the app's existing calls are answered exactly as before.
|
||||
"""
|
||||
try:
|
||||
if body.text_fallback:
|
||||
identified = identify_product(
|
||||
vector=body.vector, image_bytes=None, text=body.text, brand=body.brand,
|
||||
category=body.category, top_k=body.top_k, min_score=body.min_score,
|
||||
)
|
||||
# A Response bypasses response_model, which is the point: the
|
||||
# default path keeps its declared ImageSearchOut contract.
|
||||
return JSONResponse(content=_to_identify_out(identified).model_dump(mode="json"))
|
||||
result = search_by_vector(
|
||||
body.vector, text=body.text, brand=body.brand, category=body.category,
|
||||
top_k=body.top_k, min_score=body.min_score,
|
||||
@@ -199,3 +227,67 @@ async def image_search_endpoint(
|
||||
except InvalidVectorError as exc: # cannot happen for a model output, but the route must not 500
|
||||
raise HTTPException(status_code=422, detail=str(exc))
|
||||
return _to_image_search_out(result)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Identify: image first, label text second
|
||||
# ---------------------------------------------------------------------------
|
||||
# A phone photo of a pack scores ~0.63 against the catalog's render of the
|
||||
# same pack, so /search/image alone cannot confirm a product. This route
|
||||
# runs the ladder in app/services/product_identify.py: the image match when
|
||||
# it clears IMAGE_IDENTIFY_MIN_IMAGE_SCORE, otherwise the label text - the
|
||||
# client's `text`, else what the server reads off the photo (ocr_service) -
|
||||
# resolved through the catalogue's text embeddings and product names.
|
||||
|
||||
@router.post("/search/identify", response_model=IdentifyOut)
|
||||
async def identify_endpoint(
|
||||
file: UploadFile = File(..., description="The product photo (JPEG/PNG/WebP), ideally cropped to the pack"),
|
||||
text: Optional[str] = Form(None, max_length=500,
|
||||
description="OCR text read off the label; when absent the server reads it"),
|
||||
brand: Optional[str] = Form(None, max_length=120),
|
||||
category: Optional[str] = Form(None, max_length=120),
|
||||
top_k: int = Form(IMAGE_SEARCH_DEFAULT_TOP_K, ge=1, le=IMAGE_SEARCH_MAX_TOP_K),
|
||||
min_score: float = Form(IMAGE_SEARCH_DEFAULT_MIN_SCORE, ge=-1.0, le=1.0),
|
||||
) -> IdentifyOut:
|
||||
"""Which catalog product is in this photo.
|
||||
|
||||
`matched_by` says which rung answered and therefore which space each
|
||||
result's `score` is in; `fallback_reason` is set whenever the answer is
|
||||
a best effort rather than a confirmed match. Works text-only on a
|
||||
deployment without the image model (fallback_reason
|
||||
"image_embedder_unavailable"); 503 only when neither a vector nor server
|
||||
OCR is possible and no `text` was sent - GET /api/health -> image_vectors
|
||||
and ocr say which is missing.
|
||||
"""
|
||||
content = await file.read()
|
||||
if not content:
|
||||
raise HTTPException(status_code=400, detail="The uploaded image is empty.")
|
||||
if len(content) > IMAGE_VECTOR_MAX_BYTES:
|
||||
raise HTTPException(
|
||||
status_code=413,
|
||||
detail=f"Image is {len(content) / 1_048_576:.1f} MB; the limit is "
|
||||
f"{IMAGE_VECTOR_MAX_BYTES // 1_048_576} MB. Crop or downscale it.",
|
||||
)
|
||||
vector = None
|
||||
if await run_in_threadpool(image_embedder.available):
|
||||
vector = await run_in_threadpool(image_embedder.embedding_for_bytes, content)
|
||||
if vector is None:
|
||||
raise HTTPException(
|
||||
status_code=422,
|
||||
detail="Could not decode the image (unsupported format, corrupt data, or too many pixels).",
|
||||
)
|
||||
try:
|
||||
result = await run_in_threadpool(
|
||||
identify_product, vector=vector, image_bytes=content, text=text, brand=brand,
|
||||
category=category, top_k=top_k, min_score=min_score,
|
||||
)
|
||||
except InvalidVectorError as exc: # cannot happen for a model output, but the route must not 500
|
||||
raise HTTPException(status_code=422, detail=str(exc))
|
||||
if vector is None and result.fallback_reason == "ocr_unavailable":
|
||||
raise HTTPException(
|
||||
status_code=503,
|
||||
detail="Neither the image embedding model nor server-side OCR is available on this "
|
||||
"deployment. Send the label as `text`, or embed the photo client-side and POST "
|
||||
"the vector to /api/search/image-vector.",
|
||||
)
|
||||
return _to_identify_out(result)
|
||||
|
||||
Reference in New Issue
Block a user