Files
catalogue_backend/app/api/schemas.py
2026-09-28 15:44:01 +05:30

413 lines
16 KiB
Python

"""Pydantic request/response models for the FastAPI layer."""
from __future__ import annotations
import math
from typing import List, Optional
from pydantic import BaseModel, Field, field_validator
from app.infrastructure.settings import (
IMAGE_SEARCH_DEFAULT_MIN_SCORE,
IMAGE_SEARCH_DEFAULT_TOP_K,
IMAGE_SEARCH_MAX_TOP_K,
)
# ---------------------------------------------------------------------------
# Shared
# ---------------------------------------------------------------------------
class ProductOut(BaseModel):
image_id: str
image_url: Optional[str] = None
image_urls: List[str] = Field(default_factory=list)
brand: str
product_name: str
title: Optional[str] = None
category: Optional[str] = None
description: Optional[str] = None
price_range: Optional[str] = None
size_variants: List[str] = Field(default_factory=list)
providers: List[str] = Field(default_factory=list)
highlights: List[str] = Field(default_factory=list)
nutrients: List[str] = Field(default_factory=list)
fssai_license: Optional[str] = None
product_sku: Optional[str] = None
sku_source: Optional[str] = None
hsn_code: Optional[str] = None
final_selling_price: Optional[float] = None
selling_price: Optional[float] = None
barcode: Optional[str] = None
barcode_type: Optional[str] = None
# Mirrored from nutrition_insights onto the brand table by
# nutrition_score_sync. None means "not scored yet", never "scored zero".
nutrition_score: Optional[float] = None
health_score: Optional[float] = None
class SourceProductOut(BaseModel):
image_id: str
image_url: Optional[str] = None
image_urls: List[str] = Field(default_factory=list)
brand: str
product_name: str
title: Optional[str] = None
category: Optional[str] = None
description: Optional[str] = None
price_range: Optional[str] = None
size_variants: List[str] = Field(default_factory=list)
providers: List[str] = Field(default_factory=list)
highlights: List[str] = Field(default_factory=list)
nutrients: List[str] = Field(default_factory=list)
fssai_license: Optional[str] = None
product_sku: Optional[str] = None
sku_source: Optional[str] = None
hsn_code: Optional[str] = None
final_selling_price: Optional[float] = None
selling_price: Optional[float] = None
barcode: Optional[str] = None
barcode_type: Optional[str] = None
# Mirrored from nutrition_insights onto the brand table by
# nutrition_score_sync. None means "not scored yet", never "scored zero".
nutrition_score: Optional[float] = None
health_score: Optional[float] = None
similarity: float
# ---------------------------------------------------------------------------
# Health
# ---------------------------------------------------------------------------
class ApiKeyInfoOut(BaseModel):
"""One configured machine consumer, named but never quoted.
`fingerprint` is a truncated digest of name+secret, not the secret. It exists
so a caller who was issued a key can confirm THAT key is the one this
deployment loaded - the question a 401 cannot answer, since an undeployed key
and a wrong key fail identically.
"""
name: str
role: str
fingerprint: str
class AuthConfigOut(BaseModel):
"""
The effective auth configuration, reported by /api/health.
Unauthenticated on purpose. The failure this exists to diagnose is "nobody
can sign in", so anything gated behind an admin token is unreachable
exactly when it is needed. Nothing here is a secret: the admin username is
already the documented one, allow_any_login=true is a fact an operator
urgently needs (and an attacker discovers with a single login attempt
anyway), and the fingerprint is a truncated hash of a salted digest, not a
password. The API key block follows the same rule: it names which consumers
are configured and fingerprints their keys, so a caller can tell an
undeployed key from a rejected one, but it never renders a secret. What it buys is a one-command answer to "is this deployment
running the config I think it is?" - compare the fingerprint here against
the one printed by scripts/make_auth_secrets.py --fingerprint.
"""
enabled: bool
allow_any_login: bool
admin_username: str
password_hash_valid: bool
password_hash_iterations: Optional[int] = None
password_hash_fingerprint: str
# "process-env" | "env-file" | "default" - which one actually won.
admin_username_source: str
password_hash_source: str
# Machine consumers. Names and fingerprints only - the secrets themselves are
# never rendered here, and _parse_api_keys enforces enough entropy that the
# fingerprints do not give them away. Defaulted so a client of this schema
# still validates against a deployment predating these fields.
api_keys_count: int = 0
api_keys: List[ApiKeyInfoOut] = Field(default_factory=list)
api_keys_source: str = "default"
class ImageVectorsOut(BaseModel):
"""Why img_vector is (or is not) being filled. Reported without loading
the model. `model_present=false` after a deploy means the .tflite was not
shipped in the image - the one failure this feature absorbs silently."""
enabled: bool = True
model_path: str = ""
model_present: bool = False
runtime_importable: bool = False
state: str = "unknown"
class OcrOut(BaseModel):
"""Whether POST /api/search/identify can read a label off a photo itself.
Reported without loading the engine. `runtime_importable=false` after a
deploy means the rapidocr wheel was not installed (requirements-ocr.txt);
`onnxruntime_importable=false` means its engine was not."""
enabled: bool = True
runtime_importable: bool = False
onnxruntime_importable: bool = False
state: str = "unknown"
class HealthOut(BaseModel):
status: str
database: bool
ollama: bool
ollama_model: str
embeddings_model: str
auth: AuthConfigOut
# Defaulted so a client of this schema still validates against a
# deployment predating the field.
image_vectors: ImageVectorsOut = Field(default_factory=ImageVectorsOut)
ocr: OcrOut = Field(default_factory=OcrOut)
# ---------------------------------------------------------------------------
# Brands / catalog browsing
# ---------------------------------------------------------------------------
class BrandsOut(BaseModel):
brands: List[str]
class BrandCardOut(BaseModel):
"""A brand as shown on the home page card grid.
`name` is the same string GET /api/brands returns, so the frontend can
pass it straight back to /api/brands/{brand}/products.
"""
name: str
slug: str
product_count: int
category_count: int
categories: List[str] = Field(default_factory=list)
image_url: Optional[str] = None
initials: str
class BrandCardsOut(BaseModel):
total_brands: int
total_products: int
brands: List[BrandCardOut]
class CategoriesOut(BaseModel):
brand: str
categories: List[str]
class ProductListOut(BaseModel):
brand: str
total: int
limit: int
offset: int
products: List[ProductOut]
class AllProductsOut(BaseModel):
total: int
limit: int
offset: int
products: List[ProductOut]
# ---------------------------------------------------------------------------
# Semantic search (retrieval only, no LLM generation)
# ---------------------------------------------------------------------------
class SearchOut(BaseModel):
query: str
# Echoes the *request* param, as it always has. The brand inferred from the
# query text goes in `detected_brand` instead - repurposing this field would
# break any consumer reading it as "the filter I sent".
brand: Optional[str] = None
results: List[SourceProductOut]
# Exact in brand_catalog mode. None in hybrid mode: the merge happens in
# Python across N brand tables after per-table LIMITs, so there is no cheap
# exact count and inventing one would misreport how much was found.
total: Optional[int] = None
limit: int = 0
offset: int = 0
match_mode: str = "hybrid" # "brand_catalog" | "hybrid"
detected_brand: Optional[str] = None
detected_category: Optional[str] = None
class SuggestionOut(BaseModel):
"""One row in the search box's autocomplete dropdown."""
type: str # "brand" | "category"
value: str # what the search box / filter should use
label: str # display text
sublabel: Optional[str] = None # e.g. "128 products"
score: float = 0.0
class SuggestOut(BaseModel):
query: str
suggestions: List[SuggestionOut]
# ---------------------------------------------------------------------------
# Image search (POST /api/search/image-vector, POST /api/search/image)
# ---------------------------------------------------------------------------
class ImageVectorSearchRequest(BaseModel):
"""A phone photo's embedding, as the Nearle app computes it on-device."""
vector: List[float] = Field(
..., min_length=1024, max_length=1024,
description="L2-normalised MobileNetV3-Small embedding, 1024 floats",
)
text: Optional[str] = Field(None, max_length=500, description="OCR text read off the label")
brand: Optional[str] = Field(None, max_length=120, description="Restrict to one brand (no fallback)")
category: Optional[str] = Field(None, max_length=120)
top_k: int = Field(IMAGE_SEARCH_DEFAULT_TOP_K, ge=1, le=IMAGE_SEARCH_MAX_TOP_K)
min_score: float = Field(IMAGE_SEARCH_DEFAULT_MIN_SCORE, ge=-1.0, le=1.0,
description="Drop matches with cosine similarity below this")
# With it, the route runs the identify ladder: when the best image score
# is under IMAGE_IDENTIFY_MIN_IMAGE_SCORE (or another photo is within
# IMAGE_SEARCH_MIN_MARGIN of it), `text` is resolved against the
# catalogue instead, and the response is an IdentifyOut (ImageSearchOut
# plus matched_by / fallback_reason / ...). Left out, it is ON whenever
# `text` is sent: a label that names the product must not lose to a 0.5
# cosine, which it did when text only broke exact ties. `false` keeps the
# old image-only ranking.
text_fallback: Optional[bool] = Field(
None,
description="Resolve `text` when the image match cannot be confirmed; the response then "
"carries the IdentifyOut fields. Default: on when `text` is sent. "
"false = image-only ranking, text breaks ties only",
)
@field_validator("vector")
@classmethod
def _finite_and_nonzero(cls, v: List[float]) -> List[float]:
if not all(math.isfinite(x) for x in v):
raise ValueError("vector contains NaN or infinite values")
if math.sqrt(sum(x * x for x in v)) < 1e-6:
raise ValueError("vector is all zeros")
return v
class ImageMatchOut(ProductOut):
"""One catalog product that looks like the photo: the product card plus
how close it is. `score` is cosine similarity (1 - pgvector distance);
`text_overlap` is the label-text tie-break weight, 0 when no text was sent."""
score: float
text_overlap: float = 0.0
class ImageSearchOut(BaseModel):
results: List[ImageMatchOut]
total: int
detected_brand: Optional[str] = None
scoped_to_brand: bool = False
scope_fallback: bool = False
min_score: float = 0.0
top_k: int = 0
query_text: Optional[str] = None
# "confirmed" only when the best match clears IMAGE_IDENTIFY_MIN_IMAGE_SCORE
# AND leads the best different photo by IMAGE_SEARCH_MIN_MARGIN (on
# /identify: when the ladder confirmed it). "low": show the results as a
# list to pick from, never as the answer. "none": no results.
match_confidence: str = "none"
# Best score minus the best DIFFERENT photo's; None when there was no rival.
# Image space only - None when the rows came from the text rung.
margin: Optional[float] = None
class IdentifyOut(ImageSearchOut):
"""POST /search/identify: an ImageSearchOut plus which rung answered.
`matched_by` names the space each result's `score` is in: "image_vector"
(cosine of the 1024-d photo embedding) or "text" (cosine of the 384-d
MiniLM embedding of the label). `image_top_score` always carries the
image side. The answer is confirmed when matched_by is "text", or
"image_vector" with no `fallback_reason`; otherwise `fallback_reason`
says why the best effort shown is unconfirmed (see product_identify.py).
"""
matched_by: str = "none" # "image_vector" | "text" | "label_exact"
# | "discovery_pending" | "none"
ocr_text: Optional[str] = None # the label text the ladder used
ocr_source: Optional[str] = None # "client" | "server"
image_top_score: Optional[float] = None
fallback_reason: Optional[str] = None
# Capture-to-catalog (ENABLE_CAPTURE_DISCOVERY). All None when the answer
# was confirmed or the feature is off. See app/services/capture_discovery.py.
discovery_status: Optional[str] = None # "pending" | "exists" | "needs_input" | "busy"
discovery_job_id: Optional[str] = None # poll GET /api/search/identify/jobs/{id}
discovery_message: Optional[str] = None # one line to show the colleague
provisional: Optional["ProvisionalProductOut"] = None
class ProvisionalProductOut(BaseModel):
"""What the label says, before the pipeline has run. Not a catalog row."""
brand: str
product_name: str
size: Optional[str] = None
category: Optional[str] = None
hsn_code: Optional[str] = None
gst_percent: Optional[float] = None
hsn_gst_needs_review: Optional[bool] = None
visible_in_search: bool = True # False: brand is not in ACTIVE_BRANDS
source: str = "label"
class CaptureJobOut(BaseModel):
"""GET /search/identify/jobs/{job_id}: one capture-to-catalog job."""
job_id: str
status: str # queued | running | done | rejected | failed | interrupted
created_at: float
updated_at: float
provisional: ProvisionalProductOut
product: Optional[ProductOut] = None # the stored row, once done
image_id: Optional[str] = None
disposition: Optional[str] = None # inserted | backfilled | unchanged
validation_status: Optional[str] = None
retail_presence: Optional[dict] = None
photo_used_as_image: bool = False
detail: Optional[str] = None
warnings: List[str] = Field(default_factory=list)
# ---------------------------------------------------------------------------
# RAG chat
# ---------------------------------------------------------------------------
class ChatTurn(BaseModel):
role: str = Field(..., description="'user' or 'assistant'")
content: str
class ChatRequest(BaseModel):
query: str = Field(..., min_length=1, max_length=2000)
brand: Optional[str] = Field(None, description="Restrict retrieval to a single brand")
category: Optional[str] = Field(None, description="Restrict retrieval to a category")
top_k: Optional[int] = Field(None, ge=1, le=15)
history: Optional[List[ChatTurn]] = Field(default=None, description="Prior turns for follow-up questions")
class ChatResponseOut(BaseModel):
answer: str
query: str
brand: Optional[str] = None
detected_category: Optional[str] = Field(
None, description="Product category auto-detected from the query and used to scope retrieval, if any"
)
sources: List[SourceProductOut]
# ---------------------------------------------------------------------------
# Catalog generation (admin/ingestion trigger)
# ---------------------------------------------------------------------------
class CatalogGenerateRequest(BaseModel):
brand: str = Field(..., min_length=1, max_length=120)
max_products: int = Field(50, ge=1, le=300)
class CatalogJobOut(BaseModel):
job_id: str
brand: str
status: str
detail: Optional[str] = None