"""Pydantic request/response models for the FastAPI layer.""" from __future__ import annotations import math from typing import List, Optional from pydantic import BaseModel, Field, field_validator from app.infrastructure.settings import ( IMAGE_SEARCH_DEFAULT_MIN_SCORE, IMAGE_SEARCH_DEFAULT_TOP_K, IMAGE_SEARCH_MAX_TOP_K, ) # --------------------------------------------------------------------------- # Shared # --------------------------------------------------------------------------- class ProductOut(BaseModel): image_id: str image_url: Optional[str] = None image_urls: List[str] = Field(default_factory=list) brand: str product_name: str title: Optional[str] = None category: Optional[str] = None description: Optional[str] = None price_range: Optional[str] = None size_variants: List[str] = Field(default_factory=list) providers: List[str] = Field(default_factory=list) highlights: List[str] = Field(default_factory=list) nutrients: List[str] = Field(default_factory=list) fssai_license: Optional[str] = None product_sku: Optional[str] = None sku_source: Optional[str] = None hsn_code: Optional[str] = None final_selling_price: Optional[float] = None selling_price: Optional[float] = None barcode: Optional[str] = None barcode_type: Optional[str] = None # Mirrored from nutrition_insights onto the brand table by # nutrition_score_sync. None means "not scored yet", never "scored zero". nutrition_score: Optional[float] = None health_score: Optional[float] = None class SourceProductOut(BaseModel): image_id: str image_url: Optional[str] = None image_urls: List[str] = Field(default_factory=list) brand: str product_name: str title: Optional[str] = None category: Optional[str] = None description: Optional[str] = None price_range: Optional[str] = None size_variants: List[str] = Field(default_factory=list) providers: List[str] = Field(default_factory=list) highlights: List[str] = Field(default_factory=list) nutrients: List[str] = Field(default_factory=list) fssai_license: Optional[str] = None product_sku: Optional[str] = None sku_source: Optional[str] = None hsn_code: Optional[str] = None final_selling_price: Optional[float] = None selling_price: Optional[float] = None barcode: Optional[str] = None barcode_type: Optional[str] = None # Mirrored from nutrition_insights onto the brand table by # nutrition_score_sync. None means "not scored yet", never "scored zero". nutrition_score: Optional[float] = None health_score: Optional[float] = None similarity: float # --------------------------------------------------------------------------- # Health # --------------------------------------------------------------------------- class ApiKeyInfoOut(BaseModel): """One configured machine consumer, named but never quoted. `fingerprint` is a truncated digest of name+secret, not the secret. It exists so a caller who was issued a key can confirm THAT key is the one this deployment loaded - the question a 401 cannot answer, since an undeployed key and a wrong key fail identically. """ name: str role: str fingerprint: str class AuthConfigOut(BaseModel): """ The effective auth configuration, reported by /api/health. Unauthenticated on purpose. The failure this exists to diagnose is "nobody can sign in", so anything gated behind an admin token is unreachable exactly when it is needed. Nothing here is a secret: the admin username is already the documented one, allow_any_login=true is a fact an operator urgently needs (and an attacker discovers with a single login attempt anyway), and the fingerprint is a truncated hash of a salted digest, not a password. The API key block follows the same rule: it names which consumers are configured and fingerprints their keys, so a caller can tell an undeployed key from a rejected one, but it never renders a secret. What it buys is a one-command answer to "is this deployment running the config I think it is?" - compare the fingerprint here against the one printed by scripts/make_auth_secrets.py --fingerprint. """ enabled: bool allow_any_login: bool admin_username: str password_hash_valid: bool password_hash_iterations: Optional[int] = None password_hash_fingerprint: str # "process-env" | "env-file" | "default" - which one actually won. admin_username_source: str password_hash_source: str # Machine consumers. Names and fingerprints only - the secrets themselves are # never rendered here, and _parse_api_keys enforces enough entropy that the # fingerprints do not give them away. Defaulted so a client of this schema # still validates against a deployment predating these fields. api_keys_count: int = 0 api_keys: List[ApiKeyInfoOut] = Field(default_factory=list) api_keys_source: str = "default" class ImageVectorsOut(BaseModel): """Why img_vector is (or is not) being filled. Reported without loading the model. `model_present=false` after a deploy means the .tflite was not shipped in the image - the one failure this feature absorbs silently.""" enabled: bool = True model_path: str = "" model_present: bool = False runtime_importable: bool = False state: str = "unknown" class OcrOut(BaseModel): """Whether POST /api/search/identify can read a label off a photo itself. Reported without loading the engine. `runtime_importable=false` after a deploy means the rapidocr wheel was not installed (requirements-ocr.txt); `onnxruntime_importable=false` means its engine was not.""" enabled: bool = True runtime_importable: bool = False onnxruntime_importable: bool = False state: str = "unknown" class HealthOut(BaseModel): status: str database: bool ollama: bool ollama_model: str embeddings_model: str auth: AuthConfigOut # Defaulted so a client of this schema still validates against a # deployment predating the field. image_vectors: ImageVectorsOut = Field(default_factory=ImageVectorsOut) ocr: OcrOut = Field(default_factory=OcrOut) # --------------------------------------------------------------------------- # Brands / catalog browsing # --------------------------------------------------------------------------- class BrandsOut(BaseModel): brands: List[str] class BrandCardOut(BaseModel): """A brand as shown on the home page card grid. `name` is the same string GET /api/brands returns, so the frontend can pass it straight back to /api/brands/{brand}/products. """ name: str slug: str product_count: int category_count: int categories: List[str] = Field(default_factory=list) image_url: Optional[str] = None initials: str class BrandCardsOut(BaseModel): total_brands: int total_products: int brands: List[BrandCardOut] class CategoriesOut(BaseModel): brand: str categories: List[str] class ProductListOut(BaseModel): brand: str total: int limit: int offset: int products: List[ProductOut] class AllProductsOut(BaseModel): total: int limit: int offset: int products: List[ProductOut] # --------------------------------------------------------------------------- # Semantic search (retrieval only, no LLM generation) # --------------------------------------------------------------------------- class SearchOut(BaseModel): query: str # Echoes the *request* param, as it always has. The brand inferred from the # query text goes in `detected_brand` instead - repurposing this field would # break any consumer reading it as "the filter I sent". brand: Optional[str] = None results: List[SourceProductOut] # Exact in brand_catalog mode. None in hybrid mode: the merge happens in # Python across N brand tables after per-table LIMITs, so there is no cheap # exact count and inventing one would misreport how much was found. total: Optional[int] = None limit: int = 0 offset: int = 0 match_mode: str = "hybrid" # "brand_catalog" | "hybrid" detected_brand: Optional[str] = None detected_category: Optional[str] = None class SuggestionOut(BaseModel): """One row in the search box's autocomplete dropdown.""" type: str # "brand" | "category" value: str # what the search box / filter should use label: str # display text sublabel: Optional[str] = None # e.g. "128 products" score: float = 0.0 class SuggestOut(BaseModel): query: str suggestions: List[SuggestionOut] # --------------------------------------------------------------------------- # Image search (POST /api/search/image-vector, POST /api/search/image) # --------------------------------------------------------------------------- class ImageVectorSearchRequest(BaseModel): """A phone photo's embedding, as the Nearle app computes it on-device.""" vector: List[float] = Field( ..., min_length=1024, max_length=1024, description="L2-normalised MobileNetV3-Small embedding, 1024 floats", ) text: Optional[str] = Field(None, max_length=500, description="OCR text read off the label") brand: Optional[str] = Field(None, max_length=120, description="Restrict to one brand (no fallback)") category: Optional[str] = Field(None, max_length=120) top_k: int = Field(IMAGE_SEARCH_DEFAULT_TOP_K, ge=1, le=IMAGE_SEARCH_MAX_TOP_K) min_score: float = Field(IMAGE_SEARCH_DEFAULT_MIN_SCORE, ge=-1.0, le=1.0, description="Drop matches with cosine similarity below this") # With it, the route runs the identify ladder: when the best image score # is under IMAGE_IDENTIFY_MIN_IMAGE_SCORE (or another photo is within # IMAGE_SEARCH_MIN_MARGIN of it), `text` is resolved against the # catalogue instead, and the response is an IdentifyOut (ImageSearchOut # plus matched_by / fallback_reason / ...). Left out, it is ON whenever # `text` is sent: a label that names the product must not lose to a 0.5 # cosine, which it did when text only broke exact ties. `false` keeps the # old image-only ranking. text_fallback: Optional[bool] = Field( None, description="Resolve `text` when the image match cannot be confirmed; the response then " "carries the IdentifyOut fields. Default: on when `text` is sent. " "false = image-only ranking, text breaks ties only", ) @field_validator("vector") @classmethod def _finite_and_nonzero(cls, v: List[float]) -> List[float]: if not all(math.isfinite(x) for x in v): raise ValueError("vector contains NaN or infinite values") if math.sqrt(sum(x * x for x in v)) < 1e-6: raise ValueError("vector is all zeros") return v class ImageMatchOut(ProductOut): """One catalog product that looks like the photo: the product card plus how close it is. `score` is cosine similarity (1 - pgvector distance); `text_overlap` is the label-text tie-break weight, 0 when no text was sent.""" score: float text_overlap: float = 0.0 class ImageSearchOut(BaseModel): results: List[ImageMatchOut] total: int detected_brand: Optional[str] = None scoped_to_brand: bool = False scope_fallback: bool = False min_score: float = 0.0 top_k: int = 0 query_text: Optional[str] = None # "confirmed" only when the best match clears IMAGE_IDENTIFY_MIN_IMAGE_SCORE # AND leads the best different photo by IMAGE_SEARCH_MIN_MARGIN (on # /identify: when the ladder confirmed it). "low": show the results as a # list to pick from, never as the answer. "none": no results. match_confidence: str = "none" # Best score minus the best DIFFERENT photo's; None when there was no rival. # Image space only - None when the rows came from the text rung. margin: Optional[float] = None class IdentifyOut(ImageSearchOut): """POST /search/identify: an ImageSearchOut plus which rung answered. `matched_by` names the space each result's `score` is in: "image_vector" (cosine of the 1024-d photo embedding) or "text" (cosine of the 384-d MiniLM embedding of the label). `image_top_score` always carries the image side. The answer is confirmed when matched_by is "text", or "image_vector" with no `fallback_reason`; otherwise `fallback_reason` says why the best effort shown is unconfirmed (see product_identify.py). """ matched_by: str = "none" # "image_vector" | "text" | "label_exact" # | "discovery_pending" | "none" ocr_text: Optional[str] = None # the label text the ladder used ocr_source: Optional[str] = None # "client" | "server" image_top_score: Optional[float] = None fallback_reason: Optional[str] = None # Capture-to-catalog (ENABLE_CAPTURE_DISCOVERY). All None when the answer # was confirmed or the feature is off. See app/services/capture_discovery.py. discovery_status: Optional[str] = None # "pending" | "exists" | "needs_input" | "busy" discovery_job_id: Optional[str] = None # poll GET /api/search/identify/jobs/{id} discovery_message: Optional[str] = None # one line to show the colleague provisional: Optional["ProvisionalProductOut"] = None class ProvisionalProductOut(BaseModel): """What the label says, before the pipeline has run. Not a catalog row.""" brand: str product_name: str size: Optional[str] = None category: Optional[str] = None hsn_code: Optional[str] = None gst_percent: Optional[float] = None hsn_gst_needs_review: Optional[bool] = None visible_in_search: bool = True # False: brand is not in ACTIVE_BRANDS source: str = "label" class CaptureJobOut(BaseModel): """GET /search/identify/jobs/{job_id}: one capture-to-catalog job.""" job_id: str status: str # queued | running | done | rejected | failed | interrupted created_at: float updated_at: float provisional: ProvisionalProductOut product: Optional[ProductOut] = None # the stored row, once done image_id: Optional[str] = None disposition: Optional[str] = None # inserted | backfilled | unchanged validation_status: Optional[str] = None retail_presence: Optional[dict] = None photo_used_as_image: bool = False detail: Optional[str] = None warnings: List[str] = Field(default_factory=list) # --------------------------------------------------------------------------- # RAG chat # --------------------------------------------------------------------------- class ChatTurn(BaseModel): role: str = Field(..., description="'user' or 'assistant'") content: str class ChatRequest(BaseModel): query: str = Field(..., min_length=1, max_length=2000) brand: Optional[str] = Field(None, description="Restrict retrieval to a single brand") category: Optional[str] = Field(None, description="Restrict retrieval to a category") top_k: Optional[int] = Field(None, ge=1, le=15) history: Optional[List[ChatTurn]] = Field(default=None, description="Prior turns for follow-up questions") class ChatResponseOut(BaseModel): answer: str query: str brand: Optional[str] = None detected_category: Optional[str] = Field( None, description="Product category auto-detected from the query and used to scope retrieval, if any" ) sources: List[SourceProductOut] # --------------------------------------------------------------------------- # Catalog generation (admin/ingestion trigger) # --------------------------------------------------------------------------- class CatalogGenerateRequest(BaseModel): brand: str = Field(..., min_length=1, max_length=120) max_products: int = Field(50, ge=1, le=300) class CatalogJobOut(BaseModel): job_id: str brand: str status: str detail: Optional[str] = None