updates on catalog search and suggestions in backend

This commit is contained in:
sriram
2026-08-20 13:03:16 +05:30
parent e224043e26
commit fbb1356e47
14 changed files with 1177 additions and 11 deletions

View File

@@ -5,25 +5,44 @@ from typing import Optional
from fastapi import APIRouter, Query
from app.api.schemas import SearchOut, SourceProductOut
from app.services.rag_service import retrieve
from app.infrastructure.settings import SEARCH_DEFAULT_TOP_K, SEARCH_MAX_TOP_K
from app.services.catalog_search import search_catalog
router = APIRouter(tags=["search"])
@router.get("/search", response_model=SearchOut)
def semantic_search(
def catalog_search_endpoint(
q: str = Query(..., min_length=1, max_length=500, description="Free-text search query"),
brand: Optional[str] = Query(None, description="Restrict search to a single brand"),
category: Optional[str] = Query(None, description="Restrict search to a category"),
top_k: int = Query(10, ge=1, le=50),
top_k: int = Query(SEARCH_DEFAULT_TOP_K, ge=1, le=SEARCH_MAX_TOP_K),
offset: int = Query(0, ge=0, description="Pagination offset (brand listings only)"),
) -> SearchOut:
"""Pure vector similarity search over the catalog - no LLM call, just
pgvector ranking. This is what powers the instant search-as-you-type
grid in the React 'Search' tab. For a conversational, LLM-generated
answer use POST /api/chat instead."""
results = retrieve(q, brand=brand, top_k=top_k, category=category)
"""Catalog search for the React 'Browse & Search' grid - no LLM call.
Two behaviours, chosen from the query text:
* A bare brand name ("Amul", "Colgate", "coke") returns that brand's whole
catalog - the same rows as GET /api/brands/{brand}/products.
* Anything else ("Amul Butter", "low sugar biscuit") ranks name matches
first, then semantically similar products.
This does NOT share rag_service.retrieve() with /api/chat: that path clamps
results to RAG_MAX_TOP_K to protect the LLM prompt budget, which is why a
brand search used to come back with only 15 products.
For a conversational, LLM-generated answer use POST /api/chat instead.
"""
result = search_catalog(q, brand=brand, category=category, limit=top_k, offset=offset)
return SearchOut(
query=q,
brand=brand,
results=[SourceProductOut(**r.to_dict()) for r in results],
results=[SourceProductOut(**p.to_dict()) for p in result.products],
total=result.total,
limit=result.limit,
offset=result.offset,
match_mode=result.match_mode,
detected_brand=result.detected_brand,
detected_category=result.detected_category,
)

View File

@@ -0,0 +1,35 @@
from __future__ import annotations
from fastapi import APIRouter, Query
from app.api.schemas import SuggestOut, SuggestionOut
from app.infrastructure.settings import SUGGEST_DEFAULT_LIMIT
from app.services.suggest_service import suggest as suggest_service
router = APIRouter(tags=["search"])
@router.get("/suggest", response_model=SuggestOut)
def search_suggest(
q: str = Query(..., min_length=1, max_length=64, description="Partial search text"),
limit: int = Query(SUGGEST_DEFAULT_LIMIT, ge=1, le=20),
) -> SuggestOut:
"""Autocomplete for the catalog search box: brand and category names.
Answers from in-process caches, so it is safe to call on every keystroke.
Product names are deliberately not suggested - brand tables have no
trigram index, so that would mean an unindexed scan per keystroke.
Public, matching GET /api/search.
"""
results = suggest_service(q, limit=limit)
return SuggestOut(
query=q,
suggestions=[
SuggestionOut(
type=s.type, value=s.value, label=s.label,
sublabel=s.sublabel, score=s.score,
)
for s in results
],
)

View File

@@ -126,8 +126,34 @@ class AllProductsOut(BaseModel):
class SearchOut(BaseModel):
query: str
# Echoes the *request* param, as it always has. The brand inferred from the
# query text goes in `detected_brand` instead - repurposing this field would
# break any consumer reading it as "the filter I sent".
brand: Optional[str] = None
results: List[SourceProductOut]
# Exact in brand_catalog mode. None in hybrid mode: the merge happens in
# Python across N brand tables after per-table LIMITs, so there is no cheap
# exact count and inventing one would misreport how much was found.
total: Optional[int] = None
limit: int = 0
offset: int = 0
match_mode: str = "hybrid" # "brand_catalog" | "hybrid"
detected_brand: Optional[str] = None
detected_category: Optional[str] = None
class SuggestionOut(BaseModel):
"""One row in the search box's autocomplete dropdown."""
type: str # "brand" | "category"
value: str # what the search box / filter should use
label: str # display text
sublabel: Optional[str] = None # e.g. "128 products"
score: float = 0.0
class SuggestOut(BaseModel):
query: str
suggestions: List[SuggestionOut]
# ---------------------------------------------------------------------------