Electronics Catalog: API, MCP server, frontend and deployment

Verified catalogue of mobiles and laptops sold in India, collected from
real retail listings (FastAPI backend, React frontend, Postgres/pgvector).

- REST API under /api/elec (read-only catalogue; admin endpoints need login)
- MCP server (FastMCP) at /mcp/ with list_categories, search_products,
  get_product and price_history tools
- Real ratings and reviews read from product pages and search results
- Production Dockerfile (requirements-api.txt, no PyTorch) and
  .env.production.example; remote database only via an explicit
  ELEC_ALLOW_REMOTE_DB host/name allowlist
- docs/API.md: endpoint and MCP reference with live examples

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
sriram
2026-10-01 12:17:42 +05:30
commit c7e4d59188
115 changed files with 14329 additions and 0 deletions

View File

View File

@@ -0,0 +1,31 @@
"""
Minimal in-process background job dispatcher for long-running admin jobs
(catalog ingestion, store seeding, ML model training, nutrition
enrichment).
This deliberately does NOT use Starlette's `BackgroundTasks`. BackgroundTasks
run *synchronously after the response is sent*: an async background task is
awaited directly on the server's event loop, and a sync one is awaited in the
request's thread. Either way the request handler does not return until the job
finishes. For jobs that take minutes (LLM calls, web scraping, ML training,
Open Food Facts lookups), that turns a "kick off a job and return 202" endpoint
into a blocking call and, for async tasks, freezes the whole API event loop for
the duration.
A daemon thread returns control to the caller immediately, and the job's
progress stays visible via the job_store polling endpoints the UI already
uses. Daemon threads are a deliberate, documented trade-off (see
`app/api/job_store.py`): state is process-local and not safe across multiple
uvicorn workers - fine for this project's intended single-process, CPU-only
deployment.
"""
from __future__ import annotations
import threading
from typing import Any, Callable
def run_in_background(func: Callable[[], Any], *, name: str) -> None:
"""Start `func` on a new daemon thread and return immediately."""
thread = threading.Thread(target=func, name=name, daemon=True)
thread.start()

144
backend/app/api/deps.py Normal file
View File

@@ -0,0 +1,144 @@
"""
Request-scoped authentication dependencies.
Guards are attached per route, not as middleware matching on paths. Two
reasons that matters here:
* A path-matching middleware silently stops guarding a route the moment
somebody renames it. A ``Depends`` on the route function cannot drift out
of sync with the route it protects.
* FastAPI reflects these into the OpenAPI schema, so ``/docs`` shows which
operations need a credential instead of implying everything is open.
The guard therefore holds regardless of which host the request arrives on -
through the frontend's nginx on ``{$DOMAIN}``, or directly on ``api.{$DOMAIN}``.
Usage::
@router.post("/thing", dependencies=[Depends(require_admin)])
def create_thing(): ...
@router.post("/other", dependencies=[Depends(require_permission("add_product"))])
def other_thing(): ...
@router.post("/who", ...)
def who(principal: Principal = Depends(get_principal)): ...
"""
from __future__ import annotations
from typing import Callable, Optional
from fastapi import Depends, HTTPException, status
from fastapi.security import APIKeyHeader, HTTPAuthorizationCredentials, HTTPBearer
from app.infrastructure.security import (
AuthError,
Principal,
anonymous_principal,
decode_access_token,
principal_for_api_key,
)
from app.infrastructure.settings import AUTH_ENABLED
# auto_error=False on both: with two accepted credential types, letting either
# scheme raise on its own would reject a request that carried the *other* one.
# get_principal decides, once it has seen both.
_bearer_scheme = HTTPBearer(auto_error=False, description="Access token from POST /api/auth/login")
_api_key_scheme = APIKeyHeader(
name="X-API-Key",
auto_error=False,
description="Static key for machine consumers (see API_KEYS)",
)
_UNAUTHENTICATED = HTTPException(
status_code=status.HTTP_401_UNAUTHORIZED,
detail="Not authenticated. Send a bearer token from POST /api/auth/login, or an X-API-Key header.",
headers={"WWW-Authenticate": "Bearer"},
)
def get_principal(
credentials: Optional[HTTPAuthorizationCredentials] = Depends(_bearer_scheme),
api_key: Optional[str] = Depends(_api_key_scheme),
) -> Principal:
"""Resolve the caller, or raise 401. Use this to require *any* valid credential."""
if not AUTH_ENABLED:
return anonymous_principal()
if credentials is not None and credentials.credentials:
try:
return decode_access_token(credentials.credentials)
except AuthError as exc:
raise HTTPException(
status_code=status.HTTP_401_UNAUTHORIZED,
detail=str(exc),
headers={"WWW-Authenticate": "Bearer"},
) from exc
if api_key:
try:
return principal_for_api_key(api_key)
except AuthError as exc:
raise HTTPException(
status_code=status.HTTP_401_UNAUTHORIZED, detail=str(exc)
) from exc
raise _UNAUTHENTICATED
def get_optional_principal(
credentials: Optional[HTTPAuthorizationCredentials] = Depends(_bearer_scheme),
api_key: Optional[str] = Depends(_api_key_scheme),
) -> Optional[Principal]:
"""
Resolve the caller if they presented a valid credential, else None.
For endpoints that are public but behave differently when signed in. A
credential that is present but *invalid* still raises - failing open there
would mean a typo'd token silently downgrades to anonymous access.
"""
if not AUTH_ENABLED:
return anonymous_principal()
if credentials is None and not api_key:
return None
return get_principal(credentials, api_key)
def require_role(*roles: str) -> Callable[[Principal], Principal]:
"""Require the caller to hold one of ``roles``."""
allowed = frozenset(roles)
def _dependency(principal: Principal = Depends(get_principal)) -> Principal:
if principal.role not in allowed:
raise HTTPException(
status_code=status.HTTP_403_FORBIDDEN,
detail=(
f"This operation requires the {' or '.join(sorted(allowed))} role; "
f"you are signed in as '{principal.role}'."
),
)
return principal
return _dependency
def require_permission(permission: str) -> Callable[[Principal], Principal]:
"""
Require a specific permission. ``admin`` passes every check - see
``Principal.has_permission``.
"""
def _dependency(principal: Principal = Depends(get_principal)) -> Principal:
if not principal.has_permission(permission):
raise HTTPException(
status_code=status.HTTP_403_FORBIDDEN,
detail=f"This operation requires the '{permission}' permission.",
)
return principal
return _dependency
# The two guards used most often, named so route decorators stay readable.
require_admin = require_role("admin")
require_authenticated = get_principal

View File

@@ -0,0 +1,59 @@
"""
Tiny in-memory job tracker for background catalog-generation tasks.
Deliberately not a queue/Celery/Redis setup - the original project already
had celery+redis in requirements.txt but nothing wired it up, and adding a
broker is unnecessary operational weight for a single-developer, CPU-only
project. A process-local dict is enough to let the React UI show
"running -> done/failed" status for a brand ingestion job started from the
admin panel.
NOTE: state is lost on server restart, and is per-process (not safe for
multiple uvicorn workers). For this project's intended scale (one backend
process on a personal machine) that's a fine trade-off; see the docs'
"Scaling beyond a single machine" section if this ever needs to change.
"""
from __future__ import annotations
import threading
import time
import uuid
from dataclasses import dataclass, field
from typing import Dict, Optional
@dataclass
class Job:
job_id: str
brand: str
status: str = "pending" # pending -> running -> done | failed
detail: Optional[str] = None
created_at: float = field(default_factory=time.time)
updated_at: float = field(default_factory=time.time)
class JobStore:
def __init__(self) -> None:
self._jobs: Dict[str, Job] = {}
self._lock = threading.Lock()
def create(self, brand: str) -> Job:
job = Job(job_id=str(uuid.uuid4()), brand=brand)
with self._lock:
self._jobs[job.job_id] = job
return job
def update(self, job_id: str, status: str, detail: Optional[str] = None) -> None:
with self._lock:
job = self._jobs.get(job_id)
if job:
job.status = status
job.detail = detail
job.updated_at = time.time()
def get(self, job_id: str) -> Optional[Job]:
with self._lock:
return self._jobs.get(job_id)
job_store = JobStore()

View File

View File

@@ -0,0 +1,357 @@
"""
Authentication router - issues and inspects access tokens.
This replaces an earlier version that returned a role profile without issuing
anything, accepted an empty password, and granted `admin` to any username that
asked for the role. It decided which buttons the UI drew; it protected nothing.
Now the token this returns is the credential every write endpoint checks (see
app/api/deps.py), so the rules hold for curl and partner scripts too, not just
for the React app.
Accounts come from the environment - two of them, admin and user, configured as
PBKDF2 digests. That is deliberately not a user database: this project has no
user table, no registration flow and no password reset, and inventing one here
would be a bigger change than the problem calls for. Machine consumers get
API_KEYS instead. If per-user accounts become a real requirement, this module
is the seam to replace.
For local work there is AUTH_ALLOW_ANY_LOGIN, which skips the password check
here and nowhere else - the token still gets signed and every guard downstream
still checks it. It is off by default and logs a warning at startup when on.
"""
from __future__ import annotations
import logging
import threading
import time
from typing import Dict, List, Tuple
from fastapi import APIRouter, Depends, HTTPException, Request, status
from pydantic import BaseModel, Field
from app.api.deps import get_principal
from app.infrastructure.security import (
ROLE_PERMISSIONS,
Principal,
create_access_token,
hash_is_wellformed,
password_hash_fingerprint,
verify_password,
)
from app.infrastructure.settings import (
AUTH_ADMIN_PASSWORD_HASH,
AUTH_ADMIN_USERNAME,
AUTH_ALLOW_ANY_LOGIN,
AUTH_ENABLED,
AUTH_LOCKOUT_SECONDS,
AUTH_MAX_LOGIN_ATTEMPTS,
AUTH_USER_PASSWORD_HASH,
AUTH_USER_USERNAME,
config_source,
)
logger = logging.getLogger(__name__)
router = APIRouter(prefix="/auth", tags=["auth"])
if AUTH_ENABLED and AUTH_ALLOW_ANY_LOGIN:
logger.warning(
"AUTH_ALLOW_ANY_LOGIN=true: /api/auth/login accepts ANY password, so anyone "
"who can reach this port can sign in as admin. Local development only - "
"set it to false in backend/.env before exposing this server."
)
class LoginRequest(BaseModel):
username: str = Field(min_length=1, max_length=150)
password: str = Field(min_length=1, max_length=1024)
class UserProfile(BaseModel):
username: str
role: str
display_name: str
email: str
permissions: List[str] = Field(default_factory=list)
class LoginResponse(BaseModel):
access_token: str
token_type: str = "bearer"
expires_in: int = Field(description="Token lifetime in seconds")
user: UserProfile
# A syntactically valid hash of an unguessable value. Never matches any real
# password; it exists only so the unknown-username path in login() does the
# same PBKDF2 work as the known one, keeping the two indistinguishable by timing.
_DUMMY_HASH = (
"pbkdf2_sha256$600000$YWJjZGVmZ2hpamtsbW5vcA==$"
"S1cVFrGD4pDkGqSjbEbaVSTONzGhCT9BOaWPQ2vwvvA="
)
def _accounts() -> Dict[str, dict]:
"""
The configured accounts, read per call so a settings reload is picked up.
Usernames are compared case-insensitively (matching what the login form
sends), but the password is not touched - the previous version lowercased
it before comparing, which silently shrank the effective keyspace.
An account with a blank password hash is omitted entirely rather than
included with an unmatchable digest. Both spellings deny the login, but
only omission keeps it out of the account table, so nothing downstream can
treat it as a real account. This is how the optional `user` account is
switched off: leave AUTH_USER_PASSWORD_HASH unset and only `admin` exists.
"""
accounts = {
AUTH_ADMIN_USERNAME.lower(): {
"password_hash": AUTH_ADMIN_PASSWORD_HASH,
"role": "admin",
"display_name": "System Administrator",
"email": "admin@nutritionintel.com",
},
}
if AUTH_USER_PASSWORD_HASH:
accounts[AUTH_USER_USERNAME.lower()] = {
"password_hash": AUTH_USER_PASSWORD_HASH,
"role": "user",
"display_name": "Product & Store Manager",
"email": "user@nutritionintel.com",
}
return accounts
# ---------------------------------------------------------------------------
# Failed-login throttle
# ---------------------------------------------------------------------------
# In-process and per-worker: with several uvicorn workers a determined attacker
# gets AUTH_MAX_LOGIN_ATTEMPTS per worker, not overall. That is a real limit,
# not a rounding error - but it still turns an unbounded password oracle into a
# rate-limited one without adding Redis to the deployment. Move this to a shared
# store if you ever run many workers.
_failures: Dict[Tuple[str, str], Tuple[int, float]] = {}
_failures_lock = threading.Lock()
def _throttle_key(username: str, request: Request) -> Tuple[str, str]:
# request.client.host is the real client IP because uvicorn runs with
# --proxy-headers behind nginx/Caddy (see backend/Dockerfile); without that
# every request would appear to come from the proxy and share one bucket.
client = request.client.host if request.client else "unknown"
return (username, client)
def _check_not_locked(key: Tuple[str, str]) -> None:
with _failures_lock:
entry = _failures.get(key)
if entry is None:
return
count, first_seen = entry
if time.time() - first_seen > AUTH_LOCKOUT_SECONDS:
del _failures[key]
return
if count >= AUTH_MAX_LOGIN_ATTEMPTS:
retry_after = int(AUTH_LOCKOUT_SECONDS - (time.time() - first_seen))
raise HTTPException(
status_code=status.HTTP_429_TOO_MANY_REQUESTS,
detail=f"Too many failed sign-in attempts. Try again in {retry_after}s.",
headers={"Retry-After": str(max(retry_after, 1))},
)
def _record_failure(key: Tuple[str, str]) -> None:
now = time.time()
with _failures_lock:
count, first_seen = _failures.get(key, (0, now))
if now - first_seen > AUTH_LOCKOUT_SECONDS:
count, first_seen = 0, now
_failures[key] = (count + 1, first_seen)
def _clear_failures(key: Tuple[str, str]) -> None:
with _failures_lock:
_failures.pop(key, None)
# ---------------------------------------------------------------------------
# Routes
# ---------------------------------------------------------------------------
@router.post("/login", response_model=LoginResponse)
def login(payload: LoginRequest, request: Request) -> LoginResponse:
"""Exchange a username and password for an access token."""
if not AUTH_ENABLED:
raise HTTPException(
status_code=status.HTTP_503_SERVICE_UNAVAILABLE,
detail=(
"Authentication is disabled on this server (AUTH_ENABLED=false), so no "
"token can be issued. Every endpoint is open; sign-in is not required."
),
)
username = payload.username.strip().lower()
key = _throttle_key(username, request)
if AUTH_ALLOW_ANY_LOGIN:
# Dev bypass: any password gets in. The username still picks the
# account, so `admin` lands on the admin pages and `user` on the user
# ones; anything else is an unconfigured name and gets the lower of the
# two roles rather than silently minting an admin. Throttling is skipped
# because there is no longer a password to guess.
account = _accounts().get(username) or {
"role": "user",
"display_name": payload.username.strip() or username,
"email": f"{username}@nutritionintel.com",
}
logger.warning(
"AUTH_ALLOW_ANY_LOGIN: signing in %r as %s without checking the password",
username,
account["role"],
)
else:
_check_not_locked(key)
account = _accounts().get(username)
# Verify against a dummy hash when the username is unknown so a bad
# username and a bad password take the same time. Otherwise the response
# latency alone enumerates valid usernames.
stored_hash = account["password_hash"] if account else _DUMMY_HASH
# A hash that does not parse can never match, and verify_password bails
# out of one before doing any PBKDF2 work - measured here, 0.16ms against
# 439ms for a real digest. That inverts the very property _DUMMY_HASH
# exists to protect: an account whose configured hash is corrupt would
# answer ~2700x faster than every other username, announcing which
# account is broken to anyone with a stopwatch. So spend the same work
# regardless; the result is a rejection either way.
hash_usable = hash_is_wellformed(stored_hash)
password_ok = verify_password(
payload.password, stored_hash if hash_usable else _DUMMY_HASH
)
if account is None or not password_ok:
_record_failure(key)
# The reason goes to the LOG, never to the caller - the response
# below is byte-identical whichever of these it was, so nothing here
# can be used to enumerate usernames. It is computed after both the
# lookup and the PBKDF2 call above, so it adds no timing signal
# either. Without it, a deployment whose configured hash or admin
# username has drifted is indistinguishable from someone simply
# typing the wrong password, and this is exactly how a production
# sign-in outage stayed unexplained: the log said "Failed sign-in
# for 'admin'" and nothing more.
if account is None:
logger.warning(
"Failed sign-in for %r from %s: reason=unknown-username. "
"Configured accounts: %s (AUTH_ADMIN_USERNAME source=%s).",
username,
key[1],
", ".join(sorted(_accounts())),
config_source("AUTH_ADMIN_USERNAME"),
)
elif not hash_usable:
# ERROR, not WARNING: this is a broken deployment, not a bad
# guess. No password can ever match, so every sign-in to this
# account will 401 until the hash itself is replaced.
logger.error(
"Failed sign-in for %r from %s: reason=malformed-hash. The configured "
"password hash does not parse as pbkdf2_sha256$<iterations>$<b64 salt>$"
"<b64 digest> (fingerprint=%s, source=%s). Nobody can sign in to this "
"account until it is regenerated with scripts/make_auth_secrets.py.",
username,
key[1],
password_hash_fingerprint(stored_hash) or "(empty)",
config_source("AUTH_ADMIN_PASSWORD_HASH"),
)
else:
logger.warning(
"Failed sign-in for %r from %s: reason=bad-password. The account exists "
"and its hash parses (fingerprint=%s, source=%s); the password did not "
"match. If this IS the password you deployed, then the running config "
"carries a different hash than the file you are reading - compare that "
"fingerprint against: python scripts/make_auth_secrets.py "
"--fingerprint .env.production",
username,
key[1],
password_hash_fingerprint(stored_hash),
config_source("AUTH_ADMIN_PASSWORD_HASH"),
)
# One message for every failure mode, for the same reason.
raise HTTPException(
status_code=status.HTTP_401_UNAUTHORIZED,
detail="Invalid username or password.",
)
_clear_failures(key)
role = account["role"]
permissions = ROLE_PERMISSIONS.get(role, [])
token, expires_in = create_access_token(username, role, permissions)
logger.info("Issued token for %r (role=%s)", username, role)
return LoginResponse(
access_token=token,
expires_in=expires_in,
user=UserProfile(
username=username,
role=role,
display_name=account["display_name"],
email=account["email"],
permissions=permissions,
),
)
@router.get("/me", response_model=UserProfile)
def me(principal: Principal = Depends(get_principal)) -> UserProfile:
"""
Who the presented credential belongs to. 401 if it is missing or expired.
The frontend calls this on boot to check a restored session before showing
the app, so an expired token lands on the login page rather than on a
dashboard whose every request then fails.
"""
account = _accounts().get(principal.username, {})
return UserProfile(
username=principal.username,
role=principal.role,
display_name=account.get("display_name", principal.username.title()),
email=account.get("email", f"{principal.username}@nutritionintel.com"),
permissions=principal.permissions,
)
@router.get("/roles")
def list_roles() -> dict:
"""
The available roles and what each may do.
Note there are no demo credentials here any more. The passwords are set per
deployment via AUTH_ADMIN_PASSWORD_HASH / AUTH_USER_PASSWORD_HASH; this
endpoint used to publish working ones to anyone who asked.
"""
return {
"roles": [
{
"id": "admin",
"name": "Admin",
"description": (
"Full access: catalog brand cards, project details, Excel/CSV "
"train/test uploads, discount allocation, analytics and nutrition. "
"Implicitly holds every permission."
),
"permissions": ROLE_PERMISSIONS["admin"],
},
{
"id": "user",
"name": "User",
"description": (
"Combined user and store role: single or batch product uploads with "
"image and DB/JSON sync, store inventory, profit analytics, nutrition."
),
"permissions": ROLE_PERMISSIONS["user"],
},
]
}

View File

@@ -0,0 +1,188 @@
"""Read-only catalogue API: category -> brand -> product -> per-platform offers.
Only VERIFIED products are served (see repository.refresh_verification), and
every price is returned with the site, URL, source type and time it was seen.
Money is returned as a decimal string, never a float.
"""
from __future__ import annotations
from decimal import Decimal
from typing import Any, Dict, List, Optional
from fastapi import APIRouter, HTTPException, Query
from app.electronics.db.connection import connect
from app.electronics.db.repository import product_rating_and_reviews
from app.electronics.reviews import select_reviews
router = APIRouter(prefix="/elec", tags=["electronics"])
def _money(value: Optional[Decimal]) -> Optional[str]:
return None if value is None else format(value, "f")
def _clean(row: Dict[str, Any]) -> Dict[str, Any]:
out = {}
for k, v in row.items():
if isinstance(v, Decimal):
out[k] = _money(v) if k in ("price", "mrp", "best_price", "min_price", "max_price") else float(v)
elif hasattr(v, "isoformat"):
out[k] = v.isoformat()
else:
out[k] = v
return out
@router.get("/categories")
def categories() -> List[dict]:
with connect() as conn:
rows = conn.execute(
"SELECT c.slug, c.name, coalesce(sum(s.product_count), 0)::int AS product_count "
"FROM elec.category c LEFT JOIN elec.v_brand_summary s ON s.category = c.slug "
"GROUP BY c.slug, c.name ORDER BY c.name"
).fetchall()
return [_clean(r) for r in rows]
@router.get("/brands")
def brands(category: str = Query(...)) -> List[dict]:
with connect() as conn:
rows = conn.execute(
"""
SELECT b.name AS brand, b.slug AS brand_slug,
coalesce(s.product_count, 0)::int AS product_count, s.min_price, s.max_price,
(SELECT image_url FROM elec.v_brand_catalog v
WHERE v.brand_slug = b.slug AND v.category = %(c)s AND v.image_url IS NOT NULL
ORDER BY v.platform_count DESC LIMIT 1) AS sample_image
FROM elec.brand b
JOIN elec.brand_category bc ON bc.brand_id = b.id
JOIN elec.category c ON c.id = bc.category_id AND c.slug = %(c)s
LEFT JOIN elec.v_brand_summary s ON s.brand_slug = b.slug AND s.category = %(c)s
ORDER BY coalesce(s.product_count, 0) DESC, b.name
""",
{"c": category},
).fetchall()
return [_clean(r) for r in rows]
@router.get("/products")
def products(
category: Optional[str] = None,
brand: Optional[str] = None,
q: Optional[str] = Query(None, max_length=100),
min_price: Optional[Decimal] = None,
max_price: Optional[Decimal] = None,
in_stock: bool = False,
site: Optional[str] = None,
tn_only: bool = False,
limit: int = Query(48, ge=1, le=200),
offset: int = Query(0, ge=0),
) -> dict:
where, params = ["TRUE"], {}
if category:
where.append("v.category = %(category)s"); params["category"] = category
if brand:
where.append("v.brand_slug = %(brand)s"); params["brand"] = brand
if q:
where.append("(v.display_name ILIKE %(q)s OR v.brand ILIKE %(q)s)"); params["q"] = f"%{q}%"
if min_price is not None:
where.append("v.best_price >= %(min_price)s"); params["min_price"] = min_price
if max_price is not None:
where.append("v.best_price <= %(max_price)s"); params["max_price"] = max_price
if in_stock:
where.append("EXISTS (SELECT 1 FROM elec.v_product_availability a WHERE a.product_id = v.product_id AND a.in_stock)")
if site:
where.append("EXISTS (SELECT 1 FROM elec.v_product_availability a WHERE a.product_id = v.product_id AND a.domain = %(site)s)")
params["site"] = site
if tn_only:
where.append("v.sold_by_tn_retailer")
sql_where = " AND ".join(where)
with connect() as conn:
total = conn.execute(f"SELECT count(*) AS n FROM elec.v_brand_catalog v WHERE {sql_where}", params).fetchone()["n"]
rows = conn.execute(
f"SELECT v.* FROM elec.v_brand_catalog v WHERE {sql_where} "
f"ORDER BY v.platform_count DESC, v.best_price NULLS LAST, v.display_name "
f"LIMIT %(limit)s OFFSET %(offset)s",
{**params, "limit": limit, "offset": offset},
).fetchall()
return {"total": total, "products": [_clean(r) for r in rows]}
@router.get("/products/{product_id}")
def product(product_id: int) -> dict:
with connect() as conn:
row = conn.execute("SELECT * FROM elec.v_brand_catalog WHERE product_id = %s", (product_id,)).fetchone()
if not row:
raise HTTPException(status_code=404, detail="Product not found or not verified")
specs = conn.execute("SELECT spec_sources FROM elec.product WHERE id = %s", (product_id,)).fetchone()
offers = conn.execute(
"SELECT * FROM elec.v_product_availability WHERE product_id = %s "
"ORDER BY (price IS NULL), (source_type = 'search_snippet'), price, site",
(product_id,),
).fetchall()
images = conn.execute(
"SELECT i.url, i.source_type, s.name AS site, l.source_url AS found_on "
"FROM elec.product_image i JOIN elec.source_listing l ON l.id = i.source_listing_id "
"JOIN elec.site s ON s.id = l.site_id WHERE i.product_id = %s ORDER BY i.rank, i.id",
(product_id,),
).fetchall()
rated = product_rating_and_reviews(conn, product_id)
result = _clean(row)
result["spec_sources"] = specs["spec_sources"] if specs else {}
result["offers"] = [_clean(o) for o in offers]
result["images"] = [dict(i) for i in images]
result["rating"] = _overall_rating(rated["sources"])
overall = result["rating"]["value"] if result["rating"] else None
result["reviews"] = [_clean(r) for r in select_reviews(overall, rated["reviews"])]
return result
def _overall_rating(sources: List[dict]) -> Optional[dict]:
"""The product's rating across the platforms that state one: the mean
weighted by each platform's rating count (a platform that states no count
weighs as 1). None when no platform states a rating - never a guess."""
if not sources:
return None
weight = lambda s: max(int(s["review_count"] or 0), 1) # noqa: E731
total = sum(weight(s) for s in sources)
value = sum(Decimal(s["rating"]) * weight(s) for s in sources) / total
counts = [s["review_count"] for s in sources if s["review_count"]]
return {
"value": round(float(value), 1),
"count": sum(counts) if counts else None,
"sources": [
{"site": s["site"], "rating": float(s["rating"]), "review_count": s["review_count"], "source_url": s["source_url"]}
for s in sources
],
}
@router.get("/products/{product_id}/price-history")
def price_history(product_id: int) -> List[dict]:
with connect() as conn:
rows = conn.execute(
"""
SELECT s.name AS site, h.price, h.mrp, h.in_stock, h.source_type, h.observed_at
FROM elec.price_history h
JOIN elec.product_listing_map m ON m.listing_id = h.listing_id AND m.review_status IN ('auto','approved')
JOIN elec.source_listing l ON l.id = h.listing_id
JOIN elec.site s ON s.id = l.site_id
WHERE m.product_id = %s AND h.price IS NOT NULL
ORDER BY h.observed_at
""",
(product_id,),
).fetchall()
return [_clean(r) for r in rows]
@router.get("/sites")
def sites() -> List[dict]:
with connect() as conn:
rows = conn.execute(
"SELECT s.name, s.domain, s.kind, s.region, s.policy, s.probe_outcome, s.probed_at, "
"s.breaker_until, s.breaker_reason, s.probe_evidence->>'reason' AS probe_reason, "
"(SELECT count(*) FROM elec.source_listing l WHERE l.site_id = s.id)::int AS listings "
"FROM elec.site s ORDER BY (s.kind = 'brand_official'), s.name"
).fetchall()
return [_clean(r) for r in rows]

View File

@@ -0,0 +1,124 @@
"""Admin endpoints: start a collection run, see runs, probe sites, review matches."""
from __future__ import annotations
import threading
import time
import uuid
from typing import Dict, List, Optional
from fastapi import APIRouter, Depends, HTTPException
from pydantic import BaseModel, Field
from app.api.background import run_in_background
from app.api.deps import require_admin
from app.electronics.db import repository as repo
from app.electronics.reference import load_reference
router = APIRouter(prefix="/elec/admin", tags=["electronics-admin"], dependencies=[Depends(require_admin)])
_jobs: Dict[str, dict] = {}
_jobs_lock = threading.Lock()
_run_lock = threading.Lock() # one collection at a time: polite to sites, kind to 8 GB RAM
class RunRequest(BaseModel):
category: str = Field(..., examples=["mobiles"])
brands: List[str] = Field(default_factory=list, description="brand slugs; empty = all for the category")
limit: int = Field(10, ge=1, le=60, description="max models per brand")
expand: int = Field(6, ge=0, le=30)
budget: int = Field(150, ge=10, le=1000, description="max search queries")
fetch_pages: bool = True
use_llm: bool = True
def _job_update(job_id: str, **fields) -> None:
with _jobs_lock:
_jobs[job_id].update(fields, updated_at=time.time())
@router.post("/runs", status_code=202)
def start_run(req: RunRequest) -> dict:
ref = load_reference()
if req.category not in ref.categories:
raise HTTPException(400, f"unknown category {req.category!r}")
brands = req.brands or [b.slug for b in ref.brands_for(req.category)]
bad = [b for b in brands if b not in ref.brands or req.category not in ref.brands[b].categories]
if bad:
raise HTTPException(400, f"not allow-listed for {req.category}: {bad}")
if _run_lock.locked():
raise HTTPException(409, "A collection run is already in progress")
job_id = str(uuid.uuid4())
with _jobs_lock:
_jobs[job_id] = {"job_id": job_id, "status": "queued", "log": [], "stats": {}, "created_at": time.time()}
def work() -> None:
from app.electronics.collector import Collector, RunOptions
with _run_lock:
_job_update(job_id, status="running")
def progress(msg: str) -> None:
with _jobs_lock:
_jobs[job_id]["log"] = (_jobs[job_id]["log"] + [msg])[-200:]
try:
opts = RunOptions(category=req.category, brands=brands, max_products_per_brand=req.limit,
expand_per_brand=req.expand, search_budget=req.budget,
fetch_pages=req.fetch_pages, use_llm=req.use_llm)
stats = Collector(opts, progress=progress).run()
_job_update(job_id, status="done", stats=stats)
except Exception as exc: # noqa: BLE001 - reported to the UI
_job_update(job_id, status="failed", error=repr(exc))
run_in_background(work, name=f"elec-run-{job_id[:8]}")
return {"job_id": job_id}
@router.get("/runs/{job_id}")
def get_job(job_id: str) -> dict:
with _jobs_lock:
job = _jobs.get(job_id)
if not job:
raise HTTPException(404, "unknown job")
return dict(job)
@router.get("/runs")
def list_runs(limit: int = 20) -> List[dict]:
rows = repo.recent_runs(limit)
for r in rows:
for k in ("started_at", "ended_at"):
if r.get(k):
r[k] = r[k].isoformat()
return rows
@router.get("/review")
def review_queue() -> List[dict]:
return [{**r, "confidence": float(r["confidence"])} for r in repo.review_queue()]
class ReviewDecision(BaseModel):
approve: bool
@router.post("/review/{listing_id}")
def review(listing_id: int, decision: ReviewDecision) -> dict:
if not repo.set_review(listing_id, decision.approve):
raise HTTPException(404, "no pending match for that listing")
return {"ok": True, "products": repo.refresh_verification()}
@router.post("/sites/{domain}/probe")
def probe(domain: str) -> dict:
from app.electronics.net.polite_client import PoliteClient
from app.electronics.probe.site_probe import probe_site
from app.electronics.search.engine import SearchEngine
site = load_reference().sites.get(domain)
if not site:
raise HTTPException(404, "unknown site")
with PoliteClient() as client:
res = probe_site(site, client, SearchEngine(budget=4))
repo.set_probe_result(domain, res["outcome"], res["robots_allowed"], res["evidence"])
return res

View File

@@ -0,0 +1,36 @@
from __future__ import annotations
import logging
from fastapi import APIRouter
from app.api.schemas import AuthConfigOut, HealthOut, SearchStatusOut
from app.electronics.db.connection import check_connection
from app.infrastructure.security import auth_config_summary
from app.infrastructure.settings import (
DB_NAME, EMBEDDINGS_MODEL, OLLAMA_MODEL_NAME, USE_DDG_SEARCH, USE_GOOGLE_CSE,
)
from app.services import ollama_service
logger = logging.getLogger(__name__)
router = APIRouter(tags=["health"])
@router.get("/health", response_model=HealthOut)
def health() -> HealthOut:
"""Liveness/readiness probe used by the React app to show a banner when
Postgres or Ollama aren't reachable, instead of failing silently.
Ollama being down does not make the service degraded: it is only used to
fill spec gaps, and the pipeline runs deterministically without it."""
db_ok = check_connection()
return HealthOut(
status="ok" if db_ok else "degraded",
database=db_ok,
database_name=DB_NAME,
ollama=bool(ollama_service._ensure_client()),
ollama_model=OLLAMA_MODEL_NAME,
embeddings_model=EMBEDDINGS_MODEL,
search=SearchStatusOut(ddg=USE_DDG_SEARCH, google_cse=USE_GOOGLE_CSE),
auth=AuthConfigOut(**auth_config_summary()),
)

View File

@@ -0,0 +1,73 @@
"""Pydantic response models for the health endpoint. The electronics
catalogue's own models live in app/electronics/api_models.py."""
from __future__ import annotations
from typing import List, Optional
from pydantic import BaseModel, Field
class ApiKeyInfoOut(BaseModel):
"""One configured machine consumer, named but never quoted.
`fingerprint` is a truncated digest of name+secret, not the secret. It exists
so a caller who was issued a key can confirm THAT key is the one this
deployment loaded - the question a 401 cannot answer, since an undeployed key
and a wrong key fail identically.
"""
name: str
role: str
fingerprint: str
class AuthConfigOut(BaseModel):
"""
The effective auth configuration, reported by /api/health.
Unauthenticated on purpose. The failure this exists to diagnose is "nobody
can sign in", so anything gated behind an admin token is unreachable
exactly when it is needed. Nothing here is a secret: the admin username is
already the documented one, allow_any_login=true is a fact an operator
urgently needs (and an attacker discovers with a single login attempt
anyway), and the fingerprint is a truncated hash of a salted digest, not a
password. The API key block follows the same rule: it names which consumers
are configured and fingerprints their keys, so a caller can tell an
undeployed key from a rejected one, but it never renders a secret. What it buys is a one-command answer to "is this deployment
running the config I think it is?" - compare the fingerprint here against
the one printed by scripts/make_auth_secrets.py --fingerprint.
"""
enabled: bool
allow_any_login: bool
admin_username: str
password_hash_valid: bool
password_hash_iterations: Optional[int] = None
password_hash_fingerprint: str
# "process-env" | "env-file" | "default" - which one actually won.
admin_username_source: str
password_hash_source: str
# Machine consumers. Names and fingerprints only - the secrets themselves are
# never rendered here, and _parse_api_keys enforces enough entropy that the
# fingerprints do not give them away. Defaulted so a client of this schema
# still validates against a deployment predating these fields.
api_keys_count: int = 0
api_keys: List[ApiKeyInfoOut] = Field(default_factory=list)
api_keys_source: str = "default"
class SearchStatusOut(BaseModel):
"""Which web-search providers discovery can use right now."""
ddg: bool
google_cse: bool
class HealthOut(BaseModel):
status: str
database: bool
database_name: str
ollama: bool
ollama_model: str
embeddings_model: str
search: SearchStatusOut
auth: AuthConfigOut