Electronics Catalog: API, MCP server, frontend and deployment

Verified catalogue of mobiles and laptops sold in India, collected from
real retail listings (FastAPI backend, React frontend, Postgres/pgvector).

- REST API under /api/elec (read-only catalogue; admin endpoints need login)
- MCP server (FastMCP) at /mcp/ with list_categories, search_products,
  get_product and price_history tools
- Real ratings and reviews read from product pages and search results
- Production Dockerfile (requirements-api.txt, no PyTorch) and
  .env.production.example; remote database only via an explicit
  ELEC_ALLOW_REMOTE_DB host/name allowlist
- docs/API.md: endpoint and MCP reference with live examples

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
sriram
2026-10-01 12:17:42 +05:30
commit c7e4d59188
115 changed files with 14329 additions and 0 deletions

0
backend/app/__init__.py Normal file
View File

View File

View File

@@ -0,0 +1,31 @@
"""
Minimal in-process background job dispatcher for long-running admin jobs
(catalog ingestion, store seeding, ML model training, nutrition
enrichment).
This deliberately does NOT use Starlette's `BackgroundTasks`. BackgroundTasks
run *synchronously after the response is sent*: an async background task is
awaited directly on the server's event loop, and a sync one is awaited in the
request's thread. Either way the request handler does not return until the job
finishes. For jobs that take minutes (LLM calls, web scraping, ML training,
Open Food Facts lookups), that turns a "kick off a job and return 202" endpoint
into a blocking call and, for async tasks, freezes the whole API event loop for
the duration.
A daemon thread returns control to the caller immediately, and the job's
progress stays visible via the job_store polling endpoints the UI already
uses. Daemon threads are a deliberate, documented trade-off (see
`app/api/job_store.py`): state is process-local and not safe across multiple
uvicorn workers - fine for this project's intended single-process, CPU-only
deployment.
"""
from __future__ import annotations
import threading
from typing import Any, Callable
def run_in_background(func: Callable[[], Any], *, name: str) -> None:
"""Start `func` on a new daemon thread and return immediately."""
thread = threading.Thread(target=func, name=name, daemon=True)
thread.start()

144
backend/app/api/deps.py Normal file
View File

@@ -0,0 +1,144 @@
"""
Request-scoped authentication dependencies.
Guards are attached per route, not as middleware matching on paths. Two
reasons that matters here:
* A path-matching middleware silently stops guarding a route the moment
somebody renames it. A ``Depends`` on the route function cannot drift out
of sync with the route it protects.
* FastAPI reflects these into the OpenAPI schema, so ``/docs`` shows which
operations need a credential instead of implying everything is open.
The guard therefore holds regardless of which host the request arrives on -
through the frontend's nginx on ``{$DOMAIN}``, or directly on ``api.{$DOMAIN}``.
Usage::
@router.post("/thing", dependencies=[Depends(require_admin)])
def create_thing(): ...
@router.post("/other", dependencies=[Depends(require_permission("add_product"))])
def other_thing(): ...
@router.post("/who", ...)
def who(principal: Principal = Depends(get_principal)): ...
"""
from __future__ import annotations
from typing import Callable, Optional
from fastapi import Depends, HTTPException, status
from fastapi.security import APIKeyHeader, HTTPAuthorizationCredentials, HTTPBearer
from app.infrastructure.security import (
AuthError,
Principal,
anonymous_principal,
decode_access_token,
principal_for_api_key,
)
from app.infrastructure.settings import AUTH_ENABLED
# auto_error=False on both: with two accepted credential types, letting either
# scheme raise on its own would reject a request that carried the *other* one.
# get_principal decides, once it has seen both.
_bearer_scheme = HTTPBearer(auto_error=False, description="Access token from POST /api/auth/login")
_api_key_scheme = APIKeyHeader(
name="X-API-Key",
auto_error=False,
description="Static key for machine consumers (see API_KEYS)",
)
_UNAUTHENTICATED = HTTPException(
status_code=status.HTTP_401_UNAUTHORIZED,
detail="Not authenticated. Send a bearer token from POST /api/auth/login, or an X-API-Key header.",
headers={"WWW-Authenticate": "Bearer"},
)
def get_principal(
credentials: Optional[HTTPAuthorizationCredentials] = Depends(_bearer_scheme),
api_key: Optional[str] = Depends(_api_key_scheme),
) -> Principal:
"""Resolve the caller, or raise 401. Use this to require *any* valid credential."""
if not AUTH_ENABLED:
return anonymous_principal()
if credentials is not None and credentials.credentials:
try:
return decode_access_token(credentials.credentials)
except AuthError as exc:
raise HTTPException(
status_code=status.HTTP_401_UNAUTHORIZED,
detail=str(exc),
headers={"WWW-Authenticate": "Bearer"},
) from exc
if api_key:
try:
return principal_for_api_key(api_key)
except AuthError as exc:
raise HTTPException(
status_code=status.HTTP_401_UNAUTHORIZED, detail=str(exc)
) from exc
raise _UNAUTHENTICATED
def get_optional_principal(
credentials: Optional[HTTPAuthorizationCredentials] = Depends(_bearer_scheme),
api_key: Optional[str] = Depends(_api_key_scheme),
) -> Optional[Principal]:
"""
Resolve the caller if they presented a valid credential, else None.
For endpoints that are public but behave differently when signed in. A
credential that is present but *invalid* still raises - failing open there
would mean a typo'd token silently downgrades to anonymous access.
"""
if not AUTH_ENABLED:
return anonymous_principal()
if credentials is None and not api_key:
return None
return get_principal(credentials, api_key)
def require_role(*roles: str) -> Callable[[Principal], Principal]:
"""Require the caller to hold one of ``roles``."""
allowed = frozenset(roles)
def _dependency(principal: Principal = Depends(get_principal)) -> Principal:
if principal.role not in allowed:
raise HTTPException(
status_code=status.HTTP_403_FORBIDDEN,
detail=(
f"This operation requires the {' or '.join(sorted(allowed))} role; "
f"you are signed in as '{principal.role}'."
),
)
return principal
return _dependency
def require_permission(permission: str) -> Callable[[Principal], Principal]:
"""
Require a specific permission. ``admin`` passes every check - see
``Principal.has_permission``.
"""
def _dependency(principal: Principal = Depends(get_principal)) -> Principal:
if not principal.has_permission(permission):
raise HTTPException(
status_code=status.HTTP_403_FORBIDDEN,
detail=f"This operation requires the '{permission}' permission.",
)
return principal
return _dependency
# The two guards used most often, named so route decorators stay readable.
require_admin = require_role("admin")
require_authenticated = get_principal

View File

@@ -0,0 +1,59 @@
"""
Tiny in-memory job tracker for background catalog-generation tasks.
Deliberately not a queue/Celery/Redis setup - the original project already
had celery+redis in requirements.txt but nothing wired it up, and adding a
broker is unnecessary operational weight for a single-developer, CPU-only
project. A process-local dict is enough to let the React UI show
"running -> done/failed" status for a brand ingestion job started from the
admin panel.
NOTE: state is lost on server restart, and is per-process (not safe for
multiple uvicorn workers). For this project's intended scale (one backend
process on a personal machine) that's a fine trade-off; see the docs'
"Scaling beyond a single machine" section if this ever needs to change.
"""
from __future__ import annotations
import threading
import time
import uuid
from dataclasses import dataclass, field
from typing import Dict, Optional
@dataclass
class Job:
job_id: str
brand: str
status: str = "pending" # pending -> running -> done | failed
detail: Optional[str] = None
created_at: float = field(default_factory=time.time)
updated_at: float = field(default_factory=time.time)
class JobStore:
def __init__(self) -> None:
self._jobs: Dict[str, Job] = {}
self._lock = threading.Lock()
def create(self, brand: str) -> Job:
job = Job(job_id=str(uuid.uuid4()), brand=brand)
with self._lock:
self._jobs[job.job_id] = job
return job
def update(self, job_id: str, status: str, detail: Optional[str] = None) -> None:
with self._lock:
job = self._jobs.get(job_id)
if job:
job.status = status
job.detail = detail
job.updated_at = time.time()
def get(self, job_id: str) -> Optional[Job]:
with self._lock:
return self._jobs.get(job_id)
job_store = JobStore()

View File

View File

@@ -0,0 +1,357 @@
"""
Authentication router - issues and inspects access tokens.
This replaces an earlier version that returned a role profile without issuing
anything, accepted an empty password, and granted `admin` to any username that
asked for the role. It decided which buttons the UI drew; it protected nothing.
Now the token this returns is the credential every write endpoint checks (see
app/api/deps.py), so the rules hold for curl and partner scripts too, not just
for the React app.
Accounts come from the environment - two of them, admin and user, configured as
PBKDF2 digests. That is deliberately not a user database: this project has no
user table, no registration flow and no password reset, and inventing one here
would be a bigger change than the problem calls for. Machine consumers get
API_KEYS instead. If per-user accounts become a real requirement, this module
is the seam to replace.
For local work there is AUTH_ALLOW_ANY_LOGIN, which skips the password check
here and nowhere else - the token still gets signed and every guard downstream
still checks it. It is off by default and logs a warning at startup when on.
"""
from __future__ import annotations
import logging
import threading
import time
from typing import Dict, List, Tuple
from fastapi import APIRouter, Depends, HTTPException, Request, status
from pydantic import BaseModel, Field
from app.api.deps import get_principal
from app.infrastructure.security import (
ROLE_PERMISSIONS,
Principal,
create_access_token,
hash_is_wellformed,
password_hash_fingerprint,
verify_password,
)
from app.infrastructure.settings import (
AUTH_ADMIN_PASSWORD_HASH,
AUTH_ADMIN_USERNAME,
AUTH_ALLOW_ANY_LOGIN,
AUTH_ENABLED,
AUTH_LOCKOUT_SECONDS,
AUTH_MAX_LOGIN_ATTEMPTS,
AUTH_USER_PASSWORD_HASH,
AUTH_USER_USERNAME,
config_source,
)
logger = logging.getLogger(__name__)
router = APIRouter(prefix="/auth", tags=["auth"])
if AUTH_ENABLED and AUTH_ALLOW_ANY_LOGIN:
logger.warning(
"AUTH_ALLOW_ANY_LOGIN=true: /api/auth/login accepts ANY password, so anyone "
"who can reach this port can sign in as admin. Local development only - "
"set it to false in backend/.env before exposing this server."
)
class LoginRequest(BaseModel):
username: str = Field(min_length=1, max_length=150)
password: str = Field(min_length=1, max_length=1024)
class UserProfile(BaseModel):
username: str
role: str
display_name: str
email: str
permissions: List[str] = Field(default_factory=list)
class LoginResponse(BaseModel):
access_token: str
token_type: str = "bearer"
expires_in: int = Field(description="Token lifetime in seconds")
user: UserProfile
# A syntactically valid hash of an unguessable value. Never matches any real
# password; it exists only so the unknown-username path in login() does the
# same PBKDF2 work as the known one, keeping the two indistinguishable by timing.
_DUMMY_HASH = (
"pbkdf2_sha256$600000$YWJjZGVmZ2hpamtsbW5vcA==$"
"S1cVFrGD4pDkGqSjbEbaVSTONzGhCT9BOaWPQ2vwvvA="
)
def _accounts() -> Dict[str, dict]:
"""
The configured accounts, read per call so a settings reload is picked up.
Usernames are compared case-insensitively (matching what the login form
sends), but the password is not touched - the previous version lowercased
it before comparing, which silently shrank the effective keyspace.
An account with a blank password hash is omitted entirely rather than
included with an unmatchable digest. Both spellings deny the login, but
only omission keeps it out of the account table, so nothing downstream can
treat it as a real account. This is how the optional `user` account is
switched off: leave AUTH_USER_PASSWORD_HASH unset and only `admin` exists.
"""
accounts = {
AUTH_ADMIN_USERNAME.lower(): {
"password_hash": AUTH_ADMIN_PASSWORD_HASH,
"role": "admin",
"display_name": "System Administrator",
"email": "admin@nutritionintel.com",
},
}
if AUTH_USER_PASSWORD_HASH:
accounts[AUTH_USER_USERNAME.lower()] = {
"password_hash": AUTH_USER_PASSWORD_HASH,
"role": "user",
"display_name": "Product & Store Manager",
"email": "user@nutritionintel.com",
}
return accounts
# ---------------------------------------------------------------------------
# Failed-login throttle
# ---------------------------------------------------------------------------
# In-process and per-worker: with several uvicorn workers a determined attacker
# gets AUTH_MAX_LOGIN_ATTEMPTS per worker, not overall. That is a real limit,
# not a rounding error - but it still turns an unbounded password oracle into a
# rate-limited one without adding Redis to the deployment. Move this to a shared
# store if you ever run many workers.
_failures: Dict[Tuple[str, str], Tuple[int, float]] = {}
_failures_lock = threading.Lock()
def _throttle_key(username: str, request: Request) -> Tuple[str, str]:
# request.client.host is the real client IP because uvicorn runs with
# --proxy-headers behind nginx/Caddy (see backend/Dockerfile); without that
# every request would appear to come from the proxy and share one bucket.
client = request.client.host if request.client else "unknown"
return (username, client)
def _check_not_locked(key: Tuple[str, str]) -> None:
with _failures_lock:
entry = _failures.get(key)
if entry is None:
return
count, first_seen = entry
if time.time() - first_seen > AUTH_LOCKOUT_SECONDS:
del _failures[key]
return
if count >= AUTH_MAX_LOGIN_ATTEMPTS:
retry_after = int(AUTH_LOCKOUT_SECONDS - (time.time() - first_seen))
raise HTTPException(
status_code=status.HTTP_429_TOO_MANY_REQUESTS,
detail=f"Too many failed sign-in attempts. Try again in {retry_after}s.",
headers={"Retry-After": str(max(retry_after, 1))},
)
def _record_failure(key: Tuple[str, str]) -> None:
now = time.time()
with _failures_lock:
count, first_seen = _failures.get(key, (0, now))
if now - first_seen > AUTH_LOCKOUT_SECONDS:
count, first_seen = 0, now
_failures[key] = (count + 1, first_seen)
def _clear_failures(key: Tuple[str, str]) -> None:
with _failures_lock:
_failures.pop(key, None)
# ---------------------------------------------------------------------------
# Routes
# ---------------------------------------------------------------------------
@router.post("/login", response_model=LoginResponse)
def login(payload: LoginRequest, request: Request) -> LoginResponse:
"""Exchange a username and password for an access token."""
if not AUTH_ENABLED:
raise HTTPException(
status_code=status.HTTP_503_SERVICE_UNAVAILABLE,
detail=(
"Authentication is disabled on this server (AUTH_ENABLED=false), so no "
"token can be issued. Every endpoint is open; sign-in is not required."
),
)
username = payload.username.strip().lower()
key = _throttle_key(username, request)
if AUTH_ALLOW_ANY_LOGIN:
# Dev bypass: any password gets in. The username still picks the
# account, so `admin` lands on the admin pages and `user` on the user
# ones; anything else is an unconfigured name and gets the lower of the
# two roles rather than silently minting an admin. Throttling is skipped
# because there is no longer a password to guess.
account = _accounts().get(username) or {
"role": "user",
"display_name": payload.username.strip() or username,
"email": f"{username}@nutritionintel.com",
}
logger.warning(
"AUTH_ALLOW_ANY_LOGIN: signing in %r as %s without checking the password",
username,
account["role"],
)
else:
_check_not_locked(key)
account = _accounts().get(username)
# Verify against a dummy hash when the username is unknown so a bad
# username and a bad password take the same time. Otherwise the response
# latency alone enumerates valid usernames.
stored_hash = account["password_hash"] if account else _DUMMY_HASH
# A hash that does not parse can never match, and verify_password bails
# out of one before doing any PBKDF2 work - measured here, 0.16ms against
# 439ms for a real digest. That inverts the very property _DUMMY_HASH
# exists to protect: an account whose configured hash is corrupt would
# answer ~2700x faster than every other username, announcing which
# account is broken to anyone with a stopwatch. So spend the same work
# regardless; the result is a rejection either way.
hash_usable = hash_is_wellformed(stored_hash)
password_ok = verify_password(
payload.password, stored_hash if hash_usable else _DUMMY_HASH
)
if account is None or not password_ok:
_record_failure(key)
# The reason goes to the LOG, never to the caller - the response
# below is byte-identical whichever of these it was, so nothing here
# can be used to enumerate usernames. It is computed after both the
# lookup and the PBKDF2 call above, so it adds no timing signal
# either. Without it, a deployment whose configured hash or admin
# username has drifted is indistinguishable from someone simply
# typing the wrong password, and this is exactly how a production
# sign-in outage stayed unexplained: the log said "Failed sign-in
# for 'admin'" and nothing more.
if account is None:
logger.warning(
"Failed sign-in for %r from %s: reason=unknown-username. "
"Configured accounts: %s (AUTH_ADMIN_USERNAME source=%s).",
username,
key[1],
", ".join(sorted(_accounts())),
config_source("AUTH_ADMIN_USERNAME"),
)
elif not hash_usable:
# ERROR, not WARNING: this is a broken deployment, not a bad
# guess. No password can ever match, so every sign-in to this
# account will 401 until the hash itself is replaced.
logger.error(
"Failed sign-in for %r from %s: reason=malformed-hash. The configured "
"password hash does not parse as pbkdf2_sha256$<iterations>$<b64 salt>$"
"<b64 digest> (fingerprint=%s, source=%s). Nobody can sign in to this "
"account until it is regenerated with scripts/make_auth_secrets.py.",
username,
key[1],
password_hash_fingerprint(stored_hash) or "(empty)",
config_source("AUTH_ADMIN_PASSWORD_HASH"),
)
else:
logger.warning(
"Failed sign-in for %r from %s: reason=bad-password. The account exists "
"and its hash parses (fingerprint=%s, source=%s); the password did not "
"match. If this IS the password you deployed, then the running config "
"carries a different hash than the file you are reading - compare that "
"fingerprint against: python scripts/make_auth_secrets.py "
"--fingerprint .env.production",
username,
key[1],
password_hash_fingerprint(stored_hash),
config_source("AUTH_ADMIN_PASSWORD_HASH"),
)
# One message for every failure mode, for the same reason.
raise HTTPException(
status_code=status.HTTP_401_UNAUTHORIZED,
detail="Invalid username or password.",
)
_clear_failures(key)
role = account["role"]
permissions = ROLE_PERMISSIONS.get(role, [])
token, expires_in = create_access_token(username, role, permissions)
logger.info("Issued token for %r (role=%s)", username, role)
return LoginResponse(
access_token=token,
expires_in=expires_in,
user=UserProfile(
username=username,
role=role,
display_name=account["display_name"],
email=account["email"],
permissions=permissions,
),
)
@router.get("/me", response_model=UserProfile)
def me(principal: Principal = Depends(get_principal)) -> UserProfile:
"""
Who the presented credential belongs to. 401 if it is missing or expired.
The frontend calls this on boot to check a restored session before showing
the app, so an expired token lands on the login page rather than on a
dashboard whose every request then fails.
"""
account = _accounts().get(principal.username, {})
return UserProfile(
username=principal.username,
role=principal.role,
display_name=account.get("display_name", principal.username.title()),
email=account.get("email", f"{principal.username}@nutritionintel.com"),
permissions=principal.permissions,
)
@router.get("/roles")
def list_roles() -> dict:
"""
The available roles and what each may do.
Note there are no demo credentials here any more. The passwords are set per
deployment via AUTH_ADMIN_PASSWORD_HASH / AUTH_USER_PASSWORD_HASH; this
endpoint used to publish working ones to anyone who asked.
"""
return {
"roles": [
{
"id": "admin",
"name": "Admin",
"description": (
"Full access: catalog brand cards, project details, Excel/CSV "
"train/test uploads, discount allocation, analytics and nutrition. "
"Implicitly holds every permission."
),
"permissions": ROLE_PERMISSIONS["admin"],
},
{
"id": "user",
"name": "User",
"description": (
"Combined user and store role: single or batch product uploads with "
"image and DB/JSON sync, store inventory, profit analytics, nutrition."
),
"permissions": ROLE_PERMISSIONS["user"],
},
]
}

View File

@@ -0,0 +1,188 @@
"""Read-only catalogue API: category -> brand -> product -> per-platform offers.
Only VERIFIED products are served (see repository.refresh_verification), and
every price is returned with the site, URL, source type and time it was seen.
Money is returned as a decimal string, never a float.
"""
from __future__ import annotations
from decimal import Decimal
from typing import Any, Dict, List, Optional
from fastapi import APIRouter, HTTPException, Query
from app.electronics.db.connection import connect
from app.electronics.db.repository import product_rating_and_reviews
from app.electronics.reviews import select_reviews
router = APIRouter(prefix="/elec", tags=["electronics"])
def _money(value: Optional[Decimal]) -> Optional[str]:
return None if value is None else format(value, "f")
def _clean(row: Dict[str, Any]) -> Dict[str, Any]:
out = {}
for k, v in row.items():
if isinstance(v, Decimal):
out[k] = _money(v) if k in ("price", "mrp", "best_price", "min_price", "max_price") else float(v)
elif hasattr(v, "isoformat"):
out[k] = v.isoformat()
else:
out[k] = v
return out
@router.get("/categories")
def categories() -> List[dict]:
with connect() as conn:
rows = conn.execute(
"SELECT c.slug, c.name, coalesce(sum(s.product_count), 0)::int AS product_count "
"FROM elec.category c LEFT JOIN elec.v_brand_summary s ON s.category = c.slug "
"GROUP BY c.slug, c.name ORDER BY c.name"
).fetchall()
return [_clean(r) for r in rows]
@router.get("/brands")
def brands(category: str = Query(...)) -> List[dict]:
with connect() as conn:
rows = conn.execute(
"""
SELECT b.name AS brand, b.slug AS brand_slug,
coalesce(s.product_count, 0)::int AS product_count, s.min_price, s.max_price,
(SELECT image_url FROM elec.v_brand_catalog v
WHERE v.brand_slug = b.slug AND v.category = %(c)s AND v.image_url IS NOT NULL
ORDER BY v.platform_count DESC LIMIT 1) AS sample_image
FROM elec.brand b
JOIN elec.brand_category bc ON bc.brand_id = b.id
JOIN elec.category c ON c.id = bc.category_id AND c.slug = %(c)s
LEFT JOIN elec.v_brand_summary s ON s.brand_slug = b.slug AND s.category = %(c)s
ORDER BY coalesce(s.product_count, 0) DESC, b.name
""",
{"c": category},
).fetchall()
return [_clean(r) for r in rows]
@router.get("/products")
def products(
category: Optional[str] = None,
brand: Optional[str] = None,
q: Optional[str] = Query(None, max_length=100),
min_price: Optional[Decimal] = None,
max_price: Optional[Decimal] = None,
in_stock: bool = False,
site: Optional[str] = None,
tn_only: bool = False,
limit: int = Query(48, ge=1, le=200),
offset: int = Query(0, ge=0),
) -> dict:
where, params = ["TRUE"], {}
if category:
where.append("v.category = %(category)s"); params["category"] = category
if brand:
where.append("v.brand_slug = %(brand)s"); params["brand"] = brand
if q:
where.append("(v.display_name ILIKE %(q)s OR v.brand ILIKE %(q)s)"); params["q"] = f"%{q}%"
if min_price is not None:
where.append("v.best_price >= %(min_price)s"); params["min_price"] = min_price
if max_price is not None:
where.append("v.best_price <= %(max_price)s"); params["max_price"] = max_price
if in_stock:
where.append("EXISTS (SELECT 1 FROM elec.v_product_availability a WHERE a.product_id = v.product_id AND a.in_stock)")
if site:
where.append("EXISTS (SELECT 1 FROM elec.v_product_availability a WHERE a.product_id = v.product_id AND a.domain = %(site)s)")
params["site"] = site
if tn_only:
where.append("v.sold_by_tn_retailer")
sql_where = " AND ".join(where)
with connect() as conn:
total = conn.execute(f"SELECT count(*) AS n FROM elec.v_brand_catalog v WHERE {sql_where}", params).fetchone()["n"]
rows = conn.execute(
f"SELECT v.* FROM elec.v_brand_catalog v WHERE {sql_where} "
f"ORDER BY v.platform_count DESC, v.best_price NULLS LAST, v.display_name "
f"LIMIT %(limit)s OFFSET %(offset)s",
{**params, "limit": limit, "offset": offset},
).fetchall()
return {"total": total, "products": [_clean(r) for r in rows]}
@router.get("/products/{product_id}")
def product(product_id: int) -> dict:
with connect() as conn:
row = conn.execute("SELECT * FROM elec.v_brand_catalog WHERE product_id = %s", (product_id,)).fetchone()
if not row:
raise HTTPException(status_code=404, detail="Product not found or not verified")
specs = conn.execute("SELECT spec_sources FROM elec.product WHERE id = %s", (product_id,)).fetchone()
offers = conn.execute(
"SELECT * FROM elec.v_product_availability WHERE product_id = %s "
"ORDER BY (price IS NULL), (source_type = 'search_snippet'), price, site",
(product_id,),
).fetchall()
images = conn.execute(
"SELECT i.url, i.source_type, s.name AS site, l.source_url AS found_on "
"FROM elec.product_image i JOIN elec.source_listing l ON l.id = i.source_listing_id "
"JOIN elec.site s ON s.id = l.site_id WHERE i.product_id = %s ORDER BY i.rank, i.id",
(product_id,),
).fetchall()
rated = product_rating_and_reviews(conn, product_id)
result = _clean(row)
result["spec_sources"] = specs["spec_sources"] if specs else {}
result["offers"] = [_clean(o) for o in offers]
result["images"] = [dict(i) for i in images]
result["rating"] = _overall_rating(rated["sources"])
overall = result["rating"]["value"] if result["rating"] else None
result["reviews"] = [_clean(r) for r in select_reviews(overall, rated["reviews"])]
return result
def _overall_rating(sources: List[dict]) -> Optional[dict]:
"""The product's rating across the platforms that state one: the mean
weighted by each platform's rating count (a platform that states no count
weighs as 1). None when no platform states a rating - never a guess."""
if not sources:
return None
weight = lambda s: max(int(s["review_count"] or 0), 1) # noqa: E731
total = sum(weight(s) for s in sources)
value = sum(Decimal(s["rating"]) * weight(s) for s in sources) / total
counts = [s["review_count"] for s in sources if s["review_count"]]
return {
"value": round(float(value), 1),
"count": sum(counts) if counts else None,
"sources": [
{"site": s["site"], "rating": float(s["rating"]), "review_count": s["review_count"], "source_url": s["source_url"]}
for s in sources
],
}
@router.get("/products/{product_id}/price-history")
def price_history(product_id: int) -> List[dict]:
with connect() as conn:
rows = conn.execute(
"""
SELECT s.name AS site, h.price, h.mrp, h.in_stock, h.source_type, h.observed_at
FROM elec.price_history h
JOIN elec.product_listing_map m ON m.listing_id = h.listing_id AND m.review_status IN ('auto','approved')
JOIN elec.source_listing l ON l.id = h.listing_id
JOIN elec.site s ON s.id = l.site_id
WHERE m.product_id = %s AND h.price IS NOT NULL
ORDER BY h.observed_at
""",
(product_id,),
).fetchall()
return [_clean(r) for r in rows]
@router.get("/sites")
def sites() -> List[dict]:
with connect() as conn:
rows = conn.execute(
"SELECT s.name, s.domain, s.kind, s.region, s.policy, s.probe_outcome, s.probed_at, "
"s.breaker_until, s.breaker_reason, s.probe_evidence->>'reason' AS probe_reason, "
"(SELECT count(*) FROM elec.source_listing l WHERE l.site_id = s.id)::int AS listings "
"FROM elec.site s ORDER BY (s.kind = 'brand_official'), s.name"
).fetchall()
return [_clean(r) for r in rows]

View File

@@ -0,0 +1,124 @@
"""Admin endpoints: start a collection run, see runs, probe sites, review matches."""
from __future__ import annotations
import threading
import time
import uuid
from typing import Dict, List, Optional
from fastapi import APIRouter, Depends, HTTPException
from pydantic import BaseModel, Field
from app.api.background import run_in_background
from app.api.deps import require_admin
from app.electronics.db import repository as repo
from app.electronics.reference import load_reference
router = APIRouter(prefix="/elec/admin", tags=["electronics-admin"], dependencies=[Depends(require_admin)])
_jobs: Dict[str, dict] = {}
_jobs_lock = threading.Lock()
_run_lock = threading.Lock() # one collection at a time: polite to sites, kind to 8 GB RAM
class RunRequest(BaseModel):
category: str = Field(..., examples=["mobiles"])
brands: List[str] = Field(default_factory=list, description="brand slugs; empty = all for the category")
limit: int = Field(10, ge=1, le=60, description="max models per brand")
expand: int = Field(6, ge=0, le=30)
budget: int = Field(150, ge=10, le=1000, description="max search queries")
fetch_pages: bool = True
use_llm: bool = True
def _job_update(job_id: str, **fields) -> None:
with _jobs_lock:
_jobs[job_id].update(fields, updated_at=time.time())
@router.post("/runs", status_code=202)
def start_run(req: RunRequest) -> dict:
ref = load_reference()
if req.category not in ref.categories:
raise HTTPException(400, f"unknown category {req.category!r}")
brands = req.brands or [b.slug for b in ref.brands_for(req.category)]
bad = [b for b in brands if b not in ref.brands or req.category not in ref.brands[b].categories]
if bad:
raise HTTPException(400, f"not allow-listed for {req.category}: {bad}")
if _run_lock.locked():
raise HTTPException(409, "A collection run is already in progress")
job_id = str(uuid.uuid4())
with _jobs_lock:
_jobs[job_id] = {"job_id": job_id, "status": "queued", "log": [], "stats": {}, "created_at": time.time()}
def work() -> None:
from app.electronics.collector import Collector, RunOptions
with _run_lock:
_job_update(job_id, status="running")
def progress(msg: str) -> None:
with _jobs_lock:
_jobs[job_id]["log"] = (_jobs[job_id]["log"] + [msg])[-200:]
try:
opts = RunOptions(category=req.category, brands=brands, max_products_per_brand=req.limit,
expand_per_brand=req.expand, search_budget=req.budget,
fetch_pages=req.fetch_pages, use_llm=req.use_llm)
stats = Collector(opts, progress=progress).run()
_job_update(job_id, status="done", stats=stats)
except Exception as exc: # noqa: BLE001 - reported to the UI
_job_update(job_id, status="failed", error=repr(exc))
run_in_background(work, name=f"elec-run-{job_id[:8]}")
return {"job_id": job_id}
@router.get("/runs/{job_id}")
def get_job(job_id: str) -> dict:
with _jobs_lock:
job = _jobs.get(job_id)
if not job:
raise HTTPException(404, "unknown job")
return dict(job)
@router.get("/runs")
def list_runs(limit: int = 20) -> List[dict]:
rows = repo.recent_runs(limit)
for r in rows:
for k in ("started_at", "ended_at"):
if r.get(k):
r[k] = r[k].isoformat()
return rows
@router.get("/review")
def review_queue() -> List[dict]:
return [{**r, "confidence": float(r["confidence"])} for r in repo.review_queue()]
class ReviewDecision(BaseModel):
approve: bool
@router.post("/review/{listing_id}")
def review(listing_id: int, decision: ReviewDecision) -> dict:
if not repo.set_review(listing_id, decision.approve):
raise HTTPException(404, "no pending match for that listing")
return {"ok": True, "products": repo.refresh_verification()}
@router.post("/sites/{domain}/probe")
def probe(domain: str) -> dict:
from app.electronics.net.polite_client import PoliteClient
from app.electronics.probe.site_probe import probe_site
from app.electronics.search.engine import SearchEngine
site = load_reference().sites.get(domain)
if not site:
raise HTTPException(404, "unknown site")
with PoliteClient() as client:
res = probe_site(site, client, SearchEngine(budget=4))
repo.set_probe_result(domain, res["outcome"], res["robots_allowed"], res["evidence"])
return res

View File

@@ -0,0 +1,36 @@
from __future__ import annotations
import logging
from fastapi import APIRouter
from app.api.schemas import AuthConfigOut, HealthOut, SearchStatusOut
from app.electronics.db.connection import check_connection
from app.infrastructure.security import auth_config_summary
from app.infrastructure.settings import (
DB_NAME, EMBEDDINGS_MODEL, OLLAMA_MODEL_NAME, USE_DDG_SEARCH, USE_GOOGLE_CSE,
)
from app.services import ollama_service
logger = logging.getLogger(__name__)
router = APIRouter(tags=["health"])
@router.get("/health", response_model=HealthOut)
def health() -> HealthOut:
"""Liveness/readiness probe used by the React app to show a banner when
Postgres or Ollama aren't reachable, instead of failing silently.
Ollama being down does not make the service degraded: it is only used to
fill spec gaps, and the pipeline runs deterministically without it."""
db_ok = check_connection()
return HealthOut(
status="ok" if db_ok else "degraded",
database=db_ok,
database_name=DB_NAME,
ollama=bool(ollama_service._ensure_client()),
ollama_model=OLLAMA_MODEL_NAME,
embeddings_model=EMBEDDINGS_MODEL,
search=SearchStatusOut(ddg=USE_DDG_SEARCH, google_cse=USE_GOOGLE_CSE),
auth=AuthConfigOut(**auth_config_summary()),
)

View File

@@ -0,0 +1,73 @@
"""Pydantic response models for the health endpoint. The electronics
catalogue's own models live in app/electronics/api_models.py."""
from __future__ import annotations
from typing import List, Optional
from pydantic import BaseModel, Field
class ApiKeyInfoOut(BaseModel):
"""One configured machine consumer, named but never quoted.
`fingerprint` is a truncated digest of name+secret, not the secret. It exists
so a caller who was issued a key can confirm THAT key is the one this
deployment loaded - the question a 401 cannot answer, since an undeployed key
and a wrong key fail identically.
"""
name: str
role: str
fingerprint: str
class AuthConfigOut(BaseModel):
"""
The effective auth configuration, reported by /api/health.
Unauthenticated on purpose. The failure this exists to diagnose is "nobody
can sign in", so anything gated behind an admin token is unreachable
exactly when it is needed. Nothing here is a secret: the admin username is
already the documented one, allow_any_login=true is a fact an operator
urgently needs (and an attacker discovers with a single login attempt
anyway), and the fingerprint is a truncated hash of a salted digest, not a
password. The API key block follows the same rule: it names which consumers
are configured and fingerprints their keys, so a caller can tell an
undeployed key from a rejected one, but it never renders a secret. What it buys is a one-command answer to "is this deployment
running the config I think it is?" - compare the fingerprint here against
the one printed by scripts/make_auth_secrets.py --fingerprint.
"""
enabled: bool
allow_any_login: bool
admin_username: str
password_hash_valid: bool
password_hash_iterations: Optional[int] = None
password_hash_fingerprint: str
# "process-env" | "env-file" | "default" - which one actually won.
admin_username_source: str
password_hash_source: str
# Machine consumers. Names and fingerprints only - the secrets themselves are
# never rendered here, and _parse_api_keys enforces enough entropy that the
# fingerprints do not give them away. Defaulted so a client of this schema
# still validates against a deployment predating these fields.
api_keys_count: int = 0
api_keys: List[ApiKeyInfoOut] = Field(default_factory=list)
api_keys_source: str = "default"
class SearchStatusOut(BaseModel):
"""Which web-search providers discovery can use right now."""
ddg: bool
google_cse: bool
class HealthOut(BaseModel):
status: str
database: bool
database_name: str
ollama: bool
ollama_model: str
embeddings_model: str
search: SearchStatusOut
auth: AuthConfigOut

View File

View File

@@ -0,0 +1,261 @@
"""Command line for the electronics pipeline. Run from backend/:
python -m app.electronics.cli migrate
python -m app.electronics.cli seed-reference
python -m app.electronics.cli probe [--site croma.com] [--all]
python -m app.electronics.cli collect --category mobiles --brand samsung --brand xiaomi --limit 15
python -m app.electronics.cli reviews [--category mobiles]
python -m app.electronics.cli report
python -m app.electronics.cli review [--approve ID | --reject ID]
python -m app.electronics.cli verify-grounding
"""
from __future__ import annotations
import json
import logging
import re
from decimal import Decimal
from typing import List, Optional
import typer
app = typer.Typer(add_completion=False, help="Electronics catalogue: search-first, evidence-backed collection.")
def _setup_logging(verbose: bool) -> None:
logging.basicConfig(level=logging.DEBUG if verbose else logging.INFO,
format="%(asctime)s %(levelname)s %(name)s: %(message)s")
for noisy in ("httpx", "httpcore", "primp", "ddgs", "urllib3", "sentence_transformers"):
logging.getLogger(noisy).setLevel(logging.WARNING)
@app.command()
def migrate() -> None:
"""Create/upgrade the elec schema in the local electronics_catalog database."""
from app.electronics.db.migrate import run_migrations
applied = run_migrations()
typer.echo(f"Applied: {', '.join(applied) if applied else 'nothing (up to date)'}")
@app.command("seed-reference")
def seed_reference() -> None:
"""Load brands, aliases, categories and sites from reference/*.yaml."""
from app.electronics.db import repository as repo
from app.electronics.reference import load_reference
typer.echo(json.dumps(repo.seed_reference(load_reference())))
@app.command()
def probe(site: List[str] = typer.Option([], "--site", help="Domain(s) to probe; default all probe-policy sites"),
include_official: bool = typer.Option(False, "--official", help="Also probe brand official sites"),
verbose: bool = False) -> None:
"""Grade sites A/B/C: may they be scraped, or only searched?"""
_setup_logging(verbose)
from app.electronics.db import repository as repo
from app.electronics.net.polite_client import PoliteClient
from app.electronics.probe.site_probe import probe_site
from app.electronics.reference import load_reference
from app.electronics.search.engine import SearchEngine
ref = load_reference()
if site:
targets = [s for s in ref.sites.values() if s.domain in site]
else:
targets = [s for s in ref.sites.values() if include_official or s.kind != "brand_official"]
run_id = repo.start_run("probe", {"sites": [s.domain for s in targets]})
client = PoliteClient(on_fetch=lambda r, host: repo.log_fetch(run_id, r.url, host, r.status, r.bytes,
r.outcome, r.robots_allowed))
engine = SearchEngine(budget=len(targets) * 3)
results = {}
try:
for s in targets:
res = probe_site(s, client, engine)
repo.set_probe_result(s.domain, res["outcome"], res["robots_allowed"], res["evidence"])
results[s.domain] = res["outcome"]
typer.echo(f"{s.name:28} {s.domain:24} {res['outcome']} {res['evidence'].get('reason')}")
finally:
client.close()
repo.finish_run(run_id, "done", {"grades": results})
@app.command()
def collect(category: str = typer.Option(..., help="mobiles | laptops"),
brand: List[str] = typer.Option([], "--brand", help="Brand slug(s); default all brands of the category"),
limit: int = typer.Option(15, help="Max models per brand"),
expand: int = typer.Option(8, help="Models per brand looked up on other platforms"),
budget: int = typer.Option(200, help="Max search queries this run"),
no_fetch: bool = typer.Option(False, "--no-fetch", help="Search results only; fetch no pages"),
no_llm: bool = typer.Option(False, "--no-llm", help="Deterministic spec parsing only"),
no_embed: bool = typer.Option(False, "--no-embed"),
reprobe: bool = False,
verbose: bool = False) -> None:
"""Discover and collect real listings for allow-listed brands."""
_setup_logging(verbose)
from app.electronics.collector import Collector, RunOptions
from app.electronics.reference import load_reference
ref = load_reference()
if category not in ref.categories:
raise typer.BadParameter(f"unknown category {category!r}; use one of {list(ref.categories)}")
brands = brand or [b.slug for b in ref.brands_for(category)]
unknown = [b for b in brands if b not in ref.brands or category not in ref.brands[b].categories]
if unknown:
raise typer.BadParameter(f"not allow-listed for {category}: {unknown}")
opts = RunOptions(category=category, brands=brands, max_products_per_brand=limit, expand_per_brand=expand,
search_budget=budget, use_llm=not no_llm, fetch_pages=not no_fetch, reprobe=reprobe)
stats = Collector(opts, progress=typer.echo).run(embed=not no_embed)
typer.echo(json.dumps(stats, indent=2, sort_keys=True))
@app.command()
def prices(limit: int = typer.Option(40, help="Max Google queries (free tier: 100/day)"),
category: Optional[str] = typer.Option(None, help="mobiles | laptops (default both)")) -> None:
"""Fill missing prices on search-only platforms from Google's structured data (no site fetches)."""
_setup_logging(False)
from app.electronics.price_lookup import lookup_prices
stats = lookup_prices(limit=limit, category=category, progress=typer.echo)
typer.echo(json.dumps(stats, indent=2, default=str))
if stats.get("error"):
typer.echo("\nGoogle search is not usable yet: " + str(stats["error"]))
raise typer.Exit(code=1)
@app.command()
def rematch(category: str = typer.Option(..., help="mobiles | laptops"),
no_embed: bool = typer.Option(False, "--no-embed")) -> None:
"""Rebuild products from stored listings with the current matching rules (no network)."""
_setup_logging(False)
from app.electronics.match.rematch import rematch as run_rematch
stats = run_rematch(category)
if not no_embed:
from app.electronics.collector import embed_verified_products
try:
stats["embedded"] = embed_verified_products()
except Exception as exc: # noqa: BLE001
typer.echo(f"Embedding skipped: {exc}")
typer.echo(json.dumps(stats, indent=2))
@app.command()
def reviews(category: Optional[str] = typer.Option(None, help="mobiles | laptops (default both)"),
limit: int = typer.Option(200, help="Max product pages to re-read"),
verbose: bool = False) -> None:
"""Re-read ratings and customer reviews from the product pages already on file.
Only pages the collector itself reads (scraped / brand official listings)
are fetched, politely (robots.txt, per-site pacing, circuit breaker). A
rating or review is stored only when the page's own schema.org data states
it; nothing is generated.
"""
_setup_logging(verbose)
from rapidfuzz import fuzz
from app.electronics.db import repository as repo
from app.electronics.extract.jsonld import extract_products
from app.electronics.net.polite_client import PoliteClient
rows = repo.listings_for_review_backfill(category)[:limit]
run_id = repo.start_run("reviews", {"category": category, "pages": len(rows)})
client = PoliteClient(on_fetch=lambda r, host: repo.log_fetch(run_id, r.url, host, r.status, r.bytes,
r.outcome, r.robots_allowed))
stats = {"pages": 0, "pages_ok": 0, "rated": 0, "reviews_stored": 0, "no_matching_product": 0}
status, error = "done", None
try:
for row in rows:
stats["pages"] += 1
res = client.get(row["source_url"])
if not res.ok:
continue
stats["pages_ok"] += 1
products = extract_products(res.text)
# The same product the listing was stored from: its SKU, else its name.
match = next((p for p in products if p.get("sku") and p["sku"] == row["source_sku"]), None)
if match is None:
title = (row["title"] or "").lower()
scored = [(fuzz.token_set_ratio(p["name"].lower(), title), p) for p in products]
scored = [sp for sp in scored if sp[0] >= 85]
match = max(scored, key=lambda sp: sp[0])[1] if scored else None
if match is None:
stats["no_matching_product"] += 1
continue
if match.get("rating") is not None and Decimal(0) < match["rating"] <= Decimal(5):
repo.update_listing_rating(row["listing_id"], match["rating"], match.get("review_count"))
stats["rated"] += 1
if match.get("reviews"):
stats["reviews_stored"] += repo.save_reviews(row["listing_id"], match["reviews"])
typer.echo(f" {row['domain']:22} rating={match.get('rating')} reviews={len(match.get('reviews') or [])}")
except Exception as exc: # noqa: BLE001
status, error = "failed", repr(exc)
raise
finally:
client.close()
repo.finish_run(run_id, status, stats, error)
typer.echo(json.dumps(stats, indent=2))
@app.command()
def report() -> None:
"""Counts per brand/category and per site."""
from app.electronics.db.connection import connect
with connect() as conn:
typer.echo("Sites:")
for r in conn.execute("SELECT name, domain, policy, probe_outcome, breaker_until FROM elec.site "
"WHERE kind <> 'brand_official' ORDER BY name"):
typer.echo(f" {r['name']:22} {r['policy']:9} grade={r['probe_outcome'] or '-'}"
f"{' breaker until ' + str(r['breaker_until']) if r['breaker_until'] else ''}")
typer.echo("\nProducts by status:")
for r in conn.execute("SELECT verification_status, count(*) n FROM elec.product GROUP BY 1"):
typer.echo(f" {r['verification_status']:12} {r['n']}")
typer.echo("\nVerified catalogue (brand / category):")
for r in conn.execute("SELECT * FROM elec.v_brand_summary ORDER BY category, brand"):
typer.echo(f" {r['brand']:10} {r['category']:8} products={r['product_count']:3} "
f"price ₹{r['min_price']}–₹{r['max_price']} max_platforms={r['max_platforms']}")
typer.echo("\nListings by site and source type:")
for r in conn.execute("SELECT s.name, l.source_type, count(*) n, count(l.price) priced "
"FROM elec.source_listing l JOIN elec.site s ON s.id = l.site_id "
"GROUP BY 1, 2 ORDER BY 1, 2"):
typer.echo(f" {r['name']:22} {r['source_type']:15} {r['n']:4} (with price: {r['priced']})")
@app.command()
def review(approve: Optional[int] = typer.Option(None, help="listing id to approve"),
reject: Optional[int] = typer.Option(None, help="listing id to reject")) -> None:
"""Show uncertain listing-to-product matches, or approve/reject one."""
from app.electronics.db import repository as repo
if approve or reject:
ok = repo.set_review(approve or reject, approve is not None)
refreshed = repo.refresh_verification()
typer.echo(f"{'updated' if ok else 'nothing pending for that listing'}; products: {refreshed}")
return
for r in repo.review_queue():
typer.echo(f"[{r['listing_id']}] {r['site']}: {r['listing_title']}\n -> {r['product']} "
f"({r['method']}, {r['confidence']}) {r['source_url']}")
@app.command("verify-grounding")
def verify_grounding(sample: int = 100) -> None:
"""Audit: every stored price must appear in the evidence text stored with it."""
from app.electronics.db import repository as repo
bad = 0
rows = repo.grounding_sample(sample)
for r in rows:
digits = re.sub(r"\D", "", r["evidence_text"].replace(".00", ""))
price = r["price"]
whole = str(int(price)) if price == price.to_integral() else str(price)
if whole.replace(".", "") not in digits:
bad += 1
typer.echo(f"NOT GROUNDED listing {r['id']}: price {price} not in evidence ({r['source_url']})")
typer.echo(f"Checked {len(rows)} priced listings; {bad} without evidence.")
raise typer.Exit(code=1 if bad else 0)
if __name__ == "__main__":
app()

View File

@@ -0,0 +1,581 @@
"""Search-first collection of real product listings.
For one category and a set of allow-listed brands:
1. DISCOVER web search `site:<platform> <brand> <category term>` on every
registered platform (marketplaces, national chains, Tamil Nadu
chains, the brand's own site). Only URLs that are single product
pages on a registered platform are kept.
2. EXPAND for each model found, search `<brand> <model> price` to find the
same model on other platforms.
3. COLLECT per URL, by the platform's probe grade:
A/B and breaker closed -> fetch the page politely and read
JSON-LD / meta / spec tables
C (or fetch refused) -> use the search result itself: its
title, snippet price and stock text
4. MATCH link the listing to one canonical variant (match.matcher)
5. ENRICH specs (deterministic, LLM gap-fill grounded in page text) and
images (only from the product's own listings, validated live)
6. VERIFY products with listings on ≥2 sites (≥1 a retailer) become
verified and visible.
Nothing in this module invents a product, price or image: every value is read
from a page or a search result, and stored with that URL and text.
"""
from __future__ import annotations
import hashlib
import json
import logging
import re
from dataclasses import dataclass, field
from decimal import Decimal
from typing import Callable, Dict, List, Optional, Tuple
from urllib.parse import urlparse
from rapidfuzz import fuzz
from app.electronics.db import repository as repo
from app.electronics.extract.html_fallback import extract_page, spec_tables, visible_text
from app.electronics.extract.jsonld import extract_products
from app.electronics.extract.serp_parser import clean_result_title, read_price, read_rating, read_stock
from app.electronics.match.matcher import decide
from app.electronics.models import Listing
from app.electronics.net.breaker import CircuitBreaker
from app.electronics.net.polite_client import PoliteClient
from app.electronics.normalise.brand_alias import looks_like_device_title
from app.electronics.normalise.llm_fill import fill_missing
from app.electronics.normalise.spec_normaliser import normalise_specs
from app.electronics.normalise.title_parser import ParsedTitle, fill_from_context, parse_title, variant_key
from app.electronics.probe.site_probe import probe_site
from app.electronics.reference import BrandRef, SiteRef, load_reference, site_for_url
from app.electronics.search.engine import SearchEngine
from app.electronics.search.providers import SearchHit
from app.infrastructure.settings import ELEC_PROBE_TTL_DAYS, MIN_IMAGE_BYTES
from bs4 import BeautifulSoup
logger = logging.getLogger(__name__)
# Titles that belong to another category even when the brand matches.
_OFF_CATEGORY = {
"mobiles": re.compile(r"\b(?:tab|tablet|pad|watch|buds|earbuds|laptop|book|monitor|tv|television|band)\b", re.I),
"laptops": re.compile(r"\b(?:tablet|tab|monitor|mouse|keyboard|phone|smartphone|printer|desktop|all[- ]in[- ]one)\b", re.I),
}
_LISTING_PAGE = re.compile(r"/(?:search|s|c|category|categories|brand|brands|compare|offers?|deals?)(?:/|\?|$)", re.I)
@dataclass
class RunOptions:
category: str
brands: List[str] # brand slugs
max_products_per_brand: int = 15
expand_per_brand: int = 8 # models to look up on other platforms
search_budget: int = 200
use_llm: bool = True
fetch_pages: bool = True
find_images: bool = True
reprobe: bool = False
@dataclass
class RunStats:
counts: Dict[str, int] = field(default_factory=dict)
def inc(self, key: str, n: int = 1) -> None:
self.counts[key] = self.counts.get(key, 0) + n
def source_sku(site: SiteRef, url: str) -> str:
rx = site.product_url_re
if rx is not None:
m = rx.search(url)
if m and m.groups() and m.group(1):
return m.group(1)
p = urlparse(url)
return (p.netloc.lower().removeprefix("www.") + p.path.rstrip("/").lower())[:300]
_INDIA_PATH = re.compile(r"^/(?:in|in-en|en-in|en_in|in_en)(?:/|$)", re.I)
def is_product_url(site: SiteRef, url: str) -> bool:
p = urlparse(url)
if _LISTING_PAGE.search(p.path):
return False
if site.kind == "brand_official":
# Only the brand's Indian storefront: www.samsung.com/in/..., not
# us.samsung.com or news.samsung.com. Domains that are Indian already
# (oneplus.in, motorola.co.in) qualify as they are.
host = (p.hostname or "").lower()
if host not in (site.domain, "www." + site.domain, "in." + site.domain):
return False
if not site.domain.endswith((".in", ".co.in")) and not _INDIA_PATH.search(p.path) and not host.startswith("in."):
return False
rx = site.product_url_re
if rx is not None:
return bool(rx.search(url))
return p.path.count("/") >= 2 # brand sites: at least /section/product
def embed_verified_products() -> int:
"""MiniLM vectors for verified products that do not have one yet."""
rows = repo.products_without_embedding()
if not rows:
return 0
from app.services.embeddings_service import embed_texts
texts = [
f"{r['brand']} {r['display_name']} {r['category']} "
+ " ".join(f"{k} {v}" for k, v in (r["canonical_specs"] or {}).items())
for r in rows
]
for r, vec in zip(rows, embed_texts(texts)):
repo.set_embedding(r["id"], vec)
return len(rows)
class Collector:
def __init__(self, options: RunOptions, *, progress: Optional[Callable[[str], None]] = None) -> None:
self.opt = options
self.ref = load_reference()
self.stats = RunStats()
self.progress = progress or (lambda msg: logger.info(msg))
self.run_id: Optional[int] = None
self.ids = repo.id_maps()
self.site_rows = {r["domain"]: r for r in repo.sites()}
self.breaker = CircuitBreaker(on_trip=self._on_trip)
for r in self.site_rows.values():
if r.get("breaker_until"):
self.breaker.preload(r["domain"], r["breaker_until"].timestamp(), r.get("breaker_reason") or "")
self.client = PoliteClient(breaker=self.breaker, on_fetch=self._on_fetch)
self.engine = SearchEngine(budget=options.search_budget)
self._touched_products: Dict[int, List[Tuple[int, Listing]]] = {}
# -- callbacks -----------------------------------------------------------
def _on_trip(self, host: str, reason: str, until: float) -> None:
self.stats.inc("breaker_trips")
self.progress(f"Circuit breaker opened for {host}: {reason}. Falling back to web search for it.")
repo.trip_breaker(host, reason, until)
def _on_fetch(self, result, host: str) -> None:
self.stats.inc(f"fetch_{result.outcome}")
repo.log_fetch(self.run_id, result.url, host, result.status, result.bytes, result.outcome, result.robots_allowed)
# -- grading -------------------------------------------------------------
def _grade(self, site: SiteRef) -> str:
if site.policy == "serp_only":
return "C"
host = site.domain
if self.breaker.is_open(host) or self.breaker.is_open("www." + host):
return "C"
return (self.site_rows.get(site.domain) or {}).get("probe_outcome") or "C"
def ensure_probes(self, sites: List[SiteRef]) -> None:
from datetime import datetime, timedelta, timezone
stale_before = datetime.now(timezone.utc) - timedelta(days=ELEC_PROBE_TTL_DAYS)
for site in sites:
row = self.site_rows.get(site.domain) or {}
if not self.opt.reprobe and row.get("probed_at") and row["probed_at"] > stale_before:
continue
self.progress(f"Probing {site.name} ({site.domain})")
result = probe_site(site, self.client, self.engine)
repo.set_probe_result(site.domain, result["outcome"], result["robots_allowed"], result["evidence"])
row.update(probe_outcome=result["outcome"])
self.site_rows[site.domain] = row
self.stats.inc(f"probe_{result['outcome']}")
self.progress(f" -> grade {result['outcome']}: {result['evidence'].get('reason')}")
# -- discovery -----------------------------------------------------------
def _accept_hit(self, hit: SearchHit, brand: BrandRef) -> Optional[Tuple[SiteRef, ParsedTitle]]:
site = site_for_url(hit.url)
if site is None or not self._site_allowed_for(site, brand):
return None
if not is_product_url(site, hit.url):
return None
title = clean_result_title(hit.title)
if not looks_like_device_title(title) or _OFF_CATEGORY[self.opt.category].search(title):
return None
parsed = parse_title(title, self.opt.category, expected_brand=brand.slug)
if parsed.brand is None or not parsed.model_norm:
return None
fill_from_context(parsed, self.opt.category, snippet=hit.snippet or "")
return site, parsed
def _site_allowed_for(self, site: SiteRef, brand: BrandRef) -> bool:
if site.kind == "brand_official":
return site.brand_slug == brand.slug
return True
def _platforms_for(self, brand: BrandRef) -> List[SiteRef]:
return [s for s in self.ref.sites.values()
if s.kind != "brand_official" or s.brand_slug == brand.slug]
def _search_names(self, brand: BrandRef) -> List[str]:
"""The brand, plus the sub-brands phones are actually sold under
("Redmi", "POCO", "iQOO") - a search for "Xiaomi" alone misses most
Redmi listings."""
names = [brand.name]
if self.opt.category == "mobiles":
names += [s.upper() if len(s) <= 4 else s.title() for s in brand.sub_brands
if s not in ("mi", "iphone", "pixel", "narzo")][:2]
return names
def _collect_hits(self, hits: Optional[List[SearchHit]], brand: BrandRef, query: str,
found: Dict, models: Dict[str, ParsedTitle]) -> int:
new = 0
for hit in hits or []:
accepted = self._accept_hit(hit, brand)
if not accepted:
continue
site_ref, parsed = accepted
key = (site_ref.domain, source_sku(site_ref, hit.url))
if key not in found:
found[key] = (hit, site_ref, parsed, query)
new += 1
models.setdefault(parsed.model_norm, parsed)
return new
def discover(self, brand: BrandRef) -> Dict[Tuple[str, str], Tuple[SearchHit, SiteRef, ParsedTitle, str]]:
category = self.ref.categories[self.opt.category]
found: Dict[Tuple[str, str], Tuple[SearchHit, SiteRef, ParsedTitle, str]] = {}
models: Dict[str, ParsedTitle] = {}
per_site_target = max(4, self.opt.max_products_per_brand)
for site in self._platforms_for(brand):
got = 0
for name in self._search_names(brand):
for terms in category.query_terms or category.search_terms:
query = f"site:{site.domain} {name} {terms}"
hits = self.engine.text(query, max_results=20)
if hits is None:
self.stats.inc("search_unavailable")
continue
got += self._collect_hits(hits, brand, query, found, models)
if got >= per_site_target:
break
if got >= per_site_target:
break
self.progress(f"{brand.name}: {len(found)} listing URLs, {len(models)} models from platform searches")
# Cross-platform: look each variant up by name, to find the same product
# on platforms the site: searches missed. Phones are grouped by model
# line; laptops by full configuration (line + CPU + RAM + storage),
# because one laptop line is sold in dozens of configurations and only
# the exact one confirms a product.
for p in self._expansion_targets(found):
query = f"{brand.name} {p.model or p.model_norm} {self._variant_terms(p)} price"
self._collect_hits(self.engine.text(re.sub(r"\s+", " ", query), max_results=20), brand, query, found, models)
self.progress(f"{brand.name}: {len(found)} listing URLs after cross-platform search")
return found
def _expansion_targets(self, found: Dict) -> List[ParsedTitle]:
"""Which variants to look up on other platforms, most useful first:
variants seen on the most sites, then ones whose page we can read with
a price (grade A/B platforms), since one more site verifies those."""
groups: Dict[str, Dict] = {}
for (domain, _), (_, site, parsed, _) in found.items():
if self.opt.category == "laptops":
key = variant_key(parsed, "laptops")
if key is None:
continue
else:
key = parsed.model_norm
g = groups.setdefault(key, {"parsed": parsed, "sites": set(), "readable": False})
g["sites"].add(domain)
g["readable"] = g["readable"] or self._grade(site) in ("A", "B")
ranked = sorted(groups.values(), key=lambda g: (-len(g["sites"]), not g["readable"]))
return [g["parsed"] for g in ranked[: self.opt.expand_per_brand]]
@staticmethod
def _variant_terms(p: ParsedTitle) -> str:
parts = []
if p.processor:
# "ryzen 5 7530u" -> "Ryzen 5 7530U", "i5-1334u" -> "i5-1334U"
parts.append(" ".join(t.upper() if any(ch.isdigit() for ch in t) and len(t) > 2 else t.title()
for t in p.processor.split()))
if p.ram_gb:
parts.append(f"{format(p.ram_gb.normalize(), 'f')}GB RAM")
if p.storage_gb:
parts.append(f"{format(p.storage_gb.normalize(), 'f')}GB")
return " ".join(parts)
# -- listing construction --------------------------------------------------
def _base_listing(self, site: SiteRef, url: str, parsed: ParsedTitle, title: str, source_type: str,
evidence: str, confidence: float, parser: str, query: str) -> Listing:
l = Listing(
site_domain=site.domain, source_sku=source_sku(site, url), source_url=url, source_type=source_type,
brand_slug=parsed.brand.brand_slug, category=self.opt.category, title=title,
evidence_text=evidence, confidence=confidence, parser=parser, family=parsed.brand.family,
model=parsed.model, model_number=parsed.mpn, ram_gb=parsed.ram_gb, storage_gb=parsed.storage_gb,
colour=parsed.colour, search_query=query,
)
l.model_norm = parsed.model_norm
l.processor = parsed.processor
l.variant_key = variant_key(parsed, self.opt.category)
return l
def listing_from_search(self, hit: SearchHit, site: SiteRef, parsed: ParsedTitle, query: str) -> Listing:
title = clean_result_title(hit.title)
evidence = f"{hit.title} — {hit.snippet}".strip(" —")
reading = read_price(f"{hit.title} {hit.snippet}")
price, mrp = reading.price, reading.mrp
# A snippet that names a different RAM/storage than the title is about
# another variant; its price cannot be trusted for this one.
snippet_variant = parse_title(hit.snippet or "", self.opt.category)
for a, b in ((snippet_variant.storage_gb, parsed.storage_gb), (snippet_variant.ram_gb, parsed.ram_gb)):
if a is not None and b is not None and a != b:
price = mrp = None
self.stats.inc("snippet_price_variant_conflict")
# Truncated titles ("... - (16 GB ...") lose the variant; the snippet
# of the same result usually states it.
fill_from_context(parsed, self.opt.category, snippet=hit.snippet or "")
parser, confidence = f"serp:{hit.provider}", (0.55 if price is not None else 0.45)
in_stock = read_stock(hit.snippet or "")
availability = None if in_stock is None else ("InStock" if in_stock else "OutOfStock")
# Structured offer the search engine read from the page itself (Google
# pagemap). Better than snippet text, and still no request to the site.
if hit.offer and price is None:
try:
offered = Decimal(str(hit.offer["price"]).replace(",", ""))
except Exception: # noqa: BLE001
offered = None
if offered is not None and Decimal(500) <= offered <= Decimal(1000000):
price, mrp = offered, None
evidence = f"{evidence} || search-engine offer data: {json.dumps(hit.offer['raw'], default=str)[:600]}"
parser, confidence = f"serp:{hit.provider}:pagemap", 0.65
av = (hit.offer.get("availability") or "").lower().replace(" ", "")
if "instock" in av:
in_stock, availability = True, "InStock"
elif "outofstock" in av or "soldout" in av:
in_stock, availability = False, "OutOfStock"
# Rating: the engine's structured data first (read from the page
# itself), else an explicit "x out of 5" in this result's own text.
rating, review_count = None, None
if hit.rating and hit.rating.get("rating") is not None:
rating = Decimal(str(hit.rating["rating"]))
review_count = hit.rating.get("review_count")
evidence = f"{evidence} || search-engine rating data: {json.dumps(hit.rating.get('raw'), default=str)[:300]}"
else:
stated = read_rating(f"{hit.title} {hit.snippet}")
if stated.rating is not None:
rating, review_count = stated.rating, stated.review_count
listing = self._base_listing(site, hit.url, parsed, title, "search_snippet", evidence,
confidence, parser, query)
listing.price, listing.mrp = price, mrp
listing.in_stock, listing.availability = in_stock, availability
listing.rating, listing.review_count = rating, review_count
return listing
def listing_from_page(self, hit: SearchHit, site: SiteRef, parsed_hit: ParsedTitle, query: str,
html: str, final_url: str) -> Optional[Listing]:
products = extract_products(html)
page = extract_page(html)
name = None
product = None
for p in products:
pp = parse_title(p["name"], self.opt.category, expected_brand=parsed_hit.brand.brand_slug)
if pp.brand and pp.model_norm and fuzz.token_set_ratio(pp.model_norm, parsed_hit.model_norm) >= 85:
product, name = p, p["name"]
break
if name is None:
name = page.get("name")
if not name:
return None
parsed = parse_title(name, self.opt.category, expected_brand=parsed_hit.brand.brand_slug)
if parsed.brand is None or not parsed.model_norm:
return None
if fuzz.token_set_ratio(parsed.model_norm, parsed_hit.model_norm) < 85:
# The URL did not lead to the product the search result named.
self.stats.inc("page_title_mismatch")
return None
# Fill variant fields the page name leaves out from the search title
# of the same URL (both are statements by the same site).
for attr in ("ram_gb", "storage_gb", "colour", "mpn", "processor"):
if getattr(parsed, attr) is None and getattr(parsed_hit, attr) is not None:
setattr(parsed, attr, getattr(parsed_hit, attr))
grade = self._grade(site)
source_type = "brand_official" if site.kind == "brand_official" else "scraped_page"
if product is not None:
price = product.get("price")
currency = product.get("currency")
evidence = product["evidence"]
parser = "jsonld"
else:
price, currency, evidence, parser = page.get("price"), page.get("currency"), page.get("evidence") or "", "html_meta"
if currency not in (None, "INR"):
price = None
if currency is None and site.kind == "brand_official":
price = None # a brand's global site may not be quoting rupees
if price is not None and not (Decimal(500) <= price <= Decimal(1000000)):
price = None
evidence = evidence or f"{name} ({final_url})"
listing = self._base_listing(
site, final_url if final_url.startswith("http") else hit.url, parsed, name, source_type,
evidence, 0.9 if (price is not None and parser == "jsonld") else 0.75, f"{parser}:grade{grade}", query,
)
# The site's own SKU when the page states it; otherwise its canonical URL.
listing.source_sku = ((product or {}).get("sku") or source_sku(site, final_url or hit.url))[:300]
listing.price = price
listing.in_stock = (product or page).get("in_stock")
listing.availability = (product or page).get("availability")
listing.image_urls = list(dict.fromkeys((product or {}).get("images", []) + page.get("images", [])))[:8]
if product:
listing.gtin = product.get("gtin")
listing.model_number = listing.model_number or product.get("mpn")
listing.rating = product.get("rating")
listing.review_count = product.get("review_count")
listing.reviews = list(product.get("reviews") or [])
listing.colour = listing.colour or product.get("color")
raw_specs = dict((product or {}).get("properties") or {})
raw_specs.update({k: v for k, v in spec_tables(BeautifulSoup(html, "lxml")).items() if k not in raw_specs})
listing.specs_raw = dict(list(raw_specs.items())[:150])
listing.specs, listing.spec_sources = normalise_specs(self.opt.category, raw_specs)
if self.opt.use_llm:
wanted = [k for k in self.ref.spec_keys.get(self.opt.category, {}) if k not in listing.specs]
if wanted:
text = "\n".join(f"{k}: {v}" for k, v in raw_specs.items()) or visible_text(html, 3500)
extra, extra_src = fill_missing(self.opt.category, text, wanted)
listing.specs.update(extra)
listing.spec_sources.update(extra_src)
self.stats.inc("llm_specs_kept", len(extra))
if self.opt.category == "laptops":
# "13th Gen Intel Core i7/ 16GB RAM" in a title names no CPU model;
# the page's own spec table usually does.
spec_texts = tuple(str(v) for k, v in raw_specs.items() if "processor" in k.lower() or "cpu" in k.lower())
spec_texts += (str(listing.specs.get("processor") or ""),)
before = parsed.processor
fill_from_context(parsed, self.opt.category, spec_texts=spec_texts)
if parsed.processor != before:
listing.processor = parsed.processor
listing.variant_key = variant_key(parsed, self.opt.category)
listing.content_hash = hashlib.sha1(html.encode("utf-8", "ignore")).hexdigest()
return listing
# -- persistence -----------------------------------------------------------
def store(self, listing: Listing) -> Optional[int]:
try:
listing_id = repo.upsert_listing(listing, self.ids, self.run_id)
except ValueError as exc:
self.stats.inc("rejected_listing")
logger.info("Listing rejected (%s): %s", exc, listing.source_url)
return None
self.stats.inc(f"listing_{listing.source_type}")
if listing.reviews:
self.stats.inc("reviews_stored", repo.save_reviews(listing_id, listing.reviews))
if listing.price is not None:
self.stats.inc("listing_with_price")
decision = decide(listing, repo.product_candidates(listing.brand_slug, listing.category))
if decision is None:
self.stats.inc("listing_unmatched_no_variant")
return listing_id
product_id = decision.product_id or repo.create_product(listing, self.ids)
if decision.product_id is None:
self.stats.inc("product_created")
repo.map_listing(listing_id, product_id, decision.method, decision.confidence, decision.review_status)
if decision.review_status == "pending":
self.stats.inc("match_pending_review")
else:
repo.merge_product_specs(product_id, listing.specs, listing.spec_sources, listing.source_url)
self._touched_products.setdefault(product_id, []).append((listing_id, listing))
return listing_id
# -- images ----------------------------------------------------------------
def attach_images(self) -> None:
rank_for = {"brand_official": 10, "scraped_page": 20, "search_snippet": 50}
for product_id, entries in self._touched_products.items():
if repo.product_image_count(product_id) >= 3:
continue
added = 0
for listing_id, listing in sorted(entries, key=lambda e: rank_for[e[1].source_type]):
for url in listing.image_urls:
if added >= 3:
break
if self.client.check_image(url, MIN_IMAGE_BYTES):
repo.add_image(product_id, url, listing_id, listing.source_type, rank_for[listing.source_type])
added += 1
self.stats.inc("images_from_pages")
if added or not self.opt.find_images:
continue
added = self._images_from_search(product_id, entries)
self.stats.inc("images_from_search", added)
def _images_from_search(self, product_id: int, entries: List[Tuple[int, Listing]]) -> int:
"""Image search results are used only when the page an image sits on is
one of THIS product's own listings (same site, same product), and the
image result's title names the model."""
_, listing = entries[0]
brand = self.ref.brands[listing.brand_slug]
listing_by_site = {l.site_domain: lid for lid, l in entries}
hits = self.engine.images(f"{brand.name} {listing.model or listing.model_norm}", max_results=15) or []
added = 0
for hit in hits:
site = site_for_url(hit.url)
if site is None or site.domain not in listing_by_site:
continue
parsed = parse_title(clean_result_title(hit.title), listing.category, expected_brand=brand.slug)
if not parsed.model_norm or fuzz.token_set_ratio(parsed.model_norm, listing.model_norm or "") < 90:
continue
if self.client.check_image(hit.image_url, MIN_IMAGE_BYTES):
repo.add_image(product_id, hit.image_url, listing_by_site[site.domain], "search_image", 60)
added += 1
if added >= 2:
break
return added
# -- embeddings ------------------------------------------------------------
def embed(self) -> int:
return embed_verified_products()
# -- the run ---------------------------------------------------------------
def run(self, *, embed: bool = True) -> Dict[str, int]:
self.run_id = repo.start_run("collect", {
"category": self.opt.category, "brands": self.opt.brands,
"max_products_per_brand": self.opt.max_products_per_brand,
"search_budget": self.opt.search_budget,
})
status, error = "done", None
try:
brands = [self.ref.brands[b] for b in self.opt.brands]
probe_targets = {s.domain: s for b in brands for s in self._platforms_for(b) if s.policy == "probe"}
if self.opt.fetch_pages:
self.ensure_probes(list(probe_targets.values()))
for brand in brands:
found = self.discover(brand)
# Keep the most common models first, up to the per-brand limit.
by_model: Dict[str, int] = {}
for (_, _), (_, _, parsed, _) in found.items():
by_model[parsed.model_norm] = by_model.get(parsed.model_norm, 0) + 1
keep = set(sorted(by_model, key=lambda m: -by_model[m])[: self.opt.max_products_per_brand])
for (domain, _sku), (hit, site, parsed, query) in found.items():
if parsed.model_norm not in keep:
continue
listing = None
if self.opt.fetch_pages and self._grade(site) in ("A", "B"):
res = self.client.get(hit.url)
if res.ok:
listing = self.listing_from_page(hit, site, parsed, query, res.text, res.final_url)
if listing is None:
self.stats.inc("page_unusable_fell_back_to_search")
if listing is None:
listing = self.listing_from_search(hit, site, parsed, query)
self.store(listing)
self.progress(f"{brand.name}: stored listings; {self.stats.counts}")
self.attach_images()
verification = repo.refresh_verification()
self.stats.counts.update({f"products_{k}": v for k, v in verification.items()})
if embed:
try:
self.stats.inc("embedded", self.embed())
except Exception as exc: # noqa: BLE001 - embeddings are optional
logger.warning("Embedding step skipped: %s", exc)
self.stats.counts.update({f"search_{k}": v for k, v in self.engine.stats.items()})
except Exception as exc:
status, error = "failed", repr(exc)
logger.exception("Collection run failed")
raise
finally:
repo.finish_run(self.run_id, status, self.stats.counts, error)
self.client.close()
return self.stats.counts

View File

View File

@@ -0,0 +1,68 @@
"""Connections to the local electronics database.
settings.py has already refused to load unless DB_HOST is local and DB_NAME is
electronics_catalog, so nothing here can reach another database.
"""
from __future__ import annotations
import logging
from contextlib import contextmanager
from typing import Iterator
import psycopg
from psycopg.rows import dict_row
from app.infrastructure.settings import (
DB_CONNECT_TIMEOUT_SECONDS,
DB_HOST,
DB_NAME,
DB_PASSWORD,
DB_PORT,
DB_USER,
)
logger = logging.getLogger(__name__)
def connect(*, autocommit: bool = False) -> psycopg.Connection:
conn = psycopg.connect(
host=DB_HOST,
port=DB_PORT,
dbname=DB_NAME,
user=DB_USER,
password=DB_PASSWORD,
connect_timeout=DB_CONNECT_TIMEOUT_SECONDS,
autocommit=autocommit,
row_factory=dict_row,
)
try:
from pgvector.psycopg import register_vector
register_vector(conn)
except Exception: # noqa: BLE001 - the extension is created by migration 0001
pass
return conn
@contextmanager
def transaction() -> Iterator[psycopg.Connection]:
"""A connection whose work is committed on success, rolled back on error."""
conn = connect()
try:
yield conn
conn.commit()
except Exception:
conn.rollback()
raise
finally:
conn.close()
def check_connection() -> bool:
try:
with connect(autocommit=True) as conn:
conn.execute("SELECT 1")
return True
except Exception as exc: # noqa: BLE001 - a health probe reports, never raises
logger.debug("Database unreachable: %s", exc)
return False

View File

@@ -0,0 +1,63 @@
"""Tiny migration runner for the numbered SQL files in ./migrations.
Each file runs once, in its own transaction, and is recorded in
elec.schema_migrations with a checksum. Editing an applied file is refused
rather than silently ignored - add a new numbered file instead.
"""
from __future__ import annotations
import hashlib
import logging
from pathlib import Path
from typing import List
from app.electronics.db.connection import connect
logger = logging.getLogger(__name__)
MIGRATIONS_DIR = Path(__file__).resolve().parent / "migrations"
_BOOTSTRAP = """
CREATE SCHEMA IF NOT EXISTS elec;
CREATE TABLE IF NOT EXISTS elec.schema_migrations (
version TEXT PRIMARY KEY,
checksum TEXT NOT NULL,
applied_at TIMESTAMPTZ NOT NULL DEFAULT now()
);
"""
def _files() -> List[Path]:
return sorted(MIGRATIONS_DIR.glob("[0-9][0-9][0-9][0-9]_*.sql"))
def run_migrations() -> List[str]:
"""Apply pending migrations. Returns the versions applied by this call."""
applied_now: List[str] = []
with connect() as conn:
conn.execute(_BOOTSTRAP)
conn.commit()
done = {
r["version"]: r["checksum"]
for r in conn.execute("SELECT version, checksum FROM elec.schema_migrations")
}
for path in _files():
sql = path.read_text(encoding="utf-8")
checksum = hashlib.sha256(sql.encode("utf-8")).hexdigest()
version = path.stem
if version in done:
if done[version] != checksum:
raise RuntimeError(
f"Migration {version} was edited after it was applied. "
f"Revert the edit and add a new numbered migration instead."
)
continue
logger.info("Applying migration %s", version)
with conn.transaction():
conn.execute(sql)
conn.execute(
"INSERT INTO elec.schema_migrations (version, checksum) VALUES (%s, %s)",
(version, checksum),
)
applied_now.append(version)
return applied_now

View File

@@ -0,0 +1,4 @@
-- Extensions and the dedicated schema. Everything this project owns lives in
-- schema `elec` of database `electronics_catalog`.
CREATE EXTENSION IF NOT EXISTS vector;
CREATE SCHEMA IF NOT EXISTS elec;

View File

@@ -0,0 +1,57 @@
-- Reference data: brands, categories, retail sites. Seeded from
-- app/electronics/reference/*.yaml by `elec seed-reference`.
CREATE TABLE elec.brand (
id SERIAL PRIMARY KEY,
name TEXT NOT NULL UNIQUE,
slug TEXT NOT NULL UNIQUE,
parent_brand_id INT REFERENCES elec.brand(id),
is_popular BOOLEAN NOT NULL DEFAULT TRUE,
official_domains TEXT[] NOT NULL DEFAULT '{}',
created_at TIMESTAMPTZ NOT NULL DEFAULT now()
);
-- Every spelling that resolves to a brand. Sub-brands (Redmi, iQOO, Pixel)
-- resolve to their parent and are remembered as the product family.
CREATE TABLE elec.brand_alias (
alias TEXT PRIMARY KEY CHECK (alias = lower(alias)),
brand_id INT NOT NULL REFERENCES elec.brand(id) ON DELETE CASCADE,
is_sub_brand BOOLEAN NOT NULL DEFAULT FALSE
);
CREATE TABLE elec.category (
id SERIAL PRIMARY KEY,
slug TEXT NOT NULL UNIQUE,
name TEXT NOT NULL UNIQUE
);
CREATE TABLE elec.brand_category (
brand_id INT NOT NULL REFERENCES elec.brand(id) ON DELETE CASCADE,
category_id INT NOT NULL REFERENCES elec.category(id) ON DELETE CASCADE,
PRIMARY KEY (brand_id, category_id)
);
-- A retail platform or a brand's own site, with the outcome of its probe.
-- probe_outcome A = fetchable with structured product data (scraped)
-- B = fetchable, product data from page HTML/state (scraped)
-- C = not fetched: serp_only policy, robots.txt disallow,
-- block/CAPTCHA, or unreachable -> web search only
CREATE TABLE elec.site (
id SERIAL PRIMARY KEY,
domain TEXT NOT NULL UNIQUE,
name TEXT NOT NULL,
kind TEXT NOT NULL CHECK (kind IN ('marketplace','national_chain','tn_regional','brand_official')),
region TEXT NOT NULL CHECK (region IN ('national','TN')),
policy TEXT NOT NULL CHECK (policy IN ('probe','serp_only')),
brand_id INT REFERENCES elec.brand(id),
product_url TEXT,
pincode_param TEXT,
enabled BOOLEAN NOT NULL DEFAULT TRUE,
probe_outcome CHAR(1) CHECK (probe_outcome IN ('A','B','C')),
robots_allowed BOOLEAN,
probe_evidence JSONB NOT NULL DEFAULT '{}'::jsonb,
probed_at TIMESTAMPTZ,
breaker_until TIMESTAMPTZ,
breaker_reason TEXT,
CHECK (kind <> 'brand_official' OR brand_id IS NOT NULL)
);

View File

@@ -0,0 +1,160 @@
-- Runs, fetch audit trail, search cache, listings, prices, canonical products.
CREATE TABLE elec.crawl_run (
id BIGSERIAL PRIMARY KEY,
kind TEXT NOT NULL,
params JSONB NOT NULL DEFAULT '{}'::jsonb,
status TEXT NOT NULL DEFAULT 'running' CHECK (status IN ('running','done','failed')),
stats JSONB NOT NULL DEFAULT '{}'::jsonb,
error TEXT,
started_at TIMESTAMPTZ NOT NULL DEFAULT now(),
ended_at TIMESTAMPTZ
);
-- Every HTTP request made to a retail or brand site. Evidence that the
-- crawler obeyed robots.txt and its rate limits.
CREATE TABLE elec.fetch_log (
id BIGSERIAL PRIMARY KEY,
crawl_run_id BIGINT REFERENCES elec.crawl_run(id) ON DELETE SET NULL,
url TEXT NOT NULL,
host TEXT NOT NULL,
status INT,
bytes INT,
outcome TEXT NOT NULL,
robots_allowed BOOLEAN,
fetched_at TIMESTAMPTZ NOT NULL DEFAULT now()
);
CREATE INDEX fetch_log_host_time ON elec.fetch_log (host, fetched_at DESC);
CREATE TABLE elec.search_cache (
provider TEXT NOT NULL,
kind TEXT NOT NULL CHECK (kind IN ('text','images')),
query TEXT NOT NULL,
results JSONB NOT NULL,
fetched_at TIMESTAMPTZ NOT NULL DEFAULT now(),
PRIMARY KEY (provider, kind, query)
);
-- One canonical product = one real-world variant (model + RAM + storage).
-- verification_status becomes 'verified' only when the product has a brand
-- official page, or listings on at least two different sites.
CREATE TABLE elec.product (
id BIGSERIAL PRIMARY KEY,
brand_id INT NOT NULL REFERENCES elec.brand(id),
category_id INT NOT NULL REFERENCES elec.category(id),
family TEXT,
model TEXT NOT NULL,
model_norm TEXT NOT NULL,
variant_key TEXT NOT NULL UNIQUE,
display_name TEXT NOT NULL,
ram_gb NUMERIC(6,1),
storage_gb NUMERIC(7,1),
processor TEXT,
mpn TEXT,
gtin TEXT,
canonical_specs JSONB NOT NULL DEFAULT '{}'::jsonb,
spec_sources JSONB NOT NULL DEFAULT '{}'::jsonb,
verification_status TEXT NOT NULL DEFAULT 'unverified'
CHECK (verification_status IN ('verified','unverified','rejected')),
evidence_count INT NOT NULL DEFAULT 0,
embedding vector(384),
created_at TIMESTAMPTZ NOT NULL DEFAULT now(),
updated_at TIMESTAMPTZ NOT NULL DEFAULT now()
);
CREATE INDEX product_brand_cat ON elec.product (brand_id, category_id);
CREATE INDEX product_specs_gin ON elec.product USING GIN (canonical_specs);
CREATE INDEX product_embedding_hnsw ON elec.product USING hnsw (embedding vector_cosine_ops);
-- The latest state of one product page on one site, or of one search result
-- that points at such a page. Nothing is stored without the URL it came from
-- and the text that the values were read from.
CREATE TABLE elec.source_listing (
id BIGSERIAL PRIMARY KEY,
site_id INT NOT NULL REFERENCES elec.site(id),
source_sku TEXT NOT NULL,
source_url TEXT NOT NULL CHECK (source_url ~ '^https?://'),
source_type TEXT NOT NULL CHECK (source_type IN ('scraped_page','search_snippet','brand_official')),
brand_id INT NOT NULL REFERENCES elec.brand(id),
category_id INT NOT NULL REFERENCES elec.category(id),
family TEXT,
title TEXT NOT NULL,
model TEXT,
model_number TEXT,
ram_gb NUMERIC(6,1),
storage_gb NUMERIC(7,1),
colour TEXT,
price NUMERIC(12,2) CHECK (price IS NULL OR price BETWEEN 500 AND 1000000),
mrp NUMERIC(12,2) CHECK (mrp IS NULL OR mrp BETWEEN 500 AND 1000000),
currency TEXT NOT NULL DEFAULT 'INR' CHECK (currency = 'INR'),
availability TEXT,
in_stock BOOLEAN,
pincode TEXT,
pincode_applied BOOLEAN NOT NULL DEFAULT FALSE,
rating NUMERIC(3,2) CHECK (rating IS NULL OR rating BETWEEN 0 AND 5),
review_count INT,
gtin TEXT,
image_urls TEXT[] NOT NULL DEFAULT '{}',
specs_raw JSONB NOT NULL DEFAULT '{}'::jsonb,
specs JSONB NOT NULL DEFAULT '{}'::jsonb,
evidence_text TEXT NOT NULL CHECK (length(evidence_text) > 0),
search_query TEXT,
confidence NUMERIC(3,2) NOT NULL CHECK (confidence BETWEEN 0 AND 1),
parser TEXT NOT NULL,
content_hash TEXT,
first_seen_at TIMESTAMPTZ NOT NULL DEFAULT now(),
last_seen_at TIMESTAMPTZ NOT NULL DEFAULT now(),
crawl_run_id BIGINT REFERENCES elec.crawl_run(id) ON DELETE SET NULL,
UNIQUE (site_id, source_sku),
CHECK (pincode_applied = FALSE OR pincode IS NOT NULL)
);
CREATE INDEX listing_brand_cat ON elec.source_listing (brand_id, category_id);
-- Append-only price observations. UPDATE is refused by a trigger.
CREATE TABLE elec.price_history (
id BIGSERIAL PRIMARY KEY,
listing_id BIGINT NOT NULL REFERENCES elec.source_listing(id) ON DELETE CASCADE,
price NUMERIC(12,2) CHECK (price IS NULL OR price BETWEEN 500 AND 1000000),
mrp NUMERIC(12,2),
availability TEXT,
in_stock BOOLEAN,
source_type TEXT NOT NULL,
pincode TEXT,
pincode_applied BOOLEAN NOT NULL DEFAULT FALSE,
evidence_text TEXT NOT NULL CHECK (length(evidence_text) > 0),
observed_at TIMESTAMPTZ NOT NULL DEFAULT now(),
crawl_run_id BIGINT REFERENCES elec.crawl_run(id) ON DELETE SET NULL
);
CREATE INDEX price_history_listing_time ON elec.price_history (listing_id, observed_at DESC);
CREATE FUNCTION elec.refuse_update() RETURNS trigger LANGUAGE plpgsql AS $$
BEGIN
RAISE EXCEPTION 'elec.price_history is append-only';
END $$;
CREATE TRIGGER price_history_append_only BEFORE UPDATE ON elec.price_history
FOR EACH ROW EXECUTE FUNCTION elec.refuse_update();
-- Which canonical product a listing belongs to, and how sure we are.
-- Only 'auto' and 'approved' links count as evidence or appear in views.
CREATE TABLE elec.product_listing_map (
listing_id BIGINT PRIMARY KEY REFERENCES elec.source_listing(id) ON DELETE CASCADE,
product_id BIGINT NOT NULL REFERENCES elec.product(id) ON DELETE CASCADE,
method TEXT NOT NULL CHECK (method IN ('gtin','mpn','variant_key','fuzzy','manual')),
confidence NUMERIC(3,2) NOT NULL CHECK (confidence BETWEEN 0 AND 1),
review_status TEXT NOT NULL CHECK (review_status IN ('auto','pending','approved','rejected')),
created_at TIMESTAMPTZ NOT NULL DEFAULT now(),
reviewed_at TIMESTAMPTZ
);
CREATE INDEX map_product ON elec.product_listing_map (product_id);
-- Images are URLs only (never downloaded), each tied to the listing it was
-- found on and checked live.
CREATE TABLE elec.product_image (
id BIGSERIAL PRIMARY KEY,
product_id BIGINT NOT NULL REFERENCES elec.product(id) ON DELETE CASCADE,
url TEXT NOT NULL CHECK (url ~ '^https?://'),
source_listing_id BIGINT NOT NULL REFERENCES elec.source_listing(id) ON DELETE CASCADE,
source_type TEXT NOT NULL,
rank INT NOT NULL DEFAULT 100,
validated_at TIMESTAMPTZ NOT NULL DEFAULT now(),
UNIQUE (product_id, url)
);

View File

@@ -0,0 +1,71 @@
-- Read-side views. Public views only ever show VERIFIED products and links
-- that are 'auto' or 'approved'.
CREATE VIEW elec.v_product_availability AS
SELECT p.id AS product_id,
b.name AS brand,
c.slug AS category,
p.display_name,
s.id AS site_id,
s.name AS site,
s.domain,
s.kind AS site_kind,
s.region AS site_region,
l.id AS listing_id,
l.source_url,
l.source_type,
l.title AS listing_title,
l.colour,
l.price,
l.mrp,
l.in_stock,
l.availability,
l.pincode,
l.pincode_applied,
l.confidence,
l.last_seen_at AS observed_at
FROM elec.product p
JOIN elec.brand b ON b.id = p.brand_id
JOIN elec.category c ON c.id = p.category_id
JOIN elec.product_listing_map m ON m.product_id = p.id AND m.review_status IN ('auto','approved')
JOIN elec.source_listing l ON l.id = m.listing_id
JOIN elec.site s ON s.id = l.site_id
WHERE p.verification_status = 'verified';
-- Cheapest known price per product. Scraped prices are preferred over search
-- snippet prices; a listing known to be out of stock is skipped.
CREATE VIEW elec.v_best_price AS
SELECT DISTINCT ON (product_id)
product_id, site, domain, source_url, source_type, price, mrp, in_stock, observed_at
FROM elec.v_product_availability
WHERE price IS NOT NULL AND in_stock IS DISTINCT FROM FALSE
ORDER BY product_id, (source_type = 'search_snippet'), price, observed_at DESC;
CREATE VIEW elec.v_brand_catalog AS
SELECT p.id AS product_id, b.name AS brand, b.slug AS brand_slug, c.slug AS category,
p.family, p.display_name, p.model, p.ram_gb, p.storage_gb, p.processor,
p.canonical_specs,
bp.price AS best_price,
bp.site AS best_price_site,
bp.source_type AS best_price_source_type,
(SELECT count(DISTINCT a.site_id) FROM elec.v_product_availability a
WHERE a.product_id = p.id) AS platform_count,
(SELECT coalesce(bool_or(a.site_region = 'TN'), FALSE) FROM elec.v_product_availability a
WHERE a.product_id = p.id) AS sold_by_tn_retailer,
(SELECT i.url FROM elec.product_image i WHERE i.product_id = p.id
ORDER BY i.rank, i.id LIMIT 1) AS image_url,
p.updated_at
FROM elec.product p
JOIN elec.brand b ON b.id = p.brand_id
JOIN elec.category c ON c.id = p.category_id
LEFT JOIN elec.v_best_price bp ON bp.product_id = p.id
WHERE p.verification_status = 'verified';
CREATE VIEW elec.v_brand_summary AS
SELECT brand, brand_slug, category,
count(*) AS product_count,
min(best_price) AS min_price,
max(best_price) AS max_price,
max(platform_count) AS max_platforms
FROM elec.v_brand_catalog
GROUP BY brand, brand_slug, category;

View File

@@ -0,0 +1,44 @@
-- Search results carry cached, sometimes seller-specific prices. A price that
-- disagrees sharply with the product-page price for the same product (or is
-- below what the category can cost) is kept with its evidence but flagged, and
-- is never used as the "best price". Set by repository.flag_price_outliers().
ALTER TABLE elec.source_listing ADD COLUMN price_outlier BOOLEAN NOT NULL DEFAULT FALSE;
CREATE OR REPLACE VIEW elec.v_product_availability AS
SELECT p.id AS product_id,
b.name AS brand,
c.slug AS category,
p.display_name,
s.id AS site_id,
s.name AS site,
s.domain,
s.kind AS site_kind,
s.region AS site_region,
l.id AS listing_id,
l.source_url,
l.source_type,
l.title AS listing_title,
l.colour,
l.price,
l.mrp,
l.in_stock,
l.availability,
l.pincode,
l.pincode_applied,
l.confidence,
l.last_seen_at AS observed_at,
l.price_outlier
FROM elec.product p
JOIN elec.brand b ON b.id = p.brand_id
JOIN elec.category c ON c.id = p.category_id
JOIN elec.product_listing_map m ON m.product_id = p.id AND m.review_status IN ('auto','approved')
JOIN elec.source_listing l ON l.id = m.listing_id
JOIN elec.site s ON s.id = l.site_id
WHERE p.verification_status = 'verified';
CREATE OR REPLACE VIEW elec.v_best_price AS
SELECT DISTINCT ON (product_id)
product_id, site, domain, source_url, source_type, price, mrp, in_stock, observed_at
FROM elec.v_product_availability
WHERE price IS NOT NULL AND NOT price_outlier AND in_stock IS DISTINCT FROM FALSE
ORDER BY product_id, (source_type = 'search_snippet'), price, observed_at DESC;

View File

@@ -0,0 +1,30 @@
-- Best price: a product whose every priced listing is out of stock still has a
-- price worth showing. In-stock (or unknown-stock) prices still win; an
-- out-of-stock price is used only when nothing else is priced. Same columns
-- as 0005, so v_brand_catalog keeps working unchanged.
CREATE OR REPLACE VIEW elec.v_best_price AS
SELECT DISTINCT ON (product_id)
product_id, site, domain, source_url, source_type, price, mrp, in_stock, observed_at
FROM elec.v_product_availability
WHERE price IS NOT NULL AND NOT price_outlier
ORDER BY product_id, (in_stock IS FALSE), (source_type = 'search_snippet'), price, observed_at DESC;
-- Individual customer reviews, exactly as a product page publishes them in its
-- schema.org JSON-LD. Nothing here is generated: every row is a review the
-- listing's own page stated. Sentiment is derived only from the reviewer's
-- own star rating (>=4 positive, >=3 neutral, <3 negative); NULL when the
-- review states no rating.
CREATE TABLE elec.listing_review (
id BIGSERIAL PRIMARY KEY,
listing_id BIGINT NOT NULL REFERENCES elec.source_listing(id) ON DELETE CASCADE,
author TEXT,
rating NUMERIC(2,1) CHECK (rating IS NULL OR rating BETWEEN 0 AND 5),
title TEXT,
body TEXT NOT NULL,
review_date TEXT,
sentiment TEXT CHECK (sentiment IS NULL OR sentiment IN ('positive','neutral','negative')),
content_hash TEXT NOT NULL,
fetched_at TIMESTAMPTZ NOT NULL DEFAULT now(),
UNIQUE (listing_id, content_hash)
);
CREATE INDEX listing_review_listing_idx ON elec.listing_review (listing_id);

View File

@@ -0,0 +1,588 @@
"""All SQL used by the pipeline. psycopg3, no ORM - the same style as the
original project, with each function owning one statement or one small unit
of work."""
from __future__ import annotations
import hashlib
import json
import re
from datetime import datetime, timezone
from decimal import Decimal
from typing import Any, Dict, List, Optional
from psycopg.types.json import Jsonb
from app.electronics.db.connection import connect, transaction
from app.electronics.models import Listing
from app.electronics.reference import Reference, slugify
def _json(value: Any) -> Jsonb:
return Jsonb(json.loads(json.dumps(value, default=str)))
# ---------------------------------------------------------------------------
# Reference data
# ---------------------------------------------------------------------------
def seed_reference(ref: Reference) -> Dict[str, int]:
"""Idempotent upsert of brands, aliases, categories and sites."""
counts = {"brands": 0, "aliases": 0, "categories": 0, "sites": 0}
with transaction() as conn:
for c in ref.categories.values():
conn.execute(
"INSERT INTO elec.category (slug, name) VALUES (%s, %s) "
"ON CONFLICT (slug) DO UPDATE SET name = EXCLUDED.name",
(c.slug, c.name),
)
counts["categories"] += 1
for b in ref.brands.values():
row = conn.execute(
"INSERT INTO elec.brand (name, slug, official_domains) VALUES (%s, %s, %s) "
"ON CONFLICT (slug) DO UPDATE SET name = EXCLUDED.name, official_domains = EXCLUDED.official_domains "
"RETURNING id",
(b.name, b.slug, list(b.official)),
).fetchone()
counts["brands"] += 1
for alias in b.aliases:
conn.execute(
"INSERT INTO elec.brand_alias (alias, brand_id, is_sub_brand) VALUES (%s, %s, FALSE) "
"ON CONFLICT (alias) DO UPDATE SET brand_id = EXCLUDED.brand_id, is_sub_brand = FALSE",
(alias, row["id"]),
)
counts["aliases"] += 1
for sub in b.sub_brands:
conn.execute(
"INSERT INTO elec.brand_alias (alias, brand_id, is_sub_brand) VALUES (%s, %s, TRUE) "
"ON CONFLICT (alias) DO UPDATE SET brand_id = EXCLUDED.brand_id, is_sub_brand = TRUE",
(sub, row["id"]),
)
counts["aliases"] += 1
for cat in b.categories:
conn.execute(
"INSERT INTO elec.brand_category (brand_id, category_id) "
"SELECT %s, id FROM elec.category WHERE slug = %s ON CONFLICT DO NOTHING",
(row["id"], cat),
)
for s in ref.sites.values():
conn.execute(
"""
INSERT INTO elec.site (domain, name, kind, region, policy, brand_id, product_url, pincode_param)
VALUES (%s, %s, %s, %s, %s, (SELECT id FROM elec.brand WHERE slug = %s), %s, %s)
ON CONFLICT (domain) DO UPDATE SET
name = EXCLUDED.name, kind = EXCLUDED.kind, region = EXCLUDED.region,
policy = EXCLUDED.policy, brand_id = EXCLUDED.brand_id,
product_url = EXCLUDED.product_url, pincode_param = EXCLUDED.pincode_param
""",
(s.domain, s.name, s.kind, s.region, s.policy, s.brand_slug, s.product_url, s.pincode_param),
)
counts["sites"] += 1
return counts
def id_maps() -> Dict[str, Dict[str, int]]:
with connect() as conn:
return {
"brand": {r["slug"]: r["id"] for r in conn.execute("SELECT id, slug FROM elec.brand")},
"category": {r["slug"]: r["id"] for r in conn.execute("SELECT id, slug FROM elec.category")},
"site": {r["domain"]: r["id"] for r in conn.execute("SELECT id, domain FROM elec.site")},
}
def sites() -> List[dict]:
with connect() as conn:
return list(conn.execute("SELECT * FROM elec.site ORDER BY kind, name"))
def set_probe_result(domain: str, outcome: str, robots_allowed: Optional[bool], evidence: dict) -> None:
with transaction() as conn:
conn.execute(
"UPDATE elec.site SET probe_outcome = %s, robots_allowed = %s, probe_evidence = %s, probed_at = now() "
"WHERE domain = %s",
(outcome, robots_allowed, _json(evidence), domain),
)
def trip_breaker(domain: str, reason: str, until_epoch: float) -> None:
with transaction() as conn:
conn.execute(
"UPDATE elec.site SET breaker_until = to_timestamp(%s), breaker_reason = %s, "
"probe_outcome = 'C' WHERE domain = %s OR %s LIKE '%%.' || domain",
(until_epoch, reason, domain, domain),
)
# ---------------------------------------------------------------------------
# Runs and fetch log
# ---------------------------------------------------------------------------
def start_run(kind: str, params: dict) -> int:
with transaction() as conn:
return conn.execute(
"INSERT INTO elec.crawl_run (kind, params) VALUES (%s, %s) RETURNING id", (kind, _json(params))
).fetchone()["id"]
def finish_run(run_id: int, status: str, stats: dict, error: Optional[str] = None) -> None:
with transaction() as conn:
conn.execute(
"UPDATE elec.crawl_run SET status = %s, stats = %s, error = %s, ended_at = now() WHERE id = %s",
(status, _json(stats), error, run_id),
)
def log_fetch(run_id: Optional[int], url: str, host: str, status: Optional[int], nbytes: int,
outcome: str, robots_allowed: Optional[bool]) -> None:
with transaction() as conn:
conn.execute(
"INSERT INTO elec.fetch_log (crawl_run_id, url, host, status, bytes, outcome, robots_allowed) "
"VALUES (%s, %s, %s, %s, %s, %s, %s)",
(run_id, url, host, status, nbytes, outcome, robots_allowed),
)
def recent_runs(limit: int = 20) -> List[dict]:
with connect() as conn:
return list(conn.execute("SELECT * FROM elec.crawl_run ORDER BY id DESC LIMIT %s", (limit,)))
# ---------------------------------------------------------------------------
# Search cache
# ---------------------------------------------------------------------------
def search_cache_get(provider: str, kind: str, query: str, ttl_hours: int) -> Optional[List[dict]]:
with connect() as conn:
row = conn.execute(
"SELECT results FROM elec.search_cache WHERE provider = %s AND kind = %s AND query = %s "
"AND fetched_at > now() - make_interval(hours => %s)",
(provider, kind, query, ttl_hours),
).fetchone()
return row["results"] if row else None
def search_cache_put(provider: str, kind: str, query: str, results: List[dict]) -> None:
with transaction() as conn:
conn.execute(
"INSERT INTO elec.search_cache (provider, kind, query, results) VALUES (%s, %s, %s, %s) "
"ON CONFLICT (provider, kind, query) DO UPDATE SET results = EXCLUDED.results, fetched_at = now()",
(provider, kind, query, _json(results)),
)
def google_queries_today() -> int:
with connect() as conn:
return conn.execute(
"SELECT count(*) AS n FROM elec.search_cache WHERE provider = 'google' AND fetched_at::date = current_date"
).fetchone()["n"]
# ---------------------------------------------------------------------------
# Listings and prices
# ---------------------------------------------------------------------------
def upsert_listing(listing: Listing, ids: Dict[str, Dict[str, int]], run_id: Optional[int]) -> int:
"""Write the latest state of a listing and append one price observation."""
listing.validate()
site_id = ids["site"][listing.site_domain]
brand_id = ids["brand"][listing.brand_slug]
category_id = ids["category"][listing.category]
with transaction() as conn:
existing = conn.execute(
"SELECT id, source_type, price FROM elec.source_listing WHERE site_id = %s AND source_sku = %s",
(site_id, listing.source_sku),
).fetchone()
# A scraped page is better evidence than a search snippet about the
# same page. Never let a later snippet overwrite scraped values.
if existing and existing["source_type"] in ("scraped_page", "brand_official") and listing.source_type == "search_snippet":
conn.execute("UPDATE elec.source_listing SET last_seen_at = now() WHERE id = %s", (existing["id"],))
return existing["id"]
params = dict(
site_id=site_id, source_sku=listing.source_sku, source_url=listing.source_url,
source_type=listing.source_type, brand_id=brand_id, category_id=category_id,
family=listing.family, title=listing.title[:500], model=listing.model,
model_number=listing.model_number, ram_gb=listing.ram_gb, storage_gb=listing.storage_gb,
colour=listing.colour, price=listing.price, mrp=listing.mrp, availability=listing.availability,
in_stock=listing.in_stock, pincode=listing.pincode, pincode_applied=listing.pincode_applied,
rating=listing.rating, review_count=listing.review_count, gtin=listing.gtin,
image_urls=listing.image_urls[:12], specs_raw=_json(listing.specs_raw), specs=_json(listing.specs),
evidence_text=listing.evidence_text[:4000], search_query=listing.search_query,
confidence=round(listing.confidence, 2), parser=listing.parser, content_hash=listing.content_hash,
crawl_run_id=run_id,
)
row = conn.execute(
"""
INSERT INTO elec.source_listing (
site_id, source_sku, source_url, source_type, brand_id, category_id, family, title, model,
model_number, ram_gb, storage_gb, colour, price, mrp, availability, in_stock, pincode,
pincode_applied, rating, review_count, gtin, image_urls, specs_raw, specs, evidence_text,
search_query, confidence, parser, content_hash, crawl_run_id)
VALUES (
%(site_id)s, %(source_sku)s, %(source_url)s, %(source_type)s, %(brand_id)s, %(category_id)s,
%(family)s, %(title)s, %(model)s, %(model_number)s, %(ram_gb)s, %(storage_gb)s, %(colour)s,
%(price)s, %(mrp)s, %(availability)s, %(in_stock)s, %(pincode)s, %(pincode_applied)s,
%(rating)s, %(review_count)s, %(gtin)s, %(image_urls)s, %(specs_raw)s, %(specs)s,
%(evidence_text)s, %(search_query)s, %(confidence)s, %(parser)s, %(content_hash)s,
%(crawl_run_id)s)
ON CONFLICT (site_id, source_sku) DO UPDATE SET
source_url = EXCLUDED.source_url, source_type = EXCLUDED.source_type,
brand_id = EXCLUDED.brand_id, category_id = EXCLUDED.category_id, family = EXCLUDED.family,
title = EXCLUDED.title, model = EXCLUDED.model, model_number = EXCLUDED.model_number,
ram_gb = EXCLUDED.ram_gb, storage_gb = EXCLUDED.storage_gb, colour = EXCLUDED.colour,
price = EXCLUDED.price, mrp = EXCLUDED.mrp, availability = EXCLUDED.availability,
in_stock = EXCLUDED.in_stock, pincode = EXCLUDED.pincode,
pincode_applied = EXCLUDED.pincode_applied, rating = EXCLUDED.rating,
review_count = EXCLUDED.review_count, gtin = EXCLUDED.gtin, image_urls = EXCLUDED.image_urls,
specs_raw = EXCLUDED.specs_raw, specs = EXCLUDED.specs, evidence_text = EXCLUDED.evidence_text,
search_query = EXCLUDED.search_query, confidence = EXCLUDED.confidence, parser = EXCLUDED.parser,
content_hash = EXCLUDED.content_hash, crawl_run_id = EXCLUDED.crawl_run_id, last_seen_at = now()
RETURNING id
""",
params,
).fetchone()
listing_id = row["id"]
if listing.price is not None or listing.in_stock is not None:
conn.execute(
"INSERT INTO elec.price_history (listing_id, price, mrp, availability, in_stock, source_type, "
"pincode, pincode_applied, evidence_text, crawl_run_id) VALUES (%s,%s,%s,%s,%s,%s,%s,%s,%s,%s)",
(listing_id, listing.price, listing.mrp, listing.availability, listing.in_stock,
listing.source_type, listing.pincode, listing.pincode_applied,
listing.evidence_text[:2000], run_id),
)
return listing_id
# ---------------------------------------------------------------------------
# Ratings and reviews
# ---------------------------------------------------------------------------
def save_reviews(listing_id: int, reviews: List[Dict[str, Any]]) -> int:
"""Store the reviews a listing's page publishes. Idempotent per review
text; an empty list changes nothing (a later search-only sighting must
not erase what the page said). Returns the number of new rows."""
from app.electronics.reviews import sentiment_for
added = 0
with transaction() as conn:
for r in reviews:
body = (r.get("body") or "").strip()
if not body:
continue
digest = hashlib.sha1(f"{r.get('author') or ''}|{body}".encode("utf-8", "ignore")).hexdigest()
row = conn.execute(
"INSERT INTO elec.listing_review (listing_id, author, rating, title, body, review_date, sentiment, "
"content_hash) VALUES (%s,%s,%s,%s,%s,%s,%s,%s) "
"ON CONFLICT (listing_id, content_hash) DO NOTHING RETURNING id",
(listing_id, (r.get("author") or None) and str(r["author"])[:200], r.get("rating"),
(r.get("title") or None) and str(r["title"])[:300], body[:4000],
(r.get("review_date") or None) and str(r["review_date"])[:40],
sentiment_for(r.get("rating")), digest),
).fetchone()
added += 1 if row else 0
return added
def update_listing_rating(listing_id: int, rating: Optional[Decimal], review_count: Optional[int]) -> None:
"""Refresh only the rating fields of a listing (used by the review backfill)."""
with transaction() as conn:
conn.execute(
"UPDATE elec.source_listing SET rating = %s, review_count = %s WHERE id = %s",
(rating, review_count, listing_id),
)
def product_rating_and_reviews(conn, product_id: int) -> Dict[str, Any]:
"""Per-platform ratings and all stored reviews for a verified product's
approved listings, each with the page it was read from."""
sources = conn.execute(
"SELECT a.site, a.source_url, l.rating, l.review_count FROM elec.v_product_availability a "
"JOIN elec.source_listing l ON l.id = a.listing_id "
"WHERE a.product_id = %s AND l.rating > 0 ORDER BY l.review_count DESC NULLS LAST, a.site",
(product_id,),
).fetchall()
reviews = conn.execute(
"SELECT a.site, a.source_url, r.author, r.rating, r.title, r.body, r.review_date, r.sentiment "
"FROM elec.v_product_availability a JOIN elec.listing_review r ON r.listing_id = a.listing_id "
"WHERE a.product_id = %s",
(product_id,),
).fetchall()
return {"sources": [dict(s) for s in sources], "reviews": [dict(r) for r in reviews]}
def listings_for_review_backfill(category: Optional[str] = None) -> List[dict]:
"""Page-read listings of verified products, for re-reading ratings/reviews."""
sql = (
"SELECT a.listing_id, a.source_url, a.domain, a.site_kind, a.category, l.source_sku, l.title "
"FROM elec.v_product_availability a JOIN elec.source_listing l ON l.id = a.listing_id "
"WHERE a.source_type IN ('scraped_page','brand_official')"
)
params: tuple = ()
if category:
sql += " AND a.category = %s"
params = (category,)
with connect() as conn:
return list(conn.execute(sql + " ORDER BY a.listing_id", params))
# ---------------------------------------------------------------------------
# Products, matching, images
# ---------------------------------------------------------------------------
def product_candidates(brand_slug: str, category: str) -> List[dict]:
with connect() as conn:
return list(conn.execute(
"SELECT p.id, p.variant_key, p.model_norm, p.ram_gb, p.storage_gb, p.processor, p.mpn, p.gtin "
"FROM elec.product p JOIN elec.brand b ON b.id = p.brand_id JOIN elec.category c ON c.id = p.category_id "
"WHERE b.slug = %s AND c.slug = %s AND p.verification_status <> 'rejected'",
(brand_slug, category),
))
def _cpu_label(processor: Optional[str]) -> str:
""""ryzen 5 7530u" -> "Ryzen 5 7530U", "i5-1334u" -> "i5-1334U"."""
if not processor:
return ""
def fmt(t: str) -> str:
if re.fullmatch(r"i[3579]-\w+", t):
return "i" + t[1:].upper() # i5-1334U
if any(ch.isdigit() for ch in t):
return t.upper() # 7530U, M5
return t.title() # Ryzen, Core, Ultra
return " ".join(fmt(t) for t in processor.split())
def product_display_name(listing: Listing) -> str:
variant = [x for x in (
_cpu_label(listing.processor) if listing.category == "laptops" else "",
f"{_fmt_gb(listing.ram_gb)} RAM" if listing.ram_gb else "",
_fmt_gb(listing.storage_gb) if listing.storage_gb else "",
) if x]
if listing.category == "laptops" and not listing.processor and listing.model_number:
variant.insert(0, listing.model_number) # the part number is what tells it apart
return " ".join(x for x in [load_brand_name(listing.brand_slug), listing.model,
f"({', '.join(variant)})" if variant else ""] if x)
def create_product(listing: Listing, ids: Dict[str, Dict[str, int]]) -> int:
display = product_display_name(listing)
with transaction() as conn:
row = conn.execute(
"""
INSERT INTO elec.product (brand_id, category_id, family, model, model_norm, variant_key, display_name,
ram_gb, storage_gb, processor, mpn, gtin)
VALUES (%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s)
ON CONFLICT (variant_key) DO UPDATE SET updated_at = now()
RETURNING id
""",
(ids["brand"][listing.brand_slug], ids["category"][listing.category], listing.family,
listing.model or listing.model_norm, listing.model_norm, listing.variant_key, display,
listing.ram_gb, listing.storage_gb, listing.processor, listing.model_number, listing.gtin),
).fetchone()
return row["id"]
def _fmt_gb(value: Optional[Decimal]) -> str:
if value is None:
return ""
if value >= 1024 and value % 1024 == 0:
return f"{int(value // 1024)}TB"
return f"{format(value.normalize(), 'f')}GB"
_BRAND_NAMES: Dict[str, str] = {}
def load_brand_name(slug: str) -> str:
if not _BRAND_NAMES:
from app.electronics.reference import load_reference
_BRAND_NAMES.update({s: b.name for s, b in load_reference().brands.items()})
return _BRAND_NAMES.get(slug, slug.title())
def map_listing(listing_id: int, product_id: int, method: str, confidence: float, review_status: str) -> None:
with transaction() as conn:
conn.execute(
"""
INSERT INTO elec.product_listing_map (listing_id, product_id, method, confidence, review_status)
VALUES (%s, %s, %s, %s, %s)
ON CONFLICT (listing_id) DO UPDATE SET
product_id = EXCLUDED.product_id, method = EXCLUDED.method, confidence = EXCLUDED.confidence,
review_status = CASE WHEN elec.product_listing_map.review_status IN ('approved','rejected')
AND elec.product_listing_map.product_id = EXCLUDED.product_id
THEN elec.product_listing_map.review_status
ELSE EXCLUDED.review_status END
""",
(listing_id, product_id, method, round(confidence, 2), review_status),
)
def merge_product_specs(product_id: int, specs: Dict[str, Any], sources: Dict[str, str], source_url: str) -> None:
"""Add spec keys the product does not have yet. Existing values win:
specs are only ever filled, never overwritten by a later source."""
if not specs:
return
with transaction() as conn:
row = conn.execute(
"SELECT canonical_specs, spec_sources FROM elec.product WHERE id = %s FOR UPDATE", (product_id,)
).fetchone()
current, cur_src = dict(row["canonical_specs"] or {}), dict(row["spec_sources"] or {})
changed = False
for key, value in specs.items():
if key not in current:
current[key] = value
cur_src[key] = {"url": source_url, "from": sources.get(key, "")}
changed = True
if changed:
conn.execute(
"UPDATE elec.product SET canonical_specs = %s, spec_sources = %s, updated_at = now() WHERE id = %s",
(_json(current), _json(cur_src), product_id),
)
def add_image(product_id: int, url: str, listing_id: int, source_type: str, rank: int) -> None:
with transaction() as conn:
conn.execute(
"INSERT INTO elec.product_image (product_id, url, source_listing_id, source_type, rank) "
"VALUES (%s, %s, %s, %s, %s) ON CONFLICT (product_id, url) DO UPDATE SET validated_at = now()",
(product_id, url, listing_id, source_type, rank),
)
def product_image_count(product_id: int) -> int:
with connect() as conn:
return conn.execute("SELECT count(*) AS n FROM elec.product_image WHERE product_id = %s",
(product_id,)).fetchone()["n"]
# The least a new device in the category can plausibly cost. Anything below is
# an accessory, an EMI or an offer amount that slipped through.
CATEGORY_MIN_PRICE = {"mobiles": 3000, "laptops": 15000}
OUTLIER_TOLERANCE = 0.35
def flag_price_outliers() -> int:
"""Flag prices that cannot be trusted as this product's price:
* below the category's floor (CATEGORY_MIN_PRICE);
* a search-result price more than OUTLIER_TOLERANCE away from the price
read off a product page for the same product;
* with no page price, a search-result price that far from the median of
at least three prices for the product.
Flagged prices stay stored with their evidence; they are just never used as
the best price. Returns the number flagged."""
floor_cases = " ".join(f"WHEN '{k}' THEN {v}" for k, v in CATEGORY_MIN_PRICE.items())
with transaction() as conn:
conn.execute("UPDATE elec.source_listing SET price_outlier = FALSE WHERE price_outlier")
cur = conn.execute(
f"""
WITH prices AS (
SELECT l.id, m.product_id, l.price, l.source_type, c.slug
FROM elec.source_listing l
JOIN elec.product_listing_map m ON m.listing_id = l.id AND m.review_status IN ('auto','approved')
JOIN elec.category c ON c.id = l.category_id
WHERE l.price IS NOT NULL
),
ref AS (
SELECT product_id,
percentile_cont(0.5) WITHIN GROUP (ORDER BY price)
FILTER (WHERE source_type <> 'search_snippet') AS page_median,
percentile_cont(0.5) WITHIN GROUP (ORDER BY price) AS all_median,
count(*) AS n
FROM prices GROUP BY product_id
)
UPDATE elec.source_listing l SET price_outlier = TRUE
FROM prices p JOIN ref r ON r.product_id = p.product_id
WHERE l.id = p.id AND (
p.price < CASE p.slug {floor_cases} ELSE 0 END
OR (p.source_type = 'search_snippet' AND r.page_median IS NOT NULL
AND abs(p.price - r.page_median) / r.page_median > %(tol)s)
OR (p.source_type = 'search_snippet' AND r.page_median IS NULL AND r.n >= 3
AND abs(p.price - r.all_median) / r.all_median > %(tol)s)
)
""",
{"tol": OUTLIER_TOLERANCE},
)
return cur.rowcount
def refresh_verification() -> Dict[str, int]:
flag_price_outliers()
"""A product is VERIFIED when auto/approved listings on at least two
different sites point at it, and at least one of them is a retailer
(so it is actually sold). Everything else stays unverified and hidden."""
with transaction() as conn:
conn.execute(
"""
WITH ev AS (
SELECT m.product_id,
count(DISTINCT l.site_id) AS sites,
count(DISTINCT l.site_id) FILTER (WHERE s.kind <> 'brand_official') AS retail_sites
FROM elec.product_listing_map m
JOIN elec.source_listing l ON l.id = m.listing_id
JOIN elec.site s ON s.id = l.site_id
WHERE m.review_status IN ('auto','approved')
GROUP BY m.product_id
)
UPDATE elec.product p SET
evidence_count = coalesce(ev.sites, 0),
verification_status = CASE
WHEN p.verification_status = 'rejected' THEN 'rejected'
WHEN coalesce(ev.sites, 0) >= 2 AND coalesce(ev.retail_sites, 0) >= 1 THEN 'verified'
ELSE 'unverified' END,
updated_at = now()
FROM elec.product p2 LEFT JOIN ev ON ev.product_id = p2.id
WHERE p.id = p2.id
"""
)
rows = conn.execute(
"SELECT verification_status AS s, count(*) AS n FROM elec.product GROUP BY 1"
).fetchall()
return {r["s"]: r["n"] for r in rows}
def products_without_embedding(limit: int = 500) -> List[dict]:
with connect() as conn:
return list(conn.execute(
"SELECT p.id, p.display_name, p.canonical_specs, b.name AS brand, c.name AS category "
"FROM elec.product p JOIN elec.brand b ON b.id = p.brand_id JOIN elec.category c ON c.id = p.category_id "
"WHERE p.embedding IS NULL AND p.verification_status = 'verified' LIMIT %s", (limit,)))
def set_embedding(product_id: int, vector: List[float]) -> None:
import numpy as np
with transaction() as conn:
conn.execute("UPDATE elec.product SET embedding = %s WHERE id = %s", (np.array(vector), product_id))
def review_queue(limit: int = 100) -> List[dict]:
with connect() as conn:
return list(conn.execute(
"""
SELECT m.listing_id, m.product_id, m.method, m.confidence, l.title AS listing_title, l.source_url,
s.name AS site, p.display_name AS product
FROM elec.product_listing_map m
JOIN elec.source_listing l ON l.id = m.listing_id
JOIN elec.site s ON s.id = l.site_id
JOIN elec.product p ON p.id = m.product_id
WHERE m.review_status = 'pending'
ORDER BY m.confidence DESC, m.listing_id LIMIT %s
""", (limit,)))
def set_review(listing_id: int, approve: bool) -> bool:
with transaction() as conn:
cur = conn.execute(
"UPDATE elec.product_listing_map SET review_status = %s, reviewed_at = now() "
"WHERE listing_id = %s AND review_status = 'pending'",
("approved" if approve else "rejected", listing_id),
)
return cur.rowcount > 0
def now_utc() -> datetime:
return datetime.now(timezone.utc)
def grounding_sample(n: int = 50) -> List[dict]:
with connect() as conn:
return list(conn.execute(
"SELECT id, source_url, source_type, price, evidence_text FROM elec.source_listing "
"WHERE price IS NOT NULL ORDER BY random() LIMIT %s", (n,)))
def slug(text: str) -> str:
return slugify(text)

View File

@@ -0,0 +1,123 @@
"""Product facts from page markup when there is no usable JSON-LD.
Only machine-readable markup is trusted for the price: OpenGraph/product meta
tags and schema.org microdata (itemprop="price"). Free text on the page is not
scanned for rupee amounts - a product page shows EMIs, offers and other
products' prices, and picking the wrong one is worse than picking none.
Spec tables (<table>, <dl>) supply specifications.
"""
from __future__ import annotations
import json
import re
from decimal import Decimal, InvalidOperation
from typing import Any, Dict, List, Optional
from bs4 import BeautifulSoup
def _dec(value: Optional[str]) -> Optional[Decimal]:
if not value:
return None
try:
return Decimal(re.sub(r"[^\d.]", "", value))
except InvalidOperation:
return None
def _meta(soup: BeautifulSoup, *names: str) -> Optional[str]:
for name in names:
tag = soup.find("meta", attrs={"property": name}) or soup.find("meta", attrs={"name": name})
if tag and tag.get("content"):
return tag["content"].strip()
return None
def spec_tables(soup: BeautifulSoup, limit: int = 200) -> Dict[str, str]:
specs: Dict[str, str] = {}
for row in soup.select("table tr"):
cells = row.find_all(["th", "td"])
if len(cells) == 2:
k, v = (c.get_text(" ", strip=True) for c in cells)
if k and v and len(k) <= 60 and len(v) <= 200:
specs.setdefault(k, v)
if len(specs) >= limit:
return specs
for dl in soup.find_all("dl"):
for dt in dl.find_all("dt"):
dd = dt.find_next_sibling("dd")
if dd:
k, v = dt.get_text(" ", strip=True), dd.get_text(" ", strip=True)
if k and v and len(k) <= 60 and len(v) <= 200:
specs.setdefault(k, v)
return specs
def extract_page(html: str) -> Dict[str, Any]:
soup = BeautifulSoup(html, "lxml")
title = _meta(soup, "og:title", "twitter:title")
if not title:
h1 = soup.find("h1")
title = h1.get_text(" ", strip=True) if h1 else None
images: List[str] = []
for name in ("og:image", "og:image:secure_url", "twitter:image"):
v = _meta(soup, name)
if v and v.startswith("http") and v not in images:
images.append(v)
price = _dec(_meta(soup, "product:price:amount", "og:price:amount"))
currency = _meta(soup, "product:price:currency", "og:price:currency")
evidence = ""
if price is not None:
evidence = f"meta product:price:amount={price} currency={currency}"
else:
tag = soup.find(attrs={"itemprop": "price"})
if tag is not None:
raw = tag.get("content") or tag.get_text(" ", strip=True)
price = _dec(raw)
cur_tag = soup.find(attrs={"itemprop": "priceCurrency"})
currency = (cur_tag.get("content") if cur_tag else None) or currency
if price is not None:
evidence = f'itemprop="price" {raw} currency={currency}'
availability = _meta(soup, "product:availability", "og:availability")
in_stock = None
if availability:
low = availability.lower().replace(" ", "")
in_stock = True if "instock" in low else False if ("outofstock" in low or "oos" == low) else None
return {
"name": title,
"images": images,
"price": price,
"currency": currency,
"availability": availability,
"in_stock": in_stock,
"properties": spec_tables(soup),
"evidence": evidence,
}
_STATE_RE = re.compile(
r"<script[^>]*id=\"__NEXT_DATA__\"[^>]*>(.*?)</script>"
r"|window\.__(?:INITIAL|PRELOADED)_STATE__\s*=\s*(\{.*?\})\s*;?\s*</script>",
re.DOTALL,
)
def embedded_state(html: str) -> Optional[Any]:
"""The page's server-rendered application state, when it embeds one."""
for m in _STATE_RE.finditer(html):
raw = m.group(1) or m.group(2)
try:
return json.loads(raw)
except (json.JSONDecodeError, TypeError):
continue
return None
def visible_text(html: str, limit: int = 6000) -> str:
soup = BeautifulSoup(html, "lxml")
for tag in soup(["script", "style", "noscript", "svg", "header", "footer", "nav"]):
tag.decompose()
return re.sub(r"\s+", " ", soup.get_text(" ", strip=True))[:limit]

View File

@@ -0,0 +1,202 @@
"""schema.org Product data embedded in a page as JSON-LD.
This is the preferred source on any page: it is what the site publishes for
search engines, so it is stable and states price, currency and availability
explicitly.
"""
from __future__ import annotations
import json
import re
from decimal import Decimal, InvalidOperation
from typing import Any, Dict, Iterable, List, Optional
from bs4 import BeautifulSoup
_PRODUCT_TYPES = {"product", "productgroup", "productmodel", "individualproduct"}
def _types(node: dict) -> set:
t = node.get("@type")
if isinstance(t, list):
return {str(x).lower() for x in t}
return {str(t).lower()} if t else set()
def _walk(node: Any) -> Iterable[dict]:
if isinstance(node, dict):
yield node
for v in node.values():
yield from _walk(v)
elif isinstance(node, list):
for item in node:
yield from _walk(item)
def json_ld_blocks(html: str) -> List[Any]:
soup = BeautifulSoup(html, "lxml")
blocks = []
for tag in soup.find_all("script", attrs={"type": re.compile(r"ld\+json", re.I)}):
raw = (tag.string or tag.get_text() or "").strip()
if not raw:
continue
try:
blocks.append(json.loads(raw))
except json.JSONDecodeError:
# Some sites put several objects or trailing commas in one tag.
try:
blocks.append(json.loads(re.sub(r",\s*([}\]])", r"\1", raw)))
except json.JSONDecodeError:
continue
return blocks
def _dec(value: Any) -> Optional[Decimal]:
if value is None or value == "":
return None
try:
return Decimal(str(value).replace(",", "").strip())
except InvalidOperation:
return None
def _text(value: Any) -> Optional[str]:
if isinstance(value, dict):
value = value.get("name") or value.get("@value")
if isinstance(value, list):
value = value[0] if value else None
return str(value).strip() if value not in (None, "") else None
def _images(value: Any) -> List[str]:
out: List[str] = []
for v in value if isinstance(value, list) else [value]:
if isinstance(v, dict):
v = v.get("url") or v.get("contentUrl")
if isinstance(v, str) and v.startswith(("http://", "https://")):
out.append(v)
return out
def _availability(value: Any) -> tuple:
text = (_text(value) or "").lower()
if not text:
return None, None
if "instock" in text or "limitedavailability" in text or "onlineonly" in text:
return "InStock", True
if any(k in text for k in ("outofstock", "soldout", "discontinued", "preorder", "presale")):
return text.rsplit("/", 1)[-1], False
return text.rsplit("/", 1)[-1], None
def _offer(offers: Any) -> Dict[str, Any]:
"""The price/availability of the product's (lowest) offer."""
candidates = offers if isinstance(offers, list) else [offers]
best: Dict[str, Any] = {}
for o in candidates:
if not isinstance(o, dict):
continue
price = _dec(o.get("price"))
if price is None:
price = _dec(o.get("lowPrice"))
if price is None and isinstance(o.get("priceSpecification"), dict):
price = _dec(o["priceSpecification"].get("price"))
currency = _text(o.get("priceCurrency")) or (
_text(o["priceSpecification"].get("priceCurrency")) if isinstance(o.get("priceSpecification"), dict) else None
)
availability, in_stock = _availability(o.get("availability"))
entry = {"price": price, "currency": currency, "availability": availability, "in_stock": in_stock,
"raw": {k: o.get(k) for k in ("price", "lowPrice", "priceCurrency", "availability") if k in o}}
if price is not None and (not best or best.get("price") is None or price < best["price"]):
best = entry
elif not best:
best = entry
return best
MAX_REVIEWS_PER_PAGE = 30
def _review_rating(value: Any) -> Optional[Decimal]:
"""A reviewer's star rating, rescaled to 0-5 when the page uses another scale."""
if not isinstance(value, dict):
return None
rating = _dec(value.get("ratingValue"))
if rating is None:
return None
best = _dec(value.get("bestRating")) or Decimal(5)
if best <= 0:
return None
if best != 5:
rating = rating * Decimal(5) / best
if not (Decimal(0) <= rating <= Decimal(5)):
return None
return rating.quantize(Decimal("0.1"))
def _reviews(node: dict) -> List[Dict[str, Any]]:
"""Customer reviews published on the Product node (schema.org Review).
Only reviews with text are kept - a bare star with no words is not
something a reader can weigh. Nothing is paraphrased or summarised: body,
title and author are the page's own strings.
"""
raw = node.get("review") or node.get("reviews") or []
out: List[Dict[str, Any]] = []
for r in raw if isinstance(raw, list) else [raw]:
if not isinstance(r, dict):
continue
body = _text(r.get("reviewBody")) or _text(r.get("description"))
if not body:
continue
out.append({
"author": _text(r.get("author")),
"rating": _review_rating(r.get("reviewRating")),
"title": _text(r.get("name")) or _text(r.get("headline")),
"body": body[:4000],
"review_date": _text(r.get("datePublished")) or _text(r.get("dateCreated")),
})
if len(out) >= MAX_REVIEWS_PER_PAGE:
break
return out
def extract_products(html: str) -> List[Dict[str, Any]]:
"""All schema.org Product nodes on the page, flattened to plain fields."""
products: List[Dict[str, Any]] = []
for block in json_ld_blocks(html):
for node in _walk(block):
if not (_types(node) & _PRODUCT_TYPES):
continue
name = _text(node.get("name"))
if not name:
continue
offer = _offer(node.get("offers")) if node.get("offers") else {}
if not offer and isinstance(node.get("hasVariant"), list):
offer = _offer([v.get("offers") for v in node["hasVariant"] if isinstance(v, dict) and v.get("offers")])
props = {}
for p in node.get("additionalProperty") or []:
if isinstance(p, dict) and p.get("name") and p.get("value") not in (None, ""):
props[str(p["name"])] = str(p["value"])
rating = node.get("aggregateRating") if isinstance(node.get("aggregateRating"), dict) else {}
products.append({
"name": name,
"brand": _text(node.get("brand")),
"sku": _text(node.get("sku")) or _text(node.get("productID")),
"mpn": _text(node.get("mpn")),
"gtin": next((_text(node.get(k)) for k in ("gtin13", "gtin", "gtin12", "gtin14", "gtin8") if node.get(k)), None),
"color": _text(node.get("color")),
"images": _images(node.get("image")),
"description": _text(node.get("description")),
"price": offer.get("price"),
"currency": offer.get("currency"),
"availability": offer.get("availability"),
"in_stock": offer.get("in_stock"),
# Sites publish 0 for "no ratings yet"; that is not a rating.
"rating": (_dec(rating.get("ratingValue")) or None),
"review_count": int(_dec(rating.get("reviewCount") or rating.get("ratingCount")) or 0) or None,
"reviews": _reviews(node),
"properties": props,
"evidence": json.dumps({"name": name, "offers": offer.get("raw")}, default=str)[:1500],
})
return products

View File

@@ -0,0 +1,207 @@
"""Read prices and stock state out of text we did not render ourselves:
search-result titles/snippets, and visible page text.
The rules lean hard towards NOT returning a price. A snippet usually carries
several rupee amounts - the selling price, the MRP, an EMI, a bank discount, an
exchange value, "₹X off" - and taking the wrong one is worse than taking none.
An amount is only a price when nothing around it says it is something else.
"""
from __future__ import annotations
import re
from dataclasses import dataclass
from decimal import Decimal, InvalidOperation
from typing import List, Optional
PRICE_MIN = Decimal("500")
PRICE_MAX = Decimal("1000000")
# ₹ / Rs / Rs. / INR followed by an amount with Indian (1,29,999) or western
# (129,999) grouping, or none.
_AMOUNT = r"(\d{1,3}(?:,\d{2,3})+(?:\.\d{1,2})?|\d+(?:\.\d{1,2})?)"
_MONEY_RE = re.compile(r"(?:₹|\bRs\.?|\bINR)\s?" + _AMOUNT, re.IGNORECASE)
# Words that make an amount something other than the selling price.
_REJECT_BEFORE = re.compile(
r"(?:emi|save|saving|savings|cashback|cash\s*back|exchange|bank|discount|coupon|"
r"extra|instant|up\s*to|upto|flat|off\s+upto|worth|delivery|shipping|fee|charges?|"
r"starting|starts|from|onwards|min(?:imum)?|as\s+low\s+as|down\s*payment|per\s+month)\W*$",
re.IGNORECASE,
)
_REJECT_AFTER = re.compile(
r"^\W{0,3}(?:off\b|/\s*m(?:o|onth)?\b|per\s+month|p\.?m\.?\b|a\s+month|emi\b|/-?\s*emi|"
r"cashback|discount|savings?|onwards|\+\s*shipping|delivery)",
re.IGNORECASE,
)
_MRP_BEFORE = re.compile(r"(?:m\.?\s?r\.?\s?p\.?|list\s+price|was|original\s+price)[\s:]*$", re.IGNORECASE)
_RANGE_BETWEEN = re.compile(r"^\s*(?:-|–|—|to)\s*$", re.IGNORECASE)
_OUT_OF_STOCK = re.compile(
r"\b(?:out\s+of\s+stock|currently\s+unavailable|sold\s+out|coming\s+soon|notify\s+me|"
r"temporarily\s+unavailable|not\s+available)\b",
re.IGNORECASE,
)
_IN_STOCK = re.compile(r"\b(?:in\s+stock|available\s+now|buy\s+now|add\s+to\s+cart)\b", re.IGNORECASE)
@dataclass(frozen=True)
class Amount:
value: Decimal
kind: str # price | mrp | rejected
reason: str
start: int
end: int
raw: str
def parse_amount(raw: str) -> Optional[Decimal]:
try:
value = Decimal(raw.replace(",", ""))
except InvalidOperation:
return None
return value
def find_amounts(text: str) -> List[Amount]:
"""Every rupee amount in `text`, each classified as price, mrp or rejected."""
out: List[Amount] = []
if not text:
return out
matches = list(_MONEY_RE.finditer(text))
for i, m in enumerate(matches):
value = parse_amount(m.group(1))
if value is None:
continue
before = text[max(0, m.start() - 28): m.start()]
after = text[m.end(): m.end() + 22]
kind, reason = "price", ""
if _MRP_BEFORE.search(before):
kind, reason = "mrp", "labelled MRP"
elif _REJECT_BEFORE.search(before):
kind, reason = "rejected", f"preceded by {_REJECT_BEFORE.search(before).group(0).strip()!r}"
elif _REJECT_AFTER.search(after):
kind, reason = "rejected", f"followed by {_REJECT_AFTER.search(after).group(0).strip()!r}"
# A range ("₹10,999 - ₹12,999") names no single price.
if kind == "price":
if i + 1 < len(matches) and _RANGE_BETWEEN.match(text[m.end(): matches[i + 1].start()]):
kind, reason = "rejected", "start of a price range"
elif i > 0 and _RANGE_BETWEEN.match(text[matches[i - 1].end(): m.start()]):
kind, reason = "rejected", "end of a price range"
if kind != "rejected" and not (PRICE_MIN <= value <= PRICE_MAX):
kind, reason = "rejected", "outside plausible range"
out.append(Amount(value, kind, reason, m.start(), m.end(), m.group(0)))
return out
@dataclass(frozen=True)
class PriceReading:
price: Optional[Decimal]
mrp: Optional[Decimal]
evidence: str # the exact substring the price was read from ("" if none)
def read_price(text: str) -> PriceReading:
"""The single selling price stated in `text`, or None.
If the text states two different unlabelled prices, it is ambiguous (a
listing page snippet often shows several variants) and None is returned.
"""
amounts = find_amounts(text)
prices = [a for a in amounts if a.kind == "price"]
mrps = [a for a in amounts if a.kind == "mrp"]
distinct = {a.value for a in prices}
price: Optional[Decimal] = None
evidence = ""
if len(distinct) == 1:
price = prices[0].value
evidence = prices[0].raw
mrp = mrps[0].value if mrps else None
if price is not None and mrp is not None and mrp < price:
mrp = None # an "MRP" below the selling price was misread; drop it
return PriceReading(price, mrp, evidence)
def read_stock(text: str) -> Optional[bool]:
"""True/False only when the text says so; None when it does not."""
if not text:
return None
if _OUT_OF_STOCK.search(text):
return False
if _IN_STOCK.search(text):
return True
return None
@dataclass(frozen=True)
class RatingReading:
rating: Optional[Decimal]
review_count: Optional[int]
evidence: str # the exact substring the rating was read from ("" if none)
# Only ratings the text states explicitly on a 5-point scale:
# "4.3 out of 5 stars", "Rating: 4.3/5", "Rated 4.3 / 5", "4.3★", "4.3 ★ (1,234 ratings)"
_RATING_PATTERNS = (
re.compile(r"\b([0-5](?:\.\d{1,2})?)\s*out\s+of\s*5(?:\.0)?\b(?:\s*stars?)?", re.IGNORECASE),
# "x/5" only with a rating word before it or "stars" after it - a bare
# "1/5" is as likely a sensor size or a fraction.
re.compile(r"\brat(?:ing|ed)\s*[:\-]?\s*([0-5](?:\.\d{1,2})?)\s*/\s*5(?:\.0)?\b", re.IGNORECASE),
re.compile(r"\b([0-5](?:\.\d{1,2})?)\s*/\s*5(?:\.0)?\s*stars?\b", re.IGNORECASE),
re.compile(r"\b([0-5](?:\.\d{1,2})?)\s*(?:★|☆|⭐)"),
re.compile(r"\brat(?:ing|ed)\s*[:\-]?\s*([0-5](?:\.\d{1,2})?)\s*(?:stars?|★)", re.IGNORECASE),
)
_RATING_COUNT = re.compile(
r"^[\s()\-|·,.:]*(?:stars?)?[\s()\-|·,.:]*(\d{1,3}(?:,\d{2,3})+|\d+)\s*(?:customer\s+)?(?:ratings?|reviews?|votes?)\b",
re.IGNORECASE,
)
def read_rating(text: str) -> RatingReading:
"""The product rating a search title/snippet states, or None.
Only an explicit "x out of 5" / "x/5" / "x★" statement counts; bare
numbers never do. If the text states two different ratings it is
ambiguous (several products on one results page) and None is returned.
"""
if not text:
return RatingReading(None, None, "")
found = []
for pattern in _RATING_PATTERNS:
for m in pattern.finditer(text):
try:
value = Decimal(m.group(1))
except InvalidOperation:
continue
if Decimal(0) < value <= Decimal(5):
found.append((value, m))
if not found or len({v for v, _ in found}) != 1:
return RatingReading(None, None, "")
value, m = min(found, key=lambda f: f[1].start())
count = None
tail = _RATING_COUNT.match(text[m.end(): m.end() + 40])
if tail:
count = int(tail.group(1).replace(",", ""))
evidence = text[m.start(): m.end() + (tail.end() if tail else 0)].strip()
return RatingReading(value, count, evidence)
# Titles returned by search engines carry the site name; it is not part of the
# product title.
_TITLE_SUFFIX = re.compile(
r"\s*(?:[|\-–:]\s*)?(?:buy\s+online.*|online\s+at\s+best\s+price.*|"
r"at\s+best\s+price.*|price\s+in\s+india.*|"
r"amazon\.in.*|flipkart(?:\.com)?.*|croma.*|reliance\s+digital.*|vijay\s+sales.*|"
r"tata\s+cliq.*|poorvika.*|sangeetha.*|vasanth.*|viveks.*)$",
re.IGNORECASE,
)
_TITLE_PREFIX = re.compile(r"^(?:buy\s+|amazon\.in\s*:\s*)", re.IGNORECASE)
def clean_result_title(title: str) -> str:
t = (title or "").strip()
# Engines truncate with "..." and sometimes run several results' titles
# together after it; everything past the first ellipsis is not this page.
t = re.split(r"\s*(?:\.\.\.|…)", t, maxsplit=1)[0]
t = _TITLE_PREFIX.sub("", t)
t = _TITLE_SUFFIX.sub("", t)
return t.strip(" -|:–")

View File

@@ -0,0 +1,124 @@
"""Link a listing to its canonical product (one real-world variant).
From most to least certain:
1. GTIN - same barcode -> auto
2. MPN - same manufacturer part number (laptops) -> auto
3. variant key - same brand, model, RAM and storage -> auto
4. fuzzy - model names ≥ AUTO_RATIO similar AND every
hard attribute (RAM, storage, processor) equal -> auto
≥ REVIEW_RATIO -> pending (review queue)
5. otherwise a new product is created for the variant.
A listing that states too little to identify a variant (no storage on a
phone title, for example) is stored but not linked to any product.
"""
from __future__ import annotations
from dataclasses import dataclass
from decimal import Decimal
from typing import List, Optional
from rapidfuzz import fuzz
from app.electronics.normalise.title_parser import laptop_line, processor_is_specific
AUTO_RATIO = 92
REVIEW_RATIO = 85
@dataclass
class MatchDecision:
product_id: Optional[int] # None -> create a new product
method: str
confidence: float
review_status: str
def _eq(a: Optional[Decimal], b: Optional[Decimal]) -> bool:
if a is None or b is None:
return a is None and b is None
return Decimal(a) == Decimal(b)
def _number_tokens(model_norm: str) -> set:
"""Tokens that carry a digit ("s25", "a37", "15", "2a", "8a"). Two models
whose number tokens differ are different products, however similar the
rest of the name is: Galaxy S25 vs S26, A27 vs A37, iPhone 15 vs 16."""
return {t for t in (model_norm or "").split() if any(ch.isdigit() for ch in t)}
def _lines_compatible(a: str, b: str) -> bool:
"""One model line is the other plus/minus extra words, and they agree on
every number token they both carry ("15" is not "15s", "slim 3" is not
"slim 5")."""
ta, tb = set(a.split()), set(b.split())
return bool(ta and tb) and (ta <= tb or tb <= ta)
def decide(listing, candidates: List[dict]) -> Optional[MatchDecision]:
"""`listing` is a models.Listing with variant_key/model_norm set;
`candidates` are product rows of the same brand and category."""
if not listing.variant_key:
return None
if listing.gtin:
for c in candidates:
if c.get("gtin") and c["gtin"] == listing.gtin:
return MatchDecision(c["id"], "gtin", 0.99, "auto")
if listing.model_number:
for c in candidates:
if c.get("mpn") and c["mpn"].lower() == listing.model_number.lower():
return MatchDecision(c["id"], "mpn", 0.97, "auto")
for c in candidates:
if c["variant_key"] == listing.variant_key:
return MatchDecision(c["id"], "variant_key", 0.95, "auto")
# Laptops: the same configuration (exact CPU model, RAM, storage) within a
# compatible model line - "ideapad slim 3" and "ideapad slim 3 15amn8" -
# is the same product. Two compatible candidates means the title is too
# vague to choose ("pavilion" vs "pavilion 14" and "pavilion 15"): review.
if listing.category == "laptops" and processor_is_specific(listing.processor) \
and listing.ram_gb is not None and listing.storage_gb is not None:
line = laptop_line(listing.model_norm)
same_config = [
c for c in candidates
if c.get("processor") == listing.processor
and _eq(c.get("ram_gb"), listing.ram_gb) and _eq(c.get("storage_gb"), listing.storage_gb)
and _lines_compatible(line, laptop_line(c["model_norm"]))
]
if len(same_config) == 1:
return MatchDecision(same_config[0]["id"], "variant_key", 0.85, "auto")
if len(same_config) > 1:
return MatchDecision(same_config[0]["id"], "variant_key", 0.6, "pending")
# One side does not state the RAM ("Apple iPhone 15 (128 GB)"). Same model
# and storage with exactly one candidate is the same variant; with several
# candidates it is ambiguous and goes to review.
same_model = [c for c in candidates
if c["model_norm"] == listing.model_norm and _eq(c.get("storage_gb"), listing.storage_gb)
and (c.get("ram_gb") is None) != (listing.ram_gb is None)]
if len(same_model) == 1:
return MatchDecision(same_model[0]["id"], "variant_key", 0.85, "auto")
if len(same_model) > 1:
return MatchDecision(same_model[0]["id"], "variant_key", 0.6, "pending")
best, best_score = None, 0.0
for c in candidates:
if not (_eq(c.get("storage_gb"), listing.storage_gb) and _eq(c.get("ram_gb"), listing.ram_gb)):
continue
if listing.category == "laptops" and (c.get("processor") or listing.processor) and c.get("processor") != listing.processor:
continue
if _number_tokens(c["model_norm"]) != _number_tokens(listing.model_norm):
continue
score = fuzz.token_set_ratio(c["model_norm"], listing.model_norm or "")
# token_set_ratio treats "galaxy s24" and "galaxy s24 ultra" as a
# subset match (100). Different words mean different models.
extra = set((listing.model_norm or "").split()) ^ set(c["model_norm"].split())
if extra:
score = min(score, fuzz.ratio(c["model_norm"], listing.model_norm or ""))
if score > best_score:
best, best_score = c, score
if best is not None and best_score >= AUTO_RATIO:
return MatchDecision(best["id"], "fuzzy", round(best_score / 100 * 0.9, 2), "auto")
if best is not None and best_score >= REVIEW_RATIO:
return MatchDecision(best["id"], "fuzzy", round(best_score / 100 * 0.8, 2), "pending")
return MatchDecision(None, "variant_key", 0.9, "auto")

View File

@@ -0,0 +1,105 @@
"""Rebuild canonical products for a category from the listings already stored.
Products and listing links are derived data: every fact lives on the listing
(title, snippet evidence, specs, URL). When the parsing or matching rules
improve, this re-runs them over the stored listings - no network requests -
and keeps each image attached to the listing it was found on.
Review decisions (approved/rejected links) are lost, because the products they
pointed at are rebuilt; uncertain matches simply come back to the queue.
"""
from __future__ import annotations
import logging
from decimal import Decimal
from typing import Dict, List
from app.electronics.db import repository as repo
from app.electronics.db.connection import connect, transaction
from app.electronics.match.matcher import decide
from app.electronics.models import Listing
from app.electronics.normalise.title_parser import fill_from_context, parse_title, variant_key
logger = logging.getLogger(__name__)
_ORDER = {"brand_official": 0, "scraped_page": 1, "search_snippet": 2}
def _listing_from_row(row: dict, category: str) -> Listing:
parsed = parse_title(row["title"], category, expected_brand=row["brand_slug"])
snippet = ""
if row["source_type"] == "search_snippet" and " — " in row["evidence_text"]:
snippet = row["evidence_text"].split(" — ", 1)[1].split(" || ", 1)[0]
raw = row["specs_raw"] or {}
spec_texts = tuple(str(v) for k, v in raw.items() if "processor" in k.lower() or "cpu" in k.lower())
spec_texts += (str((row["specs"] or {}).get("processor") or ""),)
fill_from_context(parsed, category, snippet=snippet, spec_texts=spec_texts)
l = Listing(
site_domain=row["domain"], source_sku=row["source_sku"], source_url=row["source_url"],
source_type=row["source_type"], brand_slug=row["brand_slug"], category=category,
title=row["title"], evidence_text=row["evidence_text"], confidence=float(row["confidence"]),
parser=row["parser"], family=parsed.brand.family if parsed.brand else row["family"],
model=parsed.model, model_number=row["model_number"] or parsed.mpn,
ram_gb=parsed.ram_gb, storage_gb=parsed.storage_gb, colour=row["colour"],
gtin=row["gtin"], specs=row["specs"] or {},
)
l.model_norm, l.processor = parsed.model_norm, parsed.processor
l.variant_key = variant_key(parsed, category) if parsed.brand else None
return l
def rematch(category: str) -> Dict[str, int]:
stats: Dict[str, int] = {"listings": 0, "linked": 0, "pending": 0, "unlinked": 0, "products": 0, "images": 0}
with connect() as conn:
rows = conn.execute(
"""
SELECT l.*, b.slug AS brand_slug, s.domain
FROM elec.source_listing l
JOIN elec.brand b ON b.id = l.brand_id
JOIN elec.site s ON s.id = l.site_id
JOIN elec.category c ON c.id = l.category_id
WHERE c.slug = %s
""",
(category,),
).fetchall()
images = conn.execute(
"""
SELECT i.url, i.source_listing_id, i.source_type, i.rank FROM elec.product_image i
JOIN elec.product p ON p.id = i.product_id JOIN elec.category c ON c.id = p.category_id
WHERE c.slug = %s
""",
(category,),
).fetchall()
with transaction() as conn:
# Maps and images cascade from the products.
conn.execute(
"DELETE FROM elec.product p USING elec.category c WHERE c.id = p.category_id AND c.slug = %s",
(category,),
)
ids = repo.id_maps()
product_of_listing: Dict[int, int] = {}
rows.sort(key=lambda r: (_ORDER.get(r["source_type"], 9), r["id"]))
for row in rows:
stats["listings"] += 1
listing = _listing_from_row(row, category)
decision = decide(listing, repo.product_candidates(listing.brand_slug, category))
if decision is None:
stats["unlinked"] += 1
continue
product_id = decision.product_id or repo.create_product(listing, ids)
stats["products"] += decision.product_id is None
repo.map_listing(row["id"], product_id, decision.method, decision.confidence, decision.review_status)
product_of_listing[row["id"]] = product_id
if decision.review_status == "pending":
stats["pending"] += 1
else:
stats["linked"] += 1
repo.merge_product_specs(product_id, listing.specs, {}, listing.source_url)
for img in images:
pid = product_of_listing.get(img["source_listing_id"])
if pid is not None:
repo.add_image(pid, img["url"], img["source_listing_id"], img["source_type"], img["rank"])
stats["images"] += 1
stats.update({f"products_{k}": v for k, v in repo.refresh_verification().items()})
return stats

View File

@@ -0,0 +1,68 @@
"""The record a collector produces for one product page / search result."""
from __future__ import annotations
from dataclasses import dataclass, field
from decimal import Decimal
from typing import Any, Dict, List, Optional
SOURCE_TYPES = ("scraped_page", "search_snippet", "brand_official")
@dataclass
class Listing:
site_domain: str
source_sku: str
source_url: str
source_type: str
brand_slug: str
category: str
title: str
evidence_text: str
confidence: float
parser: str
family: Optional[str] = None
model: Optional[str] = None
model_number: Optional[str] = None
ram_gb: Optional[Decimal] = None
storage_gb: Optional[Decimal] = None
colour: Optional[str] = None
price: Optional[Decimal] = None
mrp: Optional[Decimal] = None
availability: Optional[str] = None
in_stock: Optional[bool] = None
pincode: Optional[str] = None
pincode_applied: bool = False
rating: Optional[Decimal] = None
review_count: Optional[int] = None
# Customer reviews the page itself publishes (schema.org Review); stored
# in elec.listing_review, not on the listing row.
reviews: List[Dict[str, Any]] = field(default_factory=list)
gtin: Optional[str] = None
image_urls: List[str] = field(default_factory=list)
specs_raw: Dict[str, Any] = field(default_factory=dict)
specs: Dict[str, Any] = field(default_factory=dict)
spec_sources: Dict[str, str] = field(default_factory=dict)
search_query: Optional[str] = None
content_hash: Optional[str] = None
# Not stored on the listing; used for matching.
variant_key: Optional[str] = None
model_norm: Optional[str] = None
processor: Optional[str] = None
def validate(self) -> None:
"""The anti-fabrication contract, checked before anything is written."""
if self.source_type not in SOURCE_TYPES:
raise ValueError(f"bad source_type {self.source_type!r}")
if not self.source_url.startswith(("http://", "https://")):
raise ValueError("listing without a real source URL")
if not self.evidence_text.strip():
raise ValueError("listing without evidence text")
if self.price is not None:
if not (Decimal(500) <= self.price <= Decimal(1000000)):
raise ValueError(f"implausible price {self.price}")
if self.mrp is not None and self.price is not None and self.mrp < self.price:
self.mrp = None
if self.pincode_applied and not self.pincode:
raise ValueError("pincode_applied without a pincode")
if not 0 <= self.confidence <= 1:
raise ValueError("confidence out of range")

View File

View File

@@ -0,0 +1,61 @@
"""Per-host circuit breaker.
One 403, 429, 503 or CAPTCHA page opens the breaker for that host for
ELEC_BREAKER_COOLDOWN_HOURS. While it is open the host is not requested at all
and its products are collected from web search results instead. There is no
retry-with-a-different-identity: a block is an answer.
"""
from __future__ import annotations
import threading
import time
from typing import Callable, Dict, Optional, Tuple
from app.infrastructure.settings import ELEC_BREAKER_COOLDOWN_HOURS
class CircuitBreaker:
def __init__(
self,
cooldown_seconds: float = ELEC_BREAKER_COOLDOWN_HOURS * 3600,
on_trip: Optional[Callable[[str, str, float], None]] = None,
clock: Callable[[], float] = time.time,
) -> None:
self.cooldown = cooldown_seconds
self.on_trip = on_trip
self._clock = clock
self._open: Dict[str, Tuple[float, str]] = {}
self._lock = threading.Lock()
@staticmethod
def _key(host: str) -> str:
host = host.lower()
return host[4:] if host.startswith("www.") else host
def preload(self, host: str, until_epoch: float, reason: str) -> None:
"""Restore a breaker that was opened in an earlier run (elec.site)."""
if until_epoch > self._clock():
with self._lock:
self._open[self._key(host)] = (until_epoch, reason)
def trip(self, host: str, reason: str) -> None:
until = self._clock() + self.cooldown
with self._lock:
self._open[self._key(host)] = (until, reason)
if self.on_trip:
self.on_trip(self._key(host), reason, until)
def is_open(self, host: str) -> bool:
key = self._key(host)
with self._lock:
entry = self._open.get(key)
if not entry:
return False
if entry[0] <= self._clock():
del self._open[key]
return False
return True
def reason(self, host: str) -> Optional[str]:
entry = self._open.get(self._key(host))
return entry[1] if entry else None

View File

@@ -0,0 +1,223 @@
"""The only way this project fetches a retail or brand web page.
What it guarantees, for every request:
* robots.txt is consulted first (protego). If robots.txt cannot be read
because the server errors or blocks it, the site is treated as disallowed.
* at least ELEC_SITE_MIN_INTERVAL_SECONDS between requests to one host.
* an honest User-Agent naming the project and a contact address.
* no JavaScript, no cookies kept between requests, no proxies, no retries on
403/429 - a block is respected, not worked around.
* a size cap on the response body.
* a circuit breaker: a 403/429/CAPTCHA response opens it for the host, and
every later request to that host is refused until the cooldown passes.
* every request is reported to `on_fetch` (the fetch_log table).
"""
from __future__ import annotations
import logging
import re
import threading
import time
from dataclasses import dataclass
from typing import Callable, Dict, Optional, Tuple
from urllib.parse import urlparse
import httpx
from protego import Protego
from app.electronics.net.breaker import CircuitBreaker
from app.infrastructure.settings import (
ELEC_MAX_PAGE_BYTES,
ELEC_SITE_MIN_INTERVAL_SECONDS,
REQUEST_TIMEOUT_SECONDS,
USER_AGENT,
)
logger = logging.getLogger(__name__)
ROBOTS_TTL_SECONDS = 24 * 3600
# Pages that are a bot check rather than content. Matched on the first 20 KB.
_CAPTCHA_MARKERS = re.compile(
r"captcha|robot\s*check|are\s+you\s+a\s+robot|verify\s+you\s+are\s+human|"
r"/errors/validatecaptcha|px-captcha|cf-challenge|challenge-platform|access\s+denied|"
r"unusual\s+traffic|request\s+blocked|bot\s+detection|akamai.*reference",
re.IGNORECASE,
)
@dataclass
class FetchResult:
url: str
final_url: str
status: Optional[int]
text: str
outcome: str # ok | robots_disallowed | blocked | captcha | breaker_open | http_error | network_error | too_large | not_html
robots_allowed: Optional[bool]
bytes: int = 0
@property
def ok(self) -> bool:
return self.outcome == "ok"
class PoliteClient:
def __init__(
self,
*,
min_interval: float = ELEC_SITE_MIN_INTERVAL_SECONDS,
breaker: Optional[CircuitBreaker] = None,
on_fetch: Optional[Callable[[FetchResult, str], None]] = None,
transport: Optional[httpx.BaseTransport] = None,
sleep: Callable[[float], None] = time.sleep,
clock: Callable[[], float] = time.monotonic,
) -> None:
self.min_interval = min_interval
self.breaker = breaker or CircuitBreaker()
self.on_fetch = on_fetch
self._sleep = sleep
self._clock = clock
self._last: Dict[str, float] = {}
self._locks: Dict[str, threading.Lock] = {}
self._robots: Dict[str, Tuple[float, Optional[Protego], bool]] = {}
self._guard = threading.Lock()
self._client = httpx.Client(
headers={
"User-Agent": USER_AGENT,
"Accept": "text/html,application/xhtml+xml,application/json;q=0.9,*/*;q=0.5",
"Accept-Language": "en-IN,en;q=0.9",
},
follow_redirects=True,
timeout=REQUEST_TIMEOUT_SECONDS,
transport=transport,
)
def close(self) -> None:
self._client.close()
def __enter__(self) -> "PoliteClient":
return self
def __exit__(self, *exc) -> None:
self.close()
# -- pacing --------------------------------------------------------------
def _host_lock(self, host: str) -> threading.Lock:
with self._guard:
return self._locks.setdefault(host, threading.Lock())
def _wait_turn(self, host: str) -> None:
last = self._last.get(host)
if last is not None:
gap = self.min_interval - (self._clock() - last)
if gap > 0:
self._sleep(gap)
self._last[host] = self._clock()
# -- robots.txt ----------------------------------------------------------
def _robots_for(self, scheme: str, host: str) -> Tuple[Optional[Protego], bool]:
"""(parser, reachable). parser None + reachable True = no robots.txt
(everything allowed); reachable False = could not read it (deny)."""
cached = self._robots.get(host)
if cached and self._clock() - cached[0] < ROBOTS_TTL_SECONDS:
return cached[1], cached[2]
url = f"{scheme}://{host}/robots.txt"
parser: Optional[Protego] = None
reachable = False
self._wait_turn(host)
try:
resp = self._client.get(url)
if resp.status_code == 200:
parser, reachable = Protego.parse(resp.text), True
elif resp.status_code in (404, 410):
parser, reachable = None, True
else:
reachable = False
if resp.status_code in (403, 429):
self.breaker.trip(host, f"robots.txt returned HTTP {resp.status_code}")
except httpx.HTTPError as exc:
logger.info("robots.txt unreachable for %s: %s", host, exc)
self._robots[host] = (self._clock(), parser, reachable)
return parser, reachable
def robots_allowed(self, url: str) -> bool:
p = urlparse(url)
parser, reachable = self._robots_for(p.scheme or "https", p.netloc.lower())
if not reachable:
return False
return True if parser is None else bool(parser.can_fetch(url, USER_AGENT))
# -- fetch ---------------------------------------------------------------
def _report(self, result: FetchResult) -> FetchResult:
if self.on_fetch:
try:
self.on_fetch(result, urlparse(result.url).netloc.lower())
except Exception as exc: # noqa: BLE001 - logging must never break a crawl
logger.debug("fetch log failed: %s", exc)
return result
def get(self, url: str, *, check_robots: bool = True, accept_non_html: bool = False) -> FetchResult:
host = urlparse(url).netloc.lower()
if self.breaker.is_open(host):
return FetchResult(url, url, None, "", "breaker_open", None)
with self._host_lock(host):
allowed: Optional[bool] = None
if check_robots:
allowed = self.robots_allowed(url)
if not allowed:
return self._report(FetchResult(url, url, None, "", "robots_disallowed", False))
self._wait_turn(host)
try:
with self._client.stream("GET", url) as resp:
status = resp.status_code
final = str(resp.url)
ctype = resp.headers.get("content-type", "").lower()
body = bytearray()
too_large = False
for chunk in resp.iter_bytes():
body.extend(chunk)
if len(body) > ELEC_MAX_PAGE_BYTES:
too_large = True
break
encoding = resp.encoding or "utf-8"
except httpx.HTTPError as exc:
logger.info("fetch failed %s: %s", url, exc)
return self._report(FetchResult(url, url, None, "", "network_error", allowed))
text = bytes(body).decode(encoding, errors="replace") if body else ""
n = len(body)
if status in (403, 429, 503) or (status == 200 and _CAPTCHA_MARKERS.search(text[:20000]) and len(text) < 60000):
outcome = "captcha" if status == 200 or _CAPTCHA_MARKERS.search(text[:20000]) else "blocked"
self.breaker.trip(host, f"HTTP {status} ({outcome})")
return self._report(FetchResult(url, final, status, "", outcome, allowed, n))
if status != 200:
return self._report(FetchResult(url, final, status, "", "http_error", allowed, n))
if too_large:
return self._report(FetchResult(url, final, status, "", "too_large", allowed, n))
if not accept_non_html and "html" not in ctype and "json" not in ctype:
return self._report(FetchResult(url, final, status, "", "not_html", allowed, n))
return self._report(FetchResult(url, final, status, text, "ok", allowed, n))
def check_image(self, url: str, min_bytes: int) -> bool:
"""One ranged GET to confirm a URL serves a real image. Paced per host
like any request; robots.txt is not consulted because this fetches a
single file the product page itself references, as a browser would."""
host = urlparse(url).netloc.lower()
if not url.startswith(("http://", "https://")) or self.breaker.is_open(host):
return False
with self._host_lock(host):
self._wait_turn(host)
try:
with self._client.stream("GET", url, headers={"Accept": "image/*", "Range": f"bytes=0-{min_bytes * 4}"}) as resp:
if resp.status_code not in (200, 206):
return False
if not resp.headers.get("content-type", "").lower().startswith("image/"):
return False
got = 0
for chunk in resp.iter_bytes():
got += len(chunk)
if got >= min_bytes:
return True
return got >= min_bytes
except httpx.HTTPError:
return False

View File

@@ -0,0 +1,86 @@
"""Resolve the brand of a product title against the closed allow-list."""
from __future__ import annotations
import re
from dataclasses import dataclass
from functools import lru_cache
from typing import List, Optional, Tuple
from app.electronics.reference import load_reference
@dataclass(frozen=True)
class BrandMatch:
brand_slug: str
brand_name: str
family: Optional[str] # sub-brand (Redmi, iQOO, Pixel...) when the title used one
matched: str # the alias text found in the title
@lru_cache(maxsize=1)
def _alias_table() -> List[Tuple[str, str, bool]]:
"""(alias, brand_slug, is_sub_brand), longest alias first."""
ref = load_reference()
rows: List[Tuple[str, str, bool]] = []
for b in ref.brands.values():
for a in b.aliases:
rows.append((a, b.slug, False))
for s in b.sub_brands:
rows.append((s, b.slug, True))
rows.sort(key=lambda r: -len(r[0]))
return rows
def resolve_brand(title: str, *, expected: Optional[str] = None) -> Optional[BrandMatch]:
"""The allow-listed brand a title starts with (or names within its first
few words), or None. `expected` restricts the match to one brand slug.
Only the start of the title is considered: "Case for Samsung Galaxy S24"
is an accessory, not a Samsung phone.
"""
if not title:
return None
ref = load_reference()
head = " ".join(re.findall(r"[a-z0-9+]+", title.lower())[:3])
for alias, slug, is_sub in _alias_table():
if expected and slug != expected:
continue
pattern = r"(?:^|\s)" + re.escape(alias) + r"(?:\s|$)"
m = re.search(pattern, head)
if not m:
continue
# The brand/sub-brand must be the first or second word ("Apple iPhone",
# "Samsung Galaxy", "Xiaomi Redmi Note") - not buried later.
if len(head[: m.start()].split()) > 1:
continue
# "Google Pixel 8", "Xiaomi Redmi Note 13": the parent brand matched,
# but the family is the sub-brand that follows it.
sub = alias if is_sub else next(
(s for s in ref.brands[slug].sub_brands if re.search(r"(?:^|\s)" + re.escape(s) + r"(?:\s|$)", head)),
None,
)
return BrandMatch(slug, ref.brands[slug].name, _family_casing(sub) if sub else None, alias)
return None
_CASING = {"iphone": "iPhone", "iqoo": "iQOO", "macbook": "MacBook", "rog": "ROG", "tuf": "TUF",
"cmf": "CMF", "loq": "LOQ", "poco": "POCO", "mi": "Mi", "xps": "XPS", "thinkpad": "ThinkPad",
"ideapad": "IdeaPad", "thinkbook": "ThinkBook", "vivobook": "Vivobook", "zenbook": "Zenbook"}
def _family_casing(sub: str) -> str:
return _CASING.get(sub, sub.title())
# Words that mark an accessory or a non-product page, not a device.
_NOT_A_DEVICE = re.compile(
r"\b(?:case|cover|back\s+cover|tempered|screen\s+guard|protector|charger|adapter|cable|"
r"skin|sleeve|bag|backpack|stand|holder|refurbished|renewed|pre-?owned|used|"
r"compare|vs\.?|versus|review|specifications?\s+and|price\s+list|best\s+\w+\s+under|"
r"top\s+\d+|all\s+models)\b",
re.IGNORECASE,
)
def looks_like_device_title(title: str) -> bool:
return bool(title) and not _NOT_A_DEVICE.search(title)

View File

@@ -0,0 +1,55 @@
"""Is a value actually stated in the text it supposedly came from?
Every value the LLM returns passes through value_in_source() against the exact
text the model was shown. Anything that cannot be found there is discarded,
which is what stops a small model's guess becoming a stored fact.
"""
from __future__ import annotations
import re
from decimal import Decimal, InvalidOperation
from typing import Union
_WS = re.compile(r"\s+")
def _norm_text(text: str) -> str:
text = text.lower().replace(" ", " ")
text = re.sub(r"[^\w.+ ]+", " ", text)
text = re.sub(r"(?<!\d)\.|\.(?!\d)", " ", text) # sentence full stops, not decimals
# "5000mAh" and "5000 mAh" must compare equal.
text = re.sub(r"(?<=\d)(?=[a-z])|(?<=[a-z])(?=\d)", " ", text)
return _WS.sub(" ", text).strip()
def _numbers_in(text: str) -> set:
found = set()
for raw in re.findall(r"\d[\d,]*(?:\.\d+)?", text):
try:
found.add(Decimal(raw.replace(",", "")).normalize())
except InvalidOperation:
continue
return found
def value_in_source(value: Union[str, int, float, Decimal, None], source: str) -> bool:
if value is None or not source:
return False
if isinstance(value, bool):
return False
if isinstance(value, (int, float, Decimal)):
try:
return Decimal(str(value)).normalize() in _numbers_in(source)
except InvalidOperation:
return False
text = str(value).strip()
if not text:
return False
# A string with a number in it ("5000 mAh", "Snapdragon 8 Gen 3") must have
# every one of its numbers in the source, and its words too.
nums = _numbers_in(text)
if nums and not nums <= _numbers_in(source):
return False
words = [w for w in _norm_text(text).split() if not re.fullmatch(r"[\d.,]+", w)]
hay = f" {_norm_text(source)} "
return all(f" {w} " in hay for w in words) if words else bool(nums)

View File

@@ -0,0 +1,71 @@
"""Fill MISSING spec keys from page text with the local LLM - and keep only
what the text actually says.
The model sees one block of text that we fetched (a spec section or a
description) and is asked to copy values out of it. Every value it returns is:
1. checked by grounding.value_in_source() against that same text, and
2. normalised by spec_normaliser (units, plausible ranges).
Anything failing either step is dropped. Prices, product names and images are
never asked of the model.
"""
from __future__ import annotations
import json
import logging
from typing import Any, Dict, Iterable, Tuple
from app.electronics.normalise.grounding import value_in_source
from app.electronics.normalise.spec_normaliser import normalise_value
from app.infrastructure.settings import ELEC_USE_LLM
logger = logging.getLogger(__name__)
MAX_SOURCE_CHARS = 3500
SYSTEM_PROMPT = (
"You copy product specifications out of the text you are given. "
"Rules: use ONLY the given text; copy each value exactly as written, including its unit; "
"if the text does not state a value, use null; never guess, estimate or use outside knowledge. "
"Reply with one JSON object whose keys are exactly the requested keys."
)
def fill_missing(
category: str,
source_text: str,
missing_keys: Iterable[str],
*,
generate=None,
) -> Tuple[Dict[str, Any], Dict[str, str]]:
"""(specs, sources) for whichever of `missing_keys` the text states."""
keys = [k for k in missing_keys if k != "colour"]
text = (source_text or "").strip()[:MAX_SOURCE_CHARS]
if not keys or not text or not ELEC_USE_LLM:
return {}, {}
if generate is None:
from app.services.ollama_service import generate_json as generate
prompt = (
f"Requested keys: {json.dumps(keys)}\n\n"
f"Text:\n\"\"\"\n{text}\n\"\"\"\n\n"
"JSON:"
)
reply = generate(SYSTEM_PROMPT, prompt)
if not isinstance(reply, dict):
return {}, {}
specs: Dict[str, Any] = {}
sources: Dict[str, str] = {}
for key in keys:
raw = reply.get(key)
if raw is None or isinstance(raw, (dict, list, bool)):
continue
if not value_in_source(raw, text):
logger.debug("LLM value %r for %s not found in source text; dropped", raw, key)
continue
value = normalise_value(category, key, raw)
if value is None:
continue
specs[key] = value
sources[key] = f"llm-extracted: {str(raw)[:80]}"
return specs, sources

View File

@@ -0,0 +1,101 @@
"""Map raw spec labels/values from a page to canonical keys and units.
Deterministic and table-driven (reference/spec_keys.yaml). A value that cannot
be parsed, or lands outside the plausible range for its key, is dropped - never
estimated.
"""
from __future__ import annotations
import re
from decimal import Decimal, InvalidOperation
from functools import lru_cache
from typing import Any, Dict, Optional, Tuple
from app.electronics.reference import load_reference
def _label(text: str) -> str:
return re.sub(r"[^a-z0-9]+", " ", str(text).lower()).strip()
@lru_cache(maxsize=None)
def _synonyms(category: str) -> Dict[str, str]:
table: Dict[str, str] = {}
for key, spec in load_reference().spec_keys.get(category, {}).items():
for syn in [key.replace("_", " "), *spec.get("synonyms", [])]:
table.setdefault(_label(syn), key)
return table
def canonical_key(category: str, label: str) -> Optional[str]:
return _synonyms(category).get(_label(label))
# unit -> (regex for the unit in text, factor into the canonical unit)
_UNIT_PATTERNS = {
"GB": [(r"tb", Decimal(1024)), (r"gb", Decimal(1)), (r"mb", Decimal(1) / 1024)],
"inch": [(r"(?:inch(?:es)?|in\b|\"|”)", Decimal(1)), (r"cm", Decimal(1) / Decimal("2.54"))],
"Hz": [(r"hz", Decimal(1))],
"MP": [(r"mp|megapixel", Decimal(1))],
"mAh": [(r"mah", Decimal(1))],
"kg": [(r"kg|kilogram", Decimal(1)), (r"(?<![k])g\b|grams?", Decimal("0.001"))],
"Wh": [(r"wh|watt\s*hours?", Decimal(1))],
}
def _to_number(value: str, unit: str) -> Optional[Decimal]:
text = str(value).lower().replace(",", "")
patterns = _UNIT_PATTERNS.get(unit, [])
# Prefer an amount written in the canonical unit ("39.62 cm (15.6 inch)" -> 15.6).
for unit_re, factor in patterns:
m = re.search(r"(\d+(?:\.\d+)?)\s*(?:" + unit_re + r")", text)
if m:
try:
return (Decimal(m.group(1)) * factor).quantize(Decimal("0.01")).normalize()
except InvalidOperation:
return None
# A bare number is accepted only when nothing else is in the value.
m = re.fullmatch(r"\s*(\d+(?:\.\d+)?)\s*", text)
if m:
return Decimal(m.group(1)).normalize()
return None
def normalise_value(category: str, key: str, value: Any) -> Optional[Any]:
spec = load_reference().spec_keys.get(category, {}).get(key)
if spec is None or value is None:
return None
text = str(value).strip()
if not text or text.lower() in {"na", "n/a", "-", "none", "not applicable", "no"}:
return None
kind = spec.get("type")
if kind == "number":
num = _to_number(text, spec.get("unit", ""))
if num is None:
return None
lo, hi = spec.get("range", [None, None])
if (lo is not None and num < Decimal(str(lo))) or (hi is not None and num > Decimal(str(hi))):
return None
return float(num) if num != num.to_integral() else int(num)
if kind == "enum":
low = text.lower()
for canon, words in spec.get("values", {}).items():
if any(re.search(r"\b" + re.escape(w) + r"\b", low) for w in words):
return canon
return None
return re.sub(r"\s+", " ", text)[:120]
def normalise_specs(category: str, raw: Dict[str, Any]) -> Tuple[Dict[str, Any], Dict[str, str]]:
"""(specs, sources): canonical key -> value, and key -> the raw label it came from."""
specs: Dict[str, Any] = {}
sources: Dict[str, str] = {}
for label, value in (raw or {}).items():
key = canonical_key(category, label)
if not key or key in specs:
continue
norm = normalise_value(category, key, value)
if norm is not None:
specs[key] = norm
sources[key] = f"{label}: {value}"[:200]
return specs, sources

View File

@@ -0,0 +1,374 @@
"""Split a retail product title into model, variant and a matching key.
Everything returned is read from the title text; a value the title does not
state is None. Titles differ a lot between sites:
Samsung Galaxy S24 5G (Onyx Black, 8GB RAM, 256GB Storage) Amazon
SAMSUNG Galaxy S24 5G (Onyx Black, 256 GB) (8 GB RAM) Flipkart
Samsung Galaxy S24 5G (8GB RAM, 256GB, Onyx Black) Croma
Redmi Note 13 Pro 5G (8GB + 256GB)
Apple iPhone 15 (128 GB) - Black
HP 15s, 13th Gen Intel Core i5-1334U, 16GB DDR4, 512GB SSD, ... fd0112TU
so the model is taken from the text before the first bracket/comma, and RAM /
storage / colour from anywhere in the title.
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
from decimal import Decimal
from typing import List, Optional
from app.electronics.normalise.brand_alias import BrandMatch, resolve_brand
_NUM = r"(\d+(?:\.\d+)?)"
# "8GB RAM", "8 GB LPDDR5X RAM", "RAM 8GB", "16GB DDR4" (laptops)
_RAM_RES = [
re.compile(_NUM + r"\s*GB\s*(?:LP)?(?:DDR\s?\d\w?\s*)?RAM\b", re.IGNORECASE),
re.compile(r"\bRAM\s*[:\-]?\s*" + _NUM + r"\s*GB", re.IGNORECASE),
re.compile(_NUM + r"\s*GB\s*(?:LP)?DDR\s?\d", re.IGNORECASE),
re.compile(_NUM + r"\s*GB\s*(?:unified\s+memory|memory)\b", re.IGNORECASE),
]
# "8GB + 256GB", "8/256", "8GB/256GB", "12+512GB"
_PAIR_RE = re.compile(r"(?<![\d.])(\d{1,2})\s*(?:GB)?\s*[+/]\s*(\d{2,4}|1|2)\s*(GB|TB)?\b", re.IGNORECASE)
# explicit storage: "256GB Storage", "512GB SSD", "1TB", "256 GB ROM"
_STORAGE_LABELLED = re.compile(
_NUM + r"\s*(GB|TB)\s*(?:SSD|ROM|storage|internal(?:\s+storage)?|HDD|eMMC|UFS|NVMe|PCIe)\b", re.IGNORECASE
)
_SIZE_ANY = re.compile(r"(?<![\d.])" + _NUM + r"\s*(GB|TB)\b", re.IGNORECASE)
# Order matters: the first pattern that matches wins. AMD comes before the
# Intel "Core N" pattern, because retail titles write core counts as words
# ("Ryzen 3 Quad Core 7320U"), and "Core 7320U" must not read as Intel.
_CORE_COUNT = r"(?:(?:Dual|Quad|Hexa|Octa|Six|Eight)\s+Core\s+)?"
_PROCESSOR_RES = [
re.compile(r"\b(?:AMD\s+)?Ryzen\s+R?(\d)\s*(?:Pro\s+)?" + _CORE_COUNT + r"[- ]?(\d{4}[A-Z]{0,3})\b", re.IGNORECASE),
re.compile(r"\b(?:AMD\s+)?(Athlon)\s+(?:Silver\s+|Gold\s+)?" + _CORE_COUNT + r"(\d{4}[A-Z]{0,2})\b", re.IGNORECASE),
re.compile(r"\b(?:Intel\s+)?Core\s+Ultra\s+(\d)\s*[- ]?\s*(\d{3}[A-Z]{0,2})\b", re.IGNORECASE),
re.compile(r"\b(?:Intel\s+)?Core\s+(?:\(?\d+(?:th|nd|rd|st)\s+Gen\)?\s+)?(i[3579])\s*[- ]?\s*(\d{4,5}[A-Z]{0,3})\b", re.IGNORECASE),
re.compile(r"\b(?:Intel\s+)?Core\s+(i[3579])\s+\d+(?:th|nd|rd|st)\s+Gen\s+(\d{4,5}[A-Z]{0,3})\b", re.IGNORECASE),
re.compile(r"(?<!Dual )(?<!Quad )(?<!Hexa )(?<!Octa )(?<!Six )(?<!Eight )\b(?:Intel\s+)?Core\s+([3579])\s*[- ]?\s*(\d{3}[A-Z]{0,2})\b", re.IGNORECASE),
re.compile(r"\bApple\s+(M[1-9])(?:\s+(Pro|Max|Ultra))?\b", re.IGNORECASE),
re.compile(r"\b(M[1-9])\s*(Pro|Max|Ultra)?\s+chip\b", re.IGNORECASE),
re.compile(r"\bSnapdragon\s+(X\d?)\s*(Elite|Plus)?\s*(X\d{1,2}-\d{3})?", re.IGNORECASE),
re.compile(r"\b(?:Intel\s+)?(Celeron|Pentium(?:\s+Silver|\s+Gold)?)\s+(N?\d{3,5}[A-Z]?)\b", re.IGNORECASE),
re.compile(r"\bMediaTek\s+(Kompanio\s+\d{3,4}|MT\d{4})\b", re.IGNORECASE),
]
_COLOUR_WORDS = re.compile(
r"\b(black|white|blue|green|red|grey|gray|silver|gold|purple|violet|pink|yellow|orange|"
r"cream|titanium|graphite|midnight|starlight|mint|lavender|bronze|copper|beige|teal|"
r"navy|jade|coral|onyx|marble|obsidian|porcelain|hazel|aqua|lime|sand|charcoal)\b",
re.IGNORECASE,
)
# Tokens that describe the device class or connectivity, not the model.
_MODEL_NOISE = re.compile(
r"\b(?:5g|4g|lte|smartphone|smart\s+phone|mobile\s+phone|mobile|phone|dual\s+sim|"
r"laptop|notebook|thin\s+and\s+light|thin\s+&\s+light|gaming|new|latest|with\b.*$)",
re.IGNORECASE,
)
_SPEC_TOKEN = re.compile(
r"^(?:\d+(?:\.\d+)?\s*(?:gb|tb|mp|mah|hz|inch|inches|cm|w|kg|g)|\d+(?:th|nd|rd|st)|gen|ddr\d?\w*|"
r"lpddr\d\w*|ssd|hdd|fhd|qhd|uhd|oled|ips|win|windows|20\d\d)$",
re.IGNORECASE,
)
@dataclass
class ParsedTitle:
title: str
brand: Optional[BrandMatch]
model: Optional[str] = None # "Galaxy S24", "Redmi Note 13 Pro", "15s"
model_norm: Optional[str] = None # "galaxy s24", matching form
ram_gb: Optional[Decimal] = None
storage_gb: Optional[Decimal] = None
colour: Optional[str] = None
processor: Optional[str] = None # "i5-1334u", "ryzen 5 7530u", "m3"
mpn: Optional[str] = None # laptop part number when stated
network: Optional[str] = None
notes: List[str] = field(default_factory=list)
def _dec(value: str, unit: str = "GB") -> Decimal:
d = Decimal(value)
if unit.upper() == "TB":
d = d * 1024
return d.normalize() if d == d.to_integral() else d
def _parse_ram_storage(text: str, category: str):
ram = storage = None
for rx in _RAM_RES:
m = rx.search(text)
if m:
ram = _dec(m.group(1))
break
m = _PAIR_RE.search(text)
if m:
a, b, unit = m.group(1), m.group(2), (m.group(3) or "GB")
pair_ram, pair_storage = _dec(a), _dec(b, unit)
# "Core Ultra 5/ 16GB RAM/ 512GB" is not a 5 GB / 16 GB pair: a real
# pair has device-sized storage.
min_pair_storage = Decimal(16) if category == "mobiles" else Decimal(64)
if pair_storage > pair_ram and pair_storage >= min_pair_storage:
ram = ram if ram is not None else pair_ram
storage = pair_storage
if storage is None:
m = _STORAGE_LABELLED.search(text)
if m:
storage = _dec(m.group(1), m.group(2))
if storage is None:
# Unlabelled sizes: the storage is the largest one that is not the RAM.
min_storage = Decimal(16) if category == "mobiles" else Decimal(32)
sizes = [_dec(v, u) for v, u in _SIZE_ANY.findall(text)]
candidates = [s for s in sizes if s != ram and s >= min_storage]
if candidates:
storage = max(candidates)
if ram is None:
# "(8 GB RAM)" handled above; an unlabelled small size next to a larger
# one ("8GB 256GB") is the RAM.
sizes = [_dec(v, u) for v, u in _SIZE_ANY.findall(text)]
small = [s for s in sizes if s <= (24 if category == "mobiles" else 64) and (storage is None or s < storage)]
if len(set(small)) == 1 and storage is not None:
ram = small[0]
plausible_ram = Decimal(32) if category == "mobiles" else Decimal(128)
if ram is not None and not (Decimal(1) <= ram <= plausible_ram):
ram = None
return ram, storage
def parse_processor(text: str) -> Optional[str]:
"""Normalised CPU name ("i5-1334u", "ryzen 3 7320u", "core ultra 5 125h",
"core 5 120u", "athlon 7120u", "m2"), or None."""
for rx in _PROCESSOR_RES:
m = rx.search(text or "")
if not m:
continue
parts = [g for g in m.groups() if g]
matched = m.group(0).lower()
if "ryzen" in matched:
return f"ryzen {parts[0]} {parts[1]}".lower()
if "athlon" in matched:
return f"athlon {parts[1]}".lower()
if "ultra" in matched and "core" in matched:
return f"core ultra {parts[0]} {parts[1]}".lower()
if "core" in matched and parts[0].lower().startswith("i"):
return f"{parts[0]}-{parts[1]}".lower()
if "core" in matched:
return f"core {parts[0]} {parts[1]}".lower()
return " ".join(parts).lower()
return None
_parse_processor = parse_processor
def processor_is_specific(processor: Optional[str]) -> bool:
"""True for a CPU named down to its model number ("i5-1334u"), which
together with brand, model line, RAM and storage identifies a laptop
configuration. "m2" (Apple) also counts."""
if not processor:
return False
return bool(re.search(r"\d{3,}", processor)) or bool(re.fullmatch(r"m[1-9](?: (?:pro|max|ultra))?", processor))
_MPN_RE = re.compile(
r"(?<![\w-])([A-Z0-9]{2,8}-[A-Z0-9]{2,10}(?:-[A-Z0-9]{1,6})?|"
r"[A-Z0-9]{2,6}[A-Z]{0,4}\d{2,6}[A-Z]{1,4}\d{0,4}[A-Z]{0,3}|\d{2}[A-Z]{2}\d{3,}[A-Z0-9]{2,})(?![\w-])",
re.IGNORECASE,
)
def _parse_mpn(text: str, processor: Optional[str]) -> Optional[str]:
"""A manufacturer part number such as 82XV00BHIN or fd0112TU, when the
title states one (laptops). Tokens that are specs or CPU names are not."""
candidates = []
for m in _MPN_RE.finditer(text):
tok = m.group(1)
low = tok.lower()
if len(tok) < 6 or len(tok) > 20:
continue
if sum(c.isdigit() for c in tok) < 2 or sum(c.isalpha() for c in tok) < 2:
continue
if _SPEC_TOKEN.match(low) or re.match(r"^(?:i[3579]|m[1-9]|rtx|gtx|rx|ddr|lpddr)", low):
continue
if processor and low in processor.replace("-", " ").split() + [processor.replace(" ", "")]:
continue
if re.search(r"\d+(?:gb|tb|mp|mah|hz|w)$", low):
continue
if re.search(r"-(?:core|inch|cell|bit|gen|thread)s?$|^\d+-", low) and not re.search(r"[a-z]\d", low.split("-")[-1]):
continue # "10-Core", "15-inch", "3-Cell" describe hardware, not a part number
candidates.append(tok)
return candidates[-1].upper() if candidates else None
def _parse_colour(title: str) -> Optional[str]:
# Inside brackets first: "(Onyx Black, 8GB RAM, 256GB Storage)"
for group in re.findall(r"\(([^()]*)\)", title):
for part in re.split(r"[,|/]", group):
part = part.strip()
if part and not re.search(r"\d", part) and _COLOUR_WORDS.search(part):
return part.title()
# Trailing "- Black"
m = re.search(r"[-–|,]\s*([A-Za-z][A-Za-z ]{2,30})\s*$", title)
if m and _COLOUR_WORDS.search(m.group(1)) and not re.search(r"\d", m.group(1)):
return m.group(1).strip().title()
return None
def normalise_model(model: str) -> str:
text = model.lower()
text = re.sub(r"[()\[\],|]", " ", text)
text = _MODEL_NOISE.sub(" ", text)
text = re.sub(r"\+", " plus ", text)
text = re.sub(r"[^a-z0-9 ]+", " ", text)
tokens = [t for t in text.split() if not _SPEC_TOKEN.match(t)]
return " ".join(tokens)
_LAPTOP_SPEC_START = re.compile(
r"\b(?:intel|amd|apple\s+m[1-9]|m[1-9]\s+chip|core\s+(?:i[3579]|ultra)|ryzen|snapdragon|celeron|pentium|"
r"mediatek|\d+(?:th|nd|rd|st)\s+gen|\d+(?:\.\d+)?\s*(?:-|\s)?(?:inch|cm|\"))",
re.IGNORECASE,
)
_DISPLAY_NOISE = re.compile(
r"\b(?:5g|4g|lte|smartphone|smart\s+phone|mobile\s+phone|dual\s+sim|laptop|notebook|"
r"thin\s+and\s+light|thin\s+&\s+light|gaming|new|latest)\b",
re.IGNORECASE,
)
def _model_from_title(title: str, brand: Optional[BrandMatch], category: str, colour: Optional[str],
mpn: Optional[str] = None):
"""(display model, matching model_norm) from the head of the title."""
head = re.sub(r"^\s*buy\s+", "", title, flags=re.IGNORECASE)
if mpn:
# A part number is not the model name. HP writes the line into it
# ("15-fc0500AU" is an HP 15), so that prefix is kept.
prefix = mpn.split("-", 1)[0] if "-" in mpn and len(mpn.split("-", 1)[0]) <= 4 else ""
head = re.sub(re.escape(mpn), f" {prefix} ", head, flags=re.IGNORECASE)
# A short model token in brackets right after the name is part of it:
# "Nothing Phone (2a) 5G (Black, 128 GB)".
head = re.sub(r"\((?:19|20)\d\d\)", " ", head) # "(2026)" is a model year, not part of the name
head = re.sub(r"\(([A-Za-z0-9+ ]{1,6})\)", lambda m: " " + m.group(1) + " "
if not re.search(r"\d\s*(?:gb|tb)", m.group(1), re.I) else m.group(0), head, count=1)
head = re.split(r"\s[-–|]\s|[(,|\[:]", head, maxsplit=1)[0]
if category == "laptops":
m = _LAPTOP_SPEC_START.search(head)
if m and m.start() > 0:
head = head[: m.start()]
if brand:
# Drop the parent brand's own name ("Samsung Galaxy S24" -> "Galaxy S24",
# "Apple iPhone 15" -> "iPhone 15"); a sub-brand stays ("Redmi Note 13").
from app.electronics.reference import load_reference
parent_aliases = sorted(load_reference().brands[brand.brand_slug].aliases, key=len, reverse=True)
for alias in parent_aliases:
head = re.sub(r"^\s*" + re.escape(alias) + r"\b", "", head, flags=re.IGNORECASE).strip()
head = re.sub(r"\b\d+\s*GB\s*RAM\b", " ", head, flags=re.IGNORECASE)
head = _PAIR_RE.sub(" ", head)
head = _SIZE_ANY.sub(" ", head)
if colour:
head = re.sub(re.escape(colour), " ", head, flags=re.IGNORECASE)
words = head.split()
while len(words) > 1 and _COLOUR_WORDS.fullmatch(words[-1]):
words.pop() # "iPhone 15 Black" -> "iPhone 15"
head = " ".join(words)
display = re.sub(r"\s+", " ", _DISPLAY_NOISE.sub(" ", head)).strip(" -–")
norm = normalise_model(head)
if not norm:
return None, None
return display or head.strip(), norm
def parse_title(title: str, category: str, *, expected_brand: Optional[str] = None) -> ParsedTitle:
title = re.sub(r"\s+", " ", (title or "")).strip()
brand = resolve_brand(title, expected=expected_brand)
parsed = ParsedTitle(title=title, brand=brand)
if not title:
return parsed
parsed.ram_gb, parsed.storage_gb = _parse_ram_storage(title, category)
parsed.colour = _parse_colour(title)
if re.search(r"\b5G\b", title, re.IGNORECASE):
parsed.network = "5G"
if category == "laptops":
parsed.processor = _parse_processor(title)
parsed.mpn = _parse_mpn(title, parsed.processor)
parsed.model, parsed.model_norm = _model_from_title(title, brand, category, parsed.colour, parsed.mpn)
return parsed
def variant_key(parsed: ParsedTitle, category: str) -> Optional[str]:
"""The identity of one real-world variant, or None if the title does not
state enough to tell variants apart."""
if not parsed.brand or not parsed.model_norm:
return None
b = parsed.brand.brand_slug
fmt = lambda d: "na" if d is None else format(d.normalize(), "f") # noqa: E731
if category == "laptops":
# A laptop configuration is its model line + CPU + RAM + storage. That
# is what every site states (a part number is shown by only a few), so
# it is the key whenever it is complete; the MPN is the fallback.
line = laptop_line(parsed.model_norm)
if line and processor_is_specific(parsed.processor) and parsed.ram_gb and parsed.storage_gb:
return f"{b}|laptops|{line}|{parsed.processor}|{fmt(parsed.ram_gb)}|{fmt(parsed.storage_gb)}"
if parsed.mpn:
return f"{b}|laptops|mpn:{parsed.mpn.lower()}"
return None
if parsed.storage_gb is None:
return None
return f"{b}|{category}|{parsed.model_norm}|{fmt(parsed.ram_gb)}|{fmt(parsed.storage_gb)}"
# Lenovo/Asus machine-type codes ("15amn8", "15irh10", "14iah8", "x1504za")
# name a chassis generation, and one site prints them where another does not.
_MACHINE_CODE = re.compile(r"^(?:\d{2}[a-z]{2,4}\d{1,2}|[a-z]\d{4}[a-z]{1,3})$")
def laptop_line(model_norm: Optional[str]) -> str:
"""The model line used for matching: "ideapad slim 3 15amn8" -> "ideapad slim 3"."""
tokens = [t for t in (model_norm or "").split() if not _MACHINE_CODE.match(t)]
return " ".join(tokens)
def fill_from_context(parsed: ParsedTitle, category: str, *, snippet: str = "",
spec_texts: tuple = ()) -> ParsedTitle:
"""Fill variant fields a (often truncated) title leaves out, from text the
same site published about the same page: its search snippet, or the spec
table of the fetched page. Only unambiguous values are taken - a snippet
naming two different storage sizes is describing several variants."""
if snippet and (parsed.ram_gb is None or parsed.storage_gb is None):
sizes = {_dec(v, u) for v, u in _SIZE_ANY.findall(snippet)}
if len(sizes) <= 2:
ram, storage = _parse_ram_storage(snippet, category)
if parsed.storage_gb is None and storage is not None:
parsed.storage_gb = storage
if parsed.ram_gb is None and ram is not None and ram != parsed.storage_gb:
parsed.ram_gb = ram
if category == "laptops" and not processor_is_specific(parsed.processor):
# A title that already names a CPU family ("Snapdragon X", "Core i7")
# is only completed from the page's own spec table, never from a
# snippet - snippets often run several products' titles together.
sources = list(spec_texts) + ([snippet] if parsed.processor is None and snippet else [])
found = set()
for text in sources:
found |= {cpu for cpu in all_processors(text) if processor_is_specific(cpu)}
if len(found) == 1:
parsed.processor = found.pop()
return parsed
def all_processors(text: str) -> set:
"""Every CPU named anywhere in `text` (a snippet can name several)."""
found = set()
for rx in _PROCESSOR_RES:
for m in rx.finditer(text or ""):
cpu = parse_processor(m.group(0))
if cpu:
found.add(cpu)
return found

View File

@@ -0,0 +1,102 @@
"""Fill missing prices on search-only platforms (Amazon.in, Flipkart, Croma...)
from Google Programmable Search, without fetching those sites.
For each listing that has no price (or an unconfirmed one), search Google for
that product on that site. A price is taken only when:
* the result is the SAME product page (its site product id equals the
listing's), and
* Google's structured data for the page (pagemap offer / product:price meta)
states an INR price.
The listing is updated through the normal path, so the price is stored with
its evidence, appended to price_history, and outlier-checked.
"""
from __future__ import annotations
import logging
from typing import Callable, Dict, Optional
from app.electronics.collector import Collector, RunOptions, RunStats, source_sku
from app.electronics.db import repository as repo
from app.electronics.db.connection import connect
from app.electronics.normalise.title_parser import parse_title
from app.electronics.reference import load_reference, site_for_url
from app.electronics.search.engine import SearchEngine
logger = logging.getLogger(__name__)
def _listings_needing_price(limit: int, category: Optional[str]) -> list:
with connect() as conn:
return conn.execute(
"""
SELECT l.id, l.title, l.source_sku, l.source_url, s.domain, b.slug AS brand_slug, c.slug AS category,
(p.verification_status = 'verified') AS verified
FROM elec.source_listing l
JOIN elec.site s ON s.id = l.site_id
JOIN elec.brand b ON b.id = l.brand_id
JOIN elec.category c ON c.id = l.category_id
JOIN elec.product_listing_map m ON m.listing_id = l.id AND m.review_status IN ('auto','approved')
JOIN elec.product p ON p.id = m.product_id
WHERE l.source_type = 'search_snippet'
AND (l.price IS NULL OR l.price_outlier)
AND (s.policy = 'serp_only' OR coalesce(s.probe_outcome, 'C') = 'C')
AND (%(category)s::text IS NULL OR c.slug = %(category)s)
ORDER BY (p.verification_status = 'verified') DESC, l.last_seen_at DESC
LIMIT %(limit)s
""",
{"limit": limit, "category": category},
).fetchall()
def lookup_prices(limit: int = 40, category: Optional[str] = None,
progress: Callable[[str], None] = logger.info) -> Dict[str, object]:
stats: Dict[str, object] = {"checked": 0, "priced": 0, "no_same_page": 0, "no_structured_price": 0}
engine = SearchEngine(budget=limit)
if not engine.google.enabled:
stats["error"] = "Google Programmable Search is not configured (GOOGLE_API_KEY / GOOGLE_CSE_ID)"
return stats
ids = repo.id_maps()
ref = load_reference()
run_id = repo.start_run("price_lookup", {"limit": limit, "category": category})
try:
for row in _listings_needing_price(limit, category):
if not engine.google.enabled:
break
site = ref.sites[row["domain"]]
query = f"site:{row['domain']} {row['title'][:110]}"
hits = engine.text(query, max_results=10, providers="google")
stats["checked"] += 1
if hits is None:
continue
same = [h for h in hits
if (s := site_for_url(h.url)) is not None and s.domain == site.domain
and source_sku(site, h.url) == row["source_sku"]]
if not same:
stats["no_same_page"] += 1
continue
hit = next((h for h in same if h.offer), None)
if hit is None:
stats["no_structured_price"] += 1
continue
collector = Collector.__new__(Collector) # only its listing builder is used
collector.opt = RunOptions(category=row["category"], brands=[row["brand_slug"]])
collector.stats = RunStats()
parsed = parse_title(row["title"], row["category"], expected_brand=row["brand_slug"])
if parsed.brand is None:
continue
listing = collector.listing_from_search(hit, site, parsed, query)
listing.source_sku = row["source_sku"]
if listing.price is None:
stats["no_structured_price"] += 1
continue
repo.upsert_listing(listing, ids, run_id)
stats["priced"] += 1
progress(f"{site.name}: {row['title'][:70]} -> Rs {listing.price}")
stats["products"] = repo.refresh_verification()
if engine.google.error:
stats["error"] = engine.google.error
repo.finish_run(run_id, "done", {k: v for k, v in stats.items() if k != "products"})
except Exception as exc:
repo.finish_run(run_id, "failed", {}, repr(exc))
raise
return stats

View File

@@ -0,0 +1,99 @@
"""Decide, per site, whether it may be scraped or only searched.
A robots.txt allows product pages, HTTP 200 without a bot check, and the
page carries a schema.org Product with an INR offer -> scrape
B fetchable, product name/specs readable from the HTML, but no
structured price -> scrape specs/images,
price from search
C serp_only policy, robots.txt disallows, blocked / CAPTCHA, or the page
has no product data without JavaScript -> web search only
The probe looks at 2-3 real product URLs for the site, found through web
search, so it grades the pages the collector would actually fetch.
"""
from __future__ import annotations
import logging
from typing import Dict, List, Optional
from app.electronics.extract.html_fallback import embedded_state, extract_page
from app.electronics.extract.jsonld import extract_products
from app.electronics.net.polite_client import PoliteClient
from app.electronics.reference import SiteRef, load_reference, site_for_url
from app.electronics.search.engine import SearchEngine
logger = logging.getLogger(__name__)
def sample_product_urls(site: SiteRef, engine: SearchEngine, limit: int = 3) -> List[str]:
ref = load_reference()
urls: List[str] = []
if site.kind == "brand_official":
brand = ref.brands[site.brand_slug]
terms = [ref.categories[c].search_terms[0] for c in brand.categories]
queries = [f"site:{site.domain} {brand.name} {t}" for t in terms]
else:
queries = [f"site:{site.domain} samsung galaxy 5g", f"site:{site.domain} lenovo laptop"]
rx = site.product_url_re
for q in queries:
for hit in engine.text(q, max_results=15) or []:
s = site_for_url(hit.url)
if not s or s.domain != site.domain:
continue
if rx is not None and not rx.search(hit.url):
continue
if hit.url not in urls:
urls.append(hit.url)
if len(urls) >= limit:
return urls
return urls
def grade_page(html: str) -> Dict[str, object]:
products = extract_products(html)
priced = [p for p in products if p.get("price") is not None and (p.get("currency") in (None, "INR"))]
page = extract_page(html)
return {
"jsonld_products": len(products),
"jsonld_priced": len(priced),
"meta_price": page.get("price") is not None,
"has_title": bool(page.get("name")),
"spec_rows": len(page.get("properties") or {}),
"embedded_state": embedded_state(html) is not None,
}
def probe_site(site: SiteRef, client: PoliteClient, engine: SearchEngine) -> Dict[str, object]:
"""Returns {"outcome", "robots_allowed", "evidence"}; never raises."""
if site.policy == "serp_only":
return {"outcome": "C", "robots_allowed": None,
"evidence": {"reason": "policy serp_only: this site is never fetched directly"}}
urls = sample_product_urls(site, engine)
if not urls:
return {"outcome": "C", "robots_allowed": None,
"evidence": {"reason": "no product URLs found through web search"}}
pages: List[dict] = []
robots_any: Optional[bool] = None
for url in urls:
res = client.get(url)
entry = {"url": url, "status": res.status, "outcome": res.outcome}
robots_any = res.robots_allowed if robots_any is None else (robots_any or bool(res.robots_allowed))
if res.ok:
entry.update(grade_page(res.text))
pages.append(entry)
if res.outcome in ("captcha", "blocked", "breaker_open"):
break
ok_pages = [p for p in pages if p["outcome"] == "ok"]
if any(p["outcome"] in ("captcha", "blocked") for p in pages):
outcome, reason = "C", "blocked or bot check - not fetched again until the breaker cools down"
elif all(p["outcome"] == "robots_disallowed" for p in pages):
outcome, reason = "C", "robots.txt disallows product pages"
elif not ok_pages:
outcome, reason = "C", "product pages could not be fetched"
elif any(p.get("jsonld_priced") for p in ok_pages):
outcome, reason = "A", "schema.org Product with an INR offer"
elif any(p.get("has_title") and (p.get("spec_rows") or p.get("meta_price") or p.get("jsonld_products")) for p in ok_pages):
outcome, reason = "B", "product details readable from HTML; no structured price"
else:
outcome, reason = "C", "no product data without JavaScript"
return {"outcome": outcome, "robots_allowed": robots_any, "evidence": {"reason": reason, "pages": pages}}

View File

@@ -0,0 +1,132 @@
"""Reference data (brands, categories, sites, spec dictionary) loaded from YAML."""
from __future__ import annotations
import re
from dataclasses import dataclass, field
from functools import lru_cache
from pathlib import Path
from typing import Dict, List, Optional
import yaml
_DIR = Path(__file__).resolve().parent
def slugify(text: str) -> str:
return re.sub(r"[^a-z0-9]+", "-", text.lower()).strip("-")
@dataclass(frozen=True)
class BrandRef:
name: str
slug: str
categories: tuple
aliases: tuple
sub_brands: tuple
official: tuple
@dataclass(frozen=True)
class CategoryRef:
slug: str
name: str
search_terms: tuple
query_terms: tuple = ()
@dataclass(frozen=True)
class SiteRef:
domain: str
name: str
kind: str
region: str
policy: str
product_url: Optional[str] = None
pincode_param: Optional[str] = None
brand_slug: Optional[str] = None
@property
def product_url_re(self) -> Optional[re.Pattern]:
return re.compile(self.product_url) if self.product_url else None
@dataclass(frozen=True)
class Reference:
brands: Dict[str, BrandRef]
categories: Dict[str, CategoryRef]
sites: Dict[str, SiteRef]
spec_keys: Dict[str, dict] = field(default_factory=dict)
def brands_for(self, category: str) -> List[BrandRef]:
return [b for b in self.brands.values() if category in b.categories]
def _load_yaml(name: str) -> dict:
return yaml.safe_load((_DIR / name).read_text(encoding="utf-8")) or {}
@lru_cache(maxsize=1)
def load_reference() -> Reference:
raw_brands = _load_yaml("brands.yaml")
brands: Dict[str, BrandRef] = {}
for b in raw_brands.get("brands", []):
slug = slugify(b["name"])
brands[slug] = BrandRef(
name=b["name"],
slug=slug,
categories=tuple(b.get("categories", [])),
aliases=tuple(a.lower() for a in b.get("aliases", [])),
sub_brands=tuple(s.lower() for s in b.get("sub_brands", [])),
official=tuple(b.get("official", [])),
)
categories = {
c["slug"]: CategoryRef(c["slug"], c["name"], tuple(c.get("search_terms", [])),
tuple(c.get("query_terms", [])))
for c in raw_brands.get("categories", [])
}
sites: Dict[str, SiteRef] = {}
for s in _load_yaml("sites.yaml").get("sites", []):
sites[s["domain"]] = SiteRef(
domain=s["domain"],
name=s["name"],
kind=s["kind"],
region=s["region"],
policy=s["policy"],
product_url=s.get("product_url"),
pincode_param=s.get("pincode_param"),
)
# Every brand's official domains become sites of their own. They are
# probed like any retailer - an official page is the best evidence there is.
for b in brands.values():
for domain in b.official:
sites.setdefault(
domain,
SiteRef(
domain=domain,
name=f"{b.name} (official)",
kind="brand_official",
region="national",
policy="probe",
brand_slug=b.slug,
),
)
spec_keys = _load_yaml("spec_keys.yaml").get("categories", {})
return Reference(brands=brands, categories=categories, sites=sites, spec_keys=spec_keys)
def site_for_url(url: str) -> Optional[SiteRef]:
"""The registered site a URL belongs to (subdomains included), or None."""
from urllib.parse import urlparse
host = (urlparse(url).hostname or "").lower()
if not host:
return None
ref = load_reference()
best: Optional[SiteRef] = None
for domain, site in ref.sites.items():
if host == domain or host.endswith("." + domain):
if best is None or len(domain) > len(best.domain):
best = site
return best

View File

@@ -0,0 +1,100 @@
# Brand allow-list. A listing whose brand does not resolve to one of these is
# rejected - the catalogue is closed-world by design.
#
# aliases spellings seen on retail pages (matched case-insensitively,
# longest alias first, as a whole word at the start of a title)
# sub_brands product families sold under a parent brand. They resolve to the
# parent, and are kept as the product family.
# official the brand's own Indian web domains. A product page on one of
# these is the strongest evidence that a product exists.
brands:
- name: Samsung
categories: [mobiles, laptops]
aliases: [samsung]
official: [samsung.com]
- name: Apple
categories: [mobiles, laptops]
aliases: [apple]
sub_brands: [iphone, macbook]
official: [apple.com]
- name: Xiaomi
categories: [mobiles]
aliases: [xiaomi]
sub_brands: [redmi, poco, mi]
official: [mi.com]
- name: OnePlus
categories: [mobiles]
aliases: [oneplus, one plus]
official: [oneplus.in]
- name: Vivo
categories: [mobiles]
aliases: [vivo]
sub_brands: [iqoo]
official: [vivo.com, iqoo.com]
- name: Oppo
categories: [mobiles]
aliases: [oppo]
official: [oppo.com]
- name: Realme
categories: [mobiles]
aliases: [realme]
sub_brands: [narzo]
official: [realme.com]
- name: Motorola
categories: [mobiles]
aliases: [motorola, moto]
official: [motorola.co.in, motorola.com]
- name: Google
categories: [mobiles]
aliases: [google]
sub_brands: [pixel]
official: [store.google.com]
- name: Nothing
categories: [mobiles]
aliases: [nothing]
sub_brands: [cmf]
official: [nothing.tech]
- name: HP
categories: [laptops]
aliases: [hp, hewlett packard]
sub_brands: [omen, victus, pavilion, envy, spectre]
official: [hp.com]
- name: Dell
categories: [laptops]
aliases: [dell]
sub_brands: [alienware, inspiron, vostro, latitude, xps]
official: [dell.com]
- name: Lenovo
categories: [laptops]
aliases: [lenovo]
sub_brands: [thinkpad, ideapad, legion, yoga, thinkbook, loq]
official: [lenovo.com]
- name: Asus
categories: [laptops]
aliases: [asus]
sub_brands: [rog, tuf, vivobook, zenbook]
official: [asus.com]
- name: Acer
categories: [laptops]
aliases: [acer]
sub_brands: [aspire, nitro, predator, swift]
official: [acer.com]
- name: MSI
categories: [laptops]
aliases: [msi]
official: [msi.com]
# search_terms: the category word used when probing brand sites.
# query_terms: appended to `site:<platform> <brand>` during discovery. They
# read like the variant part of a product title, which is what
# makes search engines return single product pages rather than
# category or blog pages.
categories:
- slug: mobiles
name: Mobiles
search_terms: [smartphone, mobile phone]
query_terms: ["5G 8GB RAM 128GB", "5G 8GB 256GB", "12GB RAM 256GB"]
- slug: laptops
name: Laptops
search_terms: [laptop]
query_terms: ["laptop 16GB RAM 512GB SSD", "laptop 8GB RAM 512GB SSD"]

View File

@@ -0,0 +1,77 @@
# Retail platforms.
#
# kind marketplace | national_chain | tn_regional
# region national | TN (TN = a Tamil Nadu retail chain)
# policy serp_only -> NEVER fetched directly; everything comes from web
# search results (titles, snippets, image results)
# probe -> fetched only if the site probe grades it A or B
# (robots.txt allows, HTTP 200, no CAPTCHA); otherwise
# it falls back to search results like serp_only
# product_url regex a URL must match to count as a single product page.
# Group 1, when present, is the site's own product id.
# pincode_param optional query parameter the site accepts for a delivery
# pincode. Only sites that actually honour it get
# pincode_applied=true on their prices.
#
# Brand official sites are generated from brands.yaml (kind brand_official).
sites:
- domain: amazon.in
name: Amazon.in
kind: marketplace
region: national
policy: serp_only
product_url: '/(?:dp|gp/product)/([A-Z0-9]{10})'
- domain: flipkart.com
name: Flipkart
kind: marketplace
region: national
policy: serp_only
product_url: '/p/(itm[0-9a-z]+)'
- domain: croma.com
name: Croma
kind: national_chain
region: national
policy: probe
product_url: '/p/(\d{5,})'
- domain: reliancedigital.in
name: Reliance Digital
kind: national_chain
region: national
policy: probe
product_url: '(?:/p/|/product/[^?#]*?-)(\d{6,})'
- domain: vijaysales.com
name: Vijay Sales
kind: national_chain
region: national
policy: probe
product_url: '/p/(?:P?)(\d{3,})/'
- domain: tatacliq.com
name: Tata CLiQ
kind: marketplace
region: national
policy: probe
product_url: '/p-(mp\d+)'
- domain: poorvika.com
name: Poorvika
kind: tn_regional
region: TN
policy: probe
product_url: '/([a-z0-9-]{8,})/p/?$'
- domain: sangeethamobiles.com
name: Sangeetha Mobiles
kind: tn_regional
region: TN
policy: probe
product_url: '(?i)/product-?details/(?:[^/?#]+/)?(\d+)'
- domain: vasanthandco.in
name: Vasanth & Co
kind: tn_regional
region: TN
policy: probe
product_url: '/(?:product|products)/([a-z0-9-]{8,})'
- domain: viveks.com
name: Viveks
kind: tn_regional
region: TN
policy: probe
product_url: '/([a-z0-9-]{8,})\.html$'

View File

@@ -0,0 +1,127 @@
# Canonical specification keys per category.
#
# type number | text | enum
# unit canonical unit for numbers (values are converted into it)
# synonyms spec labels seen on retail/brand pages (case-insensitive,
# punctuation ignored). A label maps to the first key that lists it.
# range plausible [min, max] after conversion; values outside are dropped
# values allowed canonical values for enums, each with its match words
#
# A value is only ever stored if it was read from a page or snippet. Nothing
# here supplies a default.
categories:
mobiles:
ram_gb:
type: number
unit: GB
range: [1, 32]
synonyms: [ram, memory ram, ram size, ram capacity, installed ram, system memory]
storage_gb:
type: number
unit: GB
range: [8, 2048]
synonyms: [internal storage, storage, rom, internal memory, storage capacity, inbuilt memory, memory storage capacity]
display_inch:
type: number
unit: inch
range: [3, 9]
synonyms: [display size, screen size, display, screen size inches, standing screen display size]
display_type:
type: enum
synonyms: [display type, screen type, display technology, panel type]
values:
AMOLED: [amoled, super amoled, dynamic amoled, pole amoled, fluid amoled]
OLED: [oled, super retina, ltpo oled]
LCD: [lcd, ips lcd, tft, ips]
refresh_hz:
type: number
unit: Hz
range: [30, 240]
synonyms: [refresh rate, screen refresh rate, display refresh rate]
processor:
type: text
synonyms: [processor, chipset, processor name, soc, cpu, processor brand]
rear_camera_mp:
type: number
unit: MP
range: [2, 250]
synonyms: [rear camera, primary camera, main camera, back camera, rear camera resolution, primary camera resolution]
front_camera_mp:
type: number
unit: MP
range: [2, 60]
synonyms: [front camera, secondary camera, selfie camera, front camera resolution]
battery_mah:
type: number
unit: mAh
range: [1000, 10000]
synonyms: [battery capacity, battery, battery power, battery capacity mah]
os:
type: enum
synonyms: [operating system, os, os version]
values:
Android: [android]
iOS: [ios]
network:
type: enum
synonyms: [network type, network, cellular technology, connectivity technology, network connectivity]
values:
5G: [5g]
4G: [4g, lte]
colour:
type: text
synonyms: [colour, color, colour name, color name]
laptops:
processor:
type: text
synonyms: [processor, processor name, cpu, processor model, processor type]
ram_gb:
type: number
unit: GB
range: [2, 128]
synonyms: [ram, ram size, memory, system memory, installed ram, ram capacity]
storage_gb:
type: number
unit: GB
range: [32, 8192]
synonyms: [ssd capacity, storage, hard disk size, hard drive size, storage capacity, ssd, internal storage]
storage_type:
type: enum
synonyms: [storage type, hard disk type, hard drive interface, drive type]
values:
SSD: [ssd, nvme, solid state]
HDD: [hdd, hard disk drive]
eMMC: [emmc]
display_inch:
type: number
unit: inch
range: [10, 19]
synonyms: [screen size, display size, standing screen display size, display]
resolution:
type: text
synonyms: [resolution, screen resolution, display resolution, maximum display resolution]
gpu:
type: text
synonyms: [graphics, graphics processor, gpu, graphic processor, graphics coprocessor, graphics card]
os:
type: enum
synonyms: [operating system, os]
values:
Windows: [windows]
macOS: [macos, mac os]
ChromeOS: [chrome os, chromeos]
Linux: [linux, ubuntu]
DOS: [dos, free dos, freedos]
weight_kg:
type: number
unit: kg
range: [0.5, 5]
synonyms: [weight, item weight, product weight, laptop weight]
battery_wh:
type: number
unit: Wh
range: [20, 120]
synonyms: [battery capacity, battery, battery power]
colour:
type: text
synonyms: [colour, color]

View File

@@ -0,0 +1,96 @@
"""Which real customer reviews to show for a product, and in what mix.
Every review passed in here was read from a product page's own schema.org
data (see extract/jsonld.py); this module only classifies and selects - it
never writes, rewrites or summarises review text.
Sentiment is the reviewer's own star rating, nothing inferred:
>= 4 positive, >= 3 neutral, < 3 negative.
The mix follows the product's overall rating, so the reviews shown read like
the rating does:
rating >= 4.0 mostly positive, some neutral, a little negative
3.0 < rating < 4.0 mostly neutral, some positive, a little negative
rating <= 3.0 mostly negative, a little positive and neutral
When a group has too few reviews its slots go to the other groups, in the
same priority order. Nothing is ever padded: if only 3 real reviews exist,
3 are shown.
"""
from __future__ import annotations
from decimal import Decimal
from typing import Any, Dict, List, Optional, Sequence, Tuple
POSITIVE, NEUTRAL, NEGATIVE = "positive", "neutral", "negative"
MAX_REVIEWS = 10
# (group, share of MAX_REVIEWS), highest priority first.
_MIX_HIGH: Tuple[Tuple[str, int], ...] = ((POSITIVE, 6), (NEUTRAL, 3), (NEGATIVE, 1))
_MIX_MID: Tuple[Tuple[str, int], ...] = ((NEUTRAL, 5), (POSITIVE, 3), (NEGATIVE, 2))
_MIX_LOW: Tuple[Tuple[str, int], ...] = ((NEGATIVE, 6), (POSITIVE, 2), (NEUTRAL, 2))
def sentiment_for(rating: Any) -> Optional[str]:
"""The group a reviewer's own star rating puts a review in; None when the
review states no rating."""
if rating is None:
return None
try:
value = Decimal(str(rating))
except Exception: # noqa: BLE001
return None
if value >= 4:
return POSITIVE
if value >= 3:
return NEUTRAL
return NEGATIVE
def mix_for(product_rating: Any) -> Tuple[Tuple[str, int], ...]:
if product_rating is None:
return _MIX_MID # no overall rating stated: a balanced view
value = Decimal(str(product_rating))
if value >= 4:
return _MIX_HIGH
if value > 3:
return _MIX_MID
return _MIX_LOW
def _rank_key(review: Dict[str, Any]) -> tuple:
# Newest first (ISO dates sort as text), then the more substantial review.
return (str(review.get("review_date") or ""), len(review.get("body") or ""))
def select_reviews(product_rating: Any, reviews: Sequence[Dict[str, Any]],
max_n: int = MAX_REVIEWS) -> List[Dict[str, Any]]:
"""Up to `max_n` of `reviews`, mixed by sentiment as described above.
Reviews without a star rating have no sentiment and are not shown: there
is no honest way to place them in the mix.
"""
groups: Dict[str, List[Dict[str, Any]]] = {POSITIVE: [], NEUTRAL: [], NEGATIVE: []}
seen = set()
for r in reviews:
s = r.get("sentiment") or sentiment_for(r.get("rating"))
key = (r.get("body") or "").strip().lower()
if s is None or not key or key in seen:
continue
seen.add(key)
groups[s].append({**r, "sentiment": s})
for g in groups.values():
g.sort(key=_rank_key, reverse=True)
mix = mix_for(product_rating)
scale = max_n / MAX_REVIEWS
quota = {g: int(round(n * scale)) for g, n in mix}
picked: Dict[str, List[Dict[str, Any]]] = {g: groups[g][: quota[g]] for g, _ in mix}
# Hand unused slots to the other groups, in priority order.
spare = max_n - sum(len(v) for v in picked.values())
for g, _ in mix:
if spare <= 0:
break
extra = groups[g][len(picked[g]): len(picked[g]) + spare]
picked[g].extend(extra)
spare -= len(extra)
return [r for g, _ in mix for r in picked[g]]

View File

@@ -0,0 +1,65 @@
"""Cached, budgeted access to the search providers.
Results are cached in elec.search_cache so a re-run does not query again
within SEARCH_CACHE_TTL_HOURS, and each run has a query budget so a large
brand list cannot hammer the providers.
"""
from __future__ import annotations
import logging
from typing import Dict, List, Optional
from app.electronics.db import repository as repo
from app.electronics.search.providers import DuckDuckGoProvider, GoogleCseProvider, SearchHit
from app.infrastructure.settings import GOOGLE_CSE_DAILY_QUOTA, SEARCH_CACHE_TTL_HOURS
logger = logging.getLogger(__name__)
class SearchEngine:
def __init__(self, *, budget: int = 200, use_cache: bool = True) -> None:
self.budget = budget
self.use_cache = use_cache
self.used = 0
self.stats: Dict[str, int] = {"cache_hits": 0, "queries": 0, "unavailable": 0}
self.ddg = DuckDuckGoProvider()
self.google = GoogleCseProvider(
quota_left=lambda: GOOGLE_CSE_DAILY_QUOTA - repo.google_queries_today()
)
def _ask(self, provider, kind: str, query: str, max_results: int) -> Optional[List[SearchHit]]:
if not provider.enabled:
return None
cached = repo.search_cache_get(provider.name, kind, query, SEARCH_CACHE_TTL_HOURS) if self.use_cache else None
if cached is not None:
self.stats["cache_hits"] += 1
return [SearchHit.from_dict(d) for d in cached]
if self.used >= self.budget:
logger.info("Search budget (%d) spent; skipping %r", self.budget, query)
return None
self.used += 1
self.stats["queries"] += 1
self.stats[f"queries_{provider.name}"] = self.stats.get(f"queries_{provider.name}", 0) + 1
hits = provider.text(query, max_results) if kind == "text" else provider.images(query, max_results)
if hits is None:
self.stats["unavailable"] += 1
return None
repo.search_cache_put(provider.name, kind, query, [h.to_dict() for h in hits])
return hits
def _run(self, kind: str, query: str, max_results: int, providers: str) -> Optional[List[SearchHit]]:
"""providers: "default" = DuckDuckGo, with Google only when DuckDuckGo
gives no answer (keeps the 100/day Google quota for price lookups);
"google" = Google only."""
if providers == "google":
return self._ask(self.google, kind, query, max_results)
hits = self._ask(self.ddg, kind, query, max_results)
if hits is None:
hits = self._ask(self.google, kind, query, max_results)
return hits
def text(self, query: str, max_results: int = 20, *, providers: str = "default") -> Optional[List[SearchHit]]:
return self._run("text", query, max_results, providers)
def images(self, query: str, max_results: int = 15) -> Optional[List[SearchHit]]:
return self._run("images", query, max_results, "default")

View File

@@ -0,0 +1,234 @@
"""Web search: DuckDuckGo (ddgs, no key) and, when configured, Google
Programmable Search.
Three outcomes, never collapsed: a list of hits, an empty list ("we asked and
nothing matched"), or None ("we could not ask" - throttled, offline, no
quota). A throttle is not evidence that a product is not sold anywhere.
"""
from __future__ import annotations
import logging
import threading
import time
from dataclasses import asdict, dataclass
from typing import Callable, List, Optional
import requests
from app.infrastructure.settings import (
GOOGLE_API_KEY,
GOOGLE_CSE_ID,
SEARCH_MIN_INTERVAL_SECONDS,
SEARCH_REGION,
USE_DDG_SEARCH,
USE_GOOGLE_CSE,
)
logger = logging.getLogger(__name__)
@dataclass
class SearchHit:
url: str
title: str
snippet: str
provider: str
rank: int
image_url: Optional[str] = None # image searches: the image itself (url = page it is on)
# Structured offer data the search engine itself extracted from the page
# (Google CSE "pagemap"): {"price", "currency", "availability", "raw"}.
offer: Optional[dict] = None
# Aggregate rating the search engine extracted from the page's own
# structured data (Google CSE "pagemap"): {"rating", "review_count", "raw"}.
rating: Optional[dict] = None
def to_dict(self) -> dict:
return asdict(self)
@classmethod
def from_dict(cls, d: dict) -> "SearchHit":
return cls(**{k: d.get(k) for k in ("url", "title", "snippet", "provider", "rank", "image_url", "offer", "rating")})
class _Pacer:
def __init__(self, interval: float, sleep: Callable[[float], None] = time.sleep) -> None:
self.interval = interval
self._sleep = sleep
self._last = 0.0
self._lock = threading.Lock()
def wait(self) -> None:
with self._lock:
gap = self.interval - (time.monotonic() - self._last)
if gap > 0:
self._sleep(gap)
self._last = time.monotonic()
class DuckDuckGoProvider:
name = "ddg"
def __init__(self, interval: float = SEARCH_MIN_INTERVAL_SECONDS) -> None:
self._pacer = _Pacer(interval)
self.enabled = USE_DDG_SEARCH
# "auto" rotates ddgs's engines; yahoo is a second opinion when it is throttled.
BACKENDS = ("auto", "yahoo")
def text(self, query: str, max_results: int = 20) -> Optional[List[SearchHit]]:
"""Hits, or None when no backend answered. ddgs reports a throttle and
a genuinely empty result the same way ("No results found"), so an
empty answer is treated as unknown rather than as "not listed"."""
if not self.enabled:
return None
try:
from ddgs import DDGS
except ImportError:
logger.warning("ddgs is not installed; DuckDuckGo search unavailable")
return None
for backend in self.BACKENDS:
self._pacer.wait()
try:
with DDGS(timeout=20) as ddgs:
rows = list(ddgs.text(query, region=SEARCH_REGION, safesearch="moderate",
max_results=max_results, backend=backend) or [])
except Exception as exc: # noqa: BLE001 - ddgs raises many types on throttling
logger.info("DuckDuckGo(%s) text search gave no answer (%s): %s", backend, query, exc)
continue
hits = [
SearchHit(r.get("href") or r.get("url") or "", r.get("title") or "", r.get("body") or "",
f"{self.name}", i)
for i, r in enumerate(rows)
if (r.get("href") or r.get("url") or "").startswith("http")
]
if hits:
return hits
return None
def images(self, query: str, max_results: int = 15) -> Optional[List[SearchHit]]:
if not self.enabled:
return None
try:
from ddgs import DDGS
except ImportError:
return None
self._pacer.wait()
try:
with DDGS(timeout=20) as ddgs:
rows = list(ddgs.images(query, region=SEARCH_REGION, safesearch="moderate",
max_results=max_results) or [])
except Exception as exc: # noqa: BLE001
if "no results" in str(exc).lower():
return []
logger.info("DuckDuckGo image search failed (%s): %s", query, exc)
return None
return [
SearchHit(r.get("url") or "", r.get("title") or "", "", self.name, i, image_url=r.get("image"))
for i, r in enumerate(rows)
if str(r.get("image") or "").startswith("http") and str(r.get("url") or "").startswith("http")
]
class GoogleCseProvider:
name = "google"
ENDPOINT = "https://www.googleapis.com/customsearch/v1"
def __init__(self, quota_left: Callable[[], int] = lambda: 100) -> None:
self.enabled = USE_GOOGLE_CSE
self.error: Optional[str] = None
self._quota_left = quota_left
self._pacer = _Pacer(1.0)
def _call(self, query: str, extra: dict) -> Optional[List[dict]]:
if not self.enabled:
return None
if self._quota_left() <= 0:
self.error = "daily query quota used up"
return None
self._pacer.wait()
try:
resp = requests.get(
self.ENDPOINT,
params={"key": GOOGLE_API_KEY, "cx": GOOGLE_CSE_ID, "q": query, "gl": "in", "num": 10, **extra},
timeout=20,
)
except requests.RequestException as exc:
logger.info("Google CSE failed: %s", exc)
return None
if resp.status_code in (400, 401, 403):
# A key/project problem will not fix itself mid-run: stop asking.
try:
message = resp.json().get("error", {}).get("message", "")
except ValueError:
message = resp.text[:200]
self.enabled = False
self.error = f"HTTP {resp.status_code}: {message}"
logger.warning("Google Programmable Search disabled for this run - %s", self.error)
return None
if resp.status_code != 200:
logger.info("Google CSE HTTP %s: %s", resp.status_code, resp.text[:200])
return None
return resp.json().get("items", []) or []
def text(self, query: str, max_results: int = 10) -> Optional[List[SearchHit]]:
items = self._call(query, {})
if items is None:
return None
return [SearchHit(i.get("link", ""), i.get("title", ""), i.get("snippet", ""), self.name, n,
offer=pagemap_offer(i.get("pagemap") or {}),
rating=pagemap_rating(i.get("pagemap") or {}))
for n, i in enumerate(items[:max_results]) if i.get("link")]
def images(self, query: str, max_results: int = 10) -> Optional[List[SearchHit]]:
items = self._call(query, {"searchType": "image"})
if items is None:
return None
return [SearchHit((i.get("image") or {}).get("contextLink", ""), i.get("title", ""), "", self.name, n,
image_url=i.get("link"))
for n, i in enumerate(items[:max_results]) if i.get("link")]
def pagemap_offer(pagemap: dict) -> Optional[dict]:
"""The offer Google extracted from the page's own structured data
(schema.org Offer, or product:price meta tags), if any. INR only."""
candidates = []
for offer in pagemap.get("offer") or []:
candidates.append((offer.get("price"), offer.get("pricecurrency"), offer.get("availability"), offer))
for meta in pagemap.get("metatags") or []:
price = meta.get("product:price:amount") or meta.get("og:price:amount")
if price:
candidates.append((price, meta.get("product:price:currency") or meta.get("og:price:currency"),
meta.get("product:availability") or meta.get("og:availability"),
{k: v for k, v in meta.items() if "price" in k or "availability" in k}))
for price, currency, availability, raw in candidates:
if price and (currency or "").upper() == "INR":
return {"price": str(price), "currency": "INR", "availability": availability, "raw": raw}
return None
def pagemap_rating(pagemap: dict) -> Optional[dict]:
"""The aggregate rating Google extracted from the page's own structured
data (schema.org AggregateRating), if any. Only a value on a 5-point
scale is accepted."""
for node in pagemap.get("aggregaterating") or []:
try:
value = float(str(node.get("ratingvalue", "")).replace(",", "."))
except ValueError:
continue
best = node.get("bestrating")
try:
if best not in (None, "") and float(best) != 5:
continue
except ValueError:
continue
if not 0 < value <= 5:
continue
count = None
for key in ("reviewcount", "ratingcount"):
digits = "".join(ch for ch in str(node.get(key) or "") if ch.isdigit())
if digits:
count = int(digits)
break
return {"rating": round(value, 2), "review_count": count,
"raw": {k: v for k, v in node.items() if k in ("ratingvalue", "reviewcount", "ratingcount", "bestrating")}}
return None

View File

View File

@@ -0,0 +1,427 @@
"""
Password hashing, access-token issuance/verification, and the Principal that
represents an authenticated caller.
Two kinds of credential reach this module:
* Interactive users. ``POST /api/auth/login`` exchanges a username/password
for a short-lived signed JWT. No password is ever stored - only a PBKDF2
digest, read from the environment (``AUTH_ADMIN_PASSWORD_HASH`` /
``AUTH_USER_PASSWORD_HASH``). Generate those with
``python scripts/make_auth_secrets.py``.
* Machine consumers. A static key sent as ``X-API-Key``, mapped to a role by
``API_KEYS``. These do not expire, so treat one as a long-lived secret and
give each consumer its own so it can be revoked individually.
PBKDF2-HMAC-SHA256 is used rather than bcrypt or argon2 deliberately: it is in
the standard library, so the slim Python image needs no compiled dependency,
and at the iteration count below it meets OWASP's current guidance. The
encoded form carries its own iteration count, so raising the constant later
does not invalidate hashes already issued.
"""
from __future__ import annotations
import base64
import hashlib
import hmac
import logging
import secrets
import time
from dataclasses import dataclass, field
from typing import Dict, List, Optional, Tuple
import jwt
from app.infrastructure.settings import (
API_KEYS,
AUTH_ADMIN_PASSWORD_HASH,
AUTH_ADMIN_USERNAME,
AUTH_ALLOW_ANY_LOGIN,
AUTH_ENABLED,
AUTH_SECRET_KEY,
AUTH_TOKEN_TTL_MINUTES,
config_source,
)
logger = logging.getLogger(__name__)
# ---------------------------------------------------------------------------
# Roles and permissions
# ---------------------------------------------------------------------------
# These mirror the permission strings the React UI already keys its navigation
# off, so the server now enforces the same vocabulary the client was only
# displaying. `admin` is a superuser: has_permission() grants it everything
# rather than requiring every new permission to be added to this list.
ROLE_PERMISSIONS: Dict[str, List[str]] = {
"admin": [
"view_catalog",
"view_project_details",
"upload_train_test",
"allocate_discounts",
"manage_analytics",
"manage_nutrition",
],
"user": [
"add_product",
"upload_batch_products",
"update_db_and_json",
"fetch_images",
"upload_store_inventory",
"view_store_analytics",
"view_nutrition_insights",
"optimize_profits",
],
# An outside API client that may send spreadsheets for catalog ingestion and
# do NOTHING else. One permission, deliberately.
#
# This role exists because API keys carry no per-key scoping:
# principal_for_api_key() derives permissions entirely from the role, so
# "upload-only" can only be expressed as a role. Reusing `user` would have
# been less code and would also have handed an outside contributor
# add_product, upload_batch_products and upload_store_inventory - real
# write access to the catalog - to solve a problem that needed one verb.
#
# WHAT A LEAKED UPLOADER KEY COSTS. Real CPU: this permission starts the
# 11-stage pipeline, which is the point of the endpoint. The bound is not
# "this role cannot work" but "all ingestion, from every source, shares one
# worker" - batch_worker runs a single batch at a time behind a queue of
# BATCH_QUEUE_MAX, past which POST /api/uploads/catalog answers 429. So a
# key can occupy the ingestion worker; it cannot multiply it, and it cannot
# touch the request path the healthcheck reads.
#
# What it still cannot do: read the catalog, read another caller's
# submissions (every read on that router is filtered by submitted_by), or
# cancel, resume or delete anything.
"uploader": [
"upload_catalog",
],
}
VALID_ROLES = frozenset(ROLE_PERMISSIONS)
JWT_ALGORITHM = "HS256"
JWT_ISSUER = "brand-catalog-rag"
# OWASP's floor for PBKDF2-HMAC-SHA256 at time of writing.
_PBKDF2_ITERATIONS = 600_000
_PBKDF2_PREFIX = "pbkdf2_sha256"
@dataclass(frozen=True)
class Principal:
"""Whoever is making the current request, once their credential checks out."""
username: str
role: str
permissions: List[str] = field(default_factory=list)
# "user" - logged in via /api/auth/login, carrying a JWT
# "api_key" - a machine consumer from API_KEYS
# "anonymous" - AUTH_ENABLED=false; no credential was checked at all
kind: str = "user"
def has_permission(self, permission: str) -> bool:
return self.role == "admin" or permission in self.permissions
class AuthError(Exception):
"""A credential was absent, malformed, expired, or simply wrong."""
# ---------------------------------------------------------------------------
# Password hashing
# ---------------------------------------------------------------------------
def hash_password(password: str, *, iterations: int = _PBKDF2_ITERATIONS) -> str:
"""Return an encoded digest: ``pbkdf2_sha256$<iterations>$<salt>$<hash>``."""
salt = secrets.token_bytes(16)
digest = hashlib.pbkdf2_hmac("sha256", password.encode("utf-8"), salt, iterations)
return "$".join(
(
_PBKDF2_PREFIX,
str(iterations),
base64.b64encode(salt).decode("ascii"),
base64.b64encode(digest).decode("ascii"),
)
)
def _parse_encoded_hash(encoded: str) -> Optional[Tuple[bytes, bytes, int]]:
"""
Split an encoded digest into ``(salt, digest, iterations)``, or None if it
is not one.
One parser, three callers. `verify_password` needs the parts, while
`hash_is_wellformed` and `describe_password_hash` need only the verdict -
and a login failing because the *configured* hash is corrupt is a different
incident from a wrong password, so the two must agree on what "corrupt"
means. Two copies of this parse would eventually disagree.
Values arrive here straight from the environment, so a hash pasted into a
deployment platform's form field as "pbkdf2_sha256$..." is unwrapped rather
than rejected: the surrounding quotes are almost never intended as part of
the secret, and the failure they cause otherwise is a silent 401.
"""
if not encoded:
return None
encoded = encoded.strip().strip("'\"")
try:
prefix, raw_iterations, raw_salt, raw_digest = encoded.split("$")
if prefix != _PBKDF2_PREFIX:
return None
# validate=True so junk is rejected rather than silently discarded:
# b64decode's default drops non-alphabet characters, which would let a
# subtly corrupted hash decode to the wrong bytes and fail as a "wrong
# password" instead of as the configuration error it is.
# binascii.Error subclasses ValueError, so it is caught below.
salt = base64.b64decode(raw_salt, validate=True)
digest = base64.b64decode(raw_digest, validate=True)
iterations = int(raw_iterations)
except (ValueError, TypeError):
return None
# A structurally valid string that decodes to nothing is still unusable,
# and PBKDF2 rejects a non-positive iteration count by raising.
if not salt or not digest or iterations < 1:
return None
return salt, digest, iterations
def hash_is_wellformed(encoded: str) -> bool:
"""Whether a configured digest can be checked against at all.
Distinct from "does the password match": this asks whether the credential
*store* is usable, which is a deployment fault rather than a sign-in one.
"""
return _parse_encoded_hash(encoded) is not None
def password_hash_fingerprint(encoded: str) -> str:
"""
A short, non-reversible identifier for a configured digest.
Safe to log and to publish: it is a truncated SHA-256 of the *encoded
digest*, and that digest already embeds a 16-byte random salt, so this says
which credential is loaded without saying anything about the password
behind it. It exists so a running deployment can be compared against the
config it was supposed to have been built from - the failure this project
actually hit - without moving a secret in order to do the comparison.
"""
if not encoded:
return ""
return hashlib.sha256(encoded.strip().strip("'\"").encode("utf-8")).hexdigest()[:12]
def api_key_fingerprint(name: str, secret: str) -> str:
"""
A short, non-reversible identifier for a configured API key.
Same purpose as password_hash_fingerprint - say *which* credential is loaded
without moving the credential - but the safety argument is different and
worth stating. That function digests an encoded hash which already embeds a
16-byte random salt. An API key has no salt, so the name is mixed in here to
keep two consumers that were mistakenly issued the same secret from
fingerprinting identically, and settings._parse_api_keys enforces a minimum
secret length so the digest cannot be walked back with a wordlist.
"""
if not secret:
return ""
cleaned = secret.strip().strip("'\"")
material = f"{name}:{cleaned}"
return hashlib.sha256(material.encode("utf-8")).hexdigest()[:12]
def describe_api_keys() -> List[Dict[str, object]]:
"""Every configured key as {name, role, fingerprint}, sorted by name.
Sorted so two deployments' /api/health output can be diffed line for line;
API_KEYS is keyed by secret, whose iteration order says nothing useful.
"""
return sorted(
(
{"name": name, "role": role, "fingerprint": api_key_fingerprint(name, secret)}
for secret, (name, role) in API_KEYS.items()
),
key=lambda entry: entry["name"],
)
def describe_password_hash(encoded: str) -> Dict[str, object]:
"""A loggable/publishable summary of a configured digest. Never its bytes."""
parsed = _parse_encoded_hash(encoded)
return {
"valid": parsed is not None,
"algorithm": _PBKDF2_PREFIX if parsed is not None else None,
"iterations": parsed[2] if parsed is not None else None,
"fingerprint": password_hash_fingerprint(encoded),
}
def auth_config_summary() -> Dict[str, object]:
"""
The effective authentication configuration, in a form safe to both log and
publish. Contains no password and no hash - only the fingerprint.
This is deliberately one function with two callers (the startup log in
app/main.py and GET /api/health), because its entire purpose is letting two
*deployments* be compared, and that only works if both report the same
fields computed the same way.
`*_source` is the field that earns this its keep. A value of "process-env"
means the container's own environment supplied it and the .env file baked
into the image was ignored - which is invisible from anywhere else, and is
precisely how a corrected credential can keep failing after a redeploy.
The same argument is why the API keys are summarised here. backend/Dockerfile
copies .env.production in at BUILD time, so a key added to that file and then
merely restarted is not present in the running process - and from outside,
an undeployed key is indistinguishable from a wrong one, because both are
just a 401. Publishing the names and fingerprints answers "is my key on this
deployment?" without anyone having to send the secret to find out.
"""
described = describe_password_hash(AUTH_ADMIN_PASSWORD_HASH)
return {
"enabled": AUTH_ENABLED,
"allow_any_login": AUTH_ALLOW_ANY_LOGIN,
"admin_username": AUTH_ADMIN_USERNAME,
"password_hash_valid": bool(described["valid"]),
"password_hash_iterations": described["iterations"],
"password_hash_fingerprint": described["fingerprint"],
"admin_username_source": config_source("AUTH_ADMIN_USERNAME"),
"password_hash_source": config_source("AUTH_ADMIN_PASSWORD_HASH"),
"api_keys_count": len(API_KEYS),
"api_keys": describe_api_keys(),
"api_keys_source": config_source("API_KEYS"),
}
def verify_password(password: str, encoded: str) -> bool:
"""
Check a password against an encoded digest.
Returns False rather than raising on a malformed digest: a typo in
AUTH_ADMIN_PASSWORD_HASH must fail the login, not 500 the endpoint and
hand the caller a stack trace describing the credential store.
"""
if not encoded:
return False
parsed = _parse_encoded_hash(encoded)
if parsed is None:
logger.error(
"A configured password hash is malformed and cannot be used. Regenerate "
"it with: python scripts/make_auth_secrets.py"
)
return False
salt, digest, iterations = parsed
candidate = hashlib.pbkdf2_hmac("sha256", password.encode("utf-8"), salt, iterations)
return hmac.compare_digest(candidate, digest)
# ---------------------------------------------------------------------------
# Access tokens
# ---------------------------------------------------------------------------
def create_access_token(
username: str,
role: str,
permissions: List[str],
*,
ttl_minutes: Optional[int] = None,
) -> tuple[str, int]:
"""Issue a signed JWT. Returns ``(token, expires_in_seconds)``."""
ttl = (ttl_minutes if ttl_minutes is not None else AUTH_TOKEN_TTL_MINUTES) * 60
now = int(time.time())
payload = {
"sub": username,
"role": role,
"perms": permissions,
"iss": JWT_ISSUER,
"iat": now,
"exp": now + ttl,
}
return jwt.encode(payload, AUTH_SECRET_KEY, algorithm=JWT_ALGORITHM), ttl
def decode_access_token(token: str) -> Principal:
"""
Verify a JWT and return the Principal it names.
The algorithm is pinned to a single-item allow-list rather than read from
the token header. That is what closes the two classic JWT bypasses: a token
presenting ``alg: none``, and one presenting ``alg: HS256`` against a key
the server intended to use asymmetrically.
"""
try:
payload = jwt.decode(
token,
AUTH_SECRET_KEY,
algorithms=[JWT_ALGORITHM],
issuer=JWT_ISSUER,
options={"require": ["exp", "iat", "sub"]},
)
except jwt.ExpiredSignatureError as exc:
raise AuthError("Token has expired. Sign in again.") from exc
except jwt.InvalidTokenError as exc:
raise AuthError("Invalid authentication token.") from exc
role = payload.get("role")
if role not in VALID_ROLES:
raise AuthError("Token names an unknown role.")
perms = payload.get("perms")
return Principal(
username=str(payload["sub"]),
role=role,
# Fall back to the role's current grants if the token predates a
# permission change, rather than trusting an arbitrary claim shape.
permissions=list(perms) if isinstance(perms, list) else ROLE_PERMISSIONS.get(role, []),
kind="user",
)
# ---------------------------------------------------------------------------
# API keys (machine consumers)
# ---------------------------------------------------------------------------
def principal_for_api_key(presented: str) -> Principal:
"""
Resolve an ``X-API-Key`` value to a Principal.
Every configured key is compared even after a match, using compare_digest,
so the time taken does not reveal how far down the list a near-miss got.
"""
matched: Optional[tuple[str, str]] = None
for secret, (name, role) in API_KEYS.items():
if hmac.compare_digest(presented, secret):
matched = (name, role)
if matched is None:
raise AuthError("Invalid API key.")
name, role = matched
return Principal(
username=name,
role=role,
permissions=ROLE_PERMISSIONS.get(role, []),
kind="api_key",
)
def anonymous_principal() -> Principal:
"""
The stand-in used when ``AUTH_ENABLED=false``.
It is deliberately an admin: disabling auth is meant to make local
development frictionless, and a half-privileged anonymous caller would
produce confusing 403s instead. Nothing calls this when auth is on.
"""
return Principal(
username="anonymous",
role="admin",
permissions=ROLE_PERMISSIONS["admin"],
kind="anonymous",
)
if not AUTH_ENABLED:
logger.warning(
"AUTH_ENABLED=false: every endpoint is unauthenticated, including catalog "
"generation, ML training, and the upload endpoints. This is for local "
"development only - never run it on a host reachable from the internet."
)

View File

@@ -0,0 +1,373 @@
"""
Centralized configuration for the Electronics Catalog backend.
Every credential is read ONLY from the environment (backend/.env via
python-dotenv, or real OS variables). Non-secret values keep safe local
defaults.
LOCAL-ONLY GUARD
----------------
This project is a copy of the grocery catalogue, whose .env files pointed at a
remote production database. To make it impossible to write electronics data
there by accident, settings refuse to load unless DB_HOST is a local host and
DB_NAME is the dedicated electronics database. See _guard_local_database().
The one exception is an explicit production opt-in (ELEC_ALLOW_REMOTE_DB plus
an exact host/database allowlist), used only by the production deployment.
"""
from __future__ import annotations
import os
from pathlib import Path
# Snapshotted BEFORE load_dotenv, and that ordering is the entire point.
# load_dotenv() is called without override=True, so a variable already in the
# process environment silently beats the .env file and keeps beating it no
# matter how many times the file is corrected. That is not hypothetical here:
# the deployment platform injects its Environment tab into the container, so a
# stale value left in that tab overrides the credentials baked into the image
# (backend/Dockerfile copies .env.production to /app/.env) and the only symptom
# is a 401 that nothing explains. Comparing a name against this set answers
# "which of the two won?" - see config_source() below.
_PREEXISTING_ENV = frozenset(os.environ)
try:
from dotenv import load_dotenv
# backend/.env (one level up from this file: app/infrastructure/settings.py)
_env_path = Path(__file__).resolve().parents[2] / ".env"
load_dotenv(_env_path)
except ImportError:
# python-dotenv not installed - fall back to whatever is already in the
# process environment (e.g. set by the shell, Docker, systemd, CI, etc.)
pass
# Names whose raw value arrived wrapped in quotes or padded with whitespace.
# Recorded rather than merely fixed: stripping keeps the login working, but the
# only place the original shape is still visible is right here, before the value
# is normalised. A quoted hash is the signature of a value pasted into a web
# form, so surfacing it at startup is what stops the next person rediscovering
# it from a 401. See DB_PASSWORD in .env.production for the counter-case where
# the quotes ARE part of the secret - which is why this warns, and does not fail.
_ENV_NEEDED_CLEANUP = set()
def _clean(name: str, raw: str) -> str:
"""Strip surrounding quotes/whitespace off an env value, remembering if it mattered."""
cleaned = raw.strip().strip("'\"")
if cleaned != raw:
_ENV_NEEDED_CLEANUP.add(name)
return cleaned
def cleaned_env_names() -> list:
"""Which settings needed quote/whitespace stripping. Reported at startup."""
return sorted(_ENV_NEEDED_CLEANUP)
def config_source(name: str) -> str:
"""
Where a setting's value actually came from: the process environment, the
.env file, or this module's own default.
Reported at startup for the AUTH_* values (see app/main.py) so that an
override arriving from outside the image is visible in the logs instead of
being inferred from a failing login.
"""
if name in _PREEXISTING_ENV:
return "process-env"
if name in os.environ:
return "env-file"
return "default"
def _bool(name: str, default: str) -> bool:
return os.getenv(name, default).strip().lower() in {"1", "true", "yes"}
def _require(name: str, *, feature_flag: str) -> str:
"""Read a required secret. Raises if missing and the owning feature is enabled."""
value = os.getenv(name)
if not value:
raise RuntimeError(
f"Missing required environment variable '{name}'. It is required because "
f"'{feature_flag}' is enabled. Set it in backend/.env (copy from "
f".env.example) or disable the feature by setting {feature_flag}=false."
)
return value
# ---------------------------------------------------------------------------
# Paths
# ---------------------------------------------------------------------------
_BACKEND_ROOT = Path(__file__).resolve().parents[2]
DATA_DIR = Path(os.getenv("DATA_DIR", "").strip() or _BACKEND_ROOT / "data")
# ---------------------------------------------------------------------------
# Ollama (local LLM) - only ever used to read text we fetched, never to invent
# ---------------------------------------------------------------------------
USE_OLLAMA = _bool("USE_OLLAMA", "true")
OLLAMA_BASE_URL = os.getenv("OLLAMA_BASE_URL", "http://localhost:11434")
OLLAMA_MODEL_NAME = os.getenv("OLLAMA_MODEL_NAME", "qwen2.5:1.5b")
OLLAMA_TIMEOUT_SECONDS = int(os.getenv("OLLAMA_TIMEOUT_SECONDS", "120"))
# ---------------------------------------------------------------------------
# Embeddings (sentence-transformers, CPU-friendly)
# ---------------------------------------------------------------------------
USE_EMBEDDINGS = _bool("USE_EMBEDDINGS", "true")
EMBEDDINGS_MODEL = os.getenv("EMBEDDINGS_MODEL", "sentence-transformers/all-MiniLM-L6-v2")
EMBEDDINGS_DIM = int(os.getenv("EMBEDDINGS_DIM", "384"))
# ---------------------------------------------------------------------------
# Postgres / pgvector - the LOCAL electronics database only
# ---------------------------------------------------------------------------
ELECTRONICS_DB_NAME = "electronics_catalog"
# The test suite uses its own database on the same local server.
ALLOWED_DB_NAMES = frozenset({ELECTRONICS_DB_NAME, ELECTRONICS_DB_NAME + "_test"})
LOCAL_DB_HOSTS = frozenset({"localhost", "127.0.0.1", "::1", "host.docker.internal", "postgres"})
DB_HOST = os.getenv("DB_HOST", "127.0.0.1").strip()
DB_PORT = os.getenv("DB_PORT", "5433").strip()
DB_NAME = os.getenv("DB_NAME", ELECTRONICS_DB_NAME).strip()
DB_USER = os.getenv("DB_USER", "postgres").strip()
DB_PASSWORD = _require("DB_PASSWORD", feature_flag="the electronics database")
DB_CONNECT_TIMEOUT_SECONDS = int(os.getenv("DB_CONNECT_TIMEOUT_SECONDS", "5"))
def _csv_set(name: str) -> frozenset:
return frozenset(v.strip() for v in os.getenv(name, "").split(",") if v.strip())
# Production opt-in. Off by default: without ELEC_ALLOW_REMOTE_DB=true the
# guard below behaves exactly as it always has. With it on, only the host(s)
# and database name(s) listed here are accepted - never "any remote host".
ELEC_ALLOW_REMOTE_DB = _bool("ELEC_ALLOW_REMOTE_DB", "false")
ELEC_REMOTE_DB_HOSTS = _csv_set("ELEC_REMOTE_DB_HOSTS")
ELEC_REMOTE_DB_NAMES = _csv_set("ELEC_REMOTE_DB_NAMES")
def _guard_local_database(host: str, name: str, *, allow_remote: bool = False,
remote_hosts: frozenset = frozenset(), remote_names: frozenset = frozenset()) -> None:
"""Refuse to run against anything but the local electronics database,
unless the production opt-in names this exact host and database."""
if allow_remote and host in remote_hosts:
if name not in remote_names:
raise RuntimeError(
f"DB_NAME={name!r} is not in ELEC_REMOTE_DB_NAMES for remote host {host!r}."
)
return
if host not in LOCAL_DB_HOSTS:
raise RuntimeError(
f"DB_HOST={host!r} is not a local host. This project only runs against the "
f"local Docker database (see docker-compose.yml); it must never touch the "
f"remote catalogue database."
)
if name not in ALLOWED_DB_NAMES:
raise RuntimeError(
f"DB_NAME={name!r}; expected {ELECTRONICS_DB_NAME!r}. The electronics data "
f"lives in its own database so existing databases are never modified."
)
_guard_local_database(DB_HOST, DB_NAME, allow_remote=ELEC_ALLOW_REMOTE_DB,
remote_hosts=ELEC_REMOTE_DB_HOSTS, remote_names=ELEC_REMOTE_DB_NAMES)
# ---------------------------------------------------------------------------
# Web search (discovery, prices, images)
# ---------------------------------------------------------------------------
# DuckDuckGo needs no key. Google Programmable Search is used in addition when
# both GOOGLE_API_KEY and GOOGLE_CSE_ID are set (100 free queries/day).
USE_DDG_SEARCH = _bool("USE_DDG_SEARCH", "true")
GOOGLE_API_KEY = os.getenv("GOOGLE_API_KEY", "").strip()
GOOGLE_CSE_ID = os.getenv("GOOGLE_CSE_ID", "").strip()
USE_GOOGLE_CSE = bool(GOOGLE_API_KEY and GOOGLE_CSE_ID) and _bool("USE_GOOGLE_CSE", "true")
GOOGLE_CSE_DAILY_QUOTA = int(os.getenv("GOOGLE_CSE_DAILY_QUOTA", "100"))
SEARCH_REGION = os.getenv("SEARCH_REGION", "in-en")
# Minimum pause between two search queries to the same provider.
SEARCH_MIN_INTERVAL_SECONDS = float(os.getenv("SEARCH_MIN_INTERVAL_SECONDS", "2.5"))
SEARCH_CACHE_TTL_HOURS = int(os.getenv("SEARCH_CACHE_TTL_HOURS", "24"))
# ---------------------------------------------------------------------------
# Polite fetching of retailer / brand pages
# ---------------------------------------------------------------------------
# An honest User-Agent with a contact address. Set ELEC_CONTACT to a real
# address before running a crawl.
ELEC_CONTACT = os.getenv("ELEC_CONTACT", "admin@example.com").strip()
USER_AGENT = os.getenv(
"USER_AGENT", f"ElectronicsCatalogBot/0.1 (+mailto:{ELEC_CONTACT}; local research)"
)
REQUEST_TIMEOUT_SECONDS = int(os.getenv("REQUEST_TIMEOUT_SECONDS", "20"))
ELEC_SITE_MIN_INTERVAL_SECONDS = float(os.getenv("ELEC_SITE_MIN_INTERVAL_SECONDS", "3"))
ELEC_BREAKER_COOLDOWN_HOURS = float(os.getenv("ELEC_BREAKER_COOLDOWN_HOURS", "24"))
ELEC_MAX_PAGE_BYTES = int(os.getenv("ELEC_MAX_PAGE_BYTES", str(3 * 1024 * 1024)))
ELEC_PROBE_TTL_DAYS = int(os.getenv("ELEC_PROBE_TTL_DAYS", "7"))
MIN_IMAGE_BYTES = int(os.getenv("MIN_IMAGE_BYTES", "3000"))
# Reference pincodes (Tamil Nadu). "pincode:City" pairs, comma-separated. The
# first one is the default. A pincode is stored against a price only when the
# site actually accepted it.
ELEC_REFERENCE_PINCODES = [
tuple(p.split(":", 1)) if ":" in p else (p, "")
for p in (x.strip() for x in os.getenv("ELEC_REFERENCE_PINCODES", "641001:Coimbatore,600001:Chennai").split(","))
if p
]
# The LLM may only fill spec gaps from text we fetched; set false to run fully
# deterministic.
ELEC_USE_LLM = _bool("ELEC_USE_LLM", "true")
# ---------------------------------------------------------------------------
# FastAPI / web server
# ---------------------------------------------------------------------------
API_CORS_ORIGINS = [
origin.strip()
for origin in os.getenv("API_CORS_ORIGINS", "http://localhost:5173,http://127.0.0.1:5173").split(",")
if origin.strip()
]
# ---------------------------------------------------------------------------
# Authentication
# ---------------------------------------------------------------------------
# CORS above is not access control - browsers enforce it, and curl ignores it
# entirely. These settings are what actually guards the write/compute endpoints
# (catalog generation, ML training, uploads, chat).
#
# AUTH_ENABLED=false turns every guard off, restoring the old behaviour where
# any caller could reach any endpoint. It exists so a fresh checkout still runs
# without generating secrets first; app/infrastructure/security.py logs a
# warning at import when it is off. Never deploy with it off.
AUTH_ENABLED = _bool("AUTH_ENABLED", "true")
# Signs and verifies access tokens. Changing it invalidates every issued token,
# which is the intended way to force everyone to sign in again. Generate with:
# python scripts/make_auth_secrets.py
AUTH_SECRET_KEY = (
_require("AUTH_SECRET_KEY", feature_flag="AUTH_ENABLED")
if AUTH_ENABLED
else os.getenv("AUTH_SECRET_KEY", "")
)
# How long an issued token stays valid. 12h by default: long enough that a
# working day needs one sign-in, short enough that a leaked token expires.
AUTH_TOKEN_TTL_MINUTES = int(os.getenv("AUTH_TOKEN_TTL_MINUTES", "720"))
# The interactive accounts. Only PBKDF2 digests are stored - never a password.
# `make_auth_secrets.py` prints the lines ready to paste.
#
# `admin` is required whenever auth is on: without it nobody could sign in.
AUTH_ADMIN_USERNAME = _clean(
"AUTH_ADMIN_USERNAME", os.getenv("AUTH_ADMIN_USERNAME", "admin")
)
AUTH_ADMIN_PASSWORD_HASH = _clean(
"AUTH_ADMIN_PASSWORD_HASH",
(
_require("AUTH_ADMIN_PASSWORD_HASH", feature_flag="AUTH_ENABLED")
if AUTH_ENABLED
else os.getenv("AUTH_ADMIN_PASSWORD_HASH", "")
),
)
# The second `user` account is OPTIONAL, and left unset in this deployment.
# An empty hash is how the account is switched off: auth.py builds its account
# table from these values and omits any entry whose hash is blank, so there is
# nothing to sign in to. Setting the hash again re-enables it with no code
# change - which is exactly what the test suite does in tests/conftest.py.
AUTH_USER_USERNAME = _clean(
"AUTH_USER_USERNAME", os.getenv("AUTH_USER_USERNAME", "user")
)
AUTH_USER_PASSWORD_HASH = _clean(
"AUTH_USER_PASSWORD_HASH", os.getenv("AUTH_USER_PASSWORD_HASH", "")
)
# Failed-login throttle, applied per username+client-IP. Prevents an exposed
# login endpoint from being a free password oracle.
AUTH_MAX_LOGIN_ATTEMPTS = int(os.getenv("AUTH_MAX_LOGIN_ATTEMPTS", "10"))
AUTH_LOCKOUT_SECONDS = int(os.getenv("AUTH_LOCKOUT_SECONDS", "300"))
# Local-development escape hatch: accept ANY password at /api/auth/login, so a
# developer who does not have the configured passwords to hand can still reach
# the admin and user pages. The username still selects the role, and the token
# issued is a normal signed one - so every downstream guard, /api/auth/me, and
# the React route gating all behave exactly as they do in production. What is
# skipped is only the password check.
#
# This is NOT the same as AUTH_ENABLED=false. That disables every guard *and*
# makes /api/auth/login return 503, which breaks the login page outright. This
# flag keeps the whole auth machinery running and unlocks just the front door.
#
# Anyone who can reach the API can sign in as admin while it is on. Keep it
# false anywhere the port is reachable by someone you would not hand the admin
# password to.
AUTH_ALLOW_ANY_LOGIN = _bool("AUTH_ALLOW_ANY_LOGIN", "false")
# Shortest acceptable API key secret. token_urlsafe(32) yields 43 characters, so
# this rejects hand-typed values without rejecting anything the documented
# generator produces.
API_KEY_MIN_LENGTH = 32
def _parse_api_keys(raw: str) -> dict:
"""
Parse ``API_KEYS`` - ``name:role:secret`` triples, comma-separated.
Keyed by secret because that is what an inbound request presents. One entry
per consumer is the point: a shared key cannot be revoked for one caller
without breaking all of them.
Secrets must be at least API_KEY_MIN_LENGTH characters. That is not about
guessing the key over the network - the lockout and the network itself make
online brute force impractical - but about what /api/health publishes. It
reports a truncated digest of every configured key so a deployment can be
checked against the config it was built from, and a digest of a *raw* secret
is only safe when the secret is unguessable offline. An admin password hash
embeds a random salt, so its fingerprint discloses nothing; an API key has no
salt, and a hand-picked "changeme" would fall to a wordlist in seconds.
Generate one with: python -c "import secrets; print(secrets.token_urlsafe(32))"
"""
parsed: dict = {}
for entry in raw.split(","):
entry = entry.strip()
if not entry:
continue
parts = entry.split(":")
if len(parts) != 3:
raise RuntimeError(
f"Malformed API_KEYS entry {entry!r}. Expected 'name:role:secret', "
f"comma-separated between entries."
)
name, role, secret = (p.strip() for p in parts)
# MUST stay in step with ROLE_PERMISSIONS in app/infrastructure/security.py,
# which is the source of truth. It is duplicated rather than imported
# because security.py imports THIS module, so importing it back here
# would be a cycle. A role added there but not here is rejected at boot
# with the message below - loud, and before any request is served.
if role not in {"admin", "user", "uploader"}:
raise RuntimeError(
f"API_KEYS entry {name!r} has role {role!r}; expected 'admin', 'user' "
f"or 'uploader'."
)
if not secret:
raise RuntimeError(f"API_KEYS entry {name!r} has an empty secret.")
if len(secret) < API_KEY_MIN_LENGTH:
raise RuntimeError(
f"API_KEYS entry {name!r} has a {len(secret)}-character secret; at least "
f"{API_KEY_MIN_LENGTH} are required, because /api/health publishes a digest "
f"of it. Generate one with: "
f"python -c \"import secrets; print(secrets.token_urlsafe(32))\""
)
parsed[secret] = (name, role)
return parsed
# Machine consumers of api.<domain>. Empty by default - browser sessions go
# through /api/auth/login instead, and a key that nobody needs is only risk.
#
# NAME THE KEY FOR ITS FUNCTION, NOT THE PERSON HOLDING IT.
# /api/health is public and reports {name, role, fingerprint} for every
# configured key (describe_api_keys in security.py). The secret is never
# exposed, but the NAME is - so `catalog-drop:uploader:...` is right and
# `priya-laptop:uploader:...` publishes a colleague's name to anyone who
# curls the health endpoint.
API_KEYS = _parse_api_keys(os.getenv("API_KEYS", ""))

133
backend/app/main.py Normal file
View File

@@ -0,0 +1,133 @@
"""
FastAPI application entry point for the Electronics Catalog (local only).
Run with (from backend/):
.venv\\Scripts\\python -m uvicorn app.main:app --host 127.0.0.1 --port 8000
"""
from __future__ import annotations
import logging
import os
import threading
from contextlib import asynccontextmanager
from pathlib import Path
from fastapi import FastAPI, HTTPException, Request
from fastapi.encoders import jsonable_encoder
from fastapi.exceptions import RequestValidationError
from fastapi.middleware.cors import CORSMiddleware
from fastapi.responses import FileResponse, JSONResponse
from fastapi.staticfiles import StaticFiles
from app.api.routers import auth, elec, elec_admin, health
from app.infrastructure.security import auth_config_summary
from app.infrastructure.settings import API_CORS_ORIGINS, DB_HOST, DB_NAME, DB_PORT, cleaned_env_names
from app.mcp_server import mcp
from fastmcp.utilities.lifespan import combine_lifespans
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
)
logger = logging.getLogger(__name__)
def _init_schema() -> None:
"""Apply pending migrations and make sure reference data exists. Runs on a
thread so the API answers /api/health even while Postgres is starting."""
try:
from app.electronics.db import repository as repo
from app.electronics.db.migrate import run_migrations
from app.electronics.reference import load_reference
applied = run_migrations()
if applied:
logger.info("Applied migrations: %s", ", ".join(applied))
repo.seed_reference(load_reference())
except Exception as exc: # noqa: BLE001 - the health endpoint reports the DB state
logger.warning("Schema initialisation skipped: %s", exc)
@asynccontextmanager
async def lifespan(_app: FastAPI):
logger.info("Electronics Catalog using database %s at %s:%s", DB_NAME, DB_HOST, DB_PORT)
threading.Thread(target=_init_schema, daemon=True).start()
yield
# MCP endpoint (FastMCP) for AI assistants, served at /mcp/ by this same app.
# Its session manager must start with the app, hence the combined lifespan.
mcp_app = mcp.http_app(path="/")
app = FastAPI(
title="Electronics Catalog API",
description=(
"Local, evidence-backed catalogue of electronics sold in India (Tamil Nadu focus). "
"Products, prices, availability and images are collected from real retail listings "
"found through web search; nothing is generated."
),
version="1.0.0",
lifespan=combine_lifespans(lifespan, mcp_app.lifespan),
)
_allow_credentials = "*" not in API_CORS_ORIGINS
app.add_middleware(
CORSMiddleware,
allow_origins=API_CORS_ORIGINS,
allow_credentials=_allow_credentials,
allow_methods=["*"],
allow_headers=["*"],
)
def _json_safe(value):
if isinstance(value, float) and (value != value or value in (float("inf"), float("-inf"))):
return str(value)
if isinstance(value, dict):
return {k: _json_safe(v) for k, v in value.items()}
if isinstance(value, (list, tuple)):
return [_json_safe(v) for v in value]
return value
@app.exception_handler(RequestValidationError)
async def _validation_error_as_422(request: Request, exc: RequestValidationError) -> JSONResponse:
return JSONResponse(status_code=422, content={"detail": _json_safe(jsonable_encoder(exc.errors()))})
_auth_cfg = auth_config_summary()
logger.info(
"Auth config: enabled=%s allow_any_login=%s admin_username=%r hash_valid=%s",
_auth_cfg["enabled"], _auth_cfg["allow_any_login"], _auth_cfg["admin_username"],
_auth_cfg["password_hash_valid"],
)
if _auth_cfg["allow_any_login"]:
logger.warning("AUTH_ALLOW_ANY_LOGIN=true: any password is accepted. Keep the API on 127.0.0.1.")
if cleaned_env_names():
logger.warning("Settings arrived quoted or padded and were cleaned: %s", ", ".join(cleaned_env_names()))
app.include_router(health.router, prefix="/api")
app.include_router(auth.router, prefix="/api")
app.include_router(elec.router, prefix="/api")
app.include_router(elec_admin.router, prefix="/api")
app.mount("/mcp", mcp_app)
_dist_override = os.getenv("FRONTEND_DIST_DIR", "").strip()
FRONTEND_DIST = Path(_dist_override) if _dist_override else Path(__file__).resolve().parents[2] / "frontend" / "dist"
if (FRONTEND_DIST / "assets").exists():
logger.info("Serving built frontend from %s", FRONTEND_DIST)
app.mount("/assets", StaticFiles(directory=str(FRONTEND_DIST / "assets")), name="assets")
@app.get("/{full_path:path}")
def serve_frontend(full_path: str):
if full_path.startswith(("api", "mcp", "docs", "redoc", "openapi.json")):
raise HTTPException(status_code=404, detail="Not found")
file_path = FRONTEND_DIST / full_path
if file_path.exists() and file_path.is_file():
return FileResponse(file_path)
return FileResponse(FRONTEND_DIST / "index.html")
else:
@app.get("/")
def root() -> dict:
return {"service": "Electronics Catalog API", "docs": "/docs", "health": "/api/health", "mcp": "/mcp/"}

121
backend/app/mcp_server.py Normal file
View File

@@ -0,0 +1,121 @@
"""MCP server (FastMCP) over the read-only catalogue.
Mounted by app/main.py at /mcp/, in the same process as the REST API. Each tool
calls the same function the matching /api/elec endpoint uses, so the two can
never disagree. Only read-only catalogue data is exposed: no admin, no login,
no collection runs.
Images are returned as URLs (the retailer's own image address, as stored in
elec.product_image); nothing is downloaded or re-hosted.
"""
from __future__ import annotations
from decimal import Decimal
from typing import Any, Dict, List, Optional
import anyio
from fastapi import HTTPException
from fastmcp import FastMCP
from fastmcp.exceptions import ToolError
from app.api.routers import elec
mcp = FastMCP(
name="Electronics Catalog",
instructions=(
"Verified catalogue of mobiles and laptops sold in India (Tamil Nadu focus). "
"Every product is confirmed by real listings on at least two retail platforms; "
"prices, ratings and reviews come with the page they were read from. Prices are "
"rupee strings. Use search_products to find products, then get_product for "
"per-platform offers, specs, images, rating and reviews."
),
)
_SEARCH_FIELDS = (
"product_id", "brand", "category", "display_name", "ram_gb", "storage_gb",
"best_price", "best_price_site", "platform_count", "sold_by_tn_retailer", "image_url",
)
async def _run(fn, *args):
"""The catalogue functions use blocking DB calls; keep them off the event loop."""
try:
return await anyio.to_thread.run_sync(lambda: fn(*args))
except HTTPException as exc:
raise ToolError(str(exc.detail)) from exc
@mcp.tool
async def list_categories() -> List[Dict[str, Any]]:
"""List product categories (e.g. mobiles, laptops) with how many verified products each has."""
return await _run(elec.categories)
@mcp.tool
async def search_products(
query: Optional[str] = None,
category: Optional[str] = None,
brand: Optional[str] = None,
max_price: Optional[float] = None,
min_price: Optional[float] = None,
limit: int = 20,
) -> Dict[str, Any]:
"""Search verified products.
Args:
query: Text to match in the product or brand name, e.g. "galaxy s25", "vivobook".
category: Category slug: "mobiles" or "laptops".
brand: Brand slug, e.g. "samsung", "xiaomi", "hp", "lenovo".
max_price: Highest best price in rupees.
min_price: Lowest best price in rupees.
limit: Maximum products to return (1-100).
Returns the total match count and, per product: id, name, variant, best price (rupee
string) and the platform offering it, number of platforms, and an image URL (or null).
"""
limit = max(1, min(int(limit), 100))
result = await _run(
elec.products,
category, brand, (query or None),
None if min_price is None else Decimal(str(min_price)),
None if max_price is None else Decimal(str(max_price)),
False, None, False, limit, 0,
)
return {
"total": result["total"],
"products": [{k: p.get(k) for k in _SEARCH_FIELDS} for p in result["products"]],
}
@mcp.tool
async def get_product(product_id: int) -> Dict[str, Any]:
"""Full details of one product by its product_id (from search_products).
Returns per-platform offers (price, MRP, source URL, when seen), normalised specs,
image URLs, the overall rating with per-platform sources (null if none is published),
and up to 10 real customer reviews (often empty).
"""
d = await _run(elec.product, int(product_id))
return {
"product_id": d["product_id"],
"brand": d.get("brand"),
"category": d.get("category"),
"display_name": d.get("display_name"),
"best_price": d.get("best_price"),
"best_price_site": d.get("best_price_site"),
"specs": d.get("canonical_specs") or {},
"offers": [
{k: o.get(k) for k in ("site", "price", "mrp", "source_url", "observed_at", "price_outlier")}
for o in d.get("offers", [])
],
"image_urls": [i["url"] for i in d.get("images", [])],
"rating": d.get("rating"),
"reviews": d.get("reviews", []),
}
@mcp.tool
async def price_history(product_id: int) -> List[Dict[str, Any]]:
"""Every price observed for a product, per platform, oldest first (rupee strings, ISO times)."""
rows = await _run(elec.price_history, int(product_id))
return [{k: r.get(k) for k in ("site", "price", "mrp", "observed_at")} for r in rows]

View File

View File

@@ -0,0 +1,52 @@
from __future__ import annotations
from typing import List, Optional, TYPE_CHECKING
from app.infrastructure.settings import EMBEDDINGS_MODEL
if TYPE_CHECKING: # pragma: no cover - typing only, no runtime cost
from sentence_transformers import SentenceTransformer
_model_singleton: Optional["SentenceTransformer"] = None
def get_device() -> str:
"""Prefer CUDA if available, otherwise CPU.
Imports torch lazily: on an 8GB RAM / CPU-only laptop there is no
benefit to importing torch (and paying its startup/memory cost) until
an embedding is actually requested, so the FastAPI process can boot
and answer /api/health almost instantly.
"""
import torch # local import - see docstring
return "cuda" if torch.cuda.is_available() else "cpu"
def get_embedding_model() -> "SentenceTransformer":
global _model_singleton
if _model_singleton is None:
from sentence_transformers import SentenceTransformer # local import - see get_device()
device = get_device()
_model_singleton = SentenceTransformer(EMBEDDINGS_MODEL, device=device)
return _model_singleton
def embed_texts(texts: List[str]) -> List[List[float]]:
"""Embed a batch of texts into normalized 384-dim vectors (MiniLM-L6-v2).
Normalized so that pgvector's cosine-distance operator (`<=>`) behaves
consistently for the RAG retrieval step in `app.services.vector_store`.
"""
if not texts:
return []
model = get_embedding_model()
embeddings = model.encode(
texts,
batch_size=32,
normalize_embeddings=True,
convert_to_numpy=True,
show_progress_bar=False,
)
return embeddings.tolist()

View File

@@ -0,0 +1,171 @@
"""
Thin client for the local Ollama server.
In this project the LLM is an EXTRACTOR, never a source: it is only ever
handed text that was fetched from a real page or search result, and every
value it returns is checked against that text by
app.electronics.normalise.grounding before it is kept. There are deliberately
no functions here that ask the model to list products, prices or images.
"""
from __future__ import annotations
import json
import re
import time
import requests
from app.infrastructure.settings import OLLAMA_BASE_URL, OLLAMA_MODEL_NAME, USE_OLLAMA, OLLAMA_TIMEOUT_SECONDS
# Reachability is asked once per this many seconds, not once per caller.
#
# WHY THIS CACHE EXISTS. The probe below costs up to 5 seconds when nothing is
# listening, and `_ensure_client` is called per ROW by stage 2 of the ingestion
# pipeline (store_catalog_pipeline.stage_2_row_intake -> fetch_product_details).
# Uncached, a 2000-row sheet ingested with use_llm on, against a configured but
# unreachable Ollama, spends up to ~2.8 hours doing nothing but timing out - and
# presents as a batch that has hung rather than one that has failed. That is not
# hypothetical: USE_OLLAMA=true pointing at localhost:11434 is the default
# developer configuration, and `ollama serve` is not always running beside it.
#
# /api/health calls this too (app/api/routers/system.py), so the same cache
# stops a down Ollama adding 5s to every health request.
#
# A TTL rather than a permanent memo, deliberately: this is a liveness fact, not
# configuration. Cached forever, an Ollama started after the API would never be
# noticed and /api/health would report it down until a redeploy.
_PROBE_TTL_SECONDS = 30.0
_probe_cache: tuple[float, bool] | None = None
def reset_reachability_cache() -> None:
"""Forget the cached probe. For tests, and for anything that knows the
answer just changed."""
global _probe_cache
_probe_cache = None
def _ensure_client():
"""None when Ollama is switched off, True/False for reachable or not.
Three return values, not two - `system.py` relies on telling "disabled" from
"configured but down", so do not collapse this to a bool.
"""
if not USE_OLLAMA:
# No network call on this path, so nothing worth caching.
return None
global _probe_cache
now = time.monotonic()
if _probe_cache is not None and now - _probe_cache[0] < _PROBE_TTL_SECONDS:
return _probe_cache[1]
# Verify Ollama is reachable
try:
resp = requests.get(f"{OLLAMA_BASE_URL}/api/tags", timeout=5)
reachable = resp.status_code == 200
except Exception:
reachable = False
_probe_cache = (now, reachable)
return reachable
def _generate(
system: str,
user_prompt: str,
max_retries: int = 2,
*,
temperature: float = 0.0,
json_mode: bool = False,
) -> str:
"""Call Ollama's chat endpoint and return text safely.
Retries up to `max_retries` times when the response is empty, since
small local models (e.g. qwen2.5:1.5b) sometimes return empty content
for complex JSON prompts on the first attempt.
"""
if not _ensure_client():
return ""
for attempt in range(max_retries + 1):
try:
resp = requests.post(
f"{OLLAMA_BASE_URL}/api/chat",
json={
"model": OLLAMA_MODEL_NAME,
"messages": [
{"role": "system", "content": system},
{"role": "user", "content": user_prompt},
],
"stream": False,
"options": {"temperature": temperature},
**({"format": "json"} if json_mode else {}),
},
timeout=OLLAMA_TIMEOUT_SECONDS,
)
resp.raise_for_status()
data = resp.json()
content = (data.get("message", {}).get("content", "") or "").strip()
if content:
return content
if attempt < max_retries:
import time
time.sleep(1.0)
except Exception:
if attempt >= max_retries:
return ""
import time
time.sleep(1.0)
return ""
def _extract_json(text: str) -> dict | None:
"""Extract JSON from model response, trying multiple strategies.
Handles both JSON objects {...} and JSON arrays [...] since small
local models frequently return bare arrays instead of an object with
a ``products`` key.
"""
if not text:
return None
# Try fenced code block (object or array)
match = re.search(r"```(?:json)?\s*(\{[\s\S]*?\}|\[[\s\S]*?\])\s*```", text)
if match:
try:
return json.loads(match.group(1))
except json.JSONDecodeError:
pass
# Try first JSON value in text (greedy - object)
brace = re.search(r"\{[\s\S]*\}", text)
if brace:
try:
return json.loads(brace.group(0))
except json.JSONDecodeError:
pass
# Try first JSON value in text (greedy - array)
bracket = re.search(r"\[[\s\S]*\]", text)
if bracket:
try:
return json.loads(bracket.group(0))
except json.JSONDecodeError:
pass
# Try parsing entire text
try:
return json.loads(text.strip())
except json.JSONDecodeError:
pass
# Fallback: try to fix common issues
cleaned = text.strip()
cleaned = re.sub(r"(?<=[:,\[])\s*'", '"', cleaned)
cleaned = re.sub(r"'\s*(?=[,:\}\]])", '"', cleaned)
try:
return json.loads(cleaned)
except json.JSONDecodeError:
return None
def generate_json(system_prompt: str, user_prompt: str) -> dict | list | None:
"""Deterministic JSON-mode completion. None when Ollama is unavailable or
the reply is not JSON - callers must treat that as "no extra data"."""
return _extract_json(_generate(system_prompt, user_prompt, max_retries=1, json_mode=True))