Electronics Catalog: API, MCP server, frontend and deployment
Verified catalogue of mobiles and laptops sold in India, collected from real retail listings (FastAPI backend, React frontend, Postgres/pgvector). - REST API under /api/elec (read-only catalogue; admin endpoints need login) - MCP server (FastMCP) at /mcp/ with list_categories, search_products, get_product and price_history tools - Real ratings and reviews read from product pages and search results - Production Dockerfile (requirements-api.txt, no PyTorch) and .env.production.example; remote database only via an explicit ELEC_ALLOW_REMOTE_DB host/name allowlist - docs/API.md: endpoint and MCP reference with live examples Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
0
backend/app/__init__.py
Normal file
0
backend/app/__init__.py
Normal file
0
backend/app/api/__init__.py
Normal file
0
backend/app/api/__init__.py
Normal file
31
backend/app/api/background.py
Normal file
31
backend/app/api/background.py
Normal file
@@ -0,0 +1,31 @@
|
||||
"""
|
||||
Minimal in-process background job dispatcher for long-running admin jobs
|
||||
(catalog ingestion, store seeding, ML model training, nutrition
|
||||
enrichment).
|
||||
|
||||
This deliberately does NOT use Starlette's `BackgroundTasks`. BackgroundTasks
|
||||
run *synchronously after the response is sent*: an async background task is
|
||||
awaited directly on the server's event loop, and a sync one is awaited in the
|
||||
request's thread. Either way the request handler does not return until the job
|
||||
finishes. For jobs that take minutes (LLM calls, web scraping, ML training,
|
||||
Open Food Facts lookups), that turns a "kick off a job and return 202" endpoint
|
||||
into a blocking call and, for async tasks, freezes the whole API event loop for
|
||||
the duration.
|
||||
|
||||
A daemon thread returns control to the caller immediately, and the job's
|
||||
progress stays visible via the job_store polling endpoints the UI already
|
||||
uses. Daemon threads are a deliberate, documented trade-off (see
|
||||
`app/api/job_store.py`): state is process-local and not safe across multiple
|
||||
uvicorn workers - fine for this project's intended single-process, CPU-only
|
||||
deployment.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import threading
|
||||
from typing import Any, Callable
|
||||
|
||||
|
||||
def run_in_background(func: Callable[[], Any], *, name: str) -> None:
|
||||
"""Start `func` on a new daemon thread and return immediately."""
|
||||
thread = threading.Thread(target=func, name=name, daemon=True)
|
||||
thread.start()
|
||||
144
backend/app/api/deps.py
Normal file
144
backend/app/api/deps.py
Normal file
@@ -0,0 +1,144 @@
|
||||
"""
|
||||
Request-scoped authentication dependencies.
|
||||
|
||||
Guards are attached per route, not as middleware matching on paths. Two
|
||||
reasons that matters here:
|
||||
|
||||
* A path-matching middleware silently stops guarding a route the moment
|
||||
somebody renames it. A ``Depends`` on the route function cannot drift out
|
||||
of sync with the route it protects.
|
||||
* FastAPI reflects these into the OpenAPI schema, so ``/docs`` shows which
|
||||
operations need a credential instead of implying everything is open.
|
||||
|
||||
The guard therefore holds regardless of which host the request arrives on -
|
||||
through the frontend's nginx on ``{$DOMAIN}``, or directly on ``api.{$DOMAIN}``.
|
||||
|
||||
Usage::
|
||||
|
||||
@router.post("/thing", dependencies=[Depends(require_admin)])
|
||||
def create_thing(): ...
|
||||
|
||||
@router.post("/other", dependencies=[Depends(require_permission("add_product"))])
|
||||
def other_thing(): ...
|
||||
|
||||
@router.post("/who", ...)
|
||||
def who(principal: Principal = Depends(get_principal)): ...
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Callable, Optional
|
||||
|
||||
from fastapi import Depends, HTTPException, status
|
||||
from fastapi.security import APIKeyHeader, HTTPAuthorizationCredentials, HTTPBearer
|
||||
|
||||
from app.infrastructure.security import (
|
||||
AuthError,
|
||||
Principal,
|
||||
anonymous_principal,
|
||||
decode_access_token,
|
||||
principal_for_api_key,
|
||||
)
|
||||
from app.infrastructure.settings import AUTH_ENABLED
|
||||
|
||||
# auto_error=False on both: with two accepted credential types, letting either
|
||||
# scheme raise on its own would reject a request that carried the *other* one.
|
||||
# get_principal decides, once it has seen both.
|
||||
_bearer_scheme = HTTPBearer(auto_error=False, description="Access token from POST /api/auth/login")
|
||||
_api_key_scheme = APIKeyHeader(
|
||||
name="X-API-Key",
|
||||
auto_error=False,
|
||||
description="Static key for machine consumers (see API_KEYS)",
|
||||
)
|
||||
|
||||
_UNAUTHENTICATED = HTTPException(
|
||||
status_code=status.HTTP_401_UNAUTHORIZED,
|
||||
detail="Not authenticated. Send a bearer token from POST /api/auth/login, or an X-API-Key header.",
|
||||
headers={"WWW-Authenticate": "Bearer"},
|
||||
)
|
||||
|
||||
|
||||
def get_principal(
|
||||
credentials: Optional[HTTPAuthorizationCredentials] = Depends(_bearer_scheme),
|
||||
api_key: Optional[str] = Depends(_api_key_scheme),
|
||||
) -> Principal:
|
||||
"""Resolve the caller, or raise 401. Use this to require *any* valid credential."""
|
||||
if not AUTH_ENABLED:
|
||||
return anonymous_principal()
|
||||
|
||||
if credentials is not None and credentials.credentials:
|
||||
try:
|
||||
return decode_access_token(credentials.credentials)
|
||||
except AuthError as exc:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_401_UNAUTHORIZED,
|
||||
detail=str(exc),
|
||||
headers={"WWW-Authenticate": "Bearer"},
|
||||
) from exc
|
||||
|
||||
if api_key:
|
||||
try:
|
||||
return principal_for_api_key(api_key)
|
||||
except AuthError as exc:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_401_UNAUTHORIZED, detail=str(exc)
|
||||
) from exc
|
||||
|
||||
raise _UNAUTHENTICATED
|
||||
|
||||
|
||||
def get_optional_principal(
|
||||
credentials: Optional[HTTPAuthorizationCredentials] = Depends(_bearer_scheme),
|
||||
api_key: Optional[str] = Depends(_api_key_scheme),
|
||||
) -> Optional[Principal]:
|
||||
"""
|
||||
Resolve the caller if they presented a valid credential, else None.
|
||||
|
||||
For endpoints that are public but behave differently when signed in. A
|
||||
credential that is present but *invalid* still raises - failing open there
|
||||
would mean a typo'd token silently downgrades to anonymous access.
|
||||
"""
|
||||
if not AUTH_ENABLED:
|
||||
return anonymous_principal()
|
||||
if credentials is None and not api_key:
|
||||
return None
|
||||
return get_principal(credentials, api_key)
|
||||
|
||||
|
||||
def require_role(*roles: str) -> Callable[[Principal], Principal]:
|
||||
"""Require the caller to hold one of ``roles``."""
|
||||
allowed = frozenset(roles)
|
||||
|
||||
def _dependency(principal: Principal = Depends(get_principal)) -> Principal:
|
||||
if principal.role not in allowed:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_403_FORBIDDEN,
|
||||
detail=(
|
||||
f"This operation requires the {' or '.join(sorted(allowed))} role; "
|
||||
f"you are signed in as '{principal.role}'."
|
||||
),
|
||||
)
|
||||
return principal
|
||||
|
||||
return _dependency
|
||||
|
||||
|
||||
def require_permission(permission: str) -> Callable[[Principal], Principal]:
|
||||
"""
|
||||
Require a specific permission. ``admin`` passes every check - see
|
||||
``Principal.has_permission``.
|
||||
"""
|
||||
|
||||
def _dependency(principal: Principal = Depends(get_principal)) -> Principal:
|
||||
if not principal.has_permission(permission):
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_403_FORBIDDEN,
|
||||
detail=f"This operation requires the '{permission}' permission.",
|
||||
)
|
||||
return principal
|
||||
|
||||
return _dependency
|
||||
|
||||
|
||||
# The two guards used most often, named so route decorators stay readable.
|
||||
require_admin = require_role("admin")
|
||||
require_authenticated = get_principal
|
||||
59
backend/app/api/job_store.py
Normal file
59
backend/app/api/job_store.py
Normal file
@@ -0,0 +1,59 @@
|
||||
"""
|
||||
Tiny in-memory job tracker for background catalog-generation tasks.
|
||||
|
||||
Deliberately not a queue/Celery/Redis setup - the original project already
|
||||
had celery+redis in requirements.txt but nothing wired it up, and adding a
|
||||
broker is unnecessary operational weight for a single-developer, CPU-only
|
||||
project. A process-local dict is enough to let the React UI show
|
||||
"running -> done/failed" status for a brand ingestion job started from the
|
||||
admin panel.
|
||||
|
||||
NOTE: state is lost on server restart, and is per-process (not safe for
|
||||
multiple uvicorn workers). For this project's intended scale (one backend
|
||||
process on a personal machine) that's a fine trade-off; see the docs'
|
||||
"Scaling beyond a single machine" section if this ever needs to change.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import threading
|
||||
import time
|
||||
import uuid
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Dict, Optional
|
||||
|
||||
|
||||
@dataclass
|
||||
class Job:
|
||||
job_id: str
|
||||
brand: str
|
||||
status: str = "pending" # pending -> running -> done | failed
|
||||
detail: Optional[str] = None
|
||||
created_at: float = field(default_factory=time.time)
|
||||
updated_at: float = field(default_factory=time.time)
|
||||
|
||||
|
||||
class JobStore:
|
||||
def __init__(self) -> None:
|
||||
self._jobs: Dict[str, Job] = {}
|
||||
self._lock = threading.Lock()
|
||||
|
||||
def create(self, brand: str) -> Job:
|
||||
job = Job(job_id=str(uuid.uuid4()), brand=brand)
|
||||
with self._lock:
|
||||
self._jobs[job.job_id] = job
|
||||
return job
|
||||
|
||||
def update(self, job_id: str, status: str, detail: Optional[str] = None) -> None:
|
||||
with self._lock:
|
||||
job = self._jobs.get(job_id)
|
||||
if job:
|
||||
job.status = status
|
||||
job.detail = detail
|
||||
job.updated_at = time.time()
|
||||
|
||||
def get(self, job_id: str) -> Optional[Job]:
|
||||
with self._lock:
|
||||
return self._jobs.get(job_id)
|
||||
|
||||
|
||||
job_store = JobStore()
|
||||
0
backend/app/api/routers/__init__.py
Normal file
0
backend/app/api/routers/__init__.py
Normal file
357
backend/app/api/routers/auth.py
Normal file
357
backend/app/api/routers/auth.py
Normal file
@@ -0,0 +1,357 @@
|
||||
"""
|
||||
Authentication router - issues and inspects access tokens.
|
||||
|
||||
This replaces an earlier version that returned a role profile without issuing
|
||||
anything, accepted an empty password, and granted `admin` to any username that
|
||||
asked for the role. It decided which buttons the UI drew; it protected nothing.
|
||||
Now the token this returns is the credential every write endpoint checks (see
|
||||
app/api/deps.py), so the rules hold for curl and partner scripts too, not just
|
||||
for the React app.
|
||||
|
||||
Accounts come from the environment - two of them, admin and user, configured as
|
||||
PBKDF2 digests. That is deliberately not a user database: this project has no
|
||||
user table, no registration flow and no password reset, and inventing one here
|
||||
would be a bigger change than the problem calls for. Machine consumers get
|
||||
API_KEYS instead. If per-user accounts become a real requirement, this module
|
||||
is the seam to replace.
|
||||
|
||||
For local work there is AUTH_ALLOW_ANY_LOGIN, which skips the password check
|
||||
here and nowhere else - the token still gets signed and every guard downstream
|
||||
still checks it. It is off by default and logs a warning at startup when on.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import threading
|
||||
import time
|
||||
from typing import Dict, List, Tuple
|
||||
|
||||
from fastapi import APIRouter, Depends, HTTPException, Request, status
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
from app.api.deps import get_principal
|
||||
from app.infrastructure.security import (
|
||||
ROLE_PERMISSIONS,
|
||||
Principal,
|
||||
create_access_token,
|
||||
hash_is_wellformed,
|
||||
password_hash_fingerprint,
|
||||
verify_password,
|
||||
)
|
||||
from app.infrastructure.settings import (
|
||||
AUTH_ADMIN_PASSWORD_HASH,
|
||||
AUTH_ADMIN_USERNAME,
|
||||
AUTH_ALLOW_ANY_LOGIN,
|
||||
AUTH_ENABLED,
|
||||
AUTH_LOCKOUT_SECONDS,
|
||||
AUTH_MAX_LOGIN_ATTEMPTS,
|
||||
AUTH_USER_PASSWORD_HASH,
|
||||
AUTH_USER_USERNAME,
|
||||
config_source,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
router = APIRouter(prefix="/auth", tags=["auth"])
|
||||
|
||||
if AUTH_ENABLED and AUTH_ALLOW_ANY_LOGIN:
|
||||
logger.warning(
|
||||
"AUTH_ALLOW_ANY_LOGIN=true: /api/auth/login accepts ANY password, so anyone "
|
||||
"who can reach this port can sign in as admin. Local development only - "
|
||||
"set it to false in backend/.env before exposing this server."
|
||||
)
|
||||
|
||||
|
||||
class LoginRequest(BaseModel):
|
||||
username: str = Field(min_length=1, max_length=150)
|
||||
password: str = Field(min_length=1, max_length=1024)
|
||||
|
||||
|
||||
class UserProfile(BaseModel):
|
||||
username: str
|
||||
role: str
|
||||
display_name: str
|
||||
email: str
|
||||
permissions: List[str] = Field(default_factory=list)
|
||||
|
||||
|
||||
class LoginResponse(BaseModel):
|
||||
access_token: str
|
||||
token_type: str = "bearer"
|
||||
expires_in: int = Field(description="Token lifetime in seconds")
|
||||
user: UserProfile
|
||||
|
||||
|
||||
# A syntactically valid hash of an unguessable value. Never matches any real
|
||||
# password; it exists only so the unknown-username path in login() does the
|
||||
# same PBKDF2 work as the known one, keeping the two indistinguishable by timing.
|
||||
_DUMMY_HASH = (
|
||||
"pbkdf2_sha256$600000$YWJjZGVmZ2hpamtsbW5vcA==$"
|
||||
"S1cVFrGD4pDkGqSjbEbaVSTONzGhCT9BOaWPQ2vwvvA="
|
||||
)
|
||||
|
||||
|
||||
def _accounts() -> Dict[str, dict]:
|
||||
"""
|
||||
The configured accounts, read per call so a settings reload is picked up.
|
||||
|
||||
Usernames are compared case-insensitively (matching what the login form
|
||||
sends), but the password is not touched - the previous version lowercased
|
||||
it before comparing, which silently shrank the effective keyspace.
|
||||
|
||||
An account with a blank password hash is omitted entirely rather than
|
||||
included with an unmatchable digest. Both spellings deny the login, but
|
||||
only omission keeps it out of the account table, so nothing downstream can
|
||||
treat it as a real account. This is how the optional `user` account is
|
||||
switched off: leave AUTH_USER_PASSWORD_HASH unset and only `admin` exists.
|
||||
"""
|
||||
accounts = {
|
||||
AUTH_ADMIN_USERNAME.lower(): {
|
||||
"password_hash": AUTH_ADMIN_PASSWORD_HASH,
|
||||
"role": "admin",
|
||||
"display_name": "System Administrator",
|
||||
"email": "admin@nutritionintel.com",
|
||||
},
|
||||
}
|
||||
|
||||
if AUTH_USER_PASSWORD_HASH:
|
||||
accounts[AUTH_USER_USERNAME.lower()] = {
|
||||
"password_hash": AUTH_USER_PASSWORD_HASH,
|
||||
"role": "user",
|
||||
"display_name": "Product & Store Manager",
|
||||
"email": "user@nutritionintel.com",
|
||||
}
|
||||
|
||||
return accounts
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Failed-login throttle
|
||||
# ---------------------------------------------------------------------------
|
||||
# In-process and per-worker: with several uvicorn workers a determined attacker
|
||||
# gets AUTH_MAX_LOGIN_ATTEMPTS per worker, not overall. That is a real limit,
|
||||
# not a rounding error - but it still turns an unbounded password oracle into a
|
||||
# rate-limited one without adding Redis to the deployment. Move this to a shared
|
||||
# store if you ever run many workers.
|
||||
_failures: Dict[Tuple[str, str], Tuple[int, float]] = {}
|
||||
_failures_lock = threading.Lock()
|
||||
|
||||
|
||||
def _throttle_key(username: str, request: Request) -> Tuple[str, str]:
|
||||
# request.client.host is the real client IP because uvicorn runs with
|
||||
# --proxy-headers behind nginx/Caddy (see backend/Dockerfile); without that
|
||||
# every request would appear to come from the proxy and share one bucket.
|
||||
client = request.client.host if request.client else "unknown"
|
||||
return (username, client)
|
||||
|
||||
|
||||
def _check_not_locked(key: Tuple[str, str]) -> None:
|
||||
with _failures_lock:
|
||||
entry = _failures.get(key)
|
||||
if entry is None:
|
||||
return
|
||||
count, first_seen = entry
|
||||
if time.time() - first_seen > AUTH_LOCKOUT_SECONDS:
|
||||
del _failures[key]
|
||||
return
|
||||
if count >= AUTH_MAX_LOGIN_ATTEMPTS:
|
||||
retry_after = int(AUTH_LOCKOUT_SECONDS - (time.time() - first_seen))
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_429_TOO_MANY_REQUESTS,
|
||||
detail=f"Too many failed sign-in attempts. Try again in {retry_after}s.",
|
||||
headers={"Retry-After": str(max(retry_after, 1))},
|
||||
)
|
||||
|
||||
|
||||
def _record_failure(key: Tuple[str, str]) -> None:
|
||||
now = time.time()
|
||||
with _failures_lock:
|
||||
count, first_seen = _failures.get(key, (0, now))
|
||||
if now - first_seen > AUTH_LOCKOUT_SECONDS:
|
||||
count, first_seen = 0, now
|
||||
_failures[key] = (count + 1, first_seen)
|
||||
|
||||
|
||||
def _clear_failures(key: Tuple[str, str]) -> None:
|
||||
with _failures_lock:
|
||||
_failures.pop(key, None)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Routes
|
||||
# ---------------------------------------------------------------------------
|
||||
@router.post("/login", response_model=LoginResponse)
|
||||
def login(payload: LoginRequest, request: Request) -> LoginResponse:
|
||||
"""Exchange a username and password for an access token."""
|
||||
if not AUTH_ENABLED:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_503_SERVICE_UNAVAILABLE,
|
||||
detail=(
|
||||
"Authentication is disabled on this server (AUTH_ENABLED=false), so no "
|
||||
"token can be issued. Every endpoint is open; sign-in is not required."
|
||||
),
|
||||
)
|
||||
|
||||
username = payload.username.strip().lower()
|
||||
key = _throttle_key(username, request)
|
||||
|
||||
if AUTH_ALLOW_ANY_LOGIN:
|
||||
# Dev bypass: any password gets in. The username still picks the
|
||||
# account, so `admin` lands on the admin pages and `user` on the user
|
||||
# ones; anything else is an unconfigured name and gets the lower of the
|
||||
# two roles rather than silently minting an admin. Throttling is skipped
|
||||
# because there is no longer a password to guess.
|
||||
account = _accounts().get(username) or {
|
||||
"role": "user",
|
||||
"display_name": payload.username.strip() or username,
|
||||
"email": f"{username}@nutritionintel.com",
|
||||
}
|
||||
logger.warning(
|
||||
"AUTH_ALLOW_ANY_LOGIN: signing in %r as %s without checking the password",
|
||||
username,
|
||||
account["role"],
|
||||
)
|
||||
else:
|
||||
_check_not_locked(key)
|
||||
|
||||
account = _accounts().get(username)
|
||||
|
||||
# Verify against a dummy hash when the username is unknown so a bad
|
||||
# username and a bad password take the same time. Otherwise the response
|
||||
# latency alone enumerates valid usernames.
|
||||
stored_hash = account["password_hash"] if account else _DUMMY_HASH
|
||||
|
||||
# A hash that does not parse can never match, and verify_password bails
|
||||
# out of one before doing any PBKDF2 work - measured here, 0.16ms against
|
||||
# 439ms for a real digest. That inverts the very property _DUMMY_HASH
|
||||
# exists to protect: an account whose configured hash is corrupt would
|
||||
# answer ~2700x faster than every other username, announcing which
|
||||
# account is broken to anyone with a stopwatch. So spend the same work
|
||||
# regardless; the result is a rejection either way.
|
||||
hash_usable = hash_is_wellformed(stored_hash)
|
||||
password_ok = verify_password(
|
||||
payload.password, stored_hash if hash_usable else _DUMMY_HASH
|
||||
)
|
||||
|
||||
if account is None or not password_ok:
|
||||
_record_failure(key)
|
||||
# The reason goes to the LOG, never to the caller - the response
|
||||
# below is byte-identical whichever of these it was, so nothing here
|
||||
# can be used to enumerate usernames. It is computed after both the
|
||||
# lookup and the PBKDF2 call above, so it adds no timing signal
|
||||
# either. Without it, a deployment whose configured hash or admin
|
||||
# username has drifted is indistinguishable from someone simply
|
||||
# typing the wrong password, and this is exactly how a production
|
||||
# sign-in outage stayed unexplained: the log said "Failed sign-in
|
||||
# for 'admin'" and nothing more.
|
||||
if account is None:
|
||||
logger.warning(
|
||||
"Failed sign-in for %r from %s: reason=unknown-username. "
|
||||
"Configured accounts: %s (AUTH_ADMIN_USERNAME source=%s).",
|
||||
username,
|
||||
key[1],
|
||||
", ".join(sorted(_accounts())),
|
||||
config_source("AUTH_ADMIN_USERNAME"),
|
||||
)
|
||||
elif not hash_usable:
|
||||
# ERROR, not WARNING: this is a broken deployment, not a bad
|
||||
# guess. No password can ever match, so every sign-in to this
|
||||
# account will 401 until the hash itself is replaced.
|
||||
logger.error(
|
||||
"Failed sign-in for %r from %s: reason=malformed-hash. The configured "
|
||||
"password hash does not parse as pbkdf2_sha256$<iterations>$<b64 salt>$"
|
||||
"<b64 digest> (fingerprint=%s, source=%s). Nobody can sign in to this "
|
||||
"account until it is regenerated with scripts/make_auth_secrets.py.",
|
||||
username,
|
||||
key[1],
|
||||
password_hash_fingerprint(stored_hash) or "(empty)",
|
||||
config_source("AUTH_ADMIN_PASSWORD_HASH"),
|
||||
)
|
||||
else:
|
||||
logger.warning(
|
||||
"Failed sign-in for %r from %s: reason=bad-password. The account exists "
|
||||
"and its hash parses (fingerprint=%s, source=%s); the password did not "
|
||||
"match. If this IS the password you deployed, then the running config "
|
||||
"carries a different hash than the file you are reading - compare that "
|
||||
"fingerprint against: python scripts/make_auth_secrets.py "
|
||||
"--fingerprint .env.production",
|
||||
username,
|
||||
key[1],
|
||||
password_hash_fingerprint(stored_hash),
|
||||
config_source("AUTH_ADMIN_PASSWORD_HASH"),
|
||||
)
|
||||
# One message for every failure mode, for the same reason.
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_401_UNAUTHORIZED,
|
||||
detail="Invalid username or password.",
|
||||
)
|
||||
|
||||
_clear_failures(key)
|
||||
role = account["role"]
|
||||
permissions = ROLE_PERMISSIONS.get(role, [])
|
||||
token, expires_in = create_access_token(username, role, permissions)
|
||||
logger.info("Issued token for %r (role=%s)", username, role)
|
||||
|
||||
return LoginResponse(
|
||||
access_token=token,
|
||||
expires_in=expires_in,
|
||||
user=UserProfile(
|
||||
username=username,
|
||||
role=role,
|
||||
display_name=account["display_name"],
|
||||
email=account["email"],
|
||||
permissions=permissions,
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
@router.get("/me", response_model=UserProfile)
|
||||
def me(principal: Principal = Depends(get_principal)) -> UserProfile:
|
||||
"""
|
||||
Who the presented credential belongs to. 401 if it is missing or expired.
|
||||
|
||||
The frontend calls this on boot to check a restored session before showing
|
||||
the app, so an expired token lands on the login page rather than on a
|
||||
dashboard whose every request then fails.
|
||||
"""
|
||||
account = _accounts().get(principal.username, {})
|
||||
return UserProfile(
|
||||
username=principal.username,
|
||||
role=principal.role,
|
||||
display_name=account.get("display_name", principal.username.title()),
|
||||
email=account.get("email", f"{principal.username}@nutritionintel.com"),
|
||||
permissions=principal.permissions,
|
||||
)
|
||||
|
||||
|
||||
@router.get("/roles")
|
||||
def list_roles() -> dict:
|
||||
"""
|
||||
The available roles and what each may do.
|
||||
|
||||
Note there are no demo credentials here any more. The passwords are set per
|
||||
deployment via AUTH_ADMIN_PASSWORD_HASH / AUTH_USER_PASSWORD_HASH; this
|
||||
endpoint used to publish working ones to anyone who asked.
|
||||
"""
|
||||
return {
|
||||
"roles": [
|
||||
{
|
||||
"id": "admin",
|
||||
"name": "Admin",
|
||||
"description": (
|
||||
"Full access: catalog brand cards, project details, Excel/CSV "
|
||||
"train/test uploads, discount allocation, analytics and nutrition. "
|
||||
"Implicitly holds every permission."
|
||||
),
|
||||
"permissions": ROLE_PERMISSIONS["admin"],
|
||||
},
|
||||
{
|
||||
"id": "user",
|
||||
"name": "User",
|
||||
"description": (
|
||||
"Combined user and store role: single or batch product uploads with "
|
||||
"image and DB/JSON sync, store inventory, profit analytics, nutrition."
|
||||
),
|
||||
"permissions": ROLE_PERMISSIONS["user"],
|
||||
},
|
||||
]
|
||||
}
|
||||
188
backend/app/api/routers/elec.py
Normal file
188
backend/app/api/routers/elec.py
Normal file
@@ -0,0 +1,188 @@
|
||||
"""Read-only catalogue API: category -> brand -> product -> per-platform offers.
|
||||
|
||||
Only VERIFIED products are served (see repository.refresh_verification), and
|
||||
every price is returned with the site, URL, source type and time it was seen.
|
||||
Money is returned as a decimal string, never a float.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from decimal import Decimal
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from fastapi import APIRouter, HTTPException, Query
|
||||
|
||||
from app.electronics.db.connection import connect
|
||||
from app.electronics.db.repository import product_rating_and_reviews
|
||||
from app.electronics.reviews import select_reviews
|
||||
|
||||
router = APIRouter(prefix="/elec", tags=["electronics"])
|
||||
|
||||
|
||||
def _money(value: Optional[Decimal]) -> Optional[str]:
|
||||
return None if value is None else format(value, "f")
|
||||
|
||||
|
||||
def _clean(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
out = {}
|
||||
for k, v in row.items():
|
||||
if isinstance(v, Decimal):
|
||||
out[k] = _money(v) if k in ("price", "mrp", "best_price", "min_price", "max_price") else float(v)
|
||||
elif hasattr(v, "isoformat"):
|
||||
out[k] = v.isoformat()
|
||||
else:
|
||||
out[k] = v
|
||||
return out
|
||||
|
||||
|
||||
@router.get("/categories")
|
||||
def categories() -> List[dict]:
|
||||
with connect() as conn:
|
||||
rows = conn.execute(
|
||||
"SELECT c.slug, c.name, coalesce(sum(s.product_count), 0)::int AS product_count "
|
||||
"FROM elec.category c LEFT JOIN elec.v_brand_summary s ON s.category = c.slug "
|
||||
"GROUP BY c.slug, c.name ORDER BY c.name"
|
||||
).fetchall()
|
||||
return [_clean(r) for r in rows]
|
||||
|
||||
|
||||
@router.get("/brands")
|
||||
def brands(category: str = Query(...)) -> List[dict]:
|
||||
with connect() as conn:
|
||||
rows = conn.execute(
|
||||
"""
|
||||
SELECT b.name AS brand, b.slug AS brand_slug,
|
||||
coalesce(s.product_count, 0)::int AS product_count, s.min_price, s.max_price,
|
||||
(SELECT image_url FROM elec.v_brand_catalog v
|
||||
WHERE v.brand_slug = b.slug AND v.category = %(c)s AND v.image_url IS NOT NULL
|
||||
ORDER BY v.platform_count DESC LIMIT 1) AS sample_image
|
||||
FROM elec.brand b
|
||||
JOIN elec.brand_category bc ON bc.brand_id = b.id
|
||||
JOIN elec.category c ON c.id = bc.category_id AND c.slug = %(c)s
|
||||
LEFT JOIN elec.v_brand_summary s ON s.brand_slug = b.slug AND s.category = %(c)s
|
||||
ORDER BY coalesce(s.product_count, 0) DESC, b.name
|
||||
""",
|
||||
{"c": category},
|
||||
).fetchall()
|
||||
return [_clean(r) for r in rows]
|
||||
|
||||
|
||||
@router.get("/products")
|
||||
def products(
|
||||
category: Optional[str] = None,
|
||||
brand: Optional[str] = None,
|
||||
q: Optional[str] = Query(None, max_length=100),
|
||||
min_price: Optional[Decimal] = None,
|
||||
max_price: Optional[Decimal] = None,
|
||||
in_stock: bool = False,
|
||||
site: Optional[str] = None,
|
||||
tn_only: bool = False,
|
||||
limit: int = Query(48, ge=1, le=200),
|
||||
offset: int = Query(0, ge=0),
|
||||
) -> dict:
|
||||
where, params = ["TRUE"], {}
|
||||
if category:
|
||||
where.append("v.category = %(category)s"); params["category"] = category
|
||||
if brand:
|
||||
where.append("v.brand_slug = %(brand)s"); params["brand"] = brand
|
||||
if q:
|
||||
where.append("(v.display_name ILIKE %(q)s OR v.brand ILIKE %(q)s)"); params["q"] = f"%{q}%"
|
||||
if min_price is not None:
|
||||
where.append("v.best_price >= %(min_price)s"); params["min_price"] = min_price
|
||||
if max_price is not None:
|
||||
where.append("v.best_price <= %(max_price)s"); params["max_price"] = max_price
|
||||
if in_stock:
|
||||
where.append("EXISTS (SELECT 1 FROM elec.v_product_availability a WHERE a.product_id = v.product_id AND a.in_stock)")
|
||||
if site:
|
||||
where.append("EXISTS (SELECT 1 FROM elec.v_product_availability a WHERE a.product_id = v.product_id AND a.domain = %(site)s)")
|
||||
params["site"] = site
|
||||
if tn_only:
|
||||
where.append("v.sold_by_tn_retailer")
|
||||
sql_where = " AND ".join(where)
|
||||
with connect() as conn:
|
||||
total = conn.execute(f"SELECT count(*) AS n FROM elec.v_brand_catalog v WHERE {sql_where}", params).fetchone()["n"]
|
||||
rows = conn.execute(
|
||||
f"SELECT v.* FROM elec.v_brand_catalog v WHERE {sql_where} "
|
||||
f"ORDER BY v.platform_count DESC, v.best_price NULLS LAST, v.display_name "
|
||||
f"LIMIT %(limit)s OFFSET %(offset)s",
|
||||
{**params, "limit": limit, "offset": offset},
|
||||
).fetchall()
|
||||
return {"total": total, "products": [_clean(r) for r in rows]}
|
||||
|
||||
|
||||
@router.get("/products/{product_id}")
|
||||
def product(product_id: int) -> dict:
|
||||
with connect() as conn:
|
||||
row = conn.execute("SELECT * FROM elec.v_brand_catalog WHERE product_id = %s", (product_id,)).fetchone()
|
||||
if not row:
|
||||
raise HTTPException(status_code=404, detail="Product not found or not verified")
|
||||
specs = conn.execute("SELECT spec_sources FROM elec.product WHERE id = %s", (product_id,)).fetchone()
|
||||
offers = conn.execute(
|
||||
"SELECT * FROM elec.v_product_availability WHERE product_id = %s "
|
||||
"ORDER BY (price IS NULL), (source_type = 'search_snippet'), price, site",
|
||||
(product_id,),
|
||||
).fetchall()
|
||||
images = conn.execute(
|
||||
"SELECT i.url, i.source_type, s.name AS site, l.source_url AS found_on "
|
||||
"FROM elec.product_image i JOIN elec.source_listing l ON l.id = i.source_listing_id "
|
||||
"JOIN elec.site s ON s.id = l.site_id WHERE i.product_id = %s ORDER BY i.rank, i.id",
|
||||
(product_id,),
|
||||
).fetchall()
|
||||
rated = product_rating_and_reviews(conn, product_id)
|
||||
result = _clean(row)
|
||||
result["spec_sources"] = specs["spec_sources"] if specs else {}
|
||||
result["offers"] = [_clean(o) for o in offers]
|
||||
result["images"] = [dict(i) for i in images]
|
||||
result["rating"] = _overall_rating(rated["sources"])
|
||||
overall = result["rating"]["value"] if result["rating"] else None
|
||||
result["reviews"] = [_clean(r) for r in select_reviews(overall, rated["reviews"])]
|
||||
return result
|
||||
|
||||
|
||||
def _overall_rating(sources: List[dict]) -> Optional[dict]:
|
||||
"""The product's rating across the platforms that state one: the mean
|
||||
weighted by each platform's rating count (a platform that states no count
|
||||
weighs as 1). None when no platform states a rating - never a guess."""
|
||||
if not sources:
|
||||
return None
|
||||
weight = lambda s: max(int(s["review_count"] or 0), 1) # noqa: E731
|
||||
total = sum(weight(s) for s in sources)
|
||||
value = sum(Decimal(s["rating"]) * weight(s) for s in sources) / total
|
||||
counts = [s["review_count"] for s in sources if s["review_count"]]
|
||||
return {
|
||||
"value": round(float(value), 1),
|
||||
"count": sum(counts) if counts else None,
|
||||
"sources": [
|
||||
{"site": s["site"], "rating": float(s["rating"]), "review_count": s["review_count"], "source_url": s["source_url"]}
|
||||
for s in sources
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
@router.get("/products/{product_id}/price-history")
|
||||
def price_history(product_id: int) -> List[dict]:
|
||||
with connect() as conn:
|
||||
rows = conn.execute(
|
||||
"""
|
||||
SELECT s.name AS site, h.price, h.mrp, h.in_stock, h.source_type, h.observed_at
|
||||
FROM elec.price_history h
|
||||
JOIN elec.product_listing_map m ON m.listing_id = h.listing_id AND m.review_status IN ('auto','approved')
|
||||
JOIN elec.source_listing l ON l.id = h.listing_id
|
||||
JOIN elec.site s ON s.id = l.site_id
|
||||
WHERE m.product_id = %s AND h.price IS NOT NULL
|
||||
ORDER BY h.observed_at
|
||||
""",
|
||||
(product_id,),
|
||||
).fetchall()
|
||||
return [_clean(r) for r in rows]
|
||||
|
||||
|
||||
@router.get("/sites")
|
||||
def sites() -> List[dict]:
|
||||
with connect() as conn:
|
||||
rows = conn.execute(
|
||||
"SELECT s.name, s.domain, s.kind, s.region, s.policy, s.probe_outcome, s.probed_at, "
|
||||
"s.breaker_until, s.breaker_reason, s.probe_evidence->>'reason' AS probe_reason, "
|
||||
"(SELECT count(*) FROM elec.source_listing l WHERE l.site_id = s.id)::int AS listings "
|
||||
"FROM elec.site s ORDER BY (s.kind = 'brand_official'), s.name"
|
||||
).fetchall()
|
||||
return [_clean(r) for r in rows]
|
||||
124
backend/app/api/routers/elec_admin.py
Normal file
124
backend/app/api/routers/elec_admin.py
Normal file
@@ -0,0 +1,124 @@
|
||||
"""Admin endpoints: start a collection run, see runs, probe sites, review matches."""
|
||||
from __future__ import annotations
|
||||
|
||||
import threading
|
||||
import time
|
||||
import uuid
|
||||
from typing import Dict, List, Optional
|
||||
|
||||
from fastapi import APIRouter, Depends, HTTPException
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
from app.api.background import run_in_background
|
||||
from app.api.deps import require_admin
|
||||
from app.electronics.db import repository as repo
|
||||
from app.electronics.reference import load_reference
|
||||
|
||||
router = APIRouter(prefix="/elec/admin", tags=["electronics-admin"], dependencies=[Depends(require_admin)])
|
||||
|
||||
_jobs: Dict[str, dict] = {}
|
||||
_jobs_lock = threading.Lock()
|
||||
_run_lock = threading.Lock() # one collection at a time: polite to sites, kind to 8 GB RAM
|
||||
|
||||
|
||||
class RunRequest(BaseModel):
|
||||
category: str = Field(..., examples=["mobiles"])
|
||||
brands: List[str] = Field(default_factory=list, description="brand slugs; empty = all for the category")
|
||||
limit: int = Field(10, ge=1, le=60, description="max models per brand")
|
||||
expand: int = Field(6, ge=0, le=30)
|
||||
budget: int = Field(150, ge=10, le=1000, description="max search queries")
|
||||
fetch_pages: bool = True
|
||||
use_llm: bool = True
|
||||
|
||||
|
||||
def _job_update(job_id: str, **fields) -> None:
|
||||
with _jobs_lock:
|
||||
_jobs[job_id].update(fields, updated_at=time.time())
|
||||
|
||||
|
||||
@router.post("/runs", status_code=202)
|
||||
def start_run(req: RunRequest) -> dict:
|
||||
ref = load_reference()
|
||||
if req.category not in ref.categories:
|
||||
raise HTTPException(400, f"unknown category {req.category!r}")
|
||||
brands = req.brands or [b.slug for b in ref.brands_for(req.category)]
|
||||
bad = [b for b in brands if b not in ref.brands or req.category not in ref.brands[b].categories]
|
||||
if bad:
|
||||
raise HTTPException(400, f"not allow-listed for {req.category}: {bad}")
|
||||
if _run_lock.locked():
|
||||
raise HTTPException(409, "A collection run is already in progress")
|
||||
job_id = str(uuid.uuid4())
|
||||
with _jobs_lock:
|
||||
_jobs[job_id] = {"job_id": job_id, "status": "queued", "log": [], "stats": {}, "created_at": time.time()}
|
||||
|
||||
def work() -> None:
|
||||
from app.electronics.collector import Collector, RunOptions
|
||||
|
||||
with _run_lock:
|
||||
_job_update(job_id, status="running")
|
||||
|
||||
def progress(msg: str) -> None:
|
||||
with _jobs_lock:
|
||||
_jobs[job_id]["log"] = (_jobs[job_id]["log"] + [msg])[-200:]
|
||||
|
||||
try:
|
||||
opts = RunOptions(category=req.category, brands=brands, max_products_per_brand=req.limit,
|
||||
expand_per_brand=req.expand, search_budget=req.budget,
|
||||
fetch_pages=req.fetch_pages, use_llm=req.use_llm)
|
||||
stats = Collector(opts, progress=progress).run()
|
||||
_job_update(job_id, status="done", stats=stats)
|
||||
except Exception as exc: # noqa: BLE001 - reported to the UI
|
||||
_job_update(job_id, status="failed", error=repr(exc))
|
||||
|
||||
run_in_background(work, name=f"elec-run-{job_id[:8]}")
|
||||
return {"job_id": job_id}
|
||||
|
||||
|
||||
@router.get("/runs/{job_id}")
|
||||
def get_job(job_id: str) -> dict:
|
||||
with _jobs_lock:
|
||||
job = _jobs.get(job_id)
|
||||
if not job:
|
||||
raise HTTPException(404, "unknown job")
|
||||
return dict(job)
|
||||
|
||||
|
||||
@router.get("/runs")
|
||||
def list_runs(limit: int = 20) -> List[dict]:
|
||||
rows = repo.recent_runs(limit)
|
||||
for r in rows:
|
||||
for k in ("started_at", "ended_at"):
|
||||
if r.get(k):
|
||||
r[k] = r[k].isoformat()
|
||||
return rows
|
||||
|
||||
|
||||
@router.get("/review")
|
||||
def review_queue() -> List[dict]:
|
||||
return [{**r, "confidence": float(r["confidence"])} for r in repo.review_queue()]
|
||||
|
||||
|
||||
class ReviewDecision(BaseModel):
|
||||
approve: bool
|
||||
|
||||
|
||||
@router.post("/review/{listing_id}")
|
||||
def review(listing_id: int, decision: ReviewDecision) -> dict:
|
||||
if not repo.set_review(listing_id, decision.approve):
|
||||
raise HTTPException(404, "no pending match for that listing")
|
||||
return {"ok": True, "products": repo.refresh_verification()}
|
||||
|
||||
|
||||
@router.post("/sites/{domain}/probe")
|
||||
def probe(domain: str) -> dict:
|
||||
from app.electronics.net.polite_client import PoliteClient
|
||||
from app.electronics.probe.site_probe import probe_site
|
||||
from app.electronics.search.engine import SearchEngine
|
||||
|
||||
site = load_reference().sites.get(domain)
|
||||
if not site:
|
||||
raise HTTPException(404, "unknown site")
|
||||
with PoliteClient() as client:
|
||||
res = probe_site(site, client, SearchEngine(budget=4))
|
||||
repo.set_probe_result(domain, res["outcome"], res["robots_allowed"], res["evidence"])
|
||||
return res
|
||||
36
backend/app/api/routers/health.py
Normal file
36
backend/app/api/routers/health.py
Normal file
@@ -0,0 +1,36 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
from fastapi import APIRouter
|
||||
|
||||
from app.api.schemas import AuthConfigOut, HealthOut, SearchStatusOut
|
||||
from app.electronics.db.connection import check_connection
|
||||
from app.infrastructure.security import auth_config_summary
|
||||
from app.infrastructure.settings import (
|
||||
DB_NAME, EMBEDDINGS_MODEL, OLLAMA_MODEL_NAME, USE_DDG_SEARCH, USE_GOOGLE_CSE,
|
||||
)
|
||||
from app.services import ollama_service
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
router = APIRouter(tags=["health"])
|
||||
|
||||
|
||||
@router.get("/health", response_model=HealthOut)
|
||||
def health() -> HealthOut:
|
||||
"""Liveness/readiness probe used by the React app to show a banner when
|
||||
Postgres or Ollama aren't reachable, instead of failing silently.
|
||||
|
||||
Ollama being down does not make the service degraded: it is only used to
|
||||
fill spec gaps, and the pipeline runs deterministically without it."""
|
||||
db_ok = check_connection()
|
||||
return HealthOut(
|
||||
status="ok" if db_ok else "degraded",
|
||||
database=db_ok,
|
||||
database_name=DB_NAME,
|
||||
ollama=bool(ollama_service._ensure_client()),
|
||||
ollama_model=OLLAMA_MODEL_NAME,
|
||||
embeddings_model=EMBEDDINGS_MODEL,
|
||||
search=SearchStatusOut(ddg=USE_DDG_SEARCH, google_cse=USE_GOOGLE_CSE),
|
||||
auth=AuthConfigOut(**auth_config_summary()),
|
||||
)
|
||||
73
backend/app/api/schemas.py
Normal file
73
backend/app/api/schemas.py
Normal file
@@ -0,0 +1,73 @@
|
||||
"""Pydantic response models for the health endpoint. The electronics
|
||||
catalogue's own models live in app/electronics/api_models.py."""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import List, Optional
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
|
||||
class ApiKeyInfoOut(BaseModel):
|
||||
"""One configured machine consumer, named but never quoted.
|
||||
|
||||
`fingerprint` is a truncated digest of name+secret, not the secret. It exists
|
||||
so a caller who was issued a key can confirm THAT key is the one this
|
||||
deployment loaded - the question a 401 cannot answer, since an undeployed key
|
||||
and a wrong key fail identically.
|
||||
"""
|
||||
|
||||
name: str
|
||||
role: str
|
||||
fingerprint: str
|
||||
|
||||
|
||||
class AuthConfigOut(BaseModel):
|
||||
"""
|
||||
The effective auth configuration, reported by /api/health.
|
||||
|
||||
Unauthenticated on purpose. The failure this exists to diagnose is "nobody
|
||||
can sign in", so anything gated behind an admin token is unreachable
|
||||
exactly when it is needed. Nothing here is a secret: the admin username is
|
||||
already the documented one, allow_any_login=true is a fact an operator
|
||||
urgently needs (and an attacker discovers with a single login attempt
|
||||
anyway), and the fingerprint is a truncated hash of a salted digest, not a
|
||||
password. The API key block follows the same rule: it names which consumers
|
||||
are configured and fingerprints their keys, so a caller can tell an
|
||||
undeployed key from a rejected one, but it never renders a secret. What it buys is a one-command answer to "is this deployment
|
||||
running the config I think it is?" - compare the fingerprint here against
|
||||
the one printed by scripts/make_auth_secrets.py --fingerprint.
|
||||
"""
|
||||
|
||||
enabled: bool
|
||||
allow_any_login: bool
|
||||
admin_username: str
|
||||
password_hash_valid: bool
|
||||
password_hash_iterations: Optional[int] = None
|
||||
password_hash_fingerprint: str
|
||||
# "process-env" | "env-file" | "default" - which one actually won.
|
||||
admin_username_source: str
|
||||
password_hash_source: str
|
||||
# Machine consumers. Names and fingerprints only - the secrets themselves are
|
||||
# never rendered here, and _parse_api_keys enforces enough entropy that the
|
||||
# fingerprints do not give them away. Defaulted so a client of this schema
|
||||
# still validates against a deployment predating these fields.
|
||||
api_keys_count: int = 0
|
||||
api_keys: List[ApiKeyInfoOut] = Field(default_factory=list)
|
||||
api_keys_source: str = "default"
|
||||
|
||||
|
||||
class SearchStatusOut(BaseModel):
|
||||
"""Which web-search providers discovery can use right now."""
|
||||
ddg: bool
|
||||
google_cse: bool
|
||||
|
||||
|
||||
class HealthOut(BaseModel):
|
||||
status: str
|
||||
database: bool
|
||||
database_name: str
|
||||
ollama: bool
|
||||
ollama_model: str
|
||||
embeddings_model: str
|
||||
search: SearchStatusOut
|
||||
auth: AuthConfigOut
|
||||
0
backend/app/electronics/__init__.py
Normal file
0
backend/app/electronics/__init__.py
Normal file
261
backend/app/electronics/cli.py
Normal file
261
backend/app/electronics/cli.py
Normal file
@@ -0,0 +1,261 @@
|
||||
"""Command line for the electronics pipeline. Run from backend/:
|
||||
|
||||
python -m app.electronics.cli migrate
|
||||
python -m app.electronics.cli seed-reference
|
||||
python -m app.electronics.cli probe [--site croma.com] [--all]
|
||||
python -m app.electronics.cli collect --category mobiles --brand samsung --brand xiaomi --limit 15
|
||||
python -m app.electronics.cli reviews [--category mobiles]
|
||||
python -m app.electronics.cli report
|
||||
python -m app.electronics.cli review [--approve ID | --reject ID]
|
||||
python -m app.electronics.cli verify-grounding
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
from decimal import Decimal
|
||||
from typing import List, Optional
|
||||
|
||||
import typer
|
||||
|
||||
app = typer.Typer(add_completion=False, help="Electronics catalogue: search-first, evidence-backed collection.")
|
||||
|
||||
|
||||
def _setup_logging(verbose: bool) -> None:
|
||||
logging.basicConfig(level=logging.DEBUG if verbose else logging.INFO,
|
||||
format="%(asctime)s %(levelname)s %(name)s: %(message)s")
|
||||
for noisy in ("httpx", "httpcore", "primp", "ddgs", "urllib3", "sentence_transformers"):
|
||||
logging.getLogger(noisy).setLevel(logging.WARNING)
|
||||
|
||||
|
||||
@app.command()
|
||||
def migrate() -> None:
|
||||
"""Create/upgrade the elec schema in the local electronics_catalog database."""
|
||||
from app.electronics.db.migrate import run_migrations
|
||||
|
||||
applied = run_migrations()
|
||||
typer.echo(f"Applied: {', '.join(applied) if applied else 'nothing (up to date)'}")
|
||||
|
||||
|
||||
@app.command("seed-reference")
|
||||
def seed_reference() -> None:
|
||||
"""Load brands, aliases, categories and sites from reference/*.yaml."""
|
||||
from app.electronics.db import repository as repo
|
||||
from app.electronics.reference import load_reference
|
||||
|
||||
typer.echo(json.dumps(repo.seed_reference(load_reference())))
|
||||
|
||||
|
||||
@app.command()
|
||||
def probe(site: List[str] = typer.Option([], "--site", help="Domain(s) to probe; default all probe-policy sites"),
|
||||
include_official: bool = typer.Option(False, "--official", help="Also probe brand official sites"),
|
||||
verbose: bool = False) -> None:
|
||||
"""Grade sites A/B/C: may they be scraped, or only searched?"""
|
||||
_setup_logging(verbose)
|
||||
from app.electronics.db import repository as repo
|
||||
from app.electronics.net.polite_client import PoliteClient
|
||||
from app.electronics.probe.site_probe import probe_site
|
||||
from app.electronics.reference import load_reference
|
||||
from app.electronics.search.engine import SearchEngine
|
||||
|
||||
ref = load_reference()
|
||||
if site:
|
||||
targets = [s for s in ref.sites.values() if s.domain in site]
|
||||
else:
|
||||
targets = [s for s in ref.sites.values() if include_official or s.kind != "brand_official"]
|
||||
run_id = repo.start_run("probe", {"sites": [s.domain for s in targets]})
|
||||
client = PoliteClient(on_fetch=lambda r, host: repo.log_fetch(run_id, r.url, host, r.status, r.bytes,
|
||||
r.outcome, r.robots_allowed))
|
||||
engine = SearchEngine(budget=len(targets) * 3)
|
||||
results = {}
|
||||
try:
|
||||
for s in targets:
|
||||
res = probe_site(s, client, engine)
|
||||
repo.set_probe_result(s.domain, res["outcome"], res["robots_allowed"], res["evidence"])
|
||||
results[s.domain] = res["outcome"]
|
||||
typer.echo(f"{s.name:28} {s.domain:24} {res['outcome']} {res['evidence'].get('reason')}")
|
||||
finally:
|
||||
client.close()
|
||||
repo.finish_run(run_id, "done", {"grades": results})
|
||||
|
||||
|
||||
@app.command()
|
||||
def collect(category: str = typer.Option(..., help="mobiles | laptops"),
|
||||
brand: List[str] = typer.Option([], "--brand", help="Brand slug(s); default all brands of the category"),
|
||||
limit: int = typer.Option(15, help="Max models per brand"),
|
||||
expand: int = typer.Option(8, help="Models per brand looked up on other platforms"),
|
||||
budget: int = typer.Option(200, help="Max search queries this run"),
|
||||
no_fetch: bool = typer.Option(False, "--no-fetch", help="Search results only; fetch no pages"),
|
||||
no_llm: bool = typer.Option(False, "--no-llm", help="Deterministic spec parsing only"),
|
||||
no_embed: bool = typer.Option(False, "--no-embed"),
|
||||
reprobe: bool = False,
|
||||
verbose: bool = False) -> None:
|
||||
"""Discover and collect real listings for allow-listed brands."""
|
||||
_setup_logging(verbose)
|
||||
from app.electronics.collector import Collector, RunOptions
|
||||
from app.electronics.reference import load_reference
|
||||
|
||||
ref = load_reference()
|
||||
if category not in ref.categories:
|
||||
raise typer.BadParameter(f"unknown category {category!r}; use one of {list(ref.categories)}")
|
||||
brands = brand or [b.slug for b in ref.brands_for(category)]
|
||||
unknown = [b for b in brands if b not in ref.brands or category not in ref.brands[b].categories]
|
||||
if unknown:
|
||||
raise typer.BadParameter(f"not allow-listed for {category}: {unknown}")
|
||||
opts = RunOptions(category=category, brands=brands, max_products_per_brand=limit, expand_per_brand=expand,
|
||||
search_budget=budget, use_llm=not no_llm, fetch_pages=not no_fetch, reprobe=reprobe)
|
||||
stats = Collector(opts, progress=typer.echo).run(embed=not no_embed)
|
||||
typer.echo(json.dumps(stats, indent=2, sort_keys=True))
|
||||
|
||||
|
||||
@app.command()
|
||||
def prices(limit: int = typer.Option(40, help="Max Google queries (free tier: 100/day)"),
|
||||
category: Optional[str] = typer.Option(None, help="mobiles | laptops (default both)")) -> None:
|
||||
"""Fill missing prices on search-only platforms from Google's structured data (no site fetches)."""
|
||||
_setup_logging(False)
|
||||
from app.electronics.price_lookup import lookup_prices
|
||||
|
||||
stats = lookup_prices(limit=limit, category=category, progress=typer.echo)
|
||||
typer.echo(json.dumps(stats, indent=2, default=str))
|
||||
if stats.get("error"):
|
||||
typer.echo("\nGoogle search is not usable yet: " + str(stats["error"]))
|
||||
raise typer.Exit(code=1)
|
||||
|
||||
|
||||
@app.command()
|
||||
def rematch(category: str = typer.Option(..., help="mobiles | laptops"),
|
||||
no_embed: bool = typer.Option(False, "--no-embed")) -> None:
|
||||
"""Rebuild products from stored listings with the current matching rules (no network)."""
|
||||
_setup_logging(False)
|
||||
from app.electronics.match.rematch import rematch as run_rematch
|
||||
|
||||
stats = run_rematch(category)
|
||||
if not no_embed:
|
||||
from app.electronics.collector import embed_verified_products
|
||||
|
||||
try:
|
||||
stats["embedded"] = embed_verified_products()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
typer.echo(f"Embedding skipped: {exc}")
|
||||
typer.echo(json.dumps(stats, indent=2))
|
||||
|
||||
|
||||
@app.command()
|
||||
def reviews(category: Optional[str] = typer.Option(None, help="mobiles | laptops (default both)"),
|
||||
limit: int = typer.Option(200, help="Max product pages to re-read"),
|
||||
verbose: bool = False) -> None:
|
||||
"""Re-read ratings and customer reviews from the product pages already on file.
|
||||
|
||||
Only pages the collector itself reads (scraped / brand official listings)
|
||||
are fetched, politely (robots.txt, per-site pacing, circuit breaker). A
|
||||
rating or review is stored only when the page's own schema.org data states
|
||||
it; nothing is generated.
|
||||
"""
|
||||
_setup_logging(verbose)
|
||||
from rapidfuzz import fuzz
|
||||
|
||||
from app.electronics.db import repository as repo
|
||||
from app.electronics.extract.jsonld import extract_products
|
||||
from app.electronics.net.polite_client import PoliteClient
|
||||
|
||||
rows = repo.listings_for_review_backfill(category)[:limit]
|
||||
run_id = repo.start_run("reviews", {"category": category, "pages": len(rows)})
|
||||
client = PoliteClient(on_fetch=lambda r, host: repo.log_fetch(run_id, r.url, host, r.status, r.bytes,
|
||||
r.outcome, r.robots_allowed))
|
||||
stats = {"pages": 0, "pages_ok": 0, "rated": 0, "reviews_stored": 0, "no_matching_product": 0}
|
||||
status, error = "done", None
|
||||
try:
|
||||
for row in rows:
|
||||
stats["pages"] += 1
|
||||
res = client.get(row["source_url"])
|
||||
if not res.ok:
|
||||
continue
|
||||
stats["pages_ok"] += 1
|
||||
products = extract_products(res.text)
|
||||
# The same product the listing was stored from: its SKU, else its name.
|
||||
match = next((p for p in products if p.get("sku") and p["sku"] == row["source_sku"]), None)
|
||||
if match is None:
|
||||
title = (row["title"] or "").lower()
|
||||
scored = [(fuzz.token_set_ratio(p["name"].lower(), title), p) for p in products]
|
||||
scored = [sp for sp in scored if sp[0] >= 85]
|
||||
match = max(scored, key=lambda sp: sp[0])[1] if scored else None
|
||||
if match is None:
|
||||
stats["no_matching_product"] += 1
|
||||
continue
|
||||
if match.get("rating") is not None and Decimal(0) < match["rating"] <= Decimal(5):
|
||||
repo.update_listing_rating(row["listing_id"], match["rating"], match.get("review_count"))
|
||||
stats["rated"] += 1
|
||||
if match.get("reviews"):
|
||||
stats["reviews_stored"] += repo.save_reviews(row["listing_id"], match["reviews"])
|
||||
typer.echo(f" {row['domain']:22} rating={match.get('rating')} reviews={len(match.get('reviews') or [])}")
|
||||
except Exception as exc: # noqa: BLE001
|
||||
status, error = "failed", repr(exc)
|
||||
raise
|
||||
finally:
|
||||
client.close()
|
||||
repo.finish_run(run_id, status, stats, error)
|
||||
typer.echo(json.dumps(stats, indent=2))
|
||||
|
||||
|
||||
@app.command()
|
||||
def report() -> None:
|
||||
"""Counts per brand/category and per site."""
|
||||
from app.electronics.db.connection import connect
|
||||
|
||||
with connect() as conn:
|
||||
typer.echo("Sites:")
|
||||
for r in conn.execute("SELECT name, domain, policy, probe_outcome, breaker_until FROM elec.site "
|
||||
"WHERE kind <> 'brand_official' ORDER BY name"):
|
||||
typer.echo(f" {r['name']:22} {r['policy']:9} grade={r['probe_outcome'] or '-'}"
|
||||
f"{' breaker until ' + str(r['breaker_until']) if r['breaker_until'] else ''}")
|
||||
typer.echo("\nProducts by status:")
|
||||
for r in conn.execute("SELECT verification_status, count(*) n FROM elec.product GROUP BY 1"):
|
||||
typer.echo(f" {r['verification_status']:12} {r['n']}")
|
||||
typer.echo("\nVerified catalogue (brand / category):")
|
||||
for r in conn.execute("SELECT * FROM elec.v_brand_summary ORDER BY category, brand"):
|
||||
typer.echo(f" {r['brand']:10} {r['category']:8} products={r['product_count']:3} "
|
||||
f"price ₹{r['min_price']}–₹{r['max_price']} max_platforms={r['max_platforms']}")
|
||||
typer.echo("\nListings by site and source type:")
|
||||
for r in conn.execute("SELECT s.name, l.source_type, count(*) n, count(l.price) priced "
|
||||
"FROM elec.source_listing l JOIN elec.site s ON s.id = l.site_id "
|
||||
"GROUP BY 1, 2 ORDER BY 1, 2"):
|
||||
typer.echo(f" {r['name']:22} {r['source_type']:15} {r['n']:4} (with price: {r['priced']})")
|
||||
|
||||
|
||||
@app.command()
|
||||
def review(approve: Optional[int] = typer.Option(None, help="listing id to approve"),
|
||||
reject: Optional[int] = typer.Option(None, help="listing id to reject")) -> None:
|
||||
"""Show uncertain listing-to-product matches, or approve/reject one."""
|
||||
from app.electronics.db import repository as repo
|
||||
|
||||
if approve or reject:
|
||||
ok = repo.set_review(approve or reject, approve is not None)
|
||||
refreshed = repo.refresh_verification()
|
||||
typer.echo(f"{'updated' if ok else 'nothing pending for that listing'}; products: {refreshed}")
|
||||
return
|
||||
for r in repo.review_queue():
|
||||
typer.echo(f"[{r['listing_id']}] {r['site']}: {r['listing_title']}\n -> {r['product']} "
|
||||
f"({r['method']}, {r['confidence']}) {r['source_url']}")
|
||||
|
||||
|
||||
@app.command("verify-grounding")
|
||||
def verify_grounding(sample: int = 100) -> None:
|
||||
"""Audit: every stored price must appear in the evidence text stored with it."""
|
||||
from app.electronics.db import repository as repo
|
||||
|
||||
bad = 0
|
||||
rows = repo.grounding_sample(sample)
|
||||
for r in rows:
|
||||
digits = re.sub(r"\D", "", r["evidence_text"].replace(".00", ""))
|
||||
price = r["price"]
|
||||
whole = str(int(price)) if price == price.to_integral() else str(price)
|
||||
if whole.replace(".", "") not in digits:
|
||||
bad += 1
|
||||
typer.echo(f"NOT GROUNDED listing {r['id']}: price {price} not in evidence ({r['source_url']})")
|
||||
typer.echo(f"Checked {len(rows)} priced listings; {bad} without evidence.")
|
||||
raise typer.Exit(code=1 if bad else 0)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
app()
|
||||
581
backend/app/electronics/collector.py
Normal file
581
backend/app/electronics/collector.py
Normal file
@@ -0,0 +1,581 @@
|
||||
"""Search-first collection of real product listings.
|
||||
|
||||
For one category and a set of allow-listed brands:
|
||||
|
||||
1. DISCOVER web search `site:<platform> <brand> <category term>` on every
|
||||
registered platform (marketplaces, national chains, Tamil Nadu
|
||||
chains, the brand's own site). Only URLs that are single product
|
||||
pages on a registered platform are kept.
|
||||
2. EXPAND for each model found, search `<brand> <model> price` to find the
|
||||
same model on other platforms.
|
||||
3. COLLECT per URL, by the platform's probe grade:
|
||||
A/B and breaker closed -> fetch the page politely and read
|
||||
JSON-LD / meta / spec tables
|
||||
C (or fetch refused) -> use the search result itself: its
|
||||
title, snippet price and stock text
|
||||
4. MATCH link the listing to one canonical variant (match.matcher)
|
||||
5. ENRICH specs (deterministic, LLM gap-fill grounded in page text) and
|
||||
images (only from the product's own listings, validated live)
|
||||
6. VERIFY products with listings on ≥2 sites (≥1 a retailer) become
|
||||
verified and visible.
|
||||
|
||||
Nothing in this module invents a product, price or image: every value is read
|
||||
from a page or a search result, and stored with that URL and text.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from decimal import Decimal
|
||||
from typing import Callable, Dict, List, Optional, Tuple
|
||||
from urllib.parse import urlparse
|
||||
|
||||
from rapidfuzz import fuzz
|
||||
|
||||
from app.electronics.db import repository as repo
|
||||
from app.electronics.extract.html_fallback import extract_page, spec_tables, visible_text
|
||||
from app.electronics.extract.jsonld import extract_products
|
||||
from app.electronics.extract.serp_parser import clean_result_title, read_price, read_rating, read_stock
|
||||
from app.electronics.match.matcher import decide
|
||||
from app.electronics.models import Listing
|
||||
from app.electronics.net.breaker import CircuitBreaker
|
||||
from app.electronics.net.polite_client import PoliteClient
|
||||
from app.electronics.normalise.brand_alias import looks_like_device_title
|
||||
from app.electronics.normalise.llm_fill import fill_missing
|
||||
from app.electronics.normalise.spec_normaliser import normalise_specs
|
||||
from app.electronics.normalise.title_parser import ParsedTitle, fill_from_context, parse_title, variant_key
|
||||
from app.electronics.probe.site_probe import probe_site
|
||||
from app.electronics.reference import BrandRef, SiteRef, load_reference, site_for_url
|
||||
from app.electronics.search.engine import SearchEngine
|
||||
from app.electronics.search.providers import SearchHit
|
||||
from app.infrastructure.settings import ELEC_PROBE_TTL_DAYS, MIN_IMAGE_BYTES
|
||||
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Titles that belong to another category even when the brand matches.
|
||||
_OFF_CATEGORY = {
|
||||
"mobiles": re.compile(r"\b(?:tab|tablet|pad|watch|buds|earbuds|laptop|book|monitor|tv|television|band)\b", re.I),
|
||||
"laptops": re.compile(r"\b(?:tablet|tab|monitor|mouse|keyboard|phone|smartphone|printer|desktop|all[- ]in[- ]one)\b", re.I),
|
||||
}
|
||||
_LISTING_PAGE = re.compile(r"/(?:search|s|c|category|categories|brand|brands|compare|offers?|deals?)(?:/|\?|$)", re.I)
|
||||
|
||||
|
||||
@dataclass
|
||||
class RunOptions:
|
||||
category: str
|
||||
brands: List[str] # brand slugs
|
||||
max_products_per_brand: int = 15
|
||||
expand_per_brand: int = 8 # models to look up on other platforms
|
||||
search_budget: int = 200
|
||||
use_llm: bool = True
|
||||
fetch_pages: bool = True
|
||||
find_images: bool = True
|
||||
reprobe: bool = False
|
||||
|
||||
|
||||
@dataclass
|
||||
class RunStats:
|
||||
counts: Dict[str, int] = field(default_factory=dict)
|
||||
|
||||
def inc(self, key: str, n: int = 1) -> None:
|
||||
self.counts[key] = self.counts.get(key, 0) + n
|
||||
|
||||
|
||||
def source_sku(site: SiteRef, url: str) -> str:
|
||||
rx = site.product_url_re
|
||||
if rx is not None:
|
||||
m = rx.search(url)
|
||||
if m and m.groups() and m.group(1):
|
||||
return m.group(1)
|
||||
p = urlparse(url)
|
||||
return (p.netloc.lower().removeprefix("www.") + p.path.rstrip("/").lower())[:300]
|
||||
|
||||
|
||||
_INDIA_PATH = re.compile(r"^/(?:in|in-en|en-in|en_in|in_en)(?:/|$)", re.I)
|
||||
|
||||
|
||||
def is_product_url(site: SiteRef, url: str) -> bool:
|
||||
p = urlparse(url)
|
||||
if _LISTING_PAGE.search(p.path):
|
||||
return False
|
||||
if site.kind == "brand_official":
|
||||
# Only the brand's Indian storefront: www.samsung.com/in/..., not
|
||||
# us.samsung.com or news.samsung.com. Domains that are Indian already
|
||||
# (oneplus.in, motorola.co.in) qualify as they are.
|
||||
host = (p.hostname or "").lower()
|
||||
if host not in (site.domain, "www." + site.domain, "in." + site.domain):
|
||||
return False
|
||||
if not site.domain.endswith((".in", ".co.in")) and not _INDIA_PATH.search(p.path) and not host.startswith("in."):
|
||||
return False
|
||||
rx = site.product_url_re
|
||||
if rx is not None:
|
||||
return bool(rx.search(url))
|
||||
return p.path.count("/") >= 2 # brand sites: at least /section/product
|
||||
|
||||
|
||||
def embed_verified_products() -> int:
|
||||
"""MiniLM vectors for verified products that do not have one yet."""
|
||||
rows = repo.products_without_embedding()
|
||||
if not rows:
|
||||
return 0
|
||||
from app.services.embeddings_service import embed_texts
|
||||
|
||||
texts = [
|
||||
f"{r['brand']} {r['display_name']} {r['category']} "
|
||||
+ " ".join(f"{k} {v}" for k, v in (r["canonical_specs"] or {}).items())
|
||||
for r in rows
|
||||
]
|
||||
for r, vec in zip(rows, embed_texts(texts)):
|
||||
repo.set_embedding(r["id"], vec)
|
||||
return len(rows)
|
||||
|
||||
|
||||
class Collector:
|
||||
def __init__(self, options: RunOptions, *, progress: Optional[Callable[[str], None]] = None) -> None:
|
||||
self.opt = options
|
||||
self.ref = load_reference()
|
||||
self.stats = RunStats()
|
||||
self.progress = progress or (lambda msg: logger.info(msg))
|
||||
self.run_id: Optional[int] = None
|
||||
self.ids = repo.id_maps()
|
||||
self.site_rows = {r["domain"]: r for r in repo.sites()}
|
||||
self.breaker = CircuitBreaker(on_trip=self._on_trip)
|
||||
for r in self.site_rows.values():
|
||||
if r.get("breaker_until"):
|
||||
self.breaker.preload(r["domain"], r["breaker_until"].timestamp(), r.get("breaker_reason") or "")
|
||||
self.client = PoliteClient(breaker=self.breaker, on_fetch=self._on_fetch)
|
||||
self.engine = SearchEngine(budget=options.search_budget)
|
||||
self._touched_products: Dict[int, List[Tuple[int, Listing]]] = {}
|
||||
|
||||
# -- callbacks -----------------------------------------------------------
|
||||
def _on_trip(self, host: str, reason: str, until: float) -> None:
|
||||
self.stats.inc("breaker_trips")
|
||||
self.progress(f"Circuit breaker opened for {host}: {reason}. Falling back to web search for it.")
|
||||
repo.trip_breaker(host, reason, until)
|
||||
|
||||
def _on_fetch(self, result, host: str) -> None:
|
||||
self.stats.inc(f"fetch_{result.outcome}")
|
||||
repo.log_fetch(self.run_id, result.url, host, result.status, result.bytes, result.outcome, result.robots_allowed)
|
||||
|
||||
# -- grading -------------------------------------------------------------
|
||||
def _grade(self, site: SiteRef) -> str:
|
||||
if site.policy == "serp_only":
|
||||
return "C"
|
||||
host = site.domain
|
||||
if self.breaker.is_open(host) or self.breaker.is_open("www." + host):
|
||||
return "C"
|
||||
return (self.site_rows.get(site.domain) or {}).get("probe_outcome") or "C"
|
||||
|
||||
def ensure_probes(self, sites: List[SiteRef]) -> None:
|
||||
from datetime import datetime, timedelta, timezone
|
||||
|
||||
stale_before = datetime.now(timezone.utc) - timedelta(days=ELEC_PROBE_TTL_DAYS)
|
||||
for site in sites:
|
||||
row = self.site_rows.get(site.domain) or {}
|
||||
if not self.opt.reprobe and row.get("probed_at") and row["probed_at"] > stale_before:
|
||||
continue
|
||||
self.progress(f"Probing {site.name} ({site.domain})")
|
||||
result = probe_site(site, self.client, self.engine)
|
||||
repo.set_probe_result(site.domain, result["outcome"], result["robots_allowed"], result["evidence"])
|
||||
row.update(probe_outcome=result["outcome"])
|
||||
self.site_rows[site.domain] = row
|
||||
self.stats.inc(f"probe_{result['outcome']}")
|
||||
self.progress(f" -> grade {result['outcome']}: {result['evidence'].get('reason')}")
|
||||
|
||||
# -- discovery -----------------------------------------------------------
|
||||
def _accept_hit(self, hit: SearchHit, brand: BrandRef) -> Optional[Tuple[SiteRef, ParsedTitle]]:
|
||||
site = site_for_url(hit.url)
|
||||
if site is None or not self._site_allowed_for(site, brand):
|
||||
return None
|
||||
if not is_product_url(site, hit.url):
|
||||
return None
|
||||
title = clean_result_title(hit.title)
|
||||
if not looks_like_device_title(title) or _OFF_CATEGORY[self.opt.category].search(title):
|
||||
return None
|
||||
parsed = parse_title(title, self.opt.category, expected_brand=brand.slug)
|
||||
if parsed.brand is None or not parsed.model_norm:
|
||||
return None
|
||||
fill_from_context(parsed, self.opt.category, snippet=hit.snippet or "")
|
||||
return site, parsed
|
||||
|
||||
def _site_allowed_for(self, site: SiteRef, brand: BrandRef) -> bool:
|
||||
if site.kind == "brand_official":
|
||||
return site.brand_slug == brand.slug
|
||||
return True
|
||||
|
||||
def _platforms_for(self, brand: BrandRef) -> List[SiteRef]:
|
||||
return [s for s in self.ref.sites.values()
|
||||
if s.kind != "brand_official" or s.brand_slug == brand.slug]
|
||||
|
||||
def _search_names(self, brand: BrandRef) -> List[str]:
|
||||
"""The brand, plus the sub-brands phones are actually sold under
|
||||
("Redmi", "POCO", "iQOO") - a search for "Xiaomi" alone misses most
|
||||
Redmi listings."""
|
||||
names = [brand.name]
|
||||
if self.opt.category == "mobiles":
|
||||
names += [s.upper() if len(s) <= 4 else s.title() for s in brand.sub_brands
|
||||
if s not in ("mi", "iphone", "pixel", "narzo")][:2]
|
||||
return names
|
||||
|
||||
def _collect_hits(self, hits: Optional[List[SearchHit]], brand: BrandRef, query: str,
|
||||
found: Dict, models: Dict[str, ParsedTitle]) -> int:
|
||||
new = 0
|
||||
for hit in hits or []:
|
||||
accepted = self._accept_hit(hit, brand)
|
||||
if not accepted:
|
||||
continue
|
||||
site_ref, parsed = accepted
|
||||
key = (site_ref.domain, source_sku(site_ref, hit.url))
|
||||
if key not in found:
|
||||
found[key] = (hit, site_ref, parsed, query)
|
||||
new += 1
|
||||
models.setdefault(parsed.model_norm, parsed)
|
||||
return new
|
||||
|
||||
def discover(self, brand: BrandRef) -> Dict[Tuple[str, str], Tuple[SearchHit, SiteRef, ParsedTitle, str]]:
|
||||
category = self.ref.categories[self.opt.category]
|
||||
found: Dict[Tuple[str, str], Tuple[SearchHit, SiteRef, ParsedTitle, str]] = {}
|
||||
models: Dict[str, ParsedTitle] = {}
|
||||
per_site_target = max(4, self.opt.max_products_per_brand)
|
||||
for site in self._platforms_for(brand):
|
||||
got = 0
|
||||
for name in self._search_names(brand):
|
||||
for terms in category.query_terms or category.search_terms:
|
||||
query = f"site:{site.domain} {name} {terms}"
|
||||
hits = self.engine.text(query, max_results=20)
|
||||
if hits is None:
|
||||
self.stats.inc("search_unavailable")
|
||||
continue
|
||||
got += self._collect_hits(hits, brand, query, found, models)
|
||||
if got >= per_site_target:
|
||||
break
|
||||
if got >= per_site_target:
|
||||
break
|
||||
self.progress(f"{brand.name}: {len(found)} listing URLs, {len(models)} models from platform searches")
|
||||
|
||||
# Cross-platform: look each variant up by name, to find the same product
|
||||
# on platforms the site: searches missed. Phones are grouped by model
|
||||
# line; laptops by full configuration (line + CPU + RAM + storage),
|
||||
# because one laptop line is sold in dozens of configurations and only
|
||||
# the exact one confirms a product.
|
||||
for p in self._expansion_targets(found):
|
||||
query = f"{brand.name} {p.model or p.model_norm} {self._variant_terms(p)} price"
|
||||
self._collect_hits(self.engine.text(re.sub(r"\s+", " ", query), max_results=20), brand, query, found, models)
|
||||
self.progress(f"{brand.name}: {len(found)} listing URLs after cross-platform search")
|
||||
return found
|
||||
|
||||
def _expansion_targets(self, found: Dict) -> List[ParsedTitle]:
|
||||
"""Which variants to look up on other platforms, most useful first:
|
||||
variants seen on the most sites, then ones whose page we can read with
|
||||
a price (grade A/B platforms), since one more site verifies those."""
|
||||
groups: Dict[str, Dict] = {}
|
||||
for (domain, _), (_, site, parsed, _) in found.items():
|
||||
if self.opt.category == "laptops":
|
||||
key = variant_key(parsed, "laptops")
|
||||
if key is None:
|
||||
continue
|
||||
else:
|
||||
key = parsed.model_norm
|
||||
g = groups.setdefault(key, {"parsed": parsed, "sites": set(), "readable": False})
|
||||
g["sites"].add(domain)
|
||||
g["readable"] = g["readable"] or self._grade(site) in ("A", "B")
|
||||
ranked = sorted(groups.values(), key=lambda g: (-len(g["sites"]), not g["readable"]))
|
||||
return [g["parsed"] for g in ranked[: self.opt.expand_per_brand]]
|
||||
|
||||
@staticmethod
|
||||
def _variant_terms(p: ParsedTitle) -> str:
|
||||
parts = []
|
||||
if p.processor:
|
||||
# "ryzen 5 7530u" -> "Ryzen 5 7530U", "i5-1334u" -> "i5-1334U"
|
||||
parts.append(" ".join(t.upper() if any(ch.isdigit() for ch in t) and len(t) > 2 else t.title()
|
||||
for t in p.processor.split()))
|
||||
if p.ram_gb:
|
||||
parts.append(f"{format(p.ram_gb.normalize(), 'f')}GB RAM")
|
||||
if p.storage_gb:
|
||||
parts.append(f"{format(p.storage_gb.normalize(), 'f')}GB")
|
||||
return " ".join(parts)
|
||||
|
||||
# -- listing construction --------------------------------------------------
|
||||
def _base_listing(self, site: SiteRef, url: str, parsed: ParsedTitle, title: str, source_type: str,
|
||||
evidence: str, confidence: float, parser: str, query: str) -> Listing:
|
||||
l = Listing(
|
||||
site_domain=site.domain, source_sku=source_sku(site, url), source_url=url, source_type=source_type,
|
||||
brand_slug=parsed.brand.brand_slug, category=self.opt.category, title=title,
|
||||
evidence_text=evidence, confidence=confidence, parser=parser, family=parsed.brand.family,
|
||||
model=parsed.model, model_number=parsed.mpn, ram_gb=parsed.ram_gb, storage_gb=parsed.storage_gb,
|
||||
colour=parsed.colour, search_query=query,
|
||||
)
|
||||
l.model_norm = parsed.model_norm
|
||||
l.processor = parsed.processor
|
||||
l.variant_key = variant_key(parsed, self.opt.category)
|
||||
return l
|
||||
|
||||
def listing_from_search(self, hit: SearchHit, site: SiteRef, parsed: ParsedTitle, query: str) -> Listing:
|
||||
title = clean_result_title(hit.title)
|
||||
evidence = f"{hit.title} — {hit.snippet}".strip(" —")
|
||||
reading = read_price(f"{hit.title} {hit.snippet}")
|
||||
price, mrp = reading.price, reading.mrp
|
||||
# A snippet that names a different RAM/storage than the title is about
|
||||
# another variant; its price cannot be trusted for this one.
|
||||
snippet_variant = parse_title(hit.snippet or "", self.opt.category)
|
||||
for a, b in ((snippet_variant.storage_gb, parsed.storage_gb), (snippet_variant.ram_gb, parsed.ram_gb)):
|
||||
if a is not None and b is not None and a != b:
|
||||
price = mrp = None
|
||||
self.stats.inc("snippet_price_variant_conflict")
|
||||
# Truncated titles ("... - (16 GB ...") lose the variant; the snippet
|
||||
# of the same result usually states it.
|
||||
fill_from_context(parsed, self.opt.category, snippet=hit.snippet or "")
|
||||
parser, confidence = f"serp:{hit.provider}", (0.55 if price is not None else 0.45)
|
||||
in_stock = read_stock(hit.snippet or "")
|
||||
availability = None if in_stock is None else ("InStock" if in_stock else "OutOfStock")
|
||||
# Structured offer the search engine read from the page itself (Google
|
||||
# pagemap). Better than snippet text, and still no request to the site.
|
||||
if hit.offer and price is None:
|
||||
try:
|
||||
offered = Decimal(str(hit.offer["price"]).replace(",", ""))
|
||||
except Exception: # noqa: BLE001
|
||||
offered = None
|
||||
if offered is not None and Decimal(500) <= offered <= Decimal(1000000):
|
||||
price, mrp = offered, None
|
||||
evidence = f"{evidence} || search-engine offer data: {json.dumps(hit.offer['raw'], default=str)[:600]}"
|
||||
parser, confidence = f"serp:{hit.provider}:pagemap", 0.65
|
||||
av = (hit.offer.get("availability") or "").lower().replace(" ", "")
|
||||
if "instock" in av:
|
||||
in_stock, availability = True, "InStock"
|
||||
elif "outofstock" in av or "soldout" in av:
|
||||
in_stock, availability = False, "OutOfStock"
|
||||
# Rating: the engine's structured data first (read from the page
|
||||
# itself), else an explicit "x out of 5" in this result's own text.
|
||||
rating, review_count = None, None
|
||||
if hit.rating and hit.rating.get("rating") is not None:
|
||||
rating = Decimal(str(hit.rating["rating"]))
|
||||
review_count = hit.rating.get("review_count")
|
||||
evidence = f"{evidence} || search-engine rating data: {json.dumps(hit.rating.get('raw'), default=str)[:300]}"
|
||||
else:
|
||||
stated = read_rating(f"{hit.title} {hit.snippet}")
|
||||
if stated.rating is not None:
|
||||
rating, review_count = stated.rating, stated.review_count
|
||||
listing = self._base_listing(site, hit.url, parsed, title, "search_snippet", evidence,
|
||||
confidence, parser, query)
|
||||
listing.price, listing.mrp = price, mrp
|
||||
listing.in_stock, listing.availability = in_stock, availability
|
||||
listing.rating, listing.review_count = rating, review_count
|
||||
return listing
|
||||
|
||||
def listing_from_page(self, hit: SearchHit, site: SiteRef, parsed_hit: ParsedTitle, query: str,
|
||||
html: str, final_url: str) -> Optional[Listing]:
|
||||
products = extract_products(html)
|
||||
page = extract_page(html)
|
||||
name = None
|
||||
product = None
|
||||
for p in products:
|
||||
pp = parse_title(p["name"], self.opt.category, expected_brand=parsed_hit.brand.brand_slug)
|
||||
if pp.brand and pp.model_norm and fuzz.token_set_ratio(pp.model_norm, parsed_hit.model_norm) >= 85:
|
||||
product, name = p, p["name"]
|
||||
break
|
||||
if name is None:
|
||||
name = page.get("name")
|
||||
if not name:
|
||||
return None
|
||||
parsed = parse_title(name, self.opt.category, expected_brand=parsed_hit.brand.brand_slug)
|
||||
if parsed.brand is None or not parsed.model_norm:
|
||||
return None
|
||||
if fuzz.token_set_ratio(parsed.model_norm, parsed_hit.model_norm) < 85:
|
||||
# The URL did not lead to the product the search result named.
|
||||
self.stats.inc("page_title_mismatch")
|
||||
return None
|
||||
# Fill variant fields the page name leaves out from the search title
|
||||
# of the same URL (both are statements by the same site).
|
||||
for attr in ("ram_gb", "storage_gb", "colour", "mpn", "processor"):
|
||||
if getattr(parsed, attr) is None and getattr(parsed_hit, attr) is not None:
|
||||
setattr(parsed, attr, getattr(parsed_hit, attr))
|
||||
|
||||
grade = self._grade(site)
|
||||
source_type = "brand_official" if site.kind == "brand_official" else "scraped_page"
|
||||
if product is not None:
|
||||
price = product.get("price")
|
||||
currency = product.get("currency")
|
||||
evidence = product["evidence"]
|
||||
parser = "jsonld"
|
||||
else:
|
||||
price, currency, evidence, parser = page.get("price"), page.get("currency"), page.get("evidence") or "", "html_meta"
|
||||
if currency not in (None, "INR"):
|
||||
price = None
|
||||
if currency is None and site.kind == "brand_official":
|
||||
price = None # a brand's global site may not be quoting rupees
|
||||
if price is not None and not (Decimal(500) <= price <= Decimal(1000000)):
|
||||
price = None
|
||||
evidence = evidence or f"{name} ({final_url})"
|
||||
listing = self._base_listing(
|
||||
site, final_url if final_url.startswith("http") else hit.url, parsed, name, source_type,
|
||||
evidence, 0.9 if (price is not None and parser == "jsonld") else 0.75, f"{parser}:grade{grade}", query,
|
||||
)
|
||||
# The site's own SKU when the page states it; otherwise its canonical URL.
|
||||
listing.source_sku = ((product or {}).get("sku") or source_sku(site, final_url or hit.url))[:300]
|
||||
listing.price = price
|
||||
listing.in_stock = (product or page).get("in_stock")
|
||||
listing.availability = (product or page).get("availability")
|
||||
listing.image_urls = list(dict.fromkeys((product or {}).get("images", []) + page.get("images", [])))[:8]
|
||||
if product:
|
||||
listing.gtin = product.get("gtin")
|
||||
listing.model_number = listing.model_number or product.get("mpn")
|
||||
listing.rating = product.get("rating")
|
||||
listing.review_count = product.get("review_count")
|
||||
listing.reviews = list(product.get("reviews") or [])
|
||||
listing.colour = listing.colour or product.get("color")
|
||||
raw_specs = dict((product or {}).get("properties") or {})
|
||||
raw_specs.update({k: v for k, v in spec_tables(BeautifulSoup(html, "lxml")).items() if k not in raw_specs})
|
||||
listing.specs_raw = dict(list(raw_specs.items())[:150])
|
||||
listing.specs, listing.spec_sources = normalise_specs(self.opt.category, raw_specs)
|
||||
if self.opt.use_llm:
|
||||
wanted = [k for k in self.ref.spec_keys.get(self.opt.category, {}) if k not in listing.specs]
|
||||
if wanted:
|
||||
text = "\n".join(f"{k}: {v}" for k, v in raw_specs.items()) or visible_text(html, 3500)
|
||||
extra, extra_src = fill_missing(self.opt.category, text, wanted)
|
||||
listing.specs.update(extra)
|
||||
listing.spec_sources.update(extra_src)
|
||||
self.stats.inc("llm_specs_kept", len(extra))
|
||||
if self.opt.category == "laptops":
|
||||
# "13th Gen Intel Core i7/ 16GB RAM" in a title names no CPU model;
|
||||
# the page's own spec table usually does.
|
||||
spec_texts = tuple(str(v) for k, v in raw_specs.items() if "processor" in k.lower() or "cpu" in k.lower())
|
||||
spec_texts += (str(listing.specs.get("processor") or ""),)
|
||||
before = parsed.processor
|
||||
fill_from_context(parsed, self.opt.category, spec_texts=spec_texts)
|
||||
if parsed.processor != before:
|
||||
listing.processor = parsed.processor
|
||||
listing.variant_key = variant_key(parsed, self.opt.category)
|
||||
listing.content_hash = hashlib.sha1(html.encode("utf-8", "ignore")).hexdigest()
|
||||
return listing
|
||||
|
||||
# -- persistence -----------------------------------------------------------
|
||||
def store(self, listing: Listing) -> Optional[int]:
|
||||
try:
|
||||
listing_id = repo.upsert_listing(listing, self.ids, self.run_id)
|
||||
except ValueError as exc:
|
||||
self.stats.inc("rejected_listing")
|
||||
logger.info("Listing rejected (%s): %s", exc, listing.source_url)
|
||||
return None
|
||||
self.stats.inc(f"listing_{listing.source_type}")
|
||||
if listing.reviews:
|
||||
self.stats.inc("reviews_stored", repo.save_reviews(listing_id, listing.reviews))
|
||||
if listing.price is not None:
|
||||
self.stats.inc("listing_with_price")
|
||||
decision = decide(listing, repo.product_candidates(listing.brand_slug, listing.category))
|
||||
if decision is None:
|
||||
self.stats.inc("listing_unmatched_no_variant")
|
||||
return listing_id
|
||||
product_id = decision.product_id or repo.create_product(listing, self.ids)
|
||||
if decision.product_id is None:
|
||||
self.stats.inc("product_created")
|
||||
repo.map_listing(listing_id, product_id, decision.method, decision.confidence, decision.review_status)
|
||||
if decision.review_status == "pending":
|
||||
self.stats.inc("match_pending_review")
|
||||
else:
|
||||
repo.merge_product_specs(product_id, listing.specs, listing.spec_sources, listing.source_url)
|
||||
self._touched_products.setdefault(product_id, []).append((listing_id, listing))
|
||||
return listing_id
|
||||
|
||||
# -- images ----------------------------------------------------------------
|
||||
def attach_images(self) -> None:
|
||||
rank_for = {"brand_official": 10, "scraped_page": 20, "search_snippet": 50}
|
||||
for product_id, entries in self._touched_products.items():
|
||||
if repo.product_image_count(product_id) >= 3:
|
||||
continue
|
||||
added = 0
|
||||
for listing_id, listing in sorted(entries, key=lambda e: rank_for[e[1].source_type]):
|
||||
for url in listing.image_urls:
|
||||
if added >= 3:
|
||||
break
|
||||
if self.client.check_image(url, MIN_IMAGE_BYTES):
|
||||
repo.add_image(product_id, url, listing_id, listing.source_type, rank_for[listing.source_type])
|
||||
added += 1
|
||||
self.stats.inc("images_from_pages")
|
||||
if added or not self.opt.find_images:
|
||||
continue
|
||||
added = self._images_from_search(product_id, entries)
|
||||
self.stats.inc("images_from_search", added)
|
||||
|
||||
def _images_from_search(self, product_id: int, entries: List[Tuple[int, Listing]]) -> int:
|
||||
"""Image search results are used only when the page an image sits on is
|
||||
one of THIS product's own listings (same site, same product), and the
|
||||
image result's title names the model."""
|
||||
_, listing = entries[0]
|
||||
brand = self.ref.brands[listing.brand_slug]
|
||||
listing_by_site = {l.site_domain: lid for lid, l in entries}
|
||||
hits = self.engine.images(f"{brand.name} {listing.model or listing.model_norm}", max_results=15) or []
|
||||
added = 0
|
||||
for hit in hits:
|
||||
site = site_for_url(hit.url)
|
||||
if site is None or site.domain not in listing_by_site:
|
||||
continue
|
||||
parsed = parse_title(clean_result_title(hit.title), listing.category, expected_brand=brand.slug)
|
||||
if not parsed.model_norm or fuzz.token_set_ratio(parsed.model_norm, listing.model_norm or "") < 90:
|
||||
continue
|
||||
if self.client.check_image(hit.image_url, MIN_IMAGE_BYTES):
|
||||
repo.add_image(product_id, hit.image_url, listing_by_site[site.domain], "search_image", 60)
|
||||
added += 1
|
||||
if added >= 2:
|
||||
break
|
||||
return added
|
||||
|
||||
# -- embeddings ------------------------------------------------------------
|
||||
def embed(self) -> int:
|
||||
return embed_verified_products()
|
||||
|
||||
# -- the run ---------------------------------------------------------------
|
||||
def run(self, *, embed: bool = True) -> Dict[str, int]:
|
||||
self.run_id = repo.start_run("collect", {
|
||||
"category": self.opt.category, "brands": self.opt.brands,
|
||||
"max_products_per_brand": self.opt.max_products_per_brand,
|
||||
"search_budget": self.opt.search_budget,
|
||||
})
|
||||
status, error = "done", None
|
||||
try:
|
||||
brands = [self.ref.brands[b] for b in self.opt.brands]
|
||||
probe_targets = {s.domain: s for b in brands for s in self._platforms_for(b) if s.policy == "probe"}
|
||||
if self.opt.fetch_pages:
|
||||
self.ensure_probes(list(probe_targets.values()))
|
||||
for brand in brands:
|
||||
found = self.discover(brand)
|
||||
# Keep the most common models first, up to the per-brand limit.
|
||||
by_model: Dict[str, int] = {}
|
||||
for (_, _), (_, _, parsed, _) in found.items():
|
||||
by_model[parsed.model_norm] = by_model.get(parsed.model_norm, 0) + 1
|
||||
keep = set(sorted(by_model, key=lambda m: -by_model[m])[: self.opt.max_products_per_brand])
|
||||
for (domain, _sku), (hit, site, parsed, query) in found.items():
|
||||
if parsed.model_norm not in keep:
|
||||
continue
|
||||
listing = None
|
||||
if self.opt.fetch_pages and self._grade(site) in ("A", "B"):
|
||||
res = self.client.get(hit.url)
|
||||
if res.ok:
|
||||
listing = self.listing_from_page(hit, site, parsed, query, res.text, res.final_url)
|
||||
if listing is None:
|
||||
self.stats.inc("page_unusable_fell_back_to_search")
|
||||
if listing is None:
|
||||
listing = self.listing_from_search(hit, site, parsed, query)
|
||||
self.store(listing)
|
||||
self.progress(f"{brand.name}: stored listings; {self.stats.counts}")
|
||||
self.attach_images()
|
||||
verification = repo.refresh_verification()
|
||||
self.stats.counts.update({f"products_{k}": v for k, v in verification.items()})
|
||||
if embed:
|
||||
try:
|
||||
self.stats.inc("embedded", self.embed())
|
||||
except Exception as exc: # noqa: BLE001 - embeddings are optional
|
||||
logger.warning("Embedding step skipped: %s", exc)
|
||||
self.stats.counts.update({f"search_{k}": v for k, v in self.engine.stats.items()})
|
||||
except Exception as exc:
|
||||
status, error = "failed", repr(exc)
|
||||
logger.exception("Collection run failed")
|
||||
raise
|
||||
finally:
|
||||
repo.finish_run(self.run_id, status, self.stats.counts, error)
|
||||
self.client.close()
|
||||
return self.stats.counts
|
||||
0
backend/app/electronics/db/__init__.py
Normal file
0
backend/app/electronics/db/__init__.py
Normal file
68
backend/app/electronics/db/connection.py
Normal file
68
backend/app/electronics/db/connection.py
Normal file
@@ -0,0 +1,68 @@
|
||||
"""Connections to the local electronics database.
|
||||
|
||||
settings.py has already refused to load unless DB_HOST is local and DB_NAME is
|
||||
electronics_catalog, so nothing here can reach another database.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from contextlib import contextmanager
|
||||
from typing import Iterator
|
||||
|
||||
import psycopg
|
||||
from psycopg.rows import dict_row
|
||||
|
||||
from app.infrastructure.settings import (
|
||||
DB_CONNECT_TIMEOUT_SECONDS,
|
||||
DB_HOST,
|
||||
DB_NAME,
|
||||
DB_PASSWORD,
|
||||
DB_PORT,
|
||||
DB_USER,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def connect(*, autocommit: bool = False) -> psycopg.Connection:
|
||||
conn = psycopg.connect(
|
||||
host=DB_HOST,
|
||||
port=DB_PORT,
|
||||
dbname=DB_NAME,
|
||||
user=DB_USER,
|
||||
password=DB_PASSWORD,
|
||||
connect_timeout=DB_CONNECT_TIMEOUT_SECONDS,
|
||||
autocommit=autocommit,
|
||||
row_factory=dict_row,
|
||||
)
|
||||
try:
|
||||
from pgvector.psycopg import register_vector
|
||||
|
||||
register_vector(conn)
|
||||
except Exception: # noqa: BLE001 - the extension is created by migration 0001
|
||||
pass
|
||||
return conn
|
||||
|
||||
|
||||
@contextmanager
|
||||
def transaction() -> Iterator[psycopg.Connection]:
|
||||
"""A connection whose work is committed on success, rolled back on error."""
|
||||
conn = connect()
|
||||
try:
|
||||
yield conn
|
||||
conn.commit()
|
||||
except Exception:
|
||||
conn.rollback()
|
||||
raise
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
|
||||
def check_connection() -> bool:
|
||||
try:
|
||||
with connect(autocommit=True) as conn:
|
||||
conn.execute("SELECT 1")
|
||||
return True
|
||||
except Exception as exc: # noqa: BLE001 - a health probe reports, never raises
|
||||
logger.debug("Database unreachable: %s", exc)
|
||||
return False
|
||||
63
backend/app/electronics/db/migrate.py
Normal file
63
backend/app/electronics/db/migrate.py
Normal file
@@ -0,0 +1,63 @@
|
||||
"""Tiny migration runner for the numbered SQL files in ./migrations.
|
||||
|
||||
Each file runs once, in its own transaction, and is recorded in
|
||||
elec.schema_migrations with a checksum. Editing an applied file is refused
|
||||
rather than silently ignored - add a new numbered file instead.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import logging
|
||||
from pathlib import Path
|
||||
from typing import List
|
||||
|
||||
from app.electronics.db.connection import connect
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
MIGRATIONS_DIR = Path(__file__).resolve().parent / "migrations"
|
||||
|
||||
_BOOTSTRAP = """
|
||||
CREATE SCHEMA IF NOT EXISTS elec;
|
||||
CREATE TABLE IF NOT EXISTS elec.schema_migrations (
|
||||
version TEXT PRIMARY KEY,
|
||||
checksum TEXT NOT NULL,
|
||||
applied_at TIMESTAMPTZ NOT NULL DEFAULT now()
|
||||
);
|
||||
"""
|
||||
|
||||
|
||||
def _files() -> List[Path]:
|
||||
return sorted(MIGRATIONS_DIR.glob("[0-9][0-9][0-9][0-9]_*.sql"))
|
||||
|
||||
|
||||
def run_migrations() -> List[str]:
|
||||
"""Apply pending migrations. Returns the versions applied by this call."""
|
||||
applied_now: List[str] = []
|
||||
with connect() as conn:
|
||||
conn.execute(_BOOTSTRAP)
|
||||
conn.commit()
|
||||
done = {
|
||||
r["version"]: r["checksum"]
|
||||
for r in conn.execute("SELECT version, checksum FROM elec.schema_migrations")
|
||||
}
|
||||
for path in _files():
|
||||
sql = path.read_text(encoding="utf-8")
|
||||
checksum = hashlib.sha256(sql.encode("utf-8")).hexdigest()
|
||||
version = path.stem
|
||||
if version in done:
|
||||
if done[version] != checksum:
|
||||
raise RuntimeError(
|
||||
f"Migration {version} was edited after it was applied. "
|
||||
f"Revert the edit and add a new numbered migration instead."
|
||||
)
|
||||
continue
|
||||
logger.info("Applying migration %s", version)
|
||||
with conn.transaction():
|
||||
conn.execute(sql)
|
||||
conn.execute(
|
||||
"INSERT INTO elec.schema_migrations (version, checksum) VALUES (%s, %s)",
|
||||
(version, checksum),
|
||||
)
|
||||
applied_now.append(version)
|
||||
return applied_now
|
||||
4
backend/app/electronics/db/migrations/0001_schema.sql
Normal file
4
backend/app/electronics/db/migrations/0001_schema.sql
Normal file
@@ -0,0 +1,4 @@
|
||||
-- Extensions and the dedicated schema. Everything this project owns lives in
|
||||
-- schema `elec` of database `electronics_catalog`.
|
||||
CREATE EXTENSION IF NOT EXISTS vector;
|
||||
CREATE SCHEMA IF NOT EXISTS elec;
|
||||
57
backend/app/electronics/db/migrations/0002_reference.sql
Normal file
57
backend/app/electronics/db/migrations/0002_reference.sql
Normal file
@@ -0,0 +1,57 @@
|
||||
-- Reference data: brands, categories, retail sites. Seeded from
|
||||
-- app/electronics/reference/*.yaml by `elec seed-reference`.
|
||||
|
||||
CREATE TABLE elec.brand (
|
||||
id SERIAL PRIMARY KEY,
|
||||
name TEXT NOT NULL UNIQUE,
|
||||
slug TEXT NOT NULL UNIQUE,
|
||||
parent_brand_id INT REFERENCES elec.brand(id),
|
||||
is_popular BOOLEAN NOT NULL DEFAULT TRUE,
|
||||
official_domains TEXT[] NOT NULL DEFAULT '{}',
|
||||
created_at TIMESTAMPTZ NOT NULL DEFAULT now()
|
||||
);
|
||||
|
||||
-- Every spelling that resolves to a brand. Sub-brands (Redmi, iQOO, Pixel)
|
||||
-- resolve to their parent and are remembered as the product family.
|
||||
CREATE TABLE elec.brand_alias (
|
||||
alias TEXT PRIMARY KEY CHECK (alias = lower(alias)),
|
||||
brand_id INT NOT NULL REFERENCES elec.brand(id) ON DELETE CASCADE,
|
||||
is_sub_brand BOOLEAN NOT NULL DEFAULT FALSE
|
||||
);
|
||||
|
||||
CREATE TABLE elec.category (
|
||||
id SERIAL PRIMARY KEY,
|
||||
slug TEXT NOT NULL UNIQUE,
|
||||
name TEXT NOT NULL UNIQUE
|
||||
);
|
||||
|
||||
CREATE TABLE elec.brand_category (
|
||||
brand_id INT NOT NULL REFERENCES elec.brand(id) ON DELETE CASCADE,
|
||||
category_id INT NOT NULL REFERENCES elec.category(id) ON DELETE CASCADE,
|
||||
PRIMARY KEY (brand_id, category_id)
|
||||
);
|
||||
|
||||
-- A retail platform or a brand's own site, with the outcome of its probe.
|
||||
-- probe_outcome A = fetchable with structured product data (scraped)
|
||||
-- B = fetchable, product data from page HTML/state (scraped)
|
||||
-- C = not fetched: serp_only policy, robots.txt disallow,
|
||||
-- block/CAPTCHA, or unreachable -> web search only
|
||||
CREATE TABLE elec.site (
|
||||
id SERIAL PRIMARY KEY,
|
||||
domain TEXT NOT NULL UNIQUE,
|
||||
name TEXT NOT NULL,
|
||||
kind TEXT NOT NULL CHECK (kind IN ('marketplace','national_chain','tn_regional','brand_official')),
|
||||
region TEXT NOT NULL CHECK (region IN ('national','TN')),
|
||||
policy TEXT NOT NULL CHECK (policy IN ('probe','serp_only')),
|
||||
brand_id INT REFERENCES elec.brand(id),
|
||||
product_url TEXT,
|
||||
pincode_param TEXT,
|
||||
enabled BOOLEAN NOT NULL DEFAULT TRUE,
|
||||
probe_outcome CHAR(1) CHECK (probe_outcome IN ('A','B','C')),
|
||||
robots_allowed BOOLEAN,
|
||||
probe_evidence JSONB NOT NULL DEFAULT '{}'::jsonb,
|
||||
probed_at TIMESTAMPTZ,
|
||||
breaker_until TIMESTAMPTZ,
|
||||
breaker_reason TEXT,
|
||||
CHECK (kind <> 'brand_official' OR brand_id IS NOT NULL)
|
||||
);
|
||||
160
backend/app/electronics/db/migrations/0003_observations.sql
Normal file
160
backend/app/electronics/db/migrations/0003_observations.sql
Normal file
@@ -0,0 +1,160 @@
|
||||
-- Runs, fetch audit trail, search cache, listings, prices, canonical products.
|
||||
|
||||
CREATE TABLE elec.crawl_run (
|
||||
id BIGSERIAL PRIMARY KEY,
|
||||
kind TEXT NOT NULL,
|
||||
params JSONB NOT NULL DEFAULT '{}'::jsonb,
|
||||
status TEXT NOT NULL DEFAULT 'running' CHECK (status IN ('running','done','failed')),
|
||||
stats JSONB NOT NULL DEFAULT '{}'::jsonb,
|
||||
error TEXT,
|
||||
started_at TIMESTAMPTZ NOT NULL DEFAULT now(),
|
||||
ended_at TIMESTAMPTZ
|
||||
);
|
||||
|
||||
-- Every HTTP request made to a retail or brand site. Evidence that the
|
||||
-- crawler obeyed robots.txt and its rate limits.
|
||||
CREATE TABLE elec.fetch_log (
|
||||
id BIGSERIAL PRIMARY KEY,
|
||||
crawl_run_id BIGINT REFERENCES elec.crawl_run(id) ON DELETE SET NULL,
|
||||
url TEXT NOT NULL,
|
||||
host TEXT NOT NULL,
|
||||
status INT,
|
||||
bytes INT,
|
||||
outcome TEXT NOT NULL,
|
||||
robots_allowed BOOLEAN,
|
||||
fetched_at TIMESTAMPTZ NOT NULL DEFAULT now()
|
||||
);
|
||||
CREATE INDEX fetch_log_host_time ON elec.fetch_log (host, fetched_at DESC);
|
||||
|
||||
CREATE TABLE elec.search_cache (
|
||||
provider TEXT NOT NULL,
|
||||
kind TEXT NOT NULL CHECK (kind IN ('text','images')),
|
||||
query TEXT NOT NULL,
|
||||
results JSONB NOT NULL,
|
||||
fetched_at TIMESTAMPTZ NOT NULL DEFAULT now(),
|
||||
PRIMARY KEY (provider, kind, query)
|
||||
);
|
||||
|
||||
-- One canonical product = one real-world variant (model + RAM + storage).
|
||||
-- verification_status becomes 'verified' only when the product has a brand
|
||||
-- official page, or listings on at least two different sites.
|
||||
CREATE TABLE elec.product (
|
||||
id BIGSERIAL PRIMARY KEY,
|
||||
brand_id INT NOT NULL REFERENCES elec.brand(id),
|
||||
category_id INT NOT NULL REFERENCES elec.category(id),
|
||||
family TEXT,
|
||||
model TEXT NOT NULL,
|
||||
model_norm TEXT NOT NULL,
|
||||
variant_key TEXT NOT NULL UNIQUE,
|
||||
display_name TEXT NOT NULL,
|
||||
ram_gb NUMERIC(6,1),
|
||||
storage_gb NUMERIC(7,1),
|
||||
processor TEXT,
|
||||
mpn TEXT,
|
||||
gtin TEXT,
|
||||
canonical_specs JSONB NOT NULL DEFAULT '{}'::jsonb,
|
||||
spec_sources JSONB NOT NULL DEFAULT '{}'::jsonb,
|
||||
verification_status TEXT NOT NULL DEFAULT 'unverified'
|
||||
CHECK (verification_status IN ('verified','unverified','rejected')),
|
||||
evidence_count INT NOT NULL DEFAULT 0,
|
||||
embedding vector(384),
|
||||
created_at TIMESTAMPTZ NOT NULL DEFAULT now(),
|
||||
updated_at TIMESTAMPTZ NOT NULL DEFAULT now()
|
||||
);
|
||||
CREATE INDEX product_brand_cat ON elec.product (brand_id, category_id);
|
||||
CREATE INDEX product_specs_gin ON elec.product USING GIN (canonical_specs);
|
||||
CREATE INDEX product_embedding_hnsw ON elec.product USING hnsw (embedding vector_cosine_ops);
|
||||
|
||||
-- The latest state of one product page on one site, or of one search result
|
||||
-- that points at such a page. Nothing is stored without the URL it came from
|
||||
-- and the text that the values were read from.
|
||||
CREATE TABLE elec.source_listing (
|
||||
id BIGSERIAL PRIMARY KEY,
|
||||
site_id INT NOT NULL REFERENCES elec.site(id),
|
||||
source_sku TEXT NOT NULL,
|
||||
source_url TEXT NOT NULL CHECK (source_url ~ '^https?://'),
|
||||
source_type TEXT NOT NULL CHECK (source_type IN ('scraped_page','search_snippet','brand_official')),
|
||||
brand_id INT NOT NULL REFERENCES elec.brand(id),
|
||||
category_id INT NOT NULL REFERENCES elec.category(id),
|
||||
family TEXT,
|
||||
title TEXT NOT NULL,
|
||||
model TEXT,
|
||||
model_number TEXT,
|
||||
ram_gb NUMERIC(6,1),
|
||||
storage_gb NUMERIC(7,1),
|
||||
colour TEXT,
|
||||
price NUMERIC(12,2) CHECK (price IS NULL OR price BETWEEN 500 AND 1000000),
|
||||
mrp NUMERIC(12,2) CHECK (mrp IS NULL OR mrp BETWEEN 500 AND 1000000),
|
||||
currency TEXT NOT NULL DEFAULT 'INR' CHECK (currency = 'INR'),
|
||||
availability TEXT,
|
||||
in_stock BOOLEAN,
|
||||
pincode TEXT,
|
||||
pincode_applied BOOLEAN NOT NULL DEFAULT FALSE,
|
||||
rating NUMERIC(3,2) CHECK (rating IS NULL OR rating BETWEEN 0 AND 5),
|
||||
review_count INT,
|
||||
gtin TEXT,
|
||||
image_urls TEXT[] NOT NULL DEFAULT '{}',
|
||||
specs_raw JSONB NOT NULL DEFAULT '{}'::jsonb,
|
||||
specs JSONB NOT NULL DEFAULT '{}'::jsonb,
|
||||
evidence_text TEXT NOT NULL CHECK (length(evidence_text) > 0),
|
||||
search_query TEXT,
|
||||
confidence NUMERIC(3,2) NOT NULL CHECK (confidence BETWEEN 0 AND 1),
|
||||
parser TEXT NOT NULL,
|
||||
content_hash TEXT,
|
||||
first_seen_at TIMESTAMPTZ NOT NULL DEFAULT now(),
|
||||
last_seen_at TIMESTAMPTZ NOT NULL DEFAULT now(),
|
||||
crawl_run_id BIGINT REFERENCES elec.crawl_run(id) ON DELETE SET NULL,
|
||||
UNIQUE (site_id, source_sku),
|
||||
CHECK (pincode_applied = FALSE OR pincode IS NOT NULL)
|
||||
);
|
||||
CREATE INDEX listing_brand_cat ON elec.source_listing (brand_id, category_id);
|
||||
|
||||
-- Append-only price observations. UPDATE is refused by a trigger.
|
||||
CREATE TABLE elec.price_history (
|
||||
id BIGSERIAL PRIMARY KEY,
|
||||
listing_id BIGINT NOT NULL REFERENCES elec.source_listing(id) ON DELETE CASCADE,
|
||||
price NUMERIC(12,2) CHECK (price IS NULL OR price BETWEEN 500 AND 1000000),
|
||||
mrp NUMERIC(12,2),
|
||||
availability TEXT,
|
||||
in_stock BOOLEAN,
|
||||
source_type TEXT NOT NULL,
|
||||
pincode TEXT,
|
||||
pincode_applied BOOLEAN NOT NULL DEFAULT FALSE,
|
||||
evidence_text TEXT NOT NULL CHECK (length(evidence_text) > 0),
|
||||
observed_at TIMESTAMPTZ NOT NULL DEFAULT now(),
|
||||
crawl_run_id BIGINT REFERENCES elec.crawl_run(id) ON DELETE SET NULL
|
||||
);
|
||||
CREATE INDEX price_history_listing_time ON elec.price_history (listing_id, observed_at DESC);
|
||||
|
||||
CREATE FUNCTION elec.refuse_update() RETURNS trigger LANGUAGE plpgsql AS $$
|
||||
BEGIN
|
||||
RAISE EXCEPTION 'elec.price_history is append-only';
|
||||
END $$;
|
||||
CREATE TRIGGER price_history_append_only BEFORE UPDATE ON elec.price_history
|
||||
FOR EACH ROW EXECUTE FUNCTION elec.refuse_update();
|
||||
|
||||
-- Which canonical product a listing belongs to, and how sure we are.
|
||||
-- Only 'auto' and 'approved' links count as evidence or appear in views.
|
||||
CREATE TABLE elec.product_listing_map (
|
||||
listing_id BIGINT PRIMARY KEY REFERENCES elec.source_listing(id) ON DELETE CASCADE,
|
||||
product_id BIGINT NOT NULL REFERENCES elec.product(id) ON DELETE CASCADE,
|
||||
method TEXT NOT NULL CHECK (method IN ('gtin','mpn','variant_key','fuzzy','manual')),
|
||||
confidence NUMERIC(3,2) NOT NULL CHECK (confidence BETWEEN 0 AND 1),
|
||||
review_status TEXT NOT NULL CHECK (review_status IN ('auto','pending','approved','rejected')),
|
||||
created_at TIMESTAMPTZ NOT NULL DEFAULT now(),
|
||||
reviewed_at TIMESTAMPTZ
|
||||
);
|
||||
CREATE INDEX map_product ON elec.product_listing_map (product_id);
|
||||
|
||||
-- Images are URLs only (never downloaded), each tied to the listing it was
|
||||
-- found on and checked live.
|
||||
CREATE TABLE elec.product_image (
|
||||
id BIGSERIAL PRIMARY KEY,
|
||||
product_id BIGINT NOT NULL REFERENCES elec.product(id) ON DELETE CASCADE,
|
||||
url TEXT NOT NULL CHECK (url ~ '^https?://'),
|
||||
source_listing_id BIGINT NOT NULL REFERENCES elec.source_listing(id) ON DELETE CASCADE,
|
||||
source_type TEXT NOT NULL,
|
||||
rank INT NOT NULL DEFAULT 100,
|
||||
validated_at TIMESTAMPTZ NOT NULL DEFAULT now(),
|
||||
UNIQUE (product_id, url)
|
||||
);
|
||||
71
backend/app/electronics/db/migrations/0004_views.sql
Normal file
71
backend/app/electronics/db/migrations/0004_views.sql
Normal file
@@ -0,0 +1,71 @@
|
||||
-- Read-side views. Public views only ever show VERIFIED products and links
|
||||
-- that are 'auto' or 'approved'.
|
||||
|
||||
CREATE VIEW elec.v_product_availability AS
|
||||
SELECT p.id AS product_id,
|
||||
b.name AS brand,
|
||||
c.slug AS category,
|
||||
p.display_name,
|
||||
s.id AS site_id,
|
||||
s.name AS site,
|
||||
s.domain,
|
||||
s.kind AS site_kind,
|
||||
s.region AS site_region,
|
||||
l.id AS listing_id,
|
||||
l.source_url,
|
||||
l.source_type,
|
||||
l.title AS listing_title,
|
||||
l.colour,
|
||||
l.price,
|
||||
l.mrp,
|
||||
l.in_stock,
|
||||
l.availability,
|
||||
l.pincode,
|
||||
l.pincode_applied,
|
||||
l.confidence,
|
||||
l.last_seen_at AS observed_at
|
||||
FROM elec.product p
|
||||
JOIN elec.brand b ON b.id = p.brand_id
|
||||
JOIN elec.category c ON c.id = p.category_id
|
||||
JOIN elec.product_listing_map m ON m.product_id = p.id AND m.review_status IN ('auto','approved')
|
||||
JOIN elec.source_listing l ON l.id = m.listing_id
|
||||
JOIN elec.site s ON s.id = l.site_id
|
||||
WHERE p.verification_status = 'verified';
|
||||
|
||||
-- Cheapest known price per product. Scraped prices are preferred over search
|
||||
-- snippet prices; a listing known to be out of stock is skipped.
|
||||
CREATE VIEW elec.v_best_price AS
|
||||
SELECT DISTINCT ON (product_id)
|
||||
product_id, site, domain, source_url, source_type, price, mrp, in_stock, observed_at
|
||||
FROM elec.v_product_availability
|
||||
WHERE price IS NOT NULL AND in_stock IS DISTINCT FROM FALSE
|
||||
ORDER BY product_id, (source_type = 'search_snippet'), price, observed_at DESC;
|
||||
|
||||
CREATE VIEW elec.v_brand_catalog AS
|
||||
SELECT p.id AS product_id, b.name AS brand, b.slug AS brand_slug, c.slug AS category,
|
||||
p.family, p.display_name, p.model, p.ram_gb, p.storage_gb, p.processor,
|
||||
p.canonical_specs,
|
||||
bp.price AS best_price,
|
||||
bp.site AS best_price_site,
|
||||
bp.source_type AS best_price_source_type,
|
||||
(SELECT count(DISTINCT a.site_id) FROM elec.v_product_availability a
|
||||
WHERE a.product_id = p.id) AS platform_count,
|
||||
(SELECT coalesce(bool_or(a.site_region = 'TN'), FALSE) FROM elec.v_product_availability a
|
||||
WHERE a.product_id = p.id) AS sold_by_tn_retailer,
|
||||
(SELECT i.url FROM elec.product_image i WHERE i.product_id = p.id
|
||||
ORDER BY i.rank, i.id LIMIT 1) AS image_url,
|
||||
p.updated_at
|
||||
FROM elec.product p
|
||||
JOIN elec.brand b ON b.id = p.brand_id
|
||||
JOIN elec.category c ON c.id = p.category_id
|
||||
LEFT JOIN elec.v_best_price bp ON bp.product_id = p.id
|
||||
WHERE p.verification_status = 'verified';
|
||||
|
||||
CREATE VIEW elec.v_brand_summary AS
|
||||
SELECT brand, brand_slug, category,
|
||||
count(*) AS product_count,
|
||||
min(best_price) AS min_price,
|
||||
max(best_price) AS max_price,
|
||||
max(platform_count) AS max_platforms
|
||||
FROM elec.v_brand_catalog
|
||||
GROUP BY brand, brand_slug, category;
|
||||
@@ -0,0 +1,44 @@
|
||||
-- Search results carry cached, sometimes seller-specific prices. A price that
|
||||
-- disagrees sharply with the product-page price for the same product (or is
|
||||
-- below what the category can cost) is kept with its evidence but flagged, and
|
||||
-- is never used as the "best price". Set by repository.flag_price_outliers().
|
||||
ALTER TABLE elec.source_listing ADD COLUMN price_outlier BOOLEAN NOT NULL DEFAULT FALSE;
|
||||
|
||||
CREATE OR REPLACE VIEW elec.v_product_availability AS
|
||||
SELECT p.id AS product_id,
|
||||
b.name AS brand,
|
||||
c.slug AS category,
|
||||
p.display_name,
|
||||
s.id AS site_id,
|
||||
s.name AS site,
|
||||
s.domain,
|
||||
s.kind AS site_kind,
|
||||
s.region AS site_region,
|
||||
l.id AS listing_id,
|
||||
l.source_url,
|
||||
l.source_type,
|
||||
l.title AS listing_title,
|
||||
l.colour,
|
||||
l.price,
|
||||
l.mrp,
|
||||
l.in_stock,
|
||||
l.availability,
|
||||
l.pincode,
|
||||
l.pincode_applied,
|
||||
l.confidence,
|
||||
l.last_seen_at AS observed_at,
|
||||
l.price_outlier
|
||||
FROM elec.product p
|
||||
JOIN elec.brand b ON b.id = p.brand_id
|
||||
JOIN elec.category c ON c.id = p.category_id
|
||||
JOIN elec.product_listing_map m ON m.product_id = p.id AND m.review_status IN ('auto','approved')
|
||||
JOIN elec.source_listing l ON l.id = m.listing_id
|
||||
JOIN elec.site s ON s.id = l.site_id
|
||||
WHERE p.verification_status = 'verified';
|
||||
|
||||
CREATE OR REPLACE VIEW elec.v_best_price AS
|
||||
SELECT DISTINCT ON (product_id)
|
||||
product_id, site, domain, source_url, source_type, price, mrp, in_stock, observed_at
|
||||
FROM elec.v_product_availability
|
||||
WHERE price IS NOT NULL AND NOT price_outlier AND in_stock IS DISTINCT FROM FALSE
|
||||
ORDER BY product_id, (source_type = 'search_snippet'), price, observed_at DESC;
|
||||
@@ -0,0 +1,30 @@
|
||||
-- Best price: a product whose every priced listing is out of stock still has a
|
||||
-- price worth showing. In-stock (or unknown-stock) prices still win; an
|
||||
-- out-of-stock price is used only when nothing else is priced. Same columns
|
||||
-- as 0005, so v_brand_catalog keeps working unchanged.
|
||||
CREATE OR REPLACE VIEW elec.v_best_price AS
|
||||
SELECT DISTINCT ON (product_id)
|
||||
product_id, site, domain, source_url, source_type, price, mrp, in_stock, observed_at
|
||||
FROM elec.v_product_availability
|
||||
WHERE price IS NOT NULL AND NOT price_outlier
|
||||
ORDER BY product_id, (in_stock IS FALSE), (source_type = 'search_snippet'), price, observed_at DESC;
|
||||
|
||||
-- Individual customer reviews, exactly as a product page publishes them in its
|
||||
-- schema.org JSON-LD. Nothing here is generated: every row is a review the
|
||||
-- listing's own page stated. Sentiment is derived only from the reviewer's
|
||||
-- own star rating (>=4 positive, >=3 neutral, <3 negative); NULL when the
|
||||
-- review states no rating.
|
||||
CREATE TABLE elec.listing_review (
|
||||
id BIGSERIAL PRIMARY KEY,
|
||||
listing_id BIGINT NOT NULL REFERENCES elec.source_listing(id) ON DELETE CASCADE,
|
||||
author TEXT,
|
||||
rating NUMERIC(2,1) CHECK (rating IS NULL OR rating BETWEEN 0 AND 5),
|
||||
title TEXT,
|
||||
body TEXT NOT NULL,
|
||||
review_date TEXT,
|
||||
sentiment TEXT CHECK (sentiment IS NULL OR sentiment IN ('positive','neutral','negative')),
|
||||
content_hash TEXT NOT NULL,
|
||||
fetched_at TIMESTAMPTZ NOT NULL DEFAULT now(),
|
||||
UNIQUE (listing_id, content_hash)
|
||||
);
|
||||
CREATE INDEX listing_review_listing_idx ON elec.listing_review (listing_id);
|
||||
588
backend/app/electronics/db/repository.py
Normal file
588
backend/app/electronics/db/repository.py
Normal file
@@ -0,0 +1,588 @@
|
||||
"""All SQL used by the pipeline. psycopg3, no ORM - the same style as the
|
||||
original project, with each function owning one statement or one small unit
|
||||
of work."""
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import re
|
||||
from datetime import datetime, timezone
|
||||
from decimal import Decimal
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from psycopg.types.json import Jsonb
|
||||
|
||||
from app.electronics.db.connection import connect, transaction
|
||||
from app.electronics.models import Listing
|
||||
from app.electronics.reference import Reference, slugify
|
||||
|
||||
|
||||
def _json(value: Any) -> Jsonb:
|
||||
return Jsonb(json.loads(json.dumps(value, default=str)))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Reference data
|
||||
# ---------------------------------------------------------------------------
|
||||
def seed_reference(ref: Reference) -> Dict[str, int]:
|
||||
"""Idempotent upsert of brands, aliases, categories and sites."""
|
||||
counts = {"brands": 0, "aliases": 0, "categories": 0, "sites": 0}
|
||||
with transaction() as conn:
|
||||
for c in ref.categories.values():
|
||||
conn.execute(
|
||||
"INSERT INTO elec.category (slug, name) VALUES (%s, %s) "
|
||||
"ON CONFLICT (slug) DO UPDATE SET name = EXCLUDED.name",
|
||||
(c.slug, c.name),
|
||||
)
|
||||
counts["categories"] += 1
|
||||
for b in ref.brands.values():
|
||||
row = conn.execute(
|
||||
"INSERT INTO elec.brand (name, slug, official_domains) VALUES (%s, %s, %s) "
|
||||
"ON CONFLICT (slug) DO UPDATE SET name = EXCLUDED.name, official_domains = EXCLUDED.official_domains "
|
||||
"RETURNING id",
|
||||
(b.name, b.slug, list(b.official)),
|
||||
).fetchone()
|
||||
counts["brands"] += 1
|
||||
for alias in b.aliases:
|
||||
conn.execute(
|
||||
"INSERT INTO elec.brand_alias (alias, brand_id, is_sub_brand) VALUES (%s, %s, FALSE) "
|
||||
"ON CONFLICT (alias) DO UPDATE SET brand_id = EXCLUDED.brand_id, is_sub_brand = FALSE",
|
||||
(alias, row["id"]),
|
||||
)
|
||||
counts["aliases"] += 1
|
||||
for sub in b.sub_brands:
|
||||
conn.execute(
|
||||
"INSERT INTO elec.brand_alias (alias, brand_id, is_sub_brand) VALUES (%s, %s, TRUE) "
|
||||
"ON CONFLICT (alias) DO UPDATE SET brand_id = EXCLUDED.brand_id, is_sub_brand = TRUE",
|
||||
(sub, row["id"]),
|
||||
)
|
||||
counts["aliases"] += 1
|
||||
for cat in b.categories:
|
||||
conn.execute(
|
||||
"INSERT INTO elec.brand_category (brand_id, category_id) "
|
||||
"SELECT %s, id FROM elec.category WHERE slug = %s ON CONFLICT DO NOTHING",
|
||||
(row["id"], cat),
|
||||
)
|
||||
for s in ref.sites.values():
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO elec.site (domain, name, kind, region, policy, brand_id, product_url, pincode_param)
|
||||
VALUES (%s, %s, %s, %s, %s, (SELECT id FROM elec.brand WHERE slug = %s), %s, %s)
|
||||
ON CONFLICT (domain) DO UPDATE SET
|
||||
name = EXCLUDED.name, kind = EXCLUDED.kind, region = EXCLUDED.region,
|
||||
policy = EXCLUDED.policy, brand_id = EXCLUDED.brand_id,
|
||||
product_url = EXCLUDED.product_url, pincode_param = EXCLUDED.pincode_param
|
||||
""",
|
||||
(s.domain, s.name, s.kind, s.region, s.policy, s.brand_slug, s.product_url, s.pincode_param),
|
||||
)
|
||||
counts["sites"] += 1
|
||||
return counts
|
||||
|
||||
|
||||
def id_maps() -> Dict[str, Dict[str, int]]:
|
||||
with connect() as conn:
|
||||
return {
|
||||
"brand": {r["slug"]: r["id"] for r in conn.execute("SELECT id, slug FROM elec.brand")},
|
||||
"category": {r["slug"]: r["id"] for r in conn.execute("SELECT id, slug FROM elec.category")},
|
||||
"site": {r["domain"]: r["id"] for r in conn.execute("SELECT id, domain FROM elec.site")},
|
||||
}
|
||||
|
||||
|
||||
def sites() -> List[dict]:
|
||||
with connect() as conn:
|
||||
return list(conn.execute("SELECT * FROM elec.site ORDER BY kind, name"))
|
||||
|
||||
|
||||
def set_probe_result(domain: str, outcome: str, robots_allowed: Optional[bool], evidence: dict) -> None:
|
||||
with transaction() as conn:
|
||||
conn.execute(
|
||||
"UPDATE elec.site SET probe_outcome = %s, robots_allowed = %s, probe_evidence = %s, probed_at = now() "
|
||||
"WHERE domain = %s",
|
||||
(outcome, robots_allowed, _json(evidence), domain),
|
||||
)
|
||||
|
||||
|
||||
def trip_breaker(domain: str, reason: str, until_epoch: float) -> None:
|
||||
with transaction() as conn:
|
||||
conn.execute(
|
||||
"UPDATE elec.site SET breaker_until = to_timestamp(%s), breaker_reason = %s, "
|
||||
"probe_outcome = 'C' WHERE domain = %s OR %s LIKE '%%.' || domain",
|
||||
(until_epoch, reason, domain, domain),
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Runs and fetch log
|
||||
# ---------------------------------------------------------------------------
|
||||
def start_run(kind: str, params: dict) -> int:
|
||||
with transaction() as conn:
|
||||
return conn.execute(
|
||||
"INSERT INTO elec.crawl_run (kind, params) VALUES (%s, %s) RETURNING id", (kind, _json(params))
|
||||
).fetchone()["id"]
|
||||
|
||||
|
||||
def finish_run(run_id: int, status: str, stats: dict, error: Optional[str] = None) -> None:
|
||||
with transaction() as conn:
|
||||
conn.execute(
|
||||
"UPDATE elec.crawl_run SET status = %s, stats = %s, error = %s, ended_at = now() WHERE id = %s",
|
||||
(status, _json(stats), error, run_id),
|
||||
)
|
||||
|
||||
|
||||
def log_fetch(run_id: Optional[int], url: str, host: str, status: Optional[int], nbytes: int,
|
||||
outcome: str, robots_allowed: Optional[bool]) -> None:
|
||||
with transaction() as conn:
|
||||
conn.execute(
|
||||
"INSERT INTO elec.fetch_log (crawl_run_id, url, host, status, bytes, outcome, robots_allowed) "
|
||||
"VALUES (%s, %s, %s, %s, %s, %s, %s)",
|
||||
(run_id, url, host, status, nbytes, outcome, robots_allowed),
|
||||
)
|
||||
|
||||
|
||||
def recent_runs(limit: int = 20) -> List[dict]:
|
||||
with connect() as conn:
|
||||
return list(conn.execute("SELECT * FROM elec.crawl_run ORDER BY id DESC LIMIT %s", (limit,)))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Search cache
|
||||
# ---------------------------------------------------------------------------
|
||||
def search_cache_get(provider: str, kind: str, query: str, ttl_hours: int) -> Optional[List[dict]]:
|
||||
with connect() as conn:
|
||||
row = conn.execute(
|
||||
"SELECT results FROM elec.search_cache WHERE provider = %s AND kind = %s AND query = %s "
|
||||
"AND fetched_at > now() - make_interval(hours => %s)",
|
||||
(provider, kind, query, ttl_hours),
|
||||
).fetchone()
|
||||
return row["results"] if row else None
|
||||
|
||||
|
||||
def search_cache_put(provider: str, kind: str, query: str, results: List[dict]) -> None:
|
||||
with transaction() as conn:
|
||||
conn.execute(
|
||||
"INSERT INTO elec.search_cache (provider, kind, query, results) VALUES (%s, %s, %s, %s) "
|
||||
"ON CONFLICT (provider, kind, query) DO UPDATE SET results = EXCLUDED.results, fetched_at = now()",
|
||||
(provider, kind, query, _json(results)),
|
||||
)
|
||||
|
||||
|
||||
def google_queries_today() -> int:
|
||||
with connect() as conn:
|
||||
return conn.execute(
|
||||
"SELECT count(*) AS n FROM elec.search_cache WHERE provider = 'google' AND fetched_at::date = current_date"
|
||||
).fetchone()["n"]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Listings and prices
|
||||
# ---------------------------------------------------------------------------
|
||||
def upsert_listing(listing: Listing, ids: Dict[str, Dict[str, int]], run_id: Optional[int]) -> int:
|
||||
"""Write the latest state of a listing and append one price observation."""
|
||||
listing.validate()
|
||||
site_id = ids["site"][listing.site_domain]
|
||||
brand_id = ids["brand"][listing.brand_slug]
|
||||
category_id = ids["category"][listing.category]
|
||||
with transaction() as conn:
|
||||
existing = conn.execute(
|
||||
"SELECT id, source_type, price FROM elec.source_listing WHERE site_id = %s AND source_sku = %s",
|
||||
(site_id, listing.source_sku),
|
||||
).fetchone()
|
||||
# A scraped page is better evidence than a search snippet about the
|
||||
# same page. Never let a later snippet overwrite scraped values.
|
||||
if existing and existing["source_type"] in ("scraped_page", "brand_official") and listing.source_type == "search_snippet":
|
||||
conn.execute("UPDATE elec.source_listing SET last_seen_at = now() WHERE id = %s", (existing["id"],))
|
||||
return existing["id"]
|
||||
params = dict(
|
||||
site_id=site_id, source_sku=listing.source_sku, source_url=listing.source_url,
|
||||
source_type=listing.source_type, brand_id=brand_id, category_id=category_id,
|
||||
family=listing.family, title=listing.title[:500], model=listing.model,
|
||||
model_number=listing.model_number, ram_gb=listing.ram_gb, storage_gb=listing.storage_gb,
|
||||
colour=listing.colour, price=listing.price, mrp=listing.mrp, availability=listing.availability,
|
||||
in_stock=listing.in_stock, pincode=listing.pincode, pincode_applied=listing.pincode_applied,
|
||||
rating=listing.rating, review_count=listing.review_count, gtin=listing.gtin,
|
||||
image_urls=listing.image_urls[:12], specs_raw=_json(listing.specs_raw), specs=_json(listing.specs),
|
||||
evidence_text=listing.evidence_text[:4000], search_query=listing.search_query,
|
||||
confidence=round(listing.confidence, 2), parser=listing.parser, content_hash=listing.content_hash,
|
||||
crawl_run_id=run_id,
|
||||
)
|
||||
row = conn.execute(
|
||||
"""
|
||||
INSERT INTO elec.source_listing (
|
||||
site_id, source_sku, source_url, source_type, brand_id, category_id, family, title, model,
|
||||
model_number, ram_gb, storage_gb, colour, price, mrp, availability, in_stock, pincode,
|
||||
pincode_applied, rating, review_count, gtin, image_urls, specs_raw, specs, evidence_text,
|
||||
search_query, confidence, parser, content_hash, crawl_run_id)
|
||||
VALUES (
|
||||
%(site_id)s, %(source_sku)s, %(source_url)s, %(source_type)s, %(brand_id)s, %(category_id)s,
|
||||
%(family)s, %(title)s, %(model)s, %(model_number)s, %(ram_gb)s, %(storage_gb)s, %(colour)s,
|
||||
%(price)s, %(mrp)s, %(availability)s, %(in_stock)s, %(pincode)s, %(pincode_applied)s,
|
||||
%(rating)s, %(review_count)s, %(gtin)s, %(image_urls)s, %(specs_raw)s, %(specs)s,
|
||||
%(evidence_text)s, %(search_query)s, %(confidence)s, %(parser)s, %(content_hash)s,
|
||||
%(crawl_run_id)s)
|
||||
ON CONFLICT (site_id, source_sku) DO UPDATE SET
|
||||
source_url = EXCLUDED.source_url, source_type = EXCLUDED.source_type,
|
||||
brand_id = EXCLUDED.brand_id, category_id = EXCLUDED.category_id, family = EXCLUDED.family,
|
||||
title = EXCLUDED.title, model = EXCLUDED.model, model_number = EXCLUDED.model_number,
|
||||
ram_gb = EXCLUDED.ram_gb, storage_gb = EXCLUDED.storage_gb, colour = EXCLUDED.colour,
|
||||
price = EXCLUDED.price, mrp = EXCLUDED.mrp, availability = EXCLUDED.availability,
|
||||
in_stock = EXCLUDED.in_stock, pincode = EXCLUDED.pincode,
|
||||
pincode_applied = EXCLUDED.pincode_applied, rating = EXCLUDED.rating,
|
||||
review_count = EXCLUDED.review_count, gtin = EXCLUDED.gtin, image_urls = EXCLUDED.image_urls,
|
||||
specs_raw = EXCLUDED.specs_raw, specs = EXCLUDED.specs, evidence_text = EXCLUDED.evidence_text,
|
||||
search_query = EXCLUDED.search_query, confidence = EXCLUDED.confidence, parser = EXCLUDED.parser,
|
||||
content_hash = EXCLUDED.content_hash, crawl_run_id = EXCLUDED.crawl_run_id, last_seen_at = now()
|
||||
RETURNING id
|
||||
""",
|
||||
params,
|
||||
).fetchone()
|
||||
listing_id = row["id"]
|
||||
if listing.price is not None or listing.in_stock is not None:
|
||||
conn.execute(
|
||||
"INSERT INTO elec.price_history (listing_id, price, mrp, availability, in_stock, source_type, "
|
||||
"pincode, pincode_applied, evidence_text, crawl_run_id) VALUES (%s,%s,%s,%s,%s,%s,%s,%s,%s,%s)",
|
||||
(listing_id, listing.price, listing.mrp, listing.availability, listing.in_stock,
|
||||
listing.source_type, listing.pincode, listing.pincode_applied,
|
||||
listing.evidence_text[:2000], run_id),
|
||||
)
|
||||
return listing_id
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Ratings and reviews
|
||||
# ---------------------------------------------------------------------------
|
||||
def save_reviews(listing_id: int, reviews: List[Dict[str, Any]]) -> int:
|
||||
"""Store the reviews a listing's page publishes. Idempotent per review
|
||||
text; an empty list changes nothing (a later search-only sighting must
|
||||
not erase what the page said). Returns the number of new rows."""
|
||||
from app.electronics.reviews import sentiment_for
|
||||
|
||||
added = 0
|
||||
with transaction() as conn:
|
||||
for r in reviews:
|
||||
body = (r.get("body") or "").strip()
|
||||
if not body:
|
||||
continue
|
||||
digest = hashlib.sha1(f"{r.get('author') or ''}|{body}".encode("utf-8", "ignore")).hexdigest()
|
||||
row = conn.execute(
|
||||
"INSERT INTO elec.listing_review (listing_id, author, rating, title, body, review_date, sentiment, "
|
||||
"content_hash) VALUES (%s,%s,%s,%s,%s,%s,%s,%s) "
|
||||
"ON CONFLICT (listing_id, content_hash) DO NOTHING RETURNING id",
|
||||
(listing_id, (r.get("author") or None) and str(r["author"])[:200], r.get("rating"),
|
||||
(r.get("title") or None) and str(r["title"])[:300], body[:4000],
|
||||
(r.get("review_date") or None) and str(r["review_date"])[:40],
|
||||
sentiment_for(r.get("rating")), digest),
|
||||
).fetchone()
|
||||
added += 1 if row else 0
|
||||
return added
|
||||
|
||||
|
||||
def update_listing_rating(listing_id: int, rating: Optional[Decimal], review_count: Optional[int]) -> None:
|
||||
"""Refresh only the rating fields of a listing (used by the review backfill)."""
|
||||
with transaction() as conn:
|
||||
conn.execute(
|
||||
"UPDATE elec.source_listing SET rating = %s, review_count = %s WHERE id = %s",
|
||||
(rating, review_count, listing_id),
|
||||
)
|
||||
|
||||
|
||||
def product_rating_and_reviews(conn, product_id: int) -> Dict[str, Any]:
|
||||
"""Per-platform ratings and all stored reviews for a verified product's
|
||||
approved listings, each with the page it was read from."""
|
||||
sources = conn.execute(
|
||||
"SELECT a.site, a.source_url, l.rating, l.review_count FROM elec.v_product_availability a "
|
||||
"JOIN elec.source_listing l ON l.id = a.listing_id "
|
||||
"WHERE a.product_id = %s AND l.rating > 0 ORDER BY l.review_count DESC NULLS LAST, a.site",
|
||||
(product_id,),
|
||||
).fetchall()
|
||||
reviews = conn.execute(
|
||||
"SELECT a.site, a.source_url, r.author, r.rating, r.title, r.body, r.review_date, r.sentiment "
|
||||
"FROM elec.v_product_availability a JOIN elec.listing_review r ON r.listing_id = a.listing_id "
|
||||
"WHERE a.product_id = %s",
|
||||
(product_id,),
|
||||
).fetchall()
|
||||
return {"sources": [dict(s) for s in sources], "reviews": [dict(r) for r in reviews]}
|
||||
|
||||
|
||||
def listings_for_review_backfill(category: Optional[str] = None) -> List[dict]:
|
||||
"""Page-read listings of verified products, for re-reading ratings/reviews."""
|
||||
sql = (
|
||||
"SELECT a.listing_id, a.source_url, a.domain, a.site_kind, a.category, l.source_sku, l.title "
|
||||
"FROM elec.v_product_availability a JOIN elec.source_listing l ON l.id = a.listing_id "
|
||||
"WHERE a.source_type IN ('scraped_page','brand_official')"
|
||||
)
|
||||
params: tuple = ()
|
||||
if category:
|
||||
sql += " AND a.category = %s"
|
||||
params = (category,)
|
||||
with connect() as conn:
|
||||
return list(conn.execute(sql + " ORDER BY a.listing_id", params))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Products, matching, images
|
||||
# ---------------------------------------------------------------------------
|
||||
def product_candidates(brand_slug: str, category: str) -> List[dict]:
|
||||
with connect() as conn:
|
||||
return list(conn.execute(
|
||||
"SELECT p.id, p.variant_key, p.model_norm, p.ram_gb, p.storage_gb, p.processor, p.mpn, p.gtin "
|
||||
"FROM elec.product p JOIN elec.brand b ON b.id = p.brand_id JOIN elec.category c ON c.id = p.category_id "
|
||||
"WHERE b.slug = %s AND c.slug = %s AND p.verification_status <> 'rejected'",
|
||||
(brand_slug, category),
|
||||
))
|
||||
|
||||
|
||||
def _cpu_label(processor: Optional[str]) -> str:
|
||||
""""ryzen 5 7530u" -> "Ryzen 5 7530U", "i5-1334u" -> "i5-1334U"."""
|
||||
if not processor:
|
||||
return ""
|
||||
def fmt(t: str) -> str:
|
||||
if re.fullmatch(r"i[3579]-\w+", t):
|
||||
return "i" + t[1:].upper() # i5-1334U
|
||||
if any(ch.isdigit() for ch in t):
|
||||
return t.upper() # 7530U, M5
|
||||
return t.title() # Ryzen, Core, Ultra
|
||||
return " ".join(fmt(t) for t in processor.split())
|
||||
|
||||
|
||||
def product_display_name(listing: Listing) -> str:
|
||||
variant = [x for x in (
|
||||
_cpu_label(listing.processor) if listing.category == "laptops" else "",
|
||||
f"{_fmt_gb(listing.ram_gb)} RAM" if listing.ram_gb else "",
|
||||
_fmt_gb(listing.storage_gb) if listing.storage_gb else "",
|
||||
) if x]
|
||||
if listing.category == "laptops" and not listing.processor and listing.model_number:
|
||||
variant.insert(0, listing.model_number) # the part number is what tells it apart
|
||||
return " ".join(x for x in [load_brand_name(listing.brand_slug), listing.model,
|
||||
f"({', '.join(variant)})" if variant else ""] if x)
|
||||
|
||||
|
||||
def create_product(listing: Listing, ids: Dict[str, Dict[str, int]]) -> int:
|
||||
display = product_display_name(listing)
|
||||
with transaction() as conn:
|
||||
row = conn.execute(
|
||||
"""
|
||||
INSERT INTO elec.product (brand_id, category_id, family, model, model_norm, variant_key, display_name,
|
||||
ram_gb, storage_gb, processor, mpn, gtin)
|
||||
VALUES (%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s)
|
||||
ON CONFLICT (variant_key) DO UPDATE SET updated_at = now()
|
||||
RETURNING id
|
||||
""",
|
||||
(ids["brand"][listing.brand_slug], ids["category"][listing.category], listing.family,
|
||||
listing.model or listing.model_norm, listing.model_norm, listing.variant_key, display,
|
||||
listing.ram_gb, listing.storage_gb, listing.processor, listing.model_number, listing.gtin),
|
||||
).fetchone()
|
||||
return row["id"]
|
||||
|
||||
|
||||
def _fmt_gb(value: Optional[Decimal]) -> str:
|
||||
if value is None:
|
||||
return ""
|
||||
if value >= 1024 and value % 1024 == 0:
|
||||
return f"{int(value // 1024)}TB"
|
||||
return f"{format(value.normalize(), 'f')}GB"
|
||||
|
||||
|
||||
_BRAND_NAMES: Dict[str, str] = {}
|
||||
|
||||
|
||||
def load_brand_name(slug: str) -> str:
|
||||
if not _BRAND_NAMES:
|
||||
from app.electronics.reference import load_reference
|
||||
|
||||
_BRAND_NAMES.update({s: b.name for s, b in load_reference().brands.items()})
|
||||
return _BRAND_NAMES.get(slug, slug.title())
|
||||
|
||||
|
||||
def map_listing(listing_id: int, product_id: int, method: str, confidence: float, review_status: str) -> None:
|
||||
with transaction() as conn:
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO elec.product_listing_map (listing_id, product_id, method, confidence, review_status)
|
||||
VALUES (%s, %s, %s, %s, %s)
|
||||
ON CONFLICT (listing_id) DO UPDATE SET
|
||||
product_id = EXCLUDED.product_id, method = EXCLUDED.method, confidence = EXCLUDED.confidence,
|
||||
review_status = CASE WHEN elec.product_listing_map.review_status IN ('approved','rejected')
|
||||
AND elec.product_listing_map.product_id = EXCLUDED.product_id
|
||||
THEN elec.product_listing_map.review_status
|
||||
ELSE EXCLUDED.review_status END
|
||||
""",
|
||||
(listing_id, product_id, method, round(confidence, 2), review_status),
|
||||
)
|
||||
|
||||
|
||||
def merge_product_specs(product_id: int, specs: Dict[str, Any], sources: Dict[str, str], source_url: str) -> None:
|
||||
"""Add spec keys the product does not have yet. Existing values win:
|
||||
specs are only ever filled, never overwritten by a later source."""
|
||||
if not specs:
|
||||
return
|
||||
with transaction() as conn:
|
||||
row = conn.execute(
|
||||
"SELECT canonical_specs, spec_sources FROM elec.product WHERE id = %s FOR UPDATE", (product_id,)
|
||||
).fetchone()
|
||||
current, cur_src = dict(row["canonical_specs"] or {}), dict(row["spec_sources"] or {})
|
||||
changed = False
|
||||
for key, value in specs.items():
|
||||
if key not in current:
|
||||
current[key] = value
|
||||
cur_src[key] = {"url": source_url, "from": sources.get(key, "")}
|
||||
changed = True
|
||||
if changed:
|
||||
conn.execute(
|
||||
"UPDATE elec.product SET canonical_specs = %s, spec_sources = %s, updated_at = now() WHERE id = %s",
|
||||
(_json(current), _json(cur_src), product_id),
|
||||
)
|
||||
|
||||
|
||||
def add_image(product_id: int, url: str, listing_id: int, source_type: str, rank: int) -> None:
|
||||
with transaction() as conn:
|
||||
conn.execute(
|
||||
"INSERT INTO elec.product_image (product_id, url, source_listing_id, source_type, rank) "
|
||||
"VALUES (%s, %s, %s, %s, %s) ON CONFLICT (product_id, url) DO UPDATE SET validated_at = now()",
|
||||
(product_id, url, listing_id, source_type, rank),
|
||||
)
|
||||
|
||||
|
||||
def product_image_count(product_id: int) -> int:
|
||||
with connect() as conn:
|
||||
return conn.execute("SELECT count(*) AS n FROM elec.product_image WHERE product_id = %s",
|
||||
(product_id,)).fetchone()["n"]
|
||||
|
||||
|
||||
# The least a new device in the category can plausibly cost. Anything below is
|
||||
# an accessory, an EMI or an offer amount that slipped through.
|
||||
CATEGORY_MIN_PRICE = {"mobiles": 3000, "laptops": 15000}
|
||||
OUTLIER_TOLERANCE = 0.35
|
||||
|
||||
|
||||
def flag_price_outliers() -> int:
|
||||
"""Flag prices that cannot be trusted as this product's price:
|
||||
* below the category's floor (CATEGORY_MIN_PRICE);
|
||||
* a search-result price more than OUTLIER_TOLERANCE away from the price
|
||||
read off a product page for the same product;
|
||||
* with no page price, a search-result price that far from the median of
|
||||
at least three prices for the product.
|
||||
Flagged prices stay stored with their evidence; they are just never used as
|
||||
the best price. Returns the number flagged."""
|
||||
floor_cases = " ".join(f"WHEN '{k}' THEN {v}" for k, v in CATEGORY_MIN_PRICE.items())
|
||||
with transaction() as conn:
|
||||
conn.execute("UPDATE elec.source_listing SET price_outlier = FALSE WHERE price_outlier")
|
||||
cur = conn.execute(
|
||||
f"""
|
||||
WITH prices AS (
|
||||
SELECT l.id, m.product_id, l.price, l.source_type, c.slug
|
||||
FROM elec.source_listing l
|
||||
JOIN elec.product_listing_map m ON m.listing_id = l.id AND m.review_status IN ('auto','approved')
|
||||
JOIN elec.category c ON c.id = l.category_id
|
||||
WHERE l.price IS NOT NULL
|
||||
),
|
||||
ref AS (
|
||||
SELECT product_id,
|
||||
percentile_cont(0.5) WITHIN GROUP (ORDER BY price)
|
||||
FILTER (WHERE source_type <> 'search_snippet') AS page_median,
|
||||
percentile_cont(0.5) WITHIN GROUP (ORDER BY price) AS all_median,
|
||||
count(*) AS n
|
||||
FROM prices GROUP BY product_id
|
||||
)
|
||||
UPDATE elec.source_listing l SET price_outlier = TRUE
|
||||
FROM prices p JOIN ref r ON r.product_id = p.product_id
|
||||
WHERE l.id = p.id AND (
|
||||
p.price < CASE p.slug {floor_cases} ELSE 0 END
|
||||
OR (p.source_type = 'search_snippet' AND r.page_median IS NOT NULL
|
||||
AND abs(p.price - r.page_median) / r.page_median > %(tol)s)
|
||||
OR (p.source_type = 'search_snippet' AND r.page_median IS NULL AND r.n >= 3
|
||||
AND abs(p.price - r.all_median) / r.all_median > %(tol)s)
|
||||
)
|
||||
""",
|
||||
{"tol": OUTLIER_TOLERANCE},
|
||||
)
|
||||
return cur.rowcount
|
||||
|
||||
|
||||
def refresh_verification() -> Dict[str, int]:
|
||||
flag_price_outliers()
|
||||
"""A product is VERIFIED when auto/approved listings on at least two
|
||||
different sites point at it, and at least one of them is a retailer
|
||||
(so it is actually sold). Everything else stays unverified and hidden."""
|
||||
with transaction() as conn:
|
||||
conn.execute(
|
||||
"""
|
||||
WITH ev AS (
|
||||
SELECT m.product_id,
|
||||
count(DISTINCT l.site_id) AS sites,
|
||||
count(DISTINCT l.site_id) FILTER (WHERE s.kind <> 'brand_official') AS retail_sites
|
||||
FROM elec.product_listing_map m
|
||||
JOIN elec.source_listing l ON l.id = m.listing_id
|
||||
JOIN elec.site s ON s.id = l.site_id
|
||||
WHERE m.review_status IN ('auto','approved')
|
||||
GROUP BY m.product_id
|
||||
)
|
||||
UPDATE elec.product p SET
|
||||
evidence_count = coalesce(ev.sites, 0),
|
||||
verification_status = CASE
|
||||
WHEN p.verification_status = 'rejected' THEN 'rejected'
|
||||
WHEN coalesce(ev.sites, 0) >= 2 AND coalesce(ev.retail_sites, 0) >= 1 THEN 'verified'
|
||||
ELSE 'unverified' END,
|
||||
updated_at = now()
|
||||
FROM elec.product p2 LEFT JOIN ev ON ev.product_id = p2.id
|
||||
WHERE p.id = p2.id
|
||||
"""
|
||||
)
|
||||
rows = conn.execute(
|
||||
"SELECT verification_status AS s, count(*) AS n FROM elec.product GROUP BY 1"
|
||||
).fetchall()
|
||||
return {r["s"]: r["n"] for r in rows}
|
||||
|
||||
|
||||
def products_without_embedding(limit: int = 500) -> List[dict]:
|
||||
with connect() as conn:
|
||||
return list(conn.execute(
|
||||
"SELECT p.id, p.display_name, p.canonical_specs, b.name AS brand, c.name AS category "
|
||||
"FROM elec.product p JOIN elec.brand b ON b.id = p.brand_id JOIN elec.category c ON c.id = p.category_id "
|
||||
"WHERE p.embedding IS NULL AND p.verification_status = 'verified' LIMIT %s", (limit,)))
|
||||
|
||||
|
||||
def set_embedding(product_id: int, vector: List[float]) -> None:
|
||||
import numpy as np
|
||||
|
||||
with transaction() as conn:
|
||||
conn.execute("UPDATE elec.product SET embedding = %s WHERE id = %s", (np.array(vector), product_id))
|
||||
|
||||
|
||||
def review_queue(limit: int = 100) -> List[dict]:
|
||||
with connect() as conn:
|
||||
return list(conn.execute(
|
||||
"""
|
||||
SELECT m.listing_id, m.product_id, m.method, m.confidence, l.title AS listing_title, l.source_url,
|
||||
s.name AS site, p.display_name AS product
|
||||
FROM elec.product_listing_map m
|
||||
JOIN elec.source_listing l ON l.id = m.listing_id
|
||||
JOIN elec.site s ON s.id = l.site_id
|
||||
JOIN elec.product p ON p.id = m.product_id
|
||||
WHERE m.review_status = 'pending'
|
||||
ORDER BY m.confidence DESC, m.listing_id LIMIT %s
|
||||
""", (limit,)))
|
||||
|
||||
|
||||
def set_review(listing_id: int, approve: bool) -> bool:
|
||||
with transaction() as conn:
|
||||
cur = conn.execute(
|
||||
"UPDATE elec.product_listing_map SET review_status = %s, reviewed_at = now() "
|
||||
"WHERE listing_id = %s AND review_status = 'pending'",
|
||||
("approved" if approve else "rejected", listing_id),
|
||||
)
|
||||
return cur.rowcount > 0
|
||||
|
||||
|
||||
def now_utc() -> datetime:
|
||||
return datetime.now(timezone.utc)
|
||||
|
||||
|
||||
def grounding_sample(n: int = 50) -> List[dict]:
|
||||
with connect() as conn:
|
||||
return list(conn.execute(
|
||||
"SELECT id, source_url, source_type, price, evidence_text FROM elec.source_listing "
|
||||
"WHERE price IS NOT NULL ORDER BY random() LIMIT %s", (n,)))
|
||||
|
||||
|
||||
def slug(text: str) -> str:
|
||||
return slugify(text)
|
||||
0
backend/app/electronics/extract/__init__.py
Normal file
0
backend/app/electronics/extract/__init__.py
Normal file
123
backend/app/electronics/extract/html_fallback.py
Normal file
123
backend/app/electronics/extract/html_fallback.py
Normal file
@@ -0,0 +1,123 @@
|
||||
"""Product facts from page markup when there is no usable JSON-LD.
|
||||
|
||||
Only machine-readable markup is trusted for the price: OpenGraph/product meta
|
||||
tags and schema.org microdata (itemprop="price"). Free text on the page is not
|
||||
scanned for rupee amounts - a product page shows EMIs, offers and other
|
||||
products' prices, and picking the wrong one is worse than picking none.
|
||||
Spec tables (<table>, <dl>) supply specifications.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from decimal import Decimal, InvalidOperation
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
|
||||
def _dec(value: Optional[str]) -> Optional[Decimal]:
|
||||
if not value:
|
||||
return None
|
||||
try:
|
||||
return Decimal(re.sub(r"[^\d.]", "", value))
|
||||
except InvalidOperation:
|
||||
return None
|
||||
|
||||
|
||||
def _meta(soup: BeautifulSoup, *names: str) -> Optional[str]:
|
||||
for name in names:
|
||||
tag = soup.find("meta", attrs={"property": name}) or soup.find("meta", attrs={"name": name})
|
||||
if tag and tag.get("content"):
|
||||
return tag["content"].strip()
|
||||
return None
|
||||
|
||||
|
||||
def spec_tables(soup: BeautifulSoup, limit: int = 200) -> Dict[str, str]:
|
||||
specs: Dict[str, str] = {}
|
||||
for row in soup.select("table tr"):
|
||||
cells = row.find_all(["th", "td"])
|
||||
if len(cells) == 2:
|
||||
k, v = (c.get_text(" ", strip=True) for c in cells)
|
||||
if k and v and len(k) <= 60 and len(v) <= 200:
|
||||
specs.setdefault(k, v)
|
||||
if len(specs) >= limit:
|
||||
return specs
|
||||
for dl in soup.find_all("dl"):
|
||||
for dt in dl.find_all("dt"):
|
||||
dd = dt.find_next_sibling("dd")
|
||||
if dd:
|
||||
k, v = dt.get_text(" ", strip=True), dd.get_text(" ", strip=True)
|
||||
if k and v and len(k) <= 60 and len(v) <= 200:
|
||||
specs.setdefault(k, v)
|
||||
return specs
|
||||
|
||||
|
||||
def extract_page(html: str) -> Dict[str, Any]:
|
||||
soup = BeautifulSoup(html, "lxml")
|
||||
title = _meta(soup, "og:title", "twitter:title")
|
||||
if not title:
|
||||
h1 = soup.find("h1")
|
||||
title = h1.get_text(" ", strip=True) if h1 else None
|
||||
images: List[str] = []
|
||||
for name in ("og:image", "og:image:secure_url", "twitter:image"):
|
||||
v = _meta(soup, name)
|
||||
if v and v.startswith("http") and v not in images:
|
||||
images.append(v)
|
||||
|
||||
price = _dec(_meta(soup, "product:price:amount", "og:price:amount"))
|
||||
currency = _meta(soup, "product:price:currency", "og:price:currency")
|
||||
evidence = ""
|
||||
if price is not None:
|
||||
evidence = f"meta product:price:amount={price} currency={currency}"
|
||||
else:
|
||||
tag = soup.find(attrs={"itemprop": "price"})
|
||||
if tag is not None:
|
||||
raw = tag.get("content") or tag.get_text(" ", strip=True)
|
||||
price = _dec(raw)
|
||||
cur_tag = soup.find(attrs={"itemprop": "priceCurrency"})
|
||||
currency = (cur_tag.get("content") if cur_tag else None) or currency
|
||||
if price is not None:
|
||||
evidence = f'itemprop="price" {raw} currency={currency}'
|
||||
|
||||
availability = _meta(soup, "product:availability", "og:availability")
|
||||
in_stock = None
|
||||
if availability:
|
||||
low = availability.lower().replace(" ", "")
|
||||
in_stock = True if "instock" in low else False if ("outofstock" in low or "oos" == low) else None
|
||||
|
||||
return {
|
||||
"name": title,
|
||||
"images": images,
|
||||
"price": price,
|
||||
"currency": currency,
|
||||
"availability": availability,
|
||||
"in_stock": in_stock,
|
||||
"properties": spec_tables(soup),
|
||||
"evidence": evidence,
|
||||
}
|
||||
|
||||
|
||||
_STATE_RE = re.compile(
|
||||
r"<script[^>]*id=\"__NEXT_DATA__\"[^>]*>(.*?)</script>"
|
||||
r"|window\.__(?:INITIAL|PRELOADED)_STATE__\s*=\s*(\{.*?\})\s*;?\s*</script>",
|
||||
re.DOTALL,
|
||||
)
|
||||
|
||||
|
||||
def embedded_state(html: str) -> Optional[Any]:
|
||||
"""The page's server-rendered application state, when it embeds one."""
|
||||
for m in _STATE_RE.finditer(html):
|
||||
raw = m.group(1) or m.group(2)
|
||||
try:
|
||||
return json.loads(raw)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
continue
|
||||
return None
|
||||
|
||||
|
||||
def visible_text(html: str, limit: int = 6000) -> str:
|
||||
soup = BeautifulSoup(html, "lxml")
|
||||
for tag in soup(["script", "style", "noscript", "svg", "header", "footer", "nav"]):
|
||||
tag.decompose()
|
||||
return re.sub(r"\s+", " ", soup.get_text(" ", strip=True))[:limit]
|
||||
202
backend/app/electronics/extract/jsonld.py
Normal file
202
backend/app/electronics/extract/jsonld.py
Normal file
@@ -0,0 +1,202 @@
|
||||
"""schema.org Product data embedded in a page as JSON-LD.
|
||||
|
||||
This is the preferred source on any page: it is what the site publishes for
|
||||
search engines, so it is stable and states price, currency and availability
|
||||
explicitly.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from decimal import Decimal, InvalidOperation
|
||||
from typing import Any, Dict, Iterable, List, Optional
|
||||
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
_PRODUCT_TYPES = {"product", "productgroup", "productmodel", "individualproduct"}
|
||||
|
||||
|
||||
def _types(node: dict) -> set:
|
||||
t = node.get("@type")
|
||||
if isinstance(t, list):
|
||||
return {str(x).lower() for x in t}
|
||||
return {str(t).lower()} if t else set()
|
||||
|
||||
|
||||
def _walk(node: Any) -> Iterable[dict]:
|
||||
if isinstance(node, dict):
|
||||
yield node
|
||||
for v in node.values():
|
||||
yield from _walk(v)
|
||||
elif isinstance(node, list):
|
||||
for item in node:
|
||||
yield from _walk(item)
|
||||
|
||||
|
||||
def json_ld_blocks(html: str) -> List[Any]:
|
||||
soup = BeautifulSoup(html, "lxml")
|
||||
blocks = []
|
||||
for tag in soup.find_all("script", attrs={"type": re.compile(r"ld\+json", re.I)}):
|
||||
raw = (tag.string or tag.get_text() or "").strip()
|
||||
if not raw:
|
||||
continue
|
||||
try:
|
||||
blocks.append(json.loads(raw))
|
||||
except json.JSONDecodeError:
|
||||
# Some sites put several objects or trailing commas in one tag.
|
||||
try:
|
||||
blocks.append(json.loads(re.sub(r",\s*([}\]])", r"\1", raw)))
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
return blocks
|
||||
|
||||
|
||||
def _dec(value: Any) -> Optional[Decimal]:
|
||||
if value is None or value == "":
|
||||
return None
|
||||
try:
|
||||
return Decimal(str(value).replace(",", "").strip())
|
||||
except InvalidOperation:
|
||||
return None
|
||||
|
||||
|
||||
def _text(value: Any) -> Optional[str]:
|
||||
if isinstance(value, dict):
|
||||
value = value.get("name") or value.get("@value")
|
||||
if isinstance(value, list):
|
||||
value = value[0] if value else None
|
||||
return str(value).strip() if value not in (None, "") else None
|
||||
|
||||
|
||||
def _images(value: Any) -> List[str]:
|
||||
out: List[str] = []
|
||||
for v in value if isinstance(value, list) else [value]:
|
||||
if isinstance(v, dict):
|
||||
v = v.get("url") or v.get("contentUrl")
|
||||
if isinstance(v, str) and v.startswith(("http://", "https://")):
|
||||
out.append(v)
|
||||
return out
|
||||
|
||||
|
||||
def _availability(value: Any) -> tuple:
|
||||
text = (_text(value) or "").lower()
|
||||
if not text:
|
||||
return None, None
|
||||
if "instock" in text or "limitedavailability" in text or "onlineonly" in text:
|
||||
return "InStock", True
|
||||
if any(k in text for k in ("outofstock", "soldout", "discontinued", "preorder", "presale")):
|
||||
return text.rsplit("/", 1)[-1], False
|
||||
return text.rsplit("/", 1)[-1], None
|
||||
|
||||
|
||||
def _offer(offers: Any) -> Dict[str, Any]:
|
||||
"""The price/availability of the product's (lowest) offer."""
|
||||
candidates = offers if isinstance(offers, list) else [offers]
|
||||
best: Dict[str, Any] = {}
|
||||
for o in candidates:
|
||||
if not isinstance(o, dict):
|
||||
continue
|
||||
price = _dec(o.get("price"))
|
||||
if price is None:
|
||||
price = _dec(o.get("lowPrice"))
|
||||
if price is None and isinstance(o.get("priceSpecification"), dict):
|
||||
price = _dec(o["priceSpecification"].get("price"))
|
||||
currency = _text(o.get("priceCurrency")) or (
|
||||
_text(o["priceSpecification"].get("priceCurrency")) if isinstance(o.get("priceSpecification"), dict) else None
|
||||
)
|
||||
availability, in_stock = _availability(o.get("availability"))
|
||||
entry = {"price": price, "currency": currency, "availability": availability, "in_stock": in_stock,
|
||||
"raw": {k: o.get(k) for k in ("price", "lowPrice", "priceCurrency", "availability") if k in o}}
|
||||
if price is not None and (not best or best.get("price") is None or price < best["price"]):
|
||||
best = entry
|
||||
elif not best:
|
||||
best = entry
|
||||
return best
|
||||
|
||||
|
||||
MAX_REVIEWS_PER_PAGE = 30
|
||||
|
||||
|
||||
def _review_rating(value: Any) -> Optional[Decimal]:
|
||||
"""A reviewer's star rating, rescaled to 0-5 when the page uses another scale."""
|
||||
if not isinstance(value, dict):
|
||||
return None
|
||||
rating = _dec(value.get("ratingValue"))
|
||||
if rating is None:
|
||||
return None
|
||||
best = _dec(value.get("bestRating")) or Decimal(5)
|
||||
if best <= 0:
|
||||
return None
|
||||
if best != 5:
|
||||
rating = rating * Decimal(5) / best
|
||||
if not (Decimal(0) <= rating <= Decimal(5)):
|
||||
return None
|
||||
return rating.quantize(Decimal("0.1"))
|
||||
|
||||
|
||||
def _reviews(node: dict) -> List[Dict[str, Any]]:
|
||||
"""Customer reviews published on the Product node (schema.org Review).
|
||||
|
||||
Only reviews with text are kept - a bare star with no words is not
|
||||
something a reader can weigh. Nothing is paraphrased or summarised: body,
|
||||
title and author are the page's own strings.
|
||||
"""
|
||||
raw = node.get("review") or node.get("reviews") or []
|
||||
out: List[Dict[str, Any]] = []
|
||||
for r in raw if isinstance(raw, list) else [raw]:
|
||||
if not isinstance(r, dict):
|
||||
continue
|
||||
body = _text(r.get("reviewBody")) or _text(r.get("description"))
|
||||
if not body:
|
||||
continue
|
||||
out.append({
|
||||
"author": _text(r.get("author")),
|
||||
"rating": _review_rating(r.get("reviewRating")),
|
||||
"title": _text(r.get("name")) or _text(r.get("headline")),
|
||||
"body": body[:4000],
|
||||
"review_date": _text(r.get("datePublished")) or _text(r.get("dateCreated")),
|
||||
})
|
||||
if len(out) >= MAX_REVIEWS_PER_PAGE:
|
||||
break
|
||||
return out
|
||||
|
||||
|
||||
def extract_products(html: str) -> List[Dict[str, Any]]:
|
||||
"""All schema.org Product nodes on the page, flattened to plain fields."""
|
||||
products: List[Dict[str, Any]] = []
|
||||
for block in json_ld_blocks(html):
|
||||
for node in _walk(block):
|
||||
if not (_types(node) & _PRODUCT_TYPES):
|
||||
continue
|
||||
name = _text(node.get("name"))
|
||||
if not name:
|
||||
continue
|
||||
offer = _offer(node.get("offers")) if node.get("offers") else {}
|
||||
if not offer and isinstance(node.get("hasVariant"), list):
|
||||
offer = _offer([v.get("offers") for v in node["hasVariant"] if isinstance(v, dict) and v.get("offers")])
|
||||
props = {}
|
||||
for p in node.get("additionalProperty") or []:
|
||||
if isinstance(p, dict) and p.get("name") and p.get("value") not in (None, ""):
|
||||
props[str(p["name"])] = str(p["value"])
|
||||
rating = node.get("aggregateRating") if isinstance(node.get("aggregateRating"), dict) else {}
|
||||
products.append({
|
||||
"name": name,
|
||||
"brand": _text(node.get("brand")),
|
||||
"sku": _text(node.get("sku")) or _text(node.get("productID")),
|
||||
"mpn": _text(node.get("mpn")),
|
||||
"gtin": next((_text(node.get(k)) for k in ("gtin13", "gtin", "gtin12", "gtin14", "gtin8") if node.get(k)), None),
|
||||
"color": _text(node.get("color")),
|
||||
"images": _images(node.get("image")),
|
||||
"description": _text(node.get("description")),
|
||||
"price": offer.get("price"),
|
||||
"currency": offer.get("currency"),
|
||||
"availability": offer.get("availability"),
|
||||
"in_stock": offer.get("in_stock"),
|
||||
# Sites publish 0 for "no ratings yet"; that is not a rating.
|
||||
"rating": (_dec(rating.get("ratingValue")) or None),
|
||||
"review_count": int(_dec(rating.get("reviewCount") or rating.get("ratingCount")) or 0) or None,
|
||||
"reviews": _reviews(node),
|
||||
"properties": props,
|
||||
"evidence": json.dumps({"name": name, "offers": offer.get("raw")}, default=str)[:1500],
|
||||
})
|
||||
return products
|
||||
207
backend/app/electronics/extract/serp_parser.py
Normal file
207
backend/app/electronics/extract/serp_parser.py
Normal file
@@ -0,0 +1,207 @@
|
||||
"""Read prices and stock state out of text we did not render ourselves:
|
||||
search-result titles/snippets, and visible page text.
|
||||
|
||||
The rules lean hard towards NOT returning a price. A snippet usually carries
|
||||
several rupee amounts - the selling price, the MRP, an EMI, a bank discount, an
|
||||
exchange value, "₹X off" - and taking the wrong one is worse than taking none.
|
||||
An amount is only a price when nothing around it says it is something else.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from decimal import Decimal, InvalidOperation
|
||||
from typing import List, Optional
|
||||
|
||||
PRICE_MIN = Decimal("500")
|
||||
PRICE_MAX = Decimal("1000000")
|
||||
|
||||
# ₹ / Rs / Rs. / INR followed by an amount with Indian (1,29,999) or western
|
||||
# (129,999) grouping, or none.
|
||||
_AMOUNT = r"(\d{1,3}(?:,\d{2,3})+(?:\.\d{1,2})?|\d+(?:\.\d{1,2})?)"
|
||||
_MONEY_RE = re.compile(r"(?:₹|\bRs\.?|\bINR)\s?" + _AMOUNT, re.IGNORECASE)
|
||||
|
||||
# Words that make an amount something other than the selling price.
|
||||
_REJECT_BEFORE = re.compile(
|
||||
r"(?:emi|save|saving|savings|cashback|cash\s*back|exchange|bank|discount|coupon|"
|
||||
r"extra|instant|up\s*to|upto|flat|off\s+upto|worth|delivery|shipping|fee|charges?|"
|
||||
r"starting|starts|from|onwards|min(?:imum)?|as\s+low\s+as|down\s*payment|per\s+month)\W*$",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_REJECT_AFTER = re.compile(
|
||||
r"^\W{0,3}(?:off\b|/\s*m(?:o|onth)?\b|per\s+month|p\.?m\.?\b|a\s+month|emi\b|/-?\s*emi|"
|
||||
r"cashback|discount|savings?|onwards|\+\s*shipping|delivery)",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_MRP_BEFORE = re.compile(r"(?:m\.?\s?r\.?\s?p\.?|list\s+price|was|original\s+price)[\s:]*$", re.IGNORECASE)
|
||||
_RANGE_BETWEEN = re.compile(r"^\s*(?:-|–|—|to)\s*$", re.IGNORECASE)
|
||||
|
||||
_OUT_OF_STOCK = re.compile(
|
||||
r"\b(?:out\s+of\s+stock|currently\s+unavailable|sold\s+out|coming\s+soon|notify\s+me|"
|
||||
r"temporarily\s+unavailable|not\s+available)\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_IN_STOCK = re.compile(r"\b(?:in\s+stock|available\s+now|buy\s+now|add\s+to\s+cart)\b", re.IGNORECASE)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Amount:
|
||||
value: Decimal
|
||||
kind: str # price | mrp | rejected
|
||||
reason: str
|
||||
start: int
|
||||
end: int
|
||||
raw: str
|
||||
|
||||
|
||||
def parse_amount(raw: str) -> Optional[Decimal]:
|
||||
try:
|
||||
value = Decimal(raw.replace(",", ""))
|
||||
except InvalidOperation:
|
||||
return None
|
||||
return value
|
||||
|
||||
|
||||
def find_amounts(text: str) -> List[Amount]:
|
||||
"""Every rupee amount in `text`, each classified as price, mrp or rejected."""
|
||||
out: List[Amount] = []
|
||||
if not text:
|
||||
return out
|
||||
matches = list(_MONEY_RE.finditer(text))
|
||||
for i, m in enumerate(matches):
|
||||
value = parse_amount(m.group(1))
|
||||
if value is None:
|
||||
continue
|
||||
before = text[max(0, m.start() - 28): m.start()]
|
||||
after = text[m.end(): m.end() + 22]
|
||||
kind, reason = "price", ""
|
||||
if _MRP_BEFORE.search(before):
|
||||
kind, reason = "mrp", "labelled MRP"
|
||||
elif _REJECT_BEFORE.search(before):
|
||||
kind, reason = "rejected", f"preceded by {_REJECT_BEFORE.search(before).group(0).strip()!r}"
|
||||
elif _REJECT_AFTER.search(after):
|
||||
kind, reason = "rejected", f"followed by {_REJECT_AFTER.search(after).group(0).strip()!r}"
|
||||
# A range ("₹10,999 - ₹12,999") names no single price.
|
||||
if kind == "price":
|
||||
if i + 1 < len(matches) and _RANGE_BETWEEN.match(text[m.end(): matches[i + 1].start()]):
|
||||
kind, reason = "rejected", "start of a price range"
|
||||
elif i > 0 and _RANGE_BETWEEN.match(text[matches[i - 1].end(): m.start()]):
|
||||
kind, reason = "rejected", "end of a price range"
|
||||
if kind != "rejected" and not (PRICE_MIN <= value <= PRICE_MAX):
|
||||
kind, reason = "rejected", "outside plausible range"
|
||||
out.append(Amount(value, kind, reason, m.start(), m.end(), m.group(0)))
|
||||
return out
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PriceReading:
|
||||
price: Optional[Decimal]
|
||||
mrp: Optional[Decimal]
|
||||
evidence: str # the exact substring the price was read from ("" if none)
|
||||
|
||||
|
||||
def read_price(text: str) -> PriceReading:
|
||||
"""The single selling price stated in `text`, or None.
|
||||
|
||||
If the text states two different unlabelled prices, it is ambiguous (a
|
||||
listing page snippet often shows several variants) and None is returned.
|
||||
"""
|
||||
amounts = find_amounts(text)
|
||||
prices = [a for a in amounts if a.kind == "price"]
|
||||
mrps = [a for a in amounts if a.kind == "mrp"]
|
||||
distinct = {a.value for a in prices}
|
||||
price: Optional[Decimal] = None
|
||||
evidence = ""
|
||||
if len(distinct) == 1:
|
||||
price = prices[0].value
|
||||
evidence = prices[0].raw
|
||||
mrp = mrps[0].value if mrps else None
|
||||
if price is not None and mrp is not None and mrp < price:
|
||||
mrp = None # an "MRP" below the selling price was misread; drop it
|
||||
return PriceReading(price, mrp, evidence)
|
||||
|
||||
|
||||
def read_stock(text: str) -> Optional[bool]:
|
||||
"""True/False only when the text says so; None when it does not."""
|
||||
if not text:
|
||||
return None
|
||||
if _OUT_OF_STOCK.search(text):
|
||||
return False
|
||||
if _IN_STOCK.search(text):
|
||||
return True
|
||||
return None
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RatingReading:
|
||||
rating: Optional[Decimal]
|
||||
review_count: Optional[int]
|
||||
evidence: str # the exact substring the rating was read from ("" if none)
|
||||
|
||||
|
||||
# Only ratings the text states explicitly on a 5-point scale:
|
||||
# "4.3 out of 5 stars", "Rating: 4.3/5", "Rated 4.3 / 5", "4.3★", "4.3 ★ (1,234 ratings)"
|
||||
_RATING_PATTERNS = (
|
||||
re.compile(r"\b([0-5](?:\.\d{1,2})?)\s*out\s+of\s*5(?:\.0)?\b(?:\s*stars?)?", re.IGNORECASE),
|
||||
# "x/5" only with a rating word before it or "stars" after it - a bare
|
||||
# "1/5" is as likely a sensor size or a fraction.
|
||||
re.compile(r"\brat(?:ing|ed)\s*[:\-]?\s*([0-5](?:\.\d{1,2})?)\s*/\s*5(?:\.0)?\b", re.IGNORECASE),
|
||||
re.compile(r"\b([0-5](?:\.\d{1,2})?)\s*/\s*5(?:\.0)?\s*stars?\b", re.IGNORECASE),
|
||||
re.compile(r"\b([0-5](?:\.\d{1,2})?)\s*(?:★|☆|⭐)"),
|
||||
re.compile(r"\brat(?:ing|ed)\s*[:\-]?\s*([0-5](?:\.\d{1,2})?)\s*(?:stars?|★)", re.IGNORECASE),
|
||||
)
|
||||
_RATING_COUNT = re.compile(
|
||||
r"^[\s()\-|·,.:]*(?:stars?)?[\s()\-|·,.:]*(\d{1,3}(?:,\d{2,3})+|\d+)\s*(?:customer\s+)?(?:ratings?|reviews?|votes?)\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
def read_rating(text: str) -> RatingReading:
|
||||
"""The product rating a search title/snippet states, or None.
|
||||
|
||||
Only an explicit "x out of 5" / "x/5" / "x★" statement counts; bare
|
||||
numbers never do. If the text states two different ratings it is
|
||||
ambiguous (several products on one results page) and None is returned.
|
||||
"""
|
||||
if not text:
|
||||
return RatingReading(None, None, "")
|
||||
found = []
|
||||
for pattern in _RATING_PATTERNS:
|
||||
for m in pattern.finditer(text):
|
||||
try:
|
||||
value = Decimal(m.group(1))
|
||||
except InvalidOperation:
|
||||
continue
|
||||
if Decimal(0) < value <= Decimal(5):
|
||||
found.append((value, m))
|
||||
if not found or len({v for v, _ in found}) != 1:
|
||||
return RatingReading(None, None, "")
|
||||
value, m = min(found, key=lambda f: f[1].start())
|
||||
count = None
|
||||
tail = _RATING_COUNT.match(text[m.end(): m.end() + 40])
|
||||
if tail:
|
||||
count = int(tail.group(1).replace(",", ""))
|
||||
evidence = text[m.start(): m.end() + (tail.end() if tail else 0)].strip()
|
||||
return RatingReading(value, count, evidence)
|
||||
|
||||
|
||||
# Titles returned by search engines carry the site name; it is not part of the
|
||||
# product title.
|
||||
_TITLE_SUFFIX = re.compile(
|
||||
r"\s*(?:[|\-–:]\s*)?(?:buy\s+online.*|online\s+at\s+best\s+price.*|"
|
||||
r"at\s+best\s+price.*|price\s+in\s+india.*|"
|
||||
r"amazon\.in.*|flipkart(?:\.com)?.*|croma.*|reliance\s+digital.*|vijay\s+sales.*|"
|
||||
r"tata\s+cliq.*|poorvika.*|sangeetha.*|vasanth.*|viveks.*)$",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_TITLE_PREFIX = re.compile(r"^(?:buy\s+|amazon\.in\s*:\s*)", re.IGNORECASE)
|
||||
|
||||
|
||||
def clean_result_title(title: str) -> str:
|
||||
t = (title or "").strip()
|
||||
# Engines truncate with "..." and sometimes run several results' titles
|
||||
# together after it; everything past the first ellipsis is not this page.
|
||||
t = re.split(r"\s*(?:\.\.\.|…)", t, maxsplit=1)[0]
|
||||
t = _TITLE_PREFIX.sub("", t)
|
||||
t = _TITLE_SUFFIX.sub("", t)
|
||||
return t.strip(" -|:–")
|
||||
0
backend/app/electronics/match/__init__.py
Normal file
0
backend/app/electronics/match/__init__.py
Normal file
124
backend/app/electronics/match/matcher.py
Normal file
124
backend/app/electronics/match/matcher.py
Normal file
@@ -0,0 +1,124 @@
|
||||
"""Link a listing to its canonical product (one real-world variant).
|
||||
|
||||
From most to least certain:
|
||||
1. GTIN - same barcode -> auto
|
||||
2. MPN - same manufacturer part number (laptops) -> auto
|
||||
3. variant key - same brand, model, RAM and storage -> auto
|
||||
4. fuzzy - model names ≥ AUTO_RATIO similar AND every
|
||||
hard attribute (RAM, storage, processor) equal -> auto
|
||||
≥ REVIEW_RATIO -> pending (review queue)
|
||||
5. otherwise a new product is created for the variant.
|
||||
|
||||
A listing that states too little to identify a variant (no storage on a
|
||||
phone title, for example) is stored but not linked to any product.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from decimal import Decimal
|
||||
from typing import List, Optional
|
||||
|
||||
from rapidfuzz import fuzz
|
||||
|
||||
from app.electronics.normalise.title_parser import laptop_line, processor_is_specific
|
||||
|
||||
AUTO_RATIO = 92
|
||||
REVIEW_RATIO = 85
|
||||
|
||||
|
||||
@dataclass
|
||||
class MatchDecision:
|
||||
product_id: Optional[int] # None -> create a new product
|
||||
method: str
|
||||
confidence: float
|
||||
review_status: str
|
||||
|
||||
|
||||
def _eq(a: Optional[Decimal], b: Optional[Decimal]) -> bool:
|
||||
if a is None or b is None:
|
||||
return a is None and b is None
|
||||
return Decimal(a) == Decimal(b)
|
||||
|
||||
|
||||
def _number_tokens(model_norm: str) -> set:
|
||||
"""Tokens that carry a digit ("s25", "a37", "15", "2a", "8a"). Two models
|
||||
whose number tokens differ are different products, however similar the
|
||||
rest of the name is: Galaxy S25 vs S26, A27 vs A37, iPhone 15 vs 16."""
|
||||
return {t for t in (model_norm or "").split() if any(ch.isdigit() for ch in t)}
|
||||
|
||||
|
||||
def _lines_compatible(a: str, b: str) -> bool:
|
||||
"""One model line is the other plus/minus extra words, and they agree on
|
||||
every number token they both carry ("15" is not "15s", "slim 3" is not
|
||||
"slim 5")."""
|
||||
ta, tb = set(a.split()), set(b.split())
|
||||
return bool(ta and tb) and (ta <= tb or tb <= ta)
|
||||
|
||||
|
||||
def decide(listing, candidates: List[dict]) -> Optional[MatchDecision]:
|
||||
"""`listing` is a models.Listing with variant_key/model_norm set;
|
||||
`candidates` are product rows of the same brand and category."""
|
||||
if not listing.variant_key:
|
||||
return None
|
||||
if listing.gtin:
|
||||
for c in candidates:
|
||||
if c.get("gtin") and c["gtin"] == listing.gtin:
|
||||
return MatchDecision(c["id"], "gtin", 0.99, "auto")
|
||||
if listing.model_number:
|
||||
for c in candidates:
|
||||
if c.get("mpn") and c["mpn"].lower() == listing.model_number.lower():
|
||||
return MatchDecision(c["id"], "mpn", 0.97, "auto")
|
||||
for c in candidates:
|
||||
if c["variant_key"] == listing.variant_key:
|
||||
return MatchDecision(c["id"], "variant_key", 0.95, "auto")
|
||||
|
||||
# Laptops: the same configuration (exact CPU model, RAM, storage) within a
|
||||
# compatible model line - "ideapad slim 3" and "ideapad slim 3 15amn8" -
|
||||
# is the same product. Two compatible candidates means the title is too
|
||||
# vague to choose ("pavilion" vs "pavilion 14" and "pavilion 15"): review.
|
||||
if listing.category == "laptops" and processor_is_specific(listing.processor) \
|
||||
and listing.ram_gb is not None and listing.storage_gb is not None:
|
||||
line = laptop_line(listing.model_norm)
|
||||
same_config = [
|
||||
c for c in candidates
|
||||
if c.get("processor") == listing.processor
|
||||
and _eq(c.get("ram_gb"), listing.ram_gb) and _eq(c.get("storage_gb"), listing.storage_gb)
|
||||
and _lines_compatible(line, laptop_line(c["model_norm"]))
|
||||
]
|
||||
if len(same_config) == 1:
|
||||
return MatchDecision(same_config[0]["id"], "variant_key", 0.85, "auto")
|
||||
if len(same_config) > 1:
|
||||
return MatchDecision(same_config[0]["id"], "variant_key", 0.6, "pending")
|
||||
|
||||
# One side does not state the RAM ("Apple iPhone 15 (128 GB)"). Same model
|
||||
# and storage with exactly one candidate is the same variant; with several
|
||||
# candidates it is ambiguous and goes to review.
|
||||
same_model = [c for c in candidates
|
||||
if c["model_norm"] == listing.model_norm and _eq(c.get("storage_gb"), listing.storage_gb)
|
||||
and (c.get("ram_gb") is None) != (listing.ram_gb is None)]
|
||||
if len(same_model) == 1:
|
||||
return MatchDecision(same_model[0]["id"], "variant_key", 0.85, "auto")
|
||||
if len(same_model) > 1:
|
||||
return MatchDecision(same_model[0]["id"], "variant_key", 0.6, "pending")
|
||||
|
||||
best, best_score = None, 0.0
|
||||
for c in candidates:
|
||||
if not (_eq(c.get("storage_gb"), listing.storage_gb) and _eq(c.get("ram_gb"), listing.ram_gb)):
|
||||
continue
|
||||
if listing.category == "laptops" and (c.get("processor") or listing.processor) and c.get("processor") != listing.processor:
|
||||
continue
|
||||
if _number_tokens(c["model_norm"]) != _number_tokens(listing.model_norm):
|
||||
continue
|
||||
score = fuzz.token_set_ratio(c["model_norm"], listing.model_norm or "")
|
||||
# token_set_ratio treats "galaxy s24" and "galaxy s24 ultra" as a
|
||||
# subset match (100). Different words mean different models.
|
||||
extra = set((listing.model_norm or "").split()) ^ set(c["model_norm"].split())
|
||||
if extra:
|
||||
score = min(score, fuzz.ratio(c["model_norm"], listing.model_norm or ""))
|
||||
if score > best_score:
|
||||
best, best_score = c, score
|
||||
if best is not None and best_score >= AUTO_RATIO:
|
||||
return MatchDecision(best["id"], "fuzzy", round(best_score / 100 * 0.9, 2), "auto")
|
||||
if best is not None and best_score >= REVIEW_RATIO:
|
||||
return MatchDecision(best["id"], "fuzzy", round(best_score / 100 * 0.8, 2), "pending")
|
||||
return MatchDecision(None, "variant_key", 0.9, "auto")
|
||||
105
backend/app/electronics/match/rematch.py
Normal file
105
backend/app/electronics/match/rematch.py
Normal file
@@ -0,0 +1,105 @@
|
||||
"""Rebuild canonical products for a category from the listings already stored.
|
||||
|
||||
Products and listing links are derived data: every fact lives on the listing
|
||||
(title, snippet evidence, specs, URL). When the parsing or matching rules
|
||||
improve, this re-runs them over the stored listings - no network requests -
|
||||
and keeps each image attached to the listing it was found on.
|
||||
|
||||
Review decisions (approved/rejected links) are lost, because the products they
|
||||
pointed at are rebuilt; uncertain matches simply come back to the queue.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from decimal import Decimal
|
||||
from typing import Dict, List
|
||||
|
||||
from app.electronics.db import repository as repo
|
||||
from app.electronics.db.connection import connect, transaction
|
||||
from app.electronics.match.matcher import decide
|
||||
from app.electronics.models import Listing
|
||||
from app.electronics.normalise.title_parser import fill_from_context, parse_title, variant_key
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_ORDER = {"brand_official": 0, "scraped_page": 1, "search_snippet": 2}
|
||||
|
||||
|
||||
def _listing_from_row(row: dict, category: str) -> Listing:
|
||||
parsed = parse_title(row["title"], category, expected_brand=row["brand_slug"])
|
||||
snippet = ""
|
||||
if row["source_type"] == "search_snippet" and " — " in row["evidence_text"]:
|
||||
snippet = row["evidence_text"].split(" — ", 1)[1].split(" || ", 1)[0]
|
||||
raw = row["specs_raw"] or {}
|
||||
spec_texts = tuple(str(v) for k, v in raw.items() if "processor" in k.lower() or "cpu" in k.lower())
|
||||
spec_texts += (str((row["specs"] or {}).get("processor") or ""),)
|
||||
fill_from_context(parsed, category, snippet=snippet, spec_texts=spec_texts)
|
||||
l = Listing(
|
||||
site_domain=row["domain"], source_sku=row["source_sku"], source_url=row["source_url"],
|
||||
source_type=row["source_type"], brand_slug=row["brand_slug"], category=category,
|
||||
title=row["title"], evidence_text=row["evidence_text"], confidence=float(row["confidence"]),
|
||||
parser=row["parser"], family=parsed.brand.family if parsed.brand else row["family"],
|
||||
model=parsed.model, model_number=row["model_number"] or parsed.mpn,
|
||||
ram_gb=parsed.ram_gb, storage_gb=parsed.storage_gb, colour=row["colour"],
|
||||
gtin=row["gtin"], specs=row["specs"] or {},
|
||||
)
|
||||
l.model_norm, l.processor = parsed.model_norm, parsed.processor
|
||||
l.variant_key = variant_key(parsed, category) if parsed.brand else None
|
||||
return l
|
||||
|
||||
|
||||
def rematch(category: str) -> Dict[str, int]:
|
||||
stats: Dict[str, int] = {"listings": 0, "linked": 0, "pending": 0, "unlinked": 0, "products": 0, "images": 0}
|
||||
with connect() as conn:
|
||||
rows = conn.execute(
|
||||
"""
|
||||
SELECT l.*, b.slug AS brand_slug, s.domain
|
||||
FROM elec.source_listing l
|
||||
JOIN elec.brand b ON b.id = l.brand_id
|
||||
JOIN elec.site s ON s.id = l.site_id
|
||||
JOIN elec.category c ON c.id = l.category_id
|
||||
WHERE c.slug = %s
|
||||
""",
|
||||
(category,),
|
||||
).fetchall()
|
||||
images = conn.execute(
|
||||
"""
|
||||
SELECT i.url, i.source_listing_id, i.source_type, i.rank FROM elec.product_image i
|
||||
JOIN elec.product p ON p.id = i.product_id JOIN elec.category c ON c.id = p.category_id
|
||||
WHERE c.slug = %s
|
||||
""",
|
||||
(category,),
|
||||
).fetchall()
|
||||
with transaction() as conn:
|
||||
# Maps and images cascade from the products.
|
||||
conn.execute(
|
||||
"DELETE FROM elec.product p USING elec.category c WHERE c.id = p.category_id AND c.slug = %s",
|
||||
(category,),
|
||||
)
|
||||
|
||||
ids = repo.id_maps()
|
||||
product_of_listing: Dict[int, int] = {}
|
||||
rows.sort(key=lambda r: (_ORDER.get(r["source_type"], 9), r["id"]))
|
||||
for row in rows:
|
||||
stats["listings"] += 1
|
||||
listing = _listing_from_row(row, category)
|
||||
decision = decide(listing, repo.product_candidates(listing.brand_slug, category))
|
||||
if decision is None:
|
||||
stats["unlinked"] += 1
|
||||
continue
|
||||
product_id = decision.product_id or repo.create_product(listing, ids)
|
||||
stats["products"] += decision.product_id is None
|
||||
repo.map_listing(row["id"], product_id, decision.method, decision.confidence, decision.review_status)
|
||||
product_of_listing[row["id"]] = product_id
|
||||
if decision.review_status == "pending":
|
||||
stats["pending"] += 1
|
||||
else:
|
||||
stats["linked"] += 1
|
||||
repo.merge_product_specs(product_id, listing.specs, {}, listing.source_url)
|
||||
for img in images:
|
||||
pid = product_of_listing.get(img["source_listing_id"])
|
||||
if pid is not None:
|
||||
repo.add_image(pid, img["url"], img["source_listing_id"], img["source_type"], img["rank"])
|
||||
stats["images"] += 1
|
||||
stats.update({f"products_{k}": v for k, v in repo.refresh_verification().items()})
|
||||
return stats
|
||||
68
backend/app/electronics/models.py
Normal file
68
backend/app/electronics/models.py
Normal file
@@ -0,0 +1,68 @@
|
||||
"""The record a collector produces for one product page / search result."""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from decimal import Decimal
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
SOURCE_TYPES = ("scraped_page", "search_snippet", "brand_official")
|
||||
|
||||
|
||||
@dataclass
|
||||
class Listing:
|
||||
site_domain: str
|
||||
source_sku: str
|
||||
source_url: str
|
||||
source_type: str
|
||||
brand_slug: str
|
||||
category: str
|
||||
title: str
|
||||
evidence_text: str
|
||||
confidence: float
|
||||
parser: str
|
||||
family: Optional[str] = None
|
||||
model: Optional[str] = None
|
||||
model_number: Optional[str] = None
|
||||
ram_gb: Optional[Decimal] = None
|
||||
storage_gb: Optional[Decimal] = None
|
||||
colour: Optional[str] = None
|
||||
price: Optional[Decimal] = None
|
||||
mrp: Optional[Decimal] = None
|
||||
availability: Optional[str] = None
|
||||
in_stock: Optional[bool] = None
|
||||
pincode: Optional[str] = None
|
||||
pincode_applied: bool = False
|
||||
rating: Optional[Decimal] = None
|
||||
review_count: Optional[int] = None
|
||||
# Customer reviews the page itself publishes (schema.org Review); stored
|
||||
# in elec.listing_review, not on the listing row.
|
||||
reviews: List[Dict[str, Any]] = field(default_factory=list)
|
||||
gtin: Optional[str] = None
|
||||
image_urls: List[str] = field(default_factory=list)
|
||||
specs_raw: Dict[str, Any] = field(default_factory=dict)
|
||||
specs: Dict[str, Any] = field(default_factory=dict)
|
||||
spec_sources: Dict[str, str] = field(default_factory=dict)
|
||||
search_query: Optional[str] = None
|
||||
content_hash: Optional[str] = None
|
||||
# Not stored on the listing; used for matching.
|
||||
variant_key: Optional[str] = None
|
||||
model_norm: Optional[str] = None
|
||||
processor: Optional[str] = None
|
||||
|
||||
def validate(self) -> None:
|
||||
"""The anti-fabrication contract, checked before anything is written."""
|
||||
if self.source_type not in SOURCE_TYPES:
|
||||
raise ValueError(f"bad source_type {self.source_type!r}")
|
||||
if not self.source_url.startswith(("http://", "https://")):
|
||||
raise ValueError("listing without a real source URL")
|
||||
if not self.evidence_text.strip():
|
||||
raise ValueError("listing without evidence text")
|
||||
if self.price is not None:
|
||||
if not (Decimal(500) <= self.price <= Decimal(1000000)):
|
||||
raise ValueError(f"implausible price {self.price}")
|
||||
if self.mrp is not None and self.price is not None and self.mrp < self.price:
|
||||
self.mrp = None
|
||||
if self.pincode_applied and not self.pincode:
|
||||
raise ValueError("pincode_applied without a pincode")
|
||||
if not 0 <= self.confidence <= 1:
|
||||
raise ValueError("confidence out of range")
|
||||
0
backend/app/electronics/net/__init__.py
Normal file
0
backend/app/electronics/net/__init__.py
Normal file
61
backend/app/electronics/net/breaker.py
Normal file
61
backend/app/electronics/net/breaker.py
Normal file
@@ -0,0 +1,61 @@
|
||||
"""Per-host circuit breaker.
|
||||
|
||||
One 403, 429, 503 or CAPTCHA page opens the breaker for that host for
|
||||
ELEC_BREAKER_COOLDOWN_HOURS. While it is open the host is not requested at all
|
||||
and its products are collected from web search results instead. There is no
|
||||
retry-with-a-different-identity: a block is an answer.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import threading
|
||||
import time
|
||||
from typing import Callable, Dict, Optional, Tuple
|
||||
|
||||
from app.infrastructure.settings import ELEC_BREAKER_COOLDOWN_HOURS
|
||||
|
||||
|
||||
class CircuitBreaker:
|
||||
def __init__(
|
||||
self,
|
||||
cooldown_seconds: float = ELEC_BREAKER_COOLDOWN_HOURS * 3600,
|
||||
on_trip: Optional[Callable[[str, str, float], None]] = None,
|
||||
clock: Callable[[], float] = time.time,
|
||||
) -> None:
|
||||
self.cooldown = cooldown_seconds
|
||||
self.on_trip = on_trip
|
||||
self._clock = clock
|
||||
self._open: Dict[str, Tuple[float, str]] = {}
|
||||
self._lock = threading.Lock()
|
||||
|
||||
@staticmethod
|
||||
def _key(host: str) -> str:
|
||||
host = host.lower()
|
||||
return host[4:] if host.startswith("www.") else host
|
||||
|
||||
def preload(self, host: str, until_epoch: float, reason: str) -> None:
|
||||
"""Restore a breaker that was opened in an earlier run (elec.site)."""
|
||||
if until_epoch > self._clock():
|
||||
with self._lock:
|
||||
self._open[self._key(host)] = (until_epoch, reason)
|
||||
|
||||
def trip(self, host: str, reason: str) -> None:
|
||||
until = self._clock() + self.cooldown
|
||||
with self._lock:
|
||||
self._open[self._key(host)] = (until, reason)
|
||||
if self.on_trip:
|
||||
self.on_trip(self._key(host), reason, until)
|
||||
|
||||
def is_open(self, host: str) -> bool:
|
||||
key = self._key(host)
|
||||
with self._lock:
|
||||
entry = self._open.get(key)
|
||||
if not entry:
|
||||
return False
|
||||
if entry[0] <= self._clock():
|
||||
del self._open[key]
|
||||
return False
|
||||
return True
|
||||
|
||||
def reason(self, host: str) -> Optional[str]:
|
||||
entry = self._open.get(self._key(host))
|
||||
return entry[1] if entry else None
|
||||
223
backend/app/electronics/net/polite_client.py
Normal file
223
backend/app/electronics/net/polite_client.py
Normal file
@@ -0,0 +1,223 @@
|
||||
"""The only way this project fetches a retail or brand web page.
|
||||
|
||||
What it guarantees, for every request:
|
||||
* robots.txt is consulted first (protego). If robots.txt cannot be read
|
||||
because the server errors or blocks it, the site is treated as disallowed.
|
||||
* at least ELEC_SITE_MIN_INTERVAL_SECONDS between requests to one host.
|
||||
* an honest User-Agent naming the project and a contact address.
|
||||
* no JavaScript, no cookies kept between requests, no proxies, no retries on
|
||||
403/429 - a block is respected, not worked around.
|
||||
* a size cap on the response body.
|
||||
* a circuit breaker: a 403/429/CAPTCHA response opens it for the host, and
|
||||
every later request to that host is refused until the cooldown passes.
|
||||
* every request is reported to `on_fetch` (the fetch_log table).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
import threading
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from typing import Callable, Dict, Optional, Tuple
|
||||
from urllib.parse import urlparse
|
||||
|
||||
import httpx
|
||||
from protego import Protego
|
||||
|
||||
from app.electronics.net.breaker import CircuitBreaker
|
||||
from app.infrastructure.settings import (
|
||||
ELEC_MAX_PAGE_BYTES,
|
||||
ELEC_SITE_MIN_INTERVAL_SECONDS,
|
||||
REQUEST_TIMEOUT_SECONDS,
|
||||
USER_AGENT,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
ROBOTS_TTL_SECONDS = 24 * 3600
|
||||
|
||||
# Pages that are a bot check rather than content. Matched on the first 20 KB.
|
||||
_CAPTCHA_MARKERS = re.compile(
|
||||
r"captcha|robot\s*check|are\s+you\s+a\s+robot|verify\s+you\s+are\s+human|"
|
||||
r"/errors/validatecaptcha|px-captcha|cf-challenge|challenge-platform|access\s+denied|"
|
||||
r"unusual\s+traffic|request\s+blocked|bot\s+detection|akamai.*reference",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class FetchResult:
|
||||
url: str
|
||||
final_url: str
|
||||
status: Optional[int]
|
||||
text: str
|
||||
outcome: str # ok | robots_disallowed | blocked | captcha | breaker_open | http_error | network_error | too_large | not_html
|
||||
robots_allowed: Optional[bool]
|
||||
bytes: int = 0
|
||||
|
||||
@property
|
||||
def ok(self) -> bool:
|
||||
return self.outcome == "ok"
|
||||
|
||||
|
||||
class PoliteClient:
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
min_interval: float = ELEC_SITE_MIN_INTERVAL_SECONDS,
|
||||
breaker: Optional[CircuitBreaker] = None,
|
||||
on_fetch: Optional[Callable[[FetchResult, str], None]] = None,
|
||||
transport: Optional[httpx.BaseTransport] = None,
|
||||
sleep: Callable[[float], None] = time.sleep,
|
||||
clock: Callable[[], float] = time.monotonic,
|
||||
) -> None:
|
||||
self.min_interval = min_interval
|
||||
self.breaker = breaker or CircuitBreaker()
|
||||
self.on_fetch = on_fetch
|
||||
self._sleep = sleep
|
||||
self._clock = clock
|
||||
self._last: Dict[str, float] = {}
|
||||
self._locks: Dict[str, threading.Lock] = {}
|
||||
self._robots: Dict[str, Tuple[float, Optional[Protego], bool]] = {}
|
||||
self._guard = threading.Lock()
|
||||
self._client = httpx.Client(
|
||||
headers={
|
||||
"User-Agent": USER_AGENT,
|
||||
"Accept": "text/html,application/xhtml+xml,application/json;q=0.9,*/*;q=0.5",
|
||||
"Accept-Language": "en-IN,en;q=0.9",
|
||||
},
|
||||
follow_redirects=True,
|
||||
timeout=REQUEST_TIMEOUT_SECONDS,
|
||||
transport=transport,
|
||||
)
|
||||
|
||||
def close(self) -> None:
|
||||
self._client.close()
|
||||
|
||||
def __enter__(self) -> "PoliteClient":
|
||||
return self
|
||||
|
||||
def __exit__(self, *exc) -> None:
|
||||
self.close()
|
||||
|
||||
# -- pacing --------------------------------------------------------------
|
||||
def _host_lock(self, host: str) -> threading.Lock:
|
||||
with self._guard:
|
||||
return self._locks.setdefault(host, threading.Lock())
|
||||
|
||||
def _wait_turn(self, host: str) -> None:
|
||||
last = self._last.get(host)
|
||||
if last is not None:
|
||||
gap = self.min_interval - (self._clock() - last)
|
||||
if gap > 0:
|
||||
self._sleep(gap)
|
||||
self._last[host] = self._clock()
|
||||
|
||||
# -- robots.txt ----------------------------------------------------------
|
||||
def _robots_for(self, scheme: str, host: str) -> Tuple[Optional[Protego], bool]:
|
||||
"""(parser, reachable). parser None + reachable True = no robots.txt
|
||||
(everything allowed); reachable False = could not read it (deny)."""
|
||||
cached = self._robots.get(host)
|
||||
if cached and self._clock() - cached[0] < ROBOTS_TTL_SECONDS:
|
||||
return cached[1], cached[2]
|
||||
url = f"{scheme}://{host}/robots.txt"
|
||||
parser: Optional[Protego] = None
|
||||
reachable = False
|
||||
self._wait_turn(host)
|
||||
try:
|
||||
resp = self._client.get(url)
|
||||
if resp.status_code == 200:
|
||||
parser, reachable = Protego.parse(resp.text), True
|
||||
elif resp.status_code in (404, 410):
|
||||
parser, reachable = None, True
|
||||
else:
|
||||
reachable = False
|
||||
if resp.status_code in (403, 429):
|
||||
self.breaker.trip(host, f"robots.txt returned HTTP {resp.status_code}")
|
||||
except httpx.HTTPError as exc:
|
||||
logger.info("robots.txt unreachable for %s: %s", host, exc)
|
||||
self._robots[host] = (self._clock(), parser, reachable)
|
||||
return parser, reachable
|
||||
|
||||
def robots_allowed(self, url: str) -> bool:
|
||||
p = urlparse(url)
|
||||
parser, reachable = self._robots_for(p.scheme or "https", p.netloc.lower())
|
||||
if not reachable:
|
||||
return False
|
||||
return True if parser is None else bool(parser.can_fetch(url, USER_AGENT))
|
||||
|
||||
# -- fetch ---------------------------------------------------------------
|
||||
def _report(self, result: FetchResult) -> FetchResult:
|
||||
if self.on_fetch:
|
||||
try:
|
||||
self.on_fetch(result, urlparse(result.url).netloc.lower())
|
||||
except Exception as exc: # noqa: BLE001 - logging must never break a crawl
|
||||
logger.debug("fetch log failed: %s", exc)
|
||||
return result
|
||||
|
||||
def get(self, url: str, *, check_robots: bool = True, accept_non_html: bool = False) -> FetchResult:
|
||||
host = urlparse(url).netloc.lower()
|
||||
if self.breaker.is_open(host):
|
||||
return FetchResult(url, url, None, "", "breaker_open", None)
|
||||
with self._host_lock(host):
|
||||
allowed: Optional[bool] = None
|
||||
if check_robots:
|
||||
allowed = self.robots_allowed(url)
|
||||
if not allowed:
|
||||
return self._report(FetchResult(url, url, None, "", "robots_disallowed", False))
|
||||
self._wait_turn(host)
|
||||
try:
|
||||
with self._client.stream("GET", url) as resp:
|
||||
status = resp.status_code
|
||||
final = str(resp.url)
|
||||
ctype = resp.headers.get("content-type", "").lower()
|
||||
body = bytearray()
|
||||
too_large = False
|
||||
for chunk in resp.iter_bytes():
|
||||
body.extend(chunk)
|
||||
if len(body) > ELEC_MAX_PAGE_BYTES:
|
||||
too_large = True
|
||||
break
|
||||
encoding = resp.encoding or "utf-8"
|
||||
except httpx.HTTPError as exc:
|
||||
logger.info("fetch failed %s: %s", url, exc)
|
||||
return self._report(FetchResult(url, url, None, "", "network_error", allowed))
|
||||
|
||||
text = bytes(body).decode(encoding, errors="replace") if body else ""
|
||||
n = len(body)
|
||||
if status in (403, 429, 503) or (status == 200 and _CAPTCHA_MARKERS.search(text[:20000]) and len(text) < 60000):
|
||||
outcome = "captcha" if status == 200 or _CAPTCHA_MARKERS.search(text[:20000]) else "blocked"
|
||||
self.breaker.trip(host, f"HTTP {status} ({outcome})")
|
||||
return self._report(FetchResult(url, final, status, "", outcome, allowed, n))
|
||||
if status != 200:
|
||||
return self._report(FetchResult(url, final, status, "", "http_error", allowed, n))
|
||||
if too_large:
|
||||
return self._report(FetchResult(url, final, status, "", "too_large", allowed, n))
|
||||
if not accept_non_html and "html" not in ctype and "json" not in ctype:
|
||||
return self._report(FetchResult(url, final, status, "", "not_html", allowed, n))
|
||||
return self._report(FetchResult(url, final, status, text, "ok", allowed, n))
|
||||
|
||||
def check_image(self, url: str, min_bytes: int) -> bool:
|
||||
"""One ranged GET to confirm a URL serves a real image. Paced per host
|
||||
like any request; robots.txt is not consulted because this fetches a
|
||||
single file the product page itself references, as a browser would."""
|
||||
host = urlparse(url).netloc.lower()
|
||||
if not url.startswith(("http://", "https://")) or self.breaker.is_open(host):
|
||||
return False
|
||||
with self._host_lock(host):
|
||||
self._wait_turn(host)
|
||||
try:
|
||||
with self._client.stream("GET", url, headers={"Accept": "image/*", "Range": f"bytes=0-{min_bytes * 4}"}) as resp:
|
||||
if resp.status_code not in (200, 206):
|
||||
return False
|
||||
if not resp.headers.get("content-type", "").lower().startswith("image/"):
|
||||
return False
|
||||
got = 0
|
||||
for chunk in resp.iter_bytes():
|
||||
got += len(chunk)
|
||||
if got >= min_bytes:
|
||||
return True
|
||||
return got >= min_bytes
|
||||
except httpx.HTTPError:
|
||||
return False
|
||||
0
backend/app/electronics/normalise/__init__.py
Normal file
0
backend/app/electronics/normalise/__init__.py
Normal file
86
backend/app/electronics/normalise/brand_alias.py
Normal file
86
backend/app/electronics/normalise/brand_alias.py
Normal file
@@ -0,0 +1,86 @@
|
||||
"""Resolve the brand of a product title against the closed allow-list."""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from functools import lru_cache
|
||||
from typing import List, Optional, Tuple
|
||||
|
||||
from app.electronics.reference import load_reference
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class BrandMatch:
|
||||
brand_slug: str
|
||||
brand_name: str
|
||||
family: Optional[str] # sub-brand (Redmi, iQOO, Pixel...) when the title used one
|
||||
matched: str # the alias text found in the title
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def _alias_table() -> List[Tuple[str, str, bool]]:
|
||||
"""(alias, brand_slug, is_sub_brand), longest alias first."""
|
||||
ref = load_reference()
|
||||
rows: List[Tuple[str, str, bool]] = []
|
||||
for b in ref.brands.values():
|
||||
for a in b.aliases:
|
||||
rows.append((a, b.slug, False))
|
||||
for s in b.sub_brands:
|
||||
rows.append((s, b.slug, True))
|
||||
rows.sort(key=lambda r: -len(r[0]))
|
||||
return rows
|
||||
|
||||
|
||||
def resolve_brand(title: str, *, expected: Optional[str] = None) -> Optional[BrandMatch]:
|
||||
"""The allow-listed brand a title starts with (or names within its first
|
||||
few words), or None. `expected` restricts the match to one brand slug.
|
||||
|
||||
Only the start of the title is considered: "Case for Samsung Galaxy S24"
|
||||
is an accessory, not a Samsung phone.
|
||||
"""
|
||||
if not title:
|
||||
return None
|
||||
ref = load_reference()
|
||||
head = " ".join(re.findall(r"[a-z0-9+]+", title.lower())[:3])
|
||||
for alias, slug, is_sub in _alias_table():
|
||||
if expected and slug != expected:
|
||||
continue
|
||||
pattern = r"(?:^|\s)" + re.escape(alias) + r"(?:\s|$)"
|
||||
m = re.search(pattern, head)
|
||||
if not m:
|
||||
continue
|
||||
# The brand/sub-brand must be the first or second word ("Apple iPhone",
|
||||
# "Samsung Galaxy", "Xiaomi Redmi Note") - not buried later.
|
||||
if len(head[: m.start()].split()) > 1:
|
||||
continue
|
||||
# "Google Pixel 8", "Xiaomi Redmi Note 13": the parent brand matched,
|
||||
# but the family is the sub-brand that follows it.
|
||||
sub = alias if is_sub else next(
|
||||
(s for s in ref.brands[slug].sub_brands if re.search(r"(?:^|\s)" + re.escape(s) + r"(?:\s|$)", head)),
|
||||
None,
|
||||
)
|
||||
return BrandMatch(slug, ref.brands[slug].name, _family_casing(sub) if sub else None, alias)
|
||||
return None
|
||||
|
||||
|
||||
_CASING = {"iphone": "iPhone", "iqoo": "iQOO", "macbook": "MacBook", "rog": "ROG", "tuf": "TUF",
|
||||
"cmf": "CMF", "loq": "LOQ", "poco": "POCO", "mi": "Mi", "xps": "XPS", "thinkpad": "ThinkPad",
|
||||
"ideapad": "IdeaPad", "thinkbook": "ThinkBook", "vivobook": "Vivobook", "zenbook": "Zenbook"}
|
||||
|
||||
|
||||
def _family_casing(sub: str) -> str:
|
||||
return _CASING.get(sub, sub.title())
|
||||
|
||||
|
||||
# Words that mark an accessory or a non-product page, not a device.
|
||||
_NOT_A_DEVICE = re.compile(
|
||||
r"\b(?:case|cover|back\s+cover|tempered|screen\s+guard|protector|charger|adapter|cable|"
|
||||
r"skin|sleeve|bag|backpack|stand|holder|refurbished|renewed|pre-?owned|used|"
|
||||
r"compare|vs\.?|versus|review|specifications?\s+and|price\s+list|best\s+\w+\s+under|"
|
||||
r"top\s+\d+|all\s+models)\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
def looks_like_device_title(title: str) -> bool:
|
||||
return bool(title) and not _NOT_A_DEVICE.search(title)
|
||||
55
backend/app/electronics/normalise/grounding.py
Normal file
55
backend/app/electronics/normalise/grounding.py
Normal file
@@ -0,0 +1,55 @@
|
||||
"""Is a value actually stated in the text it supposedly came from?
|
||||
|
||||
Every value the LLM returns passes through value_in_source() against the exact
|
||||
text the model was shown. Anything that cannot be found there is discarded,
|
||||
which is what stops a small model's guess becoming a stored fact.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from decimal import Decimal, InvalidOperation
|
||||
from typing import Union
|
||||
|
||||
_WS = re.compile(r"\s+")
|
||||
|
||||
|
||||
def _norm_text(text: str) -> str:
|
||||
text = text.lower().replace(" ", " ")
|
||||
text = re.sub(r"[^\w.+ ]+", " ", text)
|
||||
text = re.sub(r"(?<!\d)\.|\.(?!\d)", " ", text) # sentence full stops, not decimals
|
||||
# "5000mAh" and "5000 mAh" must compare equal.
|
||||
text = re.sub(r"(?<=\d)(?=[a-z])|(?<=[a-z])(?=\d)", " ", text)
|
||||
return _WS.sub(" ", text).strip()
|
||||
|
||||
|
||||
def _numbers_in(text: str) -> set:
|
||||
found = set()
|
||||
for raw in re.findall(r"\d[\d,]*(?:\.\d+)?", text):
|
||||
try:
|
||||
found.add(Decimal(raw.replace(",", "")).normalize())
|
||||
except InvalidOperation:
|
||||
continue
|
||||
return found
|
||||
|
||||
|
||||
def value_in_source(value: Union[str, int, float, Decimal, None], source: str) -> bool:
|
||||
if value is None or not source:
|
||||
return False
|
||||
if isinstance(value, bool):
|
||||
return False
|
||||
if isinstance(value, (int, float, Decimal)):
|
||||
try:
|
||||
return Decimal(str(value)).normalize() in _numbers_in(source)
|
||||
except InvalidOperation:
|
||||
return False
|
||||
text = str(value).strip()
|
||||
if not text:
|
||||
return False
|
||||
# A string with a number in it ("5000 mAh", "Snapdragon 8 Gen 3") must have
|
||||
# every one of its numbers in the source, and its words too.
|
||||
nums = _numbers_in(text)
|
||||
if nums and not nums <= _numbers_in(source):
|
||||
return False
|
||||
words = [w for w in _norm_text(text).split() if not re.fullmatch(r"[\d.,]+", w)]
|
||||
hay = f" {_norm_text(source)} "
|
||||
return all(f" {w} " in hay for w in words) if words else bool(nums)
|
||||
71
backend/app/electronics/normalise/llm_fill.py
Normal file
71
backend/app/electronics/normalise/llm_fill.py
Normal file
@@ -0,0 +1,71 @@
|
||||
"""Fill MISSING spec keys from page text with the local LLM - and keep only
|
||||
what the text actually says.
|
||||
|
||||
The model sees one block of text that we fetched (a spec section or a
|
||||
description) and is asked to copy values out of it. Every value it returns is:
|
||||
1. checked by grounding.value_in_source() against that same text, and
|
||||
2. normalised by spec_normaliser (units, plausible ranges).
|
||||
Anything failing either step is dropped. Prices, product names and images are
|
||||
never asked of the model.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
from typing import Any, Dict, Iterable, Tuple
|
||||
|
||||
from app.electronics.normalise.grounding import value_in_source
|
||||
from app.electronics.normalise.spec_normaliser import normalise_value
|
||||
from app.infrastructure.settings import ELEC_USE_LLM
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
MAX_SOURCE_CHARS = 3500
|
||||
|
||||
SYSTEM_PROMPT = (
|
||||
"You copy product specifications out of the text you are given. "
|
||||
"Rules: use ONLY the given text; copy each value exactly as written, including its unit; "
|
||||
"if the text does not state a value, use null; never guess, estimate or use outside knowledge. "
|
||||
"Reply with one JSON object whose keys are exactly the requested keys."
|
||||
)
|
||||
|
||||
|
||||
def fill_missing(
|
||||
category: str,
|
||||
source_text: str,
|
||||
missing_keys: Iterable[str],
|
||||
*,
|
||||
generate=None,
|
||||
) -> Tuple[Dict[str, Any], Dict[str, str]]:
|
||||
"""(specs, sources) for whichever of `missing_keys` the text states."""
|
||||
keys = [k for k in missing_keys if k != "colour"]
|
||||
text = (source_text or "").strip()[:MAX_SOURCE_CHARS]
|
||||
if not keys or not text or not ELEC_USE_LLM:
|
||||
return {}, {}
|
||||
if generate is None:
|
||||
from app.services.ollama_service import generate_json as generate
|
||||
|
||||
prompt = (
|
||||
f"Requested keys: {json.dumps(keys)}\n\n"
|
||||
f"Text:\n\"\"\"\n{text}\n\"\"\"\n\n"
|
||||
"JSON:"
|
||||
)
|
||||
reply = generate(SYSTEM_PROMPT, prompt)
|
||||
if not isinstance(reply, dict):
|
||||
return {}, {}
|
||||
|
||||
specs: Dict[str, Any] = {}
|
||||
sources: Dict[str, str] = {}
|
||||
for key in keys:
|
||||
raw = reply.get(key)
|
||||
if raw is None or isinstance(raw, (dict, list, bool)):
|
||||
continue
|
||||
if not value_in_source(raw, text):
|
||||
logger.debug("LLM value %r for %s not found in source text; dropped", raw, key)
|
||||
continue
|
||||
value = normalise_value(category, key, raw)
|
||||
if value is None:
|
||||
continue
|
||||
specs[key] = value
|
||||
sources[key] = f"llm-extracted: {str(raw)[:80]}"
|
||||
return specs, sources
|
||||
101
backend/app/electronics/normalise/spec_normaliser.py
Normal file
101
backend/app/electronics/normalise/spec_normaliser.py
Normal file
@@ -0,0 +1,101 @@
|
||||
"""Map raw spec labels/values from a page to canonical keys and units.
|
||||
|
||||
Deterministic and table-driven (reference/spec_keys.yaml). A value that cannot
|
||||
be parsed, or lands outside the plausible range for its key, is dropped - never
|
||||
estimated.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from decimal import Decimal, InvalidOperation
|
||||
from functools import lru_cache
|
||||
from typing import Any, Dict, Optional, Tuple
|
||||
|
||||
from app.electronics.reference import load_reference
|
||||
|
||||
|
||||
def _label(text: str) -> str:
|
||||
return re.sub(r"[^a-z0-9]+", " ", str(text).lower()).strip()
|
||||
|
||||
|
||||
@lru_cache(maxsize=None)
|
||||
def _synonyms(category: str) -> Dict[str, str]:
|
||||
table: Dict[str, str] = {}
|
||||
for key, spec in load_reference().spec_keys.get(category, {}).items():
|
||||
for syn in [key.replace("_", " "), *spec.get("synonyms", [])]:
|
||||
table.setdefault(_label(syn), key)
|
||||
return table
|
||||
|
||||
|
||||
def canonical_key(category: str, label: str) -> Optional[str]:
|
||||
return _synonyms(category).get(_label(label))
|
||||
|
||||
|
||||
# unit -> (regex for the unit in text, factor into the canonical unit)
|
||||
_UNIT_PATTERNS = {
|
||||
"GB": [(r"tb", Decimal(1024)), (r"gb", Decimal(1)), (r"mb", Decimal(1) / 1024)],
|
||||
"inch": [(r"(?:inch(?:es)?|in\b|\"|”)", Decimal(1)), (r"cm", Decimal(1) / Decimal("2.54"))],
|
||||
"Hz": [(r"hz", Decimal(1))],
|
||||
"MP": [(r"mp|megapixel", Decimal(1))],
|
||||
"mAh": [(r"mah", Decimal(1))],
|
||||
"kg": [(r"kg|kilogram", Decimal(1)), (r"(?<![k])g\b|grams?", Decimal("0.001"))],
|
||||
"Wh": [(r"wh|watt\s*hours?", Decimal(1))],
|
||||
}
|
||||
|
||||
|
||||
def _to_number(value: str, unit: str) -> Optional[Decimal]:
|
||||
text = str(value).lower().replace(",", "")
|
||||
patterns = _UNIT_PATTERNS.get(unit, [])
|
||||
# Prefer an amount written in the canonical unit ("39.62 cm (15.6 inch)" -> 15.6).
|
||||
for unit_re, factor in patterns:
|
||||
m = re.search(r"(\d+(?:\.\d+)?)\s*(?:" + unit_re + r")", text)
|
||||
if m:
|
||||
try:
|
||||
return (Decimal(m.group(1)) * factor).quantize(Decimal("0.01")).normalize()
|
||||
except InvalidOperation:
|
||||
return None
|
||||
# A bare number is accepted only when nothing else is in the value.
|
||||
m = re.fullmatch(r"\s*(\d+(?:\.\d+)?)\s*", text)
|
||||
if m:
|
||||
return Decimal(m.group(1)).normalize()
|
||||
return None
|
||||
|
||||
|
||||
def normalise_value(category: str, key: str, value: Any) -> Optional[Any]:
|
||||
spec = load_reference().spec_keys.get(category, {}).get(key)
|
||||
if spec is None or value is None:
|
||||
return None
|
||||
text = str(value).strip()
|
||||
if not text or text.lower() in {"na", "n/a", "-", "none", "not applicable", "no"}:
|
||||
return None
|
||||
kind = spec.get("type")
|
||||
if kind == "number":
|
||||
num = _to_number(text, spec.get("unit", ""))
|
||||
if num is None:
|
||||
return None
|
||||
lo, hi = spec.get("range", [None, None])
|
||||
if (lo is not None and num < Decimal(str(lo))) or (hi is not None and num > Decimal(str(hi))):
|
||||
return None
|
||||
return float(num) if num != num.to_integral() else int(num)
|
||||
if kind == "enum":
|
||||
low = text.lower()
|
||||
for canon, words in spec.get("values", {}).items():
|
||||
if any(re.search(r"\b" + re.escape(w) + r"\b", low) for w in words):
|
||||
return canon
|
||||
return None
|
||||
return re.sub(r"\s+", " ", text)[:120]
|
||||
|
||||
|
||||
def normalise_specs(category: str, raw: Dict[str, Any]) -> Tuple[Dict[str, Any], Dict[str, str]]:
|
||||
"""(specs, sources): canonical key -> value, and key -> the raw label it came from."""
|
||||
specs: Dict[str, Any] = {}
|
||||
sources: Dict[str, str] = {}
|
||||
for label, value in (raw or {}).items():
|
||||
key = canonical_key(category, label)
|
||||
if not key or key in specs:
|
||||
continue
|
||||
norm = normalise_value(category, key, value)
|
||||
if norm is not None:
|
||||
specs[key] = norm
|
||||
sources[key] = f"{label}: {value}"[:200]
|
||||
return specs, sources
|
||||
374
backend/app/electronics/normalise/title_parser.py
Normal file
374
backend/app/electronics/normalise/title_parser.py
Normal file
@@ -0,0 +1,374 @@
|
||||
"""Split a retail product title into model, variant and a matching key.
|
||||
|
||||
Everything returned is read from the title text; a value the title does not
|
||||
state is None. Titles differ a lot between sites:
|
||||
|
||||
Samsung Galaxy S24 5G (Onyx Black, 8GB RAM, 256GB Storage) Amazon
|
||||
SAMSUNG Galaxy S24 5G (Onyx Black, 256 GB) (8 GB RAM) Flipkart
|
||||
Samsung Galaxy S24 5G (8GB RAM, 256GB, Onyx Black) Croma
|
||||
Redmi Note 13 Pro 5G (8GB + 256GB)
|
||||
Apple iPhone 15 (128 GB) - Black
|
||||
HP 15s, 13th Gen Intel Core i5-1334U, 16GB DDR4, 512GB SSD, ... fd0112TU
|
||||
|
||||
so the model is taken from the text before the first bracket/comma, and RAM /
|
||||
storage / colour from anywhere in the title.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from decimal import Decimal
|
||||
from typing import List, Optional
|
||||
|
||||
from app.electronics.normalise.brand_alias import BrandMatch, resolve_brand
|
||||
|
||||
_NUM = r"(\d+(?:\.\d+)?)"
|
||||
|
||||
# "8GB RAM", "8 GB LPDDR5X RAM", "RAM 8GB", "16GB DDR4" (laptops)
|
||||
_RAM_RES = [
|
||||
re.compile(_NUM + r"\s*GB\s*(?:LP)?(?:DDR\s?\d\w?\s*)?RAM\b", re.IGNORECASE),
|
||||
re.compile(r"\bRAM\s*[:\-]?\s*" + _NUM + r"\s*GB", re.IGNORECASE),
|
||||
re.compile(_NUM + r"\s*GB\s*(?:LP)?DDR\s?\d", re.IGNORECASE),
|
||||
re.compile(_NUM + r"\s*GB\s*(?:unified\s+memory|memory)\b", re.IGNORECASE),
|
||||
]
|
||||
# "8GB + 256GB", "8/256", "8GB/256GB", "12+512GB"
|
||||
_PAIR_RE = re.compile(r"(?<![\d.])(\d{1,2})\s*(?:GB)?\s*[+/]\s*(\d{2,4}|1|2)\s*(GB|TB)?\b", re.IGNORECASE)
|
||||
# explicit storage: "256GB Storage", "512GB SSD", "1TB", "256 GB ROM"
|
||||
_STORAGE_LABELLED = re.compile(
|
||||
_NUM + r"\s*(GB|TB)\s*(?:SSD|ROM|storage|internal(?:\s+storage)?|HDD|eMMC|UFS|NVMe|PCIe)\b", re.IGNORECASE
|
||||
)
|
||||
_SIZE_ANY = re.compile(r"(?<![\d.])" + _NUM + r"\s*(GB|TB)\b", re.IGNORECASE)
|
||||
|
||||
# Order matters: the first pattern that matches wins. AMD comes before the
|
||||
# Intel "Core N" pattern, because retail titles write core counts as words
|
||||
# ("Ryzen 3 Quad Core 7320U"), and "Core 7320U" must not read as Intel.
|
||||
_CORE_COUNT = r"(?:(?:Dual|Quad|Hexa|Octa|Six|Eight)\s+Core\s+)?"
|
||||
_PROCESSOR_RES = [
|
||||
re.compile(r"\b(?:AMD\s+)?Ryzen\s+R?(\d)\s*(?:Pro\s+)?" + _CORE_COUNT + r"[- ]?(\d{4}[A-Z]{0,3})\b", re.IGNORECASE),
|
||||
re.compile(r"\b(?:AMD\s+)?(Athlon)\s+(?:Silver\s+|Gold\s+)?" + _CORE_COUNT + r"(\d{4}[A-Z]{0,2})\b", re.IGNORECASE),
|
||||
re.compile(r"\b(?:Intel\s+)?Core\s+Ultra\s+(\d)\s*[- ]?\s*(\d{3}[A-Z]{0,2})\b", re.IGNORECASE),
|
||||
re.compile(r"\b(?:Intel\s+)?Core\s+(?:\(?\d+(?:th|nd|rd|st)\s+Gen\)?\s+)?(i[3579])\s*[- ]?\s*(\d{4,5}[A-Z]{0,3})\b", re.IGNORECASE),
|
||||
re.compile(r"\b(?:Intel\s+)?Core\s+(i[3579])\s+\d+(?:th|nd|rd|st)\s+Gen\s+(\d{4,5}[A-Z]{0,3})\b", re.IGNORECASE),
|
||||
re.compile(r"(?<!Dual )(?<!Quad )(?<!Hexa )(?<!Octa )(?<!Six )(?<!Eight )\b(?:Intel\s+)?Core\s+([3579])\s*[- ]?\s*(\d{3}[A-Z]{0,2})\b", re.IGNORECASE),
|
||||
re.compile(r"\bApple\s+(M[1-9])(?:\s+(Pro|Max|Ultra))?\b", re.IGNORECASE),
|
||||
re.compile(r"\b(M[1-9])\s*(Pro|Max|Ultra)?\s+chip\b", re.IGNORECASE),
|
||||
re.compile(r"\bSnapdragon\s+(X\d?)\s*(Elite|Plus)?\s*(X\d{1,2}-\d{3})?", re.IGNORECASE),
|
||||
re.compile(r"\b(?:Intel\s+)?(Celeron|Pentium(?:\s+Silver|\s+Gold)?)\s+(N?\d{3,5}[A-Z]?)\b", re.IGNORECASE),
|
||||
re.compile(r"\bMediaTek\s+(Kompanio\s+\d{3,4}|MT\d{4})\b", re.IGNORECASE),
|
||||
]
|
||||
|
||||
_COLOUR_WORDS = re.compile(
|
||||
r"\b(black|white|blue|green|red|grey|gray|silver|gold|purple|violet|pink|yellow|orange|"
|
||||
r"cream|titanium|graphite|midnight|starlight|mint|lavender|bronze|copper|beige|teal|"
|
||||
r"navy|jade|coral|onyx|marble|obsidian|porcelain|hazel|aqua|lime|sand|charcoal)\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
# Tokens that describe the device class or connectivity, not the model.
|
||||
_MODEL_NOISE = re.compile(
|
||||
r"\b(?:5g|4g|lte|smartphone|smart\s+phone|mobile\s+phone|mobile|phone|dual\s+sim|"
|
||||
r"laptop|notebook|thin\s+and\s+light|thin\s+&\s+light|gaming|new|latest|with\b.*$)",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_SPEC_TOKEN = re.compile(
|
||||
r"^(?:\d+(?:\.\d+)?\s*(?:gb|tb|mp|mah|hz|inch|inches|cm|w|kg|g)|\d+(?:th|nd|rd|st)|gen|ddr\d?\w*|"
|
||||
r"lpddr\d\w*|ssd|hdd|fhd|qhd|uhd|oled|ips|win|windows|20\d\d)$",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class ParsedTitle:
|
||||
title: str
|
||||
brand: Optional[BrandMatch]
|
||||
model: Optional[str] = None # "Galaxy S24", "Redmi Note 13 Pro", "15s"
|
||||
model_norm: Optional[str] = None # "galaxy s24", matching form
|
||||
ram_gb: Optional[Decimal] = None
|
||||
storage_gb: Optional[Decimal] = None
|
||||
colour: Optional[str] = None
|
||||
processor: Optional[str] = None # "i5-1334u", "ryzen 5 7530u", "m3"
|
||||
mpn: Optional[str] = None # laptop part number when stated
|
||||
network: Optional[str] = None
|
||||
notes: List[str] = field(default_factory=list)
|
||||
|
||||
|
||||
def _dec(value: str, unit: str = "GB") -> Decimal:
|
||||
d = Decimal(value)
|
||||
if unit.upper() == "TB":
|
||||
d = d * 1024
|
||||
return d.normalize() if d == d.to_integral() else d
|
||||
|
||||
|
||||
def _parse_ram_storage(text: str, category: str):
|
||||
ram = storage = None
|
||||
for rx in _RAM_RES:
|
||||
m = rx.search(text)
|
||||
if m:
|
||||
ram = _dec(m.group(1))
|
||||
break
|
||||
m = _PAIR_RE.search(text)
|
||||
if m:
|
||||
a, b, unit = m.group(1), m.group(2), (m.group(3) or "GB")
|
||||
pair_ram, pair_storage = _dec(a), _dec(b, unit)
|
||||
# "Core Ultra 5/ 16GB RAM/ 512GB" is not a 5 GB / 16 GB pair: a real
|
||||
# pair has device-sized storage.
|
||||
min_pair_storage = Decimal(16) if category == "mobiles" else Decimal(64)
|
||||
if pair_storage > pair_ram and pair_storage >= min_pair_storage:
|
||||
ram = ram if ram is not None else pair_ram
|
||||
storage = pair_storage
|
||||
if storage is None:
|
||||
m = _STORAGE_LABELLED.search(text)
|
||||
if m:
|
||||
storage = _dec(m.group(1), m.group(2))
|
||||
if storage is None:
|
||||
# Unlabelled sizes: the storage is the largest one that is not the RAM.
|
||||
min_storage = Decimal(16) if category == "mobiles" else Decimal(32)
|
||||
sizes = [_dec(v, u) for v, u in _SIZE_ANY.findall(text)]
|
||||
candidates = [s for s in sizes if s != ram and s >= min_storage]
|
||||
if candidates:
|
||||
storage = max(candidates)
|
||||
if ram is None:
|
||||
# "(8 GB RAM)" handled above; an unlabelled small size next to a larger
|
||||
# one ("8GB 256GB") is the RAM.
|
||||
sizes = [_dec(v, u) for v, u in _SIZE_ANY.findall(text)]
|
||||
small = [s for s in sizes if s <= (24 if category == "mobiles" else 64) and (storage is None or s < storage)]
|
||||
if len(set(small)) == 1 and storage is not None:
|
||||
ram = small[0]
|
||||
plausible_ram = Decimal(32) if category == "mobiles" else Decimal(128)
|
||||
if ram is not None and not (Decimal(1) <= ram <= plausible_ram):
|
||||
ram = None
|
||||
return ram, storage
|
||||
|
||||
|
||||
def parse_processor(text: str) -> Optional[str]:
|
||||
"""Normalised CPU name ("i5-1334u", "ryzen 3 7320u", "core ultra 5 125h",
|
||||
"core 5 120u", "athlon 7120u", "m2"), or None."""
|
||||
for rx in _PROCESSOR_RES:
|
||||
m = rx.search(text or "")
|
||||
if not m:
|
||||
continue
|
||||
parts = [g for g in m.groups() if g]
|
||||
matched = m.group(0).lower()
|
||||
if "ryzen" in matched:
|
||||
return f"ryzen {parts[0]} {parts[1]}".lower()
|
||||
if "athlon" in matched:
|
||||
return f"athlon {parts[1]}".lower()
|
||||
if "ultra" in matched and "core" in matched:
|
||||
return f"core ultra {parts[0]} {parts[1]}".lower()
|
||||
if "core" in matched and parts[0].lower().startswith("i"):
|
||||
return f"{parts[0]}-{parts[1]}".lower()
|
||||
if "core" in matched:
|
||||
return f"core {parts[0]} {parts[1]}".lower()
|
||||
return " ".join(parts).lower()
|
||||
return None
|
||||
|
||||
|
||||
_parse_processor = parse_processor
|
||||
|
||||
|
||||
def processor_is_specific(processor: Optional[str]) -> bool:
|
||||
"""True for a CPU named down to its model number ("i5-1334u"), which
|
||||
together with brand, model line, RAM and storage identifies a laptop
|
||||
configuration. "m2" (Apple) also counts."""
|
||||
if not processor:
|
||||
return False
|
||||
return bool(re.search(r"\d{3,}", processor)) or bool(re.fullmatch(r"m[1-9](?: (?:pro|max|ultra))?", processor))
|
||||
|
||||
|
||||
_MPN_RE = re.compile(
|
||||
r"(?<![\w-])([A-Z0-9]{2,8}-[A-Z0-9]{2,10}(?:-[A-Z0-9]{1,6})?|"
|
||||
r"[A-Z0-9]{2,6}[A-Z]{0,4}\d{2,6}[A-Z]{1,4}\d{0,4}[A-Z]{0,3}|\d{2}[A-Z]{2}\d{3,}[A-Z0-9]{2,})(?![\w-])",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
def _parse_mpn(text: str, processor: Optional[str]) -> Optional[str]:
|
||||
"""A manufacturer part number such as 82XV00BHIN or fd0112TU, when the
|
||||
title states one (laptops). Tokens that are specs or CPU names are not."""
|
||||
candidates = []
|
||||
for m in _MPN_RE.finditer(text):
|
||||
tok = m.group(1)
|
||||
low = tok.lower()
|
||||
if len(tok) < 6 or len(tok) > 20:
|
||||
continue
|
||||
if sum(c.isdigit() for c in tok) < 2 or sum(c.isalpha() for c in tok) < 2:
|
||||
continue
|
||||
if _SPEC_TOKEN.match(low) or re.match(r"^(?:i[3579]|m[1-9]|rtx|gtx|rx|ddr|lpddr)", low):
|
||||
continue
|
||||
if processor and low in processor.replace("-", " ").split() + [processor.replace(" ", "")]:
|
||||
continue
|
||||
if re.search(r"\d+(?:gb|tb|mp|mah|hz|w)$", low):
|
||||
continue
|
||||
if re.search(r"-(?:core|inch|cell|bit|gen|thread)s?$|^\d+-", low) and not re.search(r"[a-z]\d", low.split("-")[-1]):
|
||||
continue # "10-Core", "15-inch", "3-Cell" describe hardware, not a part number
|
||||
candidates.append(tok)
|
||||
return candidates[-1].upper() if candidates else None
|
||||
|
||||
|
||||
def _parse_colour(title: str) -> Optional[str]:
|
||||
# Inside brackets first: "(Onyx Black, 8GB RAM, 256GB Storage)"
|
||||
for group in re.findall(r"\(([^()]*)\)", title):
|
||||
for part in re.split(r"[,|/]", group):
|
||||
part = part.strip()
|
||||
if part and not re.search(r"\d", part) and _COLOUR_WORDS.search(part):
|
||||
return part.title()
|
||||
# Trailing "- Black"
|
||||
m = re.search(r"[-–|,]\s*([A-Za-z][A-Za-z ]{2,30})\s*$", title)
|
||||
if m and _COLOUR_WORDS.search(m.group(1)) and not re.search(r"\d", m.group(1)):
|
||||
return m.group(1).strip().title()
|
||||
return None
|
||||
|
||||
|
||||
def normalise_model(model: str) -> str:
|
||||
text = model.lower()
|
||||
text = re.sub(r"[()\[\],|]", " ", text)
|
||||
text = _MODEL_NOISE.sub(" ", text)
|
||||
text = re.sub(r"\+", " plus ", text)
|
||||
text = re.sub(r"[^a-z0-9 ]+", " ", text)
|
||||
tokens = [t for t in text.split() if not _SPEC_TOKEN.match(t)]
|
||||
return " ".join(tokens)
|
||||
|
||||
|
||||
_LAPTOP_SPEC_START = re.compile(
|
||||
r"\b(?:intel|amd|apple\s+m[1-9]|m[1-9]\s+chip|core\s+(?:i[3579]|ultra)|ryzen|snapdragon|celeron|pentium|"
|
||||
r"mediatek|\d+(?:th|nd|rd|st)\s+gen|\d+(?:\.\d+)?\s*(?:-|\s)?(?:inch|cm|\"))",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_DISPLAY_NOISE = re.compile(
|
||||
r"\b(?:5g|4g|lte|smartphone|smart\s+phone|mobile\s+phone|dual\s+sim|laptop|notebook|"
|
||||
r"thin\s+and\s+light|thin\s+&\s+light|gaming|new|latest)\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
def _model_from_title(title: str, brand: Optional[BrandMatch], category: str, colour: Optional[str],
|
||||
mpn: Optional[str] = None):
|
||||
"""(display model, matching model_norm) from the head of the title."""
|
||||
head = re.sub(r"^\s*buy\s+", "", title, flags=re.IGNORECASE)
|
||||
if mpn:
|
||||
# A part number is not the model name. HP writes the line into it
|
||||
# ("15-fc0500AU" is an HP 15), so that prefix is kept.
|
||||
prefix = mpn.split("-", 1)[0] if "-" in mpn and len(mpn.split("-", 1)[0]) <= 4 else ""
|
||||
head = re.sub(re.escape(mpn), f" {prefix} ", head, flags=re.IGNORECASE)
|
||||
# A short model token in brackets right after the name is part of it:
|
||||
# "Nothing Phone (2a) 5G (Black, 128 GB)".
|
||||
head = re.sub(r"\((?:19|20)\d\d\)", " ", head) # "(2026)" is a model year, not part of the name
|
||||
head = re.sub(r"\(([A-Za-z0-9+ ]{1,6})\)", lambda m: " " + m.group(1) + " "
|
||||
if not re.search(r"\d\s*(?:gb|tb)", m.group(1), re.I) else m.group(0), head, count=1)
|
||||
head = re.split(r"\s[-–|]\s|[(,|\[:]", head, maxsplit=1)[0]
|
||||
if category == "laptops":
|
||||
m = _LAPTOP_SPEC_START.search(head)
|
||||
if m and m.start() > 0:
|
||||
head = head[: m.start()]
|
||||
if brand:
|
||||
# Drop the parent brand's own name ("Samsung Galaxy S24" -> "Galaxy S24",
|
||||
# "Apple iPhone 15" -> "iPhone 15"); a sub-brand stays ("Redmi Note 13").
|
||||
from app.electronics.reference import load_reference
|
||||
|
||||
parent_aliases = sorted(load_reference().brands[brand.brand_slug].aliases, key=len, reverse=True)
|
||||
for alias in parent_aliases:
|
||||
head = re.sub(r"^\s*" + re.escape(alias) + r"\b", "", head, flags=re.IGNORECASE).strip()
|
||||
head = re.sub(r"\b\d+\s*GB\s*RAM\b", " ", head, flags=re.IGNORECASE)
|
||||
head = _PAIR_RE.sub(" ", head)
|
||||
head = _SIZE_ANY.sub(" ", head)
|
||||
if colour:
|
||||
head = re.sub(re.escape(colour), " ", head, flags=re.IGNORECASE)
|
||||
words = head.split()
|
||||
while len(words) > 1 and _COLOUR_WORDS.fullmatch(words[-1]):
|
||||
words.pop() # "iPhone 15 Black" -> "iPhone 15"
|
||||
head = " ".join(words)
|
||||
display = re.sub(r"\s+", " ", _DISPLAY_NOISE.sub(" ", head)).strip(" -–")
|
||||
norm = normalise_model(head)
|
||||
if not norm:
|
||||
return None, None
|
||||
return display or head.strip(), norm
|
||||
|
||||
|
||||
def parse_title(title: str, category: str, *, expected_brand: Optional[str] = None) -> ParsedTitle:
|
||||
title = re.sub(r"\s+", " ", (title or "")).strip()
|
||||
brand = resolve_brand(title, expected=expected_brand)
|
||||
parsed = ParsedTitle(title=title, brand=brand)
|
||||
if not title:
|
||||
return parsed
|
||||
|
||||
parsed.ram_gb, parsed.storage_gb = _parse_ram_storage(title, category)
|
||||
parsed.colour = _parse_colour(title)
|
||||
if re.search(r"\b5G\b", title, re.IGNORECASE):
|
||||
parsed.network = "5G"
|
||||
if category == "laptops":
|
||||
parsed.processor = _parse_processor(title)
|
||||
parsed.mpn = _parse_mpn(title, parsed.processor)
|
||||
|
||||
parsed.model, parsed.model_norm = _model_from_title(title, brand, category, parsed.colour, parsed.mpn)
|
||||
return parsed
|
||||
|
||||
|
||||
def variant_key(parsed: ParsedTitle, category: str) -> Optional[str]:
|
||||
"""The identity of one real-world variant, or None if the title does not
|
||||
state enough to tell variants apart."""
|
||||
if not parsed.brand or not parsed.model_norm:
|
||||
return None
|
||||
b = parsed.brand.brand_slug
|
||||
fmt = lambda d: "na" if d is None else format(d.normalize(), "f") # noqa: E731
|
||||
if category == "laptops":
|
||||
# A laptop configuration is its model line + CPU + RAM + storage. That
|
||||
# is what every site states (a part number is shown by only a few), so
|
||||
# it is the key whenever it is complete; the MPN is the fallback.
|
||||
line = laptop_line(parsed.model_norm)
|
||||
if line and processor_is_specific(parsed.processor) and parsed.ram_gb and parsed.storage_gb:
|
||||
return f"{b}|laptops|{line}|{parsed.processor}|{fmt(parsed.ram_gb)}|{fmt(parsed.storage_gb)}"
|
||||
if parsed.mpn:
|
||||
return f"{b}|laptops|mpn:{parsed.mpn.lower()}"
|
||||
return None
|
||||
if parsed.storage_gb is None:
|
||||
return None
|
||||
return f"{b}|{category}|{parsed.model_norm}|{fmt(parsed.ram_gb)}|{fmt(parsed.storage_gb)}"
|
||||
|
||||
|
||||
# Lenovo/Asus machine-type codes ("15amn8", "15irh10", "14iah8", "x1504za")
|
||||
# name a chassis generation, and one site prints them where another does not.
|
||||
_MACHINE_CODE = re.compile(r"^(?:\d{2}[a-z]{2,4}\d{1,2}|[a-z]\d{4}[a-z]{1,3})$")
|
||||
|
||||
|
||||
def laptop_line(model_norm: Optional[str]) -> str:
|
||||
"""The model line used for matching: "ideapad slim 3 15amn8" -> "ideapad slim 3"."""
|
||||
tokens = [t for t in (model_norm or "").split() if not _MACHINE_CODE.match(t)]
|
||||
return " ".join(tokens)
|
||||
|
||||
|
||||
def fill_from_context(parsed: ParsedTitle, category: str, *, snippet: str = "",
|
||||
spec_texts: tuple = ()) -> ParsedTitle:
|
||||
"""Fill variant fields a (often truncated) title leaves out, from text the
|
||||
same site published about the same page: its search snippet, or the spec
|
||||
table of the fetched page. Only unambiguous values are taken - a snippet
|
||||
naming two different storage sizes is describing several variants."""
|
||||
if snippet and (parsed.ram_gb is None or parsed.storage_gb is None):
|
||||
sizes = {_dec(v, u) for v, u in _SIZE_ANY.findall(snippet)}
|
||||
if len(sizes) <= 2:
|
||||
ram, storage = _parse_ram_storage(snippet, category)
|
||||
if parsed.storage_gb is None and storage is not None:
|
||||
parsed.storage_gb = storage
|
||||
if parsed.ram_gb is None and ram is not None and ram != parsed.storage_gb:
|
||||
parsed.ram_gb = ram
|
||||
if category == "laptops" and not processor_is_specific(parsed.processor):
|
||||
# A title that already names a CPU family ("Snapdragon X", "Core i7")
|
||||
# is only completed from the page's own spec table, never from a
|
||||
# snippet - snippets often run several products' titles together.
|
||||
sources = list(spec_texts) + ([snippet] if parsed.processor is None and snippet else [])
|
||||
found = set()
|
||||
for text in sources:
|
||||
found |= {cpu for cpu in all_processors(text) if processor_is_specific(cpu)}
|
||||
if len(found) == 1:
|
||||
parsed.processor = found.pop()
|
||||
return parsed
|
||||
|
||||
|
||||
def all_processors(text: str) -> set:
|
||||
"""Every CPU named anywhere in `text` (a snippet can name several)."""
|
||||
found = set()
|
||||
for rx in _PROCESSOR_RES:
|
||||
for m in rx.finditer(text or ""):
|
||||
cpu = parse_processor(m.group(0))
|
||||
if cpu:
|
||||
found.add(cpu)
|
||||
return found
|
||||
102
backend/app/electronics/price_lookup.py
Normal file
102
backend/app/electronics/price_lookup.py
Normal file
@@ -0,0 +1,102 @@
|
||||
"""Fill missing prices on search-only platforms (Amazon.in, Flipkart, Croma...)
|
||||
from Google Programmable Search, without fetching those sites.
|
||||
|
||||
For each listing that has no price (or an unconfirmed one), search Google for
|
||||
that product on that site. A price is taken only when:
|
||||
* the result is the SAME product page (its site product id equals the
|
||||
listing's), and
|
||||
* Google's structured data for the page (pagemap offer / product:price meta)
|
||||
states an INR price.
|
||||
The listing is updated through the normal path, so the price is stored with
|
||||
its evidence, appended to price_history, and outlier-checked.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import Callable, Dict, Optional
|
||||
|
||||
from app.electronics.collector import Collector, RunOptions, RunStats, source_sku
|
||||
from app.electronics.db import repository as repo
|
||||
from app.electronics.db.connection import connect
|
||||
from app.electronics.normalise.title_parser import parse_title
|
||||
from app.electronics.reference import load_reference, site_for_url
|
||||
from app.electronics.search.engine import SearchEngine
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def _listings_needing_price(limit: int, category: Optional[str]) -> list:
|
||||
with connect() as conn:
|
||||
return conn.execute(
|
||||
"""
|
||||
SELECT l.id, l.title, l.source_sku, l.source_url, s.domain, b.slug AS brand_slug, c.slug AS category,
|
||||
(p.verification_status = 'verified') AS verified
|
||||
FROM elec.source_listing l
|
||||
JOIN elec.site s ON s.id = l.site_id
|
||||
JOIN elec.brand b ON b.id = l.brand_id
|
||||
JOIN elec.category c ON c.id = l.category_id
|
||||
JOIN elec.product_listing_map m ON m.listing_id = l.id AND m.review_status IN ('auto','approved')
|
||||
JOIN elec.product p ON p.id = m.product_id
|
||||
WHERE l.source_type = 'search_snippet'
|
||||
AND (l.price IS NULL OR l.price_outlier)
|
||||
AND (s.policy = 'serp_only' OR coalesce(s.probe_outcome, 'C') = 'C')
|
||||
AND (%(category)s::text IS NULL OR c.slug = %(category)s)
|
||||
ORDER BY (p.verification_status = 'verified') DESC, l.last_seen_at DESC
|
||||
LIMIT %(limit)s
|
||||
""",
|
||||
{"limit": limit, "category": category},
|
||||
).fetchall()
|
||||
|
||||
|
||||
def lookup_prices(limit: int = 40, category: Optional[str] = None,
|
||||
progress: Callable[[str], None] = logger.info) -> Dict[str, object]:
|
||||
stats: Dict[str, object] = {"checked": 0, "priced": 0, "no_same_page": 0, "no_structured_price": 0}
|
||||
engine = SearchEngine(budget=limit)
|
||||
if not engine.google.enabled:
|
||||
stats["error"] = "Google Programmable Search is not configured (GOOGLE_API_KEY / GOOGLE_CSE_ID)"
|
||||
return stats
|
||||
ids = repo.id_maps()
|
||||
ref = load_reference()
|
||||
run_id = repo.start_run("price_lookup", {"limit": limit, "category": category})
|
||||
try:
|
||||
for row in _listings_needing_price(limit, category):
|
||||
if not engine.google.enabled:
|
||||
break
|
||||
site = ref.sites[row["domain"]]
|
||||
query = f"site:{row['domain']} {row['title'][:110]}"
|
||||
hits = engine.text(query, max_results=10, providers="google")
|
||||
stats["checked"] += 1
|
||||
if hits is None:
|
||||
continue
|
||||
same = [h for h in hits
|
||||
if (s := site_for_url(h.url)) is not None and s.domain == site.domain
|
||||
and source_sku(site, h.url) == row["source_sku"]]
|
||||
if not same:
|
||||
stats["no_same_page"] += 1
|
||||
continue
|
||||
hit = next((h for h in same if h.offer), None)
|
||||
if hit is None:
|
||||
stats["no_structured_price"] += 1
|
||||
continue
|
||||
collector = Collector.__new__(Collector) # only its listing builder is used
|
||||
collector.opt = RunOptions(category=row["category"], brands=[row["brand_slug"]])
|
||||
collector.stats = RunStats()
|
||||
parsed = parse_title(row["title"], row["category"], expected_brand=row["brand_slug"])
|
||||
if parsed.brand is None:
|
||||
continue
|
||||
listing = collector.listing_from_search(hit, site, parsed, query)
|
||||
listing.source_sku = row["source_sku"]
|
||||
if listing.price is None:
|
||||
stats["no_structured_price"] += 1
|
||||
continue
|
||||
repo.upsert_listing(listing, ids, run_id)
|
||||
stats["priced"] += 1
|
||||
progress(f"{site.name}: {row['title'][:70]} -> Rs {listing.price}")
|
||||
stats["products"] = repo.refresh_verification()
|
||||
if engine.google.error:
|
||||
stats["error"] = engine.google.error
|
||||
repo.finish_run(run_id, "done", {k: v for k, v in stats.items() if k != "products"})
|
||||
except Exception as exc:
|
||||
repo.finish_run(run_id, "failed", {}, repr(exc))
|
||||
raise
|
||||
return stats
|
||||
0
backend/app/electronics/probe/__init__.py
Normal file
0
backend/app/electronics/probe/__init__.py
Normal file
99
backend/app/electronics/probe/site_probe.py
Normal file
99
backend/app/electronics/probe/site_probe.py
Normal file
@@ -0,0 +1,99 @@
|
||||
"""Decide, per site, whether it may be scraped or only searched.
|
||||
|
||||
A robots.txt allows product pages, HTTP 200 without a bot check, and the
|
||||
page carries a schema.org Product with an INR offer -> scrape
|
||||
B fetchable, product name/specs readable from the HTML, but no
|
||||
structured price -> scrape specs/images,
|
||||
price from search
|
||||
C serp_only policy, robots.txt disallows, blocked / CAPTCHA, or the page
|
||||
has no product data without JavaScript -> web search only
|
||||
|
||||
The probe looks at 2-3 real product URLs for the site, found through web
|
||||
search, so it grades the pages the collector would actually fetch.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import Dict, List, Optional
|
||||
|
||||
from app.electronics.extract.html_fallback import embedded_state, extract_page
|
||||
from app.electronics.extract.jsonld import extract_products
|
||||
from app.electronics.net.polite_client import PoliteClient
|
||||
from app.electronics.reference import SiteRef, load_reference, site_for_url
|
||||
from app.electronics.search.engine import SearchEngine
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def sample_product_urls(site: SiteRef, engine: SearchEngine, limit: int = 3) -> List[str]:
|
||||
ref = load_reference()
|
||||
urls: List[str] = []
|
||||
if site.kind == "brand_official":
|
||||
brand = ref.brands[site.brand_slug]
|
||||
terms = [ref.categories[c].search_terms[0] for c in brand.categories]
|
||||
queries = [f"site:{site.domain} {brand.name} {t}" for t in terms]
|
||||
else:
|
||||
queries = [f"site:{site.domain} samsung galaxy 5g", f"site:{site.domain} lenovo laptop"]
|
||||
rx = site.product_url_re
|
||||
for q in queries:
|
||||
for hit in engine.text(q, max_results=15) or []:
|
||||
s = site_for_url(hit.url)
|
||||
if not s or s.domain != site.domain:
|
||||
continue
|
||||
if rx is not None and not rx.search(hit.url):
|
||||
continue
|
||||
if hit.url not in urls:
|
||||
urls.append(hit.url)
|
||||
if len(urls) >= limit:
|
||||
return urls
|
||||
return urls
|
||||
|
||||
|
||||
def grade_page(html: str) -> Dict[str, object]:
|
||||
products = extract_products(html)
|
||||
priced = [p for p in products if p.get("price") is not None and (p.get("currency") in (None, "INR"))]
|
||||
page = extract_page(html)
|
||||
return {
|
||||
"jsonld_products": len(products),
|
||||
"jsonld_priced": len(priced),
|
||||
"meta_price": page.get("price") is not None,
|
||||
"has_title": bool(page.get("name")),
|
||||
"spec_rows": len(page.get("properties") or {}),
|
||||
"embedded_state": embedded_state(html) is not None,
|
||||
}
|
||||
|
||||
|
||||
def probe_site(site: SiteRef, client: PoliteClient, engine: SearchEngine) -> Dict[str, object]:
|
||||
"""Returns {"outcome", "robots_allowed", "evidence"}; never raises."""
|
||||
if site.policy == "serp_only":
|
||||
return {"outcome": "C", "robots_allowed": None,
|
||||
"evidence": {"reason": "policy serp_only: this site is never fetched directly"}}
|
||||
urls = sample_product_urls(site, engine)
|
||||
if not urls:
|
||||
return {"outcome": "C", "robots_allowed": None,
|
||||
"evidence": {"reason": "no product URLs found through web search"}}
|
||||
pages: List[dict] = []
|
||||
robots_any: Optional[bool] = None
|
||||
for url in urls:
|
||||
res = client.get(url)
|
||||
entry = {"url": url, "status": res.status, "outcome": res.outcome}
|
||||
robots_any = res.robots_allowed if robots_any is None else (robots_any or bool(res.robots_allowed))
|
||||
if res.ok:
|
||||
entry.update(grade_page(res.text))
|
||||
pages.append(entry)
|
||||
if res.outcome in ("captcha", "blocked", "breaker_open"):
|
||||
break
|
||||
ok_pages = [p for p in pages if p["outcome"] == "ok"]
|
||||
if any(p["outcome"] in ("captcha", "blocked") for p in pages):
|
||||
outcome, reason = "C", "blocked or bot check - not fetched again until the breaker cools down"
|
||||
elif all(p["outcome"] == "robots_disallowed" for p in pages):
|
||||
outcome, reason = "C", "robots.txt disallows product pages"
|
||||
elif not ok_pages:
|
||||
outcome, reason = "C", "product pages could not be fetched"
|
||||
elif any(p.get("jsonld_priced") for p in ok_pages):
|
||||
outcome, reason = "A", "schema.org Product with an INR offer"
|
||||
elif any(p.get("has_title") and (p.get("spec_rows") or p.get("meta_price") or p.get("jsonld_products")) for p in ok_pages):
|
||||
outcome, reason = "B", "product details readable from HTML; no structured price"
|
||||
else:
|
||||
outcome, reason = "C", "no product data without JavaScript"
|
||||
return {"outcome": outcome, "robots_allowed": robots_any, "evidence": {"reason": reason, "pages": pages}}
|
||||
132
backend/app/electronics/reference/__init__.py
Normal file
132
backend/app/electronics/reference/__init__.py
Normal file
@@ -0,0 +1,132 @@
|
||||
"""Reference data (brands, categories, sites, spec dictionary) loaded from YAML."""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from functools import lru_cache
|
||||
from pathlib import Path
|
||||
from typing import Dict, List, Optional
|
||||
|
||||
import yaml
|
||||
|
||||
_DIR = Path(__file__).resolve().parent
|
||||
|
||||
|
||||
def slugify(text: str) -> str:
|
||||
return re.sub(r"[^a-z0-9]+", "-", text.lower()).strip("-")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class BrandRef:
|
||||
name: str
|
||||
slug: str
|
||||
categories: tuple
|
||||
aliases: tuple
|
||||
sub_brands: tuple
|
||||
official: tuple
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class CategoryRef:
|
||||
slug: str
|
||||
name: str
|
||||
search_terms: tuple
|
||||
query_terms: tuple = ()
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SiteRef:
|
||||
domain: str
|
||||
name: str
|
||||
kind: str
|
||||
region: str
|
||||
policy: str
|
||||
product_url: Optional[str] = None
|
||||
pincode_param: Optional[str] = None
|
||||
brand_slug: Optional[str] = None
|
||||
|
||||
@property
|
||||
def product_url_re(self) -> Optional[re.Pattern]:
|
||||
return re.compile(self.product_url) if self.product_url else None
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Reference:
|
||||
brands: Dict[str, BrandRef]
|
||||
categories: Dict[str, CategoryRef]
|
||||
sites: Dict[str, SiteRef]
|
||||
spec_keys: Dict[str, dict] = field(default_factory=dict)
|
||||
|
||||
def brands_for(self, category: str) -> List[BrandRef]:
|
||||
return [b for b in self.brands.values() if category in b.categories]
|
||||
|
||||
|
||||
def _load_yaml(name: str) -> dict:
|
||||
return yaml.safe_load((_DIR / name).read_text(encoding="utf-8")) or {}
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def load_reference() -> Reference:
|
||||
raw_brands = _load_yaml("brands.yaml")
|
||||
brands: Dict[str, BrandRef] = {}
|
||||
for b in raw_brands.get("brands", []):
|
||||
slug = slugify(b["name"])
|
||||
brands[slug] = BrandRef(
|
||||
name=b["name"],
|
||||
slug=slug,
|
||||
categories=tuple(b.get("categories", [])),
|
||||
aliases=tuple(a.lower() for a in b.get("aliases", [])),
|
||||
sub_brands=tuple(s.lower() for s in b.get("sub_brands", [])),
|
||||
official=tuple(b.get("official", [])),
|
||||
)
|
||||
categories = {
|
||||
c["slug"]: CategoryRef(c["slug"], c["name"], tuple(c.get("search_terms", [])),
|
||||
tuple(c.get("query_terms", [])))
|
||||
for c in raw_brands.get("categories", [])
|
||||
}
|
||||
|
||||
sites: Dict[str, SiteRef] = {}
|
||||
for s in _load_yaml("sites.yaml").get("sites", []):
|
||||
sites[s["domain"]] = SiteRef(
|
||||
domain=s["domain"],
|
||||
name=s["name"],
|
||||
kind=s["kind"],
|
||||
region=s["region"],
|
||||
policy=s["policy"],
|
||||
product_url=s.get("product_url"),
|
||||
pincode_param=s.get("pincode_param"),
|
||||
)
|
||||
# Every brand's official domains become sites of their own. They are
|
||||
# probed like any retailer - an official page is the best evidence there is.
|
||||
for b in brands.values():
|
||||
for domain in b.official:
|
||||
sites.setdefault(
|
||||
domain,
|
||||
SiteRef(
|
||||
domain=domain,
|
||||
name=f"{b.name} (official)",
|
||||
kind="brand_official",
|
||||
region="national",
|
||||
policy="probe",
|
||||
brand_slug=b.slug,
|
||||
),
|
||||
)
|
||||
|
||||
spec_keys = _load_yaml("spec_keys.yaml").get("categories", {})
|
||||
return Reference(brands=brands, categories=categories, sites=sites, spec_keys=spec_keys)
|
||||
|
||||
|
||||
def site_for_url(url: str) -> Optional[SiteRef]:
|
||||
"""The registered site a URL belongs to (subdomains included), or None."""
|
||||
from urllib.parse import urlparse
|
||||
|
||||
host = (urlparse(url).hostname or "").lower()
|
||||
if not host:
|
||||
return None
|
||||
ref = load_reference()
|
||||
best: Optional[SiteRef] = None
|
||||
for domain, site in ref.sites.items():
|
||||
if host == domain or host.endswith("." + domain):
|
||||
if best is None or len(domain) > len(best.domain):
|
||||
best = site
|
||||
return best
|
||||
100
backend/app/electronics/reference/brands.yaml
Normal file
100
backend/app/electronics/reference/brands.yaml
Normal file
@@ -0,0 +1,100 @@
|
||||
# Brand allow-list. A listing whose brand does not resolve to one of these is
|
||||
# rejected - the catalogue is closed-world by design.
|
||||
#
|
||||
# aliases spellings seen on retail pages (matched case-insensitively,
|
||||
# longest alias first, as a whole word at the start of a title)
|
||||
# sub_brands product families sold under a parent brand. They resolve to the
|
||||
# parent, and are kept as the product family.
|
||||
# official the brand's own Indian web domains. A product page on one of
|
||||
# these is the strongest evidence that a product exists.
|
||||
brands:
|
||||
- name: Samsung
|
||||
categories: [mobiles, laptops]
|
||||
aliases: [samsung]
|
||||
official: [samsung.com]
|
||||
- name: Apple
|
||||
categories: [mobiles, laptops]
|
||||
aliases: [apple]
|
||||
sub_brands: [iphone, macbook]
|
||||
official: [apple.com]
|
||||
- name: Xiaomi
|
||||
categories: [mobiles]
|
||||
aliases: [xiaomi]
|
||||
sub_brands: [redmi, poco, mi]
|
||||
official: [mi.com]
|
||||
- name: OnePlus
|
||||
categories: [mobiles]
|
||||
aliases: [oneplus, one plus]
|
||||
official: [oneplus.in]
|
||||
- name: Vivo
|
||||
categories: [mobiles]
|
||||
aliases: [vivo]
|
||||
sub_brands: [iqoo]
|
||||
official: [vivo.com, iqoo.com]
|
||||
- name: Oppo
|
||||
categories: [mobiles]
|
||||
aliases: [oppo]
|
||||
official: [oppo.com]
|
||||
- name: Realme
|
||||
categories: [mobiles]
|
||||
aliases: [realme]
|
||||
sub_brands: [narzo]
|
||||
official: [realme.com]
|
||||
- name: Motorola
|
||||
categories: [mobiles]
|
||||
aliases: [motorola, moto]
|
||||
official: [motorola.co.in, motorola.com]
|
||||
- name: Google
|
||||
categories: [mobiles]
|
||||
aliases: [google]
|
||||
sub_brands: [pixel]
|
||||
official: [store.google.com]
|
||||
- name: Nothing
|
||||
categories: [mobiles]
|
||||
aliases: [nothing]
|
||||
sub_brands: [cmf]
|
||||
official: [nothing.tech]
|
||||
- name: HP
|
||||
categories: [laptops]
|
||||
aliases: [hp, hewlett packard]
|
||||
sub_brands: [omen, victus, pavilion, envy, spectre]
|
||||
official: [hp.com]
|
||||
- name: Dell
|
||||
categories: [laptops]
|
||||
aliases: [dell]
|
||||
sub_brands: [alienware, inspiron, vostro, latitude, xps]
|
||||
official: [dell.com]
|
||||
- name: Lenovo
|
||||
categories: [laptops]
|
||||
aliases: [lenovo]
|
||||
sub_brands: [thinkpad, ideapad, legion, yoga, thinkbook, loq]
|
||||
official: [lenovo.com]
|
||||
- name: Asus
|
||||
categories: [laptops]
|
||||
aliases: [asus]
|
||||
sub_brands: [rog, tuf, vivobook, zenbook]
|
||||
official: [asus.com]
|
||||
- name: Acer
|
||||
categories: [laptops]
|
||||
aliases: [acer]
|
||||
sub_brands: [aspire, nitro, predator, swift]
|
||||
official: [acer.com]
|
||||
- name: MSI
|
||||
categories: [laptops]
|
||||
aliases: [msi]
|
||||
official: [msi.com]
|
||||
|
||||
# search_terms: the category word used when probing brand sites.
|
||||
# query_terms: appended to `site:<platform> <brand>` during discovery. They
|
||||
# read like the variant part of a product title, which is what
|
||||
# makes search engines return single product pages rather than
|
||||
# category or blog pages.
|
||||
categories:
|
||||
- slug: mobiles
|
||||
name: Mobiles
|
||||
search_terms: [smartphone, mobile phone]
|
||||
query_terms: ["5G 8GB RAM 128GB", "5G 8GB 256GB", "12GB RAM 256GB"]
|
||||
- slug: laptops
|
||||
name: Laptops
|
||||
search_terms: [laptop]
|
||||
query_terms: ["laptop 16GB RAM 512GB SSD", "laptop 8GB RAM 512GB SSD"]
|
||||
77
backend/app/electronics/reference/sites.yaml
Normal file
77
backend/app/electronics/reference/sites.yaml
Normal file
@@ -0,0 +1,77 @@
|
||||
# Retail platforms.
|
||||
#
|
||||
# kind marketplace | national_chain | tn_regional
|
||||
# region national | TN (TN = a Tamil Nadu retail chain)
|
||||
# policy serp_only -> NEVER fetched directly; everything comes from web
|
||||
# search results (titles, snippets, image results)
|
||||
# probe -> fetched only if the site probe grades it A or B
|
||||
# (robots.txt allows, HTTP 200, no CAPTCHA); otherwise
|
||||
# it falls back to search results like serp_only
|
||||
# product_url regex a URL must match to count as a single product page.
|
||||
# Group 1, when present, is the site's own product id.
|
||||
# pincode_param optional query parameter the site accepts for a delivery
|
||||
# pincode. Only sites that actually honour it get
|
||||
# pincode_applied=true on their prices.
|
||||
#
|
||||
# Brand official sites are generated from brands.yaml (kind brand_official).
|
||||
sites:
|
||||
- domain: amazon.in
|
||||
name: Amazon.in
|
||||
kind: marketplace
|
||||
region: national
|
||||
policy: serp_only
|
||||
product_url: '/(?:dp|gp/product)/([A-Z0-9]{10})'
|
||||
- domain: flipkart.com
|
||||
name: Flipkart
|
||||
kind: marketplace
|
||||
region: national
|
||||
policy: serp_only
|
||||
product_url: '/p/(itm[0-9a-z]+)'
|
||||
- domain: croma.com
|
||||
name: Croma
|
||||
kind: national_chain
|
||||
region: national
|
||||
policy: probe
|
||||
product_url: '/p/(\d{5,})'
|
||||
- domain: reliancedigital.in
|
||||
name: Reliance Digital
|
||||
kind: national_chain
|
||||
region: national
|
||||
policy: probe
|
||||
product_url: '(?:/p/|/product/[^?#]*?-)(\d{6,})'
|
||||
- domain: vijaysales.com
|
||||
name: Vijay Sales
|
||||
kind: national_chain
|
||||
region: national
|
||||
policy: probe
|
||||
product_url: '/p/(?:P?)(\d{3,})/'
|
||||
- domain: tatacliq.com
|
||||
name: Tata CLiQ
|
||||
kind: marketplace
|
||||
region: national
|
||||
policy: probe
|
||||
product_url: '/p-(mp\d+)'
|
||||
- domain: poorvika.com
|
||||
name: Poorvika
|
||||
kind: tn_regional
|
||||
region: TN
|
||||
policy: probe
|
||||
product_url: '/([a-z0-9-]{8,})/p/?$'
|
||||
- domain: sangeethamobiles.com
|
||||
name: Sangeetha Mobiles
|
||||
kind: tn_regional
|
||||
region: TN
|
||||
policy: probe
|
||||
product_url: '(?i)/product-?details/(?:[^/?#]+/)?(\d+)'
|
||||
- domain: vasanthandco.in
|
||||
name: Vasanth & Co
|
||||
kind: tn_regional
|
||||
region: TN
|
||||
policy: probe
|
||||
product_url: '/(?:product|products)/([a-z0-9-]{8,})'
|
||||
- domain: viveks.com
|
||||
name: Viveks
|
||||
kind: tn_regional
|
||||
region: TN
|
||||
policy: probe
|
||||
product_url: '/([a-z0-9-]{8,})\.html$'
|
||||
127
backend/app/electronics/reference/spec_keys.yaml
Normal file
127
backend/app/electronics/reference/spec_keys.yaml
Normal file
@@ -0,0 +1,127 @@
|
||||
# Canonical specification keys per category.
|
||||
#
|
||||
# type number | text | enum
|
||||
# unit canonical unit for numbers (values are converted into it)
|
||||
# synonyms spec labels seen on retail/brand pages (case-insensitive,
|
||||
# punctuation ignored). A label maps to the first key that lists it.
|
||||
# range plausible [min, max] after conversion; values outside are dropped
|
||||
# values allowed canonical values for enums, each with its match words
|
||||
#
|
||||
# A value is only ever stored if it was read from a page or snippet. Nothing
|
||||
# here supplies a default.
|
||||
categories:
|
||||
mobiles:
|
||||
ram_gb:
|
||||
type: number
|
||||
unit: GB
|
||||
range: [1, 32]
|
||||
synonyms: [ram, memory ram, ram size, ram capacity, installed ram, system memory]
|
||||
storage_gb:
|
||||
type: number
|
||||
unit: GB
|
||||
range: [8, 2048]
|
||||
synonyms: [internal storage, storage, rom, internal memory, storage capacity, inbuilt memory, memory storage capacity]
|
||||
display_inch:
|
||||
type: number
|
||||
unit: inch
|
||||
range: [3, 9]
|
||||
synonyms: [display size, screen size, display, screen size inches, standing screen display size]
|
||||
display_type:
|
||||
type: enum
|
||||
synonyms: [display type, screen type, display technology, panel type]
|
||||
values:
|
||||
AMOLED: [amoled, super amoled, dynamic amoled, pole amoled, fluid amoled]
|
||||
OLED: [oled, super retina, ltpo oled]
|
||||
LCD: [lcd, ips lcd, tft, ips]
|
||||
refresh_hz:
|
||||
type: number
|
||||
unit: Hz
|
||||
range: [30, 240]
|
||||
synonyms: [refresh rate, screen refresh rate, display refresh rate]
|
||||
processor:
|
||||
type: text
|
||||
synonyms: [processor, chipset, processor name, soc, cpu, processor brand]
|
||||
rear_camera_mp:
|
||||
type: number
|
||||
unit: MP
|
||||
range: [2, 250]
|
||||
synonyms: [rear camera, primary camera, main camera, back camera, rear camera resolution, primary camera resolution]
|
||||
front_camera_mp:
|
||||
type: number
|
||||
unit: MP
|
||||
range: [2, 60]
|
||||
synonyms: [front camera, secondary camera, selfie camera, front camera resolution]
|
||||
battery_mah:
|
||||
type: number
|
||||
unit: mAh
|
||||
range: [1000, 10000]
|
||||
synonyms: [battery capacity, battery, battery power, battery capacity mah]
|
||||
os:
|
||||
type: enum
|
||||
synonyms: [operating system, os, os version]
|
||||
values:
|
||||
Android: [android]
|
||||
iOS: [ios]
|
||||
network:
|
||||
type: enum
|
||||
synonyms: [network type, network, cellular technology, connectivity technology, network connectivity]
|
||||
values:
|
||||
5G: [5g]
|
||||
4G: [4g, lte]
|
||||
colour:
|
||||
type: text
|
||||
synonyms: [colour, color, colour name, color name]
|
||||
laptops:
|
||||
processor:
|
||||
type: text
|
||||
synonyms: [processor, processor name, cpu, processor model, processor type]
|
||||
ram_gb:
|
||||
type: number
|
||||
unit: GB
|
||||
range: [2, 128]
|
||||
synonyms: [ram, ram size, memory, system memory, installed ram, ram capacity]
|
||||
storage_gb:
|
||||
type: number
|
||||
unit: GB
|
||||
range: [32, 8192]
|
||||
synonyms: [ssd capacity, storage, hard disk size, hard drive size, storage capacity, ssd, internal storage]
|
||||
storage_type:
|
||||
type: enum
|
||||
synonyms: [storage type, hard disk type, hard drive interface, drive type]
|
||||
values:
|
||||
SSD: [ssd, nvme, solid state]
|
||||
HDD: [hdd, hard disk drive]
|
||||
eMMC: [emmc]
|
||||
display_inch:
|
||||
type: number
|
||||
unit: inch
|
||||
range: [10, 19]
|
||||
synonyms: [screen size, display size, standing screen display size, display]
|
||||
resolution:
|
||||
type: text
|
||||
synonyms: [resolution, screen resolution, display resolution, maximum display resolution]
|
||||
gpu:
|
||||
type: text
|
||||
synonyms: [graphics, graphics processor, gpu, graphic processor, graphics coprocessor, graphics card]
|
||||
os:
|
||||
type: enum
|
||||
synonyms: [operating system, os]
|
||||
values:
|
||||
Windows: [windows]
|
||||
macOS: [macos, mac os]
|
||||
ChromeOS: [chrome os, chromeos]
|
||||
Linux: [linux, ubuntu]
|
||||
DOS: [dos, free dos, freedos]
|
||||
weight_kg:
|
||||
type: number
|
||||
unit: kg
|
||||
range: [0.5, 5]
|
||||
synonyms: [weight, item weight, product weight, laptop weight]
|
||||
battery_wh:
|
||||
type: number
|
||||
unit: Wh
|
||||
range: [20, 120]
|
||||
synonyms: [battery capacity, battery, battery power]
|
||||
colour:
|
||||
type: text
|
||||
synonyms: [colour, color]
|
||||
96
backend/app/electronics/reviews.py
Normal file
96
backend/app/electronics/reviews.py
Normal file
@@ -0,0 +1,96 @@
|
||||
"""Which real customer reviews to show for a product, and in what mix.
|
||||
|
||||
Every review passed in here was read from a product page's own schema.org
|
||||
data (see extract/jsonld.py); this module only classifies and selects - it
|
||||
never writes, rewrites or summarises review text.
|
||||
|
||||
Sentiment is the reviewer's own star rating, nothing inferred:
|
||||
>= 4 positive, >= 3 neutral, < 3 negative.
|
||||
|
||||
The mix follows the product's overall rating, so the reviews shown read like
|
||||
the rating does:
|
||||
rating >= 4.0 mostly positive, some neutral, a little negative
|
||||
3.0 < rating < 4.0 mostly neutral, some positive, a little negative
|
||||
rating <= 3.0 mostly negative, a little positive and neutral
|
||||
When a group has too few reviews its slots go to the other groups, in the
|
||||
same priority order. Nothing is ever padded: if only 3 real reviews exist,
|
||||
3 are shown.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from decimal import Decimal
|
||||
from typing import Any, Dict, List, Optional, Sequence, Tuple
|
||||
|
||||
POSITIVE, NEUTRAL, NEGATIVE = "positive", "neutral", "negative"
|
||||
MAX_REVIEWS = 10
|
||||
|
||||
# (group, share of MAX_REVIEWS), highest priority first.
|
||||
_MIX_HIGH: Tuple[Tuple[str, int], ...] = ((POSITIVE, 6), (NEUTRAL, 3), (NEGATIVE, 1))
|
||||
_MIX_MID: Tuple[Tuple[str, int], ...] = ((NEUTRAL, 5), (POSITIVE, 3), (NEGATIVE, 2))
|
||||
_MIX_LOW: Tuple[Tuple[str, int], ...] = ((NEGATIVE, 6), (POSITIVE, 2), (NEUTRAL, 2))
|
||||
|
||||
|
||||
def sentiment_for(rating: Any) -> Optional[str]:
|
||||
"""The group a reviewer's own star rating puts a review in; None when the
|
||||
review states no rating."""
|
||||
if rating is None:
|
||||
return None
|
||||
try:
|
||||
value = Decimal(str(rating))
|
||||
except Exception: # noqa: BLE001
|
||||
return None
|
||||
if value >= 4:
|
||||
return POSITIVE
|
||||
if value >= 3:
|
||||
return NEUTRAL
|
||||
return NEGATIVE
|
||||
|
||||
|
||||
def mix_for(product_rating: Any) -> Tuple[Tuple[str, int], ...]:
|
||||
if product_rating is None:
|
||||
return _MIX_MID # no overall rating stated: a balanced view
|
||||
value = Decimal(str(product_rating))
|
||||
if value >= 4:
|
||||
return _MIX_HIGH
|
||||
if value > 3:
|
||||
return _MIX_MID
|
||||
return _MIX_LOW
|
||||
|
||||
|
||||
def _rank_key(review: Dict[str, Any]) -> tuple:
|
||||
# Newest first (ISO dates sort as text), then the more substantial review.
|
||||
return (str(review.get("review_date") or ""), len(review.get("body") or ""))
|
||||
|
||||
|
||||
def select_reviews(product_rating: Any, reviews: Sequence[Dict[str, Any]],
|
||||
max_n: int = MAX_REVIEWS) -> List[Dict[str, Any]]:
|
||||
"""Up to `max_n` of `reviews`, mixed by sentiment as described above.
|
||||
|
||||
Reviews without a star rating have no sentiment and are not shown: there
|
||||
is no honest way to place them in the mix.
|
||||
"""
|
||||
groups: Dict[str, List[Dict[str, Any]]] = {POSITIVE: [], NEUTRAL: [], NEGATIVE: []}
|
||||
seen = set()
|
||||
for r in reviews:
|
||||
s = r.get("sentiment") or sentiment_for(r.get("rating"))
|
||||
key = (r.get("body") or "").strip().lower()
|
||||
if s is None or not key or key in seen:
|
||||
continue
|
||||
seen.add(key)
|
||||
groups[s].append({**r, "sentiment": s})
|
||||
for g in groups.values():
|
||||
g.sort(key=_rank_key, reverse=True)
|
||||
|
||||
mix = mix_for(product_rating)
|
||||
scale = max_n / MAX_REVIEWS
|
||||
quota = {g: int(round(n * scale)) for g, n in mix}
|
||||
picked: Dict[str, List[Dict[str, Any]]] = {g: groups[g][: quota[g]] for g, _ in mix}
|
||||
# Hand unused slots to the other groups, in priority order.
|
||||
spare = max_n - sum(len(v) for v in picked.values())
|
||||
for g, _ in mix:
|
||||
if spare <= 0:
|
||||
break
|
||||
extra = groups[g][len(picked[g]): len(picked[g]) + spare]
|
||||
picked[g].extend(extra)
|
||||
spare -= len(extra)
|
||||
return [r for g, _ in mix for r in picked[g]]
|
||||
0
backend/app/electronics/search/__init__.py
Normal file
0
backend/app/electronics/search/__init__.py
Normal file
65
backend/app/electronics/search/engine.py
Normal file
65
backend/app/electronics/search/engine.py
Normal file
@@ -0,0 +1,65 @@
|
||||
"""Cached, budgeted access to the search providers.
|
||||
|
||||
Results are cached in elec.search_cache so a re-run does not query again
|
||||
within SEARCH_CACHE_TTL_HOURS, and each run has a query budget so a large
|
||||
brand list cannot hammer the providers.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import Dict, List, Optional
|
||||
|
||||
from app.electronics.db import repository as repo
|
||||
from app.electronics.search.providers import DuckDuckGoProvider, GoogleCseProvider, SearchHit
|
||||
from app.infrastructure.settings import GOOGLE_CSE_DAILY_QUOTA, SEARCH_CACHE_TTL_HOURS
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class SearchEngine:
|
||||
def __init__(self, *, budget: int = 200, use_cache: bool = True) -> None:
|
||||
self.budget = budget
|
||||
self.use_cache = use_cache
|
||||
self.used = 0
|
||||
self.stats: Dict[str, int] = {"cache_hits": 0, "queries": 0, "unavailable": 0}
|
||||
self.ddg = DuckDuckGoProvider()
|
||||
self.google = GoogleCseProvider(
|
||||
quota_left=lambda: GOOGLE_CSE_DAILY_QUOTA - repo.google_queries_today()
|
||||
)
|
||||
|
||||
def _ask(self, provider, kind: str, query: str, max_results: int) -> Optional[List[SearchHit]]:
|
||||
if not provider.enabled:
|
||||
return None
|
||||
cached = repo.search_cache_get(provider.name, kind, query, SEARCH_CACHE_TTL_HOURS) if self.use_cache else None
|
||||
if cached is not None:
|
||||
self.stats["cache_hits"] += 1
|
||||
return [SearchHit.from_dict(d) for d in cached]
|
||||
if self.used >= self.budget:
|
||||
logger.info("Search budget (%d) spent; skipping %r", self.budget, query)
|
||||
return None
|
||||
self.used += 1
|
||||
self.stats["queries"] += 1
|
||||
self.stats[f"queries_{provider.name}"] = self.stats.get(f"queries_{provider.name}", 0) + 1
|
||||
hits = provider.text(query, max_results) if kind == "text" else provider.images(query, max_results)
|
||||
if hits is None:
|
||||
self.stats["unavailable"] += 1
|
||||
return None
|
||||
repo.search_cache_put(provider.name, kind, query, [h.to_dict() for h in hits])
|
||||
return hits
|
||||
|
||||
def _run(self, kind: str, query: str, max_results: int, providers: str) -> Optional[List[SearchHit]]:
|
||||
"""providers: "default" = DuckDuckGo, with Google only when DuckDuckGo
|
||||
gives no answer (keeps the 100/day Google quota for price lookups);
|
||||
"google" = Google only."""
|
||||
if providers == "google":
|
||||
return self._ask(self.google, kind, query, max_results)
|
||||
hits = self._ask(self.ddg, kind, query, max_results)
|
||||
if hits is None:
|
||||
hits = self._ask(self.google, kind, query, max_results)
|
||||
return hits
|
||||
|
||||
def text(self, query: str, max_results: int = 20, *, providers: str = "default") -> Optional[List[SearchHit]]:
|
||||
return self._run("text", query, max_results, providers)
|
||||
|
||||
def images(self, query: str, max_results: int = 15) -> Optional[List[SearchHit]]:
|
||||
return self._run("images", query, max_results, "default")
|
||||
234
backend/app/electronics/search/providers.py
Normal file
234
backend/app/electronics/search/providers.py
Normal file
@@ -0,0 +1,234 @@
|
||||
"""Web search: DuckDuckGo (ddgs, no key) and, when configured, Google
|
||||
Programmable Search.
|
||||
|
||||
Three outcomes, never collapsed: a list of hits, an empty list ("we asked and
|
||||
nothing matched"), or None ("we could not ask" - throttled, offline, no
|
||||
quota). A throttle is not evidence that a product is not sold anywhere.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import threading
|
||||
import time
|
||||
from dataclasses import asdict, dataclass
|
||||
from typing import Callable, List, Optional
|
||||
|
||||
import requests
|
||||
|
||||
from app.infrastructure.settings import (
|
||||
GOOGLE_API_KEY,
|
||||
GOOGLE_CSE_ID,
|
||||
SEARCH_MIN_INTERVAL_SECONDS,
|
||||
SEARCH_REGION,
|
||||
USE_DDG_SEARCH,
|
||||
USE_GOOGLE_CSE,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class SearchHit:
|
||||
url: str
|
||||
title: str
|
||||
snippet: str
|
||||
provider: str
|
||||
rank: int
|
||||
image_url: Optional[str] = None # image searches: the image itself (url = page it is on)
|
||||
# Structured offer data the search engine itself extracted from the page
|
||||
# (Google CSE "pagemap"): {"price", "currency", "availability", "raw"}.
|
||||
offer: Optional[dict] = None
|
||||
# Aggregate rating the search engine extracted from the page's own
|
||||
# structured data (Google CSE "pagemap"): {"rating", "review_count", "raw"}.
|
||||
rating: Optional[dict] = None
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return asdict(self)
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, d: dict) -> "SearchHit":
|
||||
return cls(**{k: d.get(k) for k in ("url", "title", "snippet", "provider", "rank", "image_url", "offer", "rating")})
|
||||
|
||||
|
||||
class _Pacer:
|
||||
def __init__(self, interval: float, sleep: Callable[[float], None] = time.sleep) -> None:
|
||||
self.interval = interval
|
||||
self._sleep = sleep
|
||||
self._last = 0.0
|
||||
self._lock = threading.Lock()
|
||||
|
||||
def wait(self) -> None:
|
||||
with self._lock:
|
||||
gap = self.interval - (time.monotonic() - self._last)
|
||||
if gap > 0:
|
||||
self._sleep(gap)
|
||||
self._last = time.monotonic()
|
||||
|
||||
|
||||
class DuckDuckGoProvider:
|
||||
name = "ddg"
|
||||
|
||||
def __init__(self, interval: float = SEARCH_MIN_INTERVAL_SECONDS) -> None:
|
||||
self._pacer = _Pacer(interval)
|
||||
self.enabled = USE_DDG_SEARCH
|
||||
|
||||
# "auto" rotates ddgs's engines; yahoo is a second opinion when it is throttled.
|
||||
BACKENDS = ("auto", "yahoo")
|
||||
|
||||
def text(self, query: str, max_results: int = 20) -> Optional[List[SearchHit]]:
|
||||
"""Hits, or None when no backend answered. ddgs reports a throttle and
|
||||
a genuinely empty result the same way ("No results found"), so an
|
||||
empty answer is treated as unknown rather than as "not listed"."""
|
||||
if not self.enabled:
|
||||
return None
|
||||
try:
|
||||
from ddgs import DDGS
|
||||
except ImportError:
|
||||
logger.warning("ddgs is not installed; DuckDuckGo search unavailable")
|
||||
return None
|
||||
for backend in self.BACKENDS:
|
||||
self._pacer.wait()
|
||||
try:
|
||||
with DDGS(timeout=20) as ddgs:
|
||||
rows = list(ddgs.text(query, region=SEARCH_REGION, safesearch="moderate",
|
||||
max_results=max_results, backend=backend) or [])
|
||||
except Exception as exc: # noqa: BLE001 - ddgs raises many types on throttling
|
||||
logger.info("DuckDuckGo(%s) text search gave no answer (%s): %s", backend, query, exc)
|
||||
continue
|
||||
hits = [
|
||||
SearchHit(r.get("href") or r.get("url") or "", r.get("title") or "", r.get("body") or "",
|
||||
f"{self.name}", i)
|
||||
for i, r in enumerate(rows)
|
||||
if (r.get("href") or r.get("url") or "").startswith("http")
|
||||
]
|
||||
if hits:
|
||||
return hits
|
||||
return None
|
||||
|
||||
def images(self, query: str, max_results: int = 15) -> Optional[List[SearchHit]]:
|
||||
if not self.enabled:
|
||||
return None
|
||||
try:
|
||||
from ddgs import DDGS
|
||||
except ImportError:
|
||||
return None
|
||||
self._pacer.wait()
|
||||
try:
|
||||
with DDGS(timeout=20) as ddgs:
|
||||
rows = list(ddgs.images(query, region=SEARCH_REGION, safesearch="moderate",
|
||||
max_results=max_results) or [])
|
||||
except Exception as exc: # noqa: BLE001
|
||||
if "no results" in str(exc).lower():
|
||||
return []
|
||||
logger.info("DuckDuckGo image search failed (%s): %s", query, exc)
|
||||
return None
|
||||
return [
|
||||
SearchHit(r.get("url") or "", r.get("title") or "", "", self.name, i, image_url=r.get("image"))
|
||||
for i, r in enumerate(rows)
|
||||
if str(r.get("image") or "").startswith("http") and str(r.get("url") or "").startswith("http")
|
||||
]
|
||||
|
||||
|
||||
class GoogleCseProvider:
|
||||
name = "google"
|
||||
ENDPOINT = "https://www.googleapis.com/customsearch/v1"
|
||||
|
||||
def __init__(self, quota_left: Callable[[], int] = lambda: 100) -> None:
|
||||
self.enabled = USE_GOOGLE_CSE
|
||||
self.error: Optional[str] = None
|
||||
self._quota_left = quota_left
|
||||
self._pacer = _Pacer(1.0)
|
||||
|
||||
def _call(self, query: str, extra: dict) -> Optional[List[dict]]:
|
||||
if not self.enabled:
|
||||
return None
|
||||
if self._quota_left() <= 0:
|
||||
self.error = "daily query quota used up"
|
||||
return None
|
||||
self._pacer.wait()
|
||||
try:
|
||||
resp = requests.get(
|
||||
self.ENDPOINT,
|
||||
params={"key": GOOGLE_API_KEY, "cx": GOOGLE_CSE_ID, "q": query, "gl": "in", "num": 10, **extra},
|
||||
timeout=20,
|
||||
)
|
||||
except requests.RequestException as exc:
|
||||
logger.info("Google CSE failed: %s", exc)
|
||||
return None
|
||||
if resp.status_code in (400, 401, 403):
|
||||
# A key/project problem will not fix itself mid-run: stop asking.
|
||||
try:
|
||||
message = resp.json().get("error", {}).get("message", "")
|
||||
except ValueError:
|
||||
message = resp.text[:200]
|
||||
self.enabled = False
|
||||
self.error = f"HTTP {resp.status_code}: {message}"
|
||||
logger.warning("Google Programmable Search disabled for this run - %s", self.error)
|
||||
return None
|
||||
if resp.status_code != 200:
|
||||
logger.info("Google CSE HTTP %s: %s", resp.status_code, resp.text[:200])
|
||||
return None
|
||||
return resp.json().get("items", []) or []
|
||||
|
||||
def text(self, query: str, max_results: int = 10) -> Optional[List[SearchHit]]:
|
||||
items = self._call(query, {})
|
||||
if items is None:
|
||||
return None
|
||||
return [SearchHit(i.get("link", ""), i.get("title", ""), i.get("snippet", ""), self.name, n,
|
||||
offer=pagemap_offer(i.get("pagemap") or {}),
|
||||
rating=pagemap_rating(i.get("pagemap") or {}))
|
||||
for n, i in enumerate(items[:max_results]) if i.get("link")]
|
||||
|
||||
def images(self, query: str, max_results: int = 10) -> Optional[List[SearchHit]]:
|
||||
items = self._call(query, {"searchType": "image"})
|
||||
if items is None:
|
||||
return None
|
||||
return [SearchHit((i.get("image") or {}).get("contextLink", ""), i.get("title", ""), "", self.name, n,
|
||||
image_url=i.get("link"))
|
||||
for n, i in enumerate(items[:max_results]) if i.get("link")]
|
||||
|
||||
|
||||
def pagemap_offer(pagemap: dict) -> Optional[dict]:
|
||||
"""The offer Google extracted from the page's own structured data
|
||||
(schema.org Offer, or product:price meta tags), if any. INR only."""
|
||||
candidates = []
|
||||
for offer in pagemap.get("offer") or []:
|
||||
candidates.append((offer.get("price"), offer.get("pricecurrency"), offer.get("availability"), offer))
|
||||
for meta in pagemap.get("metatags") or []:
|
||||
price = meta.get("product:price:amount") or meta.get("og:price:amount")
|
||||
if price:
|
||||
candidates.append((price, meta.get("product:price:currency") or meta.get("og:price:currency"),
|
||||
meta.get("product:availability") or meta.get("og:availability"),
|
||||
{k: v for k, v in meta.items() if "price" in k or "availability" in k}))
|
||||
for price, currency, availability, raw in candidates:
|
||||
if price and (currency or "").upper() == "INR":
|
||||
return {"price": str(price), "currency": "INR", "availability": availability, "raw": raw}
|
||||
return None
|
||||
|
||||
|
||||
def pagemap_rating(pagemap: dict) -> Optional[dict]:
|
||||
"""The aggregate rating Google extracted from the page's own structured
|
||||
data (schema.org AggregateRating), if any. Only a value on a 5-point
|
||||
scale is accepted."""
|
||||
for node in pagemap.get("aggregaterating") or []:
|
||||
try:
|
||||
value = float(str(node.get("ratingvalue", "")).replace(",", "."))
|
||||
except ValueError:
|
||||
continue
|
||||
best = node.get("bestrating")
|
||||
try:
|
||||
if best not in (None, "") and float(best) != 5:
|
||||
continue
|
||||
except ValueError:
|
||||
continue
|
||||
if not 0 < value <= 5:
|
||||
continue
|
||||
count = None
|
||||
for key in ("reviewcount", "ratingcount"):
|
||||
digits = "".join(ch for ch in str(node.get(key) or "") if ch.isdigit())
|
||||
if digits:
|
||||
count = int(digits)
|
||||
break
|
||||
return {"rating": round(value, 2), "review_count": count,
|
||||
"raw": {k: v for k, v in node.items() if k in ("ratingvalue", "reviewcount", "ratingcount", "bestrating")}}
|
||||
return None
|
||||
0
backend/app/infrastructure/__init__.py
Normal file
0
backend/app/infrastructure/__init__.py
Normal file
427
backend/app/infrastructure/security.py
Normal file
427
backend/app/infrastructure/security.py
Normal file
@@ -0,0 +1,427 @@
|
||||
"""
|
||||
Password hashing, access-token issuance/verification, and the Principal that
|
||||
represents an authenticated caller.
|
||||
|
||||
Two kinds of credential reach this module:
|
||||
|
||||
* Interactive users. ``POST /api/auth/login`` exchanges a username/password
|
||||
for a short-lived signed JWT. No password is ever stored - only a PBKDF2
|
||||
digest, read from the environment (``AUTH_ADMIN_PASSWORD_HASH`` /
|
||||
``AUTH_USER_PASSWORD_HASH``). Generate those with
|
||||
``python scripts/make_auth_secrets.py``.
|
||||
|
||||
* Machine consumers. A static key sent as ``X-API-Key``, mapped to a role by
|
||||
``API_KEYS``. These do not expire, so treat one as a long-lived secret and
|
||||
give each consumer its own so it can be revoked individually.
|
||||
|
||||
PBKDF2-HMAC-SHA256 is used rather than bcrypt or argon2 deliberately: it is in
|
||||
the standard library, so the slim Python image needs no compiled dependency,
|
||||
and at the iteration count below it meets OWASP's current guidance. The
|
||||
encoded form carries its own iteration count, so raising the constant later
|
||||
does not invalidate hashes already issued.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import hashlib
|
||||
import hmac
|
||||
import logging
|
||||
import secrets
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Dict, List, Optional, Tuple
|
||||
|
||||
import jwt
|
||||
|
||||
from app.infrastructure.settings import (
|
||||
API_KEYS,
|
||||
AUTH_ADMIN_PASSWORD_HASH,
|
||||
AUTH_ADMIN_USERNAME,
|
||||
AUTH_ALLOW_ANY_LOGIN,
|
||||
AUTH_ENABLED,
|
||||
AUTH_SECRET_KEY,
|
||||
AUTH_TOKEN_TTL_MINUTES,
|
||||
config_source,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Roles and permissions
|
||||
# ---------------------------------------------------------------------------
|
||||
# These mirror the permission strings the React UI already keys its navigation
|
||||
# off, so the server now enforces the same vocabulary the client was only
|
||||
# displaying. `admin` is a superuser: has_permission() grants it everything
|
||||
# rather than requiring every new permission to be added to this list.
|
||||
ROLE_PERMISSIONS: Dict[str, List[str]] = {
|
||||
"admin": [
|
||||
"view_catalog",
|
||||
"view_project_details",
|
||||
"upload_train_test",
|
||||
"allocate_discounts",
|
||||
"manage_analytics",
|
||||
"manage_nutrition",
|
||||
],
|
||||
"user": [
|
||||
"add_product",
|
||||
"upload_batch_products",
|
||||
"update_db_and_json",
|
||||
"fetch_images",
|
||||
"upload_store_inventory",
|
||||
"view_store_analytics",
|
||||
"view_nutrition_insights",
|
||||
"optimize_profits",
|
||||
],
|
||||
# An outside API client that may send spreadsheets for catalog ingestion and
|
||||
# do NOTHING else. One permission, deliberately.
|
||||
#
|
||||
# This role exists because API keys carry no per-key scoping:
|
||||
# principal_for_api_key() derives permissions entirely from the role, so
|
||||
# "upload-only" can only be expressed as a role. Reusing `user` would have
|
||||
# been less code and would also have handed an outside contributor
|
||||
# add_product, upload_batch_products and upload_store_inventory - real
|
||||
# write access to the catalog - to solve a problem that needed one verb.
|
||||
#
|
||||
# WHAT A LEAKED UPLOADER KEY COSTS. Real CPU: this permission starts the
|
||||
# 11-stage pipeline, which is the point of the endpoint. The bound is not
|
||||
# "this role cannot work" but "all ingestion, from every source, shares one
|
||||
# worker" - batch_worker runs a single batch at a time behind a queue of
|
||||
# BATCH_QUEUE_MAX, past which POST /api/uploads/catalog answers 429. So a
|
||||
# key can occupy the ingestion worker; it cannot multiply it, and it cannot
|
||||
# touch the request path the healthcheck reads.
|
||||
#
|
||||
# What it still cannot do: read the catalog, read another caller's
|
||||
# submissions (every read on that router is filtered by submitted_by), or
|
||||
# cancel, resume or delete anything.
|
||||
"uploader": [
|
||||
"upload_catalog",
|
||||
],
|
||||
}
|
||||
|
||||
VALID_ROLES = frozenset(ROLE_PERMISSIONS)
|
||||
|
||||
JWT_ALGORITHM = "HS256"
|
||||
JWT_ISSUER = "brand-catalog-rag"
|
||||
|
||||
# OWASP's floor for PBKDF2-HMAC-SHA256 at time of writing.
|
||||
_PBKDF2_ITERATIONS = 600_000
|
||||
_PBKDF2_PREFIX = "pbkdf2_sha256"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Principal:
|
||||
"""Whoever is making the current request, once their credential checks out."""
|
||||
|
||||
username: str
|
||||
role: str
|
||||
permissions: List[str] = field(default_factory=list)
|
||||
# "user" - logged in via /api/auth/login, carrying a JWT
|
||||
# "api_key" - a machine consumer from API_KEYS
|
||||
# "anonymous" - AUTH_ENABLED=false; no credential was checked at all
|
||||
kind: str = "user"
|
||||
|
||||
def has_permission(self, permission: str) -> bool:
|
||||
return self.role == "admin" or permission in self.permissions
|
||||
|
||||
|
||||
class AuthError(Exception):
|
||||
"""A credential was absent, malformed, expired, or simply wrong."""
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Password hashing
|
||||
# ---------------------------------------------------------------------------
|
||||
def hash_password(password: str, *, iterations: int = _PBKDF2_ITERATIONS) -> str:
|
||||
"""Return an encoded digest: ``pbkdf2_sha256$<iterations>$<salt>$<hash>``."""
|
||||
salt = secrets.token_bytes(16)
|
||||
digest = hashlib.pbkdf2_hmac("sha256", password.encode("utf-8"), salt, iterations)
|
||||
return "$".join(
|
||||
(
|
||||
_PBKDF2_PREFIX,
|
||||
str(iterations),
|
||||
base64.b64encode(salt).decode("ascii"),
|
||||
base64.b64encode(digest).decode("ascii"),
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
def _parse_encoded_hash(encoded: str) -> Optional[Tuple[bytes, bytes, int]]:
|
||||
"""
|
||||
Split an encoded digest into ``(salt, digest, iterations)``, or None if it
|
||||
is not one.
|
||||
|
||||
One parser, three callers. `verify_password` needs the parts, while
|
||||
`hash_is_wellformed` and `describe_password_hash` need only the verdict -
|
||||
and a login failing because the *configured* hash is corrupt is a different
|
||||
incident from a wrong password, so the two must agree on what "corrupt"
|
||||
means. Two copies of this parse would eventually disagree.
|
||||
|
||||
Values arrive here straight from the environment, so a hash pasted into a
|
||||
deployment platform's form field as "pbkdf2_sha256$..." is unwrapped rather
|
||||
than rejected: the surrounding quotes are almost never intended as part of
|
||||
the secret, and the failure they cause otherwise is a silent 401.
|
||||
"""
|
||||
if not encoded:
|
||||
return None
|
||||
encoded = encoded.strip().strip("'\"")
|
||||
try:
|
||||
prefix, raw_iterations, raw_salt, raw_digest = encoded.split("$")
|
||||
if prefix != _PBKDF2_PREFIX:
|
||||
return None
|
||||
# validate=True so junk is rejected rather than silently discarded:
|
||||
# b64decode's default drops non-alphabet characters, which would let a
|
||||
# subtly corrupted hash decode to the wrong bytes and fail as a "wrong
|
||||
# password" instead of as the configuration error it is.
|
||||
# binascii.Error subclasses ValueError, so it is caught below.
|
||||
salt = base64.b64decode(raw_salt, validate=True)
|
||||
digest = base64.b64decode(raw_digest, validate=True)
|
||||
iterations = int(raw_iterations)
|
||||
except (ValueError, TypeError):
|
||||
return None
|
||||
# A structurally valid string that decodes to nothing is still unusable,
|
||||
# and PBKDF2 rejects a non-positive iteration count by raising.
|
||||
if not salt or not digest or iterations < 1:
|
||||
return None
|
||||
return salt, digest, iterations
|
||||
|
||||
|
||||
def hash_is_wellformed(encoded: str) -> bool:
|
||||
"""Whether a configured digest can be checked against at all.
|
||||
|
||||
Distinct from "does the password match": this asks whether the credential
|
||||
*store* is usable, which is a deployment fault rather than a sign-in one.
|
||||
"""
|
||||
return _parse_encoded_hash(encoded) is not None
|
||||
|
||||
|
||||
def password_hash_fingerprint(encoded: str) -> str:
|
||||
"""
|
||||
A short, non-reversible identifier for a configured digest.
|
||||
|
||||
Safe to log and to publish: it is a truncated SHA-256 of the *encoded
|
||||
digest*, and that digest already embeds a 16-byte random salt, so this says
|
||||
which credential is loaded without saying anything about the password
|
||||
behind it. It exists so a running deployment can be compared against the
|
||||
config it was supposed to have been built from - the failure this project
|
||||
actually hit - without moving a secret in order to do the comparison.
|
||||
"""
|
||||
if not encoded:
|
||||
return ""
|
||||
return hashlib.sha256(encoded.strip().strip("'\"").encode("utf-8")).hexdigest()[:12]
|
||||
|
||||
|
||||
def api_key_fingerprint(name: str, secret: str) -> str:
|
||||
"""
|
||||
A short, non-reversible identifier for a configured API key.
|
||||
|
||||
Same purpose as password_hash_fingerprint - say *which* credential is loaded
|
||||
without moving the credential - but the safety argument is different and
|
||||
worth stating. That function digests an encoded hash which already embeds a
|
||||
16-byte random salt. An API key has no salt, so the name is mixed in here to
|
||||
keep two consumers that were mistakenly issued the same secret from
|
||||
fingerprinting identically, and settings._parse_api_keys enforces a minimum
|
||||
secret length so the digest cannot be walked back with a wordlist.
|
||||
"""
|
||||
if not secret:
|
||||
return ""
|
||||
cleaned = secret.strip().strip("'\"")
|
||||
material = f"{name}:{cleaned}"
|
||||
return hashlib.sha256(material.encode("utf-8")).hexdigest()[:12]
|
||||
|
||||
|
||||
def describe_api_keys() -> List[Dict[str, object]]:
|
||||
"""Every configured key as {name, role, fingerprint}, sorted by name.
|
||||
|
||||
Sorted so two deployments' /api/health output can be diffed line for line;
|
||||
API_KEYS is keyed by secret, whose iteration order says nothing useful.
|
||||
"""
|
||||
return sorted(
|
||||
(
|
||||
{"name": name, "role": role, "fingerprint": api_key_fingerprint(name, secret)}
|
||||
for secret, (name, role) in API_KEYS.items()
|
||||
),
|
||||
key=lambda entry: entry["name"],
|
||||
)
|
||||
|
||||
|
||||
def describe_password_hash(encoded: str) -> Dict[str, object]:
|
||||
"""A loggable/publishable summary of a configured digest. Never its bytes."""
|
||||
parsed = _parse_encoded_hash(encoded)
|
||||
return {
|
||||
"valid": parsed is not None,
|
||||
"algorithm": _PBKDF2_PREFIX if parsed is not None else None,
|
||||
"iterations": parsed[2] if parsed is not None else None,
|
||||
"fingerprint": password_hash_fingerprint(encoded),
|
||||
}
|
||||
|
||||
|
||||
def auth_config_summary() -> Dict[str, object]:
|
||||
"""
|
||||
The effective authentication configuration, in a form safe to both log and
|
||||
publish. Contains no password and no hash - only the fingerprint.
|
||||
|
||||
This is deliberately one function with two callers (the startup log in
|
||||
app/main.py and GET /api/health), because its entire purpose is letting two
|
||||
*deployments* be compared, and that only works if both report the same
|
||||
fields computed the same way.
|
||||
|
||||
`*_source` is the field that earns this its keep. A value of "process-env"
|
||||
means the container's own environment supplied it and the .env file baked
|
||||
into the image was ignored - which is invisible from anywhere else, and is
|
||||
precisely how a corrected credential can keep failing after a redeploy.
|
||||
|
||||
The same argument is why the API keys are summarised here. backend/Dockerfile
|
||||
copies .env.production in at BUILD time, so a key added to that file and then
|
||||
merely restarted is not present in the running process - and from outside,
|
||||
an undeployed key is indistinguishable from a wrong one, because both are
|
||||
just a 401. Publishing the names and fingerprints answers "is my key on this
|
||||
deployment?" without anyone having to send the secret to find out.
|
||||
"""
|
||||
described = describe_password_hash(AUTH_ADMIN_PASSWORD_HASH)
|
||||
return {
|
||||
"enabled": AUTH_ENABLED,
|
||||
"allow_any_login": AUTH_ALLOW_ANY_LOGIN,
|
||||
"admin_username": AUTH_ADMIN_USERNAME,
|
||||
"password_hash_valid": bool(described["valid"]),
|
||||
"password_hash_iterations": described["iterations"],
|
||||
"password_hash_fingerprint": described["fingerprint"],
|
||||
"admin_username_source": config_source("AUTH_ADMIN_USERNAME"),
|
||||
"password_hash_source": config_source("AUTH_ADMIN_PASSWORD_HASH"),
|
||||
"api_keys_count": len(API_KEYS),
|
||||
"api_keys": describe_api_keys(),
|
||||
"api_keys_source": config_source("API_KEYS"),
|
||||
}
|
||||
|
||||
|
||||
def verify_password(password: str, encoded: str) -> bool:
|
||||
"""
|
||||
Check a password against an encoded digest.
|
||||
|
||||
Returns False rather than raising on a malformed digest: a typo in
|
||||
AUTH_ADMIN_PASSWORD_HASH must fail the login, not 500 the endpoint and
|
||||
hand the caller a stack trace describing the credential store.
|
||||
"""
|
||||
if not encoded:
|
||||
return False
|
||||
parsed = _parse_encoded_hash(encoded)
|
||||
if parsed is None:
|
||||
logger.error(
|
||||
"A configured password hash is malformed and cannot be used. Regenerate "
|
||||
"it with: python scripts/make_auth_secrets.py"
|
||||
)
|
||||
return False
|
||||
|
||||
salt, digest, iterations = parsed
|
||||
candidate = hashlib.pbkdf2_hmac("sha256", password.encode("utf-8"), salt, iterations)
|
||||
return hmac.compare_digest(candidate, digest)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Access tokens
|
||||
# ---------------------------------------------------------------------------
|
||||
def create_access_token(
|
||||
username: str,
|
||||
role: str,
|
||||
permissions: List[str],
|
||||
*,
|
||||
ttl_minutes: Optional[int] = None,
|
||||
) -> tuple[str, int]:
|
||||
"""Issue a signed JWT. Returns ``(token, expires_in_seconds)``."""
|
||||
ttl = (ttl_minutes if ttl_minutes is not None else AUTH_TOKEN_TTL_MINUTES) * 60
|
||||
now = int(time.time())
|
||||
payload = {
|
||||
"sub": username,
|
||||
"role": role,
|
||||
"perms": permissions,
|
||||
"iss": JWT_ISSUER,
|
||||
"iat": now,
|
||||
"exp": now + ttl,
|
||||
}
|
||||
return jwt.encode(payload, AUTH_SECRET_KEY, algorithm=JWT_ALGORITHM), ttl
|
||||
|
||||
|
||||
def decode_access_token(token: str) -> Principal:
|
||||
"""
|
||||
Verify a JWT and return the Principal it names.
|
||||
|
||||
The algorithm is pinned to a single-item allow-list rather than read from
|
||||
the token header. That is what closes the two classic JWT bypasses: a token
|
||||
presenting ``alg: none``, and one presenting ``alg: HS256`` against a key
|
||||
the server intended to use asymmetrically.
|
||||
"""
|
||||
try:
|
||||
payload = jwt.decode(
|
||||
token,
|
||||
AUTH_SECRET_KEY,
|
||||
algorithms=[JWT_ALGORITHM],
|
||||
issuer=JWT_ISSUER,
|
||||
options={"require": ["exp", "iat", "sub"]},
|
||||
)
|
||||
except jwt.ExpiredSignatureError as exc:
|
||||
raise AuthError("Token has expired. Sign in again.") from exc
|
||||
except jwt.InvalidTokenError as exc:
|
||||
raise AuthError("Invalid authentication token.") from exc
|
||||
|
||||
role = payload.get("role")
|
||||
if role not in VALID_ROLES:
|
||||
raise AuthError("Token names an unknown role.")
|
||||
|
||||
perms = payload.get("perms")
|
||||
return Principal(
|
||||
username=str(payload["sub"]),
|
||||
role=role,
|
||||
# Fall back to the role's current grants if the token predates a
|
||||
# permission change, rather than trusting an arbitrary claim shape.
|
||||
permissions=list(perms) if isinstance(perms, list) else ROLE_PERMISSIONS.get(role, []),
|
||||
kind="user",
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# API keys (machine consumers)
|
||||
# ---------------------------------------------------------------------------
|
||||
def principal_for_api_key(presented: str) -> Principal:
|
||||
"""
|
||||
Resolve an ``X-API-Key`` value to a Principal.
|
||||
|
||||
Every configured key is compared even after a match, using compare_digest,
|
||||
so the time taken does not reveal how far down the list a near-miss got.
|
||||
"""
|
||||
matched: Optional[tuple[str, str]] = None
|
||||
for secret, (name, role) in API_KEYS.items():
|
||||
if hmac.compare_digest(presented, secret):
|
||||
matched = (name, role)
|
||||
if matched is None:
|
||||
raise AuthError("Invalid API key.")
|
||||
|
||||
name, role = matched
|
||||
return Principal(
|
||||
username=name,
|
||||
role=role,
|
||||
permissions=ROLE_PERMISSIONS.get(role, []),
|
||||
kind="api_key",
|
||||
)
|
||||
|
||||
|
||||
def anonymous_principal() -> Principal:
|
||||
"""
|
||||
The stand-in used when ``AUTH_ENABLED=false``.
|
||||
|
||||
It is deliberately an admin: disabling auth is meant to make local
|
||||
development frictionless, and a half-privileged anonymous caller would
|
||||
produce confusing 403s instead. Nothing calls this when auth is on.
|
||||
"""
|
||||
return Principal(
|
||||
username="anonymous",
|
||||
role="admin",
|
||||
permissions=ROLE_PERMISSIONS["admin"],
|
||||
kind="anonymous",
|
||||
)
|
||||
|
||||
|
||||
if not AUTH_ENABLED:
|
||||
logger.warning(
|
||||
"AUTH_ENABLED=false: every endpoint is unauthenticated, including catalog "
|
||||
"generation, ML training, and the upload endpoints. This is for local "
|
||||
"development only - never run it on a host reachable from the internet."
|
||||
)
|
||||
373
backend/app/infrastructure/settings.py
Normal file
373
backend/app/infrastructure/settings.py
Normal file
@@ -0,0 +1,373 @@
|
||||
"""
|
||||
Centralized configuration for the Electronics Catalog backend.
|
||||
|
||||
Every credential is read ONLY from the environment (backend/.env via
|
||||
python-dotenv, or real OS variables). Non-secret values keep safe local
|
||||
defaults.
|
||||
|
||||
LOCAL-ONLY GUARD
|
||||
----------------
|
||||
This project is a copy of the grocery catalogue, whose .env files pointed at a
|
||||
remote production database. To make it impossible to write electronics data
|
||||
there by accident, settings refuse to load unless DB_HOST is a local host and
|
||||
DB_NAME is the dedicated electronics database. See _guard_local_database().
|
||||
The one exception is an explicit production opt-in (ELEC_ALLOW_REMOTE_DB plus
|
||||
an exact host/database allowlist), used only by the production deployment.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
# Snapshotted BEFORE load_dotenv, and that ordering is the entire point.
|
||||
# load_dotenv() is called without override=True, so a variable already in the
|
||||
# process environment silently beats the .env file and keeps beating it no
|
||||
# matter how many times the file is corrected. That is not hypothetical here:
|
||||
# the deployment platform injects its Environment tab into the container, so a
|
||||
# stale value left in that tab overrides the credentials baked into the image
|
||||
# (backend/Dockerfile copies .env.production to /app/.env) and the only symptom
|
||||
# is a 401 that nothing explains. Comparing a name against this set answers
|
||||
# "which of the two won?" - see config_source() below.
|
||||
_PREEXISTING_ENV = frozenset(os.environ)
|
||||
|
||||
try:
|
||||
from dotenv import load_dotenv
|
||||
|
||||
# backend/.env (one level up from this file: app/infrastructure/settings.py)
|
||||
_env_path = Path(__file__).resolve().parents[2] / ".env"
|
||||
load_dotenv(_env_path)
|
||||
except ImportError:
|
||||
# python-dotenv not installed - fall back to whatever is already in the
|
||||
# process environment (e.g. set by the shell, Docker, systemd, CI, etc.)
|
||||
pass
|
||||
|
||||
|
||||
# Names whose raw value arrived wrapped in quotes or padded with whitespace.
|
||||
# Recorded rather than merely fixed: stripping keeps the login working, but the
|
||||
# only place the original shape is still visible is right here, before the value
|
||||
# is normalised. A quoted hash is the signature of a value pasted into a web
|
||||
# form, so surfacing it at startup is what stops the next person rediscovering
|
||||
# it from a 401. See DB_PASSWORD in .env.production for the counter-case where
|
||||
# the quotes ARE part of the secret - which is why this warns, and does not fail.
|
||||
_ENV_NEEDED_CLEANUP = set()
|
||||
|
||||
|
||||
def _clean(name: str, raw: str) -> str:
|
||||
"""Strip surrounding quotes/whitespace off an env value, remembering if it mattered."""
|
||||
cleaned = raw.strip().strip("'\"")
|
||||
if cleaned != raw:
|
||||
_ENV_NEEDED_CLEANUP.add(name)
|
||||
return cleaned
|
||||
|
||||
|
||||
def cleaned_env_names() -> list:
|
||||
"""Which settings needed quote/whitespace stripping. Reported at startup."""
|
||||
return sorted(_ENV_NEEDED_CLEANUP)
|
||||
|
||||
|
||||
def config_source(name: str) -> str:
|
||||
"""
|
||||
Where a setting's value actually came from: the process environment, the
|
||||
.env file, or this module's own default.
|
||||
|
||||
Reported at startup for the AUTH_* values (see app/main.py) so that an
|
||||
override arriving from outside the image is visible in the logs instead of
|
||||
being inferred from a failing login.
|
||||
"""
|
||||
if name in _PREEXISTING_ENV:
|
||||
return "process-env"
|
||||
if name in os.environ:
|
||||
return "env-file"
|
||||
return "default"
|
||||
|
||||
|
||||
def _bool(name: str, default: str) -> bool:
|
||||
return os.getenv(name, default).strip().lower() in {"1", "true", "yes"}
|
||||
|
||||
|
||||
def _require(name: str, *, feature_flag: str) -> str:
|
||||
"""Read a required secret. Raises if missing and the owning feature is enabled."""
|
||||
value = os.getenv(name)
|
||||
if not value:
|
||||
raise RuntimeError(
|
||||
f"Missing required environment variable '{name}'. It is required because "
|
||||
f"'{feature_flag}' is enabled. Set it in backend/.env (copy from "
|
||||
f".env.example) or disable the feature by setting {feature_flag}=false."
|
||||
)
|
||||
return value
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Paths
|
||||
# ---------------------------------------------------------------------------
|
||||
_BACKEND_ROOT = Path(__file__).resolve().parents[2]
|
||||
DATA_DIR = Path(os.getenv("DATA_DIR", "").strip() or _BACKEND_ROOT / "data")
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Ollama (local LLM) - only ever used to read text we fetched, never to invent
|
||||
# ---------------------------------------------------------------------------
|
||||
USE_OLLAMA = _bool("USE_OLLAMA", "true")
|
||||
OLLAMA_BASE_URL = os.getenv("OLLAMA_BASE_URL", "http://localhost:11434")
|
||||
OLLAMA_MODEL_NAME = os.getenv("OLLAMA_MODEL_NAME", "qwen2.5:1.5b")
|
||||
OLLAMA_TIMEOUT_SECONDS = int(os.getenv("OLLAMA_TIMEOUT_SECONDS", "120"))
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Embeddings (sentence-transformers, CPU-friendly)
|
||||
# ---------------------------------------------------------------------------
|
||||
USE_EMBEDDINGS = _bool("USE_EMBEDDINGS", "true")
|
||||
EMBEDDINGS_MODEL = os.getenv("EMBEDDINGS_MODEL", "sentence-transformers/all-MiniLM-L6-v2")
|
||||
EMBEDDINGS_DIM = int(os.getenv("EMBEDDINGS_DIM", "384"))
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Postgres / pgvector - the LOCAL electronics database only
|
||||
# ---------------------------------------------------------------------------
|
||||
ELECTRONICS_DB_NAME = "electronics_catalog"
|
||||
# The test suite uses its own database on the same local server.
|
||||
ALLOWED_DB_NAMES = frozenset({ELECTRONICS_DB_NAME, ELECTRONICS_DB_NAME + "_test"})
|
||||
LOCAL_DB_HOSTS = frozenset({"localhost", "127.0.0.1", "::1", "host.docker.internal", "postgres"})
|
||||
|
||||
DB_HOST = os.getenv("DB_HOST", "127.0.0.1").strip()
|
||||
DB_PORT = os.getenv("DB_PORT", "5433").strip()
|
||||
DB_NAME = os.getenv("DB_NAME", ELECTRONICS_DB_NAME).strip()
|
||||
DB_USER = os.getenv("DB_USER", "postgres").strip()
|
||||
DB_PASSWORD = _require("DB_PASSWORD", feature_flag="the electronics database")
|
||||
DB_CONNECT_TIMEOUT_SECONDS = int(os.getenv("DB_CONNECT_TIMEOUT_SECONDS", "5"))
|
||||
|
||||
|
||||
def _csv_set(name: str) -> frozenset:
|
||||
return frozenset(v.strip() for v in os.getenv(name, "").split(",") if v.strip())
|
||||
|
||||
|
||||
# Production opt-in. Off by default: without ELEC_ALLOW_REMOTE_DB=true the
|
||||
# guard below behaves exactly as it always has. With it on, only the host(s)
|
||||
# and database name(s) listed here are accepted - never "any remote host".
|
||||
ELEC_ALLOW_REMOTE_DB = _bool("ELEC_ALLOW_REMOTE_DB", "false")
|
||||
ELEC_REMOTE_DB_HOSTS = _csv_set("ELEC_REMOTE_DB_HOSTS")
|
||||
ELEC_REMOTE_DB_NAMES = _csv_set("ELEC_REMOTE_DB_NAMES")
|
||||
|
||||
|
||||
def _guard_local_database(host: str, name: str, *, allow_remote: bool = False,
|
||||
remote_hosts: frozenset = frozenset(), remote_names: frozenset = frozenset()) -> None:
|
||||
"""Refuse to run against anything but the local electronics database,
|
||||
unless the production opt-in names this exact host and database."""
|
||||
if allow_remote and host in remote_hosts:
|
||||
if name not in remote_names:
|
||||
raise RuntimeError(
|
||||
f"DB_NAME={name!r} is not in ELEC_REMOTE_DB_NAMES for remote host {host!r}."
|
||||
)
|
||||
return
|
||||
if host not in LOCAL_DB_HOSTS:
|
||||
raise RuntimeError(
|
||||
f"DB_HOST={host!r} is not a local host. This project only runs against the "
|
||||
f"local Docker database (see docker-compose.yml); it must never touch the "
|
||||
f"remote catalogue database."
|
||||
)
|
||||
if name not in ALLOWED_DB_NAMES:
|
||||
raise RuntimeError(
|
||||
f"DB_NAME={name!r}; expected {ELECTRONICS_DB_NAME!r}. The electronics data "
|
||||
f"lives in its own database so existing databases are never modified."
|
||||
)
|
||||
|
||||
|
||||
_guard_local_database(DB_HOST, DB_NAME, allow_remote=ELEC_ALLOW_REMOTE_DB,
|
||||
remote_hosts=ELEC_REMOTE_DB_HOSTS, remote_names=ELEC_REMOTE_DB_NAMES)
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Web search (discovery, prices, images)
|
||||
# ---------------------------------------------------------------------------
|
||||
# DuckDuckGo needs no key. Google Programmable Search is used in addition when
|
||||
# both GOOGLE_API_KEY and GOOGLE_CSE_ID are set (100 free queries/day).
|
||||
USE_DDG_SEARCH = _bool("USE_DDG_SEARCH", "true")
|
||||
GOOGLE_API_KEY = os.getenv("GOOGLE_API_KEY", "").strip()
|
||||
GOOGLE_CSE_ID = os.getenv("GOOGLE_CSE_ID", "").strip()
|
||||
USE_GOOGLE_CSE = bool(GOOGLE_API_KEY and GOOGLE_CSE_ID) and _bool("USE_GOOGLE_CSE", "true")
|
||||
GOOGLE_CSE_DAILY_QUOTA = int(os.getenv("GOOGLE_CSE_DAILY_QUOTA", "100"))
|
||||
SEARCH_REGION = os.getenv("SEARCH_REGION", "in-en")
|
||||
# Minimum pause between two search queries to the same provider.
|
||||
SEARCH_MIN_INTERVAL_SECONDS = float(os.getenv("SEARCH_MIN_INTERVAL_SECONDS", "2.5"))
|
||||
SEARCH_CACHE_TTL_HOURS = int(os.getenv("SEARCH_CACHE_TTL_HOURS", "24"))
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Polite fetching of retailer / brand pages
|
||||
# ---------------------------------------------------------------------------
|
||||
# An honest User-Agent with a contact address. Set ELEC_CONTACT to a real
|
||||
# address before running a crawl.
|
||||
ELEC_CONTACT = os.getenv("ELEC_CONTACT", "admin@example.com").strip()
|
||||
USER_AGENT = os.getenv(
|
||||
"USER_AGENT", f"ElectronicsCatalogBot/0.1 (+mailto:{ELEC_CONTACT}; local research)"
|
||||
)
|
||||
REQUEST_TIMEOUT_SECONDS = int(os.getenv("REQUEST_TIMEOUT_SECONDS", "20"))
|
||||
ELEC_SITE_MIN_INTERVAL_SECONDS = float(os.getenv("ELEC_SITE_MIN_INTERVAL_SECONDS", "3"))
|
||||
ELEC_BREAKER_COOLDOWN_HOURS = float(os.getenv("ELEC_BREAKER_COOLDOWN_HOURS", "24"))
|
||||
ELEC_MAX_PAGE_BYTES = int(os.getenv("ELEC_MAX_PAGE_BYTES", str(3 * 1024 * 1024)))
|
||||
ELEC_PROBE_TTL_DAYS = int(os.getenv("ELEC_PROBE_TTL_DAYS", "7"))
|
||||
MIN_IMAGE_BYTES = int(os.getenv("MIN_IMAGE_BYTES", "3000"))
|
||||
|
||||
# Reference pincodes (Tamil Nadu). "pincode:City" pairs, comma-separated. The
|
||||
# first one is the default. A pincode is stored against a price only when the
|
||||
# site actually accepted it.
|
||||
ELEC_REFERENCE_PINCODES = [
|
||||
tuple(p.split(":", 1)) if ":" in p else (p, "")
|
||||
for p in (x.strip() for x in os.getenv("ELEC_REFERENCE_PINCODES", "641001:Coimbatore,600001:Chennai").split(","))
|
||||
if p
|
||||
]
|
||||
|
||||
# The LLM may only fill spec gaps from text we fetched; set false to run fully
|
||||
# deterministic.
|
||||
ELEC_USE_LLM = _bool("ELEC_USE_LLM", "true")
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# FastAPI / web server
|
||||
# ---------------------------------------------------------------------------
|
||||
API_CORS_ORIGINS = [
|
||||
origin.strip()
|
||||
for origin in os.getenv("API_CORS_ORIGINS", "http://localhost:5173,http://127.0.0.1:5173").split(",")
|
||||
if origin.strip()
|
||||
]
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Authentication
|
||||
# ---------------------------------------------------------------------------
|
||||
# CORS above is not access control - browsers enforce it, and curl ignores it
|
||||
# entirely. These settings are what actually guards the write/compute endpoints
|
||||
# (catalog generation, ML training, uploads, chat).
|
||||
#
|
||||
# AUTH_ENABLED=false turns every guard off, restoring the old behaviour where
|
||||
# any caller could reach any endpoint. It exists so a fresh checkout still runs
|
||||
# without generating secrets first; app/infrastructure/security.py logs a
|
||||
# warning at import when it is off. Never deploy with it off.
|
||||
AUTH_ENABLED = _bool("AUTH_ENABLED", "true")
|
||||
|
||||
# Signs and verifies access tokens. Changing it invalidates every issued token,
|
||||
# which is the intended way to force everyone to sign in again. Generate with:
|
||||
# python scripts/make_auth_secrets.py
|
||||
AUTH_SECRET_KEY = (
|
||||
_require("AUTH_SECRET_KEY", feature_flag="AUTH_ENABLED")
|
||||
if AUTH_ENABLED
|
||||
else os.getenv("AUTH_SECRET_KEY", "")
|
||||
)
|
||||
|
||||
# How long an issued token stays valid. 12h by default: long enough that a
|
||||
# working day needs one sign-in, short enough that a leaked token expires.
|
||||
AUTH_TOKEN_TTL_MINUTES = int(os.getenv("AUTH_TOKEN_TTL_MINUTES", "720"))
|
||||
|
||||
# The interactive accounts. Only PBKDF2 digests are stored - never a password.
|
||||
# `make_auth_secrets.py` prints the lines ready to paste.
|
||||
#
|
||||
# `admin` is required whenever auth is on: without it nobody could sign in.
|
||||
AUTH_ADMIN_USERNAME = _clean(
|
||||
"AUTH_ADMIN_USERNAME", os.getenv("AUTH_ADMIN_USERNAME", "admin")
|
||||
)
|
||||
AUTH_ADMIN_PASSWORD_HASH = _clean(
|
||||
"AUTH_ADMIN_PASSWORD_HASH",
|
||||
(
|
||||
_require("AUTH_ADMIN_PASSWORD_HASH", feature_flag="AUTH_ENABLED")
|
||||
if AUTH_ENABLED
|
||||
else os.getenv("AUTH_ADMIN_PASSWORD_HASH", "")
|
||||
),
|
||||
)
|
||||
|
||||
# The second `user` account is OPTIONAL, and left unset in this deployment.
|
||||
# An empty hash is how the account is switched off: auth.py builds its account
|
||||
# table from these values and omits any entry whose hash is blank, so there is
|
||||
# nothing to sign in to. Setting the hash again re-enables it with no code
|
||||
# change - which is exactly what the test suite does in tests/conftest.py.
|
||||
AUTH_USER_USERNAME = _clean(
|
||||
"AUTH_USER_USERNAME", os.getenv("AUTH_USER_USERNAME", "user")
|
||||
)
|
||||
AUTH_USER_PASSWORD_HASH = _clean(
|
||||
"AUTH_USER_PASSWORD_HASH", os.getenv("AUTH_USER_PASSWORD_HASH", "")
|
||||
)
|
||||
|
||||
# Failed-login throttle, applied per username+client-IP. Prevents an exposed
|
||||
# login endpoint from being a free password oracle.
|
||||
AUTH_MAX_LOGIN_ATTEMPTS = int(os.getenv("AUTH_MAX_LOGIN_ATTEMPTS", "10"))
|
||||
AUTH_LOCKOUT_SECONDS = int(os.getenv("AUTH_LOCKOUT_SECONDS", "300"))
|
||||
|
||||
# Local-development escape hatch: accept ANY password at /api/auth/login, so a
|
||||
# developer who does not have the configured passwords to hand can still reach
|
||||
# the admin and user pages. The username still selects the role, and the token
|
||||
# issued is a normal signed one - so every downstream guard, /api/auth/me, and
|
||||
# the React route gating all behave exactly as they do in production. What is
|
||||
# skipped is only the password check.
|
||||
#
|
||||
# This is NOT the same as AUTH_ENABLED=false. That disables every guard *and*
|
||||
# makes /api/auth/login return 503, which breaks the login page outright. This
|
||||
# flag keeps the whole auth machinery running and unlocks just the front door.
|
||||
#
|
||||
# Anyone who can reach the API can sign in as admin while it is on. Keep it
|
||||
# false anywhere the port is reachable by someone you would not hand the admin
|
||||
# password to.
|
||||
AUTH_ALLOW_ANY_LOGIN = _bool("AUTH_ALLOW_ANY_LOGIN", "false")
|
||||
|
||||
|
||||
# Shortest acceptable API key secret. token_urlsafe(32) yields 43 characters, so
|
||||
# this rejects hand-typed values without rejecting anything the documented
|
||||
# generator produces.
|
||||
API_KEY_MIN_LENGTH = 32
|
||||
|
||||
|
||||
def _parse_api_keys(raw: str) -> dict:
|
||||
"""
|
||||
Parse ``API_KEYS`` - ``name:role:secret`` triples, comma-separated.
|
||||
|
||||
Keyed by secret because that is what an inbound request presents. One entry
|
||||
per consumer is the point: a shared key cannot be revoked for one caller
|
||||
without breaking all of them.
|
||||
|
||||
Secrets must be at least API_KEY_MIN_LENGTH characters. That is not about
|
||||
guessing the key over the network - the lockout and the network itself make
|
||||
online brute force impractical - but about what /api/health publishes. It
|
||||
reports a truncated digest of every configured key so a deployment can be
|
||||
checked against the config it was built from, and a digest of a *raw* secret
|
||||
is only safe when the secret is unguessable offline. An admin password hash
|
||||
embeds a random salt, so its fingerprint discloses nothing; an API key has no
|
||||
salt, and a hand-picked "changeme" would fall to a wordlist in seconds.
|
||||
Generate one with: python -c "import secrets; print(secrets.token_urlsafe(32))"
|
||||
"""
|
||||
parsed: dict = {}
|
||||
for entry in raw.split(","):
|
||||
entry = entry.strip()
|
||||
if not entry:
|
||||
continue
|
||||
parts = entry.split(":")
|
||||
if len(parts) != 3:
|
||||
raise RuntimeError(
|
||||
f"Malformed API_KEYS entry {entry!r}. Expected 'name:role:secret', "
|
||||
f"comma-separated between entries."
|
||||
)
|
||||
name, role, secret = (p.strip() for p in parts)
|
||||
# MUST stay in step with ROLE_PERMISSIONS in app/infrastructure/security.py,
|
||||
# which is the source of truth. It is duplicated rather than imported
|
||||
# because security.py imports THIS module, so importing it back here
|
||||
# would be a cycle. A role added there but not here is rejected at boot
|
||||
# with the message below - loud, and before any request is served.
|
||||
if role not in {"admin", "user", "uploader"}:
|
||||
raise RuntimeError(
|
||||
f"API_KEYS entry {name!r} has role {role!r}; expected 'admin', 'user' "
|
||||
f"or 'uploader'."
|
||||
)
|
||||
if not secret:
|
||||
raise RuntimeError(f"API_KEYS entry {name!r} has an empty secret.")
|
||||
if len(secret) < API_KEY_MIN_LENGTH:
|
||||
raise RuntimeError(
|
||||
f"API_KEYS entry {name!r} has a {len(secret)}-character secret; at least "
|
||||
f"{API_KEY_MIN_LENGTH} are required, because /api/health publishes a digest "
|
||||
f"of it. Generate one with: "
|
||||
f"python -c \"import secrets; print(secrets.token_urlsafe(32))\""
|
||||
)
|
||||
parsed[secret] = (name, role)
|
||||
return parsed
|
||||
|
||||
|
||||
# Machine consumers of api.<domain>. Empty by default - browser sessions go
|
||||
# through /api/auth/login instead, and a key that nobody needs is only risk.
|
||||
#
|
||||
# NAME THE KEY FOR ITS FUNCTION, NOT THE PERSON HOLDING IT.
|
||||
# /api/health is public and reports {name, role, fingerprint} for every
|
||||
# configured key (describe_api_keys in security.py). The secret is never
|
||||
# exposed, but the NAME is - so `catalog-drop:uploader:...` is right and
|
||||
# `priya-laptop:uploader:...` publishes a colleague's name to anyone who
|
||||
# curls the health endpoint.
|
||||
API_KEYS = _parse_api_keys(os.getenv("API_KEYS", ""))
|
||||
|
||||
133
backend/app/main.py
Normal file
133
backend/app/main.py
Normal file
@@ -0,0 +1,133 @@
|
||||
"""
|
||||
FastAPI application entry point for the Electronics Catalog (local only).
|
||||
|
||||
Run with (from backend/):
|
||||
.venv\\Scripts\\python -m uvicorn app.main:app --host 127.0.0.1 --port 8000
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import threading
|
||||
from contextlib import asynccontextmanager
|
||||
from pathlib import Path
|
||||
|
||||
from fastapi import FastAPI, HTTPException, Request
|
||||
from fastapi.encoders import jsonable_encoder
|
||||
from fastapi.exceptions import RequestValidationError
|
||||
from fastapi.middleware.cors import CORSMiddleware
|
||||
from fastapi.responses import FileResponse, JSONResponse
|
||||
from fastapi.staticfiles import StaticFiles
|
||||
|
||||
from app.api.routers import auth, elec, elec_admin, health
|
||||
from app.infrastructure.security import auth_config_summary
|
||||
from app.infrastructure.settings import API_CORS_ORIGINS, DB_HOST, DB_NAME, DB_PORT, cleaned_env_names
|
||||
from app.mcp_server import mcp
|
||||
from fastmcp.utilities.lifespan import combine_lifespans
|
||||
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
|
||||
)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def _init_schema() -> None:
|
||||
"""Apply pending migrations and make sure reference data exists. Runs on a
|
||||
thread so the API answers /api/health even while Postgres is starting."""
|
||||
try:
|
||||
from app.electronics.db import repository as repo
|
||||
from app.electronics.db.migrate import run_migrations
|
||||
from app.electronics.reference import load_reference
|
||||
|
||||
applied = run_migrations()
|
||||
if applied:
|
||||
logger.info("Applied migrations: %s", ", ".join(applied))
|
||||
repo.seed_reference(load_reference())
|
||||
except Exception as exc: # noqa: BLE001 - the health endpoint reports the DB state
|
||||
logger.warning("Schema initialisation skipped: %s", exc)
|
||||
|
||||
|
||||
@asynccontextmanager
|
||||
async def lifespan(_app: FastAPI):
|
||||
logger.info("Electronics Catalog using database %s at %s:%s", DB_NAME, DB_HOST, DB_PORT)
|
||||
threading.Thread(target=_init_schema, daemon=True).start()
|
||||
yield
|
||||
|
||||
|
||||
# MCP endpoint (FastMCP) for AI assistants, served at /mcp/ by this same app.
|
||||
# Its session manager must start with the app, hence the combined lifespan.
|
||||
mcp_app = mcp.http_app(path="/")
|
||||
|
||||
app = FastAPI(
|
||||
title="Electronics Catalog API",
|
||||
description=(
|
||||
"Local, evidence-backed catalogue of electronics sold in India (Tamil Nadu focus). "
|
||||
"Products, prices, availability and images are collected from real retail listings "
|
||||
"found through web search; nothing is generated."
|
||||
),
|
||||
version="1.0.0",
|
||||
lifespan=combine_lifespans(lifespan, mcp_app.lifespan),
|
||||
)
|
||||
|
||||
_allow_credentials = "*" not in API_CORS_ORIGINS
|
||||
app.add_middleware(
|
||||
CORSMiddleware,
|
||||
allow_origins=API_CORS_ORIGINS,
|
||||
allow_credentials=_allow_credentials,
|
||||
allow_methods=["*"],
|
||||
allow_headers=["*"],
|
||||
)
|
||||
|
||||
|
||||
def _json_safe(value):
|
||||
if isinstance(value, float) and (value != value or value in (float("inf"), float("-inf"))):
|
||||
return str(value)
|
||||
if isinstance(value, dict):
|
||||
return {k: _json_safe(v) for k, v in value.items()}
|
||||
if isinstance(value, (list, tuple)):
|
||||
return [_json_safe(v) for v in value]
|
||||
return value
|
||||
|
||||
|
||||
@app.exception_handler(RequestValidationError)
|
||||
async def _validation_error_as_422(request: Request, exc: RequestValidationError) -> JSONResponse:
|
||||
return JSONResponse(status_code=422, content={"detail": _json_safe(jsonable_encoder(exc.errors()))})
|
||||
|
||||
|
||||
_auth_cfg = auth_config_summary()
|
||||
logger.info(
|
||||
"Auth config: enabled=%s allow_any_login=%s admin_username=%r hash_valid=%s",
|
||||
_auth_cfg["enabled"], _auth_cfg["allow_any_login"], _auth_cfg["admin_username"],
|
||||
_auth_cfg["password_hash_valid"],
|
||||
)
|
||||
if _auth_cfg["allow_any_login"]:
|
||||
logger.warning("AUTH_ALLOW_ANY_LOGIN=true: any password is accepted. Keep the API on 127.0.0.1.")
|
||||
if cleaned_env_names():
|
||||
logger.warning("Settings arrived quoted or padded and were cleaned: %s", ", ".join(cleaned_env_names()))
|
||||
|
||||
app.include_router(health.router, prefix="/api")
|
||||
app.include_router(auth.router, prefix="/api")
|
||||
app.include_router(elec.router, prefix="/api")
|
||||
app.include_router(elec_admin.router, prefix="/api")
|
||||
app.mount("/mcp", mcp_app)
|
||||
|
||||
_dist_override = os.getenv("FRONTEND_DIST_DIR", "").strip()
|
||||
FRONTEND_DIST = Path(_dist_override) if _dist_override else Path(__file__).resolve().parents[2] / "frontend" / "dist"
|
||||
|
||||
if (FRONTEND_DIST / "assets").exists():
|
||||
logger.info("Serving built frontend from %s", FRONTEND_DIST)
|
||||
app.mount("/assets", StaticFiles(directory=str(FRONTEND_DIST / "assets")), name="assets")
|
||||
|
||||
@app.get("/{full_path:path}")
|
||||
def serve_frontend(full_path: str):
|
||||
if full_path.startswith(("api", "mcp", "docs", "redoc", "openapi.json")):
|
||||
raise HTTPException(status_code=404, detail="Not found")
|
||||
file_path = FRONTEND_DIST / full_path
|
||||
if file_path.exists() and file_path.is_file():
|
||||
return FileResponse(file_path)
|
||||
return FileResponse(FRONTEND_DIST / "index.html")
|
||||
else:
|
||||
@app.get("/")
|
||||
def root() -> dict:
|
||||
return {"service": "Electronics Catalog API", "docs": "/docs", "health": "/api/health", "mcp": "/mcp/"}
|
||||
121
backend/app/mcp_server.py
Normal file
121
backend/app/mcp_server.py
Normal file
@@ -0,0 +1,121 @@
|
||||
"""MCP server (FastMCP) over the read-only catalogue.
|
||||
|
||||
Mounted by app/main.py at /mcp/, in the same process as the REST API. Each tool
|
||||
calls the same function the matching /api/elec endpoint uses, so the two can
|
||||
never disagree. Only read-only catalogue data is exposed: no admin, no login,
|
||||
no collection runs.
|
||||
|
||||
Images are returned as URLs (the retailer's own image address, as stored in
|
||||
elec.product_image); nothing is downloaded or re-hosted.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from decimal import Decimal
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
import anyio
|
||||
from fastapi import HTTPException
|
||||
from fastmcp import FastMCP
|
||||
from fastmcp.exceptions import ToolError
|
||||
|
||||
from app.api.routers import elec
|
||||
|
||||
mcp = FastMCP(
|
||||
name="Electronics Catalog",
|
||||
instructions=(
|
||||
"Verified catalogue of mobiles and laptops sold in India (Tamil Nadu focus). "
|
||||
"Every product is confirmed by real listings on at least two retail platforms; "
|
||||
"prices, ratings and reviews come with the page they were read from. Prices are "
|
||||
"rupee strings. Use search_products to find products, then get_product for "
|
||||
"per-platform offers, specs, images, rating and reviews."
|
||||
),
|
||||
)
|
||||
|
||||
_SEARCH_FIELDS = (
|
||||
"product_id", "brand", "category", "display_name", "ram_gb", "storage_gb",
|
||||
"best_price", "best_price_site", "platform_count", "sold_by_tn_retailer", "image_url",
|
||||
)
|
||||
|
||||
|
||||
async def _run(fn, *args):
|
||||
"""The catalogue functions use blocking DB calls; keep them off the event loop."""
|
||||
try:
|
||||
return await anyio.to_thread.run_sync(lambda: fn(*args))
|
||||
except HTTPException as exc:
|
||||
raise ToolError(str(exc.detail)) from exc
|
||||
|
||||
|
||||
@mcp.tool
|
||||
async def list_categories() -> List[Dict[str, Any]]:
|
||||
"""List product categories (e.g. mobiles, laptops) with how many verified products each has."""
|
||||
return await _run(elec.categories)
|
||||
|
||||
|
||||
@mcp.tool
|
||||
async def search_products(
|
||||
query: Optional[str] = None,
|
||||
category: Optional[str] = None,
|
||||
brand: Optional[str] = None,
|
||||
max_price: Optional[float] = None,
|
||||
min_price: Optional[float] = None,
|
||||
limit: int = 20,
|
||||
) -> Dict[str, Any]:
|
||||
"""Search verified products.
|
||||
|
||||
Args:
|
||||
query: Text to match in the product or brand name, e.g. "galaxy s25", "vivobook".
|
||||
category: Category slug: "mobiles" or "laptops".
|
||||
brand: Brand slug, e.g. "samsung", "xiaomi", "hp", "lenovo".
|
||||
max_price: Highest best price in rupees.
|
||||
min_price: Lowest best price in rupees.
|
||||
limit: Maximum products to return (1-100).
|
||||
|
||||
Returns the total match count and, per product: id, name, variant, best price (rupee
|
||||
string) and the platform offering it, number of platforms, and an image URL (or null).
|
||||
"""
|
||||
limit = max(1, min(int(limit), 100))
|
||||
result = await _run(
|
||||
elec.products,
|
||||
category, brand, (query or None),
|
||||
None if min_price is None else Decimal(str(min_price)),
|
||||
None if max_price is None else Decimal(str(max_price)),
|
||||
False, None, False, limit, 0,
|
||||
)
|
||||
return {
|
||||
"total": result["total"],
|
||||
"products": [{k: p.get(k) for k in _SEARCH_FIELDS} for p in result["products"]],
|
||||
}
|
||||
|
||||
|
||||
@mcp.tool
|
||||
async def get_product(product_id: int) -> Dict[str, Any]:
|
||||
"""Full details of one product by its product_id (from search_products).
|
||||
|
||||
Returns per-platform offers (price, MRP, source URL, when seen), normalised specs,
|
||||
image URLs, the overall rating with per-platform sources (null if none is published),
|
||||
and up to 10 real customer reviews (often empty).
|
||||
"""
|
||||
d = await _run(elec.product, int(product_id))
|
||||
return {
|
||||
"product_id": d["product_id"],
|
||||
"brand": d.get("brand"),
|
||||
"category": d.get("category"),
|
||||
"display_name": d.get("display_name"),
|
||||
"best_price": d.get("best_price"),
|
||||
"best_price_site": d.get("best_price_site"),
|
||||
"specs": d.get("canonical_specs") or {},
|
||||
"offers": [
|
||||
{k: o.get(k) for k in ("site", "price", "mrp", "source_url", "observed_at", "price_outlier")}
|
||||
for o in d.get("offers", [])
|
||||
],
|
||||
"image_urls": [i["url"] for i in d.get("images", [])],
|
||||
"rating": d.get("rating"),
|
||||
"reviews": d.get("reviews", []),
|
||||
}
|
||||
|
||||
|
||||
@mcp.tool
|
||||
async def price_history(product_id: int) -> List[Dict[str, Any]]:
|
||||
"""Every price observed for a product, per platform, oldest first (rupee strings, ISO times)."""
|
||||
rows = await _run(elec.price_history, int(product_id))
|
||||
return [{k: r.get(k) for k in ("site", "price", "mrp", "observed_at")} for r in rows]
|
||||
0
backend/app/services/__init__.py
Normal file
0
backend/app/services/__init__.py
Normal file
52
backend/app/services/embeddings_service.py
Normal file
52
backend/app/services/embeddings_service.py
Normal file
@@ -0,0 +1,52 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import List, Optional, TYPE_CHECKING
|
||||
|
||||
from app.infrastructure.settings import EMBEDDINGS_MODEL
|
||||
|
||||
if TYPE_CHECKING: # pragma: no cover - typing only, no runtime cost
|
||||
from sentence_transformers import SentenceTransformer
|
||||
|
||||
_model_singleton: Optional["SentenceTransformer"] = None
|
||||
|
||||
|
||||
def get_device() -> str:
|
||||
"""Prefer CUDA if available, otherwise CPU.
|
||||
|
||||
Imports torch lazily: on an 8GB RAM / CPU-only laptop there is no
|
||||
benefit to importing torch (and paying its startup/memory cost) until
|
||||
an embedding is actually requested, so the FastAPI process can boot
|
||||
and answer /api/health almost instantly.
|
||||
"""
|
||||
import torch # local import - see docstring
|
||||
|
||||
return "cuda" if torch.cuda.is_available() else "cpu"
|
||||
|
||||
|
||||
def get_embedding_model() -> "SentenceTransformer":
|
||||
global _model_singleton
|
||||
if _model_singleton is None:
|
||||
from sentence_transformers import SentenceTransformer # local import - see get_device()
|
||||
|
||||
device = get_device()
|
||||
_model_singleton = SentenceTransformer(EMBEDDINGS_MODEL, device=device)
|
||||
return _model_singleton
|
||||
|
||||
|
||||
def embed_texts(texts: List[str]) -> List[List[float]]:
|
||||
"""Embed a batch of texts into normalized 384-dim vectors (MiniLM-L6-v2).
|
||||
|
||||
Normalized so that pgvector's cosine-distance operator (`<=>`) behaves
|
||||
consistently for the RAG retrieval step in `app.services.vector_store`.
|
||||
"""
|
||||
if not texts:
|
||||
return []
|
||||
model = get_embedding_model()
|
||||
embeddings = model.encode(
|
||||
texts,
|
||||
batch_size=32,
|
||||
normalize_embeddings=True,
|
||||
convert_to_numpy=True,
|
||||
show_progress_bar=False,
|
||||
)
|
||||
return embeddings.tolist()
|
||||
171
backend/app/services/ollama_service.py
Normal file
171
backend/app/services/ollama_service.py
Normal file
@@ -0,0 +1,171 @@
|
||||
"""
|
||||
Thin client for the local Ollama server.
|
||||
|
||||
In this project the LLM is an EXTRACTOR, never a source: it is only ever
|
||||
handed text that was fetched from a real page or search result, and every
|
||||
value it returns is checked against that text by
|
||||
app.electronics.normalise.grounding before it is kept. There are deliberately
|
||||
no functions here that ask the model to list products, prices or images.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
import time
|
||||
|
||||
import requests
|
||||
|
||||
from app.infrastructure.settings import OLLAMA_BASE_URL, OLLAMA_MODEL_NAME, USE_OLLAMA, OLLAMA_TIMEOUT_SECONDS
|
||||
|
||||
|
||||
# Reachability is asked once per this many seconds, not once per caller.
|
||||
#
|
||||
# WHY THIS CACHE EXISTS. The probe below costs up to 5 seconds when nothing is
|
||||
# listening, and `_ensure_client` is called per ROW by stage 2 of the ingestion
|
||||
# pipeline (store_catalog_pipeline.stage_2_row_intake -> fetch_product_details).
|
||||
# Uncached, a 2000-row sheet ingested with use_llm on, against a configured but
|
||||
# unreachable Ollama, spends up to ~2.8 hours doing nothing but timing out - and
|
||||
# presents as a batch that has hung rather than one that has failed. That is not
|
||||
# hypothetical: USE_OLLAMA=true pointing at localhost:11434 is the default
|
||||
# developer configuration, and `ollama serve` is not always running beside it.
|
||||
#
|
||||
# /api/health calls this too (app/api/routers/system.py), so the same cache
|
||||
# stops a down Ollama adding 5s to every health request.
|
||||
#
|
||||
# A TTL rather than a permanent memo, deliberately: this is a liveness fact, not
|
||||
# configuration. Cached forever, an Ollama started after the API would never be
|
||||
# noticed and /api/health would report it down until a redeploy.
|
||||
_PROBE_TTL_SECONDS = 30.0
|
||||
_probe_cache: tuple[float, bool] | None = None
|
||||
|
||||
|
||||
def reset_reachability_cache() -> None:
|
||||
"""Forget the cached probe. For tests, and for anything that knows the
|
||||
answer just changed."""
|
||||
global _probe_cache
|
||||
_probe_cache = None
|
||||
|
||||
|
||||
def _ensure_client():
|
||||
"""None when Ollama is switched off, True/False for reachable or not.
|
||||
|
||||
Three return values, not two - `system.py` relies on telling "disabled" from
|
||||
"configured but down", so do not collapse this to a bool.
|
||||
"""
|
||||
if not USE_OLLAMA:
|
||||
# No network call on this path, so nothing worth caching.
|
||||
return None
|
||||
|
||||
global _probe_cache
|
||||
now = time.monotonic()
|
||||
if _probe_cache is not None and now - _probe_cache[0] < _PROBE_TTL_SECONDS:
|
||||
return _probe_cache[1]
|
||||
|
||||
# Verify Ollama is reachable
|
||||
try:
|
||||
resp = requests.get(f"{OLLAMA_BASE_URL}/api/tags", timeout=5)
|
||||
reachable = resp.status_code == 200
|
||||
except Exception:
|
||||
reachable = False
|
||||
|
||||
_probe_cache = (now, reachable)
|
||||
return reachable
|
||||
|
||||
|
||||
def _generate(
|
||||
system: str,
|
||||
user_prompt: str,
|
||||
max_retries: int = 2,
|
||||
*,
|
||||
temperature: float = 0.0,
|
||||
json_mode: bool = False,
|
||||
) -> str:
|
||||
"""Call Ollama's chat endpoint and return text safely.
|
||||
|
||||
Retries up to `max_retries` times when the response is empty, since
|
||||
small local models (e.g. qwen2.5:1.5b) sometimes return empty content
|
||||
for complex JSON prompts on the first attempt.
|
||||
"""
|
||||
if not _ensure_client():
|
||||
return ""
|
||||
for attempt in range(max_retries + 1):
|
||||
try:
|
||||
resp = requests.post(
|
||||
f"{OLLAMA_BASE_URL}/api/chat",
|
||||
json={
|
||||
"model": OLLAMA_MODEL_NAME,
|
||||
"messages": [
|
||||
{"role": "system", "content": system},
|
||||
{"role": "user", "content": user_prompt},
|
||||
],
|
||||
"stream": False,
|
||||
"options": {"temperature": temperature},
|
||||
**({"format": "json"} if json_mode else {}),
|
||||
},
|
||||
timeout=OLLAMA_TIMEOUT_SECONDS,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
content = (data.get("message", {}).get("content", "") or "").strip()
|
||||
if content:
|
||||
return content
|
||||
if attempt < max_retries:
|
||||
import time
|
||||
time.sleep(1.0)
|
||||
except Exception:
|
||||
if attempt >= max_retries:
|
||||
return ""
|
||||
import time
|
||||
time.sleep(1.0)
|
||||
return ""
|
||||
|
||||
|
||||
def _extract_json(text: str) -> dict | None:
|
||||
"""Extract JSON from model response, trying multiple strategies.
|
||||
|
||||
Handles both JSON objects {...} and JSON arrays [...] since small
|
||||
local models frequently return bare arrays instead of an object with
|
||||
a ``products`` key.
|
||||
"""
|
||||
if not text:
|
||||
return None
|
||||
# Try fenced code block (object or array)
|
||||
match = re.search(r"```(?:json)?\s*(\{[\s\S]*?\}|\[[\s\S]*?\])\s*```", text)
|
||||
if match:
|
||||
try:
|
||||
return json.loads(match.group(1))
|
||||
except json.JSONDecodeError:
|
||||
pass
|
||||
# Try first JSON value in text (greedy - object)
|
||||
brace = re.search(r"\{[\s\S]*\}", text)
|
||||
if brace:
|
||||
try:
|
||||
return json.loads(brace.group(0))
|
||||
except json.JSONDecodeError:
|
||||
pass
|
||||
# Try first JSON value in text (greedy - array)
|
||||
bracket = re.search(r"\[[\s\S]*\]", text)
|
||||
if bracket:
|
||||
try:
|
||||
return json.loads(bracket.group(0))
|
||||
except json.JSONDecodeError:
|
||||
pass
|
||||
# Try parsing entire text
|
||||
try:
|
||||
return json.loads(text.strip())
|
||||
except json.JSONDecodeError:
|
||||
pass
|
||||
# Fallback: try to fix common issues
|
||||
cleaned = text.strip()
|
||||
cleaned = re.sub(r"(?<=[:,\[])\s*'", '"', cleaned)
|
||||
cleaned = re.sub(r"'\s*(?=[,:\}\]])", '"', cleaned)
|
||||
try:
|
||||
return json.loads(cleaned)
|
||||
except json.JSONDecodeError:
|
||||
return None
|
||||
|
||||
|
||||
def generate_json(system_prompt: str, user_prompt: str) -> dict | list | None:
|
||||
"""Deterministic JSON-mode completion. None when Ollama is unavailable or
|
||||
the reply is not JSON - callers must treat that as "no extra data"."""
|
||||
return _extract_json(_generate(system_prompt, user_prompt, max_retries=1, json_mode=True))
|
||||
Reference in New Issue
Block a user