Electronics Catalog: API, MCP server, frontend and deployment
Verified catalogue of mobiles and laptops sold in India, collected from real retail listings (FastAPI backend, React frontend, Postgres/pgvector). - REST API under /api/elec (read-only catalogue; admin endpoints need login) - MCP server (FastMCP) at /mcp/ with list_categories, search_products, get_product and price_history tools - Real ratings and reviews read from product pages and search results - Production Dockerfile (requirements-api.txt, no PyTorch) and .env.production.example; remote database only via an explicit ELEC_ALLOW_REMOTE_DB host/name allowlist - docs/API.md: endpoint and MCP reference with live examples Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
0
backend/app/electronics/normalise/__init__.py
Normal file
0
backend/app/electronics/normalise/__init__.py
Normal file
86
backend/app/electronics/normalise/brand_alias.py
Normal file
86
backend/app/electronics/normalise/brand_alias.py
Normal file
@@ -0,0 +1,86 @@
|
||||
"""Resolve the brand of a product title against the closed allow-list."""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from functools import lru_cache
|
||||
from typing import List, Optional, Tuple
|
||||
|
||||
from app.electronics.reference import load_reference
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class BrandMatch:
|
||||
brand_slug: str
|
||||
brand_name: str
|
||||
family: Optional[str] # sub-brand (Redmi, iQOO, Pixel...) when the title used one
|
||||
matched: str # the alias text found in the title
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def _alias_table() -> List[Tuple[str, str, bool]]:
|
||||
"""(alias, brand_slug, is_sub_brand), longest alias first."""
|
||||
ref = load_reference()
|
||||
rows: List[Tuple[str, str, bool]] = []
|
||||
for b in ref.brands.values():
|
||||
for a in b.aliases:
|
||||
rows.append((a, b.slug, False))
|
||||
for s in b.sub_brands:
|
||||
rows.append((s, b.slug, True))
|
||||
rows.sort(key=lambda r: -len(r[0]))
|
||||
return rows
|
||||
|
||||
|
||||
def resolve_brand(title: str, *, expected: Optional[str] = None) -> Optional[BrandMatch]:
|
||||
"""The allow-listed brand a title starts with (or names within its first
|
||||
few words), or None. `expected` restricts the match to one brand slug.
|
||||
|
||||
Only the start of the title is considered: "Case for Samsung Galaxy S24"
|
||||
is an accessory, not a Samsung phone.
|
||||
"""
|
||||
if not title:
|
||||
return None
|
||||
ref = load_reference()
|
||||
head = " ".join(re.findall(r"[a-z0-9+]+", title.lower())[:3])
|
||||
for alias, slug, is_sub in _alias_table():
|
||||
if expected and slug != expected:
|
||||
continue
|
||||
pattern = r"(?:^|\s)" + re.escape(alias) + r"(?:\s|$)"
|
||||
m = re.search(pattern, head)
|
||||
if not m:
|
||||
continue
|
||||
# The brand/sub-brand must be the first or second word ("Apple iPhone",
|
||||
# "Samsung Galaxy", "Xiaomi Redmi Note") - not buried later.
|
||||
if len(head[: m.start()].split()) > 1:
|
||||
continue
|
||||
# "Google Pixel 8", "Xiaomi Redmi Note 13": the parent brand matched,
|
||||
# but the family is the sub-brand that follows it.
|
||||
sub = alias if is_sub else next(
|
||||
(s for s in ref.brands[slug].sub_brands if re.search(r"(?:^|\s)" + re.escape(s) + r"(?:\s|$)", head)),
|
||||
None,
|
||||
)
|
||||
return BrandMatch(slug, ref.brands[slug].name, _family_casing(sub) if sub else None, alias)
|
||||
return None
|
||||
|
||||
|
||||
_CASING = {"iphone": "iPhone", "iqoo": "iQOO", "macbook": "MacBook", "rog": "ROG", "tuf": "TUF",
|
||||
"cmf": "CMF", "loq": "LOQ", "poco": "POCO", "mi": "Mi", "xps": "XPS", "thinkpad": "ThinkPad",
|
||||
"ideapad": "IdeaPad", "thinkbook": "ThinkBook", "vivobook": "Vivobook", "zenbook": "Zenbook"}
|
||||
|
||||
|
||||
def _family_casing(sub: str) -> str:
|
||||
return _CASING.get(sub, sub.title())
|
||||
|
||||
|
||||
# Words that mark an accessory or a non-product page, not a device.
|
||||
_NOT_A_DEVICE = re.compile(
|
||||
r"\b(?:case|cover|back\s+cover|tempered|screen\s+guard|protector|charger|adapter|cable|"
|
||||
r"skin|sleeve|bag|backpack|stand|holder|refurbished|renewed|pre-?owned|used|"
|
||||
r"compare|vs\.?|versus|review|specifications?\s+and|price\s+list|best\s+\w+\s+under|"
|
||||
r"top\s+\d+|all\s+models)\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
def looks_like_device_title(title: str) -> bool:
|
||||
return bool(title) and not _NOT_A_DEVICE.search(title)
|
||||
55
backend/app/electronics/normalise/grounding.py
Normal file
55
backend/app/electronics/normalise/grounding.py
Normal file
@@ -0,0 +1,55 @@
|
||||
"""Is a value actually stated in the text it supposedly came from?
|
||||
|
||||
Every value the LLM returns passes through value_in_source() against the exact
|
||||
text the model was shown. Anything that cannot be found there is discarded,
|
||||
which is what stops a small model's guess becoming a stored fact.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from decimal import Decimal, InvalidOperation
|
||||
from typing import Union
|
||||
|
||||
_WS = re.compile(r"\s+")
|
||||
|
||||
|
||||
def _norm_text(text: str) -> str:
|
||||
text = text.lower().replace(" ", " ")
|
||||
text = re.sub(r"[^\w.+ ]+", " ", text)
|
||||
text = re.sub(r"(?<!\d)\.|\.(?!\d)", " ", text) # sentence full stops, not decimals
|
||||
# "5000mAh" and "5000 mAh" must compare equal.
|
||||
text = re.sub(r"(?<=\d)(?=[a-z])|(?<=[a-z])(?=\d)", " ", text)
|
||||
return _WS.sub(" ", text).strip()
|
||||
|
||||
|
||||
def _numbers_in(text: str) -> set:
|
||||
found = set()
|
||||
for raw in re.findall(r"\d[\d,]*(?:\.\d+)?", text):
|
||||
try:
|
||||
found.add(Decimal(raw.replace(",", "")).normalize())
|
||||
except InvalidOperation:
|
||||
continue
|
||||
return found
|
||||
|
||||
|
||||
def value_in_source(value: Union[str, int, float, Decimal, None], source: str) -> bool:
|
||||
if value is None or not source:
|
||||
return False
|
||||
if isinstance(value, bool):
|
||||
return False
|
||||
if isinstance(value, (int, float, Decimal)):
|
||||
try:
|
||||
return Decimal(str(value)).normalize() in _numbers_in(source)
|
||||
except InvalidOperation:
|
||||
return False
|
||||
text = str(value).strip()
|
||||
if not text:
|
||||
return False
|
||||
# A string with a number in it ("5000 mAh", "Snapdragon 8 Gen 3") must have
|
||||
# every one of its numbers in the source, and its words too.
|
||||
nums = _numbers_in(text)
|
||||
if nums and not nums <= _numbers_in(source):
|
||||
return False
|
||||
words = [w for w in _norm_text(text).split() if not re.fullmatch(r"[\d.,]+", w)]
|
||||
hay = f" {_norm_text(source)} "
|
||||
return all(f" {w} " in hay for w in words) if words else bool(nums)
|
||||
71
backend/app/electronics/normalise/llm_fill.py
Normal file
71
backend/app/electronics/normalise/llm_fill.py
Normal file
@@ -0,0 +1,71 @@
|
||||
"""Fill MISSING spec keys from page text with the local LLM - and keep only
|
||||
what the text actually says.
|
||||
|
||||
The model sees one block of text that we fetched (a spec section or a
|
||||
description) and is asked to copy values out of it. Every value it returns is:
|
||||
1. checked by grounding.value_in_source() against that same text, and
|
||||
2. normalised by spec_normaliser (units, plausible ranges).
|
||||
Anything failing either step is dropped. Prices, product names and images are
|
||||
never asked of the model.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
from typing import Any, Dict, Iterable, Tuple
|
||||
|
||||
from app.electronics.normalise.grounding import value_in_source
|
||||
from app.electronics.normalise.spec_normaliser import normalise_value
|
||||
from app.infrastructure.settings import ELEC_USE_LLM
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
MAX_SOURCE_CHARS = 3500
|
||||
|
||||
SYSTEM_PROMPT = (
|
||||
"You copy product specifications out of the text you are given. "
|
||||
"Rules: use ONLY the given text; copy each value exactly as written, including its unit; "
|
||||
"if the text does not state a value, use null; never guess, estimate or use outside knowledge. "
|
||||
"Reply with one JSON object whose keys are exactly the requested keys."
|
||||
)
|
||||
|
||||
|
||||
def fill_missing(
|
||||
category: str,
|
||||
source_text: str,
|
||||
missing_keys: Iterable[str],
|
||||
*,
|
||||
generate=None,
|
||||
) -> Tuple[Dict[str, Any], Dict[str, str]]:
|
||||
"""(specs, sources) for whichever of `missing_keys` the text states."""
|
||||
keys = [k for k in missing_keys if k != "colour"]
|
||||
text = (source_text or "").strip()[:MAX_SOURCE_CHARS]
|
||||
if not keys or not text or not ELEC_USE_LLM:
|
||||
return {}, {}
|
||||
if generate is None:
|
||||
from app.services.ollama_service import generate_json as generate
|
||||
|
||||
prompt = (
|
||||
f"Requested keys: {json.dumps(keys)}\n\n"
|
||||
f"Text:\n\"\"\"\n{text}\n\"\"\"\n\n"
|
||||
"JSON:"
|
||||
)
|
||||
reply = generate(SYSTEM_PROMPT, prompt)
|
||||
if not isinstance(reply, dict):
|
||||
return {}, {}
|
||||
|
||||
specs: Dict[str, Any] = {}
|
||||
sources: Dict[str, str] = {}
|
||||
for key in keys:
|
||||
raw = reply.get(key)
|
||||
if raw is None or isinstance(raw, (dict, list, bool)):
|
||||
continue
|
||||
if not value_in_source(raw, text):
|
||||
logger.debug("LLM value %r for %s not found in source text; dropped", raw, key)
|
||||
continue
|
||||
value = normalise_value(category, key, raw)
|
||||
if value is None:
|
||||
continue
|
||||
specs[key] = value
|
||||
sources[key] = f"llm-extracted: {str(raw)[:80]}"
|
||||
return specs, sources
|
||||
101
backend/app/electronics/normalise/spec_normaliser.py
Normal file
101
backend/app/electronics/normalise/spec_normaliser.py
Normal file
@@ -0,0 +1,101 @@
|
||||
"""Map raw spec labels/values from a page to canonical keys and units.
|
||||
|
||||
Deterministic and table-driven (reference/spec_keys.yaml). A value that cannot
|
||||
be parsed, or lands outside the plausible range for its key, is dropped - never
|
||||
estimated.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from decimal import Decimal, InvalidOperation
|
||||
from functools import lru_cache
|
||||
from typing import Any, Dict, Optional, Tuple
|
||||
|
||||
from app.electronics.reference import load_reference
|
||||
|
||||
|
||||
def _label(text: str) -> str:
|
||||
return re.sub(r"[^a-z0-9]+", " ", str(text).lower()).strip()
|
||||
|
||||
|
||||
@lru_cache(maxsize=None)
|
||||
def _synonyms(category: str) -> Dict[str, str]:
|
||||
table: Dict[str, str] = {}
|
||||
for key, spec in load_reference().spec_keys.get(category, {}).items():
|
||||
for syn in [key.replace("_", " "), *spec.get("synonyms", [])]:
|
||||
table.setdefault(_label(syn), key)
|
||||
return table
|
||||
|
||||
|
||||
def canonical_key(category: str, label: str) -> Optional[str]:
|
||||
return _synonyms(category).get(_label(label))
|
||||
|
||||
|
||||
# unit -> (regex for the unit in text, factor into the canonical unit)
|
||||
_UNIT_PATTERNS = {
|
||||
"GB": [(r"tb", Decimal(1024)), (r"gb", Decimal(1)), (r"mb", Decimal(1) / 1024)],
|
||||
"inch": [(r"(?:inch(?:es)?|in\b|\"|”)", Decimal(1)), (r"cm", Decimal(1) / Decimal("2.54"))],
|
||||
"Hz": [(r"hz", Decimal(1))],
|
||||
"MP": [(r"mp|megapixel", Decimal(1))],
|
||||
"mAh": [(r"mah", Decimal(1))],
|
||||
"kg": [(r"kg|kilogram", Decimal(1)), (r"(?<![k])g\b|grams?", Decimal("0.001"))],
|
||||
"Wh": [(r"wh|watt\s*hours?", Decimal(1))],
|
||||
}
|
||||
|
||||
|
||||
def _to_number(value: str, unit: str) -> Optional[Decimal]:
|
||||
text = str(value).lower().replace(",", "")
|
||||
patterns = _UNIT_PATTERNS.get(unit, [])
|
||||
# Prefer an amount written in the canonical unit ("39.62 cm (15.6 inch)" -> 15.6).
|
||||
for unit_re, factor in patterns:
|
||||
m = re.search(r"(\d+(?:\.\d+)?)\s*(?:" + unit_re + r")", text)
|
||||
if m:
|
||||
try:
|
||||
return (Decimal(m.group(1)) * factor).quantize(Decimal("0.01")).normalize()
|
||||
except InvalidOperation:
|
||||
return None
|
||||
# A bare number is accepted only when nothing else is in the value.
|
||||
m = re.fullmatch(r"\s*(\d+(?:\.\d+)?)\s*", text)
|
||||
if m:
|
||||
return Decimal(m.group(1)).normalize()
|
||||
return None
|
||||
|
||||
|
||||
def normalise_value(category: str, key: str, value: Any) -> Optional[Any]:
|
||||
spec = load_reference().spec_keys.get(category, {}).get(key)
|
||||
if spec is None or value is None:
|
||||
return None
|
||||
text = str(value).strip()
|
||||
if not text or text.lower() in {"na", "n/a", "-", "none", "not applicable", "no"}:
|
||||
return None
|
||||
kind = spec.get("type")
|
||||
if kind == "number":
|
||||
num = _to_number(text, spec.get("unit", ""))
|
||||
if num is None:
|
||||
return None
|
||||
lo, hi = spec.get("range", [None, None])
|
||||
if (lo is not None and num < Decimal(str(lo))) or (hi is not None and num > Decimal(str(hi))):
|
||||
return None
|
||||
return float(num) if num != num.to_integral() else int(num)
|
||||
if kind == "enum":
|
||||
low = text.lower()
|
||||
for canon, words in spec.get("values", {}).items():
|
||||
if any(re.search(r"\b" + re.escape(w) + r"\b", low) for w in words):
|
||||
return canon
|
||||
return None
|
||||
return re.sub(r"\s+", " ", text)[:120]
|
||||
|
||||
|
||||
def normalise_specs(category: str, raw: Dict[str, Any]) -> Tuple[Dict[str, Any], Dict[str, str]]:
|
||||
"""(specs, sources): canonical key -> value, and key -> the raw label it came from."""
|
||||
specs: Dict[str, Any] = {}
|
||||
sources: Dict[str, str] = {}
|
||||
for label, value in (raw or {}).items():
|
||||
key = canonical_key(category, label)
|
||||
if not key or key in specs:
|
||||
continue
|
||||
norm = normalise_value(category, key, value)
|
||||
if norm is not None:
|
||||
specs[key] = norm
|
||||
sources[key] = f"{label}: {value}"[:200]
|
||||
return specs, sources
|
||||
374
backend/app/electronics/normalise/title_parser.py
Normal file
374
backend/app/electronics/normalise/title_parser.py
Normal file
@@ -0,0 +1,374 @@
|
||||
"""Split a retail product title into model, variant and a matching key.
|
||||
|
||||
Everything returned is read from the title text; a value the title does not
|
||||
state is None. Titles differ a lot between sites:
|
||||
|
||||
Samsung Galaxy S24 5G (Onyx Black, 8GB RAM, 256GB Storage) Amazon
|
||||
SAMSUNG Galaxy S24 5G (Onyx Black, 256 GB) (8 GB RAM) Flipkart
|
||||
Samsung Galaxy S24 5G (8GB RAM, 256GB, Onyx Black) Croma
|
||||
Redmi Note 13 Pro 5G (8GB + 256GB)
|
||||
Apple iPhone 15 (128 GB) - Black
|
||||
HP 15s, 13th Gen Intel Core i5-1334U, 16GB DDR4, 512GB SSD, ... fd0112TU
|
||||
|
||||
so the model is taken from the text before the first bracket/comma, and RAM /
|
||||
storage / colour from anywhere in the title.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from decimal import Decimal
|
||||
from typing import List, Optional
|
||||
|
||||
from app.electronics.normalise.brand_alias import BrandMatch, resolve_brand
|
||||
|
||||
_NUM = r"(\d+(?:\.\d+)?)"
|
||||
|
||||
# "8GB RAM", "8 GB LPDDR5X RAM", "RAM 8GB", "16GB DDR4" (laptops)
|
||||
_RAM_RES = [
|
||||
re.compile(_NUM + r"\s*GB\s*(?:LP)?(?:DDR\s?\d\w?\s*)?RAM\b", re.IGNORECASE),
|
||||
re.compile(r"\bRAM\s*[:\-]?\s*" + _NUM + r"\s*GB", re.IGNORECASE),
|
||||
re.compile(_NUM + r"\s*GB\s*(?:LP)?DDR\s?\d", re.IGNORECASE),
|
||||
re.compile(_NUM + r"\s*GB\s*(?:unified\s+memory|memory)\b", re.IGNORECASE),
|
||||
]
|
||||
# "8GB + 256GB", "8/256", "8GB/256GB", "12+512GB"
|
||||
_PAIR_RE = re.compile(r"(?<![\d.])(\d{1,2})\s*(?:GB)?\s*[+/]\s*(\d{2,4}|1|2)\s*(GB|TB)?\b", re.IGNORECASE)
|
||||
# explicit storage: "256GB Storage", "512GB SSD", "1TB", "256 GB ROM"
|
||||
_STORAGE_LABELLED = re.compile(
|
||||
_NUM + r"\s*(GB|TB)\s*(?:SSD|ROM|storage|internal(?:\s+storage)?|HDD|eMMC|UFS|NVMe|PCIe)\b", re.IGNORECASE
|
||||
)
|
||||
_SIZE_ANY = re.compile(r"(?<![\d.])" + _NUM + r"\s*(GB|TB)\b", re.IGNORECASE)
|
||||
|
||||
# Order matters: the first pattern that matches wins. AMD comes before the
|
||||
# Intel "Core N" pattern, because retail titles write core counts as words
|
||||
# ("Ryzen 3 Quad Core 7320U"), and "Core 7320U" must not read as Intel.
|
||||
_CORE_COUNT = r"(?:(?:Dual|Quad|Hexa|Octa|Six|Eight)\s+Core\s+)?"
|
||||
_PROCESSOR_RES = [
|
||||
re.compile(r"\b(?:AMD\s+)?Ryzen\s+R?(\d)\s*(?:Pro\s+)?" + _CORE_COUNT + r"[- ]?(\d{4}[A-Z]{0,3})\b", re.IGNORECASE),
|
||||
re.compile(r"\b(?:AMD\s+)?(Athlon)\s+(?:Silver\s+|Gold\s+)?" + _CORE_COUNT + r"(\d{4}[A-Z]{0,2})\b", re.IGNORECASE),
|
||||
re.compile(r"\b(?:Intel\s+)?Core\s+Ultra\s+(\d)\s*[- ]?\s*(\d{3}[A-Z]{0,2})\b", re.IGNORECASE),
|
||||
re.compile(r"\b(?:Intel\s+)?Core\s+(?:\(?\d+(?:th|nd|rd|st)\s+Gen\)?\s+)?(i[3579])\s*[- ]?\s*(\d{4,5}[A-Z]{0,3})\b", re.IGNORECASE),
|
||||
re.compile(r"\b(?:Intel\s+)?Core\s+(i[3579])\s+\d+(?:th|nd|rd|st)\s+Gen\s+(\d{4,5}[A-Z]{0,3})\b", re.IGNORECASE),
|
||||
re.compile(r"(?<!Dual )(?<!Quad )(?<!Hexa )(?<!Octa )(?<!Six )(?<!Eight )\b(?:Intel\s+)?Core\s+([3579])\s*[- ]?\s*(\d{3}[A-Z]{0,2})\b", re.IGNORECASE),
|
||||
re.compile(r"\bApple\s+(M[1-9])(?:\s+(Pro|Max|Ultra))?\b", re.IGNORECASE),
|
||||
re.compile(r"\b(M[1-9])\s*(Pro|Max|Ultra)?\s+chip\b", re.IGNORECASE),
|
||||
re.compile(r"\bSnapdragon\s+(X\d?)\s*(Elite|Plus)?\s*(X\d{1,2}-\d{3})?", re.IGNORECASE),
|
||||
re.compile(r"\b(?:Intel\s+)?(Celeron|Pentium(?:\s+Silver|\s+Gold)?)\s+(N?\d{3,5}[A-Z]?)\b", re.IGNORECASE),
|
||||
re.compile(r"\bMediaTek\s+(Kompanio\s+\d{3,4}|MT\d{4})\b", re.IGNORECASE),
|
||||
]
|
||||
|
||||
_COLOUR_WORDS = re.compile(
|
||||
r"\b(black|white|blue|green|red|grey|gray|silver|gold|purple|violet|pink|yellow|orange|"
|
||||
r"cream|titanium|graphite|midnight|starlight|mint|lavender|bronze|copper|beige|teal|"
|
||||
r"navy|jade|coral|onyx|marble|obsidian|porcelain|hazel|aqua|lime|sand|charcoal)\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
# Tokens that describe the device class or connectivity, not the model.
|
||||
_MODEL_NOISE = re.compile(
|
||||
r"\b(?:5g|4g|lte|smartphone|smart\s+phone|mobile\s+phone|mobile|phone|dual\s+sim|"
|
||||
r"laptop|notebook|thin\s+and\s+light|thin\s+&\s+light|gaming|new|latest|with\b.*$)",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_SPEC_TOKEN = re.compile(
|
||||
r"^(?:\d+(?:\.\d+)?\s*(?:gb|tb|mp|mah|hz|inch|inches|cm|w|kg|g)|\d+(?:th|nd|rd|st)|gen|ddr\d?\w*|"
|
||||
r"lpddr\d\w*|ssd|hdd|fhd|qhd|uhd|oled|ips|win|windows|20\d\d)$",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class ParsedTitle:
|
||||
title: str
|
||||
brand: Optional[BrandMatch]
|
||||
model: Optional[str] = None # "Galaxy S24", "Redmi Note 13 Pro", "15s"
|
||||
model_norm: Optional[str] = None # "galaxy s24", matching form
|
||||
ram_gb: Optional[Decimal] = None
|
||||
storage_gb: Optional[Decimal] = None
|
||||
colour: Optional[str] = None
|
||||
processor: Optional[str] = None # "i5-1334u", "ryzen 5 7530u", "m3"
|
||||
mpn: Optional[str] = None # laptop part number when stated
|
||||
network: Optional[str] = None
|
||||
notes: List[str] = field(default_factory=list)
|
||||
|
||||
|
||||
def _dec(value: str, unit: str = "GB") -> Decimal:
|
||||
d = Decimal(value)
|
||||
if unit.upper() == "TB":
|
||||
d = d * 1024
|
||||
return d.normalize() if d == d.to_integral() else d
|
||||
|
||||
|
||||
def _parse_ram_storage(text: str, category: str):
|
||||
ram = storage = None
|
||||
for rx in _RAM_RES:
|
||||
m = rx.search(text)
|
||||
if m:
|
||||
ram = _dec(m.group(1))
|
||||
break
|
||||
m = _PAIR_RE.search(text)
|
||||
if m:
|
||||
a, b, unit = m.group(1), m.group(2), (m.group(3) or "GB")
|
||||
pair_ram, pair_storage = _dec(a), _dec(b, unit)
|
||||
# "Core Ultra 5/ 16GB RAM/ 512GB" is not a 5 GB / 16 GB pair: a real
|
||||
# pair has device-sized storage.
|
||||
min_pair_storage = Decimal(16) if category == "mobiles" else Decimal(64)
|
||||
if pair_storage > pair_ram and pair_storage >= min_pair_storage:
|
||||
ram = ram if ram is not None else pair_ram
|
||||
storage = pair_storage
|
||||
if storage is None:
|
||||
m = _STORAGE_LABELLED.search(text)
|
||||
if m:
|
||||
storage = _dec(m.group(1), m.group(2))
|
||||
if storage is None:
|
||||
# Unlabelled sizes: the storage is the largest one that is not the RAM.
|
||||
min_storage = Decimal(16) if category == "mobiles" else Decimal(32)
|
||||
sizes = [_dec(v, u) for v, u in _SIZE_ANY.findall(text)]
|
||||
candidates = [s for s in sizes if s != ram and s >= min_storage]
|
||||
if candidates:
|
||||
storage = max(candidates)
|
||||
if ram is None:
|
||||
# "(8 GB RAM)" handled above; an unlabelled small size next to a larger
|
||||
# one ("8GB 256GB") is the RAM.
|
||||
sizes = [_dec(v, u) for v, u in _SIZE_ANY.findall(text)]
|
||||
small = [s for s in sizes if s <= (24 if category == "mobiles" else 64) and (storage is None or s < storage)]
|
||||
if len(set(small)) == 1 and storage is not None:
|
||||
ram = small[0]
|
||||
plausible_ram = Decimal(32) if category == "mobiles" else Decimal(128)
|
||||
if ram is not None and not (Decimal(1) <= ram <= plausible_ram):
|
||||
ram = None
|
||||
return ram, storage
|
||||
|
||||
|
||||
def parse_processor(text: str) -> Optional[str]:
|
||||
"""Normalised CPU name ("i5-1334u", "ryzen 3 7320u", "core ultra 5 125h",
|
||||
"core 5 120u", "athlon 7120u", "m2"), or None."""
|
||||
for rx in _PROCESSOR_RES:
|
||||
m = rx.search(text or "")
|
||||
if not m:
|
||||
continue
|
||||
parts = [g for g in m.groups() if g]
|
||||
matched = m.group(0).lower()
|
||||
if "ryzen" in matched:
|
||||
return f"ryzen {parts[0]} {parts[1]}".lower()
|
||||
if "athlon" in matched:
|
||||
return f"athlon {parts[1]}".lower()
|
||||
if "ultra" in matched and "core" in matched:
|
||||
return f"core ultra {parts[0]} {parts[1]}".lower()
|
||||
if "core" in matched and parts[0].lower().startswith("i"):
|
||||
return f"{parts[0]}-{parts[1]}".lower()
|
||||
if "core" in matched:
|
||||
return f"core {parts[0]} {parts[1]}".lower()
|
||||
return " ".join(parts).lower()
|
||||
return None
|
||||
|
||||
|
||||
_parse_processor = parse_processor
|
||||
|
||||
|
||||
def processor_is_specific(processor: Optional[str]) -> bool:
|
||||
"""True for a CPU named down to its model number ("i5-1334u"), which
|
||||
together with brand, model line, RAM and storage identifies a laptop
|
||||
configuration. "m2" (Apple) also counts."""
|
||||
if not processor:
|
||||
return False
|
||||
return bool(re.search(r"\d{3,}", processor)) or bool(re.fullmatch(r"m[1-9](?: (?:pro|max|ultra))?", processor))
|
||||
|
||||
|
||||
_MPN_RE = re.compile(
|
||||
r"(?<![\w-])([A-Z0-9]{2,8}-[A-Z0-9]{2,10}(?:-[A-Z0-9]{1,6})?|"
|
||||
r"[A-Z0-9]{2,6}[A-Z]{0,4}\d{2,6}[A-Z]{1,4}\d{0,4}[A-Z]{0,3}|\d{2}[A-Z]{2}\d{3,}[A-Z0-9]{2,})(?![\w-])",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
def _parse_mpn(text: str, processor: Optional[str]) -> Optional[str]:
|
||||
"""A manufacturer part number such as 82XV00BHIN or fd0112TU, when the
|
||||
title states one (laptops). Tokens that are specs or CPU names are not."""
|
||||
candidates = []
|
||||
for m in _MPN_RE.finditer(text):
|
||||
tok = m.group(1)
|
||||
low = tok.lower()
|
||||
if len(tok) < 6 or len(tok) > 20:
|
||||
continue
|
||||
if sum(c.isdigit() for c in tok) < 2 or sum(c.isalpha() for c in tok) < 2:
|
||||
continue
|
||||
if _SPEC_TOKEN.match(low) or re.match(r"^(?:i[3579]|m[1-9]|rtx|gtx|rx|ddr|lpddr)", low):
|
||||
continue
|
||||
if processor and low in processor.replace("-", " ").split() + [processor.replace(" ", "")]:
|
||||
continue
|
||||
if re.search(r"\d+(?:gb|tb|mp|mah|hz|w)$", low):
|
||||
continue
|
||||
if re.search(r"-(?:core|inch|cell|bit|gen|thread)s?$|^\d+-", low) and not re.search(r"[a-z]\d", low.split("-")[-1]):
|
||||
continue # "10-Core", "15-inch", "3-Cell" describe hardware, not a part number
|
||||
candidates.append(tok)
|
||||
return candidates[-1].upper() if candidates else None
|
||||
|
||||
|
||||
def _parse_colour(title: str) -> Optional[str]:
|
||||
# Inside brackets first: "(Onyx Black, 8GB RAM, 256GB Storage)"
|
||||
for group in re.findall(r"\(([^()]*)\)", title):
|
||||
for part in re.split(r"[,|/]", group):
|
||||
part = part.strip()
|
||||
if part and not re.search(r"\d", part) and _COLOUR_WORDS.search(part):
|
||||
return part.title()
|
||||
# Trailing "- Black"
|
||||
m = re.search(r"[-–|,]\s*([A-Za-z][A-Za-z ]{2,30})\s*$", title)
|
||||
if m and _COLOUR_WORDS.search(m.group(1)) and not re.search(r"\d", m.group(1)):
|
||||
return m.group(1).strip().title()
|
||||
return None
|
||||
|
||||
|
||||
def normalise_model(model: str) -> str:
|
||||
text = model.lower()
|
||||
text = re.sub(r"[()\[\],|]", " ", text)
|
||||
text = _MODEL_NOISE.sub(" ", text)
|
||||
text = re.sub(r"\+", " plus ", text)
|
||||
text = re.sub(r"[^a-z0-9 ]+", " ", text)
|
||||
tokens = [t for t in text.split() if not _SPEC_TOKEN.match(t)]
|
||||
return " ".join(tokens)
|
||||
|
||||
|
||||
_LAPTOP_SPEC_START = re.compile(
|
||||
r"\b(?:intel|amd|apple\s+m[1-9]|m[1-9]\s+chip|core\s+(?:i[3579]|ultra)|ryzen|snapdragon|celeron|pentium|"
|
||||
r"mediatek|\d+(?:th|nd|rd|st)\s+gen|\d+(?:\.\d+)?\s*(?:-|\s)?(?:inch|cm|\"))",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_DISPLAY_NOISE = re.compile(
|
||||
r"\b(?:5g|4g|lte|smartphone|smart\s+phone|mobile\s+phone|dual\s+sim|laptop|notebook|"
|
||||
r"thin\s+and\s+light|thin\s+&\s+light|gaming|new|latest)\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
def _model_from_title(title: str, brand: Optional[BrandMatch], category: str, colour: Optional[str],
|
||||
mpn: Optional[str] = None):
|
||||
"""(display model, matching model_norm) from the head of the title."""
|
||||
head = re.sub(r"^\s*buy\s+", "", title, flags=re.IGNORECASE)
|
||||
if mpn:
|
||||
# A part number is not the model name. HP writes the line into it
|
||||
# ("15-fc0500AU" is an HP 15), so that prefix is kept.
|
||||
prefix = mpn.split("-", 1)[0] if "-" in mpn and len(mpn.split("-", 1)[0]) <= 4 else ""
|
||||
head = re.sub(re.escape(mpn), f" {prefix} ", head, flags=re.IGNORECASE)
|
||||
# A short model token in brackets right after the name is part of it:
|
||||
# "Nothing Phone (2a) 5G (Black, 128 GB)".
|
||||
head = re.sub(r"\((?:19|20)\d\d\)", " ", head) # "(2026)" is a model year, not part of the name
|
||||
head = re.sub(r"\(([A-Za-z0-9+ ]{1,6})\)", lambda m: " " + m.group(1) + " "
|
||||
if not re.search(r"\d\s*(?:gb|tb)", m.group(1), re.I) else m.group(0), head, count=1)
|
||||
head = re.split(r"\s[-–|]\s|[(,|\[:]", head, maxsplit=1)[0]
|
||||
if category == "laptops":
|
||||
m = _LAPTOP_SPEC_START.search(head)
|
||||
if m and m.start() > 0:
|
||||
head = head[: m.start()]
|
||||
if brand:
|
||||
# Drop the parent brand's own name ("Samsung Galaxy S24" -> "Galaxy S24",
|
||||
# "Apple iPhone 15" -> "iPhone 15"); a sub-brand stays ("Redmi Note 13").
|
||||
from app.electronics.reference import load_reference
|
||||
|
||||
parent_aliases = sorted(load_reference().brands[brand.brand_slug].aliases, key=len, reverse=True)
|
||||
for alias in parent_aliases:
|
||||
head = re.sub(r"^\s*" + re.escape(alias) + r"\b", "", head, flags=re.IGNORECASE).strip()
|
||||
head = re.sub(r"\b\d+\s*GB\s*RAM\b", " ", head, flags=re.IGNORECASE)
|
||||
head = _PAIR_RE.sub(" ", head)
|
||||
head = _SIZE_ANY.sub(" ", head)
|
||||
if colour:
|
||||
head = re.sub(re.escape(colour), " ", head, flags=re.IGNORECASE)
|
||||
words = head.split()
|
||||
while len(words) > 1 and _COLOUR_WORDS.fullmatch(words[-1]):
|
||||
words.pop() # "iPhone 15 Black" -> "iPhone 15"
|
||||
head = " ".join(words)
|
||||
display = re.sub(r"\s+", " ", _DISPLAY_NOISE.sub(" ", head)).strip(" -–")
|
||||
norm = normalise_model(head)
|
||||
if not norm:
|
||||
return None, None
|
||||
return display or head.strip(), norm
|
||||
|
||||
|
||||
def parse_title(title: str, category: str, *, expected_brand: Optional[str] = None) -> ParsedTitle:
|
||||
title = re.sub(r"\s+", " ", (title or "")).strip()
|
||||
brand = resolve_brand(title, expected=expected_brand)
|
||||
parsed = ParsedTitle(title=title, brand=brand)
|
||||
if not title:
|
||||
return parsed
|
||||
|
||||
parsed.ram_gb, parsed.storage_gb = _parse_ram_storage(title, category)
|
||||
parsed.colour = _parse_colour(title)
|
||||
if re.search(r"\b5G\b", title, re.IGNORECASE):
|
||||
parsed.network = "5G"
|
||||
if category == "laptops":
|
||||
parsed.processor = _parse_processor(title)
|
||||
parsed.mpn = _parse_mpn(title, parsed.processor)
|
||||
|
||||
parsed.model, parsed.model_norm = _model_from_title(title, brand, category, parsed.colour, parsed.mpn)
|
||||
return parsed
|
||||
|
||||
|
||||
def variant_key(parsed: ParsedTitle, category: str) -> Optional[str]:
|
||||
"""The identity of one real-world variant, or None if the title does not
|
||||
state enough to tell variants apart."""
|
||||
if not parsed.brand or not parsed.model_norm:
|
||||
return None
|
||||
b = parsed.brand.brand_slug
|
||||
fmt = lambda d: "na" if d is None else format(d.normalize(), "f") # noqa: E731
|
||||
if category == "laptops":
|
||||
# A laptop configuration is its model line + CPU + RAM + storage. That
|
||||
# is what every site states (a part number is shown by only a few), so
|
||||
# it is the key whenever it is complete; the MPN is the fallback.
|
||||
line = laptop_line(parsed.model_norm)
|
||||
if line and processor_is_specific(parsed.processor) and parsed.ram_gb and parsed.storage_gb:
|
||||
return f"{b}|laptops|{line}|{parsed.processor}|{fmt(parsed.ram_gb)}|{fmt(parsed.storage_gb)}"
|
||||
if parsed.mpn:
|
||||
return f"{b}|laptops|mpn:{parsed.mpn.lower()}"
|
||||
return None
|
||||
if parsed.storage_gb is None:
|
||||
return None
|
||||
return f"{b}|{category}|{parsed.model_norm}|{fmt(parsed.ram_gb)}|{fmt(parsed.storage_gb)}"
|
||||
|
||||
|
||||
# Lenovo/Asus machine-type codes ("15amn8", "15irh10", "14iah8", "x1504za")
|
||||
# name a chassis generation, and one site prints them where another does not.
|
||||
_MACHINE_CODE = re.compile(r"^(?:\d{2}[a-z]{2,4}\d{1,2}|[a-z]\d{4}[a-z]{1,3})$")
|
||||
|
||||
|
||||
def laptop_line(model_norm: Optional[str]) -> str:
|
||||
"""The model line used for matching: "ideapad slim 3 15amn8" -> "ideapad slim 3"."""
|
||||
tokens = [t for t in (model_norm or "").split() if not _MACHINE_CODE.match(t)]
|
||||
return " ".join(tokens)
|
||||
|
||||
|
||||
def fill_from_context(parsed: ParsedTitle, category: str, *, snippet: str = "",
|
||||
spec_texts: tuple = ()) -> ParsedTitle:
|
||||
"""Fill variant fields a (often truncated) title leaves out, from text the
|
||||
same site published about the same page: its search snippet, or the spec
|
||||
table of the fetched page. Only unambiguous values are taken - a snippet
|
||||
naming two different storage sizes is describing several variants."""
|
||||
if snippet and (parsed.ram_gb is None or parsed.storage_gb is None):
|
||||
sizes = {_dec(v, u) for v, u in _SIZE_ANY.findall(snippet)}
|
||||
if len(sizes) <= 2:
|
||||
ram, storage = _parse_ram_storage(snippet, category)
|
||||
if parsed.storage_gb is None and storage is not None:
|
||||
parsed.storage_gb = storage
|
||||
if parsed.ram_gb is None and ram is not None and ram != parsed.storage_gb:
|
||||
parsed.ram_gb = ram
|
||||
if category == "laptops" and not processor_is_specific(parsed.processor):
|
||||
# A title that already names a CPU family ("Snapdragon X", "Core i7")
|
||||
# is only completed from the page's own spec table, never from a
|
||||
# snippet - snippets often run several products' titles together.
|
||||
sources = list(spec_texts) + ([snippet] if parsed.processor is None and snippet else [])
|
||||
found = set()
|
||||
for text in sources:
|
||||
found |= {cpu for cpu in all_processors(text) if processor_is_specific(cpu)}
|
||||
if len(found) == 1:
|
||||
parsed.processor = found.pop()
|
||||
return parsed
|
||||
|
||||
|
||||
def all_processors(text: str) -> set:
|
||||
"""Every CPU named anywhere in `text` (a snippet can name several)."""
|
||||
found = set()
|
||||
for rx in _PROCESSOR_RES:
|
||||
for m in rx.finditer(text or ""):
|
||||
cpu = parse_processor(m.group(0))
|
||||
if cpu:
|
||||
found.add(cpu)
|
||||
return found
|
||||
Reference in New Issue
Block a user