Electronics Catalog: API, MCP server, frontend and deployment

Verified catalogue of mobiles and laptops sold in India, collected from
real retail listings (FastAPI backend, React frontend, Postgres/pgvector).

- REST API under /api/elec (read-only catalogue; admin endpoints need login)
- MCP server (FastMCP) at /mcp/ with list_categories, search_products,
  get_product and price_history tools
- Real ratings and reviews read from product pages and search results
- Production Dockerfile (requirements-api.txt, no PyTorch) and
  .env.production.example; remote database only via an explicit
  ELEC_ALLOW_REMOTE_DB host/name allowlist
- docs/API.md: endpoint and MCP reference with live examples

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
sriram
2026-10-01 12:17:42 +05:30
commit c7e4d59188
115 changed files with 14329 additions and 0 deletions

View File

@@ -0,0 +1,86 @@
"""Resolve the brand of a product title against the closed allow-list."""
from __future__ import annotations
import re
from dataclasses import dataclass
from functools import lru_cache
from typing import List, Optional, Tuple
from app.electronics.reference import load_reference
@dataclass(frozen=True)
class BrandMatch:
brand_slug: str
brand_name: str
family: Optional[str] # sub-brand (Redmi, iQOO, Pixel...) when the title used one
matched: str # the alias text found in the title
@lru_cache(maxsize=1)
def _alias_table() -> List[Tuple[str, str, bool]]:
"""(alias, brand_slug, is_sub_brand), longest alias first."""
ref = load_reference()
rows: List[Tuple[str, str, bool]] = []
for b in ref.brands.values():
for a in b.aliases:
rows.append((a, b.slug, False))
for s in b.sub_brands:
rows.append((s, b.slug, True))
rows.sort(key=lambda r: -len(r[0]))
return rows
def resolve_brand(title: str, *, expected: Optional[str] = None) -> Optional[BrandMatch]:
"""The allow-listed brand a title starts with (or names within its first
few words), or None. `expected` restricts the match to one brand slug.
Only the start of the title is considered: "Case for Samsung Galaxy S24"
is an accessory, not a Samsung phone.
"""
if not title:
return None
ref = load_reference()
head = " ".join(re.findall(r"[a-z0-9+]+", title.lower())[:3])
for alias, slug, is_sub in _alias_table():
if expected and slug != expected:
continue
pattern = r"(?:^|\s)" + re.escape(alias) + r"(?:\s|$)"
m = re.search(pattern, head)
if not m:
continue
# The brand/sub-brand must be the first or second word ("Apple iPhone",
# "Samsung Galaxy", "Xiaomi Redmi Note") - not buried later.
if len(head[: m.start()].split()) > 1:
continue
# "Google Pixel 8", "Xiaomi Redmi Note 13": the parent brand matched,
# but the family is the sub-brand that follows it.
sub = alias if is_sub else next(
(s for s in ref.brands[slug].sub_brands if re.search(r"(?:^|\s)" + re.escape(s) + r"(?:\s|$)", head)),
None,
)
return BrandMatch(slug, ref.brands[slug].name, _family_casing(sub) if sub else None, alias)
return None
_CASING = {"iphone": "iPhone", "iqoo": "iQOO", "macbook": "MacBook", "rog": "ROG", "tuf": "TUF",
"cmf": "CMF", "loq": "LOQ", "poco": "POCO", "mi": "Mi", "xps": "XPS", "thinkpad": "ThinkPad",
"ideapad": "IdeaPad", "thinkbook": "ThinkBook", "vivobook": "Vivobook", "zenbook": "Zenbook"}
def _family_casing(sub: str) -> str:
return _CASING.get(sub, sub.title())
# Words that mark an accessory or a non-product page, not a device.
_NOT_A_DEVICE = re.compile(
r"\b(?:case|cover|back\s+cover|tempered|screen\s+guard|protector|charger|adapter|cable|"
r"skin|sleeve|bag|backpack|stand|holder|refurbished|renewed|pre-?owned|used|"
r"compare|vs\.?|versus|review|specifications?\s+and|price\s+list|best\s+\w+\s+under|"
r"top\s+\d+|all\s+models)\b",
re.IGNORECASE,
)
def looks_like_device_title(title: str) -> bool:
return bool(title) and not _NOT_A_DEVICE.search(title)

View File

@@ -0,0 +1,55 @@
"""Is a value actually stated in the text it supposedly came from?
Every value the LLM returns passes through value_in_source() against the exact
text the model was shown. Anything that cannot be found there is discarded,
which is what stops a small model's guess becoming a stored fact.
"""
from __future__ import annotations
import re
from decimal import Decimal, InvalidOperation
from typing import Union
_WS = re.compile(r"\s+")
def _norm_text(text: str) -> str:
text = text.lower().replace(" ", " ")
text = re.sub(r"[^\w.+ ]+", " ", text)
text = re.sub(r"(?<!\d)\.|\.(?!\d)", " ", text) # sentence full stops, not decimals
# "5000mAh" and "5000 mAh" must compare equal.
text = re.sub(r"(?<=\d)(?=[a-z])|(?<=[a-z])(?=\d)", " ", text)
return _WS.sub(" ", text).strip()
def _numbers_in(text: str) -> set:
found = set()
for raw in re.findall(r"\d[\d,]*(?:\.\d+)?", text):
try:
found.add(Decimal(raw.replace(",", "")).normalize())
except InvalidOperation:
continue
return found
def value_in_source(value: Union[str, int, float, Decimal, None], source: str) -> bool:
if value is None or not source:
return False
if isinstance(value, bool):
return False
if isinstance(value, (int, float, Decimal)):
try:
return Decimal(str(value)).normalize() in _numbers_in(source)
except InvalidOperation:
return False
text = str(value).strip()
if not text:
return False
# A string with a number in it ("5000 mAh", "Snapdragon 8 Gen 3") must have
# every one of its numbers in the source, and its words too.
nums = _numbers_in(text)
if nums and not nums <= _numbers_in(source):
return False
words = [w for w in _norm_text(text).split() if not re.fullmatch(r"[\d.,]+", w)]
hay = f" {_norm_text(source)} "
return all(f" {w} " in hay for w in words) if words else bool(nums)

View File

@@ -0,0 +1,71 @@
"""Fill MISSING spec keys from page text with the local LLM - and keep only
what the text actually says.
The model sees one block of text that we fetched (a spec section or a
description) and is asked to copy values out of it. Every value it returns is:
1. checked by grounding.value_in_source() against that same text, and
2. normalised by spec_normaliser (units, plausible ranges).
Anything failing either step is dropped. Prices, product names and images are
never asked of the model.
"""
from __future__ import annotations
import json
import logging
from typing import Any, Dict, Iterable, Tuple
from app.electronics.normalise.grounding import value_in_source
from app.electronics.normalise.spec_normaliser import normalise_value
from app.infrastructure.settings import ELEC_USE_LLM
logger = logging.getLogger(__name__)
MAX_SOURCE_CHARS = 3500
SYSTEM_PROMPT = (
"You copy product specifications out of the text you are given. "
"Rules: use ONLY the given text; copy each value exactly as written, including its unit; "
"if the text does not state a value, use null; never guess, estimate or use outside knowledge. "
"Reply with one JSON object whose keys are exactly the requested keys."
)
def fill_missing(
category: str,
source_text: str,
missing_keys: Iterable[str],
*,
generate=None,
) -> Tuple[Dict[str, Any], Dict[str, str]]:
"""(specs, sources) for whichever of `missing_keys` the text states."""
keys = [k for k in missing_keys if k != "colour"]
text = (source_text or "").strip()[:MAX_SOURCE_CHARS]
if not keys or not text or not ELEC_USE_LLM:
return {}, {}
if generate is None:
from app.services.ollama_service import generate_json as generate
prompt = (
f"Requested keys: {json.dumps(keys)}\n\n"
f"Text:\n\"\"\"\n{text}\n\"\"\"\n\n"
"JSON:"
)
reply = generate(SYSTEM_PROMPT, prompt)
if not isinstance(reply, dict):
return {}, {}
specs: Dict[str, Any] = {}
sources: Dict[str, str] = {}
for key in keys:
raw = reply.get(key)
if raw is None or isinstance(raw, (dict, list, bool)):
continue
if not value_in_source(raw, text):
logger.debug("LLM value %r for %s not found in source text; dropped", raw, key)
continue
value = normalise_value(category, key, raw)
if value is None:
continue
specs[key] = value
sources[key] = f"llm-extracted: {str(raw)[:80]}"
return specs, sources

View File

@@ -0,0 +1,101 @@
"""Map raw spec labels/values from a page to canonical keys and units.
Deterministic and table-driven (reference/spec_keys.yaml). A value that cannot
be parsed, or lands outside the plausible range for its key, is dropped - never
estimated.
"""
from __future__ import annotations
import re
from decimal import Decimal, InvalidOperation
from functools import lru_cache
from typing import Any, Dict, Optional, Tuple
from app.electronics.reference import load_reference
def _label(text: str) -> str:
return re.sub(r"[^a-z0-9]+", " ", str(text).lower()).strip()
@lru_cache(maxsize=None)
def _synonyms(category: str) -> Dict[str, str]:
table: Dict[str, str] = {}
for key, spec in load_reference().spec_keys.get(category, {}).items():
for syn in [key.replace("_", " "), *spec.get("synonyms", [])]:
table.setdefault(_label(syn), key)
return table
def canonical_key(category: str, label: str) -> Optional[str]:
return _synonyms(category).get(_label(label))
# unit -> (regex for the unit in text, factor into the canonical unit)
_UNIT_PATTERNS = {
"GB": [(r"tb", Decimal(1024)), (r"gb", Decimal(1)), (r"mb", Decimal(1) / 1024)],
"inch": [(r"(?:inch(?:es)?|in\b|\"|”)", Decimal(1)), (r"cm", Decimal(1) / Decimal("2.54"))],
"Hz": [(r"hz", Decimal(1))],
"MP": [(r"mp|megapixel", Decimal(1))],
"mAh": [(r"mah", Decimal(1))],
"kg": [(r"kg|kilogram", Decimal(1)), (r"(?<![k])g\b|grams?", Decimal("0.001"))],
"Wh": [(r"wh|watt\s*hours?", Decimal(1))],
}
def _to_number(value: str, unit: str) -> Optional[Decimal]:
text = str(value).lower().replace(",", "")
patterns = _UNIT_PATTERNS.get(unit, [])
# Prefer an amount written in the canonical unit ("39.62 cm (15.6 inch)" -> 15.6).
for unit_re, factor in patterns:
m = re.search(r"(\d+(?:\.\d+)?)\s*(?:" + unit_re + r")", text)
if m:
try:
return (Decimal(m.group(1)) * factor).quantize(Decimal("0.01")).normalize()
except InvalidOperation:
return None
# A bare number is accepted only when nothing else is in the value.
m = re.fullmatch(r"\s*(\d+(?:\.\d+)?)\s*", text)
if m:
return Decimal(m.group(1)).normalize()
return None
def normalise_value(category: str, key: str, value: Any) -> Optional[Any]:
spec = load_reference().spec_keys.get(category, {}).get(key)
if spec is None or value is None:
return None
text = str(value).strip()
if not text or text.lower() in {"na", "n/a", "-", "none", "not applicable", "no"}:
return None
kind = spec.get("type")
if kind == "number":
num = _to_number(text, spec.get("unit", ""))
if num is None:
return None
lo, hi = spec.get("range", [None, None])
if (lo is not None and num < Decimal(str(lo))) or (hi is not None and num > Decimal(str(hi))):
return None
return float(num) if num != num.to_integral() else int(num)
if kind == "enum":
low = text.lower()
for canon, words in spec.get("values", {}).items():
if any(re.search(r"\b" + re.escape(w) + r"\b", low) for w in words):
return canon
return None
return re.sub(r"\s+", " ", text)[:120]
def normalise_specs(category: str, raw: Dict[str, Any]) -> Tuple[Dict[str, Any], Dict[str, str]]:
"""(specs, sources): canonical key -> value, and key -> the raw label it came from."""
specs: Dict[str, Any] = {}
sources: Dict[str, str] = {}
for label, value in (raw or {}).items():
key = canonical_key(category, label)
if not key or key in specs:
continue
norm = normalise_value(category, key, value)
if norm is not None:
specs[key] = norm
sources[key] = f"{label}: {value}"[:200]
return specs, sources

View File

@@ -0,0 +1,374 @@
"""Split a retail product title into model, variant and a matching key.
Everything returned is read from the title text; a value the title does not
state is None. Titles differ a lot between sites:
Samsung Galaxy S24 5G (Onyx Black, 8GB RAM, 256GB Storage) Amazon
SAMSUNG Galaxy S24 5G (Onyx Black, 256 GB) (8 GB RAM) Flipkart
Samsung Galaxy S24 5G (8GB RAM, 256GB, Onyx Black) Croma
Redmi Note 13 Pro 5G (8GB + 256GB)
Apple iPhone 15 (128 GB) - Black
HP 15s, 13th Gen Intel Core i5-1334U, 16GB DDR4, 512GB SSD, ... fd0112TU
so the model is taken from the text before the first bracket/comma, and RAM /
storage / colour from anywhere in the title.
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
from decimal import Decimal
from typing import List, Optional
from app.electronics.normalise.brand_alias import BrandMatch, resolve_brand
_NUM = r"(\d+(?:\.\d+)?)"
# "8GB RAM", "8 GB LPDDR5X RAM", "RAM 8GB", "16GB DDR4" (laptops)
_RAM_RES = [
re.compile(_NUM + r"\s*GB\s*(?:LP)?(?:DDR\s?\d\w?\s*)?RAM\b", re.IGNORECASE),
re.compile(r"\bRAM\s*[:\-]?\s*" + _NUM + r"\s*GB", re.IGNORECASE),
re.compile(_NUM + r"\s*GB\s*(?:LP)?DDR\s?\d", re.IGNORECASE),
re.compile(_NUM + r"\s*GB\s*(?:unified\s+memory|memory)\b", re.IGNORECASE),
]
# "8GB + 256GB", "8/256", "8GB/256GB", "12+512GB"
_PAIR_RE = re.compile(r"(?<![\d.])(\d{1,2})\s*(?:GB)?\s*[+/]\s*(\d{2,4}|1|2)\s*(GB|TB)?\b", re.IGNORECASE)
# explicit storage: "256GB Storage", "512GB SSD", "1TB", "256 GB ROM"
_STORAGE_LABELLED = re.compile(
_NUM + r"\s*(GB|TB)\s*(?:SSD|ROM|storage|internal(?:\s+storage)?|HDD|eMMC|UFS|NVMe|PCIe)\b", re.IGNORECASE
)
_SIZE_ANY = re.compile(r"(?<![\d.])" + _NUM + r"\s*(GB|TB)\b", re.IGNORECASE)
# Order matters: the first pattern that matches wins. AMD comes before the
# Intel "Core N" pattern, because retail titles write core counts as words
# ("Ryzen 3 Quad Core 7320U"), and "Core 7320U" must not read as Intel.
_CORE_COUNT = r"(?:(?:Dual|Quad|Hexa|Octa|Six|Eight)\s+Core\s+)?"
_PROCESSOR_RES = [
re.compile(r"\b(?:AMD\s+)?Ryzen\s+R?(\d)\s*(?:Pro\s+)?" + _CORE_COUNT + r"[- ]?(\d{4}[A-Z]{0,3})\b", re.IGNORECASE),
re.compile(r"\b(?:AMD\s+)?(Athlon)\s+(?:Silver\s+|Gold\s+)?" + _CORE_COUNT + r"(\d{4}[A-Z]{0,2})\b", re.IGNORECASE),
re.compile(r"\b(?:Intel\s+)?Core\s+Ultra\s+(\d)\s*[- ]?\s*(\d{3}[A-Z]{0,2})\b", re.IGNORECASE),
re.compile(r"\b(?:Intel\s+)?Core\s+(?:\(?\d+(?:th|nd|rd|st)\s+Gen\)?\s+)?(i[3579])\s*[- ]?\s*(\d{4,5}[A-Z]{0,3})\b", re.IGNORECASE),
re.compile(r"\b(?:Intel\s+)?Core\s+(i[3579])\s+\d+(?:th|nd|rd|st)\s+Gen\s+(\d{4,5}[A-Z]{0,3})\b", re.IGNORECASE),
re.compile(r"(?<!Dual )(?<!Quad )(?<!Hexa )(?<!Octa )(?<!Six )(?<!Eight )\b(?:Intel\s+)?Core\s+([3579])\s*[- ]?\s*(\d{3}[A-Z]{0,2})\b", re.IGNORECASE),
re.compile(r"\bApple\s+(M[1-9])(?:\s+(Pro|Max|Ultra))?\b", re.IGNORECASE),
re.compile(r"\b(M[1-9])\s*(Pro|Max|Ultra)?\s+chip\b", re.IGNORECASE),
re.compile(r"\bSnapdragon\s+(X\d?)\s*(Elite|Plus)?\s*(X\d{1,2}-\d{3})?", re.IGNORECASE),
re.compile(r"\b(?:Intel\s+)?(Celeron|Pentium(?:\s+Silver|\s+Gold)?)\s+(N?\d{3,5}[A-Z]?)\b", re.IGNORECASE),
re.compile(r"\bMediaTek\s+(Kompanio\s+\d{3,4}|MT\d{4})\b", re.IGNORECASE),
]
_COLOUR_WORDS = re.compile(
r"\b(black|white|blue|green|red|grey|gray|silver|gold|purple|violet|pink|yellow|orange|"
r"cream|titanium|graphite|midnight|starlight|mint|lavender|bronze|copper|beige|teal|"
r"navy|jade|coral|onyx|marble|obsidian|porcelain|hazel|aqua|lime|sand|charcoal)\b",
re.IGNORECASE,
)
# Tokens that describe the device class or connectivity, not the model.
_MODEL_NOISE = re.compile(
r"\b(?:5g|4g|lte|smartphone|smart\s+phone|mobile\s+phone|mobile|phone|dual\s+sim|"
r"laptop|notebook|thin\s+and\s+light|thin\s+&\s+light|gaming|new|latest|with\b.*$)",
re.IGNORECASE,
)
_SPEC_TOKEN = re.compile(
r"^(?:\d+(?:\.\d+)?\s*(?:gb|tb|mp|mah|hz|inch|inches|cm|w|kg|g)|\d+(?:th|nd|rd|st)|gen|ddr\d?\w*|"
r"lpddr\d\w*|ssd|hdd|fhd|qhd|uhd|oled|ips|win|windows|20\d\d)$",
re.IGNORECASE,
)
@dataclass
class ParsedTitle:
title: str
brand: Optional[BrandMatch]
model: Optional[str] = None # "Galaxy S24", "Redmi Note 13 Pro", "15s"
model_norm: Optional[str] = None # "galaxy s24", matching form
ram_gb: Optional[Decimal] = None
storage_gb: Optional[Decimal] = None
colour: Optional[str] = None
processor: Optional[str] = None # "i5-1334u", "ryzen 5 7530u", "m3"
mpn: Optional[str] = None # laptop part number when stated
network: Optional[str] = None
notes: List[str] = field(default_factory=list)
def _dec(value: str, unit: str = "GB") -> Decimal:
d = Decimal(value)
if unit.upper() == "TB":
d = d * 1024
return d.normalize() if d == d.to_integral() else d
def _parse_ram_storage(text: str, category: str):
ram = storage = None
for rx in _RAM_RES:
m = rx.search(text)
if m:
ram = _dec(m.group(1))
break
m = _PAIR_RE.search(text)
if m:
a, b, unit = m.group(1), m.group(2), (m.group(3) or "GB")
pair_ram, pair_storage = _dec(a), _dec(b, unit)
# "Core Ultra 5/ 16GB RAM/ 512GB" is not a 5 GB / 16 GB pair: a real
# pair has device-sized storage.
min_pair_storage = Decimal(16) if category == "mobiles" else Decimal(64)
if pair_storage > pair_ram and pair_storage >= min_pair_storage:
ram = ram if ram is not None else pair_ram
storage = pair_storage
if storage is None:
m = _STORAGE_LABELLED.search(text)
if m:
storage = _dec(m.group(1), m.group(2))
if storage is None:
# Unlabelled sizes: the storage is the largest one that is not the RAM.
min_storage = Decimal(16) if category == "mobiles" else Decimal(32)
sizes = [_dec(v, u) for v, u in _SIZE_ANY.findall(text)]
candidates = [s for s in sizes if s != ram and s >= min_storage]
if candidates:
storage = max(candidates)
if ram is None:
# "(8 GB RAM)" handled above; an unlabelled small size next to a larger
# one ("8GB 256GB") is the RAM.
sizes = [_dec(v, u) for v, u in _SIZE_ANY.findall(text)]
small = [s for s in sizes if s <= (24 if category == "mobiles" else 64) and (storage is None or s < storage)]
if len(set(small)) == 1 and storage is not None:
ram = small[0]
plausible_ram = Decimal(32) if category == "mobiles" else Decimal(128)
if ram is not None and not (Decimal(1) <= ram <= plausible_ram):
ram = None
return ram, storage
def parse_processor(text: str) -> Optional[str]:
"""Normalised CPU name ("i5-1334u", "ryzen 3 7320u", "core ultra 5 125h",
"core 5 120u", "athlon 7120u", "m2"), or None."""
for rx in _PROCESSOR_RES:
m = rx.search(text or "")
if not m:
continue
parts = [g for g in m.groups() if g]
matched = m.group(0).lower()
if "ryzen" in matched:
return f"ryzen {parts[0]} {parts[1]}".lower()
if "athlon" in matched:
return f"athlon {parts[1]}".lower()
if "ultra" in matched and "core" in matched:
return f"core ultra {parts[0]} {parts[1]}".lower()
if "core" in matched and parts[0].lower().startswith("i"):
return f"{parts[0]}-{parts[1]}".lower()
if "core" in matched:
return f"core {parts[0]} {parts[1]}".lower()
return " ".join(parts).lower()
return None
_parse_processor = parse_processor
def processor_is_specific(processor: Optional[str]) -> bool:
"""True for a CPU named down to its model number ("i5-1334u"), which
together with brand, model line, RAM and storage identifies a laptop
configuration. "m2" (Apple) also counts."""
if not processor:
return False
return bool(re.search(r"\d{3,}", processor)) or bool(re.fullmatch(r"m[1-9](?: (?:pro|max|ultra))?", processor))
_MPN_RE = re.compile(
r"(?<![\w-])([A-Z0-9]{2,8}-[A-Z0-9]{2,10}(?:-[A-Z0-9]{1,6})?|"
r"[A-Z0-9]{2,6}[A-Z]{0,4}\d{2,6}[A-Z]{1,4}\d{0,4}[A-Z]{0,3}|\d{2}[A-Z]{2}\d{3,}[A-Z0-9]{2,})(?![\w-])",
re.IGNORECASE,
)
def _parse_mpn(text: str, processor: Optional[str]) -> Optional[str]:
"""A manufacturer part number such as 82XV00BHIN or fd0112TU, when the
title states one (laptops). Tokens that are specs or CPU names are not."""
candidates = []
for m in _MPN_RE.finditer(text):
tok = m.group(1)
low = tok.lower()
if len(tok) < 6 or len(tok) > 20:
continue
if sum(c.isdigit() for c in tok) < 2 or sum(c.isalpha() for c in tok) < 2:
continue
if _SPEC_TOKEN.match(low) or re.match(r"^(?:i[3579]|m[1-9]|rtx|gtx|rx|ddr|lpddr)", low):
continue
if processor and low in processor.replace("-", " ").split() + [processor.replace(" ", "")]:
continue
if re.search(r"\d+(?:gb|tb|mp|mah|hz|w)$", low):
continue
if re.search(r"-(?:core|inch|cell|bit|gen|thread)s?$|^\d+-", low) and not re.search(r"[a-z]\d", low.split("-")[-1]):
continue # "10-Core", "15-inch", "3-Cell" describe hardware, not a part number
candidates.append(tok)
return candidates[-1].upper() if candidates else None
def _parse_colour(title: str) -> Optional[str]:
# Inside brackets first: "(Onyx Black, 8GB RAM, 256GB Storage)"
for group in re.findall(r"\(([^()]*)\)", title):
for part in re.split(r"[,|/]", group):
part = part.strip()
if part and not re.search(r"\d", part) and _COLOUR_WORDS.search(part):
return part.title()
# Trailing "- Black"
m = re.search(r"[-–|,]\s*([A-Za-z][A-Za-z ]{2,30})\s*$", title)
if m and _COLOUR_WORDS.search(m.group(1)) and not re.search(r"\d", m.group(1)):
return m.group(1).strip().title()
return None
def normalise_model(model: str) -> str:
text = model.lower()
text = re.sub(r"[()\[\],|]", " ", text)
text = _MODEL_NOISE.sub(" ", text)
text = re.sub(r"\+", " plus ", text)
text = re.sub(r"[^a-z0-9 ]+", " ", text)
tokens = [t for t in text.split() if not _SPEC_TOKEN.match(t)]
return " ".join(tokens)
_LAPTOP_SPEC_START = re.compile(
r"\b(?:intel|amd|apple\s+m[1-9]|m[1-9]\s+chip|core\s+(?:i[3579]|ultra)|ryzen|snapdragon|celeron|pentium|"
r"mediatek|\d+(?:th|nd|rd|st)\s+gen|\d+(?:\.\d+)?\s*(?:-|\s)?(?:inch|cm|\"))",
re.IGNORECASE,
)
_DISPLAY_NOISE = re.compile(
r"\b(?:5g|4g|lte|smartphone|smart\s+phone|mobile\s+phone|dual\s+sim|laptop|notebook|"
r"thin\s+and\s+light|thin\s+&\s+light|gaming|new|latest)\b",
re.IGNORECASE,
)
def _model_from_title(title: str, brand: Optional[BrandMatch], category: str, colour: Optional[str],
mpn: Optional[str] = None):
"""(display model, matching model_norm) from the head of the title."""
head = re.sub(r"^\s*buy\s+", "", title, flags=re.IGNORECASE)
if mpn:
# A part number is not the model name. HP writes the line into it
# ("15-fc0500AU" is an HP 15), so that prefix is kept.
prefix = mpn.split("-", 1)[0] if "-" in mpn and len(mpn.split("-", 1)[0]) <= 4 else ""
head = re.sub(re.escape(mpn), f" {prefix} ", head, flags=re.IGNORECASE)
# A short model token in brackets right after the name is part of it:
# "Nothing Phone (2a) 5G (Black, 128 GB)".
head = re.sub(r"\((?:19|20)\d\d\)", " ", head) # "(2026)" is a model year, not part of the name
head = re.sub(r"\(([A-Za-z0-9+ ]{1,6})\)", lambda m: " " + m.group(1) + " "
if not re.search(r"\d\s*(?:gb|tb)", m.group(1), re.I) else m.group(0), head, count=1)
head = re.split(r"\s[-–|]\s|[(,|\[:]", head, maxsplit=1)[0]
if category == "laptops":
m = _LAPTOP_SPEC_START.search(head)
if m and m.start() > 0:
head = head[: m.start()]
if brand:
# Drop the parent brand's own name ("Samsung Galaxy S24" -> "Galaxy S24",
# "Apple iPhone 15" -> "iPhone 15"); a sub-brand stays ("Redmi Note 13").
from app.electronics.reference import load_reference
parent_aliases = sorted(load_reference().brands[brand.brand_slug].aliases, key=len, reverse=True)
for alias in parent_aliases:
head = re.sub(r"^\s*" + re.escape(alias) + r"\b", "", head, flags=re.IGNORECASE).strip()
head = re.sub(r"\b\d+\s*GB\s*RAM\b", " ", head, flags=re.IGNORECASE)
head = _PAIR_RE.sub(" ", head)
head = _SIZE_ANY.sub(" ", head)
if colour:
head = re.sub(re.escape(colour), " ", head, flags=re.IGNORECASE)
words = head.split()
while len(words) > 1 and _COLOUR_WORDS.fullmatch(words[-1]):
words.pop() # "iPhone 15 Black" -> "iPhone 15"
head = " ".join(words)
display = re.sub(r"\s+", " ", _DISPLAY_NOISE.sub(" ", head)).strip(" -–")
norm = normalise_model(head)
if not norm:
return None, None
return display or head.strip(), norm
def parse_title(title: str, category: str, *, expected_brand: Optional[str] = None) -> ParsedTitle:
title = re.sub(r"\s+", " ", (title or "")).strip()
brand = resolve_brand(title, expected=expected_brand)
parsed = ParsedTitle(title=title, brand=brand)
if not title:
return parsed
parsed.ram_gb, parsed.storage_gb = _parse_ram_storage(title, category)
parsed.colour = _parse_colour(title)
if re.search(r"\b5G\b", title, re.IGNORECASE):
parsed.network = "5G"
if category == "laptops":
parsed.processor = _parse_processor(title)
parsed.mpn = _parse_mpn(title, parsed.processor)
parsed.model, parsed.model_norm = _model_from_title(title, brand, category, parsed.colour, parsed.mpn)
return parsed
def variant_key(parsed: ParsedTitle, category: str) -> Optional[str]:
"""The identity of one real-world variant, or None if the title does not
state enough to tell variants apart."""
if not parsed.brand or not parsed.model_norm:
return None
b = parsed.brand.brand_slug
fmt = lambda d: "na" if d is None else format(d.normalize(), "f") # noqa: E731
if category == "laptops":
# A laptop configuration is its model line + CPU + RAM + storage. That
# is what every site states (a part number is shown by only a few), so
# it is the key whenever it is complete; the MPN is the fallback.
line = laptop_line(parsed.model_norm)
if line and processor_is_specific(parsed.processor) and parsed.ram_gb and parsed.storage_gb:
return f"{b}|laptops|{line}|{parsed.processor}|{fmt(parsed.ram_gb)}|{fmt(parsed.storage_gb)}"
if parsed.mpn:
return f"{b}|laptops|mpn:{parsed.mpn.lower()}"
return None
if parsed.storage_gb is None:
return None
return f"{b}|{category}|{parsed.model_norm}|{fmt(parsed.ram_gb)}|{fmt(parsed.storage_gb)}"
# Lenovo/Asus machine-type codes ("15amn8", "15irh10", "14iah8", "x1504za")
# name a chassis generation, and one site prints them where another does not.
_MACHINE_CODE = re.compile(r"^(?:\d{2}[a-z]{2,4}\d{1,2}|[a-z]\d{4}[a-z]{1,3})$")
def laptop_line(model_norm: Optional[str]) -> str:
"""The model line used for matching: "ideapad slim 3 15amn8" -> "ideapad slim 3"."""
tokens = [t for t in (model_norm or "").split() if not _MACHINE_CODE.match(t)]
return " ".join(tokens)
def fill_from_context(parsed: ParsedTitle, category: str, *, snippet: str = "",
spec_texts: tuple = ()) -> ParsedTitle:
"""Fill variant fields a (often truncated) title leaves out, from text the
same site published about the same page: its search snippet, or the spec
table of the fetched page. Only unambiguous values are taken - a snippet
naming two different storage sizes is describing several variants."""
if snippet and (parsed.ram_gb is None or parsed.storage_gb is None):
sizes = {_dec(v, u) for v, u in _SIZE_ANY.findall(snippet)}
if len(sizes) <= 2:
ram, storage = _parse_ram_storage(snippet, category)
if parsed.storage_gb is None and storage is not None:
parsed.storage_gb = storage
if parsed.ram_gb is None and ram is not None and ram != parsed.storage_gb:
parsed.ram_gb = ram
if category == "laptops" and not processor_is_specific(parsed.processor):
# A title that already names a CPU family ("Snapdragon X", "Core i7")
# is only completed from the page's own spec table, never from a
# snippet - snippets often run several products' titles together.
sources = list(spec_texts) + ([snippet] if parsed.processor is None and snippet else [])
found = set()
for text in sources:
found |= {cpu for cpu in all_processors(text) if processor_is_specific(cpu)}
if len(found) == 1:
parsed.processor = found.pop()
return parsed
def all_processors(text: str) -> set:
"""Every CPU named anywhere in `text` (a snippet can name several)."""
found = set()
for rx in _PROCESSOR_RES:
for m in rx.finditer(text or ""):
cpu = parse_processor(m.group(0))
if cpu:
found.add(cpu)
return found