Image capturing Flow updates

This commit is contained in:
sriram
2026-09-24 17:18:38 +05:30
parent f933ea10a1
commit fe208f4715
14 changed files with 1877 additions and 6 deletions

View File

@@ -219,6 +219,28 @@ def test_unknown_brand_passes_through_unchanged() -> None:
assert resolve_parent_brand("Totally Made Up Brand") == "Totally Made Up Brand"
@pytest.mark.parametrize("brand,parent", [
("Godrej Fab", "godrej"),
("GODREJ FAB", "godrej"),
("Godrej Ezee", "godrej"),
("Amul Taaza", "amul"),
("Hindustan Unilever Rin", "hindustan unilever"),
# The prefix inherits the parent's own answer, quirks included.
("Sunfeast Farmlite", "itc"),
])
def test_parent_followed_by_an_unregistered_line_joins_the_parent(brand: str, parent: str) -> None:
"""The field-capture case: a label reading "Godrej Fab" must write into
brand_godrej, not build a brand_godrej_fab table beside it."""
assert resolve_parent_brand(brand) == parent
def test_leading_brand_rule_never_uses_a_fuzzy_word() -> None:
""""fab" is only known as part of the alias "parle fab". Accepting it as a
leading brand would send another company's detergent to Parle."""
assert resolve_parent_brand("Fab Detergent Powder") == "Fab Detergent Powder"
assert resolve_parent_brand("Sun Pharma") == "Sun Pharma"
def test_new_brand_round_trips_to_its_own_table() -> None:
"""ITC is the worked example: it must own brand_itc, not merge elsewhere."""
assert resolve_parent_brand("ITC") == "itc"

View File

@@ -0,0 +1,331 @@
"""Capture-to-catalog: a photographed product the catalog does not have.
No database, no network, no worker thread: the DB reads, the pipeline and the
post-pass are patched, and jobs are written under a tmp CAPTURE_DIR. The
pipeline's own stages are covered by the store-catalog tests; here the
assertions are about the decisions this module makes around it.
"""
from __future__ import annotations
import io
import math
from typing import Any, Dict, List
import pytest
from app.api.routers import search as search_router
from app.core.store_catalog_pipeline import PipelineResult
from app.services import capture_discovery as cd
from app.services.image_match import ImageSearchResult
from app.services.product_identify import IdentifyResult
GODREJ_FAB_LABEL = ("NEW Godrej fab Detergent Powder Superior cleaning 1kg MRP Rs.110 "
"(incl of all taxes) www.godrejfab.com Mfd by Godrej Consumer Products Ltd")
# ---------------------------------------------------------------------------
# Fixtures
# ---------------------------------------------------------------------------
@pytest.fixture(autouse=True)
def isolated(tmp_path, monkeypatch):
"""Jobs on tmp, no worker thread, clean in-process state, DB reads patched."""
monkeypatch.setattr(cd.settings, "CAPTURE_DIR", tmp_path / "captures")
monkeypatch.setattr(cd.settings, "CAPTURE_MAX_PER_CLIENT_PER_HOUR", 20)
monkeypatch.setattr(cd.settings, "CAPTURE_PUBLIC_BASE_URL", "")
monkeypatch.setattr(cd, "_ensure_worker", lambda: None)
monkeypatch.setattr(cd, "_jobs", {})
monkeypatch.setattr(cd, "_live_ids", set())
monkeypatch.setattr(cd, "_inflight", {})
monkeypatch.setattr(cd, "_starts", {})
import queue as _queue
monkeypatch.setattr(cd, "_queue", _queue.Queue(maxsize=8))
monkeypatch.setattr(cd, "_brand_table_exists", lambda parent: False)
import app.services.vector_store as vs
monkeypatch.setattr(vs, "get_product_by_image_id", lambda brand, image_id: None)
def _miss(label: str = GODREJ_FAB_LABEL, **kw) -> cd.CaptureOutcome:
args = dict(label_text=label, image_bytes=b"\xff\xd8\xff\xe0fakejpeg", vector=None,
brand=None, category=None, client="10.0.0.1")
args.update(kw)
return cd.handle_miss(**args)
# ---------------------------------------------------------------------------
# parse_label
# ---------------------------------------------------------------------------
def test_godrej_fab_label_becomes_a_godrej_detergent() -> None:
p = cd.parse_label(GODREJ_FAB_LABEL)
assert p.usable
assert (p.brand, p.parent) == ("Godrej", "godrej")
assert p.product_name == "Godrej Fab Detergent Powder" # marketing copy trimmed
assert p.size == "1kg"
assert p.category == "Detergents & Fabric Care"
assert (p.hsn_code, p.gst_percent, p.hsn_gst_needs_review) == ("3402", 18, False)
def test_sub_brand_printed_alone_reaches_its_parent() -> None:
p = cd.parse_label("Surf excel easy wash detergent powder 1 kg")
assert p.parent == "hindustan unilever"
assert p.brand == "Surf Excel"
assert p.size == "1kg"
def test_a_brand_that_is_also_a_category_word_is_not_cut_short() -> None:
p = cd.parse_label("Dairy Milk Silk 60g")
assert p.product_name == "Dairy Milk Silk"
assert p.parent == "cadbury"
assert p.category == "Chocolates"
def test_bare_line_name_is_never_given_a_brand() -> None:
""""Fab" alone is Parle's biscuit line in the registry; a detergent label
without its brand must ask, not guess."""
p = cd.parse_label("fab detergent powder 500 g")
assert not p.usable
assert "brand" in p.problem.lower()
def test_brand_hint_completes_a_label_without_the_brand() -> None:
p = cd.parse_label("fab detergent powder 500 g", brand_hint="Godrej")
assert p.usable
assert p.product_name == "Godrej Fab Detergent Powder"
assert p.size == "500g"
@pytest.mark.parametrize("text", ["", " ", None])
def test_unreadable_label_asks_for_input(text) -> None:
p = cd.parse_label(text)
assert not p.usable
assert "type the product name" in p.problem.lower()
def test_brand_only_label_asks_for_the_product_name() -> None:
p = cd.parse_label("Godrej 1kg")
assert not p.usable
assert "product name" in p.problem.lower()
def test_expected_image_id_matches_the_pipeline_key() -> None:
p = cd.parse_label(GODREJ_FAB_LABEL)
assert cd.expected_image_id(p) == "godrej_godrej_fab_detergent_powder_1kg"
# ---------------------------------------------------------------------------
# Confirmation rule and brand gate
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("matched_by,reason,confirmed", [
("text", "image_below_threshold", True),
("image_vector", None, True),
("image_vector", "text_no_match", False),
("none", "ocr_empty", False),
])
def test_is_confirmed_follows_the_documented_client_rule(matched_by, reason, confirmed) -> None:
assert cd.is_confirmed(matched_by, reason) is confirmed
def test_known_brand_is_a_registry_parent_or_an_existing_table() -> None:
assert cd.is_known_brand("godrej", table_exists=lambda p: False)
assert cd.is_known_brand("idhayam", table_exists=lambda p: True)
assert not cd.is_known_brand("Xyzzy Foods", table_exists=lambda p: False)
assert not cd.is_known_brand("", table_exists=lambda p: True)
# ---------------------------------------------------------------------------
# handle_miss
# ---------------------------------------------------------------------------
def test_miss_queues_a_job_with_a_provisional_card() -> None:
out = _miss()
assert out.status == cd.PENDING
assert out.job_id and len(out.job_id) == 32
assert out.provisional["product_name"] == "Godrej Fab Detergent Powder"
assert out.provisional["source"] == "label"
job = cd.get_job(out.job_id)
assert job.status == cd.QUEUED
assert job.photo_ext == "jpg"
assert cd.photo_path(f"{job.job_id}.jpg") is not None
def test_same_product_twice_reuses_the_running_job() -> None:
first = _miss()
second = _miss(client="10.0.0.2")
assert second.status == cd.PENDING
assert second.job_id == first.job_id
def test_stored_product_is_returned_not_rediscovered(monkeypatch) -> None:
import app.services.vector_store as vs
row = {"image_id": "godrej_godrej_fab_detergent_powder_1kg", "product_name": "Godrej Fab 1kg"}
monkeypatch.setattr(vs, "get_product_by_image_id",
lambda brand, image_id: row if image_id == row["image_id"] else None)
out = _miss()
assert out.status == cd.EXISTS
assert out.existing is row
assert out.job_id is None
def test_unknown_brand_never_mints_a_table() -> None:
out = _miss(label="Xyzzy Soap 100g", brand="Xyzzy")
assert out.status == cd.NEEDS_INPUT
assert out.job_id is None
assert "not a brand in the catalog" in out.message
def test_unreadable_label_is_needs_input_never_not_found() -> None:
out = _miss(label=None)
assert out.status == cd.NEEDS_INPUT
assert "not found" not in (out.message or "").lower()
def test_rate_limit_answers_busy_with_the_card(monkeypatch) -> None:
monkeypatch.setattr(cd.settings, "CAPTURE_MAX_PER_CLIENT_PER_HOUR", 1)
assert _miss().status == cd.PENDING
out = _miss(label="Amul Taaza Toned Milk 500 ml")
assert out.status == cd.BUSY
assert out.provisional["brand"] == "Amul"
def test_discovery_failure_never_breaks_identify(monkeypatch) -> None:
monkeypatch.setattr(cd, "parse_label", lambda *a, **k: 1 / 0)
out = _miss()
assert out.status == cd.NEEDS_INPUT
def test_restart_marks_an_orphaned_job_interrupted(monkeypatch) -> None:
out = _miss()
monkeypatch.setattr(cd, "_live_ids", set())
monkeypatch.setattr(cd, "_jobs", {})
assert cd.get_job(out.job_id).status == cd.INTERRUPTED
@pytest.mark.parametrize("name", ["../etc/passwd", "abc.jpg", "0" * 32 + ".exe", ""])
def test_photo_path_rejects_anything_but_a_job_photo(name) -> None:
assert cd.photo_path(name) is None
# ---------------------------------------------------------------------------
# run_job
# ---------------------------------------------------------------------------
def _queued_job() -> cd.CaptureJob:
return cd.get_job(_miss().job_id)
def _result(disposition: str = "inserted") -> PipelineResult:
r = PipelineResult()
r.products = [{"image_id": "godrej_godrej_fab_detergent_powder_1kg", "disposition": disposition}]
return r
@pytest.fixture
def pipeline(monkeypatch):
from app.core import store_catalog_pipeline as p
seen: Dict[str, Any] = {"marked": []}
def fake_run(filename, content, **kw):
seen["csv"] = content.decode("utf-8")
seen["kw"] = kw
return seen["result"]
seen["result"] = _result()
monkeypatch.setattr(p, "run_pipeline", fake_run)
monkeypatch.setattr(cd, "_retail_check", lambda job: {"status": "found"})
monkeypatch.setattr(cd, "mark_captured_row",
lambda job, url: seen["marked"].append((job.image_id, url)) or False)
return seen
def test_inserted_product_is_marked_for_review(pipeline) -> None:
job = cd.run_job(_queued_job())
assert job.status == cd.DONE
assert job.disposition == "inserted"
assert job.retail_presence == {"status": "found"}
assert pipeline["marked"] == [("godrej_godrej_fab_detergent_powder_1kg", None)]
assert "Godrej Fab Detergent Powder" in pipeline["csv"]
assert "Detergents & Fabric Care" in pipeline["csv"]
assert pipeline["kw"]["fetch_images"] is True
def test_existing_row_is_never_downgraded(pipeline) -> None:
pipeline["result"] = _result("backfilled")
job = cd.run_job(_queued_job())
assert job.status == cd.DONE
assert pipeline["marked"] == []
def test_rejected_product_reports_why(pipeline) -> None:
r = PipelineResult()
r.rejections = [{"reason": "title does not name a product"}]
pipeline["result"] = r
job = cd.run_job(_queued_job())
assert job.status == cd.REJECTED
assert "title does not name a product" in job.detail
def test_storage_error_fails_the_job(pipeline) -> None:
r = _result()
r.storage_error = "connection refused"
pipeline["result"] = r
job = cd.run_job(_queued_job())
assert job.status == cd.FAILED
def test_photo_url_offered_only_with_a_public_base(pipeline, monkeypatch) -> None:
monkeypatch.setattr(cd.settings, "CAPTURE_PUBLIC_BASE_URL", "https://api.example.com/")
job = cd.run_job(_queued_job())
url = pipeline["marked"][0][1]
assert url == f"https://api.example.com/api/search/captures/{job.job_id}.jpg"
# ---------------------------------------------------------------------------
# HTTP
# ---------------------------------------------------------------------------
def _unit():
v = [math.cos(i / 7.0) for i in range(1024)]
n = math.sqrt(sum(x * x for x in v))
return [x / n for x in v]
@pytest.fixture
def unconfirmed(monkeypatch):
monkeypatch.setattr(search_router.image_embedder, "available", lambda: True)
monkeypatch.setattr(search_router.image_embedder, "embedding_for_bytes", lambda data: _unit())
rival = {"image_id": "hul_surf_excel_1kg", "product_name": "Surf Excel 1kg", "brand": "Hindustan Unilever",
"category": "Detergents & Fabric Care", "score": 0.52, "text_overlap": 0.0}
result = IdentifyResult(search=ImageSearchResult(rows=[rival], top_k=10), matched_by="image_vector",
ocr_text=GODREJ_FAB_LABEL, ocr_source="server", image_top_score=0.52,
fallback_reason="text_no_match")
monkeypatch.setattr(search_router, "identify_product", lambda **kw: result)
def _post(client):
return client.post("/api/search/identify",
files={"file": ("p.jpg", io.BytesIO(b"\xff\xd8\xff\xe0x"), "image/jpeg")})
def test_identify_with_discovery_off_is_unchanged(client, unconfirmed, monkeypatch) -> None:
monkeypatch.setattr(search_router.settings, "ENABLE_CAPTURE_DISCOVERY", False)
body = _post(client).json()
assert body["discovery_status"] is None
assert body["matched_by"] == "image_vector"
assert body["total"] == 1
def test_identify_miss_starts_discovery(client, unconfirmed, monkeypatch) -> None:
monkeypatch.setattr(search_router.settings, "ENABLE_CAPTURE_DISCOVERY", True)
body = _post(client).json()
assert body["discovery_status"] == "pending"
assert body["matched_by"] == "discovery_pending"
assert body["results"] == [] # the rival detergent is not offered
assert body["provisional"]["product_name"] == "Godrej Fab Detergent Powder"
job = client.get(f"/api/search/identify/jobs/{body['discovery_job_id']}").json()
assert job["status"] == "queued"
assert job["provisional"]["brand"] == "Godrej"
def test_job_endpoint_404s_for_an_unknown_id(client) -> None:
assert client.get("/api/search/identify/jobs/" + "0" * 32).status_code == 404
assert client.get("/api/search/identify/jobs/not-an-id").status_code == 404
def test_admin_capture_list_requires_admin(client) -> None:
assert client.get("/api/admin/captures").status_code in (401, 403)

View File

@@ -0,0 +1,49 @@
"""
Laundry products resolve to "Detergents & Fabric Care", the name the rest of
the system already uses (title_validator, HSN_GST_TABLE, consumability).
The registry used to put detergents under "Household Cleaning", whose HSN row
carries needs_review=True, so every detergent was flagged for review over a
naming mismatch rather than any real doubt about its tax code.
No database, no network.
"""
from __future__ import annotations
import pytest
from app.services.category_registry import ALL_CATEGORIES, detect_category_from_text
from app.services.enrichment.hsn_gst.models import resolve_hsn_gst
@pytest.mark.parametrize("title", [
"Godrej Fab Detergent Powder 1kg",
"Surf Excel Matic Liquid Laundry Detergent",
"Tide Washing Powder",
"Rin Detergent Bar",
])
def test_laundry_resolves_to_detergents_with_a_clean_hsn(title: str) -> None:
category = detect_category_from_text(title)
assert category == "Detergents & Fabric Care"
info = resolve_hsn_gst(category, title)
assert (info.hsn_code, info.gst_percent, info.hsn_gst_needs_review) == ("3402", 18, False)
@pytest.mark.parametrize("title", ["Vim Dishwash Bar", "Lizol Floor Cleaner", "Dettol Handwash"])
def test_other_cleaners_stay_household_cleaning(title: str) -> None:
assert detect_category_from_text(title) == "Household Cleaning"
def test_fab_is_not_a_laundry_keyword() -> None:
"""Parle Fab is a biscuit. A bare "Godrej Fab" gets its category from the
label text, not from the product-line name."""
assert detect_category_from_text("Parle Fab Biscuits") == "Biscuits & Cookies"
assert detect_category_from_text("Godrej Fab") is None
def test_face_wash_is_not_laundry() -> None:
assert detect_category_from_text("Nivea Face Wash") == "Skin & Bath Care"
def test_category_names_are_unique() -> None:
assert len(ALL_CATEGORIES) == len(set(ALL_CATEGORIES))

View File

@@ -46,6 +46,9 @@ def _row(score: float = 0.631, image_id: str = "britannia_marie_gold_300g") -> D
IMAGE_SEARCH_KEYS = {"results", "total", "detected_brand", "scoped_to_brand", "scope_fallback",
"min_score", "top_k", "query_text"}
IDENTIFY_KEYS = IMAGE_SEARCH_KEYS | {"matched_by", "ocr_text", "ocr_source", "image_top_score", "fallback_reason"}
# Capture-to-catalog fields: always present, None unless discovery ran (see
# tests/test_capture_discovery.py). Additive - no existing key changed.
IDENTIFY_KEYS |= {"discovery_status", "discovery_job_id", "discovery_message", "provisional"}
def _post_identify(client, data: bytes, **form):