imag vector generation with dimentionality reduction

This commit is contained in:
sriram
2026-09-18 15:26:25 +05:30
parent afa0bfa743
commit d6296bd1f0
16 changed files with 1724 additions and 48 deletions

View File

@@ -0,0 +1,184 @@
#!/usr/bin/env python3
"""
Build `mobilenet_v3_small_embedder.tflite` - the image model behind img_vector.
WHY THIS EXISTS
---------------
The Nearle Flutter app embeds product photos with a MobileNetV3-Small TFLite
model whose output is the 1024-wide penultimate layer (input [1,224,224,3]
float32 in 0..1, output [1,1024], L2-normalised afterwards). That file is a
hand-made export, not a package: nothing installs it. When the app's own copy
is not to hand, this script produces an equivalent one from Keras' pretrained
ImageNet weights, so the catalog can be vectorised at all.
WHAT IT BUILDS
--------------
Input(224, 224, 3) float32, 0..1 - the app divides by 255 and
-> Rescaling(scale=2, offset=-1) nothing else; MobileNetV3 wants -1..1, so
-> MobileNetV3Small backbone the rescale lives INSIDE the graph
(imagenet weights, include_top=True,
include_preprocessing=False)
-> "Conv_2" 1x1 conv (576 -> 1024) + hard-swish <- the embedding
-> Flatten [1, 1024]
`Conv_2` is the only 1024-wide layer in MobileNetV3-Small, so it is the layer
any "1024-d MobileNetV3-Small embedder" taps. Dropout and the 1000-way Logits
layer after it are discarded. No normalisation in the graph - the app
normalises after inference, and so does app/services/image_embedder.py.
IS IT THE SAME AS THE APP'S FILE?
---------------------------------
Same architecture, same pretrained weights, same tap, same input convention.
Whether the numbers agree to the last decimal depends on how the app's copy
was exported (TF version, converter flags), and that can only be checked
against the app itself: take one cropped photo from the app, note the first
eight values of its `[VECTOR][IMAGE]` log line, run
`image_embedder.embedding_for_bytes()` on the same file, and compare. Agreement
to 3-4 decimals means the two systems' vectors are interchangeable. If they
differ, catalog<->catalog search still works (the vectors are self-consistent);
drop the app's real file over this one and run
`python -m scripts.backfill_image_vectors --all --force --apply`.
HOW TO RUN IT
-------------
TensorFlow is NOT a dependency of this project and must not become one (the
container is memory-capped and already carries torch). Run the export in a
throwaway container, from backend/:
docker run --rm -v "${PWD}:/work" -w /work python:3.11-slim sh -c \\
"pip install -q tensorflow-cpu && python scripts/export_mobilenet_embedder.py \\
--out app/services/models/mobilenet/mobilenet_v3_small_embedder.tflite"
Or, without Docker, in a throwaway venv at a SHORT path (TensorFlow's wheel
trips Windows' path-length limit inside a deep temp directory; TF has no
wheel for Python 3.14, so use 3.13 or 3.11):
py -3.13 -m venv C:/tfexport_venv
C:/tfexport_venv/Scripts/pip install "tensorflow==2.21.*" "ai-edge-litert>=2.2.0"
C:/tfexport_venv/Scripts/python scripts/export_mobilenet_embedder.py --out ...
The committed file was produced this way on 2026-09-17 with TensorFlow 2.21.0
/ Keras 3.15.1. The pretrained weights are fixed, so re-running produces the
same model. The script self-checks the result with the same runtime the
backend uses (ai-edge-litert) when it is importable, else with tf.lite.
"""
from __future__ import annotations
import argparse
import sys
import tempfile
from pathlib import Path
INPUT_SIZE = 224
EMBED_DIM = 1024
TAP_LAYER = "Conv_2" # Keras' name for the 1x1 conv that widens 576 -> 1024
def build_embedder():
"""The Keras model described in the module docstring."""
import tensorflow as tf
from tensorflow import keras
base = keras.applications.MobileNetV3Small(
input_shape=(INPUT_SIZE, INPUT_SIZE, 3),
weights="imagenet",
include_top=True, # we need the head's Conv_2, which include_top=False drops
include_preprocessing=False, # the 0..1 -> -1..1 rescale is added explicitly below
)
# Keras 2 named it "Conv_2", Keras 3 "conv_2"; match case-insensitively.
names = [layer.name for layer in base.layers]
try:
idx = [n.lower() for n in names].index(TAP_LAYER.lower())
except ValueError:
raise SystemExit(f"no layer named {TAP_LAYER} in MobileNetV3Small; layers: {names}")
conv = base.layers[idx]
# The layer right after it is the hard-swish activation whose output is
# the embedding (then come dropout and the 1000-way logits, discarded).
act = base.layers[idx + 1]
if int(conv.output.shape[-1]) != EMBED_DIM or int(act.output.shape[-1]) != EMBED_DIM:
raise SystemExit(f"{conv.name}/{act.name} are not {EMBED_DIM} wide: "
f"{conv.output.shape[-1]}/{act.output.shape[-1]}")
if "activation" not in act.name.lower() and "swish" not in act.name.lower():
raise SystemExit(f"layer after {conv.name} is {act.name}, expected the hard-swish activation")
inputs = keras.Input(shape=(INPUT_SIZE, INPUT_SIZE, 3), dtype="float32", name="image_0_1")
x = keras.layers.Rescaling(scale=2.0, offset=-1.0, name="rescale_0_1_to_pm1")(inputs)
features = keras.Model(base.input, act.output, name="mobilenet_v3_small_features")(x)
outputs = keras.layers.Flatten(name="embedding")(features)
model = keras.Model(inputs, outputs, name="mobilenet_v3_small_embedder")
if tuple(model.output.shape) != (None, EMBED_DIM):
raise SystemExit(f"unexpected output shape {model.output.shape}")
return model
def convert_to_tflite(model) -> bytes:
import tensorflow as tf
with tempfile.TemporaryDirectory() as tmp:
saved = Path(tmp) / "saved_model"
# Keras 3 exports a SavedModel via .export(); Keras 2 via tf.saved_model.save.
if hasattr(model, "export"):
model.export(str(saved))
else:
tf.saved_model.save(model, str(saved))
converter = tf.lite.TFLiteConverter.from_saved_model(str(saved))
# Float32, builtin ops only, no quantisation: the backend runtime is
# plain ai-edge-litert and must not need SELECT_TF_OPS.
converter.target_spec.supported_ops = [tf.lite.OpsSet.TFLITE_BUILTINS]
return converter.convert()
def self_check(path: Path) -> None:
import numpy as np
try:
from ai_edge_litert.interpreter import Interpreter
runtime = "ai-edge-litert"
except ImportError:
import tensorflow as tf
Interpreter = tf.lite.Interpreter
runtime = "tf.lite"
it = Interpreter(model_path=str(path))
it.allocate_tensors()
inp, out = it.get_input_details(), it.get_output_details()
assert len(inp) == 1 and list(inp[0]["shape"]) == [1, INPUT_SIZE, INPUT_SIZE, 3], inp
assert np.dtype(inp[0]["dtype"]) == np.float32, inp[0]["dtype"]
assert len(out) == 1 and list(out[0]["shape"]) == [1, EMBED_DIM], out
rng = np.random.default_rng(0)
for label, x in (
("zeros", np.zeros((1, INPUT_SIZE, INPUT_SIZE, 3), np.float32)),
("random", rng.random((1, INPUT_SIZE, INPUT_SIZE, 3), dtype=np.float32)),
):
it.set_tensor(inp[0]["index"], x)
it.invoke()
v = it.get_tensor(out[0]["index"])[0]
norm = float(np.linalg.norm(v))
assert np.all(np.isfinite(v)) and norm > 0, label
print(f" {label:6s} first 8: {np.round(v[:8], 5).tolist()} norm={norm:.4f} "
f"zeros={int((v == 0).sum())}/{EMBED_DIM}")
print(f" self-check passed with {runtime}: input [1,{INPUT_SIZE},{INPUT_SIZE},3] float32, output [1,{EMBED_DIM}]")
def main() -> int:
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("--out", required=True, help="where to write the .tflite")
args = ap.parse_args()
out = Path(args.out)
out.parent.mkdir(parents=True, exist_ok=True)
import tensorflow as tf
print(f"tensorflow {tf.__version__}, keras {tf.keras.__version__ if hasattr(tf.keras, '__version__') else '?'}")
model = build_embedder()
print(f"model: {model.name}, params={model.count_params():,}, tap={TAP_LAYER}+hard-swish")
data = convert_to_tflite(model)
out.write_bytes(data)
print(f"wrote {out} ({len(data) / 1e6:.1f} MB)")
self_check(out)
return 0
if __name__ == "__main__":
sys.exit(main())