from __future__ import annotations from typing import List, Optional, TYPE_CHECKING from app.infrastructure.settings import EMBEDDINGS_MODEL if TYPE_CHECKING: # pragma: no cover - typing only, no runtime cost from sentence_transformers import SentenceTransformer _model_singleton: Optional["SentenceTransformer"] = None def get_device() -> str: """Prefer CUDA if available, otherwise CPU. Imports torch lazily: on an 8GB RAM / CPU-only laptop there is no benefit to importing torch (and paying its startup/memory cost) until an embedding is actually requested, so the FastAPI process can boot and answer /api/health almost instantly. """ import torch # local import - see docstring return "cuda" if torch.cuda.is_available() else "cpu" def get_embedding_model() -> "SentenceTransformer": global _model_singleton if _model_singleton is None: from sentence_transformers import SentenceTransformer # local import - see get_device() device = get_device() _model_singleton = SentenceTransformer(EMBEDDINGS_MODEL, device=device) return _model_singleton def embed_texts(texts: List[str]) -> List[List[float]]: """Embed a batch of texts into normalized 384-dim vectors (MiniLM-L6-v2). Normalized so that pgvector's cosine-distance operator (`<=>`) behaves consistently for the RAG retrieval step in `app.services.vector_store`. """ if not texts: return [] model = get_embedding_model() embeddings = model.encode( texts, batch_size=32, normalize_embeddings=True, convert_to_numpy=True, show_progress_bar=False, ) return embeddings.tolist()