"""Fuzzy ingredient→grocery_item matcher. Ingredient-centric: for each ingredient, finds the best-matching grocery item. Scores combine partial_token_sort_ratio with a precision term (ingredient sig-words / grocery sig-words) so long branded product names that contain an ingredient word incidentally rank lower than items whose primary purpose IS that ingredient. Manual matches (source='manual') are preserved across runs. """ from __future__ import annotations import re import uuid as _uuid_mod from dataclasses import dataclass from decimal import Decimal from uuid import UUID from rapidfuzz import fuzz, process from sqlalchemy.dialects.postgresql import insert as _pg_insert from sqlalchemy.orm import Session from app.models import ( GroceryItem, Ingredient, IngredientGroceryMatch, IngredientMatchSource, ) # Words that describe quantity/preparation but don't identify the ingredient itself. # Filtering these from sig-word sets keeps "Chicken Thighs, Boneless Skinless" # from requiring "boneless" to appear in the grocery name. _STOP_WORDS = frozenset({ "fresh", "organic", "whole", "large", "small", "medium", "low", "free", "light", "dark", "raw", "dried", "frozen", "canned", "extra", "virgin", "pure", "natural", "classic", "style", "boneless", "skinless", "lean", "grain", "long", "jarred", "roasted", "smoked", "cooked", "and", "with", "for", "the", }) @dataclass class MatchResult: ingredient_id: UUID grocery_item_id: UUID confidence: float def _sig_words(text: str) -> frozenset: """Lowercase alpha tokens >2 chars, stop-words removed.""" tokens = re.sub(r"[^a-z ]", " ", text.lower()).split() return frozenset(t for t in tokens if len(t) > 2 and t not in _STOP_WORDS) def run_match_job( db: Session, *, source_filter: str = "lucky_california", threshold: float = 0.82, ) -> int: """Refresh AUTO ingredient_grocery_match rows for grocery items from `source_filter`. For each ingredient the scorer is: combined = partial_token_sort_ratio × (overlap / grocery_sig_count) where overlap = ingredient sig-words that appear in the grocery sig-words. This means a product like "Milton's Olive Oil Crackers" (6 sig-words) scores half of "Bertolli Olive Oil" (3 sig-words) for the ingredient "Olive Oil", so the simpler/more specific product wins. 100% recall is required: every significant ingredient word must appear in the grocery name. This eliminates cross-category noise such as "Ginger, Fresh" → "Pampers Complete Clean Baby Fresh Scent Wipes". Manual matches (source='manual') are NOT touched. Returns number of AUTO rows written. """ grocery_rows = ( db.query(GroceryItem) .filter(GroceryItem.source == source_filter) .all() ) if not grocery_rows: return 0 grocery_names_lower = [gi.name.lower() for gi in grocery_rows] grocery_sig = [_sig_words(gi.name) for gi in grocery_rows] grocery_ids = [gi.id for gi in grocery_rows] # Purge existing AUTO matches for this source's items before re-matching. db.query(IngredientGroceryMatch).filter( IngredientGroceryMatch.grocery_item_id.in_([gi.id for gi in grocery_rows]), IngredientGroceryMatch.source == IngredientMatchSource.AUTO, ).delete(synchronize_session=False) db.flush() ingredients = db.query(Ingredient).all() written = 0 for ingredient in ingredients: ing_sig = _sig_words(ingredient.name) if not ing_sig: continue # Include aliases as additional query variants. queries = [ingredient.name.lower()] + [ a.lower() for a in (ingredient.aliases or []) if a ] best_idx: int | None = None best_combined = 0.0 for query in queries: results = process.extract( query, grocery_names_lower, scorer=fuzz.partial_token_sort_ratio, limit=20, ) for _text, score, idx in results: if score < threshold * 100: continue gsig = grocery_sig[idx] if not gsig: continue # 100% recall: every ingredient sig-word must appear in the grocery name. if not ing_sig.issubset(gsig): continue # Precision penalises grocery items with many extra words. precision = len(ing_sig) / len(gsig) combined = (score / 100.0) * precision if combined > best_combined: best_combined = combined best_idx = idx if best_idx is None: continue # ON CONFLICT DO NOTHING preserves any existing MANUAL match for the same pair. stmt = ( _pg_insert(IngredientGroceryMatch.__table__) .values( id=_uuid_mod.uuid4(), ingredient_id=ingredient.id, grocery_item_id=grocery_ids[best_idx], confidence=Decimal(str(round(best_combined, 3))), source=IngredientMatchSource.AUTO, ) .on_conflict_do_nothing(index_elements=["ingredient_id", "grocery_item_id"]) ) db.execute(stmt) written += 1 db.commit() return written