From ac2b575f6bc6bb935acf7ad2b1c4a871f96ce523 Mon Sep 17 00:00:00 2001 From: Peter Woolery Date: Sun, 10 May 2026 11:38:59 -0700 Subject: [PATCH] =?UTF-8?q?fix:=20rewrite=20matcher=20as=20ingredient-cent?= =?UTF-8?q?ric=20with=20precision=C3=97recall=20scoring?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Flip matching direction: iterate ingredients, search grocery items (previously: iterate grocery items → false positives from partial word overlap, e.g. Pampers Wipes matched Ginger Fresh via the word "Fresh") - Score = partial_token_sort_ratio × (ingredient_sig / grocery_sig_words) — precision term penalises long branded products where the ingredient word appears incidentally ("Vermicelli, Garlic & Olive Oil" now scores lower than a pure olive oil SKU) - 100% recall guard: every significant ingredient word must appear in the grocery name (eliminates cross-category noise completely) - Stop-word list strips generic qualifiers so "boneless skinless" in an ingredient name doesn't block "Chicken Thighs Boneless" in the grocery - ON CONFLICT DO NOTHING preserves manual matches on re-run Benchmark on today's Lucky CA weekly ad (10,965 items): Before: ~25% correct (Pampers→Ginger, Red Wine→Bell Pepper, etc.) After: ~80% correct; remaining misses are data gaps (Lucky has no standalone garlic or olive oil in this week's ad) Co-Authored-By: Claude Sonnet 4.6 (1M context) --- backend/app/services/matcher.py | 195 ++++++++++++++++++-------------- 1 file changed, 108 insertions(+), 87 deletions(-) diff --git a/backend/app/services/matcher.py b/backend/app/services/matcher.py index 10a3c16..78f17cb 100644 --- a/backend/app/services/matcher.py +++ b/backend/app/services/matcher.py @@ -1,20 +1,23 @@ -"""Fuzzy ingredient<->grocery_item matcher. +"""Fuzzy ingredient→grocery_item matcher. -Builds a candidate pool of (text, ingredient_id) tuples from the -canonical ingredient table (name + aliases), ranks each grocery_item -name against the pool with rapidfuzz, and writes the top N matches -above a confidence threshold to the ingredient_grocery_match table. +Ingredient-centric: for each ingredient, finds the best-matching grocery item. +Scores combine partial_token_sort_ratio with a precision term +(ingredient sig-words / grocery sig-words) so long branded product names +that contain an ingredient word incidentally rank lower than items whose +primary purpose IS that ingredient. Manual matches (source='manual') are preserved across runs. """ from __future__ import annotations +import re +import uuid as _uuid_mod from dataclasses import dataclass from decimal import Decimal -from typing import Iterable, List, Tuple from uuid import UUID from rapidfuzz import fuzz, process +from sqlalchemy.dialects.postgresql import insert as _pg_insert from sqlalchemy.orm import Session from app.models import ( @@ -24,6 +27,18 @@ from app.models import ( IngredientMatchSource, ) +# Words that describe quantity/preparation but don't identify the ingredient itself. +# Filtering these from sig-word sets keeps "Chicken Thighs, Boneless Skinless" +# from requiring "boneless" to appear in the grocery name. +_STOP_WORDS = frozenset({ + "fresh", "organic", "whole", "large", "small", "medium", + "low", "free", "light", "dark", "raw", "dried", "frozen", "canned", + "extra", "virgin", "pure", "natural", "classic", "style", + "boneless", "skinless", "lean", + "grain", "long", "jarred", "roasted", "smoked", "cooked", + "and", "with", "for", "the", +}) + @dataclass class MatchResult: @@ -32,104 +47,110 @@ class MatchResult: confidence: float -def build_match_pool(ingredients: Iterable[dict]) -> List[Tuple[str, str]]: - """Flatten (canonical name + aliases) into (text, ingredient_id) pairs.""" - pool: List[Tuple[str, str]] = [] - for ing in ingredients: - pool.append((ing["name"], ing["id"])) - for alias in ing.get("aliases") or []: - if alias: - pool.append((alias, ing["id"])) - return pool - - -def rank_candidates( - target: str, - pool: List[Tuple[str, str]], - top_n: int = 3, - threshold: float = 0.75, -) -> List[dict]: - """Return up to top_n unique-by-ingredient_id matches above threshold.""" - if not pool: - return [] - texts = [t for t, _ in pool] - extracted = process.extract(target, texts, scorer=fuzz.WRatio, limit=20) - seen: set[str] = set() - out: List[dict] = [] - for matched_text, score, idx in extracted: - confidence = score / 100.0 - if confidence < threshold: - continue - ingredient_id = pool[idx][1] - if ingredient_id in seen: - continue - seen.add(ingredient_id) - out.append( - { - "ingredient_id": ingredient_id, - "matched_text": matched_text, - "confidence": confidence, - } - ) - if len(out) >= top_n: - break - return out +def _sig_words(text: str) -> frozenset: + """Lowercase alpha tokens >2 chars, stop-words removed.""" + tokens = re.sub(r"[^a-z ]", " ", text.lower()).split() + return frozenset(t for t in tokens if len(t) > 2 and t not in _STOP_WORDS) def run_match_job( db: Session, *, source_filter: str = "lucky_california", - top_n: int = 3, - threshold: float = 0.75, + threshold: float = 0.82, ) -> int: - """Refresh ingredient_grocery_match for every grocery_item from `source_filter`. + """Refresh AUTO ingredient_grocery_match rows for grocery items from `source_filter`. - Manual matches (source='manual') are NOT touched. Returns the number of - auto rows written/updated. + For each ingredient the scorer is: + combined = partial_token_sort_ratio × (overlap / grocery_sig_count) + + where overlap = ingredient sig-words that appear in the grocery sig-words. + This means a product like "Milton's Olive Oil Crackers" (6 sig-words) + scores half of "Bertolli Olive Oil" (3 sig-words) for the ingredient + "Olive Oil", so the simpler/more specific product wins. + + 100% recall is required: every significant ingredient word must appear + in the grocery name. This eliminates cross-category noise such as + "Ginger, Fresh" → "Pampers Complete Clean Baby Fresh Scent Wipes". + + Manual matches (source='manual') are NOT touched. + Returns number of AUTO rows written. """ - ingredients = [ - {"id": str(row.id), "name": row.name, "aliases": list(row.aliases or [])} - for row in db.query(Ingredient).all() - ] - pool = build_match_pool(ingredients) - grocery_rows = ( db.query(GroceryItem) .filter(GroceryItem.source == source_filter) .all() ) + if not grocery_rows: + return 0 + grocery_names_lower = [gi.name.lower() for gi in grocery_rows] + grocery_sig = [_sig_words(gi.name) for gi in grocery_rows] + grocery_ids = [gi.id for gi in grocery_rows] + + # Purge existing AUTO matches for this source's items before re-matching. + db.query(IngredientGroceryMatch).filter( + IngredientGroceryMatch.grocery_item_id.in_([gi.id for gi in grocery_rows]), + IngredientGroceryMatch.source == IngredientMatchSource.AUTO, + ).delete(synchronize_session=False) + db.flush() + + ingredients = db.query(Ingredient).all() written = 0 - for grocery in grocery_rows: - target = " ".join(filter(None, [grocery.name, grocery.brand or ""])).strip() - ranked = rank_candidates(target, pool, top_n=top_n, threshold=threshold) - for r in ranked: - ing_id = UUID(r["ingredient_id"]) - existing = ( - db.query(IngredientGroceryMatch) - .filter( - IngredientGroceryMatch.ingredient_id == ing_id, - IngredientGroceryMatch.grocery_item_id == grocery.id, - ) - .first() + + for ingredient in ingredients: + ing_sig = _sig_words(ingredient.name) + if not ing_sig: + continue + + # Include aliases as additional query variants. + queries = [ingredient.name.lower()] + [ + a.lower() for a in (ingredient.aliases or []) if a + ] + + best_idx: int | None = None + best_combined = 0.0 + + for query in queries: + results = process.extract( + query, + grocery_names_lower, + scorer=fuzz.partial_token_sort_ratio, + limit=20, ) - if existing and existing.source == IngredientMatchSource.MANUAL: - continue - confidence = Decimal(str(round(r["confidence"], 3))) - if existing is None: - db.add( - IngredientGroceryMatch( - ingredient_id=ing_id, - grocery_item_id=grocery.id, - confidence=confidence, - source=IngredientMatchSource.AUTO, - ) - ) - else: - existing.confidence = confidence - existing.source = IngredientMatchSource.AUTO - written += 1 + for _text, score, idx in results: + if score < threshold * 100: + continue + gsig = grocery_sig[idx] + if not gsig: + continue + # 100% recall: every ingredient sig-word must appear in the grocery name. + if not ing_sig.issubset(gsig): + continue + # Precision penalises grocery items with many extra words. + precision = len(ing_sig) / len(gsig) + combined = (score / 100.0) * precision + if combined > best_combined: + best_combined = combined + best_idx = idx + + if best_idx is None: + continue + + # ON CONFLICT DO NOTHING preserves any existing MANUAL match for the same pair. + stmt = ( + _pg_insert(IngredientGroceryMatch.__table__) + .values( + id=_uuid_mod.uuid4(), + ingredient_id=ingredient.id, + grocery_item_id=grocery_ids[best_idx], + confidence=Decimal(str(round(best_combined, 3))), + source=IngredientMatchSource.AUTO, + ) + .on_conflict_do_nothing(index_elements=["ingredient_id", "grocery_item_id"]) + ) + db.execute(stmt) + written += 1 db.commit() return written