Public Access
- Add exclusion words: soda, rotisserie, tuna/tonno/salmon/sardine/anchovy to prevent beverages, prepared poultry, and seafood-in-oil from matching raw cooking ingredients - Add min precision floor (0.45): grocery sig-word count must be ≤ 2× the ingredient's sig-word count, catching long branded products that pass the word-overlap recall check but are clearly wrong category matches (e.g. "Garlic Herb Rotisserie Chicken" precision=0.25 now rejected) Result: all previously wrong matches now show '—' (no match) rather than a wrong product; correct matches unchanged Co-Authored-By: Claude Sonnet 4.6 (1M context) <noreply@anthropic.com>
195 lines
7.2 KiB
Python
195 lines
7.2 KiB
Python
"""Fuzzy ingredient→grocery_item matcher.
|
||
|
||
Ingredient-centric: for each ingredient, finds the best-matching grocery item.
|
||
Scores combine partial_token_sort_ratio with a precision term
|
||
(ingredient sig-words / grocery sig-words) so long branded product names
|
||
that contain an ingredient word incidentally rank lower than items whose
|
||
primary purpose IS that ingredient.
|
||
|
||
Manual matches (source='manual') are preserved across runs.
|
||
"""
|
||
from __future__ import annotations
|
||
|
||
import re
|
||
import uuid as _uuid_mod
|
||
from dataclasses import dataclass
|
||
from decimal import Decimal
|
||
from uuid import UUID
|
||
|
||
from rapidfuzz import fuzz, process
|
||
from sqlalchemy.dialects.postgresql import insert as _pg_insert
|
||
from sqlalchemy.orm import Session
|
||
|
||
from app.models import (
|
||
GroceryItem,
|
||
Ingredient,
|
||
IngredientGroceryMatch,
|
||
IngredientMatchSource,
|
||
)
|
||
|
||
# Words that describe quantity/preparation but don't identify the ingredient itself.
|
||
# Filtering these from sig-word sets keeps "Chicken Thighs, Boneless Skinless"
|
||
# from requiring "boneless" to appear in the grocery name.
|
||
_STOP_WORDS = frozenset({
|
||
"fresh", "organic", "whole", "large", "small", "medium",
|
||
"low", "free", "light", "dark", "raw", "dried", "frozen", "canned",
|
||
"extra", "virgin", "pure", "natural", "classic", "style",
|
||
"boneless", "skinless", "lean",
|
||
"grain", "long", "jarred", "roasted", "smoked", "cooked",
|
||
"and", "with", "for", "the",
|
||
})
|
||
|
||
# If any of these words appear in a grocery item's sig-words but NOT in the
|
||
# ingredient's sig-words, the match is rejected outright. Prevents category
|
||
# cross-contamination: "Garlic" must not match "Garlic Bread", "Lime" must not
|
||
# match "Lime Margarita", etc.
|
||
_EXCLUSION_WORDS = frozenset({
|
||
# Baked goods / bread products
|
||
"bread", "loaf", "rolls", "bun", "buns", "croissant",
|
||
"cracker", "crackers", "cookie", "cookies", "cake", "cupcake", "muffin", "bagel",
|
||
# Chips / snack foods
|
||
"chips",
|
||
# Pasta / noodles
|
||
"pasta", "noodle", "noodles", "vermicelli", "spaghetti", "linguine",
|
||
"fettuccine", "penne", "rigatoni", "macaroni", "rotini", "orzo",
|
||
# Alcoholic / mixed beverages
|
||
"margarita", "rita", "cocktail", "beer", "ale", "lager", "cider", "malt",
|
||
"wine", "spirits", "liquor",
|
||
"vodka", "tequila", "whiskey", "rum", "gin", "bourbon",
|
||
"lemonade", "limeade", "seltzer", "soda",
|
||
# Butter / spreads (prevents "Garlic & Herb Butter Spread" matching "Garlic")
|
||
"butter", "spread", "margarine",
|
||
# Prepared proteins / seafood-in-oil (prevents "Tuna in Olive Oil" matching "Olive Oil")
|
||
"tuna", "tonno", "salmon", "sardine", "anchovy",
|
||
# Prepared poultry (prevents "Garlic Herb Rotisserie Chicken" matching "Garlic")
|
||
"rotisserie",
|
||
# Baby / personal care (belt-and-suspenders after stop-word rework)
|
||
"baby", "wipes", "diaper",
|
||
})
|
||
|
||
|
||
@dataclass
|
||
class MatchResult:
|
||
ingredient_id: UUID
|
||
grocery_item_id: UUID
|
||
confidence: float
|
||
|
||
|
||
def _sig_words(text: str) -> frozenset:
|
||
"""Lowercase alpha tokens >2 chars, stop-words removed."""
|
||
tokens = re.sub(r"[^a-z ]", " ", text.lower()).split()
|
||
return frozenset(t for t in tokens if len(t) > 2 and t not in _STOP_WORDS)
|
||
|
||
|
||
def run_match_job(
|
||
db: Session,
|
||
*,
|
||
source_filter: str = "lucky_california",
|
||
threshold: float = 0.82,
|
||
) -> int:
|
||
"""Refresh AUTO ingredient_grocery_match rows for grocery items from `source_filter`.
|
||
|
||
For each ingredient the scorer is:
|
||
combined = partial_token_sort_ratio × (overlap / grocery_sig_count)
|
||
|
||
where overlap = ingredient sig-words that appear in the grocery sig-words.
|
||
This means a product like "Milton's Olive Oil Crackers" (6 sig-words)
|
||
scores half of "Bertolli Olive Oil" (3 sig-words) for the ingredient
|
||
"Olive Oil", so the simpler/more specific product wins.
|
||
|
||
100% recall is required: every significant ingredient word must appear
|
||
in the grocery name. This eliminates cross-category noise such as
|
||
"Ginger, Fresh" → "Pampers Complete Clean Baby Fresh Scent Wipes".
|
||
|
||
Manual matches (source='manual') are NOT touched.
|
||
Returns number of AUTO rows written.
|
||
"""
|
||
grocery_rows = (
|
||
db.query(GroceryItem)
|
||
.filter(GroceryItem.source == source_filter)
|
||
.all()
|
||
)
|
||
if not grocery_rows:
|
||
return 0
|
||
|
||
grocery_names_lower = [gi.name.lower() for gi in grocery_rows]
|
||
grocery_sig = [_sig_words(gi.name) for gi in grocery_rows]
|
||
grocery_ids = [gi.id for gi in grocery_rows]
|
||
|
||
# Purge existing AUTO matches for this source's items before re-matching.
|
||
db.query(IngredientGroceryMatch).filter(
|
||
IngredientGroceryMatch.grocery_item_id.in_([gi.id for gi in grocery_rows]),
|
||
IngredientGroceryMatch.source == IngredientMatchSource.AUTO,
|
||
).delete(synchronize_session=False)
|
||
db.flush()
|
||
|
||
ingredients = db.query(Ingredient).all()
|
||
written = 0
|
||
|
||
for ingredient in ingredients:
|
||
ing_sig = _sig_words(ingredient.name)
|
||
if not ing_sig:
|
||
continue
|
||
|
||
# Include aliases as additional query variants.
|
||
queries = [ingredient.name.lower()] + [
|
||
a.lower() for a in (ingredient.aliases or []) if a
|
||
]
|
||
|
||
best_idx: int | None = None
|
||
best_combined = 0.0
|
||
|
||
for query in queries:
|
||
results = process.extract(
|
||
query,
|
||
grocery_names_lower,
|
||
scorer=fuzz.partial_token_sort_ratio,
|
||
limit=20,
|
||
)
|
||
for _text, score, idx in results:
|
||
if score < threshold * 100:
|
||
continue
|
||
gsig = grocery_sig[idx]
|
||
if not gsig:
|
||
continue
|
||
# 100% recall: every ingredient sig-word must appear in the grocery name.
|
||
if not ing_sig.issubset(gsig):
|
||
continue
|
||
# Category exclusion: reject if grocery has a disqualifying word
|
||
# (e.g. "bread", "chips", "margarita") absent from the ingredient.
|
||
bad_words = (gsig & _EXCLUSION_WORDS) - ing_sig
|
||
if bad_words:
|
||
continue
|
||
# Precision penalises grocery items with many extra words.
|
||
precision = len(ing_sig) / len(gsig)
|
||
# Hard floor: grocery must not have >2× the sig-words of the ingredient.
|
||
# Catches long branded products that sneak past exclusion words, e.g.
|
||
# "Garlic Herb Rotisserie Chicken" for "Garlic".
|
||
if precision < 0.45:
|
||
continue
|
||
combined = (score / 100.0) * precision
|
||
if combined > best_combined:
|
||
best_combined = combined
|
||
best_idx = idx
|
||
|
||
if best_idx is None:
|
||
continue
|
||
|
||
# ON CONFLICT DO NOTHING preserves any existing MANUAL match for the same pair.
|
||
stmt = (
|
||
_pg_insert(IngredientGroceryMatch.__table__)
|
||
.values(
|
||
id=_uuid_mod.uuid4(),
|
||
ingredient_id=ingredient.id,
|
||
grocery_item_id=grocery_ids[best_idx],
|
||
confidence=Decimal(str(round(best_combined, 3))),
|
||
source=IngredientMatchSource.AUTO,
|
||
)
|
||
.on_conflict_do_nothing(index_elements=["ingredient_id", "grocery_item_id"])
|
||
)
|
||
db.execute(stmt)
|
||
written += 1
|
||
|
||
db.commit()
|
||
return written
|