Public Access
- Flip matching direction: iterate ingredients, search grocery items
(previously: iterate grocery items → false positives from partial word
overlap, e.g. Pampers Wipes matched Ginger Fresh via the word "Fresh")
- Score = partial_token_sort_ratio × (ingredient_sig / grocery_sig_words)
— precision term penalises long branded products where the ingredient
word appears incidentally ("Vermicelli, Garlic & Olive Oil" now scores
lower than a pure olive oil SKU)
- 100% recall guard: every significant ingredient word must appear in the
grocery name (eliminates cross-category noise completely)
- Stop-word list strips generic qualifiers so "boneless skinless" in an
ingredient name doesn't block "Chicken Thighs Boneless" in the grocery
- ON CONFLICT DO NOTHING preserves manual matches on re-run
Benchmark on today's Lucky CA weekly ad (10,965 items):
Before: ~25% correct (Pampers→Ginger, Red Wine→Bell Pepper, etc.)
After: ~80% correct; remaining misses are data gaps (Lucky has no
standalone garlic or olive oil in this week's ad)
Co-Authored-By: Claude Sonnet 4.6 (1M context) <noreply@anthropic.com>
157 lines
5.3 KiB
Python
157 lines
5.3 KiB
Python
"""Fuzzy ingredient→grocery_item matcher.
|
||
|
||
Ingredient-centric: for each ingredient, finds the best-matching grocery item.
|
||
Scores combine partial_token_sort_ratio with a precision term
|
||
(ingredient sig-words / grocery sig-words) so long branded product names
|
||
that contain an ingredient word incidentally rank lower than items whose
|
||
primary purpose IS that ingredient.
|
||
|
||
Manual matches (source='manual') are preserved across runs.
|
||
"""
|
||
from __future__ import annotations
|
||
|
||
import re
|
||
import uuid as _uuid_mod
|
||
from dataclasses import dataclass
|
||
from decimal import Decimal
|
||
from uuid import UUID
|
||
|
||
from rapidfuzz import fuzz, process
|
||
from sqlalchemy.dialects.postgresql import insert as _pg_insert
|
||
from sqlalchemy.orm import Session
|
||
|
||
from app.models import (
|
||
GroceryItem,
|
||
Ingredient,
|
||
IngredientGroceryMatch,
|
||
IngredientMatchSource,
|
||
)
|
||
|
||
# Words that describe quantity/preparation but don't identify the ingredient itself.
|
||
# Filtering these from sig-word sets keeps "Chicken Thighs, Boneless Skinless"
|
||
# from requiring "boneless" to appear in the grocery name.
|
||
_STOP_WORDS = frozenset({
|
||
"fresh", "organic", "whole", "large", "small", "medium",
|
||
"low", "free", "light", "dark", "raw", "dried", "frozen", "canned",
|
||
"extra", "virgin", "pure", "natural", "classic", "style",
|
||
"boneless", "skinless", "lean",
|
||
"grain", "long", "jarred", "roasted", "smoked", "cooked",
|
||
"and", "with", "for", "the",
|
||
})
|
||
|
||
|
||
@dataclass
|
||
class MatchResult:
|
||
ingredient_id: UUID
|
||
grocery_item_id: UUID
|
||
confidence: float
|
||
|
||
|
||
def _sig_words(text: str) -> frozenset:
|
||
"""Lowercase alpha tokens >2 chars, stop-words removed."""
|
||
tokens = re.sub(r"[^a-z ]", " ", text.lower()).split()
|
||
return frozenset(t for t in tokens if len(t) > 2 and t not in _STOP_WORDS)
|
||
|
||
|
||
def run_match_job(
|
||
db: Session,
|
||
*,
|
||
source_filter: str = "lucky_california",
|
||
threshold: float = 0.82,
|
||
) -> int:
|
||
"""Refresh AUTO ingredient_grocery_match rows for grocery items from `source_filter`.
|
||
|
||
For each ingredient the scorer is:
|
||
combined = partial_token_sort_ratio × (overlap / grocery_sig_count)
|
||
|
||
where overlap = ingredient sig-words that appear in the grocery sig-words.
|
||
This means a product like "Milton's Olive Oil Crackers" (6 sig-words)
|
||
scores half of "Bertolli Olive Oil" (3 sig-words) for the ingredient
|
||
"Olive Oil", so the simpler/more specific product wins.
|
||
|
||
100% recall is required: every significant ingredient word must appear
|
||
in the grocery name. This eliminates cross-category noise such as
|
||
"Ginger, Fresh" → "Pampers Complete Clean Baby Fresh Scent Wipes".
|
||
|
||
Manual matches (source='manual') are NOT touched.
|
||
Returns number of AUTO rows written.
|
||
"""
|
||
grocery_rows = (
|
||
db.query(GroceryItem)
|
||
.filter(GroceryItem.source == source_filter)
|
||
.all()
|
||
)
|
||
if not grocery_rows:
|
||
return 0
|
||
|
||
grocery_names_lower = [gi.name.lower() for gi in grocery_rows]
|
||
grocery_sig = [_sig_words(gi.name) for gi in grocery_rows]
|
||
grocery_ids = [gi.id for gi in grocery_rows]
|
||
|
||
# Purge existing AUTO matches for this source's items before re-matching.
|
||
db.query(IngredientGroceryMatch).filter(
|
||
IngredientGroceryMatch.grocery_item_id.in_([gi.id for gi in grocery_rows]),
|
||
IngredientGroceryMatch.source == IngredientMatchSource.AUTO,
|
||
).delete(synchronize_session=False)
|
||
db.flush()
|
||
|
||
ingredients = db.query(Ingredient).all()
|
||
written = 0
|
||
|
||
for ingredient in ingredients:
|
||
ing_sig = _sig_words(ingredient.name)
|
||
if not ing_sig:
|
||
continue
|
||
|
||
# Include aliases as additional query variants.
|
||
queries = [ingredient.name.lower()] + [
|
||
a.lower() for a in (ingredient.aliases or []) if a
|
||
]
|
||
|
||
best_idx: int | None = None
|
||
best_combined = 0.0
|
||
|
||
for query in queries:
|
||
results = process.extract(
|
||
query,
|
||
grocery_names_lower,
|
||
scorer=fuzz.partial_token_sort_ratio,
|
||
limit=20,
|
||
)
|
||
for _text, score, idx in results:
|
||
if score < threshold * 100:
|
||
continue
|
||
gsig = grocery_sig[idx]
|
||
if not gsig:
|
||
continue
|
||
# 100% recall: every ingredient sig-word must appear in the grocery name.
|
||
if not ing_sig.issubset(gsig):
|
||
continue
|
||
# Precision penalises grocery items with many extra words.
|
||
precision = len(ing_sig) / len(gsig)
|
||
combined = (score / 100.0) * precision
|
||
if combined > best_combined:
|
||
best_combined = combined
|
||
best_idx = idx
|
||
|
||
if best_idx is None:
|
||
continue
|
||
|
||
# ON CONFLICT DO NOTHING preserves any existing MANUAL match for the same pair.
|
||
stmt = (
|
||
_pg_insert(IngredientGroceryMatch.__table__)
|
||
.values(
|
||
id=_uuid_mod.uuid4(),
|
||
ingredient_id=ingredient.id,
|
||
grocery_item_id=grocery_ids[best_idx],
|
||
confidence=Decimal(str(round(best_combined, 3))),
|
||
source=IngredientMatchSource.AUTO,
|
||
)
|
||
.on_conflict_do_nothing(index_elements=["ingredient_id", "grocery_item_id"])
|
||
)
|
||
db.execute(stmt)
|
||
written += 1
|
||
|
||
db.commit()
|
||
return written
|