262 lines
15 KiB
Python
262 lines
15 KiB
Python
"""Conservative, resumable Open Food Facts enrichment of imported products."""
|
||
import hashlib
|
||
import json
|
||
from pathlib import Path
|
||
import re
|
||
import time
|
||
import tomllib
|
||
import unicodedata
|
||
from datetime import datetime, timezone
|
||
|
||
import click
|
||
import requests
|
||
|
||
from .grocy import Grocy, imported_products
|
||
from .history import private_json
|
||
from .product_metadata import read_metadata, replace_metadata
|
||
|
||
FIELDS = 'code,product_name,product_name_en,brands,quantity,ingredients_text_en,ingredients_text,allergens_tags,traces_tags,nutriments,nutrition_grades,nova_group,labels_tags,packaging,serving_size,image_front_url,countries_tags,last_modified_t'
|
||
NONFOOD = {'Household & Cleaning', 'Beauty & Toiletries', 'Home & Garden', 'Health, Medicines & Wellbeing', 'Pets'}
|
||
NONFOOD_BRANDS = {'Miniml', 'Heart & Soul', 'INTERNATIONAL GREETINGS', 'Caroline Gardner', 'Colgate'}
|
||
ALIASES = {'m s': 'marks spencer', 'm s food':'marks spencer', 'marks and spencer': 'marks spencer', 'marks spencers':'marks spencer', 'kelloggs special k': 'kelloggs', 'kellogg s': 'kelloggs', 'jacob s': 'jacobs', 'mcvitie s': 'mcvities', 'nairn s': 'nairns', 'garner s': 'garners', 'wrigley s extra': 'extra', 'nestle shredded wheat': 'shredded wheat', 'arnotts': 'arnotts', 'tncc': 'natural confectionery co'}
|
||
IGNORE = {'the', 'and', 'with', 'of', 'in', 'a', 'an', 'bag', 'sharing', 'multipack', 'pack', 'baked', 'snacks', 'breakfast', 'cereal', 'biscuits', 'biscuit', 'tinned', 'tin', 'can', 'single', 'jar', 'bottle', 'small', 'classic', 'favourites', 'ready', 'meal', 'cook'}
|
||
|
||
|
||
def words(text):
|
||
text = unicodedata.normalize('NFKD', str(text)).encode('ascii', 'ignore').decode().lower()
|
||
return ' '.join(re.findall(r'[a-z0-9]+', text.replace("'", '').replace('’', '')))
|
||
|
||
|
||
def brand_key(text):
|
||
key = words(text)
|
||
return ALIASES.get(key, key)
|
||
|
||
|
||
def pack(text):
|
||
"""Only unambiguous metric quantities; retain multipack structure."""
|
||
text = str(text).lower().replace('×', 'x').replace(',', '.')
|
||
text = text.split('(')[0].strip()
|
||
m = re.fullmatch(r'(?:(\d+)\s*x\s*)?(\d+(?:\.\d+)?)\s*(kg|g|ml|cl|l)', text)
|
||
if not m:
|
||
return None
|
||
n, amount, unit = m.groups()
|
||
return (int(n or 1), round(float(amount) * {'kg':1000, 'g':1, 'ml':1, 'cl':10, 'l':1000}[unit], 3), 'g' if unit in ('g','kg') else 'ml')
|
||
|
||
|
||
def tokens(name, brand):
|
||
name = words(re.sub(r'\b\d+(?:\.\d+)?\s*(?:kg|g|ml|cl|l)\b', '', name, flags=re.I))
|
||
remove = set(words(brand).split()) | set(brand_key(brand).split()) | IGNORE
|
||
return set(name.split()) - remove
|
||
|
||
|
||
def candidate_info(product, metadata, candidate):
|
||
brand = metadata.get('ocado', {}).get('brand', '')
|
||
expected = metadata.get('ocado', {}).get('size', {}).get('value', metadata.get('size', ''))
|
||
title = candidate.get('product_name') or candidate.get('product_name_en', '')
|
||
a, b = tokens(product['name'], brand), tokens(title, brand)
|
||
brands = [brand_key(x) for x in candidate.get('brands', '').split(',')]
|
||
brand_ok = bool(brand) and brand_key(brand) in brands
|
||
size_ok = pack(expected) is not None and pack(expected) == pack(candidate.get('quantity', ''))
|
||
score = len(a & b) / max(1, len(a | b))
|
||
exact = brand_ok and size_ok and bool(a) and a == b and valid_barcode(candidate.get('code', ''))
|
||
return {'code': candidate.get('code'), 'name':title, 'brand':candidate.get('brands'), 'quantity':candidate.get('quantity'), 'brand_match':brand_ok, 'size_match':size_ok, 'name_score':round(score, 3), 'exact':exact}
|
||
|
||
|
||
def choose_match(product, metadata, candidates):
|
||
ranked = sorted([candidate_info(product, metadata, c) for c in candidates.values()], key=lambda c:(c['exact'], c['brand_match'], c['name_score'], c['size_match']), reverse=True)
|
||
matches = [c for c in ranked if c['exact']]
|
||
return ranked, candidates[matches[0]['code']] if len(matches) == 1 else None
|
||
|
||
|
||
def valid_barcode(code):
|
||
if not isinstance(code, str) or not code.isascii() or not code.isdigit() or len(code) not in (8,12,13,14):
|
||
return False
|
||
return (sum(int(x) * (3 if i % 2 == 0 else 1) for i, x in enumerate(reversed(code[:-1]))) + int(code[-1])) % 10 == 0
|
||
|
||
|
||
class OpenFoodFacts:
|
||
def __init__(self, cache, contact, refresh=False):
|
||
self.refresh = refresh
|
||
self.cache = cache
|
||
self.session = requests.Session()
|
||
self.session.headers['User-Agent'] = f'ocado-grocy/0.1 ({contact})'
|
||
self.last = 0
|
||
|
||
def fetch(self, path, params, *, offline=False):
|
||
key = hashlib.sha256(json.dumps([path, params], sort_keys=True).encode()).hexdigest()
|
||
target = self.cache / (key + '.json')
|
||
if target.exists() and (offline or not self.refresh):
|
||
return json.loads(target.read_text())
|
||
if offline:
|
||
raise ValueError('Search not cached; rerun without --offline')
|
||
# Shared spacing for search and product reads: <= 10 requests per minute.
|
||
time.sleep(max(0, 6.2 - (time.monotonic() - self.last)))
|
||
self.last = time.monotonic()
|
||
url = path if path.startswith('https://search.openfoodfacts.org/') else 'https://world.openfoodfacts.org' + path
|
||
response = self.session.get(url, params=params, timeout=45)
|
||
response.raise_for_status()
|
||
data = response.json()
|
||
private_json(target, data)
|
||
return data
|
||
|
||
def search(self, product, metadata, offline=False):
|
||
"""Fetch one bounded page for this product, never a whole brand catalog."""
|
||
brand = metadata.get('ocado', {}).get('brand', '')
|
||
terms = sorted(tokens(product['name'], brand))
|
||
tags = {brand_key(brand).replace(' ', '-'),
|
||
re.sub(r'[^a-z0-9]+', '-', unicodedata.normalize('NFKD', brand)
|
||
.encode('ascii', 'ignore').decode().lower()).strip('-')}
|
||
if not brand or not terms:
|
||
return {}, {'query': '', 'search_count': 0, 'search_complete': True}
|
||
query = 'brands_tags:(' + ' OR '.join(sorted(tags)) + ') ' + ' '.join(terms)
|
||
data = self.fetch('https://search.openfoodfacts.org/search',
|
||
{'q': query, 'page_size': 100, 'page': 1, 'fields': FIELDS, 'langs': 'en'}, offline=offline)
|
||
if data.get('errors') or data.get('timed_out') or data.get('warnings'):
|
||
raise ValueError('Incomplete product search; retry later')
|
||
hits = data.get('hits', [])
|
||
complete = bool(data.get('is_count_exact')) and data.get('count') == len(hits)
|
||
candidates = {}
|
||
for hit in hits:
|
||
hit = dict(hit)
|
||
hit['code'] = str(hit['code'])
|
||
if isinstance(hit.get('brands'), list):
|
||
hit['brands'] = ','.join(hit['brands'])
|
||
candidates[hit['code']] = hit
|
||
return candidates, {'query': query, 'search_count': data.get('count'),
|
||
'search_complete': complete}
|
||
|
||
def product(self, code, offline=False):
|
||
data = self.fetch(f'/api/v3.6/product/{code}.json', {'fields':FIELDS + ',nutrition'}, offline=offline)
|
||
product = data.get('product')
|
||
if not product or str(product.get('code')) != code:
|
||
raise ValueError('Open Food Facts did not return requested barcode')
|
||
return product
|
||
|
||
|
||
def apply_match(client, product, candidate, barcodes):
|
||
"""Write only the barcode relation and description; never stock or name."""
|
||
code = str(candidate['code'])
|
||
if not valid_barcode(code):
|
||
raise ValueError('Invalid GTIN check digit')
|
||
owners = {int(b['product_id']) for b in barcodes if b['barcode'] == code}
|
||
if owners - {int(product['id'])}:
|
||
raise ValueError('Barcode already belongs to another Grocy product')
|
||
# Re-read just before mutation to retain user edits and skip deleted products.
|
||
current = client.get(f"/objects/products/{product['id']}")
|
||
if not current:
|
||
raise ValueError('Grocy product was deleted')
|
||
metadata = read_metadata(current.get('description'))
|
||
if not metadata or metadata.get('Ocado product ID') != read_metadata(product.get('description')).get('Ocado product ID'):
|
||
raise ValueError('Imported product identity changed')
|
||
if metadata.get('Barcode') and metadata['Barcode'] != code:
|
||
raise ValueError('Existing metadata barcode differs; review required')
|
||
enrichment = {k:v for k,v in candidate.items() if v not in (None, '', [], {})}
|
||
enrichment.update(source='Open Food Facts', url=f'https://world.openfoodfacts.org/product/{code}', database_license='ODbL', contents_license='Database Contents License', images_license='CC BY-SA')
|
||
metadata['Barcode'] = code
|
||
metadata['openfoodfacts'] = enrichment
|
||
updated = replace_metadata(current['description'], metadata)
|
||
if not owners:
|
||
row = {'product_id':int(product['id']), 'barcode':code, 'qu_id':current['qu_id_purchase'], 'amount':1}
|
||
client.post('/objects/product_barcodes', row)
|
||
barcodes.append(row)
|
||
if read_metadata(current['description']) != metadata:
|
||
client.request('PUT', f"/objects/products/{product['id']}", {'description':updated})
|
||
return 'updated'
|
||
return 'updated' if not owners else 'unchanged'
|
||
|
||
|
||
@click.command()
|
||
@click.option('--config', type=click.Path(path_type=Path), default=Path('config.toml'))
|
||
@click.option('--cache', type=click.Path(path_type=Path), default=Path('history/openfoodfacts'))
|
||
@click.option('--contact', help='Contact for the Open Food Facts User-Agent; defaults to config [openfoodfacts].contact.')
|
||
@click.option('--refresh-cache', is_flag=True, help='Fetch fresh OFF responses, replacing cached responses.')
|
||
@click.option('--apply', is_flag=True, help='Apply unambiguous matches and explicitly reviewed mappings.')
|
||
@click.option('--offline', is_flag=True, help='Use cached Open Food Facts responses only.')
|
||
@click.option('--mapping', type=click.Path(exists=True, path_type=Path), help='Reviewed JSON mapping of Grocy product ID to barcode.')
|
||
@click.option('--limit', type=int, help='Search at most this many eligible imported products.')
|
||
@click.option('--order-id', 'order_ids', multiple=True, help='Limit to products imported from these orders.')
|
||
def main(config, cache, contact, refresh_cache, apply, offline, mapping, limit, order_ids):
|
||
"""Match Ocado imports to Open Food Facts. Defaults to a read-only preview."""
|
||
settings = tomllib.loads(config.read_text())
|
||
contact = contact or settings.get('openfoodfacts', {}).get('contact')
|
||
if not contact:
|
||
raise click.UsageError('Set --contact or [openfoodfacts].contact in config.toml')
|
||
if refresh_cache and offline:
|
||
raise click.UsageError('--refresh-cache cannot be combined with --offline')
|
||
client = Grocy(**settings['grocy'])
|
||
state = config.parent / settings.get('import', {}).get('state_file', 'imports.sqlite3')
|
||
products = imported_products(client, state, order_ids)
|
||
mappings = json.loads(mapping.read_text()) if mapping else {}
|
||
return run_openfoodfacts(client, products, cache, contact, apply=apply, offline=offline,
|
||
refresh_cache=refresh_cache, mappings=mappings, limit=limit)
|
||
|
||
|
||
def run_openfoodfacts(client, products, cache, contact, *, apply=True, offline=False,
|
||
refresh_cache=False, mappings=None, limit=None):
|
||
barcodes = client.get('/objects/product_barcodes')
|
||
off = OpenFoodFacts(cache / 'responses', contact, refresh_cache)
|
||
mappings = mappings or {}
|
||
if not isinstance(mappings, dict) or any(not isinstance(v, str) or not valid_barcode(v) for v in mappings.values()):
|
||
raise click.UsageError('Mapping must be a JSON object with barcode strings and valid GTIN check digits')
|
||
report = {'created_at':datetime.now(timezone.utc).isoformat(), 'apply':apply, 'products':[]}
|
||
if apply:
|
||
private_json(cache / ('before-' + datetime.now().strftime('%Y%m%d-%H%M%S') + '.json'), {'products':products, 'barcodes':barcodes})
|
||
searched = 0
|
||
for product in products:
|
||
row = {'product_id':product['id'], 'name':product['name']}
|
||
metadata = read_metadata(product.get('description'))
|
||
ocado = metadata.get('ocado', {})
|
||
row['size'] = ocado.get('size', {}).get('value', metadata.get('size',''))
|
||
try:
|
||
if not metadata:
|
||
row['status'] = 'missing_metadata'
|
||
elif set(ocado.get('categoryPath', [])) & NONFOOD or ocado.get('brand') in NONFOOD_BRANDS:
|
||
row['status'] = 'non_food'
|
||
elif limit is not None and searched >= limit:
|
||
row['status'] = 'not_searched'
|
||
else:
|
||
searched += 1
|
||
code = mappings.get(str(product['id']))
|
||
method = 'reviewed_mapping'
|
||
established = metadata.get('openfoodfacts', {}).get('code')
|
||
if not code and established == metadata.get('Barcode') and valid_barcode(established):
|
||
code = established
|
||
method = 'existing_match'
|
||
if code:
|
||
candidate = off.product(str(code), offline)
|
||
row['match_method'] = method
|
||
row['candidates'] = [candidate_info(product, metadata, candidate)]
|
||
else:
|
||
click.echo(f"Searching Open Food Facts for {product['name']} ({row['size']})...")
|
||
candidates, search = off.search(product, metadata, offline)
|
||
ranked, candidate = choose_match(product, metadata, candidates)
|
||
row.update(search, candidates=ranked[:10])
|
||
# A capped result cannot establish that a barcode is unique.
|
||
if not search['search_complete']:
|
||
candidate = None
|
||
row['match_method'] = 'exact_brand_name_pack'
|
||
if candidate:
|
||
# Search is indexed separately; fetch current detail before writes.
|
||
if apply:
|
||
fresh = off.product(str(candidate['code']), offline)
|
||
row['current_candidate'] = candidate_info(product, metadata, fresh)
|
||
if not code and not row['current_candidate']['exact']:
|
||
raise ValueError('Current product details no longer match search result')
|
||
candidate = fresh
|
||
row['barcode'] = candidate['code']
|
||
row['status'] = apply_match(client, product, candidate, barcodes) if apply else 'matched'
|
||
else:
|
||
row['status'] = 'review' if row['candidates'] or not row.get('search_complete', True) else 'no_match'
|
||
except (requests.RequestException, ValueError, RuntimeError) as exc:
|
||
row.update(status='error', error=str(exc))
|
||
report['products'].append(row)
|
||
private_json(cache / ('applied-report.json' if apply else 'report.json'), report)
|
||
click.echo(f"{product['id']} {row['status']}: {product['name']}")
|
||
report['counts'] = {s:sum(r['status']==s for r in report['products']) for s in sorted({r['status'] for r in report['products']})}
|
||
private_json(cache / ('applied-report.json' if apply else 'report.json'), report)
|
||
click.echo(json.dumps(report['counts']))
|
||
return report
|
||
|
||
|
||
if __name__ == '__main__':
|
||
main()
|