ocado-grocy/ocado_grocy/openfoodfacts.py

263 lines
15 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""Conservative, resumable Open Food Facts enrichment of imported products."""
import hashlib
import json
from pathlib import Path
import re
import time
import tomllib
import unicodedata
from datetime import datetime, timezone
import click
import requests
from .grocy import Grocy, imported_products
from .paths import config_file, archive_directory, journal_file
from .history import private_json
from .product_metadata import read_metadata, replace_metadata
FIELDS = 'code,product_name,product_name_en,brands,quantity,ingredients_text_en,ingredients_text,allergens_tags,traces_tags,nutriments,nutrition_grades,nova_group,labels_tags,packaging,serving_size,image_front_url,countries_tags,last_modified_t'
NONFOOD = {'Household & Cleaning', 'Beauty & Toiletries', 'Home & Garden', 'Health, Medicines & Wellbeing', 'Pets'}
NONFOOD_BRANDS = {'Miniml', 'Heart & Soul', 'INTERNATIONAL GREETINGS', 'Caroline Gardner', 'Colgate'}
ALIASES = {'m s': 'marks spencer', 'm s food':'marks spencer', 'marks and spencer': 'marks spencer', 'marks spencers':'marks spencer', 'kelloggs special k': 'kelloggs', 'kellogg s': 'kelloggs', 'jacob s': 'jacobs', 'mcvitie s': 'mcvities', 'nairn s': 'nairns', 'garner s': 'garners', 'wrigley s extra': 'extra', 'nestle shredded wheat': 'shredded wheat', 'arnotts': 'arnotts', 'tncc': 'natural confectionery co'}
IGNORE = {'the', 'and', 'with', 'of', 'in', 'a', 'an', 'bag', 'sharing', 'multipack', 'pack', 'baked', 'snacks', 'breakfast', 'cereal', 'biscuits', 'biscuit', 'tinned', 'tin', 'can', 'single', 'jar', 'bottle', 'small', 'classic', 'favourites', 'ready', 'meal', 'cook'}
def words(text):
text = unicodedata.normalize('NFKD', str(text)).encode('ascii', 'ignore').decode().lower()
return ' '.join(re.findall(r'[a-z0-9]+', text.replace("'", '').replace('’', '')))
def brand_key(text):
key = words(text)
return ALIASES.get(key, key)
def pack(text):
"""Only unambiguous metric quantities; retain multipack structure."""
text = str(text).lower().replace('×', 'x').replace(',', '.')
text = text.split('(')[0].strip()
m = re.fullmatch(r'(?:(\d+)\s*x\s*)?(\d+(?:\.\d+)?)\s*(kg|g|ml|cl|l)', text)
if not m:
return None
n, amount, unit = m.groups()
return (int(n or 1), round(float(amount) * {'kg':1000, 'g':1, 'ml':1, 'cl':10, 'l':1000}[unit], 3), 'g' if unit in ('g','kg') else 'ml')
def tokens(name, brand):
name = words(re.sub(r'\b\d+(?:\.\d+)?\s*(?:kg|g|ml|cl|l)\b', '', name, flags=re.I))
remove = set(words(brand).split()) | set(brand_key(brand).split()) | IGNORE
return set(name.split()) - remove
def candidate_info(product, metadata, candidate):
brand = metadata.get('ocado', {}).get('brand', '')
expected = metadata.get('ocado', {}).get('size', {}).get('value', metadata.get('size', ''))
title = candidate.get('product_name') or candidate.get('product_name_en', '')
a, b = tokens(product['name'], brand), tokens(title, brand)
brands = [brand_key(x) for x in candidate.get('brands', '').split(',')]
brand_ok = bool(brand) and brand_key(brand) in brands
size_ok = pack(expected) is not None and pack(expected) == pack(candidate.get('quantity', ''))
score = len(a & b) / max(1, len(a | b))
exact = brand_ok and size_ok and bool(a) and a == b and valid_barcode(candidate.get('code', ''))
return {'code': candidate.get('code'), 'name':title, 'brand':candidate.get('brands'), 'quantity':candidate.get('quantity'), 'brand_match':brand_ok, 'size_match':size_ok, 'name_score':round(score, 3), 'exact':exact}
def choose_match(product, metadata, candidates):
ranked = sorted([candidate_info(product, metadata, c) for c in candidates.values()], key=lambda c:(c['exact'], c['brand_match'], c['name_score'], c['size_match']), reverse=True)
matches = [c for c in ranked if c['exact']]
return ranked, candidates[matches[0]['code']] if len(matches) == 1 else None
def valid_barcode(code):
if not isinstance(code, str) or not code.isascii() or not code.isdigit() or len(code) not in (8,12,13,14):
return False
return (sum(int(x) * (3 if i % 2 == 0 else 1) for i, x in enumerate(reversed(code[:-1]))) + int(code[-1])) % 10 == 0
class OpenFoodFacts:
def __init__(self, cache, contact, refresh=False):
self.refresh = refresh
self.cache = cache
self.session = requests.Session()
self.session.headers['User-Agent'] = f'ocado-grocy/0.1 ({contact})'
self.last = 0
def fetch(self, path, params, *, offline=False):
key = hashlib.sha256(json.dumps([path, params], sort_keys=True).encode()).hexdigest()
target = self.cache / (key + '.json')
if target.exists() and (offline or not self.refresh):
return json.loads(target.read_text())
if offline:
raise ValueError('Search not cached; rerun without --offline')
# Shared spacing for search and product reads: <= 10 requests per minute.
time.sleep(max(0, 6.2 - (time.monotonic() - self.last)))
self.last = time.monotonic()
url = path if path.startswith('https://search.openfoodfacts.org/') else 'https://world.openfoodfacts.org' + path
response = self.session.get(url, params=params, timeout=45)
response.raise_for_status()
data = response.json()
private_json(target, data)
return data
def search(self, product, metadata, offline=False):
"""Fetch one bounded page for this product, never a whole brand catalog."""
brand = metadata.get('ocado', {}).get('brand', '')
terms = sorted(tokens(product['name'], brand))
tags = {brand_key(brand).replace(' ', '-'),
re.sub(r'[^a-z0-9]+', '-', unicodedata.normalize('NFKD', brand)
.encode('ascii', 'ignore').decode().lower()).strip('-')}
if not brand or not terms:
return {}, {'query': '', 'search_count': 0, 'search_complete': True}
query = 'brands_tags:(' + ' OR '.join(sorted(tags)) + ') ' + ' '.join(terms)
data = self.fetch('https://search.openfoodfacts.org/search',
{'q': query, 'page_size': 100, 'page': 1, 'fields': FIELDS, 'langs': 'en'}, offline=offline)
if data.get('errors') or data.get('timed_out') or data.get('warnings'):
raise ValueError('Incomplete product search; retry later')
hits = data.get('hits', [])
complete = bool(data.get('is_count_exact')) and data.get('count') == len(hits)
candidates = {}
for hit in hits:
hit = dict(hit)
hit['code'] = str(hit['code'])
if isinstance(hit.get('brands'), list):
hit['brands'] = ','.join(hit['brands'])
candidates[hit['code']] = hit
return candidates, {'query': query, 'search_count': data.get('count'),
'search_complete': complete}
def product(self, code, offline=False):
data = self.fetch(f'/api/v3.6/product/{code}.json', {'fields':FIELDS + ',nutrition'}, offline=offline)
product = data.get('product')
if not product or str(product.get('code')) != code:
raise ValueError('Open Food Facts did not return requested barcode')
return product
def apply_match(client, product, candidate, barcodes):
"""Write only the barcode relation and description; never stock or name."""
code = str(candidate['code'])
if not valid_barcode(code):
raise ValueError('Invalid GTIN check digit')
owners = {int(b['product_id']) for b in barcodes if b['barcode'] == code}
if owners - {int(product['id'])}:
raise ValueError('Barcode already belongs to another Grocy product')
# Re-read just before mutation to retain user edits and skip deleted products.
current = client.get(f"/objects/products/{product['id']}")
if not current:
raise ValueError('Grocy product was deleted')
metadata = read_metadata(current.get('description'))
if not metadata or metadata.get('Ocado product ID') != read_metadata(product.get('description')).get('Ocado product ID'):
raise ValueError('Imported product identity changed')
if metadata.get('Barcode') and metadata['Barcode'] != code:
raise ValueError('Existing metadata barcode differs; review required')
enrichment = {k:v for k,v in candidate.items() if v not in (None, '', [], {})}
enrichment.update(source='Open Food Facts', url=f'https://world.openfoodfacts.org/product/{code}', database_license='ODbL', contents_license='Database Contents License', images_license='CC BY-SA')
metadata['Barcode'] = code
metadata['openfoodfacts'] = enrichment
updated = replace_metadata(current['description'], metadata)
if not owners:
row = {'product_id':int(product['id']), 'barcode':code, 'qu_id':current['qu_id_purchase'], 'amount':1}
client.post('/objects/product_barcodes', row)
barcodes.append(row)
if read_metadata(current['description']) != metadata:
client.request('PUT', f"/objects/products/{product['id']}", {'description':updated})
return 'updated'
return 'updated' if not owners else 'unchanged'
@click.command()
@click.option('--config', type=click.Path(path_type=Path), default=config_file, show_default='~/.config/ocado-grocy/config.toml')
@click.option('--cache', type=click.Path(path_type=Path), default=lambda: archive_directory() / 'openfoodfacts', show_default='archive/openfoodfacts')
@click.option('--contact', help='Contact for the Open Food Facts User-Agent; defaults to config [openfoodfacts].contact.')
@click.option('--refresh-cache', is_flag=True, help='Fetch fresh OFF responses, replacing cached responses.')
@click.option('--apply', is_flag=True, help='Apply unambiguous matches and explicitly reviewed mappings.')
@click.option('--offline', is_flag=True, help='Use cached Open Food Facts responses only.')
@click.option('--mapping', type=click.Path(exists=True, path_type=Path), help='Reviewed JSON mapping of Grocy product ID to barcode.')
@click.option('--limit', type=int, help='Search at most this many eligible imported products.')
@click.option('--order-id', 'order_ids', multiple=True, help='Limit to products imported from these orders.')
def main(config, cache, contact, refresh_cache, apply, offline, mapping, limit, order_ids):
"""Match Ocado imports to Open Food Facts. Defaults to a read-only preview."""
settings = tomllib.loads(config.read_text())
contact = contact or settings.get('openfoodfacts', {}).get('contact')
if not contact:
raise click.UsageError('Set --contact or [openfoodfacts].contact in config.toml')
if refresh_cache and offline:
raise click.UsageError('--refresh-cache cannot be combined with --offline')
client = Grocy(**settings['grocy'])
state = journal_file(settings, config)
products = imported_products(client, state, order_ids)
mappings = json.loads(mapping.read_text()) if mapping else {}
return run_openfoodfacts(client, products, cache, contact, apply=apply, offline=offline,
refresh_cache=refresh_cache, mappings=mappings, limit=limit)
def run_openfoodfacts(client, products, cache, contact, *, apply=True, offline=False,
refresh_cache=False, mappings=None, limit=None):
barcodes = client.get('/objects/product_barcodes')
off = OpenFoodFacts(cache / 'responses', contact, refresh_cache)
mappings = mappings or {}
if not isinstance(mappings, dict) or any(not isinstance(v, str) or not valid_barcode(v) for v in mappings.values()):
raise click.UsageError('Mapping must be a JSON object with barcode strings and valid GTIN check digits')
report = {'created_at':datetime.now(timezone.utc).isoformat(), 'apply':apply, 'products':[]}
if apply:
private_json(cache / ('before-' + datetime.now().strftime('%Y%m%d-%H%M%S') + '.json'), {'products':products, 'barcodes':barcodes})
searched = 0
for product in products:
row = {'product_id':product['id'], 'name':product['name']}
metadata = read_metadata(product.get('description'))
ocado = metadata.get('ocado', {})
row['size'] = ocado.get('size', {}).get('value', metadata.get('size',''))
try:
if not metadata:
row['status'] = 'missing_metadata'
elif set(ocado.get('categoryPath', [])) & NONFOOD or ocado.get('brand') in NONFOOD_BRANDS:
row['status'] = 'non_food'
elif limit is not None and searched >= limit:
row['status'] = 'not_searched'
else:
searched += 1
code = mappings.get(str(product['id']))
method = 'reviewed_mapping'
established = metadata.get('openfoodfacts', {}).get('code')
if not code and established == metadata.get('Barcode') and valid_barcode(established):
code = established
method = 'existing_match'
if code:
candidate = off.product(str(code), offline)
row['match_method'] = method
row['candidates'] = [candidate_info(product, metadata, candidate)]
else:
click.echo(f"Searching Open Food Facts for {product['name']} ({row['size']})...")
candidates, search = off.search(product, metadata, offline)
ranked, candidate = choose_match(product, metadata, candidates)
row.update(search, candidates=ranked[:10])
# A capped result cannot establish that a barcode is unique.
if not search['search_complete']:
candidate = None
row['match_method'] = 'exact_brand_name_pack'
if candidate:
# Search is indexed separately; fetch current detail before writes.
if apply:
fresh = off.product(str(candidate['code']), offline)
row['current_candidate'] = candidate_info(product, metadata, fresh)
if not code and not row['current_candidate']['exact']:
raise ValueError('Current product details no longer match search result')
candidate = fresh
row['barcode'] = candidate['code']
row['status'] = apply_match(client, product, candidate, barcodes) if apply else 'matched'
else:
row['status'] = 'review' if row['candidates'] or not row.get('search_complete', True) else 'no_match'
except (requests.RequestException, ValueError, RuntimeError) as exc:
row.update(status='error', error=str(exc))
report['products'].append(row)
private_json(cache / ('applied-report.json' if apply else 'report.json'), report)
click.echo(f"{product['id']} {row['status']}: {product['name']}")
report['counts'] = {s:sum(r['status']==s for r in report['products']) for s in sorted({r['status'] for r in report['products']})}
private_json(cache / ('applied-report.json' if apply else 'report.json'), report)
click.echo(json.dumps(report['counts']))
return report
if __name__ == '__main__':
main()