Initial commit.
This commit is contained in:
commit
628e5c3823
26 changed files with 3478 additions and 0 deletions
283
ocado_grocy/openfoodfacts.py
Normal file
283
ocado_grocy/openfoodfacts.py
Normal file
|
|
@ -0,0 +1,283 @@
|
|||
"""Conservative, resumable Open Food Facts enrichment of imported products."""
|
||||
import hashlib
|
||||
import json
|
||||
from pathlib import Path
|
||||
import re
|
||||
import time
|
||||
import tomllib
|
||||
import unicodedata
|
||||
from datetime import datetime, timezone
|
||||
|
||||
import click
|
||||
import requests
|
||||
|
||||
from .grocy import Grocy, imported_products
|
||||
from .history import private_json
|
||||
from .product_metadata import read_metadata, replace_metadata
|
||||
|
||||
FIELDS = 'code,product_name,product_name_en,brands,quantity,ingredients_text_en,ingredients_text,allergens_tags,traces_tags,nutriments,nutrition_grades,nova_group,labels_tags,packaging,serving_size,image_front_url,countries_tags,last_modified_t'
|
||||
NONFOOD = {'Household & Cleaning', 'Beauty & Toiletries', 'Home & Garden', 'Health, Medicines & Wellbeing', 'Pets'}
|
||||
NONFOOD_BRANDS = {'Miniml', 'Heart & Soul', 'INTERNATIONAL GREETINGS', 'Caroline Gardner', 'Colgate'}
|
||||
ALIASES = {'m s': 'marks spencer', 'm s food':'marks spencer', 'marks and spencer': 'marks spencer', 'marks spencers':'marks spencer', 'kelloggs special k': 'kelloggs', 'kellogg s': 'kelloggs', 'jacob s': 'jacobs', 'mcvitie s': 'mcvities', 'nairn s': 'nairns', 'garner s': 'garners', 'wrigley s extra': 'extra', 'nestle shredded wheat': 'shredded wheat', 'arnotts': 'arnotts', 'tncc': 'natural confectionery co'}
|
||||
IGNORE = {'the', 'and', 'with', 'of', 'in', 'a', 'an', 'bag', 'sharing', 'multipack', 'pack', 'baked', 'snacks', 'breakfast', 'cereal', 'biscuits', 'biscuit', 'tinned', 'tin', 'can', 'single', 'jar', 'bottle', 'small', 'classic', 'favourites', 'ready', 'meal', 'cook'}
|
||||
|
||||
|
||||
def words(text):
|
||||
text = unicodedata.normalize('NFKD', str(text)).encode('ascii', 'ignore').decode().lower()
|
||||
return ' '.join(re.findall(r'[a-z0-9]+', text.replace("'", '').replace('’', '')))
|
||||
|
||||
|
||||
def brand_key(text):
|
||||
key = words(text)
|
||||
return ALIASES.get(key, key)
|
||||
|
||||
|
||||
def pack(text):
|
||||
"""Only unambiguous metric quantities; retain multipack structure."""
|
||||
text = str(text).lower().replace('×', 'x').replace(',', '.')
|
||||
text = text.split('(')[0].strip()
|
||||
m = re.fullmatch(r'(?:(\d+)\s*x\s*)?(\d+(?:\.\d+)?)\s*(kg|g|ml|cl|l)', text)
|
||||
if not m:
|
||||
return None
|
||||
n, amount, unit = m.groups()
|
||||
return (int(n or 1), round(float(amount) * {'kg':1000, 'g':1, 'ml':1, 'cl':10, 'l':1000}[unit], 3), 'g' if unit in ('g','kg') else 'ml')
|
||||
|
||||
|
||||
def tokens(name, brand):
|
||||
name = words(re.sub(r'\b\d+(?:\.\d+)?\s*(?:kg|g|ml|cl|l)\b', '', name, flags=re.I))
|
||||
remove = set(words(brand).split()) | set(brand_key(brand).split()) | IGNORE
|
||||
return set(name.split()) - remove
|
||||
|
||||
|
||||
def candidate_info(product, metadata, candidate):
|
||||
brand = metadata.get('ocado', {}).get('brand', '')
|
||||
expected = metadata.get('ocado', {}).get('size', {}).get('value', metadata.get('size', ''))
|
||||
title = candidate.get('product_name') or candidate.get('product_name_en', '')
|
||||
a, b = tokens(product['name'], brand), tokens(title, brand)
|
||||
brands = [brand_key(x) for x in candidate.get('brands', '').split(',')]
|
||||
brand_ok = bool(brand) and brand_key(brand) in brands
|
||||
size_ok = pack(expected) is not None and pack(expected) == pack(candidate.get('quantity', ''))
|
||||
score = len(a & b) / max(1, len(a | b))
|
||||
exact = brand_ok and size_ok and bool(a) and a == b and valid_barcode(candidate.get('code', ''))
|
||||
return {'code': candidate.get('code'), 'name':title, 'brand':candidate.get('brands'), 'quantity':candidate.get('quantity'), 'brand_match':brand_ok, 'size_match':size_ok, 'name_score':round(score, 3), 'exact':exact}
|
||||
|
||||
|
||||
def choose_match(product, metadata, candidates):
|
||||
ranked = sorted([candidate_info(product, metadata, c) for c in candidates.values()], key=lambda c:(c['exact'], c['brand_match'], c['name_score'], c['size_match']), reverse=True)
|
||||
matches = [c for c in ranked if c['exact']]
|
||||
return ranked, candidates[matches[0]['code']] if len(matches) == 1 else None
|
||||
|
||||
|
||||
def valid_barcode(code):
|
||||
if not isinstance(code, str) or not code.isascii() or not code.isdigit() or len(code) not in (8,12,13,14):
|
||||
return False
|
||||
return (sum(int(x) * (3 if i % 2 == 0 else 1) for i, x in enumerate(reversed(code[:-1]))) + int(code[-1])) % 10 == 0
|
||||
|
||||
|
||||
class OpenFoodFacts:
|
||||
def __init__(self, cache, contact, refresh=False):
|
||||
self.refresh = refresh
|
||||
self.cache = cache
|
||||
self.session = requests.Session()
|
||||
self.session.headers['User-Agent'] = f'ocado-grocy/0.1 ({contact})'
|
||||
self.last = 0
|
||||
|
||||
def fetch(self, path, params, *, offline=False):
|
||||
key = hashlib.sha256(json.dumps([path, params], sort_keys=True).encode()).hexdigest()
|
||||
target = self.cache / (key + '.json')
|
||||
if target.exists() and (offline or not self.refresh):
|
||||
return json.loads(target.read_text())
|
||||
if offline:
|
||||
raise ValueError('Search not cached; rerun without --offline')
|
||||
# Shared spacing for search and product reads: <= 10 requests per minute.
|
||||
time.sleep(max(0, 6.2 - (time.monotonic() - self.last)))
|
||||
self.last = time.monotonic()
|
||||
url = path if path.startswith('https://search.openfoodfacts.org/') else 'https://world.openfoodfacts.org' + path
|
||||
response = self.session.get(url, params=params, timeout=45)
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
private_json(target, data)
|
||||
return data
|
||||
|
||||
def catalog(self, brands, offline=False):
|
||||
tags = sorted({brand_key(b).replace(' ', '-') for b in brands if b})
|
||||
if not tags:
|
||||
return []
|
||||
|
||||
def group(tags):
|
||||
query = 'brands_tags:(' + ' OR '.join(tags) + ')'
|
||||
result = []
|
||||
page = 1
|
||||
while True:
|
||||
data = self.fetch('https://search.openfoodfacts.org/search', {'q':query, 'page_size':1000, 'page':page, 'fields':FIELDS}, offline=offline)
|
||||
if data.get('timed_out') or data.get('warnings'):
|
||||
raise ValueError('Incomplete catalog search; retry later')
|
||||
if not data.get('is_count_exact', False):
|
||||
if len(tags) == 1:
|
||||
raise ValueError('Brand catalog exceeds search limit: ' + tags[0])
|
||||
middle = len(tags) // 2
|
||||
return group(tags[:middle]) + group(tags[middle:])
|
||||
for hit in data.get('hits', []):
|
||||
hit = dict(hit)
|
||||
if isinstance(hit.get('brands'), list):
|
||||
hit['brands'] = ','.join(hit['brands'])
|
||||
result.append(hit)
|
||||
click.echo(f"Catalog ({len(tags)} brands) page {page}: {len(result)} of {data.get('count')} products")
|
||||
if page >= data.get('page_count', 1):
|
||||
if len(result) != data.get('count'):
|
||||
raise ValueError('Incomplete catalog response')
|
||||
return result
|
||||
page += 1
|
||||
|
||||
# OFF also retains apostrophes as tag separators (e.g. nairn-s).
|
||||
alternatives = sorted({re.sub(r'[^a-z0-9]+', '-', unicodedata.normalize('NFKD', b).encode('ascii', 'ignore').decode().lower()).strip('-') for b in brands if b} - set(tags))
|
||||
result = group(tags)
|
||||
if alternatives:
|
||||
result += group(alternatives)
|
||||
return list({str(p['code']):p for p in result}.values())
|
||||
|
||||
def product(self, code, offline=False):
|
||||
data = self.fetch(f'/api/v3.6/product/{code}.json', {'fields':FIELDS + ',nutrition'}, offline=offline)
|
||||
product = data.get('product')
|
||||
if not product or str(product.get('code')) != code:
|
||||
raise ValueError('Open Food Facts did not return requested barcode')
|
||||
return product
|
||||
|
||||
|
||||
def apply_match(client, product, candidate, barcodes):
|
||||
"""Write only the barcode relation and description; never stock or name."""
|
||||
code = str(candidate['code'])
|
||||
if not valid_barcode(code):
|
||||
raise ValueError('Invalid GTIN check digit')
|
||||
owners = {int(b['product_id']) for b in barcodes if b['barcode'] == code}
|
||||
if owners - {int(product['id'])}:
|
||||
raise ValueError('Barcode already belongs to another Grocy product')
|
||||
# Re-read just before mutation to retain user edits and skip deleted products.
|
||||
current = client.get(f"/objects/products/{product['id']}")
|
||||
if not current:
|
||||
raise ValueError('Grocy product was deleted')
|
||||
metadata = read_metadata(current.get('description'))
|
||||
if not metadata or metadata.get('Ocado product ID') != read_metadata(product.get('description')).get('Ocado product ID'):
|
||||
raise ValueError('Imported product identity changed')
|
||||
if metadata.get('Barcode') and metadata['Barcode'] != code:
|
||||
raise ValueError('Existing metadata barcode differs; review required')
|
||||
enrichment = {k:v for k,v in candidate.items() if v not in (None, '', [], {})}
|
||||
enrichment.update(source='Open Food Facts', url=f'https://world.openfoodfacts.org/product/{code}', database_license='ODbL', contents_license='Database Contents License', images_license='CC BY-SA')
|
||||
metadata['Barcode'] = code
|
||||
metadata['openfoodfacts'] = enrichment
|
||||
updated = replace_metadata(current['description'], metadata)
|
||||
if not owners:
|
||||
row = {'product_id':int(product['id']), 'barcode':code, 'qu_id':current['qu_id_purchase'], 'amount':1}
|
||||
client.post('/objects/product_barcodes', row)
|
||||
barcodes.append(row)
|
||||
if read_metadata(current['description']) != metadata:
|
||||
client.request('PUT', f"/objects/products/{product['id']}", {'description':updated})
|
||||
return 'updated'
|
||||
return 'updated' if not owners else 'unchanged'
|
||||
|
||||
|
||||
@click.command()
|
||||
@click.option('--config', type=click.Path(path_type=Path), default=Path('config.toml'))
|
||||
@click.option('--cache', type=click.Path(path_type=Path), default=Path('history/openfoodfacts'))
|
||||
@click.option('--contact', help='Contact for the Open Food Facts User-Agent; defaults to config [openfoodfacts].contact.')
|
||||
@click.option('--refresh-cache', is_flag=True, help='Fetch fresh OFF responses, replacing cached responses.')
|
||||
@click.option('--apply', is_flag=True, help='Apply unambiguous matches and explicitly reviewed mappings.')
|
||||
@click.option('--offline', is_flag=True, help='Use cached Open Food Facts responses only.')
|
||||
@click.option('--mapping', type=click.Path(exists=True, path_type=Path), help='Reviewed JSON mapping of Grocy product ID to barcode.')
|
||||
@click.option('--limit', type=int, help='Limit product comparisons after downloading the catalog.')
|
||||
@click.option('--order-id', 'order_ids', multiple=True, help='Limit to products imported from these orders.')
|
||||
def main(config, cache, contact, refresh_cache, apply, offline, mapping, limit, order_ids):
|
||||
"""Match Ocado imports to Open Food Facts. Defaults to a read-only preview."""
|
||||
settings = tomllib.loads(config.read_text())
|
||||
contact = contact or settings.get('openfoodfacts', {}).get('contact')
|
||||
if not contact:
|
||||
raise click.UsageError('Set --contact or [openfoodfacts].contact in config.toml')
|
||||
if refresh_cache and offline:
|
||||
raise click.UsageError('--refresh-cache cannot be combined with --offline')
|
||||
client = Grocy(**settings['grocy'])
|
||||
state = config.parent / settings.get('import', {}).get('state_file', 'imports.sqlite3')
|
||||
products = imported_products(client, state, order_ids)
|
||||
mappings = json.loads(mapping.read_text()) if mapping else {}
|
||||
return run_openfoodfacts(client, products, cache, contact, apply=apply, offline=offline,
|
||||
refresh_cache=refresh_cache, mappings=mappings, limit=limit)
|
||||
|
||||
|
||||
def run_openfoodfacts(client, products, cache, contact, *, apply=True, offline=False,
|
||||
refresh_cache=False, mappings=None, limit=None):
|
||||
barcodes = client.get('/objects/product_barcodes')
|
||||
off = OpenFoodFacts(cache / 'responses', contact, refresh_cache)
|
||||
mappings = mappings or {}
|
||||
if not isinstance(mappings, dict) or any(not isinstance(v, str) or not valid_barcode(v) for v in mappings.values()):
|
||||
raise click.UsageError('Mapping must be a JSON object with barcode strings and valid GTIN check digits')
|
||||
def needs_search(product):
|
||||
metadata = read_metadata(product.get('description'))
|
||||
code = metadata.get('openfoodfacts', {}).get('code')
|
||||
return not mappings.get(str(product['id'])) and not (valid_barcode(code) and code == metadata.get('Barcode'))
|
||||
brands = {read_metadata(p.get('description')).get('ocado', {}).get('brand') for p in products if needs_search(p)}
|
||||
catalog = off.catalog(brands, offline)
|
||||
private_json(cache / 'catalog.json', catalog)
|
||||
by_brand = {}
|
||||
for candidate in catalog:
|
||||
for brand in candidate.get('brands', '').split(','):
|
||||
by_brand.setdefault(brand_key(brand), {})[str(candidate['code'])] = candidate
|
||||
report = {'created_at':datetime.now(timezone.utc).isoformat(), 'apply':apply, 'products':[]}
|
||||
if apply:
|
||||
private_json(cache / ('before-' + datetime.now().strftime('%Y%m%d-%H%M%S') + '.json'), {'products':products, 'barcodes':barcodes})
|
||||
searched = 0
|
||||
for product in products:
|
||||
row = {'product_id':product['id'], 'name':product['name']}
|
||||
metadata = read_metadata(product.get('description'))
|
||||
ocado = metadata.get('ocado', {})
|
||||
row['size'] = ocado.get('size', {}).get('value', metadata.get('size',''))
|
||||
try:
|
||||
if not metadata:
|
||||
row['status'] = 'missing_metadata'
|
||||
elif set(ocado.get('categoryPath', [])) & NONFOOD or ocado.get('brand') in NONFOOD_BRANDS:
|
||||
row['status'] = 'non_food'
|
||||
elif limit is not None and searched >= limit:
|
||||
row['status'] = 'not_searched'
|
||||
else:
|
||||
searched += 1
|
||||
code = mappings.get(str(product['id']))
|
||||
method = 'reviewed_mapping'
|
||||
established = metadata.get('openfoodfacts', {}).get('code')
|
||||
if not code and established == metadata.get('Barcode') and valid_barcode(established):
|
||||
code = established
|
||||
method = 'existing_match'
|
||||
if code:
|
||||
candidate = next((c for c in catalog if str(c.get('code')) == str(code)), None) or off.product(str(code), offline)
|
||||
row['match_method'] = method
|
||||
row['candidates'] = [candidate_info(product, metadata, candidate)]
|
||||
else:
|
||||
query = 'brand catalog: ' + ocado.get('brand', '')
|
||||
result = {'count':len(catalog)}
|
||||
candidates = by_brand.get(brand_key(ocado.get('brand', '')), {})
|
||||
result['count'] = len(candidates)
|
||||
ranked, candidate = choose_match(product, metadata, candidates)
|
||||
row.update(query=query, search_count=result.get('count'), candidates=ranked[:10])
|
||||
row['match_method'] = 'exact_brand_name_pack'
|
||||
if candidate:
|
||||
# Catalog is indexed separately; fetch current detail before writes.
|
||||
if apply:
|
||||
fresh = off.product(str(candidate['code']), offline)
|
||||
row['current_candidate'] = candidate_info(product, metadata, fresh)
|
||||
if not code and not row['current_candidate']['exact']:
|
||||
raise ValueError('Current product details no longer match catalog')
|
||||
candidate = fresh
|
||||
row['barcode'] = candidate['code']
|
||||
row['status'] = apply_match(client, product, candidate, barcodes) if apply else 'matched'
|
||||
else:
|
||||
row['status'] = 'review' if row['candidates'] else 'no_match'
|
||||
except (requests.RequestException, ValueError, RuntimeError) as exc:
|
||||
row.update(status='error', error=str(exc))
|
||||
report['products'].append(row)
|
||||
private_json(cache / ('applied-report.json' if apply else 'report.json'), report)
|
||||
click.echo(f"{product['id']} {row['status']}: {product['name']}")
|
||||
report['counts'] = {s:sum(r['status']==s for r in report['products']) for s in sorted({r['status'] for r in report['products']})}
|
||||
private_json(cache / ('applied-report.json' if apply else 'report.json'), report)
|
||||
click.echo(json.dumps(report['counts']))
|
||||
return report
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Loading…
Add table
Add a link
Reference in a new issue