Limit order discovery and Open Food Facts searches

This commit is contained in:
Edward Betts 2026-10-07 11:16:50 +01:00
parent 628e5c3823
commit 41772b4d1a
5 changed files with 185 additions and 78 deletions

View file

@ -66,7 +66,9 @@ def wait_for_orders(page, timeout, echo):
raise ImportError("Timed out waiting for order history; rerun to reuse the saved browser session")
def discover_orders(page, echo):
def discover_orders(page, echo, *, completed=(), order_ids=()):
completed = set(completed)
requested = set(order_ids)
last_count = -1
stable = 0
for _ in range(500):
@ -76,8 +78,17 @@ def discover_orders(page, echo):
else:
stable = 0
echo(f"Loaded {count} order links...")
links = order_links(page.content())
visible = {link.order_id for link in links}
# Ocado lists newest orders first. Keep the whole visible batch so
# incomplete orders alongside the completion boundary remain eligible.
if requested:
if requested <= visible:
return links
elif completed & visible:
return links
if page.locator('[data-test="order-list-no-more-orders-label"]').count():
return order_links(page.content())
return links
if stable >= 10:
raise ImportError("Order history stopped loading before Ocado's end-of-list marker; retry after checking the browser")
last_count = count
@ -109,8 +120,8 @@ def download_orders(profile, archive, *, since=None, limit=None, login_only=Fals
if login_only:
echo("Signed-in browser session saved.")
return [], []
links = discover_orders(page, echo)
private_json(archive / "orders.json", [asdict(link) for link in links])
links = discover_orders(page, echo, completed=completed, order_ids=order_ids)
links = save_order_manifest(archive, links)
links = select_orders(links, since, limit, order_ids, completed)
receipts, failures = [], []
for index, link in enumerate(links,1):
@ -161,6 +172,17 @@ def download_orders(profile, archive, *, since=None, limit=None, login_only=Fals
context.close()
def save_order_manifest(archive, links):
"""Retain older history and pending imports after an incremental scan."""
path = archive / "orders.json"
saved = [OrderLink(**row) for row in json.loads(path.read_text())] if path.exists() else []
merged = {link.order_id: link for link in saved}
merged.update((link.order_id, link) for link in links)
links = sorted(merged.values(), key=lambda link: (link.purchased_date, link.order_id), reverse=True)
private_json(path, [asdict(link) for link in links])
return links
def select_orders(links, since=None, limit=None, order_ids=(), completed=()):
missing = set(order_ids) - {link.order_id for link in links}
if missing:

View file

@ -99,42 +99,31 @@ class OpenFoodFacts:
private_json(target, data)
return data
def catalog(self, brands, offline=False):
tags = sorted({brand_key(b).replace(' ', '-') for b in brands if b})
if not tags:
return []
def group(tags):
query = 'brands_tags:(' + ' OR '.join(tags) + ')'
result = []
page = 1
while True:
data = self.fetch('https://search.openfoodfacts.org/search', {'q':query, 'page_size':1000, 'page':page, 'fields':FIELDS}, offline=offline)
if data.get('timed_out') or data.get('warnings'):
raise ValueError('Incomplete catalog search; retry later')
if not data.get('is_count_exact', False):
if len(tags) == 1:
raise ValueError('Brand catalog exceeds search limit: ' + tags[0])
middle = len(tags) // 2
return group(tags[:middle]) + group(tags[middle:])
for hit in data.get('hits', []):
hit = dict(hit)
if isinstance(hit.get('brands'), list):
hit['brands'] = ','.join(hit['brands'])
result.append(hit)
click.echo(f"Catalog ({len(tags)} brands) page {page}: {len(result)} of {data.get('count')} products")
if page >= data.get('page_count', 1):
if len(result) != data.get('count'):
raise ValueError('Incomplete catalog response')
return result
page += 1
# OFF also retains apostrophes as tag separators (e.g. nairn-s).
alternatives = sorted({re.sub(r'[^a-z0-9]+', '-', unicodedata.normalize('NFKD', b).encode('ascii', 'ignore').decode().lower()).strip('-') for b in brands if b} - set(tags))
result = group(tags)
if alternatives:
result += group(alternatives)
return list({str(p['code']):p for p in result}.values())
def search(self, product, metadata, offline=False):
"""Fetch one bounded page for this product, never a whole brand catalog."""
brand = metadata.get('ocado', {}).get('brand', '')
terms = sorted(tokens(product['name'], brand))
tags = {brand_key(brand).replace(' ', '-'),
re.sub(r'[^a-z0-9]+', '-', unicodedata.normalize('NFKD', brand)
.encode('ascii', 'ignore').decode().lower()).strip('-')}
if not brand or not terms:
return {}, {'query': '', 'search_count': 0, 'search_complete': True}
query = 'brands_tags:(' + ' OR '.join(sorted(tags)) + ') ' + ' '.join(terms)
data = self.fetch('https://search.openfoodfacts.org/search',
{'q': query, 'page_size': 100, 'page': 1, 'fields': FIELDS, 'langs': 'en'}, offline=offline)
if data.get('errors') or data.get('timed_out') or data.get('warnings'):
raise ValueError('Incomplete product search; retry later')
hits = data.get('hits', [])
complete = bool(data.get('is_count_exact')) and data.get('count') == len(hits)
candidates = {}
for hit in hits:
hit = dict(hit)
hit['code'] = str(hit['code'])
if isinstance(hit.get('brands'), list):
hit['brands'] = ','.join(hit['brands'])
candidates[hit['code']] = hit
return candidates, {'query': query, 'search_count': data.get('count'),
'search_complete': complete}
def product(self, code, offline=False):
data = self.fetch(f'/api/v3.6/product/{code}.json', {'fields':FIELDS + ',nutrition'}, offline=offline)
@ -184,7 +173,7 @@ def apply_match(client, product, candidate, barcodes):
@click.option('--apply', is_flag=True, help='Apply unambiguous matches and explicitly reviewed mappings.')
@click.option('--offline', is_flag=True, help='Use cached Open Food Facts responses only.')
@click.option('--mapping', type=click.Path(exists=True, path_type=Path), help='Reviewed JSON mapping of Grocy product ID to barcode.')
@click.option('--limit', type=int, help='Limit product comparisons after downloading the catalog.')
@click.option('--limit', type=int, help='Search at most this many eligible imported products.')
@click.option('--order-id', 'order_ids', multiple=True, help='Limit to products imported from these orders.')
def main(config, cache, contact, refresh_cache, apply, offline, mapping, limit, order_ids):
"""Match Ocado imports to Open Food Facts. Defaults to a read-only preview."""
@ -209,17 +198,6 @@ def run_openfoodfacts(client, products, cache, contact, *, apply=True, offline=F
mappings = mappings or {}
if not isinstance(mappings, dict) or any(not isinstance(v, str) or not valid_barcode(v) for v in mappings.values()):
raise click.UsageError('Mapping must be a JSON object with barcode strings and valid GTIN check digits')
def needs_search(product):
metadata = read_metadata(product.get('description'))
code = metadata.get('openfoodfacts', {}).get('code')
return not mappings.get(str(product['id'])) and not (valid_barcode(code) and code == metadata.get('Barcode'))
brands = {read_metadata(p.get('description')).get('ocado', {}).get('brand') for p in products if needs_search(p)}
catalog = off.catalog(brands, offline)
private_json(cache / 'catalog.json', catalog)
by_brand = {}
for candidate in catalog:
for brand in candidate.get('brands', '').split(','):
by_brand.setdefault(brand_key(brand), {})[str(candidate['code'])] = candidate
report = {'created_at':datetime.now(timezone.utc).isoformat(), 'apply':apply, 'products':[]}
if apply:
private_json(cache / ('before-' + datetime.now().strftime('%Y%m%d-%H%M%S') + '.json'), {'products':products, 'barcodes':barcodes})
@ -245,29 +223,30 @@ def run_openfoodfacts(client, products, cache, contact, *, apply=True, offline=F
code = established
method = 'existing_match'
if code:
candidate = next((c for c in catalog if str(c.get('code')) == str(code)), None) or off.product(str(code), offline)
candidate = off.product(str(code), offline)
row['match_method'] = method
row['candidates'] = [candidate_info(product, metadata, candidate)]
else:
query = 'brand catalog: ' + ocado.get('brand', '')
result = {'count':len(catalog)}
candidates = by_brand.get(brand_key(ocado.get('brand', '')), {})
result['count'] = len(candidates)
click.echo(f"Searching Open Food Facts for {product['name']} ({row['size']})...")
candidates, search = off.search(product, metadata, offline)
ranked, candidate = choose_match(product, metadata, candidates)
row.update(query=query, search_count=result.get('count'), candidates=ranked[:10])
row.update(search, candidates=ranked[:10])
# A capped result cannot establish that a barcode is unique.
if not search['search_complete']:
candidate = None
row['match_method'] = 'exact_brand_name_pack'
if candidate:
# Catalog is indexed separately; fetch current detail before writes.
# Search is indexed separately; fetch current detail before writes.
if apply:
fresh = off.product(str(candidate['code']), offline)
row['current_candidate'] = candidate_info(product, metadata, fresh)
if not code and not row['current_candidate']['exact']:
raise ValueError('Current product details no longer match catalog')
raise ValueError('Current product details no longer match search result')
candidate = fresh
row['barcode'] = candidate['code']
row['status'] = apply_match(client, product, candidate, barcodes) if apply else 'matched'
else:
row['status'] = 'review' if row['candidates'] else 'no_match'
row['status'] = 'review' if row['candidates'] or not row.get('search_complete', True) else 'no_match'
except (requests.RequestException, ValueError, RuntimeError) as exc:
row.update(status='error', error=str(exc))
report['products'].append(row)