Initial commit.

This commit is contained in:
Edward Betts 2026-09-19 11:32:53 +01:00
commit 628e5c3823
26 changed files with 3478 additions and 0 deletions

1
ocado_grocy/__init__.py Normal file
View file

@ -0,0 +1 @@
"""Ocado receipt importer."""

3
ocado_grocy/__main__.py Normal file
View file

@ -0,0 +1,3 @@
from .cli import main
main()

4
ocado_grocy/cli.py Normal file
View file

@ -0,0 +1,4 @@
"""Main command: acquire orders through Playwright."""
from .history import main
__all__ = ["main"]

View file

@ -0,0 +1,53 @@
"""Durable product-level retries, independent of completed stock orders."""
import json
import sqlite3
class RetryQueue:
def __init__(self, state, server):
self.server = server
self.db = sqlite3.connect(state)
with self.db:
self.db.execute('''CREATE TABLE IF NOT EXISTS enrichment_retries (
server TEXT, stage TEXT, product_id INTEGER, error TEXT,
PRIMARY KEY(server,stage,product_id))''')
self.db.execute('CREATE TABLE IF NOT EXISTS enrichment_retry_migrations (server TEXT PRIMARY KEY)')
def start(self, stage, products):
self.results(stage, [{'product_id':p['id'], 'status':'error', 'error':'Interrupted enrichment'} for p in products])
def results(self, stage, rows):
with self.db:
for row in rows:
key = (self.server,stage,int(row['product_id']))
if row['status'] == 'error':
self.db.execute('INSERT OR REPLACE INTO enrichment_retries VALUES (?,?,?,?)', (*key,row.get('error','Enrichment failed')))
else:
self.db.execute('DELETE FROM enrichment_retries WHERE server=? AND stage=? AND product_id=?',key)
def pending(self, stage):
return {r[0] for r in self.db.execute('SELECT product_id FROM enrichment_retries WHERE server=? AND stage=?',(self.server,stage))}
def migrate_reports(self, archive, allowed_ids):
"""Import old errors once; newer successful reports supersede old failures."""
if self.db.execute('SELECT 1 FROM enrichment_retry_migrations WHERE server=?',(self.server,)).fetchone():
return
events = []
for path in (archive/'enrichment').glob('*/summary.json'):
summary = json.loads(path.read_text())
if summary.get('server',self.server) != self.server:
continue
for stage in ('images','openfoodfacts'):
report_path = path.parent / f'{stage}.json'
if report_path.exists():
data = json.loads(report_path.read_text())
events.append((report_path.stat().st_mtime,stage,data.get('products',[])))
elif any(error.lower().startswith('images:' if stage=='images' else 'open food facts:') for error in summary.get('errors',[])):
events.append((path.stat().st_mtime,stage,[{'product_id':pid,'status':'error','error':'Previous enrichment failed'} for pid in summary.get('product_ids',[])]))
for _,stage,rows in sorted(events,key=lambda event:event[0]):
self.results(stage,[r for r in rows if int(r['product_id']) in allowed_ids])
with self.db:
self.db.execute('INSERT INTO enrichment_retry_migrations VALUES (?)',(self.server,))
def close(self):
self.db.close()

304
ocado_grocy/grocy.py Normal file
View file

@ -0,0 +1,304 @@
"""Small Grocy API client and resumable importer."""
from decimal import Decimal
from dataclasses import replace
import hashlib
import html
import json
import re
from pathlib import Path
import sqlite3
import requests
from .product_metadata import read_metadata
from .receipt import ImportError, number, receipt_date
def normalized(name):
return " ".join(name.casefold().split())
class Grocy:
def __init__(self, url, api_key, timeout=30):
self.url = url.rstrip("/")
if not self.url.endswith("/api"):
self.url += "/api"
self.session = requests.Session()
self.session.headers.update({"GROCY-API-KEY": api_key,
"User-Agent": "ocado-grocy/0.1"})
self.timeout = timeout
def request(self, method, path, data=None):
try:
response = self.session.request(method, self.url + path, json=data,
timeout=self.timeout, allow_redirects=False)
except requests.RequestException as exc:
raise ImportError(f"Grocy {method} {path} failed ({type(exc).__name__}); "
"check connectivity before retrying") from exc
if not 200 <= response.status_code < 300:
raise ImportError(f"Grocy {method} {path}: HTTP {response.status_code}")
try:
return response.json() if response.content else None
except ValueError as exc:
raise ImportError(f"Grocy {method} {path} returned non-JSON data") from exc
def get(self, path):
return self.request("GET", path)
def post(self, path, data):
return self.request("POST", path, data)
def named_object(self, entity, name, **fields):
matches = [x for x in self.get(f"/objects/{entity}")
if normalized(x["name"]) == normalized(name)]
if len(matches) > 1:
raise ImportError(f"Multiple Grocy {entity} named {name!r}")
if matches:
return int(matches[0]["id"])
return int(self.post(f"/objects/{entity}", {"name": name, **fields})["created_object_id"])
class Journal:
"""Commit intent before each write; ambiguous writes require explicit reconciliation."""
def __init__(self, path: Path):
path.parent.mkdir(parents=True, exist_ok=True)
self.db = sqlite3.connect(path)
self.db.execute("""CREATE TABLE IF NOT EXISTS imports (
server TEXT, order_id TEXT, line INTEGER, fingerprint TEXT,
status TEXT, product_id INTEGER, PRIMARY KEY(server, order_id, line))""")
self.db.commit()
def check(self, server, order, line, fingerprint):
row = self.db.execute("SELECT fingerprint, status FROM imports WHERE server=? AND order_id=? AND line=?",
(server, order, line)).fetchone()
if not row:
return False
if row[0] != fingerprint:
raise ImportError(f"Order {order} line {line} changed since a previous import; reconcile it manually")
if row[1] != "done":
raise ImportError(f"Order {order} line {line} has an uncertain previous write. "
"Check Grocy stock/logs and reconcile the journal before retrying (see README).")
return True
def record(self, server, order, line, fingerprint, status, product_id):
with self.db:
self.db.execute("INSERT OR REPLACE INTO imports VALUES (?, ?, ?, ?, ?, ?)",
(server, order, line, fingerprint, status, product_id))
def marker(item):
return f"Ocado product ID: {item.product_id}" if item.product_id else ""
def description(item):
details = {"Ocado product ID": item.product_id, "Barcode": item.barcode,
"Product URL": item.url, **item.metadata}
return "<p>" + html.escape(marker(item) or "Imported from Ocado") + "</p><pre>" + html.escape(
json.dumps(details, ensure_ascii=False, indent=2, default=str)) + "</pre>"
def match_product(item, products, barcodes, mappings):
override = mappings.get(item.product_id or item.name)
if override is not None:
matches = [p for p in products if int(p["id"]) == int(override["product_id"])]
if not matches:
raise ImportError(f"Mapped Grocy product does not exist: {item.name}")
return matches[0], override
matches = []
if item.product_id:
matches = [p for p in products if f"<p>{html.escape(marker(item))}</p>" in (p.get("description") or "")]
if not matches and item.barcode:
ids = {int(b["product_id"]) for b in barcodes if b["barcode"] == item.barcode}
matches = [p for p in products if int(p["id"]) in ids]
if not matches:
matches = [p for p in products if normalized(p["name"]) == normalized(item.name)]
if len(matches) > 1:
raise ImportError(f"Ambiguous existing product: {item.name}; add a config product mapping")
return (matches[0] if matches else None), {}
def item_fingerprint(receipt, item):
return hashlib.sha256(json.dumps({
"name": item.name, "product_id": item.product_id, "quantity": str(item.quantity.normalize()),
"total": str(item.total.normalize()), "best_before": item.best_before,
"date": receipt.purchased_date}, sort_keys=True).encode()).hexdigest()
def align_recorded_lines(receipt, server, journal):
"""Restore journal positions when Ocado reorders equal-expiry receipt rows."""
records = journal.db.execute(
"SELECT line, fingerprint FROM imports WHERE server=? AND order_id=? ORDER BY line",
(server, receipt.order_id)).fetchall()
remaining = list(receipt.items)
aligned = [None] * len(remaining)
for line, fingerprint in records:
candidates = [i for i,item in enumerate(remaining) if item_fingerprint(receipt,item) == fingerprint]
if not candidates or not 1 <= line <= len(aligned):
raise ImportError(f"Order {receipt.order_id} line {line} changed since its import; review before continuing")
aligned[line - 1] = remaining.pop(candidates[0])
rest = iter(remaining)
return replace(receipt, items=[item if item is not None else next(rest) for item in aligned])
def refresh_imported_products(receipt, client, journal, *, seen=None):
"""Refresh details of journal-linked products, including excluded perishables."""
seen = seen if seen is not None else set()
receipt = align_recorded_lines(receipt, client.url, journal)
records = journal.db.execute(
"SELECT line, product_id FROM imports WHERE server=? AND order_id=? AND status='done' ORDER BY line",
(client.url, receipt.order_id)).fetchall()
count = 0
for line, product_id in records:
if product_id in seen:
continue
if not 1 <= line <= len(receipt.items):
raise ImportError(f"Receipt line {line} no longer exists; cannot refresh order {receipt.order_id}")
item = receipt.items[line - 1]
product = client.get(f"/objects/products/{product_id}")
old = product.get("description") or ""
expected = f"<p>{html.escape(marker(item))}</p>" if item.product_id else ""
if expected and expected not in old:
raise ImportError(f"Product identity changed for order {receipt.order_id}, line {line}; review before refreshing")
if not expected and normalized(product["name"]) != normalized(item.name):
raise ImportError(f"Product name changed for order {receipt.order_id}, line {line}")
metadata = read_metadata(old)
for key, value in item.metadata.items():
if value not in (None, "", {}, []):
if isinstance(value, dict) and isinstance(metadata.get(key), dict):
metadata[key] = {**metadata[key], **value}
else:
metadata[key] = value
barcode = item.barcode or metadata.pop("Barcode", "")
url = item.url or metadata.pop("Product URL", "")
for key in ("Ocado product ID", "Barcode", "Product URL"):
metadata.pop(key, None)
updated = description(replace(item, metadata=metadata, barcode=barcode, url=url))
if old != updated:
client.request("PUT", f"/objects/products/{product_id}", {"description": updated})
count += 1
seen.add(product_id)
return count
def import_receipt(receipt, client, journal, config, echo=print, *, include_lines=None):
if not receipt.purchased_date:
raise ImportError("No purchase/delivery date found; supply --purchased-date YYYY-MM-DD")
products = client.get("/objects/products")
barcodes = client.get("/objects/product_barcodes")
plans = []
# Validate every existing product/conversion before any mutations.
for line, item in enumerate(receipt.items, 1):
if include_lines is not None and line not in include_lines:
continue
fingerprint = item_fingerprint(receipt, item)
if journal.check(client.url, receipt.order_id, line, fingerprint):
echo(f"Skipped already imported: {item.name}")
continue
product, override = match_product(item, products, barcodes, config.get("products", {}))
factor = number(override.get("stock_per_purchase", 1), "stock conversion")
if product:
if product.get("enable_tare_weight_handling") or product.get("no_own_stock"):
raise ImportError(f"Unsupported tare-weight or parent-only product: {item.name}")
if "stock_per_purchase" not in override and product["qu_id_purchase"] != product["qu_id_stock"]:
details = client.get(f'/stock/products/{product["id"]}')
factor = number(details.get("qu_conversion_factor_purchase_to_stock"), "stock conversion")
if factor <= 0:
raise ImportError(f"Stock conversion must be positive: {item.name}")
plans.append((line, item, fingerprint, product, factor))
if not plans:
return 0
location = client.named_object("locations", config.get("location", "Ocado imports"))
unit = client.named_object("quantity_units", config.get("quantity_unit", "Pack"), name_plural="Packs")
store = client.named_object("shopping_locations", "Ocado")
count = 0
for line, item, fingerprint, product, factor in plans:
if product is None:
# Recheck the local list: the same product can occur on multiple receipt lines.
product, _ = match_product(item, products, barcodes, config.get("products", {}))
if product is None:
payload = {"name": item.name, "description": description(item), "location_id": location,
"shopping_location_id": store, "qu_id_purchase": unit, "qu_id_stock": unit,
"qu_id_consume": unit, "qu_id_price": unit}
product_id = int(client.post("/objects/products", payload)["created_object_id"])
product = {"id": product_id, **payload}
products.append(product)
if item.barcode:
client.post("/objects/product_barcodes", {"product_id": product_id, "barcode": item.barcode,
"qu_id": unit, "amount": 1})
product_id = int(product["id"])
amount = item.quantity * factor
note = f"Ocado {receipt.purchased_date}"
if not item.best_before:
note += " Expiry unknown."
payload = {"amount": float(amount), "price": float(item.total / amount),
"best_before_date": item.best_before or "2999-12-31",
"purchased_date": receipt.purchased_date, "transaction_type": "purchase",
"shopping_location_id": store, "stock_label_type": 1, "note": note}
journal.record(client.url, receipt.order_id, line, fingerprint, "pending", product_id)
client.post(f"/stock/products/{product_id}/add", payload)
journal.record(client.url, receipt.order_id, line, fingerprint, "done", product_id)
echo(f"Imported {item.quantity} × {item.name}: £{item.total:.2f}")
count += 1
return count
def update_existing_notes(client, journal):
"""Edit only surviving journal-linked stock entries; never book new stock."""
imported = {(str(order), int(line), int(product)) for order, line, product in journal.db.execute(
"SELECT order_id, line, product_id FROM imports WHERE server=? AND status='done'", (client.url,))}
changed = {}
preserved_fields = ("amount", "best_before_date", "price", "open", "location_id",
"shopping_location_id", "purchased_date")
for row in client.get("/objects/stock"):
old = row.get("note") or ""
match = re.match(r"Ocado order (\d+), line (\d+)[.;]", old)
dated = re.fullmatch(r"Ocado (\d{4}-\d{2}-\d{2}), line (\d+)\.(?: Expiry unknown\.)?", old)
if match:
linked = (match[1], int(match[2]), int(row["product_id"])) in imported
elif dated:
linked = any(line == int(dated[2]) and product == int(row["product_id"])
for _, line, product in imported)
else:
linked = False
if not linked:
continue
try:
current = client.get(f"/stock/entry/{row['id']}")
except ImportError:
if not any(entry["id"] == row["id"] for entry in client.get("/objects/stock")):
continue
raise
if current is None or current.get("note") != old:
continue
row = current
purchased = receipt_date(row["purchased_date"])
if not purchased:
raise ImportError(f"Stock entry {row['id']} has no purchase date")
note = f"Ocado {purchased}"
if "Expiry unknown" in old:
note += " Expiry unknown."
payload = {field: row[field] for field in preserved_fields}
try:
client.request("PUT", f"/stock/entry/{row['id']}", {**payload, "note": note})
except ImportError:
if not any(entry["id"] == row["id"] for entry in client.get("/objects/stock")):
continue
raise
changed[row["id"]] = {**payload, "note": note}
for row in client.get("/objects/stock"):
if row["id"] in changed:
if row["note"] != changed[row["id"]]["note"]:
raise ImportError(f"Verification failed for stock entry {row['id']}")
return len(changed)
def imported_products(client, state, order_ids=()):
"""Live products from completed journal lines, optionally scoped to orders."""
query = "SELECT DISTINCT product_id FROM imports WHERE server=? AND status='done'"
params = [client.url]
if order_ids:
query += ' AND order_id IN (' + ','.join('?' for _ in order_ids) + ')'
params.extend(order_ids)
with sqlite3.connect(f'{Path(state).resolve().as_uri()}?mode=ro', uri=True) as db:
ids = {int(row[0]) for row in db.execute(query, params)}
return [p for p in client.get('/objects/products') if int(p['id']) in ids]

326
ocado_grocy/history.py Normal file
View file

@ -0,0 +1,326 @@
"""Browse Ocado history with a persistent, visible Playwright browser."""
from dataclasses import asdict, dataclass
from datetime import datetime
import json
import os
from pathlib import Path
import re
import sqlite3
import time
import tomllib
from urllib.parse import urlsplit
import click
from lxml import html
from .grocy import Grocy, Journal, import_receipt, refresh_imported_products, align_recorded_lines
from .pantry import classify
from .receipt import ImportError, parse_document, parse_ocado_order, receipt_date
@dataclass(frozen=True)
class OrderLink:
order_id: str
purchased_date: str
url: str
def private_json(path, data):
path.parent.mkdir(parents=True, exist_ok=True, mode=0o700)
temporary = path.with_suffix(path.suffix + ".tmp")
fd = os.open(temporary, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600)
with os.fdopen(fd, "w") as stream:
json.dump(data, stream, ensure_ascii=False, indent=2, default=str)
temporary.replace(path)
def order_links(content):
tree = html.fromstring(content)
found = {}
for anchor in tree.xpath('//a[contains(@href,"/orders/")]'):
href = anchor.get("href")
match = re.fullmatch(r"/orders/(\d+)/details", href)
text = " ".join(" ".join(anchor.itertext()).split())
if not match or not re.search(r"\bDelivered\b", text):
continue
stamp = re.search(r"([A-Z][a-z]{2} \d{1,2}, \d{4})", text)
if not stamp:
raise ImportError(f"No full delivery date on order {match[1]}")
purchased = datetime.strptime(stamp[1], "%b %d, %Y").date().isoformat()
found[match[1]] = OrderLink(match[1], purchased, "https://www.ocado.com" + href)
return list(found.values())
def wait_for_orders(page, timeout, echo):
deadline = time.monotonic() + timeout
notice = 0
while time.monotonic() < deadline:
if page.is_closed():
raise ImportError("Browser closed before sign-in completed")
if urlsplit(page.url).hostname == "www.ocado.com" and page.locator('a[href*="/orders/"]').count():
return
if time.monotonic() >= notice:
echo("Waiting for order history. Complete sign-in or any CAPTCHA in the browser.")
notice = time.monotonic() + 30
page.wait_for_timeout(1000)
raise ImportError("Timed out waiting for order history; rerun to reuse the saved browser session")
def discover_orders(page, echo):
last_count = -1
stable = 0
for _ in range(500):
count = page.locator('a[href*="/orders/"]').count()
if count == last_count:
stable += 1
else:
stable = 0
echo(f"Loaded {count} order links...")
if page.locator('[data-test="order-list-no-more-orders-label"]').count():
return order_links(page.content())
if stable >= 10:
raise ImportError("Order history stopped loading before Ocado's end-of-list marker; retry after checking the browser")
last_count = count
bottom = page.get_by_role("button", name="Back to top", exact=True)
if bottom.count():
bottom.scroll_into_view_if_needed()
else:
page.locator('a[href*="/orders/"]').last.scroll_into_view_if_needed()
page.mouse.wheel(0,1200)
page.wait_for_timeout(1500)
raise ImportError("Order list did not reach its end; refusing to call the download complete")
def download_orders(profile, archive, *, since=None, limit=None, login_only=False,
login_timeout=300, echo=print, order_ids=(), refresh=False, completed=()):
try:
from playwright.sync_api import sync_playwright, Error as BrowserError
except ModuleNotFoundError as exc:
raise ImportError("Install dependencies: pip install -e . and python -m playwright install chromium") from exc
profile.mkdir(parents=True, exist_ok=True, mode=0o700)
profile.chmod(0o700)
with sync_playwright() as playwright:
context = playwright.chromium.launch_persistent_context(
str(profile.resolve()), headless=False, viewport={"width":1280, "height":900})
try:
page = context.pages[0] if context.pages else context.new_page()
page.goto("https://www.ocado.com/orders", wait_until="domcontentloaded")
wait_for_orders(page, login_timeout, echo)
if login_only:
echo("Signed-in browser session saved.")
return [], []
links = discover_orders(page, echo)
private_json(archive / "orders.json", [asdict(link) for link in links])
links = select_orders(links, since, limit, order_ids, completed)
receipts, failures = [], []
for index, link in enumerate(links,1):
path = archive / f"{link.order_id}.json"
if path.exists() and not refresh:
receipt = parse_document(json.loads(path.read_text()))
if receipt.order_id != link.order_id or receipt.purchased_date != link.purchased_date:
raise ImportError(f"Cached receipt does not match order {link.order_id}")
receipts.append(receipt)
echo(f"[{index}/{len(links)}] Cached {link.order_id} ({link.purchased_date})")
continue
try:
endpoint = f"/api/order/v6/orders/{link.order_id}/decorated"
with page.expect_response(lambda response: urlsplit(response.url).path == endpoint,
timeout=45000) as captured:
page.goto(link.url, wait_until="domcontentloaded")
response = captured.value
if response.status != 200:
raise ImportError(f"Ocado receipt returned HTTP {response.status}")
data = response.json()
page.wait_for_selector('[data-test="order-online-receipt"]', timeout=30000)
hrefs = page.locator('[data-test="receipt-element-product-link"][href]').evaluate_all(
'(elements) => elements.map(element => element.getAttribute("href"))')
urls = {}
for href in hrefs:
match = re.search(r"/(\d+)$", href)
if match:
urls[match[1]] = "https://www.ocado.com" + href
receipt = parse_ocado_order(data, expected_order_id=link.order_id, product_urls=urls)
if receipt.purchased_date != link.purchased_date:
raise ImportError("Order-list date does not match the receipt API date")
private_json(path, receipt.export())
receipts.append(receipt)
echo(f"[{index}/{len(links)}] Saved {link.order_id}: {len(receipt.items)} lines")
except (ImportError, BrowserError, ValueError, KeyError) as exc:
message = str(exc) if isinstance(exc, ImportError) else type(exc).__name__
failures.append({"order_id":link.order_id,"error":message})
echo(f"[{index}/{len(links)}] Could not read {link.order_id}: {message}")
page.wait_for_timeout(750)
private_json(archive / "download-errors.json", failures)
return receipts, failures
finally:
try:
# Persistent profile is authoritative; this also preserves a portable cookie backup.
context.storage_state(path=str(profile / "auth.json"))
(profile / "auth.json").chmod(0o600)
finally:
context.close()
def select_orders(links, since=None, limit=None, order_ids=(), completed=()):
missing = set(order_ids) - {link.order_id for link in links}
if missing:
raise ImportError("Orders not found in delivered history: " + ", ".join(sorted(missing)))
links = sorted((link for link in links if not since or link.purchased_date >= since),
key=lambda link:(link.purchased_date,link.order_id), reverse=True)
if order_ids:
links = [link for link in links if link.order_id in order_ids]
if not order_ids:
links = [link for link in links if link.order_id not in completed]
if limit:
links = links[:limit]
return links
def cached_orders(archive, since=None, limit=None, order_ids=(), completed=()):
links = [OrderLink(**row) for row in json.loads((archive / "orders.json").read_text())]
links = select_orders(links, since, limit, order_ids, completed)
receipts, failures = [], []
for link in links:
path = archive / f"{link.order_id}.json"
if path.exists():
receipt = parse_document(json.loads(path.read_text()))
if receipt.order_id != link.order_id or receipt.purchased_date != link.purchased_date:
raise ImportError(f"Cached receipt does not match order {link.order_id}")
receipts.append(receipt)
else:
failures.append({"order_id":link.order_id,"error":"Receipt has not been downloaded"})
return receipts, failures
def run_import(receipts, config, config_path, archive, dry_run, failures, echo=print, refresh_existing=True,
all_products=True):
report = {"orders":len(receipts), "download_errors":failures, "items":[], "imported_lines":0, "refreshed_products":0}
settings = config.get("import", {})
journal = None
refreshed = set()
try:
if not dry_run:
connection = config.get("grocy", {})
url = os.environ.get("GROCY_URL", connection.get("url", ""))
key = os.environ.get("GROCY_API_KEY", connection.get("api_key", ""))
if not url or not key:
raise ImportError("Configure Grocy URL and API key before importing")
api = Grocy(url,key)
state = Path(settings.get("state_file", "imports.sqlite3"))
if not state.is_absolute():
state = config_path.resolve().parent / state
journal = Journal(state)
for receipt in receipts:
if not dry_run:
receipt = align_recorded_lines(receipt, api.url, journal)
if not dry_run and refresh_existing:
count = refresh_imported_products(receipt, api, journal, seen=refreshed)
report["refreshed_products"] += count
if count:
echo(f"Refreshed details for {count} previously imported products from {receipt.order_id}")
selected = set()
for line,item in enumerate(receipt.items,1):
decision = classify(item,config.get("pantry",{}).get("overrides",{}))
report["items"].append({"order_id":receipt.order_id, "date":receipt.purchased_date,
"line":line,"name":item.name,"product_id":item.product_id,
"quantity":str(item.quantity),"total":str(item.total),
**asdict(decision)})
if decision.include or all_products:
selected.add(line)
report["items"][-1]["include"] = True
if all_products:
report["items"][-1]["reason"] = "all products requested"
report["items"][-1]["uncertain"] = False
echo(f"Order {receipt.order_id} ({receipt.purchased_date}): {len(selected)}/{len(receipt.items)} selected lines")
if not dry_run:
report["imported_lines"] += import_receipt(
receipt,api,journal,{**settings,"products":config.get("products",{})},
echo=echo, include_lines=selected)
finally:
private_json(archive / "report.json",report)
if journal:
journal.db.close()
return report
@click.command(context_settings={"help_option_names":["-h","--help"]})
@click.option("--config", "config_path", type=click.Path(path_type=Path,dir_okay=False),default="config.toml",show_default=True)
@click.option("--profile", type=click.Path(path_type=Path,file_okay=False),default=".ocado-browser",show_default=True)
@click.option("--archive", type=click.Path(path_type=Path,file_okay=False),default="history",show_default=True)
@click.option("--since", help="Only orders delivered on or after YYYY-MM-DD.")
@click.option("--limit", type=click.IntRange(min=1),help="Process at most this many new delivered orders (or explicitly selected orders).")
@click.option("--dry-run", is_flag=True,help="Download and report selections without writing to Grocy.")
@click.option("--cached", is_flag=True,help="Use downloaded receipts without opening a browser.")
@click.option("--login-only", is_flag=True,help="Sign in and save cookies without downloading or importing orders.")
@click.option("--login-timeout",type=click.IntRange(min=1),default=300,show_default=True)
@click.option("--order-id", "order_ids", multiple=True, help="Process a specific delivered order; repeat for several.")
@click.option("--all-products/--shelf-stable-only", default=True, show_default=True,
help="Include all products, or restrict imports to shelf-stable items.")
@click.option("--refresh", is_flag=True, help="Download receipts again even when they are cached.")
@click.option("--refresh-existing/--no-refresh-existing", default=True, show_default=True,
help="Update details of previously imported products, including perishables.")
@click.option("--images/--no-images", default=True, show_default=True, help="Add missing Ocado product pictures after importing.")
@click.option("--openfoodfacts/--no-openfoodfacts", default=True, show_default=True, help="Match and enrich imported products after importing.")
def main(config_path,profile,archive,since,limit,dry_run,cached,login_only,login_timeout,refresh_existing,
order_ids,all_products,refresh,images,openfoodfacts):
"""Import delivered Ocado orders with Playwright; include all products by default."""
try:
if cached and login_only:
raise ImportError("--cached and --login-only cannot be combined")
if cached and refresh:
raise ImportError("--cached and --refresh cannot be combined")
config = tomllib.loads(config_path.read_text()) if config_path.exists() else {}
from .order_state import completed_orders, mark_orders
connection = config.get('grocy', {})
server = Grocy(os.environ.get('GROCY_URL', connection.get('url', '')),
os.environ.get('GROCY_API_KEY', connection.get('api_key', ''))).url
state = Path(config.get('import', {}).get('state_file', 'imports.sqlite3'))
if not state.is_absolute():
state = config_path.resolve().parent / state
completed = completed_orders(state, server, archive, config)
if not dry_run and not login_only:
mark_orders(state, server, completed, 'complete')
retry_errors = []
if not dry_run and not login_only and (images or openfoodfacts):
from .postprocess import retry_enrichment
retry_errors = retry_enrichment(config, config_path, archive, images=images, openfoodfacts=openfoodfacts)['errors']
since = receipt_date(since)
archive.mkdir(parents=True,exist_ok=True,mode=0o700)
if cached:
receipts,failures = cached_orders(archive,since,limit,order_ids,completed)
else:
receipts,failures = download_orders(profile,archive,since=since,limit=limit,
login_only=login_only,login_timeout=login_timeout,echo=click.echo,order_ids=order_ids,refresh=refresh,completed=completed)
if login_only:
return
if not receipts and not failures:
if retry_errors:
raise ImportError('Enrichment retries still need attention: ' + '; '.join(retry_errors))
click.echo('No new delivered orders to import.')
return
processing = [receipt.order_id for receipt in receipts]
if not dry_run:
mark_orders(state, server, processing, 'processing')
report = run_import(receipts,config,config_path,archive,dry_run,failures,echo=click.echo,
refresh_existing=refresh_existing,all_products=all_products)
selected = sum(row["include"] for row in report["items"])
uncertain = sum(row["uncertain"] for row in report["items"])
click.echo(f"{len(receipts)} orders; {selected} selected lines; {report['imported_lines']} imported; "
f"{uncertain} uncertain lines. Report: {archive / 'report.json'}")
if not dry_run:
from .postprocess import enrich_orders
enrichment = enrich_orders(config, config_path, archive,
[receipt.order_id for receipt in receipts], images=images, openfoodfacts=openfoodfacts)
report['enrichment'] = enrichment
private_json(archive / 'report.json', report)
mark_orders(state, server, processing, 'complete')
if enrichment['errors'] or retry_errors:
raise ImportError('Stock import completed; enrichment needs a retry. ' + '; '.join(enrichment['errors'] + retry_errors))
if failures:
raise ImportError(f"{len(failures)} orders could not be downloaded; see report and retry")
except (ImportError,OSError,ValueError,KeyError,sqlite3.Error) as exc:
raise click.ClickException(str(exc)) from exc
if __name__ == "__main__":
main()

113
ocado_grocy/images.py Normal file
View file

@ -0,0 +1,113 @@
"""Copy saved Ocado product photos into Grocy's native product pictures."""
import base64
import hashlib
import json
from pathlib import Path
import tomllib
from urllib.parse import quote, urlsplit
import click
import requests
from .grocy import Grocy, imported_products
from .history import private_json
from .product_metadata import read_metadata
def image_type(data):
if data.startswith(b'\xff\xd8\xff'):
return 'jpg'
if data.startswith(b'\x89PNG\r\n\x1a\n'):
return 'png'
if data.startswith(b'RIFF') and data[8:12] == b'WEBP':
return 'webp'
raise ValueError('Image response is not JPEG, PNG or WebP')
def file_path(filename):
return '/files/productpictures/' + quote(base64.b64encode(filename.encode()).decode(), safe='')
def add_picture(client, downloader, product):
if product.get('picture_file_name'):
return 'existing_picture'
metadata = read_metadata(product.get('description'))
url = metadata.get('image_url')
if not url:
return 'no_image_url'
parsed = urlsplit(url)
if parsed.scheme != 'https' or parsed.hostname not in {'www.ocado.com', 'ocado.com'}:
raise ValueError('Expected an HTTPS Ocado image URL')
# Separate session: never send the Grocy API key to the image host.
response = downloader.get(url, timeout=30, allow_redirects=False)
response.raise_for_status()
if response.status_code != 200:
raise ValueError('Image URL did not return HTTP 200')
data = response.content
extension = image_type(data)
if len(data) > 10_000_000:
raise ValueError('Image exceeds 10 MB')
filename = f"ocado-{int(product['id'])}-{hashlib.sha256(data).hexdigest()[:16]}.{extension}"
current = client.get(f"/objects/products/{product['id']}")
if not current:
return 'deleted_product'
if current.get('picture_file_name'):
return 'existing_picture'
if read_metadata(current.get('description')).get('Ocado product ID') != metadata.get('Ocado product ID'):
raise ValueError('Product identity changed')
path = file_path(filename)
uploaded = client.session.put(client.url + path, data=data,
headers={'Content-Type':'application/octet-stream'},
timeout=client.timeout, allow_redirects=False)
uploaded.raise_for_status()
if uploaded.status_code not in (200, 201, 204):
raise ValueError('Unexpected upload response')
# Check the stored file before setting the product's picture reference.
stored = client.session.get(client.url + path, timeout=client.timeout, allow_redirects=False)
stored.raise_for_status()
if stored.status_code != 200 or stored.content != data:
raise ValueError('Uploaded image verification failed')
client.request('PUT', f"/objects/products/{product['id']}", {'picture_file_name':filename})
return 'updated'
@click.command()
@click.option('--config', type=click.Path(path_type=Path), default=Path('config.toml'))
@click.option('--report', type=click.Path(path_type=Path), default=Path('history/images/report.json'))
@click.option('--apply', is_flag=True, help='Upload and assign missing product pictures.')
@click.option('--order-id', 'order_ids', multiple=True, help='Limit to products imported from these orders.')
def main(config, report, apply, order_ids):
"""Add Ocado images to journal-linked products; keep existing pictures."""
settings = tomllib.loads(config.read_text())
client = Grocy(**settings['grocy'])
state = config.parent / settings.get('import', {}).get('state_file', 'imports.sqlite3')
products = imported_products(client, state, order_ids)
return run_images(client, products, report, apply=apply)
def run_images(client, products, report, *, apply=True, raise_errors=True):
downloader = requests.Session()
downloader.headers['User-Agent'] = 'ocado-grocy/0.1'
results = []
private_json(report, {'apply':apply, 'products':results})
for product in products:
row = {'product_id':product['id'], 'name':product['name']}
try:
if apply:
row['status'] = add_picture(client, downloader, product)
else:
row['status'] = 'existing_picture' if product.get('picture_file_name') else 'ready' if read_metadata(product.get('description')).get('image_url') else 'no_image_url'
except (requests.RequestException, ValueError) as exc:
row.update(status='error', error=str(exc))
results.append(row)
private_json(report, {'apply':apply, 'products':results})
click.echo(f"{product['id']} {row['status']}: {product['name']}")
counts = {s:sum(r['status']==s for r in results) for s in sorted({r['status'] for r in results})}
click.echo(json.dumps(counts))
if counts.get('error') and raise_errors:
raise click.ClickException(f"{counts['error']} image updates failed; see {report}")
return {'products':results, 'counts':counts}
if __name__ == '__main__':
main()

View file

@ -0,0 +1,283 @@
"""Conservative, resumable Open Food Facts enrichment of imported products."""
import hashlib
import json
from pathlib import Path
import re
import time
import tomllib
import unicodedata
from datetime import datetime, timezone
import click
import requests
from .grocy import Grocy, imported_products
from .history import private_json
from .product_metadata import read_metadata, replace_metadata
FIELDS = 'code,product_name,product_name_en,brands,quantity,ingredients_text_en,ingredients_text,allergens_tags,traces_tags,nutriments,nutrition_grades,nova_group,labels_tags,packaging,serving_size,image_front_url,countries_tags,last_modified_t'
NONFOOD = {'Household & Cleaning', 'Beauty & Toiletries', 'Home & Garden', 'Health, Medicines & Wellbeing', 'Pets'}
NONFOOD_BRANDS = {'Miniml', 'Heart & Soul', 'INTERNATIONAL GREETINGS', 'Caroline Gardner', 'Colgate'}
ALIASES = {'m s': 'marks spencer', 'm s food':'marks spencer', 'marks and spencer': 'marks spencer', 'marks spencers':'marks spencer', 'kelloggs special k': 'kelloggs', 'kellogg s': 'kelloggs', 'jacob s': 'jacobs', 'mcvitie s': 'mcvities', 'nairn s': 'nairns', 'garner s': 'garners', 'wrigley s extra': 'extra', 'nestle shredded wheat': 'shredded wheat', 'arnotts': 'arnotts', 'tncc': 'natural confectionery co'}
IGNORE = {'the', 'and', 'with', 'of', 'in', 'a', 'an', 'bag', 'sharing', 'multipack', 'pack', 'baked', 'snacks', 'breakfast', 'cereal', 'biscuits', 'biscuit', 'tinned', 'tin', 'can', 'single', 'jar', 'bottle', 'small', 'classic', 'favourites', 'ready', 'meal', 'cook'}
def words(text):
text = unicodedata.normalize('NFKD', str(text)).encode('ascii', 'ignore').decode().lower()
return ' '.join(re.findall(r'[a-z0-9]+', text.replace("'", '').replace('’', '')))
def brand_key(text):
key = words(text)
return ALIASES.get(key, key)
def pack(text):
"""Only unambiguous metric quantities; retain multipack structure."""
text = str(text).lower().replace('×', 'x').replace(',', '.')
text = text.split('(')[0].strip()
m = re.fullmatch(r'(?:(\d+)\s*x\s*)?(\d+(?:\.\d+)?)\s*(kg|g|ml|cl|l)', text)
if not m:
return None
n, amount, unit = m.groups()
return (int(n or 1), round(float(amount) * {'kg':1000, 'g':1, 'ml':1, 'cl':10, 'l':1000}[unit], 3), 'g' if unit in ('g','kg') else 'ml')
def tokens(name, brand):
name = words(re.sub(r'\b\d+(?:\.\d+)?\s*(?:kg|g|ml|cl|l)\b', '', name, flags=re.I))
remove = set(words(brand).split()) | set(brand_key(brand).split()) | IGNORE
return set(name.split()) - remove
def candidate_info(product, metadata, candidate):
brand = metadata.get('ocado', {}).get('brand', '')
expected = metadata.get('ocado', {}).get('size', {}).get('value', metadata.get('size', ''))
title = candidate.get('product_name') or candidate.get('product_name_en', '')
a, b = tokens(product['name'], brand), tokens(title, brand)
brands = [brand_key(x) for x in candidate.get('brands', '').split(',')]
brand_ok = bool(brand) and brand_key(brand) in brands
size_ok = pack(expected) is not None and pack(expected) == pack(candidate.get('quantity', ''))
score = len(a & b) / max(1, len(a | b))
exact = brand_ok and size_ok and bool(a) and a == b and valid_barcode(candidate.get('code', ''))
return {'code': candidate.get('code'), 'name':title, 'brand':candidate.get('brands'), 'quantity':candidate.get('quantity'), 'brand_match':brand_ok, 'size_match':size_ok, 'name_score':round(score, 3), 'exact':exact}
def choose_match(product, metadata, candidates):
ranked = sorted([candidate_info(product, metadata, c) for c in candidates.values()], key=lambda c:(c['exact'], c['brand_match'], c['name_score'], c['size_match']), reverse=True)
matches = [c for c in ranked if c['exact']]
return ranked, candidates[matches[0]['code']] if len(matches) == 1 else None
def valid_barcode(code):
if not isinstance(code, str) or not code.isascii() or not code.isdigit() or len(code) not in (8,12,13,14):
return False
return (sum(int(x) * (3 if i % 2 == 0 else 1) for i, x in enumerate(reversed(code[:-1]))) + int(code[-1])) % 10 == 0
class OpenFoodFacts:
def __init__(self, cache, contact, refresh=False):
self.refresh = refresh
self.cache = cache
self.session = requests.Session()
self.session.headers['User-Agent'] = f'ocado-grocy/0.1 ({contact})'
self.last = 0
def fetch(self, path, params, *, offline=False):
key = hashlib.sha256(json.dumps([path, params], sort_keys=True).encode()).hexdigest()
target = self.cache / (key + '.json')
if target.exists() and (offline or not self.refresh):
return json.loads(target.read_text())
if offline:
raise ValueError('Search not cached; rerun without --offline')
# Shared spacing for search and product reads: <= 10 requests per minute.
time.sleep(max(0, 6.2 - (time.monotonic() - self.last)))
self.last = time.monotonic()
url = path if path.startswith('https://search.openfoodfacts.org/') else 'https://world.openfoodfacts.org' + path
response = self.session.get(url, params=params, timeout=45)
response.raise_for_status()
data = response.json()
private_json(target, data)
return data
def catalog(self, brands, offline=False):
tags = sorted({brand_key(b).replace(' ', '-') for b in brands if b})
if not tags:
return []
def group(tags):
query = 'brands_tags:(' + ' OR '.join(tags) + ')'
result = []
page = 1
while True:
data = self.fetch('https://search.openfoodfacts.org/search', {'q':query, 'page_size':1000, 'page':page, 'fields':FIELDS}, offline=offline)
if data.get('timed_out') or data.get('warnings'):
raise ValueError('Incomplete catalog search; retry later')
if not data.get('is_count_exact', False):
if len(tags) == 1:
raise ValueError('Brand catalog exceeds search limit: ' + tags[0])
middle = len(tags) // 2
return group(tags[:middle]) + group(tags[middle:])
for hit in data.get('hits', []):
hit = dict(hit)
if isinstance(hit.get('brands'), list):
hit['brands'] = ','.join(hit['brands'])
result.append(hit)
click.echo(f"Catalog ({len(tags)} brands) page {page}: {len(result)} of {data.get('count')} products")
if page >= data.get('page_count', 1):
if len(result) != data.get('count'):
raise ValueError('Incomplete catalog response')
return result
page += 1
# OFF also retains apostrophes as tag separators (e.g. nairn-s).
alternatives = sorted({re.sub(r'[^a-z0-9]+', '-', unicodedata.normalize('NFKD', b).encode('ascii', 'ignore').decode().lower()).strip('-') for b in brands if b} - set(tags))
result = group(tags)
if alternatives:
result += group(alternatives)
return list({str(p['code']):p for p in result}.values())
def product(self, code, offline=False):
data = self.fetch(f'/api/v3.6/product/{code}.json', {'fields':FIELDS + ',nutrition'}, offline=offline)
product = data.get('product')
if not product or str(product.get('code')) != code:
raise ValueError('Open Food Facts did not return requested barcode')
return product
def apply_match(client, product, candidate, barcodes):
"""Write only the barcode relation and description; never stock or name."""
code = str(candidate['code'])
if not valid_barcode(code):
raise ValueError('Invalid GTIN check digit')
owners = {int(b['product_id']) for b in barcodes if b['barcode'] == code}
if owners - {int(product['id'])}:
raise ValueError('Barcode already belongs to another Grocy product')
# Re-read just before mutation to retain user edits and skip deleted products.
current = client.get(f"/objects/products/{product['id']}")
if not current:
raise ValueError('Grocy product was deleted')
metadata = read_metadata(current.get('description'))
if not metadata or metadata.get('Ocado product ID') != read_metadata(product.get('description')).get('Ocado product ID'):
raise ValueError('Imported product identity changed')
if metadata.get('Barcode') and metadata['Barcode'] != code:
raise ValueError('Existing metadata barcode differs; review required')
enrichment = {k:v for k,v in candidate.items() if v not in (None, '', [], {})}
enrichment.update(source='Open Food Facts', url=f'https://world.openfoodfacts.org/product/{code}', database_license='ODbL', contents_license='Database Contents License', images_license='CC BY-SA')
metadata['Barcode'] = code
metadata['openfoodfacts'] = enrichment
updated = replace_metadata(current['description'], metadata)
if not owners:
row = {'product_id':int(product['id']), 'barcode':code, 'qu_id':current['qu_id_purchase'], 'amount':1}
client.post('/objects/product_barcodes', row)
barcodes.append(row)
if read_metadata(current['description']) != metadata:
client.request('PUT', f"/objects/products/{product['id']}", {'description':updated})
return 'updated'
return 'updated' if not owners else 'unchanged'
@click.command()
@click.option('--config', type=click.Path(path_type=Path), default=Path('config.toml'))
@click.option('--cache', type=click.Path(path_type=Path), default=Path('history/openfoodfacts'))
@click.option('--contact', help='Contact for the Open Food Facts User-Agent; defaults to config [openfoodfacts].contact.')
@click.option('--refresh-cache', is_flag=True, help='Fetch fresh OFF responses, replacing cached responses.')
@click.option('--apply', is_flag=True, help='Apply unambiguous matches and explicitly reviewed mappings.')
@click.option('--offline', is_flag=True, help='Use cached Open Food Facts responses only.')
@click.option('--mapping', type=click.Path(exists=True, path_type=Path), help='Reviewed JSON mapping of Grocy product ID to barcode.')
@click.option('--limit', type=int, help='Limit product comparisons after downloading the catalog.')
@click.option('--order-id', 'order_ids', multiple=True, help='Limit to products imported from these orders.')
def main(config, cache, contact, refresh_cache, apply, offline, mapping, limit, order_ids):
"""Match Ocado imports to Open Food Facts. Defaults to a read-only preview."""
settings = tomllib.loads(config.read_text())
contact = contact or settings.get('openfoodfacts', {}).get('contact')
if not contact:
raise click.UsageError('Set --contact or [openfoodfacts].contact in config.toml')
if refresh_cache and offline:
raise click.UsageError('--refresh-cache cannot be combined with --offline')
client = Grocy(**settings['grocy'])
state = config.parent / settings.get('import', {}).get('state_file', 'imports.sqlite3')
products = imported_products(client, state, order_ids)
mappings = json.loads(mapping.read_text()) if mapping else {}
return run_openfoodfacts(client, products, cache, contact, apply=apply, offline=offline,
refresh_cache=refresh_cache, mappings=mappings, limit=limit)
def run_openfoodfacts(client, products, cache, contact, *, apply=True, offline=False,
refresh_cache=False, mappings=None, limit=None):
barcodes = client.get('/objects/product_barcodes')
off = OpenFoodFacts(cache / 'responses', contact, refresh_cache)
mappings = mappings or {}
if not isinstance(mappings, dict) or any(not isinstance(v, str) or not valid_barcode(v) for v in mappings.values()):
raise click.UsageError('Mapping must be a JSON object with barcode strings and valid GTIN check digits')
def needs_search(product):
metadata = read_metadata(product.get('description'))
code = metadata.get('openfoodfacts', {}).get('code')
return not mappings.get(str(product['id'])) and not (valid_barcode(code) and code == metadata.get('Barcode'))
brands = {read_metadata(p.get('description')).get('ocado', {}).get('brand') for p in products if needs_search(p)}
catalog = off.catalog(brands, offline)
private_json(cache / 'catalog.json', catalog)
by_brand = {}
for candidate in catalog:
for brand in candidate.get('brands', '').split(','):
by_brand.setdefault(brand_key(brand), {})[str(candidate['code'])] = candidate
report = {'created_at':datetime.now(timezone.utc).isoformat(), 'apply':apply, 'products':[]}
if apply:
private_json(cache / ('before-' + datetime.now().strftime('%Y%m%d-%H%M%S') + '.json'), {'products':products, 'barcodes':barcodes})
searched = 0
for product in products:
row = {'product_id':product['id'], 'name':product['name']}
metadata = read_metadata(product.get('description'))
ocado = metadata.get('ocado', {})
row['size'] = ocado.get('size', {}).get('value', metadata.get('size',''))
try:
if not metadata:
row['status'] = 'missing_metadata'
elif set(ocado.get('categoryPath', [])) & NONFOOD or ocado.get('brand') in NONFOOD_BRANDS:
row['status'] = 'non_food'
elif limit is not None and searched >= limit:
row['status'] = 'not_searched'
else:
searched += 1
code = mappings.get(str(product['id']))
method = 'reviewed_mapping'
established = metadata.get('openfoodfacts', {}).get('code')
if not code and established == metadata.get('Barcode') and valid_barcode(established):
code = established
method = 'existing_match'
if code:
candidate = next((c for c in catalog if str(c.get('code')) == str(code)), None) or off.product(str(code), offline)
row['match_method'] = method
row['candidates'] = [candidate_info(product, metadata, candidate)]
else:
query = 'brand catalog: ' + ocado.get('brand', '')
result = {'count':len(catalog)}
candidates = by_brand.get(brand_key(ocado.get('brand', '')), {})
result['count'] = len(candidates)
ranked, candidate = choose_match(product, metadata, candidates)
row.update(query=query, search_count=result.get('count'), candidates=ranked[:10])
row['match_method'] = 'exact_brand_name_pack'
if candidate:
# Catalog is indexed separately; fetch current detail before writes.
if apply:
fresh = off.product(str(candidate['code']), offline)
row['current_candidate'] = candidate_info(product, metadata, fresh)
if not code and not row['current_candidate']['exact']:
raise ValueError('Current product details no longer match catalog')
candidate = fresh
row['barcode'] = candidate['code']
row['status'] = apply_match(client, product, candidate, barcodes) if apply else 'matched'
else:
row['status'] = 'review' if row['candidates'] else 'no_match'
except (requests.RequestException, ValueError, RuntimeError) as exc:
row.update(status='error', error=str(exc))
report['products'].append(row)
private_json(cache / ('applied-report.json' if apply else 'report.json'), report)
click.echo(f"{product['id']} {row['status']}: {product['name']}")
report['counts'] = {s:sum(r['status']==s for r in report['products']) for s in sorted({r['status'] for r in report['products']})}
private_json(cache / ('applied-report.json' if apply else 'report.json'), report)
click.echo(json.dumps(report['counts']))
return report
if __name__ == '__main__':
main()

View file

@ -0,0 +1,56 @@
"""Order-level completion, with conservative migration of the line journal."""
import json
import sqlite3
from types import SimpleNamespace
from .grocy import align_recorded_lines, item_fingerprint
from .pantry import classify
from .receipt import parse_document
def completed_orders(state, server, archive, config):
"""Read only: a downloaded receipt alone is never evidence of an import."""
if not state.exists():
return set()
with sqlite3.connect(f'{state.resolve().as_uri()}?mode=ro', uri=True) as db:
tables = {r[0] for r in db.execute("SELECT name FROM sqlite_master WHERE type='table'")}
tracked = dict(db.execute('SELECT order_id,status FROM loaded_orders WHERE server=?', (server,))) if 'loaded_orders' in tables else {}
completed = {order for order, status in tracked.items() if status == 'complete'}
if 'imports' not in tables:
return completed
legacy = {r[0] for r in db.execute('SELECT DISTINCT order_id FROM imports WHERE server=?', (server,))} - tracked.keys()
for order in legacy:
path = archive / f'{order}.json'
if not path.exists():
continue
try:
receipt = parse_document(json.loads(path.read_text()))
if receipt.order_id != order:
continue
receipt = align_recorded_lines(receipt, server, SimpleNamespace(db=db))
rows = {line:(fingerprint,status) for line,fingerprint,status in db.execute(
'SELECT line,fingerprint,status FROM imports WHERE server=? AND order_id=?', (server,order))}
if any(status != 'done' for _,status in rows.values()):
continue
selected = [(line,item) for line,item in enumerate(receipt.items,1)
if classify(item,config.get('pantry',{}).get('overrides',{})).include]
if all(rows.get(line) == (item_fingerprint(receipt,item),'done') for line,item in selected):
completed.add(order)
except (ValueError, KeyError, TypeError):
# A changed or incomplete receipt needs the normal reconciliation path.
continue
return completed
def mark_orders(state, server, orders, status):
if not orders:
return
state.parent.mkdir(parents=True, exist_ok=True)
with sqlite3.connect(state) as db:
db.execute('''CREATE TABLE IF NOT EXISTS loaded_orders (
server TEXT NOT NULL, order_id TEXT NOT NULL, status TEXT NOT NULL,
updated_at TEXT NOT NULL DEFAULT CURRENT_TIMESTAMP,
PRIMARY KEY(server,order_id))''')
db.executemany('''INSERT INTO loaded_orders(server,order_id,status) VALUES (?,?,?)
ON CONFLICT(server,order_id) DO UPDATE SET status=excluded.status, updated_at=CURRENT_TIMESTAMP''',
[(server,order,status) for order in orders])

45
ocado_grocy/pantry.py Normal file
View file

@ -0,0 +1,45 @@
"""Select shelf-stable stock using Ocado metadata, without a product-name list."""
from dataclasses import dataclass
@dataclass(frozen=True)
class Decision:
include: bool
reason: str
uncertain: bool = False
PERISHABLE_CATEGORIES = {"fresh & chilled food", "frozen food", "bakery"}
SHELF_STABLE_CATEGORIES = {
"food cupboard", "treats & snacks", "soft drinks, tea & coffee", "beer, wine & spirits",
"household & cleaning", "home care & cleaning", "health, beauty & personal care",
"beauty & toiletries", "pets", "health, medicines & wellbeing", "home & garden",
"m&s food cupboard", "m&s snacks & treats",
}
def classify(item, overrides=None):
overrides = overrides or {}
key = item.product_id or item.name
if key in overrides:
value = overrides[key]
if not isinstance(value, bool):
raise ValueError(f"pantry override {key!r} must be true or false")
return Decision(value, "explicit product override")
storage_type = item.metadata.get("storage_type")
if storage_type in {"FRIDGE", "FREEZER"}:
return Decision(False, "requires cold storage")
product = item.metadata.get("ocado", {})
storage = str(product.get("storage", "")).casefold()
if "keep refrigerated" in storage or "keep frozen" in storage:
return Decision(False, "requires cold storage")
categories = product.get("categoryPath", [])
if isinstance(categories, str):
categories = [categories]
categories = {str(category).strip().casefold() for category in categories}
if categories & PERISHABLE_CATEGORIES:
return Decision(False, "fresh, chilled, frozen or bakery category")
if categories & SHELF_STABLE_CATEGORIES:
return Decision(True, "shelf-stable product category")
# CUPBOARD alone also describes bananas, onions and potatoes.
return Decision(False, "missing or ambiguous product category", uncertain=True)

112
ocado_grocy/postprocess.py Normal file
View file

@ -0,0 +1,112 @@
"""Enrich products belonging to the orders just processed, including reruns."""
import hashlib
import os
from pathlib import Path
import click
from .grocy import Grocy, imported_products
from .history import private_json
from .images import run_images
from .openfoodfacts import run_openfoodfacts
from .enrichment_retry import RetryQueue
def enrich_orders(config, config_path, archive, order_ids, *, images=True, openfoodfacts=True):
if not order_ids or not (images or openfoodfacts):
return {'errors':[]}
connection = config.get('grocy', {})
client = Grocy(os.environ.get('GROCY_URL', connection.get('url', '')),
os.environ.get('GROCY_API_KEY', connection.get('api_key', '')))
state = Path(config.get('import', {}).get('state_file', 'imports.sqlite3'))
if not state.is_absolute():
state = config_path.resolve().parent / state
products = imported_products(client, state, order_ids)
key = order_ids[0] if len(order_ids) == 1 and order_ids[0].isdigit() else hashlib.sha256(','.join(sorted(order_ids)).encode()).hexdigest()[:16]
output = archive / 'enrichment' / key
queue = RetryQueue(state, client.url)
summary = {'server':client.url, 'order_ids':list(order_ids), 'product_ids':[p['id'] for p in products], 'errors':[]}
if not products:
queue.close()
private_json(output / 'summary.json', summary)
return summary
if images:
try:
click.echo(f'Adding missing pictures for {len(products)} imported products...')
queue.start('images', products)
result = run_images(client, products, output / 'images.json', raise_errors=False)
queue.results('images', result['products'])
summary['images'] = result['counts']
if result['counts'].get('error'):
summary['errors'].append(f"Images: {result['counts']['error']} products failed")
except Exception as exc:
summary['errors'].append(f'Images: {exc}')
if openfoodfacts:
try:
queue.start('openfoodfacts', products)
contact = config.get('openfoodfacts', {}).get('contact')
if not contact:
raise ValueError('Set [openfoodfacts].contact in config.toml or use --no-openfoodfacts')
click.echo(f'Matching {len(products)} imported products on Open Food Facts...')
result = run_openfoodfacts(client, products, archive / 'openfoodfacts', contact)
queue.results('openfoodfacts', result['products'])
private_json(output / 'openfoodfacts.json', result)
summary['openfoodfacts'] = result['counts']
if result['counts'].get('error'):
summary['errors'].append(f"Open Food Facts: {result['counts']['error']} products failed; see {output / 'openfoodfacts.json'}")
except Exception as exc:
summary['errors'].append(f'Open Food Facts: {exc}')
queue.close()
private_json(output / 'summary.json', summary)
return summary
def retry_enrichment(config, config_path, archive, *, images=True, openfoodfacts=True):
connection = config.get('grocy', {})
client = Grocy(os.environ.get('GROCY_URL', connection.get('url','')),
os.environ.get('GROCY_API_KEY', connection.get('api_key','')))
state = config_path.resolve().parent / config.get('import',{}).get('state_file','imports.sqlite3')
if not state.exists():
return {'errors':[]}
queue = RetryQueue(state, client.url)
try:
# Journal scope protects other Grocy installations and non-imported products.
tables = {r[0] for r in queue.db.execute("SELECT name FROM sqlite_master WHERE type='table'")}
if 'imports' not in tables:
return {'errors':[]}
allowed = {int(r[0]) for r in queue.db.execute("SELECT DISTINCT product_id FROM imports WHERE server=? AND status='done'",(client.url,))}
queue.migrate_reports(archive, allowed)
enabled = [('images',images),('openfoodfacts',openfoodfacts)]
summary = {'errors':[]}
live = None
for stage, active in enabled:
pending = queue.pending(stage)
if not active or not pending:
continue
if live is None:
live = {int(p['id']):p for p in client.get('/objects/products')}
products = [live[pid] for pid in sorted(pending & allowed & live.keys())]
queue.results(stage,[{'product_id':pid,'status':'deleted_or_out_of_scope'} for pid in pending - {int(p['id']) for p in products}])
if not products:
continue
click.echo(f'Retrying {stage} for {len(products)} previously failed products...')
output = archive / 'enrichment' / 'retries'
try:
if stage == 'images':
result = run_images(client, products, output/'images.json', raise_errors=False)
else:
contact = config.get('openfoodfacts',{}).get('contact')
if not contact:
raise ValueError('Set [openfoodfacts].contact in config.toml')
result = run_openfoodfacts(client, products, archive/'openfoodfacts',contact)
private_json(output/'openfoodfacts.json',result)
queue.results(stage,result['products'])
summary[stage] = result['counts']
if result['counts'].get('error'):
summary['errors'].append(f"{stage}: {result['counts']['error']} retries failed")
except Exception as exc:
summary['errors'].append(f'{stage}: {exc}')
private_json(archive/'enrichment'/'retry-summary.json',summary)
return summary
finally:
queue.close()

View file

@ -0,0 +1,38 @@
"""Read imported metadata even when Grocy's sanitizer removes the pre tag."""
import json
from html import escape
from lxml import html
def read_metadata(description):
try:
text = html.fromstring(description or '<p></p>').text_content()
except (html.etree.ParserError, ValueError):
return {}
for index, character in enumerate(text):
if character == '{':
try:
value, _ = json.JSONDecoder().raw_decode(text[index:])
if isinstance(value, dict) and 'Ocado product ID' in value:
return value
except ValueError:
pass
return {}
def replace_metadata(description, metadata):
tree = html.fragment_fromstring(description, create_parent='div')
for node in tree.iter():
for attr in ('text', 'tail'):
text = getattr(node, attr) or ''
for index, character in enumerate(text):
if character != '{':
continue
try:
old, end = json.JSONDecoder().raw_decode(text[index:])
except ValueError:
continue
if isinstance(old, dict) and 'Ocado product ID' in old:
setattr(node, attr, text[:index] + json.dumps(metadata, ensure_ascii=False, indent=2) + text[index + end:])
return escape(tree.text or '') + ''.join(html.tostring(child, encoding='unicode') for child in tree)
raise ValueError('Cannot safely locate imported metadata in description')

161
ocado_grocy/receipt.py Normal file
View file

@ -0,0 +1,161 @@
"""Parse receipt data only: never treat the live basket as an order."""
from dataclasses import asdict, dataclass, field
from datetime import date, datetime
from zoneinfo import ZoneInfo
from decimal import Decimal, InvalidOperation
import json
class ImportError(ValueError):
"""An input or integration error suitable for displaying to the user."""
def number(value, label, *, money=False):
if isinstance(value, dict):
currency = value.get("currency", "GBP")
if currency not in ("GBP", "GBX"):
raise ImportError(f"Unsupported currency: {currency}")
result = number(value.get("amount"), label, money=money)
return result / 100 if currency == "GBX" else result
text = str(value).strip().replace("£", "").replace(",", "")
try:
result = Decimal(text)
except InvalidOperation as exc:
raise ImportError(f"Invalid {label}: {value!r}") from exc
if not result.is_finite() or result < 0:
raise ImportError(f"Invalid {label}: {value!r}")
return result
def receipt_date(value):
if not value:
return None
text = str(value).strip()
try:
return date.fromisoformat(text[:10]).isoformat()
except ValueError:
pass
for fmt in ("%d/%m/%Y", "%d %B %Y", "%d %b %Y", "%A %d %B %Y"):
try:
return datetime.strptime(text, fmt).date().isoformat()
except ValueError:
pass
raise ImportError(f"Unrecognised date: {value!r}; use YYYY-MM-DD")
@dataclass
class Item:
name: str
quantity: Decimal
total: Decimal
product_id: str = ""
barcode: str = ""
url: str = ""
best_before: str | None = None
metadata: dict = field(default_factory=dict)
@property
def unit_price(self):
return self.total / self.quantity
@dataclass
class Receipt:
order_id: str
purchased_date: str | None
items: list[Item]
metadata: dict = field(default_factory=dict)
def export(self):
return asdict(self)
def parse_document(data):
"""Read a normalized cached receipt."""
if not isinstance(data, dict) or not isinstance(data.get("items"), list):
raise ImportError("Receipt JSON requires an items array")
if data.get("currency", "GBP") != "GBP":
raise ImportError("Only GBP receipts are supported")
order_id = str(data.get("order_id") or "").strip()
if not order_id:
raise ImportError("Receipt has no order ID")
items = []
for row in data["items"]:
if not isinstance(row, dict):
raise ImportError("Receipt items must be objects")
status = str(row.get("status", "")).lower()
if status in {"unavailable", "cancelled", "rejected", "not delivered"}:
continue
quantity = number(row.get("delivered_quantity", row.get("quantity")), "quantity")
if quantity == 0:
continue
name = str(row.get("name") or "").strip()
if not name:
raise ImportError("Receipt item has no name")
total = number(row.get("total"), f"line total for {name}", money=True)
metadata = dict(row.get("metadata") or {})
metadata.update({k: v for k, v in row.items() if k not in {
"name", "quantity", "delivered_quantity", "total", "product_id",
"barcode", "url", "best_before", "metadata"}})
items.append(Item(name, quantity, total, str(row.get("product_id") or ""),
str(row.get("barcode") or ""), str(row.get("url") or ""),
receipt_date(row.get("best_before")), metadata))
if not items:
raise ImportError("Receipt contains no delivered items")
return Receipt(order_id, receipt_date(data.get("purchased_date")), items, data.get("metadata", {}))
def product_metadata(product):
return {key: product[key] for key in ("brand", "categoryPath", "size", "packSize",
"description", "ingredients", "nutritionalInformation", "allergens", "storage",
"guaranteedProductLife", "retailerProductId") if key in product}
def parse_ocado_order(data, *, expected_order_id=None, product_urls=None):
"""Parse the order JSON loaded by Playwright, never the page's live basket."""
try:
order = data["entities"]["order"][data["result"]]
order_id = str(order["orderId"])
if expected_order_id and order_id != str(expected_order_id):
raise ImportError("Ocado API order does not match the requested order")
if order["status"] != "DELIVERED":
raise ImportError("Ocado order is not delivered")
dates = order["dates"]
delivered_at = datetime.fromisoformat(dates["deliveryStartDate"].replace("Z", "+00:00"))
if delivered_at.tzinfo is not None:
delivered_at = delivered_at.astimezone(ZoneInfo(dates.get("timeZoneId", "Europe/London")))
purchased_date = delivered_at.date().isoformat()
groups = order["groupedProducts"]
products = list(groups["products"])
for original in groups.get("substitutes", []):
for replacement in original.get("substitutes", []):
if replacement.get("status") == "ACCEPTED":
products.append({"storageType": original.get("storageType"),
"substituted_for": original["name"], **replacement})
rows = []
catalogue = data["entities"].get("product", {})
for product in products:
info = catalogue.get(product["productId"], {})
retailer_id = str(product["retailerProductId"])
metadata = {"ocado": product_metadata(info), "storage_type": product.get("storageType"),
"pack_info": product.get("packInfo", {}), "promotions": product.get("promotions", []),
"original_line_price": product["prices"].get("retail"),
"price_per_item": product["prices"].get("pricePerItem"),
"image_url": (product.get("image") or {}).get("src", "")}
if product.get("substituted_for"):
metadata["substituted_for"] = product["substituted_for"]
rows.append({"name": " ".join(product["name"].split()), "quantity": product["quantity"],
"total": product["prices"]["offered"], "product_id": retailer_id,
"best_before": product.get("expirationDate"),
"url": (product_urls or {}).get(retailer_id, ""), "metadata": metadata})
# Stable ordering is independent of the browser's equal-expiry sorting.
rows.sort(key=lambda row:(row["best_before"] or "9999-12-31", row["product_id"], row["name"]))
receipt = parse_document({"order_id":order_id,"purchased_date":purchased_date,"items":rows})
if sum(item.quantity for item in receipt.items) != number(order["totalItems"], "receipt item count"):
raise ImportError("Delivered quantities do not match the order summary")
if sum(item.total for item in receipt.items) != number(order["orderTotals"]["itemPriceAfterPromos"], "receipt product total"):
raise ImportError("Delivered line prices do not match the order summary")
receipt.metadata = {"source":"ocado-order-api", "order_totals":order["orderTotals"]}
return receipt
except (KeyError, TypeError, AttributeError) as exc:
raise ImportError("Unrecognised Ocado order API structure") from exc