Initial commit.
This commit is contained in:
commit
628e5c3823
26 changed files with 3478 additions and 0 deletions
1
ocado_grocy/__init__.py
Normal file
1
ocado_grocy/__init__.py
Normal file
|
|
@ -0,0 +1 @@
|
|||
"""Ocado receipt importer."""
|
||||
3
ocado_grocy/__main__.py
Normal file
3
ocado_grocy/__main__.py
Normal file
|
|
@ -0,0 +1,3 @@
|
|||
from .cli import main
|
||||
|
||||
main()
|
||||
4
ocado_grocy/cli.py
Normal file
4
ocado_grocy/cli.py
Normal file
|
|
@ -0,0 +1,4 @@
|
|||
"""Main command: acquire orders through Playwright."""
|
||||
from .history import main
|
||||
|
||||
__all__ = ["main"]
|
||||
53
ocado_grocy/enrichment_retry.py
Normal file
53
ocado_grocy/enrichment_retry.py
Normal file
|
|
@ -0,0 +1,53 @@
|
|||
"""Durable product-level retries, independent of completed stock orders."""
|
||||
import json
|
||||
import sqlite3
|
||||
|
||||
|
||||
class RetryQueue:
|
||||
def __init__(self, state, server):
|
||||
self.server = server
|
||||
self.db = sqlite3.connect(state)
|
||||
with self.db:
|
||||
self.db.execute('''CREATE TABLE IF NOT EXISTS enrichment_retries (
|
||||
server TEXT, stage TEXT, product_id INTEGER, error TEXT,
|
||||
PRIMARY KEY(server,stage,product_id))''')
|
||||
self.db.execute('CREATE TABLE IF NOT EXISTS enrichment_retry_migrations (server TEXT PRIMARY KEY)')
|
||||
|
||||
def start(self, stage, products):
|
||||
self.results(stage, [{'product_id':p['id'], 'status':'error', 'error':'Interrupted enrichment'} for p in products])
|
||||
|
||||
def results(self, stage, rows):
|
||||
with self.db:
|
||||
for row in rows:
|
||||
key = (self.server,stage,int(row['product_id']))
|
||||
if row['status'] == 'error':
|
||||
self.db.execute('INSERT OR REPLACE INTO enrichment_retries VALUES (?,?,?,?)', (*key,row.get('error','Enrichment failed')))
|
||||
else:
|
||||
self.db.execute('DELETE FROM enrichment_retries WHERE server=? AND stage=? AND product_id=?',key)
|
||||
|
||||
def pending(self, stage):
|
||||
return {r[0] for r in self.db.execute('SELECT product_id FROM enrichment_retries WHERE server=? AND stage=?',(self.server,stage))}
|
||||
|
||||
def migrate_reports(self, archive, allowed_ids):
|
||||
"""Import old errors once; newer successful reports supersede old failures."""
|
||||
if self.db.execute('SELECT 1 FROM enrichment_retry_migrations WHERE server=?',(self.server,)).fetchone():
|
||||
return
|
||||
events = []
|
||||
for path in (archive/'enrichment').glob('*/summary.json'):
|
||||
summary = json.loads(path.read_text())
|
||||
if summary.get('server',self.server) != self.server:
|
||||
continue
|
||||
for stage in ('images','openfoodfacts'):
|
||||
report_path = path.parent / f'{stage}.json'
|
||||
if report_path.exists():
|
||||
data = json.loads(report_path.read_text())
|
||||
events.append((report_path.stat().st_mtime,stage,data.get('products',[])))
|
||||
elif any(error.lower().startswith('images:' if stage=='images' else 'open food facts:') for error in summary.get('errors',[])):
|
||||
events.append((path.stat().st_mtime,stage,[{'product_id':pid,'status':'error','error':'Previous enrichment failed'} for pid in summary.get('product_ids',[])]))
|
||||
for _,stage,rows in sorted(events,key=lambda event:event[0]):
|
||||
self.results(stage,[r for r in rows if int(r['product_id']) in allowed_ids])
|
||||
with self.db:
|
||||
self.db.execute('INSERT INTO enrichment_retry_migrations VALUES (?)',(self.server,))
|
||||
|
||||
def close(self):
|
||||
self.db.close()
|
||||
304
ocado_grocy/grocy.py
Normal file
304
ocado_grocy/grocy.py
Normal file
|
|
@ -0,0 +1,304 @@
|
|||
"""Small Grocy API client and resumable importer."""
|
||||
from decimal import Decimal
|
||||
from dataclasses import replace
|
||||
import hashlib
|
||||
import html
|
||||
import json
|
||||
import re
|
||||
from pathlib import Path
|
||||
import sqlite3
|
||||
|
||||
import requests
|
||||
from .product_metadata import read_metadata
|
||||
|
||||
from .receipt import ImportError, number, receipt_date
|
||||
|
||||
|
||||
def normalized(name):
|
||||
return " ".join(name.casefold().split())
|
||||
|
||||
|
||||
class Grocy:
|
||||
def __init__(self, url, api_key, timeout=30):
|
||||
self.url = url.rstrip("/")
|
||||
if not self.url.endswith("/api"):
|
||||
self.url += "/api"
|
||||
self.session = requests.Session()
|
||||
self.session.headers.update({"GROCY-API-KEY": api_key,
|
||||
"User-Agent": "ocado-grocy/0.1"})
|
||||
self.timeout = timeout
|
||||
|
||||
def request(self, method, path, data=None):
|
||||
try:
|
||||
response = self.session.request(method, self.url + path, json=data,
|
||||
timeout=self.timeout, allow_redirects=False)
|
||||
except requests.RequestException as exc:
|
||||
raise ImportError(f"Grocy {method} {path} failed ({type(exc).__name__}); "
|
||||
"check connectivity before retrying") from exc
|
||||
if not 200 <= response.status_code < 300:
|
||||
raise ImportError(f"Grocy {method} {path}: HTTP {response.status_code}")
|
||||
try:
|
||||
return response.json() if response.content else None
|
||||
except ValueError as exc:
|
||||
raise ImportError(f"Grocy {method} {path} returned non-JSON data") from exc
|
||||
|
||||
def get(self, path):
|
||||
return self.request("GET", path)
|
||||
|
||||
def post(self, path, data):
|
||||
return self.request("POST", path, data)
|
||||
|
||||
def named_object(self, entity, name, **fields):
|
||||
matches = [x for x in self.get(f"/objects/{entity}")
|
||||
if normalized(x["name"]) == normalized(name)]
|
||||
if len(matches) > 1:
|
||||
raise ImportError(f"Multiple Grocy {entity} named {name!r}")
|
||||
if matches:
|
||||
return int(matches[0]["id"])
|
||||
return int(self.post(f"/objects/{entity}", {"name": name, **fields})["created_object_id"])
|
||||
|
||||
|
||||
class Journal:
|
||||
"""Commit intent before each write; ambiguous writes require explicit reconciliation."""
|
||||
def __init__(self, path: Path):
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
self.db = sqlite3.connect(path)
|
||||
self.db.execute("""CREATE TABLE IF NOT EXISTS imports (
|
||||
server TEXT, order_id TEXT, line INTEGER, fingerprint TEXT,
|
||||
status TEXT, product_id INTEGER, PRIMARY KEY(server, order_id, line))""")
|
||||
self.db.commit()
|
||||
|
||||
def check(self, server, order, line, fingerprint):
|
||||
row = self.db.execute("SELECT fingerprint, status FROM imports WHERE server=? AND order_id=? AND line=?",
|
||||
(server, order, line)).fetchone()
|
||||
if not row:
|
||||
return False
|
||||
if row[0] != fingerprint:
|
||||
raise ImportError(f"Order {order} line {line} changed since a previous import; reconcile it manually")
|
||||
if row[1] != "done":
|
||||
raise ImportError(f"Order {order} line {line} has an uncertain previous write. "
|
||||
"Check Grocy stock/logs and reconcile the journal before retrying (see README).")
|
||||
return True
|
||||
|
||||
def record(self, server, order, line, fingerprint, status, product_id):
|
||||
with self.db:
|
||||
self.db.execute("INSERT OR REPLACE INTO imports VALUES (?, ?, ?, ?, ?, ?)",
|
||||
(server, order, line, fingerprint, status, product_id))
|
||||
|
||||
|
||||
def marker(item):
|
||||
return f"Ocado product ID: {item.product_id}" if item.product_id else ""
|
||||
|
||||
|
||||
def description(item):
|
||||
details = {"Ocado product ID": item.product_id, "Barcode": item.barcode,
|
||||
"Product URL": item.url, **item.metadata}
|
||||
return "<p>" + html.escape(marker(item) or "Imported from Ocado") + "</p><pre>" + html.escape(
|
||||
json.dumps(details, ensure_ascii=False, indent=2, default=str)) + "</pre>"
|
||||
|
||||
|
||||
def match_product(item, products, barcodes, mappings):
|
||||
override = mappings.get(item.product_id or item.name)
|
||||
if override is not None:
|
||||
matches = [p for p in products if int(p["id"]) == int(override["product_id"])]
|
||||
if not matches:
|
||||
raise ImportError(f"Mapped Grocy product does not exist: {item.name}")
|
||||
return matches[0], override
|
||||
matches = []
|
||||
if item.product_id:
|
||||
matches = [p for p in products if f"<p>{html.escape(marker(item))}</p>" in (p.get("description") or "")]
|
||||
if not matches and item.barcode:
|
||||
ids = {int(b["product_id"]) for b in barcodes if b["barcode"] == item.barcode}
|
||||
matches = [p for p in products if int(p["id"]) in ids]
|
||||
if not matches:
|
||||
matches = [p for p in products if normalized(p["name"]) == normalized(item.name)]
|
||||
if len(matches) > 1:
|
||||
raise ImportError(f"Ambiguous existing product: {item.name}; add a config product mapping")
|
||||
return (matches[0] if matches else None), {}
|
||||
|
||||
|
||||
def item_fingerprint(receipt, item):
|
||||
return hashlib.sha256(json.dumps({
|
||||
"name": item.name, "product_id": item.product_id, "quantity": str(item.quantity.normalize()),
|
||||
"total": str(item.total.normalize()), "best_before": item.best_before,
|
||||
"date": receipt.purchased_date}, sort_keys=True).encode()).hexdigest()
|
||||
|
||||
|
||||
def align_recorded_lines(receipt, server, journal):
|
||||
"""Restore journal positions when Ocado reorders equal-expiry receipt rows."""
|
||||
records = journal.db.execute(
|
||||
"SELECT line, fingerprint FROM imports WHERE server=? AND order_id=? ORDER BY line",
|
||||
(server, receipt.order_id)).fetchall()
|
||||
remaining = list(receipt.items)
|
||||
aligned = [None] * len(remaining)
|
||||
for line, fingerprint in records:
|
||||
candidates = [i for i,item in enumerate(remaining) if item_fingerprint(receipt,item) == fingerprint]
|
||||
if not candidates or not 1 <= line <= len(aligned):
|
||||
raise ImportError(f"Order {receipt.order_id} line {line} changed since its import; review before continuing")
|
||||
aligned[line - 1] = remaining.pop(candidates[0])
|
||||
rest = iter(remaining)
|
||||
return replace(receipt, items=[item if item is not None else next(rest) for item in aligned])
|
||||
|
||||
|
||||
def refresh_imported_products(receipt, client, journal, *, seen=None):
|
||||
"""Refresh details of journal-linked products, including excluded perishables."""
|
||||
seen = seen if seen is not None else set()
|
||||
receipt = align_recorded_lines(receipt, client.url, journal)
|
||||
records = journal.db.execute(
|
||||
"SELECT line, product_id FROM imports WHERE server=? AND order_id=? AND status='done' ORDER BY line",
|
||||
(client.url, receipt.order_id)).fetchall()
|
||||
count = 0
|
||||
for line, product_id in records:
|
||||
if product_id in seen:
|
||||
continue
|
||||
if not 1 <= line <= len(receipt.items):
|
||||
raise ImportError(f"Receipt line {line} no longer exists; cannot refresh order {receipt.order_id}")
|
||||
item = receipt.items[line - 1]
|
||||
product = client.get(f"/objects/products/{product_id}")
|
||||
old = product.get("description") or ""
|
||||
expected = f"<p>{html.escape(marker(item))}</p>" if item.product_id else ""
|
||||
if expected and expected not in old:
|
||||
raise ImportError(f"Product identity changed for order {receipt.order_id}, line {line}; review before refreshing")
|
||||
if not expected and normalized(product["name"]) != normalized(item.name):
|
||||
raise ImportError(f"Product name changed for order {receipt.order_id}, line {line}")
|
||||
metadata = read_metadata(old)
|
||||
for key, value in item.metadata.items():
|
||||
if value not in (None, "", {}, []):
|
||||
if isinstance(value, dict) and isinstance(metadata.get(key), dict):
|
||||
metadata[key] = {**metadata[key], **value}
|
||||
else:
|
||||
metadata[key] = value
|
||||
barcode = item.barcode or metadata.pop("Barcode", "")
|
||||
url = item.url or metadata.pop("Product URL", "")
|
||||
for key in ("Ocado product ID", "Barcode", "Product URL"):
|
||||
metadata.pop(key, None)
|
||||
updated = description(replace(item, metadata=metadata, barcode=barcode, url=url))
|
||||
if old != updated:
|
||||
client.request("PUT", f"/objects/products/{product_id}", {"description": updated})
|
||||
count += 1
|
||||
seen.add(product_id)
|
||||
return count
|
||||
|
||||
|
||||
def import_receipt(receipt, client, journal, config, echo=print, *, include_lines=None):
|
||||
if not receipt.purchased_date:
|
||||
raise ImportError("No purchase/delivery date found; supply --purchased-date YYYY-MM-DD")
|
||||
products = client.get("/objects/products")
|
||||
barcodes = client.get("/objects/product_barcodes")
|
||||
plans = []
|
||||
# Validate every existing product/conversion before any mutations.
|
||||
for line, item in enumerate(receipt.items, 1):
|
||||
if include_lines is not None and line not in include_lines:
|
||||
continue
|
||||
fingerprint = item_fingerprint(receipt, item)
|
||||
if journal.check(client.url, receipt.order_id, line, fingerprint):
|
||||
echo(f"Skipped already imported: {item.name}")
|
||||
continue
|
||||
product, override = match_product(item, products, barcodes, config.get("products", {}))
|
||||
factor = number(override.get("stock_per_purchase", 1), "stock conversion")
|
||||
if product:
|
||||
if product.get("enable_tare_weight_handling") or product.get("no_own_stock"):
|
||||
raise ImportError(f"Unsupported tare-weight or parent-only product: {item.name}")
|
||||
if "stock_per_purchase" not in override and product["qu_id_purchase"] != product["qu_id_stock"]:
|
||||
details = client.get(f'/stock/products/{product["id"]}')
|
||||
factor = number(details.get("qu_conversion_factor_purchase_to_stock"), "stock conversion")
|
||||
if factor <= 0:
|
||||
raise ImportError(f"Stock conversion must be positive: {item.name}")
|
||||
plans.append((line, item, fingerprint, product, factor))
|
||||
if not plans:
|
||||
return 0
|
||||
location = client.named_object("locations", config.get("location", "Ocado imports"))
|
||||
unit = client.named_object("quantity_units", config.get("quantity_unit", "Pack"), name_plural="Packs")
|
||||
store = client.named_object("shopping_locations", "Ocado")
|
||||
count = 0
|
||||
for line, item, fingerprint, product, factor in plans:
|
||||
if product is None:
|
||||
# Recheck the local list: the same product can occur on multiple receipt lines.
|
||||
product, _ = match_product(item, products, barcodes, config.get("products", {}))
|
||||
if product is None:
|
||||
payload = {"name": item.name, "description": description(item), "location_id": location,
|
||||
"shopping_location_id": store, "qu_id_purchase": unit, "qu_id_stock": unit,
|
||||
"qu_id_consume": unit, "qu_id_price": unit}
|
||||
product_id = int(client.post("/objects/products", payload)["created_object_id"])
|
||||
product = {"id": product_id, **payload}
|
||||
products.append(product)
|
||||
if item.barcode:
|
||||
client.post("/objects/product_barcodes", {"product_id": product_id, "barcode": item.barcode,
|
||||
"qu_id": unit, "amount": 1})
|
||||
product_id = int(product["id"])
|
||||
amount = item.quantity * factor
|
||||
note = f"Ocado {receipt.purchased_date}"
|
||||
if not item.best_before:
|
||||
note += " Expiry unknown."
|
||||
payload = {"amount": float(amount), "price": float(item.total / amount),
|
||||
"best_before_date": item.best_before or "2999-12-31",
|
||||
"purchased_date": receipt.purchased_date, "transaction_type": "purchase",
|
||||
"shopping_location_id": store, "stock_label_type": 1, "note": note}
|
||||
journal.record(client.url, receipt.order_id, line, fingerprint, "pending", product_id)
|
||||
client.post(f"/stock/products/{product_id}/add", payload)
|
||||
journal.record(client.url, receipt.order_id, line, fingerprint, "done", product_id)
|
||||
echo(f"Imported {item.quantity} × {item.name}: £{item.total:.2f}")
|
||||
count += 1
|
||||
return count
|
||||
|
||||
|
||||
def update_existing_notes(client, journal):
|
||||
"""Edit only surviving journal-linked stock entries; never book new stock."""
|
||||
imported = {(str(order), int(line), int(product)) for order, line, product in journal.db.execute(
|
||||
"SELECT order_id, line, product_id FROM imports WHERE server=? AND status='done'", (client.url,))}
|
||||
changed = {}
|
||||
preserved_fields = ("amount", "best_before_date", "price", "open", "location_id",
|
||||
"shopping_location_id", "purchased_date")
|
||||
for row in client.get("/objects/stock"):
|
||||
old = row.get("note") or ""
|
||||
match = re.match(r"Ocado order (\d+), line (\d+)[.;]", old)
|
||||
dated = re.fullmatch(r"Ocado (\d{4}-\d{2}-\d{2}), line (\d+)\.(?: Expiry unknown\.)?", old)
|
||||
if match:
|
||||
linked = (match[1], int(match[2]), int(row["product_id"])) in imported
|
||||
elif dated:
|
||||
linked = any(line == int(dated[2]) and product == int(row["product_id"])
|
||||
for _, line, product in imported)
|
||||
else:
|
||||
linked = False
|
||||
if not linked:
|
||||
continue
|
||||
try:
|
||||
current = client.get(f"/stock/entry/{row['id']}")
|
||||
except ImportError:
|
||||
if not any(entry["id"] == row["id"] for entry in client.get("/objects/stock")):
|
||||
continue
|
||||
raise
|
||||
if current is None or current.get("note") != old:
|
||||
continue
|
||||
row = current
|
||||
purchased = receipt_date(row["purchased_date"])
|
||||
if not purchased:
|
||||
raise ImportError(f"Stock entry {row['id']} has no purchase date")
|
||||
note = f"Ocado {purchased}"
|
||||
if "Expiry unknown" in old:
|
||||
note += " Expiry unknown."
|
||||
payload = {field: row[field] for field in preserved_fields}
|
||||
try:
|
||||
client.request("PUT", f"/stock/entry/{row['id']}", {**payload, "note": note})
|
||||
except ImportError:
|
||||
if not any(entry["id"] == row["id"] for entry in client.get("/objects/stock")):
|
||||
continue
|
||||
raise
|
||||
changed[row["id"]] = {**payload, "note": note}
|
||||
for row in client.get("/objects/stock"):
|
||||
if row["id"] in changed:
|
||||
if row["note"] != changed[row["id"]]["note"]:
|
||||
raise ImportError(f"Verification failed for stock entry {row['id']}")
|
||||
return len(changed)
|
||||
|
||||
|
||||
def imported_products(client, state, order_ids=()):
|
||||
"""Live products from completed journal lines, optionally scoped to orders."""
|
||||
query = "SELECT DISTINCT product_id FROM imports WHERE server=? AND status='done'"
|
||||
params = [client.url]
|
||||
if order_ids:
|
||||
query += ' AND order_id IN (' + ','.join('?' for _ in order_ids) + ')'
|
||||
params.extend(order_ids)
|
||||
with sqlite3.connect(f'{Path(state).resolve().as_uri()}?mode=ro', uri=True) as db:
|
||||
ids = {int(row[0]) for row in db.execute(query, params)}
|
||||
return [p for p in client.get('/objects/products') if int(p['id']) in ids]
|
||||
326
ocado_grocy/history.py
Normal file
326
ocado_grocy/history.py
Normal file
|
|
@ -0,0 +1,326 @@
|
|||
"""Browse Ocado history with a persistent, visible Playwright browser."""
|
||||
from dataclasses import asdict, dataclass
|
||||
from datetime import datetime
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
import re
|
||||
import sqlite3
|
||||
import time
|
||||
import tomllib
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
import click
|
||||
from lxml import html
|
||||
|
||||
from .grocy import Grocy, Journal, import_receipt, refresh_imported_products, align_recorded_lines
|
||||
from .pantry import classify
|
||||
from .receipt import ImportError, parse_document, parse_ocado_order, receipt_date
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class OrderLink:
|
||||
order_id: str
|
||||
purchased_date: str
|
||||
url: str
|
||||
|
||||
|
||||
def private_json(path, data):
|
||||
path.parent.mkdir(parents=True, exist_ok=True, mode=0o700)
|
||||
temporary = path.with_suffix(path.suffix + ".tmp")
|
||||
fd = os.open(temporary, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600)
|
||||
with os.fdopen(fd, "w") as stream:
|
||||
json.dump(data, stream, ensure_ascii=False, indent=2, default=str)
|
||||
temporary.replace(path)
|
||||
|
||||
|
||||
def order_links(content):
|
||||
tree = html.fromstring(content)
|
||||
found = {}
|
||||
for anchor in tree.xpath('//a[contains(@href,"/orders/")]'):
|
||||
href = anchor.get("href")
|
||||
match = re.fullmatch(r"/orders/(\d+)/details", href)
|
||||
text = " ".join(" ".join(anchor.itertext()).split())
|
||||
if not match or not re.search(r"\bDelivered\b", text):
|
||||
continue
|
||||
stamp = re.search(r"([A-Z][a-z]{2} \d{1,2}, \d{4})", text)
|
||||
if not stamp:
|
||||
raise ImportError(f"No full delivery date on order {match[1]}")
|
||||
purchased = datetime.strptime(stamp[1], "%b %d, %Y").date().isoformat()
|
||||
found[match[1]] = OrderLink(match[1], purchased, "https://www.ocado.com" + href)
|
||||
return list(found.values())
|
||||
|
||||
|
||||
def wait_for_orders(page, timeout, echo):
|
||||
deadline = time.monotonic() + timeout
|
||||
notice = 0
|
||||
while time.monotonic() < deadline:
|
||||
if page.is_closed():
|
||||
raise ImportError("Browser closed before sign-in completed")
|
||||
if urlsplit(page.url).hostname == "www.ocado.com" and page.locator('a[href*="/orders/"]').count():
|
||||
return
|
||||
if time.monotonic() >= notice:
|
||||
echo("Waiting for order history. Complete sign-in or any CAPTCHA in the browser.")
|
||||
notice = time.monotonic() + 30
|
||||
page.wait_for_timeout(1000)
|
||||
raise ImportError("Timed out waiting for order history; rerun to reuse the saved browser session")
|
||||
|
||||
|
||||
def discover_orders(page, echo):
|
||||
last_count = -1
|
||||
stable = 0
|
||||
for _ in range(500):
|
||||
count = page.locator('a[href*="/orders/"]').count()
|
||||
if count == last_count:
|
||||
stable += 1
|
||||
else:
|
||||
stable = 0
|
||||
echo(f"Loaded {count} order links...")
|
||||
if page.locator('[data-test="order-list-no-more-orders-label"]').count():
|
||||
return order_links(page.content())
|
||||
if stable >= 10:
|
||||
raise ImportError("Order history stopped loading before Ocado's end-of-list marker; retry after checking the browser")
|
||||
last_count = count
|
||||
bottom = page.get_by_role("button", name="Back to top", exact=True)
|
||||
if bottom.count():
|
||||
bottom.scroll_into_view_if_needed()
|
||||
else:
|
||||
page.locator('a[href*="/orders/"]').last.scroll_into_view_if_needed()
|
||||
page.mouse.wheel(0,1200)
|
||||
page.wait_for_timeout(1500)
|
||||
raise ImportError("Order list did not reach its end; refusing to call the download complete")
|
||||
|
||||
|
||||
def download_orders(profile, archive, *, since=None, limit=None, login_only=False,
|
||||
login_timeout=300, echo=print, order_ids=(), refresh=False, completed=()):
|
||||
try:
|
||||
from playwright.sync_api import sync_playwright, Error as BrowserError
|
||||
except ModuleNotFoundError as exc:
|
||||
raise ImportError("Install dependencies: pip install -e . and python -m playwright install chromium") from exc
|
||||
profile.mkdir(parents=True, exist_ok=True, mode=0o700)
|
||||
profile.chmod(0o700)
|
||||
with sync_playwright() as playwright:
|
||||
context = playwright.chromium.launch_persistent_context(
|
||||
str(profile.resolve()), headless=False, viewport={"width":1280, "height":900})
|
||||
try:
|
||||
page = context.pages[0] if context.pages else context.new_page()
|
||||
page.goto("https://www.ocado.com/orders", wait_until="domcontentloaded")
|
||||
wait_for_orders(page, login_timeout, echo)
|
||||
if login_only:
|
||||
echo("Signed-in browser session saved.")
|
||||
return [], []
|
||||
links = discover_orders(page, echo)
|
||||
private_json(archive / "orders.json", [asdict(link) for link in links])
|
||||
links = select_orders(links, since, limit, order_ids, completed)
|
||||
receipts, failures = [], []
|
||||
for index, link in enumerate(links,1):
|
||||
path = archive / f"{link.order_id}.json"
|
||||
if path.exists() and not refresh:
|
||||
receipt = parse_document(json.loads(path.read_text()))
|
||||
if receipt.order_id != link.order_id or receipt.purchased_date != link.purchased_date:
|
||||
raise ImportError(f"Cached receipt does not match order {link.order_id}")
|
||||
receipts.append(receipt)
|
||||
echo(f"[{index}/{len(links)}] Cached {link.order_id} ({link.purchased_date})")
|
||||
continue
|
||||
try:
|
||||
endpoint = f"/api/order/v6/orders/{link.order_id}/decorated"
|
||||
with page.expect_response(lambda response: urlsplit(response.url).path == endpoint,
|
||||
timeout=45000) as captured:
|
||||
page.goto(link.url, wait_until="domcontentloaded")
|
||||
response = captured.value
|
||||
if response.status != 200:
|
||||
raise ImportError(f"Ocado receipt returned HTTP {response.status}")
|
||||
data = response.json()
|
||||
page.wait_for_selector('[data-test="order-online-receipt"]', timeout=30000)
|
||||
hrefs = page.locator('[data-test="receipt-element-product-link"][href]').evaluate_all(
|
||||
'(elements) => elements.map(element => element.getAttribute("href"))')
|
||||
urls = {}
|
||||
for href in hrefs:
|
||||
match = re.search(r"/(\d+)$", href)
|
||||
if match:
|
||||
urls[match[1]] = "https://www.ocado.com" + href
|
||||
receipt = parse_ocado_order(data, expected_order_id=link.order_id, product_urls=urls)
|
||||
if receipt.purchased_date != link.purchased_date:
|
||||
raise ImportError("Order-list date does not match the receipt API date")
|
||||
private_json(path, receipt.export())
|
||||
receipts.append(receipt)
|
||||
echo(f"[{index}/{len(links)}] Saved {link.order_id}: {len(receipt.items)} lines")
|
||||
except (ImportError, BrowserError, ValueError, KeyError) as exc:
|
||||
message = str(exc) if isinstance(exc, ImportError) else type(exc).__name__
|
||||
failures.append({"order_id":link.order_id,"error":message})
|
||||
echo(f"[{index}/{len(links)}] Could not read {link.order_id}: {message}")
|
||||
page.wait_for_timeout(750)
|
||||
private_json(archive / "download-errors.json", failures)
|
||||
return receipts, failures
|
||||
finally:
|
||||
try:
|
||||
# Persistent profile is authoritative; this also preserves a portable cookie backup.
|
||||
context.storage_state(path=str(profile / "auth.json"))
|
||||
(profile / "auth.json").chmod(0o600)
|
||||
finally:
|
||||
context.close()
|
||||
|
||||
|
||||
def select_orders(links, since=None, limit=None, order_ids=(), completed=()):
|
||||
missing = set(order_ids) - {link.order_id for link in links}
|
||||
if missing:
|
||||
raise ImportError("Orders not found in delivered history: " + ", ".join(sorted(missing)))
|
||||
links = sorted((link for link in links if not since or link.purchased_date >= since),
|
||||
key=lambda link:(link.purchased_date,link.order_id), reverse=True)
|
||||
if order_ids:
|
||||
links = [link for link in links if link.order_id in order_ids]
|
||||
if not order_ids:
|
||||
links = [link for link in links if link.order_id not in completed]
|
||||
if limit:
|
||||
links = links[:limit]
|
||||
return links
|
||||
|
||||
|
||||
def cached_orders(archive, since=None, limit=None, order_ids=(), completed=()):
|
||||
links = [OrderLink(**row) for row in json.loads((archive / "orders.json").read_text())]
|
||||
links = select_orders(links, since, limit, order_ids, completed)
|
||||
receipts, failures = [], []
|
||||
for link in links:
|
||||
path = archive / f"{link.order_id}.json"
|
||||
if path.exists():
|
||||
receipt = parse_document(json.loads(path.read_text()))
|
||||
if receipt.order_id != link.order_id or receipt.purchased_date != link.purchased_date:
|
||||
raise ImportError(f"Cached receipt does not match order {link.order_id}")
|
||||
receipts.append(receipt)
|
||||
else:
|
||||
failures.append({"order_id":link.order_id,"error":"Receipt has not been downloaded"})
|
||||
return receipts, failures
|
||||
|
||||
|
||||
def run_import(receipts, config, config_path, archive, dry_run, failures, echo=print, refresh_existing=True,
|
||||
all_products=True):
|
||||
report = {"orders":len(receipts), "download_errors":failures, "items":[], "imported_lines":0, "refreshed_products":0}
|
||||
settings = config.get("import", {})
|
||||
journal = None
|
||||
refreshed = set()
|
||||
try:
|
||||
if not dry_run:
|
||||
connection = config.get("grocy", {})
|
||||
url = os.environ.get("GROCY_URL", connection.get("url", ""))
|
||||
key = os.environ.get("GROCY_API_KEY", connection.get("api_key", ""))
|
||||
if not url or not key:
|
||||
raise ImportError("Configure Grocy URL and API key before importing")
|
||||
api = Grocy(url,key)
|
||||
state = Path(settings.get("state_file", "imports.sqlite3"))
|
||||
if not state.is_absolute():
|
||||
state = config_path.resolve().parent / state
|
||||
journal = Journal(state)
|
||||
for receipt in receipts:
|
||||
if not dry_run:
|
||||
receipt = align_recorded_lines(receipt, api.url, journal)
|
||||
if not dry_run and refresh_existing:
|
||||
count = refresh_imported_products(receipt, api, journal, seen=refreshed)
|
||||
report["refreshed_products"] += count
|
||||
if count:
|
||||
echo(f"Refreshed details for {count} previously imported products from {receipt.order_id}")
|
||||
selected = set()
|
||||
for line,item in enumerate(receipt.items,1):
|
||||
decision = classify(item,config.get("pantry",{}).get("overrides",{}))
|
||||
report["items"].append({"order_id":receipt.order_id, "date":receipt.purchased_date,
|
||||
"line":line,"name":item.name,"product_id":item.product_id,
|
||||
"quantity":str(item.quantity),"total":str(item.total),
|
||||
**asdict(decision)})
|
||||
if decision.include or all_products:
|
||||
selected.add(line)
|
||||
report["items"][-1]["include"] = True
|
||||
if all_products:
|
||||
report["items"][-1]["reason"] = "all products requested"
|
||||
report["items"][-1]["uncertain"] = False
|
||||
echo(f"Order {receipt.order_id} ({receipt.purchased_date}): {len(selected)}/{len(receipt.items)} selected lines")
|
||||
if not dry_run:
|
||||
report["imported_lines"] += import_receipt(
|
||||
receipt,api,journal,{**settings,"products":config.get("products",{})},
|
||||
echo=echo, include_lines=selected)
|
||||
finally:
|
||||
private_json(archive / "report.json",report)
|
||||
if journal:
|
||||
journal.db.close()
|
||||
return report
|
||||
|
||||
|
||||
@click.command(context_settings={"help_option_names":["-h","--help"]})
|
||||
@click.option("--config", "config_path", type=click.Path(path_type=Path,dir_okay=False),default="config.toml",show_default=True)
|
||||
@click.option("--profile", type=click.Path(path_type=Path,file_okay=False),default=".ocado-browser",show_default=True)
|
||||
@click.option("--archive", type=click.Path(path_type=Path,file_okay=False),default="history",show_default=True)
|
||||
@click.option("--since", help="Only orders delivered on or after YYYY-MM-DD.")
|
||||
@click.option("--limit", type=click.IntRange(min=1),help="Process at most this many new delivered orders (or explicitly selected orders).")
|
||||
@click.option("--dry-run", is_flag=True,help="Download and report selections without writing to Grocy.")
|
||||
@click.option("--cached", is_flag=True,help="Use downloaded receipts without opening a browser.")
|
||||
@click.option("--login-only", is_flag=True,help="Sign in and save cookies without downloading or importing orders.")
|
||||
@click.option("--login-timeout",type=click.IntRange(min=1),default=300,show_default=True)
|
||||
@click.option("--order-id", "order_ids", multiple=True, help="Process a specific delivered order; repeat for several.")
|
||||
@click.option("--all-products/--shelf-stable-only", default=True, show_default=True,
|
||||
help="Include all products, or restrict imports to shelf-stable items.")
|
||||
@click.option("--refresh", is_flag=True, help="Download receipts again even when they are cached.")
|
||||
@click.option("--refresh-existing/--no-refresh-existing", default=True, show_default=True,
|
||||
help="Update details of previously imported products, including perishables.")
|
||||
@click.option("--images/--no-images", default=True, show_default=True, help="Add missing Ocado product pictures after importing.")
|
||||
@click.option("--openfoodfacts/--no-openfoodfacts", default=True, show_default=True, help="Match and enrich imported products after importing.")
|
||||
def main(config_path,profile,archive,since,limit,dry_run,cached,login_only,login_timeout,refresh_existing,
|
||||
order_ids,all_products,refresh,images,openfoodfacts):
|
||||
"""Import delivered Ocado orders with Playwright; include all products by default."""
|
||||
try:
|
||||
if cached and login_only:
|
||||
raise ImportError("--cached and --login-only cannot be combined")
|
||||
if cached and refresh:
|
||||
raise ImportError("--cached and --refresh cannot be combined")
|
||||
config = tomllib.loads(config_path.read_text()) if config_path.exists() else {}
|
||||
from .order_state import completed_orders, mark_orders
|
||||
connection = config.get('grocy', {})
|
||||
server = Grocy(os.environ.get('GROCY_URL', connection.get('url', '')),
|
||||
os.environ.get('GROCY_API_KEY', connection.get('api_key', ''))).url
|
||||
state = Path(config.get('import', {}).get('state_file', 'imports.sqlite3'))
|
||||
if not state.is_absolute():
|
||||
state = config_path.resolve().parent / state
|
||||
completed = completed_orders(state, server, archive, config)
|
||||
if not dry_run and not login_only:
|
||||
mark_orders(state, server, completed, 'complete')
|
||||
retry_errors = []
|
||||
if not dry_run and not login_only and (images or openfoodfacts):
|
||||
from .postprocess import retry_enrichment
|
||||
retry_errors = retry_enrichment(config, config_path, archive, images=images, openfoodfacts=openfoodfacts)['errors']
|
||||
since = receipt_date(since)
|
||||
archive.mkdir(parents=True,exist_ok=True,mode=0o700)
|
||||
if cached:
|
||||
receipts,failures = cached_orders(archive,since,limit,order_ids,completed)
|
||||
else:
|
||||
receipts,failures = download_orders(profile,archive,since=since,limit=limit,
|
||||
login_only=login_only,login_timeout=login_timeout,echo=click.echo,order_ids=order_ids,refresh=refresh,completed=completed)
|
||||
if login_only:
|
||||
return
|
||||
if not receipts and not failures:
|
||||
if retry_errors:
|
||||
raise ImportError('Enrichment retries still need attention: ' + '; '.join(retry_errors))
|
||||
click.echo('No new delivered orders to import.')
|
||||
return
|
||||
processing = [receipt.order_id for receipt in receipts]
|
||||
if not dry_run:
|
||||
mark_orders(state, server, processing, 'processing')
|
||||
report = run_import(receipts,config,config_path,archive,dry_run,failures,echo=click.echo,
|
||||
refresh_existing=refresh_existing,all_products=all_products)
|
||||
selected = sum(row["include"] for row in report["items"])
|
||||
uncertain = sum(row["uncertain"] for row in report["items"])
|
||||
click.echo(f"{len(receipts)} orders; {selected} selected lines; {report['imported_lines']} imported; "
|
||||
f"{uncertain} uncertain lines. Report: {archive / 'report.json'}")
|
||||
if not dry_run:
|
||||
from .postprocess import enrich_orders
|
||||
enrichment = enrich_orders(config, config_path, archive,
|
||||
[receipt.order_id for receipt in receipts], images=images, openfoodfacts=openfoodfacts)
|
||||
report['enrichment'] = enrichment
|
||||
private_json(archive / 'report.json', report)
|
||||
mark_orders(state, server, processing, 'complete')
|
||||
if enrichment['errors'] or retry_errors:
|
||||
raise ImportError('Stock import completed; enrichment needs a retry. ' + '; '.join(enrichment['errors'] + retry_errors))
|
||||
if failures:
|
||||
raise ImportError(f"{len(failures)} orders could not be downloaded; see report and retry")
|
||||
except (ImportError,OSError,ValueError,KeyError,sqlite3.Error) as exc:
|
||||
raise click.ClickException(str(exc)) from exc
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
113
ocado_grocy/images.py
Normal file
113
ocado_grocy/images.py
Normal file
|
|
@ -0,0 +1,113 @@
|
|||
"""Copy saved Ocado product photos into Grocy's native product pictures."""
|
||||
import base64
|
||||
import hashlib
|
||||
import json
|
||||
from pathlib import Path
|
||||
import tomllib
|
||||
from urllib.parse import quote, urlsplit
|
||||
|
||||
import click
|
||||
import requests
|
||||
|
||||
from .grocy import Grocy, imported_products
|
||||
from .history import private_json
|
||||
from .product_metadata import read_metadata
|
||||
|
||||
|
||||
def image_type(data):
|
||||
if data.startswith(b'\xff\xd8\xff'):
|
||||
return 'jpg'
|
||||
if data.startswith(b'\x89PNG\r\n\x1a\n'):
|
||||
return 'png'
|
||||
if data.startswith(b'RIFF') and data[8:12] == b'WEBP':
|
||||
return 'webp'
|
||||
raise ValueError('Image response is not JPEG, PNG or WebP')
|
||||
|
||||
|
||||
def file_path(filename):
|
||||
return '/files/productpictures/' + quote(base64.b64encode(filename.encode()).decode(), safe='')
|
||||
|
||||
|
||||
def add_picture(client, downloader, product):
|
||||
if product.get('picture_file_name'):
|
||||
return 'existing_picture'
|
||||
metadata = read_metadata(product.get('description'))
|
||||
url = metadata.get('image_url')
|
||||
if not url:
|
||||
return 'no_image_url'
|
||||
parsed = urlsplit(url)
|
||||
if parsed.scheme != 'https' or parsed.hostname not in {'www.ocado.com', 'ocado.com'}:
|
||||
raise ValueError('Expected an HTTPS Ocado image URL')
|
||||
# Separate session: never send the Grocy API key to the image host.
|
||||
response = downloader.get(url, timeout=30, allow_redirects=False)
|
||||
response.raise_for_status()
|
||||
if response.status_code != 200:
|
||||
raise ValueError('Image URL did not return HTTP 200')
|
||||
data = response.content
|
||||
extension = image_type(data)
|
||||
if len(data) > 10_000_000:
|
||||
raise ValueError('Image exceeds 10 MB')
|
||||
filename = f"ocado-{int(product['id'])}-{hashlib.sha256(data).hexdigest()[:16]}.{extension}"
|
||||
current = client.get(f"/objects/products/{product['id']}")
|
||||
if not current:
|
||||
return 'deleted_product'
|
||||
if current.get('picture_file_name'):
|
||||
return 'existing_picture'
|
||||
if read_metadata(current.get('description')).get('Ocado product ID') != metadata.get('Ocado product ID'):
|
||||
raise ValueError('Product identity changed')
|
||||
path = file_path(filename)
|
||||
uploaded = client.session.put(client.url + path, data=data,
|
||||
headers={'Content-Type':'application/octet-stream'},
|
||||
timeout=client.timeout, allow_redirects=False)
|
||||
uploaded.raise_for_status()
|
||||
if uploaded.status_code not in (200, 201, 204):
|
||||
raise ValueError('Unexpected upload response')
|
||||
# Check the stored file before setting the product's picture reference.
|
||||
stored = client.session.get(client.url + path, timeout=client.timeout, allow_redirects=False)
|
||||
stored.raise_for_status()
|
||||
if stored.status_code != 200 or stored.content != data:
|
||||
raise ValueError('Uploaded image verification failed')
|
||||
client.request('PUT', f"/objects/products/{product['id']}", {'picture_file_name':filename})
|
||||
return 'updated'
|
||||
|
||||
|
||||
@click.command()
|
||||
@click.option('--config', type=click.Path(path_type=Path), default=Path('config.toml'))
|
||||
@click.option('--report', type=click.Path(path_type=Path), default=Path('history/images/report.json'))
|
||||
@click.option('--apply', is_flag=True, help='Upload and assign missing product pictures.')
|
||||
@click.option('--order-id', 'order_ids', multiple=True, help='Limit to products imported from these orders.')
|
||||
def main(config, report, apply, order_ids):
|
||||
"""Add Ocado images to journal-linked products; keep existing pictures."""
|
||||
settings = tomllib.loads(config.read_text())
|
||||
client = Grocy(**settings['grocy'])
|
||||
state = config.parent / settings.get('import', {}).get('state_file', 'imports.sqlite3')
|
||||
products = imported_products(client, state, order_ids)
|
||||
return run_images(client, products, report, apply=apply)
|
||||
|
||||
|
||||
def run_images(client, products, report, *, apply=True, raise_errors=True):
|
||||
downloader = requests.Session()
|
||||
downloader.headers['User-Agent'] = 'ocado-grocy/0.1'
|
||||
results = []
|
||||
private_json(report, {'apply':apply, 'products':results})
|
||||
for product in products:
|
||||
row = {'product_id':product['id'], 'name':product['name']}
|
||||
try:
|
||||
if apply:
|
||||
row['status'] = add_picture(client, downloader, product)
|
||||
else:
|
||||
row['status'] = 'existing_picture' if product.get('picture_file_name') else 'ready' if read_metadata(product.get('description')).get('image_url') else 'no_image_url'
|
||||
except (requests.RequestException, ValueError) as exc:
|
||||
row.update(status='error', error=str(exc))
|
||||
results.append(row)
|
||||
private_json(report, {'apply':apply, 'products':results})
|
||||
click.echo(f"{product['id']} {row['status']}: {product['name']}")
|
||||
counts = {s:sum(r['status']==s for r in results) for s in sorted({r['status'] for r in results})}
|
||||
click.echo(json.dumps(counts))
|
||||
if counts.get('error') and raise_errors:
|
||||
raise click.ClickException(f"{counts['error']} image updates failed; see {report}")
|
||||
return {'products':results, 'counts':counts}
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
283
ocado_grocy/openfoodfacts.py
Normal file
283
ocado_grocy/openfoodfacts.py
Normal file
|
|
@ -0,0 +1,283 @@
|
|||
"""Conservative, resumable Open Food Facts enrichment of imported products."""
|
||||
import hashlib
|
||||
import json
|
||||
from pathlib import Path
|
||||
import re
|
||||
import time
|
||||
import tomllib
|
||||
import unicodedata
|
||||
from datetime import datetime, timezone
|
||||
|
||||
import click
|
||||
import requests
|
||||
|
||||
from .grocy import Grocy, imported_products
|
||||
from .history import private_json
|
||||
from .product_metadata import read_metadata, replace_metadata
|
||||
|
||||
FIELDS = 'code,product_name,product_name_en,brands,quantity,ingredients_text_en,ingredients_text,allergens_tags,traces_tags,nutriments,nutrition_grades,nova_group,labels_tags,packaging,serving_size,image_front_url,countries_tags,last_modified_t'
|
||||
NONFOOD = {'Household & Cleaning', 'Beauty & Toiletries', 'Home & Garden', 'Health, Medicines & Wellbeing', 'Pets'}
|
||||
NONFOOD_BRANDS = {'Miniml', 'Heart & Soul', 'INTERNATIONAL GREETINGS', 'Caroline Gardner', 'Colgate'}
|
||||
ALIASES = {'m s': 'marks spencer', 'm s food':'marks spencer', 'marks and spencer': 'marks spencer', 'marks spencers':'marks spencer', 'kelloggs special k': 'kelloggs', 'kellogg s': 'kelloggs', 'jacob s': 'jacobs', 'mcvitie s': 'mcvities', 'nairn s': 'nairns', 'garner s': 'garners', 'wrigley s extra': 'extra', 'nestle shredded wheat': 'shredded wheat', 'arnotts': 'arnotts', 'tncc': 'natural confectionery co'}
|
||||
IGNORE = {'the', 'and', 'with', 'of', 'in', 'a', 'an', 'bag', 'sharing', 'multipack', 'pack', 'baked', 'snacks', 'breakfast', 'cereal', 'biscuits', 'biscuit', 'tinned', 'tin', 'can', 'single', 'jar', 'bottle', 'small', 'classic', 'favourites', 'ready', 'meal', 'cook'}
|
||||
|
||||
|
||||
def words(text):
|
||||
text = unicodedata.normalize('NFKD', str(text)).encode('ascii', 'ignore').decode().lower()
|
||||
return ' '.join(re.findall(r'[a-z0-9]+', text.replace("'", '').replace('’', '')))
|
||||
|
||||
|
||||
def brand_key(text):
|
||||
key = words(text)
|
||||
return ALIASES.get(key, key)
|
||||
|
||||
|
||||
def pack(text):
|
||||
"""Only unambiguous metric quantities; retain multipack structure."""
|
||||
text = str(text).lower().replace('×', 'x').replace(',', '.')
|
||||
text = text.split('(')[0].strip()
|
||||
m = re.fullmatch(r'(?:(\d+)\s*x\s*)?(\d+(?:\.\d+)?)\s*(kg|g|ml|cl|l)', text)
|
||||
if not m:
|
||||
return None
|
||||
n, amount, unit = m.groups()
|
||||
return (int(n or 1), round(float(amount) * {'kg':1000, 'g':1, 'ml':1, 'cl':10, 'l':1000}[unit], 3), 'g' if unit in ('g','kg') else 'ml')
|
||||
|
||||
|
||||
def tokens(name, brand):
|
||||
name = words(re.sub(r'\b\d+(?:\.\d+)?\s*(?:kg|g|ml|cl|l)\b', '', name, flags=re.I))
|
||||
remove = set(words(brand).split()) | set(brand_key(brand).split()) | IGNORE
|
||||
return set(name.split()) - remove
|
||||
|
||||
|
||||
def candidate_info(product, metadata, candidate):
|
||||
brand = metadata.get('ocado', {}).get('brand', '')
|
||||
expected = metadata.get('ocado', {}).get('size', {}).get('value', metadata.get('size', ''))
|
||||
title = candidate.get('product_name') or candidate.get('product_name_en', '')
|
||||
a, b = tokens(product['name'], brand), tokens(title, brand)
|
||||
brands = [brand_key(x) for x in candidate.get('brands', '').split(',')]
|
||||
brand_ok = bool(brand) and brand_key(brand) in brands
|
||||
size_ok = pack(expected) is not None and pack(expected) == pack(candidate.get('quantity', ''))
|
||||
score = len(a & b) / max(1, len(a | b))
|
||||
exact = brand_ok and size_ok and bool(a) and a == b and valid_barcode(candidate.get('code', ''))
|
||||
return {'code': candidate.get('code'), 'name':title, 'brand':candidate.get('brands'), 'quantity':candidate.get('quantity'), 'brand_match':brand_ok, 'size_match':size_ok, 'name_score':round(score, 3), 'exact':exact}
|
||||
|
||||
|
||||
def choose_match(product, metadata, candidates):
|
||||
ranked = sorted([candidate_info(product, metadata, c) for c in candidates.values()], key=lambda c:(c['exact'], c['brand_match'], c['name_score'], c['size_match']), reverse=True)
|
||||
matches = [c for c in ranked if c['exact']]
|
||||
return ranked, candidates[matches[0]['code']] if len(matches) == 1 else None
|
||||
|
||||
|
||||
def valid_barcode(code):
|
||||
if not isinstance(code, str) or not code.isascii() or not code.isdigit() or len(code) not in (8,12,13,14):
|
||||
return False
|
||||
return (sum(int(x) * (3 if i % 2 == 0 else 1) for i, x in enumerate(reversed(code[:-1]))) + int(code[-1])) % 10 == 0
|
||||
|
||||
|
||||
class OpenFoodFacts:
|
||||
def __init__(self, cache, contact, refresh=False):
|
||||
self.refresh = refresh
|
||||
self.cache = cache
|
||||
self.session = requests.Session()
|
||||
self.session.headers['User-Agent'] = f'ocado-grocy/0.1 ({contact})'
|
||||
self.last = 0
|
||||
|
||||
def fetch(self, path, params, *, offline=False):
|
||||
key = hashlib.sha256(json.dumps([path, params], sort_keys=True).encode()).hexdigest()
|
||||
target = self.cache / (key + '.json')
|
||||
if target.exists() and (offline or not self.refresh):
|
||||
return json.loads(target.read_text())
|
||||
if offline:
|
||||
raise ValueError('Search not cached; rerun without --offline')
|
||||
# Shared spacing for search and product reads: <= 10 requests per minute.
|
||||
time.sleep(max(0, 6.2 - (time.monotonic() - self.last)))
|
||||
self.last = time.monotonic()
|
||||
url = path if path.startswith('https://search.openfoodfacts.org/') else 'https://world.openfoodfacts.org' + path
|
||||
response = self.session.get(url, params=params, timeout=45)
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
private_json(target, data)
|
||||
return data
|
||||
|
||||
def catalog(self, brands, offline=False):
|
||||
tags = sorted({brand_key(b).replace(' ', '-') for b in brands if b})
|
||||
if not tags:
|
||||
return []
|
||||
|
||||
def group(tags):
|
||||
query = 'brands_tags:(' + ' OR '.join(tags) + ')'
|
||||
result = []
|
||||
page = 1
|
||||
while True:
|
||||
data = self.fetch('https://search.openfoodfacts.org/search', {'q':query, 'page_size':1000, 'page':page, 'fields':FIELDS}, offline=offline)
|
||||
if data.get('timed_out') or data.get('warnings'):
|
||||
raise ValueError('Incomplete catalog search; retry later')
|
||||
if not data.get('is_count_exact', False):
|
||||
if len(tags) == 1:
|
||||
raise ValueError('Brand catalog exceeds search limit: ' + tags[0])
|
||||
middle = len(tags) // 2
|
||||
return group(tags[:middle]) + group(tags[middle:])
|
||||
for hit in data.get('hits', []):
|
||||
hit = dict(hit)
|
||||
if isinstance(hit.get('brands'), list):
|
||||
hit['brands'] = ','.join(hit['brands'])
|
||||
result.append(hit)
|
||||
click.echo(f"Catalog ({len(tags)} brands) page {page}: {len(result)} of {data.get('count')} products")
|
||||
if page >= data.get('page_count', 1):
|
||||
if len(result) != data.get('count'):
|
||||
raise ValueError('Incomplete catalog response')
|
||||
return result
|
||||
page += 1
|
||||
|
||||
# OFF also retains apostrophes as tag separators (e.g. nairn-s).
|
||||
alternatives = sorted({re.sub(r'[^a-z0-9]+', '-', unicodedata.normalize('NFKD', b).encode('ascii', 'ignore').decode().lower()).strip('-') for b in brands if b} - set(tags))
|
||||
result = group(tags)
|
||||
if alternatives:
|
||||
result += group(alternatives)
|
||||
return list({str(p['code']):p for p in result}.values())
|
||||
|
||||
def product(self, code, offline=False):
|
||||
data = self.fetch(f'/api/v3.6/product/{code}.json', {'fields':FIELDS + ',nutrition'}, offline=offline)
|
||||
product = data.get('product')
|
||||
if not product or str(product.get('code')) != code:
|
||||
raise ValueError('Open Food Facts did not return requested barcode')
|
||||
return product
|
||||
|
||||
|
||||
def apply_match(client, product, candidate, barcodes):
|
||||
"""Write only the barcode relation and description; never stock or name."""
|
||||
code = str(candidate['code'])
|
||||
if not valid_barcode(code):
|
||||
raise ValueError('Invalid GTIN check digit')
|
||||
owners = {int(b['product_id']) for b in barcodes if b['barcode'] == code}
|
||||
if owners - {int(product['id'])}:
|
||||
raise ValueError('Barcode already belongs to another Grocy product')
|
||||
# Re-read just before mutation to retain user edits and skip deleted products.
|
||||
current = client.get(f"/objects/products/{product['id']}")
|
||||
if not current:
|
||||
raise ValueError('Grocy product was deleted')
|
||||
metadata = read_metadata(current.get('description'))
|
||||
if not metadata or metadata.get('Ocado product ID') != read_metadata(product.get('description')).get('Ocado product ID'):
|
||||
raise ValueError('Imported product identity changed')
|
||||
if metadata.get('Barcode') and metadata['Barcode'] != code:
|
||||
raise ValueError('Existing metadata barcode differs; review required')
|
||||
enrichment = {k:v for k,v in candidate.items() if v not in (None, '', [], {})}
|
||||
enrichment.update(source='Open Food Facts', url=f'https://world.openfoodfacts.org/product/{code}', database_license='ODbL', contents_license='Database Contents License', images_license='CC BY-SA')
|
||||
metadata['Barcode'] = code
|
||||
metadata['openfoodfacts'] = enrichment
|
||||
updated = replace_metadata(current['description'], metadata)
|
||||
if not owners:
|
||||
row = {'product_id':int(product['id']), 'barcode':code, 'qu_id':current['qu_id_purchase'], 'amount':1}
|
||||
client.post('/objects/product_barcodes', row)
|
||||
barcodes.append(row)
|
||||
if read_metadata(current['description']) != metadata:
|
||||
client.request('PUT', f"/objects/products/{product['id']}", {'description':updated})
|
||||
return 'updated'
|
||||
return 'updated' if not owners else 'unchanged'
|
||||
|
||||
|
||||
@click.command()
|
||||
@click.option('--config', type=click.Path(path_type=Path), default=Path('config.toml'))
|
||||
@click.option('--cache', type=click.Path(path_type=Path), default=Path('history/openfoodfacts'))
|
||||
@click.option('--contact', help='Contact for the Open Food Facts User-Agent; defaults to config [openfoodfacts].contact.')
|
||||
@click.option('--refresh-cache', is_flag=True, help='Fetch fresh OFF responses, replacing cached responses.')
|
||||
@click.option('--apply', is_flag=True, help='Apply unambiguous matches and explicitly reviewed mappings.')
|
||||
@click.option('--offline', is_flag=True, help='Use cached Open Food Facts responses only.')
|
||||
@click.option('--mapping', type=click.Path(exists=True, path_type=Path), help='Reviewed JSON mapping of Grocy product ID to barcode.')
|
||||
@click.option('--limit', type=int, help='Limit product comparisons after downloading the catalog.')
|
||||
@click.option('--order-id', 'order_ids', multiple=True, help='Limit to products imported from these orders.')
|
||||
def main(config, cache, contact, refresh_cache, apply, offline, mapping, limit, order_ids):
|
||||
"""Match Ocado imports to Open Food Facts. Defaults to a read-only preview."""
|
||||
settings = tomllib.loads(config.read_text())
|
||||
contact = contact or settings.get('openfoodfacts', {}).get('contact')
|
||||
if not contact:
|
||||
raise click.UsageError('Set --contact or [openfoodfacts].contact in config.toml')
|
||||
if refresh_cache and offline:
|
||||
raise click.UsageError('--refresh-cache cannot be combined with --offline')
|
||||
client = Grocy(**settings['grocy'])
|
||||
state = config.parent / settings.get('import', {}).get('state_file', 'imports.sqlite3')
|
||||
products = imported_products(client, state, order_ids)
|
||||
mappings = json.loads(mapping.read_text()) if mapping else {}
|
||||
return run_openfoodfacts(client, products, cache, contact, apply=apply, offline=offline,
|
||||
refresh_cache=refresh_cache, mappings=mappings, limit=limit)
|
||||
|
||||
|
||||
def run_openfoodfacts(client, products, cache, contact, *, apply=True, offline=False,
|
||||
refresh_cache=False, mappings=None, limit=None):
|
||||
barcodes = client.get('/objects/product_barcodes')
|
||||
off = OpenFoodFacts(cache / 'responses', contact, refresh_cache)
|
||||
mappings = mappings or {}
|
||||
if not isinstance(mappings, dict) or any(not isinstance(v, str) or not valid_barcode(v) for v in mappings.values()):
|
||||
raise click.UsageError('Mapping must be a JSON object with barcode strings and valid GTIN check digits')
|
||||
def needs_search(product):
|
||||
metadata = read_metadata(product.get('description'))
|
||||
code = metadata.get('openfoodfacts', {}).get('code')
|
||||
return not mappings.get(str(product['id'])) and not (valid_barcode(code) and code == metadata.get('Barcode'))
|
||||
brands = {read_metadata(p.get('description')).get('ocado', {}).get('brand') for p in products if needs_search(p)}
|
||||
catalog = off.catalog(brands, offline)
|
||||
private_json(cache / 'catalog.json', catalog)
|
||||
by_brand = {}
|
||||
for candidate in catalog:
|
||||
for brand in candidate.get('brands', '').split(','):
|
||||
by_brand.setdefault(brand_key(brand), {})[str(candidate['code'])] = candidate
|
||||
report = {'created_at':datetime.now(timezone.utc).isoformat(), 'apply':apply, 'products':[]}
|
||||
if apply:
|
||||
private_json(cache / ('before-' + datetime.now().strftime('%Y%m%d-%H%M%S') + '.json'), {'products':products, 'barcodes':barcodes})
|
||||
searched = 0
|
||||
for product in products:
|
||||
row = {'product_id':product['id'], 'name':product['name']}
|
||||
metadata = read_metadata(product.get('description'))
|
||||
ocado = metadata.get('ocado', {})
|
||||
row['size'] = ocado.get('size', {}).get('value', metadata.get('size',''))
|
||||
try:
|
||||
if not metadata:
|
||||
row['status'] = 'missing_metadata'
|
||||
elif set(ocado.get('categoryPath', [])) & NONFOOD or ocado.get('brand') in NONFOOD_BRANDS:
|
||||
row['status'] = 'non_food'
|
||||
elif limit is not None and searched >= limit:
|
||||
row['status'] = 'not_searched'
|
||||
else:
|
||||
searched += 1
|
||||
code = mappings.get(str(product['id']))
|
||||
method = 'reviewed_mapping'
|
||||
established = metadata.get('openfoodfacts', {}).get('code')
|
||||
if not code and established == metadata.get('Barcode') and valid_barcode(established):
|
||||
code = established
|
||||
method = 'existing_match'
|
||||
if code:
|
||||
candidate = next((c for c in catalog if str(c.get('code')) == str(code)), None) or off.product(str(code), offline)
|
||||
row['match_method'] = method
|
||||
row['candidates'] = [candidate_info(product, metadata, candidate)]
|
||||
else:
|
||||
query = 'brand catalog: ' + ocado.get('brand', '')
|
||||
result = {'count':len(catalog)}
|
||||
candidates = by_brand.get(brand_key(ocado.get('brand', '')), {})
|
||||
result['count'] = len(candidates)
|
||||
ranked, candidate = choose_match(product, metadata, candidates)
|
||||
row.update(query=query, search_count=result.get('count'), candidates=ranked[:10])
|
||||
row['match_method'] = 'exact_brand_name_pack'
|
||||
if candidate:
|
||||
# Catalog is indexed separately; fetch current detail before writes.
|
||||
if apply:
|
||||
fresh = off.product(str(candidate['code']), offline)
|
||||
row['current_candidate'] = candidate_info(product, metadata, fresh)
|
||||
if not code and not row['current_candidate']['exact']:
|
||||
raise ValueError('Current product details no longer match catalog')
|
||||
candidate = fresh
|
||||
row['barcode'] = candidate['code']
|
||||
row['status'] = apply_match(client, product, candidate, barcodes) if apply else 'matched'
|
||||
else:
|
||||
row['status'] = 'review' if row['candidates'] else 'no_match'
|
||||
except (requests.RequestException, ValueError, RuntimeError) as exc:
|
||||
row.update(status='error', error=str(exc))
|
||||
report['products'].append(row)
|
||||
private_json(cache / ('applied-report.json' if apply else 'report.json'), report)
|
||||
click.echo(f"{product['id']} {row['status']}: {product['name']}")
|
||||
report['counts'] = {s:sum(r['status']==s for r in report['products']) for s in sorted({r['status'] for r in report['products']})}
|
||||
private_json(cache / ('applied-report.json' if apply else 'report.json'), report)
|
||||
click.echo(json.dumps(report['counts']))
|
||||
return report
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
56
ocado_grocy/order_state.py
Normal file
56
ocado_grocy/order_state.py
Normal file
|
|
@ -0,0 +1,56 @@
|
|||
"""Order-level completion, with conservative migration of the line journal."""
|
||||
import json
|
||||
import sqlite3
|
||||
from types import SimpleNamespace
|
||||
|
||||
from .grocy import align_recorded_lines, item_fingerprint
|
||||
from .pantry import classify
|
||||
from .receipt import parse_document
|
||||
|
||||
|
||||
def completed_orders(state, server, archive, config):
|
||||
"""Read only: a downloaded receipt alone is never evidence of an import."""
|
||||
if not state.exists():
|
||||
return set()
|
||||
with sqlite3.connect(f'{state.resolve().as_uri()}?mode=ro', uri=True) as db:
|
||||
tables = {r[0] for r in db.execute("SELECT name FROM sqlite_master WHERE type='table'")}
|
||||
tracked = dict(db.execute('SELECT order_id,status FROM loaded_orders WHERE server=?', (server,))) if 'loaded_orders' in tables else {}
|
||||
completed = {order for order, status in tracked.items() if status == 'complete'}
|
||||
if 'imports' not in tables:
|
||||
return completed
|
||||
legacy = {r[0] for r in db.execute('SELECT DISTINCT order_id FROM imports WHERE server=?', (server,))} - tracked.keys()
|
||||
for order in legacy:
|
||||
path = archive / f'{order}.json'
|
||||
if not path.exists():
|
||||
continue
|
||||
try:
|
||||
receipt = parse_document(json.loads(path.read_text()))
|
||||
if receipt.order_id != order:
|
||||
continue
|
||||
receipt = align_recorded_lines(receipt, server, SimpleNamespace(db=db))
|
||||
rows = {line:(fingerprint,status) for line,fingerprint,status in db.execute(
|
||||
'SELECT line,fingerprint,status FROM imports WHERE server=? AND order_id=?', (server,order))}
|
||||
if any(status != 'done' for _,status in rows.values()):
|
||||
continue
|
||||
selected = [(line,item) for line,item in enumerate(receipt.items,1)
|
||||
if classify(item,config.get('pantry',{}).get('overrides',{})).include]
|
||||
if all(rows.get(line) == (item_fingerprint(receipt,item),'done') for line,item in selected):
|
||||
completed.add(order)
|
||||
except (ValueError, KeyError, TypeError):
|
||||
# A changed or incomplete receipt needs the normal reconciliation path.
|
||||
continue
|
||||
return completed
|
||||
|
||||
|
||||
def mark_orders(state, server, orders, status):
|
||||
if not orders:
|
||||
return
|
||||
state.parent.mkdir(parents=True, exist_ok=True)
|
||||
with sqlite3.connect(state) as db:
|
||||
db.execute('''CREATE TABLE IF NOT EXISTS loaded_orders (
|
||||
server TEXT NOT NULL, order_id TEXT NOT NULL, status TEXT NOT NULL,
|
||||
updated_at TEXT NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
PRIMARY KEY(server,order_id))''')
|
||||
db.executemany('''INSERT INTO loaded_orders(server,order_id,status) VALUES (?,?,?)
|
||||
ON CONFLICT(server,order_id) DO UPDATE SET status=excluded.status, updated_at=CURRENT_TIMESTAMP''',
|
||||
[(server,order,status) for order in orders])
|
||||
45
ocado_grocy/pantry.py
Normal file
45
ocado_grocy/pantry.py
Normal file
|
|
@ -0,0 +1,45 @@
|
|||
"""Select shelf-stable stock using Ocado metadata, without a product-name list."""
|
||||
from dataclasses import dataclass
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Decision:
|
||||
include: bool
|
||||
reason: str
|
||||
uncertain: bool = False
|
||||
|
||||
|
||||
PERISHABLE_CATEGORIES = {"fresh & chilled food", "frozen food", "bakery"}
|
||||
SHELF_STABLE_CATEGORIES = {
|
||||
"food cupboard", "treats & snacks", "soft drinks, tea & coffee", "beer, wine & spirits",
|
||||
"household & cleaning", "home care & cleaning", "health, beauty & personal care",
|
||||
"beauty & toiletries", "pets", "health, medicines & wellbeing", "home & garden",
|
||||
"m&s food cupboard", "m&s snacks & treats",
|
||||
}
|
||||
|
||||
|
||||
def classify(item, overrides=None):
|
||||
overrides = overrides or {}
|
||||
key = item.product_id or item.name
|
||||
if key in overrides:
|
||||
value = overrides[key]
|
||||
if not isinstance(value, bool):
|
||||
raise ValueError(f"pantry override {key!r} must be true or false")
|
||||
return Decision(value, "explicit product override")
|
||||
storage_type = item.metadata.get("storage_type")
|
||||
if storage_type in {"FRIDGE", "FREEZER"}:
|
||||
return Decision(False, "requires cold storage")
|
||||
product = item.metadata.get("ocado", {})
|
||||
storage = str(product.get("storage", "")).casefold()
|
||||
if "keep refrigerated" in storage or "keep frozen" in storage:
|
||||
return Decision(False, "requires cold storage")
|
||||
categories = product.get("categoryPath", [])
|
||||
if isinstance(categories, str):
|
||||
categories = [categories]
|
||||
categories = {str(category).strip().casefold() for category in categories}
|
||||
if categories & PERISHABLE_CATEGORIES:
|
||||
return Decision(False, "fresh, chilled, frozen or bakery category")
|
||||
if categories & SHELF_STABLE_CATEGORIES:
|
||||
return Decision(True, "shelf-stable product category")
|
||||
# CUPBOARD alone also describes bananas, onions and potatoes.
|
||||
return Decision(False, "missing or ambiguous product category", uncertain=True)
|
||||
112
ocado_grocy/postprocess.py
Normal file
112
ocado_grocy/postprocess.py
Normal file
|
|
@ -0,0 +1,112 @@
|
|||
"""Enrich products belonging to the orders just processed, including reruns."""
|
||||
import hashlib
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
import click
|
||||
|
||||
from .grocy import Grocy, imported_products
|
||||
from .history import private_json
|
||||
from .images import run_images
|
||||
from .openfoodfacts import run_openfoodfacts
|
||||
from .enrichment_retry import RetryQueue
|
||||
|
||||
|
||||
def enrich_orders(config, config_path, archive, order_ids, *, images=True, openfoodfacts=True):
|
||||
if not order_ids or not (images or openfoodfacts):
|
||||
return {'errors':[]}
|
||||
connection = config.get('grocy', {})
|
||||
client = Grocy(os.environ.get('GROCY_URL', connection.get('url', '')),
|
||||
os.environ.get('GROCY_API_KEY', connection.get('api_key', '')))
|
||||
state = Path(config.get('import', {}).get('state_file', 'imports.sqlite3'))
|
||||
if not state.is_absolute():
|
||||
state = config_path.resolve().parent / state
|
||||
products = imported_products(client, state, order_ids)
|
||||
key = order_ids[0] if len(order_ids) == 1 and order_ids[0].isdigit() else hashlib.sha256(','.join(sorted(order_ids)).encode()).hexdigest()[:16]
|
||||
output = archive / 'enrichment' / key
|
||||
queue = RetryQueue(state, client.url)
|
||||
summary = {'server':client.url, 'order_ids':list(order_ids), 'product_ids':[p['id'] for p in products], 'errors':[]}
|
||||
if not products:
|
||||
queue.close()
|
||||
private_json(output / 'summary.json', summary)
|
||||
return summary
|
||||
if images:
|
||||
try:
|
||||
click.echo(f'Adding missing pictures for {len(products)} imported products...')
|
||||
queue.start('images', products)
|
||||
result = run_images(client, products, output / 'images.json', raise_errors=False)
|
||||
queue.results('images', result['products'])
|
||||
summary['images'] = result['counts']
|
||||
if result['counts'].get('error'):
|
||||
summary['errors'].append(f"Images: {result['counts']['error']} products failed")
|
||||
except Exception as exc:
|
||||
summary['errors'].append(f'Images: {exc}')
|
||||
if openfoodfacts:
|
||||
try:
|
||||
queue.start('openfoodfacts', products)
|
||||
contact = config.get('openfoodfacts', {}).get('contact')
|
||||
if not contact:
|
||||
raise ValueError('Set [openfoodfacts].contact in config.toml or use --no-openfoodfacts')
|
||||
click.echo(f'Matching {len(products)} imported products on Open Food Facts...')
|
||||
result = run_openfoodfacts(client, products, archive / 'openfoodfacts', contact)
|
||||
queue.results('openfoodfacts', result['products'])
|
||||
private_json(output / 'openfoodfacts.json', result)
|
||||
summary['openfoodfacts'] = result['counts']
|
||||
if result['counts'].get('error'):
|
||||
summary['errors'].append(f"Open Food Facts: {result['counts']['error']} products failed; see {output / 'openfoodfacts.json'}")
|
||||
except Exception as exc:
|
||||
summary['errors'].append(f'Open Food Facts: {exc}')
|
||||
queue.close()
|
||||
private_json(output / 'summary.json', summary)
|
||||
return summary
|
||||
|
||||
|
||||
def retry_enrichment(config, config_path, archive, *, images=True, openfoodfacts=True):
|
||||
connection = config.get('grocy', {})
|
||||
client = Grocy(os.environ.get('GROCY_URL', connection.get('url','')),
|
||||
os.environ.get('GROCY_API_KEY', connection.get('api_key','')))
|
||||
state = config_path.resolve().parent / config.get('import',{}).get('state_file','imports.sqlite3')
|
||||
if not state.exists():
|
||||
return {'errors':[]}
|
||||
queue = RetryQueue(state, client.url)
|
||||
try:
|
||||
# Journal scope protects other Grocy installations and non-imported products.
|
||||
tables = {r[0] for r in queue.db.execute("SELECT name FROM sqlite_master WHERE type='table'")}
|
||||
if 'imports' not in tables:
|
||||
return {'errors':[]}
|
||||
allowed = {int(r[0]) for r in queue.db.execute("SELECT DISTINCT product_id FROM imports WHERE server=? AND status='done'",(client.url,))}
|
||||
queue.migrate_reports(archive, allowed)
|
||||
enabled = [('images',images),('openfoodfacts',openfoodfacts)]
|
||||
summary = {'errors':[]}
|
||||
live = None
|
||||
for stage, active in enabled:
|
||||
pending = queue.pending(stage)
|
||||
if not active or not pending:
|
||||
continue
|
||||
if live is None:
|
||||
live = {int(p['id']):p for p in client.get('/objects/products')}
|
||||
products = [live[pid] for pid in sorted(pending & allowed & live.keys())]
|
||||
queue.results(stage,[{'product_id':pid,'status':'deleted_or_out_of_scope'} for pid in pending - {int(p['id']) for p in products}])
|
||||
if not products:
|
||||
continue
|
||||
click.echo(f'Retrying {stage} for {len(products)} previously failed products...')
|
||||
output = archive / 'enrichment' / 'retries'
|
||||
try:
|
||||
if stage == 'images':
|
||||
result = run_images(client, products, output/'images.json', raise_errors=False)
|
||||
else:
|
||||
contact = config.get('openfoodfacts',{}).get('contact')
|
||||
if not contact:
|
||||
raise ValueError('Set [openfoodfacts].contact in config.toml')
|
||||
result = run_openfoodfacts(client, products, archive/'openfoodfacts',contact)
|
||||
private_json(output/'openfoodfacts.json',result)
|
||||
queue.results(stage,result['products'])
|
||||
summary[stage] = result['counts']
|
||||
if result['counts'].get('error'):
|
||||
summary['errors'].append(f"{stage}: {result['counts']['error']} retries failed")
|
||||
except Exception as exc:
|
||||
summary['errors'].append(f'{stage}: {exc}')
|
||||
private_json(archive/'enrichment'/'retry-summary.json',summary)
|
||||
return summary
|
||||
finally:
|
||||
queue.close()
|
||||
38
ocado_grocy/product_metadata.py
Normal file
38
ocado_grocy/product_metadata.py
Normal file
|
|
@ -0,0 +1,38 @@
|
|||
"""Read imported metadata even when Grocy's sanitizer removes the pre tag."""
|
||||
import json
|
||||
from html import escape
|
||||
from lxml import html
|
||||
|
||||
|
||||
def read_metadata(description):
|
||||
try:
|
||||
text = html.fromstring(description or '<p></p>').text_content()
|
||||
except (html.etree.ParserError, ValueError):
|
||||
return {}
|
||||
for index, character in enumerate(text):
|
||||
if character == '{':
|
||||
try:
|
||||
value, _ = json.JSONDecoder().raw_decode(text[index:])
|
||||
if isinstance(value, dict) and 'Ocado product ID' in value:
|
||||
return value
|
||||
except ValueError:
|
||||
pass
|
||||
return {}
|
||||
|
||||
|
||||
def replace_metadata(description, metadata):
|
||||
tree = html.fragment_fromstring(description, create_parent='div')
|
||||
for node in tree.iter():
|
||||
for attr in ('text', 'tail'):
|
||||
text = getattr(node, attr) or ''
|
||||
for index, character in enumerate(text):
|
||||
if character != '{':
|
||||
continue
|
||||
try:
|
||||
old, end = json.JSONDecoder().raw_decode(text[index:])
|
||||
except ValueError:
|
||||
continue
|
||||
if isinstance(old, dict) and 'Ocado product ID' in old:
|
||||
setattr(node, attr, text[:index] + json.dumps(metadata, ensure_ascii=False, indent=2) + text[index + end:])
|
||||
return escape(tree.text or '') + ''.join(html.tostring(child, encoding='unicode') for child in tree)
|
||||
raise ValueError('Cannot safely locate imported metadata in description')
|
||||
161
ocado_grocy/receipt.py
Normal file
161
ocado_grocy/receipt.py
Normal file
|
|
@ -0,0 +1,161 @@
|
|||
"""Parse receipt data only: never treat the live basket as an order."""
|
||||
from dataclasses import asdict, dataclass, field
|
||||
from datetime import date, datetime
|
||||
from zoneinfo import ZoneInfo
|
||||
from decimal import Decimal, InvalidOperation
|
||||
import json
|
||||
|
||||
|
||||
class ImportError(ValueError):
|
||||
"""An input or integration error suitable for displaying to the user."""
|
||||
|
||||
|
||||
def number(value, label, *, money=False):
|
||||
if isinstance(value, dict):
|
||||
currency = value.get("currency", "GBP")
|
||||
if currency not in ("GBP", "GBX"):
|
||||
raise ImportError(f"Unsupported currency: {currency}")
|
||||
result = number(value.get("amount"), label, money=money)
|
||||
return result / 100 if currency == "GBX" else result
|
||||
text = str(value).strip().replace("£", "").replace(",", "")
|
||||
try:
|
||||
result = Decimal(text)
|
||||
except InvalidOperation as exc:
|
||||
raise ImportError(f"Invalid {label}: {value!r}") from exc
|
||||
if not result.is_finite() or result < 0:
|
||||
raise ImportError(f"Invalid {label}: {value!r}")
|
||||
return result
|
||||
|
||||
|
||||
def receipt_date(value):
|
||||
if not value:
|
||||
return None
|
||||
text = str(value).strip()
|
||||
try:
|
||||
return date.fromisoformat(text[:10]).isoformat()
|
||||
except ValueError:
|
||||
pass
|
||||
for fmt in ("%d/%m/%Y", "%d %B %Y", "%d %b %Y", "%A %d %B %Y"):
|
||||
try:
|
||||
return datetime.strptime(text, fmt).date().isoformat()
|
||||
except ValueError:
|
||||
pass
|
||||
raise ImportError(f"Unrecognised date: {value!r}; use YYYY-MM-DD")
|
||||
|
||||
|
||||
@dataclass
|
||||
class Item:
|
||||
name: str
|
||||
quantity: Decimal
|
||||
total: Decimal
|
||||
product_id: str = ""
|
||||
barcode: str = ""
|
||||
url: str = ""
|
||||
best_before: str | None = None
|
||||
metadata: dict = field(default_factory=dict)
|
||||
|
||||
@property
|
||||
def unit_price(self):
|
||||
return self.total / self.quantity
|
||||
|
||||
|
||||
@dataclass
|
||||
class Receipt:
|
||||
order_id: str
|
||||
purchased_date: str | None
|
||||
items: list[Item]
|
||||
metadata: dict = field(default_factory=dict)
|
||||
|
||||
def export(self):
|
||||
return asdict(self)
|
||||
|
||||
|
||||
def parse_document(data):
|
||||
"""Read a normalized cached receipt."""
|
||||
if not isinstance(data, dict) or not isinstance(data.get("items"), list):
|
||||
raise ImportError("Receipt JSON requires an items array")
|
||||
if data.get("currency", "GBP") != "GBP":
|
||||
raise ImportError("Only GBP receipts are supported")
|
||||
order_id = str(data.get("order_id") or "").strip()
|
||||
if not order_id:
|
||||
raise ImportError("Receipt has no order ID")
|
||||
items = []
|
||||
for row in data["items"]:
|
||||
if not isinstance(row, dict):
|
||||
raise ImportError("Receipt items must be objects")
|
||||
status = str(row.get("status", "")).lower()
|
||||
if status in {"unavailable", "cancelled", "rejected", "not delivered"}:
|
||||
continue
|
||||
quantity = number(row.get("delivered_quantity", row.get("quantity")), "quantity")
|
||||
if quantity == 0:
|
||||
continue
|
||||
name = str(row.get("name") or "").strip()
|
||||
if not name:
|
||||
raise ImportError("Receipt item has no name")
|
||||
total = number(row.get("total"), f"line total for {name}", money=True)
|
||||
metadata = dict(row.get("metadata") or {})
|
||||
metadata.update({k: v for k, v in row.items() if k not in {
|
||||
"name", "quantity", "delivered_quantity", "total", "product_id",
|
||||
"barcode", "url", "best_before", "metadata"}})
|
||||
items.append(Item(name, quantity, total, str(row.get("product_id") or ""),
|
||||
str(row.get("barcode") or ""), str(row.get("url") or ""),
|
||||
receipt_date(row.get("best_before")), metadata))
|
||||
if not items:
|
||||
raise ImportError("Receipt contains no delivered items")
|
||||
return Receipt(order_id, receipt_date(data.get("purchased_date")), items, data.get("metadata", {}))
|
||||
|
||||
|
||||
def product_metadata(product):
|
||||
return {key: product[key] for key in ("brand", "categoryPath", "size", "packSize",
|
||||
"description", "ingredients", "nutritionalInformation", "allergens", "storage",
|
||||
"guaranteedProductLife", "retailerProductId") if key in product}
|
||||
|
||||
|
||||
def parse_ocado_order(data, *, expected_order_id=None, product_urls=None):
|
||||
"""Parse the order JSON loaded by Playwright, never the page's live basket."""
|
||||
try:
|
||||
order = data["entities"]["order"][data["result"]]
|
||||
order_id = str(order["orderId"])
|
||||
if expected_order_id and order_id != str(expected_order_id):
|
||||
raise ImportError("Ocado API order does not match the requested order")
|
||||
if order["status"] != "DELIVERED":
|
||||
raise ImportError("Ocado order is not delivered")
|
||||
dates = order["dates"]
|
||||
delivered_at = datetime.fromisoformat(dates["deliveryStartDate"].replace("Z", "+00:00"))
|
||||
if delivered_at.tzinfo is not None:
|
||||
delivered_at = delivered_at.astimezone(ZoneInfo(dates.get("timeZoneId", "Europe/London")))
|
||||
purchased_date = delivered_at.date().isoformat()
|
||||
groups = order["groupedProducts"]
|
||||
products = list(groups["products"])
|
||||
for original in groups.get("substitutes", []):
|
||||
for replacement in original.get("substitutes", []):
|
||||
if replacement.get("status") == "ACCEPTED":
|
||||
products.append({"storageType": original.get("storageType"),
|
||||
"substituted_for": original["name"], **replacement})
|
||||
rows = []
|
||||
catalogue = data["entities"].get("product", {})
|
||||
for product in products:
|
||||
info = catalogue.get(product["productId"], {})
|
||||
retailer_id = str(product["retailerProductId"])
|
||||
metadata = {"ocado": product_metadata(info), "storage_type": product.get("storageType"),
|
||||
"pack_info": product.get("packInfo", {}), "promotions": product.get("promotions", []),
|
||||
"original_line_price": product["prices"].get("retail"),
|
||||
"price_per_item": product["prices"].get("pricePerItem"),
|
||||
"image_url": (product.get("image") or {}).get("src", "")}
|
||||
if product.get("substituted_for"):
|
||||
metadata["substituted_for"] = product["substituted_for"]
|
||||
rows.append({"name": " ".join(product["name"].split()), "quantity": product["quantity"],
|
||||
"total": product["prices"]["offered"], "product_id": retailer_id,
|
||||
"best_before": product.get("expirationDate"),
|
||||
"url": (product_urls or {}).get(retailer_id, ""), "metadata": metadata})
|
||||
# Stable ordering is independent of the browser's equal-expiry sorting.
|
||||
rows.sort(key=lambda row:(row["best_before"] or "9999-12-31", row["product_id"], row["name"]))
|
||||
receipt = parse_document({"order_id":order_id,"purchased_date":purchased_date,"items":rows})
|
||||
if sum(item.quantity for item in receipt.items) != number(order["totalItems"], "receipt item count"):
|
||||
raise ImportError("Delivered quantities do not match the order summary")
|
||||
if sum(item.total for item in receipt.items) != number(order["orderTotals"]["itemPriceAfterPromos"], "receipt product total"):
|
||||
raise ImportError("Delivered line prices do not match the order summary")
|
||||
receipt.metadata = {"source":"ocado-order-api", "order_totals":order["orderTotals"]}
|
||||
return receipt
|
||||
except (KeyError, TypeError, AttributeError) as exc:
|
||||
raise ImportError("Unrecognised Ocado order API structure") from exc
|
||||
Loading…
Add table
Add a link
Reference in a new issue