"""Browse Ocado history with a persistent, visible Playwright browser.""" from dataclasses import asdict, dataclass from datetime import datetime import json import os from pathlib import Path import re import sqlite3 import time import tomllib from urllib.parse import urlsplit import click from lxml import html from .grocy import Grocy, Journal, import_receipt, refresh_imported_products, align_recorded_lines from .paths import config_file, archive_directory, browser_directory, journal_file from .pantry import classify from .receipt import ImportError, parse_document, parse_ocado_order, receipt_date @dataclass(frozen=True) class OrderLink: order_id: str purchased_date: str url: str def private_json(path, data): path.parent.mkdir(parents=True, exist_ok=True, mode=0o700) temporary = path.with_suffix(path.suffix + ".tmp") fd = os.open(temporary, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600) with os.fdopen(fd, "w") as stream: json.dump(data, stream, ensure_ascii=False, indent=2, default=str) temporary.replace(path) def order_links(content): tree = html.fromstring(content) found = {} for anchor in tree.xpath('//a[contains(@href,"/orders/")]'): href = anchor.get("href") match = re.fullmatch(r"/orders/(\d+)/details", href) text = " ".join(" ".join(anchor.itertext()).split()) if not match or not re.search(r"\bDelivered\b", text): continue stamp = re.search(r"([A-Z][a-z]{2} \d{1,2}, \d{4})", text) if not stamp: raise ImportError(f"No full delivery date on order {match[1]}") purchased = datetime.strptime(stamp[1], "%b %d, %Y").date().isoformat() found[match[1]] = OrderLink(match[1], purchased, "https://www.ocado.com" + href) return list(found.values()) def wait_for_orders(page, timeout, echo): deadline = time.monotonic() + timeout notice = 0 while time.monotonic() < deadline: if page.is_closed(): raise ImportError("Browser closed before sign-in completed") if urlsplit(page.url).hostname == "www.ocado.com" and page.locator('a[href*="/orders/"]').count(): return if time.monotonic() >= notice: echo("Waiting for order history. Complete sign-in or any CAPTCHA in the browser.") notice = time.monotonic() + 30 page.wait_for_timeout(1000) raise ImportError("Timed out waiting for order history; rerun to reuse the saved browser session") def discover_orders(page, echo, *, completed=(), order_ids=()): completed = set(completed) requested = set(order_ids) last_count = -1 stable = 0 for _ in range(500): count = page.locator('a[href*="/orders/"]').count() if count == last_count: stable += 1 else: stable = 0 echo(f"Loaded {count} order links...") links = order_links(page.content()) visible = {link.order_id for link in links} # Ocado lists newest orders first. Keep the whole visible batch so # incomplete orders alongside the completion boundary remain eligible. if requested: if requested <= visible: return links elif completed & visible: return links if page.locator('[data-test="order-list-no-more-orders-label"]').count(): return links if stable >= 10: raise ImportError("Order history stopped loading before Ocado's end-of-list marker; retry after checking the browser") last_count = count bottom = page.get_by_role("button", name="Back to top", exact=True) if bottom.count(): bottom.scroll_into_view_if_needed() else: page.locator('a[href*="/orders/"]').last.scroll_into_view_if_needed() page.mouse.wheel(0,1200) page.wait_for_timeout(1500) raise ImportError("Order list did not reach its end; refusing to call the download complete") def download_orders(profile, archive, *, since=None, limit=None, login_only=False, login_timeout=300, echo=print, order_ids=(), refresh=False, completed=()): try: from playwright.sync_api import sync_playwright, Error as BrowserError except ModuleNotFoundError as exc: raise ImportError("Install dependencies: pip install -e . and python -m playwright install chromium") from exc profile.mkdir(parents=True, exist_ok=True, mode=0o700) profile.chmod(0o700) with sync_playwright() as playwright: context = playwright.chromium.launch_persistent_context( str(profile.resolve()), headless=False, viewport={"width":1280, "height":900}) try: page = context.pages[0] if context.pages else context.new_page() page.goto("https://www.ocado.com/orders", wait_until="domcontentloaded") wait_for_orders(page, login_timeout, echo) if login_only: echo("Signed-in browser session saved.") return [], [] links = discover_orders(page, echo, completed=completed, order_ids=order_ids) links = save_order_manifest(archive, links) links = select_orders(links, since, limit, order_ids, completed) receipts, failures = [], [] for index, link in enumerate(links,1): path = archive / f"{link.order_id}.json" if path.exists() and not refresh: receipt = parse_document(json.loads(path.read_text())) if receipt.order_id != link.order_id or receipt.purchased_date != link.purchased_date: raise ImportError(f"Cached receipt does not match order {link.order_id}") receipts.append(receipt) echo(f"[{index}/{len(links)}] Cached {link.order_id} ({link.purchased_date})") continue try: endpoint = f"/api/order/v6/orders/{link.order_id}/decorated" with page.expect_response(lambda response: urlsplit(response.url).path == endpoint, timeout=45000) as captured: page.goto(link.url, wait_until="domcontentloaded") response = captured.value if response.status != 200: raise ImportError(f"Ocado receipt returned HTTP {response.status}") data = response.json() page.wait_for_selector('[data-test="order-online-receipt"]', timeout=30000) hrefs = page.locator('[data-test="receipt-element-product-link"][href]').evaluate_all( '(elements) => elements.map(element => element.getAttribute("href"))') urls = {} for href in hrefs: match = re.search(r"/(\d+)$", href) if match: urls[match[1]] = "https://www.ocado.com" + href receipt = parse_ocado_order(data, expected_order_id=link.order_id, product_urls=urls) if receipt.purchased_date != link.purchased_date: raise ImportError("Order-list date does not match the receipt API date") private_json(path, receipt.export()) receipts.append(receipt) echo(f"[{index}/{len(links)}] Saved {link.order_id}: {len(receipt.items)} lines") except (ImportError, BrowserError, ValueError, KeyError) as exc: message = str(exc) if isinstance(exc, ImportError) else type(exc).__name__ failures.append({"order_id":link.order_id,"error":message}) echo(f"[{index}/{len(links)}] Could not read {link.order_id}: {message}") page.wait_for_timeout(750) private_json(archive / "download-errors.json", failures) return receipts, failures finally: try: # Persistent profile is authoritative; this also preserves a portable cookie backup. context.storage_state(path=str(profile / "auth.json")) (profile / "auth.json").chmod(0o600) finally: context.close() def save_order_manifest(archive, links): """Retain older history and pending imports after an incremental scan.""" path = archive / "orders.json" saved = [OrderLink(**row) for row in json.loads(path.read_text())] if path.exists() else [] merged = {link.order_id: link for link in saved} merged.update((link.order_id, link) for link in links) links = sorted(merged.values(), key=lambda link: (link.purchased_date, link.order_id), reverse=True) private_json(path, [asdict(link) for link in links]) return links def select_orders(links, since=None, limit=None, order_ids=(), completed=()): missing = set(order_ids) - {link.order_id for link in links} if missing: raise ImportError("Orders not found in delivered history: " + ", ".join(sorted(missing))) links = sorted((link for link in links if not since or link.purchased_date >= since), key=lambda link:(link.purchased_date,link.order_id), reverse=True) if order_ids: links = [link for link in links if link.order_id in order_ids] if not order_ids: links = [link for link in links if link.order_id not in completed] if limit: links = links[:limit] return links def cached_orders(archive, since=None, limit=None, order_ids=(), completed=()): links = [OrderLink(**row) for row in json.loads((archive / "orders.json").read_text())] links = select_orders(links, since, limit, order_ids, completed) receipts, failures = [], [] for link in links: path = archive / f"{link.order_id}.json" if path.exists(): receipt = parse_document(json.loads(path.read_text())) if receipt.order_id != link.order_id or receipt.purchased_date != link.purchased_date: raise ImportError(f"Cached receipt does not match order {link.order_id}") receipts.append(receipt) else: failures.append({"order_id":link.order_id,"error":"Receipt has not been downloaded"}) return receipts, failures def run_import(receipts, config, config_path, archive, dry_run, failures, echo=print, refresh_existing=True, all_products=True): report = {"orders":len(receipts), "download_errors":failures, "items":[], "imported_lines":0, "refreshed_products":0} settings = config.get("import", {}) journal = None refreshed = set() try: if not dry_run: connection = config.get("grocy", {}) url = os.environ.get("GROCY_URL", connection.get("url", "")) key = os.environ.get("GROCY_API_KEY", connection.get("api_key", "")) if not url or not key: raise ImportError("Configure Grocy URL and API key before importing") api = Grocy(url,key) state = journal_file(config, config_path) journal = Journal(state) for receipt in receipts: if not dry_run: receipt = align_recorded_lines(receipt, api.url, journal) if not dry_run and refresh_existing: count = refresh_imported_products(receipt, api, journal, seen=refreshed) report["refreshed_products"] += count if count: echo(f"Refreshed details for {count} previously imported products from {receipt.order_id}") selected = set() for line,item in enumerate(receipt.items,1): decision = classify(item,config.get("pantry",{}).get("overrides",{})) report["items"].append({"order_id":receipt.order_id, "date":receipt.purchased_date, "line":line,"name":item.name,"product_id":item.product_id, "quantity":str(item.quantity),"total":str(item.total), **asdict(decision)}) if decision.include or all_products: selected.add(line) report["items"][-1]["include"] = True if all_products: report["items"][-1]["reason"] = "all products requested" report["items"][-1]["uncertain"] = False echo(f"Order {receipt.order_id} ({receipt.purchased_date}): {len(selected)}/{len(receipt.items)} selected lines") if not dry_run: report["imported_lines"] += import_receipt( receipt,api,journal,{**settings,"products":config.get("products",{})}, echo=echo, include_lines=selected) finally: private_json(archive / "report.json",report) if journal: journal.db.close() return report @click.command(context_settings={"help_option_names":["-h","--help"]}) @click.option("--config", "config_path", type=click.Path(path_type=Path,dir_okay=False),default=config_file,show_default="~/.config/ocado-grocy/config.toml") @click.option("--profile", type=click.Path(path_type=Path,file_okay=False),default=browser_directory,show_default="~/.local/state/ocado-grocy/browser") @click.option("--archive", type=click.Path(path_type=Path,file_okay=False),default=archive_directory,show_default="~/.local/share/ocado-grocy/history") @click.option("--since", help="Only orders delivered on or after YYYY-MM-DD.") @click.option("--limit", type=click.IntRange(min=1),help="Process at most this many new delivered orders (or explicitly selected orders).") @click.option("--dry-run", is_flag=True,help="Download and report selections without writing to Grocy.") @click.option("--cached", is_flag=True,help="Use downloaded receipts without opening a browser.") @click.option("--login-only", is_flag=True,help="Sign in and save cookies without downloading or importing orders.") @click.option("--login-timeout",type=click.IntRange(min=1),default=300,show_default=True) @click.option("--order-id", "order_ids", multiple=True, help="Process a specific delivered order; repeat for several.") @click.option("--all-products/--shelf-stable-only", default=True, show_default=True, help="Include all products, or restrict imports to shelf-stable items.") @click.option("--refresh", is_flag=True, help="Download receipts again even when they are cached.") @click.option("--refresh-existing/--no-refresh-existing", default=True, show_default=True, help="Update details of previously imported products, including perishables.") @click.option("--images/--no-images", default=True, show_default=True, help="Add missing Ocado product pictures after importing.") @click.option("--openfoodfacts/--no-openfoodfacts", default=True, show_default=True, help="Match and enrich imported products after importing.") def main(config_path,profile,archive,since,limit,dry_run,cached,login_only,login_timeout,refresh_existing, order_ids,all_products,refresh,images,openfoodfacts): """Import delivered Ocado orders with Playwright; include all products by default.""" try: if cached and login_only: raise ImportError("--cached and --login-only cannot be combined") if cached and refresh: raise ImportError("--cached and --refresh cannot be combined") config = tomllib.loads(config_path.read_text()) if config_path.exists() else {} from .order_state import completed_orders, mark_orders connection = config.get('grocy', {}) server = Grocy(os.environ.get('GROCY_URL', connection.get('url', '')), os.environ.get('GROCY_API_KEY', connection.get('api_key', ''))).url state = journal_file(config, config_path) completed = completed_orders(state, server, archive, config) if not dry_run and not login_only: mark_orders(state, server, completed, 'complete') retry_errors = [] if not dry_run and not login_only and (images or openfoodfacts): from .postprocess import retry_enrichment retry_errors = retry_enrichment(config, config_path, archive, images=images, openfoodfacts=openfoodfacts)['errors'] since = receipt_date(since) archive.mkdir(parents=True,exist_ok=True,mode=0o700) if cached: receipts,failures = cached_orders(archive,since,limit,order_ids,completed) else: receipts,failures = download_orders(profile,archive,since=since,limit=limit, login_only=login_only,login_timeout=login_timeout,echo=click.echo,order_ids=order_ids,refresh=refresh,completed=completed) if login_only: return if not receipts and not failures: if retry_errors: raise ImportError('Enrichment retries still need attention: ' + '; '.join(retry_errors)) click.echo('No new delivered orders to import.') return processing = [receipt.order_id for receipt in receipts] if not dry_run: mark_orders(state, server, processing, 'processing') report = run_import(receipts,config,config_path,archive,dry_run,failures,echo=click.echo, refresh_existing=refresh_existing,all_products=all_products) selected = sum(row["include"] for row in report["items"]) uncertain = sum(row["uncertain"] for row in report["items"]) click.echo(f"{len(receipts)} orders; {selected} selected lines; {report['imported_lines']} imported; " f"{uncertain} uncertain lines. Report: {archive / 'report.json'}") if not dry_run: from .postprocess import enrich_orders enrichment = enrich_orders(config, config_path, archive, [receipt.order_id for receipt in receipts], images=images, openfoodfacts=openfoodfacts) report['enrichment'] = enrichment private_json(archive / 'report.json', report) mark_orders(state, server, processing, 'complete') if enrichment['errors'] or retry_errors: raise ImportError('Stock import completed; enrichment needs a retry. ' + '; '.join(enrichment['errors'] + retry_errors)) if failures: raise ImportError(f"{len(failures)} orders could not be downloaded; see report and retry") except (ImportError,OSError,ValueError,KeyError,sqlite3.Error) as exc: raise click.ClickException(str(exc)) from exc if __name__ == "__main__": main()