ocado-grocy/ocado_grocy/history.py

345 lines
18 KiB
Python

"""Browse Ocado history with a persistent, visible Playwright browser."""
from dataclasses import asdict, dataclass
from datetime import datetime
import json
import os
from pathlib import Path
import re
import sqlite3
import time
import tomllib
from urllib.parse import urlsplit
import click
from lxml import html
from .grocy import Grocy, Journal, import_receipt, refresh_imported_products, align_recorded_lines
from .paths import config_file, archive_directory, browser_directory, journal_file
from .pantry import classify
from .receipt import ImportError, parse_document, parse_ocado_order, receipt_date
@dataclass(frozen=True)
class OrderLink:
order_id: str
purchased_date: str
url: str
def private_json(path, data):
path.parent.mkdir(parents=True, exist_ok=True, mode=0o700)
temporary = path.with_suffix(path.suffix + ".tmp")
fd = os.open(temporary, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600)
with os.fdopen(fd, "w") as stream:
json.dump(data, stream, ensure_ascii=False, indent=2, default=str)
temporary.replace(path)
def order_links(content):
tree = html.fromstring(content)
found = {}
for anchor in tree.xpath('//a[contains(@href,"/orders/")]'):
href = anchor.get("href")
match = re.fullmatch(r"/orders/(\d+)/details", href)
text = " ".join(" ".join(anchor.itertext()).split())
if not match or not re.search(r"\bDelivered\b", text):
continue
stamp = re.search(r"([A-Z][a-z]{2} \d{1,2}, \d{4})", text)
if not stamp:
raise ImportError(f"No full delivery date on order {match[1]}")
purchased = datetime.strptime(stamp[1], "%b %d, %Y").date().isoformat()
found[match[1]] = OrderLink(match[1], purchased, "https://www.ocado.com" + href)
return list(found.values())
def wait_for_orders(page, timeout, echo):
deadline = time.monotonic() + timeout
notice = 0
while time.monotonic() < deadline:
if page.is_closed():
raise ImportError("Browser closed before sign-in completed")
if urlsplit(page.url).hostname == "www.ocado.com" and page.locator('a[href*="/orders/"]').count():
return
if time.monotonic() >= notice:
echo("Waiting for order history. Complete sign-in or any CAPTCHA in the browser.")
notice = time.monotonic() + 30
page.wait_for_timeout(1000)
raise ImportError("Timed out waiting for order history; rerun to reuse the saved browser session")
def discover_orders(page, echo, *, completed=(), order_ids=()):
completed = set(completed)
requested = set(order_ids)
last_count = -1
stable = 0
for _ in range(500):
count = page.locator('a[href*="/orders/"]').count()
if count == last_count:
stable += 1
else:
stable = 0
echo(f"Loaded {count} order links...")
links = order_links(page.content())
visible = {link.order_id for link in links}
# Ocado lists newest orders first. Keep the whole visible batch so
# incomplete orders alongside the completion boundary remain eligible.
if requested:
if requested <= visible:
return links
elif completed & visible:
return links
if page.locator('[data-test="order-list-no-more-orders-label"]').count():
return links
if stable >= 10:
raise ImportError("Order history stopped loading before Ocado's end-of-list marker; retry after checking the browser")
last_count = count
bottom = page.get_by_role("button", name="Back to top", exact=True)
if bottom.count():
bottom.scroll_into_view_if_needed()
else:
page.locator('a[href*="/orders/"]').last.scroll_into_view_if_needed()
page.mouse.wheel(0,1200)
page.wait_for_timeout(1500)
raise ImportError("Order list did not reach its end; refusing to call the download complete")
def download_orders(profile, archive, *, since=None, limit=None, login_only=False,
login_timeout=300, echo=print, order_ids=(), refresh=False, completed=()):
try:
from playwright.sync_api import sync_playwright, Error as BrowserError
except ModuleNotFoundError as exc:
raise ImportError("Install dependencies: pip install -e . and python -m playwright install chromium") from exc
profile.mkdir(parents=True, exist_ok=True, mode=0o700)
profile.chmod(0o700)
with sync_playwright() as playwright:
context = playwright.chromium.launch_persistent_context(
str(profile.resolve()), headless=False, viewport={"width":1280, "height":900})
try:
page = context.pages[0] if context.pages else context.new_page()
page.goto("https://www.ocado.com/orders", wait_until="domcontentloaded")
wait_for_orders(page, login_timeout, echo)
if login_only:
echo("Signed-in browser session saved.")
return [], []
links = discover_orders(page, echo, completed=completed, order_ids=order_ids)
links = save_order_manifest(archive, links)
links = select_orders(links, since, limit, order_ids, completed)
receipts, failures = [], []
for index, link in enumerate(links,1):
path = archive / f"{link.order_id}.json"
if path.exists() and not refresh:
receipt = parse_document(json.loads(path.read_text()))
if receipt.order_id != link.order_id or receipt.purchased_date != link.purchased_date:
raise ImportError(f"Cached receipt does not match order {link.order_id}")
receipts.append(receipt)
echo(f"[{index}/{len(links)}] Cached {link.order_id} ({link.purchased_date})")
continue
try:
endpoint = f"/api/order/v6/orders/{link.order_id}/decorated"
with page.expect_response(lambda response: urlsplit(response.url).path == endpoint,
timeout=45000) as captured:
page.goto(link.url, wait_until="domcontentloaded")
response = captured.value
if response.status != 200:
raise ImportError(f"Ocado receipt returned HTTP {response.status}")
data = response.json()
page.wait_for_selector('[data-test="order-online-receipt"]', timeout=30000)
hrefs = page.locator('[data-test="receipt-element-product-link"][href]').evaluate_all(
'(elements) => elements.map(element => element.getAttribute("href"))')
urls = {}
for href in hrefs:
match = re.search(r"/(\d+)$", href)
if match:
urls[match[1]] = "https://www.ocado.com" + href
receipt = parse_ocado_order(data, expected_order_id=link.order_id, product_urls=urls)
if receipt.purchased_date != link.purchased_date:
raise ImportError("Order-list date does not match the receipt API date")
private_json(path, receipt.export())
receipts.append(receipt)
echo(f"[{index}/{len(links)}] Saved {link.order_id}: {len(receipt.items)} lines")
except (ImportError, BrowserError, ValueError, KeyError) as exc:
message = str(exc) if isinstance(exc, ImportError) else type(exc).__name__
failures.append({"order_id":link.order_id,"error":message})
echo(f"[{index}/{len(links)}] Could not read {link.order_id}: {message}")
page.wait_for_timeout(750)
private_json(archive / "download-errors.json", failures)
return receipts, failures
finally:
try:
# Persistent profile is authoritative; this also preserves a portable cookie backup.
context.storage_state(path=str(profile / "auth.json"))
(profile / "auth.json").chmod(0o600)
finally:
context.close()
def save_order_manifest(archive, links):
"""Retain older history and pending imports after an incremental scan."""
path = archive / "orders.json"
saved = [OrderLink(**row) for row in json.loads(path.read_text())] if path.exists() else []
merged = {link.order_id: link for link in saved}
merged.update((link.order_id, link) for link in links)
links = sorted(merged.values(), key=lambda link: (link.purchased_date, link.order_id), reverse=True)
private_json(path, [asdict(link) for link in links])
return links
def select_orders(links, since=None, limit=None, order_ids=(), completed=()):
missing = set(order_ids) - {link.order_id for link in links}
if missing:
raise ImportError("Orders not found in delivered history: " + ", ".join(sorted(missing)))
links = sorted((link for link in links if not since or link.purchased_date >= since),
key=lambda link:(link.purchased_date,link.order_id), reverse=True)
if order_ids:
links = [link for link in links if link.order_id in order_ids]
if not order_ids:
links = [link for link in links if link.order_id not in completed]
if limit:
links = links[:limit]
return links
def cached_orders(archive, since=None, limit=None, order_ids=(), completed=()):
links = [OrderLink(**row) for row in json.loads((archive / "orders.json").read_text())]
links = select_orders(links, since, limit, order_ids, completed)
receipts, failures = [], []
for link in links:
path = archive / f"{link.order_id}.json"
if path.exists():
receipt = parse_document(json.loads(path.read_text()))
if receipt.order_id != link.order_id or receipt.purchased_date != link.purchased_date:
raise ImportError(f"Cached receipt does not match order {link.order_id}")
receipts.append(receipt)
else:
failures.append({"order_id":link.order_id,"error":"Receipt has not been downloaded"})
return receipts, failures
def run_import(receipts, config, config_path, archive, dry_run, failures, echo=print, refresh_existing=True,
all_products=True):
report = {"orders":len(receipts), "download_errors":failures, "items":[], "imported_lines":0, "refreshed_products":0}
settings = config.get("import", {})
journal = None
refreshed = set()
try:
if not dry_run:
connection = config.get("grocy", {})
url = os.environ.get("GROCY_URL", connection.get("url", ""))
key = os.environ.get("GROCY_API_KEY", connection.get("api_key", ""))
if not url or not key:
raise ImportError("Configure Grocy URL and API key before importing")
api = Grocy(url,key)
state = journal_file(config, config_path)
journal = Journal(state)
for receipt in receipts:
if not dry_run:
receipt = align_recorded_lines(receipt, api.url, journal)
if not dry_run and refresh_existing:
count = refresh_imported_products(receipt, api, journal, seen=refreshed)
report["refreshed_products"] += count
if count:
echo(f"Refreshed details for {count} previously imported products from {receipt.order_id}")
selected = set()
for line,item in enumerate(receipt.items,1):
decision = classify(item,config.get("pantry",{}).get("overrides",{}))
report["items"].append({"order_id":receipt.order_id, "date":receipt.purchased_date,
"line":line,"name":item.name,"product_id":item.product_id,
"quantity":str(item.quantity),"total":str(item.total),
**asdict(decision)})
if decision.include or all_products:
selected.add(line)
report["items"][-1]["include"] = True
if all_products:
report["items"][-1]["reason"] = "all products requested"
report["items"][-1]["uncertain"] = False
echo(f"Order {receipt.order_id} ({receipt.purchased_date}): {len(selected)}/{len(receipt.items)} selected lines")
if not dry_run:
report["imported_lines"] += import_receipt(
receipt,api,journal,{**settings,"products":config.get("products",{})},
echo=echo, include_lines=selected)
finally:
private_json(archive / "report.json",report)
if journal:
journal.db.close()
return report
@click.command(context_settings={"help_option_names":["-h","--help"]})
@click.option("--config", "config_path", type=click.Path(path_type=Path,dir_okay=False),default=config_file,show_default="~/.config/ocado-grocy/config.toml")
@click.option("--profile", type=click.Path(path_type=Path,file_okay=False),default=browser_directory,show_default="~/.local/state/ocado-grocy/browser")
@click.option("--archive", type=click.Path(path_type=Path,file_okay=False),default=archive_directory,show_default="~/.local/share/ocado-grocy/history")
@click.option("--since", help="Only orders delivered on or after YYYY-MM-DD.")
@click.option("--limit", type=click.IntRange(min=1),help="Process at most this many new delivered orders (or explicitly selected orders).")
@click.option("--dry-run", is_flag=True,help="Download and report selections without writing to Grocy.")
@click.option("--cached", is_flag=True,help="Use downloaded receipts without opening a browser.")
@click.option("--login-only", is_flag=True,help="Sign in and save cookies without downloading or importing orders.")
@click.option("--login-timeout",type=click.IntRange(min=1),default=300,show_default=True)
@click.option("--order-id", "order_ids", multiple=True, help="Process a specific delivered order; repeat for several.")
@click.option("--all-products/--shelf-stable-only", default=True, show_default=True,
help="Include all products, or restrict imports to shelf-stable items.")
@click.option("--refresh", is_flag=True, help="Download receipts again even when they are cached.")
@click.option("--refresh-existing/--no-refresh-existing", default=True, show_default=True,
help="Update details of previously imported products, including perishables.")
@click.option("--images/--no-images", default=True, show_default=True, help="Add missing Ocado product pictures after importing.")
@click.option("--openfoodfacts/--no-openfoodfacts", default=True, show_default=True, help="Match and enrich imported products after importing.")
def main(config_path,profile,archive,since,limit,dry_run,cached,login_only,login_timeout,refresh_existing,
order_ids,all_products,refresh,images,openfoodfacts):
"""Import delivered Ocado orders with Playwright; include all products by default."""
try:
if cached and login_only:
raise ImportError("--cached and --login-only cannot be combined")
if cached and refresh:
raise ImportError("--cached and --refresh cannot be combined")
config = tomllib.loads(config_path.read_text()) if config_path.exists() else {}
from .order_state import completed_orders, mark_orders
connection = config.get('grocy', {})
server = Grocy(os.environ.get('GROCY_URL', connection.get('url', '')),
os.environ.get('GROCY_API_KEY', connection.get('api_key', ''))).url
state = journal_file(config, config_path)
completed = completed_orders(state, server, archive, config)
if not dry_run and not login_only:
mark_orders(state, server, completed, 'complete')
retry_errors = []
if not dry_run and not login_only and (images or openfoodfacts):
from .postprocess import retry_enrichment
retry_errors = retry_enrichment(config, config_path, archive, images=images, openfoodfacts=openfoodfacts)['errors']
since = receipt_date(since)
archive.mkdir(parents=True,exist_ok=True,mode=0o700)
if cached:
receipts,failures = cached_orders(archive,since,limit,order_ids,completed)
else:
receipts,failures = download_orders(profile,archive,since=since,limit=limit,
login_only=login_only,login_timeout=login_timeout,echo=click.echo,order_ids=order_ids,refresh=refresh,completed=completed)
if login_only:
return
if not receipts and not failures:
if retry_errors:
raise ImportError('Enrichment retries still need attention: ' + '; '.join(retry_errors))
click.echo('No new delivered orders to import.')
return
processing = [receipt.order_id for receipt in receipts]
if not dry_run:
mark_orders(state, server, processing, 'processing')
report = run_import(receipts,config,config_path,archive,dry_run,failures,echo=click.echo,
refresh_existing=refresh_existing,all_products=all_products)
selected = sum(row["include"] for row in report["items"])
uncertain = sum(row["uncertain"] for row in report["items"])
click.echo(f"{len(receipts)} orders; {selected} selected lines; {report['imported_lines']} imported; "
f"{uncertain} uncertain lines. Report: {archive / 'report.json'}")
if not dry_run:
from .postprocess import enrich_orders
enrichment = enrich_orders(config, config_path, archive,
[receipt.order_id for receipt in receipts], images=images, openfoodfacts=openfoodfacts)
report['enrichment'] = enrichment
private_json(archive / 'report.json', report)
mark_orders(state, server, processing, 'complete')
if enrichment['errors'] or retry_errors:
raise ImportError('Stock import completed; enrichment needs a retry. ' + '; '.join(enrichment['errors'] + retry_errors))
if failures:
raise ImportError(f"{len(failures)} orders could not be downloaded; see report and retry")
except (ImportError,OSError,ValueError,KeyError,sqlite3.Error) as exc:
raise click.ClickException(str(exc)) from exc
if __name__ == "__main__":
main()