326 lines
17 KiB
Python
326 lines
17 KiB
Python
"""Browse Ocado history with a persistent, visible Playwright browser."""
|
|
from dataclasses import asdict, dataclass
|
|
from datetime import datetime
|
|
import json
|
|
import os
|
|
from pathlib import Path
|
|
import re
|
|
import sqlite3
|
|
import time
|
|
import tomllib
|
|
from urllib.parse import urlsplit
|
|
|
|
import click
|
|
from lxml import html
|
|
|
|
from .grocy import Grocy, Journal, import_receipt, refresh_imported_products, align_recorded_lines
|
|
from .pantry import classify
|
|
from .receipt import ImportError, parse_document, parse_ocado_order, receipt_date
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class OrderLink:
|
|
order_id: str
|
|
purchased_date: str
|
|
url: str
|
|
|
|
|
|
def private_json(path, data):
|
|
path.parent.mkdir(parents=True, exist_ok=True, mode=0o700)
|
|
temporary = path.with_suffix(path.suffix + ".tmp")
|
|
fd = os.open(temporary, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600)
|
|
with os.fdopen(fd, "w") as stream:
|
|
json.dump(data, stream, ensure_ascii=False, indent=2, default=str)
|
|
temporary.replace(path)
|
|
|
|
|
|
def order_links(content):
|
|
tree = html.fromstring(content)
|
|
found = {}
|
|
for anchor in tree.xpath('//a[contains(@href,"/orders/")]'):
|
|
href = anchor.get("href")
|
|
match = re.fullmatch(r"/orders/(\d+)/details", href)
|
|
text = " ".join(" ".join(anchor.itertext()).split())
|
|
if not match or not re.search(r"\bDelivered\b", text):
|
|
continue
|
|
stamp = re.search(r"([A-Z][a-z]{2} \d{1,2}, \d{4})", text)
|
|
if not stamp:
|
|
raise ImportError(f"No full delivery date on order {match[1]}")
|
|
purchased = datetime.strptime(stamp[1], "%b %d, %Y").date().isoformat()
|
|
found[match[1]] = OrderLink(match[1], purchased, "https://www.ocado.com" + href)
|
|
return list(found.values())
|
|
|
|
|
|
def wait_for_orders(page, timeout, echo):
|
|
deadline = time.monotonic() + timeout
|
|
notice = 0
|
|
while time.monotonic() < deadline:
|
|
if page.is_closed():
|
|
raise ImportError("Browser closed before sign-in completed")
|
|
if urlsplit(page.url).hostname == "www.ocado.com" and page.locator('a[href*="/orders/"]').count():
|
|
return
|
|
if time.monotonic() >= notice:
|
|
echo("Waiting for order history. Complete sign-in or any CAPTCHA in the browser.")
|
|
notice = time.monotonic() + 30
|
|
page.wait_for_timeout(1000)
|
|
raise ImportError("Timed out waiting for order history; rerun to reuse the saved browser session")
|
|
|
|
|
|
def discover_orders(page, echo):
|
|
last_count = -1
|
|
stable = 0
|
|
for _ in range(500):
|
|
count = page.locator('a[href*="/orders/"]').count()
|
|
if count == last_count:
|
|
stable += 1
|
|
else:
|
|
stable = 0
|
|
echo(f"Loaded {count} order links...")
|
|
if page.locator('[data-test="order-list-no-more-orders-label"]').count():
|
|
return order_links(page.content())
|
|
if stable >= 10:
|
|
raise ImportError("Order history stopped loading before Ocado's end-of-list marker; retry after checking the browser")
|
|
last_count = count
|
|
bottom = page.get_by_role("button", name="Back to top", exact=True)
|
|
if bottom.count():
|
|
bottom.scroll_into_view_if_needed()
|
|
else:
|
|
page.locator('a[href*="/orders/"]').last.scroll_into_view_if_needed()
|
|
page.mouse.wheel(0,1200)
|
|
page.wait_for_timeout(1500)
|
|
raise ImportError("Order list did not reach its end; refusing to call the download complete")
|
|
|
|
|
|
def download_orders(profile, archive, *, since=None, limit=None, login_only=False,
|
|
login_timeout=300, echo=print, order_ids=(), refresh=False, completed=()):
|
|
try:
|
|
from playwright.sync_api import sync_playwright, Error as BrowserError
|
|
except ModuleNotFoundError as exc:
|
|
raise ImportError("Install dependencies: pip install -e . and python -m playwright install chromium") from exc
|
|
profile.mkdir(parents=True, exist_ok=True, mode=0o700)
|
|
profile.chmod(0o700)
|
|
with sync_playwright() as playwright:
|
|
context = playwright.chromium.launch_persistent_context(
|
|
str(profile.resolve()), headless=False, viewport={"width":1280, "height":900})
|
|
try:
|
|
page = context.pages[0] if context.pages else context.new_page()
|
|
page.goto("https://www.ocado.com/orders", wait_until="domcontentloaded")
|
|
wait_for_orders(page, login_timeout, echo)
|
|
if login_only:
|
|
echo("Signed-in browser session saved.")
|
|
return [], []
|
|
links = discover_orders(page, echo)
|
|
private_json(archive / "orders.json", [asdict(link) for link in links])
|
|
links = select_orders(links, since, limit, order_ids, completed)
|
|
receipts, failures = [], []
|
|
for index, link in enumerate(links,1):
|
|
path = archive / f"{link.order_id}.json"
|
|
if path.exists() and not refresh:
|
|
receipt = parse_document(json.loads(path.read_text()))
|
|
if receipt.order_id != link.order_id or receipt.purchased_date != link.purchased_date:
|
|
raise ImportError(f"Cached receipt does not match order {link.order_id}")
|
|
receipts.append(receipt)
|
|
echo(f"[{index}/{len(links)}] Cached {link.order_id} ({link.purchased_date})")
|
|
continue
|
|
try:
|
|
endpoint = f"/api/order/v6/orders/{link.order_id}/decorated"
|
|
with page.expect_response(lambda response: urlsplit(response.url).path == endpoint,
|
|
timeout=45000) as captured:
|
|
page.goto(link.url, wait_until="domcontentloaded")
|
|
response = captured.value
|
|
if response.status != 200:
|
|
raise ImportError(f"Ocado receipt returned HTTP {response.status}")
|
|
data = response.json()
|
|
page.wait_for_selector('[data-test="order-online-receipt"]', timeout=30000)
|
|
hrefs = page.locator('[data-test="receipt-element-product-link"][href]').evaluate_all(
|
|
'(elements) => elements.map(element => element.getAttribute("href"))')
|
|
urls = {}
|
|
for href in hrefs:
|
|
match = re.search(r"/(\d+)$", href)
|
|
if match:
|
|
urls[match[1]] = "https://www.ocado.com" + href
|
|
receipt = parse_ocado_order(data, expected_order_id=link.order_id, product_urls=urls)
|
|
if receipt.purchased_date != link.purchased_date:
|
|
raise ImportError("Order-list date does not match the receipt API date")
|
|
private_json(path, receipt.export())
|
|
receipts.append(receipt)
|
|
echo(f"[{index}/{len(links)}] Saved {link.order_id}: {len(receipt.items)} lines")
|
|
except (ImportError, BrowserError, ValueError, KeyError) as exc:
|
|
message = str(exc) if isinstance(exc, ImportError) else type(exc).__name__
|
|
failures.append({"order_id":link.order_id,"error":message})
|
|
echo(f"[{index}/{len(links)}] Could not read {link.order_id}: {message}")
|
|
page.wait_for_timeout(750)
|
|
private_json(archive / "download-errors.json", failures)
|
|
return receipts, failures
|
|
finally:
|
|
try:
|
|
# Persistent profile is authoritative; this also preserves a portable cookie backup.
|
|
context.storage_state(path=str(profile / "auth.json"))
|
|
(profile / "auth.json").chmod(0o600)
|
|
finally:
|
|
context.close()
|
|
|
|
|
|
def select_orders(links, since=None, limit=None, order_ids=(), completed=()):
|
|
missing = set(order_ids) - {link.order_id for link in links}
|
|
if missing:
|
|
raise ImportError("Orders not found in delivered history: " + ", ".join(sorted(missing)))
|
|
links = sorted((link for link in links if not since or link.purchased_date >= since),
|
|
key=lambda link:(link.purchased_date,link.order_id), reverse=True)
|
|
if order_ids:
|
|
links = [link for link in links if link.order_id in order_ids]
|
|
if not order_ids:
|
|
links = [link for link in links if link.order_id not in completed]
|
|
if limit:
|
|
links = links[:limit]
|
|
return links
|
|
|
|
|
|
def cached_orders(archive, since=None, limit=None, order_ids=(), completed=()):
|
|
links = [OrderLink(**row) for row in json.loads((archive / "orders.json").read_text())]
|
|
links = select_orders(links, since, limit, order_ids, completed)
|
|
receipts, failures = [], []
|
|
for link in links:
|
|
path = archive / f"{link.order_id}.json"
|
|
if path.exists():
|
|
receipt = parse_document(json.loads(path.read_text()))
|
|
if receipt.order_id != link.order_id or receipt.purchased_date != link.purchased_date:
|
|
raise ImportError(f"Cached receipt does not match order {link.order_id}")
|
|
receipts.append(receipt)
|
|
else:
|
|
failures.append({"order_id":link.order_id,"error":"Receipt has not been downloaded"})
|
|
return receipts, failures
|
|
|
|
|
|
def run_import(receipts, config, config_path, archive, dry_run, failures, echo=print, refresh_existing=True,
|
|
all_products=True):
|
|
report = {"orders":len(receipts), "download_errors":failures, "items":[], "imported_lines":0, "refreshed_products":0}
|
|
settings = config.get("import", {})
|
|
journal = None
|
|
refreshed = set()
|
|
try:
|
|
if not dry_run:
|
|
connection = config.get("grocy", {})
|
|
url = os.environ.get("GROCY_URL", connection.get("url", ""))
|
|
key = os.environ.get("GROCY_API_KEY", connection.get("api_key", ""))
|
|
if not url or not key:
|
|
raise ImportError("Configure Grocy URL and API key before importing")
|
|
api = Grocy(url,key)
|
|
state = Path(settings.get("state_file", "imports.sqlite3"))
|
|
if not state.is_absolute():
|
|
state = config_path.resolve().parent / state
|
|
journal = Journal(state)
|
|
for receipt in receipts:
|
|
if not dry_run:
|
|
receipt = align_recorded_lines(receipt, api.url, journal)
|
|
if not dry_run and refresh_existing:
|
|
count = refresh_imported_products(receipt, api, journal, seen=refreshed)
|
|
report["refreshed_products"] += count
|
|
if count:
|
|
echo(f"Refreshed details for {count} previously imported products from {receipt.order_id}")
|
|
selected = set()
|
|
for line,item in enumerate(receipt.items,1):
|
|
decision = classify(item,config.get("pantry",{}).get("overrides",{}))
|
|
report["items"].append({"order_id":receipt.order_id, "date":receipt.purchased_date,
|
|
"line":line,"name":item.name,"product_id":item.product_id,
|
|
"quantity":str(item.quantity),"total":str(item.total),
|
|
**asdict(decision)})
|
|
if decision.include or all_products:
|
|
selected.add(line)
|
|
report["items"][-1]["include"] = True
|
|
if all_products:
|
|
report["items"][-1]["reason"] = "all products requested"
|
|
report["items"][-1]["uncertain"] = False
|
|
echo(f"Order {receipt.order_id} ({receipt.purchased_date}): {len(selected)}/{len(receipt.items)} selected lines")
|
|
if not dry_run:
|
|
report["imported_lines"] += import_receipt(
|
|
receipt,api,journal,{**settings,"products":config.get("products",{})},
|
|
echo=echo, include_lines=selected)
|
|
finally:
|
|
private_json(archive / "report.json",report)
|
|
if journal:
|
|
journal.db.close()
|
|
return report
|
|
|
|
|
|
@click.command(context_settings={"help_option_names":["-h","--help"]})
|
|
@click.option("--config", "config_path", type=click.Path(path_type=Path,dir_okay=False),default="config.toml",show_default=True)
|
|
@click.option("--profile", type=click.Path(path_type=Path,file_okay=False),default=".ocado-browser",show_default=True)
|
|
@click.option("--archive", type=click.Path(path_type=Path,file_okay=False),default="history",show_default=True)
|
|
@click.option("--since", help="Only orders delivered on or after YYYY-MM-DD.")
|
|
@click.option("--limit", type=click.IntRange(min=1),help="Process at most this many new delivered orders (or explicitly selected orders).")
|
|
@click.option("--dry-run", is_flag=True,help="Download and report selections without writing to Grocy.")
|
|
@click.option("--cached", is_flag=True,help="Use downloaded receipts without opening a browser.")
|
|
@click.option("--login-only", is_flag=True,help="Sign in and save cookies without downloading or importing orders.")
|
|
@click.option("--login-timeout",type=click.IntRange(min=1),default=300,show_default=True)
|
|
@click.option("--order-id", "order_ids", multiple=True, help="Process a specific delivered order; repeat for several.")
|
|
@click.option("--all-products/--shelf-stable-only", default=True, show_default=True,
|
|
help="Include all products, or restrict imports to shelf-stable items.")
|
|
@click.option("--refresh", is_flag=True, help="Download receipts again even when they are cached.")
|
|
@click.option("--refresh-existing/--no-refresh-existing", default=True, show_default=True,
|
|
help="Update details of previously imported products, including perishables.")
|
|
@click.option("--images/--no-images", default=True, show_default=True, help="Add missing Ocado product pictures after importing.")
|
|
@click.option("--openfoodfacts/--no-openfoodfacts", default=True, show_default=True, help="Match and enrich imported products after importing.")
|
|
def main(config_path,profile,archive,since,limit,dry_run,cached,login_only,login_timeout,refresh_existing,
|
|
order_ids,all_products,refresh,images,openfoodfacts):
|
|
"""Import delivered Ocado orders with Playwright; include all products by default."""
|
|
try:
|
|
if cached and login_only:
|
|
raise ImportError("--cached and --login-only cannot be combined")
|
|
if cached and refresh:
|
|
raise ImportError("--cached and --refresh cannot be combined")
|
|
config = tomllib.loads(config_path.read_text()) if config_path.exists() else {}
|
|
from .order_state import completed_orders, mark_orders
|
|
connection = config.get('grocy', {})
|
|
server = Grocy(os.environ.get('GROCY_URL', connection.get('url', '')),
|
|
os.environ.get('GROCY_API_KEY', connection.get('api_key', ''))).url
|
|
state = Path(config.get('import', {}).get('state_file', 'imports.sqlite3'))
|
|
if not state.is_absolute():
|
|
state = config_path.resolve().parent / state
|
|
completed = completed_orders(state, server, archive, config)
|
|
if not dry_run and not login_only:
|
|
mark_orders(state, server, completed, 'complete')
|
|
retry_errors = []
|
|
if not dry_run and not login_only and (images or openfoodfacts):
|
|
from .postprocess import retry_enrichment
|
|
retry_errors = retry_enrichment(config, config_path, archive, images=images, openfoodfacts=openfoodfacts)['errors']
|
|
since = receipt_date(since)
|
|
archive.mkdir(parents=True,exist_ok=True,mode=0o700)
|
|
if cached:
|
|
receipts,failures = cached_orders(archive,since,limit,order_ids,completed)
|
|
else:
|
|
receipts,failures = download_orders(profile,archive,since=since,limit=limit,
|
|
login_only=login_only,login_timeout=login_timeout,echo=click.echo,order_ids=order_ids,refresh=refresh,completed=completed)
|
|
if login_only:
|
|
return
|
|
if not receipts and not failures:
|
|
if retry_errors:
|
|
raise ImportError('Enrichment retries still need attention: ' + '; '.join(retry_errors))
|
|
click.echo('No new delivered orders to import.')
|
|
return
|
|
processing = [receipt.order_id for receipt in receipts]
|
|
if not dry_run:
|
|
mark_orders(state, server, processing, 'processing')
|
|
report = run_import(receipts,config,config_path,archive,dry_run,failures,echo=click.echo,
|
|
refresh_existing=refresh_existing,all_products=all_products)
|
|
selected = sum(row["include"] for row in report["items"])
|
|
uncertain = sum(row["uncertain"] for row in report["items"])
|
|
click.echo(f"{len(receipts)} orders; {selected} selected lines; {report['imported_lines']} imported; "
|
|
f"{uncertain} uncertain lines. Report: {archive / 'report.json'}")
|
|
if not dry_run:
|
|
from .postprocess import enrich_orders
|
|
enrichment = enrich_orders(config, config_path, archive,
|
|
[receipt.order_id for receipt in receipts], images=images, openfoodfacts=openfoodfacts)
|
|
report['enrichment'] = enrichment
|
|
private_json(archive / 'report.json', report)
|
|
mark_orders(state, server, processing, 'complete')
|
|
if enrichment['errors'] or retry_errors:
|
|
raise ImportError('Stock import completed; enrichment needs a retry. ' + '; '.join(enrichment['errors'] + retry_errors))
|
|
if failures:
|
|
raise ImportError(f"{len(failures)} orders could not be downloaded; see report and retry")
|
|
except (ImportError,OSError,ValueError,KeyError,sqlite3.Error) as exc:
|
|
raise click.ClickException(str(exc)) from exc
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|