From 628e5c3823b292e81266fae1a46907af2dd2dfe1 Mon Sep 17 00:00:00 2001 From: Edward Betts Date: Sat, 19 Sep 2026 11:32:53 +0100 Subject: [PATCH] Initial commit. --- .gitignore | 24 + README.md | 178 ++++++ config.example.toml | 20 + ocado_grocy/__init__.py | 1 + ocado_grocy/__main__.py | 3 + ocado_grocy/cli.py | 4 + ocado_grocy/enrichment_retry.py | 53 ++ ocado_grocy/grocy.py | 304 ++++++++++ ocado_grocy/history.py | 326 +++++++++++ ocado_grocy/images.py | 113 ++++ ocado_grocy/openfoodfacts.py | 283 +++++++++ ocado_grocy/order_state.py | 56 ++ ocado_grocy/pantry.py | 45 ++ ocado_grocy/postprocess.py | 112 ++++ ocado_grocy/product_metadata.py | 38 ++ ocado_grocy/receipt.py | 161 +++++ pyproject.toml | 25 + tests/fixtures/history_receipt.json | 870 ++++++++++++++++++++++++++++ tests/test_enrichment_retry.py | 85 +++ tests/test_grocy.py | 151 +++++ tests/test_history.py | 140 +++++ tests/test_images.py | 87 +++ tests/test_openfoodfacts.py | 139 +++++ tests/test_order_state.py | 58 ++ tests/test_postprocess.py | 84 +++ tests/test_receipt.py | 118 ++++ 26 files changed, 3478 insertions(+) create mode 100644 .gitignore create mode 100644 README.md create mode 100644 config.example.toml create mode 100644 ocado_grocy/__init__.py create mode 100644 ocado_grocy/__main__.py create mode 100644 ocado_grocy/cli.py create mode 100644 ocado_grocy/enrichment_retry.py create mode 100644 ocado_grocy/grocy.py create mode 100644 ocado_grocy/history.py create mode 100644 ocado_grocy/images.py create mode 100644 ocado_grocy/openfoodfacts.py create mode 100644 ocado_grocy/order_state.py create mode 100644 ocado_grocy/pantry.py create mode 100644 ocado_grocy/postprocess.py create mode 100644 ocado_grocy/product_metadata.py create mode 100644 ocado_grocy/receipt.py create mode 100644 pyproject.toml create mode 100644 tests/fixtures/history_receipt.json create mode 100644 tests/test_enrichment_retry.py create mode 100644 tests/test_grocy.py create mode 100644 tests/test_history.py create mode 100644 tests/test_images.py create mode 100644 tests/test_openfoodfacts.py create mode 100644 tests/test_order_state.py create mode 100644 tests/test_postprocess.py create mode 100644 tests/test_receipt.py diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..e4136bc --- /dev/null +++ b/.gitignore @@ -0,0 +1,24 @@ +# Python environments and generated files +.venv/ +__pycache__/ +*.py[cod] +.pytest_cache/ +*.egg-info/ +build/ +dist/ + +# Credentials and browser sessions +config.toml +.env +.env.* +!.env.example +auth.json +.ocado-browser/ + +# Local receipts, reports, and import journal +history/ +*.sqlite3* +*.html +!tests/fixtures/*.html + +PLAN.md diff --git a/README.md b/README.md new file mode 100644 index 0000000..409b722 --- /dev/null +++ b/README.md @@ -0,0 +1,178 @@ +# Ocado → Grocy + +Import new delivered Ocado orders using a visible Python Playwright browser. Completed orders are skipped by default. Login cookies persist between runs. After importing, the command adds missing Ocado pictures and runs conservative Open Food Facts enrichment for products in the processed orders. All products are included by default, including fresh, chilled and frozen items. Use `--shelf-stable-only` to restrict an import. + +## Install + +Python 3.11+ is required. + +```sh +python3 -m venv .venv +.venv/bin/pip install -e '.[test]' +.venv/bin/python -m playwright install chromium +``` + +Copy `config.example.toml` to `config.toml` if you do not already have a config, then set the Grocy URL and API key. `GROCY_URL` and `GROCY_API_KEY` override the config. + +## Run + +```sh +# Preview new delivered orders, write no stock +.venv/bin/ocado-grocy --dry-run + +# Import all products from orders not previously completed +.venv/bin/ocado-grocy + +# Import one complete order, including perishables +.venv/bin/ocado-grocy --order-id 1000000000001 --all-products + +# Sign in and save the browser session without importing +.venv/bin/ocado-grocy --login-only + +# Limit history by date or number of orders +.venv/bin/ocado-grocy --since 2026-01-01 --limit 10 --dry-run + +# Reuse downloaded receipt JSON without opening a browser +.venv/bin/ocado-grocy --cached + +# Fetch current details again, including updates to already imported products +.venv/bin/ocado-grocy --order-id 1000000000001 --refresh +``` + +Images and Open Food Facts enrichment are enabled by default. Set `[openfoodfacts].contact` in `config.toml` for Open Food Facts requests. Use `--no-images` or `--no-openfoodfacts` to disable either step. `--dry-run` performs neither enrichment step and writes nothing to Grocy. `--cached` only skips Ocado browsing; enrichment can still use the network. + +When an order is explicitly selected with `--order-id`, enrichment also runs for its already-imported lines, so repeating the command retries missing pictures or Open Food Facts work without adding stock again. It only processes live products linked to completed journal entries for those orders. Image and Open Food Facts failures are recorded separately; one does not prevent the other from running. An enrichment failure exits with an error after preserving the completed stock import. The next normal run retries only failed products before checking for new orders, even when all orders are already loaded. Existing error reports are recovered into the retry queue once. `--no-images` and `--no-openfoodfacts` defer retries for the disabled stage; `--dry-run` and `--login-only` do not retry. Retry results are saved in `history/enrichment/retries/` and `history/enrichment/retry-summary.json`. Successful retries leave the queue, while unresolved failures remain queued. Reports are in `history/enrichment/ORDER_ID/` (a stable batch identifier is used for multiple orders). + +```sh +# Import and enrich the latest complete delivered order +.venv/bin/ocado-grocy --limit 1 --all-products + +# Retry enrichment for a cached order; imported stock is skipped +.venv/bin/ocado-grocy --cached --order-id 1000000000002 --all-products + +# Run only Open Food Facts for one imported order +.venv/bin/ocado-grocy-off --order-id 1000000000002 --apply +``` + +`python -m ocado_grocy` and `ocado-grocy-history` run the same command. `--order-id` may be repeated. `--help` lists options. There is no standalone HTML-file importer: acquisition always uses Playwright, or normalized JSON cached from a previous browser run. + +## Authentication and downloads + +Playwright launches Chromium with `headless=False`. Complete sign-in and any CAPTCHA in that window. The private `.ocado-browser/` profile saves cookies and local storage for reuse. `.ocado-browser/auth.json` is an additional cookie/storage backup. The application does not store your password. Both the profile and the receipt archive are excluded from Git. + +`--profile`, `--archive`, `--config`, and `--login-timeout` customize locations and login waiting time. Profile/archive paths are relative to the working directory. Do not run two browser processes with the same profile. + +The command checks the delivered-order list through Ocado's explicit end-of-list marker, but filters out completed orders before loading their receipts. `--limit` counts new orders after this filtering. `--order-id` explicitly selects an old order for retry, refreshing, or importing previously excluded items; `--refresh` alone does not reselect completed orders. It reads each order's authenticated `/api/order/v6/orders/ID/decorated` JSON response as the browser loads it. It does not parse prices or quantities from rendered HTML or use the live trolley. Accepted substitutions are included; unavailable originals and rejected replacements are excluded. Discontinued products work without a product-page link. + +Delivered item quantities and paid line prices must reconcile with the API order summary before importing. Delivery, bag charges and order-level credits are not stock items. Exact expiry dates and delivery dates come from the API, so historical orders do not depend on relative UI labels such as “Expired” or “Tuesday”. + +Normalized receipts are saved to `history/ORDER_ID.json`, excluding account/payment details. Successful downloads are reused unless `--refresh` is specified. `history/orders.json` keeps the full manifest even when you use `--limit` or `--order-id`. Failed orders are listed in `history/download-errors.json` and retried on the next browser run. + +## Shelf-stable selection + +All products are imported by default. With `--shelf-stable-only`, no maintained list of pantry products or product-name heuristics is needed. Selection uses Ocado's `storageType` and `categoryPath`: + +- Refrigerated/frozen products and fresh food/bakery categories are excluded. +- Shelf-stable food/drink and household/personal-care categories are included. +- Missing or ambiguous categories are left for review. `CUPBOARD` alone is insufficient: Ocado also uses it for bananas, onions and potatoes. + +Every decision is recorded in `history/report.json`. Catalogue classifications are imperfect, so the filter is conservative. If needed, optional config overrides can resolve a particular product; they are not required to run the importer: + +```toml +[pantry.overrides] +"123456011" = true +"987654011" = false +``` + +`--all-products` is the default and bypasses the optional shelf-stable filter for a complete order. Existing imported stock is not removed when a classification changes. + +## Grocy dates, units and product details + +The order's delivery date is stored as Grocy's **purchased date**. Grocy's technical record creation timestamp remains the actual import time. Expiry dates are retained even when in the past; missing dates use `2999-12-31` with an “Expiry unknown” note. The backfill does not infer which items you have consumed. Remove used-up stock through Grocy and reruns will not restore it. + +New products use one **Pack** per purchased retail unit: two 1.25 L drinks add two packs; one five-banana pack adds one pack. Descriptions retain brand, categories, storage, pack information, promotions, and source/image links where available. Ocado image URLs are retained in metadata. Use the image command below to copy them into Grocy product pictures. Notes stay short: `Ocado YYYY-MM-DD` + +New products use the configurable `Ocado imports` location. Existing products retain their locations and units. Matching uses explicit mappings, imported Ocado product IDs, exact barcodes when available, then names ignoring case/repeated whitespace. It does not guess that different names represent the same product. To map an Ocado product to existing Grocy stock units: + +```toml +[products."72496011"] +product_id = 123 +stock_per_purchase = 3 +``` + +Without an override, existing Grocy purchase-to-stock conversions are used. Paid prices are divided by stock quantity as required by the [Grocy API](https://github.com/grocy/grocy/blob/master/grocy.openapi.json). Tare-weight and parent-only products are rejected. Configure Grocy's currency as GBP. + +Previously imported products are refreshed by default, including fresh/chilled products omitted from the shelf-stable backfill. Refreshing descriptions does not add stock or change notes. Existing enriched fields are kept when new metadata lacks a replacement. The newest available order is considered first for a product appearing in several orders. `--no-refresh-existing` disables this. + +## Duplicate protection and recovery + +The journal also tracks completion per order and Grocy server in `loaded_orders`. Order completion tracks the stock import separately from enrichment failures. After the stock import and enrichment attempt, the order is complete; failed picture or Open Food Facts work stays in a persistent, per-product retry queue. Interrupted stock imports remain eligible for processing; individual completed stock lines still prevent duplicate bookings. Orders with no eligible stock lines are also tracked. Uncertain Open Food Facts candidates are a successful review result, not a processing failure. Dry runs never mark orders loaded. + +Existing journal data is migrated conservatively: an old order counts as loaded only when its cached receipt reconciles with completed journal entries for all selected shelf-stable lines, with no uncertain writes. Downloaded receipts alone do not count as loaded. This preserves earlier shelf-stable backfills without importing their excluded perishables; use `--order-id ID` when you want those too. A cached check with no new orders exits without Grocy access when there are no pending enrichment retries. + +Keep `imports.sqlite3` and the cached receipts. Journal paths are relative to the config. The same journal supports single-order and history imports. Completed lines are skipped, including stock you subsequently consumed or deleted in Grocy. Changing/deleting the journal removes that protection. + +API rows are ordered deterministically. Previously imported rows are matched by their stored fingerprints and restored to their original journal positions, including imports made by the earlier HTML importer. Actual changes in product identity, quantities, prices or dates require review rather than automatic reimport. + +A `pending` journal entry is committed before each stock write and marked `done` after a successful response. A timeout might occur after Grocy accepts stock, so pending entries are not retried automatically. Check Grocy for the purchase-date note and match the product to its journal row, then reconcile that specific journal row: + +```sql +SELECT * FROM imports WHERE order_id = 'ORDER'; + +-- Only if the stock booking succeeded: +UPDATE imports SET status = 'done' +WHERE server = 'https://grocy.example.com/api' AND order_id = 'ORDER' AND line = 1; + +-- OR, only if the stock booking definitely did not occur: +DELETE FROM imports +WHERE server = 'https://grocy.example.com/api' AND order_id = 'ORDER' AND line = 1; +``` + +Use one importer at a time. Grocy cannot transact an entire receipt atomically; earlier completed writes remain after a later failure. Reports are saved even when an import fails. + +## Verification + +```sh +.venv/bin/python -m pytest -q +``` + +Tests use sanitized real API data and a fake Grocy API. They cover delivered orders, substitutions, discontinued products, dates, summary reconciliation, category filtering, conversions, metadata refresh, stable line identity and interrupted imports. Tests do not mutate live stock. + + +## Open Food Facts enrichment + +```sh +.venv/bin/pip install -e '.[test]' +.venv/bin/ocado-grocy-off # preview, no Grocy writes +.venv/bin/ocado-grocy-off --apply # apply clear matches +.venv/bin/ocado-grocy-off --offline # repeat a preview from cached data +.venv/bin/ocado-grocy-off --refresh-cache # download a fresh catalog +``` + +Set `[openfoodfacts].contact` in `config.toml` (or pass `--contact`) for the identifying User-Agent required by [Open Food Facts](https://openfoodfacts.github.io/openfoodfacts-server/api/). Requests are spaced at least 6.2 seconds apart. Run one enrichment process at a time. + +The script reads the import journal for this Grocy server and matches live imported products against brand catalogs from the official [Search-a-licious API](https://openfoodfacts.github.io/search-a-licious/users/ref-openapi/). Large searches are split to avoid truncated results. Recognized household and personal-care categories are skipped. Missing data, different pack sizes, name/variant differences and multiple matching barcodes require review. Automatic matching requires the same brand, meaningful name words and metric pack size, including multipack structure, plus a valid GTIN check digit. Previously established Open Food Facts barcode matches are retained on reruns, including manually reviewed matches. Current product details are fetched again and checked before applying a catalog match. + +Writes add a barcode to Grocy's barcode table and an `openfoodfacts` section to the product description. Available nutrition, ingredient/allergen information, labels, packaging, serving size and image/source links are retained with attribution. Nutrition values keep their source units and basis; they are not copied into Grocy's calories-per-stock-unit field. Product titles, stock, purchase dates and stock notes are never edited. Deleted stock is never recreated. Product descriptions and barcodes are snapshotted before applying, and reruns are idempotent. Future Ocado metadata refreshes preserve the enrichment even after Grocy sanitizes the description HTML. + +Private caches, snapshots and reports live in `history/openfoodfacts/`. `report.json` contains the preview and up to ten ranked candidates per product; `applied-report.json` records writes and failures. The catalog is only a source of candidates: Open Food Facts is crowdsourced, and matching by name cannot prove which physical barcode was on a previously purchased pack. Ambiguous matches are left unchanged. Check the current package for authoritative allergen information. + +For a manually verified match, create a JSON mapping of **Grocy product ID** to **barcode string** (retain leading zeroes), then run: + +```sh +.venv/bin/ocado-grocy-off --mapping history/openfoodfacts/reviewed.json +.venv/bin/ocado-grocy-off --mapping history/openfoodfacts/reviewed.json --apply +``` + +Mappings explicitly override name/size matching, so verify the variant and retail pack first. Existing barcode conflicts are reported instead of reassigned. Errors can leave an earlier barcode addition completed; rerunning reconciles it without duplicating the barcode. Open Food Facts data carries ODbL/database-content attribution, and image links carry CC BY-SA attribution. + +## Product pictures + +Copy the saved Ocado image URLs into native Grocy product pictures: + +```sh +.venv/bin/ocado-grocy-images # preview +.venv/bin/ocado-grocy-images --apply # upload missing pictures +``` + +This covers all existing journal-linked imported products, including non-food products and products whose stock has been used up. It keeps existing pictures and skips deleted products. The main import command runs this step automatically; the standalone command can also be rerun at any time, optionally with `--order-id ORDER_ID`. Downloads use a separate session so Grocy credentials never go to Ocado. Each uploaded file is read back and verified before assigning it to the product, using the [Grocy files API](https://github.com/grocy/grocy/blob/master/grocy.openapi.json). Only the picture field is changed; titles, descriptions and stock remain untouched. Results are saved in `history/images/report.json`. Failed updates can be retried safely. diff --git a/config.example.toml b/config.example.toml new file mode 100644 index 0000000..b3c1fe4 --- /dev/null +++ b/config.example.toml @@ -0,0 +1,20 @@ +[grocy] +url = "https://grocy.example.com/" +api_key = "replace-with-your-key" + +[import] +# Used only for newly created products. Existing products keep their location. +location = "Ocado imports" +quantity_unit = "Pack" +# Relative to this config file. Keep this journal to prevent duplicate imports. +state_file = "imports.sqlite3" + +# Optional: map an Ocado retailer product ID (or exact receipt name if no ID) +# to an existing Grocy product. Stock units per purchased pack can be overridden. +# [products."72496011"] +# product_id = 123 +# stock_per_purchase = 3 + +[openfoodfacts] +# Identifies the application to Open Food Facts. +contact = "you@example.com" diff --git a/ocado_grocy/__init__.py b/ocado_grocy/__init__.py new file mode 100644 index 0000000..b66d246 --- /dev/null +++ b/ocado_grocy/__init__.py @@ -0,0 +1 @@ +"""Ocado receipt importer.""" diff --git a/ocado_grocy/__main__.py b/ocado_grocy/__main__.py new file mode 100644 index 0000000..4e28416 --- /dev/null +++ b/ocado_grocy/__main__.py @@ -0,0 +1,3 @@ +from .cli import main + +main() diff --git a/ocado_grocy/cli.py b/ocado_grocy/cli.py new file mode 100644 index 0000000..9311d20 --- /dev/null +++ b/ocado_grocy/cli.py @@ -0,0 +1,4 @@ +"""Main command: acquire orders through Playwright.""" +from .history import main + +__all__ = ["main"] diff --git a/ocado_grocy/enrichment_retry.py b/ocado_grocy/enrichment_retry.py new file mode 100644 index 0000000..cb34c01 --- /dev/null +++ b/ocado_grocy/enrichment_retry.py @@ -0,0 +1,53 @@ +"""Durable product-level retries, independent of completed stock orders.""" +import json +import sqlite3 + + +class RetryQueue: + def __init__(self, state, server): + self.server = server + self.db = sqlite3.connect(state) + with self.db: + self.db.execute('''CREATE TABLE IF NOT EXISTS enrichment_retries ( + server TEXT, stage TEXT, product_id INTEGER, error TEXT, + PRIMARY KEY(server,stage,product_id))''') + self.db.execute('CREATE TABLE IF NOT EXISTS enrichment_retry_migrations (server TEXT PRIMARY KEY)') + + def start(self, stage, products): + self.results(stage, [{'product_id':p['id'], 'status':'error', 'error':'Interrupted enrichment'} for p in products]) + + def results(self, stage, rows): + with self.db: + for row in rows: + key = (self.server,stage,int(row['product_id'])) + if row['status'] == 'error': + self.db.execute('INSERT OR REPLACE INTO enrichment_retries VALUES (?,?,?,?)', (*key,row.get('error','Enrichment failed'))) + else: + self.db.execute('DELETE FROM enrichment_retries WHERE server=? AND stage=? AND product_id=?',key) + + def pending(self, stage): + return {r[0] for r in self.db.execute('SELECT product_id FROM enrichment_retries WHERE server=? AND stage=?',(self.server,stage))} + + def migrate_reports(self, archive, allowed_ids): + """Import old errors once; newer successful reports supersede old failures.""" + if self.db.execute('SELECT 1 FROM enrichment_retry_migrations WHERE server=?',(self.server,)).fetchone(): + return + events = [] + for path in (archive/'enrichment').glob('*/summary.json'): + summary = json.loads(path.read_text()) + if summary.get('server',self.server) != self.server: + continue + for stage in ('images','openfoodfacts'): + report_path = path.parent / f'{stage}.json' + if report_path.exists(): + data = json.loads(report_path.read_text()) + events.append((report_path.stat().st_mtime,stage,data.get('products',[]))) + elif any(error.lower().startswith('images:' if stage=='images' else 'open food facts:') for error in summary.get('errors',[])): + events.append((path.stat().st_mtime,stage,[{'product_id':pid,'status':'error','error':'Previous enrichment failed'} for pid in summary.get('product_ids',[])])) + for _,stage,rows in sorted(events,key=lambda event:event[0]): + self.results(stage,[r for r in rows if int(r['product_id']) in allowed_ids]) + with self.db: + self.db.execute('INSERT INTO enrichment_retry_migrations VALUES (?)',(self.server,)) + + def close(self): + self.db.close() diff --git a/ocado_grocy/grocy.py b/ocado_grocy/grocy.py new file mode 100644 index 0000000..7164ad9 --- /dev/null +++ b/ocado_grocy/grocy.py @@ -0,0 +1,304 @@ +"""Small Grocy API client and resumable importer.""" +from decimal import Decimal +from dataclasses import replace +import hashlib +import html +import json +import re +from pathlib import Path +import sqlite3 + +import requests +from .product_metadata import read_metadata + +from .receipt import ImportError, number, receipt_date + + +def normalized(name): + return " ".join(name.casefold().split()) + + +class Grocy: + def __init__(self, url, api_key, timeout=30): + self.url = url.rstrip("/") + if not self.url.endswith("/api"): + self.url += "/api" + self.session = requests.Session() + self.session.headers.update({"GROCY-API-KEY": api_key, + "User-Agent": "ocado-grocy/0.1"}) + self.timeout = timeout + + def request(self, method, path, data=None): + try: + response = self.session.request(method, self.url + path, json=data, + timeout=self.timeout, allow_redirects=False) + except requests.RequestException as exc: + raise ImportError(f"Grocy {method} {path} failed ({type(exc).__name__}); " + "check connectivity before retrying") from exc + if not 200 <= response.status_code < 300: + raise ImportError(f"Grocy {method} {path}: HTTP {response.status_code}") + try: + return response.json() if response.content else None + except ValueError as exc: + raise ImportError(f"Grocy {method} {path} returned non-JSON data") from exc + + def get(self, path): + return self.request("GET", path) + + def post(self, path, data): + return self.request("POST", path, data) + + def named_object(self, entity, name, **fields): + matches = [x for x in self.get(f"/objects/{entity}") + if normalized(x["name"]) == normalized(name)] + if len(matches) > 1: + raise ImportError(f"Multiple Grocy {entity} named {name!r}") + if matches: + return int(matches[0]["id"]) + return int(self.post(f"/objects/{entity}", {"name": name, **fields})["created_object_id"]) + + +class Journal: + """Commit intent before each write; ambiguous writes require explicit reconciliation.""" + def __init__(self, path: Path): + path.parent.mkdir(parents=True, exist_ok=True) + self.db = sqlite3.connect(path) + self.db.execute("""CREATE TABLE IF NOT EXISTS imports ( + server TEXT, order_id TEXT, line INTEGER, fingerprint TEXT, + status TEXT, product_id INTEGER, PRIMARY KEY(server, order_id, line))""") + self.db.commit() + + def check(self, server, order, line, fingerprint): + row = self.db.execute("SELECT fingerprint, status FROM imports WHERE server=? AND order_id=? AND line=?", + (server, order, line)).fetchone() + if not row: + return False + if row[0] != fingerprint: + raise ImportError(f"Order {order} line {line} changed since a previous import; reconcile it manually") + if row[1] != "done": + raise ImportError(f"Order {order} line {line} has an uncertain previous write. " + "Check Grocy stock/logs and reconcile the journal before retrying (see README).") + return True + + def record(self, server, order, line, fingerprint, status, product_id): + with self.db: + self.db.execute("INSERT OR REPLACE INTO imports VALUES (?, ?, ?, ?, ?, ?)", + (server, order, line, fingerprint, status, product_id)) + + +def marker(item): + return f"Ocado product ID: {item.product_id}" if item.product_id else "" + + +def description(item): + details = {"Ocado product ID": item.product_id, "Barcode": item.barcode, + "Product URL": item.url, **item.metadata} + return "

" + html.escape(marker(item) or "Imported from Ocado") + "

" + html.escape(
+        json.dumps(details, ensure_ascii=False, indent=2, default=str)) + "
" + + +def match_product(item, products, barcodes, mappings): + override = mappings.get(item.product_id or item.name) + if override is not None: + matches = [p for p in products if int(p["id"]) == int(override["product_id"])] + if not matches: + raise ImportError(f"Mapped Grocy product does not exist: {item.name}") + return matches[0], override + matches = [] + if item.product_id: + matches = [p for p in products if f"

{html.escape(marker(item))}

" in (p.get("description") or "")] + if not matches and item.barcode: + ids = {int(b["product_id"]) for b in barcodes if b["barcode"] == item.barcode} + matches = [p for p in products if int(p["id"]) in ids] + if not matches: + matches = [p for p in products if normalized(p["name"]) == normalized(item.name)] + if len(matches) > 1: + raise ImportError(f"Ambiguous existing product: {item.name}; add a config product mapping") + return (matches[0] if matches else None), {} + + +def item_fingerprint(receipt, item): + return hashlib.sha256(json.dumps({ + "name": item.name, "product_id": item.product_id, "quantity": str(item.quantity.normalize()), + "total": str(item.total.normalize()), "best_before": item.best_before, + "date": receipt.purchased_date}, sort_keys=True).encode()).hexdigest() + + +def align_recorded_lines(receipt, server, journal): + """Restore journal positions when Ocado reorders equal-expiry receipt rows.""" + records = journal.db.execute( + "SELECT line, fingerprint FROM imports WHERE server=? AND order_id=? ORDER BY line", + (server, receipt.order_id)).fetchall() + remaining = list(receipt.items) + aligned = [None] * len(remaining) + for line, fingerprint in records: + candidates = [i for i,item in enumerate(remaining) if item_fingerprint(receipt,item) == fingerprint] + if not candidates or not 1 <= line <= len(aligned): + raise ImportError(f"Order {receipt.order_id} line {line} changed since its import; review before continuing") + aligned[line - 1] = remaining.pop(candidates[0]) + rest = iter(remaining) + return replace(receipt, items=[item if item is not None else next(rest) for item in aligned]) + + +def refresh_imported_products(receipt, client, journal, *, seen=None): + """Refresh details of journal-linked products, including excluded perishables.""" + seen = seen if seen is not None else set() + receipt = align_recorded_lines(receipt, client.url, journal) + records = journal.db.execute( + "SELECT line, product_id FROM imports WHERE server=? AND order_id=? AND status='done' ORDER BY line", + (client.url, receipt.order_id)).fetchall() + count = 0 + for line, product_id in records: + if product_id in seen: + continue + if not 1 <= line <= len(receipt.items): + raise ImportError(f"Receipt line {line} no longer exists; cannot refresh order {receipt.order_id}") + item = receipt.items[line - 1] + product = client.get(f"/objects/products/{product_id}") + old = product.get("description") or "" + expected = f"

{html.escape(marker(item))}

" if item.product_id else "" + if expected and expected not in old: + raise ImportError(f"Product identity changed for order {receipt.order_id}, line {line}; review before refreshing") + if not expected and normalized(product["name"]) != normalized(item.name): + raise ImportError(f"Product name changed for order {receipt.order_id}, line {line}") + metadata = read_metadata(old) + for key, value in item.metadata.items(): + if value not in (None, "", {}, []): + if isinstance(value, dict) and isinstance(metadata.get(key), dict): + metadata[key] = {**metadata[key], **value} + else: + metadata[key] = value + barcode = item.barcode or metadata.pop("Barcode", "") + url = item.url or metadata.pop("Product URL", "") + for key in ("Ocado product ID", "Barcode", "Product URL"): + metadata.pop(key, None) + updated = description(replace(item, metadata=metadata, barcode=barcode, url=url)) + if old != updated: + client.request("PUT", f"/objects/products/{product_id}", {"description": updated}) + count += 1 + seen.add(product_id) + return count + + +def import_receipt(receipt, client, journal, config, echo=print, *, include_lines=None): + if not receipt.purchased_date: + raise ImportError("No purchase/delivery date found; supply --purchased-date YYYY-MM-DD") + products = client.get("/objects/products") + barcodes = client.get("/objects/product_barcodes") + plans = [] + # Validate every existing product/conversion before any mutations. + for line, item in enumerate(receipt.items, 1): + if include_lines is not None and line not in include_lines: + continue + fingerprint = item_fingerprint(receipt, item) + if journal.check(client.url, receipt.order_id, line, fingerprint): + echo(f"Skipped already imported: {item.name}") + continue + product, override = match_product(item, products, barcodes, config.get("products", {})) + factor = number(override.get("stock_per_purchase", 1), "stock conversion") + if product: + if product.get("enable_tare_weight_handling") or product.get("no_own_stock"): + raise ImportError(f"Unsupported tare-weight or parent-only product: {item.name}") + if "stock_per_purchase" not in override and product["qu_id_purchase"] != product["qu_id_stock"]: + details = client.get(f'/stock/products/{product["id"]}') + factor = number(details.get("qu_conversion_factor_purchase_to_stock"), "stock conversion") + if factor <= 0: + raise ImportError(f"Stock conversion must be positive: {item.name}") + plans.append((line, item, fingerprint, product, factor)) + if not plans: + return 0 + location = client.named_object("locations", config.get("location", "Ocado imports")) + unit = client.named_object("quantity_units", config.get("quantity_unit", "Pack"), name_plural="Packs") + store = client.named_object("shopping_locations", "Ocado") + count = 0 + for line, item, fingerprint, product, factor in plans: + if product is None: + # Recheck the local list: the same product can occur on multiple receipt lines. + product, _ = match_product(item, products, barcodes, config.get("products", {})) + if product is None: + payload = {"name": item.name, "description": description(item), "location_id": location, + "shopping_location_id": store, "qu_id_purchase": unit, "qu_id_stock": unit, + "qu_id_consume": unit, "qu_id_price": unit} + product_id = int(client.post("/objects/products", payload)["created_object_id"]) + product = {"id": product_id, **payload} + products.append(product) + if item.barcode: + client.post("/objects/product_barcodes", {"product_id": product_id, "barcode": item.barcode, + "qu_id": unit, "amount": 1}) + product_id = int(product["id"]) + amount = item.quantity * factor + note = f"Ocado {receipt.purchased_date}" + if not item.best_before: + note += " Expiry unknown." + payload = {"amount": float(amount), "price": float(item.total / amount), + "best_before_date": item.best_before or "2999-12-31", + "purchased_date": receipt.purchased_date, "transaction_type": "purchase", + "shopping_location_id": store, "stock_label_type": 1, "note": note} + journal.record(client.url, receipt.order_id, line, fingerprint, "pending", product_id) + client.post(f"/stock/products/{product_id}/add", payload) + journal.record(client.url, receipt.order_id, line, fingerprint, "done", product_id) + echo(f"Imported {item.quantity} × {item.name}: £{item.total:.2f}") + count += 1 + return count + + +def update_existing_notes(client, journal): + """Edit only surviving journal-linked stock entries; never book new stock.""" + imported = {(str(order), int(line), int(product)) for order, line, product in journal.db.execute( + "SELECT order_id, line, product_id FROM imports WHERE server=? AND status='done'", (client.url,))} + changed = {} + preserved_fields = ("amount", "best_before_date", "price", "open", "location_id", + "shopping_location_id", "purchased_date") + for row in client.get("/objects/stock"): + old = row.get("note") or "" + match = re.match(r"Ocado order (\d+), line (\d+)[.;]", old) + dated = re.fullmatch(r"Ocado (\d{4}-\d{2}-\d{2}), line (\d+)\.(?: Expiry unknown\.)?", old) + if match: + linked = (match[1], int(match[2]), int(row["product_id"])) in imported + elif dated: + linked = any(line == int(dated[2]) and product == int(row["product_id"]) + for _, line, product in imported) + else: + linked = False + if not linked: + continue + try: + current = client.get(f"/stock/entry/{row['id']}") + except ImportError: + if not any(entry["id"] == row["id"] for entry in client.get("/objects/stock")): + continue + raise + if current is None or current.get("note") != old: + continue + row = current + purchased = receipt_date(row["purchased_date"]) + if not purchased: + raise ImportError(f"Stock entry {row['id']} has no purchase date") + note = f"Ocado {purchased}" + if "Expiry unknown" in old: + note += " Expiry unknown." + payload = {field: row[field] for field in preserved_fields} + try: + client.request("PUT", f"/stock/entry/{row['id']}", {**payload, "note": note}) + except ImportError: + if not any(entry["id"] == row["id"] for entry in client.get("/objects/stock")): + continue + raise + changed[row["id"]] = {**payload, "note": note} + for row in client.get("/objects/stock"): + if row["id"] in changed: + if row["note"] != changed[row["id"]]["note"]: + raise ImportError(f"Verification failed for stock entry {row['id']}") + return len(changed) + + +def imported_products(client, state, order_ids=()): + """Live products from completed journal lines, optionally scoped to orders.""" + query = "SELECT DISTINCT product_id FROM imports WHERE server=? AND status='done'" + params = [client.url] + if order_ids: + query += ' AND order_id IN (' + ','.join('?' for _ in order_ids) + ')' + params.extend(order_ids) + with sqlite3.connect(f'{Path(state).resolve().as_uri()}?mode=ro', uri=True) as db: + ids = {int(row[0]) for row in db.execute(query, params)} + return [p for p in client.get('/objects/products') if int(p['id']) in ids] diff --git a/ocado_grocy/history.py b/ocado_grocy/history.py new file mode 100644 index 0000000..45a267e --- /dev/null +++ b/ocado_grocy/history.py @@ -0,0 +1,326 @@ +"""Browse Ocado history with a persistent, visible Playwright browser.""" +from dataclasses import asdict, dataclass +from datetime import datetime +import json +import os +from pathlib import Path +import re +import sqlite3 +import time +import tomllib +from urllib.parse import urlsplit + +import click +from lxml import html + +from .grocy import Grocy, Journal, import_receipt, refresh_imported_products, align_recorded_lines +from .pantry import classify +from .receipt import ImportError, parse_document, parse_ocado_order, receipt_date + + +@dataclass(frozen=True) +class OrderLink: + order_id: str + purchased_date: str + url: str + + +def private_json(path, data): + path.parent.mkdir(parents=True, exist_ok=True, mode=0o700) + temporary = path.with_suffix(path.suffix + ".tmp") + fd = os.open(temporary, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600) + with os.fdopen(fd, "w") as stream: + json.dump(data, stream, ensure_ascii=False, indent=2, default=str) + temporary.replace(path) + + +def order_links(content): + tree = html.fromstring(content) + found = {} + for anchor in tree.xpath('//a[contains(@href,"/orders/")]'): + href = anchor.get("href") + match = re.fullmatch(r"/orders/(\d+)/details", href) + text = " ".join(" ".join(anchor.itertext()).split()) + if not match or not re.search(r"\bDelivered\b", text): + continue + stamp = re.search(r"([A-Z][a-z]{2} \d{1,2}, \d{4})", text) + if not stamp: + raise ImportError(f"No full delivery date on order {match[1]}") + purchased = datetime.strptime(stamp[1], "%b %d, %Y").date().isoformat() + found[match[1]] = OrderLink(match[1], purchased, "https://www.ocado.com" + href) + return list(found.values()) + + +def wait_for_orders(page, timeout, echo): + deadline = time.monotonic() + timeout + notice = 0 + while time.monotonic() < deadline: + if page.is_closed(): + raise ImportError("Browser closed before sign-in completed") + if urlsplit(page.url).hostname == "www.ocado.com" and page.locator('a[href*="/orders/"]').count(): + return + if time.monotonic() >= notice: + echo("Waiting for order history. Complete sign-in or any CAPTCHA in the browser.") + notice = time.monotonic() + 30 + page.wait_for_timeout(1000) + raise ImportError("Timed out waiting for order history; rerun to reuse the saved browser session") + + +def discover_orders(page, echo): + last_count = -1 + stable = 0 + for _ in range(500): + count = page.locator('a[href*="/orders/"]').count() + if count == last_count: + stable += 1 + else: + stable = 0 + echo(f"Loaded {count} order links...") + if page.locator('[data-test="order-list-no-more-orders-label"]').count(): + return order_links(page.content()) + if stable >= 10: + raise ImportError("Order history stopped loading before Ocado's end-of-list marker; retry after checking the browser") + last_count = count + bottom = page.get_by_role("button", name="Back to top", exact=True) + if bottom.count(): + bottom.scroll_into_view_if_needed() + else: + page.locator('a[href*="/orders/"]').last.scroll_into_view_if_needed() + page.mouse.wheel(0,1200) + page.wait_for_timeout(1500) + raise ImportError("Order list did not reach its end; refusing to call the download complete") + + +def download_orders(profile, archive, *, since=None, limit=None, login_only=False, + login_timeout=300, echo=print, order_ids=(), refresh=False, completed=()): + try: + from playwright.sync_api import sync_playwright, Error as BrowserError + except ModuleNotFoundError as exc: + raise ImportError("Install dependencies: pip install -e . and python -m playwright install chromium") from exc + profile.mkdir(parents=True, exist_ok=True, mode=0o700) + profile.chmod(0o700) + with sync_playwright() as playwright: + context = playwright.chromium.launch_persistent_context( + str(profile.resolve()), headless=False, viewport={"width":1280, "height":900}) + try: + page = context.pages[0] if context.pages else context.new_page() + page.goto("https://www.ocado.com/orders", wait_until="domcontentloaded") + wait_for_orders(page, login_timeout, echo) + if login_only: + echo("Signed-in browser session saved.") + return [], [] + links = discover_orders(page, echo) + private_json(archive / "orders.json", [asdict(link) for link in links]) + links = select_orders(links, since, limit, order_ids, completed) + receipts, failures = [], [] + for index, link in enumerate(links,1): + path = archive / f"{link.order_id}.json" + if path.exists() and not refresh: + receipt = parse_document(json.loads(path.read_text())) + if receipt.order_id != link.order_id or receipt.purchased_date != link.purchased_date: + raise ImportError(f"Cached receipt does not match order {link.order_id}") + receipts.append(receipt) + echo(f"[{index}/{len(links)}] Cached {link.order_id} ({link.purchased_date})") + continue + try: + endpoint = f"/api/order/v6/orders/{link.order_id}/decorated" + with page.expect_response(lambda response: urlsplit(response.url).path == endpoint, + timeout=45000) as captured: + page.goto(link.url, wait_until="domcontentloaded") + response = captured.value + if response.status != 200: + raise ImportError(f"Ocado receipt returned HTTP {response.status}") + data = response.json() + page.wait_for_selector('[data-test="order-online-receipt"]', timeout=30000) + hrefs = page.locator('[data-test="receipt-element-product-link"][href]').evaluate_all( + '(elements) => elements.map(element => element.getAttribute("href"))') + urls = {} + for href in hrefs: + match = re.search(r"/(\d+)$", href) + if match: + urls[match[1]] = "https://www.ocado.com" + href + receipt = parse_ocado_order(data, expected_order_id=link.order_id, product_urls=urls) + if receipt.purchased_date != link.purchased_date: + raise ImportError("Order-list date does not match the receipt API date") + private_json(path, receipt.export()) + receipts.append(receipt) + echo(f"[{index}/{len(links)}] Saved {link.order_id}: {len(receipt.items)} lines") + except (ImportError, BrowserError, ValueError, KeyError) as exc: + message = str(exc) if isinstance(exc, ImportError) else type(exc).__name__ + failures.append({"order_id":link.order_id,"error":message}) + echo(f"[{index}/{len(links)}] Could not read {link.order_id}: {message}") + page.wait_for_timeout(750) + private_json(archive / "download-errors.json", failures) + return receipts, failures + finally: + try: + # Persistent profile is authoritative; this also preserves a portable cookie backup. + context.storage_state(path=str(profile / "auth.json")) + (profile / "auth.json").chmod(0o600) + finally: + context.close() + + +def select_orders(links, since=None, limit=None, order_ids=(), completed=()): + missing = set(order_ids) - {link.order_id for link in links} + if missing: + raise ImportError("Orders not found in delivered history: " + ", ".join(sorted(missing))) + links = sorted((link for link in links if not since or link.purchased_date >= since), + key=lambda link:(link.purchased_date,link.order_id), reverse=True) + if order_ids: + links = [link for link in links if link.order_id in order_ids] + if not order_ids: + links = [link for link in links if link.order_id not in completed] + if limit: + links = links[:limit] + return links + + +def cached_orders(archive, since=None, limit=None, order_ids=(), completed=()): + links = [OrderLink(**row) for row in json.loads((archive / "orders.json").read_text())] + links = select_orders(links, since, limit, order_ids, completed) + receipts, failures = [], [] + for link in links: + path = archive / f"{link.order_id}.json" + if path.exists(): + receipt = parse_document(json.loads(path.read_text())) + if receipt.order_id != link.order_id or receipt.purchased_date != link.purchased_date: + raise ImportError(f"Cached receipt does not match order {link.order_id}") + receipts.append(receipt) + else: + failures.append({"order_id":link.order_id,"error":"Receipt has not been downloaded"}) + return receipts, failures + + +def run_import(receipts, config, config_path, archive, dry_run, failures, echo=print, refresh_existing=True, + all_products=True): + report = {"orders":len(receipts), "download_errors":failures, "items":[], "imported_lines":0, "refreshed_products":0} + settings = config.get("import", {}) + journal = None + refreshed = set() + try: + if not dry_run: + connection = config.get("grocy", {}) + url = os.environ.get("GROCY_URL", connection.get("url", "")) + key = os.environ.get("GROCY_API_KEY", connection.get("api_key", "")) + if not url or not key: + raise ImportError("Configure Grocy URL and API key before importing") + api = Grocy(url,key) + state = Path(settings.get("state_file", "imports.sqlite3")) + if not state.is_absolute(): + state = config_path.resolve().parent / state + journal = Journal(state) + for receipt in receipts: + if not dry_run: + receipt = align_recorded_lines(receipt, api.url, journal) + if not dry_run and refresh_existing: + count = refresh_imported_products(receipt, api, journal, seen=refreshed) + report["refreshed_products"] += count + if count: + echo(f"Refreshed details for {count} previously imported products from {receipt.order_id}") + selected = set() + for line,item in enumerate(receipt.items,1): + decision = classify(item,config.get("pantry",{}).get("overrides",{})) + report["items"].append({"order_id":receipt.order_id, "date":receipt.purchased_date, + "line":line,"name":item.name,"product_id":item.product_id, + "quantity":str(item.quantity),"total":str(item.total), + **asdict(decision)}) + if decision.include or all_products: + selected.add(line) + report["items"][-1]["include"] = True + if all_products: + report["items"][-1]["reason"] = "all products requested" + report["items"][-1]["uncertain"] = False + echo(f"Order {receipt.order_id} ({receipt.purchased_date}): {len(selected)}/{len(receipt.items)} selected lines") + if not dry_run: + report["imported_lines"] += import_receipt( + receipt,api,journal,{**settings,"products":config.get("products",{})}, + echo=echo, include_lines=selected) + finally: + private_json(archive / "report.json",report) + if journal: + journal.db.close() + return report + + +@click.command(context_settings={"help_option_names":["-h","--help"]}) +@click.option("--config", "config_path", type=click.Path(path_type=Path,dir_okay=False),default="config.toml",show_default=True) +@click.option("--profile", type=click.Path(path_type=Path,file_okay=False),default=".ocado-browser",show_default=True) +@click.option("--archive", type=click.Path(path_type=Path,file_okay=False),default="history",show_default=True) +@click.option("--since", help="Only orders delivered on or after YYYY-MM-DD.") +@click.option("--limit", type=click.IntRange(min=1),help="Process at most this many new delivered orders (or explicitly selected orders).") +@click.option("--dry-run", is_flag=True,help="Download and report selections without writing to Grocy.") +@click.option("--cached", is_flag=True,help="Use downloaded receipts without opening a browser.") +@click.option("--login-only", is_flag=True,help="Sign in and save cookies without downloading or importing orders.") +@click.option("--login-timeout",type=click.IntRange(min=1),default=300,show_default=True) +@click.option("--order-id", "order_ids", multiple=True, help="Process a specific delivered order; repeat for several.") +@click.option("--all-products/--shelf-stable-only", default=True, show_default=True, + help="Include all products, or restrict imports to shelf-stable items.") +@click.option("--refresh", is_flag=True, help="Download receipts again even when they are cached.") +@click.option("--refresh-existing/--no-refresh-existing", default=True, show_default=True, + help="Update details of previously imported products, including perishables.") +@click.option("--images/--no-images", default=True, show_default=True, help="Add missing Ocado product pictures after importing.") +@click.option("--openfoodfacts/--no-openfoodfacts", default=True, show_default=True, help="Match and enrich imported products after importing.") +def main(config_path,profile,archive,since,limit,dry_run,cached,login_only,login_timeout,refresh_existing, + order_ids,all_products,refresh,images,openfoodfacts): + """Import delivered Ocado orders with Playwright; include all products by default.""" + try: + if cached and login_only: + raise ImportError("--cached and --login-only cannot be combined") + if cached and refresh: + raise ImportError("--cached and --refresh cannot be combined") + config = tomllib.loads(config_path.read_text()) if config_path.exists() else {} + from .order_state import completed_orders, mark_orders + connection = config.get('grocy', {}) + server = Grocy(os.environ.get('GROCY_URL', connection.get('url', '')), + os.environ.get('GROCY_API_KEY', connection.get('api_key', ''))).url + state = Path(config.get('import', {}).get('state_file', 'imports.sqlite3')) + if not state.is_absolute(): + state = config_path.resolve().parent / state + completed = completed_orders(state, server, archive, config) + if not dry_run and not login_only: + mark_orders(state, server, completed, 'complete') + retry_errors = [] + if not dry_run and not login_only and (images or openfoodfacts): + from .postprocess import retry_enrichment + retry_errors = retry_enrichment(config, config_path, archive, images=images, openfoodfacts=openfoodfacts)['errors'] + since = receipt_date(since) + archive.mkdir(parents=True,exist_ok=True,mode=0o700) + if cached: + receipts,failures = cached_orders(archive,since,limit,order_ids,completed) + else: + receipts,failures = download_orders(profile,archive,since=since,limit=limit, + login_only=login_only,login_timeout=login_timeout,echo=click.echo,order_ids=order_ids,refresh=refresh,completed=completed) + if login_only: + return + if not receipts and not failures: + if retry_errors: + raise ImportError('Enrichment retries still need attention: ' + '; '.join(retry_errors)) + click.echo('No new delivered orders to import.') + return + processing = [receipt.order_id for receipt in receipts] + if not dry_run: + mark_orders(state, server, processing, 'processing') + report = run_import(receipts,config,config_path,archive,dry_run,failures,echo=click.echo, + refresh_existing=refresh_existing,all_products=all_products) + selected = sum(row["include"] for row in report["items"]) + uncertain = sum(row["uncertain"] for row in report["items"]) + click.echo(f"{len(receipts)} orders; {selected} selected lines; {report['imported_lines']} imported; " + f"{uncertain} uncertain lines. Report: {archive / 'report.json'}") + if not dry_run: + from .postprocess import enrich_orders + enrichment = enrich_orders(config, config_path, archive, + [receipt.order_id for receipt in receipts], images=images, openfoodfacts=openfoodfacts) + report['enrichment'] = enrichment + private_json(archive / 'report.json', report) + mark_orders(state, server, processing, 'complete') + if enrichment['errors'] or retry_errors: + raise ImportError('Stock import completed; enrichment needs a retry. ' + '; '.join(enrichment['errors'] + retry_errors)) + if failures: + raise ImportError(f"{len(failures)} orders could not be downloaded; see report and retry") + except (ImportError,OSError,ValueError,KeyError,sqlite3.Error) as exc: + raise click.ClickException(str(exc)) from exc + + +if __name__ == "__main__": + main() diff --git a/ocado_grocy/images.py b/ocado_grocy/images.py new file mode 100644 index 0000000..fe759e0 --- /dev/null +++ b/ocado_grocy/images.py @@ -0,0 +1,113 @@ +"""Copy saved Ocado product photos into Grocy's native product pictures.""" +import base64 +import hashlib +import json +from pathlib import Path +import tomllib +from urllib.parse import quote, urlsplit + +import click +import requests + +from .grocy import Grocy, imported_products +from .history import private_json +from .product_metadata import read_metadata + + +def image_type(data): + if data.startswith(b'\xff\xd8\xff'): + return 'jpg' + if data.startswith(b'\x89PNG\r\n\x1a\n'): + return 'png' + if data.startswith(b'RIFF') and data[8:12] == b'WEBP': + return 'webp' + raise ValueError('Image response is not JPEG, PNG or WebP') + + +def file_path(filename): + return '/files/productpictures/' + quote(base64.b64encode(filename.encode()).decode(), safe='') + + +def add_picture(client, downloader, product): + if product.get('picture_file_name'): + return 'existing_picture' + metadata = read_metadata(product.get('description')) + url = metadata.get('image_url') + if not url: + return 'no_image_url' + parsed = urlsplit(url) + if parsed.scheme != 'https' or parsed.hostname not in {'www.ocado.com', 'ocado.com'}: + raise ValueError('Expected an HTTPS Ocado image URL') + # Separate session: never send the Grocy API key to the image host. + response = downloader.get(url, timeout=30, allow_redirects=False) + response.raise_for_status() + if response.status_code != 200: + raise ValueError('Image URL did not return HTTP 200') + data = response.content + extension = image_type(data) + if len(data) > 10_000_000: + raise ValueError('Image exceeds 10 MB') + filename = f"ocado-{int(product['id'])}-{hashlib.sha256(data).hexdigest()[:16]}.{extension}" + current = client.get(f"/objects/products/{product['id']}") + if not current: + return 'deleted_product' + if current.get('picture_file_name'): + return 'existing_picture' + if read_metadata(current.get('description')).get('Ocado product ID') != metadata.get('Ocado product ID'): + raise ValueError('Product identity changed') + path = file_path(filename) + uploaded = client.session.put(client.url + path, data=data, + headers={'Content-Type':'application/octet-stream'}, + timeout=client.timeout, allow_redirects=False) + uploaded.raise_for_status() + if uploaded.status_code not in (200, 201, 204): + raise ValueError('Unexpected upload response') + # Check the stored file before setting the product's picture reference. + stored = client.session.get(client.url + path, timeout=client.timeout, allow_redirects=False) + stored.raise_for_status() + if stored.status_code != 200 or stored.content != data: + raise ValueError('Uploaded image verification failed') + client.request('PUT', f"/objects/products/{product['id']}", {'picture_file_name':filename}) + return 'updated' + + +@click.command() +@click.option('--config', type=click.Path(path_type=Path), default=Path('config.toml')) +@click.option('--report', type=click.Path(path_type=Path), default=Path('history/images/report.json')) +@click.option('--apply', is_flag=True, help='Upload and assign missing product pictures.') +@click.option('--order-id', 'order_ids', multiple=True, help='Limit to products imported from these orders.') +def main(config, report, apply, order_ids): + """Add Ocado images to journal-linked products; keep existing pictures.""" + settings = tomllib.loads(config.read_text()) + client = Grocy(**settings['grocy']) + state = config.parent / settings.get('import', {}).get('state_file', 'imports.sqlite3') + products = imported_products(client, state, order_ids) + return run_images(client, products, report, apply=apply) + + +def run_images(client, products, report, *, apply=True, raise_errors=True): + downloader = requests.Session() + downloader.headers['User-Agent'] = 'ocado-grocy/0.1' + results = [] + private_json(report, {'apply':apply, 'products':results}) + for product in products: + row = {'product_id':product['id'], 'name':product['name']} + try: + if apply: + row['status'] = add_picture(client, downloader, product) + else: + row['status'] = 'existing_picture' if product.get('picture_file_name') else 'ready' if read_metadata(product.get('description')).get('image_url') else 'no_image_url' + except (requests.RequestException, ValueError) as exc: + row.update(status='error', error=str(exc)) + results.append(row) + private_json(report, {'apply':apply, 'products':results}) + click.echo(f"{product['id']} {row['status']}: {product['name']}") + counts = {s:sum(r['status']==s for r in results) for s in sorted({r['status'] for r in results})} + click.echo(json.dumps(counts)) + if counts.get('error') and raise_errors: + raise click.ClickException(f"{counts['error']} image updates failed; see {report}") + return {'products':results, 'counts':counts} + + +if __name__ == '__main__': + main() diff --git a/ocado_grocy/openfoodfacts.py b/ocado_grocy/openfoodfacts.py new file mode 100644 index 0000000..d50a47f --- /dev/null +++ b/ocado_grocy/openfoodfacts.py @@ -0,0 +1,283 @@ +"""Conservative, resumable Open Food Facts enrichment of imported products.""" +import hashlib +import json +from pathlib import Path +import re +import time +import tomllib +import unicodedata +from datetime import datetime, timezone + +import click +import requests + +from .grocy import Grocy, imported_products +from .history import private_json +from .product_metadata import read_metadata, replace_metadata + +FIELDS = 'code,product_name,product_name_en,brands,quantity,ingredients_text_en,ingredients_text,allergens_tags,traces_tags,nutriments,nutrition_grades,nova_group,labels_tags,packaging,serving_size,image_front_url,countries_tags,last_modified_t' +NONFOOD = {'Household & Cleaning', 'Beauty & Toiletries', 'Home & Garden', 'Health, Medicines & Wellbeing', 'Pets'} +NONFOOD_BRANDS = {'Miniml', 'Heart & Soul', 'INTERNATIONAL GREETINGS', 'Caroline Gardner', 'Colgate'} +ALIASES = {'m s': 'marks spencer', 'm s food':'marks spencer', 'marks and spencer': 'marks spencer', 'marks spencers':'marks spencer', 'kelloggs special k': 'kelloggs', 'kellogg s': 'kelloggs', 'jacob s': 'jacobs', 'mcvitie s': 'mcvities', 'nairn s': 'nairns', 'garner s': 'garners', 'wrigley s extra': 'extra', 'nestle shredded wheat': 'shredded wheat', 'arnotts': 'arnotts', 'tncc': 'natural confectionery co'} +IGNORE = {'the', 'and', 'with', 'of', 'in', 'a', 'an', 'bag', 'sharing', 'multipack', 'pack', 'baked', 'snacks', 'breakfast', 'cereal', 'biscuits', 'biscuit', 'tinned', 'tin', 'can', 'single', 'jar', 'bottle', 'small', 'classic', 'favourites', 'ready', 'meal', 'cook'} + + +def words(text): + text = unicodedata.normalize('NFKD', str(text)).encode('ascii', 'ignore').decode().lower() + return ' '.join(re.findall(r'[a-z0-9]+', text.replace("'", '').replace('’', ''))) + + +def brand_key(text): + key = words(text) + return ALIASES.get(key, key) + + +def pack(text): + """Only unambiguous metric quantities; retain multipack structure.""" + text = str(text).lower().replace('×', 'x').replace(',', '.') + text = text.split('(')[0].strip() + m = re.fullmatch(r'(?:(\d+)\s*x\s*)?(\d+(?:\.\d+)?)\s*(kg|g|ml|cl|l)', text) + if not m: + return None + n, amount, unit = m.groups() + return (int(n or 1), round(float(amount) * {'kg':1000, 'g':1, 'ml':1, 'cl':10, 'l':1000}[unit], 3), 'g' if unit in ('g','kg') else 'ml') + + +def tokens(name, brand): + name = words(re.sub(r'\b\d+(?:\.\d+)?\s*(?:kg|g|ml|cl|l)\b', '', name, flags=re.I)) + remove = set(words(brand).split()) | set(brand_key(brand).split()) | IGNORE + return set(name.split()) - remove + + +def candidate_info(product, metadata, candidate): + brand = metadata.get('ocado', {}).get('brand', '') + expected = metadata.get('ocado', {}).get('size', {}).get('value', metadata.get('size', '')) + title = candidate.get('product_name') or candidate.get('product_name_en', '') + a, b = tokens(product['name'], brand), tokens(title, brand) + brands = [brand_key(x) for x in candidate.get('brands', '').split(',')] + brand_ok = bool(brand) and brand_key(brand) in brands + size_ok = pack(expected) is not None and pack(expected) == pack(candidate.get('quantity', '')) + score = len(a & b) / max(1, len(a | b)) + exact = brand_ok and size_ok and bool(a) and a == b and valid_barcode(candidate.get('code', '')) + return {'code': candidate.get('code'), 'name':title, 'brand':candidate.get('brands'), 'quantity':candidate.get('quantity'), 'brand_match':brand_ok, 'size_match':size_ok, 'name_score':round(score, 3), 'exact':exact} + + +def choose_match(product, metadata, candidates): + ranked = sorted([candidate_info(product, metadata, c) for c in candidates.values()], key=lambda c:(c['exact'], c['brand_match'], c['name_score'], c['size_match']), reverse=True) + matches = [c for c in ranked if c['exact']] + return ranked, candidates[matches[0]['code']] if len(matches) == 1 else None + + +def valid_barcode(code): + if not isinstance(code, str) or not code.isascii() or not code.isdigit() or len(code) not in (8,12,13,14): + return False + return (sum(int(x) * (3 if i % 2 == 0 else 1) for i, x in enumerate(reversed(code[:-1]))) + int(code[-1])) % 10 == 0 + + +class OpenFoodFacts: + def __init__(self, cache, contact, refresh=False): + self.refresh = refresh + self.cache = cache + self.session = requests.Session() + self.session.headers['User-Agent'] = f'ocado-grocy/0.1 ({contact})' + self.last = 0 + + def fetch(self, path, params, *, offline=False): + key = hashlib.sha256(json.dumps([path, params], sort_keys=True).encode()).hexdigest() + target = self.cache / (key + '.json') + if target.exists() and (offline or not self.refresh): + return json.loads(target.read_text()) + if offline: + raise ValueError('Search not cached; rerun without --offline') + # Shared spacing for search and product reads: <= 10 requests per minute. + time.sleep(max(0, 6.2 - (time.monotonic() - self.last))) + self.last = time.monotonic() + url = path if path.startswith('https://search.openfoodfacts.org/') else 'https://world.openfoodfacts.org' + path + response = self.session.get(url, params=params, timeout=45) + response.raise_for_status() + data = response.json() + private_json(target, data) + return data + + def catalog(self, brands, offline=False): + tags = sorted({brand_key(b).replace(' ', '-') for b in brands if b}) + if not tags: + return [] + + def group(tags): + query = 'brands_tags:(' + ' OR '.join(tags) + ')' + result = [] + page = 1 + while True: + data = self.fetch('https://search.openfoodfacts.org/search', {'q':query, 'page_size':1000, 'page':page, 'fields':FIELDS}, offline=offline) + if data.get('timed_out') or data.get('warnings'): + raise ValueError('Incomplete catalog search; retry later') + if not data.get('is_count_exact', False): + if len(tags) == 1: + raise ValueError('Brand catalog exceeds search limit: ' + tags[0]) + middle = len(tags) // 2 + return group(tags[:middle]) + group(tags[middle:]) + for hit in data.get('hits', []): + hit = dict(hit) + if isinstance(hit.get('brands'), list): + hit['brands'] = ','.join(hit['brands']) + result.append(hit) + click.echo(f"Catalog ({len(tags)} brands) page {page}: {len(result)} of {data.get('count')} products") + if page >= data.get('page_count', 1): + if len(result) != data.get('count'): + raise ValueError('Incomplete catalog response') + return result + page += 1 + + # OFF also retains apostrophes as tag separators (e.g. nairn-s). + alternatives = sorted({re.sub(r'[^a-z0-9]+', '-', unicodedata.normalize('NFKD', b).encode('ascii', 'ignore').decode().lower()).strip('-') for b in brands if b} - set(tags)) + result = group(tags) + if alternatives: + result += group(alternatives) + return list({str(p['code']):p for p in result}.values()) + + def product(self, code, offline=False): + data = self.fetch(f'/api/v3.6/product/{code}.json', {'fields':FIELDS + ',nutrition'}, offline=offline) + product = data.get('product') + if not product or str(product.get('code')) != code: + raise ValueError('Open Food Facts did not return requested barcode') + return product + + +def apply_match(client, product, candidate, barcodes): + """Write only the barcode relation and description; never stock or name.""" + code = str(candidate['code']) + if not valid_barcode(code): + raise ValueError('Invalid GTIN check digit') + owners = {int(b['product_id']) for b in barcodes if b['barcode'] == code} + if owners - {int(product['id'])}: + raise ValueError('Barcode already belongs to another Grocy product') + # Re-read just before mutation to retain user edits and skip deleted products. + current = client.get(f"/objects/products/{product['id']}") + if not current: + raise ValueError('Grocy product was deleted') + metadata = read_metadata(current.get('description')) + if not metadata or metadata.get('Ocado product ID') != read_metadata(product.get('description')).get('Ocado product ID'): + raise ValueError('Imported product identity changed') + if metadata.get('Barcode') and metadata['Barcode'] != code: + raise ValueError('Existing metadata barcode differs; review required') + enrichment = {k:v for k,v in candidate.items() if v not in (None, '', [], {})} + enrichment.update(source='Open Food Facts', url=f'https://world.openfoodfacts.org/product/{code}', database_license='ODbL', contents_license='Database Contents License', images_license='CC BY-SA') + metadata['Barcode'] = code + metadata['openfoodfacts'] = enrichment + updated = replace_metadata(current['description'], metadata) + if not owners: + row = {'product_id':int(product['id']), 'barcode':code, 'qu_id':current['qu_id_purchase'], 'amount':1} + client.post('/objects/product_barcodes', row) + barcodes.append(row) + if read_metadata(current['description']) != metadata: + client.request('PUT', f"/objects/products/{product['id']}", {'description':updated}) + return 'updated' + return 'updated' if not owners else 'unchanged' + + +@click.command() +@click.option('--config', type=click.Path(path_type=Path), default=Path('config.toml')) +@click.option('--cache', type=click.Path(path_type=Path), default=Path('history/openfoodfacts')) +@click.option('--contact', help='Contact for the Open Food Facts User-Agent; defaults to config [openfoodfacts].contact.') +@click.option('--refresh-cache', is_flag=True, help='Fetch fresh OFF responses, replacing cached responses.') +@click.option('--apply', is_flag=True, help='Apply unambiguous matches and explicitly reviewed mappings.') +@click.option('--offline', is_flag=True, help='Use cached Open Food Facts responses only.') +@click.option('--mapping', type=click.Path(exists=True, path_type=Path), help='Reviewed JSON mapping of Grocy product ID to barcode.') +@click.option('--limit', type=int, help='Limit product comparisons after downloading the catalog.') +@click.option('--order-id', 'order_ids', multiple=True, help='Limit to products imported from these orders.') +def main(config, cache, contact, refresh_cache, apply, offline, mapping, limit, order_ids): + """Match Ocado imports to Open Food Facts. Defaults to a read-only preview.""" + settings = tomllib.loads(config.read_text()) + contact = contact or settings.get('openfoodfacts', {}).get('contact') + if not contact: + raise click.UsageError('Set --contact or [openfoodfacts].contact in config.toml') + if refresh_cache and offline: + raise click.UsageError('--refresh-cache cannot be combined with --offline') + client = Grocy(**settings['grocy']) + state = config.parent / settings.get('import', {}).get('state_file', 'imports.sqlite3') + products = imported_products(client, state, order_ids) + mappings = json.loads(mapping.read_text()) if mapping else {} + return run_openfoodfacts(client, products, cache, contact, apply=apply, offline=offline, + refresh_cache=refresh_cache, mappings=mappings, limit=limit) + + +def run_openfoodfacts(client, products, cache, contact, *, apply=True, offline=False, + refresh_cache=False, mappings=None, limit=None): + barcodes = client.get('/objects/product_barcodes') + off = OpenFoodFacts(cache / 'responses', contact, refresh_cache) + mappings = mappings or {} + if not isinstance(mappings, dict) or any(not isinstance(v, str) or not valid_barcode(v) for v in mappings.values()): + raise click.UsageError('Mapping must be a JSON object with barcode strings and valid GTIN check digits') + def needs_search(product): + metadata = read_metadata(product.get('description')) + code = metadata.get('openfoodfacts', {}).get('code') + return not mappings.get(str(product['id'])) and not (valid_barcode(code) and code == metadata.get('Barcode')) + brands = {read_metadata(p.get('description')).get('ocado', {}).get('brand') for p in products if needs_search(p)} + catalog = off.catalog(brands, offline) + private_json(cache / 'catalog.json', catalog) + by_brand = {} + for candidate in catalog: + for brand in candidate.get('brands', '').split(','): + by_brand.setdefault(brand_key(brand), {})[str(candidate['code'])] = candidate + report = {'created_at':datetime.now(timezone.utc).isoformat(), 'apply':apply, 'products':[]} + if apply: + private_json(cache / ('before-' + datetime.now().strftime('%Y%m%d-%H%M%S') + '.json'), {'products':products, 'barcodes':barcodes}) + searched = 0 + for product in products: + row = {'product_id':product['id'], 'name':product['name']} + metadata = read_metadata(product.get('description')) + ocado = metadata.get('ocado', {}) + row['size'] = ocado.get('size', {}).get('value', metadata.get('size','')) + try: + if not metadata: + row['status'] = 'missing_metadata' + elif set(ocado.get('categoryPath', [])) & NONFOOD or ocado.get('brand') in NONFOOD_BRANDS: + row['status'] = 'non_food' + elif limit is not None and searched >= limit: + row['status'] = 'not_searched' + else: + searched += 1 + code = mappings.get(str(product['id'])) + method = 'reviewed_mapping' + established = metadata.get('openfoodfacts', {}).get('code') + if not code and established == metadata.get('Barcode') and valid_barcode(established): + code = established + method = 'existing_match' + if code: + candidate = next((c for c in catalog if str(c.get('code')) == str(code)), None) or off.product(str(code), offline) + row['match_method'] = method + row['candidates'] = [candidate_info(product, metadata, candidate)] + else: + query = 'brand catalog: ' + ocado.get('brand', '') + result = {'count':len(catalog)} + candidates = by_brand.get(brand_key(ocado.get('brand', '')), {}) + result['count'] = len(candidates) + ranked, candidate = choose_match(product, metadata, candidates) + row.update(query=query, search_count=result.get('count'), candidates=ranked[:10]) + row['match_method'] = 'exact_brand_name_pack' + if candidate: + # Catalog is indexed separately; fetch current detail before writes. + if apply: + fresh = off.product(str(candidate['code']), offline) + row['current_candidate'] = candidate_info(product, metadata, fresh) + if not code and not row['current_candidate']['exact']: + raise ValueError('Current product details no longer match catalog') + candidate = fresh + row['barcode'] = candidate['code'] + row['status'] = apply_match(client, product, candidate, barcodes) if apply else 'matched' + else: + row['status'] = 'review' if row['candidates'] else 'no_match' + except (requests.RequestException, ValueError, RuntimeError) as exc: + row.update(status='error', error=str(exc)) + report['products'].append(row) + private_json(cache / ('applied-report.json' if apply else 'report.json'), report) + click.echo(f"{product['id']} {row['status']}: {product['name']}") + report['counts'] = {s:sum(r['status']==s for r in report['products']) for s in sorted({r['status'] for r in report['products']})} + private_json(cache / ('applied-report.json' if apply else 'report.json'), report) + click.echo(json.dumps(report['counts'])) + return report + + +if __name__ == '__main__': + main() diff --git a/ocado_grocy/order_state.py b/ocado_grocy/order_state.py new file mode 100644 index 0000000..5d8f3ec --- /dev/null +++ b/ocado_grocy/order_state.py @@ -0,0 +1,56 @@ +"""Order-level completion, with conservative migration of the line journal.""" +import json +import sqlite3 +from types import SimpleNamespace + +from .grocy import align_recorded_lines, item_fingerprint +from .pantry import classify +from .receipt import parse_document + + +def completed_orders(state, server, archive, config): + """Read only: a downloaded receipt alone is never evidence of an import.""" + if not state.exists(): + return set() + with sqlite3.connect(f'{state.resolve().as_uri()}?mode=ro', uri=True) as db: + tables = {r[0] for r in db.execute("SELECT name FROM sqlite_master WHERE type='table'")} + tracked = dict(db.execute('SELECT order_id,status FROM loaded_orders WHERE server=?', (server,))) if 'loaded_orders' in tables else {} + completed = {order for order, status in tracked.items() if status == 'complete'} + if 'imports' not in tables: + return completed + legacy = {r[0] for r in db.execute('SELECT DISTINCT order_id FROM imports WHERE server=?', (server,))} - tracked.keys() + for order in legacy: + path = archive / f'{order}.json' + if not path.exists(): + continue + try: + receipt = parse_document(json.loads(path.read_text())) + if receipt.order_id != order: + continue + receipt = align_recorded_lines(receipt, server, SimpleNamespace(db=db)) + rows = {line:(fingerprint,status) for line,fingerprint,status in db.execute( + 'SELECT line,fingerprint,status FROM imports WHERE server=? AND order_id=?', (server,order))} + if any(status != 'done' for _,status in rows.values()): + continue + selected = [(line,item) for line,item in enumerate(receipt.items,1) + if classify(item,config.get('pantry',{}).get('overrides',{})).include] + if all(rows.get(line) == (item_fingerprint(receipt,item),'done') for line,item in selected): + completed.add(order) + except (ValueError, KeyError, TypeError): + # A changed or incomplete receipt needs the normal reconciliation path. + continue + return completed + + +def mark_orders(state, server, orders, status): + if not orders: + return + state.parent.mkdir(parents=True, exist_ok=True) + with sqlite3.connect(state) as db: + db.execute('''CREATE TABLE IF NOT EXISTS loaded_orders ( + server TEXT NOT NULL, order_id TEXT NOT NULL, status TEXT NOT NULL, + updated_at TEXT NOT NULL DEFAULT CURRENT_TIMESTAMP, + PRIMARY KEY(server,order_id))''') + db.executemany('''INSERT INTO loaded_orders(server,order_id,status) VALUES (?,?,?) + ON CONFLICT(server,order_id) DO UPDATE SET status=excluded.status, updated_at=CURRENT_TIMESTAMP''', + [(server,order,status) for order in orders]) diff --git a/ocado_grocy/pantry.py b/ocado_grocy/pantry.py new file mode 100644 index 0000000..407306f --- /dev/null +++ b/ocado_grocy/pantry.py @@ -0,0 +1,45 @@ +"""Select shelf-stable stock using Ocado metadata, without a product-name list.""" +from dataclasses import dataclass + + +@dataclass(frozen=True) +class Decision: + include: bool + reason: str + uncertain: bool = False + + +PERISHABLE_CATEGORIES = {"fresh & chilled food", "frozen food", "bakery"} +SHELF_STABLE_CATEGORIES = { + "food cupboard", "treats & snacks", "soft drinks, tea & coffee", "beer, wine & spirits", + "household & cleaning", "home care & cleaning", "health, beauty & personal care", + "beauty & toiletries", "pets", "health, medicines & wellbeing", "home & garden", + "m&s food cupboard", "m&s snacks & treats", +} + + +def classify(item, overrides=None): + overrides = overrides or {} + key = item.product_id or item.name + if key in overrides: + value = overrides[key] + if not isinstance(value, bool): + raise ValueError(f"pantry override {key!r} must be true or false") + return Decision(value, "explicit product override") + storage_type = item.metadata.get("storage_type") + if storage_type in {"FRIDGE", "FREEZER"}: + return Decision(False, "requires cold storage") + product = item.metadata.get("ocado", {}) + storage = str(product.get("storage", "")).casefold() + if "keep refrigerated" in storage or "keep frozen" in storage: + return Decision(False, "requires cold storage") + categories = product.get("categoryPath", []) + if isinstance(categories, str): + categories = [categories] + categories = {str(category).strip().casefold() for category in categories} + if categories & PERISHABLE_CATEGORIES: + return Decision(False, "fresh, chilled, frozen or bakery category") + if categories & SHELF_STABLE_CATEGORIES: + return Decision(True, "shelf-stable product category") + # CUPBOARD alone also describes bananas, onions and potatoes. + return Decision(False, "missing or ambiguous product category", uncertain=True) diff --git a/ocado_grocy/postprocess.py b/ocado_grocy/postprocess.py new file mode 100644 index 0000000..cc4bb54 --- /dev/null +++ b/ocado_grocy/postprocess.py @@ -0,0 +1,112 @@ +"""Enrich products belonging to the orders just processed, including reruns.""" +import hashlib +import os +from pathlib import Path + +import click + +from .grocy import Grocy, imported_products +from .history import private_json +from .images import run_images +from .openfoodfacts import run_openfoodfacts +from .enrichment_retry import RetryQueue + + +def enrich_orders(config, config_path, archive, order_ids, *, images=True, openfoodfacts=True): + if not order_ids or not (images or openfoodfacts): + return {'errors':[]} + connection = config.get('grocy', {}) + client = Grocy(os.environ.get('GROCY_URL', connection.get('url', '')), + os.environ.get('GROCY_API_KEY', connection.get('api_key', ''))) + state = Path(config.get('import', {}).get('state_file', 'imports.sqlite3')) + if not state.is_absolute(): + state = config_path.resolve().parent / state + products = imported_products(client, state, order_ids) + key = order_ids[0] if len(order_ids) == 1 and order_ids[0].isdigit() else hashlib.sha256(','.join(sorted(order_ids)).encode()).hexdigest()[:16] + output = archive / 'enrichment' / key + queue = RetryQueue(state, client.url) + summary = {'server':client.url, 'order_ids':list(order_ids), 'product_ids':[p['id'] for p in products], 'errors':[]} + if not products: + queue.close() + private_json(output / 'summary.json', summary) + return summary + if images: + try: + click.echo(f'Adding missing pictures for {len(products)} imported products...') + queue.start('images', products) + result = run_images(client, products, output / 'images.json', raise_errors=False) + queue.results('images', result['products']) + summary['images'] = result['counts'] + if result['counts'].get('error'): + summary['errors'].append(f"Images: {result['counts']['error']} products failed") + except Exception as exc: + summary['errors'].append(f'Images: {exc}') + if openfoodfacts: + try: + queue.start('openfoodfacts', products) + contact = config.get('openfoodfacts', {}).get('contact') + if not contact: + raise ValueError('Set [openfoodfacts].contact in config.toml or use --no-openfoodfacts') + click.echo(f'Matching {len(products)} imported products on Open Food Facts...') + result = run_openfoodfacts(client, products, archive / 'openfoodfacts', contact) + queue.results('openfoodfacts', result['products']) + private_json(output / 'openfoodfacts.json', result) + summary['openfoodfacts'] = result['counts'] + if result['counts'].get('error'): + summary['errors'].append(f"Open Food Facts: {result['counts']['error']} products failed; see {output / 'openfoodfacts.json'}") + except Exception as exc: + summary['errors'].append(f'Open Food Facts: {exc}') + queue.close() + private_json(output / 'summary.json', summary) + return summary + + +def retry_enrichment(config, config_path, archive, *, images=True, openfoodfacts=True): + connection = config.get('grocy', {}) + client = Grocy(os.environ.get('GROCY_URL', connection.get('url','')), + os.environ.get('GROCY_API_KEY', connection.get('api_key',''))) + state = config_path.resolve().parent / config.get('import',{}).get('state_file','imports.sqlite3') + if not state.exists(): + return {'errors':[]} + queue = RetryQueue(state, client.url) + try: + # Journal scope protects other Grocy installations and non-imported products. + tables = {r[0] for r in queue.db.execute("SELECT name FROM sqlite_master WHERE type='table'")} + if 'imports' not in tables: + return {'errors':[]} + allowed = {int(r[0]) for r in queue.db.execute("SELECT DISTINCT product_id FROM imports WHERE server=? AND status='done'",(client.url,))} + queue.migrate_reports(archive, allowed) + enabled = [('images',images),('openfoodfacts',openfoodfacts)] + summary = {'errors':[]} + live = None + for stage, active in enabled: + pending = queue.pending(stage) + if not active or not pending: + continue + if live is None: + live = {int(p['id']):p for p in client.get('/objects/products')} + products = [live[pid] for pid in sorted(pending & allowed & live.keys())] + queue.results(stage,[{'product_id':pid,'status':'deleted_or_out_of_scope'} for pid in pending - {int(p['id']) for p in products}]) + if not products: + continue + click.echo(f'Retrying {stage} for {len(products)} previously failed products...') + output = archive / 'enrichment' / 'retries' + try: + if stage == 'images': + result = run_images(client, products, output/'images.json', raise_errors=False) + else: + contact = config.get('openfoodfacts',{}).get('contact') + if not contact: + raise ValueError('Set [openfoodfacts].contact in config.toml') + result = run_openfoodfacts(client, products, archive/'openfoodfacts',contact) + private_json(output/'openfoodfacts.json',result) + queue.results(stage,result['products']) + summary[stage] = result['counts'] + if result['counts'].get('error'): + summary['errors'].append(f"{stage}: {result['counts']['error']} retries failed") + except Exception as exc: + summary['errors'].append(f'{stage}: {exc}') + private_json(archive/'enrichment'/'retry-summary.json',summary) + return summary + finally: + queue.close() diff --git a/ocado_grocy/product_metadata.py b/ocado_grocy/product_metadata.py new file mode 100644 index 0000000..976534c --- /dev/null +++ b/ocado_grocy/product_metadata.py @@ -0,0 +1,38 @@ +"""Read imported metadata even when Grocy's sanitizer removes the pre tag.""" +import json +from html import escape +from lxml import html + + +def read_metadata(description): + try: + text = html.fromstring(description or '

').text_content() + except (html.etree.ParserError, ValueError): + return {} + for index, character in enumerate(text): + if character == '{': + try: + value, _ = json.JSONDecoder().raw_decode(text[index:]) + if isinstance(value, dict) and 'Ocado product ID' in value: + return value + except ValueError: + pass + return {} + + +def replace_metadata(description, metadata): + tree = html.fragment_fromstring(description, create_parent='div') + for node in tree.iter(): + for attr in ('text', 'tail'): + text = getattr(node, attr) or '' + for index, character in enumerate(text): + if character != '{': + continue + try: + old, end = json.JSONDecoder().raw_decode(text[index:]) + except ValueError: + continue + if isinstance(old, dict) and 'Ocado product ID' in old: + setattr(node, attr, text[:index] + json.dumps(metadata, ensure_ascii=False, indent=2) + text[index + end:]) + return escape(tree.text or '') + ''.join(html.tostring(child, encoding='unicode') for child in tree) + raise ValueError('Cannot safely locate imported metadata in description') diff --git a/ocado_grocy/receipt.py b/ocado_grocy/receipt.py new file mode 100644 index 0000000..e7ea997 --- /dev/null +++ b/ocado_grocy/receipt.py @@ -0,0 +1,161 @@ +"""Parse receipt data only: never treat the live basket as an order.""" +from dataclasses import asdict, dataclass, field +from datetime import date, datetime +from zoneinfo import ZoneInfo +from decimal import Decimal, InvalidOperation +import json + + +class ImportError(ValueError): + """An input or integration error suitable for displaying to the user.""" + + +def number(value, label, *, money=False): + if isinstance(value, dict): + currency = value.get("currency", "GBP") + if currency not in ("GBP", "GBX"): + raise ImportError(f"Unsupported currency: {currency}") + result = number(value.get("amount"), label, money=money) + return result / 100 if currency == "GBX" else result + text = str(value).strip().replace("£", "").replace(",", "") + try: + result = Decimal(text) + except InvalidOperation as exc: + raise ImportError(f"Invalid {label}: {value!r}") from exc + if not result.is_finite() or result < 0: + raise ImportError(f"Invalid {label}: {value!r}") + return result + + +def receipt_date(value): + if not value: + return None + text = str(value).strip() + try: + return date.fromisoformat(text[:10]).isoformat() + except ValueError: + pass + for fmt in ("%d/%m/%Y", "%d %B %Y", "%d %b %Y", "%A %d %B %Y"): + try: + return datetime.strptime(text, fmt).date().isoformat() + except ValueError: + pass + raise ImportError(f"Unrecognised date: {value!r}; use YYYY-MM-DD") + + +@dataclass +class Item: + name: str + quantity: Decimal + total: Decimal + product_id: str = "" + barcode: str = "" + url: str = "" + best_before: str | None = None + metadata: dict = field(default_factory=dict) + + @property + def unit_price(self): + return self.total / self.quantity + + +@dataclass +class Receipt: + order_id: str + purchased_date: str | None + items: list[Item] + metadata: dict = field(default_factory=dict) + + def export(self): + return asdict(self) + + +def parse_document(data): + """Read a normalized cached receipt.""" + if not isinstance(data, dict) or not isinstance(data.get("items"), list): + raise ImportError("Receipt JSON requires an items array") + if data.get("currency", "GBP") != "GBP": + raise ImportError("Only GBP receipts are supported") + order_id = str(data.get("order_id") or "").strip() + if not order_id: + raise ImportError("Receipt has no order ID") + items = [] + for row in data["items"]: + if not isinstance(row, dict): + raise ImportError("Receipt items must be objects") + status = str(row.get("status", "")).lower() + if status in {"unavailable", "cancelled", "rejected", "not delivered"}: + continue + quantity = number(row.get("delivered_quantity", row.get("quantity")), "quantity") + if quantity == 0: + continue + name = str(row.get("name") or "").strip() + if not name: + raise ImportError("Receipt item has no name") + total = number(row.get("total"), f"line total for {name}", money=True) + metadata = dict(row.get("metadata") or {}) + metadata.update({k: v for k, v in row.items() if k not in { + "name", "quantity", "delivered_quantity", "total", "product_id", + "barcode", "url", "best_before", "metadata"}}) + items.append(Item(name, quantity, total, str(row.get("product_id") or ""), + str(row.get("barcode") or ""), str(row.get("url") or ""), + receipt_date(row.get("best_before")), metadata)) + if not items: + raise ImportError("Receipt contains no delivered items") + return Receipt(order_id, receipt_date(data.get("purchased_date")), items, data.get("metadata", {})) + + +def product_metadata(product): + return {key: product[key] for key in ("brand", "categoryPath", "size", "packSize", + "description", "ingredients", "nutritionalInformation", "allergens", "storage", + "guaranteedProductLife", "retailerProductId") if key in product} + + +def parse_ocado_order(data, *, expected_order_id=None, product_urls=None): + """Parse the order JSON loaded by Playwright, never the page's live basket.""" + try: + order = data["entities"]["order"][data["result"]] + order_id = str(order["orderId"]) + if expected_order_id and order_id != str(expected_order_id): + raise ImportError("Ocado API order does not match the requested order") + if order["status"] != "DELIVERED": + raise ImportError("Ocado order is not delivered") + dates = order["dates"] + delivered_at = datetime.fromisoformat(dates["deliveryStartDate"].replace("Z", "+00:00")) + if delivered_at.tzinfo is not None: + delivered_at = delivered_at.astimezone(ZoneInfo(dates.get("timeZoneId", "Europe/London"))) + purchased_date = delivered_at.date().isoformat() + groups = order["groupedProducts"] + products = list(groups["products"]) + for original in groups.get("substitutes", []): + for replacement in original.get("substitutes", []): + if replacement.get("status") == "ACCEPTED": + products.append({"storageType": original.get("storageType"), + "substituted_for": original["name"], **replacement}) + rows = [] + catalogue = data["entities"].get("product", {}) + for product in products: + info = catalogue.get(product["productId"], {}) + retailer_id = str(product["retailerProductId"]) + metadata = {"ocado": product_metadata(info), "storage_type": product.get("storageType"), + "pack_info": product.get("packInfo", {}), "promotions": product.get("promotions", []), + "original_line_price": product["prices"].get("retail"), + "price_per_item": product["prices"].get("pricePerItem"), + "image_url": (product.get("image") or {}).get("src", "")} + if product.get("substituted_for"): + metadata["substituted_for"] = product["substituted_for"] + rows.append({"name": " ".join(product["name"].split()), "quantity": product["quantity"], + "total": product["prices"]["offered"], "product_id": retailer_id, + "best_before": product.get("expirationDate"), + "url": (product_urls or {}).get(retailer_id, ""), "metadata": metadata}) + # Stable ordering is independent of the browser's equal-expiry sorting. + rows.sort(key=lambda row:(row["best_before"] or "9999-12-31", row["product_id"], row["name"])) + receipt = parse_document({"order_id":order_id,"purchased_date":purchased_date,"items":rows}) + if sum(item.quantity for item in receipt.items) != number(order["totalItems"], "receipt item count"): + raise ImportError("Delivered quantities do not match the order summary") + if sum(item.total for item in receipt.items) != number(order["orderTotals"]["itemPriceAfterPromos"], "receipt product total"): + raise ImportError("Delivered line prices do not match the order summary") + receipt.metadata = {"source":"ocado-order-api", "order_totals":order["orderTotals"]} + return receipt + except (KeyError, TypeError, AttributeError) as exc: + raise ImportError("Unrecognised Ocado order API structure") from exc diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..39c3361 --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,25 @@ +[build-system] +requires = ["setuptools>=68"] +build-backend = "setuptools.build_meta" + +[project] +name = "ocado-grocy" +version = "0.1.0" +description = "Import Ocado orders into Grocy using Playwright" +requires-python = ">=3.11" +dependencies = ["lxml>=5,<7", "click>=8.1,<9", "requests>=2.31,<3", "playwright>=1.50,<2"] + +[project.optional-dependencies] +test = ["pytest>=8,<10"] + +[project.scripts] +ocado-grocy = "ocado_grocy.cli:main" +ocado-grocy-history = "ocado_grocy.history:main" +ocado-grocy-off = "ocado_grocy.openfoodfacts:main" +ocado-grocy-images = "ocado_grocy.images:main" + +[tool.setuptools.packages.find] +include = ["ocado_grocy*"] + +[tool.pytest.ini_options] +testpaths = ["tests"] diff --git a/tests/fixtures/history_receipt.json b/tests/fixtures/history_receipt.json new file mode 100644 index 0000000..ba4661c --- /dev/null +++ b/tests/fixtures/history_receipt.json @@ -0,0 +1,870 @@ +{ + "_comment": "Synthetic API fixture. All identifiers, product details, dates and prices are invented.", + "result": "1000000000001", + "entities": { + "order": { + "1000000000001": { + "orderId": "1000000000001", + "status": "DELIVERED", + "dates": { + "deliveryStartDate": "2020-01-02T12:00:00Z", + "timeZoneId": "Europe/London" + }, + "orderTotals": { + "itemPriceAfterPromos": { + "amount": "28.75", + "currency": "GBP" + } + }, + "totalItems": 23, + "groupedProducts": { + "products": [ + { + "productId": "example-product-1", + "retailerProductId": "90000001", + "name": "Example Bananas", + "quantity": 1, + "prices": { + "retail": { + "amount": "1.50", + "currency": "GBP" + }, + "offered": { + "amount": "1.25", + "currency": "GBP" + }, + "pricePerItem": { + "amount": "1.25", + "currency": "GBP" + } + }, + "storageType": "CUPBOARD", + "expirationDate": "2020-01-05", + "isInCurrentCatalog": true, + "image": { + "src": "https://www.ocado.com/images-v3/example-1.jpg" + }, + "packInfo": { + "packSizeDescription": "250g" + }, + "promotions": [] + }, + { + "productId": "example-product-2", + "retailerProductId": "90000002", + "name": "Example Lettuce", + "quantity": 1, + "prices": { + "retail": { + "amount": "1.50", + "currency": "GBP" + }, + "offered": { + "amount": "1.25", + "currency": "GBP" + }, + "pricePerItem": { + "amount": "1.25", + "currency": "GBP" + } + }, + "storageType": "FRIDGE", + "expirationDate": "2020-01-05", + "isInCurrentCatalog": true, + "image": { + "src": "https://www.ocado.com/images-v3/example-2.jpg" + }, + "packInfo": { + "packSizeDescription": "250g" + }, + "promotions": [] + }, + { + "productId": "example-product-3", + "retailerProductId": "90000003", + "name": "Example Feta", + "quantity": 1, + "prices": { + "retail": { + "amount": "1.50", + "currency": "GBP" + }, + "offered": { + "amount": "1.25", + "currency": "GBP" + }, + "pricePerItem": { + "amount": "1.25", + "currency": "GBP" + } + }, + "storageType": "FRIDGE", + "expirationDate": "2030-01-01", + "isInCurrentCatalog": true, + "image": { + "src": "https://www.ocado.com/images-v3/example-3.jpg" + }, + "packInfo": { + "packSizeDescription": "250g" + }, + "promotions": [] + }, + { + "productId": "example-product-4", + "retailerProductId": "90000004", + "name": "Example Milk", + "quantity": 1, + "prices": { + "retail": { + "amount": "1.50", + "currency": "GBP" + }, + "offered": { + "amount": "1.25", + "currency": "GBP" + }, + "pricePerItem": { + "amount": "1.25", + "currency": "GBP" + } + }, + "storageType": "FRIDGE", + "expirationDate": "2020-01-05", + "isInCurrentCatalog": true, + "image": { + "src": "https://www.ocado.com/images-v3/example-4.jpg" + }, + "packInfo": { + "packSizeDescription": "250g" + }, + "promotions": [] + }, + { + "productId": "example-product-5", + "retailerProductId": "90000005", + "name": "Example Yogurt", + "quantity": 1, + "prices": { + "retail": { + "amount": "1.50", + "currency": "GBP" + }, + "offered": { + "amount": "1.25", + "currency": "GBP" + }, + "pricePerItem": { + "amount": "1.25", + "currency": "GBP" + } + }, + "storageType": "FRIDGE", + "expirationDate": "2020-01-05", + "isInCurrentCatalog": true, + "image": { + "src": "https://www.ocado.com/images-v3/example-5.jpg" + }, + "packInfo": { + "packSizeDescription": "250g" + }, + "promotions": [] + }, + { + "productId": "example-product-6", + "retailerProductId": "90000006", + "name": "Example Frozen Peas", + "quantity": 1, + "prices": { + "retail": { + "amount": "1.50", + "currency": "GBP" + }, + "offered": { + "amount": "1.25", + "currency": "GBP" + }, + "pricePerItem": { + "amount": "1.25", + "currency": "GBP" + } + }, + "storageType": "FREEZER", + "expirationDate": "2020-01-05", + "isInCurrentCatalog": true, + "image": { + "src": "https://www.ocado.com/images-v3/example-6.jpg" + }, + "packInfo": { + "packSizeDescription": "250g" + }, + "promotions": [] + }, + { + "productId": "example-product-7", + "retailerProductId": "90000007", + "name": "Example Bread", + "quantity": 1, + "prices": { + "retail": { + "amount": "1.50", + "currency": "GBP" + }, + "offered": { + "amount": "1.25", + "currency": "GBP" + }, + "pricePerItem": { + "amount": "1.25", + "currency": "GBP" + } + }, + "storageType": "FRIDGE", + "expirationDate": "2020-01-05", + "isInCurrentCatalog": true, + "image": { + "src": "https://www.ocado.com/images-v3/example-7.jpg" + }, + "packInfo": { + "packSizeDescription": "250g" + }, + "promotions": [] + }, + { + "productId": "example-product-8", + "retailerProductId": "90000008", + "name": "Example Apples", + "quantity": 1, + "prices": { + "retail": { + "amount": "1.50", + "currency": "GBP" + }, + "offered": { + "amount": "1.25", + "currency": "GBP" + }, + "pricePerItem": { + "amount": "1.25", + "currency": "GBP" + } + }, + "storageType": "FRIDGE", + "expirationDate": "2020-01-05", + "isInCurrentCatalog": true, + "image": { + "src": "https://www.ocado.com/images-v3/example-8.jpg" + }, + "packInfo": { + "packSizeDescription": "250g" + }, + "promotions": [] + }, + { + "productId": "example-product-9", + "retailerProductId": "90000009", + "name": "Example Carrots", + "quantity": 1, + "prices": { + "retail": { + "amount": "1.50", + "currency": "GBP" + }, + "offered": { + "amount": "1.25", + "currency": "GBP" + }, + "pricePerItem": { + "amount": "1.25", + "currency": "GBP" + } + }, + "storageType": "FRIDGE", + "expirationDate": "2020-01-05", + "isInCurrentCatalog": true, + "image": { + "src": "https://www.ocado.com/images-v3/example-9.jpg" + }, + "packInfo": { + "packSizeDescription": "250g" + }, + "promotions": [] + }, + { + "productId": "example-product-10", + "retailerProductId": "90000010", + "name": "Example Spinach", + "quantity": 1, + "prices": { + "retail": { + "amount": "1.50", + "currency": "GBP" + }, + "offered": { + "amount": "1.25", + "currency": "GBP" + }, + "pricePerItem": { + "amount": "1.25", + "currency": "GBP" + } + }, + "storageType": "FRIDGE", + "expirationDate": "2020-01-05", + "isInCurrentCatalog": true, + "image": { + "src": "https://www.ocado.com/images-v3/example-10.jpg" + }, + "packInfo": { + "packSizeDescription": "250g" + }, + "promotions": [] + }, + { + "productId": "example-product-11", + "retailerProductId": "90000011", + "name": "Example Frozen Berries", + "quantity": 1, + "prices": { + "retail": { + "amount": "1.50", + "currency": "GBP" + }, + "offered": { + "amount": "1.25", + "currency": "GBP" + }, + "pricePerItem": { + "amount": "1.25", + "currency": "GBP" + } + }, + "storageType": "FREEZER", + "expirationDate": "2020-01-05", + "isInCurrentCatalog": true, + "image": { + "src": "https://www.ocado.com/images-v3/example-11.jpg" + }, + "packInfo": { + "packSizeDescription": "250g" + }, + "promotions": [] + }, + { + "productId": "example-product-12", + "retailerProductId": "90000012", + "name": "Example Fresh Pasta", + "quantity": 1, + "prices": { + "retail": { + "amount": "1.50", + "currency": "GBP" + }, + "offered": { + "amount": "1.25", + "currency": "GBP" + }, + "pricePerItem": { + "amount": "1.25", + "currency": "GBP" + } + }, + "storageType": "FRIDGE", + "expirationDate": "2020-01-05", + "isInCurrentCatalog": true, + "image": { + "src": "https://www.ocado.com/images-v3/example-12.jpg" + }, + "packInfo": { + "packSizeDescription": "250g" + }, + "promotions": [] + }, + { + "productId": "example-product-13", + "retailerProductId": "90000013", + "name": "Example Rice", + "quantity": 1, + "prices": { + "retail": { + "amount": "1.50", + "currency": "GBP" + }, + "offered": { + "amount": "1.25", + "currency": "GBP" + }, + "pricePerItem": { + "amount": "1.25", + "currency": "GBP" + } + }, + "storageType": "CUPBOARD", + "expirationDate": null, + "isInCurrentCatalog": true, + "image": { + "src": "https://www.ocado.com/images-v3/example-13.jpg" + }, + "packInfo": { + "packSizeDescription": "250g" + }, + "promotions": [] + }, + { + "productId": "example-product-14", + "retailerProductId": "90000014", + "name": "Example Dried Pasta", + "quantity": 2, + "prices": { + "retail": { + "amount": "3.00", + "currency": "GBP" + }, + "offered": { + "amount": "2.50", + "currency": "GBP" + }, + "pricePerItem": { + "amount": "1.25", + "currency": "GBP" + } + }, + "storageType": "CUPBOARD", + "expirationDate": null, + "isInCurrentCatalog": true, + "image": { + "src": "https://www.ocado.com/images-v3/example-14.jpg" + }, + "packInfo": { + "packSizeDescription": "250g" + }, + "promotions": [] + }, + { + "productId": "example-product-15", + "retailerProductId": "90000015", + "name": "Example Tea", + "quantity": 1, + "prices": { + "retail": { + "amount": "1.50", + "currency": "GBP" + }, + "offered": { + "amount": "1.25", + "currency": "GBP" + }, + "pricePerItem": { + "amount": "1.25", + "currency": "GBP" + } + }, + "storageType": "CUPBOARD", + "expirationDate": null, + "isInCurrentCatalog": true, + "image": { + "src": "https://www.ocado.com/images-v3/example-15.jpg" + }, + "packInfo": { + "packSizeDescription": "250g" + }, + "promotions": [] + }, + { + "productId": "example-product-16", + "retailerProductId": "90000016", + "name": "Example Coffee", + "quantity": 1, + "prices": { + "retail": { + "amount": "1.50", + "currency": "GBP" + }, + "offered": { + "amount": "1.25", + "currency": "GBP" + }, + "pricePerItem": { + "amount": "1.25", + "currency": "GBP" + } + }, + "storageType": "CUPBOARD", + "expirationDate": null, + "isInCurrentCatalog": true, + "image": { + "src": "https://www.ocado.com/images-v3/example-16.jpg" + }, + "packInfo": { + "packSizeDescription": "250g" + }, + "promotions": [] + }, + { + "productId": "example-product-17", + "retailerProductId": "90000017", + "name": "Example Biscuits", + "quantity": 1, + "prices": { + "retail": { + "amount": "1.50", + "currency": "GBP" + }, + "offered": { + "amount": "1.25", + "currency": "GBP" + }, + "pricePerItem": { + "amount": "1.25", + "currency": "GBP" + } + }, + "storageType": "CUPBOARD", + "expirationDate": null, + "isInCurrentCatalog": true, + "image": { + "src": "https://www.ocado.com/images-v3/example-17.jpg" + }, + "packInfo": { + "packSizeDescription": "250g" + }, + "promotions": [] + }, + { + "productId": "example-product-18", + "retailerProductId": "90000018", + "name": "Example Tinned Tomatoes", + "quantity": 1, + "prices": { + "retail": { + "amount": "1.50", + "currency": "GBP" + }, + "offered": { + "amount": "1.25", + "currency": "GBP" + }, + "pricePerItem": { + "amount": "1.25", + "currency": "GBP" + } + }, + "storageType": "CUPBOARD", + "expirationDate": null, + "isInCurrentCatalog": true, + "image": { + "src": "https://www.ocado.com/images-v3/example-18.jpg" + }, + "packInfo": { + "packSizeDescription": "250g" + }, + "promotions": [] + }, + { + "productId": "example-product-19", + "retailerProductId": "90000019", + "name": "Example Soap", + "quantity": 1, + "prices": { + "retail": { + "amount": "1.50", + "currency": "GBP" + }, + "offered": { + "amount": "1.25", + "currency": "GBP" + }, + "pricePerItem": { + "amount": "1.25", + "currency": "GBP" + } + }, + "storageType": "CUPBOARD", + "expirationDate": null, + "isInCurrentCatalog": true, + "image": { + "src": "https://www.ocado.com/images-v3/example-19.jpg" + }, + "packInfo": { + "packSizeDescription": "250g" + }, + "promotions": [] + }, + { + "productId": "example-product-20", + "retailerProductId": "90000020", + "name": "Example Cleaning Cloths", + "quantity": 2, + "prices": { + "retail": { + "amount": "3.00", + "currency": "GBP" + }, + "offered": { + "amount": "2.50", + "currency": "GBP" + }, + "pricePerItem": { + "amount": "1.25", + "currency": "GBP" + } + }, + "storageType": "CUPBOARD", + "expirationDate": null, + "isInCurrentCatalog": true, + "image": { + "src": "https://www.ocado.com/images-v3/example-20.jpg" + }, + "packInfo": { + "packSizeDescription": "250g" + }, + "promotions": [] + }, + { + "productId": "example-product-21", + "retailerProductId": "90000021", + "name": "Example Pet Food", + "quantity": 1, + "prices": { + "retail": { + "amount": "1.50", + "currency": "GBP" + }, + "offered": { + "amount": "1.25", + "currency": "GBP" + }, + "pricePerItem": { + "amount": "1.25", + "currency": "GBP" + } + }, + "storageType": "CUPBOARD", + "expirationDate": null, + "isInCurrentCatalog": true, + "image": { + "src": "https://www.ocado.com/images-v3/example-21.jpg" + }, + "packInfo": { + "packSizeDescription": "250g" + }, + "promotions": [] + } + ], + "substitutes": [] + } + } + }, + "product": { + "example-product-1": { + "retailerProductId": "90000001", + "brand": "Example Brand", + "categoryPath": [ + "Fresh & Chilled Food" + ], + "size": { + "value": "250g" + } + }, + "example-product-2": { + "retailerProductId": "90000002", + "brand": "Example Brand", + "categoryPath": [ + "Fresh & Chilled Food" + ], + "size": { + "value": "250g" + } + }, + "example-product-3": { + "retailerProductId": "90000003", + "brand": "Example Brand", + "categoryPath": [ + "Fresh & Chilled Food" + ], + "size": { + "value": "250g" + } + }, + "example-product-4": { + "retailerProductId": "90000004", + "brand": "Example Brand", + "categoryPath": [ + "Fresh & Chilled Food" + ], + "size": { + "value": "250g" + } + }, + "example-product-5": { + "retailerProductId": "90000005", + "brand": "Example Brand", + "categoryPath": [ + "Fresh & Chilled Food" + ], + "size": { + "value": "250g" + } + }, + "example-product-6": { + "retailerProductId": "90000006", + "brand": "Example Brand", + "categoryPath": [ + "Fresh & Chilled Food" + ], + "size": { + "value": "250g" + } + }, + "example-product-7": { + "retailerProductId": "90000007", + "brand": "Example Brand", + "categoryPath": [ + "Fresh & Chilled Food" + ], + "size": { + "value": "250g" + } + }, + "example-product-8": { + "retailerProductId": "90000008", + "brand": "Example Brand", + "categoryPath": [ + "Fresh & Chilled Food" + ], + "size": { + "value": "250g" + } + }, + "example-product-9": { + "retailerProductId": "90000009", + "brand": "Example Brand", + "categoryPath": [ + "Fresh & Chilled Food" + ], + "size": { + "value": "250g" + } + }, + "example-product-10": { + "retailerProductId": "90000010", + "brand": "Example Brand", + "categoryPath": [ + "Fresh & Chilled Food" + ], + "size": { + "value": "250g" + } + }, + "example-product-11": { + "retailerProductId": "90000011", + "brand": "Example Brand", + "categoryPath": [ + "Fresh & Chilled Food" + ], + "size": { + "value": "250g" + } + }, + "example-product-12": { + "retailerProductId": "90000012", + "brand": "Example Brand", + "categoryPath": [ + "Fresh & Chilled Food" + ], + "size": { + "value": "250g" + } + }, + "example-product-13": { + "retailerProductId": "90000013", + "brand": "Example Brand", + "categoryPath": [ + "Food Cupboard" + ], + "size": { + "value": "250g" + } + }, + "example-product-14": { + "retailerProductId": "90000014", + "brand": "Example Brand", + "categoryPath": [ + "Food Cupboard" + ], + "size": { + "value": "250g" + } + }, + "example-product-15": { + "retailerProductId": "90000015", + "brand": "Example Brand", + "categoryPath": [ + "Food Cupboard" + ], + "size": { + "value": "250g" + } + }, + "example-product-16": { + "retailerProductId": "90000016", + "brand": "Example Brand", + "categoryPath": [ + "Food Cupboard" + ], + "size": { + "value": "250g" + } + }, + "example-product-17": { + "retailerProductId": "90000017", + "brand": "Example Brand", + "categoryPath": [ + "Food Cupboard" + ], + "size": { + "value": "250g" + } + }, + "example-product-18": { + "retailerProductId": "90000018", + "brand": "Example Brand", + "categoryPath": [ + "Food Cupboard" + ], + "size": { + "value": "250g" + } + }, + "example-product-19": { + "retailerProductId": "90000019", + "brand": "Example Brand", + "categoryPath": [ + "Household & Cleaning" + ], + "size": { + "value": "250g" + } + }, + "example-product-20": { + "retailerProductId": "90000020", + "brand": "Example Brand", + "categoryPath": [ + "Household & Cleaning" + ], + "size": { + "value": "250g" + } + }, + "example-product-21": { + "retailerProductId": "90000021", + "brand": "Example Brand", + "categoryPath": [ + "Pets" + ], + "size": { + "value": "250g" + } + } + } + } +} diff --git a/tests/test_enrichment_retry.py b/tests/test_enrichment_retry.py new file mode 100644 index 0000000..4c553d9 --- /dev/null +++ b/tests/test_enrichment_retry.py @@ -0,0 +1,85 @@ +import sqlite3 +from types import SimpleNamespace + +from click.testing import CliRunner + +from ocado_grocy.enrichment_retry import RetryQueue +from ocado_grocy.history import private_json, main +from ocado_grocy import postprocess +from ocado_grocy.order_state import mark_orders + + +def test_legacy_report_recovery_is_scoped_and_does_not_resurrect_old_errors(tmp_path): + output = tmp_path/'enrichment'/'old' + private_json(output/'summary.json',{'order_ids':['123'],'errors':['Open Food Facts: failed']}) + private_json(output/'openfoodfacts.json',{'products':[ + {'product_id':1,'status':'error','error':'ConnectionError'}, + {'product_id':2,'status':'unchanged'}, {'product_id':99,'status':'error'}]}) + queue = RetryQueue(tmp_path/'state.sqlite3','server') + queue.migrate_reports(tmp_path,{1,2}) + assert queue.pending('openfoodfacts') == {1} + queue.results('openfoodfacts',[{'product_id':1,'status':'unchanged'}]) + queue.migrate_reports(tmp_path,{1,2}) + assert not queue.pending('openfoodfacts') + queue.close() + + +def test_queue_separates_stages_and_servers_and_keeps_only_failures(tmp_path): + queue = RetryQueue(tmp_path/'state.sqlite3','server') + queue.start('images',[{'id':1},{'id':2}]) + queue.start('openfoodfacts',[{'id':1}]) + queue.results('images',[{'product_id':1,'status':'updated'},{'product_id':2,'status':'error'}]) + assert queue.pending('images') == {2} + assert queue.pending('openfoodfacts') == {1} + other = RetryQueue(tmp_path/'state.sqlite3','other') + assert not other.pending('images') + other.close() + queue.close() + + +def test_retry_targets_only_failed_live_imports_and_clears_success(tmp_path, monkeypatch): + state = tmp_path/'imports.sqlite3' + with sqlite3.connect(state) as db: + db.execute('CREATE TABLE imports(server,product_id,status)') + db.executemany('INSERT INTO imports VALUES (?,?,?)',[('server',1,'done'),('server',2,'done'),('server',3,'done')]) + queue = RetryQueue(state,'server') + queue.start('openfoodfacts',[{'id':1},{'id':3}]) + queue.close() + class Client: + url = 'server' + def get(self,path): + assert path == '/objects/products' + return [{'id':1},{'id':2}] # 3 was deleted + monkeypatch.setattr(postprocess,'Grocy',lambda *a:Client()) + calls = [] + def run(client, products, cache, contact): + calls.append(products) + return {'counts':{'unchanged':1},'products':[{'product_id':1,'status':'unchanged'}]} + monkeypatch.setattr(postprocess,'run_openfoodfacts',run) + result = postprocess.retry_enrichment({'openfoodfacts':{'contact':'test@example.com'}},tmp_path/'config.toml',tmp_path) + assert calls == [[{'id':1}]] and not result['errors'] + queue = RetryQueue(state,'server') + assert not queue.pending('openfoodfacts') + queue.close() + + +def test_main_retries_even_with_no_new_orders(tmp_path, monkeypatch): + from ocado_grocy import history + monkeypatch.setattr(history,'cached_orders',lambda *a:([],[])) + calls = [] + monkeypatch.setattr(postprocess,'retry_enrichment',lambda *a,**kw:calls.append(True) or {'errors':[]}) + result = CliRunner().invoke(main,['--cached','--config',str(tmp_path/'config.toml'),'--archive',str(tmp_path)]) + assert result.exit_code == 0 and calls == [True] + assert 'No new delivered orders' in result.output + calls.clear() + result = CliRunner().invoke(main,['--cached','--dry-run','--config',str(tmp_path/'config.toml'),'--archive',str(tmp_path)]) + assert result.exit_code == 0 and not calls + + +def test_retry_failure_keeps_error_exit_even_without_new_orders(tmp_path, monkeypatch): + from ocado_grocy import history + monkeypatch.setattr(history,'cached_orders',lambda *a:([],[])) + monkeypatch.setattr(postprocess,'retry_enrichment',lambda *a,**kw:{'errors':['ConnectionError']}) + result = CliRunner().invoke(main,['--cached','--config',str(tmp_path/'config.toml'),'--archive',str(tmp_path)]) + assert result.exit_code == 1 + assert 'Enrichment retries still need attention' in result.output diff --git a/tests/test_grocy.py b/tests/test_grocy.py new file mode 100644 index 0000000..476f880 --- /dev/null +++ b/tests/test_grocy.py @@ -0,0 +1,151 @@ +from decimal import Decimal + +import pytest + +from ocado_grocy.grocy import Journal, import_receipt +from ocado_grocy.receipt import ImportError, Item, Receipt + + +class FakeGrocy: + url = "https://grocy.test/api" + + def __init__(self, products=None, fail=False): + self.products = products or [] + self.posts = [] + self.fail = fail + + def get(self, path): + if path == "/objects/products": + return self.products.copy() + if path == "/objects/product_barcodes": + return [] + if path.startswith("/stock/products/"): + return {"qu_conversion_factor_purchase_to_stock": 3} + raise AssertionError(path) + + def named_object(self, *args, **kwargs): + return 1 + + def post(self, path, payload): + self.posts.append((path, payload)) + if path == "/objects/products": + self.products.append({"id": len(self.products) + 1, **payload}) + return {"created_object_id": len(self.products)} + if self.fail and path.endswith("/add"): + raise ImportError("Network failed after stock may have been written") + return [] + + +def receipt(): + return Receipt("123", "2026-09-08", [Item("Apples", Decimal(2), Decimal("4.50"), "ocado123")]) + + +def test_import_creation_prices_and_duplicate_protection(tmp_path): + api = FakeGrocy() + journal = Journal(tmp_path / "state.sqlite3") + assert import_receipt(receipt(), api, journal, {}) == 1 + assert api.posts[0][0] == "/objects/products" + stock = api.posts[-1][1] + assert stock["amount"] == 2 + assert stock["price"] == 2.25 + assert stock["purchased_date"] == "2026-09-08" + before = len(api.posts) + assert import_receipt(receipt(), api, journal, {}) == 0 + assert len(api.posts) == before + + +def test_existing_product_conversion(tmp_path): + api = FakeGrocy([{"id": 9, "name": "Apples", "qu_id_purchase": 1, "qu_id_stock": 2}]) + import_receipt(receipt(), api, Journal(tmp_path / "state.sqlite3"), {}) + assert len(api.posts) == 1 + assert api.posts[0][0] == "/stock/products/9/add" + assert api.posts[0][1]["amount"] == 6 + assert api.posts[0][1]["price"] == .75 + + +def test_uncertain_write_not_retried(tmp_path): + api = FakeGrocy(fail=True) + journal = Journal(tmp_path / "state.sqlite3") + with pytest.raises(ImportError, match="Network failed"): + import_receipt(receipt(), api, journal, {}) + writes = len(api.posts) + with pytest.raises(ImportError, match="uncertain"): + import_receipt(receipt(), api, journal, {}) + assert len(api.posts) == writes + + +def test_changed_receipt_not_reimported(tmp_path): + api = FakeGrocy() + journal = Journal(tmp_path / "state.sqlite3") + import_receipt(receipt(), api, journal, {}) + changed = receipt() + changed.items[0].quantity = Decimal(3) + with pytest.raises(ImportError, match="changed"): + import_receipt(changed, api, journal, {}) + + +def test_preflight_rejects_ambiguity_before_writes(tmp_path): + api = FakeGrocy([{"id": 1, "name": "Apples"}, {"id": 2, "name": "Apples"}]) + with pytest.raises(ImportError, match="Ambiguous"): + import_receipt(receipt(), api, Journal(tmp_path / "state.sqlite3"), {}) + assert not api.posts + + +def test_same_product_two_lines_created_once(tmp_path): + api = FakeGrocy() + order = receipt() + order.items.append(order.items[0]) + import_receipt(order, api, Journal(tmp_path / "state.sqlite3"), {}) + assert len([p for p, _ in api.posts if p == "/objects/products"]) == 1 + assert len([p for p, _ in api.posts if p.endswith("/add")]) == 2 + + +@pytest.mark.parametrize('old_note', ['Ocado order 123, line 1.', 'Ocado 2026-09-08, line 1.']) +def test_note_updates_never_recreate_deleted_stock(tmp_path, old_note): + from copy import deepcopy + from ocado_grocy.grocy import update_existing_notes + class Api: + url='https://test/api' + def __init__(self): + self.rows=[{'id':10,'product_id':1,'amount':2,'price':1.25,'open':0, + 'best_before_date':'2027-01-01','purchased_date':'2026-09-08', + 'location_id':1,'shopping_location_id':1,'note':old_note}] + self.writes=[] + def get(self,path): + if path=='/stock/entry/10': + return deepcopy(self.rows[0]) + assert path=='/objects/stock' + return deepcopy(self.rows) + def request(self,method,path,data): + assert method=='PUT' and path=='/stock/entry/10' + self.writes.append(data) + self.rows[0].update(data) + journal=Journal(tmp_path/'state.sqlite3') + api=Api() + journal.record(api.url,'123',1,'existing','done',1) + journal.record(api.url,'123',2,'deleted','done',2) # Missing from live stock. + assert update_existing_notes(api,journal)==1 + assert api.rows[0]['note']=='Ocado 2026-09-08' + assert len(api.rows)==1 + assert api.rows[0]['amount']==2 + assert update_existing_notes(api,journal)==0 + assert len(api.writes)==1 + + +def test_note_update_skips_stock_deleted_after_listing(tmp_path): + from ocado_grocy.grocy import update_existing_notes + class Api: + url='https://test/api' + removed=False + def get(self,path): + if path=='/stock/entry/10': + self.removed=True + return None # Grocy returns JSON null for a deleted entry. + assert path=='/objects/stock' + return [] if self.removed else [{'id':10,'product_id':1,'note':'Ocado 2026-09-08, line 1.'}] + def request(self,*args,**kwargs): + pytest.fail('Must not write deleted stock') + journal=Journal(tmp_path/'state.sqlite3') + api=Api() + journal.record(api.url,'123',1,'existing','done',1) + assert update_existing_notes(api,journal)==0 diff --git a/tests/test_history.py b/tests/test_history.py new file mode 100644 index 0000000..c523a60 --- /dev/null +++ b/tests/test_history.py @@ -0,0 +1,140 @@ +from dataclasses import replace +from decimal import Decimal +import json +from pathlib import Path + +import pytest + +from ocado_grocy.history import order_links,private_json,cached_orders,select_orders,OrderLink +from ocado_grocy.pantry import classify +from ocado_grocy.receipt import ImportError,Item,parse_ocado_order +from ocado_grocy.grocy import (Journal,import_receipt,description,refresh_imported_products, + item_fingerprint,align_recorded_lines) +from test_grocy import FakeGrocy,receipt + +FIXTURE=Path(__file__).parent/'fixtures'/'history_receipt.json' + + +def test_storage_and_categories_select_without_a_product_list(): + order=parse_ocado_order(json.loads(FIXTURE.read_text())) + assert sum(classify(i).include for i in order.items)==9 + assert not any(classify(i).uncertain for i in order.items) + banana=next(i for i in order.items if 'Bananas' in i.name) + assert banana.metadata['storage_type']=='CUPBOARD' + assert not classify(banana).include + feta=next(i for i in order.items if 'Feta' in i.name) + assert feta.best_before.startswith('2030') + assert not classify(feta).include + + +def test_names_do_not_override_missing_metadata(): + item=Item('Shelf-stable sounding Chocolate Biscuits',Decimal(1),Decimal(1),product_id='55') + assert classify(item).uncertain + assert not classify(item).include + assert classify(item,{'55':True}).include + with pytest.raises(ValueError):classify(item,{'55':'yes'}) + + +def test_cold_storage_takes_precedence_over_category(): + item=Item('Chocolate pudding',Decimal(1),Decimal(1),metadata={ + 'storage_type':'FRIDGE','ocado':{'categoryPath':['Treats & Snacks']}}) + assert not classify(item).include + + +def test_order_links_ignore_cancelled_and_future_and_keep_year(): + links=order_links(''' + Sep 8, 2026 5:00pm£56.75Delivered + Sep 8, 2025 5:00pm View order Delivered + Sep 8, 2025 5:00pm You cancelled this order + Fri 11 Sep You're editing this order''') + assert [(r.order_id,r.purchased_date) for r in links]==[('1','2026-09-08'),('2','2025-09-08')] + + +def test_selection_does_not_change_complete_manifest(tmp_path): + links=[OrderLink('1','2026-09-08','https://www.ocado.com/orders/1/details'), + OrderLink('2','2025-09-08','https://www.ocado.com/orders/2/details')] + assert select_orders(links,limit=1)==links[:1] + assert select_orders(links,order_ids=('2',))==links[1:] + assert len(links)==2 + with pytest.raises(ImportError,match='not found'):select_orders(links,order_ids=('missing',)) + + +def test_partial_import_preserves_original_line_numbers(tmp_path): + order=receipt() + order.items.extend([Item('Milk',Decimal(1),Decimal(1)),Item('Tea',Decimal(1),Decimal(2))]) + api=FakeGrocy() + journal=Journal(tmp_path/'journal.sqlite3') + assert import_receipt(order,api,journal,{},include_lines={3})==1 + assert api.posts[-1][1]['note']=='Ocado 2026-09-08 Expiry unknown.' + assert journal.db.execute('SELECT line FROM imports').fetchone()[0]==3 + assert import_receipt(order,api,journal,{},include_lines={3})==0 + assert import_receipt(order,api,journal,{})==2 + assert journal.db.execute('SELECT COUNT(*) FROM imports').fetchone()[0]==3 + + +def test_cache_permissions_and_missing_order(tmp_path): + private_json(tmp_path/'orders.json',[{'order_id':'1','purchased_date':'2026-09-08','url':'https://www.ocado.com/orders/1/details'}]) + assert (tmp_path/'orders.json').stat().st_mode & 0o777==0o600 + orders,failures=cached_orders(tmp_path) + assert not orders and failures[0]['order_id']=='1' + + +@pytest.mark.parametrize("sanitized", [False, True]) +def test_refresh_existing_includes_perishables_without_booking_stock(tmp_path, sanitized): + order=parse_ocado_order(json.loads(FIXTURE.read_text())) + index,item=next((i,item) for i,item in enumerate(order.items,1) if 'Lettuce' in item.name) + old=replace(item,metadata={'additional_details':{'brand':'Preserved'}, 'openfoodfacts':{'code':'00000048'}}) + class Api: + url='https://test/api' + def __init__(self): + self.product={'id':1,'name':item.name,'description':description(old)} + if sanitized: + self.product['description'] = self.product['description'].replace('
', '').replace('
', '') + self.writes=[] + def get(self,path): + assert path=='/objects/products/1' + return self.product + def request(self,method,path,data): + assert method=='PUT' and path=='/objects/products/1' + self.writes.append(data) + self.product.update(data) + api=Api() + journal=Journal(tmp_path/'state.sqlite3') + journal.record(api.url,order.order_id,index,item_fingerprint(order,item),'done',1) + assert not classify(item).include + assert refresh_imported_products(order,api,journal)==1 + assert 'FRIDGE' in api.product['description'] + assert 'Preserved' in api.product['description'] + assert '00000048' in api.product['description'] + assert refresh_imported_products(order,api,journal)==0 + assert len(api.writes)==1 + + +def test_reordered_receipt_recovers_original_journal_lines(tmp_path): + original=receipt() + original.items.append(Item('Tea',Decimal(1),Decimal(2))) + api=FakeGrocy() + journal=Journal(tmp_path/'state.sqlite3') + import_receipt(original,api,journal,{}) + reordered=replace(original,items=list(reversed(original.items))) + restored=align_recorded_lines(reordered,api.url,journal) + assert restored.items==original.items + assert import_receipt(restored,api,journal,{})==0 + + +@pytest.mark.parametrize("flags,expected", [([],21), (["--all-products"],21), (["--shelf-stable-only"],9)]) +def test_cached_cli_preview_does_not_write_stock_or_use_browser(tmp_path,monkeypatch,flags,expected): + from click.testing import CliRunner + from ocado_grocy.cli import main + order=parse_ocado_order(json.loads(FIXTURE.read_text())) + private_json(tmp_path/'orders.json',[{'order_id':order.order_id,'purchased_date':order.purchased_date, + 'url':f'https://www.ocado.com/orders/{order.order_id}/details'}]) + private_json(tmp_path/f'{order.order_id}.json',order.export()) + monkeypatch.setattr('requests.Session.request',lambda *a,**kw:pytest.fail('Unexpected Grocy request')) + monkeypatch.setattr('ocado_grocy.history.download_orders',lambda *a,**kw:pytest.fail('Unexpected browser')) + result=CliRunner().invoke(main,flags+['--cached','--dry-run','--archive',str(tmp_path), + '--config',str(tmp_path/'missing-config.toml')]) + assert result.exit_code==0,result.output + report=json.loads((tmp_path/'report.json').read_text()) + assert sum(i['include'] for i in report['items'])==expected + assert report['imported_lines']==0 diff --git a/tests/test_images.py b/tests/test_images.py new file mode 100644 index 0000000..9b5b036 --- /dev/null +++ b/tests/test_images.py @@ -0,0 +1,87 @@ +import json +from types import SimpleNamespace +import pytest + +from ocado_grocy.images import add_picture, file_path, image_type + + +DATA = b'\xff\xd8\xfftest-image' + + +def product(): + return {'id':72, 'name':'Ocado title', 'description':'

Ocado product ID: 123

' + json.dumps({'Ocado product ID':'123','image_url':'https://www.ocado.com/images-v3/test.jpg'}), 'picture_file_name':None} + + +def response(data=DATA): + return SimpleNamespace(status_code=200, content=data, raise_for_status=lambda:None) + + +class API: + url = 'https://grocy.test/api' + timeout = 30 + def __init__(self): + self.product = product() + self.writes = [] + self.stored = DATA + self.session = self + def get(self, path, **kwargs): + if path.startswith(self.url): + assert '/files/productpictures/' in path + return response(self.stored) + assert path == '/objects/products/72' + return self.product.copy() + def put(self, path, **kwargs): + assert path.startswith(self.url + '/files/productpictures/') + assert kwargs['data'] == DATA + assert kwargs['headers'] == {'Content-Type':'application/octet-stream'} + self.writes.append(path) + return response() + def request(self, method, path, payload): + assert method == 'PUT' and path == '/objects/products/72' + assert set(payload) == {'picture_file_name'} + self.writes.append(payload) + self.product.update(payload) + + +class Downloader: + def get(self, url, **kwargs): + assert url == 'https://www.ocado.com/images-v3/test.jpg' + assert set(kwargs) == {'timeout','allow_redirects'} + return response() + + +def test_upload_verifies_file_and_only_sets_picture_field(): + api = API() + assert add_picture(api, Downloader(), product()) == 'updated' + assert len(api.writes) == 2 + assert api.product['name'] == 'Ocado title' + assert api.product['description'] == product()['description'] + assert add_picture(api, Downloader(), api.product) == 'existing_picture' + assert len(api.writes) == 2 + + +def test_failed_file_verification_does_not_assign_picture(): + api = API() + api.stored = b'wrong file' + with pytest.raises(ValueError, match='verification'): + add_picture(api, Downloader(), product()) + assert api.product['picture_file_name'] is None + + +def test_concurrent_picture_or_deleted_product_is_preserved(): + api = API() + api.product['picture_file_name'] = 'user-picture.jpg' + assert add_picture(api, Downloader(), product()) == 'existing_picture' + api.product = {} + assert add_picture(api, Downloader(), product()) == 'deleted_product' + assert not api.writes + + +def test_rejects_non_images_and_external_urls(): + with pytest.raises(ValueError): + image_type(b'not an image') + p = product() + p['description'] = p['description'].replace('www.ocado.com', 'example.com') + with pytest.raises(ValueError, match='Ocado image URL'): + add_picture(API(), Downloader(), p) + assert file_path('test.jpg') == '/files/productpictures/dGVzdC5qcGc%3D' diff --git a/tests/test_openfoodfacts.py b/tests/test_openfoodfacts.py new file mode 100644 index 0000000..6e3c14c --- /dev/null +++ b/tests/test_openfoodfacts.py @@ -0,0 +1,139 @@ +import json +import pytest +from ocado_grocy.openfoodfacts import candidate_info, pack, valid_barcode, apply_match +from ocado_grocy.product_metadata import read_metadata, replace_metadata + + +def metadata(): + return {'Ocado product ID':'123', 'Barcode':'', 'ocado':{'brand':'M&S', 'size':{'value':'250g'}}, 'custom':'keep me'} + + +def product(): + return {'id':42, 'name':'M&S Ginger Snaps', 'qu_id_purchase':1, 'description':'

Ocado product ID: 123

' + json.dumps(metadata()) + '

User comment

'} + + +def candidate(): + return {'code':'00000048', 'product_name':'Ginger Snaps', 'brands':'Marks & Spencer', 'quantity':'250 g', 'ingredients_text':'Ginger & flour', 'nutriments':{'energy-kcal_100g':450}} + + +def test_matching_rejects_variant_brand_and_pack_changes(): + assert candidate_info(product(), metadata(), candidate())['exact'] + for change in ({'quantity':'300 g'}, {'quantity':'2 x 125g'}, {'brands':'Tesco'}, {'product_name':'Reduced Sugar Ginger Snaps'}, {'code':'00000041'}): + assert not candidate_info(product(), metadata(), {**candidate(), **change})['exact'] + assert pack('1.25L') == pack('1250ml') + assert pack('250g') != pack('250ml') + assert pack('5 per pack') is None + assert not valid_barcode('123456011') + + +def test_metadata_survives_sanitizer_and_retains_user_comments(): + p = product() + m = read_metadata(p['description']) + m['openfoodfacts'] = {'ingredients':'Ginger & flour < 1%'} + updated = replace_metadata(p['description'], m) + assert read_metadata(updated) == m + assert '

User comment

' in updated + assert 'Ginger & flour < 1%' in updated + + +class API: + def __init__(self): + self.p = product() + self.writes = [] + def get(self, path): + assert path == '/objects/products/42' + return self.p.copy() + def post(self, path, data): + assert path == '/objects/product_barcodes' + self.writes.append((path, data)) + def request(self, method, path, data): + assert method == 'PUT' and path == '/objects/products/42' + assert set(data) == {'description'} + self.writes.append((path, data)) + self.p.update(data) + + +def test_enrichment_is_idempotent_and_never_changes_stock_or_title(): + api = API() + barcodes = [] + assert apply_match(api, product(), candidate(), barcodes) == 'updated' + assert apply_match(api, product(), candidate(), barcodes) == 'unchanged' + assert len(api.writes) == 2 + assert api.p['name'] == product()['name'] + assert read_metadata(api.p['description'])['custom'] == 'keep me' + assert read_metadata(api.p['description'])['openfoodfacts']['nutriments']['energy-kcal_100g'] == 450 + + +def test_conflicting_barcode_and_deleted_product_do_not_write(): + api = API() + with pytest.raises(ValueError, match='another Grocy product'): + apply_match(api, product(), candidate(), [{'barcode':'00000048','product_id':99}]) + api.p = {} + with pytest.raises(ValueError, match='deleted'): + apply_match(api, product(), candidate(), []) + assert not api.writes + + +def test_two_exact_barcodes_are_ambiguous(): + from ocado_grocy.openfoodfacts import choose_match + first = candidate() + second = {**first, 'code':'00000055'} + ranked, match = choose_match(product(), metadata(), {first['code']:first, second['code']:second}) + assert len(ranked) == 2 + assert match is None + + +def test_truncated_catalog_is_split_and_pages_must_be_complete(tmp_path): + from ocado_grocy.openfoodfacts import OpenFoodFacts + class OFF(OpenFoodFacts): + def fetch(self, path, params, **kwargs): + if params['q'] == 'brands_tags:(a OR b)': + return {'is_count_exact':False, 'hits':[]} + code = '00000048' if params['q'] == 'brands_tags:(a)' else '00000055' + return {'is_count_exact':True, 'count':1, 'page_count':1, 'hits':[{'code':code,'brands':['a']}]} + off = OFF(tmp_path, 'test@example.com') + assert len(off.catalog({'a','b'})) == 2 + off.fetch = lambda *args, **kwargs: {'is_count_exact':True, 'count':2, 'page_count':1, 'hits':[]} + with pytest.raises(ValueError, match='Incomplete'): + off.catalog({'a'}) + + +def test_ms_food_brand_and_missing_barcode_recovery(): + assert candidate_info(product(), metadata(), {**candidate(), 'brands':'M&S Food'})['exact'] + api = API() + barcodes = [] + apply_match(api, product(), candidate(), barcodes) + # Simulate an interrupted attempt with metadata present but no barcode row. + assert apply_match(api, api.p.copy(), candidate(), []) == 'updated' + assert len([p for p, data in api.writes if p == '/objects/products/42']) == 1 + + +def test_established_barcode_match_survives_a_missing_catalog_entry(tmp_path, monkeypatch): + from ocado_grocy import openfoodfacts as module + p = product() + m = metadata() + m.update(Barcode=candidate()['code'], openfoodfacts=candidate()) + p['description'] = replace_metadata(p['description'], m) + class OFF: + def __init__(self, *a): pass + def catalog(self, *a): return [] + def product(self, code, offline): + assert code == candidate()['code'] + return candidate() + class Client: + def get(self, path): + assert path == '/objects/product_barcodes' + return [] + monkeypatch.setattr(module, 'OpenFoodFacts', OFF) + result = module.run_openfoodfacts(Client(), [p], tmp_path, 'test@example.com', apply=False) + assert result['products'][0]['match_method'] == 'existing_match' + assert result['products'][0]['status'] == 'matched' + + +def test_review_ranks_the_right_name_ahead_of_unrelated_matching_pack(): + from ocado_grocy.openfoodfacts import choose_match + missing_size = {**candidate(), 'quantity':''} + unrelated = {**candidate(), 'code':'00000055', 'product_name':'Tomato Ketchup'} + ranked, matched = choose_match(product(), metadata(), {c['code']:c for c in [missing_size, unrelated]}) + assert ranked[0]['code'] == missing_size['code'] + assert matched is None diff --git a/tests/test_order_state.py b/tests/test_order_state.py new file mode 100644 index 0000000..bb5fd2a --- /dev/null +++ b/tests/test_order_state.py @@ -0,0 +1,58 @@ +from dataclasses import replace + +from click.testing import CliRunner + +from ocado_grocy.grocy import Journal, item_fingerprint +from ocado_grocy.history import OrderLink, main, private_json, select_orders +from ocado_grocy.order_state import completed_orders, mark_orders +from test_grocy import receipt + + +def test_completed_filter_precedes_limit_and_explicit_order_bypasses_it(): + links = [OrderLink(str(i),f'2026-09-{i:02d}','url') for i in (1,2,3)] + assert [r.order_id for r in select_orders(links,limit=1,completed={'3'})] == ['2'] + assert [r.order_id for r in select_orders(links,order_ids=('3',),completed={'3'})] == ['3'] + + +def test_legacy_migration_requires_all_selected_lines_and_never_just_a_cache(tmp_path): + state = tmp_path/'imports.sqlite3' + order = receipt() + item = replace(order.items[0], metadata={'ocado':{'categoryPath':['Food Cupboard']}}) + order = replace(order,items=[item,replace(item,name='Second product')]) + private_json(tmp_path/'123.json',order.export()) + journal = Journal(state) + assert completed_orders(state,'server',tmp_path,{}) == set() + journal.record('server','123',1,item_fingerprint(order,item),'done',1) + assert completed_orders(state,'server',tmp_path,{}) == set() + journal.record('server','123',2,item_fingerprint(order,order.items[1]),'pending',2) + assert completed_orders(state,'server',tmp_path,{}) == set() + journal.record('server','123',2,item_fingerprint(order,order.items[1]),'done',2) + assert completed_orders(state,'server',tmp_path,{}) == {'123'} + assert completed_orders(state,'other-server',tmp_path,{}) == set() + # Enrichment failure must override legacy done stock lines. + mark_orders(state,'server',['123'],'processing') + assert completed_orders(state,'server',tmp_path,{}) == set() + mark_orders(state,'server',['123'],'complete') + assert completed_orders(state,'server',tmp_path,{}) == {'123'} + journal.db.close() + + +def test_empty_selection_completion_and_read_only_missing_state(tmp_path): + state = tmp_path/'imports.sqlite3' + assert completed_orders(state,'server',tmp_path,{}) == set() + assert not state.exists() + mark_orders(state,'server',['123'],'complete') + assert completed_orders(state,'server',tmp_path,{}) == {'123'} + + +def test_default_cached_run_skips_receipt_loading_and_all_processing(tmp_path, monkeypatch): + from ocado_grocy import history + private_json(tmp_path/'orders.json',[{'order_id':'123','purchased_date':'2026-09-08','url':'url'}]) + # No receipt file: skipping must happen before trying to load it. + mark_orders(tmp_path/'imports.sqlite3','/api',['123'],'complete') + def unexpected(*args, **kwargs): + raise AssertionError('Completed order was processed again') + monkeypatch.setattr(history,'run_import',unexpected) + result = CliRunner().invoke(main,['--cached','--archive',str(tmp_path),'--config',str(tmp_path/'config.toml')]) + assert result.exit_code == 0, result.output + assert 'No new delivered orders' in result.output diff --git a/tests/test_postprocess.py b/tests/test_postprocess.py new file mode 100644 index 0000000..1cc0ee7 --- /dev/null +++ b/tests/test_postprocess.py @@ -0,0 +1,84 @@ +import sqlite3 +from pathlib import Path + +from click.testing import CliRunner +import pytest + +from ocado_grocy.grocy import imported_products +from ocado_grocy.history import main +from ocado_grocy import postprocess + + +def test_scope_uses_completed_lines_for_selected_order_and_server(tmp_path): + state = tmp_path / 'journal.sqlite3' + with sqlite3.connect(state) as db: + db.execute('CREATE TABLE imports(server,order_id,status,product_id)') + db.executemany('INSERT INTO imports VALUES (?,?,?,?)', [ + ('server','latest','done',1), ('server','older','done',2), + ('server','latest','pending',3), ('other','latest','done',4), + ('server','latest','done',5)]) + class Client: + url = 'server' + def get(self, path): + assert path == '/objects/products' + return [{'id':i} for i in range(1,5)] # 5 was deleted + assert imported_products(Client(), state, ['latest']) == [{'id':1}] + + +@pytest.mark.parametrize('dry_run', [False, True]) +def test_main_runs_enrichment_even_when_no_stock_added_but_never_in_dry_run(tmp_path, monkeypatch, dry_run): + from ocado_grocy import history + from test_grocy import receipt + monkeypatch.setattr(history, 'cached_orders', lambda *a: ([receipt()], [])) + monkeypatch.setattr(history, 'run_import', lambda *a, **kw: {'items':[], 'imported_lines':0}) + calls = [] + monkeypatch.setattr(postprocess, 'enrich_orders', lambda *a, **kw: calls.append((a,kw)) or {'errors':[]}) + args = ['--cached','--archive',str(tmp_path),'--config',str(tmp_path/'config.toml')] + if dry_run: + args.append('--dry-run') + result = CliRunner().invoke(main, args) + assert result.exit_code == 0, result.output + assert len(calls) == (0 if dry_run else 1) + if calls: + assert calls[0][0][3] == ['123'] + assert calls[0][1] == {'images':True,'openfoodfacts':True} + + +def test_picture_failure_does_not_prevent_off_and_errors_are_recorded(tmp_path, monkeypatch): + from types import SimpleNamespace + client = SimpleNamespace(url='server') + products = [{'id':42}] + monkeypatch.setattr(postprocess, 'Grocy', lambda *a: client) + monkeypatch.setattr(postprocess, 'imported_products', lambda *a: products) + def fail_images(*args, **kwargs): + raise ValueError('image unavailable') + monkeypatch.setattr(postprocess, 'run_images', fail_images) + calls = [] + def off(api, selected, cache, contact): + assert api is client and selected == products + calls.append(True) + return {'counts':{'unchanged':1}, 'products':[]} + monkeypatch.setattr(postprocess, 'run_openfoodfacts', off) + result = postprocess.enrich_orders({'openfoodfacts':{'contact':'test@example.com'}}, tmp_path/'config.toml',tmp_path,['123']) + assert calls and result['openfoodfacts'] == {'unchanged':1} + assert result['errors'] == ['Images: image unavailable'] + assert (tmp_path/'enrichment/123/summary.json').exists() + + +def test_disabled_enrichment_does_not_access_grocy(tmp_path, monkeypatch): + monkeypatch.setattr(postprocess, 'Grocy', lambda *a: (_ for _ in ()).throw(AssertionError('Unexpected API access'))) + assert postprocess.enrich_orders({}, tmp_path/'config.toml',tmp_path,['123'],images=False,openfoodfacts=False) == {'errors':[]} + + +def test_enrichment_failure_reports_stock_completion_and_nonzero_exit(tmp_path, monkeypatch): + from ocado_grocy import history + from test_grocy import receipt + monkeypatch.setattr(history, 'cached_orders', lambda *a: ([receipt()], [])) + monkeypatch.setattr(history, 'run_import', lambda *a, **kw: {'items':[], 'imported_lines':1}) + monkeypatch.setattr(postprocess, 'enrich_orders', lambda *a, **kw: {'errors':['OFF unavailable']}) + result = CliRunner().invoke(main, ['--cached','--archive',str(tmp_path),'--config',str(tmp_path/'config.toml')]) + assert result.exit_code == 1 + assert 'Stock import completed; enrichment needs a retry' in result.output + import json + saved = json.loads((tmp_path/'report.json').read_text()) + assert saved['imported_lines'] == 1 and saved['enrichment']['errors'] == ['OFF unavailable'] diff --git a/tests/test_receipt.py b/tests/test_receipt.py new file mode 100644 index 0000000..6ca9cdd --- /dev/null +++ b/tests/test_receipt.py @@ -0,0 +1,118 @@ +from copy import deepcopy +from decimal import Decimal +import json +from pathlib import Path + +import pytest + +from ocado_grocy.receipt import ImportError, number, parse_document, parse_ocado_order + +FIXTURE = Path(__file__).parent / 'fixtures' / 'history_receipt.json' + + +def api_data(): + return json.loads(FIXTURE.read_text()) + + +def test_synthetic_api_receipt(): + receipt = parse_ocado_order(api_data()) + assert receipt.order_id == '1000000000001' + assert receipt.purchased_date == '2020-01-02' + assert len(receipt.items) == 21 + assert sum(i.quantity for i in receipt.items) == 23 + assert sum(i.total for i in receipt.items) == Decimal('28.75') + lettuce = next(i for i in receipt.items if 'Lettuce' in i.name) + assert lettuce.best_before == '2020-01-05' + assert lettuce.metadata['storage_type'] == 'FRIDGE' + cloths = next(i for i in receipt.items if 'Cleaning Cloths' in i.name) + assert cloths.quantity == 2 + assert cloths.unit_price == Decimal('1.25') + assert cloths.best_before is None + + +def test_order_identity_and_delivery_status(): + with pytest.raises(ImportError,match='does not match'): + parse_ocado_order(api_data(),expected_order_id='123') + data=api_data() + data['entities']['order'][data['result']]['status']='CONFIRMED' + with pytest.raises(ImportError,match='not delivered'): + parse_ocado_order(data) + + +def test_summary_detects_partial_receipt(): + data=api_data() + data['entities']['order'][data['result']]['groupedProducts']['products'].pop() + with pytest.raises(ImportError,match='quantities'): + parse_ocado_order(data) + + +def test_summary_detects_wrong_price(): + data=api_data() + data['entities']['order'][data['result']]['groupedProducts']['products'][0]['prices']['offered']['amount']='9.99' + with pytest.raises(ImportError,match='prices'): + parse_ocado_order(data) + + +def test_only_accepted_substitutes_are_imported(): + data=api_data() + order=data['entities']['order'][data['result']] + original=order['groupedProducts']['products'].pop(0) + accepted={**deepcopy(original),'productId':'replacement','retailerProductId':'999','name':'Replacement bananas','status':'ACCEPTED'} + rejected={**deepcopy(original),'productId':'rejected','retailerProductId':'888','name':'Rejected bananas','status':'REJECTED'} + original['substitutes']=[accepted,rejected] + order['groupedProducts']['substitutes']=[original] + receipt=parse_ocado_order(data) + assert len(receipt.items)==21 + assert any(i.product_id=='999' for i in receipt.items) + assert not any(i.product_id in {'888',original['retailerProductId']} for i in receipt.items) + assert next(i for i in receipt.items if i.product_id=='999').metadata['substituted_for']==original['name'] + + +def test_input_order_does_not_change_stable_line_order(): + first=api_data() + second=deepcopy(first) + second['entities']['order'][second['result']]['groupedProducts']['products'].reverse() + assert parse_ocado_order(first)==parse_ocado_order(second) + + +def test_discontinued_product_does_not_need_html_link(): + data=api_data() + product=data['entities']['order'][data['result']]['groupedProducts']['products'][0] + product['isInCurrentCatalog']=False + receipt=parse_ocado_order(data) + assert any(i.product_id==product['retailerProductId'] for i in receipt.items) + + +def test_purchase_date_is_local_delivery_date(): + data=api_data() + data['entities']['order'][data['result']]['dates']['deliveryStartDate']='2026-09-01T23:30:00Z' + assert parse_ocado_order(data).purchased_date=='2026-09-02' + + +@pytest.mark.parametrize('value',['NaN','Infinity','-1','nonsense',None]) +def test_invalid_numbers(value): + with pytest.raises(ImportError):number(value,'quantity') + + +def test_free_and_weighted_items(): + receipt=parse_document({'order_id':'1','items':[ + {'name':'Free gift','quantity':1,'total':0}, + {'name':'Apples','quantity':'0.75','total':'1.50'}]}) + assert receipt.items[0].unit_price==0 + assert receipt.items[1].unit_price==2 + + +def test_receipt_json_roundtrip(): + receipt=parse_ocado_order(api_data()) + assert parse_document(json.loads(json.dumps(receipt.export(),default=str)))==receipt + + +def test_live_basket_cannot_be_mistaken_for_order(): + with pytest.raises(ImportError,match='structure'): + parse_ocado_order({'data':{'basket':{'items':[{'name':'Not an order'}]}}}) + + +def test_non_gbp_order_rejected(): + data=api_data() + data['entities']['order'][data['result']]['groupedProducts']['products'][0]['prices']['offered']['currency']='EUR' + with pytest.raises(ImportError,match='currency'):parse_ocado_order(data)