diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 0000000..1bdcfaf --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,48 @@ +# Agent Notes + +## Bot detection (blank page symptom) + +Goodreads returns `` to headless browsers +it identifies as automated. Symptoms: +- `page.content()` → bare empty HTML skeleton +- `page.title()` → empty string +- Screenshot is fully white +- `.exportBooks` div count is 0 + +**Root cause:** Playwright's default headless Chromium exposes `navigator.webdriver = true` +and announces itself via the `AutomationControlled` Blink feature. + +**Fix applied:** +- Launch with `--disable-blink-features=AutomationControlled` +- Set a realistic `user_agent` and `viewport` on the context +- `context.add_init_script(...)` to set `navigator.webdriver = undefined` +- Guard the `logged_in` check: a blank page must not be treated as "logged in" + +## Goodreads Import/Export page (`/review/import`) + +The "Export Library" button is present in the static HTML as: +```html + +``` + +### Why `get_by_role("button", name="Export Library")` can fail + +`page.goto()` only waits for the `load` event. JavaScript may still be running +(React hydration, etc.) when the button is queried, so `count()` can return 0 +or `is_visible()` can return `False` even though the button exists in the DOM. + +**Fix applied:** call `page.wait_for_load_state("networkidle")` after `goto`, +then use `wait_for(state="visible", timeout=10s)` instead of `is_visible()`. +A CSS-class fallback (`button.js-LibraryExport`) is also tried in case the +accessible-name lookup fails. + +### Export flow + +1. Click "Export Library" → Goodreads POSTs to `/review_porter/export/` +2. The `#exportFile` div is cleared while generation is in progress. +3. When done, a download link reappears inside `#exportFile a`. +4. The CSV is fetched via `/review_porter/export//goodreads_export.csv`. + Expect repeated 404s before the 200 arrives (Goodreads generates it async). diff --git a/backup.py b/backup.py index 2bb98ba..7698cb6 100755 --- a/backup.py +++ b/backup.py @@ -44,11 +44,23 @@ def run_backup(page: Page) -> None: """Run backup.""" import_url = "https://www.goodreads.com/review/import" page.goto(import_url) + page.wait_for_load_state("networkidle") + if refresh_backup: print("backup requested") export_button = page.get_by_role("button", name="Export Library") - if export_button.count() > 0 and export_button.first.is_visible(): + # Fall back to CSS class in case accessible name lookup fails + if export_button.count() == 0: + export_button = page.locator("button.js-LibraryExport") + found = False + if export_button.count() > 0: + try: + export_button.first.wait_for(state="visible", timeout=10 * 1000) + found = True + except TimeoutError: + pass + if found: export_button.first.click() print("waiting for export to be ready...") @@ -105,27 +117,37 @@ def run( mode: str = "backup", ) -> int: """Download export.""" - browser = playwright.chromium.launch(headless=headless) + browser = playwright.chromium.launch( + headless=headless, + args=["--disable-blink-features=AutomationControlled"], + ) + # Spoof user agent and viewport to look like a regular desktop browser. + context_kwargs: dict = dict( + user_agent=( + "Mozilla/5.0 (X11; Linux x86_64) " + "AppleWebKit/537.36 (KHTML, like Gecko) " + "Chrome/131.0.0.0 Safari/537.36" + ), + viewport={"width": 1280, "height": 800}, + ) auth_json = os.path.join(script_dir, "auth.json") - if has_auth(auth_json) and har_path: - context = browser.new_context( - storage_state=auth_json, - record_har_path=har_path, - record_har_mode="minimal", - ) - elif has_auth(auth_json): - context = browser.new_context(storage_state=auth_json) - elif har_path: - context = browser.new_context( - record_har_path=har_path, - record_har_mode="minimal", - ) - else: - context = browser.new_context() + if has_auth(auth_json): + context_kwargs["storage_state"] = auth_json + if har_path: + context_kwargs["record_har_path"] = har_path + context_kwargs["record_har_mode"] = "minimal" + context = browser.new_context(**context_kwargs) + # Remove the navigator.webdriver flag that headless Chrome exposes. + context.add_init_script( + "Object.defineProperty(navigator, 'webdriver', {get: () => undefined})" + ) page = context.new_page() page.goto("https://www.goodreads.com/") - logged_in = not page.get_by_role("link", name="Sign In").is_visible() + page.wait_for_load_state("domcontentloaded") + # A blank page (bot-detection response) has no body content; treat as not logged in. + page_has_content = page.locator("body").inner_text().strip() != "" + logged_in = page_has_content and not page.get_by_role("link", name="Sign In").is_visible() if mode == "check_login": print("logged in" if logged_in else "not logged in")