Fix bot detection causing blank page from Goodreads

Goodreads returns an empty HTML page to headless browsers it identifies
as automated (navigator.webdriver=true, AutomationControlled feature).
This caused the Export Library button to never be found.

- Launch Chromium with --disable-blink-features=AutomationControlled
- Set a realistic user agent and viewport on the browser context
- Strip navigator.webdriver via an init script
- Guard the logged_in check so a blank page isn't mistaken for a session
- Wait for networkidle before querying the export button, and use
  wait_for(visible) instead of is_visible() to tolerate JS render delay
- Add CSS class fallback (button.js-LibraryExport) if role lookup fails
- Add AGENTS.md documenting the bot detection pattern and export flow

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
Edward Betts 2026-02-18 11:53:42 +00:00
parent ae932b26b9
commit 56dff875e7
2 changed files with 88 additions and 18 deletions

View file

@ -44,11 +44,23 @@ def run_backup(page: Page) -> None:
"""Run backup."""
import_url = "https://www.goodreads.com/review/import"
page.goto(import_url)
page.wait_for_load_state("networkidle")
if refresh_backup:
print("backup requested")
export_button = page.get_by_role("button", name="Export Library")
if export_button.count() > 0 and export_button.first.is_visible():
# Fall back to CSS class in case accessible name lookup fails
if export_button.count() == 0:
export_button = page.locator("button.js-LibraryExport")
found = False
if export_button.count() > 0:
try:
export_button.first.wait_for(state="visible", timeout=10 * 1000)
found = True
except TimeoutError:
pass
if found:
export_button.first.click()
print("waiting for export to be ready...")
@ -105,27 +117,37 @@ def run(
mode: str = "backup",
) -> int:
"""Download export."""
browser = playwright.chromium.launch(headless=headless)
browser = playwright.chromium.launch(
headless=headless,
args=["--disable-blink-features=AutomationControlled"],
)
# Spoof user agent and viewport to look like a regular desktop browser.
context_kwargs: dict = dict(
user_agent=(
"Mozilla/5.0 (X11; Linux x86_64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/131.0.0.0 Safari/537.36"
),
viewport={"width": 1280, "height": 800},
)
auth_json = os.path.join(script_dir, "auth.json")
if has_auth(auth_json) and har_path:
context = browser.new_context(
storage_state=auth_json,
record_har_path=har_path,
record_har_mode="minimal",
)
elif has_auth(auth_json):
context = browser.new_context(storage_state=auth_json)
elif har_path:
context = browser.new_context(
record_har_path=har_path,
record_har_mode="minimal",
)
else:
context = browser.new_context()
if has_auth(auth_json):
context_kwargs["storage_state"] = auth_json
if har_path:
context_kwargs["record_har_path"] = har_path
context_kwargs["record_har_mode"] = "minimal"
context = browser.new_context(**context_kwargs)
# Remove the navigator.webdriver flag that headless Chrome exposes.
context.add_init_script(
"Object.defineProperty(navigator, 'webdriver', {get: () => undefined})"
)
page = context.new_page()
page.goto("https://www.goodreads.com/")
logged_in = not page.get_by_role("link", name="Sign In").is_visible()
page.wait_for_load_state("domcontentloaded")
# A blank page (bot-detection response) has no body content; treat as not logged in.
page_has_content = page.locator("body").inner_text().strip() != ""
logged_in = page_has_content and not page.get_by_role("link", name="Sign In").is_visible()
if mode == "check_login":
print("logged in" if logged_in else "not logged in")