Add conference detail pages and cached flight searches

This commit is contained in:
Edward Betts 2026-10-02 10:46:26 +01:00
parent 6b253f2d04
commit dcaf723336
24 changed files with 12198 additions and 12 deletions

9053
agenda/airport_index.json Normal file

File diff suppressed because it is too large Load diff

147
agenda/airport_lookup.py Normal file
View file

@ -0,0 +1,147 @@
"""Local airport names from OurAirports' public-domain data (1 October 2026)."""
import json
import typing
import unicodedata
from functools import lru_cache
from pathlib import Path
import pycountry
import yaml
from agenda.types import StrDict
def normalized(value: str) -> str:
"""Match names regardless of case or accents."""
return "".join(
c
for c in unicodedata.normalize("NFKD", value.casefold())
if not unicodedata.combining(c)
).strip()
@lru_cache(maxsize=1)
def airports() -> dict[str, list[typing.Any]]:
"""Load the bundled index once; no network access during autocomplete."""
return typing.cast(
dict[str, list[typing.Any]],
json.loads(Path(__file__).with_name("airport_index.json").read_text()),
)
@lru_cache(maxsize=1)
def airport_groups() -> dict[tuple[str, str], list[str]]:
"""Named cross-border groups from the conference location mappings."""
mapping: dict[str, str | list[str]] = json.loads(
Path(__file__).with_name("conference_airports.json").read_text()
)
return {
(country.upper(), normalized(city)): codes
for key, codes in mapping.items()
for country, city in [key.split(":", 1)]
if isinstance(codes, list)
}
def code_list(query: str) -> tuple[str, ...]:
"""Recognize one or more exact IATA codes, without fuzzy matching."""
codes = tuple(dict.fromkeys(part.strip().upper() for part in query.split(",")))
return (
codes
if all(len(code) == 3 and code.isascii() and code.isalpha() for code in codes)
else ()
)
def preferred_names(personal_data: str | None) -> dict[str, str]:
"""Read personal airport names afresh so edits appear without a reload."""
if personal_data is None:
return {}
path = Path(personal_data) / "airports.yaml"
if not path.exists():
return {}
records: dict[str, StrDict] = yaml.safe_load(path.read_text()) or {}
names = {}
for key, record in records.items():
code = str(record.get("iata") or key).upper()
name = record.get("name")
if isinstance(name, str) and name.strip():
names[code] = name.strip()
return names
def group_match(
codes: tuple[str, ...], preferred: dict[str, str] | None = None
) -> StrDict | None:
"""Represent multiple airports as one selectable destination."""
names = []
for code in codes:
record = airports().get(code)
if code == "LON":
names.append("London (all airports)")
elif record:
names.append((preferred or {}).get(code, str(record[0])))
else:
return None
return {"code": ",".join(codes), "name": " / ".join(names)}
def matches(query: str, personal_data: str | None = None) -> list[StrDict]:
"""Prefer scheduled airports serving an exact city; accept country suffixes."""
query = query.strip()
preferred = preferred_names(personal_data)
codes = code_list(query)
if codes:
match = group_match(codes, preferred)
return [match] if match else []
city, _, suffix = query.partition(",")
needle = normalized(city)
if not needle:
return []
country: str | None = None
if suffix.strip():
try:
subdivision = typing.cast(typing.Any, pycountry.subdivisions).get(
code="US-" + suffix.strip().upper()
)
country = (
"US"
if subdivision
else str(pycountry.countries.lookup(suffix.strip()).alpha_2)
)
except LookupError:
pass
for (group_country, group_city), group_codes in airport_groups().items():
if needle == group_city and country in (None, group_country):
match = group_match(tuple(group_codes), preferred)
if match:
return [match]
candidates: list[tuple[tuple[int, int, int, str], StrDict]] = []
for code, record in airports().items():
name, municipality, iso_country, scheduled = record
if country and iso_country != country:
continue
name_text, city_text = normalized(name), normalized(municipality)
display_name = preferred.get(code, str(name))
preferred_text = normalized(display_name)
if (
needle not in name_text
and needle not in city_text
and needle not in preferred_text
):
continue
rank = (
int(not scheduled),
int(city_text != needle),
int(
not (name_text.startswith(needle) or preferred_text.startswith(needle))
),
code,
)
candidates.append((rank, {"code": code, "name": display_name}))
candidates.sort(key=lambda item: item[0])
result = [record for _, record in candidates[:10]]
if needle == "london" and country in (None, "GB"):
result.insert(0, {"code": "LON", "name": "London (all airports)"})
return result[:10]

View file

@ -0,0 +1,46 @@
{
"be:brussels": "BRU",
"de:berlin": "BER",
"de:hamburg": "HAM",
"de:heidelberg": "FRA",
"de:karlsruhe": "STR",
"de:chemnitz": "DRS",
"de:mildenberg": "BER",
"es:barcelona": "BCN",
"it:bologna": [
"BLQ",
"VRN",
"VCE"
],
"cy:limassol": "LCA",
"al:albania": "TIA",
"dk:funen": [
"BLL",
"CPH"
],
"ca:vancouver": "YVR",
"co:bogotá": "BOG",
"cl:santiago": "SCL",
"us:pasadena, california": "LAX",
"us:petaluma, california": "SFO",
"us:st. louis, mo": "STL",
"us:seattle": "SEA",
"us:doe bay resort & spa, orcas island, wa": "SEA",
"jp:asahikawa, hokkaido": "AKJ",
"se:malmö": [
"MMX",
"CPH"
],
"ch:bern": [
"BRN",
"BSL"
],
"gb:belfast": [
"BFS",
"BHD"
],
"de:bonn": [
"CGN",
"DUS"
]
}

110
agenda/conference_detail.py Normal file
View file

@ -0,0 +1,110 @@
"""Conference identities and airport suggestions, without network lookups."""
import json
import math
import re
import unicodedata
from pathlib import Path
import yaml
from agenda import airport_lookup
from agenda.types import StrDict
EUROPE = frozenset(
"al ad at be ba bg hr cy cz dk ee fi fr de gr hu is ie it lv li lt lu mt "
"md mc me nl mk no pl pt ro sm rs sk si es se ch tr ua gb va".split()
)
def conference_slug(conf: StrDict) -> str:
"""Use a readable title or series and location; the URL also includes a date."""
title = str(conf.get("series") or conf["name"])
ascii_title = (
unicodedata.normalize("NFKD", title).encode("ascii", "ignore").decode()
)
slug = re.sub(r"[^a-z0-9]+", "-", ascii_title.lower()).strip("-") or "conference"
location = (
unicodedata.normalize("NFKD", str(conf.get("location", "")))
.encode("ascii", "ignore")
.decode()
)
location_slug = re.sub(r"[^a-z0-9]+", "-", location.lower()).strip("-")
return f"{slug}-{location_slug}" if location_slug else slug
def distance_km(lat: float, lon: float, other_lat: float, other_lon: float) -> float:
"""Great-circle distance."""
a, b = math.radians(lat), math.radians(other_lat)
dlat, dlon = b - a, math.radians(other_lon - lon)
hav = math.sin(dlat / 2) ** 2 + math.cos(a) * math.cos(b) * math.sin(dlon / 2) ** 2
return 6371 * 2 * math.asin(math.sqrt(min(1, hav)))
def suggest_airport(conf: StrDict, personal_data: str) -> tuple[str | None, str]:
"""Prefer explicit mappings, then city matches, then nearby known airports."""
mapping_path = Path(__file__).with_name("conference_airports.json")
mapping: dict[str, str | list[str]] = json.loads(mapping_path.read_text())
overrides = Path(personal_data) / "conference_airports.yaml"
if overrides.exists():
mapping.update(yaml.safe_load(overrides.read_text()) or {})
location = airport_lookup.normalized(str(conf.get("location", "")))
country = str(conf.get("country", "")).lower()
mapped = {
airport_lookup.normalized(key): value for key, value in mapping.items()
}.get(f"{country}:{location}")
if mapped:
return (
",".join(mapped) if isinstance(mapped, list) else mapped
), "location mapping"
airports_path = Path(personal_data) / "airports.yaml"
airports: dict[str, StrDict] = (
yaml.safe_load(airports_path.read_text()) or {}
if airports_path.exists()
else {}
)
candidates = {
code: airport
for code, airport in airports.items()
if not country or str(airport.get("country", "")).lower() == country
}
for code, airport in candidates.items():
if location and location == airport_lookup.normalized(
str(airport.get("city", ""))
):
return code, "airport city matches conference location"
if conf.get("latitude") is not None and conf.get("longitude") is not None:
distances = [
(
distance_km(
float(conf["latitude"]),
float(conf["longitude"]),
float(airport["latitude"]),
float(airport["longitude"]),
),
code,
)
for code, airport in candidates.items()
if airport.get("latitude") is not None
and airport.get("longitude") is not None
]
if distances:
distance, code = min(distances)
if distance <= 150:
return (
code,
f"nearest airport in your data ({distance:.0f} km from venue)",
)
return None, "No airport mapping yet; enter a destination airport code."
def is_short_haul(conf: StrDict) -> bool:
"""Treat known European destinations within 3,500 km as short haul."""
if str(conf.get("country", "")).lower() not in EUROPE:
return False
if conf.get("latitude") is not None and conf.get("longitude") is not None:
return (
distance_km(51.47, -0.45, float(conf["latitude"]), float(conf["longitude"]))
<= 3500
)
return True

View file

@ -0,0 +1,343 @@
"""On-demand flight searches with shared disk caching and refresh throttling."""
import fcntl
import hashlib
import json
import logging
import os
import re
import tempfile
import typing
from datetime import date, datetime, timedelta, timezone
from pathlib import Path
import flask
from agenda import airport_lookup, flight_search_cache, google_flights
from agenda.types import StrDict
LONDON_AIRPORTS = ("LHR", "LGW", "STN", "LTN", "LCY", "SEN")
CACHE_VERSION = 4
logger = logging.getLogger(__name__)
REFRESH_INTERVAL = timedelta(hours=6)
ERROR_INTERVAL = timedelta(minutes=15)
def cache_path(
data_dir: str, start: date, end: date, destination: str, short_haul: bool
) -> Path:
"""Cache by all search inputs, independently of conference names."""
key = json.dumps(
[
CACHE_VERSION,
start.isoformat(),
end.isoformat(),
destination,
short_haul,
"GBP",
"en-GB",
"GB",
]
)
return (
Path(data_dir)
/ "conference-flights"
/ (hashlib.sha256(key.encode()).hexdigest() + ".json")
)
def read_cache(path: Path) -> StrDict | None:
"""Read cached results; a missing or damaged cache is a cache miss."""
try:
value = json.loads(path.read_text())
if not isinstance(value, dict) or not isinstance(value.get("updated_at"), str):
return None
for field in ("updated_at", "attempted_at"):
if field in value and datetime.fromisoformat(value[field]).tzinfo is None:
return None
if not isinstance(value.get("searches"), list):
return None
return typing.cast(StrDict, value)
except (OSError, ValueError, TypeError):
return None
def write_cache(path: Path, value: StrDict) -> None:
"""Replace JSON atomically so concurrent readers never see partial results."""
with tempfile.NamedTemporaryFile(mode="w", dir=path.parent, delete=False) as output:
temporary = output.name
json.dump(value, output)
try:
os.replace(temporary, path)
finally:
if os.path.exists(temporary):
os.unlink(temporary)
def airport_matches(query: str) -> list[StrDict]:
"""Look up locally, preferring airport names in the app's personal data."""
personal_data = (
str(flask.current_app.config["PERSONAL_DATA"])
if flask.has_app_context()
else None
)
return airport_lookup.matches(query, personal_data)
def resolve_airport(query: str) -> tuple[str, str]:
"""Resolve a code or the best name match, and expose its name for confirmation."""
codes = airport_lookup.code_list(query)
if len(codes) > 1:
for code in codes:
if not valid_airport(code):
raise ValueError(f"Unknown IATA airport code {code!r}.")
matches = airport_matches(",".join(codes))
if not matches:
raise ValueError("Unknown airport in destination group")
return str(matches[0]["code"]), str(matches[0]["name"])
code = query.strip().upper()
if valid_airport(code):
matches = airport_matches(code)
return code, str(matches[0]["name"]) if matches else code
if len(code) == 3 and code.isascii() and code.isalpha():
raise ValueError(
f"Unknown IATA airport code {code!r}. Enter a valid code or a longer city/airport name."
)
matches = airport_matches(query)
if not matches:
raise ValueError(
f"No airport found for {query!r}. Try a city, airport name or IATA code."
)
return str(matches[0]["code"]), str(matches[0]["name"])
def error_detail(exc: Exception) -> str:
"""Describe retry wrappers and underlying causes without losing the useful error."""
messages: list[str] = []
current: BaseException | None = exc
seen: set[int] = set()
while current is not None and id(current) not in seen and len(messages) < 3:
seen.add(id(current))
last_attempt = getattr(current, "last_attempt", None)
if last_attempt is not None:
underlying = last_attempt.exception()
if underlying is not None and id(underlying) not in seen:
current = underlying
continue
raw_message = str(current)
if "Browser logs:" in raw_message:
# Playwright repeats its long command before the useful stderr.
diagnostic = list(dict.fromkeys(re.findall(r"\[err\] (.+)", raw_message)))
if diagnostic:
raw_message = (
raw_message.split("Browser logs:", 1)[0]
+ " "
+ " ".join(diagnostic)
)
message = " ".join(raw_message.split())[:1000]
messages.append(
f"{type(current).__name__}: {message or 'No error message provided'}"
)
current = current.__cause__ or current.__context__
return " — ".join(messages)
def date_unavailable_message(exc: BaseException) -> str | None:
"""Use Google's date-range rejection as the banner through route wrappers."""
current: BaseException | None = exc
seen: set[int] = set()
while current is not None and id(current) not in seen:
seen.add(id(current))
if isinstance(current, google_flights.FlightDateUnavailableError):
return str(current)
current = current.__cause__ or current.__context__
return None
def flight_rank(result: StrDict) -> tuple[int, int]:
"""Prefer fewer stops, then BA-operated legs, preserving Google's ranking."""
ba_legs = [
leg.get("operating_airline_code", leg.get("airline_code")) == "BA"
for leg in result["legs"]
]
ba_preference = 0 if ba_legs and all(ba_legs) else 1 if any(ba_legs) else 2
return int(result["stops"]), ba_preference
def search_day(origin: str, destination: str, day: date, direct: bool) -> list[StrDict]:
"""Reuse a route/date cache before making any paced provider requests."""
directory = flight_search_cache.search_directory.get()
if directory is None:
raise RuntimeError(
"Flight searches must run inside a cached conference lookup."
)
return flight_search_cache.cached_day(
directory,
origin,
destination,
day.isoformat(),
direct,
lambda: fetch_day(origin, destination, day, direct, directory),
)
def fetch_day(
origin: str, destination: str, day: date, direct: bool, directory: Path
) -> list[StrDict]:
"""Read the browser search page, preferring direct and one-stop flights."""
max_stops = 0 if direct else 1
rows = google_flights.search(origin, destination, day, max_stops)
rows = [row for row in rows if row["stops"] <= max_stops]
if not rows and not direct:
rows = google_flights.search(origin, destination, day, 2)
rows = [row for row in rows if row["stops"] <= 2]
rows.sort(key=flight_rank)
return rows[:5]
def valid_airport(code: str) -> bool:
"""Accept known IATA codes and London's metropolitan code."""
return code == "LON" or code in airport_lookup.airports()
def search_origin(
origin: str, destination: str, start: date, end: date, flexible: bool
) -> StrDict:
"""Expand each direction from the closest date, stopping when flights are found."""
days = range(1, 5) if flexible else range(1, 3)
outbound: list[StrDict] = []
inbound: list[StrDict] = []
outbound_dates: list[str] = []
inbound_dates: list[str] = []
for offset in days:
departure = start - timedelta(days=offset if flexible else offset + 1)
if not outbound and departure >= date.today():
outbound_dates.append(departure.isoformat())
# Only stop expanding when flights arrive before the opening date.
outbound = [
row
for row in search_day(origin, destination, departure, origin == "BRS")
if date.fromisoformat(row["arrival"][:10]) < start
]
if not inbound and (flexible or offset == 1):
return_day = end + timedelta(days=offset)
inbound_dates.append(return_day.isoformat())
inbound = search_day(destination, origin, return_day, origin == "BRS")
if outbound and inbound:
break
return {
"origin": origin,
"outbound": outbound,
"inbound": inbound,
"outbound_dates": outbound_dates,
"inbound_dates": inbound_dates,
}
def lookup(
path: Path, start: date, end: date, destination: str, short_haul: bool
) -> StrDict:
"""Serialize searches across workers, reuse recent results, and cache errors."""
path.parent.mkdir(parents=True, exist_ok=True)
# A shared lock limits all conference requests, not just identical searches.
with (path.parent / "search.lock").open(
"a+"
) as lock, flight_search_cache.use_directory(
path.parent
), google_flights.use_browser(
path.parent
):
try:
fcntl.flock(lock, fcntl.LOCK_EX | fcntl.LOCK_NB)
except BlockingIOError:
raise ValueError(
"Another flight lookup is running. Please try again shortly."
) from None
cached = read_cache(path)
now = datetime.now(timezone.utc)
legacy_empty_or_failed = (
cached
and cached.get("transport") != flight_search_cache.TRANSPORT
and (
cached.get("error")
or not any(
search.get("outbound") or search.get("inbound")
for search in cached.get("searches", [])
)
)
)
retry_old_parse_failure = (
cached
and cached.get("error")
and "Unrecognized Google Flights result format"
in cached.get("error_details", "")
and cached.get("parser_version") != google_flights.PARSER_VERSION
)
needs_earlier_departure = (
cached
and not cached.get("error")
and any(
search.get("origin") == "LON"
and not search.get("outbound")
and "outbound_dates" not in search
for search in cached.get("searches", [])
)
)
if (
cached
and not needs_earlier_departure
and not retry_old_parse_failure
and not legacy_empty_or_failed
and (not cached.get("error") or cached.get("error_details"))
):
timestamp = str(cached.get("attempted_at", cached["updated_at"]))
interval = ERROR_INTERVAL if cached.get("error") else REFRESH_INTERVAL
if now - datetime.fromisoformat(timestamp) < interval:
return cached
try:
searches = []
if short_haul:
bristol = search_origin("BRS", destination, start, end, flexible=True)
searches.append(bristol)
if not bristol["outbound"] or not bristol["inbound"]:
searches.append(
search_origin("LON", destination, start, end, flexible=False)
)
else:
searches.append(
search_origin("LON", destination, start, end, flexible=False)
)
value: StrDict = {
"updated_at": now.isoformat(),
"destination": destination,
"searches": searches,
"transport": flight_search_cache.TRANSPORT,
"parser_version": google_flights.PARSER_VERSION,
}
except Exception as exc:
logger.exception(
"Conference flight lookup failed for %s (%s to %s)",
destination,
start,
end,
)
# Preserve previously successful data and its original timestamp.
value = dict(
cached
or {
"updated_at": now.isoformat(),
"destination": destination,
"searches": [],
}
)
value.update(
error=flight_search_cache.cooldown_message(path.parent)
or date_unavailable_message(exc)
or "Flight lookup failed. Please try again in 15 minutes.",
error_details=error_detail(exc),
attempted_at=now.isoformat(),
transport=flight_search_cache.TRANSPORT,
parser_version=google_flights.PARSER_VERSION,
)
write_cache(path, value)
return value

View file

@ -1,6 +1,7 @@
"""Prepare conference lists, country filters, series summaries, and timelines."""
import decimal
import hashlib
import os.path
import typing
from collections import defaultdict
@ -9,6 +10,7 @@ from datetime import date, timedelta
import yaml
import agenda.conference
import agenda.conference_detail
from agenda.types import StrDict, Trip
@ -26,6 +28,7 @@ def build_conference_list(data_dir: str, trips: list[Trip]) -> list[StrDict]:
for conf in items:
conf.update(agenda.conference.validate_conference_date_fields(conf))
conf["slug"] = agenda.conference_detail.conference_slug(conf)
price = conf.get("price")
if price:
@ -40,6 +43,15 @@ def build_conference_list(data_dir: str, trips: list[Trip]) -> list[StrDict]:
if this_trip := conference_trip_lookup.get(key):
conf["linked_trip"] = this_trip
identities: defaultdict[tuple[date, str], list[StrDict]] = defaultdict(list)
for conf in items:
identities[(conf["sort_date"], conf["slug"])].append(conf)
for group in identities.values():
if len(group) > 1:
for conf in group:
identity = str((conf["name"], conf.get("url")))
conf["slug"] += "-" + hashlib.sha256(identity.encode()).hexdigest()[:8]
items.sort(key=lambda item: item["sort_date"])
return items
@ -172,6 +184,8 @@ def build_conference_timeline(
{
"name": conf["name"],
"url": conf.get("url"),
"slug": conf.get("slug"),
"event_date": conf["sort_date"].isoformat(),
"lane": lane,
"left_pct": left_pct,
"width_pct": width_pct,

281
agenda/conference_page.py Normal file
View file

@ -0,0 +1,281 @@
"""Conference detail page and explicit flight lookup action."""
import hmac
import secrets
import typing
from datetime import date, datetime, timedelta
import flask
import werkzeug
import agenda
import agenda.conference
import agenda.conference_detail as detail
import agenda.conference_flights as flights
import agenda.conference_list
import agenda.flight_search_cache
import agenda.google_flights
import agenda.trip
from agenda.types import StrDict
blueprint = flask.Blueprint("conference", __name__)
@blueprint.app_template_filter("conference_date")
def format_date(value: typing.Any, part: str = "full") -> str:
"""Render dates without converting flight times out of the airport's local time."""
parsed = value
if isinstance(value, str):
try:
parsed = (
datetime.fromisoformat(value)
if "T" in value or " " in value
else date.fromisoformat(value)
)
except ValueError:
return value
if isinstance(parsed, datetime):
if part == "time":
return parsed.strftime("%H:%M")
if part == "date":
return parsed.strftime("%a %-d %b %Y")
return parsed.strftime("%a %-d %b %Y at %H:%M")
if isinstance(parsed, date):
return parsed.strftime("%a %-d %b %Y")
return str(value)
def searched_dates(
search: StrDict, direction: str, start: date, end: date
) -> list[str]:
"""Read checked dates; reconstruct the old policy for legacy cached results."""
field = direction + "_dates"
if field in search:
return typing.cast(list[str], search[field])
rows = search.get(direction, [])
if rows:
return sorted({str(row["departure"])[:10] for row in rows})
offsets = (
range(1, 5)
if search["origin"] == "BRS"
else (2 if direction == "outbound" else 1,)
)
return [
(
start - timedelta(days=offset)
if direction == "outbound"
else end + timedelta(days=offset)
).isoformat()
for offset in offsets
if direction != "outbound" or start - timedelta(days=offset) >= date.today()
]
def google_flights_link(
origin: str, destination: str, start: date, end: date, search: StrDict | None = None
) -> StrDict:
"""Open a round trip with the displayed route, dates and stop preference."""
departure = start - timedelta(days=1 if origin == "BRS" else 2)
returning = end + timedelta(days=1)
if search:
for direction in ("outbound", "inbound"):
rows = search.get(direction, [])
checked = search.get(direction + "_dates", [])
selected = (
str(rows[0]["departure"])[:10]
if rows
else checked[-1] if checked else None
)
if selected:
if direction == "outbound":
departure = date.fromisoformat(selected)
else:
returning = date.fromisoformat(selected)
max_stops = 0 if origin == "BRS" else 1
if (
search
and origin != "BRS"
and any(
row.get("stops", 0) > 1
for direction in ("outbound", "inbound")
for row in search.get(direction, [])
)
):
max_stops = 2
return {
"url": agenda.google_flights.search_url(
origin, destination, departure, max_stops, returning
),
"departure": departure,
"return": returning,
}
def find_conference(event_date: str, slug: str) -> StrDict:
"""Resolve a dated conference identity, including approximate dates."""
for conf in agenda.conference_list.build_conference_list(
flask.current_app.config["PERSONAL_DATA"], agenda.trip.build_trip_list()
):
if conf["sort_date"].isoformat() == event_date and conf["slug"] == slug:
return conf
flask.abort(404)
@blueprint.route("/conference/<event_date>/<slug>", methods=["GET", "POST"])
def page(event_date: str, slug: str) -> str | werkzeug.Response:
"""GET reads only cached data; POST explicitly requests a flight search."""
conf = find_conference(event_date, slug)
config = flask.current_app.config
suggested, airport_reason = detail.suggest_airport(conf, config["PERSONAL_DATA"])
parameters = (
flask.request.form if flask.request.method == "POST" else flask.request.args
)
default_location = str(conf.get("location") or "")
airport_input = parameters.get("airport", default_location).strip()[:200]
destination = airport_input.upper()
resolved_name = ""
resolution_error = ""
if airport_input:
try:
lookup_query = (
suggested
if suggested and airport_input == default_location
else airport_input
)
destination, resolved_name = flights.resolve_airport(lookup_query)
except ValueError as exc:
resolution_error = str(exc)
eligible = (
conf["date_status"] in agenda.conference.DATED_STATUSES
and conf["start_date"] > date.today()
and not conf.get("online")
and (
str(conf.get("country", "")).lower() != "gb"
or str(conf.get("location", "")).partition(",")[0].strip().casefold()
== "belfast"
)
)
cache = flights.cache_path(
config["DATA_DIR"],
conf["start_date"],
conf["end_date"],
destination,
detail.is_short_haul(conf),
)
token = flask.session.setdefault(
"conference_flight_token", secrets.token_urlsafe(32)
)
if flask.request.method == "POST":
if not hmac.compare_digest(
str(token), flask.request.form.get("csrf_token", "")
):
flask.abort(400)
if not eligible:
flask.abort(400)
if resolution_error or destination in {"BRS", "LON"}:
flask.flash(
resolution_error or "Choose a destination outside Bristol or London.",
"warning",
)
else:
try:
flights.lookup(
cache,
conf["start_date"],
conf["end_date"],
destination,
detail.is_short_haul(conf),
)
except ValueError as exc:
flask.flash(str(exc), "warning")
return flask.redirect(
flask.url_for(
"conference.page",
event_date=event_date,
slug=slug,
airport=airport_input if resolution_error else destination,
)
)
excluded = {
"name",
"location",
"country",
"venue",
"address",
"latitude",
"longitude",
"url",
"dates",
"date_status",
"start",
"end",
"start_date",
"end_date",
"sort_date",
"latest_date",
"display_date",
"has_exact_dates",
"series",
"series_detail",
"linked_trip",
"slug",
}
fields = [
(key.replace("_", " ").capitalize(), value)
for key, value in conf.items()
if key not in excluded and value is not None
]
cached = flights.read_cache(cache) if destination and not resolution_error else None
if cached:
for search in cached.get("searches", []):
for direction in ("outbound", "inbound"):
search[direction + "_dates"] = searched_dates(
search, direction, conf["start_date"], conf["end_date"]
)
search["google_flights"] = google_flights_link(
search["origin"],
destination,
conf["start_date"],
conf["end_date"],
search,
)
default_link = (
google_flights_link(
"BRS" if detail.is_short_haul(conf) else "LON",
destination,
conf["start_date"],
conf["end_date"],
)
if eligible and not resolution_error and destination
else None
)
return flask.render_template(
"conference_detail.html",
conf=conf,
fields=fields,
get_country=agenda.get_country,
airport=airport_input,
resolved_airport=destination if not resolution_error else "",
resolved_name=resolved_name,
resolution_error=resolution_error,
airport_reason=airport_reason,
suggested=suggested,
short_haul=detail.is_short_haul(conf),
eligible=eligible,
cached=cached,
default_google_flights=default_link,
csrf_token=token,
cooldown=agenda.flight_search_cache.cooldown_message(cache.parent),
)
@blueprint.get("/conference/airports")
def airport_search() -> flask.Response:
"""Autocomplete airport names locally; this endpoint never searches Google."""
query = flask.request.args.get("q", "").strip()[:200]
if len(query) < 2:
return flask.jsonify([])
try:
return flask.jsonify(flights.airport_matches(query))
except ValueError as exc:
return flask.jsonify(error=str(exc))

View file

@ -0,0 +1,87 @@
"""Bounded, best-effort diagnostic snapshots for failed flight searches."""
import gzip
import json
import logging
import os
import tempfile
import time
from datetime import datetime, timezone
from pathlib import Path
from agenda.types import StrDict
MAX_RECORDS = 100
RETENTION_SECONDS = 30 * 24 * 60 * 60
logger = logging.getLogger(__name__)
def write_failure(directory: Path, details: StrDict) -> Path | None:
"""Save one compressed JSON snapshot; logging failures never break lookups."""
temporary: str | None = None
try:
folder = directory / "errors"
folder.mkdir(parents=True, exist_ok=True)
truncated: list[str] = []
def limit(value: str, size: int, field: str) -> str:
data = value.encode("utf-8")
if len(data) > size:
truncated.append(field)
return data[:size].decode("utf-8", errors="ignore")
return value
snapshot = dict(details)
snapshot["format_version"] = 1
for field, size in (
("page_html", 2 * 1024 * 1024),
("flight_script", 1024 * 1024),
("page_text", 64 * 1024),
):
if isinstance(snapshot.get(field), str):
snapshot[field] = limit(snapshot[field], size, field)
responses = []
for index, response in enumerate(snapshot.get("http_errors", [])[:5]):
item = dict(response)
if isinstance(item.get("body"), str):
item["body"] = limit(
item["body"], 256 * 1024, f"http_errors[{index}].body"
)
responses.append(item)
snapshot["http_errors"] = responses
snapshot["truncated_fields"] = truncated
snapshot["recorded_at"] = datetime.now(timezone.utc).isoformat()
stamp = datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%S%fZ")
with tempfile.NamedTemporaryFile(
dir=folder, prefix=".snapshot-", delete=False
) as temporary_output:
temporary = temporary_output.name
path = (
folder
/ f"{stamp}-{Path(temporary).name.removeprefix('.snapshot-')}.json.gz"
)
with gzip.open(temporary, "wt", encoding="utf-8") as compressed_output:
json.dump(snapshot, compressed_output, ensure_ascii=False)
os.chmod(temporary, 0o640)
os.replace(temporary, path)
temporary = None
records = sorted(
folder.glob("*.json.gz"),
key=lambda record: record.stat().st_mtime,
reverse=True,
)
cutoff = time.time() - RETENTION_SECONDS
for index, record in enumerate(records):
if index >= MAX_RECORDS or record.stat().st_mtime < cutoff:
record.unlink()
logger.info("Saved flight failure diagnostics to %s", path)
return path
except Exception:
logger.warning("Unable to save flight failure diagnostics", exc_info=True)
return None
finally:
if temporary is not None:
try:
Path(temporary).unlink(missing_ok=True)
except OSError:
pass

View file

@ -0,0 +1,141 @@
"""Shared pacing, HTTP 429 cooldowns and caches for individual flight searches."""
import contextvars
import fcntl
import hashlib
import json
import time
import typing
from contextlib import contextmanager
from datetime import datetime, timezone
from pathlib import Path
from agenda.types import StrDict
REQUEST_INTERVAL = 3.0
RATE_LIMIT_COOLDOWN = 30 * 60
DAY_CACHE_SECONDS = 6 * 60 * 60
DAY_CACHE_VERSION = 1
TRANSPORT = "playwright-chromium"
search_directory: contextvars.ContextVar[Path | None] = contextvars.ContextVar(
"conference_flight_search_directory", default=None
)
class RateLimitCooldownError(RuntimeError):
"""Google flight requests are temporarily paused across all workers."""
@contextmanager
def use_directory(directory: Path) -> typing.Iterator[None]:
"""Share a lookup's configured cache directory with its individual searches."""
token = search_directory.set(directory)
try:
yield
finally:
search_directory.reset(token)
def read_state(path: Path) -> StrDict:
"""Read shared JSON state, treating damaged files as a cache miss."""
try:
state = json.loads(path.read_text())
return typing.cast(StrDict, state) if isinstance(state, dict) else {}
except (OSError, ValueError):
return {}
def cooldown_message(directory: Path) -> str | None:
"""Describe an active shared cooldown without making any requests."""
state = read_state(directory / "rate-limit.json")
until = state.get("cooldown_until")
if not isinstance(until, (float, int)) or until <= time.time():
return None
when = datetime.fromtimestamp(until, timezone.utc).strftime(
"%a %-d %b at %H:%M UTC"
)
return f"Google returned HTTP 429. All flight searches are paused until {when}. Cached results remain available."
def write_state(path: Path, state: StrDict) -> None:
"""Use the agenda's atomic JSON writer for state and day caches."""
# Imported lazily to avoid a module cycle.
from agenda.conference_flights import write_cache
write_cache(path, state)
def wait_for_request(directory: Path) -> None:
"""Space browser searches across processes."""
directory.mkdir(parents=True, exist_ok=True)
with (directory / "request.lock").open("a+") as lock:
fcntl.flock(lock, fcntl.LOCK_EX)
message = cooldown_message(directory)
if message:
raise RateLimitCooldownError(message)
state_path = directory / "rate-limit.json"
state = read_state(state_path)
last = state.get("last_request_at", 0)
last = float(last) if isinstance(last, (float, int)) else 0.0
delay = min(REQUEST_INTERVAL, max(0.0, last + REQUEST_INTERVAL - time.time()))
if delay:
time.sleep(delay)
state["last_request_at"] = time.time()
write_state(state_path, state)
def block_requests(directory: Path) -> None:
"""Set a shared cooldown immediately after the first HTTP 429."""
directory.mkdir(parents=True, exist_ok=True)
with (directory / "request.lock").open("a+") as lock:
fcntl.flock(lock, fcntl.LOCK_EX)
path = directory / "rate-limit.json"
state = read_state(path)
state["cooldown_until"] = time.time() + RATE_LIMIT_COOLDOWN
state["transport"] = TRANSPORT
write_state(path, state)
def cached_day(
directory: Path,
origin: str,
destination: str,
day: str,
direct: bool,
fetch: typing.Callable[[], list[StrDict]],
) -> list[StrDict]:
"""Reuse a route/date's results across conferences, including empty results."""
key = json.dumps(
[
DAY_CACHE_VERSION,
origin,
destination,
day,
direct,
"BEST",
"BA",
"GBP",
"en-GB",
"GB",
]
)
folder = directory / "days"
folder.mkdir(parents=True, exist_ok=True)
path = folder / (hashlib.sha256(key.encode()).hexdigest() + ".json")
# Recheck under the lock so concurrent lookups cannot duplicate a fetch.
with path.with_suffix(".lock").open("a+") as lock:
fcntl.flock(lock, fcntl.LOCK_EX)
state = read_state(path)
timestamp = state.get("fetched_at")
if (
isinstance(timestamp, (float, int))
and time.time() - timestamp < DAY_CACHE_SECONDS
and isinstance(state.get("flights"), list)
and (state["flights"] or state.get("transport") == TRANSPORT)
):
return typing.cast(list[StrDict], state["flights"])
rows = fetch()
write_state(
path, {"fetched_at": time.time(), "flights": rows, "transport": TRANSPORT}
)
return rows

View file

@ -1,5 +1,6 @@
"""Geomob events."""
import logging
import os
import subprocess
from pathlib import Path
@ -13,9 +14,13 @@ import lxml.html # type: ignore[import-untyped]
import requests
import agenda.add_new_conference
import agenda.conference_list
import agenda.mail
import agenda.utils
logger = logging.getLogger(__name__)
AGENDA_BASE_URL = "https://edwardbetts.com/agenda"
@dataclass(frozen=True)
class GeomobEvent:
@ -56,6 +61,7 @@ def geomob_email(
new_events: list[GeomobEvent],
base_url: str,
recording_results: dict[GeomobEvent, str] | None = None,
conference_links: dict[GeomobEvent, str] | None = None,
) -> tuple[str, str]:
"""Generate email subject and body for new events.
@ -80,6 +86,8 @@ def geomob_email(
else:
assert "//" not in url, f"Double slash found in URL: {url}"
event_details = f"Date: {event.date}\nURL: {url}\nHashtag: {event.hashtag}\n"
if conference_links and event in conference_links:
event_details += f"Conference page: {conference_links[event]}\n"
if recording_results is not None:
event_details += f"Recording: {recording_results[event]}\n"
body_lines.append(event_details)
@ -89,6 +97,31 @@ def geomob_email(
return (subject, body)
def conference_page_links(
events: list[GeomobEvent], base_url: str, personal_data: str
) -> dict[GeomobEvent, str]:
"""Find saved events using the same dated identities as the conference pages."""
try:
conferences = agenda.conference_list.build_conference_list(personal_data, [])
except Exception:
logger.warning(
"Could not find conference links for Geomob email", exc_info=True
)
return {}
by_identity = {
(conference.get("url"), conference["start_date"]): conference
for conference in conferences
}
links = {}
for event in events:
conference = by_identity.get((base_url + event.href, event.date))
if conference is not None:
links[event] = (
f"{AGENDA_BASE_URL}/conference/{conference['sort_date'].isoformat()}/{conference['slug']}"
)
return links
def record_event(event: GeomobEvent, base_url: str) -> str:
"""Record an event, reporting failures without suppressing its notification.
@ -183,5 +216,10 @@ def update(config: flask.config.Config) -> None:
base_url = "https://thegeomob.com"
recording_results = {event: record_event(event, base_url) for event in new_events}
subject, body = geomob_email(new_events, base_url, recording_results)
links = (
conference_page_links(new_events, base_url, config["PERSONAL_DATA"])
if config.get("PERSONAL_DATA")
else {}
)
subject, body = geomob_email(new_events, base_url, recording_results, links)
agenda.mail.send_mail(config, subject, body)

417
agenda/google_flights.py Normal file
View file

@ -0,0 +1,417 @@
"""Read Google Flights in Chromium via Playwright, without a flight API client.
Query schema and embedded result format documented by AWeirdDev/fast-flights.
See licenses/fast-flights-MIT.txt. Google's undocumented format may change.
"""
import base64
import contextvars
import json
import logging
import os
import re
import shutil
import typing
from contextlib import contextmanager
from datetime import date, datetime
from pathlib import Path
from urllib.parse import urlencode, urlsplit
from playwright.sync_api import (
BrowserContext,
Page,
Playwright,
Response,
sync_playwright,
)
from agenda import flight_diagnostics, flight_search_cache
from agenda.types import StrDict
logger = logging.getLogger(__name__)
PARSER_VERSION = 2
LONDON_AIRPORTS = ("LHR", "LGW", "STN", "LTN", "LCY", "SEN")
class FlightDateUnavailableError(ValueError):
"""Google explicitly rejected a search beyond its available date range."""
def date_limit_message(text: str) -> str | None:
"""Recognize a provider date-range error without assuming a fixed horizon."""
normalized_text = " ".join(text.split())
if re.search(
r"(?:date|flight|departure|return)[^.]{0,160}too far[^.]{0,80}future",
normalized_text,
re.IGNORECASE,
):
return "Requested flight date is too far in the future. Please try again closer to departure."
return None
def chromium_executable() -> str:
"""Avoid Debian's shell wrapper: Apache hides its /proc/cpuinfo probe."""
binary = Path("/usr/lib/chromium/chromium")
if binary.is_file() and os.access(binary, os.X_OK):
return str(binary)
executable = shutil.which("chromium")
if executable is None:
raise RuntimeError("Chromium is not installed")
return executable
def varint(value: int) -> bytes:
"""Encode an unsigned protobuf integer."""
result = bytearray()
while value > 127:
result.append((value & 127) | 128)
value >>= 7
result.append(value)
return bytes(result)
def field(number: int, value: str | bytes | int) -> bytes:
"""Encode the small subset of protobuf needed by a search URL."""
if isinstance(value, int):
return varint(number << 3) + varint(value)
data = value.encode() if isinstance(value, str) else value
return varint((number << 3) | 2) + varint(len(data)) + data
def search_segment(origin: str, destination: str, day: date, max_stops: int) -> bytes:
"""Encode airports, departure date and stop limit for one direction."""
segment = field(2, day.isoformat()) + field(5, max_stops)
for number, code in ((13, origin), (14, destination)):
for member in code.split(","):
for airport in LONDON_AIRPORTS if member == "LON" else (member,):
segment += field(number, field(1, 1) + field(2, airport))
return segment
def search_url(
origin: str,
destination: str,
day: date,
max_stops: int,
return_day: date | None = None,
) -> str:
"""One adult, economy, British locale; one-way or a prefilled return trip."""
info = field(3, search_segment(origin, destination, day, max_stops))
if return_day is not None:
info += field(3, search_segment(destination, origin, return_day, max_stops))
info += field(8, 1) + field(9, 1) + field(19, 1 if return_day is not None else 2)
return "https://www.google.com/travel/flights/search?" + urlencode(
{
"tfs": base64.urlsafe_b64encode(info).decode().rstrip("="),
"tfu": "EgQIABABIgA",
"hl": "en-GB",
"gl": "GB",
"curr": "GBP",
}
)
def timestamp(values: list[int], clock: list[int | None] | None) -> str:
"""Google omits zero hours/minutes; preserve midnight and overnight dates."""
padded = [*(clock or []), None, None]
return datetime(
values[0], values[1], values[2], padded[0] or 0, padded[1] or 0
).isoformat()
def parse_results(script: str, url: str) -> list[StrDict]:
"""Read Google's best flights followed by its remaining itineraries."""
marker = re.search(r"\bdata\s*:\s*", script)
if marker is None:
raise ValueError("Google Flights page did not contain flight data")
payload, _ = json.JSONDecoder().raw_decode(script[marker.end() :])
if not isinstance(payload, list) or len(payload) < 4:
message = date_limit_message(json.dumps(payload))
if message:
raise FlightDateUnavailableError(message)
raise ValueError("Unrecognized Google Flights result format")
airlines: dict[str, str] = {}
try:
airlines = dict(payload[7][1][1])
except (IndexError, TypeError, ValueError):
pass
rows: list[StrDict] = []
malformed = 0
for section in payload[2:4]:
if not section:
continue
for item in section[0] or []:
try:
itinerary = item[0]
raw_legs = itinerary[2]
if not raw_legs:
continue
legs = []
for leg in raw_legs:
airline = leg[22]
code = str(airline[0])
operating = airline[2] if len(airline) > 2 else None
legs.append(
{
"airline": airlines.get(code, code),
"airline_code": code,
"operating_airline_code": operating or code,
"number": f"{code} {airline[1]}",
"origin": leg[3],
"destination": leg[6],
}
)
fare = item[1][0] if item[1] else []
price = fare[1] if len(fare) > 1 else None
rows.append(
{
"price": price,
"currency": "GBP",
"stops": len(legs) - 1,
"duration": itinerary[9],
"departure": timestamp(raw_legs[0][20], raw_legs[0][8]),
"arrival": timestamp(raw_legs[-1][21], raw_legs[-1][10]),
"legs": legs,
"url": url,
}
)
except (IndexError, TypeError, ValueError) as exc:
malformed += 1
if not rows:
last_error = exc
if malformed and not rows:
raise ValueError("Could not decode Google Flights itineraries") from last_error
return rows
class BrowserSearch:
"""One lazy browser session shared by the dates in a conference lookup."""
def __init__(self, directory: Path) -> None:
self.directory = directory
self.playwright: Playwright | None = None
self.context: BrowserContext | None = None
self.page: Page | None = None
self.rate_limited = False
self.response: Response | None = None
self.error_responses: list[Response] = []
self.flight_script: str | None = None
def start(self) -> Page:
"""Use installed Chromium, including when running as Apache's www-data."""
if self.page is None:
executable = chromium_executable()
self.playwright = sync_playwright().start()
self.context = self.playwright.chromium.launch_persistent_context(
str(self.directory / "browser-profile"),
executable_path=executable,
headless=True,
locale="en-GB",
timezone_id="Europe/London",
args=["--disable-dev-shm-usage"],
env={
**os.environ,
"XDG_CONFIG_HOME": str(self.directory / "browser-config"),
"XDG_CACHE_HOME": str(self.directory / "browser-cache"),
},
)
self.page = self.context.new_page()
self.page.on("response", self.check_response)
self.page.set_default_timeout(20000)
return self.page
def check_response(self, response: Response) -> None:
"""Observe rate limits on flight data requests as well as navigation."""
host = urlsplit(response.url).hostname or ""
is_google = host == "google.com" or host.endswith(".google.com")
if response.status >= 400 and is_google and len(self.error_responses) < 5:
self.error_responses.append(response)
if response.status == 429 and is_google:
if not self.rate_limited:
flight_search_cache.block_requests(self.directory)
self.rate_limited = True
def check_cooldown(self) -> None:
"""Surface a subrequest's 429 instead of reporting missing flight data."""
if self.rate_limited:
raise flight_search_cache.RateLimitCooldownError(
flight_search_cache.cooldown_message(self.directory)
)
def search(
self, origin: str, destination: str, day: date, max_stops: int
) -> list[StrDict]:
"""Pace searches and capture failed responses without extra Google requests."""
flight_search_cache.wait_for_request(self.directory)
self.response = None
self.error_responses = []
self.flight_script = None
try:
return self.perform_search(origin, destination, day, max_stops)
except Exception as exc:
try:
self.record_failure(origin, destination, day, max_stops, exc)
except Exception:
logger.warning(
"Unable to capture flight failure diagnostics", exc_info=True
)
raise
def perform_search(
self, origin: str, destination: str, day: date, max_stops: int
) -> list[StrDict]:
"""Navigate once, honour consent, and stop immediately on rate limiting."""
page = self.start()
url = search_url(origin, destination, day, max_stops)
response = page.goto(url, wait_until="domcontentloaded", timeout=45000)
self.response = response
if response is not None and response.status == 429:
flight_search_cache.block_requests(self.directory)
raise flight_search_cache.RateLimitCooldownError(
flight_search_cache.cooldown_message(self.directory)
)
self.check_cooldown()
if response is not None and response.status >= 400:
raise RuntimeError(f"Google Flights returned HTTP {response.status}")
if "consent.google." in page.url:
page.get_by_role("button", name="Reject all", exact=True).click()
page.wait_for_url("**/travel/flights/**", timeout=20000)
script = page.locator("script.ds\\:1")
try:
script.wait_for(state="attached", timeout=20000)
except Exception as exc:
self.check_cooldown()
text = " ".join(page.locator("body").inner_text().split())[:400]
raise RuntimeError(
f"Google Flights did not load results ({page.url}): {text}"
) from exc
self.check_cooldown()
self.flight_script = script.inner_text()
return self.read_results(page, self.flight_script, url, day)
def read_results(
self, page: Page, script: str, url: str, day: date
) -> list[StrDict]:
"""Interpret provider messages when the embedded data isn't a result list."""
try:
return parse_results(script, url)
except FlightDateUnavailableError:
raise
except ValueError as exc:
text = " ".join(page.locator("body").inner_text().split())
message = date_limit_message(text)
if message:
raise FlightDateUnavailableError(message) from None
if re.search(
r"\b(?:no (?:(?:non[-\u2010-\u2015 ]?stop|direct) )?flights found|no results found|no matching flights)\b",
text,
re.IGNORECASE,
):
return []
ahead = (day - date.today()).days
context = f"Google Flights returned unusable search data for {day.strftime('%a %-d %b %Y')} ({ahead} days ahead)."
if ahead > 300:
context += " This date may be beyond the available airline schedules; try again closer to departure."
raise ValueError(context) from exc
def record_failure(
self, origin: str, destination: str, day: date, max_stops: int, exc: Exception
) -> None:
"""Capture already-loaded data; unavailable page fields are recorded explicitly."""
from agenda.conference_flights import error_detail
details: StrDict = {
"origin": origin,
"destination": destination,
"departure_date": day.isoformat(),
"max_stops": max_stops,
"currency": "GBP",
"language": "en-GB",
"country": "GB",
"requested_url": search_url(origin, destination, day, max_stops),
"error": error_detail(exc),
"flight_script": self.flight_script,
"transport": flight_search_cache.TRANSPORT,
"parser_version": PARSER_VERSION,
}
capture_errors: list[str] = []
if self.page is not None:
captures: tuple[tuple[str, typing.Callable[[], str]], ...] = (
("page_url", lambda: self.page.url if self.page else ""),
("page_html", lambda: self.page.content() if self.page else ""),
(
"page_text",
lambda: (
self.page.locator("body").inner_text(timeout=2000)
if self.page
else ""
),
),
)
for field, fetch in captures:
try:
details[field] = fetch()
except Exception as capture_error:
capture_errors.append(
f"{field}: {type(capture_error).__name__}: {capture_error}"
)
responses = list(self.error_responses[:5])
if self.response is not None:
details["http_status"] = self.response.status
if self.response.status >= 400 and self.response not in responses:
responses = [self.response, *responses][:5]
errors = []
for response in responses:
try:
errors.append(
{
"url": response.url,
"status": response.status,
"body": response.text(),
}
)
except Exception as capture_error:
capture_errors.append(
f"response body: {type(capture_error).__name__}: {capture_error}"
)
details["http_errors"] = errors
details["capture_errors"] = capture_errors
flight_diagnostics.write_failure(self.directory, details)
def close(self) -> None:
"""Always release Chromium, including on failed searches."""
try:
if self.context is not None:
self.context.close()
finally:
if self.playwright is not None:
self.playwright.stop()
session: contextvars.ContextVar[BrowserSearch | None] = contextvars.ContextVar(
"conference_browser_session", default=None
)
@contextmanager
def use_browser(directory: Path) -> typing.Iterator[None]:
"""Lazy initialization keeps cache hits free of browser startup costs."""
browser = BrowserSearch(directory)
token = session.set(browser)
try:
yield
finally:
session.reset(token)
browser.close()
def search(origin: str, destination: str, day: date, max_stops: int) -> list[StrDict]:
"""Attach the route/date to browser failures for display on the page."""
browser = session.get()
if browser is None:
raise RuntimeError("Flight searches require a conference browser session")
try:
return browser.search(origin, destination, day, max_stops)
except Exception as exc:
raise RuntimeError(f"{origin} → {destination} on {day.isoformat()}") from exc

View file

@ -861,3 +861,142 @@ Example:
name: Brussels for FOSDEM
private: false
```
### Conference detail pages and airport suggestions
Conference titles in the list and series pages link to
`/conference/YYYY-MM-DD/series-or-title-location`. The date is the conference's
start date (or the earliest date for an approximate event). The detail page keeps
an external website link and displays the venue, address, coordinates, attendance
fields and remaining conference metadata.
Airport suggestions start with `agenda/conference_airports.json`. You can override
or extend these in `personal-data/conference_airports.yaml`, using lowercase
country codes and casefolded location names:
```yaml
"be:brussels": CRL
"us:pasadena, california": LAX
"se:malmö": [MMX, CPH]
"ch:bern": [BRN, BSL]
"de:bonn": [CGN, DUS]
"dk:funen": [BLL, CPH]
"it:bologna": [BLQ, VRN, VCE]
```
Mappings may name several airports. Malmö (also entered as Malmo or Malmö,
Sweden) searches MMX and CPH together, including cross-border airports explicitly
listed in the mapping. Bern searches BRN and Basel (BSL) together; Bonn searches Cologne/Bonn (CGN)
and Düsseldorf (DUS) together. Funen includes Billund (BLL) and Copenhagen (CPH). Bologna includes BLQ, Verona
(VRN) and Venice Marco Polo (VCE). You can also enter comma-separated IATA codes such as
`MMX,CPH`. Each route/date uses one combined Google search, so Bristol's direct
flights to CPH are considered without doubling the number of requests. Results
show the actual airport for each flight leg.
Otherwise the page matches cities in `airports.yaml`, then suggests the nearest
known airport within 150 km of the venue, limited to the same country when known.
The flight lookup form accepts IATA codes, cities and airport names, with local
autocomplete from the bundled public-domain OurAirports index
(`agenda/airport_index.json`, downloaded from https://ourairports.com/data/).
Names from `PERSONAL_DATA/airports.yaml` (normally
`~/src/personal-data/airports.yaml`) override bundled labels for individual airports,
combined destinations and autocomplete. Missing or blank names fall back to the
bundled index, and edits take effect on the next request. Both personal and
bundled names remain searchable.
Scheduled-service airports are preferred; three-letter inputs are treated as
IATA codes. London resolves to LON (all London airports). The highest-ranked match is
used and its name/code shown; pick an autocomplete result for a specific airport.
Country suffixes narrow the matches; US state abbreviations such as CA are
recognized too. For example,
`Copenhagen, Denmark` resolves to CPH. A nearby airport may
still require a substantial ground transfer. Unknown locations need an IATA code
or city/airport name entered manually.
“Open in Google Flights” opens a prefilled round trip in a new tab, with the
current airport group, outbound/return dates, stops, one adult in economy and
GBP/en-GB settings. Before a lookup it uses the initial Bristol or London dates;
for cached results each origin has its own link using the displayed flights'
departure dates (or the last checked dates when no flights were found). Links
are generated locally and work during the shared server cooldown.
Only the flight lookup button contacts Google Flights from the server. Searches use Python
Playwright with installed Chromium, one adult in economy, language
`en-GB`, country `GB`
and currency `GBP`. Overseas European destinations within 3,500 km of London
prefer direct Bristol flights. Start with the day before and the day after the
conference, expanding each direction independently to at most four days away
only when no suitable flights are found. Stop searching a direction at the first
date with flights; when both nearest dates work, this needs just two searches. If either direction has no suitable results, London is
searched too. Other destinations start with London. London means LHR, LGW, STN,
LTN, LCY and SEN together, initially with at most one stop; departures are two days
before the conference to allow overnight travel, trying three days before if
no suitable outbound flights are found. Returns are the day after. Outbound
flights must arrive before the opening date. Prices are **one-way fares**, shown
separately for outbound and return travel. Searches use Google’s best-flight group first,
then prefer direct flights and BA within each stop count (all BA legs first, then
some BA legs, then other airlines). Operating airline is used when available,
otherwise the advertised airline. Google’s order is preserved within each group.
If a London search has no direct or one-stop results, it retries with at most two
stops. Bristol always remains direct-only.
Results live in `DATA_DIR/conference-flights` (normally
`/home/edward/lib/data/conference-flights`), keyed by dates, destination and search
policy. Updates reuse results for six hours. Empty results are also cached;
errors have a 15-minute retry cooldown and preserve any previous successful
results. Failures show the route/date and underlying exception type/message;
full tracebacks are logged on the server. Explicit Google date-range rejections
ask you to retry closer to departure, rather than after 15 minutes. Confirmed
“no flights found” and “no non-stop flights found” pages count as empty results;
Bristol searches can then continue to nearby dates and show “No direct flights
found” with the checked dates if none work; other unusable responses remain
errors and include the requested date. No fixed airline booking horizon is assumed. Legacy cached failures without details
can be retried immediately. A shared process lock prevents simultaneous conference flight searches.
The page shows when results were last fetched, the destination airport names,
and the departure dates checked when no suitable flights were found. Belfast is an exception to the UK exclusion: it searches Belfast International
(BFS) and Belfast City (BHD) together, starting with direct flights from Bristol.
Conferences with tentative start/end dates can also search for flights; the flight
section displays those provisional dates. Other domestic, past, online and
approximate-date conferences display details but do not offer flight lookup.
Individual route/date searches (including empty results) are cached for six hours
under `conference-flights/days`, so overlapping conferences reuse the same Google
lookup. Browser searches are spaced at least three seconds apart using shared disk state
across Apache workers. Chromium loads the page’s assets normally. HTTP 429 is not immediately
retried: it pauses all uncached Google flight requests for 30 minutes. The page
shows the shared cooldown and continues to display cached results. Failed navigations are not automatically retried.
Search URL encoding and result parsing follow the format documented by
https://github.com/AWeirdDev/flights (MIT attribution in
`licenses/fast-flights-MIT.txt`). `flights` is no longer an application dependency. `run.fcgi` uses `/usr/bin/python3`; a project virtualenv is no longer
needed. System Python must provide Playwright and the existing Flask dependencies,
and Chromium must be installed. On Debian we launch
`/usr/lib/chromium/chromium` directly: Apache's `ProcSubset=pid` hides
`/proc/cpuinfo`, which causes `/usr/bin/chromium`'s shell-wrapper CPU check to
fail incorrectly. Other installations fall back to `chromium` on PATH. The browser is started only on a cache miss,
reused for that lookup’s dates, and closed after the lookup. Its persistent profile
in `conference-flights/browser-profile` preserves consent choices. The shared
search lock prevents multiple workers opening that profile simultaneously.
Google’s result format is undocumented, and using a browser does not guarantee
avoidance of rate limits or verification challenges. Existing successful cached
fares remain usable; old empty results and failures can be refreshed using the
browser, subject to the existing shared cooldown.
Failed browser searches save compressed JSON diagnostics in
`DATA_DIR/conference-flights/errors/*.json.gz`. Each contains the UTC timestamp,
route, requested date, stop limit, locale/currency, request and final page URLs,
HTTP status, exception details, raw flight-data script, visible page text, page
HTML, and up to five Google HTTP error response bodies. Capture uses the loaded
page and received responses; it makes no extra Google requests. Capture failures
are logged and cannot replace the original lookup error. Local cooldown refusals
and successful searches do not create snapshots.
Snapshots are retained for 30 days, with at most 100 files; cleanup runs when a
new snapshot is written. Captures are capped at 2 MiB of HTML, 1 MiB of flight
script, 64 KiB of visible text and 256 KiB per HTTP error body, with any truncation
recorded. Snapshot files have mode 0640. Previous failures cannot be recovered
from these snapshots; recording begins with this change. Use Python's `gzip.open`
to read the JSON for analysis or replay the saved `flight_script` through
`agenda.google_flights.parse_results`.

View file

@ -0,0 +1,21 @@
MIT License
Copyright (c) 2024-PRESENT fast-flights Contributors
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
SOFTWARE.

View file

@ -11,3 +11,4 @@ flask
requests
emoji
timezonefinder
playwright

View file

@ -1,8 +1,9 @@
#!/usr/bin/python3
from flipflop import WSGIServer
from flipflop import WSGIServer # type: ignore[import-untyped]
import sys
sys.path.append('/home/edward/src/agenda') # isort:skip
from web_view import app # isort:skip
if __name__ == '__main__':
sys.path.append("/home/edward/src/agenda") # isort:skip
from web_view import app # isort:skip
if __name__ == "__main__":
WSGIServer(app).run()

View file

@ -0,0 +1,152 @@
{% extends "base.html" %}
{% from "macros.html" import trip_link with context %}
{% block title %}{{ conf.name }} - Edward Betts{% endblock %}
{% block style %}
<style>
.conference-page h1 { overflow-wrap: anywhere; }
.conference-details dd { margin-bottom: 1rem; }
.flight-table td { vertical-align: middle; padding: .8rem .5rem; }
.flight-date { display: block; color: var(--bs-secondary-color, #6c757d); font-size: .85rem; white-space: nowrap; }
.flight-time { display: block; font-size: 1.1rem; font-weight: 600; }
.flight-price a { font-weight: 600; white-space: nowrap; }
@media (max-width: 575.98px) {
.conference-page { padding-left: 1rem; padding-right: 1rem; }
.conference-page h1 { font-size: 1.8rem; }
.conference-details { display: grid; grid-template-columns: minmax(0, 2fr) minmax(0, 3fr); gap: .5rem 1rem; margin: 0; }
.conference-details dt, .conference-details dd { width: auto; padding: 0; margin: 0; font-size: .9rem; }
#flight-search { flex-direction: column; align-items: stretch !important; gap: .75rem !important; }
#flight-search input { max-width: none !important; }
.flight-table, .flight-table tbody { display: block; width: 100%; }
.flight-table thead { display: none; }
.flight-table tr { display: grid; grid-template-columns: minmax(0, 1fr) minmax(0, 1fr); grid-template-areas: "route price" "departure arrival" "stops duration"; gap: .75rem 1rem; border: 1px solid #dee2e6; border-radius: .65rem; padding: 1rem; margin-bottom: .75rem; }
.flight-table td { display: block; min-width: 0; padding: 0; border: 0; box-shadow: none; background: transparent; overflow-wrap: anywhere; }
.flight-table td::before { content: attr(data-label); display: block; font-size: .75rem; color: var(--bs-secondary-color, #6c757d); margin-bottom: .2rem; }
.flight-departure { grid-area: departure; }
.flight-arrival { grid-area: arrival; }
.flight-route { grid-area: route; font-size: .9rem; }
.flight-stops { grid-area: stops; font-size: .85rem; }
.flight-duration { grid-area: duration; font-size: .85rem; }
.flight-price { grid-area: price; text-align: right; font-size: 1.15rem; }
.flight-price a { white-space: normal; }
.flight-route::before, .flight-price::before, .flight-stops::before, .flight-duration::before { display: none !important; }
.flight-date { font-size: .8rem; white-space: normal; }
}
</style>
{% endblock %}
{% block content %}
<div class="container py-4 conference-page">
<a href="{{ url_for('conference_list') }}">← Conferences</a>
<h1 class="mt-3">{{ conf.name }}</h1>
<p class="lead">{{ conf.display_date }}{% if not conf.has_exact_dates %} <span class="badge text-bg-secondary">{{ conf.date_status }}</span>{% endif %}</p>
{% if conf.url %}<p><a href="{{ conf.url }}" class="btn btn-outline-primary">Conference website ↗</a></p>{% endif %}
<div class="row g-4">
<div class="col-lg-6">
<h2 class="h4">Location and venue</h2>
<p>{{ conf.location }}{% if conf.country and get_country(conf.country) %} · {{ get_country(conf.country).flag }} {{ get_country(conf.country).name }}{% endif %}</p>
{% if conf.venue %}<p class="fw-semibold mb-1">{{ conf.venue }}</p>{% endif %}
{% if conf.address %}<p>{{ conf.address }}</p>{% endif %}
{% if conf.latitude is defined and conf.longitude is defined and conf.latitude is not none and conf.longitude is not none %}
<p><a href="https://www.openstreetmap.org/?mlat={{ conf.latitude }}&amp;mlon={{ conf.longitude }}#map=16/{{ conf.latitude }}/{{ conf.longitude }}">View venue on map</a> <span class="text-muted">({{ conf.latitude }}, {{ conf.longitude }})</span></p>
{% elif conf.venue or conf.address %}
<p><a href="https://www.google.com/maps/search/?api=1&amp;query={{ ((conf.venue or '') ~ ' ' ~ (conf.address or '') ~ ' ' ~ conf.location)|urlencode }}">Find venue on map</a></p>
{% endif %}
{% if conf.series_detail %}<p>Series: <a href="{{ url_for('conference_series_page', series_id=conf.series) }}">{{ conf.series_detail.name }}</a></p>{% endif %}
{% if conf.linked_trip %}<p>{{ trip_link(conf.linked_trip) }}</p>{% endif %}
</div>
<div class="col-lg-6">
<h2 class="h4">Conference details</h2>
<dl class="row conference-details">
{% for label, value in fields %}
<dt class="col-sm-4">{{ label }}</dt>
<dd class="col-sm-8 text-break">{% if value is sameas true %}Yes{% elif value is sameas false %}No{% elif value is string and (value.startswith('https://') or value.startswith('http://')) %}<a href="{{ value }}">{{ value }}</a>{% else %}{{ value|conference_date }}{% endif %}</dd>
{% endfor %}
</dl>
</div>
</div>
<section class="card mt-4"><div class="card-body">
<h2 class="h4">Flights</h2>
{% if cooldown %}<p class="alert alert-warning">{{ cooldown }}</p>{% endif %}
{% if eligible %}
{% if conf.date_status == 'tentative' %}<p class="text-muted">Using tentative conference dates: {{ conf.start_date|conference_date }} – {{ conf.end_date|conference_date }}.</p>{% endif %}
<p>{% if short_haul %}Direct flights from Bristol first. Start the day before and return the day after; if no flights are found, check progressively further dates, up to four days each side. Each direction stops at the nearest date with flights. London is checked if either direction has no Bristol flights.{% else %}Flights from London, departing two days before and returning the day after the conference. If none are found, try departing one day earlier. Direct or one-stop flights first.{% endif %}</p>
<form method="post" class="d-flex flex-wrap align-items-end gap-3" id="flight-search">
<input type="hidden" name="csrf_token" value="{{ csrf_token }}">
<div><label for="airport" class="form-label">Destination city or airport</label>
<input id="airport" name="airport" value="{{ airport }}" class="form-control" list="airport-suggestions" placeholder="Copenhagen, Denmark or CPH" maxlength="200" autocomplete="off" required style="max-width: 24rem" aria-describedby="airport-help"><datalist id="airport-suggestions"></datalist></div>
<button class="btn btn-primary" type="submit"{% if cooldown %} disabled{% endif %}>{{ 'Update flights' if cached else 'Look up flights' }}</button>
</form>
{% if resolved_airport %}<p class="small mt-2 mb-1">Destination: <strong>{{ resolved_name }}{% if resolved_name != resolved_airport %} ({{ resolved_airport }}){% endif %}</strong></p>{% endif %}
{% if default_google_flights and (not cached or not cached.searches) %}
<p class="mt-3"><a class="btn btn-outline-secondary" href="{{ default_google_flights.url }}" target="_blank" rel="noopener">Open in Google Flights ↗</a><span class="small text-muted d-block mt-1">Round trip: {{ default_google_flights.departure|conference_date }} – {{ default_google_flights.return|conference_date }}</span></p>
{% endif %}
{% if resolution_error %}<p class="text-danger small mt-2">{{ resolution_error }}</p>{% endif %}
<p id="airport-help" class="text-muted small mt-2">{% if suggested %}Suggested {{ suggested }}: {{ airport_reason }}. Enter a city, airport name or IATA code to choose another airport.{% else %}{{ airport_reason }}{% endif %}</p>
<p class="small text-muted">Direct flights first, with British Airways preferred within each stop count. Other airlines are included. Two-stop flights from London are checked only if no direct or one-stop options are found.</p>
<p class="small text-muted">Prices in GBP for one adult, economy. Each direction is priced separately. Results are cached for six hours; updates within that time reuse the cache.</p>
{% else %}
<p class="text-muted">Flight lookup is available for future conferences overseas or in Belfast with confirmed or tentative dates and an in-person venue.</p>
{% endif %}
{% if cached %}
<p>Last updated: <time class="cache-updated" datetime="{{ cached.updated_at }}">{{ cached.updated_at|conference_date }} UTC</time></p>
{% if cached.error %}<div class="alert alert-warning"><p class="mb-0">{{ cached.error }}{% if cached.searches %} Showing previously cached results.{% endif %}</p>{% if cached.error_details %}<p class="small text-break mt-2 mb-0">{{ cached.error_details }}</p>{% endif %}</div>{% endif %}
{% for search in cached.searches %}
<h3 class="h5 mt-4">{{ 'Bristol (BRS), direct only' if search.origin == 'BRS' else 'London (LON)' }} → {{ resolved_name or cached.destination }}{% if resolved_name and resolved_name != cached.destination %} ({{ cached.destination }}){% endif %}</h3>
<p><a class="btn btn-outline-secondary" href="{{ search.google_flights.url }}" target="_blank" rel="noopener">Open in Google Flights ↗</a><span class="small text-muted d-block mt-1">Round trip: {{ search.google_flights.departure|conference_date }} – {{ search.google_flights.return|conference_date }}</span></p>
{% for label, rows, days in [('Outbound', search.outbound, search.outbound_dates), ('Return', search.inbound, search.inbound_dates)] %}
<h4 class="h6">{{ label }} · one-way fares</h4>
{% if rows %}<p class="small text-muted mb-2">Departure and arrival times are local to each airport.</p>{% endif %}
{% if rows %}<div class="table-responsive"><table class="table table-sm flight-table">
<thead><tr><th>Departure</th><th>Arrival</th><th>Flight</th><th>Stops</th><th>Duration</th><th>Price</th></tr></thead><tbody>
{% for flight in rows %}<tr>
<td class="flight-departure" data-label="Departure"><time datetime="{{ flight.departure }}"><span class="flight-date">{{ flight.departure|conference_date('date') }}</span><span class="flight-time">{{ flight.departure|conference_date('time') }}</span></time></td>
<td class="flight-arrival" data-label="Arrival"><time datetime="{{ flight.arrival }}"><span class="flight-date">{{ flight.arrival|conference_date('date') }}</span><span class="flight-time">{{ flight.arrival|conference_date('time') }}</span></time></td>
<td class="flight-route" data-label="Flight">{% for leg in flight.legs %}{{ leg.airline }} {{ leg.number }} ({{ leg.origin }}–{{ leg.destination }}){% if not loop.last %}<br>{% endif %}{% endfor %}</td>
<td class="flight-stops" data-label="Stops">{{ 'Direct' if flight.stops == 0 else flight.stops|string ~ (' stop' if flight.stops == 1 else ' stops') }}</td>
<td class="flight-duration" data-label="Duration">{% if flight.duration is defined %}{{ flight.duration // 60 }}h {{ flight.duration % 60 }}m{% else %}—{% endif %}</td>
<td class="flight-price" data-label="Price"><a href="{{ flight.url }}">{% if flight.price is not none %}{{ '£' if flight.currency == 'GBP' else flight.currency ~ ' ' }}{{ '%.2f'|format(flight.price) }}{% else %}Check price{% endif %} ↗</a></td>
</tr>{% endfor %}
</tbody></table></div>{% else %}<p class="text-muted">{{ 'No direct flights found' if search.origin == 'BRS' else 'No suitable flights found' }}{% if days %} for departures on {{ days|map('conference_date')|join(', ') }}{% else %}: no eligible departure dates{% endif %}.{% if search.origin == 'BRS' and days %} Direct flights may not operate every day.{% endif %}</p>{% endif %}
{% endfor %}
{% endfor %}
{% endif %}
</div></section>
</div>
<script>
const airportInput = document.getElementById('airport');
const suggestions = document.getElementById('airport-suggestions');
let suggestionTimer;
let suggestionRequest;
if (airportInput) airportInput.addEventListener('input', function () {
clearTimeout(suggestionTimer);
if (suggestionRequest) suggestionRequest.abort();
suggestions.replaceChildren();
const query = airportInput.value.trim();
if (query.length < 2) return;
suggestionTimer = setTimeout(async function () {
suggestionRequest = new AbortController();
try {
const response = await fetch({{ url_for('conference.airport_search')|tojson }} + '?q=' + encodeURIComponent(query), {signal: suggestionRequest.signal});
const matches = await response.json();
if (airportInput.value.trim() !== query || !Array.isArray(matches)) return;
suggestions.replaceChildren(...matches.map(function (match) {
const option = document.createElement('option');
option.value = match.code;
option.label = match.name;
return option;
}));
} catch (error) {
// Typed names still resolve on submission if autocomplete is unavailable.
}
}, 250);
});
const form = document.getElementById('flight-search');
if (form) form.addEventListener('submit', function () {
const button = form.querySelector('button');
button.disabled = true;
button.textContent = 'Looking up flights…';
});
document.querySelectorAll('.cache-updated[datetime]').forEach(function (element) {
element.textContent = new Date(element.dateTime).toLocaleString('en-GB', {day: 'numeric', month: 'short', year: 'numeric', hour: '2-digit', minute: '2-digit', timeZoneName: 'short'});
});
</script>
{% endblock %}

View file

@ -189,8 +189,7 @@ tr.conf-hl > td {
style="left: {{ conf.left_pct }}%; width: {{ conf.width_pct }}%; top: {{ top_px }}px; background: {{ color }};"
title="{{ conf.label }}"
data-conf-key="{{ conf.key }}">
{% if conf.url %}<a href="{{ conf.url }}">{{ conf.name }}</a>
{% else %}<span>{{ conf.name }}</span>{% endif %}
<a href="{{ url_for('conference.page', event_date=conf.event_date, slug=conf.slug) }}">{{ conf.name }}</a>
</div>
{% endfor %}
@ -224,8 +223,7 @@ tr.conf-hl > td {
{% endif %}
</td>
<td class="conf-name" data-label="Conference">
{% if item.url %}<a href="{{ item.url }}">{{ item.name }}</a>
{% else %}{{ item.name }}{% endif %}
<a href="{{ url_for('conference.page', event_date=item.sort_date.isoformat(), slug=item.slug) }}">{{ item.name }}</a>
{% if item.going and not (item.accommodation_booked or item.travel_booked) %}
<span class="badge text-bg-primary ms-1">{{ badge }}</span>
{% endif %}

View file

@ -54,8 +54,7 @@
{% endif %}
</td>
<td>
{% if item.url %}<a href="{{ item.url }}">{{ item.name }}</a>
{% else %}{{ item.name }}{% endif %}
<a href="{{ url_for('conference.page', event_date=item.sort_date.isoformat(), slug=item.slug) }}">{{ item.name }}</a>
</td>
<td>
{% set country = get_country(item.country) if item.country else None %}

View file

@ -0,0 +1,587 @@
"""Conference detail pages, airport resolution and flight caching."""
import typing
from datetime import date, datetime, timedelta, timezone
from pathlib import Path
import pytest
import yaml
import agenda.conference_detail as detail
import agenda.conference_flights as flights
import agenda.conference_list
import agenda.conference_page as page
import agenda.flight_search_cache as cache
import agenda.trip
import web_view
from agenda.types import StrDict
def test_airport_suggestions(tmp_path: Path) -> None:
"""Mappings can be overridden; proximity stays within the conference country."""
conf: StrDict = {"location": "Brussels", "country": "be"}
assert detail.suggest_airport(conf, str(tmp_path))[0] == "BRU"
(tmp_path / "conference_airports.yaml").write_text("be:brussels: CRL\n")
assert detail.suggest_airport(conf, str(tmp_path))[0] == "CRL"
(tmp_path / "airports.yaml").write_text(
yaml.safe_dump(
{
"AAA": {
"city": "Elsewhere",
"country": "fr",
"latitude": 50,
"longitude": 4,
},
"BBB": {
"city": "Another",
"country": "be",
"latitude": 50.1,
"longitude": 4.1,
},
}
)
)
assert (
detail.suggest_airport(
{"location": "Town", "country": "be", "latitude": 50, "longitude": 4},
str(tmp_path),
)[0]
== "BBB"
)
assert detail.suggest_airport({"location": "TBC"}, str(tmp_path))[0] is None
def test_short_haul() -> None:
"""European geography determines Bristol preference, with a distance limit."""
assert detail.is_short_haul({"country": "BE"})
assert not detail.is_short_haul({"country": "us"})
assert not detail.is_short_haul({"country": "fr", "latitude": -21, "longitude": 55})
def test_page_is_cache_only(tmp_path: Path, monkeypatch: typing.Any) -> None:
"""GET displays venue/details without initiating a search; POST requires a token."""
conf: StrDict = {
"name": "Mapping 2099",
"location": "Brussels",
"country": "be",
"topic": "Maps",
"start": date(2099, 5, 5),
"end": date(2099, 5, 6),
"venue": "Test Venue",
"address": "123 Test Street",
"latitude": 50.8,
"longitude": 4.3,
"url": "https://example.com/conference",
"description": "A great conference",
}
(tmp_path / "conferences.yaml").write_text(yaml.safe_dump([conf]))
monkeypatch.setitem(web_view.app.config, "PERSONAL_DATA", str(tmp_path))
monkeypatch.setitem(web_view.app.config, "DATA_DIR", str(tmp_path))
monkeypatch.setattr(agenda.trip, "build_trip_list", lambda: [])
monkeypatch.setattr(
flights, "lookup", lambda *args: pytest.fail("GET made a lookup")
)
monkeypatch.setattr(flights, "valid_airport", lambda code: code == "BRU")
url = f"/conference/2099-05-05/{detail.conference_slug(conf)}"
with web_view.app.test_client() as client:
response = client.get(url)
assert response.status_code == 200
for value in (
b"Test Venue",
b"123 Test Street",
b"View venue on map",
b"A great conference",
b"Look up flights",
b"Conference website",
):
assert value in response.data
assert client.post(url, data={"airport": "BRU"}).status_code == 400
assert client.get("/conference/2099-05-05/missing").status_code == 404
cached: StrDict = {
"updated_at": datetime.now(timezone.utc).isoformat(),
"destination": "BRU",
"searches": [],
}
path = flights.cache_path(
str(tmp_path), conf["start"], conf["end"], "BRU", True
)
path.parent.mkdir()
flights.write_cache(path, cached)
response = client.get(url)
assert b"Last updated" in response.data and b"Update flights" in response.data
with client.session_transaction() as session:
token = session["conference_flight_token"]
calls: list[str] = []
def lookup(*args: typing.Any) -> StrDict:
calls.append(args[3])
return cached
monkeypatch.setattr(flights, "lookup", lookup)
assert (
client.post(url, data={"airport": "BRU", "csrf_token": token}).status_code
== 302
)
assert calls == ["BRU"]
monkeypatch.setattr(
flights,
"resolve_airport",
lambda query: ("CPH", "Copenhagen Kastrup Airport"),
)
response = client.post(
url + "?airport=BRU",
data={"airport": "Copenhagen, Denmark", "csrf_token": token},
)
assert response.status_code == 302
assert response.headers["Location"].endswith("airport=CPH")
assert calls == ["BRU", "CPH"]
def test_slug_collision(tmp_path: Path) -> None:
"""Events of one series on one day receive distinct reproducible URLs."""
conferences = [
{
"name": name,
"series": "maps",
"location": "Berlin",
"start": date(2099, 5, 1),
}
for name in ("Maps morning", "Maps evening")
]
(tmp_path / "conferences.yaml").write_text(yaml.safe_dump(conferences))
rows = agenda.conference_list.build_conference_list(str(tmp_path), [])
assert rows[0]["slug"] != rows[1]["slug"]
assert (
detail.conference_slug({"name": "FOSDEM", "location": "Brussels"})
== "fosdem-brussels"
)
@pytest.mark.parametrize(
"short_haul,bristol_complete,expected",
[(True, True, ["BRS"]), (True, False, ["BRS", "LON"]), (False, True, ["LON"])],
)
def test_cache_and_fallback(
tmp_path: Path,
monkeypatch: typing.Any,
short_haul: bool,
bristol_complete: bool,
expected: list[str],
) -> None:
"""Bristol preference, London fallback, and refresh throttling."""
calls: list[str] = []
def search(
origin: str, destination: str, start: date, end: date, flexible: bool
) -> StrDict:
calls.append(origin)
return {
"origin": origin,
"outbound": [{"price": 50}],
"inbound": [{"price": 60}] if bristol_complete else [],
}
monkeypatch.setattr(flights, "search_origin", search)
start, end = date(2099, 1, 5), date(2099, 1, 6)
path = flights.cache_path(str(tmp_path), start, end, "BRU", short_haul)
result = flights.lookup(path, start, end, "BRU", short_haul)
assert calls == expected
assert flights.lookup(path, start, end, "BRU", short_haul) == result
assert calls == expected
def test_failed_refresh_keeps_results(tmp_path: Path, monkeypatch: typing.Any) -> None:
"""Failures retain old results and get a short retry cooldown."""
path = flights.cache_path(
str(tmp_path), date(2099, 1, 5), date(2099, 1, 6), "BRU", True
)
path.parent.mkdir()
old: StrDict = {
"updated_at": (datetime.now(timezone.utc) - timedelta(days=1)).isoformat(),
"destination": "BRU",
"searches": [{"origin": "BRS", "outbound": [1], "inbound": [2]}],
}
flights.write_cache(path, old)
calls: list[bool] = []
def fail(*args: typing.Any, **kwargs: typing.Any) -> StrDict:
calls.append(True)
raise RuntimeError("network unavailable")
monkeypatch.setattr(flights, "search_origin", fail)
result = flights.lookup(path, date(2099, 1, 5), date(2099, 1, 6), "BRU", True)
assert result["updated_at"] == old["updated_at"]
assert result["searches"] == old["searches"]
assert result["error"]
assert "RuntimeError: network unavailable" in result["error_details"]
flights.lookup(path, date(2099, 1, 5), date(2099, 1, 6), "BRU", True)
assert len(calls) == 1
@pytest.mark.parametrize(
"outbound_offset,inbound_offset,expected_calls",
[(1, 1, 2), (3, 1, 4), (1, 3, 4), (None, None, 8)],
)
def test_flexible_direct_dates(
monkeypatch: typing.Any,
outbound_offset: int | None,
inbound_offset: int | None,
expected_calls: int,
) -> None:
"""Each direction expands only as needed, excluding late outbound arrivals."""
calls: list[tuple[str, str, date, bool]] = []
start, end = date(2099, 1, 5), date(2099, 1, 6)
def search(origin: str, destination: str, day: date, direct: bool) -> list[StrDict]:
calls.append((origin, destination, day, direct))
if origin == "BRS":
offset = (start - day).days
if offset == 1 and outbound_offset != 1:
return [{"arrival": start.isoformat() + "T18:00:00", "price": 30}]
available = offset == outbound_offset
else:
available = (day - end).days == inbound_offset
return (
[{"arrival": day.isoformat() + "T18:00:00", "price": 30}]
if available
else []
)
monkeypatch.setattr(flights, "search_day", search)
result = flights.search_origin("BRS", "BRU", start, end, True)
assert len(calls) == expected_calls
assert all(call[3] for call in calls)
assert bool(result["outbound"]) == (outbound_offset is not None)
assert bool(result["inbound"]) == (inbound_offset is not None)
assert sum(call[0] == "BRS" for call in calls) == (outbound_offset or 4)
assert sum(call[0] == "BRU" for call in calls) == (inbound_offset or 4)
def test_bad_cache(tmp_path: Path) -> None:
"""Invalid JSON and naive timestamps are ignored."""
path = tmp_path / "cache.json"
path.write_text('{"updated_at": "2026-01-01T12:00:00", "searches": []}')
assert flights.read_cache(path) is None
path.write_text("bad json")
assert flights.read_cache(path) is None
def test_airport_autocomplete(monkeypatch: typing.Any) -> None:
"""Autocomplete uses the local index and returns airport names and codes."""
queries: list[str] = []
def matches(query: str) -> list[StrDict]:
queries.append(query)
return [{"code": "CPH", "name": "Copenhagen Kastrup Airport"}]
monkeypatch.setattr(flights, "airport_matches", matches)
with web_view.app.test_client() as client:
assert client.get("/conference/airports?q=C").json == []
response = client.get("/conference/airports?q=Copenhagen%2C%20Denmark")
assert response.json == [{"code": "CPH", "name": "Copenhagen Kastrup Airport"}]
assert queries == ["Copenhagen, Denmark"]
def test_multi_airport_location_mapping(tmp_path: Path) -> None:
"""Mappings support airport lists, accents, and personal overrides."""
for location in ("Malmö", "Malmo"):
assert (
detail.suggest_airport(
{"country": "se", "location": location}, str(tmp_path)
)[0]
== "MMX,CPH"
)
(tmp_path / "conference_airports.yaml").write_text('"se:malmo": [CPH, MMX]\n')
assert (
detail.suggest_airport({"country": "se", "location": "Malmö"}, str(tmp_path))[0]
== "CPH,MMX"
)
@pytest.mark.parametrize("available_offset", [2, 3, None])
def test_london_earlier_departure(
monkeypatch: typing.Any, available_offset: int | None
) -> None:
"""London expands the outbound by one day, without repeating the return search."""
start, end = date(2099, 1, 5), date(2099, 1, 6)
calls: list[tuple[str, date]] = []
def search(origin: str, destination: str, day: date, direct: bool) -> list[StrDict]:
calls.append((origin, day))
if origin == "LON" and (start - day).days != available_offset:
return []
return [{"arrival": day.isoformat() + "T18:00:00"}]
monkeypatch.setattr(flights, "search_day", search)
result = flights.search_origin("LON", "BKW", start, end, False)
expected = ["2099-01-03"] if available_offset == 2 else ["2099-01-03", "2099-01-02"]
assert result["outbound_dates"] == expected
assert result["inbound_dates"] == ["2099-01-07"]
assert sum(origin == "BKW" for origin, _ in calls) == 1
assert bool(result["outbound"]) == (available_offset is not None)
def test_legacy_empty_search_can_expand(
tmp_path: Path, monkeypatch: typing.Any
) -> None:
"""An old fresh empty London cache can immediately try the new earlier date."""
start, end = date(2099, 1, 5), date(2099, 1, 6)
path = flights.cache_path(str(tmp_path), start, end, "BKW", False)
path.parent.mkdir()
flights.write_cache(
path,
{
"updated_at": datetime.now(timezone.utc).isoformat(),
"searches": [{"origin": "LON", "outbound": [], "inbound": []}],
"transport": cache.TRANSPORT,
},
)
calls: list[date] = []
def search(origin: str, destination: str, day: date, direct: bool) -> list[StrDict]:
calls.append(day)
return []
monkeypatch.setattr(flights, "search_day", search)
result = flights.lookup(path, start, end, "BKW", False)
assert result["searches"][0]["outbound_dates"] == ["2099-01-03", "2099-01-02"]
assert len(calls) == 3
assert flights.lookup(path, start, end, "BKW", False) == result
assert len(calls) == 3
def test_empty_flights_show_airport_and_dates(
tmp_path: Path, monkeypatch: typing.Any
) -> None:
"""Cached empty results show full airport names and checked departure dates."""
conf: StrDict = {
"name": "Mapping 2099",
"location": "Raleigh",
"country": "us",
"start": date(2099, 5, 5),
"end": date(2099, 5, 6),
}
(tmp_path / "conferences.yaml").write_text(yaml.safe_dump([conf]))
monkeypatch.setitem(web_view.app.config, "PERSONAL_DATA", str(tmp_path))
monkeypatch.setitem(web_view.app.config, "DATA_DIR", str(tmp_path))
monkeypatch.setattr(agenda.trip, "build_trip_list", lambda: [])
path = flights.cache_path(str(tmp_path), conf["start"], conf["end"], "BKW", False)
path.parent.mkdir()
cached: StrDict = {
"updated_at": datetime.now(timezone.utc).isoformat(),
"destination": "BKW",
"searches": [
{
"origin": "LON",
"outbound": [],
"inbound": [],
"outbound_dates": ["2099-05-03", "2099-05-02"],
"inbound_dates": ["2099-05-07"],
}
],
}
flights.write_cache(path, cached)
url = "/conference/2099-05-05/" + detail.conference_slug(conf) + "?airport=BKW"
with web_view.app.test_client() as client:
response = client.get(url)
assert response.status_code == 200
assert b"Raleigh County Memorial Airport (BKW)</h3>" in response.data
for day in ("2099-05-03", "2099-05-02", "2099-05-07"):
assert page.format_date(day).encode() in response.data
# Legacy cache data reconstructs the single London departure date.
del cached["searches"][0]["outbound_dates"]
del cached["searches"][0]["inbound_dates"]
flights.write_cache(path, cached)
response = client.get(url)
assert page.format_date("2099-05-03").encode() in response.data
@pytest.mark.parametrize(
"location,online,eligible",
[("Belfast", False, True), ("Belfast", True, False), ("London", False, False)],
)
def test_belfast_domestic_exception(
tmp_path: Path, monkeypatch: typing.Any, location: str, online: bool, eligible: bool
) -> None:
"""Belfast alone permits UK flight lookups; online events remain excluded."""
conf: StrDict = {
"name": "Test conference",
"location": location,
"country": "gb",
"start": date(2099, 5, 5),
"end": date(2099, 5, 6),
"online": online,
}
(tmp_path / "conferences.yaml").write_text(yaml.safe_dump([conf]))
monkeypatch.setitem(web_view.app.config, "PERSONAL_DATA", str(tmp_path))
monkeypatch.setitem(web_view.app.config, "DATA_DIR", str(tmp_path))
monkeypatch.setattr(agenda.trip, "build_trip_list", lambda: [])
calls: list[tuple[str, bool]] = []
def lookup(
path: Path, start: date, end: date, destination: str, short_haul: bool
) -> StrDict:
calls.append((destination, short_haul))
return {}
monkeypatch.setattr(flights, "lookup", lookup)
url = "/conference/2099-05-05/" + detail.conference_slug(conf)
with web_view.app.test_client() as client:
response = client.get(url)
assert response.status_code == 200
assert (b"Look up flights" in response.data) == eligible
assert calls == []
if eligible:
assert b"BFS,BHD" in response.data
with client.session_transaction() as session:
token = session["conference_flight_token"]
response = client.post(url, data={"airport": location, "csrf_token": token})
assert response.status_code == (302 if eligible else 400)
assert calls == ([("BFS,BHD", True)] if eligible else [])
def test_personal_airport_names(tmp_path: Path, monkeypatch: typing.Any) -> None:
"""Personal names apply to codes, groups and autocomplete, with live updates."""
path = tmp_path / "airports.yaml"
path.write_text(
yaml.safe_dump({"BSL": {"name": "My Basel Airport"}, "BRN": {"name": ""}})
)
monkeypatch.setitem(web_view.app.config, "PERSONAL_DATA", str(tmp_path))
with web_view.app.app_context():
assert flights.resolve_airport("BSL") == ("BSL", "My Basel Airport")
code, name = flights.resolve_airport("Bern")
assert code == "BRN,BSL"
assert name == "Bern Airport / My Basel Airport"
assert flights.resolve_airport("BRN,BSL")[1] == name
assert flights.airport_matches("My Basel Airport")[0] == {
"code": "BSL",
"name": "My Basel Airport",
}
assert any(
match["code"] == "BSL" for match in flights.airport_matches("EuroAirport")
)
with web_view.app.test_client() as client:
assert client.get("/conference/airports?q=BSL").json == [
{"code": "BSL", "name": "My Basel Airport"}
]
path.write_text(yaml.safe_dump({"BSL": {"name": "Renamed Basel Airport"}}))
assert client.get("/conference/airports?q=BSL").json == [
{"code": "BSL", "name": "Renamed Basel Airport"}
]
path.unlink()
fallback = client.get("/conference/airports?q=BSL").json
assert isinstance(fallback, list)
assert "EuroAirport" in fallback[0]["name"]
def test_date_limit_banner(tmp_path: Path, monkeypatch: typing.Any) -> None:
"""A wrapped date rejection asks to retry nearer departure, keeping cached fares."""
from agenda.google_flights import FlightDateUnavailableError
start, end = date(2099, 1, 5), date(2099, 1, 6)
path = flights.cache_path(str(tmp_path), start, end, "AKJ", False)
path.parent.mkdir()
old: StrDict = {
"updated_at": (datetime.now(timezone.utc) - timedelta(days=1)).isoformat(),
"destination": "AKJ",
"searches": [{"origin": "LON", "outbound": [{"price": 100}], "inbound": []}],
}
flights.write_cache(path, old)
def fail(*args: typing.Any, **kwargs: typing.Any) -> StrDict:
try:
raise FlightDateUnavailableError(
"Requested flight date is too far in the future. Please try again closer to departure."
)
except FlightDateUnavailableError as exc:
raise RuntimeError("AKJ → LON on 2099-01-07") from exc
monkeypatch.setattr(flights, "search_origin", fail)
result = flights.lookup(path, start, end, "AKJ", False)
assert "closer to departure" in result["error"]
assert "15 minutes" not in result["error"]
assert result["searches"] == old["searches"]
assert result["updated_at"] == old["updated_at"]
def test_previous_parser_error_can_retry_once(
tmp_path: Path, monkeypatch: typing.Any
) -> None:
"""The parser fix unlocks a recent old-format error, but fresh errors still throttle."""
start, end = date(2099, 6, 2), date(2099, 6, 4)
path = flights.cache_path(str(tmp_path), start, end, "BLQ", True)
path.parent.mkdir()
flights.write_cache(
path,
{
"updated_at": datetime.now(timezone.utc).isoformat(),
"searches": [],
"error": "Flight lookup failed",
"error_details": "ValueError: Unrecognized Google Flights result format",
"transport": cache.TRANSPORT,
},
)
calls: list[bool] = []
def fail(*args: typing.Any, **kwargs: typing.Any) -> StrDict:
calls.append(True)
raise ValueError("Unrecognized Google Flights result format")
monkeypatch.setattr(flights, "search_origin", fail)
refreshed = flights.lookup(path, start, end, "BLQ", True)
assert len(calls) == 1
assert flights.lookup(path, start, end, "BLQ", True) == refreshed
assert len(calls) == 1
@pytest.mark.parametrize(
"status,eligible", [("exact", True), ("tentative", True), ("approximate", False)]
)
def test_flight_lookup_with_tentative_dates(
tmp_path: Path, monkeypatch: typing.Any, status: str, eligible: bool
) -> None:
"""Tentative start/end dates support lookup and links; approximate ranges don't."""
start, end = date(2099, 6, 2), date(2099, 6, 4)
dates: StrDict = {"status": status}
dates.update(
{"earliest": start, "latest": end}
if status == "approximate"
else {"start": start, "end": end}
)
conf: StrDict = {
"name": "PyCon test",
"location": "Bologna",
"country": "it",
"dates": dates,
}
(tmp_path / "conferences.yaml").write_text(yaml.safe_dump([conf]))
monkeypatch.setitem(web_view.app.config, "PERSONAL_DATA", str(tmp_path))
monkeypatch.setitem(web_view.app.config, "DATA_DIR", str(tmp_path))
monkeypatch.setattr(agenda.trip, "build_trip_list", lambda: [])
calls: list[tuple[date, date, str]] = []
def lookup(
path: Path, departure: date, returning: date, destination: str, short_haul: bool
) -> StrDict:
calls.append((departure, returning, destination))
return {}
monkeypatch.setattr(flights, "lookup", lookup)
url = "/conference/2099-06-02/" + detail.conference_slug(conf)
with web_view.app.test_client() as client:
response = client.get(url)
assert response.status_code == 200
assert (b"Look up flights" in response.data) == eligible
assert (b"Open in Google Flights" in response.data) == eligible
assert (b"Using tentative conference dates" in response.data) == (
status == "tentative"
)
if status == "tentative":
assert page.format_date(start).encode() in response.data
assert page.format_date(end).encode() in response.data
assert calls == []
with client.session_transaction() as session:
token = session["conference_flight_token"]
response = client.post(url, data={"airport": "BLQ", "csrf_token": token})
assert response.status_code == (302 if eligible else 400)
assert calls == ([(start, end, "BLQ")] if eligible else [])

View file

@ -0,0 +1,148 @@
"""Offline failure recording, replay and retention coverage."""
import gzip
import json
import os
import stat
import time
import typing
from datetime import date
from pathlib import Path
from types import SimpleNamespace
import pytest
from playwright.sync_api import Page
from agenda import flight_diagnostics, flight_search_cache, google_flights
def read_snapshot(path: Path) -> dict[str, typing.Any]:
"""Read the same artifact an administrator can use for analysis."""
with gzip.open(path, "rt", encoding="utf-8") as source:
return typing.cast(dict[str, typing.Any], json.load(source))
def test_failed_page_saved_for_replay(tmp_path: Path, monkeypatch: typing.Any) -> None:
"""A parser failure saves the route, raw script and loaded response content."""
script = "AF_initDataCallback({data:[]});"
page_text = "Something went wrong"
html = "<html><body>Something went wrong</body></html>"
url = google_flights.search_url("AKJ", "LON", date(2099, 9, 12), 1)
page = SimpleNamespace(
url=url,
goto=lambda *args, **kwargs: SimpleNamespace(status=200),
content=lambda: html,
locator=lambda selector: SimpleNamespace(
wait_for=lambda **kwargs: None,
inner_text=lambda **kwargs: page_text if selector == "body" else script,
),
)
browser = google_flights.BrowserSearch(tmp_path)
browser.page = typing.cast(Page, page)
monkeypatch.setattr(browser, "start", lambda: browser.page)
with pytest.raises(ValueError, match="unusable search data"):
browser.search("AKJ", "LON", date(2099, 9, 12), 1)
path = next((tmp_path / "errors").glob("*.json.gz"))
snapshot = read_snapshot(path)
assert snapshot["origin"] == "AKJ"
assert snapshot["destination"] == "LON"
assert snapshot["departure_date"] == "2099-09-12"
assert snapshot["http_status"] == 200
assert snapshot["requested_url"] == snapshot["page_url"] == url
assert snapshot["flight_script"] == script
assert snapshot["page_html"] == html
assert snapshot["page_text"] == page_text
assert snapshot["truncated_fields"] == []
assert stat.S_IMODE(path.stat().st_mode) == 0o640
with pytest.raises(ValueError, match="Unrecognized"):
google_flights.parse_results(
snapshot["flight_script"], snapshot["requested_url"]
)
def test_recording_failure_preserves_lookup_error(
tmp_path: Path, monkeypatch: typing.Any
) -> None:
"""Unavailable diagnostic storage cannot hide the real lookup error."""
browser = google_flights.BrowserSearch(tmp_path)
def fail(*args: typing.Any) -> typing.Any:
raise ValueError("Original flight failure")
def cannot_record(*args: typing.Any) -> None:
raise PermissionError("No access to diagnostics")
monkeypatch.setattr(browser, "perform_search", fail)
monkeypatch.setattr(browser, "record_failure", cannot_record)
with pytest.raises(ValueError, match="Original flight failure"):
browser.search("AKJ", "LON", date(2099, 9, 12), 1)
def test_snapshot_retention_and_truncation(
tmp_path: Path, monkeypatch: typing.Any
) -> None:
"""Old and excess snapshots are removed; large raw captures have marked limits."""
monkeypatch.setattr(flight_diagnostics, "MAX_RECORDS", 2)
first = flight_diagnostics.write_failure(tmp_path, {"error": "old"})
assert first is not None
old = time.time() - flight_diagnostics.RETENTION_SECONDS - 1
os.utime(first, (old, old))
second = flight_diagnostics.write_failure(tmp_path, {"error": "recent"})
assert second is not None and not first.exists()
third = flight_diagnostics.write_failure(tmp_path, {"error": "recent"})
fourth = flight_diagnostics.write_failure(
tmp_path, {"flight_script": "x" * (1024 * 1024 + 1)}
)
assert third is not None and fourth is not None
assert not second.exists()
assert len(list((tmp_path / "errors").glob("*.json.gz"))) == 2
snapshot = read_snapshot(fourth)
assert snapshot["truncated_fields"] == ["flight_script"]
assert len(snapshot["flight_script"].encode()) == 1024 * 1024
def test_storage_error_is_best_effort(tmp_path: Path) -> None:
"""An unwritable destination logs a warning and returns without raising."""
(tmp_path / "errors").write_text("not a directory")
assert flight_diagnostics.write_failure(tmp_path, {"error": "original"}) is None
def test_http_error_captured_once_during_cooldown(
tmp_path: Path, monkeypatch: typing.Any
) -> None:
"""Capture a received 429 body, but don't save duplicate local cooldown refusals."""
calls: list[str] = []
url = "https://www.google.com/travel/flights/search"
body = "Too Many Requests"
def goto(requested_url: str, **kwargs: typing.Any) -> typing.Any:
calls.append(requested_url)
return SimpleNamespace(status=429, url=url, text=lambda: body)
page = SimpleNamespace(
url=url,
goto=goto,
content=lambda: body,
locator=lambda selector: SimpleNamespace(inner_text=lambda **kwargs: body),
)
browser = google_flights.BrowserSearch(tmp_path)
browser.page = typing.cast(Page, page)
monkeypatch.setattr(browser, "start", lambda: browser.page)
for _ in range(2):
with pytest.raises(flight_search_cache.RateLimitCooldownError):
browser.search("AKJ", "LON", date(2099, 9, 12), 1)
records = list((tmp_path / "errors").glob("*.json.gz"))
assert len(records) == len(calls) == 1
snapshot = read_snapshot(records[0])
assert snapshot["http_status"] == 429
assert snapshot["http_errors"] == [{"status": 429, "url": url, "body": body}]
def test_success_does_not_create_diagnostics(
tmp_path: Path, monkeypatch: typing.Any
) -> None:
"""Successful searches, including empty results, do not accumulate snapshots."""
browser = google_flights.BrowserSearch(tmp_path)
monkeypatch.setattr(browser, "perform_search", lambda *args: [])
assert browser.search("AKJ", "LON", date(2099, 9, 12), 1) == []
assert not (tmp_path / "errors").exists()

View file

@ -0,0 +1,83 @@
"""Offline tests for shared flight pacing, cooldowns and route/date caching."""
import typing
from datetime import date
from pathlib import Path
from types import SimpleNamespace
import pytest
import agenda.conference_flights as flights
import agenda.flight_search_cache as cache
from agenda.types import StrDict
@pytest.fixture
def clock(monkeypatch: typing.Any) -> tuple[list[float], list[float]]:
"""Advance simulated time instead of sleeping or contacting Google."""
now = [1000.0]
sleeps: list[float] = []
def current_time() -> float:
return now[0]
def sleep(seconds: float) -> None:
sleeps.append(seconds)
now[0] += seconds
monkeypatch.setattr(cache, "time", SimpleNamespace(time=current_time, sleep=sleep))
return now, sleeps
def test_shared_request_spacing(
tmp_path: Path, clock: tuple[list[float], list[float]]
) -> None:
"""Separate callers share the same three-second spacing state."""
cache.wait_for_request(tmp_path)
cache.wait_for_request(tmp_path)
cache.wait_for_request(tmp_path)
assert clock[1] == [3.0, 3.0]
assert cache.read_state(tmp_path / "rate-limit.json")["last_request_at"] == 1006.0
@pytest.mark.parametrize("rows", [[], [{"price": 50}]])
def test_day_cache_reuses_results_during_cooldown(
tmp_path: Path, clock: tuple[list[float], list[float]], rows: list[StrDict]
) -> None:
"""Successful and empty route/date results survive a global request cooldown."""
calls: list[bool] = []
def fetch() -> list[StrDict]:
calls.append(True)
return rows
assert cache.cached_day(tmp_path, "BRS", "CPH", "2099-01-04", True, fetch) == rows
cache.block_requests(tmp_path)
assert cache.cached_day(tmp_path, "BRS", "CPH", "2099-01-04", True, fetch) == rows
assert len(calls) == 1
clock[0][0] += cache.DAY_CACHE_SECONDS + 1
cache.cached_day(tmp_path, "BRS", "CPH", "2099-01-04", True, fetch)
assert len(calls) == 2
def test_overlapping_conferences_reuse_day_cache(
tmp_path: Path, monkeypatch: typing.Any
) -> None:
"""Two conferences sharing a departure date fetch that outbound only once."""
calls: list[tuple[str, str, date]] = []
def fetch(
origin: str, destination: str, day: date, direct: bool, directory: Path
) -> list[StrDict]:
calls.append((origin, destination, day))
return [{"arrival": day.isoformat() + "T18:00:00", "price": 50}]
monkeypatch.setattr(flights, "fetch_day", fetch)
start = date(2099, 1, 5)
for end in (date(2099, 1, 6), date(2099, 1, 7)):
path = flights.cache_path(str(tmp_path), start, end, "CPH", False)
result = flights.lookup(path, start, end, "CPH", False)
assert result["searches"][0]["outbound"]
assert len(calls) == 3
assert sum(origin == "LON" for origin, _, _ in calls) == 1
assert cache.search_directory.get() is None

View file

@ -1,12 +1,14 @@
"""Check automatic Geomob recording and failure notifications."""
import subprocess
from pathlib import Path
from datetime import date
from unittest.mock import patch
import pytest
import yaml
from agenda.geomob import GeomobEvent, geomob_email, record_event
from agenda.geomob import GeomobEvent, conference_page_links, geomob_email, record_event
EVENT = GeomobEvent(date(2026, 10, 15), "/post/oct-2026", "#geomobLON")
BASE_URL = "https://thegeomob.com"
@ -87,3 +89,44 @@ def test_duplicate_event() -> None:
assert "Already recorded" in record_event(EVENT, BASE_URL)
assert run.call_count == 4
assert run.call_args.args[0] == ["git", "push"]
def test_email_keeps_event_url_and_adds_conference_page(tmp_path: Path) -> None:
"""A saved event links to its canonical agenda page alongside the original URL."""
event = GeomobEvent(
date(2026, 10, 28), "/post/oct-28th-2026-geomobcgn-details", "#geomobCGN"
)
conference = {
"name": "Geomob Cologne",
"location": "Cologne",
"country": "de",
"dates": {"status": "exact", "start": event.date, "end": event.date},
"url": BASE_URL + event.href,
}
(tmp_path / "conferences.yaml").write_text(yaml.safe_dump([conference]))
links = conference_page_links([event], BASE_URL, str(tmp_path))
expected = (
"https://edwardbetts.com/agenda/conference/2026-10-28/geomob-cologne-cologne"
)
assert links == {event: expected}
subject, body = geomob_email(
[event], BASE_URL, {event: "Recorded and pushed."}, links
)
assert subject == "1 New Geomob Event(s) Announced"
assert "URL: https://thegeomob.com/post/oct-28th-2026-geomobcgn-details" in body
assert f"Conference page: {expected}" in body
assert "Recording: Recorded and pushed." in body
changed_date = GeomobEvent(date(2026, 10, 29), event.href, event.hashtag)
assert conference_page_links([changed_date], BASE_URL, str(tmp_path)) == {}
def test_missing_conference_data_does_not_hide_event(tmp_path: Path) -> None:
"""Unrecorded events still generate their original notification without a broken link."""
links = conference_page_links([EVENT], BASE_URL, str(tmp_path))
assert links == {}
_, body = geomob_email(
[EVENT], BASE_URL, {EVENT: "FAILED at add conference"}, links
)
assert "URL: " + BASE_URL + EVENT.href in body
assert "FAILED at add conference" in body
assert "Conference page:" not in body

View file

@ -0,0 +1,335 @@
"""Offline browser and parser coverage, with no requests to Google."""
import base64
import json
import typing
from datetime import date
from pathlib import Path
from types import SimpleNamespace
from urllib.parse import parse_qs, urlsplit
import pytest
from playwright.sync_api import Page, Response
from agenda import (
airport_lookup,
conference_flights,
flight_search_cache,
google_flights,
)
def itinerary(airline: str = "BA", price: int | None = 424) -> list[typing.Any]:
"""A Google itinerary with omitted zero minutes and overnight arrival."""
leg: list[typing.Any] = [None] * 23
leg[3], leg[6] = "LHR", "LAX"
leg[8], leg[10] = [23], [None, 30]
leg[20], leg[21] = [2027, 3, 30], [2027, 3, 31]
leg[22] = [airline, "135", "BA"]
details: list[typing.Any] = [None] * 10
details[2], details[9] = [leg], 670
return [details, [[None, price]] if price is not None else None]
def result_script() -> str:
"""Represent best and other groups, airline metadata, and an unpriced fare."""
payload: list[typing.Any] = [None] * 8
payload[2] = [[itinerary()]]
payload[3] = [[itinerary("AA", None)]]
payload[7] = [None, [[], [["BA", "British Airways"], ["AA", "American Airlines"]]]]
return (
"AF_initDataCallback({key: 'ds:1', data:"
+ json.dumps(payload)
+ ", sideChannel: {}});"
)
def test_results_and_sparse_times() -> None:
"""Best flights retain order; codeshares, missing fares and midnight parse."""
rows = google_flights.parse_results(result_script(), "https://example.com")
assert [row["price"] for row in rows] == [424, None]
assert rows[0]["departure"] == "2027-03-30T23:00:00"
assert rows[0]["arrival"] == "2027-03-31T00:30:00"
assert rows[0]["currency"] == "GBP"
assert rows[0]["duration"] == 670
assert rows[1]["legs"][0]["airline"] == "American Airlines"
assert rows[1]["legs"][0]["operating_airline_code"] == "BA"
assert conference_flights.flight_rank(rows[1]) == (0, 0)
@pytest.mark.parametrize(
"script", ["no data", "data:{}", "data:[null,null,[[[[]]]],null]"]
)
def test_unrecognized_data_is_error(script: str) -> None:
"""Broken parsing must not silently cache a successful empty search."""
with pytest.raises(ValueError):
google_flights.parse_results(script, "https://example.com")
def test_search_url() -> None:
"""London includes all six airports; locale, currency and date are explicit."""
url = google_flights.search_url("LON", "LAX", date(2027, 3, 30), 1)
query = parse_qs(urlsplit(url).query)
assert query["hl"] == ["en-GB"]
assert query["gl"] == ["GB"]
assert query["curr"] == ["GBP"]
encoded = query["tfs"][0]
data = base64.urlsafe_b64decode(encoded + "=" * (-len(encoded) % 4))
assert all(
code.encode() in data for code in (*google_flights.LONDON_AIRPORTS, "LAX")
)
assert b"2027-03-30" in data
@pytest.mark.parametrize(
"query,code",
[
("SFO", "SFO"),
("sfo", "SFO"),
("Copenhagen, Denmark", "CPH"),
("Sacramento, CA", "SMF"),
("London", "LON"),
],
)
def test_local_airports(query: str, code: str) -> None:
"""Exact codes avoid fuzzy matching; useful scheduled airports come first."""
assert airport_lookup.matches(query)[0]["code"] == code
assert conference_flights.resolve_airport(query)[0] == code
def test_unknown_iata_is_not_fuzzy() -> None:
"""Three letters must not accidentally resolve to another airport."""
assert airport_lookup.matches("ZZZ") == []
with pytest.raises(ValueError, match="Unknown IATA"):
conference_flights.resolve_airport("ZZZ")
def test_browser_429_stops_without_retry(
tmp_path: Path, monkeypatch: typing.Any
) -> None:
"""A page navigation's 429 blocks other searches without a second navigation."""
calls: list[str] = []
def goto(url: str, **kwargs: typing.Any) -> typing.Any:
calls.append(url)
return SimpleNamespace(status=429)
browser = google_flights.BrowserSearch(tmp_path)
monkeypatch.setattr(browser, "start", lambda: SimpleNamespace(goto=goto))
with pytest.raises(flight_search_cache.RateLimitCooldownError):
browser.search("LON", "LAX", date(2027, 3, 30), 1)
with pytest.raises(flight_search_cache.RateLimitCooldownError):
browser.search("LON", "LAX", date(2027, 3, 31), 1)
assert len(calls) == 1
assert flight_search_cache.cooldown_message(tmp_path)
def test_stops_fallback_and_ba_ranking(tmp_path: Path, monkeypatch: typing.Any) -> None:
"""Only fall back to two stops for London; BA preference keeps other airlines."""
calls: list[int] = []
def search(
origin: str, destination: str, day: date, max_stops: int
) -> list[dict[str, typing.Any]]:
calls.append(max_stops)
return [
{"stops": 2, "legs": [{"airline_code": "BA"}]},
{"stops": 1, "legs": [{"airline_code": "AA"}]},
{"stops": 1, "legs": [{"airline_code": "BA"}]},
{"stops": 0, "legs": [{"airline_code": "AA"}]},
]
monkeypatch.setattr(google_flights, "search", search)
rows = conference_flights.fetch_day(
"LON", "LAX", date(2027, 3, 30), False, tmp_path
)
assert calls == [1]
assert [(r["stops"], r["legs"][0]["airline_code"]) for r in rows] == [
(0, "AA"),
(1, "BA"),
(1, "AA"),
]
calls.clear()
def no_results(*args: typing.Any) -> list[dict[str, typing.Any]]:
calls.append(args[-1])
return []
monkeypatch.setattr(google_flights, "search", no_results)
assert (
conference_flights.fetch_day("LON", "LAX", date(2027, 3, 30), False, tmp_path)
== []
)
assert calls == [1, 2]
calls.clear()
assert (
conference_flights.fetch_day("BRS", "CPH", date(2027, 3, 30), True, tmp_path)
== []
)
assert calls == [0]
def test_data_request_429_uses_shared_cooldown(tmp_path: Path) -> None:
"""A 429 in the loaded page's Google data request pauses all searches too."""
browser = google_flights.BrowserSearch(tmp_path)
response = SimpleNamespace(
status=429, url="https://www.google.com/_/FlightsUi/data/batchexecute"
)
browser.check_response(typing.cast(Response, response))
with pytest.raises(flight_search_cache.RateLimitCooldownError):
browser.check_cooldown()
assert (
flight_search_cache.read_state(tmp_path / "rate-limit.json")["transport"]
== flight_search_cache.TRANSPORT
)
def test_browser_errors_show_diagnostic_stderr() -> None:
"""Long Playwright launch commands must not hide Chromium's actual failure."""
exc = RuntimeError(
"BrowserType.launch_persistent_context: Target closed\nBrowser logs:\n"
"<launching> /usr/bin/chromium "
+ "--long-option " * 200
+ "\n[pid=123][err] grep: /proc/cpuinfo: No such file or directory\n"
"[pid=123][err] The hardware lacks SSE3 support.\n"
)
details = conference_flights.error_detail(exc)
assert "Target closed" in details
assert "/proc/cpuinfo: No such file or directory" in details
assert "SSE3" in details
assert "--long-option" not in details
@pytest.mark.parametrize(
"query", ["Malmö", "Malmo", "Malmö, Sweden", "MMX,CPH", "mmx, cph"]
)
def test_malmo_airport_group(query: str) -> None:
"""Malmö includes nearby Copenhagen across the border; explicit codes still work."""
assert conference_flights.resolve_airport(query)[0] == "MMX,CPH"
assert airport_lookup.matches(query)[0]["code"] == "MMX,CPH"
assert conference_flights.resolve_airport("MMX")[0] == "MMX"
assert conference_flights.resolve_airport("CPH")[0] == "CPH"
@pytest.mark.parametrize("origin,destination", [("BRS", "MMX,CPH"), ("MMX,CPH", "BRS")])
def test_combined_airport_search_url(origin: str, destination: str) -> None:
"""Outbound and return searches include each airport as its own protobuf entry."""
query = parse_qs(
urlsplit(
google_flights.search_url(origin, destination, date(2026, 10, 19), 0)
).query
)
encoded = query["tfs"][0]
data = base64.urlsafe_b64decode(encoded + "=" * (-len(encoded) % 4))
assert b"MMX,CPH" not in data
assert all(code.encode() in data for code in ("MMX", "CPH", "BRS"))
assert data.count(b"MMX") == data.count(b"CPH") == 1
assert b"\x28\x00" in data # Explicit non-stop filter.
def test_bristol_copenhagen_serves_malmo(
tmp_path: Path, monkeypatch: typing.Any
) -> None:
"""CPH flights satisfy the Bristol preference without searching London or MMX separately."""
calls: list[tuple[str, str, bool]] = []
def search(
origin: str, destination: str, day: date, direct: bool
) -> list[dict[str, typing.Any]]:
calls.append((origin, destination, direct))
return [
{
"arrival": day.isoformat() + "T18:00:00",
"legs": [
{
"origin": "BRS" if origin == "BRS" else "CPH",
"destination": "CPH" if origin == "BRS" else "BRS",
}
],
}
]
monkeypatch.setattr(conference_flights, "search_day", search)
start, end = date(2099, 1, 5), date(2099, 1, 6)
path = conference_flights.cache_path(str(tmp_path), start, end, "MMX,CPH", True)
result = conference_flights.lookup(path, start, end, "MMX,CPH", True)
assert calls == [("BRS", "MMX,CPH", True), ("MMX,CPH", "BRS", True)]
assert [row["origin"] for row in result["searches"]] == ["BRS"]
@pytest.mark.parametrize("query", ["Bern", "Bern, Switzerland", "BRN,BSL"])
def test_bern_includes_basel(query: str) -> None:
"""Bern's destination group includes Basel, while BSL alone stays explicit."""
code, name = conference_flights.resolve_airport(query)
assert code == "BRN,BSL"
assert "Basel" in name
assert conference_flights.resolve_airport("BSL")[0] == "BSL"
@pytest.mark.parametrize("query", ["Bonn", "Bonn, Germany", "CGN,DUS"])
def test_bonn_includes_dusseldorf(query: str) -> None:
"""Bonn uses Cologne/Bonn and Düsseldorf; DUS remains individually selectable."""
code, name = conference_flights.resolve_airport(query)
assert code == "CGN,DUS"
assert "Düsseldorf" in name
assert conference_flights.resolve_airport("DUS")[0] == "DUS"
@pytest.mark.parametrize(
"script",
[
'data:["Requested flight date is too far in the future."]',
'data:{"error":"Dates are too far in the future"}',
],
)
def test_provider_future_date_error(script: str) -> None:
"""Explicit provider date rejections aren't mistaken for unknown result formats."""
with pytest.raises(
google_flights.FlightDateUnavailableError, match="too far in the future"
):
google_flights.parse_results(script, "https://example.com")
@pytest.mark.parametrize(
"text,expected",
[
("Requested flight date is too far in the future.", "future"),
("No flights found", "empty"),
(
"No non-stop flights found There might not be daily non-stop flights to Bologna (BLQ). Try changing your dates, or search flights with more stops.",
"empty",
),
("No nonstop flights found", "empty"),
("No non‑stop flights found", "empty"),
("No direct flights found", "empty"),
("Something went wrong", "error"),
],
)
def test_unusable_page_data(tmp_path: Path, text: str, expected: str) -> None:
"""Only confirmed empty pages become empty results; unrelated failures remain errors."""
page = SimpleNamespace(
locator=lambda selector: SimpleNamespace(inner_text=lambda: text)
)
browser = google_flights.BrowserSearch(tmp_path)
args = (
typing.cast(Page, page),
"data:[]",
"https://example.com",
date(2027, 9, 12),
)
if expected == "future":
with pytest.raises(google_flights.FlightDateUnavailableError):
browser.read_results(*args)
elif expected == "empty":
assert browser.read_results(*args) == []
else:
with pytest.raises(ValueError, match="12 Sep 2027"):
browser.read_results(*args)
@pytest.mark.parametrize("query", ["Funen", "Funen, Denmark", "BLL,CPH"])
def test_funen_includes_copenhagen(query: str) -> None:
"""Funen considers both Billund and Copenhagen in a single search."""
code, name = conference_flights.resolve_airport(query)
assert code == "BLL,CPH"
assert "Billund" in name and "Copenhagen" in name

View file

@ -21,6 +21,7 @@ import agenda.accommodation
import agenda.conference
import agenda.conference_ical
import agenda.conference_list
import agenda.conference_page
import agenda.data
import agenda.error_mail
import agenda.fx
@ -41,6 +42,7 @@ from agenda.types import StrDict, Trip
app = flask.Flask(__name__)
app.debug = False
app.config.from_object("config.default")
app.register_blueprint(agenda.conference_page.blueprint)
app.wsgi_app = ProxyFix(app.wsgi_app, x_for=1, x_proto=1, x_host=1) # type: ignore[method-assign]
agenda.error_mail.setup_error_mail(app)