"""Scrape a public FlightAware history page for a specific past flight's track, without using AeroAPI (whose free tier excludes historical data). Two requests per flight, verified against a real flight (QFA1) during design: 1. GET the flight's public activity page, which lists direct links to every recent/upcoming specific instance of that flight number in the form /live/flight/{callsign}/history/{YYYYMMDD}/{HHMM}Z/{FROM_ICAO}/{TO_ICAO} - this sidesteps the fact that a BCBP barcode never encodes a time-of-day, which the history URL otherwise requires. 2. GET that specific history URL + "/google_earth", which returns a KML file with a gx:Track of / pairs. Low volume by design (one flight per day) - no retry-hammering on failure, a transient error just leaves the row queued for next day's run. """ import re import urllib.error import urllib.request import xml.etree.ElementTree as ET from datetime import date, datetime import config BASE = "https://www.flightaware.com" KML_NAMESPACES = { "kml": "http://www.opengis.net/kml/2.2", "gx": "http://www.google.com/kml/ext/2.2", } class FlightAwareTransientError(Exception): """Network/timeout/5xx - safe to retry on a later run.""" def _fetch(url: str) -> bytes: req = urllib.request.Request(url, headers={"User-Agent": config.FLIGHTAWARE_USER_AGENT}) try: with urllib.request.urlopen(req, timeout=20) as resp: return resp.read() except urllib.error.HTTPError as e: if e.code == 404: raise raise FlightAwareTransientError(f"HTTP {e.code} fetching {url}") from e except urllib.error.URLError as e: raise FlightAwareTransientError(f"error fetching {url}: {e}") from e def find_history_url( icao_callsign: str, flight_date: date, from_icao: str, to_icao: str ) -> str | None: """Returns the specific history page URL for this flight instance, or None if no matching link is found on the activity page (definitive - the flight isn't there, don't retry). Raises FlightAwareTransientError on network-level failures (retry next run).""" activity_url = f"{BASE}/live/flight/{icao_callsign}" try: html = _fetch(activity_url).decode("utf-8", errors="replace") except urllib.error.HTTPError: return None date_str = flight_date.strftime("%Y%m%d") pattern = re.compile( rf"/live/flight/{re.escape(icao_callsign)}/history/{date_str}/" rf"(\d{{4}})Z/{re.escape(from_icao)}/{re.escape(to_icao)}" ) match = pattern.search(html) if not match: return None return f"{BASE}{match.group(0)}" def fetch_kml(history_url: str) -> bytes | None: """Fetches the KML for a history URL found by find_history_url. Returns None if the specific flight has no track available (definitive).""" kml_url = history_url.rstrip("/") + "/google_earth" try: return _fetch(kml_url) except urllib.error.HTTPError: return None def parse_kml(kml_bytes: bytes) -> tuple[list[list[float]], list[int]] | None: """Returns (coordinates, times) where coordinates is [[lon,lat,elev],...] and times is [unix_seconds,...], aligned. None if no track found.""" try: root = ET.fromstring(kml_bytes) except ET.ParseError: return None track = root.find(".//gx:Track", KML_NAMESPACES) if track is None: return None coordinates: list[list[float]] = [] times: list[int] = [] for child in track: tag = child.tag.split("}")[-1] if tag == "when": ts = child.text.strip() dt = datetime.fromisoformat(ts.replace("Z", "+00:00")) times.append(int(dt.timestamp())) elif tag == "coord": parts = child.text.strip().split() lon, lat = float(parts[0]), float(parts[1]) elev = float(parts[2]) if len(parts) > 2 else 0.0 coordinates.append([lon, lat, elev]) if not coordinates: return None if len(times) != len(coordinates): times = [] return coordinates, times