Files
boarding-pass/scripts/build_reference_data.py
bsncubed 408fde440d Initial commit: boarding pass to AirTrail pipeline
Photograph/screenshot a boarding pass, decode its BCBP barcode (PDF417/
Aztec/QR/DataMatrix), fetch the historical flight track from FlightAware
the day after the flight, stage it for review, and write it into AirTrail
via its REST API on approval. ntfy notifications carry signed
Approve/Deny/Investigate actions.
2026-07-04 22:21:02 +10:00

86 lines
2.8 KiB
Python

"""One-off dev script: fetch and trim public airport/airline reference data.
Not run inside the container - output is committed to data/*.csv and loaded
at runtime by reference_data.py with zero network dependency.
Usage: python3 scripts/build_reference_data.py
"""
import csv
import io
import urllib.request
from pathlib import Path
DATA_DIR = Path(__file__).resolve().parent.parent / "data"
AIRPORTS_URL = "https://davidmegginson.github.io/ourairports-data/airports.csv"
AIRLINES_URL = "https://raw.githubusercontent.com/jpatokal/openflights/master/data/airlines.dat"
USER_AGENT = "Mozilla/5.0 (compatible; boarding-pass-pipeline/1.0)"
def fetch(url: str) -> bytes:
req = urllib.request.Request(url, headers={"User-Agent": USER_AGENT})
with urllib.request.urlopen(req, timeout=30) as resp:
return resp.read()
def build_airports() -> None:
raw = fetch(AIRPORTS_URL).decode("utf-8", errors="replace")
reader = csv.DictReader(io.StringIO(raw))
out_path = DATA_DIR / "airports.csv"
seen = set()
rows = []
for row in reader:
iata = (row.get("iata_code") or "").strip().upper()
icao = (row.get("ident") or row.get("gps_code") or "").strip().upper()
name = (row.get("name") or "").strip()
if not iata or not icao or len(iata) != 3 or len(icao) != 4:
continue
if iata in seen:
continue
seen.add(iata)
rows.append((iata, icao, name))
rows.sort(key=lambda r: r[0])
with out_path.open("w", newline="", encoding="utf-8") as f:
writer = csv.writer(f)
writer.writerow(["iata", "icao", "name"])
writer.writerows(rows)
print(f"wrote {len(rows)} airports to {out_path}")
def build_airlines() -> None:
raw = fetch(AIRLINES_URL).decode("utf-8", errors="replace")
reader = csv.reader(io.StringIO(raw))
out_path = DATA_DIR / "airlines.csv"
seen = set()
rows = []
for cols in reader:
# OpenFlights airlines.dat columns (no header):
# id, name, alias, iata, icao, callsign, country, active
if len(cols) < 5:
continue
name = cols[1].strip().strip('"')
iata = cols[3].strip().strip('"').upper()
icao = cols[4].strip().strip('"').upper()
if not iata or not icao or iata == r"\N" or icao == r"\N":
continue
if len(iata) != 2 or len(icao) != 3:
continue
if iata in seen:
continue
seen.add(iata)
rows.append((iata, icao, name))
rows.sort(key=lambda r: r[0])
with out_path.open("w", newline="", encoding="utf-8") as f:
writer = csv.writer(f)
writer.writerow(["iata", "icao", "name"])
writer.writerows(rows)
print(f"wrote {len(rows)} airlines to {out_path}")
if __name__ == "__main__":
DATA_DIR.mkdir(exist_ok=True)
build_airports()
build_airlines()