408fde440d
Photograph/screenshot a boarding pass, decode its BCBP barcode (PDF417/ Aztec/QR/DataMatrix), fetch the historical flight track from FlightAware the day after the flight, stage it for review, and write it into AirTrail via its REST API on approval. ntfy notifications carry signed Approve/Deny/Investigate actions.
86 lines
2.8 KiB
Python
86 lines
2.8 KiB
Python
"""One-off dev script: fetch and trim public airport/airline reference data.
|
|
|
|
Not run inside the container - output is committed to data/*.csv and loaded
|
|
at runtime by reference_data.py with zero network dependency.
|
|
|
|
Usage: python3 scripts/build_reference_data.py
|
|
"""
|
|
|
|
import csv
|
|
import io
|
|
import urllib.request
|
|
from pathlib import Path
|
|
|
|
DATA_DIR = Path(__file__).resolve().parent.parent / "data"
|
|
|
|
AIRPORTS_URL = "https://davidmegginson.github.io/ourairports-data/airports.csv"
|
|
AIRLINES_URL = "https://raw.githubusercontent.com/jpatokal/openflights/master/data/airlines.dat"
|
|
|
|
USER_AGENT = "Mozilla/5.0 (compatible; boarding-pass-pipeline/1.0)"
|
|
|
|
|
|
def fetch(url: str) -> bytes:
|
|
req = urllib.request.Request(url, headers={"User-Agent": USER_AGENT})
|
|
with urllib.request.urlopen(req, timeout=30) as resp:
|
|
return resp.read()
|
|
|
|
|
|
def build_airports() -> None:
|
|
raw = fetch(AIRPORTS_URL).decode("utf-8", errors="replace")
|
|
reader = csv.DictReader(io.StringIO(raw))
|
|
out_path = DATA_DIR / "airports.csv"
|
|
seen = set()
|
|
rows = []
|
|
for row in reader:
|
|
iata = (row.get("iata_code") or "").strip().upper()
|
|
icao = (row.get("ident") or row.get("gps_code") or "").strip().upper()
|
|
name = (row.get("name") or "").strip()
|
|
if not iata or not icao or len(iata) != 3 or len(icao) != 4:
|
|
continue
|
|
if iata in seen:
|
|
continue
|
|
seen.add(iata)
|
|
rows.append((iata, icao, name))
|
|
rows.sort(key=lambda r: r[0])
|
|
with out_path.open("w", newline="", encoding="utf-8") as f:
|
|
writer = csv.writer(f)
|
|
writer.writerow(["iata", "icao", "name"])
|
|
writer.writerows(rows)
|
|
print(f"wrote {len(rows)} airports to {out_path}")
|
|
|
|
|
|
def build_airlines() -> None:
|
|
raw = fetch(AIRLINES_URL).decode("utf-8", errors="replace")
|
|
reader = csv.reader(io.StringIO(raw))
|
|
out_path = DATA_DIR / "airlines.csv"
|
|
seen = set()
|
|
rows = []
|
|
for cols in reader:
|
|
# OpenFlights airlines.dat columns (no header):
|
|
# id, name, alias, iata, icao, callsign, country, active
|
|
if len(cols) < 5:
|
|
continue
|
|
name = cols[1].strip().strip('"')
|
|
iata = cols[3].strip().strip('"').upper()
|
|
icao = cols[4].strip().strip('"').upper()
|
|
if not iata or not icao or iata == r"\N" or icao == r"\N":
|
|
continue
|
|
if len(iata) != 2 or len(icao) != 3:
|
|
continue
|
|
if iata in seen:
|
|
continue
|
|
seen.add(iata)
|
|
rows.append((iata, icao, name))
|
|
rows.sort(key=lambda r: r[0])
|
|
with out_path.open("w", newline="", encoding="utf-8") as f:
|
|
writer = csv.writer(f)
|
|
writer.writerow(["iata", "icao", "name"])
|
|
writer.writerows(rows)
|
|
print(f"wrote {len(rows)} airlines to {out_path}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
DATA_DIR.mkdir(exist_ok=True)
|
|
build_airports()
|
|
build_airlines()
|