mirror of
https://github.com/PlaneQuery/OpenAirframes.git
synced 2026-09-22 13:50:43 +02:00
fix: stop a failed Transport Canada build from erasing release history
- fall back to a single-day rebuild only on FileNotFoundError; a rate limit, schema change or truncated download previously took the same path and republished one day as the whole dataset, which the next run then read back as its base - move the monotonic download_date assert out of the try so corruption cannot select the destructive branch - walk back through releases like the ADS-B reader, so one missing optional asset does not strand the accumulation - authenticate release reads and verify downloaded asset size - read the previous CSV with keep_default_na=False so literal NA values round-trip - write the download atomically and reject non-zip responses Generated-by: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -20,7 +20,8 @@ if not zip_path.exists():
|
|||||||
url = "https://wwwapps.tc.gc.ca/saf-sec-sur/2/ccarcs-riacc/download/ccarcsdb.zip"
|
url = "https://wwwapps.tc.gc.ca/saf-sec-sur/2/ccarcs-riacc/download/ccarcsdb.zip"
|
||||||
from urllib.request import Request, urlopen
|
from urllib.request import Request, urlopen
|
||||||
|
|
||||||
# CCARCS rejects default urllib agents.
|
# CCARCS 403s a default urllib agent. Any browser-like UA works; the exact
|
||||||
|
# version string is not load-bearing.
|
||||||
req = Request(
|
req = Request(
|
||||||
url,
|
url,
|
||||||
headers={
|
headers={
|
||||||
@@ -34,23 +35,39 @@ if not zip_path.exists():
|
|||||||
|
|
||||||
with urlopen(req, timeout=120) as r:
|
with urlopen(req, timeout=120) as r:
|
||||||
body = r.read()
|
body = r.read()
|
||||||
zip_path.write_bytes(body)
|
# TC serves an HTML maintenance page with a 200, which would otherwise be cached
|
||||||
|
# under a .zip name and re-read on every later run.
|
||||||
|
if body[:2] != b"PK":
|
||||||
|
raise RuntimeError(f"{url} did not return a zip (got {body[:40]!r})")
|
||||||
|
tmp_path = zip_path.with_suffix(".part")
|
||||||
|
tmp_path.write_bytes(body)
|
||||||
|
tmp_path.replace(zip_path)
|
||||||
|
|
||||||
OUT_ROOT = Path("data/openairframes")
|
OUT_ROOT = Path("data/openairframes")
|
||||||
OUT_ROOT.mkdir(parents=True, exist_ok=True)
|
OUT_ROOT.mkdir(parents=True, exist_ok=True)
|
||||||
from derive_from_tc_ccarcs import convert_tc_ccarcs_to_df
|
from derive_from_tc_ccarcs import convert_tc_ccarcs_to_df
|
||||||
# Row-fingerprint dedup is source-agnostic; reused rather than forked.
|
# Named for FAA but column-agnostic: fingerprints every column except download_date.
|
||||||
from derive_from_faa_master_txt import concat_faa_historical_df
|
from derive_from_faa_master_txt import concat_faa_historical_df
|
||||||
from get_latest_release import get_latest_aircraft_tc_csv_df
|
from get_latest_release import get_latest_aircraft_tc_csv_df
|
||||||
df_new = convert_tc_ccarcs_to_df(zip_path, date_str)
|
df_new = convert_tc_ccarcs_to_df(zip_path, date_str)
|
||||||
|
|
||||||
|
# Only a genuine first run may rebuild from a single day. Every other failure -- a rate
|
||||||
|
# limit, a schema change, a truncated download -- must stop the run, because this file
|
||||||
|
# becomes tomorrow's base and silently republishing one day erases the whole history.
|
||||||
try:
|
try:
|
||||||
df_base, start_date_str = get_latest_aircraft_tc_csv_df()
|
df_base, start_date_str = get_latest_aircraft_tc_csv_df()
|
||||||
df_base = concat_faa_historical_df(df_base, df_new)
|
except FileNotFoundError as e:
|
||||||
assert df_base['download_date'].is_monotonic_increasing, "download_date is not monotonic increasing"
|
print(f"No existing Transport Canada release found, bootstrapping from today only: {e}")
|
||||||
except Exception as e:
|
df_base = None
|
||||||
print(f"No existing Transport Canada release found, using only new data: {e}")
|
|
||||||
df_base = df_new
|
|
||||||
start_date_str = date_str
|
start_date_str = date_str
|
||||||
|
|
||||||
|
if df_base is not None:
|
||||||
|
missing = set(df_base.columns) ^ set(df_new.columns)
|
||||||
|
if missing:
|
||||||
|
raise SystemExit(f"Column set changed since the last release: {sorted(missing)}")
|
||||||
|
df_base = concat_faa_historical_df(df_base, df_new)
|
||||||
|
assert df_base['download_date'].is_monotonic_increasing, "download_date is not monotonic increasing"
|
||||||
|
else:
|
||||||
|
df_base = df_new
|
||||||
|
|
||||||
df_base.to_csv(OUT_ROOT / f"openairframes_tc_{start_date_str}_{date_str}.csv", index=False)
|
df_base.to_csv(OUT_ROOT / f"openairframes_tc_{start_date_str}_{date_str}.csv", index=False)
|
||||||
|
|||||||
@@ -3,6 +3,7 @@ from __future__ import annotations
|
|||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Iterable, Optional
|
from typing import Iterable, Optional
|
||||||
|
import os
|
||||||
import re
|
import re
|
||||||
import urllib.request
|
import urllib.request
|
||||||
import urllib.error
|
import urllib.error
|
||||||
@@ -184,19 +185,45 @@ def download_latest_aircraft_tc_csv(
|
|||||||
Path to the downloaded file
|
Path to the downloaded file
|
||||||
"""
|
"""
|
||||||
output_dir = Path(output_dir)
|
output_dir = Path(output_dir)
|
||||||
assets = get_latest_release_assets(repo, github_token=github_token)
|
github_token = github_token or os.environ.get("GITHUB_TOKEN")
|
||||||
asset = pick_asset(assets, name_regex=r"^openairframes_tc_.*\.csv$")
|
|
||||||
saved_to = download_asset(asset, output_dir / asset.name, github_token=github_token)
|
# Walk back through releases rather than reading only `latest`. The TC asset is
|
||||||
print(f"Downloaded: {asset.name} ({asset.size} bytes) -> {saved_to}")
|
# optional, so a single failed build publishes a release without it; anchoring on
|
||||||
return saved_to
|
# `latest` would then make the caller rebuild history from one day and republish
|
||||||
|
# that as the whole dataset.
|
||||||
|
for release in get_releases(repo, github_token=github_token, per_page=30):
|
||||||
|
assets = get_release_assets_from_release_data(release)
|
||||||
|
try:
|
||||||
|
asset = pick_asset(assets, name_regex=r"^openairframes_tc_.*\.csv$")
|
||||||
|
except FileNotFoundError:
|
||||||
|
continue
|
||||||
|
saved_to = download_asset(asset, output_dir / asset.name, github_token=github_token)
|
||||||
|
if asset.size and saved_to.stat().st_size != asset.size:
|
||||||
|
raise RuntimeError(
|
||||||
|
f"{asset.name}: downloaded {saved_to.stat().st_size} bytes, expected {asset.size}"
|
||||||
|
)
|
||||||
|
print(f"Downloaded: {asset.name} ({asset.size} bytes) -> {saved_to}")
|
||||||
|
return saved_to
|
||||||
|
|
||||||
|
raise FileNotFoundError(
|
||||||
|
"No release in the last 30 releases has an asset matching 'openairframes_tc_.*\\.csv$'"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def get_latest_aircraft_tc_csv_df():
|
def get_latest_aircraft_tc_csv_df():
|
||||||
|
"""Return (DataFrame, start_date_str) for the most recent published TC release.
|
||||||
|
|
||||||
|
Raises FileNotFoundError when no recent release carries a TC asset, and ValueError
|
||||||
|
when the asset filename has no parseable start date.
|
||||||
|
"""
|
||||||
csv_path = download_latest_aircraft_tc_csv()
|
csv_path = download_latest_aircraft_tc_csv()
|
||||||
import pandas as pd
|
import pandas as pd
|
||||||
df = pd.read_csv(csv_path, dtype=str)
|
# keep_default_na=False: a literal "NA"/"N/A" in the source would otherwise read back
|
||||||
|
# as NaN -> "" while the fresh parse keeps the string, so the row fingerprints would
|
||||||
|
# never match and every affected record would re-append on every run.
|
||||||
|
df = pd.read_csv(csv_path, dtype=str, keep_default_na=False)
|
||||||
df = df.fillna("")
|
df = df.fillna("")
|
||||||
# Filename pattern: openairframes_tc_{start_date}_{end_date}.csv
|
# Only the start date is taken; the end date is always the run's own date.
|
||||||
match = re.search(r"openairframes_tc_(\d{4}-\d{2}-\d{2})_", str(csv_path))
|
match = re.search(r"openairframes_tc_(\d{4}-\d{2}-\d{2})_", str(csv_path))
|
||||||
if not match:
|
if not match:
|
||||||
raise ValueError(f"Could not extract date from filename: {csv_path.name}")
|
raise ValueError(f"Could not extract date from filename: {csv_path.name}")
|
||||||
|
|||||||
Reference in New Issue
Block a user