mirror of
https://github.com/PlaneQuery/OpenAirframes.git
synced 2026-09-14 01:48:55 +02:00
feat: build a daily Transport Canada aircraft register release
- parse the headerless latin1 CCARCS export against its declared column layout - derive transponder_code_hex from the 24-bit Mode S binary, populated for all 34,913 rows - expand marks to C- and vintage CF- registrations and drop owner mailing addresses - mirror the FAA build: same concat-with-latest-release dedup and output conventions Generated-by: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,56 @@
|
||||
from pathlib import Path
|
||||
from datetime import datetime, timezone
|
||||
import argparse
|
||||
|
||||
parser = argparse.ArgumentParser(description="Create daily Transport Canada release")
|
||||
parser.add_argument("--date", type=str, help="Date to process (YYYY-MM-DD format, default: today)")
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.date:
|
||||
date_str = args.date
|
||||
else:
|
||||
date_str = datetime.now(timezone.utc).strftime("%Y-%m-%d")
|
||||
|
||||
out_dir = Path("data/tc_ccarcs")
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
zip_name = f"ccarcsdb_{date_str}.zip"
|
||||
|
||||
zip_path = out_dir / zip_name
|
||||
if not zip_path.exists():
|
||||
url = "https://wwwapps.tc.gc.ca/saf-sec-sur/2/ccarcs-riacc/download/ccarcsdb.zip"
|
||||
from urllib.request import Request, urlopen
|
||||
|
||||
# CCARCS rejects default urllib agents.
|
||||
req = Request(
|
||||
url,
|
||||
headers={
|
||||
"User-Agent": (
|
||||
"Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 "
|
||||
"(KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36"
|
||||
)
|
||||
},
|
||||
method="GET",
|
||||
)
|
||||
|
||||
with urlopen(req, timeout=120) as r:
|
||||
body = r.read()
|
||||
zip_path.write_bytes(body)
|
||||
|
||||
OUT_ROOT = Path("data/openairframes")
|
||||
OUT_ROOT.mkdir(parents=True, exist_ok=True)
|
||||
from derive_from_tc_ccarcs import convert_tc_ccarcs_to_df
|
||||
# Row-fingerprint dedup is source-agnostic; reused rather than forked.
|
||||
from derive_from_faa_master_txt import concat_faa_historical_df
|
||||
from get_latest_release import get_latest_aircraft_tc_csv_df
|
||||
df_new = convert_tc_ccarcs_to_df(zip_path, date_str)
|
||||
|
||||
try:
|
||||
df_base, start_date_str = get_latest_aircraft_tc_csv_df()
|
||||
df_base = concat_faa_historical_df(df_base, df_new)
|
||||
assert df_base['download_date'].is_monotonic_increasing, "download_date is not monotonic increasing"
|
||||
except Exception as e:
|
||||
print(f"No existing Transport Canada release found, using only new data: {e}")
|
||||
df_base = df_new
|
||||
start_date_str = date_str
|
||||
|
||||
df_base.to_csv(OUT_ROOT / f"openairframes_tc_{start_date_str}_{date_str}.csv", index=False)
|
||||
@@ -0,0 +1,172 @@
|
||||
from pathlib import Path
|
||||
import csv
|
||||
import io
|
||||
import zipfile
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from derive_from_faa_master_txt import normalize
|
||||
|
||||
# CCARCS ships headerless, latin1, comma-delimited exports. Column names come from
|
||||
# carslayout.txt in the same archive and must stay in file order.
|
||||
CARSCURR_COLUMNS = [
|
||||
"MARK", "REGISTRATION_SUB_TYPE_E", "REGISTRATION_SUB_TYPE_F", "COMMON_NAME",
|
||||
"MODEL_NAME", "MANUFACTURERS_SERIAL_NUMBER", "MANUFACTURER_SERIAL_COMPRESSED",
|
||||
"ID_PLATE_MANUFACTURERS_NAME", "BASIS_FOR_REGISTRATION", "BASIS_FOR_REGISTRATION_F",
|
||||
"AIRCRAFT_CATEGORY_E", "AIRCRAFT_CATEGORY_F", "DATE_OF_IMPORT", "ENGINE_MANUF",
|
||||
"POWERGLIDER_FLAG", "ENGINE_CATEGORY_E", "ENGINE_CATEGORY_F", "NUMBER_OF_ENGINES",
|
||||
"NUMBER_OF_SEATS", "AIR_WEIGHT_KILOS", "SALE_REPORTED", "ISSUE_DATE",
|
||||
"EFFECTIVE_DATE", "INEFFECTIVE_DATE", "REGISTERED_PURPOSE_E", "REGISTERED_PURPOSE_F",
|
||||
"FLIGHT_AUTHORITY_E", "FLIGHT_AUTHORITY_F", "MANUFACTURE_OR_ASSEMBLY",
|
||||
"COUNTRY_MANUFACTURE_ASS_E", "COUNTRY_MANUFACTURE_ASS_F", "DATE_MANUFACTURE_ASSEMBLY",
|
||||
"BASE_OF_OPERATIONS_CTRY_E", "BASE_OF_OPERATIONS_CTRY_F", "BASE_PROVINCE_OR_STATE_E",
|
||||
"BASE_PROVINCE_OR_STATE_F", "CITY_AIRPORT", "TYPE_CERTIFICATE_NUMBER",
|
||||
"REGISTRATION_AUTH_STATUS_E", "REGISTRATION_AUTH_STATUS_F", "MULTIPLE_OWNER_FLAG",
|
||||
"MODIFIED_DATE", "MODE_S_TRANSPONDER_BINARY", "PHYSICAL_FILE_REGION_E",
|
||||
"PHYSICAL_FILE_REGION_F", "EX_MILITARY_MARK", "TRIMMED_MARK",
|
||||
]
|
||||
|
||||
CARSOWNR_COLUMNS = [
|
||||
"MARK_LINK", "FULL_NAME", "TRADE_NAME", "STREET_NAME", "STREET_NAME2", "CITY",
|
||||
"PROVINCE_OR_STATE_E", "PROVINCE_OR_STATE_F", "POSTAL_CODE", "COUNTRY_E", "COUNTRY_F",
|
||||
"TYPE_OF_OWNER_E", "TYPE_OF_OWNER_F", "ACTIVE_FLAG", "CARE_OF", "REGION_E", "REGION_F",
|
||||
"OWNER_NAME_OLD_FORMAT", "MAIL_RECIPIENT", "TRIMMED_MARK",
|
||||
]
|
||||
|
||||
# Owner mailing addresses are dropped rather than republished; see NOTICE.
|
||||
OWNER_PII_COLUMNS = ["STREET_NAME", "STREET_NAME2", "CITY", "POSTAL_CODE", "CARE_OF"]
|
||||
|
||||
|
||||
def _read_ccarcs_entry(zip_path: Path, entry: str, columns: list[str]) -> pd.DataFrame:
|
||||
"""Read one headerless CCARCS export into a DataFrame, dropping the Oracle footer.
|
||||
|
||||
The export ends with a bare "N rows selected." line and a blank line; both are
|
||||
narrower than the declared column count. Any *other* width mismatch is silent
|
||||
field loss, so it raises instead.
|
||||
"""
|
||||
with zipfile.ZipFile(zip_path) as z:
|
||||
text = z.read(entry).decode("latin1")
|
||||
|
||||
rows = []
|
||||
ragged = 0
|
||||
for row in csv.reader(io.StringIO(text)):
|
||||
if len(row) == len(columns):
|
||||
rows.append([cell.strip() for cell in row])
|
||||
elif len(row) <= 1:
|
||||
ragged += 1 # footer or trailing blank
|
||||
else:
|
||||
raise ValueError(
|
||||
f"{entry}: row with {len(row)} fields, expected {len(columns)}"
|
||||
)
|
||||
|
||||
if ragged > 2:
|
||||
raise ValueError(f"{entry}: {ragged} ragged rows, expected at most 2")
|
||||
|
||||
return pd.DataFrame(rows, columns=columns)
|
||||
|
||||
|
||||
def tc_full_registration(mark: str) -> str:
|
||||
"""Expand a trimmed CCARCS mark into the full Canadian registration.
|
||||
|
||||
Three-character marks are vintage CF- registrations; everything else takes the
|
||||
modern C- prefix.
|
||||
"""
|
||||
mark = (mark or "").strip().upper()
|
||||
if not mark:
|
||||
return ""
|
||||
return f"CF-{mark}" if len(mark) == 3 else f"C-{mark}"
|
||||
|
||||
|
||||
def binary_to_hex(binary: str) -> str:
|
||||
"""Convert a 24-bit Mode S binary string to uppercase hex."""
|
||||
binary = (binary or "").strip()
|
||||
if not binary or any(c not in "01" for c in binary):
|
||||
return ""
|
||||
return f"{int(binary, 2):06X}"
|
||||
|
||||
|
||||
def _merge_owners(df_ownr: pd.DataFrame) -> pd.DataFrame:
|
||||
"""Collapse one row per registered party into one row per mark.
|
||||
|
||||
A co-owned mark repeats with a different party each time; keeping only the mail
|
||||
recipient would silently drop the rest. Each field is deduplicated independently
|
||||
and blanks are skipped, so values are not index-parallel across columns.
|
||||
"""
|
||||
def join_unique(series: pd.Series) -> str:
|
||||
seen = []
|
||||
for value in series:
|
||||
value = (value or "").strip()
|
||||
if value and value not in seen:
|
||||
seen.append(value)
|
||||
return ", ".join(seen)
|
||||
|
||||
grouped = df_ownr.groupby("TRIMMED_MARK", sort=False).agg(
|
||||
owner_name=("FULL_NAME", join_unique),
|
||||
owner_province_or_state=("PROVINCE_OR_STATE_E", join_unique),
|
||||
owner_country=("COUNTRY_E", join_unique),
|
||||
owner_type=("TYPE_OF_OWNER_E", join_unique),
|
||||
owner_party_count=("FULL_NAME", "size"),
|
||||
).reset_index()
|
||||
|
||||
# A party row states its own type ("Individual"); that stops being true of the
|
||||
# mark once several parties share it.
|
||||
grouped.loc[grouped["owner_party_count"] > 1, "owner_type"] = "Co-owner"
|
||||
return grouped
|
||||
|
||||
|
||||
def convert_tc_ccarcs_to_df(zip_path: Path, date: str) -> pd.DataFrame:
|
||||
"""Build the OpenAirframes Transport Canada frame from a CCARCS zip."""
|
||||
df = _read_ccarcs_entry(zip_path, "carscurr.txt", CARSCURR_COLUMNS)
|
||||
df_ownr = _read_ccarcs_entry(zip_path, "carsownr.txt", CARSOWNR_COLUMNS)
|
||||
df_ownr = df_ownr.drop(columns=OWNER_PII_COLUMNS)
|
||||
|
||||
df = df.merge(_merge_owners(df_ownr), on="TRIMMED_MARK", how="left")
|
||||
|
||||
out = pd.DataFrame({
|
||||
"download_date": date,
|
||||
"transponder_code_hex": df["MODE_S_TRANSPONDER_BINARY"].map(binary_to_hex),
|
||||
"registration_number": df["TRIMMED_MARK"].map(tc_full_registration),
|
||||
"mark": df["TRIMMED_MARK"],
|
||||
"aircraft_manufacturer": df["COMMON_NAME"],
|
||||
"aircraft_model": df["MODEL_NAME"],
|
||||
"serial_number": df["MANUFACTURERS_SERIAL_NUMBER"],
|
||||
"aircraft_category": df["AIRCRAFT_CATEGORY_E"],
|
||||
"engine_manufacturer": df["ENGINE_MANUF"],
|
||||
"engine_category": df["ENGINE_CATEGORY_E"],
|
||||
"number_of_engines": df["NUMBER_OF_ENGINES"],
|
||||
"number_of_seats": df["NUMBER_OF_SEATS"],
|
||||
"max_weight_kilos": df["AIR_WEIGHT_KILOS"],
|
||||
"registration_status": df["REGISTRATION_AUTH_STATUS_E"],
|
||||
"registration_sub_type": df["REGISTRATION_SUB_TYPE_E"],
|
||||
"basis_for_registration": df["BASIS_FOR_REGISTRATION"],
|
||||
"registered_purpose": df["REGISTERED_PURPOSE_E"],
|
||||
"flight_authority": df["FLIGHT_AUTHORITY_E"],
|
||||
"type_certificate_number": df["TYPE_CERTIFICATE_NUMBER"],
|
||||
"country_manufacture": df["COUNTRY_MANUFACTURE_ASS_E"],
|
||||
"date_manufacture_assembly": df["DATE_MANUFACTURE_ASSEMBLY"],
|
||||
"base_country": df["BASE_OF_OPERATIONS_CTRY_E"],
|
||||
"base_province_or_state": df["BASE_PROVINCE_OR_STATE_E"],
|
||||
"city_airport": df["CITY_AIRPORT"],
|
||||
"ex_military_mark": df["EX_MILITARY_MARK"],
|
||||
"multiple_owner_flag": df["MULTIPLE_OWNER_FLAG"],
|
||||
"owner_name": df["owner_name"],
|
||||
"owner_type": df["owner_type"],
|
||||
"owner_province_or_state": df["owner_province_or_state"],
|
||||
"owner_country": df["owner_country"],
|
||||
"issue_date": df["ISSUE_DATE"],
|
||||
"effective_date": df["EFFECTIVE_DATE"],
|
||||
"ineffective_date": df["INEFFECTIVE_DATE"],
|
||||
"modified_date": df["MODIFIED_DATE"],
|
||||
})
|
||||
|
||||
out.insert(3, "openairframes_id", (
|
||||
normalize(out["aircraft_manufacturer"])
|
||||
+ "|"
|
||||
+ normalize(out["aircraft_model"])
|
||||
+ "|"
|
||||
+ normalize(out["serial_number"])
|
||||
))
|
||||
|
||||
out = out.fillna("")
|
||||
out = out.replace("None", "")
|
||||
return out
|
||||
@@ -167,6 +167,43 @@ def get_latest_aircraft_faa_csv_df():
|
||||
return df, date_str
|
||||
|
||||
|
||||
def download_latest_aircraft_tc_csv(
|
||||
output_dir: Path = Path("downloads"),
|
||||
github_token: Optional[str] = None,
|
||||
repo: str = REPO,
|
||||
) -> Path:
|
||||
"""
|
||||
Download the latest openairframes_tc_*.csv file from the latest GitHub release.
|
||||
|
||||
Args:
|
||||
output_dir: Directory to save the downloaded file (default: "downloads")
|
||||
github_token: Optional GitHub token for authentication
|
||||
repo: GitHub repository in format "owner/repo" (default: REPO)
|
||||
|
||||
Returns:
|
||||
Path to the downloaded file
|
||||
"""
|
||||
output_dir = Path(output_dir)
|
||||
assets = get_latest_release_assets(repo, github_token=github_token)
|
||||
asset = pick_asset(assets, name_regex=r"^openairframes_tc_.*\.csv$")
|
||||
saved_to = download_asset(asset, output_dir / asset.name, github_token=github_token)
|
||||
print(f"Downloaded: {asset.name} ({asset.size} bytes) -> {saved_to}")
|
||||
return saved_to
|
||||
|
||||
|
||||
def get_latest_aircraft_tc_csv_df():
|
||||
csv_path = download_latest_aircraft_tc_csv()
|
||||
import pandas as pd
|
||||
df = pd.read_csv(csv_path, dtype=str)
|
||||
df = df.fillna("")
|
||||
# Filename pattern: openairframes_tc_{start_date}_{end_date}.csv
|
||||
match = re.search(r"openairframes_tc_(\d{4}-\d{2}-\d{2})_", str(csv_path))
|
||||
if not match:
|
||||
raise ValueError(f"Could not extract date from filename: {csv_path.name}")
|
||||
|
||||
return df, match.group(1)
|
||||
|
||||
|
||||
def download_latest_aircraft_adsb_csv(
|
||||
output_dir: Path = Path("downloads"),
|
||||
github_token: Optional[str] = None,
|
||||
|
||||
Reference in New Issue
Block a user