Merge pull request #72 from anchildress1/feat/parallel-registry-build

Add Transport Canada, build registry sources in parallel, and join them into one asset
This commit is contained in:
JG
2026-09-13 18:45:36 -04:00
committed by GitHub
10 changed files with 935 additions and 141 deletions
@@ -10,11 +10,19 @@ on:
description: 'Date to process (YYYY-MM-DD format, default: yesterday)'
required: false
type: string
bootstrap_source:
description: 'Onboarding only: the one source id permitted to rebuild from a single day'
required: false
type: string
permissions:
contents: write
actions: write
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: false
jobs:
trigger-releases:
runs-on: ubuntu-latest
@@ -42,39 +50,81 @@ jobs:
ref: 'develop'
});
build-faa:
runs-on: ubuntu-24.04-arm
# One thread per registry. Adding a source is one matrix entry plus
# src/create_daily_<source>_release.py; join-registry picks it up by artifact pattern.
build-registry-source:
if: github.event_name != 'schedule'
strategy:
fail-fast: false
matrix:
include:
- source: faa
required: true
- source: tc
required: false
uses: ./.github/workflows/registry-source.yaml
with:
source: ${{ matrix.source }}
date: ${{ inputs.date }}
required: ${{ matrix.required }}
# Standing permission would also fire on an outage. A source bootstraps only when a
# human dispatches the run naming it.
allow_bootstrap: ${{ inputs.bootstrap_source == matrix.source }}
join-registry:
needs: build-registry-source
# No always(): a tolerated source fails its step without failing its leg, so this only
# blocks when a required source could not be built.
if: github.event_name != 'schedule'
runs-on: ubuntu-24.04-arm
timeout-minutes: 20
permissions:
contents: read
steps:
- name: Checkout
uses: actions/checkout@v6
with:
fetch-depth: 0
uses: actions/checkout@v7
- name: Setup Python
uses: actions/setup-python@v6
uses: actions/setup-python@v7
with:
python-version: "3.14"
cache: 'pip'
cache-dependency-path: requirements.txt
- name: Install dependencies
run: |
python -m pip install --upgrade pip
pip install -r requirements.txt
- name: Run FAA release script
run: |
python src/create_daily_faa_release.py ${{ inputs.date && format('--date {0}', inputs.date) || '' }}
ls -lah data/faa_releasable
ls -lah data/openairframes
- name: Upload FAA artifacts
uses: actions/upload-artifact@v4
- name: Download every source thread
uses: actions/download-artifact@v8
with:
name: faa-release
path: |
data/openairframes/openairframes_faa_*.csv
data/faa_releasable/ReleasableAircraft_*.zip
pattern: registry-*
path: artifacts/registry
merge-multiple: true
- name: Join sources into one registry
env:
RUN_DATE: ${{ inputs.date }}
run: |
python src/build_registry.py --input-dir artifacts/registry ${RUN_DATE:+--date "$RUN_DATE"}
- name: Stage release assets
run: |
mkdir -p data/release-out
# Every per-source CSV must ship: each source reads its OWN previous asset to
# accumulate, so one that is never published can never be anything but day one.
cp artifacts/registry/* data/release-out/
cp data/openairframes/openairframes_registry_*.csv data/release-out/
ls -lah data/release-out
- name: Upload registry
uses: actions/upload-artifact@v7
with:
name: union-registry
path: data/release-out
retention-days: 1
if-no-files-found: error
resolve-dates:
runs-on: ubuntu-latest
@@ -233,7 +283,7 @@ jobs:
create-release:
runs-on: ubuntu-latest
needs: [resolve-dates, build-faa, adsb-to-aircraft, adsb-reduce, build-community, build-adsbexchange-json, build-mictronics-db]
needs: [resolve-dates, join-registry, adsb-to-aircraft, adsb-reduce, build-community, build-adsbexchange-json, build-mictronics-db]
if: github.event_name != 'schedule' && !cancelled()
steps:
- name: Check ADS-B workflow status
@@ -246,12 +296,13 @@ jobs:
with:
sparse-checkout: |
.github
NOTICE
sparse-checkout-cone-mode: false
- name: Download FAA artifacts
- name: Download joined registry
uses: actions/download-artifact@v5
with:
name: faa-release
name: union-registry
path: artifacts/faa
- name: Download ADS-B artifacts
@@ -311,6 +362,12 @@ jobs:
# Find files from artifacts using find (handles nested structures)
CSV_FILE_FAA=$(find artifacts/faa -name "openairframes_faa_*.csv" -type f 2>/dev/null | head -1)
CSV_FILE_REGISTRY=$(find artifacts/faa -name "openairframes_registry_*.csv" -type f 2>/dev/null | head -1)
# Every per-source registry CSV, whatever sources the matrix ran.
SOURCE_CSVS=$(find artifacts/faa -name "openairframes_*.csv" -type f 2>/dev/null \
| grep -vE '/openairframes_(registry|community|adsb)_[0-9]{4}-[0-9]{2}-[0-9]{2}_[0-9]{4}-[0-9]{2}-[0-9]{2}\.csv$' | sort)
echo "Per-source registry CSVs found:"
echo "$SOURCE_CSVS"
# Prefer concatenated file (with date range) over single-day file
CSV_FILE_ADSB=$(find artifacts/adsb -name "openairframes_adsb_*_*.csv.gz" -type f 2>/dev/null | head -1)
if [ -z "$CSV_FILE_ADSB" ]; then
@@ -332,6 +389,14 @@ jobs:
if [ -z "$JSON_FILE_ADSBX" ] || [ ! -f "$JSON_FILE_ADSBX" ]; then
MISSING_FILES="$MISSING_FILES ADSBX_JSON"
fi
if [ -z "$CSV_FILE_REGISTRY" ] || [ ! -f "$CSV_FILE_REGISTRY" ]; then
MISSING_FILES="$MISSING_FILES REGISTRY_CSV"
fi
# NOTICE carries the terms that make each asset redistributable. Shipping data
# without it removes the permission, so it is required rather than optional.
if [ ! -f NOTICE ]; then
MISSING_FILES="$MISSING_FILES NOTICE"
fi
# Optional files - warn but don't fail
OPTIONAL_MISSING=""
@@ -369,12 +434,24 @@ jobs:
fi
if [ -n "$OPTIONAL_MISSING" ]; then
echo "WARNING: Optional files missing:$OPTIONAL_MISSING (will continue without them)"
echo "::warning title=Missing optional release assets::$OPTIONAL_MISSING"
fi
echo "date=$DATE" >> "$GITHUB_OUTPUT"
echo "tag=$TAG" >> "$GITHUB_OUTPUT"
echo "csv_file_faa=$CSV_FILE_FAA" >> "$GITHUB_OUTPUT"
echo "csv_file_registry=$CSV_FILE_REGISTRY" >> "$GITHUB_OUTPUT"
{
echo "source_csvs<<SOURCE_CSVS_EOF"
echo "$SOURCE_CSVS"
echo "SOURCE_CSVS_EOF"
echo "source_basenames<<SOURCE_BASENAMES_EOF"
echo "$SOURCE_CSVS" | while read -r f; do
[ -n "$f" ] && echo "- $(basename "$f")"
done
echo "SOURCE_BASENAMES_EOF"
} >> "$GITHUB_OUTPUT"
echo "csv_basename_registry=$(basename "$CSV_FILE_REGISTRY")" >> "$GITHUB_OUTPUT"
echo "csv_basename_faa=$CSV_BASENAME_FAA" >> "$GITHUB_OUTPUT"
echo "csv_file_adsb=$CSV_FILE_ADSB" >> "$GITHUB_OUTPUT"
echo "csv_basename_adsb=$CSV_BASENAME_ADSB" >> "$GITHUB_OUTPUT"
@@ -399,7 +476,14 @@ jobs:
- name: Delete existing release if exists
run: |
echo "Attempting to delete release: ${{ steps.meta.outputs.tag }}"
gh release delete "${{ steps.meta.outputs.tag }}" --yes --cleanup-tag || echo "No existing release to delete"
# `|| echo` here would swallow a 403 or a partial delete, leaving yesterday's
# asset attached alongside today's; the next run then matches two and rebuilds
# the dataset from a single day.
if gh release view "${{ steps.meta.outputs.tag }}" >/dev/null 2>&1; then
gh release delete "${{ steps.meta.outputs.tag }}" --yes --cleanup-tag
else
echo "No existing release to delete"
fi
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
@@ -413,14 +497,18 @@ jobs:
Automated daily snapshot generated at 06:00 UTC for ${{ steps.meta.outputs.date }}.
Assets:
- ${{ steps.meta.outputs.csv_basename_faa }}
- NOTICE (source terms; required for redistribution)
- ${{ steps.meta.outputs.csv_basename_registry }}
${{ steps.meta.outputs.source_basenames }}
${{ steps.meta.outputs.csv_basename_adsb && format('- {0}', steps.meta.outputs.csv_basename_adsb) || '' }}
- ${{ steps.meta.outputs.csv_basename_community }}
- ${{ steps.meta.outputs.zip_basename }}
- ${{ steps.meta.outputs.json_basename_adsbx }}
${{ steps.meta.outputs.zip_basename_mictronics && format('- {0}', steps.meta.outputs.zip_basename_mictronics) || '' }}
files: |
${{ steps.meta.outputs.csv_file_faa }}
${{ steps.meta.outputs.csv_file_registry }}
${{ steps.meta.outputs.source_csvs }}
NOTICE
${{ steps.meta.outputs.csv_file_adsb }}
${{ steps.meta.outputs.csv_file_community }}
${{ steps.meta.outputs.zip_file }}
+104
View File
@@ -0,0 +1,104 @@
name: registry-source
# One registry source, on its own thread. Called once per entry in the caller's matrix, so
# adding a registry is a matrix entry plus src/create_daily_<source>_release.py — no new job.
on:
workflow_call:
inputs:
source:
description: 'Source id; must match src/create_daily_<source>_release.py'
required: true
type: string
date:
description: 'Date to process (YYYY-MM-DD, default: today UTC)'
required: false
type: string
python-version:
required: false
type: string
default: '3.14'
required:
description: 'Fail the thread when this source cannot be built'
required: false
type: boolean
default: false
allow_bootstrap:
description: 'Onboarding only: permit a single-day rebuild when no asset has ever been published'
required: false
type: boolean
default: false
# Keyed on the source: without it every matrix leg shares one group and the legs cancel
# each other, which is the opposite of running them in parallel.
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}-${{ inputs.source }}
cancel-in-progress: false
jobs:
build:
runs-on: ubuntu-24.04-arm
timeout-minutes: 30
permissions:
contents: read
steps:
- name: Checkout
uses: actions/checkout@v7
- name: Setup Python
uses: actions/setup-python@v7
with:
python-version: ${{ inputs.python-version }}
cache: 'pip'
cache-dependency-path: requirements.txt
- name: Install dependencies
run: |
python -m pip install --upgrade pip
pip install -r requirements.txt
# Deliberately outside the tolerated step below: a typo in the matrix is a config
# error, and must go red even for an optional source.
- name: Check the source has a build script
env:
SOURCE: ${{ inputs.source }}
run: |
script="src/create_daily_${SOURCE}_release.py"
if [ ! -f "$script" ]; then
echo "::error title=Unknown registry source::$script does not exist"
exit 1
fi
- name: Build ${{ inputs.source }} registry
continue-on-error: ${{ inputs.required == false }}
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
SOURCE: ${{ inputs.source }}
RUN_DATE: ${{ inputs.date }}
BOOTSTRAP: ${{ inputs.allow_bootstrap && '1' || '' }}
run: |
python "src/create_daily_${SOURCE}_release.py" ${RUN_DATE:+--date "$RUN_DATE"} ${BOOTSTRAP:+--allow-bootstrap}
# Stage into one directory so every leg's artifact has the same root; a second
# search path would move the root to the common ancestor for some legs only.
shopt -s nullglob
built=(data/openairframes/openairframes_"${SOURCE}"_*.csv)
if [ ${#built[@]} -ne 1 ]; then
echo "::error title=${SOURCE} produced no registry CSV::expected one openairframes_${SOURCE}_*.csv, found ${#built[@]}"
exit 1
fi
# Stage into one directory so every leg's artifact has the same root; a second
# search path would move the root to the common ancestor for some legs only.
mkdir -p data/registry-out
cp "${built[@]}" data/registry-out/
cp data/faa_releasable/ReleasableAircraft_*.zip data/registry-out/ 2>/dev/null || true
ls -lah data/registry-out
- name: Upload ${{ inputs.source }} registry
uses: actions/upload-artifact@v7
with:
name: registry-${{ inputs.source }}
path: data/registry-out
retention-days: 1
# A tolerated source that failed has nothing to upload; only a required
# source missing its artifact is an error.
if-no-files-found: ${{ inputs.required && 'error' || 'ignore' }}
+61 -100
View File
@@ -1,121 +1,82 @@
## Never interpolate `${{ github.event.* }}` into a `run:` block
## Never interpolate `${{ github.event.* }}` or `${{ inputs.* }}` into a `run:` block
Actions substitutes `${{ }}` as raw text before the shell parses it, and issue bodies here are
public and unauthenticated. Pass untrusted values through `env:` and quote them.
Pass them via `env:` and quote the shell variable. A quoted heredoc does not help — the body can
contain the delimiter and close it early. Never fix this by escaping or renaming the delimiter.
A quoted heredoc delimiter does not save you: the body can *contain* the delimiter, close the
heredoc early, and execute every line after it. `validate-community-submission.yaml` had that shape.
Never fix this class of bug by escaping, sanitizing, or renaming the delimiter — move it to `env:`.
## Invocation
## Run everything from the repo root
`src/create_daily_faa_release.py` must be invoked as a **script** (`python src/create_daily_faa_release.py`).
It uses bare sibling imports, so `-m src.create_daily_faa_release` raises `ModuleNotFoundError`.
Everything under `src/adsb/` and `src/contributions/` is the opposite — `python -m`, package-relative.
Output paths are CWD-relative.
- `src/*.py` at the root of `src/` are scripts: `python src/create_daily_faa_release.py`. Bare
sibling imports, so `-m` raises `ModuleNotFoundError`.
- `src/adsb/*`, `src/contributions/*` are packages: `python -m`.
- Run from the repo root; output paths are CWD-relative.
## Verification
There is no test framework, linter, or packaging config. **Do not add one unprompted**, and do not
treat "nothing broke" as verification.
- No test framework, linter, or packaging config. Do not add one unprompted.
- Never `gh workflow run` to test a change — every dispatch pulls tens of GB.
- Never commit generated data. The product is a GitHub Release; jobs pass state as artifacts.
- ADS-B has no cheap end-to-end check. Exercise `compress_multi_icao_df` on a hand-built frame.
The ADS-B path has no cheap end-to-end check — one day of input is tens of GB. Exercise
`compress_multi_icao_df` / `compress_df_polars` directly against a small hand-built Polars frame.
## Release invariants
**Never `gh workflow run` to test a change.** Every dispatch pulls tens of GB and fans out over date
matrices. Reason about the YAML statically.
- `FINAL_COLUMN_ORDER` (`compress_adsb_to_aircraft_data.py`) is the only definition of the ADS-B
column contract. `pl.concat` matches by position after `.select()`; a forked copy corrupts the
release with no error.
- Empty string, never null, in every released frame.
- `openairframes_id` = `normalize(manufacturer)|normalize(model)|normalize(serial)`. Reuse
`derive_from_faa_master_txt.normalize()`.
- Each source's daily build reads its own previous release asset and appends. Falling back to a
single-day rebuild on anything but `FileNotFoundError` republishes one day as the whole dataset,
which the next run then reads back as its base. Keep the fallback narrow.
- Python `3.14` for FAA/community/vendor jobs, `3.12` for ADS-B. Match the surrounding job.
**Never commit generated data.** The product is a GitHub Release; jobs pass state as artifacts.
## Registry sources
## ADS-B invariants
- Judge **redistribution**, not access. A public licence travels to this project; a bilateral
permission granted to another project does not — that alone disqualifies Taiwan, Estonia, Chile.
Non-commercial-only terms are a separate, independent bar.
- `NOTICE` carries the terms that make each asset redistributable and is a required release file.
Never edit or drop an entry. Transport Canada requires both its notices together.
- `LICENSE` is MIT and covers code only. Claim nothing about released data.
- Owner/registrant mailing addresses are published for every registry. FAA and TC must not diverge.
- CCARCS `ACTIVE_FLAG` is not "current owner": 1,932 Registered marks carry only `I` parties, and
those rows are the `MAIL_RECIPIENT`. Prefer `A`, fall back to all.
- CCARCS addresses come from the single `MAIL_RECIPIENT == "Y"` row, never merged across co-owners.
- `FINAL_COLUMN_ORDER` (`compress_adsb_to_aircraft_data.py`) is the only definition of the released
column contract. `pl.concat` matches by **position** after `.select()`, and
`get_latest_release.get_latest_aircraft_adsb_csv_df` parses released CSVs against the same order —
so a forked copy corrupts the release with no error anywhere.
- `load_parquet_part()` deleting its source parquet is **deliberate**: the raw part is many GB and
the runner would otherwise exhaust disk. Do not defer the delete to make reruns easier; that raises
peak disk by the size of the part.
- A released row means "most informative observation for this ICAO on this UTC day" — non-empty
fields not a subset of another row's, tie-broken by signature frequency. It is not a registry record.
- HTTP 404 is terminal in the release fetch. Restoring the retry makes the Dec-31 next-year-repo
probe stall ~45 minutes on a repo that does not exist yet.
## ADS-B
## Fork and upstream
`src/get_latest_release.py` pins `REPO = "PlaneQuery/openairframes"` on purpose: this fork reads
**upstream's** releases wherever it runs. Do not repoint it at `github.repository` without being asked.
Upstream develops on `develop` and PRs into `main`. The daily release **deletes the existing release
and tag** before recreating them.
- `load_parquet_part()` deleting its source parquet is deliberate — disk pressure. Do not defer it.
- A released row is the most informative observation for that ICAO on that UTC day, not a registry
record.
- HTTP 404 is terminal in the release fetch; restoring the retry stalls the Dec-31 probe ~45 min.
## Community submissions are automation-owned
Merging to `community/**` or `schemas/**` force-pushes every open `community`-labeled PR branch back
onto main. Anything you hand-edit on such a branch is destroyed on the next merge.
- Merging `community/**` or `schemas/**` force-pushes every open `community` PR branch onto main.
Hand edits there are destroyed.
- Never hand-author `community/` files — the filename encodes `sha256(content)[:8]`.
- A tag's JSON type is fixed by its first submission and enforced forever. Emergent from
`build_tag_type_registry` + `validate_submission`; written nowhere in the schema.
- Adding `community_submission.v2.schema.json` promotes it atomically across every reader and
writer. One-way door — only on request.
- Never hand-author files in `community/` — the filename encodes `sha256(content)[:8]`, so an edit
orphans the hash and duplicates on re-approval.
- Never invent or copy a `contributor_uuid`; it is derived from the GitHub user id.
- Do not reintroduce a hardcoded `"main"` or `v1` filename. Both are resolved at runtime now.
## Fork
**A tag's JSON type is fixed by its first-ever submission and enforced forever** — emergent from
`build_tag_type_registry` + `validate_submission`, and written nowhere in the schema. Retyping or
renaming an existing tag breaks every future contributor, not just the current one.
`get_latest_release.REPO` pins upstream `PlaneQuery/openairframes` on purpose. Do not repoint it.
Upstream develops on `develop`. The daily release deletes the existing release and tag first.
Dropping a `community_submission.v2.schema.json` into `schemas/` promotes it atomically across every
reader and writer. That is a one-way door for contributors — only on explicit request.
## Do not chase
## Conventions
- `process-historical-faa.yaml` is dead: missing `src/get_historical_faa.py`,
`scripts/concat_csvs.py`, and uses the disabled `::set-output`.
- `af-klm-fleet/package.json``npm run validate` has no `scripts/validate.js`.
- `af-klm-fleet/` and `community-routes/` are unwired; nothing in CI touches them.
`af-klm-fleet/README.md` is generated.
- **Empty string, not null**, everywhere in released frames.
- Reuse `derive_from_faa_master_txt.normalize()` for `openairframes_id`; never re-derive the format.
- Python `3.14` for FAA/community/vendor jobs, `3.12` for ADS-B jobs (pyarrow pin + multiprocessing).
Deliberate. Match the surrounding job; do not unify.
## Flag, do not silently fix
## References with no target — do not chase as regressions
| Reference | Missing |
|---|---|
| `process-historical-faa.yaml` | `src/get_historical_faa.py`, `scripts/concat_csvs.py` |
| `af-klm-fleet/package.json``npm run validate` | `af-klm-fleet/scripts/validate.js` |
`process-historical-faa.yaml` is dead, not stale — it also uses the disabled `::set-output`.
Repair-vs-delete is the owner's call; leave it alone unprompted.
## `af-klm-fleet/` and `community-routes/` are unwired
Nothing in CI touches either, and nothing consumes `community-routes/`. `af-klm-fleet/` is a vendored
project by a different author with its own license — its aircraft model is unrelated to
`schemas/community_submission.*`, so do not merge the two. Its `README.md` is generated by
`generate-readme.js`; hand edits are overwritten.
## Warts left standing — flag, do not silently fix
- `NUMBER_PARTS` is restated by the matrix and four hand-written upload steps in
`adsb-to-aircraft-for-day.yaml`; changing the constant alone silently drops data. YAML cannot loop
upload steps and one merged artifact would force every map job to download all parts, so any real
fix is a restructure.
- `MAX_WORKERS = OS_CPU_COUNT if OS_CPU_COUNT > 4 else 1` collapses to a single worker on a ≤4-core
runner, shrinking `files_per_batch` with it. Possibly intentional memory control — do not raise it
without measuring peak RSS on the target runner.
- `update-community-prs.yaml` runs `regenerate_pr_schema || true` then force-pushes, so a
regeneration failure ships anyway. Making it fatal leaves PRs un-rebased instead — a judgment call.
- `approve_submission.py` wraps its schema update in a bare `except Exception`, so a submission can
merge without its new tags reaching the schema.
## Workflow authoring
The user's global GitHub Actions rules apply. Existing workflows violate most of them.
**Do not bulk-remediate** — bring only the file you were asked to touch up to standard, and surface
the rest in chat.
## External sources
`registry.faa.gov` and ADS-B Exchange are **required** — the release fails without them. Mictronics is
**tolerated**: it retries, then the job continues without it. adsb.lol may simply not have published a
given day, in which case the previous CSV is re-released rather than failing.
FAA refreshes at 05:30 UTC; the release cron fires at 06:00 UTC. That 30-minute margin is the reason
for the schedule.
- `NUMBER_PARTS` is restated by the matrix and four upload steps in `adsb-to-aircraft-for-day.yaml`.
- `MAX_WORKERS = ... if OS_CPU_COUNT > 4 else 1` collapses to one worker on a ≤4-core runner.
- `update-community-prs.yaml` runs `regenerate_pr_schema || true`, then force-pushes.
- `approve_submission.py` wraps its schema update in a bare `except Exception`.
- Existing workflows violate the global GHA rules. Fix only the file you were asked to touch.
+66
View File
@@ -0,0 +1,66 @@
OpenAirframes — Data Source Notices
===================================
The LICENSE file covers the *code* in this repository. It does not cover the
*data* published in releases. Several upstream registries permit redistribution
only on condition that specific notices travel with the data. Those conditions
are reproduced below. This file is published as an asset on every release; anyone
redistributing a release asset further must carry the corresponding notice with it.
Removing or altering a notice in this file removes the permission that makes the
corresponding asset redistributable.
Transport Canada — Canadian Civil Aircraft Register (CCARCS)
------------------------------------------------------------
Asset: openairframes_tc_*.csv
Source: https://wwwapps.tc.gc.ca/saf-sec-sur/2/ccarcs-riacc/download/ccarcsdb.zip
Redistribution is permitted under the Government of Canada terms recorded for this
dataset. They require both of the following notices, verbatim, and require that the
two reach the consumer together. The governing instrument is not published on the
CCARCS download page, so no licence name or URL is asserted here.
Reproduced and distributed with the permission of the Government of Canada.
This product has been produced by or for the OpenAirframes project and
includes data provided by the Government of Canada. The incorporation of
data sourced from the Government of Canada within this product shall not be
construed as constituting an endorsement by the Government of Canada of our
product.
Registered-owner mailing addresses are redistributed, matching the registrant
addresses the FAA asset already carries. Because CCARCS lists one row per party,
the published address is that of the single designated mail recipient rather than
a merge across co-owners.
FAA — Releasable Aircraft Database
-----------------------------------
Source: https://registry.faa.gov/database/ReleasableAircraft.zip
ReleasableAircraft_*.zip is redistributed unmodified. It is a work of the United
States federal government, not subject to copyright protection in the United
States (17 U.S.C. § 105). No notice is required; it is credited for provenance.
openairframes_faa_*.csv is a derived product built by this repository —
normalized, joined, deduplicated, and extended with an identifier this project
defines. Section 105 disclaims copyright in the government's own work and says
nothing about a derivative, so no claim is made here about the CSV's status.
Other redistributed assets
---------------------------
The following assets are republished from third parties whose terms have not
been assessed in this repository. They are listed for provenance only; nothing
here asserts a licence over them.
openairframes_adsb_*.csv.gz derived from adsb.lol daily globe history
(github.com/adsblol), with registration data
from tar1090-db
basic-ac-db_*.json.gz ADS-B Exchange, downloads.adsbexchange.com
mictronics-db_*.zip Mictronics, www.mictronics.de
Community submissions under community/ are contributed by their authors through
the repository's submission workflow and are published with the attribution each
contributor selected.
+23 -2
View File
@@ -26,13 +26,34 @@ df = pd.read_csv(url)
df
```
![](docs/images/df_adsb_example_0.png)
- **openairframes_faa.csv**
All [FAA registration data](https://www.faa.gov/licenses_certificates/aircraft_certification/aircraft_registry/releasable_aircraft_download) from 2023-08-16 to present (~260 MB)
- **openairframes_registry.csv**
Every national registry in one table, one row per registration record, with a `source`
column naming the registry it came from. Currently the FAA (United States) and Transport
Canada. Identifier columns (`transponder_code_hex`, `registration_number`,
`openairframes_id`) lead the table and are populated for every source.
- **openairframes_faa.csv**
All [FAA registration data](https://www.faa.gov/licenses_certificates/aircraft_certification/aircraft_registry/releasable_aircraft_download) from 2023-08-16 to present (~275 MB).
Superseded by `openairframes_registry.csv`; still published so existing consumers keep working.
- **openairframes_tc.csv**
The [Transport Canada Civil Aircraft Register](https://wwwapps.tc.gc.ca/saf-sec-sur/2/ccarcs-riacc/RchSimp.aspx),
~35k aircraft with full ICAO 24-bit hex coverage. Also folded into
`openairframes_registry.csv`; published separately for the same reason as the FAA CSV.
- **ReleasableAircraft_{date}.zip**
A daily snapshot of the FAA database, which updates at **05:30 UTC**
- **basic-ac-db.json.gz**
[ADS-B Exchange](https://www.adsbexchange.com/) basic aircraft database, republished unmodified.
- **mictronics-db.zip**
[Mictronics](https://www.mictronics.de/aircraft-database/) aircraft database, republished
unmodified. Best effort — the release ships without it when the source is unavailable.
Redistribution terms for the underlying sources travel with the release in **NOTICE**. Some
registries permit redistribution only on condition that specific notices reach you with the data.
---
## For Contributors
+128
View File
@@ -0,0 +1,128 @@
"""Join the per-source registry CSVs into one union table.
Every source publishes its own `openairframes_<source>_{start}_{end}.csv` on its own thread.
This reads whatever landed, aligns them on the union of columns, and writes a single
`openairframes_registry_{start}_{end}.csv` discriminated by the `source` column.
Adding a registry means adding a source to the workflow matrix; nothing here changes.
Usage:
python src/build_registry.py --input-dir artifacts/registry --date 2026-08-31
"""
from datetime import datetime, timezone
from pathlib import Path
import argparse
import re
import sys
import pandas as pd
# Sources that are registries. Community and ADS-B are published separately: they are
# observations and contributions, not registration records, and do not share this schema.
FILENAME_RE = re.compile(
r"\Aopenairframes_(?P<source>[a-z0-9_]+?)_"
r"(?P<start>\d{4}-\d{2}-\d{2})_(?P<end>\d{4}-\d{2}-\d{2})\.csv\Z"
)
EXCLUDED_SOURCES = {"community", "adsb", "registry"}
# Identifier columns lead the union so the table is usable without reading 70 headers.
LEADING_COLUMNS = [
"download_date",
"source",
"transponder_code_hex",
"registration_number",
"openairframes_id",
]
def discover(input_dir: Path) -> list[tuple[str, str, str, Path]]:
"""Return (source, start, end, path) for each per-source registry CSV found."""
found = []
for path in sorted(input_dir.rglob("openairframes_*.csv")):
match = FILENAME_RE.match(path.name)
if not match:
print(f" SKIP {path.name}: does not match {FILENAME_RE.pattern}")
continue
source = match.group("source")
if source in EXCLUDED_SOURCES:
print(f" SKIP {path.name}: {source!r} is published as its own asset")
continue
found.append((source, match.group("start"), match.group("end"), path))
return found
def build(input_dir: Path, date_str: str) -> tuple[pd.DataFrame, str, str]:
parts = discover(input_dir)
if not parts:
raise SystemExit(f"No per-source registry CSVs found under {input_dir}")
seen = [p[0] for p in parts]
duplicated = {s for s in seen if seen.count(s) > 1}
if duplicated:
raise SystemExit(f"More than one file claims source {sorted(duplicated)}")
frames = []
for source, _, _, path in parts:
# keep_default_na=False so a literal "NA" survives the round trip unchanged.
df = pd.read_csv(path, dtype=str, keep_default_na=False)
if "source" not in df.columns:
raise SystemExit(f"{path.name}: no source column; cannot discriminate rows")
if df.empty:
print(f" {source}: empty, skipping")
continue
actual = {v.strip().lower() for v in df["source"].unique()}
if actual != {source.lower()}:
# A file whose rows disagree with its name would duplicate another source into
# the union under the wrong label.
raise SystemExit(
f"{path.name}: filename says {source!r}, rows say {sorted(actual)}"
)
print(f" {source}: {len(df)} rows, {len(df.columns)} columns from {path.name}")
frames.append(df)
if not frames:
raise SystemExit("Every discovered source was empty; refusing to publish an empty registry")
if len(frames) > 1:
shared = set.intersection(*(set(f.columns) for f in frames)) - set(LEADING_COLUMNS)
print(f" columns shared across sources ({len(shared)}): {sorted(shared)}")
columns = list(dict.fromkeys(c for df in frames for c in df.columns))
ordered = [c for c in LEADING_COLUMNS if c in columns]
ordered += [c for c in columns if c not in ordered]
# reindex rather than concat directly: a source missing a column must yield an empty
# cell, never a shifted row.
df_union = pd.concat([df.reindex(columns=ordered) for df in frames], ignore_index=True)
df_union = df_union.fillna("")
# Earliest start across sources, not per-source coverage: a source added today still
# carries the oldest source's start date in the filename.
start = min(p[1] for p in parts)
end = max(p[2] for p in parts + [("", "", date_str, Path())])
return df_union, start, end
def main() -> None:
parser = argparse.ArgumentParser(description="Join per-source registry CSVs into one table")
parser.add_argument("--input-dir", default="artifacts/registry", help="Directory to search")
parser.add_argument("--output-dir", default="data/openairframes", help="Where to write")
parser.add_argument("--date", help="Run date (YYYY-MM-DD, default: today UTC)")
args = parser.parse_args()
date_str = args.date or datetime.now(timezone.utc).strftime("%Y-%m-%d")
df, start, end = build(Path(args.input_dir), date_str)
out_dir = Path(args.output_dir)
out_dir.mkdir(parents=True, exist_ok=True)
out_path = out_dir / f"openairframes_registry_{start}_{end}.csv"
df.to_csv(out_path, index=False)
print(f"Wrote {out_path}: {len(df)} rows, {len(df.columns)} columns")
print(" rows per source:")
for source, count in df["source"].value_counts().items():
print(f" {source}: {count}")
if __name__ == "__main__":
main()
+22 -5
View File
@@ -4,6 +4,10 @@ import argparse
parser = argparse.ArgumentParser(description="Create daily FAA release")
parser.add_argument("--date", type=str, help="Date to process (YYYY-MM-DD format, default: today)")
parser.add_argument("--allow-bootstrap", action="store_true",
help="Permit rebuilding from a single day when no published asset is found. "
"Onboarding only: a missing asset is otherwise indistinguishable from a "
"transient outage, and rebuilding would erase the accumulated history.")
args = parser.parse_args()
if args.date:
@@ -37,13 +41,26 @@ from derive_from_faa_master_txt import convert_faa_master_txt_to_df, concat_faa_
from get_latest_release import get_latest_aircraft_faa_csv_df
df_new = convert_faa_master_txt_to_df(zip_path, date_str)
# Only a genuine first run may rebuild from a single day. A rate limit, a parse error or a
# non-monotonic download_date must stop the run: this file becomes tomorrow's base, so
# silently republishing one day erases the accumulated history.
try:
df_base, start_date_str = get_latest_aircraft_faa_csv_df()
df_base = concat_faa_historical_df(df_base, df_new)
assert df_base['download_date'].is_monotonic_increasing, "download_date is not monotonic increasing"
except Exception as e:
print(f"No existing FAA release found, using only new data: {e}")
df_base = df_new
except FileNotFoundError as e:
if not args.allow_bootstrap:
raise SystemExit(
f"No published FAA asset found: {e}\n"
"This is indistinguishable from a transient outage, and rebuilding from one day "
"would erase the accumulated history. Pass --allow-bootstrap when onboarding."
) from None
print(f"Bootstrapping FAA from today only (--allow-bootstrap): {e}")
df_base = None
start_date_str = date_str
if df_base is not None:
df_base = concat_faa_historical_df(df_base, df_new)
assert df_base['download_date'].is_monotonic_increasing, "download_date is not monotonic increasing"
else:
df_base = df_new
df_base.to_csv(OUT_ROOT / f"openairframes_faa_{start_date_str}_{date_str}.csv", index=False)
+83
View File
@@ -0,0 +1,83 @@
from pathlib import Path
from datetime import datetime, timezone
import argparse
parser = argparse.ArgumentParser(description="Create daily Transport Canada release")
parser.add_argument("--date", type=str, help="Date to process (YYYY-MM-DD format, default: today)")
parser.add_argument("--allow-bootstrap", action="store_true",
help="Permit rebuilding from a single day when no published asset is found. "
"Onboarding only: a missing asset is otherwise indistinguishable from a "
"transient outage, and rebuilding would erase the accumulated history.")
args = parser.parse_args()
if args.date:
date_str = args.date
else:
date_str = datetime.now(timezone.utc).strftime("%Y-%m-%d")
out_dir = Path("data/tc_ccarcs")
out_dir.mkdir(parents=True, exist_ok=True)
zip_name = f"ccarcsdb_{date_str}.zip"
zip_path = out_dir / zip_name
if not zip_path.exists():
url = "https://wwwapps.tc.gc.ca/saf-sec-sur/2/ccarcs-riacc/download/ccarcsdb.zip"
from urllib.request import Request, urlopen
# CCARCS 403s a default urllib agent. Any browser-like UA works; the exact
# version string is not load-bearing.
req = Request(
url,
headers={
"User-Agent": (
"Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36"
)
},
method="GET",
)
with urlopen(req, timeout=120) as r:
body = r.read()
# TC serves an HTML maintenance page with a 200, which would otherwise be cached
# under a .zip name and re-read on every later run.
if body[:2] != b"PK":
raise RuntimeError(f"{url} did not return a zip (got {body[:40]!r})")
tmp_path = zip_path.with_suffix(".part")
tmp_path.write_bytes(body)
tmp_path.replace(zip_path)
OUT_ROOT = Path("data/openairframes")
OUT_ROOT.mkdir(parents=True, exist_ok=True)
from derive_from_tc_ccarcs import convert_tc_ccarcs_to_df
# Named for FAA but column-agnostic: fingerprints every column except download_date.
from derive_from_faa_master_txt import concat_faa_historical_df
from get_latest_release import get_latest_aircraft_tc_csv_df
df_new = convert_tc_ccarcs_to_df(zip_path, date_str)
# Only a genuine first run may rebuild from a single day. Every other failure -- a rate
# limit, a schema change, a truncated download -- must stop the run, because this file
# becomes tomorrow's base and silently republishing one day erases the whole history.
try:
df_base, start_date_str = get_latest_aircraft_tc_csv_df()
except FileNotFoundError as e:
if not args.allow_bootstrap:
raise SystemExit(
f"No published Transport Canada asset found: {e}\n"
"This is indistinguishable from a transient outage, and rebuilding from one day "
"would erase the accumulated history. Pass --allow-bootstrap when onboarding."
) from None
print(f"Bootstrapping Transport Canada from today only (--allow-bootstrap): {e}")
df_base = None
start_date_str = date_str
if df_base is not None:
missing = set(df_base.columns) ^ set(df_new.columns)
if missing:
raise SystemExit(f"Column set changed since the last release: {sorted(missing)}")
df_base = concat_faa_historical_df(df_base, df_new)
assert df_base['download_date'].is_monotonic_increasing, "download_date is not monotonic increasing"
else:
df_base = df_new
df_base.to_csv(OUT_ROOT / f"openairframes_tc_{start_date_str}_{date_str}.csv", index=False)
+248
View File
@@ -0,0 +1,248 @@
from pathlib import Path
import csv
import io
import re
import zipfile
import pandas as pd
from derive_from_faa_master_txt import normalize
# CCARCS ships headerless, latin1, comma-delimited exports. Column names come from
# carslayout.txt in the same archive and must stay in file order.
CARSCURR_COLUMNS = [
"MARK", "REGISTRATION_SUB_TYPE_E", "REGISTRATION_SUB_TYPE_F", "COMMON_NAME",
"MODEL_NAME", "MANUFACTURERS_SERIAL_NUMBER", "MANUFACTURER_SERIAL_COMPRESSED",
"ID_PLATE_MANUFACTURERS_NAME", "BASIS_FOR_REGISTRATION", "BASIS_FOR_REGISTRATION_F",
"AIRCRAFT_CATEGORY_E", "AIRCRAFT_CATEGORY_F", "DATE_OF_IMPORT", "ENGINE_MANUF",
"POWERGLIDER_FLAG", "ENGINE_CATEGORY_E", "ENGINE_CATEGORY_F", "NUMBER_OF_ENGINES",
"NUMBER_OF_SEATS", "AIR_WEIGHT_KILOS", "SALE_REPORTED", "ISSUE_DATE",
"EFFECTIVE_DATE", "INEFFECTIVE_DATE", "REGISTERED_PURPOSE_E", "REGISTERED_PURPOSE_F",
"FLIGHT_AUTHORITY_E", "FLIGHT_AUTHORITY_F", "MANUFACTURE_OR_ASSEMBLY",
"COUNTRY_MANUFACTURE_ASS_E", "COUNTRY_MANUFACTURE_ASS_F", "DATE_MANUFACTURE_ASSEMBLY",
"BASE_OF_OPERATIONS_CTRY_E", "BASE_OF_OPERATIONS_CTRY_F", "BASE_PROVINCE_OR_STATE_E",
"BASE_PROVINCE_OR_STATE_F", "CITY_AIRPORT", "TYPE_CERTIFICATE_NUMBER",
"REGISTRATION_AUTH_STATUS_E", "REGISTRATION_AUTH_STATUS_F", "MULTIPLE_OWNER_FLAG",
"MODIFIED_DATE", "MODE_S_TRANSPONDER_BINARY", "PHYSICAL_FILE_REGION_E",
"PHYSICAL_FILE_REGION_F", "EX_MILITARY_MARK", "TRIMMED_MARK",
]
CARSOWNR_COLUMNS = [
"MARK_LINK", "FULL_NAME", "TRADE_NAME", "STREET_NAME", "STREET_NAME2", "CITY",
"PROVINCE_OR_STATE_E", "PROVINCE_OR_STATE_F", "POSTAL_CODE", "COUNTRY_E", "COUNTRY_F",
"TYPE_OF_OWNER_E", "TYPE_OF_OWNER_F", "ACTIVE_FLAG", "CARE_OF", "REGION_E", "REGION_F",
"OWNER_NAME_OLD_FORMAT", "MAIL_RECIPIENT", "TRIMMED_MARK",
]
# Mailing address of the single designated recipient, matching the registrant_* address
# the FAA build already publishes. Addresses are per-party, so they are taken from the one
# MAIL_RECIPIENT row rather than merged across co-owners.
# registrant_zip_code holds the Canadian postal code: the name is the FAA's, and a union
# table needs one column per concept, not one per country's vocabulary.
OWNER_ADDRESS_COLUMNS = {
"STREET_NAME": "registrant_street_1",
"STREET_NAME2": "registrant_street_2",
"CITY": "registrant_city",
"POSTAL_CODE": "registrant_zip_code",
"CARE_OF": "registrant_care_of",
}
FOOTER_RE = re.compile(r"\s*(\d+) rows selected\.\s*")
# Floor, not an expectation: Canada's register is ~35k aircraft and carsownr is larger
# still, so 1000 only catches a grossly truncated export. The footer row-count check above
# is what actually validates the parse; this guards the case where the footer agrees with a
# near-empty body.
MIN_EXPECTED_ROWS = 1000
def _read_ccarcs_entry(zip_path: Path, entry: str, columns: list[str]) -> pd.DataFrame:
"""Read one headerless CCARCS export into a DataFrame.
Raises:
ValueError: on any row whose width is neither the declared column count nor a
blank/footer line, on a missing or disagreeing "N rows selected." footer, or
on a row count below MIN_EXPECTED_ROWS.
"""
with zipfile.ZipFile(zip_path) as z:
text = z.read(entry).decode("latin1")
rows = []
declared = None
# newline="" so a CRLF export does not leave \r on the final field of every row.
for row in csv.reader(io.StringIO(text, newline="")):
if len(row) == len(columns):
rows.append([cell.strip() for cell in row])
continue
if not row or not any(cell.strip() for cell in row):
continue # trailing blank line
match = FOOTER_RE.fullmatch(row[0]) if len(row) == 1 else None
if match:
declared = int(match.group(1))
continue
raise ValueError(
f"{entry}: row with {len(row)} fields, expected {len(columns)}: {row[:3]!r}"
)
# The spool footer is a free checksum from the source; a short export is otherwise
# indistinguishable from a genuinely smaller register.
if declared is None:
raise ValueError(f"{entry}: no 'N rows selected.' footer; export is truncated")
if declared != len(rows):
raise ValueError(f"{entry}: footer declares {declared} rows, parsed {len(rows)}")
if len(rows) < MIN_EXPECTED_ROWS:
raise ValueError(f"{entry}: only {len(rows)} rows, expected >= {MIN_EXPECTED_ROWS}")
return pd.DataFrame(rows, columns=columns)
def tc_full_registration(mark: str) -> str:
"""Expand a trimmed CCARCS mark into the full Canadian registration.
CCARCS stores the bare mark in both MARK and TRIMMED_MARK, so the prefix has to be
reconstructed: three-character marks are vintage CF- registrations, everything else
takes the modern C- prefix. Returns "" for a blank mark.
"""
mark = (mark or "").strip().upper()
if not mark:
return ""
return f"CF-{mark}" if len(mark) == 3 else f"C-{mark}"
def binary_to_hex(binary: str) -> str:
"""Convert a 24-bit Mode S binary string to a 6-digit uppercase hex address.
Returns "" for empty, non-binary, or non-24-bit input. Width is checked because
this column is the join key against ADS-B data: a short field would otherwise
zero-pad into a plausible address belonging to a different aircraft.
"""
binary = (binary or "").strip()
if len(binary) != 24 or any(c not in "01" for c in binary):
return ""
return f"{int(binary, 2):06X}"
def _merge_owners(df_ownr: pd.DataFrame) -> pd.DataFrame:
"""Collapse the active registered parties for each mark into a single row.
A co-owned mark repeats with a different party each time; keeping only the mail
recipient would silently drop the rest.
Each field is deduplicated and blank-skipped independently, so the values are NOT
index-parallel: a mark with three owners can emit three names but one province.
Consumers must not split on ", " and zip the columns together.
"""
# ACTIVE_FLAG is "A"/"I", but "I" does not mean "former owner": 1,932 currently
# Registered marks carry only "I" parties, and those rows are the MAIL_RECIPIENT.
# So prefer active parties where a mark has any, and fall back to all of them
# rather than publishing a registered aircraft with no owner at all.
all_parties = df_ownr
active = df_ownr[df_ownr["ACTIVE_FLAG"].str.upper() == "A"]
marks_with_active = set(active["TRIMMED_MARK"])
df_ownr = pd.concat([
active,
df_ownr[~df_ownr["TRIMMED_MARK"].isin(marks_with_active)],
])
def join_unique(series: pd.Series) -> str:
seen = []
for value in series:
value = (value or "").strip()
if value and value not in seen:
seen.append(value)
return ", ".join(seen)
def count_distinct(series: pd.Series) -> int:
return len({v.strip() for v in series if v and v.strip()})
grouped = df_ownr.groupby("TRIMMED_MARK", sort=False).agg(
registrant_name=("FULL_NAME", join_unique),
registrant_state=("PROVINCE_OR_STATE_E", join_unique),
registrant_country=("COUNTRY_E", join_unique),
registrant_type=("TYPE_OF_OWNER_E", join_unique),
registrant_party_count=("FULL_NAME", count_distinct),
).reset_index()
# A party row states its own type ("Individual"); that stops being true of the mark
# once several parties share it. Counting distinct names rather than rows keeps this
# consistent with owner_name, which is also deduplicated.
grouped.loc[grouped["registrant_party_count"] > 1, "registrant_type"] = "Co-owner"
# Taken from the unfiltered frame: the designated recipient is the designated
# recipient even when its own party row is flagged inactive.
recipient = (
all_parties[all_parties["MAIL_RECIPIENT"].str.upper() == "Y"]
.drop_duplicates(subset="TRIMMED_MARK", keep="first")
.rename(columns=OWNER_ADDRESS_COLUMNS)
)
return grouped.merge(
recipient[["TRIMMED_MARK", *OWNER_ADDRESS_COLUMNS.values()]],
on="TRIMMED_MARK",
how="left",
)
def convert_tc_ccarcs_to_df(zip_path: Path, date: str) -> pd.DataFrame:
"""Build the OpenAirframes Transport Canada frame from a CCARCS zip."""
df = _read_ccarcs_entry(zip_path, "carscurr.txt", CARSCURR_COLUMNS)
df_ownr = _read_ccarcs_entry(zip_path, "carsownr.txt", CARSOWNR_COLUMNS)
df = df.merge(_merge_owners(df_ownr), on="TRIMMED_MARK", how="left")
out = pd.DataFrame({
"download_date": date,
# The FAA frame already carries `source`; it is the union discriminator.
"source": "TC",
"transponder_code_hex": df["MODE_S_TRANSPONDER_BINARY"].map(binary_to_hex),
"registration_number": df["TRIMMED_MARK"].map(tc_full_registration),
"mark": df["TRIMMED_MARK"],
"aircraft_manufacturer": df["COMMON_NAME"],
"aircraft_model": df["MODEL_NAME"],
"serial_number": df["MANUFACTURERS_SERIAL_NUMBER"],
"aircraft_category": df["AIRCRAFT_CATEGORY_E"],
"engine_manufacturer": df["ENGINE_MANUF"],
"engine_category": df["ENGINE_CATEGORY_E"],
"aircraft_number_of_engines": df["NUMBER_OF_ENGINES"],
"aircraft_number_of_seats": df["NUMBER_OF_SEATS"],
"max_weight_kilos": df["AIR_WEIGHT_KILOS"],
"status": df["REGISTRATION_AUTH_STATUS_E"],
"registration_sub_type": df["REGISTRATION_SUB_TYPE_E"],
"basis_for_registration": df["BASIS_FOR_REGISTRATION"],
"registered_purpose": df["REGISTERED_PURPOSE_E"],
"flight_authority": df["FLIGHT_AUTHORITY_E"],
"type_certificate_number": df["TYPE_CERTIFICATE_NUMBER"],
"country_manufacture": df["COUNTRY_MANUFACTURE_ASS_E"],
"date_manufacture_assembly": df["DATE_MANUFACTURE_ASSEMBLY"],
"base_country": df["BASE_OF_OPERATIONS_CTRY_E"],
"base_province_or_state": df["BASE_PROVINCE_OR_STATE_E"],
"city_airport": df["CITY_AIRPORT"],
"ex_military_mark": df["EX_MILITARY_MARK"],
"multiple_owner_flag": df["MULTIPLE_OWNER_FLAG"],
"registrant_name": df["registrant_name"],
"registrant_type": df["registrant_type"],
"registrant_state": df["registrant_state"],
"registrant_country": df["registrant_country"],
"registrant_care_of": df["registrant_care_of"],
"registrant_street_1": df["registrant_street_1"],
"registrant_street_2": df["registrant_street_2"],
"registrant_city": df["registrant_city"],
"registrant_zip_code": df["registrant_zip_code"],
"issue_date": df["ISSUE_DATE"],
"effective_date": df["EFFECTIVE_DATE"],
"ineffective_date": df["INEFFECTIVE_DATE"],
"modified_date": df["MODIFIED_DATE"],
})
# Position matches the FAA frame (after registration_number). Ordering is cosmetic:
# concat_faa_historical_df reindexes df_new to the base's columns before merging.
out.insert(3, "openairframes_id", (
normalize(out["aircraft_manufacturer"])
+ "|"
+ normalize(out["aircraft_model"])
+ "|"
+ normalize(out["serial_number"])
))
out = out.fillna("")
out = out.replace("None", "")
return out
+87 -9
View File
@@ -3,6 +3,7 @@ from __future__ import annotations
from dataclasses import dataclass
from pathlib import Path
from typing import Iterable, Optional
import os
import re
import urllib.request
import urllib.error
@@ -138,15 +139,29 @@ def download_latest_aircraft_csv(
Path to the downloaded file
"""
output_dir = Path(output_dir)
assets = get_latest_release_assets(repo, github_token=github_token)
try:
asset = pick_asset(assets, name_regex=r"^openairframes_faa_.*\.csv$")
except FileNotFoundError:
# Fallback to old naming pattern
asset = pick_asset(assets, name_regex=r"^openairframes_\d{4}-\d{2}-\d{2}_.*\.csv$")
saved_to = download_asset(asset, output_dir / asset.name, github_token=github_token)
print(f"Downloaded: {asset.name} ({asset.size} bytes) -> {saved_to}")
return saved_to
github_token = github_token or os.environ.get("GITHUB_TOKEN")
for release in get_releases(repo, github_token=github_token, per_page=30):
assets = get_release_assets_from_release_data(release)
try:
asset = pick_asset(assets, name_regex=r"^openairframes_faa_.*\.csv$")
except FileNotFoundError:
try:
# Fallback to old naming pattern
asset = pick_asset(assets, name_regex=r"^openairframes_\d{4}-\d{2}-\d{2}_.*\.csv$")
except FileNotFoundError:
continue
saved_to = download_asset(asset, output_dir / asset.name, github_token=github_token)
if asset.size and saved_to.stat().st_size != asset.size:
raise RuntimeError(
f"{asset.name}: downloaded {saved_to.stat().st_size} bytes, expected {asset.size}"
)
print(f"Downloaded: {asset.name} ({asset.size} bytes) -> {saved_to}")
return saved_to
raise FileNotFoundError(
"No release in the last 30 releases has an asset matching 'openairframes_faa_.*\\.csv$'"
)
def get_latest_aircraft_faa_csv_df():
csv_path = download_latest_aircraft_csv()
@@ -167,6 +182,69 @@ def get_latest_aircraft_faa_csv_df():
return df, date_str
def download_latest_aircraft_tc_csv(
output_dir: Path = Path("downloads"),
github_token: Optional[str] = None,
repo: str = REPO,
) -> Path:
"""
Download the latest openairframes_tc_*.csv file from the latest GitHub release.
Args:
output_dir: Directory to save the downloaded file (default: "downloads")
github_token: Optional GitHub token for authentication
repo: GitHub repository in format "owner/repo" (default: REPO)
Returns:
Path to the downloaded file
"""
output_dir = Path(output_dir)
github_token = github_token or os.environ.get("GITHUB_TOKEN")
# Walk back through releases rather than reading only `latest`. The TC asset is
# optional, so a single failed build publishes a release without it; anchoring on
# `latest` would then make the caller rebuild history from one day and republish
# that as the whole dataset.
for release in get_releases(repo, github_token=github_token, per_page=30):
assets = get_release_assets_from_release_data(release)
try:
asset = pick_asset(assets, name_regex=r"^openairframes_tc_.*\.csv$")
except FileNotFoundError:
continue
saved_to = download_asset(asset, output_dir / asset.name, github_token=github_token)
if asset.size and saved_to.stat().st_size != asset.size:
raise RuntimeError(
f"{asset.name}: downloaded {saved_to.stat().st_size} bytes, expected {asset.size}"
)
print(f"Downloaded: {asset.name} ({asset.size} bytes) -> {saved_to}")
return saved_to
raise FileNotFoundError(
"No release in the last 30 releases has an asset matching 'openairframes_tc_.*\\.csv$'"
)
def get_latest_aircraft_tc_csv_df():
"""Return (DataFrame, start_date_str) for the most recent published TC release.
Raises FileNotFoundError when no recent release carries a TC asset, and ValueError
when the asset filename has no parseable start date.
"""
csv_path = download_latest_aircraft_tc_csv()
import pandas as pd
# keep_default_na=False: a literal "NA"/"N/A" in the source would otherwise read back
# as NaN -> "" while the fresh parse keeps the string, so the row fingerprints would
# never match and every affected record would re-append on every run.
df = pd.read_csv(csv_path, dtype=str, keep_default_na=False)
df = df.fillna("")
# Only the start date is taken; the end date is always the run's own date.
match = re.search(r"openairframes_tc_(\d{4}-\d{2}-\d{2})_", str(csv_path))
if not match:
raise ValueError(f"Could not extract date from filename: {csv_path.name}")
return df, match.group(1)
def download_latest_aircraft_adsb_csv(
output_dir: Path = Path("downloads"),
github_token: Optional[str] = None,