diff --git a/.github/workflows/openairframes-daily-release.yaml b/.github/workflows/openairframes-daily-release.yaml index ca9003d..7dcb5ce 100644 --- a/.github/workflows/openairframes-daily-release.yaml +++ b/.github/workflows/openairframes-daily-release.yaml @@ -10,11 +10,19 @@ on: description: 'Date to process (YYYY-MM-DD format, default: yesterday)' required: false type: string + bootstrap_source: + description: 'Onboarding only: the one source id permitted to rebuild from a single day' + required: false + type: string permissions: contents: write actions: write +concurrency: + group: ${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: false + jobs: trigger-releases: runs-on: ubuntu-latest @@ -42,39 +50,81 @@ jobs: ref: 'develop' }); - build-faa: - runs-on: ubuntu-24.04-arm + # One thread per registry. Adding a source is one matrix entry plus + # src/create_daily__release.py; join-registry picks it up by artifact pattern. + build-registry-source: if: github.event_name != 'schedule' + strategy: + fail-fast: false + matrix: + include: + - source: faa + required: true + - source: tc + required: false + uses: ./.github/workflows/registry-source.yaml + with: + source: ${{ matrix.source }} + date: ${{ inputs.date }} + required: ${{ matrix.required }} + # Standing permission would also fire on an outage. A source bootstraps only when a + # human dispatches the run naming it. + allow_bootstrap: ${{ inputs.bootstrap_source == matrix.source }} + + join-registry: + needs: build-registry-source + # No always(): a tolerated source fails its step without failing its leg, so this only + # blocks when a required source could not be built. + if: github.event_name != 'schedule' + runs-on: ubuntu-24.04-arm + timeout-minutes: 20 + permissions: + contents: read steps: - name: Checkout - uses: actions/checkout@v6 - with: - fetch-depth: 0 + uses: actions/checkout@v7 - name: Setup Python - uses: actions/setup-python@v6 + uses: actions/setup-python@v7 with: python-version: "3.14" + cache: 'pip' + cache-dependency-path: requirements.txt - name: Install dependencies run: | python -m pip install --upgrade pip pip install -r requirements.txt - - name: Run FAA release script - run: | - python src/create_daily_faa_release.py ${{ inputs.date && format('--date {0}', inputs.date) || '' }} - ls -lah data/faa_releasable - ls -lah data/openairframes - - - name: Upload FAA artifacts - uses: actions/upload-artifact@v4 + - name: Download every source thread + uses: actions/download-artifact@v8 with: - name: faa-release - path: | - data/openairframes/openairframes_faa_*.csv - data/faa_releasable/ReleasableAircraft_*.zip + pattern: registry-* + path: artifacts/registry + merge-multiple: true + + - name: Join sources into one registry + env: + RUN_DATE: ${{ inputs.date }} + run: | + python src/build_registry.py --input-dir artifacts/registry ${RUN_DATE:+--date "$RUN_DATE"} + + - name: Stage release assets + run: | + mkdir -p data/release-out + # Every per-source CSV must ship: each source reads its OWN previous asset to + # accumulate, so one that is never published can never be anything but day one. + cp artifacts/registry/* data/release-out/ + cp data/openairframes/openairframes_registry_*.csv data/release-out/ + ls -lah data/release-out + + - name: Upload registry + uses: actions/upload-artifact@v7 + with: + name: union-registry + path: data/release-out retention-days: 1 + if-no-files-found: error resolve-dates: runs-on: ubuntu-latest @@ -233,7 +283,7 @@ jobs: create-release: runs-on: ubuntu-latest - needs: [resolve-dates, build-faa, adsb-to-aircraft, adsb-reduce, build-community, build-adsbexchange-json, build-mictronics-db] + needs: [resolve-dates, join-registry, adsb-to-aircraft, adsb-reduce, build-community, build-adsbexchange-json, build-mictronics-db] if: github.event_name != 'schedule' && !cancelled() steps: - name: Check ADS-B workflow status @@ -246,12 +296,13 @@ jobs: with: sparse-checkout: | .github + NOTICE sparse-checkout-cone-mode: false - - name: Download FAA artifacts + - name: Download joined registry uses: actions/download-artifact@v5 with: - name: faa-release + name: union-registry path: artifacts/faa - name: Download ADS-B artifacts @@ -311,6 +362,12 @@ jobs: # Find files from artifacts using find (handles nested structures) CSV_FILE_FAA=$(find artifacts/faa -name "openairframes_faa_*.csv" -type f 2>/dev/null | head -1) + CSV_FILE_REGISTRY=$(find artifacts/faa -name "openairframes_registry_*.csv" -type f 2>/dev/null | head -1) + # Every per-source registry CSV, whatever sources the matrix ran. + SOURCE_CSVS=$(find artifacts/faa -name "openairframes_*.csv" -type f 2>/dev/null \ + | grep -vE '/openairframes_(registry|community|adsb)_[0-9]{4}-[0-9]{2}-[0-9]{2}_[0-9]{4}-[0-9]{2}-[0-9]{2}\.csv$' | sort) + echo "Per-source registry CSVs found:" + echo "$SOURCE_CSVS" # Prefer concatenated file (with date range) over single-day file CSV_FILE_ADSB=$(find artifacts/adsb -name "openairframes_adsb_*_*.csv.gz" -type f 2>/dev/null | head -1) if [ -z "$CSV_FILE_ADSB" ]; then @@ -332,6 +389,14 @@ jobs: if [ -z "$JSON_FILE_ADSBX" ] || [ ! -f "$JSON_FILE_ADSBX" ]; then MISSING_FILES="$MISSING_FILES ADSBX_JSON" fi + if [ -z "$CSV_FILE_REGISTRY" ] || [ ! -f "$CSV_FILE_REGISTRY" ]; then + MISSING_FILES="$MISSING_FILES REGISTRY_CSV" + fi + # NOTICE carries the terms that make each asset redistributable. Shipping data + # without it removes the permission, so it is required rather than optional. + if [ ! -f NOTICE ]; then + MISSING_FILES="$MISSING_FILES NOTICE" + fi # Optional files - warn but don't fail OPTIONAL_MISSING="" @@ -369,12 +434,24 @@ jobs: fi if [ -n "$OPTIONAL_MISSING" ]; then - echo "WARNING: Optional files missing:$OPTIONAL_MISSING (will continue without them)" + echo "::warning title=Missing optional release assets::$OPTIONAL_MISSING" fi echo "date=$DATE" >> "$GITHUB_OUTPUT" echo "tag=$TAG" >> "$GITHUB_OUTPUT" echo "csv_file_faa=$CSV_FILE_FAA" >> "$GITHUB_OUTPUT" + echo "csv_file_registry=$CSV_FILE_REGISTRY" >> "$GITHUB_OUTPUT" + { + echo "source_csvs<> "$GITHUB_OUTPUT" + echo "csv_basename_registry=$(basename "$CSV_FILE_REGISTRY")" >> "$GITHUB_OUTPUT" echo "csv_basename_faa=$CSV_BASENAME_FAA" >> "$GITHUB_OUTPUT" echo "csv_file_adsb=$CSV_FILE_ADSB" >> "$GITHUB_OUTPUT" echo "csv_basename_adsb=$CSV_BASENAME_ADSB" >> "$GITHUB_OUTPUT" @@ -399,7 +476,14 @@ jobs: - name: Delete existing release if exists run: | echo "Attempting to delete release: ${{ steps.meta.outputs.tag }}" - gh release delete "${{ steps.meta.outputs.tag }}" --yes --cleanup-tag || echo "No existing release to delete" + # `|| echo` here would swallow a 403 or a partial delete, leaving yesterday's + # asset attached alongside today's; the next run then matches two and rebuilds + # the dataset from a single day. + if gh release view "${{ steps.meta.outputs.tag }}" >/dev/null 2>&1; then + gh release delete "${{ steps.meta.outputs.tag }}" --yes --cleanup-tag + else + echo "No existing release to delete" + fi env: GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} @@ -413,14 +497,18 @@ jobs: Automated daily snapshot generated at 06:00 UTC for ${{ steps.meta.outputs.date }}. Assets: - - ${{ steps.meta.outputs.csv_basename_faa }} + - NOTICE (source terms; required for redistribution) + - ${{ steps.meta.outputs.csv_basename_registry }} + ${{ steps.meta.outputs.source_basenames }} ${{ steps.meta.outputs.csv_basename_adsb && format('- {0}', steps.meta.outputs.csv_basename_adsb) || '' }} - ${{ steps.meta.outputs.csv_basename_community }} - ${{ steps.meta.outputs.zip_basename }} - ${{ steps.meta.outputs.json_basename_adsbx }} ${{ steps.meta.outputs.zip_basename_mictronics && format('- {0}', steps.meta.outputs.zip_basename_mictronics) || '' }} files: | - ${{ steps.meta.outputs.csv_file_faa }} + ${{ steps.meta.outputs.csv_file_registry }} + ${{ steps.meta.outputs.source_csvs }} + NOTICE ${{ steps.meta.outputs.csv_file_adsb }} ${{ steps.meta.outputs.csv_file_community }} ${{ steps.meta.outputs.zip_file }} diff --git a/.github/workflows/registry-source.yaml b/.github/workflows/registry-source.yaml new file mode 100644 index 0000000..d39fe46 --- /dev/null +++ b/.github/workflows/registry-source.yaml @@ -0,0 +1,104 @@ +name: registry-source + +# One registry source, on its own thread. Called once per entry in the caller's matrix, so +# adding a registry is a matrix entry plus src/create_daily__release.py — no new job. + +on: + workflow_call: + inputs: + source: + description: 'Source id; must match src/create_daily__release.py' + required: true + type: string + date: + description: 'Date to process (YYYY-MM-DD, default: today UTC)' + required: false + type: string + python-version: + required: false + type: string + default: '3.14' + required: + description: 'Fail the thread when this source cannot be built' + required: false + type: boolean + default: false + allow_bootstrap: + description: 'Onboarding only: permit a single-day rebuild when no asset has ever been published' + required: false + type: boolean + default: false + +# Keyed on the source: without it every matrix leg shares one group and the legs cancel +# each other, which is the opposite of running them in parallel. +concurrency: + group: ${{ github.workflow }}-${{ github.ref }}-${{ inputs.source }} + cancel-in-progress: false + +jobs: + build: + runs-on: ubuntu-24.04-arm + timeout-minutes: 30 + permissions: + contents: read + steps: + - name: Checkout + uses: actions/checkout@v7 + + - name: Setup Python + uses: actions/setup-python@v7 + with: + python-version: ${{ inputs.python-version }} + cache: 'pip' + cache-dependency-path: requirements.txt + + - name: Install dependencies + run: | + python -m pip install --upgrade pip + pip install -r requirements.txt + + # Deliberately outside the tolerated step below: a typo in the matrix is a config + # error, and must go red even for an optional source. + - name: Check the source has a build script + env: + SOURCE: ${{ inputs.source }} + run: | + script="src/create_daily_${SOURCE}_release.py" + if [ ! -f "$script" ]; then + echo "::error title=Unknown registry source::$script does not exist" + exit 1 + fi + + - name: Build ${{ inputs.source }} registry + continue-on-error: ${{ inputs.required == false }} + env: + GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} + SOURCE: ${{ inputs.source }} + RUN_DATE: ${{ inputs.date }} + BOOTSTRAP: ${{ inputs.allow_bootstrap && '1' || '' }} + run: | + python "src/create_daily_${SOURCE}_release.py" ${RUN_DATE:+--date "$RUN_DATE"} ${BOOTSTRAP:+--allow-bootstrap} + # Stage into one directory so every leg's artifact has the same root; a second + # search path would move the root to the common ancestor for some legs only. + shopt -s nullglob + built=(data/openairframes/openairframes_"${SOURCE}"_*.csv) + if [ ${#built[@]} -ne 1 ]; then + echo "::error title=${SOURCE} produced no registry CSV::expected one openairframes_${SOURCE}_*.csv, found ${#built[@]}" + exit 1 + fi + # Stage into one directory so every leg's artifact has the same root; a second + # search path would move the root to the common ancestor for some legs only. + mkdir -p data/registry-out + cp "${built[@]}" data/registry-out/ + cp data/faa_releasable/ReleasableAircraft_*.zip data/registry-out/ 2>/dev/null || true + ls -lah data/registry-out + + - name: Upload ${{ inputs.source }} registry + uses: actions/upload-artifact@v7 + with: + name: registry-${{ inputs.source }} + path: data/registry-out + retention-days: 1 + # A tolerated source that failed has nothing to upload; only a required + # source missing its artifact is an error. + if-no-files-found: ${{ inputs.required && 'error' || 'ignore' }} diff --git a/AGENTS.md b/AGENTS.md index b523458..1dc9c76 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -1,121 +1,82 @@ -## Never interpolate `${{ github.event.* }}` into a `run:` block +## Never interpolate `${{ github.event.* }}` or `${{ inputs.* }}` into a `run:` block -Actions substitutes `${{ }}` as raw text before the shell parses it, and issue bodies here are -public and unauthenticated. Pass untrusted values through `env:` and quote them. +Pass them via `env:` and quote the shell variable. A quoted heredoc does not help — the body can +contain the delimiter and close it early. Never fix this by escaping or renaming the delimiter. -A quoted heredoc delimiter does not save you: the body can *contain* the delimiter, close the -heredoc early, and execute every line after it. `validate-community-submission.yaml` had that shape. -Never fix this class of bug by escaping, sanitizing, or renaming the delimiter — move it to `env:`. +## Invocation -## Run everything from the repo root - -`src/create_daily_faa_release.py` must be invoked as a **script** (`python src/create_daily_faa_release.py`). -It uses bare sibling imports, so `-m src.create_daily_faa_release` raises `ModuleNotFoundError`. -Everything under `src/adsb/` and `src/contributions/` is the opposite — `python -m`, package-relative. - -Output paths are CWD-relative. +- `src/*.py` at the root of `src/` are scripts: `python src/create_daily_faa_release.py`. Bare + sibling imports, so `-m` raises `ModuleNotFoundError`. +- `src/adsb/*`, `src/contributions/*` are packages: `python -m`. +- Run from the repo root; output paths are CWD-relative. ## Verification -There is no test framework, linter, or packaging config. **Do not add one unprompted**, and do not -treat "nothing broke" as verification. +- No test framework, linter, or packaging config. Do not add one unprompted. +- Never `gh workflow run` to test a change — every dispatch pulls tens of GB. +- Never commit generated data. The product is a GitHub Release; jobs pass state as artifacts. +- ADS-B has no cheap end-to-end check. Exercise `compress_multi_icao_df` on a hand-built frame. -The ADS-B path has no cheap end-to-end check — one day of input is tens of GB. Exercise -`compress_multi_icao_df` / `compress_df_polars` directly against a small hand-built Polars frame. +## Release invariants -**Never `gh workflow run` to test a change.** Every dispatch pulls tens of GB and fans out over date -matrices. Reason about the YAML statically. +- `FINAL_COLUMN_ORDER` (`compress_adsb_to_aircraft_data.py`) is the only definition of the ADS-B + column contract. `pl.concat` matches by position after `.select()`; a forked copy corrupts the + release with no error. +- Empty string, never null, in every released frame. +- `openairframes_id` = `normalize(manufacturer)|normalize(model)|normalize(serial)`. Reuse + `derive_from_faa_master_txt.normalize()`. +- Each source's daily build reads its own previous release asset and appends. Falling back to a + single-day rebuild on anything but `FileNotFoundError` republishes one day as the whole dataset, + which the next run then reads back as its base. Keep the fallback narrow. +- Python `3.14` for FAA/community/vendor jobs, `3.12` for ADS-B. Match the surrounding job. -**Never commit generated data.** The product is a GitHub Release; jobs pass state as artifacts. +## Registry sources -## ADS-B invariants +- Judge **redistribution**, not access. A public licence travels to this project; a bilateral + permission granted to another project does not — that alone disqualifies Taiwan, Estonia, Chile. + Non-commercial-only terms are a separate, independent bar. +- `NOTICE` carries the terms that make each asset redistributable and is a required release file. + Never edit or drop an entry. Transport Canada requires both its notices together. +- `LICENSE` is MIT and covers code only. Claim nothing about released data. +- Owner/registrant mailing addresses are published for every registry. FAA and TC must not diverge. +- CCARCS `ACTIVE_FLAG` is not "current owner": 1,932 Registered marks carry only `I` parties, and + those rows are the `MAIL_RECIPIENT`. Prefer `A`, fall back to all. +- CCARCS addresses come from the single `MAIL_RECIPIENT == "Y"` row, never merged across co-owners. -- `FINAL_COLUMN_ORDER` (`compress_adsb_to_aircraft_data.py`) is the only definition of the released - column contract. `pl.concat` matches by **position** after `.select()`, and - `get_latest_release.get_latest_aircraft_adsb_csv_df` parses released CSVs against the same order — - so a forked copy corrupts the release with no error anywhere. -- `load_parquet_part()` deleting its source parquet is **deliberate**: the raw part is many GB and - the runner would otherwise exhaust disk. Do not defer the delete to make reruns easier; that raises - peak disk by the size of the part. -- A released row means "most informative observation for this ICAO on this UTC day" — non-empty - fields not a subset of another row's, tie-broken by signature frequency. It is not a registry record. -- HTTP 404 is terminal in the release fetch. Restoring the retry makes the Dec-31 next-year-repo - probe stall ~45 minutes on a repo that does not exist yet. +## ADS-B -## Fork and upstream - -`src/get_latest_release.py` pins `REPO = "PlaneQuery/openairframes"` on purpose: this fork reads -**upstream's** releases wherever it runs. Do not repoint it at `github.repository` without being asked. - -Upstream develops on `develop` and PRs into `main`. The daily release **deletes the existing release -and tag** before recreating them. +- `load_parquet_part()` deleting its source parquet is deliberate — disk pressure. Do not defer it. +- A released row is the most informative observation for that ICAO on that UTC day, not a registry + record. +- HTTP 404 is terminal in the release fetch; restoring the retry stalls the Dec-31 probe ~45 min. ## Community submissions are automation-owned -Merging to `community/**` or `schemas/**` force-pushes every open `community`-labeled PR branch back -onto main. Anything you hand-edit on such a branch is destroyed on the next merge. +- Merging `community/**` or `schemas/**` force-pushes every open `community` PR branch onto main. + Hand edits there are destroyed. +- Never hand-author `community/` files — the filename encodes `sha256(content)[:8]`. +- A tag's JSON type is fixed by its first submission and enforced forever. Emergent from + `build_tag_type_registry` + `validate_submission`; written nowhere in the schema. +- Adding `community_submission.v2.schema.json` promotes it atomically across every reader and + writer. One-way door — only on request. -- Never hand-author files in `community/` — the filename encodes `sha256(content)[:8]`, so an edit - orphans the hash and duplicates on re-approval. -- Never invent or copy a `contributor_uuid`; it is derived from the GitHub user id. -- Do not reintroduce a hardcoded `"main"` or `v1` filename. Both are resolved at runtime now. +## Fork -**A tag's JSON type is fixed by its first-ever submission and enforced forever** — emergent from -`build_tag_type_registry` + `validate_submission`, and written nowhere in the schema. Retyping or -renaming an existing tag breaks every future contributor, not just the current one. +`get_latest_release.REPO` pins upstream `PlaneQuery/openairframes` on purpose. Do not repoint it. +Upstream develops on `develop`. The daily release deletes the existing release and tag first. -Dropping a `community_submission.v2.schema.json` into `schemas/` promotes it atomically across every -reader and writer. That is a one-way door for contributors — only on explicit request. +## Do not chase -## Conventions +- `process-historical-faa.yaml` is dead: missing `src/get_historical_faa.py`, + `scripts/concat_csvs.py`, and uses the disabled `::set-output`. +- `af-klm-fleet/package.json` → `npm run validate` has no `scripts/validate.js`. +- `af-klm-fleet/` and `community-routes/` are unwired; nothing in CI touches them. + `af-klm-fleet/README.md` is generated. -- **Empty string, not null**, everywhere in released frames. -- Reuse `derive_from_faa_master_txt.normalize()` for `openairframes_id`; never re-derive the format. -- Python `3.14` for FAA/community/vendor jobs, `3.12` for ADS-B jobs (pyarrow pin + multiprocessing). - Deliberate. Match the surrounding job; do not unify. +## Flag, do not silently fix -## References with no target — do not chase as regressions - -| Reference | Missing | -|---|---| -| `process-historical-faa.yaml` | `src/get_historical_faa.py`, `scripts/concat_csvs.py` | -| `af-klm-fleet/package.json` → `npm run validate` | `af-klm-fleet/scripts/validate.js` | - -`process-historical-faa.yaml` is dead, not stale — it also uses the disabled `::set-output`. -Repair-vs-delete is the owner's call; leave it alone unprompted. - -## `af-klm-fleet/` and `community-routes/` are unwired - -Nothing in CI touches either, and nothing consumes `community-routes/`. `af-klm-fleet/` is a vendored -project by a different author with its own license — its aircraft model is unrelated to -`schemas/community_submission.*`, so do not merge the two. Its `README.md` is generated by -`generate-readme.js`; hand edits are overwritten. - -## Warts left standing — flag, do not silently fix - -- `NUMBER_PARTS` is restated by the matrix and four hand-written upload steps in - `adsb-to-aircraft-for-day.yaml`; changing the constant alone silently drops data. YAML cannot loop - upload steps and one merged artifact would force every map job to download all parts, so any real - fix is a restructure. -- `MAX_WORKERS = OS_CPU_COUNT if OS_CPU_COUNT > 4 else 1` collapses to a single worker on a ≤4-core - runner, shrinking `files_per_batch` with it. Possibly intentional memory control — do not raise it - without measuring peak RSS on the target runner. -- `update-community-prs.yaml` runs `regenerate_pr_schema || true` then force-pushes, so a - regeneration failure ships anyway. Making it fatal leaves PRs un-rebased instead — a judgment call. -- `approve_submission.py` wraps its schema update in a bare `except Exception`, so a submission can - merge without its new tags reaching the schema. - -## Workflow authoring - -The user's global GitHub Actions rules apply. Existing workflows violate most of them. -**Do not bulk-remediate** — bring only the file you were asked to touch up to standard, and surface -the rest in chat. - -## External sources - -`registry.faa.gov` and ADS-B Exchange are **required** — the release fails without them. Mictronics is -**tolerated**: it retries, then the job continues without it. adsb.lol may simply not have published a -given day, in which case the previous CSV is re-released rather than failing. - -FAA refreshes at 05:30 UTC; the release cron fires at 06:00 UTC. That 30-minute margin is the reason -for the schedule. +- `NUMBER_PARTS` is restated by the matrix and four upload steps in `adsb-to-aircraft-for-day.yaml`. +- `MAX_WORKERS = ... if OS_CPU_COUNT > 4 else 1` collapses to one worker on a ≤4-core runner. +- `update-community-prs.yaml` runs `regenerate_pr_schema || true`, then force-pushes. +- `approve_submission.py` wraps its schema update in a bare `except Exception`. +- Existing workflows violate the global GHA rules. Fix only the file you were asked to touch. diff --git a/NOTICE b/NOTICE new file mode 100644 index 0000000..fcfffde --- /dev/null +++ b/NOTICE @@ -0,0 +1,66 @@ +OpenAirframes — Data Source Notices +=================================== + +The LICENSE file covers the *code* in this repository. It does not cover the +*data* published in releases. Several upstream registries permit redistribution +only on condition that specific notices travel with the data. Those conditions +are reproduced below. This file is published as an asset on every release; anyone +redistributing a release asset further must carry the corresponding notice with it. + +Removing or altering a notice in this file removes the permission that makes the +corresponding asset redistributable. + + +Transport Canada — Canadian Civil Aircraft Register (CCARCS) +------------------------------------------------------------ +Asset: openairframes_tc_*.csv +Source: https://wwwapps.tc.gc.ca/saf-sec-sur/2/ccarcs-riacc/download/ccarcsdb.zip + +Redistribution is permitted under the Government of Canada terms recorded for this +dataset. They require both of the following notices, verbatim, and require that the +two reach the consumer together. The governing instrument is not published on the +CCARCS download page, so no licence name or URL is asserted here. + + Reproduced and distributed with the permission of the Government of Canada. + + This product has been produced by or for the OpenAirframes project and + includes data provided by the Government of Canada. The incorporation of + data sourced from the Government of Canada within this product shall not be + construed as constituting an endorsement by the Government of Canada of our + product. + +Registered-owner mailing addresses are redistributed, matching the registrant +addresses the FAA asset already carries. Because CCARCS lists one row per party, +the published address is that of the single designated mail recipient rather than +a merge across co-owners. + + +FAA — Releasable Aircraft Database +----------------------------------- +Source: https://registry.faa.gov/database/ReleasableAircraft.zip + +ReleasableAircraft_*.zip is redistributed unmodified. It is a work of the United +States federal government, not subject to copyright protection in the United +States (17 U.S.C. § 105). No notice is required; it is credited for provenance. + +openairframes_faa_*.csv is a derived product built by this repository — +normalized, joined, deduplicated, and extended with an identifier this project +defines. Section 105 disclaims copyright in the government's own work and says +nothing about a derivative, so no claim is made here about the CSV's status. + + +Other redistributed assets +--------------------------- +The following assets are republished from third parties whose terms have not +been assessed in this repository. They are listed for provenance only; nothing +here asserts a licence over them. + + openairframes_adsb_*.csv.gz derived from adsb.lol daily globe history + (github.com/adsblol), with registration data + from tar1090-db + basic-ac-db_*.json.gz ADS-B Exchange, downloads.adsbexchange.com + mictronics-db_*.zip Mictronics, www.mictronics.de + +Community submissions under community/ are contributed by their authors through +the repository's submission workflow and are published with the attribution each +contributor selected. diff --git a/README.md b/README.md index 8c8eb94..b3b2e30 100644 --- a/README.md +++ b/README.md @@ -26,13 +26,34 @@ df = pd.read_csv(url) df ``` ![](docs/images/df_adsb_example_0.png) -- **openairframes_faa.csv** - All [FAA registration data](https://www.faa.gov/licenses_certificates/aircraft_certification/aircraft_registry/releasable_aircraft_download) from 2023-08-16 to present (~260 MB) +- **openairframes_registry.csv** + Every national registry in one table, one row per registration record, with a `source` + column naming the registry it came from. Currently the FAA (United States) and Transport + Canada. Identifier columns (`transponder_code_hex`, `registration_number`, + `openairframes_id`) lead the table and are populated for every source. +- **openairframes_faa.csv** + All [FAA registration data](https://www.faa.gov/licenses_certificates/aircraft_certification/aircraft_registry/releasable_aircraft_download) from 2023-08-16 to present (~275 MB). + Superseded by `openairframes_registry.csv`; still published so existing consumers keep working. + +- **openairframes_tc.csv** + The [Transport Canada Civil Aircraft Register](https://wwwapps.tc.gc.ca/saf-sec-sur/2/ccarcs-riacc/RchSimp.aspx), + ~35k aircraft with full ICAO 24-bit hex coverage. Also folded into + `openairframes_registry.csv`; published separately for the same reason as the FAA CSV. - **ReleasableAircraft_{date}.zip** A daily snapshot of the FAA database, which updates at **05:30 UTC** +- **basic-ac-db.json.gz** + [ADS-B Exchange](https://www.adsbexchange.com/) basic aircraft database, republished unmodified. + +- **mictronics-db.zip** + [Mictronics](https://www.mictronics.de/aircraft-database/) aircraft database, republished + unmodified. Best effort — the release ships without it when the source is unavailable. + +Redistribution terms for the underlying sources travel with the release in **NOTICE**. Some +registries permit redistribution only on condition that specific notices reach you with the data. + --- ## For Contributors diff --git a/src/build_registry.py b/src/build_registry.py new file mode 100644 index 0000000..f3e582c --- /dev/null +++ b/src/build_registry.py @@ -0,0 +1,128 @@ +"""Join the per-source registry CSVs into one union table. + +Every source publishes its own `openairframes__{start}_{end}.csv` on its own thread. +This reads whatever landed, aligns them on the union of columns, and writes a single +`openairframes_registry_{start}_{end}.csv` discriminated by the `source` column. + +Adding a registry means adding a source to the workflow matrix; nothing here changes. + +Usage: + python src/build_registry.py --input-dir artifacts/registry --date 2026-08-31 +""" +from datetime import datetime, timezone +from pathlib import Path +import argparse +import re +import sys + +import pandas as pd + +# Sources that are registries. Community and ADS-B are published separately: they are +# observations and contributions, not registration records, and do not share this schema. +FILENAME_RE = re.compile( + r"\Aopenairframes_(?P[a-z0-9_]+?)_" + r"(?P\d{4}-\d{2}-\d{2})_(?P\d{4}-\d{2}-\d{2})\.csv\Z" +) +EXCLUDED_SOURCES = {"community", "adsb", "registry"} + +# Identifier columns lead the union so the table is usable without reading 70 headers. +LEADING_COLUMNS = [ + "download_date", + "source", + "transponder_code_hex", + "registration_number", + "openairframes_id", +] + + +def discover(input_dir: Path) -> list[tuple[str, str, str, Path]]: + """Return (source, start, end, path) for each per-source registry CSV found.""" + found = [] + for path in sorted(input_dir.rglob("openairframes_*.csv")): + match = FILENAME_RE.match(path.name) + if not match: + print(f" SKIP {path.name}: does not match {FILENAME_RE.pattern}") + continue + source = match.group("source") + if source in EXCLUDED_SOURCES: + print(f" SKIP {path.name}: {source!r} is published as its own asset") + continue + found.append((source, match.group("start"), match.group("end"), path)) + return found + + +def build(input_dir: Path, date_str: str) -> tuple[pd.DataFrame, str, str]: + parts = discover(input_dir) + if not parts: + raise SystemExit(f"No per-source registry CSVs found under {input_dir}") + + seen = [p[0] for p in parts] + duplicated = {s for s in seen if seen.count(s) > 1} + if duplicated: + raise SystemExit(f"More than one file claims source {sorted(duplicated)}") + + frames = [] + for source, _, _, path in parts: + # keep_default_na=False so a literal "NA" survives the round trip unchanged. + df = pd.read_csv(path, dtype=str, keep_default_na=False) + if "source" not in df.columns: + raise SystemExit(f"{path.name}: no source column; cannot discriminate rows") + if df.empty: + print(f" {source}: empty, skipping") + continue + actual = {v.strip().lower() for v in df["source"].unique()} + if actual != {source.lower()}: + # A file whose rows disagree with its name would duplicate another source into + # the union under the wrong label. + raise SystemExit( + f"{path.name}: filename says {source!r}, rows say {sorted(actual)}" + ) + print(f" {source}: {len(df)} rows, {len(df.columns)} columns from {path.name}") + frames.append(df) + + if not frames: + raise SystemExit("Every discovered source was empty; refusing to publish an empty registry") + + if len(frames) > 1: + shared = set.intersection(*(set(f.columns) for f in frames)) - set(LEADING_COLUMNS) + print(f" columns shared across sources ({len(shared)}): {sorted(shared)}") + + columns = list(dict.fromkeys(c for df in frames for c in df.columns)) + ordered = [c for c in LEADING_COLUMNS if c in columns] + ordered += [c for c in columns if c not in ordered] + + # reindex rather than concat directly: a source missing a column must yield an empty + # cell, never a shifted row. + df_union = pd.concat([df.reindex(columns=ordered) for df in frames], ignore_index=True) + df_union = df_union.fillna("") + + # Earliest start across sources, not per-source coverage: a source added today still + # carries the oldest source's start date in the filename. + start = min(p[1] for p in parts) + end = max(p[2] for p in parts + [("", "", date_str, Path())]) + return df_union, start, end + + +def main() -> None: + parser = argparse.ArgumentParser(description="Join per-source registry CSVs into one table") + parser.add_argument("--input-dir", default="artifacts/registry", help="Directory to search") + parser.add_argument("--output-dir", default="data/openairframes", help="Where to write") + parser.add_argument("--date", help="Run date (YYYY-MM-DD, default: today UTC)") + args = parser.parse_args() + + date_str = args.date or datetime.now(timezone.utc).strftime("%Y-%m-%d") + df, start, end = build(Path(args.input_dir), date_str) + + out_dir = Path(args.output_dir) + out_dir.mkdir(parents=True, exist_ok=True) + out_path = out_dir / f"openairframes_registry_{start}_{end}.csv" + df.to_csv(out_path, index=False) + + print(f"Wrote {out_path}: {len(df)} rows, {len(df.columns)} columns") + print(" rows per source:") + for source, count in df["source"].value_counts().items(): + print(f" {source}: {count}") + + +if __name__ == "__main__": + main() diff --git a/src/create_daily_faa_release.py b/src/create_daily_faa_release.py index 4e7adfd..8e5e49f 100644 --- a/src/create_daily_faa_release.py +++ b/src/create_daily_faa_release.py @@ -4,6 +4,10 @@ import argparse parser = argparse.ArgumentParser(description="Create daily FAA release") parser.add_argument("--date", type=str, help="Date to process (YYYY-MM-DD format, default: today)") +parser.add_argument("--allow-bootstrap", action="store_true", + help="Permit rebuilding from a single day when no published asset is found. " + "Onboarding only: a missing asset is otherwise indistinguishable from a " + "transient outage, and rebuilding would erase the accumulated history.") args = parser.parse_args() if args.date: @@ -37,13 +41,26 @@ from derive_from_faa_master_txt import convert_faa_master_txt_to_df, concat_faa_ from get_latest_release import get_latest_aircraft_faa_csv_df df_new = convert_faa_master_txt_to_df(zip_path, date_str) +# Only a genuine first run may rebuild from a single day. A rate limit, a parse error or a +# non-monotonic download_date must stop the run: this file becomes tomorrow's base, so +# silently republishing one day erases the accumulated history. try: df_base, start_date_str = get_latest_aircraft_faa_csv_df() - df_base = concat_faa_historical_df(df_base, df_new) - assert df_base['download_date'].is_monotonic_increasing, "download_date is not monotonic increasing" -except Exception as e: - print(f"No existing FAA release found, using only new data: {e}") - df_base = df_new +except FileNotFoundError as e: + if not args.allow_bootstrap: + raise SystemExit( + f"No published FAA asset found: {e}\n" + "This is indistinguishable from a transient outage, and rebuilding from one day " + "would erase the accumulated history. Pass --allow-bootstrap when onboarding." + ) from None + print(f"Bootstrapping FAA from today only (--allow-bootstrap): {e}") + df_base = None start_date_str = date_str +if df_base is not None: + df_base = concat_faa_historical_df(df_base, df_new) + assert df_base['download_date'].is_monotonic_increasing, "download_date is not monotonic increasing" +else: + df_base = df_new + df_base.to_csv(OUT_ROOT / f"openairframes_faa_{start_date_str}_{date_str}.csv", index=False) \ No newline at end of file diff --git a/src/create_daily_tc_release.py b/src/create_daily_tc_release.py new file mode 100644 index 0000000..ba7ae3b --- /dev/null +++ b/src/create_daily_tc_release.py @@ -0,0 +1,83 @@ +from pathlib import Path +from datetime import datetime, timezone +import argparse + +parser = argparse.ArgumentParser(description="Create daily Transport Canada release") +parser.add_argument("--date", type=str, help="Date to process (YYYY-MM-DD format, default: today)") +parser.add_argument("--allow-bootstrap", action="store_true", + help="Permit rebuilding from a single day when no published asset is found. " + "Onboarding only: a missing asset is otherwise indistinguishable from a " + "transient outage, and rebuilding would erase the accumulated history.") +args = parser.parse_args() + +if args.date: + date_str = args.date +else: + date_str = datetime.now(timezone.utc).strftime("%Y-%m-%d") + +out_dir = Path("data/tc_ccarcs") +out_dir.mkdir(parents=True, exist_ok=True) +zip_name = f"ccarcsdb_{date_str}.zip" + +zip_path = out_dir / zip_name +if not zip_path.exists(): + url = "https://wwwapps.tc.gc.ca/saf-sec-sur/2/ccarcs-riacc/download/ccarcsdb.zip" + from urllib.request import Request, urlopen + + # CCARCS 403s a default urllib agent. Any browser-like UA works; the exact + # version string is not load-bearing. + req = Request( + url, + headers={ + "User-Agent": ( + "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 " + "(KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36" + ) + }, + method="GET", + ) + + with urlopen(req, timeout=120) as r: + body = r.read() + # TC serves an HTML maintenance page with a 200, which would otherwise be cached + # under a .zip name and re-read on every later run. + if body[:2] != b"PK": + raise RuntimeError(f"{url} did not return a zip (got {body[:40]!r})") + tmp_path = zip_path.with_suffix(".part") + tmp_path.write_bytes(body) + tmp_path.replace(zip_path) + +OUT_ROOT = Path("data/openairframes") +OUT_ROOT.mkdir(parents=True, exist_ok=True) +from derive_from_tc_ccarcs import convert_tc_ccarcs_to_df +# Named for FAA but column-agnostic: fingerprints every column except download_date. +from derive_from_faa_master_txt import concat_faa_historical_df +from get_latest_release import get_latest_aircraft_tc_csv_df +df_new = convert_tc_ccarcs_to_df(zip_path, date_str) + +# Only a genuine first run may rebuild from a single day. Every other failure -- a rate +# limit, a schema change, a truncated download -- must stop the run, because this file +# becomes tomorrow's base and silently republishing one day erases the whole history. +try: + df_base, start_date_str = get_latest_aircraft_tc_csv_df() +except FileNotFoundError as e: + if not args.allow_bootstrap: + raise SystemExit( + f"No published Transport Canada asset found: {e}\n" + "This is indistinguishable from a transient outage, and rebuilding from one day " + "would erase the accumulated history. Pass --allow-bootstrap when onboarding." + ) from None + print(f"Bootstrapping Transport Canada from today only (--allow-bootstrap): {e}") + df_base = None + start_date_str = date_str + +if df_base is not None: + missing = set(df_base.columns) ^ set(df_new.columns) + if missing: + raise SystemExit(f"Column set changed since the last release: {sorted(missing)}") + df_base = concat_faa_historical_df(df_base, df_new) + assert df_base['download_date'].is_monotonic_increasing, "download_date is not monotonic increasing" +else: + df_base = df_new + +df_base.to_csv(OUT_ROOT / f"openairframes_tc_{start_date_str}_{date_str}.csv", index=False) diff --git a/src/derive_from_tc_ccarcs.py b/src/derive_from_tc_ccarcs.py new file mode 100644 index 0000000..5bf3c45 --- /dev/null +++ b/src/derive_from_tc_ccarcs.py @@ -0,0 +1,248 @@ +from pathlib import Path +import csv +import io +import re +import zipfile + +import pandas as pd + +from derive_from_faa_master_txt import normalize + +# CCARCS ships headerless, latin1, comma-delimited exports. Column names come from +# carslayout.txt in the same archive and must stay in file order. +CARSCURR_COLUMNS = [ + "MARK", "REGISTRATION_SUB_TYPE_E", "REGISTRATION_SUB_TYPE_F", "COMMON_NAME", + "MODEL_NAME", "MANUFACTURERS_SERIAL_NUMBER", "MANUFACTURER_SERIAL_COMPRESSED", + "ID_PLATE_MANUFACTURERS_NAME", "BASIS_FOR_REGISTRATION", "BASIS_FOR_REGISTRATION_F", + "AIRCRAFT_CATEGORY_E", "AIRCRAFT_CATEGORY_F", "DATE_OF_IMPORT", "ENGINE_MANUF", + "POWERGLIDER_FLAG", "ENGINE_CATEGORY_E", "ENGINE_CATEGORY_F", "NUMBER_OF_ENGINES", + "NUMBER_OF_SEATS", "AIR_WEIGHT_KILOS", "SALE_REPORTED", "ISSUE_DATE", + "EFFECTIVE_DATE", "INEFFECTIVE_DATE", "REGISTERED_PURPOSE_E", "REGISTERED_PURPOSE_F", + "FLIGHT_AUTHORITY_E", "FLIGHT_AUTHORITY_F", "MANUFACTURE_OR_ASSEMBLY", + "COUNTRY_MANUFACTURE_ASS_E", "COUNTRY_MANUFACTURE_ASS_F", "DATE_MANUFACTURE_ASSEMBLY", + "BASE_OF_OPERATIONS_CTRY_E", "BASE_OF_OPERATIONS_CTRY_F", "BASE_PROVINCE_OR_STATE_E", + "BASE_PROVINCE_OR_STATE_F", "CITY_AIRPORT", "TYPE_CERTIFICATE_NUMBER", + "REGISTRATION_AUTH_STATUS_E", "REGISTRATION_AUTH_STATUS_F", "MULTIPLE_OWNER_FLAG", + "MODIFIED_DATE", "MODE_S_TRANSPONDER_BINARY", "PHYSICAL_FILE_REGION_E", + "PHYSICAL_FILE_REGION_F", "EX_MILITARY_MARK", "TRIMMED_MARK", +] + +CARSOWNR_COLUMNS = [ + "MARK_LINK", "FULL_NAME", "TRADE_NAME", "STREET_NAME", "STREET_NAME2", "CITY", + "PROVINCE_OR_STATE_E", "PROVINCE_OR_STATE_F", "POSTAL_CODE", "COUNTRY_E", "COUNTRY_F", + "TYPE_OF_OWNER_E", "TYPE_OF_OWNER_F", "ACTIVE_FLAG", "CARE_OF", "REGION_E", "REGION_F", + "OWNER_NAME_OLD_FORMAT", "MAIL_RECIPIENT", "TRIMMED_MARK", +] + +# Mailing address of the single designated recipient, matching the registrant_* address +# the FAA build already publishes. Addresses are per-party, so they are taken from the one +# MAIL_RECIPIENT row rather than merged across co-owners. +# registrant_zip_code holds the Canadian postal code: the name is the FAA's, and a union +# table needs one column per concept, not one per country's vocabulary. +OWNER_ADDRESS_COLUMNS = { + "STREET_NAME": "registrant_street_1", + "STREET_NAME2": "registrant_street_2", + "CITY": "registrant_city", + "POSTAL_CODE": "registrant_zip_code", + "CARE_OF": "registrant_care_of", +} + + +FOOTER_RE = re.compile(r"\s*(\d+) rows selected\.\s*") + +# Floor, not an expectation: Canada's register is ~35k aircraft and carsownr is larger +# still, so 1000 only catches a grossly truncated export. The footer row-count check above +# is what actually validates the parse; this guards the case where the footer agrees with a +# near-empty body. +MIN_EXPECTED_ROWS = 1000 + + +def _read_ccarcs_entry(zip_path: Path, entry: str, columns: list[str]) -> pd.DataFrame: + """Read one headerless CCARCS export into a DataFrame. + + Raises: + ValueError: on any row whose width is neither the declared column count nor a + blank/footer line, on a missing or disagreeing "N rows selected." footer, or + on a row count below MIN_EXPECTED_ROWS. + """ + with zipfile.ZipFile(zip_path) as z: + text = z.read(entry).decode("latin1") + + rows = [] + declared = None + # newline="" so a CRLF export does not leave \r on the final field of every row. + for row in csv.reader(io.StringIO(text, newline="")): + if len(row) == len(columns): + rows.append([cell.strip() for cell in row]) + continue + if not row or not any(cell.strip() for cell in row): + continue # trailing blank line + match = FOOTER_RE.fullmatch(row[0]) if len(row) == 1 else None + if match: + declared = int(match.group(1)) + continue + raise ValueError( + f"{entry}: row with {len(row)} fields, expected {len(columns)}: {row[:3]!r}" + ) + + # The spool footer is a free checksum from the source; a short export is otherwise + # indistinguishable from a genuinely smaller register. + if declared is None: + raise ValueError(f"{entry}: no 'N rows selected.' footer; export is truncated") + if declared != len(rows): + raise ValueError(f"{entry}: footer declares {declared} rows, parsed {len(rows)}") + if len(rows) < MIN_EXPECTED_ROWS: + raise ValueError(f"{entry}: only {len(rows)} rows, expected >= {MIN_EXPECTED_ROWS}") + + return pd.DataFrame(rows, columns=columns) + + +def tc_full_registration(mark: str) -> str: + """Expand a trimmed CCARCS mark into the full Canadian registration. + + CCARCS stores the bare mark in both MARK and TRIMMED_MARK, so the prefix has to be + reconstructed: three-character marks are vintage CF- registrations, everything else + takes the modern C- prefix. Returns "" for a blank mark. + """ + mark = (mark or "").strip().upper() + if not mark: + return "" + return f"CF-{mark}" if len(mark) == 3 else f"C-{mark}" + + +def binary_to_hex(binary: str) -> str: + """Convert a 24-bit Mode S binary string to a 6-digit uppercase hex address. + + Returns "" for empty, non-binary, or non-24-bit input. Width is checked because + this column is the join key against ADS-B data: a short field would otherwise + zero-pad into a plausible address belonging to a different aircraft. + """ + binary = (binary or "").strip() + if len(binary) != 24 or any(c not in "01" for c in binary): + return "" + return f"{int(binary, 2):06X}" + + +def _merge_owners(df_ownr: pd.DataFrame) -> pd.DataFrame: + """Collapse the active registered parties for each mark into a single row. + + A co-owned mark repeats with a different party each time; keeping only the mail + recipient would silently drop the rest. + + Each field is deduplicated and blank-skipped independently, so the values are NOT + index-parallel: a mark with three owners can emit three names but one province. + Consumers must not split on ", " and zip the columns together. + """ + # ACTIVE_FLAG is "A"/"I", but "I" does not mean "former owner": 1,932 currently + # Registered marks carry only "I" parties, and those rows are the MAIL_RECIPIENT. + # So prefer active parties where a mark has any, and fall back to all of them + # rather than publishing a registered aircraft with no owner at all. + all_parties = df_ownr + active = df_ownr[df_ownr["ACTIVE_FLAG"].str.upper() == "A"] + marks_with_active = set(active["TRIMMED_MARK"]) + df_ownr = pd.concat([ + active, + df_ownr[~df_ownr["TRIMMED_MARK"].isin(marks_with_active)], + ]) + def join_unique(series: pd.Series) -> str: + seen = [] + for value in series: + value = (value or "").strip() + if value and value not in seen: + seen.append(value) + return ", ".join(seen) + + def count_distinct(series: pd.Series) -> int: + return len({v.strip() for v in series if v and v.strip()}) + + grouped = df_ownr.groupby("TRIMMED_MARK", sort=False).agg( + registrant_name=("FULL_NAME", join_unique), + registrant_state=("PROVINCE_OR_STATE_E", join_unique), + registrant_country=("COUNTRY_E", join_unique), + registrant_type=("TYPE_OF_OWNER_E", join_unique), + registrant_party_count=("FULL_NAME", count_distinct), + ).reset_index() + + # A party row states its own type ("Individual"); that stops being true of the mark + # once several parties share it. Counting distinct names rather than rows keeps this + # consistent with owner_name, which is also deduplicated. + grouped.loc[grouped["registrant_party_count"] > 1, "registrant_type"] = "Co-owner" + + # Taken from the unfiltered frame: the designated recipient is the designated + # recipient even when its own party row is flagged inactive. + recipient = ( + all_parties[all_parties["MAIL_RECIPIENT"].str.upper() == "Y"] + .drop_duplicates(subset="TRIMMED_MARK", keep="first") + .rename(columns=OWNER_ADDRESS_COLUMNS) + ) + return grouped.merge( + recipient[["TRIMMED_MARK", *OWNER_ADDRESS_COLUMNS.values()]], + on="TRIMMED_MARK", + how="left", + ) + + +def convert_tc_ccarcs_to_df(zip_path: Path, date: str) -> pd.DataFrame: + """Build the OpenAirframes Transport Canada frame from a CCARCS zip.""" + df = _read_ccarcs_entry(zip_path, "carscurr.txt", CARSCURR_COLUMNS) + df_ownr = _read_ccarcs_entry(zip_path, "carsownr.txt", CARSOWNR_COLUMNS) + + df = df.merge(_merge_owners(df_ownr), on="TRIMMED_MARK", how="left") + + out = pd.DataFrame({ + "download_date": date, + # The FAA frame already carries `source`; it is the union discriminator. + "source": "TC", + "transponder_code_hex": df["MODE_S_TRANSPONDER_BINARY"].map(binary_to_hex), + "registration_number": df["TRIMMED_MARK"].map(tc_full_registration), + "mark": df["TRIMMED_MARK"], + "aircraft_manufacturer": df["COMMON_NAME"], + "aircraft_model": df["MODEL_NAME"], + "serial_number": df["MANUFACTURERS_SERIAL_NUMBER"], + "aircraft_category": df["AIRCRAFT_CATEGORY_E"], + "engine_manufacturer": df["ENGINE_MANUF"], + "engine_category": df["ENGINE_CATEGORY_E"], + "aircraft_number_of_engines": df["NUMBER_OF_ENGINES"], + "aircraft_number_of_seats": df["NUMBER_OF_SEATS"], + "max_weight_kilos": df["AIR_WEIGHT_KILOS"], + "status": df["REGISTRATION_AUTH_STATUS_E"], + "registration_sub_type": df["REGISTRATION_SUB_TYPE_E"], + "basis_for_registration": df["BASIS_FOR_REGISTRATION"], + "registered_purpose": df["REGISTERED_PURPOSE_E"], + "flight_authority": df["FLIGHT_AUTHORITY_E"], + "type_certificate_number": df["TYPE_CERTIFICATE_NUMBER"], + "country_manufacture": df["COUNTRY_MANUFACTURE_ASS_E"], + "date_manufacture_assembly": df["DATE_MANUFACTURE_ASSEMBLY"], + "base_country": df["BASE_OF_OPERATIONS_CTRY_E"], + "base_province_or_state": df["BASE_PROVINCE_OR_STATE_E"], + "city_airport": df["CITY_AIRPORT"], + "ex_military_mark": df["EX_MILITARY_MARK"], + "multiple_owner_flag": df["MULTIPLE_OWNER_FLAG"], + "registrant_name": df["registrant_name"], + "registrant_type": df["registrant_type"], + "registrant_state": df["registrant_state"], + "registrant_country": df["registrant_country"], + "registrant_care_of": df["registrant_care_of"], + "registrant_street_1": df["registrant_street_1"], + "registrant_street_2": df["registrant_street_2"], + "registrant_city": df["registrant_city"], + "registrant_zip_code": df["registrant_zip_code"], + "issue_date": df["ISSUE_DATE"], + "effective_date": df["EFFECTIVE_DATE"], + "ineffective_date": df["INEFFECTIVE_DATE"], + "modified_date": df["MODIFIED_DATE"], + }) + + # Position matches the FAA frame (after registration_number). Ordering is cosmetic: + # concat_faa_historical_df reindexes df_new to the base's columns before merging. + out.insert(3, "openairframes_id", ( + normalize(out["aircraft_manufacturer"]) + + "|" + + normalize(out["aircraft_model"]) + + "|" + + normalize(out["serial_number"]) + )) + + out = out.fillna("") + out = out.replace("None", "") + return out diff --git a/src/get_latest_release.py b/src/get_latest_release.py index 27a2eca..e2738ff 100644 --- a/src/get_latest_release.py +++ b/src/get_latest_release.py @@ -3,6 +3,7 @@ from __future__ import annotations from dataclasses import dataclass from pathlib import Path from typing import Iterable, Optional +import os import re import urllib.request import urllib.error @@ -138,15 +139,29 @@ def download_latest_aircraft_csv( Path to the downloaded file """ output_dir = Path(output_dir) - assets = get_latest_release_assets(repo, github_token=github_token) - try: - asset = pick_asset(assets, name_regex=r"^openairframes_faa_.*\.csv$") - except FileNotFoundError: - # Fallback to old naming pattern - asset = pick_asset(assets, name_regex=r"^openairframes_\d{4}-\d{2}-\d{2}_.*\.csv$") - saved_to = download_asset(asset, output_dir / asset.name, github_token=github_token) - print(f"Downloaded: {asset.name} ({asset.size} bytes) -> {saved_to}") - return saved_to + github_token = github_token or os.environ.get("GITHUB_TOKEN") + + for release in get_releases(repo, github_token=github_token, per_page=30): + assets = get_release_assets_from_release_data(release) + try: + asset = pick_asset(assets, name_regex=r"^openairframes_faa_.*\.csv$") + except FileNotFoundError: + try: + # Fallback to old naming pattern + asset = pick_asset(assets, name_regex=r"^openairframes_\d{4}-\d{2}-\d{2}_.*\.csv$") + except FileNotFoundError: + continue + saved_to = download_asset(asset, output_dir / asset.name, github_token=github_token) + if asset.size and saved_to.stat().st_size != asset.size: + raise RuntimeError( + f"{asset.name}: downloaded {saved_to.stat().st_size} bytes, expected {asset.size}" + ) + print(f"Downloaded: {asset.name} ({asset.size} bytes) -> {saved_to}") + return saved_to + + raise FileNotFoundError( + "No release in the last 30 releases has an asset matching 'openairframes_faa_.*\\.csv$'" + ) def get_latest_aircraft_faa_csv_df(): csv_path = download_latest_aircraft_csv() @@ -167,6 +182,69 @@ def get_latest_aircraft_faa_csv_df(): return df, date_str +def download_latest_aircraft_tc_csv( + output_dir: Path = Path("downloads"), + github_token: Optional[str] = None, + repo: str = REPO, +) -> Path: + """ + Download the latest openairframes_tc_*.csv file from the latest GitHub release. + + Args: + output_dir: Directory to save the downloaded file (default: "downloads") + github_token: Optional GitHub token for authentication + repo: GitHub repository in format "owner/repo" (default: REPO) + + Returns: + Path to the downloaded file + """ + output_dir = Path(output_dir) + github_token = github_token or os.environ.get("GITHUB_TOKEN") + + # Walk back through releases rather than reading only `latest`. The TC asset is + # optional, so a single failed build publishes a release without it; anchoring on + # `latest` would then make the caller rebuild history from one day and republish + # that as the whole dataset. + for release in get_releases(repo, github_token=github_token, per_page=30): + assets = get_release_assets_from_release_data(release) + try: + asset = pick_asset(assets, name_regex=r"^openairframes_tc_.*\.csv$") + except FileNotFoundError: + continue + saved_to = download_asset(asset, output_dir / asset.name, github_token=github_token) + if asset.size and saved_to.stat().st_size != asset.size: + raise RuntimeError( + f"{asset.name}: downloaded {saved_to.stat().st_size} bytes, expected {asset.size}" + ) + print(f"Downloaded: {asset.name} ({asset.size} bytes) -> {saved_to}") + return saved_to + + raise FileNotFoundError( + "No release in the last 30 releases has an asset matching 'openairframes_tc_.*\\.csv$'" + ) + + +def get_latest_aircraft_tc_csv_df(): + """Return (DataFrame, start_date_str) for the most recent published TC release. + + Raises FileNotFoundError when no recent release carries a TC asset, and ValueError + when the asset filename has no parseable start date. + """ + csv_path = download_latest_aircraft_tc_csv() + import pandas as pd + # keep_default_na=False: a literal "NA"/"N/A" in the source would otherwise read back + # as NaN -> "" while the fresh parse keeps the string, so the row fingerprints would + # never match and every affected record would re-append on every run. + df = pd.read_csv(csv_path, dtype=str, keep_default_na=False) + df = df.fillna("") + # Only the start date is taken; the end date is always the run's own date. + match = re.search(r"openairframes_tc_(\d{4}-\d{2}-\d{2})_", str(csv_path)) + if not match: + raise ValueError(f"Could not extract date from filename: {csv_path.name}") + + return df, match.group(1) + + def download_latest_aircraft_adsb_csv( output_dir: Path = Path("downloads"), github_token: Optional[str] = None,