mirror of
https://github.com/PlaneQuery/OpenAirframes.git
synced 2026-09-15 02:15:25 +02:00
Merge pull request #72 from anchildress1/feat/parallel-registry-build
Add Transport Canada, build registry sources in parallel, and join them into one asset
This commit is contained in:
@@ -10,11 +10,19 @@ on:
|
|||||||
description: 'Date to process (YYYY-MM-DD format, default: yesterday)'
|
description: 'Date to process (YYYY-MM-DD format, default: yesterday)'
|
||||||
required: false
|
required: false
|
||||||
type: string
|
type: string
|
||||||
|
bootstrap_source:
|
||||||
|
description: 'Onboarding only: the one source id permitted to rebuild from a single day'
|
||||||
|
required: false
|
||||||
|
type: string
|
||||||
|
|
||||||
permissions:
|
permissions:
|
||||||
contents: write
|
contents: write
|
||||||
actions: write
|
actions: write
|
||||||
|
|
||||||
|
concurrency:
|
||||||
|
group: ${{ github.workflow }}-${{ github.ref }}
|
||||||
|
cancel-in-progress: false
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
trigger-releases:
|
trigger-releases:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
@@ -42,39 +50,81 @@ jobs:
|
|||||||
ref: 'develop'
|
ref: 'develop'
|
||||||
});
|
});
|
||||||
|
|
||||||
build-faa:
|
# One thread per registry. Adding a source is one matrix entry plus
|
||||||
runs-on: ubuntu-24.04-arm
|
# src/create_daily_<source>_release.py; join-registry picks it up by artifact pattern.
|
||||||
|
build-registry-source:
|
||||||
if: github.event_name != 'schedule'
|
if: github.event_name != 'schedule'
|
||||||
|
strategy:
|
||||||
|
fail-fast: false
|
||||||
|
matrix:
|
||||||
|
include:
|
||||||
|
- source: faa
|
||||||
|
required: true
|
||||||
|
- source: tc
|
||||||
|
required: false
|
||||||
|
uses: ./.github/workflows/registry-source.yaml
|
||||||
|
with:
|
||||||
|
source: ${{ matrix.source }}
|
||||||
|
date: ${{ inputs.date }}
|
||||||
|
required: ${{ matrix.required }}
|
||||||
|
# Standing permission would also fire on an outage. A source bootstraps only when a
|
||||||
|
# human dispatches the run naming it.
|
||||||
|
allow_bootstrap: ${{ inputs.bootstrap_source == matrix.source }}
|
||||||
|
|
||||||
|
join-registry:
|
||||||
|
needs: build-registry-source
|
||||||
|
# No always(): a tolerated source fails its step without failing its leg, so this only
|
||||||
|
# blocks when a required source could not be built.
|
||||||
|
if: github.event_name != 'schedule'
|
||||||
|
runs-on: ubuntu-24.04-arm
|
||||||
|
timeout-minutes: 20
|
||||||
|
permissions:
|
||||||
|
contents: read
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout
|
- name: Checkout
|
||||||
uses: actions/checkout@v6
|
uses: actions/checkout@v7
|
||||||
with:
|
|
||||||
fetch-depth: 0
|
|
||||||
|
|
||||||
- name: Setup Python
|
- name: Setup Python
|
||||||
uses: actions/setup-python@v6
|
uses: actions/setup-python@v7
|
||||||
with:
|
with:
|
||||||
python-version: "3.14"
|
python-version: "3.14"
|
||||||
|
cache: 'pip'
|
||||||
|
cache-dependency-path: requirements.txt
|
||||||
|
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
run: |
|
run: |
|
||||||
python -m pip install --upgrade pip
|
python -m pip install --upgrade pip
|
||||||
pip install -r requirements.txt
|
pip install -r requirements.txt
|
||||||
|
|
||||||
- name: Run FAA release script
|
- name: Download every source thread
|
||||||
run: |
|
uses: actions/download-artifact@v8
|
||||||
python src/create_daily_faa_release.py ${{ inputs.date && format('--date {0}', inputs.date) || '' }}
|
|
||||||
ls -lah data/faa_releasable
|
|
||||||
ls -lah data/openairframes
|
|
||||||
|
|
||||||
- name: Upload FAA artifacts
|
|
||||||
uses: actions/upload-artifact@v4
|
|
||||||
with:
|
with:
|
||||||
name: faa-release
|
pattern: registry-*
|
||||||
path: |
|
path: artifacts/registry
|
||||||
data/openairframes/openairframes_faa_*.csv
|
merge-multiple: true
|
||||||
data/faa_releasable/ReleasableAircraft_*.zip
|
|
||||||
|
- name: Join sources into one registry
|
||||||
|
env:
|
||||||
|
RUN_DATE: ${{ inputs.date }}
|
||||||
|
run: |
|
||||||
|
python src/build_registry.py --input-dir artifacts/registry ${RUN_DATE:+--date "$RUN_DATE"}
|
||||||
|
|
||||||
|
- name: Stage release assets
|
||||||
|
run: |
|
||||||
|
mkdir -p data/release-out
|
||||||
|
# Every per-source CSV must ship: each source reads its OWN previous asset to
|
||||||
|
# accumulate, so one that is never published can never be anything but day one.
|
||||||
|
cp artifacts/registry/* data/release-out/
|
||||||
|
cp data/openairframes/openairframes_registry_*.csv data/release-out/
|
||||||
|
ls -lah data/release-out
|
||||||
|
|
||||||
|
- name: Upload registry
|
||||||
|
uses: actions/upload-artifact@v7
|
||||||
|
with:
|
||||||
|
name: union-registry
|
||||||
|
path: data/release-out
|
||||||
retention-days: 1
|
retention-days: 1
|
||||||
|
if-no-files-found: error
|
||||||
|
|
||||||
resolve-dates:
|
resolve-dates:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
@@ -233,7 +283,7 @@ jobs:
|
|||||||
|
|
||||||
create-release:
|
create-release:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
needs: [resolve-dates, build-faa, adsb-to-aircraft, adsb-reduce, build-community, build-adsbexchange-json, build-mictronics-db]
|
needs: [resolve-dates, join-registry, adsb-to-aircraft, adsb-reduce, build-community, build-adsbexchange-json, build-mictronics-db]
|
||||||
if: github.event_name != 'schedule' && !cancelled()
|
if: github.event_name != 'schedule' && !cancelled()
|
||||||
steps:
|
steps:
|
||||||
- name: Check ADS-B workflow status
|
- name: Check ADS-B workflow status
|
||||||
@@ -246,12 +296,13 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
sparse-checkout: |
|
sparse-checkout: |
|
||||||
.github
|
.github
|
||||||
|
NOTICE
|
||||||
sparse-checkout-cone-mode: false
|
sparse-checkout-cone-mode: false
|
||||||
|
|
||||||
- name: Download FAA artifacts
|
- name: Download joined registry
|
||||||
uses: actions/download-artifact@v5
|
uses: actions/download-artifact@v5
|
||||||
with:
|
with:
|
||||||
name: faa-release
|
name: union-registry
|
||||||
path: artifacts/faa
|
path: artifacts/faa
|
||||||
|
|
||||||
- name: Download ADS-B artifacts
|
- name: Download ADS-B artifacts
|
||||||
@@ -311,6 +362,12 @@ jobs:
|
|||||||
|
|
||||||
# Find files from artifacts using find (handles nested structures)
|
# Find files from artifacts using find (handles nested structures)
|
||||||
CSV_FILE_FAA=$(find artifacts/faa -name "openairframes_faa_*.csv" -type f 2>/dev/null | head -1)
|
CSV_FILE_FAA=$(find artifacts/faa -name "openairframes_faa_*.csv" -type f 2>/dev/null | head -1)
|
||||||
|
CSV_FILE_REGISTRY=$(find artifacts/faa -name "openairframes_registry_*.csv" -type f 2>/dev/null | head -1)
|
||||||
|
# Every per-source registry CSV, whatever sources the matrix ran.
|
||||||
|
SOURCE_CSVS=$(find artifacts/faa -name "openairframes_*.csv" -type f 2>/dev/null \
|
||||||
|
| grep -vE '/openairframes_(registry|community|adsb)_[0-9]{4}-[0-9]{2}-[0-9]{2}_[0-9]{4}-[0-9]{2}-[0-9]{2}\.csv$' | sort)
|
||||||
|
echo "Per-source registry CSVs found:"
|
||||||
|
echo "$SOURCE_CSVS"
|
||||||
# Prefer concatenated file (with date range) over single-day file
|
# Prefer concatenated file (with date range) over single-day file
|
||||||
CSV_FILE_ADSB=$(find artifacts/adsb -name "openairframes_adsb_*_*.csv.gz" -type f 2>/dev/null | head -1)
|
CSV_FILE_ADSB=$(find artifacts/adsb -name "openairframes_adsb_*_*.csv.gz" -type f 2>/dev/null | head -1)
|
||||||
if [ -z "$CSV_FILE_ADSB" ]; then
|
if [ -z "$CSV_FILE_ADSB" ]; then
|
||||||
@@ -332,6 +389,14 @@ jobs:
|
|||||||
if [ -z "$JSON_FILE_ADSBX" ] || [ ! -f "$JSON_FILE_ADSBX" ]; then
|
if [ -z "$JSON_FILE_ADSBX" ] || [ ! -f "$JSON_FILE_ADSBX" ]; then
|
||||||
MISSING_FILES="$MISSING_FILES ADSBX_JSON"
|
MISSING_FILES="$MISSING_FILES ADSBX_JSON"
|
||||||
fi
|
fi
|
||||||
|
if [ -z "$CSV_FILE_REGISTRY" ] || [ ! -f "$CSV_FILE_REGISTRY" ]; then
|
||||||
|
MISSING_FILES="$MISSING_FILES REGISTRY_CSV"
|
||||||
|
fi
|
||||||
|
# NOTICE carries the terms that make each asset redistributable. Shipping data
|
||||||
|
# without it removes the permission, so it is required rather than optional.
|
||||||
|
if [ ! -f NOTICE ]; then
|
||||||
|
MISSING_FILES="$MISSING_FILES NOTICE"
|
||||||
|
fi
|
||||||
|
|
||||||
# Optional files - warn but don't fail
|
# Optional files - warn but don't fail
|
||||||
OPTIONAL_MISSING=""
|
OPTIONAL_MISSING=""
|
||||||
@@ -369,12 +434,24 @@ jobs:
|
|||||||
fi
|
fi
|
||||||
|
|
||||||
if [ -n "$OPTIONAL_MISSING" ]; then
|
if [ -n "$OPTIONAL_MISSING" ]; then
|
||||||
echo "WARNING: Optional files missing:$OPTIONAL_MISSING (will continue without them)"
|
echo "::warning title=Missing optional release assets::$OPTIONAL_MISSING"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
echo "date=$DATE" >> "$GITHUB_OUTPUT"
|
echo "date=$DATE" >> "$GITHUB_OUTPUT"
|
||||||
echo "tag=$TAG" >> "$GITHUB_OUTPUT"
|
echo "tag=$TAG" >> "$GITHUB_OUTPUT"
|
||||||
echo "csv_file_faa=$CSV_FILE_FAA" >> "$GITHUB_OUTPUT"
|
echo "csv_file_faa=$CSV_FILE_FAA" >> "$GITHUB_OUTPUT"
|
||||||
|
echo "csv_file_registry=$CSV_FILE_REGISTRY" >> "$GITHUB_OUTPUT"
|
||||||
|
{
|
||||||
|
echo "source_csvs<<SOURCE_CSVS_EOF"
|
||||||
|
echo "$SOURCE_CSVS"
|
||||||
|
echo "SOURCE_CSVS_EOF"
|
||||||
|
echo "source_basenames<<SOURCE_BASENAMES_EOF"
|
||||||
|
echo "$SOURCE_CSVS" | while read -r f; do
|
||||||
|
[ -n "$f" ] && echo "- $(basename "$f")"
|
||||||
|
done
|
||||||
|
echo "SOURCE_BASENAMES_EOF"
|
||||||
|
} >> "$GITHUB_OUTPUT"
|
||||||
|
echo "csv_basename_registry=$(basename "$CSV_FILE_REGISTRY")" >> "$GITHUB_OUTPUT"
|
||||||
echo "csv_basename_faa=$CSV_BASENAME_FAA" >> "$GITHUB_OUTPUT"
|
echo "csv_basename_faa=$CSV_BASENAME_FAA" >> "$GITHUB_OUTPUT"
|
||||||
echo "csv_file_adsb=$CSV_FILE_ADSB" >> "$GITHUB_OUTPUT"
|
echo "csv_file_adsb=$CSV_FILE_ADSB" >> "$GITHUB_OUTPUT"
|
||||||
echo "csv_basename_adsb=$CSV_BASENAME_ADSB" >> "$GITHUB_OUTPUT"
|
echo "csv_basename_adsb=$CSV_BASENAME_ADSB" >> "$GITHUB_OUTPUT"
|
||||||
@@ -399,7 +476,14 @@ jobs:
|
|||||||
- name: Delete existing release if exists
|
- name: Delete existing release if exists
|
||||||
run: |
|
run: |
|
||||||
echo "Attempting to delete release: ${{ steps.meta.outputs.tag }}"
|
echo "Attempting to delete release: ${{ steps.meta.outputs.tag }}"
|
||||||
gh release delete "${{ steps.meta.outputs.tag }}" --yes --cleanup-tag || echo "No existing release to delete"
|
# `|| echo` here would swallow a 403 or a partial delete, leaving yesterday's
|
||||||
|
# asset attached alongside today's; the next run then matches two and rebuilds
|
||||||
|
# the dataset from a single day.
|
||||||
|
if gh release view "${{ steps.meta.outputs.tag }}" >/dev/null 2>&1; then
|
||||||
|
gh release delete "${{ steps.meta.outputs.tag }}" --yes --cleanup-tag
|
||||||
|
else
|
||||||
|
echo "No existing release to delete"
|
||||||
|
fi
|
||||||
env:
|
env:
|
||||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||||
|
|
||||||
@@ -413,14 +497,18 @@ jobs:
|
|||||||
Automated daily snapshot generated at 06:00 UTC for ${{ steps.meta.outputs.date }}.
|
Automated daily snapshot generated at 06:00 UTC for ${{ steps.meta.outputs.date }}.
|
||||||
|
|
||||||
Assets:
|
Assets:
|
||||||
- ${{ steps.meta.outputs.csv_basename_faa }}
|
- NOTICE (source terms; required for redistribution)
|
||||||
|
- ${{ steps.meta.outputs.csv_basename_registry }}
|
||||||
|
${{ steps.meta.outputs.source_basenames }}
|
||||||
${{ steps.meta.outputs.csv_basename_adsb && format('- {0}', steps.meta.outputs.csv_basename_adsb) || '' }}
|
${{ steps.meta.outputs.csv_basename_adsb && format('- {0}', steps.meta.outputs.csv_basename_adsb) || '' }}
|
||||||
- ${{ steps.meta.outputs.csv_basename_community }}
|
- ${{ steps.meta.outputs.csv_basename_community }}
|
||||||
- ${{ steps.meta.outputs.zip_basename }}
|
- ${{ steps.meta.outputs.zip_basename }}
|
||||||
- ${{ steps.meta.outputs.json_basename_adsbx }}
|
- ${{ steps.meta.outputs.json_basename_adsbx }}
|
||||||
${{ steps.meta.outputs.zip_basename_mictronics && format('- {0}', steps.meta.outputs.zip_basename_mictronics) || '' }}
|
${{ steps.meta.outputs.zip_basename_mictronics && format('- {0}', steps.meta.outputs.zip_basename_mictronics) || '' }}
|
||||||
files: |
|
files: |
|
||||||
${{ steps.meta.outputs.csv_file_faa }}
|
${{ steps.meta.outputs.csv_file_registry }}
|
||||||
|
${{ steps.meta.outputs.source_csvs }}
|
||||||
|
NOTICE
|
||||||
${{ steps.meta.outputs.csv_file_adsb }}
|
${{ steps.meta.outputs.csv_file_adsb }}
|
||||||
${{ steps.meta.outputs.csv_file_community }}
|
${{ steps.meta.outputs.csv_file_community }}
|
||||||
${{ steps.meta.outputs.zip_file }}
|
${{ steps.meta.outputs.zip_file }}
|
||||||
|
|||||||
@@ -0,0 +1,104 @@
|
|||||||
|
name: registry-source
|
||||||
|
|
||||||
|
# One registry source, on its own thread. Called once per entry in the caller's matrix, so
|
||||||
|
# adding a registry is a matrix entry plus src/create_daily_<source>_release.py — no new job.
|
||||||
|
|
||||||
|
on:
|
||||||
|
workflow_call:
|
||||||
|
inputs:
|
||||||
|
source:
|
||||||
|
description: 'Source id; must match src/create_daily_<source>_release.py'
|
||||||
|
required: true
|
||||||
|
type: string
|
||||||
|
date:
|
||||||
|
description: 'Date to process (YYYY-MM-DD, default: today UTC)'
|
||||||
|
required: false
|
||||||
|
type: string
|
||||||
|
python-version:
|
||||||
|
required: false
|
||||||
|
type: string
|
||||||
|
default: '3.14'
|
||||||
|
required:
|
||||||
|
description: 'Fail the thread when this source cannot be built'
|
||||||
|
required: false
|
||||||
|
type: boolean
|
||||||
|
default: false
|
||||||
|
allow_bootstrap:
|
||||||
|
description: 'Onboarding only: permit a single-day rebuild when no asset has ever been published'
|
||||||
|
required: false
|
||||||
|
type: boolean
|
||||||
|
default: false
|
||||||
|
|
||||||
|
# Keyed on the source: without it every matrix leg shares one group and the legs cancel
|
||||||
|
# each other, which is the opposite of running them in parallel.
|
||||||
|
concurrency:
|
||||||
|
group: ${{ github.workflow }}-${{ github.ref }}-${{ inputs.source }}
|
||||||
|
cancel-in-progress: false
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
build:
|
||||||
|
runs-on: ubuntu-24.04-arm
|
||||||
|
timeout-minutes: 30
|
||||||
|
permissions:
|
||||||
|
contents: read
|
||||||
|
steps:
|
||||||
|
- name: Checkout
|
||||||
|
uses: actions/checkout@v7
|
||||||
|
|
||||||
|
- name: Setup Python
|
||||||
|
uses: actions/setup-python@v7
|
||||||
|
with:
|
||||||
|
python-version: ${{ inputs.python-version }}
|
||||||
|
cache: 'pip'
|
||||||
|
cache-dependency-path: requirements.txt
|
||||||
|
|
||||||
|
- name: Install dependencies
|
||||||
|
run: |
|
||||||
|
python -m pip install --upgrade pip
|
||||||
|
pip install -r requirements.txt
|
||||||
|
|
||||||
|
# Deliberately outside the tolerated step below: a typo in the matrix is a config
|
||||||
|
# error, and must go red even for an optional source.
|
||||||
|
- name: Check the source has a build script
|
||||||
|
env:
|
||||||
|
SOURCE: ${{ inputs.source }}
|
||||||
|
run: |
|
||||||
|
script="src/create_daily_${SOURCE}_release.py"
|
||||||
|
if [ ! -f "$script" ]; then
|
||||||
|
echo "::error title=Unknown registry source::$script does not exist"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
- name: Build ${{ inputs.source }} registry
|
||||||
|
continue-on-error: ${{ inputs.required == false }}
|
||||||
|
env:
|
||||||
|
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||||
|
SOURCE: ${{ inputs.source }}
|
||||||
|
RUN_DATE: ${{ inputs.date }}
|
||||||
|
BOOTSTRAP: ${{ inputs.allow_bootstrap && '1' || '' }}
|
||||||
|
run: |
|
||||||
|
python "src/create_daily_${SOURCE}_release.py" ${RUN_DATE:+--date "$RUN_DATE"} ${BOOTSTRAP:+--allow-bootstrap}
|
||||||
|
# Stage into one directory so every leg's artifact has the same root; a second
|
||||||
|
# search path would move the root to the common ancestor for some legs only.
|
||||||
|
shopt -s nullglob
|
||||||
|
built=(data/openairframes/openairframes_"${SOURCE}"_*.csv)
|
||||||
|
if [ ${#built[@]} -ne 1 ]; then
|
||||||
|
echo "::error title=${SOURCE} produced no registry CSV::expected one openairframes_${SOURCE}_*.csv, found ${#built[@]}"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
# Stage into one directory so every leg's artifact has the same root; a second
|
||||||
|
# search path would move the root to the common ancestor for some legs only.
|
||||||
|
mkdir -p data/registry-out
|
||||||
|
cp "${built[@]}" data/registry-out/
|
||||||
|
cp data/faa_releasable/ReleasableAircraft_*.zip data/registry-out/ 2>/dev/null || true
|
||||||
|
ls -lah data/registry-out
|
||||||
|
|
||||||
|
- name: Upload ${{ inputs.source }} registry
|
||||||
|
uses: actions/upload-artifact@v7
|
||||||
|
with:
|
||||||
|
name: registry-${{ inputs.source }}
|
||||||
|
path: data/registry-out
|
||||||
|
retention-days: 1
|
||||||
|
# A tolerated source that failed has nothing to upload; only a required
|
||||||
|
# source missing its artifact is an error.
|
||||||
|
if-no-files-found: ${{ inputs.required && 'error' || 'ignore' }}
|
||||||
@@ -1,121 +1,82 @@
|
|||||||
## Never interpolate `${{ github.event.* }}` into a `run:` block
|
## Never interpolate `${{ github.event.* }}` or `${{ inputs.* }}` into a `run:` block
|
||||||
|
|
||||||
Actions substitutes `${{ }}` as raw text before the shell parses it, and issue bodies here are
|
Pass them via `env:` and quote the shell variable. A quoted heredoc does not help — the body can
|
||||||
public and unauthenticated. Pass untrusted values through `env:` and quote them.
|
contain the delimiter and close it early. Never fix this by escaping or renaming the delimiter.
|
||||||
|
|
||||||
A quoted heredoc delimiter does not save you: the body can *contain* the delimiter, close the
|
## Invocation
|
||||||
heredoc early, and execute every line after it. `validate-community-submission.yaml` had that shape.
|
|
||||||
Never fix this class of bug by escaping, sanitizing, or renaming the delimiter — move it to `env:`.
|
|
||||||
|
|
||||||
## Run everything from the repo root
|
- `src/*.py` at the root of `src/` are scripts: `python src/create_daily_faa_release.py`. Bare
|
||||||
|
sibling imports, so `-m` raises `ModuleNotFoundError`.
|
||||||
`src/create_daily_faa_release.py` must be invoked as a **script** (`python src/create_daily_faa_release.py`).
|
- `src/adsb/*`, `src/contributions/*` are packages: `python -m`.
|
||||||
It uses bare sibling imports, so `-m src.create_daily_faa_release` raises `ModuleNotFoundError`.
|
- Run from the repo root; output paths are CWD-relative.
|
||||||
Everything under `src/adsb/` and `src/contributions/` is the opposite — `python -m`, package-relative.
|
|
||||||
|
|
||||||
Output paths are CWD-relative.
|
|
||||||
|
|
||||||
## Verification
|
## Verification
|
||||||
|
|
||||||
There is no test framework, linter, or packaging config. **Do not add one unprompted**, and do not
|
- No test framework, linter, or packaging config. Do not add one unprompted.
|
||||||
treat "nothing broke" as verification.
|
- Never `gh workflow run` to test a change — every dispatch pulls tens of GB.
|
||||||
|
- Never commit generated data. The product is a GitHub Release; jobs pass state as artifacts.
|
||||||
|
- ADS-B has no cheap end-to-end check. Exercise `compress_multi_icao_df` on a hand-built frame.
|
||||||
|
|
||||||
The ADS-B path has no cheap end-to-end check — one day of input is tens of GB. Exercise
|
## Release invariants
|
||||||
`compress_multi_icao_df` / `compress_df_polars` directly against a small hand-built Polars frame.
|
|
||||||
|
|
||||||
**Never `gh workflow run` to test a change.** Every dispatch pulls tens of GB and fans out over date
|
- `FINAL_COLUMN_ORDER` (`compress_adsb_to_aircraft_data.py`) is the only definition of the ADS-B
|
||||||
matrices. Reason about the YAML statically.
|
column contract. `pl.concat` matches by position after `.select()`; a forked copy corrupts the
|
||||||
|
release with no error.
|
||||||
|
- Empty string, never null, in every released frame.
|
||||||
|
- `openairframes_id` = `normalize(manufacturer)|normalize(model)|normalize(serial)`. Reuse
|
||||||
|
`derive_from_faa_master_txt.normalize()`.
|
||||||
|
- Each source's daily build reads its own previous release asset and appends. Falling back to a
|
||||||
|
single-day rebuild on anything but `FileNotFoundError` republishes one day as the whole dataset,
|
||||||
|
which the next run then reads back as its base. Keep the fallback narrow.
|
||||||
|
- Python `3.14` for FAA/community/vendor jobs, `3.12` for ADS-B. Match the surrounding job.
|
||||||
|
|
||||||
**Never commit generated data.** The product is a GitHub Release; jobs pass state as artifacts.
|
## Registry sources
|
||||||
|
|
||||||
## ADS-B invariants
|
- Judge **redistribution**, not access. A public licence travels to this project; a bilateral
|
||||||
|
permission granted to another project does not — that alone disqualifies Taiwan, Estonia, Chile.
|
||||||
|
Non-commercial-only terms are a separate, independent bar.
|
||||||
|
- `NOTICE` carries the terms that make each asset redistributable and is a required release file.
|
||||||
|
Never edit or drop an entry. Transport Canada requires both its notices together.
|
||||||
|
- `LICENSE` is MIT and covers code only. Claim nothing about released data.
|
||||||
|
- Owner/registrant mailing addresses are published for every registry. FAA and TC must not diverge.
|
||||||
|
- CCARCS `ACTIVE_FLAG` is not "current owner": 1,932 Registered marks carry only `I` parties, and
|
||||||
|
those rows are the `MAIL_RECIPIENT`. Prefer `A`, fall back to all.
|
||||||
|
- CCARCS addresses come from the single `MAIL_RECIPIENT == "Y"` row, never merged across co-owners.
|
||||||
|
|
||||||
- `FINAL_COLUMN_ORDER` (`compress_adsb_to_aircraft_data.py`) is the only definition of the released
|
## ADS-B
|
||||||
column contract. `pl.concat` matches by **position** after `.select()`, and
|
|
||||||
`get_latest_release.get_latest_aircraft_adsb_csv_df` parses released CSVs against the same order —
|
|
||||||
so a forked copy corrupts the release with no error anywhere.
|
|
||||||
- `load_parquet_part()` deleting its source parquet is **deliberate**: the raw part is many GB and
|
|
||||||
the runner would otherwise exhaust disk. Do not defer the delete to make reruns easier; that raises
|
|
||||||
peak disk by the size of the part.
|
|
||||||
- A released row means "most informative observation for this ICAO on this UTC day" — non-empty
|
|
||||||
fields not a subset of another row's, tie-broken by signature frequency. It is not a registry record.
|
|
||||||
- HTTP 404 is terminal in the release fetch. Restoring the retry makes the Dec-31 next-year-repo
|
|
||||||
probe stall ~45 minutes on a repo that does not exist yet.
|
|
||||||
|
|
||||||
## Fork and upstream
|
- `load_parquet_part()` deleting its source parquet is deliberate — disk pressure. Do not defer it.
|
||||||
|
- A released row is the most informative observation for that ICAO on that UTC day, not a registry
|
||||||
`src/get_latest_release.py` pins `REPO = "PlaneQuery/openairframes"` on purpose: this fork reads
|
record.
|
||||||
**upstream's** releases wherever it runs. Do not repoint it at `github.repository` without being asked.
|
- HTTP 404 is terminal in the release fetch; restoring the retry stalls the Dec-31 probe ~45 min.
|
||||||
|
|
||||||
Upstream develops on `develop` and PRs into `main`. The daily release **deletes the existing release
|
|
||||||
and tag** before recreating them.
|
|
||||||
|
|
||||||
## Community submissions are automation-owned
|
## Community submissions are automation-owned
|
||||||
|
|
||||||
Merging to `community/**` or `schemas/**` force-pushes every open `community`-labeled PR branch back
|
- Merging `community/**` or `schemas/**` force-pushes every open `community` PR branch onto main.
|
||||||
onto main. Anything you hand-edit on such a branch is destroyed on the next merge.
|
Hand edits there are destroyed.
|
||||||
|
- Never hand-author `community/` files — the filename encodes `sha256(content)[:8]`.
|
||||||
|
- A tag's JSON type is fixed by its first submission and enforced forever. Emergent from
|
||||||
|
`build_tag_type_registry` + `validate_submission`; written nowhere in the schema.
|
||||||
|
- Adding `community_submission.v2.schema.json` promotes it atomically across every reader and
|
||||||
|
writer. One-way door — only on request.
|
||||||
|
|
||||||
- Never hand-author files in `community/` — the filename encodes `sha256(content)[:8]`, so an edit
|
## Fork
|
||||||
orphans the hash and duplicates on re-approval.
|
|
||||||
- Never invent or copy a `contributor_uuid`; it is derived from the GitHub user id.
|
|
||||||
- Do not reintroduce a hardcoded `"main"` or `v1` filename. Both are resolved at runtime now.
|
|
||||||
|
|
||||||
**A tag's JSON type is fixed by its first-ever submission and enforced forever** — emergent from
|
`get_latest_release.REPO` pins upstream `PlaneQuery/openairframes` on purpose. Do not repoint it.
|
||||||
`build_tag_type_registry` + `validate_submission`, and written nowhere in the schema. Retyping or
|
Upstream develops on `develop`. The daily release deletes the existing release and tag first.
|
||||||
renaming an existing tag breaks every future contributor, not just the current one.
|
|
||||||
|
|
||||||
Dropping a `community_submission.v2.schema.json` into `schemas/` promotes it atomically across every
|
## Do not chase
|
||||||
reader and writer. That is a one-way door for contributors — only on explicit request.
|
|
||||||
|
|
||||||
## Conventions
|
- `process-historical-faa.yaml` is dead: missing `src/get_historical_faa.py`,
|
||||||
|
`scripts/concat_csvs.py`, and uses the disabled `::set-output`.
|
||||||
|
- `af-klm-fleet/package.json` → `npm run validate` has no `scripts/validate.js`.
|
||||||
|
- `af-klm-fleet/` and `community-routes/` are unwired; nothing in CI touches them.
|
||||||
|
`af-klm-fleet/README.md` is generated.
|
||||||
|
|
||||||
- **Empty string, not null**, everywhere in released frames.
|
## Flag, do not silently fix
|
||||||
- Reuse `derive_from_faa_master_txt.normalize()` for `openairframes_id`; never re-derive the format.
|
|
||||||
- Python `3.14` for FAA/community/vendor jobs, `3.12` for ADS-B jobs (pyarrow pin + multiprocessing).
|
|
||||||
Deliberate. Match the surrounding job; do not unify.
|
|
||||||
|
|
||||||
## References with no target — do not chase as regressions
|
- `NUMBER_PARTS` is restated by the matrix and four upload steps in `adsb-to-aircraft-for-day.yaml`.
|
||||||
|
- `MAX_WORKERS = ... if OS_CPU_COUNT > 4 else 1` collapses to one worker on a ≤4-core runner.
|
||||||
| Reference | Missing |
|
- `update-community-prs.yaml` runs `regenerate_pr_schema || true`, then force-pushes.
|
||||||
|---|---|
|
- `approve_submission.py` wraps its schema update in a bare `except Exception`.
|
||||||
| `process-historical-faa.yaml` | `src/get_historical_faa.py`, `scripts/concat_csvs.py` |
|
- Existing workflows violate the global GHA rules. Fix only the file you were asked to touch.
|
||||||
| `af-klm-fleet/package.json` → `npm run validate` | `af-klm-fleet/scripts/validate.js` |
|
|
||||||
|
|
||||||
`process-historical-faa.yaml` is dead, not stale — it also uses the disabled `::set-output`.
|
|
||||||
Repair-vs-delete is the owner's call; leave it alone unprompted.
|
|
||||||
|
|
||||||
## `af-klm-fleet/` and `community-routes/` are unwired
|
|
||||||
|
|
||||||
Nothing in CI touches either, and nothing consumes `community-routes/`. `af-klm-fleet/` is a vendored
|
|
||||||
project by a different author with its own license — its aircraft model is unrelated to
|
|
||||||
`schemas/community_submission.*`, so do not merge the two. Its `README.md` is generated by
|
|
||||||
`generate-readme.js`; hand edits are overwritten.
|
|
||||||
|
|
||||||
## Warts left standing — flag, do not silently fix
|
|
||||||
|
|
||||||
- `NUMBER_PARTS` is restated by the matrix and four hand-written upload steps in
|
|
||||||
`adsb-to-aircraft-for-day.yaml`; changing the constant alone silently drops data. YAML cannot loop
|
|
||||||
upload steps and one merged artifact would force every map job to download all parts, so any real
|
|
||||||
fix is a restructure.
|
|
||||||
- `MAX_WORKERS = OS_CPU_COUNT if OS_CPU_COUNT > 4 else 1` collapses to a single worker on a ≤4-core
|
|
||||||
runner, shrinking `files_per_batch` with it. Possibly intentional memory control — do not raise it
|
|
||||||
without measuring peak RSS on the target runner.
|
|
||||||
- `update-community-prs.yaml` runs `regenerate_pr_schema || true` then force-pushes, so a
|
|
||||||
regeneration failure ships anyway. Making it fatal leaves PRs un-rebased instead — a judgment call.
|
|
||||||
- `approve_submission.py` wraps its schema update in a bare `except Exception`, so a submission can
|
|
||||||
merge without its new tags reaching the schema.
|
|
||||||
|
|
||||||
## Workflow authoring
|
|
||||||
|
|
||||||
The user's global GitHub Actions rules apply. Existing workflows violate most of them.
|
|
||||||
**Do not bulk-remediate** — bring only the file you were asked to touch up to standard, and surface
|
|
||||||
the rest in chat.
|
|
||||||
|
|
||||||
## External sources
|
|
||||||
|
|
||||||
`registry.faa.gov` and ADS-B Exchange are **required** — the release fails without them. Mictronics is
|
|
||||||
**tolerated**: it retries, then the job continues without it. adsb.lol may simply not have published a
|
|
||||||
given day, in which case the previous CSV is re-released rather than failing.
|
|
||||||
|
|
||||||
FAA refreshes at 05:30 UTC; the release cron fires at 06:00 UTC. That 30-minute margin is the reason
|
|
||||||
for the schedule.
|
|
||||||
|
|||||||
@@ -0,0 +1,66 @@
|
|||||||
|
OpenAirframes — Data Source Notices
|
||||||
|
===================================
|
||||||
|
|
||||||
|
The LICENSE file covers the *code* in this repository. It does not cover the
|
||||||
|
*data* published in releases. Several upstream registries permit redistribution
|
||||||
|
only on condition that specific notices travel with the data. Those conditions
|
||||||
|
are reproduced below. This file is published as an asset on every release; anyone
|
||||||
|
redistributing a release asset further must carry the corresponding notice with it.
|
||||||
|
|
||||||
|
Removing or altering a notice in this file removes the permission that makes the
|
||||||
|
corresponding asset redistributable.
|
||||||
|
|
||||||
|
|
||||||
|
Transport Canada — Canadian Civil Aircraft Register (CCARCS)
|
||||||
|
------------------------------------------------------------
|
||||||
|
Asset: openairframes_tc_*.csv
|
||||||
|
Source: https://wwwapps.tc.gc.ca/saf-sec-sur/2/ccarcs-riacc/download/ccarcsdb.zip
|
||||||
|
|
||||||
|
Redistribution is permitted under the Government of Canada terms recorded for this
|
||||||
|
dataset. They require both of the following notices, verbatim, and require that the
|
||||||
|
two reach the consumer together. The governing instrument is not published on the
|
||||||
|
CCARCS download page, so no licence name or URL is asserted here.
|
||||||
|
|
||||||
|
Reproduced and distributed with the permission of the Government of Canada.
|
||||||
|
|
||||||
|
This product has been produced by or for the OpenAirframes project and
|
||||||
|
includes data provided by the Government of Canada. The incorporation of
|
||||||
|
data sourced from the Government of Canada within this product shall not be
|
||||||
|
construed as constituting an endorsement by the Government of Canada of our
|
||||||
|
product.
|
||||||
|
|
||||||
|
Registered-owner mailing addresses are redistributed, matching the registrant
|
||||||
|
addresses the FAA asset already carries. Because CCARCS lists one row per party,
|
||||||
|
the published address is that of the single designated mail recipient rather than
|
||||||
|
a merge across co-owners.
|
||||||
|
|
||||||
|
|
||||||
|
FAA — Releasable Aircraft Database
|
||||||
|
-----------------------------------
|
||||||
|
Source: https://registry.faa.gov/database/ReleasableAircraft.zip
|
||||||
|
|
||||||
|
ReleasableAircraft_*.zip is redistributed unmodified. It is a work of the United
|
||||||
|
States federal government, not subject to copyright protection in the United
|
||||||
|
States (17 U.S.C. § 105). No notice is required; it is credited for provenance.
|
||||||
|
|
||||||
|
openairframes_faa_*.csv is a derived product built by this repository —
|
||||||
|
normalized, joined, deduplicated, and extended with an identifier this project
|
||||||
|
defines. Section 105 disclaims copyright in the government's own work and says
|
||||||
|
nothing about a derivative, so no claim is made here about the CSV's status.
|
||||||
|
|
||||||
|
|
||||||
|
Other redistributed assets
|
||||||
|
---------------------------
|
||||||
|
The following assets are republished from third parties whose terms have not
|
||||||
|
been assessed in this repository. They are listed for provenance only; nothing
|
||||||
|
here asserts a licence over them.
|
||||||
|
|
||||||
|
openairframes_adsb_*.csv.gz derived from adsb.lol daily globe history
|
||||||
|
(github.com/adsblol), with registration data
|
||||||
|
from tar1090-db
|
||||||
|
basic-ac-db_*.json.gz ADS-B Exchange, downloads.adsbexchange.com
|
||||||
|
mictronics-db_*.zip Mictronics, www.mictronics.de
|
||||||
|
|
||||||
|
Community submissions under community/ are contributed by their authors through
|
||||||
|
the repository's submission workflow and are published with the attribution each
|
||||||
|
contributor selected.
|
||||||
@@ -26,13 +26,34 @@ df = pd.read_csv(url)
|
|||||||
df
|
df
|
||||||
```
|
```
|
||||||

|

|
||||||
- **openairframes_faa.csv**
|
- **openairframes_registry.csv**
|
||||||
All [FAA registration data](https://www.faa.gov/licenses_certificates/aircraft_certification/aircraft_registry/releasable_aircraft_download) from 2023-08-16 to present (~260 MB)
|
Every national registry in one table, one row per registration record, with a `source`
|
||||||
|
column naming the registry it came from. Currently the FAA (United States) and Transport
|
||||||
|
Canada. Identifier columns (`transponder_code_hex`, `registration_number`,
|
||||||
|
`openairframes_id`) lead the table and are populated for every source.
|
||||||
|
|
||||||
|
- **openairframes_faa.csv**
|
||||||
|
All [FAA registration data](https://www.faa.gov/licenses_certificates/aircraft_certification/aircraft_registry/releasable_aircraft_download) from 2023-08-16 to present (~275 MB).
|
||||||
|
Superseded by `openairframes_registry.csv`; still published so existing consumers keep working.
|
||||||
|
|
||||||
|
- **openairframes_tc.csv**
|
||||||
|
The [Transport Canada Civil Aircraft Register](https://wwwapps.tc.gc.ca/saf-sec-sur/2/ccarcs-riacc/RchSimp.aspx),
|
||||||
|
~35k aircraft with full ICAO 24-bit hex coverage. Also folded into
|
||||||
|
`openairframes_registry.csv`; published separately for the same reason as the FAA CSV.
|
||||||
|
|
||||||
- **ReleasableAircraft_{date}.zip**
|
- **ReleasableAircraft_{date}.zip**
|
||||||
A daily snapshot of the FAA database, which updates at **05:30 UTC**
|
A daily snapshot of the FAA database, which updates at **05:30 UTC**
|
||||||
|
|
||||||
|
- **basic-ac-db.json.gz**
|
||||||
|
[ADS-B Exchange](https://www.adsbexchange.com/) basic aircraft database, republished unmodified.
|
||||||
|
|
||||||
|
- **mictronics-db.zip**
|
||||||
|
[Mictronics](https://www.mictronics.de/aircraft-database/) aircraft database, republished
|
||||||
|
unmodified. Best effort — the release ships without it when the source is unavailable.
|
||||||
|
|
||||||
|
Redistribution terms for the underlying sources travel with the release in **NOTICE**. Some
|
||||||
|
registries permit redistribution only on condition that specific notices reach you with the data.
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
## For Contributors
|
## For Contributors
|
||||||
|
|||||||
@@ -0,0 +1,128 @@
|
|||||||
|
"""Join the per-source registry CSVs into one union table.
|
||||||
|
|
||||||
|
Every source publishes its own `openairframes_<source>_{start}_{end}.csv` on its own thread.
|
||||||
|
This reads whatever landed, aligns them on the union of columns, and writes a single
|
||||||
|
`openairframes_registry_{start}_{end}.csv` discriminated by the `source` column.
|
||||||
|
|
||||||
|
Adding a registry means adding a source to the workflow matrix; nothing here changes.
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
python src/build_registry.py --input-dir artifacts/registry --date 2026-08-31
|
||||||
|
"""
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
from pathlib import Path
|
||||||
|
import argparse
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
|
||||||
|
import pandas as pd
|
||||||
|
|
||||||
|
# Sources that are registries. Community and ADS-B are published separately: they are
|
||||||
|
# observations and contributions, not registration records, and do not share this schema.
|
||||||
|
FILENAME_RE = re.compile(
|
||||||
|
r"\Aopenairframes_(?P<source>[a-z0-9_]+?)_"
|
||||||
|
r"(?P<start>\d{4}-\d{2}-\d{2})_(?P<end>\d{4}-\d{2}-\d{2})\.csv\Z"
|
||||||
|
)
|
||||||
|
EXCLUDED_SOURCES = {"community", "adsb", "registry"}
|
||||||
|
|
||||||
|
# Identifier columns lead the union so the table is usable without reading 70 headers.
|
||||||
|
LEADING_COLUMNS = [
|
||||||
|
"download_date",
|
||||||
|
"source",
|
||||||
|
"transponder_code_hex",
|
||||||
|
"registration_number",
|
||||||
|
"openairframes_id",
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def discover(input_dir: Path) -> list[tuple[str, str, str, Path]]:
|
||||||
|
"""Return (source, start, end, path) for each per-source registry CSV found."""
|
||||||
|
found = []
|
||||||
|
for path in sorted(input_dir.rglob("openairframes_*.csv")):
|
||||||
|
match = FILENAME_RE.match(path.name)
|
||||||
|
if not match:
|
||||||
|
print(f" SKIP {path.name}: does not match {FILENAME_RE.pattern}")
|
||||||
|
continue
|
||||||
|
source = match.group("source")
|
||||||
|
if source in EXCLUDED_SOURCES:
|
||||||
|
print(f" SKIP {path.name}: {source!r} is published as its own asset")
|
||||||
|
continue
|
||||||
|
found.append((source, match.group("start"), match.group("end"), path))
|
||||||
|
return found
|
||||||
|
|
||||||
|
|
||||||
|
def build(input_dir: Path, date_str: str) -> tuple[pd.DataFrame, str, str]:
|
||||||
|
parts = discover(input_dir)
|
||||||
|
if not parts:
|
||||||
|
raise SystemExit(f"No per-source registry CSVs found under {input_dir}")
|
||||||
|
|
||||||
|
seen = [p[0] for p in parts]
|
||||||
|
duplicated = {s for s in seen if seen.count(s) > 1}
|
||||||
|
if duplicated:
|
||||||
|
raise SystemExit(f"More than one file claims source {sorted(duplicated)}")
|
||||||
|
|
||||||
|
frames = []
|
||||||
|
for source, _, _, path in parts:
|
||||||
|
# keep_default_na=False so a literal "NA" survives the round trip unchanged.
|
||||||
|
df = pd.read_csv(path, dtype=str, keep_default_na=False)
|
||||||
|
if "source" not in df.columns:
|
||||||
|
raise SystemExit(f"{path.name}: no source column; cannot discriminate rows")
|
||||||
|
if df.empty:
|
||||||
|
print(f" {source}: empty, skipping")
|
||||||
|
continue
|
||||||
|
actual = {v.strip().lower() for v in df["source"].unique()}
|
||||||
|
if actual != {source.lower()}:
|
||||||
|
# A file whose rows disagree with its name would duplicate another source into
|
||||||
|
# the union under the wrong label.
|
||||||
|
raise SystemExit(
|
||||||
|
f"{path.name}: filename says {source!r}, rows say {sorted(actual)}"
|
||||||
|
)
|
||||||
|
print(f" {source}: {len(df)} rows, {len(df.columns)} columns from {path.name}")
|
||||||
|
frames.append(df)
|
||||||
|
|
||||||
|
if not frames:
|
||||||
|
raise SystemExit("Every discovered source was empty; refusing to publish an empty registry")
|
||||||
|
|
||||||
|
if len(frames) > 1:
|
||||||
|
shared = set.intersection(*(set(f.columns) for f in frames)) - set(LEADING_COLUMNS)
|
||||||
|
print(f" columns shared across sources ({len(shared)}): {sorted(shared)}")
|
||||||
|
|
||||||
|
columns = list(dict.fromkeys(c for df in frames for c in df.columns))
|
||||||
|
ordered = [c for c in LEADING_COLUMNS if c in columns]
|
||||||
|
ordered += [c for c in columns if c not in ordered]
|
||||||
|
|
||||||
|
# reindex rather than concat directly: a source missing a column must yield an empty
|
||||||
|
# cell, never a shifted row.
|
||||||
|
df_union = pd.concat([df.reindex(columns=ordered) for df in frames], ignore_index=True)
|
||||||
|
df_union = df_union.fillna("")
|
||||||
|
|
||||||
|
# Earliest start across sources, not per-source coverage: a source added today still
|
||||||
|
# carries the oldest source's start date in the filename.
|
||||||
|
start = min(p[1] for p in parts)
|
||||||
|
end = max(p[2] for p in parts + [("", "", date_str, Path())])
|
||||||
|
return df_union, start, end
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
parser = argparse.ArgumentParser(description="Join per-source registry CSVs into one table")
|
||||||
|
parser.add_argument("--input-dir", default="artifacts/registry", help="Directory to search")
|
||||||
|
parser.add_argument("--output-dir", default="data/openairframes", help="Where to write")
|
||||||
|
parser.add_argument("--date", help="Run date (YYYY-MM-DD, default: today UTC)")
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
date_str = args.date or datetime.now(timezone.utc).strftime("%Y-%m-%d")
|
||||||
|
df, start, end = build(Path(args.input_dir), date_str)
|
||||||
|
|
||||||
|
out_dir = Path(args.output_dir)
|
||||||
|
out_dir.mkdir(parents=True, exist_ok=True)
|
||||||
|
out_path = out_dir / f"openairframes_registry_{start}_{end}.csv"
|
||||||
|
df.to_csv(out_path, index=False)
|
||||||
|
|
||||||
|
print(f"Wrote {out_path}: {len(df)} rows, {len(df.columns)} columns")
|
||||||
|
print(" rows per source:")
|
||||||
|
for source, count in df["source"].value_counts().items():
|
||||||
|
print(f" {source}: {count}")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -4,6 +4,10 @@ import argparse
|
|||||||
|
|
||||||
parser = argparse.ArgumentParser(description="Create daily FAA release")
|
parser = argparse.ArgumentParser(description="Create daily FAA release")
|
||||||
parser.add_argument("--date", type=str, help="Date to process (YYYY-MM-DD format, default: today)")
|
parser.add_argument("--date", type=str, help="Date to process (YYYY-MM-DD format, default: today)")
|
||||||
|
parser.add_argument("--allow-bootstrap", action="store_true",
|
||||||
|
help="Permit rebuilding from a single day when no published asset is found. "
|
||||||
|
"Onboarding only: a missing asset is otherwise indistinguishable from a "
|
||||||
|
"transient outage, and rebuilding would erase the accumulated history.")
|
||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
|
|
||||||
if args.date:
|
if args.date:
|
||||||
@@ -37,13 +41,26 @@ from derive_from_faa_master_txt import convert_faa_master_txt_to_df, concat_faa_
|
|||||||
from get_latest_release import get_latest_aircraft_faa_csv_df
|
from get_latest_release import get_latest_aircraft_faa_csv_df
|
||||||
df_new = convert_faa_master_txt_to_df(zip_path, date_str)
|
df_new = convert_faa_master_txt_to_df(zip_path, date_str)
|
||||||
|
|
||||||
|
# Only a genuine first run may rebuild from a single day. A rate limit, a parse error or a
|
||||||
|
# non-monotonic download_date must stop the run: this file becomes tomorrow's base, so
|
||||||
|
# silently republishing one day erases the accumulated history.
|
||||||
try:
|
try:
|
||||||
df_base, start_date_str = get_latest_aircraft_faa_csv_df()
|
df_base, start_date_str = get_latest_aircraft_faa_csv_df()
|
||||||
df_base = concat_faa_historical_df(df_base, df_new)
|
except FileNotFoundError as e:
|
||||||
assert df_base['download_date'].is_monotonic_increasing, "download_date is not monotonic increasing"
|
if not args.allow_bootstrap:
|
||||||
except Exception as e:
|
raise SystemExit(
|
||||||
print(f"No existing FAA release found, using only new data: {e}")
|
f"No published FAA asset found: {e}\n"
|
||||||
df_base = df_new
|
"This is indistinguishable from a transient outage, and rebuilding from one day "
|
||||||
|
"would erase the accumulated history. Pass --allow-bootstrap when onboarding."
|
||||||
|
) from None
|
||||||
|
print(f"Bootstrapping FAA from today only (--allow-bootstrap): {e}")
|
||||||
|
df_base = None
|
||||||
start_date_str = date_str
|
start_date_str = date_str
|
||||||
|
|
||||||
|
if df_base is not None:
|
||||||
|
df_base = concat_faa_historical_df(df_base, df_new)
|
||||||
|
assert df_base['download_date'].is_monotonic_increasing, "download_date is not monotonic increasing"
|
||||||
|
else:
|
||||||
|
df_base = df_new
|
||||||
|
|
||||||
df_base.to_csv(OUT_ROOT / f"openairframes_faa_{start_date_str}_{date_str}.csv", index=False)
|
df_base.to_csv(OUT_ROOT / f"openairframes_faa_{start_date_str}_{date_str}.csv", index=False)
|
||||||
@@ -0,0 +1,83 @@
|
|||||||
|
from pathlib import Path
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
import argparse
|
||||||
|
|
||||||
|
parser = argparse.ArgumentParser(description="Create daily Transport Canada release")
|
||||||
|
parser.add_argument("--date", type=str, help="Date to process (YYYY-MM-DD format, default: today)")
|
||||||
|
parser.add_argument("--allow-bootstrap", action="store_true",
|
||||||
|
help="Permit rebuilding from a single day when no published asset is found. "
|
||||||
|
"Onboarding only: a missing asset is otherwise indistinguishable from a "
|
||||||
|
"transient outage, and rebuilding would erase the accumulated history.")
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
if args.date:
|
||||||
|
date_str = args.date
|
||||||
|
else:
|
||||||
|
date_str = datetime.now(timezone.utc).strftime("%Y-%m-%d")
|
||||||
|
|
||||||
|
out_dir = Path("data/tc_ccarcs")
|
||||||
|
out_dir.mkdir(parents=True, exist_ok=True)
|
||||||
|
zip_name = f"ccarcsdb_{date_str}.zip"
|
||||||
|
|
||||||
|
zip_path = out_dir / zip_name
|
||||||
|
if not zip_path.exists():
|
||||||
|
url = "https://wwwapps.tc.gc.ca/saf-sec-sur/2/ccarcs-riacc/download/ccarcsdb.zip"
|
||||||
|
from urllib.request import Request, urlopen
|
||||||
|
|
||||||
|
# CCARCS 403s a default urllib agent. Any browser-like UA works; the exact
|
||||||
|
# version string is not load-bearing.
|
||||||
|
req = Request(
|
||||||
|
url,
|
||||||
|
headers={
|
||||||
|
"User-Agent": (
|
||||||
|
"Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 "
|
||||||
|
"(KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36"
|
||||||
|
)
|
||||||
|
},
|
||||||
|
method="GET",
|
||||||
|
)
|
||||||
|
|
||||||
|
with urlopen(req, timeout=120) as r:
|
||||||
|
body = r.read()
|
||||||
|
# TC serves an HTML maintenance page with a 200, which would otherwise be cached
|
||||||
|
# under a .zip name and re-read on every later run.
|
||||||
|
if body[:2] != b"PK":
|
||||||
|
raise RuntimeError(f"{url} did not return a zip (got {body[:40]!r})")
|
||||||
|
tmp_path = zip_path.with_suffix(".part")
|
||||||
|
tmp_path.write_bytes(body)
|
||||||
|
tmp_path.replace(zip_path)
|
||||||
|
|
||||||
|
OUT_ROOT = Path("data/openairframes")
|
||||||
|
OUT_ROOT.mkdir(parents=True, exist_ok=True)
|
||||||
|
from derive_from_tc_ccarcs import convert_tc_ccarcs_to_df
|
||||||
|
# Named for FAA but column-agnostic: fingerprints every column except download_date.
|
||||||
|
from derive_from_faa_master_txt import concat_faa_historical_df
|
||||||
|
from get_latest_release import get_latest_aircraft_tc_csv_df
|
||||||
|
df_new = convert_tc_ccarcs_to_df(zip_path, date_str)
|
||||||
|
|
||||||
|
# Only a genuine first run may rebuild from a single day. Every other failure -- a rate
|
||||||
|
# limit, a schema change, a truncated download -- must stop the run, because this file
|
||||||
|
# becomes tomorrow's base and silently republishing one day erases the whole history.
|
||||||
|
try:
|
||||||
|
df_base, start_date_str = get_latest_aircraft_tc_csv_df()
|
||||||
|
except FileNotFoundError as e:
|
||||||
|
if not args.allow_bootstrap:
|
||||||
|
raise SystemExit(
|
||||||
|
f"No published Transport Canada asset found: {e}\n"
|
||||||
|
"This is indistinguishable from a transient outage, and rebuilding from one day "
|
||||||
|
"would erase the accumulated history. Pass --allow-bootstrap when onboarding."
|
||||||
|
) from None
|
||||||
|
print(f"Bootstrapping Transport Canada from today only (--allow-bootstrap): {e}")
|
||||||
|
df_base = None
|
||||||
|
start_date_str = date_str
|
||||||
|
|
||||||
|
if df_base is not None:
|
||||||
|
missing = set(df_base.columns) ^ set(df_new.columns)
|
||||||
|
if missing:
|
||||||
|
raise SystemExit(f"Column set changed since the last release: {sorted(missing)}")
|
||||||
|
df_base = concat_faa_historical_df(df_base, df_new)
|
||||||
|
assert df_base['download_date'].is_monotonic_increasing, "download_date is not monotonic increasing"
|
||||||
|
else:
|
||||||
|
df_base = df_new
|
||||||
|
|
||||||
|
df_base.to_csv(OUT_ROOT / f"openairframes_tc_{start_date_str}_{date_str}.csv", index=False)
|
||||||
@@ -0,0 +1,248 @@
|
|||||||
|
from pathlib import Path
|
||||||
|
import csv
|
||||||
|
import io
|
||||||
|
import re
|
||||||
|
import zipfile
|
||||||
|
|
||||||
|
import pandas as pd
|
||||||
|
|
||||||
|
from derive_from_faa_master_txt import normalize
|
||||||
|
|
||||||
|
# CCARCS ships headerless, latin1, comma-delimited exports. Column names come from
|
||||||
|
# carslayout.txt in the same archive and must stay in file order.
|
||||||
|
CARSCURR_COLUMNS = [
|
||||||
|
"MARK", "REGISTRATION_SUB_TYPE_E", "REGISTRATION_SUB_TYPE_F", "COMMON_NAME",
|
||||||
|
"MODEL_NAME", "MANUFACTURERS_SERIAL_NUMBER", "MANUFACTURER_SERIAL_COMPRESSED",
|
||||||
|
"ID_PLATE_MANUFACTURERS_NAME", "BASIS_FOR_REGISTRATION", "BASIS_FOR_REGISTRATION_F",
|
||||||
|
"AIRCRAFT_CATEGORY_E", "AIRCRAFT_CATEGORY_F", "DATE_OF_IMPORT", "ENGINE_MANUF",
|
||||||
|
"POWERGLIDER_FLAG", "ENGINE_CATEGORY_E", "ENGINE_CATEGORY_F", "NUMBER_OF_ENGINES",
|
||||||
|
"NUMBER_OF_SEATS", "AIR_WEIGHT_KILOS", "SALE_REPORTED", "ISSUE_DATE",
|
||||||
|
"EFFECTIVE_DATE", "INEFFECTIVE_DATE", "REGISTERED_PURPOSE_E", "REGISTERED_PURPOSE_F",
|
||||||
|
"FLIGHT_AUTHORITY_E", "FLIGHT_AUTHORITY_F", "MANUFACTURE_OR_ASSEMBLY",
|
||||||
|
"COUNTRY_MANUFACTURE_ASS_E", "COUNTRY_MANUFACTURE_ASS_F", "DATE_MANUFACTURE_ASSEMBLY",
|
||||||
|
"BASE_OF_OPERATIONS_CTRY_E", "BASE_OF_OPERATIONS_CTRY_F", "BASE_PROVINCE_OR_STATE_E",
|
||||||
|
"BASE_PROVINCE_OR_STATE_F", "CITY_AIRPORT", "TYPE_CERTIFICATE_NUMBER",
|
||||||
|
"REGISTRATION_AUTH_STATUS_E", "REGISTRATION_AUTH_STATUS_F", "MULTIPLE_OWNER_FLAG",
|
||||||
|
"MODIFIED_DATE", "MODE_S_TRANSPONDER_BINARY", "PHYSICAL_FILE_REGION_E",
|
||||||
|
"PHYSICAL_FILE_REGION_F", "EX_MILITARY_MARK", "TRIMMED_MARK",
|
||||||
|
]
|
||||||
|
|
||||||
|
CARSOWNR_COLUMNS = [
|
||||||
|
"MARK_LINK", "FULL_NAME", "TRADE_NAME", "STREET_NAME", "STREET_NAME2", "CITY",
|
||||||
|
"PROVINCE_OR_STATE_E", "PROVINCE_OR_STATE_F", "POSTAL_CODE", "COUNTRY_E", "COUNTRY_F",
|
||||||
|
"TYPE_OF_OWNER_E", "TYPE_OF_OWNER_F", "ACTIVE_FLAG", "CARE_OF", "REGION_E", "REGION_F",
|
||||||
|
"OWNER_NAME_OLD_FORMAT", "MAIL_RECIPIENT", "TRIMMED_MARK",
|
||||||
|
]
|
||||||
|
|
||||||
|
# Mailing address of the single designated recipient, matching the registrant_* address
|
||||||
|
# the FAA build already publishes. Addresses are per-party, so they are taken from the one
|
||||||
|
# MAIL_RECIPIENT row rather than merged across co-owners.
|
||||||
|
# registrant_zip_code holds the Canadian postal code: the name is the FAA's, and a union
|
||||||
|
# table needs one column per concept, not one per country's vocabulary.
|
||||||
|
OWNER_ADDRESS_COLUMNS = {
|
||||||
|
"STREET_NAME": "registrant_street_1",
|
||||||
|
"STREET_NAME2": "registrant_street_2",
|
||||||
|
"CITY": "registrant_city",
|
||||||
|
"POSTAL_CODE": "registrant_zip_code",
|
||||||
|
"CARE_OF": "registrant_care_of",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
FOOTER_RE = re.compile(r"\s*(\d+) rows selected\.\s*")
|
||||||
|
|
||||||
|
# Floor, not an expectation: Canada's register is ~35k aircraft and carsownr is larger
|
||||||
|
# still, so 1000 only catches a grossly truncated export. The footer row-count check above
|
||||||
|
# is what actually validates the parse; this guards the case where the footer agrees with a
|
||||||
|
# near-empty body.
|
||||||
|
MIN_EXPECTED_ROWS = 1000
|
||||||
|
|
||||||
|
|
||||||
|
def _read_ccarcs_entry(zip_path: Path, entry: str, columns: list[str]) -> pd.DataFrame:
|
||||||
|
"""Read one headerless CCARCS export into a DataFrame.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
ValueError: on any row whose width is neither the declared column count nor a
|
||||||
|
blank/footer line, on a missing or disagreeing "N rows selected." footer, or
|
||||||
|
on a row count below MIN_EXPECTED_ROWS.
|
||||||
|
"""
|
||||||
|
with zipfile.ZipFile(zip_path) as z:
|
||||||
|
text = z.read(entry).decode("latin1")
|
||||||
|
|
||||||
|
rows = []
|
||||||
|
declared = None
|
||||||
|
# newline="" so a CRLF export does not leave \r on the final field of every row.
|
||||||
|
for row in csv.reader(io.StringIO(text, newline="")):
|
||||||
|
if len(row) == len(columns):
|
||||||
|
rows.append([cell.strip() for cell in row])
|
||||||
|
continue
|
||||||
|
if not row or not any(cell.strip() for cell in row):
|
||||||
|
continue # trailing blank line
|
||||||
|
match = FOOTER_RE.fullmatch(row[0]) if len(row) == 1 else None
|
||||||
|
if match:
|
||||||
|
declared = int(match.group(1))
|
||||||
|
continue
|
||||||
|
raise ValueError(
|
||||||
|
f"{entry}: row with {len(row)} fields, expected {len(columns)}: {row[:3]!r}"
|
||||||
|
)
|
||||||
|
|
||||||
|
# The spool footer is a free checksum from the source; a short export is otherwise
|
||||||
|
# indistinguishable from a genuinely smaller register.
|
||||||
|
if declared is None:
|
||||||
|
raise ValueError(f"{entry}: no 'N rows selected.' footer; export is truncated")
|
||||||
|
if declared != len(rows):
|
||||||
|
raise ValueError(f"{entry}: footer declares {declared} rows, parsed {len(rows)}")
|
||||||
|
if len(rows) < MIN_EXPECTED_ROWS:
|
||||||
|
raise ValueError(f"{entry}: only {len(rows)} rows, expected >= {MIN_EXPECTED_ROWS}")
|
||||||
|
|
||||||
|
return pd.DataFrame(rows, columns=columns)
|
||||||
|
|
||||||
|
|
||||||
|
def tc_full_registration(mark: str) -> str:
|
||||||
|
"""Expand a trimmed CCARCS mark into the full Canadian registration.
|
||||||
|
|
||||||
|
CCARCS stores the bare mark in both MARK and TRIMMED_MARK, so the prefix has to be
|
||||||
|
reconstructed: three-character marks are vintage CF- registrations, everything else
|
||||||
|
takes the modern C- prefix. Returns "" for a blank mark.
|
||||||
|
"""
|
||||||
|
mark = (mark or "").strip().upper()
|
||||||
|
if not mark:
|
||||||
|
return ""
|
||||||
|
return f"CF-{mark}" if len(mark) == 3 else f"C-{mark}"
|
||||||
|
|
||||||
|
|
||||||
|
def binary_to_hex(binary: str) -> str:
|
||||||
|
"""Convert a 24-bit Mode S binary string to a 6-digit uppercase hex address.
|
||||||
|
|
||||||
|
Returns "" for empty, non-binary, or non-24-bit input. Width is checked because
|
||||||
|
this column is the join key against ADS-B data: a short field would otherwise
|
||||||
|
zero-pad into a plausible address belonging to a different aircraft.
|
||||||
|
"""
|
||||||
|
binary = (binary or "").strip()
|
||||||
|
if len(binary) != 24 or any(c not in "01" for c in binary):
|
||||||
|
return ""
|
||||||
|
return f"{int(binary, 2):06X}"
|
||||||
|
|
||||||
|
|
||||||
|
def _merge_owners(df_ownr: pd.DataFrame) -> pd.DataFrame:
|
||||||
|
"""Collapse the active registered parties for each mark into a single row.
|
||||||
|
|
||||||
|
A co-owned mark repeats with a different party each time; keeping only the mail
|
||||||
|
recipient would silently drop the rest.
|
||||||
|
|
||||||
|
Each field is deduplicated and blank-skipped independently, so the values are NOT
|
||||||
|
index-parallel: a mark with three owners can emit three names but one province.
|
||||||
|
Consumers must not split on ", " and zip the columns together.
|
||||||
|
"""
|
||||||
|
# ACTIVE_FLAG is "A"/"I", but "I" does not mean "former owner": 1,932 currently
|
||||||
|
# Registered marks carry only "I" parties, and those rows are the MAIL_RECIPIENT.
|
||||||
|
# So prefer active parties where a mark has any, and fall back to all of them
|
||||||
|
# rather than publishing a registered aircraft with no owner at all.
|
||||||
|
all_parties = df_ownr
|
||||||
|
active = df_ownr[df_ownr["ACTIVE_FLAG"].str.upper() == "A"]
|
||||||
|
marks_with_active = set(active["TRIMMED_MARK"])
|
||||||
|
df_ownr = pd.concat([
|
||||||
|
active,
|
||||||
|
df_ownr[~df_ownr["TRIMMED_MARK"].isin(marks_with_active)],
|
||||||
|
])
|
||||||
|
def join_unique(series: pd.Series) -> str:
|
||||||
|
seen = []
|
||||||
|
for value in series:
|
||||||
|
value = (value or "").strip()
|
||||||
|
if value and value not in seen:
|
||||||
|
seen.append(value)
|
||||||
|
return ", ".join(seen)
|
||||||
|
|
||||||
|
def count_distinct(series: pd.Series) -> int:
|
||||||
|
return len({v.strip() for v in series if v and v.strip()})
|
||||||
|
|
||||||
|
grouped = df_ownr.groupby("TRIMMED_MARK", sort=False).agg(
|
||||||
|
registrant_name=("FULL_NAME", join_unique),
|
||||||
|
registrant_state=("PROVINCE_OR_STATE_E", join_unique),
|
||||||
|
registrant_country=("COUNTRY_E", join_unique),
|
||||||
|
registrant_type=("TYPE_OF_OWNER_E", join_unique),
|
||||||
|
registrant_party_count=("FULL_NAME", count_distinct),
|
||||||
|
).reset_index()
|
||||||
|
|
||||||
|
# A party row states its own type ("Individual"); that stops being true of the mark
|
||||||
|
# once several parties share it. Counting distinct names rather than rows keeps this
|
||||||
|
# consistent with owner_name, which is also deduplicated.
|
||||||
|
grouped.loc[grouped["registrant_party_count"] > 1, "registrant_type"] = "Co-owner"
|
||||||
|
|
||||||
|
# Taken from the unfiltered frame: the designated recipient is the designated
|
||||||
|
# recipient even when its own party row is flagged inactive.
|
||||||
|
recipient = (
|
||||||
|
all_parties[all_parties["MAIL_RECIPIENT"].str.upper() == "Y"]
|
||||||
|
.drop_duplicates(subset="TRIMMED_MARK", keep="first")
|
||||||
|
.rename(columns=OWNER_ADDRESS_COLUMNS)
|
||||||
|
)
|
||||||
|
return grouped.merge(
|
||||||
|
recipient[["TRIMMED_MARK", *OWNER_ADDRESS_COLUMNS.values()]],
|
||||||
|
on="TRIMMED_MARK",
|
||||||
|
how="left",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def convert_tc_ccarcs_to_df(zip_path: Path, date: str) -> pd.DataFrame:
|
||||||
|
"""Build the OpenAirframes Transport Canada frame from a CCARCS zip."""
|
||||||
|
df = _read_ccarcs_entry(zip_path, "carscurr.txt", CARSCURR_COLUMNS)
|
||||||
|
df_ownr = _read_ccarcs_entry(zip_path, "carsownr.txt", CARSOWNR_COLUMNS)
|
||||||
|
|
||||||
|
df = df.merge(_merge_owners(df_ownr), on="TRIMMED_MARK", how="left")
|
||||||
|
|
||||||
|
out = pd.DataFrame({
|
||||||
|
"download_date": date,
|
||||||
|
# The FAA frame already carries `source`; it is the union discriminator.
|
||||||
|
"source": "TC",
|
||||||
|
"transponder_code_hex": df["MODE_S_TRANSPONDER_BINARY"].map(binary_to_hex),
|
||||||
|
"registration_number": df["TRIMMED_MARK"].map(tc_full_registration),
|
||||||
|
"mark": df["TRIMMED_MARK"],
|
||||||
|
"aircraft_manufacturer": df["COMMON_NAME"],
|
||||||
|
"aircraft_model": df["MODEL_NAME"],
|
||||||
|
"serial_number": df["MANUFACTURERS_SERIAL_NUMBER"],
|
||||||
|
"aircraft_category": df["AIRCRAFT_CATEGORY_E"],
|
||||||
|
"engine_manufacturer": df["ENGINE_MANUF"],
|
||||||
|
"engine_category": df["ENGINE_CATEGORY_E"],
|
||||||
|
"aircraft_number_of_engines": df["NUMBER_OF_ENGINES"],
|
||||||
|
"aircraft_number_of_seats": df["NUMBER_OF_SEATS"],
|
||||||
|
"max_weight_kilos": df["AIR_WEIGHT_KILOS"],
|
||||||
|
"status": df["REGISTRATION_AUTH_STATUS_E"],
|
||||||
|
"registration_sub_type": df["REGISTRATION_SUB_TYPE_E"],
|
||||||
|
"basis_for_registration": df["BASIS_FOR_REGISTRATION"],
|
||||||
|
"registered_purpose": df["REGISTERED_PURPOSE_E"],
|
||||||
|
"flight_authority": df["FLIGHT_AUTHORITY_E"],
|
||||||
|
"type_certificate_number": df["TYPE_CERTIFICATE_NUMBER"],
|
||||||
|
"country_manufacture": df["COUNTRY_MANUFACTURE_ASS_E"],
|
||||||
|
"date_manufacture_assembly": df["DATE_MANUFACTURE_ASSEMBLY"],
|
||||||
|
"base_country": df["BASE_OF_OPERATIONS_CTRY_E"],
|
||||||
|
"base_province_or_state": df["BASE_PROVINCE_OR_STATE_E"],
|
||||||
|
"city_airport": df["CITY_AIRPORT"],
|
||||||
|
"ex_military_mark": df["EX_MILITARY_MARK"],
|
||||||
|
"multiple_owner_flag": df["MULTIPLE_OWNER_FLAG"],
|
||||||
|
"registrant_name": df["registrant_name"],
|
||||||
|
"registrant_type": df["registrant_type"],
|
||||||
|
"registrant_state": df["registrant_state"],
|
||||||
|
"registrant_country": df["registrant_country"],
|
||||||
|
"registrant_care_of": df["registrant_care_of"],
|
||||||
|
"registrant_street_1": df["registrant_street_1"],
|
||||||
|
"registrant_street_2": df["registrant_street_2"],
|
||||||
|
"registrant_city": df["registrant_city"],
|
||||||
|
"registrant_zip_code": df["registrant_zip_code"],
|
||||||
|
"issue_date": df["ISSUE_DATE"],
|
||||||
|
"effective_date": df["EFFECTIVE_DATE"],
|
||||||
|
"ineffective_date": df["INEFFECTIVE_DATE"],
|
||||||
|
"modified_date": df["MODIFIED_DATE"],
|
||||||
|
})
|
||||||
|
|
||||||
|
# Position matches the FAA frame (after registration_number). Ordering is cosmetic:
|
||||||
|
# concat_faa_historical_df reindexes df_new to the base's columns before merging.
|
||||||
|
out.insert(3, "openairframes_id", (
|
||||||
|
normalize(out["aircraft_manufacturer"])
|
||||||
|
+ "|"
|
||||||
|
+ normalize(out["aircraft_model"])
|
||||||
|
+ "|"
|
||||||
|
+ normalize(out["serial_number"])
|
||||||
|
))
|
||||||
|
|
||||||
|
out = out.fillna("")
|
||||||
|
out = out.replace("None", "")
|
||||||
|
return out
|
||||||
@@ -3,6 +3,7 @@ from __future__ import annotations
|
|||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Iterable, Optional
|
from typing import Iterable, Optional
|
||||||
|
import os
|
||||||
import re
|
import re
|
||||||
import urllib.request
|
import urllib.request
|
||||||
import urllib.error
|
import urllib.error
|
||||||
@@ -138,15 +139,29 @@ def download_latest_aircraft_csv(
|
|||||||
Path to the downloaded file
|
Path to the downloaded file
|
||||||
"""
|
"""
|
||||||
output_dir = Path(output_dir)
|
output_dir = Path(output_dir)
|
||||||
assets = get_latest_release_assets(repo, github_token=github_token)
|
github_token = github_token or os.environ.get("GITHUB_TOKEN")
|
||||||
try:
|
|
||||||
asset = pick_asset(assets, name_regex=r"^openairframes_faa_.*\.csv$")
|
for release in get_releases(repo, github_token=github_token, per_page=30):
|
||||||
except FileNotFoundError:
|
assets = get_release_assets_from_release_data(release)
|
||||||
# Fallback to old naming pattern
|
try:
|
||||||
asset = pick_asset(assets, name_regex=r"^openairframes_\d{4}-\d{2}-\d{2}_.*\.csv$")
|
asset = pick_asset(assets, name_regex=r"^openairframes_faa_.*\.csv$")
|
||||||
saved_to = download_asset(asset, output_dir / asset.name, github_token=github_token)
|
except FileNotFoundError:
|
||||||
print(f"Downloaded: {asset.name} ({asset.size} bytes) -> {saved_to}")
|
try:
|
||||||
return saved_to
|
# Fallback to old naming pattern
|
||||||
|
asset = pick_asset(assets, name_regex=r"^openairframes_\d{4}-\d{2}-\d{2}_.*\.csv$")
|
||||||
|
except FileNotFoundError:
|
||||||
|
continue
|
||||||
|
saved_to = download_asset(asset, output_dir / asset.name, github_token=github_token)
|
||||||
|
if asset.size and saved_to.stat().st_size != asset.size:
|
||||||
|
raise RuntimeError(
|
||||||
|
f"{asset.name}: downloaded {saved_to.stat().st_size} bytes, expected {asset.size}"
|
||||||
|
)
|
||||||
|
print(f"Downloaded: {asset.name} ({asset.size} bytes) -> {saved_to}")
|
||||||
|
return saved_to
|
||||||
|
|
||||||
|
raise FileNotFoundError(
|
||||||
|
"No release in the last 30 releases has an asset matching 'openairframes_faa_.*\\.csv$'"
|
||||||
|
)
|
||||||
|
|
||||||
def get_latest_aircraft_faa_csv_df():
|
def get_latest_aircraft_faa_csv_df():
|
||||||
csv_path = download_latest_aircraft_csv()
|
csv_path = download_latest_aircraft_csv()
|
||||||
@@ -167,6 +182,69 @@ def get_latest_aircraft_faa_csv_df():
|
|||||||
return df, date_str
|
return df, date_str
|
||||||
|
|
||||||
|
|
||||||
|
def download_latest_aircraft_tc_csv(
|
||||||
|
output_dir: Path = Path("downloads"),
|
||||||
|
github_token: Optional[str] = None,
|
||||||
|
repo: str = REPO,
|
||||||
|
) -> Path:
|
||||||
|
"""
|
||||||
|
Download the latest openairframes_tc_*.csv file from the latest GitHub release.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
output_dir: Directory to save the downloaded file (default: "downloads")
|
||||||
|
github_token: Optional GitHub token for authentication
|
||||||
|
repo: GitHub repository in format "owner/repo" (default: REPO)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Path to the downloaded file
|
||||||
|
"""
|
||||||
|
output_dir = Path(output_dir)
|
||||||
|
github_token = github_token or os.environ.get("GITHUB_TOKEN")
|
||||||
|
|
||||||
|
# Walk back through releases rather than reading only `latest`. The TC asset is
|
||||||
|
# optional, so a single failed build publishes a release without it; anchoring on
|
||||||
|
# `latest` would then make the caller rebuild history from one day and republish
|
||||||
|
# that as the whole dataset.
|
||||||
|
for release in get_releases(repo, github_token=github_token, per_page=30):
|
||||||
|
assets = get_release_assets_from_release_data(release)
|
||||||
|
try:
|
||||||
|
asset = pick_asset(assets, name_regex=r"^openairframes_tc_.*\.csv$")
|
||||||
|
except FileNotFoundError:
|
||||||
|
continue
|
||||||
|
saved_to = download_asset(asset, output_dir / asset.name, github_token=github_token)
|
||||||
|
if asset.size and saved_to.stat().st_size != asset.size:
|
||||||
|
raise RuntimeError(
|
||||||
|
f"{asset.name}: downloaded {saved_to.stat().st_size} bytes, expected {asset.size}"
|
||||||
|
)
|
||||||
|
print(f"Downloaded: {asset.name} ({asset.size} bytes) -> {saved_to}")
|
||||||
|
return saved_to
|
||||||
|
|
||||||
|
raise FileNotFoundError(
|
||||||
|
"No release in the last 30 releases has an asset matching 'openairframes_tc_.*\\.csv$'"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def get_latest_aircraft_tc_csv_df():
|
||||||
|
"""Return (DataFrame, start_date_str) for the most recent published TC release.
|
||||||
|
|
||||||
|
Raises FileNotFoundError when no recent release carries a TC asset, and ValueError
|
||||||
|
when the asset filename has no parseable start date.
|
||||||
|
"""
|
||||||
|
csv_path = download_latest_aircraft_tc_csv()
|
||||||
|
import pandas as pd
|
||||||
|
# keep_default_na=False: a literal "NA"/"N/A" in the source would otherwise read back
|
||||||
|
# as NaN -> "" while the fresh parse keeps the string, so the row fingerprints would
|
||||||
|
# never match and every affected record would re-append on every run.
|
||||||
|
df = pd.read_csv(csv_path, dtype=str, keep_default_na=False)
|
||||||
|
df = df.fillna("")
|
||||||
|
# Only the start date is taken; the end date is always the run's own date.
|
||||||
|
match = re.search(r"openairframes_tc_(\d{4}-\d{2}-\d{2})_", str(csv_path))
|
||||||
|
if not match:
|
||||||
|
raise ValueError(f"Could not extract date from filename: {csv_path.name}")
|
||||||
|
|
||||||
|
return df, match.group(1)
|
||||||
|
|
||||||
|
|
||||||
def download_latest_aircraft_adsb_csv(
|
def download_latest_aircraft_adsb_csv(
|
||||||
output_dir: Path = Path("downloads"),
|
output_dir: Path = Path("downloads"),
|
||||||
github_token: Optional[str] = None,
|
github_token: Optional[str] = None,
|
||||||
|
|||||||
Reference in New Issue
Block a user