From b340cc025efee709ac3c416ce9c4fe3f32088767 Mon Sep 17 00:00:00 2001 From: Joseph Magly <1159087+jmagly@users.noreply.github.com> Date: Fri, 21 Aug 2026 19:31:58 -0400 Subject: [PATCH 1/4] docs: define Jetson support acceptance plan --- .aiwg/testing/jetson-support-assessment.md | 273 +++++++++++++++++++++ docs/conditional-testing.md | 6 + docs/jetson-support.md | 147 +++++++++++ 3 files changed, 426 insertions(+) create mode 100644 .aiwg/testing/jetson-support-assessment.md create mode 100644 docs/jetson-support.md diff --git a/.aiwg/testing/jetson-support-assessment.md b/.aiwg/testing/jetson-support-assessment.md new file mode 100644 index 0000000..685e2f5 --- /dev/null +++ b/.aiwg/testing/jetson-support-assessment.md @@ -0,0 +1,273 @@ +# Jetson support acceptance and release-validation strategy + +Date: 2026-08-21 +Scope: OBLITERATUS issue #31 +Status: assessment, not implementation + +## Executive conclusion + +Issue #31 cannot be validated or released on a generic ARM builder alone. +Generic ARM64 CI can cover repository-level portability, importability, +packaging, and CPU-only behavior, but Jetson support depends on the NVIDIA +JetPack software stack and on execution on Jetson-class hardware or a JetPack +container running on that hardware. + +The lowest-risk support model is: + +1. keep the default PR gate CPU-only and deterministic; +2. add a Jetson-specific conditional gate on a self-hosted ARM64 Jetson runner; +3. treat generic ARM64 CI as a preflight layer, not as acceptance evidence; +4. require real-device smoke evidence before declaring Jetson support. + +## What the current repo policy already says + +The project context already establishes that the default PR baseline is +CPU-safe, deterministic, and must not require accelerator, network, or remote +credentials. Conditional accelerator checks are explicitly outside the default +CPU job. + +The current conditional policy already has: + +- a CUDA gate on `self-hosted, linux, x64, cuda`; +- a documented waiver model for unavailable accelerator lanes; +- a GitHub Actions routing model that relies on runner labels and groups. + +That means Jetson support should be added as a new conditional lane, not folded +into the standard CPU PR job. + +## Evidence-based split: generic ARM64 vs Jetson device + +### Generic ARM64 builders can validate + +- Python packaging and metadata. +- Pure-Python imports and CLI/help entry points. +- CPU-only unit and boundary tests. +- Static/configuration logic that does not require CUDA or JetPack runtime + libraries. +- The fact that a workflow can target `ARM64` self-hosted runners. + +This is enough for portability regressions, but not enough for Jetson runtime +support. + +### Jetson hardware or JetPack container is required for + +- CUDA device discovery. +- CUDA execution and memory behavior. +- cuDNN / TensorRT / JetPack-specific library compatibility. +- Any claim that NVIDIA-provided Jetson PyTorch wheels run correctly. +- Any claim that the installed runtime actually sees a Jetson GPU. + +NVIDIA’s Jetson PyTorch installation guide says the provided wheels are meant +to be installed on top of a specified JetPack version on a Jetson device, and +verification is done by importing `torch` on the Jetson platform. + +NVIDIA’s Jetson Linux validation guide also says CUDA samples can be run +natively on the target or inside the JetPack container, which is the right +acceptance bar for hardware-backed validation. + +## Support tiers + +### Tier 0: source-only portability + +Purpose: prove the codebase does not contain obvious ARM-incompatible +assumptions. + +Environment: + +- GitHub-hosted Linux or self-hosted ARM64 runner; +- CPU-only; +- no NVIDIA driver, no JetPack, no CUDA. + +Checks: + +- `python -m build --sdist --wheel` +- install the built wheel into a clean venv +- `python -m obliteratus --help` +- import and module smoke tests +- all CPU-only pytest markers + +Acceptance: + +- no architecture-specific syntax/runtime breakage; +- no packaging or import regressions. + +### Tier 1: Jetson preflight on generic ARM64 + +Purpose: catch obvious Jetson-adjacent integration mistakes before touching +hardware. + +Environment: + +- ARM64 Linux runner without JetPack; +- optional cross-check against Jetson-targeted config files and dependency + pins. + +Checks: + +- validate Jetson-specific config manifests and workflow wiring; +- verify that Jetson jobs are gated behind dedicated labels and environment + variables; +- confirm that no Jetson-only dependency is pulled into the default PR job. + +Acceptance: + +- the repo can express a Jetson lane cleanly; +- default CI remains CPU-only. + +### Tier 2: Jetson hardware smoke + +Purpose: prove the runtime actually works on a Jetson device. + +Environment: + +- self-hosted Linux ARM64 Jetson runner; +- JetPack installed and active; +- NVIDIA runtime / CUDA stack available. + +Checks: + +- `torch` import from the JetPack-aligned wheel set; +- `torch.cuda.is_available()` and device enumeration; +- a tiny tensor operation on GPU; +- a minimal model-load or model-adapter smoke if the feature requires it; +- a short CUDA sample or equivalent library smoke; +- optional containerized run using the JetPack container on the target. + +Acceptance: + +- the device is recognized as CUDA-capable; +- the relevant Jetson wheel/container stack works on the intended JetPack + version; +- the workflow emits logs and artifacts that identify the exact JetPack / CUDA / + wheel set used. + +### Tier 3: release validation + +Purpose: certify a tagged release candidate. + +Environment: + +- Tier 2 Jetson hardware; +- pinned dependency set and reproducible artifacts; +- retained logs and hashes. + +Checks: + +- Tier 2 smoke; +- the repo’s full release-validation suite; +- supply-chain checks for vendor artifacts; +- any release note or compatibility matrix update. + +Acceptance: + +- tagged releases declare Jetson support only when Tier 2 and Tier 3 pass + against the exact tagged commit. + +## Recommended runner labels + +Use separate labels so the workflow can route cleanly and fail obviously if the +hardware is absent. + +- `self-hosted, linux, arm64, jetson` +- optional refinement: `self-hosted, linux, arm64, jetson, orin` +- optional refinement: `self-hosted, linux, arm64, jetpack-` + +Keep the generic ARM64 preflight on a separate label such as: + +- `self-hosted, linux, arm64, arm64-preflight` + +This avoids accidentally treating a generic ARM machine as a Jetson target. + +GitHub’s self-hosted runner routing is label-based, and jobs remain queued if +no matching runner is online. That makes an explicit Jetson label the right +mechanism for a physical-device lane. + +## Exact acceptance gates + +Jetson support should be considered ready only when all of these are true: + +1. The default PR gate still passes with no Jetson dependency. +2. A dedicated Jetson job passes on real hardware or an equivalent JetPack + target container. +3. The exact supported JetPack version is documented. +4. The exact NVIDIA wheel/container provenance is pinned and hashed. +5. The repo has a regression test that fails if Jetson detection or the CUDA + smoke is broken. +6. The release notes state the supported JetPack / CUDA / TensorRT matrix. + +## Memory and thermal constraints + +Jetson support should assume small-device variability, even on higher-memory +SKUs. + +Practical guardrails: + +- use tiny models and short sequences for smoke tests; +- cap wall time tightly; +- assert memory-sensitive paths with low-footprint fixtures; +- avoid benchmark-length runs on the Jetson lane; +- collect `nvidia-smi`-equivalent or device telemetry only if the platform + exposes it; +- do not rely on long repeated sweeps for acceptance. + +If a test is intended to validate memory pressure, it should do so with a +deterministic threshold and a short timeout, not with a human-observed +benchmark. + +## Rollback and waiver policy + +If Jetson hardware is unavailable, the repo should: + +- keep the Jetson lane optional and separately labeled; +- preserve the generic ARM64 preflight; +- publish a time-bounded waiver in the existing conditional-policy pattern; +- avoid claiming Jetson support in release notes. + +If a Jetson regression is found after support is published: + +- revert or patch the exact regression on the supported branch; +- keep the Jetson lane red until the hardware smoke passes again; +- do not widen the default PR gate to hide the failure. + +## Supply-chain and security considerations + +Jetson support increases supply-chain risk because the runtime depends on vendor +wheels and platform-specific system packages. + +Controls to require: + +- pin exact NVIDIA wheel URLs and versions; +- verify hashes for every downloaded binary artifact; +- prefer official JetPack / NVIDIA container images over ad hoc mixed library + stacks; +- do not mix arbitrary patched CUDA/TensorRT libraries into the supported + matrix; +- record the exact JetPack release, CUDA version, and TensorRT version in the + evidence trail; +- treat any vendor wheel update as a release-affecting change. + +NVIDIA’s TensorRT documentation explicitly notes that JetPack deployments must +remain on a TensorRT 10.x release supported by the JetPack version. That means +the Jetson compatibility matrix must be version-pinned, not “latest by default.” + +## Recommended implementation path + +1. Keep issue #31 open as a feature/specification item until the acceptance + matrix is merged. +2. Add a Jetson conditional policy entry and workflow job with an explicit + `jetson` runner label. +3. Add a minimal Jetson smoke test that checks import, CUDA discovery, and one + tiny GPU operation. +4. Add a generic ARM64 preflight job to catch packaging and configuration + regressions early. +5. Document the supported JetPack / CUDA / TensorRT matrix in the repo docs. +6. Promote Jetson claims only after a real-device run has passed on the exact + release candidate. + +## Source URLs + +- https://docs.nvidia.com/deeplearning/frameworks/install-pytorch-jetson-platform/index.html +- https://docs.nvidia.com/jetson/archives/r36.5/DeveloperGuide/SD/TestPlanValidation.html +- https://docs.nvidia.com/deeplearning/tensorrt/latest/installing-tensorrt/installing.html +- https://docs.github.com/en/actions/reference/runners/self-hosted-runners +- https://docs.github.com/en/actions/reference/runners/github-hosted-runners diff --git a/docs/conditional-testing.md b/docs/conditional-testing.md index bc38504..10dc0ec 100644 --- a/docs/conditional-testing.md +++ b/docs/conditional-testing.md @@ -80,6 +80,12 @@ uv run --extra dev python scripts/run_conditional_gate.py cuda-runtime uv run --extra dev python scripts/run_conditional_gate.py bitsandbytes-runtime ``` +Jetson CUDA support is tracked separately from this generic x64 CUDA lane. A +generic Linux ARM build can prove package portability, but it does not prove +Jetson GPU support because Jetson depends on a JetPack/L4T-matched CUDA, cuDNN, +and PyTorch runtime. The support plan, recommended container path, and acceptance +criteria are documented in [NVIDIA Jetson support plan](jetson-support.md). + ## Apple MPS and MLX MPS uses a runner labeled `self-hosted`, `macOS`, `ARM64`, and `mps`; enable its diff --git a/docs/jetson-support.md b/docs/jetson-support.md new file mode 100644 index 0000000..8722a1b --- /dev/null +++ b/docs/jetson-support.md @@ -0,0 +1,147 @@ +# NVIDIA Jetson support plan + +Issue: https://github.com/elder-plinius/OBLITERATUS/issues/31 + +Internal testing assessment: +[.aiwg/testing/jetson-support-assessment.md](../.aiwg/testing/jetson-support-assessment.md) + +## Decision + +OBLITERATUS should treat Jetson AGX support as a distinct platform target, not +as a generic Linux ARM build. Generic `linux/aarch64` packaging can prove that +the Python package resolves and imports on ARM, but it does not prove CUDA on +Jetson because Jetson devices depend on JetPack/L4T-matched CUDA, cuDNN, and +PyTorch builds. + +The first supported target should be Jetson AGX Orin 64 GB on JetPack 6.x, run +through either NVIDIA's Jetson PyTorch instructions or a JetPack-compatible +container. Xavier-era JetPack 5 can be documented as best-effort until a runner +exists. + +## Evidence + +NVIDIA documents Jetson PyTorch as JetPack-specific pip wheels with GPU +acceleration and cuDNN support, installed on top of the matching JetPack +version: + +- https://docs.nvidia.com/deeplearning/frameworks/install-pytorch-jetson-platform/index.html +- https://catalog.ngc.nvidia.com/orgs/nvidia/containers/l4t-pytorch + +The NVIDIA NGC `l4t-pytorch` container is explicitly a Jetson/JetPack image and +requires matching the container tag to the installed JetPack/L4T release. The +`jetson-containers` project provides a practical composition path for PyTorch +and Transformers on Jetson, including compatible autotag selection: + +- https://github.com/dusty-nv/jetson-containers + +PyTorch upstream support is moving: PyTorch 2.11 release notes say PyPI wheels +now include CUDA 13.0 for Linux aarch64, while CUDA 12.6 and 12.8 wheels remain +available from `download.pytorch.org`. That improves generic ARM CUDA, but it +does not remove the need to match Jetson's installed JetPack/CUDA stack: + +- https://github.com/pytorch/pytorch/releases + +bitsandbytes currently publishes Linux aarch64 CUDA builds and lists targets +for CUDA 11.8 through 13.x. AGX Orin's Ampere GPU should satisfy the compute +capability requirement for 4-bit and 8-bit quantization, but the runtime still +has to be validated on the Jetson image actually used by contributors: + +- https://huggingface.co/docs/bitsandbytes/en/installation + +## Current repository gaps + +- `pyproject.toml` maps Linux and Windows Torch resolution to the CPU-only + PyTorch index. That is correct for default CI, but it means the locked default + environment will not discover Jetson CUDA. +- `uv.lock` includes a Linux aarch64 bitsandbytes wheel, but Torch is locked as + `2.13.0+cpu` on Linux. A Jetson runtime must intentionally override Torch + after syncing the default lock, similar to the existing CUDA conditional lane. +- `.github/workflows/conditional-tests.yml` has one CUDA lane labeled + `self-hosted, linux, x64, cuda` and installs an official `cu130` Torch build. + It cannot run on Jetson because the runner labels and Torch source are x64 + CUDA-specific. +- `docs/conditional-testing.md` documents generic CUDA and bitsandbytes probes, + not Jetson labels, JetPack requirements, or container startup. +- `Dockerfile` is based on `python:3.11-slim`, installs `requirements.txt`, and + is documented as local-only. It is not a Jetson image and should not be used + for a support claim. +- `obliteratus/device.py` can already detect CUDA through PyTorch, so no special + Jetson detection is needed for the first milestone. The blocker is installing + a CUDA-enabled Jetson PyTorch runtime and proving the existing device and + bitsandbytes contracts there. +- `tests/conditional/test_cuda_runtime.py` is a good starting probe, but it does + not record Jetson-specific platform evidence such as JetPack/L4T release, + CUDA version, Python version, architecture, Torch build, and bitsandbytes + binary availability. + +## Implementation options + +### Option A: Native Jetson runner, minimal repo changes + +Attach a self-hosted runner on the Jetson with labels such as +`self-hosted`, `linux`, `ARM64`, `jetson`, and `cuda`. Add a new conditional +gate `jetson-cuda-runtime` that: + +- syncs OBLITERATUS without replacing the system/vendor Jetson Torch build; +- verifies `platform.machine()` is `aarch64`; +- records `/etc/nv_tegra_release`, `nvcc --version`, `torch.__version__`, + `torch.version.cuda`, `torch.cuda.get_device_name(0)`, and + `bitsandbytes` import/runtime status; +- runs the existing CUDA and bitsandbytes probes. + +Tradeoff: lowest abstraction and fastest to validate real hardware, but it +requires maintaining the Jetson's host Python/runtime state carefully. + +### Option B: Jetson container lane + +Base a Jetson-specific image on an NVIDIA L4T PyTorch image or build it with +`jetson-containers` from `pytorch` plus `transformers`. Install OBLITERATUS into +that image with dependency constraints that do not replace the container's +Torch. Run the same `jetson-cuda-runtime` probe inside the container. + +Tradeoff: best reproducibility and closest to the reporter's concern about +patched/static libraries, but the image has to track JetPack/L4T tags and may +need per-JetPack dependency pins. + +### Option C: Generic Linux ARM build plus separate Jetson UAT + +Add a hosted or self-hosted Linux ARM packaging job that installs OBLITERATUS +with CPU Torch on aarch64, then keep Jetson CUDA as manual UAT evidence until a +stable runner exists. + +Tradeoff: useful packaging signal for ARM contributors, but it must not be +presented as Jetson GPU support because it does not exercise CUDA, cuDNN, +device placement, quantization kernels, or JetPack compatibility. + +## Recommended path + +Use Option B for the support claim and Option C as a cheap early-warning signal. +The container path matches Jetson's dependency model, avoids polluting the +default lock with Jetson-only Torch URLs, and gives contributors a repeatable +recipe. A native runner can still be used to execute the container and collect +evidence. Treat generic ARM64 as Tier 0/Tier 1 preflight only; require +real-device Jetson smoke and release validation before publishing a support +claim. + +## Acceptance criteria + +- `docs/conditional-testing.md` documents Jetson as a separate conditional gate + with supported JetPack versions, runner labels, and local/container commands. +- `ci/conditional-test-policy.json` includes a `jetson-cuda-runtime` gate with a + tracking issue and evidence freshness rule. +- `.github/workflows/conditional-tests.yml` has a Jetson-selected job that runs + only on a labeled Jetson runner or through a Jetson-compatible container. +- The gate records JetPack/L4T, architecture, Torch CUDA version, CUDA device + name, memory, and bitsandbytes status in JSON evidence. +- The gate runs without skips: + `python scripts/run_conditional_gate.py jetson-cuda-runtime`. +- README support language says Jetson AGX Orin is supported only for the + specific JetPack/runtime combinations that have fresh green conditional + evidence. + +## User guidance until support lands + +Do not install from the default OBLITERATUS lock and expect Jetson CUDA to work. +Start from a JetPack-compatible PyTorch runtime, then install OBLITERATUS +without replacing Torch. If using a container, prefer an L4T PyTorch image or a +`jetson-containers` build that matches the device's JetPack/L4T release. From 8814dbe3590042f2537b9882c28aa46b34a82601 Mon Sep 17 00:00:00 2001 From: Joseph Magly <1159087+jmagly@users.noreply.github.com> Date: Fri, 21 Aug 2026 19:35:21 -0400 Subject: [PATCH 2/4] docs: specify Jetson support path --- .../jetson-support-architecture.md | 71 +++++++++ .../REF-BITSANDBYTES-INSTALL-assessment.yaml | 13 ++ .../REF-JETPACK-62-assessment.yaml | 14 ++ .../REF-JETPACK-7-DOWNLOADS-assessment.yaml | 13 ++ ...REF-JETSON-PYTORCH-INSTALL-assessment.yaml | 14 ++ ...EF-JETSON-PYTORCH-RELEASES-assessment.yaml | 14 ++ .../REF-UV-PYTORCH-assessment.yaml | 13 ++ .../sources/REF-BITSANDBYTES-INSTALL.yaml | 10 ++ .aiwg/research/sources/REF-JETPACK-62.yaml | 10 ++ .../sources/REF-JETPACK-7-DOWNLOADS.yaml | 10 ++ .../sources/REF-JETSON-PYTORCH-INSTALL.yaml | 11 ++ .../sources/REF-JETSON-PYTORCH-RELEASES.yaml | 11 ++ .aiwg/research/sources/REF-UV-PYTORCH.yaml | 11 ++ docs/conditional-testing.md | 2 +- docs/platforms/jetson.md | 136 ++++++++++++++++++ 15 files changed, 352 insertions(+), 1 deletion(-) create mode 100644 .aiwg/architecture/jetson-support-architecture.md create mode 100644 .aiwg/research/quality-assessments/REF-BITSANDBYTES-INSTALL-assessment.yaml create mode 100644 .aiwg/research/quality-assessments/REF-JETPACK-62-assessment.yaml create mode 100644 .aiwg/research/quality-assessments/REF-JETPACK-7-DOWNLOADS-assessment.yaml create mode 100644 .aiwg/research/quality-assessments/REF-JETSON-PYTORCH-INSTALL-assessment.yaml create mode 100644 .aiwg/research/quality-assessments/REF-JETSON-PYTORCH-RELEASES-assessment.yaml create mode 100644 .aiwg/research/quality-assessments/REF-UV-PYTORCH-assessment.yaml create mode 100644 .aiwg/research/sources/REF-BITSANDBYTES-INSTALL.yaml create mode 100644 .aiwg/research/sources/REF-JETPACK-62.yaml create mode 100644 .aiwg/research/sources/REF-JETPACK-7-DOWNLOADS.yaml create mode 100644 .aiwg/research/sources/REF-JETSON-PYTORCH-INSTALL.yaml create mode 100644 .aiwg/research/sources/REF-JETSON-PYTORCH-RELEASES.yaml create mode 100644 .aiwg/research/sources/REF-UV-PYTORCH.yaml create mode 100644 docs/platforms/jetson.md diff --git a/.aiwg/architecture/jetson-support-architecture.md b/.aiwg/architecture/jetson-support-architecture.md new file mode 100644 index 0000000..c0433e5 --- /dev/null +++ b/.aiwg/architecture/jetson-support-architecture.md @@ -0,0 +1,71 @@ +# Jetson support architecture note + +Date: 2026-08-21 +Issue: https://github.com/elder-plinius/OBLITERATUS/issues/31 + +## Summary + +OBLITERATUS should support NVIDIA Jetson AGX as a JetPack-specific conditional +runtime target. A generic ARM build is useful as import/build coverage, but it +is not sufficient for Jetson CUDA support because the working runtime depends on +JetPack, L4T, CUDA, cuDNN, TensorRT, and NVIDIA's Jetson-compatible PyTorch +build or container. + +## Current repo state + +- `obliteratus/device.py` uses `torch.cuda.is_available()` for CUDA discovery. +- `docs/conditional-testing.md` defines hardware gates outside the mandatory + PR workflow. +- `.github/workflows/conditional-tests.yml` has an x64 CUDA gate using + `UV_TORCH_BACKEND=cu130`. +- `pyproject.toml` locks Linux PR resolution to CPU-only PyTorch through the + `pytorch-cpu` index. +- `Dockerfile` is a generic local `python:3.11-slim` image, not a Jetson image. + +These choices are correct for contributor PR turnaround, but they do not create +a Jetson-compatible runtime. + +## Required changes before claiming support + +1. Add a `jetson-runtime` conditional gate and policy entry. +2. Add a Jetson install path that preserves NVIDIA's Jetson PyTorch stack. +3. Add a Jetson container recipe or documented base-image override for the + selected JetPack tier. +4. Add a tiny hardware probe that records JetPack/L4T, CUDA, PyTorch, device + name, and OBLITERATUS `device=auto` behavior. +5. Decide whether `bitsandbytes` is supported, tier-limited, or disabled on the + selected Jetson stack based on hardware evidence. + +## Design constraints + +- Keep PR CI CPU-only and offline. +- Keep x64 CUDA and Jetson CUDA evidence separate. +- Do not let `uv sync` replace NVIDIA's Jetson PyTorch wheel/container runtime + with the generic CPU lock. +- Treat JetPack 6.x Orin and JetPack 7.x Thor/Orin as separate evidence tiers. +- Avoid publishing support claims without non-skipped conditional evidence. + +## Recommended first implementation + +Implement Jetson AGX Orin 64 GB on JetPack 6.2 first. Add a self-hosted runner +with labels `self-hosted`, `linux`, `ARM64`, `jetson`, `orin`, and +`jetpack-6`. The first gate should run only CUDA discovery, a small matrix +operation, OBLITERATUS device selection, and the existing offloaded-surgery +probe. Add `bitsandbytes` only after the same runner proves NF4 quantization +works with the selected PyTorch/CUDA stack. + +## Evidence basis + +- `REF-JETSON-PYTORCH-INSTALL` records NVIDIA's Jetson-specific PyTorch install + path and states the packages are intended for specified JetPack versions. +- `REF-JETSON-PYTORCH-RELEASES` maps PyTorch releases to NVIDIA framework + containers/wheels and JetPack versions. +- `REF-JETPACK-62` records JetPack 6.2 as Jetson Linux 36.4.3 and CUDA + 12.6. +- `REF-JETPACK-7-DOWNLOADS` records the current JetPack 7.2.1 stack as Jetson + Linux 39.2.1, Ubuntu 24.04, and CUDA 13.2.1. +- `REF-UV-PYTORCH` explains that PyTorch uses separate accelerator indexes and + local-version builds such as CPU and CUDA variants. +- `REF-BITSANDBYTES-INSTALL` documents Linux `aarch64` CUDA support targets, + but OBLITERATUS should still require project-specific Jetson evidence before + claiming quantization support. diff --git a/.aiwg/research/quality-assessments/REF-BITSANDBYTES-INSTALL-assessment.yaml b/.aiwg/research/quality-assessments/REF-BITSANDBYTES-INSTALL-assessment.yaml new file mode 100644 index 0000000..335284b --- /dev/null +++ b/.aiwg/research/quality-assessments/REF-BITSANDBYTES-INSTALL-assessment.yaml @@ -0,0 +1,13 @@ +ref_id: REF-BITSANDBYTES-INSTALL +grade: LOW +baseline: LOW +source_type: library_documentation +upgrades: [] +downgrades: + - Library support matrix is not Jetson-specific OBLITERATUS runtime evidence. +allowed_language: + - "bitsandbytes documents Linux aarch64 CUDA support..." + - "Quantization should still be proven on the selected Jetson stack..." +forbidden_language: + - "bitsandbytes is confirmed for OBLITERATUS on Jetson..." + - "Jetson quantization support can be claimed without hardware evidence..." diff --git a/.aiwg/research/quality-assessments/REF-JETPACK-62-assessment.yaml b/.aiwg/research/quality-assessments/REF-JETPACK-62-assessment.yaml new file mode 100644 index 0000000..4928843 --- /dev/null +++ b/.aiwg/research/quality-assessments/REF-JETPACK-62-assessment.yaml @@ -0,0 +1,14 @@ +ref_id: REF-JETPACK-62 +grade: MODERATE +baseline: LOW +source_type: vendor_documentation +upgrades: + - Primary NVIDIA release notes for the recommended JetPack 6.2 target tier. +downgrades: + - Release notes establish platform stack contents, not OBLITERATUS runtime support. +allowed_language: + - "NVIDIA lists JetPack 6.2 as..." + - "JetPack 6.2 is an appropriate explicit platform tier..." +forbidden_language: + - "JetPack 6.2 support is certified..." + - "Generic ARM evidence proves this stack..." diff --git a/.aiwg/research/quality-assessments/REF-JETPACK-7-DOWNLOADS-assessment.yaml b/.aiwg/research/quality-assessments/REF-JETPACK-7-DOWNLOADS-assessment.yaml new file mode 100644 index 0000000..0cfa88a --- /dev/null +++ b/.aiwg/research/quality-assessments/REF-JETPACK-7-DOWNLOADS-assessment.yaml @@ -0,0 +1,13 @@ +ref_id: REF-JETPACK-7-DOWNLOADS +grade: MODERATE +baseline: LOW +source_type: vendor_documentation +upgrades: + - Current NVIDIA release/download page for JetPack 7 stack contents. +downgrades: + - Download page is not project runtime evidence. +allowed_language: + - "NVIDIA currently lists..." + - "JetPack 7 should be a separate evidence tier..." +forbidden_language: + - "JetPack 7 works with OBLITERATUS..." diff --git a/.aiwg/research/quality-assessments/REF-JETSON-PYTORCH-INSTALL-assessment.yaml b/.aiwg/research/quality-assessments/REF-JETSON-PYTORCH-INSTALL-assessment.yaml new file mode 100644 index 0000000..e388aef --- /dev/null +++ b/.aiwg/research/quality-assessments/REF-JETSON-PYTORCH-INSTALL-assessment.yaml @@ -0,0 +1,14 @@ +ref_id: REF-JETSON-PYTORCH-INSTALL +grade: MODERATE +baseline: LOW +source_type: vendor_documentation +upgrades: + - Primary vendor documentation for Jetson PyTorch installation. +downgrades: + - Vendor documentation can change without project-controlled reproducibility. +allowed_language: + - "NVIDIA documents..." + - "NVIDIA's install guide states..." +forbidden_language: + - "OBLITERATUS supports Jetson..." + - "This guarantees compatibility..." diff --git a/.aiwg/research/quality-assessments/REF-JETSON-PYTORCH-RELEASES-assessment.yaml b/.aiwg/research/quality-assessments/REF-JETSON-PYTORCH-RELEASES-assessment.yaml new file mode 100644 index 0000000..914dfb8 --- /dev/null +++ b/.aiwg/research/quality-assessments/REF-JETSON-PYTORCH-RELEASES-assessment.yaml @@ -0,0 +1,14 @@ +ref_id: REF-JETSON-PYTORCH-RELEASES +grade: MODERATE +baseline: LOW +source_type: vendor_documentation +upgrades: + - Primary vendor compatibility table for Jetson PyTorch releases. +downgrades: + - Release tables are necessary but not sufficient runtime evidence. +allowed_language: + - "NVIDIA maps..." + - "The compatibility table lists..." +forbidden_language: + - "All listed combinations work for OBLITERATUS..." + - "No hardware testing is needed..." diff --git a/.aiwg/research/quality-assessments/REF-UV-PYTORCH-assessment.yaml b/.aiwg/research/quality-assessments/REF-UV-PYTORCH-assessment.yaml new file mode 100644 index 0000000..9f2f867 --- /dev/null +++ b/.aiwg/research/quality-assessments/REF-UV-PYTORCH-assessment.yaml @@ -0,0 +1,13 @@ +ref_id: REF-UV-PYTORCH +grade: MODERATE +baseline: LOW +source_type: tool_documentation +upgrades: + - Primary tool documentation for uv PyTorch resolution behavior. +downgrades: + - Packaging behavior still needs validation in this repository's lock model. +allowed_language: + - "uv documents..." + - "PyTorch accelerator variants need explicit resolver handling..." +forbidden_language: + - "uv automatically solves Jetson packaging..." diff --git a/.aiwg/research/sources/REF-BITSANDBYTES-INSTALL.yaml b/.aiwg/research/sources/REF-BITSANDBYTES-INSTALL.yaml new file mode 100644 index 0000000..258908e --- /dev/null +++ b/.aiwg/research/sources/REF-BITSANDBYTES-INSTALL.yaml @@ -0,0 +1,10 @@ +id: REF-BITSANDBYTES-INSTALL +title: bitsandbytes Installation Guide +source_type: library_documentation +publisher: Hugging Face +url: https://huggingface.co/docs/bitsandbytes/installation +accessed_at: "2026-08-21T23:31:01Z" +relevant_claims: + - bitsandbytes supports NVIDIA CUDA GPUs with compute capability 6.0 or newer. + - Linux aarch64 CUDA builds are documented for CUDA Toolkit 11.8 through 13.2 with specific SM targets. + - LLM.int8 requires Turing-class or newer hardware, while NF4/FP4 quantization requires Pascal-class or newer hardware. diff --git a/.aiwg/research/sources/REF-JETPACK-62.yaml b/.aiwg/research/sources/REF-JETPACK-62.yaml new file mode 100644 index 0000000..d83b272 --- /dev/null +++ b/.aiwg/research/sources/REF-JETPACK-62.yaml @@ -0,0 +1,10 @@ +id: REF-JETPACK-62 +title: JetPack 6.2 Release Notes +source_type: vendor_documentation +publisher: NVIDIA +url: https://docs.nvidia.com/jetson/archives/jetpack-archived/jetpack-62/release-notes/index.html +accessed_at: "2026-08-21T23:31:01Z" +relevant_claims: + - JetPack 6.2 includes Jetson Linux 36.4.3. + - JetPack 6.2 includes a compute stack with CUDA 12.6, TensorRT 10.3, cuDNN 9.3, VPI 3.2, DLA 3.1, and DLFW 24.0. + - JetPack 6.2 targets Jetson Orin modules and includes updated Super Mode behavior for Orin Nano and Orin NX modules. diff --git a/.aiwg/research/sources/REF-JETPACK-7-DOWNLOADS.yaml b/.aiwg/research/sources/REF-JETPACK-7-DOWNLOADS.yaml new file mode 100644 index 0000000..8fdad1b --- /dev/null +++ b/.aiwg/research/sources/REF-JETPACK-7-DOWNLOADS.yaml @@ -0,0 +1,10 @@ +id: REF-JETPACK-7-DOWNLOADS +title: NVIDIA JetPack SDK Downloads and Notes +source_type: vendor_documentation +publisher: NVIDIA +url: https://developer.nvidia.com/embedded/jetpack/downloads +accessed_at: "2026-08-21T23:31:01Z" +relevant_claims: + - The current JetPack 7.2.1 release is paired with Jetson Linux 39.2.1. + - JetPack 7.2.1 lists CUDA 13.2.1, TensorRT 10.16.2, and cuDNN 9.20.0. + - JetPack 7 uses an Ubuntu 24.04 L4T base and aligns Jetson software with SBSA. diff --git a/.aiwg/research/sources/REF-JETSON-PYTORCH-INSTALL.yaml b/.aiwg/research/sources/REF-JETSON-PYTORCH-INSTALL.yaml new file mode 100644 index 0000000..674e4bc --- /dev/null +++ b/.aiwg/research/sources/REF-JETSON-PYTORCH-INSTALL.yaml @@ -0,0 +1,11 @@ +id: REF-JETSON-PYTORCH-INSTALL +title: Installing PyTorch for Jetson Platform +source_type: vendor_documentation +publisher: NVIDIA +url: https://docs.nvidia.com/deeplearning/frameworks/install-pytorch-jetson-platform/index.html +accessed_at: "2026-08-21T23:31:01Z" +relevant_claims: + - NVIDIA provides Jetson PyTorch pip wheels with GPU acceleration and cuDNN support. + - The packages are intended to be installed on top of a specified JetPack version. + - Installation prerequisites include JetPack on the Jetson device and system packages. + - PyTorch installation verification starts by importing torch on the Jetson platform. diff --git a/.aiwg/research/sources/REF-JETSON-PYTORCH-RELEASES.yaml b/.aiwg/research/sources/REF-JETSON-PYTORCH-RELEASES.yaml new file mode 100644 index 0000000..25eb9d9 --- /dev/null +++ b/.aiwg/research/sources/REF-JETSON-PYTORCH-RELEASES.yaml @@ -0,0 +1,11 @@ +id: REF-JETSON-PYTORCH-RELEASES +title: PyTorch for Jetson Platform Release Notes +source_type: vendor_documentation +publisher: NVIDIA +url: https://docs.nvidia.com/deeplearning/frameworks/install-pytorch-jetson-platform-release-notes/pytorch-jetson-rel.html +accessed_at: "2026-08-21T23:31:01Z" +relevant_claims: + - NVIDIA's compatibility table maps PyTorch versions to NVIDIA framework containers or wheels and JetPack versions. + - JetPack 6.2 entries map to NVIDIA framework containers 25.02 through 25.06. + - JetPack 7.x entries map to NVIDIA framework containers 25.08 and later. + - NVIDIA notes that standalone iGPU containers are no longer produced starting with the 26.03 release. diff --git a/.aiwg/research/sources/REF-UV-PYTORCH.yaml b/.aiwg/research/sources/REF-UV-PYTORCH.yaml new file mode 100644 index 0000000..5aaa54c --- /dev/null +++ b/.aiwg/research/sources/REF-UV-PYTORCH.yaml @@ -0,0 +1,11 @@ +id: REF-UV-PYTORCH +title: Using uv with PyTorch +source_type: tool_documentation +publisher: Astral +url: https://docs.astral.sh/uv/guides/integration/pytorch/ +accessed_at: "2026-08-21T23:31:01Z" +relevant_claims: + - uv can manage PyTorch dependencies while controlling accelerator selection. + - PyTorch wheels use dedicated indexes outside PyPI for many builds. + - PyTorch encodes accelerator builds in local version specifiers such as +cpu and +cu130. + - Different PyTorch accelerator builds are published on different indexes. diff --git a/docs/conditional-testing.md b/docs/conditional-testing.md index 10dc0ec..655e3da 100644 --- a/docs/conditional-testing.md +++ b/docs/conditional-testing.md @@ -84,7 +84,7 @@ Jetson CUDA support is tracked separately from this generic x64 CUDA lane. A generic Linux ARM build can prove package portability, but it does not prove Jetson GPU support because Jetson depends on a JetPack/L4T-matched CUDA, cuDNN, and PyTorch runtime. The support plan, recommended container path, and acceptance -criteria are documented in [NVIDIA Jetson support plan](jetson-support.md). +criteria are documented in [NVIDIA Jetson support plan](platforms/jetson.md). ## Apple MPS and MLX diff --git a/docs/platforms/jetson.md b/docs/platforms/jetson.md new file mode 100644 index 0000000..e3edd7c --- /dev/null +++ b/docs/platforms/jetson.md @@ -0,0 +1,136 @@ +# NVIDIA Jetson support plan + +Issue: https://github.com/elder-plinius/OBLITERATUS/issues/31 + +## Status + +Native Jetson AGX support is not claimed yet. OBLITERATUS should treat Jetson +as a dedicated conditional runtime lane, not as part of the default pull-request +gate and not as a generic Linux ARM build. + +The current OBLITERATUS CUDA path delegates discovery to PyTorch through +`torch.cuda.is_available()`. If a Jetson AGX host reports no CUDA inside +OBLITERATUS, the first thing to verify is the JetPack/L4T/PyTorch/container +stack, because NVIDIA publishes Jetson-specific PyTorch builds intended for +specified JetPack versions. + +## Decision + +Support Jetson through a JetPack-pinned runtime contract: + +- Keep ordinary PR CI CPU-only, offline, and architecture-neutral. +- Add a Jetson conditional gate once a Jetson runner is available. +- Prefer an NVIDIA-supported Jetson PyTorch container or NVIDIA Jetson PyTorch + wheel for the exact JetPack release under test. +- Do not use the existing x64 CUDA gate as Jetson evidence. +- Do not treat a generic `linux/arm64` build as evidence that CUDA works on + Jetson. + +## Why generic ARM is insufficient + +Jetson support couples at least five moving pieces: + +- Jetson hardware family and compute capability. +- JetPack version. +- Jetson Linux/L4T version and Ubuntu base image. +- CUDA, cuDNN, TensorRT, and related NVIDIA libraries. +- PyTorch build or container version. + +NVIDIA's Jetson PyTorch documentation says the PyTorch packages are installed +on top of a specified JetPack version, and the compatibility table maps PyTorch +versions to NVIDIA framework containers/wheels and JetPack versions. A generic +ARM build can prove that Python code imports on `aarch64`; it cannot prove that +CUDA, cuDNN, TensorRT, or PyTorch CUDA dispatch works on Jetson. + +## Initial support matrix + +Start with the hardware reported in issue #31: Jetson AGX devices with 64 GB +unified memory. + +Recommended first support tier: + +| Tier | Hardware | JetPack | OS / CUDA baseline | Evidence requirement | +| --- | --- | --- | --- | --- | +| Target | Jetson AGX Orin 64 GB | 6.2 | Jetson Linux 36.4.3 / CUDA 12.6 | Native Jetson runner or NVIDIA Jetson container on Jetson hardware | +| Evaluate | Jetson AGX Thor | 7.x | Jetson Linux 38/39 / Ubuntu 24.04 / CUDA 13.x family | Separate runner and issue before claiming support | +| Legacy | Jetson AGX Xavier | 5.1.x | Jetson Linux 35.x / Ubuntu 20.04 / CUDA 11.x family | Defer unless a maintainer/user provides hardware and demand | + +Do not collapse these tiers into one "ARM64" support claim. + +## Installation shape + +The generic local Dockerfile uses `python:3.11-slim` and is not the Jetson +runtime image. A Jetson runtime should use one of these approaches: + +1. Start from an NVIDIA Jetson-compatible PyTorch framework container for the + selected JetPack version, then install OBLITERATUS without replacing the + container's validated PyTorch stack. +2. On a flashed Jetson host, install the NVIDIA Jetson PyTorch wheel matching + the installed JetPack release, then install OBLITERATUS in a virtual + environment without allowing dependency resolution to replace `torch`. + +The lock policy should make the Jetson torch source explicit. The existing +Linux PR lock intentionally uses CPU-only PyTorch. A Jetson install path needs +an override or separate constraints file that preserves NVIDIA's Jetson PyTorch +runtime. + +## Conditional gate + +Add a new gate instead of modifying the x64 CUDA gate: + +- Gate id: `jetson-runtime` +- Runner labels: `self-hosted`, `linux`, `ARM64`, `jetson` +- Optional labels by tier: `orin`, `jetpack-6` or `thor`, `jetpack-7` +- Trigger: manual dispatch and release/scheduled validation only +- Evidence retention: same 30-day conditional-evidence policy as other hardware + gates + +The gate should verify: + +- `platform.machine()` is `aarch64` or equivalent ARM64. +- `torch.cuda.is_available()` is true. +- `torch.version.cuda` is not `None`. +- `torch.cuda.get_device_name(0)` identifies the Jetson GPU class. +- OBLITERATUS resolves `device=auto` to `cuda`. +- A small CUDA tensor operation completes with finite output. +- The existing offloaded-surgery CUDA probe passes. +- `bitsandbytes` NF4/4-bit quantization is either proven on that exact Jetson + stack or documented as unsupported for the tier. +- A tiny Hugging Face model run passes only when the model-download gate is + explicitly selected and the runner has the required account/cache policy. + +## Acceptance criteria + +Jetson support can be claimed for a tier only after all of the following are +true: + +- The supported Jetson module and JetPack version are named in this document. +- The install instructions pin the JetPack-compatible PyTorch container or + wheel source. +- The Jetson conditional gate produces non-skipped green evidence on the exact + commit being claimed. +- The release notes distinguish generic ARM importability from Jetson CUDA + support. +- The docs state memory expectations for 64 GB unified memory and recommend + small models for validation before large ablation runs. + +## Operational notes + +Jetson's unified memory is shared by the OS, CUDA, model weights, activations, +and file cache. Treat "64 GB" as a capacity class, not guaranteed usable model +memory. Use small models for smoke tests, then move larger GPU validation to +dedicated CUDA hosts such as Titan when those resources are available. + +Logging into a Hugging Face account is expected only for gated/private models, +license-gated models, or rate-limit avoidance. It should not be required for the +offline CPU PR gate or for the Jetson CUDA hardware probe. Any model-download +validation must remain an explicit conditional gate. + +## Sources + +- `REF-JETSON-PYTORCH-INSTALL`: NVIDIA, Installing PyTorch for Jetson Platform. +- `REF-JETSON-PYTORCH-RELEASES`: NVIDIA, PyTorch for Jetson Platform release notes. +- `REF-JETPACK-62`: NVIDIA, JetPack 6.2 release notes. +- `REF-JETPACK-7-DOWNLOADS`: NVIDIA, JetPack SDK downloads and notes. +- `REF-UV-PYTORCH`: Astral, Using uv with PyTorch. +- `REF-BITSANDBYTES-INSTALL`: Hugging Face, bitsandbytes installation guide. From 1e870ccee08a358058956f43e3546a8e403bea70 Mon Sep 17 00:00:00 2001 From: Joseph Magly <1159087+jmagly@users.noreply.github.com> Date: Fri, 21 Aug 2026 19:40:38 -0400 Subject: [PATCH 3/4] docs: harden Jetson runtime decision --- .../adr/ADR-001-jetson-runtime-support.md | 95 +++++++++++ .../jetson-support-architecture.md | 71 --------- .../REF-BITSANDBYTES-INSTALL-assessment.yaml | 5 +- .../REF-GITHUB-SELF-HOSTED-assessment.yaml | 13 ++ .../sources/REF-BITSANDBYTES-INSTALL.yaml | 3 +- .../sources/REF-GITHUB-SELF-HOSTED.yaml | 10 ++ .aiwg/testing/jetson-support-assessment.md | 24 ++- docs/jetson-support.md | 147 ------------------ docs/platforms/jetson.md | 46 ++++-- 9 files changed, 168 insertions(+), 246 deletions(-) create mode 100644 .aiwg/architecture/adr/ADR-001-jetson-runtime-support.md delete mode 100644 .aiwg/architecture/jetson-support-architecture.md create mode 100644 .aiwg/research/quality-assessments/REF-GITHUB-SELF-HOSTED-assessment.yaml create mode 100644 .aiwg/research/sources/REF-GITHUB-SELF-HOSTED.yaml delete mode 100644 docs/jetson-support.md diff --git a/.aiwg/architecture/adr/ADR-001-jetson-runtime-support.md b/.aiwg/architecture/adr/ADR-001-jetson-runtime-support.md new file mode 100644 index 0000000..5060166 --- /dev/null +++ b/.aiwg/architecture/adr/ADR-001-jetson-runtime-support.md @@ -0,0 +1,95 @@ +# ADR-001: Use a JetPack-pinned runtime for Jetson support + +Status: Proposed +Date: 2026-08-21 +Issue: https://github.com/elder-plinius/OBLITERATUS/issues/31 +Public plan: [docs/platforms/jetson.md](../../../docs/platforms/jetson.md) + +## Context + +OBLITERATUS currently resolves CPU-only PyTorch for Linux through its default +uv lock. Its hardware CUDA gate is explicitly x64 and replaces that locked +package with an upstream cu130 build. Application device discovery already +delegates to `torch.cuda.is_available()`. + +Jetson is therefore not primarily an application detection change. It is a +platform-runtime problem: the Jetson module, JetPack/L4T release, Python, +CUDA-family libraries, and PyTorch build or container must be compatible. A +generic ARM64 build can detect Python packaging defects but cannot establish +that this runtime works on Jetson hardware. + +## Options considered + +| Option | Benefit | Limitation | Decision | +| --- | --- | --- | --- | +| Generic ARM64 build only | Cheap packaging and CPU portability signal | Exercises no Jetson CUDA, L4T, or device memory behavior | Keep as preflight only | +| Native Jetson virtual environment | Direct and simple hardware validation | Host packages drift and the default lock can replace vendor PyTorch | Supported fallback | +| JetPack-compatible container on Jetson hardware | Reproducible runtime boundary and retained image provenance | Requires per-JetPack maintenance and a physical Jetson runner | Preferred | +| Build PyTorch and the complete CUDA stack from source | Maximum version control | High build cost and long-term platform maintenance burden | Reject for the initial tier | + +`jetson-containers` may accelerate prototyping, but it is a community project, +not the project's vendor trust anchor. A supported image must ultimately pin +audited base-image and package provenance. + +## Decision + +1. Keep the default PR environment and lock CPU-only. +2. Treat generic ARM64 CI as packaging preflight, never Jetson support evidence. +3. Define one exact initial matrix only after a runner is available: Jetson + module, JetPack patch, L4T, Python, PyTorch image or wheel, CUDA, cuDNN, + TensorRT, and bitsandbytes status. +4. Prefer an image derived from a JetPack-compatible NVIDIA PyTorch runtime and + run it on physical Jetson hardware. Preserve the image digest and binary + hashes in release evidence. +5. Add a separate `jetson-runtime` conditional gate. Its exact base labels are + `self-hosted`, `linux`, `ARM64`, and `jetson`; tier labels identify the module + and JetPack major version. +6. Run the hardware lane only by trusted manual, scheduled, or release + dispatch. Never execute untrusted pull-request code on the persistent + self-hosted Jetson runner. +7. Do not install the generic PyPI Linux-aarch64 bitsandbytes wheel on Jetson. + Upstream documents that wheel as SBSA/server ARM and requires a Jetson source + build. Quantization remains unsupported until a pinned source build passes + the exact hardware gate. + +## Consequences + +- Jetson support will have a narrower, explicit matrix rather than a broad ARM + claim. +- The implementation needs a Jetson-specific constraints/profile boundary so + `uv sync` cannot replace vendor PyTorch or select an incompatible + bitsandbytes wheel. +- Cross-building an ARM64 image may shorten image assembly, but only an + on-device run supplies CUDA support evidence. +- Vendor upgrades become release-affecting changes and must produce fresh gate + evidence. + +## Delivery and migration + +1. Inventory the available Jetson and record its exact JetPack/L4T stack. +2. Select and pin the initial Orin/JetPack 6.2.x matrix at an exact patch. +3. Add generic ARM64 package/import preflight without changing default PR + dependencies. +4. Add the Jetson constraints/profile and preferred container recipe. +5. Add the `jetson-runtime` policy entry, test probe, trusted workflow job, and + retained evidence artifact. +6. Validate CUDA discovery, device selection, a tiny CUDA operation, and the + offloaded-surgery probe. Validate source-built bitsandbytes separately. +7. Claim support only for matrices with fresh green evidence on the exact + release commit. + +## Rollback + +If the Jetson lane regresses, disable the affected matrix entry and remove its +support claim while retaining the generic ARM64 preflight. Revert the +Jetson-specific image/profile independently; do not change or weaken the +default CPU lock or x64 CUDA gate. Restore a matrix only after fresh physical +Jetson evidence passes. + +## Review record + +The 2026-08-21 review panel covered NVIDIA/platform compatibility, +build/maintainability, test/release evidence, and security/supply chain. It +required the Jetson-specific bitsandbytes source-build rule, physical-hardware +container evidence, one canonical public plan, exact gate naming, and an +untrusted-code restriction for the self-hosted runner. diff --git a/.aiwg/architecture/jetson-support-architecture.md b/.aiwg/architecture/jetson-support-architecture.md deleted file mode 100644 index c0433e5..0000000 --- a/.aiwg/architecture/jetson-support-architecture.md +++ /dev/null @@ -1,71 +0,0 @@ -# Jetson support architecture note - -Date: 2026-08-21 -Issue: https://github.com/elder-plinius/OBLITERATUS/issues/31 - -## Summary - -OBLITERATUS should support NVIDIA Jetson AGX as a JetPack-specific conditional -runtime target. A generic ARM build is useful as import/build coverage, but it -is not sufficient for Jetson CUDA support because the working runtime depends on -JetPack, L4T, CUDA, cuDNN, TensorRT, and NVIDIA's Jetson-compatible PyTorch -build or container. - -## Current repo state - -- `obliteratus/device.py` uses `torch.cuda.is_available()` for CUDA discovery. -- `docs/conditional-testing.md` defines hardware gates outside the mandatory - PR workflow. -- `.github/workflows/conditional-tests.yml` has an x64 CUDA gate using - `UV_TORCH_BACKEND=cu130`. -- `pyproject.toml` locks Linux PR resolution to CPU-only PyTorch through the - `pytorch-cpu` index. -- `Dockerfile` is a generic local `python:3.11-slim` image, not a Jetson image. - -These choices are correct for contributor PR turnaround, but they do not create -a Jetson-compatible runtime. - -## Required changes before claiming support - -1. Add a `jetson-runtime` conditional gate and policy entry. -2. Add a Jetson install path that preserves NVIDIA's Jetson PyTorch stack. -3. Add a Jetson container recipe or documented base-image override for the - selected JetPack tier. -4. Add a tiny hardware probe that records JetPack/L4T, CUDA, PyTorch, device - name, and OBLITERATUS `device=auto` behavior. -5. Decide whether `bitsandbytes` is supported, tier-limited, or disabled on the - selected Jetson stack based on hardware evidence. - -## Design constraints - -- Keep PR CI CPU-only and offline. -- Keep x64 CUDA and Jetson CUDA evidence separate. -- Do not let `uv sync` replace NVIDIA's Jetson PyTorch wheel/container runtime - with the generic CPU lock. -- Treat JetPack 6.x Orin and JetPack 7.x Thor/Orin as separate evidence tiers. -- Avoid publishing support claims without non-skipped conditional evidence. - -## Recommended first implementation - -Implement Jetson AGX Orin 64 GB on JetPack 6.2 first. Add a self-hosted runner -with labels `self-hosted`, `linux`, `ARM64`, `jetson`, `orin`, and -`jetpack-6`. The first gate should run only CUDA discovery, a small matrix -operation, OBLITERATUS device selection, and the existing offloaded-surgery -probe. Add `bitsandbytes` only after the same runner proves NF4 quantization -works with the selected PyTorch/CUDA stack. - -## Evidence basis - -- `REF-JETSON-PYTORCH-INSTALL` records NVIDIA's Jetson-specific PyTorch install - path and states the packages are intended for specified JetPack versions. -- `REF-JETSON-PYTORCH-RELEASES` maps PyTorch releases to NVIDIA framework - containers/wheels and JetPack versions. -- `REF-JETPACK-62` records JetPack 6.2 as Jetson Linux 36.4.3 and CUDA - 12.6. -- `REF-JETPACK-7-DOWNLOADS` records the current JetPack 7.2.1 stack as Jetson - Linux 39.2.1, Ubuntu 24.04, and CUDA 13.2.1. -- `REF-UV-PYTORCH` explains that PyTorch uses separate accelerator indexes and - local-version builds such as CPU and CUDA variants. -- `REF-BITSANDBYTES-INSTALL` documents Linux `aarch64` CUDA support targets, - but OBLITERATUS should still require project-specific Jetson evidence before - claiming quantization support. diff --git a/.aiwg/research/quality-assessments/REF-BITSANDBYTES-INSTALL-assessment.yaml b/.aiwg/research/quality-assessments/REF-BITSANDBYTES-INSTALL-assessment.yaml index 335284b..db71a09 100644 --- a/.aiwg/research/quality-assessments/REF-BITSANDBYTES-INSTALL-assessment.yaml +++ b/.aiwg/research/quality-assessments/REF-BITSANDBYTES-INSTALL-assessment.yaml @@ -6,8 +6,9 @@ upgrades: [] downgrades: - Library support matrix is not Jetson-specific OBLITERATUS runtime evidence. allowed_language: - - "bitsandbytes documents Linux aarch64 CUDA support..." - - "Quantization should still be proven on the selected Jetson stack..." + - "bitsandbytes documents generic Linux aarch64 CUDA wheels for SBSA/server ARM..." + - "Jetson requires a pinned source build that must be proven on the selected stack..." forbidden_language: - "bitsandbytes is confirmed for OBLITERATUS on Jetson..." - "Jetson quantization support can be claimed without hardware evidence..." + - "The generic Linux aarch64 wheel is compatible with Jetson..." diff --git a/.aiwg/research/quality-assessments/REF-GITHUB-SELF-HOSTED-assessment.yaml b/.aiwg/research/quality-assessments/REF-GITHUB-SELF-HOSTED-assessment.yaml new file mode 100644 index 0000000..9b05249 --- /dev/null +++ b/.aiwg/research/quality-assessments/REF-GITHUB-SELF-HOSTED-assessment.yaml @@ -0,0 +1,13 @@ +ref_id: REF-GITHUB-SELF-HOSTED +grade: MODERATE +baseline: LOW +source_type: platform_documentation +upgrades: + - Primary GitHub documentation for the runner trust boundary used by the proposed workflow. +downgrades: + - Documentation establishes platform risk, not proof of this repository's future runner configuration. +allowed_language: + - "GitHub warns that persistent self-hosted runners can be compromised by untrusted workflow code..." + - "The Jetson lane must be limited to trusted refs and maintainer-controlled dispatch..." +forbidden_language: + - "A runner label alone safely isolates untrusted pull requests..." diff --git a/.aiwg/research/sources/REF-BITSANDBYTES-INSTALL.yaml b/.aiwg/research/sources/REF-BITSANDBYTES-INSTALL.yaml index 258908e..2cf8a89 100644 --- a/.aiwg/research/sources/REF-BITSANDBYTES-INSTALL.yaml +++ b/.aiwg/research/sources/REF-BITSANDBYTES-INSTALL.yaml @@ -6,5 +6,6 @@ url: https://huggingface.co/docs/bitsandbytes/installation accessed_at: "2026-08-21T23:31:01Z" relevant_claims: - bitsandbytes supports NVIDIA CUDA GPUs with compute capability 6.0 or newer. - - Linux aarch64 CUDA builds are documented for CUDA Toolkit 11.8 through 13.2 with specific SM targets. + - Linux aarch64 CUDA wheels are documented for CUDA Toolkit 11.8 through 13.2 with specific SM targets. + - NVIDIA Jetson L4T/JetPack requires a source build; the published Linux aarch64 wheels target SBSA/server ARM and are not Jetson-compatible. - LLM.int8 requires Turing-class or newer hardware, while NF4/FP4 quantization requires Pascal-class or newer hardware. diff --git a/.aiwg/research/sources/REF-GITHUB-SELF-HOSTED.yaml b/.aiwg/research/sources/REF-GITHUB-SELF-HOSTED.yaml new file mode 100644 index 0000000..36027c2 --- /dev/null +++ b/.aiwg/research/sources/REF-GITHUB-SELF-HOSTED.yaml @@ -0,0 +1,10 @@ +id: REF-GITHUB-SELF-HOSTED +title: Self-hosted runners reference +source_type: platform_documentation +publisher: GitHub +url: https://docs.github.com/en/actions/reference/security/secure-use#hardening-for-self-hosted-runners +accessed_at: "2026-08-21T23:55:00Z" +relevant_claims: + - GitHub warns that self-hosted runners do not provide clean, ephemeral isolation for every job. + - GitHub recommends using self-hosted runners only with private repositories because repository forks can execute dangerous code through pull requests. + - A persistent Jetson runner must not execute untrusted pull-request code. diff --git a/.aiwg/testing/jetson-support-assessment.md b/.aiwg/testing/jetson-support-assessment.md index 685e2f5..71aef40 100644 --- a/.aiwg/testing/jetson-support-assessment.md +++ b/.aiwg/testing/jetson-support-assessment.md @@ -120,7 +120,8 @@ Purpose: prove the runtime actually works on a Jetson device. Environment: -- self-hosted Linux ARM64 Jetson runner; +- self-hosted Linux ARM64 Jetson runner with the exact labels `self-hosted`, + `linux`, `ARM64`, and `jetson`; - JetPack installed and active; - NVIDIA runtime / CUDA stack available. @@ -131,7 +132,8 @@ Checks: - a tiny tensor operation on GPU; - a minimal model-load or model-adapter smoke if the feature requires it; - a short CUDA sample or equivalent library smoke; -- optional containerized run using the JetPack container on the target. +- optional containerized run using the JetPack-compatible container on that + Jetson hardware. Acceptance: @@ -168,13 +170,13 @@ Acceptance: Use separate labels so the workflow can route cleanly and fail obviously if the hardware is absent. -- `self-hosted, linux, arm64, jetson` -- optional refinement: `self-hosted, linux, arm64, jetson, orin` -- optional refinement: `self-hosted, linux, arm64, jetpack-` +- `self-hosted, linux, ARM64, jetson` +- optional refinement: `self-hosted, linux, ARM64, jetson, orin` +- optional refinement: `self-hosted, linux, ARM64, jetson, jetpack-` Keep the generic ARM64 preflight on a separate label such as: -- `self-hosted, linux, arm64, arm64-preflight` +- `self-hosted, linux, ARM64, arm64-preflight` This avoids accidentally treating a generic ARM machine as a Jetson target. @@ -187,13 +189,15 @@ mechanism for a physical-device lane. Jetson support should be considered ready only when all of these are true: 1. The default PR gate still passes with no Jetson dependency. -2. A dedicated Jetson job passes on real hardware or an equivalent JetPack - target container. +2. A dedicated `jetson-runtime` job passes on physical Jetson hardware, either + natively or inside the pinned JetPack-compatible container. 3. The exact supported JetPack version is documented. 4. The exact NVIDIA wheel/container provenance is pinned and hashed. 5. The repo has a regression test that fails if Jetson detection or the CUDA smoke is broken. 6. The release notes state the supported JetPack / CUDA / TensorRT matrix. +7. The workflow retains the named `conditional-jetson-` log and + environment artifact for 30 days against the exact candidate commit. ## Memory and thermal constraints @@ -245,6 +249,10 @@ Controls to require: - record the exact JetPack release, CUDA version, and TensorRT version in the evidence trail; - treat any vendor wheel update as a release-affecting change. +- build bitsandbytes from a pinned source revision for Jetson, record the build + inputs and hash, and never substitute the generic SBSA Linux-aarch64 wheel; +- never execute untrusted pull-request code on the persistent self-hosted + Jetson runner. NVIDIA’s TensorRT documentation explicitly notes that JetPack deployments must remain on a TensorRT 10.x release supported by the JetPack version. That means diff --git a/docs/jetson-support.md b/docs/jetson-support.md deleted file mode 100644 index 8722a1b..0000000 --- a/docs/jetson-support.md +++ /dev/null @@ -1,147 +0,0 @@ -# NVIDIA Jetson support plan - -Issue: https://github.com/elder-plinius/OBLITERATUS/issues/31 - -Internal testing assessment: -[.aiwg/testing/jetson-support-assessment.md](../.aiwg/testing/jetson-support-assessment.md) - -## Decision - -OBLITERATUS should treat Jetson AGX support as a distinct platform target, not -as a generic Linux ARM build. Generic `linux/aarch64` packaging can prove that -the Python package resolves and imports on ARM, but it does not prove CUDA on -Jetson because Jetson devices depend on JetPack/L4T-matched CUDA, cuDNN, and -PyTorch builds. - -The first supported target should be Jetson AGX Orin 64 GB on JetPack 6.x, run -through either NVIDIA's Jetson PyTorch instructions or a JetPack-compatible -container. Xavier-era JetPack 5 can be documented as best-effort until a runner -exists. - -## Evidence - -NVIDIA documents Jetson PyTorch as JetPack-specific pip wheels with GPU -acceleration and cuDNN support, installed on top of the matching JetPack -version: - -- https://docs.nvidia.com/deeplearning/frameworks/install-pytorch-jetson-platform/index.html -- https://catalog.ngc.nvidia.com/orgs/nvidia/containers/l4t-pytorch - -The NVIDIA NGC `l4t-pytorch` container is explicitly a Jetson/JetPack image and -requires matching the container tag to the installed JetPack/L4T release. The -`jetson-containers` project provides a practical composition path for PyTorch -and Transformers on Jetson, including compatible autotag selection: - -- https://github.com/dusty-nv/jetson-containers - -PyTorch upstream support is moving: PyTorch 2.11 release notes say PyPI wheels -now include CUDA 13.0 for Linux aarch64, while CUDA 12.6 and 12.8 wheels remain -available from `download.pytorch.org`. That improves generic ARM CUDA, but it -does not remove the need to match Jetson's installed JetPack/CUDA stack: - -- https://github.com/pytorch/pytorch/releases - -bitsandbytes currently publishes Linux aarch64 CUDA builds and lists targets -for CUDA 11.8 through 13.x. AGX Orin's Ampere GPU should satisfy the compute -capability requirement for 4-bit and 8-bit quantization, but the runtime still -has to be validated on the Jetson image actually used by contributors: - -- https://huggingface.co/docs/bitsandbytes/en/installation - -## Current repository gaps - -- `pyproject.toml` maps Linux and Windows Torch resolution to the CPU-only - PyTorch index. That is correct for default CI, but it means the locked default - environment will not discover Jetson CUDA. -- `uv.lock` includes a Linux aarch64 bitsandbytes wheel, but Torch is locked as - `2.13.0+cpu` on Linux. A Jetson runtime must intentionally override Torch - after syncing the default lock, similar to the existing CUDA conditional lane. -- `.github/workflows/conditional-tests.yml` has one CUDA lane labeled - `self-hosted, linux, x64, cuda` and installs an official `cu130` Torch build. - It cannot run on Jetson because the runner labels and Torch source are x64 - CUDA-specific. -- `docs/conditional-testing.md` documents generic CUDA and bitsandbytes probes, - not Jetson labels, JetPack requirements, or container startup. -- `Dockerfile` is based on `python:3.11-slim`, installs `requirements.txt`, and - is documented as local-only. It is not a Jetson image and should not be used - for a support claim. -- `obliteratus/device.py` can already detect CUDA through PyTorch, so no special - Jetson detection is needed for the first milestone. The blocker is installing - a CUDA-enabled Jetson PyTorch runtime and proving the existing device and - bitsandbytes contracts there. -- `tests/conditional/test_cuda_runtime.py` is a good starting probe, but it does - not record Jetson-specific platform evidence such as JetPack/L4T release, - CUDA version, Python version, architecture, Torch build, and bitsandbytes - binary availability. - -## Implementation options - -### Option A: Native Jetson runner, minimal repo changes - -Attach a self-hosted runner on the Jetson with labels such as -`self-hosted`, `linux`, `ARM64`, `jetson`, and `cuda`. Add a new conditional -gate `jetson-cuda-runtime` that: - -- syncs OBLITERATUS without replacing the system/vendor Jetson Torch build; -- verifies `platform.machine()` is `aarch64`; -- records `/etc/nv_tegra_release`, `nvcc --version`, `torch.__version__`, - `torch.version.cuda`, `torch.cuda.get_device_name(0)`, and - `bitsandbytes` import/runtime status; -- runs the existing CUDA and bitsandbytes probes. - -Tradeoff: lowest abstraction and fastest to validate real hardware, but it -requires maintaining the Jetson's host Python/runtime state carefully. - -### Option B: Jetson container lane - -Base a Jetson-specific image on an NVIDIA L4T PyTorch image or build it with -`jetson-containers` from `pytorch` plus `transformers`. Install OBLITERATUS into -that image with dependency constraints that do not replace the container's -Torch. Run the same `jetson-cuda-runtime` probe inside the container. - -Tradeoff: best reproducibility and closest to the reporter's concern about -patched/static libraries, but the image has to track JetPack/L4T tags and may -need per-JetPack dependency pins. - -### Option C: Generic Linux ARM build plus separate Jetson UAT - -Add a hosted or self-hosted Linux ARM packaging job that installs OBLITERATUS -with CPU Torch on aarch64, then keep Jetson CUDA as manual UAT evidence until a -stable runner exists. - -Tradeoff: useful packaging signal for ARM contributors, but it must not be -presented as Jetson GPU support because it does not exercise CUDA, cuDNN, -device placement, quantization kernels, or JetPack compatibility. - -## Recommended path - -Use Option B for the support claim and Option C as a cheap early-warning signal. -The container path matches Jetson's dependency model, avoids polluting the -default lock with Jetson-only Torch URLs, and gives contributors a repeatable -recipe. A native runner can still be used to execute the container and collect -evidence. Treat generic ARM64 as Tier 0/Tier 1 preflight only; require -real-device Jetson smoke and release validation before publishing a support -claim. - -## Acceptance criteria - -- `docs/conditional-testing.md` documents Jetson as a separate conditional gate - with supported JetPack versions, runner labels, and local/container commands. -- `ci/conditional-test-policy.json` includes a `jetson-cuda-runtime` gate with a - tracking issue and evidence freshness rule. -- `.github/workflows/conditional-tests.yml` has a Jetson-selected job that runs - only on a labeled Jetson runner or through a Jetson-compatible container. -- The gate records JetPack/L4T, architecture, Torch CUDA version, CUDA device - name, memory, and bitsandbytes status in JSON evidence. -- The gate runs without skips: - `python scripts/run_conditional_gate.py jetson-cuda-runtime`. -- README support language says Jetson AGX Orin is supported only for the - specific JetPack/runtime combinations that have fresh green conditional - evidence. - -## User guidance until support lands - -Do not install from the default OBLITERATUS lock and expect Jetson CUDA to work. -Start from a JetPack-compatible PyTorch runtime, then install OBLITERATUS -without replacing Torch. If using a container, prefer an L4T PyTorch image or a -`jetson-containers` build that matches the device's JetPack/L4T release. diff --git a/docs/platforms/jetson.md b/docs/platforms/jetson.md index e3edd7c..4be9af7 100644 --- a/docs/platforms/jetson.md +++ b/docs/platforms/jetson.md @@ -42,17 +42,19 @@ versions to NVIDIA framework containers/wheels and JetPack versions. A generic ARM build can prove that Python code imports on `aarch64`; it cannot prove that CUDA, cuDNN, TensorRT, or PyTorch CUDA dispatch works on Jetson. -## Initial support matrix +## Proposed initial support matrix Start with the hardware reported in issue #31: Jetson AGX devices with 64 GB unified memory. -Recommended first support tier: +This is a planning target, not a current support claim. The implementation must +replace the JetPack family with the exact patch installed on the available +runner before publishing compatibility: | Tier | Hardware | JetPack | OS / CUDA baseline | Evidence requirement | | --- | --- | --- | --- | --- | -| Target | Jetson AGX Orin 64 GB | 6.2 | Jetson Linux 36.4.3 / CUDA 12.6 | Native Jetson runner or NVIDIA Jetson container on Jetson hardware | -| Evaluate | Jetson AGX Thor | 7.x | Jetson Linux 38/39 / Ubuntu 24.04 / CUDA 13.x family | Separate runner and issue before claiming support | +| Initial candidate | Jetson AGX Orin 64 GB | 6.2.x, exact patch TBD | L4T/CUDA values from the selected patch | Native Jetson runner or pinned compatible container on Jetson hardware | +| Evaluate later | Jetson AGX Thor | 7.x, exact release TBD | Select only after NVIDIA's PyTorch compatibility table covers the release | Separate runner and evidence before claiming support | | Legacy | Jetson AGX Xavier | 5.1.x | Jetson Linux 35.x / Ubuntu 20.04 / CUDA 11.x family | Defer unless a maintainer/user provides hardware and demand | Do not collapse these tiers into one "ARM64" support claim. @@ -72,7 +74,10 @@ runtime image. A Jetson runtime should use one of these approaches: The lock policy should make the Jetson torch source explicit. The existing Linux PR lock intentionally uses CPU-only PyTorch. A Jetson install path needs an override or separate constraints file that preserves NVIDIA's Jetson PyTorch -runtime. +runtime. It must also prevent the generic PyPI Linux-aarch64 bitsandbytes wheel +from being selected: upstream documents that wheel as SBSA/server ARM and says +Jetson L4T/JetPack requires a source build. Until a pinned source build passes +on the selected device, bitsandbytes is unsupported for that tier. ## Conditional gate @@ -85,6 +90,9 @@ Add a new gate instead of modifying the x64 CUDA gate: - Evidence retention: same 30-day conditional-evidence policy as other hardware gates +The job must run only from a trusted ref or reviewed maintainer dispatch. A +persistent self-hosted Jetson must never execute untrusted pull-request code. + The gate should verify: - `platform.machine()` is `aarch64` or equivalent ARM64. @@ -94,8 +102,8 @@ The gate should verify: - OBLITERATUS resolves `device=auto` to `cuda`. - A small CUDA tensor operation completes with finite output. - The existing offloaded-surgery CUDA probe passes. -- `bitsandbytes` NF4/4-bit quantization is either proven on that exact Jetson - stack or documented as unsupported for the tier. +- A pinned, source-built `bitsandbytes` NF4/4-bit path is proven on that exact + Jetson stack, or bitsandbytes is documented as unsupported for the tier. - A tiny Hugging Face model run passes only when the model-download gate is explicitly selected and the runner has the required account/cache policy. @@ -109,6 +117,9 @@ true: wheel source. - The Jetson conditional gate produces non-skipped green evidence on the exact commit being claimed. +- The named `jetson-runtime` workflow job uses the documented runner labels and + retains `conditional-jetson-` logs and environment metadata for + 30 days. - The release notes distinguish generic ARM importability from Jetson CUDA support. - The docs state memory expectations for 64 GB unified memory and recommend @@ -121,16 +132,17 @@ and file cache. Treat "64 GB" as a capacity class, not guaranteed usable model memory. Use small models for smoke tests, then move larger GPU validation to dedicated CUDA hosts such as Titan when those resources are available. -Logging into a Hugging Face account is expected only for gated/private models, -license-gated models, or rate-limit avoidance. It should not be required for the -offline CPU PR gate or for the Jetson CUDA hardware probe. Any model-download -validation must remain an explicit conditional gate. +The Jetson hardware probe does not require a Hugging Face login. Model download +testing remains the separate, explicitly selected `model-download-runtime` +gate; credentials are relevant only when that selected model itself requires +them. ## Sources -- `REF-JETSON-PYTORCH-INSTALL`: NVIDIA, Installing PyTorch for Jetson Platform. -- `REF-JETSON-PYTORCH-RELEASES`: NVIDIA, PyTorch for Jetson Platform release notes. -- `REF-JETPACK-62`: NVIDIA, JetPack 6.2 release notes. -- `REF-JETPACK-7-DOWNLOADS`: NVIDIA, JetPack SDK downloads and notes. -- `REF-UV-PYTORCH`: Astral, Using uv with PyTorch. -- `REF-BITSANDBYTES-INSTALL`: Hugging Face, bitsandbytes installation guide. +- [NVIDIA: Installing PyTorch for Jetson Platform](https://docs.nvidia.com/deeplearning/frameworks/install-pytorch-jetson-platform/index.html) +- [NVIDIA: PyTorch for Jetson compatibility table](https://docs.nvidia.com/deeplearning/frameworks/install-pytorch-jetson-platform-release-notes/pytorch-jetson-rel.html) +- [NVIDIA: JetPack 6.2 release notes](https://docs.nvidia.com/jetson/archives/jetpack-archived/jetpack-62/release-notes/index.html) +- [NVIDIA: current JetPack downloads and notes](https://developer.nvidia.com/embedded/jetpack/downloads) +- [Astral: Using uv with PyTorch](https://docs.astral.sh/uv/guides/integration/pytorch/) +- [Hugging Face: bitsandbytes installation guide](https://huggingface.co/docs/bitsandbytes/installation) +- [GitHub: secure use of self-hosted runners](https://docs.github.com/en/actions/reference/security/secure-use#hardening-for-self-hosted-runners) From 06e80af7b63b3eb55c33e9bd29b31671a3bb0e7a Mon Sep 17 00:00:00 2001 From: Joseph Magly <1159087+jmagly@users.noreply.github.com> Date: Fri, 21 Aug 2026 20:12:35 -0400 Subject: [PATCH 4/4] feat: add Jetson contributor validation path --- .../adr/ADR-001-jetson-runtime-support.md | 13 +- .aiwg/testing/jetson-support-assessment.md | 5 + .github/ISSUE_TEMPLATE/jetson-runtime.yml | 77 ++++++ .github/actionlint.yaml | 2 +- .github/workflows/ci.yml | 39 +++ .github/workflows/conditional-tests.yml | 58 +++- .gitignore | 1 + CONTRIBUTING.md | 11 +- README.md | 18 +- ci/conditional-test-policy.json | 13 + ci/pr-test-policy.json | 4 + ci/test-risk-map.json | 11 +- docs/conditional-testing.md | 23 +- docs/platforms/jetson.md | 67 +++++ pyproject.toml | 2 +- scripts/jetson_support.py | 254 ++++++++++++++++++ scripts/run_conditional_gate.py | 12 +- scripts/setup_jetson.py | 190 +++++++++++++ tests/conditional/test_jetson_runtime.py | 53 ++++ tests/test_ci_policy.py | 18 ++ tests/test_conditional_gate_scripts.py | 23 ++ tests/test_jetson_support_tooling.py | 254 ++++++++++++++++++ uv.lock | 10 +- 23 files changed, 1128 insertions(+), 30 deletions(-) create mode 100644 .github/ISSUE_TEMPLATE/jetson-runtime.yml create mode 100644 scripts/jetson_support.py create mode 100644 scripts/setup_jetson.py create mode 100644 tests/conditional/test_jetson_runtime.py create mode 100644 tests/test_jetson_support_tooling.py diff --git a/.aiwg/architecture/adr/ADR-001-jetson-runtime-support.md b/.aiwg/architecture/adr/ADR-001-jetson-runtime-support.md index 5060166..afa1cbc 100644 --- a/.aiwg/architecture/adr/ADR-001-jetson-runtime-support.md +++ b/.aiwg/architecture/adr/ADR-001-jetson-runtime-support.md @@ -5,6 +5,11 @@ Date: 2026-08-21 Issue: https://github.com/elder-plinius/OBLITERATUS/issues/31 Public plan: [docs/platforms/jetson.md](../../../docs/platforms/jetson.md) +Implementation status: generic ARM64 preflight, vendor-PyTorch-preserving +bootstrap, physical-device gate, sanitized evidence collector, manual runner +workflow, and contributor issue form are implemented. The ADR remains Proposed +until physical-device evidence fixes the initial supported matrix. + ## Context OBLITERATUS currently resolves CPU-only PyTorch for Linux through its default @@ -69,10 +74,12 @@ audited base-image and package provenance. 1. Inventory the available Jetson and record its exact JetPack/L4T stack. 2. Select and pin the initial Orin/JetPack 6.2.x matrix at an exact patch. 3. Add generic ARM64 package/import preflight without changing default PR - dependencies. -4. Add the Jetson constraints/profile and preferred container recipe. + dependencies. (Implemented.) +4. Add the Jetson constraints/profile and preferred container recipe. (The + native/container bootstrap boundary is implemented; an exact image remains + dependent on the selected hardware matrix.) 5. Add the `jetson-runtime` policy entry, test probe, trusted workflow job, and - retained evidence artifact. + retained evidence artifact. (Implemented.) 6. Validate CUDA discovery, device selection, a tiny CUDA operation, and the offloaded-surgery probe. Validate source-built bitsandbytes separately. 7. Claim support only for matrices with fresh green evidence on the exact diff --git a/.aiwg/testing/jetson-support-assessment.md b/.aiwg/testing/jetson-support-assessment.md index 71aef40..54233f4 100644 --- a/.aiwg/testing/jetson-support-assessment.md +++ b/.aiwg/testing/jetson-support-assessment.md @@ -260,6 +260,11 @@ the Jetson compatibility matrix must be version-pinned, not “latest by default ## Recommended implementation path +Repository status as of 2026-08-21: steps 2 through 4 are implemented with a +manual-only physical runner, a hosted ARM64 preflight, CPU-testable policy +contracts, and a sanitized contributor evidence flow. Physical Jetson results +are still required before completing steps 5 and 6 or publishing support. + 1. Keep issue #31 open as a feature/specification item until the acceptance matrix is merged. 2. Add a Jetson conditional policy entry and workflow job with an explicit diff --git a/.github/ISSUE_TEMPLATE/jetson-runtime.yml b/.github/ISSUE_TEMPLATE/jetson-runtime.yml new file mode 100644 index 0000000..eb5ea7e --- /dev/null +++ b/.github/ISSUE_TEMPLATE/jetson-runtime.yml @@ -0,0 +1,77 @@ +name: Jetson runtime report +description: Report a physical Jetson bootstrap, build, or CUDA validation result +title: "[Jetson] " +labels: + - enhancement +body: + - type: markdown + attributes: + value: | + Thank you for testing OBLITERATUS on physical Jetson hardware. Run the + documented `jetson-runtime` probe and attach its sanitized report. The + report intentionally omits host identity, network, serial, token, and + local-path data. + - type: dropdown + id: device + attributes: + label: Jetson device + options: + - Jetson AGX Orin + - Jetson Orin NX + - Jetson Orin Nano + - Jetson AGX Xavier + - Jetson AGX Thor + validations: + required: true + - type: input + id: jetpack + attributes: + label: Exact JetPack version + description: Include the patch version, for example 6.2.1. + validations: + required: true + - type: input + id: commit + attributes: + label: Exact OBLITERATUS commit + description: Paste the 40-character commit SHA that was tested. + validations: + required: true + - type: textarea + id: reproduction + attributes: + label: Reproduction commands + description: Provide the smallest command sequence that demonstrates the result. + validations: + required: true + - type: textarea + id: expected + attributes: + label: Expected result + validations: + required: true + - type: textarea + id: actual + attributes: + label: Actual result + description: Include the complete error text, but remove any secrets before submitting. + validations: + required: true + - type: textarea + id: evidence + attributes: + label: Sanitized Jetson evidence + description: Attach conditional-evidence/jetson-report.json or paste its JSON contents. + validations: + required: true + - type: checkboxes + id: confirmations + attributes: + label: Confirmations + options: + - label: I ran this validation on physical Jetson hardware. + required: true + - label: I reviewed the report and removed any secrets or personal identifiers. + required: true + validations: + required: true diff --git a/.github/actionlint.yaml b/.github/actionlint.yaml index a85f6a0..22ac661 100644 --- a/.github/actionlint.yaml +++ b/.github/actionlint.yaml @@ -1,6 +1,6 @@ self-hosted-runner: # Project-owned capability labels used by conditional test runners. - labels: [cuda, mps, mlx] + labels: [cuda, jetson, mps, mlx] # Configuration variables in array of strings defined in your repository or # organization. `null` means disabling configuration variables check. diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index e712d58..3354cbd 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -225,6 +225,45 @@ jobs: scripts/check_supply_chain_policy.py scripts/gemma4_12b_recursive_loop.py || true + arm64-preflight: + name: Linux ARM64 preflight + if: github.event_name == 'pull_request' || (github.event_name == 'push' && github.ref == 'refs/heads/main') + runs-on: ubuntu-24.04-arm + timeout-minutes: 10 + env: + CUDA_VISIBLE_DEVICES: "" + HF_DATASETS_OFFLINE: "1" + HF_HUB_DISABLE_TELEMETRY: "1" + HF_HUB_OFFLINE: "1" + TRANSFORMERS_OFFLINE: "1" + TEST_ENV: /tmp/obliteratus-arm64-preflight + steps: + - name: Check out repository + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + - name: Set up Python + uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 + with: + python-version: "3.12" + cache: pip + cache-dependency-path: | + pyproject.toml + uv.lock + - name: Install locked CPU runtime + run: | + python -m pip install "uv==${UV_VERSION}" + UV_PROJECT_ENVIRONMENT="$TEST_ENV" \ + uv sync --locked --no-default-groups --extra dev --no-editable + - name: Verify ARM64 package and core runtime + run: | + "$TEST_ENV/bin/python" -c \ + 'import platform; assert platform.machine().lower() in {"aarch64", "arm64"}' + "$TEST_ENV/bin/python" -m build --wheel + "$TEST_ENV/bin/python" -c 'import obliteratus; print(obliteratus.__version__)' + "$TEST_ENV/bin/python" -m obliteratus --help + "$TEST_ENV/bin/python" -m pytest \ + tests/test_module_imports.py tests/test_device_boundaries.py \ + tests/test_jetson_support_tooling.py -q --no-cov + pr-core: name: Pull request core if: github.event_name == 'pull_request' || (github.event_name == 'push' && github.ref == 'refs/heads/main') diff --git a/.github/workflows/conditional-tests.yml b/.github/workflows/conditional-tests.yml index 16442b8..c23f020 100644 --- a/.github/workflows/conditional-tests.yml +++ b/.github/workflows/conditional-tests.yml @@ -23,6 +23,10 @@ on: description: Run CUDA and bitsandbytes on the labeled self-hosted runner type: boolean default: false + run_jetson: + description: Run Jetson CUDA on the trusted labeled physical runner + type: boolean + default: false run_mps: description: Run MPS on the labeled Apple Silicon runner type: boolean @@ -239,7 +243,7 @@ jobs: run: | python -m pip install "uv==${UV_VERSION}" UV_PROJECT_ENVIRONMENT="$CONDITIONAL_ENV" \ - uv sync --locked --no-default-groups --extra dev --no-editable + uv sync --locked --no-default-groups --extra dev --extra quantization --no-editable CUDA_TORCH_VERSION="$("$CONDITIONAL_ENV/bin/python" -c \ 'import torch; print(torch.__version__.split("+", 1)[0])')" UV_TORCH_BACKEND=cu130 uv pip install \ @@ -264,6 +268,52 @@ jobs: if-no-files-found: error retention-days: 30 + jetson: + name: NVIDIA Jetson runtime + needs: policy + if: github.event_name == 'workflow_dispatch' && inputs.run_jetson + runs-on: [self-hosted, linux, ARM64, jetson] + timeout-minutes: 45 + env: + JETSON_ENV: /tmp/obliteratus-jetson-${{ github.run_id }}-${{ github.run_attempt }} + JETSON_TOOLS: /tmp/obliteratus-jetson-tools-${{ github.run_id }}-${{ github.run_attempt }} + steps: + - name: Check out trusted candidate + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + - name: Install pinned bootstrap tooling outside the JetPack runtime + run: | + python3 -m venv "$JETSON_TOOLS" + "$JETSON_TOOLS/bin/python" -m pip install "uv==${UV_VERSION}" + - name: Preserve JetPack PyTorch and install OBLITERATUS + run: >- + python3 scripts/setup_jetson.py + --python python3 + --uv-python "$JETSON_TOOLS/bin/python" + --venv "$JETSON_ENV" + - name: Run physical Jetson CUDA probe + run: >- + "$JETSON_ENV/bin/python" scripts/run_conditional_gate.py jetson-runtime + - name: Collect sanitized Jetson evidence + if: always() + run: | + JETSON_PYTHON=python3 + if [ -x "$JETSON_ENV/bin/python" ]; then + JETSON_PYTHON="$JETSON_ENV/bin/python" + fi + "$JETSON_PYTHON" scripts/jetson_support.py \ + --check \ + --gate-evidence conditional-evidence/jetson-runtime.json \ + --output conditional-evidence/jetson-report.json \ + --issue-body conditional-evidence/jetson-issue.md + - name: Upload Jetson evidence + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: conditional-jetson-${{ github.run_attempt }} + path: conditional-evidence/ + if-no-files-found: error + retention-days: 30 + mps: name: Apple MPS runtime needs: policy @@ -393,20 +443,22 @@ jobs: summary: name: Conditional result and freshness summary if: always() - needs: [policy, model_runtime, network_services, operator_ui, cuda, mps, mlx, remote] + needs: [policy, model_runtime, network_services, operator_ui, cuda, jetson, mps, mlx, remote] runs-on: ubuntu-latest timeout-minutes: 5 env: CONDITIONAL_RESULTS: >- {"policy":"${{ needs.policy.result }}","model_runtime":"${{ needs.model_runtime.result }}", "network_services":"${{ needs.network_services.result }}","operator_ui":"${{ needs.operator_ui.result }}", - "cuda":"${{ needs.cuda.result }}","mps":"${{ needs.mps.result }}","mlx":"${{ needs.mlx.result }}", + "cuda":"${{ needs.cuda.result }}","jetson":"${{ needs.jetson.result }}", + "mps":"${{ needs.mps.result }}","mlx":"${{ needs.mlx.result }}", "remote":"${{ needs.remote.result }}"} CONDITIONAL_SELECTED: >- {"model_runtime":${{ github.event_name != 'workflow_dispatch' || inputs.run_model }}, "network_services":${{ github.event_name != 'workflow_dispatch' || inputs.run_network }}, "operator_ui":${{ github.event_name != 'workflow_dispatch' || inputs.run_ui }}, "cuda":${{ (github.event_name == 'workflow_dispatch' && inputs.run_cuda) || (github.event_name != 'workflow_dispatch' && vars.ENABLE_CUDA_GATE == 'true') }}, + "jetson":${{ github.event_name == 'workflow_dispatch' && inputs.run_jetson }}, "mps":${{ (github.event_name == 'workflow_dispatch' && inputs.run_mps) || (github.event_name != 'workflow_dispatch' && vars.ENABLE_MPS_GATE == 'true') }}, "mlx":${{ (github.event_name == 'workflow_dispatch' && inputs.run_mlx) || (github.event_name != 'workflow_dispatch' && vars.ENABLE_MLX_GATE == 'true') }}, "remote":${{ (github.event_name == 'workflow_dispatch' && inputs.run_remote) || (github.event_name != 'workflow_dispatch' && vars.ENABLE_REMOTE_GATE == 'true') }}} diff --git a/.gitignore b/.gitignore index dc750a1..db7a058 100644 --- a/.gitignore +++ b/.gitignore @@ -8,6 +8,7 @@ build/ .eggs/ *.egg .venv/ +.venv-jetson/ venv/ env/ .env diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 03612c1..d1a1d2c 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -206,8 +206,8 @@ them in the managed environment. ## Conditional and hardware testing -GPU, MPS, MLX, model-download, external-evaluation, network, operator-UI, and -remote-execution checks are conditional release or risk-surface gates. A unit test +GPU, Jetson, MPS, MLX, model-download, external-evaluation, network, operator-UI, +and remote-execution checks are conditional release or risk-surface gates. A unit test with a mocked device is still required; hardware evidence complements deterministic contract coverage and never replaces it. @@ -222,6 +222,13 @@ Current operator hardware includes Titan for CUDA/bitsandbytes probes and Mutsu, machines are attached to a public pull-request workflow; CI or a maintainer will record whether the mapped conditional gate ran. +Jetson contributors do not need project-owned hardware access. Follow the +[Jetson contributor bootstrap](docs/platforms/jetson.md#experimental-contributor-bootstrap) +on a physical device, then submit the generated sanitized evidence through the +[Jetson runtime report](https://github.com/elder-plinius/OBLITERATUS/issues/new?template=jetson-runtime.yml). Maintainers +will reproduce, add missing test depth, and integrate compatible changes. Never +attach a contributor-controlled runner to untrusted pull-request execution. + ## Security and supply-chain expectations - Treat issue text, pull requests, patches, model repositories, checkpoints, diff --git a/README.md b/README.md index 9743dcd..84d20c5 100644 --- a/README.md +++ b/README.md @@ -162,6 +162,11 @@ obliteratus ui --auth user:pass # add basic auth The `obliteratus ui` command adds a Rich terminal startup with GPU detection and hardware-appropriate model recommendations. You can also run `python app.py` directly (same thing the Space uses). +Install `.[spaces,quantization]` instead when the UI must load supported +bitsandbytes 8-bit or 4-bit models. Jetson users must follow the dedicated +[Jetson bootstrap](docs/platforms/jetson.md); its bitsandbytes path is not yet +supported. + ### 3. Google Colab (free GPU) [![Open in Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/github/elder-plinius/OBLITERATUS/blob/main/notebooks/abliterate.ipynb) @@ -457,6 +462,12 @@ This sets `CUDA_VISIBLE_DEVICES` before CUDA initializes. The model is then shar The `--dtype` flag controls the precision of model weights, which directly determines how much VRAM you need. Lower precision means smaller memory footprint at the cost of some numerical fidelity: +Install the optional backend before selecting a bitsandbytes mode: + +```bash +pip install -e ".[quantization]" +``` + | Dtype | Bytes/param | 7B model | 70B model | 405B model | |-------|-----------|---------|----------|-----------| | `float32` | 4 | 28 GB | 280 GB | 1620 GB | @@ -783,9 +794,10 @@ run the full Python matrix with 75% repository statement coverage, 60% repositor branch coverage, touched-module regression checks, mature-scope floors, deterministic property/order-repeat checks, selective mutation, package contracts, Windows portability, and supply-chain certification. -Eight environment-bound test files run through the separately documented conditional -workflow for model downloads, network services, operator UI, CUDA, bitsandbytes, MPS, -MLX, and least-privileged remote execution. +Nine environment-bound test files run through the separately documented conditional +workflow for model downloads, network services, operator UI, CUDA, bitsandbytes, +physical Jetson CUDA, MPS, MLX, and least-privileged remote execution. The Jetson +lane is a trusted manual hardware probe; hosted ARM64 CI checks portability only. ## License diff --git a/ci/conditional-test-policy.json b/ci/conditional-test-policy.json index 0394c84..b1ab181 100644 --- a/ci/conditional-test-policy.json +++ b/ci/conditional-test-policy.json @@ -154,6 +154,19 @@ "expected_cost": "included in cuda", "coverage_paths": ["obliteratus/models/loader.py"] }, + { + "id": "jetson-runtime", + "job": "jetson", + "marker": "gpu", + "runner": "self-hosted, linux, ARM64, jetson", + "prerequisites": "trusted manual dispatch on physical Jetson hardware with a JetPack-aligned CUDA PyTorch runtime", + "expected_cost": "under 30 self-hosted runner-minutes", + "coverage_paths": [ + "obliteratus/device.py", + "obliteratus/models/loader.py", + "obliteratus/models/offload_surgery.py" + ] + }, { "id": "mps-runtime", "job": "mps", diff --git a/ci/pr-test-policy.json b/ci/pr-test-policy.json index 95cfeb9..7077141 100644 --- a/ci/pr-test-policy.json +++ b/ci/pr-test-policy.json @@ -13,13 +13,17 @@ ".github/workflows/**", "ci/**", "scripts/check_*.py", + "scripts/jetson_*.py", "scripts/select_pr_tests.py", + "scripts/setup_jetson.py", "pyproject.toml", "uv.lock" ], "infrastructure_tests": [ "tests/test_aiwg_workspace_contracts.py", "tests/test_ci_policy.py", + "tests/test_conditional_gate_scripts.py", + "tests/test_jetson_support_tooling.py", "tests/test_pr_test_selection.py", "tests/test_quality_policy.py", "tests/test_quality_gate_scripts.py", diff --git a/ci/test-risk-map.json b/ci/test-risk-map.json index 289ef64..de745b7 100644 --- a/ci/test-risk-map.json +++ b/ci/test-risk-map.json @@ -78,7 +78,8 @@ "tests/test_offload_surgery.py", "tests/test_persistence_contracts.py", "tests/test_persistence_pipeline.py", - "tests/conditional/test_cuda_runtime.py" + "tests/conditional/test_cuda_runtime.py", + "tests/conditional/test_jetson_runtime.py" ] }, { @@ -134,7 +135,8 @@ "tests/test_model_profile.py", "tests/test_model_profile_contracts.py", "tests/test_runtime_contracts.py", - "tests/test_study_presets.py" + "tests/test_study_presets.py", + "tests/conditional/test_jetson_runtime.py" ] }, { @@ -442,6 +444,7 @@ ], "conditional_gates": [ "cuda-runtime", + "jetson-runtime", "mps-runtime" ] }, @@ -466,6 +469,7 @@ "conditional_gates": [ "model-download-runtime", "cuda-runtime", + "jetson-runtime", "bitsandbytes-runtime" ] }, @@ -489,7 +493,8 @@ "tests/conditional/test_cuda_runtime.py" ], "conditional_gates": [ - "cuda-runtime" + "cuda-runtime", + "jetson-runtime" ] }, { diff --git a/docs/conditional-testing.md b/docs/conditional-testing.md index 655e3da..07b4d61 100644 --- a/docs/conditional-testing.md +++ b/docs/conditional-testing.md @@ -70,21 +70,28 @@ probes. For an operator run on the labeled machine: ```bash -uv sync --locked --extra dev +uv sync --locked --extra dev --extra quantization CUDA_TORCH_VERSION="$(.venv/bin/python -c \ 'import torch; print(torch.__version__.split("+", 1)[0])')" UV_TORCH_BACKEND=cu130 uv pip install --python .venv/bin/python \ --reinstall-package torch "torch==$CUDA_TORCH_VERSION" uv pip check --python .venv/bin/python -uv run --extra dev python scripts/run_conditional_gate.py cuda-runtime -uv run --extra dev python scripts/run_conditional_gate.py bitsandbytes-runtime +uv run --extra dev --extra quantization python scripts/run_conditional_gate.py cuda-runtime +uv run --extra dev --extra quantization python scripts/run_conditional_gate.py bitsandbytes-runtime ``` -Jetson CUDA support is tracked separately from this generic x64 CUDA lane. A -generic Linux ARM build can prove package portability, but it does not prove -Jetson GPU support because Jetson depends on a JetPack/L4T-matched CUDA, cuDNN, -and PyTorch runtime. The support plan, recommended container path, and acceptance -criteria are documented in [NVIDIA Jetson support plan](platforms/jetson.md). +Jetson CUDA support is tracked separately from this generic x64 CUDA lane. The +mandatory `Linux ARM64 preflight` uses GitHub's hosted `ubuntu-24.04-arm` runner +to prove locked CPU packaging, imports, CLI startup, and Jetson tooling contracts. +It is not GPU evidence. Jetson depends on a JetPack/L4T-matched CUDA, cuDNN, and +PyTorch runtime. + +Physical testing uses only a trusted manual dispatch on the labels `self-hosted`, +`linux`, `ARM64`, and `jetson`. The job preserves NVIDIA's vendor PyTorch, runs +`jetson-runtime`, and uploads the sanitized `conditional-jetson-` +artifact for 30 days. It never runs for a pull request, schedule, or release. +Contributor bootstrap, runner isolation, reporting commands, and acceptance +criteria are documented in the [NVIDIA Jetson support plan](platforms/jetson.md). ## Apple MPS and MLX diff --git a/docs/platforms/jetson.md b/docs/platforms/jetson.md index 4be9af7..2120546 100644 --- a/docs/platforms/jetson.md +++ b/docs/platforms/jetson.md @@ -79,6 +79,41 @@ from being selected: upstream documents that wheel as SBSA/server ARM and says Jetson L4T/JetPack requires a source build. Until a pinned source build passes on the selected device, bitsandbytes is unsupported for that tier. +### Experimental contributor bootstrap + +Start with NVIDIA's PyTorch wheel or PyTorch iGPU container for the exact +JetPack patch installed on the device. Confirm that `python3 -c 'import torch; +assert torch.cuda.is_available()'` succeeds before installing OBLITERATUS. Then, +from a checkout of the exact commit under test, run: + +```bash +python3 -m venv .venv-jetson-tools +.venv-jetson-tools/bin/python -m pip install "uv==0.12.4" +.venv-jetson-tools/bin/python scripts/setup_jetson.py \ + --python python3 \ + --uv-python .venv-jetson-tools/bin/python \ + --venv .venv-jetson +.venv-jetson/bin/python scripts/run_conditional_gate.py jetson-runtime +.venv-jetson/bin/python scripts/jetson_support.py \ + --check \ + --gate-evidence conditional-evidence/jetson-runtime.json \ + --output conditional-evidence/jetson-report.json \ + --issue-body conditional-evidence/jetson-issue.md +``` + +The bootstrap validates ARM64, L4T, and CUDA before changing the environment. +It creates a virtual environment with `--system-site-packages`, exports the +committed lock, and installs locked OBLITERATUS dependencies without replacing +the vendor `torch`. It also excludes bitsandbytes. The generic bitsandbytes +package is now an explicit `quantization` extra for supported non-Jetson +environments; Jetson quantization remains a separate source-build milestone. + +For JetPack 6.2, NVIDIA publishes the +`nvcr.io/nvidia/pytorch:25.06-py3-igpu` container. Run it only on Jetson hardware +with the NVIDIA runtime, mount a reviewed checkout, and use the same bootstrap +inside the container. Match other JetPack releases through NVIDIA's +compatibility table rather than substituting a `latest` tag. + ## Conditional gate Add a new gate instead of modifying the x64 CUDA gate: @@ -93,6 +128,35 @@ Add a new gate instead of modifying the x64 CUDA gate: The job must run only from a trusted ref or reviewed maintainer dispatch. A persistent self-hosted Jetson must never execute untrusted pull-request code. +### Attaching a contributor-owned runner + +Register the runner using GitHub's self-hosted runner instructions, on the +Jetson itself, and add the custom label `jetson`. GitHub supplies the +`self-hosted`, `linux`, and `ARM64` default labels. Verify that the repository +shows exactly these required labels before dispatching the job: + +```text +self-hosted, linux, ARM64, jetson +``` + +Use a dedicated, non-personal runner account and a disposable or resettable +workspace. Do not place Hugging Face, SSH, cloud, or signing credentials on the +runner. Only a maintainer should manually dispatch `Conditional tests` against +a reviewed commit; the Jetson job is deliberately unavailable to pull-request, +scheduled, and release triggers. Remove the runner registration token after +setup and keep the runner offline when it is not being used for reviewed work. + +### Reporting results without a project-owned Jetson + +Open the [Jetson runtime report](https://github.com/elder-plinius/OBLITERATUS/issues/new?template=jetson-runtime.yml) +issue form and attach `conditional-evidence/jetson-report.json`, or paste the +generated `conditional-evidence/jetson-issue.md`. The collector reports only an +allow-listed architecture, OS/JetPack, PyTorch/CUDA, device-class, test-result, +and commit profile. It excludes environment variables, hostnames, usernames, +network addresses, device serials, tokens, and local filesystem paths. Review +the file yourself before publishing it. A failed report is useful evidence and +does not imply that the contributor must diagnose the compatibility problem. + The gate should verify: - `platform.machine()` is `aarch64` or equivalent ARM64. @@ -146,3 +210,6 @@ them. - [Astral: Using uv with PyTorch](https://docs.astral.sh/uv/guides/integration/pytorch/) - [Hugging Face: bitsandbytes installation guide](https://huggingface.co/docs/bitsandbytes/installation) - [GitHub: secure use of self-hosted runners](https://docs.github.com/en/actions/reference/security/secure-use#hardening-for-self-hosted-runners) +- [GitHub: use self-hosted runner labels](https://docs.github.com/en/actions/how-tos/manage-runners/self-hosted-runners/use-in-a-workflow) +- [GitHub: hosted ARM64 runners](https://docs.github.com/en/actions/reference/runners/github-hosted-runners) +- [NVIDIA: PyTorch 25.06 for JetPack 6.2](https://docs.nvidia.com/deeplearning/frameworks/pytorch-release-notes/rel-25-06.html) diff --git a/pyproject.toml b/pyproject.toml index c044473..1bd7a82 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -34,7 +34,6 @@ dependencies = [ "numpy>=1.24", "scikit-learn>=1.3", "tqdm>=4.64", - "bitsandbytes>=0.46.1", ] [project.urls] @@ -51,6 +50,7 @@ dev = [ "pytest-cov==7.1.0", "ruff==0.16.2", ] +quantization = ["bitsandbytes>=0.46.1"] spaces = ["gradio>=6.7,<7.0"] [dependency-groups] diff --git a/scripts/jetson_support.py b/scripts/jetson_support.py new file mode 100644 index 0000000..6ce37a4 --- /dev/null +++ b/scripts/jetson_support.py @@ -0,0 +1,254 @@ +#!/usr/bin/env python3 +"""Collect privacy-safe Jetson runtime evidence for CI and issue reports.""" + +from __future__ import annotations + +import argparse +import importlib.metadata +import json +import os +import platform +import re +import subprocess +import sys +from datetime import datetime, timezone +from pathlib import Path +from typing import Sequence + + +ISSUE_URL = "https://github.com/elder-plinius/OBLITERATUS/issues/31" +JETSON_RELEASE = Path("/etc/nv_tegra_release") +OS_RELEASE = Path("/etc/os-release") +SHA = re.compile(r"^[0-9a-f]{40}$") + + +def _read_first_line(path: Path, *, limit: int = 500) -> str | None: + try: + return path.read_text(encoding="utf-8", errors="replace").splitlines()[0][:limit] + except (OSError, IndexError): + return None + + +def _os_release(path: Path) -> dict[str, str]: + try: + lines = path.read_text(encoding="utf-8", errors="replace").splitlines() + except OSError: + return {} + values: dict[str, str] = {} + for line in lines: + key, separator, value = line.partition("=") + if separator and key in {"ID", "VERSION_ID", "PRETTY_NAME"}: + values[key.lower()] = value.strip().strip('"')[:200] + return values + + +def _capture(command: Sequence[str], *, timeout: int = 10) -> str | None: + try: + result = subprocess.run( + command, + check=False, + capture_output=True, + text=True, + timeout=timeout, + ) + except (OSError, subprocess.TimeoutExpired): + return None + if result.returncode != 0: + return None + value = result.stdout.strip() + return value[:500] or None + + +def _candidate_sha() -> str: + candidate = os.environ.get("GITHUB_SHA", "") + if SHA.fullmatch(candidate): + return candidate + local = _capture(["git", "rev-parse", "HEAD"]) + return local if local is not None and SHA.fullmatch(local) else "local" + + +def collect_host_facts( + *, + tegra_release: Path = JETSON_RELEASE, + os_release: Path = OS_RELEASE, +) -> dict[str, object]: + """Return an allow-listed host profile without identity or network data.""" + + return { + "architecture": platform.machine(), + "python_version": platform.python_version(), + "os": _os_release(os_release), + "l4t_release": _read_first_line(tegra_release), + "jetpack_package": _capture( + ["dpkg-query", "-W", "-f=${Version}", "nvidia-jetpack"], + ), + } + + +def collect_runtime_facts() -> dict[str, object]: + """Return an allow-listed PyTorch/GPU profile without serials or file paths.""" + + facts: dict[str, object] = { + "torch_imported": False, + "torch_version": None, + "torch_cuda_version": None, + "cuda_available": False, + "cuda_device_count": 0, + "device_name": None, + "compute_capability": None, + "total_memory_gb": None, + "bitsandbytes_version": None, + } + try: + import torch + except Exception as exc: # pragma: no cover - exact vendor loader failures vary + facts["torch_import_error"] = type(exc).__name__ + return facts + + facts.update({ + "torch_imported": True, + "torch_version": str(torch.__version__), + "torch_cuda_version": torch.version.cuda, + "cuda_available": bool(torch.cuda.is_available()), + "cuda_device_count": int(torch.cuda.device_count()), + }) + if facts["cuda_available"] and facts["cuda_device_count"]: + properties = torch.cuda.get_device_properties(0) + facts.update({ + "device_name": str(properties.name)[:200], + "compute_capability": list(torch.cuda.get_device_capability(0)), + "total_memory_gb": round(properties.total_memory / 1024 ** 3, 2), + }) + try: + facts["bitsandbytes_version"] = importlib.metadata.version("bitsandbytes") + except importlib.metadata.PackageNotFoundError: + pass + return facts + + +def validate_report(report: dict[str, object]) -> tuple[list[str], list[str]]: + """Return blocking errors and non-blocking compatibility warnings.""" + + host = report["host"] + runtime = report["runtime"] + assert isinstance(host, dict) + assert isinstance(runtime, dict) + errors: list[str] = [] + warnings: list[str] = [] + if str(host.get("architecture", "")).lower() not in {"aarch64", "arm64"}: + errors.append("host architecture is not ARM64") + if not host.get("l4t_release"): + errors.append("/etc/nv_tegra_release is unavailable; this is not a Jetson L4T runtime") + if not runtime.get("torch_imported"): + errors.append("PyTorch could not be imported from the JetPack-aligned runtime") + elif not runtime.get("torch_cuda_version"): + errors.append("PyTorch is not a CUDA build") + elif not runtime.get("cuda_available"): + errors.append("PyTorch cannot access the Jetson CUDA device") + if runtime.get("bitsandbytes_version"): + warnings.append( + "bitsandbytes is installed but remains unsupported until its pinned Jetson " + "source build passes the separate quantization probe", + ) + return errors, warnings + + +def _gate_summary(path: Path | None) -> dict[str, object] | None: + if path is None: + return None + try: + value = json.loads(path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError): + return {"status": "unavailable"} + if not isinstance(value, dict): + return {"status": "invalid"} + summary: dict[str, object] = {} + for key in ("gate", "status"): + if isinstance(value.get(key), str): + summary[key] = value[key][:100] + git_sha = value.get("git_sha") + if isinstance(git_sha, str) and (SHA.fullmatch(git_sha) or git_sha == "local"): + summary["git_sha"] = git_sha + counts = value.get("counts") + if isinstance(counts, dict): + summary["counts"] = { + key: counts[key] + for key in ("tests", "failures", "errors", "skipped") + if isinstance(counts.get(key), int) and counts[key] >= 0 + } + return summary + + +def build_report(*, gate_evidence: Path | None = None) -> dict[str, object]: + report: dict[str, object] = { + "schema_version": 1, + "generated_at": datetime.now(timezone.utc).isoformat(), + "issue": ISSUE_URL, + "git_sha": _candidate_sha(), + "host": collect_host_facts(), + "runtime": collect_runtime_facts(), + } + errors, warnings = validate_report(report) + report["validation"] = {"errors": errors, "warnings": warnings} + gate = _gate_summary(gate_evidence) + if gate is not None: + report["gate_evidence"] = gate + return report + + +def issue_body(report: dict[str, object]) -> str: + return "\n".join([ + "## Jetson runtime report", + "", + "### What happened", + "", + "", + "### Reproduction", + "", + "", + "### Sanitized environment evidence", + "", + "```json", + json.dumps(report, indent=2, sort_keys=True), + "```", + "", + "This report intentionally excludes environment variables, hostnames, usernames,", + "network addresses, GPU serials, tokens, and local filesystem paths.", + "", + ]) + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--output", + type=Path, + default=Path("conditional-evidence/jetson-report.json"), + ) + parser.add_argument("--gate-evidence", type=Path) + parser.add_argument("--issue-body", type=Path) + parser.add_argument("--check", action="store_true") + args = parser.parse_args() + + report = build_report(gate_evidence=args.gate_evidence) + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(json.dumps(report, indent=2, sort_keys=True) + "\n") + if args.issue_body is not None: + args.issue_body.parent.mkdir(parents=True, exist_ok=True) + args.issue_body.write_text(issue_body(report), encoding="utf-8") + validation = report["validation"] + assert isinstance(validation, dict) + errors = validation["errors"] + warnings = validation["warnings"] + assert isinstance(errors, list) + assert isinstance(warnings, list) + for warning in warnings: + print(f"WARNING: {warning}", file=sys.stderr) + for error in errors: + print(f"ERROR: {error}", file=sys.stderr) + print(args.output) + return 2 if args.check and errors else 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/run_conditional_gate.py b/scripts/run_conditional_gate.py index 1e78ac9..6a73e53 100644 --- a/scripts/run_conditional_gate.py +++ b/scripts/run_conditional_gate.py @@ -7,6 +7,7 @@ import argparse import importlib.util import json import os +import platform import subprocess import sys from datetime import datetime, timezone @@ -21,24 +22,31 @@ GATES = { "operator-ui": "tests/conditional/test_operator_ui.py", "cuda-runtime": "tests/conditional/test_cuda_runtime.py", "bitsandbytes-runtime": "tests/conditional/test_cuda_runtime.py", + "jetson-runtime": "tests/conditional/test_jetson_runtime.py", "mps-runtime": "tests/conditional/test_mps_runtime.py", "mlx-runtime": "tests/conditional/test_mlx_runtime.py", "remote-execution": "tests/conditional/test_remote_runtime.py", } +JETSON_RELEASE = Path("/etc/nv_tegra_release") def missing_prerequisites(gate: str) -> list[str]: missing: list[str] = [] - if gate in {"cuda-runtime", "bitsandbytes-runtime", "mps-runtime"}: + if gate in {"cuda-runtime", "bitsandbytes-runtime", "jetson-runtime", "mps-runtime"}: import torch - if gate.startswith("cuda") or gate.startswith("bitsandbytes"): + if gate.startswith(("cuda", "bitsandbytes", "jetson")): if not torch.cuda.is_available(): missing.append("a CUDA-capable PyTorch runtime") elif not (hasattr(torch.backends, "mps") and torch.backends.mps.is_available()): missing.append("an available Apple MPS backend") if gate == "bitsandbytes-runtime" and importlib.util.find_spec("bitsandbytes") is None: missing.append("bitsandbytes") + if gate == "jetson-runtime": + if platform.machine().lower() not in {"aarch64", "arm64"}: + missing.append("an ARM64 host") + if not JETSON_RELEASE.is_file(): + missing.append("a Jetson L4T runtime") if gate == "mlx-runtime": for module in ("mlx", "mlx_lm"): if importlib.util.find_spec(module) is None: diff --git a/scripts/setup_jetson.py b/scripts/setup_jetson.py new file mode 100644 index 0000000..68ba6de --- /dev/null +++ b/scripts/setup_jetson.py @@ -0,0 +1,190 @@ +#!/usr/bin/env python3 +"""Create an OBLITERATUS venv without replacing JetPack's PyTorch runtime.""" + +from __future__ import annotations + +import argparse +import re +import subprocess +import sys +import tempfile +from pathlib import Path +from typing import Sequence + + +EXCLUDED_PACKAGES = {"bitsandbytes", "torch"} +NORMALIZE = re.compile(r"[-_.]+") + + +def _run(command: Sequence[str], *, cwd: Path) -> None: + print("+ " + " ".join(command)) + subprocess.run(command, cwd=cwd, check=True) + + +def _require_new_or_reusable_venv(venv: Path, *, reuse: bool) -> None: + if not venv.exists(): + return + if not reuse: + raise ValueError(f"virtual environment already exists: {venv}; pass --reuse to use it") + config = venv / "pyvenv.cfg" + try: + contents = config.read_text(encoding="utf-8").lower() + except OSError as exc: + raise ValueError(f"existing path is not a reusable virtual environment: {venv}") from exc + if "include-system-site-packages = true" not in contents: + raise ValueError(f"existing virtual environment does not expose JetPack packages: {venv}") + + +def _require_safe_target(venv: Path, project: Path) -> None: + resolved = venv.resolve() + forbidden = {Path("/").resolve(), Path.home().resolve(), project.resolve()} + if resolved in forbidden: + raise ValueError(f"refusing unsafe virtual environment target: {resolved}") + + +def _require_exclusions(requirements: Path) -> None: + emitted: set[str] = set() + for raw_line in requirements.read_text(encoding="utf-8").splitlines(): + line = raw_line.strip() + if not line or line.startswith(("#", "--")): + continue + name = re.split(r"[<>=!~;@\[]", line, maxsplit=1)[0].strip() + emitted.add(NORMALIZE.sub("-", name).lower()) + unexpected = sorted(EXCLUDED_PACKAGES & emitted) + if unexpected: + raise RuntimeError(f"Jetson export contains forbidden packages: {unexpected}") + + +def prepare( + *, + project: Path, + venv: Path, + python: str, + uv_python: str, + reuse: bool, +) -> None: + project = project.resolve() + support_script = project / "scripts" / "jetson_support.py" + if not support_script.is_file() or not (project / "uv.lock").is_file(): + raise ValueError(f"not an OBLITERATUS checkout: {project}") + _require_safe_target(venv, project) + _require_new_or_reusable_venv(venv, reuse=reuse) + + with tempfile.TemporaryDirectory(prefix="obliteratus-jetson-") as temp_value: + temp = Path(temp_value) + _run( + [python, str(support_script), "--check", "--output", str(temp / "host.json")], + cwd=project, + ) + _run([uv_python, "-m", "uv", "--version"], cwd=project) + if not venv.exists(): + _run([python, "-m", "venv", "--system-site-packages", str(venv)], cwd=project) + + target_python = venv / "bin" / "python" + _run( + [ + str(target_python), + str(support_script), + "--check", + "--output", + str(temp / "venv.json"), + ], + cwd=project, + ) + requirements = temp / "requirements-jetson.txt" + _run( + [ + uv_python, + "-m", + "uv", + "export", + "--locked", + "--no-default-groups", + "--extra", + "dev", + "--no-emit-project", + "--no-emit-package", + "torch", + "--no-emit-package", + "bitsandbytes", + "--no-annotate", + "--no-header", + "--no-hashes", + "--output-file", + str(requirements), + ], + cwd=project, + ) + _require_exclusions(requirements) + _run( + [ + uv_python, + "-m", + "uv", + "pip", + "install", + "--python", + str(target_python), + "--no-deps", + "--requirements", + str(requirements), + ], + cwd=project, + ) + _run( + [ + uv_python, + "-m", + "uv", + "pip", + "install", + "--python", + str(target_python), + "--no-deps", + "--editable", + str(project), + ], + cwd=project, + ) + _run( + [uv_python, "-m", "uv", "pip", "check", "--python", str(target_python)], + cwd=project, + ) + + print("Jetson environment prepared without replacing vendor PyTorch.") + print(f"Run: {target_python} scripts/run_conditional_gate.py jetson-runtime") + print( + f"Then: {target_python} scripts/jetson_support.py --check " + "--gate-evidence conditional-evidence/jetson-runtime.json " + "--issue-body conditional-evidence/jetson-issue.md", + ) + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--project", type=Path, default=Path(__file__).parents[1]) + parser.add_argument("--venv", type=Path, default=Path(".venv-jetson")) + parser.add_argument("--python", default=sys.executable) + parser.add_argument( + "--uv-python", + default=sys.executable, + help="interpreter containing the pinned uv module (defaults to this interpreter)", + ) + parser.add_argument("--reuse", action="store_true") + args = parser.parse_args() + try: + prepare( + project=args.project, + venv=args.venv, + python=args.python, + uv_python=args.uv_python, + reuse=args.reuse, + ) + except (OSError, RuntimeError, subprocess.CalledProcessError, ValueError) as exc: + print(f"ERROR: {exc}", file=sys.stderr) + return 2 + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/conditional/test_jetson_runtime.py b/tests/conditional/test_jetson_runtime.py new file mode 100644 index 0000000..22d3d34 --- /dev/null +++ b/tests/conditional/test_jetson_runtime.py @@ -0,0 +1,53 @@ +"""Physical NVIDIA Jetson CUDA placement and operation probe.""" + +from __future__ import annotations + +import platform +from pathlib import Path + +import pytest +import torch +import torch.nn as nn +from accelerate.hooks import AlignDevicesHook, add_hook_to_module + +from obliteratus import device +from obliteratus.abliterate import AbliterationPipeline + + +pytestmark = pytest.mark.gpu + + +def test_jetson_cuda_runtime_contract(): + assert platform.machine().lower() in {"aarch64", "arm64"} + assert Path("/etc/nv_tegra_release").is_file() + assert torch.version.cuda is not None + assert torch.cuda.is_available() + assert torch.cuda.device_count() > 0 + assert device.is_cuda() + assert device.get_device("auto") == "cuda" + tensor = torch.arange(16, device="cuda", dtype=torch.float32).reshape(4, 4) + result = tensor @ tensor.T + assert result.device.type == "cuda" + assert torch.isfinite(result).all() + + +def test_jetson_cuda_offloaded_surgery_contract(): + module = nn.Module() + module.proj = nn.Linear(4, 4, bias=False) + original = module.proj.weight.detach().clone() + hook = AlignDevicesHook(execution_device="cuda", offload=True) + add_hook_to_module(module.proj, hook) + + count = AbliterationPipeline._project_out_advanced( + module, + torch.tensor([[1.0], [0.0], [0.0], [0.0]], device="cuda"), + ["proj"], + ) + output = module.proj(torch.ones(1, 4, device="cuda")) + + expected = original.clone() + expected[:, 0] = 0 + assert count == 1 + assert output.device.type == "cuda" + assert module.proj.weight.device.type == "meta" + torch.testing.assert_close(hook.weights_map["weight"], expected) diff --git a/tests/test_ci_policy.py b/tests/test_ci_policy.py index 576b005..aa3adfc 100644 --- a/tests/test_ci_policy.py +++ b/tests/test_ci_policy.py @@ -129,6 +129,24 @@ def test_pull_request_gate_is_fast_risk_mapped_and_uses_shared_floor(): assert "tests/conditional/" in policy["excluded_test_prefixes"] +def test_arm64_preflight_proves_portability_without_claiming_jetson_cuda(): + workflow = WORKFLOW.read_text(encoding="utf-8") + arm = workflow.split(" arm64-preflight:\n", maxsplit=1)[1].split( + " pr-core:\n", + maxsplit=1, + )[0] + + assert "runs-on: ubuntu-24.04-arm" in arm + assert "github.event_name == 'pull_request'" in arm + assert "github.event_name == 'push' && github.ref == 'refs/heads/main'" in arm + assert "-m build --wheel" in arm + assert "import obliteratus" in arm + assert "-m obliteratus --help" in arm + assert "tests/test_device_boundaries.py" in arm + assert "tests/test_jetson_support_tooling.py" in arm + assert "jetson-runtime" not in arm + + def test_release_depth_jobs_only_run_for_tags_or_manual_validation(): workflow = WORKFLOW.read_text(encoding="utf-8") release_condition = ( diff --git a/tests/test_conditional_gate_scripts.py b/tests/test_conditional_gate_scripts.py index 0dc1f34..80d4822 100644 --- a/tests/test_conditional_gate_scripts.py +++ b/tests/test_conditional_gate_scripts.py @@ -42,6 +42,7 @@ def test_cuda_job_replaces_locked_cpu_torch_with_same_version_cuda_build(): cuda_job = workflow.split(" cuda:\n", maxsplit=1)[1].split(" mps:\n", maxsplit=1)[0] assert "torch.__version__.split" in cuda_job + assert "--extra quantization" in cuda_job assert "UV_TORCH_BACKEND=cu130 uv pip install" in cuda_job assert "--reinstall-package torch" in cuda_job assert '"torch==$CUDA_TORCH_VERSION"' in cuda_job @@ -49,6 +50,28 @@ def test_cuda_job_replaces_locked_cpu_torch_with_same_version_cuda_build(): assert 'uv pip check --python "$CONDITIONAL_ENV/bin/python"' in cuda_job +def test_jetson_job_is_manual_physical_trusted_and_retains_sanitized_evidence(): + workflow = (ROOT / ".github" / "workflows" / "conditional-tests.yml").read_text() + jetson_job = workflow.split(" jetson:\n", maxsplit=1)[1].split( + " mps:\n", + maxsplit=1, + )[0] + + assert "github.event_name == 'workflow_dispatch' && inputs.run_jetson" in jetson_job + assert "runs-on: [self-hosted, linux, ARM64, jetson]" in jetson_job + assert "scripts/setup_jetson.py" in jetson_job + assert "scripts/run_conditional_gate.py jetson-runtime" in jetson_job + assert "scripts/jetson_support.py" in jetson_job + assert "conditional-jetson-${{ github.run_attempt }}" in jetson_job + assert "retention-days: 30" in jetson_job + assert "actions/setup-python" not in jetson_job + + policy = json.loads((ROOT / "ci" / "conditional-test-policy.json").read_text()) + gate = next(value for value in policy["gates"] if value["id"] == "jetson-runtime") + assert gate["job"] == "jetson" + assert gate["runner"] == "self-hosted, linux, ARM64, jetson" + + def test_policy_rejects_unknown_cpu_exclusion_gate(tmp_path): policy = json.loads((ROOT / "ci" / "conditional-test-policy.json").read_text()) quality = { diff --git a/tests/test_jetson_support_tooling.py b/tests/test_jetson_support_tooling.py new file mode 100644 index 0000000..2a0ddb8 --- /dev/null +++ b/tests/test_jetson_support_tooling.py @@ -0,0 +1,254 @@ +"""CPU-testable contracts for the experimental Jetson support path.""" + +from __future__ import annotations + +import importlib.metadata +import json +import sys +import tomllib +from pathlib import Path +from types import SimpleNamespace + +import pytest +import yaml + +from scripts import jetson_support +from scripts import run_conditional_gate +from scripts import setup_jetson + + +ROOT = Path(__file__).parents[1] + + +def test_host_evidence_is_allow_listed(monkeypatch, tmp_path): + tegra = tmp_path / "nv_tegra_release" + tegra.write_text("# R36 (release), REVISION: 4.3\nserial=secret\n") + os_release = tmp_path / "os-release" + os_release.write_text( + 'ID=ubuntu\nVERSION_ID="22.04"\nPRETTY_NAME="Ubuntu 22.04"\nSECRET=value\n' + ) + monkeypatch.setattr(jetson_support.platform, "machine", lambda: "aarch64") + monkeypatch.setattr(jetson_support.platform, "python_version", lambda: "3.10.12") + monkeypatch.setattr(jetson_support, "_capture", lambda command, **kwargs: "6.2+b17") + + facts = jetson_support.collect_host_facts( + tegra_release=tegra, + os_release=os_release, + ) + + assert facts == { + "architecture": "aarch64", + "python_version": "3.10.12", + "os": { + "id": "ubuntu", + "version_id": "22.04", + "pretty_name": "Ubuntu 22.04", + }, + "l4t_release": "# R36 (release), REVISION: 4.3", + "jetpack_package": "6.2+b17", + } + assert "secret" not in json.dumps(facts).lower() + + +def test_runtime_evidence_reports_cuda_without_device_identity(monkeypatch): + properties = SimpleNamespace(name="Orin", total_memory=64 * 1024**3) + fake_cuda = SimpleNamespace( + is_available=lambda: True, + device_count=lambda: 1, + get_device_properties=lambda _index: properties, + get_device_capability=lambda _index: (8, 7), + ) + fake_torch = SimpleNamespace( + __version__="2.8.0a0+nv25.06", + version=SimpleNamespace(cuda="12.6"), + cuda=fake_cuda, + ) + monkeypatch.setitem(sys.modules, "torch", fake_torch) + + def missing_package(_name): + raise importlib.metadata.PackageNotFoundError + + monkeypatch.setattr(jetson_support.importlib.metadata, "version", missing_package) + + facts = jetson_support.collect_runtime_facts() + + assert facts["cuda_available"] is True + assert facts["device_name"] == "Orin" + assert facts["compute_capability"] == [8, 7] + assert facts["total_memory_gb"] == 64.0 + assert set(facts) == { + "torch_imported", + "torch_version", + "torch_cuda_version", + "cuda_available", + "cuda_device_count", + "device_name", + "compute_capability", + "total_memory_gb", + "bitsandbytes_version", + } + + +def test_report_validation_blocks_non_jetson_or_non_cuda_and_warns_on_bnb(): + report = { + "host": {"architecture": "x86_64", "l4t_release": None}, + "runtime": { + "torch_imported": True, + "torch_cuda_version": None, + "cuda_available": False, + "bitsandbytes_version": "0.47.0", + }, + } + + errors, warnings = jetson_support.validate_report(report) + + assert len(errors) == 3 + assert any("ARM64" in error for error in errors) + assert any("Jetson L4T" in error for error in errors) + assert any("not a CUDA build" in error for error in errors) + assert warnings and "unsupported" in warnings[0] + + +def test_gate_evidence_and_issue_body_cannot_copy_arbitrary_fields(tmp_path): + evidence = tmp_path / "gate.json" + evidence.write_text(json.dumps({ + "gate": "jetson-runtime", + "status": "passed", + "git_sha": "a" * 40, + "counts": {"tests": 1, "token": "nested-secret"}, + "token": "must-not-escape", + "hostname": "must-not-escape", + })) + + summary = jetson_support._gate_summary(evidence) + body = jetson_support.issue_body({"gate_evidence": summary}) + + assert summary == { + "gate": "jetson-runtime", + "status": "passed", + "git_sha": "a" * 40, + "counts": {"tests": 1}, + } + assert "must-not-escape" not in body + assert "nested-secret" not in body + assert "excludes environment variables" in body + + +def test_dependency_export_rejects_torch_and_bitsandbytes(tmp_path): + requirements = tmp_path / "requirements.txt" + requirements.write_text("transformers==4.56.0\npytest==8.4.1\n") + setup_jetson._require_exclusions(requirements) + + for forbidden in ( + "torch==2.8.0", + "torch @ https://example.invalid/torch.whl", + "bitsandbytes[diagnostics]==0.47.0", + ): + requirements.write_text(forbidden + "\n") + with pytest.raises(RuntimeError, match="forbidden packages"): + setup_jetson._require_exclusions(requirements) + + +def test_jetson_bootstrap_preserves_vendor_runtime_and_uses_locked_no_deps( + monkeypatch, + tmp_path, +): + project = tmp_path / "project" + (project / "scripts").mkdir(parents=True) + (project / "scripts" / "jetson_support.py").write_text("# fixture\n") + (project / "uv.lock").write_text("# fixture\n") + venv = tmp_path / "jetson-venv" + commands: list[list[str]] = [] + + def fake_run(command, *, cwd): + command = list(command) + commands.append(command) + if command[1:4] == ["-m", "venv", "--system-site-packages"]: + (venv / "bin").mkdir(parents=True) + (venv / "pyvenv.cfg").write_text("include-system-site-packages = true\n") + if "--output-file" in command: + output = Path(command[command.index("--output-file") + 1]) + output.write_text("transformers==4.56.0\n") + + monkeypatch.setattr(setup_jetson, "_run", fake_run) + + setup_jetson.prepare( + project=project, + venv=venv, + python="vendor-python", + uv_python="tool-python", + reuse=False, + ) + + assert commands[0][0] == "vendor-python" + assert "--check" in commands[0] + export = next(command for command in commands if "export" in command) + assert export.count("--no-emit-package") == 2 + assert "torch" in export and "bitsandbytes" in export + installs = [command for command in commands if "install" in command] + assert len(installs) == 2 + assert all("--no-deps" in command for command in installs) + assert any("check" in command for command in commands) + + +def test_bootstrap_rejects_unsafe_or_non_vendor_reusable_targets(tmp_path): + project = tmp_path / "project" + project.mkdir() + with pytest.raises(ValueError, match="unsafe"): + setup_jetson._require_safe_target(project, project) + + existing = tmp_path / "existing" + existing.mkdir() + (existing / "pyvenv.cfg").write_text("include-system-site-packages = false\n") + with pytest.raises(ValueError, match="does not expose JetPack"): + setup_jetson._require_new_or_reusable_venv(existing, reuse=True) + + +def test_bitsandbytes_is_opt_in_for_quantization_only(): + metadata = tomllib.loads((ROOT / "pyproject.toml").read_text()) + base = metadata["project"]["dependencies"] + extras = metadata["project"]["optional-dependencies"] + + assert not any(value.startswith("bitsandbytes") for value in base) + assert extras["quantization"] == ["bitsandbytes>=0.46.1"] + + +def test_jetson_issue_form_requires_reproducible_sanitized_hardware_evidence(): + form = yaml.safe_load( + (ROOT / ".github" / "ISSUE_TEMPLATE" / "jetson-runtime.yml").read_text(), + ) + fields = {value.get("id"): value for value in form["body"] if value.get("id")} + + assert set(fields) == { + "device", + "jetpack", + "commit", + "reproduction", + "expected", + "actual", + "evidence", + "confirmations", + } + assert all(value.get("validations", {}).get("required") for value in fields.values()) + confirmations = fields["confirmations"]["attributes"]["options"] + assert all(option["required"] for option in confirmations) + assert any("secrets" in option["label"] for option in confirmations) + + +def test_jetson_conditional_prerequisites_are_physical_and_cuda(monkeypatch, tmp_path): + fake_torch = SimpleNamespace(cuda=SimpleNamespace(is_available=lambda: True)) + monkeypatch.setitem(sys.modules, "torch", fake_torch) + monkeypatch.setattr(run_conditional_gate.platform, "machine", lambda: "aarch64") + tegra = tmp_path / "nv_tegra_release" + tegra.write_text("# R36\n") + monkeypatch.setattr(run_conditional_gate, "JETSON_RELEASE", tegra) + assert run_conditional_gate.missing_prerequisites("jetson-runtime") == [] + + monkeypatch.setattr(run_conditional_gate.platform, "machine", lambda: "x86_64") + fake_torch.cuda.is_available = lambda: False + tegra.unlink() + assert run_conditional_gate.missing_prerequisites("jetson-runtime") == [ + "a CUDA-capable PyTorch runtime", + "an ARM64 host", + "a Jetson L4T runtime", + ] diff --git a/uv.lock b/uv.lock index 53f8a6d..bb2c903 100644 --- a/uv.lock +++ b/uv.lock @@ -15,7 +15,7 @@ resolution-markers = [ ] [options] -exclude-newer = "2026-08-11T18:17:23.461940614Z" +exclude-newer = "2026-08-19T00:05:48.361273363Z" exclude-newer-span = "P3D" [manifest] @@ -2602,7 +2602,6 @@ version = "0.1.2" source = { editable = "." } dependencies = [ { name = "accelerate" }, - { name = "bitsandbytes" }, { name = "datasets" }, { name = "matplotlib", version = "3.10.9", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.11'" }, { name = "matplotlib", version = "3.11.1", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version >= '3.11'" }, @@ -2631,6 +2630,9 @@ dev = [ { name = "pytest-cov" }, { name = "ruff" }, ] +quantization = [ + { name = "bitsandbytes" }, +] spaces = [ { name = "gradio" }, ] @@ -2652,7 +2654,7 @@ quality = [ [package.metadata] requires-dist = [ { name = "accelerate", specifier = ">=0.24" }, - { name = "bitsandbytes", specifier = ">=0.46.1" }, + { name = "bitsandbytes", marker = "extra == 'quantization'", specifier = ">=0.46.1" }, { name = "build", marker = "extra == 'dev'", specifier = "==1.2.2.post1" }, { name = "datasets", specifier = ">=2.14" }, { name = "gradio", marker = "extra == 'spaces'", specifier = ">=6.7,<7.0" }, @@ -2674,7 +2676,7 @@ requires-dist = [ { name = "tqdm", specifier = ">=4.64" }, { name = "transformers", specifier = ">=4.40" }, ] -provides-extras = ["dev", "spaces"] +provides-extras = ["dev", "quantization", "spaces"] [package.metadata.requires-dev] ci = [