"""Build a deterministic, synthetic Hugging Face causal language model.""" from __future__ import annotations import json from pathlib import Path import torch from tokenizers import Tokenizer from tokenizers.models import WordLevel from tokenizers.pre_tokenizers import Whitespace from transformers import GPT2Config, GPT2LMHeadModel, PreTrainedTokenizerFast FIXTURE_SEED = 20260814 FIXTURE_VOCAB = { "": 0, "": 1, "": 2, "harmful": 3, "harmless": 4, "request": 5, "answer": 6, "hello": 7, "world": 8, "safe": 9, "test": 10, } def build_tiny_offline_model(destination: Path) -> Path: """Create a tiny random-init GPT-2 model without downloads or caches.""" destination = Path(destination) destination.mkdir(parents=True, exist_ok=False) torch.manual_seed(FIXTURE_SEED) tokenizer_backend = Tokenizer(WordLevel(FIXTURE_VOCAB, unk_token="")) tokenizer_backend.pre_tokenizer = Whitespace() tokenizer = PreTrainedTokenizerFast( tokenizer_object=tokenizer_backend, pad_token="", eos_token="", unk_token="", ) config = GPT2Config( vocab_size=len(FIXTURE_VOCAB), n_positions=128, n_ctx=128, n_embd=16, n_layer=1, n_head=2, n_inner=32, bos_token_id=1, eos_token_id=1, pad_token_id=0, ) model = GPT2LMHeadModel(config) model.save_pretrained(destination, safe_serialization=True) tokenizer.save_pretrained(destination) manifest = { "fixture": "tiny-offline-gpt2", "provenance": "generated locally from configuration with random initialization", "training_data": None, "third_party_weights": None, "license": "AGPL-3.0-only (part of the OBLITERATUS test suite)", "seed": FIXTURE_SEED, "architecture": { "model_type": "gpt2", "layers": 1, "hidden_size": 16, "attention_heads": 2, "vocabulary_size": len(FIXTURE_VOCAB), }, } (destination / "fixture-provenance.json").write_text( json.dumps(manifest, indent=2, sort_keys=True) + "\n", encoding="utf-8", ) return destination