From 941869d9096b91f381d40eae362d27194ab96a17 Mon Sep 17 00:00:00 2001 From: Joseph Magly <1159087+jmagly@users.noreply.github.com> Date: Sat, 29 Aug 2026 03:54:53 -0400 Subject: [PATCH] feat: add tune-only Qwen3.8 E02 candidate (#185) --- app.py | 108 ++++++++++++++++++-------- docs/QWEN38_27B_RESEARCH_ROADMAP.md | 32 ++++---- docs/qwen38-experiment-matrix-v1.json | 16 +++- obliteratus/abliterate.py | 31 ++++++++ obliteratus/experiment_protocol.py | 15 ++++ tests/test_abliterate.py | 2 +- tests/test_experiment_protocol.py | 16 +++- 7 files changed, 170 insertions(+), 50 deletions(-) diff --git a/app.py b/app.py index de6ce08..5c23316 100644 --- a/app.py +++ b/app.py @@ -533,6 +533,7 @@ MODELS = _build_model_choices() METHODS = { "adaptive (telemetry-recommended)": "adaptive", "Qwen3.8 E01 (held-out causal baseline)": "qwen38_e01", + "Qwen3.8 E02 (tune-only multi-direction)": "qwen38_e02", "advanced (recommended)": "advanced", "basic (fast, single direction)": "basic", "aggressive (maximum removal)": "aggressive", @@ -2284,19 +2285,40 @@ def obliterate(model_choice: str, method_choice: str, "harmful": hashlib.sha256((custom_harmful or "").encode()).hexdigest(), "harmless": hashlib.sha256((custom_harmless or "").encode()).hexdigest(), } - if method == "qwen38_e01": + if method in {"qwen38_e01", "qwen38_e02"}: + experiment = "E01" if method == "qwen38_e01" else "E02" + experiment_settings = { + "E01": { + "direction_method": "diff_means", + "n_directions": 1, + "regularization": 0.0, + "refinement_passes": 1, + "norm_preserve": False, + "layer_selection": "all_except_first", + "projection_target": "attention", + "verify_sample_size": 200, + "evaluation_split": "final_test", + }, + "E02": { + "direction_method": "svd", + "n_directions": 4, + "regularization": 0.0, + "refinement_passes": 1, + "norm_preserve": True, + "layer_selection": "middle60", + "projection_target": "output", + "rdo_refinement": True, + "winsorize_activations": True, + "use_kl_optimization": True, + "verify_sample_size": 142, + "evaluation_split": "optimizer_tune", + }, + }[experiment] run_config["immutable_experiment"] = { "protocol": "qwen38-v1", - "experiment": "E01", - "direction_method": "diff_means", - "n_directions": 1, - "regularization": 0.0, - "refinement_passes": 1, - "norm_preserve": False, + "experiment": experiment, "use_chat_template": True, - "layer_selection": "all_except_first", - "projection_target": "attention", - "verify_sample_size": 200, + **experiment_settings, } try: run_archive = RunArchive() @@ -2394,18 +2416,28 @@ def obliterate(model_choice: str, method_choice: str, evaluation_harmful = None evaluation_harmless = None - if method == "qwen38_e01": + if method in {"qwen38_e01", "qwen38_e02"}: if model_id.rstrip("/").lower() != "qwen/qwen3.8-27b": - raise ValueError("Qwen3.8 E01 requires Qwen/Qwen3.8-27B") + raise ValueError("Qwen3.8 experiments require Qwen/Qwen3.8-27B") if use_custom or dataset_key != "builtin": - raise ValueError("Qwen3.8 E01 requires the immutable built-in corpus") - from obliteratus.experiment_protocol import build_qwen38_split + raise ValueError( + "Qwen3.8 experiments require the immutable built-in corpus" + ) + from obliteratus.experiment_protocol import ( + build_qwen38_split, + qwen38_evaluation_pairs, + ) split = build_qwen38_split(harmful_all, harmless_all) train_harmful = [pair[0] for pair in split.train] train_harmless = [pair[1] for pair in split.train] - evaluation_harmful = [pair[0] for pair in split.test] - evaluation_harmless = [pair[1] for pair in split.test] + experiment = "E01" if method == "qwen38_e01" else "E02" + evaluation_pairs = qwen38_evaluation_pairs( + split, + experiment, + ) + evaluation_harmful = [pair[0] for pair in evaluation_pairs] + evaluation_harmless = [pair[1] for pair in evaluation_pairs] harmful_selected = train_harmful harmless_selected = train_harmless n = len(train_harmful) @@ -2414,8 +2446,13 @@ def obliterate(model_choice: str, method_choice: str, "Experiment protocol qwen38-v1: 500 direction-train, " "142 optimizer-tune, 200 untouched final-test pairs" ) + on_log( + f"{experiment} evaluation partition: " + f"{'optimizer-tune' if experiment == 'E02' else 'final-test'} " + f"({len(evaluation_pairs)} pairs)" + ) on_log(f"Split manifest: {split.manifest['manifest_sha256']}") - immutable_e01 = True + immutable_qwen38 = True else: # Apply volume cap (-1 = use all) if prompt_volume > 0: @@ -2424,7 +2461,7 @@ def obliterate(model_choice: str, method_choice: str, n = min(len(harmful_all), len(harmless_all)) harmful_selected = harmful_all[:n] harmless_selected = harmless_all[:n] - immutable_e01 = False + immutable_qwen38 = False run_archive.record_dataset( run_id, identifier="custom" if use_custom else dataset_key, @@ -2452,15 +2489,16 @@ def obliterate(model_choice: str, method_choice: str, pipeline.run_informed() else: from obliteratus.abliterate import AbliterationPipeline - if immutable_e01: - # E01 is immutable: ignore mutable advanced controls and - # run the registered causal baseline exactly as reviewed. + if immutable_qwen38: + # Promotion experiments are immutable: ignore mutable + # advanced controls and run the registered candidate. + is_e02 = method == "qwen38_e02" advanced_options = { - "n_directions": 1, - "direction_method": "diff_means", + "n_directions": 4 if is_e02 else 1, + "direction_method": "svd" if is_e02 else "diff_means", "regularization": 0.0, "refinement_passes": 1, - "norm_preserve": False, + "norm_preserve": is_e02, "project_biases": False, "use_chat_template": True, "use_whitened_svd": False, @@ -2477,13 +2515,13 @@ def obliterate(model_choice: str, method_choice: str, "expert_transplant": False, "use_wasserstein_optimal": False, "spectral_cascade": False, - "layer_selection": "all_except_first", - "winsorize_activations": False, - "use_kl_optimization": False, + "layer_selection": "middle60" if is_e02 else "all_except_first", + "winsorize_activations": is_e02, + "use_kl_optimization": is_e02, "float_layer_interpolation": False, - "rdo_refinement": False, + "rdo_refinement": is_e02, "cot_aware": False, - "verify_sample_size": 200, + "verify_sample_size": 142 if is_e02 else 200, } else: advanced_options = { @@ -2537,8 +2575,16 @@ def obliterate(model_choice: str, method_choice: str, transplant_blend=float(adv_transplant_blend), spectral_bands=int(adv_spectral_bands), spectral_threshold=float(adv_spectral_threshold), - winsorize_percentile=float(adv_winsorize_percentile), - kl_budget=float(adv_kl_budget), + winsorize_percentile=( + 0.01 + if immutable_qwen38 and method == "qwen38_e02" + else float(adv_winsorize_percentile) + ), + kl_budget=( + 0.5 + if immutable_qwen38 and method == "qwen38_e02" + else float(adv_kl_budget) + ), n_sae_features=int(adv_n_sae_features), **advanced_options, ) diff --git a/docs/QWEN38_27B_RESEARCH_ROADMAP.md b/docs/QWEN38_27B_RESEARCH_ROADMAP.md index 044b390..04db834 100644 --- a/docs/QWEN38_27B_RESEARCH_ROADMAP.md +++ b/docs/QWEN38_27B_RESEARCH_ROADMAP.md @@ -2,15 +2,18 @@ ## Decision summary -The current Qwen3.8-27B result is a useful capability-preserving control, not a -successful refusal-removal result: +The promotion-grade E01 run is a useful causal control, not a successful +refusal-removal result: -- refusal rate: 90.9% (30/33 still refusing); -- coherence: 80% (8/10); -- perplexity: 3.11; -- first-token KL divergence: 0.2405; -- 39 middle-depth layers and 312 matrices modified; -- 13 layers partially reverted by the current correction pass. +- archive run: `run-4787f0f2c2ee4a6b8ca8840684eff99b`; +- refusal rate: 92% (184/200 still refusing); +- coherence: 70% (7/10); +- capability: 83% (5/6 checks); +- perplexity: 3.06; +- sequence-token KL divergence: 0.0513; +- first-token KL divergence: 0.1235; +- 62 layers and 62 attention-output matrices modified; +- checkpoint: 53,812,173,294 bytes, 35 files, all SHA-256 inventoried. Do not increase surgery strength or enable the full advanced toggle set yet. The next engineering release should make the optimizer and evaluation reliable, @@ -211,7 +214,7 @@ method needs an isolated ablation and an interaction test before composition. |---|---|---|---|---| | E00 | Is the pristine evaluator stable? | final-test only, 3 seeds | none | metric variance and baselines recorded | | E01 | Does the Arditi control work? | 400/100 | DIM, one layer/direction sweep | lower refusal with <=1.25x PPL | -| E02 | How much data is needed? | 33/64/128/256/400 | fixed DIM/SVD setup | direction and outcome stability plateau | +| E02 | Does a low-rank writer intervention improve E01? | 500/142; final test prohibited | SVD-4 + RDO, middle-60% residual writers, KL rollback | tune refusal <30% and coherence >=80% | | E03 | Which architecture component carries refusal? | 400/100 | DeltaNet vs full-attn vs MLP groups | best causal effect per KL unit | | E04 | Does SVD rank help? | 400/100 | k=1..7, joint held-out selection | nondominated gain over E01 | | E05 | Does RDO improve targeting? | 400/100 | one and multiple RDO directions | held-out gain over E01/E04 | @@ -253,12 +256,11 @@ from 60% refusal in S2 to 100% in several other strata. ## Recommended next operational run -Do not treat another 54 GB permanent checkpoint as the next experiment. After -P0 and hook-mode P1 exist, run E00--E03 first. If an immediate pre-fix diagnostic -is unavoidable, use the last stable settings but increase direction-training -volume to at least 256 pairs, keep all destructive optional techniques off, and -do not promote or publish the output because the current KL and spectral gates -are not trustworthy enough for release qualification. +Run the pre-registered E02 candidate on the 142-pair optimizer-tune partition. +E02 must not read the 200-pair final partition. Retain its complete archive even +if it fails. If it misses either target, use only tune evidence to define the +next isolated ablation. If it passes both targets, reload that saved checkpoint +and run the final partition exactly once for release qualification. ## Evidence base diff --git a/docs/qwen38-experiment-matrix-v1.json b/docs/qwen38-experiment-matrix-v1.json index 0f39f0f..4ca08d9 100644 --- a/docs/qwen38-experiment-matrix-v1.json +++ b/docs/qwen38-experiment-matrix-v1.json @@ -42,8 +42,20 @@ }, { "id": "E02", - "purpose": "direction training-volume ablation", - "depends_on": "E01" + "purpose": "tune-only multi-direction residual-writer candidate", + "depends_on": "E01", + "method": "qwen38_e02", + "evaluation_split": "optimizer_tune", + "direction_method": "svd", + "directions": 4, + "layer_selection": "middle60", + "projection_target": "output", + "regularization": 0.0, + "norm_preserve": true, + "refinement_passes": 1, + "rdo_refinement": true, + "winsorize_activations": true, + "kl_optimization": true }, { "id": "E03", diff --git a/obliteratus/abliterate.py b/obliteratus/abliterate.py index 054c593..24fe35d 100644 --- a/obliteratus/abliterate.py +++ b/obliteratus/abliterate.py @@ -218,6 +218,37 @@ METHODS = { "projection_target": "attention", "verify_sample_size": 200, }, + "qwen38_e02": { + "label": "Qwen3.8 E02 (held-out tune candidate)", + "description": ( + "Pre-registered Qwen3.8-27B tune-split candidate: a four-direction " + "refusal subspace projected from validated residual writers in the " + "middle 60% of layers, with RDO refinement and KL rollback." + ), + "n_directions": 4, + "direction_method": "svd", + "norm_preserve": True, + "regularization": 0.0, + "refinement_passes": 1, + "project_biases": False, + "use_chat_template": True, + "use_whitened_svd": False, + "true_iterative_refinement": False, + "use_jailbreak_contrast": False, + "layer_adaptive_strength": False, + "safety_neuron_masking": False, + "per_expert_directions": False, + "attention_head_surgery": False, + "use_sae_features": False, + "invert_refusal": False, + "use_kl_optimization": True, + "winsorize_activations": True, + "winsorize_percentile": 0.01, + "layer_selection": "middle60", + "projection_target": "output", + "rdo_refinement": True, + "verify_sample_size": 142, + }, "basic": { "label": "Basic (Arditi et al.)", "description": "Single refusal direction via difference-in-means", diff --git a/obliteratus/experiment_protocol.py b/obliteratus/experiment_protocol.py index 36a4492..022c6fe 100644 --- a/obliteratus/experiment_protocol.py +++ b/obliteratus/experiment_protocol.py @@ -25,6 +25,21 @@ class PromptSplit: manifest: dict[str, object] +def qwen38_evaluation_pairs( + split: PromptSplit, + experiment: str, +) -> tuple[tuple[str, str], ...]: + """Return the only evaluation partition authorized for an experiment.""" + partitions = { + "E01": split.test, + "E02": split.tune, + } + try: + return partitions[experiment] + except KeyError as exc: + raise ValueError(f"unregistered Qwen3.8 experiment: {experiment}") from exc + + def build_qwen38_split( harmful: list[str], harmless: list[str], diff --git a/tests/test_abliterate.py b/tests/test_abliterate.py index d87acc8..ef2c4ab 100644 --- a/tests/test_abliterate.py +++ b/tests/test_abliterate.py @@ -163,7 +163,7 @@ class TestStages: class TestMethods: def test_methods_exist(self): - assert set(METHODS.keys()) == {"basic", "advanced", "aggressive", "informed", "surgical", "inverted", "nuclear", "optimized", "failspy", "gabliteration", "heretic", "rdo", "spectral_cascade", "som", "qwen38_e01"} + assert set(METHODS.keys()) == {"basic", "advanced", "aggressive", "informed", "surgical", "inverted", "nuclear", "optimized", "failspy", "gabliteration", "heretic", "rdo", "spectral_cascade", "som", "qwen38_e01", "qwen38_e02"} def test_basic_single_direction(self): cfg = METHODS["basic"] diff --git a/tests/test_experiment_protocol.py b/tests/test_experiment_protocol.py index f73e2b2..6795cc1 100644 --- a/tests/test_experiment_protocol.py +++ b/tests/test_experiment_protocol.py @@ -4,7 +4,10 @@ import json import pytest -from obliteratus.experiment_protocol import build_qwen38_split +from obliteratus.experiment_protocol import ( + build_qwen38_split, + qwen38_evaluation_pairs, +) def _corpus(size: int = 842) -> tuple[list[str], list[str]]: @@ -41,6 +44,17 @@ def test_qwen38_split_manifest_contains_no_prompt_text(): assert len(split.manifest["manifest_sha256"]) == 64 +def test_qwen38_experiment_evaluation_partitions_are_fail_closed(): + harmful, harmless = _corpus() + split = build_qwen38_split(harmful, harmless) + + assert qwen38_evaluation_pairs(split, "E01") is split.test + assert qwen38_evaluation_pairs(split, "E02") is split.tune + assert not set(split.test) & set(qwen38_evaluation_pairs(split, "E02")) + with pytest.raises(ValueError, match="unregistered Qwen3.8 experiment"): + qwen38_evaluation_pairs(split, "E03") + + def test_qwen38_split_rejects_wrong_size_and_duplicate_pairs(): harmful, harmless = _corpus(4) with pytest.raises(ValueError, match="split sizes total"):