From 13b13ef973db4cadee2a1d0005c60527488e81e8 Mon Sep 17 00:00:00 2001 From: Joseph Magly <1159087+jmagly@users.noreply.github.com> Date: Wed, 5 Aug 2026 13:46:00 -0400 Subject: [PATCH] Make spectral certification sample-aware --- obliteratus/abliterate.py | 12 ++++- .../analysis/spectral_certification.py | 54 ++++++++++++------- obliteratus/architecture_profiles.py | 2 +- obliteratus/tourney.py | 4 +- tests/test_breakthrough_modules.py | 43 +++++++++++++++ 5 files changed, 91 insertions(+), 24 deletions(-) diff --git a/obliteratus/abliterate.py b/obliteratus/abliterate.py index c751498..f35b365 100644 --- a/obliteratus/abliterate.py +++ b/obliteratus/abliterate.py @@ -6431,7 +6431,7 @@ class AbliterationPipeline: self._quality_metrics["kl_divergence"] = None # 5. Spectral certification of abliteration completeness (BBP phase transition) - # Provides a formal guarantee that no linear refusal signal survives. + # Statistical diagnostic for detectable residual linear refusal signal. # We re-collect a small batch of post-abliteration activations on # cert layers (the original activations were freed after DISTILL). self._quality_metrics["spectral_certification"] = None @@ -6470,11 +6470,14 @@ class AbliterationPipeline: if CertificationLevel.RED in levels: overall = "RED (incomplete)" overall_level = "RED" + elif CertificationLevel.INCONCLUSIVE in levels: + overall = "INCONCLUSIVE (insufficient samples)" + overall_level = "INCONCLUSIVE" elif CertificationLevel.YELLOW in levels: overall = "YELLOW (distributed refusal detected)" overall_level = "YELLOW" else: - overall = "GREEN (certified complete)" + overall = "GREEN (no detectable residual signal)" overall_level = "GREEN" self._quality_metrics["spectral_certification"] = overall_level @@ -6493,6 +6496,11 @@ class AbliterationPipeline: elif overall_level == "YELLOW": self.log(" Recommendation: distributed refusal detected — " "consider GRP-Obliteration or 'informed' method") + elif overall_level == "INCONCLUSIVE": + n_required = max(c.n_samples_required for c in cert_results) + n_used = min(c.n_samples_used for c in cert_results) + self.log(f" Recommendation: collect more activation samples " + f"({n_used} used; up to {n_required} recommended)") else: self.log(" Spectral certification: skipped (insufficient activation data)") except Exception as e: diff --git a/obliteratus/analysis/spectral_certification.py b/obliteratus/analysis/spectral_certification.py index 32f04f8..1b010cb 100644 --- a/obliteratus/analysis/spectral_certification.py +++ b/obliteratus/analysis/spectral_certification.py @@ -7,13 +7,14 @@ distributes refusal into many low-energy dimensions, defeating single- direction abliteration. GRP-Obliteration (Russinovich et al., Microsoft, Feb 2026) reorganizes safety representations entirely. -This module uses random matrix theory to build a *spectral certificate* +This module uses random matrix theory to build a *spectral diagnostic* for abliteration completeness. After abliteration, it computes the covariance of residual activations and applies the BBP phase transition to determine whether any detectable refusal signal survives. Contributions: - 1. **Spectral certificate**: Three-tier certification (Green/Yellow/Red) + 1. **Spectral diagnostic**: Green/Yellow/Red classification when the + estimated sample requirement is met, otherwise Inconclusive based on eigenvalue analysis relative to BBP threshold 2. **Non-isotropic BBP extension**: Extends Paper Theorem 4 to anisotropic activation covariance (heuristic extension) @@ -43,7 +44,7 @@ logger = logging.getLogger(__name__) class CertificationLevel(Enum): - """Three-tier certification for abliteration completeness.""" + """Spectral diagnostic outcome for abliteration completeness.""" GREEN = "certified_complete" """All eigenvalues below BBP threshold. No detectable linear refusal @@ -58,10 +59,15 @@ class CertificationLevel(Enum): """Clear eigenvalue spikes above threshold. Abliteration failed to remove all refusal signal. Re-run with more directions.""" + INCONCLUSIVE = "insufficient_samples" + """The observed activation sample is too small for a reliable BBP-based + classification at this dimensionality. Collect more samples before + interpreting the spectral result.""" + @dataclass class SpectralCertificate: - """Formal certificate of abliteration completeness.""" + """Statistical diagnostic of abliteration completeness.""" # Certification level: CertificationLevel @@ -120,8 +126,9 @@ class SpectralCertifier: """Certify abliteration completeness via random matrix theory. Uses the BBP phase transition and Marchenko-Pastur distribution - to provide formal guarantees about whether residual refusal signal - exists in the post-abliteration model. + to assess whether a detectable residual refusal signal exists in the + post-abliteration model. This is a finite-sample diagnostic, not a formal + guarantee; results are inconclusive when the sample requirement is unmet. """ def __init__( @@ -157,7 +164,8 @@ class SpectralCertifier: layer_idx: Layer index (for logging). Returns: - SpectralCertificate with formal certification. + SpectralCertificate with a statistical classification, or an + inconclusive result when the sample requirement is unmet. """ n_h, d = harmful_activations.shape n_b = harmless_activations.shape[0] @@ -239,8 +247,13 @@ class SpectralCertifier: n_required = max(self.min_samples, int(d / max(rho ** 2, 0.01))) is_sufficient = n >= n_required - # Certification level - if n_above == 0 and not is_distributed: + # Certification level. Do not emit a definitive traffic-light result + # when the diagnostic's own sample requirement is unmet: doing so made + # tiny, high-dimensional batches appear authoritatively RED or GREEN. + if not is_sufficient: + level = CertificationLevel.INCONCLUSIVE + confidence = min(0.99, self.confidence_level * (n / max(n_required, 1))) + elif n_above == 0 and not is_distributed: level = CertificationLevel.GREEN confidence = min(0.99, self.confidence_level * (n / max(n_required, 1))) elif is_distributed: @@ -251,9 +264,16 @@ class SpectralCertifier: confidence = min(0.99, self.confidence_level) # Recommendations - if level == CertificationLevel.GREEN: + if level == CertificationLevel.INCONCLUSIVE: recommendation = ( - f"Abliteration is spectrally certified complete. " + f"Spectral diagnostic is inconclusive: only {n} samples were " + f"available, while {n_required} are recommended for reliable " + f"detection at this dimensionality." + ) + action = "more_samples" + elif level == CertificationLevel.GREEN: + recommendation = ( + f"No detectable linear refusal component was found. " f"No linear refusal component with eigenvalue above " f"BBP threshold ({bbp_threshold:.4f}) detected." ) @@ -274,13 +294,6 @@ class SpectralCertifier: ) action = "more_directions" - if not is_sufficient: - recommendation += ( - f" WARNING: Only {n} samples used, {n_required} recommended " - f"for reliable detection at this dimensionality." - ) - action = "more_samples" if level == CertificationLevel.GREEN else action - return SpectralCertificate( level=level, confidence=confidence, @@ -340,10 +353,13 @@ class SpectralCertifier: if not layer_certificates: return None - # Worst level wins + # A sufficient RED result remains actionable. Otherwise, any + # insufficient layer makes the aggregate result inconclusive. levels = [c.level for c in layer_certificates.values()] if CertificationLevel.RED in levels: worst = CertificationLevel.RED + elif CertificationLevel.INCONCLUSIVE in levels: + worst = CertificationLevel.INCONCLUSIVE elif CertificationLevel.YELLOW in levels: worst = CertificationLevel.YELLOW else: diff --git a/obliteratus/architecture_profiles.py b/obliteratus/architecture_profiles.py index bf9a9ee..716bb0f 100644 --- a/obliteratus/architecture_profiles.py +++ b/obliteratus/architecture_profiles.py @@ -430,7 +430,7 @@ def _apply_recommended_defaults(profile: ArchitectureProfile): "Small MoE with reasoning (e.g. Qwen3-30B-A3B in think mode). " "Most fragile combination — MoE expert specialization extends into " "reasoning (Korinsky 2025). Gentle surgical approach. Stop at first " - "GREEN spectral cert to avoid over-ablation." + "Conclusive GREEN spectral diagnostic to avoid over-ablation." ) profile.research_citations = [ "Korinsky 2025: MoE abliteration damages reasoning substantially", diff --git a/obliteratus/tourney.py b/obliteratus/tourney.py index 4f0e9c6..75434ab 100644 --- a/obliteratus/tourney.py +++ b/obliteratus/tourney.py @@ -60,7 +60,7 @@ def composite_score(metrics: dict[str, Any]) -> float: 25% coherence — model must still be useful 20% KL divergence — minimal capability damage 10% perplexity — fluency preservation - 5% spectral cert — formal completeness guarantee + 5% spectral diagnostic — residual-signal assessment 5% degenerate penalty — penalize broken output """ rr = metrics.get("refusal_rate") @@ -115,7 +115,7 @@ class Contender: error: str | None = None round_eliminated: int = 0 # 0 = still alive / winner direction_method: str = "" # which direction extraction was used - spectral_cert: str = "" # GREEN/YELLOW/RED/"" + spectral_cert: str = "" # GREEN/YELLOW/RED/INCONCLUSIVE/"" @dataclass diff --git a/tests/test_breakthrough_modules.py b/tests/test_breakthrough_modules.py index d894f7b..2992670 100644 --- a/tests/test_breakthrough_modules.py +++ b/tests/test_breakthrough_modules.py @@ -616,6 +616,49 @@ class TestSpectralCertification: assert result.n_samples_used == 20 assert result.n_samples_required >= 50 + assert result.is_sample_sufficient is False + assert result.level == CertificationLevel.INCONCLUSIVE + assert result.suggested_action == "more_samples" + + def test_insufficient_strong_signal_is_not_reported_red(self): + """An undersampled spike is evidence to collect data, not a RED result.""" + torch.manual_seed(42) + d = 128 + n = 5 + direction = torch.randn(d) + direction = direction / direction.norm() + harmful = torch.randn(n, d) * 0.3 + 5.0 * direction + harmless = torch.randn(n, d) * 0.3 + + result = SpectralCertifier(min_samples=50).certify(harmful, harmless) + + assert result.n_eigenvalues_above_threshold > 0 + assert result.is_sample_sufficient is False + assert result.level == CertificationLevel.INCONCLUSIVE + assert result.confidence < 0.95 + assert "inconclusive" in result.recommendation.lower() + + def test_overall_prefers_sufficient_red_over_inconclusive(self): + """A reliable RED layer remains actionable in a mixed result set.""" + torch.manual_seed(42) + d = 16 + direction = torch.randn(d) + direction = direction / direction.norm() + sufficient = SpectralCertifier(min_samples=30).certify( + torch.randn(100, d) * 0.3 + 5.0 * direction, + torch.randn(100, d) * 0.3, + ) + insufficient = SpectralCertifier(min_samples=50).certify( + torch.randn(5, d), torch.randn(5, d) + ) + + overall = SpectralCertifier().overall_certification( + {0: insufficient, 1: sufficient} + ) + + assert sufficient.level == CertificationLevel.RED + assert insufficient.level == CertificationLevel.INCONCLUSIVE + assert overall is sufficient def test_certify_all_layers(self): harmful_dict, harmless_dict, _ = _make_multilayer_activations(n_layers=4)