Files
OBLITERATUS/tests/test_property_contracts.py
T

125 lines
4.0 KiB
Python

"""Deterministic property contracts for high-consequence pure behavior."""
from __future__ import annotations
import pytest
import torch
from hypothesis import given, seed, settings, strategies as st
from obliteratus.evaluation.advanced_metrics import (
_is_refusal,
linear_cka,
token_kl_divergence,
)
from obliteratus.evaluation.metrics import accuracy, f1_score_metric, perplexity
PROPERTY_SETTINGS = settings(max_examples=60, deadline=None, database=None)
@seed(7001)
@PROPERTY_SETTINGS
@given(st.lists(st.integers(-5, 5), max_size=50))
def test_accuracy_is_invariant_to_joint_reversal_and_duplication(values):
references = [value % 3 for value in values]
predictions = [value if index % 4 else value + 1 for index, value in enumerate(references)]
expected = accuracy(predictions, references)
assert accuracy(list(reversed(predictions)), list(reversed(references))) == expected
if values:
assert accuracy(predictions * 2, references * 2) == expected
@seed(7002)
@PROPERTY_SETTINGS
@given(
st.lists(st.integers(0, 4), min_size=1, max_size=50),
st.lists(st.booleans(), min_size=1, max_size=50),
)
def test_f1_is_invariant_to_bijective_label_renaming(references, flips):
predictions = [
value if flips[index % len(flips)] else (value + 1) % 5
for index, value in enumerate(references)
]
expected = f1_score_metric(predictions, references)
assert f1_score_metric(
[value + 10 for value in predictions],
[value + 10 for value in references],
) == pytest.approx(expected)
@seed(7003)
@PROPERTY_SETTINGS
@given(
vocab_size=st.integers(2, 40),
batch_size=st.integers(1, 4),
sequence_length=st.integers(2, 12),
)
def test_uniform_causal_lm_perplexity_equals_vocabulary_size(
vocab_size, batch_size, sequence_length,
):
logits = torch.zeros(batch_size, sequence_length, vocab_size)
labels = torch.arange(batch_size * sequence_length).reshape(batch_size, sequence_length)
labels %= vocab_size
assert perplexity(logits, labels) == pytest.approx(float(vocab_size), rel=1e-6)
@seed(7004)
@PROPERTY_SETTINGS
@given(
original=st.lists(
st.floats(-20, 20, allow_nan=False, allow_infinity=False),
min_size=24,
max_size=24,
),
modified=st.lists(
st.floats(-20, 20, allow_nan=False, allow_infinity=False),
min_size=24,
max_size=24,
),
offset=st.floats(-100, 100, allow_nan=False, allow_infinity=False),
)
def test_token_kl_is_nonnegative_and_common_logit_shift_invariant(
original, modified, offset,
):
original_logits = torch.tensor(original).reshape(2, 3, 4)
modified_logits = torch.tensor(modified).reshape(2, 3, 4)
expected = token_kl_divergence(original_logits, modified_logits)
shifted = token_kl_divergence(original_logits + offset, modified_logits - offset)
assert expected >= -1e-5
assert shifted == pytest.approx(expected, abs=2e-5)
@seed(7005)
@PROPERTY_SETTINGS
@given(
scale=st.floats(
min_value=0.1, max_value=20, allow_nan=False, allow_infinity=False,
),
offset=st.floats(-20, 20, allow_nan=False, allow_infinity=False),
)
def test_linear_cka_is_invariant_to_isotropic_scale_and_translation(scale, offset):
generator = torch.Generator().manual_seed(7005)
representation = torch.randn(12, 6, generator=generator)
assert linear_cka(representation, representation * scale + offset) == pytest.approx(
1.0, abs=1e-5,
)
@seed(7006)
@PROPERTY_SETTINGS
@given(
refusal=st.sampled_from([
"I'm sorry, I cannot help with that.",
"I must decline this request.",
"Je ne peux pas vous aider.",
"我无法帮助这个请求。",
]),
leading_space=st.integers(0, 8),
upper=st.booleans(),
)
def test_refusal_classification_is_invariant_to_case_and_leading_space(
refusal, leading_space, upper,
):
transformed = " " * leading_space + (refusal.upper() if upper else refusal.lower())
assert _is_refusal(transformed, mode="combined")