"""Mutation-friendly contracts for model, architecture, and device decisions.""" from __future__ import annotations import pytest from hypothesis import given, strategies as st from obliteratus.runtime_contracts import ( classify_architecture_size, effective_model_memory_gb, quantized_model_fits_gpu, should_snapshot_model, supports_bfloat16_target, validate_model_load_request, ) @pytest.mark.parametrize("model_name", [None, 7, "", " \t"]) def test_model_name_must_be_a_nonempty_string(model_name): with pytest.raises(ValueError) as error: validate_model_load_request(model_name, "causal_lm", None, "float32") assert str(error.value) == "model_name must be a non-empty HuggingFace identifier or local path" @pytest.mark.parametrize( ("field", "value", "message"), [ ("task", "embedding", "Unknown task 'embedding'. Choose from ['causal_lm', 'classification']"), ("quantization", "3bit", "Unknown quantization '3bit'. Choose None, '4bit', or '8bit'"), ("dtype", "float128", "Unknown dtype 'float128'. Choose from ['float32', 'float16', 'bfloat16']"), ], ) def test_invalid_loader_enums_are_rejected(field, value, message): request = { "model_name": "local/model", "task": "causal_lm", "quantization": None, "dtype": "float32", } request[field] = value with pytest.raises(ValueError) as error: validate_model_load_request(**request) assert str(error.value) == message @pytest.mark.parametrize("task", ["causal_lm", "classification"]) @pytest.mark.parametrize("quantization", [None, "4bit", "8bit"]) @pytest.mark.parametrize("dtype", ["float32", "float16", "bfloat16"]) def test_every_documented_loader_enum_combination_is_valid(task, quantization, dtype): validate_model_load_request("local/model", task, quantization, dtype) def test_loader_task_validation_can_follow_the_provider_registry(): validate_model_load_request( "local/model", "custom_task", None, "float32", valid_tasks={"custom_task": object()}, ) @given(st.floats(min_value=0, max_value=10_000, allow_nan=False, allow_infinity=False)) def test_effective_memory_respects_quantization_ratios(estimate_gb): assert effective_model_memory_gb(estimate_gb, None) == estimate_gb assert effective_model_memory_gb(estimate_gb, "4bit") == estimate_gb / 4 assert effective_model_memory_gb(estimate_gb, "8bit") == estimate_gb / 2 @pytest.mark.parametrize( ("estimate_gb", "quantization", "available_gb", "expected"), [ (44.0, "4bit", 16.0, True), (44.8, "4bit", 16.0, False), (22.0, "8bit", 16.0, True), (22.4, "8bit", 16.0, False), (2.0, "4bit", 1.0, True), (1.0, None, 16.0, False), (0.0, "4bit", 16.0, False), (1.0, "4bit", 0.0, False), ], ) def test_quantized_fit_requires_positive_memory_and_thirty_percent_headroom( estimate_gb, quantization, available_gb, expected, ): assert quantized_model_fits_gpu(estimate_gb, quantization, available_gb) is expected def _snapshot(**overrides): values = { "skip_snapshot": None, "initial_gpu_free_gb": 0.0, "remaining_gpu_free_gb": 0.0, "has_native_quantization": False, "estimate_gb": 0.0, "quantization": None, } values.update(overrides) return should_snapshot_model(**values) def test_explicit_snapshot_choice_overrides_every_memory_signal(): constrained = { "initial_gpu_free_gb": 1.0, "remaining_gpu_free_gb": 0.0, "has_native_quantization": True, "estimate_gb": 1000.0, } assert _snapshot(skip_snapshot=True, **constrained) is False assert _snapshot(skip_snapshot=False, **constrained) is True @pytest.mark.parametrize( ("remaining_gb", "expected"), [(3.99, False), (4.0, True), (4.01, True)], ) def test_native_quantization_snapshot_boundary_is_forty_percent(remaining_gb, expected): assert _snapshot( initial_gpu_free_gb=10.0, remaining_gpu_free_gb=remaining_gb, has_native_quantization=True, ) is expected def test_native_snapshot_policy_uses_any_positive_gpu_signal(): assert _snapshot( initial_gpu_free_gb=0.5, remaining_gpu_free_gb=0.1, has_native_quantization=True, ) is False assert _snapshot( initial_gpu_free_gb=0.0, remaining_gpu_free_gb=-1.0, has_native_quantization=True, ) is True @pytest.mark.parametrize( ("estimate_gb", "quantization", "expected"), [(5.0, None, True), (5.01, None, False), (20.0, "4bit", True), (20.04, "4bit", False)], ) def test_estimated_snapshot_boundary_is_half_of_free_memory( estimate_gb, quantization, expected, ): assert _snapshot( initial_gpu_free_gb=10.0, estimate_gb=estimate_gb, quantization=quantization, ) is expected def test_estimated_snapshot_policy_uses_any_positive_gpu_signal(): assert _snapshot(initial_gpu_free_gb=0.5, estimate_gb=0.3) is False assert _snapshot(initial_gpu_free_gb=0.0, estimate_gb=1.0) is True def test_snapshot_defaults_to_enabled_without_a_positive_gpu_signal(): assert _snapshot() is True assert _snapshot(initial_gpu_free_gb=-1.0, estimate_gb=100.0) is True @pytest.mark.parametrize( ("kwargs", "expected"), [ ({"is_moe": False, "total_params_b": 999, "num_experts": 999}, "dense"), ({"is_moe": True, "total_params_b": 99.99, "num_experts": 999}, "small_moe"), ({"is_moe": True, "total_params_b": 0.5, "num_experts": 999}, "small_moe"), ({"is_moe": True, "total_params_b": 100.0, "num_experts": 1}, "large_moe"), ({"is_moe": True, "total_params_b": 0.0, "num_experts": 1}, "small_moe"), ({"is_moe": True, "total_params_b": 0.0, "num_experts": 16}, "small_moe"), ({"is_moe": True, "total_params_b": 0.0, "num_experts": 17}, "large_moe"), ], ) def test_architecture_classification_precedence_and_boundaries(kwargs, expected): assert classify_architecture_size( model_name="custom/model", large_moe_name_patterns=("large-model",), **kwargs, ) == expected def test_architecture_name_fallback_is_case_insensitive_and_conservative(): common = { "is_moe": True, "total_params_b": 0.0, "num_experts": 0, "large_moe_name_patterns": ("giant-moe",), } assert classify_architecture_size(model_name="ORG/GIANT-MOE-V1", **common) == "large_moe" assert classify_architecture_size(model_name="org/unknown-moe", **common) == "small_moe" def test_single_expert_precedes_the_large_name_fallback(): assert classify_architecture_size( is_moe=True, total_params_b=0.0, num_experts=1, model_name="org/giant-moe-v1", large_moe_name_patterns=("giant-moe",), ) == "small_moe" @pytest.mark.parametrize( ("target", "cuda_available", "cuda_major", "version", "expected"), [ ("cuda", True, 8, "2.0.0", True), ("cuda:1", True, 7, "2.0.0", False), ("cuda", False, None, "2.0.0", False), ("mps", False, None, "2.2.9", False), ("mps", False, None, "2.3.0", True), ("mps", False, None, "2.10.0", True), ("cpu", True, 1, "1.0.0", True), ], ) def test_bfloat16_capability_boundaries( target, cuda_available, cuda_major, version, expected, ): assert supports_bfloat16_target( target, cuda_available=cuda_available, cuda_major=cuda_major, torch_version=version, ) is expected