Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
24 changes: 16 additions & 8 deletions src/modelinfo/architecture.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,7 +17,8 @@ def extract_architecture(tensors: Dict[str, Any], config: Dict[str, Any] = None)
if gen_arch:
arch_str = str(gen_arch)
num_layers = metadata.get(f"{arch_str}.block_count", 0)
kv_heads = metadata.get(f"{arch_str}.attention.head_count_kv", 0)
kv_heads = metadata.get(f"{arch_str}.attention.head_count_kv",
metadata.get(f"{arch_str}.attention.head_count", 0))

key_length = metadata.get(f"{arch_str}.attention.key_length")
if not key_length:
Expand All @@ -35,13 +36,16 @@ def extract_architecture(tensors: Dict[str, Any], config: Dict[str, Any] = None)

# 2. Attempt explicit SafeTensors config.json
if config:
num_layers = config.get("num_hidden_layers", 0)
num_attention_heads = config.get("num_attention_heads", 1)
# Multimodal models keep the decoder dimensions in text_config.
if isinstance(config.get("text_config"), dict):
config = config["text_config"]
num_layers = config.get("num_hidden_layers", config.get("n_layer", 0))
num_attention_heads = config.get("num_attention_heads", config.get("n_head", 1))
num_key_value_heads = config.get("num_key_value_heads", num_attention_heads)
hidden_size = config.get("hidden_size", 0)
hidden_size = config.get("hidden_size", config.get("n_embd", 0))

if num_attention_heads > 0:
head_dim = hidden_size // num_attention_heads
head_dim = config.get("head_dim") or hidden_size // num_attention_heads
kv_dim = num_key_value_heads * head_dim
if num_layers > 0 and kv_dim > 0:
return num_layers, kv_dim, False
Expand All @@ -65,18 +69,22 @@ def extract_architecture(tensors: Dict[str, Any], config: Dict[str, Any] = None)
if len(parts) > idx + 1 and parts[idx+1].isdigit():
layers_set.add(int(parts[idx+1]))

if name.endswith("k_proj.weight") or name.endswith("attn.k.weight") or name.endswith("k_proj.w"):
if is_gguf and len(parts) > 1 and parts[0] == "blk" and parts[1].isdigit():
layers_set.add(int(parts[1]))

if (is_gguf and name.endswith("attn_k.weight")) or name.endswith("k_proj.weight") or name.endswith("attn.k.weight") or name.endswith("k_proj.w"):
found_k_proj = True
shape = meta.get("shape", [])
if len(shape) >= 2:
kv_dim = shape[-1] if is_gguf else shape[0]

if "qkv_proj.weight" in name or "c_attn.weight" in name:
if "qkv_proj.weight" in name or "c_attn.weight" in name or (is_gguf and name.endswith("attn_qkv.weight")):
found_fused = True
if not found_k_proj:
shape = meta.get("shape", [])
if len(shape) >= 2:
kv_dim = (shape[-1] if is_gguf else shape[0]) // 3
output_dim = shape[-1] if is_gguf or name.endswith("c_attn.weight") else shape[0]
kv_dim = output_dim // 3

num_layers = len(layers_set)
if found_fused and not found_k_proj and kv_dim > 0:
Expand Down
53 changes: 35 additions & 18 deletions src/modelinfo/calculator.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,27 +17,33 @@
"I8": 1.0,
"U64": 8.0,
"U32": 4.0,
"U16": 2.0,
"U8": 1.0,
"BOOL": 1.0,
"Q8_0": 1.0625,
"Q8_1": 1.0625,
"Q8_K": 1.0625,
"Q6_K": 0.828125,
"Q8_1": 1.125,
"Q8_K": 1.140625,
"Q6_K": 0.8203125,
"Q5_0": 0.6875,
"Q5_1": 0.75,
"Q5_K": 0.6875,
"Q4_0": 0.5625,
"Q4_1": 0.625,
"Q4_K": 0.59375,
"Q3_K": 0.4375,
"Q2_K": 0.34375,
"IQ4_NL": 0.53125,
"Q4_K": 0.5625,
"Q3_K": 0.4296875,
"Q2_K": 0.328125,
"IQ4_NL": 0.5625,
"IQ4_XS": 0.53125,
"IQ3_S": 0.4375,
"IQ3_XXS": 0.385,
"IQ2_S": 0.3125,
"IQ2_XS": 0.296875,
"IQ2_XXS": 0.28125,
"IQ3_S": 0.4296875,
"IQ3_XXS": 0.3828125,
"IQ2_S": 0.3203125,
"IQ2_XS": 0.2890625,
"IQ2_XXS": 0.2578125,
"IQ1_M": 0.21875,
"IQ1_S": 0.1953125,
"TQ1_0": 54 / 256,
"TQ2_0": 66 / 256,
"MXFP4": 17 / 32,
"Q8": 1.06,
"Q6": 0.82,
"Q5": 0.68,
Expand Down Expand Up @@ -73,19 +79,30 @@

if is_lazy:
base_memory_bytes = tensors.get("__metadata__", {}).get("total_size", 0.0)
# Assume predominantly FP16/BF16 for modern Hub architectures
primary_dtype = "BF16"
# Lazy mode estimates parameter count from the configured weight dtype.
# Retain the BF16 fallback when no recognized dtype is available.
dtype_config = config or {}
if isinstance(dtype_config.get("text_config"), dict):
dtype_config = dtype_config["text_config"]
configured_dtype = dtype_config.get("dtype") or dtype_config.get("torch_dtype")
primary_dtype = {
"float64": "F64", "float32": "F32", "float16": "F16",
"bfloat16": "BF16",
}.get(str(configured_dtype).lower(), "BF16")
dtype_counts[primary_dtype] = 1
total_params = int(base_memory_bytes / 2.0)
total_params = int(base_memory_bytes / _get_bytes_per_param(primary_dtype))
else:
for name, metadata in tensors.items():
if name == "__metadata__":
continue

shape = metadata.get("shape", [])
if not shape:
shape = metadata.get("shape")
if shape is None:
continue

if (not isinstance(shape, (list, tuple))
or any(type(dimension) is not int or dimension < 0 for dimension in shape)):

Check warning on line 103 in src/modelinfo/calculator.py

View check run for this annotation

Codacy Production / Codacy Static Code Analysis

src/modelinfo/calculator.py#L103

Use isinstance() rather than type() for a typecheck.
raise ValueError(f"Invalid tensor shape for {name!r}: expected nonnegative integer dimensions")

param_count = math.prod(shape)
total_params += param_count

Expand Down
1 change: 1 addition & 0 deletions src/modelinfo/parsers/gguf.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,7 @@
18: "IQ3_XXS", 19: "IQ1_S", 20: "IQ4_NL", 21: "IQ3_S", 22: "IQ2_S",
23: "IQ4_XS", 24: "I8", 25: "I16", 26: "I32", 27: "I64", 28: "F64",
29: "IQ1_M", 30: "BF16", 31: "Q4_0_4_4", 32: "Q4_0_4_8", 33: "Q4_0_8_8",
34: "TQ1_0", 35: "TQ2_0", 39: "MXFP4",
}

def _read_gguf_value(f: Any, val_type: int) -> Any:
Expand Down
4 changes: 2 additions & 2 deletions tests/test_calculator.py
Original file line number Diff line number Diff line change
Expand Up @@ -115,8 +115,8 @@ def test_framework_overhead_included():
def test_explicit_gguf_quantization_byte_multipliers():
"""Verify that explicit ggml_type enums are exactly mapped."""
assert _get_bytes_per_param("Q8_0") == 1.0625
assert _get_bytes_per_param("Q4_K") == 0.59375
assert _get_bytes_per_param("IQ2_XXS") == 0.28125
assert _get_bytes_per_param("Q4_K") == 144 / 256
assert _get_bytes_per_param("IQ2_XXS") == 66 / 256
assert _get_bytes_per_param("F8_E5M2") == 1.0

def test_topology_penalties():
Expand Down
17 changes: 17 additions & 0 deletions tests/test_explicit_head_dimension.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,17 @@
from modelinfo.architecture import extract_architecture
from modelinfo.calculator import calculate_footprint


def test_explicit_head_dimension_overrides_hidden_size_quotient():
# Gemma2 2B configuration: 2304 / 8 is 288, but head_dim is 256.
config = {"num_hidden_layers": 26, "num_attention_heads": 8,
"num_key_value_heads": 4, "hidden_size": 2304, "head_dim": 256}
assert extract_architecture({}, config) == (26, 1024, False)
result = calculate_footprint({}, config=config, context_length=100)
assert result["kv_cache_bytes"] == 4 * 26 * 1024 * 100


def test_config_without_explicit_dimension_retains_quotient():
config = {"num_hidden_layers": 2, "num_attention_heads": 4,
"hidden_size": 512}
assert extract_architecture({}, config) == (2, 512, False)
14 changes: 14 additions & 0 deletions tests/test_gguf_mha_default.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,14 @@
from modelinfo.architecture import extract_architecture


def test_gguf_omitted_kv_heads_defaults_to_attention_heads():
metadata = {'general.architecture': 'llama', 'llama.block_count': 32,
'llama.embedding_length': 4096, 'llama.attention.head_count': 32}
assert extract_architecture({'__metadata__': metadata}) == (32, 4096, False)


def test_gguf_explicit_grouped_query_heads_are_preserved():
metadata = {'general.architecture': 'llama', 'llama.block_count': 32,
'llama.embedding_length': 4096, 'llama.attention.head_count': 32,
'llama.attention.head_count_kv': 8}
assert extract_architecture({'__metadata__': metadata}) == (32, 1024, False)
18 changes: 18 additions & 0 deletions tests/test_gguf_native_names.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,18 @@
from modelinfo.architecture import extract_architecture


def test_native_gguf_key_projection_names():
tensors = {
"__metadata__": {"general.architecture": "llama"},
"blk.0.attn_k.weight": {"shape": [4096, 1024]},
"blk.1.attn_k.weight": {"shape": [4096, 1024]},
}
assert extract_architecture(tensors) == (2, 1024, False)


def test_native_gguf_fused_projection_names():
tensors = {
"__metadata__": {"general.architecture": "gpt2"},
"blk.0.attn_qkv.weight": {"shape": [768, 2304]},
}
assert extract_architecture(tensors) == (1, 768, True)
16 changes: 16 additions & 0 deletions tests/test_gpt2_config.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,16 @@
"""GPT2Config serializes n_layer/n_head/n_embd rather than generic aliases."""
from modelinfo.architecture import extract_architecture
from modelinfo.calculator import calculate_footprint


def test_gpt2_config_provides_exact_architecture_without_tensor_scan():
config = {"model_type": "gpt2", "n_layer": 12, "n_head": 12, "n_embd": 768}
assert extract_architecture({}, config) == (12, 768, False)
result = calculate_footprint({}, context_length=1024, config=config)
assert result["kv_cache_bytes"] == 4 * 12 * 768 * 1024


def test_generic_config_fields_take_precedence_over_gpt2_aliases():
config = {"n_layer": 12, "n_head": 12, "n_embd": 768,
"num_hidden_layers": 2, "num_attention_heads": 4, "hidden_size": 256}
assert extract_architecture({}, config) == (2, 256, False)
11 changes: 11 additions & 0 deletions tests/test_gpt2_projection_layout.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,11 @@
from modelinfo.architecture import extract_architecture


def test_gpt2_conv1d_projection_uses_output_axis():
tensors = {"transformer.h.0.attn.c_attn.weight": {"shape": [768, 2304]}}
assert extract_architecture(tensors) == (1, 768, True)


def test_linear_qkv_projection_retains_output_first_layout():
tensors = {"model.layers.0.qkv_proj.weight": {"shape": [2304, 768]}}
assert extract_architecture(tensors) == (1, 768, True)
15 changes: 15 additions & 0 deletions tests/test_invalid_tensor_shapes.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,15 @@
import pytest
from modelinfo.calculator import calculate_footprint


@pytest.mark.parametrize('shape', [[-1, 4], [True, 4], [1.5, 4], ['3', 4], '34'])
def test_invalid_tensor_dimensions_are_rejected(shape):
with pytest.raises(ValueError, match='shape'):
calculate_footprint({'weight': {'shape': shape, 'dtype': 'F16'}})


def test_zero_length_dimensions_and_scalar_shapes_remain_valid():
result = calculate_footprint({'empty': {'shape': [0, 4], 'dtype': 'F16'},
'scalar': {'shape': [], 'dtype': 'F32'}})
assert result['total_params'] == 1
assert result['base_memory_bytes'] == 4
15 changes: 15 additions & 0 deletions tests/test_lazy_dtype.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,15 @@
import pytest

from modelinfo.calculator import calculate_footprint


@pytest.mark.parametrize('key,dtype,expected,width', [
('torch_dtype', 'float32', 'F32', 4),
('dtype', 'float64', 'F64', 8),
('torch_dtype', 'float16', 'F16', 2),
])
def test_lazy_parameter_estimate_respects_declared_dtype(key, dtype, expected, width):
result = calculate_footprint({'__metadata__': {'lazy_fetch': True, 'total_size': 1024}}, config={key: dtype})
assert result['primary_dtype'] == expected
assert result['total_params'] == 1024 // width
assert result['base_memory_bytes'] == 1024
21 changes: 21 additions & 0 deletions tests/test_multimodal_text_config.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,21 @@
from copy import deepcopy

from modelinfo.calculator import calculate_footprint


def test_lazy_multimodal_checkpoint_uses_text_backbone_kv_dimensions():
config = {
'model_type': 'llava',
'vision_config': {'hidden_size': 1024, 'num_hidden_layers': 24},
'text_config': {
'hidden_size': 4096, 'num_hidden_layers': 32,
'num_attention_heads': 32, 'num_key_value_heads': 8,
},
}
original = deepcopy(config)
tensors = {'__metadata__': {'lazy_fetch': True, 'total_size': 1024}}
result = calculate_footprint(tensors, context_length=2048, config=config)
assert result['num_layers'] == 32
assert result['kv_dim'] == 1024
assert result['kv_cache_bytes'] == 2 * 32 * 1024 * 2048 * 2
assert config == original
9 changes: 9 additions & 0 deletions tests/test_primitive_dtype_sizes.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,9 @@
import pytest

from modelinfo.calculator import calculate_footprint


@pytest.mark.parametrize(("dtype", "size"), [("BOOL", 1), ("U8", 1), ("U16", 2)])
def test_unsigned_and_boolean_tensor_sizes(dtype, size):
result = calculate_footprint({"buffer": {"dtype": dtype, "shape": [256]}})
assert result["base_memory_bytes"] == 256 * size
35 changes: 35 additions & 0 deletions tests/test_quantized_block_sizes.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,35 @@
"""Expected byte lengths verified with sizeof GGML quantization block structures.

Reference: https://github.com/ggml-org/ggml/blob/master/src/ggml-common.h
This tests packed tensor storage, excluding GGUF file metadata and alignment.
"""
import pytest
from modelinfo.calculator import calculate_footprint


@pytest.mark.parametrize("dtype,block_size,block_bytes", [
('Q8_0', 32, 34),
('Q8_1', 32, 36),
('Q8_K', 256, 292),
('Q6_K', 256, 210),
('Q5_0', 32, 22),
('Q5_1', 32, 24),
('Q5_K', 256, 176),
('Q4_0', 32, 18),
('Q4_1', 32, 20),
('Q4_K', 256, 144),
('Q3_K', 256, 110),
('Q2_K', 256, 84),
('IQ4_NL', 32, 18),
('IQ4_XS', 256, 136),
('IQ3_S', 256, 110),
('IQ3_XXS', 256, 98),
('IQ2_S', 256, 82),
('IQ2_XS', 256, 74),
('IQ2_XXS', 256, 66),
('IQ1_M', 256, 56),
('IQ1_S', 256, 50),
])
def test_quantized_tensor_storage_matches_ggml_blocks(dtype, block_size, block_bytes):
result = calculate_footprint({"weight": {"shape": [3, block_size], "dtype": dtype}})
assert result["base_memory_bytes"] == 3 * block_bytes
19 changes: 19 additions & 0 deletions tests/test_recent_gguf_types.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,19 @@
import io
import struct

import pytest

from modelinfo.calculator import calculate_footprint
from modelinfo.parsers.gguf import parse_gguf_header


@pytest.mark.parametrize('type_id,dtype,block_size,block_bytes', [
(34, 'TQ1_0', 256, 54), (35, 'TQ2_0', 256, 66), (39, 'MXFP4', 32, 17),
])
def test_gguf_ternary_and_mxfp4_memory(type_id, dtype, block_size, block_bytes):
payload = (b'GGUF' + struct.pack('<IQQ', 3, 1, 0)
+ struct.pack('<Q', 6) + b'weight'
+ struct.pack('<IQIQ', 1, block_size, type_id, 0))
tensors = parse_gguf_header(io.BytesIO(payload))
assert tensors['weight']['dtype'] == dtype
assert calculate_footprint(tensors)['base_memory_bytes'] == block_bytes
19 changes: 19 additions & 0 deletions tests/test_scalar_tensors.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,19 @@
from modelinfo.calculator import calculate_footprint


def test_scalar_tensor_contributes_one_parameter():
result = calculate_footprint({"scale": {"shape": [], "dtype": "F32"}})
assert result["total_params"] == 1
assert result["base_memory_bytes"] == 4
assert result["primary_dtype"] == "F32"


def test_missing_shape_is_not_assumed_scalar():
result = calculate_footprint({"unknown": {"dtype": "F32"}})
assert result["total_params"] == 0


def test_zero_length_tensor_stays_empty():
result = calculate_footprint({"empty": {"shape": [0, 3], "dtype": "F32"}})
assert result["total_params"] == 0
assert result["base_memory_bytes"] == 0
Loading