From 8c80de1b98be95de56e37d24fe55f8e9d50db835 Mon Sep 17 00:00:00 2001 From: Rupayon Haldar <80724680+rupayon123@users.noreply.github.com> Date: Sun, 13 Sep 2026 21:11:21 -0400 Subject: [PATCH 01/12] fix: include scalar tensors in memory estimates How to test: PYTHONPATH=src pytest tests/test_scalar_tensors.py; PYTHONPATH=src pytest --- src/modelinfo/calculator.py | 4 ++-- tests/test_scalar_tensors.py | 19 +++++++++++++++++++ 2 files changed, 21 insertions(+), 2 deletions(-) create mode 100644 tests/test_scalar_tensors.py diff --git a/src/modelinfo/calculator.py b/src/modelinfo/calculator.py index 6649e58..93b4852 100644 --- a/src/modelinfo/calculator.py +++ b/src/modelinfo/calculator.py @@ -82,8 +82,8 @@ def calculate_footprint( if name == "__metadata__": continue - shape = metadata.get("shape", []) - if not shape: + shape = metadata.get("shape") + if shape is None: continue param_count = math.prod(shape) diff --git a/tests/test_scalar_tensors.py b/tests/test_scalar_tensors.py new file mode 100644 index 0000000..7905b45 --- /dev/null +++ b/tests/test_scalar_tensors.py @@ -0,0 +1,19 @@ +from modelinfo.calculator import calculate_footprint + + +def test_scalar_tensor_contributes_one_parameter(): + result = calculate_footprint({"scale": {"shape": [], "dtype": "F32"}}) + assert result["total_params"] == 1 + assert result["base_memory_bytes"] == 4 + assert result["primary_dtype"] == "F32" + + +def test_missing_shape_is_not_assumed_scalar(): + result = calculate_footprint({"unknown": {"dtype": "F32"}}) + assert result["total_params"] == 0 + + +def test_zero_length_tensor_stays_empty(): + result = calculate_footprint({"empty": {"shape": [0, 3], "dtype": "F32"}}) + assert result["total_params"] == 0 + assert result["base_memory_bytes"] == 0 From f4d38f1c451ab9a7023c1447bc25c582dc89ca7a Mon Sep 17 00:00:00 2001 From: Rupayon Haldar <80724680+rupayon123@users.noreply.github.com> Date: Sun, 13 Sep 2026 21:11:56 -0400 Subject: [PATCH 02/12] fix: honor explicit attention head dimensions How to test: PYTHONPATH=src pytest tests/test_explicit_head_dimension.py; PYTHONPATH=src pytest --- src/modelinfo/architecture.py | 2 +- tests/test_explicit_head_dimension.py | 17 +++++++++++++++++ 2 files changed, 18 insertions(+), 1 deletion(-) create mode 100644 tests/test_explicit_head_dimension.py diff --git a/src/modelinfo/architecture.py b/src/modelinfo/architecture.py index bef7237..6c31bc6 100644 --- a/src/modelinfo/architecture.py +++ b/src/modelinfo/architecture.py @@ -41,7 +41,7 @@ def extract_architecture(tensors: Dict[str, Any], config: Dict[str, Any] = None) hidden_size = config.get("hidden_size", 0) if num_attention_heads > 0: - head_dim = hidden_size // num_attention_heads + head_dim = config.get("head_dim") or hidden_size // num_attention_heads kv_dim = num_key_value_heads * head_dim if num_layers > 0 and kv_dim > 0: return num_layers, kv_dim, False diff --git a/tests/test_explicit_head_dimension.py b/tests/test_explicit_head_dimension.py new file mode 100644 index 0000000..26dff56 --- /dev/null +++ b/tests/test_explicit_head_dimension.py @@ -0,0 +1,17 @@ +from modelinfo.architecture import extract_architecture +from modelinfo.calculator import calculate_footprint + + +def test_explicit_head_dimension_overrides_hidden_size_quotient(): + # Gemma2 2B configuration: 2304 / 8 is 288, but head_dim is 256. + config = {"num_hidden_layers": 26, "num_attention_heads": 8, + "num_key_value_heads": 4, "hidden_size": 2304, "head_dim": 256} + assert extract_architecture({}, config) == (26, 1024, False) + result = calculate_footprint({}, config=config, context_length=100) + assert result["kv_cache_bytes"] == 4 * 26 * 1024 * 100 + + +def test_config_without_explicit_dimension_retains_quotient(): + config = {"num_hidden_layers": 2, "num_attention_heads": 4, + "hidden_size": 512} + assert extract_architecture({}, config) == (2, 512, False) From 65dc96711bd5e7ce4bf880c84178a60b75d1ed35 Mon Sep 17 00:00:00 2001 From: Rupayon Haldar <80724680+rupayon123@users.noreply.github.com> Date: Sun, 13 Sep 2026 21:12:49 -0400 Subject: [PATCH 03/12] fix: recognize native gguf tensor names during shape inference How to test: PYTHONPATH=src pytest tests/test_gguf_native_names.py; PYTHONPATH=src pytest --- src/modelinfo/architecture.py | 7 +++++-- tests/test_gguf_native_names.py | 18 ++++++++++++++++++ 2 files changed, 23 insertions(+), 2 deletions(-) create mode 100644 tests/test_gguf_native_names.py diff --git a/src/modelinfo/architecture.py b/src/modelinfo/architecture.py index 6c31bc6..e55db55 100644 --- a/src/modelinfo/architecture.py +++ b/src/modelinfo/architecture.py @@ -65,13 +65,16 @@ def extract_architecture(tensors: Dict[str, Any], config: Dict[str, Any] = None) if len(parts) > idx + 1 and parts[idx+1].isdigit(): layers_set.add(int(parts[idx+1])) - if name.endswith("k_proj.weight") or name.endswith("attn.k.weight") or name.endswith("k_proj.w"): + if is_gguf and len(parts) > 1 and parts[0] == "blk" and parts[1].isdigit(): + layers_set.add(int(parts[1])) + + if (is_gguf and name.endswith("attn_k.weight")) or name.endswith("k_proj.weight") or name.endswith("attn.k.weight") or name.endswith("k_proj.w"): found_k_proj = True shape = meta.get("shape", []) if len(shape) >= 2: kv_dim = shape[-1] if is_gguf else shape[0] - if "qkv_proj.weight" in name or "c_attn.weight" in name: + if "qkv_proj.weight" in name or "c_attn.weight" in name or (is_gguf and name.endswith("attn_qkv.weight")): found_fused = True if not found_k_proj: shape = meta.get("shape", []) diff --git a/tests/test_gguf_native_names.py b/tests/test_gguf_native_names.py new file mode 100644 index 0000000..a070d18 --- /dev/null +++ b/tests/test_gguf_native_names.py @@ -0,0 +1,18 @@ +from modelinfo.architecture import extract_architecture + + +def test_native_gguf_key_projection_names(): + tensors = { + "__metadata__": {"general.architecture": "llama"}, + "blk.0.attn_k.weight": {"shape": [4096, 1024]}, + "blk.1.attn_k.weight": {"shape": [4096, 1024]}, + } + assert extract_architecture(tensors) == (2, 1024, False) + + +def test_native_gguf_fused_projection_names(): + tensors = { + "__metadata__": {"general.architecture": "gpt2"}, + "blk.0.attn_qkv.weight": {"shape": [768, 2304]}, + } + assert extract_architecture(tensors) == (1, 768, True) From aae7d685500773805ff675095a1eee274691d3d6 Mon Sep 17 00:00:00 2001 From: Rupayon Haldar <80724680+rupayon123@users.noreply.github.com> Date: Sun, 13 Sep 2026 21:13:32 -0400 Subject: [PATCH 04/12] fix: infer gpt2 cache dimensions from conv1d output axis How to test: PYTHONPATH=src pytest tests/test_gpt2_projection_layout.py; PYTHONPATH=src pytest --- src/modelinfo/architecture.py | 3 ++- tests/test_gpt2_projection_layout.py | 11 +++++++++++ 2 files changed, 13 insertions(+), 1 deletion(-) create mode 100644 tests/test_gpt2_projection_layout.py diff --git a/src/modelinfo/architecture.py b/src/modelinfo/architecture.py index e55db55..400ab1a 100644 --- a/src/modelinfo/architecture.py +++ b/src/modelinfo/architecture.py @@ -79,7 +79,8 @@ def extract_architecture(tensors: Dict[str, Any], config: Dict[str, Any] = None) if not found_k_proj: shape = meta.get("shape", []) if len(shape) >= 2: - kv_dim = (shape[-1] if is_gguf else shape[0]) // 3 + output_dim = shape[-1] if is_gguf or name.endswith("c_attn.weight") else shape[0] + kv_dim = output_dim // 3 num_layers = len(layers_set) if found_fused and not found_k_proj and kv_dim > 0: diff --git a/tests/test_gpt2_projection_layout.py b/tests/test_gpt2_projection_layout.py new file mode 100644 index 0000000..d1efd3e --- /dev/null +++ b/tests/test_gpt2_projection_layout.py @@ -0,0 +1,11 @@ +from modelinfo.architecture import extract_architecture + + +def test_gpt2_conv1d_projection_uses_output_axis(): + tensors = {"transformer.h.0.attn.c_attn.weight": {"shape": [768, 2304]}} + assert extract_architecture(tensors) == (1, 768, True) + + +def test_linear_qkv_projection_retains_output_first_layout(): + tensors = {"model.layers.0.qkv_proj.weight": {"shape": [2304, 768]}} + assert extract_architecture(tensors) == (1, 768, True) From 9365204c1e4b898f32a22358c62059b78a6f3f0e Mon Sep 17 00:00:00 2001 From: Rupayon Haldar <80724680+rupayon123@users.noreply.github.com> Date: Sun, 13 Sep 2026 21:15:07 -0400 Subject: [PATCH 05/12] fix: account for boolean and unsigned byte tensor storage How to test: PYTHONPATH=src pytest tests/test_primitive_dtype_sizes.py; PYTHONPATH=src pytest --- src/modelinfo/calculator.py | 3 +++ tests/test_primitive_dtype_sizes.py | 9 +++++++++ 2 files changed, 12 insertions(+) create mode 100644 tests/test_primitive_dtype_sizes.py diff --git a/src/modelinfo/calculator.py b/src/modelinfo/calculator.py index 93b4852..b74b7ca 100644 --- a/src/modelinfo/calculator.py +++ b/src/modelinfo/calculator.py @@ -17,6 +17,9 @@ "I8": 1.0, "U64": 8.0, "U32": 4.0, + "U16": 2.0, + "U8": 1.0, + "BOOL": 1.0, "Q8_0": 1.0625, "Q8_1": 1.0625, "Q8_K": 1.0625, diff --git a/tests/test_primitive_dtype_sizes.py b/tests/test_primitive_dtype_sizes.py new file mode 100644 index 0000000..d06fbbd --- /dev/null +++ b/tests/test_primitive_dtype_sizes.py @@ -0,0 +1,9 @@ +import pytest + +from modelinfo.calculator import calculate_footprint + + +@pytest.mark.parametrize(("dtype", "size"), [("BOOL", 1), ("U8", 1), ("U16", 2)]) +def test_unsigned_and_boolean_tensor_sizes(dtype, size): + result = calculate_footprint({"buffer": {"dtype": dtype, "shape": [256]}}) + assert result["base_memory_bytes"] == 256 * size From 7659eadeb28aed3f6d5ed8773256592a5eec318f Mon Sep 17 00:00:00 2001 From: Rupayon Haldar <80724680+rupayon123@users.noreply.github.com> Date: Mon, 14 Sep 2026 08:26:48 -0400 Subject: [PATCH 06/12] Read GPT-2 configuration aliases for exact KV cache estimates --- src/modelinfo/architecture.py | 6 +++--- tests/test_gpt2_config.py | 16 ++++++++++++++++ 2 files changed, 19 insertions(+), 3 deletions(-) create mode 100644 tests/test_gpt2_config.py diff --git a/src/modelinfo/architecture.py b/src/modelinfo/architecture.py index 400ab1a..76257a8 100644 --- a/src/modelinfo/architecture.py +++ b/src/modelinfo/architecture.py @@ -35,10 +35,10 @@ def extract_architecture(tensors: Dict[str, Any], config: Dict[str, Any] = None) # 2. Attempt explicit SafeTensors config.json if config: - num_layers = config.get("num_hidden_layers", 0) - num_attention_heads = config.get("num_attention_heads", 1) + num_layers = config.get("num_hidden_layers", config.get("n_layer", 0)) + num_attention_heads = config.get("num_attention_heads", config.get("n_head", 1)) num_key_value_heads = config.get("num_key_value_heads", num_attention_heads) - hidden_size = config.get("hidden_size", 0) + hidden_size = config.get("hidden_size", config.get("n_embd", 0)) if num_attention_heads > 0: head_dim = config.get("head_dim") or hidden_size // num_attention_heads diff --git a/tests/test_gpt2_config.py b/tests/test_gpt2_config.py new file mode 100644 index 0000000..adcc0fe --- /dev/null +++ b/tests/test_gpt2_config.py @@ -0,0 +1,16 @@ +"""GPT2Config serializes n_layer/n_head/n_embd rather than generic aliases.""" +from modelinfo.architecture import extract_architecture +from modelinfo.calculator import calculate_footprint + + +def test_gpt2_config_provides_exact_architecture_without_tensor_scan(): + config = {"model_type": "gpt2", "n_layer": 12, "n_head": 12, "n_embd": 768} + assert extract_architecture({}, config) == (12, 768, False) + result = calculate_footprint({}, context_length=1024, config=config) + assert result["kv_cache_bytes"] == 4 * 12 * 768 * 1024 + + +def test_generic_config_fields_take_precedence_over_gpt2_aliases(): + config = {"n_layer": 12, "n_head": 12, "n_embd": 768, + "num_hidden_layers": 2, "num_attention_heads": 4, "hidden_size": 256} + assert extract_architecture({}, config) == (2, 256, False) From dd7cc5fa6b24d7b52c20a64dde0f807e1ef8a7a0 Mon Sep 17 00:00:00 2001 From: Rupayon Haldar <80724680+rupayon123@users.noreply.github.com> Date: Mon, 14 Sep 2026 08:28:32 -0400 Subject: [PATCH 07/12] Match quantized memory estimates to GGML block storage sizes --- src/modelinfo/calculator.py | 24 ++++++++++---------- tests/test_calculator.py | 4 ++-- tests/test_quantized_block_sizes.py | 35 +++++++++++++++++++++++++++++ 3 files changed, 49 insertions(+), 14 deletions(-) create mode 100644 tests/test_quantized_block_sizes.py diff --git a/src/modelinfo/calculator.py b/src/modelinfo/calculator.py index b74b7ca..2295b5e 100644 --- a/src/modelinfo/calculator.py +++ b/src/modelinfo/calculator.py @@ -21,24 +21,24 @@ "U8": 1.0, "BOOL": 1.0, "Q8_0": 1.0625, - "Q8_1": 1.0625, - "Q8_K": 1.0625, - "Q6_K": 0.828125, + "Q8_1": 1.125, + "Q8_K": 1.140625, + "Q6_K": 0.8203125, "Q5_0": 0.6875, "Q5_1": 0.75, "Q5_K": 0.6875, "Q4_0": 0.5625, "Q4_1": 0.625, - "Q4_K": 0.59375, - "Q3_K": 0.4375, - "Q2_K": 0.34375, - "IQ4_NL": 0.53125, + "Q4_K": 0.5625, + "Q3_K": 0.4296875, + "Q2_K": 0.328125, + "IQ4_NL": 0.5625, "IQ4_XS": 0.53125, - "IQ3_S": 0.4375, - "IQ3_XXS": 0.385, - "IQ2_S": 0.3125, - "IQ2_XS": 0.296875, - "IQ2_XXS": 0.28125, + "IQ3_S": 0.4296875, + "IQ3_XXS": 0.3828125, + "IQ2_S": 0.3203125, + "IQ2_XS": 0.2890625, + "IQ2_XXS": 0.2578125, "IQ1_M": 0.21875, "IQ1_S": 0.1953125, "Q8": 1.06, diff --git a/tests/test_calculator.py b/tests/test_calculator.py index 94cf3ea..b55d34a 100644 --- a/tests/test_calculator.py +++ b/tests/test_calculator.py @@ -115,8 +115,8 @@ def test_framework_overhead_included(): def test_explicit_gguf_quantization_byte_multipliers(): """Verify that explicit ggml_type enums are exactly mapped.""" assert _get_bytes_per_param("Q8_0") == 1.0625 - assert _get_bytes_per_param("Q4_K") == 0.59375 - assert _get_bytes_per_param("IQ2_XXS") == 0.28125 + assert _get_bytes_per_param("Q4_K") == 144 / 256 + assert _get_bytes_per_param("IQ2_XXS") == 66 / 256 assert _get_bytes_per_param("F8_E5M2") == 1.0 def test_topology_penalties(): diff --git a/tests/test_quantized_block_sizes.py b/tests/test_quantized_block_sizes.py new file mode 100644 index 0000000..d0c4992 --- /dev/null +++ b/tests/test_quantized_block_sizes.py @@ -0,0 +1,35 @@ +"""Expected byte lengths verified with sizeof GGML quantization block structures. + +Reference: https://github.com/ggml-org/ggml/blob/master/src/ggml-common.h +This tests packed tensor storage, excluding GGUF file metadata and alignment. +""" +import pytest +from modelinfo.calculator import calculate_footprint + + +@pytest.mark.parametrize("dtype,block_size,block_bytes", [ + ('Q8_0', 32, 34), + ('Q8_1', 32, 36), + ('Q8_K', 256, 292), + ('Q6_K', 256, 210), + ('Q5_0', 32, 22), + ('Q5_1', 32, 24), + ('Q5_K', 256, 176), + ('Q4_0', 32, 18), + ('Q4_1', 32, 20), + ('Q4_K', 256, 144), + ('Q3_K', 256, 110), + ('Q2_K', 256, 84), + ('IQ4_NL', 32, 18), + ('IQ4_XS', 256, 136), + ('IQ3_S', 256, 110), + ('IQ3_XXS', 256, 98), + ('IQ2_S', 256, 82), + ('IQ2_XS', 256, 74), + ('IQ2_XXS', 256, 66), + ('IQ1_M', 256, 56), + ('IQ1_S', 256, 50), +]) +def test_quantized_tensor_storage_matches_ggml_blocks(dtype, block_size, block_bytes): + result = calculate_footprint({"weight": {"shape": [3, block_size], "dtype": dtype}}) + assert result["base_memory_bytes"] == 3 * block_bytes From bd7e9ebe773277d2466230350728232b71ce8cb8 Mon Sep 17 00:00:00 2001 From: Rupayon Haldar <80724680+rupayon123@users.noreply.github.com> Date: Tue, 15 Sep 2026 00:18:36 -0400 Subject: [PATCH 08/12] Recognize ternary and MXFP4 GGUF tensor storage --- src/modelinfo/calculator.py | 3 +++ src/modelinfo/parsers/gguf.py | 1 + tests/test_recent_gguf_types.py | 19 +++++++++++++++++++ 3 files changed, 23 insertions(+) create mode 100644 tests/test_recent_gguf_types.py diff --git a/src/modelinfo/calculator.py b/src/modelinfo/calculator.py index 2295b5e..3b892cd 100644 --- a/src/modelinfo/calculator.py +++ b/src/modelinfo/calculator.py @@ -41,6 +41,9 @@ "IQ2_XXS": 0.2578125, "IQ1_M": 0.21875, "IQ1_S": 0.1953125, + "TQ1_0": 54 / 256, + "TQ2_0": 66 / 256, + "MXFP4": 17 / 32, "Q8": 1.06, "Q6": 0.82, "Q5": 0.68, diff --git a/src/modelinfo/parsers/gguf.py b/src/modelinfo/parsers/gguf.py index 3af2fb4..cc5dd8b 100644 --- a/src/modelinfo/parsers/gguf.py +++ b/src/modelinfo/parsers/gguf.py @@ -8,6 +8,7 @@ 18: "IQ3_XXS", 19: "IQ1_S", 20: "IQ4_NL", 21: "IQ3_S", 22: "IQ2_S", 23: "IQ4_XS", 24: "I8", 25: "I16", 26: "I32", 27: "I64", 28: "F64", 29: "IQ1_M", 30: "BF16", 31: "Q4_0_4_4", 32: "Q4_0_4_8", 33: "Q4_0_8_8", + 34: "TQ1_0", 35: "TQ2_0", 39: "MXFP4", } def _read_gguf_value(f: Any, val_type: int) -> Any: diff --git a/tests/test_recent_gguf_types.py b/tests/test_recent_gguf_types.py new file mode 100644 index 0000000..c7e6688 --- /dev/null +++ b/tests/test_recent_gguf_types.py @@ -0,0 +1,19 @@ +import io +import struct + +import pytest + +from modelinfo.calculator import calculate_footprint +from modelinfo.parsers.gguf import parse_gguf_header + + +@pytest.mark.parametrize('type_id,dtype,block_size,block_bytes', [ + (34, 'TQ1_0', 256, 54), (35, 'TQ2_0', 256, 66), (39, 'MXFP4', 32, 17), +]) +def test_gguf_ternary_and_mxfp4_memory(type_id, dtype, block_size, block_bytes): + payload = (b'GGUF' + struct.pack(' Date: Tue, 15 Sep 2026 00:24:27 -0400 Subject: [PATCH 09/12] Reject invalid tensor dimensions before calculating memory --- src/modelinfo/calculator.py | 5 ++++- tests/test_invalid_tensor_shapes.py | 15 +++++++++++++++ 2 files changed, 19 insertions(+), 1 deletion(-) create mode 100644 tests/test_invalid_tensor_shapes.py diff --git a/src/modelinfo/calculator.py b/src/modelinfo/calculator.py index 3b892cd..3a369b0 100644 --- a/src/modelinfo/calculator.py +++ b/src/modelinfo/calculator.py @@ -91,7 +91,10 @@ def calculate_footprint( shape = metadata.get("shape") if shape is None: continue - + if (not isinstance(shape, (list, tuple)) + or any(type(dimension) is not int or dimension < 0 for dimension in shape)): + raise ValueError(f"Invalid tensor shape for {name!r}: expected nonnegative integer dimensions") + param_count = math.prod(shape) total_params += param_count diff --git a/tests/test_invalid_tensor_shapes.py b/tests/test_invalid_tensor_shapes.py new file mode 100644 index 0000000..4d14a66 --- /dev/null +++ b/tests/test_invalid_tensor_shapes.py @@ -0,0 +1,15 @@ +import pytest +from modelinfo.calculator import calculate_footprint + + +@pytest.mark.parametrize('shape', [[-1, 4], [True, 4], [1.5, 4], ['3', 4], '34']) +def test_invalid_tensor_dimensions_are_rejected(shape): + with pytest.raises(ValueError, match='shape'): + calculate_footprint({'weight': {'shape': shape, 'dtype': 'F16'}}) + + +def test_zero_length_dimensions_and_scalar_shapes_remain_valid(): + result = calculate_footprint({'empty': {'shape': [0, 4], 'dtype': 'F16'}, + 'scalar': {'shape': [], 'dtype': 'F32'}}) + assert result['total_params'] == 1 + assert result['base_memory_bytes'] == 4 From 769f892b10b47211a8bcb4605bb9383ad1583c41 Mon Sep 17 00:00:00 2001 From: Rupayon Haldar <80724680+rupayon123@users.noreply.github.com> Date: Wed, 16 Sep 2026 10:52:51 -0400 Subject: [PATCH 10/12] Read decoder cache dimensions from multimodal text configs --- src/modelinfo/architecture.py | 3 +++ tests/test_multimodal_text_config.py | 21 +++++++++++++++++++++ 2 files changed, 24 insertions(+) create mode 100644 tests/test_multimodal_text_config.py diff --git a/src/modelinfo/architecture.py b/src/modelinfo/architecture.py index 76257a8..ca0366e 100644 --- a/src/modelinfo/architecture.py +++ b/src/modelinfo/architecture.py @@ -35,6 +35,9 @@ def extract_architecture(tensors: Dict[str, Any], config: Dict[str, Any] = None) # 2. Attempt explicit SafeTensors config.json if config: + # Multimodal models keep the decoder dimensions in text_config. + if isinstance(config.get("text_config"), dict): + config = config["text_config"] num_layers = config.get("num_hidden_layers", config.get("n_layer", 0)) num_attention_heads = config.get("num_attention_heads", config.get("n_head", 1)) num_key_value_heads = config.get("num_key_value_heads", num_attention_heads) diff --git a/tests/test_multimodal_text_config.py b/tests/test_multimodal_text_config.py new file mode 100644 index 0000000..1760dc4 --- /dev/null +++ b/tests/test_multimodal_text_config.py @@ -0,0 +1,21 @@ +from copy import deepcopy + +from modelinfo.calculator import calculate_footprint + + +def test_lazy_multimodal_checkpoint_uses_text_backbone_kv_dimensions(): + config = { + 'model_type': 'llava', + 'vision_config': {'hidden_size': 1024, 'num_hidden_layers': 24}, + 'text_config': { + 'hidden_size': 4096, 'num_hidden_layers': 32, + 'num_attention_heads': 32, 'num_key_value_heads': 8, + }, + } + original = deepcopy(config) + tensors = {'__metadata__': {'lazy_fetch': True, 'total_size': 1024}} + result = calculate_footprint(tensors, context_length=2048, config=config) + assert result['num_layers'] == 32 + assert result['kv_dim'] == 1024 + assert result['kv_cache_bytes'] == 2 * 32 * 1024 * 2048 * 2 + assert config == original From 2cd130e42134834dc62cb529af873a042ced8f15 Mon Sep 17 00:00:00 2001 From: Rupayon Haldar <80724680+rupayon123@users.noreply.github.com> Date: Wed, 16 Sep 2026 11:00:34 -0400 Subject: [PATCH 11/12] Use declared weight precision for lazy parameter estimates --- src/modelinfo/calculator.py | 14 +++++++++++--- tests/test_lazy_dtype.py | 15 +++++++++++++++ 2 files changed, 26 insertions(+), 3 deletions(-) create mode 100644 tests/test_lazy_dtype.py diff --git a/src/modelinfo/calculator.py b/src/modelinfo/calculator.py index 3a369b0..b7b54dc 100644 --- a/src/modelinfo/calculator.py +++ b/src/modelinfo/calculator.py @@ -79,10 +79,18 @@ def calculate_footprint( if is_lazy: base_memory_bytes = tensors.get("__metadata__", {}).get("total_size", 0.0) - # Assume predominantly FP16/BF16 for modern Hub architectures - primary_dtype = "BF16" + # Lazy mode estimates parameter count from the configured weight dtype. + # Retain the BF16 fallback when no recognized dtype is available. + dtype_config = config or {} + if isinstance(dtype_config.get("text_config"), dict): + dtype_config = dtype_config["text_config"] + configured_dtype = dtype_config.get("dtype") or dtype_config.get("torch_dtype") + primary_dtype = { + "float64": "F64", "float32": "F32", "float16": "F16", + "bfloat16": "BF16", + }.get(str(configured_dtype).lower(), "BF16") dtype_counts[primary_dtype] = 1 - total_params = int(base_memory_bytes / 2.0) + total_params = int(base_memory_bytes / _get_bytes_per_param(primary_dtype)) else: for name, metadata in tensors.items(): if name == "__metadata__": diff --git a/tests/test_lazy_dtype.py b/tests/test_lazy_dtype.py new file mode 100644 index 0000000..925cce3 --- /dev/null +++ b/tests/test_lazy_dtype.py @@ -0,0 +1,15 @@ +import pytest + +from modelinfo.calculator import calculate_footprint + + +@pytest.mark.parametrize('key,dtype,expected,width', [ + ('torch_dtype', 'float32', 'F32', 4), + ('dtype', 'float64', 'F64', 8), + ('torch_dtype', 'float16', 'F16', 2), +]) +def test_lazy_parameter_estimate_respects_declared_dtype(key, dtype, expected, width): + result = calculate_footprint({'__metadata__': {'lazy_fetch': True, 'total_size': 1024}}, config={key: dtype}) + assert result['primary_dtype'] == expected + assert result['total_params'] == 1024 // width + assert result['base_memory_bytes'] == 1024 From b956af9f957232dacf133dae0a58d7c11cc212b3 Mon Sep 17 00:00:00 2001 From: Rupayon Haldar <80724680+rupayon123@users.noreply.github.com> Date: Wed, 16 Sep 2026 11:02:40 -0400 Subject: [PATCH 12/12] Apply GGUF multi-head attention default when KV heads are omitted --- src/modelinfo/architecture.py | 3 ++- tests/test_gguf_mha_default.py | 14 ++++++++++++++ 2 files changed, 16 insertions(+), 1 deletion(-) create mode 100644 tests/test_gguf_mha_default.py diff --git a/src/modelinfo/architecture.py b/src/modelinfo/architecture.py index ca0366e..d2f4a1f 100644 --- a/src/modelinfo/architecture.py +++ b/src/modelinfo/architecture.py @@ -17,7 +17,8 @@ def extract_architecture(tensors: Dict[str, Any], config: Dict[str, Any] = None) if gen_arch: arch_str = str(gen_arch) num_layers = metadata.get(f"{arch_str}.block_count", 0) - kv_heads = metadata.get(f"{arch_str}.attention.head_count_kv", 0) + kv_heads = metadata.get(f"{arch_str}.attention.head_count_kv", + metadata.get(f"{arch_str}.attention.head_count", 0)) key_length = metadata.get(f"{arch_str}.attention.key_length") if not key_length: diff --git a/tests/test_gguf_mha_default.py b/tests/test_gguf_mha_default.py new file mode 100644 index 0000000..fdea452 --- /dev/null +++ b/tests/test_gguf_mha_default.py @@ -0,0 +1,14 @@ +from modelinfo.architecture import extract_architecture + + +def test_gguf_omitted_kv_heads_defaults_to_attention_heads(): + metadata = {'general.architecture': 'llama', 'llama.block_count': 32, + 'llama.embedding_length': 4096, 'llama.attention.head_count': 32} + assert extract_architecture({'__metadata__': metadata}) == (32, 4096, False) + + +def test_gguf_explicit_grouped_query_heads_are_preserved(): + metadata = {'general.architecture': 'llama', 'llama.block_count': 32, + 'llama.embedding_length': 4096, 'llama.attention.head_count': 32, + 'llama.attention.head_count_kv': 8} + assert extract_architecture({'__metadata__': metadata}) == (32, 1024, False)