Skip to content

[bug]: FP4 Mistral 3 Text Encoder (for FLUX.2 DEV) May be Misconfigured #9565

Description

@the-space-fish

Is there an existing issue for this problem?

  • I have searched the existing issues

Install method

Invoke's Launcher

Operating system

Windows

GPU vendor

Nvidia (CUDA)

GPU model

RTX 4090

GPU VRAM

24GB

Version number

6.14

Browser

Chrome 152.0.7977.66

System Information

{
"version": "6.14.0",
"dependencies": {
"absl-py" : "2.4.0",
"accelerate" : "1.14.0",
"annotated-doc" : "0.0.4",
"annotated-types" : "0.7.0",
"anyio" : "4.14.1",
"argon2-cffi" : "25.1.0",
"argon2-cffi-bindings" : "25.1.0",
"arrow" : "1.4.0",
"asttokens" : "3.0.1",
"async-lru" : "2.3.0",
"attrs" : "26.1.0",
"babel" : "2.18.0",
"bcrypt" : "3.2.2",
"beautifulsoup4" : "4.15.0",
"bidict" : "0.23.1",
"bitsandbytes" : "0.49.2",
"blake3" : "1.0.9",
"bleach" : "6.4.0",
"certifi" : "2026.6.17",
"cffi" : "2.0.0",
"charset-normalizer" : "3.4.7",
"click" : "8.4.2",
"colorama" : "0.4.6",
"coloredlogs" : "15.0.1",
"comm" : "0.2.3",
"compel" : "2.4.0",
"contourpy" : "1.3.3",
"cryptography" : "49.0.0",
"CUDA" : "12.8",
"cycler" : "0.12.1",
"debugpy" : "1.8.21",
"decorator" : "5.3.1",
"defusedxml" : "0.7.1",
"Deprecated" : "1.3.1",
"diffusers" : "0.39.0",
"dnspython" : "2.8.0",
"dynamicprompts" : "0.31.0",
"ecdsa" : "0.19.2",
"einops" : "0.8.2",
"email-validator" : "2.3.0",
"executing" : "2.2.1",
"fastapi" : "0.141.1",
"fastapi-events" : "0.12.2",
"fastjsonschema" : "2.21.2",
"filelock" : "3.29.4",
"flatbuffers" : "25.12.19",
"fonttools" : "4.63.0",
"fqdn" : "1.5.1",
"fsspec" : "2026.6.0",
"gguf" : "0.19.0",
"h11" : "0.16.0",
"hf-xet" : "1.5.1",
"httpcore" : "1.0.9",
"httptools" : "0.8.0",
"httpx" : "0.28.1",
"huggingface_hub" : "1.21.0",
"humanfriendly" : "10.0",
"idna" : "3.18",
"ImageIO" : "2.37.4",
"imageio-ffmpeg" : "0.6.0",
"importlib_metadata" : "9.0.0",
"InvokeAI" : "6.14.0",
"ipykernel" : "7.3.0",
"ipython" : "9.15.0",
"ipython_pygments_lexers" : "1.1.1",
"isoduration" : "20.11.0",
"jax" : "0.7.1",
"jaxlib" : "0.7.1",
"jedi" : "0.20.0",
"Jinja2" : "3.1.6",
"json5" : "0.15.0",
"jsonpointer" : "3.1.1",
"jsonschema" : "4.26.0",
"jsonschema-specifications": "2025.9.1",
"jupyter-events" : "0.12.1",
"jupyter-lsp" : "2.3.1",
"jupyter_builder" : "1.0.2",
"jupyter_client" : "8.9.1",
"jupyter_core" : "5.9.1",
"jupyter_server" : "2.20.0",
"jupyter_server_terminals" : "0.5.4",
"jupyterlab" : "4.6.0",
"jupyterlab_pygments" : "0.3.0",
"jupyterlab_server" : "2.28.0",
"kiwisolver" : "1.5.0",
"lark" : "1.3.1",
"markdown-it-py" : "4.2.0",
"MarkupSafe" : "3.0.3",
"matplotlib" : "3.11.0",
"matplotlib-inline" : "0.2.2",
"mdurl" : "0.1.2",
"mediapipe" : "0.10.14",
"mistral_common" : "1.11.6",
"mistune" : "3.3.2",
"ml_dtypes" : "0.5.4",
"mpmath" : "1.3.0",
"nbclient" : "0.11.0",
"nbconvert" : "7.17.1",
"nbformat" : "5.10.4",
"nest-asyncio2" : "1.7.2",
"networkx" : "3.6.1",
"notebook" : "7.6.0",
"notebook_shim" : "0.2.4",
"numpy" : "1.26.4",
"onnx" : "1.16.1",
"onnxruntime" : "1.19.2",
"opencv-contrib-python" : "4.11.0.86",
"opt_einsum" : "3.4.0",
"packaging" : "26.2",
"pandocfilters" : "1.5.1",
"parso" : "0.8.7",
"passlib" : "1.7.4",
"picklescan" : "1.0.4",
"pillow" : "12.2.0",
"platformdirs" : "4.10.0",
"prometheus_client" : "0.25.0",
"prompt_toolkit" : "3.0.52",
"protobuf" : "4.25.9",
"psutil" : "7.2.2",
"pure_eval" : "0.2.3",
"pyasn1" : "0.6.3",
"pycountry" : "26.2.16",
"pycparser" : "3.0",
"pydantic" : "2.13.4",
"pydantic-extra-types" : "2.11.1",
"pydantic-settings" : "2.14.2",
"pydantic_core" : "2.46.4",
"Pygments" : "2.20.0",
"pyparsing" : "3.3.2",
"PyPatchMatch" : "1.0.2",
"pyreadline3" : "3.5.6",
"python-dateutil" : "2.9.0.post0",
"python-dotenv" : "1.2.2",
"python-engineio" : "4.13.3",
"python-jose" : "3.5.0",
"python-json-logger" : "4.1.0",
"python-multipart" : "0.0.32",
"python-socketio" : "5.16.3",
"PyWavelets" : "1.9.0",
"pywinpty" : "3.0.5",
"PyYAML" : "6.0.3",
"pyzmq" : "27.1.0",
"referencing" : "0.37.0",
"regex" : "2026.5.9",
"requests" : "2.34.2",
"rfc3339-validator" : "0.1.4",
"rfc3986-validator" : "0.1.1",
"rfc3987-syntax" : "1.1.0",
"rich" : "15.0.0",
"rpds-py" : "2026.5.1",
"rsa" : "4.9.1",
"safetensors" : "0.8.0",
"scipy" : "1.17.1",
"semver" : "3.0.4",
"Send2Trash" : "2.1.0",
"sentencepiece" : "0.2.0",
"setuptools" : "82.0.1",
"shellingham" : "1.5.4",
"simple-websocket" : "1.1.0",
"six" : "1.17.0",
"sniffio" : "1.3.1",
"sounddevice" : "0.5.5",
"soupsieve" : "2.8.4",
"spandrel" : "0.4.2",
"stack-data" : "0.6.3",
"starlette" : "0.48.0",
"sympy" : "1.14.0",
"terminado" : "0.18.1",
"tiktoken" : "0.13.0",
"tinycss2" : "1.5.1",
"tokenizers" : "0.22.2",
"torch" : "2.7.1+cu128",
"torchsde" : "0.2.6",
"torchvision" : "0.22.1+cu128",
"tornado" : "6.5.7",
"tqdm" : "4.68.3",
"traitlets" : "5.15.1",
"trampoline" : "0.1.2",
"transformers" : "5.5.4",
"typer" : "0.25.1",
"typing-inspection" : "0.4.2",
"typing_extensions" : "4.15.0",
"tzdata" : "2026.2",
"uri-template" : "1.3.0",
"urllib3" : "2.7.0",
"uvicorn" : "0.49.0",
"watchfiles" : "1.2.0",
"wcwidth" : "0.8.1",
"webcolors" : "25.10.0",
"webencodings" : "0.5.1",
"websocket-client" : "1.9.0",
"websockets" : "16.0",
"wrapt" : "2.2.2",
"wsproto" : "1.3.2",
"zipp" : "4.1.0"
},
"config": {
"schema_version": "4.0.3",
"legacy_models_yaml_path": null,
"host": "0.0.0.0",
"port": 9090,
"allow_origins": [],
"allow_credentials": true,
"allow_methods": [""],
"allow_headers": ["
"],
"ssl_certfile": null,
"ssl_keyfile": null,
"base_url": null,
"forwarded_allow_ips": "127.0.0.1",
"http_compression_level": 9,
"log_tokenization": false,
"patchmatch": true,
"models_dir": "models",
"convert_cache_dir": "models\.convert_cache",
"download_cache_dir": "models\.download_cache",
"legacy_conf_dir": "configs",
"db_dir": "databases",
"outputs_dir": "outputs",
"image_subfolder_strategy": "flat",
"custom_nodes_dir": "nodes",
"style_presets_dir": "style_presets",
"workflow_thumbnails_dir": "workflow_thumbnails",
"log_handlers": ["console"],
"log_format": "color",
"log_level": "info",
"log_sql": false,
"log_level_network": "warning",
"use_memory_db": false,
"dev_reload": false,
"profile_graphs": false,
"profile_prefix": null,
"profiles_dir": "profiles",
"max_cache_ram_gb": null,
"max_cache_vram_gb": null,
"log_memory_usage": false,
"model_cache_keep_alive_min": 0,
"device_working_mem_gb": 3,
"enable_partial_loading": true,
"keep_ram_copy_of_weights": true,
"ram": null,
"vram": null,
"lazy_offload": true,
"pytorch_cuda_alloc_conf": null,
"device": "auto",
"generation_devices": "auto",
"offload_text_encoders_to_idle_gpus": true,
"precision": "auto",
"sequential_guidance": false,
"wan_memory_optimization": false,
"pid_memory_optimization": false,
"attention_type": "auto",
"attention_slice_size": "auto",
"force_tiled_decode": false,
"pil_compress_level": 6,
"max_queue_size": 10000,
"session_queue_mode": "round_robin",
"clear_queue_on_startup": false,
"max_queue_history": null,
"allow_nodes": null,
"deny_nodes": null,
"node_cache_size": 512,
"hashing_algorithm": "blake3_single",
"remote_api_tokens": [ {"url_regex": "civitai.com", "token": "**********"} ],
"scan_models_on_startup": false,
"allow_private_download_urls": false,
"download_proxy": null,
"unsafe_disable_picklescan": false,
"allow_unknown_models": true,
"multiuser": false,
"strict_password_checking": false,
"external_alibabacloud_api_key": null,
"external_alibabacloud_base_url": null,
"external_gemini_api_key": null,
"external_openai_api_key": null,
"external_gemini_base_url": null,
"external_openai_base_url": null,
"external_seedream_api_key": null,
"external_seedream_base_url": null
},
"set_config_fields": ["host", "legacy_models_yaml_path", "pil_compress_level", "remote_api_tokens"

What happened

In setting up FLUX.2 Dev, I initially imported a known-good copy of the Mistral 3 Small 24B text encoder in FP4 that I use in Swarm/ComfyUI (the same file from the same Comfy-Org repository that Invoke would otherwise download), and it produced a size/shape mismatch error when attempting to initiate a generation. I tried to re-download the same text encoder with Invoke through its Starter Model section just in case the import misconfigured something, but got the same error. The full error sequence is attached:

MistralFP4Error.txt

Snippet:

[2026-09-01 16:46:02,417]::[InvokeAI]::INFO --> Executing queue item 31823, session 66b0b828-1c34-45fa-9adb-4baefa396cc2 on cuda:0
[2026-09-01 16:46:18,523]::[MistralEncoderCheckpointLoader]::INFO --> Dequantized 208 Comfy-Org-style quantized weights
[2026-09-01 16:46:18,570]::[MistralEncoderCheckpointLoader]::INFO --> Mistral encoder config (checkpoint): layers=30, hidden=5120, heads=32, kv_heads=8, intermediate=32768
[2026-09-01 16:46:19,490]::[InvokeAI]::ERROR --> Error while invoking session 66b0b828-1c34-45fa-9adb-4baefa396cc2, invocation b71ffe7a-bbd0-47ff-87e8-c801c0cd64b4 (flux2_dev_text_encoder): Error(s) in loading state_dict for MistralModel:
        size mismatch for layers.0.self_attn.q_proj.weight: copying a param with shape torch.Size([4096, 2560]) from checkpoint, the shape in current model is torch.Size([4096, 5120]).
        size mismatch for layers.0.self_attn.k_proj.weight: copying a param with shape torch.Size([1024, 2560]) from checkpoint, the shape in current model is torch.Size([1024, 5120]).
        size mismatch for layers.0.mlp.gate_proj.weight: copying a param with shape torch.Size([32768, 2560]) from checkpoint, the shape in current model is torch.Size([32768, 5120]).
        size mismatch for layers.0.mlp.up_proj.weight: copying a param with shape torch.Size([32768, 2560]) from checkpoint, the shape in current model is torch.Size([32768, 5120]).
        size mismatch for layers.1.self_attn.q_proj.weight: copying a param with shape torch.Size([4096, 2560]) from checkpoint, the shape in current model is torch.Size([4096, 5120]).
        size mismatch for layers.1.self_attn.k_proj.weight: copying a param with shape torch.Size([1024, 2560]) from checkpoint, the shape in current model is torch.Size([1024, 5120]).
        size mismatch for layers.1.self_attn.o_proj.weight: copying a param with shape torch.Size([5120, 2048]) from checkpoint, the shape in current model is torch.Size([5120, 4096]).
        size mismatch for layers.1.mlp.gate_proj.weight: copying a param with shape torch.Size([32768, 2560]) from checkpoint, the shape in current model is torch.Size([32768, 5120]).
        size mismatch for layers.1.mlp.up_proj.weight: copying a param with shape torch.Size([32768, 2560]) from checkpoint, the shape in current model is torch.Size([32768, 5120]).
        size mismatch for layers.1.mlp.down_proj.weight: copying a param with shape torch.Size([5120, 16384]) from checkpoint, the shape in current model is torch.Size([5120, 32768]).
...
[2026-09-01 16:46:19,490]::[InvokeAI]::ERROR --> Traceback (most recent call last):
  File "D:\LMIM\InvokeAI\.venv\Lib\site-packages\invokeai\app\services\session_processor\session_processor_default.py", line 278, in run_node
    output = invocation.invoke_internal(context=context, services=self._services)
             ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "D:\LMIM\InvokeAI\.venv\Lib\site-packages\invokeai\app\invocations\baseinvocation.py", line 248, in invoke_internal
    output = self.invoke(context)
             ^^^^^^^^^^^^^^^^^^^^
  File "D:\LMIM\InvokeAI\.venv\Lib\site-packages\torch\utils\_contextlib.py", line 116, in decorate_context
    return func(*args, **kwargs)
           ^^^^^^^^^^^^^^^^^^^^^
  File "D:\LMIM\InvokeAI\.venv\Lib\site-packages\invokeai\app\invocations\flux2_dev_text_encoder.py", line 105, in invoke
    mistral_embeds = self._encode_prompt(context, exit_stack)
                     ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "D:\LMIM\InvokeAI\.venv\Lib\site-packages\invokeai\app\invocations\flux2_dev_text_encoder.py", line 131, in _encode_prompt
    text_encoder_info = context.models.load(self.mistral_encoder.text_encoder)
                        ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "D:\LMIM\InvokeAI\.venv\Lib\site-packages\invokeai\app\services\shared\invocation_context.py", line 553, in load
    return self._services.model_manager.load.load_model(model, submodel_type, user_id=self._data.queue_item.user_id)
           ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "D:\LMIM\InvokeAI\.venv\Lib\site-packages\invokeai\app\services\model_load\model_load_default.py", line 100, in load_model
    ).load_model(model_config, submodel_type)
      ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "D:\LMIM\InvokeAI\.venv\Lib\site-packages\invokeai\backend\model_manager\load\load_default.py", line 224, in load_model
    cache_record = self._load_and_cache(model_config, submodel_type)
                   ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "D:\LMIM\InvokeAI\.venv\Lib\site-packages\invokeai\backend\model_manager\load\load_default.py", line 309, in _load_and_cache
    loaded_model = put_in_eval_mode(self._load_model(config, submodel_type))
                                    ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "D:\LMIM\InvokeAI\.venv\Lib\site-packages\invokeai\backend\model_manager\load\model_loaders\mistral_encoder.py", line 908, in _load_model
    return self._load_text_encoder(config)
           ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "D:\LMIM\InvokeAI\.venv\Lib\site-packages\invokeai\backend\model_manager\load\model_loaders\mistral_encoder.py", line 955, in _load_text_encoder
    missing, unexpected = model.load_state_dict(sd, strict=False, assign=True)
                          ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "D:\LMIM\InvokeAI\.venv\Lib\site-packages\torch\nn\modules\module.py", line 2593, in load_state_dict
    raise RuntimeError(
RuntimeError: Error(s) in loading state_dict for MistralModel:

What you expected to happen

Invoke to process the text encoding with the indicated FP4 Mistral model.

How to reproduce the problem

Download the FP4 Mistral 3 Small 24B text encoder from the Starter Models, along with some variant of FLUX.2 Dev, and attempt to generate an image with them.

Additional context

If relevant, I was trying to use a Q4 GGUF of the FLUX.2 Dev model itself, but it doesn't look like it got that far.

I also attempted to try the FP8 version of the encoder from the Starter Models, which seemed to get further, but then crashes Invoke silently with no errors to console (not sure if this one is just overloading my RAM resources):

[2026-09-01 16:17:38,504]::[InvokeAI]::INFO --> Executing queue item 31821, session dcf63456-1fe3-47d6-b40a-0be6d4e929cf on cuda:0
[2026-09-01 16:18:19,887]::[MistralEncoderCheckpointLoader]::INFO --> Dequantized 210 Comfy-Org-style quantized weights
[2026-09-01 16:18:19,904]::[MistralEncoderCheckpointLoader]::INFO --> Mistral encoder config (checkpoint): layers=30, hidden=5120, heads=32, kv_heads=8, intermediate=32768
[2026-09-01 16:18:20,435]::[MistralEncoderCheckpointLoader]::INFO --> Replaced model.norm with Identity for 30-layer cow Mistral (final_norm=False).
[2026-09-01 16:18:58,414]::[ModelManagerService]::INFO --> [MODEL CACHE] Loaded model 'e5f8c8e0-027e-4f98-8984-b7dd8640a673:text_encoder' (MistralModel) onto cuda device #0 in 37.73s. Total model size: 33080.59MB, VRAM: 19960.59MB (60.3%)
Desired action:
1. Generate images with the browser-based interface
2. Open the developer console
3. Command-line help
Q - Quit

To update, download and run the installer from https://github.com/invoke-ai/InvokeAI/releases/latest

Discord username

No response

Activity

Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment

Metadata

Metadata

Labels

bugSomething isn't working

Type

No type

Projects

No projects

    Milestone

    No milestone

    Relationships

    None yet

    Development

    No branches or pull requests

    Issue actions