Skip to content

[bug]: Multi-GPU generation intermittently stalls during l2i / post-denoise processing #9546

Description

@Baelin-fishin

Is there an existing issue for this problem?

  • I have searched the existing issues

Install method

Invoke's Launcher

Operating system

Windows

GPU vendor

Nvidia (CUDA)

GPU model

RTX 5070 ti, RTX 3070 ti, RTX 3070

GPU VRAM

16GB, 8GB, 8GB

Version number

v6.14.0

Browser

No response

System Information

{
"version": "6.14.0",
"dependencies": {
"absl-py" : "2.4.0",
"accelerate" : "1.14.0",
"annotated-doc" : "0.0.4",
"annotated-types" : "0.7.0",
"anyio" : "4.14.1",
"argon2-cffi" : "25.1.0",
"argon2-cffi-bindings" : "25.1.0",
"arrow" : "1.4.0",
"asttokens" : "3.0.1",
"async-lru" : "2.3.0",
"attrs" : "26.1.0",
"babel" : "2.18.0",
"bcrypt" : "3.2.2",
"beautifulsoup4" : "4.15.0",
"bidict" : "0.23.1",
"bitsandbytes" : "0.49.2",
"blake3" : "1.0.9",
"bleach" : "6.4.0",
"certifi" : "2026.6.17",
"cffi" : "2.0.0",
"charset-normalizer" : "3.4.7",
"click" : "8.4.2",
"colorama" : "0.4.6",
"coloredlogs" : "15.0.1",
"comm" : "0.2.3",
"compel" : "2.4.0",
"contourpy" : "1.3.3",
"cryptography" : "49.0.0",
"CUDA" : "12.8",
"cycler" : "0.12.1",
"debugpy" : "1.8.21",
"decorator" : "5.3.1",
"defusedxml" : "0.7.1",
"Deprecated" : "1.3.1",
"diffusers" : "0.39.0",
"dnspython" : "2.8.0",
"dynamicprompts" : "0.31.0",
"ecdsa" : "0.19.2",
"einops" : "0.8.2",
"email-validator" : "2.3.0",
"executing" : "2.2.1",
"fastapi" : "0.141.1",
"fastapi-events" : "0.12.2",
"fastjsonschema" : "2.21.2",
"filelock" : "3.29.4",
"flatbuffers" : "25.12.19",
"fonttools" : "4.63.0",
"fqdn" : "1.5.1",
"fsspec" : "2026.6.0",
"gguf" : "0.19.0",
"h11" : "0.16.0",
"hf-xet" : "1.5.1",
"httpcore" : "1.0.9",
"httptools" : "0.8.0",
"httpx" : "0.28.1",
"huggingface_hub" : "1.21.0",
"humanfriendly" : "10.0",
"idna" : "3.18",
"ImageIO" : "2.37.4",
"imageio-ffmpeg" : "0.6.0",
"importlib_metadata" : "9.0.0",
"InvokeAI" : "6.14.0",
"ipykernel" : "7.3.0",
"ipython" : "9.15.0",
"ipython_pygments_lexers" : "1.1.1",
"isoduration" : "20.11.0",
"jax" : "0.7.1",
"jaxlib" : "0.7.1",
"jedi" : "0.20.0",
"Jinja2" : "3.1.6",
"json5" : "0.15.0",
"jsonpointer" : "3.1.1",
"jsonschema" : "4.26.0",
"jsonschema-specifications": "2025.9.1",
"jupyter-events" : "0.12.1",
"jupyter-lsp" : "2.3.1",
"jupyter_builder" : "1.0.2",
"jupyter_client" : "8.9.1",
"jupyter_core" : "5.9.1",
"jupyter_server" : "2.20.0",
"jupyter_server_terminals" : "0.5.4",
"jupyterlab" : "4.6.0",
"jupyterlab_pygments" : "0.3.0",
"jupyterlab_server" : "2.28.0",
"kiwisolver" : "1.5.0",
"lark" : "1.3.1",
"markdown-it-py" : "4.2.0",
"MarkupSafe" : "3.0.3",
"matplotlib" : "3.11.0",
"matplotlib-inline" : "0.2.2",
"mdurl" : "0.1.2",
"mediapipe" : "0.10.14",
"mistral_common" : "1.11.6",
"mistune" : "3.3.2",
"ml_dtypes" : "0.5.4",
"mpmath" : "1.3.0",
"nbclient" : "0.11.0",
"nbconvert" : "7.17.1",
"nbformat" : "5.10.4",
"nest-asyncio2" : "1.7.2",
"networkx" : "3.6.1",
"notebook" : "7.6.0",
"notebook_shim" : "0.2.4",
"numpy" : "1.26.4",
"onnx" : "1.16.1",
"onnxruntime" : "1.19.2",
"opencv-contrib-python" : "4.11.0.86",
"opt_einsum" : "3.4.0",
"packaging" : "26.2",
"pandocfilters" : "1.5.1",
"parso" : "0.8.7",
"passlib" : "1.7.4",
"picklescan" : "1.0.4",
"pillow" : "12.2.0",
"platformdirs" : "4.10.0",
"prometheus_client" : "0.25.0",
"prompt_toolkit" : "3.0.52",
"protobuf" : "4.25.9",
"psutil" : "7.2.2",
"pure_eval" : "0.2.3",
"pyasn1" : "0.6.3",
"pycountry" : "26.2.16",
"pycparser" : "3.0",
"pydantic" : "2.13.4",
"pydantic-extra-types" : "2.11.1",
"pydantic-settings" : "2.14.2",
"pydantic_core" : "2.46.4",
"Pygments" : "2.20.0",
"pyparsing" : "3.3.2",
"PyPatchMatch" : "1.0.2",
"pyreadline3" : "3.5.6",
"python-dateutil" : "2.9.0.post0",
"python-dotenv" : "1.2.2",
"python-engineio" : "4.13.3",
"python-jose" : "3.5.0",
"python-json-logger" : "4.1.0",
"python-multipart" : "0.0.32",
"python-socketio" : "5.16.3",
"PyWavelets" : "1.9.0",
"pywinpty" : "3.0.5",
"PyYAML" : "6.0.3",
"pyzmq" : "27.1.0",
"referencing" : "0.37.0",
"regex" : "2026.5.9",
"requests" : "2.34.2",
"rfc3339-validator" : "0.1.4",
"rfc3986-validator" : "0.1.1",
"rfc3987-syntax" : "1.1.0",
"rich" : "15.0.0",
"rpds-py" : "2026.5.1",
"rsa" : "4.9.1",
"safetensors" : "0.8.0",
"scipy" : "1.17.1",
"semver" : "3.0.4",
"Send2Trash" : "2.1.0",
"sentencepiece" : "0.2.0",
"setuptools" : "82.0.1",
"shellingham" : "1.5.4",
"simple-websocket" : "1.1.0",
"six" : "1.17.0",
"sounddevice" : "0.5.5",
"soupsieve" : "2.8.4",
"spandrel" : "0.4.2",
"stack-data" : "0.6.3",
"starlette" : "0.48.0",
"sympy" : "1.14.0",
"terminado" : "0.18.1",
"tiktoken" : "0.13.0",
"tinycss2" : "1.5.1",
"tokenizers" : "0.22.2",
"torch" : "2.7.1+cu128",
"torchsde" : "0.2.6",
"torchvision" : "0.22.1+cu128",
"tornado" : "6.5.7",
"tqdm" : "4.68.3",
"traitlets" : "5.15.1",
"trampoline" : "0.1.2",
"transformers" : "5.5.4",
"typer" : "0.25.1",
"typing-inspection" : "0.4.2",
"typing_extensions" : "4.15.0",
"tzdata" : "2026.2",
"uri-template" : "1.3.0",
"urllib3" : "2.7.0",
"uvicorn" : "0.49.0",
"watchfiles" : "1.2.0",
"wcwidth" : "0.8.1",
"webcolors" : "25.10.0",
"webencodings" : "0.5.1",
"websocket-client" : "1.9.0",
"websockets" : "16.0",
"wrapt" : "2.2.2",
"wsproto" : "1.3.2",
"zipp" : "4.1.0"
},
"config": {
"schema_version": "4.0.3",
"legacy_models_yaml_path": null,
"host": "127.0.0.1",
"port": 9090,
"allow_origins": [],
"allow_credentials": true,
"allow_methods": [""],
"allow_headers": ["
"],
"ssl_certfile": null,
"ssl_keyfile": null,
"base_url": null,
"forwarded_allow_ips": "127.0.0.1",
"http_compression_level": 9,
"log_tokenization": false,
"patchmatch": true,
"models_dir": "models",
"convert_cache_dir": "models\.convert_cache",
"download_cache_dir": "models\.download_cache",
"legacy_conf_dir": "configs",
"db_dir": "databases",
"outputs_dir": "D:\AI\InvokeV6\Outputs",
"image_subfolder_strategy": "date",
"custom_nodes_dir": "nodes",
"style_presets_dir": "style_presets",
"workflow_thumbnails_dir": "workflow_thumbnails",
"log_handlers": ["console"],
"log_format": "color",
"log_level": "info",
"log_sql": false,
"log_level_network": "warning",
"use_memory_db": false,
"dev_reload": false,
"profile_graphs": false,
"profile_prefix": null,
"profiles_dir": "profiles",
"max_cache_ram_gb": null,
"max_cache_vram_gb": null,
"log_memory_usage": false,
"model_cache_keep_alive_min": 0,
"device_working_mem_gb": 3,
"enable_partial_loading": true,
"keep_ram_copy_of_weights": true,
"ram": null,
"vram": null,
"lazy_offload": true,
"pytorch_cuda_alloc_conf": null,
"device": "auto",
"generation_devices": ["cuda:1", "cuda:0"],
"offload_text_encoders_to_idle_gpus": false,
"precision": "auto",
"sequential_guidance": false,
"wan_memory_optimization": false,
"pid_memory_optimization": false,
"attention_type": "auto",
"attention_slice_size": "auto",
"force_tiled_decode": false,
"pil_compress_level": 1,
"max_queue_size": 10000,
"session_queue_mode": "FIFO",
"clear_queue_on_startup": false,
"max_queue_history": null,
"allow_nodes": null,
"deny_nodes": null,
"node_cache_size": 512,
"hashing_algorithm": "blake3_single",
"remote_api_tokens": null,
"scan_models_on_startup": false,
"allow_private_download_urls": false,
"download_proxy": null,
"unsafe_disable_picklescan": false,
"allow_unknown_models": true,
"multiuser": false,
"strict_password_checking": false,
"external_alibabacloud_api_key": null,
"external_alibabacloud_base_url": null,
"external_gemini_api_key": null,
"external_openai_api_key": null,
"external_gemini_base_url": null,
"external_openai_base_url": null,
"external_seedream_api_key": null,
"external_seedream_base_url": null
},
"set_config_fields": [
"legacy_models_yaml_path",
"outputs_dir",
"session_queue_mode",
"image_subfolder_strategy",
"offload_text_encoders_to_idle_gpus",
"generation_devices"
]
}

What happened

Multi-GPU generation in InvokeAI 6.14.0 pre-release works functionally and both GPUs are able to denoise concurrently at their expected individual speeds. However, overall batch throughput is inconsistent because one of the concurrent workers intermittently incurs substantial additional execution time outside the main denoising workload, most visibly in the l2i stage.

I originally observed this using an RTX 5070 Ti + RTX 3070 Ti. To rule out a Blackwell/Ampere interaction or the large performance difference between the cards, I physically removed the RTX 5070 Ti and replaced it with an RTX 3070.

The same behavior is reproducible with an RTX 3070 + RTX 3070 Ti.

Both GPUs can simultaneously reach normal utilization, clocks and denoising throughput. The slowdown therefore appears to occur elsewhere in concurrent graph execution rather than being an inability to execute CUDA workloads simultaneously.

Example from an RTX 3070 + RTX 3070 Ti run:

Worker A:
denoise_latents: 9.176s
l2i: 0.662s
Total graph: 9.859s

Worker B:
denoise_latents: 11.726s
l2i: 2.765s
Total graph: 14.761s

The l2i stage on Worker B therefore took more than 4x as long. Other runs have also shown intermittent multi-second l2i times.

This does not happen consistently during equivalent single-GPU testing.

Variables tested/eliminated:

  • RTX 5070 Ti + RTX 3070 Ti reproduced issue
  • RTX 3070 + RTX 3070 Ti reproduced issue
  • Each GPU tested individually
  • Output directory tested on HDD and NVMe
  • Image preview disabled
  • CPU noise disabled
  • LoRAs disabled
  • Model cache shows zero misses during affected runs

There may be contention/synchronization around VAE/l2i or another shared resource, but I don't have enough instrumentation to determine the cause.

I realize 6.14.0 is pre-release software. I'm reporting this because the multi-GPU behaviour is reproducible and the testing allowed several possible hardware/configuration causes to be eliminated.

What you expected to happen

Both GPU workers should execute independently, with each GPU completing the workflow with approximately the same non-denoising overhead seen when that GPU is used alone.

Aggregate multi-GPU throughput should therefore approach the combined throughput of the individual GPUs, allowing for normal scheduling overhead.

How to reproduce the problem

  1. Configure two NVIDIA GPUs in invoke.yaml:

generation_devices:

  • cuda:0
  • cuda:1
  1. Restart InvokeAI.

  2. Queue multiple otherwise identical images so that both GPUs receive work concurrently.

  3. Allow several consecutive generations to run.

  4. Compare the graph statistics for the two workers, particularly denoise_latents and l2i.

  5. Observe that both GPUs denoise concurrently at their expected individual speeds, but one worker will intermittently incur substantially longer execution time in l2i and/or surrounding graph stages.

The behaviour was reproduced using both:

  • RTX 5070 Ti + RTX 3070 Ti
  • RTX 3070 + RTX 3070 Ti

3070_3070Ti console log.txt

Additional context

The RTX 3070 + RTX 3070 Ti test was performed specifically to eliminate the RTX 5070 Ti/RTX3070Ti Blackwell/Ampere architecture as a variable. The 5070 Ti was physically removed from the system and replaced by the RTX 3070; the 3070 Ti remained installed in the second slot.

The issue remained reproducible with the two Ampere GPUs.

During concurrent denoising, both cards show expected utilization and performance. Example observed denoising rates were approximately:

GPU worker 1: 2.99-3.00 it/s
GPU worker 2: 2.57-2.58 it/s

This suggests concurrent CUDA execution itself is functioning correctly.

One particularly interesting timing example showed one worker completing its graph only ~18 ms before the other worker loaded its AutoencoderKL for the VAE/l2i stage. The latter worker subsequently reported l2i = 2.765s versus 0.662s for the other worker. This may or may not be causal, but could be useful when investigating synchronization between workers.

I can provide the complete InvokeAI console logs and nvidia-smi monitoring output from the testing if useful.

Secondary observation: During wider multi-GPU testing I encountered at least one denoise_latents failure reporting "CUDA error: an illegal memory access was encountered". I have not established whether that is related to this performance issue.

Discord username

No response

Activity

Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment

Metadata

Metadata

Assignees

Labels

bugSomething isn't working

Type

No type

Projects

No projects

    Milestone

    No milestone

    Relationships

    None yet

    Development

    No branches or pull requests

    Issue actions