invoke-ai / invoke-ai/InvokeAI

[bug]: Z-Image GGUF fails with "Multiple dispatch failed for 'torch._ops.aten.where.self'" on certain quantized models

Open
#9,464 0 comments 1 reaction 1 assignee Claimed by @Pfannkuchensack View on GitHub
bug
Dominant language
Python
Stars
28.2k
Forks
3k
Avg merge
6d 5h
Merged PRs (30d)
19

Description

### Is there an existing issue for this problem?

- [x] I have searched the existing issues

### Install method

Invoke's Launcher

### Operating system

Windows

### GPU vendor

Nvidia (CUDA)

### GPU model

RTX 3060

### GPU VRAM

12GB

### Version number

6.14.0-rc1

### Browser

_No response_

### System Information

{
"version": "6.14.0-rc1",
"dependencies": {
"absl-py" : "2.5.0",
"accelerate" : "1.14.0",
"annotated-doc" : "0.0.5",
"annotated-types" : "0.8.0",
"anyio" : "4.14.2",
"argon2-cffi" : "25.1.0",
"argon2-cffi-bindings" : "25.1.0",
"arrow" : "1.4.0",
"asttokens" : "3.0.2",
"async-lru" : "2.3.0",
"attrs" : "26.1.0",
"babel" : "2.18.0",
"bcrypt" : "3.2.2",
"beautifulsoup4" : "4.15.0",
"bidict" : "0.23.1",
"bitsandbytes" : "0.50.0",
"blake3" : "1.0.9",
"bleach" : "6.4.0",
"certifi" : "2022.12.7",
"cffi" : "2.1.0",
"charset-normalizer" : "2.1.1",
"click" : "8.4.2",
"colorama" : "0.4.6",
"coloredlogs" : "15.0.1",
"comm" : "0.2.3",
"compel" : "2.4.0",
"contourpy" : "1.3.3",
"cryptography" : "50.0.0",
"CUDA" : "12.8",
"cycler" : "0.12.1",
"debugpy" : "1.8.21",
"decorator" : "5.3.1",
"defusedxml" : "0.7.1",
"Deprecated" : "1.3.1",
"diffusers" : "0.39.0",
"dnspython" : "2.8.0",
"dynamicprompts" : "0.31.0",
"ecdsa" : "0.19.2",
"einops" : "0.8.2",
"email-validator" : "2.3.0",
"executing" : "2.2.1",
"fastapi" : "0.118.3",
"fastapi-events" : "0.12.2",
"fastjsonschema" : "2.22.1",
"filelock" : "3.29.0",
"flatbuffers" : "25.12.19",
"fonttools" : "4.63.0",
"fqdn" : "1.5.1",
"fsspec" : "2026.4.0",
"gguf" : "0.19.0",
"h11" : "0.16.0",
"hf-xet" : "1.5.2",
"httpcore" : "1.0.9",
"httptools" : "0.8.0",
"httpx" : "0.28.1",
"huggingface_hub" : "1.26.0",
"humanfriendly" : "10.0",
"idna" : "3.4",
"ImageIO" : "2.37.4",
"imageio-ffmpeg" : "0.6.0",
"importlib_metadata" : "7.1.0",
"InvokeAI" : "6.14.0rc1",
"ipykernel" : "7.3.0",
"ipython" : "9.16.0",
"ipython_pygments_lexers" : "1.1.1",
"isoduration" : "20.11.0",
"jax" : "0.7.1",
"jaxlib" : "0.7.1",
"jedi" : "0.20.0",
"Jinja2" : "3.1.6",
"json5" : "0.15.0",
"jsonpointer" : "3.1.1",
"jsonschema" : "4.26.0",
"jsonschema-specifications": "2025.9.1",
"jupyter-events" : "0.12.1",
"jupyter-lsp" : "2.3.1",
"jupyter_client" : "8.9.1",
"jupyter_core" : "5.9.1",
"jupyter_server" : "2.20.0",
"jupyter_server_terminals" : "0.5.4",
"jupyterlab" : "4.1.6",
"jupyterlab_pygments" : "0.3.0",
"jupyterlab_server" : "2.24.0",
"kiwisolver" : "1.5.0",
"lark" : "1.3.1",
"markdown-it-py" : "4.2.0",
"MarkupSafe" : "3.0.3",
"matplotlib" : "3.11.1",
"matplotlib-inline" : "0.2.2",
"mdurl" : "0.1.2",
"mediapipe" : "0.10.14",
"mistune" : "3.3.4",
"ml_dtypes" : "0.5.4",
"mpmath" : "1.3.0",
"nbclient" : "0.11.0",
"nbconvert" : "7.17.1",
"nbformat" : "5.10.4",
"nest-asyncio2" : "1.7.2",
"networkx" : "3.6.1",
"notebook" : "7.1.3",
"notebook_shim" : "0.2.4",
"numpy" : "1.26.4",
"onnx" : "1.16.1",
"onnxruntime" : "1.19.2",
"opencv-contrib-python" : "4.11.0.86",
"opt_einsum" : "3.4.0",
"packaging" : "24.1",
"pandocfilters" : "1.5.1",
"parso" : "0.8.7",
"passlib" : "1.7.4",
"picklescan" : "1.0.5",
"pillow" : "12.2.0",
"platformdirs" : "4.11.0",
"prometheus_client" : "0.26.0",
"prompt_toolkit" : "3.0.53",
"protobuf" : "4.25.9",
"psutil" : "7.2.2",
"pure_eval" : "0.2.3",
"pyasn1" : "0.6.4",
"pycparser" : "3.0",
"pydantic" : "2.13.4",
"pydantic-settings" : "2.14.2",
"pydantic_core" : "2.46.4",
"Pygments" : "2.20.0",
"pyparsing" : "3.3.2",
"PyPatchMatch" : "1.0.2",
"pyreadline3" : "3.5.6",
"python-dateutil" : "2.9.0.post0",
"python-dotenv" : "1.2.2",
"python-engineio" : "4.13.4",
"python-jose" : "3.5.0",
"python-json-logger" : "4.1.0",
"python-multipart" : "0.0.32",
"python-socketio" : "5.16.3",
"PyWavelets" : "1.9.0",
"pywinpty" : "3.0.5",
"PyYAML" : "6.0.3",
"pyzmq" : "27.1.0",
"referencing" : "0.37.0",
"regex" : "2026.7.19",
"requests" : "2.28.1",
"rfc3339-validator" : "0.1.4",
"rfc3986-validator" : "0.1.1",
"rfc3987-syntax" : "1.1.0",
"rich" : "15.0.0",
"rpds-py" : "2026.6.3",
"rsa" : "4.9.1",
"safetensors" : "0.8.0",
"scipy" : "1.17.1",
"semver" : "3.0.4",
"Send2Trash" : "2.1.0",
"sentencepiece" : "0.2.0",
"setuptools" : "78.1.0",
"shellingham" : "1.5.4",
"simple-websocket" : "1.1.0",
"six" : "1.17.0",
"sounddevice" : "0.5.5",
"soupsieve" : "2.9.1",
"spandrel" : "0.4.2",
"stack-data" : "0.6.3",
"starlette" : "0.48.0",
"sympy" : "1.14.0",
"terminado" : "0.18.1",
"tinycss2" : "1.5.1",
"tokenizers" : "0.22.2",
"torch" : "2.11.0+cu128",
"torchsde" : "0.2.6",
"torchvision" : "0.26.0+cu128",
"tornado" : "6.5.7",
"tqdm" : "4.66.5",
"traitlets" : "5.16.0",
"trampoline" : "0.1.2",
"transformers" : "5.5.4",
"typer" : "0.27.0",
"typing-inspection" : "0.4.2",
"typing_extensions" : "4.15.0",
"tzdata" : "2026.3",
"uri-template" : "1.3.0",
"urllib3" : "1.26.13",
"uvicorn" : "0.52.0",
"watchfiles" : "1.2.0",
"wcwidth" : "0.8.2",
"webcolors" : "25.10.0",
"webencodings" : "0.5.1",
"websocket-client" : "1.9.0",
"websockets" : "17.0.1",
"wrapt" : "2.3.0",
"wsproto" : "1.3.2",
"zipp" : "3.19.2"
},
"config": {
"schema_version": "4.0.3",
"legacy_models_yaml_path": null,
"host": "127.0.0.1",
"port": 9090,
"allow_origins": [],
"allow_credentials": true,
"allow_methods": ["*"],
"allow_headers": ["*"],
"ssl_certfile": null,
"ssl_keyfile": null,
"base_url": null,
"forwarded_allow_ips": "127.0.0.1",
"log_tokenization": false,
"patchmatch": true,
"models_dir": "models",
"convert_cache_dir": "models\\.convert_cache",
"download_cache_dir": "models\\.download_cache",
"legacy_conf_dir": "configs",
"db_dir": "databases",
"outputs_dir": "W:\\AI\\InvokeAI\\outputs",
"image_subfolder_strategy": "date",
"custom_nodes_dir": "nodes",
"style_presets_dir": "style_presets",
"workflow_thumbnails_dir": "workflow_thumbnails",
"log_handlers": ["console"],
"log_format": "color",
"log_level": "info",
"log_sql": false,
"log_level_network": "warning",
"use_memory_db": false,
"dev_reload": false,
"profile_graphs": false,
"profile_prefix": null,
"profiles_dir": "profiles",
"max_cache_ram_gb": null,
"max_cache_vram_gb": null,
"log_memory_usage": false,
"model_cache_keep_alive_min": 0,
"device_working_mem_gb": 0.9,
"enable_partial_loading": true,
"keep_ram_copy_of_weights": true,
"ram": null,
"vram": null,
"lazy_offload": true,
"pytorch_cuda_alloc_conf": "backend:cudaMallocAsync",
"device": "cuda",
"generation_devices": "auto",
"offload_text_encoders_to_idle_gpus": true,
"precision": "bfloat16",
"sequential_guidance": false,
"attention_type": "torch-sdp",
"attention_slice_size": "auto",
"force_tiled_decode": true,
"pil_compress_level": 1,
"max_queue_size": 10000,
"session_queue_mode": "round_robin",
"clear_queue_on_startup": false,
"max_queue_history": 0,
"allow_nodes": null,
"deny_nodes": null,
"node_cache_size": 512,
"hashing_algorithm": "random",
"remote_api_tokens": [
{"url_regex": "civitai.com", "token": "**********"},
{"url_regex": "huggingface.co", "token": "**********"},
{"url_regex": "civitai.red", "token": "**********"}
],
"scan_models_on_startup": false,
"unsafe_disable_picklescan": false,
"allow_unknown_models": true,
"multiuser": false,
"strict_password_checking": false,
"external_alibabacloud_api_key": null,
"external_alibabacloud_base_url": null,
"external_gemini_api_key": null,
"external_openai_api_key": null,
"external_gemini_base_url": null,
"external_openai_base_url": null,
"external_seedream_api_key": null,
"external_seedream_base_url": null
},
"set_config_fields": [
"attention_type", "max_queue_history", "precision", "hashing_algorithm",
"pytorch_cuda_alloc_conf", "image_subfolder_strategy", "device", "force_tiled_decode",
"outputs_dir", "legacy_models_yaml_path", "device_working_mem_gb", "remote_api_tokens"
]
}

### What happened

Some quantized GGUF models fail, such as this one: https://huggingface.co/unsloth/Z-Image-Turbo-GGUF

The problem was reported by a user to the Discord: https://discord.com/channels/1020123559063990373/1149506274971631688/1534429470453399553

I asked Claude about the error, and here's what it told me: "Some Z-Image GGUF models fail to run with a TypeError during the transformer's forward pass, specifically when torch.where() is called on a tensor that's still in InvokeAI's quantized GGMLTensor form. Other Z-Image GGUFs (converted from BF16 source via stable-diffusion.cpp, uniform Q8_0 quantization) load and run without issue — so this appears specific to how certain GGUF files were quantized, not a general GGUF-loading problem.

"My guess (not confirmed) is that GGMLTensor only implements __torch_dispatch__ handlers for the common ops needed by a typical forward pass (linear, matmul, etc.), and aten.where.self isn't among them. The failure seems to depend on whether the pad_token tensor specifically ends up quantized/wrapped as a GGMLTensor in a given file — my own conversions may leave small tensors like this unquantized, while some third-party GGUFs (possibly using a mixed/per-tensor quantization approach) quantize it too, exposing this code path."

```
[2026-08-05 10:30:15,498]::[InvokeAI]::ERROR --> Error while invoking session dd977b1e-552e-4e27-94be-b0806d1e7a91, invocation 95590a46-f898-4912-9d36-ba759fd5b108 (z_image_denoise): Multiple dispatch f
ailed for 'torch._ops.aten.where.self'; all __torch_dispatch__ handlers returned NotImplemented:

- tensor subclass

For more information, try re-running with TORCH_LOGS=not_implemented
[2026-08-05 10:30:15,499]::[InvokeAI]::ERROR --> Traceback (most recent call last):
File "W:\AI\InvokeAI\.venv\Lib\site-packages\invokeai\app\services\session_processor\session_processor_default.py", line 148, in run_node
output = invocation.invoke_internal(context=context, services=self._services)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "W:\AI\InvokeAI\.venv\Lib\site-packages\invokeai\app\invocations\baseinvocation.py", line 248, in invoke_internal
output = self.invoke(context)
^^^^^^^^^^^^^^^^^^^^
File "W:\AI\InvokeAI\.venv\Lib\site-packages\torch\utils\_contextlib.py", line 124, in decorate_context
return func(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^
File "W:\AI\InvokeAI\.venv\Lib\site-packages\invokeai\app\invocations\z_image_denoise.py", line 130, in invoke
latents = self._run_diffusion(context)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "W:\AI\InvokeAI\.venv\Lib\site-packages\invokeai\app\invocations\z_image_denoise.py", line 722, in _run_diffusion
model_output = transformer(
^^^^^^^^^^^^
File "W:\AI\InvokeAI\.venv\Lib\site-packages\torch\nn\modules\module.py", line 1779, in _wrapped_call_impl
return self._call_impl(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "W:\AI\InvokeAI\.venv\Lib\site-packages\torch\nn\modules\module.py", line 1790, in _call_impl
return forward_call(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "W:\AI\InvokeAI\.venv\Lib\site-packages\diffusers\models\transformers\transformer_z_image.py", line 981, in forward
x, x_freqs, x_mask, _, x_noise_tensor = self._prepare_sequence(
^^^^^^^^^^^^^^^^^^^^^^^
File "W:\AI\InvokeAI\.venv\Lib\site-packages\diffusers\models\transformers\transformer_z_image.py", line 781, in _prepare_sequence
feats_cat = torch.where(mask, pad_token, feats_cat)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
TypeError: Multiple dispatch failed for 'torch._ops.aten.where.self'; all __torch_dispatch__ handlers returned NotImplemented:

- tensor subclass

For more information, try re-running with TORCH_LOGS=not_implemented
```

### What you expected to happen

I expected the model to work, like many other GGUF quants of ZiT have.

### How to reproduce the problem

1. Install one of the models from https://huggingface.co/unsloth/Z-Image-Turbo-GGUF (I used Q8_0).
2. Try to generate an image with it.
3. Get the error.

### Additional context

GGUF files I converted myself (using stable-diffusion-gguf-converter, BF16 source with uniform Q8_0 quantization) load and run correctly on the same InvokeAI install. Only some third-party GGUFs fail this way.

### Discord username

shanedk

Contributor guide

No contributing guide indexed for this repository

Assessment

This issue has not been assessed yet.

Get new issues in your inbox

A short digest of beginner-friendly GitHub issues.