Skip to content

[bug]: Z-Image GGUF fails with "Multiple dispatch failed for 'torch._ops.aten.where.self'" on certain quantized models #9464

Description

@shanedk

Is there an existing issue for this problem?

  • I have searched the existing issues

Install method

Invoke's Launcher

Operating system

Windows

GPU vendor

Nvidia (CUDA)

GPU model

RTX 3060

GPU VRAM

12GB

Version number

6.14.0-rc1

Browser

No response

System Information

{
"version": "6.14.0-rc1",
"dependencies": {
"absl-py" : "2.5.0",
"accelerate" : "1.14.0",
"annotated-doc" : "0.0.5",
"annotated-types" : "0.8.0",
"anyio" : "4.14.2",
"argon2-cffi" : "25.1.0",
"argon2-cffi-bindings" : "25.1.0",
"arrow" : "1.4.0",
"asttokens" : "3.0.2",
"async-lru" : "2.3.0",
"attrs" : "26.1.0",
"babel" : "2.18.0",
"bcrypt" : "3.2.2",
"beautifulsoup4" : "4.15.0",
"bidict" : "0.23.1",
"bitsandbytes" : "0.50.0",
"blake3" : "1.0.9",
"bleach" : "6.4.0",
"certifi" : "2022.12.7",
"cffi" : "2.1.0",
"charset-normalizer" : "2.1.1",
"click" : "8.4.2",
"colorama" : "0.4.6",
"coloredlogs" : "15.0.1",
"comm" : "0.2.3",
"compel" : "2.4.0",
"contourpy" : "1.3.3",
"cryptography" : "50.0.0",
"CUDA" : "12.8",
"cycler" : "0.12.1",
"debugpy" : "1.8.21",
"decorator" : "5.3.1",
"defusedxml" : "0.7.1",
"Deprecated" : "1.3.1",
"diffusers" : "0.39.0",
"dnspython" : "2.8.0",
"dynamicprompts" : "0.31.0",
"ecdsa" : "0.19.2",
"einops" : "0.8.2",
"email-validator" : "2.3.0",
"executing" : "2.2.1",
"fastapi" : "0.118.3",
"fastapi-events" : "0.12.2",
"fastjsonschema" : "2.22.1",
"filelock" : "3.29.0",
"flatbuffers" : "25.12.19",
"fonttools" : "4.63.0",
"fqdn" : "1.5.1",
"fsspec" : "2026.4.0",
"gguf" : "0.19.0",
"h11" : "0.16.0",
"hf-xet" : "1.5.2",
"httpcore" : "1.0.9",
"httptools" : "0.8.0",
"httpx" : "0.28.1",
"huggingface_hub" : "1.26.0",
"humanfriendly" : "10.0",
"idna" : "3.4",
"ImageIO" : "2.37.4",
"imageio-ffmpeg" : "0.6.0",
"importlib_metadata" : "7.1.0",
"InvokeAI" : "6.14.0rc1",
"ipykernel" : "7.3.0",
"ipython" : "9.16.0",
"ipython_pygments_lexers" : "1.1.1",
"isoduration" : "20.11.0",
"jax" : "0.7.1",
"jaxlib" : "0.7.1",
"jedi" : "0.20.0",
"Jinja2" : "3.1.6",
"json5" : "0.15.0",
"jsonpointer" : "3.1.1",
"jsonschema" : "4.26.0",
"jsonschema-specifications": "2025.9.1",
"jupyter-events" : "0.12.1",
"jupyter-lsp" : "2.3.1",
"jupyter_client" : "8.9.1",
"jupyter_core" : "5.9.1",
"jupyter_server" : "2.20.0",
"jupyter_server_terminals" : "0.5.4",
"jupyterlab" : "4.1.6",
"jupyterlab_pygments" : "0.3.0",
"jupyterlab_server" : "2.24.0",
"kiwisolver" : "1.5.0",
"lark" : "1.3.1",
"markdown-it-py" : "4.2.0",
"MarkupSafe" : "3.0.3",
"matplotlib" : "3.11.1",
"matplotlib-inline" : "0.2.2",
"mdurl" : "0.1.2",
"mediapipe" : "0.10.14",
"mistune" : "3.3.4",
"ml_dtypes" : "0.5.4",
"mpmath" : "1.3.0",
"nbclient" : "0.11.0",
"nbconvert" : "7.17.1",
"nbformat" : "5.10.4",
"nest-asyncio2" : "1.7.2",
"networkx" : "3.6.1",
"notebook" : "7.1.3",
"notebook_shim" : "0.2.4",
"numpy" : "1.26.4",
"onnx" : "1.16.1",
"onnxruntime" : "1.19.2",
"opencv-contrib-python" : "4.11.0.86",
"opt_einsum" : "3.4.0",
"packaging" : "24.1",
"pandocfilters" : "1.5.1",
"parso" : "0.8.7",
"passlib" : "1.7.4",
"picklescan" : "1.0.5",
"pillow" : "12.2.0",
"platformdirs" : "4.11.0",
"prometheus_client" : "0.26.0",
"prompt_toolkit" : "3.0.53",
"protobuf" : "4.25.9",
"psutil" : "7.2.2",
"pure_eval" : "0.2.3",
"pyasn1" : "0.6.4",
"pycparser" : "3.0",
"pydantic" : "2.13.4",
"pydantic-settings" : "2.14.2",
"pydantic_core" : "2.46.4",
"Pygments" : "2.20.0",
"pyparsing" : "3.3.2",
"PyPatchMatch" : "1.0.2",
"pyreadline3" : "3.5.6",
"python-dateutil" : "2.9.0.post0",
"python-dotenv" : "1.2.2",
"python-engineio" : "4.13.4",
"python-jose" : "3.5.0",
"python-json-logger" : "4.1.0",
"python-multipart" : "0.0.32",
"python-socketio" : "5.16.3",
"PyWavelets" : "1.9.0",
"pywinpty" : "3.0.5",
"PyYAML" : "6.0.3",
"pyzmq" : "27.1.0",
"referencing" : "0.37.0",
"regex" : "2026.7.19",
"requests" : "2.28.1",
"rfc3339-validator" : "0.1.4",
"rfc3986-validator" : "0.1.1",
"rfc3987-syntax" : "1.1.0",
"rich" : "15.0.0",
"rpds-py" : "2026.6.3",
"rsa" : "4.9.1",
"safetensors" : "0.8.0",
"scipy" : "1.17.1",
"semver" : "3.0.4",
"Send2Trash" : "2.1.0",
"sentencepiece" : "0.2.0",
"setuptools" : "78.1.0",
"shellingham" : "1.5.4",
"simple-websocket" : "1.1.0",
"six" : "1.17.0",
"sounddevice" : "0.5.5",
"soupsieve" : "2.9.1",
"spandrel" : "0.4.2",
"stack-data" : "0.6.3",
"starlette" : "0.48.0",
"sympy" : "1.14.0",
"terminado" : "0.18.1",
"tinycss2" : "1.5.1",
"tokenizers" : "0.22.2",
"torch" : "2.11.0+cu128",
"torchsde" : "0.2.6",
"torchvision" : "0.26.0+cu128",
"tornado" : "6.5.7",
"tqdm" : "4.66.5",
"traitlets" : "5.16.0",
"trampoline" : "0.1.2",
"transformers" : "5.5.4",
"typer" : "0.27.0",
"typing-inspection" : "0.4.2",
"typing_extensions" : "4.15.0",
"tzdata" : "2026.3",
"uri-template" : "1.3.0",
"urllib3" : "1.26.13",
"uvicorn" : "0.52.0",
"watchfiles" : "1.2.0",
"wcwidth" : "0.8.2",
"webcolors" : "25.10.0",
"webencodings" : "0.5.1",
"websocket-client" : "1.9.0",
"websockets" : "17.0.1",
"wrapt" : "2.3.0",
"wsproto" : "1.3.2",
"zipp" : "3.19.2"
},
"config": {
"schema_version": "4.0.3",
"legacy_models_yaml_path": null,
"host": "127.0.0.1",
"port": 9090,
"allow_origins": [],
"allow_credentials": true,
"allow_methods": [""],
"allow_headers": ["
"],
"ssl_certfile": null,
"ssl_keyfile": null,
"base_url": null,
"forwarded_allow_ips": "127.0.0.1",
"log_tokenization": false,
"patchmatch": true,
"models_dir": "models",
"convert_cache_dir": "models\.convert_cache",
"download_cache_dir": "models\.download_cache",
"legacy_conf_dir": "configs",
"db_dir": "databases",
"outputs_dir": "W:\AI\InvokeAI\outputs",
"image_subfolder_strategy": "date",
"custom_nodes_dir": "nodes",
"style_presets_dir": "style_presets",
"workflow_thumbnails_dir": "workflow_thumbnails",
"log_handlers": ["console"],
"log_format": "color",
"log_level": "info",
"log_sql": false,
"log_level_network": "warning",
"use_memory_db": false,
"dev_reload": false,
"profile_graphs": false,
"profile_prefix": null,
"profiles_dir": "profiles",
"max_cache_ram_gb": null,
"max_cache_vram_gb": null,
"log_memory_usage": false,
"model_cache_keep_alive_min": 0,
"device_working_mem_gb": 0.9,
"enable_partial_loading": true,
"keep_ram_copy_of_weights": true,
"ram": null,
"vram": null,
"lazy_offload": true,
"pytorch_cuda_alloc_conf": "backend:cudaMallocAsync",
"device": "cuda",
"generation_devices": "auto",
"offload_text_encoders_to_idle_gpus": true,
"precision": "bfloat16",
"sequential_guidance": false,
"attention_type": "torch-sdp",
"attention_slice_size": "auto",
"force_tiled_decode": true,
"pil_compress_level": 1,
"max_queue_size": 10000,
"session_queue_mode": "round_robin",
"clear_queue_on_startup": false,
"max_queue_history": 0,
"allow_nodes": null,
"deny_nodes": null,
"node_cache_size": 512,
"hashing_algorithm": "random",
"remote_api_tokens": [
{"url_regex": "civitai.com", "token": ""},
{"url_regex": "huggingface.co", "token": "
"},
{"url_regex": "civitai.red", "token": "**********"}
],
"scan_models_on_startup": false,
"unsafe_disable_picklescan": false,
"allow_unknown_models": true,
"multiuser": false,
"strict_password_checking": false,
"external_alibabacloud_api_key": null,
"external_alibabacloud_base_url": null,
"external_gemini_api_key": null,
"external_openai_api_key": null,
"external_gemini_base_url": null,
"external_openai_base_url": null,
"external_seedream_api_key": null,
"external_seedream_base_url": null
},
"set_config_fields": [
"attention_type", "max_queue_history", "precision", "hashing_algorithm",
"pytorch_cuda_alloc_conf", "image_subfolder_strategy", "device", "force_tiled_decode",
"outputs_dir", "legacy_models_yaml_path", "device_working_mem_gb", "remote_api_tokens"
]
}

What happened

Some quantized GGUF models fail, such as this one: https://huggingface.co/unsloth/Z-Image-Turbo-GGUF

The problem was reported by a user to the Discord: https://discord.com/channels/1020123559063990373/1149506274971631688/1534429470453399553

I asked Claude about the error, and here's what it told me: "Some Z-Image GGUF models fail to run with a TypeError during the transformer's forward pass, specifically when torch.where() is called on a tensor that's still in InvokeAI's quantized GGMLTensor form. Other Z-Image GGUFs (converted from BF16 source via stable-diffusion.cpp, uniform Q8_0 quantization) load and run without issue — so this appears specific to how certain GGUF files were quantized, not a general GGUF-loading problem.

"My guess (not confirmed) is that GGMLTensor only implements torch_dispatch handlers for the common ops needed by a typical forward pass (linear, matmul, etc.), and aten.where.self isn't among them. The failure seems to depend on whether the pad_token tensor specifically ends up quantized/wrapped as a GGMLTensor in a given file — my own conversions may leave small tensors like this unquantized, while some third-party GGUFs (possibly using a mixed/per-tensor quantization approach) quantize it too, exposing this code path."

[2026-08-05 10:30:15,498]::[InvokeAI]::ERROR --> Error while invoking session dd977b1e-552e-4e27-94be-b0806d1e7a91, invocation 95590a46-f898-4912-9d36-ba759fd5b108 (z_image_denoise): Multiple dispatch f
ailed for 'torch._ops.aten.where.self'; all __torch_dispatch__ handlers returned NotImplemented:

  - tensor subclass <class 'invokeai.backend.quantization.gguf.ggml_tensor.GGMLTensor'>

For more information, try re-running with TORCH_LOGS=not_implemented
[2026-08-05 10:30:15,499]::[InvokeAI]::ERROR --> Traceback (most recent call last):
  File "W:\AI\InvokeAI\.venv\Lib\site-packages\invokeai\app\services\session_processor\session_processor_default.py", line 148, in run_node
    output = invocation.invoke_internal(context=context, services=self._services)
             ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "W:\AI\InvokeAI\.venv\Lib\site-packages\invokeai\app\invocations\baseinvocation.py", line 248, in invoke_internal
    output = self.invoke(context)
             ^^^^^^^^^^^^^^^^^^^^
  File "W:\AI\InvokeAI\.venv\Lib\site-packages\torch\utils\_contextlib.py", line 124, in decorate_context
    return func(*args, **kwargs)
           ^^^^^^^^^^^^^^^^^^^^^
  File "W:\AI\InvokeAI\.venv\Lib\site-packages\invokeai\app\invocations\z_image_denoise.py", line 130, in invoke
    latents = self._run_diffusion(context)
              ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "W:\AI\InvokeAI\.venv\Lib\site-packages\invokeai\app\invocations\z_image_denoise.py", line 722, in _run_diffusion
    model_output = transformer(
                   ^^^^^^^^^^^^
  File "W:\AI\InvokeAI\.venv\Lib\site-packages\torch\nn\modules\module.py", line 1779, in _wrapped_call_impl
    return self._call_impl(*args, **kwargs)
           ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "W:\AI\InvokeAI\.venv\Lib\site-packages\torch\nn\modules\module.py", line 1790, in _call_impl
    return forward_call(*args, **kwargs)
           ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "W:\AI\InvokeAI\.venv\Lib\site-packages\diffusers\models\transformers\transformer_z_image.py", line 981, in forward
    x, x_freqs, x_mask, _, x_noise_tensor = self._prepare_sequence(
                                            ^^^^^^^^^^^^^^^^^^^^^^^
  File "W:\AI\InvokeAI\.venv\Lib\site-packages\diffusers\models\transformers\transformer_z_image.py", line 781, in _prepare_sequence
    feats_cat = torch.where(mask, pad_token, feats_cat)
                ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
TypeError: Multiple dispatch failed for 'torch._ops.aten.where.self'; all __torch_dispatch__ handlers returned NotImplemented:

  - tensor subclass <class 'invokeai.backend.quantization.gguf.ggml_tensor.GGMLTensor'>

For more information, try re-running with TORCH_LOGS=not_implemented

What you expected to happen

I expected the model to work, like many other GGUF quants of ZiT have.

How to reproduce the problem

  1. Install one of the models from https://huggingface.co/unsloth/Z-Image-Turbo-GGUF (I used Q8_0).
  2. Try to generate an image with it.
  3. Get the error.

Additional context

GGUF files I converted myself (using stable-diffusion-gguf-converter, BF16 source with uniform Q8_0 quantization) load and run correctly on the same InvokeAI install. Only some third-party GGUFs fail this way.

Discord username

shanedk

Metadata

Metadata

Assignees

No one assigned

    Labels

    bugSomething isn't working

    Type

    No type

    Projects

    No projects

    Milestone

    No milestone

    Relationships

    None yet

    Development

    No branches or pull requests

    Issue actions