Is there an existing issue for this problem?
Install method
Invoke's Launcher
Operating system
Windows
GPU vendor
Nvidia (CUDA)
GPU model
RTX 3060
GPU VRAM
12GB
Version number
6.14.0-rc1
Browser
No response
System Information
{
"version": "6.14.0-rc1",
"dependencies": {
"absl-py" : "2.5.0",
"accelerate" : "1.14.0",
"annotated-doc" : "0.0.5",
"annotated-types" : "0.8.0",
"anyio" : "4.14.2",
"argon2-cffi" : "25.1.0",
"argon2-cffi-bindings" : "25.1.0",
"arrow" : "1.4.0",
"asttokens" : "3.0.2",
"async-lru" : "2.3.0",
"attrs" : "26.1.0",
"babel" : "2.18.0",
"bcrypt" : "3.2.2",
"beautifulsoup4" : "4.15.0",
"bidict" : "0.23.1",
"bitsandbytes" : "0.50.0",
"blake3" : "1.0.9",
"bleach" : "6.4.0",
"certifi" : "2022.12.7",
"cffi" : "2.1.0",
"charset-normalizer" : "2.1.1",
"click" : "8.4.2",
"colorama" : "0.4.6",
"coloredlogs" : "15.0.1",
"comm" : "0.2.3",
"compel" : "2.4.0",
"contourpy" : "1.3.3",
"cryptography" : "50.0.0",
"CUDA" : "12.8",
"cycler" : "0.12.1",
"debugpy" : "1.8.21",
"decorator" : "5.3.1",
"defusedxml" : "0.7.1",
"Deprecated" : "1.3.1",
"diffusers" : "0.39.0",
"dnspython" : "2.8.0",
"dynamicprompts" : "0.31.0",
"ecdsa" : "0.19.2",
"einops" : "0.8.2",
"email-validator" : "2.3.0",
"executing" : "2.2.1",
"fastapi" : "0.118.3",
"fastapi-events" : "0.12.2",
"fastjsonschema" : "2.22.1",
"filelock" : "3.29.0",
"flatbuffers" : "25.12.19",
"fonttools" : "4.63.0",
"fqdn" : "1.5.1",
"fsspec" : "2026.4.0",
"gguf" : "0.19.0",
"h11" : "0.16.0",
"hf-xet" : "1.5.2",
"httpcore" : "1.0.9",
"httptools" : "0.8.0",
"httpx" : "0.28.1",
"huggingface_hub" : "1.26.0",
"humanfriendly" : "10.0",
"idna" : "3.4",
"ImageIO" : "2.37.4",
"imageio-ffmpeg" : "0.6.0",
"importlib_metadata" : "7.1.0",
"InvokeAI" : "6.14.0rc1",
"ipykernel" : "7.3.0",
"ipython" : "9.16.0",
"ipython_pygments_lexers" : "1.1.1",
"isoduration" : "20.11.0",
"jax" : "0.7.1",
"jaxlib" : "0.7.1",
"jedi" : "0.20.0",
"Jinja2" : "3.1.6",
"json5" : "0.15.0",
"jsonpointer" : "3.1.1",
"jsonschema" : "4.26.0",
"jsonschema-specifications": "2025.9.1",
"jupyter-events" : "0.12.1",
"jupyter-lsp" : "2.3.1",
"jupyter_client" : "8.9.1",
"jupyter_core" : "5.9.1",
"jupyter_server" : "2.20.0",
"jupyter_server_terminals" : "0.5.4",
"jupyterlab" : "4.1.6",
"jupyterlab_pygments" : "0.3.0",
"jupyterlab_server" : "2.24.0",
"kiwisolver" : "1.5.0",
"lark" : "1.3.1",
"markdown-it-py" : "4.2.0",
"MarkupSafe" : "3.0.3",
"matplotlib" : "3.11.1",
"matplotlib-inline" : "0.2.2",
"mdurl" : "0.1.2",
"mediapipe" : "0.10.14",
"mistune" : "3.3.4",
"ml_dtypes" : "0.5.4",
"mpmath" : "1.3.0",
"nbclient" : "0.11.0",
"nbconvert" : "7.17.1",
"nbformat" : "5.10.4",
"nest-asyncio2" : "1.7.2",
"networkx" : "3.6.1",
"notebook" : "7.1.3",
"notebook_shim" : "0.2.4",
"numpy" : "1.26.4",
"onnx" : "1.16.1",
"onnxruntime" : "1.19.2",
"opencv-contrib-python" : "4.11.0.86",
"opt_einsum" : "3.4.0",
"packaging" : "24.1",
"pandocfilters" : "1.5.1",
"parso" : "0.8.7",
"passlib" : "1.7.4",
"picklescan" : "1.0.5",
"pillow" : "12.2.0",
"platformdirs" : "4.11.0",
"prometheus_client" : "0.26.0",
"prompt_toolkit" : "3.0.53",
"protobuf" : "4.25.9",
"psutil" : "7.2.2",
"pure_eval" : "0.2.3",
"pyasn1" : "0.6.4",
"pycparser" : "3.0",
"pydantic" : "2.13.4",
"pydantic-settings" : "2.14.2",
"pydantic_core" : "2.46.4",
"Pygments" : "2.20.0",
"pyparsing" : "3.3.2",
"PyPatchMatch" : "1.0.2",
"pyreadline3" : "3.5.6",
"python-dateutil" : "2.9.0.post0",
"python-dotenv" : "1.2.2",
"python-engineio" : "4.13.4",
"python-jose" : "3.5.0",
"python-json-logger" : "4.1.0",
"python-multipart" : "0.0.32",
"python-socketio" : "5.16.3",
"PyWavelets" : "1.9.0",
"pywinpty" : "3.0.5",
"PyYAML" : "6.0.3",
"pyzmq" : "27.1.0",
"referencing" : "0.37.0",
"regex" : "2026.7.19",
"requests" : "2.28.1",
"rfc3339-validator" : "0.1.4",
"rfc3986-validator" : "0.1.1",
"rfc3987-syntax" : "1.1.0",
"rich" : "15.0.0",
"rpds-py" : "2026.6.3",
"rsa" : "4.9.1",
"safetensors" : "0.8.0",
"scipy" : "1.17.1",
"semver" : "3.0.4",
"Send2Trash" : "2.1.0",
"sentencepiece" : "0.2.0",
"setuptools" : "78.1.0",
"shellingham" : "1.5.4",
"simple-websocket" : "1.1.0",
"six" : "1.17.0",
"sounddevice" : "0.5.5",
"soupsieve" : "2.9.1",
"spandrel" : "0.4.2",
"stack-data" : "0.6.3",
"starlette" : "0.48.0",
"sympy" : "1.14.0",
"terminado" : "0.18.1",
"tinycss2" : "1.5.1",
"tokenizers" : "0.22.2",
"torch" : "2.11.0+cu128",
"torchsde" : "0.2.6",
"torchvision" : "0.26.0+cu128",
"tornado" : "6.5.7",
"tqdm" : "4.66.5",
"traitlets" : "5.16.0",
"trampoline" : "0.1.2",
"transformers" : "5.5.4",
"typer" : "0.27.0",
"typing-inspection" : "0.4.2",
"typing_extensions" : "4.15.0",
"tzdata" : "2026.3",
"uri-template" : "1.3.0",
"urllib3" : "1.26.13",
"uvicorn" : "0.52.0",
"watchfiles" : "1.2.0",
"wcwidth" : "0.8.2",
"webcolors" : "25.10.0",
"webencodings" : "0.5.1",
"websocket-client" : "1.9.0",
"websockets" : "17.0.1",
"wrapt" : "2.3.0",
"wsproto" : "1.3.2",
"zipp" : "3.19.2"
},
"config": {
"schema_version": "4.0.3",
"legacy_models_yaml_path": null,
"host": "127.0.0.1",
"port": 9090,
"allow_origins": [],
"allow_credentials": true,
"allow_methods": [""],
"allow_headers": [""],
"ssl_certfile": null,
"ssl_keyfile": null,
"base_url": null,
"forwarded_allow_ips": "127.0.0.1",
"log_tokenization": false,
"patchmatch": true,
"models_dir": "models",
"convert_cache_dir": "models\.convert_cache",
"download_cache_dir": "models\.download_cache",
"legacy_conf_dir": "configs",
"db_dir": "databases",
"outputs_dir": "W:\AI\InvokeAI\outputs",
"image_subfolder_strategy": "date",
"custom_nodes_dir": "nodes",
"style_presets_dir": "style_presets",
"workflow_thumbnails_dir": "workflow_thumbnails",
"log_handlers": ["console"],
"log_format": "color",
"log_level": "info",
"log_sql": false,
"log_level_network": "warning",
"use_memory_db": false,
"dev_reload": false,
"profile_graphs": false,
"profile_prefix": null,
"profiles_dir": "profiles",
"max_cache_ram_gb": null,
"max_cache_vram_gb": null,
"log_memory_usage": false,
"model_cache_keep_alive_min": 0,
"device_working_mem_gb": 0.9,
"enable_partial_loading": true,
"keep_ram_copy_of_weights": true,
"ram": null,
"vram": null,
"lazy_offload": true,
"pytorch_cuda_alloc_conf": "backend:cudaMallocAsync",
"device": "cuda",
"generation_devices": "auto",
"offload_text_encoders_to_idle_gpus": true,
"precision": "bfloat16",
"sequential_guidance": false,
"attention_type": "torch-sdp",
"attention_slice_size": "auto",
"force_tiled_decode": true,
"pil_compress_level": 1,
"max_queue_size": 10000,
"session_queue_mode": "round_robin",
"clear_queue_on_startup": false,
"max_queue_history": 0,
"allow_nodes": null,
"deny_nodes": null,
"node_cache_size": 512,
"hashing_algorithm": "random",
"remote_api_tokens": [
{"url_regex": "civitai.com", "token": ""},
{"url_regex": "huggingface.co", "token": ""},
{"url_regex": "civitai.red", "token": "**********"}
],
"scan_models_on_startup": false,
"unsafe_disable_picklescan": false,
"allow_unknown_models": true,
"multiuser": false,
"strict_password_checking": false,
"external_alibabacloud_api_key": null,
"external_alibabacloud_base_url": null,
"external_gemini_api_key": null,
"external_openai_api_key": null,
"external_gemini_base_url": null,
"external_openai_base_url": null,
"external_seedream_api_key": null,
"external_seedream_base_url": null
},
"set_config_fields": [
"attention_type", "max_queue_history", "precision", "hashing_algorithm",
"pytorch_cuda_alloc_conf", "image_subfolder_strategy", "device", "force_tiled_decode",
"outputs_dir", "legacy_models_yaml_path", "device_working_mem_gb", "remote_api_tokens"
]
}
What happened
Some quantized GGUF models fail, such as this one: https://huggingface.co/unsloth/Z-Image-Turbo-GGUF
The problem was reported by a user to the Discord: https://discord.com/channels/1020123559063990373/1149506274971631688/1534429470453399553
I asked Claude about the error, and here's what it told me: "Some Z-Image GGUF models fail to run with a TypeError during the transformer's forward pass, specifically when torch.where() is called on a tensor that's still in InvokeAI's quantized GGMLTensor form. Other Z-Image GGUFs (converted from BF16 source via stable-diffusion.cpp, uniform Q8_0 quantization) load and run without issue — so this appears specific to how certain GGUF files were quantized, not a general GGUF-loading problem.
"My guess (not confirmed) is that GGMLTensor only implements torch_dispatch handlers for the common ops needed by a typical forward pass (linear, matmul, etc.), and aten.where.self isn't among them. The failure seems to depend on whether the pad_token tensor specifically ends up quantized/wrapped as a GGMLTensor in a given file — my own conversions may leave small tensors like this unquantized, while some third-party GGUFs (possibly using a mixed/per-tensor quantization approach) quantize it too, exposing this code path."
[2026-08-05 10:30:15,498]::[InvokeAI]::ERROR --> Error while invoking session dd977b1e-552e-4e27-94be-b0806d1e7a91, invocation 95590a46-f898-4912-9d36-ba759fd5b108 (z_image_denoise): Multiple dispatch f
ailed for 'torch._ops.aten.where.self'; all __torch_dispatch__ handlers returned NotImplemented:
- tensor subclass <class 'invokeai.backend.quantization.gguf.ggml_tensor.GGMLTensor'>
For more information, try re-running with TORCH_LOGS=not_implemented
[2026-08-05 10:30:15,499]::[InvokeAI]::ERROR --> Traceback (most recent call last):
File "W:\AI\InvokeAI\.venv\Lib\site-packages\invokeai\app\services\session_processor\session_processor_default.py", line 148, in run_node
output = invocation.invoke_internal(context=context, services=self._services)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "W:\AI\InvokeAI\.venv\Lib\site-packages\invokeai\app\invocations\baseinvocation.py", line 248, in invoke_internal
output = self.invoke(context)
^^^^^^^^^^^^^^^^^^^^
File "W:\AI\InvokeAI\.venv\Lib\site-packages\torch\utils\_contextlib.py", line 124, in decorate_context
return func(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^
File "W:\AI\InvokeAI\.venv\Lib\site-packages\invokeai\app\invocations\z_image_denoise.py", line 130, in invoke
latents = self._run_diffusion(context)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "W:\AI\InvokeAI\.venv\Lib\site-packages\invokeai\app\invocations\z_image_denoise.py", line 722, in _run_diffusion
model_output = transformer(
^^^^^^^^^^^^
File "W:\AI\InvokeAI\.venv\Lib\site-packages\torch\nn\modules\module.py", line 1779, in _wrapped_call_impl
return self._call_impl(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "W:\AI\InvokeAI\.venv\Lib\site-packages\torch\nn\modules\module.py", line 1790, in _call_impl
return forward_call(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "W:\AI\InvokeAI\.venv\Lib\site-packages\diffusers\models\transformers\transformer_z_image.py", line 981, in forward
x, x_freqs, x_mask, _, x_noise_tensor = self._prepare_sequence(
^^^^^^^^^^^^^^^^^^^^^^^
File "W:\AI\InvokeAI\.venv\Lib\site-packages\diffusers\models\transformers\transformer_z_image.py", line 781, in _prepare_sequence
feats_cat = torch.where(mask, pad_token, feats_cat)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
TypeError: Multiple dispatch failed for 'torch._ops.aten.where.self'; all __torch_dispatch__ handlers returned NotImplemented:
- tensor subclass <class 'invokeai.backend.quantization.gguf.ggml_tensor.GGMLTensor'>
For more information, try re-running with TORCH_LOGS=not_implemented
What you expected to happen
I expected the model to work, like many other GGUF quants of ZiT have.
How to reproduce the problem
- Install one of the models from https://huggingface.co/unsloth/Z-Image-Turbo-GGUF (I used Q8_0).
- Try to generate an image with it.
- Get the error.
Additional context
GGUF files I converted myself (using stable-diffusion-gguf-converter, BF16 source with uniform Q8_0 quantization) load and run correctly on the same InvokeAI install. Only some third-party GGUFs fail this way.
Discord username
shanedk
Is there an existing issue for this problem?
Install method
Invoke's Launcher
Operating system
Windows
GPU vendor
Nvidia (CUDA)
GPU model
RTX 3060
GPU VRAM
12GB
Version number
6.14.0-rc1
Browser
No response
System Information
{
"version": "6.14.0-rc1",
"dependencies": {
"absl-py" : "2.5.0",
"accelerate" : "1.14.0",
"annotated-doc" : "0.0.5",
"annotated-types" : "0.8.0",
"anyio" : "4.14.2",
"argon2-cffi" : "25.1.0",
"argon2-cffi-bindings" : "25.1.0",
"arrow" : "1.4.0",
"asttokens" : "3.0.2",
"async-lru" : "2.3.0",
"attrs" : "26.1.0",
"babel" : "2.18.0",
"bcrypt" : "3.2.2",
"beautifulsoup4" : "4.15.0",
"bidict" : "0.23.1",
"bitsandbytes" : "0.50.0",
"blake3" : "1.0.9",
"bleach" : "6.4.0",
"certifi" : "2022.12.7",
"cffi" : "2.1.0",
"charset-normalizer" : "2.1.1",
"click" : "8.4.2",
"colorama" : "0.4.6",
"coloredlogs" : "15.0.1",
"comm" : "0.2.3",
"compel" : "2.4.0",
"contourpy" : "1.3.3",
"cryptography" : "50.0.0",
"CUDA" : "12.8",
"cycler" : "0.12.1",
"debugpy" : "1.8.21",
"decorator" : "5.3.1",
"defusedxml" : "0.7.1",
"Deprecated" : "1.3.1",
"diffusers" : "0.39.0",
"dnspython" : "2.8.0",
"dynamicprompts" : "0.31.0",
"ecdsa" : "0.19.2",
"einops" : "0.8.2",
"email-validator" : "2.3.0",
"executing" : "2.2.1",
"fastapi" : "0.118.3",
"fastapi-events" : "0.12.2",
"fastjsonschema" : "2.22.1",
"filelock" : "3.29.0",
"flatbuffers" : "25.12.19",
"fonttools" : "4.63.0",
"fqdn" : "1.5.1",
"fsspec" : "2026.4.0",
"gguf" : "0.19.0",
"h11" : "0.16.0",
"hf-xet" : "1.5.2",
"httpcore" : "1.0.9",
"httptools" : "0.8.0",
"httpx" : "0.28.1",
"huggingface_hub" : "1.26.0",
"humanfriendly" : "10.0",
"idna" : "3.4",
"ImageIO" : "2.37.4",
"imageio-ffmpeg" : "0.6.0",
"importlib_metadata" : "7.1.0",
"InvokeAI" : "6.14.0rc1",
"ipykernel" : "7.3.0",
"ipython" : "9.16.0",
"ipython_pygments_lexers" : "1.1.1",
"isoduration" : "20.11.0",
"jax" : "0.7.1",
"jaxlib" : "0.7.1",
"jedi" : "0.20.0",
"Jinja2" : "3.1.6",
"json5" : "0.15.0",
"jsonpointer" : "3.1.1",
"jsonschema" : "4.26.0",
"jsonschema-specifications": "2025.9.1",
"jupyter-events" : "0.12.1",
"jupyter-lsp" : "2.3.1",
"jupyter_client" : "8.9.1",
"jupyter_core" : "5.9.1",
"jupyter_server" : "2.20.0",
"jupyter_server_terminals" : "0.5.4",
"jupyterlab" : "4.1.6",
"jupyterlab_pygments" : "0.3.0",
"jupyterlab_server" : "2.24.0",
"kiwisolver" : "1.5.0",
"lark" : "1.3.1",
"markdown-it-py" : "4.2.0",
"MarkupSafe" : "3.0.3",
"matplotlib" : "3.11.1",
"matplotlib-inline" : "0.2.2",
"mdurl" : "0.1.2",
"mediapipe" : "0.10.14",
"mistune" : "3.3.4",
"ml_dtypes" : "0.5.4",
"mpmath" : "1.3.0",
"nbclient" : "0.11.0",
"nbconvert" : "7.17.1",
"nbformat" : "5.10.4",
"nest-asyncio2" : "1.7.2",
"networkx" : "3.6.1",
"notebook" : "7.1.3",
"notebook_shim" : "0.2.4",
"numpy" : "1.26.4",
"onnx" : "1.16.1",
"onnxruntime" : "1.19.2",
"opencv-contrib-python" : "4.11.0.86",
"opt_einsum" : "3.4.0",
"packaging" : "24.1",
"pandocfilters" : "1.5.1",
"parso" : "0.8.7",
"passlib" : "1.7.4",
"picklescan" : "1.0.5",
"pillow" : "12.2.0",
"platformdirs" : "4.11.0",
"prometheus_client" : "0.26.0",
"prompt_toolkit" : "3.0.53",
"protobuf" : "4.25.9",
"psutil" : "7.2.2",
"pure_eval" : "0.2.3",
"pyasn1" : "0.6.4",
"pycparser" : "3.0",
"pydantic" : "2.13.4",
"pydantic-settings" : "2.14.2",
"pydantic_core" : "2.46.4",
"Pygments" : "2.20.0",
"pyparsing" : "3.3.2",
"PyPatchMatch" : "1.0.2",
"pyreadline3" : "3.5.6",
"python-dateutil" : "2.9.0.post0",
"python-dotenv" : "1.2.2",
"python-engineio" : "4.13.4",
"python-jose" : "3.5.0",
"python-json-logger" : "4.1.0",
"python-multipart" : "0.0.32",
"python-socketio" : "5.16.3",
"PyWavelets" : "1.9.0",
"pywinpty" : "3.0.5",
"PyYAML" : "6.0.3",
"pyzmq" : "27.1.0",
"referencing" : "0.37.0",
"regex" : "2026.7.19",
"requests" : "2.28.1",
"rfc3339-validator" : "0.1.4",
"rfc3986-validator" : "0.1.1",
"rfc3987-syntax" : "1.1.0",
"rich" : "15.0.0",
"rpds-py" : "2026.6.3",
"rsa" : "4.9.1",
"safetensors" : "0.8.0",
"scipy" : "1.17.1",
"semver" : "3.0.4",
"Send2Trash" : "2.1.0",
"sentencepiece" : "0.2.0",
"setuptools" : "78.1.0",
"shellingham" : "1.5.4",
"simple-websocket" : "1.1.0",
"six" : "1.17.0",
"sounddevice" : "0.5.5",
"soupsieve" : "2.9.1",
"spandrel" : "0.4.2",
"stack-data" : "0.6.3",
"starlette" : "0.48.0",
"sympy" : "1.14.0",
"terminado" : "0.18.1",
"tinycss2" : "1.5.1",
"tokenizers" : "0.22.2",
"torch" : "2.11.0+cu128",
"torchsde" : "0.2.6",
"torchvision" : "0.26.0+cu128",
"tornado" : "6.5.7",
"tqdm" : "4.66.5",
"traitlets" : "5.16.0",
"trampoline" : "0.1.2",
"transformers" : "5.5.4",
"typer" : "0.27.0",
"typing-inspection" : "0.4.2",
"typing_extensions" : "4.15.0",
"tzdata" : "2026.3",
"uri-template" : "1.3.0",
"urllib3" : "1.26.13",
"uvicorn" : "0.52.0",
"watchfiles" : "1.2.0",
"wcwidth" : "0.8.2",
"webcolors" : "25.10.0",
"webencodings" : "0.5.1",
"websocket-client" : "1.9.0",
"websockets" : "17.0.1",
"wrapt" : "2.3.0",
"wsproto" : "1.3.2",
"zipp" : "3.19.2"
},
"config": {
"schema_version": "4.0.3",
"legacy_models_yaml_path": null,
"host": "127.0.0.1",
"port": 9090,
"allow_origins": [],
"allow_credentials": true,
"allow_methods": [""],
"allow_headers": [""],
"ssl_certfile": null,
"ssl_keyfile": null,
"base_url": null,
"forwarded_allow_ips": "127.0.0.1",
"log_tokenization": false,
"patchmatch": true,
"models_dir": "models",
"convert_cache_dir": "models\.convert_cache",
"download_cache_dir": "models\.download_cache",
"legacy_conf_dir": "configs",
"db_dir": "databases",
"outputs_dir": "W:\AI\InvokeAI\outputs",
"image_subfolder_strategy": "date",
"custom_nodes_dir": "nodes",
"style_presets_dir": "style_presets",
"workflow_thumbnails_dir": "workflow_thumbnails",
"log_handlers": ["console"],
"log_format": "color",
"log_level": "info",
"log_sql": false,
"log_level_network": "warning",
"use_memory_db": false,
"dev_reload": false,
"profile_graphs": false,
"profile_prefix": null,
"profiles_dir": "profiles",
"max_cache_ram_gb": null,
"max_cache_vram_gb": null,
"log_memory_usage": false,
"model_cache_keep_alive_min": 0,
"device_working_mem_gb": 0.9,
"enable_partial_loading": true,
"keep_ram_copy_of_weights": true,
"ram": null,
"vram": null,
"lazy_offload": true,
"pytorch_cuda_alloc_conf": "backend:cudaMallocAsync",
"device": "cuda",
"generation_devices": "auto",
"offload_text_encoders_to_idle_gpus": true,
"precision": "bfloat16",
"sequential_guidance": false,
"attention_type": "torch-sdp",
"attention_slice_size": "auto",
"force_tiled_decode": true,
"pil_compress_level": 1,
"max_queue_size": 10000,
"session_queue_mode": "round_robin",
"clear_queue_on_startup": false,
"max_queue_history": 0,
"allow_nodes": null,
"deny_nodes": null,
"node_cache_size": 512,
"hashing_algorithm": "random",
"remote_api_tokens": [
{"url_regex": "civitai.com", "token": ""},
{"url_regex": "huggingface.co", "token": ""},
{"url_regex": "civitai.red", "token": "**********"}
],
"scan_models_on_startup": false,
"unsafe_disable_picklescan": false,
"allow_unknown_models": true,
"multiuser": false,
"strict_password_checking": false,
"external_alibabacloud_api_key": null,
"external_alibabacloud_base_url": null,
"external_gemini_api_key": null,
"external_openai_api_key": null,
"external_gemini_base_url": null,
"external_openai_base_url": null,
"external_seedream_api_key": null,
"external_seedream_base_url": null
},
"set_config_fields": [
"attention_type", "max_queue_history", "precision", "hashing_algorithm",
"pytorch_cuda_alloc_conf", "image_subfolder_strategy", "device", "force_tiled_decode",
"outputs_dir", "legacy_models_yaml_path", "device_working_mem_gb", "remote_api_tokens"
]
}
What happened
Some quantized GGUF models fail, such as this one: https://huggingface.co/unsloth/Z-Image-Turbo-GGUF
The problem was reported by a user to the Discord: https://discord.com/channels/1020123559063990373/1149506274971631688/1534429470453399553
I asked Claude about the error, and here's what it told me: "Some Z-Image GGUF models fail to run with a TypeError during the transformer's forward pass, specifically when torch.where() is called on a tensor that's still in InvokeAI's quantized GGMLTensor form. Other Z-Image GGUFs (converted from BF16 source via stable-diffusion.cpp, uniform Q8_0 quantization) load and run without issue — so this appears specific to how certain GGUF files were quantized, not a general GGUF-loading problem.
"My guess (not confirmed) is that GGMLTensor only implements torch_dispatch handlers for the common ops needed by a typical forward pass (linear, matmul, etc.), and aten.where.self isn't among them. The failure seems to depend on whether the pad_token tensor specifically ends up quantized/wrapped as a GGMLTensor in a given file — my own conversions may leave small tensors like this unquantized, while some third-party GGUFs (possibly using a mixed/per-tensor quantization approach) quantize it too, exposing this code path."
What you expected to happen
I expected the model to work, like many other GGUF quants of ZiT have.
How to reproduce the problem
Additional context
GGUF files I converted myself (using stable-diffusion-gguf-converter, BF16 source with uniform Q8_0 quantization) load and run correctly on the same InvokeAI install. Only some third-party GGUFs fail this way.
Discord username
shanedk