[bug]: Error generating simple image on NVIDIA using FLUX.2 Klein 4B
I maintainer di solito rispondono entro 8 giorni
@Pfannkuchensack ci sta già lavorando.
Dal 20/9/2026.
Valutazione
Questa issue non è ancora stata valutata.
Descrizione
Is there an existing issue for this problem?
- I have searched the existing issues
Install method
Invoke's Launcher
Operating system
Linux
GPU vendor
Nvidia (CUDA)
GPU model
RTX 2080 Ti
GPU VRAM
11GB
Version number
v6.14.1
Browser
Firefox 156.0
System Information
{
"version": "6.14.1",
"dependencies": {
"absl-py" : "2.4.0",
"accelerate" : "1.14.0",
"annotated-doc" : "0.0.4",
"annotated-types" : "0.7.0",
"anyio" : "4.14.1",
"attrs" : "26.1.0",
"bcrypt" : "3.2.2",
"bidict" : "0.23.1",
"bitsandbytes" : "0.49.2",
"blake3" : "1.0.9",
"certifi" : "2026.6.17",
"cffi" : "2.0.0",
"charset-normalizer" : "3.4.7",
"click" : "8.4.2",
"coloredlogs" : "15.0.1",
"compel" : "2.4.0",
"contourpy" : "1.3.3",
"cryptography" : "49.0.0",
"CUDA" : "12.8",
"cycler" : "0.12.1",
"Deprecated" : "1.3.1",
"diffusers" : "0.40.0",
"dnspython" : "2.8.0",
"dynamicprompts" : "0.31.0",
"ecdsa" : "0.19.2",
"einops" : "0.8.2",
"email-validator" : "2.3.0",
"fastapi" : "0.141.1",
"fastapi-events" : "0.12.2",
"filelock" : "3.29.4",
"flatbuffers" : "25.12.19",
"fonttools" : "4.63.0",
"fsspec" : "2026.6.0",
"gguf" : "0.19.0",
"h11" : "0.16.0",
"hf-xet" : "1.6.0",
"httpcore" : "1.0.9",
"httptools" : "0.8.0",
"httpx" : "0.28.1",
"huggingface_hub" : "1.28.0",
"humanfriendly" : "10.0",
"idna" : "3.18",
"ImageIO" : "2.37.4",
"imageio-ffmpeg" : "0.6.0",
"importlib_metadata" : "9.0.0",
"InvokeAI" : "6.14.1",
"jax" : "0.7.1",
"jaxlib" : "0.7.1",
"Jinja2" : "3.1.6",
"jsonschema" : "4.26.0",
"jsonschema-specifications": "2025.9.1",
"kiwisolver" : "1.5.0",
"markdown-it-py" : "4.2.0",
"MarkupSafe" : "3.0.3",
"matplotlib" : "3.11.0",
"mdurl" : "0.1.2",
"mediapipe" : "0.10.14",
"mistral_common" : "1.11.6",
"ml_dtypes" : "0.5.4",
"mpmath" : "1.3.0",
"networkx" : "3.6.1",
"numpy" : "1.26.4",
"nvidia-cublas-cu12" : "12.8.3.14",
"nvidia-cuda-cupti-cu12" : "12.8.57",
"nvidia-cuda-nvrtc-cu12" : "12.8.61",
"nvidia-cuda-runtime-cu12" : "12.8.57",
"nvidia-cudnn-cu12" : "9.7.1.26",
"nvidia-cufft-cu12" : "11.3.3.41",
"nvidia-cufile-cu12" : "1.13.0.11",
"nvidia-curand-cu12" : "10.3.9.55",
"nvidia-cusolver-cu12" : "11.7.2.55",
"nvidia-cusparse-cu12" : "12.5.7.53",
"nvidia-cusparselt-cu12" : "0.6.3",
"nvidia-nccl-cu12" : "2.26.2",
"nvidia-nvjitlink-cu12" : "12.8.61",
"nvidia-nvtx-cu12" : "12.8.55",
"onnx" : "1.16.1",
"onnxruntime" : "1.19.2",
"opencv-contrib-python" : "4.11.0.86",
"opt_einsum" : "3.4.0",
"packaging" : "26.2",
"passlib" : "1.7.4",
"picklescan" : "1.0.4",
"pillow" : "12.2.0",
"prompt_toolkit" : "3.0.52",
"protobuf" : "4.25.9",
"psutil" : "7.2.2",
"pyasn1" : "0.6.3",
"pycountry" : "26.2.16",
"pycparser" : "3.0",
"pydantic" : "2.13.4",
"pydantic-extra-types" : "2.11.1",
"pydantic-settings" : "2.14.2",
"pydantic_core" : "2.46.4",
"Pygments" : "2.20.0",
"pyparsing" : "3.3.2",
"PyPatchMatch" : "1.0.2",
"python-dateutil" : "2.9.0.post0",
"python-dotenv" : "1.2.2",
"python-engineio" : "4.13.3",
"python-jose" : "3.5.0",
"python-multipart" : "0.0.32",
"python-socketio" : "5.16.3",
"PyWavelets" : "1.9.0",
"PyYAML" : "6.0.3",
"referencing" : "0.37.0",
"regex" : "2026.5.9",
"requests" : "2.34.2",
"rich" : "15.0.0",
"rpds-py" : "2026.5.1",
"rsa" : "4.9.1",
"safetensors" : "0.8.0",
"scipy" : "1.17.1",
"semver" : "3.0.4",
"sentencepiece" : "0.2.0",
"setuptools" : "82.0.1",
"shellingham" : "1.5.4",
"simple-websocket" : "1.1.0",
"six" : "1.17.0",
"sounddevice" : "0.5.5",
"spandrel" : "0.4.2",
"starlette" : "0.48.0",
"sympy" : "1.14.0",
"tiktoken" : "0.13.0",
"tokenizers" : "0.22.2",
"torch" : "2.7.1+cu128",
"torchsde" : "0.2.6",
"torchvision" : "0.22.1+cu128",
"tqdm" : "4.68.3",
"trampoline" : "0.1.2",
"transformers" : "5.5.4",
"triton" : "3.3.1",
"typer" : "0.25.1",
"typing-inspection" : "0.4.2",
"typing_extensions" : "4.15.0",
"urllib3" : "2.7.0",
"uvicorn" : "0.49.0",
"uvloop" : "0.22.1",
"watchfiles" : "1.2.0",
"wcwidth" : "0.8.1",
"websockets" : "16.0",
"wrapt" : "2.2.2",
"wsproto" : "1.3.2",
"xformers" : "0.0.31.post1",
"zipp" : "4.1.0"
},
"config": {
"schema_version": "4.0.3",
"legacy_models_yaml_path": null,
"host": "127.0.0.1",
"port": 9090,
"allow_origins": [],
"allow_credentials": true,
"allow_methods": [""],
"allow_headers": [""],
"ssl_certfile": null,
"ssl_keyfile": null,
"base_url": null,
"forwarded_allow_ips": "127.0.0.1",
"http_compression_level": 9,
"log_tokenization": false,
"patchmatch": true,
"models_dir": "models",
"convert_cache_dir": "models/.convert_cache",
"download_cache_dir": "models/.download_cache",
"legacy_conf_dir": "configs",
"db_dir": "databases",
"outputs_dir": "outputs",
"image_subfolder_strategy": "type",
"custom_nodes_dir": "nodes",
"style_presets_dir": "style_presets",
"workflow_thumbnails_dir": "workflow_thumbnails",
"log_handlers": ["console"],
"log_format": "color",
"log_level": "info",
"log_sql": false,
"log_level_network": "warning",
"use_memory_db": false,
"dev_reload": false,
"profile_graphs": false,
"profile_prefix": null,
"profiles_dir": "profiles",
"max_cache_ram_gb": null,
"max_cache_vram_gb": null,
"log_memory_usage": false,
"model_cache_keep_alive_min": 0,
"device_working_mem_gb": 3,
"enable_partial_loading": true,
"keep_ram_copy_of_weights": true,
"ram": null,
"vram": null,
"lazy_offload": true,
"pytorch_cuda_alloc_conf": null,
"device": "auto",
"generation_devices": "auto",
"offload_text_encoders_to_idle_gpus": true,
"precision": "auto",
"sequential_guidance": false,
"wan_memory_optimization": false,
"pid_memory_optimization": false,
"attention_type": "auto",
"attention_slice_size": "auto",
"force_tiled_decode": false,
"pil_compress_level": 1,
"max_queue_size": 10000,
"session_queue_mode": "round_robin",
"clear_queue_on_startup": false,
"max_queue_history": null,
"allow_nodes": null,
"deny_nodes": null,
"node_cache_size": 512,
"hashing_algorithm": "blake3_single",
"remote_api_tokens": null,
"scan_models_on_startup": false,
"allow_private_download_urls": false,
"download_proxy": null,
"unsafe_disable_picklescan": false,
"allow_unknown_models": true,
"multiuser": false,
"strict_password_checking": false,
"external_alibabacloud_api_key": null,
"external_alibabacloud_base_url": null,
"external_gemini_api_key": null,
"external_openai_api_key": null,
"external_gemini_base_url": null,
"external_openai_base_url": null,
"external_seedream_api_key": null,
"external_seedream_base_url": null
},
"set_config_fields": ["legacy_models_yaml_path", "image_subfolder_strategy"]
}
What happened
Opened the canvas, selected the model FLUX.2 Klein 4B, and wrote a simple test prompt ("Vase on a table."). It got to 25% completion before it failed fatally.
What you expected to happen
I expected to generate the image without error.
How to reproduce the problem
After a fresh install of the latest release of InvokeAI on my Bazitte 44 (NVIDIA Edition), I launched it and downloaded the FLUX.2 Klein starting pack only.
Opened the canvas, selected the model FLUX.2 Klein 4B, and wrote a simple test prompt. It got to 25% completion before it failed fatally.
Additional context
Here are the logs generated from start to error:
Started Invoke process with PID 25698
[2026-09-19 14:45:46,551]::[InvokeAI]::INFO --> Using torch device: NVIDIA GeForce RTX 2080 Ti
[2026-09-19 14:45:46,557]::[InvokeAI]::INFO --> cuDNN version: 90701
`Siglip2ImageProcessorFast` is deprecated. The `Fast` suffix for image processors has been removed; use `Siglip2ImageProcessor` instead.
>> patchmatch.patch_match: INFO - Compiling and loading c extensions from "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/patchmatch".
>> patchmatch.patch_match: ERROR - patchmatch failed to load or compile (Command 'make clean && make' returned non-zero exit status 2.).
>> patchmatch.patch_match: INFO - Refer to https://invoke-ai.github.io/InvokeAI/installation/060_INSTALL_PATCHMATCH/ for installation instructions.
[2026-09-19 14:45:49,697]::[InvokeAI]::INFO --> Patchmatch not loaded (nonfatal)
[2026-09-19 14:45:51,156]::[InvokeAI]::INFO --> InvokeAI version 6.14.1
[2026-09-19 14:45:51,156]::[InvokeAI]::INFO --> Root directory = /MyApps/InstalledApps
[2026-09-19 14:45:51,157]::[InvokeAI]::INFO --> Initializing database at /MyApps/InstalledApps/databases/invokeai.db
[2026-09-19 14:45:51,179]::[InvokeAI]::INFO --> JWT secret loaded from database
[2026-09-19 14:45:51,181]::[ModelManagerService]::INFO --> [MODEL CACHE] Calculated model RAM cache size: 7757.50 MB. Heuristics applied: [1, 2].
[2026-09-19 14:45:51,181]::[ModelManagerService]::INFO --> Model cache global RAM budget: 7.58 GB across 1 device cache(s).
[2026-09-19 14:45:51,185]::[ModelInstallService]::INFO --> Restoring incomplete installs
[2026-09-19 14:45:51,186]::[ModelInstallService]::INFO --> Finished restoring incomplete installs
[2026-09-19 14:45:51,250]::[InvokeAI]::INFO --> Invoke running on http://127.0.0.1:9090 (Press CTRL+C to quit)
[2026-09-19 14:47:35,574]::[InvokeAI]::INFO --> Executing queue item 5, session 2bfc8573-7cc7-41a4-8619-914faf6370ab on cuda:0
[2026-09-19 14:47:51,317]::[Qwen3EncoderGGUFLoader]::INFO --> Detected llama.cpp GGUF format, converting keys to PyTorch format
[2026-09-19 14:47:51,318]::[Qwen3EncoderGGUFLoader]::INFO --> Qwen3 GGUF Encoder config detected: layers=36, hidden=2560, heads=32, kv_heads=8, intermediate=9728, head_dim=128
[2026-09-19 14:47:53,107]::[Qwen3EncoderGGUFLoader]::INFO --> Dequantized embed_tokens weight for embedding lookups
[2026-09-19 14:47:53,107]::[Qwen3EncoderGGUFLoader]::INFO --> Tied lm_head.weight to embed_tokens.weight (GGUF tied embeddings)
[2026-09-19 14:47:53,905]::[ModelManagerService]::INFO --> [MODEL CACHE] Loaded model '165da151-1eb2-4457-821b-ea83785b1858:text_encoder' (Qwen3ForCausalLM) onto cuda device #0 in 0.75s. Total model size: 4326.88MB, VRAM: 4326.88MB (100.0%)
[2026-09-19 14:47:54,435]::[ModelManagerService]::INFO --> [MODEL CACHE] Loaded model '165da151-1eb2-4457-821b-ea83785b1858:tokenizer' (Qwen2Tokenizer) onto cuda device #0 in 0.00s. Total model size: 0.00MB, VRAM: 0.00MB (0.0%)
[2026-09-19 14:48:03,076]::[ModelManagerService]::INFO --> [MODEL CACHE] Loaded model '271b3dc7-6a26-42fa-a2ca-c08f2ddb9c88:vae' (AutoencoderKLFlux2) onto cuda device #0 in 0.25s. Total model size: 160.31MB, VRAM: 160.31MB (100.0%)
[2026-09-19 14:48:03,128]::[invokeai.backend.util.attention]::INFO --> SDPA materializes its attention score matrix on cuda for head_dim=128 (this torch build), so working-memory estimates reserve an extra ~8.1 GiB for it.
[2026-09-19 14:48:04,153]::[InvokeAI]::WARNING --> Loading 0.0146484375 MB into VRAM, but only -2477.013671875 MB were requested. This is the minimum set of weights in VRAM required to run the model.
[2026-09-19 14:48:04,237]::[ModelManagerService]::INFO --> [MODEL CACHE] Loaded model '993aa3a1-6dc5-4e56-b57e-7464c663ced1:transformer' (Flux2Transformer2DModel) onto cuda device #0 in 0.11s. Total model size: 7411.51MB, VRAM: 0.01MB (0.0%)
Denoising (#0): 0%| | 0/4 [00:00<?, ?it/s][2026-09-19 14:48:14,233]::[InvokeAI]::ERROR --> Error while invoking session 2bfc8573-7cc7-41a4-8619-914faf6370ab, invocation db3c60db-68c3-4d90-acc2-20e1b475dca6 (flux2_denoise): CUDA error: CUBLAS_STATUS_EXECUTION_FAILED when calling `cublasSgemmStridedBatched( handle, opa, opb, m, n, k, &alpha, a, lda, stridea, b, ldb, strideb, &beta, c, ldc, stridec, num_batches)`
[2026-09-19 14:48:14,233]::[InvokeAI]::ERROR --> Traceback (most recent call last):
File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/invokeai/app/services/session_processor/session_processor_default.py", line 278, in run_node
output = invocation.invoke_internal(context=context, services=self._services)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/invokeai/app/invocations/baseinvocation.py", line 248, in invoke_internal
output = self.invoke(context)
^^^^^^^^^^^^^^^^^^^^
File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/torch/utils/_contextlib.py", line 116, in decorate_context
return func(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^
File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/invokeai/app/invocations/flux2_denoise.py", line 257, in invoke
latents = self._run_diffusion(context)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/invokeai/app/invocations/flux2_denoise.py", line 571, in _run_diffusion
x = denoise(
^^^^^^^^
File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/invokeai/backend/flux2/denoise.py", line 148, in denoise
output = model(
^^^^^^
File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
return self._call_impl(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
return forward_call(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/diffusers/utils/peft_utils.py", line 323, in wrapper
result = forward_fn(self, *args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/diffusers/models/transformers/transformer_flux2.py", line 1408, in forward
hidden_states = block(
^^^^^^
File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
return self._call_impl(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
return forward_call(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/diffusers/models/transformers/transformer_flux2.py", line 859, in forward
attn_output = self.attn(
^^^^^^^^^^
File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
return self._call_impl(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
return forward_call(*args, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/diffusers/models/transformers/transformer_flux2.py", line 804, in forward
return self.processor(self, hidden_states, attention_mask, image_rotary_emb, **kwargs)
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/diffusers/models/transformers/transformer_flux2.py", line 612, in __call__
hidden_states = dispatch_attention_fn(
^^^^^^^^^^^^^^^^^^^^^^
File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/diffusers/models/attention_dispatch.py", line 439, in dispatch_attention_fn
return backend_fn(**kwargs)
^^^^^^^^^^^^^^^^^^^^
File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/diffusers/models/attention_dispatch.py", line 3707, in _native_attention
out = torch.nn.functional.scaled_dot_product_attention(
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
RuntimeError: CUDA error: CUBLAS_STATUS_EXECUTION_FAILED when calling `cublasSgemmStridedBatched( handle, opa, opb, m, n, k, &alpha, a, lda, stridea, b, ldb, strideb, &beta, c, ldc, stridec, num_batches)`
Denoising (#0): 0%| | 0/4 [00:09<?, ?it/s]
[2026-09-19 14:48:14,252]::[InvokeAI]::INFO --> Graph stats: 2bfc8573-7cc7-41a4-8619-914faf6370ab
Node Calls Seconds VRAM Change
integer 1 0.006s +0.000G
flux2_klein_model_loader 1 0.012s +0.000G
string 1 0.005s +0.000G
core_metadata 1 0.001s +0.000G
flux2_klein_text_encoder 1 27.155s +4.288G
collect 1 0.001s +0.000G
flux2_denoise 1 11.463s -3.907G
TOTAL GRAPH EXECUTION TIME: 38.643s
TOTAL GRAPH WALL TIME: 38.651s
RAM used by InvokeAI process: 5.37G (+4.326G)
RAM used to load models: 11.62G
VRAM in use: 0.008G
RAM cache statistics:
Model cache hits: 4
Model cache misses: 8
Models cached: 3
Models cleared from cache: 1
Cache high water mark: 7.39/7.58G
Discord username
No response
- Lingua principale
- Python
- Stelle
- 28.3k
- Fork
- 3k
- Merge medio
- 9g 22h
- PR unite (30g)
- 12
Preparare l'ambiente
Non abbiamo ancora controllato i file di configurazione di questo progetto. Parti dal suo README e consulta la nostra guida al primo contributo per i passaggi generali.
Come iniziare
- Leggi tutta la issue e poi la guida ai contributi del progetto.
- Commenta sulla issue per dire che te ne occupi tu — evita che due persone facciano lo stesso lavoro.
- Fai un fork del repository e lavora su un branch.
- Apri una pull request che faccia riferimento al numero della issue.
Altre issue di invoke-ai/InvokeAI
-
enhancement
Difficoltà 5/5 Più di una settimana Idoneità per principianti 35/100
I maintainer di solito rispondono entro 8 giorni
-
[bug]: 6.14.1 regression with `pytorch_cuda_alloc_conf: backend:cudaMallocAsync` — Z-Image bf16 + LoRA takes ~30 min whenever the transformer is (re)loaded (VRAM overflows into Windows shared memory)Forse già presa @lstein l’ha presa 4 giorni fa. Aperta
invoke-ai/InvokeAI#9597 · 2 commenti · 1 assegnatario ·
I maintainer di solito rispondono entro 8 giorni
-
enhancement
Difficoltà 5/5 Più di una settimana Idoneità per principianti 25/100
invoke-ai/InvokeAI#9594 · 2 commenti · 7 reazioni ·
I maintainer di solito rispondono entro 8 giorni
-
bug
Difficoltà 4/5 3-5 giorni Idoneità per principianti 35/100
invoke-ai/InvokeAI#9586 · 1 commento ·
I maintainer di solito rispondono entro 8 giorni
-
[bug]: Crashing Krea2 checkpoints after recent update (Anything but the smallest Q8_0 GGUF crashes, Even standard checkpoints)Forse già presa @Pfannkuchensack l’ha presa 8 giorni fa. Apertabug
Difficoltà 4/5 3-5 giorni Idoneità per principianti 38/100
invoke-ai/InvokeAI#9585 · 9 commenti · 1 assegnatario ·
I maintainer di solito rispondono entro 8 giorni
Tutte le issue di invoke-ai/InvokeAI
Issue simili
-
Difficoltà 2/5 1-3 ore Idoneità per principianti 84/100
PedestrianDynamics/pyFDS-Evac#343 ·
I maintainer di solito rispondono entro 1 giorno
-
Difficoltà 2/5 1-3 ore Idoneità per principianti 88/100
theskumar/python-dotenv#708 ·
-
Difficoltà 1/5 Meno di un'ora Idoneità per principianti 88/100
I maintainer di solito rispondono entro 2 giorni
-
Docs Timedelta
Difficoltà 2/5 1-3 ore Idoneità per principianti 72/100
pandas-dev/pandas#69919 ·
I maintainer di solito rispondono entro 1 giorno
-
API documentation
Difficoltà 2/5 1-3 ore Idoneità per principianti 72/100
zephyrproject-rtos/west#1009 · 2 commenti ·
I maintainer di solito rispondono entro 3 giorni