Hacktoberfest 2026: le issue che i maintainer hanno segnato per ottobre, aperte e adatte ai principianti. Sfoglia le issue Hacktoberfest

[bug]: Error generating simple image on NVIDIA using FLUX.2 Klein 4B

Aperta
#9,593 14 commenti 0 reazioni 1 assegnatario Vedi su GitHub

I maintainer di solito rispondono entro 8 giorni

@Pfannkuchensack ci sta già lavorando.

Dal 20/9/2026.

Valutazione

Questa issue non è ancora stata valutata.

Descrizione

bug
Is there an existing issue for this problem?
  • I have searched the existing issues
Install method

Invoke's Launcher

Operating system

Linux

GPU vendor

Nvidia (CUDA)

GPU model

RTX 2080 Ti

GPU VRAM

11GB

Version number

v6.14.1

Browser

Firefox 156.0

System Information

{
"version": "6.14.1",
"dependencies": {
"absl-py" : "2.4.0",
"accelerate" : "1.14.0",
"annotated-doc" : "0.0.4",
"annotated-types" : "0.7.0",
"anyio" : "4.14.1",
"attrs" : "26.1.0",
"bcrypt" : "3.2.2",
"bidict" : "0.23.1",
"bitsandbytes" : "0.49.2",
"blake3" : "1.0.9",
"certifi" : "2026.6.17",
"cffi" : "2.0.0",
"charset-normalizer" : "3.4.7",
"click" : "8.4.2",
"coloredlogs" : "15.0.1",
"compel" : "2.4.0",
"contourpy" : "1.3.3",
"cryptography" : "49.0.0",
"CUDA" : "12.8",
"cycler" : "0.12.1",
"Deprecated" : "1.3.1",
"diffusers" : "0.40.0",
"dnspython" : "2.8.0",
"dynamicprompts" : "0.31.0",
"ecdsa" : "0.19.2",
"einops" : "0.8.2",
"email-validator" : "2.3.0",
"fastapi" : "0.141.1",
"fastapi-events" : "0.12.2",
"filelock" : "3.29.4",
"flatbuffers" : "25.12.19",
"fonttools" : "4.63.0",
"fsspec" : "2026.6.0",
"gguf" : "0.19.0",
"h11" : "0.16.0",
"hf-xet" : "1.6.0",
"httpcore" : "1.0.9",
"httptools" : "0.8.0",
"httpx" : "0.28.1",
"huggingface_hub" : "1.28.0",
"humanfriendly" : "10.0",
"idna" : "3.18",
"ImageIO" : "2.37.4",
"imageio-ffmpeg" : "0.6.0",
"importlib_metadata" : "9.0.0",
"InvokeAI" : "6.14.1",
"jax" : "0.7.1",
"jaxlib" : "0.7.1",
"Jinja2" : "3.1.6",
"jsonschema" : "4.26.0",
"jsonschema-specifications": "2025.9.1",
"kiwisolver" : "1.5.0",
"markdown-it-py" : "4.2.0",
"MarkupSafe" : "3.0.3",
"matplotlib" : "3.11.0",
"mdurl" : "0.1.2",
"mediapipe" : "0.10.14",
"mistral_common" : "1.11.6",
"ml_dtypes" : "0.5.4",
"mpmath" : "1.3.0",
"networkx" : "3.6.1",
"numpy" : "1.26.4",
"nvidia-cublas-cu12" : "12.8.3.14",
"nvidia-cuda-cupti-cu12" : "12.8.57",
"nvidia-cuda-nvrtc-cu12" : "12.8.61",
"nvidia-cuda-runtime-cu12" : "12.8.57",
"nvidia-cudnn-cu12" : "9.7.1.26",
"nvidia-cufft-cu12" : "11.3.3.41",
"nvidia-cufile-cu12" : "1.13.0.11",
"nvidia-curand-cu12" : "10.3.9.55",
"nvidia-cusolver-cu12" : "11.7.2.55",
"nvidia-cusparse-cu12" : "12.5.7.53",
"nvidia-cusparselt-cu12" : "0.6.3",
"nvidia-nccl-cu12" : "2.26.2",
"nvidia-nvjitlink-cu12" : "12.8.61",
"nvidia-nvtx-cu12" : "12.8.55",
"onnx" : "1.16.1",
"onnxruntime" : "1.19.2",
"opencv-contrib-python" : "4.11.0.86",
"opt_einsum" : "3.4.0",
"packaging" : "26.2",
"passlib" : "1.7.4",
"picklescan" : "1.0.4",
"pillow" : "12.2.0",
"prompt_toolkit" : "3.0.52",
"protobuf" : "4.25.9",
"psutil" : "7.2.2",
"pyasn1" : "0.6.3",
"pycountry" : "26.2.16",
"pycparser" : "3.0",
"pydantic" : "2.13.4",
"pydantic-extra-types" : "2.11.1",
"pydantic-settings" : "2.14.2",
"pydantic_core" : "2.46.4",
"Pygments" : "2.20.0",
"pyparsing" : "3.3.2",
"PyPatchMatch" : "1.0.2",
"python-dateutil" : "2.9.0.post0",
"python-dotenv" : "1.2.2",
"python-engineio" : "4.13.3",
"python-jose" : "3.5.0",
"python-multipart" : "0.0.32",
"python-socketio" : "5.16.3",
"PyWavelets" : "1.9.0",
"PyYAML" : "6.0.3",
"referencing" : "0.37.0",
"regex" : "2026.5.9",
"requests" : "2.34.2",
"rich" : "15.0.0",
"rpds-py" : "2026.5.1",
"rsa" : "4.9.1",
"safetensors" : "0.8.0",
"scipy" : "1.17.1",
"semver" : "3.0.4",
"sentencepiece" : "0.2.0",
"setuptools" : "82.0.1",
"shellingham" : "1.5.4",
"simple-websocket" : "1.1.0",
"six" : "1.17.0",
"sounddevice" : "0.5.5",
"spandrel" : "0.4.2",
"starlette" : "0.48.0",
"sympy" : "1.14.0",
"tiktoken" : "0.13.0",
"tokenizers" : "0.22.2",
"torch" : "2.7.1+cu128",
"torchsde" : "0.2.6",
"torchvision" : "0.22.1+cu128",
"tqdm" : "4.68.3",
"trampoline" : "0.1.2",
"transformers" : "5.5.4",
"triton" : "3.3.1",
"typer" : "0.25.1",
"typing-inspection" : "0.4.2",
"typing_extensions" : "4.15.0",
"urllib3" : "2.7.0",
"uvicorn" : "0.49.0",
"uvloop" : "0.22.1",
"watchfiles" : "1.2.0",
"wcwidth" : "0.8.1",
"websockets" : "16.0",
"wrapt" : "2.2.2",
"wsproto" : "1.3.2",
"xformers" : "0.0.31.post1",
"zipp" : "4.1.0"
},
"config": {
"schema_version": "4.0.3",
"legacy_models_yaml_path": null,
"host": "127.0.0.1",
"port": 9090,
"allow_origins": [],
"allow_credentials": true,
"allow_methods": [""],
"allow_headers": ["
"],
"ssl_certfile": null,
"ssl_keyfile": null,
"base_url": null,
"forwarded_allow_ips": "127.0.0.1",
"http_compression_level": 9,
"log_tokenization": false,
"patchmatch": true,
"models_dir": "models",
"convert_cache_dir": "models/.convert_cache",
"download_cache_dir": "models/.download_cache",
"legacy_conf_dir": "configs",
"db_dir": "databases",
"outputs_dir": "outputs",
"image_subfolder_strategy": "type",
"custom_nodes_dir": "nodes",
"style_presets_dir": "style_presets",
"workflow_thumbnails_dir": "workflow_thumbnails",
"log_handlers": ["console"],
"log_format": "color",
"log_level": "info",
"log_sql": false,
"log_level_network": "warning",
"use_memory_db": false,
"dev_reload": false,
"profile_graphs": false,
"profile_prefix": null,
"profiles_dir": "profiles",
"max_cache_ram_gb": null,
"max_cache_vram_gb": null,
"log_memory_usage": false,
"model_cache_keep_alive_min": 0,
"device_working_mem_gb": 3,
"enable_partial_loading": true,
"keep_ram_copy_of_weights": true,
"ram": null,
"vram": null,
"lazy_offload": true,
"pytorch_cuda_alloc_conf": null,
"device": "auto",
"generation_devices": "auto",
"offload_text_encoders_to_idle_gpus": true,
"precision": "auto",
"sequential_guidance": false,
"wan_memory_optimization": false,
"pid_memory_optimization": false,
"attention_type": "auto",
"attention_slice_size": "auto",
"force_tiled_decode": false,
"pil_compress_level": 1,
"max_queue_size": 10000,
"session_queue_mode": "round_robin",
"clear_queue_on_startup": false,
"max_queue_history": null,
"allow_nodes": null,
"deny_nodes": null,
"node_cache_size": 512,
"hashing_algorithm": "blake3_single",
"remote_api_tokens": null,
"scan_models_on_startup": false,
"allow_private_download_urls": false,
"download_proxy": null,
"unsafe_disable_picklescan": false,
"allow_unknown_models": true,
"multiuser": false,
"strict_password_checking": false,
"external_alibabacloud_api_key": null,
"external_alibabacloud_base_url": null,
"external_gemini_api_key": null,
"external_openai_api_key": null,
"external_gemini_base_url": null,
"external_openai_base_url": null,
"external_seedream_api_key": null,
"external_seedream_base_url": null
},
"set_config_fields": ["legacy_models_yaml_path", "image_subfolder_strategy"]
}

What happened

Opened the canvas, selected the model FLUX.2 Klein 4B, and wrote a simple test prompt ("Vase on a table."). It got to 25% completion before it failed fatally.

What you expected to happen

I expected to generate the image without error.

How to reproduce the problem

After a fresh install of the latest release of InvokeAI on my Bazitte 44 (NVIDIA Edition), I launched it and downloaded the FLUX.2 Klein starting pack only.
Opened the canvas, selected the model FLUX.2 Klein 4B, and wrote a simple test prompt. It got to 25% completion before it failed fatally.

Additional context

Here are the logs generated from start to error:

Started Invoke process with PID 25698
[2026-09-19 14:45:46,551]::[InvokeAI]::INFO --> Using torch device: NVIDIA GeForce RTX 2080 Ti
[2026-09-19 14:45:46,557]::[InvokeAI]::INFO --> cuDNN version: 90701
`Siglip2ImageProcessorFast` is deprecated. The `Fast` suffix for image processors has been removed; use `Siglip2ImageProcessor` instead.
>> patchmatch.patch_match: INFO - Compiling and loading c extensions from "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/patchmatch".
>> patchmatch.patch_match: ERROR - patchmatch failed to load or compile (Command 'make clean && make' returned non-zero exit status 2.).
>> patchmatch.patch_match: INFO - Refer to https://invoke-ai.github.io/InvokeAI/installation/060_INSTALL_PATCHMATCH/ for installation instructions.
[2026-09-19 14:45:49,697]::[InvokeAI]::INFO --> Patchmatch not loaded (nonfatal)
[2026-09-19 14:45:51,156]::[InvokeAI]::INFO --> InvokeAI version 6.14.1
[2026-09-19 14:45:51,156]::[InvokeAI]::INFO --> Root directory = /MyApps/InstalledApps
[2026-09-19 14:45:51,157]::[InvokeAI]::INFO --> Initializing database at /MyApps/InstalledApps/databases/invokeai.db
[2026-09-19 14:45:51,179]::[InvokeAI]::INFO --> JWT secret loaded from database
[2026-09-19 14:45:51,181]::[ModelManagerService]::INFO --> [MODEL CACHE] Calculated model RAM cache size: 7757.50 MB. Heuristics applied: [1, 2].
[2026-09-19 14:45:51,181]::[ModelManagerService]::INFO --> Model cache global RAM budget: 7.58 GB across 1 device cache(s).
[2026-09-19 14:45:51,185]::[ModelInstallService]::INFO --> Restoring incomplete installs
[2026-09-19 14:45:51,186]::[ModelInstallService]::INFO --> Finished restoring incomplete installs
[2026-09-19 14:45:51,250]::[InvokeAI]::INFO --> Invoke running on http://127.0.0.1:9090 (Press CTRL+C to quit)
[2026-09-19 14:47:35,574]::[InvokeAI]::INFO --> Executing queue item 5, session 2bfc8573-7cc7-41a4-8619-914faf6370ab on cuda:0
[2026-09-19 14:47:51,317]::[Qwen3EncoderGGUFLoader]::INFO --> Detected llama.cpp GGUF format, converting keys to PyTorch format
[2026-09-19 14:47:51,318]::[Qwen3EncoderGGUFLoader]::INFO --> Qwen3 GGUF Encoder config detected: layers=36, hidden=2560, heads=32, kv_heads=8, intermediate=9728, head_dim=128
[2026-09-19 14:47:53,107]::[Qwen3EncoderGGUFLoader]::INFO --> Dequantized embed_tokens weight for embedding lookups
[2026-09-19 14:47:53,107]::[Qwen3EncoderGGUFLoader]::INFO --> Tied lm_head.weight to embed_tokens.weight (GGUF tied embeddings)
[2026-09-19 14:47:53,905]::[ModelManagerService]::INFO --> [MODEL CACHE] Loaded model '165da151-1eb2-4457-821b-ea83785b1858:text_encoder' (Qwen3ForCausalLM) onto cuda device #0 in 0.75s. Total model size: 4326.88MB, VRAM: 4326.88MB (100.0%)
[2026-09-19 14:47:54,435]::[ModelManagerService]::INFO --> [MODEL CACHE] Loaded model '165da151-1eb2-4457-821b-ea83785b1858:tokenizer' (Qwen2Tokenizer) onto cuda device #0 in 0.00s. Total model size: 0.00MB, VRAM: 0.00MB (0.0%)
[2026-09-19 14:48:03,076]::[ModelManagerService]::INFO --> [MODEL CACHE] Loaded model '271b3dc7-6a26-42fa-a2ca-c08f2ddb9c88:vae' (AutoencoderKLFlux2) onto cuda device #0 in 0.25s. Total model size: 160.31MB, VRAM: 160.31MB (100.0%)
[2026-09-19 14:48:03,128]::[invokeai.backend.util.attention]::INFO --> SDPA materializes its attention score matrix on cuda for head_dim=128 (this torch build), so working-memory estimates reserve an extra ~8.1 GiB for it.
[2026-09-19 14:48:04,153]::[InvokeAI]::WARNING --> Loading 0.0146484375 MB into VRAM, but only -2477.013671875 MB were requested. This is the minimum set of weights in VRAM required to run the model.
[2026-09-19 14:48:04,237]::[ModelManagerService]::INFO --> [MODEL CACHE] Loaded model '993aa3a1-6dc5-4e56-b57e-7464c663ced1:transformer' (Flux2Transformer2DModel) onto cuda device #0 in 0.11s. Total model size: 7411.51MB, VRAM: 0.01MB (0.0%)
Denoising (#0):   0%|                                                                                                                                                                            | 0/4 [00:00<?, ?it/s][2026-09-19 14:48:14,233]::[InvokeAI]::ERROR --> Error while invoking session 2bfc8573-7cc7-41a4-8619-914faf6370ab, invocation db3c60db-68c3-4d90-acc2-20e1b475dca6 (flux2_denoise): CUDA error: CUBLAS_STATUS_EXECUTION_FAILED when calling `cublasSgemmStridedBatched( handle, opa, opb, m, n, k, &alpha, a, lda, stridea, b, ldb, strideb, &beta, c, ldc, stridec, num_batches)`
[2026-09-19 14:48:14,233]::[InvokeAI]::ERROR --> Traceback (most recent call last):
  File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/invokeai/app/services/session_processor/session_processor_default.py", line 278, in run_node
    output = invocation.invoke_internal(context=context, services=self._services)
             ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/invokeai/app/invocations/baseinvocation.py", line 248, in invoke_internal
    output = self.invoke(context)
             ^^^^^^^^^^^^^^^^^^^^
  File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/torch/utils/_contextlib.py", line 116, in decorate_context
    return func(*args, **kwargs)
           ^^^^^^^^^^^^^^^^^^^^^
  File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/invokeai/app/invocations/flux2_denoise.py", line 257, in invoke
    latents = self._run_diffusion(context)
              ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/invokeai/app/invocations/flux2_denoise.py", line 571, in _run_diffusion
    x = denoise(
        ^^^^^^^^
  File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/invokeai/backend/flux2/denoise.py", line 148, in denoise
    output = model(
             ^^^^^^
  File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
    return self._call_impl(*args, **kwargs)
           ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
    return forward_call(*args, **kwargs)
           ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/diffusers/utils/peft_utils.py", line 323, in wrapper
    result = forward_fn(self, *args, **kwargs)
             ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/diffusers/models/transformers/transformer_flux2.py", line 1408, in forward
    hidden_states = block(
                    ^^^^^^
  File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
    return self._call_impl(*args, **kwargs)
           ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
    return forward_call(*args, **kwargs)
           ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/diffusers/models/transformers/transformer_flux2.py", line 859, in forward
    attn_output = self.attn(
                  ^^^^^^^^^^
  File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1751, in _wrapped_call_impl
    return self._call_impl(*args, **kwargs)
           ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/torch/nn/modules/module.py", line 1762, in _call_impl
    return forward_call(*args, **kwargs)
           ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/diffusers/models/transformers/transformer_flux2.py", line 804, in forward
    return self.processor(self, hidden_states, attention_mask, image_rotary_emb, **kwargs)
           ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/diffusers/models/transformers/transformer_flux2.py", line 612, in __call__
    hidden_states = dispatch_attention_fn(
                    ^^^^^^^^^^^^^^^^^^^^^^
  File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/diffusers/models/attention_dispatch.py", line 439, in dispatch_attention_fn
    return backend_fn(**kwargs)
           ^^^^^^^^^^^^^^^^^^^^
  File "/MyApps/InstalledApps/.venv/lib/python3.12/site-packages/diffusers/models/attention_dispatch.py", line 3707, in _native_attention
    out = torch.nn.functional.scaled_dot_product_attention(
          ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
RuntimeError: CUDA error: CUBLAS_STATUS_EXECUTION_FAILED when calling `cublasSgemmStridedBatched( handle, opa, opb, m, n, k, &alpha, a, lda, stridea, b, ldb, strideb, &beta, c, ldc, stridec, num_batches)`

Denoising (#0):   0%|                                                                                                                                                                            | 0/4 [00:09<?, ?it/s]
[2026-09-19 14:48:14,252]::[InvokeAI]::INFO --> Graph stats: 2bfc8573-7cc7-41a4-8619-914faf6370ab
                          Node   Calls   Seconds VRAM Change
                       integer       1    0.006s     +0.000G
      flux2_klein_model_loader       1    0.012s     +0.000G
                        string       1    0.005s     +0.000G
                 core_metadata       1    0.001s     +0.000G
      flux2_klein_text_encoder       1   27.155s     +4.288G
                       collect       1    0.001s     +0.000G
                 flux2_denoise       1   11.463s     -3.907G
TOTAL GRAPH EXECUTION TIME:  38.643s
TOTAL GRAPH WALL TIME:  38.651s
RAM used by InvokeAI process: 5.37G (+4.326G)
RAM used to load models: 11.62G
VRAM in use: 0.008G
RAM cache statistics:
   Model cache hits: 4
   Model cache misses: 8
   Models cached: 3
   Models cleared from cache: 1
   Cache high water mark: 7.39/7.58G

Discord username

No response

Lingua principale
Python
Stelle
28.3k
Fork
3k
Merge medio
9g 22h
PR unite (30g)
12

Preparare l'ambiente

Non abbiamo ancora controllato i file di configurazione di questo progetto. Parti dal suo README e consulta la nostra guida al primo contributo per i passaggi generali.

Come iniziare

  1. Leggi tutta la issue e poi la guida ai contributi del progetto.
  2. Commenta sulla issue per dire che te ne occupi tu — evita che due persone facciano lo stesso lavoro.
  3. Fai un fork del repository e lavora su un branch.
  4. Apri una pull request che faccia riferimento al numero della issue.

Altre issue di invoke-ai/InvokeAI

Tutte le issue di invoke-ai/InvokeAI

Issue simili

Altre issue su Python

Ricevi le nuove issue nella tua casella

Un breve riepilogo di issue GitHub adatte ai principianti.