Repository navigation
Support mistral3 architecture #439
Description
Activity
I intend to make a PR with the following change, if @city96 agrees with it.
diff --git a/loader.py b/loader.py index 7cefb11..cd48359 100644 --- a/loader.py +++ b/loader.py @@ -393,7 +393,7 @@ def gguf_tekken_tokenizer_loader(path, temb_shape): model_str = get_field(reader, "tokenizer.ggml.model", str) if model_str == "gpt2": - if temb_shape == (131072, 5120): # probably Mistral + if temb_shape in {(131072, 5120), (131072, 3072)}: # Mistral variants ^M data = { "config": {"num_vocab_tokens": 150000, "default_vocab_size": 131072}, "vocab": [], @@ -483,7 +483,7 @@ def gguf_clip_loader(path): # TODO: pass model_options["vocab_size"] to loader somehow temb_key = "token_embd.weight" if temb_key in sd and sd[temb_key].shape[0] >= (64 * 1024): - if arch == "llama" and sd[temb_key].shape == (131072, 5120): + if arch == "llama" and sd[temb_key].shape in {(131072, 5120), (131072, 3072)}:^M # non-standard Comfy-Org tokenizer sd["tekken_model"] = gguf_tekken_tokenizer_loader(path, sd[temb_key].shape) elif arch == "gemma3":I just managed to make inference work using a quantized version of the text encoder from ComfyUI at https://huggingface.co/Comfy-Org/ERNIE-Image/tree/main/text_encoders. I also removed the vision layers which are not needed to generate text embeddings.
I also had to make a little change in
convert_hf_to_gguf.pyfrom llama.cpp to make it work. I'm also planning to upload the quantization to HuggingFace.Reacted by jacobairI did some more tests to support the official K-quants from MistralAI and came to the following diff. With this patch, it's possible to generate the text embeddings with them. The patch I proposed in the above comment only supported my own quantizations for de TE-only version of the ministral-3-3B model.
diff --git a/loader.py b/loader.py index 7cefb11..da71f2e 100644 --- a/loader.py +++ b/loader.py @@ -10,7 +10,8 @@ from .ops import GGMLTensor from .dequant import is_quantized, dequantize_tensor IMG_ARCH_LIST = {"flux", "sd1", "sdxl", "sd3", "aura", "hidream", "cosmos", "ltxv", "hyvid", "wan", "lumina2", "qwen_image"} -TXT_ARCH_LIST = {"t5", "t5encoder", "llama", "qwen2vl", "qwen3", "qwen3vl", "gemma3"} +TXT_ARCH_LIST = {"t5", "t5encoder", "llama", "qwen2vl", "qwen3", "qwen3vl", + "gemma3", "mistral3"} VIS_TYPE_LIST = {"clip-vision", "mmproj"} def get_orig_shape(reader, tensor_name): @@ -393,7 +394,7 @@ def gguf_tekken_tokenizer_loader(path, temb_shape): model_str = get_field(reader, "tokenizer.ggml.model", str) if model_str == "gpt2": - if temb_shape == (131072, 5120): # probably Mistral + if temb_shape in {(131072, 5120), (131072, 3072)}: # Mistral variants data = { "config": {"num_vocab_tokens": 150000, "default_vocab_size": 131072}, "vocab": [], @@ -479,11 +480,11 @@ def gguf_clip_loader(path): logging.warning(f"Dequantizing {temb_key} to prevent runtime OOM.") sd[temb_key] = dequantize_tensor(sd[temb_key], dtype=torch.float16) sd = sd_map_replace(sd, T5_SD_MAP) - elif arch in {"llama", "qwen2vl", "qwen3", "qwen3vl", "gemma3"}: + elif arch in {"llama", "qwen2vl", "qwen3", "qwen3vl", "gemma3", "mistral3"}: # TODO: pass model_options["vocab_size"] to loader somehow temb_key = "token_embd.weight" if temb_key in sd and sd[temb_key].shape[0] >= (64 * 1024): - if arch == "llama" and sd[temb_key].shape == (131072, 5120): + if arch in {"llama", "mistral3"} and sd[temb_key].shape in {(131072, 5120), (131072, 3072)}: # non-standard Comfy-Org tokenizer sd["tekken_model"] = gguf_tekken_tokenizer_loader(path, sd[temb_key].shape) elif arch == "gemma3": @@ -496,7 +497,7 @@ def gguf_clip_loader(path): sd = gemma3_norm_corrections(sd) else: sd = sd_map_replace(sd, LLAMA_SD_MAP) - if arch == "llama": + if arch in {"llama", "mistral3"}: sd = llama_permute(sd, 32, 8) # L3 / Mistral if arch == "qwen2vl": vsd = gguf_mmproj_loader(path)I'm guessing from these lines in
loader.pythat there was an intention to include mistral as llama, maybe because llama.cpp produced K-quants of Mistral models with such architecture.if arch == "llama": sd = llama_permute(sd, 32, 8) # L3 / MistralHere are the results using an NVFP4 quantization of ernie-image-turbo and an official Q5_K_M GGUF file from MistralAI HF repo at https://huggingface.co/mistralai/Ministral-3-3B-Instruct-2512-GGUF
Apr 18 08:21:58 akari python[4326]: got prompt Apr 18 08:21:58 akari python[4326]: Using pytorch attention in VAE Apr 18 08:21:58 akari python[4326]: Using pytorch attention in VAE Apr 18 08:21:58 akari python[4326]: VAE load device: cuda:0, offload device: cpu, dtype: torch.bfloat16 Apr 18 08:22:04 akari python[4326]: gguf qtypes: F32 (53), Q6_K (27), Q5_K (156) Apr 18 08:22:04 akari python[4326]: Attempting to recreate tekken tokenizer from GGUF file metadata... Apr 18 08:22:10 akari python[4326]: Created tekken tokenizer with vocab size of 130072 (+1000) Apr 18 08:22:10 akari python[4326]: Dequantizing token_embd.weight to prevent runtime OOM. Apr 18 08:22:11 akari python[4326]: [MultiGPU Core Patching] text_encoder_device_patched returning device: cuda:0 (current_text_encoder_device=cuda:0) Apr 18 08:22:12 akari python[4326]: CLIP/text encoder model load device: cuda:0, offload device: cpu, current: cpu, dtype: torch.float16 Apr 18 08:22:12 akari python[4326]: Requested to load ErnieTEModel_ Apr 18 08:22:13 akari python[4326]: loaded completely; 6189.74 MB usable, 2804.54 MB loaded, full load: True Apr 18 08:22:13 akari python[4326]: Found quantization metadata version 1 Apr 18 08:22:13 akari python[4326]: Detected mixed precision quantization Apr 18 08:22:13 akari python[4326]: Using mixed precision operations Apr 18 08:22:13 akari python[4326]: model weight dtype torch.bfloat16, manual cast: torch.bfloat16 Apr 18 08:22:13 akari python[4326]: model_type FLOW Apr 18 08:22:13 akari python[4326]: Using sage attention mode: auto Apr 18 08:22:13 akari python[4326]: Requested to load ErnieImage Apr 18 08:22:14 akari python[4326]: Unloaded partially: 2804.54 MB freed, 0.00 MB remains loaded, 2348.30 MB buffer reserved, lowvram patches: 0 Apr 18 08:22:14 akari python[4326]: Model ErnieImage prepared for dynamic VRAM loading. 4558MB Staged. 0 patches attached. Apr 18 08:22:21 akari python[4326]: [827B blob data] Apr 18 08:22:21 akari python[4326]: Requested to load AutoencoderKL Apr 18 08:22:21 akari python[4326]: 0 models unloaded. Apr 18 08:22:21 akari python[4326]: Model AutoencoderKL prepared for dynamic VRAM loading. 160MB Staged. 0 patches attached. Apr 18 08:22:21 akari python[4326]: Prompt executed in 23.82 seconds- added 8 commits that reference this issue
on Apr 18, 2026 It seems there was already an earlier PR #436 that achieves the same purpose.
thanks @insecure-erasure
