Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions conversion/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -28,6 +28,7 @@
"BailingMoeForCausalLM": "bailingmoe",
"BailingMoeV2ForCausalLM": "bailingmoe",
"BailingMoeV3ForCausalLM": "bailingmoe3",
"BailingMoeV3VLForConditionalGeneration": "bailingmoe3",
"BambaForCausalLM": "granite",
"BertForMaskedLM": "bert",
"BertForSequenceClassification": "bert",
Expand Down Expand Up @@ -301,6 +302,7 @@
"Gemma4ForConditionalGeneration": "gemma",
"Gemma4UnifiedForConditionalGeneration": "gemma",
"Glm4vForConditionalGeneration": "qwen3vl",
"BailingMoeV3VLForConditionalGeneration": "bailingmoe3",
"Glm4vMoeForConditionalGeneration": "qwen3vl",
"Glm5vForConditionalGeneration": "kimivl",
"GlmOcrForConditionalGeneration": "qwen3vl",
Expand Down
114 changes: 112 additions & 2 deletions conversion/bailingmoe3.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,7 +9,9 @@
if TYPE_CHECKING:
from torch import Tensor

from .base import ModelBase, TextModel, gguf
from .base import ModelBase, MmprojModel, TextModel, gguf

from .qwen3vl import Qwen3VLVisionModel


@ModelBase.register("BailingMoeV3ForCausalLM")
Expand Down Expand Up @@ -74,7 +76,7 @@ def set_gguf_parameters(self):

self.gguf_writer.add_expert_feed_forward_length(self.hparams["moe_intermediate_size"])
self.gguf_writer.add_expert_shared_feed_forward_length(self.hparams["moe_shared_expert_intermediate_size"])
self.gguf_writer.add_expert_shared_count(self.hparams["num_shared_experts"])
self.gguf_writer.add_expert_shared_count(self.hparams.get("num_shared_experts", 1))
self.gguf_writer.add_leading_dense_block_count(self.hparams["first_k_dense_replace"])
self.gguf_writer.add_expert_weights_scale(self.hparams["routed_scaling_factor"])
self.gguf_writer.add_expert_weights_norm(self.hparams["norm_topk_prob"])
Expand Down Expand Up @@ -191,3 +193,111 @@ def prepare_tensors(self):
experts = [name for layer in self._experts for name in layer]
if experts:
raise ValueError(f"Unprocessed experts: {experts}")


@ModelBase.register("BailingMoeV3VLForConditionalGeneration")
@ModelBase.example("inclusionAI/Ling-3.0-flash-VL")
class BailingMoeV3VLModel(BailingMoeV3Model):
model_arch = gguf.MODEL_ARCH.BAILINGMOE3

def index_tensors(self, remote_hf_model_id: str | None = None):
# hoist text_config before the shared BailingMoeV3 logic runs:
# ModelBase.__init__ calls this with the raw VL config, where the text
# dims still live under text_config
if "text_config" in self.hparams:
self.hparams = {**self.hparams, **self.hparams["text_config"]}
return super().index_tensors(remote_hf_model_id=remote_hf_model_id)

def set_gguf_parameters(self):
super().set_gguf_parameters()
mrope_section = self.hparams.get("mrope_section")
if mrope_section is None:
raise ValueError("BailingMoeV3VL requires mrope_section in the config")
if sum(mrope_section[:3]) * 2 != self.hparams["qk_rope_head_dim"]:
raise ValueError(
f"mrope_section {mrope_section[:3]} counts rope pairs and must sum to"
f" qk_rope_head_dim / 2 = {self.hparams['qk_rope_head_dim'] // 2}"
)
# mrope_section is [t, h, w]; pad to the 4-wide sections array
self.gguf_writer.add_rope_dimension_sections(list(mrope_section[:3]) + [0])

@classmethod
def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:
name, gen = item

# Skip projector tensors; the vision tower is skipped by TextModel.filter_tensors
if name.startswith("linear_proj"):
return None

return super().filter_tensors(item)


@ModelBase.register("BailingMoeV3VLForConditionalGeneration")
@ModelBase.example("inclusionAI/Ling-3.0-flash-VL")
class BailingMoeV3VLVisionModel(Qwen3VLVisionModel):
def __init__(self, *args, **kwargs):
super().__init__(*args, **kwargs)
assert self.hparams_vision is not None

if self.hparams_vision.get("disable_merger_proj") is not True:
raise ValueError("BailingMoeV3VL requires disable_merger_proj=true")

# out_hidden_size is the vision encoder output (post spatial merge, pre linear_proj)
self.image_emb_dim = self.hparams_vision.get("out_hidden_size")
if self.image_emb_dim is None:
raise ValueError("BailingMoeV3VL vision config requires out_hidden_size")

def set_gguf_parameters(self):
assert self.hparams_vision is not None
MmprojModel.set_gguf_parameters(self) # skip Qwen3VLVisionModel parameters
self.gguf_writer.add_clip_projector_type(gguf.VisionProjectorType.LING3VL)
self.gguf_writer.add_vision_use_gelu(True)

merge_size = self.hparams_vision.get("spatial_merge_size")
if merge_size is not None:
self.gguf_writer.add_vision_spatial_merge_size(int(merge_size))

rms_norm_eps = self.global_config.get("text_config", {}).get("rms_norm_eps", 1e-6)
self.gguf_writer.add_vision_attention_layernorm_eps(rms_norm_eps)

@classmethod
def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:
name, gen = item

if name.startswith("lm_head."):
return None

if name.startswith("linear_proj"):
# top-level projector MLP: linear_proj.0 -> mm.0, linear_proj.2 -> mm.2
parts = name.split(".")
if len(parts) != 3:
raise ValueError(f"Unexpected linear_proj tensor: {name}")
idx, suffix = int(parts[1]), parts[2]
name = f"mm.{idx}.{suffix}"
# the qwen3vl filter keeps only visual.*; skip it for the renamed projector tensors
return MmprojModel.filter_tensors((name, gen))

if name.startswith("model.visual."):
name = name.replace("model.visual.", "visual.", 1)

if not name.startswith("visual."):
return None

return super().filter_tensors((name, gen))

def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
assert self.hparams_vision is not None

if name.startswith("mm.0.") or name.startswith("mm.2."):
# top-level projector MLP (linear_proj.0 / linear_proj.2, renamed by filter_tensors)
yield (name, data_torch)
return

if name == "visual.merger.norm.weight" or name == "visual.merger.norm.bias":
# the merger is norm-only for Ling: per-patch LayerNorm before the spatial merge
new_name = f"mm.input_norm.{name.split('.')[-1]}"
yield (new_name, data_torch)
return

# Ling has no patch bias; the Conv3D split below matches the stock qwen3vl path
yield from Qwen3VLVisionModel.modify_tensors(self, data_torch, name, bid)
3 changes: 3 additions & 0 deletions gguf-py/gguf/constants.py
Original file line number Diff line number Diff line change
Expand Up @@ -650,6 +650,7 @@ class VISION_PROJECTOR_TYPE(IntEnum):
GEMMA3N = auto()
GEMMA3 = auto()
QWEN3VL = auto()
LING3VL = auto()
STEP3VL = auto()
COGVLM = auto()

Expand Down Expand Up @@ -1407,6 +1408,7 @@ class MODEL_TENSOR(IntEnum):
VISION_PROJECTOR_TYPE.MERGER: "qwen2vl_merger",
VISION_PROJECTOR_TYPE.GEMMA3: "gemma3",
VISION_PROJECTOR_TYPE.QWEN3VL: "qwen3vl_merger",
VISION_PROJECTOR_TYPE.LING3VL: "ling3vl",
VISION_PROJECTOR_TYPE.STEP3VL: "step3vl",
}

Expand Down Expand Up @@ -5825,6 +5827,7 @@ class VisionProjectorType:
QWEN25VL = "qwen2.5vl_merger"
EXAONE4_5 = "exaone4_5"
QWEN3VL = "qwen3vl_merger"
LING3VL = "ling3vl"
STEP3VL = "step3vl"
ULTRAVOX = "ultravox"
INTERNVL = "internvl"
Expand Down
5 changes: 4 additions & 1 deletion src/llama-model.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -2955,7 +2955,6 @@ llama_rope_type llama_model_rope_type(const llama_model * model) {
case LLM_ARCH_GRANITE_SWA:
case LLM_ARCH_CHAMELEON:
case LLM_ARCH_BAILINGMOE:
case LLM_ARCH_BAILINGMOE3:
case LLM_ARCH_NEO_BERT:
case LLM_ARCH_SMOLLM3:
case LLM_ARCH_ARCEE:
Expand All @@ -2970,6 +2969,10 @@ llama_rope_type llama_model_rope_type(const llama_model * model) {
case LLM_ARCH_DOTS3NOTE:
case LLM_ARCH_NANBEIGE:
case LLM_ARCH_POCKETTTS:
return LLAMA_ROPE_TYPE_NORM;
case LLM_ARCH_BAILINGMOE3:
// VL files carry mrope sections; text-only files keep NORM rope
return model->hparams.use_mrope() ? LLAMA_ROPE_TYPE_MROPE : LLAMA_ROPE_TYPE_NORM;
// HY_V4 rotates consecutive pairs, matching the reference implementation
case LLM_ARCH_HY_V4:
return LLAMA_ROPE_TYPE_NORM;
Expand Down
39 changes: 31 additions & 8 deletions src/models/bailingmoe3.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -15,6 +15,7 @@ void llama_model_bailingmoe3::load_arch_hparams(llama_model_loader & ml) {
hparams.kda_safe_gate = true;
}
ml.get_key(LLM_KV_KDA_GATE_LOWER_BOUND, hparams.kda_gate_lower_bound);
ml.get_key_or_arr(LLM_KV_ROPE_DIMENSION_SECTIONS, hparams.rope_sections, 4, false);
ml.get_key_or_arr(LLM_KV_EXPERT_FEED_FORWARD_LENGTH, hparams.n_ff_exp_arr, hparams.n_layer_all);
ml.get_key(LLM_KV_EXPERT_SHARED_FEED_FORWARD_LENGTH, hparams.n_ff_shexp, false);
ml.get_key(LLM_KV_EXPERT_SHARED_COUNT, hparams.n_expert_shared);
Expand Down Expand Up @@ -233,6 +234,10 @@ llama_model_bailingmoe3::graph::graph(const llama_model & model, const llm_graph
const int64_t d_conv = hparams.ssm_d_conv;
const int64_t n_seqs = ubatch.n_seqs;
const int64_t n_seq_tokens = ubatch.n_seq_tokens;

const bool use_mrope = hparams.use_mrope();
int sections[4];
std::copy(std::begin(hparams.rope_sections), std::begin(hparams.rope_sections) + 4, sections);
const int64_t qk_head_dim = hparams.n_embd_head_k_mla();
const int64_t v_head_dim = hparams.n_embd_head_v_mla();
const int64_t qk_rope_head_dim = hparams.n_rot();
Expand Down Expand Up @@ -326,10 +331,17 @@ llama_model_bailingmoe3::graph::graph(const llama_model & model, const llm_graph
ggml_row_size(kv_all->type, kv_lora_rank + qk_rope_head_dim),
ggml_row_size(kv_all->type, kv_lora_rank));

q_pe = ggml_rope_ext(ctx0, q_pe, inp_pos, nullptr, n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,
ext_factor, attn_factor, beta_fast, beta_slow);
k_pe = ggml_rope_ext(ctx0, k_pe, inp_pos, nullptr, n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,
ext_factor, attn_factor, beta_fast, beta_slow);
if (use_mrope) {
q_pe = ggml_rope_multi(ctx0, q_pe, inp_pos, nullptr, n_rot, sections, rope_type,
n_ctx_orig, freq_base, freq_scale, ext_factor, attn_factor, beta_fast, beta_slow);
k_pe = ggml_rope_multi(ctx0, k_pe, inp_pos, nullptr, n_rot, sections, rope_type,
n_ctx_orig, freq_base, freq_scale, ext_factor, attn_factor, beta_fast, beta_slow);
} else {
q_pe = ggml_rope_ext(ctx0, q_pe, inp_pos, nullptr, n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,
ext_factor, attn_factor, beta_fast, beta_slow);
k_pe = ggml_rope_ext(ctx0, k_pe, inp_pos, nullptr, n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,
ext_factor, attn_factor, beta_fast, beta_slow);
}
kv = build_norm(kv, layer.attn_kv_a_norm, nullptr, LLM_NORM_RMS, il);

q_nope = ggml_permute(ctx0, q_nope, 0, 2, 1, 3);
Expand Down Expand Up @@ -482,10 +494,21 @@ llama_model_bailingmoe3::graph_mtp::graph_mtp(const llama_model & model, const l
ggml_row_size(kv_all->type, kv_lora_rank + qk_rope_head_dim),
ggml_row_size(kv_all->type, kv_lora_rank));

q_pe = ggml_rope_ext(ctx0, q_pe, inp_pos, nullptr, n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,
ext_factor, attn_factor, beta_fast, beta_slow);
k_pe = ggml_rope_ext(ctx0, k_pe, inp_pos, nullptr, n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,
ext_factor, attn_factor, beta_fast, beta_slow);
const bool use_mrope = hparams.use_mrope();
int sections[4];
std::copy(std::begin(hparams.rope_sections), std::begin(hparams.rope_sections) + 4, sections);

if (use_mrope) {
q_pe = ggml_rope_multi(ctx0, q_pe, inp_pos, nullptr, n_rot, sections, rope_type,
n_ctx_orig, freq_base, freq_scale, ext_factor, attn_factor, beta_fast, beta_slow);
k_pe = ggml_rope_multi(ctx0, k_pe, inp_pos, nullptr, n_rot, sections, rope_type,
n_ctx_orig, freq_base, freq_scale, ext_factor, attn_factor, beta_fast, beta_slow);
} else {
q_pe = ggml_rope_ext(ctx0, q_pe, inp_pos, nullptr, n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,
ext_factor, attn_factor, beta_fast, beta_slow);
k_pe = ggml_rope_ext(ctx0, k_pe, inp_pos, nullptr, n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,
ext_factor, attn_factor, beta_fast, beta_slow);
}
kv = build_norm(kv, layer.attn_kv_a_norm, nullptr, LLM_NORM_RMS, il);

q_nope = ggml_permute(ctx0, q_nope, 0, 2, 1, 3);
Expand Down
8 changes: 7 additions & 1 deletion tests/test-llama-archs.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -332,7 +332,13 @@ static gguf_context_ptr get_gguf_ctx(const llm_arch arch, const bool moe) {

ms.add_kv(LLM_KV_ATTENTION_INDEXER_BLOCK_SIZE, uint32_t(4));
ms.add_kv(LLM_KV_ATTENTION_INDEXER_LOCAL_BLOCKS, uint32_t(1));
ms.add_kv(LLM_KV_ROPE_DIMENSION_SECTIONS, std::vector<uint32_t>({n_embd_head/4, n_embd_head/4, n_embd_head/4, n_embd_head/4}));
// mrope sections count rope pairs; Ling 3.0 VL files carry [t, h, w] sections
// summing to n_rot / 2 (n_rot is 64 in this fixture)
if (arch == LLM_ARCH_BAILINGMOE3) {
ms.add_kv(LLM_KV_ROPE_DIMENSION_SECTIONS, std::vector<uint32_t>({8, 12, 12, 0}));
} else {
ms.add_kv(LLM_KV_ROPE_DIMENSION_SECTIONS, std::vector<uint32_t>({n_embd_head/4, n_embd_head/4, n_embd_head/4, n_embd_head/4}));
}

if (arch == LLM_ARCH_HY_V4) {
ms.add_kv(LLM_KV_HYPER_CONNECTION_COUNT, uint32_t(4));
Expand Down
1 change: 1 addition & 0 deletions tools/mtmd/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -55,6 +55,7 @@ add_library(mtmd
models/qwen2vl.cpp
models/minimax-m3.cpp
models/qwen3vl.cpp
models/ling3vl.cpp
models/mimovl.cpp
models/qwen3a.cpp
models/mimo-audio.cpp
Expand Down
2 changes: 2 additions & 0 deletions tools/mtmd/clip-impl.h
Original file line number Diff line number Diff line change
Expand Up @@ -450,6 +450,7 @@ enum projector_type {
PROJECTOR_TYPE_GLM_EDGE,
PROJECTOR_TYPE_QWEN2VL,
PROJECTOR_TYPE_QWEN3VL,
PROJECTOR_TYPE_LING3VL,
PROJECTOR_TYPE_STEP3VL,
PROJECTOR_TYPE_GEMMA3,
PROJECTOR_TYPE_GEMMA3NV,
Expand Down Expand Up @@ -516,6 +517,7 @@ static std::map<projector_type, std::string> PROJECTOR_TYPE_NAMES = {
{ PROJECTOR_TYPE_QWEN2VL, "qwen2vl_merger"},
{ PROJECTOR_TYPE_QWEN25VL, "qwen2.5vl_merger"},
{ PROJECTOR_TYPE_QWEN3VL, "qwen3vl_merger"},
{ PROJECTOR_TYPE_LING3VL, "ling3vl"},
{ PROJECTOR_TYPE_STEP3VL, "step3vl"},
{ PROJECTOR_TYPE_GEMMA3, "gemma3"},
{ PROJECTOR_TYPE_GEMMA3NV, "gemma3nv"},
Expand Down
21 changes: 21 additions & 0 deletions tools/mtmd/clip.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -976,6 +976,10 @@ static std::unique_ptr<clip_graph> clip_get_graph_builder(clip_ctx * ctx, const
{
builder = std::make_unique<clip_graph_qwen3vl>(ctx, img);
} break;
case PROJECTOR_TYPE_LING3VL:
{
builder = std::make_unique<clip_graph_ling3vl>(ctx, img);
} break;
case PROJECTOR_TYPE_EXAONE4_5:
{
builder = std::make_unique<clip_graph_exaone4_5>(ctx, img);
Expand Down Expand Up @@ -1651,6 +1655,7 @@ struct clip_model_loader {
case PROJECTOR_TYPE_QWEN2VL:
case PROJECTOR_TYPE_QWEN25VL:
case PROJECTOR_TYPE_QWEN3VL:
case PROJECTOR_TYPE_LING3VL:
{
hparams.n_merge = 2; // default value for Qwen 2 and 2.5
hparams.image_resize_algo = RESIZE_ALGO_BICUBIC;
Expand Down Expand Up @@ -2478,6 +2483,15 @@ struct clip_model_loader {
model.mm_1_w = get_tensor(string_format(TN_LLAVA_PROJ, 2, "weight"));
model.mm_1_b = get_tensor(string_format(TN_LLAVA_PROJ, 2, "bias"));
} break;
case PROJECTOR_TYPE_LING3VL:
{
model.mm_input_norm_w = get_tensor(TN_MM_INP_NORM); // merger.norm
model.mm_input_norm_b = get_tensor(TN_MM_INP_NORM_B); // merger.norm
model.mm_0_w = get_tensor(string_format(TN_LLAVA_PROJ, 0, "weight")); // linear_proj.0
model.mm_0_b = get_tensor(string_format(TN_LLAVA_PROJ, 0, "bias"));
model.mm_1_w = get_tensor(string_format(TN_LLAVA_PROJ, 2, "weight")); // linear_proj.2
model.mm_1_b = get_tensor(string_format(TN_LLAVA_PROJ, 2, "bias"));
} break;
case PROJECTOR_TYPE_MIMOVL:
{
model.mm_0_w = get_tensor(string_format(TN_LLAVA_PROJ, 0, "weight"));
Expand Down Expand Up @@ -4038,6 +4052,7 @@ int clip_n_output_tokens_x(const clip_ctx * ctx, const clip_image_f32 * img) {
case PROJECTOR_TYPE_QWEN2VL:
case PROJECTOR_TYPE_QWEN25VL:
case PROJECTOR_TYPE_QWEN3VL:
case PROJECTOR_TYPE_LING3VL:
case PROJECTOR_TYPE_EXAONE4_5:
case PROJECTOR_TYPE_MIMOVL:
case PROJECTOR_TYPE_GLM4V:
Expand All @@ -4064,6 +4079,7 @@ int clip_n_output_tokens_y(const clip_ctx * ctx, const clip_image_f32 * img) {
case PROJECTOR_TYPE_QWEN2VL:
case PROJECTOR_TYPE_QWEN25VL:
case PROJECTOR_TYPE_QWEN3VL:
case PROJECTOR_TYPE_LING3VL:
case PROJECTOR_TYPE_EXAONE4_5:
case PROJECTOR_TYPE_MIMOVL:
case PROJECTOR_TYPE_GLM4V:
Expand Down Expand Up @@ -4144,6 +4160,7 @@ int clip_n_output_tokens(const clip_ctx * ctx, const clip_image_f32 * img) {
case PROJECTOR_TYPE_QWEN2VL:
case PROJECTOR_TYPE_QWEN25VL:
case PROJECTOR_TYPE_QWEN3VL:
case PROJECTOR_TYPE_LING3VL:
case PROJECTOR_TYPE_EXAONE4_5:
case PROJECTOR_TYPE_MIMOVL:
case PROJECTOR_TYPE_MINIMAX_M3:
Expand Down Expand Up @@ -4782,6 +4799,7 @@ bool clip_encode(struct clip_ctx * ctx, struct clip_encode_params * params) {
} break;
case PROJECTOR_TYPE_QWEN2VL:
case PROJECTOR_TYPE_QWEN3VL:
case PROJECTOR_TYPE_LING3VL:
case PROJECTOR_TYPE_GLM4V:
{
const int merge_ratio = hparams.n_merge;
Expand Down Expand Up @@ -5946,6 +5964,8 @@ int clip_n_mmproj_embd(const struct clip_ctx * ctx) {
case PROJECTOR_TYPE_QWEN3VL:
// main path + deepstack paths
return ctx->model.mm_1_b->ne[0] * (1 + ctx->model.n_deepstack_layers);
case PROJECTOR_TYPE_LING3VL:
return ctx->model.mm_1_b->ne[0];
case PROJECTOR_TYPE_MIMOVL:
return ctx->model.mm_1_w->ne[1];
case PROJECTOR_TYPE_STEP3VL:
Expand Down Expand Up @@ -6039,6 +6059,7 @@ int clip_model_n_temporal_merge(const struct clip_ctx * ctx) {
case PROJECTOR_TYPE_QWEN2VL:
case PROJECTOR_TYPE_QWEN25VL:
case PROJECTOR_TYPE_QWEN3VL:
case PROJECTOR_TYPE_LING3VL:
return 2;
default:
return 1;
Expand Down
Loading
Loading