kingjones777 commited on
Commit
1329d8f
·
verified ·
1 Parent(s): 7e5701f

Upload k2-horizon-on-rocmfpx.patch with huggingface_hub

Browse files
Files changed (1) hide show
  1. k2-horizon-on-rocmfpx.patch +2166 -0
k2-horizon-on-rocmfpx.patch ADDED
@@ -0,0 +1,2166 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ diff --git a/conversion/__init__.py b/conversion/__init__.py
2
+ index 495e345..ccf7f2b 100644
3
+ --- a/conversion/__init__.py
4
+ +++ b/conversion/__init__.py
5
+ @@ -117,6 +117,8 @@ TEXT_MODEL_MAP: dict[str, str] = {
6
+ "JinaBertForMaskedLM": "bert",
7
+ "JinaBertModel": "bert",
8
+ "JinaEmbeddingsV5Model": "bert",
9
+ + "K2HorizonForCausalLM": "k2_horizon",
10
+ + "K2AuroraForCausalLM": "k2_horizon", # TODO: DELETE
11
+ "KORMoForCausalLM": "qwen",
12
+ "KimiK25ForConditionalGeneration": "deepseek",
13
+ "KimiLinearForCausalLM": "kimi_linear",
14
+ diff --git a/conversion/base.py b/conversion/base.py
15
+ index 6b45e5b..8e91a2c 100644
16
+ --- a/conversion/base.py
17
+ +++ b/conversion/base.py
18
+ @@ -219,6 +219,8 @@ class ModelBase:
19
+
20
+ prefix = "model" if not self.is_mistral_format else "consolidated"
21
+ part_names: list[str] = ModelBase.get_model_part_names(self.dir_model, prefix, ".safetensors")
22
+ + if not part_names and not self.is_mistral_format:
23
+ + part_names = ModelBase.get_model_part_names(self.dir_model, "pytorch_model", ".safetensors")
24
+ is_safetensors: bool = len(part_names) > 0
25
+ if not is_safetensors:
26
+ part_names = ModelBase.get_model_part_names(self.dir_model, "pytorch_model", ".bin")
27
+ @@ -1461,6 +1463,12 @@ class TextModel(ModelBase):
28
+ if chkhsh == "9e454714343b69b99b71795c1d27a68c2a1d15dab111f4d353109f966af29da7":
29
+ # ref: https://huggingface.co/LiquidAI/LFM2.5-8B-A1B
30
+ res = "lfm2"
31
+ + if chkhsh == "1f9825a388f700a6b591722f17d470cbbcf10973ece35d2fd14239a14110ae1a":
32
+ + # ref: https://huggingface.co/IFM/K2-Horizon-0.9B
33
+ + res = "k2-horizon"
34
+ + if chkhsh == "a9af07a84191f55098b248ae6f3dfe9e32d3190bebe8eafd91c1ddec9bc3449f":
35
+ + # ref: https://huggingface.co/IFM/K2-Horizon-36B
36
+ + res = "k2-horizon"
37
+ if chkhsh == "0ef9807a4087ebef797fc749390439009c3b9eda9ad1a097abbe738f486c01e5":
38
+ # ref: https://huggingface.co/meta-llama/Meta-Llama-3-8B
39
+ res = "llama-bpe"
40
+ diff --git a/conversion/k2_horizon.py b/conversion/k2_horizon.py
41
+ new file mode 100644
42
+ index 0000000..2d77222
43
+ --- /dev/null
44
+ +++ b/conversion/k2_horizon.py
45
+ @@ -0,0 +1,212 @@
46
+ +from __future__ import annotations
47
+ +
48
+ +import re
49
+ +from pathlib import Path
50
+ +from typing import Iterable
51
+ +
52
+ +import torch
53
+ +from torch import Tensor
54
+ +
55
+ +from .base import ModelBase, TextModel, gguf
56
+ +
57
+ +@ModelBase.register(
58
+ + "K2HorizonForCausalLM",
59
+ + "K2AuroraForCausalLM", # TODO: DELETE
60
+ +)
61
+ +@ModelBase.example(
62
+ + "IFM/K2-Horizon-0.9B",
63
+ + "IFM/K2-Horizon-36B",
64
+ +)
65
+ +class K2HorizonModel(TextModel):
66
+ + model_arch = gguf.MODEL_ARCH.K2HORIZON
67
+ +
68
+ + def set_vocab(self):
69
+ + super().set_vocab()
70
+ +
71
+ + template_path = (
72
+ + Path(__file__).parent.parent
73
+ + / "models"
74
+ + / "templates"
75
+ + / "k2-horizon.jinja"
76
+ + )
77
+ + template = template_path.read_text(encoding="utf-8")
78
+ + self.gguf_writer.remove_key(gguf.Keys.Tokenizer.CHAT_TEMPLATE)
79
+ + self.gguf_writer.add_chat_template(template)
80
+ +
81
+ + def set_gguf_parameters(self):
82
+ + super().set_gguf_parameters()
83
+ + # generic
84
+ + rope_head_dim = self.hparams.get("rope_head_dim")
85
+ + norm_groups = int(self.hparams.get("layernorm_num_groups", 1))
86
+ +
87
+ + self.gguf_writer.add_group_norm_groups(norm_groups)
88
+ + if rope_head_dim is not None:
89
+ + self.gguf_writer.add_rope_dimension_count(int(rope_head_dim))
90
+ +
91
+ + # moe
92
+ + num_experts = int(self.hparams.get("num_experts", 0))
93
+ + if num_experts > 0:
94
+ + moe_ff = int(self.hparams["moe_intermediate_size"])
95
+ + dense_layers = self.hparams.get("num_dense_layers")
96
+ + mlp_only_layers = {int(layer) for layer in self.hparams.get("mlp_only_layers", [])}
97
+ + sparse_step = int(self.hparams.get("decoder_sparse_step", 1))
98
+ + shared_experts = int(self.hparams.get("num_shared_experts", 0))
99
+ + router_scale = self.hparams.get("router_scaling_factor")
100
+ + normalize_topk = bool(self.hparams.get("norm_topk_prob", False))
101
+ + router_func = self.hparams.get("router_score_func")
102
+ +
103
+ + if dense_layers is None:
104
+ + dense_layers = 0
105
+ + while dense_layers in mlp_only_layers:
106
+ + dense_layers += 1
107
+ +
108
+ + self.gguf_writer.add_expert_feed_forward_length(moe_ff)
109
+ + self.gguf_writer.add_leading_dense_block_count(dense_layers)
110
+ + self.gguf_writer.add_moe_every_n_layers(sparse_step)
111
+ + self.gguf_writer.add_expert_shared_count(shared_experts)
112
+ + self.gguf_writer.add_expert_weights_norm(normalize_topk)
113
+ + if shared_experts > 0:
114
+ + self.gguf_writer.add_expert_shared_feed_forward_length(moe_ff * shared_experts)
115
+ + if router_scale is not None:
116
+ + self.gguf_writer.add_expert_weights_scale(float(router_scale))
117
+ + match router_func:
118
+ + case "sigmoid":
119
+ + gating_func = gguf.ExpertGatingFuncType.SIGMOID
120
+ + case "softmax":
121
+ + gating_func = gguf.ExpertGatingFuncType.SOFTMAX
122
+ + case _:
123
+ + raise ValueError(f"Unsupported router_score_func: {router_func!r}")
124
+ + self.gguf_writer.add_expert_gating_func(gating_func)
125
+ +
126
+ + # mova
127
+ + value_experts = int(self.hparams.get("mova_num_experts", 0))
128
+ + value_experts_used = int(self.hparams.get("mova_num_experts_per_tok", 0))
129
+ +
130
+ + if value_experts > 0 and value_experts_used > 0:
131
+ + assert value_experts_used <= value_experts
132
+ + self.gguf_writer.add_attention_value_expert_count(value_experts)
133
+ + self.gguf_writer.add_attention_value_expert_used_count(value_experts_used)
134
+ +
135
+ + # gate func, only making sure it exists and is softplus
136
+ + gate_func = self.hparams.get("attention_gate_func")
137
+ + if gate_func not in (None, "softplus"):
138
+ + raise ValueError(f"Unsupported attention_gate_func: {gate_func!r}")
139
+ +
140
+ + _experts: list[dict[str, Tensor]] | None = None
141
+ + _value_experts: list[dict[str, Tensor]] | None = None
142
+ + def modify_tensors(
143
+ + self,
144
+ + data_torch: Tensor,
145
+ + name: str,
146
+ + bid: int | None
147
+ + ) -> Iterable[tuple[str, Tensor]]:
148
+ + # MoE: router
149
+ + if name.endswith(".mlp.gate.bias"):
150
+ + assert bid is not None
151
+ + yield (
152
+ + self.format_tensor_name(
153
+ + gguf.MODEL_TENSOR.FFN_EXP_PROBS_B,
154
+ + bid,
155
+ + ".bias"
156
+ + ),
157
+ + data_torch
158
+ + )
159
+ + return
160
+ +
161
+ + # MoE: actual up down or gate
162
+ + is_moe_tensor = re.fullmatch(r"model\.layers\.\d+\.mlp\.experts\.\d+\.(down_proj|gate_proj|up_proj)\.weight", name)
163
+ + if is_moe_tensor:
164
+ + assert bid is not None
165
+ + num_experts = int(self.hparams["num_experts"])
166
+ +
167
+ + # allocate on first layer that has experts
168
+ + if self._experts is None:
169
+ + self._experts = [{} for _ in range(self.block_count)]
170
+ +
171
+ + # atp, this_blocks_experts contains all experts
172
+ + this_blocks_experts = self._experts[bid]
173
+ + this_blocks_experts[name] = data_torch
174
+ +
175
+ + # filling up self._experts until up down gate are all inside, then continue
176
+ + if len(this_blocks_experts) < num_experts * 3:
177
+ + return
178
+ +
179
+ + for projection in ("down_proj", "gate_proj", "up_proj"):
180
+ + tensors = []
181
+ + for expert_id in range(num_experts):
182
+ + expert_name = f"model.layers.{bid}.mlp.experts.{expert_id}.{projection}.weight"
183
+ + tensors.append(this_blocks_experts.pop(expert_name))
184
+ + merged = torch.stack(tensors, dim=0)
185
+ + merged_name = f"model.layers.{bid}.mlp.experts.{projection}.weight"
186
+ + yield from super().modify_tensors(
187
+ + merged,
188
+ + merged_name,
189
+ + bid
190
+ + )
191
+ + return
192
+ +
193
+ + # MoVA
194
+ + is_mova_weights = re.fullmatch(r"model\.layers\.\d+\.self_attn\.v_experts\.\d+\.weight", name)
195
+ + if is_mova_weights:
196
+ + assert bid is not None
197
+ + num_value_experts = int(self.hparams["mova_num_experts"])
198
+ + if self._value_experts is None:
199
+ + self._value_experts = [{} for _ in range(self.block_count)]
200
+ +
201
+ + this_blocks_value_expert = self._value_experts[bid]
202
+ + this_blocks_value_expert[name] = data_torch
203
+ +
204
+ + # no need to * 3 because no up down gate like normal moe
205
+ + if len(this_blocks_value_expert) < num_value_experts:
206
+ + return
207
+ +
208
+ + tensors = []
209
+ + for value_exp_id in range(num_value_experts):
210
+ + value_exp_name = f"model.layers.{bid}.self_attn.v_experts.{value_exp_id}.weight"
211
+ + tensors.append(this_blocks_value_expert.pop(value_exp_name))
212
+ +
213
+ + merged = torch.stack(tensors, dim = 0)
214
+ + merged_name = f"model.layers.{bid}.self_attn.v_experts.weight"
215
+ + yield from super().modify_tensors(
216
+ + merged,
217
+ + merged_name,
218
+ + bid
219
+ + )
220
+ + return
221
+ +
222
+ + # fallback, the default way basically
223
+ + yield from super().modify_tensors(
224
+ + data_torch,
225
+ + name,
226
+ + bid
227
+ + )
228
+ +
229
+ + def prepare_tensors(self):
230
+ + super().prepare_tensors()
231
+ +
232
+ + # this is just checks basically
233
+ + if self._experts is not None:
234
+ + remaining_experts = [
235
+ + name
236
+ + for block in self._experts
237
+ + for name in block
238
+ + ]
239
+ +
240
+ + if remaining_experts:
241
+ + raise ValueError(
242
+ + f"Unprocessed MoE experts: {remaining_experts}"
243
+ + )
244
+ +
245
+ + if self._value_experts is not None:
246
+ + remaining_value_experts = [
247
+ + name
248
+ + for block in self._value_experts
249
+ + for name in block
250
+ + ]
251
+ +
252
+ + if remaining_value_experts:
253
+ + raise ValueError(
254
+ + "Unprocessed MoVA value experts: "
255
+ + f"{remaining_value_experts}"
256
+ + )
257
+ +
258
+ diff --git a/convert_hf_to_gguf_update.py b/convert_hf_to_gguf_update.py
259
+ index 85b502f..ed1b561 100755
260
+ --- a/convert_hf_to_gguf_update.py
261
+ +++ b/convert_hf_to_gguf_update.py
262
+ @@ -190,6 +190,9 @@ pre_computed_hashes = [
263
+ # jina-v2-de variants
264
+ {"name": "jina-v2-de", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/aari1995/German_Semantic_V3", "chkhsh": "b3d1dd861f1d4c5c0d2569ce36baf3f90fe8a102db3de50dd71ff860d91be3df"},
265
+ {"name": "gpt-2", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/evilfreelancer/ruGPT3XL", "chkhsh": "0fe1cf6eda062318a1af7270f3331a85c539a01778ff948e24388e949c5282f4"},
266
+ + # K2 Horizon. 2 hashes because various sets of tokens depending on size
267
+ + {"name": "k2-horizon", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/IFM/K2-Horizon-0.9B", "chkhsh": "1f9825a388f700a6b591722f17d470cbbcf10973ece35d2fd14239a14110ae1a"},
268
+ + {"name": "k2-horizon", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/IFM/K2-Horizon-36B", "chkhsh": "a9af07a84191f55098b248ae6f3dfe9e32d3190bebe8eafd91c1ddec9bc3449f"},
269
+ ]
270
+
271
+
272
+ diff --git a/gguf-py/gguf/constants.py b/gguf-py/gguf/constants.py
273
+ index 4013f81..125bf3b 100644
274
+ --- a/gguf-py/gguf/constants.py
275
+ +++ b/gguf-py/gguf/constants.py
276
+ @@ -217,6 +217,8 @@ class Keys:
277
+ class Rope:
278
+ DIMENSION_COUNT = "{arch}.rope.dimension_count"
279
+ DIMENSION_COUNT_SWA = "{arch}.rope.dimension_count_swa"
280
+ + VALUE_EXPERT_COUNT = "{arch}.attention.value_expert_count"
281
+ + VALUE_EXPERT_USED_COUNT = "{arch}.attention.value_expert_used_count"
282
+ DIMENSION_SECTIONS = "{arch}.rope.dimension_sections"
283
+ FREQ_BASE = "{arch}.rope.freq_base"
284
+ FREQ_BASE_SWA = "{arch}.rope.freq_base_swa"
285
+ @@ -625,6 +627,7 @@ class MODEL_TENSOR(IntEnum):
286
+ MOE_LATENT_DOWN = auto() # nemotron 3 super
287
+ MOE_LATENT_UP = auto() # nemotron 3 super
288
+ ATTN_Q_NORM = auto()
289
+ + K2HORIZON = auto()
290
+ ATTN_K_NORM = auto()
291
+ LAYER_OUT_NORM = auto()
292
+ LAYER_OUT_SCALE = auto()
293
+ @@ -888,6 +891,9 @@ class MODEL_TENSOR(IntEnum):
294
+ V_DS_NORM = auto() # qwen3vl
295
+ V_DS_FC1 = auto() # qwen3vl
296
+ V_DS_FC2 = auto() # qwen3vl
297
+ + ATTN_V_GATE = auto() # K2Horizon
298
+ + ATTN_V_EXP = auto() # K2Horizon
299
+ +
300
+ V_MERGER_LN1 = auto() # minicpmv4_6
301
+ V_MERGER_ATTN_Q = auto() # minicpmv4_6
302
+ V_MERGER_ATTN_K = auto() # minicpmv4_6
303
+ @@ -1374,6 +1380,7 @@ TENSOR_NAMES: dict[MODEL_TENSOR, str] = {
304
+ MODEL_TENSOR.ENC_FFN_NORM: "enc.blk.{bid}.ffn_norm",
305
+ MODEL_TENSOR.ENC_FFN_GATE: "enc.blk.{bid}.ffn_gate",
306
+ MODEL_TENSOR.ENC_FFN_DOWN: "enc.blk.{bid}.ffn_down",
307
+ + MODEL_ARCH.K2HORIZON: "k2-horizon",
308
+ MODEL_TENSOR.ENC_FFN_UP: "enc.blk.{bid}.ffn_up",
309
+ MODEL_TENSOR.ENC_OUTPUT_NORM: "enc.output_norm",
310
+ MODEL_TENSOR.CLS: "cls",
311
+ @@ -1620,6 +1627,10 @@ TENSOR_NAMES: dict[MODEL_TENSOR, str] = {
312
+ MODEL_TENSOR.DSPARK_MARKOV_W2: "markov_w2",
313
+ MODEL_TENSOR.DSPARK_CONF_PROJ: "conf_proj",
314
+ MODEL_TENSOR.D2T: "d2t",
315
+ + # K2 Horizon
316
+ + MODEL_TENSOR.ATTN_V_GATE: "blk.{bid}.attn_v_gate",
317
+ + MODEL_TENSOR.ATTN_V_EXP: "blk.{bid}.attn_v_exps",
318
+ +
319
+ }
320
+
321
+ MODEL_TENSORS: dict[MODEL_ARCH, list[MODEL_TENSOR]] = {
322
+ @@ -4711,6 +4722,39 @@ MODEL_TENSOR_SKIP: dict[MODEL_ARCH, list[MODEL_TENSOR]] = {
323
+ MODEL_TENSOR.ROPE_FREQS,
324
+ MODEL_TENSOR.ATTN_ROT_EMBD,
325
+ ],
326
+ + MODEL_ARCH.K2HORIZON: [
327
+ + MODEL_TENSOR.TOKEN_EMBD,
328
+ + MODEL_TENSOR.OUTPUT_NORM,
329
+ + MODEL_TENSOR.OUTPUT,
330
+ + MODEL_TENSOR.ATTN_NORM,
331
+ + MODEL_TENSOR.ATTN_Q,
332
+ + MODEL_TENSOR.ATTN_Q_NORM,
333
+ + MODEL_TENSOR.ATTN_K,
334
+ + MODEL_TENSOR.ATTN_K_NORM,
335
+ + MODEL_TENSOR.ATTN_V,
336
+ + MODEL_TENSOR.ATTN_V_GATE, # MoVA
337
+ + MODEL_TENSOR.ATTN_V_EXP, # MoVA
338
+ + MODEL_TENSOR.ATTN_OUT,
339
+ + MODEL_TENSOR.ATTN_GATE,
340
+ + MODEL_TENSOR.FFN_NORM,
341
+ +
342
+ + # Dense MLP
343
+ + MODEL_TENSOR.FFN_GATE,
344
+ + MODEL_TENSOR.FFN_UP,
345
+ + MODEL_TENSOR.FFN_DOWN,
346
+ +
347
+ + # MoE
348
+ + MODEL_TENSOR.FFN_GATE_INP,
349
+ + MODEL_TENSOR.FFN_EXP_PROBS_B,
350
+ + MODEL_TENSOR.FFN_GATE_EXP,
351
+ + MODEL_TENSOR.FFN_UP_EXP,
352
+ + MODEL_TENSOR.FFN_DOWN_EXP,
353
+ +
354
+ + # Shared Expert
355
+ + MODEL_TENSOR.FFN_GATE_SHEXP,
356
+ + MODEL_TENSOR.FFN_UP_SHEXP,
357
+ + MODEL_TENSOR.FFN_DOWN_SHEXP,
358
+ + ]
359
+ }
360
+
361
+ #
362
+ diff --git a/gguf-py/gguf/gguf_writer.py b/gguf-py/gguf/gguf_writer.py
363
+ index cb26462..ac59e5a 100644
364
+ --- a/gguf-py/gguf/gguf_writer.py
365
+ +++ b/gguf-py/gguf/gguf_writer.py
366
+ @@ -1373,6 +1373,12 @@ class GGUFWriter:
367
+ def add_xielu_eps(self, values: Sequence[float]):
368
+ self.add_array(Keys.xIELU.EPS, values)
369
+
370
+ + def add_attention_value_expert_count(self, count: int):
371
+ + self.add_uint32(Keys.Attention.VALUE_EXPERT_COUNT.format(arch=self.arch), count)
372
+ +
373
+ + def add_attention_value_expert_used_count(self, count: int):
374
+ + self.add_uint32(Keys.Attention.VALUE_EXPERT_USED_COUNT.format(arch=self.arch), count)
375
+ +
376
+ # diffusion models
377
+
378
+ def add_diffusion_shift_logits(self, value: bool) -> None:
379
+ diff --git a/gguf-py/gguf/tensor_mapping.py b/gguf-py/gguf/tensor_mapping.py
380
+ index 7125cb4..841d1f2 100644
381
+ --- a/gguf-py/gguf/tensor_mapping.py
382
+ +++ b/gguf-py/gguf/tensor_mapping.py
383
+ @@ -392,6 +392,7 @@ class TensorNameMap:
384
+ "transformer.h.{bid}.ln_2", # gpt2 refact qwen jais exaone
385
+ "h.{bid}.post_attention_layernorm", # bloom
386
+ "transformer.blocks.{bid}.norm_2", # mpt
387
+ + "model.layers.{bid}.self_attn.attn_gate_proj", # K2Horizon
388
+ "model.layers.{bid}.post_attention_layernorm", # llama-hf nemotron olmoe phimoe
389
+ "layers.{bid}.ffn_norm", # llama-pth
390
+ "model.layers.{bid}.ln2", # yi
391
+ @@ -2319,6 +2320,14 @@ class TensorNameMap:
392
+ MODEL_TENSOR.NEXTN_SHARED_HEAD_NORM: (
393
+ "model.layers.{bid}.shared_head.norm",
394
+ ),
395
+ +
396
+ + MODEL_TENSOR.ATTN_V_GATE: (
397
+ + "model.layers.{bid}.self_attn.v_router",
398
+ + ),
399
+ +
400
+ + MODEL_TENSOR.ATTN_V_EXP: (
401
+ + "model.layers.{bid}.self_attn.v_experts",
402
+ + ),
403
+ }
404
+
405
+ # architecture-specific block mappings
406
+ diff --git a/models/templates/k2-horizon.jinja b/models/templates/k2-horizon.jinja
407
+ new file mode 100644
408
+ index 0000000..9dc680f
409
+ --- /dev/null
410
+ +++ b/models/templates/k2-horizon.jinja
411
+ @@ -0,0 +1,883 @@
412
+ +{{- bos_token }}
413
+ +{%- if tool_presentation is defined -%}
414
+ + {{- raise_exception("Unsupported argument: tool_presentation. Use tool_presentation_format with one of: json, xml, markdown.") -}}
415
+ +{%- endif -%}
416
+ +{%- if tool_calling_format is defined -%}
417
+ + {{- raise_exception("Unsupported argument: tool_calling_format. Use tool_call_format with one of: json, xml, xml_typed.") -}}
418
+ +{%- endif -%}
419
+ +{%- if tool_format is defined -%}
420
+ + {{- raise_exception("Unsupported argument: tool_format. Use tool_call_format with one of: json, xml, xml_typed.") -}}
421
+ +{%- endif -%}
422
+ +{%- set tool_presentation_fmt = tool_presentation_format | default('markdown') -%}
423
+ +{%- set tool_call_fmt = tool_call_format | default('xml') -%}
424
+ +{%- if tool_presentation_fmt != 'json' and tool_presentation_fmt != 'xml' and tool_presentation_fmt != 'markdown' -%}
425
+ + {{- raise_exception("Unsupported tool_presentation_format: '" ~ tool_presentation_fmt ~ "'. Supported formats: json, xml, markdown.") -}}
426
+ +{%- endif -%}
427
+ +{%- if tool_call_fmt != 'json' and tool_call_fmt != 'xml' and tool_call_fmt != 'xml_typed' -%}
428
+ + {{- raise_exception("Unsupported tool_call_format: '" ~ tool_call_fmt ~ "'. Supported formats: json, xml, xml_typed.") -}}
429
+ +{%- endif -%}
430
+ +
431
+ +{#- Renderability state, computed during validate_tools (single walk, no extra -#}
432
+ +{#- traversal at render time): ok = working flag for the tool being validated; -#}
433
+ +{#- bad = pipe-delimited indices of tools that must render as verbatim JSON. -#}
434
+ +{%- set RB = namespace(ok=true, bad='|') -%}
435
+ +
436
+ +{%- macro value_contains_mapping(v) -%}
437
+ +{%- if v is mapping -%}
438
+ +true
439
+ +{%- elif v is sequence and v is not string -%}
440
+ +{%- set f = namespace(x='false') -%}
441
+ +{%- for c in v -%}{%- if value_contains_mapping(c) == 'true' -%}{%- set f.x = 'true' -%}{%- endif -%}{%- endfor -%}
442
+ +{{- f.x -}}
443
+ +{%- else -%}
444
+ +false
445
+ +{%- endif -%}
446
+ +{%- endmacro -%}
447
+ +
448
+ +{%- macro render_compact_type_name(type_name, spec) -%}
449
+ +{%- if type_name == "array" -%}
450
+ +array[{%- if 'items' in spec -%}{{ render_compact_type(spec['items']) }}{%- else -%}any{%- endif -%}]
451
+ +{%- elif type_name -%}
452
+ +{{- type_name -}}
453
+ +{%- else -%}
454
+ +any
455
+ +{%- endif -%}
456
+ +{%- endmacro -%}
457
+ +
458
+ +{%- macro render_compact_type(spec) -%}
459
+ +{%- if spec is not mapping -%}
460
+ +any
461
+ +{%- elif spec.type is defined and spec.type is sequence and spec.type is not string and spec.type | length > 0 -%}
462
+ +{%- for type_name in spec.type -%}{{ render_compact_type_name(type_name, spec) }}{%- if not loop.last -%}|{%- endif -%}{%- endfor -%}
463
+ +{%- elif spec.type is defined and spec.type is sequence and spec.type is not string -%}
464
+ +any
465
+ +{%- elif spec.type -%}
466
+ +{{- render_compact_type_name(spec.type, spec) -}}
467
+ +{%- elif spec['$ref'] is string -%}
468
+ +{{- spec['$ref'].split('/') | last -}}
469
+ +{%- elif spec.oneOf -%}
470
+ +oneOf[{%- for variant in spec.oneOf -%}{{ render_compact_type(variant) }}{%- if not loop.last -%}|{%- endif -%}{%- endfor -%}]
471
+ +{%- elif spec.anyOf -%}
472
+ +anyOf[{%- for variant in spec.anyOf -%}{{ render_compact_type(variant) }}{%- if not loop.last -%}|{%- endif -%}{%- endfor -%}]
473
+ +{%- elif spec.properties -%}
474
+ +object
475
+ +{%- elif 'items' in spec -%}
476
+ +array[{{ render_compact_type(spec['items']) }}]
477
+ +{%- else -%}
478
+ +any
479
+ +{%- endif -%}
480
+ +{%- endmacro -%}
481
+ +
482
+ +{%- macro render_markdown_type_name(type_name, spec) -%}
483
+ +{%- if type_name == "array" -%}
484
+ +array of {% if 'items' in spec %}{{ render_markdown_type(spec['items']) }}{% else %}any{% endif %}
485
+ +{%- elif type_name -%}
486
+ +{{- type_name -}}
487
+ +{%- else -%}
488
+ +any
489
+ +{%- endif -%}
490
+ +{%- endmacro -%}
491
+ +
492
+ +{%- macro render_markdown_type(spec) -%}
493
+ +{%- if spec is true -%}
494
+ +True
495
+ +{%- elif spec is false -%}
496
+ +False
497
+ +{%- elif spec is not mapping -%}
498
+ +any
499
+ +{%- elif spec.type is defined and spec.type is sequence and spec.type is not string and spec.type | length > 0 -%}
500
+ +{%- for type_name in spec.type -%}{{ render_markdown_type_name(type_name, spec) }}{% if not loop.last %} or {% endif %}{%- endfor -%}
501
+ +{%- elif spec.type is defined and spec.type is sequence and spec.type is not string -%}
502
+ +any
503
+ +{%- elif spec.type -%}
504
+ +{{- render_markdown_type_name(spec.type, spec) -}}
505
+ +{%- elif spec['$ref'] is string -%}
506
+ +{{- spec['$ref'].split('/') | last -}}
507
+ +{%- elif spec.oneOf -%}
508
+ +oneOf[{%- for variant in spec.oneOf -%}{{ render_markdown_type(variant) }}{% if not loop.last %} or {% endif %}{%- endfor -%}]
509
+ +{%- elif spec.anyOf -%}
510
+ +anyOf[{%- for variant in spec.anyOf -%}{{ render_markdown_type(variant) }}{% if not loop.last %} or {% endif %}{%- endfor -%}]
511
+ +{%- elif spec.properties -%}
512
+ +object
513
+ +{%- elif 'items' in spec -%}
514
+ +array of {{ render_markdown_type(spec['items']) }}
515
+ +{%- else -%}
516
+ +any
517
+ +{%- endif -%}
518
+ +{%- endmacro -%}
519
+ +
520
+ +{%- macro render_xml_text(value) -%}
521
+ +{{- value.split() | join(" ") -}}
522
+ +{%- endmacro -%}
523
+ +
524
+ +{%- macro render_python_string(value) -%}
525
+ +'{{- value.split() | join(" ") | replace("\\", "\\\\") | replace("'", "\\'") -}}'
526
+ +{%- endmacro -%}
527
+ +
528
+ +{%- macro render_python_repr(value) -%}
529
+ +{%- if value is string -%}
530
+ +{{ render_python_string(value) }}
531
+ +{%- elif value is true -%}
532
+ +True
533
+ +{%- elif value is false -%}
534
+ +False
535
+ +{%- elif value is none -%}
536
+ +None
537
+ +{%- elif value is mapping -%}
538
+ +{{- "{" -}}
539
+ +{%- for key, child in value | items -%}
540
+ +{{ render_python_repr(key) }}: {{ render_python_repr(child) }}{%- if not loop.last -%}, {% endif -%}
541
+ +{%- endfor -%}
542
+ +{{- "}" -}}
543
+ +{%- elif value is sequence -%}
544
+ +{{- "[" -}}
545
+ +{%- for child in value -%}
546
+ +{{ render_python_repr(child) }}{%- if not loop.last -%}, {% endif -%}
547
+ +{%- endfor -%}
548
+ +{{- "]" -}}
549
+ +{%- else -%}
550
+ +{{- value -}}
551
+ +{%- endif -%}
552
+ +{%- endmacro -%}
553
+ +
554
+ +{%- macro render_xml_value(value) -%}
555
+ +{%- if value is string -%}{{ render_xml_text(value) }}{%- else -%}{{ render_python_repr(value) }}{%- endif -%}
556
+ +{%- endmacro -%}
557
+ +
558
+ +{%- macro render_xml_enum_value(value) -%}
559
+ +{%- if value is string -%}"{{- value | replace("\\", "\\\\") | replace("\"", "\\\"") -}}"{%- else -%}"{{- render_python_repr(value) | replace("\\", "\\\\") | replace("\"", "\\\"") -}}"{%- endif -%}
560
+ +{%- endmacro -%}
561
+ +
562
+ +{%- macro render_xml_enum(values) -%}
563
+ +{%- for value in values -%}{{ render_xml_enum_value(value) }}{%- if not loop.last -%}|{%- endif -%}{%- endfor -%}
564
+ +{%- endmacro -%}
565
+ +
566
+ +{%- macro render_xml_default_attr(value) -%}
567
+ +{{- " default=" }}{%- if value is string -%}"{{- value | replace("\\", "\\\\") | replace("\"", "\\\"") -}}"{%- else -%}{{ render_xml_value(value) }}{%- endif -%}
568
+ +{%- endmacro -%}
569
+ +
570
+ +{%- macro render_xml_attr(name, value) -%}
571
+ +{{- " " + name + "=" }}{%- if value == "" -%}""{%- else -%}{{ render_xml_value(value) }}{%- endif -%}
572
+ +{%- endmacro -%}
573
+ +
574
+ +{%- macro validate_schema(spec, path, lenient=false, classify=true, in_variant=false) -%}
575
+ +{%- if spec is mapping -%}
576
+ + {%- if not lenient -%}
577
+ + {%- if spec.required is defined -%}
578
+ + {%- if spec.required is string or spec.required is not sequence -%}
579
+ + {{- raise_exception("Schema '" + path + "' has 'required' but it is not a list.") -}}
580
+ + {%- endif -%}
581
+ + {%- if spec.required | length > 0 and not spec.properties and not in_variant -%}
582
+ + {{- raise_exception("Schema '" + path + "' has required fields but no properties object to define them.") -}}
583
+ + {%- endif -%}
584
+ + {%- if spec.properties -%}
585
+ + {%- for required_name in spec.required -%}
586
+ + {%- if required_name not in spec.properties -%}
587
+ + {{- raise_exception("Schema '" + path + "' marks '" + required_name + "' as required, but that property is not defined in properties.") -}}
588
+ + {%- endif -%}
589
+ + {%- endfor -%}
590
+ + {%- endif -%}
591
+ + {%- endif -%}
592
+ + {%- endif -%}
593
+ + {#- renderability classification, piggybacking on this walk (no raises here): -#}
594
+ + {#- constructs the pretty renderer does not fully handle flip RB.ok so the -#}
595
+ + {#- tool falls back to verbatim JSON. Skipped entirely for json presentation. -#}
596
+ + {%- if classify -%}
597
+ + {%- for key, value in spec | items -%}
598
+ + {%- if key == '$ref' -%}
599
+ + {#- llama.cpp's Jinja has no dictionary constructor, so $ref inlining stays -#}
600
+ + {#- template-local by falling back to the exact JSON presentation. -#}
601
+ + {%- set RB.ok = false -%}
602
+ + {%- elif key == '$defs' or key == 'definitions' -%}
603
+ + {%- if value is mapping -%}
604
+ + {%- for dk, dv in value | items -%}
605
+ + {{- validate_schema(dv, path + ".$defs." + dk, true) -}}
606
+ + {%- endfor -%}
607
+ + {%- else -%}{%- set RB.ok = false -%}{%- endif -%}
608
+ + {%- elif key == 'type' -%}
609
+ + {%- if value is mapping -%}{%- set RB.ok = false -%}{%- endif -%}
610
+ + {%- elif key == 'enum' -%}
611
+ + {%- if value is string or value is mapping or value is not sequence -%}{%- set RB.ok = false -%}{%- endif -%}
612
+ + {%- elif key == 'items' -%}
613
+ + {#- any items shape renders: mapping structurally, others via repr detail -#}
614
+ + {%- elif key == 'oneOf' or key == 'anyOf' -%}
615
+ + {%- if value is mapping or value is string or value is not sequence -%}{%- set RB.ok = false -%}{%- endif -%}
616
+ + {%- elif key == 'required' -%}
617
+ + {%- if value and not spec.properties -%}{%- set RB.ok = false -%}{%- endif -%}
618
+ + {%- elif ('|' ~ key ~ '|') in '|description|default|title|examples|properties|patternProperties|additionalProperties|returns|' -%}
619
+ + {%- elif value is mapping -%}
620
+ + {%- for uk, uv in value | items -%}
621
+ + {%- if value_contains_mapping(uv) == 'true' -%}{%- set RB.ok = false -%}{%- endif -%}
622
+ + {%- endfor -%}
623
+ + {%- elif value is sequence and value is not string -%}
624
+ + {%- if value_contains_mapping(value) == 'true' -%}{%- set RB.ok = false -%}{%- endif -%}
625
+ + {%- endif -%}
626
+ + {%- endfor -%}
627
+ + {%- endif -%}
628
+ + {%- if spec.properties -%}
629
+ + {%- for child_name, child_spec in spec.properties | items -%}
630
+ + {{- validate_schema(child_spec, path + "." + child_name, lenient, classify) -}}
631
+ + {%- endfor -%}
632
+ + {%- endif -%}
633
+ + {%- if 'items' in spec -%}{{- validate_schema(spec['items'], path + "[]", lenient, classify) -}}{%- endif -%}
634
+ + {%- if spec.oneOf -%}
635
+ + {%- for variant in spec.oneOf -%}{{- validate_schema(variant, path + ".oneOf[" + (loop.index0 | string) + "]", lenient, classify, true) -}}{%- endfor -%}
636
+ + {%- endif -%}
637
+ + {%- if spec.anyOf -%}
638
+ + {%- for variant in spec.anyOf -%}{{- validate_schema(variant, path + ".anyOf[" + (loop.index0 | string) + "]", lenient, classify, true) -}}{%- endfor -%}
639
+ + {%- endif -%}
640
+ + {%- if spec.additionalProperties is mapping -%}{{- validate_schema(spec.additionalProperties, path + ".additionalProperties", lenient, classify) -}}{%- endif -%}
641
+ + {%- if spec.patternProperties is mapping -%}
642
+ + {%- for pattern, pattern_spec in spec.patternProperties | items -%}
643
+ + {{- validate_schema(pattern_spec, path + ".patternProperties[" + pattern + "]", lenient, classify) -}}
644
+ + {%- endfor -%}
645
+ + {%- endif -%}
646
+ + {%- if spec.returns is mapping -%}{{- validate_schema(spec.returns, path + ".returns", lenient, classify) -}}{%- endif -%}
647
+ +{%- endif -%}
648
+ +{%- endmacro -%}
649
+ +
650
+ +{%- macro validate_tools(tools_list, classify=true) -%}
651
+ +{%- set RB.bad = '|' -%}
652
+ +{%- for tool in tools_list -%}
653
+ + {%- set fn = tool.function if tool.function is defined else tool -%}
654
+ + {%- set RB.ok = true -%}
655
+ + {%- if fn.parameters is defined and fn.parameters is string -%}
656
+ + {{- raise_exception("tool.function.parameters must be a dict, not a JSON string. Parse it before passing to the template.") -}}
657
+ + {%- endif -%}
658
+ + {%- if fn.parameters is not defined or fn.parameters is none -%}
659
+ + {%- if fn.arguments is defined -%}
660
+ + {{- raise_exception("Tool '" + fn.name + "' has 'arguments' instead of 'parameters'. Rename 'arguments' to 'parameters'.") -}}
661
+ + {%- else -%}
662
+ + {{- raise_exception("Tool '" + fn.name + "' is missing required 'parameters' field. Each tool must have a 'parameters' dict with 'type', 'properties', and 'required' keys.") -}}
663
+ + {%- endif -%}
664
+ + {%- endif -%}
665
+ + {{- validate_schema(fn.parameters, "tool." + fn.name + ".parameters", false, classify) -}}
666
+ + {%- if classify -%}
667
+ + {%- if fn.parameters is mapping -%}
668
+ + {#- unknown container-valued keys at the parameters ROOT are never rendered -#}
669
+ + {#- by the pretty path (root extras are dropped) -> verbatim fallback. -#}
670
+ + {%- for rk, rv in fn.parameters | items -%}
671
+ + {%- if rk not in ['type', 'description', 'enum', 'default', 'properties', 'required', 'optional', 'title', 'items', 'oneOf', 'anyOf', 'additionalProperties', 'patternProperties', 'returns', 'examples', '$defs', 'definitions', '$ref'] -%}
672
+ + {%- if rv is mapping or (rv is sequence and rv is not string) -%}{%- set RB.ok = false -%}{%- endif -%}
673
+ + {%- endif -%}
674
+ + {%- endfor -%}
675
+ + {%- else -%}
676
+ + {%- set RB.ok = false -%}
677
+ + {%- endif -%}
678
+ + {%- endif -%}
679
+ + {%- if fn.returns is mapping -%}{{- validate_schema(fn.returns, "tool." + fn.name + ".returns", false, classify) -}}{%- endif -%}
680
+ + {%- if classify and fn.returns is not defined and fn.response is mapping -%}{{- validate_schema(fn.response, "tool." + fn.name + ".response", true) -}}{%- endif -%}
681
+ + {#- unknown container-valued keys at the FUNCTION level are never rendered -> fallback. -#}
682
+ + {%- if classify -%}
683
+ + {%- for fk, fv in fn | items -%}
684
+ + {%- if fk not in ['name', 'description', 'parameters', 'returns', 'response', 'type', 'function'] -%}
685
+ + {%- if fv is mapping or (fv is sequence and fv is not string) -%}{%- set RB.ok = false -%}{%- endif -%}
686
+ + {%- endif -%}
687
+ + {%- endfor -%}
688
+ + {%- endif -%}
689
+ + {%- if not RB.ok -%}{%- set RB.bad = RB.bad ~ loop.index0 ~ '|' -%}{%- endif -%}
690
+ +{%- endfor -%}
691
+ +{%- endmacro -%}
692
+ +
693
+ +{%- macro render_tools_json(tools_list) -%}
694
+ +{{- "<ifm|tools>" }}
695
+ +{%- for tool in tools_list %}
696
+ +{{- "\n" }}
697
+ +{{- tool | tojson }}
698
+ +{%- endfor %}
699
+ +{{- "\n</ifm|tools>" }}
700
+ +{%- endmacro -%}
701
+ +
702
+ +{%- macro render_xml_schema_attrs(spec, include_value_attrs) -%}
703
+ +{%- if spec is mapping -%}
704
+ +{%- set structural_keys = ["type", "description", "enum", "default", "properties", "required", "items", "oneOf", "anyOf", "additionalProperties", "patternProperties", "returns"] -%}
705
+ +{%- if include_value_attrs and spec.enum -%}{{- " enum=" }}{{ render_xml_enum(spec.enum) }}{%- endif -%}
706
+ +{%- if include_value_attrs and spec.default is defined -%}{{ render_xml_default_attr(spec.default) }}{%- endif -%}
707
+ +{%- if spec.additionalProperties is defined and spec.additionalProperties is not mapping -%}{{ render_xml_attr("additionalProperties", spec.additionalProperties) }}{%- endif -%}
708
+ +{%- if spec.patternProperties is defined and spec.patternProperties is not mapping -%}{{ render_xml_attr("patternProperties", spec.patternProperties) }}{%- endif -%}
709
+ +{%- for key, value in spec | items -%}
710
+ + {%- if key not in structural_keys -%}
711
+ +{{ render_xml_attr(key, value) }}
712
+ + {%- endif -%}
713
+ +{%- endfor -%}
714
+ +{%- endif -%}
715
+ +{%- endmacro -%}
716
+ +
717
+ +{%- macro xml_schema_has_children(spec, include_properties, include_description) -%}
718
+ +{%- if spec is not mapping -%}
719
+ +false
720
+ +{%- elif (include_description and spec.description is defined) or (include_properties and spec.properties) or 'items' in spec or spec.oneOf or spec.anyOf or spec.additionalProperties is mapping or spec.patternProperties is mapping or spec.returns is defined -%}
721
+ +true
722
+ +{%- else -%}
723
+ +false
724
+ +{%- endif -%}
725
+ +{%- endmacro -%}
726
+ +
727
+ +{%- macro render_xml_schema_node(tag, spec, include_properties) -%}
728
+ +{%- if spec is mapping -%}
729
+ +{{- "<" + tag + " type=" + render_compact_type(spec) }}{{ render_xml_schema_attrs(spec, true) }}
730
+ +{%- if xml_schema_has_children(spec, include_properties, true) == 'true' -%}
731
+ +{{- ">" }}{{ render_xml_schema_children(spec, include_properties, true) }}{{- "</" + tag + ">" }}
732
+ +{%- else -%}
733
+ +{{- "/>" }}
734
+ +{%- endif -%}
735
+ +{%- else -%}
736
+ +{{- "<" + tag + ">" }}{{ render_xml_value(spec) }}{{- "</" + tag + ">" }}
737
+ +{%- endif -%}
738
+ +{%- endmacro -%}
739
+ +
740
+ +{%- macro render_xml_pattern_property(pattern, spec) -%}
741
+ +{%- if spec is mapping -%}
742
+ +{{- "<patternProperty" }}{{ render_xml_attr("pattern", pattern) }}{{- " type=" + render_compact_type(spec) }}{{ render_xml_schema_attrs(spec, true) }}
743
+ +{%- if xml_schema_has_children(spec, true, true) == 'true' -%}
744
+ +{{- ">" }}{{ render_xml_schema_children(spec, true, true) }}{{- "</patternProperty>" }}
745
+ +{%- else -%}
746
+ +{{- "/>" }}
747
+ +{%- endif -%}
748
+ +{%- else -%}
749
+ +{{- "<patternProperty" }}{{ render_xml_attr("pattern", pattern) }}{{- ">" }}{{ render_xml_value(spec) }}{{- "</patternProperty>" }}
750
+ +{%- endif -%}
751
+ +{%- endmacro -%}
752
+ +
753
+ +{%- macro render_xml_schema_children(spec, include_properties, include_description) -%}
754
+ +{%- if include_description and spec.description is defined -%}{{- "<description>" }}{{ spec.description }}{{- "</description>" }}{%- endif -%}
755
+ +{%- if include_properties and spec.properties -%}
756
+ +{%- for child_name, child_spec in spec.properties | items -%}
757
+ +{{- render_xml_param(child_name, child_spec, spec.required or []) }}
758
+ +{%- endfor -%}
759
+ +{%- endif -%}
760
+ +{%- if 'items' in spec -%}{{ render_xml_schema_node("items", spec['items'], true) }}{%- endif -%}
761
+ +{%- if spec.oneOf -%}
762
+ +{{- "<oneOf>" }}
763
+ +{%- for variant in spec.oneOf -%}{{ render_xml_schema_node("variant", variant, true) }}{%- endfor -%}
764
+ +{{- "</oneOf>" }}
765
+ +{%- endif -%}
766
+ +{%- if spec.anyOf -%}
767
+ +{{- "<anyOf>" }}
768
+ +{%- for variant in spec.anyOf -%}{{ render_xml_schema_node("variant", variant, true) }}{%- endfor -%}
769
+ +{{- "</anyOf>" }}
770
+ +{%- endif -%}
771
+ +{%- if spec.additionalProperties is mapping -%}{{ render_xml_schema_node("additionalProperties", spec.additionalProperties, true) }}{%- endif -%}
772
+ +{%- if spec.patternProperties is mapping -%}
773
+ +{{- "<patternProperties>" }}
774
+ +{%- for pattern, pattern_spec in spec.patternProperties | items -%}{{ render_xml_pattern_property(pattern, pattern_spec) }}{%- endfor -%}
775
+ +{{- "</patternProperties>" }}
776
+ +{%- elif spec.patternProperties is defined -%}<patternProperties>{{ render_xml_value(spec.patternProperties) }}</patternProperties>{%- endif -%}
777
+ +{%- if spec.returns is mapping -%}{{ render_xml_schema_node("returns", spec.returns, true) }}{%- elif spec.returns is defined -%}<returns>{{ render_xml_value(spec.returns) }}</returns>{%- endif -%}
778
+ +{%- endmacro -%}
779
+ +
780
+ +{%- macro render_xml_param(name, spec, required_list) -%}
781
+ +{{- "<param name=" + name + " type=" + render_compact_type(spec) }}
782
+ +{%- if name in (required_list or []) -%}{{- " required=true" }}{%- endif -%}
783
+ +{%- if spec.enum -%}{{- " enum=" }}{{ render_xml_enum(spec.enum) }}{%- endif -%}
784
+ +{%- if spec.default is defined -%}{{ render_xml_default_attr(spec.default) }}{%- endif -%}
785
+ +{{- render_xml_schema_attrs(spec, false) }}
786
+ +{%- if spec.description or xml_schema_has_children(spec, true, false) == 'true' -%}
787
+ +{{- ">" }}
788
+ +{%- if spec.description -%}{{ spec.description }}{%- endif -%}
789
+ +{{- render_xml_schema_children(spec, true, false) }}
790
+ +{{- "</param>" }}
791
+ +{%- else -%}
792
+ +{{- "/>" }}
793
+ +{%- endif -%}
794
+ +{%- endmacro -%}
795
+ +
796
+ +{%- macro render_tools_xml(tools_list) -%}
797
+ +{{- "<ifm|tools>" }}
798
+ +{%- for tool in tools_list -%}
799
+ + {%- set fn = tool.function if tool.function is defined else tool -%}
800
+ + {%- set fnp = namespace(p=fn.parameters) -%}
801
+ +{{- "\n<function name=" + fn.name + ">" }}
802
+ +{%- if fn.description -%}
803
+ +{{- "<description>" }}{{ fn.description }}{{- "</description>" }}
804
+ +{%- endif -%}
805
+ +{{- "<parameters>" }}
806
+ +{%- if fnp.p and fnp.p.properties -%}
807
+ + {%- for pname, pspec in fnp.p.properties | items -%}
808
+ +{{- render_xml_param(pname, pspec, fnp.p.required or []) }}
809
+ + {%- endfor -%}
810
+ +{%- elif fnp.p is mapping and (fnp.p.oneOf or fnp.p.anyOf or 'items' in fnp.p) -%}
811
+ +{{- render_xml_schema_children(fnp.p, true, false) }}
812
+ +{%- endif -%}
813
+ +{{- "</parameters>" }}
814
+ +{%- set fn_ret = fn.returns if fn.returns is defined else fn.response -%}
815
+ +{%- if fn_ret is mapping -%}{{ render_xml_schema_node("returns", fn_ret, true) }}{%- elif fn_ret is defined -%}<returns>{{ render_xml_value(fn_ret) }}</returns>{%- endif -%}
816
+ +{{- "</function>" }}
817
+ +{%- endfor -%}
818
+ +{{- "\n</ifm|tools>" }}
819
+ +{%- endmacro -%}
820
+ +
821
+ +{%- macro render_markdown_literal(value) -%}
822
+ +{%- if value is string and value == "" -%}""
823
+ +{%- elif value is string -%}`{{ value | replace("\n", "\\n") }}`
824
+ +{%- else -%}`{{ render_python_repr(value) }}`
825
+ +{%- endif -%}
826
+ +{%- endmacro -%}
827
+ +
828
+ +{%- macro render_allowed_values(values) -%}
829
+ +{%- for value in values -%}{{ render_markdown_literal(value) }}{% if not loop.last %}, {% endif %}{%- endfor -%}
830
+ +{%- endmacro -%}
831
+ +
832
+ +{%- macro render_markdown_value(value) -%}
833
+ +{%- if value is string and value == "" -%}""{%- elif value is string -%}{{ value }}{%- else -%}{{ render_python_repr(value) }}{%- endif -%}
834
+ +{%- endmacro -%}
835
+ +
836
+ +{%- macro render_markdown_detail(indent, label, value) -%}
837
+ +{{- "\n" + indent + " - " + label + ": " }}{{ render_markdown_value(value) }}
838
+ +{%- endmacro -%}
839
+ +
840
+ +{%- macro render_markdown_metadata_detail(label, value) -%}
841
+ +{{- "\n- " + label + ": " }}{{ render_markdown_value(value) }}
842
+ +{%- endmacro -%}
843
+ +
844
+ +{%- macro render_markdown_schema_annotations(spec, indent, include_value_details) -%}
845
+ +{%- if include_value_details and spec.description is defined -%}{{ render_markdown_detail(indent, "Description", spec.description | replace("\n", "\n" + indent + " ")) }}{%- endif -%}
846
+ +{%- if include_value_details and spec.enum is defined -%}{{- "\n" + indent + " - Allowed values: " }}{{ render_allowed_values(spec.enum) }}{%- endif -%}
847
+ +{%- if include_value_details and spec.default is defined -%}{{- "\n" + indent + " - Default: " }}{{ render_markdown_literal(spec.default) }}{%- endif -%}
848
+ +{%- if spec.additionalProperties is defined -%}
849
+ + {%- if spec.additionalProperties is mapping -%}
850
+ +{{- "\n" + indent + " - Additional properties *(" + render_markdown_type(spec.additionalProperties) + ")*" }}
851
+ +{{- render_markdown_schema_details(spec.additionalProperties, indent + " ", true) }}
852
+ + {%- else -%}
853
+ +{{ render_markdown_detail(indent, "Additional properties", spec.additionalProperties) }}
854
+ + {%- endif -%}
855
+ +{%- endif -%}
856
+ +{%- endmacro -%}
857
+ +
858
+ +{%- macro render_markdown_metadata_annotations(spec) -%}
859
+ +{%- if spec.description is defined -%}{{ render_markdown_metadata_detail("Description", spec.description | replace("\n", "\n ")) }}{%- endif -%}
860
+ +{%- if spec.enum is defined -%}{{- "\n- Allowed values: " }}{{ render_allowed_values(spec.enum) }}{%- endif -%}
861
+ +{%- if spec.default is defined -%}{{- "\n- Default: " }}{{ render_markdown_literal(spec.default) }}{%- endif -%}
862
+ +{%- if spec.additionalProperties is defined -%}
863
+ + {%- if spec.additionalProperties is mapping -%}
864
+ +{{- "\n- Additional properties *(" + render_markdown_type(spec.additionalProperties) + ")*" }}
865
+ +{{- render_markdown_schema_details(spec.additionalProperties, "", true) }}
866
+ + {%- else -%}
867
+ +{{ render_markdown_metadata_detail("Additional properties", spec.additionalProperties) }}
868
+ + {%- endif -%}
869
+ +{%- endif -%}
870
+ +{%- endmacro -%}
871
+ +
872
+ +{%- macro render_markdown_schema_extras(spec, indent) -%}
873
+ +{%- set rendered_keys = ["type", "description", "enum", "default", "properties", "required", "items", "oneOf", "anyOf", "additionalProperties", "patternProperties", "returns"] -%}
874
+ +{%- for key, value in spec | items -%}
875
+ + {%- if key not in rendered_keys -%}
876
+ +{{- "\n" + indent + " - " + key + ": " }}{{ render_markdown_value(value) }}
877
+ + {%- endif -%}
878
+ +{%- endfor -%}
879
+ +{%- endmacro -%}
880
+ +
881
+ +{%- macro render_markdown_metadata_extras(spec) -%}
882
+ +{%- set rendered_keys = ["type", "description", "enum", "default", "properties", "required", "items", "oneOf", "anyOf", "additionalProperties", "patternProperties", "returns"] -%}
883
+ +{%- for key, value in spec | items -%}
884
+ + {%- if key not in rendered_keys -%}
885
+ +{{- "\n- " + key + ": " }}{{ render_markdown_value(value) }}
886
+ + {%- endif -%}
887
+ +{%- endfor -%}
888
+ +{%- endmacro -%}
889
+ +
890
+ +{%- macro markdown_schema_has_extra(spec) -%}
891
+ +{%- set rendered_keys = ["type", "description", "enum", "default", "properties", "required", "items", "oneOf", "anyOf", "additionalProperties", "patternProperties", "returns"] -%}
892
+ +{%- set found = namespace(value='false') -%}
893
+ +{%- for key, value in spec | items -%}
894
+ + {%- if key not in rendered_keys -%}{%- set found.value = 'true' -%}{%- endif -%}
895
+ +{%- endfor -%}
896
+ +{{- found.value -}}
897
+ +{%- endmacro -%}
898
+ +
899
+ +{%- macro markdown_parameter_schema_has_details(spec) -%}
900
+ +{%- if spec.description is defined or spec.enum is defined or spec.default is defined or spec.additionalProperties is defined or spec.patternProperties is defined or 'items' in spec or spec.oneOf or spec.anyOf or spec.returns is defined or markdown_schema_has_extra(spec) == 'true' -%}
901
+ +true
902
+ +{%- else -%}
903
+ +false
904
+ +{%- endif -%}
905
+ +{%- endmacro -%}
906
+ +
907
+ +{%- macro render_markdown_schema_structure(spec, indent, include_properties) -%}
908
+ +{%- if include_properties and spec.properties -%}
909
+ + {%- for child_name, child_spec in spec.properties | items -%}
910
+ +{{- render_markdown_param(child_name, child_spec, spec.required or [], indent + " ") }}
911
+ + {%- endfor -%}
912
+ +{%- endif -%}
913
+ +{%- if 'items' in spec and spec['items'] is mapping -%}
914
+ +{{- "\n" + indent + " - Items *(" + render_markdown_type(spec['items']) + ")*" }}
915
+ +{{- render_markdown_schema_details(spec['items'], indent + " ", true) }}
916
+ +{%- elif 'items' in spec -%}
917
+ +{{ render_markdown_detail(indent, "Items", spec['items']) }}
918
+ +{%- endif -%}
919
+ +{%- if spec.oneOf -%}
920
+ +{{- "\n" + indent + " - oneOf:" }}
921
+ + {%- for variant in spec.oneOf -%}
922
+ +{{- "\n" + indent + " - Variant " }}{{ loop.index }}{{- " *(" + render_markdown_type(variant) + ")*" }}
923
+ +{{- render_markdown_schema_details(variant, indent + " ", true) }}
924
+ + {%- endfor -%}
925
+ +{%- endif -%}
926
+ +{%- if spec.anyOf -%}
927
+ +{{- "\n" + indent + " - anyOf:" }}
928
+ + {%- for variant in spec.anyOf -%}
929
+ +{{- "\n" + indent + " - Variant " }}{{ loop.index }}{{- " *(" + render_markdown_type(variant) + ")*" }}
930
+ +{{- render_markdown_schema_details(variant, indent + " ", true) }}
931
+ + {%- endfor -%}
932
+ +{%- endif -%}
933
+ +{%- if spec.patternProperties is mapping -%}
934
+ +{{- "\n" + indent + " - Pattern properties:" }}
935
+ + {%- for pattern, pattern_spec in spec.patternProperties | items -%}
936
+ + {%- if pattern_spec is mapping -%}
937
+ +{{- "\n" + indent + " - `" + pattern + "` *(" + render_markdown_type(pattern_spec) + ")*" }}
938
+ +{{- render_markdown_schema_details(pattern_spec, indent + " ", true) }}
939
+ + {%- else -%}
940
+ +{{- "\n" + indent + " - `" + pattern + "`: " }}{{ render_markdown_value(pattern_spec) }}
941
+ + {%- endif -%}
942
+ + {%- endfor -%}
943
+ +{%- elif spec.patternProperties is defined -%}
944
+ +{{ render_markdown_detail(indent, "Pattern properties", spec.patternProperties) }}
945
+ +{%- endif -%}
946
+ +{%- if spec.returns is mapping -%}
947
+ +{{- "\n" + indent + " - Returns *(" + render_markdown_type(spec.returns) + ")*" }}
948
+ +{{- render_markdown_schema_details(spec.returns, indent + " ", true) }}
949
+ +{%- elif spec.returns is defined -%}
950
+ +{{ render_markdown_detail(indent, "Returns", spec.returns) }}
951
+ +{%- endif -%}
952
+ +{%- endmacro -%}
953
+ +
954
+ +{%- macro render_markdown_schema_details(spec, indent, include_value_details) -%}
955
+ +{%- if spec is mapping -%}
956
+ +{{- render_markdown_schema_annotations(spec, indent, include_value_details) }}
957
+ +{{- render_markdown_schema_structure(spec, indent, true) }}
958
+ +{{- render_markdown_schema_extras(spec, indent) }}
959
+ +{%- elif spec is not boolean -%}
960
+ +{{- "\n" + indent + " - Value: " }}{{ render_markdown_literal(spec) }}
961
+ +{%- endif -%}
962
+ +{%- endmacro -%}
963
+ +
964
+ +{%- macro render_markdown_parameter_schema(spec) -%}
965
+ +{%- if spec is mapping -%}
966
+ +{{- render_markdown_metadata_annotations(spec) }}
967
+ +{%- if 'items' in spec and spec['items'] is mapping -%}
968
+ +{{- "\n- Items *(" + render_markdown_type(spec['items']) + ")*" }}
969
+ +{{- render_markdown_schema_details(spec['items'], "", true) }}
970
+ +{%- elif 'items' in spec -%}
971
+ +{{ render_markdown_metadata_detail("Items", spec['items']) }}
972
+ +{%- endif -%}
973
+ +{%- if spec.oneOf -%}
974
+ +{{- "\n- oneOf:" }}
975
+ + {%- for variant in spec.oneOf -%}
976
+ +{{- "\n - Variant " }}{{ loop.index }}{{- " *(" + render_markdown_type(variant) + ")*" }}
977
+ +{{- render_markdown_schema_details(variant, " ", true) }}
978
+ + {%- endfor -%}
979
+ +{%- endif -%}
980
+ +{%- if spec.anyOf -%}
981
+ +{{- "\n- anyOf:" }}
982
+ + {%- for variant in spec.anyOf -%}
983
+ +{{- "\n - Variant " }}{{ loop.index }}{{- " *(" + render_markdown_type(variant) + ")*" }}
984
+ +{{- render_markdown_schema_details(variant, " ", true) }}
985
+ + {%- endfor -%}
986
+ +{%- endif -%}
987
+ +{%- if spec.patternProperties is mapping -%}
988
+ +{{- "\n- Pattern properties:" }}
989
+ + {%- for pattern, pattern_spec in spec.patternProperties | items -%}
990
+ + {%- if pattern_spec is mapping -%}
991
+ +{{- "\n - `" + pattern + "` *(" + render_markdown_type(pattern_spec) + ")*" }}
992
+ +{{- render_markdown_schema_details(pattern_spec, " ", true) }}
993
+ + {%- else -%}
994
+ +{{- "\n - `" + pattern + "`: " }}{{ render_markdown_value(pattern_spec) }}
995
+ + {%- endif -%}
996
+ + {%- endfor -%}
997
+ +{%- elif spec.patternProperties is defined -%}
998
+ +{{ render_markdown_metadata_detail("Pattern properties", spec.patternProperties) }}
999
+ +{%- endif -%}
1000
+ +{%- if spec.returns is mapping -%}
1001
+ +{{- "\n- Returns *(" + render_markdown_type(spec.returns) + ")*" }}
1002
+ +{{- render_markdown_schema_details(spec.returns, "", true) }}
1003
+ +{%- elif spec.returns is defined -%}
1004
+ +{{ render_markdown_metadata_detail("Returns", spec.returns) }}
1005
+ +{%- endif -%}
1006
+ +{{- render_markdown_metadata_extras(spec) }}
1007
+ +{%- endif -%}
1008
+ +{%- endmacro -%}
1009
+ +
1010
+ +{%- macro render_markdown_param(name, spec, required_list, indent) -%}
1011
+ +{{- "\n" + indent + "- `" + name + "` *(" + render_markdown_type(spec) }}
1012
+ +{%- if name in (required_list or []) -%}{{- ", required" }}{%- endif -%}
1013
+ +{{- ")*" }}
1014
+ +{%- if spec.description -%}{{- " - " + spec.description | replace("\n", "\n" + indent + " ") }}{%- endif -%}
1015
+ +{%- if spec.enum -%}
1016
+ +{{- "\n" + indent + " - Allowed values: " }}{{ render_allowed_values(spec.enum) }}
1017
+ +{%- endif -%}
1018
+ +{%- if spec.default is defined -%}
1019
+ +{{- "\n" + indent + " - Default: " }}{{ render_markdown_literal(spec.default) }}
1020
+ +{%- endif -%}
1021
+ +{{- render_markdown_schema_details(spec, indent, false) }}
1022
+ +{%- endmacro -%}
1023
+ +
1024
+ +{%- macro render_tools_markdown(tools_list) -%}
1025
+ +{{- "<ifm|tools>" }}
1026
+ +{%- for tool in tools_list -%}
1027
+ + {%- set fn = tool.function if tool.function is defined else tool -%}
1028
+ + {%- set fnp = namespace(p=fn.parameters) -%}
1029
+ +{{- "\n## " + fn.name }}
1030
+ +{%- if fn.description -%}
1031
+ +{{- "\n" + fn.description }}
1032
+ +{%- endif -%}
1033
+ +{{- "\n\n**Parameters**" }}
1034
+ +{%- if fnp.p and fnp.p.properties -%}
1035
+ + {%- for pname, pspec in fnp.p.properties | items -%}
1036
+ +{{- render_markdown_param(pname, pspec, fnp.p.required or [], "") }}
1037
+ + {%- endfor -%}
1038
+ +{%- elif fnp.p is mapping and (fnp.p.oneOf or fnp.p.anyOf or 'items' in fnp.p) -%}
1039
+ +{{- render_markdown_parameter_schema(fnp.p) }}
1040
+ +{%- else -%}
1041
+ +{{- "\n- None" }}
1042
+ +{%- endif -%}
1043
+ +{%- set fn_ret = fn.returns if fn.returns is defined else fn.response -%}
1044
+ +{%- if fn_ret is mapping -%}
1045
+ +{{- "\n\n**Returns**" }}
1046
+ +{{- "\n- Return *(" + render_markdown_type(fn_ret) + ")*" }}
1047
+ +{{- render_markdown_schema_details(fn_ret, "", true) }}
1048
+ +{%- elif fn_ret is defined -%}
1049
+ +{{- "\n\n**Returns**\n- " }}{{ render_markdown_value(fn_ret) }}
1050
+ +{%- endif -%}
1051
+ +{%- if not loop.last -%}{{- "\n" }}{%- endif -%}
1052
+ +{%- endfor -%}
1053
+ +{{- "\n</ifm|tools>" }}
1054
+ +{%- endmacro -%}
1055
+ +
1056
+ +{%- macro render_tool_presentation(tools_list, fmt) -%}
1057
+ +{%- if fmt == 'json' -%}
1058
+ +{{- render_tools_json(tools_list) }}
1059
+ +{%- elif RB.bad != '|' -%}
1060
+ +{#- some tool uses constructs the pretty renderers cannot represent (verdicts -#}
1061
+ +{#- computed during validate_tools): render the WHOLE toolset exactly as the -#}
1062
+ +{#- json presentation would, so the block stays uniform and model-familiar. -#}
1063
+ +{{- render_tools_json(tools_list) }}
1064
+ +{%- elif fmt == 'xml' -%}
1065
+ +{{- render_tools_xml(tools_list) }}
1066
+ +{%- elif fmt == 'markdown' -%}
1067
+ +{{- render_tools_markdown(tools_list) }}
1068
+ +{%- else -%}
1069
+ +{{- raise_exception("Unsupported tool_presentation_format: '" + fmt + "'. Supported formats: json, xml, markdown.") }}
1070
+ +{%- endif -%}
1071
+ +{%- endmacro -%}
1072
+ +
1073
+ +{%- macro render_call_instructions(fmt) -%}
1074
+ +{%- if fmt == 'json' -%}
1075
+ +{{- "Wrap all tool calls in a single <ifm|tool_calls></ifm|tool_calls> block. For each call, emit one JSON object with the function name and arguments on the same line inside <ifm|tool_call></ifm|tool_call> tags:\n\n<ifm|tool_calls>\n<ifm|tool_call>{\"name\": <function-name>, \"arguments\": <args-json-object>}</ifm|tool_call>\n</ifm|tool_calls>" }}
1076
+ +{%- elif fmt == 'xml' -%}
1077
+ +{{- "Wrap all tool calls in a single <ifm|tool_calls></ifm|tool_calls> block. For each call, write the function name at the start of <ifm|tool_call>, followed by paired <ifm|arg_key> and <ifm|arg_value> tags for each argument:\n\n<ifm|tool_calls>\n<ifm|tool_call>$FUNCTION_NAME\n<ifm|arg_key>$PARAMETER_NAME</ifm|arg_key>\n<ifm|arg_value>$PARAMETER_VALUE</ifm|arg_value>\n...\n</ifm|tool_call>\n</ifm|tool_calls>\n\nString and scalar parameters should be written as plain text. Array and object parameters should be written as JSON literals." }}
1078
+ +{%- elif fmt == 'xml_typed' -%}
1079
+ +{{- "Wrap all tool calls in a single <ifm|tool_calls></ifm|tool_calls> block. For each call, write the function name at the start of <ifm|tool_call>, followed by <ifm|arg_key>, <ifm|arg_type>, and <ifm|arg_value> tags for each argument:\n\n<ifm|tool_calls>\n<ifm|tool_call>$FUNCTION_NAME\n<ifm|arg_key>$PARAMETER_NAME</ifm|arg_key>\n<ifm|arg_type>$ARGUMENT_TYPE</ifm|arg_type>\n<ifm|arg_value>$PARAMETER_VALUE</ifm|arg_value>\n...\n</ifm|tool_call>\n</ifm|tool_calls>\n\nUse the parameter type shown in the tool definition. If that type contains anyOf or oneOf, use the actual argument value type instead. String and scalar parameters should be written as plain text. Array and object parameters should be written as JSON literals." }}
1080
+ +{%- else -%}
1081
+ +{{- raise_exception("Unsupported tool_call_format: '" + fmt + "'. Supported formats: json, xml, xml_typed.") }}
1082
+ +{%- endif -%}
1083
+ +{%- endmacro -%}
1084
+ +
1085
+ +{%- macro render_system_with_tools(tools_list, system_content, presentation_fmt, call_fmt) -%}
1086
+ +{{- "<|ifm|im_start|>system\n# Tools\nYou may call one or more tools to assist with the user query.\n\nAvailable tools are:\n\n" }}
1087
+ +{{- render_tool_presentation(tools_list, presentation_fmt) }}
1088
+ +{{- "\n\nWhen calling tools, you MUST follow the tool-call format below:\n\n" }}
1089
+ +{{- render_call_instructions(call_fmt) }}
1090
+ +{%- if system_content -%}
1091
+ +{{- "\n\n" + system_content }}
1092
+ +{%- endif -%}
1093
+ +{{- "<|ifm|im_end|>" }}
1094
+ +{%- endmacro -%}
1095
+ +
1096
+ +{%- macro render_argument_value(value) -%}
1097
+ +{%- if value is string -%}{{- value -}}{%- else -%}{{- value | tojson -}}{%- endif -%}
1098
+ +{%- endmacro -%}
1099
+ +
1100
+ +{%- macro render_value_type(value) -%}
1101
+ +{%- if value is none -%}null
1102
+ +{%- elif value is boolean -%}boolean
1103
+ +{%- elif value is integer -%}integer
1104
+ +{%- elif value is number -%}number
1105
+ +{%- elif value is string -%}string
1106
+ +{%- elif value is mapping -%}object
1107
+ +{%- elif value is sequence -%}array
1108
+ +{%- else -%}any
1109
+ +{%- endif -%}
1110
+ +{%- endmacro -%}
1111
+ +
1112
+ +{%- macro schema_has_combinator(spec) -%}
1113
+ +{%- if spec.oneOf or spec.anyOf -%}
1114
+ +true
1115
+ +{%- elif spec.type is defined and spec.type is sequence and spec.type is not string and spec.type | length > 1 -%}
1116
+ +true
1117
+ +{%- elif spec.type == "array" and 'items' in spec -%}
1118
+ +{{- schema_has_combinator(spec['items']) -}}
1119
+ +{%- elif spec.properties -%}
1120
+ + {%- set found = namespace(value='false') -%}
1121
+ + {%- for child_name, child_spec in spec.properties | items -%}
1122
+ + {%- if schema_has_combinator(child_spec) == 'true' -%}
1123
+ + {%- set found.value = 'true' -%}
1124
+ + {%- endif -%}
1125
+ + {%- endfor -%}
1126
+ +{{- found.value -}}
1127
+ +{%- else -%}
1128
+ +false
1129
+ +{%- endif -%}
1130
+ +{%- endmacro -%}
1131
+ +
1132
+ +{%- macro render_arg_type(tools_list, tool_name, arg_name, value) -%}
1133
+ +{%- set found = namespace(type='any') -%}
1134
+ +{%- for tool in tools_list -%}
1135
+ + {%- set fn = tool.function if tool.function is defined else tool -%}
1136
+ + {%- if fn.name == tool_name and fn.parameters and fn.parameters.properties and arg_name in fn.parameters.properties -%}
1137
+ + {%- set spec = fn.parameters.properties[arg_name] -%}
1138
+ + {%- if spec is mapping and spec['$ref'] is string -%}
1139
+ + {%- set found.type = render_value_type(value) -%}
1140
+ + {%- elif schema_has_combinator(spec) == 'true' -%}
1141
+ + {%- set found.type = render_value_type(value) -%}
1142
+ + {%- else -%}
1143
+ + {%- set found.type = render_compact_type(spec) -%}
1144
+ + {%- endif -%}
1145
+ + {%- endif -%}
1146
+ +{%- endfor -%}
1147
+ +{{- found.type -}}
1148
+ +{%- endmacro -%}
1149
+ +
1150
+ +{%- macro render_tool_calls_block(tool_calls, fmt, tools_list) -%}
1151
+ +{{- "<ifm|tool_calls>" }}
1152
+ +{%- for raw_tool_call in tool_calls -%}
1153
+ + {%- set tool_call = raw_tool_call.function if raw_tool_call.function else raw_tool_call -%}
1154
+ + {%- if tool_call.arguments is string -%}
1155
+ + {{- raise_exception("tool_call.arguments must be a dict, not a JSON string. Parse it before passing to the template.") -}}
1156
+ + {%- endif -%}
1157
+ + {%- if fmt == 'json' -%}
1158
+ +{{- "\n<ifm|tool_call>{\"name\": \"" + tool_call.name + "\", \"arguments\": " }}{{ tool_call.arguments | tojson }}{{- "}</ifm|tool_call>" }}
1159
+ + {%- elif fmt == 'xml' or fmt == 'xml_typed' -%}
1160
+ +{{- "\n<ifm|tool_call>" + tool_call.name + "\n" }}
1161
+ + {%- for key, value in tool_call.arguments | items -%}
1162
+ +{{- "<ifm|arg_key>" + key + "</ifm|arg_key>\n" }}
1163
+ +{%- if fmt == 'xml_typed' -%}
1164
+ +{{- "<ifm|arg_type>" + render_arg_type(tools_list, tool_call.name, key, value) + "</ifm|arg_type>\n" }}
1165
+ +{%- endif -%}
1166
+ +{{- "<ifm|arg_value>" }}{{ render_argument_value(value) }}{{- "</ifm|arg_value>\n" }}
1167
+ + {%- endfor -%}
1168
+ +{{- "</ifm|tool_call>" }}
1169
+ + {%- else -%}
1170
+ + {{- raise_exception("Unsupported tool_call_format: '" + fmt + "'. Supported formats: json, xml, xml_typed.") -}}
1171
+ + {%- endif -%}
1172
+ +{%- endfor -%}
1173
+ +{{- "\n</ifm|tool_calls>" }}
1174
+ +{%- endmacro -%}
1175
+ +
1176
+ +{%- macro render_tool_response_messages(raw_content) -%}
1177
+ +{%- if raw_content is string -%}
1178
+ +{{- '<|ifm|im_start|>tool\n' + raw_content + '<|ifm|im_end|>' }}
1179
+ +{%- elif raw_content is sequence and raw_content is not string and raw_content is not mapping -%}
1180
+ + {%- if raw_content | length == 0 -%}
1181
+ + {{- raise_exception("tool message content list must not be empty.") -}}
1182
+ + {%- endif -%}
1183
+ +{{- '<|ifm|im_start|>tool\n' -}}
1184
+ + {%- for item in raw_content -%}
1185
+ + {%- if not loop.first -%}{{- '\n' -}}{%- endif -%}
1186
+ + {%- if item is string -%}
1187
+ +{{- item -}}
1188
+ + {%- elif item is mapping and item.text is string -%}
1189
+ +{{- item.text -}}
1190
+ + {%- else -%}
1191
+ +{{- (item | tojson) -}}
1192
+ + {%- endif -%}
1193
+ + {%- endfor -%}
1194
+ +{{- '<|ifm|im_end|>' -}}
1195
+ +{%- else -%}
1196
+ +{{- '<|ifm|im_start|>tool\n' }}{{ raw_content | tojson }}{{- '<|ifm|im_end|>' }}
1197
+ +{%- endif -%}
1198
+ +{%- endmacro -%}
1199
+ +
1200
+ +{%- set available_tools = tools if tools else [] -%}
1201
+ +{%- if (not available_tools) and messages[0].role == 'system' and messages[0].get('tools') -%}
1202
+ + {%- set available_tools = messages[0]['tools'] -%}
1203
+ +{%- endif -%}
1204
+ +{%- if available_tools -%}
1205
+ + {{- validate_tools(available_tools, tool_presentation_fmt != 'json') }}
1206
+ + {%- set system_content = '' -%}
1207
+ + {%- if messages[0].role == 'system' and messages[0].content -%}
1208
+ + {%- set system_content = messages[0].content -%}
1209
+ + {%- endif -%}
1210
+ + {{- render_system_with_tools(available_tools, system_content, tool_presentation_fmt, tool_call_fmt) }}
1211
+ +{%- else -%}
1212
+ + {%- if messages[0].role == 'system' -%}
1213
+ + {{- '<|ifm|im_start|>system\n' + messages[0].content + '<|ifm|im_end|>' }}
1214
+ + {%- endif -%}
1215
+ +{%- endif -%}
1216
+ +
1217
+ +{%- for message in messages -%}
1218
+ + {%- if message.content is string -%}
1219
+ + {%- set content = message.content -%}
1220
+ + {%- else -%}
1221
+ + {%- set content = '' -%}
1222
+ + {%- endif -%}
1223
+ + {%- if (message.role == "user") or (message.role == "system" and not loop.first) -%}
1224
+ + {{- '<|ifm|im_start|>' + message.role + '\n' + content + '<|ifm|im_end|>' }}
1225
+ + {%- elif message.role == "assistant" -%}
1226
+ + {%- set thinking_content = '' -%}
1227
+ + {%- set think_tag = 'ifm|think' -%}
1228
+ + {%- if message.think is defined and message.think is string -%}
1229
+ + {%- set thinking_content = message.think -%}
1230
+ + {%- set think_tag = 'ifm|think' -%}
1231
+ + {%- elif message.think_fast is defined and message.think_fast is string -%}
1232
+ + {%- set thinking_content = message.think_fast -%}
1233
+ + {%- set think_tag = 'ifm|think_fast' -%}
1234
+ + {%- elif message.think_faster is defined and message.think_faster is string -%}
1235
+ + {%- set thinking_content = message.think_faster -%}
1236
+ + {%- set think_tag = 'ifm|think_faster' -%}
1237
+ + {%- elif message.reasoning_content is defined and message.reasoning_content is string -%}
1238
+ + {%- set thinking_content = message.reasoning_content -%}
1239
+ + {%- set think_tag = 'ifm|think' -%}
1240
+ + {%- elif message.reasoning is defined and message.reasoning is string -%}
1241
+ + {%- set thinking_content = message.reasoning -%}
1242
+ + {%- set think_tag = 'ifm|think' -%}
1243
+ + {%- else -%}
1244
+ + {%- if '</ifm|think>' in content -%}
1245
+ + {%- set thinking_content = content.split('</ifm|think>')[0].rstrip('\n').split('<ifm|think>')[-1].lstrip('\n') -%}
1246
+ + {%- set content = content.split('</ifm|think>')[-1].lstrip('\n') -%}
1247
+ + {%- set think_tag = 'ifm|think' -%}
1248
+ + {%- elif '</ifm|think_fast>' in content -%}
1249
+ + {%- set thinking_content = content.split('</ifm|think_fast>')[0].rstrip('\n').split('<ifm|think_fast>')[-1].lstrip('\n') -%}
1250
+ + {%- set content = content.split('</ifm|think_fast>')[-1].lstrip('\n') -%}
1251
+ + {%- set think_tag = 'ifm|think_fast' -%}
1252
+ + {%- elif '</ifm|think_faster>' in content -%}
1253
+ + {%- set thinking_content = content.split('</ifm|think_faster>')[0].rstrip('\n').split('<ifm|think_faster>')[-1].lstrip('\n') -%}
1254
+ + {%- set content = content.split('</ifm|think_faster>')[-1].lstrip('\n') -%}
1255
+ + {%- set think_tag = 'ifm|think_faster' -%}
1256
+ + {%- endif -%}
1257
+ + {%- endif -%}
1258
+ + {{- '<|ifm|im_start|>' + message.role }}
1259
+ + {% generation %}
1260
+ + {%- if think_tag -%}
1261
+ + {%- if thinking_content -%}
1262
+ + {{- '<' + think_tag + '>\n' + thinking_content + '\n</' + think_tag + '>\n' + content.lstrip('\n') }}
1263
+ + {%- else -%}
1264
+ + {{- '<' + think_tag + '>\n</' + think_tag + '>\n' + content.lstrip('\n') }}
1265
+ + {%- endif -%}
1266
+ + {%- else -%}
1267
+ + {{- content }}
1268
+ + {%- endif -%}
1269
+ + {%- if message.tool_calls -%}
1270
+ + {%- if content -%}
1271
+ + {{- '\n' }}
1272
+ + {%- endif -%}
1273
+ + {{- render_tool_calls_block(message.tool_calls, tool_call_fmt, available_tools) }}
1274
+ + {%- endif -%}
1275
+ + {{- '<|ifm|im_end|>' -}}
1276
+ + {%- endgeneration -%}
1277
+ + {%- elif message.role == "tool" -%}
1278
+ + {{- render_tool_response_messages(message.content) }}
1279
+ + {%- endif -%}
1280
+ +{%- endfor -%}
1281
+ +{%- if add_generation_prompt -%}
1282
+ + {%- set effort = reasoning_effort | default('high') -%}
1283
+ + {%- if enable_thinking is defined and enable_thinking is false -%}
1284
+ + {{- '<|ifm|im_start|>assistant\n<ifm|think>\n</ifm|think>\n' }}
1285
+ + {%- elif effort == 'high' -%}
1286
+ + {{- '<|ifm|im_start|>assistant\n<ifm|think>\n' }}
1287
+ + {%- elif effort == 'medium' -%}
1288
+ + {{- '<|ifm|im_start|>assistant\n<ifm|think_fast>\n' }}
1289
+ + {%- elif effort == 'low' -%}
1290
+ + {{- '<|ifm|im_start|>assistant\n<ifm|think_faster>\n' }}
1291
+ + {%- else -%}
1292
+ + {{- raise_exception("Unsupported reasoning_effort: '" + effort + "'. Supported values: high, medium, low.") -}}
1293
+ + {%- endif -%}
1294
+ +{%- endif -%}
1295
+ diff --git a/src/llama-arch.cpp b/src/llama-arch.cpp
1296
+ index 3f1c8d4..7ebfdff 100644
1297
+ --- a/src/llama-arch.cpp
1298
+ +++ b/src/llama-arch.cpp
1299
+ @@ -139,6 +139,7 @@ static const std::map<llm_arch, const char *> LLM_ARCH_NAMES = {
1300
+ { LLM_ARCH_MISTRAL3, "mistral3" },
1301
+ { LLM_ARCH_EAGLE3, "eagle3" },
1302
+ { LLM_ARCH_DFLASH, "dflash" },
1303
+ + { LLM_ARCH_K2_HORIZON, "k2-horizon" },
1304
+ { LLM_ARCH_MISTRAL4, "mistral4" },
1305
+ { LLM_ARCH_PADDLEOCR, "paddleocr" },
1306
+ { LLM_ARCH_MIMO2, "mimo2" },
1307
+ @@ -380,6 +381,10 @@ static const std::map<llm_kv, const char *> LLM_KV_NAMES = {
1308
+ { LLM_KV_XIELU_BETA, "xielu.beta" },
1309
+ { LLM_KV_XIELU_EPS, "xielu.eps" },
1310
+
1311
+ + // K2 Horizon MoVA
1312
+ + { LLM_KV_ATTENTION_VALUE_EXPERT_COUNT, "%s.attention.value_expert_count"},
1313
+ + { LLM_KV_ATTENTION_VALUE_EXPERT_USED_COUNT, "%s.attention.value_expert_used_count"},
1314
+ +
1315
+ // deprecated
1316
+ { LLM_KV_TOKENIZER_PREFIX_ID, "tokenizer.ggml.prefix_token_id" },
1317
+ { LLM_KV_TOKENIZER_SUFFIX_ID, "tokenizer.ggml.suffix_token_id" },
1318
+ @@ -658,6 +663,8 @@ static const std::map<llm_tensor, const char *> LLM_TENSOR_NAMES = {
1319
+ { LLM_TENSOR_HC_ATTN_FN, "blk.%d.hc_attn_fn" },
1320
+ { LLM_TENSOR_HC_ATTN_SCALE, "blk.%d.hc_attn_scale" },
1321
+ { LLM_TENSOR_HC_FFN_BASE, "blk.%d.hc_ffn_base" },
1322
+ + { LLM_TENSOR_ATTN_V_GATE, "blk.%d.attn_v_gate"},
1323
+ + { LLM_TENSOR_ATTN_V_EXPS, "blk.%d.attn_v_exps"},
1324
+ { LLM_TENSOR_HC_FFN_FN, "blk.%d.hc_ffn_fn" },
1325
+ { LLM_TENSOR_HC_FFN_SCALE, "blk.%d.hc_ffn_scale" },
1326
+ { LLM_TENSOR_FFN_GATE_TID2EID, "blk.%d.ffn_gate_tid2eid" },
1327
+ @@ -947,6 +954,9 @@ static const std::map<llm_tensor, llm_tensor_info> LLM_TENSOR_INFOS = {
1328
+ {LLM_TENSOR_NEXTN_EH_PROJ, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
1329
+ {LLM_TENSOR_NEXTN_E_PROJ, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
1330
+ {LLM_TENSOR_NEXTN_H_PROJ, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
1331
+ + // K2 Horizon MoVA
1332
+ + {LLM_TENSOR_ATTN_V_GATE, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
1333
+ + {LLM_TENSOR_ATTN_V_EXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
1334
+ {LLM_TENSOR_NEXTN_EMBED_TOKENS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_GET_ROWS}},
1335
+ {LLM_TENSOR_NEXTN_ENORM, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL}},
1336
+ {LLM_TENSOR_NEXTN_HNORM, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL}},
1337
+ diff --git a/src/llama-arch.h b/src/llama-arch.h
1338
+ index b447273..85e8167 100644
1339
+ --- a/src/llama-arch.h
1340
+ +++ b/src/llama-arch.h
1341
+ @@ -155,6 +155,7 @@ enum llm_arch {
1342
+ LLM_ARCH_EAGLE3,
1343
+ LLM_ARCH_MINIMAX_M3,
1344
+ LLM_ARCH_DFLASH,
1345
+ + LLM_ARCH_K2_HORIZON,
1346
+ LLM_ARCH_UNKNOWN,
1347
+ };
1348
+
1349
+ @@ -391,6 +392,10 @@ enum llm_kv {
1350
+ LLM_KV_DENSE_2_FEAT_OUT,
1351
+ LLM_KV_DENSE_3_FEAT_IN,
1352
+ LLM_KV_DENSE_3_FEAT_OUT,
1353
+ +
1354
+ + // K2 Horizon MoVA
1355
+ + LLM_KV_ATTENTION_VALUE_EXPERT_COUNT,
1356
+ + LLM_KV_ATTENTION_VALUE_EXPERT_USED_COUNT,
1357
+ };
1358
+
1359
+ enum llm_tensor {
1360
+ @@ -667,6 +672,9 @@ enum llm_tensor {
1361
+ LLM_TENSOR_NEXTN_H_PROJ,
1362
+ LLM_TENSOR_NEXTN_EMBED_TOKENS,
1363
+ LLM_TENSOR_NEXTN_ENORM,
1364
+ + // K2 Horizon MoVA
1365
+ + LLM_TENSOR_ATTN_V_GATE,
1366
+ + LLM_TENSOR_ATTN_V_EXPS,
1367
+ LLM_TENSOR_NEXTN_HNORM,
1368
+ LLM_TENSOR_NEXTN_SHARED_HEAD_HEAD,
1369
+ LLM_TENSOR_NEXTN_SHARED_HEAD_NORM,
1370
+ diff --git a/src/llama-hparams.h b/src/llama-hparams.h
1371
+ index 2ffe84c..bef7c5d 100644
1372
+ --- a/src/llama-hparams.h
1373
+ +++ b/src/llama-hparams.h
1374
+ @@ -65,6 +65,10 @@ struct llama_hparams {
1375
+ // note: deepseek2 using MLA converts into MQA with larger heads, then decompresses to MHA
1376
+ uint32_t n_embd_head_k_mla_impl = 0;
1377
+ uint32_t n_embd_head_v_mla_impl = 0;
1378
+ + // K2 Horizon MoVA
1379
+ + uint32_t n_value_expert = 0;
1380
+ + uint32_t n_value_expert_used = 0;
1381
+ +
1382
+
1383
+ // for WavTokenizer
1384
+ struct llama_hparams_posnet posnet;
1385
+ diff --git a/src/llama-model.cpp b/src/llama-model.cpp
1386
+ index 8dd0efa..9eb162e 100644
1387
+ --- a/src/llama-model.cpp
1388
+ +++ b/src/llama-model.cpp
1389
+ @@ -320,7 +320,9 @@ static llama_model * llama_model_mapping(llm_arch arch, const llama_model_params
1390
+ return new llama_model_zaya(params);
1391
+ case LLM_ARCH_STEP35:
1392
+ return new llama_model_step35(params);
1393
+ - default:
1394
+ + case LLM_ARCH_K2_HORIZON:
1395
+ + return new llama_model_k2_horizon(params);
1396
+ + default:
1397
+ throw std::runtime_error(std::string("unsupported model architecture: '") + llm_arch_name(arch) + "'");
1398
+ }
1399
+
1400
+ @@ -2583,6 +2585,7 @@ llama_rope_type llama_model_rope_type(const llama_model * model) {
1401
+ case LLM_ARCH_STEP35:
1402
+ case LLM_ARCH_HYV3:
1403
+ case LLM_ARCH_TALKIE:
1404
+ + case LLM_ARCH_K2_HORIZON:
1405
+ case LLM_ARCH_MELLUM:
1406
+ case LLM_ARCH_ZAYA:
1407
+ return LLAMA_ROPE_TYPE_NEOX;
1408
+ diff --git a/src/llama-model.h b/src/llama-model.h
1409
+ index 53ec1e8..eae26ae 100644
1410
+ --- a/src/llama-model.h
1411
+ +++ b/src/llama-model.h
1412
+ @@ -297,6 +297,10 @@ struct llama_layer {
1413
+ struct ggml_tensor * ffn_norm_exps = nullptr;
1414
+ struct ggml_tensor * ffn_norm_enc = nullptr;
1415
+
1416
+ + // K2 Horizon MoVA
1417
+ + struct ggml_tensor * attn_v_gate = nullptr;
1418
+ + struct ggml_tensor * attn_v_gate_b = nullptr;
1419
+ + struct ggml_tensor * attn_v_exps = nullptr;
1420
+ // ff
1421
+ struct ggml_tensor * ffn_gate = nullptr; // w1
1422
+ struct ggml_tensor * ffn_down = nullptr; // w2
1423
+ diff --git a/src/llama-vocab.cpp b/src/llama-vocab.cpp
1424
+ index 0c899d1..c78ef00 100644
1425
+ --- a/src/llama-vocab.cpp
1426
+ +++ b/src/llama-vocab.cpp
1427
+ @@ -526,6 +526,11 @@ struct llm_tokenizer_bpe : llm_tokenizer {
1428
+ "(?:'[sS]|'[tT]|'[rR][eE]|'[vV][eE]|'[mM]|'[lL][lL]|'[dD])|[^\\r\\n\\p{L}\\p{N}]?\\p{L}+|\\p{N}+| ?[^\\s\\p{L}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
1429
+ };
1430
+ break;
1431
+ + case LLAMA_VOCAB_PRE_TYPE_K2_HORIZON:
1432
+ + regex_exprs = {
1433
+ + "(?:'[sS]|'[tT]|'[rR][eE]|'[vV][eE]|'[mM]|'[lL][lL]|'[dD])|[^\\r\\n\\p{L}\\p{N}]?(?:\\p{L}|\\p{M}|\\u200C|\\u200D)+|\\p{N}{1,3}| ?[^\\s\\p{L}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
1434
+ + };
1435
+ + break;
1436
+ case LLAMA_VOCAB_PRE_TYPE_WHITESPACE:
1437
+ // whitespace pre-tokenizer (jinaai/jina-embeddings-v2-base-zh)
1438
+ regex_exprs = {
1439
+ @@ -2333,6 +2338,10 @@ void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) {
1440
+ tokenizer_pre == "solar-open") {
1441
+ pre_type = LLAMA_VOCAB_PRE_TYPE_SOLAR_OPEN;
1442
+ clean_spaces = false;
1443
+ + } else if (
1444
+ + tokenizer_pre == "k2-horizon") {
1445
+ + pre_type = LLAMA_VOCAB_PRE_TYPE_K2_HORIZON;
1446
+ + clean_spaces = false;
1447
+ } else {
1448
+ throw std::runtime_error(format("unknown pre-tokenizer type: '%s'", tokenizer_pre.c_str()));
1449
+ }
1450
+ diff --git a/src/llama-vocab.h b/src/llama-vocab.h
1451
+ index 9d8e4bd..83b29f8 100644
1452
+ --- a/src/llama-vocab.h
1453
+ +++ b/src/llama-vocab.h
1454
+ @@ -64,6 +64,7 @@ enum llama_vocab_pre_type {
1455
+ LLAMA_VOCAB_PRE_TYPE_WHITESPACE = 53,
1456
+ LLAMA_VOCAB_PRE_TYPE_LAGUNA = 56,
1457
+ LLAMA_VOCAB_PRE_TYPE_MELLUM2 = 57,
1458
+ + LLAMA_VOCAB_PRE_TYPE_K2_HORIZON = 58,
1459
+ };
1460
+
1461
+ struct LLM_KV;
1462
+ diff --git a/src/models/k2-horizon.cpp b/src/models/k2-horizon.cpp
1463
+ new file mode 100644
1464
+ index 0000000..9776364
1465
+ --- /dev/null
1466
+ +++ b/src/models/k2-horizon.cpp
1467
+ @@ -0,0 +1,663 @@
1468
+ +#include "models.h"
1469
+ +
1470
+ +void llama_model_k2_horizon::load_arch_hparams(llama_model_loader & ml) {
1471
+ + // generic
1472
+ + ml.get_key(LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, hparams.f_norm_rms_eps);
1473
+ + ml.get_key(LLM_KV_ATTENTION_GROUPNORM_GROUPS, hparams.n_norm_groups, false);
1474
+ +
1475
+ + hparams.f_norm_group_eps = hparams.f_norm_rms_eps;
1476
+ + if (hparams.n_norm_groups == 0) hparams.n_norm_groups = 1;
1477
+ +
1478
+ + // moe
1479
+ + if (hparams.n_expert > 0) {
1480
+ + ml.get_key(LLM_KV_EXPERT_FEED_FORWARD_LENGTH, hparams.n_ff_exp);
1481
+ + ml.get_key(LLM_KV_LEADING_DENSE_BLOCK_COUNT, hparams.n_layer_dense_lead, false);
1482
+ + ml.get_key(LLM_KV_MOE_EVERY_N_LAYERS, hparams.moe_every_n_layers, false);
1483
+ + ml.get_key(LLM_KV_EXPERT_SHARED_COUNT, hparams.n_expert_shared, false);
1484
+ + ml.get_key(LLM_KV_EXPERT_SHARED_FEED_FORWARD_LENGTH, hparams.n_ff_shexp, false);
1485
+ + ml.get_key(LLM_KV_EXPERT_WEIGHTS_SCALE, hparams.expert_weights_scale, false);
1486
+ + ml.get_key(LLM_KV_EXPERT_WEIGHTS_NORM, hparams.expert_weights_norm, false);
1487
+ + ml.get_key(LLM_KV_EXPERT_GATING_FUNC, hparams.expert_gating_func, false);
1488
+ + if (hparams.expert_gating_func == LLAMA_EXPERT_GATING_FUNC_TYPE_NONE) {
1489
+ + hparams.expert_gating_func = LLAMA_EXPERT_GATING_FUNC_TYPE_SIGMOID;
1490
+ + }
1491
+ + }
1492
+ +
1493
+ + // mova
1494
+ + ml.get_key(LLM_KV_ATTENTION_VALUE_EXPERT_COUNT, hparams.n_value_expert, false);
1495
+ + ml.get_key(LLM_KV_ATTENTION_VALUE_EXPERT_USED_COUNT, hparams.n_value_expert_used, false);
1496
+ + if (hparams.n_value_expert > 0) {
1497
+ + GGML_ASSERT(hparams.n_value_expert <= LLAMA_MAX_EXPERTS);
1498
+ + GGML_ASSERT(hparams.n_value_expert_used > 0);
1499
+ + GGML_ASSERT(hparams.n_value_expert_used <= hparams.n_value_expert);
1500
+ + }
1501
+ + else {
1502
+ + GGML_ASSERT(hparams.n_value_expert_used == 0);
1503
+ + }
1504
+ +
1505
+ + // model size info
1506
+ + if (hparams.n_layer == 28 && hparams.n_embd == 1536) {
1507
+ + type = LLM_TYPE_1B;
1508
+ + }
1509
+ + else if (hparams.n_layer == 48 && hparams.n_embd == 2560) {
1510
+ + type = LLM_TYPE_36B;
1511
+ + }
1512
+ + else {
1513
+ + type = LLM_TYPE_UNKNOWN;
1514
+ + }
1515
+ +}
1516
+ +
1517
+ +void llama_model_k2_horizon::load_arch_tensors(llama_model_loader & ml) {
1518
+ + GGML_UNUSED(ml);
1519
+ + LLAMA_LOAD_LOCALS; // initializing variables basically
1520
+ +
1521
+ + // embeddings
1522
+ + tok_embd = create_tensor(
1523
+ + tn(LLM_TENSOR_TOKEN_EMBD, "weight"),
1524
+ + {n_embd, n_vocab},
1525
+ + 0
1526
+ + );
1527
+ +
1528
+ + // final norm and output projection
1529
+ + output_norm = create_tensor(
1530
+ + tn(LLM_TENSOR_OUTPUT_NORM, "weight"),
1531
+ + {n_embd},
1532
+ + 0
1533
+ + );
1534
+ +
1535
+ + // output
1536
+ + output = create_tensor(
1537
+ + tn(LLM_TENSOR_OUTPUT, "weight"),
1538
+ + {n_embd, n_vocab},
1539
+ + TENSOR_NOT_REQUIRED // can be tied with embedding (indicated by tensor not found in .gguf). see next conditional
1540
+ + );
1541
+ + if (output == nullptr) {
1542
+ + output = create_tensor(
1543
+ + tn(LLM_TENSOR_TOKEN_EMBD, "weight"),
1544
+ + {n_embd, n_vocab},
1545
+ + TENSOR_DUPLICATED
1546
+ + );
1547
+ + }
1548
+ +
1549
+ + for (int i = 0; i < n_layer; i++){
1550
+ + auto & layer = layers[i];
1551
+ + const bool is_moe_layer = n_expert > 0 && static_cast<uint32_t>(i) >= hparams.n_layer_dense_lead;
1552
+ + const bool is_mova_layer = is_moe_layer && hparams.n_value_expert > 0; // in the architecture, if mova is moe as well
1553
+ +
1554
+ + // attn normalization
1555
+ + layer.attn_norm = create_tensor(
1556
+ + tn(LLM_TENSOR_ATTN_NORM, "weight", i),
1557
+ + {n_embd},
1558
+ + 0
1559
+ + );
1560
+ +
1561
+ + // query and key tensors, always dense. and their optional normalization
1562
+ + // query
1563
+ + layer.wq = create_tensor(
1564
+ + tn(LLM_TENSOR_ATTN_Q, "weight", i),
1565
+ + {n_embd, n_embd_head_k * n_head},
1566
+ + 0
1567
+ + );
1568
+ + layer.attn_q_norm = create_tensor(
1569
+ + tn(LLM_TENSOR_ATTN_Q_NORM, "weight", i),
1570
+ + {n_embd_head_k * n_head},
1571
+ + TENSOR_NOT_REQUIRED
1572
+ + );
1573
+ +
1574
+ + // key
1575
+ + layer.wk = create_tensor(
1576
+ + tn(LLM_TENSOR_ATTN_K, "weight", i),
1577
+ + {n_embd, n_embd_k_gqa},
1578
+ + 0
1579
+ + );
1580
+ + layer.attn_k_norm = create_tensor(
1581
+ + tn(LLM_TENSOR_ATTN_K_NORM, "weight", i),
1582
+ + {n_embd_k_gqa},
1583
+ + TENSOR_NOT_REQUIRED
1584
+ + );
1585
+ +
1586
+ + // value tensors, possible MoVA
1587
+ + if (is_mova_layer) {
1588
+ + layer.attn_v_gate = create_tensor(
1589
+ + tn(LLM_TENSOR_ATTN_V_GATE, "weight", i),
1590
+ + {n_embd, hparams.n_value_expert},
1591
+ + 0
1592
+ + );
1593
+ + layer.attn_v_gate_b = create_tensor(
1594
+ + tn(LLM_TENSOR_ATTN_V_GATE, "bias", i),
1595
+ + {hparams.n_value_expert},
1596
+ + TENSOR_NOT_REQUIRED
1597
+ + );
1598
+ + layer.attn_v_exps = create_tensor(
1599
+ + tn(LLM_TENSOR_ATTN_V_EXPS, "weight", i),
1600
+ + {n_embd, n_embd_v_gqa, hparams.n_value_expert},
1601
+ + 0
1602
+ + );
1603
+ + }
1604
+ + else {
1605
+ + layer.wv = create_tensor(
1606
+ + tn(LLM_TENSOR_ATTN_V, "weight", i),
1607
+ + {n_embd, n_embd_v_gqa},
1608
+ + 0
1609
+ + );
1610
+ + }
1611
+ +
1612
+ + // attn output projection
1613
+ + layer.wo = create_tensor(
1614
+ + tn(LLM_TENSOR_ATTN_OUT, "weight", i),
1615
+ + {n_embd_head_v * n_head, n_embd},
1616
+ + 0
1617
+ + );
1618
+ +
1619
+ + // optional softplus gate
1620
+ + layer.wqkv_gate = create_tensor(
1621
+ + tn(LLM_TENSOR_ATTN_GATE, "weight", i),
1622
+ + {n_embd, n_embd_head_v * n_head},
1623
+ + TENSOR_NOT_REQUIRED
1624
+ + );
1625
+ +
1626
+ + // FFN normalization
1627
+ + layer.ffn_norm = create_tensor(
1628
+ + tn(LLM_TENSOR_FFN_NORM, "weight", i),
1629
+ + {n_embd},
1630
+ + 0
1631
+ + );
1632
+ +
1633
+ + // MoE stuff
1634
+ + if (is_moe_layer) {
1635
+ + if (hparams.n_ff_exp == 0){
1636
+ + throw std::runtime_error("K2 MoE layer requires expert_feed_forward_length");
1637
+ + }
1638
+ +
1639
+ + // moe router and it's optional bias
1640
+ + layer.ffn_gate_inp = create_tensor(
1641
+ + tn(LLM_TENSOR_FFN_GATE_INP, "weight", i),
1642
+ + {n_embd, n_expert},
1643
+ + 0
1644
+ + );
1645
+ + layer.ffn_exp_probs_b = create_tensor(
1646
+ + tn(LLM_TENSOR_FFN_EXP_PROBS_B, "bias", i),
1647
+ + {n_expert},
1648
+ + TENSOR_NOT_REQUIRED
1649
+ + );
1650
+ +
1651
+ + // routed experts (up, gate, and down)
1652
+ + layer.ffn_up_exps = create_tensor(
1653
+ + tn(LLM_TENSOR_FFN_UP_EXPS, "weight", i),
1654
+ + {n_embd, hparams.n_ff_exp, n_expert},
1655
+ + 0
1656
+ + );
1657
+ + layer.ffn_gate_exps = create_tensor(
1658
+ + tn(LLM_TENSOR_FFN_GATE_EXPS, "weight", i),
1659
+ + {n_embd, hparams.n_ff_exp, n_expert},
1660
+ + 0
1661
+ + );
1662
+ + layer.ffn_down_exps = create_tensor(
1663
+ + tn(LLM_TENSOR_FFN_DOWN_EXPS, "weight", i),
1664
+ + {hparams.n_ff_exp, n_embd, n_expert},
1665
+ + 0
1666
+ + );
1667
+ +
1668
+ + // shared experts (always evaluated)
1669
+ + if (hparams.n_expert_shared > 0) {
1670
+ + int64_t n_ff_shexp;
1671
+ + if (hparams.n_ff_shexp > 0) {
1672
+ + n_ff_shexp = hparams.n_ff_shexp;
1673
+ + } else {
1674
+ + n_ff_shexp = hparams.n_ff_exp * hparams.n_expert_shared;
1675
+ + }
1676
+ +
1677
+ + // up gate down
1678
+ + layer.ffn_up_shexp = create_tensor(
1679
+ + tn(LLM_TENSOR_FFN_UP_SHEXP, "weight", i),
1680
+ + {n_embd, n_ff_shexp},
1681
+ + 0
1682
+ + );
1683
+ + layer.ffn_gate_shexp = create_tensor(
1684
+ + tn(LLM_TENSOR_FFN_GATE_SHEXP, "weight", i),
1685
+ + {n_embd, n_ff_shexp},
1686
+ + 0
1687
+ + );
1688
+ + layer.ffn_down_shexp = create_tensor(
1689
+ + tn(LLM_TENSOR_FFN_DOWN_SHEXP, "weight", i),
1690
+ + {n_ff_shexp, n_embd},
1691
+ + 0
1692
+ + );
1693
+ + }
1694
+ + }
1695
+ + else {
1696
+ + // ordinary up gate down
1697
+ + layer.ffn_up = create_tensor(
1698
+ + tn(LLM_TENSOR_FFN_UP, "weight", i),
1699
+ + {n_embd, n_ff},
1700
+ + 0
1701
+ + );
1702
+ + layer.ffn_gate = create_tensor(
1703
+ + tn(LLM_TENSOR_FFN_GATE, "weight", i),
1704
+ + {n_embd, n_ff},
1705
+ + 0
1706
+ + );
1707
+ + layer.ffn_down = create_tensor(
1708
+ + tn(LLM_TENSOR_FFN_DOWN, "weight", i),
1709
+ + {n_ff, n_embd},
1710
+ + 0
1711
+ + );
1712
+ + }
1713
+ +
1714
+ + }
1715
+ +
1716
+ +}
1717
+ +
1718
+ +// helper for grouped RMS norm
1719
+ +static ggml_tensor * k2_horizon_group_rms_norm(
1720
+ + ggml_context * ctx,
1721
+ + ggml_tensor * cur,
1722
+ + ggml_tensor * weight,
1723
+ + int64_t n_groups,
1724
+ + float eps
1725
+ +) {
1726
+ + GGML_ASSERT(n_groups > 0);
1727
+ + GGML_ASSERT(cur->ne[0] % n_groups == 0);
1728
+ +
1729
+ + const int64_t n_embd = cur->ne[0];
1730
+ + const int64_t n_tokens = cur->ne[1];
1731
+ +
1732
+ + // separate embeddings into groups
1733
+ + cur = ggml_reshape_3d(
1734
+ + ctx,
1735
+ + cur,
1736
+ + n_embd / n_groups,
1737
+ + n_groups,
1738
+ + n_tokens
1739
+ + );
1740
+ +
1741
+ + // norm it
1742
+ + cur = ggml_rms_norm(ctx, cur, eps);
1743
+ +
1744
+ + // bring back shape
1745
+ + cur = ggml_reshape_2d(ctx, cur, n_embd, n_tokens);
1746
+ +
1747
+ + // apply the learned normalization weights
1748
+ + if (weight != nullptr) {
1749
+ + cur = ggml_mul(ctx, cur, weight);
1750
+ + }
1751
+ +
1752
+ + return cur;
1753
+ +}
1754
+ +
1755
+ +ggml_tensor * llama_model_k2_horizon::graph::build_routed_value(
1756
+ + const llama_layer & layer,
1757
+ + ggml_tensor * cur,
1758
+ + int il
1759
+ +) const {
1760
+ + const int64_t n_embd = cur->ne[0];
1761
+ + const int64_t n_tokens = cur->ne[1];
1762
+ + const int64_t n_embd_gqa = hparams.n_embd_v_gqa(il);
1763
+ + const int64_t n_values = hparams.n_value_expert;
1764
+ + const int64_t n_used = hparams.n_value_expert_used;
1765
+ +
1766
+ + GGML_ASSERT(layer.attn_v_gate != nullptr);
1767
+ + GGML_ASSERT(layer.attn_v_exps != nullptr);
1768
+ + GGML_ASSERT(n_values > 0);
1769
+ + GGML_ASSERT(n_used > 0);
1770
+ +
1771
+ + // router. logits and probs
1772
+ + ggml_tensor * logits = build_lora_mm(layer.attn_v_gate, cur);
1773
+ + ggml_tensor * probs = nullptr;
1774
+ +
1775
+ + // probs
1776
+ + llama_expert_gating_func_type gating_func = static_cast<llama_expert_gating_func_type>(hparams.expert_gating_func);
1777
+ + switch(gating_func){
1778
+ + case LLAMA_EXPERT_GATING_FUNC_TYPE_SOFTMAX:
1779
+ + probs = ggml_soft_max(ctx0, logits);
1780
+ + break;
1781
+ + case LLAMA_EXPERT_GATING_FUNC_TYPE_SIGMOID:
1782
+ + probs = ggml_sigmoid(ctx0, logits);
1783
+ + break;
1784
+ + default:
1785
+ + GGML_ABORT("Unsupported K2 Horizon value-router gating function");
1786
+ + }
1787
+ +
1788
+ + // selection probs
1789
+ + ggml_tensor * selection_probs = probs;
1790
+ + if (layer.attn_v_gate_b != nullptr){
1791
+ + selection_probs = ggml_add(ctx0, probs, layer.attn_v_gate_b);
1792
+ + cb(selection_probs, "v_moe_probs_biased", il);
1793
+ + }
1794
+ +
1795
+ + // select expert values
1796
+ + ggml_tensor * selected_value_experts = ggml_argsort_top_k(ctx0, selection_probs, n_used);
1797
+ +
1798
+ + // reshaping and selecting the weights (probs) of the selected experts
1799
+ + probs = ggml_reshape_3d(ctx0, probs, 1, n_values, n_tokens);
1800
+ + ggml_tensor * selected_weights = ggml_get_rows(ctx0, probs, selected_value_experts);
1801
+ +
1802
+ + // if weights of value experts are to be normalized
1803
+ + if (hparams.expert_weights_norm) {
1804
+ + selected_weights = ggml_reshape_2d(ctx0, selected_weights, n_used, n_tokens);
1805
+ + ggml_tensor * selected_weights_sum = ggml_sum_rows(ctx0, selected_weights);
1806
+ + selected_weights_sum = ggml_clamp(ctx0, selected_weights_sum, 6.103515625e-5f, INFINITY);
1807
+ + selected_weights = ggml_div(ctx0, selected_weights, selected_weights_sum);
1808
+ + selected_weights = ggml_reshape_3d(ctx0, selected_weights, 1, n_used, n_tokens);
1809
+ + cb(selected_weights, "v_moe_weights_norm", il);
1810
+ + }
1811
+ +
1812
+ + // scaling
1813
+ + if (hparams.expert_weights_scale != 0.0f && hparams.expert_weights_scale != 1.0f) {
1814
+ + selected_weights = ggml_scale(ctx0, selected_weights, hparams.expert_weights_scale);
1815
+ + cb(selected_weights, "v_moe_weights_scaled", il);
1816
+ + }
1817
+ +
1818
+ + // labeling
1819
+ + cb(logits, "v_moe_logits", il);
1820
+ + cb(probs, "v_moe_probs", il);
1821
+ + cb(selected_value_experts->src[0], "v_moe_argsort", il);
1822
+ + cb(selected_value_experts, "v_moe_topk", il);
1823
+ + cb(selected_weights, "v_moe_weights", il);
1824
+ +
1825
+ + ggml_tensor * value_inp = ggml_reshape_3d(ctx0, cur, n_embd, 1, n_tokens);
1826
+ + // computing only on selected experts (the _id in the api)
1827
+ + ggml_tensor * values = build_lora_mm_id(layer.attn_v_exps, value_inp, selected_value_experts);
1828
+ + values = ggml_silu(ctx0, values);
1829
+ + values = ggml_mul(ctx0, values, selected_weights);
1830
+ + cb(values, "v_moe_weighted", il);
1831
+ +
1832
+ + // sum the multiple value outputs
1833
+ + ggml_tensor * value_parts[LLAMA_MAX_EXPERTS] = {};
1834
+ + for(int64_t i = 0; i < n_used; i++) {
1835
+ + value_parts[i] = ggml_view_2d(ctx0, values, n_embd_gqa, n_tokens, values->nb[2], i * values->nb[1]);
1836
+ + }
1837
+ + ggml_tensor * value_out = value_parts[0];
1838
+ + for (int64_t i = 1; i < n_used; ++i) {
1839
+ + value_out = ggml_add(ctx0, value_out, value_parts[i]);
1840
+ + }
1841
+ +
1842
+ + // making it contiguous in case it isn't (for one expert only)
1843
+ + if (n_used == 1) value_out = ggml_cont(ctx0, value_out);
1844
+ +
1845
+ + cb(value_out, "Vcur_routed", il);
1846
+ + return value_out;
1847
+ +}
1848
+ +
1849
+ +llama_model_k2_horizon::graph::graph(
1850
+ + const llama_model & model,
1851
+ + const llm_graph_params & params
1852
+ +) : llm_graph_context(params) {
1853
+ + const int64_t n_embd_head = hparams.n_embd_head_v();
1854
+ + GGML_ASSERT(n_embd_head == hparams.n_embd_head_k());
1855
+ +
1856
+ + // initialization or placeholders for computational artifacts
1857
+ + ggml_tensor * cur;
1858
+ + ggml_tensor * inpL = build_inp_embd(model.tok_embd);
1859
+ + ggml_tensor * inp_pos = build_inp_pos();
1860
+ + auto * inp_attn = build_attn_inp_kv();
1861
+ + ggml_tensor * inp_out_ids = build_inp_out_ids();
1862
+ +
1863
+ +
1864
+ + for (int il = 0; il < n_layer; ++il) {
1865
+ + res->t_layer_inp[il] = inpL;
1866
+ + ggml_tensor * inpSA = inpL; // for residuals
1867
+ +
1868
+ + const bool is_moe_layer = n_expert > 0 && static_cast<uint32_t>(il) >= hparams.n_layer_dense_lead;
1869
+ + const bool is_mova_layer = is_moe_layer && hparams.n_value_expert > 0;
1870
+ +
1871
+ + // ============ grouped rms norm
1872
+ + cur = k2_horizon_group_rms_norm(
1873
+ + ctx0,
1874
+ + inpL,
1875
+ + model.layers[il].attn_norm,
1876
+ + hparams.n_norm_groups,
1877
+ + hparams.f_norm_rms_eps
1878
+ + );
1879
+ + cb(cur, "attn_norm", il);
1880
+ +
1881
+ + // ============ setup attention tensors
1882
+ + ggml_tensor * attn_inp = cur;
1883
+ +
1884
+ + // query
1885
+ + ggml_tensor * Qcur = build_lora_mm(model.layers[il].wq, cur, model.layers[il].wq_s);
1886
+ + if (model.layers[il].attn_q_norm != nullptr) {
1887
+ + Qcur = k2_horizon_group_rms_norm(
1888
+ + ctx0,
1889
+ + Qcur,
1890
+ + model.layers[il].attn_q_norm,
1891
+ + n_head,
1892
+ + hparams.f_norm_rms_eps
1893
+ + );
1894
+ + }
1895
+ +
1896
+ + // key
1897
+ + ggml_tensor * Kcur = build_lora_mm(model.layers[il].wk, cur, model.layers[il].wk_s);
1898
+ + if (model.layers[il].attn_k_norm != nullptr) {
1899
+ + Kcur = k2_horizon_group_rms_norm(
1900
+ + ctx0,
1901
+ + Kcur,
1902
+ + model.layers[il].attn_k_norm,
1903
+ + n_head_kv,
1904
+ + hparams.f_norm_rms_eps
1905
+ + );
1906
+ + }
1907
+ +
1908
+ + // value
1909
+ + ggml_tensor * Vcur;
1910
+ + if (is_mova_layer) {
1911
+ + Vcur = build_routed_value(model.layers[il], cur, il); // handle MoVA
1912
+ + }
1913
+ + else {
1914
+ + Vcur = build_lora_mm(model.layers[il].wv, cur, model.layers[il].wv_s);
1915
+ + }
1916
+ +
1917
+ + // reshaping
1918
+ + Qcur = ggml_reshape_3d(ctx0, Qcur, n_embd_head, n_head, n_tokens);
1919
+ + Kcur = ggml_reshape_3d(ctx0, Kcur, n_embd_head, n_head_kv, n_tokens);
1920
+ + Vcur = ggml_reshape_3d(ctx0, Vcur, n_embd_head, n_head_kv, n_tokens);
1921
+ +
1922
+ + // applying RoPE
1923
+ + Qcur = ggml_rope_ext(
1924
+ + ctx0,
1925
+ + Qcur,
1926
+ + inp_pos,
1927
+ + nullptr,
1928
+ + n_rot,
1929
+ + rope_type,
1930
+ + n_ctx_orig,
1931
+ + freq_base,
1932
+ + freq_scale,
1933
+ + ext_factor,
1934
+ + attn_factor,
1935
+ + beta_fast,
1936
+ + beta_slow
1937
+ + );
1938
+ + Kcur = ggml_rope_ext(
1939
+ + ctx0,
1940
+ + Kcur,
1941
+ + inp_pos,
1942
+ + nullptr,
1943
+ + n_rot,
1944
+ + rope_type,
1945
+ + n_ctx_orig,
1946
+ + freq_base,
1947
+ + freq_scale,
1948
+ + ext_factor,
1949
+ + attn_factor,
1950
+ + beta_fast,
1951
+ + beta_slow
1952
+ + );
1953
+ +
1954
+ + cb(Qcur, "Qcur", il);
1955
+ + cb(Kcur, "Kcur", il);
1956
+ + cb(Vcur, "Vcur", il);
1957
+ +
1958
+ + // ============ attention (with and without gating)
1959
+ + const float kq_scale = 1.0f / sqrtf(static_cast<float>(n_embd_head));
1960
+ + if(model.layers[il].wqkv_gate == nullptr){ // without gating
1961
+ + cur = build_attn(
1962
+ + inp_attn,
1963
+ + model.layers[il].wo,
1964
+ + model.layers[il].wo_b,
1965
+ + model.layers[il].wo_s,
1966
+ + Qcur,
1967
+ + Kcur,
1968
+ + Vcur,
1969
+ + nullptr, // attention score bias
1970
+ + nullptr, // attn sink
1971
+ + nullptr, // MLA value transformation
1972
+ + kq_scale,
1973
+ + il
1974
+ + );
1975
+ + }
1976
+ + else { // with gating
1977
+ + // no output yet
1978
+ + cur = build_attn(
1979
+ + inp_attn,
1980
+ + nullptr,
1981
+ + nullptr,
1982
+ + nullptr,
1983
+ + Qcur,
1984
+ + Kcur,
1985
+ + Vcur,
1986
+ + nullptr,
1987
+ + nullptr,
1988
+ + nullptr,
1989
+ + kq_scale,
1990
+ + il
1991
+ + );
1992
+ +
1993
+ + // building the gate
1994
+ + constexpr float LN2 = 0.6931471805599453f;
1995
+ + constexpr float ONE_OVER_LN2 = 1.4426950408889634f;
1996
+ +
1997
+ + ggml_tensor * gate = build_lora_mm(model.layers[il].wqkv_gate, attn_inp, model.layers[il].wqkv_gate_s);
1998
+ + gate = ggml_scale(ctx0, gate, LN2);
1999
+ + gate = ggml_softplus(ctx0, gate);
2000
+ + gate = ggml_scale(ctx0, gate, ONE_OVER_LN2);
2001
+ +
2002
+ + // applying the gate
2003
+ + cur = ggml_mul(ctx0, cur, gate);
2004
+ +
2005
+ + // projection
2006
+ + cur = build_lora_mm(model.layers[il].wo, cur, model.layers[il].wo_s);
2007
+ +
2008
+ + // bias
2009
+ + if (model.layers[il].wo_b != nullptr) {
2010
+ + cur = ggml_add(ctx0, cur, model.layers[il].wo_b);
2011
+ + }
2012
+ + }
2013
+ +
2014
+ + // ============ output layer, and take (usually) last token for generation
2015
+ + if (il == n_layer - 1 && inp_out_ids != nullptr) {
2016
+ + cur = ggml_get_rows(ctx0, cur, inp_out_ids);
2017
+ + inpSA = ggml_get_rows(ctx0, inpSA, inp_out_ids); // pull the same positions for inpSA
2018
+ + }
2019
+ +
2020
+ + // ============ add residuals
2021
+ + ggml_tensor * ffn_inp = ggml_add(ctx0, cur, inpSA);
2022
+ + cb(ffn_inp, "ffn_inp", il);
2023
+ +
2024
+ + // ============ group RMSNorm before FFN
2025
+ + cur = k2_horizon_group_rms_norm(
2026
+ + ctx0,
2027
+ + ffn_inp,
2028
+ + model.layers[il].ffn_norm,
2029
+ + hparams.n_norm_groups,
2030
+ + hparams.f_norm_rms_eps
2031
+ + );
2032
+ + cb(cur, "ffn_norm", il);
2033
+ +
2034
+ + // ============ Mixture of Experts
2035
+ + if (is_moe_layer) {
2036
+ + ggml_tensor * moe_out = build_moe_ffn(
2037
+ + cur,
2038
+ + model.layers[il].ffn_gate_inp,
2039
+ + model.layers[il].ffn_up_exps,
2040
+ + model.layers[il].ffn_gate_exps,
2041
+ + model.layers[il].ffn_down_exps,
2042
+ + model.layers[il].ffn_exp_probs_b,
2043
+ + n_expert,
2044
+ + n_expert_used,
2045
+ + LLM_FFN_SILU,
2046
+ + hparams.expert_weights_norm,
2047
+ + hparams.expert_weights_scale,
2048
+ + static_cast<llama_expert_gating_func_type>(hparams.expert_gating_func),
2049
+ + il
2050
+ + );
2051
+ +
2052
+ + // shared experts
2053
+ + if (model.layers[il].ffn_gate_shexp != nullptr){
2054
+ + ggml_tensor * shared_moe_out = build_ffn(
2055
+ + cur,
2056
+ + model.layers[il].ffn_up_shexp,
2057
+ + nullptr,
2058
+ + nullptr,
2059
+ + model.layers[il].ffn_gate_shexp,
2060
+ + nullptr,
2061
+ + nullptr,
2062
+ + model.layers[il].ffn_down_shexp,
2063
+ + nullptr,
2064
+ + nullptr,
2065
+ + nullptr,
2066
+ + LLM_FFN_SILU,
2067
+ + LLM_FFN_PAR,
2068
+ + il
2069
+ + );
2070
+ + cur = ggml_add(ctx0, moe_out, shared_moe_out);
2071
+ + }
2072
+ + else{
2073
+ + cur = moe_out;
2074
+ + }
2075
+ + }
2076
+ + else { // normal non moe FFN
2077
+ + cur = build_ffn(
2078
+ + cur,
2079
+ + model.layers[il].ffn_up,
2080
+ + nullptr,
2081
+ + nullptr,
2082
+ + model.layers[il].ffn_gate,
2083
+ + nullptr,
2084
+ + nullptr,
2085
+ + model.layers[il].ffn_down,
2086
+ + nullptr,
2087
+ + nullptr,
2088
+ + nullptr,
2089
+ + LLM_FFN_SILU,
2090
+ + LLM_FFN_PAR,
2091
+ + il
2092
+ + );
2093
+ + }
2094
+ + cb(cur, "ffn_out", il);
2095
+ +
2096
+ + // ============ FFN residual
2097
+ + cur = ggml_add(ctx0, cur, ffn_inp);
2098
+ + cur = build_cvec(cur, il);
2099
+ + cb(cur, "l_out", il);
2100
+ +
2101
+ + // for next layer
2102
+ + inpL = cur;
2103
+ + }
2104
+ +
2105
+ + // final group rms norm. also becomes last layer embedding
2106
+ + cur = k2_horizon_group_rms_norm(
2107
+ + ctx0,
2108
+ + inpL,
2109
+ + model.output_norm,
2110
+ + hparams.n_norm_groups,
2111
+ + hparams.f_norm_rms_eps
2112
+ + );
2113
+ + cb(cur, "result_norm", -1);
2114
+ + res->t_embd = cur;
2115
+ +
2116
+ + // ============ vocab projection. also becomes logits
2117
+ + cur = build_lora_mm(model.output, cur,model.output_s);
2118
+ + cb(cur, "result_output", -1);
2119
+ + res->t_logits = cur;
2120
+ +
2121
+ + // build everything
2122
+ + ggml_build_forward_expand(gf, cur);
2123
+ +}
2124
+ +
2125
+ +
2126
+ +std::unique_ptr<llm_graph_context> llama_model_k2_horizon::build_arch_graph (
2127
+ + const llm_graph_params & params
2128
+ +) const {
2129
+ + return std::make_unique<graph>(*this, params);
2130
+ +}
2131
+ diff --git a/src/models/models.h b/src/models/models.h
2132
+ index 2d07f5e..fb2165d 100644
2133
+ --- a/src/models/models.h
2134
+ +++ b/src/models/models.h
2135
+ @@ -2270,3 +2270,31 @@ struct llama_model_zaya : public llama_model_base {
2136
+
2137
+ std::unique_ptr<llm_graph_context> build_arch_graph(const llm_graph_params & params) const override;
2138
+ };
2139
+ +
2140
+ +
2141
+ +struct llama_model_k2_horizon : public llama_model_base {
2142
+ + llama_model_k2_horizon(
2143
+ + const llama_model_params & params
2144
+ + ) : llama_model_base(params) {}
2145
+ +
2146
+ + void load_arch_hparams(llama_model_loader & ml) override;
2147
+ +
2148
+ + void load_arch_tensors(llama_model_loader & ml) override;
2149
+ +
2150
+ + struct graph: public llm_graph_context {
2151
+ + graph(
2152
+ + const llama_model & model,
2153
+ + const llm_graph_params & params
2154
+ + );
2155
+ +
2156
+ + ggml_tensor * build_routed_value (
2157
+ + const llama_layer & layer,
2158
+ + ggml_tensor * cur,
2159
+ + int il // layer index
2160
+ + ) const;
2161
+ + };
2162
+ +
2163
+ + std::unique_ptr<llm_graph_context> build_arch_graph(
2164
+ + const llm_graph_params & params
2165
+ + ) const override;
2166
+ +};