From a1cc3a5e0dfe8a9f78a10728482d0b5514dcec8c Mon Sep 17 00:00:00 2001 From: "Claude (Fable 5)" Date: Tue, 7 Jul 2026 20:37:53 +1200 Subject: [PATCH 1/9] model: add Hy3 (hy_v3) architecture support Adds Tencent Hunyuan 3 (HF architecture HYV3ForCausalLM, GGUF arch hy_v3): a MoE decoder stack with per-head Q/K RMSNorm, a sigmoid router with expert selection bias, an always-active ungated shared expert, and leading dense block(s) (first_k_dense_replace). The base implementation is ported from charlie12345's fork (https://github.com/charlie12345/ROCmFPX, src/models/hyv3.cpp), adapted to current mainline APIs (hparams.n_layer(), build_qkv, build_moe_ffn with fused gate_up + scale tensors, output_s). Note: blk.N.exp_probs_b is stored without a .bias suffix for compatibility with existing hy_v3 GGUFs produced by that fork. Implemented by Claude (Anthropic Fable 5), directed by Satinder Grewal (@satindergrewal). Co-Authored-By: charlie12345 --- src/llama-arch.cpp | 1 + src/llama-arch.h | 1 + src/llama-model.cpp | 3 + src/models/hy-v3.cpp | 189 +++++++++++++++++++++++++++++++++++++++++++ src/models/models.h | 12 +++ 5 files changed, 206 insertions(+) create mode 100644 src/models/hy-v3.cpp diff --git a/src/llama-arch.cpp b/src/llama-arch.cpp index b890e66fcf6..72968607db8 100644 --- a/src/llama-arch.cpp +++ b/src/llama-arch.cpp @@ -113,6 +113,7 @@ static const std::map LLM_ARCH_NAMES = { { LLM_ARCH_HUNYUAN_MOE, "hunyuan-moe" }, { LLM_ARCH_HUNYUAN_DENSE, "hunyuan-dense" }, { LLM_ARCH_HUNYUAN_VL, "hunyuan_vl" }, + { LLM_ARCH_HY_V3, "hy_v3" }, { LLM_ARCH_SMOLLM3, "smollm3" }, { LLM_ARCH_OPENAI_MOE, "gpt-oss" }, { LLM_ARCH_LFM2, "lfm2" }, diff --git a/src/llama-arch.h b/src/llama-arch.h index a4f5091e717..b74d53af4a3 100644 --- a/src/llama-arch.h +++ b/src/llama-arch.h @@ -118,6 +118,7 @@ enum llm_arch { LLM_ARCH_HUNYUAN_MOE, LLM_ARCH_HUNYUAN_DENSE, LLM_ARCH_HUNYUAN_VL, + LLM_ARCH_HY_V3, LLM_ARCH_SMOLLM3, LLM_ARCH_OPENAI_MOE, LLM_ARCH_LFM2, diff --git a/src/llama-model.cpp b/src/llama-model.cpp index adacf702d05..a4ad5a0fcf2 100644 --- a/src/llama-model.cpp +++ b/src/llama-model.cpp @@ -262,6 +262,8 @@ static llama_model * llama_model_mapping(llm_arch arch, const llama_model_params return new llama_model_hunyuan_vl(params); case LLM_ARCH_HUNYUAN_DENSE: return new llama_model_hunyuan_dense(params); + case LLM_ARCH_HY_V3: + return new llama_model_hy_v3(params); case LLM_ARCH_SMOLLM3: return new llama_model_smollm3(params); case LLM_ARCH_OPENAI_MOE: @@ -2526,6 +2528,7 @@ llama_rope_type llama_model_rope_type(const llama_model * model) { case LLM_ARCH_JAIS2: case LLM_ARCH_OPENAI_MOE: case LLM_ARCH_HUNYUAN_DENSE: + case LLM_ARCH_HY_V3: case LLM_ARCH_LFM2: case LLM_ARCH_LFM2MOE: case LLM_ARCH_SMALLTHINKER: diff --git a/src/models/hy-v3.cpp b/src/models/hy-v3.cpp new file mode 100644 index 00000000000..d92964f79f9 --- /dev/null +++ b/src/models/hy-v3.cpp @@ -0,0 +1,189 @@ +#include "models.h" + +void llama_model_hy_v3::load_arch_hparams(llama_model_loader & ml) { + ml.get_key(LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, hparams.f_norm_rms_eps); + ml.get_key(LLM_KV_EXPERT_FEED_FORWARD_LENGTH, hparams.n_ff_exp); + ml.get_key(LLM_KV_EXPERT_SHARED_FEED_FORWARD_LENGTH, hparams.n_ff_shexp, false); + ml.get_key(LLM_KV_EXPERT_GATING_FUNC, hparams.expert_gating_func, false); + ml.get_key(LLM_KV_EXPERT_WEIGHTS_SCALE, hparams.expert_weights_scale, false); + ml.get_key(LLM_KV_EXPERT_WEIGHTS_NORM, hparams.expert_weights_norm, false); + + // HY V3 uses a sigmoid router with expert selection bias by default + if (hparams.expert_gating_func == LLAMA_EXPERT_GATING_FUNC_TYPE_NONE) { + hparams.expert_gating_func = LLAMA_EXPERT_GATING_FUNC_TYPE_SIGMOID; + } + + switch (hparams.n_layer()) { + case 48: type = LLM_TYPE_30B_A3B; break; + default: type = LLM_TYPE_UNKNOWN; + } +} + +void llama_model_hy_v3::load_arch_tensors(llama_model_loader & ml) { + LLAMA_LOAD_LOCALS; + + GGML_UNUSED(ml); + + tok_embd = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, 0); + + output_norm = create_tensor(tn(LLM_TENSOR_OUTPUT_NORM, "weight"), {n_embd}, 0); + output = create_tensor(tn(LLM_TENSOR_OUTPUT, "weight"), {n_embd, n_vocab}, TENSOR_NOT_REQUIRED); + if (output == NULL) { + output = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, TENSOR_DUPLICATED); + } + + auto load_block = [&](int i, int flags) { + auto & layer = layers[i]; + const int64_t n_ff_exp = hparams.n_ff_exp ? hparams.n_ff_exp : n_ff / (n_expert_used > 0 ? n_expert_used : 1); + const int64_t n_ff_shexp = hparams.n_ff_shexp ? hparams.n_ff_shexp : n_ff_exp; + + layer.attn_norm = create_tensor(tn(LLM_TENSOR_ATTN_NORM, "weight", i), {n_embd}, flags); + + create_tensor_qkv(layer, i, n_embd, n_embd_head_k * n_head, n_embd_k_gqa, n_embd_v_gqa, flags); + layer.wo = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "weight", i), {n_embd_head_k * n_head, n_embd}, flags); + + layer.attn_q_norm = create_tensor(tn(LLM_TENSOR_ATTN_Q_NORM, "weight", i), {n_embd_head_k}, flags); + layer.attn_k_norm = create_tensor(tn(LLM_TENSOR_ATTN_K_NORM, "weight", i), {n_embd_head_k}, flags); + + layer.ffn_norm = create_tensor(tn(LLM_TENSOR_FFN_NORM, "weight", i), {n_embd}, flags); + + // dense FFN (leading dense blocks, first_k_dense_replace) + layer.ffn_gate = create_tensor(tn(LLM_TENSOR_FFN_GATE, "weight", i), {n_embd, n_ff}, TENSOR_NOT_REQUIRED); + layer.ffn_down = create_tensor(tn(LLM_TENSOR_FFN_DOWN, "weight", i), {n_ff, n_embd}, TENSOR_NOT_REQUIRED); + layer.ffn_up = create_tensor(tn(LLM_TENSOR_FFN_UP, "weight", i), {n_embd, n_ff}, TENSOR_NOT_REQUIRED); + + // MoE routed experts (sigmoid router + expert selection bias) + layer.ffn_gate_inp = create_tensor(tn(LLM_TENSOR_FFN_GATE_INP, "weight", i), {n_embd, n_expert}, TENSOR_NOT_REQUIRED); + layer.ffn_exp_probs_b = create_tensor(tn(LLM_TENSOR_FFN_EXP_PROBS_B, i), {n_expert}, TENSOR_NOT_REQUIRED); + layer.ffn_down_exps = create_tensor(tn(LLM_TENSOR_FFN_DOWN_EXPS, "weight", i), {n_ff_exp, n_embd, n_expert}, TENSOR_NOT_REQUIRED); + create_tensor_gate_up_exps(layer, i, n_embd, n_ff_exp, n_expert, TENSOR_NOT_REQUIRED); + + // shared expert (always active, no gate) + layer.ffn_gate_shexp = create_tensor(tn(LLM_TENSOR_FFN_GATE_SHEXP, "weight", i), {n_embd, n_ff_shexp}, TENSOR_NOT_REQUIRED); + layer.ffn_up_shexp = create_tensor(tn(LLM_TENSOR_FFN_UP_SHEXP, "weight", i), {n_embd, n_ff_shexp}, TENSOR_NOT_REQUIRED); + layer.ffn_down_shexp = create_tensor(tn(LLM_TENSOR_FFN_DOWN_SHEXP, "weight", i), {n_ff_shexp, n_embd}, TENSOR_NOT_REQUIRED); + }; + + for (int i = 0; i < n_layer; ++i) { + load_block(i, 0); + } +} + +std::unique_ptr llama_model_hy_v3::build_arch_graph(const llm_graph_params & params) const { + return std::make_unique(*this, params); +} + +llama_model_hy_v3::graph::graph(const llama_model & model, const llm_graph_params & params) : llm_graph_context(params) { + const int64_t n_embd_head = hparams.n_embd_head_v(); + + GGML_ASSERT(n_embd_head == hparams.n_embd_head_k()); + GGML_ASSERT(n_embd_head == n_rot); + + ggml_tensor * cur; + ggml_tensor * inpL; + + inpL = build_inp_embd(model.tok_embd); + ggml_tensor * inp_pos = build_inp_pos(); + auto * inp_attn = build_attn_inp_kv(); + ggml_tensor * inp_out_ids = build_inp_out_ids(); + + const float kq_scale = 1.0f / sqrtf(float(n_embd_head)); + + for (int il = 0; il < n_layer; ++il) { + ggml_tensor * inpSA = inpL; + + cur = build_norm(inpL, model.layers[il].attn_norm, nullptr, LLM_NORM_RMS, il); + cb(cur, "attn_norm", il); + + // self-attention + { + ggml_tensor * rope_factors = model.get_rope_factors(cparams, il); + + auto [Qcur, Kcur, Vcur] = build_qkv(model.layers[il], cur, n_embd_head, n_head, n_head_kv, il); + + Qcur = build_norm(Qcur, model.layers[il].attn_q_norm, nullptr, LLM_NORM_RMS, il); + Kcur = build_norm(Kcur, model.layers[il].attn_k_norm, nullptr, LLM_NORM_RMS, il); + + Qcur = ggml_rope_ext(ctx0, Qcur, inp_pos, rope_factors, + n_rot, rope_type, n_ctx_orig, freq_base, freq_scale, + ext_factor, attn_factor, beta_fast, beta_slow); + Kcur = ggml_rope_ext(ctx0, Kcur, inp_pos, rope_factors, + n_rot, rope_type, n_ctx_orig, freq_base, freq_scale, + ext_factor, attn_factor, beta_fast, beta_slow); + + cur = build_attn(inp_attn, + model.layers[il].wo, model.layers[il].wo_b, model.layers[il].wo_s, + Qcur, Kcur, Vcur, nullptr, nullptr, nullptr, kq_scale, il); + cb(cur, "attn_out", il); + } + + if (il == n_layer - 1 && inp_out_ids) { + cur = ggml_get_rows(ctx0, cur, inp_out_ids); + inpSA = ggml_get_rows(ctx0, inpSA, inp_out_ids); + } + + ggml_tensor * ffn_inp = ggml_add(ctx0, cur, inpSA); + cb(ffn_inp, "ffn_inp", il); + + cur = build_norm(ffn_inp, model.layers[il].ffn_norm, nullptr, LLM_NORM_RMS, il); + cb(cur, "ffn_norm", il); + + if (model.layers[il].ffn_gate_inp == nullptr) { + // dense FFN (leading dense blocks) + cur = build_ffn(cur, + model.layers[il].ffn_up, model.layers[il].ffn_up_b, model.layers[il].ffn_up_s, + model.layers[il].ffn_gate, model.layers[il].ffn_gate_b, model.layers[il].ffn_gate_s, + model.layers[il].ffn_down, model.layers[il].ffn_down_b, model.layers[il].ffn_down_s, + nullptr, + LLM_FFN_SILU, LLM_FFN_PAR, il); + cb(cur, "ffn_dense_out", il); + } else { + // MoE routed experts (sigmoid gating + expert selection bias) + ggml_tensor * moe_out = build_moe_ffn(cur, + model.layers[il].ffn_gate_inp, + model.layers[il].ffn_up_exps, + model.layers[il].ffn_gate_exps, + model.layers[il].ffn_down_exps, + model.layers[il].ffn_exp_probs_b, + n_expert, n_expert_used, + LLM_FFN_SILU, + hparams.expert_weights_norm, + hparams.expert_weights_scale, + (llama_expert_gating_func_type) hparams.expert_gating_func, + il, + nullptr, model.layers[il].ffn_gate_up_exps, + model.layers[il].ffn_up_exps_s, + model.layers[il].ffn_gate_exps_s, + model.layers[il].ffn_down_exps_s); + cb(moe_out, "ffn_moe_out", il); + + // shared expert (always active, no gate) + ggml_tensor * sh_out = build_ffn(cur, + model.layers[il].ffn_up_shexp, nullptr, model.layers[il].ffn_up_shexp_s, + model.layers[il].ffn_gate_shexp, nullptr, model.layers[il].ffn_gate_shexp_s, + model.layers[il].ffn_down_shexp, nullptr, model.layers[il].ffn_down_shexp_s, + nullptr, + LLM_FFN_SILU, LLM_FFN_PAR, il); + cb(sh_out, "ffn_shared_out", il); + + cur = ggml_add(ctx0, moe_out, sh_out); + cb(cur, "ffn_out", il); + } + + cur = ggml_add(ctx0, cur, ffn_inp); + cur = build_cvec(cur, il); + cb(cur, "l_out", il); + + inpL = cur; + } + + cur = build_norm(inpL, model.output_norm, nullptr, LLM_NORM_RMS, -1); + cb(cur, "result_norm", -1); + res->t_embd = cur; + + cur = build_lora_mm(model.output, cur, model.output_s); + cb(cur, "result_output", -1); + res->t_logits = cur; + + ggml_build_forward_expand(gf, cur); +} diff --git a/src/models/models.h b/src/models/models.h index 7a52e7bc1ab..c47d0fa57b0 100644 --- a/src/models/models.h +++ b/src/models/models.h @@ -1729,6 +1729,18 @@ struct llama_model_hunyuan_moe : public llama_model_base { std::unique_ptr build_arch_graph(const llm_graph_params & params) const override; }; +struct llama_model_hy_v3 : public llama_model_base { + llama_model_hy_v3(const struct llama_model_params & params) : llama_model_base(params) {} + void load_arch_hparams(llama_model_loader & ml) override; + void load_arch_tensors(llama_model_loader & ml) override; + + struct graph : public llm_graph_context { + graph(const llama_model & model, const llm_graph_params & params); + }; + + std::unique_ptr build_arch_graph(const llm_graph_params & params) const override; +}; + struct llama_model_hunyuan_vl : public llama_model_base { llama_model_hunyuan_vl(const struct llama_model_params & params) : llama_model_base(params) {} From f370e03feaadc020c0f7b599e34ca3b5045070fc Mon Sep 17 00:00:00 2001 From: "Claude (Fable 5)" Date: Tue, 7 Jul 2026 20:39:54 +1200 Subject: [PATCH 2/9] model: add MTP/nextn speculative decoding for hy_v3 Loads the appended NextN/MTP decoder block(s) (blk...) and adds a LLM_GRAPH_TYPE_DECODER_MTP graph so hy_v3 GGUFs converted with nextn_predict_layers work with --spec-type draft-mtp. Semantics follow vLLM's hy_v3_mtp.py (HYV3MultiTokenPredictorLayer): enorm(embed) + hnorm(prev_hidden) -> concat -> eh_proj -> full hy_v3 decoder block -> final_layernorm (stored as nextn.shared_head_norm) -> shared LM head. The checkpoint carries no MTP embed_tokens or separate shared head; both fall back to the main model's tok_embd/lm_head, matching vLLM's shared-weight loading. Hidden-state chaining is post-final_layernorm, so t_h_nextn is exposed post-norm in both the main graph and the MTP graph (same convention as qwen35moe). The main KV cache excludes the MTP layers and the MTP draft context only contains them (same filter split as step35). Measured on Hy3 299B-A15B IQ1_M (MTP block at Q8_0), RTX 5090, -ngl 12, temp 0, 300 tokens: draft acceptance 85.8% (n_max=3, mean accepted length 2.11, per-position 0.878/0.224/0.010), decode 8.81 tok/s vs 7.89 tok/s baseline (+11.7%) CPU-bound; n_max=1 acceptance 93.3%. Implemented by Claude (Anthropic Fable 5), directed by Satinder Grewal (@satindergrewal). --- src/llama-model.cpp | 2 +- src/models/hy-v3.cpp | 207 ++++++++++++++++++++++++++++++++++++++++++- src/models/models.h | 4 + 3 files changed, 209 insertions(+), 4 deletions(-) diff --git a/src/llama-model.cpp b/src/llama-model.cpp index a4ad5a0fcf2..1a518a14501 100644 --- a/src/llama-model.cpp +++ b/src/llama-model.cpp @@ -2172,7 +2172,7 @@ llama_memory_i * llama_model::create_memory(const llama_memory_params & params, filter = [&](uint32_t il) { return il >= hparams.n_layer(); }; } - if (arch == LLM_ARCH_STEP35 && hparams.n_layer_nextn > 0) { + if ((arch == LLM_ARCH_STEP35 || arch == LLM_ARCH_HY_V3) && hparams.n_layer_nextn > 0) { if (params.ctx_type == LLAMA_CONTEXT_TYPE_MTP) { filter = [&](uint32_t il) { return il >= hparams.n_layer(); }; } else { diff --git a/src/models/hy-v3.cpp b/src/models/hy-v3.cpp index d92964f79f9..47a0beaf217 100644 --- a/src/models/hy-v3.cpp +++ b/src/models/hy-v3.cpp @@ -13,6 +13,10 @@ void llama_model_hy_v3::load_arch_hparams(llama_model_loader & ml) { hparams.expert_gating_func = LLAMA_EXPERT_GATING_FUNC_TYPE_SIGMOID; } + // NextN/MTP (HY V3): extra decoder block(s) appended beyond the main stack + ml.get_key(LLM_KV_NEXTN_PREDICT_LAYERS, hparams.n_layer_nextn, false); + GGML_ASSERT(hparams.n_layer_nextn < hparams.n_layer_all && "n_layer_nextn must be < n_layer_all"); + switch (hparams.n_layer()) { case 48: type = LLM_TYPE_30B_A3B; break; default: type = LLM_TYPE_UNKNOWN; @@ -22,7 +26,14 @@ void llama_model_hy_v3::load_arch_hparams(llama_model_loader & ml) { void llama_model_hy_v3::load_arch_tensors(llama_model_loader & ml) { LLAMA_LOAD_LOCALS; - GGML_UNUSED(ml); + const bool mtp_only = (hparams.n_layer_nextn > 0) && (ml.get_weight("blk.0.attn_norm.weight") == nullptr); + // Trunk-only: the GGUF declares MTP layers in metadata but the actual MTP + // tensors live in a separate file (e.g. user split target/draft). Mark + // MTP tensors NOT_REQUIRED so the trunk loads cleanly. + const std::string mtp_probe = "blk." + std::to_string(n_layer) + ".nextn.eh_proj.weight"; + const bool trunk_only = (hparams.n_layer_nextn > 0) && (ml.get_weight(mtp_probe.c_str()) == nullptr); + const int trunk_flags = mtp_only ? TENSOR_NOT_REQUIRED : 0; + const int mtp_flags = trunk_only ? TENSOR_NOT_REQUIRED : 0; tok_embd = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, 0); @@ -65,11 +76,30 @@ void llama_model_hy_v3::load_arch_tensors(llama_model_loader & ml) { }; for (int i = 0; i < n_layer; ++i) { - load_block(i, 0); + load_block(i, trunk_flags); + } + + // NextN/MTP block(s): a full hy_v3 decoder block plus the NextN projections. + for (int i = n_layer; i < n_layer_all; ++i) { + auto & layer = layers[i]; + + load_block(i, mtp_flags); + + layer.nextn.eh_proj = create_tensor(tn(LLM_TENSOR_NEXTN_EH_PROJ, "weight", i), { 2 * n_embd, n_embd }, mtp_flags); + layer.nextn.enorm = create_tensor(tn(LLM_TENSOR_NEXTN_ENORM, "weight", i), { n_embd }, mtp_flags); + layer.nextn.hnorm = create_tensor(tn(LLM_TENSOR_NEXTN_HNORM, "weight", i), { n_embd }, mtp_flags); + layer.nextn.embed_tokens = create_tensor(tn(LLM_TENSOR_NEXTN_EMBED_TOKENS, "weight", i), { n_embd, n_vocab }, TENSOR_NOT_REQUIRED); + layer.nextn.shared_head_head = create_tensor(tn(LLM_TENSOR_NEXTN_SHARED_HEAD_HEAD, "weight", i), { n_embd, n_vocab }, TENSOR_NOT_REQUIRED); + // hy_v3 stores the MTP block's trailing final_layernorm here (applied + // after the decoder block, before the shared LM head). + layer.nextn.shared_head_norm = create_tensor(tn(LLM_TENSOR_NEXTN_SHARED_HEAD_NORM, "weight", i), { n_embd }, TENSOR_NOT_REQUIRED); } } std::unique_ptr llama_model_hy_v3::build_arch_graph(const llm_graph_params & params) const { + if (params.gtype == LLM_GRAPH_TYPE_DECODER_MTP) { + return std::make_unique(*this, params); + } return std::make_unique(*this, params); } @@ -89,6 +119,7 @@ llama_model_hy_v3::graph::graph(const llama_model & model, const llm_graph_param const float kq_scale = 1.0f / sqrtf(float(n_embd_head)); + // MTP/NextN layers are loaded as extra decoder blocks but not executed in the main pass. for (int il = 0; il < n_layer; ++il) { ggml_tensor * inpSA = inpL; @@ -117,7 +148,7 @@ llama_model_hy_v3::graph::graph(const llama_model & model, const llm_graph_param cb(cur, "attn_out", il); } - if (il == n_layer - 1 && inp_out_ids) { + if (il == n_layer - 1 && inp_out_ids && cparams.embeddings_nextn_masked) { cur = ggml_get_rows(ctx0, cur, inp_out_ids); inpSA = ggml_get_rows(ctx0, inpSA, inp_out_ids); } @@ -178,6 +209,17 @@ llama_model_hy_v3::graph::graph(const llama_model & model, const llm_graph_param } cur = build_norm(inpL, model.output_norm, nullptr, LLM_NORM_RMS, -1); + + // Post-final-norm hidden state: what the MTP draft head's hnorm consumes. + // vLLM feeds the target model's normed output states, and the MTP layer + // itself returns final_layernorm(h), so the chained state is post-norm. + cb(cur, "h_nextn", -1); + res->t_h_nextn = cur; + + if (!cparams.embeddings_nextn_masked && inp_out_ids) { + cur = ggml_get_rows(ctx0, cur, inp_out_ids); + } + cb(cur, "result_norm", -1); res->t_embd = cur; @@ -187,3 +229,162 @@ llama_model_hy_v3::graph::graph(const llama_model & model, const llm_graph_param ggml_build_forward_expand(gf, cur); } + +// LLM_GRAPH_TYPE_DECODER_MTP draft head for HY V3 (MoE). +// Semantics mirror vLLM's HYV3MultiTokenPredictorLayer (hy_v3_mtp.py): +// enorm(embed) + hnorm(prev_hidden) -> concat(e, h) -> eh_proj -> +// hy_v3 decoder block -> final_layernorm (stored as nextn.shared_head_norm) -> +// shared LM head (the main model's lm_head; the checkpoint has no separate +// MTP head or MTP embeddings). +llama_model_hy_v3::graph_mtp::graph_mtp(const llama_model & model, const llm_graph_params & params) + : llm_graph_context(params) { + GGML_ASSERT(hparams.n_layer_nextn > 0 && "HY_V3 MTP requires n_layer_nextn > 0"); + + const int64_t n_embd_head = hparams.n_embd_head_v(); + GGML_ASSERT(n_embd_head == hparams.n_embd_head_k()); + GGML_ASSERT(n_embd_head == n_rot); + + const int il = hparams.n_layer() + cparams.nextn_layer_offset; + GGML_ASSERT(cparams.nextn_layer_offset >= 0 && + cparams.nextn_layer_offset < (int) hparams.n_layer_nextn && + "nextn_layer_offset out of range [0, n_layer_nextn)"); + const auto & layer = model.layers[il]; + + GGML_ASSERT(layer.nextn.eh_proj && "MTP block missing nextn.eh_proj"); + GGML_ASSERT(layer.nextn.enorm && "MTP block missing nextn.enorm"); + GGML_ASSERT(layer.nextn.hnorm && "MTP block missing nextn.hnorm"); + + auto inp = std::make_unique(hparams.n_embd); + + inp->tokens = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, n_tokens); + ggml_set_input(inp->tokens); + + inp->embd = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, hparams.n_embd, n_tokens); + ggml_set_input(inp->embd); + ggml_set_name(inp->embd, "mtp_h_input"); + + ggml_tensor * tok_embd_w = layer.nextn.embed_tokens ? layer.nextn.embed_tokens : model.tok_embd; + + ggml_tensor * h_input = inp->embd; + ggml_tensor * tok_embd = ggml_get_rows(ctx0, tok_embd_w, inp->tokens); + cb(tok_embd, "mtp_tok_embd", il); + + res->add_input(std::move(inp)); + + ggml_tensor * inp_pos = build_inp_pos(); + ggml_tensor * inp_out_ids = build_inp_out_ids(); + auto * inp_attn = build_attn_inp_kv(); + + ggml_tensor * h_norm = build_norm(h_input, layer.nextn.hnorm, nullptr, LLM_NORM_RMS, il); + cb(h_norm, "mtp_hnorm", il); + + ggml_tensor * e_norm = build_norm(tok_embd, layer.nextn.enorm, nullptr, LLM_NORM_RMS, il); + cb(e_norm, "mtp_enorm", il); + + ggml_tensor * concat = ggml_concat(ctx0, e_norm, h_norm, /*dim=*/ 0); + cb(concat, "mtp_concat", il); + + ggml_tensor * cur = build_lora_mm(layer.nextn.eh_proj, concat); + cb(cur, "mtp_eh_proj", il); + + ggml_tensor * inpSA = cur; + + // mtp_block: a full hy_v3 decoder layer (mirrors the trunk graph) + cur = build_norm(cur, layer.attn_norm, nullptr, LLM_NORM_RMS, il); + cb(cur, "mtp_attn_norm", il); + + { + ggml_tensor * rope_factors = model.get_rope_factors(cparams, il); + + auto [Qcur, Kcur, Vcur] = build_qkv(layer, cur, n_embd_head, n_head, n_head_kv, il); + + Qcur = build_norm(Qcur, layer.attn_q_norm, nullptr, LLM_NORM_RMS, il); + Kcur = build_norm(Kcur, layer.attn_k_norm, nullptr, LLM_NORM_RMS, il); + + Qcur = ggml_rope_ext(ctx0, Qcur, inp_pos, rope_factors, + n_rot, rope_type, n_ctx_orig, freq_base, freq_scale, + ext_factor, attn_factor, beta_fast, beta_slow); + Kcur = ggml_rope_ext(ctx0, Kcur, inp_pos, rope_factors, + n_rot, rope_type, n_ctx_orig, freq_base, freq_scale, + ext_factor, attn_factor, beta_fast, beta_slow); + + const float kq_scale = 1.0f / sqrtf(float(n_embd_head)); + + cur = build_attn(inp_attn, + layer.wo, layer.wo_b, layer.wo_s, + Qcur, Kcur, Vcur, nullptr, nullptr, nullptr, kq_scale, il); + cb(cur, "mtp_attn_out", il); + } + + ggml_tensor * ffn_inp = ggml_add(ctx0, cur, inpSA); + cb(ffn_inp, "mtp_ffn_inp", il); + + cur = build_norm(ffn_inp, layer.ffn_norm, nullptr, LLM_NORM_RMS, il); + cb(cur, "mtp_ffn_norm", il); + + if (layer.ffn_gate_inp == nullptr) { + cur = build_ffn(cur, + layer.ffn_up, layer.ffn_up_b, layer.ffn_up_s, + layer.ffn_gate, layer.ffn_gate_b, layer.ffn_gate_s, + layer.ffn_down, layer.ffn_down_b, layer.ffn_down_s, + nullptr, + LLM_FFN_SILU, LLM_FFN_PAR, il); + cb(cur, "mtp_ffn_dense_out", il); + } else { + ggml_tensor * moe_out = build_moe_ffn(cur, + layer.ffn_gate_inp, + layer.ffn_up_exps, + layer.ffn_gate_exps, + layer.ffn_down_exps, + layer.ffn_exp_probs_b, + n_expert, n_expert_used, + LLM_FFN_SILU, + hparams.expert_weights_norm, + hparams.expert_weights_scale, + (llama_expert_gating_func_type) hparams.expert_gating_func, + il, + nullptr, layer.ffn_gate_up_exps, + layer.ffn_up_exps_s, + layer.ffn_gate_exps_s, + layer.ffn_down_exps_s); + cb(moe_out, "mtp_ffn_moe_out", il); + + ggml_tensor * sh_out = build_ffn(cur, + layer.ffn_up_shexp, nullptr, layer.ffn_up_shexp_s, + layer.ffn_gate_shexp, nullptr, layer.ffn_gate_shexp_s, + layer.ffn_down_shexp, nullptr, layer.ffn_down_shexp_s, + nullptr, + LLM_FFN_SILU, LLM_FFN_PAR, il); + cb(sh_out, "mtp_ffn_shared_out", il); + + cur = ggml_add(ctx0, moe_out, sh_out); + cb(cur, "mtp_ffn_out", il); + } + + cur = ggml_add(ctx0, cur, ffn_inp); + cb(cur, "mtp_post_ffn", il); + + // final_layernorm applied after the decoder block, before the shared head. + // The post-norm hidden state seeds the next MTP step (matches vLLM, where + // HYV3MultiTokenPredictorLayer returns final_layernorm(h)). + ggml_tensor * head_norm_w = layer.nextn.shared_head_norm + ? layer.nextn.shared_head_norm + : model.output_norm; + GGML_ASSERT(head_norm_w && "HY_V3 MTP: missing both nextn.shared_head_norm and output_norm"); + cur = build_norm(cur, head_norm_w, nullptr, LLM_NORM_RMS, -1); + + cb(cur, "h_nextn", -1); + res->t_h_nextn = cur; + + cur = ggml_get_rows(ctx0, cur, inp_out_ids); + cb(cur, "mtp_shared_head_norm", -1); + + ggml_tensor * head_w = layer.nextn.shared_head_head ? layer.nextn.shared_head_head : model.output; + ggml_tensor * head_s = layer.nextn.shared_head_head ? layer.nextn.shared_head_head_s : model.output_s; + GGML_ASSERT(head_w && "HY_V3 MTP: missing LM head (nextn.shared_head_head or model.output)"); + cur = build_lora_mm(head_w, cur, head_s); + cb(cur, "result_output", -1); + + res->t_logits = cur; + ggml_build_forward_expand(gf, cur); +} diff --git a/src/models/models.h b/src/models/models.h index c47d0fa57b0..beab9f6bc7d 100644 --- a/src/models/models.h +++ b/src/models/models.h @@ -1738,6 +1738,10 @@ struct llama_model_hy_v3 : public llama_model_base { graph(const llama_model & model, const llm_graph_params & params); }; + struct graph_mtp : public llm_graph_context { + graph_mtp(const llama_model & model, const llm_graph_params & params); + }; + std::unique_ptr build_arch_graph(const llm_graph_params & params) const override; }; From 6105ebb79057885467914021834296d42266aa35 Mon Sep 17 00:00:00 2001 From: "Claude (Fable 5)" Date: Tue, 7 Jul 2026 20:42:55 +1200 Subject: [PATCH 3/9] convert: add HYV3ForCausalLM (hy_v3) with MTP block export Registers MODEL_ARCH.HY_V3 in gguf-py (arch name, tensor table incl the NEXTN_* draft-head tensors) and adds HYV3Model to conversion/hunyuan.py. The NextN/MTP block sits at model.layers.; block_count is extended to include it (Step35Model pattern, including the --mtp / --no-mtp toggles) and its trailing final_layernorm is renamed to shared_head.norm so the existing NEXTN_SHARED_HEAD_NORM mapping picks it up. Everything else (mlp.router.gate, mlp.expert_bias, mlp.shared_mlp.*) resolves through existing tensor_mapping entries. Vocab is GPT-2 BPE with the existing hunyuan-dense pre-tokenizer. Implemented by Claude (Anthropic Fable 5), directed by Satinder Grewal (@satindergrewal). --- conversion/__init__.py | 1 + conversion/hunyuan.py | 103 ++++++++++++++++++++++++++++++++++++++ gguf-py/gguf/constants.py | 33 ++++++++++++ 3 files changed, 137 insertions(+) diff --git a/conversion/__init__.py b/conversion/__init__.py index 02ea6385208..02102fac8d9 100644 --- a/conversion/__init__.py +++ b/conversion/__init__.py @@ -106,6 +106,7 @@ "HunYuanDenseV1ForCausalLM": "hunyuan", "HunYuanMoEV1ForCausalLM": "hunyuan", "HunYuanVLForConditionalGeneration": "hunyuan", + "HYV3ForCausalLM": "hunyuan", "IQuestCoderForCausalLM": "llama", "InternLM2ForCausalLM": "internlm", "InternLM3ForCausalLM": "internlm", diff --git a/conversion/hunyuan.py b/conversion/hunyuan.py index 537f023aa01..4d2545f8b47 100644 --- a/conversion/hunyuan.py +++ b/conversion/hunyuan.py @@ -1,6 +1,7 @@ from __future__ import annotations import json +import re from pathlib import Path from typing import Callable, Iterable, TYPE_CHECKING @@ -355,3 +356,105 @@ def set_gguf_parameters(self): self.gguf_writer.add_context_length(ctx_len) self.gguf_writer.add_rope_dimension_sections(list(self.rope_parameters["xdrope_section"])) + + +@ModelBase.register("HYV3ForCausalLM") +class HYV3Model(TextModel): + model_arch = gguf.MODEL_ARCH.HY_V3 + + # Trunk layer count, stashed before indexing so the classmethod + # filter_tensors can identify the appended MTP block(s) (mirrors + # Step35Model). + _n_main_layers: int | None = None + + def __init__(self, *args, **kwargs): + super().__init__(*args, **kwargs) + # NextN/MTP layers are appended past num_hidden_layers; extend the + # tensor map so the MTP block's tensors resolve to blk..* names. + n_nextn = int(self.hparams.get("num_nextn_predict_layers", 0)) + if n_nextn > 0 and not self.no_mtp: + self.block_count += n_nextn + self.tensor_map = gguf.get_tensor_name_map(self.model_arch, self.block_count) + + def index_tensors(self, remote_hf_model_id: str | None = None): + type(self)._n_main_layers = self.hparams["num_hidden_layers"] + return super().index_tensors(remote_hf_model_id=remote_hf_model_id) + + def set_vocab(self): + self._set_vocab_gpt2() + + def set_gguf_parameters(self): + super().set_gguf_parameters() + self.gguf_writer.add_expert_feed_forward_length(self.hparams["moe_intermediate_size"]) + self.gguf_writer.add_expert_shared_feed_forward_length( + self.hparams["moe_intermediate_size"] * self.hparams.get("num_shared_experts", 1) + ) + self.gguf_writer.add_expert_weights_norm(self.hparams.get("route_norm", True)) + self.gguf_writer.add_expert_weights_scale(float(self.hparams.get("router_scaling_factor", 1.0))) + # sigmoid router with expert selection bias + self.gguf_writer.add_expert_gating_func(gguf.ExpertGatingFuncType.SIGMOID) + + n_nextn = int(self.hparams.get("num_nextn_predict_layers", 0)) + if n_nextn > 0 and not self.no_mtp: + self.gguf_writer.add_nextn_predict_layers(n_nextn) + + @classmethod + def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None: + if (titem := super().filter_tensors(item)) is None: + return None + name, gen = titem + + # HY V3 appends the MTP block(s) past num_hidden_layers. + assert cls._n_main_layers is not None + is_mtp = (m := re.match(r"model\.layers\.(\d+)\.", name)) is not None and int(m.group(1)) >= cls._n_main_layers + + # --no-mtp: drop the appended MTP block(s) entirely. + if is_mtp and cls.no_mtp: + return None + # --mtp: keep ONLY MTP-block tensors plus the shared embeddings/norm/ + # lm_head (so the resulting GGUF carries just the draft head). + if cls.mtp_only and not is_mtp and name not in ( + "model.embed_tokens.weight", "model.norm.weight", "lm_head.weight", + ): + return None + + # The MTP block's trailing final_layernorm (applied after the decoder + # block, before the shared LM head) maps to nextn.shared_head_norm. + if is_mtp: + name = name.replace(".final_layernorm.", ".shared_head.norm.") + + return name, gen + + _experts: list[dict[str, Tensor]] | None = None + + def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]: + # merge the per-expert tensors into stacked 3d tensors + if name.startswith("model.layers.") and ".mlp.experts." in name: + n_experts = self.find_hparam(["num_local_experts", "num_experts"]) + assert bid is not None + + if self._experts is None: + self._experts = [{} for _ in range(self.block_count)] + + self._experts[bid][name] = data_torch + + if len(self._experts[bid]) >= n_experts * 3: + for w_name in ("down_proj", "gate_proj", "up_proj"): + datas: list[Tensor] = [] + for xid in range(n_experts): + ename = f"model.layers.{bid}.mlp.experts.{xid}.{w_name}.weight" + datas.append(self._experts[bid][ename]) + del self._experts[bid][ename] + + merged = torch.stack(datas, dim=0) + yield from super().modify_tensors(merged, f"model.layers.{bid}.mlp.experts.{w_name}.weight", bid) + return + + yield from super().modify_tensors(data_torch, name, bid) + + def prepare_tensors(self): + super().prepare_tensors() + if self._experts is not None: + experts = [k for d in self._experts for k in d.keys()] + if experts: + raise ValueError(f"Unprocessed experts: {experts}") diff --git a/gguf-py/gguf/constants.py b/gguf-py/gguf/constants.py index cd4cdef8991..9bd16d1f74e 100644 --- a/gguf-py/gguf/constants.py +++ b/gguf-py/gguf/constants.py @@ -512,6 +512,7 @@ class MODEL_ARCH(IntEnum): HUNYUAN_MOE = auto() HUNYUAN_DENSE = auto() HUNYUAN_VL = auto() + HY_V3 = auto() SMOLLM3 = auto() GPT_OSS = auto() LFM2 = auto() @@ -1093,6 +1094,7 @@ class MODEL_TENSOR(IntEnum): MODEL_ARCH.HUNYUAN_MOE: "hunyuan-moe", MODEL_ARCH.HUNYUAN_DENSE: "hunyuan-dense", MODEL_ARCH.HUNYUAN_VL: "hunyuan_vl", + MODEL_ARCH.HY_V3: "hy_v3", MODEL_ARCH.SMOLLM3: "smollm3", MODEL_ARCH.GPT_OSS: "gpt-oss", MODEL_ARCH.LFM2: "lfm2", @@ -3936,6 +3938,37 @@ class MODEL_TENSOR(IntEnum): MODEL_TENSOR.FFN_DOWN, MODEL_TENSOR.FFN_UP, ], + MODEL_ARCH.HY_V3: [ + MODEL_TENSOR.TOKEN_EMBD, + MODEL_TENSOR.OUTPUT_NORM, + MODEL_TENSOR.OUTPUT, + MODEL_TENSOR.ATTN_NORM, + MODEL_TENSOR.ATTN_Q, + MODEL_TENSOR.ATTN_Q_NORM, + MODEL_TENSOR.ATTN_K, + MODEL_TENSOR.ATTN_K_NORM, + MODEL_TENSOR.ATTN_V, + MODEL_TENSOR.ATTN_OUT, + MODEL_TENSOR.FFN_NORM, + MODEL_TENSOR.FFN_GATE, + MODEL_TENSOR.FFN_DOWN, + MODEL_TENSOR.FFN_UP, + MODEL_TENSOR.FFN_GATE_INP, + MODEL_TENSOR.FFN_EXP_PROBS_B, + MODEL_TENSOR.FFN_GATE_EXP, + MODEL_TENSOR.FFN_DOWN_EXP, + MODEL_TENSOR.FFN_UP_EXP, + MODEL_TENSOR.FFN_GATE_SHEXP, + MODEL_TENSOR.FFN_DOWN_SHEXP, + MODEL_TENSOR.FFN_UP_SHEXP, + # NextN/MTP tensors (draft head) + MODEL_TENSOR.NEXTN_EH_PROJ, + MODEL_TENSOR.NEXTN_EMBED_TOKENS, + MODEL_TENSOR.NEXTN_ENORM, + MODEL_TENSOR.NEXTN_HNORM, + MODEL_TENSOR.NEXTN_SHARED_HEAD_HEAD, + MODEL_TENSOR.NEXTN_SHARED_HEAD_NORM, + ], MODEL_ARCH.SMOLLM3: [ MODEL_TENSOR.TOKEN_EMBD, MODEL_TENSOR.OUTPUT_NORM, From 97181b716362545e53e72c095f7818957f5942c0 Mon Sep 17 00:00:00 2001 From: "Claude (Fable 5)" Date: Wed, 8 Jul 2026 00:43:53 +1200 Subject: [PATCH 4/9] tests : mark hy_v3 as MoE-mandatory in test-archs hy_v3 requires expert_feed_forward_length (the arch is MoE-only, same as step35), so the synthetic Dense variant failed to load: error loading model hyperparameters: key not found in model: hy_v3.expert_feed_forward_length Adding LLM_ARCH_HY_V3 to moe_mandatory() makes the test provide the expert KVs and skip the Dense variant, mirroring the existing STEP35 handling. Addresses the test-archs review note on the hy_v3 PR. The SWA-pattern branch STEP35 also sits in does not apply: hy_v3 never reads LLM_KV_ATTENTION_SLIDING_WINDOW_PATTERN. Implemented by Claude (Anthropic Fable 5), directed by Satinder Grewal (@satindergrewal). --- tests/test-llama-archs.cpp | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/test-llama-archs.cpp b/tests/test-llama-archs.cpp index f39abe773fc..2cdf3573982 100644 --- a/tests/test-llama-archs.cpp +++ b/tests/test-llama-archs.cpp @@ -346,6 +346,7 @@ static bool moe_mandatory(const llm_arch arch) { case LLM_ARCH_ERNIE4_5: case LLM_ARCH_ERNIE4_5_MOE: case LLM_ARCH_HUNYUAN_MOE: + case LLM_ARCH_HY_V3: case LLM_ARCH_OPENAI_MOE: case LLM_ARCH_LFM2MOE: case LLM_ARCH_SMALLTHINKER: From 692d171a996db64cab3f88ab26f3dce3a3af6a84 Mon Sep 17 00:00:00 2001 From: Piotr Wilkin Date: Mon, 6 Jul 2026 12:15:36 +0200 Subject: [PATCH 5/9] jinja : add str.format() support Assisted-by: Claude Fable 5 --- common/jinja/value.cpp | 90 ++++++++++++++++++++++++++++++++++++++++++ tests/test-jinja.cpp | 30 ++++++++++++++ 2 files changed, 120 insertions(+) diff --git a/common/jinja/value.cpp b/common/jinja/value.cpp index 5055ae9ac12..8f163292a4d 100644 --- a/common/jinja/value.cpp +++ b/common/jinja/value.cpp @@ -750,6 +750,96 @@ const func_builtins & value_string_t::get_builtins() const { res->val_str.mark_input_based_on(args.get_pos(0)->val_str); return res; }}, + {"format", [](const func_args & args) -> value { + value val_input = args.get_pos(0); + if (!is_val(val_input)) { + throw raised_exception("format() first argument must be a string"); + } + const jinja::string & fmt = val_input->as_string(); + const bool fmt_is_input = fmt.all_parts_are_input(); + + std::vector pos_args; + const auto & all_args = args.get_args(); + for (size_t i = 1; i < all_args.size(); ++i) { + if (!is_val(all_args[i])) { + pos_args.push_back(all_args[i]); + } + } + + const std::string str = fmt.str(); + jinja::string result; + std::string literal; + auto flush_literal = [&]() { + if (!literal.empty()) { + result.parts.push_back({fmt_is_input, literal}); + literal.clear(); + } + }; + + size_t auto_idx = 0; + bool manual = false; + bool automatic = false; + for (size_t i = 0; i < str.size(); ++i) { + const char c = str[i]; + if (c == '{' && i + 1 < str.size() && str[i + 1] == '{') { + literal += '{'; + ++i; + continue; + } + if (c == '}' && i + 1 < str.size() && str[i + 1] == '}') { + literal += '}'; + ++i; + continue; + } + if (c == '}') { + throw raised_exception("single '}' encountered in format string"); + } + if (c != '{') { + literal += c; + continue; + } + const size_t end = str.find('}', i + 1); + if (end == std::string::npos) { + throw raised_exception("single '{' encountered in format string"); + } + const std::string field = str.substr(i + 1, end - i - 1); + i = end; + if (field.find_first_of(":!.[") != std::string::npos) { + throw not_implemented_exception("format() conversions, format specs and nested fields not implemented"); + } + value arg; + if (field.empty()) { + if (manual) { + throw raised_exception("cannot switch from manual field specification to automatic field numbering"); + } + automatic = true; + if (auto_idx >= pos_args.size()) { + throw raised_exception("format() replacement index " + std::to_string(auto_idx) + " out of range"); + } + arg = pos_args[auto_idx++]; + } else if (std::all_of(field.begin(), field.end(), [](unsigned char ch) { return std::isdigit(ch); })) { + if (automatic) { + throw raised_exception("cannot switch from automatic field numbering to manual field specification"); + } + manual = true; + const size_t idx = std::stoul(field); + if (idx >= pos_args.size()) { + throw raised_exception("format() replacement index " + field + " out of range"); + } + arg = pos_args[idx]; + } else { + arg = args.get_kwarg(field, mk_val()); + if (arg->is_undefined()) { + throw raised_exception("format() missing keyword argument '" + field + "'"); + } + } + flush_literal(); + const jinja::string arg_str = arg->as_string(); + result.parts.insert(result.parts.end(), arg_str.parts.begin(), arg_str.parts.end()); + } + flush_literal(); + return mk_val(result); + }}, {"int", [](const func_args & args) -> value { value val_input = args.get_pos(0); value val_default = args.get_kwarg_or_pos("default", 1); diff --git a/tests/test-jinja.cpp b/tests/test-jinja.cpp index d8d1892a911..90bdbc445d5 100644 --- a/tests/test-jinja.cpp +++ b/tests/test-jinja.cpp @@ -1376,6 +1376,36 @@ static void test_string_methods(testing & t) { "bXnXna" ); + test_template(t, "string.format() auto numbering", + "{{ '<{}|{}>'.format(s, 42) }}", + {{"s", "hello"}}, + "" + ); + + test_template(t, "string.format() manual numbering", + "{{ '{1}-{0}-{1}'.format('a', 'b') }}", + json::object(), + "b-a-b" + ); + + test_template(t, "string.format() named fields", + "{{ '{name} is {age}'.format(name='Bob', age=7) }}", + json::object(), + "Bob is 7" + ); + + test_template(t, "string.format() escaped braces", + "{{ '{{}} {} {{x}}'.format('mid') }}", + json::object(), + "{} mid {x}" + ); + + test_template(t, "string.format() no fields", + "{{ 'plain'.format() }}", + json::object(), + "plain" + ); + test_template(t, "undefined|capitalize", "{{ arr|capitalize }}", json::object(), From 56142c5f87a59c0e34562259c7f7edf4613b0835 Mon Sep 17 00:00:00 2001 From: Piotr Wilkin Date: Mon, 6 Jul 2026 12:15:36 +0200 Subject: [PATCH 6/9] chat : detect separator between function name and arguments in autoparser Also disambiguate per-call vs section end markers by comparing their occurrence counts in single vs parallel tool call renders. Assisted-by: Claude Fable 5 --- common/chat-auto-parser-generator.cpp | 4 + common/chat-auto-parser.h | 7 +- common/chat-diff-analyzer.cpp | 30 ++++ models/templates/tencent-Hy3.jinja | 222 ++++++++++++++++++++++++++ tests/test-chat-auto-parser.cpp | 3 + 5 files changed, 263 insertions(+), 3 deletions(-) create mode 100644 models/templates/tencent-Hy3.jinja diff --git a/common/chat-auto-parser-generator.cpp b/common/chat-auto-parser-generator.cpp index 36aab7ecbe2..b10dea64af7 100644 --- a/common/chat-auto-parser-generator.cpp +++ b/common/chat-auto-parser-generator.cpp @@ -261,6 +261,10 @@ common_peg_parser analyze_tools::build_func_parser(common_chat_peg_builder & p, bool matched_atomic = false; common_peg_parser func_parser = p.eps(); + if (!function.args_separator.empty()) { + open = open + p.space() + p.literal(function.args_separator); + } + if (!function.name_suffix.empty()) { func_parser = open + call_id_section + p.space() + args; matched_atomic = true; diff --git a/common/chat-auto-parser.h b/common/chat-auto-parser.h index 9e8113f2442..d47b09dcf6f 100644 --- a/common/chat-auto-parser.h +++ b/common/chat-auto-parser.h @@ -192,9 +192,10 @@ struct tool_format_analysis { }; struct tool_function_analysis { - std::string name_prefix; // e.g., "", "\"", ":0" - std::string close; // e.g., "", "" (for tag-based) + std::string name_prefix; // e.g., "", "\"", ":0" + std::string args_separator; // e.g., "" (marker between function name and arguments) + std::string close; // e.g., "", "" (for tag-based) }; struct tool_arguments_analysis { diff --git a/common/chat-diff-analyzer.cpp b/common/chat-diff-analyzer.cpp index b166ee5a18f..9600efed195 100644 --- a/common/chat-diff-analyzer.cpp +++ b/common/chat-diff-analyzer.cpp @@ -259,6 +259,7 @@ void autoparser::analyze_template(const common_chat_template & tmpl) { LOG_DBG("per_call_end: '%s'\n", tools.format.per_call_end.c_str()); LOG_DBG("func_name_prefix: '%s'\n", tools.function.name_prefix.c_str()); LOG_DBG("func_name_suffix: '%s'\n", tools.function.name_suffix.c_str()); + LOG_DBG("func_args_separator: '%s'\n", tools.function.args_separator.c_str()); LOG_DBG("func_close: '%s'\n", tools.function.close.c_str()); LOG_DBG("call_id_prefix: '%s'\n", tools.call_id.prefix.c_str()); LOG_DBG("call_id_suffix: '%s'\n", tools.call_id.suffix.c_str()); @@ -302,6 +303,7 @@ void autoparser::collect_preserved_tokens() { add_token(tools.format.per_call_end); add_token(tools.function.name_prefix); add_token(tools.function.name_suffix); + add_token(tools.function.args_separator); add_token(tools.function.close); add_token(tools.arguments.start); add_token(tools.arguments.end); @@ -1051,6 +1053,23 @@ void analyze_tools::check_per_call_markers() { format.section_start.clear(); format.section_end.clear(); } + + if (!format.per_call_end.empty()) { + auto count_occurrences = [](const std::string & haystack, const std::string & needle) { + size_t count = 0; + for (size_t pos = haystack.find(needle); pos != std::string::npos; + pos = haystack.find(needle, pos + needle.size())) { + count++; + } + return count; + }; + size_t calls_one = count_occurrences(one_vs_two->output_A, format.per_call_end); + size_t calls_two = count_occurrences(one_vs_two->output_B, format.per_call_end); + if (calls_one > 0 && calls_one == calls_two) { + format.section_end = format.per_call_end; + format.per_call_end.clear(); + } + } } void analyze_tools::extract_function_markers() { @@ -1132,6 +1151,17 @@ void analyze_tools::extract_function_markers() { auto suf_result = suffix_parser.parse_and_extract(diff.suffix); if (suf_result.result.success()) { function.name_suffix += suf_result.tags["ext"]; + + auto arg_start = [&](common_peg_parser_builder &p) { + return p.marker() + p.space() + p.choice({ p.literal(ARG_FIRST), p.literal(ARG_SECOND) }); + }; + auto sep_parser = build_tagged_peg_parser([&](common_peg_parser_builder &p) { + return p.tag("sep", p.zero_or_more(p.negate(arg_start(p)) + p.any())) + arg_start(p); + }); + auto sep_result = sep_parser.parse_and_extract(diff.suffix.substr(suf_result.tags["ext"].size())); + if (sep_result.result.success()) { + function.args_separator = trim_whitespace(sep_result.tags["sep"]); + } } } diff --git a/models/templates/tencent-Hy3.jinja b/models/templates/tencent-Hy3.jinja new file mode 100644 index 00000000000..4fdb6c568c6 --- /dev/null +++ b/models/templates/tencent-Hy3.jinja @@ -0,0 +1,222 @@ +{#- ----------‑‑‑ special token variables ‑‑‑---------- -#} +{%- set HYTK = ':opensource' %} +{%- set eos_token = '<|hy_eos{}|>'.format(HYTK) %} +{%- set bos_token = '<|hy_begin_of_sentence{}|>'.format(HYTK) %} +{%- set pad_token = '<|hy_pad{}|>'.format(HYTK) %} +{%- set user_token = '<|hy_User{}|>'.format(HYTK) %} +{%- set assistant_token = '<|hy_Assistant{}|>'.format(HYTK) %} +{%- set think_begin_token = ''.format(HYTK) %} +{%- set think_end_token = ''.format(HYTK) %} +{%- set toolcalls_begin_token = ''.format(HYTK) %} +{%- set toolcalls_end_token = ''.format(HYTK) %} +{%- set toolcall_begin_token = ''.format(HYTK) %} +{%- set toolcall_end_token = ''.format(HYTK) %} +{%- set toolsep_token = ''.format(HYTK) %} +{%- set argkey_begin_token = ''.format(HYTK) %} +{%- set argkey_end_token = ''.format(HYTK) %} +{%- set argvalue_begin_token = ''.format(HYTK) %} +{%- set argvalue_end_token = ''.format(HYTK) %} +{%- set toolresponses_begin_token = ''.format(HYTK) %} +{%- set toolresponses_end_token = ''.format(HYTK) %} +{%- set toolresponse_begin_token = ''.format(HYTK) %} +{%- set toolresponse_end_token = ''.format(HYTK) %} +{%- set reasoning_mode_token = '<|reasoning_mode{}|>'.format(HYTK) %} + +{#- ----------‑‑‑ hyperparameters variables ‑‑‑---------- -#} +{%- if not add_generation_prompt is defined %} + {%- set add_generation_prompt = false %} +{%- endif %} +{%- if not preserved_thinking is defined %} + {%- if not tools %} + {%- set preserved_thinking = false %} + {%- else %} + {%- set preserved_thinking = true %} + {%- endif %} +{%- endif %} +{%- if not is_training is defined %} + {%- set is_training = false %} +{%- endif %} + +{%- if not reasoning_effort is defined %} + {%- set reasoning_effort = 'no_think' %} +{%- elif reasoning_effort not in ['high', 'low', 'no_think'] %} + {%- if reasoning_effort is none %} + {{- raise_exception('reasoning_effort error : None, should be no_think/low/high') }} + {%- else %} + {{- raise_exception('reasoning_effort error : ' + reasoning_effort + ', should be no_think/low/high') }} + {%- endif %} +{%- endif %} + +{%- if fallback_strategy is defined and fallback_strategy == 'reasoning_toolcall_retry' %} + {%- set reasoning_effort = 'high' %} + {%- set add_generation_prompt = false %} +{%- endif %} +{%- if not raw_last_assistant is defined %} + {%- set raw_last_assistant = false %} +{%- endif %} + +{%- macro visible_text(content) -%} + {%- if content is string -%} + {{- content }} + {%- elif content is iterable and content is not mapping -%} + {%- for item in content -%} + {%- if item is mapping and item.type == 'text' -%} + {{- item.text }} + {%- elif item is string -%} + {{- item }} + {%- endif -%} + {%- endfor -%} + {%- elif content is none -%} + {{- '' }} + {%- else -%} + {{- content }} + {%- endif -%} +{%- endmacro -%} + +{%- set ns = namespace(last_user_index=-1) %} +{%- set sp_ns = namespace(system_prompt='', is_first_sp=true) %} +{%- for message in messages %} + {%- if message['role'] == 'system' %} + {%- set sp_ns.system_prompt = sp_ns.system_prompt + visible_text(message['content']) %} + {%- endif %} + {%- if message['role'] == 'user' %} + {%- set ns.last_user_index = loop.index0 %} + {%- endif %} +{%- endfor %} +{%- if reasoning_effort is defined and reasoning_effort is string and reasoning_effort != '' and not tools %} + {%- set sp_ns.system_prompt = sp_ns.system_prompt + reasoning_mode_token + 'reasoning_effort:' + reasoning_effort %} +{%- endif %} +{{- bos_token }} +{{- sp_ns.system_prompt }} +{%- if tools %} + {%- if sp_ns.system_prompt != '' %} + {{- '\n\n# Tools\n\nYou may call one or more functions to assist with the user query.' }} + {%- else %} + {{- '# Tools\n\nYou may call one or more functions to assist with the user query.' }} + {%- endif %} + {{- '\n\nYou are provided with function signatures within XML tags:' }} + {{- '\n\n' }} + {%- for tool in tools %} + {%- if loop.index0 > 0 %} + {{- '\n' }} + {%- endif %} + {{- tool | tojson }} + {%- endfor %} + {{- '\n\n\n' }} + {{- 'For function call returns, you should first print ' + toolcalls_begin_token + '\n' }} + {{- 'For each function call, you should return object like:\n' }} + {{- toolcall_begin_token + '{function-name}' + toolsep_token + '\n' }} + {{- argkey_begin_token + '{arg-key-1}' + argkey_end_token + '\n' }} + {{- argvalue_begin_token + '{arg-value-1}' + argvalue_end_token + '\n' }} + {{- argkey_begin_token + '{arg-key-2}' + argkey_end_token + '\n' }} + {{- argvalue_begin_token + '{arg-value-2}' + argvalue_end_token + '\n' }} + {{- '...\n' }} + {{- toolcall_end_token + '\n' }} + {%- if reasoning_effort is defined and reasoning_effort is string and reasoning_effort != '' %} + {{- 'At the end of function call returns, you should print ' + toolcalls_end_token + reasoning_mode_token + 'reasoning_effort:' + reasoning_effort }} + {%- else %} + {{- 'At the end of function call returns, you should print ' + toolcalls_end_token }} + {%- endif %} +{%- endif %} + +{%- set prev_ns = namespace(is_tool=false, is_tool_first=true) %} +{%- set last_ns = namespace(last_is_assistant=false) %} +{%- for message in messages %} + {%- if message['role'] == 'user' %} + {%- if prev_ns.is_tool %} + {{- toolresponses_end_token }} + {%- endif %} + {{- user_token + visible_text(message['content']) }} + {%- set prev_ns.is_tool = false %} + {%- endif %} + {%- if message['role'] == 'assistant' %} + {%- if is_training %} + {%- if 'reasoning_content' in message and message['reasoning_content'] is string %} + {%- set rc = message['reasoning_content'] %} + {%- elif 'reasoning' in message and message['reasoning'] is string %} + {%- set rc = message['reasoning'] %} + {%- else %} + {%- set rc = none %} + {%- endif %} + {%- if rc is not none %} + {%- set content = think_begin_token + rc + think_end_token + visible_text(message['content']) %} + {%- else %} + {%- set content = think_begin_token + think_end_token + visible_text(message['content']) %} + {%- endif %} + {%- else %} + {%- if ((preserved_thinking is defined and preserved_thinking) or loop.index0 > ns.last_user_index) %} + {%- if 'reasoning_content' in message and message['reasoning_content'] is string %} + {%- set rc = message['reasoning_content'] %} + {%- elif 'reasoning' in message and message['reasoning'] is string %} + {%- set rc = message['reasoning'] %} + {%- else %} + {%- set rc = none %} + {%- endif %} + {%- if rc is not none %} + {%- set content = think_begin_token + rc + think_end_token + visible_text(message['content']) %} + {%- else %} + {%- set content = think_begin_token + think_end_token + visible_text(message['content']) %} + {%- endif %} + {%- else %} + {%- set content = think_begin_token + think_end_token + visible_text(message['content']) %} + {%- endif %} + {%- endif %} + {%- if prev_ns.is_tool %} + {{- toolresponses_end_token }} + {%- endif %} + {{- assistant_token }} + {%- if message['tool_calls'] is defined and message['tool_calls'] %} + {%- set prev_ns.is_tool_first = true %} + {{- content }} + {{- toolcalls_begin_token + '\n' }} + {%- for tool in message['tool_calls'] %} + {%- set arguments = tool['function']['arguments'] %} + {{- toolcall_begin_token + tool['function']['name'] + toolsep_token + '\n' }} + {%- for key, value in arguments.items() %} + {{- argkey_begin_token + key + argkey_end_token + '\n' }} + {%- if value is not string %} + {%- set value = value | tojson(ensure_ascii=False) %} + {%- endif %} + {{- argvalue_begin_token + value + argvalue_end_token + '\n' }} + {%- endfor %} + {{- toolcall_end_token + '\n' }} + {%- endfor %} + {{- toolcalls_end_token + eos_token }} + {%- else %} + {%- if loop.last and raw_last_assistant %} + {{- visible_text(message['content']) }} + {%- elif not loop.last or is_training %} + {{- content + eos_token }} + {%- else %} + {{- content }} + {%- endif %} + {%- endif %} + {%- set prev_ns.is_tool = false %} + {%- endif %} + {%- if message['role'] == 'tool' %} + {%- set prev_ns.is_tool = true %} + {%- if prev_ns.is_tool_first %} + {{- toolresponses_begin_token + '\n' }} + {%- set prev_ns.is_tool_first = false %} + {%- endif %} + {{- toolresponse_begin_token + '\n' + visible_text(message['content']) + '\n' + toolresponse_end_token + '\n' }} + {%- endif %} + {%- if loop.last and message['role'] == 'assistant' %} + {%- set last_ns.last_is_assistant = true %} + {%- endif %} + +{%- endfor %} +{%- if prev_ns.is_tool %} + {{- toolresponses_end_token }} +{%- endif %} +{%- if add_generation_prompt %} + {%- if not last_ns.last_is_assistant %} + {%- if reasoning_effort is defined and reasoning_effort in ['low', 'high'] %} + {{- assistant_token + think_begin_token }} + {%- elif reasoning_effort is defined and reasoning_effort == 'no_think' %} + {{- assistant_token + think_begin_token + think_end_token }} + {%- else %} + {{- assistant_token }} + {%- endif %} + {%- endif %} +{%- endif %} \ No newline at end of file diff --git a/tests/test-chat-auto-parser.cpp b/tests/test-chat-auto-parser.cpp index d15fdd2c022..78e42c65a50 100644 --- a/tests/test-chat-auto-parser.cpp +++ b/tests/test-chat-auto-parser.cpp @@ -1944,6 +1944,9 @@ static void test_role_markers_all_templates(testing & t) { // MiniMax M2: ]~b]{user|ai} { "MiniMax-M2.jinja", "]~b]user", "]~b]ai" }, + // HunYuan V3: <|hy_User:opensource|> / <|hy_Assistant:opensource|> + { "tencent-Hy3.jinja", "<|hy_User:opensource|>", "<|hy_Assistant:opensource|>" }, + // Nemotron Nano v2: {User|Assistant}; assistant marker // is followed by a prefilled block that gets included. { "NVIDIA-Nemotron-Nano-v2.jinja", "User", "Assistant" }, From d81d88627c656b532d0b53419ff225c61984f7ac Mon Sep 17 00:00:00 2001 From: Satinder Grewal Date: Mon, 13 Jul 2026 17:31:09 +1200 Subject: [PATCH 7/9] jinja: simplify format() to '{}' only; fix template dashes Assisted-by: Claude Opus 4.8 --- common/jinja/value.cpp | 36 ++++++------------------------ models/templates/tencent-Hy3.jinja | 4 ++-- 2 files changed, 9 insertions(+), 31 deletions(-) diff --git a/common/jinja/value.cpp b/common/jinja/value.cpp index 8f163292a4d..b5c6854e51a 100644 --- a/common/jinja/value.cpp +++ b/common/jinja/value.cpp @@ -777,8 +777,6 @@ const func_builtins & value_string_t::get_builtins() const { }; size_t auto_idx = 0; - bool manual = false; - bool automatic = false; for (size_t i = 0; i < str.size(); ++i) { const char c = str[i]; if (c == '{' && i + 1 < str.size() && str[i + 1] == '{') { @@ -804,35 +802,15 @@ const func_builtins & value_string_t::get_builtins() const { } const std::string field = str.substr(i + 1, end - i - 1); i = end; - if (field.find_first_of(":!.[") != std::string::npos) { - throw not_implemented_exception("format() conversions, format specs and nested fields not implemented"); + if (!field.empty()) { + // Hy3 template only uses simple '{}' placeholders; indexed, + // named, conversion and format-spec fields are unsupported. + throw not_implemented_exception("format() only supports simple '{}' placeholders"); } - value arg; - if (field.empty()) { - if (manual) { - throw raised_exception("cannot switch from manual field specification to automatic field numbering"); - } - automatic = true; - if (auto_idx >= pos_args.size()) { - throw raised_exception("format() replacement index " + std::to_string(auto_idx) + " out of range"); - } - arg = pos_args[auto_idx++]; - } else if (std::all_of(field.begin(), field.end(), [](unsigned char ch) { return std::isdigit(ch); })) { - if (automatic) { - throw raised_exception("cannot switch from automatic field numbering to manual field specification"); - } - manual = true; - const size_t idx = std::stoul(field); - if (idx >= pos_args.size()) { - throw raised_exception("format() replacement index " + field + " out of range"); - } - arg = pos_args[idx]; - } else { - arg = args.get_kwarg(field, mk_val()); - if (arg->is_undefined()) { - throw raised_exception("format() missing keyword argument '" + field + "'"); - } + if (auto_idx >= pos_args.size()) { + throw raised_exception("format() replacement index " + std::to_string(auto_idx) + " out of range"); } + value arg = pos_args[auto_idx++]; flush_literal(); const jinja::string arg_str = arg->as_string(); result.parts.insert(result.parts.end(), arg_str.parts.begin(), arg_str.parts.end()); diff --git a/models/templates/tencent-Hy3.jinja b/models/templates/tencent-Hy3.jinja index 4fdb6c568c6..7591102ca4c 100644 --- a/models/templates/tencent-Hy3.jinja +++ b/models/templates/tencent-Hy3.jinja @@ -1,4 +1,4 @@ -{#- ----------‑‑‑ special token variables ‑‑‑---------- -#} +{#- ------------- special token variables ------------- -#} {%- set HYTK = ':opensource' %} {%- set eos_token = '<|hy_eos{}|>'.format(HYTK) %} {%- set bos_token = '<|hy_begin_of_sentence{}|>'.format(HYTK) %} @@ -22,7 +22,7 @@ {%- set toolresponse_end_token = ''.format(HYTK) %} {%- set reasoning_mode_token = '<|reasoning_mode{}|>'.format(HYTK) %} -{#- ----------‑‑‑ hyperparameters variables ‑‑‑---------- -#} +{#- ------------- hyperparameters variables ------------- -#} {%- if not add_generation_prompt is defined %} {%- set add_generation_prompt = false %} {%- endif %} From cecbf5fb07680894eab7de4aa90ecc7309cc3ea6 Mon Sep 17 00:00:00 2001 From: Satinder Grewal Date: Tue, 14 Jul 2026 01:04:22 +1200 Subject: [PATCH 8/9] jinja: reduce format() scan to a single '{' + '}' check Per review: the only special sequence is '{' directly followed by '}'. Anything else after '{' throws not_implemented_exception, which the test harness already skips in cpp mode. Deletes brace-escape handling, field extraction and kwarg filtering. Co-Authored-By: Claude Fable 5 --- common/jinja/value.cpp | 45 ++++++------------------------------------ 1 file changed, 6 insertions(+), 39 deletions(-) diff --git a/common/jinja/value.cpp b/common/jinja/value.cpp index b5c6854e51a..32600e708a3 100644 --- a/common/jinja/value.cpp +++ b/common/jinja/value.cpp @@ -758,14 +758,6 @@ const func_builtins & value_string_t::get_builtins() const { const jinja::string & fmt = val_input->as_string(); const bool fmt_is_input = fmt.all_parts_are_input(); - std::vector pos_args; - const auto & all_args = args.get_args(); - for (size_t i = 1; i < all_args.size(); ++i) { - if (!is_val(all_args[i])) { - pos_args.push_back(all_args[i]); - } - } - const std::string str = fmt.str(); jinja::string result; std::string literal; @@ -776,43 +768,18 @@ const func_builtins & value_string_t::get_builtins() const { } }; - size_t auto_idx = 0; + size_t arg_idx = 1; // positional args follow the format string for (size_t i = 0; i < str.size(); ++i) { - const char c = str[i]; - if (c == '{' && i + 1 < str.size() && str[i + 1] == '{') { - literal += '{'; - ++i; + if (str[i] != '{') { + literal += str[i]; continue; } - if (c == '}' && i + 1 < str.size() && str[i + 1] == '}') { - literal += '}'; - ++i; - continue; - } - if (c == '}') { - throw raised_exception("single '}' encountered in format string"); - } - if (c != '{') { - literal += c; - continue; - } - const size_t end = str.find('}', i + 1); - if (end == std::string::npos) { - throw raised_exception("single '{' encountered in format string"); - } - const std::string field = str.substr(i + 1, end - i - 1); - i = end; - if (!field.empty()) { - // Hy3 template only uses simple '{}' placeholders; indexed, - // named, conversion and format-spec fields are unsupported. + if (i + 1 >= str.size() || str[i + 1] != '}') { throw not_implemented_exception("format() only supports simple '{}' placeholders"); } - if (auto_idx >= pos_args.size()) { - throw raised_exception("format() replacement index " + std::to_string(auto_idx) + " out of range"); - } - value arg = pos_args[auto_idx++]; + ++i; flush_literal(); - const jinja::string arg_str = arg->as_string(); + const jinja::string arg_str = args.get_pos(arg_idx++)->as_string(); result.parts.insert(result.parts.end(), arg_str.parts.begin(), arg_str.parts.end()); } flush_literal(); From a00f69af7b4966380272b3d8e73883294afa7371 Mon Sep 17 00:00:00 2001 From: Satinder Grewal Date: Tue, 14 Jul 2026 09:43:15 +1200 Subject: [PATCH 9/9] jinja: validate int() base to avoid MSVC CRT fail-fast std::stoi with a base outside {0, 2..36} trips the invalid parameter handler on the MSVC CRT (0xc0000409 fail-fast, not a catchable exception). The builtin-function fuzzer can generate such calls, e.g. '"s".int([1, 2, 3], -366)'; adding format() to the string builtins reshuffled the fuzz pairings and exposed this on the windows CI job. Co-Authored-By: Claude Fable 5 --- common/jinja/value.cpp | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/common/jinja/value.cpp b/common/jinja/value.cpp index 32600e708a3..870596d617f 100644 --- a/common/jinja/value.cpp +++ b/common/jinja/value.cpp @@ -790,6 +790,10 @@ const func_builtins & value_string_t::get_builtins() const { value val_default = args.get_kwarg_or_pos("default", 1); value val_base = args.get_kwarg_or_pos("base", 2); const int base = val_base->is_undefined() ? 10 : val_base->as_int(); + if (base != 0 && (base < 2 || base > 36)) { + // an out-of-range base makes std::stoi fail fast on the MSVC CRT instead of throwing + throw raised_exception("int() base must be 0 or between 2 and 36"); + } if (is_val(val_input) == false) { throw raised_exception("int() first argument must be a string"); }