Compare commits

...
23 Commits
Author SHA1 Message Date
Georgi Gerganov f84632951a wip 2026-05-05 09:36:07 +03:00
Georgi Gerganov 4567954ab0 Merge branch 'master' into pr/18039 2026-05-05 09:28:39 +03:00
Georgi Gerganov 069be0ae22 Merge branch 'master' into pr/18039 2026-05-04 21:42:27 +03:00
Georgi Gerganov 459b02f6c0 Merge branch 'master' into pr/18039 2026-05-02 18:08:25 +03:00
Georgi Gerganov cb8a3a93ec Merge branch 'master' into pr/18039 2026-04-30 10:08:10 +03:00
Georgi Gerganov 91b03e4c93 Merge branch 'master' into pr/18039 2026-04-24 14:20:12 +03:00
Georgi Gerganov 5bb2d50c0f Merge branch 'master' into pr/18039 2026-03-16 15:41:24 +02:00
ruixiangw 07e2c9707c eagle3: support --eagle3 in llama-cli 2026-02-28 00:33:54 +00:00
Georgi Gerganov b8ab2cc559 Merge branch 'master' into pr/18039 2026-02-23 14:47:03 +02:00
ruixiangw 9fea2434af eagle3: fix model convert code format 2026-02-20 18:05:49 +00:00
ruixiangw b3537924ef eagle3: fix model convert issue 2026-02-20 17:54:08 +00:00
Georgi Gerganov 5e224bc190 Merge branch 'master' into pr/18039 2026-02-09 15:40:01 +02:00
Georgi Gerganov 7d4c223943 Merge branch 'master' into HEAD 2026-02-05 15:58:16 +02:00
ruixiangw 7b78bfa984 eagle3: add support for RedHtAI eagle3 speculator series models 2026-01-16 00:54:14 +00:00
ruixiangw 75883cde73 eagle3: add support for gpt-oss-120B eagle3 2026-01-10 18:33:41 +00:00
ruixiangw 13a9f31de3 eagle3: make d2t mapping optional 2026-01-10 18:30:19 +00:00
ruixiangw 3da288d78d eagle3: load lm_head from target model if not in draft model when convert GGUF 2026-01-10 14:09:50 +00:00
ruixiangw 71ba283a65 add eagle3 support for Qwen3 MoE models 2026-01-09 11:54:28 +00:00
ruixiangw c0d99e65d2 add eagle3 support for Qwen3 series models 2026-01-08 23:49:06 +00:00
Georgi Gerganov 5a79c1900f eagle3 : improve naming 2025-12-17 15:49:03 +02:00
Georgi Gerganov 3e7f376b53 Merge branch 'master' into pr/18039 2025-12-17 12:09:41 +02:00
ruixiangw ac5667dcc6 fix eagle3 logits sync bug & remove ggml_set_sync() 2025-12-16 16:53:28 +00:00
ruixiangw 8fac4b1cc8 feat: add EAGLE3 speculative decoding support
EAGLE3 is an encoder-decoder based speculative decoding method:
- Extracts features from target model at specific layers
- Uses feature fusion layer to compress target features
- Generates draft tokens with single-layer decoder
- Maps draft vocabulary to target vocabulary via d2t tensor

Key changes:
- Add LLM_ARCH_EAGLE3 architecture
- Add EAGLE3 encoder/decoder graph (src/models/eagle3.cpp)
- Add feature extraction from target model layers
- Add g_embeddings handling for decoder input
- Add GGML_TENSOR_FLAG_SYNC for GPU synchronization
- Add --eagle3 flag for speculative-simple example
- Add EAGLE3 model conversion in convert_hf_to_gguf.py
2025-12-14 18:12:33 +00:00
25 changed files with 1031 additions and 66 deletions
+10 -2
View File
@@ -3067,7 +3067,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
[](common_params & params, bool value) {
params.use_jinja = value;
}
).set_examples({LLAMA_EXAMPLE_SERVER, LLAMA_EXAMPLE_COMPLETION, LLAMA_EXAMPLE_CLI, LLAMA_EXAMPLE_MTMD}).set_env("LLAMA_ARG_JINJA"));
).set_examples({LLAMA_EXAMPLE_SERVER, LLAMA_EXAMPLE_COMPLETION, LLAMA_EXAMPLE_CLI, LLAMA_EXAMPLE_MTMD, LLAMA_EXAMPLE_SPECULATIVE}).set_env("LLAMA_ARG_JINJA"));
add_opt(common_arg(
{"--reasoning-format"}, "FORMAT",
"controls whether thought tags are allowed and/or extracted from the response, and in which format they're returned; one of:\n"
@@ -3123,7 +3123,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
[](common_params & params, const std::string & value) {
params.chat_template = value;
}
).set_examples({LLAMA_EXAMPLE_COMPLETION, LLAMA_EXAMPLE_CLI, LLAMA_EXAMPLE_SERVER, LLAMA_EXAMPLE_MTMD}).set_env("LLAMA_ARG_CHAT_TEMPLATE"));
).set_examples({LLAMA_EXAMPLE_COMPLETION, LLAMA_EXAMPLE_CLI, LLAMA_EXAMPLE_SERVER, LLAMA_EXAMPLE_MTMD, LLAMA_EXAMPLE_SPECULATIVE}).set_env("LLAMA_ARG_CHAT_TEMPLATE"));
add_opt(common_arg(
{"--chat-template-file"}, "JINJA_TEMPLATE_FILE",
string_format(
@@ -3455,6 +3455,14 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
params.speculative.draft.cache_type_v = kv_cache_type_from_str(value);
}
).set_env("LLAMA_ARG_SPEC_DRAFT_CACHE_TYPE_V"));
// TODO: rename:
add_opt(common_arg(
{"--eagle3"},
"use EAGLE3 speculative decoding with the draft model",
[](common_params & params) {
params.speculative.draft.eagle3 = true;
}
).set_examples({LLAMA_EXAMPLE_SPECULATIVE, LLAMA_EXAMPLE_CLI}));
add_opt(common_arg(
{"--spec-draft-override-tensor", "-otd", "--override-tensor-draft"}, "<tensor name pattern>=<buffer type>,...",
"override tensor buffer type for draft model", [](common_params & params, const std::string & value) {
+4 -1
View File
@@ -307,10 +307,13 @@ struct common_params_speculative_draft {
common_params_model mparams;
llama_model * model = nullptr; // a llama_model that can be shared by multiple speculative contexts
llama_model * model = nullptr; // a llama_model that can be shared by multiple speculative contexts
llama_model * model_tgt = nullptr; // the target model
llama_context_params cparams; // these are the parameters for the draft llama_context
bool eagle3 = false; // use EAGLE3 speculative decoding
int32_t n_ctx = 0; // draft context size
int32_t n_gpu_layers = -1; // number of layers to store in VRAM for the draft model (-1 - use default)
+185 -23
View File
@@ -48,6 +48,7 @@ struct common_speculative_config {
const common_params_speculative & p = common_params_speculative{}) : type(t), params(p) {}
};
static bool common_speculative_are_compatible(
const llama_model * model_tgt,
const llama_model * model_dft) {
@@ -240,7 +241,9 @@ struct common_speculative_state_draft : public common_speculative_state {
~common_speculative_state_draft() override {
llama_perf_context_print(ctx_dft);
llama_free(ctx_dft);
if (ctx_dft) {
llama_free(ctx_dft);
}
common_sampler_free(smpl);
@@ -291,11 +294,11 @@ struct common_speculative_state_draft : public common_speculative_state {
auto * spec = this;
auto & batch = spec->batch;
auto & ctx_tgt = spec->ctx_tgt;
auto & ctx_dft = spec->ctx_dft;
auto & smpl = spec->smpl;
auto & prompt_dft = spec->prompt_dft;
auto & batch = spec->batch;
auto & ctx_tgt = spec->ctx_tgt;
auto & ctx_dft = spec->ctx_dft;
auto & smpl = spec->smpl;
auto & prompt_dft = spec->prompt_dft;
auto * mem_dft = llama_get_memory(ctx_dft);
@@ -567,7 +570,52 @@ struct common_speculative_state_draft : public common_speculative_state {
};
struct common_speculative_state_eagle3 : public common_speculative_state {
common_speculative_state_eagle3(enum common_speculative_type type) : common_speculative_state(type) {}
llama_context * ctx_tgt;
common_sampler * smpl;
llama_batch batch;
struct llama_context * ctx_dft_enc = nullptr;
struct llama_context * ctx_dft_dec = nullptr;
int32_t eagle3_n_past = 0; // number of verified positions in decoder KV cache
common_speculative_state_eagle3(
enum common_speculative_type type,
llama_context * ctx_tgt,
llama_context * ctx_dft_enc,
llama_context * ctx_dft_dec)
: common_speculative_state(type)
, ctx_tgt(ctx_tgt)
, ctx_dft_enc(ctx_dft_enc)
, ctx_dft_dec(ctx_dft_dec)
{
batch = llama_batch_init(llama_n_batch(ctx_dft_dec), 0, 1);
// Initialize sampler for EAGLE3 decoder
common_params_sampling params;
params.no_perf = false;
params.top_k = 10; // set 1 for greedy sampling (argmax) to match vLLM's default behavior but >1 always gets higher acceptance rate for eagle3
params.samplers = { COMMON_SAMPLER_TYPE_TOP_K };
smpl = common_sampler_init(llama_get_model(ctx_dft_dec), params);
}
~common_speculative_state_eagle3() override {
llama_perf_context_print(ctx_dft_dec);
if (ctx_dft_dec) {
llama_free(ctx_dft_dec);
}
if (ctx_dft_enc) {
llama_free(ctx_dft_enc);
}
common_sampler_free(smpl);
llama_batch_free(batch);
}
void begin(const llama_tokens & prompt) override {
GGML_UNUSED(prompt);
@@ -577,12 +625,97 @@ struct common_speculative_state_eagle3 : public common_speculative_state {
const common_params_speculative & params,
const llama_tokens & prompt_tgt,
llama_token id_last,
llama_tokens & draft_tokens) override {
// TODO: implement
GGML_UNUSED(params);
GGML_UNUSED(prompt_tgt);
GGML_UNUSED(id_last);
GGML_UNUSED(draft_tokens);
llama_tokens & result) override {
auto * spec = this;
auto & batch = spec->batch;
auto & ctx_tgt = spec->ctx_tgt;
auto & ctx_dft_enc = spec->ctx_dft_enc;
auto & ctx_dft_dec = spec->ctx_dft_dec;
auto & smpl = spec->smpl;
//result = gen_eagle3_draft(spec, params, prompt_tgt, id_last);
const int n_embd = llama_model_n_embd(llama_get_model(ctx_dft_enc));
const int n = (int)prompt_tgt.size();
const int n_new = n - spec->eagle3_n_past;
GGML_ASSERT(n >= 1 && "prompt_tgt is empty");
GGML_ASSERT(n_new >= 1 && "must have at least 1 new token");
// Clear draft positions from decoder KV cache [n_past, inf)
llama_memory_seq_rm(llama_get_memory(ctx_dft_dec), 0, spec->eagle3_n_past, -1);
// Encoder: features → g_embeddings
const float * features = llama_get_eagle3_target_features(ctx_tgt);
GGML_ASSERT(features && "no target features");
llama_batch enc_batch = {
/*.n_tokens =*/ n_new,
/*.token =*/ nullptr,
/*.embd =*/ const_cast<float*>(features),
/*.pos =*/ nullptr,
/*.n_seq_id =*/ nullptr,
/*.seq_id =*/ nullptr,
/*.logits =*/ nullptr,
};
GGML_ASSERT(llama_encode(ctx_dft_enc, enc_batch) == 0);
const float * g_embd = llama_get_embeddings(ctx_dft_enc);
GGML_ASSERT(g_embd && "encoder output failed");
// Decoder batch: process new tokens with KV cache reuse
llama_set_eagle3_g_embeddings(ctx_dft_dec, g_embd, n_embd, n_new);
common_batch_clear(batch);
for (int i = 0; i < n_new; i++) {
const int pos = spec->eagle3_n_past + i;
const llama_token tok = (pos < n - 1) ? prompt_tgt[pos + 1] : id_last;
common_batch_add(batch, tok, pos, {0}, true);
}
GGML_ASSERT(llama_decode(ctx_dft_dec, batch) == 0);
spec->eagle3_n_past = n; // update verified positions
// Sample draft tokens
result.clear();
common_sampler_reset(smpl);
// Sample and check probability (consistent with standard speculative decoding)
auto sample_and_check = [&](int idx) -> bool {
common_sampler_sample(smpl, ctx_dft_dec, idx);
const auto * cur_p = common_sampler_get_candidates(smpl, true);
const llama_token id = cur_p->data[0].id;
common_sampler_accept(smpl, id, true);
result.push_back(id);
return cur_p->data[0].p >= params.draft.p_min;
};
// First draft token from batch decode
if (!sample_and_check(n_new - 1)) {
return;
}
// Autoregressive: use prenorm as g_embd (-1 = last output)
const float * prenorm = llama_get_embeddings_ith(ctx_dft_dec, -1);
for (int i = 1; i < params.draft.n_max; i++) {
GGML_ASSERT(prenorm && "prenorm failed");
llama_set_eagle3_g_embeddings(ctx_dft_dec, prenorm, n_embd, 1);
common_batch_clear(batch);
common_batch_add(batch, result.back(), n - 1 + i, {0}, true);
GGML_ASSERT(llama_decode(ctx_dft_dec, batch) == 0);
prenorm = llama_get_embeddings_ith(ctx_dft_dec, -1);
if (!sample_and_check(0)) {
break;
}
}
}
void accept(uint16_t n_accepted) override {
@@ -975,11 +1108,35 @@ common_speculative * common_speculative_init(
common_params_speculative & params,
llama_context * ctx_tgt) {
llama_context * ctx_dft = nullptr;
llama_context * ctx_dft_enc = nullptr;
llama_context * ctx_dft_dec = nullptr;
if (params.draft.model) {
ctx_dft = llama_init_from_model(params.draft.model, params.draft.cparams);
if (ctx_dft == nullptr) {
LOG_ERR("%s", "failed to create draft context\n");
return nullptr;
if (params.draft.eagle3) {
llama_context_params params_enc = params.draft.cparams;
params_enc.target_model = nullptr;
params_enc.embeddings = true;
ctx_dft_enc = llama_init_from_model(params.draft.model, params_enc);
if (!ctx_dft_enc) {
LOG_ERR("failed to create EAGLE3 encoder context\n");
return nullptr;
}
llama_context_params params_dec = params.draft.cparams;
params_dec.target_model = params.draft.model_tgt;
params_dec.embeddings = true;
ctx_dft_dec = llama_init_from_model(params.draft.model, params_dec);
if (!ctx_dft_dec) {
LOG_ERR("failed to create EAGLE3 decoder context\n");
return nullptr;
}
} else {
ctx_dft = llama_init_from_model(params.draft.model, params.draft.cparams);
if (ctx_dft == nullptr) {
LOG_ERR("%s", "failed to create draft context\n");
return nullptr;
}
}
}
@@ -987,7 +1144,7 @@ common_speculative * common_speculative_init(
std::vector<common_speculative_config> configs = {}; // list of speculative configs to try
{
bool has_draft = !params.draft.mparams.path.empty();
bool has_draft_eagle3 = false; // TODO PR-18039: if params.speculative.eagle3
bool has_draft_eagle3 = params.draft.eagle3;
bool has_ngram_cache = (params.type == COMMON_SPECULATIVE_TYPE_NGRAM_CACHE);
bool has_ngram_simple = (params.type == COMMON_SPECULATIVE_TYPE_NGRAM_SIMPLE);
@@ -1029,10 +1186,11 @@ common_speculative * common_speculative_init(
configs.push_back(common_speculative_config(COMMON_SPECULATIVE_TYPE_NGRAM_CACHE, params));
}
if (has_draft) {
configs.push_back(common_speculative_config(COMMON_SPECULATIVE_TYPE_DRAFT, params));
}
if (has_draft_eagle3) {
configs.push_back(common_speculative_config(COMMON_SPECULATIVE_TYPE_EAGLE3, params));
if (has_draft_eagle3) {
configs.push_back(common_speculative_config(COMMON_SPECULATIVE_TYPE_EAGLE3, params));
} else {
configs.push_back(common_speculative_config(COMMON_SPECULATIVE_TYPE_DRAFT, params));
}
}
}
@@ -1055,7 +1213,11 @@ common_speculative * common_speculative_init(
break;
}
case COMMON_SPECULATIVE_TYPE_EAGLE3: {
impls.push_back(std::make_unique<common_speculative_state_eagle3>(config.type));
impls.push_back(std::make_unique<common_speculative_state_eagle3>(config.type,
/* .ctx_tgt = */ ctx_tgt,
/* .ctx_dft_enc = */ ctx_dft_enc,
/* .ctx_dft_dec = */ ctx_dft_dec
));
break;
}
case COMMON_SPECULATIVE_TYPE_NGRAM_SIMPLE: {
+146 -1
View File
@@ -97,6 +97,7 @@ class ModelBase:
metadata_override: Path | None
dir_model_card: Path
remote_hf_model_id: str | None
target_model_dir: Path | None
# subclasses should define this!
model_arch: gguf.MODEL_ARCH
@@ -116,7 +117,7 @@ class ModelBase:
split_max_tensors: int = 0, split_max_size: int = 0, dry_run: bool = False,
small_first_shard: bool = False, hparams: dict[str, Any] | None = None, remote_hf_model_id: str | None = None,
disable_mistral_community_chat_template: bool = False,
sentence_transformers_dense_modules: bool = False,
sentence_transformers_dense_modules: bool = False, target_model_dir: Path | None = None,
fuse_gate_up_exps: bool = False):
if type(self) is ModelBase or \
type(self) is TextModel or \
@@ -136,6 +137,7 @@ class ModelBase:
self.dry_run = dry_run
self.remote_hf_model_id = remote_hf_model_id
self.sentence_transformers_dense_modules = sentence_transformers_dense_modules
self.target_model_dir = target_model_dir
self.fuse_gate_up_exps = fuse_gate_up_exps
self._gate_exp_buffer: dict[int, Tensor] = {}
self._up_exp_buffer: dict[int, Tensor] = {}
@@ -2812,6 +2814,9 @@ class StableLMModel(TextModel):
"VLlama3ForCausalLM",
"LlavaForConditionalGeneration",
"VoxtralForConditionalGeneration",
"LlamaForCausalLMEagle3",
"Eagle3Speculator",
"Eagle3DraftModel",
"IQuestCoderForCausalLM",
"LlamaModel")
class LlamaModel(TextModel):
@@ -2826,7 +2831,60 @@ class LlamaModel(TextModel):
hparams = ModelBase.load_hparams(self.dir_model, is_mistral_format=False)
self.origin_hf_arch = hparams.get('architectures', [None])[0]
# detect EAGLE-3 llama checkpoint
if "draft_vocab_size" in self.hparams and self.hparams["num_hidden_layers"] == 1:
self.is_eagle3 = True
self.model_arch = gguf.MODEL_ARCH.EAGLE3
logger.info("Detected EAGLE-3 draft model, switching to EAGLE3 architecture")
# Re-initialize tensor_map with EAGLE3 architecture
self.tensor_map = gguf.get_tensor_name_map(self.model_arch, self.block_count)
# Update gguf_writer architecture
self.gguf_writer.arch = gguf.MODEL_ARCH_NAMES[self.model_arch]
self.gguf_writer.add_architecture()
if not hasattr(self, 'target_model_dir') or not self.target_model_dir:
raise ValueError(
"EAGLE3 model requires --target-model-dir to be specified. "
"Please provide the path to the target model directory to read config.json"
)
# Read both EAGLE3 raw config and target model config
with open(self.dir_model / "config.json", 'r', encoding='utf-8') as f:
eagle3_raw_config = json.load(f)
with open(self.target_model_dir / "config.json", 'r', encoding='utf-8') as f:
target_config = json.load(f)
# EAGLE3 extract_layers
target_num_layers = target_config["num_hidden_layers"]
extract_layers = [2, target_num_layers // 2, target_num_layers - 3]
logger.info(f"EAGLE3: extract_layers = {extract_layers} (target model has {target_num_layers} layers)")
self.gguf_writer.add_array(f"{self.gguf_writer.arch}.extract_layers", extract_layers)
# EAGLE3 target_hidden_size: prefer EAGLE3 config, fallback to target config
if "target_hidden_size" in eagle3_raw_config and eagle3_raw_config["target_hidden_size"] is not None:
target_hidden_size = eagle3_raw_config["target_hidden_size"]
logger.info(f"EAGLE3: target_hidden_size = {target_hidden_size} (from EAGLE3 config)")
else:
target_hidden_size = target_config["hidden_size"]
logger.info(f"EAGLE3: target_hidden_size = {target_hidden_size} (from target model config)")
self.gguf_writer.add_uint32(f"{self.gguf_writer.arch}.target_hidden_size", target_hidden_size)
# Eagle3Speculator norm_before_residual specific handling
norm_before_residual = eagle3_raw_config.get("norm_before_residual", False)
logger.info(f"EAGLE3: norm_before_residual = {norm_before_residual} (from EAGLE3 config)")
self.gguf_writer.add_bool(f"{self.gguf_writer.arch}.norm_before_residual", norm_before_residual)
def set_vocab(self):
# For EAGLE-3 models, use tokenizer from target model if provided
if hasattr(self, 'is_eagle3') and self.is_eagle3:
if self.target_model_dir is None:
raise ValueError(
"EAGLE-3 draft model requires --target-model-dir to be specified. "
"Please provide the path to the target model directory containing the tokenizer."
)
logger.info(f"EAGLE-3: Using tokenizer from target model: {self.target_model_dir}")
# Temporarily swap dir_model to load tokenizer from target model
original_dir_model = self.dir_model
self.dir_model = self.target_model_dir
if self.origin_hf_arch == "GlmasrModel":
return self._set_vocab_glmedge()
@@ -2870,6 +2928,10 @@ class LlamaModel(TextModel):
if self.hparams.get("vocab_size", 32000) == 49152:
self.gguf_writer.add_add_bos_token(False)
# Restore original dir_model for EAGLE-3
if hasattr(self, 'is_eagle3') and self.is_eagle3:
self.dir_model = original_dir_model
def set_gguf_parameters(self):
super().set_gguf_parameters()
hparams = self.hparams
@@ -2905,7 +2967,55 @@ class LlamaModel(TextModel):
_experts: list[dict[str, Tensor]] | None = None
def index_tensors(self, remote_hf_model_id: str | None = None) -> dict[str, Callable[[], Tensor]]:
tensors = super().index_tensors(remote_hf_model_id)
# Handle Eagle3Speculator nested config
if "transformer_layer_config" in self.hparams:
self.hparams = {**self.hparams, **self.hparams["transformer_layer_config"]}
# EAGLE-3 detection: check hparams directly (before self.is_eagle3 is set)
if "draft_vocab_size" in self.hparams and self.hparams["num_hidden_layers"] == 1:
logger.info("EAGLE-3: Renaming midlayer.* or layers.0.* to model.layers.0.*")
new_tensors = {}
# EAGLE-3: rename midlayer.* to model.layers.0.* for compatibility with llama model
for name, gen in tensors.items():
if name.startswith("midlayer."):
new_name = "model.layers.0." + name[len("midlayer."):]
new_tensors[new_name] = gen
elif name.startswith("layers.0."): # layers.0.* -> model.layers.0.* (Eagle3Speculator format)
new_name = "model." + name
new_tensors[new_name] = gen
else:
new_tensors[name] = gen
return new_tensors
else:
return tensors
def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
# Eagle-3 llama checkpoint special handling
if hasattr(self, 'is_eagle3') and self.is_eagle3:
# Eagle-3 llama checkpoint special weights handling
# fc.weight: feature fusion layer
if name == "fc.weight":
yield (name, data_torch)
return
# d2t: draft to target vocabulary mapping
elif name == "d2t":
# Skip parent class processing (store for manual handling in prepare_tensors)
if not hasattr(self, '_eagle3_int_tensors'):
self._eagle3_int_tensors = {}
self._eagle3_int_tensors[name] = data_torch
return
# t2d: target to draft vocabulary mapping (not used, skip completely)
elif name == "t2d":
return
# hidden_norm: EAGLE-3 specific layer normalization
elif name == "model.layers.0.hidden_norm.weight":
yield ("blk.0.hidden_norm.weight", data_torch)
return
n_head = self.find_hparam(["n_heads", "num_attention_heads"])
n_kv_head = self.find_hparam(["n_kv_heads", "num_key_value_heads"])
@@ -2975,6 +3085,17 @@ class LlamaModel(TextModel):
yield from super().modify_tensors(data_torch, name, bid)
def generate_extra_tensors(self) -> Iterable[tuple[str, Tensor]]:
# EAGLE3: If no lm_head in draft model, load from target model
if hasattr(self, 'is_eagle3') and self.is_eagle3 and "lm_head.weight" not in self.model_tensors:
from safetensors import safe_open
for sf_file in self.target_model_dir.glob("*.safetensors"):
with safe_open(sf_file, framework="pt") as f:
if "lm_head.weight" in f.keys():
lm_head = f.get_tensor("lm_head.weight")
logger.info(f"EAGLE3: No lm_head in draft model, loaded lm_head from {sf_file.name}, shape = {lm_head.shape}")
yield ("output.weight", lm_head)
break
if rope_params := self.rope_parameters.get("full_attention", self.rope_parameters):
if rope_params.get("rope_type", '').lower() == "llama3":
base = rope_params.get("rope_theta", 10000.0)
@@ -3005,8 +3126,26 @@ class LlamaModel(TextModel):
yield (self.format_tensor_name(gguf.MODEL_TENSOR.ROPE_FREQS), torch.tensor(rope_factors, dtype=torch.float32))
def prepare_tensors(self):
# EAGLE-3: collect original dtypes BEFORE parent class converts them to F32
eagle3_original_dtypes = {}
if hasattr(self, 'is_eagle3') and self.is_eagle3:
for name, data_torch in self.get_tensors():
if name == "d2t":
eagle3_original_dtypes[name] = data_torch.dtype
super().prepare_tensors()
if hasattr(self, 'is_eagle3') and self.is_eagle3 and hasattr(self, '_eagle3_int_tensors'):
for name, data_torch in self._eagle3_int_tensors.items():
old_dtype = eagle3_original_dtypes.get(name, data_torch.dtype)
# Keep as int64 to match original torch tensor dtype
data = data_torch.to(torch.int64).numpy()
data_qtype = gguf.GGMLQuantizationType.I64
shape_str = f"{{{', '.join(str(n) for n in reversed(data.shape))}}}"
logger.info(f"{name + ',':<30} {old_dtype} --> {data_qtype.name}, shape = {shape_str}")
self.gguf_writer.add_tensor(name, data, raw_dtype=data_qtype)
if self._experts is not None:
# flatten `list[dict[str, Tensor]]` into `list[str]`
experts = [k for d in self._experts for k in d.keys()]
@@ -13244,6 +13383,7 @@ class LazyTorchTensor(gguf.LazyBase):
torch.float16: np.float16,
torch.float32: np.float32,
torch.uint8: np.uint8,
torch.int64: np.int64,
}
# only used when byteswapping data. Only correct size is needed
@@ -13406,6 +13546,10 @@ def parse_args() -> argparse.Namespace:
"--no-tensor-first-split", action="store_true",
help="do not add tensors to the first split (disabled by default)"
)
parser.add_argument(
"--target-model-dir", type=str, default=None,
help="directory containing target model tokenizer (for EAGLE-3 draft models that don't have their own tokenizer)",
)
parser.add_argument(
"--metadata", type=Path,
help="Specify the path for an authorship metadata override file"
@@ -13590,6 +13734,7 @@ def main() -> None:
small_first_shard=args.no_tensor_first_split,
remote_hf_model_id=hf_repo_id, disable_mistral_community_chat_template=disable_mistral_community_chat_template,
sentence_transformers_dense_modules=args.sentence_transformers_dense_modules,
target_model_dir=Path(args.target_model_dir) if args.target_model_dir else None,
fuse_gate_up_exps=args.fuse_gate_up_exps
)
@@ -4,6 +4,7 @@
#include "speculative.h"
#include "log.h"
#include "llama.h"
#include "chat.h"
#include <clocale>
#include <cstdio>
@@ -103,13 +104,53 @@ int main(int argc, char ** argv) {
return 1;
}
params.speculative.draft.model = model_dft.get();
params.speculative.draft.model_tgt = model_tgt;
params.speculative.draft.model = model_dft.get();
params.speculative.draft.cparams = common_context_params_to_llama(params_dft);
if (params.speculative.draft.eagle3) {
llama_set_eagle3(ctx_tgt, model_dft.get());
}
}
// Apply chat template for EAGLE3 if available which can increase the acceptance rate
std::string prompt = params.prompt;
if (params.speculative.draft.eagle3) {
auto chat_templates = common_chat_templates_init(model_tgt, params.chat_template);
if (common_chat_templates_was_explicit(chat_templates.get())) {
std::vector<common_chat_msg> chat_msgs;
common_chat_msg user_msg;
user_msg.role = "user";
user_msg.content = params.prompt;
chat_msgs.push_back(user_msg);
common_chat_templates_inputs inputs;
inputs.messages = chat_msgs;
inputs.add_generation_prompt = true;
prompt = common_chat_templates_apply(chat_templates.get(), inputs).prompt;
LOG_INF("%s: EAGLE3 chat template applied\n", __func__);
}
}
int n_predict = 0;
int n_drafted = 0;
int n_accept = 0;
// used to determine end of generation
bool has_eos = false;
// ================================================
// everything until here is standard initialization
// the relevant stuff for speculative decoding starts here
const auto t_enc_start = ggml_time_us();
// target model sampling context
common_sampler_ptr smpl(common_sampler_init(model_tgt, params.sampling));
// Tokenize the prompt
std::vector<llama_token> inp;
inp = common_tokenize(ctx_tgt, params.prompt, true, true);
inp = common_tokenize(ctx_tgt, prompt, true, true);
if (llama_n_ctx(ctx_tgt) < (uint32_t) inp.size()) {
LOG_ERR("%s: the prompt exceeds the context size (%d tokens, ctx %d)\n", __func__, (int) inp.size(), llama_n_ctx(ctx_tgt));
@@ -129,33 +170,39 @@ int main(int argc, char ** argv) {
LOG("%s", common_token_to_piece(ctx_tgt, id).c_str());
}
int n_predict = 0;
int n_drafted = 0;
int n_accept = 0;
// used to determine end of generation
bool has_eos = false;
// ================================================
// everything until here is standard initialization
// the relevant stuff for speculative decoding starts here
const auto t_enc_start = ggml_time_us();
// target model sampling context
common_sampler_ptr smpl(common_sampler_init(model_tgt, params.sampling));
// eval the prompt
llama_decode(ctx_tgt, llama_batch_get_one(inp.data(), inp.size() - 1));
llama_token id_last;
llama_tokens prompt_tgt;
int n_past;
// note: keep the last token separate!
llama_token id_last = inp.back();
// TODO: simplify
if (params.speculative.draft.eagle3) {
// Target model decodes full prompt and sample first token and intermediate features are extracted
llama_decode(ctx_tgt, llama_batch_get_one(inp.data(), inp.size()));
// all tokens currently in the target context
llama_tokens prompt_tgt(inp.begin(), inp.end() - 1);
prompt_tgt.reserve(llama_n_ctx(ctx_tgt));
id_last = common_sampler_sample(smpl.get(), ctx_tgt, -1);
common_sampler_accept(smpl.get(), id_last, true);
LOG("%s", common_token_to_piece(ctx_tgt, id_last).c_str());
n_predict++;
int n_past = inp.size() - 1;
// all tokens currently in the target context
prompt_tgt.assign(inp.begin(), inp.end());
prompt_tgt.reserve(llama_n_ctx(ctx_tgt));
n_past = inp.size();
} else {
llama_decode(ctx_tgt, llama_batch_get_one(inp.data(), inp.size() - 1));
// note: keep the last token separate!
id_last = inp.back();
// all tokens currently in the target context
prompt_tgt.assign(inp.begin(), inp.end() - 1);
prompt_tgt.reserve(llama_n_ctx(ctx_tgt));
n_past = inp.size() - 1;
}
// init the speculator
const auto & params_spec = params.speculative;
+30
View File
@@ -152,6 +152,9 @@ class Keys:
SWIGLU_CLAMP_SHEXP = "{arch}.swiglu_clamp_shexp"
DENSE_FEAT_IN_SIZE = "{arch}.{dense}_feat_in"
DENSE_FEAT_OUT_SIZE = "{arch}.{dense}_feat_out"
EAGLE3_EXTRACT_LAYERS = "{arch}.extract_layers"
EAGLE3_TARGET_HIDDEN_SIZE = "{arch}.target_hidden_size"
EAGLE3_NORM_BEFORE_RESIDUAL = "{arch}.norm_before_residual"
class Attention:
HEAD_COUNT = "{arch}.attention.head_count"
@@ -489,6 +492,7 @@ class MODEL_ARCH(IntEnum):
RND1 = auto()
PANGU_EMBED = auto()
MISTRAL3 = auto()
EAGLE3 = auto()
MISTRAL4 = auto()
PADDLEOCR = auto()
MIMO2 = auto()
@@ -844,6 +848,10 @@ class MODEL_TENSOR(IntEnum):
NEXTN_HNORM = auto()
NEXTN_SHARED_HEAD_HEAD = auto()
NEXTN_SHARED_HEAD_NORM = auto()
# EAGLE3 specific tensors
EAGLE3_FC = auto() # feature fusion layer
EAGLE3_HIDDEN_NORM = auto() # hidden normalization
EAGLE3_D2T = auto() # draft to target vocabulary mapping
# lfm2 audio
A_ENC_NORM_CONV = auto()
A_ENC_LINEAR_POS = auto()
@@ -976,6 +984,7 @@ MODEL_ARCH_NAMES: dict[MODEL_ARCH, str] = {
MODEL_ARCH.RND1: "rnd1",
MODEL_ARCH.PANGU_EMBED: "pangu-embedded",
MODEL_ARCH.MISTRAL3: "mistral3",
MODEL_ARCH.EAGLE3: "eagle3",
MODEL_ARCH.MISTRAL4: "mistral4",
MODEL_ARCH.PADDLEOCR: "paddleocr",
MODEL_ARCH.MIMO2: "mimo2",
@@ -1340,6 +1349,9 @@ TENSOR_NAMES: dict[MODEL_TENSOR, str] = {
MODEL_TENSOR.NEXTN_HNORM: "blk.{bid}.nextn.hnorm",
MODEL_TENSOR.NEXTN_SHARED_HEAD_HEAD: "blk.{bid}.nextn.shared_head_head",
MODEL_TENSOR.NEXTN_SHARED_HEAD_NORM: "blk.{bid}.nextn.shared_head_norm",
MODEL_TENSOR.EAGLE3_FC: "fc",
MODEL_TENSOR.EAGLE3_HIDDEN_NORM: "blk.{bid}.hidden_norm",
MODEL_TENSOR.EAGLE3_D2T: "d2t",
}
MODEL_TENSORS: dict[MODEL_ARCH, list[MODEL_TENSOR]] = {
@@ -3742,6 +3754,24 @@ MODEL_TENSORS: dict[MODEL_ARCH, list[MODEL_TENSOR]] = {
MODEL_TENSOR.FFN_DOWN_EXP,
MODEL_TENSOR.FFN_UP_EXP,
],
MODEL_ARCH.EAGLE3: [
MODEL_TENSOR.TOKEN_EMBD,
MODEL_TENSOR.OUTPUT_NORM,
MODEL_TENSOR.OUTPUT,
MODEL_TENSOR.ROPE_FREQS,
MODEL_TENSOR.ATTN_NORM,
MODEL_TENSOR.ATTN_Q,
MODEL_TENSOR.ATTN_K,
MODEL_TENSOR.ATTN_V,
MODEL_TENSOR.ATTN_OUT,
MODEL_TENSOR.FFN_NORM,
MODEL_TENSOR.FFN_GATE,
MODEL_TENSOR.FFN_DOWN,
MODEL_TENSOR.FFN_UP,
MODEL_TENSOR.EAGLE3_FC,
MODEL_TENSOR.EAGLE3_HIDDEN_NORM,
MODEL_TENSOR.EAGLE3_D2T,
],
MODEL_ARCH.MISTRAL4: [
MODEL_TENSOR.TOKEN_EMBD,
MODEL_TENSOR.OUTPUT_NORM,
+13
View File
@@ -126,6 +126,7 @@ static const std::map<llm_arch, const char *> LLM_ARCH_NAMES = {
{ LLM_ARCH_RND1, "rnd1" },
{ LLM_ARCH_PANGU_EMBED, "pangu-embedded" },
{ LLM_ARCH_MISTRAL3, "mistral3" },
{ LLM_ARCH_EAGLE3, "eagle3" },
{ LLM_ARCH_MISTRAL4, "mistral4" },
{ LLM_ARCH_PADDLEOCR, "paddleocr" },
{ LLM_ARCH_MIMO2, "mimo2" },
@@ -284,6 +285,10 @@ static const std::map<llm_kv, const char *> LLM_KV_NAMES = {
{ LLM_KV_CLASSIFIER_OUTPUT_LABELS, "%s.classifier.output_labels" },
{ LLM_KV_EAGLE3_EXTRACT_LAYERS, "%s.extract_layers" },
{ LLM_KV_EAGLE3_TARGET_HIDDEN_SIZE, "%s.target_hidden_size" },
{ LLM_KV_EAGLE3_NORM_BEFORE_RESIDUAL, "%s.norm_before_residual" },
{ LLM_KV_SHORTCONV_L_CACHE, "%s.shortconv.l_cache" },
// sentence-transformers dense modules feature dims
{ LLM_KV_DENSE_2_FEAT_IN, "%s.dense_2_feat_in" },
@@ -547,6 +552,10 @@ static const std::map<llm_tensor, const char *> LLM_TENSOR_NAMES = {
{ LLM_TENSOR_INDEXER_PROJ, "blk.%d.indexer.proj" },
{ LLM_TENSOR_INDEXER_ATTN_K, "blk.%d.indexer.attn_k" },
{ LLM_TENSOR_INDEXER_ATTN_Q_B, "blk.%d.indexer.attn_q_b" },
// EAGLE-3 specific layers
{ LLM_TENSOR_EAGLE3_HIDDEN_NORM, "blk.%d.hidden_norm" },
{ LLM_TENSOR_EAGLE3_FC, "fc" },
{ LLM_TENSOR_EAGLE3_D2T, "d2t" },
};
// declare information about the model weight tensors:
@@ -767,6 +776,10 @@ static const std::map<llm_tensor, llm_tensor_info> LLM_TENSOR_INFOS = {
// Nemotron 3 Super
{LLM_TENSOR_FFN_LATENT_DOWN, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL}},
{LLM_TENSOR_FFN_LATENT_UP, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL}},
// EAGLE-3 tensors
{LLM_TENSOR_EAGLE3_FC, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL_MAT}},
{LLM_TENSOR_EAGLE3_HIDDEN_NORM, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL}},
{LLM_TENSOR_EAGLE3_D2T, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_GET_ROWS}},
};
LLM_KV::LLM_KV(llm_arch arch, const char * suffix) : arch(arch), suffix(suffix) {}
+8
View File
@@ -138,6 +138,7 @@ enum llm_arch {
LLM_ARCH_MAINCODER,
LLM_ARCH_KIMI_LINEAR,
LLM_ARCH_UNKNOWN,
LLM_ARCH_EAGLE3,
};
enum llm_kv {
@@ -326,6 +327,10 @@ enum llm_kv {
LLM_KV_CLASSIFIER_OUTPUT_LABELS,
LLM_KV_EAGLE3_EXTRACT_LAYERS,
LLM_KV_EAGLE3_TARGET_HIDDEN_SIZE,
LLM_KV_EAGLE3_NORM_BEFORE_RESIDUAL,
LLM_KV_SHORTCONV_L_CACHE,
LLM_KV_XIELU_ALPHA_N,
@@ -554,6 +559,9 @@ enum llm_tensor {
LLM_TENSOR_NEXTN_HNORM,
LLM_TENSOR_NEXTN_SHARED_HEAD_HEAD,
LLM_TENSOR_NEXTN_SHARED_HEAD_NORM,
LLM_TENSOR_EAGLE3_FC, // eagle3: feature fusion layer
LLM_TENSOR_EAGLE3_HIDDEN_NORM, // eagle3: additional normalization layer
LLM_TENSOR_EAGLE3_D2T, // eagle3: draft to target vocabulary mapping
};
enum llm_tensor_layer {
+128 -6
View File
@@ -65,6 +65,8 @@ llama_context::llama_context(
cparams.cb_eval = params.cb_eval;
cparams.cb_eval_user_data = params.cb_eval_user_data;
cparams.output_layer_inp.resize(hparams.n_layer, false);
// Initialize backend samplers here so they are part of the sampling graph
// before the reserve passes run later in this function. This avoids a later
// re-reserve when graph nodes change.
@@ -1168,6 +1170,16 @@ bool llama_context::set_adapter_cvec(
return res;
}
void llama_context::set_output_layer_inp(uint32_t layer_id, bool enable) {
LLAMA_LOG_DEBUG("%s: layer_id = %d, enable = %d\n", __func__, layer_id, enable);
GGML_ASSERT(layer_id < model.hparams.n_layer);
cparams.output_layer_inp[layer_id] = enable;
sched_need_reserve = true;
}
llm_graph_result * llama_context::process_ubatch(const llama_ubatch & ubatch, llm_graph_type gtype, llama_memory_context_i * mctx, ggml_status & ret) {
if (mctx && !mctx->apply()) {
LLAMA_LOG_ERROR("%s: failed to apply memory context\n", __func__);
@@ -1225,6 +1237,14 @@ llm_graph_result * llama_context::process_ubatch(const llama_ubatch & ubatch, ll
// FIXME this call causes a crash if any model inputs were not used in the graph and were therefore not allocated
res->set_inputs(&ubatch);
// EAGLE3: Fill g_embeddings for decoder input
if (model.arch == LLM_ARCH_EAGLE3 && gtype == LLM_GRAPH_TYPE_DECODER && !eagle3.g_embeddings.empty()) {
ggml_tensor * g_embd = ggml_graph_get_tensor(gf, "inp_g_embeddings");
if (g_embd) {
ggml_backend_tensor_set(g_embd, eagle3.g_embeddings.data(), 0, ggml_nbytes(g_embd));
}
}
//LLAMA_LOG_INFO("graph set inputs time: %.3f ms\n", (ggml_time_us() - t_start_us)/1000.0);
}
@@ -1336,8 +1356,15 @@ int llama_context::encode(const llama_batch & batch_inp) {
GGML_ASSERT(embd.data != nullptr);
const uint32_t n_embd_out = hparams.n_embd_out();
GGML_ASSERT(n_tokens*n_embd_out <= (int64_t) embd.size);
ggml_backend_tensor_get_async(backend_embd, t_embd, embd.data, 0, n_tokens*n_embd_out*sizeof(float));
if (model.arch == LLM_ARCH_EAGLE3) {
// g_embeddings are stored temporarily in embd buffer
const int64_t out_embd = hparams.n_embd;
GGML_ASSERT(n_tokens * out_embd <= (int64_t) embd.size);
ggml_backend_tensor_get_async(backend_embd, t_embd, embd.data, 0, n_tokens * out_embd * sizeof(float));
} else {
GGML_ASSERT(n_tokens*n_embd_out <= (int64_t) embd.size);
ggml_backend_tensor_get_async(backend_embd, t_embd, embd.data, 0, n_tokens*n_embd_out*sizeof(float));
}
} break;
case LLAMA_POOLING_TYPE_MEAN:
case LLAMA_POOLING_TYPE_CLS:
@@ -1730,7 +1757,8 @@ int llama_context::decode(const llama_batch & batch_inp) {
auto * t_logits = res->get_logits();
auto * t_embd = cparams.embeddings ? res->get_embd() : nullptr;
if (t_embd && res->get_embd_pooled()) {
// For EAGLE3, don't override t_embd with t_embd_pooled - we need the prenorm value during eagle3 decoder autoregressive generation
if (t_embd && res->get_embd_pooled() && model.arch != LLM_ARCH_EAGLE3) {
t_embd = res->get_embd_pooled();
}
@@ -1745,7 +1773,40 @@ int llama_context::decode(const llama_batch & batch_inp) {
if (n_outputs) {
GGML_ASSERT( n_outputs_prev + n_outputs <= n_outputs_all);
GGML_ASSERT((n_outputs_prev + n_outputs)*n_vocab <= (int64_t) logits.size);
ggml_backend_tensor_get_async(backend_res, t_logits, logits_out, 0, n_outputs*n_vocab*sizeof(float));
// EAGLE3: Map draft vocab to target vocab
if (model.arch == LLM_ARCH_EAGLE3 && model.d2t) {
static thread_local std::vector<int64_t> eagle3_d2t_map;
static thread_local std::vector<float> eagle3_draft_logits;
const int64_t draft_vocab_size = t_logits->ne[0];
const uint32_t last_idx = n_outputs - 1;
// Load d2t mapping once (on first call)
if (eagle3_d2t_map.empty()) {
eagle3_d2t_map.resize(model.d2t->ne[0]);
ggml_backend_tensor_get(model.d2t, eagle3_d2t_map.data(), 0, eagle3_d2t_map.size() * sizeof(int64_t));
}
// Read only the last token's draft logits
eagle3_draft_logits.resize(draft_vocab_size);
const size_t last_offset = last_idx * draft_vocab_size * sizeof(float);
ggml_backend_tensor_get_async(backend_res, t_logits, eagle3_draft_logits.data(), last_offset, draft_vocab_size * sizeof(float));
synchronize();
// Map only the last token's draft logits to target vocab
float * last_logits_out = logits_out + last_idx * n_vocab;
std::fill(last_logits_out, last_logits_out + n_vocab, -std::numeric_limits<float>::infinity());
for (int64_t j = 0; j < draft_vocab_size; j++) {
const int64_t target_id = j + eagle3_d2t_map[j];
GGML_ASSERT(target_id >= 0 && target_id < n_vocab);
last_logits_out[target_id] = eagle3_draft_logits[j];
}
} else {
ggml_backend_tensor_get_async(backend_res, t_logits, logits_out, 0, n_outputs*n_vocab*sizeof(float));
}
}
}
@@ -1904,7 +1965,6 @@ uint32_t llama_context::output_reserve(int32_t n_outputs) {
has_embd = true;
}
size_t backend_float_count = 0;
size_t backend_token_count = 0;
@@ -2121,7 +2181,16 @@ ggml_cgraph * llama_context::graph_reserve(
auto * res = gf_res_reserve.get();
const auto gparams = graph_params(res, ubatch, mctx, LLM_GRAPH_TYPE_DEFAULT);
// EAGLE3: auto-detect encoder (embeddings+no target_model) or decoder (has target_model)
llm_graph_type gtype = LLM_GRAPH_TYPE_DEFAULT;
if (model.arch == LLM_ARCH_EAGLE3) {
if (cparams.embeddings && model.target_tok_embd == nullptr) {
gtype = LLM_GRAPH_TYPE_ENCODER;
} else if (model.target_tok_embd != nullptr) {
gtype = LLM_GRAPH_TYPE_DECODER;
}
}
const auto gparams = graph_params(res, ubatch, mctx, gtype);
res->reset();
@@ -2162,6 +2231,7 @@ llm_graph_params llama_context::graph_params(
/*.loras =*/ loras.get(),
/*.mctx =*/ mctx,
/*.cross =*/ &cross,
/*.eagle3 =*/ &eagle3,
/*.samplers =*/ sampling.samplers,
/*.n_outputs =*/ n_outputs,
/*.cb =*/ graph_get_cb(),
@@ -2224,6 +2294,54 @@ llm_graph_cb llama_context::graph_get_cb() const {
};
}
void llama_context::extract_eagle3_features(const llama_ubatch & ubatch) {
const int64_t n_tokens = ubatch.n_tokens;
const int64_t n_embd = model.hparams.n_embd;
const size_t n_layers = eagle3.extract_tensors.size();
// Allocate storage for concatenated features
const int64_t n_embd_concat = n_embd * n_layers;
eagle3.target_features.resize(n_embd_concat * n_tokens);
// Temporary buffer to hold layer features before transposing
static thread_local std::vector<float> temp_layer_features;
temp_layer_features.resize(n_embd * n_tokens);
LLAMA_LOG_DEBUG("%s: Start to extract EAGLE3 features: %zu layers, %lld tokens, %lld embd\n",
__func__, n_layers, (long long)n_tokens, (long long)n_embd);
// Extract each layer's features and interleave into token-major layout
for (size_t layer_idx = 0; layer_idx < n_layers; ++layer_idx) {
ggml_tensor * tensor = eagle3.extract_tensors[layer_idx];
GGML_ASSERT(tensor != nullptr && "EAGLE3 extraction tensor is null");
// Get the backend where this tensor is stored
ggml_backend_t backend = ggml_backend_sched_get_tensor_backend(sched.get(), tensor);
GGML_ASSERT(backend != nullptr && "EAGLE3 tensor has no backend");
// Verify tensor shape: should be [n_embd, n_tokens]
GGML_ASSERT(tensor->ne[0] == n_embd && tensor->ne[1] == n_tokens &&
"EAGLE3 extraction tensor has unexpected shape");
// Get layer features to temp buffer
const size_t size_bytes = n_embd * n_tokens * sizeof(float);
ggml_backend_tensor_get_async(backend, tensor, temp_layer_features.data(), 0, size_bytes);
ggml_backend_sched_synchronize(sched.get());
// Then copy to correct position in target_features
// target_features layout: [token_0_all_layers, token_1_all_layers, ...]
// Each token has [layer_0_embd, layer_1_embd, layer_2_embd]
for (int64_t token_idx = 0; token_idx < n_tokens; ++token_idx) {
// Source: temp_layer_features[token_idx * n_embd ... (token_idx + 1) * n_embd - 1]
const float * src = temp_layer_features.data() + token_idx * n_embd;
// Dest: target_features[token_idx * n_embd_concat + layer_idx * n_embd]
float * dest = eagle3.target_features.data() + token_idx * n_embd_concat + layer_idx * n_embd;
std::memcpy(dest, src, n_embd * sizeof(float));
}
}
}
//
// state save/load
//
@@ -3779,3 +3897,7 @@ void llama_opt_epoch(
llama_memory_breakdown llama_get_memory_breakdown(const struct llama_context * ctx) {
return ctx->memory_breakdown();
}
void llama_set_output_layer_inp(struct llama_context * ctx, uint32_t layer_id, bool enable) {
ctx->set_output_layer_inp(layer_id, enable);
}
+14
View File
@@ -121,6 +121,8 @@ struct llama_context {
int32_t il_start,
int32_t il_end);
void set_output_layer_inp(uint32_t layer_id, bool enable);
// process a single ubatch with a specific graph type
// if memory_context is provided, it will be applied first to the context's memory
// ret contains the status of the graph computation
@@ -238,6 +240,12 @@ public:
ggml_cgraph * graph_reserve(
uint32_t n_tokens, uint32_t n_seqs, uint32_t n_outputs, const llama_memory_context_i * mctx, bool split_only = false, size_t * sizes = nullptr);
// EAGLE3: Get pointer to target model features extracted for EAGLE3 encoder
const float * get_eagle3_target_features() const;
// EAGLE3: Set g_embeddings from encoder output for decoder input
void set_eagle3_g_embeddings(const float * g_embd, int32_t n_embd, int32_t n_tokens);
bool set_sampler(llama_seq_id seq_id, llama_sampler * sampler);
private:
@@ -249,6 +257,9 @@ private:
llm_graph_cb graph_get_cb() const;
// EAGLE3: Extract intermediate layer features from target model
void extract_eagle3_features(const llama_ubatch & ubatch);
// TODO: read/write lora adapters and cvec
size_t state_write_data(llama_io_write_i & io);
size_t state_read_data (llama_io_read_i & io);
@@ -269,6 +280,9 @@ private:
llama_cross cross; // TODO: tmp for handling cross-attention - need something better probably
mutable llama_eagle3 eagle3; // EAGLE3 draft model support - stores features from target model
// mutable because it's modified during graph building (const function)
std::unique_ptr<llama_memory_i> memory;
// decode output (2-dimensional array: [n_outputs][n_vocab])
+3
View File
@@ -3,6 +3,7 @@
#include "llama.h"
#include <cstdint>
#include <vector>
#define LLAMA_MAX_SEQ 256
@@ -40,6 +41,8 @@ struct llama_cparams {
bool kv_unified;
bool pipeline_parallel;
std::vector<bool> output_layer_inp;
enum llama_pooling_type pooling_type;
ggml_backend_sched_eval_callback cb_eval;
+10
View File
@@ -88,3 +88,13 @@ LLAMA_API int32_t llama_model_n_devices(const struct llama_model * model);
LLAMA_API ggml_backend_dev_t llama_model_get_device(const struct llama_model * model, int i);
LLAMA_API llama_memory_breakdown llama_get_memory_breakdown(const struct llama_context * ctx);
//
// model/context data extraction
//
LLAMA_API void llama_set_output_layer_inp(struct llama_context * ctx, uint32_t layer_id, bool enable);
LLAMA_API ggml_tensor * llama_model_get_tok_embd(const struct llama_model * model);
LLAMA_API void llama_model_set_tok_embd(struct llama_model * model, ggml_tensor * tensor);
+14 -1
View File
@@ -810,6 +810,10 @@ void llm_graph_result::reset() {
t_logits = nullptr;
t_embd = nullptr;
t_embd_pooled = nullptr;
t_layer_inp.resize(LLAMA_MAX_LAYERS);
std::fill(t_layer_inp.begin(), t_layer_inp.end(), nullptr);
t_sampled.clear();
t_sampled_probs.clear();
t_sampled_logits.clear();
@@ -838,7 +842,7 @@ void llm_graph_result::set_inputs(const llama_ubatch * ubatch) {
}
}
void llm_graph_result::set_outputs() {
void llm_graph_result::set_outputs(const llm_graph_params & params) {
if (t_logits != nullptr) {
ggml_set_output(t_logits);
}
@@ -848,6 +852,14 @@ void llm_graph_result::set_outputs() {
if (t_embd_pooled != nullptr) {
ggml_set_output(t_embd_pooled);
}
{
const auto & output_layer_inp = params.cparams.output_layer_inp;
for (size_t il = 0; il < output_layer_inp.size(); ++il) {
if (output_layer_inp[il]) {
ggml_set_output(t_layer_inp[il]);
}
}
}
for (auto & [seq_id, t] : t_sampled) {
if (t != nullptr) {
ggml_set_output(t);
@@ -951,6 +963,7 @@ llm_graph_context::llm_graph_context(const llm_graph_params & params) :
loras (params.loras),
mctx (params.mctx),
cross (params.cross),
eagle3 (params.eagle3),
samplers (params.samplers),
cb_func (params.cb),
res (params.res),
+35 -5
View File
@@ -73,6 +73,30 @@ struct llama_cross {
std::vector<std::set<llama_seq_id>> seq_ids_enc;
};
// EAGLE3 support - stores intermediate features from target model
struct llama_eagle3 {
// Configuration: which layers to extract from target model
std::vector<int> extract_layer_indices;
// Extracted features from target model (for encoder input)
// Concatenated [layer_l, layer_m, layer_h] embeddings
// Shape: [n_layers * n_embd, n_tokens] where n_layers = extract_layer_indices.size()
std::vector<float> target_features;
// Encoder output (for decoder input)
std::vector<float> g_embeddings;
// Tensor references for feature extraction from target model
std::vector<ggml_tensor *> extract_tensors;
// Clear all stored data
void clear() {
target_features.clear();
g_embeddings.clear();
extract_tensors.clear();
}
};
struct llm_graph_params;
//
@@ -544,6 +568,7 @@ struct llm_graph_params {
const llama_adapter_loras * loras;
const llama_memory_context_i * mctx;
const llama_cross * cross;
llama_eagle3 * eagle3; // non-const: we write extracted features here
std::map<llama_seq_id, llama_sampler *> samplers;
@@ -645,6 +670,8 @@ public:
ggml_tensor * get_embd() const { return t_embd; }
ggml_tensor * get_embd_pooled() const { return t_embd_pooled; }
ggml_tensor * get_layer_inp(int il) const { return t_layer_inp[il]; }
ggml_cgraph * get_gf() const { return gf; }
ggml_context * get_ctx() const { return ctx_compute.get(); }
@@ -653,7 +680,7 @@ public:
void reset();
void set_inputs(const llama_ubatch * ubatch);
void set_outputs();
void set_outputs(const llm_graph_params & params);
// try to update the existing graph result using the new graph parameters in order to reuse it
// this can only be done if we determine that the resulting graph using the new graph parameters
@@ -673,10 +700,12 @@ public:
ggml_tensor * t_embd = nullptr;
ggml_tensor * t_embd_pooled = nullptr;
std::map<llama_seq_id, ggml_tensor*> t_sampled_logits;
std::map<llama_seq_id, ggml_tensor*> t_candidates;
std::map<llama_seq_id, ggml_tensor*> t_sampled;
std::map<llama_seq_id, ggml_tensor*> t_sampled_probs;
std::vector<ggml_tensor *> t_layer_inp;
std::map<llama_seq_id, ggml_tensor *> t_sampled_logits;
std::map<llama_seq_id, ggml_tensor *> t_candidates;
std::map<llama_seq_id, ggml_tensor *> t_sampled;
std::map<llama_seq_id, ggml_tensor *> t_sampled_probs;
std::vector<llm_graph_input_ptr> inputs;
@@ -758,6 +787,7 @@ struct llm_graph_context {
const llama_adapter_loras * loras;
const llama_memory_context_i * mctx;
const llama_cross * cross;
llama_eagle3 * eagle3; // non-const: we write extracted features here
std::map<llama_seq_id, llama_sampler *> samplers;
+1 -1
View File
@@ -71,7 +71,7 @@ uint32_t llama_hparams::n_rot(uint32_t il) const {
}
uint32_t llama_hparams::n_embd_inp() const {
uint32_t n_embd_inp = n_embd;
uint32_t n_embd_inp = n_embd_inp_impl > 0 ? n_embd_inp_impl : n_embd;
if (n_deepstack_layers > 0) {
n_embd_inp += n_embd * n_deepstack_layers;
+8
View File
@@ -42,6 +42,7 @@ struct llama_hparams {
uint32_t n_ctx_train; // context size the model was trained on
uint32_t n_embd;
uint32_t n_embd_inp_impl = 0;
uint32_t n_layer;
int32_t n_layer_kv_from_start = -1; // if non-negative, the first n_layer_kv_from_start layers have KV cache
uint32_t n_expert = 0;
@@ -210,6 +211,13 @@ struct llama_hparams {
// qwen3vl deepstack
uint32_t n_deepstack_layers = 0;
// EAGLE3 draft model - layer indices to extract from target model
// e.g., for 32-layer target: [2, 16, 29] (low, middle, high)
std::array<int, 3> eagle3_extract_layers = {0, 0, 0};
// EAGLE3 draft model - apply hidden_norm before storing residual
bool eagle3_norm_before_residual = false;
// gemma4 per-layer embedding
uint32_t n_embd_per_layer = 0;
+12 -1
View File
@@ -278,6 +278,8 @@ static llama_model * llama_model_mapping(llm_arch arch, const llama_model_params
return new llama_model_qwen35moe(params);
case LLM_ARCH_MISTRAL3:
return new llama_model_mistral3(params);
case LLM_ARCH_EAGLE3:
return new llama_model_eagle3(params);
case LLM_ARCH_MIMO2:
return new llama_model_mimo2(params);
case LLM_ARCH_KIMI_LINEAR:
@@ -2071,7 +2073,7 @@ ggml_cgraph * llama_model::build_graph(const llm_graph_params & params) const {
// TODO: move reranking logic here and generalize
llm->build_dense_out(dense_2_out_layers, dense_2_out_layers_b, dense_3_out_layers);
llm->res->set_outputs();
llm->res->set_outputs(params);
return llm->res->get_gf();
}
@@ -2237,6 +2239,7 @@ llama_rope_type llama_model_rope_type(const llama_model * model) {
case LLM_ARCH_ERNIE4_5:
case LLM_ARCH_ERNIE4_5_MOE:
case LLM_ARCH_MISTRAL3:
case LLM_ARCH_EAGLE3:
case LLM_ARCH_MISTRAL4:
case LLM_ARCH_LLAMA_EMBED:
case LLM_ARCH_MAINCODER:
@@ -2515,3 +2518,11 @@ void llama_model_base::create_tensor_qkv(llama_layer & layer, int bid,
layer.wv_b = create_tensor(tn(LLM_TENSOR_ATTN_V, "bias", bid), {n_embd_v_}, TENSOR_NOT_REQUIRED);
}
}
ggml_tensor * llama_model_get_tok_embd(const struct llama_model * model) {
return model->tok_embd;
}
void llama_model_set_tok_embd(struct llama_model * model, ggml_tensor * tensor) {
model->tok_embd = tensor;
}
+10
View File
@@ -465,6 +465,9 @@ struct llama_layer {
struct ggml_tensor * ffn_act_beta = nullptr;
struct ggml_tensor * ffn_act_eps = nullptr;
// eagle3
struct ggml_tensor * eagle3_hidden_norm = nullptr;
// Kimi Linear KDA (using ssm_ prefix for consistency)
// Note: ssm_dt_b already exists above (mamba bias), reused for Kimi dt_bias
struct ggml_tensor * ssm_q_conv = nullptr;
@@ -550,6 +553,13 @@ struct llama_model {
struct ggml_tensor * per_layer_model_proj = nullptr;
struct ggml_tensor * per_layer_proj_norm = nullptr;
// eagle3
struct ggml_tensor * fc = nullptr; // feature fusion layer
struct ggml_tensor * d2t = nullptr; // draft to target vocabulary mapping
// Reference to target model's embedding layer
// This allows EAGLE3 to use target model's embeddings without copying
struct ggml_tensor * target_tok_embd = nullptr;
std::vector<llama_layer> layers;
//Dense linear projections for SentenceTransformers models like embeddinggemma
+293
View File
@@ -0,0 +1,293 @@
#include "models.h"
void llama_model_eagle3::load_arch_hparams(llama_model_loader & ml) {
ml.get_key(LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, hparams.f_norm_rms_eps);
// EAGLE3 layer extraction configuration
// Use array<int, 4> (has template instantiation), then copy first 3 elements
std::array<int, 4> extract_layers_tmp = {};
if (!ml.get_key_or_arr(LLM_KV_EAGLE3_EXTRACT_LAYERS, extract_layers_tmp, 3, false)) {
throw std::runtime_error("EAGLE3 model requires 'extract_layers' in GGUF metadata");
}
std::copy_n(extract_layers_tmp.begin(), 3, hparams.eagle3_extract_layers.begin());
LLAMA_LOG_INFO("%s: EAGLE3 extract_layers = [%d, %d, %d]\n", __func__,
hparams.eagle3_extract_layers[0],
hparams.eagle3_extract_layers[1],
hparams.eagle3_extract_layers[2]);
// EAGLE3 target model hidden size
ml.get_key(LLM_KV_EAGLE3_TARGET_HIDDEN_SIZE, hparams.n_embd_inp_impl);
LLAMA_LOG_INFO("%s: EAGLE3 target_hidden_size = %u (draft n_embd = %u)\n", __func__,
hparams.n_embd_inp_impl, hparams.n_embd);
// EAGLE3 norm_before_residual (optional, default false)
// compatible with Readhat eagle3 speculator model
ml.get_key(LLM_KV_EAGLE3_NORM_BEFORE_RESIDUAL, hparams.eagle3_norm_before_residual, false);
if (hparams.eagle3_norm_before_residual) {
LLAMA_LOG_INFO("%s: EAGLE3 norm_before_residual = true\n", __func__);
}
type = LLM_TYPE_UNKNOWN;
}
void llama_model_eagle3::load_arch_tensors(llama_model_loader & /*ml*/) {
LLAMA_LOAD_LOCALS;
const int64_t n_embd_inp = hparams.n_embd_inp();
const int64_t n_embd_attn_input = 2 * n_embd;
// Get vocab size from the d2t tensor in the GGUF file (optional - only needed if EAGLE3 has different vocab_size than target)
// d2t: draft to target vocabulary mapping
int64_t n_draft_vocab = n_vocab; // Default: same as target vocab
const struct ggml_tensor * d2t_meta = ml->get_tensor_meta("d2t");
if (d2t_meta) {
n_draft_vocab = d2t_meta->ne[0]; // update draft vocab size
d2t = create_tensor(tn(LLM_TENSOR_EAGLE3_D2T), {n_draft_vocab}, 0);
LLAMA_LOG_INFO("%s: EAGLE3 using d2t mapping (draft_vocab_size = %lld)\n", __func__, (long long)n_draft_vocab);
} else {
d2t = nullptr; // no d2t, use default vocab size
LLAMA_LOG_INFO("%s: EAGLE3 without d2t - sharing same vocab_size with target (vocab_size = %lld)\n", __func__, (long long)n_draft_vocab);
}
// Feature fusion layer: projects 3 target layers to draft hidden size
fc = create_tensor(tn(LLM_TENSOR_EAGLE3_FC, "weight"), {n_embd_inp, n_embd}, 0);
// Output layer (uses draft vocab size)
output_norm = create_tensor(tn(LLM_TENSOR_OUTPUT_NORM, "weight"), {n_embd}, 0);
output = create_tensor(tn(LLM_TENSOR_OUTPUT, "weight"), {n_embd, n_draft_vocab}, 0);
// Token embeddings (optional - Llama 3.3 70B EAGLE3 has its own)
const struct ggml_tensor * tok_embd_meta = ml->get_tensor_meta(tn(LLM_TENSOR_TOKEN_EMBD, "weight").str().c_str());
if (tok_embd_meta) {
const int64_t n_target_vocab = tok_embd_meta->ne[1];
tok_embd = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_target_vocab}, 0);
LLAMA_LOG_INFO("%s: EAGLE3 using its own token_embd (vocab = %lld)\n", __func__, (long long)n_target_vocab);
}
// Single decoder layer
for (int i = 0; i < n_layer; ++i) {
auto & layer = layers[i];
// input_layernorm: applied to token embeddings
layer.attn_norm = create_tensor(tn(LLM_TENSOR_ATTN_NORM, "weight", i), {n_embd}, 0);
// Attention takes input_embeds_normed + fused_target_normed as input
layer.wq = create_tensor(tn(LLM_TENSOR_ATTN_Q, "weight", i), {n_embd_attn_input, n_embd_head_k * n_head}, 0);
layer.wk = create_tensor(tn(LLM_TENSOR_ATTN_K, "weight", i), {n_embd_attn_input, n_embd_k_gqa}, 0);
layer.wv = create_tensor(tn(LLM_TENSOR_ATTN_V, "weight", i), {n_embd_attn_input, n_embd_v_gqa}, 0);
layer.wo = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "weight", i), {n_embd_head_k * n_head, n_embd}, 0);
// EAGLE-3 specific: hidden_norm applied to fused target features
layer.eagle3_hidden_norm = create_tensor(tn(LLM_TENSOR_EAGLE3_HIDDEN_NORM, "weight", i), {n_embd}, 0);
layer.ffn_norm = create_tensor(tn(LLM_TENSOR_FFN_NORM, "weight", i), {n_embd}, 0);
layer.ffn_gate = create_tensor(tn(LLM_TENSOR_FFN_GATE, "weight", i), {n_embd, n_ff}, 0);
layer.ffn_down = create_tensor(tn(LLM_TENSOR_FFN_DOWN, "weight", i), { n_ff, n_embd}, 0);
layer.ffn_up = create_tensor(tn(LLM_TENSOR_FFN_UP, "weight", i), {n_embd, n_ff}, 0);
// rope_freqs for llama3 rope scaling (optional - only if EAGLE3 config has rope_scaling)
layer.rope_freqs = create_tensor(tn(LLM_TENSOR_ROPE_FREQS, "weight", i), {n_rot/2}, TENSOR_NOT_REQUIRED);
}
}
std::unique_ptr<llm_graph_context> llama_model_eagle3::build_arch_graph(const llm_graph_params & params) const {
switch (params.gtype) {
case LLM_GRAPH_TYPE_ENCODER:
return std::make_unique<graph<true>>(*this, params);
case LLM_GRAPH_TYPE_DEFAULT:
case LLM_GRAPH_TYPE_DECODER:
return std::make_unique<graph<false>>(*this, params);
default:
GGML_ABORT("invalid graph type");
};
}
template <>
ggml_tensor * llama_model_eagle3::graph<true>::build_inp_embd_enc() const {
const int64_t n_embd_inp = hparams.n_embd_inp();
ggml_tensor * cur = nullptr;
// Input: Target model features (3 layers concatenated: low, mid, high)
// Data will be provided via ubatch->embd in encode_eagle3_features()
auto inp_target = std::make_unique<llm_graph_input_embd>(n_embd_inp);
inp_target->embd = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_embd_inp, n_tokens);
ggml_set_input(inp_target->embd);
cur = inp_target->embd;
cb(cur, "inp_embd", -1);
res->add_input(std::move(inp_target));
return cur;
}
// EAGLE3 Encoder: processes target model features through feature fusion layer
// Input: target_features e.g. [12288, n_tokens] from target model layers low, middle, high
// Output: g_embeddings e.g. [4096, n_tokens] stored in context
template <>
llama_model_eagle3::graph<true>::graph(const llama_model & model, const llm_graph_params & params) : llm_graph_context(params) {
ggml_tensor * cur = nullptr;
cur = build_inp_embd_enc();
// sanity check
GGML_ASSERT(hparams.n_embd_inp() == model.fc->ne[0]);
// Feature fusion layer
cur = build_lora_mm(model.fc, cur);
cb(cur, "fc_out", -1);
// Output: g_embeddings e.g. [4096, n_tokens]
res->t_embd = cur;
ggml_build_forward_expand(gf, cur);
}
// EAGLE3 Decoder: processes draft tokens using g_embeddings from encoder
// Input: draft tokens + g_embeddings from encoder
// Output: draft logits
template <>
llama_model_eagle3::graph<false>::graph(const llama_model & model, const llm_graph_params & params) : llm_graph_context(params) {
const int64_t n_embd_head = hparams.n_embd_head_v();
GGML_ASSERT(n_embd_head == hparams.n_embd_head_k());
GGML_ASSERT(n_layer == 1); // EAGLE-3 has only one decoder layer
ggml_tensor * cur;
ggml_tensor * inpL;
// EAGLE3 Decoder receives:
// 1. Token embeddings (e.g.from EAGLE3's own tok_embd for Llama 3.3 70B, or target model for Llama 3.1 8B)
// 2. g_embeddings from encoder
// Choose token_embd_eagle3: prefer EAGLE3's own if available (Llama 3.3 70B), else use target's (Llama 3.1 8B)
ggml_tensor * token_embd_eagle3 = (model.tok_embd != nullptr) ? model.tok_embd : model.target_tok_embd;
GGML_ASSERT(token_embd_eagle3 != nullptr && "EAGLE3 decoder requires token embeddings (own or from target model)");
ggml_tensor * inp_embd = build_inp_embd(token_embd_eagle3);
cb(inp_embd, "inp_embd", -1);
// TODO: refactor into llm_graph_input
ggml_tensor * inp_g = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_embd, n_tokens);
ggml_set_input(inp_g);
cb(inp_g, "inp_g_embeddings", -1); // TODO: do not change the name! refactor into llm_graph_input
inpL = inp_g;
// inp_pos - contains the positions
ggml_tensor * inp_pos = build_inp_pos();
auto * inp_attn = build_attn_inp_kv();
const float kq_scale = 1.0f/sqrtf(float(n_embd_head));
ggml_tensor * inp_out_ids = build_inp_out_ids();
// Single decoder layer (il = 0)
const int il = 0;
{
// Apply input_layernorm to the token embeddings
ggml_tensor * embd_norm = build_norm(inp_embd,
model.layers[il].attn_norm, NULL,
LLM_NORM_RMS, il);
cb(embd_norm, "embd_norm", il);
// Apply hidden_norm to inp_g
ggml_tensor * g_norm = build_norm(inp_g,
model.layers[il].eagle3_hidden_norm, NULL,
LLM_NORM_RMS, -1);
cb(g_norm, "g_norm", il);
// norm_before_residual: determines what goes into the residual connection (compatible with Readhat eagle3 speculator model)
// - false (default): use raw inp_g for residual
// - true: use normalized g_norm for residual
// inpL is the concatenated input (normalized inp_embd + normalized inp_g)
ggml_tensor * inpSA = hparams.eagle3_norm_before_residual ? g_norm : inpL;
// Concatenate normalized inp_embd and normalized inp_g
cur = ggml_concat(ctx0, embd_norm, g_norm, il);
cb(cur, "concat_embd", il);
// Self-attention with concatenated input
ggml_tensor * Qcur = build_lora_mm(model.layers[il].wq, cur);
cb(Qcur, "Qcur", il);
ggml_tensor * Kcur = build_lora_mm(model.layers[il].wk, cur);
cb(Kcur, "Kcur", il);
ggml_tensor * Vcur = build_lora_mm(model.layers[il].wv, cur);
cb(Vcur, "Vcur", il);
Qcur = ggml_reshape_3d(ctx0, Qcur, n_embd_head, n_head, n_tokens);
Kcur = ggml_reshape_3d(ctx0, Kcur, n_embd_head, n_head_kv, n_tokens);
Vcur = ggml_reshape_3d(ctx0, Vcur, n_embd_head, n_head_kv, n_tokens);
// rope freq factors, returns nullptr if not available
ggml_tensor * rope_factors = model.get_rope_factors(cparams, il);
// RoPE
Qcur = ggml_rope_ext(
ctx0, Qcur, inp_pos, rope_factors,
n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,
ext_factor, attn_factor, beta_fast, beta_slow
);
Kcur = ggml_rope_ext(
ctx0, Kcur, inp_pos, rope_factors,
n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,
ext_factor, attn_factor, beta_fast, beta_slow
);
cb(Qcur, "Qcur_rope", il);
cb(Kcur, "Kcur_rope", il);
cur = build_attn(inp_attn,
model.layers[il].wo, NULL, nullptr,
Qcur, Kcur, Vcur, nullptr, nullptr, nullptr, kq_scale, il);
if (inp_out_ids) {
cur = ggml_get_rows(ctx0, cur, inp_out_ids);
inpSA = ggml_get_rows(ctx0, inpSA, inp_out_ids);
}
// Add residual and update it
ggml_tensor * ffn_inp = ggml_add(ctx0, cur, inpSA);
cb(ffn_inp, "ffn_inp", il);
// Apply FFN norm to the sum
cur = build_norm(ffn_inp,
model.layers[il].ffn_norm, NULL,
LLM_NORM_RMS, il);
cb(cur, "post_attn_norm", il);
cur = build_ffn(cur,
model.layers[il].ffn_up, NULL, NULL,
model.layers[il].ffn_gate, NULL, NULL,
model.layers[il].ffn_down, NULL, NULL,
NULL,
LLM_FFN_SILU, LLM_FFN_PAR, il);
cb(cur, "ffn_out", il);
// Output norm with residual
cur = ggml_add(ctx0, cur, ffn_inp);
cb(cur, "eagle3_prenorm", il);
inpL = cur;
}
cur = inpL;
// Output prenorm state (for next token's g_embeddings in autoregressive generation)
ggml_set_output(cur);
res->t_embd = cur;
cur = build_norm(cur,
model.output_norm, NULL,
LLM_NORM_RMS, -1);
cb(cur, "result_norm", -1);
// lm_head - projects to draft vocabulary
cur = build_lora_mm(model.output, cur);
cb(cur, "result_output", -1);
res->t_logits = cur;
ggml_build_forward_expand(gf, cur);
}
+2
View File
@@ -123,6 +123,8 @@ llama_model_llama::graph<embed>::graph(const llama_model & model, const llm_grap
ggml_tensor * inp_out_ids = build_inp_out_ids();
for (int il = 0; il < n_layer; ++il) {
res->t_layer_inp[il] = inpL;
ggml_tensor * inpSA = inpL;
// norm
+15
View File
@@ -1784,6 +1784,21 @@ struct llama_model_qwen35moe : public llama_model_base {
std::unique_ptr<llm_graph_context> build_arch_graph(const llm_graph_params & params) const override;
};
struct llama_model_eagle3 : public llama_model_base {
llama_model_eagle3(const struct llama_model_params & params) : llama_model_base(params) {}
void load_arch_hparams(llama_model_loader & ml) override;
void load_arch_tensors(llama_model_loader & ml) override;
template <bool is_enc>
struct graph : public llm_graph_context {
graph(const llama_model & model, const llm_graph_params & params);
ggml_tensor * build_inp_embd_enc() const;
};
std::unique_ptr<llm_graph_context> build_arch_graph(const llm_graph_params & params) const override;
};
struct llama_model_mistral3 : public llama_model_base {
llama_model_mistral3(const struct llama_model_params & params) : llama_model_base(params) {}
+2
View File
@@ -75,6 +75,8 @@ llama_model_openai_moe::graph::graph(const llama_model & model, const llm_graph_
ggml_tensor * inp_out_ids = build_inp_out_ids();
for (int il = 0; il < n_layer; ++il) {
res->t_layer_inp[il] = inpL;
const float freq_base_l = model.get_rope_freq_base (cparams, il);
const float freq_scale_l = model.get_rope_freq_scale(cparams, il);
+2
View File
@@ -68,6 +68,8 @@ llama_model_qwen3::graph::graph(const llama_model & model, const llm_graph_param
ggml_tensor * inp_out_ids = build_inp_out_ids();
for (int il = 0; il < n_layer; ++il) {
res->t_layer_inp[il] = inpL;
ggml_tensor * inpSA = inpL;
// norm
+2
View File
@@ -78,6 +78,8 @@ llama_model_qwen3moe::graph::graph(const llama_model & model, const llm_graph_pa
ggml_tensor * inp_out_ids = build_inp_out_ids();
for (int il = 0; il < n_layer; ++il) {
res->t_layer_inp[il] = inpL;
ggml_tensor * inpSA = inpL;
// norm
+13 -1
View File
@@ -805,8 +805,20 @@ private:
return false;
}
params_base.speculative.draft.model = model_dft.get();
params_base.speculative.draft.model = model_dft.get();
params_base.speculative.draft.model_tgt = model;
params_base.speculative.draft.cparams = common_context_params_to_llama(params_dft);
if (params_base.speculative.draft.eagle3) {
// EAGLE3 current limitation: extracted target features are per-context; multiple slots would overwrite each other
if (params_base.n_parallel > 1) {
SRV_ERR("%s", "EAGLE3 speculative decoding is not supported with n_parallel > 1\n");
return false;
}
llama_set_eagle3(ctx, model_dft.get());
SRV_INF("%s", "EAGLE3 feature extraction enabled on target model\n");
}
}
std::string & mmproj_path = params_base.mmproj.path;