mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-09-27 21:46:57 +02:00
Compare commits
23
Commits
b10327
...
pr/18039-gg
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f84632951a | ||
|
|
4567954ab0 | ||
|
|
069be0ae22 | ||
|
|
459b02f6c0 | ||
|
|
cb8a3a93ec | ||
|
|
91b03e4c93 | ||
|
|
5bb2d50c0f | ||
|
|
07e2c9707c | ||
|
|
b8ab2cc559 | ||
|
|
9fea2434af | ||
|
|
b3537924ef | ||
|
|
5e224bc190 | ||
|
|
7d4c223943 | ||
|
|
7b78bfa984 | ||
|
|
75883cde73 | ||
|
|
13a9f31de3 | ||
|
|
3da288d78d | ||
|
|
71ba283a65 | ||
|
|
c0d99e65d2 | ||
|
|
5a79c1900f | ||
|
|
3e7f376b53 | ||
|
|
ac5667dcc6 | ||
|
|
8fac4b1cc8 |
+10
-2
@@ -3067,7 +3067,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
|
||||
[](common_params & params, bool value) {
|
||||
params.use_jinja = value;
|
||||
}
|
||||
).set_examples({LLAMA_EXAMPLE_SERVER, LLAMA_EXAMPLE_COMPLETION, LLAMA_EXAMPLE_CLI, LLAMA_EXAMPLE_MTMD}).set_env("LLAMA_ARG_JINJA"));
|
||||
).set_examples({LLAMA_EXAMPLE_SERVER, LLAMA_EXAMPLE_COMPLETION, LLAMA_EXAMPLE_CLI, LLAMA_EXAMPLE_MTMD, LLAMA_EXAMPLE_SPECULATIVE}).set_env("LLAMA_ARG_JINJA"));
|
||||
add_opt(common_arg(
|
||||
{"--reasoning-format"}, "FORMAT",
|
||||
"controls whether thought tags are allowed and/or extracted from the response, and in which format they're returned; one of:\n"
|
||||
@@ -3123,7 +3123,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
|
||||
[](common_params & params, const std::string & value) {
|
||||
params.chat_template = value;
|
||||
}
|
||||
).set_examples({LLAMA_EXAMPLE_COMPLETION, LLAMA_EXAMPLE_CLI, LLAMA_EXAMPLE_SERVER, LLAMA_EXAMPLE_MTMD}).set_env("LLAMA_ARG_CHAT_TEMPLATE"));
|
||||
).set_examples({LLAMA_EXAMPLE_COMPLETION, LLAMA_EXAMPLE_CLI, LLAMA_EXAMPLE_SERVER, LLAMA_EXAMPLE_MTMD, LLAMA_EXAMPLE_SPECULATIVE}).set_env("LLAMA_ARG_CHAT_TEMPLATE"));
|
||||
add_opt(common_arg(
|
||||
{"--chat-template-file"}, "JINJA_TEMPLATE_FILE",
|
||||
string_format(
|
||||
@@ -3455,6 +3455,14 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
|
||||
params.speculative.draft.cache_type_v = kv_cache_type_from_str(value);
|
||||
}
|
||||
).set_env("LLAMA_ARG_SPEC_DRAFT_CACHE_TYPE_V"));
|
||||
// TODO: rename:
|
||||
add_opt(common_arg(
|
||||
{"--eagle3"},
|
||||
"use EAGLE3 speculative decoding with the draft model",
|
||||
[](common_params & params) {
|
||||
params.speculative.draft.eagle3 = true;
|
||||
}
|
||||
).set_examples({LLAMA_EXAMPLE_SPECULATIVE, LLAMA_EXAMPLE_CLI}));
|
||||
add_opt(common_arg(
|
||||
{"--spec-draft-override-tensor", "-otd", "--override-tensor-draft"}, "<tensor name pattern>=<buffer type>,...",
|
||||
"override tensor buffer type for draft model", [](common_params & params, const std::string & value) {
|
||||
|
||||
+4
-1
@@ -307,10 +307,13 @@ struct common_params_speculative_draft {
|
||||
|
||||
common_params_model mparams;
|
||||
|
||||
llama_model * model = nullptr; // a llama_model that can be shared by multiple speculative contexts
|
||||
llama_model * model = nullptr; // a llama_model that can be shared by multiple speculative contexts
|
||||
llama_model * model_tgt = nullptr; // the target model
|
||||
|
||||
llama_context_params cparams; // these are the parameters for the draft llama_context
|
||||
|
||||
bool eagle3 = false; // use EAGLE3 speculative decoding
|
||||
|
||||
int32_t n_ctx = 0; // draft context size
|
||||
int32_t n_gpu_layers = -1; // number of layers to store in VRAM for the draft model (-1 - use default)
|
||||
|
||||
|
||||
+185
-23
@@ -48,6 +48,7 @@ struct common_speculative_config {
|
||||
const common_params_speculative & p = common_params_speculative{}) : type(t), params(p) {}
|
||||
};
|
||||
|
||||
|
||||
static bool common_speculative_are_compatible(
|
||||
const llama_model * model_tgt,
|
||||
const llama_model * model_dft) {
|
||||
@@ -240,7 +241,9 @@ struct common_speculative_state_draft : public common_speculative_state {
|
||||
~common_speculative_state_draft() override {
|
||||
llama_perf_context_print(ctx_dft);
|
||||
|
||||
llama_free(ctx_dft);
|
||||
if (ctx_dft) {
|
||||
llama_free(ctx_dft);
|
||||
}
|
||||
|
||||
common_sampler_free(smpl);
|
||||
|
||||
@@ -291,11 +294,11 @@ struct common_speculative_state_draft : public common_speculative_state {
|
||||
|
||||
auto * spec = this;
|
||||
|
||||
auto & batch = spec->batch;
|
||||
auto & ctx_tgt = spec->ctx_tgt;
|
||||
auto & ctx_dft = spec->ctx_dft;
|
||||
auto & smpl = spec->smpl;
|
||||
auto & prompt_dft = spec->prompt_dft;
|
||||
auto & batch = spec->batch;
|
||||
auto & ctx_tgt = spec->ctx_tgt;
|
||||
auto & ctx_dft = spec->ctx_dft;
|
||||
auto & smpl = spec->smpl;
|
||||
auto & prompt_dft = spec->prompt_dft;
|
||||
|
||||
auto * mem_dft = llama_get_memory(ctx_dft);
|
||||
|
||||
@@ -567,7 +570,52 @@ struct common_speculative_state_draft : public common_speculative_state {
|
||||
};
|
||||
|
||||
struct common_speculative_state_eagle3 : public common_speculative_state {
|
||||
common_speculative_state_eagle3(enum common_speculative_type type) : common_speculative_state(type) {}
|
||||
llama_context * ctx_tgt;
|
||||
|
||||
common_sampler * smpl;
|
||||
|
||||
llama_batch batch;
|
||||
|
||||
struct llama_context * ctx_dft_enc = nullptr;
|
||||
struct llama_context * ctx_dft_dec = nullptr;
|
||||
|
||||
int32_t eagle3_n_past = 0; // number of verified positions in decoder KV cache
|
||||
|
||||
common_speculative_state_eagle3(
|
||||
enum common_speculative_type type,
|
||||
llama_context * ctx_tgt,
|
||||
llama_context * ctx_dft_enc,
|
||||
llama_context * ctx_dft_dec)
|
||||
: common_speculative_state(type)
|
||||
, ctx_tgt(ctx_tgt)
|
||||
, ctx_dft_enc(ctx_dft_enc)
|
||||
, ctx_dft_dec(ctx_dft_dec)
|
||||
{
|
||||
batch = llama_batch_init(llama_n_batch(ctx_dft_dec), 0, 1);
|
||||
|
||||
// Initialize sampler for EAGLE3 decoder
|
||||
common_params_sampling params;
|
||||
params.no_perf = false;
|
||||
params.top_k = 10; // set 1 for greedy sampling (argmax) to match vLLM's default behavior but >1 always gets higher acceptance rate for eagle3
|
||||
params.samplers = { COMMON_SAMPLER_TYPE_TOP_K };
|
||||
smpl = common_sampler_init(llama_get_model(ctx_dft_dec), params);
|
||||
}
|
||||
|
||||
~common_speculative_state_eagle3() override {
|
||||
llama_perf_context_print(ctx_dft_dec);
|
||||
|
||||
if (ctx_dft_dec) {
|
||||
llama_free(ctx_dft_dec);
|
||||
}
|
||||
|
||||
if (ctx_dft_enc) {
|
||||
llama_free(ctx_dft_enc);
|
||||
}
|
||||
|
||||
common_sampler_free(smpl);
|
||||
|
||||
llama_batch_free(batch);
|
||||
}
|
||||
|
||||
void begin(const llama_tokens & prompt) override {
|
||||
GGML_UNUSED(prompt);
|
||||
@@ -577,12 +625,97 @@ struct common_speculative_state_eagle3 : public common_speculative_state {
|
||||
const common_params_speculative & params,
|
||||
const llama_tokens & prompt_tgt,
|
||||
llama_token id_last,
|
||||
llama_tokens & draft_tokens) override {
|
||||
// TODO: implement
|
||||
GGML_UNUSED(params);
|
||||
GGML_UNUSED(prompt_tgt);
|
||||
GGML_UNUSED(id_last);
|
||||
GGML_UNUSED(draft_tokens);
|
||||
llama_tokens & result) override {
|
||||
auto * spec = this;
|
||||
|
||||
auto & batch = spec->batch;
|
||||
auto & ctx_tgt = spec->ctx_tgt;
|
||||
auto & ctx_dft_enc = spec->ctx_dft_enc;
|
||||
auto & ctx_dft_dec = spec->ctx_dft_dec;
|
||||
auto & smpl = spec->smpl;
|
||||
|
||||
//result = gen_eagle3_draft(spec, params, prompt_tgt, id_last);
|
||||
const int n_embd = llama_model_n_embd(llama_get_model(ctx_dft_enc));
|
||||
const int n = (int)prompt_tgt.size();
|
||||
const int n_new = n - spec->eagle3_n_past;
|
||||
|
||||
GGML_ASSERT(n >= 1 && "prompt_tgt is empty");
|
||||
GGML_ASSERT(n_new >= 1 && "must have at least 1 new token");
|
||||
|
||||
// Clear draft positions from decoder KV cache [n_past, inf)
|
||||
llama_memory_seq_rm(llama_get_memory(ctx_dft_dec), 0, spec->eagle3_n_past, -1);
|
||||
|
||||
// Encoder: features → g_embeddings
|
||||
const float * features = llama_get_eagle3_target_features(ctx_tgt);
|
||||
GGML_ASSERT(features && "no target features");
|
||||
|
||||
llama_batch enc_batch = {
|
||||
/*.n_tokens =*/ n_new,
|
||||
/*.token =*/ nullptr,
|
||||
/*.embd =*/ const_cast<float*>(features),
|
||||
/*.pos =*/ nullptr,
|
||||
/*.n_seq_id =*/ nullptr,
|
||||
/*.seq_id =*/ nullptr,
|
||||
/*.logits =*/ nullptr,
|
||||
};
|
||||
GGML_ASSERT(llama_encode(ctx_dft_enc, enc_batch) == 0);
|
||||
|
||||
const float * g_embd = llama_get_embeddings(ctx_dft_enc);
|
||||
GGML_ASSERT(g_embd && "encoder output failed");
|
||||
|
||||
// Decoder batch: process new tokens with KV cache reuse
|
||||
llama_set_eagle3_g_embeddings(ctx_dft_dec, g_embd, n_embd, n_new);
|
||||
|
||||
common_batch_clear(batch);
|
||||
for (int i = 0; i < n_new; i++) {
|
||||
const int pos = spec->eagle3_n_past + i;
|
||||
const llama_token tok = (pos < n - 1) ? prompt_tgt[pos + 1] : id_last;
|
||||
common_batch_add(batch, tok, pos, {0}, true);
|
||||
}
|
||||
|
||||
GGML_ASSERT(llama_decode(ctx_dft_dec, batch) == 0);
|
||||
|
||||
spec->eagle3_n_past = n; // update verified positions
|
||||
|
||||
// Sample draft tokens
|
||||
result.clear();
|
||||
common_sampler_reset(smpl);
|
||||
|
||||
// Sample and check probability (consistent with standard speculative decoding)
|
||||
auto sample_and_check = [&](int idx) -> bool {
|
||||
common_sampler_sample(smpl, ctx_dft_dec, idx);
|
||||
|
||||
const auto * cur_p = common_sampler_get_candidates(smpl, true);
|
||||
const llama_token id = cur_p->data[0].id;
|
||||
|
||||
common_sampler_accept(smpl, id, true);
|
||||
result.push_back(id);
|
||||
|
||||
return cur_p->data[0].p >= params.draft.p_min;
|
||||
};
|
||||
|
||||
// First draft token from batch decode
|
||||
if (!sample_and_check(n_new - 1)) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Autoregressive: use prenorm as g_embd (-1 = last output)
|
||||
const float * prenorm = llama_get_embeddings_ith(ctx_dft_dec, -1);
|
||||
|
||||
for (int i = 1; i < params.draft.n_max; i++) {
|
||||
GGML_ASSERT(prenorm && "prenorm failed");
|
||||
llama_set_eagle3_g_embeddings(ctx_dft_dec, prenorm, n_embd, 1);
|
||||
|
||||
common_batch_clear(batch);
|
||||
common_batch_add(batch, result.back(), n - 1 + i, {0}, true);
|
||||
GGML_ASSERT(llama_decode(ctx_dft_dec, batch) == 0);
|
||||
|
||||
prenorm = llama_get_embeddings_ith(ctx_dft_dec, -1);
|
||||
|
||||
if (!sample_and_check(0)) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void accept(uint16_t n_accepted) override {
|
||||
@@ -975,11 +1108,35 @@ common_speculative * common_speculative_init(
|
||||
common_params_speculative & params,
|
||||
llama_context * ctx_tgt) {
|
||||
llama_context * ctx_dft = nullptr;
|
||||
|
||||
llama_context * ctx_dft_enc = nullptr;
|
||||
llama_context * ctx_dft_dec = nullptr;
|
||||
|
||||
if (params.draft.model) {
|
||||
ctx_dft = llama_init_from_model(params.draft.model, params.draft.cparams);
|
||||
if (ctx_dft == nullptr) {
|
||||
LOG_ERR("%s", "failed to create draft context\n");
|
||||
return nullptr;
|
||||
if (params.draft.eagle3) {
|
||||
llama_context_params params_enc = params.draft.cparams;
|
||||
params_enc.target_model = nullptr;
|
||||
params_enc.embeddings = true;
|
||||
ctx_dft_enc = llama_init_from_model(params.draft.model, params_enc);
|
||||
if (!ctx_dft_enc) {
|
||||
LOG_ERR("failed to create EAGLE3 encoder context\n");
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
llama_context_params params_dec = params.draft.cparams;
|
||||
params_dec.target_model = params.draft.model_tgt;
|
||||
params_dec.embeddings = true;
|
||||
ctx_dft_dec = llama_init_from_model(params.draft.model, params_dec);
|
||||
if (!ctx_dft_dec) {
|
||||
LOG_ERR("failed to create EAGLE3 decoder context\n");
|
||||
return nullptr;
|
||||
}
|
||||
} else {
|
||||
ctx_dft = llama_init_from_model(params.draft.model, params.draft.cparams);
|
||||
if (ctx_dft == nullptr) {
|
||||
LOG_ERR("%s", "failed to create draft context\n");
|
||||
return nullptr;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -987,7 +1144,7 @@ common_speculative * common_speculative_init(
|
||||
std::vector<common_speculative_config> configs = {}; // list of speculative configs to try
|
||||
{
|
||||
bool has_draft = !params.draft.mparams.path.empty();
|
||||
bool has_draft_eagle3 = false; // TODO PR-18039: if params.speculative.eagle3
|
||||
bool has_draft_eagle3 = params.draft.eagle3;
|
||||
|
||||
bool has_ngram_cache = (params.type == COMMON_SPECULATIVE_TYPE_NGRAM_CACHE);
|
||||
bool has_ngram_simple = (params.type == COMMON_SPECULATIVE_TYPE_NGRAM_SIMPLE);
|
||||
@@ -1029,10 +1186,11 @@ common_speculative * common_speculative_init(
|
||||
configs.push_back(common_speculative_config(COMMON_SPECULATIVE_TYPE_NGRAM_CACHE, params));
|
||||
}
|
||||
if (has_draft) {
|
||||
configs.push_back(common_speculative_config(COMMON_SPECULATIVE_TYPE_DRAFT, params));
|
||||
}
|
||||
if (has_draft_eagle3) {
|
||||
configs.push_back(common_speculative_config(COMMON_SPECULATIVE_TYPE_EAGLE3, params));
|
||||
if (has_draft_eagle3) {
|
||||
configs.push_back(common_speculative_config(COMMON_SPECULATIVE_TYPE_EAGLE3, params));
|
||||
} else {
|
||||
configs.push_back(common_speculative_config(COMMON_SPECULATIVE_TYPE_DRAFT, params));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1055,7 +1213,11 @@ common_speculative * common_speculative_init(
|
||||
break;
|
||||
}
|
||||
case COMMON_SPECULATIVE_TYPE_EAGLE3: {
|
||||
impls.push_back(std::make_unique<common_speculative_state_eagle3>(config.type));
|
||||
impls.push_back(std::make_unique<common_speculative_state_eagle3>(config.type,
|
||||
/* .ctx_tgt = */ ctx_tgt,
|
||||
/* .ctx_dft_enc = */ ctx_dft_enc,
|
||||
/* .ctx_dft_dec = */ ctx_dft_dec
|
||||
));
|
||||
break;
|
||||
}
|
||||
case COMMON_SPECULATIVE_TYPE_NGRAM_SIMPLE: {
|
||||
|
||||
+146
-1
@@ -97,6 +97,7 @@ class ModelBase:
|
||||
metadata_override: Path | None
|
||||
dir_model_card: Path
|
||||
remote_hf_model_id: str | None
|
||||
target_model_dir: Path | None
|
||||
|
||||
# subclasses should define this!
|
||||
model_arch: gguf.MODEL_ARCH
|
||||
@@ -116,7 +117,7 @@ class ModelBase:
|
||||
split_max_tensors: int = 0, split_max_size: int = 0, dry_run: bool = False,
|
||||
small_first_shard: bool = False, hparams: dict[str, Any] | None = None, remote_hf_model_id: str | None = None,
|
||||
disable_mistral_community_chat_template: bool = False,
|
||||
sentence_transformers_dense_modules: bool = False,
|
||||
sentence_transformers_dense_modules: bool = False, target_model_dir: Path | None = None,
|
||||
fuse_gate_up_exps: bool = False):
|
||||
if type(self) is ModelBase or \
|
||||
type(self) is TextModel or \
|
||||
@@ -136,6 +137,7 @@ class ModelBase:
|
||||
self.dry_run = dry_run
|
||||
self.remote_hf_model_id = remote_hf_model_id
|
||||
self.sentence_transformers_dense_modules = sentence_transformers_dense_modules
|
||||
self.target_model_dir = target_model_dir
|
||||
self.fuse_gate_up_exps = fuse_gate_up_exps
|
||||
self._gate_exp_buffer: dict[int, Tensor] = {}
|
||||
self._up_exp_buffer: dict[int, Tensor] = {}
|
||||
@@ -2812,6 +2814,9 @@ class StableLMModel(TextModel):
|
||||
"VLlama3ForCausalLM",
|
||||
"LlavaForConditionalGeneration",
|
||||
"VoxtralForConditionalGeneration",
|
||||
"LlamaForCausalLMEagle3",
|
||||
"Eagle3Speculator",
|
||||
"Eagle3DraftModel",
|
||||
"IQuestCoderForCausalLM",
|
||||
"LlamaModel")
|
||||
class LlamaModel(TextModel):
|
||||
@@ -2826,7 +2831,60 @@ class LlamaModel(TextModel):
|
||||
hparams = ModelBase.load_hparams(self.dir_model, is_mistral_format=False)
|
||||
self.origin_hf_arch = hparams.get('architectures', [None])[0]
|
||||
|
||||
# detect EAGLE-3 llama checkpoint
|
||||
if "draft_vocab_size" in self.hparams and self.hparams["num_hidden_layers"] == 1:
|
||||
self.is_eagle3 = True
|
||||
self.model_arch = gguf.MODEL_ARCH.EAGLE3
|
||||
logger.info("Detected EAGLE-3 draft model, switching to EAGLE3 architecture")
|
||||
# Re-initialize tensor_map with EAGLE3 architecture
|
||||
self.tensor_map = gguf.get_tensor_name_map(self.model_arch, self.block_count)
|
||||
# Update gguf_writer architecture
|
||||
self.gguf_writer.arch = gguf.MODEL_ARCH_NAMES[self.model_arch]
|
||||
self.gguf_writer.add_architecture()
|
||||
if not hasattr(self, 'target_model_dir') or not self.target_model_dir:
|
||||
raise ValueError(
|
||||
"EAGLE3 model requires --target-model-dir to be specified. "
|
||||
"Please provide the path to the target model directory to read config.json"
|
||||
)
|
||||
# Read both EAGLE3 raw config and target model config
|
||||
with open(self.dir_model / "config.json", 'r', encoding='utf-8') as f:
|
||||
eagle3_raw_config = json.load(f)
|
||||
with open(self.target_model_dir / "config.json", 'r', encoding='utf-8') as f:
|
||||
target_config = json.load(f)
|
||||
|
||||
# EAGLE3 extract_layers
|
||||
target_num_layers = target_config["num_hidden_layers"]
|
||||
extract_layers = [2, target_num_layers // 2, target_num_layers - 3]
|
||||
logger.info(f"EAGLE3: extract_layers = {extract_layers} (target model has {target_num_layers} layers)")
|
||||
self.gguf_writer.add_array(f"{self.gguf_writer.arch}.extract_layers", extract_layers)
|
||||
|
||||
# EAGLE3 target_hidden_size: prefer EAGLE3 config, fallback to target config
|
||||
if "target_hidden_size" in eagle3_raw_config and eagle3_raw_config["target_hidden_size"] is not None:
|
||||
target_hidden_size = eagle3_raw_config["target_hidden_size"]
|
||||
logger.info(f"EAGLE3: target_hidden_size = {target_hidden_size} (from EAGLE3 config)")
|
||||
else:
|
||||
target_hidden_size = target_config["hidden_size"]
|
||||
logger.info(f"EAGLE3: target_hidden_size = {target_hidden_size} (from target model config)")
|
||||
self.gguf_writer.add_uint32(f"{self.gguf_writer.arch}.target_hidden_size", target_hidden_size)
|
||||
|
||||
# Eagle3Speculator norm_before_residual specific handling
|
||||
norm_before_residual = eagle3_raw_config.get("norm_before_residual", False)
|
||||
logger.info(f"EAGLE3: norm_before_residual = {norm_before_residual} (from EAGLE3 config)")
|
||||
self.gguf_writer.add_bool(f"{self.gguf_writer.arch}.norm_before_residual", norm_before_residual)
|
||||
|
||||
def set_vocab(self):
|
||||
# For EAGLE-3 models, use tokenizer from target model if provided
|
||||
if hasattr(self, 'is_eagle3') and self.is_eagle3:
|
||||
if self.target_model_dir is None:
|
||||
raise ValueError(
|
||||
"EAGLE-3 draft model requires --target-model-dir to be specified. "
|
||||
"Please provide the path to the target model directory containing the tokenizer."
|
||||
)
|
||||
logger.info(f"EAGLE-3: Using tokenizer from target model: {self.target_model_dir}")
|
||||
# Temporarily swap dir_model to load tokenizer from target model
|
||||
original_dir_model = self.dir_model
|
||||
self.dir_model = self.target_model_dir
|
||||
|
||||
if self.origin_hf_arch == "GlmasrModel":
|
||||
return self._set_vocab_glmedge()
|
||||
|
||||
@@ -2870,6 +2928,10 @@ class LlamaModel(TextModel):
|
||||
if self.hparams.get("vocab_size", 32000) == 49152:
|
||||
self.gguf_writer.add_add_bos_token(False)
|
||||
|
||||
# Restore original dir_model for EAGLE-3
|
||||
if hasattr(self, 'is_eagle3') and self.is_eagle3:
|
||||
self.dir_model = original_dir_model
|
||||
|
||||
def set_gguf_parameters(self):
|
||||
super().set_gguf_parameters()
|
||||
hparams = self.hparams
|
||||
@@ -2905,7 +2967,55 @@ class LlamaModel(TextModel):
|
||||
|
||||
_experts: list[dict[str, Tensor]] | None = None
|
||||
|
||||
def index_tensors(self, remote_hf_model_id: str | None = None) -> dict[str, Callable[[], Tensor]]:
|
||||
tensors = super().index_tensors(remote_hf_model_id)
|
||||
|
||||
# Handle Eagle3Speculator nested config
|
||||
if "transformer_layer_config" in self.hparams:
|
||||
self.hparams = {**self.hparams, **self.hparams["transformer_layer_config"]}
|
||||
|
||||
# EAGLE-3 detection: check hparams directly (before self.is_eagle3 is set)
|
||||
if "draft_vocab_size" in self.hparams and self.hparams["num_hidden_layers"] == 1:
|
||||
logger.info("EAGLE-3: Renaming midlayer.* or layers.0.* to model.layers.0.*")
|
||||
new_tensors = {}
|
||||
# EAGLE-3: rename midlayer.* to model.layers.0.* for compatibility with llama model
|
||||
for name, gen in tensors.items():
|
||||
if name.startswith("midlayer."):
|
||||
new_name = "model.layers.0." + name[len("midlayer."):]
|
||||
new_tensors[new_name] = gen
|
||||
elif name.startswith("layers.0."): # layers.0.* -> model.layers.0.* (Eagle3Speculator format)
|
||||
new_name = "model." + name
|
||||
new_tensors[new_name] = gen
|
||||
else:
|
||||
new_tensors[name] = gen
|
||||
return new_tensors
|
||||
else:
|
||||
return tensors
|
||||
|
||||
def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
|
||||
|
||||
# Eagle-3 llama checkpoint special handling
|
||||
if hasattr(self, 'is_eagle3') and self.is_eagle3:
|
||||
# Eagle-3 llama checkpoint special weights handling
|
||||
# fc.weight: feature fusion layer
|
||||
if name == "fc.weight":
|
||||
yield (name, data_torch)
|
||||
return
|
||||
# d2t: draft to target vocabulary mapping
|
||||
elif name == "d2t":
|
||||
# Skip parent class processing (store for manual handling in prepare_tensors)
|
||||
if not hasattr(self, '_eagle3_int_tensors'):
|
||||
self._eagle3_int_tensors = {}
|
||||
self._eagle3_int_tensors[name] = data_torch
|
||||
return
|
||||
# t2d: target to draft vocabulary mapping (not used, skip completely)
|
||||
elif name == "t2d":
|
||||
return
|
||||
# hidden_norm: EAGLE-3 specific layer normalization
|
||||
elif name == "model.layers.0.hidden_norm.weight":
|
||||
yield ("blk.0.hidden_norm.weight", data_torch)
|
||||
return
|
||||
|
||||
n_head = self.find_hparam(["n_heads", "num_attention_heads"])
|
||||
n_kv_head = self.find_hparam(["n_kv_heads", "num_key_value_heads"])
|
||||
|
||||
@@ -2975,6 +3085,17 @@ class LlamaModel(TextModel):
|
||||
yield from super().modify_tensors(data_torch, name, bid)
|
||||
|
||||
def generate_extra_tensors(self) -> Iterable[tuple[str, Tensor]]:
|
||||
# EAGLE3: If no lm_head in draft model, load from target model
|
||||
if hasattr(self, 'is_eagle3') and self.is_eagle3 and "lm_head.weight" not in self.model_tensors:
|
||||
from safetensors import safe_open
|
||||
for sf_file in self.target_model_dir.glob("*.safetensors"):
|
||||
with safe_open(sf_file, framework="pt") as f:
|
||||
if "lm_head.weight" in f.keys():
|
||||
lm_head = f.get_tensor("lm_head.weight")
|
||||
logger.info(f"EAGLE3: No lm_head in draft model, loaded lm_head from {sf_file.name}, shape = {lm_head.shape}")
|
||||
yield ("output.weight", lm_head)
|
||||
break
|
||||
|
||||
if rope_params := self.rope_parameters.get("full_attention", self.rope_parameters):
|
||||
if rope_params.get("rope_type", '').lower() == "llama3":
|
||||
base = rope_params.get("rope_theta", 10000.0)
|
||||
@@ -3005,8 +3126,26 @@ class LlamaModel(TextModel):
|
||||
yield (self.format_tensor_name(gguf.MODEL_TENSOR.ROPE_FREQS), torch.tensor(rope_factors, dtype=torch.float32))
|
||||
|
||||
def prepare_tensors(self):
|
||||
# EAGLE-3: collect original dtypes BEFORE parent class converts them to F32
|
||||
eagle3_original_dtypes = {}
|
||||
if hasattr(self, 'is_eagle3') and self.is_eagle3:
|
||||
for name, data_torch in self.get_tensors():
|
||||
if name == "d2t":
|
||||
eagle3_original_dtypes[name] = data_torch.dtype
|
||||
|
||||
super().prepare_tensors()
|
||||
|
||||
if hasattr(self, 'is_eagle3') and self.is_eagle3 and hasattr(self, '_eagle3_int_tensors'):
|
||||
for name, data_torch in self._eagle3_int_tensors.items():
|
||||
old_dtype = eagle3_original_dtypes.get(name, data_torch.dtype)
|
||||
# Keep as int64 to match original torch tensor dtype
|
||||
data = data_torch.to(torch.int64).numpy()
|
||||
data_qtype = gguf.GGMLQuantizationType.I64
|
||||
|
||||
shape_str = f"{{{', '.join(str(n) for n in reversed(data.shape))}}}"
|
||||
logger.info(f"{name + ',':<30} {old_dtype} --> {data_qtype.name}, shape = {shape_str}")
|
||||
self.gguf_writer.add_tensor(name, data, raw_dtype=data_qtype)
|
||||
|
||||
if self._experts is not None:
|
||||
# flatten `list[dict[str, Tensor]]` into `list[str]`
|
||||
experts = [k for d in self._experts for k in d.keys()]
|
||||
@@ -13244,6 +13383,7 @@ class LazyTorchTensor(gguf.LazyBase):
|
||||
torch.float16: np.float16,
|
||||
torch.float32: np.float32,
|
||||
torch.uint8: np.uint8,
|
||||
torch.int64: np.int64,
|
||||
}
|
||||
|
||||
# only used when byteswapping data. Only correct size is needed
|
||||
@@ -13406,6 +13546,10 @@ def parse_args() -> argparse.Namespace:
|
||||
"--no-tensor-first-split", action="store_true",
|
||||
help="do not add tensors to the first split (disabled by default)"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--target-model-dir", type=str, default=None,
|
||||
help="directory containing target model tokenizer (for EAGLE-3 draft models that don't have their own tokenizer)",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--metadata", type=Path,
|
||||
help="Specify the path for an authorship metadata override file"
|
||||
@@ -13590,6 +13734,7 @@ def main() -> None:
|
||||
small_first_shard=args.no_tensor_first_split,
|
||||
remote_hf_model_id=hf_repo_id, disable_mistral_community_chat_template=disable_mistral_community_chat_template,
|
||||
sentence_transformers_dense_modules=args.sentence_transformers_dense_modules,
|
||||
target_model_dir=Path(args.target_model_dir) if args.target_model_dir else None,
|
||||
fuse_gate_up_exps=args.fuse_gate_up_exps
|
||||
)
|
||||
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
#include "speculative.h"
|
||||
#include "log.h"
|
||||
#include "llama.h"
|
||||
#include "chat.h"
|
||||
|
||||
#include <clocale>
|
||||
#include <cstdio>
|
||||
@@ -103,13 +104,53 @@ int main(int argc, char ** argv) {
|
||||
return 1;
|
||||
}
|
||||
|
||||
params.speculative.draft.model = model_dft.get();
|
||||
params.speculative.draft.model_tgt = model_tgt;
|
||||
params.speculative.draft.model = model_dft.get();
|
||||
params.speculative.draft.cparams = common_context_params_to_llama(params_dft);
|
||||
|
||||
if (params.speculative.draft.eagle3) {
|
||||
llama_set_eagle3(ctx_tgt, model_dft.get());
|
||||
}
|
||||
}
|
||||
|
||||
// Apply chat template for EAGLE3 if available which can increase the acceptance rate
|
||||
std::string prompt = params.prompt;
|
||||
if (params.speculative.draft.eagle3) {
|
||||
auto chat_templates = common_chat_templates_init(model_tgt, params.chat_template);
|
||||
if (common_chat_templates_was_explicit(chat_templates.get())) {
|
||||
std::vector<common_chat_msg> chat_msgs;
|
||||
common_chat_msg user_msg;
|
||||
user_msg.role = "user";
|
||||
user_msg.content = params.prompt;
|
||||
chat_msgs.push_back(user_msg);
|
||||
|
||||
common_chat_templates_inputs inputs;
|
||||
inputs.messages = chat_msgs;
|
||||
inputs.add_generation_prompt = true;
|
||||
prompt = common_chat_templates_apply(chat_templates.get(), inputs).prompt;
|
||||
LOG_INF("%s: EAGLE3 chat template applied\n", __func__);
|
||||
}
|
||||
}
|
||||
|
||||
int n_predict = 0;
|
||||
int n_drafted = 0;
|
||||
int n_accept = 0;
|
||||
|
||||
// used to determine end of generation
|
||||
bool has_eos = false;
|
||||
|
||||
// ================================================
|
||||
// everything until here is standard initialization
|
||||
// the relevant stuff for speculative decoding starts here
|
||||
|
||||
const auto t_enc_start = ggml_time_us();
|
||||
|
||||
// target model sampling context
|
||||
common_sampler_ptr smpl(common_sampler_init(model_tgt, params.sampling));
|
||||
|
||||
// Tokenize the prompt
|
||||
std::vector<llama_token> inp;
|
||||
inp = common_tokenize(ctx_tgt, params.prompt, true, true);
|
||||
inp = common_tokenize(ctx_tgt, prompt, true, true);
|
||||
|
||||
if (llama_n_ctx(ctx_tgt) < (uint32_t) inp.size()) {
|
||||
LOG_ERR("%s: the prompt exceeds the context size (%d tokens, ctx %d)\n", __func__, (int) inp.size(), llama_n_ctx(ctx_tgt));
|
||||
@@ -129,33 +170,39 @@ int main(int argc, char ** argv) {
|
||||
LOG("%s", common_token_to_piece(ctx_tgt, id).c_str());
|
||||
}
|
||||
|
||||
int n_predict = 0;
|
||||
int n_drafted = 0;
|
||||
int n_accept = 0;
|
||||
|
||||
// used to determine end of generation
|
||||
bool has_eos = false;
|
||||
|
||||
// ================================================
|
||||
// everything until here is standard initialization
|
||||
// the relevant stuff for speculative decoding starts here
|
||||
|
||||
const auto t_enc_start = ggml_time_us();
|
||||
|
||||
// target model sampling context
|
||||
common_sampler_ptr smpl(common_sampler_init(model_tgt, params.sampling));
|
||||
|
||||
// eval the prompt
|
||||
llama_decode(ctx_tgt, llama_batch_get_one(inp.data(), inp.size() - 1));
|
||||
llama_token id_last;
|
||||
llama_tokens prompt_tgt;
|
||||
int n_past;
|
||||
|
||||
// note: keep the last token separate!
|
||||
llama_token id_last = inp.back();
|
||||
// TODO: simplify
|
||||
if (params.speculative.draft.eagle3) {
|
||||
// Target model decodes full prompt and sample first token and intermediate features are extracted
|
||||
llama_decode(ctx_tgt, llama_batch_get_one(inp.data(), inp.size()));
|
||||
|
||||
// all tokens currently in the target context
|
||||
llama_tokens prompt_tgt(inp.begin(), inp.end() - 1);
|
||||
prompt_tgt.reserve(llama_n_ctx(ctx_tgt));
|
||||
id_last = common_sampler_sample(smpl.get(), ctx_tgt, -1);
|
||||
common_sampler_accept(smpl.get(), id_last, true);
|
||||
LOG("%s", common_token_to_piece(ctx_tgt, id_last).c_str());
|
||||
n_predict++;
|
||||
|
||||
int n_past = inp.size() - 1;
|
||||
// all tokens currently in the target context
|
||||
prompt_tgt.assign(inp.begin(), inp.end());
|
||||
prompt_tgt.reserve(llama_n_ctx(ctx_tgt));
|
||||
|
||||
n_past = inp.size();
|
||||
} else {
|
||||
llama_decode(ctx_tgt, llama_batch_get_one(inp.data(), inp.size() - 1));
|
||||
|
||||
// note: keep the last token separate!
|
||||
id_last = inp.back();
|
||||
|
||||
// all tokens currently in the target context
|
||||
prompt_tgt.assign(inp.begin(), inp.end() - 1);
|
||||
prompt_tgt.reserve(llama_n_ctx(ctx_tgt));
|
||||
|
||||
n_past = inp.size() - 1;
|
||||
}
|
||||
|
||||
// init the speculator
|
||||
const auto & params_spec = params.speculative;
|
||||
|
||||
@@ -152,6 +152,9 @@ class Keys:
|
||||
SWIGLU_CLAMP_SHEXP = "{arch}.swiglu_clamp_shexp"
|
||||
DENSE_FEAT_IN_SIZE = "{arch}.{dense}_feat_in"
|
||||
DENSE_FEAT_OUT_SIZE = "{arch}.{dense}_feat_out"
|
||||
EAGLE3_EXTRACT_LAYERS = "{arch}.extract_layers"
|
||||
EAGLE3_TARGET_HIDDEN_SIZE = "{arch}.target_hidden_size"
|
||||
EAGLE3_NORM_BEFORE_RESIDUAL = "{arch}.norm_before_residual"
|
||||
|
||||
class Attention:
|
||||
HEAD_COUNT = "{arch}.attention.head_count"
|
||||
@@ -489,6 +492,7 @@ class MODEL_ARCH(IntEnum):
|
||||
RND1 = auto()
|
||||
PANGU_EMBED = auto()
|
||||
MISTRAL3 = auto()
|
||||
EAGLE3 = auto()
|
||||
MISTRAL4 = auto()
|
||||
PADDLEOCR = auto()
|
||||
MIMO2 = auto()
|
||||
@@ -844,6 +848,10 @@ class MODEL_TENSOR(IntEnum):
|
||||
NEXTN_HNORM = auto()
|
||||
NEXTN_SHARED_HEAD_HEAD = auto()
|
||||
NEXTN_SHARED_HEAD_NORM = auto()
|
||||
# EAGLE3 specific tensors
|
||||
EAGLE3_FC = auto() # feature fusion layer
|
||||
EAGLE3_HIDDEN_NORM = auto() # hidden normalization
|
||||
EAGLE3_D2T = auto() # draft to target vocabulary mapping
|
||||
# lfm2 audio
|
||||
A_ENC_NORM_CONV = auto()
|
||||
A_ENC_LINEAR_POS = auto()
|
||||
@@ -976,6 +984,7 @@ MODEL_ARCH_NAMES: dict[MODEL_ARCH, str] = {
|
||||
MODEL_ARCH.RND1: "rnd1",
|
||||
MODEL_ARCH.PANGU_EMBED: "pangu-embedded",
|
||||
MODEL_ARCH.MISTRAL3: "mistral3",
|
||||
MODEL_ARCH.EAGLE3: "eagle3",
|
||||
MODEL_ARCH.MISTRAL4: "mistral4",
|
||||
MODEL_ARCH.PADDLEOCR: "paddleocr",
|
||||
MODEL_ARCH.MIMO2: "mimo2",
|
||||
@@ -1340,6 +1349,9 @@ TENSOR_NAMES: dict[MODEL_TENSOR, str] = {
|
||||
MODEL_TENSOR.NEXTN_HNORM: "blk.{bid}.nextn.hnorm",
|
||||
MODEL_TENSOR.NEXTN_SHARED_HEAD_HEAD: "blk.{bid}.nextn.shared_head_head",
|
||||
MODEL_TENSOR.NEXTN_SHARED_HEAD_NORM: "blk.{bid}.nextn.shared_head_norm",
|
||||
MODEL_TENSOR.EAGLE3_FC: "fc",
|
||||
MODEL_TENSOR.EAGLE3_HIDDEN_NORM: "blk.{bid}.hidden_norm",
|
||||
MODEL_TENSOR.EAGLE3_D2T: "d2t",
|
||||
}
|
||||
|
||||
MODEL_TENSORS: dict[MODEL_ARCH, list[MODEL_TENSOR]] = {
|
||||
@@ -3742,6 +3754,24 @@ MODEL_TENSORS: dict[MODEL_ARCH, list[MODEL_TENSOR]] = {
|
||||
MODEL_TENSOR.FFN_DOWN_EXP,
|
||||
MODEL_TENSOR.FFN_UP_EXP,
|
||||
],
|
||||
MODEL_ARCH.EAGLE3: [
|
||||
MODEL_TENSOR.TOKEN_EMBD,
|
||||
MODEL_TENSOR.OUTPUT_NORM,
|
||||
MODEL_TENSOR.OUTPUT,
|
||||
MODEL_TENSOR.ROPE_FREQS,
|
||||
MODEL_TENSOR.ATTN_NORM,
|
||||
MODEL_TENSOR.ATTN_Q,
|
||||
MODEL_TENSOR.ATTN_K,
|
||||
MODEL_TENSOR.ATTN_V,
|
||||
MODEL_TENSOR.ATTN_OUT,
|
||||
MODEL_TENSOR.FFN_NORM,
|
||||
MODEL_TENSOR.FFN_GATE,
|
||||
MODEL_TENSOR.FFN_DOWN,
|
||||
MODEL_TENSOR.FFN_UP,
|
||||
MODEL_TENSOR.EAGLE3_FC,
|
||||
MODEL_TENSOR.EAGLE3_HIDDEN_NORM,
|
||||
MODEL_TENSOR.EAGLE3_D2T,
|
||||
],
|
||||
MODEL_ARCH.MISTRAL4: [
|
||||
MODEL_TENSOR.TOKEN_EMBD,
|
||||
MODEL_TENSOR.OUTPUT_NORM,
|
||||
|
||||
@@ -126,6 +126,7 @@ static const std::map<llm_arch, const char *> LLM_ARCH_NAMES = {
|
||||
{ LLM_ARCH_RND1, "rnd1" },
|
||||
{ LLM_ARCH_PANGU_EMBED, "pangu-embedded" },
|
||||
{ LLM_ARCH_MISTRAL3, "mistral3" },
|
||||
{ LLM_ARCH_EAGLE3, "eagle3" },
|
||||
{ LLM_ARCH_MISTRAL4, "mistral4" },
|
||||
{ LLM_ARCH_PADDLEOCR, "paddleocr" },
|
||||
{ LLM_ARCH_MIMO2, "mimo2" },
|
||||
@@ -284,6 +285,10 @@ static const std::map<llm_kv, const char *> LLM_KV_NAMES = {
|
||||
|
||||
{ LLM_KV_CLASSIFIER_OUTPUT_LABELS, "%s.classifier.output_labels" },
|
||||
|
||||
{ LLM_KV_EAGLE3_EXTRACT_LAYERS, "%s.extract_layers" },
|
||||
{ LLM_KV_EAGLE3_TARGET_HIDDEN_SIZE, "%s.target_hidden_size" },
|
||||
{ LLM_KV_EAGLE3_NORM_BEFORE_RESIDUAL, "%s.norm_before_residual" },
|
||||
|
||||
{ LLM_KV_SHORTCONV_L_CACHE, "%s.shortconv.l_cache" },
|
||||
// sentence-transformers dense modules feature dims
|
||||
{ LLM_KV_DENSE_2_FEAT_IN, "%s.dense_2_feat_in" },
|
||||
@@ -547,6 +552,10 @@ static const std::map<llm_tensor, const char *> LLM_TENSOR_NAMES = {
|
||||
{ LLM_TENSOR_INDEXER_PROJ, "blk.%d.indexer.proj" },
|
||||
{ LLM_TENSOR_INDEXER_ATTN_K, "blk.%d.indexer.attn_k" },
|
||||
{ LLM_TENSOR_INDEXER_ATTN_Q_B, "blk.%d.indexer.attn_q_b" },
|
||||
// EAGLE-3 specific layers
|
||||
{ LLM_TENSOR_EAGLE3_HIDDEN_NORM, "blk.%d.hidden_norm" },
|
||||
{ LLM_TENSOR_EAGLE3_FC, "fc" },
|
||||
{ LLM_TENSOR_EAGLE3_D2T, "d2t" },
|
||||
};
|
||||
|
||||
// declare information about the model weight tensors:
|
||||
@@ -767,6 +776,10 @@ static const std::map<llm_tensor, llm_tensor_info> LLM_TENSOR_INFOS = {
|
||||
// Nemotron 3 Super
|
||||
{LLM_TENSOR_FFN_LATENT_DOWN, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL}},
|
||||
{LLM_TENSOR_FFN_LATENT_UP, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL}},
|
||||
// EAGLE-3 tensors
|
||||
{LLM_TENSOR_EAGLE3_FC, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL_MAT}},
|
||||
{LLM_TENSOR_EAGLE3_HIDDEN_NORM, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL}},
|
||||
{LLM_TENSOR_EAGLE3_D2T, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_GET_ROWS}},
|
||||
};
|
||||
|
||||
LLM_KV::LLM_KV(llm_arch arch, const char * suffix) : arch(arch), suffix(suffix) {}
|
||||
|
||||
@@ -138,6 +138,7 @@ enum llm_arch {
|
||||
LLM_ARCH_MAINCODER,
|
||||
LLM_ARCH_KIMI_LINEAR,
|
||||
LLM_ARCH_UNKNOWN,
|
||||
LLM_ARCH_EAGLE3,
|
||||
};
|
||||
|
||||
enum llm_kv {
|
||||
@@ -326,6 +327,10 @@ enum llm_kv {
|
||||
|
||||
LLM_KV_CLASSIFIER_OUTPUT_LABELS,
|
||||
|
||||
LLM_KV_EAGLE3_EXTRACT_LAYERS,
|
||||
LLM_KV_EAGLE3_TARGET_HIDDEN_SIZE,
|
||||
LLM_KV_EAGLE3_NORM_BEFORE_RESIDUAL,
|
||||
|
||||
LLM_KV_SHORTCONV_L_CACHE,
|
||||
|
||||
LLM_KV_XIELU_ALPHA_N,
|
||||
@@ -554,6 +559,9 @@ enum llm_tensor {
|
||||
LLM_TENSOR_NEXTN_HNORM,
|
||||
LLM_TENSOR_NEXTN_SHARED_HEAD_HEAD,
|
||||
LLM_TENSOR_NEXTN_SHARED_HEAD_NORM,
|
||||
LLM_TENSOR_EAGLE3_FC, // eagle3: feature fusion layer
|
||||
LLM_TENSOR_EAGLE3_HIDDEN_NORM, // eagle3: additional normalization layer
|
||||
LLM_TENSOR_EAGLE3_D2T, // eagle3: draft to target vocabulary mapping
|
||||
};
|
||||
|
||||
enum llm_tensor_layer {
|
||||
|
||||
+128
-6
@@ -65,6 +65,8 @@ llama_context::llama_context(
|
||||
cparams.cb_eval = params.cb_eval;
|
||||
cparams.cb_eval_user_data = params.cb_eval_user_data;
|
||||
|
||||
cparams.output_layer_inp.resize(hparams.n_layer, false);
|
||||
|
||||
// Initialize backend samplers here so they are part of the sampling graph
|
||||
// before the reserve passes run later in this function. This avoids a later
|
||||
// re-reserve when graph nodes change.
|
||||
@@ -1168,6 +1170,16 @@ bool llama_context::set_adapter_cvec(
|
||||
return res;
|
||||
}
|
||||
|
||||
void llama_context::set_output_layer_inp(uint32_t layer_id, bool enable) {
|
||||
LLAMA_LOG_DEBUG("%s: layer_id = %d, enable = %d\n", __func__, layer_id, enable);
|
||||
|
||||
GGML_ASSERT(layer_id < model.hparams.n_layer);
|
||||
|
||||
cparams.output_layer_inp[layer_id] = enable;
|
||||
|
||||
sched_need_reserve = true;
|
||||
}
|
||||
|
||||
llm_graph_result * llama_context::process_ubatch(const llama_ubatch & ubatch, llm_graph_type gtype, llama_memory_context_i * mctx, ggml_status & ret) {
|
||||
if (mctx && !mctx->apply()) {
|
||||
LLAMA_LOG_ERROR("%s: failed to apply memory context\n", __func__);
|
||||
@@ -1225,6 +1237,14 @@ llm_graph_result * llama_context::process_ubatch(const llama_ubatch & ubatch, ll
|
||||
// FIXME this call causes a crash if any model inputs were not used in the graph and were therefore not allocated
|
||||
res->set_inputs(&ubatch);
|
||||
|
||||
// EAGLE3: Fill g_embeddings for decoder input
|
||||
if (model.arch == LLM_ARCH_EAGLE3 && gtype == LLM_GRAPH_TYPE_DECODER && !eagle3.g_embeddings.empty()) {
|
||||
ggml_tensor * g_embd = ggml_graph_get_tensor(gf, "inp_g_embeddings");
|
||||
if (g_embd) {
|
||||
ggml_backend_tensor_set(g_embd, eagle3.g_embeddings.data(), 0, ggml_nbytes(g_embd));
|
||||
}
|
||||
}
|
||||
|
||||
//LLAMA_LOG_INFO("graph set inputs time: %.3f ms\n", (ggml_time_us() - t_start_us)/1000.0);
|
||||
}
|
||||
|
||||
@@ -1336,8 +1356,15 @@ int llama_context::encode(const llama_batch & batch_inp) {
|
||||
GGML_ASSERT(embd.data != nullptr);
|
||||
const uint32_t n_embd_out = hparams.n_embd_out();
|
||||
|
||||
GGML_ASSERT(n_tokens*n_embd_out <= (int64_t) embd.size);
|
||||
ggml_backend_tensor_get_async(backend_embd, t_embd, embd.data, 0, n_tokens*n_embd_out*sizeof(float));
|
||||
if (model.arch == LLM_ARCH_EAGLE3) {
|
||||
// g_embeddings are stored temporarily in embd buffer
|
||||
const int64_t out_embd = hparams.n_embd;
|
||||
GGML_ASSERT(n_tokens * out_embd <= (int64_t) embd.size);
|
||||
ggml_backend_tensor_get_async(backend_embd, t_embd, embd.data, 0, n_tokens * out_embd * sizeof(float));
|
||||
} else {
|
||||
GGML_ASSERT(n_tokens*n_embd_out <= (int64_t) embd.size);
|
||||
ggml_backend_tensor_get_async(backend_embd, t_embd, embd.data, 0, n_tokens*n_embd_out*sizeof(float));
|
||||
}
|
||||
} break;
|
||||
case LLAMA_POOLING_TYPE_MEAN:
|
||||
case LLAMA_POOLING_TYPE_CLS:
|
||||
@@ -1730,7 +1757,8 @@ int llama_context::decode(const llama_batch & batch_inp) {
|
||||
auto * t_logits = res->get_logits();
|
||||
auto * t_embd = cparams.embeddings ? res->get_embd() : nullptr;
|
||||
|
||||
if (t_embd && res->get_embd_pooled()) {
|
||||
// For EAGLE3, don't override t_embd with t_embd_pooled - we need the prenorm value during eagle3 decoder autoregressive generation
|
||||
if (t_embd && res->get_embd_pooled() && model.arch != LLM_ARCH_EAGLE3) {
|
||||
t_embd = res->get_embd_pooled();
|
||||
}
|
||||
|
||||
@@ -1745,7 +1773,40 @@ int llama_context::decode(const llama_batch & batch_inp) {
|
||||
if (n_outputs) {
|
||||
GGML_ASSERT( n_outputs_prev + n_outputs <= n_outputs_all);
|
||||
GGML_ASSERT((n_outputs_prev + n_outputs)*n_vocab <= (int64_t) logits.size);
|
||||
ggml_backend_tensor_get_async(backend_res, t_logits, logits_out, 0, n_outputs*n_vocab*sizeof(float));
|
||||
|
||||
// EAGLE3: Map draft vocab to target vocab
|
||||
if (model.arch == LLM_ARCH_EAGLE3 && model.d2t) {
|
||||
static thread_local std::vector<int64_t> eagle3_d2t_map;
|
||||
static thread_local std::vector<float> eagle3_draft_logits;
|
||||
|
||||
const int64_t draft_vocab_size = t_logits->ne[0];
|
||||
const uint32_t last_idx = n_outputs - 1;
|
||||
|
||||
// Load d2t mapping once (on first call)
|
||||
if (eagle3_d2t_map.empty()) {
|
||||
eagle3_d2t_map.resize(model.d2t->ne[0]);
|
||||
ggml_backend_tensor_get(model.d2t, eagle3_d2t_map.data(), 0, eagle3_d2t_map.size() * sizeof(int64_t));
|
||||
}
|
||||
|
||||
// Read only the last token's draft logits
|
||||
eagle3_draft_logits.resize(draft_vocab_size);
|
||||
const size_t last_offset = last_idx * draft_vocab_size * sizeof(float);
|
||||
ggml_backend_tensor_get_async(backend_res, t_logits, eagle3_draft_logits.data(), last_offset, draft_vocab_size * sizeof(float));
|
||||
synchronize();
|
||||
|
||||
|
||||
// Map only the last token's draft logits to target vocab
|
||||
float * last_logits_out = logits_out + last_idx * n_vocab;
|
||||
std::fill(last_logits_out, last_logits_out + n_vocab, -std::numeric_limits<float>::infinity());
|
||||
|
||||
for (int64_t j = 0; j < draft_vocab_size; j++) {
|
||||
const int64_t target_id = j + eagle3_d2t_map[j];
|
||||
GGML_ASSERT(target_id >= 0 && target_id < n_vocab);
|
||||
last_logits_out[target_id] = eagle3_draft_logits[j];
|
||||
}
|
||||
} else {
|
||||
ggml_backend_tensor_get_async(backend_res, t_logits, logits_out, 0, n_outputs*n_vocab*sizeof(float));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1904,7 +1965,6 @@ uint32_t llama_context::output_reserve(int32_t n_outputs) {
|
||||
has_embd = true;
|
||||
}
|
||||
|
||||
|
||||
size_t backend_float_count = 0;
|
||||
size_t backend_token_count = 0;
|
||||
|
||||
@@ -2121,7 +2181,16 @@ ggml_cgraph * llama_context::graph_reserve(
|
||||
|
||||
auto * res = gf_res_reserve.get();
|
||||
|
||||
const auto gparams = graph_params(res, ubatch, mctx, LLM_GRAPH_TYPE_DEFAULT);
|
||||
// EAGLE3: auto-detect encoder (embeddings+no target_model) or decoder (has target_model)
|
||||
llm_graph_type gtype = LLM_GRAPH_TYPE_DEFAULT;
|
||||
if (model.arch == LLM_ARCH_EAGLE3) {
|
||||
if (cparams.embeddings && model.target_tok_embd == nullptr) {
|
||||
gtype = LLM_GRAPH_TYPE_ENCODER;
|
||||
} else if (model.target_tok_embd != nullptr) {
|
||||
gtype = LLM_GRAPH_TYPE_DECODER;
|
||||
}
|
||||
}
|
||||
const auto gparams = graph_params(res, ubatch, mctx, gtype);
|
||||
|
||||
res->reset();
|
||||
|
||||
@@ -2162,6 +2231,7 @@ llm_graph_params llama_context::graph_params(
|
||||
/*.loras =*/ loras.get(),
|
||||
/*.mctx =*/ mctx,
|
||||
/*.cross =*/ &cross,
|
||||
/*.eagle3 =*/ &eagle3,
|
||||
/*.samplers =*/ sampling.samplers,
|
||||
/*.n_outputs =*/ n_outputs,
|
||||
/*.cb =*/ graph_get_cb(),
|
||||
@@ -2224,6 +2294,54 @@ llm_graph_cb llama_context::graph_get_cb() const {
|
||||
};
|
||||
}
|
||||
|
||||
void llama_context::extract_eagle3_features(const llama_ubatch & ubatch) {
|
||||
const int64_t n_tokens = ubatch.n_tokens;
|
||||
const int64_t n_embd = model.hparams.n_embd;
|
||||
const size_t n_layers = eagle3.extract_tensors.size();
|
||||
|
||||
// Allocate storage for concatenated features
|
||||
const int64_t n_embd_concat = n_embd * n_layers;
|
||||
eagle3.target_features.resize(n_embd_concat * n_tokens);
|
||||
|
||||
// Temporary buffer to hold layer features before transposing
|
||||
static thread_local std::vector<float> temp_layer_features;
|
||||
temp_layer_features.resize(n_embd * n_tokens);
|
||||
|
||||
LLAMA_LOG_DEBUG("%s: Start to extract EAGLE3 features: %zu layers, %lld tokens, %lld embd\n",
|
||||
__func__, n_layers, (long long)n_tokens, (long long)n_embd);
|
||||
|
||||
// Extract each layer's features and interleave into token-major layout
|
||||
for (size_t layer_idx = 0; layer_idx < n_layers; ++layer_idx) {
|
||||
ggml_tensor * tensor = eagle3.extract_tensors[layer_idx];
|
||||
GGML_ASSERT(tensor != nullptr && "EAGLE3 extraction tensor is null");
|
||||
|
||||
// Get the backend where this tensor is stored
|
||||
ggml_backend_t backend = ggml_backend_sched_get_tensor_backend(sched.get(), tensor);
|
||||
GGML_ASSERT(backend != nullptr && "EAGLE3 tensor has no backend");
|
||||
|
||||
// Verify tensor shape: should be [n_embd, n_tokens]
|
||||
GGML_ASSERT(tensor->ne[0] == n_embd && tensor->ne[1] == n_tokens &&
|
||||
"EAGLE3 extraction tensor has unexpected shape");
|
||||
|
||||
// Get layer features to temp buffer
|
||||
const size_t size_bytes = n_embd * n_tokens * sizeof(float);
|
||||
ggml_backend_tensor_get_async(backend, tensor, temp_layer_features.data(), 0, size_bytes);
|
||||
ggml_backend_sched_synchronize(sched.get());
|
||||
|
||||
// Then copy to correct position in target_features
|
||||
// target_features layout: [token_0_all_layers, token_1_all_layers, ...]
|
||||
// Each token has [layer_0_embd, layer_1_embd, layer_2_embd]
|
||||
for (int64_t token_idx = 0; token_idx < n_tokens; ++token_idx) {
|
||||
// Source: temp_layer_features[token_idx * n_embd ... (token_idx + 1) * n_embd - 1]
|
||||
const float * src = temp_layer_features.data() + token_idx * n_embd;
|
||||
// Dest: target_features[token_idx * n_embd_concat + layer_idx * n_embd]
|
||||
float * dest = eagle3.target_features.data() + token_idx * n_embd_concat + layer_idx * n_embd;
|
||||
std::memcpy(dest, src, n_embd * sizeof(float));
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
//
|
||||
// state save/load
|
||||
//
|
||||
@@ -3779,3 +3897,7 @@ void llama_opt_epoch(
|
||||
llama_memory_breakdown llama_get_memory_breakdown(const struct llama_context * ctx) {
|
||||
return ctx->memory_breakdown();
|
||||
}
|
||||
|
||||
void llama_set_output_layer_inp(struct llama_context * ctx, uint32_t layer_id, bool enable) {
|
||||
ctx->set_output_layer_inp(layer_id, enable);
|
||||
}
|
||||
|
||||
@@ -121,6 +121,8 @@ struct llama_context {
|
||||
int32_t il_start,
|
||||
int32_t il_end);
|
||||
|
||||
void set_output_layer_inp(uint32_t layer_id, bool enable);
|
||||
|
||||
// process a single ubatch with a specific graph type
|
||||
// if memory_context is provided, it will be applied first to the context's memory
|
||||
// ret contains the status of the graph computation
|
||||
@@ -238,6 +240,12 @@ public:
|
||||
ggml_cgraph * graph_reserve(
|
||||
uint32_t n_tokens, uint32_t n_seqs, uint32_t n_outputs, const llama_memory_context_i * mctx, bool split_only = false, size_t * sizes = nullptr);
|
||||
|
||||
// EAGLE3: Get pointer to target model features extracted for EAGLE3 encoder
|
||||
const float * get_eagle3_target_features() const;
|
||||
|
||||
// EAGLE3: Set g_embeddings from encoder output for decoder input
|
||||
void set_eagle3_g_embeddings(const float * g_embd, int32_t n_embd, int32_t n_tokens);
|
||||
|
||||
bool set_sampler(llama_seq_id seq_id, llama_sampler * sampler);
|
||||
|
||||
private:
|
||||
@@ -249,6 +257,9 @@ private:
|
||||
|
||||
llm_graph_cb graph_get_cb() const;
|
||||
|
||||
// EAGLE3: Extract intermediate layer features from target model
|
||||
void extract_eagle3_features(const llama_ubatch & ubatch);
|
||||
|
||||
// TODO: read/write lora adapters and cvec
|
||||
size_t state_write_data(llama_io_write_i & io);
|
||||
size_t state_read_data (llama_io_read_i & io);
|
||||
@@ -269,6 +280,9 @@ private:
|
||||
|
||||
llama_cross cross; // TODO: tmp for handling cross-attention - need something better probably
|
||||
|
||||
mutable llama_eagle3 eagle3; // EAGLE3 draft model support - stores features from target model
|
||||
// mutable because it's modified during graph building (const function)
|
||||
|
||||
std::unique_ptr<llama_memory_i> memory;
|
||||
|
||||
// decode output (2-dimensional array: [n_outputs][n_vocab])
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
#include "llama.h"
|
||||
|
||||
#include <cstdint>
|
||||
#include <vector>
|
||||
|
||||
#define LLAMA_MAX_SEQ 256
|
||||
|
||||
@@ -40,6 +41,8 @@ struct llama_cparams {
|
||||
bool kv_unified;
|
||||
bool pipeline_parallel;
|
||||
|
||||
std::vector<bool> output_layer_inp;
|
||||
|
||||
enum llama_pooling_type pooling_type;
|
||||
|
||||
ggml_backend_sched_eval_callback cb_eval;
|
||||
|
||||
@@ -88,3 +88,13 @@ LLAMA_API int32_t llama_model_n_devices(const struct llama_model * model);
|
||||
LLAMA_API ggml_backend_dev_t llama_model_get_device(const struct llama_model * model, int i);
|
||||
|
||||
LLAMA_API llama_memory_breakdown llama_get_memory_breakdown(const struct llama_context * ctx);
|
||||
|
||||
//
|
||||
// model/context data extraction
|
||||
//
|
||||
|
||||
LLAMA_API void llama_set_output_layer_inp(struct llama_context * ctx, uint32_t layer_id, bool enable);
|
||||
|
||||
LLAMA_API ggml_tensor * llama_model_get_tok_embd(const struct llama_model * model);
|
||||
LLAMA_API void llama_model_set_tok_embd(struct llama_model * model, ggml_tensor * tensor);
|
||||
|
||||
|
||||
+14
-1
@@ -810,6 +810,10 @@ void llm_graph_result::reset() {
|
||||
t_logits = nullptr;
|
||||
t_embd = nullptr;
|
||||
t_embd_pooled = nullptr;
|
||||
|
||||
t_layer_inp.resize(LLAMA_MAX_LAYERS);
|
||||
std::fill(t_layer_inp.begin(), t_layer_inp.end(), nullptr);
|
||||
|
||||
t_sampled.clear();
|
||||
t_sampled_probs.clear();
|
||||
t_sampled_logits.clear();
|
||||
@@ -838,7 +842,7 @@ void llm_graph_result::set_inputs(const llama_ubatch * ubatch) {
|
||||
}
|
||||
}
|
||||
|
||||
void llm_graph_result::set_outputs() {
|
||||
void llm_graph_result::set_outputs(const llm_graph_params & params) {
|
||||
if (t_logits != nullptr) {
|
||||
ggml_set_output(t_logits);
|
||||
}
|
||||
@@ -848,6 +852,14 @@ void llm_graph_result::set_outputs() {
|
||||
if (t_embd_pooled != nullptr) {
|
||||
ggml_set_output(t_embd_pooled);
|
||||
}
|
||||
{
|
||||
const auto & output_layer_inp = params.cparams.output_layer_inp;
|
||||
for (size_t il = 0; il < output_layer_inp.size(); ++il) {
|
||||
if (output_layer_inp[il]) {
|
||||
ggml_set_output(t_layer_inp[il]);
|
||||
}
|
||||
}
|
||||
}
|
||||
for (auto & [seq_id, t] : t_sampled) {
|
||||
if (t != nullptr) {
|
||||
ggml_set_output(t);
|
||||
@@ -951,6 +963,7 @@ llm_graph_context::llm_graph_context(const llm_graph_params & params) :
|
||||
loras (params.loras),
|
||||
mctx (params.mctx),
|
||||
cross (params.cross),
|
||||
eagle3 (params.eagle3),
|
||||
samplers (params.samplers),
|
||||
cb_func (params.cb),
|
||||
res (params.res),
|
||||
|
||||
+35
-5
@@ -73,6 +73,30 @@ struct llama_cross {
|
||||
std::vector<std::set<llama_seq_id>> seq_ids_enc;
|
||||
};
|
||||
|
||||
// EAGLE3 support - stores intermediate features from target model
|
||||
struct llama_eagle3 {
|
||||
// Configuration: which layers to extract from target model
|
||||
std::vector<int> extract_layer_indices;
|
||||
|
||||
// Extracted features from target model (for encoder input)
|
||||
// Concatenated [layer_l, layer_m, layer_h] embeddings
|
||||
// Shape: [n_layers * n_embd, n_tokens] where n_layers = extract_layer_indices.size()
|
||||
std::vector<float> target_features;
|
||||
|
||||
// Encoder output (for decoder input)
|
||||
std::vector<float> g_embeddings;
|
||||
|
||||
// Tensor references for feature extraction from target model
|
||||
std::vector<ggml_tensor *> extract_tensors;
|
||||
|
||||
// Clear all stored data
|
||||
void clear() {
|
||||
target_features.clear();
|
||||
g_embeddings.clear();
|
||||
extract_tensors.clear();
|
||||
}
|
||||
};
|
||||
|
||||
struct llm_graph_params;
|
||||
|
||||
//
|
||||
@@ -544,6 +568,7 @@ struct llm_graph_params {
|
||||
const llama_adapter_loras * loras;
|
||||
const llama_memory_context_i * mctx;
|
||||
const llama_cross * cross;
|
||||
llama_eagle3 * eagle3; // non-const: we write extracted features here
|
||||
|
||||
std::map<llama_seq_id, llama_sampler *> samplers;
|
||||
|
||||
@@ -645,6 +670,8 @@ public:
|
||||
ggml_tensor * get_embd() const { return t_embd; }
|
||||
ggml_tensor * get_embd_pooled() const { return t_embd_pooled; }
|
||||
|
||||
ggml_tensor * get_layer_inp(int il) const { return t_layer_inp[il]; }
|
||||
|
||||
ggml_cgraph * get_gf() const { return gf; }
|
||||
ggml_context * get_ctx() const { return ctx_compute.get(); }
|
||||
|
||||
@@ -653,7 +680,7 @@ public:
|
||||
void reset();
|
||||
|
||||
void set_inputs(const llama_ubatch * ubatch);
|
||||
void set_outputs();
|
||||
void set_outputs(const llm_graph_params & params);
|
||||
|
||||
// try to update the existing graph result using the new graph parameters in order to reuse it
|
||||
// this can only be done if we determine that the resulting graph using the new graph parameters
|
||||
@@ -673,10 +700,12 @@ public:
|
||||
ggml_tensor * t_embd = nullptr;
|
||||
ggml_tensor * t_embd_pooled = nullptr;
|
||||
|
||||
std::map<llama_seq_id, ggml_tensor*> t_sampled_logits;
|
||||
std::map<llama_seq_id, ggml_tensor*> t_candidates;
|
||||
std::map<llama_seq_id, ggml_tensor*> t_sampled;
|
||||
std::map<llama_seq_id, ggml_tensor*> t_sampled_probs;
|
||||
std::vector<ggml_tensor *> t_layer_inp;
|
||||
|
||||
std::map<llama_seq_id, ggml_tensor *> t_sampled_logits;
|
||||
std::map<llama_seq_id, ggml_tensor *> t_candidates;
|
||||
std::map<llama_seq_id, ggml_tensor *> t_sampled;
|
||||
std::map<llama_seq_id, ggml_tensor *> t_sampled_probs;
|
||||
|
||||
std::vector<llm_graph_input_ptr> inputs;
|
||||
|
||||
@@ -758,6 +787,7 @@ struct llm_graph_context {
|
||||
const llama_adapter_loras * loras;
|
||||
const llama_memory_context_i * mctx;
|
||||
const llama_cross * cross;
|
||||
llama_eagle3 * eagle3; // non-const: we write extracted features here
|
||||
|
||||
std::map<llama_seq_id, llama_sampler *> samplers;
|
||||
|
||||
|
||||
@@ -71,7 +71,7 @@ uint32_t llama_hparams::n_rot(uint32_t il) const {
|
||||
}
|
||||
|
||||
uint32_t llama_hparams::n_embd_inp() const {
|
||||
uint32_t n_embd_inp = n_embd;
|
||||
uint32_t n_embd_inp = n_embd_inp_impl > 0 ? n_embd_inp_impl : n_embd;
|
||||
|
||||
if (n_deepstack_layers > 0) {
|
||||
n_embd_inp += n_embd * n_deepstack_layers;
|
||||
|
||||
@@ -42,6 +42,7 @@ struct llama_hparams {
|
||||
|
||||
uint32_t n_ctx_train; // context size the model was trained on
|
||||
uint32_t n_embd;
|
||||
uint32_t n_embd_inp_impl = 0;
|
||||
uint32_t n_layer;
|
||||
int32_t n_layer_kv_from_start = -1; // if non-negative, the first n_layer_kv_from_start layers have KV cache
|
||||
uint32_t n_expert = 0;
|
||||
@@ -210,6 +211,13 @@ struct llama_hparams {
|
||||
// qwen3vl deepstack
|
||||
uint32_t n_deepstack_layers = 0;
|
||||
|
||||
// EAGLE3 draft model - layer indices to extract from target model
|
||||
// e.g., for 32-layer target: [2, 16, 29] (low, middle, high)
|
||||
std::array<int, 3> eagle3_extract_layers = {0, 0, 0};
|
||||
|
||||
// EAGLE3 draft model - apply hidden_norm before storing residual
|
||||
bool eagle3_norm_before_residual = false;
|
||||
|
||||
// gemma4 per-layer embedding
|
||||
uint32_t n_embd_per_layer = 0;
|
||||
|
||||
|
||||
+12
-1
@@ -278,6 +278,8 @@ static llama_model * llama_model_mapping(llm_arch arch, const llama_model_params
|
||||
return new llama_model_qwen35moe(params);
|
||||
case LLM_ARCH_MISTRAL3:
|
||||
return new llama_model_mistral3(params);
|
||||
case LLM_ARCH_EAGLE3:
|
||||
return new llama_model_eagle3(params);
|
||||
case LLM_ARCH_MIMO2:
|
||||
return new llama_model_mimo2(params);
|
||||
case LLM_ARCH_KIMI_LINEAR:
|
||||
@@ -2071,7 +2073,7 @@ ggml_cgraph * llama_model::build_graph(const llm_graph_params & params) const {
|
||||
// TODO: move reranking logic here and generalize
|
||||
llm->build_dense_out(dense_2_out_layers, dense_2_out_layers_b, dense_3_out_layers);
|
||||
|
||||
llm->res->set_outputs();
|
||||
llm->res->set_outputs(params);
|
||||
|
||||
return llm->res->get_gf();
|
||||
}
|
||||
@@ -2237,6 +2239,7 @@ llama_rope_type llama_model_rope_type(const llama_model * model) {
|
||||
case LLM_ARCH_ERNIE4_5:
|
||||
case LLM_ARCH_ERNIE4_5_MOE:
|
||||
case LLM_ARCH_MISTRAL3:
|
||||
case LLM_ARCH_EAGLE3:
|
||||
case LLM_ARCH_MISTRAL4:
|
||||
case LLM_ARCH_LLAMA_EMBED:
|
||||
case LLM_ARCH_MAINCODER:
|
||||
@@ -2515,3 +2518,11 @@ void llama_model_base::create_tensor_qkv(llama_layer & layer, int bid,
|
||||
layer.wv_b = create_tensor(tn(LLM_TENSOR_ATTN_V, "bias", bid), {n_embd_v_}, TENSOR_NOT_REQUIRED);
|
||||
}
|
||||
}
|
||||
|
||||
ggml_tensor * llama_model_get_tok_embd(const struct llama_model * model) {
|
||||
return model->tok_embd;
|
||||
}
|
||||
|
||||
void llama_model_set_tok_embd(struct llama_model * model, ggml_tensor * tensor) {
|
||||
model->tok_embd = tensor;
|
||||
}
|
||||
|
||||
@@ -465,6 +465,9 @@ struct llama_layer {
|
||||
struct ggml_tensor * ffn_act_beta = nullptr;
|
||||
struct ggml_tensor * ffn_act_eps = nullptr;
|
||||
|
||||
// eagle3
|
||||
struct ggml_tensor * eagle3_hidden_norm = nullptr;
|
||||
|
||||
// Kimi Linear KDA (using ssm_ prefix for consistency)
|
||||
// Note: ssm_dt_b already exists above (mamba bias), reused for Kimi dt_bias
|
||||
struct ggml_tensor * ssm_q_conv = nullptr;
|
||||
@@ -550,6 +553,13 @@ struct llama_model {
|
||||
struct ggml_tensor * per_layer_model_proj = nullptr;
|
||||
struct ggml_tensor * per_layer_proj_norm = nullptr;
|
||||
|
||||
// eagle3
|
||||
struct ggml_tensor * fc = nullptr; // feature fusion layer
|
||||
struct ggml_tensor * d2t = nullptr; // draft to target vocabulary mapping
|
||||
// Reference to target model's embedding layer
|
||||
// This allows EAGLE3 to use target model's embeddings without copying
|
||||
struct ggml_tensor * target_tok_embd = nullptr;
|
||||
|
||||
std::vector<llama_layer> layers;
|
||||
|
||||
//Dense linear projections for SentenceTransformers models like embeddinggemma
|
||||
|
||||
@@ -0,0 +1,293 @@
|
||||
#include "models.h"
|
||||
|
||||
void llama_model_eagle3::load_arch_hparams(llama_model_loader & ml) {
|
||||
ml.get_key(LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, hparams.f_norm_rms_eps);
|
||||
// EAGLE3 layer extraction configuration
|
||||
// Use array<int, 4> (has template instantiation), then copy first 3 elements
|
||||
std::array<int, 4> extract_layers_tmp = {};
|
||||
if (!ml.get_key_or_arr(LLM_KV_EAGLE3_EXTRACT_LAYERS, extract_layers_tmp, 3, false)) {
|
||||
throw std::runtime_error("EAGLE3 model requires 'extract_layers' in GGUF metadata");
|
||||
}
|
||||
std::copy_n(extract_layers_tmp.begin(), 3, hparams.eagle3_extract_layers.begin());
|
||||
LLAMA_LOG_INFO("%s: EAGLE3 extract_layers = [%d, %d, %d]\n", __func__,
|
||||
hparams.eagle3_extract_layers[0],
|
||||
hparams.eagle3_extract_layers[1],
|
||||
hparams.eagle3_extract_layers[2]);
|
||||
|
||||
// EAGLE3 target model hidden size
|
||||
ml.get_key(LLM_KV_EAGLE3_TARGET_HIDDEN_SIZE, hparams.n_embd_inp_impl);
|
||||
LLAMA_LOG_INFO("%s: EAGLE3 target_hidden_size = %u (draft n_embd = %u)\n", __func__,
|
||||
hparams.n_embd_inp_impl, hparams.n_embd);
|
||||
|
||||
// EAGLE3 norm_before_residual (optional, default false)
|
||||
// compatible with Readhat eagle3 speculator model
|
||||
ml.get_key(LLM_KV_EAGLE3_NORM_BEFORE_RESIDUAL, hparams.eagle3_norm_before_residual, false);
|
||||
if (hparams.eagle3_norm_before_residual) {
|
||||
LLAMA_LOG_INFO("%s: EAGLE3 norm_before_residual = true\n", __func__);
|
||||
}
|
||||
|
||||
type = LLM_TYPE_UNKNOWN;
|
||||
}
|
||||
|
||||
void llama_model_eagle3::load_arch_tensors(llama_model_loader & /*ml*/) {
|
||||
LLAMA_LOAD_LOCALS;
|
||||
|
||||
const int64_t n_embd_inp = hparams.n_embd_inp();
|
||||
const int64_t n_embd_attn_input = 2 * n_embd;
|
||||
|
||||
// Get vocab size from the d2t tensor in the GGUF file (optional - only needed if EAGLE3 has different vocab_size than target)
|
||||
// d2t: draft to target vocabulary mapping
|
||||
int64_t n_draft_vocab = n_vocab; // Default: same as target vocab
|
||||
const struct ggml_tensor * d2t_meta = ml->get_tensor_meta("d2t");
|
||||
if (d2t_meta) {
|
||||
n_draft_vocab = d2t_meta->ne[0]; // update draft vocab size
|
||||
d2t = create_tensor(tn(LLM_TENSOR_EAGLE3_D2T), {n_draft_vocab}, 0);
|
||||
LLAMA_LOG_INFO("%s: EAGLE3 using d2t mapping (draft_vocab_size = %lld)\n", __func__, (long long)n_draft_vocab);
|
||||
} else {
|
||||
d2t = nullptr; // no d2t, use default vocab size
|
||||
LLAMA_LOG_INFO("%s: EAGLE3 without d2t - sharing same vocab_size with target (vocab_size = %lld)\n", __func__, (long long)n_draft_vocab);
|
||||
}
|
||||
|
||||
// Feature fusion layer: projects 3 target layers to draft hidden size
|
||||
fc = create_tensor(tn(LLM_TENSOR_EAGLE3_FC, "weight"), {n_embd_inp, n_embd}, 0);
|
||||
|
||||
// Output layer (uses draft vocab size)
|
||||
output_norm = create_tensor(tn(LLM_TENSOR_OUTPUT_NORM, "weight"), {n_embd}, 0);
|
||||
output = create_tensor(tn(LLM_TENSOR_OUTPUT, "weight"), {n_embd, n_draft_vocab}, 0);
|
||||
|
||||
// Token embeddings (optional - Llama 3.3 70B EAGLE3 has its own)
|
||||
const struct ggml_tensor * tok_embd_meta = ml->get_tensor_meta(tn(LLM_TENSOR_TOKEN_EMBD, "weight").str().c_str());
|
||||
if (tok_embd_meta) {
|
||||
const int64_t n_target_vocab = tok_embd_meta->ne[1];
|
||||
tok_embd = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_target_vocab}, 0);
|
||||
LLAMA_LOG_INFO("%s: EAGLE3 using its own token_embd (vocab = %lld)\n", __func__, (long long)n_target_vocab);
|
||||
}
|
||||
|
||||
// Single decoder layer
|
||||
for (int i = 0; i < n_layer; ++i) {
|
||||
auto & layer = layers[i];
|
||||
|
||||
// input_layernorm: applied to token embeddings
|
||||
layer.attn_norm = create_tensor(tn(LLM_TENSOR_ATTN_NORM, "weight", i), {n_embd}, 0);
|
||||
|
||||
// Attention takes input_embeds_normed + fused_target_normed as input
|
||||
layer.wq = create_tensor(tn(LLM_TENSOR_ATTN_Q, "weight", i), {n_embd_attn_input, n_embd_head_k * n_head}, 0);
|
||||
layer.wk = create_tensor(tn(LLM_TENSOR_ATTN_K, "weight", i), {n_embd_attn_input, n_embd_k_gqa}, 0);
|
||||
layer.wv = create_tensor(tn(LLM_TENSOR_ATTN_V, "weight", i), {n_embd_attn_input, n_embd_v_gqa}, 0);
|
||||
layer.wo = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "weight", i), {n_embd_head_k * n_head, n_embd}, 0);
|
||||
|
||||
// EAGLE-3 specific: hidden_norm applied to fused target features
|
||||
layer.eagle3_hidden_norm = create_tensor(tn(LLM_TENSOR_EAGLE3_HIDDEN_NORM, "weight", i), {n_embd}, 0);
|
||||
|
||||
layer.ffn_norm = create_tensor(tn(LLM_TENSOR_FFN_NORM, "weight", i), {n_embd}, 0);
|
||||
layer.ffn_gate = create_tensor(tn(LLM_TENSOR_FFN_GATE, "weight", i), {n_embd, n_ff}, 0);
|
||||
layer.ffn_down = create_tensor(tn(LLM_TENSOR_FFN_DOWN, "weight", i), { n_ff, n_embd}, 0);
|
||||
layer.ffn_up = create_tensor(tn(LLM_TENSOR_FFN_UP, "weight", i), {n_embd, n_ff}, 0);
|
||||
|
||||
// rope_freqs for llama3 rope scaling (optional - only if EAGLE3 config has rope_scaling)
|
||||
layer.rope_freqs = create_tensor(tn(LLM_TENSOR_ROPE_FREQS, "weight", i), {n_rot/2}, TENSOR_NOT_REQUIRED);
|
||||
}
|
||||
}
|
||||
|
||||
std::unique_ptr<llm_graph_context> llama_model_eagle3::build_arch_graph(const llm_graph_params & params) const {
|
||||
switch (params.gtype) {
|
||||
case LLM_GRAPH_TYPE_ENCODER:
|
||||
return std::make_unique<graph<true>>(*this, params);
|
||||
case LLM_GRAPH_TYPE_DEFAULT:
|
||||
case LLM_GRAPH_TYPE_DECODER:
|
||||
return std::make_unique<graph<false>>(*this, params);
|
||||
default:
|
||||
GGML_ABORT("invalid graph type");
|
||||
};
|
||||
}
|
||||
|
||||
template <>
|
||||
ggml_tensor * llama_model_eagle3::graph<true>::build_inp_embd_enc() const {
|
||||
const int64_t n_embd_inp = hparams.n_embd_inp();
|
||||
|
||||
ggml_tensor * cur = nullptr;
|
||||
|
||||
// Input: Target model features (3 layers concatenated: low, mid, high)
|
||||
// Data will be provided via ubatch->embd in encode_eagle3_features()
|
||||
auto inp_target = std::make_unique<llm_graph_input_embd>(n_embd_inp);
|
||||
inp_target->embd = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_embd_inp, n_tokens);
|
||||
ggml_set_input(inp_target->embd);
|
||||
|
||||
cur = inp_target->embd;
|
||||
cb(cur, "inp_embd", -1);
|
||||
|
||||
res->add_input(std::move(inp_target));
|
||||
|
||||
return cur;
|
||||
}
|
||||
|
||||
// EAGLE3 Encoder: processes target model features through feature fusion layer
|
||||
// Input: target_features e.g. [12288, n_tokens] from target model layers low, middle, high
|
||||
// Output: g_embeddings e.g. [4096, n_tokens] stored in context
|
||||
template <>
|
||||
llama_model_eagle3::graph<true>::graph(const llama_model & model, const llm_graph_params & params) : llm_graph_context(params) {
|
||||
ggml_tensor * cur = nullptr;
|
||||
|
||||
cur = build_inp_embd_enc();
|
||||
|
||||
// sanity check
|
||||
GGML_ASSERT(hparams.n_embd_inp() == model.fc->ne[0]);
|
||||
|
||||
// Feature fusion layer
|
||||
cur = build_lora_mm(model.fc, cur);
|
||||
cb(cur, "fc_out", -1);
|
||||
|
||||
// Output: g_embeddings e.g. [4096, n_tokens]
|
||||
res->t_embd = cur;
|
||||
|
||||
ggml_build_forward_expand(gf, cur);
|
||||
}
|
||||
|
||||
// EAGLE3 Decoder: processes draft tokens using g_embeddings from encoder
|
||||
// Input: draft tokens + g_embeddings from encoder
|
||||
// Output: draft logits
|
||||
template <>
|
||||
llama_model_eagle3::graph<false>::graph(const llama_model & model, const llm_graph_params & params) : llm_graph_context(params) {
|
||||
const int64_t n_embd_head = hparams.n_embd_head_v();
|
||||
|
||||
GGML_ASSERT(n_embd_head == hparams.n_embd_head_k());
|
||||
GGML_ASSERT(n_layer == 1); // EAGLE-3 has only one decoder layer
|
||||
|
||||
ggml_tensor * cur;
|
||||
ggml_tensor * inpL;
|
||||
|
||||
// EAGLE3 Decoder receives:
|
||||
// 1. Token embeddings (e.g.from EAGLE3's own tok_embd for Llama 3.3 70B, or target model for Llama 3.1 8B)
|
||||
// 2. g_embeddings from encoder
|
||||
// Choose token_embd_eagle3: prefer EAGLE3's own if available (Llama 3.3 70B), else use target's (Llama 3.1 8B)
|
||||
ggml_tensor * token_embd_eagle3 = (model.tok_embd != nullptr) ? model.tok_embd : model.target_tok_embd;
|
||||
GGML_ASSERT(token_embd_eagle3 != nullptr && "EAGLE3 decoder requires token embeddings (own or from target model)");
|
||||
ggml_tensor * inp_embd = build_inp_embd(token_embd_eagle3);
|
||||
cb(inp_embd, "inp_embd", -1);
|
||||
|
||||
// TODO: refactor into llm_graph_input
|
||||
ggml_tensor * inp_g = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_embd, n_tokens);
|
||||
ggml_set_input(inp_g);
|
||||
cb(inp_g, "inp_g_embeddings", -1); // TODO: do not change the name! refactor into llm_graph_input
|
||||
|
||||
inpL = inp_g;
|
||||
|
||||
// inp_pos - contains the positions
|
||||
ggml_tensor * inp_pos = build_inp_pos();
|
||||
|
||||
auto * inp_attn = build_attn_inp_kv();
|
||||
|
||||
const float kq_scale = 1.0f/sqrtf(float(n_embd_head));
|
||||
|
||||
ggml_tensor * inp_out_ids = build_inp_out_ids();
|
||||
|
||||
// Single decoder layer (il = 0)
|
||||
const int il = 0;
|
||||
{
|
||||
// Apply input_layernorm to the token embeddings
|
||||
ggml_tensor * embd_norm = build_norm(inp_embd,
|
||||
model.layers[il].attn_norm, NULL,
|
||||
LLM_NORM_RMS, il);
|
||||
cb(embd_norm, "embd_norm", il);
|
||||
|
||||
// Apply hidden_norm to inp_g
|
||||
ggml_tensor * g_norm = build_norm(inp_g,
|
||||
model.layers[il].eagle3_hidden_norm, NULL,
|
||||
LLM_NORM_RMS, -1);
|
||||
cb(g_norm, "g_norm", il);
|
||||
|
||||
// norm_before_residual: determines what goes into the residual connection (compatible with Readhat eagle3 speculator model)
|
||||
// - false (default): use raw inp_g for residual
|
||||
// - true: use normalized g_norm for residual
|
||||
// inpL is the concatenated input (normalized inp_embd + normalized inp_g)
|
||||
ggml_tensor * inpSA = hparams.eagle3_norm_before_residual ? g_norm : inpL;
|
||||
|
||||
// Concatenate normalized inp_embd and normalized inp_g
|
||||
cur = ggml_concat(ctx0, embd_norm, g_norm, il);
|
||||
cb(cur, "concat_embd", il);
|
||||
|
||||
// Self-attention with concatenated input
|
||||
ggml_tensor * Qcur = build_lora_mm(model.layers[il].wq, cur);
|
||||
cb(Qcur, "Qcur", il);
|
||||
|
||||
ggml_tensor * Kcur = build_lora_mm(model.layers[il].wk, cur);
|
||||
cb(Kcur, "Kcur", il);
|
||||
|
||||
ggml_tensor * Vcur = build_lora_mm(model.layers[il].wv, cur);
|
||||
cb(Vcur, "Vcur", il);
|
||||
|
||||
Qcur = ggml_reshape_3d(ctx0, Qcur, n_embd_head, n_head, n_tokens);
|
||||
Kcur = ggml_reshape_3d(ctx0, Kcur, n_embd_head, n_head_kv, n_tokens);
|
||||
Vcur = ggml_reshape_3d(ctx0, Vcur, n_embd_head, n_head_kv, n_tokens);
|
||||
|
||||
// rope freq factors, returns nullptr if not available
|
||||
ggml_tensor * rope_factors = model.get_rope_factors(cparams, il);
|
||||
|
||||
// RoPE
|
||||
Qcur = ggml_rope_ext(
|
||||
ctx0, Qcur, inp_pos, rope_factors,
|
||||
n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,
|
||||
ext_factor, attn_factor, beta_fast, beta_slow
|
||||
);
|
||||
Kcur = ggml_rope_ext(
|
||||
ctx0, Kcur, inp_pos, rope_factors,
|
||||
n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,
|
||||
ext_factor, attn_factor, beta_fast, beta_slow
|
||||
);
|
||||
|
||||
cb(Qcur, "Qcur_rope", il);
|
||||
cb(Kcur, "Kcur_rope", il);
|
||||
|
||||
cur = build_attn(inp_attn,
|
||||
model.layers[il].wo, NULL, nullptr,
|
||||
Qcur, Kcur, Vcur, nullptr, nullptr, nullptr, kq_scale, il);
|
||||
|
||||
if (inp_out_ids) {
|
||||
cur = ggml_get_rows(ctx0, cur, inp_out_ids);
|
||||
inpSA = ggml_get_rows(ctx0, inpSA, inp_out_ids);
|
||||
}
|
||||
|
||||
// Add residual and update it
|
||||
ggml_tensor * ffn_inp = ggml_add(ctx0, cur, inpSA);
|
||||
cb(ffn_inp, "ffn_inp", il);
|
||||
|
||||
// Apply FFN norm to the sum
|
||||
cur = build_norm(ffn_inp,
|
||||
model.layers[il].ffn_norm, NULL,
|
||||
LLM_NORM_RMS, il);
|
||||
cb(cur, "post_attn_norm", il);
|
||||
|
||||
cur = build_ffn(cur,
|
||||
model.layers[il].ffn_up, NULL, NULL,
|
||||
model.layers[il].ffn_gate, NULL, NULL,
|
||||
model.layers[il].ffn_down, NULL, NULL,
|
||||
NULL,
|
||||
LLM_FFN_SILU, LLM_FFN_PAR, il);
|
||||
cb(cur, "ffn_out", il);
|
||||
|
||||
// Output norm with residual
|
||||
cur = ggml_add(ctx0, cur, ffn_inp);
|
||||
cb(cur, "eagle3_prenorm", il);
|
||||
|
||||
inpL = cur;
|
||||
}
|
||||
|
||||
cur = inpL;
|
||||
|
||||
// Output prenorm state (for next token's g_embeddings in autoregressive generation)
|
||||
ggml_set_output(cur);
|
||||
res->t_embd = cur;
|
||||
|
||||
cur = build_norm(cur,
|
||||
model.output_norm, NULL,
|
||||
LLM_NORM_RMS, -1);
|
||||
cb(cur, "result_norm", -1);
|
||||
|
||||
// lm_head - projects to draft vocabulary
|
||||
cur = build_lora_mm(model.output, cur);
|
||||
|
||||
cb(cur, "result_output", -1);
|
||||
res->t_logits = cur;
|
||||
|
||||
ggml_build_forward_expand(gf, cur);
|
||||
}
|
||||
@@ -123,6 +123,8 @@ llama_model_llama::graph<embed>::graph(const llama_model & model, const llm_grap
|
||||
ggml_tensor * inp_out_ids = build_inp_out_ids();
|
||||
|
||||
for (int il = 0; il < n_layer; ++il) {
|
||||
res->t_layer_inp[il] = inpL;
|
||||
|
||||
ggml_tensor * inpSA = inpL;
|
||||
|
||||
// norm
|
||||
|
||||
@@ -1784,6 +1784,21 @@ struct llama_model_qwen35moe : public llama_model_base {
|
||||
std::unique_ptr<llm_graph_context> build_arch_graph(const llm_graph_params & params) const override;
|
||||
};
|
||||
|
||||
struct llama_model_eagle3 : public llama_model_base {
|
||||
llama_model_eagle3(const struct llama_model_params & params) : llama_model_base(params) {}
|
||||
void load_arch_hparams(llama_model_loader & ml) override;
|
||||
void load_arch_tensors(llama_model_loader & ml) override;
|
||||
|
||||
template <bool is_enc>
|
||||
struct graph : public llm_graph_context {
|
||||
graph(const llama_model & model, const llm_graph_params & params);
|
||||
|
||||
ggml_tensor * build_inp_embd_enc() const;
|
||||
};
|
||||
|
||||
std::unique_ptr<llm_graph_context> build_arch_graph(const llm_graph_params & params) const override;
|
||||
};
|
||||
|
||||
|
||||
struct llama_model_mistral3 : public llama_model_base {
|
||||
llama_model_mistral3(const struct llama_model_params & params) : llama_model_base(params) {}
|
||||
|
||||
@@ -75,6 +75,8 @@ llama_model_openai_moe::graph::graph(const llama_model & model, const llm_graph_
|
||||
ggml_tensor * inp_out_ids = build_inp_out_ids();
|
||||
|
||||
for (int il = 0; il < n_layer; ++il) {
|
||||
res->t_layer_inp[il] = inpL;
|
||||
|
||||
const float freq_base_l = model.get_rope_freq_base (cparams, il);
|
||||
const float freq_scale_l = model.get_rope_freq_scale(cparams, il);
|
||||
|
||||
|
||||
@@ -68,6 +68,8 @@ llama_model_qwen3::graph::graph(const llama_model & model, const llm_graph_param
|
||||
ggml_tensor * inp_out_ids = build_inp_out_ids();
|
||||
|
||||
for (int il = 0; il < n_layer; ++il) {
|
||||
res->t_layer_inp[il] = inpL;
|
||||
|
||||
ggml_tensor * inpSA = inpL;
|
||||
|
||||
// norm
|
||||
|
||||
@@ -78,6 +78,8 @@ llama_model_qwen3moe::graph::graph(const llama_model & model, const llm_graph_pa
|
||||
ggml_tensor * inp_out_ids = build_inp_out_ids();
|
||||
|
||||
for (int il = 0; il < n_layer; ++il) {
|
||||
res->t_layer_inp[il] = inpL;
|
||||
|
||||
ggml_tensor * inpSA = inpL;
|
||||
|
||||
// norm
|
||||
|
||||
@@ -805,8 +805,20 @@ private:
|
||||
return false;
|
||||
}
|
||||
|
||||
params_base.speculative.draft.model = model_dft.get();
|
||||
params_base.speculative.draft.model = model_dft.get();
|
||||
params_base.speculative.draft.model_tgt = model;
|
||||
|
||||
params_base.speculative.draft.cparams = common_context_params_to_llama(params_dft);
|
||||
|
||||
if (params_base.speculative.draft.eagle3) {
|
||||
// EAGLE3 current limitation: extracted target features are per-context; multiple slots would overwrite each other
|
||||
if (params_base.n_parallel > 1) {
|
||||
SRV_ERR("%s", "EAGLE3 speculative decoding is not supported with n_parallel > 1\n");
|
||||
return false;
|
||||
}
|
||||
llama_set_eagle3(ctx, model_dft.get());
|
||||
SRV_INF("%s", "EAGLE3 feature extraction enabled on target model\n");
|
||||
}
|
||||
}
|
||||
|
||||
std::string & mmproj_path = params_base.mmproj.path;
|
||||
|
||||
Reference in New Issue
Block a user