mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-09-27 21:46:57 +02:00
llama-context : report graph inputs and input tensors during sched reserve (#26625)
* llama-context : report graph inputs and input tensors during sched reserve - fix the tg (token generation) graph bs label to use n_seqs instead of a hardcoded 1 - report the number of graph inputs from llm_graph_result::inputs for both the pp and tg graphs - report the number of input tensors (nodes and their src tensors flagged with GGML_TENSOR_FLAG_INPUT) - log a warning when an input tensor has an op other than GGML_OP_NONE - log a trace line for each input tensor and the nodes (name and op) that use it Assisted-by: llama.cpp:DeepSeek-v4-Flash-0731 * cont : count input tensors before reserving the sched * wip * llama-graph : name the unnamed graph input tensors - name the kv-cache idxs input tensors (attn_inp_k_idxs, attn_inp_v_idxs) - name the recurrent state copy idxs input tensor (rs_s_copy) - report the input tensor shape in the sched_reserve trace Assisted-by: pi:llama.cpp/Qwen3.8-27B * llama-context : rename "graph inputs" to "graph input objects" Assisted-by: pi:llama.cpp/Qwen3.8-27B * llama-context : report the sched reserve graph stats on a single line - print nodes, splits, input objects and input tensors in one line - when the pp and tg graphs differ, print each value as 'pp / tg' and annotate the line with the batch sizes used for each graph Assisted-by: pi:llama.cpp/Qwen3.8-27B * cont : pad logs
This commit is contained in:
+10
-1
@@ -2451,6 +2451,7 @@ ggml_tensor * llm_graph_context::build_inp_pos() const {
|
||||
|
||||
cur = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, (int64_t)n_tokens*hparams.n_pos_per_embd());
|
||||
ggml_set_input(cur);
|
||||
cb(cur, "inp_pos", -1);
|
||||
|
||||
res->add_input(std::move(inp));
|
||||
|
||||
@@ -2465,7 +2466,7 @@ ggml_tensor * llm_graph_context::build_inp_attn_scale() const {
|
||||
// this need to be 1x1xN for broadcasting
|
||||
cur = ggml_new_tensor_3d(ctx0, GGML_TYPE_F32, 1, 1, n_tokens);
|
||||
ggml_set_input(cur);
|
||||
ggml_set_name(cur, "attn_scale");
|
||||
cb(cur, "inp_attn_scale", -1);
|
||||
|
||||
res->add_input(std::move(inp));
|
||||
|
||||
@@ -2487,6 +2488,7 @@ ggml_tensor * llm_graph_context::build_inp_out_ids() const {
|
||||
|
||||
cur = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, n_outputs);
|
||||
ggml_set_input(cur);
|
||||
ggml_set_name(cur, "out_ids");
|
||||
|
||||
res->add_input(std::move(inp));
|
||||
|
||||
@@ -2500,6 +2502,7 @@ ggml_tensor * llm_graph_context::build_inp_mean() const {
|
||||
|
||||
cur = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_tokens, ubatch.n_seqs_unq);
|
||||
ggml_set_input(cur);
|
||||
ggml_set_name(cur, "mean");
|
||||
|
||||
res->add_input(std::move(inp));
|
||||
|
||||
@@ -2513,6 +2516,7 @@ ggml_tensor * llm_graph_context::build_inp_cls() const {
|
||||
|
||||
cur = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, ubatch.n_seqs_unq);
|
||||
ggml_set_input(cur);
|
||||
ggml_set_name(cur, "cls");
|
||||
|
||||
res->add_input(std::move(inp));
|
||||
|
||||
@@ -2537,6 +2541,7 @@ ggml_tensor * llm_graph_context::build_inp_cross_embd() const {
|
||||
|
||||
cur = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_embd, n_enc);
|
||||
ggml_set_input(cur);
|
||||
ggml_set_name(cur, "cross_embd");
|
||||
|
||||
res->add_input(std::move(inp));
|
||||
|
||||
@@ -2550,6 +2555,7 @@ ggml_tensor * llm_graph_context::build_inp_pos_bucket_enc() const {
|
||||
|
||||
cur = ggml_new_tensor_2d(ctx0, GGML_TYPE_I32, n_tokens, n_tokens);
|
||||
ggml_set_input(cur);
|
||||
ggml_set_name(cur, "pos_bucket_enc");
|
||||
|
||||
res->add_input(std::move(inp));
|
||||
|
||||
@@ -2567,6 +2573,7 @@ ggml_tensor * llm_graph_context::build_inp_pos_bucket_dec() const {
|
||||
|
||||
cur = ggml_new_tensor_2d(ctx0, GGML_TYPE_I32, n_kv, n_tokens);
|
||||
ggml_set_input(cur);
|
||||
ggml_set_name(cur, "pos_bucket_dec");
|
||||
|
||||
res->add_input(std::move(inp));
|
||||
|
||||
@@ -2735,6 +2742,7 @@ llm_graph_input_attn_no_cache * llm_graph_context::build_attn_inp_no_cache() con
|
||||
// note: there is no KV cache, so the number of KV values is equal to the number of tokens in the batch
|
||||
inp->self_kq_mask = ggml_new_tensor_4d(ctx0, type_mask, n_tokens, n_tokens, 1, 1);
|
||||
ggml_set_input(inp->self_kq_mask);
|
||||
cb(inp->self_kq_mask, "self_kq_mask", -1);
|
||||
|
||||
inp->self_kq_mask_cnv = inp->self_kq_mask;
|
||||
|
||||
@@ -3511,6 +3519,7 @@ static std::unique_ptr<llm_graph_input_rs> build_rs_inp_impl(
|
||||
|
||||
inp->s_copy = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, n_rs);
|
||||
ggml_set_input(inp->s_copy);
|
||||
ggml_set_name(inp->s_copy, "rs_s_copy");
|
||||
|
||||
inp->s_copy_main = ggml_view_1d(ctx0, inp->s_copy, n_seqs, 0);
|
||||
inp->s_copy_extra = ggml_view_1d(ctx0, inp->s_copy, n_rs - n_seqs, n_seqs * inp->s_copy->nb[0]);
|
||||
|
||||
Reference in New Issue
Block a user