llama-context : report graph inputs and input tensors during sched reserve (#26625)

* llama-context : report graph inputs and input tensors during sched reserve

- fix the tg (token generation) graph bs label to use n_seqs instead of a hardcoded 1
- report the number of graph inputs from llm_graph_result::inputs for both the pp and tg graphs
- report the number of input tensors (nodes and their src tensors flagged with GGML_TENSOR_FLAG_INPUT)
- log a warning when an input tensor has an op other than GGML_OP_NONE
- log a trace line for each input tensor and the nodes (name and op) that use it

Assisted-by: llama.cpp:DeepSeek-v4-Flash-0731

* cont : count input tensors before reserving the sched

* wip

* llama-graph : name the unnamed graph input tensors

- name the kv-cache idxs input tensors (attn_inp_k_idxs, attn_inp_v_idxs)
- name the recurrent state copy idxs input tensor (rs_s_copy)
- report the input tensor shape in the sched_reserve trace

Assisted-by: pi:llama.cpp/Qwen3.8-27B

* llama-context : rename "graph inputs" to "graph input objects"

Assisted-by: pi:llama.cpp/Qwen3.8-27B

* llama-context : report the sched reserve graph stats on a single line

- print nodes, splits, input objects and input tensors in one line
- when the pp and tg graphs differ, print each value as 'pp / tg'
  and annotate the line with the batch sizes used for each graph

Assisted-by: pi:llama.cpp/Qwen3.8-27B

* cont : pad logs
This commit is contained in:
Georgi Gerganov
2026-09-21 19:13:04 +03:00
committed by GitHub
parent b1c2863e2c
commit 9655061365
4 changed files with 79 additions and 18 deletions
+66 -17
View File
@@ -19,6 +19,7 @@
#include <limits>
#include <stdexcept>
#include <string>
#include <unordered_map>
//
// llama_context
@@ -579,6 +580,40 @@ void llama_context::resolve_fused_ops(const llama_memory_context_i * mctx, uint3
}
}
static int llama_graph_n_input_tensors(ggml_cgraph * gf) {
std::unordered_map<const ggml_tensor *, std::vector<ggml_tensor *>> users;
for (int i = 0; i < ggml_graph_n_nodes(gf); ++i) {
ggml_tensor * node = ggml_graph_node(gf, i);
if (node->flags & GGML_TENSOR_FLAG_INPUT) {
users[node].push_back(node);
}
for (int j = 0; j < GGML_MAX_SRC; ++j) {
ggml_tensor * src = node->src[j];
if (!src) {
break;
}
if (src->flags & GGML_TENSOR_FLAG_INPUT) {
users[src].push_back(node);
}
}
}
for (const auto & [tensor, nodes] : users) {
if (tensor->op != GGML_OP_NONE) {
LLAMA_LOG_WARN("%s: input tensor '%32s' has op %s, expected GGML_OP_NONE\n",
__func__, tensor->name, ggml_op_name(tensor->op));
}
for (const ggml_tensor * node : nodes) {
LLAMA_LOG_DEBUG("%s: input tensor '%32s' [%s, ne = { %5" PRId64 ", %5" PRId64 ", %5" PRId64 ", %5" PRId64 " }] is used by node '%s' (%s)\n",
__func__, tensor->name, ggml_type_name(tensor->type),
tensor->ne[0], tensor->ne[1], tensor->ne[2], tensor->ne[3],
node->name, ggml_op_name(node->op));
}
}
return (int) users.size();
}
void llama_context::sched_reserve() {
if (!sched_need_reserve) {
return;
@@ -624,11 +659,15 @@ void llama_context::sched_reserve() {
resolve_fused_ops(mctx.get(), n_seqs);
// reserve worst-case graph
int n_splits_pp = -1;
int n_nodes_pp = -1;
int n_splits_pp = -1;
int n_nodes_pp = -1;
int n_inputs_pp = -1;
int n_input_tensors_pp = -1;
int n_splits_tg = -1;
int n_nodes_tg = -1;
int n_splits_tg = -1;
int n_nodes_tg = -1;
int n_inputs_tg = -1;
int n_input_tensors_tg = -1;
const uint32_t n_outputs_pp = std::min(n_tokens, cparams.n_outputs_max);
@@ -648,8 +687,10 @@ void llama_context::sched_reserve() {
}
}
n_splits_pp = ggml_backend_sched_get_n_splits(sched.get());
n_nodes_pp = ggml_graph_n_nodes(gf);
n_splits_pp = ggml_backend_sched_get_n_splits(sched.get());
n_nodes_pp = ggml_graph_n_nodes(gf);
n_inputs_pp = get_gf_res_reserve()->inputs.size();
n_input_tensors_pp = this->n_input_tensors;
}
// reserve with tg (token generation) graph to get the number of splits and nodes
@@ -659,8 +700,10 @@ void llama_context::sched_reserve() {
throw std::runtime_error("failed to allocate compute tg buffers");
}
n_splits_tg = ggml_backend_sched_get_n_splits(sched.get());
n_nodes_tg = ggml_graph_n_nodes(gf);
n_splits_tg = ggml_backend_sched_get_n_splits(sched.get());
n_nodes_tg = ggml_graph_n_nodes(gf);
n_inputs_tg = get_gf_res_reserve()->inputs.size();
n_input_tensors_tg = this->n_input_tensors;
}
// reserve again with pp graph to avoid ggml-alloc reallocations during inference
@@ -698,16 +741,21 @@ void llama_context::sched_reserve() {
}
}
if (n_nodes_pp == n_nodes_tg) {
LLAMA_LOG_INFO("%s: graph nodes = %d\n", __func__, n_nodes_pp);
} else {
LLAMA_LOG_INFO("%s: graph nodes = %d (with bs=%d), %d (with bs=1)\n", __func__, n_nodes_pp, n_tokens, n_nodes_tg);
}
{
const bool diff = n_nodes_pp != n_nodes_tg || n_splits_pp != n_splits_tg ||
n_inputs_pp != n_inputs_tg || n_input_tensors_pp != n_input_tensors_tg;
if (n_splits_pp == n_splits_tg) {
LLAMA_LOG_INFO("%s: graph splits = %d\n", __func__, n_splits_pp);
} else {
LLAMA_LOG_INFO("%s: graph splits = %d (with bs=%d), %d (with bs=1)\n", __func__, n_splits_pp, n_tokens, n_splits_tg);
const auto val = [diff](int v_pp, int v_tg) -> std::string {
return diff ? format("%d / %d", v_pp, v_tg) : format("%d", v_pp);
};
LLAMA_LOG_INFO("%s: graph%s: nodes = %s, splits = %s, input objects = %s, input tensors = %s\n",
__func__,
diff ? format(" (pp bs=%d, tg bs=%d)", n_tokens, n_seqs).c_str() : "",
val(n_nodes_pp, n_nodes_tg).c_str(),
val(n_splits_pp, n_splits_tg).c_str(),
val(n_inputs_pp, n_inputs_tg).c_str(),
val(n_input_tensors_pp, n_input_tensors_tg).c_str());
}
const int64_t t_end_us = ggml_time_us();
@@ -2475,6 +2523,7 @@ ggml_cgraph * llama_context::graph_reserve(
auto * gf = model.build_graph(gparams);
this->n_input_tensors = llama_graph_n_input_tensors(gf);
this->n_outputs = save_n_outputs;
// initialize scheduler with the specified graph