mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-09-27 21:46:57 +02:00
sampling : stop short if backend sampler sampled a token
This commit modifies the graph building logic to immediately continue when a token has already been sampled by the backend sampler. It also updates the test for backend temporary sampling to include top-k and distribution samplers in the chain to verify that they are not producing any logits (they are not run).
This commit is contained in:
@@ -2100,6 +2100,7 @@ void llm_graph_context::build_sampling() const {
|
||||
if (data.sampled != nullptr) {
|
||||
res->t_sampled[seq_id] = data.sampled;
|
||||
ggml_build_forward_expand(gf, data.sampled);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (data.probs != nullptr) {
|
||||
|
||||
Reference in New Issue
Block a user