mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-09-27 21:46:57 +02:00
metal : gate mul_mm_id src1 rescale behind ggml_prec (#29029)
* metal : gate mul_mm_id src1 rescale behind ggml_prec Assisted-by: Claude Fable 5.1 * ggml-webgpu: reject MUL_MAT_ID when src1 precision is F32 * cuda/vulkan: reject MUL_MAT_ID in supports_op when src1 prec is F32 fix `supports_op` to return false for failing backends when the specified src1 precision is f32 Assisted-by: Claude Fable 5.1 --------- Co-authored-by: yomaytk <[email protected]>
This commit is contained in:
co-authored by
yomaytk
parent
f95b0d9539
commit
0f8a414b75
@@ -2302,6 +2302,10 @@ ggml_tensor * llm_graph_context::build_moe_ffn(
|
||||
}
|
||||
|
||||
experts = build_lora_mm_id(down_exps, cur, selected_experts, down_exps_s); // [n_embd, n_expert_used, n_tokens]
|
||||
if (arch == LLM_ARCH_MISTRAL4) {
|
||||
// src1 can exceed F16 range
|
||||
ggml_prec_set_src(experts, GGML_PREC_F32, 1);
|
||||
}
|
||||
cb(experts, "ffn_moe_down", il);
|
||||
|
||||
if (down_exps_s) {
|
||||
|
||||
Reference in New Issue
Block a user