Compare commits

...
Author SHA1 Message Date
Sigbjørn Skjæret 1c911f1ef9 fix name 2026-09-23 14:06:16 +02:00
Sigbjørn Skjæret 9adc548950 add gfx1201 jobs 2026-09-23 13:57:11 +02:00
Si Chen 4e416ee730 jinja : parse unary +/- before variables (#29244)
* jinja : parse unary +/- before variables

Lexer already emits unary_operator for -n / +n, and runtime executes
unary -. Parse them at multiplicative precedence so slices like
items[:-n] and GigaChat indent[:-indent_factor] work.

* jinja : keep filters/tests outside unary operands

Unary +/- must bind only the primary/postfix operand so -n|abs is
(-n)|abs, not -(n|abs). Add unary + and filter/test regression coverage.

Signed-off-by: sinksilk <[email protected]>

---------

Signed-off-by: sinksilk <[email protected]>
2026-09-23 13:29:45 +02:00
YiChen Lv ee3ecce05c metal : key the fa-vec tuned table by family instead of SKU (#29075)
* key the fa-vec tuned table by family instead of SKU

* fall back to baseline for untuned fa-vec gpu families
2026-09-23 19:23:15 +08:00
9 changed files with 1210 additions and 3974 deletions
+44 -14
View File
@@ -88,8 +88,8 @@ jobs:
hf_bucket: ggml-org/cache
save: true
gpu-rocm:
runs-on: [self-hosted, Linux, AMD]
gpu-rocm-gfx1151:
runs-on: [self-hosted, Linux, gfx1151]
steps:
- name: Clone
@@ -108,17 +108,47 @@ jobs:
rocminfo
GG_BUILD_ROCM=1 GG_BUILD_AMDGPU_TARGETS=gfx1151 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
# TODO: provision AMD GPU machine
# amd-rocm:
# runs-on: [self-hosted, Linux, AMD]
gpu-rocm-gfx1201:
runs-on: [self-hosted, Linux, gfx1201]
container: "rocm/dev-ubuntu-24.04:7.2.4-complete"
# steps:
# - name: Clone
# id: checkout
# uses: actions/checkout@v6
steps:
- name: Clone
id: checkout
uses: actions/checkout@v6
# - name: Test
# id: ggml-ci
# run: |
# amd-smi static
# GG_BUILD_ROCM=1 GG_BUILD_AMDGPU_TARGETS="gfx1101" bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
- name: Install dependencies
run: |
apt update
apt install -y build-essential git git-lfs jq cmake libssl-dev time unzip wget python3 python3-venv python3-pip
- name: ccache
uses: ggml-org/[email protected]
with:
restore: false
save: false
- name: ccache-buckets-restore
uses: ./.github/actions/ccache-buckets
with:
key: self-hosted-gpu-rocm-gfx1201
folder: llama.cpp
hf_bucket: ggml-org/cache
- name: Test
id: ggml-ci
run: |
rocminfo
GG_BUILD_ROCM=1 GG_BUILD_AMDGPU_TARGETS=gfx1201 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
- name: ccache-buckets-save
if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}
uses: ./.github/actions/ccache-buckets
env:
HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}
with:
key: self-hosted-gpu-rocm-gfx1201
folder: llama.cpp
evict-old-files: 1d
hf_bucket: ggml-org/cache
save: true
+42 -12
View File
@@ -185,17 +185,47 @@ jobs:
# a valid python environment for testing
LLAMA_FATAL_WARNINGS=OFF GG_BUILD_NINJA=1 GG_BUILD_VULKAN=1 GG_BUILD_LOW_PERF=1 ./ci/run.sh ./results/llama.cpp ./mnt/llama.cpp
# TODO: provision AMD GPU machine
# amd-vulkan:
# runs-on: [self-hosted, Linux, AMD]
gpu-vulkan-amd:
runs-on: [self-hosted, Linux, gfx1201]
container: "ubuntu:26.04"
# steps:
# - name: Clone
# id: checkout
# uses: actions/checkout@v6
steps:
- name: Clone
id: checkout
uses: actions/checkout@v6
# - name: Test
# id: ggml-ci
# run: |
# vulkaninfo --summary
# GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
- name: Install dependencies
run: |
apt update
apt install -y build-essential git git-lfs jq cmake libxcb-xinput0 libxcb-xinerama0 libxcb-cursor-dev libvulkan-dev glslc spirv-headers vulkan-tools mesa-vulkan-drivers libglvnd0 libgl1 libglx0 libegl1 libgles2 libssl-dev time unzip wget python3 python3-venv python3-pip
- name: ccache
uses: ggml-org/[email protected]
with:
restore: false
save: false
- name: ccache-buckets-restore
uses: ./.github/actions/ccache-buckets
with:
key: self-hosted-vulkan-amd
folder: llama.cpp
hf_bucket: ggml-org/cache
- name: Test
id: ggml-ci
run: |
vulkaninfo --summary
GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
- name: ccache-buckets-save
if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}
uses: ./.github/actions/ccache-buckets
env:
HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}
with:
key: self-hosted-vulkan-amd
folder: llama.cpp
evict-old-files: 1d
hf_bucket: ggml-org/cache
save: true
+11 -1
View File
@@ -437,7 +437,8 @@ private:
}
statement_ptr parse_filter_expression() {
auto operand = parse_call_member_expression();
// Filters/tests bind outside unary so -n|abs is (-n)|abs, not -(n|abs).
auto operand = parse_unary_expression();
while (is(token::pipe)) {
size_t start_pos = current;
++current; // consume pipe
@@ -448,6 +449,15 @@ private:
return operand;
}
statement_ptr parse_unary_expression() {
if (is(token::unary_operator)) {
size_t start_pos = current;
auto op = next();
return mk_stmt<unary_expression>(start_pos, op, parse_unary_expression());
}
return parse_call_member_expression();
}
statement_ptr parse_call_member_expression() {
// Handle member expressions recursively
auto member = parse_member_expression(parse_primary_expression());
+5
View File
@@ -450,6 +450,11 @@ value unary_expression::execute_impl(context & ctx) const {
} else {
throw std::runtime_error("Unary - operator requires numeric operand");
}
} else if (op.value == "+") {
if (is_val<value_int>(operand_val) || is_val<value_float>(operand_val)) {
return operand_val;
}
throw std::runtime_error("Unary + operator requires numeric operand");
}
throw std::runtime_error("Unknown unary operator '" + op.value + "'");
-1
View File
@@ -3584,7 +3584,6 @@ int ggml_metal_op_flash_attn_ext(ggml_metal_op_t ctx, int idx) {
auto cfg = use_sparse
? ggml_metal_tuning::fa_vec_baseline_cfg((int) ne00, (int) ne20)
: ggml_metal_tuning::fa_vec_pick(
props_dev->device_id,
props_dev->gpu_family,
(int) op->src[1]->type,
(int) ne00, (int) ne20, // dk, dv (ne00 == dk for FA)
File diff suppressed because it is too large Load Diff
+3 -5
View File
@@ -1,6 +1,5 @@
#pragma once
#include "ggml-metal-device.h" // enum ggml_metal_device_id
#include "ggml.h"
#include <cstdint>
@@ -32,7 +31,7 @@ constexpr int8_t FA_VEC_DOMAIN_DECODE = 0; // ne01 == 1
constexpr int8_t FA_VEC_DOMAIN_BATCH = 1; // ne01 >= 2
struct fa_vec_key_t {
int8_t device_id;
int8_t family;
int8_t dtype;
int16_t dk;
int16_t dv;
@@ -70,8 +69,7 @@ void fa_vec_set_override(fa_vec_cfg_t cfg);
void fa_vec_clear_override();
fa_vec_cfg_t fa_vec_baseline_cfg(int dk, int dv);
// device_id selects a per-SKU row; on a miss, gpu_family (0 if unknown) maps to a representative
// SKU and the table is retried. No match -> baseline.
fa_vec_cfg_t fa_vec_pick(enum ggml_metal_device_id device_id, int gpu_family, int dtype, int dk, int dv, int64_t ne11, int64_t ne01);
// Keyed by Apple GPU family; an untuned family matches no row and gets the baseline.
fa_vec_cfg_t fa_vec_pick(int gpu_family, int dtype, int dk, int dv, int64_t ne11, int64_t ne01);
} // namespace ggml_metal_tuning
+43
View File
@@ -458,6 +458,49 @@ static void test_expressions(testing & t) {
"['b']"
);
test_template(t, "array slice negative variable",
"{{ items[:-n]|string }}",
{{"items", json::array({"a", "b", "c"})}, {"n", 1}},
"['a', 'b']"
);
test_template(t, "array slice negative variable indent",
"{{ indent[:-indent_factor] }}",
{{"indent", " "}, {"indent_factor", 2}},
" "
);
test_template(t, "unary minus variable",
"{{ -n }}",
{{"n", 3}},
"-3"
);
test_template(t, "unary plus variable",
"{{ +n }}",
{{"n", -3}},
"-3"
);
test_template(t, "unary plus float",
"{{ +x }}",
{{"x", -1.5}},
"-1.5"
);
// Unary binds tighter than filter: -n|abs == (-n)|abs, not -(n|abs)
test_template(t, "unary minus then abs filter",
"{{ -n|abs }}",
{{"n", -3}},
"3"
);
test_template(t, "unary minus then number test",
"{{ -n is number }}",
{{"n", 3}},
"True"
);
test_template(t, "array slice step",
"{{ items[::2]|string }}",
{{"items", json::array({"a", "b", "c"})}},
+5 -2
View File
@@ -26,10 +26,13 @@ Sweep the grid (6 dtypes x 10 head sizes x 4 KV depths x 9 batch widths; a few h
./build/bin/ggml-metal-tuning fa-vec > fa_vec_rows.txt 2> fa_vec_sweep.log
```
`fa_vec_rows.txt` holds nothing but table rows, ready to paste into `fa_vec_tuned_table`: the min-max-regret target, the aggregate benefit gate, the short-KV drop and the pointwise compression are already applied.
`fa_vec_rows.txt` holds nothing but table rows: the min-max-regret target, the aggregate benefit gate, the short-KV drop and the pointwise compression are already applied.
The rows carry the SKU token the runtime reported, but `fa_vec_tuned_table` is keyed by Apple GPU family, so that column has to be retagged before the rows compile.
Your family number is on the `MTLGPUFamilyApple<N>` line the backend logs at init, near the top of `fa_vec_sweep.log`; `N` is the value, and the `MTLGPUFamilyCommon`/`MTLGPUFamilyMetal` lines beside it are not it.
If your log is the only sweep for that family, its rows become the family's segment; where the family already has rows, post the log and let the two be compared before anything is replaced.
A config represents a bucket only if it is no slower than the baseline config at every point that bucket covers, so a config that wins on average but loses at one batch width leaves its bucket at baseline.
`fa_vec_sweep.log` holds the per-cell timings, bucket coverage, noise floor, any cooldown activity, and every config the no-harm rule refused together with the point that refused it.
Post both: the log is what makes the rows reviewable.
Post both, always: the rows now speak for every device in the family, so the log is what makes them reviewable.
Long sweeps can be split.
`--dtype f16,q4_0` and `--dk 128,192` restrict the grid, and the emitted rows for one `(dtype, head size)` do not depend on the others.