Compare commits

...
Author SHA1 Message Date
Aaron Teo 85056e094a vendor: attempt to ignore warnings from vendored files
Signed-off-by: Aaron Teo <[email protected]>
2026-09-28 03:22:59 +08:00
Aaron Teo cda4023517 ggml-zdnn: fix compiler errors
Signed-off-by: Aaron Teo <[email protected]>
2026-09-28 02:57:13 +08:00
Aaron Teo 1c87213937 ci: clean up comments
Signed-off-by: Aaron Teo <[email protected]>
2026-09-28 02:56:50 +08:00
Aaron Teo 69022c2af8 ci: set shell to bash
Signed-off-by: Aaron Teo <[email protected]>
2026-09-28 02:39:23 +08:00
Aaron Teo 1c7986a776 ci: attempt to run a ubuntu 26.04 container
Signed-off-by: Aaron Teo <[email protected]>
2026-09-28 02:36:07 +08:00
Aaron Teo 6b5a83d97b ci: add zdnn backend build but not test
Signed-off-by: Aaron Teo <[email protected]>
2026-09-28 02:24:18 +08:00
Georgi Gerganov a97cce86a8 common : avoid side effects around params parsing (#29537)
- register --rpc unconditionally and call llama_supports_rpc() only from its handler
- print server "initialization ..." log after args are parsed

Assisted-by: pi:llama.cpp/MiMo-V2.6-Flash-RL
2026-09-27 20:18:56 +03:00
6 changed files with 64 additions and 19 deletions
+40
View File
@@ -100,6 +100,46 @@ jobs:
wget https://huggingface.co/ggml-org/models/resolve/main/tinyllamas/stories260K-be.gguf
./bin/llama-completion -m stories260K-be.gguf -p "One day, Lily met a Shoggoth" -n 500 -c 256
ubuntu-26-zdnn-s390x:
name: ubuntu-26-zdnn-s390x
runs-on: ubuntu-24.04-s390x
container: ubuntu:26.04 # required to get GCC 15.1 and binutils 2.44
defaults:
run:
shell: bash
steps:
- name: Build Dependencies
id: build_depends
run: |
apt-get update
apt-get install -y --no-install-recommends \
build-essential cmake git ca-certificates \
libssl-dev libzdnn-dev
- name: Clone
id: checkout
uses: actions/checkout@v6
- name: Toolchain workaround (GCC 15)
run: |
apt-get install -y gcc-15 g++-15
echo "CC=gcc-15" >> "$GITHUB_ENV"
echo "CXX=g++-15" >> "$GITHUB_ENV"
- name: Build with zDNN Backend
id: cmake_build
run: |
cmake -B build \
-DLLAMA_FATAL_WARNINGS=ON \
-DGGML_NATIVE=OFF \
-DGGML_VXE=ON \
-DGGML_ZDNN=ON \
-DGGML_RPC=ON \
-DCMAKE_C_FLAGS="-march=arch15" \
-DCMAKE_CXX_FLAGS="-march=arch15"
time cmake --build build --config Release -j $(nproc)
ubuntu-24-ppc64le:
runs-on: ubuntu-24.04-ppc64le
+10 -9
View File
@@ -2673,16 +2673,17 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
params.video_ffmpeg_bin_dir = value;
}
).set_examples(mmproj_examples).set_env("LLAMA_ARG_VIDEO_FFMPEG_DIR"));
if (params.is_gen_docs || llama_supports_rpc()) {
add_opt(common_arg(
{"--rpc"}, "SERVERS",
"comma-separated list of RPC servers (host:port)",
[](common_params & params, const std::string & value) {
add_rpc_devices(value);
GGML_UNUSED(params);
add_opt(common_arg(
{"--rpc"}, "SERVERS",
"comma-separated list of RPC servers (host:port)",
[](common_params & params, const std::string & value) {
if (!llama_supports_rpc()) {
throw std::invalid_argument("RPC not supported in this build");
}
).set_env("LLAMA_ARG_RPC"));
}
add_rpc_devices(value);
GGML_UNUSED(params);
}
).set_env("LLAMA_ARG_RPC"));
add_opt(common_arg(
{"-lm", "--load-mode"}, "MODE",
"model loading mode (default: auto)\n"
+2
View File
@@ -10,6 +10,8 @@ extern "C" {
// device buffer
GGML_BACKEND_API ggml_backend_buffer_type_t ggml_backend_zdnn_buffer_type(void);
GGML_BACKEND_API bool ggml_backend_is_zdnn(ggml_backend_t backend);
GGML_BACKEND_API ggml_backend_reg_t ggml_backend_zdnn_reg(void);
#ifdef __cplusplus
+6 -8
View File
@@ -438,15 +438,13 @@ static ggml_backend_i ggml_backend_zdnn_i = {
};
static ggml_guid_t ggml_backend_zdnn_guid(void) {
static const char * guid_str = "IBM-ZDNN-ACCELER";
return reinterpret_cast<ggml_guid_t>((void *)guid_str);
static char guid_str[] = "IBM-ZDNN-ACCELER";
return reinterpret_cast<ggml_guid_t>(guid_str);
}
bool ggml_backend_is_zdnn(ggml_backend_t backend) {
return backend != NULL &&
ggml_guid_matches(backend->guid, ggml_backend_zdnn_guid());
GGML_UNUSED(backend);
}
//
@@ -483,7 +481,7 @@ static void ggml_backend_zdnn_device_get_props(ggml_backend_dev_t dev, ggml_back
props->description = ggml_backend_zdnn_device_get_description(dev);
props->type = ggml_backend_zdnn_device_get_type(dev);
ggml_backend_zdnn_device_get_memory(dev, &props->memory_free, &props->memory_total);
props->caps = (ggml_backend_dev_caps) {
props->caps = {
/* .async = */ false,
/* .host_buffer = */ false,
/* .buffer_from_host_ptr = */ false,
@@ -500,7 +498,7 @@ static ggml_backend_t ggml_backend_zdnn_device_init(ggml_backend_dev_t dev, cons
}
ggml_backend_t backend = (ggml_backend *)malloc(sizeof(ggml_backend));
*backend = (ggml_backend) {
*backend = {
/* .guid = */ ggml_backend_zdnn_guid(),
/* .iface = */ ggml_backend_zdnn_i,
/* .device = */ dev,
@@ -619,13 +617,13 @@ ggml_backend_reg_t ggml_backend_zdnn_reg(void) {
atexit(ggml_zdnn_cleanup);
{
g_ggml_backend_zdnn_reg = (ggml_backend_reg) {
g_ggml_backend_zdnn_reg = {
/* .api_version = */ GGML_ZDNN_VERSION,
/* .iface = */ ggml_backend_zdnn_reg_i,
/* .context = */ NULL
};
g_ggml_backend_zdnn_device = (ggml_backend_device) {
g_ggml_backend_zdnn_device = {
/* .iface = */ ggml_backend_zdnn_device_i,
/* .reg = */ &g_ggml_backend_zdnn_reg,
/* .context = */ &g_ggml_ctx_dev_main
+2 -2
View File
@@ -102,12 +102,12 @@ int llama_server(int argc, char ** argv) {
// touch it. lifecycle is symmetric, stop_gc() runs in clean_up() before backend free
server_stream_session_manager_start();
SRV_INF("%s", "initializing ...\n");
if (!common_params_parse(argc, argv, params, LLAMA_EXAMPLE_SERVER)) {
return 1;
}
SRV_INF("%s", "initializing ...\n");
llama_backend_init();
llama_numa_init(params.numa);
+4
View File
@@ -4,3 +4,7 @@ add_library(stb INTERFACE)
add_library(vendor::stb ALIAS stb)
target_include_directories(stb INTERFACE ..)
if (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
target_compile_options(stb INTERFACE -Wno-maybe-uninitialized)
endif()