mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-09-27 21:46:57 +02:00
ggml : add get_alloc_size_n to buffer type interface
- Add ggml_backend_buft_get_alloc_size_n public API - Add optional get_alloc_size_n callback to ggml_backend_buffer_type_i - Share tensor->buffer planning between alloc_buffer_n default and get_alloc_size_n default - Replace unchecked realloc with std::vector in alloc_buffer_n default - Make ggml_backend_alloc_ctx_tensors_from_buft_size use the new API - Add test-alloc coverage for get_alloc_size_n Assisted-by: pi:llama.cpp/DeepSeek-V4-Flash-Vision-Exp
This commit is contained in:
@@ -34,14 +34,15 @@ extern "C" {
|
||||
// Backend buffer type
|
||||
//
|
||||
|
||||
GGML_API const char * ggml_backend_buft_name (ggml_backend_buffer_type_t buft);
|
||||
GGML_API ggml_backend_buffer_t ggml_backend_buft_alloc_buffer (ggml_backend_buffer_type_t buft, size_t size);
|
||||
GGML_API ggml_backend_buffer_t ggml_backend_buft_alloc_buffer_n(ggml_backend_buffer_type_t buft, struct ggml_tensor ** tensors, int n_tensors);
|
||||
GGML_API size_t ggml_backend_buft_get_alignment (ggml_backend_buffer_type_t buft);
|
||||
GGML_API size_t ggml_backend_buft_get_max_size (ggml_backend_buffer_type_t buft);
|
||||
GGML_API size_t ggml_backend_buft_get_alloc_size(ggml_backend_buffer_type_t buft, const struct ggml_tensor * tensor);
|
||||
GGML_API bool ggml_backend_buft_is_host (ggml_backend_buffer_type_t buft);
|
||||
GGML_API ggml_backend_dev_t ggml_backend_buft_get_device (ggml_backend_buffer_type_t buft);
|
||||
GGML_API const char * ggml_backend_buft_name (ggml_backend_buffer_type_t buft);
|
||||
GGML_API ggml_backend_buffer_t ggml_backend_buft_alloc_buffer (ggml_backend_buffer_type_t buft, size_t size);
|
||||
GGML_API ggml_backend_buffer_t ggml_backend_buft_alloc_buffer_n (ggml_backend_buffer_type_t buft, struct ggml_tensor ** tensors, int n_tensors);
|
||||
GGML_API size_t ggml_backend_buft_get_alignment (ggml_backend_buffer_type_t buft);
|
||||
GGML_API size_t ggml_backend_buft_get_max_size (ggml_backend_buffer_type_t buft);
|
||||
GGML_API size_t ggml_backend_buft_get_alloc_size (ggml_backend_buffer_type_t buft, const struct ggml_tensor * tensor);
|
||||
GGML_API size_t ggml_backend_buft_get_alloc_size_n(ggml_backend_buffer_type_t buft, struct ggml_tensor ** tensors, int n_tensors);
|
||||
GGML_API bool ggml_backend_buft_is_host (ggml_backend_buffer_type_t buft);
|
||||
GGML_API ggml_backend_dev_t ggml_backend_buft_get_device (ggml_backend_buffer_type_t buft);
|
||||
|
||||
//
|
||||
// Backend buffer
|
||||
|
||||
+24
-27
@@ -1117,18 +1117,18 @@ size_t ggml_gallocr_get_buffer_size(ggml_gallocr_t galloc, int buffer_id) {
|
||||
|
||||
// utils
|
||||
|
||||
ggml_backend_buffer_t ggml_backend_alloc_ctx_tensors_from_buft(struct ggml_context * ctx, ggml_backend_buffer_type_t buft) {
|
||||
GGML_ASSERT(ggml_get_no_alloc(ctx) == true);
|
||||
|
||||
int n_tensors = 0;
|
||||
static struct ggml_tensor ** ggml_backend_alloc_ctx_tensors_from_buft_collect(
|
||||
struct ggml_context * ctx, int * n_tensors) {
|
||||
int n = 0;
|
||||
for (struct ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) {
|
||||
n_tensors++;
|
||||
n++;
|
||||
}
|
||||
if (n_tensors == 0) {
|
||||
*n_tensors = n;
|
||||
if (n == 0) {
|
||||
return NULL;
|
||||
}
|
||||
|
||||
struct ggml_tensor ** tensors = (struct ggml_tensor **) malloc(n_tensors * sizeof(struct ggml_tensor *));
|
||||
struct ggml_tensor ** tensors = (struct ggml_tensor **) malloc(n * sizeof(struct ggml_tensor *));
|
||||
if (tensors == NULL) {
|
||||
return NULL;
|
||||
}
|
||||
@@ -1136,37 +1136,34 @@ ggml_backend_buffer_t ggml_backend_alloc_ctx_tensors_from_buft(struct ggml_conte
|
||||
for (struct ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) {
|
||||
tensors[i++] = t;
|
||||
}
|
||||
return tensors;
|
||||
}
|
||||
|
||||
ggml_backend_buffer_t ggml_backend_alloc_ctx_tensors_from_buft(struct ggml_context * ctx, ggml_backend_buffer_type_t buft) {
|
||||
GGML_ASSERT(ggml_get_no_alloc(ctx) == true);
|
||||
|
||||
int n_tensors = 0;
|
||||
struct ggml_tensor ** tensors = ggml_backend_alloc_ctx_tensors_from_buft_collect(ctx, &n_tensors);
|
||||
if (tensors == NULL) {
|
||||
return NULL;
|
||||
}
|
||||
|
||||
ggml_backend_buffer_t buffer = ggml_backend_buft_alloc_buffer_n(buft, tensors, n_tensors);
|
||||
free(tensors);
|
||||
return buffer;
|
||||
}
|
||||
|
||||
// TODO [TAG_ALLOC_SHARED_BUFFER_SPLIT]: reuse shared buffer-splitting logic from ggml_backend_buft_alloc_buffer_n_default
|
||||
size_t ggml_backend_alloc_ctx_tensors_from_buft_size(struct ggml_context * ctx, ggml_backend_buffer_type_t buft) {
|
||||
GGML_ASSERT(ggml_get_no_alloc(ctx) == true);
|
||||
|
||||
size_t alignment = ggml_backend_buft_get_alignment(buft);
|
||||
size_t max_size = ggml_backend_buft_get_max_size(buft);
|
||||
|
||||
size_t nbytes_total = 0;
|
||||
size_t cur_buf_size = 0;
|
||||
|
||||
for (struct ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) {
|
||||
size_t this_size = 0;
|
||||
if (t->data == NULL && t->view_src == NULL) {
|
||||
this_size = GGML_PAD(ggml_backend_buft_get_alloc_size(buft, t), alignment);
|
||||
}
|
||||
|
||||
if (cur_buf_size > 0 && (cur_buf_size + this_size) > max_size) {
|
||||
nbytes_total += cur_buf_size;
|
||||
cur_buf_size = this_size;
|
||||
} else {
|
||||
cur_buf_size += this_size;
|
||||
}
|
||||
int n_tensors = 0;
|
||||
struct ggml_tensor ** tensors = ggml_backend_alloc_ctx_tensors_from_buft_collect(ctx, &n_tensors);
|
||||
if (tensors == NULL) {
|
||||
return 0;
|
||||
}
|
||||
nbytes_total += cur_buf_size;
|
||||
|
||||
size_t nbytes_total = ggml_backend_buft_get_alloc_size_n(buft, tensors, n_tensors);
|
||||
free(tensors);
|
||||
return nbytes_total;
|
||||
}
|
||||
|
||||
|
||||
@@ -15,19 +15,21 @@ extern "C" {
|
||||
//
|
||||
|
||||
struct ggml_backend_buffer_type_i {
|
||||
const char * (*get_name) (ggml_backend_buffer_type_t buft);
|
||||
const char * (*get_name) (ggml_backend_buffer_type_t buft);
|
||||
// allocate a buffer of this type
|
||||
ggml_backend_buffer_t (*alloc_buffer) (ggml_backend_buffer_type_t buft, size_t size);
|
||||
ggml_backend_buffer_t (*alloc_buffer) (ggml_backend_buffer_type_t buft, size_t size);
|
||||
// (optional) allocate tensors from a list into a buffer of this type (defaults to alloc_buffer + linear allocator)
|
||||
ggml_backend_buffer_t (*alloc_buffer_n)(ggml_backend_buffer_type_t buft, struct ggml_tensor ** tensors, int n_tensors);
|
||||
ggml_backend_buffer_t (*alloc_buffer_n) (ggml_backend_buffer_type_t buft, struct ggml_tensor ** tensors, int n_tensors);
|
||||
// tensor alignment
|
||||
size_t (*get_alignment) (ggml_backend_buffer_type_t buft);
|
||||
size_t (*get_alignment) (ggml_backend_buffer_type_t buft);
|
||||
// (optional) max buffer size that can be allocated (defaults to SIZE_MAX)
|
||||
size_t (*get_max_size) (ggml_backend_buffer_type_t buft);
|
||||
size_t (*get_max_size) (ggml_backend_buffer_type_t buft);
|
||||
// (optional) data size needed to allocate the tensor, including padding (defaults to ggml_nbytes)
|
||||
size_t (*get_alloc_size)(ggml_backend_buffer_type_t buft, const struct ggml_tensor * tensor);
|
||||
size_t (*get_alloc_size) (ggml_backend_buffer_type_t buft, const struct ggml_tensor * tensor);
|
||||
// (optional) total data size needed to allocate the given tensors, including padding and splitting (defaults to per-tensor get_alloc_size)
|
||||
size_t (*get_alloc_size_n)(ggml_backend_buffer_type_t buft, struct ggml_tensor ** tensors, int n_tensors);
|
||||
// (optional) check if tensor data is in host memory and uses standard ggml tensor layout (defaults to false)
|
||||
bool (*is_host) (ggml_backend_buffer_type_t buft);
|
||||
bool (*is_host) (ggml_backend_buffer_type_t buft);
|
||||
};
|
||||
|
||||
struct ggml_backend_buffer_type {
|
||||
|
||||
@@ -333,13 +333,14 @@ static bool ggml_backend_meta_buffer_type_is_host(ggml_backend_buffer_type_t buf
|
||||
}
|
||||
|
||||
static const struct ggml_backend_buffer_type_i ggml_backend_meta_buffer_type_iface = {
|
||||
/* .get_name = */ ggml_backend_meta_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_meta_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ ggml_backend_meta_buffer_type_alloc_buffer_n,
|
||||
/* .get_alignment = */ ggml_backend_meta_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_meta_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_meta_buffer_type_get_alloc_size,
|
||||
/* .is_host = */ ggml_backend_meta_buffer_type_is_host,
|
||||
/* .get_name = */ ggml_backend_meta_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_meta_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ ggml_backend_meta_buffer_type_alloc_buffer_n,
|
||||
/* .get_alignment = */ ggml_backend_meta_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_meta_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_meta_buffer_type_get_alloc_size,
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ ggml_backend_meta_buffer_type_is_host,
|
||||
};
|
||||
|
||||
bool ggml_backend_buft_is_meta(ggml_backend_buffer_type_t buft) {
|
||||
|
||||
+131
-92
@@ -45,99 +45,128 @@ ggml_backend_buffer_t ggml_backend_buft_alloc_buffer(ggml_backend_buffer_type_t
|
||||
return buft->iface.alloc_buffer(buft, size);
|
||||
}
|
||||
|
||||
// TODO [TAG_ALLOC_SHARED_BUFFER_SPLIT]: extract shared buffer-splitting logic with ggml_backend_alloc_ctx_tensors_from_buft_size
|
||||
// default implementation of alloc_buffer_n
|
||||
// allocates tensors from a list into one or more buffers of the given type
|
||||
static ggml_backend_buffer_t ggml_backend_buft_alloc_buffer_n_default(ggml_backend_buffer_type_t buft, struct ggml_tensor ** tensors, int n_tensors) {
|
||||
size_t alignment = ggml_backend_buft_get_alignment(buft);
|
||||
size_t max_size = ggml_backend_buft_get_max_size(buft);
|
||||
// shared planning logic for allocating a list of tensors into one or more buffers of the given type
|
||||
struct ggml_backend_buft_alloc_buffer_n_plan_item {
|
||||
size_t size; // total bytes for this buffer
|
||||
int first; // first tensor index (inclusive)
|
||||
int last; // last tensor index (exclusive)
|
||||
};
|
||||
|
||||
ggml_backend_buffer_t * buffers = NULL;
|
||||
size_t n_buffers = 0;
|
||||
using ggml_backend_buft_alloc_buffer_n_plan_t = std::vector<ggml_backend_buft_alloc_buffer_n_plan_item>;
|
||||
|
||||
static ggml_backend_buft_alloc_buffer_n_plan_t ggml_backend_buft_alloc_buffer_n_plan(
|
||||
ggml_backend_buffer_type_t buft, struct ggml_tensor ** tensors, int n_tensors) {
|
||||
ggml_backend_buft_alloc_buffer_n_plan_t plan;
|
||||
|
||||
const size_t alignment = ggml_backend_buft_get_alignment(buft);
|
||||
const size_t max_size = ggml_backend_buft_get_max_size(buft);
|
||||
|
||||
size_t cur_buf_size = 0;
|
||||
int first = 0;
|
||||
for (int i = 0; i <= n_tensors; i++) {
|
||||
int first = 0;
|
||||
|
||||
for (int i = 0; i < n_tensors; i++) {
|
||||
size_t this_size = 0;
|
||||
if (i < n_tensors) {
|
||||
struct ggml_tensor * t = tensors[i];
|
||||
if (t->data == NULL && t->view_src == NULL) {
|
||||
this_size = GGML_PAD(ggml_backend_buft_get_alloc_size(buft, t), alignment);
|
||||
}
|
||||
struct ggml_tensor * t = tensors[i];
|
||||
if (t->data == NULL && t->view_src == NULL) {
|
||||
this_size = GGML_PAD(ggml_backend_buft_get_alloc_size(buft, t), alignment);
|
||||
}
|
||||
|
||||
// flush the current buffer if adding this tensor would exceed max_size, or if we are at the end
|
||||
bool should_flush = (i == n_tensors) || (cur_buf_size > 0 && (cur_buf_size + this_size) > max_size);
|
||||
if (should_flush && cur_buf_size > 0) {
|
||||
ggml_backend_buffer_t buffer = ggml_backend_buft_alloc_buffer(buft, cur_buf_size);
|
||||
if (buffer == NULL) {
|
||||
GGML_LOG_ERROR("%s: failed to allocate %s buffer of size %zu\n", __func__, ggml_backend_buft_name(buft), cur_buf_size);
|
||||
for (size_t b = 0; b < n_buffers; b++) {
|
||||
ggml_backend_buffer_free(buffers[b]);
|
||||
}
|
||||
free(buffers);
|
||||
return NULL;
|
||||
}
|
||||
struct ggml_tallocr tallocr = ggml_tallocr_new(buffer);
|
||||
|
||||
// allocate tensors in the current buffer
|
||||
struct ggml_tensor * t_failed = NULL;
|
||||
for (int j = first; j < i; j++) {
|
||||
struct ggml_tensor * t = tensors[j];
|
||||
if (t->data == NULL) {
|
||||
if (t->view_src == NULL) {
|
||||
if (ggml_tallocr_alloc(&tallocr, t) != GGML_STATUS_SUCCESS) {
|
||||
t_failed = t;
|
||||
break;
|
||||
}
|
||||
} else if (t->buffer == NULL) {
|
||||
if (ggml_backend_view_init(t) != GGML_STATUS_SUCCESS) {
|
||||
t_failed = t;
|
||||
break;
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if (t->view_src != NULL && t->buffer == NULL) {
|
||||
// view of a pre-allocated tensor
|
||||
if (ggml_backend_view_init(t) != GGML_STATUS_SUCCESS) {
|
||||
t_failed = t;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if (t_failed != NULL) {
|
||||
GGML_LOG_ERROR("%s: failed to initialize tensor %s\n", __func__, t_failed->name);
|
||||
for (size_t b = 0; b < n_buffers; b++) {
|
||||
ggml_backend_buffer_free(buffers[b]);
|
||||
}
|
||||
ggml_backend_buffer_free(buffer);
|
||||
free(buffers);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
buffers = (ggml_backend_buffer_t *) realloc(buffers, sizeof(ggml_backend_buffer_t) * (n_buffers + 1));
|
||||
buffers[n_buffers++] = buffer;
|
||||
// flush the current buffer if adding this tensor would exceed max_size
|
||||
if (cur_buf_size > 0 && (cur_buf_size + this_size) > max_size) {
|
||||
plan.push_back({ cur_buf_size, first, i });
|
||||
cur_buf_size = this_size;
|
||||
first = i;
|
||||
} else if (i < n_tensors) {
|
||||
first = i;
|
||||
} else {
|
||||
cur_buf_size += this_size;
|
||||
}
|
||||
}
|
||||
|
||||
if (n_buffers == 0) {
|
||||
free(buffers);
|
||||
if (cur_buf_size > 0) {
|
||||
plan.push_back({ cur_buf_size, first, n_tensors });
|
||||
}
|
||||
|
||||
return plan;
|
||||
}
|
||||
|
||||
// default implementation of alloc_buffer_n
|
||||
// allocates tensors from a list into one or more buffers of the given type
|
||||
static ggml_backend_buffer_t ggml_backend_buft_alloc_buffer_n_default(ggml_backend_buffer_type_t buft, struct ggml_tensor ** tensors, int n_tensors) {
|
||||
const ggml_backend_buft_alloc_buffer_n_plan_t plan = ggml_backend_buft_alloc_buffer_n_plan(buft, tensors, n_tensors);
|
||||
|
||||
std::vector<ggml_backend_buffer_t> buffers;
|
||||
buffers.reserve(plan.size());
|
||||
|
||||
for (const ggml_backend_buft_alloc_buffer_n_plan_item & item : plan) {
|
||||
ggml_backend_buffer_t buffer = ggml_backend_buft_alloc_buffer(buft, item.size);
|
||||
if (buffer == NULL) {
|
||||
GGML_LOG_ERROR("%s: failed to allocate %s buffer of size %zu\n", __func__, ggml_backend_buft_name(buft), item.size);
|
||||
for (ggml_backend_buffer_t b : buffers) {
|
||||
ggml_backend_buffer_free(b);
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
struct ggml_tallocr tallocr = ggml_tallocr_new(buffer);
|
||||
|
||||
// allocate tensors in the current buffer
|
||||
struct ggml_tensor * t_failed = NULL;
|
||||
for (int j = item.first; j < item.last; j++) {
|
||||
struct ggml_tensor * t = tensors[j];
|
||||
if (t->data == NULL) {
|
||||
if (t->view_src == NULL) {
|
||||
if (ggml_tallocr_alloc(&tallocr, t) != GGML_STATUS_SUCCESS) {
|
||||
t_failed = t;
|
||||
break;
|
||||
}
|
||||
} else if (t->buffer == NULL) {
|
||||
if (ggml_backend_view_init(t) != GGML_STATUS_SUCCESS) {
|
||||
t_failed = t;
|
||||
break;
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if (t->view_src != NULL && t->buffer == NULL) {
|
||||
// view of a pre-allocated tensor
|
||||
if (ggml_backend_view_init(t) != GGML_STATUS_SUCCESS) {
|
||||
t_failed = t;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if (t_failed != NULL) {
|
||||
GGML_LOG_ERROR("%s: failed to initialize tensor %s\n", __func__, t_failed->name);
|
||||
for (ggml_backend_buffer_t b : buffers) {
|
||||
ggml_backend_buffer_free(b);
|
||||
}
|
||||
ggml_backend_buffer_free(buffer);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
buffers.push_back(buffer);
|
||||
}
|
||||
|
||||
if (buffers.empty()) {
|
||||
return NULL;
|
||||
}
|
||||
|
||||
ggml_backend_buffer_t result;
|
||||
if (n_buffers == 1) {
|
||||
result = buffers[0];
|
||||
} else {
|
||||
result = ggml_backend_multi_buffer_alloc_buffer(buffers, n_buffers);
|
||||
if (buffers.size() == 1) {
|
||||
return buffers[0];
|
||||
}
|
||||
free(buffers);
|
||||
return result;
|
||||
|
||||
return ggml_backend_multi_buffer_alloc_buffer(buffers.data(), buffers.size());
|
||||
}
|
||||
|
||||
// default implementation of get_alloc_size_n
|
||||
// returns the total size that alloc_buffer_n_default would allocate for the given tensors
|
||||
static size_t ggml_backend_buft_get_alloc_size_n_default(ggml_backend_buffer_type_t buft, struct ggml_tensor ** tensors, int n_tensors) {
|
||||
const ggml_backend_buft_alloc_buffer_n_plan_t plan = ggml_backend_buft_alloc_buffer_n_plan(buft, tensors, n_tensors);
|
||||
|
||||
size_t total = 0;
|
||||
for (const ggml_backend_buft_alloc_buffer_n_plan_item & item : plan) {
|
||||
total += item.size;
|
||||
}
|
||||
return total;
|
||||
}
|
||||
|
||||
ggml_backend_buffer_t ggml_backend_buft_alloc_buffer_n(ggml_backend_buffer_type_t buft, struct ggml_tensor ** tensors, int n_tensors) {
|
||||
@@ -181,6 +210,14 @@ size_t ggml_backend_buft_get_alloc_size(ggml_backend_buffer_type_t buft, const s
|
||||
return ggml_nbytes(tensor);
|
||||
}
|
||||
|
||||
size_t ggml_backend_buft_get_alloc_size_n(ggml_backend_buffer_type_t buft, struct ggml_tensor ** tensors, int n_tensors) {
|
||||
GGML_ASSERT(buft);
|
||||
if (buft->iface.get_alloc_size_n) {
|
||||
return buft->iface.get_alloc_size_n(buft, tensors, n_tensors);
|
||||
}
|
||||
return ggml_backend_buft_get_alloc_size_n_default(buft, tensors, n_tensors);
|
||||
}
|
||||
|
||||
bool ggml_backend_buft_is_host(ggml_backend_buffer_type_t buft) {
|
||||
GGML_ASSERT(buft);
|
||||
if (buft->iface.is_host) {
|
||||
@@ -2573,13 +2610,14 @@ static bool ggml_backend_cpu_buffer_type_is_host(ggml_backend_buffer_type_t buft
|
||||
ggml_backend_buffer_type_t ggml_backend_cpu_buffer_type(void) {
|
||||
static struct ggml_backend_buffer_type ggml_backend_cpu_buffer_type = {
|
||||
/* .iface = */ {
|
||||
/* .get_name = */ ggml_backend_cpu_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_cpu_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_cpu_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ NULL, // defaults to SIZE_MAX
|
||||
/* .get_alloc_size = */ NULL, // defaults to ggml_nbytes
|
||||
/* .is_host = */ ggml_backend_cpu_buffer_type_is_host,
|
||||
/* .get_name = */ ggml_backend_cpu_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_cpu_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_cpu_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ NULL, // defaults to SIZE_MAX
|
||||
/* .get_alloc_size = */ NULL, // defaults to ggml_nbytes
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ ggml_backend_cpu_buffer_type_is_host,
|
||||
},
|
||||
/* .device = */ NULL, // FIXME ggml_backend_reg_dev_get(ggml_backend_cpu_reg(), 0),
|
||||
/* .context = */ NULL,
|
||||
@@ -2597,13 +2635,14 @@ static const char * ggml_backend_cpu_buffer_from_ptr_type_get_name(ggml_backend_
|
||||
static ggml_backend_buffer_type_t ggml_backend_cpu_buffer_from_ptr_type(void) {
|
||||
static struct ggml_backend_buffer_type ggml_backend_cpu_buffer_type = {
|
||||
/* .iface = */ {
|
||||
/* .get_name = */ ggml_backend_cpu_buffer_from_ptr_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_cpu_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_cpu_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ NULL, // defaults to SIZE_MAX
|
||||
/* .get_alloc_size = */ NULL, // defaults to ggml_nbytes
|
||||
/* .is_host = */ ggml_backend_cpu_buffer_type_is_host,
|
||||
/* .get_name = */ ggml_backend_cpu_buffer_from_ptr_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_cpu_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_cpu_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ NULL, // defaults to SIZE_MAX
|
||||
/* .get_alloc_size = */ NULL, // defaults to ggml_nbytes
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ ggml_backend_cpu_buffer_type_is_host,
|
||||
},
|
||||
/* .device = */ NULL, // FIXME ggml_backend_reg_dev_get(ggml_backend_cpu_reg(), 0),
|
||||
/* .context = */ NULL,
|
||||
|
||||
@@ -1595,13 +1595,14 @@ static bool ggml_backend_cann_buffer_type_is_host(ggml_backend_buffer_type_t buf
|
||||
* memory for CANN buffer types in the GGML backend.
|
||||
*/
|
||||
static const ggml_backend_buffer_type_i ggml_backend_cann_buffer_type_interface = {
|
||||
/* .get_name = */ ggml_backend_cann_buffer_type_name,
|
||||
/* .alloc_buffer = */ ggml_backend_cann_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_cann_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ NULL, // defaults to SIZE_MAX
|
||||
/* .get_alloc_size = */ ggml_backend_cann_buffer_type_get_alloc_size,
|
||||
/* .is_host = */ ggml_backend_cann_buffer_type_is_host,
|
||||
/* .get_name = */ ggml_backend_cann_buffer_type_name,
|
||||
/* .alloc_buffer = */ ggml_backend_cann_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_cann_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ NULL, // defaults to SIZE_MAX
|
||||
/* .get_alloc_size = */ ggml_backend_cann_buffer_type_get_alloc_size,
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ ggml_backend_cann_buffer_type_is_host,
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -1743,13 +1744,14 @@ static ggml_backend_buffer_t ggml_backend_cann_host_buffer_type_alloc_buffer(ggm
|
||||
ggml_backend_buffer_type_t ggml_backend_cann_host_buffer_type() {
|
||||
static struct ggml_backend_buffer_type ggml_backend_cann_buffer_type_host = {
|
||||
/* .iface = */ {
|
||||
/* .get_name = */ ggml_backend_cann_host_buffer_type_name,
|
||||
/* .alloc_buffer = */ ggml_backend_cann_host_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_cpu_buffer_type()->iface.get_alignment,
|
||||
/* .get_max_size = */ NULL, // defaults to SIZE_MAX
|
||||
/* .get_alloc_size = */ ggml_backend_cpu_buffer_type()->iface.get_alloc_size,
|
||||
/* .is_host = */ ggml_backend_cpu_buffer_type()->iface.is_host,
|
||||
/* .get_name = */ ggml_backend_cann_host_buffer_type_name,
|
||||
/* .alloc_buffer = */ ggml_backend_cann_host_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_cpu_buffer_type()->iface.get_alignment,
|
||||
/* .get_max_size = */ NULL, // defaults to SIZE_MAX
|
||||
/* .get_alloc_size = */ ggml_backend_cpu_buffer_type()->iface.get_alloc_size,
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ ggml_backend_cpu_buffer_type()->iface.is_host,
|
||||
},
|
||||
/* .device = */
|
||||
ggml_backend_reg_dev_get(ggml_backend_cann_reg(), 0),
|
||||
|
||||
@@ -228,13 +228,14 @@ static bool ggml_amx_init() {
|
||||
ggml_backend_buffer_type_t ggml_backend_amx_buffer_type() {
|
||||
static struct ggml_backend_buffer_type ggml_backend_buffer_type_amx = {
|
||||
/* .iface = */ {
|
||||
/* .get_name = */ ggml_backend_amx_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_amx_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ nullptr,
|
||||
/* .get_alignment = */ ggml_backend_amx_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ nullptr, // defaults to SIZE_MAX
|
||||
/* .get_alloc_size = */ ggml_backend_amx_buffer_type_get_alloc_size,
|
||||
/* .is_host = */ nullptr,
|
||||
/* .get_name = */ ggml_backend_amx_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_amx_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ nullptr,
|
||||
/* .get_alignment = */ ggml_backend_amx_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ nullptr, // defaults to SIZE_MAX
|
||||
/* .get_alloc_size = */ ggml_backend_amx_buffer_type_get_alloc_size,
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ nullptr,
|
||||
},
|
||||
/* .device = */ ggml_backend_reg_dev_get(ggml_backend_cpu_reg(), 0),
|
||||
/* .context = */ new ggml::cpu::amx::extra_buffer_type(),
|
||||
|
||||
@@ -40,13 +40,14 @@ static ggml_backend_buffer_t ggml_backend_cpu_hbm_buffer_type_alloc_buffer(ggml_
|
||||
ggml_backend_buffer_type_t ggml_backend_cpu_hbm_buffer_type(void) {
|
||||
static struct ggml_backend_buffer_type ggml_backend_cpu_buffer_type_hbm = {
|
||||
/* .iface = */ {
|
||||
/* .get_name = */ ggml_backend_cpu_hbm_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_cpu_hbm_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ nullptr,
|
||||
/* .get_alignment = */ ggml_backend_cpu_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ nullptr, // defaults to SIZE_MAX
|
||||
/* .get_alloc_size = */ nullptr, // defaults to ggml_nbytes
|
||||
/* .is_host = */ ggml_backend_cpu_buffer_type_is_host,
|
||||
/* .get_name = */ ggml_backend_cpu_hbm_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_cpu_hbm_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ nullptr,
|
||||
/* .get_alignment = */ ggml_backend_cpu_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ nullptr, // defaults to SIZE_MAX
|
||||
/* .get_alloc_size = */ nullptr, // defaults to ggml_nbytes
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ ggml_backend_cpu_buffer_type_is_host,
|
||||
},
|
||||
/* .context = */ nullptr,
|
||||
};
|
||||
|
||||
@@ -1902,13 +1902,14 @@ ggml_backend_buffer_type_t ggml_backend_cpu_kleidiai_buffer_type(void) {
|
||||
static ggml::cpu::kleidiai::extra_buffer_type ctx;
|
||||
static struct ggml_backend_buffer_type ggml_backend_cpu_buffer_type_kleidiai = {
|
||||
/* .iface = */ {
|
||||
/* .get_name = */ ggml_backend_cpu_kleidiai_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_cpu_kleidiai_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ nullptr,
|
||||
/* .get_alignment = */ ggml_backend_cpu_kleidiai_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ nullptr, // defaults to SIZE_MAX
|
||||
/* .get_alloc_size = */ ggml_backend_cpu_kleidiai_buffer_type_get_alloc_size,
|
||||
/* .is_host = */ nullptr,
|
||||
/* .get_name = */ ggml_backend_cpu_kleidiai_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_cpu_kleidiai_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ nullptr,
|
||||
/* .get_alignment = */ ggml_backend_cpu_kleidiai_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ nullptr, // defaults to SIZE_MAX
|
||||
/* .get_alloc_size = */ ggml_backend_cpu_kleidiai_buffer_type_get_alloc_size,
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ nullptr,
|
||||
},
|
||||
/* .device = */ ggml_backend_reg_dev_get(ggml_backend_cpu_reg(), 0),
|
||||
/* .context = */ &ctx,
|
||||
|
||||
@@ -5238,13 +5238,14 @@ class extra_buffer_type : ggml::cpu::extra_buffer_type {
|
||||
ggml_backend_buffer_type_t ggml_backend_cpu_repack_buffer_type(void) {
|
||||
static struct ggml_backend_buffer_type ggml_backend_cpu_buffer_type_repack = {
|
||||
/* .iface = */ {
|
||||
/* .get_name = */ ggml_backend_cpu_repack_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_cpu_repack_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ nullptr,
|
||||
/* .get_alignment = */ ggml_backend_cpu_repack_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ nullptr, // defaults to SIZE_MAX
|
||||
/* .get_alloc_size = */ nullptr, // defaults to ggml_nbytes
|
||||
/* .is_host = */ nullptr,
|
||||
/* .get_name = */ ggml_backend_cpu_repack_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_cpu_repack_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ nullptr,
|
||||
/* .get_alignment = */ ggml_backend_cpu_repack_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ nullptr, // defaults to SIZE_MAX
|
||||
/* .get_alloc_size = */ nullptr, // defaults to ggml_nbytes
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ nullptr,
|
||||
},
|
||||
/* .device = */ ggml_backend_reg_dev_get(ggml_backend_cpu_reg(), 0),
|
||||
/* .context = */ new ggml::cpu::repack::extra_buffer_type(),
|
||||
|
||||
@@ -1650,13 +1650,14 @@ ggml_backend_buffer_type_t ggml_backend_cpu_riscv64_spacemit_buffer_type(void) {
|
||||
static ggml_backend_buffer_type ggml_backend_cpu_buffer_type_riscv64_spacemit = {
|
||||
/* .iface = */
|
||||
{
|
||||
/* .get_name = */ ggml_backend_cpu_riscv64_spacemit_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_cpu_riscv64_spacemit_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_cpu_riscv64_spacemit_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ nullptr,
|
||||
/* .get_alloc_size = */ ggml_backend_cpu_riscv64_spacemit_nbytes,
|
||||
/* .is_host = */ nullptr,
|
||||
/* .get_name = */ ggml_backend_cpu_riscv64_spacemit_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_cpu_riscv64_spacemit_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_cpu_riscv64_spacemit_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ nullptr,
|
||||
/* .get_alloc_size = */ ggml_backend_cpu_riscv64_spacemit_nbytes,
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ nullptr,
|
||||
},
|
||||
/* .device = */
|
||||
ggml_backend_reg_dev_get(ggml_backend_cpu_reg(), 0),
|
||||
|
||||
@@ -926,13 +926,14 @@ static size_t ggml_backend_cuda_buffer_type_get_alloc_size(ggml_backend_buffer_t
|
||||
}
|
||||
|
||||
static const ggml_backend_buffer_type_i ggml_backend_cuda_buffer_type_interface = {
|
||||
/* .get_name = */ ggml_backend_cuda_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_cuda_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_cuda_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ NULL, // defaults to SIZE_MAX
|
||||
/* .get_alloc_size = */ ggml_backend_cuda_buffer_type_get_alloc_size,
|
||||
/* .is_host = */ NULL,
|
||||
/* .get_name = */ ggml_backend_cuda_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_cuda_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_cuda_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ NULL, // defaults to SIZE_MAX
|
||||
/* .get_alloc_size = */ ggml_backend_cuda_buffer_type_get_alloc_size,
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ NULL,
|
||||
};
|
||||
|
||||
ggml_backend_buffer_type_t ggml_backend_cuda_buffer_type(int device) {
|
||||
@@ -1311,13 +1312,14 @@ static ggml_backend_buffer_t ggml_backend_cuda_host_buffer_type_alloc_buffer(ggm
|
||||
ggml_backend_buffer_type_t ggml_backend_cuda_host_buffer_type() {
|
||||
static struct ggml_backend_buffer_type ggml_backend_cuda_buffer_type_host = {
|
||||
/* .iface = */ {
|
||||
/* .get_name = */ ggml_backend_cuda_host_buffer_type_name,
|
||||
/* .alloc_buffer = */ ggml_backend_cuda_host_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_cpu_buffer_type()->iface.get_alignment,
|
||||
/* .get_max_size = */ NULL, // defaults to SIZE_MAX
|
||||
/* .get_alloc_size = */ ggml_backend_cpu_buffer_type()->iface.get_alloc_size,
|
||||
/* .is_host = */ ggml_backend_cpu_buffer_type()->iface.is_host,
|
||||
/* .get_name = */ ggml_backend_cuda_host_buffer_type_name,
|
||||
/* .alloc_buffer = */ ggml_backend_cuda_host_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_cpu_buffer_type()->iface.get_alignment,
|
||||
/* .get_max_size = */ NULL, // defaults to SIZE_MAX
|
||||
/* .get_alloc_size = */ ggml_backend_cpu_buffer_type()->iface.get_alloc_size,
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ ggml_backend_cpu_buffer_type()->iface.is_host,
|
||||
},
|
||||
/* .device = */ ggml_backend_reg_dev_get(ggml_backend_cuda_reg(), 0),
|
||||
/* .context = */ nullptr,
|
||||
|
||||
@@ -440,13 +440,14 @@ static bool ggml_backend_et_buffer_type_is_host(ggml_backend_buffer_type_t buft)
|
||||
}
|
||||
|
||||
static const struct ggml_backend_buffer_type_i ggml_backend_et_buffer_type_i = {
|
||||
/* .get_name = */ ggml_backend_et_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_et_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_et_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_et_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_et_buffer_type_get_alloc_size,
|
||||
/* .is_host = */ ggml_backend_et_buffer_type_is_host,
|
||||
/* .get_name = */ ggml_backend_et_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_et_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_et_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_et_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_et_buffer_type_get_alloc_size,
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ ggml_backend_et_buffer_type_is_host,
|
||||
};
|
||||
|
||||
static const char * ggml_backend_et_get_name(ggml_backend_t backend) {
|
||||
|
||||
@@ -2146,23 +2146,25 @@ static bool ggml_backend_hexagon_host_buffer_type_is_host(ggml_backend_buffer_ty
|
||||
}
|
||||
|
||||
static ggml_backend_buffer_type_i ggml_backend_hexagon_buffer_type_interface = {
|
||||
/* .get_name = */ ggml_backend_hexagon_buffer_type_name,
|
||||
/* .alloc_buffer = */ ggml_backend_hexagon_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_hexagon_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_hexagon_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_hexagon_buffer_type_get_alloc_size,
|
||||
/* .is_host = */ ggml_backend_hexagon_buffer_type_is_host,
|
||||
/* .get_name = */ ggml_backend_hexagon_buffer_type_name,
|
||||
/* .alloc_buffer = */ ggml_backend_hexagon_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_hexagon_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_hexagon_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_hexagon_buffer_type_get_alloc_size,
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ ggml_backend_hexagon_buffer_type_is_host,
|
||||
};
|
||||
|
||||
static ggml_backend_buffer_type_i ggml_backend_hexagon_host_buffer_type_interface = {
|
||||
/* .get_name = */ ggml_backend_hexagon_buffer_type_name,
|
||||
/* .alloc_buffer = */ ggml_backend_hexagon_host_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_hexagon_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_hexagon_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_hexagon_buffer_type_get_alloc_size,
|
||||
/* .is_host = */ ggml_backend_hexagon_host_buffer_type_is_host,
|
||||
/* .get_name = */ ggml_backend_hexagon_buffer_type_name,
|
||||
/* .alloc_buffer = */ ggml_backend_hexagon_host_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_hexagon_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_hexagon_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_hexagon_buffer_type_get_alloc_size,
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ ggml_backend_hexagon_host_buffer_type_is_host,
|
||||
};
|
||||
|
||||
ggml_backend_hexagon_device_context::ggml_backend_hexagon_device_context(int dev_id, const ggml_hexagon_device_config & config, ggml_backend_dev_t dev)
|
||||
|
||||
@@ -310,13 +310,14 @@ static ggml_backend_buffer_type_t ggml_backend_metal_buffer_type_shared(int devi
|
||||
|
||||
ggml_backend_buffer_type buft = {
|
||||
/* .iface = */ {
|
||||
/* .get_name = */ ggml_backend_metal_buffer_type_shared_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_metal_buffer_type_shared_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_metal_buffer_type_shared_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_metal_buffer_type_shared_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_metal_buffer_type_shared_get_alloc_size,
|
||||
/* .is_host = */ ggml_backend_metal_buffer_type_shared_is_host,
|
||||
/* .get_name = */ ggml_backend_metal_buffer_type_shared_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_metal_buffer_type_shared_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_metal_buffer_type_shared_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_metal_buffer_type_shared_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_metal_buffer_type_shared_get_alloc_size,
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ ggml_backend_metal_buffer_type_shared_is_host,
|
||||
},
|
||||
/* .device = */ ggml_backend_reg_dev_get(ggml_backend_metal_reg(), i),
|
||||
/* .context = */ raw_ctx,
|
||||
@@ -386,13 +387,14 @@ static ggml_backend_buffer_type_t ggml_backend_metal_buffer_type_private(int dev
|
||||
|
||||
ggml_backend_buffer_type buft = {
|
||||
/* .iface = */ {
|
||||
/* .get_name = */ ggml_backend_metal_buffer_type_private_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_metal_buffer_type_private_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_metal_buffer_type_private_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_metal_buffer_type_private_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_metal_buffer_type_private_get_alloc_size,
|
||||
/* .is_host = */ ggml_backend_metal_buffer_type_private_is_host,
|
||||
/* .get_name = */ ggml_backend_metal_buffer_type_private_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_metal_buffer_type_private_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_metal_buffer_type_private_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_metal_buffer_type_private_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_metal_buffer_type_private_get_alloc_size,
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ ggml_backend_metal_buffer_type_private_is_host,
|
||||
},
|
||||
/* .device = */ ggml_backend_reg_dev_get(ggml_backend_metal_reg(), i),
|
||||
/* .context = */ raw_ctx,
|
||||
@@ -465,13 +467,14 @@ static ggml_backend_buffer_type_t ggml_backend_metal_buffer_type_mapped(int devi
|
||||
// https://github.com/ggml-org/llama.cpp/pull/15832#discussion_r2333177099
|
||||
ggml_backend_buffer_type buft = {
|
||||
/* .iface = */ {
|
||||
/* .get_name = */ ggml_backend_metal_buffer_type_mapped_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_metal_buffer_type_mapped_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_metal_buffer_type_mapped_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_metal_buffer_type_mapped_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_metal_buffer_type_mapped_get_alloc_size,
|
||||
/* .is_host = */ ggml_backend_metal_buffer_type_mapped_is_host,
|
||||
/* .get_name = */ ggml_backend_metal_buffer_type_mapped_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_metal_buffer_type_mapped_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_metal_buffer_type_mapped_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_metal_buffer_type_mapped_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_metal_buffer_type_mapped_get_alloc_size,
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ ggml_backend_metal_buffer_type_mapped_is_host,
|
||||
},
|
||||
/* .device = */ ggml_backend_reg_dev_get(ggml_backend_metal_reg(), i),
|
||||
/* .context = */ raw_ctx,
|
||||
|
||||
@@ -12769,13 +12769,14 @@ static size_t ggml_backend_opencl_buffer_type_get_alloc_size(ggml_backend_buffer
|
||||
}
|
||||
|
||||
static ggml_backend_buffer_type_i ggml_backend_opencl_buffer_type_interface = {
|
||||
/* .get_name = */ ggml_backend_opencl_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_opencl_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_opencl_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_opencl_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_opencl_buffer_type_get_alloc_size,
|
||||
/* .is_host = */ NULL,
|
||||
/* .get_name = */ ggml_backend_opencl_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_opencl_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_opencl_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_opencl_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_opencl_buffer_type_get_alloc_size,
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ NULL,
|
||||
};
|
||||
|
||||
//
|
||||
|
||||
@@ -630,13 +630,14 @@ static size_t ggml_backend_openvino_buffer_type_get_alloc_size(ggml_backend_buff
|
||||
}
|
||||
|
||||
static const ggml_backend_buffer_type_i ggml_backend_openvino_buffer_type_interface = {
|
||||
/* .get_name = */ ggml_backend_openvino_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_openvino_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_openvino_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_openvino_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_openvino_buffer_type_get_alloc_size,
|
||||
/* .is_host = */ nullptr,
|
||||
/* .get_name = */ ggml_backend_openvino_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_openvino_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_openvino_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_openvino_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_openvino_buffer_type_get_alloc_size,
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ nullptr,
|
||||
};
|
||||
|
||||
// Get buffer type for a specific device
|
||||
@@ -684,13 +685,14 @@ static bool ggml_backend_openvino_host_buffer_type_is_host(ggml_backend_buffer_t
|
||||
}
|
||||
|
||||
static const ggml_backend_buffer_type_i ggml_backend_openvino_host_buffer_type_interface = {
|
||||
/* .get_name = */ ggml_backend_openvino_host_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_openvino_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_openvino_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_openvino_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_openvino_buffer_type_get_alloc_size,
|
||||
/* .is_host = */ ggml_backend_openvino_host_buffer_type_is_host,
|
||||
/* .get_name = */ ggml_backend_openvino_host_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_openvino_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_openvino_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_openvino_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_openvino_buffer_type_get_alloc_size,
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ ggml_backend_openvino_host_buffer_type_is_host,
|
||||
};
|
||||
|
||||
GGML_BACKEND_API ggml_backend_buffer_type_t ggml_backend_openvino_host_buffer_type(int device) {
|
||||
|
||||
@@ -922,13 +922,14 @@ static size_t ggml_backend_rpc_buffer_type_get_alloc_size(ggml_backend_buffer_ty
|
||||
}
|
||||
|
||||
static ggml_backend_buffer_type_i ggml_backend_rpc_buffer_type_interface = {
|
||||
/* .get_name = */ ggml_backend_rpc_buffer_type_name,
|
||||
/* .alloc_buffer = */ ggml_backend_rpc_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_rpc_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_rpc_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_rpc_buffer_type_get_alloc_size,
|
||||
/* .is_host = */ NULL,
|
||||
/* .get_name = */ ggml_backend_rpc_buffer_type_name,
|
||||
/* .alloc_buffer = */ ggml_backend_rpc_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_rpc_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_rpc_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_rpc_buffer_type_get_alloc_size,
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ NULL,
|
||||
};
|
||||
|
||||
static const char * ggml_backend_rpc_name(ggml_backend_t backend) {
|
||||
|
||||
@@ -1053,13 +1053,14 @@ static size_t ggml_backend_sycl_buffer_type_get_alloc_size(ggml_backend_buffer_t
|
||||
}
|
||||
|
||||
static const ggml_backend_buffer_type_i ggml_backend_sycl_buffer_type_interface = {
|
||||
/* .get_name = */ ggml_backend_sycl_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_sycl_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_sycl_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_sycl_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_sycl_buffer_type_get_alloc_size,
|
||||
/* .is_host = */ NULL,
|
||||
/* .get_name = */ ggml_backend_sycl_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_sycl_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_sycl_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_sycl_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_sycl_buffer_type_get_alloc_size,
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ NULL,
|
||||
};
|
||||
|
||||
ggml_backend_buffer_type_t ggml_backend_sycl_buffer_type(int device) {
|
||||
@@ -1490,13 +1491,14 @@ static bool ggml_backend_sycl_split_buffer_type_is_host(ggml_backend_buffer_type
|
||||
}
|
||||
|
||||
static ggml_backend_buffer_type_i ggml_backend_sycl_split_buffer_type_interface = {
|
||||
/* .get_name = */ ggml_backend_sycl_split_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_sycl_split_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_sycl_split_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ NULL, // defaults to SIZE_MAX
|
||||
/* .get_alloc_size = */ ggml_backend_sycl_split_buffer_type_get_alloc_size,
|
||||
/* .is_host = */ ggml_backend_sycl_split_buffer_type_is_host,
|
||||
/* .get_name = */ ggml_backend_sycl_split_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_sycl_split_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_sycl_split_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ NULL, // defaults to SIZE_MAX
|
||||
/* .get_alloc_size = */ ggml_backend_sycl_split_buffer_type_get_alloc_size,
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ ggml_backend_sycl_split_buffer_type_is_host,
|
||||
};
|
||||
|
||||
ggml_backend_buffer_type_t ggml_backend_sycl_split_buffer_type(int main_device, const float * tensor_split) {
|
||||
@@ -1635,13 +1637,14 @@ static ggml_backend_buffer_type_t ggml_backend_sycl_host_buffer_type_for_device(
|
||||
for (size_t i = 0; i < bufts.size(); i++) {
|
||||
bufts[i] = {
|
||||
/* .iface = */ {
|
||||
/* .get_name = */ ggml_backend_sycl_host_buffer_type_name,
|
||||
/* .alloc_buffer = */ ggml_backend_sycl_host_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_cpu_buffer_type()->iface.get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_sycl_host_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_cpu_buffer_type()->iface.get_alloc_size,
|
||||
/* .is_host = */ ggml_backend_cpu_buffer_type()->iface.is_host,
|
||||
/* .get_name = */ ggml_backend_sycl_host_buffer_type_name,
|
||||
/* .alloc_buffer = */ ggml_backend_sycl_host_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_cpu_buffer_type()->iface.get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_sycl_host_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_cpu_buffer_type()->iface.get_alloc_size,
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ ggml_backend_cpu_buffer_type()->iface.is_host,
|
||||
},
|
||||
/* .device = */ ggml_backend_reg_dev_get(ggml_backend_sycl_reg(), i),
|
||||
/* .context = */ nullptr,
|
||||
|
||||
@@ -63,21 +63,23 @@ static size_t ggml_backend_remoting_buffer_type_get_alloc_size(ggml_backend_buff
|
||||
}
|
||||
|
||||
const ggml_backend_buffer_type_i ggml_backend_remoting_buffer_type_interface = {
|
||||
/* .get_name = */ ggml_backend_remoting_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_remoting_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_remoting_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_remoting_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_remoting_buffer_type_get_alloc_size,
|
||||
/* .is_host = */ NULL,
|
||||
/* .get_name = */ ggml_backend_remoting_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_remoting_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_remoting_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_remoting_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_remoting_buffer_type_get_alloc_size,
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ NULL,
|
||||
};
|
||||
|
||||
const ggml_backend_buffer_type_i ggml_backend_remoting_buffer_from_ptr_type_interface = {
|
||||
/* .get_name = */ ggml_backend_remoting_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ NULL,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_remoting_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_remoting_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_remoting_buffer_type_get_alloc_size,
|
||||
/* .is_host = */ NULL,
|
||||
/* .get_name = */ ggml_backend_remoting_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ NULL,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_remoting_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_remoting_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_remoting_buffer_type_get_alloc_size,
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ NULL,
|
||||
};
|
||||
|
||||
@@ -1,13 +1,14 @@
|
||||
#include "ggml-vulkan-common.h"
|
||||
|
||||
ggml_backend_buffer_type_i ggml_backend_vk_buffer_type_interface = {
|
||||
/* .get_name = */ ggml_backend_vk_buffer_type_name,
|
||||
/* .alloc_buffer = */ ggml_backend_vk_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_vk_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_vk_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_vk_buffer_type_get_alloc_size,
|
||||
/* .is_host = */ NULL,
|
||||
/* .get_name = */ ggml_backend_vk_buffer_type_name,
|
||||
/* .alloc_buffer = */ ggml_backend_vk_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_vk_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_vk_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_vk_buffer_type_get_alloc_size,
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ NULL,
|
||||
};
|
||||
|
||||
static std::vector<uint32_t> ggml_vk_find_memory_properties(const vk::PhysicalDeviceMemoryProperties* mem_props, vk::MemoryRequirements* mem_req, vk::MemoryPropertyFlags flags) {
|
||||
|
||||
@@ -13056,13 +13056,14 @@ static size_t ggml_backend_vk_host_buffer_type_get_max_size(ggml_backend_buffer_
|
||||
ggml_backend_buffer_type_t ggml_backend_vk_host_buffer_type() {
|
||||
static struct ggml_backend_buffer_type ggml_backend_vk_buffer_type_host = {
|
||||
/* .iface = */ {
|
||||
/* .get_name = */ ggml_backend_vk_host_buffer_type_name,
|
||||
/* .alloc_buffer = */ ggml_backend_vk_host_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ nullptr,
|
||||
/* .get_alignment = */ ggml_backend_vk_host_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_vk_host_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_cpu_buffer_type()->iface.get_alloc_size,
|
||||
/* .is_host = */ ggml_backend_cpu_buffer_type()->iface.is_host,
|
||||
/* .get_name = */ ggml_backend_vk_host_buffer_type_name,
|
||||
/* .alloc_buffer = */ ggml_backend_vk_host_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ nullptr,
|
||||
/* .get_alignment = */ ggml_backend_vk_host_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_vk_host_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_cpu_buffer_type()->iface.get_alloc_size,
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ ggml_backend_cpu_buffer_type()->iface.is_host,
|
||||
},
|
||||
/* .device = */ ggml_backend_reg_dev_get(ggml_backend_vk_reg(), 0),
|
||||
/* .context = */ nullptr,
|
||||
|
||||
@@ -4311,13 +4311,14 @@ static ggml_backend_buffer_type_t ggml_backend_webgpu_device_get_buffer_type(ggm
|
||||
|
||||
static struct ggml_backend_buffer_type ggml_backend_webgpu_buffer_type = {
|
||||
/* .iface = */ {
|
||||
/* .get_name = */ ggml_backend_webgpu_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_webgpu_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_webgpu_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_webgpu_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_webgpu_buffer_type_get_alloc_size,
|
||||
/* .is_host = */ NULL, // defaults to false
|
||||
/* .get_name = */ ggml_backend_webgpu_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_webgpu_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_webgpu_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ ggml_backend_webgpu_buffer_type_get_max_size,
|
||||
/* .get_alloc_size = */ ggml_backend_webgpu_buffer_type_get_alloc_size,
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ NULL, // defaults to false
|
||||
},
|
||||
/* .device = */
|
||||
dev,
|
||||
|
||||
@@ -383,13 +383,14 @@ static bool ggml_backend_zdnn_buffer_type_is_host(ggml_backend_buffer_type_t buf
|
||||
ggml_backend_buffer_type_t ggml_backend_zdnn_buffer_type(void) {
|
||||
static ggml_backend_buffer_type ggml_backend_buffer_type_zdnn = {
|
||||
/* .iface = */ {
|
||||
/* .get_name = */ ggml_backend_zdnn_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_zdnn_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_zdnn_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ NULL,
|
||||
/* .get_alloc_size = */ NULL, // defaults to ggml_nbytes
|
||||
/* .is_host = */ ggml_backend_zdnn_buffer_type_is_host,
|
||||
/* .get_name = */ ggml_backend_zdnn_buffer_type_get_name,
|
||||
/* .alloc_buffer = */ ggml_backend_zdnn_buffer_type_alloc_buffer,
|
||||
/* .alloc_buffer_n = */ NULL,
|
||||
/* .get_alignment = */ ggml_backend_zdnn_buffer_type_get_alignment,
|
||||
/* .get_max_size = */ NULL,
|
||||
/* .get_alloc_size = */ NULL, // defaults to ggml_nbytes
|
||||
/* .get_alloc_size_n = */ NULL,
|
||||
/* .is_host = */ ggml_backend_zdnn_buffer_type_is_host,
|
||||
},
|
||||
/* .device = */ &g_ggml_backend_zdnn_device,
|
||||
/* .context = */ NULL,
|
||||
|
||||
+241
-15
@@ -5,7 +5,6 @@
|
||||
#include "ggml.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <exception>
|
||||
#include <memory>
|
||||
#include <vector>
|
||||
|
||||
@@ -23,7 +22,8 @@ struct dummy_backend_context {
|
||||
ggml_backend backend;
|
||||
std::vector<ggml_backend_buffer_t> buffers;
|
||||
|
||||
bool custom_alloc_buffer_n_called = false;
|
||||
bool custom_alloc_buffer_n_called = false;
|
||||
bool custom_get_alloc_size_n_called = false;
|
||||
|
||||
size_t allocated_total() const {
|
||||
size_t n = 0;
|
||||
@@ -36,7 +36,7 @@ struct dummy_backend_context {
|
||||
|
||||
// ggml_backend_buffer_type interface
|
||||
|
||||
static const char * dummy_backend_buffer_type_get_name(ggml_backend_buffer_type_t) {
|
||||
static const char * dummy_backend_buffer_type_get_name(ggml_backend_buffer_type_t /*buft*/) {
|
||||
return "dummy_buffer_type";
|
||||
}
|
||||
|
||||
@@ -73,7 +73,16 @@ static ggml_backend_buffer_t dummy_backend_buffer_type_alloc_buffer_n_custom(
|
||||
return buffer;
|
||||
}
|
||||
|
||||
static bool dummy_backend_buffer_type_is_host(ggml_backend_buffer_type_t) {
|
||||
static size_t dummy_backend_buffer_type_get_alloc_size_n_custom(
|
||||
ggml_backend_buffer_type_t buft, ggml_tensor ** tensors, int n_tensors) {
|
||||
dummy_backend_context * ctx = (dummy_backend_context *) buft->context;
|
||||
GGML_UNUSED(tensors);
|
||||
GGML_UNUSED(n_tensors);
|
||||
ctx->custom_get_alloc_size_n_called = true;
|
||||
return 64;
|
||||
}
|
||||
|
||||
static bool dummy_backend_buffer_type_is_host(ggml_backend_buffer_type_t /*buft*/) {
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -87,29 +96,29 @@ static void dummy_backend_buffer_free_buffer(ggml_backend_buffer_t buffer) {
|
||||
ctx->buffers.erase(i);
|
||||
}
|
||||
|
||||
static void * dummy_backend_buffer_get_base(ggml_backend_buffer_t) {
|
||||
static void * dummy_backend_buffer_get_base(ggml_backend_buffer_t /*buft*/) {
|
||||
return alloc_base;
|
||||
}
|
||||
|
||||
static ggml_status dummy_backend_buffer_init_tensor(ggml_backend_buffer_t, ggml_tensor *) {
|
||||
static ggml_status dummy_backend_buffer_init_tensor(ggml_backend_buffer_t /*buft*/, ggml_tensor * /*x*/) {
|
||||
return GGML_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
static void dummy_backend_buffer_memset_tensor(ggml_backend_buffer_t, ggml_tensor *, uint8_t, size_t, size_t) {}
|
||||
static void dummy_backend_buffer_memset_tensor(ggml_backend_buffer_t /*buft*/, ggml_tensor * /*x*/, uint8_t, size_t, size_t) {}
|
||||
|
||||
static void dummy_backend_buffer_set_tensor(ggml_backend_buffer_t, ggml_tensor *, const void *, size_t, size_t) {}
|
||||
static void dummy_backend_buffer_set_tensor(ggml_backend_buffer_t /*buft*/, ggml_tensor * /*x*/, const void *, size_t, size_t) {}
|
||||
|
||||
static void dummy_backend_buffer_get_tensor(ggml_backend_buffer_t, const ggml_tensor *, void *, size_t, size_t) {}
|
||||
static void dummy_backend_buffer_get_tensor(ggml_backend_buffer_t /*buft*/, const ggml_tensor * /*x*/, void *, size_t, size_t) {}
|
||||
|
||||
static void dummy_backend_buffer_clear(ggml_backend_buffer_t, uint8_t) {}
|
||||
static void dummy_backend_buffer_clear(ggml_backend_buffer_t /*buft*/, uint8_t /*val*/) {}
|
||||
|
||||
// ggml_backend_device interface
|
||||
|
||||
static enum ggml_backend_dev_type dummy_backend_device_get_type(ggml_backend_dev_t) {
|
||||
static enum ggml_backend_dev_type dummy_backend_device_get_type(ggml_backend_dev_t /*dev*/) {
|
||||
return GGML_BACKEND_DEVICE_TYPE_CPU;
|
||||
}
|
||||
|
||||
static bool dummy_backend_device_supports_op(ggml_backend_dev_t, const ggml_tensor *) {
|
||||
static bool dummy_backend_device_supports_op(ggml_backend_dev_t /*dev*/, const ggml_tensor * /*x*/) {
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -119,7 +128,7 @@ static bool dummy_backend_device_supports_buft(ggml_backend_dev_t device, ggml_b
|
||||
|
||||
// ggml_backend interface
|
||||
|
||||
static const char * dummy_backend_get_name(ggml_backend_t) {
|
||||
static const char * dummy_backend_get_name(ggml_backend_t /*backend*/) {
|
||||
return "dummy_backend";
|
||||
}
|
||||
|
||||
@@ -244,7 +253,7 @@ static void check_all_allocated(ggml_cgraph * graph) {
|
||||
|
||||
static void check_max_size(ggml_context * ctx) {
|
||||
for (ggml_tensor * t = ggml_get_first_tensor(ctx); t; t = ggml_get_next_tensor(ctx, t)) {
|
||||
auto buft = ggml_backend_buffer_get_type(t->buffer);
|
||||
auto * buft = ggml_backend_buffer_get_type(t->buffer);
|
||||
size_t max_size = ggml_backend_buft_get_max_size(buft);
|
||||
size_t offset = (char *) t->data - (char *) ggml_backend_buffer_get_base(t->buffer);
|
||||
GGML_ASSERT(t->data >= ggml_backend_buffer_get_base(t->buffer));
|
||||
@@ -633,7 +642,7 @@ static void test_reallocation() {
|
||||
}
|
||||
}
|
||||
|
||||
static void test_backend_graph_optimize(ggml_backend_t, ggml_cgraph * graph, ggml_backend_graph_optimize_params * params) {
|
||||
static void test_backend_graph_optimize(ggml_backend_t /*backend*/, ggml_cgraph * graph, ggml_backend_graph_optimize_params * params) {
|
||||
GGML_ASSERT(graph->n_nodes == 3);
|
||||
params->add_alloc_dep(params->user_data, graph->nodes[0], graph->nodes[2]);
|
||||
}
|
||||
@@ -671,7 +680,14 @@ static void test_graph_optimize_alloc_dep() {
|
||||
// Check that the size reported by ggml_backend_alloc_ctx_tensors_from_buft_size
|
||||
// matches the actual size of the buffer allocated for the ctx tensors
|
||||
static ggml_backend_buffer_ptr check_size_matches(ggml_backend_buffer_type_t buft, ggml_context * ctx) {
|
||||
std::vector<ggml_tensor *> tensors;
|
||||
for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) {
|
||||
tensors.push_back(t);
|
||||
}
|
||||
|
||||
const size_t expected_size = ggml_backend_alloc_ctx_tensors_from_buft_size(ctx, buft);
|
||||
GGML_ASSERT(ggml_backend_buft_get_alloc_size_n(buft, tensors.data(), (int) tensors.size()) == expected_size);
|
||||
|
||||
ggml_backend_buffer_ptr buffer(ggml_backend_alloc_ctx_tensors_from_buft(ctx, buft));
|
||||
GGML_ASSERT((buffer != nullptr) == (expected_size != 0));
|
||||
if (buffer) {
|
||||
@@ -884,6 +900,205 @@ static void test_buft_alloc_buffer_n_custom_override() {
|
||||
GGML_ASSERT(x[1]->buffer == buffer.get());
|
||||
}
|
||||
|
||||
// Check that get_alloc_size_n predicts a single buffer allocation
|
||||
static void test_buft_get_alloc_size_n_single_buffer() {
|
||||
dummy_backend backend = dummy_backend_init(SIZE_MAX);
|
||||
auto [ctx, graph, ctx_ptr] = make_context();
|
||||
|
||||
ggml_tensor * x[2];
|
||||
x[0] = make_input_with_size(ctx, 8);
|
||||
x[1] = make_input_with_size(ctx, 8);
|
||||
assign_names(ctx);
|
||||
|
||||
ggml_tensor * tensors[2] = { x[0], x[1] };
|
||||
GGML_ASSERT(ggml_backend_buft_get_alloc_size_n(&backend.buffer_type, tensors, 2) == 16);
|
||||
|
||||
ggml_backend_buffer_ptr buffer(ggml_backend_buft_alloc_buffer_n(&backend.buffer_type, tensors, 2));
|
||||
GGML_ASSERT(buffer != nullptr);
|
||||
GGML_ASSERT(ggml_backend_buffer_get_size(buffer.get()) == 16);
|
||||
}
|
||||
|
||||
// Check that get_alloc_size_n accounts for splitting into multiple buffers
|
||||
static void test_buft_get_alloc_size_n_multi_buffer() {
|
||||
dummy_backend backend = dummy_backend_init(16);
|
||||
auto [ctx, graph, ctx_ptr] = make_context();
|
||||
|
||||
ggml_tensor * x[4];
|
||||
x[0] = make_input_with_size(ctx, 8);
|
||||
x[1] = make_input_with_size(ctx, 8);
|
||||
x[2] = make_input_with_size(ctx, 8);
|
||||
x[3] = make_input_with_size(ctx, 8);
|
||||
assign_names(ctx);
|
||||
|
||||
ggml_tensor * tensors[4] = { x[0], x[1], x[2], x[3] };
|
||||
GGML_ASSERT(ggml_backend_buft_get_alloc_size_n(&backend.buffer_type, tensors, 4) == 32);
|
||||
|
||||
ggml_backend_buffer_ptr buffer(ggml_backend_buft_alloc_buffer_n(&backend.buffer_type, tensors, 4));
|
||||
GGML_ASSERT(buffer != nullptr);
|
||||
GGML_ASSERT(ggml_backend_buffer_is_multi_buffer(buffer.get()));
|
||||
GGML_ASSERT(ggml_backend_buffer_get_size(buffer.get()) == 32);
|
||||
}
|
||||
|
||||
// Check that a tensor larger than max_size is still accounted for
|
||||
static void test_buft_get_alloc_size_n_single_tensor_exceeds_max() {
|
||||
dummy_backend backend = dummy_backend_init(8);
|
||||
auto [ctx, graph, ctx_ptr] = make_context();
|
||||
|
||||
ggml_tensor * x = make_input_with_size(ctx, 16);
|
||||
assign_names(ctx);
|
||||
|
||||
ggml_tensor * tensors[1] = { x };
|
||||
GGML_ASSERT(ggml_backend_buft_get_alloc_size_n(&backend.buffer_type, tensors, 1) == 16);
|
||||
|
||||
ggml_backend_buffer_ptr buffer(ggml_backend_buft_alloc_buffer_n(&backend.buffer_type, tensors, 1));
|
||||
GGML_ASSERT(buffer != nullptr);
|
||||
GGML_ASSERT(ggml_backend_buffer_get_size(buffer.get()) == 16);
|
||||
}
|
||||
|
||||
// Check that zero-size tensors report a total allocation size of 0
|
||||
static void test_buft_get_alloc_size_n_zero_size() {
|
||||
dummy_backend backend = dummy_backend_init(SIZE_MAX);
|
||||
auto [ctx, graph, ctx_ptr] = make_context();
|
||||
|
||||
ggml_tensor * x[2];
|
||||
x[0] = make_input_1d(ctx, 0);
|
||||
x[1] = make_input_1d(ctx, 0);
|
||||
assign_names(ctx);
|
||||
|
||||
ggml_tensor * tensors[2] = { x[0], x[1] };
|
||||
GGML_ASSERT(ggml_backend_buft_get_alloc_size_n(&backend.buffer_type, tensors, 2) == 0);
|
||||
GGML_ASSERT(ggml_backend_buft_alloc_buffer_n(&backend.buffer_type, tensors, 2) == nullptr);
|
||||
}
|
||||
|
||||
// Check that already-allocated tensors are not counted again
|
||||
static void test_buft_get_alloc_size_n_already_allocated() {
|
||||
dummy_backend backend = dummy_backend_init(SIZE_MAX);
|
||||
auto [ctx, graph, ctx_ptr] = make_context();
|
||||
|
||||
ggml_tensor * a[2];
|
||||
a[0] = make_input_with_size(ctx, 8);
|
||||
a[1] = make_input_with_size(ctx, 8);
|
||||
assign_names(ctx, "a");
|
||||
|
||||
ggml_backend_buffer_ptr buf_a(ggml_backend_alloc_ctx_tensors_from_buft(ctx, &backend.buffer_type));
|
||||
GGML_ASSERT(buf_a != nullptr);
|
||||
|
||||
ggml_tensor * b = make_input_with_size(ctx, 8);
|
||||
assign_names(ctx, "b");
|
||||
|
||||
ggml_tensor * tensors[3] = { a[0], a[1], b };
|
||||
GGML_ASSERT(ggml_backend_buft_get_alloc_size_n(&backend.buffer_type, tensors, 3) == 8);
|
||||
|
||||
ggml_backend_buffer_ptr buf_b(ggml_backend_buft_alloc_buffer_n(&backend.buffer_type, tensors, 3));
|
||||
GGML_ASSERT(buf_b != nullptr);
|
||||
GGML_ASSERT(backend.context->buffers.size() == 2);
|
||||
GGML_ASSERT(a[0]->buffer == buf_a.get());
|
||||
GGML_ASSERT(a[1]->buffer == buf_a.get());
|
||||
GGML_ASSERT(b->buffer == buf_b.get());
|
||||
GGML_ASSERT(ggml_backend_buffer_get_size(buf_b.get()) == 8);
|
||||
}
|
||||
|
||||
// Check that views do not add to the projected allocation size
|
||||
static void test_buft_get_alloc_size_n_views() {
|
||||
dummy_backend backend = dummy_backend_init(SIZE_MAX);
|
||||
auto [ctx, graph, ctx_ptr] = make_context();
|
||||
|
||||
ggml_tensor * base = make_input_1d(ctx, 4);
|
||||
ggml_tensor * view = ggml_view_1d(ctx, base, 2, 0);
|
||||
ggml_tensor * extra = make_input_1d(ctx, 2);
|
||||
assign_names(ctx);
|
||||
|
||||
ggml_tensor * tensors[3] = { base, view, extra };
|
||||
GGML_ASSERT(ggml_backend_buft_get_alloc_size_n(&backend.buffer_type, tensors, 3) == 24);
|
||||
|
||||
ggml_backend_buffer_ptr buffer(ggml_backend_buft_alloc_buffer_n(&backend.buffer_type, tensors, 3));
|
||||
GGML_ASSERT(buffer != nullptr);
|
||||
GGML_ASSERT(ggml_backend_buffer_get_size(buffer.get()) == 24);
|
||||
}
|
||||
|
||||
// Check that n_tensors == 0 is handled
|
||||
static void test_buft_get_alloc_size_n_n_tensors_zero() {
|
||||
dummy_backend backend = dummy_backend_init(SIZE_MAX);
|
||||
GGML_ASSERT(ggml_backend_buft_get_alloc_size_n(&backend.buffer_type, nullptr, 0) == 0);
|
||||
GGML_ASSERT(ggml_backend_buft_alloc_buffer_n(&backend.buffer_type, nullptr, 0) == nullptr);
|
||||
}
|
||||
|
||||
// Check that get_alloc_size_n respects the backend alignment
|
||||
static void test_buft_get_alloc_size_n_alignment() {
|
||||
dummy_backend backend = dummy_backend_init(SIZE_MAX, 16);
|
||||
auto [ctx, graph, ctx_ptr] = make_context();
|
||||
|
||||
ggml_tensor * x[2];
|
||||
x[0] = make_input_with_size(ctx, 4);
|
||||
x[1] = make_input_with_size(ctx, 4);
|
||||
assign_names(ctx);
|
||||
|
||||
ggml_tensor * tensors[2] = { x[0], x[1] };
|
||||
GGML_ASSERT(ggml_backend_buft_get_alloc_size_n(&backend.buffer_type, tensors, 2) == 32);
|
||||
|
||||
ggml_backend_buffer_ptr buffer(ggml_backend_buft_alloc_buffer_n(&backend.buffer_type, tensors, 2));
|
||||
GGML_ASSERT(buffer != nullptr);
|
||||
GGML_ASSERT(ggml_backend_buffer_get_size(buffer.get()) == 32);
|
||||
}
|
||||
|
||||
// Check that a backend-provided get_alloc_size_n implementation takes precedence over
|
||||
// the default one
|
||||
static void test_buft_get_alloc_size_n_custom_override() {
|
||||
dummy_backend backend = dummy_backend_init(SIZE_MAX);
|
||||
backend.buffer_type.iface.alloc_buffer_n = dummy_backend_buffer_type_alloc_buffer_n_custom;
|
||||
backend.buffer_type.iface.get_alloc_size_n = dummy_backend_buffer_type_get_alloc_size_n_custom;
|
||||
|
||||
auto [ctx, graph, ctx_ptr] = make_context();
|
||||
|
||||
ggml_tensor * x[2];
|
||||
x[0] = make_input_with_size(ctx, 8);
|
||||
x[1] = make_input_with_size(ctx, 8);
|
||||
assign_names(ctx);
|
||||
|
||||
ggml_tensor * tensors[2] = { x[0], x[1] };
|
||||
GGML_ASSERT(ggml_backend_buft_get_alloc_size_n(&backend.buffer_type, tensors, 2) == 64);
|
||||
GGML_ASSERT(backend.context->custom_get_alloc_size_n_called);
|
||||
|
||||
ggml_backend_buffer_ptr buffer(ggml_backend_buft_alloc_buffer_n(&backend.buffer_type, tensors, 2));
|
||||
GGML_ASSERT(buffer != nullptr);
|
||||
GGML_ASSERT(backend.context->custom_alloc_buffer_n_called);
|
||||
GGML_ASSERT(ggml_backend_buffer_get_size(buffer.get()) == 64);
|
||||
}
|
||||
|
||||
// Check that the custom get_alloc_size_n override is used by the context size helper
|
||||
static void test_buft_get_alloc_size_n_custom_override_ctx_size() {
|
||||
dummy_backend backend = dummy_backend_init(SIZE_MAX);
|
||||
backend.buffer_type.iface.get_alloc_size_n = dummy_backend_buffer_type_get_alloc_size_n_custom;
|
||||
|
||||
auto [ctx, graph, ctx_ptr] = make_context();
|
||||
|
||||
ggml_tensor * x[2];
|
||||
x[0] = make_input_with_size(ctx, 8);
|
||||
x[1] = make_input_with_size(ctx, 8);
|
||||
assign_names(ctx);
|
||||
|
||||
GGML_UNUSED(x);
|
||||
|
||||
GGML_ASSERT(ggml_backend_alloc_ctx_tensors_from_buft_size(ctx, &backend.buffer_type) == 64);
|
||||
}
|
||||
|
||||
// Check that querying the projected allocation size does not allocate anything
|
||||
static void test_buft_get_alloc_size_n_does_not_allocate() {
|
||||
dummy_backend backend = dummy_backend_init(SIZE_MAX);
|
||||
auto [ctx, graph, ctx_ptr] = make_context();
|
||||
|
||||
ggml_tensor * x[2];
|
||||
x[0] = make_input_with_size(ctx, 8);
|
||||
x[1] = make_input_with_size(ctx, 8);
|
||||
assign_names(ctx);
|
||||
|
||||
ggml_tensor * tensors[2] = { x[0], x[1] };
|
||||
GGML_ASSERT(ggml_backend_buft_get_alloc_size_n(&backend.buffer_type, tensors, 2) == 16);
|
||||
GGML_ASSERT(backend.context->buffers.empty());
|
||||
GGML_ASSERT(x[0]->data == nullptr);
|
||||
GGML_ASSERT(x[1]->data == nullptr);
|
||||
}
|
||||
|
||||
static void run(const char * name, void (*f)()) {
|
||||
printf("%s ", name);
|
||||
fflush(stdout);
|
||||
@@ -913,5 +1128,16 @@ int main() {
|
||||
run("test_buft_alloc_buffer_n_views", test_buft_alloc_buffer_n_views);
|
||||
run("test_alloc_ctx_tensors_from_buft_size_matches", test_alloc_ctx_tensors_from_buft_size_matches);
|
||||
run("test_buft_alloc_buffer_n_custom_override", test_buft_alloc_buffer_n_custom_override);
|
||||
run("test_buft_get_alloc_size_n_single_buffer", test_buft_get_alloc_size_n_single_buffer);
|
||||
run("test_buft_get_alloc_size_n_multi_buffer", test_buft_get_alloc_size_n_multi_buffer);
|
||||
run("test_buft_get_alloc_size_n_single_tensor_exceeds_max", test_buft_get_alloc_size_n_single_tensor_exceeds_max);
|
||||
run("test_buft_get_alloc_size_n_zero_size", test_buft_get_alloc_size_n_zero_size);
|
||||
run("test_buft_get_alloc_size_n_already_allocated", test_buft_get_alloc_size_n_already_allocated);
|
||||
run("test_buft_get_alloc_size_n_views", test_buft_get_alloc_size_n_views);
|
||||
run("test_buft_get_alloc_size_n_n_tensors_zero", test_buft_get_alloc_size_n_n_tensors_zero);
|
||||
run("test_buft_get_alloc_size_n_alignment", test_buft_get_alloc_size_n_alignment);
|
||||
run("test_buft_get_alloc_size_n_custom_override", test_buft_get_alloc_size_n_custom_override);
|
||||
run("test_buft_get_alloc_size_n_custom_override_ctx_size", test_buft_get_alloc_size_n_custom_override_ctx_size);
|
||||
run("test_buft_get_alloc_size_n_does_not_allocate", test_buft_get_alloc_size_n_does_not_allocate);
|
||||
return 0;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user