hexagon: support I32 CPY and CONT (#29379)

Assisted-by: OpenCode
This commit is contained in:
kurquhar
2026-09-24 10:48:57 -07:00
committed by GitHub
parent a72e04abe0
commit 97a418bdf4
2 changed files with 28 additions and 9 deletions
+11 -7
View File
@@ -7089,18 +7089,21 @@ static bool ggml_hexagon_supported_cpy(const struct ggml_hexagon_session * sess,
const struct ggml_tensor * src0 = op->src[0];
const struct ggml_tensor * dst = op;
// for now we can do f32 -> f16 and f16 -> f32 (without reshaping)
if (src0->type != GGML_TYPE_F32 && src0->type != GGML_TYPE_F16) return false;
if ( dst->type != GGML_TYPE_F32 && dst->type != GGML_TYPE_F16) return false;
if (src0->type != GGML_TYPE_F32 && src0->type != GGML_TYPE_F16 &&
src0->type != GGML_TYPE_I32) return false;
if (dst->type != GGML_TYPE_F32 && dst->type != GGML_TYPE_F16 &&
dst->type != GGML_TYPE_I32) return false;
const bool sametype = (src0->type == dst->type);
const bool transposed = ggml_is_transposed(src0) || ggml_is_transposed(dst);
const bool sameshape = !transposed && ggml_are_same_shape(src0, dst);
// can handle any shape and any same-type (pretty slow if reshaping is required)
// Same-type copies also support I32.
if (sametype) return true;
// cannot handle re-shaping and type conversion at the same time
// Type conversion is only supported between F32 and F16.
if (src0->type == GGML_TYPE_I32 || dst->type == GGML_TYPE_I32) return false;
if (!sameshape) return false;
return true;
@@ -7110,8 +7113,9 @@ static bool ggml_hexagon_supported_cont(const struct ggml_hexagon_session * sess
GGML_UNUSED(sess);
const struct ggml_tensor * src0 = op->src[0];
// CONT is same-type only, supports f32 and f16
if (src0->type != GGML_TYPE_F32 && src0->type != GGML_TYPE_F16) return false;
// CONT is same-type only and supports F32, F16, and I32.
if (src0->type != GGML_TYPE_F32 && src0->type != GGML_TYPE_F16 &&
src0->type != GGML_TYPE_I32) return false;
return true;
}
+17 -2
View File
@@ -140,6 +140,7 @@ static void cpy_thread_##NAME##_sameshape(unsigned int nth, unsigned int ith, vo
DEFINE_CPY_SAMESHAPE(f32, float, 4)
DEFINE_CPY_SAMESHAPE(f16, __fp16, 2)
DEFINE_CPY_SAMESHAPE(i32, int32_t, 4)
#define DEFINE_CPY_RESHAPE(NAME, ELEM_TYPE, ELEM_SIZE) \
static void cpy_thread_##NAME##_reshape(unsigned int nth, unsigned int ith, void * data) { \
@@ -226,6 +227,7 @@ static void cpy_thread_##NAME##_reshape(unsigned int nth, unsigned int ith, void
DEFINE_CPY_RESHAPE(f32, float, 4)
DEFINE_CPY_RESHAPE(f16, __fp16, 2)
DEFINE_CPY_RESHAPE(i32, int32_t, 4)
static void cpy_thread_f16_f32_sameshape(unsigned int nth, unsigned int ith, void * data) {
struct htp_copy_context * ct = (struct htp_copy_context *) data;
@@ -370,6 +372,7 @@ static int exec_cpy(struct htp_ops_context * octx, bool * use_dma) {
switch (src0->type) {
case HTP_TYPE_F32: ct.src0_type_size = 4; ct.src0_block_size = 1; ct.src0_blocks_per_row = ne00 / 1; break;
case HTP_TYPE_F16: ct.src0_type_size = 2; ct.src0_block_size = 1; ct.src0_blocks_per_row = ne00 / 1; break;
case HTP_TYPE_I32: ct.src0_type_size = 4; ct.src0_block_size = 1; ct.src0_blocks_per_row = ne00 / 1; break;
default:
return HTP_STATUS_NO_SUPPORT;
}
@@ -377,6 +380,7 @@ static int exec_cpy(struct htp_ops_context * octx, bool * use_dma) {
switch (dst->type) {
case HTP_TYPE_F32: ct.dst_type_size = 4; ct.dst_block_size = 1; ct.dst_blocks_per_row = ne0 / 1; break;
case HTP_TYPE_F16: ct.dst_type_size = 2; ct.dst_block_size = 1; ct.dst_blocks_per_row = ne0 / 1; break;
case HTP_TYPE_I32: ct.dst_type_size = 4; ct.dst_block_size = 1; ct.dst_blocks_per_row = ne0 / 1; break;
default:
return HTP_STATUS_NO_SUPPORT;
}
@@ -436,7 +440,12 @@ static int exec_cpy(struct htp_ops_context * octx, bool * use_dma) {
} else {
work_queue_func_t copy_fun = NULL;
if (sametype) {
copy_fun = (src0->type == HTP_TYPE_F32) ? cpy_thread_f32_sameshape : cpy_thread_f16_sameshape;
switch (src0->type) {
case HTP_TYPE_F32: copy_fun = cpy_thread_f32_sameshape; break;
case HTP_TYPE_F16: copy_fun = cpy_thread_f16_sameshape; break;
case HTP_TYPE_I32: copy_fun = cpy_thread_i32_sameshape; break;
default: return HTP_STATUS_NO_SUPPORT;
}
} else if (dst->type == HTP_TYPE_F16 && src0->type == HTP_TYPE_F32) {
copy_fun = cpy_thread_f16_f32_sameshape;
} else if (dst->type == HTP_TYPE_F32 && src0->type == HTP_TYPE_F16) {
@@ -482,7 +491,13 @@ static int exec_cpy(struct htp_ops_context * octx, bool * use_dma) {
ct.nelem = nelem;
ct.elem_per_thread = fastdiv(nelem + n_threads - 1, &octx->n_threads_div);
work_queue_func_t copy_fun = (src0->type == HTP_TYPE_F32) ? cpy_thread_f32_reshape : cpy_thread_f16_reshape;
work_queue_func_t copy_fun = NULL;
switch (src0->type) {
case HTP_TYPE_F32: copy_fun = cpy_thread_f32_reshape; break;
case HTP_TYPE_F16: copy_fun = cpy_thread_f16_reshape; break;
case HTP_TYPE_I32: copy_fun = cpy_thread_i32_reshape; break;
default: return HTP_STATUS_NO_SUPPORT;
}
work_queue_run(octx->ctx->work_queue, copy_fun, &ct, n_threads);
} else {
return HTP_STATUS_NO_SUPPORT;