Commit a1ded76 for stable-diffusion.cpp
commit a1ded76da5818803fca97a3b433669ef727d32cf
Author: leejet <leejet714@gmail.com>
Date: Tue Oct 6 20:10:36 2026 +0800
chore: resolve MSVC warnings in ggml extensions
diff --git a/src/core/ggml_extend.cpp b/src/core/ggml_extend.cpp
index d87353d..d16b92a 100644
--- a/src/core/ggml_extend.cpp
+++ b/src/core/ggml_extend.cpp
@@ -223,13 +223,13 @@ ggml_tensor* ggml_ext_linear(ggml_context* ctx,
x = ggml_reshape_2d(ctx, x, x->ne[0], x->ne[1] * x->ne[2] * x->ne[3]);
x = ggml_mul_mat(ctx, w, x);
if (force_prec_f32) {
- ggml_mul_mat_set_prec(x, GGML_PREC_F32);
+ ggml_prec_set_acc(x, GGML_PREC_F32);
}
x = ggml_reshape_4d(ctx, x, x->ne[0], x->ne[1] / ne2 / ne3, ne2, ne3);
} else {
x = ggml_mul_mat(ctx, w, x);
if (force_prec_f32) {
- ggml_mul_mat_set_prec(x, GGML_PREC_F32);
+ ggml_prec_set_acc(x, GGML_PREC_F32);
}
}
if (scale != 1.f) {
@@ -488,7 +488,7 @@ ggml_tensor* ggml_ext_conv_3d(ggml_context* ctx,
x = ggml_mul_mat(ctx,
ggml_reshape_2d(ctx, im2col, im2col->ne[0], im2col->ne[3] * im2col->ne[2] * im2col->ne[1]),
ggml_reshape_2d(ctx, w, w->ne[0] * w->ne[1] * w->ne[2] * IC, OC));
- ggml_mul_mat_set_prec(x, GGML_PREC_F32);
+ ggml_prec_set_acc(x, GGML_PREC_F32);
int64_t OD = im2col->ne[3] / N;
x = ggml_reshape_4d(ctx, x, im2col->ne[1] * im2col->ne[2], OD, N, OC);
@@ -683,8 +683,8 @@ ggml_tensor* ggml_ext_attention_ext(ggml_context* ctx,
v_in->type == GGML_TYPE_F32 && sd_backend_supports_cuda_mma(backend);
if (pad_head) {
// CUDA FA MMA starts at 64 channels; keep the original head's attention scale.
- q_in = ggml_pad(ctx, q_in, 64 - d_head, 0, 0, 0);
- k_in = ggml_pad(ctx, k_in, 64 - d_head, 0, 0, 0);
+ q_in = ggml_pad(ctx, q_in, static_cast<int>(64 - d_head), 0, 0, 0);
+ k_in = ggml_pad(ctx, k_in, static_cast<int>(64 - d_head), 0, 0, 0);
}
if (kv_scale != 1.0f) {
k_in = ggml_ext_scale(ctx, k_in, kv_scale);
@@ -694,7 +694,7 @@ ggml_tensor* ggml_ext_attention_ext(ggml_context* ctx,
v_in = ggml_ext_cont(ctx, ggml_permute(ctx, v_in, 0, 2, 1, 3));
v_in = ggml_reshape_3d(ctx, v_in, d_head, L_k, n_kv_head * N);
if (pad_head) {
- v_in = ggml_pad(ctx, v_in, 64 - d_head, 0, 0, 0);
+ v_in = ggml_pad(ctx, v_in, static_cast<int>(64 - d_head), 0, 0, 0);
}
if (kv_scale != 1.0f) {
v_in = ggml_ext_scale(ctx, v_in, kv_scale);
@@ -721,7 +721,7 @@ ggml_tensor* ggml_ext_attention_ext(ggml_context* ctx,
if (!ggml_backend_supports_op(backend, out)) {
return nullptr;
}
- ggml_flash_attn_ext_set_prec(out, GGML_PREC_F32);
+ ggml_prec_set_acc(out, GGML_PREC_F32);
if (kv_scale != 1.0f) {
out = ggml_ext_scale(ctx, out, 1.0f / kv_scale);
}
@@ -742,9 +742,9 @@ ggml_tensor* ggml_ext_attention_ext(ggml_context* ctx,
}
if (padded_head != d_head) {
// Keep the original head's softmax scale when padding for the CUDA kernel.
- q_in = ggml_pad(ctx, q_in, padded_head - d_head, 0, 0, 0);
- k_in = ggml_pad(ctx, k_in, padded_head - d_head, 0, 0, 0);
- v_in = ggml_pad(ctx, v_in, padded_head - d_head, 0, 0, 0);
+ q_in = ggml_pad(ctx, q_in, static_cast<int>(padded_head - d_head), 0, 0, 0);
+ k_in = ggml_pad(ctx, k_in, static_cast<int>(padded_head - d_head), 0, 0, 0);
+ v_in = ggml_pad(ctx, v_in, static_cast<int>(padded_head - d_head), 0, 0, 0);
}
if (kv_scale != 1.0f) {
k_in = ggml_ext_scale(ctx, k_in, kv_scale);
@@ -797,7 +797,7 @@ ggml_tensor* ggml_ext_attention_ext(ggml_context* ctx,
v = ggml_reshape_3d(ctx, v, L_k, d_head, n_kv_head * N); // [N * n_kv_head, d_head, L_k]
auto kq = ggml_mul_mat(ctx, k, q); // [N * n_head, L_q, L_k]
- ggml_mul_mat_set_prec(kq, GGML_PREC_F32);
+ ggml_prec_set_acc(kq, GGML_PREC_F32);
kq = ggml_scale_inplace(ctx, kq, scale);
if (mask) {
kq = ggml_add_inplace(ctx, kq, mask);