Subject: [PATCH] HIP: pass packed bf16 operands to LLVM 23 WMMA builtins

LLVM 23 expects the gfx11 and gfx12 BF16 WMMA operands as vectors of
16-bit integer storage, rather than vectors of __bf16 values.

--- a/ggml/src/ggml-cuda/mma.cuh
+++ b/ggml/src/ggml-cuda/mma.cuh
@@ -1260,13 +1260,13 @@
 #if defined(AMD_WMMA_AVAILABLE)
 #if defined(RDNA4)
-        using bf16x8_t = __attribute__((ext_vector_type(8))) __bf16;
+        using bf16x8_t = __attribute__((ext_vector_type(8))) short;
         using floatx8_t = __attribute__((ext_vector_type(8))) float;
         floatx8_t& acc_frag = reinterpret_cast<floatx8_t&>(D.x[0]);
         const bf16x8_t& a_frag = reinterpret_cast<const bf16x8_t&>(A.x[0]);
         const bf16x8_t& b_frag = reinterpret_cast<const bf16x8_t&>(B.x[0]);
         acc_frag = __builtin_amdgcn_wmma_f32_16x16x16_bf16_w32_gfx12(a_frag, b_frag, acc_frag);
 #elif defined(RDNA3)
-        using bf16x16_t = __attribute__((ext_vector_type(16))) __bf16;
+        using bf16x16_t = __attribute__((ext_vector_type(16))) short;
         using floatx8_t = __attribute__((ext_vector_type(8))) float;
         floatx8_t& acc_frag = reinterpret_cast<floatx8_t&>(D.x[0]);
         const bf16x16_t& a_frag = reinterpret_cast<const bf16x16_t&>(A.x[0]);
