Revert "musa: update compile flags (ggml-org#16265)"

Nexesenex · Nexesenex · commit a7b2d20ee1bb · 2025-10-07T17:53:39.000+02:00
This reverts commit 91a2a56.
diff --git a/ggml/src/ggml-cuda/fattn-vec.cuh b/ggml/src/ggml-cuda/fattn-vec.cuh
@@ -535,6 +535,8 @@ void ggml_cuda_flash_attn_ext_vec_case(ggml_backend_cuda_context & ctx, ggml_ten
     float logit_softcap;
     memcpy(&logit_softcap, (const float *) KQV->op_params + 2, sizeof(float));
 
+    const int cc = ggml_cuda_info().devices[ggml_cuda_get_device()].cc;
+
     if (Q->ne[1] == 1) {
         constexpr int cols_per_block = 1;
         if (logit_softcap == 0.0f) {
diff --git a/ggml/src/ggml-cuda/topk-moe.cu b/ggml/src/ggml-cuda/topk-moe.cu
@@ -13,7 +13,7 @@
 
     It is intended as fusion of softmax->top-k->get_rows pipeline for MoE models
 */
-template <int n_experts, bool with_norm>
+template <size_t n_experts, bool with_norm>
 __launch_bounds__(4 * WARP_SIZE, 1) __global__ void topk_moe_cuda(const float * logits,
                                                                   float *       weights,
                                                                   int32_t *     ids,
@@ -204,6 +204,8 @@ void ggml_cuda_op_topk_moe(ggml_backend_cuda_context & ctx,
 
     GGML_ASSERT(ids->nb[1] / ggml_type_size(ids->type) == (size_t) n_experts);
 
+    cudaStream_t stream = ctx.stream();
+
     const int n_expert_used = weights->ne[1];
 
     if (with_norm) {