load hadamard_block_size from config (#3797)

2026-04-23 17:11:21 +08:00 · 2025-09-05 17:07:58 +08:00
parent 41aee08982
commit 2cf55168ca
10 changed files with 60 additions and 30 deletions
@@ -872,16 +872,14 @@ void MoeFastHardamardWrapper(const T *x_data,
                          const int64_t dim,
                          const int num_max_tokens_per_expert,
                          bool used_in_ep_low_latency,
+                          const int hadamard_block_size,
                          OutT* out,
                          cudaStream_t &stream) {
  bool FLAGS_hardamard_use_diagonal_block_matrix = true;

-  static const char* FLAGS_hardamard_moe_block_size = std::getenv("FLAGS_hardamard_moe_block_size");
-  static const int32_t hardamard_moe_block_size = FLAGS_hardamard_moe_block_size != nullptr ?
-    stoi(std::string(FLAGS_hardamard_moe_block_size)) : 512;
  constexpr int kThreads = 128;
  if (FLAGS_hardamard_use_diagonal_block_matrix) {
-    const int VecSize = hardamard_moe_block_size / kThreads; // 128 / 128 = 1
+    const int VecSize = hadamard_block_size / kThreads;
    const int logN = int(ceil(std::log2(kThreads * VecSize)));
    constexpr int kNChunks = 1;
    DISPATCH_SP_VS(VecSize, VEC_SIZE, {
@@ -991,6 +989,7 @@ template void MoeFastHardamardWrapper<phi::dtype::float16, phi::dtype::float16>(
  const int64_t dim,
  const int num_max_tokens_per_expert,
  bool used_in_ep_low_latency,
+  const int hadamard_block_size,
  phi::dtype::float16 *out,
  cudaStream_t &stream
 );
@@ -1009,6 +1008,7 @@ template void MoeFastHardamardWrapper<phi::dtype::float16, int8_t>(
  const int64_t dim,
  const int num_max_tokens_per_expert,
  bool used_in_ep_low_latency,
+  const int hadamard_block_size,
  int8_t *out,
  cudaStream_t &stream
 );
@@ -1027,6 +1027,7 @@ template void MoeFastHardamardWrapper<phi::dtype::bfloat16, phi::dtype::bfloat16
  const int64_t dim,
  const int num_max_tokens_per_expert,
  bool used_in_ep_low_latency,
+  const int hadamard_block_size,
  phi::dtype::bfloat16 *out,
  cudaStream_t &stream
 );
@@ -1045,6 +1046,7 @@ template void MoeFastHardamardWrapper<phi::dtype::bfloat16, int8_t>(
  const int64_t dim,
  const int num_max_tokens_per_expert,
  bool used_in_ep_low_latency,
+  const int hadamard_block_size,
  int8_t *out,
  cudaStream_t &stream
 );