Fix thread group for large arrays (#1543)

awni · web-flow · commit 884af42da21d · 2024-10-30T16:25:12.000-07:00
* fix thread group for large arrays

* comment

* one more
diff --git a/mlx/backend/metal/binary.cpp b/mlx/backend/metal/binary.cpp
@@ -1,5 +1,4 @@
 // Copyright © 2024 Apple Inc.
-
 #include "mlx/backend/common/binary.h"
 #include "mlx/backend/metal/device.h"
 #include "mlx/backend/metal/kernels.h"
@@ -110,6 +109,7 @@ void binary_op_gpu_inplace(
     compute_encoder.set_output_array(outputs[1], arg_idx++);
   }
 
+  auto thread_group_size = kernel->maxTotalThreadsPerThreadgroup();
   if (bopt == BinaryOpType::General) {
     // Launch up to 3D grid of threads
     size_t dim0 = ndim > 0 ? shape[ndim - 1] : 1;
@@ -132,7 +132,6 @@ void binary_op_gpu_inplace(
           strides_b.data(), ndim * sizeof(size_t), arg_idx++);
     }
 
-    NS::UInteger thread_group_size = kernel->maxTotalThreadsPerThreadgroup();
     if (thread_group_size != 1024) {
       throw std::runtime_error("[Metal::binary] Must use 1024 sized block");
     }
@@ -142,13 +141,12 @@ void binary_op_gpu_inplace(
   } else {
     // Launch a 1D or 2D grid of threads
     size_t nthreads = out.data_size();
-    MTL::Size grid_dims = use_2d ? get_2d_grid_dims(out.shape(), out.strides())
-                                 : MTL::Size(nthreads, 1, 1);
-    NS::UInteger thread_group_size = kernel->maxTotalThreadsPerThreadgroup();
     if (thread_group_size > nthreads) {
       thread_group_size = nthreads;
     }
     MTL::Size group_dims = MTL::Size(thread_group_size, 1, 1);
+    MTL::Size grid_dims = use_2d ? get_2d_grid_dims(out.shape(), out.strides())
+                                 : MTL::Size(nthreads, 1, 1);
     compute_encoder.dispatchThreads(grid_dims, group_dims);
   }
 }
diff --git a/mlx/backend/metal/compiled.cpp b/mlx/backend/metal/compiled.cpp
@@ -421,11 +421,12 @@ void Compiled::eval_gpu(
   // Launch the kernel
   if (contiguous) {
     size_t nthreads = outputs[0].data_size();
+    MTL::Size group_dims(
+        std::min(nthreads, kernel->maxTotalThreadsPerThreadgroup()), 1, 1);
+
     MTL::Size grid_dims = use_2d
         ? get_2d_grid_dims(outputs[0].shape(), outputs[0].strides())
         : MTL::Size(nthreads, 1, 1);
-    MTL::Size group_dims(
-        std::min(nthreads, kernel->maxTotalThreadsPerThreadgroup()), 1, 1);
     compute_encoder.dispatchThreads(grid_dims, group_dims);
   } else {
     size_t dim0 = ndim > 0 ? shape[ndim - 1] : 1;
diff --git a/mlx/backend/metal/copy.cpp b/mlx/backend/metal/copy.cpp
@@ -120,6 +120,7 @@ void copy_gpu_inplace(
   compute_encoder.set_input_array(donate_in ? out : in, 0, inp_offset);
   compute_encoder.set_output_array(out, 1, out_offset);
 
+  auto thread_group_size = kernel->maxTotalThreadsPerThreadgroup();
   if (ctype == CopyType::General || ctype == CopyType::GeneralGeneral) {
     std::vector<int64_t> strides_in{strides_in_.begin(), strides_in_.end()};
     std::vector<int64_t> strides_out{strides_out_.begin(), strides_out_.end()};
@@ -145,7 +146,6 @@ void copy_gpu_inplace(
     }
 
     // NB assuming thread_group_size is a power of 2 larger than 32 x 32
-    NS::UInteger thread_group_size = kernel->maxTotalThreadsPerThreadgroup();
     if (thread_group_size != 1024) {
       throw std::runtime_error("[Metal::copy] Must use 1024 sized block");
     }
@@ -155,13 +155,12 @@ void copy_gpu_inplace(
     compute_encoder.dispatchThreads(grid_dims, group_dims);
   } else {
     size_t nthreads = out.data_size();
-    MTL::Size grid_dims = use_2d ? get_2d_grid_dims(out.shape(), out.strides())
-                                 : MTL::Size(nthreads, 1, 1);
-    NS::UInteger thread_group_size = kernel->maxTotalThreadsPerThreadgroup();
     if (thread_group_size > nthreads) {
       thread_group_size = nthreads;
     }
     MTL::Size group_dims = MTL::Size(thread_group_size, 1, 1);
+    MTL::Size grid_dims = use_2d ? get_2d_grid_dims(out.shape(), out.strides())
+                                 : MTL::Size(nthreads, 1, 1);
     compute_encoder.dispatchThreads(grid_dims, group_dims);
   }
 }
@@ -205,14 +204,14 @@ void fill_gpu(const array& val, array& out, const Stream& s) {
   compute_encoder.set_input_array(val, 0);
   compute_encoder.set_output_array(out, 1);
 
+  auto thread_group_size = kernel->maxTotalThreadsPerThreadgroup();
   size_t nthreads = out.data_size();
-  MTL::Size grid_dims = use_2d ? get_2d_grid_dims(out.shape(), out.strides())
-                               : MTL::Size(nthreads, 1, 1);
-  NS::UInteger thread_group_size = kernel->maxTotalThreadsPerThreadgroup();
   if (thread_group_size > nthreads) {
     thread_group_size = nthreads;
   }
   MTL::Size group_dims = MTL::Size(thread_group_size, 1, 1);
+  MTL::Size grid_dims = use_2d ? get_2d_grid_dims(out.shape(), out.strides())
+                               : MTL::Size(nthreads, 1, 1);
   compute_encoder.dispatchThreads(grid_dims, group_dims);
 }
 
diff --git a/mlx/backend/metal/ternary.cpp b/mlx/backend/metal/ternary.cpp
@@ -72,6 +72,7 @@ void ternary_op_gpu_inplace(
   compute_encoder.set_input_array(donate_c ? out : c, 2);
   compute_encoder.set_output_array(out, 3);
 
+  auto thread_group_size = kernel->maxTotalThreadsPerThreadgroup();
   if (topt == TernaryOpType::General) {
     // Launch up to 3D grid of threads
     size_t dim0 = ndim > 0 ? shape[ndim - 1] : 1;
@@ -93,7 +94,6 @@ void ternary_op_gpu_inplace(
       compute_encoder->setBytes(strides_c.data(), ndim * sizeof(size_t), 6);
     }
 
-    NS::UInteger thread_group_size = kernel->maxTotalThreadsPerThreadgroup();
     if (thread_group_size != 1024) {
       throw std::runtime_error("[Metal::ternary] Must use 1024 sized block");
     }
@@ -103,13 +103,12 @@ void ternary_op_gpu_inplace(
   } else {
     // Launch a 1D or 2D grid of threads
     size_t nthreads = out.data_size();
-    MTL::Size grid_dims = use_2d ? get_2d_grid_dims(out.shape(), out.strides())
-                                 : MTL::Size(nthreads, 1, 1);
-    NS::UInteger thread_group_size = kernel->maxTotalThreadsPerThreadgroup();
     if (thread_group_size > nthreads) {
       thread_group_size = nthreads;
     }
     MTL::Size group_dims = MTL::Size(thread_group_size, 1, 1);
+    MTL::Size grid_dims = use_2d ? get_2d_grid_dims(out.shape(), out.strides())
+                                 : MTL::Size(nthreads, 1, 1);
     compute_encoder.dispatchThreads(grid_dims, group_dims);
   }
 }
diff --git a/mlx/backend/metal/unary.cpp b/mlx/backend/metal/unary.cpp
@@ -47,9 +47,7 @@ void unary_op_gpu_inplace(
   kernel_name += "_" + op + type_to_name(in) + type_to_name(out);
   auto kernel = get_unary_kernel(d, kernel_name, in.dtype(), out.dtype(), op);
 
-  MTL::Size grid_dims = use_2d ? get_2d_grid_dims(in.shape(), in.strides())
-                               : MTL::Size(nthreads, 1, 1);
-  NS::UInteger thread_group_size = kernel->maxTotalThreadsPerThreadgroup();
+  auto thread_group_size = kernel->maxTotalThreadsPerThreadgroup();
   auto& compute_encoder = d.get_command_encoder(s.index);
   compute_encoder->setComputePipelineState(kernel);
   compute_encoder.set_input_array(
@@ -75,6 +73,8 @@ void unary_op_gpu_inplace(
       thread_group_size = nthreads;
     }
     MTL::Size group_dims = MTL::Size(thread_group_size, 1, 1);
+    MTL::Size grid_dims = use_2d ? get_2d_grid_dims(out.shape(), out.strides())
+                                 : MTL::Size(nthreads, 1, 1);
     compute_encoder.dispatchThreads(grid_dims, group_dims);
   }
 }
diff --git a/mlx/backend/metal/utils.cpp b/mlx/backend/metal/utils.cpp
@@ -103,6 +103,9 @@ MTL::Size get_2d_grid_dims(
   if (grid_y > UINT32_MAX || grid_x > UINT32_MAX) {
     throw std::runtime_error("Unable to safely factor shape.");
   }
+  if (grid_y > grid_x) {
+    std::swap(grid_x, grid_y);
+  }
   return MTL::Size(
       static_cast<uint32_t>(grid_x), static_cast<uint32_t>(grid_y), 1);
 }
@@ -145,6 +148,9 @@ MTL::Size get_2d_grid_dims(
   if (grid_y > UINT32_MAX || grid_x > UINT32_MAX || divisor > 1) {
     throw std::runtime_error("Unable to safely factor shape.");
   }
+  if (grid_y > grid_x) {
+    std::swap(grid_x, grid_y);
+  }
   return MTL::Size(
       static_cast<uint32_t>(grid_x), static_cast<uint32_t>(grid_y), 1);
 }

Original file line number	Diff line number	Diff line change
`@@ -1,5 +1,4 @@`
`1`	`1`	`// Copyright © 2024 Apple Inc.`
`2`		`-`
`3`	`2`	`#include "mlx/backend/common/binary.h"`
`4`	`3`	`#include "mlx/backend/metal/device.h"`
`5`	`4`	`#include "mlx/backend/metal/kernels.h"`
`@@ -110,6 +109,7 @@ void binary_op_gpu_inplace(`
`110`	`109`	`compute_encoder.set_output_array(outputs[1], arg_idx++);`
`111`	`110`	`}`
`112`	`111`
	`112`	`+ auto thread_group_size = kernel->maxTotalThreadsPerThreadgroup();`
`113`	`113`	`if (bopt == BinaryOpType::General) {`
`114`	`114`	`// Launch up to 3D grid of threads`
`115`	`115`	`size_t dim0 = ndim > 0 ? shape[ndim - 1] : 1;`
`@@ -132,7 +132,6 @@ void binary_op_gpu_inplace(`
`132`	`132`	`strides_b.data(), ndim * sizeof(size_t), arg_idx++);`
`133`	`133`	`}`
`134`	`134`
`135`		`- NS::UInteger thread_group_size = kernel->maxTotalThreadsPerThreadgroup();`
`136`	`135`	`if (thread_group_size != 1024) {`
`137`	`136`	`throw std::runtime_error("[Metal::binary] Must use 1024 sized block");`
`138`	`137`	`}`
`@@ -142,13 +141,12 @@ void binary_op_gpu_inplace(`
`142`	`141`	`} else {`
`143`	`142`	`// Launch a 1D or 2D grid of threads`
`144`	`143`	`size_t nthreads = out.data_size();`
`145`		`- MTL::Size grid_dims = use_2d ? get_2d_grid_dims(out.shape(), out.strides())`
`146`		`- : MTL::Size(nthreads, 1, 1);`
`147`		`- NS::UInteger thread_group_size = kernel->maxTotalThreadsPerThreadgroup();`
`148`	`144`	`if (thread_group_size > nthreads) {`
`149`	`145`	`thread_group_size = nthreads;`
`150`	`146`	`}`
`151`	`147`	`MTL::Size group_dims = MTL::Size(thread_group_size, 1, 1);`
	`148`	`+ MTL::Size grid_dims = use_2d ? get_2d_grid_dims(out.shape(), out.strides())`
	`149`	`+ : MTL::Size(nthreads, 1, 1);`
`152`	`150`	`compute_encoder.dispatchThreads(grid_dims, group_dims);`
`153`	`151`	`}`
`154`	`152`	`}`
Original file line number	Diff line number	Diff line change
`@@ -72,6 +72,7 @@ void ternary_op_gpu_inplace(`
`72`	`72`	`compute_encoder.set_input_array(donate_c ? out : c, 2);`
`73`	`73`	`compute_encoder.set_output_array(out, 3);`
`74`	`74`
	`75`	`+ auto thread_group_size = kernel->maxTotalThreadsPerThreadgroup();`
`75`	`76`	`if (topt == TernaryOpType::General) {`
`76`	`77`	`// Launch up to 3D grid of threads`
`77`	`78`	`size_t dim0 = ndim > 0 ? shape[ndim - 1] : 1;`
`@@ -93,7 +94,6 @@ void ternary_op_gpu_inplace(`
`93`	`94`	`compute_encoder->setBytes(strides_c.data(), ndim * sizeof(size_t), 6);`
`94`	`95`	`}`
`95`	`96`
`96`		`- NS::UInteger thread_group_size = kernel->maxTotalThreadsPerThreadgroup();`
`97`	`97`	`if (thread_group_size != 1024) {`
`98`	`98`	`throw std::runtime_error("[Metal::ternary] Must use 1024 sized block");`
`99`	`99`	`}`
`@@ -103,13 +103,12 @@ void ternary_op_gpu_inplace(`
`103`	`103`	`} else {`
`104`	`104`	`// Launch a 1D or 2D grid of threads`
`105`	`105`	`size_t nthreads = out.data_size();`
`106`		`- MTL::Size grid_dims = use_2d ? get_2d_grid_dims(out.shape(), out.strides())`
`107`		`- : MTL::Size(nthreads, 1, 1);`
`108`		`- NS::UInteger thread_group_size = kernel->maxTotalThreadsPerThreadgroup();`
`109`	`106`	`if (thread_group_size > nthreads) {`
`110`	`107`	`thread_group_size = nthreads;`
`111`	`108`	`}`
`112`	`109`	`MTL::Size group_dims = MTL::Size(thread_group_size, 1, 1);`
	`110`	`+ MTL::Size grid_dims = use_2d ? get_2d_grid_dims(out.shape(), out.strides())`
	`111`	`+ : MTL::Size(nthreads, 1, 1);`
`113`	`112`	`compute_encoder.dispatchThreads(grid_dims, group_dims);`
`114`	`113`	`}`
`115`	`114`	`}`
Original file line number	Diff line number	Diff line change
`@@ -47,9 +47,7 @@ void unary_op_gpu_inplace(`
`47`	`47`	`kernel_name += "_" + op + type_to_name(in) + type_to_name(out);`
`48`	`48`	`auto kernel = get_unary_kernel(d, kernel_name, in.dtype(), out.dtype(), op);`
`49`	`49`
`50`		`- MTL::Size grid_dims = use_2d ? get_2d_grid_dims(in.shape(), in.strides())`
`51`		`- : MTL::Size(nthreads, 1, 1);`
`52`		`- NS::UInteger thread_group_size = kernel->maxTotalThreadsPerThreadgroup();`
	`50`	`+ auto thread_group_size = kernel->maxTotalThreadsPerThreadgroup();`
`53`	`51`	`auto& compute_encoder = d.get_command_encoder(s.index);`
`54`	`52`	`compute_encoder->setComputePipelineState(kernel);`
`55`	`53`	`compute_encoder.set_input_array(`
`@@ -75,6 +73,8 @@ void unary_op_gpu_inplace(`
`75`	`73`	`thread_group_size = nthreads;`
`76`	`74`	`}`
`77`	`75`	`MTL::Size group_dims = MTL::Size(thread_group_size, 1, 1);`
	`76`	`+ MTL::Size grid_dims = use_2d ? get_2d_grid_dims(out.shape(), out.strides())`
	`77`	`+ : MTL::Size(nthreads, 1, 1);`
`78`	`78`	`compute_encoder.dispatchThreads(grid_dims, group_dims);`
`79`	`79`	`}`
`80`	`80`	`}`
Original file line number	Diff line number	Diff line change
`@@ -103,6 +103,9 @@ MTL::Size get_2d_grid_dims(`
`103`	`103`	`if (grid_y > UINT32_MAX \|\| grid_x > UINT32_MAX) {`
`104`	`104`	`throw std::runtime_error("Unable to safely factor shape.");`
`105`	`105`	`}`
	`106`	`+ if (grid_y > grid_x) {`
	`107`	`+ std::swap(grid_x, grid_y);`
	`108`	`+ }`
`106`	`109`	`return MTL::Size(`
`107`	`110`	`static_cast<uint32_t>(grid_x), static_cast<uint32_t>(grid_y), 1);`
`108`	`111`	`}`
`@@ -145,6 +148,9 @@ MTL::Size get_2d_grid_dims(`
`145`	`148`	`if (grid_y > UINT32_MAX \|\| grid_x > UINT32_MAX \|\| divisor > 1) {`
`146`	`149`	`throw std::runtime_error("Unable to safely factor shape.");`
`147`	`150`	`}`
	`151`	`+ if (grid_y > grid_x) {`
	`152`	`+ std::swap(grid_x, grid_y);`
	`153`	`+ }`
`148`	`154`	`return MTL::Size(`
`149`	`155`	`static_cast<uint32_t>(grid_x), static_cast<uint32_t>(grid_y), 1);`
`150`	`156`	`}`