RFC: Specialize for non-mixed-dtype in elementwise_util

swolchok · swolchok · commit e09068b7ce66 · 2025-03-25T12:47:50.000-07:00
Mixed dtype should be uncommon. Here is how we can specialize for the common case. Test Plan: automated tests on this PR verify we didn't break the now-deprecated runtime_out_dtypes mode; tests on the next PR will verify that everything works after migration. Also included migration for exactly one operator, op_mul, to verify that the new code compiles. ghstack-source-id: a094279 ghstack-comment-id: 2735017566 Pull Request resolved: pytorch/executorch#9388
diff --git a/kernels/portable/cpu/op_mul.cpp b/kernels/portable/cpu/op_mul.cpp
@@ -52,7 +52,10 @@ Tensor& mul_out(
       out);
 
   ET_SWITCH_REALB_TYPES(compute_type, ctx, op_name, CTYPE_COMPUTE, [&]() {
-    utils::apply_bitensor_elementwise_fn<CTYPE_COMPUTE, op_name>(
+    utils::apply_bitensor_elementwise_fn<
+        CTYPE_COMPUTE,
+        op_name,
+        utils::SupportedTensorDtypes::REALHBBF16>(
         [](const CTYPE_COMPUTE val_a, const CTYPE_COMPUTE val_b) {
           return val_a * val_b;
         },
@@ -61,8 +64,7 @@ Tensor& mul_out(
         utils::SupportedTensorDtypes::REALHBBF16,
         b,
         utils::SupportedTensorDtypes::REALHBBF16,
-        out,
-        utils::SupportedTensorDtypes::REALHBBF16);
+        out);
   });
 
   return out;
diff --git a/kernels/portable/cpu/util/dtype_util.h b/kernels/portable/cpu/util/dtype_util.h
@@ -309,6 +309,28 @@ bool check_tensor_dtype(
     SupportedTensorDtypes dtypes,
     const ScalarType compute_type);
 
+/// Return the one output type we are willing to emit specialized code
+/// to handle, given a compute type of CTYPE_COMMON and supported
+/// output types of out_dtypes.
+template <typename CTYPE_COMMON>
+inline constexpr ScalarType specialized_output_scalar_type(
+    SupportedTensorDtypes out_dtypes) {
+  switch (out_dtypes) {
+    case SupportedTensorDtypes::BOOL:
+      return ScalarType::Bool;
+    case SupportedTensorDtypes::BOOL_OR_BYTE:
+      return ScalarType::Bool;
+    case SupportedTensorDtypes::REALHBBF16:
+    case SupportedTensorDtypes::REALHBF16:
+    case SupportedTensorDtypes::REALH:
+    case SupportedTensorDtypes::FLOATHBF16:
+    case SupportedTensorDtypes::INTB:
+    case SupportedTensorDtypes::SAME_AS_COMPUTE:
+    case SupportedTensorDtypes::SAME_AS_COMMON:
+      return CppTypeToScalarType<CTYPE_COMMON>::value;
+  }
+}
+
 } // namespace internal
 } // namespace utils
 } // namespace native
diff --git a/kernels/portable/cpu/util/elementwise_util.h b/kernels/portable/cpu/util/elementwise_util.h
@@ -60,8 +60,43 @@ using op_call_result =
 
 template <
     typename CTYPE_COMMON,
+    typename CTYPE_OUT,
     typename Op,
-  typename... Args>
+    typename... Args>
+inline void dtype_specialized_elementwise_fn_impl(
+    const Op& compute_fun,
+    KernelRuntimeContext& ctx,
+    const Tensor& out,
+    Args... inputs) {
+  constexpr auto kNumInputs = sizeof...(inputs);
+  ET_DCHECK(((inputs.first->element_size() == sizeof(CTYPE_COMMON)) && ...));
+
+  std::array<const CTYPE_COMMON*, kNumInputs> inputs_data_ptrs = {
+      inputs.first->template const_data_ptr<CTYPE_COMMON>()...};
+
+  CTYPE_OUT* const data_out = out.mutable_data_ptr<CTYPE_OUT>();
+
+  ::executorch::extension::parallel_for(
+      0,
+      out.numel(),
+      ::executorch::extension::internal::GRAIN_SIZE,
+      [&](const auto begin, const auto end) {
+        const auto range =
+            BroadcastIndexesRange<kNumInputs>(out, (*inputs.first)...);
+        auto begin_it = range.begin();
+        begin_it += begin;
+        for (; (*begin_it)[0] < end; ++begin_it) {
+          const auto& indexes = *begin_it;
+          std::array<CTYPE_COMMON, kNumInputs> loaded_inputs;
+          for (const auto idx : c10::irange(kNumInputs)) {
+            loaded_inputs[idx] = inputs_data_ptrs[idx][indexes[idx + 1]];
+          }
+          data_out[indexes[0]] = std::apply(compute_fun, loaded_inputs);
+        }
+      });
+}
+
+template <typename CTYPE_COMMON, typename Op, typename... Args>
 inline bool validate_elementwise_fn_inputs(
     const Op& compute_fun,
     KernelRuntimeContext& ctx,
@@ -80,7 +115,8 @@ inline bool validate_elementwise_fn_inputs(
       ctx,
       (check_input_dtype(inputs, compute_type) && ...) &&
           internal::check_tensor_dtype(out, out_dtypes, compute_type),
-      InvalidArgument, false);
+      InvalidArgument,
+      false);
 
   return true;
 }
@@ -90,22 +126,12 @@ template <
     const char* op_name,
     typename Op,
     typename... Args>
-inline void apply_elementwise_fn(
+inline void apply_elementwise_fn_generic_impl(
     const Op& compute_fun,
     KernelRuntimeContext& ctx,
     const Tensor& out,
     SupportedTensorDtypes out_dtypes,
     Args... inputs) {
-  const bool inputs_valid = validate_elementwise_fn_inputs<CTYPE_COMMON>(
-      compute_fun,
-      ctx,
-      out,
-      out_dtypes,
-      inputs...);
-  if (!inputs_valid) {
-    return;
-  }
-
   constexpr auto kNumInputs = sizeof...(inputs);
 
   struct InputInfo {
@@ -157,6 +183,63 @@ inline void apply_elementwise_fn(
         }
       });
 }
+
+template <
+    typename CTYPE_COMMON,
+    const char* op_name,
+    typename Op,
+    typename... Args>
+inline void apply_elementwise_fn_runtime_out_dtypes(
+    const Op& compute_fun,
+    KernelRuntimeContext& ctx,
+    const Tensor& out,
+    SupportedTensorDtypes out_dtypes,
+    Args... inputs) {
+  const bool inputs_valid = validate_elementwise_fn_inputs<CTYPE_COMMON>(
+      compute_fun, ctx, out, out_dtypes, inputs...);
+  if (!inputs_valid) {
+    return;
+  }
+
+  apply_elementwise_fn_generic_impl<CTYPE_COMMON, op_name>(
+      compute_fun, ctx, out, out_dtypes, inputs...);
+}
+
+template <
+    typename CTYPE_COMMON,
+    const char* op_name,
+    SupportedTensorDtypes out_dtypes,
+    typename Op,
+    typename... Args>
+inline void apply_elementwise_fn(
+    const Op& compute_fun,
+    KernelRuntimeContext& ctx,
+    const Tensor& out,
+    Args... inputs) {
+  const bool inputs_valid = validate_elementwise_fn_inputs<CTYPE_COMMON>(
+      compute_fun, ctx, out, out_dtypes, inputs...);
+  if (!inputs_valid) {
+    return;
+  }
+
+  constexpr auto compute_type = CppTypeToScalarType<CTYPE_COMMON>::value;
+  const bool all_inputs_compute_dtype =
+      ((inputs.first->scalar_type() == compute_type) && ...);
+
+  constexpr ScalarType out_specialized_scalar_type =
+      specialized_output_scalar_type<CTYPE_COMMON>(out_dtypes);
+  if (all_inputs_compute_dtype &&
+      out.scalar_type() == out_specialized_scalar_type) {
+    using CTYPE_OUT =
+        typename ScalarTypeToCppType<out_specialized_scalar_type>::type;
+    dtype_specialized_elementwise_fn_impl<CTYPE_COMMON, CTYPE_OUT>(
+        compute_fun, ctx, out, inputs...);
+    return;
+  }
+
+  apply_elementwise_fn_generic_impl<CTYPE_COMMON, op_name>(
+      compute_fun, ctx, out, out_dtypes, inputs...);
+}
 } // namespace internal
 
 /// DEPRECATED: prefer the variant with out_dtypes in the template argument.
@@ -168,19 +251,23 @@ inline void apply_unitensor_elementwise_fn(
     SupportedTensorDtypes a_dtypes,
     const Tensor& out,
     SupportedTensorDtypes out_dtypes) {
-  internal::apply_elementwise_fn<CTYPE_COMMON, op_name>(
+  internal::apply_elementwise_fn_runtime_out_dtypes<CTYPE_COMMON, op_name>(
       compute_fun, ctx, out, out_dtypes, std::make_pair(&a, a_dtypes));
 }
 
-template <typename CTYPE_COMMON, const char* op_name, SupportedTensorDtypes out_dtypes, typename Op>
+template <
+    typename CTYPE_COMMON,
+    const char* op_name,
+    SupportedTensorDtypes out_dtypes,
+    typename Op>
 inline void apply_unitensor_elementwise_fn(
     const Op& compute_fun,
     KernelRuntimeContext& ctx,
     const Tensor& a,
     SupportedTensorDtypes a_dtypes,
     const Tensor& out) {
-  internal::apply_elementwise_fn<CTYPE_COMMON, op_name>(
-      compute_fun, ctx, out, out_dtypes, std::make_pair(&a, a_dtypes));
+  internal::apply_elementwise_fn<CTYPE_COMMON, op_name, out_dtypes>(
+      compute_fun, ctx, out, std::make_pair(&a, a_dtypes));
 }
 
 /**
@@ -196,7 +283,7 @@ inline void apply_bitensor_elementwise_fn(
     SupportedTensorDtypes b_dtypes,
     const Tensor& out,
     SupportedTensorDtypes out_dtypes) {
-  internal::apply_elementwise_fn<CTYPE_COMMON, op_name>(
+  internal::apply_elementwise_fn_runtime_out_dtypes<CTYPE_COMMON, op_name>(
       compute_fun,
       ctx,
       out,
@@ -210,7 +297,11 @@ inline void apply_bitensor_elementwise_fn(
  * perform a computation and write to the corresponding element of the output.
  * Tensor broadcasting is applied wherever it is required.
  */
-template <typename CTYPE_COMMON, const char* op_name, SupportedTensorDtypes out_dtypes, typename Op>
+template <
+    typename CTYPE_COMMON,
+    const char* op_name,
+    SupportedTensorDtypes out_dtypes,
+    typename Op>
 inline void apply_bitensor_elementwise_fn(
     const Op& compute_fun,
     KernelRuntimeContext& ctx,
@@ -219,11 +310,10 @@ inline void apply_bitensor_elementwise_fn(
     const Tensor& b,
     SupportedTensorDtypes b_dtypes,
     const Tensor& out) {
-  internal::apply_elementwise_fn<CTYPE_COMMON, op_name>(
+  internal::apply_elementwise_fn<CTYPE_COMMON, op_name, out_dtypes>(
       compute_fun,
       ctx,
       out,
-      out_dtypes,
       std::make_pair(&a, a_dtypes),
       std::make_pair(&b, b_dtypes));
 }
@@ -243,7 +333,7 @@ inline void apply_tritensor_elementwise_fn(
     SupportedTensorDtypes c_dtypes,
     const Tensor& out,
     SupportedTensorDtypes out_dtypes) {
-  internal::apply_elementwise_fn<CTYPE_COMMON, op_name>(
+  internal::apply_elementwise_fn_runtime_out_dtypes<CTYPE_COMMON, op_name>(
       compute_fun,
       ctx,
       out,
@@ -273,7 +363,11 @@ inline void apply_tritensor_elementwise_fn(
  * static constexpr const char op_name[] = "my_op";
  * apply_ternary_elementwise_fn<CTYPE_COMMON, op_name>.
  */
-template <typename CTYPE_COMMON, const char* op_name, SupportedTensorDtypes out_dtypes, typename Op>
+template <
+    typename CTYPE_COMMON,
+    const char* op_name,
+    SupportedTensorDtypes out_dtypes,
+    typename Op>
 inline void apply_tritensor_elementwise_fn(
     const Op& compute_fun,
     KernelRuntimeContext& ctx,
@@ -284,11 +378,10 @@ inline void apply_tritensor_elementwise_fn(
     const Tensor& c,
     SupportedTensorDtypes c_dtypes,
     const Tensor& out) {
-  internal::apply_elementwise_fn<CTYPE_COMMON, op_name>(
+  internal::apply_elementwise_fn<CTYPE_COMMON, op_name, out_dtypes>(
       compute_fun,
       ctx,
       out,
-      out_dtypes,
       std::make_pair(&a, a_dtypes),
       std::make_pair(&b, b_dtypes),
       std::make_pair(&c, c_dtypes));