New in V2:
  - Move the simplification logic so it's deployed via match.pd.
  - In the commit message I've also added the relevant note as to
    why we can't just do away with emitting `VEC_COND_EXPR' altogether,
    in response to a question in V1.
    
The simple implementation implemented here introduced no regressions, with
the patch having undergone regression tests for both AArch64 and x86_64.
--------------

Consider the following dot-product computation:

  uint32_t
  tcp_checksum(int n, uint8_t* data)
  {
    uint32_t sum = 0;
    for (int i=0; i<n; i+=1)
      sum += data[i] * data[i];
    return sum;
  }

At present, following vectorization and the subsequent optimization
passes, we will end up with the following GIMPLE code:

  vect__1 = .MASK_LOAD (vectp_data, 8B, loop_mask_1, { 0, ... });
  masked_op1_1 = VEC_COND_EXPR <loop_mask_1, vect__1, { 0, ... }>;
  vect_patt_1 = DOT_PROD_EXPR <vect__1, masked_op1_1, vectt_sum_1>;

While in this case the `VEC_COND_EXPR' is redundant, we cannot make the
assumption that input data for the vectorized dot product would always
already be masked.  As such, it is correct that `VEC_COND_EXPR'
should be emitted in conjunction with `DOT_PROD_EXPR' by the vectorizer
in order to emulate masked dot product operations.

Where simplification is possible, a better approach that maintains
correctness is to look at the input data, mask and else value going into
the VEC_COND_EXPR, walking back up the USE-DEF chain to see whether the
source of the input data shares the same mask and else values.  If so,
we can safely remove the VEC_COND_EXPR statement from  the cfg, thus
resulting in the more optimal variant:

  vect__1 = .MASK_LOAD (vectp_data, 8B, loop_mask_1, { 0, ... });
  vect_patt_1 = DOT_PROD_EXPR <vect__1, vect__1, vectt_sum_1>;

We can do this in `match.pd' via a simple pattern, e.g.

(simplify
 (vec_cond @0 @1 @2)
 (if (same_mask_and_else_value_p (@0, @1, @2))
  @1))

and implement it in the current patch.

        PR tree-optimization/111770

gcc/Changelog:

        * match.pd: Add `same_mask_and_else_value_p' to
        `define_predicates', applying it to `vec_cond' expression
        simplification.
        * tree.cc (same_mask_and_else_value_p): New.
        * tree.h (same_mask_and_else_value_p): Likewise.

gcc/testsuite/ChangeLog:

        * gcc.dg/vect/vect-cond-dot.c: New
---
 gcc/match.pd                              | 11 ++++++-
 gcc/testsuite/gcc.dg/vect/vect-cond-dot.c | 20 ++++++++++++
 gcc/tree.cc                               | 37 +++++++++++++++++++++++
 gcc/tree.h                                |  1 +
 4 files changed, 68 insertions(+), 1 deletion(-)
 create mode 100644 gcc/testsuite/gcc.dg/vect/vect-cond-dot.c

diff --git a/gcc/match.pd b/gcc/match.pd
index 0ba97b32cb1..3da40f6adcb 100644
--- a/gcc/match.pd
+++ b/gcc/match.pd
@@ -40,7 +40,8 @@ along with GCC; see the file COPYING3.  If not see
    HONOR_NANS
    uniform_vector_p
    expand_vec_cmp_expr_p
-   bitmask_inv_cst_vector_p)
+   bitmask_inv_cst_vector_p
+   same_mask_and_else_value_p)
 
 /* Operator lists.  */
 (define_operator_list tcc_comparison
@@ -8897,6 +8898,14 @@ DEFINE_INT_AND_FLOAT_ROUND_FN (RINT)
         (icmp @0 { csts; })
         (icmp (view_convert:utype @0) { csts; })))))))))
 
+/* Transform cases where VEC_COND_EXPR carries out a redundant operation,
+   e.g. masking out values that have already been masked out from a previous
+   masking operation such as a masked load.  */
+(simplify
+ (vec_cond @0 @1 @2)
+ (if (same_mask_and_else_value_p (@0, @1, @2))
+  @1))
+
 /* When one argument is a constant, overflow detection can be simplified.
    Currently restricted to single use so as not to interfere too much with
    ADD_OVERFLOW detection in tree-ssa-math-opts.cc.
diff --git a/gcc/testsuite/gcc.dg/vect/vect-cond-dot.c 
b/gcc/testsuite/gcc.dg/vect/vect-cond-dot.c
new file mode 100644
index 00000000000..667519d1e50
--- /dev/null
+++ b/gcc/testsuite/gcc.dg/vect/vect-cond-dot.c
@@ -0,0 +1,20 @@
+/* { dg-do compile } */
+/* { dg-additional-options "-fdump-tree-optimized" } */
+/* { dg-require-effective-target vect_masked_load } */
+#include <stdint.h>
+
+#define CHECK_DOT(IN, OUT)             \
+OUT check_dot_##OUT(int n, IN* data) {         \
+  OUT sum = 0;                         \
+  for (int i=0; i<n; i+=1) {           \
+    sum += data[i] * data[i];          \
+  }                                    \
+  return sum;                          \
+}
+
+CHECK_DOT (uint8_t, uint32_t);
+CHECK_DOT (int8_t, int32_t);
+CHECK_DOT (int16_t, int64_t);
+
+/* { dg-final { scan-tree-dump-times {vectorized 1 loops} 3 "vect"  } } */
+/* { dg-final { scan-tree-dump-not {VEC_COND_EXPR} "optimized" } } */
diff --git a/gcc/tree.cc b/gcc/tree.cc
index c8aa42b3e10..1182b1c5d3d 100644
--- a/gcc/tree.cc
+++ b/gcc/tree.cc
@@ -12343,6 +12343,43 @@ block_ultimate_origin (const_tree block)
     }
 }
 
+/* Look for masking redundancy.  When applying a mask, check whether the
+   statement defining the input values uses the same mask and else values as 
the
+   current masking operation, in which case the masking operation is redundant
+   and may be safely eliminated.  */
+
+bool
+same_mask_and_else_value_p (tree mask, tree then_val, tree else_val)
+{
+  if (then_val && TREE_CODE (then_val) == SSA_NAME)
+    {
+      /* Walk back up the use-def chain and see whether value comes from a
+        masked operation.  */
+      gimple *then_defn = SSA_NAME_DEF_STMT (then_val);
+      if (then_defn
+         && is_gimple_call (then_defn)
+         && gimple_call_internal_p (then_defn))
+       {
+         internal_fn ifn = gimple_call_internal_fn (then_defn);
+         int mask_index = internal_fn_mask_index (ifn);
+         if (mask_index == 0)
+           return false;
+
+         /* See how the defining masked operation populated inactive lanes
+            and compare this to how the current masked op handles its
+            inactive lanes.  */
+         int false_index = internal_fn_else_index (ifn);
+         tree then_mask = gimple_call_arg (then_defn, mask_index);
+         tree then_false = gimple_call_arg (then_defn, false_index);
+
+         if (mask == then_mask
+             && operand_equal_p (else_val, then_false, OEP_ONLY_CONST))
+           return true;
+       }
+    }
+  return false;
+}
+
 /* Return true iff conversion from INNER_TYPE to OUTER_TYPE generates
    no instruction.  */
 
diff --git a/gcc/tree.h b/gcc/tree.h
index 1ccbf848d9b..68adddcd5f6 100644
--- a/gcc/tree.h
+++ b/gcc/tree.h
@@ -5783,6 +5783,7 @@ extern bool prototype_p (const_tree);
 extern bool auto_var_p (const_tree);
 extern bool auto_var_in_fn_p (const_tree, const_tree);
 extern tree build_low_bits_mask (tree, unsigned);
+extern bool same_mask_and_else_value_p (tree, tree, tree);
 extern bool tree_nop_conversion_p (const_tree, const_tree);
 extern tree tree_strip_nop_conversions (tree);
 extern tree tree_strip_sign_nop_conversions (tree);
-- 
2.43.0

Reply via email to