Add an expander for isnormal using integer arithmetic.  This is
typically faster and avoids generating spurious exceptions on
signaling NaNs.  This fixes part of PR66462.

int isnormal1 (float x) { return __builtin_isnormal (x); }

Before:
        fabs    s0, s0
        mov     w0, 2139095039
        fmov    s31, w0
        fcmp    s0, s31
        movi    v31.2s, 0x80, lsl 16
        fccmp   s0, s31, 1, ls
        cset    w0, lt
        eor     w0, w0, 1
        ret

After:
        fmov    w0, s0
        ubfx    w0, w0, 23, 8
        sub     w0, w0, #1
        cmp     w0, 254
        cset    w0, cc
        ret

HFmode and BFmode are handled too.  Neither has 16-bit integer
arithmetic, so their encoding is zero-extended into a word first:

int isnormal2 (_Float16 x) { return __builtin_isnormal (x); }

Before:
        umov    w0, v0.h[0]
        mvni    v0.4h, 0x84, lsl 8
        movi    v30.4h, 0x4, lsl 8
        fcvt    s0, h0
        and     w0, w0, 32767
        fcvt    s30, h30
        dup     v31.4h, w0
        fcvt    s31, h31
        fcmp    s31, s0
        cset    w0, hi
        fcmp    s31, s30
        cset    w1, lt
        orr     w0, w0, w1
        eor     w0, w0, 1
        ret

After:
        umov    w0, v0.h[0]
        ubfx    w0, w0, 10, 5
        sub     w0, w0, #1
        cmp     w0, 30
        cset    w0, cc
        ret

gcc/ChangeLog:

        PR middle-end/66462
        * config/aarch64/aarch64.md (isnormal<mode>2): Add new expander.
        * config/aarch64/iterators.md (GPF_HF_BF): New mode iterator.
        (mantissa_bits): Add HF and BF.
        (V_INT_EQUIV): Add BF.

gcc/testsuite/ChangeLog:

        PR middle-end/66462
        * gcc.target/aarch64/pr66462.c: Add tests for isnormal, including
        _Float16 and __bf16.
---
 gcc/config/aarch64/aarch64.md              | 29 +++++++++
 gcc/config/aarch64/iterators.md            |  7 ++-
 gcc/testsuite/gcc.target/aarch64/pr66462.c | 72 ++++++++++++++++++++++
 3 files changed, 106 insertions(+), 2 deletions(-)

diff --git ./gcc/config/aarch64/aarch64.md ./gcc/config/aarch64/aarch64.md
index a32f8c627f1..58383483b12 100644
--- ./gcc/config/aarch64/aarch64.md
+++ ./gcc/config/aarch64/aarch64.md
@@ -7900,6 +7900,35 @@
 }
 )
 
+;; A number is normal iff its exponent field E is neither 0 nor all ones,
+;; i.e. (unsigned) (E - 1) < EXP_MAX - 1.
+(define_expand "isnormal<mode>2"
+ [(match_operand:SI 0 "register_operand")
+  (match_operand:GPF_HF_BF 1 "register_operand")]
+ "TARGET_FLOAT"
+{
+  scalar_int_mode imode = <V_INT_EQUIV>mode;
+  int exp_bits = GET_MODE_BITSIZE (<MODE>mode) - <mantissa_bits> - 1;
+  HOST_WIDE_INT exp_max = (HOST_WIDE_INT_1 << exp_bits) - 1;
+  rtx op = force_lowpart_subreg (imode, operands[1], <MODE>mode);
+  /* There is no 16-bit arithmetic, so work on the encoding of the 16-bit
+     formats in a word.  */
+  if (imode == HImode)
+    imode = SImode;
+  op = convert_to_mode (imode, op, 1);
+  op = expand_simple_binop (imode, LSHIFTRT, op, GEN_INT (<mantissa_bits>),
+                           NULL_RTX, 1, OPTAB_DIRECT);
+  op = expand_simple_binop (imode, AND, op, GEN_INT (exp_max),
+                           NULL_RTX, 1, OPTAB_DIRECT);
+  op = expand_simple_binop (imode, PLUS, op, constm1_rtx,
+                           NULL_RTX, 1, OPTAB_DIRECT);
+  rtx cc_reg = aarch64_gen_compare_reg (LTU, op, GEN_INT (exp_max - 1));
+  rtx cmp = gen_rtx_fmt_ee (LTU, SImode, cc_reg, const0_rtx);
+  emit_insn (gen_aarch64_cstoresi (operands[0], cmp, cc_reg));
+  DONE;
+}
+)
+
 ;; -------------------------------------------------------------------
 ;; Reload support
 ;; -------------------------------------------------------------------
diff --git ./gcc/config/aarch64/iterators.md ./gcc/config/aarch64/iterators.md
index a56edb8ebf4..76f042ff2f8 100644
--- ./gcc/config/aarch64/iterators.md
+++ ./gcc/config/aarch64/iterators.md
@@ -65,6 +65,9 @@
 ;; Iterator for all 16-bit scalar floating point modes (HF, BF)
 (define_mode_iterator HFBF [HF BF])
 
+;; Iterator for all scalar floating point modes (HF, BF, SF, DF)
+(define_mode_iterator GPF_HF_BF [HF BF SF DF])
+
 ;; Iterator for all integer modes (up to 64-bit) plus all General Purpose
 ;; Floating-point registers (32- and 64-bit modes).
 (define_mode_iterator ALLI_GPF [ALLI GPF])
@@ -1504,7 +1507,7 @@
 
 (define_mode_attr half_mask [(HI "255") (SI "65535") (DI "4294967295")])
 
-(define_mode_attr mantissa_bits [(SF "23") (DF "52")])
+(define_mode_attr mantissa_bits [(HF "10") (BF "7") (SF "23") (DF "52")])
 
 ;; For constraints used in scalar immediate vector moves
 (define_mode_attr hq [(HI "h") (QI "q")])
@@ -2484,7 +2487,7 @@
                               (V2SF "V2SI") (V4SF  "V4SI")
                               (DF   "DI")   (V2DF  "V2DI")
                               (SF   "SI")   (SI    "SI")
-                              (HF    "HI")
+                              (HF    "HI")  (BF    "HI")
                               (VNx16QI "VNx16QI")
                               (VNx8HI  "VNx8HI") (VNx8HF "VNx8HI")
                               (VNx8BF  "VNx8HI")
diff --git ./gcc/testsuite/gcc.target/aarch64/pr66462.c 
./gcc/testsuite/gcc.target/aarch64/pr66462.c
index 2bd734b3193..c6367ac16f6 100644
--- ./gcc/testsuite/gcc.target/aarch64/pr66462.c
+++ ./gcc/testsuite/gcc.target/aarch64/pr66462.c
@@ -64,6 +64,42 @@ static void t_nan (double x, bool res)
     __builtin_abort ();
 }
 
+static void t_normalf (float x, bool res)
+{
+  if (__builtin_isnormal (x) != res)
+    __builtin_abort ();
+  if (__builtin_isnormal (-x) != res)
+    __builtin_abort ();
+  if (fetestexcept (FE_INVALID))
+    __builtin_abort ();
+}
+
+static void t_normal (double x, bool res)
+{
+  if (__builtin_isnormal (x) != res)
+    __builtin_abort ();
+  if (__builtin_isnormal (-x) != res)
+    __builtin_abort ();
+  if (fetestexcept (FE_INVALID))
+    __builtin_abort ();
+}
+
+
+/* Generate the _Float16 and __bf16 testers with a macro rather than
+   spelling out the same body for each type.  */
+#define DEF_TEST(NAME, TYPE, PRED)             \
+static void NAME (TYPE x, bool res)            \
+{                                              \
+  if (PRED (x) != res)                         \
+    __builtin_abort ();                                \
+  if (PRED (-x) != res)                                \
+    __builtin_abort ();                                \
+  if (fetestexcept (FE_INVALID))               \
+    __builtin_abort ();                                \
+}
+
+DEF_TEST (t_normal16, _Float16, __builtin_isnormal)
+DEF_TEST (t_normalbf, __bf16, __builtin_isnormal)
 
 int
 main ()
@@ -106,5 +142,41 @@ main ()
   t_nan (__builtin_nans (""), 1);
   t_nan (__builtin_nan (""), 1);
 
+  t_normalf (0.0f, 0);
+  t_normalf (1.0f, 1);
+  t_normalf (__FLT_MIN__, 1);
+  t_normalf (__FLT_MAX__, 1);
+  t_normalf (__FLT_DENORM_MIN__, 0);
+  t_normalf (__builtin_inff (), 0);
+  t_normalf (__builtin_nansf (""), 0);
+  t_normalf (__builtin_nanf (""), 0);
+
+  t_normal (0.0, 0);
+  t_normal (1.0, 1);
+  t_normal (__DBL_MIN__, 1);
+  t_normal (__DBL_MAX__, 1);
+  t_normal (__DBL_DENORM_MIN__, 0);
+  t_normal (__builtin_inf (), 0);
+  t_normal (__builtin_nans (""), 0);
+  t_normal (__builtin_nan (""), 0);
+
+  t_normal16 (0.0f16, 0);
+  t_normal16 (1.0f16, 1);
+  t_normal16 (__FLT16_MIN__, 1);
+  t_normal16 (__FLT16_MAX__, 1);
+  t_normal16 (__FLT16_DENORM_MIN__, 0);
+  t_normal16 ((_Float16) __builtin_inff (), 0);
+  t_normal16 (__builtin_nansf16 (""), 0);
+  t_normal16 (__builtin_nanf16 (""), 0);
+
+  t_normalbf (0.0bf16, 0);
+  t_normalbf (1.0bf16, 1);
+  t_normalbf (__BFLT16_MIN__, 1);
+  t_normalbf (__BFLT16_MAX__, 1);
+  t_normalbf (__BFLT16_DENORM_MIN__, 0);
+  t_normalbf ((__bf16) __builtin_inff (), 0);
+  t_normalbf (__builtin_nansf16b (""), 0);
+  t_normalbf ((__bf16) __builtin_nanf (""), 0);
+
   return 0;
 }
-- 
2.54.0

Reply via email to