On Tue, Aug 11, 2026 at 3:43 AM <[email protected]> wrote:
>
> From: Kyrylo Tkachov <[email protected]>
>
> FEAT_FP16 provides half-precision forms of FCCMP and FCCMPE.  The
> conditional-compare patterns and expansion hooks currently accept only
> SFmode and DFmode comparisons.  For example, with -O2 -fno-trapping-math
> -march=armv8.2-a+fp16:
>
>   int
>   f (_Float16 a, _Float16 b, _Float16 c, _Float16 d)
>   {
>     return (a == b) && (c == d);
>   }
>
> AArch64 emitted:
>
>         fcmp    h0, h1
>         cset    w1, eq
>         fcmp    h2, h3
>         cset    w0, eq
>         and     w0, w1, w0
>
> After this patch it emits:
>
>         fcmp    h0, h1
>         fccmp   h2, h3, 0, eq
>         cset    w0, eq
>
> Use GPF_F16 for both conditional-compare patterns.  Use the stype attribute
> so HFmode keeps the single-precision scheduling type.  Teach both expansion
> hooks to select HFmode, and reject it when FEAT_FP16 is not available.
>
> The tests also cover FCCMPE, the reverse pattern used by OR expressions, and
> +nofp16.
>
> Bootstrapped and tested on aarch64-none-linux-gnu.
> Ok for trunk?


Ok.

> Thanks,
> Kyrill
>
> gcc/
>
>         * config/aarch64/aarch64.cc (aarch64_gen_ccmp_first): Handle HFmode.
>         (aarch64_gen_ccmp_next): Likewise.  Reject unsupported HFmode early.
>         * config/aarch64/aarch64.md
>         (@ccmp<CCFP_CCFPE:mode><GPF_F16:mode>): Use GPF_F16 and stype.
>         (@ccmp<CCFP_CCFPE:mode><GPF_F16:mode>_rev): Likewise.
>
> gcc/testsuite/
>
>         * gcc.target/aarch64/fccmp_1.c: New test.
>         * gcc.target/aarch64/fccmp_2.c: Likewise.
>
> Signed-off-by: Kyrylo Tkachov <[email protected]>
> ---
>  gcc/config/aarch64/aarch64.cc              | 19 ++++++---
>  gcc/config/aarch64/aarch64.md              | 16 ++++----
>  gcc/testsuite/gcc.target/aarch64/fccmp_1.c | 47 ++++++++++++++++++++++
>  gcc/testsuite/gcc.target/aarch64/fccmp_2.c | 12 ++++++
>  4 files changed, 81 insertions(+), 13 deletions(-)
>  create mode 100644 gcc/testsuite/gcc.target/aarch64/fccmp_1.c
>  create mode 100644 gcc/testsuite/gcc.target/aarch64/fccmp_2.c
>
> diff --git a/gcc/config/aarch64/aarch64.cc b/gcc/config/aarch64/aarch64.cc
> index d19ca305d82..8e1eb2d7e33 100644
> --- a/gcc/config/aarch64/aarch64.cc
> +++ b/gcc/config/aarch64/aarch64.cc
> @@ -29621,6 +29621,17 @@ aarch64_gen_ccmp_first (rtx_insn **prep_seq, 
> rtx_insn **gen_seq,
>        icode = CODE_FOR_cmpdi;
>        break;
>
> +    case E_HFmode:
> +      if (!TARGET_FP_F16INST)
> +       {
> +         end_sequence ();
> +         return NULL_RTX;
> +       }
> +      cmp_mode = HFmode;
> +      cc_mode = aarch64_select_cc_mode (code, op0, op1);
> +      icode = cc_mode == CCFPEmode ? CODE_FOR_fcmpehf : CODE_FOR_fcmphf;
> +      break;
> +
>      case E_SFmode:
>        cmp_mode = SFmode;
>        cc_mode = aarch64_select_cc_mode (code, op0, op1);
> @@ -29677,6 +29688,7 @@ aarch64_gen_ccmp_next (rtx_insn **prep_seq, rtx_insn 
> **gen_seq, rtx prev,
>    /* Exit early for modes that are ot handled to avoid O(n^2) part of 
> expand_operands. */
>    op_mode = TYPE_MODE (TREE_TYPE (treeop0));
>    if (!(op_mode == QImode || op_mode == HImode || op_mode == SImode || 
> op_mode == DImode
> +       || (op_mode == HFmode && TARGET_FP_F16INST)
>         || op_mode == SFmode || op_mode == DFmode))
>     return NULL_RTX;
>
> @@ -29702,13 +29714,10 @@ aarch64_gen_ccmp_next (rtx_insn **prep_seq, 
> rtx_insn **gen_seq, rtx prev,
>        cmp_mode = DImode;
>        break;
>
> +    case E_HFmode:
>      case E_SFmode:
> -      cmp_mode = SFmode;
> -      cc_mode = aarch64_select_cc_mode (cmp_code, op0, op1);
> -      break;
> -
>      case E_DFmode:
> -      cmp_mode = DFmode;
> +      cmp_mode = op_mode;
>        cc_mode = aarch64_select_cc_mode (cmp_code, op0, op1);
>        break;
>
> diff --git a/gcc/config/aarch64/aarch64.md b/gcc/config/aarch64/aarch64.md
> index b2185c63819..9da5de8429f 100644
> --- a/gcc/config/aarch64/aarch64.md
> +++ b/gcc/config/aarch64/aarch64.md
> @@ -1134,21 +1134,21 @@
>    }
>  )
>
> -(define_insn "@ccmp<CCFP_CCFPE:mode><GPF:mode>"
> +(define_insn "@ccmp<CCFP_CCFPE:mode><GPF_F16:mode>"
>    [(set (match_operand:CCFP_CCFPE 1 "cc_register" "")
>         (if_then_else:CCFP_CCFPE
>           (match_operator 4 "aarch64_comparison_operator"
>            [(match_operand 0 "cc_register" "")
>             (const_int 0)])
>           (compare:CCFP_CCFPE
> -           (match_operand:GPF 2 "register_operand" "w")
> -           (match_operand:GPF 3 "register_operand" "w"))
> +           (match_operand:GPF_F16 2 "register_operand" "w")
> +           (match_operand:GPF_F16 3 "register_operand" "w"))
>           (unspec:CCFP_CCFPE
>             [(match_operand 5 "immediate_operand")]
>             UNSPEC_NZCV)))]
>    "TARGET_FLOAT"
>    "fccmp<e>\\t%<s>2, %<s>3, %k5, %m4"
> -  [(set_attr "type" "fccmp<s>")]
> +  [(set_attr "type" "fccmp<stype>")]
>  )
>
>  (define_insn "@ccmp<CC_ONLY:mode><GPI:mode>_rev"
> @@ -1171,7 +1171,7 @@
>    }
>  )
>
> -(define_insn "@ccmp<CCFP_CCFPE:mode><GPF:mode>_rev"
> +(define_insn "@ccmp<CCFP_CCFPE:mode><GPF_F16:mode>_rev"
>    [(set (match_operand:CCFP_CCFPE 1 "cc_register" "")
>         (if_then_else:CCFP_CCFPE
>           (match_operator 4 "aarch64_comparison_operator"
> @@ -1181,11 +1181,11 @@
>             [(match_operand 5 "immediate_operand")]
>             UNSPEC_NZCV)
>           (compare:CCFP_CCFPE
> -           (match_operand:GPF 2 "register_operand" "w")
> -           (match_operand:GPF 3 "register_operand" "w"))))]
> +           (match_operand:GPF_F16 2 "register_operand" "w")
> +           (match_operand:GPF_F16 3 "register_operand" "w"))))]
>    "TARGET_FLOAT"
>    "fccmp<e>\\t%<s>2, %<s>3, %k5, %M4"
> -  [(set_attr "type" "fccmp<s>")]
> +  [(set_attr "type" "fccmp<stype>")]
>  )
>
>  ;; Expansion of signed mod by a power of 2 using CSNEG.
> diff --git a/gcc/testsuite/gcc.target/aarch64/fccmp_1.c 
> b/gcc/testsuite/gcc.target/aarch64/fccmp_1.c
> new file mode 100644
> index 00000000000..96d6f717136
> --- /dev/null
> +++ b/gcc/testsuite/gcc.target/aarch64/fccmp_1.c
> @@ -0,0 +1,47 @@
> +/* { dg-do compile } */
> +/* { dg-options "-O2 -fno-trapping-math -march=armv8.2-a+fp16" } */
> +/* { dg-final { check-function-bodies "**" "" } } */
> +
> +/* FCCMP and FCCMPE have half precision forms when FEAT_FP16 is available.
> +   Trapping behaviour is disabled so each conditional comparison is safe.  */
> +
> +/*
> +** hf_lt:
> +**     fcmpe   h0, h1
> +**     fccmpe  h2, h3, 0, mi
> +**     cset    w0, mi
> +**     ret
> +*/
> +int
> +hf_lt (_Float16 a, _Float16 b, _Float16 c, _Float16 d)
> +{
> +  return (a < b) && (c < d);
> +}
> +
> +/*
> +** hf_eq:
> +**     fcmp    h0, h1
> +**     fccmp   h2, h3, 0, eq
> +**     cset    w0, eq
> +**     ret
> +*/
> +int
> +hf_eq (_Float16 a, _Float16 b, _Float16 c, _Float16 d)
> +{
> +  return (a == b) && (c == d);
> +}
> +
> +/* Exercise the reverse conditional-compare pattern.  */
> +
> +/*
> +** hf_ior:
> +**     fcmpe   h0, h1
> +**     fccmpe  h2, h3, 8, pl
> +**     cset    w0, mi
> +**     ret
> +*/
> +int
> +hf_ior (_Float16 a, _Float16 b, _Float16 c, _Float16 d)
> +{
> +  return (a < b) || (c < d);
> +}
> diff --git a/gcc/testsuite/gcc.target/aarch64/fccmp_2.c 
> b/gcc/testsuite/gcc.target/aarch64/fccmp_2.c
> new file mode 100644
> index 00000000000..c54a83b2eef
> --- /dev/null
> +++ b/gcc/testsuite/gcc.target/aarch64/fccmp_2.c
> @@ -0,0 +1,12 @@
> +/* { dg-do compile } */
> +/* { dg-options "-O2 -march=armv8.2-a+nofp16" } */
> +
> +int
> +hf_and (_Float16 a, _Float16 b, _Float16 c, _Float16 d)
> +{
> +  return (a == b) && (c == d);
> +}
> +
> +/* FEAT_FP16 is required for half precision compare instructions.  */
> +/* { dg-final { scan-assembler-not {\tfcmpe?\th} } } */
> +/* { dg-final { scan-assembler-not {\tfccmpe?\th} } } */
> --
> 2.50.1 (Apple Git-155)
>

Reply via email to