Add an expander for isnormal using integer arithmetic. This is
typically faster and avoids generating spurious exceptions on
signaling NaNs. This fixes part of PR66462.
int isnormal1 (float x) { return __builtin_isnormal (x); }
Before:
fabs s0, s0
mov w0, 2139095039
fmov s31, w0
fcmp s0, s31
movi v31.2s, 0x80, lsl 16
fccmp s0, s31, 1, ls
cset w0, lt
eor w0, w0, 1
ret
After:
fmov w0, s0
ubfx w0, w0, 23, 8
sub w0, w0, #1
cmp w0, 254
cset w0, cc
ret
HFmode and BFmode are handled too. Neither has 16-bit integer
arithmetic, so their encoding is zero-extended into a word first:
int isnormal2 (_Float16 x) { return __builtin_isnormal (x); }
Before:
umov w0, v0.h[0]
mvni v0.4h, 0x84, lsl 8
movi v30.4h, 0x4, lsl 8
fcvt s0, h0
and w0, w0, 32767
fcvt s30, h30
dup v31.4h, w0
fcvt s31, h31
fcmp s31, s0
cset w0, hi
fcmp s31, s30
cset w1, lt
orr w0, w0, w1
eor w0, w0, 1
ret
After:
umov w0, v0.h[0]
ubfx w0, w0, 10, 5
sub w0, w0, #1
cmp w0, 30
cset w0, cc
ret
gcc/ChangeLog:
PR middle-end/66462
* config/aarch64/aarch64.md (isnormal<mode>2): Add new expander.
* config/aarch64/iterators.md (GPF_HF_BF): New mode iterator.
(mantissa_bits): Add HF and BF.
(V_INT_EQUIV): Add BF.
gcc/testsuite/ChangeLog:
PR middle-end/66462
* gcc.target/aarch64/pr66462.c: Add tests for isnormal, including
_Float16 and __bf16.
---
gcc/config/aarch64/aarch64.md | 29 +++++++++
gcc/config/aarch64/iterators.md | 7 ++-
gcc/testsuite/gcc.target/aarch64/pr66462.c | 72 ++++++++++++++++++++++
3 files changed, 106 insertions(+), 2 deletions(-)
diff --git ./gcc/config/aarch64/aarch64.md ./gcc/config/aarch64/aarch64.md
index a32f8c627f1..58383483b12 100644
--- ./gcc/config/aarch64/aarch64.md
+++ ./gcc/config/aarch64/aarch64.md
@@ -7900,6 +7900,35 @@
}
)
+;; A number is normal iff its exponent field E is neither 0 nor all ones,
+;; i.e. (unsigned) (E - 1) < EXP_MAX - 1.
+(define_expand "isnormal<mode>2"
+ [(match_operand:SI 0 "register_operand")
+ (match_operand:GPF_HF_BF 1 "register_operand")]
+ "TARGET_FLOAT"
+{
+ scalar_int_mode imode = <V_INT_EQUIV>mode;
+ int exp_bits = GET_MODE_BITSIZE (<MODE>mode) - <mantissa_bits> - 1;
+ HOST_WIDE_INT exp_max = (HOST_WIDE_INT_1 << exp_bits) - 1;
+ rtx op = force_lowpart_subreg (imode, operands[1], <MODE>mode);
+ /* There is no 16-bit arithmetic, so work on the encoding of the 16-bit
+ formats in a word. */
+ if (imode == HImode)
+ imode = SImode;
+ op = convert_to_mode (imode, op, 1);
+ op = expand_simple_binop (imode, LSHIFTRT, op, GEN_INT (<mantissa_bits>),
+ NULL_RTX, 1, OPTAB_DIRECT);
+ op = expand_simple_binop (imode, AND, op, GEN_INT (exp_max),
+ NULL_RTX, 1, OPTAB_DIRECT);
+ op = expand_simple_binop (imode, PLUS, op, constm1_rtx,
+ NULL_RTX, 1, OPTAB_DIRECT);
+ rtx cc_reg = aarch64_gen_compare_reg (LTU, op, GEN_INT (exp_max - 1));
+ rtx cmp = gen_rtx_fmt_ee (LTU, SImode, cc_reg, const0_rtx);
+ emit_insn (gen_aarch64_cstoresi (operands[0], cmp, cc_reg));
+ DONE;
+}
+)
+
;; -------------------------------------------------------------------
;; Reload support
;; -------------------------------------------------------------------
diff --git ./gcc/config/aarch64/iterators.md ./gcc/config/aarch64/iterators.md
index a56edb8ebf4..76f042ff2f8 100644
--- ./gcc/config/aarch64/iterators.md
+++ ./gcc/config/aarch64/iterators.md
@@ -65,6 +65,9 @@
;; Iterator for all 16-bit scalar floating point modes (HF, BF)
(define_mode_iterator HFBF [HF BF])
+;; Iterator for all scalar floating point modes (HF, BF, SF, DF)
+(define_mode_iterator GPF_HF_BF [HF BF SF DF])
+
;; Iterator for all integer modes (up to 64-bit) plus all General Purpose
;; Floating-point registers (32- and 64-bit modes).
(define_mode_iterator ALLI_GPF [ALLI GPF])
@@ -1504,7 +1507,7 @@
(define_mode_attr half_mask [(HI "255") (SI "65535") (DI "4294967295")])
-(define_mode_attr mantissa_bits [(SF "23") (DF "52")])
+(define_mode_attr mantissa_bits [(HF "10") (BF "7") (SF "23") (DF "52")])
;; For constraints used in scalar immediate vector moves
(define_mode_attr hq [(HI "h") (QI "q")])
@@ -2484,7 +2487,7 @@
(V2SF "V2SI") (V4SF "V4SI")
(DF "DI") (V2DF "V2DI")
(SF "SI") (SI "SI")
- (HF "HI")
+ (HF "HI") (BF "HI")
(VNx16QI "VNx16QI")
(VNx8HI "VNx8HI") (VNx8HF "VNx8HI")
(VNx8BF "VNx8HI")
diff --git ./gcc/testsuite/gcc.target/aarch64/pr66462.c
./gcc/testsuite/gcc.target/aarch64/pr66462.c
index 2bd734b3193..c6367ac16f6 100644
--- ./gcc/testsuite/gcc.target/aarch64/pr66462.c
+++ ./gcc/testsuite/gcc.target/aarch64/pr66462.c
@@ -64,6 +64,42 @@ static void t_nan (double x, bool res)
__builtin_abort ();
}
+static void t_normalf (float x, bool res)
+{
+ if (__builtin_isnormal (x) != res)
+ __builtin_abort ();
+ if (__builtin_isnormal (-x) != res)
+ __builtin_abort ();
+ if (fetestexcept (FE_INVALID))
+ __builtin_abort ();
+}
+
+static void t_normal (double x, bool res)
+{
+ if (__builtin_isnormal (x) != res)
+ __builtin_abort ();
+ if (__builtin_isnormal (-x) != res)
+ __builtin_abort ();
+ if (fetestexcept (FE_INVALID))
+ __builtin_abort ();
+}
+
+
+/* Generate the _Float16 and __bf16 testers with a macro rather than
+ spelling out the same body for each type. */
+#define DEF_TEST(NAME, TYPE, PRED) \
+static void NAME (TYPE x, bool res) \
+{ \
+ if (PRED (x) != res) \
+ __builtin_abort (); \
+ if (PRED (-x) != res) \
+ __builtin_abort (); \
+ if (fetestexcept (FE_INVALID)) \
+ __builtin_abort (); \
+}
+
+DEF_TEST (t_normal16, _Float16, __builtin_isnormal)
+DEF_TEST (t_normalbf, __bf16, __builtin_isnormal)
int
main ()
@@ -106,5 +142,41 @@ main ()
t_nan (__builtin_nans (""), 1);
t_nan (__builtin_nan (""), 1);
+ t_normalf (0.0f, 0);
+ t_normalf (1.0f, 1);
+ t_normalf (__FLT_MIN__, 1);
+ t_normalf (__FLT_MAX__, 1);
+ t_normalf (__FLT_DENORM_MIN__, 0);
+ t_normalf (__builtin_inff (), 0);
+ t_normalf (__builtin_nansf (""), 0);
+ t_normalf (__builtin_nanf (""), 0);
+
+ t_normal (0.0, 0);
+ t_normal (1.0, 1);
+ t_normal (__DBL_MIN__, 1);
+ t_normal (__DBL_MAX__, 1);
+ t_normal (__DBL_DENORM_MIN__, 0);
+ t_normal (__builtin_inf (), 0);
+ t_normal (__builtin_nans (""), 0);
+ t_normal (__builtin_nan (""), 0);
+
+ t_normal16 (0.0f16, 0);
+ t_normal16 (1.0f16, 1);
+ t_normal16 (__FLT16_MIN__, 1);
+ t_normal16 (__FLT16_MAX__, 1);
+ t_normal16 (__FLT16_DENORM_MIN__, 0);
+ t_normal16 ((_Float16) __builtin_inff (), 0);
+ t_normal16 (__builtin_nansf16 (""), 0);
+ t_normal16 (__builtin_nanf16 (""), 0);
+
+ t_normalbf (0.0bf16, 0);
+ t_normalbf (1.0bf16, 1);
+ t_normalbf (__BFLT16_MIN__, 1);
+ t_normalbf (__BFLT16_MAX__, 1);
+ t_normalbf (__BFLT16_DENORM_MIN__, 0);
+ t_normalbf ((__bf16) __builtin_inff (), 0);
+ t_normalbf (__builtin_nansf16b (""), 0);
+ t_normalbf ((__bf16) __builtin_nanf (""), 0);
+
return 0;
}
--
2.54.0