https://github.com/Michael-Chen-NJU updated https://github.com/llvm/llvm-project/pull/224790
From 34fcde06043bab04888cebd7c45273b43b2887fd Mon Sep 17 00:00:00 2001 From: Michael-Chen-NJU <[email protected]> Date: Fri, 18 Sep 2026 11:13:19 +0800 Subject: [PATCH 1/3] [Clang][RISCV] Add packed widening shift intrinsics Add the 32-bit forms of the RISC-V P-extension packed widening shift intrinsics to riscv_packed_simd.h using generic extend-and-shift IR.\n\nRecognize the generic widening shift pattern in the RISC-V backend and select the spec-listed RV32 instructions while retaining composed RV64 sequences.\n\nAdd Clang CodeGen, LLVM CodeGen, and intrinsic header tests for register, immediate, and masked shift amounts. --- clang/lib/Headers/riscv_packed_simd.h | 12 ++ clang/test/CodeGen/RISCV/rvp-intrinsics.c | 116 ++++++++++++++ .../riscv_packed_simd.c | 86 ++++++++++ llvm/lib/Target/RISCV/RISCVISelLowering.cpp | 23 +++ llvm/lib/Target/RISCV/RISCVInstrInfoP.td | 27 ++++ llvm/test/CodeGen/RISCV/rvp-widening-shift.ll | 147 ++++++++++++++++++ 6 files changed, 411 insertions(+) create mode 100644 llvm/test/CodeGen/RISCV/rvp-widening-shift.ll diff --git a/clang/lib/Headers/riscv_packed_simd.h b/clang/lib/Headers/riscv_packed_simd.h index db6d0d37c2e8a..96413a68919f3 100644 --- a/clang/lib/Headers/riscv_packed_simd.h +++ b/clang/lib/Headers/riscv_packed_simd.h @@ -175,6 +175,11 @@ typedef uint32_t uint32x2_t __attribute__((__vector_size__(8))); static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1) { \ return __builtin_convertvector(__rs1, rty); \ } +#define __packed_widen_shift(name, rty, ty, mask) \ + static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, \ + unsigned __shamt) { \ + return __builtin_convertvector(__rs1, rty) << (__shamt & (mask)); \ + } #define __packed_widen_binary_op(name, rty, ty, op) \ static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, \ ty __rs2) { \ @@ -730,6 +735,12 @@ __packed_widen_high4(pwcvth_u16x4, uint16x4_t, uint8x4_t) __packed_widen_high2(pwcvth_i32x2, int32x2_t, int16x2_t) __packed_widen_high2(pwcvth_u32x2, uint32x2_t, uint16x2_t) +/* Packed Widening Shift */ +__packed_widen_shift(pwsll_s_u16x4, uint16x4_t, uint8x4_t, 0xf) +__packed_widen_shift(pwsll_s_u32x2, uint32x2_t, uint16x2_t, 0x1f) +__packed_widen_shift(pwsla_s_i16x4, int16x4_t, int8x4_t, 0xf) +__packed_widen_shift(pwsla_s_i32x2, int32x2_t, int16x2_t, 0x1f) + /* Packed Widening Addition and Subtraction */ __packed_widen_binary_op(pwadd_i16x4, int16x4_t, int8x4_t, +) __packed_widen_binary_op(pwadd_i32x2, int32x2_t, int16x2_t, +) @@ -1314,6 +1325,7 @@ __packed_reinterpret(u32x2_i32x2, int32x2_t, uint32x2_t) #undef __packed_merge_builtin #undef __packed_unary_builtin #undef __packed_widen_convert +#undef __packed_widen_shift #undef __packed_widen_binary_op #undef __packed_widen_binary_acc_op #undef __packed_widen_mul diff --git a/clang/test/CodeGen/RISCV/rvp-intrinsics.c b/clang/test/CodeGen/RISCV/rvp-intrinsics.c index c6721dbeb5db8..0946f4885d938 100644 --- a/clang/test/CodeGen/RISCV/rvp-intrinsics.c +++ b/clang/test/CodeGen/RISCV/rvp-intrinsics.c @@ -8158,6 +8158,122 @@ uint32x2_t test_pwcvtu_u32x2(uint16x2_t rs1) { return __riscv_pwcvtu_u32x2(rs1); } +// RV32-LABEL: define dso_local i64 @test_pwsll_s_u16x4( +// RV32-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <4 x i8> +// RV32-NEXT: [[CONV_I:%.*]] = zext <4 x i8> [[TMP0]] to <4 x i16> +// RV32-NEXT: [[TMP1:%.*]] = trunc i32 [[SHAMT]] to i16 +// RV32-NEXT: [[TMP2:%.*]] = and i16 [[TMP1]], 15 +// RV32-NEXT: [[TMP3:%.*]] = insertelement <4 x i16> poison, i16 [[TMP2]], i64 0 +// RV32-NEXT: [[SH_PROM_I:%.*]] = shufflevector <4 x i16> [[TMP3]], <4 x i16> poison, <4 x i32> zeroinitializer +// RV32-NEXT: [[SHL_I:%.*]] = shl <4 x i16> [[CONV_I]], [[SH_PROM_I]] +// RV32-NEXT: [[TMP4:%.*]] = bitcast <4 x i16> [[SHL_I]] to i64 +// RV32-NEXT: ret i64 [[TMP4]] +// +// RV64-LABEL: define dso_local i64 @test_pwsll_s_u16x4( +// RV64-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <4 x i8> +// RV64-NEXT: [[CONV_I:%.*]] = zext <4 x i8> [[TMP0]] to <4 x i16> +// RV64-NEXT: [[TMP1:%.*]] = trunc i32 [[SHAMT]] to i16 +// RV64-NEXT: [[TMP2:%.*]] = and i16 [[TMP1]], 15 +// RV64-NEXT: [[TMP3:%.*]] = insertelement <4 x i16> poison, i16 [[TMP2]], i64 0 +// RV64-NEXT: [[SH_PROM_I:%.*]] = shufflevector <4 x i16> [[TMP3]], <4 x i16> poison, <4 x i32> zeroinitializer +// RV64-NEXT: [[SHL_I:%.*]] = shl <4 x i16> [[CONV_I]], [[SH_PROM_I]] +// RV64-NEXT: [[TMP4:%.*]] = bitcast <4 x i16> [[SHL_I]] to i64 +// RV64-NEXT: ret i64 [[TMP4]] +// +uint16x4_t test_pwsll_s_u16x4(uint8x4_t rs1, unsigned shamt) { + return __riscv_pwsll_s_u16x4(rs1, shamt); +} + +// RV32-LABEL: define dso_local i64 @test_pwsll_s_u32x2( +// RV32-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16> +// RV32-NEXT: [[CONV_I:%.*]] = zext <2 x i16> [[TMP0]] to <2 x i32> +// RV32-NEXT: [[AND_I:%.*]] = and i32 [[SHAMT]], 31 +// RV32-NEXT: [[SPLAT_SPLATINSERT_I:%.*]] = insertelement <2 x i32> poison, i32 [[AND_I]], i64 0 +// RV32-NEXT: [[SPLAT_SPLAT_I:%.*]] = shufflevector <2 x i32> [[SPLAT_SPLATINSERT_I]], <2 x i32> poison, <2 x i32> zeroinitializer +// RV32-NEXT: [[SHL_I:%.*]] = shl <2 x i32> [[CONV_I]], [[SPLAT_SPLAT_I]] +// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHL_I]] to i64 +// RV32-NEXT: ret i64 [[TMP1]] +// +// RV64-LABEL: define dso_local i64 @test_pwsll_s_u32x2( +// RV64-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16> +// RV64-NEXT: [[CONV_I:%.*]] = zext <2 x i16> [[TMP0]] to <2 x i32> +// RV64-NEXT: [[AND_I:%.*]] = and i32 [[SHAMT]], 31 +// RV64-NEXT: [[SPLAT_SPLATINSERT_I:%.*]] = insertelement <2 x i32> poison, i32 [[AND_I]], i64 0 +// RV64-NEXT: [[SPLAT_SPLAT_I:%.*]] = shufflevector <2 x i32> [[SPLAT_SPLATINSERT_I]], <2 x i32> poison, <2 x i32> zeroinitializer +// RV64-NEXT: [[SHL_I:%.*]] = shl <2 x i32> [[CONV_I]], [[SPLAT_SPLAT_I]] +// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHL_I]] to i64 +// RV64-NEXT: ret i64 [[TMP1]] +// +uint32x2_t test_pwsll_s_u32x2(uint16x2_t rs1, unsigned shamt) { + return __riscv_pwsll_s_u32x2(rs1, shamt); +} + +// RV32-LABEL: define dso_local i64 @test_pwsla_s_i16x4( +// RV32-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <4 x i8> +// RV32-NEXT: [[CONV_I:%.*]] = sext <4 x i8> [[TMP0]] to <4 x i16> +// RV32-NEXT: [[TMP1:%.*]] = trunc i32 [[SHAMT]] to i16 +// RV32-NEXT: [[TMP2:%.*]] = and i16 [[TMP1]], 15 +// RV32-NEXT: [[TMP3:%.*]] = insertelement <4 x i16> poison, i16 [[TMP2]], i64 0 +// RV32-NEXT: [[SH_PROM_I:%.*]] = shufflevector <4 x i16> [[TMP3]], <4 x i16> poison, <4 x i32> zeroinitializer +// RV32-NEXT: [[SHL_I:%.*]] = shl <4 x i16> [[CONV_I]], [[SH_PROM_I]] +// RV32-NEXT: [[TMP4:%.*]] = bitcast <4 x i16> [[SHL_I]] to i64 +// RV32-NEXT: ret i64 [[TMP4]] +// +// RV64-LABEL: define dso_local i64 @test_pwsla_s_i16x4( +// RV64-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <4 x i8> +// RV64-NEXT: [[CONV_I:%.*]] = sext <4 x i8> [[TMP0]] to <4 x i16> +// RV64-NEXT: [[TMP1:%.*]] = trunc i32 [[SHAMT]] to i16 +// RV64-NEXT: [[TMP2:%.*]] = and i16 [[TMP1]], 15 +// RV64-NEXT: [[TMP3:%.*]] = insertelement <4 x i16> poison, i16 [[TMP2]], i64 0 +// RV64-NEXT: [[SH_PROM_I:%.*]] = shufflevector <4 x i16> [[TMP3]], <4 x i16> poison, <4 x i32> zeroinitializer +// RV64-NEXT: [[SHL_I:%.*]] = shl <4 x i16> [[CONV_I]], [[SH_PROM_I]] +// RV64-NEXT: [[TMP4:%.*]] = bitcast <4 x i16> [[SHL_I]] to i64 +// RV64-NEXT: ret i64 [[TMP4]] +// +int16x4_t test_pwsla_s_i16x4(int8x4_t rs1, unsigned shamt) { + return __riscv_pwsla_s_i16x4(rs1, shamt); +} + +// RV32-LABEL: define dso_local i64 @test_pwsla_s_i32x2( +// RV32-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16> +// RV32-NEXT: [[CONV_I:%.*]] = sext <2 x i16> [[TMP0]] to <2 x i32> +// RV32-NEXT: [[AND_I:%.*]] = and i32 [[SHAMT]], 31 +// RV32-NEXT: [[SPLAT_SPLATINSERT_I:%.*]] = insertelement <2 x i32> poison, i32 [[AND_I]], i64 0 +// RV32-NEXT: [[SPLAT_SPLAT_I:%.*]] = shufflevector <2 x i32> [[SPLAT_SPLATINSERT_I]], <2 x i32> poison, <2 x i32> zeroinitializer +// RV32-NEXT: [[SHL_I:%.*]] = shl <2 x i32> [[CONV_I]], [[SPLAT_SPLAT_I]] +// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHL_I]] to i64 +// RV32-NEXT: ret i64 [[TMP1]] +// +// RV64-LABEL: define dso_local i64 @test_pwsla_s_i32x2( +// RV64-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16> +// RV64-NEXT: [[CONV_I:%.*]] = sext <2 x i16> [[TMP0]] to <2 x i32> +// RV64-NEXT: [[AND_I:%.*]] = and i32 [[SHAMT]], 31 +// RV64-NEXT: [[SPLAT_SPLATINSERT_I:%.*]] = insertelement <2 x i32> poison, i32 [[AND_I]], i64 0 +// RV64-NEXT: [[SPLAT_SPLAT_I:%.*]] = shufflevector <2 x i32> [[SPLAT_SPLATINSERT_I]], <2 x i32> poison, <2 x i32> zeroinitializer +// RV64-NEXT: [[SHL_I:%.*]] = shl <2 x i32> [[CONV_I]], [[SPLAT_SPLAT_I]] +// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHL_I]] to i64 +// RV64-NEXT: ret i64 [[TMP1]] +// +int32x2_t test_pwsla_s_i32x2(int16x2_t rs1, unsigned shamt) { + return __riscv_pwsla_s_i32x2(rs1, shamt); +} + // RV32-LABEL: define dso_local i64 @test_pwadd_i16x4( // RV32-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[RS2_COERCE:%.*]]) #[[ATTR0]] { // RV32-NEXT: [[ENTRY:.*:]] diff --git a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c index 14e64c3c3584b..910e1b8b85500 100644 --- a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c +++ b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c @@ -2317,6 +2317,92 @@ uint32x2_t test_pwcvtu_u32x2(uint16x2_t rs1) { return __riscv_pwcvtu_u32x2(rs1); } +// CHECK-LABEL: test_pwsll_s_u16x4: +// RV32: pwsll.bs +// RV64: pwcvtu.wb +// RV64: psll.hs +uint16x4_t test_pwsll_s_u16x4(uint8x4_t rs1, unsigned shamt) { + return __riscv_pwsll_s_u16x4(rs1, shamt); +} + +// CHECK-LABEL: test_pwsll_s_u32x2: +// RV32: pwsll.hs +// RV64: pwcvtu.wh +// RV64: psll.ws +uint32x2_t test_pwsll_s_u32x2(uint16x2_t rs1, unsigned shamt) { + return __riscv_pwsll_s_u32x2(rs1, shamt); +} + +// CHECK-LABEL: test_pwsla_s_i16x4: +// RV32: pwsla.bs +// RV64: pwcvtu.wb +// RV64: psext.h.b +// RV64: psll.hs +int16x4_t test_pwsla_s_i16x4(int8x4_t rs1, unsigned shamt) { + return __riscv_pwsla_s_i16x4(rs1, shamt); +} + +// CHECK-LABEL: test_pwsla_s_i32x2: +// RV32: pwsla.hs +// RV64: pwcvtu.wh +// RV64: psext.w.h +// RV64: psll.ws +int32x2_t test_pwsla_s_i32x2(int16x2_t rs1, unsigned shamt) { + return __riscv_pwsla_s_i32x2(rs1, shamt); +} + +// CHECK-LABEL: test_pwsll_s_u16x4_imm: +// RV32: pwslli.b{{[[:space:]]}}a0, a0, 3 +// RV64: pwcvtu.wb +// RV64: pslli.h{{[[:space:]]}}a0, a0, 3 +uint16x4_t test_pwsll_s_u16x4_imm(uint8x4_t rs1) { + return __riscv_pwsll_s_u16x4(rs1, 3); +} + +// CHECK-LABEL: test_pwsll_s_u32x2_imm: +// RV32: pwslli.h{{[[:space:]]}}a0, a0, 7 +// RV64: pwcvtu.wh +// RV64: pslli.w{{[[:space:]]}}a0, a0, 7 +uint32x2_t test_pwsll_s_u32x2_imm(uint16x2_t rs1) { + return __riscv_pwsll_s_u32x2(rs1, 7); +} + +// CHECK-LABEL: test_pwsla_s_i16x4_imm: +// RV32: pwslai.b{{[[:space:]]}}a0, a0, 3 +// RV64: pwcvtu.wb +// RV64: psext.h.b +// RV64: pslli.h{{[[:space:]]}}a0, a0, 3 +int16x4_t test_pwsla_s_i16x4_imm(int8x4_t rs1) { + return __riscv_pwsla_s_i16x4(rs1, 3); +} + +// CHECK-LABEL: test_pwsla_s_i32x2_imm: +// RV32: pwslai.h{{[[:space:]]}}a0, a0, 7 +// RV64: pwcvtu.wh +// RV64: psext.w.h +// RV64: pslli.w{{[[:space:]]}}a0, a0, 7 +int32x2_t test_pwsla_s_i32x2_imm(int16x2_t rs1) { + return __riscv_pwsla_s_i32x2(rs1, 7); +} + +// Verify that an out-of-range constant shift amount is masked to the maximum +// in-range value by the header implementation. +// CHECK-LABEL: test_pwsll_s_u16x4_masked_imm: +// RV32: pslli.dh{{[[:space:]]}}a0, a0, 15 +// RV64: pwcvtu.wb +// RV64: pslli.h{{[[:space:]]}}a0, a0, 15 +uint16x4_t test_pwsll_s_u16x4_masked_imm(uint8x4_t rs1) { + return __riscv_pwsll_s_u16x4(rs1, 31); +} + +// CHECK-LABEL: test_pwsll_s_u32x2_masked_imm: +// RV32: pslli.dw{{[[:space:]]}}a0, a0, 31 +// RV64: pwcvtu.wh +// RV64: pslli.w{{[[:space:]]}}a0, a0, 31 +uint32x2_t test_pwsll_s_u32x2_masked_imm(uint16x2_t rs1) { + return __riscv_pwsll_s_u32x2(rs1, 63); +} + // CHECK-LABEL: test_pwadd_i16x4: // RV32: pwadd.b // RV64: zip8p diff --git a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp index 2c05e3000c5eb..03966db5d6021 100644 --- a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp +++ b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp @@ -9545,6 +9545,29 @@ SDValue RISCVTargetLowering::LowerOperation(SDValue Op, if (!SplatVal) return SDValue(); + // The 32-bit packed widening shift intrinsics produce extend followed + // by a scalar-splat shift. Preserve that shape as a widening shift + // before generic packed-shift lowering loses the narrow source. + if (!Subtarget.is64Bit() && Op.getOpcode() == ISD::SHL) { + using namespace SDPatternMatch; + MVT VT = Op.getSimpleValueType(); + if (VT == MVT::v4i16 || VT == MVT::v2i32) { + MVT SrcVT = VT == MVT::v4i16 ? MVT::v4i8 : MVT::v2i16; + SDValue Src; + unsigned ExtendOpcode = Op.getOperand(0).getOpcode(); + if ((ExtendOpcode == ISD::SIGN_EXTEND || + ExtendOpcode == ISD::ZERO_EXTEND) && + sd_match(Op.getOperand(0), + m_OneUse(m_Node(ExtendOpcode, + m_Value(Src, m_SpecificVT(SrcVT)))))) { + unsigned Opc = ExtendOpcode == ISD::SIGN_EXTEND ? RISCVISD::PWSLA + : RISCVISD::PWSLL; + SplatVal = DAG.getZExtOrTrunc(SplatVal, SDLoc(Op), MVT::i32); + return DAG.getNode(Opc, SDLoc(Op), VT, Src, SplatVal); + } + } + } + unsigned Opc; switch (Op.getOpcode()) { default: diff --git a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td index 831a591cf6283..81bfbc7bd4326 100644 --- a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td +++ b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td @@ -2052,6 +2052,15 @@ def riscv_pnsrl : RVSDNode<"PNSRL", SDT_RISCVPackedNarrowingShift>; def riscv_pnclip : RVSDNode<"PNCLIP", SDT_RISCVPackedNarrowingShift>; def riscv_pnclipu : RVSDNode<"PNCLIPU", SDT_RISCVPackedNarrowingShift>; +// RV32 packed widening shift. +def SDT_RISCVPackedWideningShift + : SDTypeProfile<1, 2, [SDTCisVec<0>, SDTCisVec<1>, + SDTCisOpSmallerThanOp<1, 0>, + SDTCisSameNumEltsAs<0, 1>, + SDTCisVT<2, XLenVT>]>; +def riscv_pwsll : RVSDNode<"PWSLL", SDT_RISCVPackedWideningShift>; +def riscv_pwsla : RVSDNode<"PWSLA", SDT_RISCVPackedWideningShift>; + // Packed narrowing clip pair. def SDT_RISCVPackedNarrowingClip : SDTypeProfile<1, 2, [SDTCisVec<0>, SDTCisVec<1>, @@ -2581,6 +2590,24 @@ let append Predicates = [IsRV32] in { def : Pat<(v2i32 (sub (zext (v2i16 GPR:$rs1)), (zext (v2i16 GPR:$rs2)))), (PWSUBU_H GPR:$rs1, GPR:$rs2)>; + // Packed widening shift patterns. + def : Pat<(v4i16 (riscv_pwsll (v4i8 GPR:$rs1), uimm4:$imm)), + (PWSLLI_B GPR:$rs1, uimm4:$imm)>; + def : Pat<(v2i32 (riscv_pwsll (v2i16 GPR:$rs1), uimm5:$imm)), + (PWSLLI_H GPR:$rs1, uimm5:$imm)>; + def : Pat<(v4i16 (riscv_pwsla (v4i8 GPR:$rs1), uimm4:$imm)), + (PWSLAI_B GPR:$rs1, uimm4:$imm)>; + def : Pat<(v2i32 (riscv_pwsla (v2i16 GPR:$rs1), uimm5:$imm)), + (PWSLAI_H GPR:$rs1, uimm5:$imm)>; + def : Pat<(v4i16 (riscv_pwsll (v4i8 GPR:$rs1), shiftMask32:$rs2)), + (PWSLL_BS GPR:$rs1, shiftMask32:$rs2)>; + def : Pat<(v2i32 (riscv_pwsll (v2i16 GPR:$rs1), shiftMask32:$rs2)), + (PWSLL_HS GPR:$rs1, shiftMask32:$rs2)>; + def : Pat<(v4i16 (riscv_pwsla (v4i8 GPR:$rs1), shiftMask32:$rs2)), + (PWSLA_BS GPR:$rs1, shiftMask32:$rs2)>; + def : Pat<(v2i32 (riscv_pwsla (v2i16 GPR:$rs1), shiftMask32:$rs2)), + (PWSLA_HS GPR:$rs1, shiftMask32:$rs2)>; + // Packed widening multiply patterns. def : Pat<(v4i16 (riscv_pwmul (v4i8 GPR:$rs1), (v4i8 GPR:$rs2))), (PWMUL_B GPR:$rs1, GPR:$rs2)>; diff --git a/llvm/test/CodeGen/RISCV/rvp-widening-shift.ll b/llvm/test/CodeGen/RISCV/rvp-widening-shift.ll new file mode 100644 index 0000000000000..ec6a0e0032c87 --- /dev/null +++ b/llvm/test/CodeGen/RISCV/rvp-widening-shift.ll @@ -0,0 +1,147 @@ +; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6 +; RUN: llc -mtriple=riscv32 -mattr=+experimental-p -verify-machineinstrs < %s | \ +; RUN: FileCheck %s --check-prefixes=CHECK,RV32 +; RUN: llc -mtriple=riscv64 -mattr=+experimental-p -verify-machineinstrs < %s | \ +; RUN: FileCheck %s --check-prefixes=CHECK,RV64 + +define <4 x i16> @pwsll_v4i8(<4 x i8> %x, i16 %shamt) { +; RV32-LABEL: pwsll_v4i8: +; RV32: # %bb.0: +; RV32-NEXT: pwsll.bs a0, a0, a1 +; RV32-NEXT: ret +; +; RV64-LABEL: pwsll_v4i8: +; RV64: # %bb.0: +; RV64-NEXT: pwcvtu.wb a0, a0 +; RV64-NEXT: psll.hs a0, a0, a1 +; RV64-NEXT: ret + %ext = zext <4 x i8> %x to <4 x i16> + %splat.ins = insertelement <4 x i16> poison, i16 %shamt, i64 0 + %splat = shufflevector <4 x i16> %splat.ins, <4 x i16> poison, <4 x i32> zeroinitializer + %res = shl <4 x i16> %ext, %splat + ret <4 x i16> %res +} + +define <2 x i32> @pwsll_v2i16(<2 x i16> %x, i32 %shamt) { +; RV32-LABEL: pwsll_v2i16: +; RV32: # %bb.0: +; RV32-NEXT: pwsll.hs a0, a0, a1 +; RV32-NEXT: ret +; +; RV64-LABEL: pwsll_v2i16: +; RV64: # %bb.0: +; RV64-NEXT: pwcvtu.wh a0, a0 +; RV64-NEXT: psll.ws a0, a0, a1 +; RV64-NEXT: ret + %ext = zext <2 x i16> %x to <2 x i32> + %splat.ins = insertelement <2 x i32> poison, i32 %shamt, i64 0 + %splat = shufflevector <2 x i32> %splat.ins, <2 x i32> poison, <2 x i32> zeroinitializer + %res = shl <2 x i32> %ext, %splat + ret <2 x i32> %res +} + +define <4 x i16> @pwsla_v4i8(<4 x i8> %x, i16 %shamt) { +; RV32-LABEL: pwsla_v4i8: +; RV32: # %bb.0: +; RV32-NEXT: pwsla.bs a0, a0, a1 +; RV32-NEXT: ret +; +; RV64-LABEL: pwsla_v4i8: +; RV64: # %bb.0: +; RV64-NEXT: pwcvtu.wb a0, a0 +; RV64-NEXT: psext.h.b a0, a0 +; RV64-NEXT: psll.hs a0, a0, a1 +; RV64-NEXT: ret + %ext = sext <4 x i8> %x to <4 x i16> + %splat.ins = insertelement <4 x i16> poison, i16 %shamt, i64 0 + %splat = shufflevector <4 x i16> %splat.ins, <4 x i16> poison, <4 x i32> zeroinitializer + %res = shl <4 x i16> %ext, %splat + ret <4 x i16> %res +} + +define <2 x i32> @pwsla_v2i16(<2 x i16> %x, i32 %shamt) { +; RV32-LABEL: pwsla_v2i16: +; RV32: # %bb.0: +; RV32-NEXT: pwsla.hs a0, a0, a1 +; RV32-NEXT: ret +; +; RV64-LABEL: pwsla_v2i16: +; RV64: # %bb.0: +; RV64-NEXT: pwcvtu.wh a0, a0 +; RV64-NEXT: psext.w.h a0, a0 +; RV64-NEXT: psll.ws a0, a0, a1 +; RV64-NEXT: ret + %ext = sext <2 x i16> %x to <2 x i32> + %splat.ins = insertelement <2 x i32> poison, i32 %shamt, i64 0 + %splat = shufflevector <2 x i32> %splat.ins, <2 x i32> poison, <2 x i32> zeroinitializer + %res = shl <2 x i32> %ext, %splat + ret <2 x i32> %res +} + +define <4 x i16> @pwslli_v4i8(<4 x i8> %x) { +; RV32-LABEL: pwslli_v4i8: +; RV32: # %bb.0: +; RV32-NEXT: pwslli.b a0, a0, 3 +; RV32-NEXT: ret +; +; RV64-LABEL: pwslli_v4i8: +; RV64: # %bb.0: +; RV64-NEXT: pwcvtu.wb a0, a0 +; RV64-NEXT: pslli.h a0, a0, 3 +; RV64-NEXT: ret + %ext = zext <4 x i8> %x to <4 x i16> + %res = shl <4 x i16> %ext, <i16 3, i16 3, i16 3, i16 3> + ret <4 x i16> %res +} + +define <2 x i32> @pwslli_v2i16(<2 x i16> %x) { +; RV32-LABEL: pwslli_v2i16: +; RV32: # %bb.0: +; RV32-NEXT: pwslli.h a0, a0, 7 +; RV32-NEXT: ret +; +; RV64-LABEL: pwslli_v2i16: +; RV64: # %bb.0: +; RV64-NEXT: pwcvtu.wh a0, a0 +; RV64-NEXT: pslli.w a0, a0, 7 +; RV64-NEXT: ret + %ext = zext <2 x i16> %x to <2 x i32> + %res = shl <2 x i32> %ext, <i32 7, i32 7> + ret <2 x i32> %res +} + +define <4 x i16> @pwslai_v4i8(<4 x i8> %x) { +; RV32-LABEL: pwslai_v4i8: +; RV32: # %bb.0: +; RV32-NEXT: pwslai.b a0, a0, 3 +; RV32-NEXT: ret +; +; RV64-LABEL: pwslai_v4i8: +; RV64: # %bb.0: +; RV64-NEXT: pwcvtu.wb a0, a0 +; RV64-NEXT: psext.h.b a0, a0 +; RV64-NEXT: pslli.h a0, a0, 3 +; RV64-NEXT: ret + %ext = sext <4 x i8> %x to <4 x i16> + %res = shl <4 x i16> %ext, <i16 3, i16 3, i16 3, i16 3> + ret <4 x i16> %res +} + +define <2 x i32> @pwslai_v2i16(<2 x i16> %x) { +; RV32-LABEL: pwslai_v2i16: +; RV32: # %bb.0: +; RV32-NEXT: pwslai.h a0, a0, 7 +; RV32-NEXT: ret +; +; RV64-LABEL: pwslai_v2i16: +; RV64: # %bb.0: +; RV64-NEXT: pwcvtu.wh a0, a0 +; RV64-NEXT: psext.w.h a0, a0 +; RV64-NEXT: pslli.w a0, a0, 7 +; RV64-NEXT: ret + %ext = sext <2 x i16> %x to <2 x i32> + %res = shl <2 x i32> %ext, <i32 7, i32 7> + ret <2 x i32> %res +} +;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line: +; CHECK: {{.*}} From 3809d0a966eb8199ae2efee5e1ac0b2fb111257a Mon Sep 17 00:00:00 2001 From: Michael-Chen-NJU <[email protected]> Date: Tue, 22 Sep 2026 11:59:35 +0800 Subject: [PATCH 2/3] [Clang][RISCV] Use intrinsics for packed widening shifts --- clang/include/clang/Basic/BuiltinsRISCV.td | 6 ++ clang/lib/CodeGen/TargetBuiltins/RISCV.cpp | 12 +++ clang/lib/Headers/riscv_packed_simd.h | 18 ++-- clang/test/CodeGen/RISCV/rvp-intrinsics.c | 88 ++++++------------ .../riscv_packed_simd.c | 24 ++--- llvm/include/llvm/IR/IntrinsicsRISCV.td | 8 ++ llvm/lib/Target/RISCV/RISCVISelLowering.cpp | 43 +++++---- llvm/test/CodeGen/RISCV/rvp-widening-shift.ll | 89 ++++++++++++++----- 8 files changed, 159 insertions(+), 129 deletions(-) diff --git a/clang/include/clang/Basic/BuiltinsRISCV.td b/clang/include/clang/Basic/BuiltinsRISCV.td index ee840e45a65ba..2382f87ba0242 100644 --- a/clang/include/clang/Basic/BuiltinsRISCV.td +++ b/clang/include/clang/Basic/BuiltinsRISCV.td @@ -465,6 +465,12 @@ def psext_h_i32x2 : RISCVBuiltin<"_Vector<2, int>(_Vector<2, int>)">; def pzext_b_u16x4 : RISCVBuiltin<"_Vector<4, unsigned short>(_Vector<4, unsigned short>)">; def pzext_h_u32x2 : RISCVBuiltin<"_Vector<2, unsigned int>(_Vector<2, unsigned int>)">; +// Packed Widening Shifts +def pwsll_s_u16x4 : RISCVBuiltin<"_Vector<4, unsigned short>(_Vector<4, unsigned char>, unsigned int)">; +def pwsll_s_u32x2 : RISCVBuiltin<"_Vector<2, unsigned int>(_Vector<2, unsigned short>, unsigned int)">; +def pwsla_s_i16x4 : RISCVBuiltin<"_Vector<4, short>(_Vector<4, signed char>, unsigned int)">; +def pwsla_s_i32x2 : RISCVBuiltin<"_Vector<2, int>(_Vector<2, short>, unsigned int)">; + // Packed Narrowing Clip Pair (32-bit) def pnclipp_i8x4 : RISCVBuiltin<"_Vector<4, signed char>(_Vector<2, short>, _Vector<2, short>)">; diff --git a/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp b/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp index f99a05ce673aa..c1cead61a0108 100644 --- a/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp +++ b/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp @@ -1199,6 +1199,18 @@ Value *CodeGenFunction::EmitRISCVBuiltinExpr(unsigned BuiltinID, break; } + // Packed Widening Shifts + case RISCV::BI__builtin_riscv_pwsll_s_u16x4: + case RISCV::BI__builtin_riscv_pwsll_s_u32x2: + ID = Intrinsic::riscv_pwsll; + IntrinsicTypes = {ResultType, Ops[0]->getType()}; + break; + case RISCV::BI__builtin_riscv_pwsla_s_i16x4: + case RISCV::BI__builtin_riscv_pwsla_s_i32x2: + ID = Intrinsic::riscv_pwsla; + IntrinsicTypes = {ResultType, Ops[0]->getType()}; + break; + // Packed Averaging Addition and Subtraction case RISCV::BI__builtin_riscv_paadd_i8x4: case RISCV::BI__builtin_riscv_paadd_i16x2: diff --git a/clang/lib/Headers/riscv_packed_simd.h b/clang/lib/Headers/riscv_packed_simd.h index 96413a68919f3..aafad863c7efd 100644 --- a/clang/lib/Headers/riscv_packed_simd.h +++ b/clang/lib/Headers/riscv_packed_simd.h @@ -175,11 +175,6 @@ typedef uint32_t uint32x2_t __attribute__((__vector_size__(8))); static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1) { \ return __builtin_convertvector(__rs1, rty); \ } -#define __packed_widen_shift(name, rty, ty, mask) \ - static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, \ - unsigned __shamt) { \ - return __builtin_convertvector(__rs1, rty) << (__shamt & (mask)); \ - } #define __packed_widen_binary_op(name, rty, ty, op) \ static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, \ ty __rs2) { \ @@ -736,10 +731,14 @@ __packed_widen_high2(pwcvth_i32x2, int32x2_t, int16x2_t) __packed_widen_high2(pwcvth_u32x2, uint32x2_t, uint16x2_t) /* Packed Widening Shift */ -__packed_widen_shift(pwsll_s_u16x4, uint16x4_t, uint8x4_t, 0xf) -__packed_widen_shift(pwsll_s_u32x2, uint32x2_t, uint16x2_t, 0x1f) -__packed_widen_shift(pwsla_s_i16x4, int16x4_t, int8x4_t, 0xf) -__packed_widen_shift(pwsla_s_i32x2, int32x2_t, int16x2_t, 0x1f) +__packed_binary_builtin_mixed(pwsll_s_u16x4, uint16x4_t, uint8x4_t, unsigned, + __builtin_riscv_pwsll_s_u16x4) +__packed_binary_builtin_mixed(pwsll_s_u32x2, uint32x2_t, uint16x2_t, unsigned, + __builtin_riscv_pwsll_s_u32x2) +__packed_binary_builtin_mixed(pwsla_s_i16x4, int16x4_t, int8x4_t, unsigned, + __builtin_riscv_pwsla_s_i16x4) +__packed_binary_builtin_mixed(pwsla_s_i32x2, int32x2_t, int16x2_t, unsigned, + __builtin_riscv_pwsla_s_i32x2) /* Packed Widening Addition and Subtraction */ __packed_widen_binary_op(pwadd_i16x4, int16x4_t, int8x4_t, +) @@ -1325,7 +1324,6 @@ __packed_reinterpret(u32x2_i32x2, int32x2_t, uint32x2_t) #undef __packed_merge_builtin #undef __packed_unary_builtin #undef __packed_widen_convert -#undef __packed_widen_shift #undef __packed_widen_binary_op #undef __packed_widen_binary_acc_op #undef __packed_widen_mul diff --git a/clang/test/CodeGen/RISCV/rvp-intrinsics.c b/clang/test/CodeGen/RISCV/rvp-intrinsics.c index 0946f4885d938..58faf9eaf486b 100644 --- a/clang/test/CodeGen/RISCV/rvp-intrinsics.c +++ b/clang/test/CodeGen/RISCV/rvp-intrinsics.c @@ -137,7 +137,7 @@ uint64_t test_abs_u64(int64_t a) { /* Packed Splat (32-bit) */ // RV32-LABEL: define dso_local i32 @test_pmv_s_u8x4( -// RV32-SAME: i8 noundef zeroext [[X:%.*]]) #[[ATTR0:[0-9]+]] { +// RV32-SAME: i8 noundef zeroext [[X:%.*]]) #[[ATTR0]] { // RV32-NEXT: [[ENTRY:.*:]] // RV32-NEXT: [[VECINIT_I:%.*]] = insertelement <4 x i8> poison, i8 [[X]], i64 0 // RV32-NEXT: [[VECINIT3_I:%.*]] = shufflevector <4 x i8> [[VECINIT_I]], <4 x i8> poison, <4 x i32> zeroinitializer @@ -145,7 +145,7 @@ uint64_t test_abs_u64(int64_t a) { // RV32-NEXT: ret i32 [[TMP0]] // // RV64-LABEL: define dso_local i32 @test_pmv_s_u8x4( -// RV64-SAME: i8 noundef zeroext [[X:%.*]]) #[[ATTR0:[0-9]+]] { +// RV64-SAME: i8 noundef zeroext [[X:%.*]]) #[[ATTR0]] { // RV64-NEXT: [[ENTRY:.*:]] // RV64-NEXT: [[VECINIT_I:%.*]] = insertelement <4 x i8> poison, i8 [[X]], i64 0 // RV64-NEXT: [[VECINIT3_I:%.*]] = shufflevector <4 x i8> [[VECINIT_I]], <4 x i8> poison, <4 x i32> zeroinitializer @@ -8162,27 +8162,17 @@ uint32x2_t test_pwcvtu_u32x2(uint16x2_t rs1) { // RV32-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] { // RV32-NEXT: [[ENTRY:.*:]] // RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <4 x i8> -// RV32-NEXT: [[CONV_I:%.*]] = zext <4 x i8> [[TMP0]] to <4 x i16> -// RV32-NEXT: [[TMP1:%.*]] = trunc i32 [[SHAMT]] to i16 -// RV32-NEXT: [[TMP2:%.*]] = and i16 [[TMP1]], 15 -// RV32-NEXT: [[TMP3:%.*]] = insertelement <4 x i16> poison, i16 [[TMP2]], i64 0 -// RV32-NEXT: [[SH_PROM_I:%.*]] = shufflevector <4 x i16> [[TMP3]], <4 x i16> poison, <4 x i32> zeroinitializer -// RV32-NEXT: [[SHL_I:%.*]] = shl <4 x i16> [[CONV_I]], [[SH_PROM_I]] -// RV32-NEXT: [[TMP4:%.*]] = bitcast <4 x i16> [[SHL_I]] to i64 -// RV32-NEXT: ret i64 [[TMP4]] +// RV32-NEXT: [[TMP1:%.*]] = call <4 x i16> @llvm.riscv.pwsll.v4i16.v4i8(<4 x i8> [[TMP0]], i32 [[SHAMT]]) +// RV32-NEXT: [[TMP2:%.*]] = bitcast <4 x i16> [[TMP1]] to i64 +// RV32-NEXT: ret i64 [[TMP2]] // // RV64-LABEL: define dso_local i64 @test_pwsll_s_u16x4( // RV64-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] { // RV64-NEXT: [[ENTRY:.*:]] // RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <4 x i8> -// RV64-NEXT: [[CONV_I:%.*]] = zext <4 x i8> [[TMP0]] to <4 x i16> -// RV64-NEXT: [[TMP1:%.*]] = trunc i32 [[SHAMT]] to i16 -// RV64-NEXT: [[TMP2:%.*]] = and i16 [[TMP1]], 15 -// RV64-NEXT: [[TMP3:%.*]] = insertelement <4 x i16> poison, i16 [[TMP2]], i64 0 -// RV64-NEXT: [[SH_PROM_I:%.*]] = shufflevector <4 x i16> [[TMP3]], <4 x i16> poison, <4 x i32> zeroinitializer -// RV64-NEXT: [[SHL_I:%.*]] = shl <4 x i16> [[CONV_I]], [[SH_PROM_I]] -// RV64-NEXT: [[TMP4:%.*]] = bitcast <4 x i16> [[SHL_I]] to i64 -// RV64-NEXT: ret i64 [[TMP4]] +// RV64-NEXT: [[TMP1:%.*]] = call <4 x i16> @llvm.riscv.pwsll.v4i16.v4i8(<4 x i8> [[TMP0]], i32 [[SHAMT]]) +// RV64-NEXT: [[TMP2:%.*]] = bitcast <4 x i16> [[TMP1]] to i64 +// RV64-NEXT: ret i64 [[TMP2]] // uint16x4_t test_pwsll_s_u16x4(uint8x4_t rs1, unsigned shamt) { return __riscv_pwsll_s_u16x4(rs1, shamt); @@ -8192,25 +8182,17 @@ uint16x4_t test_pwsll_s_u16x4(uint8x4_t rs1, unsigned shamt) { // RV32-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] { // RV32-NEXT: [[ENTRY:.*:]] // RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16> -// RV32-NEXT: [[CONV_I:%.*]] = zext <2 x i16> [[TMP0]] to <2 x i32> -// RV32-NEXT: [[AND_I:%.*]] = and i32 [[SHAMT]], 31 -// RV32-NEXT: [[SPLAT_SPLATINSERT_I:%.*]] = insertelement <2 x i32> poison, i32 [[AND_I]], i64 0 -// RV32-NEXT: [[SPLAT_SPLAT_I:%.*]] = shufflevector <2 x i32> [[SPLAT_SPLATINSERT_I]], <2 x i32> poison, <2 x i32> zeroinitializer -// RV32-NEXT: [[SHL_I:%.*]] = shl <2 x i32> [[CONV_I]], [[SPLAT_SPLAT_I]] -// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHL_I]] to i64 -// RV32-NEXT: ret i64 [[TMP1]] +// RV32-NEXT: [[TMP1:%.*]] = call <2 x i32> @llvm.riscv.pwsll.v2i32.v2i16(<2 x i16> [[TMP0]], i32 [[SHAMT]]) +// RV32-NEXT: [[TMP2:%.*]] = bitcast <2 x i32> [[TMP1]] to i64 +// RV32-NEXT: ret i64 [[TMP2]] // // RV64-LABEL: define dso_local i64 @test_pwsll_s_u32x2( // RV64-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] { // RV64-NEXT: [[ENTRY:.*:]] // RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16> -// RV64-NEXT: [[CONV_I:%.*]] = zext <2 x i16> [[TMP0]] to <2 x i32> -// RV64-NEXT: [[AND_I:%.*]] = and i32 [[SHAMT]], 31 -// RV64-NEXT: [[SPLAT_SPLATINSERT_I:%.*]] = insertelement <2 x i32> poison, i32 [[AND_I]], i64 0 -// RV64-NEXT: [[SPLAT_SPLAT_I:%.*]] = shufflevector <2 x i32> [[SPLAT_SPLATINSERT_I]], <2 x i32> poison, <2 x i32> zeroinitializer -// RV64-NEXT: [[SHL_I:%.*]] = shl <2 x i32> [[CONV_I]], [[SPLAT_SPLAT_I]] -// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHL_I]] to i64 -// RV64-NEXT: ret i64 [[TMP1]] +// RV64-NEXT: [[TMP1:%.*]] = call <2 x i32> @llvm.riscv.pwsll.v2i32.v2i16(<2 x i16> [[TMP0]], i32 [[SHAMT]]) +// RV64-NEXT: [[TMP2:%.*]] = bitcast <2 x i32> [[TMP1]] to i64 +// RV64-NEXT: ret i64 [[TMP2]] // uint32x2_t test_pwsll_s_u32x2(uint16x2_t rs1, unsigned shamt) { return __riscv_pwsll_s_u32x2(rs1, shamt); @@ -8220,27 +8202,17 @@ uint32x2_t test_pwsll_s_u32x2(uint16x2_t rs1, unsigned shamt) { // RV32-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] { // RV32-NEXT: [[ENTRY:.*:]] // RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <4 x i8> -// RV32-NEXT: [[CONV_I:%.*]] = sext <4 x i8> [[TMP0]] to <4 x i16> -// RV32-NEXT: [[TMP1:%.*]] = trunc i32 [[SHAMT]] to i16 -// RV32-NEXT: [[TMP2:%.*]] = and i16 [[TMP1]], 15 -// RV32-NEXT: [[TMP3:%.*]] = insertelement <4 x i16> poison, i16 [[TMP2]], i64 0 -// RV32-NEXT: [[SH_PROM_I:%.*]] = shufflevector <4 x i16> [[TMP3]], <4 x i16> poison, <4 x i32> zeroinitializer -// RV32-NEXT: [[SHL_I:%.*]] = shl <4 x i16> [[CONV_I]], [[SH_PROM_I]] -// RV32-NEXT: [[TMP4:%.*]] = bitcast <4 x i16> [[SHL_I]] to i64 -// RV32-NEXT: ret i64 [[TMP4]] +// RV32-NEXT: [[TMP1:%.*]] = call <4 x i16> @llvm.riscv.pwsla.v4i16.v4i8(<4 x i8> [[TMP0]], i32 [[SHAMT]]) +// RV32-NEXT: [[TMP2:%.*]] = bitcast <4 x i16> [[TMP1]] to i64 +// RV32-NEXT: ret i64 [[TMP2]] // // RV64-LABEL: define dso_local i64 @test_pwsla_s_i16x4( // RV64-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] { // RV64-NEXT: [[ENTRY:.*:]] // RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <4 x i8> -// RV64-NEXT: [[CONV_I:%.*]] = sext <4 x i8> [[TMP0]] to <4 x i16> -// RV64-NEXT: [[TMP1:%.*]] = trunc i32 [[SHAMT]] to i16 -// RV64-NEXT: [[TMP2:%.*]] = and i16 [[TMP1]], 15 -// RV64-NEXT: [[TMP3:%.*]] = insertelement <4 x i16> poison, i16 [[TMP2]], i64 0 -// RV64-NEXT: [[SH_PROM_I:%.*]] = shufflevector <4 x i16> [[TMP3]], <4 x i16> poison, <4 x i32> zeroinitializer -// RV64-NEXT: [[SHL_I:%.*]] = shl <4 x i16> [[CONV_I]], [[SH_PROM_I]] -// RV64-NEXT: [[TMP4:%.*]] = bitcast <4 x i16> [[SHL_I]] to i64 -// RV64-NEXT: ret i64 [[TMP4]] +// RV64-NEXT: [[TMP1:%.*]] = call <4 x i16> @llvm.riscv.pwsla.v4i16.v4i8(<4 x i8> [[TMP0]], i32 [[SHAMT]]) +// RV64-NEXT: [[TMP2:%.*]] = bitcast <4 x i16> [[TMP1]] to i64 +// RV64-NEXT: ret i64 [[TMP2]] // int16x4_t test_pwsla_s_i16x4(int8x4_t rs1, unsigned shamt) { return __riscv_pwsla_s_i16x4(rs1, shamt); @@ -8250,25 +8222,17 @@ int16x4_t test_pwsla_s_i16x4(int8x4_t rs1, unsigned shamt) { // RV32-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) #[[ATTR0]] { // RV32-NEXT: [[ENTRY:.*:]] // RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16> -// RV32-NEXT: [[CONV_I:%.*]] = sext <2 x i16> [[TMP0]] to <2 x i32> -// RV32-NEXT: [[AND_I:%.*]] = and i32 [[SHAMT]], 31 -// RV32-NEXT: [[SPLAT_SPLATINSERT_I:%.*]] = insertelement <2 x i32> poison, i32 [[AND_I]], i64 0 -// RV32-NEXT: [[SPLAT_SPLAT_I:%.*]] = shufflevector <2 x i32> [[SPLAT_SPLATINSERT_I]], <2 x i32> poison, <2 x i32> zeroinitializer -// RV32-NEXT: [[SHL_I:%.*]] = shl <2 x i32> [[CONV_I]], [[SPLAT_SPLAT_I]] -// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHL_I]] to i64 -// RV32-NEXT: ret i64 [[TMP1]] +// RV32-NEXT: [[TMP1:%.*]] = call <2 x i32> @llvm.riscv.pwsla.v2i32.v2i16(<2 x i16> [[TMP0]], i32 [[SHAMT]]) +// RV32-NEXT: [[TMP2:%.*]] = bitcast <2 x i32> [[TMP1]] to i64 +// RV32-NEXT: ret i64 [[TMP2]] // // RV64-LABEL: define dso_local i64 @test_pwsla_s_i32x2( // RV64-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef signext [[SHAMT:%.*]]) #[[ATTR0]] { // RV64-NEXT: [[ENTRY:.*:]] // RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16> -// RV64-NEXT: [[CONV_I:%.*]] = sext <2 x i16> [[TMP0]] to <2 x i32> -// RV64-NEXT: [[AND_I:%.*]] = and i32 [[SHAMT]], 31 -// RV64-NEXT: [[SPLAT_SPLATINSERT_I:%.*]] = insertelement <2 x i32> poison, i32 [[AND_I]], i64 0 -// RV64-NEXT: [[SPLAT_SPLAT_I:%.*]] = shufflevector <2 x i32> [[SPLAT_SPLATINSERT_I]], <2 x i32> poison, <2 x i32> zeroinitializer -// RV64-NEXT: [[SHL_I:%.*]] = shl <2 x i32> [[CONV_I]], [[SPLAT_SPLAT_I]] -// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHL_I]] to i64 -// RV64-NEXT: ret i64 [[TMP1]] +// RV64-NEXT: [[TMP1:%.*]] = call <2 x i32> @llvm.riscv.pwsla.v2i32.v2i16(<2 x i16> [[TMP0]], i32 [[SHAMT]]) +// RV64-NEXT: [[TMP2:%.*]] = bitcast <2 x i32> [[TMP1]] to i64 +// RV64-NEXT: ret i64 [[TMP2]] // int32x2_t test_pwsla_s_i32x2(int16x2_t rs1, unsigned shamt) { return __riscv_pwsla_s_i32x2(rs1, shamt); diff --git a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c index 910e1b8b85500..91ac4557204ff 100644 --- a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c +++ b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c @@ -2385,21 +2385,25 @@ int32x2_t test_pwsla_s_i32x2_imm(int16x2_t rs1) { return __riscv_pwsla_s_i32x2(rs1, 7); } -// Verify that an out-of-range constant shift amount is masked to the maximum -// in-range value by the header implementation. -// CHECK-LABEL: test_pwsll_s_u16x4_masked_imm: -// RV32: pslli.dh{{[[:space:]]}}a0, a0, 15 +// The intrinsic uses the low 5 bits of the register-form instruction. Values +// that do not fit an immediate form must retain that register-form semantics. +// CHECK-LABEL: test_pwsll_s_u16x4_low5: +// RV32: li{{[[:space:]]}}a1, 31 +// RV32-NEXT: pwsll.bs{{[[:space:]]}}a0, a0, a1 +// RV64: li{{[[:space:]]}}a1, 31 // RV64: pwcvtu.wb -// RV64: pslli.h{{[[:space:]]}}a0, a0, 15 -uint16x4_t test_pwsll_s_u16x4_masked_imm(uint8x4_t rs1) { +// RV64: psll.hs{{[[:space:]]}}a0, a0, a1 +uint16x4_t test_pwsll_s_u16x4_low5(uint8x4_t rs1) { return __riscv_pwsll_s_u16x4(rs1, 31); } -// CHECK-LABEL: test_pwsll_s_u32x2_masked_imm: -// RV32: pslli.dw{{[[:space:]]}}a0, a0, 31 +// CHECK-LABEL: test_pwsll_s_u32x2_low5: +// RV32: li{{[[:space:]]}}a1, 63 +// RV32-NEXT: pwsll.hs{{[[:space:]]}}a0, a0, a1 +// RV64: li{{[[:space:]]}}a1, 63 // RV64: pwcvtu.wh -// RV64: pslli.w{{[[:space:]]}}a0, a0, 31 -uint32x2_t test_pwsll_s_u32x2_masked_imm(uint16x2_t rs1) { +// RV64: psll.ws{{[[:space:]]}}a0, a0, a1 +uint32x2_t test_pwsll_s_u32x2_low5(uint16x2_t rs1) { return __riscv_pwsll_s_u32x2(rs1, 63); } diff --git a/llvm/include/llvm/IR/IntrinsicsRISCV.td b/llvm/include/llvm/IR/IntrinsicsRISCV.td index 09399b0ea3f36..0236d614377dc 100644 --- a/llvm/include/llvm/IR/IntrinsicsRISCV.td +++ b/llvm/include/llvm/IR/IntrinsicsRISCV.td @@ -2076,6 +2076,14 @@ class RVPBinaryIntrinsic def int_riscv_psshl : RVPShiftIntrinsic; def int_riscv_psshlr : RVPShiftIntrinsic; + // Packed Widening Shifts. + class RVPWideningShiftIntrinsic + : DefaultAttrsIntrinsic<[llvm_anyvector_ty], + [llvm_anyvector_ty, llvm_i32_ty], + [IntrNoMem, IntrSpeculatable]>; + def int_riscv_pwsll : RVPWideningShiftIntrinsic; + def int_riscv_pwsla : RVPWideningShiftIntrinsic; + // Packed Exchanged Addition and Subtraction. def int_riscv_pas : RVPBinaryIntrinsic; def int_riscv_psa : RVPBinaryIntrinsic; diff --git a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp index 03966db5d6021..4a6f08f101c27 100644 --- a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp +++ b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp @@ -9545,29 +9545,6 @@ SDValue RISCVTargetLowering::LowerOperation(SDValue Op, if (!SplatVal) return SDValue(); - // The 32-bit packed widening shift intrinsics produce extend followed - // by a scalar-splat shift. Preserve that shape as a widening shift - // before generic packed-shift lowering loses the narrow source. - if (!Subtarget.is64Bit() && Op.getOpcode() == ISD::SHL) { - using namespace SDPatternMatch; - MVT VT = Op.getSimpleValueType(); - if (VT == MVT::v4i16 || VT == MVT::v2i32) { - MVT SrcVT = VT == MVT::v4i16 ? MVT::v4i8 : MVT::v2i16; - SDValue Src; - unsigned ExtendOpcode = Op.getOperand(0).getOpcode(); - if ((ExtendOpcode == ISD::SIGN_EXTEND || - ExtendOpcode == ISD::ZERO_EXTEND) && - sd_match(Op.getOperand(0), - m_OneUse(m_Node(ExtendOpcode, - m_Value(Src, m_SpecificVT(SrcVT)))))) { - unsigned Opc = ExtendOpcode == ISD::SIGN_EXTEND ? RISCVISD::PWSLA - : RISCVISD::PWSLL; - SplatVal = DAG.getZExtOrTrunc(SplatVal, SDLoc(Op), MVT::i32); - return DAG.getNode(Opc, SDLoc(Op), VT, Src, SplatVal); - } - } - } - unsigned Opc; switch (Op.getOpcode()) { default: @@ -13112,6 +13089,26 @@ SDValue RISCVTargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op, return DAG.getNode(getRVPShiftOpcode(IntNo), DL, Op.getValueType(), Op.getOperand(1), ShAmt); } + case Intrinsic::riscv_pwsll: + case Intrinsic::riscv_pwsla: { + MVT VT = Op.getSimpleValueType(); + SDValue Src = Op.getOperand(1); + MVT SrcVT = Src.getSimpleValueType(); + if (!((VT == MVT::v4i16 && SrcVT == MVT::v4i8) || + (VT == MVT::v2i32 && SrcVT == MVT::v2i16))) + reportFatalUsageError("unsupported packed widening shift intrinsic"); + + SDValue ShAmt = DAG.getAnyExtOrTrunc(Op.getOperand(2), DL, XLenVT); + bool IsSigned = IntNo == Intrinsic::riscv_pwsla; + if (!Subtarget.is64Bit()) { + unsigned Opc = IsSigned ? RISCVISD::PWSLA : RISCVISD::PWSLL; + return DAG.getNode(Opc, DL, VT, Src, ShAmt); + } + + unsigned ExtOpc = IsSigned ? ISD::SIGN_EXTEND : ISD::ZERO_EXTEND; + SDValue Wide = DAG.getNode(ExtOpc, DL, VT, Src); + return DAG.getNode(RISCVISD::PSHL, DL, VT, Wide, ShAmt); + } case Intrinsic::riscv_psext_b: case Intrinsic::riscv_psext_h: { EVT VT = Op.getValueType(); diff --git a/llvm/test/CodeGen/RISCV/rvp-widening-shift.ll b/llvm/test/CodeGen/RISCV/rvp-widening-shift.ll index ec6a0e0032c87..a95ce8fabbadb 100644 --- a/llvm/test/CodeGen/RISCV/rvp-widening-shift.ll +++ b/llvm/test/CodeGen/RISCV/rvp-widening-shift.ll @@ -4,6 +4,11 @@ ; RUN: llc -mtriple=riscv64 -mattr=+experimental-p -verify-machineinstrs < %s | \ ; RUN: FileCheck %s --check-prefixes=CHECK,RV64 +declare <4 x i16> @llvm.riscv.pwsll.v4i16.v4i8(<4 x i8>, i32) +declare <2 x i32> @llvm.riscv.pwsll.v2i32.v2i16(<2 x i16>, i32) +declare <4 x i16> @llvm.riscv.pwsla.v4i16.v4i8(<4 x i8>, i32) +declare <2 x i32> @llvm.riscv.pwsla.v2i32.v2i16(<2 x i16>, i32) + define <4 x i16> @pwsll_v4i8(<4 x i8> %x, i16 %shamt) { ; RV32-LABEL: pwsll_v4i8: ; RV32: # %bb.0: @@ -15,10 +20,8 @@ define <4 x i16> @pwsll_v4i8(<4 x i8> %x, i16 %shamt) { ; RV64-NEXT: pwcvtu.wb a0, a0 ; RV64-NEXT: psll.hs a0, a0, a1 ; RV64-NEXT: ret - %ext = zext <4 x i8> %x to <4 x i16> - %splat.ins = insertelement <4 x i16> poison, i16 %shamt, i64 0 - %splat = shufflevector <4 x i16> %splat.ins, <4 x i16> poison, <4 x i32> zeroinitializer - %res = shl <4 x i16> %ext, %splat + %shamt.ext = zext i16 %shamt to i32 + %res = call <4 x i16> @llvm.riscv.pwsll.v4i16.v4i8(<4 x i8> %x, i32 %shamt.ext) ret <4 x i16> %res } @@ -33,10 +36,7 @@ define <2 x i32> @pwsll_v2i16(<2 x i16> %x, i32 %shamt) { ; RV64-NEXT: pwcvtu.wh a0, a0 ; RV64-NEXT: psll.ws a0, a0, a1 ; RV64-NEXT: ret - %ext = zext <2 x i16> %x to <2 x i32> - %splat.ins = insertelement <2 x i32> poison, i32 %shamt, i64 0 - %splat = shufflevector <2 x i32> %splat.ins, <2 x i32> poison, <2 x i32> zeroinitializer - %res = shl <2 x i32> %ext, %splat + %res = call <2 x i32> @llvm.riscv.pwsll.v2i32.v2i16(<2 x i16> %x, i32 %shamt) ret <2 x i32> %res } @@ -52,10 +52,8 @@ define <4 x i16> @pwsla_v4i8(<4 x i8> %x, i16 %shamt) { ; RV64-NEXT: psext.h.b a0, a0 ; RV64-NEXT: psll.hs a0, a0, a1 ; RV64-NEXT: ret - %ext = sext <4 x i8> %x to <4 x i16> - %splat.ins = insertelement <4 x i16> poison, i16 %shamt, i64 0 - %splat = shufflevector <4 x i16> %splat.ins, <4 x i16> poison, <4 x i32> zeroinitializer - %res = shl <4 x i16> %ext, %splat + %shamt.ext = zext i16 %shamt to i32 + %res = call <4 x i16> @llvm.riscv.pwsla.v4i16.v4i8(<4 x i8> %x, i32 %shamt.ext) ret <4 x i16> %res } @@ -71,10 +69,7 @@ define <2 x i32> @pwsla_v2i16(<2 x i16> %x, i32 %shamt) { ; RV64-NEXT: psext.w.h a0, a0 ; RV64-NEXT: psll.ws a0, a0, a1 ; RV64-NEXT: ret - %ext = sext <2 x i16> %x to <2 x i32> - %splat.ins = insertelement <2 x i32> poison, i32 %shamt, i64 0 - %splat = shufflevector <2 x i32> %splat.ins, <2 x i32> poison, <2 x i32> zeroinitializer - %res = shl <2 x i32> %ext, %splat + %res = call <2 x i32> @llvm.riscv.pwsla.v2i32.v2i16(<2 x i16> %x, i32 %shamt) ret <2 x i32> %res } @@ -89,8 +84,7 @@ define <4 x i16> @pwslli_v4i8(<4 x i8> %x) { ; RV64-NEXT: pwcvtu.wb a0, a0 ; RV64-NEXT: pslli.h a0, a0, 3 ; RV64-NEXT: ret - %ext = zext <4 x i8> %x to <4 x i16> - %res = shl <4 x i16> %ext, <i16 3, i16 3, i16 3, i16 3> + %res = call <4 x i16> @llvm.riscv.pwsll.v4i16.v4i8(<4 x i8> %x, i32 3) ret <4 x i16> %res } @@ -105,8 +99,7 @@ define <2 x i32> @pwslli_v2i16(<2 x i16> %x) { ; RV64-NEXT: pwcvtu.wh a0, a0 ; RV64-NEXT: pslli.w a0, a0, 7 ; RV64-NEXT: ret - %ext = zext <2 x i16> %x to <2 x i32> - %res = shl <2 x i32> %ext, <i32 7, i32 7> + %res = call <2 x i32> @llvm.riscv.pwsll.v2i32.v2i16(<2 x i16> %x, i32 7) ret <2 x i32> %res } @@ -122,8 +115,7 @@ define <4 x i16> @pwslai_v4i8(<4 x i8> %x) { ; RV64-NEXT: psext.h.b a0, a0 ; RV64-NEXT: pslli.h a0, a0, 3 ; RV64-NEXT: ret - %ext = sext <4 x i8> %x to <4 x i16> - %res = shl <4 x i16> %ext, <i16 3, i16 3, i16 3, i16 3> + %res = call <4 x i16> @llvm.riscv.pwsla.v4i16.v4i8(<4 x i8> %x, i32 3) ret <4 x i16> %res } @@ -139,8 +131,57 @@ define <2 x i32> @pwslai_v2i16(<2 x i16> %x) { ; RV64-NEXT: psext.w.h a0, a0 ; RV64-NEXT: pslli.w a0, a0, 7 ; RV64-NEXT: ret - %ext = sext <2 x i16> %x to <2 x i32> - %res = shl <2 x i32> %ext, <i32 7, i32 7> + %res = call <2 x i32> @llvm.riscv.pwsla.v2i32.v2i16(<2 x i16> %x, i32 7) + ret <2 x i32> %res +} + +define <4 x i16> @pwslli_v4i8_31(<4 x i8> %x) { +; RV32-LABEL: pwslli_v4i8_31: +; RV32: # %bb.0: +; RV32-NEXT: li a1, 31 +; RV32-NEXT: pwsll.bs a0, a0, a1 +; RV32-NEXT: ret +; +; RV64-LABEL: pwslli_v4i8_31: +; RV64: # %bb.0: +; RV64-NEXT: li a1, 31 +; RV64-NEXT: pwcvtu.wb a0, a0 +; RV64-NEXT: psll.hs a0, a0, a1 +; RV64-NEXT: ret + %res = call <4 x i16> @llvm.riscv.pwsll.v4i16.v4i8(<4 x i8> %x, i32 31) + ret <4 x i16> %res +} + +define <4 x i16> @pwslai_v4i8_16(<4 x i8> %x) { +; RV32-LABEL: pwslai_v4i8_16: +; RV32: # %bb.0: +; RV32-NEXT: li a1, 16 +; RV32-NEXT: pwsla.bs a0, a0, a1 +; RV32-NEXT: ret +; +; RV64-LABEL: pwslai_v4i8_16: +; RV64: # %bb.0: +; RV64-NEXT: li a1, 16 +; RV64-NEXT: pwcvtu.wb a0, a0 +; RV64-NEXT: psext.h.b a0, a0 +; RV64-NEXT: psll.hs a0, a0, a1 +; RV64-NEXT: ret + %res = call <4 x i16> @llvm.riscv.pwsla.v4i16.v4i8(<4 x i8> %x, i32 16) + ret <4 x i16> %res +} + +define <2 x i32> @pwslli_v2i16_31(<2 x i16> %x) { +; RV32-LABEL: pwslli_v2i16_31: +; RV32: # %bb.0: +; RV32-NEXT: pwslli.h a0, a0, 31 +; RV32-NEXT: ret +; +; RV64-LABEL: pwslli_v2i16_31: +; RV64: # %bb.0: +; RV64-NEXT: pwcvtu.wh a0, a0 +; RV64-NEXT: pslli.w a0, a0, 31 +; RV64-NEXT: ret + %res = call <2 x i32> @llvm.riscv.pwsll.v2i32.v2i16(<2 x i16> %x, i32 31) ret <2 x i32> %res } ;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line: From 7ae05158b861705a83d4a64e730607e1b89dce02 Mon Sep 17 00:00:00 2001 From: Michael-Chen-NJU <[email protected]> Date: Sun, 27 Sep 2026 11:29:12 +0800 Subject: [PATCH 3/3] [RISCV] Use ANY_EXTEND for packed widening shift amount --- llvm/lib/Target/RISCV/RISCVISelLowering.cpp | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp index 6a988b82284a8..e76c0b692964c 100644 --- a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp +++ b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp @@ -13167,7 +13167,8 @@ SDValue RISCVTargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op, (VT == MVT::v2i32 && SrcVT == MVT::v2i16))) reportFatalUsageError("unsupported packed widening shift intrinsic"); - SDValue ShAmt = DAG.getAnyExtOrTrunc(Op.getOperand(2), DL, XLenVT); + SDValue ShAmt = + DAG.getNode(ISD::ANY_EXTEND, DL, XLenVT, Op.getOperand(2)); bool IsSigned = IntNo == Intrinsic::riscv_pwsla; if (!Subtarget.is64Bit()) { unsigned Opc = IsSigned ? RISCVISD::PWSLA : RISCVISD::PWSLL; _______________________________________________ cfe-commits mailing list [email protected] https://lists.llvm.org/cgi-bin/mailman/listinfo/cfe-commits
